From b493030755ca31c38e64235b3e6f1bdf64d6fc32 Mon Sep 17 00:00:00 2001 From: leiysky Date: Sun, 20 Sep 2026 20:20:54 +0800 Subject: [PATCH 1/5] feat: add adaptive fill admission before I/O timeouts --- ARCHITECTURE.md | 10 +- BENCHMARK.md | 2 + CHANGELOG.md | 2 + CONFIGURATION.md | 23 + README.md | 4 +- benchmarks/cache/main.rs | 4 + benchmarks/cache_soak/main.rs | 4 + benchmarks/mixed_workloads/main.rs | 6 + benchmarks/src/config.rs | 19 + benchmarks/src/report.rs | 17 + cache2/ERRORS.md | 2 + cache2/src/cache.rs | 3 +- cache2/src/config/runtime.rs | 82 ++ cache2/src/io/engine/mod.rs | 12 +- cache2/src/io/engine/recovery.rs | 50 +- cache2/src/io/engine/tests.rs | 68 ++ cache2/src/io/fill_control.rs | 786 ++++++++++++++++++++ cache2/src/io/mod.rs | 1 + cache2/src/lib.rs | 4 + cache2/src/region/runtime/metrics.rs | 1 + cache2/src/region/runtime/mod.rs | 45 +- cache2/src/region/runtime/shutdown_tests.rs | 83 +++ cache2/src/snapshot.rs | 58 +- 23 files changed, 1275 insertions(+), 11 deletions(-) create mode 100644 cache2/src/io/fill_control.rs diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index 910c28b..876805b 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -80,7 +80,7 @@ This layout converts small foreground mutations into ordered, batched Region wri 5. Return after bounded in-memory admission. 6. The shard worker writes sealed batches and publishes their L2 mappings only after write completion. -Full staging or short-path contention returns structured `ErrorKind::Overloaded`. Admission is shard-local. Success means accepted staging; `put_l2` becomes visible when publication completes. `drain` fences all mutations accepted before its operation barrier and waits for their Region writes and L2 publication, but does not issue the recovery durability syncs. +Full staging or short-path contention returns structured `ErrorKind::Overloaded`. Staging admission is shard-local; optional adaptive fill budgets are shared across shards. Success means accepted staging; `put_l2` becomes visible when publication completes. `drain` fences all mutations accepted before its operation barrier and waits for their Region writes and L2 publication, but does not issue the recovery durability syncs. ### `get` @@ -114,6 +114,14 @@ Reads and writes use independent bounded engine pools. Reclaim has separate read Each lane uses one concrete `IoEngine` for admission, submission, cancellation, statistics, and shutdown. Driver-specific constructors start POSIX workers or an io_uring driver behind the same bounded command and completion protocol. Callers share the engine through `Arc`; its final owner joins the workers. Submitted requests retain their buffers and capacity until actual completion, independently of the caller's wait deadline. +### Pre-timeout fill pressure + +`io::fill_control` owns optional background observations and adaptive fill admission. `Disabled` retains the existing path. Enabled modes preallocate a fixed observation table sized by append and reclaim worker counts and run one accounted monitoring thread. Background workers register before engine admission and retain their observations through completion validation and publication. A 100 ms monitor can detect old or stalled work without requiring another completion; aggregate validated throughput provides an approximate drain estimate. Pressure does not change terminal health or prove a device fault. + +A packed atomic token bucket reserves both encoded bytes and record count before foreground staging. Each reservation/refund makes at most four CAS attempts; full refill epochs prevent stale permits from refunding a later budget. No per-fill clock, allocation, or controller mutex is required. Failed staging can refund its reservation; accepted work always keeps its resources and proceeds without reacquiring rate credit. `Observe` records hypothetical rejection only; `Adaptive` rejects new fills immediately and skips optional reinsertion under pressure. Reads, deletes, and essential reclaim bypass fill budgets. A short mutex protects only background observations, monitor decisions, and the optional controller snapshot. Close signals and joins the monitor independently of I/O completion. + +See [adaptive fill admission](CONFIGURATION.md#adaptive-fill-admission) for rate ceilings, pause/resume thresholds, bounded bursts, phase accounting, and tuning limits. + ### Memory The managed-memory limit covers the index mapping, heat bits, L1, append buffers, reclaim buffers, metadata, cache-owned thread stacks, recovery scratch, and transient reads. Total deployment memory additionally includes allocator metadata, Tokio, process overhead, and the kernel page cache. `CacheConfig::new` rejects invalid or insufficient memory budgets before file access; actual allocation can still fail during open. diff --git a/BENCHMARK.md b/BENCHMARK.md index 924cdc9..1399504 100644 --- a/BENCHMARK.md +++ b/BENCHMARK.md @@ -114,6 +114,8 @@ Repeat sizes in `CACHE_SOAK_VALUE_BYTES` to weight a production distribution. Us | io_uring | `_IO_URING__SQPOLL_MS` | absent (disabled) | | io_uring | `_IO_URING__SQPOLL_CPU` | absent (unpinned) | +Optional fill admission uses `_FILL_CONTROL=disabled|observe|adaptive` under the same three prefixes. Enabled modes require explicit `_FILL_BYTES_PER_SECOND` and `_FILL_OPERATIONS_PER_SECOND` ceilings. The harness prints the effective options and adds a `type=fill_control` report beside cache records. Compare the modes in alternating order with identical ceilings and traffic; record rejected and hypothetical fills alongside completed throughput. High ceilings help measure instrumentation overhead, while lower ceilings and injected storage stalls exercise admission behavior. Buffered macOS results do not qualify Linux NVMe or cgroup throttling. + Select the backend with `_IO_ENGINE=posix|io-uring`. POSIX workers bound concurrent operations. io_uring ring count and aggregate in-flight limit are independent; changing one does not rewrite the other. Ring count must not exceed the in-flight limit. IOPOLL requires `_IO_MODE=direct`; SQPOLL CPU requires an idle timeout. Only the selected backend's settings are read. Each harness prints the resulting `IoEngineOptions`, and buffer estimates use its actual concurrency. The request benchmark's default read-wait capacity follows the selected read pool's in-flight limit. `CACHE_BENCH_STATS` is now `CACHE_BENCH_ACTIVITY_COUNTERS`. Machine-readable reports use `version=2`, renaming the cache record field `statistics_enabled` to `activity_counters_enabled`; the counter population is unchanged. diff --git a/CHANGELOG.md b/CHANGELOG.md index 90c54a1..d1a8d34 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -9,6 +9,8 @@ ### Improvements +- Optional `RuntimeOptions::fill_control` observes background request age, validated progress, and estimated drain time before timeout. `Observe` reports hypothetical rejections; `Adaptive` bounds new fills by encoded bytes and record count, pauses on stalled work, resumes gradually, and suppresses optional reinsertion under pressure. Reads, deletes, accepted writes, and essential reclaim retain their paths. `CacheSnapshot::fill_control` exposes pressure and accounting independently of statistics; enabled snapshots take a short controller lock. Controller storage and its monitoring stack are included in managed memory. Disabled by default; see [configuration](CONFIGURATION.md#adaptive-fill-admission) for ceilings and measurement limits. + - Background write and reclaim timeouts now enter a reversible `CacheHealth::Recovering` state: new cache fills return overload while existing reads and deletes remain available. Original requests keep their bounded buffers and Regions and are never resubmitted; all affected work must complete validation and publication before fills resume. `RuntimeOptions::io_recovery_timeout` defaults to `None` for recovery until completion or close; use `Some(duration)` to bound recovery or `Some(Duration::ZERO)` for immediate cancellation. Recovery checks at fixed one-second intervals. Close interrupts recovery and preserves the existing unfenced-write safeguards; drain may wait indefinitely. Actual I/O errors and invalid completions remain terminal. ## v0.5.0 (2026-09-16) diff --git a/CONFIGURATION.md b/CONFIGURATION.md index bd44d76..dc6638a 100644 --- a/CONFIGURATION.md +++ b/CONFIGURATION.md @@ -425,3 +425,26 @@ Set `options.reclaim_io_timeout = Duration::from_secs(30)` after constructing `R ### Background I/O recovery `options.io_recovery_timeout = None` is the default: background admission/completion timeouts enter `CacheHealth::Recovering` and wait until the original operation completes or close interrupts recovery. Set `Some(Duration::from_secs(300))` for a five-minute additional budget, or `Some(Duration::ZERO)` for immediate cancellation. Finite combined deadlines must be representable. Recovery uses fixed one-second checks, without an exponential backoff or repeated submission of issued I/O. Admission retries are allowed only when the engine returns the unsubmitted operation. New `put` and `put_l2` calls return overload while any background operation is recovering; reads and deletes keep their normal behavior. Already accepted work retains its bounded resources. Fills resume only after every affected operation passes validation and publication. Drain can wait indefinitely in the default mode; close interrupts recovery within a polling interval, then applies bounded cancellation. Normal close work still has its ordinary deadlines. Actual I/O errors, invalid completions, and finite-budget exhaustion retain terminal failure safeguards. This is not an automatic reopen of a failed instance. + +### Adaptive fill admission + +`RuntimeOptions::fill_control` defaults to `FillControlOptions::Disabled`. `Observe` records pressure and hypothetical rejections while retaining ordinary admission. `Adaptive` enforces the same decisions for new `put` and `put_l2` fills, returning `ErrorKind::Overloaded` immediately when paused or over budget. Neither mode changes foreground reads, deletes, accepted flushes, essential reclaim, or the existing timeout recovery policy. Adaptive pressure also suppresses optional hot-record reinsertion. Pressure is separate from `CacheHealth`: a pre-timeout pause leaves a healthy cache `Running`. + +```rust +use cache2::{AdaptiveFillOptions, FillControlOptions, RuntimeOptions}; + +let mut options = RuntimeOptions::default(); +// Example ceilings; choose values for the measured workload. +let limits = AdaptiveFillOptions::new(256 * 1024 * 1024, 20_000); +options.fill_control = FillControlOptions::Observe(limits); +// After evaluating pressure and would_reject, enforce the same policy: +options.fill_control = FillControlOptions::Adaptive(limits); +``` + +The two ceilings constrain logical encoded fill bytes (charged in 64-byte units) and fill records, not device bandwidth or physical IOPS. Valid ceilings are 640 bytes/s through 1 TiB/s and 10 through 655350 records/s. Budgets refill every 100 ms, with capacity for 100 ms of the configured ceilings and at least one maximum-size record. An idle or delayed monitor never accumulates more than that bounded burst. Failed staging refunds only within the current refill epoch. Admission uses bounded atomic retries and may conservatively reject contention; it never waits for rate credit. + +A dedicated monitor checks outstanding background work every 100 ms even if no I/O completes. It pauses when any request reaches one quarter of its normal deadline, estimated drain time reaches one quarter of the shortest observed deadline (capped at five seconds), or outstanding work makes no validated progress for the smaller of 500 ms and half that deadline. These are conservative pressure signals: very short deadlines or a descheduled monitor can still reach timeout first. Staging saturation together with a nonshrinking outstanding queue reduces both rates by 30% at most once per 500 ms. A paused controller resumes when work drains, or validated progress brings age and estimated drain time below one tenth of the deadline, starting at half its previous rates and at most 80% of measured service rates when available. After three progressing 500 ms samples, rates increase by approximately 2.5% of the ceilings per progressing sample (rounded to whole bytes and records per second), up to the ceilings. Floors of 640 bytes/s and 10 records/s retain probe capacity; idle time alone never increases the rates. Normal timeout recovery keeps its admission fence until all affected work is validated and published. + +`CacheSnapshot::fill_control` reports pressure, enforcement, current rates, rejections, hypothetical rejections, outstanding bytes/operations, oldest age, and estimated drain time independently of `RuntimeOptions::stats`. The monitor tracks background requests through admission, result consumption, and validation/publication, including scheduling delays. Its cumulative `admission_ns`, `completion_wait_ns`, and `validation_ns` measure those phases. Outstanding bytes exclude unflushed staging; drain time uses aggregate recent validated background throughput, including reclaim reads. A zero drain estimate means no estimate is available when work is pending. These measurements cannot distinguish device throttling from CPU scheduling, kernel queueing, or slow validation; correlate them with host metrics before diagnosing hardware. `cache_fill_pressure_changed` logs state transitions under `cache2::health`. + +Enabled modes preallocate one observation per append/reclaim worker and one 512 KiB monitoring stack, all included in `CacheConfig::minimum_memory_bytes()`. Background observations and controller snapshots take a short lock over this bounded table. Foreground fill admission uses atomics without allocation, clock reads, or locks from the controller. The monitor wakes at 100 ms intervals, including while idle. Close stops it independently of outstanding I/O. Disabled mode creates no controller, observations, or monitoring thread. diff --git a/README.md b/README.md index aa03ad3..b0fc2f2 100644 --- a/README.md +++ b/README.md @@ -106,7 +106,7 @@ The on-disk format is versioned. During 0.x, deployments should expect cold star ### Metrics -`Cache::snapshot()` provides lock-free health and resource gauges. Setting `RuntimeOptions::stats.activity_counters` to `true` adds cumulative cache and I/O counters. `Cache::detailed_snapshot()` samples L1, index, write-buffer pressure, and Region metadata for periodic diagnostics. +`Cache::snapshot()` provides health and resource gauges using atomics and, when enabled, a short fill-controller lock. Setting `RuntimeOptions::stats.activity_counters` to `true` adds cumulative cache and I/O counters. `Cache::detailed_snapshot()` samples L1, index, write-buffer pressure, and Region metadata for periodic diagnostics. `RuntimeOptions::stats` independently enables complete public request outcomes, L1-hit, L2-lookup and mutation latency (each `Off`, `Full`, or `Sampled`), and full I/O latency by read/write/reclaim role. `Cache::stats_snapshot()` combines these with the existing summary without metadata scans. Structured request rows include their timing scope and collection mode. Applications own metric conversion, timestamps, scheduling and transport. Run `cargo run --example stats -- ` for an example. Full timing avoids sampling work; sampled histograms retain actual sample counts and cannot guarantee observation of rare tail events. Recorder storage is preallocated, bounded and charged to managed memory. @@ -158,3 +158,5 @@ Licensed under the [Apache License, Version 2.0](LICENSE). ### Reclaim read deadline Set `RuntimeOptions::reclaim_io_timeout` to change the normal background reclaim deadline, for example `Duration::from_secs(30)` (default five seconds). Background write and reclaim timeouts enter `CacheHealth::Recovering`: new fills return overload while reads and deletes remain available. `RuntimeOptions::io_recovery_timeout` defaults to `None`, allowing recovery until completion or close. Use `Some(Duration::from_secs(300))` to limit the additional wait, or `Some(Duration::ZERO)` for immediate cancellation. Original requests retain their resources and are never resubmitted; fills resume after validation and publication of all affected work. Close interrupts recovery; drain may wait indefinitely. Actual I/O errors and invalid completions still fail the cache. + +Optional [adaptive fill admission](CONFIGURATION.md#adaptive-fill-admission) detects slow background progress before timeout and limits new fills by encoded bytes and record count. Start with `FillControlOptions::Observe` to inspect pressure and hypothetical rejections, then use `Adaptive` to enforce the configured limits. It is disabled by default; reads, deletes, accepted writes, and essential reclaim retain their existing paths. diff --git a/benchmarks/cache/main.rs b/benchmarks/cache/main.rs index 34b90fb..852e9c8 100644 --- a/benchmarks/cache/main.rs +++ b/benchmarks/cache/main.rs @@ -80,6 +80,7 @@ struct BenchConfig { write_clients: usize, clients: usize, io_engine: IoEngineOptions, + fill_control: cache2::FillControlOptions, io_mode: IoMode, l1_eviction_policy: L1EvictionPolicy, stats: cache2::StatsOptions, @@ -229,6 +230,7 @@ impl BenchConfig { write_clients, clients, io_engine, + fill_control: benchmarks::config::fill_control_from_env("CACHE_BENCH")?, io_mode, l1_eviction_policy, stats, @@ -246,6 +248,8 @@ impl BenchConfig { fn runtime_options(&self) -> RuntimeOptions { let mut options = RuntimeOptions::default(); options.io_engine = self.io_engine; + options.fill_control = self.fill_control; + println!("fill_control={:?}", self.fill_control); options.io_mode = self.io_mode; options.append_shards = self.append_shards; options.l1_capacity_bytes = self.l1_capacity_bytes; diff --git a/benchmarks/cache_soak/main.rs b/benchmarks/cache_soak/main.rs index 0d6db57..48e9ff8 100644 --- a/benchmarks/cache_soak/main.rs +++ b/benchmarks/cache_soak/main.rs @@ -89,6 +89,7 @@ struct SoakConfig { require_path_coverage: bool, require_reinsert_coverage: bool, io_engine: IoEngineOptions, + fill_control: cache2::FillControlOptions, io_mode: IoMode, l1_eviction_policy: L1EvictionPolicy, directory: PathBuf, @@ -193,6 +194,7 @@ impl SoakConfig { require_path_coverage, require_reinsert_coverage, io_engine, + fill_control: benchmarks::config::fill_control_from_env("CACHE_SOAK")?, io_mode, l1_eviction_policy, directory, @@ -209,6 +211,8 @@ impl SoakConfig { fn runtime_options(&self) -> RuntimeOptions { let mut options = RuntimeOptions::default(); options.io_engine = self.io_engine; + options.fill_control = self.fill_control; + println!("fill_control={:?}", self.fill_control); options.io_mode = self.io_mode; options.append_shards = self.append_shards; options.l1_capacity_bytes = self.l1_capacity_bytes; diff --git a/benchmarks/mixed_workloads/main.rs b/benchmarks/mixed_workloads/main.rs index ffd6e43..8b3c437 100644 --- a/benchmarks/mixed_workloads/main.rs +++ b/benchmarks/mixed_workloads/main.rs @@ -217,6 +217,7 @@ struct HarnessOptions { latency_sample_interval: usize, seed: u64, io_engine: IoEngineOptions, + fill_control: cache2::FillControlOptions, io_mode: IoMode, l1_eviction_policy: L1EvictionPolicy, directory: PathBuf, @@ -268,6 +269,7 @@ impl HarnessOptions { latency_sample_interval, seed, io_engine, + fill_control: benchmarks::config::fill_control_from_env("CACHE_WORKLOAD")?, io_mode, l1_eviction_policy, directory, @@ -346,6 +348,7 @@ impl HarnessOptions { latency_sample_interval: self.latency_sample_interval, seed: self.seed, io_engine: self.io_engine, + fill_control: self.fill_control, io_mode: self.io_mode, l1_eviction_policy: self.l1_eviction_policy, directory: self.directory.clone(), @@ -367,6 +370,7 @@ struct ScenarioConfig { latency_sample_interval: usize, seed: u64, io_engine: IoEngineOptions, + fill_control: cache2::FillControlOptions, io_mode: IoMode, l1_eviction_policy: L1EvictionPolicy, directory: PathBuf, @@ -383,6 +387,8 @@ impl ScenarioConfig { fn runtime_options(&self) -> RuntimeOptions { let mut options = RuntimeOptions::default(); options.io_engine = self.io_engine; + options.fill_control = self.fill_control; + println!("fill_control={:?}", self.fill_control); options.io_mode = self.io_mode; options.append_shards = self.append_shards; options.l1_capacity_bytes = self.l1_capacity_bytes; diff --git a/benchmarks/src/config.rs b/benchmarks/src/config.rs index 6560070..e4ee19b 100644 --- a/benchmarks/src/config.rs +++ b/benchmarks/src/config.rs @@ -59,6 +59,25 @@ pub fn io_engine_from_env(prefix: &str) -> io::Result { } } +/// Reads optional fill-control mode and its explicit logical rate ceilings. +pub fn fill_control_from_env(prefix: &str) -> io::Result { + let name = format!("{prefix}_FILL_CONTROL"); + let mode = setting::(&name)?.unwrap_or_else(|| "disabled".into()); + if mode == "disabled" { + return Ok(cache2::FillControlOptions::Disabled); + } + let bytes = setting(&format!("{prefix}_FILL_BYTES_PER_SECOND"))? + .ok_or_else(|| invalid("enabled fill control requires FILL_BYTES_PER_SECOND"))?; + let operations = setting(&format!("{prefix}_FILL_OPERATIONS_PER_SECOND"))? + .ok_or_else(|| invalid("enabled fill control requires FILL_OPERATIONS_PER_SECOND"))?; + let options = cache2::AdaptiveFillOptions::new(bytes, operations); + match mode.as_str() { + "observe" => Ok(cache2::FillControlOptions::Observe(options)), + "adaptive" => Ok(cache2::FillControlOptions::Adaptive(options)), + _ => Err(invalid(format!("unsupported {name}: {mode}"))), + } +} + /// Maximum active reads across the selected backend's pool. pub fn read_max_in_flight(options: IoEngineOptions) -> usize { match options { diff --git a/benchmarks/src/report.rs b/benchmarks/src/report.rs index e96ac2c..b334871 100644 --- a/benchmarks/src/report.rs +++ b/benchmarks/src/report.rs @@ -479,6 +479,23 @@ pub fn emit_cache_report( detailed.region.physical_record_count, format_bytes(cache.logical_disk_peak_bytes as f64), ); + let fill = cache.fill_control; + println!( + "report version=2 type=fill_control benchmark={} scenario={} phase={} pressure={:?} enforcing={} bytes_per_second={} operations_per_second={} rejections={} would_reject={} outstanding_operations={} outstanding_bytes={} oldest_operation_ns={} estimated_drain_ns={}", + benchmark, + scenario, + phase, + fill.pressure, + fill.enforcing, + fill.bytes_per_second, + fill.operations_per_second, + fill.rejections, + fill.would_reject, + fill.outstanding_operations, + fill.outstanding_bytes, + fill.oldest_operation_ns, + fill.estimated_drain_ns, + ); println!( "report version=2 type=cache benchmark={} scenario={} phase={} health={:?} activity_counters_enabled={} puts={} deletes={} written_bytes={} served_bytes={} l1_hits={} l1_misses={} l2_hits={} l2_misses={} l2_read_memory_misses={} l2_read_busy_misses={} l2_read_overloads={} l2_read_wait_ns={} promotions={} l1_evictions={} l1_bypasses={} write_rejections={} io_failures={} rotations={} reclaimed_regions={} reclaim_bytes={} reclaim_records={} reinsert_records={} reinsert_bytes={} reinsert_skipped={} reinsert_budget_skipped={}", benchmark, diff --git a/cache2/ERRORS.md b/cache2/ERRORS.md index c18ab88..0fdb40f 100644 --- a/cache2/ERRORS.md +++ b/cache2/ERRORS.md @@ -32,6 +32,8 @@ fn cache_value(cache: &Cache, key: &[u8], value: &[u8]) -> Result<(), Error> { Do not retry without a bound. C² deliberately exposes pressure instead of building unbounded queues. With the default immediate-read policy, read-pool or buffer pressure is `Ok(None)`. When read waiting is enabled, queue saturation, buffer pressure, and deadline expiry are `ErrorKind::Overloaded`. +With `FillControlOptions::Adaptive`, new fills also return `Overloaded` when the controller pauses or exhausts its byte/record budget. Inspect `CacheSnapshot::fill_control` to distinguish controller rejection from other admission pressure. Pre-timeout pressure does not change `CacheHealth::Running` or identify a hardware failure. `Observe` only counts hypothetical controller rejections. Reads, deletes, accepted writes, and essential reclaim retain their ordinary paths. + ## Classifications | `ErrorKind` | Meaning | Usual response | diff --git a/cache2/src/cache.rs b/cache2/src/cache.rs index f138406..ee1660c 100644 --- a/cache2/src/cache.rs +++ b/cache2/src/cache.rs @@ -396,7 +396,8 @@ impl Cache { public_result(ErrorOperation::Drain, self.data_plane.drain_async().await) } - /// Returns a lock-free operational snapshot. Activity and I/O counters are + /// Returns an operational snapshot using atomics and, when enabled, a short + /// fill-controller lock. Activity and I/O counters are /// cumulative for this open and are populated only when /// `RuntimeOptions::stats.activity_counters` is enabled; health and resource gauges are /// always available. diff --git a/cache2/src/config/runtime.rs b/cache2/src/config/runtime.rs index 56a0d90..061c270 100644 --- a/cache2/src/config/runtime.rs +++ b/cache2/src/config/runtime.rs @@ -353,6 +353,40 @@ pub enum ReadAdmission { }, } +/// Optional pre-timeout pressure observation and fill admission control. +#[non_exhaustive] +#[derive(Clone, Copy, Debug, Default, Eq, PartialEq)] +pub enum FillControlOptions { + /// No controller, timer, or additional request accounting. + #[default] + Disabled, + /// Report pressure and hypothetical rejections without changing admission. + Observe(AdaptiveFillOptions), + /// Shape new fills while allowing accepted writes and essential reclaim. + Adaptive(AdaptiveFillOptions), +} + +/// Logical fill-rate ceilings for the optional adaptive controller. +/// These are not physical-device bandwidth or IOPS guarantees. +#[non_exhaustive] +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub struct AdaptiveFillOptions { + /// Maximum encoded fill bytes per second, from 640 through 1 TiB/s. + pub max_bytes_per_second: u64, + /// Maximum fill records per second, from 10 through 655350. + pub max_operations_per_second: u32, +} + +impl AdaptiveFillOptions { + /// Creates unchecked rate ceilings. [`CacheConfig::new`] validates them. + pub const fn new(max_bytes_per_second: u64, max_operations_per_second: u32) -> Self { + Self { + max_bytes_per_second, + max_operations_per_second, + } + } +} + /// Process-local resource choices, checked together by [`CacheConfig::new`]. /// /// These values may change across opens. Warm recovery rebinds append shards @@ -387,6 +421,9 @@ pub struct RuntimeOptions { /// can wait indefinitely with `None`. Actual I/O errors and invalid /// completions still fail the instance. pub io_recovery_timeout: Option, + /// Optional pre-timeout fill pressure control. Disabled by default. + /// Enabled modes reserve a monitoring thread and bounded worker observations. + pub fill_control: FillControlOptions, /// Hash-routed append paths, from 1 through 256 (default 4). Each needs one /// Active Region, two Region-sized buffers, and a worker. The layout also needs a /// spare Region. @@ -421,6 +458,7 @@ impl Default for RuntimeOptions { read_admission: ReadAdmission::Immediate, reclaim_io_timeout: Duration::from_secs(5), io_recovery_timeout: None, + fill_control: FillControlOptions::Disabled, append_shards: DEFAULT_APPEND_SHARDS, l1_capacity_bytes: DEFAULT_L1_CAPACITY_BYTES, l1_eviction_policy: L1EvictionPolicy::Clock, @@ -496,6 +534,11 @@ impl CacheConfig { let index_slots = storage.index_slots; runtime.resolve()?; let stats_bytes = Recorder::allocation_bytes(runtime.stats)?; + let fill_bytes = crate::io::fill_control::FillController::allocation_bytes( + runtime.fill_control, + runtime.append_shards as usize + + IoPoolTopology::reclaim(runtime.io_engine).max_in_flight(), + )?; if geometry.region_count <= runtime.append_shards { return Err(invalid_config( "append shards require valid geometry with one Active Region each plus one spare Region", @@ -511,6 +554,7 @@ impl CacheConfig { let fixed_bytes = runtime_fixed_memory_bytes(index_slots, geometry.region_count)? .checked_add(l1_metadata_bytes) .and_then(|bytes| bytes.checked_add(stats_bytes)) + .and_then(|bytes| bytes.checked_add(fill_bytes)) .ok_or_else(|| invalid_config("fixed memory requirements overflow"))?; let (reserved_memory_bytes, minimum_memory_bytes) = runtime.memory_requirements(geometry, fixed_bytes)?; @@ -802,6 +846,44 @@ mod tests { use crate::ErrorKind; use crate::StorageOptions; + #[test] + fn fill_control_validates_ceilings_and_accounts_monitor_memory() { + let storage = StorageOptions::new(1024 * 1024 * 1024).build().unwrap(); + let base = CacheConfig::new(storage.clone(), RuntimeOptions::default()).unwrap(); + assert_eq!(base.runtime().fill_control, FillControlOptions::Disabled); + for (bytes, operations) in [(639, 100), ((1 << 40) + 1, 100), (640, 9), (640, 655_351)] { + let options = RuntimeOptions { + fill_control: FillControlOptions::Adaptive(AdaptiveFillOptions::new( + bytes, operations, + )), + ..RuntimeOptions::default() + }; + assert_eq!( + CacheConfig::new(storage.clone(), options) + .unwrap_err() + .kind(), + ErrorKind::InvalidInput + ); + } + let mut minimum = None; + for mode in [FillControlOptions::Observe, FillControlOptions::Adaptive] { + let config = CacheConfig::new( + storage.clone(), + RuntimeOptions { + fill_control: mode(AdaptiveFillOptions::new(64_000, 100)), + ..RuntimeOptions::default() + }, + ) + .unwrap(); + let extra = config.minimum_memory_bytes() - base.minimum_memory_bytes(); + assert!(extra > CACHE_THREAD_STACK_BYTES); + if let Some(previous) = minimum { + assert_eq!(extra, previous); + } + minimum = Some(extra); + } + } + #[test] fn recovery_timeout_allows_zero_and_rejects_overflow() { let storage = crate::StorageOptions::new(1024 * 1024 * 1024) diff --git a/cache2/src/io/engine/mod.rs b/cache2/src/io/engine/mod.rs index 6eff683..ef1058d 100644 --- a/cache2/src/io/engine/mod.rs +++ b/cache2/src/io/engine/mod.rs @@ -778,7 +778,10 @@ impl BoundedIoRequest { let original = self.deadline; loop { self.request = match self.request.wait_until(self.deadline) { - Ok(completion) => return Ok(completion), + Ok(completion) => { + recovery.returned(); + return Ok(completion); + } Err(request) => request, }; match recovery.next_deadline(original) { @@ -958,6 +961,12 @@ pub fn submit_background_io( timeout: Duration, recovery: &mut RecoveryAttempt<'_>, ) -> Result { + let bytes = match &operation { + IoOperation::Read { buffer, .. } | IoOperation::Write { buffer, .. } => buffer.len() as u64, + }; + if let Err(error) = recovery.start(bytes, timeout) { + return Err(SubmitError { error, operation }); + } let original = Instant::now() .checked_add(timeout) .unwrap_or_else(Instant::now); @@ -965,6 +974,7 @@ pub fn submit_background_io( loop { match submit_cache_io_until(engine, operation, deadline, CACHE_IO_CANCEL_GRACE) { Ok(mut request) => { + recovery.admitted(); request.deadline = original; return Ok(request); } diff --git a/cache2/src/io/engine/recovery.rs b/cache2/src/io/engine/recovery.rs index 23ea450..5c9d8f8 100644 --- a/cache2/src/io/engine/recovery.rs +++ b/cache2/src/io/engine/recovery.rs @@ -14,17 +14,22 @@ //! Reversible background timeout recovery, separate from the health latch. +use std::sync::Arc; use std::sync::atomic::AtomicBool; use std::sync::atomic::AtomicUsize; use std::sync::atomic::Ordering; use std::time::Duration; use std::time::Instant; +use crate::io::fill_control::FillController; +use crate::io::fill_control::Progress; + /// Shared across background workers. Resource ownership remains with each worker. pub struct BackgroundRecovery { timeout: Option, pending: AtomicUsize, stopped: AtomicBool, + pub controller: Option>, } impl BackgroundRecovery { @@ -33,6 +38,17 @@ impl BackgroundRecovery { timeout, pending: AtomicUsize::new(0), stopped: AtomicBool::new(false), + controller: None, + } + } + + pub fn with_controller( + timeout: Option, + controller: Option>, + ) -> Self { + Self { + controller, + ..Self::new(timeout) } } @@ -40,6 +56,7 @@ impl BackgroundRecovery { RecoveryAttempt { recovery: self, entered: false, + progress: None, } } @@ -49,6 +66,9 @@ impl BackgroundRecovery { pub fn stop(&self) { self.stopped.store(true, Ordering::Release); + if let Some(control) = &self.controller { + control.stop(); + } } } @@ -57,9 +77,28 @@ impl BackgroundRecovery { pub struct RecoveryAttempt<'a> { recovery: &'a BackgroundRecovery, entered: bool, + progress: Option>, } impl RecoveryAttempt<'_> { + pub fn start(&mut self, bytes: u64, timeout: Duration) -> std::io::Result<()> { + if let Some(control) = &self.recovery.controller { + self.progress = Some(control.observe(bytes, timeout)?); + } + Ok(()) + } + + pub fn admitted(&self) { + if let Some(progress) = &self.progress { + progress.admitted(); + } + } + pub fn returned(&self) { + if let Some(progress) = &self.progress { + progress.returned(); + } + } + /// Poll completion/admission at fixed one-second intervals, without extending /// a configured total budget. Shutdown also terminates unlimited recovery. pub fn next_deadline(&mut self, original: Instant) -> Option { @@ -80,6 +119,9 @@ impl RecoveryAttempt<'_> { }; if !self.entered { self.entered = true; + if let Some(control) = &self.recovery.controller { + control.set_recovering(true); + } if self.recovery.pending.fetch_add(1, Ordering::AcqRel) == 0 { log::warn!(target: "cache2::health", event = "cache_io_recovery_started"; "background I/O timed out; pausing cache fills while retaining owned requests"); @@ -89,8 +131,14 @@ impl RecoveryAttempt<'_> { } /// Called only after operation-result validation and publication succeed. - pub fn finish(self) { + pub fn finish(mut self) { + if let Some(progress) = self.progress.take() { + progress.finish(); + } if self.entered && self.recovery.pending.fetch_sub(1, Ordering::AcqRel) == 1 { + if let Some(control) = &self.recovery.controller { + control.set_recovering(false); + } log::info!(target: "cache2::health", event = "cache_io_recovery_completed"; "all timed-out background operations recovered and passed validation"); } diff --git a/cache2/src/io/engine/tests.rs b/cache2/src/io/engine/tests.rs index 69dc461..f0b84ee 100644 --- a/cache2/src/io/engine/tests.rs +++ b/cache2/src/io/engine/tests.rs @@ -1391,3 +1391,71 @@ fn shutdown_interrupts_unlimited_recovery_without_releasing_pending_write() { }); engine.shutdown().unwrap(); } + +#[test] +fn adaptive_pressure_pauses_before_real_engine_timeout_and_resumes_after_validation() { + use crate::AdaptiveFillOptions; + use crate::FillControlOptions; + use crate::FillPressure; + use crate::io::fill_control::FillController; + for enforcing in [false, true] { + let settings = AdaptiveFillOptions::new(1024 * 1024, 1000); + let mode = if enforcing { + FillControlOptions::Adaptive(settings) + } else { + FillControlOptions::Observe(settings) + }; + let control = FillController::new(mode, 1, 4096).unwrap().unwrap(); + let monitor = control.start().unwrap(); + let recovery = BackgroundRecovery::with_controller(None, Some(Arc::clone(&control))); + let backend = Arc::new(BlockingBackend::default()); + let engine = BackendIoEngine::new(backend.clone(), 1).unwrap(); + let memory = managed_memory(); + std::thread::scope(|scope| { + let (returned_tx, returned_rx) = mpsc::channel(); + let (validate_tx, validate_rx) = mpsc::channel(); + let recovery = &recovery; + let engine = &engine; + let memory = &memory; + scope.spawn(move || { + let mut attempt = recovery.attempt(); + let request = submit_background_io( + engine, + IoOperation::write(WritePoint::Record, write_buffer(memory, &[5; 4096]), 0), + Duration::from_secs(4), + &mut attempt, + ) + .unwrap(); + let completion = request.wait_with_recovery(engine, &mut attempt).unwrap(); + returned_tx.send(completion).unwrap(); + validate_rx.recv().unwrap(); + attempt.finish(); + }); + assert!(backend.wait_for_entered(1)); + let deadline = Instant::now() + Duration::from_secs(2); + while control.snapshot().pressure != FillPressure::Paused && Instant::now() < deadline { + std::thread::sleep(Duration::from_millis(10)); + } + let paused = control.snapshot(); + let recovered = recovery.is_recovering(); + let admission = control.try_admit(64).is_some(); + backend.release(); + let completion = returned_rx.recv_timeout(Duration::from_secs(2)).unwrap(); + assert_eq!(paused.pressure, FillPressure::Paused); + assert!(!recovered, "must react before the normal I/O timeout"); + assert_eq!(admission, !enforcing); + assert_eq!(paused.outstanding_bytes, 4096); + assert!(completion.into_io_result().0.is_ok()); + validate_tx.send(()).unwrap(); + }); + let deadline = Instant::now() + Duration::from_secs(2); + while control.snapshot().pressure == FillPressure::Paused && Instant::now() < deadline { + std::thread::sleep(Duration::from_millis(10)); + } + assert_eq!(control.snapshot().pressure, FillPressure::Throttled); + assert!(control.try_admit(64).is_some()); + assert_eq!(lock_unpoisoned(&backend.state).entered, 1); + engine.shutdown().unwrap(); + drop(monitor); + } +} diff --git a/cache2/src/io/fill_control.rs b/cache2/src/io/fill_control.rs new file mode 100644 index 0000000..8b65c7d --- /dev/null +++ b/cache2/src/io/fill_control.rs @@ -0,0 +1,786 @@ +// Copyright 2026 ScopeDB, Inc. +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! Bounded pre-timeout observation and nonblocking fill admission. + +use std::io; +use std::sync::Arc; +use std::sync::Condvar; +use std::sync::Mutex; +use std::sync::MutexGuard; +use std::sync::atomic::AtomicBool; +use std::sync::atomic::AtomicU64; +use std::sync::atomic::Ordering; +use std::thread::JoinHandle; +use std::time::Duration; +use std::time::Instant; + +use crate::AdaptiveFillOptions; +use crate::FillControlOptions; +use crate::FillControlSnapshot; +use crate::FillPressure; +use crate::managed_memory::CACHE_THREAD_STACK_BYTES; + +const TICK: Duration = Duration::from_millis(100); +const UNIT: u64 = 64; +const COUNT_MASK: u64 = 0xffff; +const CREDIT_MASK: u64 = (1 << 48) - 1; +const MAX_CAS_ATTEMPTS: usize = 4; + +#[derive(Clone, Copy)] +struct Observation { + start: Instant, + admitted: Option, + returned: Option, + bytes: u64, + timeout: Duration, +} + +struct State { + slots: Box<[Option]>, + snapshot: FillControlSnapshot, + last_tick: Instant, + last_progress: Instant, + previous_pending: usize, + completed_bytes: u64, + completed_ops: u64, + sampled_bytes: u64, + sampled_ops: u64, + window_start: Instant, + service_bytes: f64, + service_ops: f64, + clean_ticks: u32, + was_recovering: bool, +} + +pub struct FillController { + options: AdaptiveFillOptions, + enforcing: bool, + max_units: u64, + max_ops: u64, + // Epoch:16, operations:16, 64-byte units:32. One CAS reserves both dimensions. + credit: AtomicU64, + epoch: AtomicU64, + staging_pressure: AtomicBool, + paused: AtomicBool, + stopped: AtomicBool, + recovering: AtomicBool, + rejections: AtomicU64, + would_reject: AtomicU64, + state: Mutex, + wake: Condvar, +} + +pub struct FillMonitor { + control: Arc, + worker: Option>, +} + +impl Drop for FillMonitor { + fn drop(&mut self) { + self.control.stop(); + if let Some(worker) = self.worker.take() { + let _ = worker.join(); + } + } +} + +impl FillController { + pub fn allocation_bytes(options: FillControlOptions, slots: usize) -> io::Result { + let settings = match options { + FillControlOptions::Disabled => return Ok(0), + FillControlOptions::Observe(settings) | FillControlOptions::Adaptive(settings) => { + settings + } + }; + if !(640..=1 << 40).contains(&settings.max_bytes_per_second) + || !(10..=655_350).contains(&settings.max_operations_per_second) + { + return Err(io::Error::new( + io::ErrorKind::InvalidInput, + "fill rate ceilings are out of range", + )); + } + slots + .checked_mul(size_of::>()) + .and_then(|bytes| bytes.checked_add(size_of::() + CACHE_THREAD_STACK_BYTES + 256)) + .ok_or_else(|| { + io::Error::new( + io::ErrorKind::InvalidInput, + "fill controller memory overflow", + ) + }) + } + + pub fn new( + options: FillControlOptions, + slots: usize, + max_record: u64, + ) -> io::Result>> { + Self::allocation_bytes(options, slots)?; + let (settings, enforcing) = match options { + FillControlOptions::Disabled => return Ok(None), + FillControlOptions::Observe(settings) => (settings, false), + FillControlOptions::Adaptive(settings) => (settings, true), + }; + let max_units = (settings.max_bytes_per_second / 10) + .max(max_record) + .div_ceil(UNIT); + if max_units > u64::from(u32::MAX) { + return Err(io::Error::new( + io::ErrorKind::InvalidInput, + "fill burst exceeds budget representation", + )); + } + let max_ops = u64::from(settings.max_operations_per_second / 10).max(1); + let mut observations = Vec::new(); + observations.try_reserve_exact(slots).map_err(|_| { + io::Error::new( + io::ErrorKind::OutOfMemory, + "cannot allocate fill observations", + ) + })?; + observations.resize(slots, None); + let now = Instant::now(); + Ok(Some(Arc::new(Self { + options: settings, + enforcing, + max_units, + max_ops, + credit: AtomicU64::new((max_ops << 32) | max_units), + epoch: AtomicU64::new(0), + staging_pressure: AtomicBool::new(false), + paused: AtomicBool::new(false), + stopped: AtomicBool::new(false), + recovering: AtomicBool::new(false), + rejections: AtomicU64::new(0), + would_reject: AtomicU64::new(0), + state: Mutex::new(State { + slots: observations.into_boxed_slice(), + snapshot: FillControlSnapshot { + pressure: FillPressure::Healthy, + enforcing, + bytes_per_second: settings.max_bytes_per_second, + operations_per_second: settings.max_operations_per_second, + ..FillControlSnapshot::default() + }, + last_tick: now, + last_progress: now, + previous_pending: 0, + completed_bytes: 0, + completed_ops: 0, + sampled_bytes: 0, + sampled_ops: 0, + window_start: now, + service_bytes: 0., + service_ops: 0., + clean_ticks: 0, + was_recovering: false, + }), + wake: Condvar::new(), + }))) + } + + pub fn start(self: &Arc) -> io::Result { + let control = Arc::clone(self); + let worker = std::thread::Builder::new() + .name("cache2-fill-control".into()) + .stack_size(CACHE_THREAD_STACK_BYTES) + .spawn(move || { + let mut state = control.lock(); + while !control.stopped.load(Ordering::Acquire) { + state = control + .wake + .wait_timeout(state, TICK) + .unwrap_or_else(|p| p.into_inner()) + .0; + let now = Instant::now(); + if now.saturating_duration_since(state.last_tick) >= TICK { + control.tick(&mut state, now); + } + } + })?; + Ok(FillMonitor { + control: Arc::clone(self), + worker: Some(worker), + }) + } + + fn lock(&self) -> MutexGuard<'_, State> { + self.state.lock().unwrap_or_else(|p| p.into_inner()) + } + + pub fn stop(&self) { + let _state = self.lock(); + self.stopped.store(true, Ordering::Release); + self.paused.store(true, Ordering::Release); + self.wake.notify_all(); + } + + pub fn set_recovering(&self, recovering: bool) { + self.recovering.store(recovering, Ordering::Release); + if recovering && self.enforcing { + self.paused.store(true, Ordering::Release); + } + } + + pub fn suppress_reinsertion(&self) -> bool { + self.enforcing && self.lock().snapshot.pressure != FillPressure::Healthy + } + + pub fn staging_busy(&self) { + self.staging_pressure.store(true, Ordering::Relaxed); + } + + pub fn snapshot(&self) -> FillControlSnapshot { + let mut snapshot = self.lock().snapshot; + snapshot.rejections = self.rejections.load(Ordering::Relaxed); + snapshot.would_reject = self.would_reject.load(Ordering::Relaxed); + snapshot + } + + pub fn try_admit(&self, bytes: u64) -> Option> { + let units = bytes.div_ceil(UNIT); + let epoch = self.epoch.load(Ordering::Acquire); + let mut value = self.credit.load(Ordering::Relaxed); + if !self.paused.load(Ordering::Acquire) && !self.stopped.load(Ordering::Acquire) { + for _ in 0..MAX_CAS_ATTEMPTS { + if value >> 48 != epoch & COUNT_MASK || self.epoch.load(Ordering::Acquire) != epoch + { + break; + } + if value & u64::from(u32::MAX) < units || (value >> 32) & COUNT_MASK == 0 { + break; + } + let next = value - units - (1 << 32); + match self.credit.compare_exchange_weak( + value, + next, + Ordering::AcqRel, + Ordering::Relaxed, + ) { + Ok(_) => { + return Some(FillPermit { + control: self, + units, + epoch, + committed: false, + }); + } + Err(current) => value = current, + } + } + } + if self.enforcing { + self.rejections.fetch_add(1, Ordering::Relaxed); + None + } else { + self.would_reject.fetch_add(1, Ordering::Relaxed); + Some(FillPermit { + control: self, + units: 0, + epoch: 0, + committed: true, + }) + } + } + + pub fn observe(&self, bytes: u64, timeout: Duration) -> io::Result> { + let mut state = self.lock(); + let now = Instant::now(); + if state.slots.iter().all(Option::is_none) { + state.last_progress = now; + } + let index = state + .slots + .iter() + .position(Option::is_none) + .ok_or_else(|| io::Error::other("background observation capacity exhausted"))?; + state.slots[index] = Some(Observation { + start: now, + admitted: None, + returned: None, + bytes, + timeout, + }); + self.wake.notify_one(); + Ok(Progress { + control: self, + index, + succeeded: false, + }) + } + + fn tick(&self, state: &mut State, now: Instant) { + state.last_tick = now; + let elapsed = now + .saturating_duration_since(state.window_start) + .as_secs_f64(); + let decision = elapsed >= 0.5; + let mut made_progress = false; + if decision { + let bytes = state.completed_bytes.saturating_sub(state.sampled_bytes); + let ops = state.completed_ops.saturating_sub(state.sampled_ops); + made_progress = ops != 0; + state.service_bytes = bytes as f64 / elapsed; + state.service_ops = ops as f64 / elapsed; + state.sampled_bytes = state.completed_bytes; + state.sampled_ops = state.completed_ops; + state.window_start = now; + } + let mut pending = 0; + let mut bytes = 0_u64; + let mut oldest = Duration::ZERO; + let mut deadline = Duration::from_secs(5); + let mut aged = false; + for observation in state.slots.iter().flatten() { + pending += 1; + bytes = bytes.saturating_add(observation.bytes); + let age = now.saturating_duration_since(observation.start); + oldest = oldest.max(age); + deadline = deadline.min(observation.timeout); + aged |= age >= observation.timeout / 4; + } + let drain = if pending != 0 && state.service_bytes > 0. && state.service_ops > 0. { + (bytes as f64 / state.service_bytes).max(pending as f64 / state.service_ops) + } else { + 0. + }; + let no_progress = pending != 0 + && now.saturating_duration_since(state.last_progress) + >= (deadline / 2).min(Duration::from_millis(500)); + let recovering = self.recovering.load(Ordering::Acquire); + let pressure = recovering || aged || no_progress || drain >= deadline.as_secs_f64() / 4.; + let previous = state.snapshot.pressure; + let staging_busy = decision && self.staging_pressure.swap(false, Ordering::Relaxed); + if pressure { + state.snapshot.pressure = FillPressure::Paused; + state.clean_ticks = 0; + } else if previous == FillPressure::Paused || state.was_recovering { + if pending == 0 + || (made_progress && oldest < deadline / 10 && drain < deadline.as_secs_f64() / 10.) + { + state.snapshot.pressure = FillPressure::Throttled; + state.snapshot.bytes_per_second = (state.snapshot.bytes_per_second / 2).max(640); + state.snapshot.operations_per_second = + (state.snapshot.operations_per_second / 2).max(10); + if state.service_bytes > 0. { + state.snapshot.bytes_per_second = state + .snapshot + .bytes_per_second + .min((state.service_bytes * 0.8) as u64) + .max(640); + state.snapshot.operations_per_second = state + .snapshot + .operations_per_second + .min((state.service_ops * 0.8) as u32) + .max(10); + } + state.clean_ticks = 0; + } + } else if staging_busy && pending >= state.previous_pending && pending != 0 { + state.snapshot.pressure = FillPressure::Throttled; + state.snapshot.bytes_per_second = (state.snapshot.bytes_per_second * 7 / 10).max(640); + state.snapshot.operations_per_second = + (state.snapshot.operations_per_second * 7 / 10).max(10); + state.clean_ticks = 0; + } else if made_progress { + state.clean_ticks += 1; + if state.clean_ticks >= 3 { + state.snapshot.bytes_per_second = (state.snapshot.bytes_per_second + + self.options.max_bytes_per_second / 40) + .min(self.options.max_bytes_per_second); + state.snapshot.operations_per_second = (state.snapshot.operations_per_second + + self.options.max_operations_per_second.div_ceil(40)) + .min(self.options.max_operations_per_second); + if state.snapshot.bytes_per_second == self.options.max_bytes_per_second + && state.snapshot.operations_per_second + == self.options.max_operations_per_second + { + state.snapshot.pressure = FillPressure::Healthy; + } + } + } + state.was_recovering = recovering; + if decision { + state.previous_pending = pending; + } + state.snapshot.outstanding_operations = pending as u64; + state.snapshot.outstanding_bytes = bytes; + state.snapshot.oldest_operation_ns = nanos(oldest); + state.snapshot.estimated_drain_ns = (drain * 1e9).min(u64::MAX as f64) as u64; + self.paused.store( + state.snapshot.pressure == FillPressure::Paused, + Ordering::Release, + ); + // Publishing a fresh epoch bounds refunds and prevents accumulating missed ticks. + let clear = + previous != state.snapshot.pressure || state.snapshot.pressure == FillPressure::Paused; + let add_units = state.snapshot.bytes_per_second / 10 / UNIT; + let add_ops = u64::from(state.snapshot.operations_per_second / 10); + let epoch = self.epoch.fetch_add(1, Ordering::AcqRel).wrapping_add(1); + let _ = self + .credit + .try_update(Ordering::AcqRel, Ordering::Relaxed, |old| { + let units = if clear { 0 } else { old & u64::from(u32::MAX) }; + let ops = if clear { 0 } else { (old >> 32) & COUNT_MASK }; + let next_epoch = (epoch & COUNT_MASK) << 48; + let (units, ops) = if state.snapshot.pressure == FillPressure::Paused { + (0, 0) + } else { + ( + (units + add_units).min(self.max_units), + (ops + add_ops).min(self.max_ops), + ) + }; + Some(next_epoch | (ops << 32) | units) + }); + if previous != state.snapshot.pressure { + log::info!(target: "cache2::health", event = "cache_fill_pressure_changed", pressure:? = state.snapshot.pressure; + "cache fill admission pressure changed"); + } + } +} + +fn nanos(duration: Duration) -> u64 { + duration.as_nanos().min(u128::from(u64::MAX)) as u64 +} + +pub struct FillPermit<'a> { + control: &'a FillController, + units: u64, + epoch: u64, + committed: bool, +} +impl FillPermit<'_> { + pub fn commit(mut self) { + self.committed = true; + } +} +impl Drop for FillPermit<'_> { + fn drop(&mut self) { + if self.committed { + return; + } + let mut value = self.control.credit.load(Ordering::Relaxed); + for _ in 0..MAX_CAS_ATTEMPTS { + if self.control.epoch.load(Ordering::Acquire) != self.epoch + || value >> 48 != self.epoch & COUNT_MASK + { + return; + } + let units = ((value & u64::from(u32::MAX)) + self.units).min(self.control.max_units); + let ops = (((value >> 32) & COUNT_MASK) + 1).min(self.control.max_ops); + let next = (value & !CREDIT_MASK) | (ops << 32) | units; + match self.control.credit.compare_exchange_weak( + value, + next, + Ordering::AcqRel, + Ordering::Relaxed, + ) { + Ok(_) => return, + Err(current) => value = current, + } + } + } +} + +/// One worker-owned observation. Failed work is removed, never marked successful. +pub struct Progress<'a> { + control: &'a FillController, + index: usize, + succeeded: bool, +} +impl Progress<'_> { + pub fn admitted(&self) { + self.control.lock().slots[self.index] + .as_mut() + .unwrap() + .admitted = Some(Instant::now()); + } + pub fn returned(&self) { + self.control.lock().slots[self.index] + .as_mut() + .unwrap() + .returned = Some(Instant::now()); + } + pub fn finish(mut self) { + self.succeeded = true; + } +} +impl Drop for Progress<'_> { + fn drop(&mut self) { + let mut state = self.control.lock(); + let observation = state.slots[self.index].take().expect("live progress slot"); + let now = Instant::now(); + if self.succeeded { + state.completed_ops = state.completed_ops.saturating_add(1); + state.completed_bytes = state.completed_bytes.saturating_add(observation.bytes); + state.last_progress = now; + } + let admission_end = observation.admitted.unwrap_or(now); + state.snapshot.admission_ns = state.snapshot.admission_ns.saturating_add(nanos( + admission_end.saturating_duration_since(observation.start), + )); + if let Some(admitted) = observation.admitted { + let wait_end = observation.returned.unwrap_or(now); + state.snapshot.completion_wait_ns = state + .snapshot + .completion_wait_ns + .saturating_add(nanos(wait_end.saturating_duration_since(admitted))); + if let Some(returned) = observation.returned { + state.snapshot.validation_ns = state + .snapshot + .validation_ns + .saturating_add(nanos(now.saturating_duration_since(returned))); + } + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn control(enforcing: bool) -> Arc { + let settings = AdaptiveFillOptions::new(64_000, 100); + FillController::new( + if enforcing { + FillControlOptions::Adaptive(settings) + } else { + FillControlOptions::Observe(settings) + }, + 4, + 4096, + ) + .unwrap() + .unwrap() + } + + fn tick(control: &FillController, elapsed: Duration) { + let mut state = control.lock(); + let now = state.last_tick + elapsed; + control.tick(&mut state, now); + } + + #[test] + fn budget_reserves_both_dimensions_and_refunds_failed_staging() { + let control = control(true); + let initial = control.credit.load(Ordering::Relaxed); + let permit = control.try_admit(640).unwrap(); + assert_eq!( + control.credit.load(Ordering::Relaxed), + initial - 10 - (1 << 32) + ); + drop(permit); + assert_eq!(control.credit.load(Ordering::Relaxed), initial); + for _ in 0..10 { + control.try_admit(64).unwrap().commit(); + } + assert!(control.try_admit(64).is_none()); + tick(&control, TICK); + assert!(control.try_admit(64).is_some()); + } + + #[test] + fn staging_pressure_reduces_once_per_window_and_idle_does_not_raise_rates() { + let control = control(true); + let pending = control.observe(64, Duration::from_secs(30)).unwrap(); + control.staging_busy(); + tick(&control, TICK); + assert_eq!(control.snapshot().bytes_per_second, 64_000); + { + let mut state = control.lock(); + state.last_progress = state.last_tick + Duration::from_millis(500); + } + tick(&control, Duration::from_millis(500)); + assert_eq!(control.snapshot().pressure, FillPressure::Throttled); + assert_eq!(control.snapshot().bytes_per_second, 44_800); + assert_eq!(control.snapshot().operations_per_second, 70); + drop(pending); + tick(&control, Duration::from_secs(10)); + assert_eq!(control.snapshot().bytes_per_second, 44_800); + for sample in 0..3 { + { + let mut state = control.lock(); + state.completed_bytes += 64; + state.completed_ops += 1; + } + tick(&control, Duration::from_millis(500)); + assert_eq!( + control.snapshot().bytes_per_second, + if sample < 2 { 44_800 } else { 46_400 } + ); + } + } + + #[test] + fn old_refund_cannot_mint_new_epoch_credits_even_after_low_bits_wrap() { + let control = control(true); + let permit = control.try_admit(64).unwrap(); + control.epoch.store(COUNT_MASK, Ordering::Release); + tick(&control, TICK); + let before = control.credit.load(Ordering::Relaxed); + drop(permit); + assert_eq!(control.credit.load(Ordering::Relaxed), before); + } + + #[test] + fn old_request_is_detected_while_other_requests_progress() { + let control = control(true); + let old = control.observe(4096, Duration::from_secs(2)).unwrap(); + old.admitted(); + let fast = control.observe(64, Duration::from_secs(2)).unwrap(); + fast.admitted(); + fast.returned(); + fast.finish(); + { + // Many small completions keep aggregate progress healthy. Only + // the retained request's age should trigger the pause. + let mut state = control.lock(); + state.last_progress = state.last_tick + Duration::from_millis(600); + state.completed_bytes = 1_000_000; + state.completed_ops = 1_000; + } + tick(&control, Duration::from_millis(600)); + assert_eq!(control.snapshot().pressure, FillPressure::Paused); + assert!(control.try_admit(64).is_none()); + old.returned(); + assert!( + control.try_admit(64).is_none(), + "delivery does not release the admission fence" + ); + old.finish(); + tick(&control, TICK); + assert_eq!(control.snapshot().pressure, FillPressure::Throttled); + assert!(control.try_admit(64).is_some()); + } + + #[test] + fn observe_never_rejects_or_suppresses_reinsertion() { + let control = control(false); + let _old = control.observe(4096, Duration::from_secs(2)).unwrap(); + tick(&control, Duration::from_millis(600)); + assert_eq!(control.snapshot().pressure, FillPressure::Paused); + for _ in 0..20 { + control.try_admit(4096).unwrap().commit(); + } + assert_eq!(control.snapshot().would_reject, 20); + assert_eq!(control.snapshot().rejections, 0); + assert!(!control.suppress_reinsertion()); + } + + #[test] + fn idle_time_does_not_inflate_credit_or_look_like_a_stall() { + let control = control(true); + tick(&control, Duration::from_secs(3600)); + assert_eq!(control.snapshot().pressure, FillPressure::Healthy); + assert_eq!( + control.credit.load(Ordering::Relaxed) & CREDIT_MASK, + (control.max_ops << 32) | control.max_units + ); + let _new = control.observe(4096, Duration::from_secs(2)).unwrap(); + // Production ticks and observations use the same clock. + let mut state = control.lock(); + control.tick(&mut state, Instant::now()); + assert_eq!(state.snapshot.pressure, FillPressure::Healthy); + } + + #[test] + fn large_record_remains_eligible_and_stop_cannot_reopen_admission() { + let control = FillController::new( + FillControlOptions::Adaptive(AdaptiveFillOptions::new(640, 10)), + 1, + 4096, + ) + .unwrap() + .unwrap(); + control.try_admit(4096).unwrap().commit(); + for _ in 0..64 { + tick(&control, TICK); + } + assert!(control.try_admit(4096).is_some()); + control.stop(); + tick(&control, TICK); + assert!(control.try_admit(64).is_none()); + } + + #[test] + fn recovering_fence_survives_empty_observation_table() { + let control = control(true); + control.set_recovering(true); + tick(&control, TICK); + assert_eq!(control.snapshot().pressure, FillPressure::Paused); + assert!(control.try_admit(64).is_none()); + control.set_recovering(false); + tick(&control, TICK); + assert_eq!(control.snapshot().pressure, FillPressure::Throttled); + } + + #[test] + fn idle_monitor_can_be_stopped_without_io() { + let control = control(true); + let monitor = control.start().unwrap(); + drop(monitor); + assert!(control.stopped.load(Ordering::Acquire)); + } + + #[test] + fn failed_observations_release_capacity_without_reporting_success() { + let control = FillController::new( + FillControlOptions::Observe(AdaptiveFillOptions::new(640, 10)), + 1, + 4096, + ) + .unwrap() + .unwrap(); + let pending = control.observe(4096, Duration::from_secs(5)).unwrap(); + assert!(control.observe(64, Duration::from_secs(5)).is_err()); + // A failed admission still contributes its elapsed phase time. + control.lock().slots[0].as_mut().unwrap().start -= Duration::from_secs(1); + drop(pending); + assert!(control.snapshot().admission_ns >= 1_000_000_000); + let pending = control.observe(4096, Duration::from_secs(5)).unwrap(); + pending.admitted(); + control.lock().slots[0].as_mut().unwrap().admitted = + Some(Instant::now() - Duration::from_secs(1)); + drop(pending); + assert!(control.snapshot().completion_wait_ns >= 1_000_000_000); + let state = control.lock(); + assert_eq!(state.completed_ops, 0); + assert_eq!(state.completed_bytes, 0); + assert!(state.slots[0].is_none()); + } + + #[test] + fn concurrent_admission_never_exceeds_shared_budget() { + let control = control(true); + let accepted = AtomicU64::new(0); + std::thread::scope(|scope| { + for _ in 0..8 { + let control = &control; + let accepted = &accepted; + scope.spawn(move || { + for _ in 0..100 { + if let Some(permit) = control.try_admit(640) { + permit.commit(); + accepted.fetch_add(1, Ordering::Relaxed); + } + } + }); + } + }); + assert_eq!(accepted.load(Ordering::Relaxed), 10); + } +} diff --git a/cache2/src/io/mod.rs b/cache2/src/io/mod.rs index 9086108..7bda564 100644 --- a/cache2/src/io/mod.rs +++ b/cache2/src/io/mod.rs @@ -16,3 +16,4 @@ pub mod backend; pub mod engine; +pub mod fill_control; diff --git a/cache2/src/lib.rs b/cache2/src/lib.rs index 8c73851..076e522 100644 --- a/cache2/src/lib.rs +++ b/cache2/src/lib.rs @@ -35,6 +35,8 @@ pub use self::cache::Value; mod config; pub use self::config::CacheConfig; pub use self::config::StorageLayout; +pub use self::config::runtime::AdaptiveFillOptions; +pub use self::config::runtime::FillControlOptions; pub use self::config::runtime::IoEngineOptions; pub use self::config::runtime::IoMode; pub use self::config::runtime::IoUringOptions; @@ -56,6 +58,8 @@ pub use self::snapshot::CacheL1Snapshot; pub use self::snapshot::CacheReclaimSnapshot; pub use self::snapshot::CacheSnapshot; pub use self::snapshot::DetailedCacheSnapshot; +pub use self::snapshot::FillControlSnapshot; +pub use self::snapshot::FillPressure; pub use self::snapshot::RegionSnapshot; pub use self::snapshot::StartupMode; diff --git a/cache2/src/region/runtime/metrics.rs b/cache2/src/region/runtime/metrics.rs index cbe0cf2..2c03f2d 100644 --- a/cache2/src/region/runtime/metrics.rs +++ b/cache2/src/region/runtime/metrics.rs @@ -215,6 +215,7 @@ impl RuntimeMetrics { CacheSnapshot { metrics_epoch: self.metrics_epoch, health, + fill_control: crate::snapshot::FillControlSnapshot::default(), activity_counters_enabled, puts, deletes, diff --git a/cache2/src/region/runtime/mod.rs b/cache2/src/region/runtime/mod.rs index 6bd22c6..767724d 100644 --- a/cache2/src/region/runtime/mod.rs +++ b/cache2/src/region/runtime/mod.rs @@ -61,6 +61,8 @@ use crate::io::engine::ReadSlotWaiter; use crate::io::engine::build_file_engine; use crate::io::engine::recovery::BackgroundRecovery; use crate::io::engine::submit_background_io; +use crate::io::fill_control::FillController; +use crate::io::fill_control::FillMonitor; use crate::managed_memory::BufferLease; use crate::managed_memory::CACHE_THREAD_STACK_BYTES; use crate::managed_memory::ManagedMemory; @@ -427,6 +429,7 @@ struct RunningOwner { shared: Arc, shard_workers: Vec>, reclaim_workers: Vec>, + fill_monitor: Option, } struct RunningShared { @@ -839,6 +842,19 @@ impl RegionDataPlane { return Err(write_overload_error()); } }; + let fill_permit = if let Some(controller) = &running.recovery.controller { + match controller.try_admit(u64::from(record_bytes)) { + Some(permit) => Some(permit), + None => { + if running.activity_counters { + running.metrics.record_write_rejection(); + } + return Err(write_overload_error()); + } + } + } else { + None + }; let staged = self.core.try_stage_value( &running.staging, shard_id, @@ -853,6 +869,9 @@ impl RegionDataPlane { previous_bytes, current_bytes, } => { + if let Some(permit) = fill_permit { + permit.commit(); + } if ADMIT_L1 { let _published = running.memory.publish(hash, key, value, seqno); } else { @@ -875,9 +894,15 @@ impl RegionDataPlane { Ok(seqno) } RegionStageValue::NeedsProgress => { + if let Some(controller) = &running.recovery.controller { + controller.staging_busy(); + } reject_staged_write(running, control, WAKE_URGENT, operation) } RegionStageValue::NeedsRotation => { + if let Some(controller) = &running.recovery.controller { + controller.staging_busy(); + } reject_staged_write(running, control, WAKE_ROTATE | WAKE_URGENT, operation) } } @@ -1296,6 +1321,9 @@ impl RegionDataPlane { { snapshot.health = crate::snapshot::CacheHealth::Recovering; } + if let Some(controller) = &running.recovery.controller { + snapshot.fill_control = controller.snapshot(); + } snapshot.io = aggregate_io_stats( &running.read_engines, &running.write_engines, @@ -1497,6 +1525,12 @@ fn start_running( io::Error::new(io::ErrorKind::OutOfMemory, "cannot allocate shard controls") })?; shards.resize_with(shard_count, || Arc::new(ShardControl::new())); + let controller = FillController::new( + runtime.fill_control, + shard_count + reclaim_worker_count, + data.geometry.region_size, + )?; + let fill_monitor = controller.as_ref().map(FillController::start).transpose()?; let shared = Arc::new(RunningShared { core, read_engines, @@ -1506,7 +1540,7 @@ fn start_running( reclaim_engines, reclaim_control: ReclaimControl::new(), reclaim_io_timeout: runtime.reclaim_io_timeout, - recovery: BackgroundRecovery::new(runtime.io_recovery_timeout), + recovery: BackgroundRecovery::with_controller(runtime.io_recovery_timeout, controller), managed_memory, metrics, memory, @@ -1589,6 +1623,7 @@ fn start_running( shared, shard_workers, reclaim_workers, + fill_monitor, }) } @@ -1816,7 +1851,12 @@ fn reclaim_worker_result( // Keep one completion boundary per source Region while each // reclaimer rotates through a disjoint subset of append shards. let reinsert_shard = reinsert_shards.take(); - let preserve_hot = shared.core.reclaim_can_reinsert()?; + let preserve_hot = shared.core.reclaim_can_reinsert()? + && !shared + .recovery + .controller + .as_ref() + .is_some_and(|control| control.suppress_reinsertion()); let reinsert_operation = if preserve_hot { shared.operations.try_enter() } else { @@ -2073,6 +2113,7 @@ async fn drain_shards_async(shared: &RunningShared, stop: bool) -> io::Result<() fn stop_running(mut owner: RunningOwner) -> io::Result { owner.shared.recovery.stop(); + drop(owner.fill_monitor.take()); let drain = drain_shards(&owner.shared, true); let mut join_error = None; for worker in owner.shard_workers.drain(..) { diff --git a/cache2/src/region/runtime/shutdown_tests.rs b/cache2/src/region/runtime/shutdown_tests.rs index d0310a0..d2379af 100644 --- a/cache2/src/region/runtime/shutdown_tests.rs +++ b/cache2/src/region/runtime/shutdown_tests.rs @@ -178,6 +178,7 @@ fn assert_close_does_not_wait_for_read(submit_before_close: bool) { shared, shard_workers: vec![], reclaim_workers: vec![], + fill_monitor: None, }); tx.send(result).unwrap(); }); @@ -272,3 +273,85 @@ fn recovery_rejects_fills_preserves_reads_and_waits_for_all_workers() { store.close_fast().unwrap(); std::fs::remove_dir_all(root).unwrap(); } + +#[test] +fn adaptive_pressure_preserves_reads_and_deletes_and_resumes_fills() { + use crate::region::file_backend::FileRegionBackend; + use crate::region::file_backend::RegionFiles; + use crate::region::recovery::PersistentId; + use crate::region::store::RegionStore; + use crate::snapshot::CacheHealth; + let root = env::temp_dir().join(format!("cache2-fill-admission-{}", std::process::id())); + std::fs::create_dir_all(&root).unwrap(); + let files = RegionFiles::new(root.join("data"), root.join("state"), root.join("image")); + let data = DataSuperblock { + generation: 1, + cache_uuid: PersistentId::from_bytes([1; 16]).unwrap(), + data_identity: PersistentId::from_bytes([2; 16]).unwrap(), + geometry: DataGeometry { + data_file_len: DataGeometry::expected_file_len(4096, 4).unwrap(), + region_size: 4096, + region_count: 4, + }, + hash_seed: 3, + storage_fingerprint: 4, + }; + let config = RuntimeOptions { + fill_control: crate::FillControlOptions::Adaptive(crate::AdaptiveFillOptions::new( + 1_048_576, 1000, + )), + append_shards: 1, + l1_capacity_bytes: 0, + ..RuntimeOptions::default() + }; + let mut store = RegionStore::open( + 8, + FileRegionBackend::for_test_with_options(files, data, 8, config), + ) + .unwrap(); + let plane = store.data_plane_handle().unwrap(); + plane.put(b"existing", b"value").unwrap(); + plane.drain().unwrap(); + let controller = plane.shared.recovery.controller.as_ref().unwrap(); + let pending = controller + .observe(4096, Duration::from_millis(400)) + .unwrap(); + pending.admitted(); + let deadline = Instant::now() + Duration::from_secs(2); + while controller.snapshot().pressure != crate::FillPressure::Paused { + assert!(Instant::now() < deadline, "monitor did not pause fills"); + std::thread::sleep(Duration::from_millis(1)); + } + let snapshot = plane.snapshot().unwrap(); + assert_eq!(snapshot.health, CacheHealth::Running); + assert_eq!(snapshot.fill_control.pressure, crate::FillPressure::Paused); + assert_eq!( + plane.put(b"new", b"value").unwrap_err().kind(), + io::ErrorKind::WouldBlock + ); + assert_eq!( + plane.put_l2(b"new", b"value").unwrap_err().kind(), + io::ErrorKind::WouldBlock + ); + assert_eq!(plane.get(b"existing").unwrap().unwrap().value(), b"value"); + plane.delete(b"existing").unwrap(); + assert!(plane.get(b"existing").unwrap().is_none()); + pending.returned(); + assert!(plane.put(b"new", b"value").is_err()); + pending.finish(); + let deadline = Instant::now() + Duration::from_secs(2); + loop { + match plane.put(b"new", b"value") { + Ok(_) => break, + Err(error) => { + assert_eq!(error.kind(), io::ErrorKind::WouldBlock); + assert!(Instant::now() < deadline, "fills did not resume"); + std::thread::sleep(Duration::from_millis(1)); + } + } + } + plane.drain().unwrap(); + assert_eq!(plane.get(b"new").unwrap().unwrap().value(), b"value"); + store.close_fast().unwrap(); + std::fs::remove_dir_all(root).unwrap(); +} diff --git a/cache2/src/snapshot.rs b/cache2/src/snapshot.rs index 28bd540..ec50710 100644 --- a/cache2/src/snapshot.rs +++ b/cache2/src/snapshot.rs @@ -39,9 +39,57 @@ pub enum CacheHealth { Failed, } -/// Lock-free point-in-time operational counters and cache-owned resource -/// accounting. Counters are process-local and reset on every open. Concurrent -/// updates may appear across fields at slightly different instants. +/// Admission pressure, independent of terminal cache health. +#[non_exhaustive] +#[derive(Clone, Copy, Debug, Default, Eq, PartialEq)] +pub enum FillPressure { + /// Pressure observation is disabled. + #[default] + Disabled, + /// Configured rate ceilings apply without a pressure reduction. + Healthy, + /// Fill rate has been reduced while accepted work drains. + Throttled, + /// New fills are paused while outstanding work recovers. + Paused, +} + +/// Always available when fill control is enabled, independent of statistics. +#[non_exhaustive] +#[derive(Clone, Copy, Debug, Default, Eq, PartialEq)] +pub struct FillControlSnapshot { + /// Current measured pressure; Observe mode does not enforce it. + pub pressure: FillPressure, + /// Whether the controller enforces its admission decisions. + pub enforcing: bool, + /// Current encoded-byte admission rate. + pub bytes_per_second: u64, + /// Current record admission rate. + pub operations_per_second: u32, + /// Fills rejected by the controller, including bounded CAS contention. + pub rejections: u64, + /// Fills that Observe mode would have rejected. + pub would_reject: u64, + /// Background operations awaiting completion or validation. + pub outstanding_operations: u64, + /// Bytes held by those background operations; excludes unflushed staging. + pub outstanding_bytes: u64, + /// Age of the oldest background operation, including admission/validation. + pub oldest_operation_ns: u64, + /// Estimated drain time using recent validated background throughput. + pub estimated_drain_ns: u64, + /// Accumulated background admission time in nanoseconds. + pub admission_ns: u64, + /// Accumulated time from admission to result consumption, including scheduling. + pub completion_wait_ns: u64, + /// Accumulated time from result consumption through validation/publication. + pub validation_ns: u64, +} + +/// Point-in-time operational counters and cache-owned resource accounting. +/// Sampling uses atomics, plus a short controller lock when fill control is enabled. Counters are +/// process-local and reset on every open. Concurrent updates may appear across fields at slightly +/// different instants. #[non_exhaustive] #[derive(Clone, Copy, Debug, Eq, PartialEq)] pub struct CacheSnapshot { @@ -51,6 +99,8 @@ pub struct CacheSnapshot { pub metrics_epoch: u64, /// Current cache availability. pub health: CacheHealth, + /// Pre-timeout fill pressure and controller accounting. + pub fill_control: FillControlSnapshot, /// Whether optional cumulative activity and I/O counters are enabled. pub activity_counters_enabled: bool, /// Accepted `put` and `put_l2` operations. @@ -245,7 +295,7 @@ pub struct RegionSnapshot { #[non_exhaustive] #[derive(Clone, Debug, Eq, PartialEq)] pub struct DetailedCacheSnapshot { - /// Lock-free summary sampled for this diagnostic. + /// Operational summary sampled for this diagnostic. pub summary: CacheSnapshot, /// Mutations rejected specifically because an append buffer needed progress. pub write_buffer_rejections: u64, From 8071d72164b4524f7148bd8c6e5d5caede2fe58b Mon Sep 17 00:00:00 2001 From: leiysky Date: Sun, 20 Sep 2026 21:06:52 +0800 Subject: [PATCH 2/5] refactor: fail-open fill observations and rename FillLimits A full observation table must not fail cache I/O, and the 100 ms monitor must not be reset by each submit. Observe would_reject now tracks pause and budget policy only. Shared rate ceilings are FillLimits rather than AdaptiveFillOptions. --- ARCHITECTURE.md | 4 +- CHANGELOG.md | 6 +- CONFIGURATION.md | 12 +- README.md | 1 + benchmarks/src/config.rs | 2 +- benchmarks/src/report.rs | 5 +- cache2/ERRORS.md | 2 +- cache2/src/config/runtime.rs | 26 +- cache2/src/io/engine/mod.rs | 21 +- cache2/src/io/engine/recovery.rs | 50 ++- cache2/src/io/engine/tests.rs | 6 +- cache2/src/io/fill_control.rs | 410 ++++++++++++-------- cache2/src/lib.rs | 2 +- cache2/src/region/runtime/mod.rs | 26 +- cache2/src/region/runtime/shutdown_tests.rs | 16 +- cache2/src/snapshot.rs | 12 +- 16 files changed, 353 insertions(+), 248 deletions(-) diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index 876805b..20b3b2e 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -116,9 +116,9 @@ Each lane uses one concrete `IoEngine` for admission, submission, cancellation, ### Pre-timeout fill pressure -`io::fill_control` owns optional background observations and adaptive fill admission. `Disabled` retains the existing path. Enabled modes preallocate a fixed observation table sized by append and reclaim worker counts and run one accounted monitoring thread. Background workers register before engine admission and retain their observations through completion validation and publication. A 100 ms monitor can detect old or stalled work without requiring another completion; aggregate validated throughput provides an approximate drain estimate. Pressure does not change terminal health or prove a device fault. +`io::fill_control` owns optional background observations and adaptive fill admission. `Disabled` retains the existing path. Enabled modes preallocate a fixed observation table sized by append and reclaim worker counts and run one accounted monitoring thread. Background workers register before engine admission and retain their observations through completion validation and publication. A full table skips that request and counts `dropped_observations` instead of failing I/O. A 100 ms monitor poll can detect old or stalled work without requiring another completion or a per-submit wakeup; aggregate validated throughput provides an approximate drain estimate. Pressure does not change terminal health or prove a device fault. -A packed atomic token bucket reserves both encoded bytes and record count before foreground staging. Each reservation/refund makes at most four CAS attempts; full refill epochs prevent stale permits from refunding a later budget. No per-fill clock, allocation, or controller mutex is required. Failed staging can refund its reservation; accepted work always keeps its resources and proceeds without reacquiring rate credit. `Observe` records hypothetical rejection only; `Adaptive` rejects new fills immediately and skips optional reinsertion under pressure. Reads, deletes, and essential reclaim bypass fill budgets. A short mutex protects only background observations, monitor decisions, and the optional controller snapshot. Close signals and joins the monitor independently of I/O completion. +A packed atomic token bucket reserves both encoded bytes and record count before foreground staging. Each reservation/refund makes at most four CAS attempts; full refill epochs prevent stale permits from refunding a later budget. No per-fill clock, allocation, or controller mutex is required. Failed staging can refund its reservation; accepted work always keeps its resources and proceeds without reacquiring rate credit. `Observe` records pause and budget refusals as `would_reject`; Adaptive also rejects bounded CAS contention. `Adaptive` rejects new fills immediately and skips optional reinsertion under pressure. Reads, deletes, and essential reclaim bypass fill budgets. A short mutex protects only background observations, monitor decisions, and the optional controller snapshot. Close signals and joins the monitor independently of I/O completion. See [adaptive fill admission](CONFIGURATION.md#adaptive-fill-admission) for rate ceilings, pause/resume thresholds, bounded bursts, phase accounting, and tuning limits. diff --git a/CHANGELOG.md b/CHANGELOG.md index d1a8d34..5660d78 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,9 +7,11 @@ - All optional statistics now use `RuntimeOptions::stats`: replace `runtime.statistics` with `runtime.stats.activity_counters` and `CacheSnapshot::statistics_enabled` with `activity_counters_enabled`. Activity counters, terminal request counters, and latency collection remain independent and disabled by default; counter semantics and the on-disk format are unchanged. See [the configuration migration guide](CONFIGURATION.md#migrating-from-05). - The benchmarking `RegionIndexTurnoverReport` field `config` is renamed to `options`, matching the `RegionIndexTurnoverOptions` input it carries. -### Improvements +### Features -- Optional `RuntimeOptions::fill_control` observes background request age, validated progress, and estimated drain time before timeout. `Observe` reports hypothetical rejections; `Adaptive` bounds new fills by encoded bytes and record count, pauses on stalled work, resumes gradually, and suppresses optional reinsertion under pressure. Reads, deletes, accepted writes, and essential reclaim retain their paths. `CacheSnapshot::fill_control` exposes pressure and accounting independently of statistics; enabled snapshots take a short controller lock. Controller storage and its monitoring stack are included in managed memory. Disabled by default; see [configuration](CONFIGURATION.md#adaptive-fill-admission) for ceilings and measurement limits. +- Optional `RuntimeOptions::fill_control` observes background request age, validated progress, and estimated drain time before timeout. `Observe` reports pause and budget refusals as `would_reject`; `Adaptive` bounds new fills by encoded bytes and record count, pauses on stalled work, resumes gradually, and suppresses optional reinsertion under pressure. A full observation table skips that request instead of failing I/O. Reads, deletes, accepted writes, and essential reclaim retain their paths. `CacheSnapshot::fill_control` exposes pressure and accounting independently of statistics; enabled snapshots take a short controller lock. Controller storage and its monitoring stack are included in managed memory. Disabled by default; see [configuration](CONFIGURATION.md#adaptive-fill-admission) for ceilings and measurement limits. + +### Improvements - Background write and reclaim timeouts now enter a reversible `CacheHealth::Recovering` state: new cache fills return overload while existing reads and deletes remain available. Original requests keep their bounded buffers and Regions and are never resubmitted; all affected work must complete validation and publication before fills resume. `RuntimeOptions::io_recovery_timeout` defaults to `None` for recovery until completion or close; use `Some(duration)` to bound recovery or `Some(Duration::ZERO)` for immediate cancellation. Recovery checks at fixed one-second intervals. Close interrupts recovery and preserves the existing unfenced-write safeguards; drain may wait indefinitely. Actual I/O errors and invalid completions remain terminal. diff --git a/CONFIGURATION.md b/CONFIGURATION.md index dc6638a..96a01b7 100644 --- a/CONFIGURATION.md +++ b/CONFIGURATION.md @@ -431,20 +431,20 @@ Set `options.reclaim_io_timeout = Duration::from_secs(30)` after constructing `R `RuntimeOptions::fill_control` defaults to `FillControlOptions::Disabled`. `Observe` records pressure and hypothetical rejections while retaining ordinary admission. `Adaptive` enforces the same decisions for new `put` and `put_l2` fills, returning `ErrorKind::Overloaded` immediately when paused or over budget. Neither mode changes foreground reads, deletes, accepted flushes, essential reclaim, or the existing timeout recovery policy. Adaptive pressure also suppresses optional hot-record reinsertion. Pressure is separate from `CacheHealth`: a pre-timeout pause leaves a healthy cache `Running`. ```rust -use cache2::{AdaptiveFillOptions, FillControlOptions, RuntimeOptions}; +use cache2::{FillControlOptions, FillLimits, RuntimeOptions}; let mut options = RuntimeOptions::default(); // Example ceilings; choose values for the measured workload. -let limits = AdaptiveFillOptions::new(256 * 1024 * 1024, 20_000); +let limits = FillLimits::new(256 * 1024 * 1024, 20_000); options.fill_control = FillControlOptions::Observe(limits); // After evaluating pressure and would_reject, enforce the same policy: options.fill_control = FillControlOptions::Adaptive(limits); ``` -The two ceilings constrain logical encoded fill bytes (charged in 64-byte units) and fill records, not device bandwidth or physical IOPS. Valid ceilings are 640 bytes/s through 1 TiB/s and 10 through 655350 records/s. Budgets refill every 100 ms, with capacity for 100 ms of the configured ceilings and at least one maximum-size record. An idle or delayed monitor never accumulates more than that bounded burst. Failed staging refunds only within the current refill epoch. Admission uses bounded atomic retries and may conservatively reject contention; it never waits for rate credit. +The two ceilings constrain logical encoded fill bytes (charged in 64-byte units) and fill records, not device bandwidth or physical IOPS. Valid ceilings are 640 bytes/s through 1 TiB/s and 10 through 655350 records/s. Budgets refill every 100 ms, with capacity for 100 ms of the configured ceilings and at least one maximum-size record. An idle monitor never accumulates more than that bounded burst. Failed staging refunds only within the current refill epoch. Admission uses bounded atomic retries and never waits for rate credit. Adaptive may conservatively reject CAS or epoch contention; Observe counts `would_reject` only for pause and exhausted budgets. -A dedicated monitor checks outstanding background work every 100 ms even if no I/O completes. It pauses when any request reaches one quarter of its normal deadline, estimated drain time reaches one quarter of the shortest observed deadline (capped at five seconds), or outstanding work makes no validated progress for the smaller of 500 ms and half that deadline. These are conservative pressure signals: very short deadlines or a descheduled monitor can still reach timeout first. Staging saturation together with a nonshrinking outstanding queue reduces both rates by 30% at most once per 500 ms. A paused controller resumes when work drains, or validated progress brings age and estimated drain time below one tenth of the deadline, starting at half its previous rates and at most 80% of measured service rates when available. After three progressing 500 ms samples, rates increase by approximately 2.5% of the ceilings per progressing sample (rounded to whole bytes and records per second), up to the ceilings. Floors of 640 bytes/s and 10 records/s retain probe capacity; idle time alone never increases the rates. Normal timeout recovery keeps its admission fence until all affected work is validated and published. +A dedicated monitor polls outstanding background work every 100 ms even if no I/O completes. Background submissions do not reset that interval. It pauses when any request reaches one quarter of its normal deadline, estimated drain time reaches one quarter of the shortest observed deadline (capped at five seconds), or outstanding work makes no validated progress for the smaller of 500 ms and half that deadline. These are conservative pressure signals: very short deadlines or a descheduled monitor can still reach timeout first. Staging saturation together with a nonshrinking outstanding queue reduces both rates by 30% at most once per 500 ms. A paused controller resumes when work drains, or validated progress brings age and estimated drain time below one tenth of the deadline, starting at half its previous rates and at most 80% of measured service rates when available. After three progressing 500 ms samples, rates increase by approximately 2.5% of the ceilings per progressing sample (rounded to whole bytes and records per second), up to the ceilings. Floors of 640 bytes/s and 10 records/s retain probe capacity; idle time alone never increases the rates. Normal timeout recovery keeps its admission fence until all affected work is validated and published. -`CacheSnapshot::fill_control` reports pressure, enforcement, current rates, rejections, hypothetical rejections, outstanding bytes/operations, oldest age, and estimated drain time independently of `RuntimeOptions::stats`. The monitor tracks background requests through admission, result consumption, and validation/publication, including scheduling delays. Its cumulative `admission_ns`, `completion_wait_ns`, and `validation_ns` measure those phases. Outstanding bytes exclude unflushed staging; drain time uses aggregate recent validated background throughput, including reclaim reads. A zero drain estimate means no estimate is available when work is pending. These measurements cannot distinguish device throttling from CPU scheduling, kernel queueing, or slow validation; correlate them with host metrics before diagnosing hardware. `cache_fill_pressure_changed` logs state transitions under `cache2::health`. +`CacheSnapshot::fill_control` reports pressure, enforcement, current rates, rejections, hypothetical rejections, dropped observations, outstanding bytes/operations, oldest age, and estimated drain time independently of `RuntimeOptions::stats`. The monitor tracks background requests through admission, result consumption, and validation/publication, including scheduling delays. Its cumulative `admission_ns`, `completion_wait_ns`, and `validation_ns` measure those phases. Outstanding bytes exclude unflushed staging; drain time uses aggregate recent validated background throughput, including reclaim reads. A zero drain estimate means no estimate is available when work is pending. A full observation table skips that request rather than failing I/O. These measurements cannot distinguish device throttling from CPU scheduling, kernel queueing, or slow validation; correlate them with host metrics before diagnosing hardware. `cache_fill_pressure_changed` logs state transitions under `cache2::health`. -Enabled modes preallocate one observation per append/reclaim worker and one 512 KiB monitoring stack, all included in `CacheConfig::minimum_memory_bytes()`. Background observations and controller snapshots take a short lock over this bounded table. Foreground fill admission uses atomics without allocation, clock reads, or locks from the controller. The monitor wakes at 100 ms intervals, including while idle. Close stops it independently of outstanding I/O. Disabled mode creates no controller, observations, or monitoring thread. +Enabled modes preallocate one observation per append/reclaim worker and one 512 KiB monitoring stack, all included in `CacheConfig::minimum_memory_bytes()`. Background observations and controller snapshots take a short lock over this bounded table. Foreground fill admission uses atomics without allocation, clock reads, or locks from the controller. Reinsertion suppression reads atomics. The monitor waits out each 100 ms interval, including while idle. Close stops it independently of outstanding I/O. Disabled mode creates no controller, observations, or monitoring thread. diff --git a/README.md b/README.md index b0fc2f2..bd582e5 100644 --- a/README.md +++ b/README.md @@ -83,6 +83,7 @@ See the [configuration guide](CONFIGURATION.md#configuration-lifecycle) for exam | Memory | `managed_memory_limit_bytes` | 1 GiB across cache-managed allocations. | | I/O mode | `io_mode` | Buffered I/O. | | Metrics | `stats: StatsOptions` | Health/resource gauges always available; activity, request, and latency collection opt in. | +| Fill control | `fill_control: FillControlOptions` | Disabled. `Observe` reports pause/budget pressure; `Adaptive` rejects new fills before timeout. | Changing the append-shard count rebinds recovered Active Regions during a warm open. Growth uses available Free Regions; when there are not enough, the disposable cache safely starts empty. diff --git a/benchmarks/src/config.rs b/benchmarks/src/config.rs index e4ee19b..67daa01 100644 --- a/benchmarks/src/config.rs +++ b/benchmarks/src/config.rs @@ -70,7 +70,7 @@ pub fn fill_control_from_env(prefix: &str) -> io::Result Ok(cache2::FillControlOptions::Observe(options)), "adaptive" => Ok(cache2::FillControlOptions::Adaptive(options)), diff --git a/benchmarks/src/report.rs b/benchmarks/src/report.rs index b334871..3df5609 100644 --- a/benchmarks/src/report.rs +++ b/benchmarks/src/report.rs @@ -481,16 +481,17 @@ pub fn emit_cache_report( ); let fill = cache.fill_control; println!( - "report version=2 type=fill_control benchmark={} scenario={} phase={} pressure={:?} enforcing={} bytes_per_second={} operations_per_second={} rejections={} would_reject={} outstanding_operations={} outstanding_bytes={} oldest_operation_ns={} estimated_drain_ns={}", + "report version=2 type=fill_control benchmark={} scenario={} phase={} pressure={:?} enforcing={} bytes_per_second={} records_per_second={} rejections={} would_reject={} dropped_observations={} outstanding_operations={} outstanding_bytes={} oldest_operation_ns={} estimated_drain_ns={}", benchmark, scenario, phase, fill.pressure, fill.enforcing, fill.bytes_per_second, - fill.operations_per_second, + fill.records_per_second, fill.rejections, fill.would_reject, + fill.dropped_observations, fill.outstanding_operations, fill.outstanding_bytes, fill.oldest_operation_ns, diff --git a/cache2/ERRORS.md b/cache2/ERRORS.md index 0fdb40f..4084bd1 100644 --- a/cache2/ERRORS.md +++ b/cache2/ERRORS.md @@ -32,7 +32,7 @@ fn cache_value(cache: &Cache, key: &[u8], value: &[u8]) -> Result<(), Error> { Do not retry without a bound. C² deliberately exposes pressure instead of building unbounded queues. With the default immediate-read policy, read-pool or buffer pressure is `Ok(None)`. When read waiting is enabled, queue saturation, buffer pressure, and deadline expiry are `ErrorKind::Overloaded`. -With `FillControlOptions::Adaptive`, new fills also return `Overloaded` when the controller pauses or exhausts its byte/record budget. Inspect `CacheSnapshot::fill_control` to distinguish controller rejection from other admission pressure. Pre-timeout pressure does not change `CacheHealth::Running` or identify a hardware failure. `Observe` only counts hypothetical controller rejections. Reads, deletes, accepted writes, and essential reclaim retain their ordinary paths. +With `FillControlOptions::Adaptive`, new fills also return `Overloaded` when the controller pauses or exhausts its byte/record budget. Inspect `CacheSnapshot::fill_control` to distinguish controller rejection from other admission pressure. Pre-timeout pressure does not change `CacheHealth::Running` or identify a hardware failure. `Observe` counts `would_reject` for pause and budget refusals only. Reads, deletes, accepted writes, and essential reclaim retain their ordinary paths. ## Classifications diff --git a/cache2/src/config/runtime.rs b/cache2/src/config/runtime.rs index 061c270..60a8bf4 100644 --- a/cache2/src/config/runtime.rs +++ b/cache2/src/config/runtime.rs @@ -361,28 +361,28 @@ pub enum FillControlOptions { #[default] Disabled, /// Report pressure and hypothetical rejections without changing admission. - Observe(AdaptiveFillOptions), - /// Shape new fills while allowing accepted writes and essential reclaim. - Adaptive(AdaptiveFillOptions), + Observe(FillLimits), + /// Reject new fills under pressure. Accepted writes and essential reclaim continue. + Adaptive(FillLimits), } -/// Logical fill-rate ceilings for the optional adaptive controller. -/// These are not physical-device bandwidth or IOPS guarantees. +/// Logical fill-rate ceilings shared by [`FillControlOptions::Observe`] and +/// [`FillControlOptions::Adaptive`]. These are not device bandwidth or IOPS guarantees. #[non_exhaustive] #[derive(Clone, Copy, Debug, Eq, PartialEq)] -pub struct AdaptiveFillOptions { +pub struct FillLimits { /// Maximum encoded fill bytes per second, from 640 through 1 TiB/s. pub max_bytes_per_second: u64, /// Maximum fill records per second, from 10 through 655350. - pub max_operations_per_second: u32, + pub max_records_per_second: u32, } -impl AdaptiveFillOptions { +impl FillLimits { /// Creates unchecked rate ceilings. [`CacheConfig::new`] validates them. - pub const fn new(max_bytes_per_second: u64, max_operations_per_second: u32) -> Self { + pub const fn new(max_bytes_per_second: u64, max_records_per_second: u32) -> Self { Self { max_bytes_per_second, - max_operations_per_second, + max_records_per_second, } } } @@ -853,9 +853,7 @@ mod tests { assert_eq!(base.runtime().fill_control, FillControlOptions::Disabled); for (bytes, operations) in [(639, 100), ((1 << 40) + 1, 100), (640, 9), (640, 655_351)] { let options = RuntimeOptions { - fill_control: FillControlOptions::Adaptive(AdaptiveFillOptions::new( - bytes, operations, - )), + fill_control: FillControlOptions::Adaptive(FillLimits::new(bytes, operations)), ..RuntimeOptions::default() }; assert_eq!( @@ -870,7 +868,7 @@ mod tests { let config = CacheConfig::new( storage.clone(), RuntimeOptions { - fill_control: mode(AdaptiveFillOptions::new(64_000, 100)), + fill_control: mode(FillLimits::new(64_000, 100)), ..RuntimeOptions::default() }, ) diff --git a/cache2/src/io/engine/mod.rs b/cache2/src/io/engine/mod.rs index ef1058d..ebf21a0 100644 --- a/cache2/src/io/engine/mod.rs +++ b/cache2/src/io/engine/mod.rs @@ -779,14 +779,27 @@ impl BoundedIoRequest { loop { self.request = match self.request.wait_until(self.deadline) { Ok(completion) => { - recovery.returned(); + recovery.completed(); return Ok(completion); } Err(request) => request, }; match recovery.next_deadline(original) { Some(deadline) => self.deadline = deadline, - None => return self.wait(engine), + None => { + return match self.wait(engine) { + Ok(completion) => { + recovery.completed(); + Ok(completion) + } + Err(exceeded) => { + if exceeded.completion.is_some() { + recovery.completed(); + } + Err(exceeded) + } + }; + } } } } @@ -964,9 +977,7 @@ pub fn submit_background_io( let bytes = match &operation { IoOperation::Read { buffer, .. } | IoOperation::Write { buffer, .. } => buffer.len() as u64, }; - if let Err(error) = recovery.start(bytes, timeout) { - return Err(SubmitError { error, operation }); - } + recovery.start(bytes, timeout); let original = Instant::now() .checked_add(timeout) .unwrap_or_else(Instant::now); diff --git a/cache2/src/io/engine/recovery.rs b/cache2/src/io/engine/recovery.rs index 5c9d8f8..4cb9920 100644 --- a/cache2/src/io/engine/recovery.rs +++ b/cache2/src/io/engine/recovery.rs @@ -22,14 +22,14 @@ use std::time::Duration; use std::time::Instant; use crate::io::fill_control::FillController; -use crate::io::fill_control::Progress; +use crate::io::fill_control::Observation; /// Shared across background workers. Resource ownership remains with each worker. pub struct BackgroundRecovery { timeout: Option, pending: AtomicUsize, stopped: AtomicBool, - pub controller: Option>, + pub fill: Option>, } impl BackgroundRecovery { @@ -38,16 +38,13 @@ impl BackgroundRecovery { timeout, pending: AtomicUsize::new(0), stopped: AtomicBool::new(false), - controller: None, + fill: None, } } - pub fn with_controller( - timeout: Option, - controller: Option>, - ) -> Self { + pub fn with_fill(timeout: Option, fill: Option>) -> Self { Self { - controller, + fill, ..Self::new(timeout) } } @@ -56,7 +53,7 @@ impl BackgroundRecovery { RecoveryAttempt { recovery: self, entered: false, - progress: None, + observation: None, } } @@ -66,8 +63,8 @@ impl BackgroundRecovery { pub fn stop(&self) { self.stopped.store(true, Ordering::Release); - if let Some(control) = &self.controller { - control.stop(); + if let Some(fill) = &self.fill { + fill.stop(); } } } @@ -77,25 +74,24 @@ impl BackgroundRecovery { pub struct RecoveryAttempt<'a> { recovery: &'a BackgroundRecovery, entered: bool, - progress: Option>, + observation: Option>, } impl RecoveryAttempt<'_> { - pub fn start(&mut self, bytes: u64, timeout: Duration) -> std::io::Result<()> { - if let Some(control) = &self.recovery.controller { - self.progress = Some(control.observe(bytes, timeout)?); + pub fn start(&mut self, bytes: u64, timeout: Duration) { + if let Some(fill) = &self.recovery.fill { + self.observation = fill.observe(bytes, timeout); } - Ok(()) } pub fn admitted(&self) { - if let Some(progress) = &self.progress { - progress.admitted(); + if let Some(observation) = &self.observation { + observation.admitted(); } } - pub fn returned(&self) { - if let Some(progress) = &self.progress { - progress.returned(); + pub fn completed(&self) { + if let Some(observation) = &self.observation { + observation.completed(); } } @@ -119,8 +115,8 @@ impl RecoveryAttempt<'_> { }; if !self.entered { self.entered = true; - if let Some(control) = &self.recovery.controller { - control.set_recovering(true); + if let Some(fill) = &self.recovery.fill { + fill.set_recovering(true); } if self.recovery.pending.fetch_add(1, Ordering::AcqRel) == 0 { log::warn!(target: "cache2::health", event = "cache_io_recovery_started"; @@ -132,12 +128,12 @@ impl RecoveryAttempt<'_> { /// Called only after operation-result validation and publication succeed. pub fn finish(mut self) { - if let Some(progress) = self.progress.take() { - progress.finish(); + if let Some(observation) = self.observation.take() { + observation.finish(); } if self.entered && self.recovery.pending.fetch_sub(1, Ordering::AcqRel) == 1 { - if let Some(control) = &self.recovery.controller { - control.set_recovering(false); + if let Some(fill) = &self.recovery.fill { + fill.set_recovering(false); } log::info!(target: "cache2::health", event = "cache_io_recovery_completed"; "all timed-out background operations recovered and passed validation"); diff --git a/cache2/src/io/engine/tests.rs b/cache2/src/io/engine/tests.rs index f0b84ee..56eb1da 100644 --- a/cache2/src/io/engine/tests.rs +++ b/cache2/src/io/engine/tests.rs @@ -1394,12 +1394,12 @@ fn shutdown_interrupts_unlimited_recovery_without_releasing_pending_write() { #[test] fn adaptive_pressure_pauses_before_real_engine_timeout_and_resumes_after_validation() { - use crate::AdaptiveFillOptions; use crate::FillControlOptions; + use crate::FillLimits; use crate::FillPressure; use crate::io::fill_control::FillController; for enforcing in [false, true] { - let settings = AdaptiveFillOptions::new(1024 * 1024, 1000); + let settings = FillLimits::new(1024 * 1024, 1000); let mode = if enforcing { FillControlOptions::Adaptive(settings) } else { @@ -1407,7 +1407,7 @@ fn adaptive_pressure_pauses_before_real_engine_timeout_and_resumes_after_validat }; let control = FillController::new(mode, 1, 4096).unwrap().unwrap(); let monitor = control.start().unwrap(); - let recovery = BackgroundRecovery::with_controller(None, Some(Arc::clone(&control))); + let recovery = BackgroundRecovery::with_fill(None, Some(Arc::clone(&control))); let backend = Arc::new(BlockingBackend::default()); let engine = BackendIoEngine::new(backend.clone(), 1).unwrap(); let memory = managed_memory(); diff --git a/cache2/src/io/fill_control.rs b/cache2/src/io/fill_control.rs index 8b65c7d..964a540 100644 --- a/cache2/src/io/fill_control.rs +++ b/cache2/src/io/fill_control.rs @@ -20,35 +20,70 @@ use std::sync::Condvar; use std::sync::Mutex; use std::sync::MutexGuard; use std::sync::atomic::AtomicBool; +use std::sync::atomic::AtomicU8; use std::sync::atomic::AtomicU64; use std::sync::atomic::Ordering; use std::thread::JoinHandle; use std::time::Duration; use std::time::Instant; -use crate::AdaptiveFillOptions; use crate::FillControlOptions; use crate::FillControlSnapshot; +use crate::FillLimits; use crate::FillPressure; use crate::managed_memory::CACHE_THREAD_STACK_BYTES; const TICK: Duration = Duration::from_millis(100); +const TICKS_PER_SECOND: u64 = 10; const UNIT: u64 = 64; const COUNT_MASK: u64 = 0xffff; -const CREDIT_MASK: u64 = (1 << 48) - 1; const MAX_CAS_ATTEMPTS: usize = 4; +const MIN_BYTES_PER_SECOND: u64 = 640; +const MAX_BYTES_PER_SECOND: u64 = 1 << 40; +const MIN_RECORDS_PER_SECOND: u32 = 10; +const MAX_RECORDS_PER_SECOND: u32 = 655_350; +const DECISION_WINDOW: Duration = Duration::from_millis(500); +const STALL_CAP: Duration = Duration::from_millis(500); +const MAX_DRAIN_DEADLINE: Duration = Duration::from_secs(5); +const CLEAN_DECISIONS: u32 = 3; +const RAMP_DIVISOR: u32 = 40; +const STAGING_KEEP_NUM: u64 = 7; +const STAGING_KEEP_DEN: u64 = 10; +const SERVICE_HEADROOM: f64 = 0.8; + +fn packed_epoch(value: u64) -> u64 { + value >> 48 +} +fn packed_ops(value: u64) -> u64 { + (value >> 32) & COUNT_MASK +} +fn packed_units(value: u64) -> u64 { + value & u64::from(u32::MAX) +} +fn pack_credit(epoch: u64, ops: u64, units: u64) -> u64 { + (epoch & COUNT_MASK) << 48 | (ops << 32) | units +} + +fn encode_pressure(pressure: FillPressure) -> u8 { + match pressure { + FillPressure::Disabled => 0, + FillPressure::Healthy => 1, + FillPressure::Throttled => 2, + FillPressure::Paused => 3, + } +} #[derive(Clone, Copy)] -struct Observation { +struct InFlight { start: Instant, admitted: Option, - returned: Option, + completed: Option, bytes: u64, timeout: Duration, } struct State { - slots: Box<[Option]>, + slots: Box<[Option]>, snapshot: FillControlSnapshot, last_tick: Instant, last_progress: Instant, @@ -65,19 +100,20 @@ struct State { } pub struct FillController { - options: AdaptiveFillOptions, + options: FillLimits, enforcing: bool, max_units: u64, max_ops: u64, - // Epoch:16, operations:16, 64-byte units:32. One CAS reserves both dimensions. + // Epoch:16, records:16, 64-byte units:32. One CAS reserves both dimensions. credit: AtomicU64, epoch: AtomicU64, staging_pressure: AtomicBool, - paused: AtomicBool, + pressure: AtomicU8, stopped: AtomicBool, recovering: AtomicBool, rejections: AtomicU64, would_reject: AtomicU64, + dropped_observations: AtomicU64, state: Mutex, wake: Condvar, } @@ -104,8 +140,9 @@ impl FillController { settings } }; - if !(640..=1 << 40).contains(&settings.max_bytes_per_second) - || !(10..=655_350).contains(&settings.max_operations_per_second) + if !(MIN_BYTES_PER_SECOND..=MAX_BYTES_PER_SECOND).contains(&settings.max_bytes_per_second) + || !(MIN_RECORDS_PER_SECOND..=MAX_RECORDS_PER_SECOND) + .contains(&settings.max_records_per_second) { return Err(io::Error::new( io::ErrorKind::InvalidInput, @@ -113,7 +150,7 @@ impl FillController { )); } slots - .checked_mul(size_of::>()) + .checked_mul(size_of::>()) .and_then(|bytes| bytes.checked_add(size_of::() + CACHE_THREAD_STACK_BYTES + 256)) .ok_or_else(|| { io::Error::new( @@ -134,7 +171,7 @@ impl FillController { FillControlOptions::Observe(settings) => (settings, false), FillControlOptions::Adaptive(settings) => (settings, true), }; - let max_units = (settings.max_bytes_per_second / 10) + let max_units = (settings.max_bytes_per_second / TICKS_PER_SECOND) .max(max_record) .div_ceil(UNIT); if max_units > u64::from(u32::MAX) { @@ -143,7 +180,7 @@ impl FillController { "fill burst exceeds budget representation", )); } - let max_ops = u64::from(settings.max_operations_per_second / 10).max(1); + let max_ops = u64::from(settings.max_records_per_second / TICKS_PER_SECOND as u32).max(1); let mut observations = Vec::new(); observations.try_reserve_exact(slots).map_err(|_| { io::Error::new( @@ -158,21 +195,22 @@ impl FillController { enforcing, max_units, max_ops, - credit: AtomicU64::new((max_ops << 32) | max_units), + credit: AtomicU64::new(pack_credit(0, max_ops, max_units)), epoch: AtomicU64::new(0), staging_pressure: AtomicBool::new(false), - paused: AtomicBool::new(false), + pressure: AtomicU8::new(encode_pressure(FillPressure::Healthy)), stopped: AtomicBool::new(false), recovering: AtomicBool::new(false), rejections: AtomicU64::new(0), would_reject: AtomicU64::new(0), + dropped_observations: AtomicU64::new(0), state: Mutex::new(State { slots: observations.into_boxed_slice(), snapshot: FillControlSnapshot { pressure: FillPressure::Healthy, enforcing, bytes_per_second: settings.max_bytes_per_second, - operations_per_second: settings.max_operations_per_second, + records_per_second: settings.max_records_per_second, ..FillControlSnapshot::default() }, last_tick: now, @@ -200,15 +238,17 @@ impl FillController { .spawn(move || { let mut state = control.lock(); while !control.stopped.load(Ordering::Acquire) { + let now = Instant::now(); + let elapsed = now.saturating_duration_since(state.last_tick); + if elapsed >= TICK { + control.tick(&mut state, now); + continue; + } state = control .wake - .wait_timeout(state, TICK) + .wait_timeout(state, TICK - elapsed) .unwrap_or_else(|p| p.into_inner()) .0; - let now = Instant::now(); - if now.saturating_duration_since(state.last_tick) >= TICK { - control.tick(&mut state, now); - } } })?; Ok(FillMonitor { @@ -224,22 +264,28 @@ impl FillController { pub fn stop(&self) { let _state = self.lock(); self.stopped.store(true, Ordering::Release); - self.paused.store(true, Ordering::Release); + self.fence(); self.wake.notify_all(); } pub fn set_recovering(&self, recovering: bool) { self.recovering.store(recovering, Ordering::Release); if recovering && self.enforcing { - self.paused.store(true, Ordering::Release); + self.fence(); } } + fn fence(&self) { + self.pressure + .store(encode_pressure(FillPressure::Paused), Ordering::Release); + } + pub fn suppress_reinsertion(&self) -> bool { - self.enforcing && self.lock().snapshot.pressure != FillPressure::Healthy + self.enforcing + && self.pressure.load(Ordering::Acquire) != encode_pressure(FillPressure::Healthy) } - pub fn staging_busy(&self) { + pub fn note_staging_pressure(&self) { self.staging_pressure.store(true, Ordering::Relaxed); } @@ -247,46 +293,55 @@ impl FillController { let mut snapshot = self.lock().snapshot; snapshot.rejections = self.rejections.load(Ordering::Relaxed); snapshot.would_reject = self.would_reject.load(Ordering::Relaxed); + snapshot.dropped_observations = self.dropped_observations.load(Ordering::Relaxed); snapshot } pub fn try_admit(&self, bytes: u64) -> Option> { let units = bytes.div_ceil(UNIT); - let epoch = self.epoch.load(Ordering::Acquire); + if self.stopped.load(Ordering::Acquire) + || self.pressure.load(Ordering::Acquire) == encode_pressure(FillPressure::Paused) + { + return self.refuse(true); + } let mut value = self.credit.load(Ordering::Relaxed); - if !self.paused.load(Ordering::Acquire) && !self.stopped.load(Ordering::Acquire) { - for _ in 0..MAX_CAS_ATTEMPTS { - if value >> 48 != epoch & COUNT_MASK || self.epoch.load(Ordering::Acquire) != epoch - { - break; - } - if value & u64::from(u32::MAX) < units || (value >> 32) & COUNT_MASK == 0 { - break; - } - let next = value - units - (1 << 32); - match self.credit.compare_exchange_weak( - value, - next, - Ordering::AcqRel, - Ordering::Relaxed, - ) { - Ok(_) => { - return Some(FillPermit { - control: self, - units, - epoch, - committed: false, - }); - } - Err(current) => value = current, + for _ in 0..MAX_CAS_ATTEMPTS { + let epoch = self.epoch.load(Ordering::Acquire); + if packed_epoch(value) != epoch & COUNT_MASK { + value = self.credit.load(Ordering::Relaxed); + continue; + } + if packed_units(value) < units || packed_ops(value) == 0 { + return self.refuse(true); + } + match self.credit.compare_exchange_weak( + value, + value - units - (1 << 32), + Ordering::AcqRel, + Ordering::Relaxed, + ) { + Ok(_) => { + return Some(FillPermit { + control: self, + units, + epoch, + committed: false, + }); } + Err(current) => value = current, } } + self.refuse(false) + } + + fn refuse(&self, policy: bool) -> Option> { if self.enforcing { self.rejections.fetch_add(1, Ordering::Relaxed); None } else { - self.would_reject.fetch_add(1, Ordering::Relaxed); + if policy { + self.would_reject.fetch_add(1, Ordering::Relaxed); + } Some(FillPermit { control: self, units: 0, @@ -296,38 +351,64 @@ impl FillController { } } - pub fn observe(&self, bytes: u64, timeout: Duration) -> io::Result> { + pub fn observe(&self, bytes: u64, timeout: Duration) -> Option> { let mut state = self.lock(); let now = Instant::now(); - if state.slots.iter().all(Option::is_none) { + let mut free = None; + let mut occupied = 0; + for (index, slot) in state.slots.iter().enumerate() { + if slot.is_some() { + occupied += 1; + } else if free.is_none() { + free = Some(index); + } + } + let Some(index) = free else { + self.dropped_observations.fetch_add(1, Ordering::Relaxed); + return None; + }; + if occupied == 0 { state.last_progress = now; } - let index = state - .slots - .iter() - .position(Option::is_none) - .ok_or_else(|| io::Error::other("background observation capacity exhausted"))?; - state.slots[index] = Some(Observation { + state.slots[index] = Some(InFlight { start: now, admitted: None, - returned: None, + completed: None, bytes, timeout, }); - self.wake.notify_one(); - Ok(Progress { + Some(Observation { control: self, index, succeeded: false, }) } + fn clamp_rates(&self, bytes: u64, records: u32) -> (u64, u32) { + ( + bytes + .max(MIN_BYTES_PER_SECOND) + .min(self.options.max_bytes_per_second), + records + .max(MIN_RECORDS_PER_SECOND) + .min(self.options.max_records_per_second), + ) + } + + fn throttle(&self, state: &mut State, bytes: u64, records: u32) { + let (bytes, records) = self.clamp_rates(bytes, records); + state.snapshot.pressure = FillPressure::Throttled; + state.snapshot.bytes_per_second = bytes; + state.snapshot.records_per_second = records; + state.clean_ticks = 0; + } + fn tick(&self, state: &mut State, now: Instant) { state.last_tick = now; let elapsed = now .saturating_duration_since(state.window_start) .as_secs_f64(); - let decision = elapsed >= 0.5; + let decision = elapsed >= DECISION_WINDOW.as_secs_f64(); let mut made_progress = false; if decision { let bytes = state.completed_bytes.saturating_sub(state.sampled_bytes); @@ -342,15 +423,15 @@ impl FillController { let mut pending = 0; let mut bytes = 0_u64; let mut oldest = Duration::ZERO; - let mut deadline = Duration::from_secs(5); + let mut deadline = MAX_DRAIN_DEADLINE; let mut aged = false; - for observation in state.slots.iter().flatten() { + for request in state.slots.iter().flatten() { pending += 1; - bytes = bytes.saturating_add(observation.bytes); - let age = now.saturating_duration_since(observation.start); + bytes = bytes.saturating_add(request.bytes); + let age = now.saturating_duration_since(request.start); oldest = oldest.max(age); - deadline = deadline.min(observation.timeout); - aged |= age >= observation.timeout / 4; + deadline = deadline.min(request.timeout); + aged |= age >= request.timeout / 4; } let drain = if pending != 0 && state.service_bytes > 0. && state.service_ops > 0. { (bytes as f64 / state.service_bytes).max(pending as f64 / state.service_ops) @@ -358,55 +439,53 @@ impl FillController { 0. }; let no_progress = pending != 0 - && now.saturating_duration_since(state.last_progress) - >= (deadline / 2).min(Duration::from_millis(500)); + && now.saturating_duration_since(state.last_progress) >= (deadline / 2).min(STALL_CAP); let recovering = self.recovering.load(Ordering::Acquire); - let pressure = recovering || aged || no_progress || drain >= deadline.as_secs_f64() / 4.; + let pause = recovering || aged || no_progress || drain >= deadline.as_secs_f64() / 4.; let previous = state.snapshot.pressure; - let staging_busy = decision && self.staging_pressure.swap(false, Ordering::Relaxed); - if pressure { + let staging_pressure = decision && self.staging_pressure.swap(false, Ordering::Relaxed); + if pause { state.snapshot.pressure = FillPressure::Paused; state.clean_ticks = 0; } else if previous == FillPressure::Paused || state.was_recovering { if pending == 0 || (made_progress && oldest < deadline / 10 && drain < deadline.as_secs_f64() / 10.) { - state.snapshot.pressure = FillPressure::Throttled; - state.snapshot.bytes_per_second = (state.snapshot.bytes_per_second / 2).max(640); - state.snapshot.operations_per_second = - (state.snapshot.operations_per_second / 2).max(10); - if state.service_bytes > 0. { - state.snapshot.bytes_per_second = state - .snapshot - .bytes_per_second - .min((state.service_bytes * 0.8) as u64) - .max(640); - state.snapshot.operations_per_second = state - .snapshot - .operations_per_second - .min((state.service_ops * 0.8) as u32) - .max(10); - } - state.clean_ticks = 0; + let bytes = state.snapshot.bytes_per_second / 2; + let records = state.snapshot.records_per_second / 2; + let (bytes, records) = if state.service_bytes > 0. { + ( + bytes.min((state.service_bytes * SERVICE_HEADROOM) as u64), + records.min((state.service_ops * SERVICE_HEADROOM) as u32), + ) + } else { + (bytes, records) + }; + self.throttle(state, bytes, records); } - } else if staging_busy && pending >= state.previous_pending && pending != 0 { - state.snapshot.pressure = FillPressure::Throttled; - state.snapshot.bytes_per_second = (state.snapshot.bytes_per_second * 7 / 10).max(640); - state.snapshot.operations_per_second = - (state.snapshot.operations_per_second * 7 / 10).max(10); - state.clean_ticks = 0; + } else if staging_pressure && pending >= state.previous_pending && pending != 0 { + self.throttle( + state, + state.snapshot.bytes_per_second * STAGING_KEEP_NUM / STAGING_KEEP_DEN, + u32::try_from( + u64::from(state.snapshot.records_per_second) * STAGING_KEEP_NUM + / STAGING_KEEP_DEN, + ) + .unwrap_or(MIN_RECORDS_PER_SECOND), + ); } else if made_progress { state.clean_ticks += 1; - if state.clean_ticks >= 3 { - state.snapshot.bytes_per_second = (state.snapshot.bytes_per_second - + self.options.max_bytes_per_second / 40) - .min(self.options.max_bytes_per_second); - state.snapshot.operations_per_second = (state.snapshot.operations_per_second - + self.options.max_operations_per_second.div_ceil(40)) - .min(self.options.max_operations_per_second); - if state.snapshot.bytes_per_second == self.options.max_bytes_per_second - && state.snapshot.operations_per_second - == self.options.max_operations_per_second + if state.clean_ticks >= CLEAN_DECISIONS { + let (bytes, records) = self.clamp_rates( + state.snapshot.bytes_per_second + + self.options.max_bytes_per_second / u64::from(RAMP_DIVISOR), + state.snapshot.records_per_second + + self.options.max_records_per_second.div_ceil(RAMP_DIVISOR), + ); + state.snapshot.bytes_per_second = bytes; + state.snapshot.records_per_second = records; + if bytes == self.options.max_bytes_per_second + && records == self.options.max_records_per_second { state.snapshot.pressure = FillPressure::Healthy; } @@ -420,22 +499,18 @@ impl FillController { state.snapshot.outstanding_bytes = bytes; state.snapshot.oldest_operation_ns = nanos(oldest); state.snapshot.estimated_drain_ns = (drain * 1e9).min(u64::MAX as f64) as u64; - self.paused.store( - state.snapshot.pressure == FillPressure::Paused, - Ordering::Release, - ); - // Publishing a fresh epoch bounds refunds and prevents accumulating missed ticks. + self.pressure + .store(encode_pressure(state.snapshot.pressure), Ordering::Release); let clear = previous != state.snapshot.pressure || state.snapshot.pressure == FillPressure::Paused; - let add_units = state.snapshot.bytes_per_second / 10 / UNIT; - let add_ops = u64::from(state.snapshot.operations_per_second / 10); + let add_units = state.snapshot.bytes_per_second / TICKS_PER_SECOND / UNIT; + let add_ops = u64::from(state.snapshot.records_per_second) / TICKS_PER_SECOND; let epoch = self.epoch.fetch_add(1, Ordering::AcqRel).wrapping_add(1); let _ = self .credit .try_update(Ordering::AcqRel, Ordering::Relaxed, |old| { - let units = if clear { 0 } else { old & u64::from(u32::MAX) }; - let ops = if clear { 0 } else { (old >> 32) & COUNT_MASK }; - let next_epoch = (epoch & COUNT_MASK) << 48; + let units = if clear { 0 } else { packed_units(old) }; + let ops = if clear { 0 } else { packed_ops(old) }; let (units, ops) = if state.snapshot.pressure == FillPressure::Paused { (0, 0) } else { @@ -444,7 +519,7 @@ impl FillController { (ops + add_ops).min(self.max_ops), ) }; - Some(next_epoch | (ops << 32) | units) + Some(pack_credit(epoch, ops, units)) }); if previous != state.snapshot.pressure { log::info!(target: "cache2::health", event = "cache_fill_pressure_changed", pressure:? = state.snapshot.pressure; @@ -476,13 +551,15 @@ impl Drop for FillPermit<'_> { let mut value = self.control.credit.load(Ordering::Relaxed); for _ in 0..MAX_CAS_ATTEMPTS { if self.control.epoch.load(Ordering::Acquire) != self.epoch - || value >> 48 != self.epoch & COUNT_MASK + || packed_epoch(value) != self.epoch & COUNT_MASK { return; } - let units = ((value & u64::from(u32::MAX)) + self.units).min(self.control.max_units); - let ops = (((value >> 32) & COUNT_MASK) + 1).min(self.control.max_ops); - let next = (value & !CREDIT_MASK) | (ops << 32) | units; + let next = pack_credit( + packed_epoch(value), + (packed_ops(value) + 1).min(self.control.max_ops), + (packed_units(value) + self.units).min(self.control.max_units), + ); match self.control.credit.compare_exchange_weak( value, next, @@ -497,53 +574,57 @@ impl Drop for FillPermit<'_> { } /// One worker-owned observation. Failed work is removed, never marked successful. -pub struct Progress<'a> { +pub struct Observation<'a> { control: &'a FillController, index: usize, succeeded: bool, } -impl Progress<'_> { +impl Observation<'_> { pub fn admitted(&self) { - self.control.lock().slots[self.index] - .as_mut() - .unwrap() - .admitted = Some(Instant::now()); + self.stamp(|request, now| request.admitted = Some(now)); } - pub fn returned(&self) { - self.control.lock().slots[self.index] - .as_mut() - .unwrap() - .returned = Some(Instant::now()); + pub fn completed(&self) { + self.stamp(|request, now| request.completed = Some(now)); } pub fn finish(mut self) { self.succeeded = true; } + fn stamp(&self, update: impl FnOnce(&mut InFlight, Instant)) { + update( + self.control.lock().slots[self.index] + .as_mut() + .expect("live observation slot"), + Instant::now(), + ); + } } -impl Drop for Progress<'_> { +impl Drop for Observation<'_> { fn drop(&mut self) { let mut state = self.control.lock(); - let observation = state.slots[self.index].take().expect("live progress slot"); + let request = state.slots[self.index] + .take() + .expect("live observation slot"); let now = Instant::now(); if self.succeeded { state.completed_ops = state.completed_ops.saturating_add(1); - state.completed_bytes = state.completed_bytes.saturating_add(observation.bytes); + state.completed_bytes = state.completed_bytes.saturating_add(request.bytes); state.last_progress = now; } - let admission_end = observation.admitted.unwrap_or(now); + let admission_end = request.admitted.unwrap_or(now); state.snapshot.admission_ns = state.snapshot.admission_ns.saturating_add(nanos( - admission_end.saturating_duration_since(observation.start), + admission_end.saturating_duration_since(request.start), )); - if let Some(admitted) = observation.admitted { - let wait_end = observation.returned.unwrap_or(now); + if let Some(admitted) = request.admitted { + let wait_end = request.completed.unwrap_or(now); state.snapshot.completion_wait_ns = state .snapshot .completion_wait_ns .saturating_add(nanos(wait_end.saturating_duration_since(admitted))); - if let Some(returned) = observation.returned { + if let Some(completed) = request.completed { state.snapshot.validation_ns = state .snapshot .validation_ns - .saturating_add(nanos(now.saturating_duration_since(returned))); + .saturating_add(nanos(now.saturating_duration_since(completed))); } } } @@ -554,7 +635,7 @@ mod tests { use super::*; fn control(enforcing: bool) -> Arc { - let settings = AdaptiveFillOptions::new(64_000, 100); + let settings = FillLimits::new(64_000, 100); FillController::new( if enforcing { FillControlOptions::Adaptive(settings) @@ -597,7 +678,7 @@ mod tests { fn staging_pressure_reduces_once_per_window_and_idle_does_not_raise_rates() { let control = control(true); let pending = control.observe(64, Duration::from_secs(30)).unwrap(); - control.staging_busy(); + control.note_staging_pressure(); tick(&control, TICK); assert_eq!(control.snapshot().bytes_per_second, 64_000); { @@ -607,7 +688,7 @@ mod tests { tick(&control, Duration::from_millis(500)); assert_eq!(control.snapshot().pressure, FillPressure::Throttled); assert_eq!(control.snapshot().bytes_per_second, 44_800); - assert_eq!(control.snapshot().operations_per_second, 70); + assert_eq!(control.snapshot().records_per_second, 70); drop(pending); tick(&control, Duration::from_secs(10)); assert_eq!(control.snapshot().bytes_per_second, 44_800); @@ -643,7 +724,7 @@ mod tests { old.admitted(); let fast = control.observe(64, Duration::from_secs(2)).unwrap(); fast.admitted(); - fast.returned(); + fast.completed(); fast.finish(); { // Many small completions keep aggregate progress healthy. Only @@ -656,7 +737,7 @@ mod tests { tick(&control, Duration::from_millis(600)); assert_eq!(control.snapshot().pressure, FillPressure::Paused); assert!(control.try_admit(64).is_none()); - old.returned(); + old.completed(); assert!( control.try_admit(64).is_none(), "delivery does not release the admission fence" @@ -686,10 +767,9 @@ mod tests { let control = control(true); tick(&control, Duration::from_secs(3600)); assert_eq!(control.snapshot().pressure, FillPressure::Healthy); - assert_eq!( - control.credit.load(Ordering::Relaxed) & CREDIT_MASK, - (control.max_ops << 32) | control.max_units - ); + let credit = control.credit.load(Ordering::Relaxed); + assert_eq!(packed_ops(credit), control.max_ops); + assert_eq!(packed_units(credit), control.max_units); let _new = control.observe(4096, Duration::from_secs(2)).unwrap(); // Production ticks and observations use the same clock. let mut state = control.lock(); @@ -700,7 +780,7 @@ mod tests { #[test] fn large_record_remains_eligible_and_stop_cannot_reopen_admission() { let control = FillController::new( - FillControlOptions::Adaptive(AdaptiveFillOptions::new(640, 10)), + FillControlOptions::Adaptive(FillLimits::new(640, 10)), 1, 4096, ) @@ -739,14 +819,15 @@ mod tests { #[test] fn failed_observations_release_capacity_without_reporting_success() { let control = FillController::new( - FillControlOptions::Observe(AdaptiveFillOptions::new(640, 10)), + FillControlOptions::Observe(FillLimits::new(640, 10)), 1, 4096, ) .unwrap() .unwrap(); let pending = control.observe(4096, Duration::from_secs(5)).unwrap(); - assert!(control.observe(64, Duration::from_secs(5)).is_err()); + assert!(control.observe(64, Duration::from_secs(5)).is_none()); + assert_eq!(control.snapshot().dropped_observations, 1); // A failed admission still contributes its elapsed phase time. control.lock().slots[0].as_mut().unwrap().start -= Duration::from_secs(1); drop(pending); @@ -763,6 +844,25 @@ mod tests { assert!(state.slots[0].is_none()); } + #[test] + fn observe_would_reject_counts_policy_not_epoch_contention() { + let control = control(false); + control.epoch.store(1, Ordering::Release); + control.try_admit(64).unwrap().commit(); + assert_eq!(control.snapshot().would_reject, 0); + tick(&control, TICK); + for _ in 0..10 { + control.try_admit(64).unwrap().commit(); + } + control.try_admit(64).unwrap().commit(); + assert_eq!(control.snapshot().would_reject, 1); + let _old = control.observe(4096, Duration::from_secs(2)).unwrap(); + tick(&control, Duration::from_millis(600)); + assert_eq!(control.snapshot().pressure, FillPressure::Paused); + control.try_admit(64).unwrap().commit(); + assert_eq!(control.snapshot().would_reject, 2); + } + #[test] fn concurrent_admission_never_exceeds_shared_budget() { let control = control(true); diff --git a/cache2/src/lib.rs b/cache2/src/lib.rs index 076e522..0d8b567 100644 --- a/cache2/src/lib.rs +++ b/cache2/src/lib.rs @@ -35,8 +35,8 @@ pub use self::cache::Value; mod config; pub use self::config::CacheConfig; pub use self::config::StorageLayout; -pub use self::config::runtime::AdaptiveFillOptions; pub use self::config::runtime::FillControlOptions; +pub use self::config::runtime::FillLimits; pub use self::config::runtime::IoEngineOptions; pub use self::config::runtime::IoMode; pub use self::config::runtime::IoUringOptions; diff --git a/cache2/src/region/runtime/mod.rs b/cache2/src/region/runtime/mod.rs index 767724d..8ca7585 100644 --- a/cache2/src/region/runtime/mod.rs +++ b/cache2/src/region/runtime/mod.rs @@ -842,8 +842,8 @@ impl RegionDataPlane { return Err(write_overload_error()); } }; - let fill_permit = if let Some(controller) = &running.recovery.controller { - match controller.try_admit(u64::from(record_bytes)) { + let fill_permit = if let Some(fill) = &running.recovery.fill { + match fill.try_admit(u64::from(record_bytes)) { Some(permit) => Some(permit), None => { if running.activity_counters { @@ -894,14 +894,14 @@ impl RegionDataPlane { Ok(seqno) } RegionStageValue::NeedsProgress => { - if let Some(controller) = &running.recovery.controller { - controller.staging_busy(); + if let Some(fill) = &running.recovery.fill { + fill.note_staging_pressure(); } reject_staged_write(running, control, WAKE_URGENT, operation) } RegionStageValue::NeedsRotation => { - if let Some(controller) = &running.recovery.controller { - controller.staging_busy(); + if let Some(fill) = &running.recovery.fill { + fill.note_staging_pressure(); } reject_staged_write(running, control, WAKE_ROTATE | WAKE_URGENT, operation) } @@ -1321,8 +1321,8 @@ impl RegionDataPlane { { snapshot.health = crate::snapshot::CacheHealth::Recovering; } - if let Some(controller) = &running.recovery.controller { - snapshot.fill_control = controller.snapshot(); + if let Some(fill) = &running.recovery.fill { + snapshot.fill_control = fill.snapshot(); } snapshot.io = aggregate_io_stats( &running.read_engines, @@ -1525,12 +1525,12 @@ fn start_running( io::Error::new(io::ErrorKind::OutOfMemory, "cannot allocate shard controls") })?; shards.resize_with(shard_count, || Arc::new(ShardControl::new())); - let controller = FillController::new( + let fill = FillController::new( runtime.fill_control, shard_count + reclaim_worker_count, data.geometry.region_size, )?; - let fill_monitor = controller.as_ref().map(FillController::start).transpose()?; + let fill_monitor = fill.as_ref().map(FillController::start).transpose()?; let shared = Arc::new(RunningShared { core, read_engines, @@ -1540,7 +1540,7 @@ fn start_running( reclaim_engines, reclaim_control: ReclaimControl::new(), reclaim_io_timeout: runtime.reclaim_io_timeout, - recovery: BackgroundRecovery::with_controller(runtime.io_recovery_timeout, controller), + recovery: BackgroundRecovery::with_fill(runtime.io_recovery_timeout, fill), managed_memory, metrics, memory, @@ -1854,9 +1854,9 @@ fn reclaim_worker_result( let preserve_hot = shared.core.reclaim_can_reinsert()? && !shared .recovery - .controller + .fill .as_ref() - .is_some_and(|control| control.suppress_reinsertion()); + .is_some_and(|fill| fill.suppress_reinsertion()); let reinsert_operation = if preserve_hot { shared.operations.try_enter() } else { diff --git a/cache2/src/region/runtime/shutdown_tests.rs b/cache2/src/region/runtime/shutdown_tests.rs index d2379af..8df0627 100644 --- a/cache2/src/region/runtime/shutdown_tests.rs +++ b/cache2/src/region/runtime/shutdown_tests.rs @@ -297,9 +297,7 @@ fn adaptive_pressure_preserves_reads_and_deletes_and_resumes_fills() { storage_fingerprint: 4, }; let config = RuntimeOptions { - fill_control: crate::FillControlOptions::Adaptive(crate::AdaptiveFillOptions::new( - 1_048_576, 1000, - )), + fill_control: crate::FillControlOptions::Adaptive(crate::FillLimits::new(1_048_576, 1000)), append_shards: 1, l1_capacity_bytes: 0, ..RuntimeOptions::default() @@ -312,13 +310,10 @@ fn adaptive_pressure_preserves_reads_and_deletes_and_resumes_fills() { let plane = store.data_plane_handle().unwrap(); plane.put(b"existing", b"value").unwrap(); plane.drain().unwrap(); - let controller = plane.shared.recovery.controller.as_ref().unwrap(); - let pending = controller - .observe(4096, Duration::from_millis(400)) - .unwrap(); - pending.admitted(); + let fill = plane.shared.recovery.fill.as_ref().unwrap(); + fill.set_recovering(true); let deadline = Instant::now() + Duration::from_secs(2); - while controller.snapshot().pressure != crate::FillPressure::Paused { + while fill.snapshot().pressure != crate::FillPressure::Paused { assert!(Instant::now() < deadline, "monitor did not pause fills"); std::thread::sleep(Duration::from_millis(1)); } @@ -336,9 +331,8 @@ fn adaptive_pressure_preserves_reads_and_deletes_and_resumes_fills() { assert_eq!(plane.get(b"existing").unwrap().unwrap().value(), b"value"); plane.delete(b"existing").unwrap(); assert!(plane.get(b"existing").unwrap().is_none()); - pending.returned(); assert!(plane.put(b"new", b"value").is_err()); - pending.finish(); + fill.set_recovering(false); let deadline = Instant::now() + Duration::from_secs(2); loop { match plane.put(b"new", b"value") { diff --git a/cache2/src/snapshot.rs b/cache2/src/snapshot.rs index ec50710..c1be49f 100644 --- a/cache2/src/snapshot.rs +++ b/cache2/src/snapshot.rs @@ -50,7 +50,7 @@ pub enum FillPressure { Healthy, /// Fill rate has been reduced while accepted work drains. Throttled, - /// New fills are paused while outstanding work recovers. + /// New fills are paused while outstanding work is old or stalled. Paused, } @@ -64,12 +64,14 @@ pub struct FillControlSnapshot { pub enforcing: bool, /// Current encoded-byte admission rate. pub bytes_per_second: u64, - /// Current record admission rate. - pub operations_per_second: u32, - /// Fills rejected by the controller, including bounded CAS contention. + /// Current fill-record admission rate. + pub records_per_second: u32, + /// Fills rejected by Adaptive, including bounded CAS contention. pub rejections: u64, - /// Fills that Observe mode would have rejected. + /// Observe-mode pause or budget refusals; excludes CAS/epoch contention. pub would_reject: u64, + /// Background observations skipped because the table was full. + pub dropped_observations: u64, /// Background operations awaiting completion or validation. pub outstanding_operations: u64, /// Bytes held by those background operations; excludes unflushed staging. From fbc430742cec507a3ed9ba69916bbc5babda5662 Mon Sep 17 00:00:00 2001 From: leiysky Date: Sun, 20 Sep 2026 22:49:58 +0800 Subject: [PATCH 3/5] refactor: pace fill control on I/O workers Keep put competing only for staging. Workers checkpoint old I/O and pace non-essential flush so Adaptive can pause without a monitor thread. --- ARCHITECTURE.md | 6 +- CHANGELOG.md | 2 +- CONFIGURATION.md | 10 +- README.md | 6 +- cache2/ERRORS.md | 2 +- cache2/src/cache.rs | 3 +- cache2/src/config/runtime.rs | 8 +- cache2/src/io/engine/mod.rs | 26 +- cache2/src/io/engine/recovery.rs | 47 +- cache2/src/io/engine/tests.rs | 12 +- cache2/src/io/fill_control.rs | 975 +++++++++----------- cache2/src/region/runtime/mod.rs | 69 +- cache2/src/region/runtime/shutdown_tests.rs | 3 +- cache2/src/snapshot.rs | 27 +- 14 files changed, 544 insertions(+), 652 deletions(-) diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index 20b3b2e..5bd66df 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -116,11 +116,11 @@ Each lane uses one concrete `IoEngine` for admission, submission, cancellation, ### Pre-timeout fill pressure -`io::fill_control` owns optional background observations and adaptive fill admission. `Disabled` retains the existing path. Enabled modes preallocate a fixed observation table sized by append and reclaim worker counts and run one accounted monitoring thread. Background workers register before engine admission and retain their observations through completion validation and publication. A full table skips that request and counts `dropped_observations` instead of failing I/O. A 100 ms monitor poll can detect old or stalled work without requiring another completion or a per-submit wakeup; aggregate validated throughput provides an approximate drain estimate. Pressure does not change terminal health or prove a device fault. +`io::fill_control` owns optional background observations and adaptive fill admission. `Disabled` retains the existing path. Enabled modes preallocate a fixed observation table sized by append and reclaim worker counts. Background workers register before engine admission and retain their observations through completion validation and publication. A full table skips that request and counts `dropped_observations` instead of failing I/O. Workers checkpoint at one quarter of the I/O deadline, capped at 500 ms, and pause new fills without entering timeout recovery. Aggregate validated throughput provides an approximate drain estimate for snapshots and does not drive pause decisions. Pressure is `Healthy` or `Paused` and does not change terminal health or prove a device fault. -A packed atomic token bucket reserves both encoded bytes and record count before foreground staging. Each reservation/refund makes at most four CAS attempts; full refill epochs prevent stale permits from refunding a later budget. No per-fill clock, allocation, or controller mutex is required. Failed staging can refund its reservation; accepted work always keeps its resources and proceeds without reacquiring rate credit. `Observe` records pause and budget refusals as `would_reject`; Adaptive also rejects bounded CAS contention. `Adaptive` rejects new fills immediately and skips optional reinsertion under pressure. Reads, deletes, and essential reclaim bypass fill budgets. A short mutex protects only background observations, monitor decisions, and the optional controller snapshot. Close signals and joins the monitor independently of I/O completion. +Adaptive foreground `put`/`put_l2` load pause-holder state and otherwise only compete for staging. Byte and record ceilings pace non-essential background flush on shard workers; urgent, drain, and rotation flushes always proceed. A flush that cannot take a span refunds its budget and retries. `Observe` does not delay flush or reject fills and counts pause refusals as `would_reject`; it still checkpoints waits so pause is observable. Adaptive rejects new fills immediately while paused and skips optional reinsertion. Pause is released when the slow I/O completes, not after later publication. Reads, deletes, and essential reclaim bypass fill budgets. Close stops admission independently of outstanding I/O. -See [adaptive fill admission](CONFIGURATION.md#adaptive-fill-admission) for rate ceilings, pause/resume thresholds, bounded bursts, phase accounting, and tuning limits. +See [adaptive fill admission](CONFIGURATION.md#adaptive-fill-admission) for rate ceilings, pause conditions, bounded bursts, and tuning limits. ### Memory diff --git a/CHANGELOG.md b/CHANGELOG.md index 5660d78..42a7966 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -9,7 +9,7 @@ ### Features -- Optional `RuntimeOptions::fill_control` observes background request age, validated progress, and estimated drain time before timeout. `Observe` reports pause and budget refusals as `would_reject`; `Adaptive` bounds new fills by encoded bytes and record count, pauses on stalled work, resumes gradually, and suppresses optional reinsertion under pressure. A full observation table skips that request instead of failing I/O. Reads, deletes, accepted writes, and essential reclaim retain their paths. `CacheSnapshot::fill_control` exposes pressure and accounting independently of statistics; enabled snapshots take a short controller lock. Controller storage and its monitoring stack are included in managed memory. Disabled by default; see [configuration](CONFIGURATION.md#adaptive-fill-admission) for ceilings and measurement limits. +- Optional `RuntimeOptions::fill_control` observes background request age and estimated drain time before timeout. `Observe` reports pause refusals as `would_reject`. `Adaptive` pauses new `put`/`put_l2` fills when a background worker checkpoints an old I/O (one quarter of the deadline, capped at 500 ms) or timeout recovery is active, releases checkpoint pause when that I/O completes, paces non-essential background flush by encoded bytes and record count, and suppresses optional reinsertion while paused. Foreground admission loads pause-holder state and otherwise only competes for staging. A full observation table skips that request instead of failing I/O. Reads, deletes, accepted writes, and essential reclaim retain their paths. `CacheSnapshot::fill_control` exposes pressure and accounting independently of statistics. Controller storage is included in managed memory. Disabled by default; see [configuration](CONFIGURATION.md#adaptive-fill-admission) for ceilings and measurement limits. ### Improvements diff --git a/CONFIGURATION.md b/CONFIGURATION.md index 96a01b7..121756c 100644 --- a/CONFIGURATION.md +++ b/CONFIGURATION.md @@ -428,7 +428,7 @@ Set `options.reclaim_io_timeout = Duration::from_secs(30)` after constructing `R ### Adaptive fill admission -`RuntimeOptions::fill_control` defaults to `FillControlOptions::Disabled`. `Observe` records pressure and hypothetical rejections while retaining ordinary admission. `Adaptive` enforces the same decisions for new `put` and `put_l2` fills, returning `ErrorKind::Overloaded` immediately when paused or over budget. Neither mode changes foreground reads, deletes, accepted flushes, essential reclaim, or the existing timeout recovery policy. Adaptive pressure also suppresses optional hot-record reinsertion. Pressure is separate from `CacheHealth`: a pre-timeout pause leaves a healthy cache `Running`. +`RuntimeOptions::fill_control` defaults to `FillControlOptions::Disabled`. `Observe` records pause pressure and hypothetical rejections while retaining ordinary admission and flush cadence. `Adaptive` rejects new `put` and `put_l2` fills with `ErrorKind::Overloaded` when paused, and paces non-essential background flush by the configured ceilings. Neither mode changes foreground reads, deletes, accepted flushes, essential reclaim, or the existing timeout recovery policy. Adaptive also suppresses optional hot-record reinsertion while paused. Pressure is separate from `CacheHealth`: a pre-timeout pause leaves a healthy cache `Running`. ```rust use cache2::{FillControlOptions, FillLimits, RuntimeOptions}; @@ -441,10 +441,10 @@ options.fill_control = FillControlOptions::Observe(limits); options.fill_control = FillControlOptions::Adaptive(limits); ``` -The two ceilings constrain logical encoded fill bytes (charged in 64-byte units) and fill records, not device bandwidth or physical IOPS. Valid ceilings are 640 bytes/s through 1 TiB/s and 10 through 655350 records/s. Budgets refill every 100 ms, with capacity for 100 ms of the configured ceilings and at least one maximum-size record. An idle monitor never accumulates more than that bounded burst. Failed staging refunds only within the current refill epoch. Admission uses bounded atomic retries and never waits for rate credit. Adaptive may conservatively reject CAS or epoch contention; Observe counts `would_reject` only for pause and exhausted budgets. +The two ceilings constrain logical encoded fill bytes (charged in 64-byte units) and fill records flushed by shard workers, not device bandwidth or physical IOPS. Valid ceilings are 640 bytes/s through 1 TiB/s and 10 through 655350 records/s. Adaptive refill adds elapsed credit up to 100 ms of the configured ceilings and at least one maximum-size record. An idle controller never accumulates more than that bounded burst. Urgent, drain, and rotation flushes always proceed; other flushes wait for budget and back-pressure staging. Foreground admission does not consume the bucket. Observe does not delay flush and counts `would_reject` only for pause. -A dedicated monitor polls outstanding background work every 100 ms even if no I/O completes. Background submissions do not reset that interval. It pauses when any request reaches one quarter of its normal deadline, estimated drain time reaches one quarter of the shortest observed deadline (capped at five seconds), or outstanding work makes no validated progress for the smaller of 500 ms and half that deadline. These are conservative pressure signals: very short deadlines or a descheduled monitor can still reach timeout first. Staging saturation together with a nonshrinking outstanding queue reduces both rates by 30% at most once per 500 ms. A paused controller resumes when work drains, or validated progress brings age and estimated drain time below one tenth of the deadline, starting at half its previous rates and at most 80% of measured service rates when available. After three progressing 500 ms samples, rates increase by approximately 2.5% of the ceilings per progressing sample (rounded to whole bytes and records per second), up to the ceilings. Floors of 640 bytes/s and 10 records/s retain probe capacity; idle time alone never increases the rates. Normal timeout recovery keeps its admission fence until all affected work is validated and published. +Background workers checkpoint outstanding I/O at one quarter of the normal deadline, capped at 500 ms, and pause new Adaptive fills without entering timeout recovery. Checkpoint pause follows live holder counts and lifts when that I/O completes, even if later validation or reclaim scanning is still running. Observe uses the same checkpoints so it can report `would_reject`. Estimated drain time is reported for snapshots from validated throughput since open and does not pause admission. These are conservative pressure signals: very short deadlines or a descheduled worker can still reach timeout first. Idle time alone never looks like a stall. Normal timeout recovery keeps its admission fence until all affected work is validated and published. -`CacheSnapshot::fill_control` reports pressure, enforcement, current rates, rejections, hypothetical rejections, dropped observations, outstanding bytes/operations, oldest age, and estimated drain time independently of `RuntimeOptions::stats`. The monitor tracks background requests through admission, result consumption, and validation/publication, including scheduling delays. Its cumulative `admission_ns`, `completion_wait_ns`, and `validation_ns` measure those phases. Outstanding bytes exclude unflushed staging; drain time uses aggregate recent validated background throughput, including reclaim reads. A zero drain estimate means no estimate is available when work is pending. A full observation table skips that request rather than failing I/O. These measurements cannot distinguish device throttling from CPU scheduling, kernel queueing, or slow validation; correlate them with host metrics before diagnosing hardware. `cache_fill_pressure_changed` logs state transitions under `cache2::health`. +`CacheSnapshot::fill_control` reports pressure, enforcement, configured or paused rates, rejections, hypothetical pause refusals, dropped observations, outstanding bytes/operations, oldest age, and estimated drain time independently of `RuntimeOptions::stats`. Outstanding bytes exclude unflushed staging. A zero drain estimate means no estimate is available when work is pending. A full observation table skips that request rather than failing I/O. These measurements cannot distinguish device throttling from CPU scheduling, kernel queueing, or slow validation; correlate them with host metrics before diagnosing hardware. `cache_fill_pressure_changed` logs state transitions under `cache2::health`. -Enabled modes preallocate one observation per append/reclaim worker and one 512 KiB monitoring stack, all included in `CacheConfig::minimum_memory_bytes()`. Background observations and controller snapshots take a short lock over this bounded table. Foreground fill admission uses atomics without allocation, clock reads, or locks from the controller. Reinsertion suppression reads atomics. The monitor waits out each 100 ms interval, including while idle. Close stops it independently of outstanding I/O. Disabled mode creates no controller, observations, or monitoring thread. +Enabled modes preallocate one observation per append/reclaim worker, included in `CacheConfig::minimum_memory_bytes()`. Background observations use atomics over this bounded table. Foreground fill admission loads pause-holder and recovery counters without allocation, clocks, or locks from the controller. Reinsertion suppression reads atomics. Close stops admission independently of outstanding I/O. Disabled mode creates no controller or observations. diff --git a/README.md b/README.md index bd582e5..56e5778 100644 --- a/README.md +++ b/README.md @@ -83,7 +83,7 @@ See the [configuration guide](CONFIGURATION.md#configuration-lifecycle) for exam | Memory | `managed_memory_limit_bytes` | 1 GiB across cache-managed allocations. | | I/O mode | `io_mode` | Buffered I/O. | | Metrics | `stats: StatsOptions` | Health/resource gauges always available; activity, request, and latency collection opt in. | -| Fill control | `fill_control: FillControlOptions` | Disabled. `Observe` reports pause/budget pressure; `Adaptive` rejects new fills before timeout. | +| Fill control | `fill_control: FillControlOptions` | Disabled. `Observe` reports pause pressure; `Adaptive` rejects new fills before timeout. | Changing the append-shard count rebinds recovered Active Regions during a warm open. Growth uses available Free Regions; when there are not enough, the disposable cache safely starts empty. @@ -107,7 +107,7 @@ The on-disk format is versioned. During 0.x, deployments should expect cold star ### Metrics -`Cache::snapshot()` provides health and resource gauges using atomics and, when enabled, a short fill-controller lock. Setting `RuntimeOptions::stats.activity_counters` to `true` adds cumulative cache and I/O counters. `Cache::detailed_snapshot()` samples L1, index, write-buffer pressure, and Region metadata for periodic diagnostics. +`Cache::snapshot()` provides health and resource gauges using atomics. Setting `RuntimeOptions::stats.activity_counters` to `true` adds cumulative cache and I/O counters. `Cache::detailed_snapshot()` samples L1, index, write-buffer pressure, and Region metadata for periodic diagnostics. `RuntimeOptions::stats` independently enables complete public request outcomes, L1-hit, L2-lookup and mutation latency (each `Off`, `Full`, or `Sampled`), and full I/O latency by read/write/reclaim role. `Cache::stats_snapshot()` combines these with the existing summary without metadata scans. Structured request rows include their timing scope and collection mode. Applications own metric conversion, timestamps, scheduling and transport. Run `cargo run --example stats -- ` for an example. Full timing avoids sampling work; sampled histograms retain actual sample counts and cannot guarantee observation of rare tail events. Recorder storage is preallocated, bounded and charged to managed memory. @@ -160,4 +160,4 @@ Licensed under the [Apache License, Version 2.0](LICENSE). Set `RuntimeOptions::reclaim_io_timeout` to change the normal background reclaim deadline, for example `Duration::from_secs(30)` (default five seconds). Background write and reclaim timeouts enter `CacheHealth::Recovering`: new fills return overload while reads and deletes remain available. `RuntimeOptions::io_recovery_timeout` defaults to `None`, allowing recovery until completion or close. Use `Some(Duration::from_secs(300))` to limit the additional wait, or `Some(Duration::ZERO)` for immediate cancellation. Original requests retain their resources and are never resubmitted; fills resume after validation and publication of all affected work. Close interrupts recovery; drain may wait indefinitely. Actual I/O errors and invalid completions still fail the cache. -Optional [adaptive fill admission](CONFIGURATION.md#adaptive-fill-admission) detects slow background progress before timeout and limits new fills by encoded bytes and record count. Start with `FillControlOptions::Observe` to inspect pressure and hypothetical rejections, then use `Adaptive` to enforce the configured limits. It is disabled by default; reads, deletes, accepted writes, and essential reclaim retain their existing paths. +Optional [adaptive fill admission](CONFIGURATION.md#adaptive-fill-admission) detects slow background progress before timeout. `Adaptive` pauses new fills immediately and paces non-essential flush by encoded bytes and record count. Start with `FillControlOptions::Observe` to inspect pause pressure and hypothetical rejections, then use `Adaptive` to enforce. It is disabled by default; reads, deletes, accepted writes, and essential reclaim retain their existing paths. diff --git a/cache2/ERRORS.md b/cache2/ERRORS.md index 4084bd1..c7567fe 100644 --- a/cache2/ERRORS.md +++ b/cache2/ERRORS.md @@ -32,7 +32,7 @@ fn cache_value(cache: &Cache, key: &[u8], value: &[u8]) -> Result<(), Error> { Do not retry without a bound. C² deliberately exposes pressure instead of building unbounded queues. With the default immediate-read policy, read-pool or buffer pressure is `Ok(None)`. When read waiting is enabled, queue saturation, buffer pressure, and deadline expiry are `ErrorKind::Overloaded`. -With `FillControlOptions::Adaptive`, new fills also return `Overloaded` when the controller pauses or exhausts its byte/record budget. Inspect `CacheSnapshot::fill_control` to distinguish controller rejection from other admission pressure. Pre-timeout pressure does not change `CacheHealth::Running` or identify a hardware failure. `Observe` counts `would_reject` for pause and budget refusals only. Reads, deletes, accepted writes, and essential reclaim retain their ordinary paths. +With `FillControlOptions::Adaptive`, new fills also return `Overloaded` when the controller pauses. Staging back-pressure from paced flush uses the ordinary write-overload path. Inspect `CacheSnapshot::fill_control` to distinguish controller pause from other admission pressure. Pre-timeout pressure does not change `CacheHealth::Running` or identify a hardware failure. `Observe` counts `would_reject` for pause refusals only. Reads, deletes, accepted writes, and essential reclaim retain their ordinary paths. ## Classifications diff --git a/cache2/src/cache.rs b/cache2/src/cache.rs index ee1660c..c91a124 100644 --- a/cache2/src/cache.rs +++ b/cache2/src/cache.rs @@ -396,8 +396,7 @@ impl Cache { public_result(ErrorOperation::Drain, self.data_plane.drain_async().await) } - /// Returns an operational snapshot using atomics and, when enabled, a short - /// fill-controller lock. Activity and I/O counters are + /// Returns an operational snapshot using atomics. Activity and I/O counters are /// cumulative for this open and are populated only when /// `RuntimeOptions::stats.activity_counters` is enabled; health and resource gauges are /// always available. diff --git a/cache2/src/config/runtime.rs b/cache2/src/config/runtime.rs index 60a8bf4..9017c5b 100644 --- a/cache2/src/config/runtime.rs +++ b/cache2/src/config/runtime.rs @@ -422,7 +422,8 @@ pub struct RuntimeOptions { /// completions still fail the instance. pub io_recovery_timeout: Option, /// Optional pre-timeout fill pressure control. Disabled by default. - /// Enabled modes reserve a monitoring thread and bounded worker observations. + /// Enabled modes reserve bounded worker observations; Adaptive paces + /// background flush and pauses new fills when outstanding work is old. pub fill_control: FillControlOptions, /// Hash-routed append paths, from 1 through 256 (default 4). Each needs one /// Active Region, two Region-sized buffers, and a worker. The layout also needs a @@ -847,7 +848,7 @@ mod tests { use crate::StorageOptions; #[test] - fn fill_control_validates_ceilings_and_accounts_monitor_memory() { + fn fill_control_validates_ceilings_and_accounts_observation_memory() { let storage = StorageOptions::new(1024 * 1024 * 1024).build().unwrap(); let base = CacheConfig::new(storage.clone(), RuntimeOptions::default()).unwrap(); assert_eq!(base.runtime().fill_control, FillControlOptions::Disabled); @@ -874,7 +875,8 @@ mod tests { ) .unwrap(); let extra = config.minimum_memory_bytes() - base.minimum_memory_bytes(); - assert!(extra > CACHE_THREAD_STACK_BYTES); + assert!(extra > 0); + assert!(extra < CACHE_THREAD_STACK_BYTES); if let Some(previous) = minimum { assert_eq!(extra, previous); } diff --git a/cache2/src/io/engine/mod.rs b/cache2/src/io/engine/mod.rs index ebf21a0..f0570cb 100644 --- a/cache2/src/io/engine/mod.rs +++ b/cache2/src/io/engine/mod.rs @@ -777,24 +777,33 @@ impl BoundedIoRequest { ) -> Result { let original = self.deadline; loop { - self.request = match self.request.wait_until(self.deadline) { + let cap = if Instant::now() < original { + recovery.wait_cap(original) + } else { + self.deadline + }; + self.request = match self.request.wait_until(cap) { Ok(completion) => { - recovery.completed(); + recovery.clear_slow(); return Ok(completion); } Err(request) => request, }; + if Instant::now() < original { + recovery.note_slow(); + continue; + } match recovery.next_deadline(original) { Some(deadline) => self.deadline = deadline, None => { return match self.wait(engine) { Ok(completion) => { - recovery.completed(); + recovery.clear_slow(); Ok(completion) } Err(exceeded) => { if exceeded.completion.is_some() { - recovery.completed(); + recovery.clear_slow(); } Err(exceeded) } @@ -981,15 +990,20 @@ pub fn submit_background_io( let original = Instant::now() .checked_add(timeout) .unwrap_or_else(Instant::now); - let mut deadline = original; + let mut deadline = recovery.wait_cap(original); loop { match submit_cache_io_until(engine, operation, deadline, CACHE_IO_CANCEL_GRACE) { Ok(mut request) => { - recovery.admitted(); request.deadline = original; return Ok(request); } Err(error) if error.error.kind() == io::ErrorKind::TimedOut => { + if Instant::now() < original { + recovery.note_slow(); + deadline = original; + operation = error.operation; + continue; + } let Some(next) = recovery.next_deadline(original) else { return Err(error); }; diff --git a/cache2/src/io/engine/recovery.rs b/cache2/src/io/engine/recovery.rs index 4cb9920..f505b8f 100644 --- a/cache2/src/io/engine/recovery.rs +++ b/cache2/src/io/engine/recovery.rs @@ -53,6 +53,9 @@ impl BackgroundRecovery { RecoveryAttempt { recovery: self, entered: false, + slow: false, + started: None, + io_timeout: Duration::ZERO, observation: None, } } @@ -74,24 +77,49 @@ impl BackgroundRecovery { pub struct RecoveryAttempt<'a> { recovery: &'a BackgroundRecovery, entered: bool, + slow: bool, + started: Option, + io_timeout: Duration, observation: Option>, } impl RecoveryAttempt<'_> { pub fn start(&mut self, bytes: u64, timeout: Duration) { + self.started = Some(Instant::now()); + self.io_timeout = timeout; if let Some(fill) = &self.recovery.fill { self.observation = fill.observe(bytes, timeout); } } - pub fn admitted(&self) { - if let Some(observation) = &self.observation { - observation.admitted(); + /// First wait bound: a fill checkpoint before the normal I/O timeout. + pub fn wait_cap(&self, original: Instant) -> Instant { + if self.recovery.fill.is_none() || self.slow { + return original; + } + let Some(start) = self.started else { + return original; + }; + FillController::checkpoint(start, self.io_timeout).min(original) + } + + pub fn note_slow(&mut self) { + if self.slow { + return; + } + self.slow = true; + if let Some(observation) = &mut self.observation { + observation.note_slow(); } } - pub fn completed(&self) { - if let Some(observation) = &self.observation { - observation.completed(); + + pub fn clear_slow(&mut self) { + if !self.slow { + return; + } + self.slow = false; + if let Some(observation) = &mut self.observation { + observation.clear_slow(); } } @@ -131,12 +159,15 @@ impl RecoveryAttempt<'_> { if let Some(observation) = self.observation.take() { observation.finish(); } - if self.entered && self.recovery.pending.fetch_sub(1, Ordering::AcqRel) == 1 { + if self.entered { + let last = self.recovery.pending.fetch_sub(1, Ordering::AcqRel) == 1; if let Some(fill) = &self.recovery.fill { fill.set_recovering(false); } - log::info!(target: "cache2::health", event = "cache_io_recovery_completed"; + if last { + log::info!(target: "cache2::health", event = "cache_io_recovery_completed"; "all timed-out background operations recovered and passed validation"); + } } } } diff --git a/cache2/src/io/engine/tests.rs b/cache2/src/io/engine/tests.rs index 56eb1da..7344571 100644 --- a/cache2/src/io/engine/tests.rs +++ b/cache2/src/io/engine/tests.rs @@ -1393,7 +1393,7 @@ fn shutdown_interrupts_unlimited_recovery_without_releasing_pending_write() { } #[test] -fn adaptive_pressure_pauses_before_real_engine_timeout_and_resumes_after_validation() { +fn adaptive_pressure_pauses_before_real_engine_timeout_and_resumes_after_io_completes() { use crate::FillControlOptions; use crate::FillLimits; use crate::FillPressure; @@ -1406,7 +1406,6 @@ fn adaptive_pressure_pauses_before_real_engine_timeout_and_resumes_after_validat FillControlOptions::Observe(settings) }; let control = FillController::new(mode, 1, 4096).unwrap().unwrap(); - let monitor = control.start().unwrap(); let recovery = BackgroundRecovery::with_fill(None, Some(Arc::clone(&control))); let backend = Arc::new(BlockingBackend::default()); let engine = BackendIoEngine::new(backend.clone(), 1).unwrap(); @@ -1438,7 +1437,7 @@ fn adaptive_pressure_pauses_before_real_engine_timeout_and_resumes_after_validat } let paused = control.snapshot(); let recovered = recovery.is_recovering(); - let admission = control.try_admit(64).is_some(); + let admission = control.try_admit(); backend.release(); let completion = returned_rx.recv_timeout(Duration::from_secs(2)).unwrap(); assert_eq!(paused.pressure, FillPressure::Paused); @@ -1446,16 +1445,17 @@ fn adaptive_pressure_pauses_before_real_engine_timeout_and_resumes_after_validat assert_eq!(admission, !enforcing); assert_eq!(paused.outstanding_bytes, 4096); assert!(completion.into_io_result().0.is_ok()); + assert_eq!(control.snapshot().pressure, FillPressure::Healthy); + assert!(control.try_admit()); validate_tx.send(()).unwrap(); }); let deadline = Instant::now() + Duration::from_secs(2); while control.snapshot().pressure == FillPressure::Paused && Instant::now() < deadline { std::thread::sleep(Duration::from_millis(10)); } - assert_eq!(control.snapshot().pressure, FillPressure::Throttled); - assert!(control.try_admit(64).is_some()); + assert_eq!(control.snapshot().pressure, FillPressure::Healthy); + assert!(control.try_admit()); assert_eq!(lock_unpoisoned(&backend.state).entered, 1); engine.shutdown().unwrap(); - drop(monitor); } } diff --git a/cache2/src/io/fill_control.rs b/cache2/src/io/fill_control.rs index 964a540..a2b99e0 100644 --- a/cache2/src/io/fill_control.rs +++ b/cache2/src/io/fill_control.rs @@ -12,18 +12,15 @@ // See the License for the specific language governing permissions and // limitations under the License. -//! Bounded pre-timeout observation and nonblocking fill admission. +//! Bounded pre-timeout observation and worker-paced fill admission. use std::io; use std::sync::Arc; -use std::sync::Condvar; -use std::sync::Mutex; -use std::sync::MutexGuard; use std::sync::atomic::AtomicBool; use std::sync::atomic::AtomicU8; use std::sync::atomic::AtomicU64; +use std::sync::atomic::AtomicUsize; use std::sync::atomic::Ordering; -use std::thread::JoinHandle; use std::time::Duration; use std::time::Instant; @@ -31,72 +28,52 @@ use crate::FillControlOptions; use crate::FillControlSnapshot; use crate::FillLimits; use crate::FillPressure; -use crate::managed_memory::CACHE_THREAD_STACK_BYTES; -const TICK: Duration = Duration::from_millis(100); -const TICKS_PER_SECOND: u64 = 10; const UNIT: u64 = 64; -const COUNT_MASK: u64 = 0xffff; const MAX_CAS_ATTEMPTS: usize = 4; const MIN_BYTES_PER_SECOND: u64 = 640; const MAX_BYTES_PER_SECOND: u64 = 1 << 40; const MIN_RECORDS_PER_SECOND: u32 = 10; const MAX_RECORDS_PER_SECOND: u32 = 655_350; -const DECISION_WINDOW: Duration = Duration::from_millis(500); const STALL_CAP: Duration = Duration::from_millis(500); -const MAX_DRAIN_DEADLINE: Duration = Duration::from_secs(5); -const CLEAN_DECISIONS: u32 = 3; -const RAMP_DIVISOR: u32 = 40; -const STAGING_KEEP_NUM: u64 = 7; -const STAGING_KEEP_DEN: u64 = 10; -const SERVICE_HEADROOM: f64 = 0.8; - -fn packed_epoch(value: u64) -> u64 { - value >> 48 -} +const TICKS_PER_SECOND: u64 = 10; + +/// Wake shard workers this often when a non-essential flush is waiting for budget. +pub const FLUSH_RETRY: Duration = Duration::from_millis(100); + fn packed_ops(value: u64) -> u64 { - (value >> 32) & COUNT_MASK + value >> 32 } fn packed_units(value: u64) -> u64 { value & u64::from(u32::MAX) } -fn pack_credit(epoch: u64, ops: u64, units: u64) -> u64 { - (epoch & COUNT_MASK) << 48 | (ops << 32) | units +fn pack_credit(ops: u64, units: u64) -> u64 { + (ops << 32) | units } fn encode_pressure(pressure: FillPressure) -> u8 { match pressure { FillPressure::Disabled => 0, FillPressure::Healthy => 1, - FillPressure::Throttled => 2, - FillPressure::Paused => 3, + FillPressure::Paused => 2, } } -#[derive(Clone, Copy)] -struct InFlight { - start: Instant, - admitted: Option, - completed: Option, - bytes: u64, - timeout: Duration, +fn decode_pressure(value: u8) -> FillPressure { + match value { + 1 => FillPressure::Healthy, + 2 => FillPressure::Paused, + _ => FillPressure::Disabled, + } } -struct State { - slots: Box<[Option]>, - snapshot: FillControlSnapshot, - last_tick: Instant, - last_progress: Instant, - previous_pending: usize, - completed_bytes: u64, - completed_ops: u64, - sampled_bytes: u64, - sampled_ops: u64, - window_start: Instant, - service_bytes: f64, - service_ops: f64, - clean_ticks: u32, - was_recovering: bool, +fn nanos(duration: Duration) -> u64 { + duration.as_nanos().min(u128::from(u64::MAX)) as u64 +} + +struct Slot { + start_ns: AtomicU64, + bytes: AtomicU64, } pub struct FillController { @@ -104,32 +81,19 @@ pub struct FillController { enforcing: bool, max_units: u64, max_ops: u64, - // Epoch:16, records:16, 64-byte units:32. One CAS reserves both dimensions. + origin: Instant, credit: AtomicU64, - epoch: AtomicU64, - staging_pressure: AtomicBool, + last_refill_ns: AtomicU64, pressure: AtomicU8, stopped: AtomicBool, - recovering: AtomicBool, + recovering: AtomicUsize, + pause_holders: AtomicUsize, rejections: AtomicU64, would_reject: AtomicU64, dropped_observations: AtomicU64, - state: Mutex, - wake: Condvar, -} - -pub struct FillMonitor { - control: Arc, - worker: Option>, -} - -impl Drop for FillMonitor { - fn drop(&mut self) { - self.control.stop(); - if let Some(worker) = self.worker.take() { - let _ = worker.join(); - } - } + completed_bytes: AtomicU64, + completed_ops: AtomicU64, + slots: Box<[Slot]>, } impl FillController { @@ -150,8 +114,8 @@ impl FillController { )); } slots - .checked_mul(size_of::>()) - .and_then(|bytes| bytes.checked_add(size_of::() + CACHE_THREAD_STACK_BYTES + 256)) + .checked_mul(size_of::()) + .and_then(|bytes| bytes.checked_add(size_of::() + 256)) .ok_or_else(|| { io::Error::new( io::ErrorKind::InvalidInput, @@ -181,389 +145,258 @@ impl FillController { )); } let max_ops = u64::from(settings.max_records_per_second / TICKS_PER_SECOND as u32).max(1); - let mut observations = Vec::new(); - observations.try_reserve_exact(slots).map_err(|_| { + let mut table = Vec::new(); + table.try_reserve_exact(slots).map_err(|_| { io::Error::new( io::ErrorKind::OutOfMemory, "cannot allocate fill observations", ) })?; - observations.resize(slots, None); - let now = Instant::now(); + table.resize_with(slots, || Slot { + start_ns: AtomicU64::new(0), + bytes: AtomicU64::new(0), + }); Ok(Some(Arc::new(Self { options: settings, enforcing, max_units, max_ops, - credit: AtomicU64::new(pack_credit(0, max_ops, max_units)), - epoch: AtomicU64::new(0), - staging_pressure: AtomicBool::new(false), + origin: Instant::now(), + credit: AtomicU64::new(pack_credit(max_ops, max_units)), + last_refill_ns: AtomicU64::new(0), pressure: AtomicU8::new(encode_pressure(FillPressure::Healthy)), stopped: AtomicBool::new(false), - recovering: AtomicBool::new(false), + recovering: AtomicUsize::new(0), + pause_holders: AtomicUsize::new(0), rejections: AtomicU64::new(0), would_reject: AtomicU64::new(0), dropped_observations: AtomicU64::new(0), - state: Mutex::new(State { - slots: observations.into_boxed_slice(), - snapshot: FillControlSnapshot { - pressure: FillPressure::Healthy, - enforcing, - bytes_per_second: settings.max_bytes_per_second, - records_per_second: settings.max_records_per_second, - ..FillControlSnapshot::default() - }, - last_tick: now, - last_progress: now, - previous_pending: 0, - completed_bytes: 0, - completed_ops: 0, - sampled_bytes: 0, - sampled_ops: 0, - window_start: now, - service_bytes: 0., - service_ops: 0., - clean_ticks: 0, - was_recovering: false, - }), - wake: Condvar::new(), + completed_bytes: AtomicU64::new(0), + completed_ops: AtomicU64::new(0), + slots: table.into_boxed_slice(), }))) } - pub fn start(self: &Arc) -> io::Result { - let control = Arc::clone(self); - let worker = std::thread::Builder::new() - .name("cache2-fill-control".into()) - .stack_size(CACHE_THREAD_STACK_BYTES) - .spawn(move || { - let mut state = control.lock(); - while !control.stopped.load(Ordering::Acquire) { - let now = Instant::now(); - let elapsed = now.saturating_duration_since(state.last_tick); - if elapsed >= TICK { - control.tick(&mut state, now); - continue; - } - state = control - .wake - .wait_timeout(state, TICK - elapsed) - .unwrap_or_else(|p| p.into_inner()) - .0; - } - })?; - Ok(FillMonitor { - control: Arc::clone(self), - worker: Some(worker), - }) + pub fn checkpoint(start: Instant, timeout: Duration) -> Instant { + start + (timeout / 4).min(STALL_CAP) } - fn lock(&self) -> MutexGuard<'_, State> { - self.state.lock().unwrap_or_else(|p| p.into_inner()) + fn elapsed_ns(&self) -> u64 { + nanos(self.origin.elapsed()) } pub fn stop(&self) { - let _state = self.lock(); self.stopped.store(true, Ordering::Release); - self.fence(); - self.wake.notify_all(); + self.publish(); } pub fn set_recovering(&self, recovering: bool) { - self.recovering.store(recovering, Ordering::Release); - if recovering && self.enforcing { - self.fence(); + if recovering { + self.recovering.fetch_add(1, Ordering::AcqRel); + } else { + loop { + let holds = self.recovering.load(Ordering::Acquire); + if holds == 0 + || self + .recovering + .compare_exchange(holds, holds - 1, Ordering::AcqRel, Ordering::Relaxed) + .is_ok() + { + break; + } + } } + self.publish(); } - fn fence(&self) { - self.pressure - .store(encode_pressure(FillPressure::Paused), Ordering::Release); + fn hold_pause(&self) { + self.pause_holders.fetch_add(1, Ordering::AcqRel); + self.publish(); } - pub fn suppress_reinsertion(&self) -> bool { - self.enforcing - && self.pressure.load(Ordering::Acquire) != encode_pressure(FillPressure::Healthy) + fn release_pause(&self) { + self.pause_holders.fetch_sub(1, Ordering::AcqRel); + self.publish(); } - pub fn note_staging_pressure(&self) { - self.staging_pressure.store(true, Ordering::Relaxed); + fn publish(&self) { + let paused = self.stopped.load(Ordering::Acquire) + || self.recovering.load(Ordering::Acquire) != 0 + || self.pause_holders.load(Ordering::Acquire) != 0; + let next = if paused { + FillPressure::Paused + } else { + FillPressure::Healthy + }; + let previous = + decode_pressure(self.pressure.swap(encode_pressure(next), Ordering::Release)); + if previous != next && previous != FillPressure::Disabled { + log::info!(target: "cache2::health", event = "cache_fill_pressure_changed", pressure:? = next; + "cache fill admission pressure changed"); + } } - pub fn snapshot(&self) -> FillControlSnapshot { - let mut snapshot = self.lock().snapshot; - snapshot.rejections = self.rejections.load(Ordering::Relaxed); - snapshot.would_reject = self.would_reject.load(Ordering::Relaxed); - snapshot.dropped_observations = self.dropped_observations.load(Ordering::Relaxed); - snapshot + fn paused(&self) -> bool { + self.stopped.load(Ordering::Acquire) + || self.recovering.load(Ordering::Acquire) != 0 + || self.pause_holders.load(Ordering::Acquire) != 0 } - pub fn try_admit(&self, bytes: u64) -> Option> { - let units = bytes.div_ceil(UNIT); - if self.stopped.load(Ordering::Acquire) - || self.pressure.load(Ordering::Acquire) == encode_pressure(FillPressure::Paused) - { - return self.refuse(true); - } - let mut value = self.credit.load(Ordering::Relaxed); - for _ in 0..MAX_CAS_ATTEMPTS { - let epoch = self.epoch.load(Ordering::Acquire); - if packed_epoch(value) != epoch & COUNT_MASK { - value = self.credit.load(Ordering::Relaxed); - continue; - } - if packed_units(value) < units || packed_ops(value) == 0 { - return self.refuse(true); - } - match self.credit.compare_exchange_weak( - value, - value - units - (1 << 32), - Ordering::AcqRel, - Ordering::Relaxed, - ) { - Ok(_) => { - return Some(FillPermit { - control: self, - units, - epoch, - committed: false, - }); - } - Err(current) => value = current, - } - } - self.refuse(false) + pub fn suppress_reinsertion(&self) -> bool { + self.enforcing && self.paused() } - fn refuse(&self, policy: bool) -> Option> { - if self.enforcing { - self.rejections.fetch_add(1, Ordering::Relaxed); - None + pub fn snapshot(&self) -> FillControlSnapshot { + let pressure = if self.paused() { + FillPressure::Paused } else { - if policy { - self.would_reject.fetch_add(1, Ordering::Relaxed); + FillPressure::Healthy + }; + let now = self.elapsed_ns(); + let mut outstanding_operations = 0_u64; + let mut outstanding_bytes = 0_u64; + let mut oldest = 0_u64; + for slot in self.slots.iter() { + let start = slot.start_ns.load(Ordering::Acquire); + if start == 0 { + continue; } - Some(FillPermit { - control: self, - units: 0, - epoch: 0, - committed: true, - }) + outstanding_operations += 1; + outstanding_bytes = + outstanding_bytes.saturating_add(slot.bytes.load(Ordering::Relaxed)); + oldest = oldest.max(now.saturating_sub(start)); + } + let completed = self.completed_bytes.load(Ordering::Relaxed); + let drain = if outstanding_operations != 0 && completed != 0 && now != 0 { + (outstanding_bytes as f64 / (completed as f64 / (now as f64 / 1e9)) * 1e9) + .min(u64::MAX as f64) as u64 + } else { + 0 + }; + let (byte_rate, record_rate) = if pressure == FillPressure::Paused { + (0, 0) + } else { + ( + self.options.max_bytes_per_second, + self.options.max_records_per_second, + ) + }; + FillControlSnapshot { + pressure, + enforcing: self.enforcing, + bytes_per_second: byte_rate, + records_per_second: record_rate, + rejections: self.rejections.load(Ordering::Relaxed), + would_reject: self.would_reject.load(Ordering::Relaxed), + dropped_observations: self.dropped_observations.load(Ordering::Relaxed), + outstanding_operations, + outstanding_bytes, + oldest_operation_ns: oldest, + estimated_drain_ns: drain, } } - pub fn observe(&self, bytes: u64, timeout: Duration) -> Option> { - let mut state = self.lock(); - let now = Instant::now(); - let mut free = None; - let mut occupied = 0; - for (index, slot) in state.slots.iter().enumerate() { - if slot.is_some() { - occupied += 1; - } else if free.is_none() { - free = Some(index); + /// Foreground admission: Adaptive rejects only while paused. + pub fn try_admit(&self) -> bool { + if !self.enforcing { + if self.paused() { + self.would_reject.fetch_add(1, Ordering::Relaxed); } + return true; } - let Some(index) = free else { - self.dropped_observations.fetch_add(1, Ordering::Relaxed); - return None; - }; - if occupied == 0 { - state.last_progress = now; + if self.paused() { + self.rejections.fetch_add(1, Ordering::Relaxed); + false + } else { + true } - state.slots[index] = Some(InFlight { - start: now, - admitted: None, - completed: None, - bytes, - timeout, - }); - Some(Observation { - control: self, - index, - succeeded: false, - }) - } - - fn clamp_rates(&self, bytes: u64, records: u32) -> (u64, u32) { - ( - bytes - .max(MIN_BYTES_PER_SECOND) - .min(self.options.max_bytes_per_second), - records - .max(MIN_RECORDS_PER_SECOND) - .min(self.options.max_records_per_second), - ) } - fn throttle(&self, state: &mut State, bytes: u64, records: u32) { - let (bytes, records) = self.clamp_rates(bytes, records); - state.snapshot.pressure = FillPressure::Throttled; - state.snapshot.bytes_per_second = bytes; - state.snapshot.records_per_second = records; - state.clean_ticks = 0; - } - - fn tick(&self, state: &mut State, now: Instant) { - state.last_tick = now; - let elapsed = now - .saturating_duration_since(state.window_start) - .as_secs_f64(); - let decision = elapsed >= DECISION_WINDOW.as_secs_f64(); - let mut made_progress = false; - if decision { - let bytes = state.completed_bytes.saturating_sub(state.sampled_bytes); - let ops = state.completed_ops.saturating_sub(state.sampled_ops); - made_progress = ops != 0; - state.service_bytes = bytes as f64 / elapsed; - state.service_ops = ops as f64 / elapsed; - state.sampled_bytes = state.completed_bytes; - state.sampled_ops = state.completed_ops; - state.window_start = now; - } - let mut pending = 0; - let mut bytes = 0_u64; - let mut oldest = Duration::ZERO; - let mut deadline = MAX_DRAIN_DEADLINE; - let mut aged = false; - for request in state.slots.iter().flatten() { - pending += 1; - bytes = bytes.saturating_add(request.bytes); - let age = now.saturating_duration_since(request.start); - oldest = oldest.max(age); - deadline = deadline.min(request.timeout); - aged |= age >= request.timeout / 4; + /// Background flush pacing. Essential flushes always proceed. + pub fn try_flush(&self, bytes: u64, records: u32, essential: bool) -> bool { + if essential || !self.enforcing { + return true; } - let drain = if pending != 0 && state.service_bytes > 0. && state.service_ops > 0. { - (bytes as f64 / state.service_bytes).max(pending as f64 / state.service_ops) - } else { - 0. - }; - let no_progress = pending != 0 - && now.saturating_duration_since(state.last_progress) >= (deadline / 2).min(STALL_CAP); - let recovering = self.recovering.load(Ordering::Acquire); - let pause = recovering || aged || no_progress || drain >= deadline.as_secs_f64() / 4.; - let previous = state.snapshot.pressure; - let staging_pressure = decision && self.staging_pressure.swap(false, Ordering::Relaxed); - if pause { - state.snapshot.pressure = FillPressure::Paused; - state.clean_ticks = 0; - } else if previous == FillPressure::Paused || state.was_recovering { - if pending == 0 - || (made_progress && oldest < deadline / 10 && drain < deadline.as_secs_f64() / 10.) - { - let bytes = state.snapshot.bytes_per_second / 2; - let records = state.snapshot.records_per_second / 2; - let (bytes, records) = if state.service_bytes > 0. { - ( - bytes.min((state.service_bytes * SERVICE_HEADROOM) as u64), - records.min((state.service_ops * SERVICE_HEADROOM) as u32), - ) - } else { - (bytes, records) - }; - self.throttle(state, bytes, records); + self.refill(); + let raw_units = bytes.div_ceil(UNIT).max(1); + let raw_ops = u64::from(records.max(1)); + let units = raw_units.min(self.max_units); + let ops = raw_ops.min(self.max_ops); + let oversized = raw_units >= self.max_units; + let mut value = self.credit.load(Ordering::Relaxed); + for _ in 0..MAX_CAS_ATTEMPTS { + if packed_units(value) == 0 || packed_ops(value) == 0 { + return false; } - } else if staging_pressure && pending >= state.previous_pending && pending != 0 { - self.throttle( - state, - state.snapshot.bytes_per_second * STAGING_KEEP_NUM / STAGING_KEEP_DEN, - u32::try_from( - u64::from(state.snapshot.records_per_second) * STAGING_KEEP_NUM - / STAGING_KEEP_DEN, - ) - .unwrap_or(MIN_RECORDS_PER_SECOND), - ); - } else if made_progress { - state.clean_ticks += 1; - if state.clean_ticks >= CLEAN_DECISIONS { - let (bytes, records) = self.clamp_rates( - state.snapshot.bytes_per_second - + self.options.max_bytes_per_second / u64::from(RAMP_DIVISOR), - state.snapshot.records_per_second - + self.options.max_records_per_second.div_ceil(RAMP_DIVISOR), - ); - state.snapshot.bytes_per_second = bytes; - state.snapshot.records_per_second = records; - if bytes == self.options.max_bytes_per_second - && records == self.options.max_records_per_second - { - state.snapshot.pressure = FillPressure::Healthy; - } + if !oversized && (packed_units(value) < units || packed_ops(value) < ops) { + return false; + } + let next = if oversized { + 0 + } else { + pack_credit(packed_ops(value) - ops, packed_units(value) - units) + }; + match self + .credit + .compare_exchange(value, next, Ordering::AcqRel, Ordering::Relaxed) + { + Ok(_) => return true, + Err(current) => value = current, } } - state.was_recovering = recovering; - if decision { - state.previous_pending = pending; - } - state.snapshot.outstanding_operations = pending as u64; - state.snapshot.outstanding_bytes = bytes; - state.snapshot.oldest_operation_ns = nanos(oldest); - state.snapshot.estimated_drain_ns = (drain * 1e9).min(u64::MAX as f64) as u64; - self.pressure - .store(encode_pressure(state.snapshot.pressure), Ordering::Release); - let clear = - previous != state.snapshot.pressure || state.snapshot.pressure == FillPressure::Paused; - let add_units = state.snapshot.bytes_per_second / TICKS_PER_SECOND / UNIT; - let add_ops = u64::from(state.snapshot.records_per_second) / TICKS_PER_SECOND; - let epoch = self.epoch.fetch_add(1, Ordering::AcqRel).wrapping_add(1); - let _ = self - .credit - .try_update(Ordering::AcqRel, Ordering::Relaxed, |old| { - let units = if clear { 0 } else { packed_units(old) }; - let ops = if clear { 0 } else { packed_ops(old) }; - let (units, ops) = if state.snapshot.pressure == FillPressure::Paused { - (0, 0) - } else { - ( - (units + add_units).min(self.max_units), - (ops + add_ops).min(self.max_ops), - ) - }; - Some(pack_credit(epoch, ops, units)) - }); - if previous != state.snapshot.pressure { - log::info!(target: "cache2::health", event = "cache_fill_pressure_changed", pressure:? = state.snapshot.pressure; - "cache fill admission pressure changed"); - } + false } -} - -fn nanos(duration: Duration) -> u64 { - duration.as_nanos().min(u128::from(u64::MAX)) as u64 -} -pub struct FillPermit<'a> { - control: &'a FillController, - units: u64, - epoch: u64, - committed: bool, -} -impl FillPermit<'_> { - pub fn commit(mut self) { - self.committed = true; + pub fn refund_flush(&self, bytes: u64, records: u32) { + if !self.enforcing { + return; + } + let units = bytes.div_ceil(UNIT).max(1).min(self.max_units); + let ops = u64::from(records.max(1)).min(self.max_ops); + self.add_credit(ops, units); } -} -impl Drop for FillPermit<'_> { - fn drop(&mut self) { - if self.committed { + + fn refill(&self) { + let now = self.elapsed_ns(); + let last = self.last_refill_ns.load(Ordering::Relaxed); + let Some(dt) = now.checked_sub(last).filter(|dt| *dt != 0) else { + return; + }; + if self + .last_refill_ns + .compare_exchange(last, now, Ordering::Relaxed, Ordering::Relaxed) + .is_err() + { + return; + } + let add_units = self + .options + .max_bytes_per_second + .saturating_mul(dt) + .checked_div(1_000_000_000) + .unwrap_or(0) + / UNIT; + let add_ops = u64::from(self.options.max_records_per_second) + .saturating_mul(dt) + .checked_div(1_000_000_000) + .unwrap_or(0); + if add_units == 0 && add_ops == 0 { return; } - let mut value = self.control.credit.load(Ordering::Relaxed); + self.add_credit(add_ops, add_units); + } + + fn add_credit(&self, ops: u64, units: u64) { + let mut value = self.credit.load(Ordering::Relaxed); for _ in 0..MAX_CAS_ATTEMPTS { - if self.control.epoch.load(Ordering::Acquire) != self.epoch - || packed_epoch(value) != self.epoch & COUNT_MASK - { - return; - } let next = pack_credit( - packed_epoch(value), - (packed_ops(value) + 1).min(self.control.max_ops), - (packed_units(value) + self.units).min(self.control.max_units), + (packed_ops(value) + ops).min(self.max_ops), + (packed_units(value) + units).min(self.max_units), ); - match self.control.credit.compare_exchange_weak( + match self.credit.compare_exchange_weak( value, next, - Ordering::AcqRel, + Ordering::Relaxed, Ordering::Relaxed, ) { Ok(_) => return, @@ -571,6 +404,27 @@ impl Drop for FillPermit<'_> { } } } + + pub fn observe(&self, bytes: u64, _timeout: Duration) -> Option> { + let start = self.elapsed_ns().max(1); + for (index, slot) in self.slots.iter().enumerate() { + if slot + .start_ns + .compare_exchange(0, start, Ordering::AcqRel, Ordering::Acquire) + .is_ok() + { + slot.bytes.store(bytes, Ordering::Relaxed); + return Some(Observation { + control: self, + index, + succeeded: false, + slow: false, + }); + } + } + self.dropped_observations.fetch_add(1, Ordering::Relaxed); + None + } } /// One worker-owned observation. Failed work is removed, never marked successful. @@ -578,54 +432,42 @@ pub struct Observation<'a> { control: &'a FillController, index: usize, succeeded: bool, + slow: bool, } impl Observation<'_> { - pub fn admitted(&self) { - self.stamp(|request, now| request.admitted = Some(now)); + pub fn note_slow(&mut self) { + if self.slow { + return; + } + self.slow = true; + self.control.hold_pause(); } - pub fn completed(&self) { - self.stamp(|request, now| request.completed = Some(now)); + + pub fn clear_slow(&mut self) { + if !self.slow { + return; + } + self.slow = false; + self.control.release_pause(); } + pub fn finish(mut self) { self.succeeded = true; } - fn stamp(&self, update: impl FnOnce(&mut InFlight, Instant)) { - update( - self.control.lock().slots[self.index] - .as_mut() - .expect("live observation slot"), - Instant::now(), - ); - } } impl Drop for Observation<'_> { fn drop(&mut self) { - let mut state = self.control.lock(); - let request = state.slots[self.index] - .take() - .expect("live observation slot"); - let now = Instant::now(); + let slot = &self.control.slots[self.index]; + let bytes = slot.bytes.load(Ordering::Relaxed); + slot.start_ns.store(0, Ordering::Release); if self.succeeded { - state.completed_ops = state.completed_ops.saturating_add(1); - state.completed_bytes = state.completed_bytes.saturating_add(request.bytes); - state.last_progress = now; + self.control.completed_ops.fetch_add(1, Ordering::Relaxed); + self.control + .completed_bytes + .fetch_add(bytes, Ordering::Relaxed); } - let admission_end = request.admitted.unwrap_or(now); - state.snapshot.admission_ns = state.snapshot.admission_ns.saturating_add(nanos( - admission_end.saturating_duration_since(request.start), - )); - if let Some(admitted) = request.admitted { - let wait_end = request.completed.unwrap_or(now); - state.snapshot.completion_wait_ns = state - .snapshot - .completion_wait_ns - .saturating_add(nanos(wait_end.saturating_duration_since(admitted))); - if let Some(completed) = request.completed { - state.snapshot.validation_ns = state - .snapshot - .validation_ns - .saturating_add(nanos(now.saturating_duration_since(completed))); - } + if self.slow { + self.control.release_pause(); } } } @@ -649,113 +491,73 @@ mod tests { .unwrap() } - fn tick(control: &FillController, elapsed: Duration) { - let mut state = control.lock(); - let now = state.last_tick + elapsed; - control.tick(&mut state, now); + fn restore_burst(control: &FillController) { + control.credit.store( + pack_credit(control.max_ops, control.max_units), + Ordering::Release, + ); + } + + fn freeze_refill(control: &FillController) { + control + .last_refill_ns + .store(control.elapsed_ns(), Ordering::Relaxed); } #[test] - fn budget_reserves_both_dimensions_and_refunds_failed_staging() { + fn foreground_admit_ignores_flush_budget() { let control = control(true); - let initial = control.credit.load(Ordering::Relaxed); - let permit = control.try_admit(640).unwrap(); - assert_eq!( - control.credit.load(Ordering::Relaxed), - initial - 10 - (1 << 32) - ); - drop(permit); - assert_eq!(control.credit.load(Ordering::Relaxed), initial); for _ in 0..10 { - control.try_admit(64).unwrap().commit(); + assert!(control.try_flush(64, 1, false)); } - assert!(control.try_admit(64).is_none()); - tick(&control, TICK); - assert!(control.try_admit(64).is_some()); + assert!(!control.try_flush(64, 1, false)); + assert!(control.try_admit()); } #[test] - fn staging_pressure_reduces_once_per_window_and_idle_does_not_raise_rates() { + fn record_budget_paces_nonessential_flush() { let control = control(true); - let pending = control.observe(64, Duration::from_secs(30)).unwrap(); - control.note_staging_pressure(); - tick(&control, TICK); - assert_eq!(control.snapshot().bytes_per_second, 64_000); - { - let mut state = control.lock(); - state.last_progress = state.last_tick + Duration::from_millis(500); - } - tick(&control, Duration::from_millis(500)); - assert_eq!(control.snapshot().pressure, FillPressure::Throttled); - assert_eq!(control.snapshot().bytes_per_second, 44_800); - assert_eq!(control.snapshot().records_per_second, 70); - drop(pending); - tick(&control, Duration::from_secs(10)); - assert_eq!(control.snapshot().bytes_per_second, 44_800); - for sample in 0..3 { - { - let mut state = control.lock(); - state.completed_bytes += 64; - state.completed_ops += 1; - } - tick(&control, Duration::from_millis(500)); - assert_eq!( - control.snapshot().bytes_per_second, - if sample < 2 { 44_800 } else { 46_400 } - ); + for _ in 0..10 { + assert!(control.try_flush(64, 1, false)); } + assert!(!control.try_flush(64, 1, false)); + assert!(control.try_flush(64, 1, true)); + restore_burst(&control); + assert!(control.try_flush(64, 1, false)); } #[test] - fn old_refund_cannot_mint_new_epoch_credits_even_after_low_bits_wrap() { + fn byte_budget_paces_nonessential_flush() { let control = control(true); - let permit = control.try_admit(64).unwrap(); - control.epoch.store(COUNT_MASK, Ordering::Release); - tick(&control, TICK); - let before = control.credit.load(Ordering::Relaxed); - drop(permit); - assert_eq!(control.credit.load(Ordering::Relaxed), before); + assert!(control.try_flush(4096, 1, false)); + assert!(!control.try_flush(4096, 1, false)); + assert!(control.try_flush(64, 1, false)); + restore_burst(&control); + assert!(control.try_flush(4096, 1, false)); } #[test] - fn old_request_is_detected_while_other_requests_progress() { + fn old_request_pauses_while_other_requests_progress() { let control = control(true); - let old = control.observe(4096, Duration::from_secs(2)).unwrap(); - old.admitted(); + let mut old = control.observe(4096, Duration::from_secs(2)).unwrap(); let fast = control.observe(64, Duration::from_secs(2)).unwrap(); - fast.admitted(); - fast.completed(); fast.finish(); - { - // Many small completions keep aggregate progress healthy. Only - // the retained request's age should trigger the pause. - let mut state = control.lock(); - state.last_progress = state.last_tick + Duration::from_millis(600); - state.completed_bytes = 1_000_000; - state.completed_ops = 1_000; - } - tick(&control, Duration::from_millis(600)); + old.note_slow(); assert_eq!(control.snapshot().pressure, FillPressure::Paused); - assert!(control.try_admit(64).is_none()); - old.completed(); - assert!( - control.try_admit(64).is_none(), - "delivery does not release the admission fence" - ); - old.finish(); - tick(&control, TICK); - assert_eq!(control.snapshot().pressure, FillPressure::Throttled); - assert!(control.try_admit(64).is_some()); + assert!(!control.try_admit()); + drop(old); + assert_eq!(control.snapshot().pressure, FillPressure::Healthy); + assert!(control.try_admit()); } #[test] fn observe_never_rejects_or_suppresses_reinsertion() { let control = control(false); - let _old = control.observe(4096, Duration::from_secs(2)).unwrap(); - tick(&control, Duration::from_millis(600)); + let mut old = control.observe(4096, Duration::from_secs(2)).unwrap(); + old.note_slow(); assert_eq!(control.snapshot().pressure, FillPressure::Paused); for _ in 0..20 { - control.try_admit(4096).unwrap().commit(); + assert!(control.try_admit()); } assert_eq!(control.snapshot().would_reject, 20); assert_eq!(control.snapshot().rejections, 0); @@ -763,22 +565,33 @@ mod tests { } #[test] - fn idle_time_does_not_inflate_credit_or_look_like_a_stall() { + fn observe_would_reject_counts_pause_not_budget() { + let control = control(false); + for _ in 0..20 { + assert!(control.try_flush(4096, 1, false)); + assert!(control.try_admit()); + } + assert_eq!(control.snapshot().would_reject, 0); + let mut old = control.observe(4096, Duration::from_secs(2)).unwrap(); + old.note_slow(); + assert_eq!(control.snapshot().pressure, FillPressure::Paused); + assert!(control.try_admit()); + assert_eq!(control.snapshot().would_reject, 1); + assert_eq!(control.snapshot().rejections, 0); + } + + #[test] + fn idle_time_does_not_inflate_credit() { let control = control(true); - tick(&control, Duration::from_secs(3600)); - assert_eq!(control.snapshot().pressure, FillPressure::Healthy); + control.last_refill_ns.store(0, Ordering::Relaxed); + control.refill(); let credit = control.credit.load(Ordering::Relaxed); assert_eq!(packed_ops(credit), control.max_ops); assert_eq!(packed_units(credit), control.max_units); - let _new = control.observe(4096, Duration::from_secs(2)).unwrap(); - // Production ticks and observations use the same clock. - let mut state = control.lock(); - control.tick(&mut state, Instant::now()); - assert_eq!(state.snapshot.pressure, FillPressure::Healthy); } #[test] - fn large_record_remains_eligible_and_stop_cannot_reopen_admission() { + fn large_flush_remains_eligible_and_stop_cannot_reopen_admission() { let control = FillController::new( FillControlOptions::Adaptive(FillLimits::new(640, 10)), 1, @@ -786,34 +599,36 @@ mod tests { ) .unwrap() .unwrap(); - control.try_admit(4096).unwrap().commit(); - for _ in 0..64 { - tick(&control, TICK); - } - assert!(control.try_admit(4096).is_some()); + assert!(control.try_flush(4096, 1, false)); + restore_burst(&control); + assert!(control.try_flush(4096, 1, false)); control.stop(); - tick(&control, TICK); - assert!(control.try_admit(64).is_none()); + assert!(!control.try_admit()); } #[test] - fn recovering_fence_survives_empty_observation_table() { + fn recovering_fence_does_not_need_outstanding_io() { let control = control(true); control.set_recovering(true); - tick(&control, TICK); assert_eq!(control.snapshot().pressure, FillPressure::Paused); - assert!(control.try_admit(64).is_none()); + assert!(!control.try_admit()); control.set_recovering(false); - tick(&control, TICK); - assert_eq!(control.snapshot().pressure, FillPressure::Throttled); + assert_eq!(control.snapshot().pressure, FillPressure::Healthy); + assert!(control.try_admit()); } #[test] - fn idle_monitor_can_be_stopped_without_io() { + fn overlapping_recovery_holds_resume_when_all_release() { let control = control(true); - let monitor = control.start().unwrap(); - drop(monitor); - assert!(control.stopped.load(Ordering::Acquire)); + control.set_recovering(true); + control.set_recovering(true); + assert!(!control.try_admit()); + control.set_recovering(false); + assert_eq!(control.snapshot().pressure, FillPressure::Paused); + assert!(!control.try_admit()); + control.set_recovering(false); + assert_eq!(control.snapshot().pressure, FillPressure::Healthy); + assert!(control.try_admit()); } #[test] @@ -828,44 +643,18 @@ mod tests { let pending = control.observe(4096, Duration::from_secs(5)).unwrap(); assert!(control.observe(64, Duration::from_secs(5)).is_none()); assert_eq!(control.snapshot().dropped_observations, 1); - // A failed admission still contributes its elapsed phase time. - control.lock().slots[0].as_mut().unwrap().start -= Duration::from_secs(1); drop(pending); - assert!(control.snapshot().admission_ns >= 1_000_000_000); let pending = control.observe(4096, Duration::from_secs(5)).unwrap(); - pending.admitted(); - control.lock().slots[0].as_mut().unwrap().admitted = - Some(Instant::now() - Duration::from_secs(1)); drop(pending); - assert!(control.snapshot().completion_wait_ns >= 1_000_000_000); - let state = control.lock(); - assert_eq!(state.completed_ops, 0); - assert_eq!(state.completed_bytes, 0); - assert!(state.slots[0].is_none()); - } - - #[test] - fn observe_would_reject_counts_policy_not_epoch_contention() { - let control = control(false); - control.epoch.store(1, Ordering::Release); - control.try_admit(64).unwrap().commit(); - assert_eq!(control.snapshot().would_reject, 0); - tick(&control, TICK); - for _ in 0..10 { - control.try_admit(64).unwrap().commit(); - } - control.try_admit(64).unwrap().commit(); - assert_eq!(control.snapshot().would_reject, 1); - let _old = control.observe(4096, Duration::from_secs(2)).unwrap(); - tick(&control, Duration::from_millis(600)); - assert_eq!(control.snapshot().pressure, FillPressure::Paused); - control.try_admit(64).unwrap().commit(); - assert_eq!(control.snapshot().would_reject, 2); + assert_eq!(control.completed_ops.load(Ordering::Relaxed), 0); + assert_eq!(control.completed_bytes.load(Ordering::Relaxed), 0); + assert_eq!(control.slots[0].start_ns.load(Ordering::Acquire), 0); } #[test] - fn concurrent_admission_never_exceeds_shared_budget() { + fn concurrent_flush_never_exceeds_shared_budget() { let control = control(true); + freeze_refill(&control); let accepted = AtomicU64::new(0); std::thread::scope(|scope| { for _ in 0..8 { @@ -873,8 +662,7 @@ mod tests { let accepted = &accepted; scope.spawn(move || { for _ in 0..100 { - if let Some(permit) = control.try_admit(640) { - permit.commit(); + if control.try_flush(640, 1, false) { accepted.fetch_add(1, Ordering::Relaxed); } } @@ -883,4 +671,71 @@ mod tests { }); assert_eq!(accepted.load(Ordering::Relaxed), 10); } + + #[test] + fn io_completion_releases_pause_before_observation_drop() { + let control = control(true); + let mut pending = control.observe(4096, Duration::from_secs(2)).unwrap(); + pending.note_slow(); + assert!(!control.try_admit()); + pending.clear_slow(); + assert_eq!(control.snapshot().pressure, FillPressure::Healthy); + assert!(control.try_admit()); + pending.finish(); + } + + #[test] + fn concurrent_pause_holders_resume_when_all_release() { + let control = control(true); + std::thread::scope(|scope| { + for _ in 0..8 { + let control = &control; + scope.spawn(move || { + let mut pending = control.observe(64, Duration::from_secs(2)).unwrap(); + pending.note_slow(); + assert!(!control.try_admit()); + drop(pending); + }); + } + }); + assert_eq!(control.snapshot().pressure, FillPressure::Healthy); + assert_eq!(control.pause_holders.load(Ordering::Acquire), 0); + assert!(control.try_admit()); + } + + #[test] + fn overlapping_hold_and_release_cannot_stick_paused() { + let control = control(true); + for _ in 0..1_000 { + let mut first = control.observe(64, Duration::from_secs(2)).unwrap(); + let mut second = control.observe(64, Duration::from_secs(2)).unwrap(); + first.note_slow(); + second.note_slow(); + drop(first); + drop(second); + assert_eq!(control.pause_holders.load(Ordering::Acquire), 0); + assert_eq!(control.snapshot().pressure, FillPressure::Healthy); + assert!(control.try_admit()); + } + } + + #[test] + fn burst_sized_flush_proceeds_on_remaining_credit() { + let control = control(true); + freeze_refill(&control); + assert!(control.try_flush(64, 1, false)); + assert!(control.try_flush(1_000_000, 1, false)); + } + + #[test] + fn refund_restores_nonessential_budget() { + let control = control(true); + freeze_refill(&control); + for _ in 0..10 { + assert!(control.try_flush(64, 1, false)); + } + assert!(!control.try_flush(64, 1, false)); + control.refund_flush(64, 1); + assert!(control.try_flush(64, 1, false)); + } } diff --git a/cache2/src/region/runtime/mod.rs b/cache2/src/region/runtime/mod.rs index 8ca7585..2a5b01e 100644 --- a/cache2/src/region/runtime/mod.rs +++ b/cache2/src/region/runtime/mod.rs @@ -61,8 +61,8 @@ use crate::io::engine::ReadSlotWaiter; use crate::io::engine::build_file_engine; use crate::io::engine::recovery::BackgroundRecovery; use crate::io::engine::submit_background_io; +use crate::io::fill_control::FLUSH_RETRY; use crate::io::fill_control::FillController; -use crate::io::fill_control::FillMonitor; use crate::managed_memory::BufferLease; use crate::managed_memory::CACHE_THREAD_STACK_BYTES; use crate::managed_memory::ManagedMemory; @@ -429,7 +429,6 @@ struct RunningOwner { shared: Arc, shard_workers: Vec>, reclaim_workers: Vec>, - fill_monitor: Option, } struct RunningShared { @@ -842,19 +841,14 @@ impl RegionDataPlane { return Err(write_overload_error()); } }; - let fill_permit = if let Some(fill) = &running.recovery.fill { - match fill.try_admit(u64::from(record_bytes)) { - Some(permit) => Some(permit), - None => { - if running.activity_counters { - running.metrics.record_write_rejection(); - } - return Err(write_overload_error()); - } + if let Some(fill) = &running.recovery.fill + && !fill.try_admit() + { + if running.activity_counters { + running.metrics.record_write_rejection(); } - } else { - None - }; + return Err(write_overload_error()); + } let staged = self.core.try_stage_value( &running.staging, shard_id, @@ -869,9 +863,6 @@ impl RegionDataPlane { previous_bytes, current_bytes, } => { - if let Some(permit) = fill_permit { - permit.commit(); - } if ADMIT_L1 { let _published = running.memory.publish(hash, key, value, seqno); } else { @@ -894,15 +885,9 @@ impl RegionDataPlane { Ok(seqno) } RegionStageValue::NeedsProgress => { - if let Some(fill) = &running.recovery.fill { - fill.note_staging_pressure(); - } reject_staged_write(running, control, WAKE_URGENT, operation) } RegionStageValue::NeedsRotation => { - if let Some(fill) = &running.recovery.fill { - fill.note_staging_pressure(); - } reject_staged_write(running, control, WAKE_ROTATE | WAKE_URGENT, operation) } } @@ -1530,7 +1515,6 @@ fn start_running( shard_count + reclaim_worker_count, data.geometry.region_size, )?; - let fill_monitor = fill.as_ref().map(FillController::start).transpose()?; let shared = Arc::new(RunningShared { core, read_engines, @@ -1623,7 +1607,6 @@ fn start_running( shared, shard_workers, reclaim_workers, - fill_monitor, }) } @@ -1936,14 +1919,33 @@ fn shard_worker_result( )?); } if force_flush || fill.bytes >= shared.write_flush_threshold_bytes { - let engine = shared.write_engine_for(shard_id as u64); - shared.core.flush_staging_shard( - &shared.staging, - engine.as_ref(), - shard_id, - &shared.recovery, - )?; - deadline = None; + let essential = flags & (WAKE_URGENT | WAKE_ROTATE) != 0 || draining; + let bytes = fill.bytes as u64; + let records = u32::try_from(fill.records).unwrap_or(u32::MAX); + let allow = shared + .recovery + .fill + .as_ref() + .is_none_or(|control| control.try_flush(bytes, records, essential)); + if !allow { + deadline = Some(Instant::now() + FLUSH_RETRY); + } else { + let engine = shared.write_engine_for(shard_id as u64); + match shared.core.flush_staging_shard( + &shared.staging, + engine.as_ref(), + shard_id, + &shared.recovery, + )? { + Some(_) => deadline = None, + None => { + if !essential && let Some(control) = &shared.recovery.fill { + control.refund_flush(bytes, records); + } + deadline = Some(Instant::now() + STAGING_RETRY_DELAY); + } + } + } } } Ok(None) => { @@ -2113,7 +2115,6 @@ async fn drain_shards_async(shared: &RunningShared, stop: bool) -> io::Result<() fn stop_running(mut owner: RunningOwner) -> io::Result { owner.shared.recovery.stop(); - drop(owner.fill_monitor.take()); let drain = drain_shards(&owner.shared, true); let mut join_error = None; for worker in owner.shard_workers.drain(..) { diff --git a/cache2/src/region/runtime/shutdown_tests.rs b/cache2/src/region/runtime/shutdown_tests.rs index 8df0627..c47801b 100644 --- a/cache2/src/region/runtime/shutdown_tests.rs +++ b/cache2/src/region/runtime/shutdown_tests.rs @@ -178,7 +178,6 @@ fn assert_close_does_not_wait_for_read(submit_before_close: bool) { shared, shard_workers: vec![], reclaim_workers: vec![], - fill_monitor: None, }); tx.send(result).unwrap(); }); @@ -314,7 +313,7 @@ fn adaptive_pressure_preserves_reads_and_deletes_and_resumes_fills() { fill.set_recovering(true); let deadline = Instant::now() + Duration::from_secs(2); while fill.snapshot().pressure != crate::FillPressure::Paused { - assert!(Instant::now() < deadline, "monitor did not pause fills"); + assert!(Instant::now() < deadline, "controller did not pause fills"); std::thread::sleep(Duration::from_millis(1)); } let snapshot = plane.snapshot().unwrap(); diff --git a/cache2/src/snapshot.rs b/cache2/src/snapshot.rs index c1be49f..8b5e8eb 100644 --- a/cache2/src/snapshot.rs +++ b/cache2/src/snapshot.rs @@ -46,10 +46,8 @@ pub enum FillPressure { /// Pressure observation is disabled. #[default] Disabled, - /// Configured rate ceilings apply without a pressure reduction. + /// Configured rate ceilings apply. Healthy, - /// Fill rate has been reduced while accepted work drains. - Throttled, /// New fills are paused while outstanding work is old or stalled. Paused, } @@ -62,13 +60,13 @@ pub struct FillControlSnapshot { pub pressure: FillPressure, /// Whether the controller enforces its admission decisions. pub enforcing: bool, - /// Current encoded-byte admission rate. + /// Configured encoded-byte ceiling when healthy; zero while paused. pub bytes_per_second: u64, - /// Current fill-record admission rate. + /// Configured fill-record ceiling when healthy; zero while paused. pub records_per_second: u32, - /// Fills rejected by Adaptive, including bounded CAS contention. + /// Fills rejected by Adaptive while paused. pub rejections: u64, - /// Observe-mode pause or budget refusals; excludes CAS/epoch contention. + /// Observe-mode pause refusals. pub would_reject: u64, /// Background observations skipped because the table was full. pub dropped_observations: u64, @@ -76,22 +74,15 @@ pub struct FillControlSnapshot { pub outstanding_operations: u64, /// Bytes held by those background operations; excludes unflushed staging. pub outstanding_bytes: u64, - /// Age of the oldest background operation, including admission/validation. + /// Age of the oldest background operation. pub oldest_operation_ns: u64, - /// Estimated drain time using recent validated background throughput. + /// Estimated drain time using validated background throughput since open. pub estimated_drain_ns: u64, - /// Accumulated background admission time in nanoseconds. - pub admission_ns: u64, - /// Accumulated time from admission to result consumption, including scheduling. - pub completion_wait_ns: u64, - /// Accumulated time from result consumption through validation/publication. - pub validation_ns: u64, } /// Point-in-time operational counters and cache-owned resource accounting. -/// Sampling uses atomics, plus a short controller lock when fill control is enabled. Counters are -/// process-local and reset on every open. Concurrent updates may appear across fields at slightly -/// different instants. +/// Sampling uses atomics. Counters are process-local and reset on every open. +/// Concurrent updates may appear across fields at slightly different instants. #[non_exhaustive] #[derive(Clone, Copy, Debug, Eq, PartialEq)] pub struct CacheSnapshot { From 709d2eb3de7d3ac5ab4856824f829365dceed6e2 Mon Sep 17 00:00:00 2001 From: leiysky Date: Sun, 20 Sep 2026 23:50:49 +0800 Subject: [PATCH 4/5] fix: keep fill pacing usable at configured ceilings Raise the record ceiling to the packed-credit limit, preserve refill time across sub-tick wakes, and refund only consumed burst. Document that FillLimits are instance-wide. --- ARCHITECTURE.md | 2 +- CONFIGURATION.md | 19 ++- README.md | 13 +- cache2/src/config/runtime.rs | 18 ++- cache2/src/io/fill_control.rs | 199 ++++++++++++++++++++++--------- cache2/src/region/runtime/mod.rs | 20 ++-- 6 files changed, 192 insertions(+), 79 deletions(-) diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index 5bd66df..db41900 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -118,7 +118,7 @@ Each lane uses one concrete `IoEngine` for admission, submission, cancellation, `io::fill_control` owns optional background observations and adaptive fill admission. `Disabled` retains the existing path. Enabled modes preallocate a fixed observation table sized by append and reclaim worker counts. Background workers register before engine admission and retain their observations through completion validation and publication. A full table skips that request and counts `dropped_observations` instead of failing I/O. Workers checkpoint at one quarter of the I/O deadline, capped at 500 ms, and pause new fills without entering timeout recovery. Aggregate validated throughput provides an approximate drain estimate for snapshots and does not drive pause decisions. Pressure is `Healthy` or `Paused` and does not change terminal health or prove a device fault. -Adaptive foreground `put`/`put_l2` load pause-holder state and otherwise only compete for staging. Byte and record ceilings pace non-essential background flush on shard workers; urgent, drain, and rotation flushes always proceed. A flush that cannot take a span refunds its budget and retries. `Observe` does not delay flush or reject fills and counts pause refusals as `would_reject`; it still checkpoints waits so pause is observable. Adaptive rejects new fills immediately while paused and skips optional reinsertion. Pause is released when the slow I/O completes, not after later publication. Reads, deletes, and essential reclaim bypass fill budgets. Close stops admission independently of outstanding I/O. +Adaptive foreground `put`/`put_l2` load pause-holder state and otherwise only compete for staging. Byte and record ceilings are instance-wide, shared by every shard worker; they pace non-essential background flush. Urgent, drain, and rotation flushes always proceed. A flush that cannot take a span refunds the consumed budget and retries. `Observe` does not delay flush or reject fills and counts pause refusals as `would_reject`; it still checkpoints waits so pause is observable. Adaptive rejects new fills immediately while paused and skips optional reinsertion. Pause is released when the slow I/O completes, not after later publication. Reads, deletes, and essential reclaim bypass fill budgets. Close stops admission independently of outstanding I/O. See [adaptive fill admission](CONFIGURATION.md#adaptive-fill-admission) for rate ceilings, pause conditions, bounded bursts, and tuning limits. diff --git a/CONFIGURATION.md b/CONFIGURATION.md index 121756c..47d203f 100644 --- a/CONFIGURATION.md +++ b/CONFIGURATION.md @@ -104,6 +104,7 @@ The managed-memory limit is not an RSS limit. Allocator metadata, Tokio, the app | Enable or lengthen read waiting | Trades immediate misses for bounded wait | Adds the configured wait capacity, can raise p99, and returns overload when queue, memory, or deadline is exhausted | | Increase write execution capacity | Adds write I/O concurrency | Adds queue pressure and device contention; POSIX also adds worker stacks and neither backend adds foreground staging space | | Increase reclaim concurrency | Recycles more Regions concurrently | Adds one Region buffer and logical worker per request slot and can compete for device and append-staging capacity | +| Enable Adaptive fill control | Pauses new fills; paces flush | Ceilings are instance-wide; they do not change reads, deletes, or `CacheHealth` on pre-timeout pause | | Increase L1 capacity | Retains more reusable values | Reduces L2 demand but consumes retained bytes and fixed metadata from the same managed-memory limit | | Increase L1 shards | Reduces shard-local contention | Adds fixed metadata and controls; too many small shards reduce useful capacity efficiency | | Lower write flush threshold | Requests earlier partial publication | Increases write operation count and reduces batching without increasing staging capacity | @@ -379,6 +380,9 @@ Use `Cache::snapshot()` for regular telemetry and `Cache::detailed_snapshot()` f | `reinsert_budget_skipped` rises | Hot live bytes exceed the fixed reclaim allowance | Treat retention as best effort; change capacity/workload geometry rather than worker count | | High `l1_bypasses` with useful candidates | L1 contention, slots, or byte pressure | Inspect L1 occupancy, retained bytes, shards, capacity, and oversized values | | Managed-memory peak approaches the limit | Fixed or transient memory pressure | Rebalance index, L1, staging, I/O topology, and read headroom | +| `fill_control.pressure` is `Paused` while `health` is `Running` | Pre-timeout slow background I/O | Expected Adaptive load-shed; correlate with device queueing before treating it as a fault | +| Pause refusals with little device wait | Checkpoint fired without device saturation | Confirm I/O timeouts and host scheduling; `FillLimits` do not drive pause | +| `write_rejections` while fill pressure is `Healthy` | Staging or paced-flush pressure | Distinguish from pause; change ceilings only after measuring sustainable fill | | Unexpected cold start after configuration change | Static identity or append-shard rebind failed | Verify Region/index geometry and available Free Regions; cache loss is safe | Always correlate cache counters with device latency, physical IOPS, filesystem behavior, process RSS, and authoritative-backend load. Cache throughput alone can reward configurations that merely turn work into fast misses. @@ -393,8 +397,9 @@ Always correlate cache counters with device latency, physical IOPS, filesystem b 6. Tune read execution and wait semantics using hit rate, overload, and p99. 7. Tune append shards, write execution, and flush threshold using acceptance, publication, and device counters. 8. Increase reclaim concurrency only if one reclaimer cannot maintain Free Regions. -9. Re-run after selecting Buffered versus Direct or POSIX versus io_uring; backend topology values are not interchangeable. -10. Validate cold and warm opens, `drain`, overload behavior, managed-memory peak, and final value correctness before deployment. +9. Enable fill control after write and reclaim topology is stable: `Observe` with chosen instance-wide ceilings, then `Adaptive` with the same values. +10. Re-run after selecting Buffered versus Direct or POSIX versus io_uring; backend topology values are not interchangeable. +11. Validate cold and warm opens, `drain`, overload behavior, managed-memory peak, and final value correctness before deployment. Change one resource family at a time and alternate baseline and candidate runs on the same host. Use multiple fresh-cache samples for turnover and burst tests. For storage qualification, use a dataset larger than host RAM and follow [Validation](BENCHMARK.md). @@ -414,6 +419,8 @@ Change one resource family at a time and alternate baseline and candidate runs o | Maximum waiting reads | 1 through 65,536 in `Wait`; defaults to the read execution capacity | | L1 shards | 1 through 65,536 | | Write flush threshold | 4 KiB multiple from 4 KiB through 4 MiB | +| Fill bytes per second | 640 through 1 TiB/s when fill control is enabled; instance-wide | +| Fill records per second | 10 through 4,294,967,295 when fill control is enabled; instance-wide | | Managed-memory limit | Nonzero, at least L1 capacity, and large enough for the validated fixed footprint | Storage and configuration construction enforce these bounds before file access. Open checks the selected filesystem, device, and runtime environment. @@ -434,16 +441,18 @@ Set `options.reclaim_io_timeout = Duration::from_secs(30)` after constructing `R use cache2::{FillControlOptions, FillLimits, RuntimeOptions}; let mut options = RuntimeOptions::default(); -// Example ceilings; choose values for the measured workload. +// Instance-wide mixed-write starting point; not a device sequential-write spec. let limits = FillLimits::new(256 * 1024 * 1024, 20_000); options.fill_control = FillControlOptions::Observe(limits); // After evaluating pressure and would_reject, enforce the same policy: options.fill_control = FillControlOptions::Adaptive(limits); ``` -The two ceilings constrain logical encoded fill bytes (charged in 64-byte units) and fill records flushed by shard workers, not device bandwidth or physical IOPS. Valid ceilings are 640 bytes/s through 1 TiB/s and 10 through 655350 records/s. Adaptive refill adds elapsed credit up to 100 ms of the configured ceilings and at least one maximum-size record. An idle controller never accumulates more than that bounded burst. Urgent, drain, and rotation flushes always proceed; other flushes wait for budget and back-pressure staging. Foreground admission does not consume the bucket. Observe does not delay flush and counts `would_reject` only for pause. +The two ceilings are instance-wide: every append shard shares one token bucket. They constrain logical encoded fill bytes (charged in 64-byte units) and fill records from the worker's staging snapshot, not device bandwidth or physical IOPS. `put` and `put_l2` do not consume the bucket; Adaptive rejects those calls only while paused. The sealed span may grow by later encodes before the lock. Valid ceilings are 640 bytes/s through 1 TiB/s and 10 through 4,294,967,295 records/s. Adaptive refill adds elapsed credit up to 100 ms of the configured ceilings and at least one maximum-size record; intervals too short to mint a unit or record do not move the refill clock. An idle controller never accumulates more than that bounded burst. A flush larger than that burst still proceeds on remaining credit, and a flush that cannot take a span refunds only what it consumed. Urgent, drain, and rotation flushes always proceed; other flushes wait for budget and back-pressure staging into ordinary write overload. Observe does not delay flush and counts `would_reject` only for pause. -Background workers checkpoint outstanding I/O at one quarter of the normal deadline, capped at 500 ms, and pause new Adaptive fills without entering timeout recovery. Checkpoint pause follows live holder counts and lifts when that I/O completes, even if later validation or reclaim scanning is still running. Observe uses the same checkpoints so it can report `would_reject`. Estimated drain time is reported for snapshots from validated throughput since open and does not pause admission. These are conservative pressure signals: very short deadlines or a descheduled worker can still reach timeout first. Idle time alone never looks like a stall. Normal timeout recovery keeps its admission fence until all affected work is validated and published. +Background workers checkpoint outstanding I/O at one quarter of the normal deadline, capped at 500 ms, and pause new Adaptive fills without entering timeout recovery. Checkpoint pause follows live holder counts and lifts when that I/O completes, even if later validation or reclaim scanning is still running. Observe uses the same checkpoints so it can report `would_reject`. Estimated drain time is reported for snapshots from validated throughput since open and does not pause admission. These are conservative pressure signals: very short deadlines or a descheduled worker can still reach timeout first. Idle time alone never looks like a stall. Pause and the rate bucket are independent: a busy device is shed by checkpoint or timeout recovery, not by copying sequential-write specifications into `FillLimits`. Normal timeout recovery keeps its admission fence until all affected work is validated and published. + +Start with `Observe` and the same ceilings you plan to enforce. Inspect `CacheSnapshot::fill_control` (`pressure`, `would_reject`, `oldest_operation_ns`) together with device queueing, then switch to `Adaptive` without changing the numbers. The example `256 MiB/s` and `20_000` records/s is a conservative mixed-write starting point, not an NVMe datasheet copy. Both dimensions must hold: if typical encoded records are about 1 KiB, a 256 MiB/s byte ceiling needs on the order of 200_000–500_000 records/s or the record limit fires first. To pause ingest on stalls while leaving healthy flush uncapped, raise both ceilings until they no longer clip measured fill (for example 1–2 GiB/s with a matching record rate). To keep headroom for reads and reclaim on a shared NVMe, set the byte ceiling around one third to one half of measured sustainable cache write; four write workers and a 4 MiB flush threshold often land in the 256–512 MiB/s range. Do not use the legal minima (640 bytes/s, 10 records/s) as NVMe defaults. Buffered macOS results do not qualify Linux NVMe; use [Validation](BENCHMARK.md) on the target device. `CacheSnapshot::fill_control` reports pressure, enforcement, configured or paused rates, rejections, hypothetical pause refusals, dropped observations, outstanding bytes/operations, oldest age, and estimated drain time independently of `RuntimeOptions::stats`. Outstanding bytes exclude unflushed staging. A zero drain estimate means no estimate is available when work is pending. A full observation table skips that request rather than failing I/O. These measurements cannot distinguish device throttling from CPU scheduling, kernel queueing, or slow validation; correlate them with host metrics before diagnosing hardware. `cache_fill_pressure_changed` logs state transitions under `cache2::health`. diff --git a/README.md b/README.md index 56e5778..736cd63 100644 --- a/README.md +++ b/README.md @@ -83,7 +83,7 @@ See the [configuration guide](CONFIGURATION.md#configuration-lifecycle) for exam | Memory | `managed_memory_limit_bytes` | 1 GiB across cache-managed allocations. | | I/O mode | `io_mode` | Buffered I/O. | | Metrics | `stats: StatsOptions` | Health/resource gauges always available; activity, request, and latency collection opt in. | -| Fill control | `fill_control: FillControlOptions` | Disabled. `Observe` reports pause pressure; `Adaptive` rejects new fills before timeout. | +| Fill control | `fill_control: FillControlOptions` | Disabled. `Observe` reports pause; `Adaptive` rejects new fills. Ceilings are cache-wide. | Changing the append-shard count rebinds recovered Active Regions during a warm open. Growth uses available Free Regions; when there are not enough, the disposable cache safely starts empty. @@ -132,6 +132,12 @@ RUST_LOG=cache2=info cargo run --package examples --example logforth -- /tmp/cac `cache_opened` reports the index backing, mapping extent, validation mode, and whether warm mutations use copy-on-write. `cache_recovery_cold` records why a clean image was rejected or why private mapping fell back to a cold start. `cache_miss_only` records the first terminal index-validation or I/O failure. +### Reclaim read deadline + +Set `RuntimeOptions::reclaim_io_timeout` to change the normal background reclaim deadline, for example `Duration::from_secs(30)` (default five seconds). Background write and reclaim timeouts enter `CacheHealth::Recovering`: new fills return overload while reads and deletes remain available. `RuntimeOptions::io_recovery_timeout` defaults to `None`, allowing recovery until completion or close. Use `Some(Duration::from_secs(300))` to limit the additional wait, or `Some(Duration::ZERO)` for immediate cancellation. Original requests retain their resources and are never resubmitted; fills resume after validation and publication of all affected work. Close interrupts recovery; drain may wait indefinitely. Actual I/O errors and invalid completions still fail the cache. + +Optional [adaptive fill admission](CONFIGURATION.md#adaptive-fill-admission) detects slow background progress before timeout. `Adaptive` pauses new fills immediately and paces non-essential flush with instance-wide byte and record ceilings. Start with `FillControlOptions::Observe` to inspect pause pressure and hypothetical rejections, then use `Adaptive` to enforce the same limits. It is disabled by default; reads, deletes, accepted writes, and essential reclaim retain their existing paths. + ## Development C² requires Rust 1.98.0. @@ -156,8 +162,3 @@ The root workspace keeps the publishable crate, integration tests, benchmarks, e Licensed under the [Apache License, Version 2.0](LICENSE). -### Reclaim read deadline - -Set `RuntimeOptions::reclaim_io_timeout` to change the normal background reclaim deadline, for example `Duration::from_secs(30)` (default five seconds). Background write and reclaim timeouts enter `CacheHealth::Recovering`: new fills return overload while reads and deletes remain available. `RuntimeOptions::io_recovery_timeout` defaults to `None`, allowing recovery until completion or close. Use `Some(Duration::from_secs(300))` to limit the additional wait, or `Some(Duration::ZERO)` for immediate cancellation. Original requests retain their resources and are never resubmitted; fills resume after validation and publication of all affected work. Close interrupts recovery; drain may wait indefinitely. Actual I/O errors and invalid completions still fail the cache. - -Optional [adaptive fill admission](CONFIGURATION.md#adaptive-fill-admission) detects slow background progress before timeout. `Adaptive` pauses new fills immediately and paces non-essential flush by encoded bytes and record count. Start with `FillControlOptions::Observe` to inspect pause pressure and hypothetical rejections, then use `Adaptive` to enforce. It is disabled by default; reads, deletes, accepted writes, and essential reclaim retain their existing paths. diff --git a/cache2/src/config/runtime.rs b/cache2/src/config/runtime.rs index 9017c5b..390ed74 100644 --- a/cache2/src/config/runtime.rs +++ b/cache2/src/config/runtime.rs @@ -367,13 +367,17 @@ pub enum FillControlOptions { } /// Logical fill-rate ceilings shared by [`FillControlOptions::Observe`] and -/// [`FillControlOptions::Adaptive`]. These are not device bandwidth or IOPS guarantees. +/// [`FillControlOptions::Adaptive`]. They are instance-wide, not per worker, +/// and are not device bandwidth or IOPS guarantees. Foreground `put` does not +/// consume them; Adaptive uses them to pace non-essential background flush. #[non_exhaustive] #[derive(Clone, Copy, Debug, Eq, PartialEq)] pub struct FillLimits { /// Maximum encoded fill bytes per second, from 640 through 1 TiB/s. + /// Instance-wide across all shard workers. pub max_bytes_per_second: u64, - /// Maximum fill records per second, from 10 through 655350. + /// Maximum fill records per second, from 10 through 4,294,967,295. + /// Instance-wide across all shard workers. pub max_records_per_second: u32, } @@ -852,7 +856,7 @@ mod tests { let storage = StorageOptions::new(1024 * 1024 * 1024).build().unwrap(); let base = CacheConfig::new(storage.clone(), RuntimeOptions::default()).unwrap(); assert_eq!(base.runtime().fill_control, FillControlOptions::Disabled); - for (bytes, operations) in [(639, 100), ((1 << 40) + 1, 100), (640, 9), (640, 655_351)] { + for (bytes, operations) in [(639, 100), ((1 << 40) + 1, 100), (640, 9)] { let options = RuntimeOptions { fill_control: FillControlOptions::Adaptive(FillLimits::new(bytes, operations)), ..RuntimeOptions::default() @@ -864,6 +868,14 @@ mod tests { ErrorKind::InvalidInput ); } + CacheConfig::new( + storage.clone(), + RuntimeOptions { + fill_control: FillControlOptions::Adaptive(FillLimits::new(640, u32::MAX)), + ..RuntimeOptions::default() + }, + ) + .unwrap(); let mut minimum = None; for mode in [FillControlOptions::Observe, FillControlOptions::Adaptive] { let config = CacheConfig::new( diff --git a/cache2/src/io/fill_control.rs b/cache2/src/io/fill_control.rs index a2b99e0..dde7541 100644 --- a/cache2/src/io/fill_control.rs +++ b/cache2/src/io/fill_control.rs @@ -34,13 +34,20 @@ const MAX_CAS_ATTEMPTS: usize = 4; const MIN_BYTES_PER_SECOND: u64 = 640; const MAX_BYTES_PER_SECOND: u64 = 1 << 40; const MIN_RECORDS_PER_SECOND: u32 = 10; -const MAX_RECORDS_PER_SECOND: u32 = 655_350; +const MAX_RECORDS_PER_SECOND: u32 = u32::MAX; const STALL_CAP: Duration = Duration::from_millis(500); const TICKS_PER_SECOND: u64 = 10; /// Wake shard workers this often when a non-essential flush is waiting for budget. pub const FLUSH_RETRY: Duration = Duration::from_millis(100); +/// Credit taken by one non-essential flush. Zero for Observe and essential work. +#[derive(Clone, Copy)] +pub struct FlushCharge { + pub(crate) ops: u64, + pub(crate) units: u64, +} + fn packed_ops(value: u64) -> u64 { value >> 32 } @@ -313,47 +320,54 @@ impl FillController { } /// Background flush pacing. Essential flushes always proceed. - pub fn try_flush(&self, bytes: u64, records: u32, essential: bool) -> bool { + pub fn try_flush(&self, bytes: u64, records: u32, essential: bool) -> Option { if essential || !self.enforcing { - return true; + return Some(FlushCharge { ops: 0, units: 0 }); } self.refill(); let raw_units = bytes.div_ceil(UNIT).max(1); let raw_ops = u64::from(records.max(1)); let units = raw_units.min(self.max_units); let ops = raw_ops.min(self.max_ops); - let oversized = raw_units >= self.max_units; + let byte_oversized = raw_units >= self.max_units; + let record_oversized = raw_ops >= self.max_ops; let mut value = self.credit.load(Ordering::Relaxed); for _ in 0..MAX_CAS_ATTEMPTS { - if packed_units(value) == 0 || packed_ops(value) == 0 { - return false; + let have_units = packed_units(value); + let have_ops = packed_ops(value); + if have_units == 0 || have_ops == 0 { + return None; } - if !oversized && (packed_units(value) < units || packed_ops(value) < ops) { - return false; + if !byte_oversized && have_units < units { + return None; } - let next = if oversized { - 0 - } else { - pack_credit(packed_ops(value) - ops, packed_units(value) - units) - }; + if !record_oversized && have_ops < ops { + return None; + } + let take_units = if byte_oversized { have_units } else { units }; + let take_ops = if record_oversized { have_ops } else { ops }; + let next = pack_credit(have_ops - take_ops, have_units - take_units); match self .credit .compare_exchange(value, next, Ordering::AcqRel, Ordering::Relaxed) { - Ok(_) => return true, + Ok(_) => { + return Some(FlushCharge { + ops: take_ops, + units: take_units, + }); + } Err(current) => value = current, } } - false + None } - pub fn refund_flush(&self, bytes: u64, records: u32) { - if !self.enforcing { + pub fn refund_flush(&self, charge: FlushCharge) { + if !self.enforcing || (charge.ops == 0 && charge.units == 0) { return; } - let units = bytes.div_ceil(UNIT).max(1).min(self.max_units); - let ops = u64::from(records.max(1)).min(self.max_ops); - self.add_credit(ops, units); + self.add_credit(charge.ops, charge.units); } fn refill(&self) { @@ -362,13 +376,6 @@ impl FillController { let Some(dt) = now.checked_sub(last).filter(|dt| *dt != 0) else { return; }; - if self - .last_refill_ns - .compare_exchange(last, now, Ordering::Relaxed, Ordering::Relaxed) - .is_err() - { - return; - } let add_units = self .options .max_bytes_per_second @@ -383,22 +390,30 @@ impl FillController { if add_units == 0 && add_ops == 0 { return; } + if self + .last_refill_ns + .compare_exchange(last, now, Ordering::Relaxed, Ordering::Relaxed) + .is_err() + { + return; + } self.add_credit(add_ops, add_units); } fn add_credit(&self, ops: u64, units: u64) { + if ops == 0 && units == 0 { + return; + } let mut value = self.credit.load(Ordering::Relaxed); - for _ in 0..MAX_CAS_ATTEMPTS { + loop { let next = pack_credit( (packed_ops(value) + ops).min(self.max_ops), (packed_units(value) + units).min(self.max_units), ); - match self.credit.compare_exchange_weak( - value, - next, - Ordering::Relaxed, - Ordering::Relaxed, - ) { + match self + .credit + .compare_exchange(value, next, Ordering::Relaxed, Ordering::Relaxed) + { Ok(_) => return, Err(current) => value = current, } @@ -508,9 +523,9 @@ mod tests { fn foreground_admit_ignores_flush_budget() { let control = control(true); for _ in 0..10 { - assert!(control.try_flush(64, 1, false)); + assert!(control.try_flush(64, 1, false).is_some()); } - assert!(!control.try_flush(64, 1, false)); + assert!(control.try_flush(64, 1, false).is_none()); assert!(control.try_admit()); } @@ -518,22 +533,22 @@ mod tests { fn record_budget_paces_nonessential_flush() { let control = control(true); for _ in 0..10 { - assert!(control.try_flush(64, 1, false)); + assert!(control.try_flush(64, 1, false).is_some()); } - assert!(!control.try_flush(64, 1, false)); - assert!(control.try_flush(64, 1, true)); + assert!(control.try_flush(64, 1, false).is_none()); + assert!(control.try_flush(64, 1, true).is_some()); restore_burst(&control); - assert!(control.try_flush(64, 1, false)); + assert!(control.try_flush(64, 1, false).is_some()); } #[test] fn byte_budget_paces_nonessential_flush() { let control = control(true); - assert!(control.try_flush(4096, 1, false)); - assert!(!control.try_flush(4096, 1, false)); - assert!(control.try_flush(64, 1, false)); + assert!(control.try_flush(4096, 1, false).is_some()); + assert!(control.try_flush(4096, 1, false).is_none()); + assert!(control.try_flush(64, 1, false).is_some()); restore_burst(&control); - assert!(control.try_flush(4096, 1, false)); + assert!(control.try_flush(4096, 1, false).is_some()); } #[test] @@ -568,7 +583,7 @@ mod tests { fn observe_would_reject_counts_pause_not_budget() { let control = control(false); for _ in 0..20 { - assert!(control.try_flush(4096, 1, false)); + assert!(control.try_flush(4096, 1, false).is_some()); assert!(control.try_admit()); } assert_eq!(control.snapshot().would_reject, 0); @@ -599,9 +614,9 @@ mod tests { ) .unwrap() .unwrap(); - assert!(control.try_flush(4096, 1, false)); + assert!(control.try_flush(4096, 1, false).is_some()); restore_burst(&control); - assert!(control.try_flush(4096, 1, false)); + assert!(control.try_flush(4096, 1, false).is_some()); control.stop(); assert!(!control.try_admit()); } @@ -662,7 +677,7 @@ mod tests { let accepted = &accepted; scope.spawn(move || { for _ in 0..100 { - if control.try_flush(640, 1, false) { + if control.try_flush(640, 1, false).is_some() { accepted.fetch_add(1, Ordering::Relaxed); } } @@ -723,19 +738,95 @@ mod tests { fn burst_sized_flush_proceeds_on_remaining_credit() { let control = control(true); freeze_refill(&control); - assert!(control.try_flush(64, 1, false)); - assert!(control.try_flush(1_000_000, 1, false)); + assert!(control.try_flush(64, 1, false).is_some()); + assert!(control.try_flush(1_000_000, 1, false).is_some()); + restore_burst(&control); + freeze_refill(&control); + assert!(control.try_flush(64, 1, false).is_some()); + assert!(control.try_flush(64, 1_000_000, false).is_some()); } #[test] fn refund_restores_nonessential_budget() { let control = control(true); freeze_refill(&control); + let mut last = FlushCharge { ops: 0, units: 0 }; for _ in 0..10 { - assert!(control.try_flush(64, 1, false)); + last = control.try_flush(64, 1, false).unwrap(); } - assert!(!control.try_flush(64, 1, false)); - control.refund_flush(64, 1); - assert!(control.try_flush(64, 1, false)); + assert!(control.try_flush(64, 1, false).is_none()); + control.refund_flush(last); + assert!(control.try_flush(64, 1, false).is_some()); + } + + #[test] + fn sub_tick_refills_do_not_burn_elapsed_time() { + let control = FillController::new( + FillControlOptions::Adaptive(FillLimits::new(640, 10)), + 1, + 4096, + ) + .unwrap() + .unwrap(); + freeze_refill(&control); + while control.try_flush(64, 1, false).is_some() {} + let one_ms_ago = control.elapsed_ns().saturating_sub(1_000_000); + control.last_refill_ns.store(one_ms_ago, Ordering::Relaxed); + for _ in 0..32 { + control.refill(); + } + assert_eq!(control.last_refill_ns.load(Ordering::Relaxed), one_ms_ago); + std::thread::sleep(Duration::from_millis(110)); + control.refill(); + assert!(control.try_flush(64, 1, false).is_some()); + } + + #[test] + fn concurrent_refunds_are_not_dropped() { + let control = FillController::new( + FillControlOptions::Adaptive(FillLimits::new(64_000, 10_000)), + 1, + 4096, + ) + .unwrap() + .unwrap(); + freeze_refill(&control); + while control.try_flush(64, 1, false).is_some() {} + let before = control.credit.load(Ordering::Relaxed); + std::thread::scope(|scope| { + for _ in 0..32 { + let control = &control; + scope.spawn(move || { + control.refund_flush(FlushCharge { ops: 1, units: 1 }); + }); + } + }); + let credit = control.credit.load(Ordering::Relaxed); + assert_eq!( + packed_ops(credit), + (packed_ops(before) + 32).min(control.max_ops) + ); + assert_eq!( + packed_units(credit), + (packed_units(before) + 32).min(control.max_units) + ); + } + + #[test] + fn oversized_refund_restores_only_consumed_remainder() { + let control = control(true); + freeze_refill(&control); + for _ in 0..9 { + assert!(control.try_flush(64, 1, false).is_some()); + } + let charge = control.try_flush(1_000_000, 1, false).unwrap(); + assert_eq!(charge.ops, 1); + assert_eq!(charge.units, control.max_units - 9); + control.refund_flush(charge); + let credit = control.credit.load(Ordering::Relaxed); + assert_eq!(packed_ops(credit), 1); + assert_eq!(packed_units(credit), charge.units); + assert!(control.try_flush(64, 1, false).is_some()); + assert!(control.try_flush(64, 1, false).is_none()); } } diff --git a/cache2/src/region/runtime/mod.rs b/cache2/src/region/runtime/mod.rs index 2a5b01e..bc98686 100644 --- a/cache2/src/region/runtime/mod.rs +++ b/cache2/src/region/runtime/mod.rs @@ -63,6 +63,7 @@ use crate::io::engine::recovery::BackgroundRecovery; use crate::io::engine::submit_background_io; use crate::io::fill_control::FLUSH_RETRY; use crate::io::fill_control::FillController; +use crate::io::fill_control::FlushCharge; use crate::managed_memory::BufferLease; use crate::managed_memory::CACHE_THREAD_STACK_BYTES; use crate::managed_memory::ManagedMemory; @@ -1922,14 +1923,11 @@ fn shard_worker_result( let essential = flags & (WAKE_URGENT | WAKE_ROTATE) != 0 || draining; let bytes = fill.bytes as u64; let records = u32::try_from(fill.records).unwrap_or(u32::MAX); - let allow = shared - .recovery - .fill - .as_ref() - .is_none_or(|control| control.try_flush(bytes, records, essential)); - if !allow { - deadline = Some(Instant::now() + FLUSH_RETRY); - } else { + let charge = match &shared.recovery.fill { + None => Some(FlushCharge { ops: 0, units: 0 }), + Some(control) => control.try_flush(bytes, records, essential), + }; + if let Some(charge) = charge { let engine = shared.write_engine_for(shard_id as u64); match shared.core.flush_staging_shard( &shared.staging, @@ -1939,12 +1937,14 @@ fn shard_worker_result( )? { Some(_) => deadline = None, None => { - if !essential && let Some(control) = &shared.recovery.fill { - control.refund_flush(bytes, records); + if let Some(control) = &shared.recovery.fill { + control.refund_flush(charge); } deadline = Some(Instant::now() + STAGING_RETRY_DELAY); } } + } else { + deadline = Some(Instant::now() + FLUSH_RETRY); } } } From 69ccbeb725156751c1a105c19b7053dd8d3d4346 Mon Sep 17 00:00:00 2001 From: leiysky Date: Mon, 21 Sep 2026 00:02:27 +0800 Subject: [PATCH 5/5] fix: construct fill-pressure tests with the unified IoEngine #57 collapsed BackendIoEngine; the Adaptive pause test now uses IoEngine::for_test. --- cache2/src/io/engine/tests.rs | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/cache2/src/io/engine/tests.rs b/cache2/src/io/engine/tests.rs index 7344571..81b5c9f 100644 --- a/cache2/src/io/engine/tests.rs +++ b/cache2/src/io/engine/tests.rs @@ -1408,7 +1408,7 @@ fn adaptive_pressure_pauses_before_real_engine_timeout_and_resumes_after_io_comp let control = FillController::new(mode, 1, 4096).unwrap().unwrap(); let recovery = BackgroundRecovery::with_fill(None, Some(Arc::clone(&control))); let backend = Arc::new(BlockingBackend::default()); - let engine = BackendIoEngine::new(backend.clone(), 1).unwrap(); + let engine = IoEngine::for_test(backend.clone(), 1).unwrap(); let memory = managed_memory(); std::thread::scope(|scope| { let (returned_tx, returned_rx) = mpsc::channel();