diff --git a/.github/workflows/ci.yaml b/.github/workflows/ci.yaml index aa110904d..18d4216cd 100644 --- a/.github/workflows/ci.yaml +++ b/.github/workflows/ci.yaml @@ -190,6 +190,44 @@ jobs: RUNTIME_REQUIRE_IO_URING: "1" run: timeout --signal=TERM --kill-after=10s 300s cargo +1.96.0 test --locked --manifest-path cmd/racer-dataplane/Cargo.toml -p uring-runtime --no-default-features + # Keep allocator invocations separate from runtime/workspace tests: runtime + # test features must not enable simulation in the production allocator gate. + - name: Check allocator production and simulation + run: | + timeout --signal=TERM --kill-after=10s 300s cargo +1.96.0 check --locked --manifest-path cmd/racer-dataplane/Cargo.toml -p page-alloc --all-targets --no-default-features + timeout --signal=TERM --kill-after=10s 300s cargo +1.96.0 check --locked --manifest-path cmd/racer-dataplane/Cargo.toml -p page-alloc --all-targets --no-default-features --features simulation + + - name: Strict allocator Clippy + run: | + timeout --signal=TERM --kill-after=10s 300s cargo +1.96.0 clippy --locked --manifest-path cmd/racer-dataplane/Cargo.toml -p page-alloc --all-targets --no-default-features -- -D warnings + timeout --signal=TERM --kill-after=10s 300s cargo +1.96.0 clippy --locked --manifest-path cmd/racer-dataplane/Cargo.toml -p page-alloc --all-targets --no-default-features --features simulation -- -D warnings + + - name: Allocator production real I/O (no simulation) + env: + PAGE_ALLOC_REQUIRE_REAL_IO: "1" + run: timeout --signal=TERM --kill-after=10s 300s cargo +1.96.0 test --locked --manifest-path cmd/racer-dataplane/Cargo.toml -p page-alloc --no-default-features + + - name: Test allocator simulation + run: timeout --signal=TERM --kill-after=10s 300s cargo +1.96.0 test --locked --manifest-path cmd/racer-dataplane/Cargo.toml -p page-alloc --no-default-features --features simulation + + # Isolate flow features so workspace tests cannot substitute simulated I/O + # for the production pipe and socket contracts. Real-I/O failures never skip. + - name: Check flow production and simulation + run: | + timeout --signal=TERM --kill-after=10s 300s cargo +1.96.0 check --locked --manifest-path cmd/racer-dataplane/Cargo.toml -p flow-control --all-targets --no-default-features + timeout --signal=TERM --kill-after=10s 300s cargo +1.96.0 check --locked --manifest-path cmd/racer-dataplane/Cargo.toml -p flow-control --all-targets --no-default-features --features simulation + + - name: Strict flow Clippy + run: | + timeout --signal=TERM --kill-after=10s 300s cargo +1.96.0 clippy --locked --manifest-path cmd/racer-dataplane/Cargo.toml -p flow-control --all-targets --no-default-features -- -D warnings + timeout --signal=TERM --kill-after=10s 300s cargo +1.96.0 clippy --locked --manifest-path cmd/racer-dataplane/Cargo.toml -p flow-control --all-targets --no-default-features --features simulation -- -D warnings + + - name: Flow production real I/O (no simulation) + run: timeout --signal=TERM --kill-after=10s 300s cargo +1.96.0 test --locked --manifest-path cmd/racer-dataplane/Cargo.toml -p flow-control --no-default-features + + - name: Test flow simulation + run: timeout --signal=TERM --kill-after=10s 300s cargo +1.96.0 test --locked --manifest-path cmd/racer-dataplane/Cargo.toml -p flow-control --no-default-features --features simulation + racer-verbs: name: Racer Verbs (${{ matrix.configuration }}) runs-on: ubuntu-24.04 diff --git a/cmd/racer-dataplane/Cargo.lock b/cmd/racer-dataplane/Cargo.lock index 7e2adc276..85ce51f94 100644 --- a/cmd/racer-dataplane/Cargo.lock +++ b/cmd/racer-dataplane/Cargo.lock @@ -52,6 +52,17 @@ dependencies = [ "crypto-common", ] +[[package]] +name = "flow-control" +version = "0.1.0" +dependencies = [ + "futures", + "libc", + "page-alloc", + "uring-runtime", + "zeroize", +] + [[package]] name = "futures" version = "0.3.34" @@ -190,6 +201,15 @@ version = "2.8.3" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98" +[[package]] +name = "page-alloc" +version = "0.1.0" +dependencies = [ + "libc", + "uring-runtime", + "zeroize", +] + [[package]] name = "pin-project-lite" version = "0.2.17" diff --git a/cmd/racer-dataplane/Cargo.toml b/cmd/racer-dataplane/Cargo.toml index e778e57fc..abf2b281b 100644 --- a/cmd/racer-dataplane/Cargo.toml +++ b/cmd/racer-dataplane/Cargo.toml @@ -1,5 +1,5 @@ [workspace] -members = [".", "topology", "runtime", "telemetry", "verbs"] +members = [".", "topology", "runtime", "alloc", "flow", "telemetry", "verbs"] resolver = "3" [package] diff --git a/cmd/racer-dataplane/README.md b/cmd/racer-dataplane/README.md index 7e68e4e02..cf122eb56 100644 --- a/cmd/racer-dataplane/README.md +++ b/cmd/racer-dataplane/README.md @@ -40,3 +40,38 @@ timeout --signal=TERM --kill-after=10s 300s python3 hack/scripts/runtime-miri.py Run the `offload`, `scheduler`, and `memory` groups separately with the same script. Each group verifies that every exact allowlisted test actually ran. + +The `page-alloc` library provides worker-local aligned buffers, lease-fenced slab +I/O, and bounded segment reclamation. Test it separately from runtime/workspace +tests so feature unification cannot enable simulation in its production gate: + +```sh +PAGE_ALLOC_REQUIRE_REAL_IO=1 timeout --signal=TERM --kill-after=10s 300s cargo +1.96.0 test --locked --manifest-path cmd/racer-dataplane/Cargo.toml -p page-alloc --no-default-features -j 2 -- --test-threads=1 +timeout --signal=TERM --kill-after=10s 300s cargo +1.96.0 test --locked --manifest-path cmd/racer-dataplane/Cargo.toml -p page-alloc --no-default-features --features simulation -j 2 -- --test-threads=1 +``` + +The production gate requires real io_uring and direct-I/O support: capability +skips fail when `PAGE_ALLOC_REQUIRE_REAL_IO=1`. CI also checks all targets and +runs strict Clippy for each allocator mode. + +The `flow-control` library provides policy-driven quotas, charged buffers, bounded +kernel pipes, adaptive admission, and worker-local coalescing. Application policy +and real operation completion remain the caller's responsibility. Its crate docs +describe ownership and admission contracts; generate them with: + +```sh +timeout --signal=TERM --kill-after=10s 300s cargo +1.96.0 doc --locked --manifest-path cmd/racer-dataplane/Cargo.toml -p flow-control --no-deps +``` + +Validate flow separately so workspace feature unification cannot enable simulation +in its production gate: + +```sh +timeout --signal=TERM --kill-after=10s 300s cargo +1.96.0 test --locked --manifest-path cmd/racer-dataplane/Cargo.toml -p flow-control --no-default-features -j 2 -- --test-threads=1 +timeout --signal=TERM --kill-after=10s 300s cargo +1.96.0 test --locked --manifest-path cmd/racer-dataplane/Cargo.toml -p flow-control --no-default-features --features simulation -j 2 -- --test-threads=1 +``` + +The flow production tests require Linux kernel pipes, Unix sockets, and splice. +They exercise real I/O and fail rather than skipping unsupported operations. +Simulation adds simulated and mixed-descriptor contracts without removing the +real pipe tests. CI checks all targets and runs strict Clippy in each mode. diff --git a/cmd/racer-dataplane/alloc/.gitignore b/cmd/racer-dataplane/alloc/.gitignore new file mode 100644 index 000000000..b83d22266 --- /dev/null +++ b/cmd/racer-dataplane/alloc/.gitignore @@ -0,0 +1 @@ +/target/ diff --git a/cmd/racer-dataplane/alloc/Cargo.toml b/cmd/racer-dataplane/alloc/Cargo.toml new file mode 100644 index 000000000..c49074d93 --- /dev/null +++ b/cmd/racer-dataplane/alloc/Cargo.toml @@ -0,0 +1,16 @@ +[package] +name = "page-alloc" +version = "0.1.0" +edition = "2024" +license = "Apache-2.0" +publish = false +description = "Worker-local aligned buffers and lease-fenced sparse slab storage" + +[features] +default = [] +simulation = ["uring-runtime/simulation"] + +[dependencies] +libc = "0.2" +zeroize = "1" +uring-runtime = { path = "../runtime" } diff --git a/cmd/racer-dataplane/alloc/src/lib.rs b/cmd/racer-dataplane/alloc/src/lib.rs new file mode 100644 index 000000000..86b465c6d --- /dev/null +++ b/cmd/racer-dataplane/alloc/src/lib.rs @@ -0,0 +1,1196 @@ +//! Worker-local aligned storage with caller-owned accounting and I/O policy. +//! +//! This internal crate owns generic bytes, not keys, records, encryption, integrity +//! checks, versions, admission classes, or a persistence protocol. The caller +//! supplies accounting guards, a reactor, cancellation policy, and index callbacks. +//! There is no assumed OS page size and no cross-thread allocation authority. +//! +//! # Startup and binding +//! +//! Construct [`Slab::new`] with a full file path and [`Segments::new`] with the +//! same segment size, then call [`Slab::open_configured`] outside the worker's +//! latency-sensitive path. Directory traversal, locking, sizing, and probing are +//! blocking. Alignment comes from `statx(STATX_DIOALIGN)` rather than a fixed page +//! size. A configured partial table must match capacity, segment size, and +//! alignment. The slab reports physical geometry even with a partial table. +//! +//! Opening and binding are separate steps: failed binding can leave the file open +//! and locked but cannot authorize I/O. Fix the table and retry or drop the slab. +//! Rebinding to a different table is rejected even when its geometry matches. +//! The maximum record size only checks that its padded size fits a segment at +//! startup; it is not an allocation quota or a per-submission record-size limit. +//! +//! Linux traversal rejects parent components and intermediate/final symlinks. +//! Missing directories and files are created with modes 0700 and 0600, subject to +//! umask. Accepted files must be regular, owned by the effective user, singly +//! linked, and exactly mode 0600 with no special bits. Permissions are not repaired. +//! Nonblocking open prevents FIFOs from hanging startup before type validation. +//! Files are close-on-exec, exclusively locked with nonblocking flock, and use +//! direct I/O. Parent directories still must be trusted against hostile rename +//! and unlink; neither traversal nor flock stops noncooperating writers. +//! +//! [`Slab::from_devices`] instead accepts one [`DevicePlacement`] per logical +//! segment. The caller opens devices with O_EXCL and O_DIRECT and supplies common +//! alignment. Bounds use BLKGETSIZE64 for block devices and length for regular +//! files. Placements can share an `Arc` to use one +//! runtime descriptor per device. Startup never creates, resizes, or locks these +//! files. Logical extents still use segment-table offsets; submissions translate +//! them to the placement's physical range after checking lease bounds. Overlap +//! checks only compare the same inode or device identity within one slab. They +//! cannot detect whole-disk, partition, or device-mapper aliases. The caller must +//! guarantee disjoint physical storage across aliases and slabs, keep exclusive +//! ownership, and not change file flags or sizes while in use. +//! +//! In file mode, empty files are sparsely extended to capacity; nonempty size mismatches are +//! rejected without truncation. Capacity is a logical bound, not reserved disk +//! space, so later writes can fail with ENOSPC. Recycling changes metadata only: +//! it does not erase, truncate, or hole-punch disk bytes. Buffer zeroization is +//! not secure erasure of the file, and physical blocks can survive eviction. +//! +//! # Alignment and accounting +//! +//! [`Alignment`] permits arbitrary positive offset and length units; memory +//! alignment must be a supported power of two. Extent padding uses their least +//! common multiple, so offset unit 768 and length unit 512 require multiples of +//! 1536. Arithmetic and the conservative 1 GiB transfer cap are checked before +//! allocating. Larger logical records must be split by the caller. An [`Extent`] +//! alone checks a range, not its alignment or permission to access a segment. +//! +//! [`AlignedBuffer`] owns initialized stable storage that cannot be resized. +//! The primary [`Charge`] must truthfully account for its entire padded size; +//! `()` opts out of accounting. Additional guards attached with +//! [`AlignedBuffer::retain`] remain live through completion, but not idle pooling. +//! [`Slab::allocate`] maintains one exact-size idle slot. Drop zeroizes bytes +//! before pooling or freeing. Idle memory retains the primary charge. Reuse +//! replaces it with the newly admitted charge; undercharging leaves the idle +//! slot untouched. A size mismatch releases old storage even if replacement +//! fails. Occupied, borrowed, or dropped pools cannot retain returned storage. +//! Outstanding buffers do not keep the pool alive. Reclaim reaches idle bytes +//! only, never memory still owned by the kernel. +//! +//! # Completion ownership +//! +//! For writes, round the logical size, append the padded length, allocate matching +//! storage, copy bytes into the zeroed buffer, and submit [`Slab::write`]. Publish +//! the caller's mapping only after the write succeeds. Append reserves space and +//! does not roll it back on failed writes. For reads, validate the stored slot, +//! generation, and extent, acquire a lease, allocate storage, and submit a read. +//! A lease authorizes the segment's used prefix at acquisition, not just the last +//! appended record; it cannot authorize bytes appended afterward. +//! Any lease, including one from [`Segments::lease`], permits writes in that prefix. +//! Leases do not grant exclusive record ownership. The trusted caller must write +//! only reserved extents it owns and never overwrite published or readable records. +//! A lease does not prove initialization in its generation: append reserves space +//! without writing it. Recycled bytes may come from an earlier generation or a +//! different cache. Before exposing them as a valid record, the caller must check +//! record integrity, authentication, and cache identity, even after a successful read. +//! +//! Submission checks table identity, alignment, length, segment boundaries, and +//! captured used bytes. Reads and writes reject short completions. Accepted I/O +//! owns the buffer, all charges, and the lease through the runtime completion +//! fence, even when its waiting future is dropped. Writes also own their counter +//! guard. An unpolled future has submitted nothing. Continue driving the reactor +//! to complete or cancel accepted work before expecting resources to be released. +//! [`Slab::fence_writes`] neither waits for reads nor prevents new writes. Stop +//! admitting writes first if quiescence is required. It is a completion fence, +//! not a durability barrier: it never calls fsync or fdatasync. +//! +//! # Recovery and reclamation +//! +//! Slots normally progress from Free to Open to Sealed to Evicting to Free. +//! Append selects a fitting open tail or the lowest free slot. Full segments and +//! rotated tails become sealed. A valid rollover with no free slot seals the old +//! tail before returning Busy, enabling reclamation and retry. Malformed requests +//! and lease-counter overflow leave the table unchanged. New leases reject +//! eviction, but existing leases remain usable. Recycling requires no remaining +//! leases and increments generation without wrapping. +//! +//! A [`FreezeGuard`] blocks table mutation, not reads, snapshots, new read leases, +//! caller index mutation, or accepted I/O. It may outlive its table. Only one guard +//! exists at a time. Restore requires both thawing and draining all leases, +//! including reads and abandoned I/O. It validates the whole ordered image before +//! publication, seals an open tail, rebuilds free slots, and advances an epoch +//! that clears the eviction cursor and recent-read state. Invalid or busy restores +//! never partially publish. +//! Raw file opening, table configuration, and manual eviction are simulation-only +//! escape hatches. Production startup binds with [`Slab::open_configured`], and +//! [`SegmentClock`] owns the ordering of mapping removal before physical reuse. +//! The table, slab, and clock are cache-line aligned to isolate worker-local state; +//! their Rc ownership prevents moving authority to a different worker thread. +//! +//! For a checkpoint, stop mutation admission, retain a freeze guard, drive writes +//! through their completion fence, then capture compatible allocator and caller +//! metadata. Keep admission coordinated around thaw and restore. Neither the +//! image nor the completion fence proves durability or record validity. Recovery +//! must validate identity, lengths, integrity, and versions. Invalid, stale, torn, +//! or missing cache records must become misses and be refetched. +//! +//! [`SegmentClock`] shares a cursor and second-chance set across bounded sweeps. +//! Index reclamation forgets mappings without touching bytes or generations. +//! Physical reclamation selects only Sealed/Evicting slots and recycles empty slots +//! after leases drain. Index and unscored sweeps visit at most two rotations, +//! further capped by the caller. Scored reclamation visits at most +//! min(slot count, max_visits, 64) slots, including skipped slots, then ranks +//! eligible candidates. For a nonzero reserve, [`SegmentClock::reclaim`] and +//! [`SegmentClock::reclaim_scored`] succeed only when the reserve (capped at the +//! slot count) is met and no evictions remain. Busy is intentional even with enough +//! free slots while pending evictions drain. Use [`Segments::free_count`] to check +//! capacity, and retry bounded reclamation later to finish draining rather than +//! spinning. A zero reserve is a no-op, not a drain request; a zero entry budget +//! can recycle empty slots but cannot begin populated eviction. There is no +//! compaction. [`SegmentEntries`] implementations +//! must compare current mappings before removal and report within budget; an +//! error cannot undo callback side effects. Freeze does not block index sweeps. +//! +//! # Validation +//! +//! Run `cargo test -p page-alloc --no-default-features` and +//! `cargo test -p page-alloc --all-features` under an external TERM timeout. +//! Simulation provides deterministic fault and cancellation ordering, not proof +//! of filesystem security or kernel support. Descriptor-replacement test hooks +//! require an open slab and no live writes and do not revalidate geometry. +//! Real tests print explicit capability skips with `-- --nocapture`; set +//! `PAGE_ALLOC_REQUIRE_REAL_IO=1` to turn those skips into failures. A skip is not +//! evidence that the real path passed. Miri can cover the pure `buffer_tests`, +//! `geometry_tests`, and `segments::tests` filters, but not real kernel I/O. +#![deny(unsafe_op_in_unsafe_fn)] +#![deny(missing_docs)] + +mod segments; + +mod slab; + +pub use segments::{ + FreezeGuard, Generation, SegmentClock, SegmentEntries, SegmentId, SegmentLease, + SegmentSnapshot, SegmentState, Segments, +}; +pub use slab::{DevicePlacement, Slab}; +use std::{ + alloc::{Layout, alloc_zeroed, dealloc}, + cell::RefCell, + fmt, + ptr::NonNull, + rc::{Rc, Weak}, +}; +use uring_runtime::reactor::IoBuffer; + +/// Storage failures, separated from application record and admission policy. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +#[non_exhaustive] +pub enum Error { + /// The filesystem or kernel cannot provide required direct-I/O support. + Unsupported, + + /// Caller configuration or an implementation contract is invalid. + InvalidConfiguration, + + /// A transient lease, freeze, or resource limit prevents progress. + Busy, + + /// An extent or persisted allocator image is malformed. + Corrupt, + + /// A generation, segment state, or table identity is no longer valid. + Stale, + + /// Storage is unopened, locked elsewhere, or has exhausted a generation. + Unavailable, + + /// An I/O completion was short or failed without OS error detail. + Io, + + /// Synchronous OS failure; asynchronous errors retain the runtime's error type. + SystemIo { + /// Operation that failed. + operation: &'static str, + + /// Operating-system error number when available. + errno: Option, + }, +} + +impl fmt::Display for Error { + /// Describe the category, preserving operating-system diagnostics when present. + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + if let Self::SystemIo { operation, errno } = self { + return match errno { + Some(errno) => write!( + f, + "{operation}: {} (errno {errno})", + std::io::Error::from_raw_os_error(*errno) + ), + None => write!(f, "{operation}: storage I/O failed"), + }; + } + f.write_str(match self { + Self::Unsupported => "direct I/O is unsupported", + Self::InvalidConfiguration => "invalid storage configuration", + Self::Busy => "storage is busy or exhausted", + Self::Corrupt => "invalid storage extent or generation", + Self::Stale => "stale storage generation, state, or table", + Self::Unavailable => "storage is unavailable", + Self::Io => "storage I/O failed", + Self::SystemIo { .. } => unreachable!(), + }) + } +} + +impl std::error::Error for Error {} + +/// Result of an allocator operation before conversion into a caller's scope error. +pub type Result = std::result::Result; + +/// Maximum number of slots retained by a segment allocation table. +pub const MAX_SEGMENTS: u64 = 1_000_000; + +/// Validated physical dimensions, independent of record formats and slot quotas. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub struct SegmentGeometry { + slab_bytes: u64, + + segment_bytes: u64, + + segment_count: u64, + + alignment: Alignment, +} + +impl SegmentGeometry { + /// Validate physical geometry; retained tables separately enforce MAX_SEGMENTS. + pub fn new( + slab_bytes: u64, + segment_bytes: u64, + segment_count: u64, + alignment: Alignment, + ) -> Result { + if segment_bytes == 0 + || slab_bytes == 0 + || segment_count == 0 + || !slab_bytes.is_multiple_of(segment_bytes) + || segment_count > slab_bytes / segment_bytes + || !segment_bytes.is_multiple_of(alignment.offset()) + || !segment_bytes.is_multiple_of(alignment.length() as u64) + || segment_count.checked_mul(segment_bytes).is_none() + { + return Err(Error::Corrupt); + } + Ok(Self { + slab_bytes, + segment_bytes, + segment_count, + alignment, + }) + } + + /// Full logical size of the physical file. + pub fn slab_bytes(self) -> u64 { + self.slab_bytes + } + + /// Fixed physical size of each segment. + pub fn segment_bytes(self) -> u64 { + self.segment_bytes + } + + /// Number of exposed segments, possibly less than the physical file allows. + pub fn segment_count(self) -> u64 { + self.segment_count + } + + /// Direct-I/O requirements validated with these dimensions. + pub fn alignment(self) -> Alignment { + self.alignment + } + + /// Compare table dimensions only, not occupancy or alignment compatibility. + #[cfg(test)] + fn matches_segments(&self, segments: &Segments) -> bool { + segments.capacity_bytes() == self.slab_bytes + && segments.segment_bytes() == self.segment_bytes + && segments.count() as u64 == self.segment_count + } +} + +/// Caller-owned accounting retained while memory is live, including idle pooling. +pub trait Charge: 'static { + /// Whether this guard accounts for at least `bytes` live allocation bytes. + fn covers(&self, bytes: usize) -> bool; +} + +impl Charge for () { + /// Explicitly opt out of accounting for callers without admission policy. + fn covers(&self, _bytes: usize) -> bool { + true + } +} + +/// Direct-I/O alignment requirements. Offset and length units may be any positive +/// integers; only the memory alignment must be a power of two. +#[must_use] +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub struct Alignment { + memory: usize, + + offset: u64, + + length: usize, +} + +/// A nonempty, non-overflowing file range for one bounded I/O transfer. +#[must_use] +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub struct Extent { + offset: u64, + + length: usize, +} + +impl Extent { + /// Reject empty ranges, offset overflow, and lengths above the transfer cap. + pub fn new(offset: u64, length: usize) -> Result { + if length == 0 + || length > Alignment::MAX_TRANSFER_LENGTH + || offset.checked_add(length as u64).is_none() + { + return Err(Error::Corrupt); + } + Ok(Self { offset, length }) + } + + /// Starting byte offset in the file. + #[must_use] + pub fn offset(self) -> u64 { + self.offset + } + + /// Transfer length in bytes, including any padding. + #[must_use] + pub fn length(self) -> usize { + self.length + } +} + +impl Alignment { + /// Conservative single-transfer limit of 1 GiB. + /// + /// This fits the runtime's `u32` length and stays below Linux's + /// `MAX_RW_COUNT` (`i32::MAX` rounded down to a base-page boundary) on + /// supported Linux base-page sizes, without querying host state. Keeping a + /// fixed conservative bound also makes geometry checks deterministic under + /// simulation and Miri. Larger records must be split by the caller. + pub const MAX_TRANSFER_LENGTH: usize = 1 << 30; + + /// Validate alignment units without requiring offset/length powers of two. + pub fn new(memory: usize, offset: u64, length: usize) -> Result { + if !memory.is_power_of_two() || offset == 0 || length == 0 || memory > isize::MAX as usize { + return Err(Error::Unsupported); + } + Ok(Self { + memory, + offset, + length, + }) + } + + /// Required memory address alignment. + #[must_use] + pub fn memory(self) -> usize { + self.memory + } + + /// Required file offset unit. + #[must_use] + pub fn offset(self) -> u64 { + self.offset + } + + /// Required transfer length unit. + #[must_use] + pub fn length(self) -> usize { + self.length + } + + /// Round up to the least common multiple of offset and length units so the + /// next appended extent is also offset-aligned. Reject overflow and lengths + /// above [`Self::MAX_TRANSFER_LENGTH`] before allocating memory. + pub fn extent(&self, offset: u64, logical: usize) -> Result { + if !offset.is_multiple_of(self.offset) + || logical == 0 + || logical > Self::MAX_TRANSFER_LENGTH + { + return Err(Error::InvalidConfiguration); + } + let offset_unit = usize::try_from(self.offset).map_err(|_| Error::InvalidConfiguration)?; + let (mut a, mut b) = (offset_unit, self.length); + while b != 0 { + (a, b) = (b, a % b); + } + let unit = (offset_unit / a) + .checked_mul(self.length) + .ok_or(Error::InvalidConfiguration)?; + let length = logical + .checked_add(unit - 1) + .and_then(|v| (v / unit).checked_mul(unit)) + .ok_or(Error::InvalidConfiguration)?; + if length > Self::MAX_TRANSFER_LENGTH { + return Err(Error::InvalidConfiguration); + } + Extent::new(offset, length) + } + + /// Allocate zeroed stable storage and retain its primary accounting guard. + /// Length must be nonzero, length-aligned, covered by `charge`, and no larger + /// than [`Self::MAX_TRANSFER_LENGTH`]. Allocation failure returns `Busy`. + pub fn allocate(&self, length: usize, charge: C) -> Result> { + if length == 0 + || length > Self::MAX_TRANSFER_LENGTH + || !length.is_multiple_of(self.length) + || !charge.covers(length) + { + return Err(Error::InvalidConfiguration); + } + let layout = Layout::from_size_align(length, self.memory) + .map_err(|_| Error::InvalidConfiguration)?; + // SAFETY: valid nonzero layout; Allocation owns the matching deallocation. + let pointer = NonNull::new(unsafe { alloc_zeroed(layout) }).ok_or(Error::Busy)?; + Ok(AlignedBuffer { + allocation: Some(Allocation { + pointer, + layout, + charge, + retained: Vec::new(), + clean: true, + #[cfg(test)] + wipes: Rc::new(std::cell::Cell::new(0)), + }), + pool: Weak::new(), + }) + } + + /// Validate the address, file offset, transfer unit, and exact buffer length. + pub fn check(&self, extent: Extent, buffer: &AlignedBuffer) -> Result<()> { + if !(buffer.allocation().pointer.as_ptr() as usize).is_multiple_of(self.memory) + || !extent.offset.is_multiple_of(self.offset) + || !extent.length.is_multiple_of(self.length) + || buffer.len() != extent.length + { + return Err(Error::InvalidConfiguration); + } + Ok(()) + } +} + +/// Owns raw storage and accounting together; moving it transfers both exactly once. +struct Allocation { + pointer: NonNull, + + layout: Layout, + + charge: C, + + retained: Vec>, + + // True only after zeroed allocation or a complete secure wipe. Every mutable + // exposure, including the runtime's kernel-write borrow, clears this first. + clean: bool, + + #[cfg(test)] + wipes: Rc>, +} + +impl Allocation { + /// Exclusively borrow the complete initialized allocation with its original size. + fn as_mut_slice(&mut self) -> &mut [u8] { + self.clean = false; + // SAFETY: exclusive owner, initialized nonzero allocation, original layout. + unsafe { std::slice::from_raw_parts_mut(self.pointer.as_ptr(), self.layout.size()) } + } + + /// Securely erase the entire allocation once after each mutable exposure. + fn wipe(&mut self) { + if self.clean { + return; + } + #[cfg(all( + target_os = "linux", + any(target_env = "gnu", target_env = "musl"), + not(miri) + ))] + // SAFETY: this exclusive owner holds layout.size() initialized writable + // bytes. explicit_bzero cannot be eliminated as a dead store. No kernel + // operation can outlive ownership's completion fence. + unsafe { + libc::explicit_bzero(self.pointer.as_ptr().cast(), self.layout.size()) + }; + #[cfg(not(all( + target_os = "linux", + any(target_env = "gnu", target_env = "musl"), + not(miri) + )))] + { + use zeroize::Zeroize; + self.as_mut_slice().zeroize(); + } + self.clean = true; + #[cfg(test)] + self.wipes.set(self.wipes.get() + 1); + } +} + +impl Drop for Allocation { + /// Erase and free before field destruction releases the accounting guards. + fn drop(&mut self) { + self.wipe(); + // SAFETY: this owner holds the allocation and its original layout. Guards + // are dropped only after zeroization and deallocation complete. + unsafe { dealloc(self.pointer.as_ptr(), self.layout) }; + } +} + +/// Worker-local, exclusively owned, initialized storage with a stable address. +/// +/// Moving the buffer never moves its bytes. Drop zeroizes the bytes before +/// freeing or pooling them. Idle storage retains its primary charge, but not +/// additional guards attached with [`Self::retain`]. +/// +/// A buffer cannot cross a worker boundary even when its accounting guard can: +/// +/// ```compile_fail +/// fn require_send() {} +/// require_send::>(); +/// ``` +#[must_use] +pub struct AlignedBuffer { + // Always Some while publicly accessible; taken only by Drop for pool transfer. + allocation: Option>, + + pool: Weak>>, +} + +impl fmt::Debug for AlignedBuffer { + /// Show storage properties without requiring accounting guards to expose data. + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.debug_struct("AlignedBuffer") + .field("length", &self.len()) + .field("alignment", &self.allocation().layout.align()) + .field("retained", &self.allocation().retained.len()) + .finish_non_exhaustive() + } +} + +impl AlignedBuffer { + /// Borrow the allocation, which is absent only during drop's ownership transfer. + fn allocation(&self) -> &Allocation { + self.allocation.as_ref().expect("live buffer allocation") + } + + /// Mutably borrow live backing storage before any drop-time pool transfer. + fn allocation_mut(&mut self) -> &mut Allocation { + self.allocation.as_mut().expect("live buffer allocation") + } + + /// Attach a weak return destination without extending the pool's lifetime. + pub(crate) fn pooled(mut self, pool: &Rc>>) -> Self { + self.pool = Rc::downgrade(pool); + self + } + + /// Replace primary accounting only after the new guard covers the full size. + pub(crate) fn rebind(&mut self, charge: C) -> Result<()> { + if !charge.covers(self.len()) { + return Err(Error::InvalidConfiguration); + } + self.allocation_mut().charge = charge; + Ok(()) + } + + /// Retain additional accounting through the final kernel completion. + pub fn retain(&mut self, charge: Rc) { + self.allocation_mut().retained.push(charge); + } + + /// Initialized allocation length, including padding. + #[must_use] + pub fn len(&self) -> usize { + self.allocation().layout.size() + } + + /// Always false: construction rejects zero-length buffers. + #[must_use] + pub fn is_empty(&self) -> bool { + false + } + + /// Borrow all initialized bytes without moving or resizing the allocation. + #[must_use] + pub fn as_slice(&self) -> &[u8] { + // SAFETY: initialized allocation remains live throughout this borrow. + unsafe { std::slice::from_raw_parts(self.allocation().pointer.as_ptr(), self.len()) } + } + + /// Exclusively borrow all initialized bytes. + #[must_use] + pub fn as_mut_slice(&mut self) -> &mut [u8] { + self.allocation_mut().as_mut_slice() + } + + /// Compatibility accessor matching [`IoBuffer`]; this always succeeds. + pub fn bytes(&self) -> Result<&[u8]> { + Ok(self.as_slice()) + } + + /// Compatibility accessor matching [`IoBuffer`]; this always succeeds. + pub fn bytes_mut(&mut self) -> Result<&mut [u8]> { + Ok(self.as_mut_slice()) + } +} + +impl Drop for AlignedBuffer { + /// Zeroize and return storage if possible, otherwise let its owner free it. + fn drop(&mut self) { + if let Some(pool) = self.pool.upgrade() { + self.allocation_mut().wipe(); + // Caller-owned guard destructors may access the pool. Do not invoke + // them while holding its RefCell borrow. Allocation remains owned + // by this buffer if a destructor unwinds. + self.allocation_mut().retained.clear(); + if let Ok(mut idle) = pool.try_borrow_mut() + && idle.is_none() + { + *idle = Some(Self { + allocation: self.allocation.take(), + pool: Weak::new(), + }); + } + } + // Field drop securely wipes any still-dirty allocation and frees it. A + // rejected pool return or idle free needs no second wipe. Idle buffers + // never repool themselves. + } +} + +// SAFETY: owned aligned backing is initialized, stable, and live until Drop. +unsafe impl IoBuffer for AlignedBuffer { + type Error = Error; + + /// Expose initialized stable bytes to the runtime. + fn bytes(&self) -> Result<&[u8]> { + AlignedBuffer::bytes(self) + } + + /// Give the runtime exclusive access without moving the allocation. + fn bytes_mut(&mut self) -> Result<&mut [u8]> { + AlignedBuffer::bytes_mut(self) + } +} + +/// Pure allocation tests, including unwind and reentrant accounting destructors. +#[cfg(test)] +mod buffer_tests { + use super::*; + use std::cell::Cell; + + /// Tracks primary and retained bytes without exposing a Debug implementation. + struct TrackedCharge { + live: Rc>, + + bytes: usize, + } + + impl TrackedCharge { + /// Admit and record a fixed number of live bytes. + fn new(live: &Rc>, bytes: usize) -> Self { + live.set(live.get() + bytes); + Self { + live: live.clone(), + bytes, + } + } + } + + impl Charge for TrackedCharge { + /// Cover only the number of bytes admitted by this guard. + fn covers(&self, bytes: usize) -> bool { + self.bytes >= bytes + } + } + + impl Drop for TrackedCharge { + /// Return admitted bytes exactly once. + fn drop(&mut self) { + self.live.set(self.live.get() - self.bytes); + } + } + + /// Full padded capacity is erased, and clean reuse does not erase twice. + #[test] + fn secure_wipe_covers_padding_and_skips_clean_pool_lifetimes() { + let alignment = Alignment::new(4096, 512, 512).unwrap(); + let length = alignment.extent(0, 513).unwrap().length(); + let pool = Rc::new(RefCell::new(None)); + let mut buffer = alignment.allocate(length, ()).unwrap().pooled(&pool); + let wipes = buffer.allocation().wipes.clone(); + assert!(buffer.allocation().clean); + assert_eq!(buffer.as_slice(), vec![0; 1024]); + drop(buffer); + assert_eq!(wipes.get(), 0); + buffer = pool.borrow_mut().take().unwrap().pooled(&pool); + buffer.as_mut_slice().fill(0xa5); + assert!(!buffer.allocation().clean); + drop(buffer); + assert_eq!(wipes.get(), 1); + buffer = pool.borrow_mut().take().unwrap().pooled(&pool); + assert_eq!(buffer.as_slice(), vec![0; 1024]); + assert!(buffer.allocation().clean); + drop(buffer); + drop(pool); + assert_eq!(wipes.get(), 1); + } + + /// Runtime pointer writes dirty storage before submission, including padding. + #[test] + fn kernel_write_borrow_dirties_clean_reused_storage() { + let pool = Rc::new(RefCell::new(None)); + let alignment = Alignment::new(64, 1, 1).unwrap(); + let mut buffer = alignment.allocate(128, ()).unwrap().pooled(&pool); + let wipes = buffer.allocation().wipes.clone(); + for expected in 1..=2 { + assert!(buffer.allocation().clean); + let pointer = IoBuffer::bytes_mut(&mut buffer).unwrap().as_mut_ptr(); + // SAFETY: simulate a kernel completion while the exclusive owner is + // retained, before any subsequent access or release of the buffer. + unsafe { pointer.add(127).write(0x5a) }; + assert!(!buffer.allocation().clean); + assert_eq!(buffer.as_slice()[127], 0x5a); + drop(buffer); + assert_eq!(wipes.get(), expected); + buffer = pool.borrow_mut().take().unwrap().pooled(&pool); + assert_eq!(IoBuffer::bytes(&buffer).unwrap(), &[0; 128]); + } + // Even an unused mutable borrow must conservatively require a wipe. + let _ = buffer.bytes_mut().unwrap(); + drop(buffer); + assert_eq!(wipes.get(), 3); + drop(pool); + assert_eq!(wipes.get(), 3); + } + + /// Invalid transfers must fail before inspecting accounting or allocating. + #[test] + fn transfer_limit_is_enforced_before_allocation_or_charge_inspection() { + /// Detects any accounting inspection on a structurally invalid request. + struct UncheckedCharge; + + impl Charge for UncheckedCharge { + /// Panic when validation reaches accounting in the wrong order. + fn covers(&self, _: usize) -> bool { + panic!("invalid lengths must be rejected before inspecting accounting"); + } + } + let alignment = Alignment::new(1, 1, 1).unwrap(); + let max = Alignment::MAX_TRANSFER_LENGTH; + assert_eq!(Extent::new(0, max).unwrap().length(), max); + assert_eq!(alignment.extent(0, max).unwrap().length(), max); + for length in [0, max + 1, i32::MAX as usize, u32::MAX as usize, usize::MAX] { + assert_eq!(Extent::new(0, length), Err(Error::Corrupt)); + assert_eq!( + alignment.extent(0, length), + Err(Error::InvalidConfiguration) + ); + assert!(matches!( + alignment.allocate(length, UncheckedCharge), + Err(Error::InvalidConfiguration) + )); + } + let alignment = Alignment::new(8, 3, 5).unwrap(); + let rounded_limit = max / 15 * 15; + assert_eq!( + alignment.extent(0, rounded_limit).unwrap().length(), + rounded_limit + ); + assert_eq!( + alignment.extent(0, rounded_limit + 1), + Err(Error::InvalidConfiguration) + ); + assert_eq!(Extent::new(u64::MAX, 1), Err(Error::Corrupt)); + assert_eq!(alignment.extent(u64::MAX, 1), Err(Error::Corrupt)); + assert_eq!(Extent::new(u64::MAX - 1, 1).unwrap().offset(), u64::MAX - 1); + } + + /// Non-power-of-two units use the LCM and reject overflow before allocation. + #[test] + fn arbitrary_units_preserve_lcm_rounding_and_detect_overflow() { + for (offset_unit, length_unit, lcm) in [(3, 5, 15), (6, 9, 18), (768, 512, 1536)] { + let alignment = Alignment::new(64, offset_unit, length_unit).unwrap(); + assert_eq!(alignment.memory(), 64); + assert_eq!(alignment.offset(), offset_unit); + assert_eq!(alignment.length(), length_unit); + for (logical, expected) in [(1, lcm), (lcm, lcm), (lcm + 1, lcm * 2)] { + let extent = alignment.extent(offset_unit, logical).unwrap(); + assert_eq!(extent.offset(), offset_unit); + assert_eq!(extent.length(), expected); + let buffer = alignment.allocate(expected, ()).unwrap(); + alignment.check(extent, &buffer).unwrap(); + } + assert_eq!(alignment.extent(1, 1), Err(Error::InvalidConfiguration)); + } + for alignment in [ + Alignment::new(1, u64::MAX, usize::MAX - 1).unwrap(), + Alignment::new(1, 1, usize::MAX).unwrap(), + Alignment::new(1, 2, usize::MAX).unwrap(), + ] { + assert_eq!(alignment.extent(0, 2), Err(Error::InvalidConfiguration)); + } + for (memory, offset, length) in [(0, 1, 1), (3, 1, 1), (1, 0, 1), (1, 1, 0)] { + assert_eq!( + Alignment::new(memory, offset, length), + Err(Error::Unsupported) + ); + } + assert_eq!( + Alignment::new(1usize << (usize::BITS - 1), 1, 1), + Err(Error::Unsupported) + ); + } + + /// Moving a buffer preserves its address and initialized runtime-visible bytes. + #[test] + fn initialized_storage_remains_stable_across_moves_and_trait_access() { + let alignment = Alignment::new(64, 3, 5).unwrap(); + let mut buffer = alignment.allocate(15, ()).unwrap(); + assert!(!buffer.is_empty()); + assert_eq!(buffer.len(), 15); + assert_eq!(buffer.as_slice(), &[0; 15]); + let pointer = IoBuffer::bytes_mut(&mut buffer).unwrap().as_mut_ptr(); + assert!((pointer as usize).is_multiple_of(64)); + let moved = std::hint::black_box(Some(buffer)); + // SAFETY: moving the owner preserves the allocation; no intervening reborrow. + unsafe { pointer.write(42) }; + let mut buffer = moved.unwrap(); + assert_eq!(IoBuffer::bytes(&buffer).unwrap()[0], 42); + buffer.bytes_mut().unwrap()[1] = 17; + assert_eq!(buffer.bytes().unwrap()[1], 17); + assert_eq!(buffer.as_slice().as_ptr(), pointer); + alignment + .check(Extent::new(3, 15).unwrap(), &buffer) + .unwrap(); + for extent in [Extent::new(1, 15).unwrap(), Extent::new(3, 10).unwrap()] { + assert_eq!( + alignment.check(extent, &buffer), + Err(Error::InvalidConfiguration) + ); + } + assert_eq!( + Alignment::new(64, 3, 2) + .unwrap() + .check(Extent::new(3, 15).unwrap(), &buffer), + Err(Error::InvalidConfiguration) + ); + } + + /// Admission replacement is atomic and Debug does not expose guard internals. + #[test] + fn charge_validation_rebinding_and_debug_do_not_require_charge_debug() { + let live = Rc::new(Cell::new(0)); + let alignment = Alignment::new(8, 3, 5).unwrap(); + for (length, charge) in [ + (15, 14), + (14, 15), + (0, 15), + (Alignment::MAX_TRANSFER_LENGTH + 1, 15), + ] { + assert_eq!( + alignment + .allocate(length, TrackedCharge::new(&live, charge)) + .unwrap_err(), + Error::InvalidConfiguration + ); + assert_eq!(live.get(), 0); + } + let mut buffer = alignment + .allocate(15, TrackedCharge::new(&live, 15)) + .unwrap(); + assert_eq!( + buffer.rebind(TrackedCharge::new(&live, 14)), + Err(Error::InvalidConfiguration) + ); + assert_eq!(live.get(), 15); + buffer.rebind(TrackedCharge::new(&live, 20)).unwrap(); + assert_eq!(live.get(), 20); + let debug = format!("{buffer:?}"); + assert!(debug.contains("length: 15")); + assert!(debug.contains("alignment: 8")); + assert!(!debug.contains("TrackedCharge")); + drop(buffer); + assert_eq!(live.get(), 0); + } + + /// Pooling retains primary accounting while releasing completion-only guards. + #[test] + fn pool_transfers_allocation_and_primary_charge_but_not_retained_guards() { + let live = Rc::new(Cell::new(0)); + let pool = Rc::new(RefCell::new(None)); + let alignment = Alignment::new(64, 3, 5).unwrap(); + let mut buffer = alignment + .allocate(15, TrackedCharge::new(&live, 15)) + .unwrap() + .pooled(&pool); + let pointer = buffer.as_slice().as_ptr(); + buffer.as_mut_slice().fill(42); + let extra = Rc::new(TrackedCharge::new(&live, 7)); + let weak = Rc::downgrade(&extra); + buffer.retain(extra); + assert_eq!(live.get(), 22); + drop(buffer); + assert_eq!(live.get(), 15); + assert!(weak.upgrade().is_none()); + let mut reused = pool.borrow_mut().take().unwrap(); + assert_eq!(reused.as_slice().as_ptr(), pointer); + assert_eq!(reused.as_slice(), &[0; 15]); + reused.rebind(TrackedCharge::new(&live, 20)).unwrap(); + assert_eq!(live.get(), 20); + drop(reused.pooled(&pool)); + drop(pool); + assert_eq!(live.get(), 0); + } + + /// An unusable return slot frees storage exactly once instead of panicking. + #[test] + fn unavailable_borrowed_and_occupied_pools_release_exactly_once() { + let alignment = Alignment::new(8, 1, 1).unwrap(); + for scenario in 0..4 { + let live = Rc::new(Cell::new(0)); + let pool = Rc::new(RefCell::new(None)); + let mut buffer = alignment + .allocate(8, TrackedCharge::new(&live, 8)) + .unwrap() + .pooled(&pool); + buffer.as_mut_slice().fill(0x5a); + let wipes = buffer.allocation().wipes.clone(); + buffer.retain(Rc::new(TrackedCharge::new(&live, 3))); + match scenario { + 0 => { + drop(pool); + drop(buffer); + } + 1 => { + let borrow = pool.borrow(); + drop(buffer); + assert_eq!(live.get(), 0); + assert!(borrow.is_none()); + } + 2 => { + let borrow = pool.borrow_mut(); + drop(buffer); + assert_eq!(live.get(), 0); + assert!(borrow.is_none()); + } + _ => { + let idle = alignment.allocate(8, TrackedCharge::new(&live, 8)).unwrap(); + let pointer = idle.as_slice().as_ptr(); + *pool.borrow_mut() = Some(idle); + drop(buffer); + assert_eq!(live.get(), 8); + assert_eq!(pool.borrow().as_ref().unwrap().as_slice().as_ptr(), pointer); + drop(pool); + } + } + assert_eq!(live.get(), 0); + assert_eq!(wipes.get(), 1); + } + } + + /// Caller destructors run before the return slot is borrowed. + #[test] + fn retained_guard_can_inspect_pool_before_buffer_is_returned() { + /// Runs a caller-provided destructor to test reentrant pool inspection. + struct Guard(Option>); + + impl Charge for Guard { + /// Admit all sizes for this destructor-order test. + fn covers(&self, _: usize) -> bool { + true + } + } + + impl Drop for Guard { + /// Invoke the callback at most once. + fn drop(&mut self) { + if let Some(callback) = self.0.take() { + callback(); + } + } + } + let pool = Rc::new(RefCell::new(None)); + let called = Rc::new(Cell::new(false)); + let mut buffer = Alignment::new(8, 1, 1) + .unwrap() + .allocate(8, Guard(None)) + .unwrap() + .pooled(&pool); + let observed_pool = pool.clone(); + let observed_called = called.clone(); + buffer.as_mut_slice().fill(0x5a); + let wipes = buffer.allocation().wipes.clone(); + let observed_wipes = wipes.clone(); + buffer.retain(Rc::new(Guard(Some(Box::new(move || { + assert!(observed_pool.borrow_mut().is_none()); + assert_eq!(observed_wipes.get(), 1); + observed_called.set(true); + }))))); + drop(buffer); + assert!(called.get()); + assert!(pool.borrow().is_some()); + drop(pool); + assert_eq!(wipes.get(), 1); + } + + /// Unwinding through extra accounting cannot leak primary allocation ownership. + #[test] + fn retained_guard_unwind_still_drops_owned_allocation_and_primary_charge() { + /// Counts destruction and optionally injects a panic. + struct Guard { + live: Rc>, + + panic: bool, + } + + impl Charge for Guard { + /// Admit all sizes for the unwind test. + fn covers(&self, _: usize) -> bool { + true + } + } + + impl Drop for Guard { + /// Record release before injecting the requested destructor panic. + fn drop(&mut self) { + self.live.set(self.live.get() - 1); + assert!(!self.panic, "injected retained guard panic"); + } + } + let live = Rc::new(Cell::new(2)); + let pool = Rc::new(RefCell::new(None)); + let mut buffer = Alignment::new(8, 1, 1) + .unwrap() + .allocate( + 8, + Guard { + live: live.clone(), + panic: false, + }, + ) + .unwrap() + .pooled(&pool); + buffer.as_mut_slice().fill(0x5a); + let wipes = buffer.allocation().wipes.clone(); + buffer.retain(Rc::new(Guard { + live: live.clone(), + panic: true, + })); + assert!(std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| drop(buffer))).is_err()); + assert_eq!(live.get(), 0); + assert!(pool.borrow().is_none()); + assert_eq!(wipes.get(), 1); + } +} + +/// Pure geometry boundary tests independent of file I/O and checkpoint formats. +#[cfg(test)] +mod geometry_tests { + use super::*; + + /// Standard test units, not a host-page-size assumption. + fn alignment() -> Alignment { + Alignment::new(512, 512, 512).unwrap() + } + + /// Physical geometry does not impose the retained table's slot bound. + #[test] + fn geometry_accepts_partial_tables_without_checkpoint_item_policy() { + let geometry = SegmentGeometry::new(4096, 1024, 2, alignment()).unwrap(); + assert_eq!(geometry.slab_bytes(), 4096); + assert_eq!(geometry.segment_bytes(), 1024); + assert_eq!(geometry.segment_count(), 2); + assert_eq!(geometry.alignment(), alignment()); + assert!(SegmentGeometry::new(1024 * MAX_SEGMENTS, 1024, MAX_SEGMENTS, alignment()).is_ok()); + assert!( + SegmentGeometry::new( + 1024 * (MAX_SEGMENTS + 1), + 1024, + MAX_SEGMENTS + 1, + alignment() + ) + .is_ok() + ); + assert!( + SegmentGeometry::new(u64::MAX, 1, u64::MAX, Alignment::new(1, 1, 1).unwrap()).is_ok() + ); + } + + /// Reject zero units, inconsistent capacity, misalignment, and overflow. + #[test] + fn geometry_rejects_zero_misalignment_capacity_and_overflow() { + for (slab, segment, count) in [ + (0, 512, 1), + (512, 0, 1), + (512, 512, 0), + (513, 512, 1), + (512, 512, 2), + (514, 257, 2), + (u64::MAX, u64::MAX, 2), + ] { + assert_eq!( + SegmentGeometry::new(slab, segment, count, alignment()), + Err(Error::Corrupt) + ); + } + assert_eq!( + SegmentGeometry::new(1024, 1024, 1, Alignment::new(512, 512, 768).unwrap()), + Err(Error::Corrupt) + ); + } + + /// Table matching ignores occupancy but compares each configured dimension. + #[test] + fn matching_live_table_ignores_occupancy_but_requires_dimensions() { + let geometry = SegmentGeometry::new(4096, 1024, 2, alignment()).unwrap(); + let segments = Segments::new(1024); + assert!(!geometry.matches_segments(&segments)); + segments.configure(4096, 2, alignment()).unwrap(); + assert!(geometry.matches_segments(&segments)); + let _held = segments.append(512).unwrap(); + assert!(geometry.matches_segments(&segments)); + assert!( + !SegmentGeometry::new(4096, 1024, 3, alignment()) + .unwrap() + .matches_segments(&segments) + ); + assert!( + !SegmentGeometry::new(2048, 1024, 2, alignment()) + .unwrap() + .matches_segments(&segments) + ); + assert!( + !SegmentGeometry::new(4096, 512, 2, alignment()) + .unwrap() + .matches_segments(&segments) + ); + } + + /// Segment boundaries must satisfy both direct-I/O units simultaneously. + #[test] + fn divisibility_by_each_unit_is_equivalent_to_lcm_divisibility() { + let alignment = Alignment::new(512, 512, 768).unwrap(); + assert!(SegmentGeometry::new(3072, 1536, 2, alignment).is_ok()); + assert_eq!( + SegmentGeometry::new(2048, 1024, 2, alignment), + Err(Error::Corrupt) + ); + assert_eq!( + SegmentGeometry::new(1536, 768, 2, alignment), + Err(Error::Corrupt) + ); + assert!(SegmentGeometry::new(u64::MAX, 1, 1, Alignment::new(1, 1, 1).unwrap()).is_ok()); + } +} diff --git a/cmd/racer-dataplane/alloc/src/segments.rs b/cmd/racer-dataplane/alloc/src/segments.rs new file mode 100644 index 000000000..6a071d97d --- /dev/null +++ b/cmd/racer-dataplane/alloc/src/segments.rs @@ -0,0 +1,1886 @@ +//! Worker-local allocation authority, recovery images, and bounded reclamation. +//! +//! Slot reuse is metadata-only: it never erases, truncates, or compacts disk bytes. +//! Existing leases protect their captured used prefix even after eviction starts. +//! Freeze protects table mutation, not caller indexes or data-I/O completion. +use crate::{Alignment, Error, Extent, MAX_SEGMENTS, Result, SegmentGeometry}; +use std::{ + cell::{Cell, RefCell}, + collections::{BTreeSet, HashSet}, + rc::Rc, +}; + +/// Stable slot number within one allocation table, not authority to access it. +#[derive(Clone, Copy, Debug, Eq, Hash, PartialEq)] +pub struct SegmentId(pub u64); + +/// Monotonic reuse counter; zero is invalid in a restored image. +#[derive(Clone, Copy, Debug, Eq, Hash, PartialEq)] +pub struct Generation(pub u64); + +/// Persisted segment lifecycle, separate from live lease and freeze ownership. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub enum SegmentState { + /// Available for a new append. + Free, + + /// The sole appendable tail. + Open, + + /// Full or rotated away from; eligible for eviction. + Sealed, + + /// Rejects new leases while existing completion owners drain. + Evicting, +} + +/// Caller-owned recovery image; restore validates it before publishing any slot. +#[derive(Clone, Debug, Eq, PartialEq)] +pub struct SegmentSnapshot { + /// Ordered slot identity. + pub id: SegmentId, + + /// Current nonzero reuse generation. + pub generation: Generation, + + /// Persisted lifecycle state. + pub state: SegmentState, + + /// Aligned occupied prefix, including record padding. + pub used_bytes: u64, +} + +/// Unique worker-local lease that prevents reuse until its completion owner drops. +/// It authorizes the used prefix at acquisition, not just the latest append. +/// Any lease permits both reads and writes in that prefix; it is not exclusive +/// record ownership. The trusted caller must follow [`crate::Slab::write`]'s +/// ownership and publication rules. +/// A lease does not prove that bytes were initialized in this generation. Append +/// only reserves space; recycled bytes may belong to an earlier generation or a +/// different cache. The caller must check record integrity, authentication, and +/// cache identity before exposing bytes as a valid record. +/// +/// Lease counts cannot be duplicated by cloning a completion capability: +/// +/// ```compile_fail +/// fn duplicate(lease: page_alloc::SegmentLease) { let _ = lease.clone(); } +/// ``` +#[must_use = "keep the lease alive until the operation completes"] +pub struct SegmentLease { + id: SegmentId, + + generation: Generation, + + count: Rc>, + + table: TableIdentity, + + geometry: SegmentGeometry, + + start: u64, + + end: u64, +} + +impl SegmentLease { + /// Slot protected against recycling by this lease. + pub fn id(&self) -> SegmentId { + self.id + } + + /// Reuse generation captured when the lease was acquired. + pub fn generation(&self) -> Generation { + self.generation + } + + /// Borrow the nominal identity without granting table mutation authority. + pub(crate) fn table_identity(&self) -> TableIdentity { + self.table.clone() + } + + /// Validated dimensions captured independently of the table's lifetime. + pub(crate) fn geometry(&self) -> SegmentGeometry { + self.geometry + } + + /// An existing lease remains valid during eviction, but cannot authorize + /// bytes appended after it was acquired. + pub(crate) fn validate_extent(&self, extent: &Extent) -> Result<()> { + let end = extent + .offset() + .checked_add(extent.length() as u64) + .ok_or(Error::Corrupt)?; + if extent.offset() < self.start + || end > self.end + || self + .geometry() + .alignment() + .extent(extent.offset(), extent.length()) + .map_err(|_| Error::Corrupt)? + != *extent + { + return Err(Error::Corrupt); + } + Ok(()) + } +} + +impl Drop for SegmentLease { + /// Release exactly the one lease count acquired during construction. + fn drop(&mut self) { + self.count.set(self.count.get() - 1); + } +} + +/// Mutable recovery image paired with its independently retained lease counter. +struct Slot { + image: SegmentSnapshot, + + leases: Rc>, +} + +/// Holds a freeze independently of the table's lifetime. Dropping it thaws the table. +/// +/// A second guard cannot be created by cloning the first: +/// +/// ```compile_fail +/// fn duplicate(guard: page_alloc::FreezeGuard) { let _ = guard.clone(); } +/// ``` +#[must_use = "keep the guard alive while the table must remain frozen"] +#[derive(Debug)] +pub struct FreezeGuard(Rc>); + +impl Drop for FreezeGuard { + /// Release mutation exclusion even if the table has already been dropped. + fn drop(&mut self) { + self.0.set(false); + } +} + +/// Non-cloneable worker-local table whose leases and guards can outlive it. +/// Rc sharing is intentional within a worker; authority cannot cross threads. +/// +/// ```compile_fail +/// fn require_send() {} +/// require_send::(); +/// ``` +#[repr(align(64))] +pub struct Segments { + segment_bytes: u64, + + geometry: Cell>, + + slots: RefCell>, + + frozen: Rc>, + + table: TableIdentity, + + restore_epoch: Cell, + + open: Cell>, + + free: RefCell>, + + evicting: Cell, +} + +impl Segments { + /// Create an unconfigured table for Slab::open_configured to bind at startup. + pub fn new(segment_bytes: u64) -> Self { + Self { + segment_bytes, + geometry: Cell::new(None), + slots: RefCell::new(Vec::new()), + frozen: Rc::new(Cell::new(false)), + table: TableIdentity::new(), + restore_epoch: Cell::new(0), + open: Cell::new(None), + free: RefCell::new(BTreeSet::new()), + evicting: Cell::new(0), + } + } + + /// Fixed size of a physical segment. + pub fn segment_bytes(&self) -> u64 { + self.segment_bytes + } + + /// Configured physical capacity, or zero before configuration. + pub fn capacity_bytes(&self) -> u64 { + self.geometry.get().map_or(0, SegmentGeometry::slab_bytes) + } + + /// Build an entirely free table, rejecting geometry above the retained slot limit. + pub fn from_geometry(geometry: SegmentGeometry) -> Result { + let segments = Self::new(geometry.segment_bytes()); + segments.configure_table( + geometry.slab_bytes(), + usize::try_from(geometry.segment_count()).map_err(|_| Error::InvalidConfiguration)?, + geometry.alignment(), + )?; + Ok(segments) + } + + /// Whether validated geometry and the slot table have been installed. + pub fn is_configured(&self) -> bool { + self.geometry.get().is_some() + } + + /// Configured geometry, which may expose fewer slots than physical capacity. + pub fn geometry(&self) -> Option { + self.geometry.get() + } + + /// Clone identity only, without granting allocation authority. + pub(crate) fn table_identity(&self) -> TableIdentity { + self.table.clone() + } + + /// Recovery revision used to invalidate eviction cursor and recent-read state. + pub(crate) fn restore_epoch(&self) -> u64 { + self.restore_epoch.get() + } + + /// Install an entirely free bounded table exactly once, unless frozen. + #[cfg(any(test, feature = "simulation"))] + pub fn configure(&self, capacity: u64, count: usize, alignment: Alignment) -> Result<()> { + self.configure_table(capacity, count, alignment) + } + + /// Install validated geometry for startup or an explicitly constructed table. + pub(crate) fn configure_table( + &self, + capacity: u64, + count: usize, + alignment: Alignment, + ) -> Result<()> { + if self.frozen.get() { + return Err(Error::Busy); + } + if self.is_configured() || count as u64 > MAX_SEGMENTS { + return Err(Error::InvalidConfiguration); + } + let geometry = SegmentGeometry::new(capacity, self.segment_bytes, count as u64, alignment) + .map_err(|_| Error::InvalidConfiguration)?; + self.geometry.set(Some(geometry)); + *self.slots.borrow_mut() = (0..count) + .map(|id| Slot { + image: SegmentSnapshot { + id: SegmentId(id as u64), + generation: Generation(1), + state: SegmentState::Free, + used_bytes: 0, + }, + leases: Rc::new(Cell::new(0)), + }) + .collect(); + *self.free.borrow_mut() = (0..count).collect(); + Ok(()) + } + + /// Reserve an aligned used range. Malformed requests and lease overflow leave + /// state unchanged. A valid rollover without a free slot seals the open tail + /// before returning Busy, allowing reclamation to make a retry possible. + /// Reservation does not write or initialize disk bytes; see [`SegmentLease`]. + pub fn append(&self, length: usize) -> Result<(SegmentLease, Extent)> { + if self.frozen.get() { + return Err(Error::Busy); + } + let alignment = self.geometry().ok_or(Error::Unavailable)?.alignment(); + if length == 0 + || length as u64 > self.segment_bytes + || alignment.extent(0, length)?.length() != length + { + return Err(Error::InvalidConfiguration); + } + let mut slots = self.slots.borrow_mut(); + let previous_open = self.open.get(); + let usable_open = previous_open.filter(|&position| { + self.segment_bytes - slots[position].image.used_bytes >= length as u64 + }); + let position = match usable_open { + Some(position) => position, + None => match self.free.borrow().first() { + Some(&position) => position, + None => { + if let Some(previous) = previous_open { + slots[previous].image.state = SegmentState::Sealed; + self.open.set(None); + } + return Err(Error::Busy); + } + }, + }; + let slot = &slots[position]; + let offset = slot + .image + .id + .0 + .checked_mul(self.segment_bytes) + .and_then(|v| v.checked_add(slot.image.used_bytes)) + .ok_or(Error::InvalidConfiguration)?; + let extent = alignment.extent(offset, length)?; + let used_bytes = slot + .image + .used_bytes + .checked_add(length as u64) + .ok_or(Error::InvalidConfiguration)?; + // All fallible checks, including lease-counter overflow, precede mutation. + let lease = self.take_lease(slot, used_bytes)?; + if usable_open.is_none() { + if let Some(previous) = previous_open { + slots[previous].image.state = SegmentState::Sealed; + } + self.free.borrow_mut().remove(&position); + } + self.open.set(Some(position)); + let slot = &mut slots[position]; + slot.image.state = SegmentState::Open; + slot.image.used_bytes = used_bytes; + if slot.image.used_bytes == self.segment_bytes { + slot.image.state = SegmentState::Sealed; + self.open.set(None); + } + Ok((lease, extent)) + } + + /// Capture a checked prefix and increment its counter before publishing a lease. + fn take_lease(&self, slot: &Slot, used_bytes: u64) -> Result { + let start = slot + .image + .id + .0 + .checked_mul(self.segment_bytes) + .ok_or(Error::Corrupt)?; + let end = start.checked_add(used_bytes).ok_or(Error::Corrupt)?; + let geometry = self.geometry().ok_or(Error::Unavailable)?; + slot.leases + .set(slot.leases.get().checked_add(1).ok_or(Error::Busy)?); + Ok(SegmentLease { + id: slot.image.id, + generation: slot.image.generation, + count: slot.leases.clone(), + table: self.table.clone(), + geometry, + start, + end, + }) + } + + /// Convert an external slot number without truncation. + fn position(id: SegmentId) -> Result { + usize::try_from(id.0).map_err(|_| Error::Corrupt) + } + + /// Acquire the current used prefix only while the slot is Open or Sealed and + /// the generation matches. This checks allocation state, not whether bytes + /// were initialized in this generation or form valid records; see [`SegmentLease`]. + pub fn lease(&self, id: SegmentId, generation: Generation) -> Result { + let slots = self.slots.borrow(); + let slot = slots.get(Self::position(id)?).ok_or(Error::Corrupt)?; + if slot.image.generation != generation + || !matches!(slot.image.state, SegmentState::Open | SegmentState::Sealed) + { + return Err(Error::Stale); + } + self.take_lease(slot, slot.image.used_bytes) + } + + /// Stop new leases for a sealed slot; repeating eviction is harmless. + #[cfg(any(test, feature = "simulation"))] + pub fn begin_evict(&self, id: SegmentId) -> Result<()> { + self.begin_eviction(id) + } + + /// Enter eviction before the clock invokes any caller index removal. + fn begin_eviction(&self, id: SegmentId) -> Result<()> { + if self.frozen.get() { + return Err(Error::Busy); + } + let mut slots = self.slots.borrow_mut(); + let slot = slots.get_mut(Self::position(id)?).ok_or(Error::Corrupt)?; + if !matches!( + slot.image.state, + SegmentState::Sealed | SegmentState::Evicting + ) { + return Err(Error::Busy); + } + if slot.image.state != SegmentState::Evicting { + self.evicting.set(self.evicting.get() + 1); + slot.image.state = SegmentState::Evicting; + } + Ok(()) + } + + /// Reuse only an evicting, unleased slot, incrementing generation without wrap. + #[cfg(any(test, feature = "simulation"))] + pub fn recycle(&self, id: SegmentId) -> Result<()> { + self.recycle_evicted(id) + } + + /// Complete clock-authorized eviction only after outstanding leases drain. + fn recycle_evicted(&self, id: SegmentId) -> Result<()> { + if self.frozen.get() { + return Err(Error::Busy); + } + let mut slots = self.slots.borrow_mut(); + let slot = slots.get_mut(Self::position(id)?).ok_or(Error::Corrupt)?; + if slot.image.state != SegmentState::Evicting || slot.leases.get() != 0 { + return Err(Error::Busy); + } + slot.image.generation = Generation( + slot.image + .generation + .0 + .checked_add(1) + .ok_or(Error::Unavailable)?, + ); + slot.image.state = SegmentState::Free; + self.evicting.set(self.evicting.get() - 1); + slot.image.used_bytes = 0; + self.free.borrow_mut().insert(Self::position(id)?); + Ok(()) + } + + /// Copy the complete ordered image, including while the table is frozen. + pub fn snapshot(&self) -> Vec { + self.slots + .borrow() + .iter() + .map(|s| s.image.clone()) + .collect() + } + + /// Block allocation, eviction, recycling, and restore until the guard drops. + /// Reads and snapshots remain available; only one guard may exist at a time. + pub fn freeze(&self) -> Result { + if self.frozen.replace(true) { + return Err(Error::Busy); + } + Ok(FreezeGuard(self.frozen.clone())) + } + + /// Check the entire image and current restore eligibility without mutation. + pub fn validate_restore(&self, images: &[SegmentSnapshot]) -> Result<()> { + self.validate_restore_epoch(images).map(|_| ()) + } + + /// Validate recovery invariants and reserve the next nonwrapping epoch value. + fn validate_restore_epoch(&self, images: &[SegmentSnapshot]) -> Result { + if self.frozen.get() { + return Err(Error::Busy); + } + let slots = self.slots.borrow(); + if slots.iter().any(|s| s.leases.get() != 0) { + return Err(Error::Busy); + } + if images.len() != slots.len() { + return Err(Error::Corrupt); + } + let alignment = self.geometry().ok_or(Error::Unavailable)?.alignment(); + let mut open = false; + for (i, s) in images.iter().enumerate() { + if s.id.0 != i as u64 + || s.generation.0 == 0 + || s.generation.0 < slots[i].image.generation.0 + || s.used_bytes > self.segment_bytes + || !s.used_bytes.is_multiple_of(alignment.offset()) + || !s.used_bytes.is_multiple_of(alignment.length() as u64) + || (s.state == SegmentState::Free && s.used_bytes != 0) + || (s.state != SegmentState::Free && s.used_bytes == 0) + { + return Err(Error::Corrupt); + } + if s.state == SegmentState::Open { + if open || s.used_bytes == self.segment_bytes { + return Err(Error::Corrupt); + } + open = true; + } + } + self.restore_epoch + .get() + .checked_add(1) + .ok_or(Error::Unavailable) + } + + /// Validate the complete image before publishing it, sealing its open tail. + /// Frozen tables and outstanding leases return Busy without changing state. + /// A generation below the current slot generation returns Corrupt. + pub fn restore(&self, images: Vec) -> Result<()> { + let epoch = self.validate_restore_epoch(&images)?; + let mut slots = self.slots.borrow_mut(); + self.open.set(None); + self.free.borrow_mut().clear(); + self.evicting.set(0); + for (slot, mut image) in slots.iter_mut().zip(images) { + if image.state == SegmentState::Open { + image.state = SegmentState::Sealed; + } + if image.state == SegmentState::Free { + self.free.borrow_mut().insert(image.id.0 as usize); + } + if image.state == SegmentState::Evicting { + self.evicting.set(self.evicting.get() + 1); + } + slot.image = image; + } + self.restore_epoch.set(epoch); + Ok(()) + } + + /// Validate a stored mapping against current readable state and used bytes. + pub fn validate(&self, id: SegmentId, generation: Generation, extent: &Extent) -> Result<()> { + let slots = self.slots.borrow(); + let slot = slots.get(Self::position(id)?).ok_or(Error::Corrupt)?; + let start = id.0.checked_mul(self.segment_bytes).ok_or(Error::Corrupt)?; + let end = extent + .offset() + .checked_add(extent.length() as u64) + .ok_or(Error::Corrupt)?; + if slot.image.generation != generation + || !matches!(slot.image.state, SegmentState::Open | SegmentState::Sealed) + { + return Err(Error::Stale); + } + if extent.offset() < start + || end + > start + .checked_add(slot.image.used_bytes) + .ok_or(Error::Corrupt)? + { + return Err(Error::Corrupt); + } + let alignment = self.geometry().ok_or(Error::Unavailable)?.alignment(); + if alignment + .extent(extent.offset(), extent.length()) + .map_err(|_| Error::Corrupt)? + != *extent + { + return Err(Error::Corrupt); + } + Ok(()) + } + + /// Validate a live lease, including table identity and its captured used range. + /// Existing leases remain usable while their segment is Evicting. + pub fn validate_lease(&self, lease: &SegmentLease, extent: &Extent) -> Result<()> { + if !self.table.matches(&lease.table) { + return Err(Error::Stale); + } + lease.validate_extent(extent) + } + + /// Number of immediately appendable slots, excluding pending eviction. + pub fn free_count(&self) -> usize { + self.free.borrow().len() + } + + /// Number of retained slots, which may be less than physical capacity. + pub fn count(&self) -> usize { + self.slots.borrow().len() + } + + /// Inspect a valid slot without granting mutation or lease authority. + pub fn state(&self, id: SegmentId) -> Result { + self.slots + .borrow() + .get(Self::position(id)?) + .map(|s| s.image.state) + .ok_or(Error::Corrupt) + } +} + +/// Nominal, unforgeable identity shared by a table, bound file, and its leases. +/// Cloning identity does not clone allocation authority or keep the table alive. +#[derive(Clone)] +pub(crate) struct TableIdentity(Rc<()>); + +impl TableIdentity { + /// Mint one distinct identity for a newly constructed table. + fn new() -> Self { + Self(Rc::new(())) + } + + /// Compare allocation identity, never persisted slot numbers or generations. + pub(crate) fn matches(&self, other: &Self) -> bool { + Rc::ptr_eq(&self.0, &other.0) + } +} + +/// Application-owned mappings; removal must compare current entries first. +/// Callbacks own metadata and version side effects and must never exceed budget. +pub trait SegmentEntries { + /// Active unpublished writes must not lose publication authority. Read + /// leases still permit eviction and independently fence physical reuse. + fn can_evict(&self, _segment: SegmentId) -> bool { + true + } + + /// Remove at most budget current mappings, returning the number removed. + fn remove_bounded(&self, segment: SegmentId, budget: usize) -> usize; + + /// Report whether any current mapping still refers to this segment. + fn is_empty(&self, segment: SegmentId) -> bool; +} + +/// One worker-local cursor and recent-read set for index and physical reclamation. +#[repr(align(64))] +pub struct SegmentClock { + segments: Rc, + + hand: Cell, + + recent: RefCell>, + + restore_epoch: Cell, +} + +impl SegmentClock { + /// Attach bounded reclamation to one table without cloning its authority. + pub fn new(segments: Rc) -> Self { + let epoch = segments.restore_epoch(); + Self { + segments, + hand: Cell::new(0), + recent: RefCell::new(HashSet::new()), + restore_epoch: Cell::new(epoch), + } + } + + /// Give live segments a second chance; ignore completions arriving after eviction. + pub fn mark_read(&self, segment: SegmentId) -> Result<()> { + self.sync_restore(); + if !matches!( + self.segments.state(segment)?, + SegmentState::Open | SegmentState::Sealed + ) { + return Ok(()); + } + self.recent.borrow_mut().insert(segment); + Ok(()) + } + + /// Clear transient eviction history after successful table recovery. + fn sync_restore(&self) { + let epoch = self.segments.restore_epoch(); + if self.restore_epoch.replace(epoch) != epoch { + self.hand.set(0); + self.recent.borrow_mut().clear(); + } + } + + /// Advance one slot; callers only invoke this inside a nonempty bounded sweep. + fn next(&self, count: usize) -> SegmentId { + let hand = self.hand.get() % count; + self.hand.set((hand + 1) % count); + SegmentId(hand as u64) + } + + /// Forget one mapping per candidate until admission succeeds, without changing + /// bytes or generations. At most two rotations, further capped by max_visits. + /// Freeze does not block caller-owned index mutation. + pub fn reclaim_index( + &self, + entries: &impl SegmentEntries, + max_visits: usize, + mut ready: impl FnMut() -> bool, + ) -> Result<()> { + self.sync_restore(); + if ready() { + return Ok(()); + } + let count = self.segments.count(); + for _ in 0..count.saturating_mul(2).min(max_visits) { + let id = self.next(count); + if self.recent.borrow_mut().remove(&id) { + continue; + } + if entries.remove_bounded(id, 1) > 1 { + return Err(Error::InvalidConfiguration); + } + if ready() { + return Ok(()); + } + } + Err(Error::Busy) + } + + /// Visit at most min(slot count, max_visits, 64) slots, then rank eligible + /// candidates before entering eviction. Skipped slots count toward this limit. + /// Scores are soft preferences, not pins. The callback must itself bound + /// mapping inspection. + /// Existing eviction work sorts first so partial removals always make progress. + /// Unlike the legacy clock, logical heat is supplied by the caller, so recent + /// disk completions do not override value ranking. + /// For a nonzero reserve, success requires both the reserve (capped at the slot + /// count) and no pending evictions. Busy is intentional even if the reserve is + /// met while evictions remain. Use [`Segments::free_count`] to check capacity; + /// keep retrying bounded reclamation with a nonzero reserve to drain evictions. + /// A zero reserve is a no-op, not a drain request. + pub fn reclaim_scored( + &self, + entries: &impl SegmentEntries, + free_reserve: usize, + max_visits: usize, + max_entries: usize, + mut score: impl FnMut(SegmentId) -> u64, + ) -> Result<()> { + self.sync_restore(); + if free_reserve == 0 { + return Ok(()); + } + let count = self.segments.count(); + if count == 0 { + return Err(Error::Unavailable); + } + let target = free_reserve.min(count); + if self.segments.free_count() >= target && self.segments.evicting.get() == 0 { + return Ok(()); + } + let mut candidates = Vec::new(); + for visit in 0..count.min(max_visits).min(64) { + let id = self.next(count); + let state = self.segments.state(id)?; + if !matches!(state, SegmentState::Sealed | SegmentState::Evicting) { + continue; + } + if state == SegmentState::Sealed && !entries.can_evict(id) { + continue; + } + // Rank before any begin_eviction or index side effects. + let value = if state == SegmentState::Evicting { + 0 + } else { + score(id) + }; + candidates.push((state != SegmentState::Evicting, value, visit, id)); + } + candidates.sort_by_key(|(sealed, value, visit, _)| (*sealed, *value, *visit)); + let mut entries_left = max_entries; + for (_, _, _, id) in candidates { + let state = self.segments.state(id)?; + if state == SegmentState::Sealed + && (self + .segments + .free_count() + .saturating_add(self.segments.evicting.get()) + >= target + || !entries.can_evict(id)) + { + continue; + } + if entries_left == 0 && !entries.is_empty(id) { + continue; + } + self.segments.begin_eviction(id)?; + self.recent.borrow_mut().remove(&id); + if entries_left != 0 && !entries.is_empty(id) { + let removed = entries.remove_bounded(id, entries_left); + if removed > entries_left { + return Err(Error::InvalidConfiguration); + } + entries_left -= removed; + } + if !entries.is_empty(id) { + continue; + } + match self.segments.recycle_evicted(id) { + Ok(()) | Err(Error::Busy) => {} + Err(error) => return Err(error), + } + } + if self.segments.free_count() >= target && self.segments.evicting.get() == 0 { + Ok(()) + } else { + Err(Error::Busy) + } + } + + /// Reclaim a reserve with bounded visits and mapping removals, without compaction. + /// Busy leases keep segments Evicting until a later sweep. A zero reserve is a + /// no-op; a zero mapping budget can recycle empty but not populated segments. + /// For a nonzero reserve, success requires both the reserve (capped at the slot + /// count) and no pending evictions. Busy is intentional even if the reserve is + /// met while evictions remain. Use [`Segments::free_count`] to check capacity; + /// keep retrying bounded reclamation with a nonzero reserve to drain evictions. + pub fn reclaim( + &self, + entries: &impl SegmentEntries, + free_reserve: usize, + max_visits: usize, + max_entries: usize, + ) -> Result<()> { + self.sync_restore(); + if free_reserve == 0 { + return Ok(()); + } + let count = self.segments.count(); + if count == 0 { + return Err(Error::Unavailable); + } + let target = free_reserve.min(count); + let mut free = self.segments.free_count(); + let mut entries_left = max_entries; + for _ in 0..count.saturating_mul(2).min(max_visits) { + if free >= target && self.segments.evicting.get() == 0 { + return Ok(()); + } + let id = self.next(count); + let state = self.segments.state(id)?; + if !matches!(state, SegmentState::Sealed | SegmentState::Evicting) { + continue; + } + if state == SegmentState::Sealed + && (free.saturating_add(self.segments.evicting.get()) >= target + || !entries.can_evict(id)) + { + continue; + } + if self.recent.borrow_mut().remove(&id) { + continue; + } + if entries_left == 0 && !entries.is_empty(id) { + continue; + } + self.segments.begin_eviction(id)?; + if entries_left != 0 && !entries.is_empty(id) { + let removed = entries.remove_bounded(id, entries_left); + if removed > entries_left { + return Err(Error::InvalidConfiguration); + } + entries_left -= removed; + } + if !entries.is_empty(id) { + continue; + } + match self.segments.recycle_evicted(id) { + Ok(()) => free += 1, + Err(Error::Busy) => {} + Err(e) => return Err(e), + } + } + if free >= target && self.segments.evicting.get() == 0 { + Ok(()) + } else { + Err(Error::Busy) + } + } +} + +/// Pure state-machine tests, including synthetic overflow and invalid recovery images. +#[cfg(test)] +mod tests { + use super::*; + + /// Build a bounded free table with deterministic alignment. + fn segments(bytes: u64, count: usize) -> Segments { + let s = Segments::new(bytes); + s.configure( + bytes * count as u64, + count, + Alignment::new(512, 512, 512).unwrap(), + ) + .unwrap(); + s + } + + /// Lease ownership prevents reuse and old-generation acquisition. + #[test] + fn no_reuse_before_lease_drop_and_no_aba() { + let s = segments(1024, 2); + let a = s.append(1024).unwrap(); + s.begin_evict(a.0.id()).unwrap(); + assert_eq!(s.recycle(a.0.id()), Err(Error::Busy)); + drop(a); + s.recycle(SegmentId(0)).unwrap(); + assert!(s.lease(SegmentId(0), Generation(1)).is_err()); + assert_eq!(s.append(512).unwrap().0.generation(), Generation(2)); + let frozen = s.freeze().unwrap(); + assert!(s.append(512).is_err()); + drop(frozen); + } + + /// Recovery seals the append tail and rejects unaligned occupancy. + #[test] + fn restore_seals_open_and_rejects_invalid_geometry() { + let s = segments(1024, 1); + drop(s.append(512).unwrap()); + let mut snap = s.snapshot(); + s.restore(snap.clone()).unwrap(); + assert_eq!(s.snapshot()[0].state, SegmentState::Sealed); + snap[0].used_bytes = 513; + assert!(s.restore(snap).is_err()); + } + + /// Reject rollback atomically while allowing equal or newer generations. + #[test] + fn restore_rejects_generation_rollback_without_reviving_stale_mappings() { + let s = segments(1024, 2); + drop(s.append(1024).unwrap()); + let (lease, extent) = s.append(1024).unwrap(); + let id = lease.id(); + let generation = lease.generation(); + drop(lease); + let old_sealed = s.snapshot(); + s.begin_evict(id).unwrap(); + s.recycle(id).unwrap(); + let before = s.snapshot(); + let epoch = s.restore_epoch(); + + for state in [ + SegmentState::Free, + SegmentState::Open, + SegmentState::Sealed, + SegmentState::Evicting, + ] { + let mut images = old_sealed.clone(); + images[0].generation = Generation(3); + images[1].state = state; + images[1].used_bytes = match state { + SegmentState::Free => 0, + SegmentState::Open => 512, + _ => 1024, + }; + assert_eq!(s.validate_restore(&images), Err(Error::Corrupt)); + assert_eq!(s.restore(images), Err(Error::Corrupt)); + assert_eq!(s.snapshot(), before); + assert_eq!(s.restore_epoch(), epoch); + assert_eq!(*s.free.borrow(), BTreeSet::from([1])); + assert_eq!(s.open.get(), None); + assert_eq!(s.evicting.get(), 0); + } + + let (lease, new_extent) = s.append(1024).unwrap(); + assert_eq!(lease.id(), id); + assert_eq!(lease.generation(), Generation(2)); + assert_eq!(new_extent, extent); + assert_eq!(s.validate(id, generation, &extent), Err(Error::Stale)); + assert!(matches!(s.lease(id, generation), Err(Error::Stale))); + drop(lease); + + let mut images = s.snapshot(); + assert_eq!(s.validate_restore(&images), Ok(())); + s.restore(images.clone()).unwrap(); + images[1].generation = Generation(3); + assert_eq!(s.validate_restore(&images), Ok(())); + s.restore(images).unwrap(); + assert_eq!(s.snapshot()[1].generation, Generation(3)); + assert_eq!(s.validate(id, generation, &extent), Err(Error::Stale)); + } + + /// Tail rotation and generation exhaustion never wrap into stale authority. + #[test] + fn generation_exhaustion_never_wraps_and_small_tails_are_sealed() { + let s = segments(1024, 2); + drop(s.append(512).unwrap()); + let next = s.append(1024).unwrap(); + assert_eq!(next.0.id(), SegmentId(1)); + drop(next); + assert_eq!(s.state(SegmentId(0)).unwrap(), SegmentState::Sealed); + let mut snapshot = s.snapshot(); + snapshot[0].generation = Generation(u64::MAX); + s.restore(snapshot).unwrap(); + s.begin_evict(SegmentId(0)).unwrap(); + assert_eq!(s.recycle(SegmentId(0)), Err(Error::Unavailable)); + assert_eq!(s.snapshot()[0].generation, Generation(u64::MAX)); + } + + /// Validation distinguishes stale identity, corrupt ranges, and busy ownership. + #[test] + fn validation_rejects_unwritten_ranges_generations_and_busy_restore() { + let s = segments(1024, 1); + assert!(s.append(0).is_err()); + assert!(s.append(513).is_err()); + let (lease, extent) = s.append(512).unwrap(); + assert_eq!(s.validate(lease.id(), lease.generation(), &extent), Ok(())); + assert_eq!( + s.validate(lease.id(), Generation(2), &extent), + Err(Error::Stale) + ); + assert_eq!( + s.validate( + lease.id(), + lease.generation(), + &Extent::new(512, 512).unwrap() + ), + Err(Error::Corrupt) + ); + let mut images = s.snapshot(); + assert_eq!(s.restore(images.clone()), Err(Error::Busy)); + drop(lease); + images[0].generation = Generation(0); + assert_eq!(s.restore(images), Err(Error::Corrupt)); + assert_eq!(s.state(SegmentId(0)).unwrap(), SegmentState::Open); + assert_eq!(s.count(), 1); + assert_eq!(s.free_count(), 0); + assert_eq!(s.capacity_bytes(), 1024); + let frozen = s.freeze().unwrap(); + assert!(matches!(s.freeze(), Err(Error::Busy))); + assert_eq!(s.begin_evict(SegmentId(0)), Err(Error::Busy)); + assert_eq!(s.recycle(SegmentId(0)), Err(Error::Busy)); + drop(frozen); + drop(s.append(512).unwrap()); + assert!(matches!(s.append(512), Err(Error::Busy))); + } + + /// Invalid configuration does not partially install geometry or slots. + #[test] + fn configuration_rejects_zero_and_misaligned_capacity() { + let a = Alignment::new(512, 512, 512).unwrap(); + for (bytes, capacity, count) in [ + (0, 1024, 1), + (1024, 0, 1), + (513, 1026, 2), + (1024, 1024, 0), + (1024, 1024, 2), + (1024, 1025, 1), + (1024, 1024 * 1_000_001, 1_000_001), + ] { + assert_eq!( + Segments::new(bytes).configure(capacity, count, a), + Err(Error::InvalidConfiguration) + ); + } + let s = segments(1024, 1); + assert_eq!(s.configure(1024, 1, a), Err(Error::InvalidConfiguration)); + assert!(s.state(SegmentId(u64::MAX)).is_err()); + let s = Segments::new(1024); + assert_eq!(s.configure(1025, 1, a), Err(Error::InvalidConfiguration)); + assert!(!s.is_configured()); + assert_eq!(s.geometry(), None); + assert_eq!(s.capacity_bytes(), 0); + assert_eq!(s.free_count(), 0); + assert!(s.snapshot().is_empty()); + s.configure(1024, 1, a).unwrap(); + } + + /// Malformed append and synthetic saturation cannot consume or seal slots. + #[test] + fn append_failures_preserve_open_tail_free_list_and_lease_counts() { + let s = segments(1024, 1); + drop(s.append(512).unwrap()); + let before = s.snapshot(); + for length in [0, 1, 513, 1536] { + assert!(s.append(length).is_err()); + assert_eq!(s.snapshot(), before); + assert_eq!(s.open.get(), Some(0)); + assert_eq!(s.free_count(), 0); + assert_eq!(s.slots.borrow()[0].leases.get(), 0); + } + drop(s.append(512).unwrap()); + assert_eq!(s.state(SegmentId(0)), Ok(SegmentState::Sealed)); + let s = segments(1024, 2); + drop(s.append(512).unwrap()); + for (position, length) in [(0, 512), (1, 1024)] { + s.slots.borrow()[position].leases.set(usize::MAX); + let before = s.snapshot(); + assert!(matches!(s.append(length), Err(Error::Busy))); + assert_eq!(s.snapshot(), before); + assert_eq!(s.open.get(), Some(0)); + assert_eq!(*s.free.borrow(), BTreeSet::from([1])); + assert_eq!(s.slots.borrow()[position].leases.get(), usize::MAX); + s.slots.borrow()[position].leases.set(0); + } + drop(s.append(1024).unwrap()); + assert_eq!(s.state(SegmentId(0)), Ok(SegmentState::Sealed)); + let s = segments(1024, 2); + drop(s.append(512).unwrap()); + s.slots.borrow_mut()[1].image.id = SegmentId(u64::MAX); + let before = s.snapshot(); + assert!(matches!(s.append(1024), Err(Error::InvalidConfiguration))); + assert_eq!(s.snapshot(), before); + assert_eq!(s.open.get(), Some(0)); + assert_eq!(*s.free.borrow(), BTreeSet::from([1])); + assert_eq!(s.slots.borrow()[1].leases.get(), 0); + } + + /// Freeze ownership survives table destruction and releases during unwinding. + #[test] + fn freeze_guard_drop_thaws_and_can_outlive_table() { + let s = segments(1024, 1); + let frozen = s.freeze().unwrap(); + let before = s.snapshot(); + assert_eq!(s.restore(before.clone()), Err(Error::Busy)); + assert_eq!(s.validate_restore(&before), Err(Error::Busy)); + assert!(matches!(s.append(512), Err(Error::Busy))); + assert!(matches!(s.freeze(), Err(Error::Busy))); + assert_eq!(s.snapshot(), before); + drop(frozen); + drop(s.append(512).unwrap()); + let frozen = s.freeze().unwrap(); + drop(s); + drop(frozen); + let s = Segments::new(1024); + let frozen = s.freeze().unwrap(); + assert_eq!( + s.configure(1024, 1, Alignment::new(512, 512, 512).unwrap()), + Err(Error::Busy) + ); + drop(frozen); + assert!(!s.is_configured()); + assert_eq!(s.geometry(), None); + let s = segments(1024, 1); + assert!( + std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { + let _frozen = s.freeze().unwrap(); + panic!("simulated owner failure"); + })) + .is_err() + ); + drop(s.append(512).unwrap()); + } + + /// A lease retains both validated dimensions and the nominal table identity. + #[test] + fn geometry_is_shared_by_configuration_and_leases() { + let geometry = + SegmentGeometry::new(4096, 1024, 2, Alignment::new(512, 512, 512).unwrap()).unwrap(); + let s = Segments::from_geometry(geometry).unwrap(); + assert!(s.is_configured()); + assert_eq!(s.geometry(), Some(geometry)); + assert_eq!(s.count(), 2); + assert_eq!(s.free_count(), 2); + let (lease, extent) = s.append(512).unwrap(); + assert_eq!(lease.geometry(), geometry); + assert!(lease.table_identity().matches(&s.table_identity())); + assert_eq!(s.validate_lease(&lease, &extent), Ok(())); + } + + /// Large sparse capacity need not allocate an equally large slot table. + #[test] + fn large_physical_geometry_supports_bounded_partial_tables() { + let capacity = 1024 * (MAX_SEGMENTS + 1); + let alignment = Alignment::new(512, 512, 512).unwrap(); + let full = SegmentGeometry::new(capacity, 1024, MAX_SEGMENTS + 1, alignment).unwrap(); + assert!(matches!( + Segments::from_geometry(full), + Err(Error::InvalidConfiguration) + )); + let s = Segments::new(1024); + assert_eq!( + s.configure(capacity, (MAX_SEGMENTS + 1) as usize, alignment), + Err(Error::InvalidConfiguration) + ); + assert!(!s.is_configured()); + s.configure(capacity, 2, alignment).unwrap(); + assert_eq!(s.capacity_bytes(), capacity); + assert_eq!(s.count(), 2); + assert_eq!(s.geometry().unwrap().segment_count(), 2); + drop(s.append(1024).unwrap()); + assert_eq!(s.free_count(), 1); + } + + /// Existing leases authorize only their captured prefix, including during eviction. + #[test] + fn leases_validate_table_and_captured_used_range_through_eviction() { + let s = segments(1024, 1); + let other = segments(1024, 1); + let (lease, extent) = s.append(512).unwrap(); + let (_, later) = s.append(512).unwrap(); + assert_eq!(s.validate_lease(&lease, &later), Err(Error::Corrupt)); + assert_eq!(other.validate_lease(&lease, &extent), Err(Error::Stale)); + assert_eq!( + lease.validate_extent(&Extent::new(1, 511).unwrap()), + Err(Error::Corrupt) + ); + assert_eq!( + lease.validate_extent(&Extent::new(0, 1).unwrap()), + Err(Error::Corrupt) + ); + s.begin_evict(SegmentId(0)).unwrap(); + assert_eq!(s.validate_lease(&lease, &extent), Ok(())); + assert_eq!( + s.validate(lease.id(), lease.generation(), &extent), + Err(Error::Stale) + ); + assert!(matches!( + s.lease(lease.id(), lease.generation()), + Err(Error::Stale) + )); + assert!(matches!( + s.lease(SegmentId(1), Generation(1)), + Err(Error::Corrupt) + )); + drop(s); + assert_eq!(lease.validate_extent(&extent), Ok(())); + drop(lease); + } + + /// Stored malformed ranges are corruption rather than configuration errors. + #[test] + fn malformed_extents_are_corrupt_not_configuration_errors() { + let s = segments(1024, 1); + let (lease, _) = s.append(1024).unwrap(); + for extent in [ + Extent::new(1, 512).unwrap(), + Extent::new(0, 513).unwrap(), + Extent::new(1024, 512).unwrap(), + ] { + assert_eq!( + s.validate(lease.id(), lease.generation(), &extent), + Err(Error::Corrupt) + ); + } + } + + /// Recovery validates all states atomically and fails before epoch wraparound. + #[test] + fn invalid_restore_states_are_atomic_and_valid_states_round_trip() { + let s = segments(1024, 3); + drop(s.append(1024).unwrap()); + drop(s.append(512).unwrap()); + let before = s.snapshot(); + let mut cases = Vec::new(); + for state in [ + SegmentState::Open, + SegmentState::Sealed, + SegmentState::Evicting, + ] { + let mut images = before.clone(); + images[2].state = state; + cases.push(images); + } + let mut images = before.clone(); + images[0].state = SegmentState::Open; + cases.push(images); + let mut images = before.clone(); + images[0].state = SegmentState::Open; + images[0].used_bytes = 512; + cases.push(images); + for images in cases { + assert_eq!(s.restore(images), Err(Error::Corrupt)); + assert_eq!(s.snapshot(), before); + assert_eq!(s.free_count(), 1); + assert_eq!(s.open.get(), Some(1)); + assert_eq!(s.restore_epoch(), 0); + } + let mut images = before; + images[0].state = SegmentState::Evicting; + s.restore(images).unwrap(); + assert_eq!(s.restore_epoch(), 1); + assert_eq!(s.state(SegmentId(0)), Ok(SegmentState::Evicting)); + assert_eq!(s.state(SegmentId(1)), Ok(SegmentState::Sealed)); + s.recycle(SegmentId(0)).unwrap(); + assert_eq!(s.free_count(), 2); + let before = s.snapshot(); + s.restore_epoch.set(u64::MAX - 1); + assert_eq!(s.validate_restore(&before), Ok(())); + s.restore(before.clone()).unwrap(); + assert_eq!(s.restore_epoch(), u64::MAX); + assert_eq!(s.snapshot(), before); + s.restore_epoch.set(u64::MAX); + assert_eq!(s.validate_restore(&before), Err(Error::Unavailable)); + assert_eq!(s.restore(before.clone()), Err(Error::Unavailable)); + assert_eq!(s.snapshot(), before); + } +} + +/// Bounded sweep state-space coverage, including invalid callbacks and recovery. +#[cfg(test)] +mod clock_tests { + use super::*; + + /// Deterministic index occupancy and removal-call observations. + struct Entries { + counts: RefCell>, + + calls: RefCell>, + + evictable: Cell, + } + + impl Entries { + /// Populate a synthetic index without changing allocator state. + fn new(counts: Vec) -> Self { + Self { + counts: RefCell::new(counts), + calls: RefCell::new(vec![]), + evictable: Cell::new(true), + } + } + } + + impl SegmentEntries for Entries { + /// Allow tests to protect unpublished mappings before eviction begins. + fn can_evict(&self, _: SegmentId) -> bool { + self.evictable.get() + } + + /// Record and honor each bounded removal request. + fn remove_bounded(&self, id: SegmentId, budget: usize) -> usize { + self.calls.borrow_mut().push((id, budget)); + let mut counts = self.counts.borrow_mut(); + let count = &mut counts[id.0 as usize]; + let removed = (*count).min(budget); + *count -= removed; + removed + } + + /// Read current occupancy after bounded removal. + fn is_empty(&self, id: SegmentId) -> bool { + self.counts.borrow()[id.0 as usize] == 0 + } + } + + /// Create a shared table with enough slots for a bounded clock scenario. + fn segments(count: usize) -> Rc { + let segments = Rc::new(Segments::new(1024)); + segments + .configure( + 1024 * count as u64, + count, + Alignment::new(512, 512, 512).unwrap(), + ) + .unwrap(); + segments + } + + /// Empty tables and zero visits never invoke mapping removal. + #[test] + fn empty_and_zero_budget_sweeps_do_not_call_entries() { + let clock = SegmentClock::new(Rc::new(Segments::new(1024))); + let entries = Entries::new(vec![]); + assert_eq!(clock.reclaim_index(&entries, 64, || true), Ok(())); + assert_eq!( + clock.reclaim_index(&entries, 64, || false), + Err(Error::Busy) + ); + assert_eq!(clock.reclaim(&entries, 1, 64, 256), Err(Error::Unavailable)); + assert_eq!(clock.mark_read(SegmentId(0)), Err(Error::Corrupt)); + assert!(entries.calls.borrow().is_empty()); + let clock = SegmentClock::new(segments(1)); + let entries = Entries::new(vec![1]); + assert_eq!(clock.reclaim_index(&entries, 0, || false), Err(Error::Busy)); + assert!(entries.calls.borrow().is_empty()); + } + + /// Index admission can forget open mappings without reclaiming their storage. + #[test] + fn index_second_chance_removes_open_mappings_without_recycling() { + let segments = segments(2); + drop(segments.append(512).unwrap()); + let clock = SegmentClock::new(segments.clone()); + let entries = Entries::new(vec![1, 1]); + clock.mark_read(SegmentId(0)).unwrap(); + clock + .reclaim_index(&entries, 64, || entries.counts.borrow()[1] == 0) + .unwrap(); + assert_eq!(*entries.counts.borrow(), [1, 0]); + clock.mark_read(SegmentId(0)).unwrap(); + clock + .reclaim_index(&entries, 64, || entries.counts.borrow()[0] == 0) + .unwrap(); + assert_eq!(segments.state(SegmentId(0)), Ok(SegmentState::Open)); + assert!(segments.lease(SegmentId(0), Generation(1)).is_ok()); + assert_eq!(segments.free_count(), 1); + } + + /// Calls resume the prior cursor but never exceed visits or two rotations. + #[test] + fn sweep_budget_persists_cursor_and_limits_to_two_rotations() { + let clock = SegmentClock::new(segments(40)); + let entries = Entries::new(vec![10; 40]); + assert_eq!( + clock.reclaim_index(&entries, 64, || false), + Err(Error::Busy) + ); + assert_eq!(entries.calls.borrow().len(), 64); + assert_eq!(entries.calls.borrow()[63], (SegmentId(23), 1)); + entries.calls.borrow_mut().clear(); + assert_eq!(clock.reclaim_index(&entries, 1, || false), Err(Error::Busy)); + assert_eq!(*entries.calls.borrow(), [(SegmentId(24), 1)]); + let clock = SegmentClock::new(segments(2)); + let entries = Entries::new(vec![10; 2]); + assert_eq!( + clock.reclaim_index(&entries, 64, || false), + Err(Error::Busy) + ); + assert_eq!(entries.calls.borrow().len(), 4); + } + + /// Mapping budget exhaustion leaves partial eviction for the next bounded call. + #[test] + fn mapping_budget_keeps_partial_eviction_until_next_sweep() { + let segments = segments(2); + drop(segments.append(1024).unwrap()); + drop(segments.append(1024).unwrap()); + let clock = SegmentClock::new(segments.clone()); + let entries = Entries::new(vec![257, 0]); + assert_eq!(clock.reclaim(&entries, 2, 64, 256), Err(Error::Busy)); + assert_eq!(*entries.counts.borrow(), [1, 0]); + assert_eq!(*entries.calls.borrow(), [(SegmentId(0), 256)]); + assert_eq!(segments.state(SegmentId(0)), Ok(SegmentState::Evicting)); + assert_eq!(segments.free_count(), 1); + clock.reclaim(&entries, 2, 64, 256).unwrap(); + assert_eq!(segments.free_count(), 2); + assert_eq!(*entries.counts.borrow(), [0, 0]); + } + + /// Sealed vetoes preserve mappings, but cannot strand eviction already in progress. + #[test] + fn reclaim_honors_sealed_veto_but_drains_existing_eviction() { + for mapping_count in [0usize, 2] { + let segments = segments(1); + let held = segments.append(1024).unwrap().0; + let clock = SegmentClock::new(segments.clone()); + let entries = Entries::new(vec![mapping_count]); + entries.evictable.set(false); + let before = segments.snapshot(); + assert_eq!(before[0].state, SegmentState::Sealed); + + assert_eq!(clock.reclaim(&entries, 1, 2, 1), Err(Error::Busy)); + assert_eq!(segments.snapshot(), before); + assert_eq!(*entries.counts.borrow(), [mapping_count]); + assert!(entries.calls.borrow().is_empty()); + assert_eq!(segments.free_count(), 0); + + entries.evictable.set(true); + assert_eq!(clock.reclaim(&entries, 1, 1, 1), Err(Error::Busy)); + assert_eq!(segments.state(SegmentId(0)), Ok(SegmentState::Evicting)); + assert_eq!(*entries.counts.borrow(), [mapping_count.saturating_sub(1)]); + + // Once eviction starts, a later veto must not block remaining mappings. + entries.evictable.set(false); + assert_eq!(clock.reclaim(&entries, 1, 1, 1), Err(Error::Busy)); + assert_eq!(*entries.counts.borrow(), [0]); + assert_eq!(entries.calls.borrow().len(), mapping_count); + assert_eq!(segments.state(SegmentId(0)), Ok(SegmentState::Evicting)); + assert_eq!(segments.free_count(), 0); + assert_eq!(segments.snapshot()[0].generation, Generation(1)); + + // The live lease, not the veto, remains the physical reuse fence. + drop(held); + clock.reclaim(&entries, 1, 1, 0).unwrap(); + assert_eq!(segments.state(SegmentId(0)), Ok(SegmentState::Free)); + assert_eq!(segments.free_count(), 1); + assert_eq!(segments.snapshot()[0].generation, Generation(2)); + } + } + + /// Pending victims satisfy demand without evicting more leased segments. + #[test] + fn reclaim_pending_victims_count_toward_reserve() { + let segments = segments(3); + let mut leases: Vec<_> = (0..3) + .map(|_| Some(segments.append(1024).unwrap().0)) + .collect(); + let clock = SegmentClock::new(segments.clone()); + let entries = Entries::new(vec![1; 3]); + for _ in 0..6 { + assert_eq!(clock.reclaim(&entries, 1, 1, 256), Err(Error::Busy)); + assert_eq!(*entries.counts.borrow(), [0, 1, 1]); + assert_eq!(segments.evicting.get(), 1); + assert_eq!(segments.free_count(), 0); + assert_eq!(segments.state(SegmentId(0)), Ok(SegmentState::Evicting)); + assert!(segments.lease(SegmentId(1), Generation(1)).is_ok()); + assert!(segments.lease(SegmentId(2), Generation(1)).is_ok()); + } + drop(leases[0].take()); + clock.reclaim(&entries, 1, 6, 0).unwrap(); + assert_eq!(segments.free_count(), 1); + assert_eq!(segments.evicting.get(), 0); + assert_eq!(*entries.counts.borrow(), [0, 1, 1]); + assert_eq!(segments.snapshot()[0].generation, Generation(2)); + } + + /// Lowering the target still drains existing victims within each visit budget. + #[test] + fn reclaim_pending_victims_drain_after_reserve_is_met() { + for max_visits in [1, 8] { + let segments = segments(4); + let mut leases: Vec<_> = (0..3) + .map(|_| Some(segments.append(1024).unwrap().0)) + .collect(); + drop(segments.append(1024).unwrap()); + let clock = SegmentClock::new(segments.clone()); + let entries = Entries::new(vec![1; 4]); + assert_eq!(clock.reclaim(&entries, 3, 3, 3), Err(Error::Busy)); + assert_eq!(segments.evicting.get(), 3); + assert_eq!(*entries.counts.borrow(), [0, 0, 0, 1]); + + // Wrap the cursor without starting a fourth victim. + entries.evictable.set(false); + assert_eq!(clock.reclaim(&entries, 1, 1, 1), Err(Error::Busy)); + drop(leases[0].take()); + assert_eq!(clock.reclaim(&entries, 1, max_visits, 0), Err(Error::Busy)); + assert_eq!(segments.free_count(), 1); + assert_eq!(segments.evicting.get(), 2); + assert!(matches!( + segments.lease(SegmentId(1), Generation(1)), + Err(Error::Stale) + )); + + drop(leases); + let before = segments.snapshot(); + assert_eq!(clock.reclaim(&entries, 0, 8, 8), Ok(())); + assert_eq!(clock.reclaim(&entries, 1, 0, 0), Err(Error::Busy)); + assert_eq!(segments.snapshot(), before); + for _ in 0..4 { + let _ = clock.reclaim(&entries, 1, max_visits, 0); + } + assert_eq!(clock.reclaim(&entries, 1, 0, 0), Ok(())); + assert_eq!(segments.free_count(), 3); + assert_eq!(segments.evicting.get(), 0); + assert_eq!(segments.state(SegmentId(3)), Ok(SegmentState::Sealed)); + assert_eq!(*entries.counts.borrow(), [0, 0, 0, 1]); + for image in &segments.snapshot()[..3] { + assert_eq!(image.state, SegmentState::Free); + assert_eq!(image.generation, Generation(2)); + } + } + } + + /// Freeze checks precede index side effects, and leases precede physical reuse. + #[test] + fn busy_lease_and_frozen_table_preserve_reclaim_side_effect_order() { + let segments = segments(2); + let held = segments.append(1024).unwrap(); + drop(segments.append(1024).unwrap()); + let clock = SegmentClock::new(segments.clone()); + let entries = Entries::new(vec![1, 1]); + let frozen = segments.freeze().unwrap(); + assert_eq!(clock.reclaim(&entries, 2, 64, 256), Err(Error::Busy)); + assert!(entries.calls.borrow().is_empty()); + drop(frozen); + clock.mark_read(SegmentId(1)).unwrap(); + assert_eq!(clock.reclaim(&entries, 2, 64, 256), Err(Error::Busy)); + assert_eq!(segments.state(SegmentId(0)), Ok(SegmentState::Evicting)); + assert_eq!(segments.free_count(), 1); + assert_eq!(clock.mark_read(SegmentId(0)), Ok(())); + assert_eq!(clock.mark_read(SegmentId(1)), Ok(())); + assert!(clock.recent.borrow().is_empty()); + drop(held); + clock.reclaim(&entries, 2, 64, 256).unwrap(); + assert_eq!(segments.free_count(), 2); + assert!(segments.lease(SegmentId(0), Generation(1)).is_err()); + } + + /// Skipping an open segment does not consume its future second chance. + #[test] + fn physical_sweep_preserves_recent_bit_on_ineligible_open_segment() { + let segments = segments(1); + drop(segments.append(512).unwrap()); + let clock = SegmentClock::new(segments.clone()); + let entries = Entries::new(vec![1]); + clock.mark_read(SegmentId(0)).unwrap(); + assert_eq!(clock.reclaim(&entries, 1, 64, 256), Err(Error::Busy)); + drop(segments.append(512).unwrap()); + assert_eq!(clock.reclaim(&entries, 1, 1, 256), Err(Error::Busy)); + assert!(entries.calls.borrow().is_empty()); + clock.reclaim(&entries, 1, 1, 256).unwrap(); + assert_eq!(*entries.calls.borrow(), [(SegmentId(0), 256)]); + } + + /// A no-space rollover creates a reclaimable tail without reserving bytes. + #[test] + fn no_space_rollover_seals_tail_for_reclaim_and_retry() { + for retain_lease in [false, true] { + let segments = segments(1); + let mut held = Some(segments.append(512).unwrap().0); + if !retain_lease { + drop(held.take()); + } + assert!(matches!(segments.append(1024), Err(Error::Busy))); + let image = &segments.snapshot()[0]; + assert_eq!(image.state, SegmentState::Sealed); + assert_eq!(image.used_bytes, 512); + assert_eq!(image.generation, Generation(1)); + let clock = SegmentClock::new(segments.clone()); + let entries = Entries::new(vec![0]); + if retain_lease { + assert_eq!(clock.reclaim(&entries, 1, 2, 0), Err(Error::Busy)); + assert_eq!(segments.state(SegmentId(0)), Ok(SegmentState::Evicting)); + drop(held.take()); + } + clock.reclaim(&entries, 1, 2, 0).unwrap(); + assert!(entries.calls.borrow().is_empty()); + let (lease, extent) = segments.append(1024).unwrap(); + assert_eq!(lease.id(), SegmentId(0)); + assert_eq!(lease.generation(), Generation(2)); + assert_eq!(extent.offset(), 0); + assert_eq!(extent.length(), 1024); + } + } + + /// Zero mapping work cannot start populated eviction but may recycle empty slots. + #[test] + fn zero_budget_does_not_start_populated_eviction_but_recycles_empty_candidates() { + let segments = segments(2); + drop(segments.append(1024).unwrap()); + drop(segments.append(1024).unwrap()); + let clock = SegmentClock::new(segments.clone()); + let entries = Entries::new(vec![1, 0]); + clock.reclaim(&entries, 1, 2, 0).unwrap(); + assert_eq!(segments.state(SegmentId(0)), Ok(SegmentState::Sealed)); + assert_eq!(segments.state(SegmentId(1)), Ok(SegmentState::Free)); + assert_eq!(*entries.counts.borrow(), [1, 0]); + assert!(entries.calls.borrow().is_empty()); + assert_eq!(clock.reclaim(&entries, 2, 64, 0), Err(Error::Busy)); + assert_eq!(segments.state(SegmentId(0)), Ok(SegmentState::Sealed)); + } + + /// A zero reserve does not inspect or change allocation and mapping state. + #[test] + fn zero_reserve_has_no_side_effects_even_when_unconfigured() { + let unconfigured = SegmentClock::new(Rc::new(Segments::new(1024))); + assert_eq!( + unconfigured.reclaim(&Entries::new(vec![]), 0, 64, 256), + Ok(()) + ); + let segments = segments(1); + drop(segments.append(1024).unwrap()); + let clock = SegmentClock::new(segments.clone()); + let entries = Entries::new(vec![1]); + clock.mark_read(SegmentId(0)).unwrap(); + assert_eq!(clock.reclaim(&entries, 0, 64, 256), Ok(())); + assert_eq!(segments.state(SegmentId(0)), Ok(SegmentState::Sealed)); + assert!(entries.calls.borrow().is_empty()); + assert!(clock.recent.borrow().contains(&SegmentId(0))); + } + + /// Scoring completes before mutation; mapping budgets, freeze, leases and + /// generation authority remain enforced even when the cheapest slot is busy. + #[test] + fn scored_eviction_preserves_fences_and_partial_progress() { + let segments = segments(3); + drop(segments.append(1024).unwrap()); + let held = segments.append(1024).unwrap().0; + drop(segments.append(1024).unwrap()); + let clock = SegmentClock::new(segments.clone()); + let entries = Entries::new(vec![1, 2, 1]); + let scores = [30, 0, 10]; + let frozen = segments.freeze().unwrap(); + assert_eq!( + clock.reclaim_scored(&entries, 1, 64, 1, |id| scores[id.0 as usize]), + Err(Error::Busy) + ); + assert!(entries.calls.borrow().is_empty()); + drop(frozen); + assert_eq!( + clock.reclaim_scored(&entries, 1, 64, 1, |id| { + assert_eq!(segments.state(id), Ok(SegmentState::Sealed)); + scores[id.0 as usize] + }), + Err(Error::Busy) + ); + assert_eq!(*entries.counts.borrow(), [1, 1, 1]); + assert_eq!(segments.state(SegmentId(1)), Ok(SegmentState::Evicting)); + assert!(segments.lease(SegmentId(1), Generation(1)).is_err()); + assert_eq!( + clock.reclaim_scored(&entries, 1, 64, 2, |id| scores[id.0 as usize]), + Err(Error::Busy) + ); + assert_eq!(*entries.counts.borrow(), [1, 0, 1]); + assert_eq!(segments.state(SegmentId(0)), Ok(SegmentState::Sealed)); + assert_eq!(segments.state(SegmentId(1)), Ok(SegmentState::Evicting)); + drop(held); + clock + .reclaim_scored(&entries, 1, 64, 0, |_| u64::MAX) + .unwrap(); + assert_eq!(segments.free_count(), 1); + assert_eq!(segments.snapshot()[1].generation, Generation(2)); + } + + /// A score callback never turns a soft preference into immunity or a full scan. + #[test] + fn scored_eviction_caps_candidates_and_evicts_maximum_scores() { + let segments = segments(65); + for _ in 0..65 { + drop(segments.append(1024).unwrap()); + } + let clock = SegmentClock::new(segments.clone()); + let entries = Entries::new(vec![1; 65]); + let mut visits = 0; + clock + .reclaim_scored(&entries, 1, usize::MAX, 1, |_| { + visits += 1; + u64::MAX + }) + .unwrap(); + assert_eq!(visits, 64); + assert_eq!(entries.calls.borrow().len(), 1); + assert_eq!(segments.free_count(), 1); + assert_eq!(clock.hand.get(), 64); + } + + /// Pending leased victims count toward reserve even outside the next sample. + #[test] + fn scored_eviction_never_overshoots_reserve_with_multiple_leased_victims() { + let segments = segments(3); + let mut leases: Vec<_> = (0..3) + .map(|_| Some(segments.append(1024).unwrap().0)) + .collect(); + let clock = SegmentClock::new(segments.clone()); + let entries = Entries::new(vec![1; 3]); + for _ in 0..6 { + assert_eq!( + clock.reclaim_scored(&entries, 1, 1, 256, |_| 0), + Err(Error::Busy) + ); + assert_eq!(*entries.counts.borrow(), [0, 1, 1]); + assert_eq!(segments.evicting.get(), 1); + } + drop(leases[0].take()); + clock.reclaim_scored(&entries, 1, 64, 256, |_| 0).unwrap(); + assert_eq!(segments.free_count(), 1); + assert_eq!(segments.evicting.get(), 0); + assert_eq!(*entries.counts.borrow(), [0, 1, 1]); + } + + /// Lowering scored demand still waits for pending leases and bounded visits. + #[test] + fn scored_pending_victims_drain_after_reserve_is_met() { + for max_visits in [1, 8] { + let segments = segments(4); + let mut leases: Vec<_> = (0..3) + .map(|_| Some(segments.append(1024).unwrap().0)) + .collect(); + drop(segments.append(1024).unwrap()); + let clock = SegmentClock::new(segments.clone()); + let entries = Entries::new(vec![1; 4]); + assert_eq!( + clock.reclaim_scored(&entries, 3, 3, 3, |_| 0), + Err(Error::Busy) + ); + assert_eq!(segments.evicting.get(), 3); + assert_eq!(*entries.counts.borrow(), [0, 0, 0, 1]); + + assert_eq!( + clock.reclaim_scored(&entries, 1, 1, 1, |_| 0), + Err(Error::Busy) + ); + drop(leases[0].take()); + assert_eq!( + clock.reclaim_scored(&entries, 1, max_visits, 0, |_| 0), + Err(Error::Busy) + ); + assert_eq!(segments.free_count(), 1); + assert_eq!(segments.evicting.get(), 2); + assert!(matches!( + segments.lease(SegmentId(1), Generation(1)), + Err(Error::Stale) + )); + + drop(leases); + let before = segments.snapshot(); + assert_eq!(clock.reclaim_scored(&entries, 0, 8, 8, |_| 0), Ok(())); + assert_eq!( + clock.reclaim_scored(&entries, 1, 0, 0, |_| 0), + Err(Error::Busy) + ); + assert_eq!(segments.snapshot(), before); + for _ in 0..4 { + let _ = clock.reclaim_scored(&entries, 1, max_visits, 0, |_| 0); + } + assert_eq!(clock.reclaim_scored(&entries, 1, 0, 0, |_| 0), Ok(())); + assert_eq!(segments.free_count(), 3); + assert_eq!(segments.evicting.get(), 0); + assert_eq!(segments.state(SegmentId(3)), Ok(SegmentState::Sealed)); + assert_eq!(*entries.counts.borrow(), [0, 0, 0, 1]); + assert_eq!(entries.calls.borrow().len(), 3); + for image in &segments.snapshot()[..3] { + assert_eq!(image.state, SegmentState::Free); + assert_eq!(image.generation, Generation(2)); + } + } + } + + /// Meeting a lower reserve does not hide unfinished index removal. + #[test] + fn scored_pending_mappings_drain_after_reserve_is_met() { + let segments = segments(3); + for _ in 0..3 { + drop(segments.append(1024).unwrap()); + } + let clock = SegmentClock::new(segments.clone()); + let entries = Entries::new(vec![0, 3, 1]); + assert_eq!( + clock.reclaim_scored(&entries, 2, 3, 1, |_| 0), + Err(Error::Busy) + ); + assert_eq!(segments.free_count(), 1); + assert_eq!(segments.evicting.get(), 1); + assert_eq!(*entries.counts.borrow(), [0, 2, 1]); + + let before = segments.snapshot(); + assert_eq!( + clock.reclaim_scored(&entries, 1, 3, 0, |_| 0), + Err(Error::Busy) + ); + assert_eq!(segments.snapshot(), before); + assert_eq!(*entries.counts.borrow(), [0, 2, 1]); + assert_eq!(entries.calls.borrow().len(), 1); + assert_eq!( + clock.reclaim_scored(&entries, 1, 3, 1, |_| 0), + Err(Error::Busy) + ); + assert_eq!(*entries.counts.borrow(), [0, 1, 1]); + assert_eq!(segments.state(SegmentId(1)), Ok(SegmentState::Evicting)); + assert_eq!(clock.reclaim_scored(&entries, 1, 3, 1, |_| 0), Ok(())); + assert_eq!(segments.free_count(), 2); + assert_eq!(segments.evicting.get(), 0); + assert_eq!(segments.snapshot()[1].generation, Generation(2)); + assert_eq!(segments.state(SegmentId(2)), Ok(SegmentState::Sealed)); + assert_eq!(*entries.counts.borrow(), [0, 0, 1]); + assert_eq!(*entries.calls.borrow(), [(SegmentId(1), 1); 3]); + } + + /// Over-reporting callbacks return configuration errors without unsafe reuse. + #[test] + fn removal_contract_violations_return_errors_without_panicking() { + /// An intentionally broken callback that over-reports every removal. + struct InvalidEntries; + + impl SegmentEntries for InvalidEntries { + /// Violate the caller contract to test error handling. + fn remove_bounded(&self, _: SegmentId, budget: usize) -> usize { + budget + 1 + } + + /// Keep the segment populated despite the invalid removal report. + fn is_empty(&self, _: SegmentId) -> bool { + false + } + } + let segments = segments(1); + drop(segments.append(1024).unwrap()); + let clock = SegmentClock::new(segments.clone()); + assert_eq!( + clock.reclaim_index(&InvalidEntries, 1, || false), + Err(Error::InvalidConfiguration) + ); + assert_eq!( + clock.reclaim(&InvalidEntries, 1, 1, 1), + Err(Error::InvalidConfiguration) + ); + assert_eq!(segments.state(SegmentId(0)), Ok(SegmentState::Evicting)); + assert_eq!(segments.free_count(), 0); + } + + /// Every clock operation observes the restore epoch before using old history. + #[test] + fn restore_resets_cursor_and_recent_reads_before_any_clock_operation() { + let segments = segments(2); + drop(segments.append(1024).unwrap()); + drop(segments.append(1024).unwrap()); + let clock = SegmentClock::new(segments.clone()); + let entries = Entries::new(vec![10, 10]); + assert_eq!(clock.reclaim_index(&entries, 1, || false), Err(Error::Busy)); + clock.mark_read(SegmentId(0)).unwrap(); + assert_eq!(clock.hand.get(), 1); + segments.restore(segments.snapshot()).unwrap(); + entries.calls.borrow_mut().clear(); + assert_eq!(clock.reclaim_index(&entries, 1, || false), Err(Error::Busy)); + assert_eq!(*entries.calls.borrow(), [(SegmentId(0), 1)]); + clock.mark_read(SegmentId(1)).unwrap(); + segments.restore(segments.snapshot()).unwrap(); + clock.mark_read(SegmentId(0)).unwrap(); + assert_eq!(clock.hand.get(), 0); + assert_eq!(*clock.recent.borrow(), HashSet::from([SegmentId(0)])); + segments.restore(segments.snapshot()).unwrap(); + clock.reclaim(&entries, 0, 0, 0).unwrap(); + assert!(clock.recent.borrow().is_empty()); + } +} diff --git a/cmd/racer-dataplane/alloc/src/slab.rs b/cmd/racer-dataplane/alloc/src/slab.rs new file mode 100644 index 000000000..73db5aab6 --- /dev/null +++ b/cmd/racer-dataplane/alloc/src/slab.rs @@ -0,0 +1,2556 @@ +//! Sparse file or device storage and completion-owned direct I/O. +//! +//! Startup is blocking and belongs outside the latency-sensitive worker path. +//! Capacity bounds logical addresses, not reserved physical space: writes can +//! still fail with ENOSPC. Recycling never truncates or erases disk bytes. +//! Parent directories must be trusted against hostile rename and unlink, even +//! though descriptor-relative traversal rejects symlinks and parent components. +use crate::segments::TableIdentity; +use crate::{ + AlignedBuffer, Alignment, Charge, Error, Extent, Result, SegmentGeometry, SegmentLease, + Segments, +}; +use std::{ + cell::{Cell, RefCell}, + collections::HashMap, + ffi::CString, + fs::File, + future::Future, + os::{ + fd::{AsRawFd, FromRawFd}, + unix::{ffi::OsStrExt, fs::MetadataExt}, + }, + path::{Component, Path, PathBuf}, + pin::Pin, + rc::Rc, + sync::Arc, + task::{Context, Poll, Waker}, +}; +use uring_runtime::{ + Budget, Operation, Scope, + reactor::{Reactor, descriptor::Descriptor}, +}; + +/// One logical segment's range on a caller-opened device. +/// Share one Arc per device within a worker to avoid per-segment descriptors. +#[derive(Clone, Debug)] +pub struct DevicePlacement { + /// Read/write O_DIRECT file, opened exclusively by the caller for real devices. + pub file: Arc, + + /// Physical start of this segment, in bytes. + pub offset: u64, +} + +/// Startup resources, before worker-local descriptor creation. +enum Backing { + File(PathBuf), + Devices { + placements: RefCell>, + geometry: SegmentGeometry, + }, +} + +/// Direct-I/O cache storage, not a durable storage transaction. +/// `open_configured` binds submissions to one segment table; read/write require +/// a successful binding, not merely an open file. +#[repr(align(64))] +pub struct Slab { + backing: Backing, + + capacity_bytes: u64, + + segment_bytes: u64, + + max_record_bytes: usize, + + opened: RefCell>, + + writes: Rc, + + idle_buffer: Rc>>>, +} + +impl Slab { + /// Describe a worker's file; validation and blocking I/O happen at startup. + pub fn new( + path: PathBuf, + capacity_bytes: u64, + segment_bytes: u64, + max_record_bytes: usize, + ) -> Self { + Self { + backing: Backing::File(path), + capacity_bytes, + segment_bytes, + max_record_bytes, + opened: RefCell::new(None), + writes: Rc::new(WriteState::default()), + idle_buffer: Rc::new(RefCell::new(None)), + } + } + + /// Describe device ranges in logical segment order, without creating, sizing, + /// or locking files. Bounds use BLKGETSIZE64 for block devices and file length + /// for regular direct files. The caller must supply suitable alignment and keep + /// exclusive ownership, file flags, and capacity unchanged while in use. + /// Overlap checks only compare the same inode or device identity in this slab. + /// Whole-disk, partition, and device-mapper aliases are not detected. The caller + /// must guarantee disjoint physical storage across aliases and different slabs. + /// Files stay owned until `open_configured` creates worker-local descriptors. + pub fn from_devices( + placements: Vec, + segment_bytes: u64, + max_record_bytes: usize, + alignment: Alignment, + ) -> Result { + Self::from_devices_with_capacity( + placements, + segment_bytes, + max_record_bytes, + alignment, + |file, metadata| { + if metadata.mode() & libc::S_IFMT == libc::S_IFBLK { + block_device_capacity(file.as_raw_fd()) + } else { + Ok(metadata.len()) + } + }, + ) + } + + /// Keep capacity probing injectable for tests without block-device access. + fn from_devices_with_capacity( + placements: Vec, + segment_bytes: u64, + max_record_bytes: usize, + alignment: Alignment, + mut capacity: impl FnMut(&File, &std::fs::Metadata) -> Result, + ) -> Result { + let capacity_bytes = (placements.len() as u64) + .checked_mul(segment_bytes) + .filter(|&size| size != 0 && size <= i64::MAX as u64) + .ok_or(Error::InvalidConfiguration)?; + let mut slab = Self::new( + PathBuf::new(), + capacity_bytes, + segment_bytes, + max_record_bytes, + ); + let geometry = slab.validate_layout(alignment, capacity_bytes)?; + let mut files = HashMap::new(); + let mut ranges = Vec::with_capacity(placements.len()); + for placement in &placements { + let end = placement + .offset + .checked_add(segment_bytes) + .filter(|&end| end <= i64::MAX as u64) + .ok_or(Error::InvalidConfiguration)?; + if !placement.offset.is_multiple_of(alignment.offset()) { + return Err(Error::InvalidConfiguration); + } + let key = Arc::as_ptr(&placement.file); + let (metadata, capacity_bytes) = match files.entry(key) { + std::collections::hash_map::Entry::Occupied(entry) => entry.into_mut(), + std::collections::hash_map::Entry::Vacant(entry) => { + // SAFETY: F_GETFL only inspects the live caller-owned descriptor. + let flags = unsafe { libc::fcntl(placement.file.as_raw_fd(), libc::F_GETFL) }; + if flags < 0 { + return Err(system_error("fcntl-getfl", std::io::Error::last_os_error())); + } + if flags & libc::O_DIRECT == 0 + || flags & libc::O_ACCMODE != libc::O_RDWR + || flags & libc::O_APPEND != 0 + { + return Err(Error::InvalidConfiguration); + } + let metadata = placement + .file + .metadata() + .map_err(|e| system_error("fstat", e))?; + if !matches!( + metadata.mode() & libc::S_IFMT, + libc::S_IFREG | libc::S_IFBLK + ) { + return Err(Error::InvalidConfiguration); + } + let capacity_bytes = capacity(&placement.file, &metadata)?; + entry.insert((metadata, capacity_bytes)) + } + }; + if end > *capacity_bytes { + return Err(Error::InvalidConfiguration); + } + let identity = if metadata.mode() & libc::S_IFMT == libc::S_IFBLK { + (libc::S_IFBLK, metadata.rdev(), 0) + } else { + (libc::S_IFREG, metadata.dev(), metadata.ino()) + }; + ranges.push((identity, placement.offset, end)); + } + ranges.sort_unstable(); + if ranges + .windows(2) + .any(|pair| pair[0].0 == pair[1].0 && pair[0].2 > pair[1].1) + { + return Err(Error::InvalidConfiguration); + } + slab.backing = Backing::Devices { + placements: RefCell::new(placements), + geometry, + }; + Ok(slab) + } + + /// Logical capacity, not a reservation of physical disk blocks. + pub fn capacity_bytes(&self) -> u64 { + self.capacity_bytes + } + + /// Size of each physical segment. + pub fn segment_bytes(&self) -> u64 { + self.segment_bytes + } + + /// Accepted writes whose completion guards have not yet been released. + pub fn writes_in_flight(&self) -> usize { + self.writes.count.get() + } + + /// Accounted bytes in the single idle buffer slot. + pub fn idle_bytes(&self) -> usize { + self.idle_buffer.borrow().as_ref().map_or(0, |b| b.len()) + } + + /// Release only idle memory, returning its size without touching live I/O. + pub fn reclaim_idle(&self) -> usize { + let idle = self.idle_buffer.borrow_mut().take(); + idle.map_or(0, |b| b.len()) + } + + /// Wait for accepted writes to release their runtime completion fences. + /// This does not call fsync/fdatasync and does NOT promise crash durability. + /// Drive the reactor concurrently; stop new writes first if quiescence is needed. + pub fn fence_writes(&self) -> Operation<'_, (), Error> { + Box::pin(FenceWaiter { + state: self.writes.clone(), + registration: None, + }) + } + + /// Discovered or caller-supplied direct-I/O requirements, or Unavailable before startup. + pub fn alignment(&self) -> Result { + self.opened + .borrow() + .as_ref() + .map(|o| o.geometry.alignment()) + .ok_or(Error::Unavailable) + } + + /// Physical slab geometry, including the full segment capacity. A bound + /// table may deliberately expose fewer segments than this physical count. + pub fn geometry(&self) -> Result { + self.opened + .borrow() + .as_ref() + .map(|o| o.geometry) + .ok_or(Error::Unavailable) + } + + /// Configure an empty table, or validate an already configured partial table, + /// and permanently bind this slab to that table's identity. + #[cfg(any(test, feature = "simulation"))] + pub fn configure_segments(&self, segments: &Segments) -> Result<()> { + self.bind(segments) + } + + /// Attach the table identity to this backing only after all dimensions agree. + fn bind(&self, segments: &Segments) -> Result<()> { + let mut opened = self.opened.borrow_mut(); + let slab = opened.as_mut().ok_or(Error::Unavailable)?; + let geometry = slab.geometry; + let identity = segments.table_identity(); + if let Some(bound) = slab.table.as_ref() + && !bound.matches(&identity) + { + return Err(Error::InvalidConfiguration); + } + if segments.segment_bytes() != geometry.segment_bytes() { + return Err(Error::InvalidConfiguration); + } + if !segments.is_configured() { + let count = usize::try_from(geometry.segment_count()) + .map_err(|_| Error::InvalidConfiguration)?; + segments.configure_table(geometry.slab_bytes(), count, geometry.alignment())?; + } + let configured = segments.geometry().ok_or(Error::InvalidConfiguration)?; + if configured.slab_bytes() != geometry.slab_bytes() + || configured.segment_bytes() != geometry.segment_bytes() + || configured.alignment() != geometry.alignment() + || configured.segment_count() > geometry.segment_count() + { + return Err(Error::InvalidConfiguration); + } + slab.table = Some(identity); + Ok(()) + } + + /// Blocking startup helper. Do not invoke on a latency-sensitive worker. + pub fn open_configured(&self, segments: &Segments) -> Result { + // Automatic tables need one slot per physical segment; partial tables do not. + if !segments.is_configured() + && self + .capacity_bytes + .checked_div(self.segment_bytes) + .is_none_or(|count| count > crate::MAX_SEGMENTS) + { + return Err(Error::InvalidConfiguration); + } + let alignment = self.open_file()?; + self.bind(segments)?; + Ok(alignment) + } + + /// Blocking startup I/O for geometry probing and buffer allocation. Read/write + /// return `Unavailable` until `configure_segments` succeeds. + /// Parent directories must be trusted against + /// rename/unlink by other users. Linux traversal rejects symlinks and `..`; + /// newly created directories are private. Existing files must be owned by the + /// effective user, regular, singly linked, and mode 0600. + #[cfg(any(test, feature = "simulation"))] + pub fn open_now(&self) -> Result { + self.open_file() + } + + /// Prepare backing descriptors without granting submission authority. + fn open_file(&self) -> Result { + self.open_file_with_lock_handoff(|_| {}) + } + + /// Open backing storage with a hook for testing a prior lock holder's changes. + fn open_file_with_lock_handoff(&self, before_lock: impl FnOnce(&File)) -> Result { + if let Ok(a) = self.alignment() { + return Ok(a); + } + let path = match &self.backing { + Backing::File(path) => path, + Backing::Devices { + placements, + geometry, + } => { + let mut files = HashMap::new(); + let mut opened = Vec::with_capacity(placements.borrow().len()); + for placement in placements.borrow().iter() { + let file = match files.entry(Arc::as_ptr(&placement.file)) { + std::collections::hash_map::Entry::Occupied(entry) => entry.into_mut(), + std::collections::hash_map::Entry::Vacant(entry) => { + let file = placement + .file + .try_clone() + .map_err(|e| system_error("dup", e))?; + entry.insert(Rc::new(Descriptor::from(file))) + } + }; + opened.push(OpenPlacement { + file: file.clone(), + offset: placement.offset, + }); + } + *self.opened.borrow_mut() = Some(OpenSlab { + backing: OpenBacking::Devices(opened), + geometry: *geometry, + table: None, + }); + *placements.borrow_mut() = Vec::new(); + return Ok(geometry.alignment()); + } + }; + if self.segment_bytes == 0 + || self.capacity_bytes == 0 + || self.max_record_bytes == 0 + || !self.capacity_bytes.is_multiple_of(self.segment_bytes) + || self.capacity_bytes > i64::MAX as u64 + { + return Err(Error::InvalidConfiguration); + } + #[cfg(feature = "simulation")] + if let Some(sim) = uring_runtime::reactor::simulation::Simulation::current() { + if let Some(parent) = path.parent().filter(|p| !p.as_os_str().is_empty()) { + sim.create_dir_all(parent) + .map_err(|e| system_error("mkdir", e))?; + } + let file = sim + .open( + None, + path, + libc::O_CREAT + | libc::O_RDWR + | libc::O_DIRECT + | libc::O_CLOEXEC + | libc::O_NOFOLLOW, + ) + .map_err(|e| direct_error("open", e))?; + let handle = file.as_sim().expect("simulation descriptor"); + handle.lock().map_err(lock_error)?; + let stat = handle.stat().map_err(|e| system_error("statx", e))?; + // The virtual filesystem has a fixed root owner, independent of host uid. + validate_file(stat.stx_mode as u32, stat.stx_uid, 0, stat.stx_nlink as u64)?; + let a = alignment_from_stat(&stat)?; + let geometry = self.validate_layout(a, stat.stx_size)?; + if stat.stx_size == 0 { + handle + .set_len(self.capacity_bytes) + .map_err(|e| system_error("ftruncate", e))?; + } + self.publish(file, geometry); + return Ok(a); + } + let file = open_private_file(path)?; + let metadata = file.metadata().map_err(|e| system_error("fstat", e))?; + // SAFETY: geteuid has no arguments or borrowed memory. + validate_file( + metadata.mode(), + metadata.uid(), + unsafe { libc::geteuid() }, + metadata.nlink(), + )?; + before_lock(&file); + // SAFETY: flock synchronously borrows this live descriptor. + if unsafe { libc::flock(file.as_raw_fd(), libc::LOCK_EX | libc::LOCK_NB) } != 0 { + return Err(lock_error(std::io::Error::last_os_error())); + } + // Recheck metadata under the lock: a prior holder may have changed it. + let metadata = file.metadata().map_err(|e| system_error("fstat", e))?; + // SAFETY: geteuid has no arguments or borrowed memory. + validate_file( + metadata.mode(), + metadata.uid(), + unsafe { libc::geteuid() }, + metadata.nlink(), + )?; + let size = metadata.len(); + // Delay O_DIRECT until after rejecting nonregular files (including FIFOs). + // SAFETY: fcntl synchronously borrows this live descriptor. + let flags = unsafe { libc::fcntl(file.as_raw_fd(), libc::F_GETFL) }; + if flags < 0 { + return Err(system_error("fcntl-getfl", std::io::Error::last_os_error())); + } + if unsafe { + libc::fcntl( + file.as_raw_fd(), + libc::F_SETFL, + (flags | libc::O_DIRECT) & !libc::O_NONBLOCK, + ) + } < 0 + { + return Err(direct_error( + "fcntl-direct", + std::io::Error::last_os_error(), + )); + } + let a = probe(&file)?; + let geometry = self.validate_layout(a, size)?; + if size == 0 { + file.set_len(self.capacity_bytes) + .map_err(|e| system_error("ftruncate", e))?; + } + self.publish(file.into(), geometry); + Ok(a) + } + + /// Publish geometry with its owning file, initially without I/O authority. + fn publish(&self, file: Descriptor, geometry: SegmentGeometry) { + *self.opened.borrow_mut() = Some(OpenSlab { + backing: OpenBacking::File(Rc::new(file)), + geometry, + table: None, + }); + } + + /// Check physical dimensions and padded record size without changing the file. + fn validate_layout(&self, a: Alignment, size: u64) -> Result { + if a.extent(0, self.max_record_bytes)?.length() as u64 > self.segment_bytes + || !self.segment_bytes.is_multiple_of(a.offset()) + || !self.segment_bytes.is_multiple_of(a.length() as u64) + || (size != 0 && size != self.capacity_bytes) + { + return Err(Error::InvalidConfiguration); + } + SegmentGeometry::new( + self.capacity_bytes, + self.segment_bytes, + self.capacity_bytes / self.segment_bytes, + a, + ) + .map_err(|_| Error::InvalidConfiguration) + } + + /// Borrow-check a request before moving its resources into a submission. + fn submission( + &self, + extent: Extent, + buffer: &AlignedBuffer, + lease: &SegmentLease, + ) -> Result<(Rc, Extent)> { + let opened = self.opened.borrow(); + let slab = opened.as_ref().ok_or(Error::Unavailable)?; + let table = slab.table.as_ref().ok_or(Error::Unavailable)?; + slab.geometry.alignment().check(extent, buffer)?; + if !table.matches(&lease.table_identity()) { + return Err(Error::Stale); + } + let geometry = lease.geometry(); + if geometry.slab_bytes() != slab.geometry.slab_bytes() + || geometry.segment_bytes() != slab.geometry.segment_bytes() + || geometry.alignment() != slab.geometry.alignment() + { + return Err(Error::Corrupt); + } + lease.validate_extent(&extent)?; + let start = lease + .id() + .0 + .checked_mul(self.segment_bytes) + .ok_or(Error::Corrupt)?; + let end = extent + .offset() + .checked_add(extent.length() as u64) + .ok_or(Error::Corrupt)?; + if extent.offset() < start + || end + > start + .checked_add(self.segment_bytes) + .ok_or(Error::Corrupt)? + || end > self.capacity_bytes + { + return Err(Error::Corrupt); + } + match &slab.backing { + OpenBacking::File(file) => Ok((file.clone(), extent)), + OpenBacking::Devices(placements) => { + let position = usize::try_from(lease.id().0).map_err(|_| Error::Corrupt)?; + let placement = placements.get(position).ok_or(Error::Corrupt)?; + let offset = placement + .offset + .checked_add(extent.offset() - start) + .ok_or(Error::Corrupt)?; + let physical = Extent::new(offset, extent.length())?; + let end = offset + .checked_add(extent.length() as u64) + .ok_or(Error::Corrupt)?; + if end + > placement + .offset + .checked_add(self.segment_bytes) + .ok_or(Error::Corrupt)? + || end > i64::MAX as u64 + { + return Err(Error::Corrupt); + } + slab.geometry.alignment().check(physical, buffer)?; + Ok((placement.file.clone(), physical)) + } + } + } + + /// Consume a checked request so its descriptor, buffer, and lease travel together. + fn prepare( + &self, + extent: Extent, + buffer: AlignedBuffer, + lease: SegmentLease, + ) -> Result> { + let (file, extent) = self.submission(extent, &buffer, &lease)?; + Ok(Submission { + file, + extent, + buffer, + lease, + }) + } + + /// Read exactly one checked extent, retaining its buffer and lease until completion. + /// Dropping the waiting future does not release kernel-owned resources. + /// Neither a lease nor a successful read proves initialization in this + /// generation. Append only reserves space; recycled bytes may come from an + /// earlier generation or a different cache. Before exposing a valid record, + /// the caller must check its integrity, authentication, and cache identity. + pub fn read<'a, S: Scope, B: Budget>( + &'a self, + reactor: &'a Reactor, + extent: Extent, + buffer: AlignedBuffer, + lease: SegmentLease, + scope: &'a S, + ) -> Operation<'a, AlignedBuffer, S::Error> + where + S::Error: From, + { + Box::pin(async move { + self.prepare(extent, buffer, lease)? + .read(reactor, scope) + .await + }) + } + + /// Write exactly one checked extent with completion-owned accounting and lease. + /// Failed writes do not roll back the space reserved by append. + /// Any lease permits writes within its captured prefix, including a lease from + /// [`Segments::lease`]; it does not grant exclusive record ownership. The trusted + /// caller must write only reserved extents it owns, never overwrite published or + /// readable records, and publish a mapping only after the write succeeds. + pub fn write<'a, S: Scope, B: Budget>( + &'a self, + reactor: &'a Reactor, + extent: Extent, + buffer: AlignedBuffer, + lease: SegmentLease, + scope: &'a S, + ) -> Operation<'a, AlignedBuffer, S::Error> + where + S::Error: From, + { + Box::pin(async move { + let submission = self.prepare(extent, buffer, lease)?; + let fence = self.writes.acquire()?; + submission.write(reactor, scope, fence).await + }) + } + + /// Reuse one exact-size idle buffer. A size mismatch releases the old idle + /// buffer and its charge, even if the replacement allocation fails. + pub fn allocate(&self, length: usize, charge: C) -> Result> { + let alignment = self.alignment()?; + if !charge.covers(length) { + return Err(Error::InvalidConfiguration); + } + let idle = self.idle_buffer.borrow_mut().take(); + if let Some(mut buffer) = idle + && buffer.len() == length + { + buffer.rebind(charge)?; + return Ok(buffer.pooled(&self.idle_buffer)); + } + Ok(alignment + .allocate(length, charge)? + .pooled(&self.idle_buffer)) + } + + /// Replace file-mode backing for fault tests; device mode is rejected. + #[cfg(feature = "simulation")] + #[doc(hidden)] + pub fn replace_file_for_test(&self, file: File) -> Result<()> { + self.replace_descriptor_for_test(file.into()) + } + + /// Accept a virtual descriptor for file-mode fault tests; geometry stays unchanged. + #[cfg(feature = "simulation")] + #[doc(hidden)] + pub fn replace_descriptor_for_test(&self, file: Descriptor) -> Result<()> { + if self.writes_in_flight() != 0 { + return Err(Error::Busy); + } + let mut opened = self.opened.borrow_mut(); + let slab = opened.as_mut().ok_or(Error::Unavailable)?; + let OpenBacking::File(current) = &mut slab.backing else { + return Err(Error::InvalidConfiguration); + }; + *current = Rc::new(file); + Ok(()) + } +} + +/// Backing and geometry own their binding; a closed slab cannot retain authority. +struct OpenSlab { + backing: OpenBacking, + + geometry: SegmentGeometry, + + table: Option, +} + +/// Runtime owners shared by all submissions to a backing file. +enum OpenBacking { + File(Rc), + Devices(Vec), +} + +/// One logical segment's worker-local destination. +struct OpenPlacement { + file: Rc, + offset: u64, +} + +/// A validated transfer owns every resource needed to keep kernel access safe. +/// Only Slab::prepare constructs this capability, and submission consumes it. +struct Submission { + file: Rc, + + extent: Extent, + + buffer: AlignedBuffer, + + lease: SegmentLease, +} + +impl Submission { + /// Move the complete read capability into the reactor's completion ownership. + async fn read( + self, + reactor: &Reactor, + scope: &S, + ) -> std::result::Result, S::Error> + where + S::Error: From, + { + let completion = reactor + .read_at( + self.file, + self.extent.offset(), + self.buffer, + self.lease, + scope, + ) + .await?; + Self::complete(self.extent, completion.bytes, completion.buffer) + } + + /// Move the write capability and its counter guard into completion ownership. + async fn write( + self, + reactor: &Reactor, + scope: &S, + fence: WriteFence, + ) -> std::result::Result, S::Error> + where + S::Error: From, + { + let completion = reactor + .write_at( + self.file, + self.extent.offset(), + self.buffer, + (self.lease, fence), + scope, + ) + .await?; + Self::complete(self.extent, completion.bytes, completion.buffer) + } + + /// Return storage only after an exact-length completion; short I/O drops it. + fn complete>( + extent: Extent, + bytes: usize, + buffer: AlignedBuffer, + ) -> std::result::Result, E> { + if bytes != extent.length() { + return Err(Error::Io.into()); + } + Ok(buffer) + } +} + +/// Worker-local write counters and fence registrations, independent of slab lifetime. +#[derive(Default)] +struct WriteState { + count: Cell, + + waiters: RefCell>>>>, +} + +impl WriteState { + /// Increment before constructing the sole guard responsible for decrementing. + fn acquire(self: &Rc) -> Result { + self.count + .set(self.count.get().checked_add(1).ok_or(Error::Busy)?); + Ok(WriteFence(self.clone())) + } +} + +/// Cancel-safe registration waiting for all currently accepted writes to finish. +struct FenceWaiter { + state: Rc, + + registration: Option>>>, +} + +impl Future for FenceWaiter { + type Output = Result<()>; + + /// Refresh the task's registration without busy-waking the worker. + fn poll(mut self: Pin<&mut Self>, cx: &mut Context<'_>) -> Poll { + loop { + if self.state.count.get() == 0 { + self.unregister(); + if self.state.count.get() == 0 { + return Poll::Ready(Ok(())); + } + continue; + } + let waker = Rc::new(cx.waker().clone()); + if self.state.count.get() == 0 { + drop(waker); + self.unregister(); + if self.state.count.get() == 0 { + return Poll::Ready(Ok(())); + } + continue; + } + if let Some(registration) = self.registration.clone() { + let old = registration.replace(waker); + drop(old); + if self.state.count.get() == 0 { + drop(registration); + self.unregister(); + if self.state.count.get() == 0 { + return Poll::Ready(Ok(())); + } + continue; + } + let mut waiters = self.state.waiters.borrow_mut(); + if !waiters.iter().any(|w| Rc::ptr_eq(w, ®istration)) { + waiters.push(registration); + } + } else { + let registration = Rc::new(RefCell::new(waker)); + self.state.waiters.borrow_mut().push(registration.clone()); + self.registration = Some(registration); + } + return Poll::Pending; + } + } +} + +impl FenceWaiter { + /// Remove only this waiter's registration, including after cancellation. + fn unregister(&mut self) { + if let Some(registration) = self.registration.take() { + self.state + .waiters + .borrow_mut() + .retain(|w| !Rc::ptr_eq(w, ®istration)); + } + } +} + +impl Drop for FenceWaiter { + /// A canceled wait must not retain its task's waker. + fn drop(&mut self) { + self.unregister(); + } +} + +/// Unique write-count decrement authority retained by the reactor completion. +struct WriteFence(Rc); + +impl Drop for WriteFence { + /// Release the count and wake sleepers outside all registration borrows. + fn drop(&mut self) { + let remaining = self.0.count.get() - 1; + self.0.count.set(remaining); + if remaining == 0 { + let waiters = std::mem::take(&mut *self.0.waiters.borrow_mut()); + for waiter in waiters { + // Snapshot ownership without invoking the raw waker's clone callback. + let waker = waiter.borrow().clone(); + waker.wake_by_ref(); + } + } + } +} + +/// Read the block device's byte capacity without changing its contents. +fn block_device_capacity(fd: i32) -> Result { + let mut capacity = 0u64; + // Linux fs.h encodes BLKGETSIZE64 with size_t, but the output is always u64. + let request = libc::_IOR::(0x12, 114); + // SAFETY: BLKGETSIZE64 writes one u64 to this live output pointer. Invalid + // descriptors are rejected by the kernel without accessing device storage. + if unsafe { libc::ioctl(fd, request, &mut capacity) } != 0 { + return Err(system_error( + "ioctl-blkgetsize64", + std::io::Error::last_os_error(), + )); + } + Ok(capacity) +} + +/// Preserve synchronous operating-system diagnostic context. +fn system_error(operation: &'static str, error: std::io::Error) -> Error { + Error::SystemIo { + operation, + errno: error.raw_os_error(), + } +} + +/// Distinguish an already locked file from a failed locking syscall. +fn lock_error(error: std::io::Error) -> Error { + if error.raw_os_error() == Some(libc::EWOULDBLOCK) { + Error::Unavailable + } else { + system_error("flock", error) + } +} + +/// Classify explicit direct-I/O capability denials without hiding other failures. +fn direct_error(operation: &'static str, error: std::io::Error) -> Error { + match error.raw_os_error() { + Some(libc::EINVAL | libc::EOPNOTSUPP | libc::ENOSYS) => Error::Unsupported, + _ => system_error(operation, error), + } +} + +/// Discover direct-I/O requirements for an owned file descriptor. +fn probe(file: &File) -> Result { + probe_fd(file.as_raw_fd()) +} + +/// Ask Linux for descriptor-specific alignment without assuming a page size. +fn probe_fd(fd: i32) -> Result { + // SAFETY: initialized statx output and valid empty C path. Invalid FDs are + // rejected by the kernel without accessing user memory through the FD. + let mut stat: libc::statx = unsafe { std::mem::zeroed() }; + if unsafe { + libc::statx( + fd, + c"".as_ptr(), + libc::AT_EMPTY_PATH, + libc::STATX_DIOALIGN, + &mut stat, + ) + } != 0 + { + let error = std::io::Error::last_os_error(); + return Err(match error.raw_os_error() { + Some(libc::ENOSYS | libc::EOPNOTSUPP) => Error::Unsupported, + _ => system_error("statx", error), + }); + } + alignment_from_stat(&stat) +} + +/// Reject missing alignment capability or invalid values returned by statx. +fn alignment_from_stat(stat: &libc::statx) -> Result { + if stat.stx_mask & libc::STATX_DIOALIGN == 0 { + return Err(Error::Unsupported); + } + Alignment::new( + stat.stx_dio_mem_align as usize, + stat.stx_dio_offset_align as u64, + stat.stx_dio_offset_align as usize, + ) +} + +/// Require a private regular file with one link and the expected owner. +fn validate_file(mode: u32, owner: u32, expected_owner: u32, links: u64) -> Result<()> { + if mode & libc::S_IFMT != libc::S_IFREG + || mode & 0o7777 != 0o600 + || owner != expected_owner + || links != 1 + { + return Err(Error::InvalidConfiguration); + } + Ok(()) +} + +/// Resolve each directory relative to its already-open predecessor. Unlike +/// create_dir_all plus open, this never follows an intermediate symlink. +fn open_private_file(path: &Path) -> Result { + let name = path.file_name().ok_or(Error::InvalidConfiguration)?; + if path.components().any(|c| matches!(c, Component::ParentDir)) { + return Err(Error::InvalidConfiguration); + } + let anchor = if path.is_absolute() { c"/" } else { c"." }; + // SAFETY: valid C path, no borrowed memory retained by open. + let fd = unsafe { + libc::open( + anchor.as_ptr(), + libc::O_PATH | libc::O_DIRECTORY | libc::O_CLOEXEC, + ) + }; + if fd < 0 { + return Err(system_error("open-parent", std::io::Error::last_os_error())); + } + // SAFETY: open returned a new, uniquely owned descriptor. + let mut parent = unsafe { File::from_raw_fd(fd) }; + for component in path.parent().unwrap_or(Path::new("")).components() { + let Component::Normal(component) = component else { + continue; + }; + let component = + CString::new(component.as_bytes()).map_err(|_| Error::InvalidConfiguration)?; + // SAFETY: parent FD and C component remain valid throughout each syscall. + let flags = libc::O_PATH | libc::O_DIRECTORY | libc::O_CLOEXEC | libc::O_NOFOLLOW; + let mut next = unsafe { libc::openat(parent.as_raw_fd(), component.as_ptr(), flags) }; + if next < 0 && std::io::Error::last_os_error().raw_os_error() == Some(libc::ENOENT) { + if unsafe { libc::mkdirat(parent.as_raw_fd(), component.as_ptr(), 0o700) } != 0 + && std::io::Error::last_os_error().raw_os_error() != Some(libc::EEXIST) + { + return Err(system_error("mkdirat", std::io::Error::last_os_error())); + } + next = unsafe { libc::openat(parent.as_raw_fd(), component.as_ptr(), flags) }; + } + if next < 0 { + return Err(system_error("open-parent", std::io::Error::last_os_error())); + } + // SAFETY: successful openat returned a uniquely owned descriptor. + parent = unsafe { File::from_raw_fd(next) }; + } + let name = CString::new(name.as_bytes()).map_err(|_| Error::InvalidConfiguration)?; + // O_NONBLOCK prevents a malicious FIFO from hanging startup before fstat. + // SAFETY: live directory FD and NUL-terminated name; mode supplied for O_CREAT. + let fd = unsafe { + libc::openat( + parent.as_raw_fd(), + name.as_ptr(), + libc::O_RDWR | libc::O_CREAT | libc::O_CLOEXEC | libc::O_NOFOLLOW | libc::O_NONBLOCK, + 0o600, + ) + }; + if fd < 0 { + return Err(system_error("open", std::io::Error::last_os_error())); + } + // SAFETY: successful openat returned a uniquely owned descriptor. + Ok(unsafe { File::from_raw_fd(fd) }) +} + +/// Private storage invariants and synchronous Linux file validation. +#[cfg(test)] +mod tests { + use super::*; + use crate::{Generation, SegmentId}; + + /// Counts retained bytes independently of buffer pooling. + struct CountingCharge { + used: Rc>, + + bytes: usize, + } + + impl CountingCharge { + /// Admit a fixed number of bytes. + fn new(used: &Rc>, bytes: usize) -> Self { + used.set(used.get() + bytes); + Self { + used: used.clone(), + bytes, + } + } + } + + impl Charge for CountingCharge { + /// Cover only bytes actually admitted by this guard. + fn covers(&self, bytes: usize) -> bool { + self.bytes >= bytes + } + } + + impl Drop for CountingCharge { + /// Release accounting exactly once on destruction. + fn drop(&mut self) { + self.used.set(self.used.get() - self.bytes); + } + } + + /// Owns and cleans up an isolated directory inside the project build tree. + struct Directory(PathBuf); + + impl Directory { + /// Create a per-process, per-test directory without touching host temp paths. + fn new() -> Self { + static NEXT: std::sync::atomic::AtomicUsize = std::sync::atomic::AtomicUsize::new(0); + let id = NEXT.fetch_add(1, std::sync::atomic::Ordering::Relaxed); + let path = PathBuf::from(env!("CARGO_MANIFEST_DIR")) + .join("target") + .join(format!("page-alloc-test-{}-{id}", std::process::id())); + std::fs::create_dir_all(&path).unwrap(); + Self(path) + } + } + + impl Drop for Directory { + /// Remove only this test's owned directory. + fn drop(&mut self) { + let _ = std::fs::remove_dir_all(&self.0); + } + } + + /// Skip only explicit unsupported-filesystem results unless real I/O is required. + fn real_alignment(result: Result) -> Option { + real_io_result( + result, + std::env::var("PAGE_ALLOC_REQUIRE_REAL_IO").as_deref() == Ok("1"), + ) + } + + /// Apply the capability policy without changing the process environment in tests. + fn real_io_result(result: Result, required: bool) -> Option { + match result { + Ok(value) => Some(value), + Err(Error::Unsupported) => { + assert!( + !required, + "PAGE_ALLOC_REQUIRE_REAL_IO=1 forbids capability skips: filesystem does not support direct I/O geometry" + ); + eprintln!("SKIP real slab test: filesystem does not support direct I/O geometry"); + None + } + Err(error) => panic!("unexpected real slab startup failure: {error}"), + } + } + + /// Successful setup is preserved in both optional and required modes. + #[test] + fn real_io_success_is_preserved() { + for required in [false, true] { + assert_eq!(real_io_result(Ok(42), required), Some(42)); + } + } + + /// Unsupported direct opens skip only when the real-I/O gate is optional. + #[test] + fn real_io_unsupported_open_obeys_required_policy() { + for errno in [libc::EINVAL, libc::EOPNOTSUPP, libc::ENOSYS] { + let error = direct_error("open", std::io::Error::from_raw_os_error(errno)); + assert_eq!(real_io_result::<()>(Err(error), false), None); + let panic = + std::panic::catch_unwind(|| real_io_result::<()>(Err(error), true)).unwrap_err(); + let message = panic + .downcast_ref::() + .map(String::as_str) + .or_else(|| panic.downcast_ref::<&str>().copied()) + .unwrap(); + assert!(message.contains("PAGE_ALLOC_REQUIRE_REAL_IO=1 forbids capability skips")); + } + } + + /// Other open failures, including errors without errno, must never skip. + #[test] + fn real_io_unexpected_open_errors_fail_in_both_modes() { + for errno in [ + Some(libc::EACCES), + Some(libc::EIO), + Some(libc::EEXIST), + None, + ] { + let error = direct_error( + "open", + errno + .map(std::io::Error::from_raw_os_error) + .unwrap_or_else(|| std::io::Error::other("synthetic")), + ); + assert_eq!( + error, + Error::SystemIo { + operation: "open", + errno + } + ); + for required in [false, true] { + assert!( + std::panic::catch_unwind(|| real_io_result::<()>(Err(error), required)) + .is_err() + ); + } + } + } + + /// Pooling preserves the primary charge and zeroes bytes before reuse. + #[test] + fn aligned_pool_reuses_only_fenced_zeroed_admitted_storage() { + let used = Rc::new(Cell::new(0)); + let pool = Rc::new(RefCell::new(None)); + let alignment = Alignment::new(512, 512, 512).unwrap(); + let charge = || CountingCharge::new(&used, 512); + let mut buffer = alignment.allocate(512, charge()).unwrap().pooled(&pool); + let pointer = buffer.bytes().unwrap().as_ptr(); + buffer.bytes_mut().unwrap().fill(42); + let extra = Rc::new(charge()); + let weak = Rc::downgrade(&extra); + buffer.retain(extra); + assert_eq!(used.get(), 1024); + drop(buffer); + assert!(weak.upgrade().is_none()); + assert_eq!(used.get(), 512); + let mut reused = pool.borrow_mut().take().unwrap(); + assert_eq!(reused.bytes().unwrap().as_ptr(), pointer); + assert!(reused.bytes().unwrap().iter().all(|b| *b == 0)); + reused.rebind(charge()).unwrap(); + assert_eq!(used.get(), 512); + drop(reused.pooled(&pool)); + drop(pool); + assert_eq!(used.get(), 0); + } + + /// Geometry obeys the discovered units, not assumed memory-page dimensions. + #[test] + fn geometry_rounds_without_assuming_page_size() { + let a = Alignment::new(512, 512, 1024).unwrap(); + assert_eq!(a.extent(512, 1025).unwrap().length(), 2048); + assert!(a.extent(1, 1).is_err()); + assert!(a.extent(0, usize::MAX).is_err()); + assert!(Alignment::new(3, 512, 512).is_err()); + assert!(Extent::new(u64::MAX, 1).is_err()); + assert!(Extent::new(0, 0).is_err()); + assert!(Alignment::new(512, 0, 512).is_err()); + assert!(Alignment::new(512, 512, 0).is_err()); + let a = Alignment::new(512, 768, 512).unwrap(); + assert_eq!(a.extent(0, 513).unwrap().length(), 1536); + let b = a.allocate(512, ()).unwrap(); + assert!(!b.is_empty()); + assert_eq!( + a.check(Extent::new(1, 512).unwrap(), &b), + Err(Error::InvalidConfiguration) + ); + assert_eq!( + a.check(Extent::new(0, 1024).unwrap(), &b), + Err(Error::InvalidConfiguration) + ); + } + + /// Failed admission and borrowed return slots cannot leak accounted bytes. + #[test] + fn invalid_charge_is_released_and_pool_borrow_does_not_panic() { + let used = Rc::new(Cell::new(0)); + let a = Alignment::new(512, 512, 512).unwrap(); + assert!(matches!( + a.allocate(512, CountingCharge::new(&used, 511)), + Err(Error::InvalidConfiguration) + )); + assert_eq!(used.get(), 0); + let pool = Rc::new(RefCell::new(None)); + let b = a + .allocate(512, CountingCharge::new(&used, 512)) + .unwrap() + .pooled(&pool); + let borrow = pool.borrow_mut(); + drop(b); + assert_eq!(used.get(), 0); + drop(borrow); + assert!(pool.borrow().is_none()); + } + + /// Open direct files without using the slab's private-file startup path. + fn device_file(path: &Path) -> Option> { + use std::os::unix::fs::OpenOptionsExt; + let file = std::fs::OpenOptions::new() + .read(true) + .write(true) + .create_new(true) + .custom_flags(libc::O_DIRECT) + .open(path) + .map_err(|error| direct_error("open", error)); + let file = real_alignment(file)?; + file.set_len(16384).unwrap(); + Some(Arc::new(file)) + } + + /// Queried capacities bound every placement, including cached-file placements. + #[test] + fn device_capacity_bounds_are_checked_before_startup() { + let directory = Directory::new(); + let Some(file) = device_file(&directory.0.join("capacity")) else { + return; + }; + let a = Alignment::new(512, 512, 512).unwrap(); + for (capacity, offsets, valid) in [ + (0, vec![0], false), + (4095, vec![0], false), + (4096, vec![0], true), + (8191, vec![4096], false), + (8192, vec![4096], true), + (8192, vec![8192], false), + (8192, vec![0, 4096], true), + (8191, vec![0, 4096], false), + (u64::MAX, vec![0, 4096], true), + ] { + let mut queries = 0; + let result = Slab::<()>::from_devices_with_capacity( + offsets + .into_iter() + .map(|offset| DevicePlacement { + file: file.clone(), + offset, + }) + .collect(), + 4096, + 512, + a, + |_, _| { + queries += 1; + Ok(capacity) + }, + ); + if valid { + assert!(result.is_ok(), "capacity {capacity}: {:?}", result.err()); + } else { + assert!(matches!(result, Err(Error::InvalidConfiguration))); + } + assert_eq!(queries, 1); + assert_eq!(file.metadata().unwrap().len(), 16384); + } + } + + /// Failed capacity queries preserve syscall context rather than allowing a slab. + #[test] + fn device_capacity_errors_keep_errno() { + assert_eq!( + block_device_capacity(-1), + Err(Error::SystemIo { + operation: "ioctl-blkgetsize64", + errno: Some(libc::EBADF), + }) + ); + let directory = Directory::new(); + let Some(file) = device_file(&directory.0.join("capacity-error")) else { + return; + }; + assert_eq!( + block_device_capacity(file.as_raw_fd()), + Err(Error::SystemIo { + operation: "ioctl-blkgetsize64", + errno: Some(libc::ENOTTY), + }) + ); + for errno in [libc::EIO, libc::EACCES, libc::ENOTTY] { + let error = system_error( + "ioctl-blkgetsize64", + std::io::Error::from_raw_os_error(errno), + ); + let result = Slab::<()>::from_devices_with_capacity( + vec![DevicePlacement { + file: file.clone(), + offset: 0, + }], + 4096, + 512, + Alignment::new(512, 512, 512).unwrap(), + |_, _| Err(error), + ); + assert!(matches!(result, Err(actual) if actual == error)); + } + } + + /// Device layouts reject same-identity overlaps, misalignment, overflow, and invalid flags. + #[test] + fn device_layout_validation_is_read_only() { + use std::os::unix::fs::OpenOptionsExt; + let directory = Directory::new(); + let Some(file) = device_file(&directory.0.join("device")) else { + return; + }; + let Some(host_alignment) = real_alignment(probe(&file)) else { + return; + }; + for a in [ + host_alignment, + Alignment::new(4096, 4096, 4096).unwrap(), + Alignment::new(8192, 8192, 8192).unwrap(), + Alignment::new(4096, 768, 512).unwrap(), + ] { + let unit = a.extent(0, 1).unwrap().length(); + let segment_bytes = 2 * unit as u64; + let file_bytes = 4 * segment_bytes; + file.set_len(file_bytes).unwrap(); + let build = |offsets: &[u64], segment, record| { + Slab::<()>::from_devices( + offsets + .iter() + .map(|&offset| DevicePlacement { + file: file.clone(), + offset, + }) + .collect(), + segment, + record, + a, + ) + }; + for (offsets, segment, record) in [ + (vec![], segment_bytes, unit), + (vec![0], 0, unit), + (vec![0], segment_bytes, 0), + (vec![0], segment_bytes + 1, unit), + (vec![0], segment_bytes, segment_bytes as usize + 1), + (vec![0, segment_bytes], u64::MAX, unit), + (vec![0], i64::MAX as u64 + 1, unit), + (vec![1], segment_bytes, unit), + (vec![file_bytes], segment_bytes, unit), + (vec![u64::MAX], segment_bytes, unit), + ( + vec![i64::MAX as u64 - segment_bytes + 1], + segment_bytes, + unit, + ), + ] { + assert!(matches!( + build(&offsets, segment, record), + Err(Error::InvalidConfiguration) + )); + } + let duplicate = Arc::new(file.try_clone().unwrap()); + assert!(!Arc::ptr_eq(&file, &duplicate)); + for second in [file.clone(), duplicate] { + let placement = |offset| DevicePlacement { + file: second.clone(), + offset, + }; + for offset in [0, unit as u64, segment_bytes] { + assert!( + Slab::<()>::from_devices(vec![placement(offset)], segment_bytes, unit, a) + .is_ok() + ); + } + assert!( + Slab::<()>::from_devices( + vec![ + DevicePlacement { + file: file.clone(), + offset: 0 + }, + placement(segment_bytes) + ], + segment_bytes, + unit, + a + ) + .is_ok() + ); + for offset in [0, unit as u64] { + assert!(matches!( + Slab::<()>::from_devices( + vec![ + DevicePlacement { + file: file.clone(), + offset: 0 + }, + placement(offset), + ], + segment_bytes, + unit, + a + ), + Err(Error::InvalidConfiguration) + )); + } + } + for flags in [0, libc::O_DIRECT | libc::O_APPEND] { + let invalid = std::fs::OpenOptions::new() + .read(true) + .write(true) + .custom_flags(flags) + .open(directory.0.join("device")) + .unwrap(); + assert!(matches!( + Slab::<()>::from_devices( + vec![DevicePlacement { + file: Arc::new(invalid), + offset: 0, + }], + segment_bytes, + unit, + a + ), + Err(Error::InvalidConfiguration) + )); + } + let readonly = std::fs::OpenOptions::new() + .read(true) + .custom_flags(libc::O_DIRECT) + .open(directory.0.join("device")) + .unwrap(); + assert!(matches!( + Slab::<()>::from_devices( + vec![DevicePlacement { + file: Arc::new(readonly), + offset: 0, + }], + segment_bytes, + unit, + a + ), + Err(Error::InvalidConfiguration) + )); + let slab = build(&[2 * segment_bytes, segment_bytes], segment_bytes, unit).unwrap(); + assert_eq!(slab.capacity_bytes(), 2 * segment_bytes); + assert_eq!( + slab.open_configured(&Segments::new(2 * segment_bytes)), + Err(Error::InvalidConfiguration) + ); + let segments = Segments::new(segment_bytes); + assert_eq!(slab.open_configured(&segments), Ok(a)); + assert_eq!(slab.open_configured(&segments), Ok(a)); + assert_eq!(file.metadata().unwrap().len(), file_bytes); + let independent = File::open(directory.0.join("device")).unwrap(); + // SAFETY: this live independent descriptor tests that startup took no flock. + assert_eq!( + unsafe { libc::flock(independent.as_raw_fd(), libc::LOCK_EX | libc::LOCK_NB) }, + 0 + ); + let opened = slab.opened.borrow(); + let OpenBacking::Devices(placements) = &opened.as_ref().unwrap().backing else { + unreachable!() + }; + assert!(Rc::ptr_eq(&placements[0].file, &placements[1].file)); + let Backing::Devices { placements, .. } = &slab.backing else { + unreachable!() + }; + assert!(placements.borrow().is_empty()); + } + } + + /// Translation preserves logical authority and rechecks physical bounds and alignment. + #[test] + fn device_submission_checks_logical_and_physical_extents() { + let directory = Directory::new(); + let Some(file) = device_file(&directory.0.join("device")) else { + return; + }; + let Some(a) = real_alignment(probe(&file)) else { + return; + }; + let slab = Slab::<()>::from_devices( + vec![ + DevicePlacement { + file: file.clone(), + offset: 8192, + }, + DevicePlacement { file, offset: 0 }, + ], + 4096, + 4096, + a, + ) + .unwrap(); + let segments = Segments::new(4096); + let _ = slab.open_configured(&segments).unwrap(); + drop(segments.append(4096).unwrap()); + let (lease, extent) = segments.append(4096).unwrap(); + let buffer = slab.allocate(4096, ()).unwrap(); + let (_, physical) = slab.submission(extent, &buffer, &lease).unwrap(); + assert_eq!(physical, Extent::new(0, 4096).unwrap()); + assert!(matches!( + slab.submission(Extent::new(0, 4096).unwrap(), &buffer, &lease), + Err(Error::Corrupt) + )); + assert!(matches!( + slab.submission(Extent::new(8192, 4096).unwrap(), &buffer, &lease), + Err(Error::Corrupt) + )); + assert!(matches!( + slab.submission(Extent::new(6144, 4096).unwrap(), &buffer, &lease), + Err(Error::Corrupt) + )); + assert!(matches!( + slab.submission(Extent::new(4097, 4096).unwrap(), &buffer, &lease), + Err(Error::InvalidConfiguration) + )); + let other = Segments::from_geometry(slab.geometry().unwrap()).unwrap(); + let (foreign, extent) = other.append(4096).unwrap(); + assert!(matches!( + slab.submission(extent, &buffer, &foreign), + Err(Error::Stale) + )); + let extent = Extent::new(4096, 4096).unwrap(); + for (offset, expected) in [ + (1, Error::InvalidConfiguration), + (u64::MAX - 511, Error::Corrupt), + (i64::MAX as u64 - 511, Error::Corrupt), + ] { + let mut opened = slab.opened.borrow_mut(); + let OpenBacking::Devices(placements) = &mut opened.as_mut().unwrap().backing else { + unreachable!() + }; + placements[1].offset = offset; + drop(opened); + assert!( + matches!(slab.submission(extent, &buffer, &lease), Err(error) if error == expected) + ); + } + } + + /// Device submissions keep completion error handling and release their fences. + #[cfg(feature = "simulation")] + #[test] + fn device_io_faults_release_completion_resources() { + use uring_runtime::reactor::simulation::{Fault, Simulation}; + + #[derive(Clone, Copy, Debug, PartialEq)] + enum TestError { + Alloc(Error), + Runtime(uring_runtime::Error), + } + impl From for TestError { + fn from(error: Error) -> Self { + Self::Alloc(error) + } + } + impl From for TestError { + fn from(error: uring_runtime::Error) -> Self { + Self::Runtime(error) + } + } + #[derive(Clone)] + struct TestScope; + impl Scope for TestScope { + type Error = TestError; + fn check(&self) -> std::result::Result<(), TestError> { + Ok(()) + } + } + let directory = Directory::new(); + let Some(file) = device_file(&directory.0.join("device")) else { + return; + }; + let Some(a) = real_alignment(probe(&file)) else { + return; + }; + let slab = + Slab::<()>::from_devices(vec![DevicePlacement { file, offset: 4096 }], 4096, 4096, a) + .unwrap(); + let segments = Segments::new(4096); + let _ = slab.open_configured(&segments).unwrap(); + let (lease, extent) = segments.append(4096).unwrap(); + drop(lease); + let sim = Simulation::new(); + let _environment = sim.enter(); + let descriptor = sim + .open( + None, + Path::new("/device-faults"), + libc::O_CREAT | libc::O_RDWR, + ) + .unwrap(); + descriptor.as_sim().unwrap().set_len(8192).unwrap(); + let mut opened = slab.opened.borrow_mut(); + let OpenBacking::Devices(placements) = &mut opened.as_mut().unwrap().backing else { + unreachable!() + }; + placements[0].file = Rc::new(descriptor); + drop(opened); + let reactor = Reactor::::new(16, ()); + for write in [false, true] { + for (fault, expected) in [ + (Fault::Short(512), TestError::Alloc(Error::Io)), + ( + Fault::Errno(libc::EIO), + TestError::Runtime(uring_runtime::Error::Os(libc::EIO)), + ), + ] { + sim.inject(if write { "write" } else { "read" }, fault) + .unwrap(); + let lease = segments.lease(SegmentId(0), Generation(1)).unwrap(); + let buffer = slab.allocate(4096, ()).unwrap(); + let mut operation = if write { + slab.write(&reactor, extent, buffer, lease, &TestScope) + } else { + slab.read(&reactor, extent, buffer, lease, &TestScope) + }; + let mut completed = false; + for _ in 0..100 { + if let Poll::Ready(result) = operation + .as_mut() + .poll(&mut Context::from_waker(Waker::noop())) + { + assert_eq!(result.unwrap_err(), expected); + completed = true; + break; + } + reactor.poll_budgeted(64).unwrap(); + } + assert!(completed); + drop(operation); + assert_eq!(slab.writes_in_flight(), 0); + assert_eq!(reactor.in_flight(), 0); + } + } + segments.begin_evict(SegmentId(0)).unwrap(); + segments.recycle(SegmentId(0)).unwrap(); + } + + /// The live descriptor is direct, sparse, exclusively locked, and aligned. + #[test] + fn real_file_is_direct_aligned_and_sparse_without_reactor() { + let directory = Directory::new(); + let path = directory.0.join("caller-chosen.dat"); + let make = || Slab::<()>::new(path.clone(), 64 * 1024 * 1024, 32 * 1024 * 1024, 1024); + let conflicting = make(); + let slabs = make(); + assert!(!path.exists()); + let Some(alignment) = real_alignment(slabs.open_now()) else { + return; + }; + let stat = std::fs::metadata(&path).unwrap(); + assert_eq!(stat.len(), slabs.capacity_bytes()); + assert!(stat.blocks() * 512 < stat.len()); + let opened = slabs.opened.borrow(); + let OpenBacking::File(file) = &opened.as_ref().unwrap().backing else { + unreachable!() + }; + let fd = file.as_raw_fd(); + // SAFETY: descriptor and buffers remain live throughout these synchronous calls. + assert_ne!( + unsafe { libc::fcntl(fd, libc::F_GETFL) } & libc::O_DIRECT, + 0 + ); + let extent = alignment.extent(0, 31).unwrap(); + let mut buffer = slabs.allocate(extent.length(), ()).unwrap(); + buffer.bytes_mut().unwrap()[..31].fill(42); + assert_eq!( + unsafe { libc::pwrite(fd, buffer.bytes().unwrap().as_ptr().cast(), buffer.len(), 0) }, + buffer.len() as isize + ); + buffer.bytes_mut().unwrap().fill(0); + assert_eq!( + unsafe { + libc::pread( + fd, + buffer.bytes_mut().unwrap().as_mut_ptr().cast(), + extent.length(), + 0, + ) + }, + extent.length() as isize + ); + assert_eq!(&buffer.bytes().unwrap()[..31], &[42; 31]); + assert!(buffer.bytes().unwrap()[31..].iter().all(|b| *b == 0)); + // Filesystems may accept unaligned O_DIRECT I/O via buffered fallback. + // The allocator must reject it before submission regardless of the kernel. + assert_eq!( + alignment.check(Extent::new(1, 31).unwrap(), &buffer), + Err(Error::InvalidConfiguration) + ); + drop(buffer); + assert_eq!(slabs.idle_bytes(), extent.length()); + assert_eq!(slabs.reclaim_idle(), extent.length()); + assert_eq!(slabs.idle_bytes(), 0); + assert_eq!(conflicting.open_now(), Err(Error::Unavailable)); + } + + /// An oversized automatic table must fail before creating or sizing its backing. + #[test] + fn open_configured_rejects_oversized_table_without_file_side_effects() { + use std::os::unix::fs::OpenOptionsExt; + + let directory = Directory::new(); + let existing = directory.0.join("existing"); + let file = std::fs::OpenOptions::new() + .write(true) + .create_new(true) + .mode(0o600) + .open(&existing) + .unwrap(); + let paths = [ + directory.0.join("new"), + directory.0.join("missing/data"), + existing.clone(), + ]; + for path in &paths { + let segments = Segments::new(4096); + let slab = Slab::<()>::new(path.clone(), 4096 * (crate::MAX_SEGMENTS + 1), 4096, 512); + assert_eq!( + slab.open_configured(&segments), + Err(Error::InvalidConfiguration) + ); + assert!(!segments.is_configured()); + assert_eq!(segments.count(), 0); + assert_eq!(file.metadata().unwrap().len(), 0); + if path != &existing { + assert!(!path.exists()); + } + assert!(!directory.0.join("missing").exists()); + assert_eq!(slab.geometry(), Err(Error::Unavailable)); + } + + let Some(alignment) = real_alignment(probe(&file)) else { + return; + }; + let segment_bytes = alignment.extent(0, 512).unwrap().length() as u64; + let capacity = segment_bytes.checked_mul(crate::MAX_SEGMENTS + 1).unwrap(); + let retry_capacity = segment_bytes.checked_mul(2).unwrap(); + for path in &paths { + let segments = Segments::new(segment_bytes); + let slab = Slab::<()>::new(path.clone(), capacity, segment_bytes, 512); + assert_eq!( + slab.open_configured(&segments), + Err(Error::InvalidConfiguration) + ); + assert!(!segments.is_configured()); + assert_eq!(slab.geometry(), Err(Error::Unavailable)); + let retry = Slab::<()>::new(path.clone(), retry_capacity, segment_bytes, 512); + if real_alignment(retry.open_configured(&segments)).is_none() { + return; + } + assert_eq!(segments.count(), 2); + assert_eq!(std::fs::metadata(path).unwrap().len(), retry_capacity); + drop(retry); + std::fs::remove_file(path).unwrap(); + if path.parent() != Some(directory.0.as_path()) { + std::fs::remove_dir(path.parent().unwrap()).unwrap(); + } + } + } + + /// Invalid arithmetic inputs fail without creating a directory or a file. + #[test] + fn open_configured_rejects_invalid_geometry_before_open() { + let directory = Directory::new(); + for (capacity, segment, record) in [ + (0, 0, 512), + (4096, 0, 512), + (0, 4096, 512), + (4097, 4096, 512), + (4096, 8192, 512), + (4096, 4096, 0), + (u64::MAX, 1, 512), + (u64::MAX, u64::MAX, 512), + ] { + let path = directory.0.join("missing/data"); + let slab = Slab::<()>::new(path.clone(), capacity, segment, record); + let segments = Segments::new(segment); + assert_eq!( + slab.open_configured(&segments), + Err(Error::InvalidConfiguration) + ); + assert!(!segments.is_configured()); + assert_eq!(slab.geometry(), Err(Error::Unavailable)); + assert!(!path.exists()); + assert!(!path.parent().unwrap().exists()); + } + } + + /// The slot limit does not cap physical backing for an already configured table. + #[test] + fn open_configured_accepts_partial_table_above_physical_segment_limit() { + use std::os::unix::fs::OpenOptionsExt; + + let directory = Directory::new(); + let file = std::fs::OpenOptions::new() + .write(true) + .create_new(true) + .mode(0o600) + .open(directory.0.join("probe")) + .unwrap(); + let Some(alignment) = real_alignment(probe(&file)) else { + return; + }; + let segment_bytes = alignment.extent(0, 512).unwrap().length() as u64; + let capacity = segment_bytes.checked_mul(crate::MAX_SEGMENTS + 1).unwrap(); + let segments = Segments::new(segment_bytes); + segments.configure(capacity, 2, alignment).unwrap(); + let path = directory.0.join("partial"); + let slab = Slab::<()>::new(path.clone(), capacity, segment_bytes, 512); + let Some(opened_alignment) = real_alignment(slab.open_configured(&segments)) else { + return; + }; + assert_eq!(opened_alignment, alignment); + assert_eq!(segments.count(), 2); + assert_eq!(segments.capacity_bytes(), capacity); + assert_eq!( + slab.geometry().unwrap().segment_count(), + crate::MAX_SEGMENTS + 1 + ); + assert_eq!(std::fs::metadata(path).unwrap().len(), capacity); + } + + /// Invalid startup dimensions never truncate a preexisting nonempty file. + #[test] + fn open_rejects_bad_layout_and_existing_size_without_truncating() { + use std::os::unix::fs::PermissionsExt; + let directory = Directory::new(); + let probe = Slab::<()>::new(directory.0.join("capability-probe"), 4096, 4096, 512); + if real_alignment(probe.open_now()).is_none() { + return; + } + let path = directory.0.join("data"); + std::fs::write(&path, [42; 7]).unwrap(); + std::fs::set_permissions(&path, std::fs::Permissions::from_mode(0o600)).unwrap(); + let slab = Slab::<()>::new(path.clone(), 4096, 4096, 512); + assert_eq!(slab.open_now(), Err(Error::InvalidConfiguration)); + assert_eq!(std::fs::read(&path).unwrap(), [42; 7]); + for (capacity, segment, record) in [ + (0, 4096, 512), + (4096, 0, 512), + (4096, 4096, 0), + (4097, 4096, 512), + (4096, 4096, 4097), + ] { + assert_eq!( + Slab::<()>::new(directory.0.join("invalid"), capacity, segment, record).open_now(), + Err(Error::InvalidConfiguration) + ); + } + assert_eq!(slab.alignment(), Err(Error::Unavailable)); + assert!(matches!(slab.allocate(512, ()), Err(Error::Unavailable))); + } + + /// Private validation rejects wrong ranges before any runtime admission. + #[test] + fn submission_checks_alignment_segment_and_capacity() { + let directory = Directory::new(); + let slab = Slab::<()>::new(directory.0.join("data"), 8192, 4096, 512); + let segments = Segments::new(4096); + let Some(a) = real_alignment(slab.open_configured(&segments)) else { + return; + }; + let (lease, extent) = segments.append(4096).unwrap(); + let buffer = slab.allocate(4096, ()).unwrap(); + assert!(slab.submission(extent, &buffer, &lease).is_ok()); + assert!(matches!( + slab.submission(Extent::new(4096, 4096).unwrap(), &buffer, &lease), + Err(Error::Corrupt) + )); + assert!(matches!( + slab.submission(Extent::new(1, 4096).unwrap(), &buffer, &lease), + Err(Error::InvalidConfiguration) + )); + let outside_table = Segments::new(4096); + outside_table.configure(12288, 3, a).unwrap(); + drop(outside_table.append(4096).unwrap()); + drop(outside_table.append(4096).unwrap()); + let (outside, extent) = outside_table.append(4096).unwrap(); + assert!(matches!( + slab.submission(extent, &buffer, &outside), + Err(Error::Stale) + )); + assert_eq!( + outside_table.validate(SegmentId(2), Generation(1), &extent), + Ok(()) + ); + } + + /// Select the callback that runs a one-shot reentrant test action. + #[derive(Clone, Copy, PartialEq)] + enum FenceCallback { + Clone, + Drop, + /// Reenter on completion's first clone or wake, with either implementation. + Notify, + } + + /// One action and the waker callback that should run it. + type FenceAction = (FenceCallback, Box); + + thread_local! { + static FENCE_CALLBACK: RefCell> = + RefCell::new(None); + static FENCE_WAKES: Cell = const { Cell::new(0) }; + } + + /// Run caller code after releasing the callback registry's borrow. + fn run_fence_callback(event: FenceCallback) { + let callback = FENCE_CALLBACK.with(|slot| { + let mut slot = slot.borrow_mut(); + if slot.as_ref().is_some_and(|(on, _)| { + *on == event || (*on == FenceCallback::Notify && event == FenceCallback::Clone) + }) { + slot.take() + } else { + None + } + }); + if let Some((_, callback)) = callback { + callback(); + } + } + + /// A stateless waker accesses only the calling thread's test registry. + fn fence_callback_waker() -> Waker { + use std::task::{RawWaker, RawWakerVTable}; + + /// Cloning owns no data but may run the thread's registered action. + unsafe fn clone(_: *const ()) -> RawWaker { + run_fence_callback(FenceCallback::Clone); + RawWaker::new(std::ptr::null(), &VTABLE) + } + + /// Both wake forms count a notification and may run the test action. + unsafe fn wake(_: *const ()) { + FENCE_WAKES.with(|count| count.set(count.get() + 1)); + run_fence_callback(FenceCallback::Notify); + } + + /// Dropping owns no data but may run the thread's registered action. + unsafe fn drop_raw(_: *const ()) { + run_fence_callback(FenceCallback::Drop); + } + + static VTABLE: RawWakerVTable = RawWakerVTable::new(clone, wake, wake, drop_raw); + + // SAFETY: there is no raw data or shared ownership. Every callback accesses + // only its calling thread's registry, even if the Waker moves across threads. + unsafe { Waker::from_raw(RawWaker::new(std::ptr::null(), &VTABLE)) } + } + + /// Completion during clone or replacement drop is observed before returning. + #[test] + fn fence_poll_handles_reentrant_clone_and_drop() { + for event in [FenceCallback::Clone, FenceCallback::Drop] { + for new_write in [false, true] { + let state = Rc::new(WriteState::default()); + let write = state.acquire().unwrap(); + let mut waiter = FenceWaiter { + state: state.clone(), + registration: None, + }; + let waker = fence_callback_waker(); + let mut cx = Context::from_waker(&waker); + if event == FenceCallback::Drop { + assert!(Pin::new(&mut waiter).poll(&mut cx).is_pending()); + } + let next = Rc::new(RefCell::new(None)); + let saved_next = next.clone(); + let saved_state = state.clone(); + FENCE_CALLBACK.with(|slot| { + *slot.borrow_mut() = Some(( + event, + Box::new(move || { + drop(write); + if new_write { + *saved_next.borrow_mut() = Some(saved_state.acquire().unwrap()); + } + }), + )); + }); + let result = Pin::new(&mut waiter).poll(&mut cx); + FENCE_CALLBACK.with(|slot| assert!(slot.borrow().is_none())); + if new_write { + assert!(result.is_pending()); + assert_eq!(state.count.get(), 1); + assert_eq!(state.waiters.borrow().len(), 1); + let before = FENCE_WAKES.get(); + drop(next.borrow_mut().take()); + assert_eq!(FENCE_WAKES.get(), before + 1); + assert_eq!(Pin::new(&mut waiter).poll(&mut cx), Poll::Ready(Ok(()))); + } else { + assert_eq!(result, Poll::Ready(Ok(()))); + } + assert_eq!(state.count.get(), 0); + assert!(state.waiters.borrow().is_empty()); + assert!(waiter.registration.is_none()); + } + } + } + + /// Cleanup callbacks can accept a write at every zero-count exit. + #[test] + fn fence_poll_rechecks_unregister_callbacks() { + for completion in [None, Some(FenceCallback::Clone), Some(FenceCallback::Drop)] { + let slab = Slab::<()>::new(PathBuf::new(), 4096, 4096, 512); + let write = slab.writes.acquire().unwrap(); + let mut waiter = slab.fence_writes(); + let waker = fence_callback_waker(); + let mut cx = Context::from_waker(&waker); + assert!(waiter.as_mut().poll(&mut cx).is_pending()); + let next = Rc::new(RefCell::new(None)); + let saved_next = next.clone(); + let state = slab.writes.clone(); + let accept: Box = Box::new(move || { + *saved_next.borrow_mut() = Some(state.acquire().unwrap()); + }); + if let Some(event) = completion { + FENCE_CALLBACK.with(|slot| { + *slot.borrow_mut() = Some(( + event, + Box::new(move || { + drop(write); + FENCE_CALLBACK.with(|slot| { + *slot.borrow_mut() = Some((FenceCallback::Drop, accept)); + }); + }), + )); + }); + } else { + drop(write); + FENCE_CALLBACK.with(|slot| { + *slot.borrow_mut() = Some((FenceCallback::Drop, accept)); + }); + } + assert!(waiter.as_mut().poll(&mut cx).is_pending()); + FENCE_CALLBACK.with(|slot| assert!(slot.borrow().is_none())); + assert_eq!(slab.writes_in_flight(), 1); + assert_eq!(slab.writes.waiters.borrow().len(), 1); + let before = FENCE_WAKES.get(); + drop(next.borrow_mut().take()); + assert_eq!(FENCE_WAKES.get(), before + 1); + assert_eq!(waiter.as_mut().poll(&mut cx), Poll::Ready(Ok(()))); + assert_eq!(slab.writes_in_flight(), 0); + assert!(slab.writes.waiters.borrow().is_empty()); + } + } + + /// Completion callbacks can accept a write and repoll a drained registration. + #[test] + fn fence_completion_allows_reentrant_registration() { + let state = Rc::new(WriteState::default()); + let write = state.acquire().unwrap(); + let waiter = Rc::new(RefCell::new(FenceWaiter { + state: state.clone(), + registration: None, + })); + let waker = fence_callback_waker(); + assert!( + Pin::new(&mut *waiter.borrow_mut()) + .poll(&mut Context::from_waker(&waker)) + .is_pending() + ); + let next = Rc::new(RefCell::new(None)); + let saved_next = next.clone(); + let saved_state = state.clone(); + let saved_waiter = waiter.clone(); + FENCE_CALLBACK.with(|slot| { + *slot.borrow_mut() = Some(( + FenceCallback::Notify, + Box::new(move || { + *saved_next.borrow_mut() = Some(saved_state.acquire().unwrap()); + let waker = fence_callback_waker(); + assert!( + Pin::new(&mut *saved_waiter.borrow_mut()) + .poll(&mut Context::from_waker(&waker)) + .is_pending() + ); + }), + )); + }); + drop(write); + FENCE_CALLBACK.with(|slot| assert!(slot.borrow().is_none())); + assert_eq!(state.waiters.borrow().len(), 1); + assert_eq!(state.count.get(), 1); + let before = FENCE_WAKES.get(); + drop(next.borrow_mut().take()); + assert_eq!(FENCE_WAKES.get(), before + 1); + assert_eq!( + Pin::new(&mut *waiter.borrow_mut()).poll(&mut Context::from_waker(&waker)), + Poll::Ready(Ok(())) + ); + assert!(state.waiters.borrow().is_empty()); + assert!(waiter.borrow().registration.is_none()); + } + + /// Waiters sleep, replace wakers, unregister on cancellation, and register again. + #[test] + fn fence_waiters_sleep_update_wakers_and_unregister_on_cancellation() { + use std::sync::{ + Arc, + atomic::{AtomicUsize, Ordering}, + }; + use std::task::Wake; + /// Counts wake notifications without scheduling actual work. + #[derive(Default)] + struct WakeCount(AtomicUsize); + + impl Wake for WakeCount { + /// Count an owned notification. + fn wake(self: Arc) { + self.0.fetch_add(1, Ordering::Relaxed); + } + + /// Count a borrowed notification. + fn wake_by_ref(self: &Arc) { + self.0.fetch_add(1, Ordering::Relaxed); + } + } + let slab = Slab::<()>::new(PathBuf::new(), 4096, 4096, 512); + let first_write = slab.writes.acquire().unwrap(); + let last_write = slab.writes.acquire().unwrap(); + let old = Arc::new(WakeCount::default()); + let current = Arc::new(WakeCount::default()); + let other = Arc::new(WakeCount::default()); + let canceled = Arc::new(WakeCount::default()); + let poll = |op: &mut Operation<'_, (), Error>, wakes: &Arc| { + op.as_mut() + .poll(&mut Context::from_waker(&Waker::from(wakes.clone()))) + }; + let mut one = slab.fence_writes(); + let mut two = slab.fence_writes(); + let mut abandoned = slab.fence_writes(); + assert!(poll(&mut one, &old).is_pending()); + assert!(poll(&mut one, ¤t).is_pending()); + assert!(poll(&mut two, &other).is_pending()); + assert!(poll(&mut abandoned, &canceled).is_pending()); + assert_eq!(slab.writes.waiters.borrow().len(), 3); + drop(abandoned); + assert_eq!(slab.writes.waiters.borrow().len(), 2); + assert_eq!(Arc::strong_count(&canceled), 1); + drop(first_write); + for count in [&old, ¤t, &other, &canceled] { + assert_eq!(count.0.load(Ordering::Relaxed), 0); + } + drop(last_write); + assert_eq!(old.0.load(Ordering::Relaxed), 0); + assert_eq!(canceled.0.load(Ordering::Relaxed), 0); + assert_eq!(current.0.load(Ordering::Relaxed), 1); + assert_eq!(other.0.load(Ordering::Relaxed), 1); + assert!(slab.writes.waiters.borrow().is_empty()); + let next = slab.writes.acquire().unwrap(); + assert!(poll(&mut one, ¤t).is_pending()); + assert_eq!(slab.writes.waiters.borrow().len(), 1); + drop(next); + assert_eq!(current.0.load(Ordering::Relaxed), 2); + assert_eq!(poll(&mut one, ¤t), Poll::Ready(Ok(()))); + assert_eq!(poll(&mut two, &other), Poll::Ready(Ok(()))); + assert_eq!( + poll(&mut slab.fence_writes(), ¤t), + Poll::Ready(Ok(())) + ); + } + + /// Overflow cannot create a decrement guard or alter the saturated count. + #[test] + fn write_guard_overflow_is_atomic() { + let state = Rc::new(WriteState::default()); + state.count.set(usize::MAX); + assert!(matches!(state.acquire(), Err(Error::Busy))); + assert_eq!(state.count.get(), usize::MAX); + state.count.set(0); + let guard = state.acquire().unwrap(); + assert_eq!(state.count.get(), 1); + drop(guard); + assert_eq!(state.count.get(), 0); + } + + /// Validation consumes rejected resources and successful preparation retains its lease. + #[test] + fn prepared_submission_owns_resources_and_failed_binding_is_recoverable() { + let directory = Directory::new(); + let slab = Slab::::new(directory.0.join("prepared"), 8192, 4096, 512); + let Some(alignment) = real_alignment(slab.open_now()) else { + return; + }; + let table = Segments::new(4096); + table.configure(8192, 2, alignment).unwrap(); + let (lease, extent) = table.append(4096).unwrap(); + let used = Rc::new(Cell::new(0)); + let buffer = slab + .allocate(4096, CountingCharge::new(&used, 4096)) + .unwrap(); + assert!(matches!( + slab.prepare(extent, buffer, lease), + Err(Error::Unavailable) + )); + assert_eq!(slab.reclaim_idle(), 4096); + assert_eq!(used.get(), 0); + assert_eq!(slab.writes_in_flight(), 0); + slab.configure_segments(&table).unwrap(); + let lease = table.lease(SegmentId(0), Generation(1)).unwrap(); + let buffer = slab + .allocate(4096, CountingCharge::new(&used, 4096)) + .unwrap(); + let submission = slab.prepare(extent, buffer, lease).unwrap(); + table.begin_evict(SegmentId(0)).unwrap(); + assert_eq!(table.recycle(SegmentId(0)), Err(Error::Busy)); + drop(submission); + table.recycle(SegmentId(0)).unwrap(); + assert_eq!(used.get(), 4096); + slab.reclaim_idle(); + assert_eq!(used.get(), 0); + } + + /// System classification preserves errno and operation without hiding failures. + #[test] + fn system_errors_keep_operation_and_errno() { + for errno in [ + libc::EACCES, + libc::EPERM, + libc::ENOSPC, + libc::EIO, + libc::ELOOP, + ] { + assert_eq!( + direct_error("open", std::io::Error::from_raw_os_error(errno)), + Error::SystemIo { + operation: "open", + errno: Some(errno) + } + ); + assert_eq!( + lock_error(std::io::Error::from_raw_os_error(errno)), + Error::SystemIo { + operation: "flock", + errno: Some(errno) + } + ); + } + assert_eq!( + lock_error(std::io::Error::from_raw_os_error(libc::EWOULDBLOCK)), + Error::Unavailable + ); + for errno in [libc::EINVAL, libc::EOPNOTSUPP, libc::ENOSYS] { + assert_eq!( + direct_error("fcntl-direct", std::io::Error::from_raw_os_error(errno)), + Error::Unsupported + ); + } + assert_eq!( + system_error("test", std::io::Error::other("synthetic")), + Error::SystemIo { + operation: "test", + errno: None + } + ); + assert_eq!( + probe_fd(-1), + Err(Error::SystemIo { + operation: "statx", + errno: Some(libc::EBADF) + }) + ); + // SAFETY: zero is a valid initialized statx output representation. + let mut stat: libc::statx = unsafe { std::mem::zeroed() }; + assert_eq!(alignment_from_stat(&stat), Err(Error::Unsupported)); + stat.stx_mask = libc::STATX_DIOALIGN; + stat.stx_dio_mem_align = 512; + stat.stx_dio_offset_align = 512; + assert_eq!(alignment_from_stat(&stat), Alignment::new(512, 512, 512)); + } + + /// File validation requires regular type, effective owner, private mode, and one link. + #[test] + fn private_file_validation_checks_type_owner_permissions_and_links() { + assert_eq!(validate_file(libc::S_IFREG | 0o600, 17, 17, 1), Ok(())); + for (mode, owner, links) in [ + (libc::S_IFREG | 0o644, 17, 1), + (libc::S_IFREG | 0o600, 18, 1), + (libc::S_IFREG | 0o600, 17, 2), + (libc::S_IFREG | 0o4600, 17, 1), + (libc::S_IFDIR | 0o600, 17, 1), + (libc::S_IFIFO | 0o600, 17, 1), + ] { + assert_eq!( + validate_file(mode, owner, 17, links), + Err(Error::InvalidConfiguration) + ); + } + } + + /// A prior lock holder cannot leave unsafe metadata for startup to accept. + #[test] + fn lock_handoff_revalidates_private_file_metadata() { + use std::os::unix::fs::PermissionsExt; + + for change in ["permissions", "special-bits", "hard-link", "unlink"] { + let directory = Directory::new(); + let path = directory.0.join("data"); + let prior = open_private_file(&path).unwrap(); + let observer = File::open(&path).unwrap(); + // SAFETY: flock borrows a live descriptor owned by this test. + assert_eq!( + unsafe { libc::flock(prior.as_raw_fd(), libc::LOCK_EX | libc::LOCK_NB) }, + 0 + ); + let slab = Slab::<()>::new(path.clone(), 8192, 4096, 512); + let mut called = false; + let result = slab.open_file_with_lock_handoff(|_| { + called = true; + match change { + "permissions" => prior + .set_permissions(std::fs::Permissions::from_mode(0o644)) + .unwrap(), + "special-bits" => prior + .set_permissions(std::fs::Permissions::from_mode(0o4600)) + .unwrap(), + "hard-link" => std::fs::hard_link(&path, directory.0.join("alias")).unwrap(), + "unlink" => std::fs::remove_file(&path).unwrap(), + _ => unreachable!(), + } + drop(prior); + }); + assert!(called); + assert_eq!(result, Err(Error::InvalidConfiguration), "{change}"); + assert!(slab.opened.borrow().is_none()); + assert_eq!(observer.metadata().unwrap().len(), 0); + // SAFETY: the independent live descriptor checks lock cleanup on failure. + assert_eq!( + unsafe { libc::flock(observer.as_raw_fd(), libc::LOCK_EX | libc::LOCK_NB) }, + 0 + ); + } + } + + /// Valid handoffs use the current size and never truncate mismatched files. + #[test] + fn lock_handoff_uses_refreshed_size() { + for size_in_segments in [0, 2, 1] { + let directory = Directory::new(); + let path = directory.0.join("data"); + let prior = open_private_file(&path).unwrap(); + let Some(alignment) = real_alignment(probe(&prior)) else { + continue; + }; + let segment_bytes = alignment.extent(0, 512).unwrap().length() as u64; + let capacity = segment_bytes.checked_mul(2).unwrap(); + let size = segment_bytes * size_in_segments; + // SAFETY: flock borrows a live descriptor owned by this test. + assert_eq!( + unsafe { libc::flock(prior.as_raw_fd(), libc::LOCK_EX | libc::LOCK_NB) }, + 0 + ); + let slab = Slab::<()>::new(path.clone(), capacity, segment_bytes, 512); + let result = slab.open_file_with_lock_handoff(|_| { + prior.set_len(size).unwrap(); + drop(prior); + }); + if result == Err(Error::Unsupported) { + assert!(real_alignment(result).is_none()); + continue; + } + if size_in_segments == 1 { + assert_eq!(result, Err(Error::InvalidConfiguration)); + assert!(slab.opened.borrow().is_none()); + assert_eq!(std::fs::metadata(&path).unwrap().len(), size); + } else { + assert_eq!(result, Ok(alignment)); + assert_eq!(slab.alignment(), Ok(alignment)); + assert_eq!(std::fs::metadata(&path).unwrap().len(), capacity); + assert_eq!( + Slab::<()>::new(path, capacity, segment_bytes, 512).open_now(), + Err(Error::Unavailable) + ); + } + } + } + + /// Descriptor traversal rejects symlinks, FIFOs, and parent-directory escapes. + #[test] + fn descriptor_relative_open_rejects_symlinks_and_nonregular_files() { + use std::os::unix::fs::{PermissionsExt, symlink}; + let directory = Directory::new(); + let make = |path| Slab::<()>::new(path, 8192, 4096, 512); + let target = directory.0.join("target"); + std::fs::create_dir(&target).unwrap(); + let alias = directory.0.join("alias"); + symlink(&target, &alias).unwrap(); + assert!(matches!( + make(alias.join("data")).open_now(), + Err(Error::SystemIo { + operation: "open-parent", + errno: Some(libc::ENOTDIR | libc::ELOOP) + }) + )); + assert!(!target.join("data").exists()); + let data = target.join("data"); + std::fs::write(&data, [42; 7]).unwrap(); + let link = directory.0.join("link"); + symlink(&data, &link).unwrap(); + assert_eq!( + make(link).open_now(), + Err(Error::SystemIo { + operation: "open", + errno: Some(libc::ELOOP) + }) + ); + assert_eq!(std::fs::read(&data).unwrap(), [42; 7]); + std::fs::set_permissions(&data, std::fs::Permissions::from_mode(0o666)).unwrap(); + assert_eq!( + make(data.clone()).open_now(), + Err(Error::InvalidConfiguration) + ); + assert!(make(target.clone()).open_now().is_err()); + let fifo = directory.0.join("fifo"); + let fifo_c = CString::new(fifo.as_os_str().as_bytes()).unwrap(); + // SAFETY: valid C pathname and POSIX permission mode. + assert_eq!(unsafe { libc::mkfifo(fifo_c.as_ptr(), 0o600) }, 0); + assert_eq!(make(fifo).open_now(), Err(Error::InvalidConfiguration)); + assert_eq!( + make(target.join("..").join("escape")).open_now(), + Err(Error::InvalidConfiguration) + ); + let new_path = directory.0.join("new/nested/data"); + let opened = open_private_file(&new_path).unwrap(); + assert_eq!(opened.metadata().unwrap().mode() & 0o777, 0o600); + assert_eq!( + std::fs::metadata(new_path.parent().unwrap()) + .unwrap() + .mode() + & 0o777, + 0o700 + ); + } + + /// Binding is permanent and validates dimensions as well as captured used extents. + #[test] + fn binding_validates_geometry_identity_and_used_extents() { + let directory = Directory::new(); + let slab = Slab::<()>::new(directory.0.join("data"), 8192, 4096, 512); + let table = Segments::new(4096); + let Some(a) = real_alignment(slab.open_configured(&table)) else { + return; + }; + assert_eq!(table.geometry(), Some(slab.geometry().unwrap())); + slab.configure_segments(&table).unwrap(); + let other = Segments::new(4096); + other.configure(8192, 2, a).unwrap(); + assert_eq!( + slab.configure_segments(&other), + Err(Error::InvalidConfiguration) + ); + let length = a.extent(0, 512).unwrap().length(); + let (foreign, extent) = other.append(length).unwrap(); + let buffer = slab.allocate(length, ()).unwrap(); + assert!(matches!( + slab.submission(extent, &buffer, &foreign), + Err(Error::Stale) + )); + let (lease, extent) = table.append(length).unwrap(); + assert!(slab.submission(extent, &buffer, &lease).is_ok()); + assert!(matches!( + slab.submission(Extent::new(length as u64, length).unwrap(), &buffer, &lease), + Err(Error::Corrupt) + )); + let wrong = Slab::<()>::new(directory.0.join("wrong"), 8192, 4096, 512); + let _ = wrong.open_now().unwrap(); + assert_eq!( + wrong.configure_segments(&Segments::new(8192)), + Err(Error::InvalidConfiguration) + ); + let wrong_capacity = Segments::new(4096); + wrong_capacity.configure(4096, 1, a).unwrap(); + assert_eq!( + wrong.configure_segments(&wrong_capacity), + Err(Error::InvalidConfiguration) + ); + let wrong_alignment = Segments::new(4096); + let incompatible = Alignment::new(a.memory() * 2, a.offset(), a.length()).unwrap(); + wrong_alignment.configure(8192, 2, incompatible).unwrap(); + assert_eq!( + wrong.configure_segments(&wrong_alignment), + Err(Error::InvalidConfiguration) + ); + let partial = Segments::new(4096); + partial.configure(8192, 1, a).unwrap(); + wrong.configure_segments(&partial).unwrap(); + assert_eq!(partial.count(), 1); + if length < 4096 { + drop(table.append(4096 - length).unwrap()); + } + table.begin_evict(SegmentId(0)).unwrap(); + assert!(slab.submission(extent, &buffer, &lease).is_ok()); + assert_eq!(table.recycle(SegmentId(0)), Err(Error::Busy)); + drop(lease); + table.recycle(SegmentId(0)).unwrap(); + } + + /// Size mismatch releases idle accounting even when new allocation is invalid. + #[test] + fn idle_size_mismatch_releases_retained_charge_even_on_invalid_replacement() { + let directory = Directory::new(); + let slab = Slab::::new(directory.0.join("data"), 8192, 4096, 512); + let Some(a) = real_alignment(slab.open_now()) else { + return; + }; + let length = a.extent(0, 512).unwrap().length(); + let used = Rc::new(Cell::new(0)); + drop( + slab.allocate(length, CountingCharge::new(&used, length)) + .unwrap(), + ); + assert_eq!(slab.idle_bytes(), length); + assert_eq!(used.get(), length); + assert!(matches!( + slab.allocate(0, CountingCharge::new(&used, 0)), + Err(Error::InvalidConfiguration) + )); + assert_eq!(used.get(), 0); + assert_eq!(slab.idle_bytes(), 0); + } +} diff --git a/cmd/racer-dataplane/alloc/tests/workflows.rs b/cmd/racer-dataplane/alloc/tests/workflows.rs new file mode 100644 index 000000000..e245b1a59 --- /dev/null +++ b/cmd/racer-dataplane/alloc/tests/workflows.rs @@ -0,0 +1,1294 @@ +//! Public allocator workflows across restart, cancellation, accounting, and real I/O. + +use page_alloc::{Alignment, DevicePlacement, Error, Generation, SegmentId, Segments, Slab}; +#[cfg(feature = "simulation")] +use page_alloc::{Charge, SegmentState}; +#[cfg(feature = "simulation")] +use std::{cell::Cell, path::Path, rc::Rc}; +use std::{ + fs::File, + future::Future, + os::fd::FromRawFd, + path::PathBuf, + pin::Pin, + sync::Arc, + task::{Context, Poll, Waker}, +}; +#[cfg(feature = "simulation")] +use uring_runtime::reactor::simulation::{Fault, Simulation}; +use uring_runtime::{Operation, Scope, reactor::Reactor}; + +/// Distinguishes allocator validation errors from runtime completion failures. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +enum TestError { + Alloc(Error), + + Runtime(uring_runtime::Error), +} + +impl From for TestError { + /// Preserve the allocator category in workflow assertions. + fn from(error: Error) -> Self { + Self::Alloc(error) + } +} + +impl From for TestError { + /// Preserve runtime errors without enriching them as synchronous allocator errors. + fn from(error: uring_runtime::Error) -> Self { + Self::Runtime(error) + } +} + +/// An always-live caller scope for deterministic ownership tests. +#[derive(Clone)] +struct TestScope; + +impl Scope for TestScope { + type Error = TestError; + + /// Keep the scope live; cancellation is injected by dropping waiting futures. + fn check(&self) -> Result<(), TestError> { + Ok(()) + } +} + +/// Poll once without assuming that a completion wakes a host executor. +fn poll(future: &mut Pin + '_>>) -> Poll { + future + .as_mut() + .poll(&mut Context::from_waker(Waker::noop())) +} + +/// Drive a simulated operation with a fixed, finite reactor-turn budget. +#[cfg(feature = "simulation")] +fn drive( + reactor: &Reactor, + mut operation: Pin + '_>>, +) -> T { + for _ in 0..100 { + if let Poll::Ready(value) = poll(&mut operation) { + return value; + } + reactor.poll_budgeted(64).unwrap(); + } + panic!("simulation did not complete in 100 turns"); +} + +/// Caller-owned accounting for live buffers and the single idle slot. +#[cfg(feature = "simulation")] +struct CountingCharge { + used: Rc>, + + bytes: usize, +} + +#[cfg(feature = "simulation")] +impl CountingCharge { + /// Record admission before passing the guard into allocator ownership. + fn new(used: &Rc>, bytes: usize) -> Self { + used.set(used.get() + bytes); + Self { + used: used.clone(), + bytes, + } + } +} + +#[cfg(feature = "simulation")] +impl Charge for CountingCharge { + /// Cover only the bytes actually admitted by this guard. + fn covers(&self, bytes: usize) -> bool { + self.bytes >= bytes + } +} + +#[cfg(feature = "simulation")] +impl Drop for CountingCharge { + /// Return admission when the final owner releases this guard. + fn drop(&mut self) { + self.used.set(self.used.get() - self.bytes); + } +} + +/// Restored records remain readable and the sealed recovery tail is not overwritten. +#[cfg(feature = "simulation")] +#[test] +fn reopen_snapshot_reads_old_record_and_appends_without_overwriting_it() { + let simulation = Simulation::new(); + let _environment = simulation.enter(); + let reactor = Reactor::::new(16, ()); + let make_slab = || Slab::<()>::new("/alloc-workflows/restart".into(), 16384, 8192, 1024); + let slab = make_slab(); + let segments = Segments::new(slab.segment_bytes()); + let alignment = slab.open_configured(&segments).unwrap(); + let payload = b"a caller-owned record, not a cache entry"; + let padded = alignment.extent(0, payload.len()).unwrap().length(); + let (lease, extent) = segments.append(padded).unwrap(); + let (id, generation) = (lease.id(), lease.generation()); + let mut buffer = slab.allocate(padded, ()).unwrap(); + buffer.as_mut_slice()[..payload.len()].copy_from_slice(payload); + drop( + drive( + &reactor, + slab.write(&reactor, extent, buffer, lease, &TestScope), + ) + .unwrap(), + ); + drive(&reactor, slab.fence_writes()).unwrap(); + let frozen = segments.freeze().unwrap(); + let snapshot = segments.snapshot(); + assert_eq!(snapshot[0].state, SegmentState::Open); + assert_eq!(snapshot[0].used_bytes, padded as u64); + assert_eq!(snapshot[1].state, SegmentState::Free); + assert_eq!(segments.restore(snapshot.clone()), Err(Error::Busy)); + drop(frozen); + drop(slab); + drop(segments); + + let slab = make_slab(); + let segments = Segments::new(slab.segment_bytes()); + let alignment = slab.open_configured(&segments).unwrap(); + // Restore validates the entire image before changing any slot or free list. + for (invalid, label) in [ + "duplicate segment ID", + "zero generation", + "oversized sealed segment", + "misaligned sealed segment", + "nonempty free segment", + "missing segment", + ] + .into_iter() + .enumerate() + { + let mut damaged = snapshot.clone(); + match invalid { + 0 => damaged[1].id = SegmentId(0), + 1 => damaged[1].generation = Generation(0), + 2 => { + damaged[1].state = SegmentState::Sealed; + damaged[1].used_bytes = slab.segment_bytes() + alignment.length() as u64; + } + 3 => { + damaged[1].state = SegmentState::Sealed; + damaged[1].used_bytes = 513; + } + 4 => damaged[1].used_bytes = alignment.length() as u64, + _ => { + damaged.pop(); + } + } + assert_eq!(segments.restore(damaged), Err(Error::Corrupt), "{label}"); + assert_eq!(segments.free_count(), 2, "{label}"); + assert!( + segments + .snapshot() + .iter() + .all(|s| s.state == SegmentState::Free && s.used_bytes == 0), + "{label}" + ); + } + segments.validate_restore(&snapshot).unwrap(); + segments.restore(snapshot).unwrap(); + assert_eq!(segments.state(id), Ok(SegmentState::Sealed)); + assert_eq!(segments.free_count(), 1); + segments.validate(id, generation, &extent).unwrap(); + let buffer = drive( + &reactor, + slab.read( + &reactor, + extent, + slab.allocate(padded, ()).unwrap(), + segments.lease(id, generation).unwrap(), + &TestScope, + ), + ) + .unwrap(); + assert_eq!(&buffer.as_slice()[..payload.len()], payload); + assert!(buffer.as_slice()[payload.len()..].iter().all(|b| *b == 0)); + drop(buffer); + + let (next, next_extent) = segments.append(padded).unwrap(); + assert_eq!(next.id(), SegmentId(1)); + assert_eq!(next_extent.offset(), slab.segment_bytes()); + assert_eq!(segments.free_count(), 0); + let mut buffer = slab.allocate(padded, ()).unwrap(); + buffer.as_mut_slice().fill(99); + drop( + drive( + &reactor, + slab.write(&reactor, next_extent, buffer, next, &TestScope), + ) + .unwrap(), + ); + let buffer = drive( + &reactor, + slab.read( + &reactor, + extent, + slab.allocate(padded, ()).unwrap(), + segments.lease(id, generation).unwrap(), + &TestScope, + ), + ) + .unwrap(); + assert_eq!(&buffer.as_slice()[..payload.len()], payload); + assert!(buffer.as_slice()[payload.len()..].iter().all(|b| *b == 0)); + assert_eq!(slab.writes_in_flight(), 0); + assert_eq!(reactor.in_flight(), 0); +} + +/// Public pooling enforces admission, zeroization, exact-size reuse, and bounded retention. +#[cfg(feature = "simulation")] +#[test] +fn buffer_pool_keeps_only_zeroed_accounted_storage_and_rejects_undercharging() { + let simulation = Simulation::new(); + let _environment = simulation.enter(); + let slab = Slab::new("/alloc-workflows/pool".into(), 8192, 4096, 1024); + let segments = Segments::new(slab.segment_bytes()); + let alignment = slab.open_configured(&segments).unwrap(); + let size = alignment.extent(0, 31).unwrap().length(); + let live = Rc::new(Cell::new(0)); + let charge = |bytes| CountingCharge::new(&live, bytes); + let mut buffer = slab.allocate(size, charge(size)).unwrap(); + assert_eq!(buffer.len(), size); + assert!(!buffer.is_empty()); + let pointer = buffer.bytes().unwrap().as_ptr(); + assert_eq!(pointer as usize % alignment.memory(), 0); + assert!(buffer.bytes().unwrap().iter().all(|b| *b == 0)); + buffer.bytes_mut().unwrap().fill(42); + let retained = Rc::new(charge(17)); + let weak = Rc::downgrade(&retained); + buffer.retain(retained); + assert_eq!(live.get(), size + 17); + drop(buffer); + assert!(weak.upgrade().is_none()); + assert_eq!(live.get(), size); + assert_eq!(slab.idle_bytes(), size); + + // Denied admission must release its charge without consuming the idle buffer. + assert!(matches!( + slab.allocate(size, charge(size - 1)), + Err(Error::InvalidConfiguration) + )); + assert_eq!(live.get(), size); + assert_eq!(slab.idle_bytes(), size); + let reused = slab.allocate(size, charge(size)).unwrap(); + assert_eq!(reused.bytes().unwrap().as_ptr(), pointer); + assert!(reused.bytes().unwrap().iter().all(|b| *b == 0)); + assert_eq!(live.get(), size); + assert_eq!(slab.idle_bytes(), 0); + assert_eq!(slab.reclaim_idle(), 0); + + // Two simultaneous users cannot both return storage to the single idle slot. + let overflow = slab.allocate(size, charge(size)).unwrap(); + assert_eq!(live.get(), size * 2); + drop(reused); + drop(overflow); + assert_eq!(slab.idle_bytes(), size); + assert_eq!(live.get(), size); + let larger = slab.allocate(size * 2, charge(size * 2)).unwrap(); + assert_eq!(larger.len(), size * 2); + assert_eq!(live.get(), size * 2); + drop(larger); + assert_eq!(slab.reclaim_idle(), size * 2); + assert_eq!(slab.reclaim_idle(), 0); + assert_eq!(live.get(), 0); + + let outstanding = slab.allocate(size, charge(size)).unwrap(); + drop(slab.allocate(size, charge(size)).unwrap()); + assert_eq!(live.get(), size * 2); + assert_eq!(slab.idle_bytes(), size); + drop(slab); + assert_eq!(live.get(), size); + assert!(outstanding.as_slice().iter().all(|b| *b == 0)); + drop(outstanding); + assert_eq!(live.get(), 0); +} + +/// Bound submissions reject foreign identity and bytes beyond the captured prefix. +#[cfg(feature = "simulation")] +#[test] +fn bound_io_rejects_foreign_tables_and_ranges_appended_after_lease_acquisition() { + let simulation = Simulation::new(); + let _environment = simulation.enter(); + let reactor = Reactor::::new(16, ()); + let slab = Slab::<()>::new("/alloc-workflows/bound".into(), 16384, 8192, 512); + let segments = Segments::new(slab.segment_bytes()); + let alignment = slab.open_configured(&segments).unwrap(); + let other = Segments::from_geometry(segments.geometry().unwrap()).unwrap(); + assert_eq!( + slab.configure_segments(&other), + Err(Error::InvalidConfiguration) + ); + let size = alignment.extent(0, 31).unwrap().length(); + + for write in [false, true] { + let (foreign, extent) = other.append(size).unwrap(); + let buffer = slab.allocate(size, ()).unwrap(); + let operation = if write { + slab.write(&reactor, extent, buffer, foreign, &TestScope) + } else { + slab.read(&reactor, extent, buffer, foreign, &TestScope) + }; + assert!(matches!( + drive(&reactor, operation), + Err(TestError::Alloc(Error::Stale)) + )); + assert_eq!(reactor.in_flight(), 0); + assert_eq!(slab.writes_in_flight(), 0); + } + + let (early, first) = segments.append(size).unwrap(); + let early_read = segments.lease(early.id(), early.generation()).unwrap(); + let (later, second) = segments.append(size).unwrap(); + drop(later); + // Both extents are now used, but the earlier leases only authorize the first. + for (lease, write) in [(early, true), (early_read, false)] { + let buffer = slab.allocate(size, ()).unwrap(); + let operation = if write { + slab.write(&reactor, second, buffer, lease, &TestScope) + } else { + slab.read(&reactor, second, buffer, lease, &TestScope) + }; + assert!(matches!( + drive(&reactor, operation), + Err(TestError::Alloc(Error::Corrupt)) + )); + assert_eq!(reactor.in_flight(), 0); + assert_eq!(slab.writes_in_flight(), 0); + } + let mut buffer = slab.allocate(size, ()).unwrap(); + buffer.as_mut_slice().fill(73); + drop( + drive( + &reactor, + slab.write( + &reactor, + second, + buffer, + segments.lease(SegmentId(0), Generation(1)).unwrap(), + &TestScope, + ), + ) + .unwrap(), + ); + let untouched = drive( + &reactor, + slab.read( + &reactor, + first, + slab.allocate(size, ()).unwrap(), + segments.lease(SegmentId(0), Generation(1)).unwrap(), + &TestScope, + ), + ) + .unwrap(); + assert!(untouched.as_slice().iter().all(|byte| *byte == 0)); + drop(untouched); + let written = drive( + &reactor, + slab.read( + &reactor, + second, + slab.allocate(size, ()).unwrap(), + segments.lease(SegmentId(0), Generation(1)).unwrap(), + &TestScope, + ), + ) + .unwrap(); + assert!(written.as_slice().iter().all(|byte| *byte == 73)); + assert_eq!(reactor.in_flight(), 0); + assert_eq!(slab.writes_in_flight(), 0); +} + +/// Thawing and write completion are independent prerequisites for recovery. +#[cfg(feature = "simulation")] +#[test] +fn restore_waits_for_freeze_guard_and_abandoned_write_completion_independently() { + for cancel_first in [false, true] { + let simulation = Simulation::new(); + simulation.set_cancel_first(cancel_first); + let _environment = simulation.enter(); + let reactor = Reactor::::new(16, ()); + let slab = Slab::<()>::new("/alloc-workflows/freeze".into(), 8192, 4096, 512); + let segments = Segments::new(slab.segment_bytes()); + let alignment = slab.open_configured(&segments).unwrap(); + let size = alignment.extent(0, 31).unwrap().length(); + let (lease, extent) = segments.append(size).unwrap(); + simulation + .inject("write", Fault::HoldCompletion(8)) + .unwrap(); + let mut write = slab.write( + &reactor, + extent, + slab.allocate(size, ()).unwrap(), + lease, + &TestScope, + ); + assert!(poll(&mut write).is_pending()); + reactor.poll_budgeted(1).unwrap(); + drop(write); + assert_eq!(slab.writes_in_flight(), 1); + + let frozen = segments.freeze().unwrap(); + let snapshot = segments.snapshot(); + assert!(matches!(segments.append(size), Err(Error::Busy))); + assert_eq!(segments.restore(snapshot.clone()), Err(Error::Busy)); + drop(frozen); + // Thawing is not a kernel completion fence; the abandoned write owns a lease. + assert_eq!(segments.restore(snapshot.clone()), Err(Error::Busy)); + assert_eq!(segments.snapshot(), snapshot); + let frozen = segments.freeze().unwrap(); + drive(&reactor, slab.fence_writes()).unwrap(); + assert_eq!(slab.writes_in_flight(), 0); + assert_eq!(reactor.in_flight(), 0); + // Conversely, completion does not release a caller's freeze guard. + assert_eq!(segments.restore(snapshot.clone()), Err(Error::Busy)); + drop(frozen); + segments.restore(snapshot).unwrap(); + assert_eq!(segments.state(SegmentId(0)), Ok(SegmentState::Sealed)); + assert_eq!(segments.append(size).unwrap().0.id(), SegmentId(1)); + } +} + +/// Build and bind a deterministic simulated slab with two segments. +#[cfg(feature = "simulation")] +fn setup() -> (Slab, Segments) { + let slab = Slab::new(PathBuf::from("/virtual/slab.dat"), 8192, 4096, 512); + let segments = Segments::new(4096); + let _ = slab.open_configured(&segments).unwrap(); + (slab, segments) +} + +/// Public reads and writes preserve payloads and reject short or failed completions. +#[cfg(feature = "simulation")] +#[test] +fn roundtrip_short_io_and_runtime_errors() { + let sim = Simulation::new(); + let _environment = sim.enter(); + let (slab, segments) = setup::<()>(); + let reactor = Reactor::::new(16, ()); + let (lease, extent) = segments.append(4096).unwrap(); + let mut buffer = slab.allocate(extent.length(), ()).unwrap(); + buffer.as_mut_slice().fill(42); + let buffer = drive( + &reactor, + slab.write(&reactor, extent, buffer, lease, &TestScope), + ) + .unwrap(); + assert_eq!(slab.writes_in_flight(), 0); + drop(buffer); + let read = || { + slab.read( + &reactor, + extent, + slab.allocate(extent.length(), ()).unwrap(), + segments.lease(SegmentId(0), Generation(1)).unwrap(), + &TestScope, + ) + }; + let buffer = drive(&reactor, read()).unwrap(); + assert!(buffer.as_slice().iter().all(|b| *b == 42)); + drop(buffer); + // Kernel-written bytes must dirty an initially clean pool allocation. + let reused = slab.allocate(extent.length(), ()).unwrap(); + assert!(reused.as_slice().iter().all(|b| *b == 0)); + drop(reused); + sim.inject("read", Fault::Short(512)).unwrap(); + assert!(matches!( + drive(&reactor, read()), + Err(TestError::Alloc(Error::Io)) + )); + let reused = slab.allocate(extent.length(), ()).unwrap(); + assert!(reused.as_slice().iter().all(|b| *b == 0)); + drop(reused); + sim.inject("write", Fault::Short(512)).unwrap(); + let op = slab.write( + &reactor, + extent, + slab.allocate(extent.length(), ()).unwrap(), + segments.lease(SegmentId(0), Generation(1)).unwrap(), + &TestScope, + ); + assert!(matches!( + drive(&reactor, op), + Err(TestError::Alloc(Error::Io)) + )); + assert_eq!(slab.writes_in_flight(), 0); + sim.inject("read", Fault::Errno(libc::EIO)).unwrap(); + assert!(matches!( + drive(&reactor, read()), + Err(TestError::Runtime(uring_runtime::Error::Os(libc::EIO))) + )); + assert_eq!(reactor.in_flight(), 0); + segments.begin_evict(SegmentId(0)).unwrap(); + segments.recycle(SegmentId(0)).unwrap(); +} + +/// Abandonment cannot release kernel-visible memory, charges, leases, or write count. +#[cfg(feature = "simulation")] +#[test] +fn abandoned_write_retains_lease_charges_and_counter_until_completion() { + for cancel_first in [false, true] { + let sim = Simulation::new(); + sim.set_cancel_first(cancel_first); + let _environment = sim.enter(); + let (slab, segments) = setup::(); + let reactor = Reactor::::new(16, ()); + let used = Rc::new(Cell::new(0)); + let (lease, extent) = segments.append(4096).unwrap(); + let mut buffer = slab + .allocate(4096, CountingCharge::new(&used, 4096)) + .unwrap(); + buffer.as_mut_slice().fill(42); + let extra = Rc::new(CountingCharge::new(&used, 512)); + let weak = Rc::downgrade(&extra); + buffer.retain(extra); + sim.inject("write", Fault::HoldCompletion(8)).unwrap(); + let mut op = slab.write(&reactor, extent, buffer, lease, &TestScope); + assert!(poll(&mut op).is_pending()); + assert_eq!(slab.writes_in_flight(), 1); + reactor.poll_budgeted(1).unwrap(); + drop(op); + assert_eq!(slab.writes_in_flight(), 1); + assert_eq!(slab.idle_bytes(), 0); + assert_eq!(slab.reclaim_idle(), 0); + assert!(weak.upgrade().is_some()); + assert_eq!(used.get(), 4608); + segments.begin_evict(SegmentId(0)).unwrap(); + assert_eq!(segments.recycle(SegmentId(0)), Err(Error::Busy)); + let mut fence = slab.fence_writes(); + assert!(poll(&mut fence).is_pending()); + drive(&reactor, fence).unwrap(); + assert_eq!(slab.writes_in_flight(), 0); + assert_eq!(reactor.in_flight(), 0); + assert!(weak.upgrade().is_none()); + assert_eq!(used.get(), 4096); + assert_eq!(slab.idle_bytes(), 4096); + let reused = slab + .allocate(4096, CountingCharge::new(&used, 4096)) + .unwrap(); + assert!(reused.as_slice().iter().all(|b| *b == 0)); + drop(reused); + assert_eq!(slab.reclaim_idle(), 4096); + assert_eq!(used.get(), 0); + segments.recycle(SegmentId(0)).unwrap(); + } +} + +/// An unpolled write owns no runtime work, but an abandoned accepted read still does. +#[cfg(feature = "simulation")] +#[test] +fn abandoned_read_and_unpolled_write_release_at_the_correct_fence() { + let sim = Simulation::new(); + let _environment = sim.enter(); + let (slab, segments) = setup::<()>(); + let reactor = Reactor::::new(16, ()); + let (lease, extent) = segments.append(4096).unwrap(); + let op = slab.write( + &reactor, + extent, + slab.allocate(4096, ()).unwrap(), + lease, + &TestScope, + ); + drop(op); + assert_eq!(slab.writes_in_flight(), 0); + assert_eq!(reactor.in_flight(), 0); + // Seed nonzero bytes so a completed but abandoned read tests erasure, not + // merely the lifetime of an allocation that happens to remain zero. + let mut buffer = slab.allocate(4096, ()).unwrap(); + buffer.as_mut_slice().fill(42); + drop( + drive( + &reactor, + slab.write( + &reactor, + extent, + buffer, + segments.lease(SegmentId(0), Generation(1)).unwrap(), + &TestScope, + ), + ) + .unwrap(), + ); + sim.inject("read", Fault::HoldCompletion(8)).unwrap(); + let mut op = slab.read( + &reactor, + extent, + slab.allocate(4096, ()).unwrap(), + segments.lease(SegmentId(0), Generation(1)).unwrap(), + &TestScope, + ); + assert!(poll(&mut op).is_pending()); + reactor.poll_budgeted(1).unwrap(); + drop(op); + assert_eq!(slab.idle_bytes(), 0); + assert_eq!(slab.reclaim_idle(), 0); + segments.begin_evict(SegmentId(0)).unwrap(); + assert_eq!(segments.recycle(SegmentId(0)), Err(Error::Busy)); + drive(&reactor, reactor.drain()).unwrap(); + let reused = slab.allocate(4096, ()).unwrap(); + assert!(reused.as_slice().iter().all(|b| *b == 0)); + segments.recycle(SegmentId(0)).unwrap(); +} + +/// Simulation preserves file locks, size validation, and synchronous error categories. +#[cfg(feature = "simulation")] +#[test] +fn simulation_open_lock_size_and_faults() { + let sim = Simulation::new(); + let _environment = sim.enter(); + let (slab, _) = setup::<()>(); + let conflicting = Slab::<()>::new(PathBuf::from("/virtual/slab.dat"), 8192, 4096, 512); + assert_eq!(conflicting.open_now(), Err(Error::Unavailable)); + assert_eq!(slab.open_now(), slab.alignment()); + drop(slab); + let _ = conflicting.open_now().unwrap(); + drop(conflicting); + let wrong_size = Slab::<()>::new(PathBuf::from("/virtual/slab.dat"), 16384, 4096, 512); + assert_eq!(wrong_size.open_now(), Err(Error::InvalidConfiguration)); + sim.inject("open", Fault::Errno(libc::EOPNOTSUPP)).unwrap(); + assert_eq!(wrong_size.open_now(), Err(Error::Unsupported)); + sim.inject("open", Fault::Errno(libc::ENOSPC)).unwrap(); + assert_eq!( + wrong_size.open_now(), + Err(Error::SystemIo { + operation: "open", + errno: Some(libc::ENOSPC) + }) + ); +} + +/// Fault hooks cannot swap descriptors while a write completion owns the old file. +#[cfg(feature = "simulation")] +#[test] +fn replacement_hooks_are_fallible_and_refuse_live_writes() { + let sim = Simulation::new(); + let _environment = sim.enter(); + let replacement = || { + sim.open( + None, + Path::new("/replacement"), + libc::O_CREAT | libc::O_RDWR, + ) + .unwrap() + }; + let unopened = Slab::<()>::new(PathBuf::from("/unopened"), 8192, 4096, 512); + assert_eq!( + unopened.replace_descriptor_for_test(replacement()), + Err(Error::Unavailable) + ); + let (slab, segments) = setup::<()>(); + let reactor = Reactor::::new(16, ()); + let (lease, extent) = segments.append(4096).unwrap(); + sim.inject("write", Fault::HoldCompletion(8)).unwrap(); + let mut write = slab.write( + &reactor, + extent, + slab.allocate(4096, ()).unwrap(), + lease, + &TestScope, + ); + assert!(poll(&mut write).is_pending()); + assert_eq!( + slab.replace_descriptor_for_test(replacement()), + Err(Error::Busy) + ); + drop(write); + drive(&reactor, slab.fence_writes()).unwrap(); + slab.replace_descriptor_for_test(replacement()).unwrap(); +} + +/// Owns a unique project-local directory for real kernel workflows. +struct Directory(PathBuf); + +impl Directory { + /// Create an isolated test directory without using the host temporary directory. + fn new() -> Self { + static NEXT: std::sync::atomic::AtomicUsize = std::sync::atomic::AtomicUsize::new(0); + let id = NEXT.fetch_add(1, std::sync::atomic::Ordering::Relaxed); + let path = PathBuf::from(env!("CARGO_MANIFEST_DIR")) + .join("target") + .join(format!("page-alloc-workflow-{}-{id}", std::process::id())); + std::fs::create_dir_all(&path).unwrap(); + Self(path) + } +} + +impl Drop for Directory { + /// Clean up only this test's owned directory. + fn drop(&mut self) { + let _ = std::fs::remove_dir_all(&self.0); + } +} + +/// Report explicit capability skips or fail when real-kernel coverage is required. +fn capability_skip(reason: &str) { + assert_ne!( + std::env::var("PAGE_ALLOC_REQUIRE_REAL_IO").as_deref(), + Ok("1"), + "PAGE_ALLOC_REQUIRE_REAL_IO=1 forbids capability skips: {reason}" + ); + eprintln!("SKIP real slab test: {reason}"); +} + +/// Only explicit lack of direct-I/O support may skip real file workflows. +fn real_alignment(result: page_alloc::Result) -> Option { + match result { + Ok(alignment) => Some(alignment), + Err(Error::Unsupported) => { + capability_skip("filesystem does not support direct I/O geometry"); + None + } + Err(error) => panic!("unexpected real slab startup failure: {error}"), + } +} + +/// Discover alignment before choosing any slab dimensions. +fn probe_alignment(directory: &Directory) -> Option { + use std::os::{fd::AsRawFd, unix::fs::OpenOptionsExt}; + + let result = (|| { + let file = std::fs::OpenOptions::new() + .create_new(true) + .read(true) + .write(true) + .mode(0o600) + .custom_flags(libc::O_DIRECT) + .open(directory.0.join("alignment-probe")) + .map_err(|error| match error.raw_os_error() { + Some(libc::EINVAL | libc::EOPNOTSUPP | libc::ENOSYS) => Error::Unsupported, + errno => Error::SystemIo { + operation: "open", + errno, + }, + })?; + // SAFETY: stat is initialized and the file and empty path remain live. + let mut stat: libc::statx = unsafe { std::mem::zeroed() }; + if unsafe { + libc::statx( + file.as_raw_fd(), + c"".as_ptr(), + libc::AT_EMPTY_PATH, + libc::STATX_DIOALIGN, + &mut stat, + ) + } != 0 + { + return Err(match std::io::Error::last_os_error().raw_os_error() { + Some(libc::ENOSYS | libc::EOPNOTSUPP) => Error::Unsupported, + errno => Error::SystemIo { + operation: "statx", + errno, + }, + }); + } + if stat.stx_mask & libc::STATX_DIOALIGN == 0 { + return Err(Error::Unsupported); + } + Alignment::new( + stat.stx_dio_mem_align as usize, + stat.stx_dio_offset_align as u64, + stat.stx_dio_offset_align as usize, + ) + })(); + real_alignment(result) +} + +/// Probe baseline io_uring support without hiding unexpected runtime setup failures. +fn kernel_available() -> bool { + let mut params = [0u64; 15]; + // SAFETY: initialized, aligned 120-byte UAPI output; success owns a new descriptor. + let fd = unsafe { libc::syscall(libc::SYS_io_uring_setup, 2u32, params.as_mut_ptr()) }; + if fd >= 0 { + drop(unsafe { File::from_raw_fd(fd as i32) }); + return true; + } + let error = std::io::Error::last_os_error(); + match error.raw_os_error() { + Some(libc::ENOSYS | libc::EOPNOTSUPP | libc::EPERM | libc::EACCES) => { + capability_skip(&format!( + "io_uring kernel capability/permission error: {error}" + )); + false + } + _ => panic!("unexpected io_uring setup failure: {error}"), + } +} + +/// Drive real I/O with a strict in-test deadline as well as the external test timeout. +fn drive_real( + reactor: &Reactor, + mut op: Operation<'_, T, TestError>, +) -> Result { + let deadline = std::time::Instant::now() + std::time::Duration::from_secs(10); + loop { + if let Poll::Ready(value) = poll(&mut op) { + return value; + } + reactor.poll_budgeted(64).unwrap(); + assert!( + std::time::Instant::now() < deadline, + "real slab I/O did not complete" + ); + std::thread::yield_now(); + } +} + +/// The real kernel preserves payload bytes and releases write completion authority. +#[test] +fn io_uring_roundtrip_and_completion_fence() { + if !kernel_available() { + return; + } + let directory = Directory::new(); + let Some(alignment) = probe_alignment(&directory) else { + return; + }; + let (_, segment_bytes) = device_placement_sizes(alignment); + let slab = Slab::<()>::new( + directory.0.join("uring.dat"), + 2 * segment_bytes, + segment_bytes, + segment_bytes as usize, + ); + let segments = Segments::new(segment_bytes); + assert_eq!(slab.open_configured(&segments), Ok(alignment)); + let reactor = Reactor::::new(16, ()); + reactor.init().unwrap(); + let (lease, extent) = segments.append(segment_bytes as usize).unwrap(); + let mut buffer = slab.allocate(extent.length(), ()).unwrap(); + buffer.as_mut_slice().fill(73); + drop( + drive_real( + &reactor, + slab.write(&reactor, extent, buffer, lease, &TestScope), + ) + .unwrap(), + ); + assert_eq!(slab.writes_in_flight(), 0); + let mut fence = slab.fence_writes(); + assert_eq!(poll(&mut fence), Poll::Ready(Ok(()))); + let buffer = drive_real( + &reactor, + slab.read( + &reactor, + extent, + slab.allocate(extent.length(), ()).unwrap(), + segments.lease(SegmentId(0), Generation(1)).unwrap(), + &TestScope, + ), + ) + .unwrap(); + assert!(buffer.as_slice().iter().all(|byte| *byte == 73)); + drop(buffer); + assert_eq!(reactor.in_flight(), 0); + /// A completed roundtrip has no caller mappings left to remove. + struct EmptyEntries; + + impl page_alloc::SegmentEntries for EmptyEntries { + /// The smoke workflow publishes no index entries. + fn remove_bounded(&self, _: SegmentId, _: usize) -> usize { + 0 + } + + /// Both slots are logically unmapped. + fn is_empty(&self, _: SegmentId) -> bool { + true + } + } + let clock = page_alloc::SegmentClock::new(std::rc::Rc::new(segments)); + clock.reclaim(&EmptyEntries, 2, 4, 0).unwrap(); +} + +/// Keep both halves aligned even when offset and length units differ. +fn device_placement_sizes(alignment: Alignment) -> (usize, u64) { + let half_bytes = alignment.extent(0, 2048).unwrap().length(); + let segment_bytes = 2 * half_bytes as u64; + (half_bytes, segment_bytes) +} + +/// Placement test geometry must work without relying on the host's alignment. +#[test] +fn device_placement_sizes_support_4096_and_arbitrary_units() { + for (offset, length, expected_half) in [ + (1, 1, 2048), + (512, 512, 2048), + (4096, 4096, 4096), + (4096, 512, 4096), + (512, 4096, 4096), + (8192, 8192, 8192), + (8192, 512, 8192), + (512, 8192, 8192), + (65536, 65536, 65536), + (768, 512, 3072), + (3, 5, 2055), + ] { + let alignment = Alignment::new(4096, offset, length).unwrap(); + let (half_bytes, segment_bytes) = device_placement_sizes(alignment); + assert_eq!(half_bytes, expected_half); + assert_eq!(segment_bytes, 2 * expected_half as u64); + let geometry = + page_alloc::SegmentGeometry::new(3 * segment_bytes, segment_bytes, 3, alignment) + .unwrap(); + let segments = Segments::from_geometry(geometry).unwrap(); + let buffer = alignment.allocate(half_bytes, ()).unwrap(); + if half_bytes != 2048 { + assert!(matches!( + segments.append(2048), + Err(Error::InvalidConfiguration) + )); + } + for (id, physical_start) in [2 * segment_bytes, segment_bytes, 0] + .into_iter() + .enumerate() + { + for half in 0..2 { + let (lease, extent) = segments.append(half_bytes).unwrap(); + assert_eq!(lease.id(), SegmentId(id as u64)); + assert_eq!(extent.length(), half_bytes); + assert_eq!( + extent.offset(), + id as u64 * segment_bytes + half * half_bytes as u64 + ); + alignment.check(extent, &buffer).unwrap(); + let physical = + page_alloc::Extent::new(physical_start + half * half_bytes as u64, half_bytes) + .unwrap(); + alignment.check(physical, &buffer).unwrap(); + } + } + assert_eq!(segments.free_count(), 0); + let whole_segments = Segments::from_geometry(geometry).unwrap(); + let buffer = alignment.allocate(segment_bytes as usize, ()).unwrap(); + for id in 0..3 { + let (lease, extent) = whole_segments.append(segment_bytes as usize).unwrap(); + assert_eq!(lease.id(), SegmentId(id)); + assert_eq!(extent.offset(), id * segment_bytes); + alignment.check(extent, &buffer).unwrap(); + } + } +} + +/// Device placements translate whole segments and interior extents across real files. +#[test] +fn device_placements_route_real_io_and_reject_short_reads() { + use std::os::unix::fs::{FileExt, OpenOptionsExt}; + + if !kernel_available() { + return; + } + let directory = Directory::new(); + let Some(alignment) = probe_alignment(&directory) else { + return; + }; + let (half_bytes, segment_bytes) = device_placement_sizes(alignment); + let file_bytes = 4 * segment_bytes; + let files: Vec<_> = ["one", "two"] + .into_iter() + .map(|name| { + let file = std::fs::OpenOptions::new() + .create_new(true) + .read(true) + .write(true) + .custom_flags(libc::O_DIRECT) + .open(directory.0.join(name)) + .unwrap(); + file.set_len(file_bytes).unwrap(); + Arc::new(file) + }) + .collect(); + let slab = Slab::<()>::from_devices( + vec![ + DevicePlacement { + file: files[0].clone(), + offset: 2 * segment_bytes, + }, + DevicePlacement { + file: files[1].clone(), + offset: segment_bytes, + }, + DevicePlacement { + file: files[0].clone(), + offset: 0, + }, + ], + segment_bytes, + segment_bytes as usize, + alignment, + ) + .unwrap(); + assert_eq!(slab.capacity_bytes(), 3 * segment_bytes); + assert_eq!(slab.alignment(), Err(Error::Unavailable)); + let segments = Segments::new(segment_bytes); + assert_eq!(slab.open_configured(&segments), Ok(alignment)); + let reactor = Reactor::::new(16, ()); + reactor.init().unwrap(); + for id in 0..3 { + for half in 0..2 { + let (lease, extent) = segments.append(half_bytes).unwrap(); + assert_eq!(lease.id(), SegmentId(id)); + assert_eq!( + extent.offset(), + id * segment_bytes + half * half_bytes as u64 + ); + let mut buffer = slab.allocate(half_bytes, ()).unwrap(); + buffer.as_mut_slice().fill((id * 2 + half + 1) as u8); + drop( + drive_real( + &reactor, + slab.write(&reactor, extent, buffer, lease, &TestScope), + ) + .unwrap(), + ); + let read = drive_real( + &reactor, + slab.read( + &reactor, + extent, + slab.allocate(half_bytes, ()).unwrap(), + segments.lease(SegmentId(id), Generation(1)).unwrap(), + &TestScope, + ), + ) + .unwrap(); + assert!( + read.as_slice() + .iter() + .all(|&byte| byte == (id * 2 + half + 1) as u8) + ); + } + } + // Inspect physical addresses independently so symmetric read/write bugs cannot pass. + for (file, offset, value) in [ + (0, 0, 5), + (0, half_bytes as u64, 6), + (0, segment_bytes, 0), + (0, 2 * segment_bytes, 1), + (0, 2 * segment_bytes + half_bytes as u64, 2), + (0, 3 * segment_bytes, 0), + (1, 0, 0), + (1, segment_bytes, 3), + (1, segment_bytes + half_bytes as u64, 4), + (1, 2 * segment_bytes, 0), + ] { + let mut buffer = alignment.allocate(half_bytes, ()).unwrap(); + assert_eq!( + files[file].read_at(buffer.as_mut_slice(), offset).unwrap(), + half_bytes + ); + assert!(buffer.as_slice().iter().all(|&byte| byte == value)); + assert_eq!(files[file].metadata().unwrap().len(), file_bytes); + } + // External truncation violates the startup contract but must still fail closed. + files[1].set_len(segment_bytes).unwrap(); + let read = slab.read( + &reactor, + page_alloc::Extent::new(segment_bytes, segment_bytes as usize).unwrap(), + slab.allocate(segment_bytes as usize, ()).unwrap(), + segments.lease(SegmentId(1), Generation(1)).unwrap(), + &TestScope, + ); + assert!(matches!( + drive_real(&reactor, read), + Err(TestError::Alloc(Error::Io)) + )); + assert_eq!(reactor.in_flight(), 0); + assert_eq!(slab.writes_in_flight(), 0); +} + +/// Failed binding retains the open file but never grants admission to the reactor. +#[test] +fn unbound_and_failed_binding_reject_reads_and_writes_before_reactor_admission() { + let directory = Directory::new(); + let Some(alignment) = probe_alignment(&directory) else { + return; + }; + let (_, segment_bytes) = device_placement_sizes(alignment); + #[cfg(feature = "simulation")] + let failures = ["unbound", "wrong-geometry", "frozen-table"].as_slice(); + #[cfg(not(feature = "simulation"))] + let failures = ["wrong-geometry", "frozen-table"].as_slice(); + for &failure in failures { + let slab = Slab::<()>::new( + directory.0.join(failure), + 2 * segment_bytes, + segment_bytes, + segment_bytes as usize, + ); + let correct = Segments::new(segment_bytes); + match failure { + #[cfg(feature = "simulation")] + "unbound" => { + assert_eq!(slab.open_now(), Ok(alignment)); + } + "wrong-geometry" => { + let result = slab.open_configured(&Segments::new(2 * segment_bytes)); + assert_eq!(result, Err(Error::InvalidConfiguration)); + } + "frozen-table" => { + let frozen = correct.freeze().unwrap(); + let result = slab.open_configured(&correct); + assert_eq!(result, Err(Error::Busy)); + drop(frozen); + } + _ => unreachable!(), + } + assert_eq!(slab.alignment(), Ok(alignment)); + let correct = Segments::from_geometry(slab.geometry().unwrap()).unwrap(); + let (lease, extent) = correct.append(segment_bytes as usize).unwrap(); + let buffer = slab.allocate(extent.length(), ()).unwrap(); + let reactor = Reactor::::new(16, ()); + let mut read = slab.read(&reactor, extent, buffer, lease, &TestScope); + assert!(matches!( + poll(&mut read), + Poll::Ready(Err(TestError::Alloc(Error::Unavailable))) + )); + drop(read); + let mut write = slab.write( + &reactor, + extent, + slab.allocate(extent.length(), ()).unwrap(), + correct.lease(SegmentId(0), Generation(1)).unwrap(), + &TestScope, + ); + assert!(matches!( + poll(&mut write), + Poll::Ready(Err(TestError::Alloc(Error::Unavailable))) + )); + drop(write); + assert_eq!(reactor.in_flight(), 0); + assert_eq!(slab.writes_in_flight(), 0); + assert_eq!(slab.open_configured(&correct), Ok(alignment)); + // Acquiring the same binding again proves recovery is stable without kernel admission. + assert_eq!(slab.open_configured(&correct), Ok(alignment)); + // The successful roundtrip workflow separately proves bound admission and completion. + } +} + +/// Partial tables preserve file geometry, lease-fenced reuse, and unchanged disk bytes. +#[cfg(feature = "simulation")] +#[test] +fn partial_table_reclamation_changes_authority_without_erasing_storage() { + let simulation = Simulation::new(); + let _environment = simulation.enter(); + let reactor = Reactor::::new(16, ()); + let slab = Slab::<()>::new("/alloc-workflows/partial".into(), 16384, 4096, 512); + let alignment = slab.open_now().unwrap(); + let physical = slab.geometry().unwrap(); + let partial = page_alloc::SegmentGeometry::new(16384, 4096, 2, alignment).unwrap(); + let segments = Rc::new(Segments::from_geometry(partial).unwrap()); + assert_eq!(slab.open_configured(&segments), Ok(alignment)); + assert_eq!(slab.geometry().unwrap(), physical); + assert_eq!(physical.segment_count(), 4); + assert_eq!(segments.count(), 2); + assert_eq!(segments.geometry(), Some(partial)); + + let (lease, extent) = segments.append(4096).unwrap(); + let original_generation = lease.generation(); + let mut buffer = slab.allocate(4096, ()).unwrap(); + buffer.as_mut_slice().fill(91); + drop( + drive( + &reactor, + slab.write(&reactor, extent, buffer, lease, &TestScope), + ) + .unwrap(), + ); + assert_eq!(segments.free_count(), 1); + + /// One caller mapping whose removal is observable independently of disk bytes. + struct Entries { + populated: Cell, + + removals: Cell, + } + + impl page_alloc::SegmentEntries for Entries { + /// Forget the mapping only for its actual segment and a positive budget. + fn remove_bounded(&self, segment: SegmentId, budget: usize) -> usize { + if segment == SegmentId(0) && budget != 0 && self.populated.replace(false) { + self.removals.set(self.removals.get() + 1); + 1 + } else { + 0 + } + } + + /// Report current caller ownership rather than allocator occupancy. + fn is_empty(&self, segment: SegmentId) -> bool { + segment != SegmentId(0) || !self.populated.get() + } + } + let entries = Entries { + populated: Cell::new(true), + removals: Cell::new(0), + }; + let clock = page_alloc::SegmentClock::new(segments.clone()); + let held = segments.lease(SegmentId(0), original_generation).unwrap(); + assert_eq!(clock.reclaim(&entries, 2, 4, 1), Err(Error::Busy)); + assert_eq!(entries.removals.get(), 1); + assert!(!entries.populated.get()); + assert_eq!(segments.state(SegmentId(0)), Ok(SegmentState::Evicting)); + assert!(matches!( + segments.lease(SegmentId(0), original_generation), + Err(Error::Stale) + )); + + // Existing completion authority still reads bytes after caller index removal. + let buffer = drive( + &reactor, + slab.read( + &reactor, + extent, + slab.allocate(4096, ()).unwrap(), + held, + &TestScope, + ), + ) + .unwrap(); + assert!(buffer.as_slice().iter().all(|byte| *byte == 91)); + drop(buffer); + clock.reclaim(&entries, 2, 4, 0).unwrap(); + assert_eq!(segments.free_count(), 2); + assert_eq!(entries.removals.get(), 1); + assert_eq!(segments.snapshot()[0].generation, Generation(2)); + + // Reuse grants new generation authority, but does not mutate the stored bytes. + let (replacement, replacement_extent) = segments.append(4096).unwrap(); + assert_eq!(replacement.id(), SegmentId(0)); + assert_eq!(replacement.generation(), Generation(2)); + assert_eq!(replacement_extent, extent); + assert_eq!( + segments.validate(SegmentId(0), original_generation, &extent), + Err(Error::Stale) + ); + let buffer = drive( + &reactor, + slab.read( + &reactor, + replacement_extent, + slab.allocate(4096, ()).unwrap(), + replacement, + &TestScope, + ), + ) + .unwrap(); + assert!(buffer.as_slice().iter().all(|byte| *byte == 91)); + drop(buffer); + assert_eq!(slab.writes_in_flight(), 0); + assert_eq!(reactor.in_flight(), 0); + assert_eq!(slab.geometry().unwrap(), physical); + let image = segments.snapshot(); + assert_eq!(image.len(), 2); + assert_eq!(image[0].state, SegmentState::Sealed); + assert_eq!(image[0].used_bytes, 4096); + assert_eq!(image[0].generation, Generation(2)); + assert_eq!(image[1].state, SegmentState::Free); + assert_eq!(image[1].used_bytes, 0); + assert_eq!(image[1].generation, Generation(1)); + assert_eq!(entries.removals.get(), 1); + assert_eq!(segments.free_count(), 1); +} diff --git a/cmd/racer-dataplane/flow/.gitignore b/cmd/racer-dataplane/flow/.gitignore new file mode 100644 index 000000000..b83d22266 --- /dev/null +++ b/cmd/racer-dataplane/flow/.gitignore @@ -0,0 +1 @@ +/target/ diff --git a/cmd/racer-dataplane/flow/Cargo.toml b/cmd/racer-dataplane/flow/Cargo.toml new file mode 100644 index 000000000..6794e87d0 --- /dev/null +++ b/cmd/racer-dataplane/flow/Cargo.toml @@ -0,0 +1,17 @@ +[package] +name = "flow-control" +version = "0.1.0" +edition = "2024" +license = "Apache-2.0" +publish = false +description = "Policy-driven quotas, keyed coalescing, completion-owned flights, and bounded I/O resources" + +[features] +simulation = ["uring-runtime/simulation"] + +[dependencies] +futures = "0.3" +libc = "0.2" +zeroize = "1" +page-alloc = { path = "../alloc" } +uring-runtime = { path = "../runtime" } diff --git a/cmd/racer-dataplane/flow/src/admission.rs b/cmd/racer-dataplane/flow/src/admission.rs new file mode 100644 index 000000000..5351d8d93 --- /dev/null +++ b/cmd/racer-dataplane/flow/src/admission.rs @@ -0,0 +1,2617 @@ +//! Distinct admission mechanisms for peer recovery, handoff, and speculation. +//! +//! Adaptive permits retain shared capacity until their last owner leaves. Endpoint +//! circuits instead own local half-open probes, whose exclusivity survives timeout +//! and successful observation. Handoff envelopes retain target reservations while +//! queued and after dequeue. Hedge alarms never release their capacity: owners +//! retain permits through both contenders' completion fences. +use crate::{Error, Result}; +use std::{ + collections::BTreeMap, + sync::{Arc, Mutex}, + time::{Duration, Instant}, +}; + +pub use circuit::{Circuits, Probe}; +pub use handoff::{Admission as HandoffAdmission, Admitted, Handoff, Offer}; +pub use hedge::{Hedges, Permit as HedgePermit}; + +/// A retry or timeout beyond the clock's range never expires. +#[derive(Clone, Copy)] +enum Deadline { + At(Instant), + + Never, +} + +impl Deadline { + /// Keep an unrepresentable deadline distinct from an absent delay. + fn after(now: Instant, delay: Duration) -> Self { + now.checked_add(delay).map_or(Self::Never, Self::At) + } + + /// Only finite deadlines can become eligible. + fn elapsed(self, now: Instant) -> bool { + matches!(self, Self::At(at) if now >= at) + } +} + +/// Limits and time intervals for shared adaptive admission. +#[derive(Clone, Copy)] +pub struct Config { + /// Maximum aggregate active operations before adaptive reductions. + pub total: usize, + + /// Maximum active operations for one key before failure reductions. + pub per_key: usize, + + /// Maximum retained peer records, including idle backoff history. + pub capacity: usize, + + /// Peer retry delay and minimum interval between aggregate pressure reductions. + /// A retry beyond the clock's range keeps the peer circuit open indefinitely. + pub backoff: Duration, + + /// Minimum interval between one-slot verified-success recoveries. + pub recovery: Duration, + + /// Minimum idle age before an eligible peer record can be replaced. + pub retire_after: Duration, +} + +/// Caller-classified evidence, independent of application errors. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum Outcome { + /// Authenticated successful work provides recovery evidence. + Verified, + + /// A failure attributable to the immediate peer opens its circuit. + PeerFailure, + + /// Local overload reduces aggregate admission without blaming the peer. + LocalPressure, + + /// No evidence should affect admission or peer recovery. + Neutral, +} + +/// Admission observations emitted synchronously in state-transition order. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum Event { + /// Aggregate, per-key, or record capacity rejected the attempt. + Rejected, + + /// Peer backoff or an owned probe rejected the attempt. + CircuitRejected, + + /// An operation now owns active admission. + Accepted, + + /// An accepted operation owns the exclusive recovery probe. + Probe, + + /// Caller-reported local pressure was observed. + LocalPressure, + + /// Caller-reported verified success was observed. + Verified, + + /// Caller-reported immediate-link failure was observed. + LinkFailure, +} + +/// Callbacks run synchronously under the admission lock and must not reenter it. +pub trait Observer { + /// Record an admission or outcome event without reentering the owner. + fn event(&self, event: Event); + + /// Publish the current active count without reentering the owner. + fn active(&self, active: usize); + + /// Publish the current adaptive limit without reentering the owner. + fn limit(&self, limit: usize); +} + +/// Shared total and per-key admission with generation-fenced peer recovery. +pub struct Adaptive { + config: Config, + + state: Mutex>, + + observer: O, + + now: fn() -> Instant, +} + +/// Mutex-protected aggregate admission and bounded peer records. +struct State { + active: usize, + + limit: usize, + + updated: Instant, + + peers: BTreeMap, +} + +/// Per-key recovery state retained while any permit remains active. +struct Peer { + active: usize, + + limit: usize, + + /// Exhaustion blocks peer evidence and admission until all permits drain. + generation: Option, + + retry: Option, + + probe: bool, + + updated: Instant, + + /// Start of the current idle interval, independent of recovery throttling. + idle_since: Option, +} + +/// An admitted operation; the final shared owner releases capacity. +pub struct Permit { + owner: Arc>, + + key: K, + + generation: u64, + + probe: bool, +} + +impl Adaptive { + /// Validate positive limits and publish the initial aggregate ceiling. + pub fn new(config: Config, observer: O, now: fn() -> Instant) -> Result> { + if config.total == 0 + || config.per_key == 0 + || config.per_key > config.total + || config.capacity == 0 + { + return Err(Error::InvalidInput); + } + observer.limit(config.total); + Ok(Arc::new(Self { + config, + observer, + now, + state: Mutex::new(State { + active: 0, + limit: config.total, + updated: now(), + peers: BTreeMap::new(), + }), + })) + } + + /// Hint whether fully recovered limits leave spare speculative capacity. + pub fn hedge_available(&self, key: &K) -> bool { + self.state.lock().is_ok_and(|s| { + s.limit == self.config.total + && s.active.saturating_add(1) < s.limit + && s.peers.get(key).is_none_or(|p| { + p.retry.is_none() + && !p.probe + && p.limit == self.config.per_key + && p.active < p.limit + }) + }) + } + + /// Hint whether peer backoff permits an attempt; this does not reserve capacity. + pub fn available(&self, key: &K) -> bool { + let now = (self.now)(); + self.state.lock().is_ok_and(|s| { + s.peers.get(key).is_none_or(|p| { + p.generation.is_some() && !p.probe && p.retry.is_none_or(|at| at.elapsed(now)) + }) + }) + } + + /// Admit work or one exclusive recovery probe without waiting. + pub fn acquire(self: &Arc, key: &K) -> Result>> { + let indexed = key.clone(); + let owned = key.clone(); + let now = (self.now)(); + // Drop retired keys after the guard on every exit, since Drop may reenter. + let retired; + let mut state = self.state.lock().map_err(|_| Error::Unavailable)?; + if state.active >= state.limit { + self.observer.event(Event::Rejected); + return Err(Error::Overloaded); + } + if !state.peers.contains_key(key) && state.peers.len() == self.config.capacity { + retired = state + .peers + .extract_if(.., |_, p| { + p.active == 0 + && !p.probe + && p.retry.is_none_or(|retry| retry.elapsed(now)) + && p.idle_since.is_some_and(|since| { + now.saturating_duration_since(since) >= self.config.retire_after + }) + }) + .next(); + if retired.is_none() { + self.observer.event(Event::Rejected); + return Err(Error::Overloaded); + } + } + let peer = state.peers.entry(indexed).or_insert(Peer { + active: 0, + limit: self.config.per_key, + generation: Some(0), + retry: None, + probe: false, + updated: now, + idle_since: None, + }); + if peer.generation.is_none() || peer.probe || peer.retry.is_some_and(|at| !at.elapsed(now)) + { + self.observer.event(Event::CircuitRejected); + return Err(Error::Unavailable); + } + if peer.active >= peer.limit { + self.observer.event(Event::Rejected); + return Err(Error::Overloaded); + } + let probe = peer.retry.is_some(); + peer.probe = probe; + peer.active += 1; + peer.idle_since = None; + let generation = peer + .generation + .expect("admission rejects exhausted generations"); + state.active += 1; + self.observer.active(state.active); + self.observer.event(Event::Accepted); + if probe { + self.observer.event(Event::Probe); + } + Ok(Arc::new(Permit { + owner: self.clone(), + key: owned, + generation, + probe, + })) + } +} + +impl Permit { + /// Apply caller evidence while retaining admission through completion. + pub fn observe(&self, outcome: Outcome) { + let now = (self.owner.now)(); + let Ok(mut state) = self.owner.state.lock() else { + return; + }; + let config = self.owner.config; + if outcome == Outcome::LocalPressure { + self.owner.observer.event(Event::LocalPressure); + if now.saturating_duration_since(state.updated) >= config.backoff { + state.limit = (state.limit / 2).max(1); + state.updated = now; + self.owner.observer.limit(state.limit); + } + return; + } + if outcome == Outcome::Verified + && state.limit < config.total + && now.saturating_duration_since(state.updated) >= config.recovery + { + state.limit = state.limit.saturating_add(1).min(config.total); + state.updated = now; + self.owner.observer.limit(state.limit); + } + let peer = state + .peers + .get_mut(&self.key) + .expect("live permit retains key"); + let event = match outcome { + Outcome::Verified => Event::Verified, + Outcome::PeerFailure => Event::LinkFailure, + Outcome::LocalPressure => Event::LocalPressure, + Outcome::Neutral => return, + }; + self.owner.observer.event(event); + if peer.generation != Some(self.generation) { + return; + } + match outcome { + Outcome::PeerFailure => { + peer.limit = (peer.limit / 2).max(1); + peer.generation = self.generation.checked_add(1); + peer.retry = Some(Deadline::after(now, config.backoff)); + peer.updated = now; + } + Outcome::Verified => { + if peer.retry.is_some() && !self.probe { + return; + } + peer.retry = None; + if now.saturating_duration_since(peer.updated) >= config.recovery { + peer.limit = peer.limit.saturating_add(1).min(config.per_key); + peer.updated = now; + } + } + Outcome::LocalPressure | Outcome::Neutral => {} + } + } +} + +impl Drop for Permit { + /// Release final ownership and renew backoff for an unsuccessful probe. + fn drop(&mut self) { + let now = (self.owner.now)(); + let Ok(mut state) = self.owner.state.lock() else { + return; + }; + let peer = state + .peers + .get_mut(&self.key) + .expect("live permit retains key"); + peer.active -= 1; + if peer.active == 0 { + // No live permit can carry a reused generation after this point. + peer.generation.get_or_insert(0); + peer.idle_since = Some(now); + } + if self.probe { + peer.probe = false; + if peer.retry.is_some() { + peer.retry = Some(Deadline::after(now, self.owner.config.backoff)); + peer.updated = now; + } + } + state.active -= 1; + self.owner.observer.active(state.active); + } +} + +/// Bounded worker-local endpoint failure tracking, independent of adaptive limits. +mod circuit { + use super::Deadline; + use crate::{Error, Result}; + use std::{ + cell::RefCell, + collections::{BTreeMap, BTreeSet}, + marker::PhantomData, + rc::Rc, + time::{Duration, Instant}, + }; + + /// Local endpoint health with bounded records and completion-owned probes. + /// + /// The authority belongs to one worker even when keys themselves are shared: + /// + /// ```compile_fail + /// use flow_control::Circuits; + /// let circuits = Circuits::::new(1, std::time::Duration::ZERO); + /// std::thread::spawn(move || drop(circuits)); + /// ``` + /// + /// ```compile_fail + /// use flow_control::Circuits; + /// fn require_sync() {} + /// require_sync::>(); + /// ``` + pub struct Circuits { + capacity: usize, + + probe_timeout: Duration, + + states: RefCell>, + + probes: RefCell>, + + local: PhantomData>, + } + + /// Failure history and eligibility for one endpoint. + struct Circuit { + failures: u32, + + retry_at: Deadline, + + probe_until: Option, + + pending_backoff: Option>, + } + + impl Circuit { + /// Require both retry backoff and any abandoned probe timeout to expire. + fn available(&self, now: Instant) -> bool { + self.retry_at.elapsed(now) && self.probe_until.is_none_or(|until| until.elapsed(now)) + } + } + + /// Local exclusive half-open ownership; healthy acquisitions need no record. + pub struct Probe<'a, K: Ord> { + health: &'a Circuits, + + key: Option, + } + + impl Drop for Probe<'_, K> { + /// Release exclusivity without erasing the probe's timeout. + fn drop(&mut self) { + if let Some(key) = &self.key { + self.health.probes.borrow_mut().remove(key); + } + } + } + + impl Circuits { + /// Bound failure records and set eligibility delay for abandoned probes. + /// A timeout beyond the clock's range never expires. + pub const fn new(capacity: usize, probe_timeout: Duration) -> Self { + Self { + capacity, + probe_timeout, + states: RefCell::new(BTreeMap::new()), + probes: RefCell::new(BTreeSet::new()), + local: PhantomData, + } + } + + /// Acquire an exclusive half-open guard, or an untracked healthy guard. + pub fn acquire(&self, key: &K, now: Instant) -> Result> { + if !self.try_acquire(key, now) { + return Err(Error::Unavailable); + } + let probe = self.states.borrow().contains_key(key); + let key = if probe { + if self.probes.borrow().len() >= self.capacity { + return Err(Error::Overloaded); + } + // Complete application cloning before publishing exclusive ownership. + let indexed = key.clone(); + let owned = key.clone(); + self.probes.borrow_mut().insert(indexed); + Some(owned) + } else { + None + }; + Ok(Probe { health: self, key }) + } + + /// Remove failure state without releasing any owned probe. + pub fn success(&self, key: &K) { + self.states.borrow_mut().remove(key); + } + + /// Record a caller-classified failure using caller-selected retry jitter. + /// Backoff may reenter: the count is visible before the callback, and any + /// newer transition for this key takes precedence over its returned delay. + /// A retry beyond the clock's range never expires on its own. + pub fn failure( + &self, + key: &K, + now: Instant, + backoff: impl FnOnce(&K, u32) -> Duration, + ) -> Result<()> { + let mut states = self.states.borrow_mut(); + if !states.contains_key(key) && states.len() >= self.capacity { + return Err(Error::Overloaded); + } + let state = states.entry(key.clone()).or_insert(Circuit { + failures: 0, + retry_at: Deadline::At(now), + probe_until: None, + pending_backoff: None, + }); + state.failures = state.failures.saturating_add(1); + let failures = state.failures; + let pending = Rc::new(()); + state.pending_backoff = Some(Rc::clone(&pending)); + drop(states); + + let retry_at = Deadline::after(now, backoff(key, failures)); + let mut states = self.states.borrow_mut(); + // Never reinsert a removed record or overwrite a newer transition. + if let Some(state) = states.get_mut(key) + && state + .pending_backoff + .as_ref() + .is_some_and(|current| Rc::ptr_eq(current, &pending)) + { + state.retry_at = retry_at; + state.probe_until = None; + state.pending_backoff = None; + } + Ok(()) + } + + /// Routing hint only; actual work must acquire an exclusive probe. + pub fn available(&self, key: &K, now: Instant) -> bool { + !self.probes.borrow().contains(key) + && self + .states + .borrow() + .get(key) + .is_none_or(|s| s.available(now)) + } + + /// Admit an unowned probe that becomes eligible again after its timeout. + pub fn try_acquire(&self, key: &K, now: Instant) -> bool { + if self.probes.borrow().contains(key) { + return false; + } + let mut states = self.states.borrow_mut(); + let Some(state) = states.get_mut(key) else { + return true; + }; + if !state.available(now) { + return false; + } + state.probe_until = Some(Deadline::after(now, self.probe_timeout)); + state.pending_backoff = None; + true + } + + /// Forget excluded failure records without releasing owned probes. + pub fn retain(&self, keys: &[K]) { + self.states.borrow_mut().retain(|key, _| keys.contains(key)); + } + + /// Count retained failure records, not active probes. + pub fn len(&self) -> usize { + self.states.borrow().len() + } + + /// Whether no failure records remain, independently of owned probes. + pub fn is_empty(&self) -> bool { + self.states.borrow().is_empty() + } + } + + /// Endpoint state transition and ownership contracts. + #[cfg(test)] + mod tests { + use super::*; + + /// Oversized retry delays stay closed to probes until explicit success. + #[test] + fn duration_overflow_endpoint_backoff() { + let now = Instant::now(); + assert!(now.checked_add(Duration::MAX).is_none()); + for delay in [ + Duration::ZERO, + Duration::from_secs(1), + Duration::from_secs(u64::from(u32::MAX)), + Duration::MAX, + ] { + let health = Circuits::new(1, Duration::ZERO); + health.failure(&7, now, |_, _| delay).unwrap(); + assert_eq!(health.available(&7, now), delay.is_zero()); + if let Some(due) = now.checked_add(delay) { + assert!(health.acquire(&7, due).is_ok()); + } else { + let later = now + Duration::from_secs(60); + assert!(!health.available(&7, later)); + assert!(matches!(health.acquire(&7, later), Err(Error::Unavailable))); + assert_eq!( + health.failure(&8, later, |_, _| Duration::ZERO), + Err(Error::Overloaded) + ); + } + health.success(&7); + assert!(health.acquire(&7, now).is_ok()); + } + } + + /// Oversized abandoned-probe timeouts never expire or lose ownership. + #[test] + fn duration_overflow_endpoint_probe_timeout() { + let now = Instant::now(); + for delay in [ + Duration::ZERO, + Duration::from_secs(1), + Duration::from_secs(u64::from(u32::MAX)), + Duration::MAX, + ] { + let health = Circuits::new(1, delay); + health.failure(&7, now, |_, _| Duration::ZERO).unwrap(); + let probe = health.acquire(&7, now).unwrap(); + assert!(!health.available(&7, now)); + drop(probe); + assert_eq!(health.available(&7, now), delay.is_zero()); + if let Some(due) = now.checked_add(delay) { + assert!(health.try_acquire(&7, due)); + } else { + assert!(!health.try_acquire(&7, now + Duration::from_secs(60))); + } + health.success(&7); + assert!(health.acquire(&7, now).is_ok()); + } + } + + /// Backoff can inspect the circuit without borrowing conflicts. + #[test] + fn backoff_reentry_observes_failure_record() { + let now = Instant::now(); + let health = Circuits::new(1, Duration::from_secs(1)); + health + .failure(&7, now, |key, failures| { + assert_eq!((*key, failures), (7, 1)); + assert_eq!(health.len(), 1); + assert!(!health.is_empty()); + assert!(health.available(key, now)); + Duration::from_secs(2) + }) + .unwrap(); + assert!(!health.available(&7, now)); + assert!(health.available(&7, now + Duration::from_secs(2))); + } + + /// A nested success or retention change must not be undone by backoff. + #[test] + fn backoff_reentry_preserves_removal_and_capacity() { + let now = Instant::now(); + for retain in [false, true] { + let health = Circuits::new(1, Duration::ZERO); + health + .failure(&7, now, |_, _| { + if retain { + health.retain(&[]); + } else { + health.success(&7); + } + health.failure(&8, now, |_, _| Duration::ZERO).unwrap(); + Duration::from_secs(10) + }) + .unwrap(); + assert_eq!(health.len(), 1); + assert!(health.available(&7, now)); + assert_eq!( + health.failure(&9, now, |_, _| panic!("capacity is full")), + Err(Error::Overloaded) + ); + health.success(&8); + assert!(health.is_empty()); + } + } + + /// Newer failures win even when counts saturate or the key is replaced. + #[test] + fn backoff_reentry_preserves_newer_failure() { + let now = Instant::now(); + for (initial, replace) in [(0, false), (u32::MAX, false), (0, true)] { + let health = Circuits::new(1, Duration::ZERO); + if initial != 0 { + health.failure(&7, now, |_, _| Duration::ZERO).unwrap(); + health.states.borrow_mut().get_mut(&7).unwrap().failures = initial; + } + health + .failure(&7, now, |_, count| { + assert_eq!(count, initial.saturating_add(1)); + if replace { + health.success(&7); + } + health + .failure(&7, now, |_, nested_count| { + assert_eq!( + nested_count, + if replace { 1 } else { count.saturating_add(1) } + ); + Duration::from_secs(2) + }) + .unwrap(); + Duration::from_secs(10) + }) + .unwrap(); + assert!(!health.available(&7, now)); + assert!(health.available(&7, now + Duration::from_secs(2))); + assert_eq!(health.len(), 1); + } + } + + /// A probe admitted by backoff keeps its timeout after the callback. + #[test] + fn backoff_reentry_preserves_probe_timeout() { + let now = Instant::now(); + let timeout = Duration::from_secs(2); + let health = Circuits::new(1, timeout); + health.failure(&7, now, |_, _| Duration::ZERO).unwrap(); + health + .failure(&7, now, |_, _| { + drop(health.acquire(&7, now).unwrap()); + Duration::from_secs(10) + }) + .unwrap(); + assert!(!health.available(&7, now)); + assert!(health.available(&7, now + timeout)); + } + + /// Failed key cloning must not leave an exclusive probe without an owner. + #[test] + fn review_regression_probe_clone_panic_allows_acquisition_after_timeout() { + use std::{ + panic::{AssertUnwindSafe, catch_unwind}, + sync::atomic::{AtomicUsize, Ordering}, + }; + + static CLONES: AtomicUsize = AtomicUsize::new(0); + static PANIC_AT: AtomicUsize = AtomicUsize::new(usize::MAX); + + /// A single endpoint with controllable clone failure. + #[derive(Eq, Ord, PartialEq, PartialOrd)] + struct Key; + + impl Clone for Key { + /// Fail at the selected clone during probe acquisition. + fn clone(&self) -> Self { + let clone = CLONES.fetch_add(1, Ordering::SeqCst) + 1; + assert_ne!(clone, PANIC_AT.load(Ordering::SeqCst), "key clone failed"); + Self + } + } + + for panic_at in [1, 2] { + let now = Instant::now(); + let timeout = Duration::from_secs(1); + let health = Circuits::new(1, timeout); + health.failure(&Key, now, |_, _| Duration::ZERO).unwrap(); + CLONES.store(0, Ordering::SeqCst); + PANIC_AT.store(panic_at, Ordering::SeqCst); + let result = catch_unwind(AssertUnwindSafe(|| health.acquire(&Key, now))); + PANIC_AT.store(usize::MAX, Ordering::SeqCst); + assert!(result.is_err()); + assert_eq!(CLONES.load(Ordering::SeqCst), panic_at); + assert!(!health.available(&Key, now)); + let retry = now + timeout; + let probe = health + .acquire(&Key, retry) + .expect("probe must not be stranded"); + assert!(!health.available(&Key, retry + timeout)); + drop(probe); + assert!(health.acquire(&Key, retry + timeout).is_ok()); + } + } + + /// Keep an owned probe exclusive through success and retention changes. + #[test] + fn bounded_backoff_and_owned_probe_survive_retention_and_success() { + let now = Instant::now(); + let health = Circuits::new(1, Duration::from_secs(1)); + assert!(health.try_acquire(&7, now)); + health + .failure(&7, now, |key, failures| { + assert_eq!((*key, failures), (7, 1)); + Duration::from_secs(2) + }) + .unwrap(); + assert_eq!( + health.failure(&8, now, |_, _| Duration::ZERO), + Err(Error::Overloaded) + ); + assert!(!health.available(&7, now)); + let retry = now + Duration::from_secs(2); + let probe = health.acquire(&7, retry).unwrap(); + assert!(!health.try_acquire(&7, retry + Duration::from_secs(5))); + health.success(&7); + health.retain(&[]); + assert!(!health.available(&7, retry)); + drop(probe); + assert!(health.available(&7, retry)); + assert!(health.is_empty()); + } + + /// Abandonment preserves timeout and failure history never overflows. + #[test] + fn dropped_probe_preserves_timeout_and_failures_saturate() { + let now = Instant::now(); + let health = Circuits::new(1, Duration::from_secs(1)); + health.failure(&(), now, |_, _| Duration::ZERO).unwrap(); + drop(health.acquire(&(), now).unwrap()); + assert!(!health.try_acquire(&(), now)); + assert!(health.try_acquire(&(), now + Duration::from_secs(1))); + health.states.borrow_mut().get_mut(&()).unwrap().failures = u32::MAX; + health + .failure(&(), now, |_, n| { + assert_eq!(n, u32::MAX); + Duration::ZERO + }) + .unwrap(); + health.retain(&[]); + assert_eq!(health.len(), 0); + assert_eq!( + Circuits::new(0, Duration::ZERO).failure(&(), now, |_, _| Duration::ZERO), + Err(Error::Overloaded) + ); + } + + /// Probe capacity and failure records remain independent after success. + #[test] + fn full_probe_set_preserves_rejected_attempt_timeout() { + let now = Instant::now(); + let timeout = Duration::from_secs(1); + let health = Circuits::new(1, timeout); + health.failure(&7, now, |_, _| Duration::ZERO).unwrap(); + let held = health.acquire(&7, now).unwrap(); + health.success(&7); + assert!(health.is_empty()); + assert!( + health.acquire(&9, now).is_ok(), + "healthy work needs no slot" + ); + + health.failure(&8, now, |_, _| Duration::ZERO).unwrap(); + assert!(health.available(&8, now)); + assert!(matches!(health.acquire(&8, now), Err(Error::Overloaded))); + drop(held); + assert!(!health.available(&8, now)); + assert!(matches!(health.acquire(&8, now), Err(Error::Unavailable))); + assert!(health.acquire(&8, now + timeout).is_ok()); + } + } +} + +/// Round-robin delivery with structurally retained target admission. +mod handoff { + use crate::{Error, Result}; + use std::{ + collections::VecDeque, + sync::{Arc, Mutex}, + task::Waker, + }; + + /// Bounds queued, delivered, and outstanding offered items with reservations. + /// Callbacks execute under the handoff lock and must not reenter it. + pub trait Admission { + /// Ownership that retains capacity until final admitted work completes. + type Reservation; + + /// Register for released capacity before attempting reservation. + fn register(&self, waker: &Waker); + + /// Reserve capacity for one item without waiting. + fn reserve(&self) -> Result; + } + + /// Payload and reservation remain inseparable while queued or freshly popped. + pub struct Admitted { + item: T, + + reservation: R, + } + + /// Fixed batch whose occupied entries each retain their target reservation. + type Batch = [Option>; N]; + + impl Admitted { + /// Transfer both owners together to the application's completion owner. + pub fn into_parts(self) -> (T, R) { + (self.item, self.reservation) + } + } + + /// One target's installed admission, inbox, and wake registration. + struct Target { + admission: Option, + + queue: VecDeque>, + + // Sharing the registration avoids raw waker callbacks under the lock. + waker: Option>, + + closed: bool, + } + + /// Stable target slots and the next round-robin scan position. + struct State { + targets: Vec<(K, Target)>, + + cursor: usize, + } + + impl State { + /// Find a stable target without changing its admission or closed state. + fn target(&mut self, key: &K) -> Option<&mut Target> { + self.targets + .iter_mut() + .find(|(candidate, _)| candidate == key) + .map(|(_, target)| target) + } + } + + /// Shared handoff bounded by admission, not a separate queue-slot limit. + pub struct Handoff(Mutex>); + + /// Capacity reserved on a selected target before constructing its payload. + pub struct Offer { + handoff: Arc>, + + target: usize, + + reservation: A::Reservation, + } + + impl Handoff { + /// Create stable target slots; empty target lists simply reject offers. + /// Repeated keys share one slot, in first-occurrence order. + pub fn new(keys: &[K]) -> Self { + let mut targets = Vec::new(); + for key in keys { + if targets.iter().any(|(candidate, _)| candidate == key) { + continue; + } + targets.push(( + key.clone(), + Target { + admission: None, + queue: VecDeque::new(), + waker: None, + closed: false, + }, + )); + } + Self(Mutex::new(State { targets, cursor: 0 })) + } + + /// Install admission once on a known target that has not closed. + pub fn install(&self, key: &K, admission: A) -> Result<()> { + let mut state = self.0.lock().map_err(|_| Error::Unavailable)?; + let target = state.target(key).ok_or(Error::InvalidInput)?; + if target.admission.is_some() || target.closed { + return Err(Error::InvalidInput); + } + target.admission = Some(admission); + Ok(()) + } + + /// Scan open targets fairly and reserve before returning an offer. + pub fn reserve(self: &Arc, waker: &Waker) -> Result> { + let mut state = self.0.lock().map_err(|_| Error::Unavailable)?; + let mut rejection = None; + for offset in 0..state.targets.len() { + let index = (state.cursor + offset) % state.targets.len(); + let (_, target) = &state.targets[index]; + if target.closed { + continue; + } + let Some(admission) = &target.admission else { + continue; + }; + admission.register(waker); + match admission.reserve() { + Ok(reservation) => { + state.cursor = (index + 1) % state.targets.len(); + return Ok(Offer { + handoff: self.clone(), + target: index, + reservation, + }); + } + Err(Error::Overloaded) => rejection = Some(Error::Overloaded), + Err(Error::Unavailable) => { + rejection.get_or_insert(Error::Unavailable); + } + Err(error) => return Err(error), + } + } + Err(rejection.unwrap_or(Error::Overloaded)) + } + + /// Pop at most the caller's budget while retaining each item's admission. + /// Closed targets reject pops without retaining the caller's waker. + pub fn pop_batch( + &self, + key: &K, + waker: &Waker, + budget: usize, + ) -> Result> { + let waker = Arc::new(waker.clone()); + let mut state = self.0.lock().map_err(|_| Error::Unavailable)?; + let target = state.target(key).ok_or(Error::InvalidInput)?; + if target.closed { + return Err(Error::Unavailable); + } + let old = target.waker.replace(waker); + let batch = std::array::from_fn(|index| { + if index < budget { + target.queue.pop_front() + } else { + None + } + }); + drop(state); + drop(old); + Ok(batch) + } + + /// Close permanently, then drop ownership and wake outside the shared lock. + pub fn close(&self, key: &K) { + let (queued, admission, waker) = { + let mut state = self.0.lock().unwrap_or_else(|e| e.into_inner()); + let Some(target) = state.target(key) else { + return; + }; + target.closed = true; + ( + std::mem::take(&mut target.queue), + target.admission.take(), + target.waker.take(), + ) + }; + drop(queued); + drop(admission); + if let Some(waker) = waker { + waker.wake_by_ref(); + } + } + } + + impl Offer { + /// Build only on an open target; the envelope retains admission itself. + /// The closure executes under the handoff lock and must not reenter it. + pub fn deliver(self, item: impl FnOnce() -> T) -> Result<()> { + let mut state = self.handoff.0.lock().map_err(|_| Error::Unavailable)?; + let target = &mut state.targets[self.target].1; + if target.closed { + return Err(Error::Unavailable); + } + target.queue.push_back(Admitted { + item: item(), + reservation: self.reservation, + }); + let waker = target.waker.clone(); + drop(state); + if let Some(waker) = waker { + waker.wake_by_ref(); + } + Ok(()) + } + } + + #[cfg(test)] + mod tests { + use super::*; + use std::{ + sync::{ + Weak, + atomic::{AtomicUsize, Ordering}, + }, + task::{RawWaker, RawWakerVTable, Wake}, + }; + + struct TargetWake { + handoff: Weak>, + blocked: AtomicUsize, + clones: AtomicUsize, + drops: AtomicUsize, + wakes: AtomicUsize, + } + + impl TargetWake { + fn reenter(&self) { + let Some(handoff) = self.handoff.upgrade() else { + return; + }; + if let Ok(state) = handoff.0.try_lock() { + drop(state); + handoff.close(&2); + } else { + self.blocked.fetch_add(1, Ordering::SeqCst); + } + } + + fn new(handoff: &Arc>) -> (Arc, Waker) { + let state = Arc::new(Self { + handoff: Arc::downgrade(handoff), + blocked: AtomicUsize::new(0), + clones: AtomicUsize::new(0), + drops: AtomicUsize::new(0), + wakes: AtomicUsize::new(0), + }); + // SAFETY: Each raw waker owns one Arc, managed by this vtable. + let waker = unsafe { Waker::from_raw(Self::raw(state.clone())) }; + (state, waker) + } + + fn raw(state: Arc) -> RawWaker { + RawWaker::new(Arc::into_raw(state).cast(), &Self::VTABLE) + } + + const VTABLE: RawWakerVTable = RawWakerVTable::new( + |data| { + // SAFETY: Borrow the live Arc without consuming the source waker. + let state = + std::mem::ManuallyDrop::new(unsafe { Arc::::from_raw(data.cast()) }); + state.reenter(); + state.clones.fetch_add(1, Ordering::SeqCst); + Self::raw(Arc::clone(&state)) + }, + |data| { + // SAFETY: Wake consumes this raw waker's Arc exactly once. + let state = unsafe { Arc::::from_raw(data.cast()) }; + state.reenter(); + state.wakes.fetch_add(1, Ordering::SeqCst); + }, + |data| { + // SAFETY: Borrow without consuming the source waker's Arc. + let state = + std::mem::ManuallyDrop::new(unsafe { Arc::::from_raw(data.cast()) }); + state.reenter(); + state.wakes.fetch_add(1, Ordering::SeqCst); + }, + |data| { + // SAFETY: Drop consumes this raw waker's Arc exactly once. + let state = unsafe { Arc::::from_raw(data.cast()) }; + state.reenter(); + state.drops.fetch_add(1, Ordering::SeqCst); + }, + ); + } + + #[test] + fn target_waker_clone_can_reenter_pop() { + let handoff = Arc::new(Handoff::new(&[1])); + let (state, waker) = TargetWake::new(&handoff); + for _ in 0..2 { + assert!(handoff.pop_batch::<1>(&1, &waker, 0).unwrap()[0].is_none()); + } + assert!(state.clones.load(Ordering::SeqCst) > 0); + assert_eq!(state.blocked.load(Ordering::SeqCst), 0); + } + + #[test] + fn target_waker_drop_can_reenter_replacement_and_rejection() { + let handoff = Arc::new(Handoff::new(&[1])); + let (state, waker) = TargetWake::new(&handoff); + handoff.pop_batch::<0>(&1, &waker, 0).unwrap(); + state.blocked.store(0, Ordering::SeqCst); + handoff.pop_batch::<0>(&1, Waker::noop(), 0).unwrap(); + assert_eq!(state.drops.load(Ordering::SeqCst), 1); + assert_eq!(state.blocked.load(Ordering::SeqCst), 0); + handoff.close(&1); + assert!(matches!( + handoff.pop_batch::<1>(&1, &waker, 1), + Err(Error::Unavailable) + )); + assert!(matches!( + handoff.pop_batch::<1>(&2, &waker, 1), + Err(Error::InvalidInput) + )); + assert_eq!(state.blocked.load(Ordering::SeqCst), 0); + assert!(handoff.0.lock().unwrap().targets[0].1.waker.is_none()); + } + + #[test] + fn target_waker_callbacks_can_reenter_repeated_delivery() { + let handoff = Arc::new(Handoff::new(&[1])); + handoff.install(&1, Ready).unwrap(); + let (state, waker) = TargetWake::new(&handoff); + handoff.pop_batch::<0>(&1, &waker, 0).unwrap(); + state.blocked.store(0, Ordering::SeqCst); + for item in [7, 8] { + handoff + .reserve(Waker::noop()) + .unwrap() + .deliver(|| item) + .unwrap(); + } + assert_eq!(state.wakes.load(Ordering::SeqCst), 2); + assert_eq!(state.blocked.load(Ordering::SeqCst), 0); + let items = handoff.pop_batch::<2>(&1, Waker::noop(), 2).unwrap(); + assert_eq!(items.map(|item| item.unwrap().into_parts().0), [7, 8]); + } + + type ClosingHandoff = Handoff; + + #[derive(Clone)] + struct CloseDrop { + handoff: Weak, + drops: Arc, + } + + impl CloseDrop { + fn reenter(&self) { + if let Some(handoff) = self.handoff.upgrade() { + let state = handoff + .0 + .try_lock() + .expect("close must unlock before callbacks"); + assert!(state.targets[0].1.closed); + assert!(state.targets[0].1.queue.is_empty()); + assert!(state.targets[0].1.admission.is_none()); + assert!(state.targets[0].1.waker.is_none()); + drop(state); + handoff.close(&1); + assert!(matches!( + handoff.pop_batch::<1>(&1, Waker::noop(), 1), + Err(Error::Unavailable) + )); + } + } + } + + impl Drop for CloseDrop { + fn drop(&mut self) { + self.reenter(); + self.drops.fetch_add(1, Ordering::SeqCst); + } + } + + struct CloseAdmission(CloseDrop); + + impl Admission for CloseAdmission { + type Reservation = CloseDrop; + + fn register(&self, _: &Waker) {} + + fn reserve(&self) -> Result { + Ok(self.0.clone()) + } + } + + struct CloseWake { + on_drop: CloseDrop, + wakes: Arc, + } + + impl Wake for CloseWake { + fn wake(self: Arc) { + self.on_drop.reenter(); + self.wakes.fetch_add(1, Ordering::SeqCst); + } + } + + #[test] + fn close_releases_admission_and_queue_outside_lock() { + for queued in [0, 2] { + let handoff = Arc::new(ClosingHandoff::new(&[1])); + let drops = Arc::new(AtomicUsize::new(0)); + handoff + .install( + &1, + CloseAdmission(CloseDrop { + handoff: Arc::downgrade(&handoff), + drops: drops.clone(), + }), + ) + .unwrap(); + for _ in 0..queued { + handoff + .reserve(Waker::noop()) + .unwrap() + .deliver(|| CloseDrop { + handoff: Arc::downgrade(&handoff), + drops: drops.clone(), + }) + .unwrap(); + } + handoff.close(&1); + assert_eq!(drops.load(Ordering::SeqCst), 1 + 2 * queued); + handoff.close(&1); + handoff.close(&2); + assert_eq!(drops.load(Ordering::SeqCst), 1 + 2 * queued); + assert!(handoff.0.lock().unwrap().targets[0].1.admission.is_none()); + } + } + + #[test] + fn close_wakes_and_releases_waiter_outside_lock() { + let handoff = Arc::new(ClosingHandoff::new(&[1])); + let drops = Arc::new(AtomicUsize::new(0)); + let wakes = Arc::new(AtomicUsize::new(0)); + let waker = Waker::from(Arc::new(CloseWake { + on_drop: CloseDrop { + handoff: Arc::downgrade(&handoff), + drops: drops.clone(), + }, + wakes: wakes.clone(), + })); + assert!(handoff.pop_batch::<1>(&1, &waker, 1).unwrap()[0].is_none()); + drop(waker); + handoff.close(&2); + assert_eq!(wakes.load(Ordering::SeqCst), 0); + handoff.close(&1); + assert_eq!(wakes.load(Ordering::SeqCst), 1); + assert_eq!(drops.load(Ordering::SeqCst), 1); + handoff.close(&1); + assert_eq!(wakes.load(Ordering::SeqCst), 1); + assert!(handoff.0.lock().unwrap().targets[0].1.waker.is_none()); + } + + #[test] + fn close_rejects_pop_without_retaining_new_waiter() { + let handoff = Handoff::<_, Ready, ()>::new(&[1, 2]); + handoff.install(&1, Ready).unwrap(); + for key in [1, 2] { + handoff.close(&key); + for budget in [0, 1, usize::MAX] { + assert!(matches!( + handoff.pop_batch::<1>(&key, Waker::noop(), budget), + Err(Error::Unavailable) + )); + assert!(matches!( + handoff.pop_batch::<0>(&key, Waker::noop(), budget), + Err(Error::Unavailable) + )); + assert!( + handoff + .0 + .lock() + .unwrap() + .target(&key) + .unwrap() + .waker + .is_none() + ); + } + assert_eq!(handoff.install(&key, Ready), Err(Error::InvalidInput)); + } + assert!(matches!( + handoff.pop_batch::<1>(&3, Waker::noop(), 1), + Err(Error::InvalidInput) + )); + } + + struct Ready; + + impl Admission for Ready { + type Reservation = (); + + fn register(&self, _: &Waker) {} + + fn reserve(&self) -> Result<()> { + Ok(()) + } + } + + #[test] + fn handoff_constructor_keeps_first_unique_targets() { + let cases: &[(&[u8], &[u8])] = &[ + (&[], &[]), + (&[1], &[1]), + (&[1, 1, 1], &[1]), + (&[3, 1, 3, 2, 1, 3], &[3, 1, 2]), + ]; + for &(keys, expected) in cases { + let handoff = Arc::new(Handoff::<_, Ready, u8>::new(keys)); + assert_eq!( + handoff + .0 + .lock() + .unwrap() + .targets + .iter() + .map(|(key, _)| *key) + .collect::>(), + expected + ); + for key in expected { + handoff.install(key, Ready).unwrap(); + assert_eq!(handoff.install(key, Ready), Err(Error::InvalidInput)); + } + for _ in 0..2 { + for key in expected { + handoff + .reserve(Waker::noop()) + .unwrap() + .deliver(|| *key) + .unwrap(); + let [item] = handoff.pop_batch::<1>(key, Waker::noop(), 1).unwrap(); + assert_eq!(item.unwrap().into_parts(), (*key, ())); + } + } + for key in expected { + handoff.close(key); + assert_eq!(handoff.install(key, Ready), Err(Error::InvalidInput)); + } + assert!(matches!( + handoff.reserve(Waker::noop()), + Err(Error::Overloaded) + )); + } + } + } +} + +/// Shared speculative capacity and explicitly polled due-time registrations. +mod hedge { + use crate::{Error, Result}; + use std::{ + collections::BTreeMap, + sync::{Arc, Mutex}, + task::{Context, Poll, Waker}, + time::Instant, + }; + + /// One admitted speculation and its latest wake registration. + struct Alarm { + due: Instant, + + cost: usize, + + wake: Option, + } + + /// Mutex-protected capacity and monotonically assigned alarm identities. + #[derive(Default)] + struct State { + next: u64, + + used: usize, + + alarms: BTreeMap, + } + + /// Process-shared slot and cost limits, independent of worker registries. + pub struct Hedges { + slots: usize, + + capacity: usize, + + state: Mutex, + } + + /// Owns speculative capacity until both contenders reach their fences. + #[must_use = "retain the permit until both contenders are fenced"] + pub struct Permit { + owner: Arc, + + id: u64, + } + + impl Hedges { + /// Create shared ceilings; zero slots disable speculative admission. + pub fn new(slots: usize, capacity: usize) -> Arc { + Arc::new(Self { + slots, + capacity, + state: Mutex::new(State::default()), + }) + } + + /// Reserve one slot and caller-selected cost until permit drop. + pub fn acquire(self: &Arc, cost: usize, due: Instant) -> Result { + let mut state = self.state.lock().map_err(|_| Error::Unavailable)?; + if state.alarms.len() >= self.slots || cost > self.capacity - state.used { + return Err(Error::Overloaded); + } + let id = state.next.checked_add(1).ok_or(Error::Unavailable)?; + state.next = id; + state.used += cost; + state.alarms.insert( + id, + Alarm { + due, + cost, + wake: None, + }, + ); + Ok(Permit { + owner: self.clone(), + id, + }) + } + + /// Wake due registrations outside the shared lock, once per registration. + pub fn poll(&self, now: Instant) { + let wakes: Vec<_> = self + .state + .lock() + .map(|mut state| { + state + .alarms + .values_mut() + .filter(|a| now >= a.due) + .filter_map(|a| a.wake.take()) + .collect() + }) + .unwrap_or_default(); + for wake in wakes { + wake.wake(); + } + } + } + + impl Permit { + /// Register the latest waiter until due, without releasing capacity. + pub fn delay(&self, now: Instant, cx: &mut Context<'_>) -> Poll<()> { + // Waker clone and drop callbacks may reenter the hedge registry. + let wake = cx.waker().clone(); + let mut state = self.owner.state.lock().expect("hedge alarm lock"); + let alarm = state.alarms.get_mut(&self.id).expect("live hedge alarm"); + let (result, previous) = if now >= alarm.due { + (Poll::Ready(()), alarm.wake.take()) + } else { + (Poll::Pending, alarm.wake.replace(wake)) + }; + drop(state); + drop(previous); + result + } + } + + impl Drop for Permit { + /// Remove the alarm and return its cost only at final ownership release. + fn drop(&mut self) { + let alarm = self.owner.state.lock().ok().and_then(|mut state| { + let alarm = state.alarms.remove(&self.id)?; + state.used -= alarm.cost; + Some(alarm) + }); + drop(alarm); + } + } + + /// Shared capacity and alarm lifecycle contracts. + #[cfg(test)] + mod tests { + use super::*; + use std::{ + mem::ManuallyDrop, + sync::atomic::{AtomicUsize, Ordering}, + task::{RawWaker, RawWakerVTable, Wake}, + time::Duration, + }; + + /// Count alarm notifications across threads. + #[derive(Default)] + struct Counter(AtomicUsize); + + impl Wake for Counter { + /// Record one notification. + fn wake(self: Arc) { + self.0.fetch_add(1, Ordering::SeqCst); + } + } + + struct DelayCallbacks { + owner: Arc, + now: Instant, + clones: AtomicUsize, + drops: AtomicUsize, + locked_clones: AtomicUsize, + locked_drops: AtomicUsize, + } + + impl DelayCallbacks { + fn new(owner: &Arc, now: Instant) -> Arc { + Arc::new(Self { + owner: owner.clone(), + now, + clones: AtomicUsize::new(0), + drops: AtomicUsize::new(0), + locked_clones: AtomicUsize::new(0), + locked_drops: AtomicUsize::new(0), + }) + } + + fn reenter(&self, calls: &AtomicUsize, locked: &AtomicUsize) { + calls.fetch_add(1, Ordering::SeqCst); + // Record a deadlock risk without hanging or poisoning the mutex. + if self.owner.state.try_lock().is_err() { + locked.fetch_add(1, Ordering::SeqCst); + return; + } + self.owner.poll(self.now); + } + + fn raw(this: Arc) -> RawWaker { + RawWaker::new(Arc::into_raw(this).cast(), &Self::VTABLE) + } + + fn waker(this: &Arc) -> Waker { + // SAFETY: The vtable owns one Arc per raw waker and uses atomic state. + unsafe { Waker::from_raw(Self::raw(this.clone())) } + } + + const VTABLE: RawWakerVTable = + RawWakerVTable::new(Self::clone_raw, Self::drop_raw, |_| {}, Self::drop_raw); + + unsafe fn clone_raw(data: *const ()) -> RawWaker { + // SAFETY: Borrow the raw waker's Arc without consuming its reference. + let this = ManuallyDrop::new(unsafe { Arc::from_raw(data.cast::()) }); + this.reenter(&this.clones, &this.locked_clones); + Self::raw(Arc::clone(&this)) + } + + unsafe fn drop_raw(data: *const ()) { + // SAFETY: Drop or consuming wake releases exactly one owned reference. + let this = unsafe { Arc::from_raw(data.cast::()) }; + this.reenter(&this.drops, &this.locked_drops); + } + } + + #[test] + fn hedge_delay_clones_waiter_outside_lock() { + let now = Instant::now(); + let due = now + Duration::from_secs(1); + let owner = Hedges::new(1, 1); + let permit = owner.acquire(1, due).unwrap(); + let callbacks = DelayCallbacks::new(&owner, now); + let waker = DelayCallbacks::waker(&callbacks); + assert!( + permit + .delay(now, &mut Context::from_waker(&waker)) + .is_pending() + ); + // Clear the registration so cleanup does not exercise Permit::drop. + assert!( + permit + .delay(due, &mut Context::from_waker(Waker::noop())) + .is_ready() + ); + assert_eq!(callbacks.clones.load(Ordering::SeqCst), 1); + assert_eq!(callbacks.locked_clones.load(Ordering::SeqCst), 0); + } + + #[test] + fn hedge_delay_replaces_waiter_outside_lock() { + let now = Instant::now(); + let due = now + Duration::from_secs(1); + let owner = Hedges::new(1, 1); + let permit = owner.acquire(1, due).unwrap(); + let callbacks = DelayCallbacks::new(&owner, now); + let waker = DelayCallbacks::waker(&callbacks); + assert!( + permit + .delay(now, &mut Context::from_waker(&waker)) + .is_pending() + ); + let latest = Arc::new(Counter::default()); + assert!( + permit + .delay(now, &mut Context::from_waker(&Waker::from(latest.clone()))) + .is_pending() + ); + assert_eq!(callbacks.drops.load(Ordering::SeqCst), 1); + assert_eq!(callbacks.locked_drops.load(Ordering::SeqCst), 0); + owner.poll(due); + owner.poll(due); + assert_eq!(latest.0.load(Ordering::SeqCst), 1); + assert!(matches!(owner.acquire(1, due), Err(Error::Overloaded))); + } + + #[test] + fn hedge_delay_clears_waiter_outside_lock() { + for elapsed in [Duration::ZERO, Duration::from_secs(1)] { + let now = Instant::now(); + let due = now + Duration::from_secs(1); + let owner = Hedges::new(1, 1); + let permit = owner.acquire(1, due).unwrap(); + let callbacks = DelayCallbacks::new(&owner, now); + let waker = DelayCallbacks::waker(&callbacks); + assert!( + permit + .delay(now, &mut Context::from_waker(&waker)) + .is_pending() + ); + assert!( + permit + .delay(due + elapsed, &mut Context::from_waker(Waker::noop())) + .is_ready() + ); + assert_eq!(callbacks.drops.load(Ordering::SeqCst), 1); + assert_eq!(callbacks.locked_drops.load(Ordering::SeqCst), 0); + owner.poll(due + elapsed); + assert_eq!(callbacks.drops.load(Ordering::SeqCst), 1); + assert!(matches!(owner.acquire(1, due), Err(Error::Overloaded))); + } + } + + /// Slots and bytes stay charged after an alarm fires and across threads. + #[test] + fn hedge_shared_slots_costs_and_release_are_independent_of_alarm() { + /// Require shared ownership at compile time. + fn shared() {} + shared::(); + shared::(); + let now = Instant::now(); + let owner = Hedges::new(2, 10); + let a = owner.acquire(7, now).unwrap(); + assert!(matches!(owner.acquire(4, now), Err(Error::Overloaded))); + let b = + std::thread::scope(|s| s.spawn(|| owner.acquire(3, now)).join().unwrap().unwrap()); + assert!(matches!(owner.acquire(0, now), Err(Error::Overloaded))); + owner.poll(now); + assert!(matches!(owner.acquire(1, now), Err(Error::Overloaded))); + drop(a); + let c = owner.acquire(7, now).unwrap(); + drop((b, c)); + assert!(owner.acquire(10, now).is_ok()); + assert!(matches!( + Hedges::new(0, 10).acquire(0, now), + Err(Error::Overloaded) + )); + let max = Hedges::new(2, usize::MAX); + let _all = max.acquire(usize::MAX, now).unwrap(); + assert!(matches!(max.acquire(1, now), Err(Error::Overloaded))); + } + + /// Direct completion clears the waiter but retains speculative capacity. + #[test] + fn hedge_ready_delay_clears_registration_without_releasing_capacity() { + for elapsed in [Duration::ZERO, Duration::from_secs(1)] { + let now = Instant::now(); + let due = now + Duration::from_secs(1); + let owner = Hedges::new(1, 1); + let permit = owner.acquire(1, due).unwrap(); + let waiter = Arc::new(Counter::default()); + assert!( + permit + .delay(now, &mut Context::from_waker(&Waker::from(waiter.clone()))) + .is_pending() + ); + assert!( + permit + .delay(due + elapsed, &mut Context::from_waker(Waker::noop())) + .is_ready() + ); + assert_eq!(Arc::strong_count(&waiter), 1); + owner.poll(due + elapsed); + owner.poll(due + elapsed); + assert_eq!(waiter.0.load(Ordering::SeqCst), 0); + assert_eq!(Arc::strong_count(&waiter), 1); + assert!(matches!(owner.acquire(1, due), Err(Error::Overloaded))); + drop(permit); + assert!(owner.acquire(1, due).is_ok()); + } + } + + /// Only the latest registered waker fires and drop cancels notification. + #[test] + fn hedge_alarm_replaces_waker_and_drop_removes_registration() { + let now = Instant::now(); + let due = now + Duration::from_secs(1); + let owner = Hedges::new(1, 1); + let permit = owner.acquire(1, due).unwrap(); + let old = Arc::new(Counter::default()); + let current = Arc::new(Counter::default()); + assert!( + permit + .delay(now, &mut Context::from_waker(&Waker::from(old.clone()))) + .is_pending() + ); + assert!( + permit + .delay(now, &mut Context::from_waker(&Waker::from(current.clone()))) + .is_pending() + ); + owner.poll(now); + assert_eq!(current.0.load(Ordering::SeqCst), 0); + owner.poll(due); + owner.poll(due); + assert_eq!(old.0.load(Ordering::SeqCst), 0); + assert_eq!(current.0.load(Ordering::SeqCst), 1); + assert!( + permit + .delay(due, &mut Context::from_waker(Waker::noop())) + .is_ready() + ); + drop(permit); + let permit = owner.acquire(1, due).unwrap(); + assert!( + permit + .delay(now, &mut Context::from_waker(&Waker::from(current.clone()))) + .is_pending() + ); + drop(permit); + owner.poll(due); + assert_eq!(current.0.load(Ordering::SeqCst), 1); + } + + #[test] + fn hedge_permit_drop_releases_waker_outside_lock() { + struct ReenterOnDrop { + owner: std::sync::Weak, + drops: Arc, + wakes: Arc, + } + + impl Wake for ReenterOnDrop { + fn wake(self: Arc) { + self.wakes.fetch_add(1, Ordering::SeqCst); + } + } + + impl Drop for ReenterOnDrop { + fn drop(&mut self) { + let owner = self.owner.upgrade().unwrap(); + let state = owner + .state + .try_lock() + .expect("permit drop must unlock before dropping its waker"); + assert!(state.alarms.is_empty()); + assert_eq!(state.used, 0); + drop(state); + let now = Instant::now(); + owner.poll(now); + let permit = owner.acquire(owner.capacity, now).unwrap(); + drop(permit); + self.drops.fetch_add(1, Ordering::SeqCst); + } + } + + for cost in [0, 7] { + let now = Instant::now(); + let due = now + Duration::from_secs(1); + let owner = Hedges::new(1, 7); + let permit = owner.acquire(cost, due).unwrap(); + let drops = Arc::new(AtomicUsize::new(0)); + let wakes = Arc::new(AtomicUsize::new(0)); + let waker = Waker::from(Arc::new(ReenterOnDrop { + owner: Arc::downgrade(&owner), + drops: drops.clone(), + wakes: wakes.clone(), + })); + assert!( + permit + .delay(now, &mut Context::from_waker(&waker)) + .is_pending() + ); + drop(waker); + assert_eq!(drops.load(Ordering::SeqCst), 0); + drop(permit); + assert_eq!(drops.load(Ordering::SeqCst), 1); + owner.poll(due); + assert_eq!(wakes.load(Ordering::SeqCst), 0); + assert!(owner.acquire(7, due).is_ok()); + } + } + } +} + +/// Adaptive state transitions, generation fencing, and shared ownership contracts. +#[cfg(test)] +mod tests { + use super::*; + use std::sync::OnceLock; + + /// Synchronous observer recording emitted events and gauges. + #[derive(Default)] + struct Counts { + events: Mutex>, + + active: Mutex, + + limit: Mutex, + } + + static CLOCK_OWNER: OnceLock>> = OnceLock::new(); + + /// Verify clock callbacks run without the adaptive-state mutex held. + fn reentrant_clock() -> Instant { + if let Some(owner) = CLOCK_OWNER.get().and_then(std::sync::Weak::upgrade) { + let _state = owner + .state + .try_lock() + .expect("clock called while adaptive state is locked"); + } + Instant::now() + } + + impl Observer for Counts { + /// Append an event in emission order. + fn event(&self, event: Event) { + self.events.lock().unwrap().push(event); + } + + /// Save the latest active-work gauge. + fn active(&self, active: usize) { + *self.active.lock().unwrap() = active; + } + + /// Save the latest admission-limit gauge. + fn limit(&self, limit: usize) { + *self.limit.lock().unwrap() = limit; + } + } + + /// Return small limits with independent backoff and recovery intervals. + fn config() -> Config { + Config { + total: 4, + per_key: 4, + capacity: 2, + backoff: Duration::from_millis(250), + recovery: Duration::from_secs(1), + retire_after: Duration::from_secs(60), + } + } + + /// Oversized peer backoff must not panic, poison state, or retire early. + #[test] + fn duration_overflow_adaptive_failure() { + let owner = Adaptive::new( + Config { + capacity: 1, + backoff: Duration::MAX, + retire_after: Duration::ZERO, + ..config() + }, + Counts::default(), + Instant::now, + ) + .unwrap(); + let stale = owner.acquire(&1).unwrap(); + let failed = owner.acquire(&1).unwrap(); + failed.observe(Outcome::PeerFailure); + stale.observe(Outcome::Verified); + assert!(!owner.available(&1)); + assert!(!owner.hedge_available(&1)); + drop((stale, failed)); + assert_eq!(owner.state.lock().unwrap().active, 0); + assert_eq!(*owner.observer.active.lock().unwrap(), 0); + assert!(matches!(owner.acquire(&1), Err(Error::Unavailable))); + assert!(matches!(owner.acquire(&2), Err(Error::Overloaded))); + } + + /// Renewing a probe can overflow even when its previous retry was finite. + #[test] + fn duration_overflow_adaptive_probe_drop() { + let owner = Adaptive::new( + Config { + backoff: Duration::MAX, + ..config() + }, + Counts::default(), + Instant::now, + ) + .unwrap(); + drop(owner.acquire(&1).unwrap()); + owner.state.lock().unwrap().peers.get_mut(&1).unwrap().retry = + Some(Deadline::At(Instant::now())); + let probe = owner.acquire(&1).unwrap(); + drop(probe); + assert_eq!(owner.state.lock().unwrap().active, 0); + assert_eq!(*owner.observer.active.lock().unwrap(), 0); + assert!(!owner.available(&1)); + assert!(matches!(owner.acquire(&1), Err(Error::Unavailable))); + assert!(owner.acquire(&2).is_ok()); + } + + /// Elapsed-time limits accept huge durations without forming deadlines. + #[test] + fn duration_overflow_elapsed_limits_remain_valid() { + let owner = Adaptive::new( + Config { + capacity: 1, + backoff: Duration::ZERO, + recovery: Duration::MAX, + retire_after: Duration::MAX, + ..config() + }, + Counts::default(), + Instant::now, + ) + .unwrap(); + let work = owner.acquire(&1).unwrap(); + work.observe(Outcome::LocalPressure); + work.observe(Outcome::PeerFailure); + drop(work); + let probe = owner.acquire(&1).unwrap(); + probe.observe(Outcome::Verified); + drop(probe); + assert!(owner.available(&1)); + assert_eq!(owner.state.lock().unwrap().limit, config().total / 2); + assert_eq!( + owner.state.lock().unwrap().peers[&1].limit, + config().per_key / 2 + ); + assert!(matches!(owner.acquire(&2), Err(Error::Overloaded))); + assert!(owner.acquire(&1).is_ok()); + } + + /// Clone panics leave the mutex usable and all admission capacity recoverable. + #[test] + fn adaptive_clone_panic_preserves_capacity_and_mutex() { + use std::{ + panic::{AssertUnwindSafe, catch_unwind}, + sync::atomic::{AtomicUsize, Ordering}, + }; + + static CLONES: AtomicUsize = AtomicUsize::new(0); + static PANIC_AT: AtomicUsize = AtomicUsize::new(usize::MAX); + static PANIC_KEY: AtomicUsize = AtomicUsize::new(usize::MAX); + + /// Key with selectable clone failures, including during retirement. + #[derive(Eq, Ord, PartialEq, PartialOrd)] + struct Key(usize); + + impl Clone for Key { + fn clone(&self) -> Self { + let clone = CLONES.fetch_add(1, Ordering::SeqCst) + 1; + assert_ne!(clone, PANIC_AT.load(Ordering::SeqCst), "key clone failed"); + assert_ne!( + self.0, + PANIC_KEY.load(Ordering::SeqCst), + "retired key cloned" + ); + Self(self.0) + } + } + + for existing in [false, true] { + for panic_at in [1, 2] { + let owner = Adaptive::new( + Config { + total: 2, + per_key: 2, + capacity: 1, + retire_after: Duration::ZERO, + ..config() + }, + Counts::default(), + Instant::now, + ) + .unwrap(); + let held = existing.then(|| owner.acquire(&Key(1)).unwrap()); + CLONES.store(0, Ordering::SeqCst); + PANIC_AT.store(panic_at, Ordering::SeqCst); + let result = catch_unwind(AssertUnwindSafe(|| owner.acquire(&Key(1)))); + PANIC_AT.store(usize::MAX, Ordering::SeqCst); + assert!(result.is_err()); + assert_eq!(CLONES.load(Ordering::SeqCst), panic_at); + { + let state = owner.state.lock().expect("clone panic poisoned state"); + assert_eq!(state.active, usize::from(existing)); + assert_eq!(state.peers.len(), usize::from(existing)); + if existing { + assert_eq!(state.peers[&Key(1)].active, 1); + } + } + drop(held); + let one = owner.acquire(&Key(1)).unwrap(); + let two = owner.acquire(&Key(1)).unwrap(); + assert!(matches!(owner.acquire(&Key(1)), Err(Error::Overloaded))); + drop((one, two)); + assert_eq!(owner.state.lock().unwrap().active, 0); + assert_eq!(*owner.observer.active.lock().unwrap(), 0); + + PANIC_KEY.store(1, Ordering::SeqCst); + let replacement = owner.acquire(&Key(2)).expect("retirement must not clone"); + PANIC_KEY.store(usize::MAX, Ordering::SeqCst); + { + let state = owner.state.lock().unwrap(); + assert_eq!(state.peers.len(), 1); + assert!(!state.peers.contains_key(&Key(1))); + assert_eq!(state.peers[&Key(2)].active, 1); + } + drop(replacement); + assert_eq!(owner.state.lock().unwrap().active, 0); + } + } + } + + /// Retired keys can reenter after publication; rejected attempts keep accounting. + #[test] + fn adaptive_retired_key_drop_can_reenter_after_unlock() { + use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering}; + + static OWNER: OnceLock>> = OnceLock::new(); + static WATCH: AtomicBool = AtomicBool::new(false); + static DROPS: AtomicUsize = AtomicUsize::new(0); + static BLOCKED: AtomicUsize = AtomicUsize::new(0); + + #[derive(Clone, Eq, Ord, PartialEq, PartialOrd)] + struct Key(u8); + + impl Drop for Key { + fn drop(&mut self) { + if self.0 != 1 || !WATCH.load(Ordering::SeqCst) { + return; + } + let Some(owner) = OWNER.get().and_then(std::sync::Weak::upgrade) else { + return; + }; + DROPS.fetch_add(1, Ordering::SeqCst); + let Ok(state) = owner.state.try_lock() else { + BLOCKED.fetch_add(1, Ordering::SeqCst); + return; + }; + assert_eq!(state.active, 2); + assert_eq!(state.peers.len(), 2); + assert!(state.peers.keys().all(|key| key.0 != 1)); + assert_eq!(state.peers[&Key(3)].active, 1); + drop(state); + assert!(owner.available(&Key(3))); + } + } + + let owner = Adaptive::new( + Config { + total: 2, + per_key: 1, + backoff: Duration::MAX, + ..config() + }, + Counts::default(), + Instant::now, + ) + .unwrap(); + assert!(OWNER.set(Arc::downgrade(&owner)).is_ok()); + let one = owner.acquire(&Key(1)).unwrap(); + let two = owner.acquire(&Key(2)).unwrap(); + assert!(matches!(owner.acquire(&Key(3)), Err(Error::Overloaded))); + assert_eq!(owner.state.lock().unwrap().active, 2); + drop(one); + WATCH.store(true, Ordering::SeqCst); + + assert!(matches!(owner.acquire(&Key(2)), Err(Error::Overloaded))); + two.observe(Outcome::PeerFailure); + assert!(matches!(owner.acquire(&Key(2)), Err(Error::Unavailable))); + assert!(matches!(owner.acquire(&Key(3)), Err(Error::Overloaded))); + assert_eq!(DROPS.load(Ordering::SeqCst), 0); + { + let mut state = owner.state.lock().unwrap(); + assert_eq!(state.active, 1); + assert_eq!(state.peers.len(), 2); + let idle = state.peers.values_mut().next().unwrap(); + assert_eq!(idle.active, 0); + idle.idle_since = Some(Instant::now() - config().retire_after); + } + assert_eq!(*owner.observer.active.lock().unwrap(), 1); + + let replacement = owner.acquire(&Key(3)).unwrap(); + WATCH.store(false, Ordering::SeqCst); + assert_eq!(DROPS.load(Ordering::SeqCst), 1); + assert_eq!(BLOCKED.load(Ordering::SeqCst), 0); + assert_eq!(*owner.observer.active.lock().unwrap(), 2); + assert_eq!( + *owner.observer.events.lock().unwrap(), + [ + Event::Accepted, + Event::Accepted, + Event::Rejected, + Event::Rejected, + Event::LinkFailure, + Event::CircuitRejected, + Event::Rejected, + Event::Accepted, + ] + ); + drop((two, replacement)); + assert_eq!(owner.state.lock().unwrap().active, 0); + assert_eq!(*owner.observer.active.lock().unwrap(), 0); + assert!(owner.acquire(&Key(3)).is_ok()); + } + + /// Stale success cannot undo failure and shared fences retain active work. + #[test] + fn fences_generation_exclusivity_and_local_pressure() { + let owner = Adaptive::new(config(), Counts::default(), Instant::now).unwrap(); + assert!(owner.hedge_available(&1)); + let old = owner.acquire(&1).unwrap(); + let failed = owner.acquire(&1).unwrap(); + failed.observe(Outcome::PeerFailure); + old.observe(Outcome::Verified); + assert!(!owner.available(&1)); + let fence = failed.clone(); + drop((old, failed)); + assert_eq!(*owner.observer.active.lock().unwrap(), 1); + drop(fence); + owner.state.lock().unwrap().peers.get_mut(&1).unwrap().retry = + Some(Deadline::At(Instant::now())); + let probe = owner.acquire(&1).unwrap(); + assert!(matches!(owner.acquire(&1), Err(Error::Unavailable))); + probe.observe(Outcome::Verified); + assert!(!owner.available(&1)); + drop(probe); + assert!(owner.available(&1)); + let work = owner.acquire(&2).unwrap(); + owner.state.lock().unwrap().updated = Instant::now() - config().backoff; + work.observe(Outcome::LocalPressure); + assert_eq!(*owner.observer.limit.lock().unwrap(), 2); + assert!(owner.available(&2)); + owner.state.lock().unwrap().updated = Instant::now() - config().recovery; + work.observe(Outcome::Verified); + assert_eq!(*owner.observer.limit.lock().unwrap(), 3); + } + + /// Failed probes cannot clear retry, even at the last generation. + #[test] + fn generation_exhaustion_fences_failed_probe_success() { + for generation in [0, u64::MAX - 1, u64::MAX] { + let owner = Adaptive::new(config(), Counts::default(), Instant::now).unwrap(); + drop(owner.acquire(&1).unwrap()); + { + let mut state = owner.state.lock().unwrap(); + let peer = state.peers.get_mut(&1).unwrap(); + peer.generation = Some(generation); + peer.retry = Some(Deadline::At(Instant::now())); + } + let probe = owner.acquire(&1).unwrap(); + let fence = probe.clone(); + probe.observe(Outcome::PeerFailure); + probe.observe(Outcome::Verified); + assert!(owner.state.lock().unwrap().peers[&1].retry.is_some()); + assert!(!owner.available(&1)); + drop(probe); + assert_eq!(*owner.observer.active.lock().unwrap(), 1); + assert!(matches!(owner.acquire(&1), Err(Error::Unavailable))); + drop(fence); + assert_eq!(*owner.observer.active.lock().unwrap(), 0); + assert!(matches!(owner.acquire(&1), Err(Error::Unavailable))); + owner.state.lock().unwrap().peers.get_mut(&1).unwrap().retry = + Some(Deadline::At(Instant::now())); + let recovered = owner.acquire(&1).unwrap(); + recovered.observe(Outcome::Verified); + assert!(!owner.available(&1)); + drop(recovered); + assert!(owner.available(&1)); + } + } + + /// Exhaustion fences every old permit and resets only after final release. + #[test] + fn generation_exhaustion_waits_for_live_permits_before_reuse() { + let owner = Adaptive::new( + Config { + backoff: Duration::ZERO, + recovery: Duration::ZERO, + ..config() + }, + Counts::default(), + Instant::now, + ) + .unwrap(); + let old = owner.acquire(&1).unwrap(); + owner + .state + .lock() + .unwrap() + .peers + .get_mut(&1) + .unwrap() + .generation = Some(u64::MAX); + let failed = owner.acquire(&1).unwrap(); + let fence = failed.clone(); + failed.observe(Outcome::PeerFailure); + failed.observe(Outcome::PeerFailure); + assert_eq!(owner.state.lock().unwrap().peers[&1].limit, 2); + assert!(!owner.available(&1)); + assert!(!owner.hedge_available(&1)); + drop(failed); + drop(fence); + assert_eq!(*owner.observer.active.lock().unwrap(), 1); + assert!(matches!(owner.acquire(&1), Err(Error::Unavailable))); + assert!(owner.acquire(&2).is_ok()); + drop(old); + assert_eq!(*owner.observer.active.lock().unwrap(), 0); + assert!(owner.available(&1)); + let recovered = owner.acquire(&1).unwrap(); + recovered.observe(Outcome::Verified); + drop(recovered); + assert!(owner.available(&1)); + assert_eq!(owner.state.lock().unwrap().peers[&1].limit, 3); + } + + /// Success at the ceiling must not delay an eligible pressure reduction. + #[test] + fn recovery_at_ceiling_preserves_pressure_eligibility() { + fn now() -> Instant { + static NOW: OnceLock = OnceLock::new(); + *NOW.get_or_init(Instant::now) + } + + for recovery in [Duration::ZERO, config().recovery] { + let config = Config { + recovery, + ..config() + }; + let owner = Adaptive::new(config, Counts::default(), now).unwrap(); + let work = owner.acquire(&1).unwrap(); + let updated = now() - config.backoff.max(config.recovery); + owner.state.lock().unwrap().updated = updated; + + for _ in 0..2 { + work.observe(Outcome::Verified); + let state = owner.state.lock().unwrap(); + assert_eq!(state.limit, config.total); + assert_eq!(state.updated, updated); + } + work.observe(Outcome::LocalPressure); + assert_eq!(*owner.observer.limit.lock().unwrap(), config.total / 2); + assert_eq!(owner.state.lock().unwrap().updated, now()); + assert_eq!(owner.state.lock().unwrap().peers[&1].limit, config.per_key); + assert!(owner.available(&1)); + assert_eq!( + *owner.observer.events.lock().unwrap(), + [ + Event::Accepted, + Event::Verified, + Event::Verified, + Event::LocalPressure, + ] + ); + } + } + + /// Restoring the final slot still starts backoff and rate-limits pressure. + #[test] + fn recovery_restoring_slot_advances_pressure_backoff() { + fn now() -> Instant { + static NOW: OnceLock = OnceLock::new(); + *NOW.get_or_init(Instant::now) + } + + let config = config(); + let owner = Adaptive::new(config, Counts::default(), now).unwrap(); + let work = owner.acquire(&1).unwrap(); + { + let mut state = owner.state.lock().unwrap(); + state.limit = config.total - 1; + state.updated = now() - config.recovery; + } + work.observe(Outcome::Verified); + assert_eq!(*owner.observer.limit.lock().unwrap(), config.total); + assert_eq!(owner.state.lock().unwrap().updated, now()); + work.observe(Outcome::LocalPressure); + assert_eq!(*owner.observer.limit.lock().unwrap(), config.total); + + owner.state.lock().unwrap().updated = now() - config.backoff; + work.observe(Outcome::LocalPressure); + assert_eq!(*owner.observer.limit.lock().unwrap(), config.total / 2); + work.observe(Outcome::LocalPressure); + assert_eq!(*owner.observer.limit.lock().unwrap(), config.total / 2); + } + + /// Idle age and recovery use independent clocks without wall-clock sleeps. + mod idle_retirement { + use super::*; + use std::cell::Cell; + + thread_local! { + static CLOCK: Cell = Cell::new(Instant::now()); + } + + fn now() -> Instant { + CLOCK.get() + } + + fn advance(duration: Duration) { + CLOCK.set(now() + duration); + } + + #[test] + fn waits_from_last_owner_and_restarts_after_reuse() { + for outcome in [Outcome::Neutral, Outcome::Verified] { + let config = Config { + capacity: 1, + recovery: Duration::MAX, + ..config() + }; + let owner = Adaptive::new(config, Counts::default(), now).unwrap(); + let first = owner.acquire(&1).unwrap(); + let last = owner.acquire(&1).unwrap(); + let fence = last.clone(); + advance(config.retire_after); + first.observe(outcome); + drop(first); + advance(config.retire_after); + drop(last); + assert_eq!(owner.state.lock().unwrap().peers[&1].active, 1); + assert!(matches!(owner.acquire(&2), Err(Error::Overloaded))); + drop(fence); + assert!(matches!(owner.acquire(&2), Err(Error::Overloaded))); + + advance(config.retire_after - Duration::from_nanos(1)); + assert!(matches!(owner.acquire(&2), Err(Error::Overloaded))); + let reused = owner.acquire(&1).unwrap(); + advance(config.retire_after); + assert!(matches!(owner.acquire(&2), Err(Error::Overloaded))); + drop(reused); + assert!(matches!(owner.acquire(&2), Err(Error::Overloaded))); + advance(config.retire_after - Duration::from_nanos(1)); + assert!(matches!(owner.acquire(&2), Err(Error::Overloaded))); + advance(Duration::from_nanos(1)); + let replacement = owner.acquire(&2).unwrap(); + assert!(!owner.state.lock().unwrap().peers.contains_key(&1)); + drop(replacement); + assert_eq!(*owner.observer.active.lock().unwrap(), 0); + } + } + + #[test] + fn probe_release_requires_idle_age_and_expired_backoff() { + for outcome in [Outcome::Neutral, Outcome::Verified, Outcome::PeerFailure] { + let config = Config { + capacity: 1, + backoff: Duration::from_secs(3), + retire_after: Duration::from_secs(2), + ..config() + }; + let owner = Adaptive::new(config, Counts::default(), now).unwrap(); + let failed = owner.acquire(&1).unwrap(); + failed.observe(Outcome::PeerFailure); + drop(failed); + advance(config.backoff); + let probe = owner.acquire(&1).unwrap(); + probe.observe(outcome); + advance(config.backoff + config.retire_after); + assert!(matches!(owner.acquire(&2), Err(Error::Overloaded))); + drop(probe); + assert!(matches!(owner.acquire(&2), Err(Error::Overloaded))); + advance(config.retire_after - Duration::from_nanos(1)); + assert!(matches!(owner.acquire(&2), Err(Error::Overloaded))); + advance(Duration::from_nanos(1)); + if outcome != Outcome::Verified { + assert!(matches!(owner.acquire(&2), Err(Error::Overloaded))); + assert!(matches!(owner.acquire(&1), Err(Error::Unavailable))); + advance(config.backoff - config.retire_after); + } + assert!(owner.acquire(&2).is_ok()); + } + } + + #[test] + fn acquisition_and_drop_preserve_recovery_throttle() { + let config = Config { + backoff: Duration::ZERO, + recovery: Duration::from_secs(10), + ..config() + }; + let owner = Adaptive::new(config, Counts::default(), now).unwrap(); + let failed = owner.acquire(&1).unwrap(); + failed.observe(Outcome::PeerFailure); + drop(failed); + let updated = now(); + advance(config.recovery / 2); + let probe = owner.acquire(&1).unwrap(); + probe.observe(Outcome::Verified); + drop(probe); + assert_eq!(owner.state.lock().unwrap().peers[&1].updated, updated); + assert_eq!(owner.state.lock().unwrap().peers[&1].limit, 2); + advance(config.recovery / 2); + let work = owner.acquire(&1).unwrap(); + work.observe(Outcome::Verified); + drop(work); + assert_eq!(owner.state.lock().unwrap().peers[&1].limit, 3); + let updated = now(); + advance(config.recovery - Duration::from_nanos(1)); + let early = owner.acquire(&1).unwrap(); + early.observe(Outcome::Verified); + drop(early); + assert_eq!(owner.state.lock().unwrap().peers[&1].updated, updated); + assert_eq!(owner.state.lock().unwrap().peers[&1].limit, 3); + advance(Duration::from_nanos(1)); + let ready = owner.acquire(&1).unwrap(); + ready.observe(Outcome::Verified); + assert_eq!(owner.state.lock().unwrap().peers[&1].limit, 4); + } + + #[test] + fn zero_and_maximum_idle_age_preserve_live_ownership() { + for retire_after in [Duration::ZERO, Duration::MAX] { + let owner = Adaptive::new( + Config { + capacity: 1, + retire_after, + ..config() + }, + Counts::default(), + now, + ) + .unwrap(); + let work = owner.acquire(&1).unwrap(); + advance(Duration::from_secs(120)); + assert!(matches!(owner.acquire(&2), Err(Error::Overloaded))); + drop(work); + if retire_after.is_zero() { + assert!(owner.acquire(&2).is_ok()); + } else { + assert!(matches!(owner.acquire(&2), Err(Error::Overloaded))); + assert!(owner.acquire(&1).is_ok()); + } + } + } + } + + /// Only sufficiently old, idle, eligible peer records may be retired. + #[test] + fn capacity_preserves_live_work_and_stale_backoff_then_retires_idle() { + let owner = Adaptive::new(config(), Counts::default(), Instant::now).unwrap(); + let one = owner.acquire(&1).unwrap(); + let two = owner.acquire(&2).unwrap(); + assert!(matches!(owner.acquire(&3), Err(Error::Overloaded))); + one.observe(Outcome::PeerFailure); + drop(one); + owner + .state + .lock() + .unwrap() + .peers + .get_mut(&1) + .unwrap() + .idle_since = Some(Instant::now() - Duration::from_secs(61)); + assert!(matches!(owner.acquire(&3), Err(Error::Overloaded))); + assert!(owner.state.lock().unwrap().peers.contains_key(&1)); + owner.state.lock().unwrap().peers.get_mut(&1).unwrap().retry = + Some(Deadline::At(Instant::now())); + let probe = owner.acquire(&1).unwrap(); + drop(probe); + assert!(!owner.available(&1)); + assert!(matches!(owner.acquire(&3), Err(Error::Overloaded))); + drop(two); + owner + .state + .lock() + .unwrap() + .peers + .get_mut(&2) + .unwrap() + .idle_since = Some(Instant::now() - Duration::from_secs(61)); + assert!(owner.acquire(&3).is_ok()); + assert_eq!(owner.state.lock().unwrap().peers.len(), 2); + } + + /// Reject invalid geometry and enforce aggregate and per-key caps. + #[test] + fn validation_and_caps() { + for (total, per_key, capacity) in [(0, 1, 1), (1, 0, 1), (1, 2, 1), (1, 1, 0)] { + assert!(matches!( + Adaptive::::new( + Config { + total, + per_key, + capacity, + ..config() + }, + Counts::default(), + Instant::now + ), + Err(Error::InvalidInput) + )); + } + let owner = Adaptive::new( + Config { + total: 2, + per_key: 1, + ..config() + }, + Counts::default(), + Instant::now, + ) + .unwrap(); + let one = owner.acquire(&1).unwrap(); + assert!(matches!(owner.acquire(&1), Err(Error::Overloaded))); + let two = owner.acquire(&2).unwrap(); + assert!(!owner.hedge_available(&3)); + assert!(matches!(owner.acquire(&3), Err(Error::Overloaded))); + one.observe(Outcome::Neutral); + drop((one, two)); + assert_eq!(*owner.observer.active.lock().unwrap(), 0); + } + + /// Recovery saturates even when limits span the machine word. + #[test] + fn full_width_limits_recover_without_overflow() { + let owner = Adaptive::new( + Config { + total: usize::MAX, + per_key: usize::MAX, + recovery: Duration::ZERO, + ..config() + }, + Counts::default(), + Instant::now, + ) + .unwrap(); + let permit = owner.acquire(&()).unwrap(); + permit.observe(Outcome::Verified); + assert_eq!(*owner.observer.limit.lock().unwrap(), usize::MAX); + assert_eq!(owner.state.lock().unwrap().peers[&()].limit, usize::MAX); + } + + /// Probe release invokes the injected clock only after releasing adaptive state. + #[test] + fn probe_drop_calls_clock_outside_state_lock() { + let owner = Adaptive::new(config(), Counts::default(), reentrant_clock).unwrap(); + CLOCK_OWNER.set(Arc::downgrade(&owner)).unwrap(); + let failed = owner.acquire(&1).unwrap(); + failed.observe(Outcome::PeerFailure); + drop(failed); + owner.state.lock().unwrap().peers.get_mut(&1).unwrap().retry = + Some(Deadline::At(Instant::now())); + let probe = owner.acquire(&1).unwrap(); + drop(probe); + } +} diff --git a/cmd/racer-dataplane/flow/src/coalesce.rs b/cmd/racer-dataplane/flow/src/coalesce.rs new file mode 100644 index 000000000..ab5df0795 --- /dev/null +++ b/cmd/racer-dataplane/flow/src/coalesce.rs @@ -0,0 +1,2534 @@ +//! Worker-local keyed cohorts with bounded registration and leader re-election. +//! +//! Callers own execution, cancellation, admission policy, and result semantics. +//! A cloned registration retains the same waiter and leadership until its final +//! handle drops. Complete a cohort only after the real operation has finished. + +use std::cell::{Cell, RefCell}; +use std::collections::{HashMap, hash_map::RandomState}; +use std::hash::{BuildHasher, Hash}; +use std::rc::Rc; +use std::task::{Poll, Waker}; + +/// Local bounds. Zero waiters rejects all joins; zero attempts immediately +/// broadcasts the caller-supplied exhausted value without electing a leader. +#[derive(Clone, Copy, Debug)] +pub struct Limits { + /// Maximum distinct registrations sharing one cohort. + pub waiters_per_cohort: usize, + + /// Maximum leadership elections before broadcasting exhaustion. + pub attempts_per_cohort: usize, +} + +/// Admission would exceed a key, waiter, or identifier bound. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub struct CapacityError; + +/// A registration may start an attempt or observe its cohort's result. +#[derive(Clone, Debug, Eq, PartialEq)] +pub enum Event { + /// This registration owns the next attempt. + Lead, + + /// The operation finished or the attempt budget was exhausted. + Complete(V), +} + +/// Completed cohorts close admission immediately but continue charging live +/// registrations. `S::default()` is called for each map, including new cohorts. +pub struct Table { + active: RefCell, S>>, + + registrations: Cell, + + limits: Limits, + + exhausted: V, +} + +/// Local cohort ownership shared between the admission index and registrations. +type CohortOwner = Rc>>; + +impl Table { + /// Create an empty worker-local table and its attempt-exhaustion result. + pub fn new(limits: Limits, exhausted: V) -> Self { + Self { + active: RefCell::new(HashMap::default()), + registrations: Cell::new(0), + limits, + exhausted, + } + } +} + +impl Table { + /// Count distinct registrations, including readers of completed cohorts. + pub fn registration_count(&self) -> usize { + self.registrations.get() + } + + /// Count cohorts still accepting new registrations. + pub fn active_count(&self) -> usize { + self.active.borrow().len() + } +} + +impl Table { + /// Capacity bounds active keys and, multiplied by the waiter limit, all live + /// registrations (including readers of completed cohorts). Clones count once. + pub fn join( + self: &Rc, + key: K, + capacity: usize, + ) -> Result, CapacityError> { + if self.limits.waiters_per_cohort == 0 { + return Err(CapacityError); + } + if self.registrations.get() >= capacity.saturating_mul(self.limits.waiters_per_cohort) { + return Err(CapacityError); + } + let index_key = { + let active = self.active.borrow(); + if active.contains_key(&key) { + None + } else { + if active.len() >= capacity { + return Err(CapacityError); + } + drop(active); + Some(key.clone()) + } + }; + // Key cloning can admit work. Recheck all bounds and cohort membership. + if self.registrations.get() >= capacity.saturating_mul(self.limits.waiters_per_cohort) { + return Err(CapacityError); + } + let mut active = self.active.borrow_mut(); + let cohort = if let Some(cohort) = active.get(&key) { + cohort.clone() + } else { + if active.len() >= capacity { + return Err(CapacityError); + } + let cohort = Rc::new(RefCell::new(Cohort::default())); + active.insert( + index_key.expect("missing cohort key was cloned"), + cohort.clone(), + ); + cohort + }; + let id = { + let mut state = cohort.borrow_mut(); + if state.waiters.len() >= self.limits.waiters_per_cohort { + return Err(CapacityError); + } + let id = state.next; + state.next = state.next.checked_add(1).ok_or(CapacityError)?; + state.waiters.insert(id, None); + id + }; + self.registrations.set(self.registrations.get() + 1); + Ok(Registration { + owner: Rc::new(RegistrationOwner { + table: self.clone(), + key, + cohort, + id, + }), + }) + } +} + +/// A worker-local handle to one waiter. Clones retain its leadership and charge; +/// only dropping the last handle detaches it. Cloning never clones the key. +/// Handles cannot move or be shared across workers: +/// ```compile_fail +/// fn require_send() {} +/// require_send::>(); +/// ``` +/// ```compile_fail +/// fn require_sync() {} +/// require_sync::>(); +/// ``` +pub struct Registration { + owner: Rc>, +} + +/// Owns exactly one registration charge and releases it once on final Rc drop. +struct RegistrationOwner { + table: Rc>, + + key: K, + + cohort: CohortOwner, + + id: u64, +} + +impl Clone for Registration { + /// Retain the same waiter owner without duplicating its key or charge. + fn clone(&self) -> Self { + Self { + owner: self.owner.clone(), + } + } +} + +impl Registration { + /// Whether this is the last handle for this registration, not the cohort. + pub fn is_only_handle(&self) -> bool { + Rc::strong_count(&self.owner) == 1 + } + + /// Release leadership and notify the cohort after an attempt completes. + /// Retry policy and whether this caller should detach belong to the caller. + pub fn retry(&self) { + let wakers = { + let mut state = self.owner.cohort.borrow_mut(); + if state.leader == Some(self.owner.id) { + state.leader = None; + } + state.take_wakers() + }; + wake_all(wakers); + } + + /// Broadcast a caller-owned value and close admission before waking readers. + /// The operation owner must call this only after completion, not cancellation. + /// The first result is final; later calls leave it unchanged. + pub fn finish(&self, result: V) { + let wakers = { + let mut state = self.owner.cohort.borrow_mut(); + if state.result.is_some() { + return; + } + state.result = Some(result); + state.leader = None; + state.take_wakers() + }; + self.owner.remove_active(); + wake_all(wakers); + } +} + +impl Registration { + /// Register the latest wake target and elect at most once per attempt. + pub fn event(&self, waker: &Waker) -> Poll> { + let waker = waker.clone(); + let retired = { + let mut state = self.owner.cohort.borrow_mut(); + if state.result.is_some() { + Some(waker) + } else { + state.waiters.insert(self.owner.id, Some(waker)).flatten() + } + }; + drop(retired); + + // Clone and drop callbacks can retry, elect, or finish. Decide from the + // current state only after those callbacks have released their borrows. + let mut state = self.owner.cohort.borrow_mut(); + if let Some(result) = &state.result { + return Poll::Ready(Event::Complete(result.clone())); + } + if state.leader.is_none() { + if state.attempts >= self.owner.table.limits.attempts_per_cohort { + drop(state); + let result = self.owner.table.exhausted.clone(); + self.finish(result.clone()); + return Poll::Ready(Event::Complete(result)); + } + state.attempts += 1; + state.leader = Some(self.owner.id); + return Poll::Ready(Event::Lead); + } + Poll::Pending + } +} + +impl RegistrationOwner { + /// Remove only this cohort, never a replacement admitted under the same key. + fn remove_active(&self) { + let mut active = self.table.active.borrow_mut(); + if active + .get(&self.key) + .is_some_and(|entry| Rc::ptr_eq(entry, &self.cohort)) + { + active.remove(&self.key); + } + } +} + +impl Drop for RegistrationOwner { + /// Detach once, release the charge, then wake followers outside all borrows. + fn drop(&mut self) { + let (empty, retired, wakers) = { + let mut state = self.cohort.borrow_mut(); + let retired = state.waiters.remove(&self.id); + let wakers = if state.leader == Some(self.id) { + state.leader = None; + state.take_wakers() + } else { + Vec::new() + }; + (state.waiters.is_empty(), retired, wakers) + }; + self.table + .registrations + .set(self.table.registrations.get() - 1); + if empty { + self.remove_active(); + } + drop(retired); + wake_all(wakers); + } +} + +/// Election state shared by distinct registrations for the same key. +struct Cohort { + next: u64, + + leader: Option, + + attempts: usize, + + waiters: HashMap, S>, + + result: Option, +} + +impl Default for Cohort { + /// Start with no waiters, leadership, attempts, or published result. + fn default() -> Self { + Self { + next: 0, + leader: None, + attempts: 0, + waiters: HashMap::default(), + result: None, + } + } +} + +impl Cohort { + /// Drain wake targets so callbacks can run after releasing the cohort borrow. + fn take_wakers(&mut self) -> Vec { + self.waiters.values_mut().filter_map(Option::take).collect() + } +} + +/// Deliver collected notifications only after the caller releases owner borrows. +fn wake_all(wakers: Vec) { + for waker in wakers { + waker.wake(); + } +} + +/// Shared-result cohorts whose callers own execution rather than electing leaders. +/// Dropping receivers cannot remove real work. Dropping a completion sender +/// closes receivers but keeps its entry until completion is confirmed. +pub mod shared { + use futures::FutureExt; + use futures::channel::oneshot; + use futures::future::{LocalBoxFuture, Shared}; + use std::cell::RefCell; + use std::collections::BTreeMap; + use std::rc::Rc; + + /// Cloneable, worker-local result receiver independent of execution ownership. + pub type Receiver = Shared>; + + /// Ordered index retaining work until its completion owner explicitly finishes. + pub struct Table { + entries: RefCell>>, + } + + impl Default for Table { + /// Create an empty operation index. + fn default() -> Self { + Self { + entries: RefCell::new(BTreeMap::new()), + } + } + } + + impl Table { + /// Count operations, including those whose completion sender was lost. + pub fn len(&self) -> usize { + self.entries.borrow().len() + } + + /// Whether no operation remains indexed. + pub fn is_empty(&self) -> bool { + self.entries.borrow().is_empty() + } + } + + impl Table { + /// Join an existing result without taking ownership of its execution. + pub fn get(&self, key: &K) -> Option> { + self.entries.borrow().get(key).cloned() + } + } + + impl Table { + /// Called after miss-only admission, without yielding between get and start. + /// Recheck after key cloning: an occupied key joins without completion + /// authority. Start work only when a completion owner is returned. + pub fn start( + self: &Rc, + key: K, + closed: V, + ) -> (Receiver, Option>) { + let index_key = key.clone(); + let (send, receive) = oneshot::channel(); + let receive = async move { receive.await.unwrap_or(closed) } + .boxed_local() + .shared(); + let existing = { + let mut entries = self.entries.borrow_mut(); + // Keep the unused key and receiver outside this borrow on a join. + if let Some(existing) = entries.get(&index_key) { + Some(existing.clone()) + } else { + entries.insert(index_key, receive.clone()); + None + } + }; + if let Some(existing) = existing { + return (existing, None); + } + ( + receive, + Some(Completion { + table: self.clone(), + key, + send, + }), + ) + } + } + + /// Sole completion authority. Dropping it reports closure, not real completion, + /// so its table entry deliberately remains occupied. + pub struct Completion { + table: Rc>, + + key: K, + + send: oneshot::Sender, + } + + impl Completion { + /// Real completion closes admission before notifying any old readers. + pub fn finish(self, value: V) { + let removed = self.table.entries.borrow_mut().remove(&self.key); + // Final receiver destruction can reenter the table. + drop(removed); + let _ = self.send.send(value); + } + } + + /// Shared-result ownership and replacement contracts. + #[cfg(test)] + mod tests { + use super::*; + use futures::executor::block_on; + + #[test] + fn occupied_start_drops_unused_key_and_closed_value_outside_borrow() { + use std::cell::Cell; + + thread_local! { + static ON_KEY_DROP: RefCell>> = RefCell::new(None); + } + + #[derive(Clone, Eq, PartialEq, Ord, PartialOrd)] + struct Key(u32); + + impl Drop for Key { + fn drop(&mut self) { + let callback = ON_KEY_DROP.with(|slot| slot.borrow_mut().take()); + if let Some(callback) = callback { + callback(); + } + } + } + + struct Closed(Option>); + + impl Drop for Closed { + fn drop(&mut self) { + if let Some(callback) = self.0.take() { + callback(); + } + } + } + + let table = Rc::new(Table::default()); + let (first, completion) = table.start(Key(1), Rc::new(Closed(None))); + drop(first); + let key_dropped = Rc::new(Cell::new(false)); + let closed_dropped = Rc::new(Cell::new(false)); + ON_KEY_DROP.with(|slot| { + let table = table.clone(); + let observed = key_dropped.clone(); + *slot.borrow_mut() = Some(Box::new(move || { + assert_eq!(table.entries.borrow_mut().len(), 1); + observed.set(true); + })); + }); + let closed = { + let table = table.clone(); + let observed = closed_dropped.clone(); + Rc::new(Closed(Some(Box::new(move || { + assert_eq!(table.entries.borrow_mut().len(), 1); + observed.set(true); + })))) + }; + let (joined, rejected) = table.start(Key(1), closed); + assert!(rejected.is_none()); + assert!(key_dropped.get()); + assert!(closed_dropped.get()); + let result = Rc::new(Closed(None)); + completion.unwrap().finish(result.clone()); + assert!(Rc::ptr_eq(&block_on(joined), &result)); + assert!(table.is_empty()); + } + + /// Final receiver destruction can reenter through its retained closed value. + #[test] + fn completion_drops_last_receiver_outside_table_borrow() { + use std::cell::Cell; + use std::sync::Arc; + use std::task::{Context, Wake, Waker}; + + thread_local! { + static ON_DROP: RefCell>> = RefCell::new(None); + } + + struct ReentrantDrop; + + impl Wake for ReentrantDrop { + fn wake(self: Arc) { + panic!("closed result waker must only be dropped"); + } + } + + impl Drop for ReentrantDrop { + fn drop(&mut self) { + let callback = ON_DROP.with(|slot| slot.borrow_mut().take()); + if let Some(callback) = callback { + callback(); + } + } + } + + let table = Rc::new(Table::default()); + let dropped = Rc::new(Cell::new(false)); + let closed = Some(Waker::from(Arc::new(ReentrantDrop))); + let (mut receive, completion) = table.start(1, closed); + assert!( + receive + .poll_unpin(&mut Context::from_waker(Waker::noop())) + .is_pending() + ); + drop(receive); + assert_eq!(table.len(), 1); + + let nested = table.clone(); + let observed = dropped.clone(); + ON_DROP.with(|slot| { + *slot.borrow_mut() = Some(Box::new(move || { + assert!(nested.is_empty()); + assert!(nested.get(&1).is_none()); + let (replacement, completion) = nested.start(1, None); + assert_eq!(nested.len(), 1); + completion.unwrap().finish(None); + assert!(block_on(replacement).is_none()); + observed.set(true); + })); + }); + + assert!(!dropped.get()); + completion.unwrap().finish(None); + assert!(dropped.get()); + assert!(table.is_empty()); + assert!(ON_DROP.with(|slot| slot.borrow().is_none())); + } + + /// Results remain readable after completion admits a replacement. + #[test] + fn success_miss_failure_broadcast_and_replacement() { + for value in [Ok(Some(7)), Ok(None), Err("io")] { + let table = Rc::new(Table::default()); + let (first, completion) = table.start(1, Err("closed")); + let second = table.get(&1).unwrap(); + assert_eq!(table.len(), 1); + completion.unwrap().finish(value); + assert!(table.is_empty()); + let (next, completion) = table.start(1, Err("closed")); + assert_eq!(block_on(first), value); + assert_eq!(block_on(second), value); + assert_eq!(table.len(), 1); + completion.unwrap().finish(Ok(Some(8))); + assert_eq!(block_on(next), Ok(Some(8))); + } + } + + /// Neither receiver cancellation nor sender loss pretends work completed. + #[test] + fn all_readers_drop_keeps_completion_owner_and_lost_sender_stays_closed() { + let table = Rc::new(Table::default()); + let (receive, completion) = table.start(1, Err::("closed")); + drop(receive); + assert_eq!(table.len(), 1); + let late = table.get(&1).unwrap(); + completion.unwrap().finish(Ok(9)); + assert_eq!(block_on(late), Ok(9)); + assert!(table.is_empty()); + let (receive, completion) = table.start(1, Err("closed")); + drop(completion); + assert_eq!(block_on(receive), Err("closed")); + assert_eq!(block_on(table.get(&1).unwrap()), Err("closed")); + assert_eq!(table.len(), 1); + } + } +} + +pub mod flight { + //! Owned flight indexing, generation fences, and bounded lifecycle sweeps. + //! + //! Entry hooks own result interpretation and policy. A sweep refreshes each + //! selected entry once and removes it only when the hook reports quiescence. + //! Callers collect wakes while borrowed and dispatch them after releasing locks. + //! Clone incoming wakers before borrowing. Return retired wakers from `update` + //! and drop them after it returns, including unused copies of the same target. + + use std::cell::RefCell; + use std::collections::{BTreeMap, HashMap, hash_map::RandomState}; + use std::hash::{BuildHasher, Hash}; + use std::rc::Rc; + use std::task::Waker; + + /// Run a synchronous owner transaction, then notify outside its mutable borrow. + pub fn update(owner: &RefCell, f: impl FnOnce(&mut T, &mut Vec) -> R) -> R { + let mut wakes = Vec::new(); + let result = { + let mut state = owner.borrow_mut(); + f(&mut state, &mut wakes) + }; + super::wake_all(wakes); + result + } + + /// A monotonic identifier or acquisition budget has been exhausted. + #[derive(Clone, Copy, Debug, Eq, PartialEq)] + pub struct Exhausted; + + /// A registration, generation, or operation no longer matches its owner. + #[derive(Clone, Copy, Debug, Eq, PartialEq)] + pub struct Stale; + + /// Key equality belongs to the adapter; this identity fences owner, incarnation, + /// and acquisition generation independently of any application key schema. + #[derive(Clone)] + pub struct Identity { + /// Local table owner; pointer identity fences unrelated workers and tables. + pub owner: Rc<()>, + + /// Monotonic identity of one entry admission. + pub incarnation: u64, + + /// Acquisition attempt within this incarnation. + pub generation: u64, + } + + impl Identity { + /// Compare entry ownership without invalidating waiters across retries. + pub fn same_registration(&self, current: &Self) -> bool { + Rc::ptr_eq(&self.owner, ¤t.owner) && self.incarnation == current.incarnation + } + + /// Validate an acquisition capability against the current generation. + pub fn validate(&self, current: &Self) -> Result<(), Stale> { + if !self.same_registration(current) || self.generation != current.generation { + return Err(Stale); + } + Ok(()) + } + + /// Called only after the adapter confirms eligibility and completion fences. + fn advance(&mut self, limit: u64) -> Result<(), Exhausted> { + if self.generation >= limit { + return Err(Exhausted); + } + self.generation = self.generation.checked_add(1).ok_or(Exhausted)?; + Ok(()) + } + } + + /// Application lifecycle hooks. Refresh must be bounded and only enqueue wakes. + /// Quiescence must include both detached waiters and actual retained operations, + /// never merely cancellation requested by a caller or expiration of a deadline. + pub trait Entry { + /// Refresh bounded lifecycle work and enqueue notifications without waking. + fn refresh(&mut self, wakes: &mut Vec); + + /// Report that neither waiters nor retained operations need this entry. + fn quiescent(&self) -> bool; + } + + /// Owned membership with a synchronized, bounded sweep index. + /// Direct map mutation is intentionally unavailable: + /// ```compile_fail + /// let mut table = flow_control::coalesce::flight::Table::::default(); + /// table.entries.clear(); + /// ``` + /// Even an empty table belongs to its worker: + /// ```compile_fail + /// fn require_send() {} + /// require_send::>(); + /// ``` + /// ```compile_fail + /// fn require_sync() {} + /// require_sync::>(); + /// ``` + pub struct Table { + entries: HashMap, S>, + + sweep: BTreeMap, + + next_sweep_id: u64, + + cursor: Cursor, + + incarnation: Counter, + + next_waiter: Counter, + + next_operation: Counter, + + stopping: bool, + + drain_waker: Option, + + /// Keep even an empty table local to the worker that drives its lifecycle. + local: std::marker::PhantomData>, + } + + /// Membership identity owned by the table, never by mutable application state. + struct Indexed { + sweep_id: u64, + + entry: E, + } + + impl Default for Table { + /// Create an empty worker-local index with fresh identifier sequences. + fn default() -> Self { + Self { + entries: HashMap::default(), + sweep: BTreeMap::new(), + next_sweep_id: 0, + cursor: Cursor::default(), + incarnation: Counter::default(), + next_waiter: Counter::default(), + next_operation: Counter::default(), + stopping: false, + drain_waker: None, + local: std::marker::PhantomData, + } + } + } + + impl Table { + /// Count entries, including those waiting for real operation completion. + pub fn len(&self) -> usize { + self.entries.len() + } + + /// Whether all owned entries have been removed. + pub fn is_empty(&self) -> bool { + self.entries.is_empty() + } + + /// Inspect application state without exposing the membership index. + pub fn values(&self) -> impl Iterator { + self.entries.values().map(|indexed| &indexed.entry) + } + + /// Allocate a never-reused waiter identifier. + pub fn next_waiter_id(&mut self) -> Result { + self.next_waiter.next_id() + } + + /// Allocate a never-reused operation identifier. + pub fn next_operation_id(&mut self) -> Result { + self.next_operation.next_id() + } + + /// Whether shutdown was requested; the adapter must enforce admission closure. + pub fn is_stopping(&self) -> bool { + self.stopping + } + + /// Retain the latest drain task's wake target and return the previous one. + /// Clone before borrowing the owner; drop the returned waker after releasing + /// that borrow. With `update`, return it from the transaction closure. + pub fn register_drain(&mut self, waker: Waker) -> Option { + self.drain_waker.replace(waker) + } + + /// Enqueue a pending drain notification at most once. + pub fn notify_drain(&mut self, wakes: &mut Vec) { + if let Some(waker) = self.drain_waker.take() { + wakes.push(waker); + } + } + + /// Allocate an entry incarnation fenced by its local owner. + pub fn identity(&mut self, owner: Rc<()>) -> Result { + Ok(Identity { + owner, + incarnation: self.incarnation.next_id()?, + generation: 0, + }) + } + } + + impl Table { + /// Look up the current entry for a key. + pub fn get(&self, key: &K) -> Option<&E> { + self.entries.get(key).map(|indexed| &indexed.entry) + } + + /// Mutate application state without affecting table-owned sweep membership. + /// The adapter remains responsible for its own incarnation-fence semantics. + pub fn get_mut(&mut self, key: &K) -> Option<&mut E> { + self.entries.get_mut(key).map(|indexed| &mut indexed.entry) + } + + /// Whether a key currently has an owned entry. + pub fn contains_key(&self, key: &K) -> bool { + self.entries.contains_key(key) + } + } + + impl Table { + /// Insert only into a vacant key after the adapter checks admission. + /// Occupied keys, even quiescent ones, return the rejected key and entry. + /// Remove quiescent entries explicitly before admitting replacements. + /// Return rejection from `update` and drop it outside the owner borrow. + #[must_use = "drop the rejected key and entry outside the owner borrow"] + pub fn insert(&mut self, key: K, entry: E) -> Result<(), (K, E)> { + if self.entries.contains_key(&key) { + return Err((key, entry)); + } + let sweep_id = self.allocate_sweep_id(); + self.entries + .insert(key.clone(), Indexed { sweep_id, entry }); + self.sweep.insert(sweep_id, key); + Ok(()) + } + + /// Find a free internal sweep slot without consuming incarnation IDs. + fn allocate_sweep_id(&mut self) -> u64 { + // Unlike externally visible incarnation fences, these IDs can be reused + // after removal. At most len + 1 probes find a free ID, even after wrap. + // Vacant insertion needs no incarnation allocation or fallible counter. + for _ in 0..=self.sweep.len() { + self.next_sweep_id = self.next_sweep_id.wrapping_add(1); + if !self.sweep.contains_key(&self.next_sweep_id) { + return self.next_sweep_id; + } + } + unreachable!("a finite in-memory table cannot occupy every u64 sweep ID") + } + } + + impl Table { + /// Used after explicit detach or completion as well as by background sweeps. + /// Return the entry for destruction after releasing the owner borrow. + #[must_use = "drop the removed entry outside the owner borrow"] + pub fn remove_quiescent(&mut self, key: &K) -> Option { + if !self.get(key).is_some_and(Entry::quiescent) { + return None; + } + let entry = self.entries.remove(key).expect("quiescent entry"); + self.sweep.remove(&entry.sweep_id); + Some(entry.entry) + } + } + + impl Table { + /// Mark shutdown and visit each entry synchronously. The adapter must reject + /// admission, choose settlement policy, and drive actual operation completions. + pub fn stop( + &mut self, + wakes: &mut Vec, + mut stop: impl FnMut(&mut E, &mut Vec), + ) { + self.stopping = true; + for entry in self.entries.values_mut() { + stop(&mut entry.entry, wakes); + } + } + } + + impl Table { + /// Refresh at most the budgeted entry count and remove quiescent entries. + /// Return removed entries from `update` and drop them outside its borrow. + #[must_use = "drop the removed entries outside the owner borrow"] + pub fn sweep(&mut self, budget: usize, wakes: &mut Vec) -> Vec { + let mut removed = Vec::new(); + for _ in 0..budget.min(self.sweep.len()) { + let Some((_, key)) = self.cursor.next(&self.sweep) else { + break; + }; + let key = key.clone(); + if let Some(entry) = self.get_mut(&key) { + entry.refresh(wakes); + } + if let Some(entry) = self.remove_quiescent(&key) { + removed.push(entry); + } + } + if self.entries.is_empty() { + self.notify_drain(wakes); + } + removed + } + } + + /// Two-phase completion slots. Taking a resource leaves its slot occupied while + /// its destructor runs outside the table borrow. Only explicit completion clears + /// that slot; dropping an external completion token does not touch this owner. + pub struct Operations { + slots: HashMap, S>, + + /// Resource destruction and completion must run on the owning worker. + local: std::marker::PhantomData>, + } + + impl Default for Operations { + /// Create an empty local resource owner. + fn default() -> Self { + Self { + slots: HashMap::default(), + local: std::marker::PhantomData, + } + } + } + + impl Operations { + /// Count retained resources and occupied completion tombstones. + pub fn len(&self) -> usize { + self.slots.len() + } + + /// Whether every resource has been taken and explicitly completed. + pub fn is_empty(&self) -> bool { + self.slots.is_empty() + } + } + + impl Operations { + /// Retain resources under a caller-allocated unique operation identifier. + pub fn insert(&mut self, id: u64, resources: R) { + self.slots.insert(id, OperationSlot::Retained(resources)); + } + + /// Take resources for destruction outside the owner borrow, retaining a fence. + pub fn take(&mut self, id: u64) -> Result { + let slot = self.slots.get_mut(&id).ok_or(Stale)?; + match std::mem::replace(slot, OperationSlot::Completing) { + OperationSlot::Retained(resources) => Ok(resources), + OperationSlot::Completing => Err(Stale), + } + } + + /// Clear a taken slot after the caller has finished resource destruction. + pub fn complete(&mut self, id: u64) -> Result<(), Stale> { + if !matches!(self.slots.get(&id), Some(OperationSlot::Completing)) { + return Err(Stale); + } + self.slots.remove(&id); + Ok(()) + } + } + + /// An occupied operation owns either resources or their unfinished completion fence. + enum OperationSlot { + /// Resources must be taken and destroyed before completion is acknowledged. + Retained(R), + + /// Resources left the owner, but actual completion has not been confirmed. + Completing, + } + + /// Monotonic IDs are never reused, even after an entry is removed. + #[derive(Default)] + struct Counter(u64); + + impl Counter { + /// Allocate the next identifier without wrapping or reusing an old value. + fn next_id(&mut self) -> Result { + self.0 = self.0.checked_add(1).ok_or(Exhausted)?; + Ok(self.0) + } + } + + /// Stable round-robin selection without a scan, including wrap after removal. + #[derive(Default)] + struct Cursor(u64); + + impl Cursor { + /// Select the next live identifier, wrapping to the first when necessary. + fn next<'a, V>(&mut self, entries: &'a BTreeMap) -> Option<(u64, &'a V)> { + let (&id, value) = entries + .range(( + std::ops::Bound::Excluded(self.0), + std::ops::Bound::Unbounded, + )) + .next() + .or_else(|| entries.first_key_value())?; + self.0 = id; + Some((id, value)) + } + } + + /// Waiter lifecycle and result-independent flight transitions. + pub mod state { + use super::{Cursor, Exhausted, Identity, Stale}; + use std::collections::BTreeMap; + use std::ops::{Deref, DerefMut}; + use std::task::Waker; + use std::time::Instant; + + /// Request-owned policy facts, not acquisition credits or result interpretation. + pub trait WaiterPolicy { + /// Lightweight failure value retained with a waiter. + type Error: Copy; + + /// Check cancellation, deadline, and any other request-owned policy. + fn check(&self) -> Option; + + /// Return the request's fixed deadline for bounded expiry indexing. + fn deadline(&self) -> Instant; + } + + /// One attached caller's policy, acquisition eligibility, and notification state. + pub struct Waiter { + /// Caller-owned policy facts, available through dereferencing as well. + policy: W, + + /// Whether this caller can acquire rather than only observe results. + acquisition: bool, + + /// Whether partial publication permits another attempt for this caller. + pub complete: bool, + + /// Whether this caller has already received its acquisition opportunity. + pub issued: bool, + + /// Sticky policy or cancellation failure. + pub error: Option, + + /// Latest wake target, taken once when notification is enqueued. + pub waker: Option, + } + + impl Deref for Waiter { + type Target = W; + + /// Read the caller-owned policy facts. + fn deref(&self) -> &W { + &self.policy + } + } + + impl DerefMut for Waiter { + /// Update caller-owned policy facts without replacing waiter state. + fn deref_mut(&mut self) -> &mut W { + &mut self.policy + } + } + + impl Waiter { + /// Whether an unissued acquisition opportunity remains usable. + fn eligible(&self) -> bool { + self.acquisition && !self.issued && self.error.is_none() + } + } + + /// Attempt outcome held behind the retained-operation completion fence. + pub enum Outcome { + /// A publication awaiting settlement and result interpretation. + Published(O), + + /// A terminal error awaiting settlement. + Failed(E), + + /// An attempt that permits another eligible caller after settlement. + Retry, + } + + /// Flight lifecycle independent of application-specific result semantics. + pub enum Phase { + /// A leader may acquire and publish a result. + Acquiring, + + /// No retained attempt blocks election of an eligible caller. + RetryPending, + + /// An outcome exists but retained operations may still own resources. + Draining(Outcome), + + /// A fully settled complete result is available. + Complete(C), + + /// A fully settled terminal error is available. + Failed(E), + } + + /// Adapter interpretation of a successfully settled publication. + pub enum Published { + /// A result that satisfies all callers. + Complete(C), + + /// A reusable intermediate result that may require another acquisition. + Partial(P), + } + + /// Worker-local waiter index and transition state driven by an owning adapter. + /// The adapter supplies actual operation idleness; cancellation is not idleness. + pub struct State { + /// Current acquisition or settlement phase. + pub phase: Phase, + + /// Current leader's waiter identifier, retained until settlement. + pub leader: Option, + + /// Attached callers indexed by monotonic waiter identifier. + pub waiters: BTreeMap>, + + /// Deadline index used for bounded expiration work. + pub deadlines: BTreeMap<(Instant, u64), ()>, + + /// Settled intermediate result retained across another acquisition. + pub partial: Option

, + + cursor: Cursor, + + /// State transitions belong to one worker even for thread-safe policy data. + local: std::marker::PhantomData>, + } + + impl Default for State { + /// Start without callers or a retained attempt, ready for initial election. + fn default() -> Self { + Self { + phase: Phase::RetryPending, + leader: None, + waiters: BTreeMap::new(), + deadlines: BTreeMap::new(), + partial: None, + cursor: Cursor::default(), + local: std::marker::PhantomData, + } + } + } + + impl State { + /// Attach a caller using an identifier allocated by the owning table. + pub fn register(&mut self, id: u64, policy: W, acquisition: bool, complete: bool) { + self.deadlines.insert((policy.deadline(), id), ()); + self.waiters.insert( + id, + Waiter { + policy, + acquisition, + complete, + issued: false, + error: None, + waker: None, + }, + ); + } + + /// Remove a caller and its deadline without declaring operation completion. + pub fn detach(&mut self, id: u64) { + if let Some(waiter) = self.waiters.remove(&id) { + self.deadlines.remove(&(waiter.deadline(), id)); + } + } + + /// Validate attachment and incarnation while allowing acquisition retries. + pub fn validate_registration( + &self, + id: u64, + attached: bool, + registered: &Identity, + current: &Identity, + ) -> Result<(), Stale> { + if !attached + || !registered.same_registration(current) + || !self.waiters.contains_key(&id) + { + return Err(Stale); + } + Ok(()) + } + + /// Mark only the supplying caller; refresh decides whether to revoke a leader. + pub fn cancel(&mut self, id: u64, error: W::Error) { + if let Some(waiter) = self.waiters.get_mut(&id) { + waiter.error = Some(error); + } + } + + /// Caller validates its leader capability and publication before this step. + /// Notification timing remains explicit: publication wakes only on settlement, + /// while rejection/revocation also notifies before retained operations finish. + pub fn begin_completion(&mut self, outcome: Outcome) { + self.phase = Phase::Draining(outcome); + } + + /// Enqueue all parked caller notifications without invoking their wakers. + pub fn notify(&mut self, wakes: &mut Vec) { + for waiter in self.waiters.values_mut() { + if let Some(waker) = waiter.waker.take() { + wakes.push(waker); + } + } + } + + /// Check one caller in round-robin order without scanning the waiter map. + pub fn sweep_waiter(&mut self, wakes: &mut Vec) { + if let Some((id, _)) = self.cursor.next(&self.waiters) { + self.refresh_waiter(id, wakes); + } + } + + /// Validate policy/budget first in the adapter. Only an eligible waiter can + /// receive a new generation; retained operations must have settled first. + pub fn elect( + &mut self, + id: u64, + identity: &mut Identity, + limit: u64, + ) -> Result { + if !matches!(self.phase, Phase::RetryPending) { + return Ok(false); + } + let Some(waiter) = self.waiters.get_mut(&id).filter(|waiter| waiter.eligible()) + else { + return Ok(false); + }; + identity.advance(limit)?; + self.phase = Phase::Acquiring; + self.leader = Some(id); + waiter.issued = true; + Ok(true) + } + + /// Validate the active leader after the adapter checks its generation fence. + pub fn validate_leader(&self, id: u64, active: bool) -> Result<(), Stale> { + if !active || !matches!(self.phase, Phase::Acquiring) || self.leader != Some(id) { + return Err(Stale); + } + Ok(()) + } + + /// Expose an outcome only after the adapter confirms all operations completed. + pub fn settle( + &mut self, + idle: bool, + unavailable: W::Error, + split: fn(O) -> Published, + wakes: &mut Vec, + ) { + if !idle || !matches!(self.phase, Phase::Draining(_)) { + return; + } + let Phase::Draining(outcome) = + std::mem::replace(&mut self.phase, Phase::RetryPending) + else { + unreachable!() + }; + self.leader = None; + self.phase = match outcome { + Outcome::Published(value) => match split(value) { + Published::Complete(value) => { + self.partial = None; + Phase::Complete(value) + } + Published::Partial(value) => { + self.partial = Some(value); + for waiter in self.waiters.values_mut() { + if waiter.complete { + waiter.issued = false; + } + } + Phase::RetryPending + } + }, + Outcome::Failed(error) => Phase::Failed(error), + Outcome::Retry if self.waiters.values().any(Waiter::eligible) => { + Phase::RetryPending + } + Outcome::Retry => Phase::Failed(unavailable), + }; + self.notify(wakes); + } + + /// Revoke acquisition immediately, retaining its outcome behind completion. + pub fn revoke(&mut self, error: W::Error, wakes: &mut Vec) { + if let Some(waiter) = self.leader.and_then(|id| self.waiters.get_mut(&id)) { + waiter.error = Some(error); + } + self.phase = Phase::Draining(Outcome::Retry); + self.notify(wakes); + } + + /// Check the leader and up to 64 expired callers, then settle eligible outcomes. + pub fn refresh( + &mut self, + idle: bool, + unavailable: W::Error, + canceled: W::Error, + now: impl Fn() -> Instant, + split: fn(O) -> Published, + wakes: &mut Vec, + ) { + if let Some(id) = self.leader { + self.refresh_waiter(id, wakes); + } + self.refresh_expired(&now, wakes); + if matches!(self.phase, Phase::Acquiring) { + let leader = self.leader.and_then(|id| self.waiters.get(&id)); + if leader.is_none_or(|waiter| waiter.error.is_some()) { + let error = leader.and_then(|waiter| waiter.error).unwrap_or(canceled); + self.revoke(error, wakes); + } + } + self.settle(idle, unavailable, split, wakes); + if matches!(self.phase, Phase::RetryPending) + && self.partial.is_none() + && !self.waiters.values().any(Waiter::eligible) + { + self.phase = Phase::Draining(Outcome::Failed(unavailable)); + self.settle(idle, unavailable, split, wakes); + } + } + + /// Detach and notify all callers while preserving the operation drain fence. + pub fn stop(&mut self, error: W::Error, wakes: &mut Vec) { + for waiter in self.waiters.values_mut() { + waiter.error = Some(error); + } + self.phase = Phase::Draining(Outcome::Failed(error)); + self.notify(wakes); + self.waiters.clear(); + self.deadlines.clear(); + } + + /// Check one caller and enqueue its wake if its policy has failed. + fn refresh_waiter(&mut self, id: u64, wakes: &mut Vec) { + let Some(waiter) = self.waiters.get_mut(&id) else { + return; + }; + if waiter.error.is_none() { + waiter.error = waiter.check(); + } + if waiter.error.is_some() { + self.deadlines.remove(&(waiter.deadline(), id)); + if let Some(waker) = waiter.waker.take() { + wakes.push(waker); + } + } + } + + /// Consume at most 64 due deadline entries, independently of leader checks. + fn refresh_expired(&mut self, now: &impl Fn() -> Instant, wakes: &mut Vec) { + for _ in 0..64 { + let Some((&(deadline, id), _)) = self.deadlines.first_key_value() else { + break; + }; + if deadline > now() { + break; + } + self.deadlines.remove(&(deadline, id)); + self.refresh_waiter(id, wakes); + } + } + } + + /// Store an owned wake target without calling waker clone or drop callbacks. + /// Clone the input before borrowing the owner. Return the retired waker from + /// `update` and drop it after releasing the borrow. For the same target, keep + /// the stored waker and return the unused input instead. + #[must_use = "drop the retired waker outside the owner borrow"] + pub fn store_waker(slot: &mut Option, waker: Waker) -> Option { + if slot.as_ref().is_some_and(|old| old.will_wake(&waker)) { + Some(waker) + } else { + slot.replace(waker) + } + } + + /// Pure transition tests for cancellation, publication, and bounded expiry. + #[cfg(test)] + mod tests { + use super::*; + use std::cell::Cell; + use std::rc::Rc; + use std::sync::Arc; + use std::sync::atomic::{AtomicUsize, Ordering}; + use std::task::Wake; + use std::time::Duration; + + /// Mutable policy failure used to drive deterministic transitions. + struct Policy { + due: Instant, + + error: Rc>>, + } + + impl WaiterPolicy for Policy { + type Error = u8; + + /// Read the failure injected by the test. + fn check(&self) -> Option { + self.error.get() + } + + /// Return the fixed test deadline. + fn deadline(&self) -> Instant { + self.due + } + } + + /// Numeric results keep these tests independent of application semantics. + type Core = State, Policy>; + + /// Publications already carry their complete or partial interpretation. + fn split(value: Published) -> Published { + value + } + + /// Attach a caller and return its externally mutable policy failure. + fn register( + core: &mut Core, + id: u64, + acquisition: bool, + complete: bool, + ) -> Rc>> { + let error = Rc::new(Cell::new(None)); + core.register( + id, + Policy { + due: Instant::now() + Duration::from_secs(60), + error: error.clone(), + }, + acquisition, + complete, + ); + error + } + + /// Construct the initial identity for an isolated transition scenario. + fn identity() -> Identity { + Identity { + owner: Rc::new(()), + incarnation: 1, + generation: 0, + } + } + + /// Refresh with fixed unavailable and cancellation error values. + fn refresh(core: &mut Core, idle: bool, wakes: &mut Vec) { + core.refresh(idle, 9, 8, Instant::now, split, wakes); + } + + /// Leader cancellation or detach cannot bypass retained-operation drainage. + #[test] + fn cancel_or_detach_leader_drains_before_re_election() { + for detach in [false, true] { + let mut core = Core::default(); + let error = register(&mut core, 1, true, true); + register(&mut core, 2, true, true); + register(&mut core, 3, false, false); + let mut identity = identity(); + assert_eq!(core.elect(3, &mut identity, 4), Ok(false)); + assert_eq!(core.elect(1, &mut identity, 4), Ok(true)); + let old = identity.clone(); + assert_eq!(core.validate_registration(1, true, &old, &identity), Ok(())); + assert_eq!(core.elect(2, &mut identity, 4), Ok(false)); + if detach { + core.detach(1); + } else { + error.set(Some(7)); + } + let mut wakes = Vec::new(); + refresh(&mut core, false, &mut wakes); + assert!(matches!(core.phase, Phase::Draining(Outcome::Retry))); + assert_eq!(core.validate_leader(1, true), Err(Stale)); + assert_eq!(core.elect(2, &mut identity, 4), Ok(false)); + refresh(&mut core, true, &mut wakes); + assert_eq!(core.elect(2, &mut identity, 4), Ok(true)); + assert_eq!(old.validate(&identity), Err(Stale)); + assert_eq!(core.validate_registration(2, true, &old, &identity), Ok(())); + assert_eq!( + core.validate_registration(2, false, &old, &identity), + Err(Stale) + ); + assert_eq!( + core.validate_registration(4, true, &old, &identity), + Err(Stale) + ); + assert_eq!(core.validate_leader(2, true), Ok(())); + assert_eq!(core.validate_leader(2, false), Err(Stale)); + } + } + + /// Partial results renew only complete-result callers after the fence clears. + #[test] + fn partial_then_complete_and_terminal_failure_are_fenced() { + let mut core = Core::default(); + register(&mut core, 1, true, true); + register(&mut core, 2, true, false); + let mut identity = identity(); + assert_eq!(core.elect(1, &mut identity, 3), Ok(true)); + core.waiters.get_mut(&2).unwrap().issued = true; + let mut wakes = Vec::new(); + core.begin_completion(Outcome::Published(Published::Partial(17))); + core.settle(false, 9, split, &mut wakes); + assert!(core.partial.is_none()); + core.settle(true, 9, split, &mut wakes); + assert_eq!(core.partial, Some(17)); + assert!(!core.waiters[&1].issued); + assert!(core.waiters[&2].issued); + assert_eq!(core.elect(1, &mut identity, 3), Ok(true)); + core.begin_completion(Outcome::Published(Published::Complete(18))); + core.settle(true, 9, split, &mut wakes); + assert!(matches!(core.phase, Phase::Complete(18))); + assert!(core.partial.is_none()); + assert_eq!(core.elect(1, &mut identity, 3), Ok(false)); + core.phase = Phase::Draining(Outcome::Failed(6)); + core.settle(false, 9, split, &mut wakes); + assert!(matches!(core.phase, Phase::Draining(Outcome::Failed(6)))); + core.settle(true, 9, split, &mut wakes); + assert!(matches!(core.phase, Phase::Failed(6))); + } + + /// Observers cannot acquire and exhausted generations do not issue a caller. + #[test] + fn copy_only_cannot_revive_retry_and_generations_are_bounded() { + let mut core = Core::default(); + register(&mut core, 1, true, true); + register(&mut core, 2, false, false); + let mut identity = identity(); + assert_eq!(core.elect(1, &mut identity, 1), Ok(true)); + core.revoke(7, &mut Vec::new()); + core.settle(true, 9, split, &mut Vec::new()); + assert!(matches!(core.phase, Phase::Failed(9))); + core.phase = Phase::RetryPending; + register(&mut core, 3, true, true); + assert_eq!(core.elect(3, &mut identity, 1), Err(Exhausted)); + assert!(!core.waiters[&3].issued); + assert_eq!(identity.generation, 1); + } + + /// Counts wake delivery without changing transition state. + #[derive(Default)] + struct Count(AtomicUsize); + + impl Wake for Count { + /// Record one delivered notification. + fn wake(self: Arc) { + self.0.fetch_add(1, Ordering::Relaxed); + } + } + + /// Expiry work is bounded to 64 callers and only their latest wakes fire. + #[test] + fn deadlines_have_exact_quantum_and_latest_wakes_are_taken_once() { + let mut core = Core::default(); + let now = Instant::now(); + let old = Arc::new(Count::default()); + let latest = Arc::new(Count::default()); + for id in 0..65 { + core.register( + id, + Policy { + due: now, + error: Rc::new(Cell::new(Some(3))), + }, + true, + true, + ); + let old = Waker::from(old.clone()); + let latest = Waker::from(latest.clone()); + let retired = { + let slot = &mut core.waiters.get_mut(&id).unwrap().waker; + (store_waker(slot, old), store_waker(slot, latest)) + }; + drop(retired); + } + let mut wakes = Vec::new(); + core.refresh(true, 9, 8, || now, split, &mut wakes); + assert_eq!(core.deadlines.len(), 1); + assert_eq!(wakes.len(), 64); + assert!(core.waiters[&64].error.is_none()); + core.refresh(true, 9, 8, || now, split, &mut wakes); + assert!(core.deadlines.is_empty()); + assert_eq!(wakes.len(), 65); + core.notify(&mut wakes); + assert_eq!(wakes.len(), 65); + for wake in wakes { + wake.wake(); + } + assert_eq!(old.0.load(Ordering::Relaxed), 0); + assert_eq!(latest.0.load(Ordering::Relaxed), 65); + } + + /// Shutdown clears caller indexes but retains a pending completion outcome. + #[test] + fn detach_cleans_deadline_and_stop_preserves_pending_completion() { + let mut core = Core::default(); + register(&mut core, 1, true, true); + register(&mut core, 2, true, true); + core.detach(1); + assert_eq!(core.deadlines.len(), 1); + let mut identity = identity(); + assert_eq!(core.elect(2, &mut identity, 3), Ok(true)); + core.cancel(2, 6); + assert_eq!(core.waiters[&2].error, Some(6)); + let mut wakes = Vec::new(); + core.stop(8, &mut wakes); + core.settle(false, 9, split, &mut wakes); + assert!(matches!(core.phase, Phase::Draining(Outcome::Failed(8)))); + assert!(core.waiters.is_empty()); + assert!(core.deadlines.is_empty()); + core.settle(true, 9, split, &mut wakes); + assert!(matches!(core.phase, Phase::Failed(8))); + } + } + } + + /// Membership, sweep fairness, and retained-resource fence contracts. + #[cfg(test)] + mod tests { + use super::*; + use std::cell::{Cell, RefCell}; + + /// Count a retained resource until its destructor actually runs. + struct Resource(Rc>); + + impl Drop for Resource { + /// Record actual resource destruction rather than cancellation. + fn drop(&mut self) { + self.0.set(self.0.get() - 1); + } + } + + /// Minimal application entry with independently tracked waiters and resources. + struct TestEntry { + identity: Identity, + + waiters: usize, + + canceled: bool, + + refreshed: usize, + + operations: Operations, + } + + impl Entry for TestEntry { + /// Count visits and detach callers when cancellation is observed. + fn refresh(&mut self, _: &mut Vec) { + self.refreshed += 1; + if self.canceled { + self.waiters = 0; + } + } + + /// Require both caller detachment and real operation completion. + fn quiescent(&self) -> bool { + self.waiters == 0 && self.operations.is_empty() + } + } + + /// Integer keys isolate table ownership from application key semantics. + type TestTable = Table; + + /// Admit an entry with one waiter and return its initial identity. + fn insert(table: &mut TestTable, owner: &Rc<()>, key: u32) -> Identity { + let identity = table.identity(owner.clone()).unwrap(); + assert!( + table + .insert( + key, + TestEntry { + identity: identity.clone(), + waiters: 1, + canceled: false, + refreshed: 0, + operations: Operations::default(), + }, + ) + .is_ok() + ); + identity + } + + /// Cancellation cannot release resources or bypass the completion tombstone. + #[test] + fn canceled_entry_keeps_resources_until_two_phase_completion() { + let owner = Rc::new(()); + let mut table = TestTable::default(); + insert(&mut table, &owner, 1); + insert(&mut table, &owner, 2); + let live = Rc::new(Cell::new(1)); + let entry = table.get_mut(&1).unwrap(); + entry.operations.insert(1, Resource(live.clone())); + entry.canceled = true; + let mut wakes = Vec::new(); + drop(table.sweep(1, &mut wakes)); + assert_eq!(table.get(&1).unwrap().waiters, 0); + assert_eq!(table.get(&2).unwrap().waiters, 1); + assert_eq!(live.get(), 1, "cancellation is not completion"); + let resources = table.get_mut(&1).unwrap().operations.take(1).unwrap(); + assert!( + table.remove_quiescent(&1).is_none(), + "occupied during resource drop" + ); + drop(resources); + assert_eq!(live.get(), 0); + assert!( + table.remove_quiescent(&1).is_none(), + "completion must clear the tombstone" + ); + table.get_mut(&1).unwrap().operations.complete(1).unwrap(); + assert!(table.remove_quiescent(&1).is_some()); + assert_eq!(table.len(), 1); + } + + /// Occupied entries retain live resources until explicit completion. + #[test] + fn occupied_live_entry_rejects_insertion() { + occupied_entry_rejects_insertion(false, false); + } + + /// Taking resources does not let insertion bypass the completion tombstone. + #[test] + fn occupied_tombstone_rejects_insertion() { + occupied_entry_rejects_insertion(true, false); + } + + /// Even quiescent entries must be explicitly removed before replacement. + #[test] + fn occupied_quiescent_entry_rejects_insertion() { + occupied_entry_rejects_insertion(true, true); + } + + /// Rejection preserves identity, resource ownership, and sweep membership. + fn occupied_entry_rejects_insertion(take: bool, complete: bool) { + let owner = Rc::new(()); + let mut table = TestTable::default(); + let original = insert(&mut table, &owner, 1); + let live = Rc::new(Cell::new(1)); + let entry = table.get_mut(&1).unwrap(); + entry.waiters = 0; + entry.operations.insert(1, Resource(live.clone())); + if take { + drop(entry.operations.take(1).unwrap()); + } + if complete { + entry.operations.complete(1).unwrap(); + } + let sweep_id = table.entries[&1].sweep_id; + let next_sweep_id = table.next_sweep_id; + let rejected_live = Rc::new(Cell::new(1)); + let mut operations = Operations::default(); + operations.insert(2, Resource(rejected_live.clone())); + let replacement = table.identity(owner).unwrap(); + let rejected = table.insert( + 1, + TestEntry { + identity: replacement, + waiters: 1, + canceled: false, + refreshed: 0, + operations, + }, + ); + assert!(rejected.is_err()); + assert!(original.same_registration(&table.get(&1).unwrap().identity)); + assert_eq!(live.get(), usize::from(!take)); + assert_eq!(rejected_live.get(), 1); + assert_eq!(table.len(), 1); + assert_eq!(table.entries[&1].sweep_id, sweep_id); + assert_eq!(table.next_sweep_id, next_sweep_id); + assert_eq!(table.sweep.len(), 1); + assert_eq!(table.sweep[&sweep_id], 1); + drop(rejected); + assert_eq!(rejected_live.get(), 0); + let removed = table.sweep(1, &mut Vec::new()); + assert_eq!(removed.len(), usize::from(complete)); + if !complete { + let entry = table.get_mut(&1).unwrap(); + assert_eq!(entry.refreshed, 1); + assert_eq!(entry.operations.len(), 1); + if !take { + drop(entry.operations.take(1).unwrap()); + } + entry.operations.complete(1).unwrap(); + assert!(table.remove_quiescent(&1).is_some()); + } + assert!(table.is_empty()); + assert!(table.sweep.is_empty()); + } + + /// Request handle that detaches its waiter without completing operations. + struct Waiter { + table: Rc>, + + key: u32, + } + + impl Drop for Waiter { + /// Detach this request and remove its entry only if truly quiescent. + fn drop(&mut self) { + drop(update(&self.table, |table, _| { + table.get_mut(&self.key).unwrap().waiters -= 1; + table.remove_quiescent(&self.key) + })); + } + } + + /// Lost request and identity handles do not own operation resource release. + #[test] + fn dropped_waiter_and_completion_token_do_not_release_owned_operations() { + let owner = Rc::new(()); + let table = Rc::new(RefCell::new(TestTable::default())); + let token = insert(&mut table.borrow_mut(), &owner, 1); + let live = Rc::new(Cell::new(1)); + table + .borrow_mut() + .get_mut(&1) + .unwrap() + .operations + .insert(1, Resource(live.clone())); + drop(Waiter { + table: table.clone(), + key: 1, + }); + drop(token); + drop(update(&table, |table, wakes| table.sweep(100, wakes))); + assert_eq!(table.borrow().len(), 1); + assert_eq!(live.get(), 1); + let resources = table + .borrow_mut() + .get_mut(&1) + .unwrap() + .operations + .take(1) + .unwrap(); + drop(resources); + table + .borrow_mut() + .get_mut(&1) + .unwrap() + .operations + .complete(1) + .unwrap(); + drop(update(&table, |table, wakes| table.sweep(1, wakes))); + assert!(table.borrow().is_empty()); + assert_eq!(live.get(), 0); + } + + /// Owner, incarnation, generation, and resource-take fences are independent. + #[test] + fn reelection_and_recreation_fence_stale_completions() { + let owner = Rc::new(()); + let mut table = TestTable::default(); + insert(&mut table, &owner, 1); + let entry = table.get_mut(&1).unwrap(); + entry.identity.advance(2).unwrap(); + let old = entry.identity.clone(); + let live = Rc::new(Cell::new(1)); + entry.operations.insert(1, Resource(live.clone())); + assert_eq!( + entry.operations.complete(1), + Err(Stale), + "cannot skip resource release" + ); + let resources = entry.operations.take(1).unwrap(); + assert!(matches!(entry.operations.take(1), Err(Stale))); + drop(resources); + entry.operations.complete(1).unwrap(); + assert_eq!(entry.operations.complete(1), Err(Stale)); + entry.identity.advance(2).unwrap(); + assert_eq!(old.validate(&entry.identity), Err(Stale)); + assert!( + old.same_registration(&entry.identity), + "waiter survives retry" + ); + assert_eq!(entry.identity.advance(2), Err(Exhausted)); + assert_eq!(entry.identity.generation, 2); + let mut wrong_owner = old.clone(); + wrong_owner.owner = Rc::new(()); + assert_eq!(wrong_owner.validate(&old), Err(Stale)); + entry.waiters = 0; + assert!(table.remove_quiescent(&1).is_some()); + let new = insert(&mut table, &owner, 1); + assert!(!old.same_registration(&new)); + assert_eq!(old.validate(&new), Err(Stale)); + assert_eq!(live.get(), 0); + } + + /// Each budget unit visits one entry and final removal notifies drain once. + #[test] + fn sweeps_are_budgeted_fair_and_remove_index_entries() { + let owner = Rc::new(()); + let mut table = TestTable::default(); + for key in 1..=3 { + insert(&mut table, &owner, key); + } + drop(table.sweep(0, &mut Vec::new())); + assert!(table.values().all(|entry| entry.refreshed == 0)); + for key in 1..=3 { + drop(table.sweep(1, &mut Vec::new())); + assert_eq!(table.get(&key).unwrap().refreshed, 1); + } + table.get_mut(&2).unwrap().canceled = true; + drop(table.sweep(99, &mut Vec::new())); + assert!(!table.contains_key(&2)); + assert_eq!(table.sweep.len(), 2); + assert!(table.values().all(|entry| entry.refreshed == 2)); + table.drain_waker = Some(Waker::noop().clone()); + table.stop(&mut Vec::new(), |entry, _| { + entry.canceled = true; + }); + let mut wakes = Vec::new(); + drop(table.sweep(99, &mut wakes)); + assert!(table.is_empty()); + assert!(table.sweep.is_empty()); + assert_eq!(wakes.len(), 1); + drop(table.sweep(99, &mut wakes)); + assert_eq!(wakes.len(), 1, "drain wake is taken once"); + } + + /// Cursor wrap is safe after removal while externally visible IDs never wrap. + #[test] + fn cursor_wraps_after_removal_and_counters_never_wrap() { + let mut entries = BTreeMap::from([(1, ()), (2, ()), (3, ())]); + let mut cursor = Cursor::default(); + assert_eq!(cursor.next(&entries).map(|(id, _)| id), Some(1)); + entries.remove(&1); + assert_eq!(cursor.next(&entries).map(|(id, _)| id), Some(2)); + assert_eq!(cursor.next(&entries).map(|(id, _)| id), Some(3)); + assert_eq!(cursor.next(&entries).map(|(id, _)| id), Some(2)); + entries.clear(); + assert!(cursor.next(&entries).is_none()); + let mut counter = Counter(u64::MAX - 1); + assert_eq!(counter.next_id(), Ok(u64::MAX)); + assert_eq!(counter.next_id(), Err(Exhausted)); + assert_eq!(counter.next_id(), Err(Exhausted)); + let mut table = TestTable { + incarnation: counter, + ..TestTable::default() + }; + assert!(matches!(table.identity(Rc::new(())), Err(Exhausted))); + assert!(table.is_empty()); + } + + /// Replacing and removing entries updates only their table-owned sweep slots. + #[test] + fn replacement_and_controlled_access_keep_sweep_membership_synchronized() { + let owner = Rc::new(()); + let mut table = TestTable::default(); + assert!(table.get(&1).is_none()); + assert!(table.get_mut(&1).is_none()); + assert!(table.remove_quiescent(&1).is_none()); + insert(&mut table, &owner, 1); + let original = table.entries[&1].sweep_id; + table.get_mut(&1).unwrap().waiters = 0; + assert!(table.remove_quiescent(&1).is_some()); + insert(&mut table, &owner, 1); + let replacement = table.entries[&1].sweep_id; + insert(&mut table, &owner, 2); + assert_eq!(table.len(), 2); + assert_eq!(table.values().count(), 2); + assert!(!table.sweep.contains_key(&original)); + assert!(table.sweep.contains_key(&replacement)); + drop(table.sweep(2, &mut Vec::new())); + assert!(table.values().all(|entry| entry.refreshed == 1)); + table.get_mut(&1).unwrap().waiters = 0; + assert!(table.remove_quiescent(&1).is_some()); + assert!(!table.contains_key(&1)); + assert_eq!(table.sweep.len(), table.len()); + drop(table.sweep(1, &mut Vec::new())); + assert_eq!(table.get(&2).unwrap().refreshed, 2); + } + + /// Mutable application identities cannot corrupt another entry's membership. + #[test] + fn mutable_incarnation_cannot_remove_another_entries_sweep_slot() { + let owner = Rc::new(()); + let mut table = TestTable::default(); + insert(&mut table, &owner, 1); + let other = insert(&mut table, &owner, 2); + table.get_mut(&1).unwrap().identity.incarnation = other.incarnation; + table.get_mut(&1).unwrap().waiters = 0; + assert!(table.remove_quiescent(&1).is_some()); + assert_eq!(table.sweep.len(), 1); + drop(table.sweep(1, &mut Vec::new())); + assert_eq!(table.get(&2).unwrap().refreshed, 1); + table.get_mut(&2).unwrap().waiters = 0; + drop(table.sweep(1, &mut Vec::new())); + assert!(table.is_empty()); + assert!(table.sweep.is_empty()); + } + + /// Shutdown hooks may mutate identities without invalidating the sweep index. + #[test] + fn stop_identity_mutation_preserves_replacement_and_removal_membership() { + let owner = Rc::new(()); + let mut table = TestTable::default(); + for key in 1..=3 { + insert(&mut table, &owner, key); + } + table.stop(&mut Vec::new(), |entry, _| { + entry.identity.incarnation = u64::MAX; + entry.canceled = true; + }); + table.get_mut(&2).unwrap().waiters = 0; + assert!(table.remove_quiescent(&2).is_some()); + insert(&mut table, &owner, 2); + assert_eq!(table.sweep.len(), 3); + drop(table.sweep(3, &mut Vec::new())); + assert_eq!(table.len(), 1); + assert_eq!(table.sweep.len(), 1); + assert_eq!(table.get(&2).unwrap().refreshed, 1); + table.get_mut(&2).unwrap().waiters = 0; + assert!(table.remove_quiescent(&2).is_some()); + assert!(table.sweep.is_empty()); + } + + /// Duplicate application incarnations remain independent table memberships. + #[test] + fn duplicate_entry_incarnations_have_independent_bounded_sweeps() { + let owner = Rc::new(()); + let mut table = TestTable::default(); + let identity = table.identity(owner).unwrap(); + for key in 1..=3 { + assert!( + table + .insert( + key, + TestEntry { + identity: identity.clone(), + waiters: 1, + canceled: false, + refreshed: 0, + operations: Operations::default(), + }, + ) + .is_ok() + ); + } + assert_eq!(table.sweep.len(), 3); + drop(table.sweep(0, &mut Vec::new())); + assert!(table.values().all(|entry| entry.refreshed == 0)); + for key in 1..=3 { + drop(table.sweep(1, &mut Vec::new())); + assert_eq!(table.get(&key).unwrap().refreshed, 1); + assert_eq!( + table.values().map(|entry| entry.refreshed).sum::(), + key as usize + ); + } + table.get_mut(&2).unwrap().waiters = 0; + assert!(table.remove_quiescent(&2).is_some()); + drop(table.sweep(99, &mut Vec::new())); + assert_eq!(table.sweep.len(), 2); + assert!(table.values().all(|entry| entry.refreshed == 2)); + } + + /// Internal sweep IDs wrap safely without consuming monotonic incarnation IDs. + #[test] + fn sweep_ids_wrap_skip_occupied_slots_and_do_not_consume_incarnations() { + let owner = Rc::new(()); + let mut table = TestTable::default(); + let first = insert(&mut table, &owner, 1); + table.next_sweep_id = u64::MAX; + let second = insert(&mut table, &owner, 2); + assert_eq!(table.entries[&2].sweep_id, 0); + let third = insert(&mut table, &owner, 3); + assert_eq!(table.entries[&3].sweep_id, 2, "skip occupied ID 1"); + assert_eq!( + (first.incarnation, second.incarnation, third.incarnation), + (1, 2, 3) + ); + drop(table.sweep(3, &mut Vec::new())); + assert!(table.values().all(|entry| entry.refreshed == 1)); + table.incarnation = Counter(u64::MAX); + assert!(table.identity(owner).is_err()); + assert!( + table + .insert( + 4, + TestEntry { + identity: first, + waiters: 1, + canceled: false, + refreshed: 0, + operations: Operations::default(), + }, + ) + .is_ok() + ); + assert_eq!( + table.len(), + 4, + "insert does not require an incarnation allocation" + ); + table.stop(&mut Vec::new(), |entry, _| entry.canceled = true); + drop(table.sweep(4, &mut Vec::new())); + assert!(table.is_empty()); + assert!(table.sweep.is_empty()); + } + } +} + +/// Election, capacity, and final-handle cleanup contracts. +#[cfg(test)] +mod tests { + use super::*; + use std::sync::Arc; + use std::sync::atomic::{AtomicUsize, Ordering}; + use std::task::Wake; + + /// Counts delivered notifications without running an executor. + #[derive(Default)] + struct Counter(AtomicUsize); + + impl Wake for Counter { + /// Record one notification. + fn wake(self: Arc) { + self.0.fetch_add(1, Ordering::Relaxed); + } + } + + /// Small keyed table used by the cohort scenarios. + type TestTable = Table<&'static str, Result>; + + /// Construct a table with independently configurable waiter and attempt bounds. + fn table(waiters: usize, attempts: usize) -> Rc { + Rc::new(Table::new( + Limits { + waiters_per_cohort: waiters, + attempts_per_cohort: attempts, + }, + Err("exhausted"), + )) + } + + thread_local! { + static ON_KEY_CLONE: RefCell>> = RefCell::new(None); + } + + /// A key whose next clone can call back into its table. + #[derive(Debug, Eq, PartialEq, Ord, PartialOrd, Hash)] + struct CloneKey(u32); + + impl Clone for CloneKey { + /// Run the one-shot callback without retaining the callback slot borrow. + fn clone(&self) -> Self { + let callback = ON_KEY_CLONE.with(|slot| slot.borrow_mut().take()); + if let Some(callback) = callback { + callback(); + } + Self(self.0) + } + } + + /// A clone callback can admit the same key before the outer join resumes. + #[test] + fn key_clone_reentry_joins_current_cohort() { + let table = Rc::new(Table::<_, u32>::new( + Limits { + waiters_per_cohort: 2, + attempts_per_cohort: 1, + }, + 0, + )); + let admitted = Rc::new(RefCell::new(None)); + let nested = table.clone(); + let retained = admitted.clone(); + ON_KEY_CLONE.with(|slot| { + *slot.borrow_mut() = Some(Box::new(move || { + assert_eq!(nested.active_count(), 0); + *retained.borrow_mut() = Some(nested.join(CloneKey(1), 1).unwrap()); + })); + }); + let outer = table.join(CloneKey(1), 1).unwrap(); + let inner = admitted.borrow_mut().take().unwrap(); + assert_eq!(table.active_count(), 1); + assert_eq!(table.registration_count(), 2); + assert_eq!(inner.event(Waker::noop()), Poll::Ready(Event::Lead)); + assert!(outer.event(Waker::noop()).is_pending()); + inner.finish(7); + assert_eq!(outer.event(Waker::noop()), Poll::Ready(Event::Complete(7))); + assert_eq!(table.active_count(), 0); + drop((inner, outer)); + assert_eq!(table.registration_count(), 0); + } + + /// Clone callbacks may fill key, waiter, or completed-reader capacity. + #[test] + fn key_clone_reentry_rechecks_all_bounds() { + for (capacity, waiters, nested_key, complete) in + [(1, 2, 2, false), (2, 1, 1, false), (1, 2, 2, true)] + { + let table = Rc::new(Table::<_, u32>::new( + Limits { + waiters_per_cohort: waiters, + attempts_per_cohort: 1, + }, + 0, + )); + let admitted = Rc::new(RefCell::new(Vec::new())); + let nested = table.clone(); + let retained = admitted.clone(); + ON_KEY_CLONE.with(|slot| { + *slot.borrow_mut() = Some(Box::new(move || { + let count = if complete { capacity * waiters } else { 1 }; + for _ in 0..count { + let registration = nested.join(CloneKey(nested_key), capacity).unwrap(); + if complete { + registration.finish(7); + } + retained.borrow_mut().push(registration); + } + })); + }); + assert!(matches!( + table.join(CloneKey(1), capacity), + Err(CapacityError) + )); + assert_eq!(table.active_count(), usize::from(!complete)); + assert_eq!(table.registration_count(), admitted.borrow().len()); + admitted.borrow_mut().clear(); + assert_eq!(table.active_count(), 0); + assert_eq!(table.registration_count(), 0); + let next = table.join(CloneKey(1), capacity).unwrap(); + assert_eq!(table.registration_count(), 1); + drop(next); + assert_eq!(table.active_count(), 0); + assert_eq!(table.registration_count(), 0); + } + } + + /// Shared key cloning can inspect the table and start unrelated work. + #[test] + fn key_clone_reentry_shared_start_preserves_both_completions() { + use futures::executor::block_on; + + let table = Rc::new(shared::Table::default()); + let admitted = Rc::new(RefCell::new(None)); + let nested = table.clone(); + let retained = admitted.clone(); + ON_KEY_CLONE.with(|slot| { + *slot.borrow_mut() = Some(Box::new(move || { + assert!(nested.is_empty()); + assert!(nested.get(&CloneKey(1)).is_none()); + *retained.borrow_mut() = Some(nested.start(CloneKey(2), 0)); + })); + }); + let (outer, completion) = table.start(CloneKey(1), 0); + let (inner, nested_completion) = admitted.borrow_mut().take().unwrap(); + assert_eq!(table.len(), 2); + completion.unwrap().finish(7); + assert_eq!(block_on(outer), 7); + assert_eq!(table.len(), 1); + assert!(table.get(&CloneKey(1)).is_none()); + let follower = table.get(&CloneKey(2)).unwrap(); + nested_completion.unwrap().finish(8); + assert_eq!(block_on(inner), 8); + assert_eq!(block_on(follower), 8); + assert!(table.is_empty()); + } + + #[test] + fn shared_same_key_clone_reentry_joins_nested_completion() { + use futures::FutureExt; + + for result in [Ok(7), Err("failed")] { + let table = Rc::new(shared::Table::default()); + let admitted = Rc::new(RefCell::new(None)); + let nested = table.clone(); + let retained = admitted.clone(); + ON_KEY_CLONE.with(|slot| { + *slot.borrow_mut() = Some(Box::new(move || { + assert!(nested.get(&CloneKey(1)).is_none()); + *retained.borrow_mut() = Some(nested.start(CloneKey(1), Err("nested closed"))); + })); + }); + assert!(table.get(&CloneKey(1)).is_none()); + let (outer, completion) = table.start(CloneKey(1), Err("outer closed")); + let (inner, nested_completion) = admitted.borrow_mut().take().unwrap(); + let follower = table.get(&CloneKey(1)).unwrap(); + assert_eq!(table.len(), 1); + assert!(completion.is_none()); + nested_completion.unwrap().finish(result); + assert_eq!(inner.now_or_never(), Some(result)); + assert_eq!(outer.now_or_never(), Some(result)); + assert_eq!(follower.now_or_never(), Some(result)); + assert!(table.is_empty()); + let (next, next_completion) = table.start(CloneKey(1), Err("next closed")); + drop(completion); + assert_eq!(table.len(), 1); + next_completion.unwrap().finish(Ok(8)); + assert_eq!(next.now_or_never(), Some(Ok(8))); + assert!(table.is_empty()); + } + } + + #[test] + fn shared_same_key_clone_reentry_completed_before_start_keeps_new_owner() { + use futures::FutureExt; + + let table = Rc::new(shared::Table::default()); + let admitted = Rc::new(RefCell::new(None)); + let nested = table.clone(); + let retained = admitted.clone(); + ON_KEY_CLONE.with(|slot| { + *slot.borrow_mut() = Some(Box::new(move || { + let (receive, completion) = nested.start(CloneKey(1), 99); + completion.unwrap().finish(7); + *retained.borrow_mut() = Some(receive); + assert!(nested.is_empty()); + })); + }); + let (outer, completion) = table.start(CloneKey(1), 99); + assert_eq!(table.len(), 1); + let inner = admitted.borrow_mut().take().unwrap(); + assert_eq!(inner.now_or_never(), Some(7)); + let follower = table.get(&CloneKey(1)).unwrap(); + assert!(outer.clone().now_or_never().is_none()); + completion.unwrap().finish(8); + assert_eq!(outer.now_or_never(), Some(8)); + assert_eq!(follower.now_or_never(), Some(8)); + assert!(table.is_empty()); + } + + #[test] + fn shared_same_key_clone_reentry_lost_sender_stays_occupied() { + use futures::FutureExt; + + let table = Rc::new(shared::Table::default()); + let nested = table.clone(); + ON_KEY_CLONE.with(|slot| { + *slot.borrow_mut() = Some(Box::new(move || { + let (receive, completion) = nested.start(CloneKey(1), 7); + drop((receive, completion)); + assert_eq!(nested.len(), 1); + })); + }); + let (outer, completion) = table.start(CloneKey(1), 99); + assert!(completion.is_none()); + assert_eq!(outer.now_or_never(), Some(7)); + assert_eq!(table.len(), 1); + let (late, completion) = table.start(CloneKey(1), 88); + assert!(completion.is_none()); + assert_eq!(late.now_or_never(), Some(7)); + assert_eq!(table.len(), 1); + } + + /// Completion reaches both parked and unpolled readers without deleting replacements. + #[test] + fn broadcasts_success_and_failure_before_or_after_poll() { + for result in [Ok(42), Err("failed")] { + for poll_first in [false, true] { + let table = table(4, 2); + let leader = table.join("key", 1).unwrap(); + let follower = table.join("key", 1).unwrap(); + let count = Arc::new(Counter::default()); + let waker = Waker::from(count.clone()); + assert_eq!(leader.event(Waker::noop()), Poll::Ready(Event::Lead)); + if poll_first { + assert_eq!(follower.event(&waker), Poll::Pending); + } + leader.finish(result); + assert_eq!(count.0.load(Ordering::Relaxed), usize::from(poll_first)); + assert_eq!(follower.event(&waker), Poll::Ready(Event::Complete(result))); + assert_eq!(table.active_count(), 0); + let next = table.join("key", 1).unwrap(); + assert_eq!(next.event(Waker::noop()), Poll::Ready(Event::Lead)); + drop(leader); + drop(follower); + assert_eq!(table.active_count(), 1, "old cohort cannot remove new key"); + } + } + } + + /// The first result survives later finishes and remains separate from replacements. + #[test] + fn first_completion_wins_for_early_and_late_readers() { + for result in [Ok(42), Err("failed")] { + let table = table(5, 2); + let leader = table.join("key", 1).unwrap(); + let early = table.join("key", 1).unwrap(); + let late = table.join("key", 1).unwrap(); + let survivor = leader.clone(); + let count = Arc::new(Counter::default()); + let waker = Waker::from(count.clone()); + assert_eq!(leader.event(Waker::noop()), Poll::Ready(Event::Lead)); + assert_eq!(early.event(&waker), Poll::Pending); + leader.finish(result); + drop(leader); + assert_eq!(count.0.load(Ordering::Relaxed), 1); + assert_eq!(early.event(&waker), Poll::Ready(Event::Complete(result))); + assert_eq!(table.active_count(), 0); + + let next = table.join("key", 1).unwrap(); + assert_eq!(next.event(Waker::noop()), Poll::Ready(Event::Lead)); + survivor.finish(Ok(99)); + early.finish(Err("late failure")); + assert_eq!(count.0.load(Ordering::Relaxed), 1); + assert_eq!(early.event(&waker), Poll::Ready(Event::Complete(result))); + assert_eq!( + late.event(Waker::noop()), + Poll::Ready(Event::Complete(result)) + ); + assert_eq!( + survivor.event(Waker::noop()), + Poll::Ready(Event::Complete(result)) + ); + assert_eq!(table.active_count(), 1); + let next_follower = table.join("key", 1).unwrap(); + assert_eq!(next_follower.event(Waker::noop()), Poll::Pending); + next.finish(Ok(7)); + assert_eq!( + next_follower.event(Waker::noop()), + Poll::Ready(Event::Complete(Ok(7))) + ); + assert_eq!( + late.event(Waker::noop()), + Poll::Ready(Event::Complete(result)) + ); + drop((survivor, early, late, next, next_follower)); + assert_eq!(table.registration_count(), 0); + assert_eq!(table.active_count(), 0); + } + } + + /// Attempt exhaustion is also a terminal result, even for unpolled followers. + #[test] + fn first_completion_from_exhaustion_cannot_be_overwritten() { + let table = table(2, 0); + let first = table.join("key", 1).unwrap(); + let late = table.join("key", 1).unwrap(); + let exhausted = Poll::Ready(Event::Complete(Err("exhausted"))); + assert_eq!(first.event(Waker::noop()), exhausted); + first.finish(Ok(1)); + late.finish(Err("late failure")); + assert_eq!(first.event(Waker::noop()), exhausted); + assert_eq!(late.event(Waker::noop()), exhausted); + assert_eq!(table.active_count(), 0); + } + + /// Detached execution retains one waiter until its final handle is dropped. + #[test] + fn last_handle_drop_releases_leadership_and_registration() { + let table = table(2, 3); + let request = table.join("key", 1).unwrap(); + let follower = table.join("key", 1).unwrap(); + assert_eq!(request.event(Waker::noop()), Poll::Ready(Event::Lead)); + let driver = request.clone(); + assert!(!driver.is_only_handle()); + drop(request); + assert!(driver.is_only_handle()); + assert_eq!(table.registration_count(), 2); + let old = Arc::new(Counter::default()); + let latest = Arc::new(Counter::default()); + assert!(follower.event(&Waker::from(old.clone())).is_pending()); + assert!(follower.event(&Waker::from(latest.clone())).is_pending()); + drop(driver); + assert_eq!(old.0.load(Ordering::Relaxed), 0); + assert_eq!(latest.0.load(Ordering::Relaxed), 1); + assert_eq!(table.registration_count(), 1); + assert_eq!(follower.event(Waker::noop()), Poll::Ready(Event::Lead)); + drop(follower); + assert_eq!(table.registration_count(), 0); + assert_eq!(table.active_count(), 0); + } + + /// Retry and owner loss consume the same bounded election budget. + #[test] + fn retry_and_drop_share_bounded_attempts_and_broadcast_exhaustion() { + let table = table(3, 2); + let first = table.join("key", 1).unwrap(); + let second = table.join("key", 1).unwrap(); + let observer = table.join("key", 1).unwrap(); + assert_eq!(first.event(Waker::noop()), Poll::Ready(Event::Lead)); + assert!(first.event(Waker::noop()).is_pending()); + first.retry(); + assert_eq!(second.event(Waker::noop()), Poll::Ready(Event::Lead)); + drop(second); + let exhausted = Poll::Ready(Event::Complete(Err("exhausted"))); + assert_eq!(observer.event(Waker::noop()), exhausted); + assert_eq!(first.event(Waker::noop()), exhausted); + assert_eq!(table.active_count(), 0); + } + + /// Completed readers still consume global capacity until they detach. + #[test] + fn capacity_bounds_keys_waiters_and_completed_readers() { + let table = table(2, 1); + assert!(matches!(table.join("a", 0), Err(CapacityError))); + let a = table.join("a", 2).unwrap(); + let b = table.join("b", 2).unwrap(); + assert!(matches!(table.join("c", 2), Err(CapacityError))); + let a2 = table.join("a", 2).unwrap(); + assert!(matches!(table.join("a", 2), Err(CapacityError))); + let b2 = table.join("b", 2).unwrap(); + a.finish(Ok(1)); + b.finish(Ok(2)); + assert_eq!(table.active_count(), 0); + assert!(matches!(table.join("a", 2), Err(CapacityError))); + drop(a2); + let next = table.join("a", 2).unwrap(); + drop((a, b, b2, next)); + assert_eq!(table.registration_count(), 0); + assert_eq!(table.active_count(), 0); + } + + /// Zero bounds and identifier exhaustion reject work without leaking charges. + #[test] + fn zero_limits_and_id_exhaustion_never_wrap_or_leak_registrations() { + let no_waiters = table(0, 1); + assert!(matches!(no_waiters.join("key", 1), Err(CapacityError))); + assert_eq!(no_waiters.active_count(), 0); + let zero = table(1, 0); + let waiter = zero.join("key", 1).unwrap(); + assert_eq!( + waiter.event(Waker::noop()), + Poll::Ready(Event::Complete(Err("exhausted"))) + ); + let table = table(2, 1); + let first = table.join("key", usize::MAX).unwrap(); + first.owner.cohort.borrow_mut().next = u64::MAX; + assert!(matches!(table.join("key", usize::MAX), Err(CapacityError))); + assert_eq!(table.registration_count(), 1); + drop(first); + assert_eq!(table.active_count(), 0); + } +} diff --git a/cmd/racer-dataplane/flow/src/lib.rs b/cmd/racer-dataplane/flow/src/lib.rs new file mode 100644 index 000000000..1bf78968e --- /dev/null +++ b/cmd/racer-dataplane/flow/src/lib.rs @@ -0,0 +1,2507 @@ +//! Policy-driven admission, charged storage, and completion-owned flow control. +//! +//! Implement [`Class`] and [`Policy`] to supply resource limits and keyed fairness. +//! [`Quotas`] is a worker-local authority; [`SharedQuotas`] admits unkeyed work +//! across threads. Keep each [`Charge`] with its resource until completion. A +//! completion reservation bypasses stop and fair-share checks, not aggregate or +//! key-record limits. Policy can also allow selected classes after stop. +//! +//! Recycling wipes the full allocation before retaining at most two buffers of +//! at least 1 MiB. Pressure retries admission after applicable reclamation; stop +//! releases retained buffers. Shared handles never retain the recycler. +//! +//! [`ChargedBuffer`] pairs fixed initialized backing with admission. Low-level +//! raw buffer users must themselves retain sufficient charge for every allocation +//! and validate class, key, and provenance. Application authentication, error +//! classification, request deadlines, and cancellation policy stay with callers. +//! +//! Endpoint circuits, adaptive admission, handoffs, and hedge alarms use distinct +//! ownership rules. Pipes retain their charges while idle; socket-retained bytes +//! are outside the pipe capacity budget. The `simulation` feature forwards the +//! runtime's simulated descriptors without changing admission policy. +//! +//! [`coalesce`] provides worker-local keyed cohorts, shared results, and flight +//! lifecycle tracking. Callers retain execution, admission, and result policy; +//! cancellation never substitutes for real operation completion. +#![deny(unsafe_op_in_unsafe_fn)] + +/// Adaptive admission and completion-owned handoffs. +mod admission; + +/// Keyed cohorts and completion-owned flight tracking. +pub mod coalesce; + +/// Charged kernel pipes and worker-local reuse. +mod pipe; + +pub use admission::{ + Adaptive, Admitted, Circuits, Config as AdaptiveConfig, Event as AdaptiveEvent, Handoff, + HandoffAdmission, HedgePermit, Hedges, Observer as AdaptiveObserver, Offer, + Outcome as AdaptiveOutcome, Permit as AdaptivePermit, Probe, +}; +pub use pipe::{MAX_PIPE_BYTES, PipeLease, PipePool, splice_unsupported}; + +use std::{ + cell::RefCell, + collections::{HashMap, VecDeque}, + hash::Hash, + marker::PhantomData, + rc::Rc, + sync::{ + Arc, Mutex, Weak, + atomic::{AtomicBool, AtomicUsize, Ordering}, + }, +}; + +/// Admission and pipe-creation failures, independent of application error policy. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub enum Error { + /// The requested geometry or state transition is invalid. + InvalidInput, + + /// A resource or keyed fairness limit is exhausted. + Overloaded, + + /// Admission has stopped or its shared state is unavailable. + Unavailable, + + /// A kernel pipe operation failed during creation. + Io, +} + +impl std::fmt::Display for Error { + /// Describe the failure without application resource names. + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.write_str(match self { + Self::InvalidInput => "invalid flow-control input", + Self::Overloaded => "flow-control quota exhausted", + Self::Unavailable => "flow control unavailable", + Self::Io => "flow-control I/O failed", + }) + } +} + +impl std::error::Error for Error {} + +/// A flow-control operation's result. +pub type Result = std::result::Result; + +/// A dense, stable index in `0..COUNT`, unique to each resource class. +pub trait Class: Copy + Send + Sync + 'static { + /// Number of distinct resource classes. + const COUNT: usize; + + /// Return this class's unique index, strictly below `COUNT`. + fn index(self) -> usize; +} + +/// Application policy. Charges do not retain this value, so suitability and +/// release notifications are static properties of the class. +pub trait Policy: 'static { + /// Resource classes accounted by this policy. + type Class: Class; + + /// Identity used to divide local fair shares. + type Key: Clone + Eq + Hash + Send + Sync + 'static; + + /// Return the aggregate ceiling for a class. + fn limit(&self, class: Self::Class) -> usize; + + /// Return the minimum keyed share, clipped to the aggregate ceiling. + fn floor(&self, _class: Self::Class) -> usize { + 1 + } + + /// Bound the number of retained key records. + fn max_keys(&self) -> usize; + + /// Whether releasing this class wakes the shared admission waiter. + fn wakes(class: Self::Class) -> bool; + + /// Whether ordinary admission of this class remains allowed after stop. + fn allows_stopped(_class: Self::Class) -> bool { + false + } + + /// Whether this class can account for page allocator backing. + fn covers(class: Self::Class) -> bool; + + /// Observe each failed attempt, including one recovered by reclamation. + fn rejected(&self, rejection: Rejection); +} + +/// Facts observed at a failed admission attempt, before any reclamation retry. +#[derive(Clone, Copy, Debug)] +pub enum Rejection { + /// The bounded key table has no room for a new identity. + Keys { + /// Records still occupying the bounded key table. + used: usize, + + /// Maximum retained records allowed by policy. + limit: usize, + }, + + /// An aggregate or keyed share would be exceeded. + Resource { + class: C, + + used: usize, + + limit: usize, + + requested: usize, + + key_used: Option, + + key_limit: Option, + }, +} + +/// Worker-local keyed admission and recycler authority; never moves across workers. +/// +/// Only shared handles and charges may cross threads: +/// +/// ```compile_fail +/// use flow_control::{Policy, Quotas}; +/// fn move_authority(authority: Quotas

) { +/// fn require_send(_: T) {} +/// require_send(authority); +/// } +/// ``` +/// +/// ```compile_fail +/// use flow_control::{Policy, Quotas}; +/// fn share_authority(authority: &Quotas

) { +/// fn require_sync(_: &T) {} +/// require_sync(authority); +/// } +/// ``` +pub struct Quotas { + policy: P, + + totals: Arc>, + + keys: Keys, + + active_keys: Arc, + + retired_keys: Arc>>, + + stopped: Arc, + + buffers: Arc>, + + local: PhantomData>, +} + +/// Owns a live charge independently of the local quota authority and policy. +pub struct Charge { + class: P::Class, + + amount: usize, + + key: Option, + + totals: Arc>, + + local: Option>>, + + buffers: Weak>, + + stopped: Arc, +} + +impl page_alloc::Charge for Charge

{ + /// Check page-backing suitability without transferring admission ownership. + fn covers(&self, bytes: usize) -> bool { + P::covers(self.class) && self.amount >= bytes + } +} + +/// An unkeyed admission and usage handle. It retains no recycler buffers. +/// It is Send + Sync when the policy is Send + Sync. +pub struct SharedQuotas { + policy: P, + + totals: Arc>, + + stopped: Arc, +} + +impl Clone for SharedQuotas

{ + /// Share aggregate counters and stop state without retaining recycled backing. + fn clone(&self) -> Self { + Self { + policy: self.policy.clone(), + totals: self.totals.clone(), + stopped: self.stopped.clone(), + } + } +} + +impl SharedQuotas

{ + /// Register the single shared admission waiter before checking capacity. + pub fn register(&self, waker: &std::task::Waker) { + self.totals.wake.register(waker); + } + + /// Read current aggregate usage, including retained allocations. + pub fn used(&self, class: P::Class) -> usize { + self.totals.counter(class).used() + } + + /// Return the policy's aggregate ceiling. + pub fn limit(&self, class: P::Class) -> usize { + self.policy.limit(class) + } + + /// Whether the authority has requested admission shutdown. + pub fn is_stopped(&self) -> bool { + self.stopped.load(Ordering::Acquire) + } + + /// Admit a nonzero unkeyed amount without retaining the local recycler. + pub fn reserve(&self, class: P::Class, amount: usize) -> Result> { + if self.is_stopped() && !P::allows_stopped(class) { + return Err(Error::Unavailable); + } + if amount == 0 { + return Err(Error::InvalidInput); + } + let limit = self.limit(class); + self.totals + .counter(class) + .reserve(amount, limit) + .map_err(|_| { + self.policy.rejected(Rejection::Resource { + class, + used: self.used(class), + limit, + requested: amount, + key_used: None, + key_limit: None, + }); + Error::Overloaded + })?; + let charge = Charge { + class, + amount, + key: None, + totals: self.totals.clone(), + local: None, + buffers: Weak::new(), + stopped: self.stopped.clone(), + }; + // Stop may race the policy callback or counter reservation. Dropping the + // charge rolls back usage and applies the usual release wake policy. + if self.is_stopped() && !P::allows_stopped(class) { + return Err(Error::Unavailable); + } + Ok(charge) + } +} + +impl Charge

{ + /// Return the amount still owned by this charge. + pub fn amount(&self) -> usize { + self.amount + } + + /// Return the resource class selected at admission. + pub fn class(&self) -> P::Class { + self.class + } + + /// Return the admitted key, or none for aggregate-only admission. + pub fn key(&self) -> Option<&P::Key> { + self.key.as_ref() + } + + /// Check class and amount; this does not establish authority provenance. + pub fn validate(&self, class: P::Class, amount: usize) -> Result<()> { + if self.class.index() != class.index() || amount > self.amount { + Err(Error::InvalidInput) + } else { + Ok(()) + } + } + + /// Divide an admitted working set without changing its aggregate charge. + pub fn split(&mut self, amount: usize) -> Result { + if amount == 0 || amount >= self.amount { + return Err(Error::InvalidInput); + } + let key = self.key.clone(); + Ok(self.transfer(amount, self.buffers.clone(), key)) + } + + /// Caller must retain at least the capacity of every live backing allocation. + pub fn shrink(&mut self, amount: usize) -> Result<()> { + if amount == 0 || amount > self.amount { + return Err(Error::InvalidInput); + } + self.release_to(amount); + Ok(()) + } + + /// Obtain zeroed backing without transferring or subdividing this charge. + /// + /// This low-level API does not track other allocations made with the same + /// charge. The caller must retain sufficient admission for all live backing, + /// validate its class/key/provenance, and never shrink below live capacity. + pub fn buffer(&self, length: usize) -> Result> { + if length > self.amount { + return Err(Error::InvalidInput); + } + if let Some(pool) = self.buffers.upgrade() { + let recycled = { + let mut pool = pool.lock().unwrap_or_else(|e| e.into_inner()); + pool.iter() + .position(|(bytes, _)| bytes.capacity() == length) + .map(|index| pool.swap_remove(index)) + }; + if let Some((bytes, old)) = recycled { + drop(old); + debug_assert_eq!(bytes.len(), length); + return Ok(bytes); + } + } + let mut bytes = Vec::new(); + bytes + .try_reserve_exact(length) + .map_err(|_| Error::Overloaded)?; + bytes.resize(length, 0); + Ok(bytes) + } + + /// Wipe before every retention check. At most two >=1MiB buffers retain live + /// charges; only the final exclusive payload owner may return an allocation. + pub fn recycle(&mut self, mut bytes: Vec) { + wipe_payload(&mut bytes); + if self.stopped.load(Ordering::Acquire) + || bytes.capacity() < 1024 * 1024 + || bytes.capacity() > self.amount + { + return; + } + let Some(pool) = self.buffers.upgrade() else { + return; + }; + // Application cloning may reenter stop or fill the pool. Clone unlocked, + // then check retention again before transferring any admission. + let key = self.key.clone(); + let mut pool = match pool.try_lock() { + Ok(pool) => pool, + Err(std::sync::TryLockError::WouldBlock) => return, + Err(std::sync::TryLockError::Poisoned(error)) => error.into_inner(), + }; + if self.stopped.load(Ordering::Acquire) || pool.len() >= 2 { + return; + } + // SAFETY: wipe_payload initialized the entire capacity above. + unsafe { bytes.set_len(bytes.capacity()) }; + let charge = self.transfer(self.amount, Weak::new(), key); + pool.push((bytes, charge)); + } + + /// Move admission with a precloned key without touching either usage counter. + fn transfer(&mut self, amount: usize, buffers: Weak>, key: Option) -> Self { + self.amount -= amount; + Self { + class: self.class, + amount, + key, + totals: self.totals.clone(), + local: self.local.clone(), + buffers, + stopped: self.stopped.clone(), + } + } + + /// Return released admission without notifying the shared waiter. + fn release_to(&mut self, amount: usize) { + let released = self.amount - amount; + self.amount = amount; + self.totals.counter(self.class).release(released); + if let Some(local) = &self.local { + local.counter(self.class).release(released); + } + } +} + +impl Drop for Charge

{ + /// Release exactly this charge's remaining amount and apply wake policy. + fn drop(&mut self) { + self.release_to(0); + // Retire the final key owner before a wake callback retries admission. + drop(self.local.take()); + if P::wakes(self.class) { + self.totals.wake.wake(); + } + } +} + +impl Quotas

{ + /// Create a worker-local authority with no admitted work or retained backing. + pub fn new(policy: P) -> Self { + Self { + policy, + totals: Arc::new(Counters::new(P::Class::COUNT)), + keys: RefCell::default(), + active_keys: Arc::new(AtomicUsize::new(0)), + retired_keys: Arc::new(Mutex::new(VecDeque::new())), + stopped: Arc::new(AtomicBool::new(false)), + buffers: Arc::new(Mutex::new(Vec::new())), + local: PhantomData, + } + } + + /// Borrow application policy without transferring local authority. + pub fn policy(&self) -> &P { + &self.policy + } + + /// Create a transferable unkeyed handle without retaining recycler storage. + pub fn shared(&self) -> SharedQuotas

+ where + P: Clone, + { + SharedQuotas { + policy: self.policy.clone(), + totals: self.totals.clone(), + stopped: self.stopped.clone(), + } + } + + /// Read aggregate usage, including idle resources and completion owners. + pub fn used(&self, class: P::Class) -> usize { + self.totals.counter(class).used() + } + + /// Return the aggregate ceiling chosen by policy. + pub fn limit(&self, class: P::Class) -> usize { + self.policy.limit(class) + } + + /// Check that a charge originated from this authority's counters. + pub fn owns(&self, charge: &Charge

) -> bool { + Arc::ptr_eq(&self.totals, &charge.totals) + } + + /// Whether shutdown has been requested for this authority. + pub fn is_stopped(&self) -> bool { + self.stopped.load(Ordering::Acquire) + } + + /// Stop ordinary admission, reclaim idle backing, and wake the shared waiter. + pub fn stop(&self) { + self.stopped.store(true, Ordering::Release); + self.reclaim_buffers(); + self.totals.wake.wake(); + } + + /// A keyed fair-share deficit must be reclaimed from that key. Global + /// pressure can use any idle key; impossible requests have no byte remedy. + pub fn reclamation( + &self, + key: &P::Key, + class: P::Class, + amount: usize, + ) -> Option<(Option, usize)> { + // Keep the counters alive, but release the lookup borrow before callbacks. + let local = self.keys.borrow().get(key).and_then(Weak::upgrade); + let fair = self.fair_limit( + class, + self.active_keys.load(Ordering::Acquire) + usize::from(local.is_none()), + ); + if amount > fair { + return None; + } + let local_deficit = local + .as_ref() + .map_or(0, |local| local.counter(class).used()) + .saturating_sub(fair - amount); + if local_deficit != 0 { + return Some((Some(key.clone()), local_deficit)); + } + let headroom = self.limit(class).checked_sub(amount)?; + let deficit = self.used(class).saturating_sub(headroom); + (deficit != 0).then_some((None, deficit)) + } + + /// Admit nonzero keyed or aggregate-only work under ordinary policy. + pub fn reserve( + &self, + key: Option<&P::Key>, + class: P::Class, + amount: usize, + ) -> Result> { + self.reserve_reclaiming(key, class, amount, AdmissionMode::Ordinary) + } + + /// For already-admitted work during drain only. Aggregate limits still apply. + pub fn reserve_completion( + &self, + key: Option<&P::Key>, + class: P::Class, + amount: usize, + ) -> Result> { + self.reserve_reclaiming(key, class, amount, AdmissionMode::Completion) + } + + /// Sum admission retained by idle recycler allocations. + pub fn retained_buffer_bytes(&self) -> usize { + self.buffers + .lock() + .unwrap_or_else(|e| e.into_inner()) + .iter() + .map(|(_, charge)| charge.amount()) + .sum() + } + + /// Release every idle recycler allocation and its charge. + pub fn reclaim_buffers(&self) { + let reclaimed = { + let mut buffers = self.buffers.lock().unwrap_or_else(|e| e.into_inner()); + std::mem::take(&mut *buffers) + }; + drop(reclaimed); + } + + /// Divide the aggregate ceiling, honoring the clipped per-key floor. + fn fair_limit(&self, class: P::Class, active: usize) -> usize { + let limit = self.limit(class); + (limit / active.max(1)) + .max(self.policy.floor(class)) + .min(limit) + } + + /// Retry overload once after reclaiming relevant retained backing. + fn reserve_reclaiming( + &self, + key: Option<&P::Key>, + class: P::Class, + amount: usize, + mode: AdmissionMode, + ) -> Result> { + let result = self.reserve_inner(key, class, amount, mode); + if matches!(result, Err(Error::Overloaded)) { + self.reclaim_buffers_for(key, class); + return self.reserve_inner(key, class, amount, mode); + } + result + } + + /// Reclaim only when key records, fairness, or this class can benefit. + fn reclaim_buffers_for(&self, key: Option<&P::Key>, class: P::Class) { + let reclaimed = { + let mut buffers = self.buffers.lock().unwrap_or_else(|e| e.into_inner()); + // Keyed admission can exhaust records or fair shares too. Unkeyed + // admission only reclaims when this class has an idle charge. + if key.is_some() + || buffers + .iter() + .any(|(_, charge)| charge.class.index() == class.index()) + { + std::mem::take(&mut *buffers) + } else { + Vec::new() + } + }; + drop(reclaimed); + } + + /// Perform one admission attempt and report its rejection before retrying. + fn reserve_inner( + &self, + key: Option<&P::Key>, + class: P::Class, + amount: usize, + mode: AdmissionMode, + ) -> Result> { + if self.is_stopped() && mode == AdmissionMode::Ordinary && !P::allows_stopped(class) { + return Err(Error::Unavailable); + } + if amount == 0 { + return Err(Error::InvalidInput); + } + let local = key.map(|key| self.key_counters(key)).transpose()?; + // Clone before checking limits: application code may admit more work, + // change policy, stop admission, or panic without owning a charge yet. + let key = key.cloned(); + let limit = self.limit(class); + if let Some(local) = local.as_ref().filter(|_| mode == AdmissionMode::Ordinary) { + let fair = self.fair_limit(class, self.active_keys.load(Ordering::Acquire).max(1)); + let used = local.counter(class).used(); + if used.checked_add(amount).is_none_or(|next| next > fair) { + self.rejected(class, amount, Some(used), Some(fair)); + return Err(Error::Overloaded); + } + } + self.totals + .counter(class) + .reserve(amount, limit) + .map_err(|_| { + self.rejected(class, amount, None, None); + Error::Overloaded + })?; + if let Some(local) = &local { + local.counter(class).add(amount); + } + let charge = Charge { + class, + amount, + key, + totals: self.totals.clone(), + local, + buffers: Arc::downgrade(&self.buffers), + stopped: self.stopped.clone(), + }; + // Policy and key callbacks may stop admission. Drop the completed owner + // to roll back both counters while preserving completion and drain work. + if self.is_stopped() && mode == AdmissionMode::Ordinary && !P::allows_stopped(class) { + return Err(Error::Unavailable); + } + Ok(charge) + } + + /// Reuse a live key record or create one after bounded retirement cleanup. + fn key_counters(&self, key: &P::Key) -> Result>> { + let limit = self.policy.max_keys(); + let mut keys = self.keys.borrow_mut(); + let mut retired = self.retired_keys.lock().unwrap_or_else(|e| e.into_inner()); + for _ in 0..256 { + let Some(id) = retired.pop_front() else { + break; + }; + if keys + .get(&id) + .is_some_and(|counts| counts.strong_count() == 0) + { + keys.remove(&id); + } + } + drop(retired); + if !keys.contains_key(key) && keys.len() >= limit { + let used = keys.len(); + drop(keys); + self.policy.rejected(Rejection::Keys { used, limit }); + return Err(Error::Overloaded); + } + if let Some(counts) = keys.get(key).and_then(Weak::upgrade) { + return Ok(counts); + } + drop(keys); + let retirement_key = key.clone(); + let map_key = key.clone(); + let limit = self.policy.max_keys(); + let mut keys = self.keys.borrow_mut(); + // Either clone may have installed this key or filled the table. + if let Some(counts) = keys.get(key).and_then(Weak::upgrade) { + return Ok(counts); + } + if !keys.contains_key(key) && keys.len() >= limit { + let used = keys.len(); + drop(keys); + self.policy.rejected(Rejection::Keys { used, limit }); + return Err(Error::Overloaded); + } + let counts = Arc::new(Counters::keyed( + P::Class::COUNT, + retirement_key, + self.active_keys.clone(), + self.retired_keys.clone(), + )); + keys.insert(map_key, Arc::downgrade(&counts)); + Ok(counts) + } + + /// Report the current aggregate and optional keyed rejection facts. + fn rejected( + &self, + class: P::Class, + requested: usize, + key_used: Option, + key_limit: Option, + ) { + self.policy.rejected(Rejection::Resource { + class, + used: self.used(class), + limit: self.limit(class), + requested, + key_used, + key_limit, + }); + } +} + +/// Fixed initialized backing paired with its live quota charge. +pub struct ChargedBuffer { + bytes: Vec, + + charge: Option>, +} + +impl ChargedBuffer

{ + /// Allocate admitted backing; the caller validates class, key, and provenance. + pub fn new(mut charge: Charge

, length: usize) -> Result { + if length == 0 { + return Err(Error::InvalidInput); + } + let bytes = charge.buffer(length)?; + // Do not box or resize again while I/O pointers are live. + let bytes = bytes.into_boxed_slice().into_vec(); + charge.shrink(bytes.len())?; + Ok(Self { + bytes, + charge: Some(charge), + }) + } + + /// Borrow the charge retained for the fixed allocation. + pub fn charge(&self) -> &Charge

{ + self.charge.as_ref().expect("owned charge") + } + + /// Borrow initialized bytes without changing allocation geometry. + pub fn bytes(&self) -> &[u8] { + &self.bytes + } + + /// Mutate initialized bytes without changing allocation geometry. + pub fn bytes_mut(&mut self) -> &mut [u8] { + &mut self.bytes + } + + /// Transfer backing and admission together to a final completion owner. + pub fn into_parts(mut self) -> (Box<[u8]>, Charge

) { + ( + std::mem::take(&mut self.bytes).into_boxed_slice(), + self.charge.take().expect("owned charge"), + ) + } +} + +impl Drop for ChargedBuffer

{ + /// Wipe and optionally retain backing before its admission is released. + fn drop(&mut self) { + if let Some(charge) = &mut self.charge { + charge.recycle(std::mem::take(&mut self.bytes)); + } + } +} + +// SAFETY: private fixed backing and charge remain exclusively owned. +unsafe impl uring_runtime::reactor::IoBuffer for ChargedBuffer

{ + /// Buffer access uses the crate's application-independent error type. + type Error = Error; + + /// Borrow stable initialized backing while the runtime owns this buffer. + fn bytes(&self) -> Result<&[u8]> { + Ok(&self.bytes) + } + + /// Borrow stable mutable backing while the runtime owns this buffer. + fn bytes_mut(&mut self) -> Result<&mut [u8]> { + Ok(&mut self.bytes) + } +} + +/// Pending and issued reservations consume the same byte and slot budgets. +pub struct Window { + slots: usize, + + bytes: u64, + + max_item: u64, + + used: u64, + + outstanding: std::collections::BTreeMap, +} + +impl Window { + /// Require positive limits; the item ceiling may exceed the byte budget. + pub fn new(slots: usize, bytes: u64, max_item: u64) -> Result { + if slots == 0 || bytes == 0 || max_item == 0 { + return Err(Error::InvalidInput); + } + Ok(Self { + slots, + bytes, + max_item, + used: 0, + outstanding: std::collections::BTreeMap::new(), + }) + } + + /// Whether a valid length fits both limits, independent of item identity. + pub fn can_reserve(&self, length: u64) -> bool { + length != 0 + && length <= self.max_item + && self.outstanding.len() < self.slots + && length <= self.bytes - self.used + } + + /// Reserve a unique item without returning capacity on later issuance. + pub fn reserve(&mut self, key: K, length: u64) -> Result<()> { + if length == 0 || length > self.max_item { + return Err(Error::InvalidInput); + } + if !self.can_reserve(length) || self.outstanding.contains_key(&key) { + return Err(Error::Overloaded); + } + self.outstanding.insert(key, Credit::Pending(length)); + self.used += length; + Ok(()) + } + + /// Issue a pending item exactly once while retaining its full reservation. + pub fn issued(&mut self, key: K) -> Result<()> { + let entry = self.outstanding.get_mut(&key).ok_or(Error::InvalidInput)?; + let Credit::Pending(length) = *entry else { + return Err(Error::InvalidInput); + }; + *entry = Credit::Issued(length); + Ok(()) + } + + /// Release an issued item exactly once using its admitted length. + pub fn release(&mut self, key: K, length: u64) -> Result<()> { + if self.outstanding.get(&key) != Some(&Credit::Issued(length)) { + return Err(Error::InvalidInput); + } + self.outstanding.remove(&key); + self.used -= length; + Ok(()) + } + + /// Whether neither pending nor issued items retain credit. + pub fn is_empty(&self) -> bool { + self.outstanding.is_empty() + } +} + +/// A credit's lifecycle; only an exact issued credit can be released. +#[derive(Eq, PartialEq)] +enum Credit { + /// Reserved capacity whose item has not yet been issued. + Pending(u64), + + /// Issued capacity awaiting an exact-length acknowledgment. + Issued(u64), +} + +/// Admission intent determines whether existing work may finish during drain. +#[derive(Clone, Copy, Eq, PartialEq)] +enum AdmissionMode { + /// New work obeys stop state and keyed fair shares. + Ordinary, + + /// Existing work bypasses stop and fairness, but not hard limits. + Completion, +} + +/// Retained zeroed backing together with the charge that still accounts for it. +type Buffers

= Mutex, Charge

)>>; + +/// Worker-local key lookup; charges retain records independently. +type Keys = RefCell>>>; + +/// Isolate independently updated resource counters on separate cache lines. +#[repr(align(64))] +struct Counter(AtomicUsize); + +impl Counter { + /// Start one resource class with no admitted usage. + fn new() -> Self { + Self(AtomicUsize::new(0)) + } + + /// Read usage published by admission and cross-thread release. + fn used(&self) -> usize { + self.0.load(Ordering::Acquire) + } + + /// Add usage only when both arithmetic and the aggregate ceiling allow it. + fn reserve(&self, amount: usize, limit: usize) -> Result<()> { + self.0 + .fetch_update(Ordering::AcqRel, Ordering::Acquire, |used| { + used.checked_add(amount).filter(|next| *next <= limit) + }) + .map(|_| ()) + .map_err(|_| Error::Overloaded) + } + + /// Record keyed usage already covered by aggregate admission. + fn add(&self, amount: usize) { + self.0.fetch_add(amount, Ordering::AcqRel); + } + + /// Return usage owned exclusively by the releasing charge. + fn release(&self, amount: usize) { + self.0.fetch_sub(amount, Ordering::AcqRel); + } +} + +/// Shared counters outliving the worker-local authority while charges remain. +struct Counters { + retirement: Option>, + + used: Box<[Counter]>, + + wake: futures::task::AtomicWaker, +} + +impl Counters { + /// Allocate zeroed, independently padded counters for all classes. + fn new(classes: usize) -> Self { + Self { + retirement: None, + used: (0..classes).map(|_| Counter::new()).collect(), + wake: futures::task::AtomicWaker::new(), + } + } + + /// Couple one live key record to its eventual retirement notification. + fn keyed( + classes: usize, + key: K, + active: Arc, + queue: Arc>>, + ) -> Self { + let mut counters = Self::new(classes); + active.fetch_add(1, Ordering::AcqRel); + counters.retirement = Some(Retirement { key, active, queue }); + counters + } + + /// Select the padded counter using the policy's stable class index. + fn counter(&self, class: C) -> &Counter { + &self.used[class.index()] + } +} + +impl Drop for Counters { + /// Retire the complete keyed state when its final charge leaves. + fn drop(&mut self) { + if let Some(retirement) = self.retirement.take() { + retirement.retire(); + } + } +} + +/// One live key record owns both its active count and its cleanup notification. +struct Retirement { + key: K, + + active: Arc, + + queue: Arc>>, +} + +impl Retirement { + /// Publish retirement only after the last owner releases the key record. + fn retire(self) { + self.active.fetch_sub(1, Ordering::AcqRel); + self.queue + .lock() + .unwrap_or_else(|e| e.into_inner()) + .push_back(self.key); + } +} + +/// Initialize and wipe the full allocation, including truncated and spare bytes. +fn wipe_payload(bytes: &mut Vec) { + bytes.clear(); + #[cfg(all(target_os = "linux", any(target_env = "gnu", target_env = "musl")))] + if bytes.capacity() != 0 { + // SAFETY: the exclusive Vec owns capacity writable bytes. explicit_bzero + // initializes spare capacity and cannot be removed as a dead store. + unsafe { libc::explicit_bzero(bytes.as_mut_ptr().cast(), bytes.capacity()) }; + } + #[cfg(not(all(target_os = "linux", any(target_env = "gnu", target_env = "musl"))))] + { + use zeroize::Zeroize; + bytes.zeroize(); + } +} + +/// Accounting, retirement, reclamation, and cross-thread release contracts. +#[cfg(test)] +mod quota_tests { + use super::*; + + /// Unavailability does not imply that flow control has stopped. + #[test] + fn unavailable_display_is_state_neutral() { + assert_eq!(Error::Unavailable.to_string(), "flow control unavailable"); + } + + /// Distinct payload, wake-enabled, and drain-progress fixture classes. + #[derive(Clone, Copy, Debug)] + enum Resource { + Payload, + + Other, + + Progress, + } + + impl Class for Resource { + /// Number of fixture resource classes. + const COUNT: usize = 3; + + /// Map each fixture class to its stable counter. + fn index(self) -> usize { + self as usize + } + } + + /// Optional limit samples and a shared rejection log for assertions. + #[derive(Clone)] + struct TestPolicy { + limit: usize, + + max_keys: usize, + + rejected: Arc>>>, + + limit_gate: Option>, + + limit_samples: Arc>>, + } + + impl TestPolicy { + /// Construct fixed aggregate limits with an empty rejection log. + fn new(limit: usize, max_keys: usize) -> Self { + Self { + limit, + max_keys, + rejected: Arc::default(), + limit_gate: None, + limit_samples: Arc::default(), + } + } + } + + impl Policy for TestPolicy { + /// Fixture resource classes with separate usage counters. + type Class = Resource; + + /// Owned identities retained until all their charges leave. + type Key = String; + + /// All fixture classes use the same aggregate ceiling. + fn limit(&self, _: Resource) -> usize { + if let Some(gate) = &self.limit_gate { + gate.0.wait(); + gate.1.wait(); + } + self.limit_samples + .lock() + .unwrap() + .pop_front() + .unwrap_or(self.limit) + } + + /// Return the fixture's key-record bound. + fn max_keys(&self) -> usize { + self.max_keys + } + + /// Only the non-payload fixture class wakes shared waiters. + fn wakes(class: Resource) -> bool { + matches!(class, Resource::Other) + } + + /// Only payload admission may cover page backing. + fn covers(class: Resource) -> bool { + matches!(class, Resource::Payload) + } + + /// Progress reservations remain available during drain. + fn allows_stopped(class: Resource) -> bool { + matches!(class, Resource::Progress) + } + + /// Preserve rejection order and facts for assertions. + fn rejected(&self, rejection: Rejection) { + self.rejected.lock().unwrap().push(rejection); + } + } + + type TestBufferPool = Arc, Charge)>>>; + + /// Checks recycler lock availability during a synchronous charge wake. + struct PoolWake { + buffers: TestBufferPool, + + unlocked: Arc, + } + + impl std::task::Wake for PoolWake { + /// Record whether callback reentry can acquire the recycler mutex. + fn wake(self: Arc) { + self.unlocked + .store(self.buffers.try_lock().is_ok(), Ordering::SeqCst); + } + } + + /// Register a callback that probes the recycler mutex when the next charge drops. + fn watch_buffer_unlock(quotas: &Quotas) -> Arc { + let unlocked = Arc::new(AtomicBool::new(false)); + let wake = Arc::new(PoolWake { + buffers: quotas.buffers.clone(), + unlocked: unlocked.clone(), + }); + quotas.shared().register(&std::task::Waker::from(wake)); + unlocked + } + + /// Wiping initializes every allocated byte without moving or resizing backing. + #[test] + fn secure_payload_wipe_initializes_spare_capacity_and_preserves_geometry() { + for capacity in [0, 1, 15, 16, 17, 63, 64, 65, 4095, 4096, 4097] { + for initialized in [false, true] { + let mut bytes = Vec::with_capacity(capacity); + if initialized { + bytes.resize(bytes.capacity(), 0xa7); + bytes.truncate(capacity / 2); + } + let pointer = bytes.as_ptr(); + let allocated = bytes.capacity(); + wipe_payload(&mut bytes); + assert!(bytes.is_empty()); + assert_eq!(bytes.capacity(), allocated); + assert_eq!(bytes.as_ptr(), pointer); + // SAFETY: wipe_payload initializes every byte of the allocation. + unsafe { bytes.set_len(allocated) }; + assert!(bytes.iter().all(|byte| *byte == 0)); + wipe_payload(&mut bytes); + assert!(bytes.is_empty()); + } + } + } + + /// Compare full-capacity secure wiping on the same alternating workload. + #[test] + #[ignore = "release-only alternating full-capacity secure wipe comparison"] + #[allow(clippy::assertions_on_constants)] + fn secure_payload_wipe_benchmark() { + use std::{hint::black_box, time::Instant}; + use zeroize::Zeroize; + assert!(!cfg!(debug_assertions), "run with --release"); + const ITERATIONS: usize = 128; + for length in [1 << 20, 16 << 20, (16 << 20) + 16] { + let mut bytes = vec![0u8; length]; + for sample in 0..6 { + for optimized in if sample % 2 == 0 { + [false, true] + } else { + [true, false] + } { + let start = Instant::now(); + for _ in 0..ITERATIONS { + bytes.fill(black_box(0xa7)); + black_box(&bytes); + if optimized { + wipe_payload(&mut bytes); + } else { + bytes.clear(); + bytes.zeroize(); + } + // SAFETY: both primitives initialize the full capacity. + unsafe { bytes.set_len(length) }; + black_box(&bytes); + } + let elapsed = start.elapsed(); + assert!(bytes.iter().all(|byte| *byte == 0)); + if sample != 0 { + println!( + "secure_wipe length={length} optimized={optimized} sample={sample} iterations={ITERATIONS} ns_per_op={:.0}", + elapsed.as_nanos() as f64 / ITERATIONS as f64 + ); + } + } + } + } + } + + /// Fairness, retirement, and drain completion retain their separate limits. + #[test] + fn fairness_retirement_reclamation_and_completion() { + let quotas = Quotas::new(TestPolicy::new(100, 2)); + let (a, b, c) = ("a".to_owned(), "b".to_owned(), "c".to_owned()); + let first = quotas.reserve(Some(&a), Resource::Payload, 40).unwrap(); + let second = quotas.reserve(Some(&b), Resource::Payload, 40).unwrap(); + assert!(matches!( + quotas.reserve(Some(&a), Resource::Payload, 11), + Err(Error::Overloaded) + )); + assert_eq!( + quotas.reclamation(&a, Resource::Payload, 11), + Some((Some(a.clone()), 1)) + ); + assert_eq!(quotas.reclamation(&a, Resource::Payload, 51), None); + assert!(matches!( + quotas.reserve(Some(&c), Resource::Other, 1), + Err(Error::Overloaded) + )); + drop(second); + let third = quotas.reserve(Some(&a), Resource::Payload, 60).unwrap(); + assert_eq!(quotas.active_keys.load(Ordering::Acquire), 1); + assert!(!quotas.keys.borrow().contains_key(&b)); + assert_eq!( + quotas.reclamation(&a, Resource::Payload, 1), + Some((Some(a.clone()), 1)) + ); + let unkeyed = quotas.reserve(None, Resource::Other, 100).unwrap(); + assert_eq!(quotas.reclamation(&a, Resource::Other, 1), Some((None, 1))); + drop((unkeyed, first, third)); + let first = quotas.reserve(Some(&a), Resource::Payload, 60).unwrap(); + let second = quotas.reserve(Some(&b), Resource::Other, 1).unwrap(); + quotas.stop(); + assert!(matches!( + quotas.reserve(None, Resource::Payload, 1), + Err(Error::Unavailable) + )); + assert!(quotas.reserve(None, Resource::Progress, 1).is_ok()); + let completion = quotas + .reserve_completion(Some(&a), Resource::Payload, 40) + .unwrap(); + assert_eq!(quotas.used(Resource::Payload), 100); + assert!(matches!( + quotas.reserve_completion(None, Resource::Payload, 1), + Err(Error::Overloaded) + )); + drop((completion, first, second)); + assert_eq!(quotas.used(Resource::Payload), 0); + } + + /// A lower second limit must not wrap or suggest an impossible byte remedy. + #[test] + fn dynamic_limits_reclamation_revalidates_aggregate_headroom() { + for shared_usage in [false, true] { + for existing_key in [false, true] { + for (first, second, amount, expected) in [ + (10, 0, 5, None), + (10, 4, 5, None), + (10, 5, 5, Some((None, 3))), + (10, 7, 5, Some((None, 1))), + (10, 8, 5, None), + (10, 20, 5, None), + (usize::MAX, usize::MAX, usize::MAX, Some((None, 3))), + (4, 0, 5, None), + ] { + let quotas = Quotas::new(TestPolicy::new(100, 1)); + let shared = quotas.shared(); + let key = "a".to_owned(); + let owner = existing_key + .then(|| quotas.reserve(Some(&key), Resource::Other, 1).unwrap()); + let held = if shared_usage { + shared.reserve(Resource::Payload, 3).unwrap() + } else { + quotas.reserve(None, Resource::Payload, 3).unwrap() + }; + quotas + .policy + .limit_samples + .lock() + .unwrap() + .extend([first, second]); + assert_eq!( + quotas.reclamation(&key, Resource::Payload, amount), + expected + ); + assert_eq!(held.amount(), 3); + assert_eq!(quotas.used(Resource::Payload), 3); + assert_eq!(shared.used(Resource::Payload), 3); + assert_eq!( + quotas.active_keys.load(Ordering::Acquire), + usize::from(existing_key) + ); + assert!(quotas.policy.rejected.lock().unwrap().is_empty()); + drop((held, owner)); + assert_eq!(shared.used(Resource::Payload), 0); + assert_eq!(quotas.active_keys.load(Ordering::Acquire), 0); + } + } + } + } + + /// Reclamation callbacks may admit keyed work without borrowing the lookup table. + mod reclamation_callbacks { + use super::*; + + /// Application callback selected for one synchronous admission. + #[derive(Clone, Copy, Eq, PartialEq)] + enum Callback { + Limit, + + Floor, + + Clone, + } + + /// A one-shot callback, optionally delayed past the first limit sample. + struct Hook { + callback: Callback, + + skip: usize, + + action: Box, + } + + thread_local! { + static HOOK: RefCell> = const { RefCell::new(None) }; + } + + /// Take the selected callback before running it so nested calls are inert. + fn invoke(callback: Callback) { + let hook = HOOK.with(|slot| { + let mut slot = slot.borrow_mut(); + let hook = slot.as_mut()?; + if hook.callback != callback { + return None; + } + if hook.skip != 0 { + hook.skip -= 1; + return None; + } + slot.take() + }); + if let Some(hook) = hook { + (hook.action)(); + } + } + + /// A transferable identity with a worker-local clone callback. + #[derive(Debug, Eq, PartialEq, Hash)] + struct Key(u8); + + impl Clone for Key { + /// Exercise application code when reclamation returns a keyed deficit. + fn clone(&self) -> Self { + invoke(Callback::Clone); + Self(self.0) + } + } + + /// Fixed limits with one-shot application callbacks. + struct ReentrantPolicy; + + impl Policy for ReentrantPolicy { + type Class = Resource; + + type Key = Key; + + /// Permit both the initial charges and the nested admission. + fn limit(&self, _: Resource) -> usize { + invoke(Callback::Limit); + 100 + } + + /// Exercise callback reentry during fair-share calculation. + fn floor(&self, _: Resource) -> usize { + invoke(Callback::Floor); + 1 + } + + /// Leave room for nested creation as well as live-key reuse. + fn max_keys(&self) -> usize { + 3 + } + + /// No waiter is needed for this synchronous regression. + fn wakes(_: Resource) -> bool { + false + } + + /// The fixture does not allocate page backing. + fn covers(_: Resource) -> bool { + false + } + + /// Every nested admission must succeed without a retry. + fn rejected(&self, _: Rejection) { + panic!("unexpected rejection"); + } + } + + /// Check deficit results and accounting after new-key and live-key reentry. + fn check(callback: Callback, skip: usize) { + for nested_key in [0, 2] { + for (keyed, unkeyed, amount, expected) in [ + (40, 0, 11, Some((Some(Key(0)), 1))), + (0, 90, 11, Some((None, 1))), + (0, 0, 11, None), + (0, 0, 51, None), + ] { + let local_deficit = keyed != 0; + if (callback == Callback::Clone && !local_deficit) + || (skip != 0 && (local_deficit || amount > 50)) + { + continue; + } + let quotas = Rc::new(Quotas::new(ReentrantPolicy)); + let first = quotas.reserve(Some(&Key(0)), Resource::Other, 1).unwrap(); + let second = quotas.reserve(Some(&Key(1)), Resource::Other, 1).unwrap(); + let payload = (keyed + unkeyed != 0).then(|| { + quotas + .reserve( + (keyed != 0).then_some(&Key(0)), + Resource::Payload, + keyed + unkeyed, + ) + .unwrap() + }); + let nested = quotas.clone(); + HOOK.with(|slot| { + *slot.borrow_mut() = Some(Hook { + callback, + skip, + action: Box::new(move || { + let charge = nested + .reserve(Some(&Key(nested_key)), Resource::Progress, 1) + .unwrap(); + assert_eq!(nested.used(Resource::Progress), 1); + drop(charge); + assert_eq!(nested.used(Resource::Progress), 0); + }), + }); + }); + assert_eq!( + quotas.reclamation(&Key(0), Resource::Payload, amount), + expected + ); + assert!(HOOK.with(|slot| slot.borrow().is_none())); + assert_eq!(quotas.used(Resource::Payload), keyed + unkeyed); + assert_eq!(quotas.active_keys.load(Ordering::Acquire), 2); + drop((payload, first, second)); + assert_eq!(quotas.used(Resource::Payload), 0); + assert_eq!(quotas.used(Resource::Other), 0); + assert_eq!(quotas.active_keys.load(Ordering::Acquire), 0); + } + } + } + + /// The first aggregate sample runs without a key-table borrow. + #[test] + fn fair_limit_allows_keyed_reentry() { + check(Callback::Limit, 0); + } + + /// The later aggregate headroom sample also permits keyed reentry. + #[test] + fn aggregate_limit_allows_keyed_reentry() { + check(Callback::Limit, 1); + } + + /// The per-key floor callback may admit work synchronously. + #[test] + fn floor_allows_keyed_reentry() { + check(Callback::Floor, 0); + } + + /// Cloning the returned identity may admit work synchronously. + #[test] + fn key_clone_allows_keyed_reentry() { + check(Callback::Clone, 0); + } + } + + /// Lower ceilings reject without losing old charges; higher ceilings admit exactly. + #[test] + fn dynamic_limits_admission_preserves_failure_accounting_and_success_edges() { + let quotas = Quotas::new(TestPolicy::new(4, 1)); + let shared = quotas.shared(); + let key = "a".to_owned(); + quotas.policy.limit_samples.lock().unwrap().extend([10, 10]); + let held = quotas.reserve(Some(&key), Resource::Payload, 6).unwrap(); + assert!(matches!( + shared.reserve(Resource::Payload, 1), + Err(Error::Overloaded) + )); + assert!(matches!( + quotas.reserve(None, Resource::Payload, 1), + Err(Error::Overloaded) + )); + assert!(matches!( + quotas.reserve(Some(&key), Resource::Payload, 1), + Err(Error::Overloaded) + )); + assert!(matches!( + quotas.reserve_completion(Some(&key), Resource::Payload, 1), + Err(Error::Overloaded) + )); + assert!(matches!( + shared.reserve(Resource::Payload, 0), + Err(Error::InvalidInput) + )); + assert!(matches!( + quotas.reserve(Some(&key), Resource::Payload, 0), + Err(Error::InvalidInput) + )); + assert_eq!(quotas.used(Resource::Payload), 6); + assert_eq!(shared.used(Resource::Payload), 6); + assert_eq!( + held.local + .as_ref() + .unwrap() + .counter(Resource::Payload) + .used(), + 6 + ); + assert_eq!(quotas.active_keys.load(Ordering::Acquire), 1); + { + let rejected = quotas.policy.rejected.lock().unwrap(); + assert_eq!( + rejected.len(), + 7, + "local retries once; shared does not retry" + ); + for (index, rejection) in rejected.iter().enumerate() { + let (key_used, key_limit) = if matches!(index, 3 | 4) { + (Some(6), Some(4)) + } else { + (None, None) + }; + assert!(matches!(rejection, Rejection::Resource { + class: Resource::Payload, used: 6, limit: 4, requested: 1, + key_used: actual_used, key_limit: actual_limit, + } if *actual_used == key_used && *actual_limit == key_limit)); + } + } + quotas.policy.limit_samples.lock().unwrap().extend([10, 10]); + let refill = quotas.reserve(Some(&key), Resource::Payload, 4).unwrap(); + assert_eq!(shared.used(Resource::Payload), 10); + assert_eq!( + held.local + .as_ref() + .unwrap() + .counter(Resource::Payload) + .used(), + 10 + ); + drop((refill, held)); + assert_eq!(shared.used(Resource::Payload), 0); + assert_eq!(quotas.active_keys.load(Ordering::Acquire), 0); + let exact = shared.reserve(Resource::Payload, 4).unwrap(); + assert_eq!(quotas.used(Resource::Payload), 4); + drop(exact); + assert_eq!(quotas.used(Resource::Payload), 0); + assert_eq!(quotas.policy.rejected.lock().unwrap().len(), 7); + } + + /// Check nested admission at each key clone and preserve all owned counters. + fn check_quota_key_clone(skip: usize) { + use std::cell::Cell; + + /// Select one clone callback without affecting nested admission. + struct Hook { + skip: usize, + + action: Box, + } + + thread_local! { + static HOOK: RefCell> = const { RefCell::new(None) }; + } + + /// A transferable key with a worker-local clone hook. + #[derive(Debug, Eq, PartialEq, Hash)] + struct Key(u8); + + impl Clone for Key { + /// Release the hook borrow before running application code. + fn clone(&self) -> Self { + let hook = HOOK.with(|slot| { + let mut slot = slot.borrow_mut(); + let hook = slot.as_mut()?; + if hook.skip != 0 { + hook.skip -= 1; + return None; + } + slot.take() + }); + if let Some(hook) = hook { + (hook.action)(); + } + Self(self.0) + } + } + + /// Mutable ceilings and observed rejection facts. + struct ClonePolicy { + limit: Cell, + + max_keys: Cell, + + rejected: RefCell>>, + } + + impl Policy for ClonePolicy { + type Class = Resource; + + type Key = Key; + + /// Read the current aggregate ceiling. + fn limit(&self, _: Resource) -> usize { + self.limit.get() + } + + /// Read the current record ceiling. + fn max_keys(&self) -> usize { + self.max_keys.get() + } + + /// This fixture has no admission waiter. + fn wakes(_: Resource) -> bool { + false + } + + /// This fixture allocates no page backing. + fn covers(_: Resource) -> bool { + false + } + + /// Retain the rejection facts for assertions. + fn rejected(&self, rejection: Rejection) { + self.rejected.borrow_mut().push(rejection); + } + } + + for action in [ + "same", + "keys", + "fair", + "global", + "limit", + "key_limit", + "stop", + "panic", + ] { + for completion in [false, true] { + let quotas = Rc::new(Quotas::new(ClonePolicy { + limit: Cell::new(10), + max_keys: Cell::new(2), + rejected: RefCell::default(), + })); + let held = Rc::new(RefCell::new(Vec::new())); + let nested = quotas.clone(); + let nested_held = held.clone(); + HOOK.with(|slot| { + *slot.borrow_mut() = Some(Hook { + skip, + action: Box::new(move || match action { + "same" => nested_held + .borrow_mut() + .push(nested.reserve(Some(&Key(0)), Resource::Payload, 3).unwrap()), + "keys" => { + for key in [1, 2] { + let result = + nested.reserve(Some(&Key(key)), Resource::Other, 1); + if let Ok(charge) = result { + nested_held.borrow_mut().push(charge); + } else { + assert_eq!(nested.keys.borrow().len(), 2); + } + } + } + "fair" => nested_held + .borrow_mut() + .push(nested.reserve(Some(&Key(1)), Resource::Other, 1).unwrap()), + "global" => nested_held + .borrow_mut() + .push(nested.reserve(None, Resource::Payload, 4).unwrap()), + "limit" => nested.policy.limit.set(6), + "key_limit" => nested.policy.max_keys.set(0), + "stop" => nested.stop(), + "panic" => panic!("key clone failed"), + _ => unreachable!(), + }), + }); + }); + let result = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { + if completion { + quotas.reserve_completion(Some(&Key(0)), Resource::Payload, 7) + } else { + quotas.reserve(Some(&Key(0)), Resource::Payload, 7) + } + })); + assert!(HOOK.with(|slot| slot.borrow().is_none())); + if action == "panic" { + assert!(result.is_err()); + } else { + let result = result.unwrap(); + let expected = match action { + "keys" if skip < 2 || !completion => Some(Error::Overloaded), + "key_limit" if skip < 2 => Some(Error::Overloaded), + "fair" if !completion => Some(Error::Overloaded), + "global" | "limit" => Some(Error::Overloaded), + "stop" if !completion => Some(Error::Unavailable), + _ => None, + }; + if let Some(expected) = expected { + assert!( + matches!(result, Err(error) if error == expected), + "{action}" + ); + } else { + let charge = result.unwrap(); + assert_eq!(charge.key(), Some(&Key(0))); + if action == "same" { + assert!(Arc::ptr_eq( + charge.local.as_ref().unwrap(), + held.borrow()[0].local.as_ref().unwrap(), + )); + } + held.borrow_mut().push(charge); + } + } + if action == "keys" && skip < 2 { + assert!(matches!( + quotas.policy.rejected.borrow().as_slice(), + [ + Rejection::Keys { used: 2, limit: 2 }, + Rejection::Keys { used: 2, limit: 2 } + ] + )); + } + assert!(quotas.keys.borrow().len() <= 2); + let owners = held.borrow(); + for class in [Resource::Payload, Resource::Other] { + let total: usize = owners + .iter() + .filter(|charge| charge.class.index() == class.index()) + .map(Charge::amount) + .sum(); + assert_eq!(quotas.used(class), total, "{action}"); + assert!(total <= quotas.limit(class)); + for charge in owners.iter().filter(|charge| charge.local.is_some()) { + let local = charge.local.as_ref().unwrap(); + let keyed: usize = owners + .iter() + .filter(|other| { + other.key() == charge.key() && other.class.index() == class.index() + }) + .map(Charge::amount) + .sum(); + assert_eq!(local.counter(class).used(), keyed); + } + } + let active = owners + .iter() + .filter_map(Charge::key) + .collect::>() + .len(); + assert_eq!(quotas.active_keys.load(Ordering::Acquire), active); + drop(owners); + held.borrow_mut().clear(); + assert_eq!(quotas.used(Resource::Payload), 0); + assert_eq!(quotas.used(Resource::Other), 0); + assert_eq!(quotas.active_keys.load(Ordering::Acquire), 0); + } + } + } + + /// The retirement-key clone may recursively reserve keyed work. + #[test] + fn quota_key_clone_first_reentry() { + check_quota_key_clone(0); + } + + /// The map-key clone must also run without holding the map borrow. + #[test] + fn quota_key_clone_second_reentry() { + check_quota_key_clone(1); + } + + /// The final charge-key clone must precede admission checks. + #[test] + fn quota_key_clone_charge_reentry() { + check_quota_key_clone(2); + } + + /// Idle backing retains two live charges and reports pressure before retry. + #[test] + fn recycler_retains_two_live_charges_and_reports_pressure_before_retry() { + let size = 1 << 20; + let quotas = Quotas::new(TestPolicy::new(3 * size, 4)); + let mut buffers: Vec<_> = (0..3) + .map(|_| { + let charge = quotas.reserve(None, Resource::Payload, size).unwrap(); + let bytes = charge.buffer(size).unwrap(); + (charge, bytes) + }) + .collect(); + for (mut charge, mut bytes) in buffers.drain(..) { + bytes.fill(0xa7); + bytes.truncate(1); + charge.recycle(bytes); + } + assert_eq!(quotas.buffers.lock().unwrap().len(), 2); + assert_eq!(quotas.retained_buffer_bytes(), 2 * size); + assert!( + quotas + .buffers + .lock() + .unwrap() + .iter() + .all(|(b, _)| b.len() == size && b.iter().all(|v| *v == 0)) + ); + let shared = quotas.shared(); + let other = quotas.reserve(None, Resource::Other, 3 * size).unwrap(); + assert!(matches!( + quotas.reserve(None, Resource::Other, 1), + Err(Error::Overloaded) + )); + assert_eq!(quotas.retained_buffer_bytes(), 2 * size); + assert_eq!(quotas.policy.rejected.lock().unwrap().len(), 2); + let charge = quotas.reserve(None, Resource::Payload, 3 * size).unwrap(); + assert_eq!(quotas.retained_buffer_bytes(), 0); + assert_eq!(quotas.policy.rejected.lock().unwrap().len(), 3); + assert!( + matches!(quotas.policy.rejected.lock().unwrap()[2], Rejection::Resource { used, requested, .. } if used == 2 * size && requested == 3 * size) + ); + let mut charge = charge; + charge.recycle(vec![0xa7; 3 * size]); + drop((charge, other, quotas)); + assert_eq!( + shared.used(Resource::Payload), + 0, + "usage handles must not retain recycled buffers" + ); + } + + /// Recycled and reclaimed charges wake only after the recycler mutex is released. + #[test] + fn recycler_charge_wakes_after_unlock_on_take_and_reclaim() { + let size = 1 << 20; + let quotas = Quotas::new(TestPolicy::new(3 * size, 4)); + let mut charge = quotas.reserve(None, Resource::Other, size).unwrap(); + charge.recycle(vec![0xa7; size]); + + let mut owner = quotas.reserve(None, Resource::Other, size).unwrap(); + let unlocked = watch_buffer_unlock("as); + let bytes = owner.buffer(size).unwrap(); + assert!(unlocked.load(Ordering::SeqCst)); + owner.recycle(bytes); + + let unlocked = watch_buffer_unlock("as); + quotas.reclaim_buffers(); + assert!(unlocked.load(Ordering::SeqCst)); + assert_eq!(quotas.retained_buffer_bytes(), 0); + } + + /// Keyed pressure reclamation drops retained charges outside the pool lock. + #[test] + fn pressure_reclamation_drops_charges_after_unlock() { + let size = 1 << 20; + let quotas = Quotas::new(TestPolicy::new(size, 4)); + let mut charge = quotas.reserve(None, Resource::Other, size).unwrap(); + charge.recycle(vec![0xa7; size]); + + let unlocked = watch_buffer_unlock("as); + let key = "reclaim".to_owned(); + assert!(quotas.reserve(Some(&key), Resource::Other, 1).is_ok()); + assert!(unlocked.load(Ordering::SeqCst)); + assert_eq!(quotas.retained_buffer_bytes(), 0); + } + + /// Key cloning runs unlocked, and callback changes are checked before retention. + #[test] + fn recycler_key_clone_runs_before_retention_checks() { + thread_local! { + static CLONE_HOOK: RefCell>> = RefCell::new(None); + } + + #[derive(Eq, PartialEq, Hash)] + struct Key; + + impl Clone for Key { + fn clone(&self) -> Self { + let hook = CLONE_HOOK.with(|slot| slot.borrow_mut().take()); + if let Some(hook) = hook { + hook(); + } + Self + } + } + + struct ClonePolicy; + + impl Policy for ClonePolicy { + type Class = Resource; + type Key = Key; + + fn limit(&self, _: Resource) -> usize { + 4 << 20 + } + + fn max_keys(&self) -> usize { + 1 + } + + fn wakes(_: Resource) -> bool { + false + } + + fn covers(_: Resource) -> bool { + false + } + + fn rejected(&self, _: Rejection) { + panic!("unexpected rejection"); + } + } + + let size = 1 << 20; + for action in ["retain", "stop", "fill"] { + let quotas = Rc::new(Quotas::new(ClonePolicy)); + let mut idle = quotas.reserve(None, Resource::Payload, size).unwrap(); + idle.recycle(vec![0xa7; size]); + let mut donor = quotas + .reserve(Some(&Key), Resource::Payload, 2 * size) + .unwrap(); + let nested = quotas.clone(); + CLONE_HOOK.with(|slot| { + *slot.borrow_mut() = Some(Box::new(move || { + // Fail before reentry can block on the same mutex. + assert!( + nested.buffers.try_lock().is_ok(), + "key cloned under recycler lock" + ); + match action { + "stop" => nested.stop(), + "fill" => { + let mut extra = nested.reserve(None, Resource::Payload, size).unwrap(); + extra.recycle(vec![0xa7; size]); + } + _ => {} + } + })); + }); + + donor.recycle(vec![0xa7; size]); + assert!(CLONE_HOOK.with(|slot| slot.borrow().is_none())); + assert_eq!(quotas.is_stopped(), action == "stop"); + let retained = match action { + "retain" => 3 * size, + "fill" => 2 * size, + _ => 0, + }; + let owned = if action == "retain" { 0 } else { 2 * size }; + assert_eq!(donor.amount(), owned); + assert!(donor.key().is_some()); + assert_eq!( + donor + .local + .as_ref() + .unwrap() + .counter(Resource::Payload) + .used(), + 2 * size + ); + assert_eq!(quotas.retained_buffer_bytes(), retained); + assert_eq!(quotas.used(Resource::Payload), retained + owned); + drop(donor); + quotas.reclaim_buffers(); + assert_eq!(quotas.used(Resource::Payload), 0); + assert_eq!(quotas.active_keys.load(Ordering::Acquire), 0); + } + } + + /// A poisoned recycler remains usable after a prior operation panicked. + #[test] + fn recycler_recovers_poisoned_mutex() { + use std::panic::{AssertUnwindSafe, catch_unwind}; + + let size = 1 << 20; + let quotas = Quotas::new(TestPolicy::new(2 * size, 4)); + let buffers = quotas.buffers.clone(); + assert!( + catch_unwind(AssertUnwindSafe(|| { + let _buffers = buffers.lock().unwrap(); + panic!("poison recycler mutex"); + })) + .is_err() + ); + + let mut charge = quotas.reserve(None, Resource::Other, size).unwrap(); + charge.recycle(vec![0xa7; size]); + assert_eq!(quotas.retained_buffer_bytes(), size); + } + + /// Key-table pressure reports original facts before successful reclamation. + #[test] + fn key_record_rejection_is_reported_before_reclaim_retry() { + let size = 1 << 20; + let quotas = Quotas::new(TestPolicy::new(2 * size, 1)); + let (a, b) = ("a".to_owned(), "b".to_owned()); + let mut old = quotas.reserve(Some(&a), Resource::Payload, size).unwrap(); + old.recycle(vec![0xa7; size]); + drop(old); + let charge = quotas.reserve(Some(&b), Resource::Other, 1).unwrap(); + assert_eq!(quotas.retained_buffer_bytes(), 0); + assert!(!quotas.keys.borrow().contains_key(&a)); + assert!(matches!( + "as.policy.rejected.lock().unwrap()[..], + [Rejection::Keys { used: 1, limit: 1 }] + )); + drop(charge); + } + + /// Key-table rejection callbacks can inspect and admit work for a live key. + #[test] + fn key_record_rejection_allows_keyed_reentry() { + /// Keep a weak link to the authority and guard callback reentry. + struct ReentrantPolicy { + quotas: RefCell>>, + entered: std::cell::Cell, + rejected: RefCell>>, + } + + impl Policy for ReentrantPolicy { + type Class = Resource; + type Key = String; + + /// Leave room for nested admission under the existing key. + fn limit(&self, _: Resource) -> usize { + 10 + } + + /// Force a rejection for each new key while the first is live. + fn max_keys(&self) -> usize { + 1 + } + + /// This test does not register admission waiters. + fn wakes(_: Resource) -> bool { + false + } + + /// This test does not allocate page backing. + fn covers(_: Resource) -> bool { + false + } + + /// Reenter once and keep all rejection facts for assertions. + fn rejected(&self, rejection: Rejection) { + self.rejected.borrow_mut().push(rejection); + if self.entered.replace(true) { + return; + } + let quotas = self.quotas.borrow().upgrade().unwrap(); + let key = "live".to_owned(); + assert_eq!( + quotas.reclamation(&key, Resource::Payload, 10), + Some((Some(key.clone()), 1)) + ); + let nested = quotas.reserve(Some(&key), Resource::Payload, 2).unwrap(); + assert_eq!(quotas.used(Resource::Payload), 3); + drop(nested); + assert_eq!(quotas.used(Resource::Payload), 1); + } + } + + let quotas = Rc::new(Quotas::new(ReentrantPolicy { + quotas: RefCell::default(), + entered: std::cell::Cell::new(false), + rejected: RefCell::default(), + })); + *quotas.policy.quotas.borrow_mut() = Rc::downgrade("as); + let (live, other) = ("live".to_owned(), "other".to_owned()); + let charge = quotas.reserve(Some(&live), Resource::Payload, 1).unwrap(); + assert!(matches!( + quotas.reserve(Some(&other), Resource::Payload, 1), + Err(Error::Overloaded) + )); + assert!(quotas.policy.entered.get()); + assert!(matches!( + "as.policy.rejected.borrow()[..], + [ + Rejection::Keys { used: 1, limit: 1 }, + Rejection::Keys { used: 1, limit: 1 } + ] + )); + assert_eq!(quotas.used(Resource::Payload), 1); + assert_eq!(quotas.active_keys.load(Ordering::Acquire), 1); + drop(charge); + let replacement = quotas.reserve(Some(&other), Resource::Payload, 1).unwrap(); + assert!(!quotas.keys.borrow().contains_key(&live)); + drop(replacement); + assert_eq!(quotas.used(Resource::Payload), 0); + assert_eq!(quotas.active_keys.load(Ordering::Acquire), 0); + } + + /// A final keyed release frees its record before synchronous waiter reentry. + #[test] + fn keyed_release_retires_before_reentrant_wake() { + thread_local! { + static ON_WAKE: RefCell>> = RefCell::new(None); + } + + struct ReentrantWake(AtomicUsize); + + impl std::task::Wake for ReentrantWake { + fn wake(self: Arc) { + self.0.fetch_add(1, Ordering::Relaxed); + let callback = ON_WAKE.with(|slot| slot.borrow_mut().take()); + if let Some(callback) = callback { + callback(); + } + } + } + + let quotas = Rc::new(Quotas::new(TestPolicy::new(10, 1))); + let old = "old".to_owned(); + let mut charge = quotas.reserve(Some(&old), Resource::Other, 10).unwrap(); + let split = charge.split(4).unwrap(); + let wake = Arc::new(ReentrantWake(AtomicUsize::new(0))); + let waker = std::task::Waker::from(wake.clone()); + + let nested = quotas.clone(); + ON_WAKE.with(|slot| { + *slot.borrow_mut() = Some(Box::new(move || { + assert_eq!(nested.used(Resource::Other), 6); + assert_eq!(nested.active_keys.load(Ordering::Acquire), 1); + let local = nested.keys.borrow()["old"].upgrade().unwrap(); + assert_eq!(local.counter(Resource::Other).used(), 6); + assert!(nested.retired_keys.lock().unwrap().is_empty()); + assert!(matches!( + nested.reserve(Some(&"new".to_owned()), Resource::Other, 10), + Err(Error::Overloaded) + )); + })); + }); + quotas.shared().register(&waker); + drop(split); + assert_eq!(wake.0.load(Ordering::Relaxed), 1); + assert_eq!(charge.amount(), 6); + assert_eq!(quotas.used(Resource::Other), 6); + + quotas.policy.rejected.lock().unwrap().clear(); + let nested = quotas.clone(); + ON_WAKE.with(|slot| { + *slot.borrow_mut() = Some(Box::new(move || { + assert_eq!(nested.used(Resource::Other), 0); + let replacement = nested + .reserve(Some(&"new".to_owned()), Resource::Other, 10) + .expect("the final release must free the key slot before waking"); + assert_eq!(nested.used(Resource::Other), 10); + assert_eq!(nested.active_keys.load(Ordering::Acquire), 1); + assert_eq!( + replacement + .local + .as_ref() + .unwrap() + .counter(Resource::Other) + .used(), + 10 + ); + assert!(!nested.keys.borrow().contains_key("old")); + assert_eq!(nested.keys.borrow().len(), 1); + assert!(nested.retired_keys.lock().unwrap().is_empty()); + drop(replacement); + assert_eq!(nested.used(Resource::Other), 0); + assert_eq!(nested.active_keys.load(Ordering::Acquire), 0); + assert_eq!( + nested + .retired_keys + .lock() + .unwrap() + .iter() + .collect::>(), + vec!["new"] + ); + })); + }); + quotas.shared().register(&waker); + drop(charge); + assert_eq!(wake.0.load(Ordering::Relaxed), 2); + assert_eq!(quotas.used(Resource::Other), 0); + assert_eq!(quotas.active_keys.load(Ordering::Acquire), 0); + assert!(quotas.policy.rejected.lock().unwrap().is_empty()); + assert!(ON_WAKE.with(|slot| slot.borrow().is_none())); + } + + /// Charges and shared handles retain exact accounting across threads. + #[test] + fn charge_validation_and_thread_safe_shared_usage() { + /// Require transferable release and admission handles at compile time. + fn send_sync() {} + send_sync::>(); + send_sync::>(); + let quotas = Quotas::new(TestPolicy::new(100, 2)); + assert!(matches!( + quotas.reserve(None, Resource::Payload, 0), + Err(Error::InvalidInput) + )); + assert!(matches!( + quotas.reserve(None, Resource::Payload, usize::MAX), + Err(Error::Overloaded) + )); + let shared = quotas.shared(); + assert!(matches!( + shared.reserve(Resource::Payload, 0), + Err(Error::InvalidInput) + )); + let mut charge = shared.reserve(Resource::Payload, 100).unwrap(); + assert!(quotas.owns(&charge)); + assert!(!Quotas::new(TestPolicy::new(100, 2)).owns(&charge)); + assert!(page_alloc::Charge::covers(&charge, 100)); + assert!(!page_alloc::Charge::covers(&charge, 101)); + assert!(charge.validate(Resource::Payload, 100).is_ok()); + assert!(charge.validate(Resource::Other, 100).is_err()); + assert!(charge.split(100).is_err()); + assert!(charge.split(0).is_err()); + assert!(charge.shrink(0).is_err()); + assert!(charge.shrink(101).is_err()); + let split = charge.split(60).unwrap(); + charge.shrink(19).unwrap(); + assert_eq!(quotas.used(Resource::Payload), 79); + std::thread::spawn(move || drop(split)).join().unwrap(); + assert_eq!(shared.used(Resource::Payload), 19); + drop(charge); + let other = shared.reserve(Resource::Other, 1).unwrap(); + assert!(!page_alloc::Charge::covers(&other, 1)); + drop(other); + quotas.stop(); + assert!(shared.is_stopped()); + assert!(matches!( + shared.reserve(Resource::Payload, 1), + Err(Error::Unavailable) + )); + } + + /// Stop during the limit callback rolls back only ordinary admission. + #[test] + fn shared_reservation_rechecks_stop_after_limit_callback() { + for class in [Resource::Payload, Resource::Other, Resource::Progress] { + let mut quotas = Quotas::new(TestPolicy::new(10, 1)); + let existing = quotas.reserve(None, class, 3).unwrap(); + let gate = Arc::new((std::sync::Barrier::new(2), std::sync::Barrier::new(2))); + quotas.policy.limit_gate = Some(gate.clone()); + let shared = quotas.shared(); + let reservation = std::thread::spawn(move || shared.reserve(class, 7)); + + gate.0.wait(); + assert_eq!(quotas.used(class), 3); + quotas.stop(); + gate.1.wait(); + + let result = reservation.join().unwrap(); + if TestPolicy::allows_stopped(class) { + let charge = result.unwrap(); + assert_eq!(charge.amount(), 7); + assert_eq!(quotas.used(class), 10); + drop(charge); + } else { + assert!(matches!(result, Err(Error::Unavailable))); + } + assert_eq!(quotas.used(class), 3); + assert!(quotas.policy.rejected.lock().unwrap().is_empty()); + drop(existing); + assert_eq!(quotas.used(class), 0); + } + } + + /// Local callbacks cannot admit ordinary work after stopping the authority. + #[test] + fn local_reservation_rechecks_stop_after_callbacks() { + use std::cell::Cell; + + thread_local! { + static CLONE_HOOK: RefCell>> = RefCell::new(None); + } + + /// A transferable key whose next clone can stop the local authority. + #[derive(Eq, PartialEq, Hash)] + struct Key(u8); + + impl Clone for Key { + /// Run the one-shot callback without holding its borrow. + fn clone(&self) -> Self { + let hook = CLONE_HOOK.with(|hook| hook.borrow_mut().take()); + if let Some(hook) = hook { + hook(); + } + Self(self.0) + } + } + + /// Select which application callback requests shutdown. + #[derive(Clone, Copy, Debug, Eq, PartialEq)] + enum Hook { + Limit, + + Floor, + + Clone, + } + + /// Stop through a weak authority link without retaining an ownership cycle. + struct StopPolicy { + quotas: RefCell>>, + + hook: Cell>, + + rejections: Cell, + } + + impl StopPolicy { + /// Stop only at the selected policy callback. + fn stop_at(&self, hook: Hook) { + if self.hook.get() == Some(hook) { + self.hook.set(None); + self.quotas.borrow().upgrade().unwrap().stop(); + } + } + } + + impl Policy for StopPolicy { + type Class = Resource; + + type Key = Key; + + /// Leave enough capacity for both the existing and attempted charge. + fn limit(&self, _: Resource) -> usize { + self.stop_at(Hook::Limit); + 10 + } + + /// Exercise shutdown during keyed fair-share calculation. + fn floor(&self, _: Resource) -> usize { + self.stop_at(Hook::Floor); + 1 + } + + /// Permit one live identity and verify its eventual retirement. + fn max_keys(&self) -> usize { + 1 + } + + /// Preserve the fixture's class-specific release wake policy. + fn wakes(class: Resource) -> bool { + TestPolicy::wakes(class) + } + + /// Preserve the fixture's drain-progress exception. + fn allows_stopped(class: Resource) -> bool { + TestPolicy::allows_stopped(class) + } + + /// No page backing is allocated by this test. + fn covers(_: Resource) -> bool { + false + } + + /// Stop rejection must not be reported as quota pressure. + fn rejected(&self, _: Rejection) { + self.rejections.set(self.rejections.get() + 1); + } + } + + for hook in [Hook::Limit, Hook::Floor, Hook::Clone] { + // Unkeyed, new key, and existing key exercise distinct ownership paths. + for (keyed, live_key) in [(false, false), (true, false), (true, true)] { + if !keyed && hook != Hook::Limit { + continue; + } + for (class, completion) in [ + (Resource::Payload, false), + (Resource::Other, false), + (Resource::Progress, false), + (Resource::Payload, true), + ] { + if completion && hook == Hook::Floor { + continue; + } + let quotas = Rc::new(Quotas::new(StopPolicy { + quotas: RefCell::default(), + hook: Cell::new(None), + rejections: Cell::new(0), + })); + *quotas.policy.quotas.borrow_mut() = Rc::downgrade("as); + let key = Key(1); + let existing = quotas.reserve(live_key.then_some(&key), class, 3).unwrap(); + if hook == Hook::Clone { + let weak = Rc::downgrade("as); + CLONE_HOOK.with(|hook| { + *hook.borrow_mut() = + Some(Box::new(move || weak.upgrade().unwrap().stop())); + }); + } else { + quotas.policy.hook.set(Some(hook)); + } + let result = if completion { + quotas.reserve_completion(keyed.then_some(&key), class, 7) + } else { + quotas.reserve(keyed.then_some(&key), class, 7) + }; + assert!(quotas.is_stopped(), "callback {hook:?} must run"); + if completion || StopPolicy::allows_stopped(class) { + let charge = result.unwrap(); + assert_eq!(charge.amount(), 7); + assert_eq!(quotas.used(class), 10); + if let Some(local) = &charge.local { + assert_eq!(local.counter(class).used(), if live_key { 10 } else { 7 }); + } + drop(charge); + } else { + assert!(matches!(result, Err(Error::Unavailable)), "hook {hook:?}"); + } + assert_eq!(quotas.used(class), 3); + assert_eq!( + quotas.active_keys.load(Ordering::Acquire), + usize::from(live_key) + ); + if let Some(local) = &existing.local { + assert_eq!(local.counter(class).used(), 3); + } + assert_eq!(quotas.policy.rejections.get(), 0); + drop(existing); + assert_eq!(quotas.used(class), 0); + assert_eq!(quotas.active_keys.load(Ordering::Acquire), 0); + let replacement = quotas.reserve_completion(Some(&Key(2)), class, 10).unwrap(); + assert!(!quotas.keys.borrow().contains_key(&key)); + drop(replacement); + assert_eq!(quotas.used(class), 0); + assert_eq!(quotas.active_keys.load(Ordering::Acquire), 0); + } + } + } + } + + /// Cross-thread release and local stop notify the registered shared waiter. + #[test] + fn release_and_stop_wake_shared_waiters() { + /// Count notifications without accessing worker-local state. + struct WakeCount(AtomicUsize); + + impl std::task::Wake for WakeCount { + /// Record one notification. + fn wake(self: Arc) { + self.0.fetch_add(1, Ordering::Relaxed); + } + } + let count = Arc::new(WakeCount(AtomicUsize::new(0))); + let waker = std::task::Waker::from(count.clone()); + let quotas = Quotas::new(TestPolicy::new(1, 1)); + let shared = quotas.shared(); + shared.register(&waker); + let charge = shared.reserve(Resource::Other, 1).unwrap(); + assert!(matches!( + shared.reserve(Resource::Other, 1), + Err(Error::Overloaded) + )); + std::thread::spawn(move || drop(charge)).join().unwrap(); + assert_eq!(count.0.load(Ordering::Relaxed), 1); + shared.register(&waker); + quotas.stop(); + assert_eq!(count.0.load(Ordering::Relaxed), 2); + } +} diff --git a/cmd/racer-dataplane/flow/src/pipe.rs b/cmd/racer-dataplane/flow/src/pipe.rs new file mode 100644 index 000000000..cb15864d2 --- /dev/null +++ b/cmd/racer-dataplane/flow/src/pipe.rs @@ -0,0 +1,1728 @@ +//! Bounded, nonblocking, per-reader kernel pipes. +//! +//! Each lease owns both descriptors and its admission charge. Idle empty pipes +//! retain that charge in the worker-local pool. No descriptor or backing page is +//! recycled while a reader owns the lease. Writes copy into kernel +//! pipe pages before splice: socket acceptance is not a userspace page reuse fence. +use crate::{Charge, Error, Policy, Quotas, Result}; +use std::{ + cell::RefCell, + collections::VecDeque, + future::poll_fn, + io, + os::fd::{AsFd, AsRawFd, FromRawFd}, + rc::{Rc, Weak}, + task::{Poll, Waker}, +}; +use uring_runtime::reactor::descriptor::Descriptor; + +/// Maximum kernel buffer capacity per admitted reader. Pipe admission is in pipe +/// units, so total pipe capacity is bounded by the pipe quota * MAX_PIPE_BYTES. +/// Spliced bytes retained by sockets are subject to socket buffer limits instead. +pub const MAX_PIPE_BYTES: usize = 64 * 1024; + +/// Worker-local pipe admission and a FIFO of bounded, byte-charged waiters. +pub struct PipePool { + quotas: Rc>, + + pipe_class: P::Class, + + waiter_class: P::Class, + + waiter_limit: usize, + + waiting: Waiters, + + idle: Rc>>>, +} + +/// Shared local queue whose head alone attempts scheduled admission. +type Waiters = Rc>>>>>; + +/// Wakes the queue after resource return; holds no reactor reference. +struct Notify(Waiters); + +impl Drop for Notify { + /// Notify the oldest queued acquisition after lease resources are released. + fn drop(&mut self) { + wake_front(&self.0); + } +} + +/// Consume the head's notification before calling it without queue or entry borrows. +fn wake_front(waiters: &Waiters) { + let wake = waiters + .borrow() + .front() + .and_then(|entry| entry.borrow_mut().take()); + if let Some(wake) = wake { + wake.wake(); + } +} + +/// Own a queue registration and its admission through cancellation or completion. +struct Waiting { + queue: Waiters, + + entry: Rc>>, + + _reservation: Charge

, +} + +impl Drop for Waiting

{ + /// Remove exactly this waiter and notify its successor without self-polling. + fn drop(&mut self) { + { + let mut queue = self.queue.borrow_mut(); + queue.retain(|entry| !Rc::ptr_eq(entry, &self.entry)); + if queue.is_empty() { + // Do not retain queue storage after its admission charges leave. + *queue = VecDeque::new(); + } else if queue.capacity() > queue.len().saturating_mul(2) { + // Keep spare slots covered by live waiters' fixed allowances. + // Shrink geometrically rather than reallocating on every departure. + queue.shrink_to_fit(); + } + } + wake_front(&self.queue); + } +} + +/// Non-cloneable local ownership of both descriptors and their admission charge. +pub struct PipeLease { + resources: Option>, + + pool: Weak>>>, + + quotas: Weak>, + + _notify: Notify, +} + +/// Kernel pipe state; descriptors close before the trailing charge is released. +struct PipeResources { + read: Descriptor, + + write: Descriptor, + + capacity: usize, + + buffered: usize, + + // Declared after the descriptors so capacity is returned only after closing. + _reservation: Charge

, +} + +impl Drop for PipeLease

{ + /// Recycle only empty pipes on a live authority; close partial payloads. + fn drop(&mut self) { + // Reactor ownership keeps the lease alive through every accepted CQE. + // Never recycle canceled/partially drained payloads: close those pipes. + if let Some(resources) = self.resources.take() + && resources.buffered == 0 + && self.quotas.upgrade().is_some_and(|q| !q.is_stopped()) + && let Some(pool) = self.pool.upgrade() + { + pool.borrow_mut().push(resources); + } + // Notify drops after the pipe has been recycled or its charge released. + } +} + +impl PipePool

{ + /// Charge each live or idle pipe by one unit of `pipe_class`. Queued waits + /// charge bytes to `waiter_class`; `waiter_limit` bounds their count separately. + /// A zero waiter limit permits immediate acquisition only. + pub fn new( + quotas: Rc>, + pipe_class: P::Class, + waiter_class: P::Class, + waiter_limit: usize, + ) -> Self { + Self { + quotas, + pipe_class, + waiter_class, + waiter_limit, + waiting: Rc::default(), + idle: Rc::default(), + } + } + + /// Borrow the worker-local authority used by pipe and waiter admission. + pub fn quotas(&self) -> &Rc> { + &self.quotas + } + + /// Byte charge retained for each queue entry, excluding the caller's future. + /// The fixed allowance covers the queue slot, wake cell, and registration. + pub fn waiter_bytes(&self) -> usize { + std::mem::size_of::>() + 128 + } + + /// Empty retained pipes, still charged to the worker's fixed pipe budget. + pub fn idle_count(&self) -> usize { + self.idle.borrow().len() + } + + /// Drain idle pipes on stop without releasing resources still owned by leases. + fn check_running(&self) -> Result<()> { + if self.quotas.is_stopped() { + let idle = std::mem::take(&mut *self.idle.borrow_mut()); + // Charge destruction may reenter the pool; release the borrow first. + drop(idle); + return Err(Error::Unavailable); + } + Ok(()) + } + + /// FIFO scheduling above immediate raw admission. At most `waiter_limit` wait + /// without pipes or new page acquisitions; each entry charges context bytes + /// for its guard, queue slot, wake cell, and cancellation registration. + /// `check` runs before acquisition and on every queued poll. `subscribe` runs + /// only after queue capacity and its byte charge are secured, never on the + /// immediate path. Its returned closure registers cancellation notification + /// with each poll's waker and owns the registration until this wait ends. + /// The caller must arrange polls for deadlines, quota stop, or charges held + /// outside this pool; pipe returns and queue removal wake the FIFO head. + /// No timer, polling loop, or self-wake is created here. + pub async fn acquire_wait( + &self, + mut check: Check, + subscribe: Subscribe, + ) -> std::result::Result, E> + where + E: From, + Check: FnMut() -> std::result::Result<(), E>, + Subscribe: FnOnce() -> std::result::Result, + Register: FnMut(&Waker), + { + check()?; + if self.waiting.borrow().is_empty() { + match self.acquire() { + Err(Error::Overloaded) => {} + result => return result.map_err(E::from), + } + } + self.check_running()?; + if self.waiting.borrow().len() >= self.waiter_limit { + return Err(Error::Overloaded.into()); + } + let reservation = self + .quotas + .reserve(None, self.waiter_class, self.waiter_bytes())?; + // Reservation policy can synchronously admit another waiter. + if self.waiting.borrow().len() >= self.waiter_limit { + return Err(Error::Overloaded.into()); + } + let entry = Rc::new(RefCell::new(None)); + self.waiting.borrow_mut().push_back(entry.clone()); + let waiting = Waiting { + queue: self.waiting.clone(), + entry, + _reservation: reservation, + }; + // Publish before reentry; Waiting rolls back subscription errors or panics. + let mut register = subscribe()?; + poll_fn(|cx| { + register(cx.waker()); + check()?; + self.check_running()?; + // Clone and drop may reenter; publish before retiring the old waker. + let wake = cx.waker().clone(); + let old = waiting.entry.borrow_mut().replace(wake); + drop(old); + if self + .waiting + .borrow() + .front() + .is_some_and(|entry| Rc::ptr_eq(entry, &waiting.entry)) + { + match self.acquire() { + Err(Error::Overloaded) => {} + result => return Poll::Ready(result.map_err(E::from)), + } + } + Poll::Pending + }) + .await + } + + /// Reserve before creating descriptors. Exhaustion never waits for a reader. + pub fn acquire(&self) -> Result> { + self.check_running()?; + if let Some(resources) = self.idle.borrow_mut().pop() { + return Ok(self.lease(resources)); + } + let reservation = self.quotas.reserve(None, self.pipe_class, 1)?; + reservation.validate(self.pipe_class, 1)?; + Ok(self.lease(PipeResources::new(reservation)?)) + } + + /// Attach local recycling and notification to uniquely owned resources. + fn lease(&self, resources: PipeResources

) -> PipeLease

{ + PipeLease { + resources: Some(resources), + pool: Rc::downgrade(&self.idle), + quotas: Rc::downgrade(&self.quotas), + _notify: Notify(self.waiting.clone()), + } + } +} + +impl PipeResources

{ + /// Create bounded descriptors, closing them before admission on any failure. + fn new(reservation: Charge

) -> Result { + #[cfg(feature = "simulation")] + if let Some(sim) = uring_runtime::reactor::simulation::Simulation::current() { + let (read, write) = sim.pipe(MAX_PIPE_BYTES); + return Ok(Self { + read, + write, + capacity: MAX_PIPE_BYTES, + buffered: 0, + _reservation: reservation, + }); + } + let mut fds = [-1; 2]; + // SAFETY: pipe2 initializes exactly two descriptors on success. + if unsafe { libc::pipe2(fds.as_mut_ptr(), libc::O_NONBLOCK | libc::O_CLOEXEC) } < 0 { + return Err(Error::Io); + } + // SAFETY: both descriptors were newly created and have unique owners. + let read = unsafe { Descriptor::from_raw_fd(fds[0]) }; + let write = unsafe { Descriptor::from_raw_fd(fds[1]) }; + // SAFETY: fcntl operates on a live descriptor and requires no pointer. + let mut capacity = unsafe { libc::fcntl(write.as_raw_fd(), libc::F_GETPIPE_SZ) }; + if capacity < 0 { + return Err(Error::Io); + } + if capacity as usize != MAX_PIPE_BYTES { + // Request one bounded chunk once at creation, before pooling. Under + // UID pipe pressure growth may fail; retain the smaller actual size. + // SAFETY: the empty pipe can be resized without borrowing user memory. + let resized = unsafe { + libc::fcntl(write.as_raw_fd(), libc::F_SETPIPE_SZ, MAX_PIPE_BYTES as i32) + }; + if resized > 0 { + capacity = resized; + } + } + if capacity <= 0 || capacity as usize > MAX_PIPE_BYTES { + return Err(Error::Io); + } + Ok(Self { + read, + write, + capacity: capacity as usize, + buffered: 0, + _reservation: reservation, + }) + } +} + +impl PipeLease

{ + /// Transit benefits from a full bounded chunk even when the UID's default + /// pipe size has shrunk. Failure to grow is harmless: use the actual capacity. + pub fn prepare_transit(&mut self) { + let pipe = self.resources.as_mut().unwrap(); + #[cfg(feature = "simulation")] + if pipe.write.as_sim().is_some() { + return; + } + if pipe.capacity < MAX_PIPE_BYTES && pipe.buffered == 0 { + // SAFETY: live, empty pipe, bounded integer capacity, no user pointer. + let capacity = unsafe { + libc::fcntl( + pipe.write.as_raw_fd(), + libc::F_SETPIPE_SZ, + MAX_PIPE_BYTES as i32, + ) + }; + if capacity > 0 { + pipe.capacity = capacity as usize; + } + } + } + + /// Receive opaque socket pages directly into an empty bounded pipe. No user + /// buffer is borrowed or retained by this synchronous nonblocking syscall. + /// The descriptor is validated as a nonblocking stream socket. Callers must + /// not concurrently clear O_NONBLOCK through a duplicate descriptor. + pub fn try_splice_from(&mut self, socket: &Descriptor, count: usize) -> io::Result { + #[cfg(feature = "simulation")] + if socket.as_sim().is_some() || self.resources.as_ref().unwrap().write.as_sim().is_some() { + return Err(io::Error::from_raw_os_error(libc::EOPNOTSUPP)); + } + validate_socket(socket)?; + let pipe = self.resources.as_mut().unwrap(); + let count = count.min(pipe.capacity - pipe.buffered); + // SAFETY: socket is a validated nonblocking stream; this pipe owns + // both ends, offsets are null, and no userspace pointer enters the kernel. + let received = syscall_count(unsafe { + libc::splice( + socket.as_raw_fd(), + std::ptr::null_mut(), + pipe.write.as_raw_fd(), + std::ptr::null_mut(), + count, + libc::SPLICE_F_NONBLOCK | libc::SPLICE_F_MOVE, + ) + })?; + pipe.buffered += received; + Ok(received) + } + + /// Return actual kernel capacity, which may be below the requested ceiling. + pub fn capacity(&self) -> usize { + self.resources.as_ref().unwrap().capacity + } + + /// Return the exact suffix still retained in this pipe. + pub fn buffered(&self) -> usize { + self.resources.as_ref().unwrap().buffered + } + + /// Copy at most the available capacity. WouldBlock and Interrupted are exposed + /// to the caller; this method never waits or retains a borrowed buffer. + pub fn try_write(&mut self, bytes: &[u8]) -> io::Result { + let pipe = self.resources.as_mut().unwrap(); + #[cfg(feature = "simulation")] + if let Some(handle) = pipe.write.as_sim() { + let written = handle.pipe_write(bytes)?; + pipe.buffered += written; + return Ok(written); + } + // SAFETY: the initialized slice stays live for this nonblocking syscall. + // The read end is owned by this lease, so this cannot generate SIGPIPE. + let written = unsafe { + libc::write( + pipe.write.as_raw_fd(), + bytes.as_ptr().cast(), + bytes.len().min(pipe.capacity), + ) + }; + let written = syscall_count(written)?; + pipe.buffered += written; + Ok(written) + } + + /// Read currently buffered bytes, or return WouldBlock for an empty pipe. + pub fn try_read(&mut self, bytes: &mut [u8]) -> io::Result { + let pipe = self.resources.as_mut().unwrap(); + #[cfg(feature = "simulation")] + if let Some(handle) = pipe.read.as_sim() { + let read = handle.pipe_read(bytes)?; + pipe.buffered -= read; + return Ok(read); + } + // SAFETY: the destination is exclusively borrowed until read returns. + let read = unsafe { + libc::read( + pipe.read.as_raw_fd(), + bytes.as_mut_ptr().cast(), + bytes.len(), + ) + }; + let read = syscall_count(read)?; + pipe.buffered -= read; + Ok(read) + } + + /// Splice a bounded suffix using production or simulated runtime descriptors. + pub fn try_splice_descriptor( + &mut self, + socket: &Descriptor, + count: usize, + ) -> io::Result { + #[cfg(feature = "simulation")] + { + let pipe = self.resources.as_mut().unwrap(); + match (pipe.read.as_sim(), socket.as_sim()) { + (Some(read), Some(socket)) => { + let sent = read.splice(socket, count.min(pipe.buffered))?; + pipe.buffered -= sent; + return Ok(sent); + } + (None, None) => {} + _ => return Err(io::Error::from_raw_os_error(libc::EOPNOTSUPP)), + } + } + self.try_splice_to(socket, count) + } + + /// Transfer copied kernel pipe bytes to a nonblocking stream socket. The + /// caller retains both owners through this synchronous syscall. Unsupported + /// splice errors leave the bytes in the pipe for an independent copy fallback. + /// Callers must not concurrently clear O_NONBLOCK through a duplicate FD. + pub fn try_splice_to(&mut self, socket: &impl AsFd, count: usize) -> io::Result { + validate_socket(socket)?; + self.splice_to_fd(socket.as_fd().as_raw_fd(), count) + } + + /// Drain to a stream socket. Like `try_splice_to`, this safe API validates + /// O_NONBLOCK and SO_TYPE; a Descriptor alone does not prove either property. + /// Callers must not concurrently clear O_NONBLOCK through a duplicate FD. + pub fn try_splice_connection(&mut self, socket: &Descriptor) -> io::Result { + self.try_splice_descriptor(socket, self.buffered()) + } + + /// Drain a validated socket while masking only this thread's generated SIGPIPE. + fn splice_to_fd(&mut self, fd: libc::c_int, count: usize) -> io::Result { + let pipe = self.resources.as_mut().unwrap(); + if count == 0 || pipe.buffered == 0 { + return Ok(0); + } + // Unlike send, splice has no MSG_NOSIGNAL. Mask only on this worker and + // only across the syscall, consuming our own EPIPE signal before restore. + let signal = SigpipeGuard::block()?; + // SAFETY: owned live FDs, null offsets for pipe/socket, no userspace page + // pointers, and both ends are nonblocking. No vmsplice/GIFT is involved. + let result = syscall_count(unsafe { + libc::splice( + pipe.read.as_raw_fd(), + std::ptr::null_mut(), + fd, + std::ptr::null_mut(), + count.min(pipe.buffered), + libc::SPLICE_F_NONBLOCK, + ) + }); + if result + .as_ref() + .is_err_and(|e| e.raw_os_error() == Some(libc::EPIPE)) + { + signal.consume_generated(); + } + drop(signal); + if let Ok(sent) = result { + pipe.buffered -= sent; + } + result + } +} + +/// Require a live nonblocking stream socket before synchronous splice. +fn validate_socket(socket: &impl AsFd) -> io::Result<()> { + let fd = socket.as_fd().as_raw_fd(); + // SPLICE_F_NONBLOCK controls only the pipe side. O_NONBLOCK alone does not + // prevent regular-file I/O from blocking, so require a stream socket too. + // SAFETY: AsFd borrows a live descriptor for this synchronous query. + let flags = unsafe { libc::fcntl(fd, libc::F_GETFL) }; + if flags < 0 { + return Err(io::Error::last_os_error()); + } + if flags & libc::O_NONBLOCK == 0 { + return Err(io::Error::from_raw_os_error(libc::EINVAL)); + } + let mut kind: libc::c_int = 0; + let mut length = std::mem::size_of_val(&kind) as libc::socklen_t; + // SAFETY: both output pointers reference correctly sized local values. + if unsafe { + libc::getsockopt( + fd, + libc::SOL_SOCKET, + libc::SO_TYPE, + (&mut kind as *mut libc::c_int).cast(), + &mut length, + ) + } < 0 + { + return Err(io::Error::last_os_error()); + } + if kind != libc::SOCK_STREAM { + return Err(io::Error::from_raw_os_error(libc::EINVAL)); + } + Ok(()) +} + +/// Temporarily mask SIGPIPE while preserving the thread's prior signal state. +struct SigpipeGuard { + previous: libc::sigset_t, + + set: libc::sigset_t, + + was_pending: bool, +} + +impl SigpipeGuard { + /// Save the thread mask, block SIGPIPE, and remember preexisting signals. + fn block() -> io::Result { + // SAFETY: all signal set pointers refer to initialized local storage. + unsafe { + let mut previous = std::mem::zeroed(); + let mut set = std::mem::zeroed(); + libc::sigemptyset(&mut set); + libc::sigaddset(&mut set, libc::SIGPIPE); + let error = libc::pthread_sigmask(libc::SIG_BLOCK, &set, &mut previous); + if error != 0 { + return Err(io::Error::from_raw_os_error(error)); + } + // Only a successfully installed mask creates a restoring owner. + let mut guard = Self { + previous, + set, + was_pending: true, + }; + let mut pending = std::mem::zeroed(); + if libc::sigpending(&mut pending) < 0 { + return Err(io::Error::last_os_error()); + } + guard.was_pending = libc::sigismember(&pending, libc::SIGPIPE) == 1; + Ok(guard) + } + } + + /// Consume only a newly generated signal without waiting. + fn consume_generated(&self) { + if self.was_pending { + return; + } + let timeout = libc::timespec { + tv_sec: 0, + tv_nsec: 0, + }; + // SAFETY: zero timeout never waits; only our thread's blocked SIGPIPE is + // consumed. Preserve a signal that was pending before entering the guard. + while unsafe { libc::sigtimedwait(&self.set, std::ptr::null_mut(), &timeout) } < 0 { + if io::Error::last_os_error().raw_os_error() != Some(libc::EINTR) { + break; + } + } + } +} + +impl Drop for SigpipeGuard { + /// Restore the exact mask saved after successful installation. + fn drop(&mut self) { + // SAFETY: restore the exact thread mask saved by successful pthread_sigmask. + unsafe { libc::pthread_sigmask(libc::SIG_SETMASK, &self.previous, std::ptr::null_mut()) }; + } +} + +/// Convert a syscall byte count while preserving its OS error. +fn syscall_count(value: isize) -> io::Result { + if value < 0 { + Err(io::Error::last_os_error()) + } else { + Ok(value as usize) + } +} + +/// Unsupported splice leaves buffered bytes available to a copy fallback. +pub fn splice_unsupported(error: &io::Error) -> bool { + matches!( + error.raw_os_error(), + Some(libc::EINVAL | libc::ENOSYS | libc::EOPNOTSUPP) + ) +} + +/// Kernel behavior, admission, cancellation, and simulation ownership contracts. +#[cfg(test)] +mod tests { + use super::*; + use std::{ + future::Future, + sync::{ + Arc, + atomic::{AtomicUsize, Ordering}, + }, + task::{Context, Wake}, + }; + + /// Independent pipe, waiter-context, and payload fixture resources. + #[derive(Clone, Copy)] + enum ResourceClass { + Pipe, + + RequestContext, + + Plaintext, + } + + impl crate::Class for ResourceClass { + const COUNT: usize = 3; + + /// Select the fixture's stable per-class counter. + fn index(self) -> usize { + self as usize + } + } + + /// Independent pipe and context limits with pipe-release notifications. + struct TestPolicy { + pipes: usize, + + context: usize, + + on_context_limit: RefCell>>, + } + + impl Policy for TestPolicy { + type Class = ResourceClass; + + type Key = (); + + /// Give pipes their count ceiling and other classes a byte ceiling. + fn limit(&self, class: ResourceClass) -> usize { + if matches!(class, ResourceClass::RequestContext) { + let callback = self.on_context_limit.borrow_mut().take(); + if let Some(callback) = callback { + callback(); + } + } + match class { + ResourceClass::Pipe => self.pipes, + _ => self.context, + } + } + + /// All pipe fixture admission is unkeyed. + fn max_keys(&self) -> usize { + 0 + } + + /// Pipe release can notify an explicitly registered quota waiter. + fn wakes(class: ResourceClass) -> bool { + matches!(class, ResourceClass::Pipe) + } + + /// Pipe counts do not authorize userspace page backing. + fn covers(_: ResourceClass) -> bool { + false + } + + /// Rejection facts are not needed by these pipe assertions. + fn rejected(&self, _: crate::Rejection) {} + } + + /// Build a local authority with ample waiter-context capacity. + fn admission(pipes: usize) -> Rc> { + Rc::new(Quotas::new(TestPolicy { + pipes, + context: 32 * 1024 * 1024, + on_context_limit: RefCell::default(), + })) + } + + /// Build an eight-waiter pool using distinct pipe and context classes. + fn new_pool(quotas: Rc>) -> PipePool { + PipePool::new( + quotas, + ResourceClass::Pipe, + ResourceClass::RequestContext, + 8, + ) + } + + /// Create an acquisition with no cancellation or deadline source. + fn acquire_wait( + pool: &PipePool, + ) -> std::pin::Pin>> + '_>> { + Box::pin(pool.acquire_wait(|| Ok::<_, Error>(()), || Ok(|_: &Waker| {}))) + } + + /// Count progress notifications without requiring an executor. + #[derive(Default)] + struct WakeCounter(AtomicUsize); + + impl Wake for WakeCounter { + /// Count an owned wake notification. + fn wake(self: Arc) { + self.0.fetch_add(1, Ordering::Relaxed); + } + + /// Count a borrowed wake notification. + fn wake_by_ref(self: &Arc) { + self.0.fetch_add(1, Ordering::Relaxed); + } + } + + impl WakeCounter { + /// Read the number of observed progress notifications. + fn count(&self) -> usize { + self.0.load(Ordering::Relaxed) + } + } + + /// Waker callbacks use thread-local hooks without sharing worker-local state. + mod waker_callbacks { + use super::*; + use std::task::{RawWaker, RawWakerVTable}; + + type Hook = Box; + + thread_local! { + static HOOK: RefCell> = const { RefCell::new(None) }; + } + + /// Remove the hook before calling it so nested callbacks are harmless. + fn invoke(event: &str) { + let hook = HOOK.with(|slot| slot.borrow_mut().take()); + if let Some(hook) = hook { + hook(event); + } + } + + /// Install one callback on this test thread. + pub fn on_callback(hook: impl FnOnce(&str) + 'static) { + HOOK.with(|slot| *slot.borrow_mut() = Some(Box::new(hook))); + } + + /// Drop an unused hook before thread-local teardown, without a slot borrow. + pub fn clear() { + drop(HOOK.with(|slot| slot.borrow_mut().take())); + } + + /// No data pointer is owned; callbacks only access the calling thread. + fn raw() -> RawWaker { + RawWaker::new( + std::ptr::null(), + &RawWakerVTable::new( + |_| { + invoke("clone"); + raw() + }, + |_| invoke("wake"), + |_| invoke("wake_by_ref"), + |_| invoke("drop"), + ), + ) + } + + /// Build a transferable waker with no shared mutable data. + pub fn waker() -> Waker { + // SAFETY: the vtable never dereferences data and owns no allocation. + // All mutable hooks are thread-local, including on other threads. + unsafe { Waker::from_raw(raw()) } + } + } + + /// Consume the head before callbacks can borrow or notify the same queue. + #[test] + fn review_regression_pipe_waker_front_reentry() { + let entry = Rc::new(RefCell::new(Some(waker_callbacks::waker()))); + let queue: Waiters = Rc::new(RefCell::new(VecDeque::from([entry.clone()]))); + let callback_queue = queue.clone(); + let callback_entry = entry.clone(); + let called = Rc::new(std::cell::Cell::new(false)); + let callback_called = called.clone(); + waker_callbacks::on_callback(move |event| { + let queue = callback_queue.borrow_mut(); + let entry = callback_entry.borrow_mut(); + assert_eq!(event, "wake", "notification must not clone its target"); + assert!(entry.is_none()); + assert_eq!(queue.len(), 1); + drop(entry); + drop(queue); + wake_front(&callback_queue); + callback_called.set(true); + }); + wake_front(&queue); + assert!(called.get()); + assert!(entry.borrow().is_none()); + assert_eq!(queue.borrow().len(), 1); + wake_front(&queue); + } + + /// Dropping a replaced registration may synchronously notify the new target. + #[test] + fn review_regression_pipe_waker_replacement_reentry() { + let pool = new_pool(admission(1)); + let held = pool.acquire().unwrap(); + let mut wait = acquire_wait(&pool); + let old = waker_callbacks::waker(); + assert!( + wait.as_mut() + .poll(&mut Context::from_waker(&old)) + .is_pending() + ); + let queue = pool.waiting.clone(); + waker_callbacks::on_callback(move |event| { + assert_eq!(event, "drop"); + wake_front(&queue); + }); + let counter = Arc::new(WakeCounter::default()); + let new = Waker::from(counter.clone()); + let mut cx = Context::from_waker(&new); + assert!(wait.as_mut().poll(&mut cx).is_pending()); + assert_eq!(counter.count(), 1); + assert!(wait.as_mut().poll(&mut cx).is_pending()); + drop(held); + assert_eq!(counter.count(), 2); + assert!(matches!(wait.as_mut().poll(&mut cx), Poll::Ready(Ok(_)))); + assert!(pool.waiting.borrow().is_empty()); + } + + /// Consumed notifications rearm on each poll and preserve FIFO progress. + #[test] + fn review_regression_pipe_waker_rearms_and_advances_fifo() { + let quotas = admission(1); + let pool = new_pool(quotas.clone()); + let held = pool.acquire().unwrap(); + let registrations = std::cell::Cell::new(0); + let mut first = Box::pin(pool.acquire_wait( + || Ok::<_, Error>(()), + || Ok(|_: &Waker| registrations.set(registrations.get() + 1)), + )); + let mut second = acquire_wait(&pool); + let count = Arc::new(WakeCounter::default()); + let waker = Waker::from(count.clone()); + let mut cx = Context::from_waker(&waker); + assert!(first.as_mut().poll(&mut cx).is_pending()); + assert!(second.as_mut().poll(&mut cx).is_pending()); + wake_front(&pool.waiting); + wake_front(&pool.waiting); + assert_eq!(count.count(), 1); + let queue = pool.waiting.clone(); + waker_callbacks::on_callback(move |event| { + assert_eq!(event, "clone"); + wake_front(&queue); + }); + let reentrant = waker_callbacks::waker(); + assert!( + first + .as_mut() + .poll(&mut Context::from_waker(&reentrant)) + .is_pending() + ); + assert!(first.as_mut().poll(&mut cx).is_pending()); + assert_eq!(registrations.get(), 3); + assert_eq!(count.count(), 1, "pending polls must not self-wake"); + drop(held); + assert_eq!(count.count(), 2); + assert!(second.as_mut().poll(&mut cx).is_pending()); + let Poll::Ready(Ok(first_lease)) = first.as_mut().poll(&mut cx) else { + panic!("head did not acquire returned pipe"); + }; + assert_eq!(registrations.get(), 4); + assert_eq!(count.count(), 3, "head completion must notify successor"); + assert!(second.as_mut().poll(&mut cx).is_pending()); + drop(first_lease); + assert_eq!(count.count(), 4); + assert!(matches!(second.as_mut().poll(&mut cx), Poll::Ready(Ok(_)))); + assert!(pool.waiting.borrow().is_empty()); + assert_eq!(quotas.used(ResourceClass::RequestContext), 0); + } + + /// Distinguish application gate rejection from flow-control exhaustion. + #[derive(Debug, PartialEq)] + enum GateError { + Rejected, + + Flow(Error), + } + + impl From for GateError { + /// Preserve the underlying flow-control failure. + fn from(error: Error) -> Self { + Self::Flow(error) + } + } + + /// Unsupported operation errors never include backpressure or disconnects. + #[test] + fn unsupported_splice_is_distinct_from_backpressure_and_disconnect() { + for code in [libc::EINVAL, libc::ENOSYS, libc::EOPNOTSUPP] { + assert!(splice_unsupported(&io::Error::from_raw_os_error(code))); + } + for code in [libc::EAGAIN, libc::EINTR, libc::EPIPE, libc::ECONNRESET] { + assert!(!splice_unsupported(&io::Error::from_raw_os_error(code))); + } + assert!(!splice_unsupported(&io::Error::other("custom"))); + } + + /// Immediate acquisition skips subscription and failed waits roll back fully. + #[test] + fn lazy_subscription_and_failure_rollback() { + use std::cell::Cell; + let quotas = admission(1); + let pool = new_pool(quotas.clone()); + let subscribed = Cell::new(0); + let subscribe = || { + subscribed.set(subscribed.get() + 1); + assert_eq!(pool.waiting.borrow().len(), 1); + Err::(GateError::Rejected) + }; + let mut cx = Context::from_waker(Waker::noop()); + let mut immediate = Box::pin(pool.acquire_wait(|| Ok(()), subscribe)); + let Poll::Ready(Ok(held)) = immediate.as_mut().poll(&mut cx) else { + panic!("immediate acquisition failed") + }; + assert_eq!(subscribed.get(), 0); + assert_eq!(quotas.used(ResourceClass::RequestContext), 0); + let mut wait = Box::pin(pool.acquire_wait(|| Ok(()), subscribe)); + assert!(matches!( + wait.as_mut().poll(&mut cx), + Poll::Ready(Err(GateError::Rejected)) + )); + assert_eq!(subscribed.get(), 1); + assert!(pool.waiting.borrow().is_empty()); + assert_eq!(quotas.used(ResourceClass::RequestContext), 0); + assert_eq!(quotas.used(ResourceClass::Pipe), 1); + drop(held); + + let mut rejected = Box::pin(pool.acquire_wait(|| Err(GateError::Rejected), subscribe)); + assert!(matches!( + rejected.as_mut().poll(&mut cx), + Poll::Ready(Err(GateError::Rejected)) + )); + assert_eq!(subscribed.get(), 1); + assert_eq!(pool.idle_count(), 1); + + for (context, waiter_limit) in [(0, 8), (usize::MAX, 0)] { + let quotas = Rc::new(Quotas::new(TestPolicy { + pipes: 1, + context, + on_context_limit: RefCell::default(), + })); + let pool = PipePool::new( + quotas.clone(), + ResourceClass::Pipe, + ResourceClass::RequestContext, + waiter_limit, + ); + let _held = pool.acquire().unwrap(); + let mut wait = Box::pin(pool.acquire_wait(|| Ok(()), subscribe)); + assert!(matches!( + wait.as_mut().poll(&mut cx), + Poll::Ready(Err(GateError::Flow(Error::Overloaded))) + )); + assert_eq!(subscribed.get(), 1, "rejected queue must not subscribe"); + assert_eq!(quotas.used(ResourceClass::RequestContext), 0); + assert!(pool.waiting.borrow().is_empty()); + } + } + + /// Reservation callbacks cannot let an outer wait overfill a reentered queue. + #[test] + fn review_regression_waiter_limit_after_reserve_reentry() { + let quotas = admission(1); + let pool = Rc::new(PipePool::new( + quotas.clone(), + ResourceClass::Pipe, + ResourceClass::RequestContext, + 1, + )); + let held = pool.acquire().unwrap(); + let nested_pool = pool.clone(); + let nested = Rc::new(RefCell::new(Box::pin(async move { + acquire_wait(&nested_pool).await + }))); + let callback_wait = nested.clone(); + *quotas.policy.on_context_limit.borrow_mut() = Some(Box::new(move || { + let mut cx = Context::from_waker(Waker::noop()); + assert!( + callback_wait + .borrow_mut() + .as_mut() + .poll(&mut cx) + .is_pending() + ); + })); + let mut outer = Box::pin(pool.acquire_wait( + || Ok::<_, Error>(()), + || -> Result { + panic!("overloaded outer wait must not subscribe"); + }, + )); + let mut cx = Context::from_waker(Waker::noop()); + assert!(matches!( + outer.as_mut().poll(&mut cx), + Poll::Ready(Err(Error::Overloaded)) + )); + assert_eq!(pool.waiting.borrow().len(), 1); + assert_eq!( + quotas.used(ResourceClass::RequestContext), + pool.waiter_bytes() + ); + drop(held); + assert!(matches!( + nested.borrow_mut().as_mut().poll(&mut cx), + Poll::Ready(Ok(_)) + )); + assert!(pool.waiting.borrow().is_empty()); + assert_eq!(quotas.used(ResourceClass::RequestContext), 0); + } + + /// Subscription sees its own queue slot before it tries a nested acquisition. + #[test] + fn review_regression_waiter_limit_during_subscribe_reentry() { + let quotas = admission(1); + let pool = PipePool::new( + quotas.clone(), + ResourceClass::Pipe, + ResourceClass::RequestContext, + 1, + ); + let held = pool.acquire().unwrap(); + let mut nested = acquire_wait(&pool); + let mut outer = Box::pin(pool.acquire_wait( + || Ok::<_, Error>(()), + || { + let mut cx = Context::from_waker(Waker::noop()); + assert!(matches!( + nested.as_mut().poll(&mut cx), + Poll::Ready(Err(Error::Overloaded)) + )); + assert_eq!(pool.waiting.borrow().len(), 1); + assert_eq!( + quotas.used(ResourceClass::RequestContext), + pool.waiter_bytes() + ); + Ok(|_: &Waker| {}) + }, + )); + let mut cx = Context::from_waker(Waker::noop()); + assert!(outer.as_mut().poll(&mut cx).is_pending()); + assert_eq!(pool.waiting.borrow().len(), 1); + drop(held); + assert!(matches!(outer.as_mut().poll(&mut cx), Poll::Ready(Ok(_)))); + assert!(pool.waiting.borrow().is_empty()); + assert_eq!(quotas.used(ResourceClass::RequestContext), 0); + } + + /// Unwinding a subscription removes its published slot and releases its charge. + #[test] + fn review_regression_subscription_panic_rolls_back_waiting() { + let quotas = admission(1); + let pool = new_pool(quotas.clone()); + let _held = pool.acquire().unwrap(); + let published = std::cell::Cell::new(false); + let mut wait = Box::pin(pool.acquire_wait( + || Ok::<_, Error>(()), + || -> Result { + published.set(pool.waiting.borrow().len() == 1); + panic!("subscription failed"); + }, + )); + let result = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { + wait.as_mut().poll(&mut Context::from_waker(Waker::noop())) + })); + assert!(result.is_err()); + assert!(published.get()); + assert!(pool.waiting.borrow().is_empty()); + assert_eq!(pool.waiting.borrow().capacity(), 0); + assert_eq!(quotas.used(ResourceClass::RequestContext), 0); + let mut replacement = acquire_wait(&pool); + assert!( + replacement + .as_mut() + .poll(&mut Context::from_waker(Waker::noop())) + .is_pending() + ); + } + + /// Gate failure, stop, and abandonment release registration and exact charges. + #[test] + fn gate_stop_and_abandonment_drop_registration_and_exact_charge() { + use std::cell::Cell; + /// Count a caller-owned cancellation registration until its closure drops. + struct Registration(Rc>); + + impl Drop for Registration { + /// Release exactly this registration's live count. + fn drop(&mut self) { + self.0.set(self.0.get() - 1); + } + } + for failure in 0..3 { + let quotas = admission(1); + let pool = new_pool(quotas.clone()); + let held = pool.acquire().unwrap(); + let reject = Cell::new(false); + let registrations = Rc::new(Cell::new(0)); + let registered = Cell::new(0); + let mut wait = Box::pin(pool.acquire_wait( + || { + if reject.get() { + Err(GateError::Rejected) + } else { + Ok(()) + } + }, + || { + registrations.set(registrations.get() + 1); + let registration = Registration(registrations.clone()); + let registered = ®istered; + Ok(move |_: &Waker| { + let _keep = ®istration; + registered.set(registered.get() + 1); + }) + }, + )); + let counter = Arc::new(WakeCounter::default()); + let waker = Waker::from(counter.clone()); + let mut cx = Context::from_waker(&waker); + assert!(wait.as_mut().poll(&mut cx).is_pending()); + assert!(wait.as_mut().poll(&mut cx).is_pending()); + assert_eq!(registered.get(), 2); + assert_eq!(registrations.get(), 1); + assert_eq!(counter.count(), 0); + assert_eq!( + quotas.used(ResourceClass::RequestContext), + pool.waiter_bytes() + ); + let mut second = acquire_wait(&pool); + assert!(second.as_mut().poll(&mut cx).is_pending()); + match failure { + 0 => { + reject.set(true); + assert!(matches!( + wait.as_mut().poll(&mut cx), + Poll::Ready(Err(GateError::Rejected)) + )); + } + 1 => { + quotas.stop(); + assert!(matches!( + wait.as_mut().poll(&mut cx), + Poll::Ready(Err(GateError::Flow(Error::Unavailable))) + )); + } + _ => {} + } + drop(wait); + assert_eq!(registrations.get(), 0); + assert_eq!( + quotas.used(ResourceClass::RequestContext), + pool.waiter_bytes() + ); + assert_eq!(pool.waiting.borrow().len(), 1); + assert!(counter.count() > 0, "head removal must wake successor"); + drop(held); + if failure == 1 { + assert!(matches!( + second.as_mut().poll(&mut cx), + Poll::Ready(Err(Error::Unavailable)) + )); + assert_eq!(quotas.used(ResourceClass::Pipe), 0); + } else { + assert!(matches!(second.as_mut().poll(&mut cx), Poll::Ready(Ok(_)))); + } + drop(second); + assert_eq!(quotas.used(ResourceClass::RequestContext), 0); + assert!(pool.waiting.borrow().is_empty()); + } + } + + /// Both descriptor splice directions reject blocking sockets without data loss. + #[test] + fn public_descriptor_splice_validates_both_directions() { + use std::{ + io::{Read, Write}, + os::unix::net::UnixStream, + }; + let pool = new_pool(admission(1)); + let mut pipe = pool.acquire().unwrap(); + let (socket, mut peer) = UnixStream::pair().unwrap(); + let socket = Descriptor::from(std::os::fd::OwnedFd::from(socket)); + pipe.try_write(b"out").unwrap(); + assert_eq!( + pipe.try_splice_connection(&socket) + .unwrap_err() + .raw_os_error(), + Some(libc::EINVAL) + ); + assert_eq!( + pipe.try_splice_from(&socket, 1).unwrap_err().raw_os_error(), + Some(libc::EINVAL) + ); + assert_eq!(pipe.buffered(), 3); + socket.set_nonblocking().unwrap(); + assert_eq!(pipe.try_splice_connection(&socket).unwrap(), 3); + let mut bytes = [0; 3]; + peer.read_exact(&mut bytes).unwrap(); + assert_eq!(&bytes, b"out"); + peer.write_all(b"in!").unwrap(); + pipe.prepare_transit(); + assert_eq!(pipe.try_splice_from(&socket, 3).unwrap(), 3); + assert_eq!(pipe.try_read(&mut bytes).unwrap(), 3); + assert_eq!(&bytes, b"in!"); + } + + /// Simulated descriptors preserve byte counts, pooling, and partial close. + #[cfg(feature = "simulation")] + #[test] + fn simulation_uses_runtime_descriptors_and_preserves_accounting() { + let sim = uring_runtime::reactor::simulation::Simulation::new(); + let _environment = sim.enter(); + let quotas = admission(1); + let pool = new_pool(quotas.clone()); + let mut pipe = pool.acquire().unwrap(); + assert!(pipe.resources.as_ref().unwrap().read.as_sim().is_some()); + assert_eq!(pipe.capacity(), MAX_PIPE_BYTES); + pipe.prepare_transit(); + assert_eq!(pipe.try_write(b"simulated").unwrap(), 9); + let (socket, peer) = sim.socket_pair(); + assert_eq!( + pipe.try_splice_from(&socket, 1).unwrap_err().raw_os_error(), + Some(libc::EOPNOTSUPP) + ); + assert_eq!(pipe.try_splice_descriptor(&socket, 3).unwrap(), 3); + assert_eq!(pipe.buffered(), 6); + let mut bytes = [0; 9]; + assert_eq!(peer.try_recv(&mut bytes).unwrap(), 3); + assert_eq!(&bytes[..3], b"sim"); + assert_eq!(pipe.try_read(&mut bytes).unwrap(), 6); + assert_eq!(&bytes[..6], b"ulated"); + drop(pipe); + assert_eq!(pool.idle_count(), 1); + assert_eq!(quotas.used(ResourceClass::Pipe), 1); + let mut pipe = pool.acquire().unwrap(); + pipe.try_write(b"discard").unwrap(); + drop(pipe); + assert_eq!(pool.idle_count(), 0); + assert_eq!(quotas.used(ResourceClass::Pipe), 0); + } + + /// Mixed descriptor backends reject splice without consuming buffered bytes. + #[cfg(feature = "simulation")] + #[test] + fn mixed_splice_backends_preserve_buffered_suffix() { + use std::os::unix::net::UnixStream; + + let pool = new_pool(admission(2)); + let mut real_pipe = pool.acquire().unwrap(); + let (real_socket, _peer) = UnixStream::pair().unwrap(); + real_socket.set_nonblocking(true).unwrap(); + let real_socket = Descriptor::from(std::os::fd::OwnedFd::from(real_socket)); + let sim = uring_runtime::reactor::simulation::Simulation::new(); + let _environment = sim.enter(); + let mut simulated_pipe = pool.acquire().unwrap(); + let (simulated_socket, _peer) = sim.socket_pair(); + + for (pipe, socket) in [ + (&mut real_pipe, &simulated_socket), + (&mut simulated_pipe, &real_socket), + ] { + assert_eq!(pipe.try_write(b"pending").unwrap(), 7); + assert_eq!( + pipe.try_splice_descriptor(socket, 3) + .unwrap_err() + .raw_os_error(), + Some(libc::EOPNOTSUPP) + ); + assert_eq!(pipe.buffered(), 7); + let mut bytes = [0; 7]; + assert_eq!(pipe.try_read(&mut bytes).unwrap(), 7); + assert_eq!(&bytes, b"pending"); + } + } + + /// Empty pipes reuse both descriptors while partial payloads are closed. + #[test] + fn empty_pipe_reuses_descriptors_and_partial_pipe_is_closed() { + let admission = admission(1); + let pool = new_pool(admission.clone()); + let mut pipe = pool.acquire().unwrap(); + let read = pipe.resources.as_ref().unwrap().read.as_raw_fd(); + let write = pipe.resources.as_ref().unwrap().write.as_raw_fd(); + pipe.try_write(b"secret").unwrap(); + let mut bytes = [0; 6]; + assert_eq!(pipe.try_read(&mut bytes).unwrap(), 6); + drop(pipe); + assert_eq!(admission.used(ResourceClass::Pipe), 1); + let mut pipe = pool.acquire().unwrap(); + assert_eq!(pipe.resources.as_ref().unwrap().read.as_raw_fd(), read); + assert_eq!(pipe.resources.as_ref().unwrap().write.as_raw_fd(), write); + assert_eq!( + pipe.try_read(&mut bytes).unwrap_err().kind(), + io::ErrorKind::WouldBlock + ); + pipe.try_write(b"discard").unwrap(); + drop(pipe); + assert!(pool.idle.borrow().is_empty()); + assert_eq!(admission.used(ResourceClass::Pipe), 0); + // Dropped resources close their owned descriptors. Do not probe the old + // numbers: parallel tests may already have reused them for other files. + let mut replacement = pool.acquire().unwrap(); + assert_eq!(admission.used(ResourceClass::Pipe), 1); + assert_eq!(replacement.buffered(), 0); + assert_eq!( + replacement.try_read(&mut bytes).unwrap_err().kind(), + io::ErrorKind::WouldBlock + ); + } + + /// Stop observation drains idle owners but leaves active work and waiters owned. + #[test] + fn review_regression_stop_drains_idle_on_all_acquisition_paths() { + use std::io::Read; + + for path in 0..3 { + let quotas = admission(3); + let pool = new_pool(quotas.clone()); + let idle_read = pool.acquire().unwrap(); + let idle_write = pool.acquire().unwrap(); + let mut held = pool.acquire().unwrap(); + held.try_write(b"live").unwrap(); + let mut read = std::fs::File::from( + idle_read + .resources + .as_ref() + .unwrap() + .read + .as_fd() + .try_clone_to_owned() + .unwrap(), + ); + let write = idle_write + .resources + .as_ref() + .unwrap() + .write + .as_fd() + .try_clone_to_owned() + .unwrap(); + let mut first = acquire_wait(&pool); + let mut second = acquire_wait(&pool); + let mut cx = Context::from_waker(Waker::noop()); + assert!(first.as_mut().poll(&mut cx).is_pending()); + assert!(second.as_mut().poll(&mut cx).is_pending()); + drop((idle_read, idle_write)); + assert_eq!(pool.idle_count(), 2); + assert_eq!(quotas.used(ResourceClass::Pipe), 3); + quotas.stop(); + match path { + 0 => assert!(matches!(pool.acquire(), Err(Error::Unavailable))), + 1 => assert!(matches!( + second.as_mut().poll(&mut cx), + Poll::Ready(Err(Error::Unavailable)) + )), + _ => assert!(matches!( + acquire_wait(&pool).as_mut().poll(&mut cx), + Poll::Ready(Err(Error::Unavailable)) + )), + } + assert_eq!(pool.idle_count(), 0, "path {path} retained idle pipes"); + assert_eq!(quotas.used(ResourceClass::Pipe), 1); + let pending = if path == 1 { 1 } else { 2 }; + assert_eq!(pool.waiting.borrow().len(), pending); + assert_eq!( + quotas.used(ResourceClass::RequestContext), + pending * pool.waiter_bytes() + ); + // Owned duplicates observe peer closure, never a reused descriptor number. + assert_eq!(read.read(&mut [0]).unwrap(), 0); + let mut poll = libc::pollfd { + fd: write.as_raw_fd(), + events: libc::POLLOUT, + revents: 0, + }; + // SAFETY: poll borrows one initialized entry and a live owned descriptor. + assert_eq!(unsafe { libc::poll(&mut poll, 1, 0) }, 1); + assert_ne!(poll.revents & libc::POLLERR, 0); + assert!(matches!( + first.as_mut().poll(&mut cx), + Poll::Ready(Err(Error::Unavailable)) + )); + if path != 1 { + assert!(matches!( + second.as_mut().poll(&mut cx), + Poll::Ready(Err(Error::Unavailable)) + )); + } + assert!(pool.waiting.borrow().is_empty()); + assert_eq!(quotas.used(ResourceClass::RequestContext), 0); + let mut bytes = [0; 4]; + assert_eq!(held.try_read(&mut bytes).unwrap(), 4); + assert_eq!(&bytes, b"live"); + drop(held); + assert_eq!(quotas.used(ResourceClass::Pipe), 0); + assert_eq!(pool.idle_count(), 0); + assert!(matches!(pool.acquire(), Err(Error::Unavailable))); + } + } + + /// Charge destruction may reenter a stopped pool after its idle list is detached. + #[test] + fn review_regression_stop_drain_charge_destructor_reentry() { + let quotas = admission(2); + let pool = Rc::new(new_pool(quotas.clone())); + let first = pool.acquire().unwrap(); + let second = pool.acquire().unwrap(); + drop((first, second)); + quotas.stop(); + let called = Rc::new(std::cell::Cell::new(false)); + let callback_called = called.clone(); + let callback_pool = pool.clone(); + quotas.totals.wake.register(&waker_callbacks::waker()); + waker_callbacks::on_callback(move |event| { + assert_eq!(event, "wake"); + assert_eq!(callback_pool.idle_count(), 0); + assert!(matches!(callback_pool.acquire(), Err(Error::Unavailable))); + callback_called.set(true); + }); + assert!(matches!(pool.acquire(), Err(Error::Unavailable))); + waker_callbacks::clear(); + assert!(called.get()); + assert_eq!(quotas.used(ResourceClass::Pipe), 0); + assert_eq!(pool.idle_count(), 0); + } + + /// A lease outliving its pool continues to hold capacity until drop. + #[test] + fn exhaustion_and_drop_return_capacity_even_after_pool_drop() { + let admission = admission(1); + let pool = new_pool(admission.clone()); + let lease = pool.acquire().unwrap(); + assert!(matches!(pool.acquire(), Err(Error::Overloaded))); + drop(pool); + let pool = new_pool(admission); + assert!(matches!(pool.acquire(), Err(Error::Overloaded))); + drop(lease); + assert!(pool.acquire().is_ok()); + } + + /// Head and tail cancellation keep allocated queue storage within live charges. + #[test] + fn review_regression_canceled_waiter_storage_remains_charged() { + const WAITERS: usize = 1024; + for keep_tail in [false, true] { + let quotas = admission(1); + let pool = PipePool::new( + quotas.clone(), + ResourceClass::Pipe, + ResourceClass::RequestContext, + WAITERS, + ); + let held = pool.acquire().unwrap(); + let mut cx = Context::from_waker(Waker::noop()); + let mut waiting: Vec<_> = (0..WAITERS).map(|_| acquire_wait(&pool)).collect(); + for wait in &mut waiting { + assert!(wait.as_mut().poll(&mut cx).is_pending()); + } + if keep_tail { + waiting.reverse(); + } + while waiting.len() > 1 { + drop(waiting.pop()); + let queue = pool.waiting.borrow(); + assert_eq!(queue.len(), waiting.len()); + let slots = queue.capacity() * std::mem::size_of::>>>(); + let owners = queue.len() + * (std::mem::size_of::>() + + std::mem::size_of::>>() + + 2 * std::mem::size_of::()); + assert!( + slots + owners <= quotas.used(ResourceClass::RequestContext), + "{} live waiters retain {} queue slots without admission", + queue.len(), + queue.capacity(), + ); + } + drop(held); + assert!(matches!( + waiting[0].as_mut().poll(&mut cx), + Poll::Ready(Ok(_)) + )); + drop(waiting); + assert_eq!(pool.waiting.borrow().capacity(), 0); + assert_eq!(quotas.used(ResourceClass::RequestContext), 0); + } + } + + /// Scheduled acquisition is bounded and FIFO without polling itself awake. + #[test] + fn scheduled_acquisition_is_bounded_fifo_and_wakes_only_for_progress() { + let admission = admission(2); + let pool = new_pool(admission.clone()); + let held = [pool.acquire().unwrap(), pool.acquire().unwrap()]; + let count = Arc::new(WakeCounter::default()); + let waker = Waker::from(count.clone()); + let mut cx = Context::from_waker(&waker); + let mut waiting: Vec<_> = (0..8).map(|_| acquire_wait(&pool)).collect(); + for wait in &mut waiting { + assert!(wait.as_mut().poll(&mut cx).is_pending()); + } + assert_eq!(count.count(), 0, "waiting does not spin/self-wake"); + assert!(matches!( + acquire_wait(&pool).as_mut().poll(&mut cx), + Poll::Ready(Err(Error::Overloaded)) + )); + assert_eq!(pool.waiting.borrow().len(), 8); + assert_eq!( + admission.used(ResourceClass::RequestContext), + 8 * pool.waiter_bytes() + ); + assert_eq!(admission.used(ResourceClass::Pipe), 2); + assert_eq!(admission.used(ResourceClass::Plaintext), 0); + drop(held); + assert!(count.count() > 0); + // Reverse polling cannot let new arrivals jump the queue. + for wait in waiting.iter_mut().skip(1).rev() { + assert!(wait.as_mut().poll(&mut cx).is_pending()); + } + let mut leases = VecDeque::new(); + for mut wait in waiting { + if leases.len() == 2 { + leases.pop_front(); + } + let Poll::Ready(Ok(pipe)) = wait.as_mut().poll(&mut cx) else { + panic!("FIFO waiter did not progress") + }; + leases.push_back(pipe); + assert!(admission.used(ResourceClass::Pipe) <= 2); + } + assert!(pool.waiting.borrow().is_empty()); + assert_eq!(admission.used(ResourceClass::RequestContext), 0); + drop(leases); + assert_eq!( + admission.used(ResourceClass::Pipe), + 2, + "idle pipes remain admitted" + ); + drop(pool); + assert_eq!(admission.used(ResourceClass::Pipe), 0); + } + + /// Kernel descriptors have bounded capacity and independent nonblocking data. + #[test] + fn pipes_are_nonblocking_bounded_cloexec_and_independent() { + let admission = admission(2); + let pool = new_pool(admission); + let mut first = pool.acquire().unwrap(); + let mut second = pool.acquire().unwrap(); + let resources = first.resources.as_ref().unwrap(); + for fd in [&resources.read, &resources.write] { + // SAFETY: these descriptors are owned for the duration of the query. + assert_ne!( + unsafe { libc::fcntl(fd.as_raw_fd(), libc::F_GETFL) } & libc::O_NONBLOCK, + 0 + ); + assert_ne!( + unsafe { libc::fcntl(fd.as_raw_fd(), libc::F_GETFD) } & libc::FD_CLOEXEC, + 0 + ); + } + assert!(first.capacity() <= MAX_PIPE_BYTES); + assert!(first.capacity() > 0); + // Account the actual kernel capacity, including denied best-effort growth. + assert_eq!(first.capacity(), unsafe { + libc::fcntl( + first.resources.as_ref().unwrap().write.as_raw_fd(), + libc::F_GETPIPE_SZ, + ) + } as usize); + let bytes = vec![0x5a; first.capacity()]; + assert_eq!(first.try_write(&bytes).unwrap(), bytes.len()); + assert_eq!( + first.try_write(b"x").unwrap_err().kind(), + io::ErrorKind::WouldBlock + ); + let mut out = vec![0; bytes.len()]; + assert_eq!( + second.try_read(&mut out).unwrap_err().kind(), + io::ErrorKind::WouldBlock + ); + assert_eq!(first.try_read(&mut out).unwrap(), bytes.len()); + assert_eq!(out, bytes); + assert_eq!( + first.try_read(&mut out).unwrap_err().kind(), + io::ErrorKind::WouldBlock + ); + assert_eq!(second.try_write(b"second").unwrap(), 6); + assert_eq!(second.try_read(&mut out).unwrap(), 6); + assert_eq!(&out[..6], b"second"); + } + + /// Copied pages remain intact after source reuse, partial drain, and pipe reuse. + #[test] + fn copied_splice_survives_source_reuse_and_partial_drain() { + use std::{io::Read, os::unix::net::UnixStream}; + let admission = admission(1); + let pool = new_pool(admission); + let mut pipe = pool.acquire().unwrap(); + let (socket, mut peer) = UnixStream::pair().unwrap(); + socket.set_nonblocking(true).unwrap(); + let mut source = b"copied kernel pages".to_vec(); + assert_eq!(pipe.try_write(&source).unwrap(), source.len()); + source.fill(0); + assert_eq!(pipe.try_splice_to(&socket, 6).unwrap(), 6); + assert_eq!(pipe.buffered(), 13); + assert_eq!(pipe.try_splice_to(&socket, usize::MAX).unwrap(), 13); + assert_eq!(pipe.buffered(), 0); + // Socket-owned kernel pages survive both pipe closure and quota reuse. + drop(pipe); + let mut reused = pool.acquire().unwrap(); + reused.try_write(b"replacement data").unwrap(); + let mut received = [0; 19]; + peer.read_exact(&mut received).unwrap(); + assert_eq!(&received, b"copied kernel pages"); + } + + /// Blocking sockets and disconnects preserve the full buffered suffix. + #[test] + fn splice_rejects_blocking_socket_and_handles_disconnect_without_losing_bytes() { + use std::os::unix::net::UnixStream; + let admission = admission(1); + let pool = new_pool(admission); + let mut pipe = pool.acquire().unwrap(); + pipe.try_write(b"abc").unwrap(); + let (socket, peer) = UnixStream::pair().unwrap(); + assert_eq!( + pipe.try_splice_to(&socket, 3).unwrap_err().raw_os_error(), + Some(libc::EINVAL) + ); + assert_eq!(pipe.buffered(), 3); + socket.set_nonblocking(true).unwrap(); + drop(peer); + assert_eq!( + pipe.try_splice_to(&socket, 3).unwrap_err().raw_os_error(), + Some(libc::EPIPE) + ); + assert_eq!(pipe.buffered(), 3); + let mut bytes = [0; 3]; + assert_eq!(pipe.try_read(&mut bytes).unwrap(), 3); + assert_eq!(&bytes, b"abc"); + } + + /// Backpressure and invalid socket types leave fallback bytes untouched. + #[test] + fn splice_backpressure_preserves_buffered_bytes_and_socket_validation() { + use std::{io::Write, os::unix::net::UnixStream}; + let admission = admission(1); + let pool = new_pool(admission); + let mut pipe = pool.acquire().unwrap(); + pipe.try_write(b"pending").unwrap(); + let (mut socket, _peer) = UnixStream::pair().unwrap(); + socket.set_nonblocking(true).unwrap(); + let start = std::time::Instant::now(); + loop { + assert!(start.elapsed() < std::time::Duration::from_secs(5)); + match socket.write(&[1; 8192]) { + Ok(count) => assert_ne!(count, 0), + Err(error) if error.kind() == io::ErrorKind::WouldBlock => break, + Err(error) => panic!("{error}"), + } + } + assert_eq!( + pipe.try_splice_to(&socket, 7).unwrap_err().kind(), + io::ErrorKind::WouldBlock + ); + assert_eq!(pipe.buffered(), 7); + let (datagram, _peer) = std::os::unix::net::UnixDatagram::pair().unwrap(); + datagram.set_nonblocking(true).unwrap(); + assert_eq!( + pipe.try_splice_to(&datagram, 7).unwrap_err().raw_os_error(), + Some(libc::EINVAL) + ); + let file = std::fs::OpenOptions::new() + .write(true) + .open("/dev/null") + .unwrap(); + // SAFETY: change flags on this exclusively owned test descriptor. + assert_eq!( + unsafe { libc::fcntl(file.as_raw_fd(), libc::F_SETFL, libc::O_NONBLOCK) }, + 0 + ); + assert_eq!( + pipe.try_splice_to(&file, 7).unwrap_err().raw_os_error(), + Some(libc::ENOTSOCK) + ); + let mut bytes = [0; 7]; + assert_eq!(pipe.try_read(&mut bytes).unwrap(), 7); + assert_eq!(&bytes, b"pending"); + } +} diff --git a/cmd/racer-dataplane/flow/tests/drain_waker.rs b/cmd/racer-dataplane/flow/tests/drain_waker.rs new file mode 100644 index 000000000..ed33813c7 --- /dev/null +++ b/cmd/racer-dataplane/flow/tests/drain_waker.rs @@ -0,0 +1,114 @@ +//! Drain registration must not invoke waker callbacks under the owner borrow. + +use flow_control::coalesce::flight::{self, Table}; +use std::cell::{Cell, RefCell}; +use std::rc::Rc; +use std::sync::Arc; +use std::sync::atomic::{AtomicUsize, Ordering}; +use std::task::{RawWaker, RawWakerVTable, Wake, Waker}; + +thread_local! { + static ON_CLONE: RefCell>> = RefCell::new(None); + static ON_DROP: RefCell>> = RefCell::new(None); +} + +/// No pointer is dereferenced or owned; callbacks belong to the current thread. +static VTABLE: RawWakerVTable = RawWakerVTable::new( + |_| { + let callback = ON_CLONE.with(|slot| slot.borrow_mut().take()); + if let Some(callback) = callback { + callback(); + } + raw_waker() + }, + |_| {}, + |_| {}, + |_| { + let callback = ON_DROP.with(|slot| slot.borrow_mut().take()); + if let Some(callback) = callback { + callback(); + } + }, +); + +fn raw_waker() -> RawWaker { + RawWaker::new(std::ptr::null(), &VTABLE) +} + +fn callback_waker() -> Waker { + // SAFETY: The vtable owns no data and only accesses thread-local callbacks. + unsafe { Waker::from_raw(raw_waker()) } +} + +fn register(owner: &RefCell>, waker: &Waker) { + let waker = waker.clone(); + drop(flight::update(owner, |table, _| { + table.register_drain(waker) + })); +} + +#[test] +fn drain_waker_clone_can_reenter_owner() { + let owner = Rc::new(RefCell::new(Table::::default())); + let called = Rc::new(Cell::new(false)); + ON_CLONE.with(|slot| { + slot.replace(Some(Box::new({ + let owner = owner.clone(); + let called = called.clone(); + move || { + flight::update(&owner, |table, _| assert!(table.is_empty())); + called.set(true); + } + }))); + }); + register(&owner, &callback_waker()); + assert!(called.get()); +} + +#[test] +fn drain_waker_replacement_drop_can_reenter_owner() { + let owner = Rc::new(RefCell::new(Table::::default())); + register(&owner, &callback_waker()); + let called = Rc::new(Cell::new(false)); + ON_DROP.with(|slot| { + slot.replace(Some(Box::new({ + let owner = owner.clone(); + let called = called.clone(); + move || { + flight::update(&owner, |table, _| assert!(table.is_empty())); + called.set(true); + } + }))); + }); + register(&owner, Waker::noop()); + assert!(called.get()); +} + +#[derive(Default)] +struct CountWake(AtomicUsize); + +impl Wake for CountWake { + fn wake(self: Arc) { + self.0.fetch_add(1, Ordering::Relaxed); + } +} + +#[test] +fn drain_waker_notifies_latest_target_once() { + let owner = RefCell::new(Table::::default()); + let old = Arc::new(CountWake::default()); + let latest = Arc::new(CountWake::default()); + let latest_waker = Waker::from(latest.clone()); + register(&owner, &Waker::from(old.clone())); + register(&owner, &latest_waker); + register(&owner, &latest_waker); + flight::update(&owner, |table, wakes| { + table.notify_drain(wakes); + table.notify_drain(wakes); + }); + assert_eq!(old.0.load(Ordering::Relaxed), 0); + assert_eq!(latest.0.load(Ordering::Relaxed), 1); + register(&owner, &latest_waker); + flight::update(&owner, |table, wakes| table.notify_drain(wakes)); + assert_eq!(latest.0.load(Ordering::Relaxed), 2); +} diff --git a/cmd/racer-dataplane/flow/tests/workflows.rs b/cmd/racer-dataplane/flow/tests/workflows.rs new file mode 100644 index 000000000..65a092c27 --- /dev/null +++ b/cmd/racer-dataplane/flow/tests/workflows.rs @@ -0,0 +1,1697 @@ +//! Public ownership, capacity, and coalescing workflows without private-state access. + +/// Reservation lifetime and target selection contracts. +mod handoff_tests { + use flow_control::{Error, Handoff, HandoffAdmission, Result}; + use std::{ + sync::{ + Arc, Mutex, + atomic::{AtomicUsize, Ordering}, + }, + task::Waker, + }; + + /// One-slot admission used to expose early reservation release. + struct Quota(Arc); + + /// A counted reservation released only when its owner drops. + struct Held(Arc); + + impl Drop for Held { + /// Return the fixture's single slot. + fn drop(&mut self) { + self.0.fetch_sub(1, Ordering::SeqCst); + } + } + + impl HandoffAdmission for Quota { + /// Slot owner carried alongside every offered item. + type Reservation = Held; + + /// This synchronous fixture does not arrange wakeups. + fn register(&self, _: &Waker) {} + + /// Admit only when the fixture's slot is empty. + fn reserve(&self) -> Result { + self.0 + .compare_exchange(0, 1, Ordering::SeqCst, Ordering::SeqCst) + .map_err(|_| Error::Overloaded)?; + Ok(Held(self.0.clone())) + } + } + + struct ScriptedAdmission { + key: usize, + result: Result, + registered: AtomicUsize, + calls: Arc>>, + } + + impl HandoffAdmission for ScriptedAdmission { + type Reservation = usize; + + fn register(&self, _: &Waker) { + self.registered.fetch_add(1, Ordering::SeqCst); + } + + fn reserve(&self) -> Result { + assert_eq!(self.registered.swap(0, Ordering::SeqCst), 1); + self.calls.lock().unwrap().push(self.key); + self.result + } + } + + fn scripted_handoff( + results: &[Result], + calls: &Arc>>, + ) -> Arc> { + let keys: Vec<_> = (0..results.len()).collect(); + let handoff = Arc::new(Handoff::new(&keys)); + for (key, result) in results.iter().copied().enumerate() { + handoff + .install( + &key, + ScriptedAdmission { + key, + result, + registered: AtomicUsize::new(0), + calls: calls.clone(), + }, + ) + .unwrap(); + } + handoff + } + + #[test] + fn admission_failures_preserve_classification_and_capacity_retry() { + use Error::{Overloaded, Unavailable}; + for (errors, expected) in [ + (vec![Overloaded], Overloaded), + (vec![Unavailable], Unavailable), + (vec![Unavailable, Unavailable], Unavailable), + (vec![Overloaded, Overloaded], Overloaded), + (vec![Overloaded, Unavailable], Overloaded), + (vec![Unavailable, Overloaded], Overloaded), + ] { + let calls = Arc::new(Mutex::new(Vec::new())); + let results: Vec<_> = errors.iter().copied().map(Err).collect(); + let handoff = scripted_handoff(&results, &calls); + assert_eq!(handoff.reserve(Waker::noop()).err(), Some(expected)); + assert_eq!( + *calls.lock().unwrap(), + (0..errors.len()).collect::>() + ); + } + } + + #[test] + fn fatal_admission_errors_stop_scanning_without_becoming_overload() { + for error in [Error::InvalidInput, Error::Io] { + for preceding in [None, Some(Error::Overloaded), Some(Error::Unavailable)] { + let calls = Arc::new(Mutex::new(Vec::new())); + let mut results: Vec<_> = preceding.into_iter().map(Err).collect(); + results.extend([Err(error), Ok(99)]); + let handoff = scripted_handoff(&results, &calls); + assert_eq!(handoff.reserve(Waker::noop()).err(), Some(error)); + assert_eq!( + *calls.lock().unwrap(), + (0..results.len() - 1).collect::>() + ); + } + } + } + + #[test] + fn successful_admission_skips_retryable_failures_and_keeps_round_robin() { + for failures in [ + [Error::Overloaded, Error::Unavailable], + [Error::Unavailable, Error::Overloaded], + ] { + let calls = Arc::new(Mutex::new(Vec::new())); + let handoff = scripted_handoff( + &[Err(failures[0]), Err(failures[1]), Ok(12), Ok(13)], + &calls, + ); + for target in [2, 3, 2] { + handoff + .reserve(Waker::noop()) + .unwrap() + .deliver(|| ()) + .unwrap(); + let [item] = handoff.pop_batch::<1>(&target, Waker::noop(), 1).unwrap(); + assert_eq!(item.unwrap().into_parts(), ((), target + 10)); + } + assert_eq!(*calls.lock().unwrap(), [0, 1, 2, 3, 0, 1, 2]); + } + } + + #[test] + fn closed_and_uninstalled_targets_do_not_change_admission_errors() { + let calls = Arc::new(Mutex::new(Vec::new())); + let handoff = scripted_handoff(&[Err(Error::Io), Err(Error::Unavailable)], &calls); + handoff.close(&0); + assert_eq!( + handoff.reserve(Waker::noop()).err(), + Some(Error::Unavailable) + ); + assert_eq!(*calls.lock().unwrap(), [1]); + handoff.close(&1); + assert_eq!( + handoff.reserve(Waker::noop()).err(), + Some(Error::Overloaded) + ); + assert_eq!(*calls.lock().unwrap(), [1]); + + let handoff = Arc::new(Handoff::<_, _, ()>::new(&[0, 1])); + handoff + .install( + &1, + ScriptedAdmission { + key: 1, + result: Err(Error::Unavailable), + registered: AtomicUsize::new(0), + calls: calls.clone(), + }, + ) + .unwrap(); + assert_eq!( + handoff.reserve(Waker::noop()).err(), + Some(Error::Unavailable) + ); + assert_eq!(*calls.lock().unwrap(), [1, 1]); + } + + /// Offers, envelopes, and popped items retain the selected target's slot. + #[test] + fn round_robin_reserves_before_delivery_and_releases_after_close() { + let handoff = Arc::new(Handoff::<_, _, u8>::new(&[1, 2])); + let counts = [Arc::new(AtomicUsize::new(0)), Arc::new(AtomicUsize::new(0))]; + let waker = Waker::noop(); + assert!(matches!(handoff.reserve(waker), Err(Error::Overloaded))); + for (key, count) in [1, 2].into_iter().zip(&counts) { + handoff.install(&key, Quota(count.clone())).unwrap(); + } + assert_eq!( + handoff.install(&1, Quota(counts[0].clone())), + Err(Error::InvalidInput) + ); + let first = handoff.reserve(waker).unwrap(); + let second = handoff.reserve(waker).unwrap(); + assert!(matches!(handoff.reserve(waker), Err(Error::Overloaded))); + first.deliver(|| 7).unwrap(); + assert!( + handoff + .pop_batch::<2>(&1, waker, 0) + .unwrap() + .iter() + .all(Option::is_none) + ); + let [item, empty] = handoff.pop_batch::<2>(&1, waker, 1).unwrap(); + let (payload, held) = item.unwrap().into_parts(); + assert_eq!(payload, 7); + assert!(empty.is_none()); + handoff.close(&1); + assert_eq!(counts[0].load(Ordering::SeqCst), 1); + drop(held); + assert_eq!(counts[0].load(Ordering::SeqCst), 0); + handoff.close(&2); + assert_eq!( + second.deliver(|| panic!("closed target built item")), + Err(Error::Unavailable) + ); + assert_eq!(counts[1].load(Ordering::SeqCst), 0); + assert!(matches!( + handoff.pop_batch::<1>(&3, waker, 1), + Err(Error::InvalidInput) + )); + assert!(matches!( + Arc::new(Handoff::::new(&[])).reserve(waker), + Err(Error::Overloaded) + )); + } + + /// Closing and abandoning work release exactly the retained reservations. + #[test] + fn close_drains_queued_reservations_and_abandoned_offer_releases() { + let handoff = Arc::new(Handoff::<_, _, ()>::new(&[1])); + let count = Arc::new(AtomicUsize::new(0)); + handoff.install(&1, Quota(count.clone())).unwrap(); + drop(handoff.reserve(Waker::noop()).unwrap()); + assert_eq!(count.load(Ordering::SeqCst), 0); + handoff + .reserve(Waker::noop()) + .unwrap() + .deliver(|| ()) + .unwrap(); + handoff.close(&1); + handoff.close(&1); + handoff.close(&2); + assert_eq!(count.load(Ordering::SeqCst), 0); + assert_eq!(handoff.install(&1, Quota(count)), Err(Error::InvalidInput)); + } + + /// Payload-only construction cannot discard admission before dequeue. + #[test] + fn envelope_retains_reservation_through_queue_pop_and_owner_transfer() { + let handoff = Arc::new(Handoff::<_, _, u8>::new(&[1])); + let count = Arc::new(AtomicUsize::new(0)); + handoff.install(&1, Quota(count.clone())).unwrap(); + handoff + .reserve(Waker::noop()) + .unwrap() + .deliver(|| 9) + .unwrap(); + assert!(matches!( + handoff.reserve(Waker::noop()), + Err(Error::Overloaded) + )); + let [item] = handoff.pop_batch::<1>(&1, Waker::noop(), 1).unwrap(); + assert!(matches!( + handoff.reserve(Waker::noop()), + Err(Error::Overloaded) + )); + let (payload, held) = item.unwrap().into_parts(); + assert_eq!(payload, 9); + assert_eq!(count.load(Ordering::SeqCst), 1); + drop(held); + assert!(handoff.reserve(Waker::noop()).is_ok()); + } +} + +/// Fixed backing and admission transfer contracts. +mod buffer_tests { + use flow_control::{ChargedBuffer, Error, Policy, Quotas, Rejection}; + + /// Single fixture class for fixed backing. + #[derive(Clone, Copy)] + struct Class; + + impl flow_control::Class for Class { + /// Number of fixed-backing resource classes. + const COUNT: usize = 1; + + /// Select the fixture's sole counter. + fn index(self) -> usize { + 0 + } + } + + /// Fixed four-MiB backing budget without wake or page coverage policy. + struct TestPolicy; + + impl Policy for TestPolicy { + /// Single resource budget for fixed backing. + type Class = Class; + + /// One possible keyed identity for this fixture. + type Key = (); + + /// Allow four MiB of fixed backing. + fn limit(&self, _: Class) -> usize { + 4 * 1024 * 1024 + } + + /// Permit one fixture key. + fn max_keys(&self) -> usize { + 1 + } + + /// Buffer releases do not wake this synchronous fixture. + fn wakes(_: Class) -> bool { + false + } + + /// This fixture does not admit page allocator backing. + fn covers(_: Class) -> bool { + false + } + + /// Rejection details are irrelevant to these backing assertions. + fn rejected(&self, _: Rejection) {} + } + + /// Moves, reuse, and final transfer preserve backing and its exact charge. + #[test] + fn fixed_backing_recycles_and_transfers_charge_without_early_release() { + let quotas = Quotas::new(TestPolicy); + let length = 1024 * 1024; + let mut buffer = + ChargedBuffer::new(quotas.reserve(None, Class, 2 * length).unwrap(), length).unwrap(); + assert_eq!(quotas.used(Class), length); + assert_eq!(buffer.charge().amount(), length); + let pointer = buffer.bytes().as_ptr(); + buffer.bytes_mut().fill(0xa7); + let moved = buffer; + assert_eq!(moved.bytes().as_ptr(), pointer); + drop(moved); + assert_eq!(quotas.used(Class), length); + let buffer = + ChargedBuffer::new(quotas.reserve(None, Class, length).unwrap(), length).unwrap(); + assert_eq!(buffer.bytes().as_ptr(), pointer); + assert!(buffer.bytes().iter().all(|byte| *byte == 0)); + let (bytes, charge) = buffer.into_parts(); + assert_eq!(bytes.as_ptr(), pointer); + assert_eq!(quotas.used(Class), length); + drop((bytes, charge)); + assert_eq!(quotas.used(Class), 0); + } + + /// Invalid backing geometry releases the consumed reservation. + #[test] + fn invalid_lengths_return_charge() { + let quotas = Quotas::new(TestPolicy); + for length in [0, 9] { + assert!(matches!( + ChargedBuffer::new(quotas.reserve(None, Class, 8).unwrap(), length), + Err(Error::InvalidInput) + )); + assert_eq!(quotas.used(Class), 0); + } + } +} + +/// Quota transfers, retained key lifetimes, and exact release notifications. +mod quota_tests { + use flow_control::{Charge, Error, Policy, Quotas, Rejection, SharedQuotas}; + use std::{ + sync::{ + Arc, Mutex, + atomic::{AtomicUsize, Ordering}, + }, + task::{Wake, Waker}, + }; + + /// Independent wake-enabled and silent resource classes. + #[derive(Clone, Copy, Debug, Eq, PartialEq)] + enum Resource { + Waking, + + Silent, + } + + impl flow_control::Class for Resource { + /// Number of independently accounted fixture classes. + const COUNT: usize = 2; + + /// Select this resource's independently padded counter. + fn index(self) -> usize { + self as usize + } + } + + /// Configurable limits with ordered rejection observations. + #[derive(Clone)] + struct TestPolicy { + limit: usize, + + max_keys: usize, + + rejected: Arc>>>, + } + + impl TestPolicy { + /// Start a fixture with identical class limits and an empty event log. + fn new(limit: usize, max_keys: usize) -> Self { + Self { + limit, + max_keys, + rejected: Arc::default(), + } + } + } + + impl Policy for TestPolicy { + /// Resource classes for wake and accounting assertions. + type Class = Resource; + + /// Small identities make key retirement observable through admission. + type Key = u8; + + /// Return the fixture's ceiling for either resource class. + fn limit(&self, _: Resource) -> usize { + self.limit + } + + /// Bound simultaneously retained key records. + fn max_keys(&self) -> usize { + self.max_keys + } + + /// Only the waking class notifies admission waiters on drop. + fn wakes(class: Resource) -> bool { + class == Resource::Waking + } + + /// Neither test class is used for page allocator backing. + fn covers(_: Resource) -> bool { + false + } + + /// Preserve every attempted rejection, including reclamation retries. + fn rejected(&self, rejection: Rejection) { + self.rejected.lock().unwrap().push(rejection); + } + } + + /// Count notifications independently of the worker-local quota authority. + #[derive(Default)] + struct WakeCount(AtomicUsize); + + impl Wake for WakeCount { + /// Count one consumed waiter registration. + fn wake(self: Arc) { + self.0.fetch_add(1, Ordering::SeqCst); + } + } + + /// Splitting is accounting-neutral; shrinking releases both budgets silently. + #[test] + fn keyed_split_shrink_and_cross_thread_drop_keep_exact_accounting() { + /// Assert that local authority does not prevent transferable charges. + fn transferable() {} + + transferable::>(); + transferable::>(); + let quotas = Quotas::new(TestPolicy::new(100, 1)); + let shared = quotas.shared(); + let wake = Arc::new(WakeCount::default()); + shared.register(&Waker::from(wake.clone())); + let mut charge = quotas.reserve(Some(&1), Resource::Waking, 100).unwrap(); + for amount in [0, 100, usize::MAX] { + assert!(matches!(charge.split(amount), Err(Error::InvalidInput))); + assert_eq!(charge.amount(), 100); + assert_eq!(shared.used(Resource::Waking), 100); + } + let split = charge.split(40).unwrap(); + assert_eq!(split.key(), Some(&1)); + assert_eq!(split.class(), Resource::Waking); + assert!(quotas.owns(&split)); + assert_eq!(shared.used(Resource::Waking), 100); + for amount in [0, 61, usize::MAX] { + assert_eq!(charge.shrink(amount), Err(Error::InvalidInput)); + assert_eq!(charge.amount(), 60); + } + charge.shrink(60).unwrap(); + charge.shrink(20).unwrap(); + assert_eq!(shared.used(Resource::Waking), 60); + assert_eq!(wake.0.load(Ordering::SeqCst), 0, "shrink must not wake"); + let refill = quotas.reserve(Some(&1), Resource::Waking, 40).unwrap(); + assert_eq!(shared.used(Resource::Waking), 100); + std::thread::spawn(move || drop(split)).join().unwrap(); + assert_eq!(shared.used(Resource::Waking), 60); + assert_eq!(wake.0.load(Ordering::SeqCst), 1); + drop((refill, charge)); + assert_eq!(shared.used(Resource::Waking), 0); + let next = quotas.reserve(Some(&2), Resource::Waking, 100).unwrap(); + assert_eq!(next.key(), Some(&2)); + assert!(quotas.policy().rejected.lock().unwrap().is_empty()); + } + + /// Retaining backing transfers surplus admission but not the donor's key owner. + #[test] + fn recycler_transfers_surplus_and_empty_donor_still_owns_key_until_drop() { + let size = 1 << 20; + let quotas = Quotas::new(TestPolicy::new(4 * size, 1)); + let shared = quotas.shared(); + let wake = Arc::new(WakeCount::default()); + shared.register(&Waker::from(wake.clone())); + let mut donor = quotas + .reserve(Some(&1), Resource::Waking, 2 * size) + .unwrap(); + let mut bytes = donor.buffer(size).unwrap(); + bytes.fill(0xa7); + bytes.truncate(1); + donor.recycle(bytes); + assert_eq!(donor.amount(), 0); + assert_eq!(donor.key(), Some(&1)); + assert_eq!(quotas.retained_buffer_bytes(), 2 * size); + assert_eq!(shared.used(Resource::Waking), 2 * size); + assert_eq!(wake.0.load(Ordering::SeqCst), 0); + assert_eq!(donor.shrink(1), Err(Error::InvalidInput)); + assert!(matches!(donor.split(1), Err(Error::InvalidInput))); + quotas.reclaim_buffers(); + assert_eq!(shared.used(Resource::Waking), 0); + assert_eq!(wake.0.load(Ordering::SeqCst), 1); + assert!(matches!( + quotas.reserve(Some(&2), Resource::Silent, 1), + Err(Error::Overloaded) + )); + assert!(matches!( + "as.policy().rejected.lock().unwrap()[..], + [ + Rejection::Keys { used: 1, limit: 1 }, + Rejection::Keys { used: 1, limit: 1 } + ] + )); + shared.register(&Waker::from(wake.clone())); + std::thread::spawn(move || drop(donor)).join().unwrap(); + assert_eq!(wake.0.load(Ordering::SeqCst), 2, "empty drops still wake"); + let next = quotas.reserve(Some(&2), Resource::Silent, 1).unwrap(); + assert_eq!(shared.used(Resource::Waking), 0); + assert_eq!(shared.used(Resource::Silent), 1); + drop(next); + assert_eq!(shared.used(Resource::Silent), 0); + } + + /// Failed retention leaves admission with its caller, including after stop. + #[test] + fn full_recycler_and_stopped_recycler_do_not_consume_charge() { + let size = 1 << 20; + let quotas = Quotas::new(TestPolicy::new(4 * size, 1)); + for _ in 0..2 { + let mut retained = quotas.reserve(None, Resource::Silent, size).unwrap(); + retained.recycle(vec![0xa7; size]); + assert_eq!(retained.amount(), 0); + } + let mut rejected = quotas.reserve(None, Resource::Silent, size).unwrap(); + rejected.recycle(vec![0xa7; size]); + assert_eq!(rejected.amount(), size); + assert_eq!(quotas.retained_buffer_bytes(), 2 * size); + assert_eq!(quotas.used(Resource::Silent), 3 * size); + quotas.stop(); + assert_eq!(quotas.retained_buffer_bytes(), 0); + assert_eq!(quotas.used(Resource::Silent), size); + rejected.recycle(vec![0xa7; size]); + assert_eq!(rejected.amount(), size); + assert_eq!(quotas.retained_buffer_bytes(), 0); + drop(rejected); + assert_eq!(quotas.used(Resource::Silent), 0); + } + + /// Checked admission rejects arithmetic overflow with unchanged reported facts. + #[test] + fn full_width_rejections_preserve_usage_and_attempt_counts() { + let quotas = Quotas::new(TestPolicy::new(usize::MAX, 1)); + let shared = quotas.shared(); + let mut charge = shared.reserve(Resource::Silent, usize::MAX).unwrap(); + assert!(matches!( + shared.reserve(Resource::Silent, 1), + Err(Error::Overloaded) + )); + assert!(matches!( + quotas.reserve(None, Resource::Silent, 1), + Err(Error::Overloaded) + )); + assert!(matches!( + quotas.reserve_completion(None, Resource::Silent, 1), + Err(Error::Overloaded) + )); + { + let rejected = quotas.policy().rejected.lock().unwrap(); + assert_eq!( + rejected.len(), + 5, + "shared attempts once; local retries once" + ); + for event in rejected.iter() { + assert!(matches!( + event, + Rejection::Resource { + class: Resource::Silent, + used: usize::MAX, + limit: usize::MAX, + requested: 1, + key_used: None, + key_limit: None, + } + )); + } + } + charge.shrink(1).unwrap(); + let split = shared.reserve(Resource::Silent, usize::MAX - 1).unwrap(); + assert_eq!(shared.used(Resource::Silent), usize::MAX); + drop(charge); + assert_eq!(shared.used(Resource::Silent), usize::MAX - 1); + drop(split); + assert_eq!(shared.used(Resource::Silent), 0); + } + + /// A panicking key clone cannot strand fresh or transferred admission. + #[test] + fn key_clone_panic_keeps_split_and_recycle_admission_with_donor() { + use std::{ + panic::{AssertUnwindSafe, catch_unwind}, + sync::atomic::AtomicBool, + }; + + /// Enable clone failure only for this test's otherwise ordinary key. + static PANIC_ON_CLONE: AtomicBool = AtomicBool::new(false); + + /// One identity whose clone can fail after successful admission. + #[derive(Eq, Hash, PartialEq)] + struct Key; + + impl Clone for Key { + /// Panic on demand before a new charge can own admission. + fn clone(&self) -> Self { + assert!(!PANIC_ON_CLONE.load(Ordering::SeqCst), "key clone failed"); + Self + } + } + + /// Minimal policy whose sole key can panic while transferring ownership. + struct PanickingPolicy; + + impl Policy for PanickingPolicy { + /// Reuse the accounting fixture's resource classes. + type Class = Resource; + + /// Key with controllable clone failure. + type Key = Key; + + /// Allow one retained MiB and one MiB of surplus admission. + fn limit(&self, _: Resource) -> usize { + 2 << 20 + } + + /// Keep one key record for the donor and its possible split. + fn max_keys(&self) -> usize { + 1 + } + + /// This test observes accounting rather than wake delivery. + fn wakes(_: Resource) -> bool { + false + } + + /// No page allocator is involved in the transfer. + fn covers(_: Resource) -> bool { + false + } + + /// No admission rejection is expected in this panic path. + fn rejected(&self, _: Rejection) { + panic!("unexpected rejection"); + } + } + + let size = 1 << 20; + let quotas = Quotas::new(PanickingPolicy); + let mut donor = quotas + .reserve(Some(&Key), Resource::Silent, 2 * size) + .unwrap(); + PANIC_ON_CLONE.store(true, Ordering::SeqCst); + assert!(catch_unwind(AssertUnwindSafe(|| donor.split(size))).is_err()); + assert_eq!(donor.amount(), 2 * size); + assert_eq!(quotas.used(Resource::Silent), 2 * size); + assert!(catch_unwind(AssertUnwindSafe(|| donor.recycle(vec![0xa7; size]))).is_err()); + assert_eq!(donor.amount(), 2 * size); + assert_eq!(quotas.retained_buffer_bytes(), 0); + assert_eq!(quotas.used(Resource::Silent), 2 * size); + PANIC_ON_CLONE.store(false, Ordering::SeqCst); + let split = donor.split(size).unwrap(); + drop((donor, split)); + assert_eq!(quotas.used(Resource::Silent), 0); + let replacement = quotas + .reserve(Some(&Key), Resource::Silent, 2 * size) + .unwrap(); + drop(replacement); + assert_eq!(quotas.used(Resource::Silent), 0); + + // A live key record isolates the final charge-key clone from table setup. + // Both ordinary and drain admission must leave counters unchanged on panic. + for completion in [false, true] { + let donor = quotas.reserve(Some(&Key), Resource::Silent, size).unwrap(); + PANIC_ON_CLONE.store(true, Ordering::SeqCst); + let failed = catch_unwind(AssertUnwindSafe(|| { + if completion { + quotas.reserve_completion(Some(&Key), Resource::Silent, size) + } else { + quotas.reserve(Some(&Key), Resource::Silent, size) + } + })); + PANIC_ON_CLONE.store(false, Ordering::SeqCst); + assert!(failed.is_err()); + assert_eq!(quotas.used(Resource::Silent), size); + let refill = quotas.reserve(Some(&Key), Resource::Silent, size).unwrap(); + assert_eq!(quotas.used(Resource::Silent), 2 * size); + drop((donor, refill)); + assert_eq!(quotas.used(Resource::Silent), 0); + } + } +} + +/// Credit-window transition and full-width arithmetic contracts. +mod window_tests { + use flow_control::{Error, Window}; + + /// Invalid geometry never consumes credit. + #[test] + fn validates_configuration_and_item_lengths_without_consuming_credit() { + for (slots, bytes, max_item) in [(0, 1, 1), (1, 0, 1), (1, 1, 0)] { + assert!(matches!( + Window::::new(slots, bytes, max_item), + Err(Error::InvalidInput) + )); + } + let mut window = Window::new(2, 10, 6).unwrap(); + for length in [0, 7, u64::MAX] { + assert!(!window.can_reserve(length)); + assert_eq!(window.reserve("item", length), Err(Error::InvalidInput)); + assert!(window.is_empty()); + } + window.reserve("item", 6).unwrap(); + assert!(!window.is_empty()); + assert!(window.can_reserve(4)); + assert!(!window.can_reserve(5)); + } + + /// Only one exact acknowledgment of an issued item releases capacity. + #[test] + fn pending_and_issued_share_limits_and_release_exactly_once() { + let mut window = Window::new(2, 10, 10).unwrap(); + window.reserve(String::from("first"), 4).unwrap(); + assert_eq!(window.release("first".into(), 4), Err(Error::InvalidInput)); + window.issued("first".into()).unwrap(); + assert_eq!(window.issued("first".into()), Err(Error::InvalidInput)); + assert_eq!(window.reserve("first".into(), 1), Err(Error::Overloaded)); + window.reserve("second".into(), 6).unwrap(); + assert!(!window.can_reserve(1)); + assert_eq!(window.reserve("third".into(), 1), Err(Error::Overloaded)); + assert_eq!(window.release("first".into(), 3), Err(Error::InvalidInput)); + assert_eq!(window.release("first".into(), 5), Err(Error::InvalidInput)); + assert_eq!( + window.release("missing".into(), 4), + Err(Error::InvalidInput) + ); + assert_eq!(window.issued("missing".into()), Err(Error::InvalidInput)); + assert!(!window.can_reserve(1)); + window.release("first".into(), 4).unwrap(); + assert_eq!(window.release("first".into(), 4), Err(Error::InvalidInput)); + assert!(window.can_reserve(4)); + assert!(!window.can_reserve(5)); + window.issued("second".into()).unwrap(); + window.release("second".into(), 6).unwrap(); + assert!(window.is_empty()); + window.reserve("first".into(), 10).unwrap(); + window.issued("first".into()).unwrap(); + window.release("first".into(), 10).unwrap(); + assert!(window.is_empty()); + } + + /// Neither issuance nor oversized item ceilings weaken independent budgets. + #[test] + fn slot_and_byte_limits_are_independent() { + let mut slots = Window::new(1, 10, 10).unwrap(); + slots.reserve(1, 1).unwrap(); + assert!(!slots.can_reserve(1)); + assert_eq!(slots.reserve(2, 1), Err(Error::Overloaded)); + slots.issued(1).unwrap(); + assert!(!slots.can_reserve(1)); + slots.release(1, 1).unwrap(); + assert!(slots.can_reserve(10)); + let mut bytes = Window::new(100, 3, 10).unwrap(); + assert_eq!(bytes.reserve(1, 4), Err(Error::Overloaded)); + assert!(bytes.is_empty()); + bytes.reserve(1, 2).unwrap(); + assert!(bytes.can_reserve(1)); + assert_eq!(bytes.reserve(2, 2), Err(Error::Overloaded)); + bytes.reserve(2, 1).unwrap(); + assert!(!bytes.can_reserve(1)); + } + + /// Full-width arithmetic works without requiring cloneable keys. + #[test] + fn full_u64_budget_does_not_overflow_and_generic_keys_need_only_ord() { + /// Ordered fixture identity deliberately lacking Clone. + #[derive(Eq, PartialEq, Ord, PartialOrd)] + struct Key(u8); + let mut window = Window::new(usize::MAX, u64::MAX, u64::MAX).unwrap(); + window.reserve(Key(1), u64::MAX - 1).unwrap(); + window.reserve(Key(2), 1).unwrap(); + assert!(!window.can_reserve(1)); + assert_eq!(window.reserve(Key(3), 1), Err(Error::Overloaded)); + window.issued(Key(1)).unwrap(); + window.release(Key(1), u64::MAX - 1).unwrap(); + assert!(window.can_reserve(u64::MAX - 1)); + assert!(!window.can_reserve(u64::MAX)); + window.issued(Key(2)).unwrap(); + window.release(Key(2), 1).unwrap(); + assert!(window.is_empty()); + assert!(window.can_reserve(u64::MAX)); + } +} + +/// Public ownership boundaries and synchronous reentrant notification contracts. +mod coalesce_tests { + use flow_control::coalesce::flight::state::{Outcome, Phase, Published, State, WaiterPolicy}; + use flow_control::coalesce::flight::{self, Entry, Operations, Stale}; + use flow_control::coalesce::{CapacityError, Event, Limits, Table, shared}; + use futures::executor::block_on; + use std::cell::{Cell, RefCell}; + use std::future::Future; + use std::rc::Rc; + use std::sync::Arc; + use std::task::{Context, Poll, RawWaker, RawWakerVTable, Wake, Waker}; + use std::time::Instant; + + thread_local! { + /// Callback invoked only by synchronous wakes on the current test worker. + static ON_WAKE: RefCell>> = RefCell::new(None); + + /// One-shot hooks for raw waker ownership callbacks on this test worker. + static ON_COHORT_CLONE: RefCell>> = RefCell::new(None); + + static ON_COHORT_DROP: RefCell>> = RefCell::new(None); + } + + /// Stateless raw waker; callbacks use only the calling thread's test hooks. + fn cohort_raw_waker() -> RawWaker { + /// Run the clone hook without retaining its registry borrow. + unsafe fn clone(_: *const ()) -> RawWaker { + let callback = ON_COHORT_CLONE.with(|slot| slot.borrow_mut().take()); + if let Some(callback) = callback { + callback(); + } + cohort_raw_waker() + } + + /// Run the drop hook without retaining its registry borrow. + unsafe fn drop(_: *const ()) { + let callback = ON_COHORT_DROP.with(|slot| slot.borrow_mut().take()); + if let Some(callback) = callback { + callback(); + } + } + + /// Notifications need no action for these ownership callback tests. + unsafe fn wake(_: *const ()) {} + + RawWaker::new( + std::ptr::null(), + &RawWakerVTable::new(clone, wake, wake, drop), + ) + } + + /// Create a thread-safe stateless waker with worker-local test hooks. + fn cohort_waker() -> Waker { + // SAFETY: The vtable never dereferences data or shares thread-local hooks. + unsafe { Waker::from_raw(cohort_raw_waker()) } + } + + /// Register a flight wake target through the caller-owned borrow boundary. + fn register_flight_waker(owner: &RefCell>, waker: &Waker) { + let waker = waker.clone(); + let retired = flight::update(owner, |slot, _| flight::state::store_waker(slot, waker)); + drop(retired); + } + + /// Empty, different, and equal slots transfer ownership without callbacks. + #[test] + fn flight_store_waker_returns_retired_without_callbacks() { + let owner = RefCell::new(None); + let waker = cohort_waker(); + let cloned = Rc::new(Cell::new(false)); + let dropped = Rc::new(Cell::new(false)); + ON_COHORT_CLONE.with(|slot| { + let cloned = cloned.clone(); + *slot.borrow_mut() = Some(Box::new(move || cloned.set(true))); + }); + ON_COHORT_DROP.with(|slot| { + let dropped = dropped.clone(); + *slot.borrow_mut() = Some(Box::new(move || dropped.set(true))); + }); + let duplicate = cohort_waker(); + let retired = flight::update(&owner, |slot, _| { + assert!(flight::state::store_waker(slot, waker).is_none()); + let retired = flight::state::store_waker(slot, duplicate).unwrap(); + assert!(slot.as_ref().unwrap().will_wake(&retired)); + assert!(!cloned.get()); + assert!(!dropped.get()); + retired + }); + drop(retired); + assert!(dropped.get()); + let latest = Waker::from(Arc::new(Reenter)); + let replacement = latest.clone(); + let retired = flight::update(&owner, |slot, _| { + flight::state::store_waker(slot, replacement) + }); + assert!(retired.as_ref().unwrap().will_wake(&cohort_waker())); + assert!(owner.borrow().as_ref().unwrap().will_wake(&latest)); + assert!(!cloned.get()); + ON_COHORT_CLONE.with(|slot| slot.borrow_mut().take()); + } + + /// Clone callbacks can change the slot before registration borrows its owner. + #[test] + fn flight_store_waker_clone_reentry() { + let owner = Rc::new(RefCell::new(None)); + let called = Rc::new(Cell::new(false)); + ON_COHORT_CLONE.with(|slot| { + *slot.borrow_mut() = Some(Box::new({ + let owner = owner.clone(); + let called = called.clone(); + move || { + register_flight_waker(&owner, Waker::noop()); + called.set(true); + } + })); + }); + let waker = cohort_waker(); + register_flight_waker(&owner, &waker); + assert!(called.get()); + assert!(owner.borrow().as_ref().unwrap().will_wake(&waker)); + } + + /// Retired and redundant targets are dropped only after the owner is released. + fn check_flight_store_waker_drop_reentry(identical: bool) { + let owner = Rc::new(RefCell::new(None)); + let waker = cohort_waker(); + let latest = Waker::from(Arc::new(Reenter)); + register_flight_waker(&owner, &waker); + let called = Rc::new(Cell::new(false)); + ON_COHORT_DROP.with(|slot| { + *slot.borrow_mut() = Some(Box::new({ + let owner = owner.clone(); + let called = called.clone(); + let latest = latest.clone(); + move || { + let expected = if identical { + cohort_waker() + } else { + latest.clone() + }; + flight::update(&owner, |slot, _| { + assert!(slot.as_ref().unwrap().will_wake(&expected)); + }); + register_flight_waker(&owner, &latest); + called.set(true); + } + })); + }); + register_flight_waker(&owner, if identical { &waker } else { &latest }); + assert!(called.get()); + assert!(owner.borrow().as_ref().unwrap().will_wake(&latest)); + } + + /// Replacing a different target must return the old waker for deferred drop. + #[test] + fn flight_store_waker_replacement_drop_reentry() { + check_flight_store_waker_drop_reentry(false); + } + + /// Keeping the same target must return the unused incoming waker for deferred drop. + #[test] + fn flight_store_waker_identical_drop_reentry() { + check_flight_store_waker_drop_reentry(true); + } + + /// Clone and replacement callbacks may poll, retry, or finish the same cohort. + fn check_cohort_waker_event_reentry(on_clone: bool) { + for action in 0..3 { + let table = table(2); + let leader = table.join(1, 1).unwrap(); + let follower = table.join(1, 1).unwrap(); + assert_eq!(leader.event(Waker::noop()), Poll::Ready(Event::Lead)); + let waker = cohort_waker(); + if !on_clone { + assert!(follower.event(&waker).is_pending()); + } + let called = Rc::new(Cell::new(false)); + let callback: Box = Box::new({ + let leader = leader.clone(); + let follower = follower.clone(); + let called = called.clone(); + move || { + match action { + 0 => assert!(follower.event(Waker::noop()).is_pending()), + 1 => leader.retry(), + _ => leader.finish(7), + } + called.set(true); + } + }); + if on_clone { + ON_COHORT_CLONE.with(|slot| *slot.borrow_mut() = Some(callback)); + } else { + ON_COHORT_DROP.with(|slot| *slot.borrow_mut() = Some(callback)); + } + let event = follower.event(if on_clone { &waker } else { Waker::noop() }); + assert!(called.get()); + assert_eq!( + event, + match action { + 0 => Poll::Pending, + 1 => Poll::Ready(Event::Lead), + _ => Poll::Ready(Event::Complete(7)), + } + ); + assert_eq!(table.registration_count(), 2); + assert_eq!(table.active_count(), usize::from(action != 2)); + drop((leader, follower, waker)); + assert_eq!(table.registration_count(), 0); + assert_eq!(table.active_count(), 0); + } + } + + /// Cloning the incoming waker runs before borrowing or deciding the event. + #[test] + fn cohort_waker_clone_reentry() { + check_cohort_waker_event_reentry(true); + } + + /// Retiring the old waker runs before deciding the event from current state. + #[test] + fn cohort_waker_replacement_drop_reentry() { + check_cohort_waker_event_reentry(false); + } + + /// Detach updates charges and leadership before destroying the removed waker. + #[test] + fn cohort_waker_detach_drop_reentry() { + for drop_leader in [false, true] { + for action in 0..3 { + let table = table(2); + let leader = table.join(1, 1).unwrap(); + let follower = table.join(1, 1).unwrap(); + let waker = cohort_waker(); + assert_eq!(leader.event(&waker), Poll::Ready(Event::Lead)); + assert!(follower.event(&waker).is_pending()); + let (removed, remaining) = if drop_leader { + (leader, follower) + } else { + (follower, leader) + }; + let called = Rc::new(Cell::new(false)); + ON_COHORT_DROP.with(|slot| { + *slot.borrow_mut() = Some(Box::new({ + let table = table.clone(); + let remaining = remaining.clone(); + let called = called.clone(); + move || { + match action { + 0 => assert_eq!( + remaining.event(Waker::noop()), + if drop_leader { + Poll::Ready(Event::Lead) + } else { + Poll::Pending + } + ), + 1 => remaining.retry(), + _ => remaining.finish(7), + } + assert_eq!(table.registration_count(), 1); + called.set(true); + } + })); + }); + drop(removed); + assert!(called.get()); + if action == 1 { + assert_eq!(remaining.event(Waker::noop()), Poll::Ready(Event::Lead)); + } else if action == 2 { + assert_eq!( + remaining.event(Waker::noop()), + Poll::Ready(Event::Complete(7)) + ); + } + drop((remaining, waker)); + assert_eq!(table.registration_count(), 0); + assert_eq!(table.active_count(), 0); + } + } + } + + /// Last-owner waker destruction sees released capacity and can admit a new cohort. + #[test] + fn cohort_waker_last_detach_admits_replacement() { + let table = table(1); + let registration = table.join(1, 1).unwrap(); + let waker = cohort_waker(); + assert_eq!(registration.event(&waker), Poll::Ready(Event::Lead)); + let replacement = Rc::new(RefCell::new(None)); + ON_COHORT_DROP.with(|slot| { + *slot.borrow_mut() = Some(Box::new({ + let table = table.clone(); + let replacement = replacement.clone(); + move || { + assert_eq!(table.registration_count(), 0); + assert_eq!(table.active_count(), 0); + let next = table.join(1, 1).unwrap(); + assert_eq!(next.event(Waker::noop()), Poll::Ready(Event::Lead)); + replacement.replace(Some(next)); + } + })); + }); + drop(registration); + assert!(replacement.borrow().is_some()); + assert_eq!(table.registration_count(), 1); + assert_eq!(table.active_count(), 1); + drop((replacement, waker)); + assert_eq!(table.registration_count(), 0); + assert_eq!(table.active_count(), 0); + } + + /// Safe thread-local callback dispatch; no non-Send data enters the Waker itself. + struct Reenter; + + impl Wake for Reenter { + /// Invoke the current worker's callback after releasing its registry borrow. + fn wake(self: Arc) { + let callback = ON_WAKE.with(|slot| slot.borrow_mut().take()); + if let Some(callback) = callback { + callback(); + } + } + } + + /// Install a callback and return a wake target that dispatches it synchronously. + fn on_wake(callback: impl FnOnce() + 'static) -> Waker { + ON_WAKE.with(|slot| assert!(slot.borrow_mut().replace(Box::new(callback)).is_none())); + Waker::from(Arc::new(Reenter)) + } + + /// Build a table with enough attempts to test retry and final-owner-drop elections. + fn table(waiters: usize) -> Rc> { + Rc::new(Table::new( + Limits { + waiters_per_cohort: waiters, + attempts_per_cohort: 4, + }, + 99, + )) + } + + /// Joining and cloning handles do not require a cloneable result value. + #[test] + fn registration_clones_neither_keys_nor_results() { + /// Key whose clone records the allocation-time copy only. + #[derive(Eq, PartialEq)] + struct Key(Rc>); + + impl Clone for Key { + /// Count each actual key copy. + fn clone(&self) -> Self { + self.0.set(self.0.get() + 1); + Self(self.0.clone()) + } + } + + impl std::hash::Hash for Key { + /// Give the single logical test key a stable hash. + fn hash(&self, state: &mut H) { + // This test uses exactly one logical key, independent of the counter. + state.write_u8(0); + } + } + + /// Deliberately non-Clone result; only polling needs result cloning. + struct ResultValue; + + let table = Rc::new(Table::::new( + Limits { + waiters_per_cohort: 1, + attempts_per_cohort: 1, + }, + ResultValue, + )); + let copies = Rc::new(Cell::new(0)); + let first = table.join(Key(copies.clone()), 1).unwrap(); + assert_eq!(copies.get(), 1); + let second = first.clone(); + let third = second.clone(); + assert_eq!(copies.get(), 1); + first.finish(ResultValue); + drop((first, second)); + assert_eq!(table.registration_count(), 1); + drop(third); + assert_eq!(table.registration_count(), 0); + } + + /// Arbitrarily many cloned handles consume one charge until their final drop. + #[test] + fn many_clones_retain_completed_capacity_and_do_not_remove_replacements() { + let table = table(2); + let first = table.join(1, 1).unwrap(); + let mut clones = (0..32).map(|_| first.clone()).collect::>(); + first.finish(7); + let next = table.join(1, 1).unwrap(); + drop(first); + while clones.len() > 1 { + drop(clones.pop()); + assert_eq!(table.registration_count(), 2); + assert!(matches!(table.join(1, 1), Err(CapacityError))); + } + assert!(clones[0].is_only_handle()); + assert_eq!( + clones[0].event(Waker::noop()), + Poll::Ready(Event::Complete(7)) + ); + drop(clones); + assert_eq!(table.registration_count(), 1); + assert_eq!(table.active_count(), 1); + let follower = table.join(1, 1).unwrap(); + assert_eq!(next.event(Waker::noop()), Poll::Ready(Event::Lead)); + drop((next, follower)); + assert_eq!(table.registration_count(), 0); + assert_eq!(table.active_count(), 0); + } + + /// Retry and final-owner drop release all borrows before reentrant leader election. + #[test] + fn retry_and_final_drop_allow_reentrant_election() { + for drop_leader in [false, true] { + let table = table(2); + let leader = table.join(1, 1).unwrap(); + let follower = table.join(1, 1).unwrap(); + assert_eq!(leader.event(Waker::noop()), Poll::Ready(Event::Lead)); + let called = Rc::new(Cell::new(false)); + let waker = on_wake({ + let table = table.clone(); + let follower = follower.clone(); + let called = called.clone(); + move || { + assert_eq!(table.registration_count(), if drop_leader { 1 } else { 2 }); + assert_eq!(follower.event(Waker::noop()), Poll::Ready(Event::Lead)); + called.set(true); + } + }); + assert!(follower.event(&waker).is_pending()); + if drop_leader { + drop(leader); + } else { + leader.retry(); + } + assert!(called.get()); + } + } + + /// A follower may request notifications without revoking another waiter's leadership. + #[test] + fn follower_retry_notifies_without_taking_leadership() { + let table = table(2); + let leader = table.join(1, 1).unwrap(); + let follower = table.join(1, 1).unwrap(); + assert_eq!(leader.event(Waker::noop()), Poll::Ready(Event::Lead)); + let notified = Rc::new(Cell::new(false)); + let waker = on_wake({ + let leader = leader.clone(); + let follower = follower.clone(); + let notified = notified.clone(); + move || { + assert!(follower.event(Waker::noop()).is_pending()); + assert!(leader.event(Waker::noop()).is_pending()); + notified.set(true); + } + }); + assert!(follower.event(&waker).is_pending()); + follower.retry(); + assert!(notified.get()); + leader.retry(); + assert_eq!(follower.event(Waker::noop()), Poll::Ready(Event::Lead)); + } + + /// Completion removes old admission before any reader wake can admit new work. + #[test] + fn finish_allows_reentrant_admission_before_old_readers_detach() { + let table = table(3); + let leader = table.join(1, 1).unwrap(); + let follower = table.join(1, 1).unwrap(); + assert_eq!(leader.event(Waker::noop()), Poll::Ready(Event::Lead)); + let replacement = Rc::new(RefCell::new(None)); + let waker = on_wake({ + let table = table.clone(); + let follower = follower.clone(); + let replacement = replacement.clone(); + move || { + assert_eq!(table.active_count(), 0); + assert_eq!( + follower.event(Waker::noop()), + Poll::Ready(Event::Complete(7)) + ); + replacement.replace(Some(table.join(1, 1).unwrap())); + } + }); + assert!(follower.event(&waker).is_pending()); + leader.finish(7); + assert!(replacement.borrow().is_some()); + drop((leader, follower)); + assert_eq!(table.active_count(), 1); + assert_eq!(table.registration_count(), 1); + } + + /// Shared result notification can synchronously start replacement work. + #[test] + fn shared_completion_wakes_after_removal_and_keeps_new_owner() { + let table = Rc::new(shared::Table::default()); + let (mut receive, complete) = table.start(1, 99); + let replacement = Rc::new(RefCell::new(None)); + let waker = on_wake({ + let table = table.clone(); + let replacement = replacement.clone(); + move || { + assert!(table.is_empty()); + replacement.replace(Some(table.start(1, 99))); + } + }); + assert!( + std::pin::Pin::new(&mut receive) + .poll(&mut Context::from_waker(&waker)) + .is_pending() + ); + complete.unwrap().finish(7); + assert_eq!(block_on(receive), 7); + let (receive, complete) = replacement.borrow_mut().take().unwrap(); + assert_eq!(table.len(), 1); + drop(complete); + assert_eq!(block_on(receive), 99); + assert_eq!( + table.len(), + 1, + "sender loss cannot pretend execution completed" + ); + } + + /// Losing the sender wakes parked readers without removing the indexed work. + #[test] + fn shared_sender_drop_wakes_with_entry_still_indexed() { + let table = Rc::new(shared::Table::default()); + let (mut receive, complete) = table.start(1, 99); + let notified = Rc::new(Cell::new(false)); + let waker = on_wake({ + let table = table.clone(); + let notified = notified.clone(); + move || { + assert_eq!(table.len(), 1); + assert!(table.get(&1).is_some()); + notified.set(true); + } + }); + assert!( + std::pin::Pin::new(&mut receive) + .poll(&mut Context::from_waker(&waker)) + .is_pending() + ); + drop(complete); + assert!(notified.get()); + assert_eq!(block_on(receive), 99); + assert_eq!(block_on(table.get(&1).unwrap()), 99); + assert_eq!(table.len(), 1); + } + + /// Resource whose destructor can synchronously inspect its owning table. + struct ReentrantResource(Option>); + + impl Drop for ReentrantResource { + /// Reenter the owner while its operation tombstone must still be present. + fn drop(&mut self) { + self.0.take().unwrap()(); + } + } + + /// Only operation completion makes this deliberately waiter-free entry removable. + #[derive(Default)] + struct OwnedEntry(Operations); + + impl Entry for OwnedEntry { + /// This fixture has no caller policy to refresh. + fn refresh(&mut self, _: &mut Vec) {} + + /// Require explicit completion of every retained operation. + fn quiescent(&self) -> bool { + self.0.is_empty() + } + } + + /// A removable entry can still own application data with a reentrant destructor. + struct DroppableEntry { + _resource: ReentrantResource, + + quiescent: bool, + } + + impl Entry for DroppableEntry { + fn refresh(&mut self, _: &mut Vec) {} + + fn quiescent(&self) -> bool { + self.quiescent + } + } + + /// Rejected entries leave the transaction before their destructors reenter it. + #[test] + fn occupied_insertion_keeps_destructors_outside_owner_borrow() { + for quiescent in [false, true] { + let table = Rc::new(RefCell::new(flight::Table::::default())); + let original_dropped = Rc::new(Cell::new(false)); + let rejected_dropped = Rc::new(Cell::new(false)); + let unlocked = Rc::new(Cell::new(false)); + let entry = |dropped: Rc>| DroppableEntry { + _resource: ReentrantResource(Some(Box::new({ + let table = Rc::downgrade(&table); + let unlocked = unlocked.clone(); + move || { + let Some(table) = table.upgrade() else { + return; + }; + if let Ok(mut table) = table.try_borrow_mut() { + assert!(table.next_waiter_id().is_ok()); + unlocked.set(true); + } + dropped.set(true); + } + }))), + quiescent, + }; + assert!( + flight::update(&table, |table, _| { + table.insert(1, entry(original_dropped.clone())) + }) + .is_ok() + ); + let rejected = flight::update(&table, |table, _| { + table.insert(1, entry(rejected_dropped.clone())) + }); + assert!(rejected.is_err()); + assert!(!original_dropped.get(), "occupied entry must stay owned"); + assert!(!rejected_dropped.get(), "rejection must return ownership"); + drop(rejected); + assert!(rejected_dropped.get()); + assert!(unlocked.get()); + assert!(!original_dropped.get()); + let removed = flight::update(&table, |table, _| { + assert_eq!(table.len(), 1); + table.get_mut(&1).unwrap().quiescent = true; + table.remove_quiescent(&1) + }); + // The original destructor also reenters after explicit removal. + unlocked.set(false); + drop(removed); + assert!(original_dropped.get()); + assert!(unlocked.get()); + } + } + + /// Sweeping transfers entry destruction past the owner transaction. + #[test] + fn swept_entry_destructor_can_reenter_owner() { + removed_entry_destructor_can_reenter_owner(true); + } + + /// Explicit removal has the same destruction boundary as a sweep. + #[test] + fn detached_entry_destructor_can_reenter_owner() { + removed_entry_destructor_can_reenter_owner(false); + } + + /// Check both removal paths, including ineligible entries and same-key replacement. + fn removed_entry_destructor_can_reenter_owner(sweep: bool) { + let table = Rc::new(RefCell::new(flight::Table::::default())); + let dropped = Rc::new(Cell::new(0)); + let resource = ReentrantResource(Some(Box::new({ + let table = Rc::downgrade(&table); + let dropped = dropped.clone(); + move || { + let table = table.upgrade().unwrap(); + drop(flight::update(&table, |table, wakes| { + assert!(table.is_empty()); + let removed = table.sweep(1, wakes); + assert!(removed.is_empty()); + assert_eq!(table.next_waiter_id(), Ok(1)); + assert!( + table + .insert( + 1, + DroppableEntry { + _resource: ReentrantResource(Some(Box::new(|| {}))), + quiescent: false, + }, + ) + .is_ok() + ); + removed + })); + dropped.set(dropped.get() + 1); + } + }))); + assert!( + table + .borrow_mut() + .insert( + 1, + DroppableEntry { + _resource: resource, + quiescent: false, + }, + ) + .is_ok() + ); + let removed = flight::update(&table, |table, wakes| { + assert!(table.remove_quiescent(&2).is_none()); + assert!(table.remove_quiescent(&1).is_none()); + assert!(table.sweep(1, wakes).is_empty()); + table.get_mut(&1).unwrap().quiescent = true; + assert!(table.sweep(0, wakes).is_empty()); + assert_eq!(table.len(), 1); + let removed = if sweep { + table.sweep(1, wakes) + } else { + table.remove_quiescent(&1).into_iter().collect() + }; + assert_eq!(removed.len(), 1); + assert!(table.is_empty()); + assert!(table.remove_quiescent(&1).is_none()); + assert_eq!(dropped.get(), 0); + removed + }); + assert_eq!(dropped.get(), 0); + drop(removed); + assert_eq!(dropped.get(), 1); + assert_eq!(table.borrow_mut().next_waiter_id(), Ok(2)); + drop(flight::update(&table, |table, wakes| table.sweep(1, wakes))); + assert_eq!(table.borrow().len(), 1, "replacement remains indexed"); + assert_eq!(dropped.get(), 1); + } + + /// Resource drop reentrancy cannot erase the tombstone, and drain wakes run unlocked. + #[test] + fn completion_tombstone_survives_reentrant_destructor_and_shutdown() { + let table = Rc::new(RefCell::new(flight::Table::::default())); + let dropped = Rc::new(Cell::new(false)); + let resource = ReentrantResource(Some(Box::new({ + let table = Rc::downgrade(&table); + let dropped = dropped.clone(); + move || { + let table = table.upgrade().unwrap(); + drop(flight::update(&table, |table, wakes| { + let removed = table.sweep(1, wakes); + assert_eq!(table.len(), 1); + assert!(table.remove_quiescent(&1).is_none()); + removed + })); + dropped.set(true); + } + }))); + let id = flight::update(&table, |table, _| { + assert_eq!(table.next_waiter_id(), Ok(1)); + assert_eq!(table.next_waiter_id(), Ok(2)); + let id = table.next_operation_id().unwrap(); + assert!(table.insert(1, OwnedEntry::default()).is_ok()); + table.get_mut(&1).unwrap().0.insert(id, resource); + assert_eq!(table.get_mut(&1).unwrap().0.complete(id), Err(Stale)); + id + }); + flight::update(&table, |table, wakes| table.stop(wakes, |_, _| {})); + assert!(table.borrow().is_stopping()); + let resource = flight::update(&table, |table, _| { + let operations = &mut table.get_mut(&1).unwrap().0; + let resource = operations.take(id).unwrap(); + assert!(matches!(operations.take(id), Err(Stale))); + assert!(matches!(operations.take(id + 1), Err(Stale))); + assert_eq!(operations.complete(id + 1), Err(Stale)); + assert_eq!(operations.len(), 1); + resource + }); + drop(resource); + assert!(dropped.get()); + assert_eq!(table.borrow().len(), 1); + let notified = Rc::new(Cell::new(false)); + let waker = on_wake({ + let table = table.clone(); + let notified = notified.clone(); + move || { + assert!(table.borrow().is_empty()); + notified.set(true); + } + }); + drop(flight::update(&table, |table, _| { + table.register_drain(waker) + })); + drop(flight::update(&table, |table, wakes| { + let operations = &mut table.get_mut(&1).unwrap().0; + operations.complete(id).unwrap(); + assert_eq!(operations.complete(id), Err(Stale)); + table.sweep(1, wakes) + })); + assert!(notified.get()); + } + + /// An expired leader is checked separately from the 64-entry deadline quantum. + #[test] + fn expired_leader_does_not_consume_deadline_quantum_or_clear_drain_fence() { + let mut state = State::, CountingPolicy>::default(); + let mut table = flight::Table::::default(); + let mut identity = table.identity(Rc::new(())).unwrap(); + let now = Instant::now(); + let checks = Rc::new(Cell::new(0)); + let error = Rc::new(Cell::new(None)); + for id in 0..66 { + state.register( + id, + CountingPolicy { + due: now, + checks: checks.clone(), + error: error.clone(), + }, + true, + true, + ); + } + assert_eq!(state.elect(0, &mut identity, 2), Ok(true)); + error.set(Some(7)); + let mut wakes = Vec::new(); + state.refresh(false, 9, 8, || now, std::convert::identity, &mut wakes); + assert_eq!(checks.get(), 65); + assert_eq!(state.deadlines.len(), 1); + assert_eq!(state.waiters[&0].error, Some(7)); + assert!(state.waiters[&65].error.is_none()); + assert!(matches!(state.phase, Phase::Draining(Outcome::Retry))); + assert_eq!(state.elect(65, &mut identity, 2), Ok(false)); + assert_eq!(identity.generation, 1); + + state.refresh(false, 9, 8, || now, std::convert::identity, &mut wakes); + assert_eq!(checks.get(), 66); + assert!(state.deadlines.is_empty()); + assert!(matches!(state.phase, Phase::Draining(Outcome::Retry))); + state.settle(true, 9, std::convert::identity, &mut wakes); + assert!(matches!(state.phase, Phase::Failed(9))); + } + + /// Fixed-deadline policy with observable checks and externally injected failures. + struct CountingPolicy { + due: Instant, + + checks: Rc>, + + error: Rc>>, + } + + impl WaiterPolicy for CountingPolicy { + /// Identify the failure injected into a waiter. + type Error = u8; + + /// Count policy evaluation and return the current injected failure. + fn check(&self) -> Option { + self.checks.set(self.checks.get() + 1); + self.error.get() + } + + /// Return the fixed deadline shared by this test's waiters. + fn deadline(&self) -> Instant { + self.due + } + } +} diff --git a/designs/racer-page-alloc.md b/designs/racer-page-alloc.md new file mode 100644 index 000000000..3f588954a --- /dev/null +++ b/designs/racer-page-alloc.md @@ -0,0 +1,206 @@ +# Racer page allocator (`page-alloc`) + +## Summary + +`page-alloc` is the storage layer under the new Racer dataplane. It lives in +`cmd/racer-dataplane/alloc/`. It gives one worker thread three things: + +1. Memory buffers that are aligned for direct I/O and wiped before reuse. +2. A table of fixed-size segments in a sparse cache file or caller-opened files + or devices, with leases that stop reuse while I/O still points at a segment. +3. Async read and write of that storage through the `uring-runtime` reactor. + +The crate stores bytes only. It does not know about page keys, record headers, +encryption, checksums, or versions. The caller owns the page index and +tells the crate how to remove entries when a segment is evicted. No Racer +binary uses the crate yet. The dataplane executable is still a placeholder. + +## Goals and non-goals + +Goals: + +- Never reuse disk space or memory while the kernel may still read or write it. +- Keep every allocation and reuse decision on one worker, with no locks. +- Bound the work done by each reclaim call, so the hot path never stalls. +- Fail with an error instead of wrapping counters or truncating files. + +Non-goals: + +- Durability. Writes are not followed by `fsync`, and there is no log. +- Secure erasure of the file. Eviction changes metadata only. Old bytes stay on + disk until they are overwritten. +- Compaction, record integrity, or deciding when a page is published. + +## Threading model + +Live allocation and I/O authority is worker-local. `Segments` and `Slab` use +`Rc`-owned state and are neither `Send` nor `Sync` +(see `Segments` in `alloc/src/segments.rs` and `Slab` in `alloc/src/slab.rs`). +Buffers, leases, freeze guards, and the reclamation clock have the same restriction +(see `AlignedBuffer` in `alloc/src/lib.rs` and `SegmentLease`, `FreezeGuard`, and +`SegmentClock` in `alloc/src/segments.rs`). +Value types such as `Alignment` and `SegmentId`, and startup inputs such as +`DevicePlacement`, are `Send + Sync` (see their declarations in `alloc/src/lib.rs`, +`alloc/src/segments.rs`, and `alloc/src/slab.rs`, respectively). Each worker owns its +storage ranges, segment table, and buffer pool; live allocation and I/O authority +cannot move to another worker. + +## Buffers + +`AlignedBuffer` is a heap allocation from `alloc_zeroed` with the alignment that +the file needs (see `Alignment::allocate` in `alloc/src/lib.rs`). Lengths are +padded to the least common multiple of the offset and length units, so the next +record also starts aligned. +A single buffer is at most 1 GiB (see `Alignment::extent` and +`Alignment::MAX_TRANSFER_LENGTH` in `alloc/src/lib.rs`). + +Each buffer holds a caller-supplied `Charge`, so the caller can account for +memory against its own budget. The slab keeps at most one idle buffer. +After checking charge coverage, allocation reuses it only for an exact length +match; otherwise it frees the idle buffer and allocates new storage +(see `Slab::allocate` in `alloc/src/slab.rs`). On drop, a buffer fills the idle slot +only if the pool still exists and the slot is empty and can be mutably borrowed; otherwise +its storage is freed (see the `Drop` implementations for `AlignedBuffer` and +`Allocation` in `alloc/src/lib.rs`). +The retained size depends on return order, not necessarily the last size used. +This is not a general size-class pool. + +Each buffer tracks whether it is still all zeros. Any mutable access, including +handing it to the kernel for a read, marks it dirty. On drop, a dirty buffer is +wiped in full, including padding, with `explicit_bzero` (or `zeroize` where that +is not available) before it is pooled or freed (see `Allocation::as_mut_slice`, +`Allocation::wipe`, and the `IoBuffer` implementation for `AlignedBuffer` in +`alloc/src/lib.rs`). +Clean buffers skip the wipe. This keeps old page data from leaking into the +next request without paying for a wipe on every allocation. + +## Segments + +Storage is split into fixed-size logical segments. Segment `n` starts at +`n * segment_bytes`; device placements map it to a caller-supplied physical range. +Each segment has a state and a generation number: + +``` +Free -> Open -> Sealed -> Evicting -> Free (generation + 1) +``` + +- `append` reserves space at the end of the one open segment. When it is full, + the segment is sealed and the lowest free segment is opened. If none is free, + the call returns `Busy` after sealing, so reclaim can make room. +- `append` and `lease` return a `SegmentLease`. A lease records the segment ID, + generation, and how much of the segment was in use when it was taken. While + any lease exists, the segment cannot go back to `Free`. +- The caller stores `(segment, generation, extent)` in its own index. A lookup + with an old generation fails with `Stale`. This rejects stale mappings, not + stale bytes read through a lease for the current generation. + +Append reserves space without writing or initializing disk bytes. Neither a +`SegmentLease` nor a successful read proves that bytes were initialized in the +current generation. Recycled storage may still hold bytes from an earlier +generation or a different cache. Before exposing bytes as a valid record, the +caller must check record integrity, authentication, and cache identity (see +`SegmentLease` and `Segments::append` in `alloc/src/segments.rs`, and `Slab::read` +in `alloc/src/slab.rs`). + +`freeze`, `snapshot`, and `restore` support restart. Restore checks the whole +image before it changes anything, requires that no leases are live, and seals +any segment that was open before the restart. + +## Reclaim + +`SegmentClock::reclaim` in `alloc/src/segments.rs` is a bounded clock +(second-chance) sweep over sealed segments. The caller calls `mark_read` when it +serves a read from a segment. The sweep clears that mark once before it picks the +segment. Each call has limits on the number of segments it visits and the number +of index entries it removes. +For a nonzero reserve, `reclaim` and `reclaim_scored` succeed only when the reserve +(capped at the slot count) is met and no evictions remain. They intentionally +return `Busy` while evictions remain, even if enough slots are already free. +Callers should use `Segments::free_count` to check capacity and keep retrying +bounded reclamation later to drain pending evictions. A zero reserve is a no-op, +not a drain request. + +Evicting a segment happens in this order: + +1. Ask the caller `can_evict`. The caller says no while an unpublished write + still needs the segment. This check happens only before eviction starts. +2. Mark the segment `Evicting`. New leases are refused. +3. Call `remove_bounded` to drop the caller's index entries, a few per call. +4. When the index is empty and all leases are gone, bump the generation and + mark the segment `Free`. + +The key rule is: remove the index entries first, then wait for in-flight I/O, +then reuse. `reclaim_scored` visits at most `min(slot count, max_visits, 64)` +slots, including skipped slots. It ranks eligible candidates in that sample by a +caller-provided score instead of recent reads. The limit is on visited slots, +not eligible candidates. `reclaim_index` drops index entries one at a time until a +caller check (for example, an index size limit) passes. It does not free segments +and does not ask `can_evict`. + +## Storage and I/O + +`Slab::new` describes one cache file. Opening is blocking and is meant to run at +startup (see `Slab::open_configured` and `Slab::open_file` in `alloc/src/slab.rs`). +For this file-backed mode, it: + +- Walks the path without following symlinks or `..`. +- Requires a regular file owned by the current user, mode 0600, one hard link. +- Takes an exclusive non-blocking `flock` and enables `O_DIRECT`. +- Reads the required alignment from `statx(STATX_DIOALIGN)`. +- Sizes an empty file sparsely to capacity. A file with the wrong size is + rejected, not truncated. + +See `Slab::open_file`, `Slab::validate_layout`, `open_private_file`, +`validate_file`, and `probe_fd` in `alloc/src/slab.rs` for these checks. + +`Slab::from_devices` instead owns caller-opened file or block-device placements, +one per logical segment, without creating, sizing, or locking them. Files must +be read/write with `O_DIRECT` and without `O_APPEND`. The constructor checks +geometry, offset alignment, regular-file length, and block-device capacity from +`BLKGETSIZE64`. Overlap checks only compare the same inode or device identity +within the slab; they cannot detect whole-disk, partition, or device-mapper +aliases. The caller must guarantee disjoint physical storage across aliases and +slabs, keep exclusive ownership, supply suitable alignment, and keep file flags +and sizes unchanged while in use (see `Slab::from_devices` in `alloc/src/slab.rs`). + +For device placements, `open_configured` duplicates the owned files into +worker-local descriptors and releases the original placement references +(see `Slab::open_file` in `alloc/src/slab.rs`). In both modes, it then binds the +slab to one segment table. I/O is refused until binding succeeds, and a slab cannot be +rebound to a different table (see `Slab::bind` and `Slab::submission` in +`alloc/src/slab.rs`). + +`read` and `write` check the extent, alignment, and lease, then pass the buffer +and lease to the reactor (see `Slab::submission`, `Submission::read`, and +`Submission::write` in `alloc/src/slab.rs`). The reactor holds both +until the kernel reports completion, even if the caller drops the future or +cancels. So the memory and the segment both stay reserved until the kernel is +done with them. Short reads and writes are returned as errors. + +Any lease, including one from `Segments::lease`, allows reads and writes within +its captured used prefix, not just the latest append. It does not grant exclusive +record ownership. The trusted caller must write only reserved extents it owns, +never overwrite published or readable records, and publish a mapping only after +the write succeeds (see `Slab::write` in `alloc/src/slab.rs`). + +`fence_writes` waits until no write is in flight. It is a count, not a snapshot, +and it does not flush to disk. + +## Simulation + +For file-backed slabs, the `simulation` feature routes open, lock, stat, and sizing to the +`uring-runtime` simulated filesystem. Buffers, leases, and segment rules do not +change, so the same workflow tests run in both modes. + +## Testing + +- Unit tests cover padding math, wipe and reuse rules, lease and generation + checks, reclaim limits, the eviction veto, restore, and file security checks. +- `alloc/tests/workflows.rs` covers restart, short I/O, dropped and canceled + reads and writes, startup failures, and confirms that reuse does not erase + disk bytes. +- CI checks and tests the production and simulation builds as separate + commands, so feature unification cannot hide one from the other. The + production run sets `PAGE_ALLOC_REQUIRE_REAL_IO=1`, so a host without + io_uring or direct I/O fails instead of skipping. Miri runs only on + `uring-runtime`, not on this crate. diff --git a/internal/gantry/mirror/mirror_coldstart_test.go b/internal/gantry/mirror/mirror_coldstart_test.go index 9ff432f8f..b107534af 100644 --- a/internal/gantry/mirror/mirror_coldstart_test.go +++ b/internal/gantry/mirror/mirror_coldstart_test.go @@ -86,14 +86,14 @@ func (d *countingPeerDialer) Calls(addr string) int { func (s *stubColdStart) Resolve(_ context.Context, d digest.Digest, _ ifaces.OriginRefKind, _, _ string, _ int64) (*mirror.ColdStartResolution, error) { atomic.AddInt32(&s.calls, 1) - if s.err != nil { - return nil, s.err - } - if s.onResolve != nil { s.onResolve(d) } + if s.err != nil { + return nil, s.err + } + return &mirror.ColdStartResolution{Providers: s.providers, Outcome: "stub"}, nil } diff --git a/internal/gantry/mirror/rediscover_test.go b/internal/gantry/mirror/rediscover_test.go index e59386897..564c58b62 100644 --- a/internal/gantry/mirror/rediscover_test.go +++ b/internal/gantry/mirror/rediscover_test.go @@ -156,7 +156,12 @@ func TestMirror_Rediscover_ColdExhaustedFlushesHeadersBeforeLateProvider(t *test dialer.Put(lateAddr, d, body) dht := fakes.NewDHT() - coldStart := &stubColdStart{err: mirror.ErrColdStartExhausted} + coldStartEntered := make(chan struct{}) + signalColdStart := sync.OnceFunc(func() { close(coldStartEntered) }) + coldStart := &stubColdStart{ + err: mirror.ErrColdStartExhausted, + onResolve: func(digest.Digest) { signalColdStart() }, + } m := mirror.New(cfg, fakes.NewCache(), oc, mirror.WithLiveStreamThrough(), @@ -188,6 +193,13 @@ func TestMirror_Rediscover_ColdExhaustedFlushesHeadersBeforeLateProvider(t *test t.Fatalf("peer calls before advertise = %d, want 0", got) } + // Headers flush before round 0; wait until its empty lookup reaches cold-start. + select { + case <-coldStartEntered: + case <-time.After(2 * time.Second): + t.Fatal("cold-start was not entered before provider advertisement") + } + dht.Inject(d, ifaces.Provider{NodeID: "late-seed", Addr: lateAddr}) got, err := io.ReadAll(resp.Body)