From c7dbf6d87aa6d6def285b35ece8f06d12aa6d107 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Mon, 5 Oct 2026 21:59:17 +0000 Subject: [PATCH 01/82] Add incremental Racer workspace CI for stacked PRs --- .github/workflows/ci.yaml | 1 + 1 file changed, 1 insertion(+) diff --git a/.github/workflows/ci.yaml b/.github/workflows/ci.yaml index f55d84816..f6755256a 100644 --- a/.github/workflows/ci.yaml +++ b/.github/workflows/ci.yaml @@ -13,6 +13,7 @@ on: branches: - main - release-* + - pr/racer-* merge_group: types: - checks_requested From c0adfdd3860d5529d43dd4c6643b3e6615578337 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Mon, 5 Oct 2026 17:20:38 -0500 Subject: [PATCH 02/82] Update CI workflow to exclude 'pr/racer-*' branches Removed 'pr/racer-*' branch from pull request triggers. --- .github/workflows/ci.yaml | 1 - 1 file changed, 1 deletion(-) diff --git a/.github/workflows/ci.yaml b/.github/workflows/ci.yaml index f6755256a..f55d84816 100644 --- a/.github/workflows/ci.yaml +++ b/.github/workflows/ci.yaml @@ -13,7 +13,6 @@ on: branches: - main - release-* - - pr/racer-* merge_group: types: - checks_requested From e0c6e809005cddcc968e6c2358e54093d6dfeedf Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Mon, 5 Oct 2026 22:14:04 +0000 Subject: [PATCH 03/82] feat(racer): import page allocator crate --- cmd/racer-dataplane/Cargo.lock | 9 + cmd/racer-dataplane/Cargo.toml | 2 +- cmd/racer-dataplane/alloc/.gitignore | 1 + cmd/racer-dataplane/alloc/Cargo.toml | 16 + cmd/racer-dataplane/alloc/src/lib.rs | 1018 ++++++++++++ cmd/racer-dataplane/alloc/src/segments.rs | 1534 ++++++++++++++++++ cmd/racer-dataplane/alloc/src/slab.rs | 1314 +++++++++++++++ cmd/racer-dataplane/alloc/tests/workflows.rs | 1001 ++++++++++++ 8 files changed, 4894 insertions(+), 1 deletion(-) create mode 100644 cmd/racer-dataplane/alloc/.gitignore create mode 100644 cmd/racer-dataplane/alloc/Cargo.toml create mode 100644 cmd/racer-dataplane/alloc/src/lib.rs create mode 100644 cmd/racer-dataplane/alloc/src/segments.rs create mode 100644 cmd/racer-dataplane/alloc/src/slab.rs create mode 100644 cmd/racer-dataplane/alloc/tests/workflows.rs diff --git a/cmd/racer-dataplane/Cargo.lock b/cmd/racer-dataplane/Cargo.lock index 020540f57..a650c6bcd 100644 --- a/cmd/racer-dataplane/Cargo.lock +++ b/cmd/racer-dataplane/Cargo.lock @@ -184,6 +184,15 @@ version = "2.8.3" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98" +[[package]] +name = "page-alloc" +version = "0.1.0" +dependencies = [ + "libc", + "uring-runtime", + "zeroize", +] + [[package]] name = "pin-project-lite" version = "0.2.17" diff --git a/cmd/racer-dataplane/Cargo.toml b/cmd/racer-dataplane/Cargo.toml index cfd053c36..a840bccdb 100644 --- a/cmd/racer-dataplane/Cargo.toml +++ b/cmd/racer-dataplane/Cargo.toml @@ -1,5 +1,5 @@ [workspace] -members = [".", "topology", "runtime"] +members = [".", "topology", "runtime", "alloc"] resolver = "3" [package] diff --git a/cmd/racer-dataplane/alloc/.gitignore b/cmd/racer-dataplane/alloc/.gitignore new file mode 100644 index 000000000..b83d22266 --- /dev/null +++ b/cmd/racer-dataplane/alloc/.gitignore @@ -0,0 +1 @@ +/target/ diff --git a/cmd/racer-dataplane/alloc/Cargo.toml b/cmd/racer-dataplane/alloc/Cargo.toml new file mode 100644 index 000000000..c49074d93 --- /dev/null +++ b/cmd/racer-dataplane/alloc/Cargo.toml @@ -0,0 +1,16 @@ +[package] +name = "page-alloc" +version = "0.1.0" +edition = "2024" +license = "Apache-2.0" +publish = false +description = "Worker-local aligned buffers and lease-fenced sparse slab storage" + +[features] +default = [] +simulation = ["uring-runtime/simulation"] + +[dependencies] +libc = "0.2" +zeroize = "1" +uring-runtime = { path = "../runtime" } diff --git a/cmd/racer-dataplane/alloc/src/lib.rs b/cmd/racer-dataplane/alloc/src/lib.rs new file mode 100644 index 000000000..56de48150 --- /dev/null +++ b/cmd/racer-dataplane/alloc/src/lib.rs @@ -0,0 +1,1018 @@ +//! Worker-local aligned storage with caller-owned accounting and I/O policy. +//! +//! This internal crate owns generic bytes, not keys, records, encryption, integrity +//! checks, versions, admission classes, or a persistence protocol. The caller +//! supplies accounting guards, a reactor, cancellation policy, and index callbacks. +//! There is no assumed OS page size and no cross-thread allocation authority. +//! +//! # Startup and binding +//! +//! Construct [`Slab::new`] with a full file path and [`Segments::new`] with the +//! same segment size, then call [`Slab::open_configured`] outside the worker's +//! latency-sensitive path. Directory traversal, locking, sizing, and probing are +//! blocking. Alignment comes from `statx(STATX_DIOALIGN)` rather than a fixed page +//! size. A configured partial table must match capacity, segment size, and +//! alignment. The slab reports physical geometry even with a partial table. +//! +//! Opening and binding are separate steps: failed binding can leave the file open +//! and locked but cannot authorize I/O. Fix the table and retry or drop the slab. +//! Rebinding to a different table is rejected even when its geometry matches. +//! The maximum record size only checks that its padded size fits a segment at +//! startup; it is not an allocation quota or a per-submission record-size limit. +//! +//! Linux traversal rejects parent components and intermediate/final symlinks. +//! Missing directories and files are created with modes 0700 and 0600, subject to +//! umask. Accepted files must be regular, owned by the effective user, singly +//! linked, and exactly mode 0600 with no special bits. Permissions are not repaired. +//! Nonblocking open prevents FIFOs from hanging startup before type validation. +//! Files are close-on-exec, exclusively locked with nonblocking flock, and use +//! direct I/O. Parent directories still must be trusted against hostile rename +//! and unlink; neither traversal nor flock stops noncooperating writers. +//! +//! Empty files are sparsely extended to capacity; nonempty size mismatches are +//! rejected without truncation. Capacity is a logical bound, not reserved disk +//! space, so later writes can fail with ENOSPC. Recycling changes metadata only: +//! it does not erase, truncate, or hole-punch disk bytes. Buffer zeroization is +//! not secure erasure of the file, and physical blocks can survive eviction. +//! +//! # Alignment and accounting +//! +//! [`Alignment`] permits arbitrary positive offset and length units; memory +//! alignment must be a supported power of two. Extent padding uses their least +//! common multiple, so offset unit 768 and length unit 512 require multiples of +//! 1536. Arithmetic and the conservative 1 GiB transfer cap are checked before +//! allocating. Larger logical records must be split by the caller. An [`Extent`] +//! alone checks a range, not its alignment or permission to access a segment. +//! +//! [`AlignedBuffer`] owns initialized stable storage that cannot be resized. +//! The primary [`Charge`] must truthfully account for its entire padded size; +//! `()` opts out of accounting. Additional guards attached with +//! [`AlignedBuffer::retain`] remain live through completion, but not idle pooling. +//! [`Slab::allocate`] maintains one exact-size idle slot. Drop zeroizes bytes +//! before pooling or freeing. Idle memory retains the primary charge. Reuse +//! replaces it with the newly admitted charge; undercharging leaves the idle +//! slot untouched. A size mismatch releases old storage even if replacement +//! fails. Occupied, borrowed, or dropped pools cannot retain returned storage. +//! Outstanding buffers do not keep the pool alive. Reclaim reaches idle bytes +//! only, never memory still owned by the kernel. +//! +//! # Completion ownership +//! +//! For writes, round the logical size, append the padded length, allocate matching +//! storage, copy bytes into the zeroed buffer, and submit [`Slab::write`]. Publish +//! the caller's mapping only after handling completion. Append reserves space and +//! does not roll it back on failed writes. For reads, validate the stored slot, +//! generation, and extent, acquire a lease, allocate storage, and submit a read. +//! A lease authorizes the segment's used prefix at acquisition, not just the last +//! appended record; it cannot authorize bytes appended afterward. +//! +//! Submission checks table identity, alignment, length, segment boundaries, and +//! captured used bytes. Reads and writes reject short completions. Accepted I/O +//! owns the buffer, all charges, and the lease through the runtime completion +//! fence, even when its waiting future is dropped. Writes also own their counter +//! guard. An unpolled future has submitted nothing. Continue driving the reactor +//! to complete or cancel accepted work before expecting resources to be released. +//! [`Slab::fence_writes`] neither waits for reads nor prevents new writes. Stop +//! admitting writes first if quiescence is required. It is a completion fence, +//! not a durability barrier: it never calls fsync or fdatasync. +//! +//! # Recovery and reclamation +//! +//! Slots normally progress from Free to Open to Sealed to Evicting to Free. +//! Append selects a fitting open tail or the lowest free slot. Full segments and +//! rotated tails become sealed. A valid rollover with no free slot seals the old +//! tail before returning Busy, enabling reclamation and retry. Malformed requests +//! and lease-counter overflow leave the table unchanged. New leases reject +//! eviction, but existing leases remain usable. Recycling requires no remaining +//! leases and increments generation without wrapping. +//! +//! A [`FreezeGuard`] blocks table mutation, not reads, snapshots, new read leases, +//! caller index mutation, or accepted I/O. It may outlive its table. Only one guard +//! exists at a time. Restore requires both thawing and draining all leases, +//! including reads and abandoned I/O. It validates the whole ordered image before +//! publication, seals an open tail, rebuilds free slots, and advances an epoch +//! that clears the eviction cursor and recent-read state. Invalid or busy restores +//! never partially publish. +//! Raw file opening, table configuration, and manual eviction are simulation-only +//! escape hatches. Production startup binds with [`Slab::open_configured`], and +//! [`SegmentClock`] owns the ordering of mapping removal before physical reuse. +//! The table, slab, and clock are cache-line aligned to isolate worker-local state; +//! their Rc ownership prevents moving authority to a different worker thread. +//! +//! For a checkpoint, stop mutation admission, retain a freeze guard, drive writes +//! through their completion fence, then capture compatible allocator and caller +//! metadata. Keep admission coordinated around thaw and restore. Neither the +//! image nor the completion fence proves durability or record validity. Recovery +//! must validate identity, lengths, integrity, and versions. Invalid, stale, torn, +//! or missing cache records must become misses and be refetched. +//! +//! [`SegmentClock`] shares a cursor and second-chance set across bounded sweeps. +//! Index reclamation forgets mappings without touching bytes or generations. +//! Physical reclamation only visits Sealed/Evicting slots and recycles empty slots +//! after leases drain. Each sweep visits at most two rotations, further capped by +//! the caller. Insufficient progress returns Busy, not a spin loop. A zero reserve +//! is a no-op; a zero entry budget can recycle empty slots but cannot begin +//! populated eviction. There is no compaction. [`SegmentEntries`] implementations +//! must compare current mappings before removal and report within budget; an +//! error cannot undo callback side effects. Freeze does not block index sweeps. +//! +//! # Validation +//! +//! Run `cargo test -p page-alloc --no-default-features` and +//! `cargo test -p page-alloc --all-features` under an external TERM timeout. +//! Simulation provides deterministic fault and cancellation ordering, not proof +//! of filesystem security or kernel support. Descriptor-replacement test hooks +//! require an open slab and no live writes and do not revalidate geometry. +//! Real tests print explicit capability skips with `-- --nocapture`; set +//! `PAGE_ALLOC_REQUIRE_REAL_IO=1` to turn those skips into failures. A skip is not +//! evidence that the real path passed. Miri can cover the pure `buffer_tests`, +//! `geometry_tests`, and `segments::tests` filters, but not real kernel I/O. +#![deny(unsafe_op_in_unsafe_fn)] +#![deny(missing_docs)] + +mod segments; +mod slab; + +pub use segments::{ + FreezeGuard, Generation, SegmentClock, SegmentEntries, SegmentId, SegmentLease, + SegmentSnapshot, SegmentState, Segments, +}; +pub use slab::Slab; +use std::{ + alloc::{Layout, alloc_zeroed, dealloc}, + cell::RefCell, + fmt, + ptr::NonNull, + rc::{Rc, Weak}, +}; +use uring_runtime::reactor::IoBuffer; +use zeroize::Zeroize; + +/// Storage failures, separated from application record and admission policy. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +#[non_exhaustive] +pub enum Error { + /// The filesystem or kernel cannot provide required direct-I/O support. + Unsupported, + /// Caller configuration or an implementation contract is invalid. + InvalidConfiguration, + /// A transient lease, freeze, or resource limit prevents progress. + Busy, + /// An extent or persisted allocator image is malformed. + Corrupt, + /// A generation, segment state, or table identity is no longer valid. + Stale, + /// Storage is unopened, locked elsewhere, or has exhausted a generation. + Unavailable, + /// An I/O completion was short or failed without OS error detail. + Io, + /// Synchronous OS failure; asynchronous errors retain the runtime's error type. + SystemIo { + /// Operation that failed. + operation: &'static str, + + /// Operating-system error number when available. + errno: Option, + }, +} + +impl fmt::Display for Error { + /// Describe the category, preserving operating-system diagnostics when present. + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + if let Self::SystemIo { operation, errno } = self { + return match errno { + Some(errno) => write!( + f, + "{operation}: {} (errno {errno})", + std::io::Error::from_raw_os_error(*errno) + ), + None => write!(f, "{operation}: storage I/O failed"), + }; + } + f.write_str(match self { + Self::Unsupported => "direct I/O is unsupported", + Self::InvalidConfiguration => "invalid storage configuration", + Self::Busy => "storage is busy or exhausted", + Self::Corrupt => "invalid storage extent or generation", + Self::Stale => "stale storage generation, state, or table", + Self::Unavailable => "storage is unavailable", + Self::Io => "storage I/O failed", + Self::SystemIo { .. } => unreachable!(), + }) + } +} +impl std::error::Error for Error {} + +/// Result of an allocator operation before conversion into a caller's scope error. +pub type Result = std::result::Result; + +/// Maximum number of slots retained by a segment allocation table. +pub const MAX_SEGMENTS: u64 = 1_000_000; + +/// Validated physical dimensions, independent of record formats and slot quotas. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub struct SegmentGeometry { + slab_bytes: u64, + + segment_bytes: u64, + + segment_count: u64, + + alignment: Alignment, +} + +impl SegmentGeometry { + /// Validate physical geometry; retained tables separately enforce MAX_SEGMENTS. + pub fn new( + slab_bytes: u64, + segment_bytes: u64, + segment_count: u64, + alignment: Alignment, + ) -> Result { + if segment_bytes == 0 + || slab_bytes == 0 + || segment_count == 0 + || !slab_bytes.is_multiple_of(segment_bytes) + || segment_count > slab_bytes / segment_bytes + || !segment_bytes.is_multiple_of(alignment.offset()) + || !segment_bytes.is_multiple_of(alignment.length() as u64) + || segment_count.checked_mul(segment_bytes).is_none() + { + return Err(Error::Corrupt); + } + Ok(Self { + slab_bytes, + segment_bytes, + segment_count, + alignment, + }) + } + + /// Full logical size of the physical file. + pub fn slab_bytes(self) -> u64 { + self.slab_bytes + } + + /// Fixed physical size of each segment. + pub fn segment_bytes(self) -> u64 { + self.segment_bytes + } + + /// Number of exposed segments, possibly less than the physical file allows. + pub fn segment_count(self) -> u64 { + self.segment_count + } + + /// Direct-I/O requirements validated with these dimensions. + pub fn alignment(self) -> Alignment { + self.alignment + } + + /// Compare table dimensions only, not occupancy or alignment compatibility. + pub fn matches_segments(&self, segments: &Segments) -> bool { + segments.capacity_bytes() == self.slab_bytes + && segments.segment_bytes() == self.segment_bytes + && segments.count() as u64 == self.segment_count + } +} + +/// Caller-owned accounting retained while memory is live, including idle pooling. +pub trait Charge: 'static { + /// Whether this guard accounts for at least `bytes` live allocation bytes. + fn covers(&self, bytes: usize) -> bool; +} +impl Charge for () { + /// Explicitly opt out of accounting for callers without admission policy. + fn covers(&self, _bytes: usize) -> bool { + true + } +} + +/// Direct-I/O alignment requirements. Offset and length units may be any positive +/// integers; only the memory alignment must be a power of two. +#[must_use] +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub struct Alignment { + memory: usize, + + offset: u64, + + length: usize, +} +/// A nonempty, non-overflowing file range for one bounded I/O transfer. +#[must_use] +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub struct Extent { + offset: u64, + + length: usize, +} +impl Extent { + /// Reject empty ranges, offset overflow, and lengths above the transfer cap. + pub fn new(offset: u64, length: usize) -> Result { + if length == 0 + || length > Alignment::MAX_TRANSFER_LENGTH + || offset.checked_add(length as u64).is_none() + { + return Err(Error::Corrupt); + } + Ok(Self { offset, length }) + } + /// Starting byte offset in the file. + #[must_use] + pub fn offset(self) -> u64 { + self.offset + } + /// Transfer length in bytes, including any padding. + #[must_use] + pub fn length(self) -> usize { + self.length + } +} +impl Alignment { + /// Conservative single-transfer limit of 1 GiB. + /// + /// This fits the runtime's `u32` length and stays below Linux's + /// `MAX_RW_COUNT` (`i32::MAX` rounded down to a base-page boundary) on + /// supported Linux base-page sizes, without querying host state. Keeping a + /// fixed conservative bound also makes geometry checks deterministic under + /// simulation and Miri. Larger records must be split by the caller. + pub const MAX_TRANSFER_LENGTH: usize = 1 << 30; + + /// Validate alignment units without requiring offset/length powers of two. + pub fn new(memory: usize, offset: u64, length: usize) -> Result { + if !memory.is_power_of_two() || offset == 0 || length == 0 || memory > isize::MAX as usize { + return Err(Error::Unsupported); + } + Ok(Self { + memory, + offset, + length, + }) + } + /// Required memory address alignment. + #[must_use] + pub fn memory(self) -> usize { + self.memory + } + /// Required file offset unit. + #[must_use] + pub fn offset(self) -> u64 { + self.offset + } + /// Required transfer length unit. + #[must_use] + pub fn length(self) -> usize { + self.length + } + /// Round up to the least common multiple of offset and length units so the + /// next appended extent is also offset-aligned. Reject overflow and lengths + /// above [`Self::MAX_TRANSFER_LENGTH`] before allocating memory. + pub fn extent(&self, offset: u64, logical: usize) -> Result { + if !offset.is_multiple_of(self.offset) + || logical == 0 + || logical > Self::MAX_TRANSFER_LENGTH + { + return Err(Error::InvalidConfiguration); + } + let offset_unit = usize::try_from(self.offset).map_err(|_| Error::InvalidConfiguration)?; + let (mut a, mut b) = (offset_unit, self.length); + while b != 0 { + (a, b) = (b, a % b); + } + let unit = (offset_unit / a) + .checked_mul(self.length) + .ok_or(Error::InvalidConfiguration)?; + let length = logical + .checked_add(unit - 1) + .and_then(|v| (v / unit).checked_mul(unit)) + .ok_or(Error::InvalidConfiguration)?; + if length > Self::MAX_TRANSFER_LENGTH { + return Err(Error::InvalidConfiguration); + } + Extent::new(offset, length) + } + /// Allocate zeroed stable storage and retain its primary accounting guard. + /// Length must be nonzero, length-aligned, covered by `charge`, and no larger + /// than [`Self::MAX_TRANSFER_LENGTH`]. Allocation failure returns `Busy`. + pub fn allocate(&self, length: usize, charge: C) -> Result> { + if length == 0 + || length > Self::MAX_TRANSFER_LENGTH + || !length.is_multiple_of(self.length) + || !charge.covers(length) + { + return Err(Error::InvalidConfiguration); + } + let layout = Layout::from_size_align(length, self.memory) + .map_err(|_| Error::InvalidConfiguration)?; + // SAFETY: valid nonzero layout; Allocation owns the matching deallocation. + let pointer = NonNull::new(unsafe { alloc_zeroed(layout) }).ok_or(Error::Busy)?; + Ok(AlignedBuffer { + allocation: Some(Allocation { + pointer, + layout, + charge, + retained: Vec::new(), + }), + pool: Weak::new(), + }) + } + /// Validate the address, file offset, transfer unit, and exact buffer length. + pub fn check(&self, extent: Extent, buffer: &AlignedBuffer) -> Result<()> { + if !(buffer.allocation().pointer.as_ptr() as usize).is_multiple_of(self.memory) + || !extent.offset.is_multiple_of(self.offset) + || !extent.length.is_multiple_of(self.length) + || buffer.len() != extent.length + { + return Err(Error::InvalidConfiguration); + } + Ok(()) + } +} + +/// Owns raw storage and accounting together; moving it transfers both exactly once. +struct Allocation { + pointer: NonNull, + + layout: Layout, + + charge: C, + + retained: Vec>, +} +impl Allocation { + /// Exclusively borrow the complete initialized allocation with its original size. + fn as_mut_slice(&mut self) -> &mut [u8] { + // SAFETY: exclusive owner, initialized nonzero allocation, original layout. + unsafe { std::slice::from_raw_parts_mut(self.pointer.as_ptr(), self.layout.size()) } + } +} +impl Drop for Allocation { + /// Erase and free before field destruction releases the accounting guards. + fn drop(&mut self) { + self.as_mut_slice().zeroize(); + // SAFETY: this owner holds the allocation and its original layout. Guards + // are dropped only after zeroization and deallocation complete. + unsafe { dealloc(self.pointer.as_ptr(), self.layout) }; + } +} + +/// Worker-local, exclusively owned, initialized storage with a stable address. +/// +/// Moving the buffer never moves its bytes. Drop zeroizes the bytes before +/// freeing or pooling them. Idle storage retains its primary charge, but not +/// additional guards attached with [`Self::retain`]. +/// +/// A buffer cannot cross a worker boundary even when its accounting guard can: +/// +/// ```compile_fail +/// fn require_send() {} +/// require_send::>(); +/// ``` +#[must_use] +pub struct AlignedBuffer { + // Always Some while publicly accessible; taken only by Drop for pool transfer. + allocation: Option>, + + pool: Weak>>, +} +impl fmt::Debug for AlignedBuffer { + /// Show storage properties without requiring accounting guards to expose data. + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.debug_struct("AlignedBuffer") + .field("length", &self.len()) + .field("alignment", &self.allocation().layout.align()) + .field("retained", &self.allocation().retained.len()) + .finish_non_exhaustive() + } +} +impl AlignedBuffer { + /// Borrow the allocation, which is absent only during drop's ownership transfer. + fn allocation(&self) -> &Allocation { + self.allocation.as_ref().expect("live buffer allocation") + } + /// Mutably borrow live backing storage before any drop-time pool transfer. + fn allocation_mut(&mut self) -> &mut Allocation { + self.allocation.as_mut().expect("live buffer allocation") + } + /// Attach a weak return destination without extending the pool's lifetime. + pub(crate) fn pooled(mut self, pool: &Rc>>) -> Self { + self.pool = Rc::downgrade(pool); + self + } + /// Replace primary accounting only after the new guard covers the full size. + pub(crate) fn rebind(&mut self, charge: C) -> Result<()> { + if !charge.covers(self.len()) { + return Err(Error::InvalidConfiguration); + } + self.allocation_mut().charge = charge; + Ok(()) + } + /// Retain additional accounting through the final kernel completion. + pub fn retain(&mut self, charge: Rc) { + self.allocation_mut().retained.push(charge); + } + /// Initialized allocation length, including padding. + #[must_use] + pub fn len(&self) -> usize { + self.allocation().layout.size() + } + /// Always false: construction rejects zero-length buffers. + #[must_use] + pub fn is_empty(&self) -> bool { + false + } + /// Borrow all initialized bytes without moving or resizing the allocation. + #[must_use] + pub fn as_slice(&self) -> &[u8] { + // SAFETY: initialized allocation remains live throughout this borrow. + unsafe { std::slice::from_raw_parts(self.allocation().pointer.as_ptr(), self.len()) } + } + /// Exclusively borrow all initialized bytes. + #[must_use] + pub fn as_mut_slice(&mut self) -> &mut [u8] { + self.allocation_mut().as_mut_slice() + } + /// Compatibility accessor matching [`IoBuffer`]; this always succeeds. + pub fn bytes(&self) -> Result<&[u8]> { + Ok(self.as_slice()) + } + /// Compatibility accessor matching [`IoBuffer`]; this always succeeds. + pub fn bytes_mut(&mut self) -> Result<&mut [u8]> { + Ok(self.as_mut_slice()) + } +} +impl Drop for AlignedBuffer { + /// Zeroize and return storage if possible, otherwise let its owner free it. + fn drop(&mut self) { + if let Some(pool) = self.pool.upgrade() { + self.allocation_mut().as_mut_slice().zeroize(); + // Caller-owned guard destructors may access the pool. Do not invoke + // them while holding its RefCell borrow. Allocation remains owned + // by this buffer if a destructor unwinds. + self.allocation_mut().retained.clear(); + if let Ok(mut idle) = pool.try_borrow_mut() + && idle.is_none() + { + *idle = Some(Self { + allocation: self.allocation.take(), + pool: Weak::new(), + }); + } + } + // Otherwise field drop zeroizes and frees the allocation, even when the + // pool is gone, occupied, or borrowed. Idle buffers never repool themselves. + } +} +// SAFETY: owned aligned backing is initialized, stable, and live until Drop. +unsafe impl IoBuffer for AlignedBuffer { + type Error = Error; + + /// Expose initialized stable bytes to the runtime. + fn bytes(&self) -> Result<&[u8]> { + AlignedBuffer::bytes(self) + } + + /// Give the runtime exclusive access without moving the allocation. + fn bytes_mut(&mut self) -> Result<&mut [u8]> { + AlignedBuffer::bytes_mut(self) + } +} + +/// Pure allocation tests, including unwind and reentrant accounting destructors. +#[cfg(test)] +mod buffer_tests { + use super::*; + use std::cell::Cell; + + /// Tracks primary and retained bytes without exposing a Debug implementation. + struct TrackedCharge { + live: Rc>, + + bytes: usize, + } + impl TrackedCharge { + /// Admit and record a fixed number of live bytes. + fn new(live: &Rc>, bytes: usize) -> Self { + live.set(live.get() + bytes); + Self { + live: live.clone(), + bytes, + } + } + } + impl Charge for TrackedCharge { + /// Cover only the number of bytes admitted by this guard. + fn covers(&self, bytes: usize) -> bool { + self.bytes >= bytes + } + } + impl Drop for TrackedCharge { + /// Return admitted bytes exactly once. + fn drop(&mut self) { + self.live.set(self.live.get() - self.bytes); + } + } + + /// Invalid transfers must fail before inspecting accounting or allocating. + #[test] + fn transfer_limit_is_enforced_before_allocation_or_charge_inspection() { + /// Detects any accounting inspection on a structurally invalid request. + struct UncheckedCharge; + impl Charge for UncheckedCharge { + /// Panic when validation reaches accounting in the wrong order. + fn covers(&self, _: usize) -> bool { + panic!("invalid lengths must be rejected before inspecting accounting"); + } + } + let alignment = Alignment::new(1, 1, 1).unwrap(); + let max = Alignment::MAX_TRANSFER_LENGTH; + assert_eq!(Extent::new(0, max).unwrap().length(), max); + assert_eq!(alignment.extent(0, max).unwrap().length(), max); + for length in [0, max + 1, i32::MAX as usize, u32::MAX as usize, usize::MAX] { + assert_eq!(Extent::new(0, length), Err(Error::Corrupt)); + assert_eq!( + alignment.extent(0, length), + Err(Error::InvalidConfiguration) + ); + assert!(matches!( + alignment.allocate(length, UncheckedCharge), + Err(Error::InvalidConfiguration) + )); + } + let alignment = Alignment::new(8, 3, 5).unwrap(); + let rounded_limit = max / 15 * 15; + assert_eq!( + alignment.extent(0, rounded_limit).unwrap().length(), + rounded_limit + ); + assert_eq!( + alignment.extent(0, rounded_limit + 1), + Err(Error::InvalidConfiguration) + ); + assert_eq!(Extent::new(u64::MAX, 1), Err(Error::Corrupt)); + assert_eq!(alignment.extent(u64::MAX, 1), Err(Error::Corrupt)); + assert_eq!(Extent::new(u64::MAX - 1, 1).unwrap().offset(), u64::MAX - 1); + } + + /// Non-power-of-two units use the LCM and reject overflow before allocation. + #[test] + fn arbitrary_units_preserve_lcm_rounding_and_detect_overflow() { + for (offset_unit, length_unit, lcm) in [(3, 5, 15), (6, 9, 18), (768, 512, 1536)] { + let alignment = Alignment::new(64, offset_unit, length_unit).unwrap(); + assert_eq!(alignment.memory(), 64); + assert_eq!(alignment.offset(), offset_unit); + assert_eq!(alignment.length(), length_unit); + for (logical, expected) in [(1, lcm), (lcm, lcm), (lcm + 1, lcm * 2)] { + let extent = alignment.extent(offset_unit, logical).unwrap(); + assert_eq!(extent.offset(), offset_unit); + assert_eq!(extent.length(), expected); + let buffer = alignment.allocate(expected, ()).unwrap(); + alignment.check(extent, &buffer).unwrap(); + } + assert_eq!(alignment.extent(1, 1), Err(Error::InvalidConfiguration)); + } + for alignment in [ + Alignment::new(1, u64::MAX, usize::MAX - 1).unwrap(), + Alignment::new(1, 1, usize::MAX).unwrap(), + Alignment::new(1, 2, usize::MAX).unwrap(), + ] { + assert_eq!(alignment.extent(0, 2), Err(Error::InvalidConfiguration)); + } + for (memory, offset, length) in [(0, 1, 1), (3, 1, 1), (1, 0, 1), (1, 1, 0)] { + assert_eq!( + Alignment::new(memory, offset, length), + Err(Error::Unsupported) + ); + } + assert_eq!( + Alignment::new(1usize << (usize::BITS - 1), 1, 1), + Err(Error::Unsupported) + ); + } + + /// Moving a buffer preserves its address and initialized runtime-visible bytes. + #[test] + fn initialized_storage_remains_stable_across_moves_and_trait_access() { + let alignment = Alignment::new(64, 3, 5).unwrap(); + let mut buffer = alignment.allocate(15, ()).unwrap(); + assert!(!buffer.is_empty()); + assert_eq!(buffer.len(), 15); + assert_eq!(buffer.as_slice(), &[0; 15]); + let pointer = IoBuffer::bytes_mut(&mut buffer).unwrap().as_mut_ptr(); + assert!((pointer as usize).is_multiple_of(64)); + let moved = std::hint::black_box(Some(buffer)); + // SAFETY: moving the owner preserves the allocation; no intervening reborrow. + unsafe { pointer.write(42) }; + let mut buffer = moved.unwrap(); + assert_eq!(IoBuffer::bytes(&buffer).unwrap()[0], 42); + buffer.bytes_mut().unwrap()[1] = 17; + assert_eq!(buffer.bytes().unwrap()[1], 17); + assert_eq!(buffer.as_slice().as_ptr(), pointer); + alignment + .check(Extent::new(3, 15).unwrap(), &buffer) + .unwrap(); + for extent in [Extent::new(1, 15).unwrap(), Extent::new(3, 10).unwrap()] { + assert_eq!( + alignment.check(extent, &buffer), + Err(Error::InvalidConfiguration) + ); + } + assert_eq!( + Alignment::new(64, 3, 2) + .unwrap() + .check(Extent::new(3, 15).unwrap(), &buffer), + Err(Error::InvalidConfiguration) + ); + } + + /// Admission replacement is atomic and Debug does not expose guard internals. + #[test] + fn charge_validation_rebinding_and_debug_do_not_require_charge_debug() { + let live = Rc::new(Cell::new(0)); + let alignment = Alignment::new(8, 3, 5).unwrap(); + for (length, charge) in [ + (15, 14), + (14, 15), + (0, 15), + (Alignment::MAX_TRANSFER_LENGTH + 1, 15), + ] { + assert_eq!( + alignment + .allocate(length, TrackedCharge::new(&live, charge)) + .unwrap_err(), + Error::InvalidConfiguration + ); + assert_eq!(live.get(), 0); + } + let mut buffer = alignment + .allocate(15, TrackedCharge::new(&live, 15)) + .unwrap(); + assert_eq!( + buffer.rebind(TrackedCharge::new(&live, 14)), + Err(Error::InvalidConfiguration) + ); + assert_eq!(live.get(), 15); + buffer.rebind(TrackedCharge::new(&live, 20)).unwrap(); + assert_eq!(live.get(), 20); + let debug = format!("{buffer:?}"); + assert!(debug.contains("length: 15")); + assert!(debug.contains("alignment: 8")); + assert!(!debug.contains("TrackedCharge")); + drop(buffer); + assert_eq!(live.get(), 0); + } + + /// Pooling retains primary accounting while releasing completion-only guards. + #[test] + fn pool_transfers_allocation_and_primary_charge_but_not_retained_guards() { + let live = Rc::new(Cell::new(0)); + let pool = Rc::new(RefCell::new(None)); + let alignment = Alignment::new(64, 3, 5).unwrap(); + let mut buffer = alignment + .allocate(15, TrackedCharge::new(&live, 15)) + .unwrap() + .pooled(&pool); + let pointer = buffer.as_slice().as_ptr(); + buffer.as_mut_slice().fill(42); + let extra = Rc::new(TrackedCharge::new(&live, 7)); + let weak = Rc::downgrade(&extra); + buffer.retain(extra); + assert_eq!(live.get(), 22); + drop(buffer); + assert_eq!(live.get(), 15); + assert!(weak.upgrade().is_none()); + let mut reused = pool.borrow_mut().take().unwrap(); + assert_eq!(reused.as_slice().as_ptr(), pointer); + assert_eq!(reused.as_slice(), &[0; 15]); + reused.rebind(TrackedCharge::new(&live, 20)).unwrap(); + assert_eq!(live.get(), 20); + drop(reused.pooled(&pool)); + drop(pool); + assert_eq!(live.get(), 0); + } + + /// An unusable return slot frees storage exactly once instead of panicking. + #[test] + fn unavailable_borrowed_and_occupied_pools_release_exactly_once() { + let alignment = Alignment::new(8, 1, 1).unwrap(); + for scenario in 0..4 { + let live = Rc::new(Cell::new(0)); + let pool = Rc::new(RefCell::new(None)); + let mut buffer = alignment + .allocate(8, TrackedCharge::new(&live, 8)) + .unwrap() + .pooled(&pool); + buffer.retain(Rc::new(TrackedCharge::new(&live, 3))); + match scenario { + 0 => { + drop(pool); + drop(buffer); + } + 1 => { + let borrow = pool.borrow(); + drop(buffer); + assert_eq!(live.get(), 0); + assert!(borrow.is_none()); + } + 2 => { + let borrow = pool.borrow_mut(); + drop(buffer); + assert_eq!(live.get(), 0); + assert!(borrow.is_none()); + } + _ => { + let idle = alignment.allocate(8, TrackedCharge::new(&live, 8)).unwrap(); + let pointer = idle.as_slice().as_ptr(); + *pool.borrow_mut() = Some(idle); + drop(buffer); + assert_eq!(live.get(), 8); + assert_eq!(pool.borrow().as_ref().unwrap().as_slice().as_ptr(), pointer); + drop(pool); + } + } + assert_eq!(live.get(), 0); + } + } + + /// Caller destructors run before the return slot is borrowed. + #[test] + fn retained_guard_can_inspect_pool_before_buffer_is_returned() { + /// Runs a caller-provided destructor to test reentrant pool inspection. + struct Guard(Option>); + impl Charge for Guard { + /// Admit all sizes for this destructor-order test. + fn covers(&self, _: usize) -> bool { + true + } + } + impl Drop for Guard { + /// Invoke the callback at most once. + fn drop(&mut self) { + if let Some(callback) = self.0.take() { + callback(); + } + } + } + let pool = Rc::new(RefCell::new(None)); + let called = Rc::new(Cell::new(false)); + let mut buffer = Alignment::new(8, 1, 1) + .unwrap() + .allocate(8, Guard(None)) + .unwrap() + .pooled(&pool); + let observed_pool = pool.clone(); + let observed_called = called.clone(); + buffer.retain(Rc::new(Guard(Some(Box::new(move || { + assert!(observed_pool.borrow_mut().is_none()); + observed_called.set(true); + }))))); + drop(buffer); + assert!(called.get()); + assert!(pool.borrow().is_some()); + } + + /// Unwinding through extra accounting cannot leak primary allocation ownership. + #[test] + fn retained_guard_unwind_still_drops_owned_allocation_and_primary_charge() { + /// Counts destruction and optionally injects a panic. + struct Guard { + live: Rc>, + + panic: bool, + } + impl Charge for Guard { + /// Admit all sizes for the unwind test. + fn covers(&self, _: usize) -> bool { + true + } + } + impl Drop for Guard { + /// Record release before injecting the requested destructor panic. + fn drop(&mut self) { + self.live.set(self.live.get() - 1); + assert!(!self.panic, "injected retained guard panic"); + } + } + let live = Rc::new(Cell::new(2)); + let pool = Rc::new(RefCell::new(None)); + let mut buffer = Alignment::new(8, 1, 1) + .unwrap() + .allocate( + 8, + Guard { + live: live.clone(), + panic: false, + }, + ) + .unwrap() + .pooled(&pool); + buffer.retain(Rc::new(Guard { + live: live.clone(), + panic: true, + })); + assert!(std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| drop(buffer))).is_err()); + assert_eq!(live.get(), 0); + assert!(pool.borrow().is_none()); + } +} + +/// Pure geometry boundary tests independent of file I/O and checkpoint formats. +#[cfg(test)] +mod geometry_tests { + use super::*; + + /// Standard test units, not a host-page-size assumption. + fn alignment() -> Alignment { + Alignment::new(512, 512, 512).unwrap() + } + + /// Physical geometry does not impose the retained table's slot bound. + #[test] + fn geometry_accepts_partial_tables_without_checkpoint_item_policy() { + let geometry = SegmentGeometry::new(4096, 1024, 2, alignment()).unwrap(); + assert_eq!(geometry.slab_bytes(), 4096); + assert_eq!(geometry.segment_bytes(), 1024); + assert_eq!(geometry.segment_count(), 2); + assert_eq!(geometry.alignment(), alignment()); + assert!(SegmentGeometry::new(1024 * MAX_SEGMENTS, 1024, MAX_SEGMENTS, alignment()).is_ok()); + assert!( + SegmentGeometry::new( + 1024 * (MAX_SEGMENTS + 1), + 1024, + MAX_SEGMENTS + 1, + alignment() + ) + .is_ok() + ); + assert!( + SegmentGeometry::new(u64::MAX, 1, u64::MAX, Alignment::new(1, 1, 1).unwrap()).is_ok() + ); + } + + /// Reject zero units, inconsistent capacity, misalignment, and overflow. + #[test] + fn geometry_rejects_zero_misalignment_capacity_and_overflow() { + for (slab, segment, count) in [ + (0, 512, 1), + (512, 0, 1), + (512, 512, 0), + (513, 512, 1), + (512, 512, 2), + (514, 257, 2), + (u64::MAX, u64::MAX, 2), + ] { + assert_eq!( + SegmentGeometry::new(slab, segment, count, alignment()), + Err(Error::Corrupt) + ); + } + assert_eq!( + SegmentGeometry::new(1024, 1024, 1, Alignment::new(512, 512, 768).unwrap()), + Err(Error::Corrupt) + ); + } + + /// Table matching ignores occupancy but compares each configured dimension. + #[test] + fn matching_live_table_ignores_occupancy_but_requires_dimensions() { + let geometry = SegmentGeometry::new(4096, 1024, 2, alignment()).unwrap(); + let segments = Segments::new(1024); + assert!(!geometry.matches_segments(&segments)); + segments.configure(4096, 2, alignment()).unwrap(); + assert!(geometry.matches_segments(&segments)); + let _held = segments.append(512).unwrap(); + assert!(geometry.matches_segments(&segments)); + assert!( + !SegmentGeometry::new(4096, 1024, 3, alignment()) + .unwrap() + .matches_segments(&segments) + ); + assert!( + !SegmentGeometry::new(2048, 1024, 2, alignment()) + .unwrap() + .matches_segments(&segments) + ); + assert!( + !SegmentGeometry::new(4096, 512, 2, alignment()) + .unwrap() + .matches_segments(&segments) + ); + } + + /// Segment boundaries must satisfy both direct-I/O units simultaneously. + #[test] + fn divisibility_by_each_unit_is_equivalent_to_lcm_divisibility() { + let alignment = Alignment::new(512, 512, 768).unwrap(); + assert!(SegmentGeometry::new(3072, 1536, 2, alignment).is_ok()); + assert_eq!( + SegmentGeometry::new(2048, 1024, 2, alignment), + Err(Error::Corrupt) + ); + assert_eq!( + SegmentGeometry::new(1536, 768, 2, alignment), + Err(Error::Corrupt) + ); + assert!(SegmentGeometry::new(u64::MAX, 1, 1, Alignment::new(1, 1, 1).unwrap()).is_ok()); + } +} diff --git a/cmd/racer-dataplane/alloc/src/segments.rs b/cmd/racer-dataplane/alloc/src/segments.rs new file mode 100644 index 000000000..9928f194c --- /dev/null +++ b/cmd/racer-dataplane/alloc/src/segments.rs @@ -0,0 +1,1534 @@ +//! Worker-local allocation authority, recovery images, and bounded reclamation. +//! +//! Slot reuse is metadata-only: it never erases, truncates, or compacts disk bytes. +//! Existing leases protect their captured used prefix even after eviction starts. +//! Freeze protects table mutation, not caller indexes or data-I/O completion. +use crate::{Alignment, Error, Extent, MAX_SEGMENTS, Result, SegmentGeometry}; +use std::{ + cell::{Cell, RefCell}, + collections::{BTreeSet, HashSet}, + rc::Rc, +}; + +/// Stable slot number within one allocation table, not authority to access it. +#[derive(Clone, Copy, Debug, Eq, Hash, PartialEq)] +pub struct SegmentId(pub u64); +/// Monotonic reuse counter; zero is invalid in a restored image. +#[derive(Clone, Copy, Debug, Eq, Hash, PartialEq)] +pub struct Generation(pub u64); +/// Persisted segment lifecycle, separate from live lease and freeze ownership. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub enum SegmentState { + /// Available for a new append. + Free, + /// The sole appendable tail. + Open, + /// Full or rotated away from; eligible for eviction. + Sealed, + /// Rejects new leases while existing completion owners drain. + Evicting, +} +/// Caller-owned recovery image; restore validates it before publishing any slot. +#[derive(Clone, Debug, Eq, PartialEq)] +pub struct SegmentSnapshot { + /// Ordered slot identity. + pub id: SegmentId, + + /// Current nonzero reuse generation. + pub generation: Generation, + + /// Persisted lifecycle state. + pub state: SegmentState, + + /// Aligned occupied prefix, including record padding. + pub used_bytes: u64, +} + +/// Unique worker-local lease that prevents reuse until its completion owner drops. +/// It authorizes the used prefix at acquisition, not just the latest append. +/// +/// Lease counts cannot be duplicated by cloning a completion capability: +/// +/// ```compile_fail +/// fn duplicate(lease: page_alloc::SegmentLease) { let _ = lease.clone(); } +/// ``` +#[must_use = "keep the lease alive until the operation completes"] +pub struct SegmentLease { + id: SegmentId, + + generation: Generation, + + count: Rc>, + + table: TableIdentity, + + geometry: SegmentGeometry, + + start: u64, + + end: u64, +} +impl SegmentLease { + /// Slot protected against recycling by this lease. + pub fn id(&self) -> SegmentId { + self.id + } + /// Reuse generation captured when the lease was acquired. + pub fn generation(&self) -> Generation { + self.generation + } + /// Borrow the nominal identity without granting table mutation authority. + pub(crate) fn table_identity(&self) -> TableIdentity { + self.table.clone() + } + /// Validated dimensions captured independently of the table's lifetime. + pub(crate) fn geometry(&self) -> SegmentGeometry { + self.geometry + } + /// An existing lease remains valid during eviction, but cannot authorize + /// bytes appended after it was acquired. + pub(crate) fn validate_extent(&self, extent: &Extent) -> Result<()> { + let end = extent + .offset() + .checked_add(extent.length() as u64) + .ok_or(Error::Corrupt)?; + if extent.offset() < self.start + || end > self.end + || self + .geometry() + .alignment() + .extent(extent.offset(), extent.length()) + .map_err(|_| Error::Corrupt)? + != *extent + { + return Err(Error::Corrupt); + } + Ok(()) + } +} +impl Drop for SegmentLease { + /// Release exactly the one lease count acquired during construction. + fn drop(&mut self) { + self.count.set(self.count.get() - 1); + } +} +/// Mutable recovery image paired with its independently retained lease counter. +struct Slot { + image: SegmentSnapshot, + + leases: Rc>, +} +/// Holds a freeze independently of the table's lifetime. Dropping it thaws the table. +/// +/// A second guard cannot be created by cloning the first: +/// +/// ```compile_fail +/// fn duplicate(guard: page_alloc::FreezeGuard) { let _ = guard.clone(); } +/// ``` +#[must_use = "keep the guard alive while the table must remain frozen"] +#[derive(Debug)] +pub struct FreezeGuard(Rc>); +impl Drop for FreezeGuard { + /// Release mutation exclusion even if the table has already been dropped. + fn drop(&mut self) { + self.0.set(false); + } +} +/// Non-cloneable worker-local table whose leases and guards can outlive it. +/// Rc sharing is intentional within a worker; authority cannot cross threads. +/// +/// ```compile_fail +/// fn require_send() {} +/// require_send::(); +/// ``` +#[repr(align(64))] +pub struct Segments { + segment_bytes: u64, + + geometry: Cell>, + + slots: RefCell>, + + frozen: Rc>, + + table: TableIdentity, + + restore_epoch: Cell, + + open: Cell>, + + free: RefCell>, + evicting: Cell, +} +impl Segments { + /// Create an unconfigured table for Slab::open_configured to bind at startup. + pub fn new(segment_bytes: u64) -> Self { + Self { + segment_bytes, + geometry: Cell::new(None), + slots: RefCell::new(Vec::new()), + frozen: Rc::new(Cell::new(false)), + table: TableIdentity::new(), + restore_epoch: Cell::new(0), + open: Cell::new(None), + free: RefCell::new(BTreeSet::new()), + evicting: Cell::new(0), + } + } + /// Fixed size of a physical segment. + pub fn segment_bytes(&self) -> u64 { + self.segment_bytes + } + /// Configured physical capacity, or zero before configuration. + pub fn capacity_bytes(&self) -> u64 { + self.geometry.get().map_or(0, SegmentGeometry::slab_bytes) + } + /// Build an entirely free table, rejecting geometry above the retained slot limit. + pub fn from_geometry(geometry: SegmentGeometry) -> Result { + let segments = Self::new(geometry.segment_bytes()); + segments.configure_table( + geometry.slab_bytes(), + usize::try_from(geometry.segment_count()).map_err(|_| Error::InvalidConfiguration)?, + geometry.alignment(), + )?; + Ok(segments) + } + /// Whether validated geometry and the slot table have been installed. + pub fn is_configured(&self) -> bool { + self.geometry.get().is_some() + } + /// Configured geometry, which may expose fewer slots than physical capacity. + pub fn geometry(&self) -> Option { + self.geometry.get() + } + /// Clone identity only, without granting allocation authority. + pub(crate) fn table_identity(&self) -> TableIdentity { + self.table.clone() + } + /// Recovery revision used to invalidate eviction cursor and recent-read state. + pub(crate) fn restore_epoch(&self) -> u64 { + self.restore_epoch.get() + } + /// Install an entirely free bounded table exactly once, unless frozen. + #[cfg(any(test, feature = "simulation"))] + pub fn configure(&self, capacity: u64, count: usize, alignment: Alignment) -> Result<()> { + self.configure_table(capacity, count, alignment) + } + + /// Install validated geometry for startup or an explicitly constructed table. + pub(crate) fn configure_table( + &self, + capacity: u64, + count: usize, + alignment: Alignment, + ) -> Result<()> { + if self.frozen.get() { + return Err(Error::Busy); + } + if self.is_configured() || count as u64 > MAX_SEGMENTS { + return Err(Error::InvalidConfiguration); + } + let geometry = SegmentGeometry::new(capacity, self.segment_bytes, count as u64, alignment) + .map_err(|_| Error::InvalidConfiguration)?; + self.geometry.set(Some(geometry)); + *self.slots.borrow_mut() = (0..count) + .map(|id| Slot { + image: SegmentSnapshot { + id: SegmentId(id as u64), + generation: Generation(1), + state: SegmentState::Free, + used_bytes: 0, + }, + leases: Rc::new(Cell::new(0)), + }) + .collect(); + *self.free.borrow_mut() = (0..count).collect(); + Ok(()) + } + /// Reserve an aligned used range. Malformed requests and lease overflow leave + /// state unchanged. A valid rollover without a free slot seals the open tail + /// before returning Busy, allowing reclamation to make a retry possible. + pub fn append(&self, length: usize) -> Result<(SegmentLease, Extent)> { + if self.frozen.get() { + return Err(Error::Busy); + } + let alignment = self.geometry().ok_or(Error::Unavailable)?.alignment(); + if length == 0 + || length as u64 > self.segment_bytes + || alignment.extent(0, length)?.length() != length + { + return Err(Error::InvalidConfiguration); + } + let mut slots = self.slots.borrow_mut(); + let previous_open = self.open.get(); + let usable_open = previous_open.filter(|&position| { + self.segment_bytes - slots[position].image.used_bytes >= length as u64 + }); + let position = match usable_open { + Some(position) => position, + None => match self.free.borrow().first() { + Some(&position) => position, + None => { + if let Some(previous) = previous_open { + slots[previous].image.state = SegmentState::Sealed; + self.open.set(None); + } + return Err(Error::Busy); + } + }, + }; + let slot = &slots[position]; + let offset = slot + .image + .id + .0 + .checked_mul(self.segment_bytes) + .and_then(|v| v.checked_add(slot.image.used_bytes)) + .ok_or(Error::InvalidConfiguration)?; + let extent = alignment.extent(offset, length)?; + let used_bytes = slot + .image + .used_bytes + .checked_add(length as u64) + .ok_or(Error::InvalidConfiguration)?; + // All fallible checks, including lease-counter overflow, precede mutation. + let lease = self.take_lease(slot, used_bytes)?; + if usable_open.is_none() { + if let Some(previous) = previous_open { + slots[previous].image.state = SegmentState::Sealed; + } + self.free.borrow_mut().remove(&position); + } + self.open.set(Some(position)); + let slot = &mut slots[position]; + slot.image.state = SegmentState::Open; + slot.image.used_bytes = used_bytes; + if slot.image.used_bytes == self.segment_bytes { + slot.image.state = SegmentState::Sealed; + self.open.set(None); + } + Ok((lease, extent)) + } + /// Capture a checked prefix and increment its counter before publishing a lease. + fn take_lease(&self, slot: &Slot, used_bytes: u64) -> Result { + let start = slot + .image + .id + .0 + .checked_mul(self.segment_bytes) + .ok_or(Error::Corrupt)?; + let end = start.checked_add(used_bytes).ok_or(Error::Corrupt)?; + let geometry = self.geometry().ok_or(Error::Unavailable)?; + slot.leases + .set(slot.leases.get().checked_add(1).ok_or(Error::Busy)?); + Ok(SegmentLease { + id: slot.image.id, + generation: slot.image.generation, + count: slot.leases.clone(), + table: self.table.clone(), + geometry, + start, + end, + }) + } + /// Convert an external slot number without truncation. + fn position(id: SegmentId) -> Result { + usize::try_from(id.0).map_err(|_| Error::Corrupt) + } + /// Acquire the current used prefix only while the generation is readable. + pub fn lease(&self, id: SegmentId, generation: Generation) -> Result { + let slots = self.slots.borrow(); + let slot = slots.get(Self::position(id)?).ok_or(Error::Corrupt)?; + if slot.image.generation != generation + || !matches!(slot.image.state, SegmentState::Open | SegmentState::Sealed) + { + return Err(Error::Stale); + } + self.take_lease(slot, slot.image.used_bytes) + } + /// Stop new leases for a sealed slot; repeating eviction is harmless. + #[cfg(any(test, feature = "simulation"))] + pub fn begin_evict(&self, id: SegmentId) -> Result<()> { + self.begin_eviction(id) + } + + /// Enter eviction before the clock invokes any caller index removal. + fn begin_eviction(&self, id: SegmentId) -> Result<()> { + if self.frozen.get() { + return Err(Error::Busy); + } + let mut slots = self.slots.borrow_mut(); + let slot = slots.get_mut(Self::position(id)?).ok_or(Error::Corrupt)?; + if !matches!( + slot.image.state, + SegmentState::Sealed | SegmentState::Evicting + ) { + return Err(Error::Busy); + } + if slot.image.state != SegmentState::Evicting { + self.evicting.set(self.evicting.get() + 1); + slot.image.state = SegmentState::Evicting; + } + Ok(()) + } + /// Reuse only an evicting, unleased slot, incrementing generation without wrap. + #[cfg(any(test, feature = "simulation"))] + pub fn recycle(&self, id: SegmentId) -> Result<()> { + self.recycle_evicted(id) + } + + /// Complete clock-authorized eviction only after outstanding leases drain. + fn recycle_evicted(&self, id: SegmentId) -> Result<()> { + if self.frozen.get() { + return Err(Error::Busy); + } + let mut slots = self.slots.borrow_mut(); + let slot = slots.get_mut(Self::position(id)?).ok_or(Error::Corrupt)?; + if slot.image.state != SegmentState::Evicting || slot.leases.get() != 0 { + return Err(Error::Busy); + } + slot.image.generation = Generation( + slot.image + .generation + .0 + .checked_add(1) + .ok_or(Error::Unavailable)?, + ); + slot.image.state = SegmentState::Free; + self.evicting.set(self.evicting.get() - 1); + slot.image.used_bytes = 0; + self.free.borrow_mut().insert(Self::position(id)?); + Ok(()) + } + /// Copy the complete ordered image, including while the table is frozen. + pub fn snapshot(&self) -> Vec { + self.slots + .borrow() + .iter() + .map(|s| s.image.clone()) + .collect() + } + /// Block allocation, eviction, recycling, and restore until the guard drops. + /// Reads and snapshots remain available; only one guard may exist at a time. + pub fn freeze(&self) -> Result { + if self.frozen.replace(true) { + return Err(Error::Busy); + } + Ok(FreezeGuard(self.frozen.clone())) + } + /// Check the entire image and current restore eligibility without mutation. + pub fn validate_restore(&self, images: &[SegmentSnapshot]) -> Result<()> { + self.validate_restore_epoch(images).map(|_| ()) + } + /// Validate recovery invariants and reserve the next nonwrapping epoch value. + fn validate_restore_epoch(&self, images: &[SegmentSnapshot]) -> Result { + if self.frozen.get() { + return Err(Error::Busy); + } + let slots = self.slots.borrow(); + if slots.iter().any(|s| s.leases.get() != 0) { + return Err(Error::Busy); + } + if images.len() != slots.len() { + return Err(Error::Corrupt); + } + let alignment = self.geometry().ok_or(Error::Unavailable)?.alignment(); + let mut open = false; + for (i, s) in images.iter().enumerate() { + if s.id.0 != i as u64 + || s.generation.0 == 0 + || s.used_bytes > self.segment_bytes + || !s.used_bytes.is_multiple_of(alignment.offset()) + || !s.used_bytes.is_multiple_of(alignment.length() as u64) + || (s.state == SegmentState::Free && s.used_bytes != 0) + || (s.state != SegmentState::Free && s.used_bytes == 0) + { + return Err(Error::Corrupt); + } + if s.state == SegmentState::Open { + if open || s.used_bytes == self.segment_bytes { + return Err(Error::Corrupt); + } + open = true; + } + } + self.restore_epoch + .get() + .checked_add(1) + .ok_or(Error::Unavailable) + } + /// Validate the complete image before publishing it, sealing its open tail. + /// Frozen tables and outstanding leases return Busy without changing state. + pub fn restore(&self, images: Vec) -> Result<()> { + let epoch = self.validate_restore_epoch(&images)?; + let mut slots = self.slots.borrow_mut(); + self.open.set(None); + self.free.borrow_mut().clear(); + self.evicting.set(0); + for (slot, mut image) in slots.iter_mut().zip(images) { + if image.state == SegmentState::Open { + image.state = SegmentState::Sealed; + } + if image.state == SegmentState::Free { + self.free.borrow_mut().insert(image.id.0 as usize); + } + if image.state == SegmentState::Evicting { + self.evicting.set(self.evicting.get() + 1); + } + slot.image = image; + } + self.restore_epoch.set(epoch); + Ok(()) + } + /// Validate a stored mapping against current readable state and used bytes. + pub fn validate(&self, id: SegmentId, generation: Generation, extent: &Extent) -> Result<()> { + let slots = self.slots.borrow(); + let slot = slots.get(Self::position(id)?).ok_or(Error::Corrupt)?; + let start = id.0.checked_mul(self.segment_bytes).ok_or(Error::Corrupt)?; + let end = extent + .offset() + .checked_add(extent.length() as u64) + .ok_or(Error::Corrupt)?; + if slot.image.generation != generation + || !matches!(slot.image.state, SegmentState::Open | SegmentState::Sealed) + { + return Err(Error::Stale); + } + if extent.offset() < start + || end + > start + .checked_add(slot.image.used_bytes) + .ok_or(Error::Corrupt)? + { + return Err(Error::Corrupt); + } + let alignment = self.geometry().ok_or(Error::Unavailable)?.alignment(); + if alignment + .extent(extent.offset(), extent.length()) + .map_err(|_| Error::Corrupt)? + != *extent + { + return Err(Error::Corrupt); + } + Ok(()) + } + /// Validate a live lease, including table identity and its captured used range. + /// Existing leases remain usable while their segment is Evicting. + pub fn validate_lease(&self, lease: &SegmentLease, extent: &Extent) -> Result<()> { + if !self.table.matches(&lease.table) { + return Err(Error::Stale); + } + lease.validate_extent(extent) + } + /// Number of immediately appendable slots, excluding pending eviction. + pub fn free_count(&self) -> usize { + self.free.borrow().len() + } + /// Number of retained slots, which may be less than physical capacity. + pub fn count(&self) -> usize { + self.slots.borrow().len() + } + /// Inspect a valid slot without granting mutation or lease authority. + pub fn state(&self, id: SegmentId) -> Result { + self.slots + .borrow() + .get(Self::position(id)?) + .map(|s| s.image.state) + .ok_or(Error::Corrupt) + } +} + +/// Nominal, unforgeable identity shared by a table, bound file, and its leases. +/// Cloning identity does not clone allocation authority or keep the table alive. +#[derive(Clone)] +pub(crate) struct TableIdentity(Rc<()>); + +impl TableIdentity { + /// Mint one distinct identity for a newly constructed table. + fn new() -> Self { + Self(Rc::new(())) + } + + /// Compare allocation identity, never persisted slot numbers or generations. + pub(crate) fn matches(&self, other: &Self) -> bool { + Rc::ptr_eq(&self.0, &other.0) + } +} + +/// Application-owned mappings; removal must compare current entries first. +/// Callbacks own metadata and version side effects and must never exceed budget. +pub trait SegmentEntries { + /// Active unpublished writes must not lose publication authority. Read + /// leases still permit eviction and independently fence physical reuse. + fn can_evict(&self, _segment: SegmentId) -> bool { + true + } + /// Remove at most budget current mappings, returning the number removed. + fn remove_bounded(&self, segment: SegmentId, budget: usize) -> usize; + + /// Report whether any current mapping still refers to this segment. + fn is_empty(&self, segment: SegmentId) -> bool; +} + +/// One worker-local cursor and recent-read set for index and physical reclamation. +#[repr(align(64))] +pub struct SegmentClock { + segments: Rc, + + hand: Cell, + + recent: RefCell>, + + restore_epoch: Cell, +} + +impl SegmentClock { + /// Attach bounded reclamation to one table without cloning its authority. + pub fn new(segments: Rc) -> Self { + let epoch = segments.restore_epoch(); + Self { + segments, + hand: Cell::new(0), + recent: RefCell::new(HashSet::new()), + restore_epoch: Cell::new(epoch), + } + } + + /// Give live segments a second chance; ignore completions arriving after eviction. + pub fn mark_read(&self, segment: SegmentId) -> Result<()> { + self.sync_restore(); + if !matches!( + self.segments.state(segment)?, + SegmentState::Open | SegmentState::Sealed + ) { + return Ok(()); + } + self.recent.borrow_mut().insert(segment); + Ok(()) + } + + /// Clear transient eviction history after successful table recovery. + fn sync_restore(&self) { + let epoch = self.segments.restore_epoch(); + if self.restore_epoch.replace(epoch) != epoch { + self.hand.set(0); + self.recent.borrow_mut().clear(); + } + } + + /// Advance one slot; callers only invoke this inside a nonempty bounded sweep. + fn next(&self, count: usize) -> SegmentId { + let hand = self.hand.get() % count; + self.hand.set((hand + 1) % count); + SegmentId(hand as u64) + } + + /// Forget one mapping per candidate until admission succeeds, without changing + /// bytes or generations. At most two rotations, further capped by max_visits. + /// Freeze does not block caller-owned index mutation. + pub fn reclaim_index( + &self, + entries: &impl SegmentEntries, + max_visits: usize, + mut ready: impl FnMut() -> bool, + ) -> Result<()> { + self.sync_restore(); + if ready() { + return Ok(()); + } + let count = self.segments.count(); + for _ in 0..count.saturating_mul(2).min(max_visits) { + let id = self.next(count); + if self.recent.borrow_mut().remove(&id) { + continue; + } + if entries.remove_bounded(id, 1) > 1 { + return Err(Error::InvalidConfiguration); + } + if ready() { + return Ok(()); + } + } + Err(Error::Busy) + } + + /// Rank at most 64 eligible slots before entering eviction. Scores are soft + /// preferences, not pins. The callback must itself bound mapping inspection. + /// Existing eviction work sorts first so partial removals always make progress. + /// Unlike the legacy clock, logical heat is supplied by the caller, so recent + /// disk completions do not override value ranking. + pub fn reclaim_scored( + &self, + entries: &impl SegmentEntries, + free_reserve: usize, + max_visits: usize, + max_entries: usize, + mut score: impl FnMut(SegmentId) -> u64, + ) -> Result<()> { + self.sync_restore(); + if free_reserve == 0 { + return Ok(()); + } + let count = self.segments.count(); + if count == 0 { + return Err(Error::Unavailable); + } + let target = free_reserve.min(count); + if self.segments.free_count() >= target && self.segments.evicting.get() == 0 { + return Ok(()); + } + let mut candidates = Vec::new(); + for visit in 0..count.min(max_visits).min(64) { + let id = self.next(count); + let state = self.segments.state(id)?; + if !matches!(state, SegmentState::Sealed | SegmentState::Evicting) { + continue; + } + if state == SegmentState::Sealed && !entries.can_evict(id) { + continue; + } + // Rank before any begin_eviction or index side effects. + let value = if state == SegmentState::Evicting { + 0 + } else { + score(id) + }; + candidates.push((state != SegmentState::Evicting, value, visit, id)); + } + candidates.sort_by_key(|(sealed, value, visit, _)| (*sealed, *value, *visit)); + let mut entries_left = max_entries; + for (_, _, _, id) in candidates { + let state = self.segments.state(id)?; + if state == SegmentState::Sealed + && (self + .segments + .free_count() + .saturating_add(self.segments.evicting.get()) + >= target + || !entries.can_evict(id)) + { + continue; + } + if entries_left == 0 && !entries.is_empty(id) { + continue; + } + self.segments.begin_eviction(id)?; + self.recent.borrow_mut().remove(&id); + if entries_left != 0 && !entries.is_empty(id) { + let removed = entries.remove_bounded(id, entries_left); + if removed > entries_left { + return Err(Error::InvalidConfiguration); + } + entries_left -= removed; + } + if !entries.is_empty(id) { + continue; + } + match self.segments.recycle_evicted(id) { + Ok(()) | Err(Error::Busy) => {} + Err(error) => return Err(error), + } + } + if self.segments.free_count() >= target { + Ok(()) + } else { + Err(Error::Busy) + } + } + + /// Reclaim a reserve with bounded visits and mapping removals, without compaction. + /// Busy leases keep segments Evicting until a later sweep. A zero reserve is a + /// no-op; a zero mapping budget can recycle empty but not populated segments. + pub fn reclaim( + &self, + entries: &impl SegmentEntries, + free_reserve: usize, + max_visits: usize, + max_entries: usize, + ) -> Result<()> { + self.sync_restore(); + if free_reserve == 0 { + return Ok(()); + } + let count = self.segments.count(); + if count == 0 { + return Err(Error::Unavailable); + } + let target = free_reserve.min(count); + let mut free = self.segments.free_count(); + let mut entries_left = max_entries; + for _ in 0..count.saturating_mul(2).min(max_visits) { + if free >= target { + return Ok(()); + } + let id = self.next(count); + if !matches!( + self.segments.state(id)?, + SegmentState::Sealed | SegmentState::Evicting + ) { + continue; + } + if self.recent.borrow_mut().remove(&id) { + continue; + } + if entries_left == 0 && !entries.is_empty(id) { + continue; + } + self.segments.begin_eviction(id)?; + if entries_left != 0 && !entries.is_empty(id) { + let removed = entries.remove_bounded(id, entries_left); + if removed > entries_left { + return Err(Error::InvalidConfiguration); + } + entries_left -= removed; + } + if !entries.is_empty(id) { + continue; + } + match self.segments.recycle_evicted(id) { + Ok(()) => free += 1, + Err(Error::Busy) => {} + Err(e) => return Err(e), + } + } + if free >= target { + Ok(()) + } else { + Err(Error::Busy) + } + } +} + +/// Pure state-machine tests, including synthetic overflow and invalid recovery images. +#[cfg(test)] +mod tests { + use super::*; + + /// Build a bounded free table with deterministic alignment. + fn segments(bytes: u64, count: usize) -> Segments { + let s = Segments::new(bytes); + s.configure( + bytes * count as u64, + count, + Alignment::new(512, 512, 512).unwrap(), + ) + .unwrap(); + s + } + + /// Lease ownership prevents reuse and old-generation acquisition. + #[test] + fn no_reuse_before_lease_drop_and_no_aba() { + let s = segments(1024, 2); + let a = s.append(1024).unwrap(); + s.begin_evict(a.0.id()).unwrap(); + assert_eq!(s.recycle(a.0.id()), Err(Error::Busy)); + drop(a); + s.recycle(SegmentId(0)).unwrap(); + assert!(s.lease(SegmentId(0), Generation(1)).is_err()); + assert_eq!(s.append(512).unwrap().0.generation(), Generation(2)); + let frozen = s.freeze().unwrap(); + assert!(s.append(512).is_err()); + drop(frozen); + } + + /// Recovery seals the append tail and rejects unaligned occupancy. + #[test] + fn restore_seals_open_and_rejects_invalid_geometry() { + let s = segments(1024, 1); + drop(s.append(512).unwrap()); + let mut snap = s.snapshot(); + s.restore(snap.clone()).unwrap(); + assert_eq!(s.snapshot()[0].state, SegmentState::Sealed); + snap[0].used_bytes = 513; + assert!(s.restore(snap).is_err()); + } + + /// Tail rotation and generation exhaustion never wrap into stale authority. + #[test] + fn generation_exhaustion_never_wraps_and_small_tails_are_sealed() { + let s = segments(1024, 2); + drop(s.append(512).unwrap()); + let next = s.append(1024).unwrap(); + assert_eq!(next.0.id(), SegmentId(1)); + drop(next); + assert_eq!(s.state(SegmentId(0)).unwrap(), SegmentState::Sealed); + let mut snapshot = s.snapshot(); + snapshot[0].generation = Generation(u64::MAX); + s.restore(snapshot).unwrap(); + s.begin_evict(SegmentId(0)).unwrap(); + assert_eq!(s.recycle(SegmentId(0)), Err(Error::Unavailable)); + assert_eq!(s.snapshot()[0].generation, Generation(u64::MAX)); + } + + /// Validation distinguishes stale identity, corrupt ranges, and busy ownership. + #[test] + fn validation_rejects_unwritten_ranges_generations_and_busy_restore() { + let s = segments(1024, 1); + assert!(s.append(0).is_err()); + assert!(s.append(513).is_err()); + let (lease, extent) = s.append(512).unwrap(); + assert_eq!(s.validate(lease.id(), lease.generation(), &extent), Ok(())); + assert_eq!( + s.validate(lease.id(), Generation(2), &extent), + Err(Error::Stale) + ); + assert_eq!( + s.validate( + lease.id(), + lease.generation(), + &Extent::new(512, 512).unwrap() + ), + Err(Error::Corrupt) + ); + let mut images = s.snapshot(); + assert_eq!(s.restore(images.clone()), Err(Error::Busy)); + drop(lease); + images[0].generation = Generation(0); + assert_eq!(s.restore(images), Err(Error::Corrupt)); + assert_eq!(s.state(SegmentId(0)).unwrap(), SegmentState::Open); + assert_eq!(s.count(), 1); + assert_eq!(s.free_count(), 0); + assert_eq!(s.capacity_bytes(), 1024); + let frozen = s.freeze().unwrap(); + assert!(matches!(s.freeze(), Err(Error::Busy))); + assert_eq!(s.begin_evict(SegmentId(0)), Err(Error::Busy)); + assert_eq!(s.recycle(SegmentId(0)), Err(Error::Busy)); + drop(frozen); + drop(s.append(512).unwrap()); + assert!(matches!(s.append(512), Err(Error::Busy))); + } + + /// Invalid configuration does not partially install geometry or slots. + #[test] + fn configuration_rejects_zero_and_misaligned_capacity() { + let a = Alignment::new(512, 512, 512).unwrap(); + for (bytes, capacity, count) in [ + (0, 1024, 1), + (1024, 0, 1), + (513, 1026, 2), + (1024, 1024, 0), + (1024, 1024, 2), + (1024, 1025, 1), + (1024, 1024 * 1_000_001, 1_000_001), + ] { + assert_eq!( + Segments::new(bytes).configure(capacity, count, a), + Err(Error::InvalidConfiguration) + ); + } + let s = segments(1024, 1); + assert_eq!(s.configure(1024, 1, a), Err(Error::InvalidConfiguration)); + assert!(s.state(SegmentId(u64::MAX)).is_err()); + let s = Segments::new(1024); + assert_eq!(s.configure(1025, 1, a), Err(Error::InvalidConfiguration)); + assert!(!s.is_configured()); + assert_eq!(s.geometry(), None); + assert_eq!(s.capacity_bytes(), 0); + assert_eq!(s.free_count(), 0); + assert!(s.snapshot().is_empty()); + s.configure(1024, 1, a).unwrap(); + } + + /// Malformed append and synthetic saturation cannot consume or seal slots. + #[test] + fn append_failures_preserve_open_tail_free_list_and_lease_counts() { + let s = segments(1024, 1); + drop(s.append(512).unwrap()); + let before = s.snapshot(); + for length in [0, 1, 513, 1536] { + assert!(s.append(length).is_err()); + assert_eq!(s.snapshot(), before); + assert_eq!(s.open.get(), Some(0)); + assert_eq!(s.free_count(), 0); + assert_eq!(s.slots.borrow()[0].leases.get(), 0); + } + drop(s.append(512).unwrap()); + assert_eq!(s.state(SegmentId(0)), Ok(SegmentState::Sealed)); + let s = segments(1024, 2); + drop(s.append(512).unwrap()); + for (position, length) in [(0, 512), (1, 1024)] { + s.slots.borrow()[position].leases.set(usize::MAX); + let before = s.snapshot(); + assert!(matches!(s.append(length), Err(Error::Busy))); + assert_eq!(s.snapshot(), before); + assert_eq!(s.open.get(), Some(0)); + assert_eq!(*s.free.borrow(), BTreeSet::from([1])); + assert_eq!(s.slots.borrow()[position].leases.get(), usize::MAX); + s.slots.borrow()[position].leases.set(0); + } + drop(s.append(1024).unwrap()); + assert_eq!(s.state(SegmentId(0)), Ok(SegmentState::Sealed)); + let s = segments(1024, 2); + drop(s.append(512).unwrap()); + s.slots.borrow_mut()[1].image.id = SegmentId(u64::MAX); + let before = s.snapshot(); + assert!(matches!(s.append(1024), Err(Error::InvalidConfiguration))); + assert_eq!(s.snapshot(), before); + assert_eq!(s.open.get(), Some(0)); + assert_eq!(*s.free.borrow(), BTreeSet::from([1])); + assert_eq!(s.slots.borrow()[1].leases.get(), 0); + } + + /// Freeze ownership survives table destruction and releases during unwinding. + #[test] + fn freeze_guard_drop_thaws_and_can_outlive_table() { + let s = segments(1024, 1); + let frozen = s.freeze().unwrap(); + let before = s.snapshot(); + assert_eq!(s.restore(before.clone()), Err(Error::Busy)); + assert_eq!(s.validate_restore(&before), Err(Error::Busy)); + assert!(matches!(s.append(512), Err(Error::Busy))); + assert!(matches!(s.freeze(), Err(Error::Busy))); + assert_eq!(s.snapshot(), before); + drop(frozen); + drop(s.append(512).unwrap()); + let frozen = s.freeze().unwrap(); + drop(s); + drop(frozen); + let s = Segments::new(1024); + let frozen = s.freeze().unwrap(); + assert_eq!( + s.configure(1024, 1, Alignment::new(512, 512, 512).unwrap()), + Err(Error::Busy) + ); + drop(frozen); + assert!(!s.is_configured()); + assert_eq!(s.geometry(), None); + let s = segments(1024, 1); + assert!( + std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { + let _frozen = s.freeze().unwrap(); + panic!("simulated owner failure"); + })) + .is_err() + ); + drop(s.append(512).unwrap()); + } + + /// A lease retains both validated dimensions and the nominal table identity. + #[test] + fn geometry_is_shared_by_configuration_and_leases() { + let geometry = + SegmentGeometry::new(4096, 1024, 2, Alignment::new(512, 512, 512).unwrap()).unwrap(); + let s = Segments::from_geometry(geometry).unwrap(); + assert!(s.is_configured()); + assert_eq!(s.geometry(), Some(geometry)); + assert_eq!(s.count(), 2); + assert_eq!(s.free_count(), 2); + let (lease, extent) = s.append(512).unwrap(); + assert_eq!(lease.geometry(), geometry); + assert!(lease.table_identity().matches(&s.table_identity())); + assert_eq!(s.validate_lease(&lease, &extent), Ok(())); + } + + /// Large sparse capacity need not allocate an equally large slot table. + #[test] + fn large_physical_geometry_supports_bounded_partial_tables() { + let capacity = 1024 * (MAX_SEGMENTS + 1); + let alignment = Alignment::new(512, 512, 512).unwrap(); + let full = SegmentGeometry::new(capacity, 1024, MAX_SEGMENTS + 1, alignment).unwrap(); + assert!(matches!( + Segments::from_geometry(full), + Err(Error::InvalidConfiguration) + )); + let s = Segments::new(1024); + assert_eq!( + s.configure(capacity, (MAX_SEGMENTS + 1) as usize, alignment), + Err(Error::InvalidConfiguration) + ); + assert!(!s.is_configured()); + s.configure(capacity, 2, alignment).unwrap(); + assert_eq!(s.capacity_bytes(), capacity); + assert_eq!(s.count(), 2); + assert_eq!(s.geometry().unwrap().segment_count(), 2); + drop(s.append(1024).unwrap()); + assert_eq!(s.free_count(), 1); + } + + /// Existing leases authorize only their captured prefix, including during eviction. + #[test] + fn leases_validate_table_and_captured_used_range_through_eviction() { + let s = segments(1024, 1); + let other = segments(1024, 1); + let (lease, extent) = s.append(512).unwrap(); + let (_, later) = s.append(512).unwrap(); + assert_eq!(s.validate_lease(&lease, &later), Err(Error::Corrupt)); + assert_eq!(other.validate_lease(&lease, &extent), Err(Error::Stale)); + assert_eq!( + lease.validate_extent(&Extent::new(1, 511).unwrap()), + Err(Error::Corrupt) + ); + assert_eq!( + lease.validate_extent(&Extent::new(0, 1).unwrap()), + Err(Error::Corrupt) + ); + s.begin_evict(SegmentId(0)).unwrap(); + assert_eq!(s.validate_lease(&lease, &extent), Ok(())); + assert_eq!( + s.validate(lease.id(), lease.generation(), &extent), + Err(Error::Stale) + ); + assert!(matches!( + s.lease(lease.id(), lease.generation()), + Err(Error::Stale) + )); + assert!(matches!( + s.lease(SegmentId(1), Generation(1)), + Err(Error::Corrupt) + )); + drop(s); + assert_eq!(lease.validate_extent(&extent), Ok(())); + drop(lease); + } + + /// Stored malformed ranges are corruption rather than configuration errors. + #[test] + fn malformed_extents_are_corrupt_not_configuration_errors() { + let s = segments(1024, 1); + let (lease, _) = s.append(1024).unwrap(); + for extent in [ + Extent::new(1, 512).unwrap(), + Extent::new(0, 513).unwrap(), + Extent::new(1024, 512).unwrap(), + ] { + assert_eq!( + s.validate(lease.id(), lease.generation(), &extent), + Err(Error::Corrupt) + ); + } + } + + /// Recovery validates all states atomically and fails before epoch wraparound. + #[test] + fn invalid_restore_states_are_atomic_and_valid_states_round_trip() { + let s = segments(1024, 3); + drop(s.append(1024).unwrap()); + drop(s.append(512).unwrap()); + let before = s.snapshot(); + let mut cases = Vec::new(); + for state in [ + SegmentState::Open, + SegmentState::Sealed, + SegmentState::Evicting, + ] { + let mut images = before.clone(); + images[2].state = state; + cases.push(images); + } + let mut images = before.clone(); + images[0].state = SegmentState::Open; + cases.push(images); + let mut images = before.clone(); + images[0].state = SegmentState::Open; + images[0].used_bytes = 512; + cases.push(images); + for images in cases { + assert_eq!(s.restore(images), Err(Error::Corrupt)); + assert_eq!(s.snapshot(), before); + assert_eq!(s.free_count(), 1); + assert_eq!(s.open.get(), Some(1)); + assert_eq!(s.restore_epoch(), 0); + } + let mut images = before; + images[0].state = SegmentState::Evicting; + s.restore(images).unwrap(); + assert_eq!(s.restore_epoch(), 1); + assert_eq!(s.state(SegmentId(0)), Ok(SegmentState::Evicting)); + assert_eq!(s.state(SegmentId(1)), Ok(SegmentState::Sealed)); + s.recycle(SegmentId(0)).unwrap(); + assert_eq!(s.free_count(), 2); + let before = s.snapshot(); + s.restore_epoch.set(u64::MAX - 1); + assert_eq!(s.validate_restore(&before), Ok(())); + s.restore(before.clone()).unwrap(); + assert_eq!(s.restore_epoch(), u64::MAX); + assert_eq!(s.snapshot(), before); + s.restore_epoch.set(u64::MAX); + assert_eq!(s.validate_restore(&before), Err(Error::Unavailable)); + assert_eq!(s.restore(before.clone()), Err(Error::Unavailable)); + assert_eq!(s.snapshot(), before); + } +} + +/// Bounded sweep state-space coverage, including invalid callbacks and recovery. +#[cfg(test)] +mod clock_tests { + use super::*; + + /// Deterministic index occupancy and removal-call observations. + struct Entries { + counts: RefCell>, + + calls: RefCell>, + } + impl Entries { + /// Populate a synthetic index without changing allocator state. + fn new(counts: Vec) -> Self { + Self { + counts: RefCell::new(counts), + calls: RefCell::new(vec![]), + } + } + } + impl SegmentEntries for Entries { + /// Record and honor each bounded removal request. + fn remove_bounded(&self, id: SegmentId, budget: usize) -> usize { + self.calls.borrow_mut().push((id, budget)); + let mut counts = self.counts.borrow_mut(); + let count = &mut counts[id.0 as usize]; + let removed = (*count).min(budget); + *count -= removed; + removed + } + + /// Read current occupancy after bounded removal. + fn is_empty(&self, id: SegmentId) -> bool { + self.counts.borrow()[id.0 as usize] == 0 + } + } + + /// Create a shared table with enough slots for a bounded clock scenario. + fn segments(count: usize) -> Rc { + let segments = Rc::new(Segments::new(1024)); + segments + .configure( + 1024 * count as u64, + count, + Alignment::new(512, 512, 512).unwrap(), + ) + .unwrap(); + segments + } + + /// Empty tables and zero visits never invoke mapping removal. + #[test] + fn empty_and_zero_budget_sweeps_do_not_call_entries() { + let clock = SegmentClock::new(Rc::new(Segments::new(1024))); + let entries = Entries::new(vec![]); + assert_eq!(clock.reclaim_index(&entries, 64, || true), Ok(())); + assert_eq!( + clock.reclaim_index(&entries, 64, || false), + Err(Error::Busy) + ); + assert_eq!(clock.reclaim(&entries, 1, 64, 256), Err(Error::Unavailable)); + assert_eq!(clock.mark_read(SegmentId(0)), Err(Error::Corrupt)); + assert!(entries.calls.borrow().is_empty()); + let clock = SegmentClock::new(segments(1)); + let entries = Entries::new(vec![1]); + assert_eq!(clock.reclaim_index(&entries, 0, || false), Err(Error::Busy)); + assert!(entries.calls.borrow().is_empty()); + } + + /// Index admission can forget open mappings without reclaiming their storage. + #[test] + fn index_second_chance_removes_open_mappings_without_recycling() { + let segments = segments(2); + drop(segments.append(512).unwrap()); + let clock = SegmentClock::new(segments.clone()); + let entries = Entries::new(vec![1, 1]); + clock.mark_read(SegmentId(0)).unwrap(); + clock + .reclaim_index(&entries, 64, || entries.counts.borrow()[1] == 0) + .unwrap(); + assert_eq!(*entries.counts.borrow(), [1, 0]); + clock.mark_read(SegmentId(0)).unwrap(); + clock + .reclaim_index(&entries, 64, || entries.counts.borrow()[0] == 0) + .unwrap(); + assert_eq!(segments.state(SegmentId(0)), Ok(SegmentState::Open)); + assert!(segments.lease(SegmentId(0), Generation(1)).is_ok()); + assert_eq!(segments.free_count(), 1); + } + + /// Calls resume the prior cursor but never exceed visits or two rotations. + #[test] + fn sweep_budget_persists_cursor_and_limits_to_two_rotations() { + let clock = SegmentClock::new(segments(40)); + let entries = Entries::new(vec![10; 40]); + assert_eq!( + clock.reclaim_index(&entries, 64, || false), + Err(Error::Busy) + ); + assert_eq!(entries.calls.borrow().len(), 64); + assert_eq!(entries.calls.borrow()[63], (SegmentId(23), 1)); + entries.calls.borrow_mut().clear(); + assert_eq!(clock.reclaim_index(&entries, 1, || false), Err(Error::Busy)); + assert_eq!(*entries.calls.borrow(), [(SegmentId(24), 1)]); + let clock = SegmentClock::new(segments(2)); + let entries = Entries::new(vec![10; 2]); + assert_eq!( + clock.reclaim_index(&entries, 64, || false), + Err(Error::Busy) + ); + assert_eq!(entries.calls.borrow().len(), 4); + } + + /// Mapping budget exhaustion leaves partial eviction for the next bounded call. + #[test] + fn mapping_budget_keeps_partial_eviction_until_next_sweep() { + let segments = segments(2); + drop(segments.append(1024).unwrap()); + drop(segments.append(1024).unwrap()); + let clock = SegmentClock::new(segments.clone()); + let entries = Entries::new(vec![257, 0]); + assert_eq!(clock.reclaim(&entries, 2, 64, 256), Err(Error::Busy)); + assert_eq!(*entries.counts.borrow(), [1, 0]); + assert_eq!(*entries.calls.borrow(), [(SegmentId(0), 256)]); + assert_eq!(segments.state(SegmentId(0)), Ok(SegmentState::Evicting)); + assert_eq!(segments.free_count(), 1); + clock.reclaim(&entries, 2, 64, 256).unwrap(); + assert_eq!(segments.free_count(), 2); + assert_eq!(*entries.counts.borrow(), [0, 0]); + } + + /// Freeze checks precede index side effects, and leases precede physical reuse. + #[test] + fn busy_lease_and_frozen_table_preserve_reclaim_side_effect_order() { + let segments = segments(2); + let held = segments.append(1024).unwrap(); + drop(segments.append(1024).unwrap()); + let clock = SegmentClock::new(segments.clone()); + let entries = Entries::new(vec![1, 1]); + let frozen = segments.freeze().unwrap(); + assert_eq!(clock.reclaim(&entries, 2, 64, 256), Err(Error::Busy)); + assert!(entries.calls.borrow().is_empty()); + drop(frozen); + clock.mark_read(SegmentId(1)).unwrap(); + assert_eq!(clock.reclaim(&entries, 2, 64, 256), Err(Error::Busy)); + assert_eq!(segments.state(SegmentId(0)), Ok(SegmentState::Evicting)); + assert_eq!(segments.free_count(), 1); + assert_eq!(clock.mark_read(SegmentId(0)), Ok(())); + assert_eq!(clock.mark_read(SegmentId(1)), Ok(())); + assert!(clock.recent.borrow().is_empty()); + drop(held); + clock.reclaim(&entries, 2, 64, 256).unwrap(); + assert_eq!(segments.free_count(), 2); + assert!(segments.lease(SegmentId(0), Generation(1)).is_err()); + } + + /// Skipping an open segment does not consume its future second chance. + #[test] + fn physical_sweep_preserves_recent_bit_on_ineligible_open_segment() { + let segments = segments(1); + drop(segments.append(512).unwrap()); + let clock = SegmentClock::new(segments.clone()); + let entries = Entries::new(vec![1]); + clock.mark_read(SegmentId(0)).unwrap(); + assert_eq!(clock.reclaim(&entries, 1, 64, 256), Err(Error::Busy)); + drop(segments.append(512).unwrap()); + assert_eq!(clock.reclaim(&entries, 1, 1, 256), Err(Error::Busy)); + assert!(entries.calls.borrow().is_empty()); + clock.reclaim(&entries, 1, 1, 256).unwrap(); + assert_eq!(*entries.calls.borrow(), [(SegmentId(0), 256)]); + } + + /// A no-space rollover creates a reclaimable tail without reserving bytes. + #[test] + fn no_space_rollover_seals_tail_for_reclaim_and_retry() { + for retain_lease in [false, true] { + let segments = segments(1); + let mut held = Some(segments.append(512).unwrap().0); + if !retain_lease { + drop(held.take()); + } + assert!(matches!(segments.append(1024), Err(Error::Busy))); + let image = &segments.snapshot()[0]; + assert_eq!(image.state, SegmentState::Sealed); + assert_eq!(image.used_bytes, 512); + assert_eq!(image.generation, Generation(1)); + let clock = SegmentClock::new(segments.clone()); + let entries = Entries::new(vec![0]); + if retain_lease { + assert_eq!(clock.reclaim(&entries, 1, 2, 0), Err(Error::Busy)); + assert_eq!(segments.state(SegmentId(0)), Ok(SegmentState::Evicting)); + drop(held.take()); + } + clock.reclaim(&entries, 1, 2, 0).unwrap(); + assert!(entries.calls.borrow().is_empty()); + let (lease, extent) = segments.append(1024).unwrap(); + assert_eq!(lease.id(), SegmentId(0)); + assert_eq!(lease.generation(), Generation(2)); + assert_eq!(extent.offset(), 0); + assert_eq!(extent.length(), 1024); + } + } + + /// Zero mapping work cannot start populated eviction but may recycle empty slots. + #[test] + fn zero_budget_does_not_start_populated_eviction_but_recycles_empty_candidates() { + let segments = segments(2); + drop(segments.append(1024).unwrap()); + drop(segments.append(1024).unwrap()); + let clock = SegmentClock::new(segments.clone()); + let entries = Entries::new(vec![1, 0]); + clock.reclaim(&entries, 1, 2, 0).unwrap(); + assert_eq!(segments.state(SegmentId(0)), Ok(SegmentState::Sealed)); + assert_eq!(segments.state(SegmentId(1)), Ok(SegmentState::Free)); + assert_eq!(*entries.counts.borrow(), [1, 0]); + assert!(entries.calls.borrow().is_empty()); + assert_eq!(clock.reclaim(&entries, 2, 64, 0), Err(Error::Busy)); + assert_eq!(segments.state(SegmentId(0)), Ok(SegmentState::Sealed)); + } + + /// A zero reserve does not inspect or change allocation and mapping state. + #[test] + fn zero_reserve_has_no_side_effects_even_when_unconfigured() { + let unconfigured = SegmentClock::new(Rc::new(Segments::new(1024))); + assert_eq!( + unconfigured.reclaim(&Entries::new(vec![]), 0, 64, 256), + Ok(()) + ); + let segments = segments(1); + drop(segments.append(1024).unwrap()); + let clock = SegmentClock::new(segments.clone()); + let entries = Entries::new(vec![1]); + clock.mark_read(SegmentId(0)).unwrap(); + assert_eq!(clock.reclaim(&entries, 0, 64, 256), Ok(())); + assert_eq!(segments.state(SegmentId(0)), Ok(SegmentState::Sealed)); + assert!(entries.calls.borrow().is_empty()); + assert!(clock.recent.borrow().contains(&SegmentId(0))); + } + /// Scoring completes before mutation; mapping budgets, freeze, leases and + /// generation authority remain enforced even when the cheapest slot is busy. + #[test] + fn scored_eviction_preserves_fences_and_partial_progress() { + let segments = segments(3); + drop(segments.append(1024).unwrap()); + let held = segments.append(1024).unwrap().0; + drop(segments.append(1024).unwrap()); + let clock = SegmentClock::new(segments.clone()); + let entries = Entries::new(vec![1, 2, 1]); + let scores = [30, 0, 10]; + let frozen = segments.freeze().unwrap(); + assert_eq!( + clock.reclaim_scored(&entries, 1, 64, 1, |id| scores[id.0 as usize]), + Err(Error::Busy) + ); + assert!(entries.calls.borrow().is_empty()); + drop(frozen); + assert_eq!( + clock.reclaim_scored(&entries, 1, 64, 1, |id| { + assert_eq!(segments.state(id), Ok(SegmentState::Sealed)); + scores[id.0 as usize] + }), + Err(Error::Busy) + ); + assert_eq!(*entries.counts.borrow(), [1, 1, 1]); + assert_eq!(segments.state(SegmentId(1)), Ok(SegmentState::Evicting)); + assert!(segments.lease(SegmentId(1), Generation(1)).is_err()); + assert_eq!( + clock.reclaim_scored(&entries, 1, 64, 2, |id| scores[id.0 as usize]), + Err(Error::Busy) + ); + assert_eq!(*entries.counts.borrow(), [1, 0, 1]); + assert_eq!(segments.state(SegmentId(0)), Ok(SegmentState::Sealed)); + assert_eq!(segments.state(SegmentId(1)), Ok(SegmentState::Evicting)); + drop(held); + clock + .reclaim_scored(&entries, 1, 64, 0, |_| u64::MAX) + .unwrap(); + assert_eq!(segments.free_count(), 1); + assert_eq!(segments.snapshot()[1].generation, Generation(2)); + } + /// A score callback never turns a soft preference into immunity or a full scan. + #[test] + fn scored_eviction_caps_candidates_and_evicts_maximum_scores() { + let segments = segments(65); + for _ in 0..65 { + drop(segments.append(1024).unwrap()); + } + let clock = SegmentClock::new(segments.clone()); + let entries = Entries::new(vec![1; 65]); + let mut visits = 0; + clock + .reclaim_scored(&entries, 1, usize::MAX, 1, |_| { + visits += 1; + u64::MAX + }) + .unwrap(); + assert_eq!(visits, 64); + assert_eq!(entries.calls.borrow().len(), 1); + assert_eq!(segments.free_count(), 1); + assert_eq!(clock.hand.get(), 64); + } + /// Pending leased victims count toward reserve even outside the next sample. + #[test] + fn scored_eviction_never_overshoots_reserve_with_multiple_leased_victims() { + let segments = segments(3); + let mut leases: Vec<_> = (0..3) + .map(|_| Some(segments.append(1024).unwrap().0)) + .collect(); + let clock = SegmentClock::new(segments.clone()); + let entries = Entries::new(vec![1; 3]); + for _ in 0..6 { + assert_eq!( + clock.reclaim_scored(&entries, 1, 1, 256, |_| 0), + Err(Error::Busy) + ); + assert_eq!(*entries.counts.borrow(), [0, 1, 1]); + assert_eq!(segments.evicting.get(), 1); + } + drop(leases[0].take()); + clock.reclaim_scored(&entries, 1, 64, 256, |_| 0).unwrap(); + assert_eq!(segments.free_count(), 1); + assert_eq!(segments.evicting.get(), 0); + assert_eq!(*entries.counts.borrow(), [0, 1, 1]); + } + + /// Over-reporting callbacks return configuration errors without unsafe reuse. + #[test] + fn removal_contract_violations_return_errors_without_panicking() { + /// An intentionally broken callback that over-reports every removal. + struct InvalidEntries; + impl SegmentEntries for InvalidEntries { + /// Violate the caller contract to test error handling. + fn remove_bounded(&self, _: SegmentId, budget: usize) -> usize { + budget + 1 + } + + /// Keep the segment populated despite the invalid removal report. + fn is_empty(&self, _: SegmentId) -> bool { + false + } + } + let segments = segments(1); + drop(segments.append(1024).unwrap()); + let clock = SegmentClock::new(segments.clone()); + assert_eq!( + clock.reclaim_index(&InvalidEntries, 1, || false), + Err(Error::InvalidConfiguration) + ); + assert_eq!( + clock.reclaim(&InvalidEntries, 1, 1, 1), + Err(Error::InvalidConfiguration) + ); + assert_eq!(segments.state(SegmentId(0)), Ok(SegmentState::Evicting)); + assert_eq!(segments.free_count(), 0); + } + + /// Every clock operation observes the restore epoch before using old history. + #[test] + fn restore_resets_cursor_and_recent_reads_before_any_clock_operation() { + let segments = segments(2); + drop(segments.append(1024).unwrap()); + drop(segments.append(1024).unwrap()); + let clock = SegmentClock::new(segments.clone()); + let entries = Entries::new(vec![10, 10]); + assert_eq!(clock.reclaim_index(&entries, 1, || false), Err(Error::Busy)); + clock.mark_read(SegmentId(0)).unwrap(); + assert_eq!(clock.hand.get(), 1); + segments.restore(segments.snapshot()).unwrap(); + entries.calls.borrow_mut().clear(); + assert_eq!(clock.reclaim_index(&entries, 1, || false), Err(Error::Busy)); + assert_eq!(*entries.calls.borrow(), [(SegmentId(0), 1)]); + clock.mark_read(SegmentId(1)).unwrap(); + segments.restore(segments.snapshot()).unwrap(); + clock.mark_read(SegmentId(0)).unwrap(); + assert_eq!(clock.hand.get(), 0); + assert_eq!(*clock.recent.borrow(), HashSet::from([SegmentId(0)])); + segments.restore(segments.snapshot()).unwrap(); + clock.reclaim(&entries, 0, 0, 0).unwrap(); + assert!(clock.recent.borrow().is_empty()); + } +} diff --git a/cmd/racer-dataplane/alloc/src/slab.rs b/cmd/racer-dataplane/alloc/src/slab.rs new file mode 100644 index 000000000..326037186 --- /dev/null +++ b/cmd/racer-dataplane/alloc/src/slab.rs @@ -0,0 +1,1314 @@ +//! Locked sparse storage and completion-owned direct I/O. +//! +//! Startup is blocking and belongs outside the latency-sensitive worker path. +//! Capacity bounds logical addresses, not reserved physical space: writes can +//! still fail with ENOSPC. Recycling never truncates or erases disk bytes. +//! Parent directories must be trusted against hostile rename and unlink, even +//! though descriptor-relative traversal rejects symlinks and parent components. +use crate::segments::TableIdentity; +use crate::{ + AlignedBuffer, Alignment, Charge, Error, Extent, Result, SegmentGeometry, SegmentLease, + Segments, +}; +use std::{ + cell::{Cell, RefCell}, + ffi::CString, + fs::File, + future::Future, + os::{ + fd::{AsRawFd, FromRawFd}, + unix::{ffi::OsStrExt, fs::MetadataExt}, + }, + path::{Component, Path, PathBuf}, + pin::Pin, + rc::Rc, + task::{Context, Poll, Waker}, +}; +use uring_runtime::{ + Budget, Operation, Scope, + reactor::{Reactor, descriptor::Descriptor}, +}; + +/// One sparse direct-I/O cache file, not a durable storage transaction. +/// `open_configured` binds submissions to one segment table; read/write require +/// a successful binding, not merely an open file. +#[repr(align(64))] +pub struct Slab { + path: PathBuf, + + capacity_bytes: u64, + + segment_bytes: u64, + + max_record_bytes: usize, + + opened: RefCell>, + + writes: Rc, + + idle_buffer: Rc>>>, +} +impl Slab { + /// Describe a worker's file; validation and blocking I/O happen at startup. + pub fn new( + path: PathBuf, + capacity_bytes: u64, + segment_bytes: u64, + max_record_bytes: usize, + ) -> Self { + Self { + path, + capacity_bytes, + segment_bytes, + max_record_bytes, + opened: RefCell::new(None), + writes: Rc::new(WriteState::default()), + idle_buffer: Rc::new(RefCell::new(None)), + } + } + /// Logical file capacity, not a reservation of physical disk blocks. + pub fn capacity_bytes(&self) -> u64 { + self.capacity_bytes + } + /// Size of each physical segment. + pub fn segment_bytes(&self) -> u64 { + self.segment_bytes + } + /// Accepted writes whose completion guards have not yet been released. + pub fn writes_in_flight(&self) -> usize { + self.writes.count.get() + } + /// Accounted bytes in the single idle buffer slot. + pub fn idle_bytes(&self) -> usize { + self.idle_buffer.borrow().as_ref().map_or(0, |b| b.len()) + } + /// Release only idle memory, returning its size without touching live I/O. + pub fn reclaim_idle(&self) -> usize { + let idle = self.idle_buffer.borrow_mut().take(); + idle.map_or(0, |b| b.len()) + } + /// Wait for accepted writes to release their runtime completion fences. + /// This does not call fsync/fdatasync and does NOT promise crash durability. + /// Drive the reactor concurrently; stop new writes first if quiescence is needed. + pub fn fence_writes(&self) -> Operation<'_, (), Error> { + Box::pin(FenceWaiter { + state: self.writes.clone(), + registration: None, + }) + } + /// Discovered direct-I/O requirements, or Unavailable before startup. + pub fn alignment(&self) -> Result { + self.opened + .borrow() + .as_ref() + .map(|o| o.geometry.alignment()) + .ok_or(Error::Unavailable) + } + /// Physical slab geometry, including the full segment capacity. A bound + /// table may deliberately expose fewer segments than this physical count. + pub fn geometry(&self) -> Result { + self.opened + .borrow() + .as_ref() + .map(|o| o.geometry) + .ok_or(Error::Unavailable) + } + /// Configure an empty table, or validate an already configured partial table, + /// and permanently bind this slab to that table's identity. + #[cfg(any(test, feature = "simulation"))] + pub fn configure_segments(&self, segments: &Segments) -> Result<()> { + self.bind(segments) + } + + /// Attach the table identity to this file only after all dimensions agree. + fn bind(&self, segments: &Segments) -> Result<()> { + let mut opened = self.opened.borrow_mut(); + let slab = opened.as_mut().ok_or(Error::Unavailable)?; + let geometry = slab.geometry; + let identity = segments.table_identity(); + if let Some(bound) = slab.table.as_ref() + && !bound.matches(&identity) + { + return Err(Error::InvalidConfiguration); + } + if segments.segment_bytes() != geometry.segment_bytes() { + return Err(Error::InvalidConfiguration); + } + if !segments.is_configured() { + let count = usize::try_from(geometry.segment_count()) + .map_err(|_| Error::InvalidConfiguration)?; + segments.configure_table(geometry.slab_bytes(), count, geometry.alignment())?; + } + let configured = segments.geometry().ok_or(Error::InvalidConfiguration)?; + if configured.slab_bytes() != geometry.slab_bytes() + || configured.segment_bytes() != geometry.segment_bytes() + || configured.alignment() != geometry.alignment() + || configured.segment_count() > geometry.segment_count() + { + return Err(Error::InvalidConfiguration); + } + slab.table = Some(identity); + Ok(()) + } + /// Blocking startup helper. Do not invoke on a latency-sensitive worker. + pub fn open_configured(&self, segments: &Segments) -> Result { + let alignment = self.open_file()?; + self.bind(segments)?; + Ok(alignment) + } + /// Blocking startup I/O for geometry probing and buffer allocation. Read/write + /// return `Unavailable` until `configure_segments` succeeds. + /// Parent directories must be trusted against + /// rename/unlink by other users. Linux traversal rejects symlinks and `..`; + /// newly created directories are private. Existing files must be owned by the + /// effective user, regular, singly linked, and mode 0600. + #[cfg(any(test, feature = "simulation"))] + pub fn open_now(&self) -> Result { + self.open_file() + } + + /// Open and validate the physical file without granting submission authority. + fn open_file(&self) -> Result { + if let Ok(a) = self.alignment() { + return Ok(a); + } + if self.segment_bytes == 0 + || self.capacity_bytes == 0 + || self.max_record_bytes == 0 + || !self.capacity_bytes.is_multiple_of(self.segment_bytes) + || self.capacity_bytes > i64::MAX as u64 + { + return Err(Error::InvalidConfiguration); + } + #[cfg(feature = "simulation")] + if let Some(sim) = uring_runtime::reactor::simulation::Simulation::current() { + if let Some(parent) = self.path.parent().filter(|p| !p.as_os_str().is_empty()) { + sim.create_dir_all(parent) + .map_err(|e| system_error("mkdir", e))?; + } + let file = sim + .open( + None, + &self.path, + libc::O_CREAT + | libc::O_RDWR + | libc::O_DIRECT + | libc::O_CLOEXEC + | libc::O_NOFOLLOW, + ) + .map_err(|e| direct_error("open", e))?; + let handle = file.as_sim().expect("simulation descriptor"); + handle.lock().map_err(lock_error)?; + let stat = handle.stat().map_err(|e| system_error("statx", e))?; + // The virtual filesystem has a fixed root owner, independent of host uid. + validate_file(stat.stx_mode as u32, stat.stx_uid, 0, stat.stx_nlink as u64)?; + let a = alignment_from_stat(&stat)?; + let geometry = self.validate_layout(a, stat.stx_size)?; + if stat.stx_size == 0 { + handle + .set_len(self.capacity_bytes) + .map_err(|e| system_error("ftruncate", e))?; + } + self.publish(file, geometry); + return Ok(a); + } + let file = open_private_file(&self.path)?; + let metadata = file.metadata().map_err(|e| system_error("fstat", e))?; + // SAFETY: geteuid has no arguments or borrowed memory. + validate_file( + metadata.mode(), + metadata.uid(), + unsafe { libc::geteuid() }, + metadata.nlink(), + )?; + // SAFETY: flock synchronously borrows this live descriptor. + if unsafe { libc::flock(file.as_raw_fd(), libc::LOCK_EX | libc::LOCK_NB) } != 0 { + return Err(lock_error(std::io::Error::last_os_error())); + } + // Refresh size under the lock: a previous lock holder may have resized + // the file between the first fstat and acquiring our lock. + let size = file.metadata().map_err(|e| system_error("fstat", e))?.len(); + // Delay O_DIRECT until after rejecting nonregular files (including FIFOs). + // SAFETY: fcntl synchronously borrows this live descriptor. + let flags = unsafe { libc::fcntl(file.as_raw_fd(), libc::F_GETFL) }; + if flags < 0 { + return Err(system_error("fcntl-getfl", std::io::Error::last_os_error())); + } + if unsafe { + libc::fcntl( + file.as_raw_fd(), + libc::F_SETFL, + (flags | libc::O_DIRECT) & !libc::O_NONBLOCK, + ) + } < 0 + { + return Err(direct_error( + "fcntl-direct", + std::io::Error::last_os_error(), + )); + } + let a = probe(&file)?; + let geometry = self.validate_layout(a, size)?; + if size == 0 { + file.set_len(self.capacity_bytes) + .map_err(|e| system_error("ftruncate", e))?; + } + self.publish(file.into(), geometry); + Ok(a) + } + /// Publish geometry with its owning file, initially without I/O authority. + fn publish(&self, file: Descriptor, geometry: SegmentGeometry) { + *self.opened.borrow_mut() = Some(OpenSlab { + file: Rc::new(file), + geometry, + table: None, + }); + } + /// Check physical dimensions and padded record size without changing the file. + fn validate_layout(&self, a: Alignment, size: u64) -> Result { + if a.extent(0, self.max_record_bytes)?.length() as u64 > self.segment_bytes + || !self.segment_bytes.is_multiple_of(a.offset()) + || !self.segment_bytes.is_multiple_of(a.length() as u64) + || (size != 0 && size != self.capacity_bytes) + { + return Err(Error::InvalidConfiguration); + } + SegmentGeometry::new( + self.capacity_bytes, + self.segment_bytes, + self.capacity_bytes / self.segment_bytes, + a, + ) + .map_err(|_| Error::InvalidConfiguration) + } + /// Borrow-check a request before moving its resources into a submission. + fn submission( + &self, + extent: Extent, + buffer: &AlignedBuffer, + lease: &SegmentLease, + ) -> Result> { + let opened = self.opened.borrow(); + let slab = opened.as_ref().ok_or(Error::Unavailable)?; + let table = slab.table.as_ref().ok_or(Error::Unavailable)?; + slab.geometry.alignment().check(extent, buffer)?; + if !table.matches(&lease.table_identity()) { + return Err(Error::Stale); + } + let geometry = lease.geometry(); + if geometry.slab_bytes() != slab.geometry.slab_bytes() + || geometry.segment_bytes() != slab.geometry.segment_bytes() + || geometry.alignment() != slab.geometry.alignment() + { + return Err(Error::Corrupt); + } + lease.validate_extent(&extent)?; + let start = lease + .id() + .0 + .checked_mul(self.segment_bytes) + .ok_or(Error::Corrupt)?; + let end = extent + .offset() + .checked_add(extent.length() as u64) + .ok_or(Error::Corrupt)?; + if extent.offset() < start + || end + > start + .checked_add(self.segment_bytes) + .ok_or(Error::Corrupt)? + || end > self.capacity_bytes + { + return Err(Error::Corrupt); + } + Ok(slab.file.clone()) + } + /// Consume a checked request so its descriptor, buffer, and lease travel together. + fn prepare( + &self, + extent: Extent, + buffer: AlignedBuffer, + lease: SegmentLease, + ) -> Result> { + let file = self.submission(extent, &buffer, &lease)?; + Ok(Submission { + file, + extent, + buffer, + lease, + }) + } + /// Read exactly one checked extent, retaining its buffer and lease until completion. + /// Dropping the waiting future does not release kernel-owned resources. + pub fn read<'a, S: Scope, B: Budget>( + &'a self, + reactor: &'a Reactor, + extent: Extent, + buffer: AlignedBuffer, + lease: SegmentLease, + scope: &'a S, + ) -> Operation<'a, AlignedBuffer, S::Error> + where + S::Error: From, + { + Box::pin(async move { + self.prepare(extent, buffer, lease)? + .read(reactor, scope) + .await + }) + } + /// Write exactly one checked extent with completion-owned accounting and lease. + /// Failed writes do not roll back the space reserved by append. + pub fn write<'a, S: Scope, B: Budget>( + &'a self, + reactor: &'a Reactor, + extent: Extent, + buffer: AlignedBuffer, + lease: SegmentLease, + scope: &'a S, + ) -> Operation<'a, AlignedBuffer, S::Error> + where + S::Error: From, + { + Box::pin(async move { + let submission = self.prepare(extent, buffer, lease)?; + let fence = self.writes.acquire()?; + submission.write(reactor, scope, fence).await + }) + } + /// Reuse one exact-size idle buffer. A size mismatch releases the old idle + /// buffer and its charge, even if the replacement allocation fails. + pub fn allocate(&self, length: usize, charge: C) -> Result> { + let alignment = self.alignment()?; + if !charge.covers(length) { + return Err(Error::InvalidConfiguration); + } + let idle = self.idle_buffer.borrow_mut().take(); + if let Some(mut buffer) = idle + && buffer.len() == length + { + buffer.rebind(charge)?; + return Ok(buffer.pooled(&self.idle_buffer)); + } + Ok(alignment + .allocate(length, charge)? + .pooled(&self.idle_buffer)) + } + /// Replace the open file for fault-injection tests; geometry remains unchanged. + #[cfg(feature = "simulation")] + #[doc(hidden)] + pub fn replace_file_for_test(&self, file: File) -> Result<()> { + self.replace_descriptor_for_test(file.into()) + } + /// Also accepts a virtual descriptor from the simulation backend. + #[cfg(feature = "simulation")] + #[doc(hidden)] + pub fn replace_descriptor_for_test(&self, file: Descriptor) -> Result<()> { + if self.writes_in_flight() != 0 { + return Err(Error::Busy); + } + self.opened + .borrow_mut() + .as_mut() + .ok_or(Error::Unavailable)? + .file = Rc::new(file); + Ok(()) + } +} +/// A file and its geometry own their binding; a closed slab cannot retain authority. +struct OpenSlab { + file: Rc, + + geometry: SegmentGeometry, + + table: Option, +} + +/// A validated transfer owns every resource needed to keep kernel access safe. +/// Only Slab::prepare constructs this capability, and submission consumes it. +struct Submission { + file: Rc, + + extent: Extent, + + buffer: AlignedBuffer, + + lease: SegmentLease, +} + +impl Submission { + /// Move the complete read capability into the reactor's completion ownership. + async fn read( + self, + reactor: &Reactor, + scope: &S, + ) -> std::result::Result, S::Error> + where + S::Error: From, + { + let completion = reactor + .read_at( + self.file, + self.extent.offset(), + self.buffer, + self.lease, + scope, + ) + .await?; + Self::complete(self.extent, completion.bytes, completion.buffer) + } + + /// Move the write capability and its counter guard into completion ownership. + async fn write( + self, + reactor: &Reactor, + scope: &S, + fence: WriteFence, + ) -> std::result::Result, S::Error> + where + S::Error: From, + { + let completion = reactor + .write_at( + self.file, + self.extent.offset(), + self.buffer, + (self.lease, fence), + scope, + ) + .await?; + Self::complete(self.extent, completion.bytes, completion.buffer) + } + + /// Return storage only after an exact-length completion; short I/O drops it. + fn complete>( + extent: Extent, + bytes: usize, + buffer: AlignedBuffer, + ) -> std::result::Result, E> { + if bytes != extent.length() { + return Err(Error::Io.into()); + } + Ok(buffer) + } +} + +/// Worker-local write counters and fence registrations, independent of slab lifetime. +#[derive(Default)] +struct WriteState { + count: Cell, + + waiters: RefCell>>>, +} +impl WriteState { + /// Increment before constructing the sole guard responsible for decrementing. + fn acquire(self: &Rc) -> Result { + self.count + .set(self.count.get().checked_add(1).ok_or(Error::Busy)?); + Ok(WriteFence(self.clone())) + } +} +/// Cancel-safe registration waiting for all currently accepted writes to finish. +struct FenceWaiter { + state: Rc, + + registration: Option>>, +} +impl Future for FenceWaiter { + type Output = Result<()>; + + /// Refresh the task's registration without busy-waking the worker. + fn poll(mut self: Pin<&mut Self>, cx: &mut Context<'_>) -> Poll { + if self.state.count.get() == 0 { + self.unregister(); + return Poll::Ready(Ok(())); + } + if let Some(registration) = &self.registration { + registration.borrow_mut().clone_from(cx.waker()); + // Final completion drained registrations; a new write can precede + // our next poll, in which case this waiter must register again. + let mut waiters = self.state.waiters.borrow_mut(); + if !waiters.iter().any(|w| Rc::ptr_eq(w, registration)) { + waiters.push(registration.clone()); + } + } else { + let registration = Rc::new(RefCell::new(cx.waker().clone())); + self.state.waiters.borrow_mut().push(registration.clone()); + self.registration = Some(registration); + } + Poll::Pending + } +} +impl FenceWaiter { + /// Remove only this waiter's registration, including after cancellation. + fn unregister(&mut self) { + if let Some(registration) = self.registration.take() { + self.state + .waiters + .borrow_mut() + .retain(|w| !Rc::ptr_eq(w, ®istration)); + } + } +} +impl Drop for FenceWaiter { + /// A canceled wait must not retain its task's waker. + fn drop(&mut self) { + self.unregister(); + } +} +/// Unique write-count decrement authority retained by the reactor completion. +struct WriteFence(Rc); +impl Drop for WriteFence { + /// Release the count and wake sleepers outside all registration borrows. + fn drop(&mut self) { + let remaining = self.0.count.get() - 1; + self.0.count.set(remaining); + if remaining == 0 { + let waiters = std::mem::take(&mut *self.0.waiters.borrow_mut()); + for waiter in waiters { + let waker = waiter.borrow().clone(); + waker.wake(); + } + } + } +} +/// Preserve synchronous operating-system diagnostic context. +fn system_error(operation: &'static str, error: std::io::Error) -> Error { + Error::SystemIo { + operation, + errno: error.raw_os_error(), + } +} +/// Distinguish an already locked file from a failed locking syscall. +fn lock_error(error: std::io::Error) -> Error { + if error.raw_os_error() == Some(libc::EWOULDBLOCK) { + Error::Unavailable + } else { + system_error("flock", error) + } +} +/// Classify explicit direct-I/O capability denials without hiding other failures. +fn direct_error(operation: &'static str, error: std::io::Error) -> Error { + match error.raw_os_error() { + Some(libc::EINVAL | libc::EOPNOTSUPP | libc::ENOSYS) => Error::Unsupported, + _ => system_error(operation, error), + } +} +/// Discover direct-I/O requirements for an owned file descriptor. +fn probe(file: &File) -> Result { + probe_fd(file.as_raw_fd()) +} +/// Ask Linux for descriptor-specific alignment without assuming a page size. +fn probe_fd(fd: i32) -> Result { + // SAFETY: initialized statx output and valid empty C path. Invalid FDs are + // rejected by the kernel without accessing user memory through the FD. + let mut stat: libc::statx = unsafe { std::mem::zeroed() }; + if unsafe { + libc::statx( + fd, + c"".as_ptr(), + libc::AT_EMPTY_PATH, + libc::STATX_DIOALIGN, + &mut stat, + ) + } != 0 + { + let error = std::io::Error::last_os_error(); + return Err(match error.raw_os_error() { + Some(libc::ENOSYS | libc::EOPNOTSUPP) => Error::Unsupported, + _ => system_error("statx", error), + }); + } + alignment_from_stat(&stat) +} +/// Reject missing alignment capability or invalid values returned by statx. +fn alignment_from_stat(stat: &libc::statx) -> Result { + if stat.stx_mask & libc::STATX_DIOALIGN == 0 { + return Err(Error::Unsupported); + } + Alignment::new( + stat.stx_dio_mem_align as usize, + stat.stx_dio_offset_align as u64, + stat.stx_dio_offset_align as usize, + ) +} + +/// Require a private regular file with one link and the expected owner. +fn validate_file(mode: u32, owner: u32, expected_owner: u32, links: u64) -> Result<()> { + if mode & libc::S_IFMT != libc::S_IFREG + || mode & 0o7777 != 0o600 + || owner != expected_owner + || links != 1 + { + return Err(Error::InvalidConfiguration); + } + Ok(()) +} + +/// Resolve each directory relative to its already-open predecessor. Unlike +/// create_dir_all plus open, this never follows an intermediate symlink. +fn open_private_file(path: &Path) -> Result { + let name = path.file_name().ok_or(Error::InvalidConfiguration)?; + if path.components().any(|c| matches!(c, Component::ParentDir)) { + return Err(Error::InvalidConfiguration); + } + let anchor = if path.is_absolute() { c"/" } else { c"." }; + // SAFETY: valid C path, no borrowed memory retained by open. + let fd = unsafe { + libc::open( + anchor.as_ptr(), + libc::O_PATH | libc::O_DIRECTORY | libc::O_CLOEXEC, + ) + }; + if fd < 0 { + return Err(system_error("open-parent", std::io::Error::last_os_error())); + } + // SAFETY: open returned a new, uniquely owned descriptor. + let mut parent = unsafe { File::from_raw_fd(fd) }; + for component in path.parent().unwrap_or(Path::new("")).components() { + let Component::Normal(component) = component else { + continue; + }; + let component = + CString::new(component.as_bytes()).map_err(|_| Error::InvalidConfiguration)?; + // SAFETY: parent FD and C component remain valid throughout each syscall. + let flags = libc::O_PATH | libc::O_DIRECTORY | libc::O_CLOEXEC | libc::O_NOFOLLOW; + let mut next = unsafe { libc::openat(parent.as_raw_fd(), component.as_ptr(), flags) }; + if next < 0 && std::io::Error::last_os_error().raw_os_error() == Some(libc::ENOENT) { + if unsafe { libc::mkdirat(parent.as_raw_fd(), component.as_ptr(), 0o700) } != 0 + && std::io::Error::last_os_error().raw_os_error() != Some(libc::EEXIST) + { + return Err(system_error("mkdirat", std::io::Error::last_os_error())); + } + next = unsafe { libc::openat(parent.as_raw_fd(), component.as_ptr(), flags) }; + } + if next < 0 { + return Err(system_error("open-parent", std::io::Error::last_os_error())); + } + // SAFETY: successful openat returned a uniquely owned descriptor. + parent = unsafe { File::from_raw_fd(next) }; + } + let name = CString::new(name.as_bytes()).map_err(|_| Error::InvalidConfiguration)?; + // O_NONBLOCK prevents a malicious FIFO from hanging startup before fstat. + // SAFETY: live directory FD and NUL-terminated name; mode supplied for O_CREAT. + let fd = unsafe { + libc::openat( + parent.as_raw_fd(), + name.as_ptr(), + libc::O_RDWR | libc::O_CREAT | libc::O_CLOEXEC | libc::O_NOFOLLOW | libc::O_NONBLOCK, + 0o600, + ) + }; + if fd < 0 { + return Err(system_error("open", std::io::Error::last_os_error())); + } + // SAFETY: successful openat returned a uniquely owned descriptor. + Ok(unsafe { File::from_raw_fd(fd) }) +} + +/// Private storage invariants and synchronous Linux file validation. +#[cfg(test)] +mod tests { + use super::*; + use crate::{Generation, SegmentId}; + + /// Counts retained bytes independently of buffer pooling. + struct CountingCharge { + used: Rc>, + + bytes: usize, + } + impl CountingCharge { + /// Admit a fixed number of bytes. + fn new(used: &Rc>, bytes: usize) -> Self { + used.set(used.get() + bytes); + Self { + used: used.clone(), + bytes, + } + } + } + impl Charge for CountingCharge { + /// Cover only bytes actually admitted by this guard. + fn covers(&self, bytes: usize) -> bool { + self.bytes >= bytes + } + } + impl Drop for CountingCharge { + /// Release accounting exactly once on destruction. + fn drop(&mut self) { + self.used.set(self.used.get() - self.bytes); + } + } + + /// Owns and cleans up an isolated directory inside the project build tree. + struct Directory(PathBuf); + impl Directory { + /// Create a per-process, per-test directory without touching host temp paths. + fn new() -> Self { + static NEXT: std::sync::atomic::AtomicUsize = std::sync::atomic::AtomicUsize::new(0); + let id = NEXT.fetch_add(1, std::sync::atomic::Ordering::Relaxed); + let path = PathBuf::from(env!("CARGO_MANIFEST_DIR")) + .join("target") + .join(format!("page-alloc-test-{}-{id}", std::process::id())); + std::fs::create_dir_all(&path).unwrap(); + Self(path) + } + } + impl Drop for Directory { + /// Remove only this test's owned directory. + fn drop(&mut self) { + let _ = std::fs::remove_dir_all(&self.0); + } + } + + /// Skip only explicit unsupported-filesystem results unless real I/O is required. + fn real_alignment(result: Result) -> Option { + match result { + Ok(alignment) => Some(alignment), + Err(Error::Unsupported) => { + assert_ne!( + std::env::var("PAGE_ALLOC_REQUIRE_REAL_IO").as_deref(), + Ok("1"), + "PAGE_ALLOC_REQUIRE_REAL_IO=1 forbids capability skips: filesystem does not support direct I/O geometry" + ); + eprintln!("SKIP real slab test: filesystem does not support direct I/O geometry"); + None + } + Err(error) => panic!("unexpected real slab startup failure: {error}"), + } + } + + /// Pooling preserves the primary charge and zeroes bytes before reuse. + #[test] + fn aligned_pool_reuses_only_fenced_zeroed_admitted_storage() { + let used = Rc::new(Cell::new(0)); + let pool = Rc::new(RefCell::new(None)); + let alignment = Alignment::new(512, 512, 512).unwrap(); + let charge = || CountingCharge::new(&used, 512); + let mut buffer = alignment.allocate(512, charge()).unwrap().pooled(&pool); + let pointer = buffer.bytes().unwrap().as_ptr(); + buffer.bytes_mut().unwrap().fill(42); + let extra = Rc::new(charge()); + let weak = Rc::downgrade(&extra); + buffer.retain(extra); + assert_eq!(used.get(), 1024); + drop(buffer); + assert!(weak.upgrade().is_none()); + assert_eq!(used.get(), 512); + let mut reused = pool.borrow_mut().take().unwrap(); + assert_eq!(reused.bytes().unwrap().as_ptr(), pointer); + assert!(reused.bytes().unwrap().iter().all(|b| *b == 0)); + reused.rebind(charge()).unwrap(); + assert_eq!(used.get(), 512); + drop(reused.pooled(&pool)); + drop(pool); + assert_eq!(used.get(), 0); + } + + /// Geometry obeys the discovered units, not assumed memory-page dimensions. + #[test] + fn geometry_rounds_without_assuming_page_size() { + let a = Alignment::new(512, 512, 1024).unwrap(); + assert_eq!(a.extent(512, 1025).unwrap().length(), 2048); + assert!(a.extent(1, 1).is_err()); + assert!(a.extent(0, usize::MAX).is_err()); + assert!(Alignment::new(3, 512, 512).is_err()); + assert!(Extent::new(u64::MAX, 1).is_err()); + assert!(Extent::new(0, 0).is_err()); + assert!(Alignment::new(512, 0, 512).is_err()); + assert!(Alignment::new(512, 512, 0).is_err()); + let a = Alignment::new(512, 768, 512).unwrap(); + assert_eq!(a.extent(0, 513).unwrap().length(), 1536); + let b = a.allocate(512, ()).unwrap(); + assert!(!b.is_empty()); + assert_eq!( + a.check(Extent::new(1, 512).unwrap(), &b), + Err(Error::InvalidConfiguration) + ); + assert_eq!( + a.check(Extent::new(0, 1024).unwrap(), &b), + Err(Error::InvalidConfiguration) + ); + } + + /// Failed admission and borrowed return slots cannot leak accounted bytes. + #[test] + fn invalid_charge_is_released_and_pool_borrow_does_not_panic() { + let used = Rc::new(Cell::new(0)); + let a = Alignment::new(512, 512, 512).unwrap(); + assert!(matches!( + a.allocate(512, CountingCharge::new(&used, 511)), + Err(Error::InvalidConfiguration) + )); + assert_eq!(used.get(), 0); + let pool = Rc::new(RefCell::new(None)); + let b = a + .allocate(512, CountingCharge::new(&used, 512)) + .unwrap() + .pooled(&pool); + let borrow = pool.borrow_mut(); + drop(b); + assert_eq!(used.get(), 0); + drop(borrow); + assert!(pool.borrow().is_none()); + } + + /// The live descriptor is direct, sparse, exclusively locked, and aligned. + #[test] + fn real_file_is_direct_aligned_and_sparse_without_reactor() { + let directory = Directory::new(); + let path = directory.0.join("caller-chosen.dat"); + let make = || Slab::<()>::new(path.clone(), 64 * 1024 * 1024, 32 * 1024 * 1024, 1024); + let conflicting = make(); + let slabs = make(); + assert!(!path.exists()); + let Some(alignment) = real_alignment(slabs.open_now()) else { + return; + }; + let stat = std::fs::metadata(&path).unwrap(); + assert_eq!(stat.len(), slabs.capacity_bytes()); + assert!(stat.blocks() * 512 < stat.len()); + let opened = slabs.opened.borrow(); + let fd = opened.as_ref().unwrap().file.as_raw_fd(); + // SAFETY: descriptor and buffers remain live throughout these synchronous calls. + assert_ne!( + unsafe { libc::fcntl(fd, libc::F_GETFL) } & libc::O_DIRECT, + 0 + ); + let extent = alignment.extent(0, 31).unwrap(); + let mut buffer = slabs.allocate(extent.length(), ()).unwrap(); + buffer.bytes_mut().unwrap()[..31].fill(42); + assert_eq!( + unsafe { libc::pwrite(fd, buffer.bytes().unwrap().as_ptr().cast(), buffer.len(), 0) }, + buffer.len() as isize + ); + buffer.bytes_mut().unwrap().fill(0); + assert_eq!( + unsafe { + libc::pread( + fd, + buffer.bytes_mut().unwrap().as_mut_ptr().cast(), + extent.length(), + 0, + ) + }, + extent.length() as isize + ); + assert_eq!(&buffer.bytes().unwrap()[..31], &[42; 31]); + assert!(buffer.bytes().unwrap()[31..].iter().all(|b| *b == 0)); + assert_eq!( + unsafe { libc::pwrite(fd, buffer.bytes().unwrap().as_ptr().cast(), 31, 1) }, + -1 + ); + drop(buffer); + assert_eq!(slabs.idle_bytes(), extent.length()); + assert_eq!(slabs.reclaim_idle(), extent.length()); + assert_eq!(slabs.idle_bytes(), 0); + assert_eq!(conflicting.open_now(), Err(Error::Unavailable)); + } + + /// Invalid startup dimensions never truncate a preexisting nonempty file. + #[test] + fn open_rejects_bad_layout_and_existing_size_without_truncating() { + use std::os::unix::fs::PermissionsExt; + let directory = Directory::new(); + let probe = Slab::<()>::new(directory.0.join("capability-probe"), 4096, 4096, 512); + if real_alignment(probe.open_now()).is_none() { + return; + } + let path = directory.0.join("data"); + std::fs::write(&path, [42; 7]).unwrap(); + std::fs::set_permissions(&path, std::fs::Permissions::from_mode(0o600)).unwrap(); + let slab = Slab::<()>::new(path.clone(), 4096, 4096, 512); + assert_eq!(slab.open_now(), Err(Error::InvalidConfiguration)); + assert_eq!(std::fs::read(&path).unwrap(), [42; 7]); + for (capacity, segment, record) in [ + (0, 4096, 512), + (4096, 0, 512), + (4096, 4096, 0), + (4097, 4096, 512), + (4096, 4096, 4097), + ] { + assert_eq!( + Slab::<()>::new(directory.0.join("invalid"), capacity, segment, record).open_now(), + Err(Error::InvalidConfiguration) + ); + } + assert_eq!(slab.alignment(), Err(Error::Unavailable)); + assert!(matches!(slab.allocate(512, ()), Err(Error::Unavailable))); + } + + /// Private validation rejects wrong ranges before any runtime admission. + #[test] + fn submission_checks_alignment_segment_and_capacity() { + let directory = Directory::new(); + let slab = Slab::<()>::new(directory.0.join("data"), 8192, 4096, 512); + let segments = Segments::new(4096); + let Some(a) = real_alignment(slab.open_configured(&segments)) else { + return; + }; + let (lease, extent) = segments.append(4096).unwrap(); + let buffer = slab.allocate(4096, ()).unwrap(); + assert!(slab.submission(extent, &buffer, &lease).is_ok()); + assert!(matches!( + slab.submission(Extent::new(4096, 4096).unwrap(), &buffer, &lease), + Err(Error::Corrupt) + )); + assert!(matches!( + slab.submission(Extent::new(1, 4096).unwrap(), &buffer, &lease), + Err(Error::InvalidConfiguration) + )); + let outside_table = Segments::new(4096); + outside_table.configure(12288, 3, a).unwrap(); + drop(outside_table.append(4096).unwrap()); + drop(outside_table.append(4096).unwrap()); + let (outside, extent) = outside_table.append(4096).unwrap(); + assert!(matches!( + slab.submission(extent, &buffer, &outside), + Err(Error::Stale) + )); + assert_eq!( + outside_table.validate(SegmentId(2), Generation(1), &extent), + Ok(()) + ); + } + + /// Waiters sleep, replace wakers, unregister on cancellation, and register again. + #[test] + fn fence_waiters_sleep_update_wakers_and_unregister_on_cancellation() { + use std::sync::{ + Arc, + atomic::{AtomicUsize, Ordering}, + }; + use std::task::Wake; + /// Counts wake notifications without scheduling actual work. + #[derive(Default)] + struct WakeCount(AtomicUsize); + impl Wake for WakeCount { + /// Count an owned notification. + fn wake(self: Arc) { + self.0.fetch_add(1, Ordering::Relaxed); + } + + /// Count a borrowed notification. + fn wake_by_ref(self: &Arc) { + self.0.fetch_add(1, Ordering::Relaxed); + } + } + let slab = Slab::<()>::new(PathBuf::new(), 4096, 4096, 512); + let first_write = slab.writes.acquire().unwrap(); + let last_write = slab.writes.acquire().unwrap(); + let old = Arc::new(WakeCount::default()); + let current = Arc::new(WakeCount::default()); + let other = Arc::new(WakeCount::default()); + let cancelled = Arc::new(WakeCount::default()); + let poll = |op: &mut Operation<'_, (), Error>, wakes: &Arc| { + op.as_mut() + .poll(&mut Context::from_waker(&Waker::from(wakes.clone()))) + }; + let mut one = slab.fence_writes(); + let mut two = slab.fence_writes(); + let mut abandoned = slab.fence_writes(); + assert!(poll(&mut one, &old).is_pending()); + assert!(poll(&mut one, ¤t).is_pending()); + assert!(poll(&mut two, &other).is_pending()); + assert!(poll(&mut abandoned, &cancelled).is_pending()); + assert_eq!(slab.writes.waiters.borrow().len(), 3); + drop(abandoned); + assert_eq!(slab.writes.waiters.borrow().len(), 2); + drop(first_write); + for count in [&old, ¤t, &other, &cancelled] { + assert_eq!(count.0.load(Ordering::Relaxed), 0); + } + drop(last_write); + assert_eq!(old.0.load(Ordering::Relaxed), 0); + assert_eq!(cancelled.0.load(Ordering::Relaxed), 0); + assert_eq!(current.0.load(Ordering::Relaxed), 1); + assert_eq!(other.0.load(Ordering::Relaxed), 1); + assert!(slab.writes.waiters.borrow().is_empty()); + let next = slab.writes.acquire().unwrap(); + assert!(poll(&mut one, ¤t).is_pending()); + assert_eq!(slab.writes.waiters.borrow().len(), 1); + drop(next); + assert_eq!(current.0.load(Ordering::Relaxed), 2); + assert_eq!(poll(&mut one, ¤t), Poll::Ready(Ok(()))); + assert_eq!(poll(&mut two, &other), Poll::Ready(Ok(()))); + assert_eq!( + poll(&mut slab.fence_writes(), ¤t), + Poll::Ready(Ok(())) + ); + } + + /// Overflow cannot create a decrement guard or alter the saturated count. + #[test] + fn write_guard_overflow_is_atomic() { + let state = Rc::new(WriteState::default()); + state.count.set(usize::MAX); + assert!(matches!(state.acquire(), Err(Error::Busy))); + assert_eq!(state.count.get(), usize::MAX); + state.count.set(0); + let guard = state.acquire().unwrap(); + assert_eq!(state.count.get(), 1); + drop(guard); + assert_eq!(state.count.get(), 0); + } + + /// Validation consumes rejected resources and successful preparation retains its lease. + #[test] + fn prepared_submission_owns_resources_and_failed_binding_is_recoverable() { + let directory = Directory::new(); + let slab = Slab::::new(directory.0.join("prepared"), 8192, 4096, 512); + let Some(alignment) = real_alignment(slab.open_now()) else { + return; + }; + let table = Segments::new(4096); + table.configure(8192, 2, alignment).unwrap(); + let (lease, extent) = table.append(4096).unwrap(); + let used = Rc::new(Cell::new(0)); + let buffer = slab + .allocate(4096, CountingCharge::new(&used, 4096)) + .unwrap(); + assert!(matches!( + slab.prepare(extent, buffer, lease), + Err(Error::Unavailable) + )); + assert_eq!(slab.reclaim_idle(), 4096); + assert_eq!(used.get(), 0); + assert_eq!(slab.writes_in_flight(), 0); + slab.configure_segments(&table).unwrap(); + let lease = table.lease(SegmentId(0), Generation(1)).unwrap(); + let buffer = slab + .allocate(4096, CountingCharge::new(&used, 4096)) + .unwrap(); + let submission = slab.prepare(extent, buffer, lease).unwrap(); + table.begin_evict(SegmentId(0)).unwrap(); + assert_eq!(table.recycle(SegmentId(0)), Err(Error::Busy)); + drop(submission); + table.recycle(SegmentId(0)).unwrap(); + assert_eq!(used.get(), 4096); + slab.reclaim_idle(); + assert_eq!(used.get(), 0); + } + + /// System classification preserves errno and operation without hiding failures. + #[test] + fn system_errors_keep_operation_and_errno() { + for errno in [ + libc::EACCES, + libc::EPERM, + libc::ENOSPC, + libc::EIO, + libc::ELOOP, + ] { + assert_eq!( + direct_error("open", std::io::Error::from_raw_os_error(errno)), + Error::SystemIo { + operation: "open", + errno: Some(errno) + } + ); + assert_eq!( + lock_error(std::io::Error::from_raw_os_error(errno)), + Error::SystemIo { + operation: "flock", + errno: Some(errno) + } + ); + } + assert_eq!( + lock_error(std::io::Error::from_raw_os_error(libc::EWOULDBLOCK)), + Error::Unavailable + ); + for errno in [libc::EINVAL, libc::EOPNOTSUPP, libc::ENOSYS] { + assert_eq!( + direct_error("fcntl-direct", std::io::Error::from_raw_os_error(errno)), + Error::Unsupported + ); + } + assert_eq!( + system_error("test", std::io::Error::other("synthetic")), + Error::SystemIo { + operation: "test", + errno: None + } + ); + assert_eq!( + probe_fd(-1), + Err(Error::SystemIo { + operation: "statx", + errno: Some(libc::EBADF) + }) + ); + // SAFETY: zero is a valid initialized statx output representation. + let mut stat: libc::statx = unsafe { std::mem::zeroed() }; + assert_eq!(alignment_from_stat(&stat), Err(Error::Unsupported)); + stat.stx_mask = libc::STATX_DIOALIGN; + stat.stx_dio_mem_align = 512; + stat.stx_dio_offset_align = 512; + assert_eq!(alignment_from_stat(&stat), Alignment::new(512, 512, 512)); + } + + /// File validation requires regular type, effective owner, private mode, and one link. + #[test] + fn private_file_validation_checks_type_owner_permissions_and_links() { + assert_eq!(validate_file(libc::S_IFREG | 0o600, 17, 17, 1), Ok(())); + for (mode, owner, links) in [ + (libc::S_IFREG | 0o644, 17, 1), + (libc::S_IFREG | 0o600, 18, 1), + (libc::S_IFREG | 0o600, 17, 2), + (libc::S_IFREG | 0o4600, 17, 1), + (libc::S_IFDIR | 0o600, 17, 1), + (libc::S_IFIFO | 0o600, 17, 1), + ] { + assert_eq!( + validate_file(mode, owner, 17, links), + Err(Error::InvalidConfiguration) + ); + } + } + + /// Descriptor traversal rejects symlinks, FIFOs, and parent-directory escapes. + #[test] + fn descriptor_relative_open_rejects_symlinks_and_nonregular_files() { + use std::os::unix::fs::{PermissionsExt, symlink}; + let directory = Directory::new(); + let make = |path| Slab::<()>::new(path, 8192, 4096, 512); + let target = directory.0.join("target"); + std::fs::create_dir(&target).unwrap(); + let alias = directory.0.join("alias"); + symlink(&target, &alias).unwrap(); + assert!(matches!( + make(alias.join("data")).open_now(), + Err(Error::SystemIo { + operation: "open-parent", + errno: Some(libc::ENOTDIR | libc::ELOOP) + }) + )); + assert!(!target.join("data").exists()); + let data = target.join("data"); + std::fs::write(&data, [42; 7]).unwrap(); + let link = directory.0.join("link"); + symlink(&data, &link).unwrap(); + assert_eq!( + make(link).open_now(), + Err(Error::SystemIo { + operation: "open", + errno: Some(libc::ELOOP) + }) + ); + assert_eq!(std::fs::read(&data).unwrap(), [42; 7]); + std::fs::set_permissions(&data, std::fs::Permissions::from_mode(0o666)).unwrap(); + assert_eq!( + make(data.clone()).open_now(), + Err(Error::InvalidConfiguration) + ); + assert!(make(target.clone()).open_now().is_err()); + let fifo = directory.0.join("fifo"); + let fifo_c = CString::new(fifo.as_os_str().as_bytes()).unwrap(); + // SAFETY: valid C pathname and POSIX permission mode. + assert_eq!(unsafe { libc::mkfifo(fifo_c.as_ptr(), 0o600) }, 0); + assert_eq!(make(fifo).open_now(), Err(Error::InvalidConfiguration)); + assert_eq!( + make(target.join("..").join("escape")).open_now(), + Err(Error::InvalidConfiguration) + ); + let new_path = directory.0.join("new/nested/data"); + let opened = open_private_file(&new_path).unwrap(); + assert_eq!(opened.metadata().unwrap().mode() & 0o777, 0o600); + assert_eq!( + std::fs::metadata(new_path.parent().unwrap()) + .unwrap() + .mode() + & 0o777, + 0o700 + ); + } + + /// Binding is permanent and validates dimensions as well as captured used extents. + #[test] + fn binding_validates_geometry_identity_and_used_extents() { + let directory = Directory::new(); + let slab = Slab::<()>::new(directory.0.join("data"), 8192, 4096, 512); + let table = Segments::new(4096); + let Some(a) = real_alignment(slab.open_configured(&table)) else { + return; + }; + assert_eq!(table.geometry(), Some(slab.geometry().unwrap())); + slab.configure_segments(&table).unwrap(); + let other = Segments::new(4096); + other.configure(8192, 2, a).unwrap(); + assert_eq!( + slab.configure_segments(&other), + Err(Error::InvalidConfiguration) + ); + let length = a.extent(0, 512).unwrap().length(); + let (foreign, extent) = other.append(length).unwrap(); + let buffer = slab.allocate(length, ()).unwrap(); + assert!(matches!( + slab.submission(extent, &buffer, &foreign), + Err(Error::Stale) + )); + let (lease, extent) = table.append(length).unwrap(); + assert!(slab.submission(extent, &buffer, &lease).is_ok()); + assert!(matches!( + slab.submission(Extent::new(length as u64, length).unwrap(), &buffer, &lease), + Err(Error::Corrupt) + )); + let wrong = Slab::<()>::new(directory.0.join("wrong"), 8192, 4096, 512); + let _ = wrong.open_now().unwrap(); + assert_eq!( + wrong.configure_segments(&Segments::new(8192)), + Err(Error::InvalidConfiguration) + ); + let wrong_capacity = Segments::new(4096); + wrong_capacity.configure(4096, 1, a).unwrap(); + assert_eq!( + wrong.configure_segments(&wrong_capacity), + Err(Error::InvalidConfiguration) + ); + let wrong_alignment = Segments::new(4096); + let incompatible = Alignment::new(a.memory() * 2, a.offset(), a.length()).unwrap(); + wrong_alignment.configure(8192, 2, incompatible).unwrap(); + assert_eq!( + wrong.configure_segments(&wrong_alignment), + Err(Error::InvalidConfiguration) + ); + let partial = Segments::new(4096); + partial.configure(8192, 1, a).unwrap(); + wrong.configure_segments(&partial).unwrap(); + assert_eq!(partial.count(), 1); + if length < 4096 { + drop(table.append(4096 - length).unwrap()); + } + table.begin_evict(SegmentId(0)).unwrap(); + assert!(slab.submission(extent, &buffer, &lease).is_ok()); + assert_eq!(table.recycle(SegmentId(0)), Err(Error::Busy)); + drop(lease); + table.recycle(SegmentId(0)).unwrap(); + } + + /// Size mismatch releases idle accounting even when new allocation is invalid. + #[test] + fn idle_size_mismatch_releases_retained_charge_even_on_invalid_replacement() { + let directory = Directory::new(); + let slab = Slab::::new(directory.0.join("data"), 8192, 4096, 512); + let Some(a) = real_alignment(slab.open_now()) else { + return; + }; + let length = a.extent(0, 512).unwrap().length(); + let used = Rc::new(Cell::new(0)); + drop( + slab.allocate(length, CountingCharge::new(&used, length)) + .unwrap(), + ); + assert_eq!(slab.idle_bytes(), length); + assert_eq!(used.get(), length); + assert!(matches!( + slab.allocate(0, CountingCharge::new(&used, 0)), + Err(Error::InvalidConfiguration) + )); + assert_eq!(used.get(), 0); + assert_eq!(slab.idle_bytes(), 0); + } +} diff --git a/cmd/racer-dataplane/alloc/tests/workflows.rs b/cmd/racer-dataplane/alloc/tests/workflows.rs new file mode 100644 index 000000000..6c82c603a --- /dev/null +++ b/cmd/racer-dataplane/alloc/tests/workflows.rs @@ -0,0 +1,1001 @@ +//! Public allocator workflows across restart, cancellation, accounting, and real I/O. + +use page_alloc::{Alignment, Error, Generation, SegmentId, Segments, Slab}; +#[cfg(feature = "simulation")] +use page_alloc::{Charge, SegmentState}; +#[cfg(feature = "simulation")] +use std::{cell::Cell, path::Path, rc::Rc}; +use std::{ + fs::File, + future::Future, + os::fd::FromRawFd, + path::PathBuf, + pin::Pin, + task::{Context, Poll, Waker}, +}; +#[cfg(feature = "simulation")] +use uring_runtime::reactor::simulation::{Fault, Simulation}; +use uring_runtime::{Operation, Scope, reactor::Reactor}; + +/// Distinguishes allocator validation errors from runtime completion failures. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +enum TestError { + Alloc(Error), + Runtime(uring_runtime::Error), +} +impl From for TestError { + /// Preserve the allocator category in workflow assertions. + fn from(error: Error) -> Self { + Self::Alloc(error) + } +} +impl From for TestError { + /// Preserve runtime errors without enriching them as synchronous allocator errors. + fn from(error: uring_runtime::Error) -> Self { + Self::Runtime(error) + } +} + +/// An always-live caller scope for deterministic ownership tests. +#[derive(Clone)] +struct TestScope; +impl Scope for TestScope { + type Error = TestError; + + /// Keep the scope live; cancellation is injected by dropping waiting futures. + fn check(&self) -> Result<(), TestError> { + Ok(()) + } +} + +/// Poll once without assuming that a completion wakes a host executor. +fn poll(future: &mut Pin + '_>>) -> Poll { + future + .as_mut() + .poll(&mut Context::from_waker(Waker::noop())) +} + +/// Drive a simulated operation with a fixed, finite reactor-turn budget. +#[cfg(feature = "simulation")] +fn drive( + reactor: &Reactor, + mut operation: Pin + '_>>, +) -> T { + for _ in 0..100 { + if let Poll::Ready(value) = poll(&mut operation) { + return value; + } + reactor.poll_budgeted(64).unwrap(); + } + panic!("simulation did not complete in 100 turns"); +} + +/// Caller-owned accounting for live buffers and the single idle slot. +#[cfg(feature = "simulation")] +struct CountingCharge { + used: Rc>, + + bytes: usize, +} +#[cfg(feature = "simulation")] +impl CountingCharge { + /// Record admission before passing the guard into allocator ownership. + fn new(used: &Rc>, bytes: usize) -> Self { + used.set(used.get() + bytes); + Self { + used: used.clone(), + bytes, + } + } +} +#[cfg(feature = "simulation")] +impl Charge for CountingCharge { + /// Cover only the bytes actually admitted by this guard. + fn covers(&self, bytes: usize) -> bool { + self.bytes >= bytes + } +} +#[cfg(feature = "simulation")] +impl Drop for CountingCharge { + /// Return admission when the final owner releases this guard. + fn drop(&mut self) { + self.used.set(self.used.get() - self.bytes); + } +} + +/// Restored records remain readable and the sealed recovery tail is not overwritten. +#[cfg(feature = "simulation")] +#[test] +fn reopen_snapshot_reads_old_record_and_appends_without_overwriting_it() { + let simulation = Simulation::new(); + let _environment = simulation.enter(); + let reactor = Reactor::::new(16, ()); + let make_slab = || Slab::<()>::new("/alloc-workflows/restart".into(), 16384, 8192, 1024); + let slab = make_slab(); + let segments = Segments::new(slab.segment_bytes()); + let alignment = slab.open_configured(&segments).unwrap(); + let payload = b"a caller-owned record, not a cache entry"; + let padded = alignment.extent(0, payload.len()).unwrap().length(); + let (lease, extent) = segments.append(padded).unwrap(); + let (id, generation) = (lease.id(), lease.generation()); + let mut buffer = slab.allocate(padded, ()).unwrap(); + buffer.as_mut_slice()[..payload.len()].copy_from_slice(payload); + drop( + drive( + &reactor, + slab.write(&reactor, extent, buffer, lease, &TestScope), + ) + .unwrap(), + ); + drive(&reactor, slab.fence_writes()).unwrap(); + let frozen = segments.freeze().unwrap(); + let snapshot = segments.snapshot(); + assert_eq!(snapshot[0].state, SegmentState::Open); + assert_eq!(snapshot[0].used_bytes, padded as u64); + assert_eq!(snapshot[1].state, SegmentState::Free); + assert_eq!(segments.restore(snapshot.clone()), Err(Error::Busy)); + drop(frozen); + drop(slab); + drop(segments); + + let slab = make_slab(); + let segments = Segments::new(slab.segment_bytes()); + let alignment = slab.open_configured(&segments).unwrap(); + // Restore validates the entire image before changing any slot or free list. + for (invalid, label) in [ + "duplicate segment ID", + "zero generation", + "oversized sealed segment", + "misaligned sealed segment", + "nonempty free segment", + "missing segment", + ] + .into_iter() + .enumerate() + { + let mut damaged = snapshot.clone(); + match invalid { + 0 => damaged[1].id = SegmentId(0), + 1 => damaged[1].generation = Generation(0), + 2 => { + damaged[1].state = SegmentState::Sealed; + damaged[1].used_bytes = slab.segment_bytes() + alignment.length() as u64; + } + 3 => { + damaged[1].state = SegmentState::Sealed; + damaged[1].used_bytes = 513; + } + 4 => damaged[1].used_bytes = alignment.length() as u64, + _ => { + damaged.pop(); + } + } + assert_eq!(segments.restore(damaged), Err(Error::Corrupt), "{label}"); + assert_eq!(segments.free_count(), 2, "{label}"); + assert!( + segments + .snapshot() + .iter() + .all(|s| s.state == SegmentState::Free && s.used_bytes == 0), + "{label}" + ); + } + segments.validate_restore(&snapshot).unwrap(); + segments.restore(snapshot).unwrap(); + assert_eq!(segments.state(id), Ok(SegmentState::Sealed)); + assert_eq!(segments.free_count(), 1); + segments.validate(id, generation, &extent).unwrap(); + let buffer = drive( + &reactor, + slab.read( + &reactor, + extent, + slab.allocate(padded, ()).unwrap(), + segments.lease(id, generation).unwrap(), + &TestScope, + ), + ) + .unwrap(); + assert_eq!(&buffer.as_slice()[..payload.len()], payload); + assert!(buffer.as_slice()[payload.len()..].iter().all(|b| *b == 0)); + drop(buffer); + + let (next, next_extent) = segments.append(padded).unwrap(); + assert_eq!(next.id(), SegmentId(1)); + assert_eq!(next_extent.offset(), slab.segment_bytes()); + assert_eq!(segments.free_count(), 0); + let mut buffer = slab.allocate(padded, ()).unwrap(); + buffer.as_mut_slice().fill(99); + drop( + drive( + &reactor, + slab.write(&reactor, next_extent, buffer, next, &TestScope), + ) + .unwrap(), + ); + let buffer = drive( + &reactor, + slab.read( + &reactor, + extent, + slab.allocate(padded, ()).unwrap(), + segments.lease(id, generation).unwrap(), + &TestScope, + ), + ) + .unwrap(); + assert_eq!(&buffer.as_slice()[..payload.len()], payload); + assert!(buffer.as_slice()[payload.len()..].iter().all(|b| *b == 0)); + assert_eq!(slab.writes_in_flight(), 0); + assert_eq!(reactor.in_flight(), 0); +} + +/// Public pooling enforces admission, zeroization, exact-size reuse, and bounded retention. +#[cfg(feature = "simulation")] +#[test] +fn buffer_pool_keeps_only_zeroed_accounted_storage_and_rejects_undercharging() { + let simulation = Simulation::new(); + let _environment = simulation.enter(); + let slab = Slab::new("/alloc-workflows/pool".into(), 8192, 4096, 1024); + let segments = Segments::new(slab.segment_bytes()); + let alignment = slab.open_configured(&segments).unwrap(); + let size = alignment.extent(0, 31).unwrap().length(); + let live = Rc::new(Cell::new(0)); + let charge = |bytes| CountingCharge::new(&live, bytes); + let mut buffer = slab.allocate(size, charge(size)).unwrap(); + assert_eq!(buffer.len(), size); + assert!(!buffer.is_empty()); + let pointer = buffer.bytes().unwrap().as_ptr(); + assert_eq!(pointer as usize % alignment.memory(), 0); + assert!(buffer.bytes().unwrap().iter().all(|b| *b == 0)); + buffer.bytes_mut().unwrap().fill(42); + let retained = Rc::new(charge(17)); + let weak = Rc::downgrade(&retained); + buffer.retain(retained); + assert_eq!(live.get(), size + 17); + drop(buffer); + assert!(weak.upgrade().is_none()); + assert_eq!(live.get(), size); + assert_eq!(slab.idle_bytes(), size); + + // Denied admission must release its charge without consuming the idle buffer. + assert!(matches!( + slab.allocate(size, charge(size - 1)), + Err(Error::InvalidConfiguration) + )); + assert_eq!(live.get(), size); + assert_eq!(slab.idle_bytes(), size); + let reused = slab.allocate(size, charge(size)).unwrap(); + assert_eq!(reused.bytes().unwrap().as_ptr(), pointer); + assert!(reused.bytes().unwrap().iter().all(|b| *b == 0)); + assert_eq!(live.get(), size); + assert_eq!(slab.idle_bytes(), 0); + assert_eq!(slab.reclaim_idle(), 0); + + // Two simultaneous users cannot both return storage to the single idle slot. + let overflow = slab.allocate(size, charge(size)).unwrap(); + assert_eq!(live.get(), size * 2); + drop(reused); + drop(overflow); + assert_eq!(slab.idle_bytes(), size); + assert_eq!(live.get(), size); + let larger = slab.allocate(size * 2, charge(size * 2)).unwrap(); + assert_eq!(larger.len(), size * 2); + assert_eq!(live.get(), size * 2); + drop(larger); + assert_eq!(slab.reclaim_idle(), size * 2); + assert_eq!(slab.reclaim_idle(), 0); + assert_eq!(live.get(), 0); + + let outstanding = slab.allocate(size, charge(size)).unwrap(); + drop(slab.allocate(size, charge(size)).unwrap()); + assert_eq!(live.get(), size * 2); + assert_eq!(slab.idle_bytes(), size); + drop(slab); + assert_eq!(live.get(), size); + assert!(outstanding.as_slice().iter().all(|b| *b == 0)); + drop(outstanding); + assert_eq!(live.get(), 0); +} + +/// Bound submissions reject foreign identity and bytes beyond the captured prefix. +#[cfg(feature = "simulation")] +#[test] +fn bound_io_rejects_foreign_tables_and_ranges_appended_after_lease_acquisition() { + let simulation = Simulation::new(); + let _environment = simulation.enter(); + let reactor = Reactor::::new(16, ()); + let slab = Slab::<()>::new("/alloc-workflows/bound".into(), 16384, 8192, 512); + let segments = Segments::new(slab.segment_bytes()); + let alignment = slab.open_configured(&segments).unwrap(); + let other = Segments::from_geometry(segments.geometry().unwrap()).unwrap(); + assert_eq!( + slab.configure_segments(&other), + Err(Error::InvalidConfiguration) + ); + let size = alignment.extent(0, 31).unwrap().length(); + + for write in [false, true] { + let (foreign, extent) = other.append(size).unwrap(); + let buffer = slab.allocate(size, ()).unwrap(); + let operation = if write { + slab.write(&reactor, extent, buffer, foreign, &TestScope) + } else { + slab.read(&reactor, extent, buffer, foreign, &TestScope) + }; + assert!(matches!( + drive(&reactor, operation), + Err(TestError::Alloc(Error::Stale)) + )); + assert_eq!(reactor.in_flight(), 0); + assert_eq!(slab.writes_in_flight(), 0); + } + + let (early, first) = segments.append(size).unwrap(); + let early_read = segments.lease(early.id(), early.generation()).unwrap(); + let (later, second) = segments.append(size).unwrap(); + drop(later); + // Both extents are now used, but the earlier leases only authorize the first. + for (lease, write) in [(early, true), (early_read, false)] { + let buffer = slab.allocate(size, ()).unwrap(); + let operation = if write { + slab.write(&reactor, second, buffer, lease, &TestScope) + } else { + slab.read(&reactor, second, buffer, lease, &TestScope) + }; + assert!(matches!( + drive(&reactor, operation), + Err(TestError::Alloc(Error::Corrupt)) + )); + assert_eq!(reactor.in_flight(), 0); + assert_eq!(slab.writes_in_flight(), 0); + } + let mut buffer = slab.allocate(size, ()).unwrap(); + buffer.as_mut_slice().fill(73); + drop( + drive( + &reactor, + slab.write( + &reactor, + second, + buffer, + segments.lease(SegmentId(0), Generation(1)).unwrap(), + &TestScope, + ), + ) + .unwrap(), + ); + let untouched = drive( + &reactor, + slab.read( + &reactor, + first, + slab.allocate(size, ()).unwrap(), + segments.lease(SegmentId(0), Generation(1)).unwrap(), + &TestScope, + ), + ) + .unwrap(); + assert!(untouched.as_slice().iter().all(|byte| *byte == 0)); + drop(untouched); + let written = drive( + &reactor, + slab.read( + &reactor, + second, + slab.allocate(size, ()).unwrap(), + segments.lease(SegmentId(0), Generation(1)).unwrap(), + &TestScope, + ), + ) + .unwrap(); + assert!(written.as_slice().iter().all(|byte| *byte == 73)); + assert_eq!(reactor.in_flight(), 0); + assert_eq!(slab.writes_in_flight(), 0); +} + +/// Thawing and write completion are independent prerequisites for recovery. +#[cfg(feature = "simulation")] +#[test] +fn restore_waits_for_freeze_guard_and_abandoned_write_completion_independently() { + for cancel_first in [false, true] { + let simulation = Simulation::new(); + simulation.set_cancel_first(cancel_first); + let _environment = simulation.enter(); + let reactor = Reactor::::new(16, ()); + let slab = Slab::<()>::new("/alloc-workflows/freeze".into(), 8192, 4096, 512); + let segments = Segments::new(slab.segment_bytes()); + let alignment = slab.open_configured(&segments).unwrap(); + let size = alignment.extent(0, 31).unwrap().length(); + let (lease, extent) = segments.append(size).unwrap(); + simulation + .inject("write", Fault::HoldCompletion(8)) + .unwrap(); + let mut write = slab.write( + &reactor, + extent, + slab.allocate(size, ()).unwrap(), + lease, + &TestScope, + ); + assert!(poll(&mut write).is_pending()); + reactor.poll_budgeted(1).unwrap(); + drop(write); + assert_eq!(slab.writes_in_flight(), 1); + + let frozen = segments.freeze().unwrap(); + let snapshot = segments.snapshot(); + assert!(matches!(segments.append(size), Err(Error::Busy))); + assert_eq!(segments.restore(snapshot.clone()), Err(Error::Busy)); + drop(frozen); + // Thawing is not a kernel completion fence; the abandoned write owns a lease. + assert_eq!(segments.restore(snapshot.clone()), Err(Error::Busy)); + assert_eq!(segments.snapshot(), snapshot); + let frozen = segments.freeze().unwrap(); + drive(&reactor, slab.fence_writes()).unwrap(); + assert_eq!(slab.writes_in_flight(), 0); + assert_eq!(reactor.in_flight(), 0); + // Conversely, completion does not release a caller's freeze guard. + assert_eq!(segments.restore(snapshot.clone()), Err(Error::Busy)); + drop(frozen); + segments.restore(snapshot).unwrap(); + assert_eq!(segments.state(SegmentId(0)), Ok(SegmentState::Sealed)); + assert_eq!(segments.append(size).unwrap().0.id(), SegmentId(1)); + } +} + +/// Build and bind a deterministic simulated slab with two segments. +#[cfg(feature = "simulation")] +fn setup() -> (Slab, Segments) { + let slab = Slab::new(PathBuf::from("/virtual/slab.dat"), 8192, 4096, 512); + let segments = Segments::new(4096); + let _ = slab.open_configured(&segments).unwrap(); + (slab, segments) +} + +/// Public reads and writes preserve payloads and reject short or failed completions. +#[cfg(feature = "simulation")] +#[test] +fn roundtrip_short_io_and_runtime_errors() { + let sim = Simulation::new(); + let _environment = sim.enter(); + let (slab, segments) = setup::<()>(); + let reactor = Reactor::::new(16, ()); + let (lease, extent) = segments.append(4096).unwrap(); + let mut buffer = slab.allocate(extent.length(), ()).unwrap(); + buffer.as_mut_slice().fill(42); + let buffer = drive( + &reactor, + slab.write(&reactor, extent, buffer, lease, &TestScope), + ) + .unwrap(); + assert_eq!(slab.writes_in_flight(), 0); + drop(buffer); + let read = || { + slab.read( + &reactor, + extent, + slab.allocate(extent.length(), ()).unwrap(), + segments.lease(SegmentId(0), Generation(1)).unwrap(), + &TestScope, + ) + }; + let buffer = drive(&reactor, read()).unwrap(); + assert!(buffer.as_slice().iter().all(|b| *b == 42)); + drop(buffer); + sim.inject("read", Fault::Short(512)).unwrap(); + assert!(matches!( + drive(&reactor, read()), + Err(TestError::Alloc(Error::Io)) + )); + sim.inject("write", Fault::Short(512)).unwrap(); + let op = slab.write( + &reactor, + extent, + slab.allocate(extent.length(), ()).unwrap(), + segments.lease(SegmentId(0), Generation(1)).unwrap(), + &TestScope, + ); + assert!(matches!( + drive(&reactor, op), + Err(TestError::Alloc(Error::Io)) + )); + assert_eq!(slab.writes_in_flight(), 0); + sim.inject("read", Fault::Errno(libc::EIO)).unwrap(); + assert!(matches!( + drive(&reactor, read()), + Err(TestError::Runtime(uring_runtime::Error::Os(libc::EIO))) + )); + assert_eq!(reactor.in_flight(), 0); + segments.begin_evict(SegmentId(0)).unwrap(); + segments.recycle(SegmentId(0)).unwrap(); +} + +/// Abandonment cannot release kernel-visible memory, charges, leases, or write count. +#[cfg(feature = "simulation")] +#[test] +fn abandoned_write_retains_lease_charges_and_counter_until_completion() { + for cancel_first in [false, true] { + let sim = Simulation::new(); + sim.set_cancel_first(cancel_first); + let _environment = sim.enter(); + let (slab, segments) = setup::(); + let reactor = Reactor::::new(16, ()); + let used = Rc::new(Cell::new(0)); + let (lease, extent) = segments.append(4096).unwrap(); + let mut buffer = slab + .allocate(4096, CountingCharge::new(&used, 4096)) + .unwrap(); + buffer.as_mut_slice().fill(42); + let extra = Rc::new(CountingCharge::new(&used, 512)); + let weak = Rc::downgrade(&extra); + buffer.retain(extra); + sim.inject("write", Fault::HoldCompletion(8)).unwrap(); + let mut op = slab.write(&reactor, extent, buffer, lease, &TestScope); + assert!(poll(&mut op).is_pending()); + assert_eq!(slab.writes_in_flight(), 1); + reactor.poll_budgeted(1).unwrap(); + drop(op); + assert_eq!(slab.writes_in_flight(), 1); + assert_eq!(slab.idle_bytes(), 0); + assert_eq!(slab.reclaim_idle(), 0); + assert!(weak.upgrade().is_some()); + assert_eq!(used.get(), 4608); + segments.begin_evict(SegmentId(0)).unwrap(); + assert_eq!(segments.recycle(SegmentId(0)), Err(Error::Busy)); + let mut fence = slab.fence_writes(); + assert!(poll(&mut fence).is_pending()); + drive(&reactor, fence).unwrap(); + assert_eq!(slab.writes_in_flight(), 0); + assert_eq!(reactor.in_flight(), 0); + assert!(weak.upgrade().is_none()); + assert_eq!(used.get(), 4096); + assert_eq!(slab.idle_bytes(), 4096); + let reused = slab + .allocate(4096, CountingCharge::new(&used, 4096)) + .unwrap(); + assert!(reused.as_slice().iter().all(|b| *b == 0)); + drop(reused); + assert_eq!(slab.reclaim_idle(), 4096); + assert_eq!(used.get(), 0); + segments.recycle(SegmentId(0)).unwrap(); + } +} + +/// An unpolled write owns no runtime work, but an abandoned accepted read still does. +#[cfg(feature = "simulation")] +#[test] +fn abandoned_read_and_unpolled_write_release_at_the_correct_fence() { + let sim = Simulation::new(); + let _environment = sim.enter(); + let (slab, segments) = setup::<()>(); + let reactor = Reactor::::new(16, ()); + let (lease, extent) = segments.append(4096).unwrap(); + let op = slab.write( + &reactor, + extent, + slab.allocate(4096, ()).unwrap(), + lease, + &TestScope, + ); + drop(op); + assert_eq!(slab.writes_in_flight(), 0); + assert_eq!(reactor.in_flight(), 0); + sim.inject("read", Fault::HoldCompletion(8)).unwrap(); + let mut op = slab.read( + &reactor, + extent, + slab.allocate(4096, ()).unwrap(), + segments.lease(SegmentId(0), Generation(1)).unwrap(), + &TestScope, + ); + assert!(poll(&mut op).is_pending()); + reactor.poll_budgeted(1).unwrap(); + drop(op); + segments.begin_evict(SegmentId(0)).unwrap(); + assert_eq!(segments.recycle(SegmentId(0)), Err(Error::Busy)); + drive(&reactor, reactor.drain()).unwrap(); + segments.recycle(SegmentId(0)).unwrap(); +} + +/// Simulation preserves file locks, size validation, and synchronous error categories. +#[cfg(feature = "simulation")] +#[test] +fn simulation_open_lock_size_and_faults() { + let sim = Simulation::new(); + let _environment = sim.enter(); + let (slab, _) = setup::<()>(); + let conflicting = Slab::<()>::new(PathBuf::from("/virtual/slab.dat"), 8192, 4096, 512); + assert_eq!(conflicting.open_now(), Err(Error::Unavailable)); + assert_eq!(slab.open_now(), slab.alignment()); + drop(slab); + let _ = conflicting.open_now().unwrap(); + drop(conflicting); + let wrong_size = Slab::<()>::new(PathBuf::from("/virtual/slab.dat"), 16384, 4096, 512); + assert_eq!(wrong_size.open_now(), Err(Error::InvalidConfiguration)); + sim.inject("open", Fault::Errno(libc::EOPNOTSUPP)).unwrap(); + assert_eq!(wrong_size.open_now(), Err(Error::Unsupported)); + sim.inject("open", Fault::Errno(libc::ENOSPC)).unwrap(); + assert_eq!( + wrong_size.open_now(), + Err(Error::SystemIo { + operation: "open", + errno: Some(libc::ENOSPC) + }) + ); +} + +/// Fault hooks cannot swap descriptors while a write completion owns the old file. +#[cfg(feature = "simulation")] +#[test] +fn replacement_hooks_are_fallible_and_refuse_live_writes() { + let sim = Simulation::new(); + let _environment = sim.enter(); + let replacement = || { + sim.open( + None, + Path::new("/replacement"), + libc::O_CREAT | libc::O_RDWR, + ) + .unwrap() + }; + let unopened = Slab::<()>::new(PathBuf::from("/unopened"), 8192, 4096, 512); + assert_eq!( + unopened.replace_descriptor_for_test(replacement()), + Err(Error::Unavailable) + ); + let (slab, segments) = setup::<()>(); + let reactor = Reactor::::new(16, ()); + let (lease, extent) = segments.append(4096).unwrap(); + sim.inject("write", Fault::HoldCompletion(8)).unwrap(); + let mut write = slab.write( + &reactor, + extent, + slab.allocate(4096, ()).unwrap(), + lease, + &TestScope, + ); + assert!(poll(&mut write).is_pending()); + assert_eq!( + slab.replace_descriptor_for_test(replacement()), + Err(Error::Busy) + ); + drop(write); + drive(&reactor, slab.fence_writes()).unwrap(); + slab.replace_descriptor_for_test(replacement()).unwrap(); +} + +/// Owns a unique project-local directory for real kernel workflows. +struct Directory(PathBuf); +impl Directory { + /// Create an isolated test directory without using the host temporary directory. + fn new() -> Self { + static NEXT: std::sync::atomic::AtomicUsize = std::sync::atomic::AtomicUsize::new(0); + let id = NEXT.fetch_add(1, std::sync::atomic::Ordering::Relaxed); + let path = PathBuf::from(env!("CARGO_MANIFEST_DIR")) + .join("target") + .join(format!("page-alloc-workflow-{}-{id}", std::process::id())); + std::fs::create_dir_all(&path).unwrap(); + Self(path) + } +} +impl Drop for Directory { + /// Clean up only this test's owned directory. + fn drop(&mut self) { + let _ = std::fs::remove_dir_all(&self.0); + } +} + +/// Report explicit capability skips or fail when real-kernel coverage is required. +fn capability_skip(reason: &str) { + assert_ne!( + std::env::var("PAGE_ALLOC_REQUIRE_REAL_IO").as_deref(), + Ok("1"), + "PAGE_ALLOC_REQUIRE_REAL_IO=1 forbids capability skips: {reason}" + ); + eprintln!("SKIP real slab test: {reason}"); +} + +/// Only explicit lack of direct-I/O support may skip real file workflows. +fn real_alignment(result: page_alloc::Result) -> Option { + match result { + Ok(alignment) => Some(alignment), + Err(Error::Unsupported) => { + capability_skip("filesystem does not support direct I/O geometry"); + None + } + Err(error) => panic!("unexpected real slab startup failure: {error}"), + } +} + +/// Probe baseline io_uring support without hiding unexpected runtime setup failures. +fn kernel_available() -> bool { + let mut params = [0u64; 15]; + // SAFETY: initialized, aligned 120-byte UAPI output; success owns a new descriptor. + let fd = unsafe { libc::syscall(libc::SYS_io_uring_setup, 2u32, params.as_mut_ptr()) }; + if fd >= 0 { + drop(unsafe { File::from_raw_fd(fd as i32) }); + return true; + } + let error = std::io::Error::last_os_error(); + match error.raw_os_error() { + Some(libc::ENOSYS | libc::EOPNOTSUPP | libc::EPERM | libc::EACCES) => { + capability_skip(&format!( + "io_uring kernel capability/permission error: {error}" + )); + false + } + _ => panic!("unexpected io_uring setup failure: {error}"), + } +} + +/// Drive real I/O with a strict in-test deadline as well as the external test timeout. +fn drive_real( + reactor: &Reactor, + mut op: Operation<'_, T, TestError>, +) -> Result { + let deadline = std::time::Instant::now() + std::time::Duration::from_secs(10); + loop { + if let Poll::Ready(value) = poll(&mut op) { + return value; + } + reactor.poll_budgeted(64).unwrap(); + assert!( + std::time::Instant::now() < deadline, + "real slab I/O did not complete" + ); + std::thread::yield_now(); + } +} + +/// The real kernel preserves payload bytes and releases write completion authority. +#[test] +fn io_uring_roundtrip_and_completion_fence() { + if !kernel_available() { + return; + } + let directory = Directory::new(); + let slab = Slab::<()>::new(directory.0.join("uring.dat"), 8192, 4096, 512); + let segments = Segments::new(4096); + if real_alignment(slab.open_configured(&segments)).is_none() { + return; + } + let reactor = Reactor::::new(16, ()); + reactor.init().unwrap(); + let (lease, extent) = segments.append(4096).unwrap(); + let mut buffer = slab.allocate(extent.length(), ()).unwrap(); + buffer.as_mut_slice().fill(73); + drop( + drive_real( + &reactor, + slab.write(&reactor, extent, buffer, lease, &TestScope), + ) + .unwrap(), + ); + assert_eq!(slab.writes_in_flight(), 0); + let mut fence = slab.fence_writes(); + assert_eq!(poll(&mut fence), Poll::Ready(Ok(()))); + let buffer = drive_real( + &reactor, + slab.read( + &reactor, + extent, + slab.allocate(extent.length(), ()).unwrap(), + segments.lease(SegmentId(0), Generation(1)).unwrap(), + &TestScope, + ), + ) + .unwrap(); + assert!(buffer.as_slice().iter().all(|byte| *byte == 73)); + drop(buffer); + assert_eq!(reactor.in_flight(), 0); + /// A completed roundtrip has no caller mappings left to remove. + struct EmptyEntries; + impl page_alloc::SegmentEntries for EmptyEntries { + /// The smoke workflow publishes no index entries. + fn remove_bounded(&self, _: SegmentId, _: usize) -> usize { + 0 + } + + /// Both slots are logically unmapped. + fn is_empty(&self, _: SegmentId) -> bool { + true + } + } + let clock = page_alloc::SegmentClock::new(std::rc::Rc::new(segments)); + clock.reclaim(&EmptyEntries, 2, 4, 0).unwrap(); +} + +/// Failed binding retains the open file but never grants admission to the reactor. +#[test] +fn unbound_and_failed_binding_reject_reads_and_writes_before_reactor_admission() { + let directory = Directory::new(); + #[cfg(feature = "simulation")] + let failures = ["unbound", "wrong-geometry", "frozen-table"].as_slice(); + #[cfg(not(feature = "simulation"))] + let failures = ["wrong-geometry", "frozen-table"].as_slice(); + for &failure in failures { + let slab = Slab::<()>::new(directory.0.join(failure), 8192, 4096, 512); + let correct = Segments::new(4096); + match failure { + #[cfg(feature = "simulation")] + "unbound" => { + if real_alignment(slab.open_now()).is_none() { + return; + } + } + "wrong-geometry" => { + let result = slab.open_configured(&Segments::new(8192)); + if result == Err(Error::Unsupported) { + let _ = real_alignment(result); + return; + } + assert_eq!(result, Err(Error::InvalidConfiguration)); + } + "frozen-table" => { + let frozen = correct.freeze().unwrap(); + let result = slab.open_configured(&correct); + if result == Err(Error::Unsupported) { + let _ = real_alignment(result); + return; + } + assert_eq!(result, Err(Error::Busy)); + drop(frozen); + } + _ => unreachable!(), + } + let alignment = slab.alignment().unwrap(); + let correct = Segments::from_geometry(slab.geometry().unwrap()).unwrap(); + let (lease, extent) = correct.append(4096).unwrap(); + let buffer = slab.allocate(4096, ()).unwrap(); + let reactor = Reactor::::new(16, ()); + let mut read = slab.read(&reactor, extent, buffer, lease, &TestScope); + assert!(matches!( + poll(&mut read), + Poll::Ready(Err(TestError::Alloc(Error::Unavailable))) + )); + drop(read); + let mut write = slab.write( + &reactor, + extent, + slab.allocate(4096, ()).unwrap(), + correct.lease(SegmentId(0), Generation(1)).unwrap(), + &TestScope, + ); + assert!(matches!( + poll(&mut write), + Poll::Ready(Err(TestError::Alloc(Error::Unavailable))) + )); + drop(write); + assert_eq!(reactor.in_flight(), 0); + assert_eq!(slab.writes_in_flight(), 0); + assert_eq!(slab.open_configured(&correct), Ok(alignment)); + // Acquiring the same binding again proves recovery is stable without kernel admission. + assert_eq!(slab.open_configured(&correct), Ok(alignment)); + // The successful roundtrip workflow separately proves bound admission and completion. + } +} + +/// Partial tables preserve file geometry, lease-fenced reuse, and unchanged disk bytes. +#[cfg(feature = "simulation")] +#[test] +fn partial_table_reclamation_changes_authority_without_erasing_storage() { + let simulation = Simulation::new(); + let _environment = simulation.enter(); + let reactor = Reactor::::new(16, ()); + let slab = Slab::<()>::new("/alloc-workflows/partial".into(), 16384, 4096, 512); + let alignment = slab.open_now().unwrap(); + let physical = slab.geometry().unwrap(); + let partial = page_alloc::SegmentGeometry::new(16384, 4096, 2, alignment).unwrap(); + let segments = Rc::new(Segments::from_geometry(partial).unwrap()); + assert_eq!(slab.open_configured(&segments), Ok(alignment)); + assert_eq!(slab.geometry().unwrap(), physical); + assert_eq!(physical.segment_count(), 4); + assert_eq!(segments.count(), 2); + assert_eq!(segments.geometry(), Some(partial)); + + let (lease, extent) = segments.append(4096).unwrap(); + let original_generation = lease.generation(); + let mut buffer = slab.allocate(4096, ()).unwrap(); + buffer.as_mut_slice().fill(91); + drop( + drive( + &reactor, + slab.write(&reactor, extent, buffer, lease, &TestScope), + ) + .unwrap(), + ); + assert_eq!(segments.free_count(), 1); + + /// One caller mapping whose removal is observable independently of disk bytes. + struct Entries { + populated: Cell, + + removals: Cell, + } + impl page_alloc::SegmentEntries for Entries { + /// Forget the mapping only for its actual segment and a positive budget. + fn remove_bounded(&self, segment: SegmentId, budget: usize) -> usize { + if segment == SegmentId(0) && budget != 0 && self.populated.replace(false) { + self.removals.set(self.removals.get() + 1); + 1 + } else { + 0 + } + } + + /// Report current caller ownership rather than allocator occupancy. + fn is_empty(&self, segment: SegmentId) -> bool { + segment != SegmentId(0) || !self.populated.get() + } + } + let entries = Entries { + populated: Cell::new(true), + removals: Cell::new(0), + }; + let clock = page_alloc::SegmentClock::new(segments.clone()); + let held = segments.lease(SegmentId(0), original_generation).unwrap(); + assert_eq!(clock.reclaim(&entries, 2, 4, 1), Err(Error::Busy)); + assert_eq!(entries.removals.get(), 1); + assert!(!entries.populated.get()); + assert_eq!(segments.state(SegmentId(0)), Ok(SegmentState::Evicting)); + assert!(matches!( + segments.lease(SegmentId(0), original_generation), + Err(Error::Stale) + )); + + // Existing completion authority still reads bytes after caller index removal. + let buffer = drive( + &reactor, + slab.read( + &reactor, + extent, + slab.allocate(4096, ()).unwrap(), + held, + &TestScope, + ), + ) + .unwrap(); + assert!(buffer.as_slice().iter().all(|byte| *byte == 91)); + drop(buffer); + clock.reclaim(&entries, 2, 4, 0).unwrap(); + assert_eq!(segments.free_count(), 2); + assert_eq!(entries.removals.get(), 1); + assert_eq!(segments.snapshot()[0].generation, Generation(2)); + + // Reuse grants new generation authority, but does not mutate the stored bytes. + let (replacement, replacement_extent) = segments.append(4096).unwrap(); + assert_eq!(replacement.id(), SegmentId(0)); + assert_eq!(replacement.generation(), Generation(2)); + assert_eq!(replacement_extent, extent); + assert_eq!( + segments.validate(SegmentId(0), original_generation, &extent), + Err(Error::Stale) + ); + let buffer = drive( + &reactor, + slab.read( + &reactor, + replacement_extent, + slab.allocate(4096, ()).unwrap(), + replacement, + &TestScope, + ), + ) + .unwrap(); + assert!(buffer.as_slice().iter().all(|byte| *byte == 91)); + drop(buffer); + assert_eq!(slab.writes_in_flight(), 0); + assert_eq!(reactor.in_flight(), 0); + assert_eq!(slab.geometry().unwrap(), physical); + let image = segments.snapshot(); + assert_eq!(image.len(), 2); + assert_eq!(image[0].state, SegmentState::Sealed); + assert_eq!(image[0].used_bytes, 4096); + assert_eq!(image[0].generation, Generation(2)); + assert_eq!(image[1].state, SegmentState::Free); + assert_eq!(image[1].used_bytes, 0); + assert_eq!(image[1].generation, Generation(1)); + assert_eq!(entries.removals.get(), 1); + assert_eq!(segments.free_count(), 1); +} From 9dd6b0668983f05387e507770c430f1bc248ab51 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Mon, 5 Oct 2026 22:15:32 +0000 Subject: [PATCH 04/82] ci(racer): validate isolated allocator production and simulation --- .github/workflows/ci.yaml | 20 ++++++++++++++++++++ cmd/racer-dataplane/README.md | 13 +++++++++++++ 2 files changed, 33 insertions(+) diff --git a/.github/workflows/ci.yaml b/.github/workflows/ci.yaml index f55d84816..01481b0a9 100644 --- a/.github/workflows/ci.yaml +++ b/.github/workflows/ci.yaml @@ -190,6 +190,26 @@ jobs: RUNTIME_REQUIRE_IO_URING: "1" run: timeout --signal=TERM --kill-after=10s 300s cargo +1.96.0 test --locked --manifest-path cmd/racer-dataplane/Cargo.toml -p uring-runtime --no-default-features + # Keep allocator invocations separate from runtime/workspace tests: runtime + # test features must not enable simulation in the production allocator gate. + - name: Check allocator production and simulation + run: | + timeout --signal=TERM --kill-after=10s 300s cargo +1.96.0 check --locked --manifest-path cmd/racer-dataplane/Cargo.toml -p page-alloc --all-targets --no-default-features + timeout --signal=TERM --kill-after=10s 300s cargo +1.96.0 check --locked --manifest-path cmd/racer-dataplane/Cargo.toml -p page-alloc --all-targets --no-default-features --features simulation + + - name: Strict allocator Clippy + run: | + timeout --signal=TERM --kill-after=10s 300s cargo +1.96.0 clippy --locked --manifest-path cmd/racer-dataplane/Cargo.toml -p page-alloc --all-targets --no-default-features -- -D warnings + timeout --signal=TERM --kill-after=10s 300s cargo +1.96.0 clippy --locked --manifest-path cmd/racer-dataplane/Cargo.toml -p page-alloc --all-targets --no-default-features --features simulation -- -D warnings + + - name: Allocator production real I/O (no simulation) + env: + PAGE_ALLOC_REQUIRE_REAL_IO: "1" + run: timeout --signal=TERM --kill-after=10s 300s cargo +1.96.0 test --locked --manifest-path cmd/racer-dataplane/Cargo.toml -p page-alloc --no-default-features + + - name: Test allocator simulation + run: timeout --signal=TERM --kill-after=10s 300s cargo +1.96.0 test --locked --manifest-path cmd/racer-dataplane/Cargo.toml -p page-alloc --no-default-features --features simulation + runtime-miri: name: Runtime Miri (${{ matrix.group }}) runs-on: ubuntu-24.04 diff --git a/cmd/racer-dataplane/README.md b/cmd/racer-dataplane/README.md index 7e68e4e02..5c0c0f82b 100644 --- a/cmd/racer-dataplane/README.md +++ b/cmd/racer-dataplane/README.md @@ -40,3 +40,16 @@ timeout --signal=TERM --kill-after=10s 300s python3 hack/scripts/runtime-miri.py Run the `offload`, `scheduler`, and `memory` groups separately with the same script. Each group verifies that every exact allowlisted test actually ran. + +The `page-alloc` library provides worker-local aligned buffers, lease-fenced slab +I/O, and bounded segment reclamation. Test it separately from runtime/workspace +tests so feature unification cannot enable simulation in its production gate: + +```sh +PAGE_ALLOC_REQUIRE_REAL_IO=1 timeout --signal=TERM --kill-after=10s 300s cargo +1.96.0 test --locked --manifest-path cmd/racer-dataplane/Cargo.toml -p page-alloc --no-default-features -j 2 -- --test-threads=1 +timeout --signal=TERM --kill-after=10s 300s cargo +1.96.0 test --locked --manifest-path cmd/racer-dataplane/Cargo.toml -p page-alloc --no-default-features --features simulation -j 2 -- --test-threads=1 +``` + +The production gate requires real io_uring and direct-I/O support: capability +skips fail when `PAGE_ALLOC_REQUIRE_REAL_IO=1`. CI also checks all targets and +runs strict Clippy for each allocator mode. From 1064d9b568c93dbd320ea3372d56e0604646f114 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Mon, 5 Oct 2026 22:24:01 +0000 Subject: [PATCH 05/82] fix(racer): honor segment eviction veto in allocator clock --- cmd/racer-dataplane/alloc/src/segments.rs | 60 +++++++++++++++++++++-- 1 file changed, 56 insertions(+), 4 deletions(-) diff --git a/cmd/racer-dataplane/alloc/src/segments.rs b/cmd/racer-dataplane/alloc/src/segments.rs index 9928f194c..781f8a64b 100644 --- a/cmd/racer-dataplane/alloc/src/segments.rs +++ b/cmd/racer-dataplane/alloc/src/segments.rs @@ -158,6 +158,7 @@ pub struct Segments { open: Cell>, free: RefCell>, + evicting: Cell, } impl Segments { @@ -563,6 +564,7 @@ pub trait SegmentEntries { fn can_evict(&self, _segment: SegmentId) -> bool { true } + /// Remove at most budget current mappings, returning the number removed. fn remove_bounded(&self, segment: SegmentId, budget: usize) -> usize; @@ -762,10 +764,11 @@ impl SegmentClock { return Ok(()); } let id = self.next(count); - if !matches!( - self.segments.state(id)?, - SegmentState::Sealed | SegmentState::Evicting - ) { + let state = self.segments.state(id)?; + if !matches!(state, SegmentState::Sealed | SegmentState::Evicting) { + continue; + } + if state == SegmentState::Sealed && !entries.can_evict(id) { continue; } if self.recent.borrow_mut().remove(&id) { @@ -1161,6 +1164,8 @@ mod clock_tests { counts: RefCell>, calls: RefCell>, + + evictable: Cell, } impl Entries { /// Populate a synthetic index without changing allocator state. @@ -1168,10 +1173,16 @@ mod clock_tests { Self { counts: RefCell::new(counts), calls: RefCell::new(vec![]), + evictable: Cell::new(true), } } } impl SegmentEntries for Entries { + /// Allow tests to protect unpublished mappings before eviction begins. + fn can_evict(&self, _: SegmentId) -> bool { + self.evictable.get() + } + /// Record and honor each bounded removal request. fn remove_bounded(&self, id: SegmentId, budget: usize) -> usize { self.calls.borrow_mut().push((id, budget)); @@ -1282,6 +1293,47 @@ mod clock_tests { assert_eq!(*entries.counts.borrow(), [0, 0]); } + /// Sealed vetoes preserve mappings, but cannot strand eviction already in progress. + #[test] + fn reclaim_honors_sealed_veto_but_drains_existing_eviction() { + for mapping_count in [0usize, 2] { + let segments = segments(1); + let held = segments.append(1024).unwrap().0; + let clock = SegmentClock::new(segments.clone()); + let entries = Entries::new(vec![mapping_count]); + entries.evictable.set(false); + let before = segments.snapshot(); + assert_eq!(before[0].state, SegmentState::Sealed); + + assert_eq!(clock.reclaim(&entries, 1, 2, 1), Err(Error::Busy)); + assert_eq!(segments.snapshot(), before); + assert_eq!(*entries.counts.borrow(), [mapping_count]); + assert!(entries.calls.borrow().is_empty()); + assert_eq!(segments.free_count(), 0); + + entries.evictable.set(true); + assert_eq!(clock.reclaim(&entries, 1, 1, 1), Err(Error::Busy)); + assert_eq!(segments.state(SegmentId(0)), Ok(SegmentState::Evicting)); + assert_eq!(*entries.counts.borrow(), [mapping_count.saturating_sub(1)]); + + // Once eviction starts, a later veto must not block remaining mappings. + entries.evictable.set(false); + assert_eq!(clock.reclaim(&entries, 1, 1, 1), Err(Error::Busy)); + assert_eq!(*entries.counts.borrow(), [0]); + assert_eq!(entries.calls.borrow().len(), mapping_count); + assert_eq!(segments.state(SegmentId(0)), Ok(SegmentState::Evicting)); + assert_eq!(segments.free_count(), 0); + assert_eq!(segments.snapshot()[0].generation, Generation(1)); + + // The live lease, not the veto, remains the physical reuse fence. + drop(held); + clock.reclaim(&entries, 1, 1, 0).unwrap(); + assert_eq!(segments.state(SegmentId(0)), Ok(SegmentState::Free)); + assert_eq!(segments.free_count(), 1); + assert_eq!(segments.snapshot()[0].generation, Generation(2)); + } + } + /// Freeze checks precede index side effects, and leases precede physical reuse. #[test] fn busy_lease_and_frozen_table_preserve_reclaim_side_effect_order() { From 12bfce1d62cff15aa3bea26f28bd66bbc05e0e5e Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Mon, 5 Oct 2026 22:56:24 +0000 Subject: [PATCH 06/82] refactor(racer): tighten allocator helper visibility and spacing --- cmd/racer-dataplane/alloc/src/lib.rs | 48 +++++++++++++++++++- cmd/racer-dataplane/alloc/src/segments.rs | 47 +++++++++++++++++++ cmd/racer-dataplane/alloc/src/slab.rs | 41 +++++++++++++++++ cmd/racer-dataplane/alloc/tests/workflows.rs | 11 +++++ 4 files changed, 146 insertions(+), 1 deletion(-) diff --git a/cmd/racer-dataplane/alloc/src/lib.rs b/cmd/racer-dataplane/alloc/src/lib.rs index 56de48150..77fb1a9b4 100644 --- a/cmd/racer-dataplane/alloc/src/lib.rs +++ b/cmd/racer-dataplane/alloc/src/lib.rs @@ -131,6 +131,7 @@ #![deny(missing_docs)] mod segments; + mod slab; pub use segments::{ @@ -154,18 +155,25 @@ use zeroize::Zeroize; pub enum Error { /// The filesystem or kernel cannot provide required direct-I/O support. Unsupported, + /// Caller configuration or an implementation contract is invalid. InvalidConfiguration, + /// A transient lease, freeze, or resource limit prevents progress. Busy, + /// An extent or persisted allocator image is malformed. Corrupt, + /// A generation, segment state, or table identity is no longer valid. Stale, + /// Storage is unopened, locked elsewhere, or has exhausted a generation. Unavailable, + /// An I/O completion was short or failed without OS error detail. Io, + /// Synchronous OS failure; asynchronous errors retain the runtime's error type. SystemIo { /// Operation that failed. @@ -201,6 +209,7 @@ impl fmt::Display for Error { }) } } + impl std::error::Error for Error {} /// Result of an allocator operation before conversion into a caller's scope error. @@ -269,7 +278,8 @@ impl SegmentGeometry { } /// Compare table dimensions only, not occupancy or alignment compatibility. - pub fn matches_segments(&self, segments: &Segments) -> bool { + #[cfg(test)] + fn matches_segments(&self, segments: &Segments) -> bool { segments.capacity_bytes() == self.slab_bytes && segments.segment_bytes() == self.segment_bytes && segments.count() as u64 == self.segment_count @@ -281,6 +291,7 @@ pub trait Charge: 'static { /// Whether this guard accounts for at least `bytes` live allocation bytes. fn covers(&self, bytes: usize) -> bool; } + impl Charge for () { /// Explicitly opt out of accounting for callers without admission policy. fn covers(&self, _bytes: usize) -> bool { @@ -299,6 +310,7 @@ pub struct Alignment { length: usize, } + /// A nonempty, non-overflowing file range for one bounded I/O transfer. #[must_use] #[derive(Clone, Copy, Debug, Eq, PartialEq)] @@ -307,6 +319,7 @@ pub struct Extent { length: usize, } + impl Extent { /// Reject empty ranges, offset overflow, and lengths above the transfer cap. pub fn new(offset: u64, length: usize) -> Result { @@ -318,17 +331,20 @@ impl Extent { } Ok(Self { offset, length }) } + /// Starting byte offset in the file. #[must_use] pub fn offset(self) -> u64 { self.offset } + /// Transfer length in bytes, including any padding. #[must_use] pub fn length(self) -> usize { self.length } } + impl Alignment { /// Conservative single-transfer limit of 1 GiB. /// @@ -350,21 +366,25 @@ impl Alignment { length, }) } + /// Required memory address alignment. #[must_use] pub fn memory(self) -> usize { self.memory } + /// Required file offset unit. #[must_use] pub fn offset(self) -> u64 { self.offset } + /// Required transfer length unit. #[must_use] pub fn length(self) -> usize { self.length } + /// Round up to the least common multiple of offset and length units so the /// next appended extent is also offset-aligned. Reject overflow and lengths /// above [`Self::MAX_TRANSFER_LENGTH`] before allocating memory. @@ -392,6 +412,7 @@ impl Alignment { } Extent::new(offset, length) } + /// Allocate zeroed stable storage and retain its primary accounting guard. /// Length must be nonzero, length-aligned, covered by `charge`, and no larger /// than [`Self::MAX_TRANSFER_LENGTH`]. Allocation failure returns `Busy`. @@ -417,6 +438,7 @@ impl Alignment { pool: Weak::new(), }) } + /// Validate the address, file offset, transfer unit, and exact buffer length. pub fn check(&self, extent: Extent, buffer: &AlignedBuffer) -> Result<()> { if !(buffer.allocation().pointer.as_ptr() as usize).is_multiple_of(self.memory) @@ -440,6 +462,7 @@ struct Allocation { retained: Vec>, } + impl Allocation { /// Exclusively borrow the complete initialized allocation with its original size. fn as_mut_slice(&mut self) -> &mut [u8] { @@ -447,6 +470,7 @@ impl Allocation { unsafe { std::slice::from_raw_parts_mut(self.pointer.as_ptr(), self.layout.size()) } } } + impl Drop for Allocation { /// Erase and free before field destruction releases the accounting guards. fn drop(&mut self) { @@ -476,6 +500,7 @@ pub struct AlignedBuffer { pool: Weak>>, } + impl fmt::Debug for AlignedBuffer { /// Show storage properties without requiring accounting guards to expose data. fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { @@ -486,20 +511,24 @@ impl fmt::Debug for AlignedBuffer { .finish_non_exhaustive() } } + impl AlignedBuffer { /// Borrow the allocation, which is absent only during drop's ownership transfer. fn allocation(&self) -> &Allocation { self.allocation.as_ref().expect("live buffer allocation") } + /// Mutably borrow live backing storage before any drop-time pool transfer. fn allocation_mut(&mut self) -> &mut Allocation { self.allocation.as_mut().expect("live buffer allocation") } + /// Attach a weak return destination without extending the pool's lifetime. pub(crate) fn pooled(mut self, pool: &Rc>>) -> Self { self.pool = Rc::downgrade(pool); self } + /// Replace primary accounting only after the new guard covers the full size. pub(crate) fn rebind(&mut self, charge: C) -> Result<()> { if !charge.covers(self.len()) { @@ -508,40 +537,48 @@ impl AlignedBuffer { self.allocation_mut().charge = charge; Ok(()) } + /// Retain additional accounting through the final kernel completion. pub fn retain(&mut self, charge: Rc) { self.allocation_mut().retained.push(charge); } + /// Initialized allocation length, including padding. #[must_use] pub fn len(&self) -> usize { self.allocation().layout.size() } + /// Always false: construction rejects zero-length buffers. #[must_use] pub fn is_empty(&self) -> bool { false } + /// Borrow all initialized bytes without moving or resizing the allocation. #[must_use] pub fn as_slice(&self) -> &[u8] { // SAFETY: initialized allocation remains live throughout this borrow. unsafe { std::slice::from_raw_parts(self.allocation().pointer.as_ptr(), self.len()) } } + /// Exclusively borrow all initialized bytes. #[must_use] pub fn as_mut_slice(&mut self) -> &mut [u8] { self.allocation_mut().as_mut_slice() } + /// Compatibility accessor matching [`IoBuffer`]; this always succeeds. pub fn bytes(&self) -> Result<&[u8]> { Ok(self.as_slice()) } + /// Compatibility accessor matching [`IoBuffer`]; this always succeeds. pub fn bytes_mut(&mut self) -> Result<&mut [u8]> { Ok(self.as_mut_slice()) } } + impl Drop for AlignedBuffer { /// Zeroize and return storage if possible, otherwise let its owner free it. fn drop(&mut self) { @@ -564,6 +601,7 @@ impl Drop for AlignedBuffer { // pool is gone, occupied, or borrowed. Idle buffers never repool themselves. } } + // SAFETY: owned aligned backing is initialized, stable, and live until Drop. unsafe impl IoBuffer for AlignedBuffer { type Error = Error; @@ -591,6 +629,7 @@ mod buffer_tests { bytes: usize, } + impl TrackedCharge { /// Admit and record a fixed number of live bytes. fn new(live: &Rc>, bytes: usize) -> Self { @@ -601,12 +640,14 @@ mod buffer_tests { } } } + impl Charge for TrackedCharge { /// Cover only the number of bytes admitted by this guard. fn covers(&self, bytes: usize) -> bool { self.bytes >= bytes } } + impl Drop for TrackedCharge { /// Return admitted bytes exactly once. fn drop(&mut self) { @@ -619,6 +660,7 @@ mod buffer_tests { fn transfer_limit_is_enforced_before_allocation_or_charge_inspection() { /// Detects any accounting inspection on a structurally invalid request. struct UncheckedCharge; + impl Charge for UncheckedCharge { /// Panic when validation reaches accounting in the wrong order. fn covers(&self, _: usize) -> bool { @@ -840,12 +882,14 @@ mod buffer_tests { fn retained_guard_can_inspect_pool_before_buffer_is_returned() { /// Runs a caller-provided destructor to test reentrant pool inspection. struct Guard(Option>); + impl Charge for Guard { /// Admit all sizes for this destructor-order test. fn covers(&self, _: usize) -> bool { true } } + impl Drop for Guard { /// Invoke the callback at most once. fn drop(&mut self) { @@ -881,12 +925,14 @@ mod buffer_tests { panic: bool, } + impl Charge for Guard { /// Admit all sizes for the unwind test. fn covers(&self, _: usize) -> bool { true } } + impl Drop for Guard { /// Record release before injecting the requested destructor panic. fn drop(&mut self) { diff --git a/cmd/racer-dataplane/alloc/src/segments.rs b/cmd/racer-dataplane/alloc/src/segments.rs index 781f8a64b..8079a5a7b 100644 --- a/cmd/racer-dataplane/alloc/src/segments.rs +++ b/cmd/racer-dataplane/alloc/src/segments.rs @@ -13,21 +13,27 @@ use std::{ /// Stable slot number within one allocation table, not authority to access it. #[derive(Clone, Copy, Debug, Eq, Hash, PartialEq)] pub struct SegmentId(pub u64); + /// Monotonic reuse counter; zero is invalid in a restored image. #[derive(Clone, Copy, Debug, Eq, Hash, PartialEq)] pub struct Generation(pub u64); + /// Persisted segment lifecycle, separate from live lease and freeze ownership. #[derive(Clone, Copy, Debug, Eq, PartialEq)] pub enum SegmentState { /// Available for a new append. Free, + /// The sole appendable tail. Open, + /// Full or rotated away from; eligible for eviction. Sealed, + /// Rejects new leases while existing completion owners drain. Evicting, } + /// Caller-owned recovery image; restore validates it before publishing any slot. #[derive(Clone, Debug, Eq, PartialEq)] pub struct SegmentSnapshot { @@ -68,23 +74,28 @@ pub struct SegmentLease { end: u64, } + impl SegmentLease { /// Slot protected against recycling by this lease. pub fn id(&self) -> SegmentId { self.id } + /// Reuse generation captured when the lease was acquired. pub fn generation(&self) -> Generation { self.generation } + /// Borrow the nominal identity without granting table mutation authority. pub(crate) fn table_identity(&self) -> TableIdentity { self.table.clone() } + /// Validated dimensions captured independently of the table's lifetime. pub(crate) fn geometry(&self) -> SegmentGeometry { self.geometry } + /// An existing lease remains valid during eviction, but cannot authorize /// bytes appended after it was acquired. pub(crate) fn validate_extent(&self, extent: &Extent) -> Result<()> { @@ -106,18 +117,21 @@ impl SegmentLease { Ok(()) } } + impl Drop for SegmentLease { /// Release exactly the one lease count acquired during construction. fn drop(&mut self) { self.count.set(self.count.get() - 1); } } + /// Mutable recovery image paired with its independently retained lease counter. struct Slot { image: SegmentSnapshot, leases: Rc>, } + /// Holds a freeze independently of the table's lifetime. Dropping it thaws the table. /// /// A second guard cannot be created by cloning the first: @@ -128,12 +142,14 @@ struct Slot { #[must_use = "keep the guard alive while the table must remain frozen"] #[derive(Debug)] pub struct FreezeGuard(Rc>); + impl Drop for FreezeGuard { /// Release mutation exclusion even if the table has already been dropped. fn drop(&mut self) { self.0.set(false); } } + /// Non-cloneable worker-local table whose leases and guards can outlive it. /// Rc sharing is intentional within a worker; authority cannot cross threads. /// @@ -161,6 +177,7 @@ pub struct Segments { evicting: Cell, } + impl Segments { /// Create an unconfigured table for Slab::open_configured to bind at startup. pub fn new(segment_bytes: u64) -> Self { @@ -176,14 +193,17 @@ impl Segments { evicting: Cell::new(0), } } + /// Fixed size of a physical segment. pub fn segment_bytes(&self) -> u64 { self.segment_bytes } + /// Configured physical capacity, or zero before configuration. pub fn capacity_bytes(&self) -> u64 { self.geometry.get().map_or(0, SegmentGeometry::slab_bytes) } + /// Build an entirely free table, rejecting geometry above the retained slot limit. pub fn from_geometry(geometry: SegmentGeometry) -> Result { let segments = Self::new(geometry.segment_bytes()); @@ -194,22 +214,27 @@ impl Segments { )?; Ok(segments) } + /// Whether validated geometry and the slot table have been installed. pub fn is_configured(&self) -> bool { self.geometry.get().is_some() } + /// Configured geometry, which may expose fewer slots than physical capacity. pub fn geometry(&self) -> Option { self.geometry.get() } + /// Clone identity only, without granting allocation authority. pub(crate) fn table_identity(&self) -> TableIdentity { self.table.clone() } + /// Recovery revision used to invalidate eviction cursor and recent-read state. pub(crate) fn restore_epoch(&self) -> u64 { self.restore_epoch.get() } + /// Install an entirely free bounded table exactly once, unless frozen. #[cfg(any(test, feature = "simulation"))] pub fn configure(&self, capacity: u64, count: usize, alignment: Alignment) -> Result<()> { @@ -246,6 +271,7 @@ impl Segments { *self.free.borrow_mut() = (0..count).collect(); Ok(()) } + /// Reserve an aligned used range. Malformed requests and lease overflow leave /// state unchanged. A valid rollover without a free slot seals the open tail /// before returning Busy, allowing reclamation to make a retry possible. @@ -310,6 +336,7 @@ impl Segments { } Ok((lease, extent)) } + /// Capture a checked prefix and increment its counter before publishing a lease. fn take_lease(&self, slot: &Slot, used_bytes: u64) -> Result { let start = slot @@ -332,10 +359,12 @@ impl Segments { end, }) } + /// Convert an external slot number without truncation. fn position(id: SegmentId) -> Result { usize::try_from(id.0).map_err(|_| Error::Corrupt) } + /// Acquire the current used prefix only while the generation is readable. pub fn lease(&self, id: SegmentId, generation: Generation) -> Result { let slots = self.slots.borrow(); @@ -347,6 +376,7 @@ impl Segments { } self.take_lease(slot, slot.image.used_bytes) } + /// Stop new leases for a sealed slot; repeating eviction is harmless. #[cfg(any(test, feature = "simulation"))] pub fn begin_evict(&self, id: SegmentId) -> Result<()> { @@ -372,6 +402,7 @@ impl Segments { } Ok(()) } + /// Reuse only an evicting, unleased slot, incrementing generation without wrap. #[cfg(any(test, feature = "simulation"))] pub fn recycle(&self, id: SegmentId) -> Result<()> { @@ -401,6 +432,7 @@ impl Segments { self.free.borrow_mut().insert(Self::position(id)?); Ok(()) } + /// Copy the complete ordered image, including while the table is frozen. pub fn snapshot(&self) -> Vec { self.slots @@ -409,6 +441,7 @@ impl Segments { .map(|s| s.image.clone()) .collect() } + /// Block allocation, eviction, recycling, and restore until the guard drops. /// Reads and snapshots remain available; only one guard may exist at a time. pub fn freeze(&self) -> Result { @@ -417,10 +450,12 @@ impl Segments { } Ok(FreezeGuard(self.frozen.clone())) } + /// Check the entire image and current restore eligibility without mutation. pub fn validate_restore(&self, images: &[SegmentSnapshot]) -> Result<()> { self.validate_restore_epoch(images).map(|_| ()) } + /// Validate recovery invariants and reserve the next nonwrapping epoch value. fn validate_restore_epoch(&self, images: &[SegmentSnapshot]) -> Result { if self.frozen.get() { @@ -458,6 +493,7 @@ impl Segments { .checked_add(1) .ok_or(Error::Unavailable) } + /// Validate the complete image before publishing it, sealing its open tail. /// Frozen tables and outstanding leases return Busy without changing state. pub fn restore(&self, images: Vec) -> Result<()> { @@ -481,6 +517,7 @@ impl Segments { self.restore_epoch.set(epoch); Ok(()) } + /// Validate a stored mapping against current readable state and used bytes. pub fn validate(&self, id: SegmentId, generation: Generation, extent: &Extent) -> Result<()> { let slots = self.slots.borrow(); @@ -513,6 +550,7 @@ impl Segments { } Ok(()) } + /// Validate a live lease, including table identity and its captured used range. /// Existing leases remain usable while their segment is Evicting. pub fn validate_lease(&self, lease: &SegmentLease, extent: &Extent) -> Result<()> { @@ -521,14 +559,17 @@ impl Segments { } lease.validate_extent(extent) } + /// Number of immediately appendable slots, excluding pending eviction. pub fn free_count(&self) -> usize { self.free.borrow().len() } + /// Number of retained slots, which may be less than physical capacity. pub fn count(&self) -> usize { self.slots.borrow().len() } + /// Inspect a valid slot without granting mutation or lease authority. pub fn state(&self, id: SegmentId) -> Result { self.slots @@ -1167,6 +1208,7 @@ mod clock_tests { evictable: Cell, } + impl Entries { /// Populate a synthetic index without changing allocator state. fn new(counts: Vec) -> Self { @@ -1177,6 +1219,7 @@ mod clock_tests { } } } + impl SegmentEntries for Entries { /// Allow tests to protect unpublished mappings before eviction begins. fn can_evict(&self, _: SegmentId) -> bool { @@ -1441,6 +1484,7 @@ mod clock_tests { assert!(entries.calls.borrow().is_empty()); assert!(clock.recent.borrow().contains(&SegmentId(0))); } + /// Scoring completes before mutation; mapping budgets, freeze, leases and /// generation authority remain enforced even when the cheapest slot is busy. #[test] @@ -1483,6 +1527,7 @@ mod clock_tests { assert_eq!(segments.free_count(), 1); assert_eq!(segments.snapshot()[1].generation, Generation(2)); } + /// A score callback never turns a soft preference into immunity or a full scan. #[test] fn scored_eviction_caps_candidates_and_evicts_maximum_scores() { @@ -1504,6 +1549,7 @@ mod clock_tests { assert_eq!(segments.free_count(), 1); assert_eq!(clock.hand.get(), 64); } + /// Pending leased victims count toward reserve even outside the next sample. #[test] fn scored_eviction_never_overshoots_reserve_with_multiple_leased_victims() { @@ -1533,6 +1579,7 @@ mod clock_tests { fn removal_contract_violations_return_errors_without_panicking() { /// An intentionally broken callback that over-reports every removal. struct InvalidEntries; + impl SegmentEntries for InvalidEntries { /// Violate the caller contract to test error handling. fn remove_bounded(&self, _: SegmentId, budget: usize) -> usize { diff --git a/cmd/racer-dataplane/alloc/src/slab.rs b/cmd/racer-dataplane/alloc/src/slab.rs index 326037186..6bfbe97c1 100644 --- a/cmd/racer-dataplane/alloc/src/slab.rs +++ b/cmd/racer-dataplane/alloc/src/slab.rs @@ -48,6 +48,7 @@ pub struct Slab { idle_buffer: Rc>>>, } + impl Slab { /// Describe a worker's file; validation and blocking I/O happen at startup. pub fn new( @@ -66,27 +67,33 @@ impl Slab { idle_buffer: Rc::new(RefCell::new(None)), } } + /// Logical file capacity, not a reservation of physical disk blocks. pub fn capacity_bytes(&self) -> u64 { self.capacity_bytes } + /// Size of each physical segment. pub fn segment_bytes(&self) -> u64 { self.segment_bytes } + /// Accepted writes whose completion guards have not yet been released. pub fn writes_in_flight(&self) -> usize { self.writes.count.get() } + /// Accounted bytes in the single idle buffer slot. pub fn idle_bytes(&self) -> usize { self.idle_buffer.borrow().as_ref().map_or(0, |b| b.len()) } + /// Release only idle memory, returning its size without touching live I/O. pub fn reclaim_idle(&self) -> usize { let idle = self.idle_buffer.borrow_mut().take(); idle.map_or(0, |b| b.len()) } + /// Wait for accepted writes to release their runtime completion fences. /// This does not call fsync/fdatasync and does NOT promise crash durability. /// Drive the reactor concurrently; stop new writes first if quiescence is needed. @@ -96,6 +103,7 @@ impl Slab { registration: None, }) } + /// Discovered direct-I/O requirements, or Unavailable before startup. pub fn alignment(&self) -> Result { self.opened @@ -104,6 +112,7 @@ impl Slab { .map(|o| o.geometry.alignment()) .ok_or(Error::Unavailable) } + /// Physical slab geometry, including the full segment capacity. A bound /// table may deliberately expose fewer segments than this physical count. pub fn geometry(&self) -> Result { @@ -113,6 +122,7 @@ impl Slab { .map(|o| o.geometry) .ok_or(Error::Unavailable) } + /// Configure an empty table, or validate an already configured partial table, /// and permanently bind this slab to that table's identity. #[cfg(any(test, feature = "simulation"))] @@ -150,12 +160,14 @@ impl Slab { slab.table = Some(identity); Ok(()) } + /// Blocking startup helper. Do not invoke on a latency-sensitive worker. pub fn open_configured(&self, segments: &Segments) -> Result { let alignment = self.open_file()?; self.bind(segments)?; Ok(alignment) } + /// Blocking startup I/O for geometry probing and buffer allocation. Read/write /// return `Unavailable` until `configure_segments` succeeds. /// Parent directories must be trusted against @@ -256,6 +268,7 @@ impl Slab { self.publish(file.into(), geometry); Ok(a) } + /// Publish geometry with its owning file, initially without I/O authority. fn publish(&self, file: Descriptor, geometry: SegmentGeometry) { *self.opened.borrow_mut() = Some(OpenSlab { @@ -264,6 +277,7 @@ impl Slab { table: None, }); } + /// Check physical dimensions and padded record size without changing the file. fn validate_layout(&self, a: Alignment, size: u64) -> Result { if a.extent(0, self.max_record_bytes)?.length() as u64 > self.segment_bytes @@ -281,6 +295,7 @@ impl Slab { ) .map_err(|_| Error::InvalidConfiguration) } + /// Borrow-check a request before moving its resources into a submission. fn submission( &self, @@ -323,6 +338,7 @@ impl Slab { } Ok(slab.file.clone()) } + /// Consume a checked request so its descriptor, buffer, and lease travel together. fn prepare( &self, @@ -338,6 +354,7 @@ impl Slab { lease, }) } + /// Read exactly one checked extent, retaining its buffer and lease until completion. /// Dropping the waiting future does not release kernel-owned resources. pub fn read<'a, S: Scope, B: Budget>( @@ -357,6 +374,7 @@ impl Slab { .await }) } + /// Write exactly one checked extent with completion-owned accounting and lease. /// Failed writes do not roll back the space reserved by append. pub fn write<'a, S: Scope, B: Budget>( @@ -376,6 +394,7 @@ impl Slab { submission.write(reactor, scope, fence).await }) } + /// Reuse one exact-size idle buffer. A size mismatch releases the old idle /// buffer and its charge, even if the replacement allocation fails. pub fn allocate(&self, length: usize, charge: C) -> Result> { @@ -394,12 +413,14 @@ impl Slab { .allocate(length, charge)? .pooled(&self.idle_buffer)) } + /// Replace the open file for fault-injection tests; geometry remains unchanged. #[cfg(feature = "simulation")] #[doc(hidden)] pub fn replace_file_for_test(&self, file: File) -> Result<()> { self.replace_descriptor_for_test(file.into()) } + /// Also accepts a virtual descriptor from the simulation backend. #[cfg(feature = "simulation")] #[doc(hidden)] @@ -415,6 +436,7 @@ impl Slab { Ok(()) } } + /// A file and its geometry own their binding; a closed slab cannot retain authority. struct OpenSlab { file: Rc, @@ -500,6 +522,7 @@ struct WriteState { waiters: RefCell>>>, } + impl WriteState { /// Increment before constructing the sole guard responsible for decrementing. fn acquire(self: &Rc) -> Result { @@ -508,12 +531,14 @@ impl WriteState { Ok(WriteFence(self.clone())) } } + /// Cancel-safe registration waiting for all currently accepted writes to finish. struct FenceWaiter { state: Rc, registration: Option>>, } + impl Future for FenceWaiter { type Output = Result<()>; @@ -539,6 +564,7 @@ impl Future for FenceWaiter { Poll::Pending } } + impl FenceWaiter { /// Remove only this waiter's registration, including after cancellation. fn unregister(&mut self) { @@ -550,14 +576,17 @@ impl FenceWaiter { } } } + impl Drop for FenceWaiter { /// A canceled wait must not retain its task's waker. fn drop(&mut self) { self.unregister(); } } + /// Unique write-count decrement authority retained by the reactor completion. struct WriteFence(Rc); + impl Drop for WriteFence { /// Release the count and wake sleepers outside all registration borrows. fn drop(&mut self) { @@ -572,6 +601,7 @@ impl Drop for WriteFence { } } } + /// Preserve synchronous operating-system diagnostic context. fn system_error(operation: &'static str, error: std::io::Error) -> Error { Error::SystemIo { @@ -579,6 +609,7 @@ fn system_error(operation: &'static str, error: std::io::Error) -> Error { errno: error.raw_os_error(), } } + /// Distinguish an already locked file from a failed locking syscall. fn lock_error(error: std::io::Error) -> Error { if error.raw_os_error() == Some(libc::EWOULDBLOCK) { @@ -587,6 +618,7 @@ fn lock_error(error: std::io::Error) -> Error { system_error("flock", error) } } + /// Classify explicit direct-I/O capability denials without hiding other failures. fn direct_error(operation: &'static str, error: std::io::Error) -> Error { match error.raw_os_error() { @@ -594,10 +626,12 @@ fn direct_error(operation: &'static str, error: std::io::Error) -> Error { _ => system_error(operation, error), } } + /// Discover direct-I/O requirements for an owned file descriptor. fn probe(file: &File) -> Result { probe_fd(file.as_raw_fd()) } + /// Ask Linux for descriptor-specific alignment without assuming a page size. fn probe_fd(fd: i32) -> Result { // SAFETY: initialized statx output and valid empty C path. Invalid FDs are @@ -621,6 +655,7 @@ fn probe_fd(fd: i32) -> Result { } alignment_from_stat(&stat) } + /// Reject missing alignment capability or invalid values returned by statx. fn alignment_from_stat(stat: &libc::statx) -> Result { if stat.stx_mask & libc::STATX_DIOALIGN == 0 { @@ -718,6 +753,7 @@ mod tests { bytes: usize, } + impl CountingCharge { /// Admit a fixed number of bytes. fn new(used: &Rc>, bytes: usize) -> Self { @@ -728,12 +764,14 @@ mod tests { } } } + impl Charge for CountingCharge { /// Cover only bytes actually admitted by this guard. fn covers(&self, bytes: usize) -> bool { self.bytes >= bytes } } + impl Drop for CountingCharge { /// Release accounting exactly once on destruction. fn drop(&mut self) { @@ -743,6 +781,7 @@ mod tests { /// Owns and cleans up an isolated directory inside the project build tree. struct Directory(PathBuf); + impl Directory { /// Create a per-process, per-test directory without touching host temp paths. fn new() -> Self { @@ -755,6 +794,7 @@ mod tests { Self(path) } } + impl Drop for Directory { /// Remove only this test's owned directory. fn drop(&mut self) { @@ -985,6 +1025,7 @@ mod tests { /// Counts wake notifications without scheduling actual work. #[derive(Default)] struct WakeCount(AtomicUsize); + impl Wake for WakeCount { /// Count an owned notification. fn wake(self: Arc) { diff --git a/cmd/racer-dataplane/alloc/tests/workflows.rs b/cmd/racer-dataplane/alloc/tests/workflows.rs index 6c82c603a..8b8a97c2d 100644 --- a/cmd/racer-dataplane/alloc/tests/workflows.rs +++ b/cmd/racer-dataplane/alloc/tests/workflows.rs @@ -21,14 +21,17 @@ use uring_runtime::{Operation, Scope, reactor::Reactor}; #[derive(Clone, Copy, Debug, PartialEq, Eq)] enum TestError { Alloc(Error), + Runtime(uring_runtime::Error), } + impl From for TestError { /// Preserve the allocator category in workflow assertions. fn from(error: Error) -> Self { Self::Alloc(error) } } + impl From for TestError { /// Preserve runtime errors without enriching them as synchronous allocator errors. fn from(error: uring_runtime::Error) -> Self { @@ -39,6 +42,7 @@ impl From for TestError { /// An always-live caller scope for deterministic ownership tests. #[derive(Clone)] struct TestScope; + impl Scope for TestScope { type Error = TestError; @@ -77,6 +81,7 @@ struct CountingCharge { bytes: usize, } + #[cfg(feature = "simulation")] impl CountingCharge { /// Record admission before passing the guard into allocator ownership. @@ -88,6 +93,7 @@ impl CountingCharge { } } } + #[cfg(feature = "simulation")] impl Charge for CountingCharge { /// Cover only the bytes actually admitted by this guard. @@ -95,6 +101,7 @@ impl Charge for CountingCharge { self.bytes >= bytes } } + #[cfg(feature = "simulation")] impl Drop for CountingCharge { /// Return admission when the final owner releases this guard. @@ -667,6 +674,7 @@ fn replacement_hooks_are_fallible_and_refuse_live_writes() { /// Owns a unique project-local directory for real kernel workflows. struct Directory(PathBuf); + impl Directory { /// Create an isolated test directory without using the host temporary directory. fn new() -> Self { @@ -679,6 +687,7 @@ impl Directory { Self(path) } } + impl Drop for Directory { /// Clean up only this test's owned directory. fn drop(&mut self) { @@ -791,6 +800,7 @@ fn io_uring_roundtrip_and_completion_fence() { assert_eq!(reactor.in_flight(), 0); /// A completed roundtrip has no caller mappings left to remove. struct EmptyEntries; + impl page_alloc::SegmentEntries for EmptyEntries { /// The smoke workflow publishes no index entries. fn remove_bounded(&self, _: SegmentId, _: usize) -> usize { @@ -913,6 +923,7 @@ fn partial_table_reclamation_changes_authority_without_erasing_storage() { removals: Cell, } + impl page_alloc::SegmentEntries for Entries { /// Forget the mapping only for its actual segment and a positive budget. fn remove_bounded(&self, segment: SegmentId, budget: usize) -> usize { From 39f84b8d9cd5fa113759ce715dfdc851cdb8c449 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Mon, 5 Oct 2026 23:44:59 +0000 Subject: [PATCH 07/82] fix(alloc): track clean storage and avoid redundant secure wipes --- cmd/racer-dataplane/alloc/src/lib.rs | 116 +++++++++++++++++++++++++-- 1 file changed, 111 insertions(+), 5 deletions(-) diff --git a/cmd/racer-dataplane/alloc/src/lib.rs b/cmd/racer-dataplane/alloc/src/lib.rs index 77fb1a9b4..b1f70eac7 100644 --- a/cmd/racer-dataplane/alloc/src/lib.rs +++ b/cmd/racer-dataplane/alloc/src/lib.rs @@ -147,7 +147,6 @@ use std::{ rc::{Rc, Weak}, }; use uring_runtime::reactor::IoBuffer; -use zeroize::Zeroize; /// Storage failures, separated from application record and admission policy. #[derive(Clone, Copy, Debug, Eq, PartialEq)] @@ -434,6 +433,9 @@ impl Alignment { layout, charge, retained: Vec::new(), + clean: true, + #[cfg(test)] + wipes: Rc::new(std::cell::Cell::new(0)), }), pool: Weak::new(), }) @@ -461,20 +463,58 @@ struct Allocation { charge: C, retained: Vec>, + + // True only after zeroed allocation or a complete secure wipe. Every mutable + // exposure, including the runtime's kernel-write borrow, clears this first. + clean: bool, + + #[cfg(test)] + wipes: Rc>, } impl Allocation { /// Exclusively borrow the complete initialized allocation with its original size. fn as_mut_slice(&mut self) -> &mut [u8] { + self.clean = false; // SAFETY: exclusive owner, initialized nonzero allocation, original layout. unsafe { std::slice::from_raw_parts_mut(self.pointer.as_ptr(), self.layout.size()) } } + + /// Securely erase the entire allocation once after each mutable exposure. + fn wipe(&mut self) { + if self.clean { + return; + } + #[cfg(all( + target_os = "linux", + any(target_env = "gnu", target_env = "musl"), + not(miri) + ))] + // SAFETY: this exclusive owner holds layout.size() initialized writable + // bytes. explicit_bzero cannot be eliminated as a dead store. No kernel + // operation can outlive ownership's completion fence. + unsafe { + libc::explicit_bzero(self.pointer.as_ptr().cast(), self.layout.size()) + }; + #[cfg(not(all( + target_os = "linux", + any(target_env = "gnu", target_env = "musl"), + not(miri) + )))] + { + use zeroize::Zeroize; + self.as_mut_slice().zeroize(); + } + self.clean = true; + #[cfg(test)] + self.wipes.set(self.wipes.get() + 1); + } } impl Drop for Allocation { /// Erase and free before field destruction releases the accounting guards. fn drop(&mut self) { - self.as_mut_slice().zeroize(); + self.wipe(); // SAFETY: this owner holds the allocation and its original layout. Guards // are dropped only after zeroization and deallocation complete. unsafe { dealloc(self.pointer.as_ptr(), self.layout) }; @@ -583,7 +623,7 @@ impl Drop for AlignedBuffer { /// Zeroize and return storage if possible, otherwise let its owner free it. fn drop(&mut self) { if let Some(pool) = self.pool.upgrade() { - self.allocation_mut().as_mut_slice().zeroize(); + self.allocation_mut().wipe(); // Caller-owned guard destructors may access the pool. Do not invoke // them while holding its RefCell borrow. Allocation remains owned // by this buffer if a destructor unwinds. @@ -597,8 +637,9 @@ impl Drop for AlignedBuffer { }); } } - // Otherwise field drop zeroizes and frees the allocation, even when the - // pool is gone, occupied, or borrowed. Idle buffers never repool themselves. + // Field drop securely wipes any still-dirty allocation and frees it. A + // rejected pool return or idle free needs no second wipe. Idle buffers + // never repool themselves. } } @@ -655,6 +696,59 @@ mod buffer_tests { } } + /// Full padded capacity is erased, and clean reuse does not erase twice. + #[test] + fn secure_wipe_covers_padding_and_skips_clean_pool_lifetimes() { + let alignment = Alignment::new(4096, 512, 512).unwrap(); + let length = alignment.extent(0, 513).unwrap().length(); + let pool = Rc::new(RefCell::new(None)); + let mut buffer = alignment.allocate(length, ()).unwrap().pooled(&pool); + let wipes = buffer.allocation().wipes.clone(); + assert!(buffer.allocation().clean); + assert_eq!(buffer.as_slice(), vec![0; 1024]); + drop(buffer); + assert_eq!(wipes.get(), 0); + buffer = pool.borrow_mut().take().unwrap().pooled(&pool); + buffer.as_mut_slice().fill(0xa5); + assert!(!buffer.allocation().clean); + drop(buffer); + assert_eq!(wipes.get(), 1); + buffer = pool.borrow_mut().take().unwrap().pooled(&pool); + assert_eq!(buffer.as_slice(), vec![0; 1024]); + assert!(buffer.allocation().clean); + drop(buffer); + drop(pool); + assert_eq!(wipes.get(), 1); + } + + /// Runtime pointer writes dirty storage before submission, including padding. + #[test] + fn kernel_write_borrow_dirties_clean_reused_storage() { + let pool = Rc::new(RefCell::new(None)); + let alignment = Alignment::new(64, 1, 1).unwrap(); + let mut buffer = alignment.allocate(128, ()).unwrap().pooled(&pool); + let wipes = buffer.allocation().wipes.clone(); + for expected in 1..=2 { + assert!(buffer.allocation().clean); + let pointer = IoBuffer::bytes_mut(&mut buffer).unwrap().as_mut_ptr(); + // SAFETY: simulate a kernel completion while the exclusive owner is + // retained, before any subsequent access or release of the buffer. + unsafe { pointer.add(127).write(0x5a) }; + assert!(!buffer.allocation().clean); + assert_eq!(buffer.as_slice()[127], 0x5a); + drop(buffer); + assert_eq!(wipes.get(), expected); + buffer = pool.borrow_mut().take().unwrap().pooled(&pool); + assert_eq!(IoBuffer::bytes(&buffer).unwrap(), &[0; 128]); + } + // Even an unused mutable borrow must conservatively require a wipe. + let _ = buffer.bytes_mut().unwrap(); + drop(buffer); + assert_eq!(wipes.get(), 3); + drop(pool); + assert_eq!(wipes.get(), 3); + } + /// Invalid transfers must fail before inspecting accounting or allocating. #[test] fn transfer_limit_is_enforced_before_allocation_or_charge_inspection() { @@ -845,6 +939,8 @@ mod buffer_tests { .allocate(8, TrackedCharge::new(&live, 8)) .unwrap() .pooled(&pool); + buffer.as_mut_slice().fill(0x5a); + let wipes = buffer.allocation().wipes.clone(); buffer.retain(Rc::new(TrackedCharge::new(&live, 3))); match scenario { 0 => { @@ -874,6 +970,7 @@ mod buffer_tests { } } assert_eq!(live.get(), 0); + assert_eq!(wipes.get(), 1); } } @@ -907,13 +1004,19 @@ mod buffer_tests { .pooled(&pool); let observed_pool = pool.clone(); let observed_called = called.clone(); + buffer.as_mut_slice().fill(0x5a); + let wipes = buffer.allocation().wipes.clone(); + let observed_wipes = wipes.clone(); buffer.retain(Rc::new(Guard(Some(Box::new(move || { assert!(observed_pool.borrow_mut().is_none()); + assert_eq!(observed_wipes.get(), 1); observed_called.set(true); }))))); drop(buffer); assert!(called.get()); assert!(pool.borrow().is_some()); + drop(pool); + assert_eq!(wipes.get(), 1); } /// Unwinding through extra accounting cannot leak primary allocation ownership. @@ -953,6 +1056,8 @@ mod buffer_tests { ) .unwrap() .pooled(&pool); + buffer.as_mut_slice().fill(0x5a); + let wipes = buffer.allocation().wipes.clone(); buffer.retain(Rc::new(Guard { live: live.clone(), panic: true, @@ -960,6 +1065,7 @@ mod buffer_tests { assert!(std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| drop(buffer))).is_err()); assert_eq!(live.get(), 0); assert!(pool.borrow().is_none()); + assert_eq!(wipes.get(), 1); } } From cac312bdf3a9451be8bd2b7d565e5fc681fcae5e Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Mon, 5 Oct 2026 23:50:34 +0000 Subject: [PATCH 08/82] test(alloc): verify pooled erasure across read completion fences --- cmd/racer-dataplane/alloc/tests/workflows.rs | 28 ++++++++++++++++++++ 1 file changed, 28 insertions(+) diff --git a/cmd/racer-dataplane/alloc/tests/workflows.rs b/cmd/racer-dataplane/alloc/tests/workflows.rs index 8b8a97c2d..565a17a4e 100644 --- a/cmd/racer-dataplane/alloc/tests/workflows.rs +++ b/cmd/racer-dataplane/alloc/tests/workflows.rs @@ -490,11 +490,18 @@ fn roundtrip_short_io_and_runtime_errors() { let buffer = drive(&reactor, read()).unwrap(); assert!(buffer.as_slice().iter().all(|b| *b == 42)); drop(buffer); + // Kernel-written bytes must dirty an initially clean pool allocation. + let reused = slab.allocate(extent.length(), ()).unwrap(); + assert!(reused.as_slice().iter().all(|b| *b == 0)); + drop(reused); sim.inject("read", Fault::Short(512)).unwrap(); assert!(matches!( drive(&reactor, read()), Err(TestError::Alloc(Error::Io)) )); + let reused = slab.allocate(extent.length(), ()).unwrap(); + assert!(reused.as_slice().iter().all(|b| *b == 0)); + drop(reused); sim.inject("write", Fault::Short(512)).unwrap(); let op = slab.write( &reactor, @@ -588,6 +595,23 @@ fn abandoned_read_and_unpolled_write_release_at_the_correct_fence() { drop(op); assert_eq!(slab.writes_in_flight(), 0); assert_eq!(reactor.in_flight(), 0); + // Seed nonzero bytes so a completed but abandoned read tests erasure, not + // merely the lifetime of an allocation that happens to remain zero. + let mut buffer = slab.allocate(4096, ()).unwrap(); + buffer.as_mut_slice().fill(42); + drop( + drive( + &reactor, + slab.write( + &reactor, + extent, + buffer, + segments.lease(SegmentId(0), Generation(1)).unwrap(), + &TestScope, + ), + ) + .unwrap(), + ); sim.inject("read", Fault::HoldCompletion(8)).unwrap(); let mut op = slab.read( &reactor, @@ -599,9 +623,13 @@ fn abandoned_read_and_unpolled_write_release_at_the_correct_fence() { assert!(poll(&mut op).is_pending()); reactor.poll_budgeted(1).unwrap(); drop(op); + assert_eq!(slab.idle_bytes(), 0); + assert_eq!(slab.reclaim_idle(), 0); segments.begin_evict(SegmentId(0)).unwrap(); assert_eq!(segments.recycle(SegmentId(0)), Err(Error::Busy)); drive(&reactor, reactor.drain()).unwrap(); + let reused = slab.allocate(4096, ()).unwrap(); + assert!(reused.as_slice().iter().all(|b| *b == 0)); segments.recycle(SegmentId(0)).unwrap(); } From 45499b1457a387dd83bb3dac0f1be4348ec91494 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Tue, 6 Oct 2026 11:43:38 -0500 Subject: [PATCH 09/82] Refactor registration handling in slab.rs Co-authored-by: Copilot Autofix powered by AI <175728472+Copilot@users.noreply.github.com> --- cmd/racer-dataplane/alloc/src/slab.rs | 22 +++++++++++++++------- 1 file changed, 15 insertions(+), 7 deletions(-) diff --git a/cmd/racer-dataplane/alloc/src/slab.rs b/cmd/racer-dataplane/alloc/src/slab.rs index 6bfbe97c1..bf801aaf4 100644 --- a/cmd/racer-dataplane/alloc/src/slab.rs +++ b/cmd/racer-dataplane/alloc/src/slab.rs @@ -548,16 +548,24 @@ impl Future for FenceWaiter { self.unregister(); return Poll::Ready(Ok(())); } - if let Some(registration) = &self.registration { - registration.borrow_mut().clone_from(cx.waker()); - // Final completion drained registrations; a new write can precede - // our next poll, in which case this waiter must register again. + let waker = cx.waker().clone(); + if self.state.count.get() == 0 { + self.unregister(); + return Poll::Ready(Ok(())); + } + if let Some(registration) = self.registration.clone() { + let old = registration.borrow_mut().replace(waker); + drop(old); + if self.state.count.get() == 0 { + self.unregister(); + return Poll::Ready(Ok(())); + } let mut waiters = self.state.waiters.borrow_mut(); - if !waiters.iter().any(|w| Rc::ptr_eq(w, registration)) { - waiters.push(registration.clone()); + if !waiters.iter().any(|w| Rc::ptr_eq(w, ®istration)) { + waiters.push(registration); } } else { - let registration = Rc::new(RefCell::new(cx.waker().clone())); + let registration = Rc::new(RefCell::new(waker)); self.state.waiters.borrow_mut().push(registration.clone()); self.registration = Some(registration); } From 6f6ad843ad656bc84c44ea5837342ba282d304a2 Mon Sep 17 00:00:00 2001 From: "copilot-swe-agent[bot]" <198982749+Copilot@users.noreply.github.com> Date: Tue, 6 Oct 2026 17:18:13 +0000 Subject: [PATCH 10/82] fix(alloc): replace registered waker through RefCell Co-authored-by: jveski <7576912+jveski@users.noreply.github.com> --- cmd/racer-dataplane/alloc/src/slab.rs | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/cmd/racer-dataplane/alloc/src/slab.rs b/cmd/racer-dataplane/alloc/src/slab.rs index bf801aaf4..f0355f7a4 100644 --- a/cmd/racer-dataplane/alloc/src/slab.rs +++ b/cmd/racer-dataplane/alloc/src/slab.rs @@ -554,7 +554,7 @@ impl Future for FenceWaiter { return Poll::Ready(Ok(())); } if let Some(registration) = self.registration.clone() { - let old = registration.borrow_mut().replace(waker); + let old = registration.replace(waker); drop(old); if self.state.count.get() == 0 { self.unregister(); From cc5c11396b26f4afff99af2f4de2a020afacdd77 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Tue, 6 Oct 2026 17:42:01 +0000 Subject: [PATCH 11/82] test(alloc): validate alignment without kernel fallback assumptions --- cmd/racer-dataplane/alloc/src/slab.rs | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/cmd/racer-dataplane/alloc/src/slab.rs b/cmd/racer-dataplane/alloc/src/slab.rs index f0355f7a4..677d2f536 100644 --- a/cmd/racer-dataplane/alloc/src/slab.rs +++ b/cmd/racer-dataplane/alloc/src/slab.rs @@ -945,9 +945,11 @@ mod tests { ); assert_eq!(&buffer.bytes().unwrap()[..31], &[42; 31]); assert!(buffer.bytes().unwrap()[31..].iter().all(|b| *b == 0)); + // Filesystems may accept unaligned O_DIRECT I/O via buffered fallback. + // The allocator must reject it before submission regardless of the kernel. assert_eq!( - unsafe { libc::pwrite(fd, buffer.bytes().unwrap().as_ptr().cast(), 31, 1) }, - -1 + alignment.check(Extent::new(1, 31).unwrap(), &buffer), + Err(Error::InvalidConfiguration) ); drop(buffer); assert_eq!(slabs.idle_bytes(), extent.length()); From 017b2166aa3fff53e244c90f1680e657bd168f8f Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Tue, 6 Oct 2026 17:40:03 +0000 Subject: [PATCH 12/82] fix(alloc): wake slab fences outside registration borrows --- cmd/racer-dataplane/alloc/src/slab.rs | 171 +++++++++++++++++++++++++- 1 file changed, 167 insertions(+), 4 deletions(-) diff --git a/cmd/racer-dataplane/alloc/src/slab.rs b/cmd/racer-dataplane/alloc/src/slab.rs index 677d2f536..5a1832aac 100644 --- a/cmd/racer-dataplane/alloc/src/slab.rs +++ b/cmd/racer-dataplane/alloc/src/slab.rs @@ -520,7 +520,7 @@ impl Submission { struct WriteState { count: Cell, - waiters: RefCell>>>, + waiters: RefCell>>>>, } impl WriteState { @@ -536,7 +536,7 @@ impl WriteState { struct FenceWaiter { state: Rc, - registration: Option>>, + registration: Option>>>, } impl Future for FenceWaiter { @@ -548,7 +548,7 @@ impl Future for FenceWaiter { self.unregister(); return Poll::Ready(Ok(())); } - let waker = cx.waker().clone(); + let waker = Rc::new(cx.waker().clone()); if self.state.count.get() == 0 { self.unregister(); return Poll::Ready(Ok(())); @@ -603,8 +603,9 @@ impl Drop for WriteFence { if remaining == 0 { let waiters = std::mem::take(&mut *self.0.waiters.borrow_mut()); for waiter in waiters { + // Snapshot ownership without invoking the raw waker's clone callback. let waker = waiter.borrow().clone(); - waker.wake(); + waker.wake_by_ref(); } } } @@ -1024,6 +1025,167 @@ mod tests { ); } + /// Select the callback that runs a one-shot reentrant test action. + #[derive(Clone, Copy, PartialEq)] + enum FenceCallback { + Clone, + Drop, + /// Reenter on completion's first clone or wake, with either implementation. + Notify, + } + + /// One action and the waker callback that should run it. + type FenceAction = (FenceCallback, Box); + + thread_local! { + static FENCE_CALLBACK: RefCell> = + RefCell::new(None); + static FENCE_WAKES: Cell = const { Cell::new(0) }; + } + + /// Run caller code after releasing the callback registry's borrow. + fn run_fence_callback(event: FenceCallback) { + let callback = FENCE_CALLBACK.with(|slot| { + let mut slot = slot.borrow_mut(); + if slot.as_ref().is_some_and(|(on, _)| { + *on == event || (*on == FenceCallback::Notify && event == FenceCallback::Clone) + }) { + slot.take() + } else { + None + } + }); + if let Some((_, callback)) = callback { + callback(); + } + } + + /// A stateless waker accesses only the calling thread's test registry. + fn fence_callback_waker() -> Waker { + use std::task::{RawWaker, RawWakerVTable}; + + /// Cloning owns no data but may run the thread's registered action. + unsafe fn clone(_: *const ()) -> RawWaker { + run_fence_callback(FenceCallback::Clone); + RawWaker::new(std::ptr::null(), &VTABLE) + } + + /// Both wake forms count a notification and may run the test action. + unsafe fn wake(_: *const ()) { + FENCE_WAKES.with(|count| count.set(count.get() + 1)); + run_fence_callback(FenceCallback::Notify); + } + + /// Dropping owns no data but may run the thread's registered action. + unsafe fn drop_raw(_: *const ()) { + run_fence_callback(FenceCallback::Drop); + } + + static VTABLE: RawWakerVTable = RawWakerVTable::new(clone, wake, wake, drop_raw); + + // SAFETY: there is no raw data or shared ownership. Every callback accesses + // only its calling thread's registry, even if the Waker moves across threads. + unsafe { Waker::from_raw(RawWaker::new(std::ptr::null(), &VTABLE)) } + } + + /// Completion during clone or replacement drop is observed before returning. + #[test] + fn fence_poll_handles_reentrant_clone_and_drop() { + for event in [FenceCallback::Clone, FenceCallback::Drop] { + for new_write in [false, true] { + let state = Rc::new(WriteState::default()); + let write = state.acquire().unwrap(); + let mut waiter = FenceWaiter { + state: state.clone(), + registration: None, + }; + let waker = fence_callback_waker(); + let mut cx = Context::from_waker(&waker); + if event == FenceCallback::Drop { + assert!(Pin::new(&mut waiter).poll(&mut cx).is_pending()); + } + let next = Rc::new(RefCell::new(None)); + let saved_next = next.clone(); + let saved_state = state.clone(); + FENCE_CALLBACK.with(|slot| { + *slot.borrow_mut() = Some(( + event, + Box::new(move || { + drop(write); + if new_write { + *saved_next.borrow_mut() = Some(saved_state.acquire().unwrap()); + } + }), + )); + }); + let result = Pin::new(&mut waiter).poll(&mut cx); + FENCE_CALLBACK.with(|slot| assert!(slot.borrow().is_none())); + if new_write { + assert!(result.is_pending()); + assert_eq!(state.count.get(), 1); + assert_eq!(state.waiters.borrow().len(), 1); + let before = FENCE_WAKES.get(); + drop(next.borrow_mut().take()); + assert_eq!(FENCE_WAKES.get(), before + 1); + assert_eq!(Pin::new(&mut waiter).poll(&mut cx), Poll::Ready(Ok(()))); + } else { + assert_eq!(result, Poll::Ready(Ok(()))); + } + assert_eq!(state.count.get(), 0); + assert!(state.waiters.borrow().is_empty()); + assert!(waiter.registration.is_none()); + } + } + } + + /// Completion callbacks can accept a write and repoll a drained registration. + #[test] + fn fence_completion_allows_reentrant_registration() { + let state = Rc::new(WriteState::default()); + let write = state.acquire().unwrap(); + let waiter = Rc::new(RefCell::new(FenceWaiter { + state: state.clone(), + registration: None, + })); + let waker = fence_callback_waker(); + assert!( + Pin::new(&mut *waiter.borrow_mut()) + .poll(&mut Context::from_waker(&waker)) + .is_pending() + ); + let next = Rc::new(RefCell::new(None)); + let saved_next = next.clone(); + let saved_state = state.clone(); + let saved_waiter = waiter.clone(); + FENCE_CALLBACK.with(|slot| { + *slot.borrow_mut() = Some(( + FenceCallback::Notify, + Box::new(move || { + *saved_next.borrow_mut() = Some(saved_state.acquire().unwrap()); + let waker = fence_callback_waker(); + assert!( + Pin::new(&mut *saved_waiter.borrow_mut()) + .poll(&mut Context::from_waker(&waker)) + .is_pending() + ); + }), + )); + }); + drop(write); + FENCE_CALLBACK.with(|slot| assert!(slot.borrow().is_none())); + assert_eq!(state.waiters.borrow().len(), 1); + assert_eq!(state.count.get(), 1); + let before = FENCE_WAKES.get(); + drop(next.borrow_mut().take()); + assert_eq!(FENCE_WAKES.get(), before + 1); + assert_eq!( + Pin::new(&mut *waiter.borrow_mut()).poll(&mut Context::from_waker(&waker)), + Poll::Ready(Ok(())) + ); + assert!(state.waiters.borrow().is_empty()); + assert!(waiter.borrow().registration.is_none()); + } + /// Waiters sleep, replace wakers, unregister on cancellation, and register again. #[test] fn fence_waiters_sleep_update_wakers_and_unregister_on_cancellation() { @@ -1068,6 +1230,7 @@ mod tests { assert_eq!(slab.writes.waiters.borrow().len(), 3); drop(abandoned); assert_eq!(slab.writes.waiters.borrow().len(), 2); + assert_eq!(Arc::strong_count(&cancelled), 1); drop(first_write); for count in [&old, ¤t, &other, &cancelled] { assert_eq!(count.0.load(Ordering::Relaxed), 0); From fb63bb899f135df03d3f0d72f6be37f4c1465911 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 15:55:27 +0000 Subject: [PATCH 13/82] docs(racer): add page allocator design --- designs/racer-page-alloc.md | 158 ++++++++++++++++++++++++++++++++++++ 1 file changed, 158 insertions(+) create mode 100644 designs/racer-page-alloc.md diff --git a/designs/racer-page-alloc.md b/designs/racer-page-alloc.md new file mode 100644 index 000000000..84374e628 --- /dev/null +++ b/designs/racer-page-alloc.md @@ -0,0 +1,158 @@ +# Racer page allocator (`page-alloc`) + +## Summary + +`page-alloc` is the storage layer under the new Racer dataplane. It lives in +`cmd/racer-dataplane/alloc/`. It gives one worker thread three things: + +1. Memory buffers that are aligned for direct I/O and wiped before reuse. +2. A table of fixed-size segments inside one sparse cache file, with leases + that stop a segment from being reused while I/O still points at it. +3. Async read and write of that file through the `uring-runtime` reactor. + +The crate stores bytes only. It does not know about page keys, record headers, +encryption, checksums, or versions. The caller owns the page index and +tells the crate how to remove entries when a segment is evicted. No Racer +binary uses the crate yet. The dataplane executable is still a placeholder. + +## Goals and non-goals + +Goals: + +- Never reuse disk space or memory while the kernel may still read or write it. +- Keep every allocation and reuse decision on one worker, with no locks. +- Bound the work done by each reclaim call, so the hot path never stalls. +- Fail with an error instead of wrapping counters or truncating files. + +Non-goals: + +- Durability. Writes are not followed by `fsync`, and there is no log. +- Secure erasure of the file. Eviction changes metadata only. Old bytes stay on + disk until they are overwritten. +- Compaction, record integrity, or deciding when a page is published. + +## Threading model + +All state is worker-local. Types use `Rc`, `Cell`, and `RefCell`, so none of them +are `Send` or `Sync` (`alloc/src/segments.rs:153-179`, `alloc/src/slab.rs:35-50`). +Each worker owns its own file, segment table, and buffer pool. This removes lock +contention and makes ownership easy to reason about. The cost is that one worker +cannot hand its storage to another. + +## Buffers + +`AlignedBuffer` is a heap allocation from `alloc_zeroed` with the alignment that +the file needs (`alloc/src/lib.rs:415-442`). Lengths are padded to the least common +multiple of the offset and length units, so the next record also starts aligned. +A single buffer is at most 1 GiB. + +Each buffer holds a caller-supplied `Charge`, so the caller can account for +memory against its own budget. The slab keeps at most one idle buffer of the +last size used (`alloc/src/slab.rs:398-415`). It is not a general size-class pool. + +Each buffer tracks whether it is still all zeros. Any mutable access, including +handing it to the kernel for a read, marks it dirty. On drop, a dirty buffer is +wiped in full, including padding, with `explicit_bzero` (or `zeroize` where that +is not available) before it is pooled or freed (`alloc/src/lib.rs:483-520`). +Clean buffers skip the wipe. This keeps old page data from leaking into the +next request without paying for a wipe on every allocation. + +## Segments + +The file is split into fixed-size segments. Segment `n` starts at +`n * segment_bytes`. Each segment has a state and a generation number: + +``` +Free -> Open -> Sealed -> Evicting -> Free (generation + 1) +``` + +- `append` reserves space at the end of the one open segment. When it is full, + the segment is sealed and the lowest free segment is opened. If none is free, + the call returns `Busy` after sealing, so reclaim can make room. +- `append` and `lease` return a `SegmentLease`. A lease records the segment ID, + generation, and how much of the segment was in use when it was taken. While + any lease exists, the segment cannot go back to `Free`. +- The caller stores `(segment, generation, extent)` in its own index. A lookup + with an old generation fails with `Stale`, so a reused segment can never be + read as if it still held the old page. + +`freeze`, `snapshot`, and `restore` support restart. Restore checks the whole +image before it changes anything, requires that no leases are live, and seals +any segment that was open before the restart. + +## Reclaim + +`SegmentClock` is a bounded clock (second-chance) sweep over sealed segments +(`alloc/src/segments.rs:785-842`). The caller calls `mark_read` when it serves +a read from a segment. The sweep clears that mark once before it picks the +segment. Each call has limits on +the number of segments it visits and the number of index entries it removes. If it +cannot free enough space within those limits, it returns `Busy` and the caller +tries again later. + +Evicting a segment happens in this order: + +1. Ask the caller `can_evict`. The caller says no while an unpublished write + still needs the segment. This check happens only before eviction starts. +2. Mark the segment `Evicting`. New leases are refused. +3. Call `remove_bounded` to drop the caller's index entries, a few per call. +4. When the index is empty and all leases are gone, bump the generation and + mark the segment `Free`. + +The key rule is: remove the index entries first, then wait for in-flight I/O, +then reuse. `reclaim_scored` is a variant that ranks a small sample by a +caller-provided score instead of recent reads. `reclaim_index` drops index entries one at a time until a caller check +(for example, an index size limit) passes. It does not free segments and does not +ask `can_evict`. + +## File and I/O + +`Slab` owns one cache file. Opening is blocking and is meant to run at startup +(`alloc/src/slab.rs:183-279`). It: + +- Walks the path without following symlinks or `..`. +- Requires a regular file owned by the current user, mode 0600, one hard link. +- Takes an exclusive non-blocking `flock` and enables `O_DIRECT`. +- Reads the required alignment from `statx(STATX_DIOALIGN)`. +- Sizes an empty file sparsely to capacity. A file with the wrong size is + rejected, not truncated. + +The slab is then bound to one segment table. I/O is refused until binding +succeeds, and a slab cannot be rebound to a different table. + +`read` and `write` check the extent, alignment, and lease, then pass the buffer +and lease to the reactor (`alloc/src/slab.rs:461-515`). The reactor holds both +until the kernel reports completion, even if the caller drops the future or +cancels. So the memory and the segment both stay reserved until the kernel is +done with them. Short reads and writes are returned as errors. + +`fence_writes` waits until no write is in flight. It is a count, not a snapshot, +and it does not flush to disk. + +## Simulation + +The `simulation` feature routes file open, lock, stat, and sizing to the +`uring-runtime` simulated filesystem. Buffers, leases, and segment rules do not +change, so the same workflow tests run in both modes. + +## Testing + +- Unit tests cover padding math, wipe and reuse rules, lease and generation + checks, reclaim limits, the eviction veto, restore, and file security checks. +- `alloc/tests/workflows.rs` covers restart, short I/O, dropped and cancelled + reads and writes, startup failures, and confirms that reuse does not erase + disk bytes. +- CI checks and tests the production and simulation builds as separate + commands, so feature unification cannot hide one from the other. The + production run sets `PAGE_ALLOC_REQUIRE_REAL_IO=1`, so a host without + io_uring or direct I/O fails instead of skipping. Miri runs only on + `uring-runtime`, not on this crate. + +## Open questions + +- Should eviction punch holes or overwrite freed segments for hosts that need + data removed from disk? +- Is one idle buffer enough once real traffic mixes page sizes, or does the + pool need size classes? +- Should the crate run under Miri, at least for the pure buffer and segment + tests? From fc3d83cc87f102f933cdb08c9dbc83816648d07a Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 15:57:21 +0000 Subject: [PATCH 14/82] docs(racer): drop open questions from page allocator design --- designs/racer-page-alloc.md | 9 --------- 1 file changed, 9 deletions(-) diff --git a/designs/racer-page-alloc.md b/designs/racer-page-alloc.md index 84374e628..56c65cde4 100644 --- a/designs/racer-page-alloc.md +++ b/designs/racer-page-alloc.md @@ -147,12 +147,3 @@ change, so the same workflow tests run in both modes. production run sets `PAGE_ALLOC_REQUIRE_REAL_IO=1`, so a host without io_uring or direct I/O fails instead of skipping. Miri runs only on `uring-runtime`, not on this crate. - -## Open questions - -- Should eviction punch holes or overwrite freed segments for hosts that need - data removed from disk? -- Is one idle buffer enough once real traffic mixes page sizes, or does the - pool need size classes? -- Should the crate run under Miri, at least for the pure buffer and segment - tests? From 1e5f44a710ef9db692230d2428cd2fb19ad0c510 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Mon, 5 Oct 2026 22:35:11 +0000 Subject: [PATCH 15/82] feat(racer): extract flow-control library --- cmd/racer-dataplane/Cargo.lock | 11 + cmd/racer-dataplane/Cargo.toml | 2 +- cmd/racer-dataplane/flow/.gitignore | 1 + cmd/racer-dataplane/flow/Cargo.toml | 17 + cmd/racer-dataplane/flow/src/admission.rs | 1171 +++++++++++ cmd/racer-dataplane/flow/src/coalesce.rs | 1953 +++++++++++++++++++ cmd/racer-dataplane/flow/src/lib.rs | 1338 +++++++++++++ cmd/racer-dataplane/flow/src/pipe.rs | 1248 ++++++++++++ cmd/racer-dataplane/flow/tests/coalesce.rs | 414 ++++ cmd/racer-dataplane/flow/tests/contracts.rs | 671 +++++++ 10 files changed, 6825 insertions(+), 1 deletion(-) create mode 100644 cmd/racer-dataplane/flow/.gitignore create mode 100644 cmd/racer-dataplane/flow/Cargo.toml create mode 100644 cmd/racer-dataplane/flow/src/admission.rs create mode 100644 cmd/racer-dataplane/flow/src/coalesce.rs create mode 100644 cmd/racer-dataplane/flow/src/lib.rs create mode 100644 cmd/racer-dataplane/flow/src/pipe.rs create mode 100644 cmd/racer-dataplane/flow/tests/coalesce.rs create mode 100644 cmd/racer-dataplane/flow/tests/contracts.rs diff --git a/cmd/racer-dataplane/Cargo.lock b/cmd/racer-dataplane/Cargo.lock index a650c6bcd..195b8ff2c 100644 --- a/cmd/racer-dataplane/Cargo.lock +++ b/cmd/racer-dataplane/Cargo.lock @@ -52,6 +52,17 @@ dependencies = [ "crypto-common", ] +[[package]] +name = "flow-control" +version = "0.1.0" +dependencies = [ + "futures", + "libc", + "page-alloc", + "uring-runtime", + "zeroize", +] + [[package]] name = "futures" version = "0.3.34" diff --git a/cmd/racer-dataplane/Cargo.toml b/cmd/racer-dataplane/Cargo.toml index a840bccdb..3465b6901 100644 --- a/cmd/racer-dataplane/Cargo.toml +++ b/cmd/racer-dataplane/Cargo.toml @@ -1,5 +1,5 @@ [workspace] -members = [".", "topology", "runtime", "alloc"] +members = [".", "topology", "runtime", "alloc", "flow"] resolver = "3" [package] diff --git a/cmd/racer-dataplane/flow/.gitignore b/cmd/racer-dataplane/flow/.gitignore new file mode 100644 index 000000000..b83d22266 --- /dev/null +++ b/cmd/racer-dataplane/flow/.gitignore @@ -0,0 +1 @@ +/target/ diff --git a/cmd/racer-dataplane/flow/Cargo.toml b/cmd/racer-dataplane/flow/Cargo.toml new file mode 100644 index 000000000..6794e87d0 --- /dev/null +++ b/cmd/racer-dataplane/flow/Cargo.toml @@ -0,0 +1,17 @@ +[package] +name = "flow-control" +version = "0.1.0" +edition = "2024" +license = "Apache-2.0" +publish = false +description = "Policy-driven quotas, keyed coalescing, completion-owned flights, and bounded I/O resources" + +[features] +simulation = ["uring-runtime/simulation"] + +[dependencies] +futures = "0.3" +libc = "0.2" +zeroize = "1" +page-alloc = { path = "../alloc" } +uring-runtime = { path = "../runtime" } diff --git a/cmd/racer-dataplane/flow/src/admission.rs b/cmd/racer-dataplane/flow/src/admission.rs new file mode 100644 index 000000000..89145aa01 --- /dev/null +++ b/cmd/racer-dataplane/flow/src/admission.rs @@ -0,0 +1,1171 @@ +//! Distinct admission mechanisms for peer recovery, handoff, and speculation. +//! +//! Adaptive permits retain shared capacity until their last owner leaves. Endpoint +//! circuits instead own local half-open probes, whose exclusivity survives timeout +//! and successful observation. Handoff envelopes retain target reservations while +//! queued and after dequeue. Hedge alarms never release their capacity: owners +//! retain permits through both contenders' completion fences. +use crate::{Error, Result}; +use std::{ + collections::BTreeMap, + sync::{Arc, Mutex}, + time::{Duration, Instant}, +}; + +pub use circuit::{Circuits, Probe}; +pub use handoff::{Admission as HandoffAdmission, Admitted, Handoff, Offer}; +pub use hedge::{Hedges, Permit as HedgePermit}; + +/// Limits and time intervals for shared adaptive admission. +#[derive(Clone, Copy)] +pub struct Config { + /// Maximum aggregate active operations before adaptive reductions. + pub total: usize, + + /// Maximum active operations for one key before failure reductions. + pub per_key: usize, + + /// Maximum retained peer records, including idle backoff history. + pub capacity: usize, + + /// Peer retry delay and minimum interval between aggregate pressure reductions. + pub backoff: Duration, + + /// Minimum interval between one-slot verified-success recoveries. + pub recovery: Duration, + + /// Minimum idle age before an eligible peer record can be replaced. + pub retire_after: Duration, +} + +/// Caller-classified evidence, independent of application errors. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum Outcome { + /// Authenticated successful work provides recovery evidence. + Verified, + + /// A failure attributable to the immediate peer opens its circuit. + PeerFailure, + + /// Local overload reduces aggregate admission without blaming the peer. + LocalPressure, + + /// No evidence should affect admission or peer recovery. + Neutral, +} + +/// Admission observations emitted synchronously in state-transition order. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum Event { + /// Aggregate, per-key, or record capacity rejected the attempt. + Rejected, + + /// Peer backoff or an owned probe rejected the attempt. + CircuitRejected, + + /// An operation now owns active admission. + Accepted, + + /// An accepted operation owns the exclusive recovery probe. + Probe, + + /// Caller-reported local pressure was observed. + LocalPressure, + + /// Caller-reported verified success was observed. + Verified, + + /// Caller-reported immediate-link failure was observed. + LinkFailure, +} + +/// Callbacks run synchronously under the admission lock and must not reenter it. +pub trait Observer { + /// Record an admission or outcome event without reentering the owner. + fn event(&self, event: Event); + + /// Publish the current active count without reentering the owner. + fn active(&self, active: usize); + + /// Publish the current adaptive limit without reentering the owner. + fn limit(&self, limit: usize); +} + +/// Shared total and per-key admission with generation-fenced peer recovery. +pub struct Adaptive { + config: Config, + + state: Mutex>, + + observer: O, + + now: fn() -> Instant, +} + +/// Mutex-protected aggregate admission and bounded peer records. +struct State { + active: usize, + + limit: usize, + + updated: Instant, + + peers: BTreeMap, +} + +/// Per-key recovery state retained while any permit remains active. +struct Peer { + active: usize, + + limit: usize, + + generation: u64, + + retry: Option, + + probe: bool, + + updated: Instant, +} + +/// An admitted operation; the final shared owner releases capacity. +pub struct Permit { + owner: Arc>, + + key: K, + + generation: u64, + + probe: bool, +} + +impl Adaptive { + /// Validate positive limits and publish the initial aggregate ceiling. + pub fn new(config: Config, observer: O, now: fn() -> Instant) -> Result> { + if config.total == 0 + || config.per_key == 0 + || config.per_key > config.total + || config.capacity == 0 + { + return Err(Error::InvalidInput); + } + observer.limit(config.total); + Ok(Arc::new(Self { + config, + observer, + now, + state: Mutex::new(State { + active: 0, + limit: config.total, + updated: now(), + peers: BTreeMap::new(), + }), + })) + } + + /// Hint whether fully recovered limits leave spare speculative capacity. + pub fn hedge_available(&self, key: &K) -> bool { + self.state.lock().is_ok_and(|s| { + s.limit == self.config.total + && s.active.saturating_add(1) < s.limit + && s.peers.get(key).is_none_or(|p| { + p.retry.is_none() + && !p.probe + && p.limit == self.config.per_key + && p.active < p.limit + }) + }) + } + + /// Hint whether peer backoff permits an attempt; this does not reserve capacity. + pub fn available(&self, key: &K) -> bool { + let now = (self.now)(); + self.state.lock().is_ok_and(|s| { + s.peers + .get(key) + .is_none_or(|p| !p.probe && p.retry.is_none_or(|at| now >= at)) + }) + } + + /// Admit work or one exclusive recovery probe without waiting. + pub fn acquire(self: &Arc, key: &K) -> Result>> { + let now = (self.now)(); + let mut state = self.state.lock().map_err(|_| Error::Unavailable)?; + if state.active >= state.limit { + self.observer.event(Event::Rejected); + return Err(Error::Overloaded); + } + if !state.peers.contains_key(key) && state.peers.len() == self.config.capacity { + let retired = state + .peers + .iter() + .find(|(_, p)| { + p.active == 0 + && !p.probe + && p.retry.is_none_or(|retry| now >= retry) + && now.saturating_duration_since(p.updated) >= self.config.retire_after + }) + .map(|(k, _)| k.clone()); + if let Some(retired) = retired { + state.peers.remove(&retired); + } else { + self.observer.event(Event::Rejected); + return Err(Error::Overloaded); + } + } + let peer = state.peers.entry(key.clone()).or_insert(Peer { + active: 0, + limit: self.config.per_key, + generation: 0, + retry: None, + probe: false, + updated: now, + }); + if peer.probe || peer.retry.is_some_and(|at| now < at) { + self.observer.event(Event::CircuitRejected); + return Err(Error::Unavailable); + } + if peer.active >= peer.limit { + self.observer.event(Event::Rejected); + return Err(Error::Overloaded); + } + let probe = peer.retry.is_some(); + peer.probe = probe; + peer.active += 1; + let generation = peer.generation; + state.active += 1; + self.observer.active(state.active); + self.observer.event(Event::Accepted); + if probe { + self.observer.event(Event::Probe); + } + Ok(Arc::new(Permit { + owner: self.clone(), + key: key.clone(), + generation, + probe, + })) + } +} + +impl Permit { + /// Apply caller evidence while retaining admission through completion. + pub fn observe(&self, outcome: Outcome) { + let now = (self.owner.now)(); + let Ok(mut state) = self.owner.state.lock() else { + return; + }; + let config = self.owner.config; + if outcome == Outcome::LocalPressure { + self.owner.observer.event(Event::LocalPressure); + if now.saturating_duration_since(state.updated) >= config.backoff { + state.limit = (state.limit / 2).max(1); + state.updated = now; + self.owner.observer.limit(state.limit); + } + return; + } + if outcome == Outcome::Verified + && now.saturating_duration_since(state.updated) >= config.recovery + { + state.limit = state.limit.saturating_add(1).min(config.total); + state.updated = now; + self.owner.observer.limit(state.limit); + } + let peer = state + .peers + .get_mut(&self.key) + .expect("live permit retains key"); + let event = match outcome { + Outcome::Verified => Event::Verified, + Outcome::PeerFailure => Event::LinkFailure, + Outcome::LocalPressure => Event::LocalPressure, + Outcome::Neutral => return, + }; + self.owner.observer.event(event); + if peer.generation != self.generation { + return; + } + match outcome { + Outcome::PeerFailure => { + peer.limit = (peer.limit / 2).max(1); + peer.generation = peer.generation.saturating_add(1); + peer.retry = Some(now + config.backoff); + peer.updated = now; + } + Outcome::Verified => { + if peer.retry.is_some() && !self.probe { + return; + } + peer.retry = None; + if now.saturating_duration_since(peer.updated) >= config.recovery { + peer.limit = peer.limit.saturating_add(1).min(config.per_key); + peer.updated = now; + } + } + Outcome::LocalPressure | Outcome::Neutral => {} + } + } +} + +impl Drop for Permit { + /// Release final ownership and renew backoff for an unsuccessful probe. + fn drop(&mut self) { + let Ok(mut state) = self.owner.state.lock() else { + return; + }; + let peer = state + .peers + .get_mut(&self.key) + .expect("live permit retains key"); + peer.active -= 1; + if self.probe { + peer.probe = false; + if peer.retry.is_some() { + let now = (self.owner.now)(); + peer.retry = Some(now + self.owner.config.backoff); + peer.updated = now; + } + } + state.active -= 1; + self.owner.observer.active(state.active); + } +} + +/// Bounded worker-local endpoint failure tracking, independent of adaptive limits. +mod circuit { + use crate::{Error, Result}; + use std::{ + cell::RefCell, + collections::{BTreeMap, BTreeSet}, + marker::PhantomData, + rc::Rc, + time::{Duration, Instant}, + }; + + /// Local endpoint health with bounded records and completion-owned probes. + /// + /// The authority belongs to one worker even when keys themselves are shared: + /// + /// ```compile_fail + /// use flow_control::Circuits; + /// let circuits = Circuits::::new(1, std::time::Duration::ZERO); + /// std::thread::spawn(move || drop(circuits)); + /// ``` + /// + /// ```compile_fail + /// use flow_control::Circuits; + /// fn require_sync() {} + /// require_sync::>(); + /// ``` + pub struct Circuits { + capacity: usize, + + probe_timeout: Duration, + + states: RefCell>, + + probes: RefCell>, + + local: PhantomData>, + } + + /// Failure history and eligibility for one endpoint. + struct Circuit { + failures: u32, + + retry_at: Instant, + + probe_until: Option, + } + + impl Circuit { + /// Require both retry backoff and any abandoned probe timeout to expire. + fn available(&self, now: Instant) -> bool { + now >= self.retry_at && self.probe_until.is_none_or(|until| now >= until) + } + } + + /// Local exclusive half-open ownership; healthy acquisitions need no record. + pub struct Probe<'a, K: Ord> { + health: &'a Circuits, + + key: Option, + } + + impl Drop for Probe<'_, K> { + /// Release exclusivity without erasing the probe's timeout. + fn drop(&mut self) { + if let Some(key) = &self.key { + self.health.probes.borrow_mut().remove(key); + } + } + } + + impl Circuits { + /// Bound failure records and set eligibility delay for abandoned probes. + pub const fn new(capacity: usize, probe_timeout: Duration) -> Self { + Self { + capacity, + probe_timeout, + states: RefCell::new(BTreeMap::new()), + probes: RefCell::new(BTreeSet::new()), + local: PhantomData, + } + } + + /// Acquire an exclusive half-open guard, or an untracked healthy guard. + pub fn acquire(&self, key: &K, now: Instant) -> Result> { + if !self.try_acquire(key, now) { + return Err(Error::Unavailable); + } + let probe = self.states.borrow().contains_key(key); + if probe { + if self.probes.borrow().len() >= self.capacity { + return Err(Error::Overloaded); + } + self.probes.borrow_mut().insert(key.clone()); + } + Ok(Probe { + health: self, + key: probe.then(|| key.clone()), + }) + } + + /// Remove failure state without releasing any owned probe. + pub fn success(&self, key: &K) { + self.states.borrow_mut().remove(key); + } + + /// Record a caller-classified failure using caller-selected retry jitter. + pub fn failure( + &self, + key: &K, + now: Instant, + backoff: impl FnOnce(&K, u32) -> Duration, + ) -> Result<()> { + let mut states = self.states.borrow_mut(); + if !states.contains_key(key) && states.len() >= self.capacity { + return Err(Error::Overloaded); + } + let state = states.entry(key.clone()).or_insert(Circuit { + failures: 0, + retry_at: now, + probe_until: None, + }); + state.failures = state.failures.saturating_add(1); + state.retry_at = now + backoff(key, state.failures); + state.probe_until = None; + Ok(()) + } + + /// Routing hint only; actual work must acquire an exclusive probe. + pub fn available(&self, key: &K, now: Instant) -> bool { + !self.probes.borrow().contains(key) + && self + .states + .borrow() + .get(key) + .is_none_or(|s| s.available(now)) + } + + /// Admit an unowned probe that becomes eligible again after its timeout. + pub fn try_acquire(&self, key: &K, now: Instant) -> bool { + if self.probes.borrow().contains(key) { + return false; + } + let mut states = self.states.borrow_mut(); + let Some(state) = states.get_mut(key) else { + return true; + }; + if !state.available(now) { + return false; + } + state.probe_until = Some(now + self.probe_timeout); + true + } + + /// Forget excluded failure records without releasing owned probes. + pub fn retain(&self, keys: &[K]) { + self.states.borrow_mut().retain(|key, _| keys.contains(key)); + } + + /// Count retained failure records, not active probes. + pub fn len(&self) -> usize { + self.states.borrow().len() + } + + /// Whether no failure records remain, independently of owned probes. + pub fn is_empty(&self) -> bool { + self.states.borrow().is_empty() + } + } + + /// Endpoint state transition and ownership contracts. + #[cfg(test)] + mod tests { + use super::*; + + /// Keep an owned probe exclusive through success and retention changes. + #[test] + fn bounded_backoff_and_owned_probe_survive_retention_and_success() { + let now = Instant::now(); + let health = Circuits::new(1, Duration::from_secs(1)); + assert!(health.try_acquire(&7, now)); + health + .failure(&7, now, |key, failures| { + assert_eq!((*key, failures), (7, 1)); + Duration::from_secs(2) + }) + .unwrap(); + assert_eq!( + health.failure(&8, now, |_, _| Duration::ZERO), + Err(Error::Overloaded) + ); + assert!(!health.available(&7, now)); + let retry = now + Duration::from_secs(2); + let probe = health.acquire(&7, retry).unwrap(); + assert!(!health.try_acquire(&7, retry + Duration::from_secs(5))); + health.success(&7); + health.retain(&[]); + assert!(!health.available(&7, retry)); + drop(probe); + assert!(health.available(&7, retry)); + assert!(health.is_empty()); + } + + /// Abandonment preserves timeout and failure history never overflows. + #[test] + fn dropped_probe_preserves_timeout_and_failures_saturate() { + let now = Instant::now(); + let health = Circuits::new(1, Duration::from_secs(1)); + health.failure(&(), now, |_, _| Duration::ZERO).unwrap(); + drop(health.acquire(&(), now).unwrap()); + assert!(!health.try_acquire(&(), now)); + assert!(health.try_acquire(&(), now + Duration::from_secs(1))); + health.states.borrow_mut().get_mut(&()).unwrap().failures = u32::MAX; + health + .failure(&(), now, |_, n| { + assert_eq!(n, u32::MAX); + Duration::ZERO + }) + .unwrap(); + health.retain(&[]); + assert_eq!(health.len(), 0); + assert_eq!( + Circuits::new(0, Duration::ZERO).failure(&(), now, |_, _| Duration::ZERO), + Err(Error::Overloaded) + ); + } + + /// Probe capacity and failure records remain independent after success. + #[test] + fn full_probe_set_preserves_rejected_attempt_timeout() { + let now = Instant::now(); + let timeout = Duration::from_secs(1); + let health = Circuits::new(1, timeout); + health.failure(&7, now, |_, _| Duration::ZERO).unwrap(); + let held = health.acquire(&7, now).unwrap(); + health.success(&7); + assert!(health.is_empty()); + assert!( + health.acquire(&9, now).is_ok(), + "healthy work needs no slot" + ); + + health.failure(&8, now, |_, _| Duration::ZERO).unwrap(); + assert!(health.available(&8, now)); + assert!(matches!(health.acquire(&8, now), Err(Error::Overloaded))); + drop(held); + assert!(!health.available(&8, now)); + assert!(matches!(health.acquire(&8, now), Err(Error::Unavailable))); + assert!(health.acquire(&8, now + timeout).is_ok()); + } + } +} + +/// Round-robin delivery with structurally retained target admission. +mod handoff { + use crate::{Error, Result}; + use std::{ + collections::VecDeque, + sync::{Arc, Mutex}, + task::Waker, + }; + + /// Bounds queued, delivered, and outstanding offered items with reservations. + /// Callbacks execute under the handoff lock and must not reenter it. + pub trait Admission { + /// Ownership that retains capacity until final admitted work completes. + type Reservation; + + /// Register for released capacity before attempting reservation. + fn register(&self, waker: &Waker); + + /// Reserve capacity for one item without waiting. + fn reserve(&self) -> Result; + } + + /// Payload and reservation remain inseparable while queued or freshly popped. + pub struct Admitted { + item: T, + + reservation: R, + } + + /// Fixed batch whose occupied entries each retain their target reservation. + type Batch = [Option>; N]; + + impl Admitted { + /// Transfer both owners together to the application's completion owner. + pub fn into_parts(self) -> (T, R) { + (self.item, self.reservation) + } + } + + /// One target's installed admission, inbox, and wake registration. + struct Target { + admission: Option, + + queue: VecDeque>, + + waker: Option, + + closed: bool, + } + + /// Stable target slots and the next round-robin scan position. + struct State { + targets: Vec<(K, Target)>, + + cursor: usize, + } + + impl State { + /// Find a stable target without changing its admission or closed state. + fn target(&mut self, key: &K) -> Option<&mut Target> { + self.targets + .iter_mut() + .find(|(candidate, _)| candidate == key) + .map(|(_, target)| target) + } + } + + /// Shared handoff bounded by admission, not a separate queue-slot limit. + pub struct Handoff(Mutex>); + + /// Capacity reserved on a selected target before constructing its payload. + pub struct Offer { + handoff: Arc>, + + target: usize, + + reservation: A::Reservation, + } + + impl Handoff { + /// Create stable target slots; empty target lists simply reject offers. + pub fn new(keys: &[K]) -> Self { + Self(Mutex::new(State { + targets: keys + .iter() + .map(|key| { + ( + key.clone(), + Target { + admission: None, + queue: VecDeque::new(), + waker: None, + closed: false, + }, + ) + }) + .collect(), + cursor: 0, + })) + } + + /// Install admission once on a known target that has not closed. + pub fn install(&self, key: &K, admission: A) -> Result<()> { + let mut state = self.0.lock().map_err(|_| Error::Unavailable)?; + let target = state.target(key).ok_or(Error::InvalidInput)?; + if target.admission.is_some() || target.closed { + return Err(Error::InvalidInput); + } + target.admission = Some(admission); + Ok(()) + } + + /// Scan open targets fairly and reserve before returning an offer. + pub fn reserve(self: &Arc, waker: &Waker) -> Result> { + let mut state = self.0.lock().map_err(|_| Error::Unavailable)?; + for offset in 0..state.targets.len() { + let index = (state.cursor + offset) % state.targets.len(); + let (_, target) = &state.targets[index]; + if target.closed { + continue; + } + let Some(admission) = &target.admission else { + continue; + }; + admission.register(waker); + if let Ok(reservation) = admission.reserve() { + state.cursor = (index + 1) % state.targets.len(); + return Ok(Offer { + handoff: self.clone(), + target: index, + reservation, + }); + } + } + Err(Error::Overloaded) + } + + /// Pop at most the caller's budget while retaining each item's admission. + pub fn pop_batch( + &self, + key: &K, + waker: &Waker, + budget: usize, + ) -> Result> { + let mut state = self.0.lock().map_err(|_| Error::Unavailable)?; + let target = state.target(key).ok_or(Error::InvalidInput)?; + if let Some(old) = &mut target.waker { + old.clone_from(waker); + } else { + target.waker = Some(waker.clone()); + } + Ok(std::array::from_fn(|index| { + if index < budget { + target.queue.pop_front() + } else { + None + } + })) + } + + /// Close a target and drop queued ownership outside the shared lock. + pub fn close(&self, key: &K) { + let queued = { + let mut state = self.0.lock().unwrap_or_else(|e| e.into_inner()); + let Some(target) = state.target(key) else { + return; + }; + target.closed = true; + std::mem::take(&mut target.queue) + }; + drop(queued); + } + } + + impl Offer { + /// Build only on an open target; the envelope retains admission itself. + /// The closure executes under the handoff lock and must not reenter it. + pub fn deliver(self, item: impl FnOnce() -> T) -> Result<()> { + let mut state = self.handoff.0.lock().map_err(|_| Error::Unavailable)?; + let target = &mut state.targets[self.target].1; + if target.closed { + return Err(Error::Unavailable); + } + target.queue.push_back(Admitted { + item: item(), + reservation: self.reservation, + }); + let waker = target.waker.clone(); + drop(state); + if let Some(waker) = waker { + waker.wake(); + } + Ok(()) + } + } +} + +/// Shared speculative capacity and explicitly polled due-time registrations. +mod hedge { + use crate::{Error, Result}; + use std::{ + collections::BTreeMap, + sync::{Arc, Mutex}, + task::{Context, Poll, Waker}, + time::Instant, + }; + + /// One admitted speculation and its latest wake registration. + struct Alarm { + due: Instant, + + cost: usize, + + wake: Option, + } + + /// Mutex-protected capacity and monotonically assigned alarm identities. + #[derive(Default)] + struct State { + next: u64, + + used: usize, + + alarms: BTreeMap, + } + + /// Process-shared slot and cost limits, independent of worker registries. + pub struct Hedges { + slots: usize, + + capacity: usize, + + state: Mutex, + } + + /// Owns speculative capacity until both contenders reach their fences. + #[must_use = "retain the permit until both contenders are fenced"] + pub struct Permit { + owner: Arc, + + id: u64, + } + + impl Hedges { + /// Create shared ceilings; zero slots disable speculative admission. + pub fn new(slots: usize, capacity: usize) -> Arc { + Arc::new(Self { + slots, + capacity, + state: Mutex::new(State::default()), + }) + } + + /// Reserve one slot and caller-selected cost until permit drop. + pub fn acquire(self: &Arc, cost: usize, due: Instant) -> Result { + let mut state = self.state.lock().map_err(|_| Error::Unavailable)?; + if state.alarms.len() >= self.slots || cost > self.capacity - state.used { + return Err(Error::Overloaded); + } + let id = state.next.checked_add(1).ok_or(Error::Unavailable)?; + state.next = id; + state.used += cost; + state.alarms.insert( + id, + Alarm { + due, + cost, + wake: None, + }, + ); + Ok(Permit { + owner: self.clone(), + id, + }) + } + + /// Wake due registrations outside the shared lock, once per registration. + pub fn poll(&self, now: Instant) { + let wakes: Vec<_> = self + .state + .lock() + .map(|mut state| { + state + .alarms + .values_mut() + .filter(|a| now >= a.due) + .filter_map(|a| a.wake.take()) + .collect() + }) + .unwrap_or_default(); + for wake in wakes { + wake.wake(); + } + } + } + + impl Permit { + /// Register the latest waiter until due, without releasing capacity. + pub fn delay(&self, now: Instant, cx: &mut Context<'_>) -> Poll<()> { + let mut state = self.owner.state.lock().expect("hedge alarm lock"); + let alarm = state.alarms.get_mut(&self.id).expect("live hedge alarm"); + if now >= alarm.due { + Poll::Ready(()) + } else { + alarm.wake = Some(cx.waker().clone()); + Poll::Pending + } + } + } + + impl Drop for Permit { + /// Remove the alarm and return its cost only at final ownership release. + fn drop(&mut self) { + if let Ok(mut state) = self.owner.state.lock() + && let Some(alarm) = state.alarms.remove(&self.id) + { + state.used -= alarm.cost; + } + } + } + + /// Shared capacity and alarm lifecycle contracts. + #[cfg(test)] + mod tests { + use super::*; + use std::{ + sync::atomic::{AtomicUsize, Ordering}, + task::Wake, + time::Duration, + }; + + /// Count alarm notifications across threads. + #[derive(Default)] + struct Counter(AtomicUsize); + + impl Wake for Counter { + /// Record one notification. + fn wake(self: Arc) { + self.0.fetch_add(1, Ordering::SeqCst); + } + } + + /// Slots and bytes stay charged after an alarm fires and across threads. + #[test] + fn hedge_shared_slots_costs_and_release_are_independent_of_alarm() { + /// Require shared ownership at compile time. + fn shared() {} + shared::(); + shared::(); + let now = Instant::now(); + let owner = Hedges::new(2, 10); + let a = owner.acquire(7, now).unwrap(); + assert!(matches!(owner.acquire(4, now), Err(Error::Overloaded))); + let b = + std::thread::scope(|s| s.spawn(|| owner.acquire(3, now)).join().unwrap().unwrap()); + assert!(matches!(owner.acquire(0, now), Err(Error::Overloaded))); + owner.poll(now); + assert!(matches!(owner.acquire(1, now), Err(Error::Overloaded))); + drop(a); + let c = owner.acquire(7, now).unwrap(); + drop((b, c)); + assert!(owner.acquire(10, now).is_ok()); + assert!(matches!( + Hedges::new(0, 10).acquire(0, now), + Err(Error::Overloaded) + )); + let max = Hedges::new(2, usize::MAX); + let _all = max.acquire(usize::MAX, now).unwrap(); + assert!(matches!(max.acquire(1, now), Err(Error::Overloaded))); + } + + /// Only the latest registered waker fires and drop cancels notification. + #[test] + fn hedge_alarm_replaces_waker_and_drop_removes_registration() { + let now = Instant::now(); + let due = now + Duration::from_secs(1); + let owner = Hedges::new(1, 1); + let permit = owner.acquire(1, due).unwrap(); + let old = Arc::new(Counter::default()); + let current = Arc::new(Counter::default()); + assert!( + permit + .delay(now, &mut Context::from_waker(&Waker::from(old.clone()))) + .is_pending() + ); + assert!( + permit + .delay(now, &mut Context::from_waker(&Waker::from(current.clone()))) + .is_pending() + ); + owner.poll(now); + assert_eq!(current.0.load(Ordering::SeqCst), 0); + owner.poll(due); + owner.poll(due); + assert_eq!(old.0.load(Ordering::SeqCst), 0); + assert_eq!(current.0.load(Ordering::SeqCst), 1); + assert!( + permit + .delay(due, &mut Context::from_waker(Waker::noop())) + .is_ready() + ); + drop(permit); + let permit = owner.acquire(1, due).unwrap(); + assert!( + permit + .delay(now, &mut Context::from_waker(&Waker::from(current.clone()))) + .is_pending() + ); + drop(permit); + owner.poll(due); + assert_eq!(current.0.load(Ordering::SeqCst), 1); + } + } +} + +/// Adaptive state transitions, generation fencing, and shared ownership contracts. +#[cfg(test)] +mod tests { + use super::*; + + /// Synchronous observer recording emitted events and gauges. + #[derive(Default)] + struct Counts { + events: Mutex>, + + active: Mutex, + + limit: Mutex, + } + + impl Observer for Counts { + /// Append an event in emission order. + fn event(&self, event: Event) { + self.events.lock().unwrap().push(event); + } + + /// Save the latest active-work gauge. + fn active(&self, active: usize) { + *self.active.lock().unwrap() = active; + } + + /// Save the latest admission-limit gauge. + fn limit(&self, limit: usize) { + *self.limit.lock().unwrap() = limit; + } + } + + /// Return small limits with independent backoff and recovery intervals. + fn config() -> Config { + Config { + total: 4, + per_key: 4, + capacity: 2, + backoff: Duration::from_millis(250), + recovery: Duration::from_secs(1), + retire_after: Duration::from_secs(60), + } + } + + /// Stale success cannot undo failure and shared fences retain active work. + #[test] + fn fences_generation_exclusivity_and_local_pressure() { + let owner = Adaptive::new(config(), Counts::default(), Instant::now).unwrap(); + assert!(owner.hedge_available(&1)); + let old = owner.acquire(&1).unwrap(); + let failed = owner.acquire(&1).unwrap(); + failed.observe(Outcome::PeerFailure); + old.observe(Outcome::Verified); + assert!(!owner.available(&1)); + let fence = failed.clone(); + drop((old, failed)); + assert_eq!(*owner.observer.active.lock().unwrap(), 1); + drop(fence); + owner.state.lock().unwrap().peers.get_mut(&1).unwrap().retry = Some(Instant::now()); + let probe = owner.acquire(&1).unwrap(); + assert!(matches!(owner.acquire(&1), Err(Error::Unavailable))); + probe.observe(Outcome::Verified); + assert!(!owner.available(&1)); + drop(probe); + assert!(owner.available(&1)); + let work = owner.acquire(&2).unwrap(); + owner.state.lock().unwrap().updated = Instant::now() - config().backoff; + work.observe(Outcome::LocalPressure); + assert_eq!(*owner.observer.limit.lock().unwrap(), 2); + assert!(owner.available(&2)); + owner.state.lock().unwrap().updated = Instant::now() - config().recovery; + work.observe(Outcome::Verified); + assert_eq!(*owner.observer.limit.lock().unwrap(), 3); + } + + /// Only sufficiently old, idle, eligible peer records may be retired. + #[test] + fn capacity_preserves_live_work_and_stale_backoff_then_retires_idle() { + let owner = Adaptive::new(config(), Counts::default(), Instant::now).unwrap(); + let one = owner.acquire(&1).unwrap(); + let two = owner.acquire(&2).unwrap(); + assert!(matches!(owner.acquire(&3), Err(Error::Overloaded))); + one.observe(Outcome::PeerFailure); + drop(one); + owner + .state + .lock() + .unwrap() + .peers + .get_mut(&1) + .unwrap() + .updated = Instant::now() - Duration::from_secs(61); + assert!(matches!(owner.acquire(&3), Err(Error::Overloaded))); + assert!(owner.state.lock().unwrap().peers.contains_key(&1)); + owner.state.lock().unwrap().peers.get_mut(&1).unwrap().retry = Some(Instant::now()); + let probe = owner.acquire(&1).unwrap(); + drop(probe); + assert!(!owner.available(&1)); + assert!(matches!(owner.acquire(&3), Err(Error::Overloaded))); + drop(two); + owner + .state + .lock() + .unwrap() + .peers + .get_mut(&2) + .unwrap() + .updated = Instant::now() - Duration::from_secs(61); + assert!(owner.acquire(&3).is_ok()); + assert_eq!(owner.state.lock().unwrap().peers.len(), 2); + } + + /// Reject invalid geometry and enforce aggregate and per-key caps. + #[test] + fn validation_and_caps() { + for (total, per_key, capacity) in [(0, 1, 1), (1, 0, 1), (1, 2, 1), (1, 1, 0)] { + assert!(matches!( + Adaptive::::new( + Config { + total, + per_key, + capacity, + ..config() + }, + Counts::default(), + Instant::now + ), + Err(Error::InvalidInput) + )); + } + let owner = Adaptive::new( + Config { + total: 2, + per_key: 1, + ..config() + }, + Counts::default(), + Instant::now, + ) + .unwrap(); + let one = owner.acquire(&1).unwrap(); + assert!(matches!(owner.acquire(&1), Err(Error::Overloaded))); + let two = owner.acquire(&2).unwrap(); + assert!(!owner.hedge_available(&3)); + assert!(matches!(owner.acquire(&3), Err(Error::Overloaded))); + one.observe(Outcome::Neutral); + drop((one, two)); + assert_eq!(*owner.observer.active.lock().unwrap(), 0); + } + + /// Recovery saturates even when limits span the machine word. + #[test] + fn full_width_limits_recover_without_overflow() { + let owner = Adaptive::new( + Config { + total: usize::MAX, + per_key: usize::MAX, + recovery: Duration::ZERO, + ..config() + }, + Counts::default(), + Instant::now, + ) + .unwrap(); + let permit = owner.acquire(&()).unwrap(); + permit.observe(Outcome::Verified); + assert_eq!(*owner.observer.limit.lock().unwrap(), usize::MAX); + assert_eq!(owner.state.lock().unwrap().peers[&()].limit, usize::MAX); + } +} diff --git a/cmd/racer-dataplane/flow/src/coalesce.rs b/cmd/racer-dataplane/flow/src/coalesce.rs new file mode 100644 index 000000000..8aeaad5d1 --- /dev/null +++ b/cmd/racer-dataplane/flow/src/coalesce.rs @@ -0,0 +1,1953 @@ +//! Worker-local keyed cohorts with bounded registration and leader re-election. +//! +//! Callers own execution, cancellation, admission policy, and result semantics. +//! A cloned registration retains the same waiter and leadership until its final +//! handle drops. Complete a cohort only after the real operation has finished. + +use std::cell::{Cell, RefCell}; +use std::collections::{HashMap, hash_map::RandomState}; +use std::hash::{BuildHasher, Hash}; +use std::rc::Rc; +use std::task::{Poll, Waker}; + +/// Local bounds. Zero waiters rejects all joins; zero attempts immediately +/// broadcasts the caller-supplied exhausted value without electing a leader. +#[derive(Clone, Copy, Debug)] +pub struct Limits { + /// Maximum distinct registrations sharing one cohort. + pub waiters_per_cohort: usize, + + /// Maximum leadership elections before broadcasting exhaustion. + pub attempts_per_cohort: usize, +} + +/// Admission would exceed a key, waiter, or identifier bound. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub struct CapacityError; + +/// A registration may start an attempt or observe its cohort's result. +#[derive(Clone, Debug, Eq, PartialEq)] +pub enum Event { + /// This registration owns the next attempt. + Lead, + + /// The operation finished or the attempt budget was exhausted. + Complete(V), +} + +/// Completed cohorts close admission immediately but continue charging live +/// registrations. `S::default()` is called for each map, including new cohorts. +pub struct Table { + active: RefCell, S>>, + + registrations: Cell, + + limits: Limits, + + exhausted: V, +} + +/// Local cohort ownership shared between the admission index and registrations. +type CohortOwner = Rc>>; + +impl Table { + /// Create an empty worker-local table and its attempt-exhaustion result. + pub fn new(limits: Limits, exhausted: V) -> Self { + Self { + active: RefCell::new(HashMap::default()), + registrations: Cell::new(0), + limits, + exhausted, + } + } +} + +impl Table { + /// Count distinct registrations, including readers of completed cohorts. + pub fn registration_count(&self) -> usize { + self.registrations.get() + } + + /// Count cohorts still accepting new registrations. + pub fn active_count(&self) -> usize { + self.active.borrow().len() + } +} + +impl Table { + /// Capacity bounds active keys and, multiplied by the waiter limit, all live + /// registrations (including readers of completed cohorts). Clones count once. + pub fn join( + self: &Rc, + key: K, + capacity: usize, + ) -> Result, CapacityError> { + if self.registrations.get() >= capacity.saturating_mul(self.limits.waiters_per_cohort) { + return Err(CapacityError); + } + let mut active = self.active.borrow_mut(); + let cohort = if let Some(cohort) = active.get(&key) { + cohort.clone() + } else { + if active.len() >= capacity { + return Err(CapacityError); + } + let cohort = Rc::new(RefCell::new(Cohort::default())); + active.insert(key.clone(), cohort.clone()); + cohort + }; + let id = { + let mut state = cohort.borrow_mut(); + if state.waiters.len() >= self.limits.waiters_per_cohort { + return Err(CapacityError); + } + let id = state.next; + state.next = state.next.checked_add(1).ok_or(CapacityError)?; + state.waiters.insert(id, None); + id + }; + self.registrations.set(self.registrations.get() + 1); + Ok(Registration { + owner: Rc::new(RegistrationOwner { + table: self.clone(), + key, + cohort, + id, + }), + }) + } +} + +/// A worker-local handle to one waiter. Clones retain its leadership and charge; +/// only dropping the last handle detaches it. Cloning never clones the key. +/// Handles cannot move or be shared across workers: +/// ```compile_fail +/// fn require_send() {} +/// require_send::>(); +/// ``` +/// ```compile_fail +/// fn require_sync() {} +/// require_sync::>(); +/// ``` +pub struct Registration { + owner: Rc>, +} + +/// Owns exactly one registration charge and releases it once on final Rc drop. +struct RegistrationOwner { + table: Rc>, + + key: K, + + cohort: CohortOwner, + + id: u64, +} + +impl Clone for Registration { + /// Retain the same waiter owner without duplicating its key or charge. + fn clone(&self) -> Self { + Self { + owner: self.owner.clone(), + } + } +} + +impl Registration { + /// Whether this is the last handle for this registration, not the cohort. + pub fn is_only_handle(&self) -> bool { + Rc::strong_count(&self.owner) == 1 + } + + /// Release leadership and notify the cohort after an attempt completes. + /// Retry policy and whether this caller should detach belong to the caller. + pub fn retry(&self) { + let wakers = { + let mut state = self.owner.cohort.borrow_mut(); + if state.leader == Some(self.owner.id) { + state.leader = None; + } + state.take_wakers() + }; + wake_all(wakers); + } + + /// Broadcast a caller-owned value and close admission before waking readers. + /// The operation owner must call this only after completion, not cancellation. + pub fn finish(&self, result: V) { + let wakers = { + let mut state = self.owner.cohort.borrow_mut(); + state.result = Some(result); + state.leader = None; + state.take_wakers() + }; + self.owner.remove_active(); + wake_all(wakers); + } +} + +impl Registration { + /// Register the latest wake target and elect at most once per attempt. + pub fn event(&self, waker: &Waker) -> Poll> { + let mut state = self.owner.cohort.borrow_mut(); + if let Some(result) = &state.result { + return Poll::Ready(Event::Complete(result.clone())); + } + state.waiters.insert(self.owner.id, Some(waker.clone())); + if state.leader.is_none() { + if state.attempts >= self.owner.table.limits.attempts_per_cohort { + drop(state); + let result = self.owner.table.exhausted.clone(); + self.finish(result.clone()); + return Poll::Ready(Event::Complete(result)); + } + state.attempts += 1; + state.leader = Some(self.owner.id); + return Poll::Ready(Event::Lead); + } + Poll::Pending + } +} + +impl RegistrationOwner { + /// Remove only this cohort, never a replacement admitted under the same key. + fn remove_active(&self) { + let mut active = self.table.active.borrow_mut(); + if active + .get(&self.key) + .is_some_and(|entry| Rc::ptr_eq(entry, &self.cohort)) + { + active.remove(&self.key); + } + } +} + +impl Drop for RegistrationOwner { + /// Detach once, release the charge, then wake followers outside all borrows. + fn drop(&mut self) { + let (empty, wakers) = { + let mut state = self.cohort.borrow_mut(); + state.waiters.remove(&self.id); + let wakers = if state.leader == Some(self.id) { + state.leader = None; + state.take_wakers() + } else { + Vec::new() + }; + (state.waiters.is_empty(), wakers) + }; + self.table + .registrations + .set(self.table.registrations.get() - 1); + if empty { + self.remove_active(); + } + wake_all(wakers); + } +} + +/// Election state shared by distinct registrations for the same key. +struct Cohort { + next: u64, + + leader: Option, + + attempts: usize, + + waiters: HashMap, S>, + + result: Option, +} + +impl Default for Cohort { + /// Start with no waiters, leadership, attempts, or published result. + fn default() -> Self { + Self { + next: 0, + leader: None, + attempts: 0, + waiters: HashMap::default(), + result: None, + } + } +} + +impl Cohort { + /// Drain wake targets so callbacks can run after releasing the cohort borrow. + fn take_wakers(&mut self) -> Vec { + self.waiters.values_mut().filter_map(Option::take).collect() + } +} + +/// Deliver collected notifications only after the caller releases owner borrows. +fn wake_all(wakers: Vec) { + for waker in wakers { + waker.wake(); + } +} + +/// Shared-result cohorts whose callers own execution rather than electing leaders. +/// Dropping receivers cannot remove real work. Dropping a completion sender +/// closes receivers but keeps its entry until completion is confirmed. +pub mod shared { + use futures::FutureExt; + use futures::channel::oneshot; + use futures::future::{LocalBoxFuture, Shared}; + use std::cell::RefCell; + use std::collections::BTreeMap; + use std::rc::Rc; + + /// Cloneable, worker-local result receiver independent of execution ownership. + pub type Receiver = Shared>; + + /// Ordered index retaining work until its completion owner explicitly finishes. + pub struct Table { + entries: RefCell>>, + } + + impl Default for Table { + /// Create an empty operation index. + fn default() -> Self { + Self { + entries: RefCell::new(BTreeMap::new()), + } + } + } + + impl Table { + /// Count operations, including those whose completion sender was lost. + pub fn len(&self) -> usize { + self.entries.borrow().len() + } + + /// Whether no operation remains indexed. + pub fn is_empty(&self) -> bool { + self.entries.borrow().is_empty() + } + } + + impl Table { + /// Join an existing result without taking ownership of its execution. + pub fn get(&self, key: &K) -> Option> { + self.entries.borrow().get(key).cloned() + } + } + + impl Table { + /// Called after miss-only admission, without yielding between get and start. + /// The caller must not start a replacement while an entry is present. + pub fn start(self: &Rc, key: K, closed: V) -> (Receiver, Completion) { + let (send, receive) = oneshot::channel(); + let receive = async move { receive.await.unwrap_or(closed) } + .boxed_local() + .shared(); + self.entries + .borrow_mut() + .insert(key.clone(), receive.clone()); + ( + receive, + Completion { + table: self.clone(), + key, + send, + }, + ) + } + } + + /// Sole completion authority. Dropping it reports closure, not real completion, + /// so its table entry deliberately remains occupied. + pub struct Completion { + table: Rc>, + + key: K, + + send: oneshot::Sender, + } + + impl Completion { + /// Real completion closes admission before notifying any old readers. + pub fn finish(self, value: V) { + self.table.entries.borrow_mut().remove(&self.key); + let _ = self.send.send(value); + } + } + + /// Shared-result ownership and replacement contracts. + #[cfg(test)] + mod tests { + use super::*; + use futures::executor::block_on; + + /// Results remain readable after completion admits a replacement. + #[test] + fn success_miss_failure_broadcast_and_replacement() { + for value in [Ok(Some(7)), Ok(None), Err("io")] { + let table = Rc::new(Table::default()); + let (first, completion) = table.start(1, Err("closed")); + let second = table.get(&1).unwrap(); + assert_eq!(table.len(), 1); + completion.finish(value); + assert!(table.is_empty()); + let (next, completion) = table.start(1, Err("closed")); + assert_eq!(block_on(first), value); + assert_eq!(block_on(second), value); + assert_eq!(table.len(), 1); + completion.finish(Ok(Some(8))); + assert_eq!(block_on(next), Ok(Some(8))); + } + } + + /// Neither receiver cancellation nor sender loss pretends work completed. + #[test] + fn all_readers_drop_keeps_completion_owner_and_lost_sender_stays_closed() { + let table = Rc::new(Table::default()); + let (receive, completion) = table.start(1, Err::("closed")); + drop(receive); + assert_eq!(table.len(), 1); + let late = table.get(&1).unwrap(); + completion.finish(Ok(9)); + assert_eq!(block_on(late), Ok(9)); + assert!(table.is_empty()); + let (receive, completion) = table.start(1, Err("closed")); + drop(completion); + assert_eq!(block_on(receive), Err("closed")); + assert_eq!(block_on(table.get(&1).unwrap()), Err("closed")); + assert_eq!(table.len(), 1); + } + } +} + +pub mod flight { + //! Owned flight indexing, generation fences, and bounded lifecycle sweeps. + //! + //! Entry hooks own result interpretation and policy. A sweep refreshes each + //! selected entry once and removes it only when the hook reports quiescence. + //! Callers collect wakes while borrowed and dispatch them after releasing locks. + + use std::cell::RefCell; + use std::collections::{BTreeMap, HashMap, hash_map::RandomState}; + use std::hash::{BuildHasher, Hash}; + use std::rc::Rc; + use std::task::Waker; + + /// Run a synchronous owner transaction, then notify outside its mutable borrow. + pub fn update(owner: &RefCell, f: impl FnOnce(&mut T, &mut Vec) -> R) -> R { + let mut wakes = Vec::new(); + let result = { + let mut state = owner.borrow_mut(); + f(&mut state, &mut wakes) + }; + super::wake_all(wakes); + result + } + + /// A monotonic identifier or acquisition budget has been exhausted. + #[derive(Clone, Copy, Debug, Eq, PartialEq)] + pub struct Exhausted; + + /// A registration, generation, or operation no longer matches its owner. + #[derive(Clone, Copy, Debug, Eq, PartialEq)] + pub struct Stale; + + /// Key equality belongs to the adapter; this identity fences owner, incarnation, + /// and acquisition generation independently of any application key schema. + #[derive(Clone)] + pub struct Identity { + /// Local table owner; pointer identity fences unrelated workers and tables. + pub owner: Rc<()>, + + /// Monotonic identity of one entry admission. + pub incarnation: u64, + + /// Acquisition attempt within this incarnation. + pub generation: u64, + } + + impl Identity { + /// Compare entry ownership without invalidating waiters across retries. + pub fn same_registration(&self, current: &Self) -> bool { + Rc::ptr_eq(&self.owner, ¤t.owner) && self.incarnation == current.incarnation + } + + /// Validate an acquisition capability against the current generation. + pub fn validate(&self, current: &Self) -> Result<(), Stale> { + if !self.same_registration(current) || self.generation != current.generation { + return Err(Stale); + } + Ok(()) + } + + /// Called only after the adapter confirms eligibility and completion fences. + fn advance(&mut self, limit: u64) -> Result<(), Exhausted> { + if self.generation >= limit { + return Err(Exhausted); + } + self.generation = self.generation.checked_add(1).ok_or(Exhausted)?; + Ok(()) + } + } + + /// Application lifecycle hooks. Refresh must be bounded and only enqueue wakes. + /// Quiescence must include both detached waiters and actual retained operations, + /// never merely cancellation requested by a caller or expiration of a deadline. + pub trait Entry { + /// Refresh bounded lifecycle work and enqueue notifications without waking. + fn refresh(&mut self, wakes: &mut Vec); + + /// Report that neither waiters nor retained operations need this entry. + fn quiescent(&self) -> bool; + } + + /// Owned membership with a synchronized, bounded sweep index. + /// Direct map mutation is intentionally unavailable: + /// ```compile_fail + /// let mut table = flow_control::coalesce::flight::Table::::default(); + /// table.entries.clear(); + /// ``` + /// Even an empty table belongs to its worker: + /// ```compile_fail + /// fn require_send() {} + /// require_send::>(); + /// ``` + /// ```compile_fail + /// fn require_sync() {} + /// require_sync::>(); + /// ``` + pub struct Table { + entries: HashMap, S>, + + sweep: BTreeMap, + + next_sweep_id: u64, + + cursor: Cursor, + + incarnation: Counter, + + next_waiter: Counter, + + next_operation: Counter, + + stopping: bool, + + drain_waker: Option, + + /// Keep even an empty table local to the worker that drives its lifecycle. + local: std::marker::PhantomData>, + } + + /// Membership identity owned by the table, never by mutable application state. + struct Indexed { + sweep_id: u64, + + entry: E, + } + + impl Default for Table { + /// Create an empty worker-local index with fresh identifier sequences. + fn default() -> Self { + Self { + entries: HashMap::default(), + sweep: BTreeMap::new(), + next_sweep_id: 0, + cursor: Cursor::default(), + incarnation: Counter::default(), + next_waiter: Counter::default(), + next_operation: Counter::default(), + stopping: false, + drain_waker: None, + local: std::marker::PhantomData, + } + } + } + + impl Table { + /// Count entries, including those waiting for real operation completion. + pub fn len(&self) -> usize { + self.entries.len() + } + + /// Whether all owned entries have been removed. + pub fn is_empty(&self) -> bool { + self.entries.is_empty() + } + + /// Inspect application state without exposing the membership index. + pub fn values(&self) -> impl Iterator { + self.entries.values().map(|indexed| &indexed.entry) + } + + /// Allocate a never-reused waiter identifier. + pub fn next_waiter_id(&mut self) -> Result { + self.next_waiter.next_id() + } + + /// Allocate a never-reused operation identifier. + pub fn next_operation_id(&mut self) -> Result { + self.next_operation.next_id() + } + + /// Whether shutdown was requested; the adapter must enforce admission closure. + pub fn is_stopping(&self) -> bool { + self.stopping + } + + /// Retain the latest drain task's wake target. + pub fn register_drain(&mut self, waker: &Waker) { + state::store_waker(&mut self.drain_waker, waker); + } + + /// Enqueue a pending drain notification at most once. + pub fn notify_drain(&mut self, wakes: &mut Vec) { + if let Some(waker) = self.drain_waker.take() { + wakes.push(waker); + } + } + + /// Allocate an entry incarnation fenced by its local owner. + pub fn identity(&mut self, owner: Rc<()>) -> Result { + Ok(Identity { + owner, + incarnation: self.incarnation.next_id()?, + generation: 0, + }) + } + } + + impl Table { + /// Look up the current entry for a key. + pub fn get(&self, key: &K) -> Option<&E> { + self.entries.get(key).map(|indexed| &indexed.entry) + } + + /// Mutate application state without affecting table-owned sweep membership. + /// The adapter remains responsible for its own incarnation-fence semantics. + pub fn get_mut(&mut self, key: &K) -> Option<&mut E> { + self.entries.get_mut(key).map(|indexed| &mut indexed.entry) + } + + /// Whether a key currently has an owned entry. + pub fn contains_key(&self, key: &K) -> bool { + self.entries.contains_key(key) + } + } + + impl Table { + /// The adapter must have checked admission before inserting a new identity. + pub fn insert(&mut self, key: K, entry: E) { + let sweep_id = self.allocate_sweep_id(); + if let Some(previous) = self + .entries + .insert(key.clone(), Indexed { sweep_id, entry }) + { + self.sweep.remove(&previous.sweep_id); + } + self.sweep.insert(sweep_id, key); + } + + /// Find a free internal sweep slot without consuming incarnation IDs. + fn allocate_sweep_id(&mut self) -> u64 { + // Unlike externally visible incarnation fences, these IDs can be reused + // after removal. At most len + 1 probes find a free ID, even after wrap. + // This keeps insert infallible without coupling it to identity(). + for _ in 0..=self.sweep.len() { + self.next_sweep_id = self.next_sweep_id.wrapping_add(1); + if !self.sweep.contains_key(&self.next_sweep_id) { + return self.next_sweep_id; + } + } + unreachable!("a finite in-memory table cannot occupy every u64 sweep ID") + } + } + + impl Table { + /// Used after explicit detach or completion as well as by background sweeps. + pub fn remove_quiescent(&mut self, key: &K) -> bool { + if !self.get(key).is_some_and(Entry::quiescent) { + return false; + } + let entry = self.entries.remove(key).expect("quiescent entry"); + self.sweep.remove(&entry.sweep_id); + true + } + } + + impl Table { + /// Mark shutdown and visit each entry synchronously. The adapter must reject + /// admission, choose settlement policy, and drive actual operation completions. + pub fn stop( + &mut self, + wakes: &mut Vec, + mut stop: impl FnMut(&mut E, &mut Vec), + ) { + self.stopping = true; + for entry in self.entries.values_mut() { + stop(&mut entry.entry, wakes); + } + } + } + + impl Table { + /// Refresh at most the budgeted entry count and remove quiescent entries. + pub fn sweep(&mut self, budget: usize, wakes: &mut Vec) { + for _ in 0..budget.min(self.sweep.len()) { + let Some((_, key)) = self.cursor.next(&self.sweep) else { + break; + }; + let key = key.clone(); + if let Some(entry) = self.get_mut(&key) { + entry.refresh(wakes); + } + self.remove_quiescent(&key); + } + if self.entries.is_empty() { + self.notify_drain(wakes); + } + } + } + + /// Two-phase completion slots. Taking a resource leaves its slot occupied while + /// its destructor runs outside the table borrow. Only explicit completion clears + /// that slot; dropping an external completion token does not touch this owner. + pub struct Operations { + slots: HashMap, S>, + + /// Resource destruction and completion must run on the owning worker. + local: std::marker::PhantomData>, + } + + impl Default for Operations { + /// Create an empty local resource owner. + fn default() -> Self { + Self { + slots: HashMap::default(), + local: std::marker::PhantomData, + } + } + } + + impl Operations { + /// Count retained resources and occupied completion tombstones. + pub fn len(&self) -> usize { + self.slots.len() + } + + /// Whether every resource has been taken and explicitly completed. + pub fn is_empty(&self) -> bool { + self.slots.is_empty() + } + } + + impl Operations { + /// Retain resources under a caller-allocated unique operation identifier. + pub fn insert(&mut self, id: u64, resources: R) { + self.slots.insert(id, OperationSlot::Retained(resources)); + } + + /// Take resources for destruction outside the owner borrow, retaining a fence. + pub fn take(&mut self, id: u64) -> Result { + let slot = self.slots.get_mut(&id).ok_or(Stale)?; + match std::mem::replace(slot, OperationSlot::Completing) { + OperationSlot::Retained(resources) => Ok(resources), + OperationSlot::Completing => Err(Stale), + } + } + + /// Clear a taken slot after the caller has finished resource destruction. + pub fn complete(&mut self, id: u64) -> Result<(), Stale> { + if !matches!(self.slots.get(&id), Some(OperationSlot::Completing)) { + return Err(Stale); + } + self.slots.remove(&id); + Ok(()) + } + } + + /// An occupied operation owns either resources or their unfinished completion fence. + enum OperationSlot { + /// Resources must be taken and destroyed before completion is acknowledged. + Retained(R), + + /// Resources left the owner, but actual completion has not been confirmed. + Completing, + } + + /// Monotonic IDs are never reused, even after an entry is removed. + #[derive(Default)] + struct Counter(u64); + + impl Counter { + /// Allocate the next identifier without wrapping or reusing an old value. + fn next_id(&mut self) -> Result { + self.0 = self.0.checked_add(1).ok_or(Exhausted)?; + Ok(self.0) + } + } + + /// Stable round-robin selection without a scan, including wrap after removal. + #[derive(Default)] + struct Cursor(u64); + + impl Cursor { + /// Select the next live identifier, wrapping to the first when necessary. + fn next<'a, V>(&mut self, entries: &'a BTreeMap) -> Option<(u64, &'a V)> { + let (&id, value) = entries + .range(( + std::ops::Bound::Excluded(self.0), + std::ops::Bound::Unbounded, + )) + .next() + .or_else(|| entries.first_key_value())?; + self.0 = id; + Some((id, value)) + } + } + + /// Waiter lifecycle and result-independent flight transitions. + pub mod state { + use super::{Cursor, Exhausted, Identity, Stale}; + use std::collections::BTreeMap; + use std::ops::{Deref, DerefMut}; + use std::task::Waker; + use std::time::Instant; + + /// Request-owned policy facts, not acquisition credits or result interpretation. + pub trait WaiterPolicy { + /// Lightweight failure value retained with a waiter. + type Error: Copy; + + /// Check cancellation, deadline, and any other request-owned policy. + fn check(&self) -> Option; + + /// Return the request's fixed deadline for bounded expiry indexing. + fn deadline(&self) -> Instant; + } + + /// One attached caller's policy, acquisition eligibility, and notification state. + pub struct Waiter { + /// Caller-owned policy facts, available through dereferencing as well. + policy: W, + + /// Whether this caller can acquire rather than only observe results. + acquisition: bool, + + /// Whether partial publication permits another attempt for this caller. + pub complete: bool, + + /// Whether this caller has already received its acquisition opportunity. + pub issued: bool, + + /// Sticky policy or cancellation failure. + pub error: Option, + + /// Latest wake target, taken once when notification is enqueued. + pub waker: Option, + } + + impl Deref for Waiter { + type Target = W; + + /// Read the caller-owned policy facts. + fn deref(&self) -> &W { + &self.policy + } + } + + impl DerefMut for Waiter { + /// Update caller-owned policy facts without replacing waiter state. + fn deref_mut(&mut self) -> &mut W { + &mut self.policy + } + } + + impl Waiter { + /// Whether an unissued acquisition opportunity remains usable. + fn eligible(&self) -> bool { + self.acquisition && !self.issued && self.error.is_none() + } + } + + /// Attempt outcome held behind the retained-operation completion fence. + pub enum Outcome { + /// A publication awaiting settlement and result interpretation. + Published(O), + + /// A terminal error awaiting settlement. + Failed(E), + + /// An attempt that permits another eligible caller after settlement. + Retry, + } + + /// Flight lifecycle independent of application-specific result semantics. + pub enum Phase { + /// A leader may acquire and publish a result. + Acquiring, + + /// No retained attempt blocks election of an eligible caller. + RetryPending, + + /// An outcome exists but retained operations may still own resources. + Draining(Outcome), + + /// A fully settled complete result is available. + Complete(C), + + /// A fully settled terminal error is available. + Failed(E), + } + + /// Adapter interpretation of a successfully settled publication. + pub enum Published { + /// A result that satisfies all callers. + Complete(C), + + /// A reusable intermediate result that may require another acquisition. + Partial(P), + } + + /// Worker-local waiter index and transition state driven by an owning adapter. + /// The adapter supplies actual operation idleness; cancellation is not idleness. + pub struct State { + /// Current acquisition or settlement phase. + pub phase: Phase, + + /// Current leader's waiter identifier, retained until settlement. + pub leader: Option, + + /// Attached callers indexed by monotonic waiter identifier. + pub waiters: BTreeMap>, + + /// Deadline index used for bounded expiration work. + pub deadlines: BTreeMap<(Instant, u64), ()>, + + /// Settled intermediate result retained across another acquisition. + pub partial: Option

, + + cursor: Cursor, + + /// State transitions belong to one worker even for thread-safe policy data. + local: std::marker::PhantomData>, + } + + impl Default for State { + /// Start without callers or a retained attempt, ready for initial election. + fn default() -> Self { + Self { + phase: Phase::RetryPending, + leader: None, + waiters: BTreeMap::new(), + deadlines: BTreeMap::new(), + partial: None, + cursor: Cursor::default(), + local: std::marker::PhantomData, + } + } + } + + impl State { + /// Attach a caller using an identifier allocated by the owning table. + pub fn register(&mut self, id: u64, policy: W, acquisition: bool, complete: bool) { + self.deadlines.insert((policy.deadline(), id), ()); + self.waiters.insert( + id, + Waiter { + policy, + acquisition, + complete, + issued: false, + error: None, + waker: None, + }, + ); + } + + /// Remove a caller and its deadline without declaring operation completion. + pub fn detach(&mut self, id: u64) { + if let Some(waiter) = self.waiters.remove(&id) { + self.deadlines.remove(&(waiter.deadline(), id)); + } + } + + /// Validate attachment and incarnation while allowing acquisition retries. + pub fn validate_registration( + &self, + id: u64, + attached: bool, + registered: &Identity, + current: &Identity, + ) -> Result<(), Stale> { + if !attached + || !registered.same_registration(current) + || !self.waiters.contains_key(&id) + { + return Err(Stale); + } + Ok(()) + } + + /// Mark only the supplying caller; refresh decides whether to revoke a leader. + pub fn cancel(&mut self, id: u64, error: W::Error) { + if let Some(waiter) = self.waiters.get_mut(&id) { + waiter.error = Some(error); + } + } + + /// Caller validates its leader capability and publication before this step. + /// Notification timing remains explicit: publication wakes only on settlement, + /// while rejection/revocation also notifies before retained operations finish. + pub fn begin_completion(&mut self, outcome: Outcome) { + self.phase = Phase::Draining(outcome); + } + + /// Enqueue all parked caller notifications without invoking their wakers. + pub fn notify(&mut self, wakes: &mut Vec) { + for waiter in self.waiters.values_mut() { + if let Some(waker) = waiter.waker.take() { + wakes.push(waker); + } + } + } + + /// Check one caller in round-robin order without scanning the waiter map. + pub fn sweep_waiter(&mut self, wakes: &mut Vec) { + if let Some((id, _)) = self.cursor.next(&self.waiters) { + self.refresh_waiter(id, wakes); + } + } + + /// Validate policy/budget first in the adapter. Only an eligible waiter can + /// receive a new generation; retained operations must have settled first. + pub fn elect( + &mut self, + id: u64, + identity: &mut Identity, + limit: u64, + ) -> Result { + if !matches!(self.phase, Phase::RetryPending) { + return Ok(false); + } + let Some(waiter) = self.waiters.get_mut(&id).filter(|waiter| waiter.eligible()) + else { + return Ok(false); + }; + identity.advance(limit)?; + self.phase = Phase::Acquiring; + self.leader = Some(id); + waiter.issued = true; + Ok(true) + } + + /// Validate the active leader after the adapter checks its generation fence. + pub fn validate_leader(&self, id: u64, active: bool) -> Result<(), Stale> { + if !active || !matches!(self.phase, Phase::Acquiring) || self.leader != Some(id) { + return Err(Stale); + } + Ok(()) + } + + /// Expose an outcome only after the adapter confirms all operations completed. + pub fn settle( + &mut self, + idle: bool, + unavailable: W::Error, + split: fn(O) -> Published, + wakes: &mut Vec, + ) { + if !idle || !matches!(self.phase, Phase::Draining(_)) { + return; + } + let Phase::Draining(outcome) = + std::mem::replace(&mut self.phase, Phase::RetryPending) + else { + unreachable!() + }; + self.leader = None; + self.phase = match outcome { + Outcome::Published(value) => match split(value) { + Published::Complete(value) => { + self.partial = None; + Phase::Complete(value) + } + Published::Partial(value) => { + self.partial = Some(value); + for waiter in self.waiters.values_mut() { + if waiter.complete { + waiter.issued = false; + } + } + Phase::RetryPending + } + }, + Outcome::Failed(error) => Phase::Failed(error), + Outcome::Retry if self.waiters.values().any(Waiter::eligible) => { + Phase::RetryPending + } + Outcome::Retry => Phase::Failed(unavailable), + }; + self.notify(wakes); + } + + /// Revoke acquisition immediately, retaining its outcome behind completion. + pub fn revoke(&mut self, error: W::Error, wakes: &mut Vec) { + if let Some(waiter) = self.leader.and_then(|id| self.waiters.get_mut(&id)) { + waiter.error = Some(error); + } + self.phase = Phase::Draining(Outcome::Retry); + self.notify(wakes); + } + + /// Check the leader and up to 64 expired callers, then settle eligible outcomes. + pub fn refresh( + &mut self, + idle: bool, + unavailable: W::Error, + canceled: W::Error, + now: impl Fn() -> Instant, + split: fn(O) -> Published, + wakes: &mut Vec, + ) { + if let Some(id) = self.leader { + self.refresh_waiter(id, wakes); + } + self.refresh_expired(&now, wakes); + if matches!(self.phase, Phase::Acquiring) { + let leader = self.leader.and_then(|id| self.waiters.get(&id)); + if leader.is_none_or(|waiter| waiter.error.is_some()) { + let error = leader.and_then(|waiter| waiter.error).unwrap_or(canceled); + self.revoke(error, wakes); + } + } + self.settle(idle, unavailable, split, wakes); + if matches!(self.phase, Phase::RetryPending) + && self.partial.is_none() + && !self.waiters.values().any(Waiter::eligible) + { + self.phase = Phase::Draining(Outcome::Failed(unavailable)); + self.settle(idle, unavailable, split, wakes); + } + } + + /// Detach and notify all callers while preserving the operation drain fence. + pub fn stop(&mut self, error: W::Error, wakes: &mut Vec) { + for waiter in self.waiters.values_mut() { + waiter.error = Some(error); + } + self.phase = Phase::Draining(Outcome::Failed(error)); + self.notify(wakes); + self.waiters.clear(); + self.deadlines.clear(); + } + + /// Check one caller and enqueue its wake if its policy has failed. + fn refresh_waiter(&mut self, id: u64, wakes: &mut Vec) { + let Some(waiter) = self.waiters.get_mut(&id) else { + return; + }; + if waiter.error.is_none() { + waiter.error = waiter.check(); + } + if waiter.error.is_some() { + self.deadlines.remove(&(waiter.deadline(), id)); + if let Some(waker) = waiter.waker.take() { + wakes.push(waker); + } + } + } + + /// Consume at most 64 due deadline entries, independently of leader checks. + fn refresh_expired(&mut self, now: &impl Fn() -> Instant, wakes: &mut Vec) { + for _ in 0..64 { + let Some((&(deadline, id), _)) = self.deadlines.first_key_value() else { + break; + }; + if deadline > now() { + break; + } + self.deadlines.remove(&(deadline, id)); + self.refresh_waiter(id, wakes); + } + } + } + + /// Replace a wake target only when it would notify a different task. + pub fn store_waker(slot: &mut Option, waker: &Waker) { + if slot.as_ref().is_none_or(|old| !old.will_wake(waker)) { + *slot = Some(waker.clone()); + } + } + + /// Pure transition tests for cancellation, publication, and bounded expiry. + #[cfg(test)] + mod tests { + use super::*; + use std::cell::Cell; + use std::rc::Rc; + use std::sync::Arc; + use std::sync::atomic::{AtomicUsize, Ordering}; + use std::task::Wake; + use std::time::Duration; + + /// Mutable policy failure used to drive deterministic transitions. + struct Policy { + due: Instant, + + error: Rc>>, + } + + impl WaiterPolicy for Policy { + type Error = u8; + + /// Read the failure injected by the test. + fn check(&self) -> Option { + self.error.get() + } + + /// Return the fixed test deadline. + fn deadline(&self) -> Instant { + self.due + } + } + + /// Numeric results keep these tests independent of application semantics. + type Core = State, Policy>; + + /// Publications already carry their complete or partial interpretation. + fn split(value: Published) -> Published { + value + } + + /// Attach a caller and return its externally mutable policy failure. + fn register( + core: &mut Core, + id: u64, + acquisition: bool, + complete: bool, + ) -> Rc>> { + let error = Rc::new(Cell::new(None)); + core.register( + id, + Policy { + due: Instant::now() + Duration::from_secs(60), + error: error.clone(), + }, + acquisition, + complete, + ); + error + } + + /// Construct the initial identity for an isolated transition scenario. + fn identity() -> Identity { + Identity { + owner: Rc::new(()), + incarnation: 1, + generation: 0, + } + } + + /// Refresh with fixed unavailable and cancellation error values. + fn refresh(core: &mut Core, idle: bool, wakes: &mut Vec) { + core.refresh(idle, 9, 8, Instant::now, split, wakes); + } + + /// Leader cancellation or detach cannot bypass retained-operation drainage. + #[test] + fn cancel_or_detach_leader_drains_before_re_election() { + for detach in [false, true] { + let mut core = Core::default(); + let error = register(&mut core, 1, true, true); + register(&mut core, 2, true, true); + register(&mut core, 3, false, false); + let mut identity = identity(); + assert_eq!(core.elect(3, &mut identity, 4), Ok(false)); + assert_eq!(core.elect(1, &mut identity, 4), Ok(true)); + let old = identity.clone(); + assert_eq!(core.validate_registration(1, true, &old, &identity), Ok(())); + assert_eq!(core.elect(2, &mut identity, 4), Ok(false)); + if detach { + core.detach(1); + } else { + error.set(Some(7)); + } + let mut wakes = Vec::new(); + refresh(&mut core, false, &mut wakes); + assert!(matches!(core.phase, Phase::Draining(Outcome::Retry))); + assert_eq!(core.validate_leader(1, true), Err(Stale)); + assert_eq!(core.elect(2, &mut identity, 4), Ok(false)); + refresh(&mut core, true, &mut wakes); + assert_eq!(core.elect(2, &mut identity, 4), Ok(true)); + assert_eq!(old.validate(&identity), Err(Stale)); + assert_eq!(core.validate_registration(2, true, &old, &identity), Ok(())); + assert_eq!( + core.validate_registration(2, false, &old, &identity), + Err(Stale) + ); + assert_eq!( + core.validate_registration(4, true, &old, &identity), + Err(Stale) + ); + assert_eq!(core.validate_leader(2, true), Ok(())); + assert_eq!(core.validate_leader(2, false), Err(Stale)); + } + } + + /// Partial results renew only complete-result callers after the fence clears. + #[test] + fn partial_then_complete_and_terminal_failure_are_fenced() { + let mut core = Core::default(); + register(&mut core, 1, true, true); + register(&mut core, 2, true, false); + let mut identity = identity(); + assert_eq!(core.elect(1, &mut identity, 3), Ok(true)); + core.waiters.get_mut(&2).unwrap().issued = true; + let mut wakes = Vec::new(); + core.begin_completion(Outcome::Published(Published::Partial(17))); + core.settle(false, 9, split, &mut wakes); + assert!(core.partial.is_none()); + core.settle(true, 9, split, &mut wakes); + assert_eq!(core.partial, Some(17)); + assert!(!core.waiters[&1].issued); + assert!(core.waiters[&2].issued); + assert_eq!(core.elect(1, &mut identity, 3), Ok(true)); + core.begin_completion(Outcome::Published(Published::Complete(18))); + core.settle(true, 9, split, &mut wakes); + assert!(matches!(core.phase, Phase::Complete(18))); + assert!(core.partial.is_none()); + assert_eq!(core.elect(1, &mut identity, 3), Ok(false)); + core.phase = Phase::Draining(Outcome::Failed(6)); + core.settle(false, 9, split, &mut wakes); + assert!(matches!(core.phase, Phase::Draining(Outcome::Failed(6)))); + core.settle(true, 9, split, &mut wakes); + assert!(matches!(core.phase, Phase::Failed(6))); + } + + /// Observers cannot acquire and exhausted generations do not issue a caller. + #[test] + fn copy_only_cannot_revive_retry_and_generations_are_bounded() { + let mut core = Core::default(); + register(&mut core, 1, true, true); + register(&mut core, 2, false, false); + let mut identity = identity(); + assert_eq!(core.elect(1, &mut identity, 1), Ok(true)); + core.revoke(7, &mut Vec::new()); + core.settle(true, 9, split, &mut Vec::new()); + assert!(matches!(core.phase, Phase::Failed(9))); + core.phase = Phase::RetryPending; + register(&mut core, 3, true, true); + assert_eq!(core.elect(3, &mut identity, 1), Err(Exhausted)); + assert!(!core.waiters[&3].issued); + assert_eq!(identity.generation, 1); + } + + /// Counts wake delivery without changing transition state. + #[derive(Default)] + struct Count(AtomicUsize); + + impl Wake for Count { + /// Record one delivered notification. + fn wake(self: Arc) { + self.0.fetch_add(1, Ordering::Relaxed); + } + } + + /// Expiry work is bounded to 64 callers and only their latest wakes fire. + #[test] + fn deadlines_have_exact_quantum_and_latest_wakes_are_taken_once() { + let mut core = Core::default(); + let now = Instant::now(); + let old = Arc::new(Count::default()); + let latest = Arc::new(Count::default()); + for id in 0..65 { + core.register( + id, + Policy { + due: now, + error: Rc::new(Cell::new(Some(3))), + }, + true, + true, + ); + let slot = &mut core.waiters.get_mut(&id).unwrap().waker; + store_waker(slot, &Waker::from(old.clone())); + store_waker(slot, &Waker::from(latest.clone())); + } + let mut wakes = Vec::new(); + core.refresh(true, 9, 8, || now, split, &mut wakes); + assert_eq!(core.deadlines.len(), 1); + assert_eq!(wakes.len(), 64); + assert!(core.waiters[&64].error.is_none()); + core.refresh(true, 9, 8, || now, split, &mut wakes); + assert!(core.deadlines.is_empty()); + assert_eq!(wakes.len(), 65); + core.notify(&mut wakes); + assert_eq!(wakes.len(), 65); + for wake in wakes { + wake.wake(); + } + assert_eq!(old.0.load(Ordering::Relaxed), 0); + assert_eq!(latest.0.load(Ordering::Relaxed), 65); + } + + /// Shutdown clears caller indexes but retains a pending completion outcome. + #[test] + fn detach_cleans_deadline_and_stop_preserves_pending_completion() { + let mut core = Core::default(); + register(&mut core, 1, true, true); + register(&mut core, 2, true, true); + core.detach(1); + assert_eq!(core.deadlines.len(), 1); + let mut identity = identity(); + assert_eq!(core.elect(2, &mut identity, 3), Ok(true)); + core.cancel(2, 6); + assert_eq!(core.waiters[&2].error, Some(6)); + let mut wakes = Vec::new(); + core.stop(8, &mut wakes); + core.settle(false, 9, split, &mut wakes); + assert!(matches!(core.phase, Phase::Draining(Outcome::Failed(8)))); + assert!(core.waiters.is_empty()); + assert!(core.deadlines.is_empty()); + core.settle(true, 9, split, &mut wakes); + assert!(matches!(core.phase, Phase::Failed(8))); + } + } + } + + /// Membership, sweep fairness, and retained-resource fence contracts. + #[cfg(test)] + mod tests { + use super::*; + use std::cell::{Cell, RefCell}; + + /// Count a retained resource until its destructor actually runs. + struct Resource(Rc>); + + impl Drop for Resource { + /// Record actual resource destruction rather than cancellation. + fn drop(&mut self) { + self.0.set(self.0.get() - 1); + } + } + + /// Minimal application entry with independently tracked waiters and resources. + struct TestEntry { + identity: Identity, + + waiters: usize, + + canceled: bool, + + refreshed: usize, + + operations: Operations, + } + + impl Entry for TestEntry { + /// Count visits and detach callers when cancellation is observed. + fn refresh(&mut self, _: &mut Vec) { + self.refreshed += 1; + if self.canceled { + self.waiters = 0; + } + } + + /// Require both caller detachment and real operation completion. + fn quiescent(&self) -> bool { + self.waiters == 0 && self.operations.is_empty() + } + } + + /// Integer keys isolate table ownership from application key semantics. + type TestTable = Table; + + /// Admit an entry with one waiter and return its initial identity. + fn insert(table: &mut TestTable, owner: &Rc<()>, key: u32) -> Identity { + let identity = table.identity(owner.clone()).unwrap(); + table.insert( + key, + TestEntry { + identity: identity.clone(), + waiters: 1, + canceled: false, + refreshed: 0, + operations: Operations::default(), + }, + ); + identity + } + + /// Cancellation cannot release resources or bypass the completion tombstone. + #[test] + fn canceled_entry_keeps_resources_until_two_phase_completion() { + let owner = Rc::new(()); + let mut table = TestTable::default(); + insert(&mut table, &owner, 1); + insert(&mut table, &owner, 2); + let live = Rc::new(Cell::new(1)); + let entry = table.get_mut(&1).unwrap(); + entry.operations.insert(1, Resource(live.clone())); + entry.canceled = true; + let mut wakes = Vec::new(); + table.sweep(1, &mut wakes); + assert_eq!(table.get(&1).unwrap().waiters, 0); + assert_eq!(table.get(&2).unwrap().waiters, 1); + assert_eq!(live.get(), 1, "cancellation is not completion"); + let resources = table.get_mut(&1).unwrap().operations.take(1).unwrap(); + assert!(!table.remove_quiescent(&1), "occupied during resource drop"); + drop(resources); + assert_eq!(live.get(), 0); + assert!( + !table.remove_quiescent(&1), + "completion must clear the tombstone" + ); + table.get_mut(&1).unwrap().operations.complete(1).unwrap(); + assert!(table.remove_quiescent(&1)); + assert_eq!(table.len(), 1); + } + + /// Request handle that detaches its waiter without completing operations. + struct Waiter { + table: Rc>, + + key: u32, + } + + impl Drop for Waiter { + /// Detach this request and remove its entry only if truly quiescent. + fn drop(&mut self) { + let mut table = self.table.borrow_mut(); + table.get_mut(&self.key).unwrap().waiters -= 1; + table.remove_quiescent(&self.key); + } + } + + /// Lost request and identity handles do not own operation resource release. + #[test] + fn dropped_waiter_and_completion_token_do_not_release_owned_operations() { + let owner = Rc::new(()); + let table = Rc::new(RefCell::new(TestTable::default())); + let token = insert(&mut table.borrow_mut(), &owner, 1); + let live = Rc::new(Cell::new(1)); + table + .borrow_mut() + .get_mut(&1) + .unwrap() + .operations + .insert(1, Resource(live.clone())); + drop(Waiter { + table: table.clone(), + key: 1, + }); + drop(token); + table.borrow_mut().sweep(100, &mut Vec::new()); + assert_eq!(table.borrow().len(), 1); + assert_eq!(live.get(), 1); + let resources = table + .borrow_mut() + .get_mut(&1) + .unwrap() + .operations + .take(1) + .unwrap(); + drop(resources); + table + .borrow_mut() + .get_mut(&1) + .unwrap() + .operations + .complete(1) + .unwrap(); + table.borrow_mut().sweep(1, &mut Vec::new()); + assert!(table.borrow().is_empty()); + assert_eq!(live.get(), 0); + } + + /// Owner, incarnation, generation, and resource-take fences are independent. + #[test] + fn reelection_and_recreation_fence_stale_completions() { + let owner = Rc::new(()); + let mut table = TestTable::default(); + insert(&mut table, &owner, 1); + let entry = table.get_mut(&1).unwrap(); + entry.identity.advance(2).unwrap(); + let old = entry.identity.clone(); + let live = Rc::new(Cell::new(1)); + entry.operations.insert(1, Resource(live.clone())); + assert_eq!( + entry.operations.complete(1), + Err(Stale), + "cannot skip resource release" + ); + let resources = entry.operations.take(1).unwrap(); + assert!(matches!(entry.operations.take(1), Err(Stale))); + drop(resources); + entry.operations.complete(1).unwrap(); + assert_eq!(entry.operations.complete(1), Err(Stale)); + entry.identity.advance(2).unwrap(); + assert_eq!(old.validate(&entry.identity), Err(Stale)); + assert!( + old.same_registration(&entry.identity), + "waiter survives retry" + ); + assert_eq!(entry.identity.advance(2), Err(Exhausted)); + assert_eq!(entry.identity.generation, 2); + let mut wrong_owner = old.clone(); + wrong_owner.owner = Rc::new(()); + assert_eq!(wrong_owner.validate(&old), Err(Stale)); + entry.waiters = 0; + assert!(table.remove_quiescent(&1)); + let new = insert(&mut table, &owner, 1); + assert!(!old.same_registration(&new)); + assert_eq!(old.validate(&new), Err(Stale)); + assert_eq!(live.get(), 0); + } + + /// Each budget unit visits one entry and final removal notifies drain once. + #[test] + fn sweeps_are_budgeted_fair_and_remove_index_entries() { + let owner = Rc::new(()); + let mut table = TestTable::default(); + for key in 1..=3 { + insert(&mut table, &owner, key); + } + table.sweep(0, &mut Vec::new()); + assert!(table.values().all(|entry| entry.refreshed == 0)); + for key in 1..=3 { + table.sweep(1, &mut Vec::new()); + assert_eq!(table.get(&key).unwrap().refreshed, 1); + } + table.get_mut(&2).unwrap().canceled = true; + table.sweep(99, &mut Vec::new()); + assert!(!table.contains_key(&2)); + assert_eq!(table.sweep.len(), 2); + assert!(table.values().all(|entry| entry.refreshed == 2)); + table.drain_waker = Some(Waker::noop().clone()); + table.stop(&mut Vec::new(), |entry, _| { + entry.canceled = true; + }); + let mut wakes = Vec::new(); + table.sweep(99, &mut wakes); + assert!(table.is_empty()); + assert!(table.sweep.is_empty()); + assert_eq!(wakes.len(), 1); + table.sweep(99, &mut wakes); + assert_eq!(wakes.len(), 1, "drain wake is taken once"); + } + + /// Cursor wrap is safe after removal while externally visible IDs never wrap. + #[test] + fn cursor_wraps_after_removal_and_counters_never_wrap() { + let mut entries = BTreeMap::from([(1, ()), (2, ()), (3, ())]); + let mut cursor = Cursor::default(); + assert_eq!(cursor.next(&entries).map(|(id, _)| id), Some(1)); + entries.remove(&1); + assert_eq!(cursor.next(&entries).map(|(id, _)| id), Some(2)); + assert_eq!(cursor.next(&entries).map(|(id, _)| id), Some(3)); + assert_eq!(cursor.next(&entries).map(|(id, _)| id), Some(2)); + entries.clear(); + assert!(cursor.next(&entries).is_none()); + let mut counter = Counter(u64::MAX - 1); + assert_eq!(counter.next_id(), Ok(u64::MAX)); + assert_eq!(counter.next_id(), Err(Exhausted)); + assert_eq!(counter.next_id(), Err(Exhausted)); + let mut table = TestTable { + incarnation: counter, + ..TestTable::default() + }; + assert!(matches!(table.identity(Rc::new(())), Err(Exhausted))); + assert!(table.is_empty()); + } + + /// Replacing and removing entries updates only their table-owned sweep slots. + #[test] + fn replacement_and_controlled_access_keep_sweep_membership_synchronized() { + let owner = Rc::new(()); + let mut table = TestTable::default(); + assert!(table.get(&1).is_none()); + assert!(table.get_mut(&1).is_none()); + assert!(!table.remove_quiescent(&1)); + insert(&mut table, &owner, 1); + let original = table.entries[&1].sweep_id; + insert(&mut table, &owner, 1); + let replacement = table.entries[&1].sweep_id; + insert(&mut table, &owner, 2); + assert_eq!(table.len(), 2); + assert_eq!(table.values().count(), 2); + assert!(!table.sweep.contains_key(&original)); + assert!(table.sweep.contains_key(&replacement)); + table.sweep(2, &mut Vec::new()); + assert!(table.values().all(|entry| entry.refreshed == 1)); + table.get_mut(&1).unwrap().waiters = 0; + assert!(table.remove_quiescent(&1)); + assert!(!table.contains_key(&1)); + assert_eq!(table.sweep.len(), table.len()); + table.sweep(1, &mut Vec::new()); + assert_eq!(table.get(&2).unwrap().refreshed, 2); + } + + /// Mutable application identities cannot corrupt another entry's membership. + #[test] + fn mutable_incarnation_cannot_remove_another_entries_sweep_slot() { + let owner = Rc::new(()); + let mut table = TestTable::default(); + insert(&mut table, &owner, 1); + let other = insert(&mut table, &owner, 2); + table.get_mut(&1).unwrap().identity.incarnation = other.incarnation; + table.get_mut(&1).unwrap().waiters = 0; + assert!(table.remove_quiescent(&1)); + assert_eq!(table.sweep.len(), 1); + table.sweep(1, &mut Vec::new()); + assert_eq!(table.get(&2).unwrap().refreshed, 1); + table.get_mut(&2).unwrap().waiters = 0; + table.sweep(1, &mut Vec::new()); + assert!(table.is_empty()); + assert!(table.sweep.is_empty()); + } + + /// Shutdown hooks may mutate identities without invalidating the sweep index. + #[test] + fn stop_identity_mutation_preserves_replacement_and_removal_membership() { + let owner = Rc::new(()); + let mut table = TestTable::default(); + for key in 1..=3 { + insert(&mut table, &owner, key); + } + table.stop(&mut Vec::new(), |entry, _| { + entry.identity.incarnation = u64::MAX; + entry.canceled = true; + }); + insert(&mut table, &owner, 2); + assert_eq!(table.sweep.len(), 3); + table.sweep(3, &mut Vec::new()); + assert_eq!(table.len(), 1); + assert_eq!(table.sweep.len(), 1); + assert_eq!(table.get(&2).unwrap().refreshed, 1); + table.get_mut(&2).unwrap().waiters = 0; + assert!(table.remove_quiescent(&2)); + assert!(table.sweep.is_empty()); + } + + /// Duplicate application incarnations remain independent table memberships. + #[test] + fn duplicate_entry_incarnations_have_independent_bounded_sweeps() { + let owner = Rc::new(()); + let mut table = TestTable::default(); + let identity = table.identity(owner).unwrap(); + for key in 1..=3 { + table.insert( + key, + TestEntry { + identity: identity.clone(), + waiters: 1, + canceled: false, + refreshed: 0, + operations: Operations::default(), + }, + ); + } + assert_eq!(table.sweep.len(), 3); + table.sweep(0, &mut Vec::new()); + assert!(table.values().all(|entry| entry.refreshed == 0)); + for key in 1..=3 { + table.sweep(1, &mut Vec::new()); + assert_eq!(table.get(&key).unwrap().refreshed, 1); + assert_eq!( + table.values().map(|entry| entry.refreshed).sum::(), + key as usize + ); + } + table.get_mut(&2).unwrap().waiters = 0; + assert!(table.remove_quiescent(&2)); + table.sweep(99, &mut Vec::new()); + assert_eq!(table.sweep.len(), 2); + assert!(table.values().all(|entry| entry.refreshed == 2)); + } + + /// Internal sweep IDs wrap safely without consuming monotonic incarnation IDs. + #[test] + fn sweep_ids_wrap_skip_occupied_slots_and_do_not_consume_incarnations() { + let owner = Rc::new(()); + let mut table = TestTable::default(); + let first = insert(&mut table, &owner, 1); + table.next_sweep_id = u64::MAX; + let second = insert(&mut table, &owner, 2); + assert_eq!(table.entries[&2].sweep_id, 0); + let third = insert(&mut table, &owner, 3); + assert_eq!(table.entries[&3].sweep_id, 2, "skip occupied ID 1"); + assert_eq!( + (first.incarnation, second.incarnation, third.incarnation), + (1, 2, 3) + ); + table.sweep(3, &mut Vec::new()); + assert!(table.values().all(|entry| entry.refreshed == 1)); + table.incarnation = Counter(u64::MAX); + assert!(table.identity(owner).is_err()); + table.insert( + 4, + TestEntry { + identity: first, + waiters: 1, + canceled: false, + refreshed: 0, + operations: Operations::default(), + }, + ); + assert_eq!( + table.len(), + 4, + "insert does not require an incarnation allocation" + ); + table.stop(&mut Vec::new(), |entry, _| entry.canceled = true); + table.sweep(4, &mut Vec::new()); + assert!(table.is_empty()); + assert!(table.sweep.is_empty()); + } + } +} + +/// Election, capacity, and final-handle cleanup contracts. +#[cfg(test)] +mod tests { + use super::*; + use std::sync::Arc; + use std::sync::atomic::{AtomicUsize, Ordering}; + use std::task::Wake; + + /// Counts delivered notifications without running an executor. + #[derive(Default)] + struct Counter(AtomicUsize); + + impl Wake for Counter { + /// Record one notification. + fn wake(self: Arc) { + self.0.fetch_add(1, Ordering::Relaxed); + } + } + + /// Small keyed table used by the cohort scenarios. + type TestTable = Table<&'static str, Result>; + + /// Construct a table with independently configurable waiter and attempt bounds. + fn table(waiters: usize, attempts: usize) -> Rc { + Rc::new(Table::new( + Limits { + waiters_per_cohort: waiters, + attempts_per_cohort: attempts, + }, + Err("exhausted"), + )) + } + + /// Completion reaches both parked and unpolled readers without deleting replacements. + #[test] + fn broadcasts_success_and_failure_before_or_after_poll() { + for result in [Ok(42), Err("failed")] { + for poll_first in [false, true] { + let table = table(4, 2); + let leader = table.join("key", 1).unwrap(); + let follower = table.join("key", 1).unwrap(); + let count = Arc::new(Counter::default()); + let waker = Waker::from(count.clone()); + assert_eq!(leader.event(Waker::noop()), Poll::Ready(Event::Lead)); + if poll_first { + assert_eq!(follower.event(&waker), Poll::Pending); + } + leader.finish(result); + assert_eq!(count.0.load(Ordering::Relaxed), usize::from(poll_first)); + assert_eq!(follower.event(&waker), Poll::Ready(Event::Complete(result))); + assert_eq!(table.active_count(), 0); + let next = table.join("key", 1).unwrap(); + assert_eq!(next.event(Waker::noop()), Poll::Ready(Event::Lead)); + drop(leader); + drop(follower); + assert_eq!(table.active_count(), 1, "old cohort cannot remove new key"); + } + } + } + + /// Detached execution retains one waiter until its final handle is dropped. + #[test] + fn last_handle_drop_releases_leadership_and_registration() { + let table = table(2, 3); + let request = table.join("key", 1).unwrap(); + let follower = table.join("key", 1).unwrap(); + assert_eq!(request.event(Waker::noop()), Poll::Ready(Event::Lead)); + let driver = request.clone(); + assert!(!driver.is_only_handle()); + drop(request); + assert!(driver.is_only_handle()); + assert_eq!(table.registration_count(), 2); + let old = Arc::new(Counter::default()); + let latest = Arc::new(Counter::default()); + assert!(follower.event(&Waker::from(old.clone())).is_pending()); + assert!(follower.event(&Waker::from(latest.clone())).is_pending()); + drop(driver); + assert_eq!(old.0.load(Ordering::Relaxed), 0); + assert_eq!(latest.0.load(Ordering::Relaxed), 1); + assert_eq!(table.registration_count(), 1); + assert_eq!(follower.event(Waker::noop()), Poll::Ready(Event::Lead)); + drop(follower); + assert_eq!(table.registration_count(), 0); + assert_eq!(table.active_count(), 0); + } + + /// Retry and owner loss consume the same bounded election budget. + #[test] + fn retry_and_drop_share_bounded_attempts_and_broadcast_exhaustion() { + let table = table(3, 2); + let first = table.join("key", 1).unwrap(); + let second = table.join("key", 1).unwrap(); + let observer = table.join("key", 1).unwrap(); + assert_eq!(first.event(Waker::noop()), Poll::Ready(Event::Lead)); + assert!(first.event(Waker::noop()).is_pending()); + first.retry(); + assert_eq!(second.event(Waker::noop()), Poll::Ready(Event::Lead)); + drop(second); + let exhausted = Poll::Ready(Event::Complete(Err("exhausted"))); + assert_eq!(observer.event(Waker::noop()), exhausted); + assert_eq!(first.event(Waker::noop()), exhausted); + assert_eq!(table.active_count(), 0); + } + + /// Completed readers still consume global capacity until they detach. + #[test] + fn capacity_bounds_keys_waiters_and_completed_readers() { + let table = table(2, 1); + assert!(matches!(table.join("a", 0), Err(CapacityError))); + let a = table.join("a", 2).unwrap(); + let b = table.join("b", 2).unwrap(); + assert!(matches!(table.join("c", 2), Err(CapacityError))); + let a2 = table.join("a", 2).unwrap(); + assert!(matches!(table.join("a", 2), Err(CapacityError))); + let b2 = table.join("b", 2).unwrap(); + a.finish(Ok(1)); + b.finish(Ok(2)); + assert_eq!(table.active_count(), 0); + assert!(matches!(table.join("a", 2), Err(CapacityError))); + drop(a2); + let next = table.join("a", 2).unwrap(); + drop((a, b, b2, next)); + assert_eq!(table.registration_count(), 0); + assert_eq!(table.active_count(), 0); + } + + /// Zero bounds and identifier exhaustion reject work without leaking charges. + #[test] + fn zero_limits_and_id_exhaustion_never_wrap_or_leak_registrations() { + assert!(matches!(table(0, 1).join("key", 1), Err(CapacityError))); + let zero = table(1, 0); + let waiter = zero.join("key", 1).unwrap(); + assert_eq!( + waiter.event(Waker::noop()), + Poll::Ready(Event::Complete(Err("exhausted"))) + ); + let table = table(2, 1); + let first = table.join("key", usize::MAX).unwrap(); + first.owner.cohort.borrow_mut().next = u64::MAX; + assert!(matches!(table.join("key", usize::MAX), Err(CapacityError))); + assert_eq!(table.registration_count(), 1); + drop(first); + assert_eq!(table.active_count(), 0); + } +} diff --git a/cmd/racer-dataplane/flow/src/lib.rs b/cmd/racer-dataplane/flow/src/lib.rs new file mode 100644 index 000000000..11fc326e8 --- /dev/null +++ b/cmd/racer-dataplane/flow/src/lib.rs @@ -0,0 +1,1338 @@ +//! Policy-driven admission, charged storage, and completion-owned flow control. +//! +//! Implement [`Class`] and [`Policy`] to supply resource limits and keyed fairness. +//! [`Quotas`] is a worker-local authority; [`SharedQuotas`] admits unkeyed work +//! across threads. Keep each [`Charge`] with its resource until completion. A +//! completion reservation bypasses stop and fair-share checks, not aggregate or +//! key-record limits. Policy can also allow selected classes after stop. +//! +//! Recycling wipes the full allocation before retaining at most two buffers of +//! at least 1 MiB. Pressure retries admission after applicable reclamation; stop +//! releases retained buffers. Shared handles never retain the recycler. +//! +//! [`ChargedBuffer`] pairs fixed initialized backing with admission. Low-level +//! raw buffer users must themselves retain sufficient charge for every allocation +//! and validate class, key, and provenance. Application authentication, error +//! classification, request deadlines, and cancellation policy stay with callers. +//! +//! Endpoint circuits, adaptive admission, handoffs, and hedge alarms use distinct +//! ownership rules. Pipes retain their charges while idle; socket-retained bytes +//! are outside the pipe capacity budget. The `simulation` feature forwards the +//! runtime's simulated descriptors without changing admission policy. +//! +//! [`coalesce`] provides worker-local keyed cohorts, shared results, and flight +//! lifecycle tracking. Callers retain execution, admission, and result policy; +//! cancellation never substitutes for real operation completion. +#![deny(unsafe_op_in_unsafe_fn)] + +/// Adaptive admission and completion-owned handoffs. +mod admission; + +/// Keyed cohorts and completion-owned flight tracking. +pub mod coalesce; + +/// Charged kernel pipes and worker-local reuse. +mod pipe; + +pub use admission::{ + Adaptive, Admitted, Circuits, Config as AdaptiveConfig, Event as AdaptiveEvent, Handoff, + HandoffAdmission, HedgePermit, Hedges, Observer as AdaptiveObserver, Offer, + Outcome as AdaptiveOutcome, Permit as AdaptivePermit, Probe, +}; +pub use pipe::{MAX_PIPE_BYTES, PipeLease, PipePool, splice_unsupported}; + +use std::{ + cell::RefCell, + collections::{HashMap, VecDeque}, + hash::Hash, + marker::PhantomData, + rc::Rc, + sync::{ + Arc, Mutex, Weak, + atomic::{AtomicBool, AtomicUsize, Ordering}, + }, +}; + +/// Admission and pipe-creation failures, independent of application error policy. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub enum Error { + /// The requested geometry or state transition is invalid. + InvalidInput, + + /// A resource or keyed fairness limit is exhausted. + Overloaded, + + /// Admission has stopped or its shared state is unavailable. + Unavailable, + + /// A kernel pipe operation failed during creation. + Io, +} + +impl std::fmt::Display for Error { + /// Describe the failure without application resource names. + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.write_str(match self { + Self::InvalidInput => "invalid flow-control input", + Self::Overloaded => "flow-control quota exhausted", + Self::Unavailable => "flow control stopped", + Self::Io => "flow-control I/O failed", + }) + } +} + +impl std::error::Error for Error {} + +/// A flow-control operation's result. +pub type Result = std::result::Result; + +/// A dense, stable index in `0..COUNT`, unique to each resource class. +pub trait Class: Copy + Send + Sync + 'static { + /// Number of distinct resource classes. + const COUNT: usize; + + /// Return this class's unique index, strictly below `COUNT`. + fn index(self) -> usize; +} + +/// Application policy. Charges do not retain this value, so suitability and +/// release notifications are static properties of the class. +pub trait Policy: 'static { + /// Resource classes accounted by this policy. + type Class: Class; + + /// Identity used to divide local fair shares. + type Key: Clone + Eq + Hash + Send + Sync + 'static; + + /// Return the aggregate ceiling for a class. + fn limit(&self, class: Self::Class) -> usize; + + /// Return the minimum keyed share, clipped to the aggregate ceiling. + fn floor(&self, _class: Self::Class) -> usize { + 1 + } + + /// Bound the number of retained key records. + fn max_keys(&self) -> usize; + + /// Whether releasing this class wakes the shared admission waiter. + fn wakes(class: Self::Class) -> bool; + + /// Whether ordinary admission of this class remains allowed after stop. + fn allows_stopped(_class: Self::Class) -> bool { + false + } + + /// Whether this class can account for page allocator backing. + fn covers(class: Self::Class) -> bool; + + /// Observe each failed attempt, including one recovered by reclamation. + fn rejected(&self, rejection: Rejection); +} + +/// Facts observed at a failed admission attempt, before any reclamation retry. +#[derive(Clone, Copy, Debug)] +pub enum Rejection { + /// The bounded key table has no room for a new identity. + Keys { + /// Records still occupying the bounded key table. + used: usize, + + /// Maximum retained records allowed by policy. + limit: usize, + }, + + /// An aggregate or keyed share would be exceeded. + Resource { + class: C, + + used: usize, + + limit: usize, + + requested: usize, + + key_used: Option, + + key_limit: Option, + }, +} + +/// Worker-local keyed admission and recycler authority; never moves across workers. +/// +/// Only shared handles and charges may cross threads: +/// +/// ```compile_fail +/// use flow_control::{Policy, Quotas}; +/// fn move_authority(authority: Quotas

) { +/// fn require_send(_: T) {} +/// require_send(authority); +/// } +/// ``` +/// +/// ```compile_fail +/// use flow_control::{Policy, Quotas}; +/// fn share_authority(authority: &Quotas

) { +/// fn require_sync(_: &T) {} +/// require_sync(authority); +/// } +/// ``` +pub struct Quotas { + policy: P, + + totals: Arc>, + + keys: Keys, + + active_keys: Arc, + + retired_keys: Arc>>, + + stopped: Arc, + + buffers: Arc>, + + local: PhantomData>, +} + +/// Owns a live charge independently of the local quota authority and policy. +pub struct Charge { + class: P::Class, + + amount: usize, + + key: Option, + + totals: Arc>, + + local: Option>>, + + buffers: Weak>, + + stopped: Arc, +} + +impl page_alloc::Charge for Charge

{ + /// Check page-backing suitability without transferring admission ownership. + fn covers(&self, bytes: usize) -> bool { + P::covers(self.class) && self.amount >= bytes + } +} + +/// An unkeyed admission and usage handle. It retains no recycler buffers. +/// It is Send + Sync when the policy is Send + Sync. +pub struct SharedQuotas { + policy: P, + + totals: Arc>, + + stopped: Arc, +} + +impl Clone for SharedQuotas

{ + /// Share aggregate counters and stop state without retaining recycled backing. + fn clone(&self) -> Self { + Self { + policy: self.policy.clone(), + totals: self.totals.clone(), + stopped: self.stopped.clone(), + } + } +} + +impl SharedQuotas

{ + /// Register the single shared admission waiter before checking capacity. + pub fn register(&self, waker: &std::task::Waker) { + self.totals.wake.register(waker); + } + + /// Read current aggregate usage, including retained allocations. + pub fn used(&self, class: P::Class) -> usize { + self.totals.counter(class).used() + } + + /// Return the policy's aggregate ceiling. + pub fn limit(&self, class: P::Class) -> usize { + self.policy.limit(class) + } + + /// Whether the authority has requested admission shutdown. + pub fn is_stopped(&self) -> bool { + self.stopped.load(Ordering::Acquire) + } + + /// Admit a nonzero unkeyed amount without retaining the local recycler. + pub fn reserve(&self, class: P::Class, amount: usize) -> Result> { + if self.is_stopped() && !P::allows_stopped(class) { + return Err(Error::Unavailable); + } + if amount == 0 { + return Err(Error::InvalidInput); + } + let limit = self.limit(class); + self.totals + .counter(class) + .reserve(amount, limit) + .map_err(|_| { + self.policy.rejected(Rejection::Resource { + class, + used: self.used(class), + limit, + requested: amount, + key_used: None, + key_limit: None, + }); + Error::Overloaded + })?; + Ok(Charge { + class, + amount, + key: None, + totals: self.totals.clone(), + local: None, + buffers: Weak::new(), + stopped: self.stopped.clone(), + }) + } +} + +impl Charge

{ + /// Return the amount still owned by this charge. + pub fn amount(&self) -> usize { + self.amount + } + + /// Return the resource class selected at admission. + pub fn class(&self) -> P::Class { + self.class + } + + /// Return the admitted key, or none for aggregate-only admission. + pub fn key(&self) -> Option<&P::Key> { + self.key.as_ref() + } + + /// Check class and amount; this does not establish authority provenance. + pub fn validate(&self, class: P::Class, amount: usize) -> Result<()> { + if self.class.index() != class.index() || amount > self.amount { + Err(Error::InvalidInput) + } else { + Ok(()) + } + } + + /// Divide an admitted working set without changing its aggregate charge. + pub fn split(&mut self, amount: usize) -> Result { + if amount == 0 || amount >= self.amount { + return Err(Error::InvalidInput); + } + Ok(self.transfer(amount, self.buffers.clone())) + } + + /// Caller must retain at least the capacity of every live backing allocation. + pub fn shrink(&mut self, amount: usize) -> Result<()> { + if amount == 0 || amount > self.amount { + return Err(Error::InvalidInput); + } + self.release_to(amount); + Ok(()) + } + + /// Obtain zeroed backing without transferring or subdividing this charge. + /// + /// This low-level API does not track other allocations made with the same + /// charge. The caller must retain sufficient admission for all live backing, + /// validate its class/key/provenance, and never shrink below live capacity. + pub fn buffer(&self, length: usize) -> Result> { + if length > self.amount { + return Err(Error::InvalidInput); + } + if let Some(pool) = self.buffers.upgrade() { + let mut pool = pool.lock().unwrap_or_else(|e| e.into_inner()); + if let Some(index) = pool + .iter() + .position(|(bytes, _)| bytes.capacity() == length) + { + let (bytes, old) = pool.swap_remove(index); + drop(old); + debug_assert_eq!(bytes.len(), length); + return Ok(bytes); + } + } + let mut bytes = Vec::new(); + bytes + .try_reserve_exact(length) + .map_err(|_| Error::Overloaded)?; + bytes.resize(length, 0); + Ok(bytes) + } + + /// Wipe before every retention check. At most two >=1MiB buffers retain live + /// charges; only the final exclusive payload owner may return an allocation. + pub fn recycle(&mut self, mut bytes: Vec) { + wipe_payload(&mut bytes); + if self.stopped.load(Ordering::Acquire) + || bytes.capacity() < 1024 * 1024 + || bytes.capacity() > self.amount + { + return; + } + let Some(pool) = self.buffers.upgrade() else { + return; + }; + let Ok(mut pool) = pool.try_lock() else { + return; + }; + if self.stopped.load(Ordering::Acquire) || pool.len() >= 2 { + return; + } + // SAFETY: wipe_payload initialized the entire capacity above. + unsafe { bytes.set_len(bytes.capacity()) }; + let charge = self.transfer(self.amount, Weak::new()); + pool.push((bytes, charge)); + } + + /// Move admission to a new owner without touching either usage counter. + fn transfer(&mut self, amount: usize, buffers: Weak>) -> Self { + // Clone the application key before mutating ownership: its Clone may panic. + let key = self.key.clone(); + self.amount -= amount; + Self { + class: self.class, + amount, + key, + totals: self.totals.clone(), + local: self.local.clone(), + buffers, + stopped: self.stopped.clone(), + } + } + + /// Return released admission without notifying the shared waiter. + fn release_to(&mut self, amount: usize) { + let released = self.amount - amount; + self.amount = amount; + self.totals.counter(self.class).release(released); + if let Some(local) = &self.local { + local.counter(self.class).release(released); + } + } +} + +impl Drop for Charge

{ + /// Release exactly this charge's remaining amount and apply wake policy. + fn drop(&mut self) { + self.release_to(0); + if P::wakes(self.class) { + self.totals.wake.wake(); + } + } +} + +impl Quotas

{ + /// Create a worker-local authority with no admitted work or retained backing. + pub fn new(policy: P) -> Self { + Self { + policy, + totals: Arc::new(Counters::new(P::Class::COUNT)), + keys: RefCell::default(), + active_keys: Arc::new(AtomicUsize::new(0)), + retired_keys: Arc::new(Mutex::new(VecDeque::new())), + stopped: Arc::new(AtomicBool::new(false)), + buffers: Arc::new(Mutex::new(Vec::new())), + local: PhantomData, + } + } + + /// Borrow application policy without transferring local authority. + pub fn policy(&self) -> &P { + &self.policy + } + + /// Create a transferable unkeyed handle without retaining recycler storage. + pub fn shared(&self) -> SharedQuotas

+ where + P: Clone, + { + SharedQuotas { + policy: self.policy.clone(), + totals: self.totals.clone(), + stopped: self.stopped.clone(), + } + } + + /// Read aggregate usage, including idle resources and completion owners. + pub fn used(&self, class: P::Class) -> usize { + self.totals.counter(class).used() + } + + /// Return the aggregate ceiling chosen by policy. + pub fn limit(&self, class: P::Class) -> usize { + self.policy.limit(class) + } + + /// Check that a charge originated from this authority's counters. + pub fn owns(&self, charge: &Charge

) -> bool { + Arc::ptr_eq(&self.totals, &charge.totals) + } + + /// Whether shutdown has been requested for this authority. + pub fn is_stopped(&self) -> bool { + self.stopped.load(Ordering::Acquire) + } + + /// Stop ordinary admission, reclaim idle backing, and wake the shared waiter. + pub fn stop(&self) { + self.stopped.store(true, Ordering::Release); + self.reclaim_buffers(); + self.totals.wake.wake(); + } + + /// A keyed fair-share deficit must be reclaimed from that key. Global + /// pressure can use any idle key; impossible requests have no byte remedy. + pub fn reclamation( + &self, + key: &P::Key, + class: P::Class, + amount: usize, + ) -> Option<(Option, usize)> { + let keys = self.keys.borrow(); + let local = keys.get(key).and_then(Weak::upgrade); + let fair = self.fair_limit( + class, + self.active_keys.load(Ordering::Acquire) + usize::from(local.is_none()), + ); + if amount > fair { + return None; + } + let local_deficit = local + .as_ref() + .map_or(0, |local| local.counter(class).used()) + .saturating_sub(fair - amount); + if local_deficit != 0 { + return Some((Some(key.clone()), local_deficit)); + } + let deficit = self.used(class).saturating_sub(self.limit(class) - amount); + (deficit != 0).then_some((None, deficit)) + } + + /// Admit nonzero keyed or aggregate-only work under ordinary policy. + pub fn reserve( + &self, + key: Option<&P::Key>, + class: P::Class, + amount: usize, + ) -> Result> { + self.reserve_reclaiming(key, class, amount, AdmissionMode::Ordinary) + } + + /// For already-admitted work during drain only. Aggregate limits still apply. + pub fn reserve_completion( + &self, + key: Option<&P::Key>, + class: P::Class, + amount: usize, + ) -> Result> { + self.reserve_reclaiming(key, class, amount, AdmissionMode::Completion) + } + + /// Sum admission retained by idle recycler allocations. + pub fn retained_buffer_bytes(&self) -> usize { + self.buffers + .lock() + .unwrap_or_else(|e| e.into_inner()) + .iter() + .map(|(_, charge)| charge.amount()) + .sum() + } + + /// Release every idle recycler allocation and its charge. + pub fn reclaim_buffers(&self) { + self.buffers + .lock() + .unwrap_or_else(|e| e.into_inner()) + .clear(); + } + + /// Divide the aggregate ceiling, honoring the clipped per-key floor. + fn fair_limit(&self, class: P::Class, active: usize) -> usize { + let limit = self.limit(class); + (limit / active.max(1)) + .max(self.policy.floor(class)) + .min(limit) + } + + /// Retry overload once after reclaiming relevant retained backing. + fn reserve_reclaiming( + &self, + key: Option<&P::Key>, + class: P::Class, + amount: usize, + mode: AdmissionMode, + ) -> Result> { + let result = self.reserve_inner(key, class, amount, mode); + if matches!(result, Err(Error::Overloaded)) { + self.reclaim_buffers_for(key, class); + return self.reserve_inner(key, class, amount, mode); + } + result + } + + /// Reclaim only when key records, fairness, or this class can benefit. + fn reclaim_buffers_for(&self, key: Option<&P::Key>, class: P::Class) { + let mut buffers = self.buffers.lock().unwrap_or_else(|e| e.into_inner()); + // Keyed admission can exhaust records or fair shares too. Unkeyed + // admission only reclaims when this class has an idle charge. + if key.is_some() + || buffers + .iter() + .any(|(_, charge)| charge.class.index() == class.index()) + { + buffers.clear(); + } + } + + /// Perform one admission attempt and report its rejection before retrying. + fn reserve_inner( + &self, + key: Option<&P::Key>, + class: P::Class, + amount: usize, + mode: AdmissionMode, + ) -> Result> { + if self.is_stopped() && mode == AdmissionMode::Ordinary && !P::allows_stopped(class) { + return Err(Error::Unavailable); + } + if amount == 0 { + return Err(Error::InvalidInput); + } + let limit = self.limit(class); + let local = key.map(|key| self.key_counters(key)).transpose()?; + if let Some(local) = local.as_ref().filter(|_| mode == AdmissionMode::Ordinary) { + let fair = self.fair_limit(class, self.active_keys.load(Ordering::Acquire).max(1)); + let used = local.counter(class).used(); + if used.checked_add(amount).is_none_or(|next| next > fair) { + self.rejected(class, amount, Some(used), Some(fair)); + return Err(Error::Overloaded); + } + } + self.totals + .counter(class) + .reserve(amount, limit) + .map_err(|_| { + self.rejected(class, amount, None, None); + Error::Overloaded + })?; + if let Some(local) = &local { + local.counter(class).add(amount); + } + Ok(Charge { + class, + amount, + key: key.cloned(), + totals: self.totals.clone(), + local, + buffers: Arc::downgrade(&self.buffers), + stopped: self.stopped.clone(), + }) + } + + /// Reuse a live key record or create one after bounded retirement cleanup. + fn key_counters(&self, key: &P::Key) -> Result>> { + let mut keys = self.keys.borrow_mut(); + let mut retired = self.retired_keys.lock().unwrap_or_else(|e| e.into_inner()); + for _ in 0..256 { + let Some(id) = retired.pop_front() else { + break; + }; + if keys + .get(&id) + .is_some_and(|counts| counts.strong_count() == 0) + { + keys.remove(&id); + } + } + drop(retired); + if !keys.contains_key(key) && keys.len() >= self.policy.max_keys() { + self.policy.rejected(Rejection::Keys { + used: keys.len(), + limit: self.policy.max_keys(), + }); + return Err(Error::Overloaded); + } + if let Some(counts) = keys.get(key).and_then(Weak::upgrade) { + return Ok(counts); + } + let counts = Arc::new(Counters::keyed( + P::Class::COUNT, + key.clone(), + self.active_keys.clone(), + self.retired_keys.clone(), + )); + keys.insert(key.clone(), Arc::downgrade(&counts)); + Ok(counts) + } + + /// Report the current aggregate and optional keyed rejection facts. + fn rejected( + &self, + class: P::Class, + requested: usize, + key_used: Option, + key_limit: Option, + ) { + self.policy.rejected(Rejection::Resource { + class, + used: self.used(class), + limit: self.limit(class), + requested, + key_used, + key_limit, + }); + } +} + +/// Fixed initialized backing paired with its live quota charge. +pub struct ChargedBuffer { + bytes: Vec, + + charge: Option>, +} + +impl ChargedBuffer

{ + /// Allocate admitted backing; the caller validates class, key, and provenance. + pub fn new(mut charge: Charge

, length: usize) -> Result { + if length == 0 { + return Err(Error::InvalidInput); + } + let bytes = charge.buffer(length)?; + // Do not box or resize again while I/O pointers are live. + let bytes = bytes.into_boxed_slice().into_vec(); + charge.shrink(bytes.len())?; + Ok(Self { + bytes, + charge: Some(charge), + }) + } + + /// Borrow the charge retained for the fixed allocation. + pub fn charge(&self) -> &Charge

{ + self.charge.as_ref().expect("owned charge") + } + + /// Borrow initialized bytes without changing allocation geometry. + pub fn bytes(&self) -> &[u8] { + &self.bytes + } + + /// Mutate initialized bytes without changing allocation geometry. + pub fn bytes_mut(&mut self) -> &mut [u8] { + &mut self.bytes + } + + /// Transfer backing and admission together to a final completion owner. + pub fn into_parts(mut self) -> (Box<[u8]>, Charge

) { + ( + std::mem::take(&mut self.bytes).into_boxed_slice(), + self.charge.take().expect("owned charge"), + ) + } +} + +impl Drop for ChargedBuffer

{ + /// Wipe and optionally retain backing before its admission is released. + fn drop(&mut self) { + if let Some(charge) = &mut self.charge { + charge.recycle(std::mem::take(&mut self.bytes)); + } + } +} + +// SAFETY: private fixed backing and charge remain exclusively owned. +unsafe impl uring_runtime::reactor::IoBuffer for ChargedBuffer

{ + /// Buffer access uses the crate's application-independent error type. + type Error = Error; + + /// Borrow stable initialized backing while the runtime owns this buffer. + fn bytes(&self) -> Result<&[u8]> { + Ok(&self.bytes) + } + + /// Borrow stable mutable backing while the runtime owns this buffer. + fn bytes_mut(&mut self) -> Result<&mut [u8]> { + Ok(&mut self.bytes) + } +} + +/// Pending and issued reservations consume the same byte and slot budgets. +pub struct Window { + slots: usize, + + bytes: u64, + + max_item: u64, + + used: u64, + + outstanding: std::collections::BTreeMap, +} + +impl Window { + /// Require positive limits; the item ceiling may exceed the byte budget. + pub fn new(slots: usize, bytes: u64, max_item: u64) -> Result { + if slots == 0 || bytes == 0 || max_item == 0 { + return Err(Error::InvalidInput); + } + Ok(Self { + slots, + bytes, + max_item, + used: 0, + outstanding: std::collections::BTreeMap::new(), + }) + } + + /// Whether a valid length fits both limits, independent of item identity. + pub fn can_reserve(&self, length: u64) -> bool { + length != 0 + && length <= self.max_item + && self.outstanding.len() < self.slots + && length <= self.bytes - self.used + } + + /// Reserve a unique item without returning capacity on later issuance. + pub fn reserve(&mut self, key: K, length: u64) -> Result<()> { + if length == 0 || length > self.max_item { + return Err(Error::InvalidInput); + } + if !self.can_reserve(length) || self.outstanding.contains_key(&key) { + return Err(Error::Overloaded); + } + self.outstanding.insert(key, Credit::Pending(length)); + self.used += length; + Ok(()) + } + + /// Issue a pending item exactly once while retaining its full reservation. + pub fn issued(&mut self, key: K) -> Result<()> { + let entry = self.outstanding.get_mut(&key).ok_or(Error::InvalidInput)?; + let Credit::Pending(length) = *entry else { + return Err(Error::InvalidInput); + }; + *entry = Credit::Issued(length); + Ok(()) + } + + /// Release an issued item exactly once using its admitted length. + pub fn release(&mut self, key: K, length: u64) -> Result<()> { + if self.outstanding.get(&key) != Some(&Credit::Issued(length)) { + return Err(Error::InvalidInput); + } + self.outstanding.remove(&key); + self.used -= length; + Ok(()) + } + + /// Whether neither pending nor issued items retain credit. + pub fn is_empty(&self) -> bool { + self.outstanding.is_empty() + } +} + +/// A credit's lifecycle; only an exact issued credit can be released. +#[derive(Eq, PartialEq)] +enum Credit { + /// Reserved capacity whose item has not yet been issued. + Pending(u64), + + /// Issued capacity awaiting an exact-length acknowledgment. + Issued(u64), +} + +/// Admission intent determines whether existing work may finish during drain. +#[derive(Clone, Copy, Eq, PartialEq)] +enum AdmissionMode { + /// New work obeys stop state and keyed fair shares. + Ordinary, + + /// Existing work bypasses stop and fairness, but not hard limits. + Completion, +} + +/// Retained zeroed backing together with the charge that still accounts for it. +type Buffers

= Mutex, Charge

)>>; + +/// Worker-local key lookup; charges retain records independently. +type Keys = RefCell>>>; + +/// Isolate independently updated resource counters on separate cache lines. +#[repr(align(64))] +struct Counter(AtomicUsize); + +impl Counter { + /// Start one resource class with no admitted usage. + fn new() -> Self { + Self(AtomicUsize::new(0)) + } + + /// Read usage published by admission and cross-thread release. + fn used(&self) -> usize { + self.0.load(Ordering::Acquire) + } + + /// Add usage only when both arithmetic and the aggregate ceiling allow it. + fn reserve(&self, amount: usize, limit: usize) -> Result<()> { + self.0 + .fetch_update(Ordering::AcqRel, Ordering::Acquire, |used| { + used.checked_add(amount).filter(|next| *next <= limit) + }) + .map(|_| ()) + .map_err(|_| Error::Overloaded) + } + + /// Record keyed usage already covered by aggregate admission. + fn add(&self, amount: usize) { + self.0.fetch_add(amount, Ordering::AcqRel); + } + + /// Return usage owned exclusively by the releasing charge. + fn release(&self, amount: usize) { + self.0.fetch_sub(amount, Ordering::AcqRel); + } +} + +/// Shared counters outliving the worker-local authority while charges remain. +struct Counters { + retirement: Option>, + + used: Box<[Counter]>, + + wake: futures::task::AtomicWaker, +} + +impl Counters { + /// Allocate zeroed, independently padded counters for all classes. + fn new(classes: usize) -> Self { + Self { + retirement: None, + used: (0..classes).map(|_| Counter::new()).collect(), + wake: futures::task::AtomicWaker::new(), + } + } + + /// Couple one live key record to its eventual retirement notification. + fn keyed( + classes: usize, + key: K, + active: Arc, + queue: Arc>>, + ) -> Self { + let mut counters = Self::new(classes); + active.fetch_add(1, Ordering::AcqRel); + counters.retirement = Some(Retirement { key, active, queue }); + counters + } + + /// Select the padded counter using the policy's stable class index. + fn counter(&self, class: C) -> &Counter { + &self.used[class.index()] + } +} + +impl Drop for Counters { + /// Retire the complete keyed state when its final charge leaves. + fn drop(&mut self) { + if let Some(retirement) = self.retirement.take() { + retirement.retire(); + } + } +} + +/// One live key record owns both its active count and its cleanup notification. +struct Retirement { + key: K, + + active: Arc, + + queue: Arc>>, +} + +impl Retirement { + /// Publish retirement only after the last owner releases the key record. + fn retire(self) { + self.active.fetch_sub(1, Ordering::AcqRel); + self.queue + .lock() + .unwrap_or_else(|e| e.into_inner()) + .push_back(self.key); + } +} + +/// Initialize and wipe the full allocation, including truncated and spare bytes. +fn wipe_payload(bytes: &mut Vec) { + bytes.clear(); + #[cfg(all(target_os = "linux", any(target_env = "gnu", target_env = "musl")))] + if bytes.capacity() != 0 { + // SAFETY: the exclusive Vec owns capacity writable bytes. explicit_bzero + // initializes spare capacity and cannot be removed as a dead store. + unsafe { libc::explicit_bzero(bytes.as_mut_ptr().cast(), bytes.capacity()) }; + } + #[cfg(not(all(target_os = "linux", any(target_env = "gnu", target_env = "musl"))))] + { + use zeroize::Zeroize; + bytes.zeroize(); + } +} + +/// Accounting, retirement, reclamation, and cross-thread release contracts. +#[cfg(test)] +mod quota_tests { + use super::*; + + /// Distinct payload, wake-enabled, and drain-progress fixture classes. + #[derive(Clone, Copy, Debug)] + enum Resource { + Payload, + + Other, + + Progress, + } + + impl Class for Resource { + /// Number of fixture resource classes. + const COUNT: usize = 3; + + /// Map each fixture class to its stable counter. + fn index(self) -> usize { + self as usize + } + } + + /// Fixed limits and a shared rejection log for assertions. + #[derive(Clone)] + struct TestPolicy { + limit: usize, + + max_keys: usize, + + rejected: Arc>>>, + } + + impl TestPolicy { + /// Construct fixed aggregate limits with an empty rejection log. + fn new(limit: usize, max_keys: usize) -> Self { + Self { + limit, + max_keys, + rejected: Arc::default(), + } + } + } + + impl Policy for TestPolicy { + /// Fixture resource classes with separate usage counters. + type Class = Resource; + + /// Owned identities retained until all their charges leave. + type Key = String; + + /// All fixture classes use the same aggregate ceiling. + fn limit(&self, _: Resource) -> usize { + self.limit + } + + /// Return the fixture's key-record bound. + fn max_keys(&self) -> usize { + self.max_keys + } + + /// Only the non-payload fixture class wakes shared waiters. + fn wakes(class: Resource) -> bool { + matches!(class, Resource::Other) + } + + /// Only payload admission may cover page backing. + fn covers(class: Resource) -> bool { + matches!(class, Resource::Payload) + } + + /// Progress reservations remain available during drain. + fn allows_stopped(class: Resource) -> bool { + matches!(class, Resource::Progress) + } + + /// Preserve rejection order and facts for assertions. + fn rejected(&self, rejection: Rejection) { + self.rejected.lock().unwrap().push(rejection); + } + } + + /// Wiping initializes every allocated byte without moving or resizing backing. + #[test] + fn secure_payload_wipe_initializes_spare_capacity_and_preserves_geometry() { + for capacity in [0, 1, 15, 16, 17, 63, 64, 65, 4095, 4096, 4097] { + for initialized in [false, true] { + let mut bytes = Vec::with_capacity(capacity); + if initialized { + bytes.resize(bytes.capacity(), 0xa7); + bytes.truncate(capacity / 2); + } + let pointer = bytes.as_ptr(); + let allocated = bytes.capacity(); + wipe_payload(&mut bytes); + assert!(bytes.is_empty()); + assert_eq!(bytes.capacity(), allocated); + assert_eq!(bytes.as_ptr(), pointer); + // SAFETY: wipe_payload initializes every byte of the allocation. + unsafe { bytes.set_len(allocated) }; + assert!(bytes.iter().all(|byte| *byte == 0)); + wipe_payload(&mut bytes); + assert!(bytes.is_empty()); + } + } + } + + /// Compare full-capacity secure wiping on the same alternating workload. + #[test] + #[ignore = "release-only alternating full-capacity secure wipe comparison"] + #[allow(clippy::assertions_on_constants)] + fn secure_payload_wipe_benchmark() { + use std::{hint::black_box, time::Instant}; + use zeroize::Zeroize; + assert!(!cfg!(debug_assertions), "run with --release"); + const ITERATIONS: usize = 128; + for length in [1 << 20, 16 << 20, (16 << 20) + 16] { + let mut bytes = vec![0u8; length]; + for sample in 0..6 { + for optimized in if sample % 2 == 0 { + [false, true] + } else { + [true, false] + } { + let start = Instant::now(); + for _ in 0..ITERATIONS { + bytes.fill(black_box(0xa7)); + black_box(&bytes); + if optimized { + wipe_payload(&mut bytes); + } else { + bytes.clear(); + bytes.zeroize(); + } + // SAFETY: both primitives initialize the full capacity. + unsafe { bytes.set_len(length) }; + black_box(&bytes); + } + let elapsed = start.elapsed(); + assert!(bytes.iter().all(|byte| *byte == 0)); + if sample != 0 { + println!( + "secure_wipe length={length} optimized={optimized} sample={sample} iterations={ITERATIONS} ns_per_op={:.0}", + elapsed.as_nanos() as f64 / ITERATIONS as f64 + ); + } + } + } + } + } + + /// Fairness, retirement, and drain completion retain their separate limits. + #[test] + fn fairness_retirement_reclamation_and_completion() { + let quotas = Quotas::new(TestPolicy::new(100, 2)); + let (a, b, c) = ("a".to_owned(), "b".to_owned(), "c".to_owned()); + let first = quotas.reserve(Some(&a), Resource::Payload, 40).unwrap(); + let second = quotas.reserve(Some(&b), Resource::Payload, 40).unwrap(); + assert!(matches!( + quotas.reserve(Some(&a), Resource::Payload, 11), + Err(Error::Overloaded) + )); + assert_eq!( + quotas.reclamation(&a, Resource::Payload, 11), + Some((Some(a.clone()), 1)) + ); + assert_eq!(quotas.reclamation(&a, Resource::Payload, 51), None); + assert!(matches!( + quotas.reserve(Some(&c), Resource::Other, 1), + Err(Error::Overloaded) + )); + drop(second); + let third = quotas.reserve(Some(&a), Resource::Payload, 60).unwrap(); + assert_eq!(quotas.active_keys.load(Ordering::Acquire), 1); + assert!(!quotas.keys.borrow().contains_key(&b)); + assert_eq!( + quotas.reclamation(&a, Resource::Payload, 1), + Some((Some(a.clone()), 1)) + ); + let unkeyed = quotas.reserve(None, Resource::Other, 100).unwrap(); + assert_eq!(quotas.reclamation(&a, Resource::Other, 1), Some((None, 1))); + drop((unkeyed, first, third)); + let first = quotas.reserve(Some(&a), Resource::Payload, 60).unwrap(); + let second = quotas.reserve(Some(&b), Resource::Other, 1).unwrap(); + quotas.stop(); + assert!(matches!( + quotas.reserve(None, Resource::Payload, 1), + Err(Error::Unavailable) + )); + assert!(quotas.reserve(None, Resource::Progress, 1).is_ok()); + let completion = quotas + .reserve_completion(Some(&a), Resource::Payload, 40) + .unwrap(); + assert_eq!(quotas.used(Resource::Payload), 100); + assert!(matches!( + quotas.reserve_completion(None, Resource::Payload, 1), + Err(Error::Overloaded) + )); + drop((completion, first, second)); + assert_eq!(quotas.used(Resource::Payload), 0); + } + + /// Idle backing retains two live charges and reports pressure before retry. + #[test] + fn recycler_retains_two_live_charges_and_reports_pressure_before_retry() { + let size = 1 << 20; + let quotas = Quotas::new(TestPolicy::new(3 * size, 4)); + let mut buffers: Vec<_> = (0..3) + .map(|_| { + let charge = quotas.reserve(None, Resource::Payload, size).unwrap(); + let bytes = charge.buffer(size).unwrap(); + (charge, bytes) + }) + .collect(); + for (mut charge, mut bytes) in buffers.drain(..) { + bytes.fill(0xa7); + bytes.truncate(1); + charge.recycle(bytes); + } + assert_eq!(quotas.buffers.lock().unwrap().len(), 2); + assert_eq!(quotas.retained_buffer_bytes(), 2 * size); + assert!( + quotas + .buffers + .lock() + .unwrap() + .iter() + .all(|(b, _)| b.len() == size && b.iter().all(|v| *v == 0)) + ); + let shared = quotas.shared(); + let other = quotas.reserve(None, Resource::Other, 3 * size).unwrap(); + assert!(matches!( + quotas.reserve(None, Resource::Other, 1), + Err(Error::Overloaded) + )); + assert_eq!(quotas.retained_buffer_bytes(), 2 * size); + assert_eq!(quotas.policy.rejected.lock().unwrap().len(), 2); + let charge = quotas.reserve(None, Resource::Payload, 3 * size).unwrap(); + assert_eq!(quotas.retained_buffer_bytes(), 0); + assert_eq!(quotas.policy.rejected.lock().unwrap().len(), 3); + assert!( + matches!(quotas.policy.rejected.lock().unwrap()[2], Rejection::Resource { used, requested, .. } if used == 2 * size && requested == 3 * size) + ); + let mut charge = charge; + charge.recycle(vec![0xa7; 3 * size]); + drop((charge, other, quotas)); + assert_eq!( + shared.used(Resource::Payload), + 0, + "usage handles must not retain recycled buffers" + ); + } + + /// Key-table pressure reports original facts before successful reclamation. + #[test] + fn key_record_rejection_is_reported_before_reclaim_retry() { + let size = 1 << 20; + let quotas = Quotas::new(TestPolicy::new(2 * size, 1)); + let (a, b) = ("a".to_owned(), "b".to_owned()); + let mut old = quotas.reserve(Some(&a), Resource::Payload, size).unwrap(); + old.recycle(vec![0xa7; size]); + drop(old); + let charge = quotas.reserve(Some(&b), Resource::Other, 1).unwrap(); + assert_eq!(quotas.retained_buffer_bytes(), 0); + assert!(!quotas.keys.borrow().contains_key(&a)); + assert!(matches!( + "as.policy.rejected.lock().unwrap()[..], + [Rejection::Keys { used: 1, limit: 1 }] + )); + drop(charge); + } + + /// Charges and shared handles retain exact accounting across threads. + #[test] + fn charge_validation_and_thread_safe_shared_usage() { + /// Require transferable release and admission handles at compile time. + fn send_sync() {} + send_sync::>(); + send_sync::>(); + let quotas = Quotas::new(TestPolicy::new(100, 2)); + assert!(matches!( + quotas.reserve(None, Resource::Payload, 0), + Err(Error::InvalidInput) + )); + assert!(matches!( + quotas.reserve(None, Resource::Payload, usize::MAX), + Err(Error::Overloaded) + )); + let shared = quotas.shared(); + assert!(matches!( + shared.reserve(Resource::Payload, 0), + Err(Error::InvalidInput) + )); + let mut charge = shared.reserve(Resource::Payload, 100).unwrap(); + assert!(quotas.owns(&charge)); + assert!(!Quotas::new(TestPolicy::new(100, 2)).owns(&charge)); + assert!(page_alloc::Charge::covers(&charge, 100)); + assert!(!page_alloc::Charge::covers(&charge, 101)); + assert!(charge.validate(Resource::Payload, 100).is_ok()); + assert!(charge.validate(Resource::Other, 100).is_err()); + assert!(charge.split(100).is_err()); + assert!(charge.split(0).is_err()); + assert!(charge.shrink(0).is_err()); + assert!(charge.shrink(101).is_err()); + let split = charge.split(60).unwrap(); + charge.shrink(19).unwrap(); + assert_eq!(quotas.used(Resource::Payload), 79); + std::thread::spawn(move || drop(split)).join().unwrap(); + assert_eq!(shared.used(Resource::Payload), 19); + drop(charge); + let other = shared.reserve(Resource::Other, 1).unwrap(); + assert!(!page_alloc::Charge::covers(&other, 1)); + drop(other); + quotas.stop(); + assert!(shared.is_stopped()); + assert!(matches!( + shared.reserve(Resource::Payload, 1), + Err(Error::Unavailable) + )); + } + + /// Cross-thread release and local stop notify the registered shared waiter. + #[test] + fn release_and_stop_wake_shared_waiters() { + /// Count notifications without accessing worker-local state. + struct WakeCount(AtomicUsize); + + impl std::task::Wake for WakeCount { + /// Record one notification. + fn wake(self: Arc) { + self.0.fetch_add(1, Ordering::Relaxed); + } + } + let count = Arc::new(WakeCount(AtomicUsize::new(0))); + let waker = std::task::Waker::from(count.clone()); + let quotas = Quotas::new(TestPolicy::new(1, 1)); + let shared = quotas.shared(); + shared.register(&waker); + let charge = shared.reserve(Resource::Other, 1).unwrap(); + assert!(matches!( + shared.reserve(Resource::Other, 1), + Err(Error::Overloaded) + )); + std::thread::spawn(move || drop(charge)).join().unwrap(); + assert_eq!(count.0.load(Ordering::Relaxed), 1); + shared.register(&waker); + quotas.stop(); + assert_eq!(count.0.load(Ordering::Relaxed), 2); + } +} diff --git a/cmd/racer-dataplane/flow/src/pipe.rs b/cmd/racer-dataplane/flow/src/pipe.rs new file mode 100644 index 000000000..5c02e4c6b --- /dev/null +++ b/cmd/racer-dataplane/flow/src/pipe.rs @@ -0,0 +1,1248 @@ +//! Bounded, nonblocking, per-reader kernel pipes. +//! +//! Each lease owns both descriptors and its admission charge. Idle empty pipes +//! retain that charge in the worker-local pool. No descriptor or backing page is +//! recycled while a reader owns the lease. Writes copy into kernel +//! pipe pages before splice: socket acceptance is not a userspace page reuse fence. +use crate::{Charge, Error, Policy, Quotas, Result}; +use std::{ + cell::RefCell, + collections::VecDeque, + future::poll_fn, + io, + os::fd::{AsFd, AsRawFd, FromRawFd}, + rc::{Rc, Weak}, + task::{Poll, Waker}, +}; +use uring_runtime::reactor::descriptor::Descriptor; + +/// Maximum kernel buffer capacity per admitted reader. Pipe admission is in pipe +/// units, so total pipe capacity is bounded by the pipe quota * MAX_PIPE_BYTES. +/// Spliced bytes retained by sockets are subject to socket buffer limits instead. +pub const MAX_PIPE_BYTES: usize = 64 * 1024; + +/// Worker-local pipe admission and a FIFO of bounded, byte-charged waiters. +pub struct PipePool { + quotas: Rc>, + + pipe_class: P::Class, + + waiter_class: P::Class, + + waiter_limit: usize, + + waiting: Waiters, + + idle: Rc>>>, +} + +/// Shared local queue whose head alone attempts scheduled admission. +type Waiters = Rc>>>>>; + +/// Wakes the queue after resource return; holds no reactor reference. +struct Notify(Waiters); + +impl Drop for Notify { + /// Notify the oldest queued acquisition after lease resources are released. + fn drop(&mut self) { + wake_front(&self.0); + } +} + +/// Wake the FIFO head without holding a queue borrow through the callback. +fn wake_front(waiters: &Waiters) { + let wake = waiters + .borrow() + .front() + .and_then(|entry| entry.borrow().clone()); + if let Some(wake) = wake { + wake.wake(); + } +} + +/// Own a queue registration and its admission through cancellation or completion. +struct Waiting { + queue: Waiters, + + entry: Rc>>, + + _reservation: Charge

, +} + +impl Drop for Waiting

{ + /// Remove exactly this waiter and notify its successor without self-polling. + fn drop(&mut self) { + { + let mut queue = self.queue.borrow_mut(); + queue.retain(|entry| !Rc::ptr_eq(entry, &self.entry)); + if queue.is_empty() { + // Do not retain queue storage after its admission charges leave. + *queue = VecDeque::new(); + } + } + wake_front(&self.queue); + } +} + +/// Non-cloneable local ownership of both descriptors and their admission charge. +pub struct PipeLease { + resources: Option>, + + pool: Weak>>>, + + quotas: Weak>, + + _notify: Notify, +} + +/// Kernel pipe state; descriptors close before the trailing charge is released. +struct PipeResources { + read: Descriptor, + + write: Descriptor, + + capacity: usize, + + buffered: usize, + + // Declared after the descriptors so capacity is returned only after closing. + _reservation: Charge

, +} + +impl Drop for PipeLease

{ + /// Recycle only empty pipes on a live authority; close partial payloads. + fn drop(&mut self) { + // Reactor ownership keeps the lease alive through every accepted CQE. + // Never recycle canceled/partially drained payloads: close those pipes. + if let Some(resources) = self.resources.take() + && resources.buffered == 0 + && self.quotas.upgrade().is_some_and(|q| !q.is_stopped()) + && let Some(pool) = self.pool.upgrade() + { + pool.borrow_mut().push(resources); + } + // Notify drops after the pipe has been recycled or its charge released. + } +} + +impl PipePool

{ + /// Charge each live or idle pipe by one unit of `pipe_class`. Queued waits + /// charge bytes to `waiter_class`; `waiter_limit` bounds their count separately. + /// A zero waiter limit permits immediate acquisition only. + pub fn new( + quotas: Rc>, + pipe_class: P::Class, + waiter_class: P::Class, + waiter_limit: usize, + ) -> Self { + Self { + quotas, + pipe_class, + waiter_class, + waiter_limit, + waiting: Rc::default(), + idle: Rc::default(), + } + } + + /// Borrow the worker-local authority used by pipe and waiter admission. + pub fn quotas(&self) -> &Rc> { + &self.quotas + } + + /// Byte charge retained for each queue entry, excluding the caller's future. + /// The fixed allowance covers the queue slot, wake cell, and registration. + pub fn waiter_bytes(&self) -> usize { + std::mem::size_of::>() + 128 + } + + /// Empty retained pipes, still charged to the worker's fixed pipe budget. + pub fn idle_count(&self) -> usize { + self.idle.borrow().len() + } + + /// FIFO scheduling above immediate raw admission. At most `waiter_limit` wait + /// without pipes or new page acquisitions; each entry charges context bytes + /// for its guard, queue slot, wake cell, and cancellation registration. + /// `check` runs before acquisition and on every queued poll. `subscribe` runs + /// only after queue capacity and its byte charge are secured, never on the + /// immediate path. Its returned closure registers cancellation notification + /// with each poll's waker and owns the registration until this wait ends. + /// The caller must arrange polls for deadlines, quota stop, or charges held + /// outside this pool; pipe returns and queue removal wake the FIFO head. + /// No timer, polling loop, or self-wake is created here. + pub async fn acquire_wait( + &self, + mut check: Check, + subscribe: Subscribe, + ) -> std::result::Result, E> + where + E: From, + Check: FnMut() -> std::result::Result<(), E>, + Subscribe: FnOnce() -> std::result::Result, + Register: FnMut(&Waker), + { + check()?; + if self.waiting.borrow().is_empty() { + match self.acquire() { + Err(Error::Overloaded) => {} + result => return result.map_err(E::from), + } + } + if self.quotas.is_stopped() { + return Err(Error::Unavailable.into()); + } + if self.waiting.borrow().len() >= self.waiter_limit { + return Err(Error::Overloaded.into()); + } + let reservation = self + .quotas + .reserve(None, self.waiter_class, self.waiter_bytes())?; + let mut register = subscribe()?; + let entry = Rc::new(RefCell::new(None)); + self.waiting.borrow_mut().push_back(entry.clone()); + let waiting = Waiting { + queue: self.waiting.clone(), + entry, + _reservation: reservation, + }; + poll_fn(|cx| { + register(cx.waker()); + check()?; + if self.quotas.is_stopped() { + return Poll::Ready(Err(Error::Unavailable.into())); + } + *waiting.entry.borrow_mut() = Some(cx.waker().clone()); + if self + .waiting + .borrow() + .front() + .is_some_and(|entry| Rc::ptr_eq(entry, &waiting.entry)) + { + match self.acquire() { + Err(Error::Overloaded) => {} + result => return Poll::Ready(result.map_err(E::from)), + } + } + Poll::Pending + }) + .await + } + + /// Reserve before creating descriptors. Exhaustion never waits for a reader. + pub fn acquire(&self) -> Result> { + if self.quotas.is_stopped() { + return Err(Error::Unavailable); + } + if let Some(resources) = self.idle.borrow_mut().pop() { + return Ok(self.lease(resources)); + } + let reservation = self.quotas.reserve(None, self.pipe_class, 1)?; + reservation.validate(self.pipe_class, 1)?; + Ok(self.lease(PipeResources::new(reservation)?)) + } + + /// Attach local recycling and notification to uniquely owned resources. + fn lease(&self, resources: PipeResources

) -> PipeLease

{ + PipeLease { + resources: Some(resources), + pool: Rc::downgrade(&self.idle), + quotas: Rc::downgrade(&self.quotas), + _notify: Notify(self.waiting.clone()), + } + } +} + +impl PipeResources

{ + /// Create bounded descriptors, closing them before admission on any failure. + fn new(reservation: Charge

) -> Result { + #[cfg(feature = "simulation")] + if let Some(sim) = uring_runtime::reactor::simulation::Simulation::current() { + let (read, write) = sim.pipe(MAX_PIPE_BYTES); + return Ok(Self { + read, + write, + capacity: MAX_PIPE_BYTES, + buffered: 0, + _reservation: reservation, + }); + } + let mut fds = [-1; 2]; + // SAFETY: pipe2 initializes exactly two descriptors on success. + if unsafe { libc::pipe2(fds.as_mut_ptr(), libc::O_NONBLOCK | libc::O_CLOEXEC) } < 0 { + return Err(Error::Io); + } + // SAFETY: both descriptors were newly created and have unique owners. + let read = unsafe { Descriptor::from_raw_fd(fds[0]) }; + let write = unsafe { Descriptor::from_raw_fd(fds[1]) }; + // SAFETY: fcntl operates on a live descriptor and requires no pointer. + let mut capacity = unsafe { libc::fcntl(write.as_raw_fd(), libc::F_GETPIPE_SZ) }; + if capacity < 0 { + return Err(Error::Io); + } + if capacity as usize != MAX_PIPE_BYTES { + // Request one bounded chunk once at creation, before pooling. Under + // UID pipe pressure growth may fail; retain the smaller actual size. + // SAFETY: the empty pipe can be resized without borrowing user memory. + let resized = unsafe { + libc::fcntl(write.as_raw_fd(), libc::F_SETPIPE_SZ, MAX_PIPE_BYTES as i32) + }; + if resized > 0 { + capacity = resized; + } + } + if capacity <= 0 || capacity as usize > MAX_PIPE_BYTES { + return Err(Error::Io); + } + Ok(Self { + read, + write, + capacity: capacity as usize, + buffered: 0, + _reservation: reservation, + }) + } +} + +impl PipeLease

{ + /// Transit benefits from a full bounded chunk even when the UID's default + /// pipe size has shrunk. Failure to grow is harmless: use the actual capacity. + pub fn prepare_transit(&mut self) { + let pipe = self.resources.as_mut().unwrap(); + #[cfg(feature = "simulation")] + if pipe.write.as_sim().is_some() { + return; + } + if pipe.capacity < MAX_PIPE_BYTES && pipe.buffered == 0 { + // SAFETY: live, empty pipe, bounded integer capacity, no user pointer. + let capacity = unsafe { + libc::fcntl( + pipe.write.as_raw_fd(), + libc::F_SETPIPE_SZ, + MAX_PIPE_BYTES as i32, + ) + }; + if capacity > 0 { + pipe.capacity = capacity as usize; + } + } + } + + /// Receive opaque socket pages directly into an empty bounded pipe. No user + /// buffer is borrowed or retained by this synchronous nonblocking syscall. + /// The descriptor is validated as a nonblocking stream socket. Callers must + /// not concurrently clear O_NONBLOCK through a duplicate descriptor. + pub fn try_splice_from(&mut self, socket: &Descriptor, count: usize) -> io::Result { + #[cfg(feature = "simulation")] + if socket.as_sim().is_some() || self.resources.as_ref().unwrap().write.as_sim().is_some() { + return Err(io::Error::from_raw_os_error(libc::EOPNOTSUPP)); + } + validate_socket(socket)?; + let pipe = self.resources.as_mut().unwrap(); + let count = count.min(pipe.capacity - pipe.buffered); + // SAFETY: socket is a validated nonblocking stream; this pipe owns + // both ends, offsets are null, and no userspace pointer enters the kernel. + let received = syscall_count(unsafe { + libc::splice( + socket.as_raw_fd(), + std::ptr::null_mut(), + pipe.write.as_raw_fd(), + std::ptr::null_mut(), + count, + libc::SPLICE_F_NONBLOCK | libc::SPLICE_F_MOVE, + ) + })?; + pipe.buffered += received; + Ok(received) + } + + /// Return actual kernel capacity, which may be below the requested ceiling. + pub fn capacity(&self) -> usize { + self.resources.as_ref().unwrap().capacity + } + + /// Return the exact suffix still retained in this pipe. + pub fn buffered(&self) -> usize { + self.resources.as_ref().unwrap().buffered + } + + /// Copy at most the available capacity. WouldBlock and Interrupted are exposed + /// to the caller; this method never waits or retains a borrowed buffer. + pub fn try_write(&mut self, bytes: &[u8]) -> io::Result { + let pipe = self.resources.as_mut().unwrap(); + #[cfg(feature = "simulation")] + if let Some(handle) = pipe.write.as_sim() { + let written = handle.pipe_write(bytes)?; + pipe.buffered += written; + return Ok(written); + } + // SAFETY: the initialized slice stays live for this nonblocking syscall. + // The read end is owned by this lease, so this cannot generate SIGPIPE. + let written = unsafe { + libc::write( + pipe.write.as_raw_fd(), + bytes.as_ptr().cast(), + bytes.len().min(pipe.capacity), + ) + }; + let written = syscall_count(written)?; + pipe.buffered += written; + Ok(written) + } + + /// Read currently buffered bytes, or return WouldBlock for an empty pipe. + pub fn try_read(&mut self, bytes: &mut [u8]) -> io::Result { + let pipe = self.resources.as_mut().unwrap(); + #[cfg(feature = "simulation")] + if let Some(handle) = pipe.read.as_sim() { + let read = handle.pipe_read(bytes)?; + pipe.buffered -= read; + return Ok(read); + } + // SAFETY: the destination is exclusively borrowed until read returns. + let read = unsafe { + libc::read( + pipe.read.as_raw_fd(), + bytes.as_mut_ptr().cast(), + bytes.len(), + ) + }; + let read = syscall_count(read)?; + pipe.buffered -= read; + Ok(read) + } + + /// Splice a bounded suffix using production or simulated runtime descriptors. + pub fn try_splice_descriptor( + &mut self, + socket: &Descriptor, + count: usize, + ) -> io::Result { + #[cfg(feature = "simulation")] + { + let pipe = self.resources.as_mut().unwrap(); + match (pipe.read.as_sim(), socket.as_sim()) { + (Some(read), Some(socket)) => { + let sent = read.splice(socket, count.min(pipe.buffered))?; + pipe.buffered -= sent; + return Ok(sent); + } + (None, None) => {} + _ => return Err(io::Error::from_raw_os_error(libc::EOPNOTSUPP)), + } + } + self.try_splice_to(socket, count) + } + + /// Transfer copied kernel pipe bytes to a nonblocking stream socket. The + /// caller retains both owners through this synchronous syscall. Unsupported + /// splice errors leave the bytes in the pipe for an independent copy fallback. + /// Callers must not concurrently clear O_NONBLOCK through a duplicate FD. + pub fn try_splice_to(&mut self, socket: &impl AsFd, count: usize) -> io::Result { + validate_socket(socket)?; + self.splice_to_fd(socket.as_fd().as_raw_fd(), count) + } + + /// Drain to a stream socket. Like `try_splice_to`, this safe API validates + /// O_NONBLOCK and SO_TYPE; a Descriptor alone does not prove either property. + /// Callers must not concurrently clear O_NONBLOCK through a duplicate FD. + pub fn try_splice_connection(&mut self, socket: &Descriptor) -> io::Result { + self.try_splice_descriptor(socket, self.buffered()) + } + + /// Drain a validated socket while masking only this thread's generated SIGPIPE. + fn splice_to_fd(&mut self, fd: libc::c_int, count: usize) -> io::Result { + let pipe = self.resources.as_mut().unwrap(); + if count == 0 || pipe.buffered == 0 { + return Ok(0); + } + // Unlike send, splice has no MSG_NOSIGNAL. Mask only on this worker and + // only across the syscall, consuming our own EPIPE signal before restore. + let signal = SigpipeGuard::block()?; + // SAFETY: owned live FDs, null offsets for pipe/socket, no userspace page + // pointers, and both ends are nonblocking. No vmsplice/GIFT is involved. + let result = syscall_count(unsafe { + libc::splice( + pipe.read.as_raw_fd(), + std::ptr::null_mut(), + fd, + std::ptr::null_mut(), + count.min(pipe.buffered), + libc::SPLICE_F_NONBLOCK, + ) + }); + if result + .as_ref() + .is_err_and(|e| e.raw_os_error() == Some(libc::EPIPE)) + { + signal.consume_generated(); + } + drop(signal); + if let Ok(sent) = result { + pipe.buffered -= sent; + } + result + } +} + +/// Require a live nonblocking stream socket before synchronous splice. +fn validate_socket(socket: &impl AsFd) -> io::Result<()> { + let fd = socket.as_fd().as_raw_fd(); + // SPLICE_F_NONBLOCK controls only the pipe side. O_NONBLOCK alone does not + // prevent regular-file I/O from blocking, so require a stream socket too. + // SAFETY: AsFd borrows a live descriptor for this synchronous query. + let flags = unsafe { libc::fcntl(fd, libc::F_GETFL) }; + if flags < 0 { + return Err(io::Error::last_os_error()); + } + if flags & libc::O_NONBLOCK == 0 { + return Err(io::Error::from_raw_os_error(libc::EINVAL)); + } + let mut kind: libc::c_int = 0; + let mut length = std::mem::size_of_val(&kind) as libc::socklen_t; + // SAFETY: both output pointers reference correctly sized local values. + if unsafe { + libc::getsockopt( + fd, + libc::SOL_SOCKET, + libc::SO_TYPE, + (&mut kind as *mut libc::c_int).cast(), + &mut length, + ) + } < 0 + { + return Err(io::Error::last_os_error()); + } + if kind != libc::SOCK_STREAM { + return Err(io::Error::from_raw_os_error(libc::EINVAL)); + } + Ok(()) +} + +/// Temporarily mask SIGPIPE while preserving the thread's prior signal state. +struct SigpipeGuard { + previous: libc::sigset_t, + + set: libc::sigset_t, + + was_pending: bool, +} + +impl SigpipeGuard { + /// Save the thread mask, block SIGPIPE, and remember preexisting signals. + fn block() -> io::Result { + // SAFETY: all signal set pointers refer to initialized local storage. + unsafe { + let mut previous = std::mem::zeroed(); + let mut set = std::mem::zeroed(); + libc::sigemptyset(&mut set); + libc::sigaddset(&mut set, libc::SIGPIPE); + let error = libc::pthread_sigmask(libc::SIG_BLOCK, &set, &mut previous); + if error != 0 { + return Err(io::Error::from_raw_os_error(error)); + } + // Only a successfully installed mask creates a restoring owner. + let mut guard = Self { + previous, + set, + was_pending: true, + }; + let mut pending = std::mem::zeroed(); + if libc::sigpending(&mut pending) < 0 { + return Err(io::Error::last_os_error()); + } + guard.was_pending = libc::sigismember(&pending, libc::SIGPIPE) == 1; + Ok(guard) + } + } + + /// Consume only a newly generated signal without waiting. + fn consume_generated(&self) { + if self.was_pending { + return; + } + let timeout = libc::timespec { + tv_sec: 0, + tv_nsec: 0, + }; + // SAFETY: zero timeout never waits; only our thread's blocked SIGPIPE is + // consumed. Preserve a signal that was pending before entering the guard. + while unsafe { libc::sigtimedwait(&self.set, std::ptr::null_mut(), &timeout) } < 0 { + if io::Error::last_os_error().raw_os_error() != Some(libc::EINTR) { + break; + } + } + } +} + +impl Drop for SigpipeGuard { + /// Restore the exact mask saved after successful installation. + fn drop(&mut self) { + // SAFETY: restore the exact thread mask saved by successful pthread_sigmask. + unsafe { libc::pthread_sigmask(libc::SIG_SETMASK, &self.previous, std::ptr::null_mut()) }; + } +} + +/// Convert a syscall byte count while preserving its OS error. +fn syscall_count(value: isize) -> io::Result { + if value < 0 { + Err(io::Error::last_os_error()) + } else { + Ok(value as usize) + } +} + +/// Unsupported splice leaves buffered bytes available to a copy fallback. +pub fn splice_unsupported(error: &io::Error) -> bool { + matches!( + error.raw_os_error(), + Some(libc::EINVAL | libc::ENOSYS | libc::EOPNOTSUPP) + ) +} + +/// Kernel behavior, admission, cancellation, and simulation ownership contracts. +#[cfg(test)] +mod tests { + use super::*; + use std::{ + future::Future, + sync::{ + Arc, + atomic::{AtomicUsize, Ordering}, + }, + task::{Context, Wake}, + }; + + /// Independent pipe, waiter-context, and payload fixture resources. + #[derive(Clone, Copy)] + enum ResourceClass { + Pipe, + + RequestContext, + + Plaintext, + } + + impl crate::Class for ResourceClass { + const COUNT: usize = 3; + + /// Select the fixture's stable per-class counter. + fn index(self) -> usize { + self as usize + } + } + + /// Independent pipe and context limits without quota-driven wake policy. + struct TestPolicy { + pipes: usize, + + context: usize, + } + + impl Policy for TestPolicy { + type Class = ResourceClass; + + type Key = (); + + /// Give pipes their count ceiling and other classes a byte ceiling. + fn limit(&self, class: ResourceClass) -> usize { + match class { + ResourceClass::Pipe => self.pipes, + _ => self.context, + } + } + + /// All pipe fixture admission is unkeyed. + fn max_keys(&self) -> usize { + 0 + } + + /// Pipe return drives wakeups instead of shared quota release. + fn wakes(_: ResourceClass) -> bool { + false + } + + /// Pipe counts do not authorize userspace page backing. + fn covers(_: ResourceClass) -> bool { + false + } + + /// Rejection facts are not needed by these pipe assertions. + fn rejected(&self, _: crate::Rejection) {} + } + + /// Build a local authority with ample waiter-context capacity. + fn admission(pipes: usize) -> Rc> { + Rc::new(Quotas::new(TestPolicy { + pipes, + context: 32 * 1024 * 1024, + })) + } + + /// Build an eight-waiter pool using distinct pipe and context classes. + fn new_pool(quotas: Rc>) -> PipePool { + PipePool::new( + quotas, + ResourceClass::Pipe, + ResourceClass::RequestContext, + 8, + ) + } + + /// Create an acquisition with no cancellation or deadline source. + fn acquire_wait( + pool: &PipePool, + ) -> std::pin::Pin>> + '_>> { + Box::pin(pool.acquire_wait(|| Ok::<_, Error>(()), || Ok(|_: &Waker| {}))) + } + + /// Count progress notifications without requiring an executor. + #[derive(Default)] + struct WakeCounter(AtomicUsize); + + impl Wake for WakeCounter { + /// Count an owned wake notification. + fn wake(self: Arc) { + self.0.fetch_add(1, Ordering::Relaxed); + } + + /// Count a borrowed wake notification. + fn wake_by_ref(self: &Arc) { + self.0.fetch_add(1, Ordering::Relaxed); + } + } + + impl WakeCounter { + /// Read the number of observed progress notifications. + fn count(&self) -> usize { + self.0.load(Ordering::Relaxed) + } + } + + /// Distinguish application gate rejection from flow-control exhaustion. + #[derive(Debug, PartialEq)] + enum GateError { + Rejected, + + Flow(Error), + } + + impl From for GateError { + /// Preserve the underlying flow-control failure. + fn from(error: Error) -> Self { + Self::Flow(error) + } + } + + /// Unsupported operation errors never include backpressure or disconnects. + #[test] + fn unsupported_splice_is_distinct_from_backpressure_and_disconnect() { + for code in [libc::EINVAL, libc::ENOSYS, libc::EOPNOTSUPP] { + assert!(splice_unsupported(&io::Error::from_raw_os_error(code))); + } + for code in [libc::EAGAIN, libc::EINTR, libc::EPIPE, libc::ECONNRESET] { + assert!(!splice_unsupported(&io::Error::from_raw_os_error(code))); + } + assert!(!splice_unsupported(&io::Error::other("custom"))); + } + + /// Immediate acquisition skips subscription and failed waits roll back fully. + #[test] + fn lazy_subscription_and_failure_rollback() { + use std::cell::Cell; + let quotas = admission(1); + let pool = new_pool(quotas.clone()); + let subscribed = Cell::new(0); + let subscribe = || { + subscribed.set(subscribed.get() + 1); + Err::(GateError::Rejected) + }; + let mut cx = Context::from_waker(Waker::noop()); + let mut immediate = Box::pin(pool.acquire_wait(|| Ok(()), subscribe)); + let Poll::Ready(Ok(held)) = immediate.as_mut().poll(&mut cx) else { + panic!("immediate acquisition failed") + }; + assert_eq!(subscribed.get(), 0); + assert_eq!(quotas.used(ResourceClass::RequestContext), 0); + let mut wait = Box::pin(pool.acquire_wait(|| Ok(()), subscribe)); + assert!(matches!( + wait.as_mut().poll(&mut cx), + Poll::Ready(Err(GateError::Rejected)) + )); + assert_eq!(subscribed.get(), 1); + assert!(pool.waiting.borrow().is_empty()); + assert_eq!(quotas.used(ResourceClass::RequestContext), 0); + assert_eq!(quotas.used(ResourceClass::Pipe), 1); + drop(held); + + let mut rejected = Box::pin(pool.acquire_wait(|| Err(GateError::Rejected), subscribe)); + assert!(matches!( + rejected.as_mut().poll(&mut cx), + Poll::Ready(Err(GateError::Rejected)) + )); + assert_eq!(subscribed.get(), 1); + assert_eq!(pool.idle_count(), 1); + + for (context, waiter_limit) in [(0, 8), (usize::MAX, 0)] { + let quotas = Rc::new(Quotas::new(TestPolicy { pipes: 1, context })); + let pool = PipePool::new( + quotas.clone(), + ResourceClass::Pipe, + ResourceClass::RequestContext, + waiter_limit, + ); + let _held = pool.acquire().unwrap(); + let mut wait = Box::pin(pool.acquire_wait(|| Ok(()), subscribe)); + assert!(matches!( + wait.as_mut().poll(&mut cx), + Poll::Ready(Err(GateError::Flow(Error::Overloaded))) + )); + assert_eq!(subscribed.get(), 1, "rejected queue must not subscribe"); + assert_eq!(quotas.used(ResourceClass::RequestContext), 0); + assert!(pool.waiting.borrow().is_empty()); + } + } + + /// Gate failure, stop, and abandonment release registration and exact charges. + #[test] + fn gate_stop_and_abandonment_drop_registration_and_exact_charge() { + use std::cell::Cell; + /// Count a caller-owned cancellation registration until its closure drops. + struct Registration(Rc>); + + impl Drop for Registration { + /// Release exactly this registration's live count. + fn drop(&mut self) { + self.0.set(self.0.get() - 1); + } + } + for failure in 0..3 { + let quotas = admission(1); + let pool = new_pool(quotas.clone()); + let held = pool.acquire().unwrap(); + let reject = Cell::new(false); + let registrations = Rc::new(Cell::new(0)); + let registered = Cell::new(0); + let mut wait = Box::pin(pool.acquire_wait( + || { + if reject.get() { + Err(GateError::Rejected) + } else { + Ok(()) + } + }, + || { + registrations.set(registrations.get() + 1); + let registration = Registration(registrations.clone()); + let registered = ®istered; + Ok(move |_: &Waker| { + let _keep = ®istration; + registered.set(registered.get() + 1); + }) + }, + )); + let counter = Arc::new(WakeCounter::default()); + let waker = Waker::from(counter.clone()); + let mut cx = Context::from_waker(&waker); + assert!(wait.as_mut().poll(&mut cx).is_pending()); + assert!(wait.as_mut().poll(&mut cx).is_pending()); + assert_eq!(registered.get(), 2); + assert_eq!(registrations.get(), 1); + assert_eq!(counter.count(), 0); + assert_eq!( + quotas.used(ResourceClass::RequestContext), + pool.waiter_bytes() + ); + let mut second = acquire_wait(&pool); + assert!(second.as_mut().poll(&mut cx).is_pending()); + match failure { + 0 => { + reject.set(true); + assert!(matches!( + wait.as_mut().poll(&mut cx), + Poll::Ready(Err(GateError::Rejected)) + )); + } + 1 => { + quotas.stop(); + assert!(matches!( + wait.as_mut().poll(&mut cx), + Poll::Ready(Err(GateError::Flow(Error::Unavailable))) + )); + } + _ => {} + } + drop(wait); + assert_eq!(registrations.get(), 0); + assert_eq!( + quotas.used(ResourceClass::RequestContext), + pool.waiter_bytes() + ); + assert_eq!(pool.waiting.borrow().len(), 1); + assert!(counter.count() > 0, "head removal must wake successor"); + drop(held); + if failure == 1 { + assert!(matches!( + second.as_mut().poll(&mut cx), + Poll::Ready(Err(Error::Unavailable)) + )); + assert_eq!(quotas.used(ResourceClass::Pipe), 0); + } else { + assert!(matches!(second.as_mut().poll(&mut cx), Poll::Ready(Ok(_)))); + } + drop(second); + assert_eq!(quotas.used(ResourceClass::RequestContext), 0); + assert!(pool.waiting.borrow().is_empty()); + } + } + + /// Both descriptor splice directions reject blocking sockets without data loss. + #[test] + fn public_descriptor_splice_validates_both_directions() { + use std::{ + io::{Read, Write}, + os::unix::net::UnixStream, + }; + let pool = new_pool(admission(1)); + let mut pipe = pool.acquire().unwrap(); + let (socket, mut peer) = UnixStream::pair().unwrap(); + let socket = Descriptor::from(std::os::fd::OwnedFd::from(socket)); + pipe.try_write(b"out").unwrap(); + assert_eq!( + pipe.try_splice_connection(&socket) + .unwrap_err() + .raw_os_error(), + Some(libc::EINVAL) + ); + assert_eq!( + pipe.try_splice_from(&socket, 1).unwrap_err().raw_os_error(), + Some(libc::EINVAL) + ); + assert_eq!(pipe.buffered(), 3); + socket.set_nonblocking().unwrap(); + assert_eq!(pipe.try_splice_connection(&socket).unwrap(), 3); + let mut bytes = [0; 3]; + peer.read_exact(&mut bytes).unwrap(); + assert_eq!(&bytes, b"out"); + peer.write_all(b"in!").unwrap(); + pipe.prepare_transit(); + assert_eq!(pipe.try_splice_from(&socket, 3).unwrap(), 3); + assert_eq!(pipe.try_read(&mut bytes).unwrap(), 3); + assert_eq!(&bytes, b"in!"); + } + + /// Simulated descriptors preserve byte counts, pooling, and partial close. + #[cfg(feature = "simulation")] + #[test] + fn simulation_uses_runtime_descriptors_and_preserves_accounting() { + let sim = uring_runtime::reactor::simulation::Simulation::new(); + let _environment = sim.enter(); + let quotas = admission(1); + let pool = new_pool(quotas.clone()); + let mut pipe = pool.acquire().unwrap(); + assert!(pipe.resources.as_ref().unwrap().read.as_sim().is_some()); + assert_eq!(pipe.capacity(), MAX_PIPE_BYTES); + pipe.prepare_transit(); + assert_eq!(pipe.try_write(b"simulated").unwrap(), 9); + let (socket, peer) = sim.socket_pair(); + assert_eq!( + pipe.try_splice_from(&socket, 1).unwrap_err().raw_os_error(), + Some(libc::EOPNOTSUPP) + ); + assert_eq!(pipe.try_splice_descriptor(&socket, 3).unwrap(), 3); + assert_eq!(pipe.buffered(), 6); + let mut bytes = [0; 9]; + assert_eq!(peer.try_recv(&mut bytes).unwrap(), 3); + assert_eq!(&bytes[..3], b"sim"); + assert_eq!(pipe.try_read(&mut bytes).unwrap(), 6); + assert_eq!(&bytes[..6], b"ulated"); + drop(pipe); + assert_eq!(pool.idle_count(), 1); + assert_eq!(quotas.used(ResourceClass::Pipe), 1); + let mut pipe = pool.acquire().unwrap(); + pipe.try_write(b"discard").unwrap(); + drop(pipe); + assert_eq!(pool.idle_count(), 0); + assert_eq!(quotas.used(ResourceClass::Pipe), 0); + } + + /// Mixed descriptor backends reject splice without consuming buffered bytes. + #[cfg(feature = "simulation")] + #[test] + fn mixed_splice_backends_preserve_buffered_suffix() { + use std::os::unix::net::UnixStream; + + let pool = new_pool(admission(2)); + let mut real_pipe = pool.acquire().unwrap(); + let (real_socket, _peer) = UnixStream::pair().unwrap(); + real_socket.set_nonblocking(true).unwrap(); + let real_socket = Descriptor::from(std::os::fd::OwnedFd::from(real_socket)); + let sim = uring_runtime::reactor::simulation::Simulation::new(); + let _environment = sim.enter(); + let mut simulated_pipe = pool.acquire().unwrap(); + let (simulated_socket, _peer) = sim.socket_pair(); + + for (pipe, socket) in [ + (&mut real_pipe, &simulated_socket), + (&mut simulated_pipe, &real_socket), + ] { + assert_eq!(pipe.try_write(b"pending").unwrap(), 7); + assert_eq!( + pipe.try_splice_descriptor(socket, 3) + .unwrap_err() + .raw_os_error(), + Some(libc::EOPNOTSUPP) + ); + assert_eq!(pipe.buffered(), 7); + let mut bytes = [0; 7]; + assert_eq!(pipe.try_read(&mut bytes).unwrap(), 7); + assert_eq!(&bytes, b"pending"); + } + } + + /// Empty pipes reuse both descriptors while partial payloads are closed. + #[test] + fn empty_pipe_reuses_descriptors_and_partial_pipe_is_closed() { + let admission = admission(1); + let pool = new_pool(admission.clone()); + let mut pipe = pool.acquire().unwrap(); + let read = pipe.resources.as_ref().unwrap().read.as_raw_fd(); + let write = pipe.resources.as_ref().unwrap().write.as_raw_fd(); + pipe.try_write(b"secret").unwrap(); + let mut bytes = [0; 6]; + assert_eq!(pipe.try_read(&mut bytes).unwrap(), 6); + drop(pipe); + assert_eq!(admission.used(ResourceClass::Pipe), 1); + let mut pipe = pool.acquire().unwrap(); + assert_eq!(pipe.resources.as_ref().unwrap().read.as_raw_fd(), read); + assert_eq!(pipe.resources.as_ref().unwrap().write.as_raw_fd(), write); + assert_eq!( + pipe.try_read(&mut bytes).unwrap_err().kind(), + io::ErrorKind::WouldBlock + ); + pipe.try_write(b"discard").unwrap(); + drop(pipe); + assert!(pool.idle.borrow().is_empty()); + assert_eq!(admission.used(ResourceClass::Pipe), 0); + // SAFETY: only query the closed descriptor numbers, without reusing them. + assert_eq!(unsafe { libc::fcntl(read, libc::F_GETFD) }, -1); + assert_eq!(unsafe { libc::fcntl(write, libc::F_GETFD) }, -1); + } + + /// A lease outliving its pool continues to hold capacity until drop. + #[test] + fn exhaustion_and_drop_return_capacity_even_after_pool_drop() { + let admission = admission(1); + let pool = new_pool(admission.clone()); + let lease = pool.acquire().unwrap(); + assert!(matches!(pool.acquire(), Err(Error::Overloaded))); + drop(pool); + let pool = new_pool(admission); + assert!(matches!(pool.acquire(), Err(Error::Overloaded))); + drop(lease); + assert!(pool.acquire().is_ok()); + } + + /// Scheduled acquisition is bounded and FIFO without polling itself awake. + #[test] + fn scheduled_acquisition_is_bounded_fifo_and_wakes_only_for_progress() { + let admission = admission(2); + let pool = new_pool(admission.clone()); + let held = [pool.acquire().unwrap(), pool.acquire().unwrap()]; + let count = Arc::new(WakeCounter::default()); + let waker = Waker::from(count.clone()); + let mut cx = Context::from_waker(&waker); + let mut waiting: Vec<_> = (0..8).map(|_| acquire_wait(&pool)).collect(); + for wait in &mut waiting { + assert!(wait.as_mut().poll(&mut cx).is_pending()); + } + assert_eq!(count.count(), 0, "waiting does not spin/self-wake"); + assert!(matches!( + acquire_wait(&pool).as_mut().poll(&mut cx), + Poll::Ready(Err(Error::Overloaded)) + )); + assert_eq!(pool.waiting.borrow().len(), 8); + assert_eq!( + admission.used(ResourceClass::RequestContext), + 8 * pool.waiter_bytes() + ); + assert_eq!(admission.used(ResourceClass::Pipe), 2); + assert_eq!(admission.used(ResourceClass::Plaintext), 0); + drop(held); + assert!(count.count() > 0); + // Reverse polling cannot let new arrivals jump the queue. + for wait in waiting.iter_mut().skip(1).rev() { + assert!(wait.as_mut().poll(&mut cx).is_pending()); + } + let mut leases = VecDeque::new(); + for mut wait in waiting { + if leases.len() == 2 { + leases.pop_front(); + } + let Poll::Ready(Ok(pipe)) = wait.as_mut().poll(&mut cx) else { + panic!("FIFO waiter did not progress") + }; + leases.push_back(pipe); + assert!(admission.used(ResourceClass::Pipe) <= 2); + } + assert!(pool.waiting.borrow().is_empty()); + assert_eq!(admission.used(ResourceClass::RequestContext), 0); + drop(leases); + assert_eq!( + admission.used(ResourceClass::Pipe), + 2, + "idle pipes remain admitted" + ); + drop(pool); + assert_eq!(admission.used(ResourceClass::Pipe), 0); + } + + /// Kernel descriptors have bounded capacity and independent nonblocking data. + #[test] + fn pipes_are_nonblocking_bounded_cloexec_and_independent() { + let admission = admission(2); + let pool = new_pool(admission); + let mut first = pool.acquire().unwrap(); + let mut second = pool.acquire().unwrap(); + let resources = first.resources.as_ref().unwrap(); + for fd in [&resources.read, &resources.write] { + // SAFETY: these descriptors are owned for the duration of the query. + assert_ne!( + unsafe { libc::fcntl(fd.as_raw_fd(), libc::F_GETFL) } & libc::O_NONBLOCK, + 0 + ); + assert_ne!( + unsafe { libc::fcntl(fd.as_raw_fd(), libc::F_GETFD) } & libc::FD_CLOEXEC, + 0 + ); + } + assert!(first.capacity() <= MAX_PIPE_BYTES); + assert!(first.capacity() > 0); + // Account the actual kernel capacity, including denied best-effort growth. + assert_eq!(first.capacity(), unsafe { + libc::fcntl( + first.resources.as_ref().unwrap().write.as_raw_fd(), + libc::F_GETPIPE_SZ, + ) + } as usize); + let bytes = vec![0x5a; first.capacity()]; + assert_eq!(first.try_write(&bytes).unwrap(), bytes.len()); + assert_eq!( + first.try_write(b"x").unwrap_err().kind(), + io::ErrorKind::WouldBlock + ); + let mut out = vec![0; bytes.len()]; + assert_eq!( + second.try_read(&mut out).unwrap_err().kind(), + io::ErrorKind::WouldBlock + ); + assert_eq!(first.try_read(&mut out).unwrap(), bytes.len()); + assert_eq!(out, bytes); + assert_eq!( + first.try_read(&mut out).unwrap_err().kind(), + io::ErrorKind::WouldBlock + ); + assert_eq!(second.try_write(b"second").unwrap(), 6); + assert_eq!(second.try_read(&mut out).unwrap(), 6); + assert_eq!(&out[..6], b"second"); + } + + /// Copied pages remain intact after source reuse, partial drain, and pipe reuse. + #[test] + fn copied_splice_survives_source_reuse_and_partial_drain() { + use std::{io::Read, os::unix::net::UnixStream}; + let admission = admission(1); + let pool = new_pool(admission); + let mut pipe = pool.acquire().unwrap(); + let (socket, mut peer) = UnixStream::pair().unwrap(); + socket.set_nonblocking(true).unwrap(); + let mut source = b"copied kernel pages".to_vec(); + assert_eq!(pipe.try_write(&source).unwrap(), source.len()); + source.fill(0); + assert_eq!(pipe.try_splice_to(&socket, 6).unwrap(), 6); + assert_eq!(pipe.buffered(), 13); + assert_eq!(pipe.try_splice_to(&socket, usize::MAX).unwrap(), 13); + assert_eq!(pipe.buffered(), 0); + // Socket-owned kernel pages survive both pipe closure and quota reuse. + drop(pipe); + let mut reused = pool.acquire().unwrap(); + reused.try_write(b"replacement data").unwrap(); + let mut received = [0; 19]; + peer.read_exact(&mut received).unwrap(); + assert_eq!(&received, b"copied kernel pages"); + } + + /// Blocking sockets and disconnects preserve the full buffered suffix. + #[test] + fn splice_rejects_blocking_socket_and_handles_disconnect_without_losing_bytes() { + use std::os::unix::net::UnixStream; + let admission = admission(1); + let pool = new_pool(admission); + let mut pipe = pool.acquire().unwrap(); + pipe.try_write(b"abc").unwrap(); + let (socket, peer) = UnixStream::pair().unwrap(); + assert_eq!( + pipe.try_splice_to(&socket, 3).unwrap_err().raw_os_error(), + Some(libc::EINVAL) + ); + assert_eq!(pipe.buffered(), 3); + socket.set_nonblocking(true).unwrap(); + drop(peer); + assert_eq!( + pipe.try_splice_to(&socket, 3).unwrap_err().raw_os_error(), + Some(libc::EPIPE) + ); + assert_eq!(pipe.buffered(), 3); + let mut bytes = [0; 3]; + assert_eq!(pipe.try_read(&mut bytes).unwrap(), 3); + assert_eq!(&bytes, b"abc"); + } + + /// Backpressure and invalid socket types leave fallback bytes untouched. + #[test] + fn splice_backpressure_preserves_buffered_bytes_and_socket_validation() { + use std::{io::Write, os::unix::net::UnixStream}; + let admission = admission(1); + let pool = new_pool(admission); + let mut pipe = pool.acquire().unwrap(); + pipe.try_write(b"pending").unwrap(); + let (mut socket, _peer) = UnixStream::pair().unwrap(); + socket.set_nonblocking(true).unwrap(); + let start = std::time::Instant::now(); + loop { + assert!(start.elapsed() < std::time::Duration::from_secs(5)); + match socket.write(&[1; 8192]) { + Ok(count) => assert_ne!(count, 0), + Err(error) if error.kind() == io::ErrorKind::WouldBlock => break, + Err(error) => panic!("{error}"), + } + } + assert_eq!( + pipe.try_splice_to(&socket, 7).unwrap_err().kind(), + io::ErrorKind::WouldBlock + ); + assert_eq!(pipe.buffered(), 7); + let (datagram, _peer) = std::os::unix::net::UnixDatagram::pair().unwrap(); + datagram.set_nonblocking(true).unwrap(); + assert_eq!( + pipe.try_splice_to(&datagram, 7).unwrap_err().raw_os_error(), + Some(libc::EINVAL) + ); + let file = std::fs::OpenOptions::new() + .write(true) + .open("/dev/null") + .unwrap(); + // SAFETY: change flags on this exclusively owned test descriptor. + assert_eq!( + unsafe { libc::fcntl(file.as_raw_fd(), libc::F_SETFL, libc::O_NONBLOCK) }, + 0 + ); + assert_eq!( + pipe.try_splice_to(&file, 7).unwrap_err().raw_os_error(), + Some(libc::ENOTSOCK) + ); + let mut bytes = [0; 7]; + assert_eq!(pipe.try_read(&mut bytes).unwrap(), 7); + assert_eq!(&bytes, b"pending"); + } +} diff --git a/cmd/racer-dataplane/flow/tests/coalesce.rs b/cmd/racer-dataplane/flow/tests/coalesce.rs new file mode 100644 index 000000000..d15e09919 --- /dev/null +++ b/cmd/racer-dataplane/flow/tests/coalesce.rs @@ -0,0 +1,414 @@ +//! Public ownership boundaries and synchronous reentrant notification contracts. + +use flow_control::coalesce::flight::state::{Outcome, Phase, Published, State, WaiterPolicy}; +use flow_control::coalesce::flight::{self, Entry, Operations, Stale}; +use flow_control::coalesce::{CapacityError, Event, Limits, Table, shared}; +use futures::executor::block_on; +use std::cell::{Cell, RefCell}; +use std::future::Future; +use std::rc::Rc; +use std::sync::Arc; +use std::task::{Context, Poll, Wake, Waker}; +use std::time::Instant; + +thread_local! { + /// Callback invoked only by synchronous wakes on the current test worker. + static ON_WAKE: RefCell>> = RefCell::new(None); +} + +/// Safe thread-local callback dispatch; no non-Send data enters the Waker itself. +struct Reenter; + +impl Wake for Reenter { + /// Invoke the current worker's callback after releasing its registry borrow. + fn wake(self: Arc) { + let callback = ON_WAKE.with(|slot| slot.borrow_mut().take()); + if let Some(callback) = callback { + callback(); + } + } +} + +/// Install a callback and return a wake target that dispatches it synchronously. +fn on_wake(callback: impl FnOnce() + 'static) -> Waker { + ON_WAKE.with(|slot| assert!(slot.borrow_mut().replace(Box::new(callback)).is_none())); + Waker::from(Arc::new(Reenter)) +} + +/// Build a table with enough attempts to test retry and final-owner-drop elections. +fn table(waiters: usize) -> Rc> { + Rc::new(Table::new( + Limits { + waiters_per_cohort: waiters, + attempts_per_cohort: 4, + }, + 99, + )) +} + +/// Joining and cloning handles do not require a cloneable result value. +#[test] +fn registration_clones_neither_keys_nor_results() { + /// Key whose clone records the allocation-time copy only. + #[derive(Eq, PartialEq)] + struct Key(Rc>); + + impl Clone for Key { + /// Count each actual key copy. + fn clone(&self) -> Self { + self.0.set(self.0.get() + 1); + Self(self.0.clone()) + } + } + + impl std::hash::Hash for Key { + /// Give the single logical test key a stable hash. + fn hash(&self, state: &mut H) { + // This test uses exactly one logical key, independent of the counter. + state.write_u8(0); + } + } + + /// Deliberately non-Clone result; only polling needs result cloning. + struct ResultValue; + + let table = Rc::new(Table::::new( + Limits { + waiters_per_cohort: 1, + attempts_per_cohort: 1, + }, + ResultValue, + )); + let copies = Rc::new(Cell::new(0)); + let first = table.join(Key(copies.clone()), 1).unwrap(); + assert_eq!(copies.get(), 1); + let second = first.clone(); + let third = second.clone(); + assert_eq!(copies.get(), 1); + first.finish(ResultValue); + drop((first, second)); + assert_eq!(table.registration_count(), 1); + drop(third); + assert_eq!(table.registration_count(), 0); +} + +/// Arbitrarily many cloned handles consume one charge until their final drop. +#[test] +fn many_clones_retain_completed_capacity_and_do_not_remove_replacements() { + let table = table(2); + let first = table.join(1, 1).unwrap(); + let mut clones = (0..32).map(|_| first.clone()).collect::>(); + first.finish(7); + let next = table.join(1, 1).unwrap(); + drop(first); + while clones.len() > 1 { + drop(clones.pop()); + assert_eq!(table.registration_count(), 2); + assert!(matches!(table.join(1, 1), Err(CapacityError))); + } + assert!(clones[0].is_only_handle()); + assert_eq!( + clones[0].event(Waker::noop()), + Poll::Ready(Event::Complete(7)) + ); + drop(clones); + assert_eq!(table.registration_count(), 1); + assert_eq!(table.active_count(), 1); + let follower = table.join(1, 1).unwrap(); + assert_eq!(next.event(Waker::noop()), Poll::Ready(Event::Lead)); + drop((next, follower)); + assert_eq!(table.registration_count(), 0); + assert_eq!(table.active_count(), 0); +} + +/// Retry and final-owner drop release all borrows before reentrant leader election. +#[test] +fn retry_and_final_drop_allow_reentrant_election() { + for drop_leader in [false, true] { + let table = table(2); + let leader = table.join(1, 1).unwrap(); + let follower = table.join(1, 1).unwrap(); + assert_eq!(leader.event(Waker::noop()), Poll::Ready(Event::Lead)); + let called = Rc::new(Cell::new(false)); + let waker = on_wake({ + let table = table.clone(); + let follower = follower.clone(); + let called = called.clone(); + move || { + assert_eq!(table.registration_count(), if drop_leader { 1 } else { 2 }); + assert_eq!(follower.event(Waker::noop()), Poll::Ready(Event::Lead)); + called.set(true); + } + }); + assert!(follower.event(&waker).is_pending()); + if drop_leader { + drop(leader); + } else { + leader.retry(); + } + assert!(called.get()); + } +} + +/// A follower may request notifications without revoking another waiter's leadership. +#[test] +fn follower_retry_notifies_without_taking_leadership() { + let table = table(2); + let leader = table.join(1, 1).unwrap(); + let follower = table.join(1, 1).unwrap(); + assert_eq!(leader.event(Waker::noop()), Poll::Ready(Event::Lead)); + let notified = Rc::new(Cell::new(false)); + let waker = on_wake({ + let leader = leader.clone(); + let follower = follower.clone(); + let notified = notified.clone(); + move || { + assert!(follower.event(Waker::noop()).is_pending()); + assert!(leader.event(Waker::noop()).is_pending()); + notified.set(true); + } + }); + assert!(follower.event(&waker).is_pending()); + follower.retry(); + assert!(notified.get()); + leader.retry(); + assert_eq!(follower.event(Waker::noop()), Poll::Ready(Event::Lead)); +} + +/// Completion removes old admission before any reader wake can admit new work. +#[test] +fn finish_allows_reentrant_admission_before_old_readers_detach() { + let table = table(3); + let leader = table.join(1, 1).unwrap(); + let follower = table.join(1, 1).unwrap(); + assert_eq!(leader.event(Waker::noop()), Poll::Ready(Event::Lead)); + let replacement = Rc::new(RefCell::new(None)); + let waker = on_wake({ + let table = table.clone(); + let follower = follower.clone(); + let replacement = replacement.clone(); + move || { + assert_eq!(table.active_count(), 0); + assert_eq!( + follower.event(Waker::noop()), + Poll::Ready(Event::Complete(7)) + ); + replacement.replace(Some(table.join(1, 1).unwrap())); + } + }); + assert!(follower.event(&waker).is_pending()); + leader.finish(7); + assert!(replacement.borrow().is_some()); + drop((leader, follower)); + assert_eq!(table.active_count(), 1); + assert_eq!(table.registration_count(), 1); +} + +/// Shared result notification can synchronously start replacement work. +#[test] +fn shared_completion_wakes_after_removal_and_keeps_new_owner() { + let table = Rc::new(shared::Table::default()); + let (mut receive, complete) = table.start(1, 99); + let replacement = Rc::new(RefCell::new(None)); + let waker = on_wake({ + let table = table.clone(); + let replacement = replacement.clone(); + move || { + assert!(table.is_empty()); + replacement.replace(Some(table.start(1, 99))); + } + }); + assert!( + std::pin::Pin::new(&mut receive) + .poll(&mut Context::from_waker(&waker)) + .is_pending() + ); + complete.finish(7); + assert_eq!(block_on(receive), 7); + let (receive, complete) = replacement.borrow_mut().take().unwrap(); + assert_eq!(table.len(), 1); + drop(complete); + assert_eq!(block_on(receive), 99); + assert_eq!( + table.len(), + 1, + "sender loss cannot pretend execution completed" + ); +} + +/// Losing the sender wakes parked readers without removing the indexed work. +#[test] +fn shared_sender_drop_wakes_with_entry_still_indexed() { + let table = Rc::new(shared::Table::default()); + let (mut receive, complete) = table.start(1, 99); + let notified = Rc::new(Cell::new(false)); + let waker = on_wake({ + let table = table.clone(); + let notified = notified.clone(); + move || { + assert_eq!(table.len(), 1); + assert!(table.get(&1).is_some()); + notified.set(true); + } + }); + assert!( + std::pin::Pin::new(&mut receive) + .poll(&mut Context::from_waker(&waker)) + .is_pending() + ); + drop(complete); + assert!(notified.get()); + assert_eq!(block_on(receive), 99); + assert_eq!(block_on(table.get(&1).unwrap()), 99); + assert_eq!(table.len(), 1); +} + +/// Resource whose destructor can synchronously inspect its owning table. +struct ReentrantResource(Option>); + +impl Drop for ReentrantResource { + /// Reenter the owner while its operation tombstone must still be present. + fn drop(&mut self) { + self.0.take().unwrap()(); + } +} + +/// Only operation completion makes this deliberately waiter-free entry removable. +#[derive(Default)] +struct OwnedEntry(Operations); + +impl Entry for OwnedEntry { + /// This fixture has no caller policy to refresh. + fn refresh(&mut self, _: &mut Vec) {} + + /// Require explicit completion of every retained operation. + fn quiescent(&self) -> bool { + self.0.is_empty() + } +} + +/// Resource drop reentrancy cannot erase the tombstone, and drain wakes run unlocked. +#[test] +fn completion_tombstone_survives_reentrant_destructor_and_shutdown() { + let table = Rc::new(RefCell::new(flight::Table::::default())); + let dropped = Rc::new(Cell::new(false)); + let resource = ReentrantResource(Some(Box::new({ + let table = Rc::downgrade(&table); + let dropped = dropped.clone(); + move || { + let table = table.upgrade().unwrap(); + flight::update(&table, |table, wakes| { + table.sweep(1, wakes); + assert_eq!(table.len(), 1); + assert!(!table.remove_quiescent(&1)); + }); + dropped.set(true); + } + }))); + let id = flight::update(&table, |table, _| { + assert_eq!(table.next_waiter_id(), Ok(1)); + assert_eq!(table.next_waiter_id(), Ok(2)); + let id = table.next_operation_id().unwrap(); + table.insert(1, OwnedEntry::default()); + table.get_mut(&1).unwrap().0.insert(id, resource); + assert_eq!(table.get_mut(&1).unwrap().0.complete(id), Err(Stale)); + id + }); + flight::update(&table, |table, wakes| table.stop(wakes, |_, _| {})); + assert!(table.borrow().is_stopping()); + let resource = flight::update(&table, |table, _| { + let operations = &mut table.get_mut(&1).unwrap().0; + let resource = operations.take(id).unwrap(); + assert!(matches!(operations.take(id), Err(Stale))); + assert!(matches!(operations.take(id + 1), Err(Stale))); + assert_eq!(operations.complete(id + 1), Err(Stale)); + assert_eq!(operations.len(), 1); + resource + }); + drop(resource); + assert!(dropped.get()); + assert_eq!(table.borrow().len(), 1); + let notified = Rc::new(Cell::new(false)); + let waker = on_wake({ + let table = table.clone(); + let notified = notified.clone(); + move || { + assert!(table.borrow().is_empty()); + notified.set(true); + } + }); + flight::update(&table, |table, wakes| { + table.register_drain(&waker); + let operations = &mut table.get_mut(&1).unwrap().0; + operations.complete(id).unwrap(); + assert_eq!(operations.complete(id), Err(Stale)); + table.sweep(1, wakes); + }); + assert!(notified.get()); +} + +/// An expired leader is checked separately from the 64-entry deadline quantum. +#[test] +fn expired_leader_does_not_consume_deadline_quantum_or_clear_drain_fence() { + let mut state = State::, CountingPolicy>::default(); + let mut table = flight::Table::::default(); + let mut identity = table.identity(Rc::new(())).unwrap(); + let now = Instant::now(); + let checks = Rc::new(Cell::new(0)); + let error = Rc::new(Cell::new(None)); + for id in 0..66 { + state.register( + id, + CountingPolicy { + due: now, + checks: checks.clone(), + error: error.clone(), + }, + true, + true, + ); + } + assert_eq!(state.elect(0, &mut identity, 2), Ok(true)); + error.set(Some(7)); + let mut wakes = Vec::new(); + state.refresh(false, 9, 8, || now, std::convert::identity, &mut wakes); + assert_eq!(checks.get(), 65); + assert_eq!(state.deadlines.len(), 1); + assert_eq!(state.waiters[&0].error, Some(7)); + assert!(state.waiters[&65].error.is_none()); + assert!(matches!(state.phase, Phase::Draining(Outcome::Retry))); + assert_eq!(state.elect(65, &mut identity, 2), Ok(false)); + assert_eq!(identity.generation, 1); + + state.refresh(false, 9, 8, || now, std::convert::identity, &mut wakes); + assert_eq!(checks.get(), 66); + assert!(state.deadlines.is_empty()); + assert!(matches!(state.phase, Phase::Draining(Outcome::Retry))); + state.settle(true, 9, std::convert::identity, &mut wakes); + assert!(matches!(state.phase, Phase::Failed(9))); +} + +/// Fixed-deadline policy with observable checks and externally injected failures. +struct CountingPolicy { + due: Instant, + + checks: Rc>, + + error: Rc>>, +} + +impl WaiterPolicy for CountingPolicy { + /// Identify the failure injected into a waiter. + type Error = u8; + + /// Count policy evaluation and return the current injected failure. + fn check(&self) -> Option { + self.checks.set(self.checks.get() + 1); + self.error.get() + } + + /// Return the fixed deadline shared by this test's waiters. + fn deadline(&self) -> Instant { + self.due + } +} diff --git a/cmd/racer-dataplane/flow/tests/contracts.rs b/cmd/racer-dataplane/flow/tests/contracts.rs new file mode 100644 index 000000000..1c923e028 --- /dev/null +++ b/cmd/racer-dataplane/flow/tests/contracts.rs @@ -0,0 +1,671 @@ +//! Public ownership and capacity contracts exercised without private-state access. + +/// Reservation lifetime and target selection contracts. +mod handoff_tests { + use flow_control::{Error, Handoff, HandoffAdmission, Result}; + use std::{ + sync::{ + Arc, + atomic::{AtomicUsize, Ordering}, + }, + task::Waker, + }; + + /// One-slot admission used to expose early reservation release. + struct Quota(Arc); + + /// A counted reservation released only when its owner drops. + struct Held(Arc); + + impl Drop for Held { + /// Return the fixture's single slot. + fn drop(&mut self) { + self.0.fetch_sub(1, Ordering::SeqCst); + } + } + + impl HandoffAdmission for Quota { + /// Slot owner carried alongside every offered item. + type Reservation = Held; + + /// This synchronous fixture does not arrange wakeups. + fn register(&self, _: &Waker) {} + + /// Admit only when the fixture's slot is empty. + fn reserve(&self) -> Result { + self.0 + .compare_exchange(0, 1, Ordering::SeqCst, Ordering::SeqCst) + .map_err(|_| Error::Overloaded)?; + Ok(Held(self.0.clone())) + } + } + + /// Offers, envelopes, and popped items retain the selected target's slot. + #[test] + fn round_robin_reserves_before_delivery_and_releases_after_close() { + let handoff = Arc::new(Handoff::<_, _, u8>::new(&[1, 2])); + let counts = [Arc::new(AtomicUsize::new(0)), Arc::new(AtomicUsize::new(0))]; + let waker = Waker::noop(); + assert!(matches!(handoff.reserve(waker), Err(Error::Overloaded))); + for (key, count) in [1, 2].into_iter().zip(&counts) { + handoff.install(&key, Quota(count.clone())).unwrap(); + } + assert_eq!( + handoff.install(&1, Quota(counts[0].clone())), + Err(Error::InvalidInput) + ); + let first = handoff.reserve(waker).unwrap(); + let second = handoff.reserve(waker).unwrap(); + assert!(matches!(handoff.reserve(waker), Err(Error::Overloaded))); + first.deliver(|| 7).unwrap(); + assert!( + handoff + .pop_batch::<2>(&1, waker, 0) + .unwrap() + .iter() + .all(Option::is_none) + ); + let [item, empty] = handoff.pop_batch::<2>(&1, waker, 1).unwrap(); + let (payload, held) = item.unwrap().into_parts(); + assert_eq!(payload, 7); + assert!(empty.is_none()); + handoff.close(&1); + assert_eq!(counts[0].load(Ordering::SeqCst), 1); + drop(held); + assert_eq!(counts[0].load(Ordering::SeqCst), 0); + handoff.close(&2); + assert_eq!( + second.deliver(|| panic!("closed target built item")), + Err(Error::Unavailable) + ); + assert_eq!(counts[1].load(Ordering::SeqCst), 0); + assert!(matches!( + handoff.pop_batch::<1>(&3, waker, 1), + Err(Error::InvalidInput) + )); + assert!(matches!( + Arc::new(Handoff::::new(&[])).reserve(waker), + Err(Error::Overloaded) + )); + } + + /// Closing and abandoning work release exactly the retained reservations. + #[test] + fn close_drains_queued_reservations_and_abandoned_offer_releases() { + let handoff = Arc::new(Handoff::<_, _, ()>::new(&[1])); + let count = Arc::new(AtomicUsize::new(0)); + handoff.install(&1, Quota(count.clone())).unwrap(); + drop(handoff.reserve(Waker::noop()).unwrap()); + assert_eq!(count.load(Ordering::SeqCst), 0); + handoff + .reserve(Waker::noop()) + .unwrap() + .deliver(|| ()) + .unwrap(); + handoff.close(&1); + handoff.close(&1); + handoff.close(&2); + assert_eq!(count.load(Ordering::SeqCst), 0); + assert_eq!(handoff.install(&1, Quota(count)), Err(Error::InvalidInput)); + } + + /// Payload-only construction cannot discard admission before dequeue. + #[test] + fn envelope_retains_reservation_through_queue_pop_and_owner_transfer() { + let handoff = Arc::new(Handoff::<_, _, u8>::new(&[1])); + let count = Arc::new(AtomicUsize::new(0)); + handoff.install(&1, Quota(count.clone())).unwrap(); + handoff + .reserve(Waker::noop()) + .unwrap() + .deliver(|| 9) + .unwrap(); + assert!(matches!( + handoff.reserve(Waker::noop()), + Err(Error::Overloaded) + )); + let [item] = handoff.pop_batch::<1>(&1, Waker::noop(), 1).unwrap(); + assert!(matches!( + handoff.reserve(Waker::noop()), + Err(Error::Overloaded) + )); + let (payload, held) = item.unwrap().into_parts(); + assert_eq!(payload, 9); + assert_eq!(count.load(Ordering::SeqCst), 1); + drop(held); + assert!(handoff.reserve(Waker::noop()).is_ok()); + } +} + +/// Fixed backing and admission transfer contracts. +mod buffer_tests { + use flow_control::{ChargedBuffer, Error, Policy, Quotas, Rejection}; + + /// Single fixture class for fixed backing. + #[derive(Clone, Copy)] + struct Class; + + impl flow_control::Class for Class { + /// Number of fixed-backing resource classes. + const COUNT: usize = 1; + + /// Select the fixture's sole counter. + fn index(self) -> usize { + 0 + } + } + + /// Fixed four-MiB backing budget without wake or page coverage policy. + struct TestPolicy; + + impl Policy for TestPolicy { + /// Single resource budget for fixed backing. + type Class = Class; + + /// One possible keyed identity for this fixture. + type Key = (); + + /// Allow four MiB of fixed backing. + fn limit(&self, _: Class) -> usize { + 4 * 1024 * 1024 + } + + /// Permit one fixture key. + fn max_keys(&self) -> usize { + 1 + } + + /// Buffer releases do not wake this synchronous fixture. + fn wakes(_: Class) -> bool { + false + } + + /// This fixture does not admit page allocator backing. + fn covers(_: Class) -> bool { + false + } + + /// Rejection details are irrelevant to these backing assertions. + fn rejected(&self, _: Rejection) {} + } + + /// Moves, reuse, and final transfer preserve backing and its exact charge. + #[test] + fn fixed_backing_recycles_and_transfers_charge_without_early_release() { + let quotas = Quotas::new(TestPolicy); + let length = 1024 * 1024; + let mut buffer = + ChargedBuffer::new(quotas.reserve(None, Class, 2 * length).unwrap(), length).unwrap(); + assert_eq!(quotas.used(Class), length); + assert_eq!(buffer.charge().amount(), length); + let pointer = buffer.bytes().as_ptr(); + buffer.bytes_mut().fill(0xa7); + let moved = buffer; + assert_eq!(moved.bytes().as_ptr(), pointer); + drop(moved); + assert_eq!(quotas.used(Class), length); + let buffer = + ChargedBuffer::new(quotas.reserve(None, Class, length).unwrap(), length).unwrap(); + assert_eq!(buffer.bytes().as_ptr(), pointer); + assert!(buffer.bytes().iter().all(|byte| *byte == 0)); + let (bytes, charge) = buffer.into_parts(); + assert_eq!(bytes.as_ptr(), pointer); + assert_eq!(quotas.used(Class), length); + drop((bytes, charge)); + assert_eq!(quotas.used(Class), 0); + } + + /// Invalid backing geometry releases the consumed reservation. + #[test] + fn invalid_lengths_return_charge() { + let quotas = Quotas::new(TestPolicy); + for length in [0, 9] { + assert!(matches!( + ChargedBuffer::new(quotas.reserve(None, Class, 8).unwrap(), length), + Err(Error::InvalidInput) + )); + assert_eq!(quotas.used(Class), 0); + } + } +} + +/// Quota transfers, retained key lifetimes, and exact release notifications. +mod quota_tests { + use flow_control::{Charge, Error, Policy, Quotas, Rejection, SharedQuotas}; + use std::{ + sync::{ + Arc, Mutex, + atomic::{AtomicUsize, Ordering}, + }, + task::{Wake, Waker}, + }; + + /// Independent wake-enabled and silent resource classes. + #[derive(Clone, Copy, Debug, Eq, PartialEq)] + enum Resource { + Waking, + + Silent, + } + + impl flow_control::Class for Resource { + /// Number of independently accounted fixture classes. + const COUNT: usize = 2; + + /// Select this resource's independently padded counter. + fn index(self) -> usize { + self as usize + } + } + + /// Configurable limits with ordered rejection observations. + #[derive(Clone)] + struct TestPolicy { + limit: usize, + + max_keys: usize, + + rejected: Arc>>>, + } + + impl TestPolicy { + /// Start a fixture with identical class limits and an empty event log. + fn new(limit: usize, max_keys: usize) -> Self { + Self { + limit, + max_keys, + rejected: Arc::default(), + } + } + } + + impl Policy for TestPolicy { + /// Resource classes for wake and accounting assertions. + type Class = Resource; + + /// Small identities make key retirement observable through admission. + type Key = u8; + + /// Return the fixture's ceiling for either resource class. + fn limit(&self, _: Resource) -> usize { + self.limit + } + + /// Bound simultaneously retained key records. + fn max_keys(&self) -> usize { + self.max_keys + } + + /// Only the waking class notifies admission waiters on drop. + fn wakes(class: Resource) -> bool { + class == Resource::Waking + } + + /// Neither test class is used for page allocator backing. + fn covers(_: Resource) -> bool { + false + } + + /// Preserve every attempted rejection, including reclamation retries. + fn rejected(&self, rejection: Rejection) { + self.rejected.lock().unwrap().push(rejection); + } + } + + /// Count notifications independently of the worker-local quota authority. + #[derive(Default)] + struct WakeCount(AtomicUsize); + + impl Wake for WakeCount { + /// Count one consumed waiter registration. + fn wake(self: Arc) { + self.0.fetch_add(1, Ordering::SeqCst); + } + } + + /// Splitting is accounting-neutral; shrinking releases both budgets silently. + #[test] + fn keyed_split_shrink_and_cross_thread_drop_keep_exact_accounting() { + /// Assert that local authority does not prevent transferable charges. + fn transferable() {} + + transferable::>(); + transferable::>(); + let quotas = Quotas::new(TestPolicy::new(100, 1)); + let shared = quotas.shared(); + let wake = Arc::new(WakeCount::default()); + shared.register(&Waker::from(wake.clone())); + let mut charge = quotas.reserve(Some(&1), Resource::Waking, 100).unwrap(); + for amount in [0, 100, usize::MAX] { + assert!(matches!(charge.split(amount), Err(Error::InvalidInput))); + assert_eq!(charge.amount(), 100); + assert_eq!(shared.used(Resource::Waking), 100); + } + let split = charge.split(40).unwrap(); + assert_eq!(split.key(), Some(&1)); + assert_eq!(split.class(), Resource::Waking); + assert!(quotas.owns(&split)); + assert_eq!(shared.used(Resource::Waking), 100); + for amount in [0, 61, usize::MAX] { + assert_eq!(charge.shrink(amount), Err(Error::InvalidInput)); + assert_eq!(charge.amount(), 60); + } + charge.shrink(60).unwrap(); + charge.shrink(20).unwrap(); + assert_eq!(shared.used(Resource::Waking), 60); + assert_eq!(wake.0.load(Ordering::SeqCst), 0, "shrink must not wake"); + let refill = quotas.reserve(Some(&1), Resource::Waking, 40).unwrap(); + assert_eq!(shared.used(Resource::Waking), 100); + std::thread::spawn(move || drop(split)).join().unwrap(); + assert_eq!(shared.used(Resource::Waking), 60); + assert_eq!(wake.0.load(Ordering::SeqCst), 1); + drop((refill, charge)); + assert_eq!(shared.used(Resource::Waking), 0); + let next = quotas.reserve(Some(&2), Resource::Waking, 100).unwrap(); + assert_eq!(next.key(), Some(&2)); + assert!(quotas.policy().rejected.lock().unwrap().is_empty()); + } + + /// Retaining backing transfers surplus admission but not the donor's key owner. + #[test] + fn recycler_transfers_surplus_and_empty_donor_still_owns_key_until_drop() { + let size = 1 << 20; + let quotas = Quotas::new(TestPolicy::new(4 * size, 1)); + let shared = quotas.shared(); + let wake = Arc::new(WakeCount::default()); + shared.register(&Waker::from(wake.clone())); + let mut donor = quotas + .reserve(Some(&1), Resource::Waking, 2 * size) + .unwrap(); + let mut bytes = donor.buffer(size).unwrap(); + bytes.fill(0xa7); + bytes.truncate(1); + donor.recycle(bytes); + assert_eq!(donor.amount(), 0); + assert_eq!(donor.key(), Some(&1)); + assert_eq!(quotas.retained_buffer_bytes(), 2 * size); + assert_eq!(shared.used(Resource::Waking), 2 * size); + assert_eq!(wake.0.load(Ordering::SeqCst), 0); + assert_eq!(donor.shrink(1), Err(Error::InvalidInput)); + assert!(matches!(donor.split(1), Err(Error::InvalidInput))); + quotas.reclaim_buffers(); + assert_eq!(shared.used(Resource::Waking), 0); + assert_eq!(wake.0.load(Ordering::SeqCst), 1); + assert!(matches!( + quotas.reserve(Some(&2), Resource::Silent, 1), + Err(Error::Overloaded) + )); + assert!(matches!( + "as.policy().rejected.lock().unwrap()[..], + [ + Rejection::Keys { used: 1, limit: 1 }, + Rejection::Keys { used: 1, limit: 1 } + ] + )); + shared.register(&Waker::from(wake.clone())); + std::thread::spawn(move || drop(donor)).join().unwrap(); + assert_eq!(wake.0.load(Ordering::SeqCst), 2, "empty drops still wake"); + let next = quotas.reserve(Some(&2), Resource::Silent, 1).unwrap(); + assert_eq!(shared.used(Resource::Waking), 0); + assert_eq!(shared.used(Resource::Silent), 1); + drop(next); + assert_eq!(shared.used(Resource::Silent), 0); + } + + /// Failed retention leaves admission with its caller, including after stop. + #[test] + fn full_recycler_and_stopped_recycler_do_not_consume_charge() { + let size = 1 << 20; + let quotas = Quotas::new(TestPolicy::new(4 * size, 1)); + for _ in 0..2 { + let mut retained = quotas.reserve(None, Resource::Silent, size).unwrap(); + retained.recycle(vec![0xa7; size]); + assert_eq!(retained.amount(), 0); + } + let mut rejected = quotas.reserve(None, Resource::Silent, size).unwrap(); + rejected.recycle(vec![0xa7; size]); + assert_eq!(rejected.amount(), size); + assert_eq!(quotas.retained_buffer_bytes(), 2 * size); + assert_eq!(quotas.used(Resource::Silent), 3 * size); + quotas.stop(); + assert_eq!(quotas.retained_buffer_bytes(), 0); + assert_eq!(quotas.used(Resource::Silent), size); + rejected.recycle(vec![0xa7; size]); + assert_eq!(rejected.amount(), size); + assert_eq!(quotas.retained_buffer_bytes(), 0); + drop(rejected); + assert_eq!(quotas.used(Resource::Silent), 0); + } + + /// Checked admission rejects arithmetic overflow with unchanged reported facts. + #[test] + fn full_width_rejections_preserve_usage_and_attempt_counts() { + let quotas = Quotas::new(TestPolicy::new(usize::MAX, 1)); + let shared = quotas.shared(); + let mut charge = shared.reserve(Resource::Silent, usize::MAX).unwrap(); + assert!(matches!( + shared.reserve(Resource::Silent, 1), + Err(Error::Overloaded) + )); + assert!(matches!( + quotas.reserve(None, Resource::Silent, 1), + Err(Error::Overloaded) + )); + assert!(matches!( + quotas.reserve_completion(None, Resource::Silent, 1), + Err(Error::Overloaded) + )); + { + let rejected = quotas.policy().rejected.lock().unwrap(); + assert_eq!( + rejected.len(), + 5, + "shared attempts once; local retries once" + ); + for event in rejected.iter() { + assert!(matches!( + event, + Rejection::Resource { + class: Resource::Silent, + used: usize::MAX, + limit: usize::MAX, + requested: 1, + key_used: None, + key_limit: None, + } + )); + } + } + charge.shrink(1).unwrap(); + let split = shared.reserve(Resource::Silent, usize::MAX - 1).unwrap(); + assert_eq!(shared.used(Resource::Silent), usize::MAX); + drop(charge); + assert_eq!(shared.used(Resource::Silent), usize::MAX - 1); + drop(split); + assert_eq!(shared.used(Resource::Silent), 0); + } + + /// A panicking application key clone cannot strand transferred admission. + #[test] + fn key_clone_panic_keeps_split_and_recycle_admission_with_donor() { + use std::{ + panic::{AssertUnwindSafe, catch_unwind}, + sync::atomic::AtomicBool, + }; + + /// Enable clone failure only for this test's otherwise ordinary key. + static PANIC_ON_CLONE: AtomicBool = AtomicBool::new(false); + + /// One identity whose clone can fail after successful admission. + #[derive(Eq, Hash, PartialEq)] + struct Key; + + impl Clone for Key { + /// Panic on demand before a new charge can own admission. + fn clone(&self) -> Self { + assert!(!PANIC_ON_CLONE.load(Ordering::SeqCst), "key clone failed"); + Self + } + } + + /// Minimal policy whose sole key can panic while transferring ownership. + struct PanickingPolicy; + + impl Policy for PanickingPolicy { + /// Reuse the accounting fixture's resource classes. + type Class = Resource; + + /// Key with controllable clone failure. + type Key = Key; + + /// Allow one retained MiB and one MiB of surplus admission. + fn limit(&self, _: Resource) -> usize { + 2 << 20 + } + + /// Keep one key record for the donor and its possible split. + fn max_keys(&self) -> usize { + 1 + } + + /// This test observes accounting rather than wake delivery. + fn wakes(_: Resource) -> bool { + false + } + + /// No page allocator is involved in the transfer. + fn covers(_: Resource) -> bool { + false + } + + /// No admission rejection is expected in this panic path. + fn rejected(&self, _: Rejection) { + panic!("unexpected rejection"); + } + } + + let size = 1 << 20; + let quotas = Quotas::new(PanickingPolicy); + let mut donor = quotas + .reserve(Some(&Key), Resource::Silent, 2 * size) + .unwrap(); + PANIC_ON_CLONE.store(true, Ordering::SeqCst); + assert!(catch_unwind(AssertUnwindSafe(|| donor.split(size))).is_err()); + assert_eq!(donor.amount(), 2 * size); + assert_eq!(quotas.used(Resource::Silent), 2 * size); + assert!(catch_unwind(AssertUnwindSafe(|| donor.recycle(vec![0xa7; size]))).is_err()); + assert_eq!(donor.amount(), 2 * size); + assert_eq!(quotas.retained_buffer_bytes(), 0); + assert_eq!(quotas.used(Resource::Silent), 2 * size); + PANIC_ON_CLONE.store(false, Ordering::SeqCst); + let split = donor.split(size).unwrap(); + drop((donor, split)); + assert_eq!(quotas.used(Resource::Silent), 0); + let replacement = quotas + .reserve(Some(&Key), Resource::Silent, 2 * size) + .unwrap(); + drop(replacement); + assert_eq!(quotas.used(Resource::Silent), 0); + } +} + +/// Credit-window transition and full-width arithmetic contracts. +mod window_tests { + use flow_control::{Error, Window}; + + /// Invalid geometry never consumes credit. + #[test] + fn validates_configuration_and_item_lengths_without_consuming_credit() { + for (slots, bytes, max_item) in [(0, 1, 1), (1, 0, 1), (1, 1, 0)] { + assert!(matches!( + Window::::new(slots, bytes, max_item), + Err(Error::InvalidInput) + )); + } + let mut window = Window::new(2, 10, 6).unwrap(); + for length in [0, 7, u64::MAX] { + assert!(!window.can_reserve(length)); + assert_eq!(window.reserve("item", length), Err(Error::InvalidInput)); + assert!(window.is_empty()); + } + window.reserve("item", 6).unwrap(); + assert!(!window.is_empty()); + assert!(window.can_reserve(4)); + assert!(!window.can_reserve(5)); + } + + /// Only one exact acknowledgment of an issued item releases capacity. + #[test] + fn pending_and_issued_share_limits_and_release_exactly_once() { + let mut window = Window::new(2, 10, 10).unwrap(); + window.reserve(String::from("first"), 4).unwrap(); + assert_eq!(window.release("first".into(), 4), Err(Error::InvalidInput)); + window.issued("first".into()).unwrap(); + assert_eq!(window.issued("first".into()), Err(Error::InvalidInput)); + assert_eq!(window.reserve("first".into(), 1), Err(Error::Overloaded)); + window.reserve("second".into(), 6).unwrap(); + assert!(!window.can_reserve(1)); + assert_eq!(window.reserve("third".into(), 1), Err(Error::Overloaded)); + assert_eq!(window.release("first".into(), 3), Err(Error::InvalidInput)); + assert_eq!(window.release("first".into(), 5), Err(Error::InvalidInput)); + assert_eq!( + window.release("missing".into(), 4), + Err(Error::InvalidInput) + ); + assert_eq!(window.issued("missing".into()), Err(Error::InvalidInput)); + assert!(!window.can_reserve(1)); + window.release("first".into(), 4).unwrap(); + assert_eq!(window.release("first".into(), 4), Err(Error::InvalidInput)); + assert!(window.can_reserve(4)); + assert!(!window.can_reserve(5)); + window.issued("second".into()).unwrap(); + window.release("second".into(), 6).unwrap(); + assert!(window.is_empty()); + window.reserve("first".into(), 10).unwrap(); + window.issued("first".into()).unwrap(); + window.release("first".into(), 10).unwrap(); + assert!(window.is_empty()); + } + + /// Neither issuance nor oversized item ceilings weaken independent budgets. + #[test] + fn slot_and_byte_limits_are_independent() { + let mut slots = Window::new(1, 10, 10).unwrap(); + slots.reserve(1, 1).unwrap(); + assert!(!slots.can_reserve(1)); + assert_eq!(slots.reserve(2, 1), Err(Error::Overloaded)); + slots.issued(1).unwrap(); + assert!(!slots.can_reserve(1)); + slots.release(1, 1).unwrap(); + assert!(slots.can_reserve(10)); + let mut bytes = Window::new(100, 3, 10).unwrap(); + assert_eq!(bytes.reserve(1, 4), Err(Error::Overloaded)); + assert!(bytes.is_empty()); + bytes.reserve(1, 2).unwrap(); + assert!(bytes.can_reserve(1)); + assert_eq!(bytes.reserve(2, 2), Err(Error::Overloaded)); + bytes.reserve(2, 1).unwrap(); + assert!(!bytes.can_reserve(1)); + } + + /// Full-width arithmetic works without requiring cloneable keys. + #[test] + fn full_u64_budget_does_not_overflow_and_generic_keys_need_only_ord() { + /// Ordered fixture identity deliberately lacking Clone. + #[derive(Eq, PartialEq, Ord, PartialOrd)] + struct Key(u8); + let mut window = Window::new(usize::MAX, u64::MAX, u64::MAX).unwrap(); + window.reserve(Key(1), u64::MAX - 1).unwrap(); + window.reserve(Key(2), 1).unwrap(); + assert!(!window.can_reserve(1)); + assert_eq!(window.reserve(Key(3), 1), Err(Error::Overloaded)); + window.issued(Key(1)).unwrap(); + window.release(Key(1), u64::MAX - 1).unwrap(); + assert!(window.can_reserve(u64::MAX - 1)); + assert!(!window.can_reserve(u64::MAX)); + window.issued(Key(2)).unwrap(); + window.release(Key(2), 1).unwrap(); + assert!(window.is_empty()); + assert!(window.can_reserve(u64::MAX)); + } +} From 97122caa420de55e78069056705448541ced455d Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Mon, 5 Oct 2026 22:37:52 +0000 Subject: [PATCH 16/82] test(racer): gate isolated flow production and simulation --- .github/workflows/ci.yaml | 18 ++++++++++++++++++ cmd/racer-dataplane/README.md | 22 ++++++++++++++++++++++ 2 files changed, 40 insertions(+) diff --git a/.github/workflows/ci.yaml b/.github/workflows/ci.yaml index 01481b0a9..c227e1614 100644 --- a/.github/workflows/ci.yaml +++ b/.github/workflows/ci.yaml @@ -210,6 +210,24 @@ jobs: - name: Test allocator simulation run: timeout --signal=TERM --kill-after=10s 300s cargo +1.96.0 test --locked --manifest-path cmd/racer-dataplane/Cargo.toml -p page-alloc --no-default-features --features simulation + # Isolate flow features so workspace tests cannot substitute simulated I/O + # for the production pipe and socket contracts. Real-I/O failures never skip. + - name: Check flow production and simulation + run: | + timeout --signal=TERM --kill-after=10s 300s cargo +1.96.0 check --locked --manifest-path cmd/racer-dataplane/Cargo.toml -p flow-control --all-targets --no-default-features + timeout --signal=TERM --kill-after=10s 300s cargo +1.96.0 check --locked --manifest-path cmd/racer-dataplane/Cargo.toml -p flow-control --all-targets --no-default-features --features simulation + + - name: Strict flow Clippy + run: | + timeout --signal=TERM --kill-after=10s 300s cargo +1.96.0 clippy --locked --manifest-path cmd/racer-dataplane/Cargo.toml -p flow-control --all-targets --no-default-features -- -D warnings + timeout --signal=TERM --kill-after=10s 300s cargo +1.96.0 clippy --locked --manifest-path cmd/racer-dataplane/Cargo.toml -p flow-control --all-targets --no-default-features --features simulation -- -D warnings + + - name: Flow production real I/O (no simulation) + run: timeout --signal=TERM --kill-after=10s 300s cargo +1.96.0 test --locked --manifest-path cmd/racer-dataplane/Cargo.toml -p flow-control --no-default-features + + - name: Test flow simulation + run: timeout --signal=TERM --kill-after=10s 300s cargo +1.96.0 test --locked --manifest-path cmd/racer-dataplane/Cargo.toml -p flow-control --no-default-features --features simulation + runtime-miri: name: Runtime Miri (${{ matrix.group }}) runs-on: ubuntu-24.04 diff --git a/cmd/racer-dataplane/README.md b/cmd/racer-dataplane/README.md index 5c0c0f82b..cf122eb56 100644 --- a/cmd/racer-dataplane/README.md +++ b/cmd/racer-dataplane/README.md @@ -53,3 +53,25 @@ timeout --signal=TERM --kill-after=10s 300s cargo +1.96.0 test --locked --manife The production gate requires real io_uring and direct-I/O support: capability skips fail when `PAGE_ALLOC_REQUIRE_REAL_IO=1`. CI also checks all targets and runs strict Clippy for each allocator mode. + +The `flow-control` library provides policy-driven quotas, charged buffers, bounded +kernel pipes, adaptive admission, and worker-local coalescing. Application policy +and real operation completion remain the caller's responsibility. Its crate docs +describe ownership and admission contracts; generate them with: + +```sh +timeout --signal=TERM --kill-after=10s 300s cargo +1.96.0 doc --locked --manifest-path cmd/racer-dataplane/Cargo.toml -p flow-control --no-deps +``` + +Validate flow separately so workspace feature unification cannot enable simulation +in its production gate: + +```sh +timeout --signal=TERM --kill-after=10s 300s cargo +1.96.0 test --locked --manifest-path cmd/racer-dataplane/Cargo.toml -p flow-control --no-default-features -j 2 -- --test-threads=1 +timeout --signal=TERM --kill-after=10s 300s cargo +1.96.0 test --locked --manifest-path cmd/racer-dataplane/Cargo.toml -p flow-control --no-default-features --features simulation -j 2 -- --test-threads=1 +``` + +The flow production tests require Linux kernel pipes, Unix sockets, and splice. +They exercise real I/O and fail rather than skipping unsupported operations. +Simulation adds simulated and mixed-descriptor contracts without removing the +real pipe tests. CI checks all targets and runs strict Clippy in each mode. From 975176b682914fdcb4fc769aede6f0a0a3bd1869 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Mon, 5 Oct 2026 22:39:08 +0000 Subject: [PATCH 17/82] fix(racer): preserve flow quotas when admission key cloning panics --- cmd/racer-dataplane/flow/src/lib.rs | 5 ++++- cmd/racer-dataplane/flow/tests/contracts.rs | 23 ++++++++++++++++++++- 2 files changed, 26 insertions(+), 2 deletions(-) diff --git a/cmd/racer-dataplane/flow/src/lib.rs b/cmd/racer-dataplane/flow/src/lib.rs index 11fc326e8..a3aa1c494 100644 --- a/cmd/racer-dataplane/flow/src/lib.rs +++ b/cmd/racer-dataplane/flow/src/lib.rs @@ -616,6 +616,9 @@ impl Quotas

{ return Err(Error::Overloaded); } } + // Application cloning may panic. Finish it before committing admission + // so every counter increment is paired with a fully constructed owner. + let key = key.cloned(); self.totals .counter(class) .reserve(amount, limit) @@ -629,7 +632,7 @@ impl Quotas

{ Ok(Charge { class, amount, - key: key.cloned(), + key, totals: self.totals.clone(), local, buffers: Arc::downgrade(&self.buffers), diff --git a/cmd/racer-dataplane/flow/tests/contracts.rs b/cmd/racer-dataplane/flow/tests/contracts.rs index 1c923e028..d52f3d8a0 100644 --- a/cmd/racer-dataplane/flow/tests/contracts.rs +++ b/cmd/racer-dataplane/flow/tests/contracts.rs @@ -485,7 +485,7 @@ mod quota_tests { assert_eq!(shared.used(Resource::Silent), 0); } - /// A panicking application key clone cannot strand transferred admission. + /// A panicking key clone cannot strand fresh or transferred admission. #[test] fn key_clone_panic_keeps_split_and_recycle_admission_with_donor() { use std::{ @@ -566,6 +566,27 @@ mod quota_tests { .unwrap(); drop(replacement); assert_eq!(quotas.used(Resource::Silent), 0); + + // A live key record isolates the final charge-key clone from table setup. + // Both ordinary and drain admission must leave counters unchanged on panic. + for completion in [false, true] { + let donor = quotas.reserve(Some(&Key), Resource::Silent, size).unwrap(); + PANIC_ON_CLONE.store(true, Ordering::SeqCst); + let failed = catch_unwind(AssertUnwindSafe(|| { + if completion { + quotas.reserve_completion(Some(&Key), Resource::Silent, size) + } else { + quotas.reserve(Some(&Key), Resource::Silent, size) + } + })); + PANIC_ON_CLONE.store(false, Ordering::SeqCst); + assert!(failed.is_err()); + assert_eq!(quotas.used(Resource::Silent), size); + let refill = quotas.reserve(Some(&Key), Resource::Silent, size).unwrap(); + assert_eq!(quotas.used(Resource::Silent), 2 * size); + drop((donor, refill)); + assert_eq!(quotas.used(Resource::Silent), 0); + } } } From f29425368827d6fc58f2673f8b411b4573474e01 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Mon, 5 Oct 2026 22:45:24 +0000 Subject: [PATCH 18/82] fix(racer): retain flow ownership through panic and cancellation --- cmd/racer-dataplane/flow/src/admission.rs | 63 ++++++++++++++++++++--- cmd/racer-dataplane/flow/src/pipe.rs | 52 +++++++++++++++++++ 2 files changed, 108 insertions(+), 7 deletions(-) diff --git a/cmd/racer-dataplane/flow/src/admission.rs b/cmd/racer-dataplane/flow/src/admission.rs index 89145aa01..6b7befb43 100644 --- a/cmd/racer-dataplane/flow/src/admission.rs +++ b/cmd/racer-dataplane/flow/src/admission.rs @@ -420,16 +420,19 @@ mod circuit { return Err(Error::Unavailable); } let probe = self.states.borrow().contains_key(key); - if probe { + let key = if probe { if self.probes.borrow().len() >= self.capacity { return Err(Error::Overloaded); } - self.probes.borrow_mut().insert(key.clone()); - } - Ok(Probe { - health: self, - key: probe.then(|| key.clone()), - }) + // Complete application cloning before publishing exclusive ownership. + let indexed = key.clone(); + let owned = key.clone(); + self.probes.borrow_mut().insert(indexed); + Some(owned) + } else { + None + }; + Ok(Probe { health: self, key }) } /// Remove failure state without releasing any owned probe. @@ -506,6 +509,52 @@ mod circuit { mod tests { use super::*; + /// Failed key cloning must not leave an exclusive probe without an owner. + #[test] + fn review_regression_probe_clone_panic_allows_acquisition_after_timeout() { + use std::{ + panic::{AssertUnwindSafe, catch_unwind}, + sync::atomic::{AtomicUsize, Ordering}, + }; + + static CLONES: AtomicUsize = AtomicUsize::new(0); + static PANIC_AT: AtomicUsize = AtomicUsize::new(usize::MAX); + + /// A single endpoint with controllable clone failure. + #[derive(Eq, Ord, PartialEq, PartialOrd)] + struct Key; + + impl Clone for Key { + /// Fail at the selected clone during probe acquisition. + fn clone(&self) -> Self { + let clone = CLONES.fetch_add(1, Ordering::SeqCst) + 1; + assert_ne!(clone, PANIC_AT.load(Ordering::SeqCst), "key clone failed"); + Self + } + } + + for panic_at in [1, 2] { + let now = Instant::now(); + let timeout = Duration::from_secs(1); + let health = Circuits::new(1, timeout); + health.failure(&Key, now, |_, _| Duration::ZERO).unwrap(); + CLONES.store(0, Ordering::SeqCst); + PANIC_AT.store(panic_at, Ordering::SeqCst); + let result = catch_unwind(AssertUnwindSafe(|| health.acquire(&Key, now))); + PANIC_AT.store(usize::MAX, Ordering::SeqCst); + assert!(result.is_err()); + assert_eq!(CLONES.load(Ordering::SeqCst), panic_at); + assert!(!health.available(&Key, now)); + let retry = now + timeout; + let probe = health + .acquire(&Key, retry) + .expect("probe must not be stranded"); + assert!(!health.available(&Key, retry + timeout)); + drop(probe); + assert!(health.acquire(&Key, retry + timeout).is_ok()); + } + } + /// Keep an owned probe exclusive through success and retention changes. #[test] fn bounded_backoff_and_owned_probe_survive_retention_and_success() { diff --git a/cmd/racer-dataplane/flow/src/pipe.rs b/cmd/racer-dataplane/flow/src/pipe.rs index 5c02e4c6b..3623b6cec 100644 --- a/cmd/racer-dataplane/flow/src/pipe.rs +++ b/cmd/racer-dataplane/flow/src/pipe.rs @@ -78,6 +78,10 @@ impl Drop for Waiting

{ if queue.is_empty() { // Do not retain queue storage after its admission charges leave. *queue = VecDeque::new(); + } else if queue.capacity() > queue.len().saturating_mul(2) { + // Keep spare slots covered by live waiters' fixed allowances. + // Shrink geometrically rather than reallocating on every departure. + queue.shrink_to_fit(); } } wake_front(&self.queue); @@ -1043,6 +1047,54 @@ mod tests { assert!(pool.acquire().is_ok()); } + /// Head and tail cancellation keep allocated queue storage within live charges. + #[test] + fn review_regression_canceled_waiter_storage_remains_charged() { + const WAITERS: usize = 1024; + for keep_tail in [false, true] { + let quotas = admission(1); + let pool = PipePool::new( + quotas.clone(), + ResourceClass::Pipe, + ResourceClass::RequestContext, + WAITERS, + ); + let held = pool.acquire().unwrap(); + let mut cx = Context::from_waker(Waker::noop()); + let mut waiting: Vec<_> = (0..WAITERS).map(|_| acquire_wait(&pool)).collect(); + for wait in &mut waiting { + assert!(wait.as_mut().poll(&mut cx).is_pending()); + } + if keep_tail { + waiting.reverse(); + } + while waiting.len() > 1 { + drop(waiting.pop()); + let queue = pool.waiting.borrow(); + assert_eq!(queue.len(), waiting.len()); + let slots = queue.capacity() * std::mem::size_of::>>>(); + let owners = queue.len() + * (std::mem::size_of::>() + + std::mem::size_of::>>() + + 2 * std::mem::size_of::()); + assert!( + slots + owners <= quotas.used(ResourceClass::RequestContext), + "{} live waiters retain {} queue slots without admission", + queue.len(), + queue.capacity(), + ); + } + drop(held); + assert!(matches!( + waiting[0].as_mut().poll(&mut cx), + Poll::Ready(Ok(_)) + )); + drop(waiting); + assert_eq!(pool.waiting.borrow().capacity(), 0); + assert_eq!(quotas.used(ResourceClass::RequestContext), 0); + } + } + /// Scheduled acquisition is bounded and FIFO without polling itself awake. #[test] fn scheduled_acquisition_is_bounded_fifo_and_wakes_only_for_progress() { From 2f439b2a959bc123a3829101242612c540e3a6f4 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Mon, 5 Oct 2026 22:51:44 +0000 Subject: [PATCH 19/82] refactor(racer): consolidate flow integration workflows --- cmd/racer-dataplane/flow/tests/coalesce.rs | 414 ----------------- .../flow/tests/{contracts.rs => workflows.rs} | 418 +++++++++++++++++- 2 files changed, 417 insertions(+), 415 deletions(-) delete mode 100644 cmd/racer-dataplane/flow/tests/coalesce.rs rename cmd/racer-dataplane/flow/tests/{contracts.rs => workflows.rs} (62%) diff --git a/cmd/racer-dataplane/flow/tests/coalesce.rs b/cmd/racer-dataplane/flow/tests/coalesce.rs deleted file mode 100644 index d15e09919..000000000 --- a/cmd/racer-dataplane/flow/tests/coalesce.rs +++ /dev/null @@ -1,414 +0,0 @@ -//! Public ownership boundaries and synchronous reentrant notification contracts. - -use flow_control::coalesce::flight::state::{Outcome, Phase, Published, State, WaiterPolicy}; -use flow_control::coalesce::flight::{self, Entry, Operations, Stale}; -use flow_control::coalesce::{CapacityError, Event, Limits, Table, shared}; -use futures::executor::block_on; -use std::cell::{Cell, RefCell}; -use std::future::Future; -use std::rc::Rc; -use std::sync::Arc; -use std::task::{Context, Poll, Wake, Waker}; -use std::time::Instant; - -thread_local! { - /// Callback invoked only by synchronous wakes on the current test worker. - static ON_WAKE: RefCell>> = RefCell::new(None); -} - -/// Safe thread-local callback dispatch; no non-Send data enters the Waker itself. -struct Reenter; - -impl Wake for Reenter { - /// Invoke the current worker's callback after releasing its registry borrow. - fn wake(self: Arc) { - let callback = ON_WAKE.with(|slot| slot.borrow_mut().take()); - if let Some(callback) = callback { - callback(); - } - } -} - -/// Install a callback and return a wake target that dispatches it synchronously. -fn on_wake(callback: impl FnOnce() + 'static) -> Waker { - ON_WAKE.with(|slot| assert!(slot.borrow_mut().replace(Box::new(callback)).is_none())); - Waker::from(Arc::new(Reenter)) -} - -/// Build a table with enough attempts to test retry and final-owner-drop elections. -fn table(waiters: usize) -> Rc> { - Rc::new(Table::new( - Limits { - waiters_per_cohort: waiters, - attempts_per_cohort: 4, - }, - 99, - )) -} - -/// Joining and cloning handles do not require a cloneable result value. -#[test] -fn registration_clones_neither_keys_nor_results() { - /// Key whose clone records the allocation-time copy only. - #[derive(Eq, PartialEq)] - struct Key(Rc>); - - impl Clone for Key { - /// Count each actual key copy. - fn clone(&self) -> Self { - self.0.set(self.0.get() + 1); - Self(self.0.clone()) - } - } - - impl std::hash::Hash for Key { - /// Give the single logical test key a stable hash. - fn hash(&self, state: &mut H) { - // This test uses exactly one logical key, independent of the counter. - state.write_u8(0); - } - } - - /// Deliberately non-Clone result; only polling needs result cloning. - struct ResultValue; - - let table = Rc::new(Table::::new( - Limits { - waiters_per_cohort: 1, - attempts_per_cohort: 1, - }, - ResultValue, - )); - let copies = Rc::new(Cell::new(0)); - let first = table.join(Key(copies.clone()), 1).unwrap(); - assert_eq!(copies.get(), 1); - let second = first.clone(); - let third = second.clone(); - assert_eq!(copies.get(), 1); - first.finish(ResultValue); - drop((first, second)); - assert_eq!(table.registration_count(), 1); - drop(third); - assert_eq!(table.registration_count(), 0); -} - -/// Arbitrarily many cloned handles consume one charge until their final drop. -#[test] -fn many_clones_retain_completed_capacity_and_do_not_remove_replacements() { - let table = table(2); - let first = table.join(1, 1).unwrap(); - let mut clones = (0..32).map(|_| first.clone()).collect::>(); - first.finish(7); - let next = table.join(1, 1).unwrap(); - drop(first); - while clones.len() > 1 { - drop(clones.pop()); - assert_eq!(table.registration_count(), 2); - assert!(matches!(table.join(1, 1), Err(CapacityError))); - } - assert!(clones[0].is_only_handle()); - assert_eq!( - clones[0].event(Waker::noop()), - Poll::Ready(Event::Complete(7)) - ); - drop(clones); - assert_eq!(table.registration_count(), 1); - assert_eq!(table.active_count(), 1); - let follower = table.join(1, 1).unwrap(); - assert_eq!(next.event(Waker::noop()), Poll::Ready(Event::Lead)); - drop((next, follower)); - assert_eq!(table.registration_count(), 0); - assert_eq!(table.active_count(), 0); -} - -/// Retry and final-owner drop release all borrows before reentrant leader election. -#[test] -fn retry_and_final_drop_allow_reentrant_election() { - for drop_leader in [false, true] { - let table = table(2); - let leader = table.join(1, 1).unwrap(); - let follower = table.join(1, 1).unwrap(); - assert_eq!(leader.event(Waker::noop()), Poll::Ready(Event::Lead)); - let called = Rc::new(Cell::new(false)); - let waker = on_wake({ - let table = table.clone(); - let follower = follower.clone(); - let called = called.clone(); - move || { - assert_eq!(table.registration_count(), if drop_leader { 1 } else { 2 }); - assert_eq!(follower.event(Waker::noop()), Poll::Ready(Event::Lead)); - called.set(true); - } - }); - assert!(follower.event(&waker).is_pending()); - if drop_leader { - drop(leader); - } else { - leader.retry(); - } - assert!(called.get()); - } -} - -/// A follower may request notifications without revoking another waiter's leadership. -#[test] -fn follower_retry_notifies_without_taking_leadership() { - let table = table(2); - let leader = table.join(1, 1).unwrap(); - let follower = table.join(1, 1).unwrap(); - assert_eq!(leader.event(Waker::noop()), Poll::Ready(Event::Lead)); - let notified = Rc::new(Cell::new(false)); - let waker = on_wake({ - let leader = leader.clone(); - let follower = follower.clone(); - let notified = notified.clone(); - move || { - assert!(follower.event(Waker::noop()).is_pending()); - assert!(leader.event(Waker::noop()).is_pending()); - notified.set(true); - } - }); - assert!(follower.event(&waker).is_pending()); - follower.retry(); - assert!(notified.get()); - leader.retry(); - assert_eq!(follower.event(Waker::noop()), Poll::Ready(Event::Lead)); -} - -/// Completion removes old admission before any reader wake can admit new work. -#[test] -fn finish_allows_reentrant_admission_before_old_readers_detach() { - let table = table(3); - let leader = table.join(1, 1).unwrap(); - let follower = table.join(1, 1).unwrap(); - assert_eq!(leader.event(Waker::noop()), Poll::Ready(Event::Lead)); - let replacement = Rc::new(RefCell::new(None)); - let waker = on_wake({ - let table = table.clone(); - let follower = follower.clone(); - let replacement = replacement.clone(); - move || { - assert_eq!(table.active_count(), 0); - assert_eq!( - follower.event(Waker::noop()), - Poll::Ready(Event::Complete(7)) - ); - replacement.replace(Some(table.join(1, 1).unwrap())); - } - }); - assert!(follower.event(&waker).is_pending()); - leader.finish(7); - assert!(replacement.borrow().is_some()); - drop((leader, follower)); - assert_eq!(table.active_count(), 1); - assert_eq!(table.registration_count(), 1); -} - -/// Shared result notification can synchronously start replacement work. -#[test] -fn shared_completion_wakes_after_removal_and_keeps_new_owner() { - let table = Rc::new(shared::Table::default()); - let (mut receive, complete) = table.start(1, 99); - let replacement = Rc::new(RefCell::new(None)); - let waker = on_wake({ - let table = table.clone(); - let replacement = replacement.clone(); - move || { - assert!(table.is_empty()); - replacement.replace(Some(table.start(1, 99))); - } - }); - assert!( - std::pin::Pin::new(&mut receive) - .poll(&mut Context::from_waker(&waker)) - .is_pending() - ); - complete.finish(7); - assert_eq!(block_on(receive), 7); - let (receive, complete) = replacement.borrow_mut().take().unwrap(); - assert_eq!(table.len(), 1); - drop(complete); - assert_eq!(block_on(receive), 99); - assert_eq!( - table.len(), - 1, - "sender loss cannot pretend execution completed" - ); -} - -/// Losing the sender wakes parked readers without removing the indexed work. -#[test] -fn shared_sender_drop_wakes_with_entry_still_indexed() { - let table = Rc::new(shared::Table::default()); - let (mut receive, complete) = table.start(1, 99); - let notified = Rc::new(Cell::new(false)); - let waker = on_wake({ - let table = table.clone(); - let notified = notified.clone(); - move || { - assert_eq!(table.len(), 1); - assert!(table.get(&1).is_some()); - notified.set(true); - } - }); - assert!( - std::pin::Pin::new(&mut receive) - .poll(&mut Context::from_waker(&waker)) - .is_pending() - ); - drop(complete); - assert!(notified.get()); - assert_eq!(block_on(receive), 99); - assert_eq!(block_on(table.get(&1).unwrap()), 99); - assert_eq!(table.len(), 1); -} - -/// Resource whose destructor can synchronously inspect its owning table. -struct ReentrantResource(Option>); - -impl Drop for ReentrantResource { - /// Reenter the owner while its operation tombstone must still be present. - fn drop(&mut self) { - self.0.take().unwrap()(); - } -} - -/// Only operation completion makes this deliberately waiter-free entry removable. -#[derive(Default)] -struct OwnedEntry(Operations); - -impl Entry for OwnedEntry { - /// This fixture has no caller policy to refresh. - fn refresh(&mut self, _: &mut Vec) {} - - /// Require explicit completion of every retained operation. - fn quiescent(&self) -> bool { - self.0.is_empty() - } -} - -/// Resource drop reentrancy cannot erase the tombstone, and drain wakes run unlocked. -#[test] -fn completion_tombstone_survives_reentrant_destructor_and_shutdown() { - let table = Rc::new(RefCell::new(flight::Table::::default())); - let dropped = Rc::new(Cell::new(false)); - let resource = ReentrantResource(Some(Box::new({ - let table = Rc::downgrade(&table); - let dropped = dropped.clone(); - move || { - let table = table.upgrade().unwrap(); - flight::update(&table, |table, wakes| { - table.sweep(1, wakes); - assert_eq!(table.len(), 1); - assert!(!table.remove_quiescent(&1)); - }); - dropped.set(true); - } - }))); - let id = flight::update(&table, |table, _| { - assert_eq!(table.next_waiter_id(), Ok(1)); - assert_eq!(table.next_waiter_id(), Ok(2)); - let id = table.next_operation_id().unwrap(); - table.insert(1, OwnedEntry::default()); - table.get_mut(&1).unwrap().0.insert(id, resource); - assert_eq!(table.get_mut(&1).unwrap().0.complete(id), Err(Stale)); - id - }); - flight::update(&table, |table, wakes| table.stop(wakes, |_, _| {})); - assert!(table.borrow().is_stopping()); - let resource = flight::update(&table, |table, _| { - let operations = &mut table.get_mut(&1).unwrap().0; - let resource = operations.take(id).unwrap(); - assert!(matches!(operations.take(id), Err(Stale))); - assert!(matches!(operations.take(id + 1), Err(Stale))); - assert_eq!(operations.complete(id + 1), Err(Stale)); - assert_eq!(operations.len(), 1); - resource - }); - drop(resource); - assert!(dropped.get()); - assert_eq!(table.borrow().len(), 1); - let notified = Rc::new(Cell::new(false)); - let waker = on_wake({ - let table = table.clone(); - let notified = notified.clone(); - move || { - assert!(table.borrow().is_empty()); - notified.set(true); - } - }); - flight::update(&table, |table, wakes| { - table.register_drain(&waker); - let operations = &mut table.get_mut(&1).unwrap().0; - operations.complete(id).unwrap(); - assert_eq!(operations.complete(id), Err(Stale)); - table.sweep(1, wakes); - }); - assert!(notified.get()); -} - -/// An expired leader is checked separately from the 64-entry deadline quantum. -#[test] -fn expired_leader_does_not_consume_deadline_quantum_or_clear_drain_fence() { - let mut state = State::, CountingPolicy>::default(); - let mut table = flight::Table::::default(); - let mut identity = table.identity(Rc::new(())).unwrap(); - let now = Instant::now(); - let checks = Rc::new(Cell::new(0)); - let error = Rc::new(Cell::new(None)); - for id in 0..66 { - state.register( - id, - CountingPolicy { - due: now, - checks: checks.clone(), - error: error.clone(), - }, - true, - true, - ); - } - assert_eq!(state.elect(0, &mut identity, 2), Ok(true)); - error.set(Some(7)); - let mut wakes = Vec::new(); - state.refresh(false, 9, 8, || now, std::convert::identity, &mut wakes); - assert_eq!(checks.get(), 65); - assert_eq!(state.deadlines.len(), 1); - assert_eq!(state.waiters[&0].error, Some(7)); - assert!(state.waiters[&65].error.is_none()); - assert!(matches!(state.phase, Phase::Draining(Outcome::Retry))); - assert_eq!(state.elect(65, &mut identity, 2), Ok(false)); - assert_eq!(identity.generation, 1); - - state.refresh(false, 9, 8, || now, std::convert::identity, &mut wakes); - assert_eq!(checks.get(), 66); - assert!(state.deadlines.is_empty()); - assert!(matches!(state.phase, Phase::Draining(Outcome::Retry))); - state.settle(true, 9, std::convert::identity, &mut wakes); - assert!(matches!(state.phase, Phase::Failed(9))); -} - -/// Fixed-deadline policy with observable checks and externally injected failures. -struct CountingPolicy { - due: Instant, - - checks: Rc>, - - error: Rc>>, -} - -impl WaiterPolicy for CountingPolicy { - /// Identify the failure injected into a waiter. - type Error = u8; - - /// Count policy evaluation and return the current injected failure. - fn check(&self) -> Option { - self.checks.set(self.checks.get() + 1); - self.error.get() - } - - /// Return the fixed deadline shared by this test's waiters. - fn deadline(&self) -> Instant { - self.due - } -} diff --git a/cmd/racer-dataplane/flow/tests/contracts.rs b/cmd/racer-dataplane/flow/tests/workflows.rs similarity index 62% rename from cmd/racer-dataplane/flow/tests/contracts.rs rename to cmd/racer-dataplane/flow/tests/workflows.rs index d52f3d8a0..8634ee5ef 100644 --- a/cmd/racer-dataplane/flow/tests/contracts.rs +++ b/cmd/racer-dataplane/flow/tests/workflows.rs @@ -1,4 +1,4 @@ -//! Public ownership and capacity contracts exercised without private-state access. +//! Public ownership, capacity, and coalescing workflows without private-state access. /// Reservation lifetime and target selection contracts. mod handoff_tests { @@ -690,3 +690,419 @@ mod window_tests { assert!(window.can_reserve(u64::MAX)); } } + +/// Public ownership boundaries and synchronous reentrant notification contracts. +mod coalesce_tests { + use flow_control::coalesce::flight::state::{Outcome, Phase, Published, State, WaiterPolicy}; + use flow_control::coalesce::flight::{self, Entry, Operations, Stale}; + use flow_control::coalesce::{CapacityError, Event, Limits, Table, shared}; + use futures::executor::block_on; + use std::cell::{Cell, RefCell}; + use std::future::Future; + use std::rc::Rc; + use std::sync::Arc; + use std::task::{Context, Poll, Wake, Waker}; + use std::time::Instant; + + thread_local! { + /// Callback invoked only by synchronous wakes on the current test worker. + static ON_WAKE: RefCell>> = RefCell::new(None); + } + + /// Safe thread-local callback dispatch; no non-Send data enters the Waker itself. + struct Reenter; + + impl Wake for Reenter { + /// Invoke the current worker's callback after releasing its registry borrow. + fn wake(self: Arc) { + let callback = ON_WAKE.with(|slot| slot.borrow_mut().take()); + if let Some(callback) = callback { + callback(); + } + } + } + + /// Install a callback and return a wake target that dispatches it synchronously. + fn on_wake(callback: impl FnOnce() + 'static) -> Waker { + ON_WAKE.with(|slot| assert!(slot.borrow_mut().replace(Box::new(callback)).is_none())); + Waker::from(Arc::new(Reenter)) + } + + /// Build a table with enough attempts to test retry and final-owner-drop elections. + fn table(waiters: usize) -> Rc> { + Rc::new(Table::new( + Limits { + waiters_per_cohort: waiters, + attempts_per_cohort: 4, + }, + 99, + )) + } + + /// Joining and cloning handles do not require a cloneable result value. + #[test] + fn registration_clones_neither_keys_nor_results() { + /// Key whose clone records the allocation-time copy only. + #[derive(Eq, PartialEq)] + struct Key(Rc>); + + impl Clone for Key { + /// Count each actual key copy. + fn clone(&self) -> Self { + self.0.set(self.0.get() + 1); + Self(self.0.clone()) + } + } + + impl std::hash::Hash for Key { + /// Give the single logical test key a stable hash. + fn hash(&self, state: &mut H) { + // This test uses exactly one logical key, independent of the counter. + state.write_u8(0); + } + } + + /// Deliberately non-Clone result; only polling needs result cloning. + struct ResultValue; + + let table = Rc::new(Table::::new( + Limits { + waiters_per_cohort: 1, + attempts_per_cohort: 1, + }, + ResultValue, + )); + let copies = Rc::new(Cell::new(0)); + let first = table.join(Key(copies.clone()), 1).unwrap(); + assert_eq!(copies.get(), 1); + let second = first.clone(); + let third = second.clone(); + assert_eq!(copies.get(), 1); + first.finish(ResultValue); + drop((first, second)); + assert_eq!(table.registration_count(), 1); + drop(third); + assert_eq!(table.registration_count(), 0); + } + + /// Arbitrarily many cloned handles consume one charge until their final drop. + #[test] + fn many_clones_retain_completed_capacity_and_do_not_remove_replacements() { + let table = table(2); + let first = table.join(1, 1).unwrap(); + let mut clones = (0..32).map(|_| first.clone()).collect::>(); + first.finish(7); + let next = table.join(1, 1).unwrap(); + drop(first); + while clones.len() > 1 { + drop(clones.pop()); + assert_eq!(table.registration_count(), 2); + assert!(matches!(table.join(1, 1), Err(CapacityError))); + } + assert!(clones[0].is_only_handle()); + assert_eq!( + clones[0].event(Waker::noop()), + Poll::Ready(Event::Complete(7)) + ); + drop(clones); + assert_eq!(table.registration_count(), 1); + assert_eq!(table.active_count(), 1); + let follower = table.join(1, 1).unwrap(); + assert_eq!(next.event(Waker::noop()), Poll::Ready(Event::Lead)); + drop((next, follower)); + assert_eq!(table.registration_count(), 0); + assert_eq!(table.active_count(), 0); + } + + /// Retry and final-owner drop release all borrows before reentrant leader election. + #[test] + fn retry_and_final_drop_allow_reentrant_election() { + for drop_leader in [false, true] { + let table = table(2); + let leader = table.join(1, 1).unwrap(); + let follower = table.join(1, 1).unwrap(); + assert_eq!(leader.event(Waker::noop()), Poll::Ready(Event::Lead)); + let called = Rc::new(Cell::new(false)); + let waker = on_wake({ + let table = table.clone(); + let follower = follower.clone(); + let called = called.clone(); + move || { + assert_eq!(table.registration_count(), if drop_leader { 1 } else { 2 }); + assert_eq!(follower.event(Waker::noop()), Poll::Ready(Event::Lead)); + called.set(true); + } + }); + assert!(follower.event(&waker).is_pending()); + if drop_leader { + drop(leader); + } else { + leader.retry(); + } + assert!(called.get()); + } + } + + /// A follower may request notifications without revoking another waiter's leadership. + #[test] + fn follower_retry_notifies_without_taking_leadership() { + let table = table(2); + let leader = table.join(1, 1).unwrap(); + let follower = table.join(1, 1).unwrap(); + assert_eq!(leader.event(Waker::noop()), Poll::Ready(Event::Lead)); + let notified = Rc::new(Cell::new(false)); + let waker = on_wake({ + let leader = leader.clone(); + let follower = follower.clone(); + let notified = notified.clone(); + move || { + assert!(follower.event(Waker::noop()).is_pending()); + assert!(leader.event(Waker::noop()).is_pending()); + notified.set(true); + } + }); + assert!(follower.event(&waker).is_pending()); + follower.retry(); + assert!(notified.get()); + leader.retry(); + assert_eq!(follower.event(Waker::noop()), Poll::Ready(Event::Lead)); + } + + /// Completion removes old admission before any reader wake can admit new work. + #[test] + fn finish_allows_reentrant_admission_before_old_readers_detach() { + let table = table(3); + let leader = table.join(1, 1).unwrap(); + let follower = table.join(1, 1).unwrap(); + assert_eq!(leader.event(Waker::noop()), Poll::Ready(Event::Lead)); + let replacement = Rc::new(RefCell::new(None)); + let waker = on_wake({ + let table = table.clone(); + let follower = follower.clone(); + let replacement = replacement.clone(); + move || { + assert_eq!(table.active_count(), 0); + assert_eq!( + follower.event(Waker::noop()), + Poll::Ready(Event::Complete(7)) + ); + replacement.replace(Some(table.join(1, 1).unwrap())); + } + }); + assert!(follower.event(&waker).is_pending()); + leader.finish(7); + assert!(replacement.borrow().is_some()); + drop((leader, follower)); + assert_eq!(table.active_count(), 1); + assert_eq!(table.registration_count(), 1); + } + + /// Shared result notification can synchronously start replacement work. + #[test] + fn shared_completion_wakes_after_removal_and_keeps_new_owner() { + let table = Rc::new(shared::Table::default()); + let (mut receive, complete) = table.start(1, 99); + let replacement = Rc::new(RefCell::new(None)); + let waker = on_wake({ + let table = table.clone(); + let replacement = replacement.clone(); + move || { + assert!(table.is_empty()); + replacement.replace(Some(table.start(1, 99))); + } + }); + assert!( + std::pin::Pin::new(&mut receive) + .poll(&mut Context::from_waker(&waker)) + .is_pending() + ); + complete.finish(7); + assert_eq!(block_on(receive), 7); + let (receive, complete) = replacement.borrow_mut().take().unwrap(); + assert_eq!(table.len(), 1); + drop(complete); + assert_eq!(block_on(receive), 99); + assert_eq!( + table.len(), + 1, + "sender loss cannot pretend execution completed" + ); + } + + /// Losing the sender wakes parked readers without removing the indexed work. + #[test] + fn shared_sender_drop_wakes_with_entry_still_indexed() { + let table = Rc::new(shared::Table::default()); + let (mut receive, complete) = table.start(1, 99); + let notified = Rc::new(Cell::new(false)); + let waker = on_wake({ + let table = table.clone(); + let notified = notified.clone(); + move || { + assert_eq!(table.len(), 1); + assert!(table.get(&1).is_some()); + notified.set(true); + } + }); + assert!( + std::pin::Pin::new(&mut receive) + .poll(&mut Context::from_waker(&waker)) + .is_pending() + ); + drop(complete); + assert!(notified.get()); + assert_eq!(block_on(receive), 99); + assert_eq!(block_on(table.get(&1).unwrap()), 99); + assert_eq!(table.len(), 1); + } + + /// Resource whose destructor can synchronously inspect its owning table. + struct ReentrantResource(Option>); + + impl Drop for ReentrantResource { + /// Reenter the owner while its operation tombstone must still be present. + fn drop(&mut self) { + self.0.take().unwrap()(); + } + } + + /// Only operation completion makes this deliberately waiter-free entry removable. + #[derive(Default)] + struct OwnedEntry(Operations); + + impl Entry for OwnedEntry { + /// This fixture has no caller policy to refresh. + fn refresh(&mut self, _: &mut Vec) {} + + /// Require explicit completion of every retained operation. + fn quiescent(&self) -> bool { + self.0.is_empty() + } + } + + /// Resource drop reentrancy cannot erase the tombstone, and drain wakes run unlocked. + #[test] + fn completion_tombstone_survives_reentrant_destructor_and_shutdown() { + let table = Rc::new(RefCell::new(flight::Table::::default())); + let dropped = Rc::new(Cell::new(false)); + let resource = ReentrantResource(Some(Box::new({ + let table = Rc::downgrade(&table); + let dropped = dropped.clone(); + move || { + let table = table.upgrade().unwrap(); + flight::update(&table, |table, wakes| { + table.sweep(1, wakes); + assert_eq!(table.len(), 1); + assert!(!table.remove_quiescent(&1)); + }); + dropped.set(true); + } + }))); + let id = flight::update(&table, |table, _| { + assert_eq!(table.next_waiter_id(), Ok(1)); + assert_eq!(table.next_waiter_id(), Ok(2)); + let id = table.next_operation_id().unwrap(); + table.insert(1, OwnedEntry::default()); + table.get_mut(&1).unwrap().0.insert(id, resource); + assert_eq!(table.get_mut(&1).unwrap().0.complete(id), Err(Stale)); + id + }); + flight::update(&table, |table, wakes| table.stop(wakes, |_, _| {})); + assert!(table.borrow().is_stopping()); + let resource = flight::update(&table, |table, _| { + let operations = &mut table.get_mut(&1).unwrap().0; + let resource = operations.take(id).unwrap(); + assert!(matches!(operations.take(id), Err(Stale))); + assert!(matches!(operations.take(id + 1), Err(Stale))); + assert_eq!(operations.complete(id + 1), Err(Stale)); + assert_eq!(operations.len(), 1); + resource + }); + drop(resource); + assert!(dropped.get()); + assert_eq!(table.borrow().len(), 1); + let notified = Rc::new(Cell::new(false)); + let waker = on_wake({ + let table = table.clone(); + let notified = notified.clone(); + move || { + assert!(table.borrow().is_empty()); + notified.set(true); + } + }); + flight::update(&table, |table, wakes| { + table.register_drain(&waker); + let operations = &mut table.get_mut(&1).unwrap().0; + operations.complete(id).unwrap(); + assert_eq!(operations.complete(id), Err(Stale)); + table.sweep(1, wakes); + }); + assert!(notified.get()); + } + + /// An expired leader is checked separately from the 64-entry deadline quantum. + #[test] + fn expired_leader_does_not_consume_deadline_quantum_or_clear_drain_fence() { + let mut state = State::, CountingPolicy>::default(); + let mut table = flight::Table::::default(); + let mut identity = table.identity(Rc::new(())).unwrap(); + let now = Instant::now(); + let checks = Rc::new(Cell::new(0)); + let error = Rc::new(Cell::new(None)); + for id in 0..66 { + state.register( + id, + CountingPolicy { + due: now, + checks: checks.clone(), + error: error.clone(), + }, + true, + true, + ); + } + assert_eq!(state.elect(0, &mut identity, 2), Ok(true)); + error.set(Some(7)); + let mut wakes = Vec::new(); + state.refresh(false, 9, 8, || now, std::convert::identity, &mut wakes); + assert_eq!(checks.get(), 65); + assert_eq!(state.deadlines.len(), 1); + assert_eq!(state.waiters[&0].error, Some(7)); + assert!(state.waiters[&65].error.is_none()); + assert!(matches!(state.phase, Phase::Draining(Outcome::Retry))); + assert_eq!(state.elect(65, &mut identity, 2), Ok(false)); + assert_eq!(identity.generation, 1); + + state.refresh(false, 9, 8, || now, std::convert::identity, &mut wakes); + assert_eq!(checks.get(), 66); + assert!(state.deadlines.is_empty()); + assert!(matches!(state.phase, Phase::Draining(Outcome::Retry))); + state.settle(true, 9, std::convert::identity, &mut wakes); + assert!(matches!(state.phase, Phase::Failed(9))); + } + + /// Fixed-deadline policy with observable checks and externally injected failures. + struct CountingPolicy { + due: Instant, + + checks: Rc>, + + error: Rc>>, + } + + impl WaiterPolicy for CountingPolicy { + /// Identify the failure injected into a waiter. + type Error = u8; + + /// Count policy evaluation and return the current injected failure. + fn check(&self) -> Option { + self.checks.set(self.checks.get() + 1); + self.error.get() + } + + /// Return the fixed deadline shared by this test's waiters. + fn deadline(&self) -> Instant { + self.due + } + } +} From 7a5fbf7f82455c50c6cc7d35a7f869f104644211 Mon Sep 17 00:00:00 2001 From: "copilot-swe-agent[bot]" <198982749+Copilot@users.noreply.github.com> Date: Tue, 6 Oct 2026 16:48:44 +0000 Subject: [PATCH 20/82] Fix flow control review findings Co-authored-by: jveski <7576912+jveski@users.noreply.github.com> --- cmd/racer-dataplane/flow/src/admission.rs | 30 ++++- cmd/racer-dataplane/flow/src/coalesce.rs | 7 +- cmd/racer-dataplane/flow/src/lib.rs | 135 ++++++++++++++++++---- 3 files changed, 148 insertions(+), 24 deletions(-) diff --git a/cmd/racer-dataplane/flow/src/admission.rs b/cmd/racer-dataplane/flow/src/admission.rs index 6b7befb43..5f7059ab4 100644 --- a/cmd/racer-dataplane/flow/src/admission.rs +++ b/cmd/racer-dataplane/flow/src/admission.rs @@ -311,6 +311,7 @@ impl Permit { impl Drop for Permit { /// Release final ownership and renew backoff for an unsuccessful probe. fn drop(&mut self) { + let now = self.probe.then(|| (self.owner.now)()); let Ok(mut state) = self.owner.state.lock() else { return; }; @@ -322,7 +323,7 @@ impl Drop for Permit { if self.probe { peer.probe = false; if peer.retry.is_some() { - let now = (self.owner.now)(); + let now = now.expect("probe clock sampled before locking"); peer.retry = Some(now + self.owner.config.backoff); peer.updated = now; } @@ -1052,6 +1053,7 @@ mod hedge { #[cfg(test)] mod tests { use super::*; + use std::sync::OnceLock; /// Synchronous observer recording emitted events and gauges. #[derive(Default)] @@ -1063,6 +1065,19 @@ mod tests { limit: Mutex, } + static CLOCK_OWNER: OnceLock>> = OnceLock::new(); + + /// Verify clock callbacks run without the adaptive-state mutex held. + fn reentrant_clock() -> Instant { + if let Some(owner) = CLOCK_OWNER.get().and_then(std::sync::Weak::upgrade) { + let _state = owner + .state + .try_lock() + .expect("clock called while adaptive state is locked"); + } + Instant::now() + } + impl Observer for Counts { /// Append an event in emission order. fn event(&self, event: Event) { @@ -1217,4 +1232,17 @@ mod tests { assert_eq!(*owner.observer.limit.lock().unwrap(), usize::MAX); assert_eq!(owner.state.lock().unwrap().peers[&()].limit, usize::MAX); } + + /// Probe release invokes the injected clock only after releasing adaptive state. + #[test] + fn probe_drop_calls_clock_outside_state_lock() { + let owner = Adaptive::new(config(), Counts::default(), reentrant_clock).unwrap(); + CLOCK_OWNER.set(Arc::downgrade(&owner)).unwrap(); + let failed = owner.acquire(&1).unwrap(); + failed.observe(Outcome::PeerFailure); + drop(failed); + owner.state.lock().unwrap().peers.get_mut(&1).unwrap().retry = Some(Instant::now()); + let probe = owner.acquire(&1).unwrap(); + drop(probe); + } } diff --git a/cmd/racer-dataplane/flow/src/coalesce.rs b/cmd/racer-dataplane/flow/src/coalesce.rs index 8aeaad5d1..1b54c8b5c 100644 --- a/cmd/racer-dataplane/flow/src/coalesce.rs +++ b/cmd/racer-dataplane/flow/src/coalesce.rs @@ -82,6 +82,9 @@ impl Table { key: K, capacity: usize, ) -> Result, CapacityError> { + if self.limits.waiters_per_cohort == 0 { + return Err(CapacityError); + } if self.registrations.get() >= capacity.saturating_mul(self.limits.waiters_per_cohort) { return Err(CapacityError); } @@ -1935,7 +1938,9 @@ mod tests { /// Zero bounds and identifier exhaustion reject work without leaking charges. #[test] fn zero_limits_and_id_exhaustion_never_wrap_or_leak_registrations() { - assert!(matches!(table(0, 1).join("key", 1), Err(CapacityError))); + let no_waiters = table(0, 1); + assert!(matches!(no_waiters.join("key", 1), Err(CapacityError))); + assert_eq!(no_waiters.active_count(), 0); let zero = table(1, 0); let waiter = zero.join("key", 1).unwrap(); assert_eq!( diff --git a/cmd/racer-dataplane/flow/src/lib.rs b/cmd/racer-dataplane/flow/src/lib.rs index a3aa1c494..e28bb277f 100644 --- a/cmd/racer-dataplane/flow/src/lib.rs +++ b/cmd/racer-dataplane/flow/src/lib.rs @@ -348,12 +348,13 @@ impl Charge

{ return Err(Error::InvalidInput); } if let Some(pool) = self.buffers.upgrade() { - let mut pool = pool.lock().unwrap_or_else(|e| e.into_inner()); - if let Some(index) = pool - .iter() - .position(|(bytes, _)| bytes.capacity() == length) - { - let (bytes, old) = pool.swap_remove(index); + let recycled = { + let mut pool = pool.lock().unwrap_or_else(|e| e.into_inner()); + pool.iter() + .position(|(bytes, _)| bytes.capacity() == length) + .map(|index| pool.swap_remove(index)) + }; + if let Some((bytes, old)) = recycled { drop(old); debug_assert_eq!(bytes.len(), length); return Ok(bytes); @@ -380,8 +381,10 @@ impl Charge

{ let Some(pool) = self.buffers.upgrade() else { return; }; - let Ok(mut pool) = pool.try_lock() else { - return; + let mut pool = match pool.try_lock() { + Ok(pool) => pool, + Err(std::sync::TryLockError::WouldBlock) => return, + Err(std::sync::TryLockError::Poisoned(error)) => error.into_inner(), }; if self.stopped.load(Ordering::Acquire) || pool.len() >= 2 { return; @@ -548,10 +551,11 @@ impl Quotas

{ /// Release every idle recycler allocation and its charge. pub fn reclaim_buffers(&self) { - self.buffers - .lock() - .unwrap_or_else(|e| e.into_inner()) - .clear(); + let reclaimed = { + let mut buffers = self.buffers.lock().unwrap_or_else(|e| e.into_inner()); + std::mem::take(&mut *buffers) + }; + drop(reclaimed); } /// Divide the aggregate ceiling, honoring the clipped per-key floor. @@ -580,16 +584,21 @@ impl Quotas

{ /// Reclaim only when key records, fairness, or this class can benefit. fn reclaim_buffers_for(&self, key: Option<&P::Key>, class: P::Class) { - let mut buffers = self.buffers.lock().unwrap_or_else(|e| e.into_inner()); - // Keyed admission can exhaust records or fair shares too. Unkeyed - // admission only reclaims when this class has an idle charge. - if key.is_some() - || buffers - .iter() - .any(|(_, charge)| charge.class.index() == class.index()) - { - buffers.clear(); - } + let reclaimed = { + let mut buffers = self.buffers.lock().unwrap_or_else(|e| e.into_inner()); + // Keyed admission can exhaust records or fair shares too. Unkeyed + // admission only reclaims when this class has an idle charge. + if key.is_some() + || buffers + .iter() + .any(|(_, charge)| charge.class.index() == class.index()) + { + std::mem::take(&mut *buffers) + } else { + Vec::new() + } + }; + drop(reclaimed); } /// Perform one admission attempt and report its rejection before retrying. @@ -1071,6 +1080,32 @@ mod quota_tests { } } + /// Checks recycler lock availability during a synchronous charge wake. + struct PoolWake { + buffers: Arc, Charge)>>>, + + unlocked: Arc, + } + + impl std::task::Wake for PoolWake { + /// Record whether callback reentry can acquire the recycler mutex. + fn wake(self: Arc) { + self.unlocked + .store(self.buffers.try_lock().is_ok(), Ordering::SeqCst); + } + } + + /// Register a callback that probes the recycler mutex when the next charge drops. + fn watch_buffer_unlock(quotas: &Quotas) -> Arc { + let unlocked = Arc::new(AtomicBool::new(false)); + let wake = Arc::new(PoolWake { + buffers: quotas.buffers.clone(), + unlocked: unlocked.clone(), + }); + quotas.shared().register(&std::task::Waker::from(wake)); + unlocked + } + /// Wiping initializes every allocated byte without moving or resizing backing. #[test] fn secure_payload_wipe_initializes_spare_capacity_and_preserves_geometry() { @@ -1242,6 +1277,62 @@ mod quota_tests { ); } + /// Recycled and reclaimed charges wake only after the recycler mutex is released. + #[test] + fn recycler_charge_wakes_after_unlock_on_take_and_reclaim() { + let size = 1 << 20; + let quotas = Quotas::new(TestPolicy::new(3 * size, 4)); + let mut charge = quotas.reserve(None, Resource::Other, size).unwrap(); + charge.recycle(vec![0xa7; size]); + + let mut owner = quotas.reserve(None, Resource::Other, size).unwrap(); + let unlocked = watch_buffer_unlock("as); + let bytes = owner.buffer(size).unwrap(); + assert!(unlocked.load(Ordering::SeqCst)); + owner.recycle(bytes); + + let unlocked = watch_buffer_unlock("as); + quotas.reclaim_buffers(); + assert!(unlocked.load(Ordering::SeqCst)); + assert_eq!(quotas.retained_buffer_bytes(), 0); + } + + /// Keyed pressure reclamation drops retained charges outside the pool lock. + #[test] + fn pressure_reclamation_drops_charges_after_unlock() { + let size = 1 << 20; + let quotas = Quotas::new(TestPolicy::new(size, 4)); + let mut charge = quotas.reserve(None, Resource::Other, size).unwrap(); + charge.recycle(vec![0xa7; size]); + + let unlocked = watch_buffer_unlock("as); + let key = "reclaim".to_owned(); + assert!(quotas.reserve(Some(&key), Resource::Other, 1).is_ok()); + assert!(unlocked.load(Ordering::SeqCst)); + assert_eq!(quotas.retained_buffer_bytes(), 0); + } + + /// A poisoned recycler remains usable after a prior operation panicked. + #[test] + fn recycler_recovers_poisoned_mutex() { + use std::panic::{AssertUnwindSafe, catch_unwind}; + + let size = 1 << 20; + let quotas = Quotas::new(TestPolicy::new(2 * size, 4)); + let buffers = quotas.buffers.clone(); + assert!( + catch_unwind(AssertUnwindSafe(|| { + let _buffers = buffers.lock().unwrap(); + panic!("poison recycler mutex"); + })) + .is_err() + ); + + let mut charge = quotas.reserve(None, Resource::Other, size).unwrap(); + charge.recycle(vec![0xa7; size]); + assert_eq!(quotas.retained_buffer_bytes(), size); + } + /// Key-table pressure reports original facts before successful reclamation. #[test] fn key_record_rejection_is_reported_before_reclaim_retry() { From ba4aa2ddfd8bf268458c9f5cf38cad29f5389ab2 Mon Sep 17 00:00:00 2001 From: "copilot-swe-agent[bot]" <198982749+Copilot@users.noreply.github.com> Date: Tue, 6 Oct 2026 16:49:26 +0000 Subject: [PATCH 21/82] Simplify recycler test type Co-authored-by: jveski <7576912+jveski@users.noreply.github.com> --- cmd/racer-dataplane/flow/src/lib.rs | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/cmd/racer-dataplane/flow/src/lib.rs b/cmd/racer-dataplane/flow/src/lib.rs index e28bb277f..215f7aec9 100644 --- a/cmd/racer-dataplane/flow/src/lib.rs +++ b/cmd/racer-dataplane/flow/src/lib.rs @@ -1080,9 +1080,11 @@ mod quota_tests { } } + type TestBufferPool = Arc, Charge)>>>; + /// Checks recycler lock availability during a synchronous charge wake. struct PoolWake { - buffers: Arc, Charge)>>>, + buffers: TestBufferPool, unlocked: Arc, } From 360c0ebb802f6a273fdc61bd994cf6f815e52ab2 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 16:03:01 +0000 Subject: [PATCH 22/82] feat(page-alloc): add multi-placement device backing --- cmd/racer-dataplane/alloc/src/lib.rs | 12 +- cmd/racer-dataplane/alloc/src/slab.rs | 537 ++++++++++++++++++- cmd/racer-dataplane/alloc/tests/workflows.rs | 126 ++++- 3 files changed, 647 insertions(+), 28 deletions(-) diff --git a/cmd/racer-dataplane/alloc/src/lib.rs b/cmd/racer-dataplane/alloc/src/lib.rs index b1f70eac7..554d32dd5 100644 --- a/cmd/racer-dataplane/alloc/src/lib.rs +++ b/cmd/racer-dataplane/alloc/src/lib.rs @@ -29,7 +29,15 @@ //! direct I/O. Parent directories still must be trusted against hostile rename //! and unlink; neither traversal nor flock stops noncooperating writers. //! -//! Empty files are sparsely extended to capacity; nonempty size mismatches are +//! [`Slab::from_devices`] instead accepts one [`DevicePlacement`] per logical +//! segment. The caller opens devices with O_EXCL and O_DIRECT, checks capacity, +//! and supplies common alignment. Placements can share an `Arc` to use one +//! runtime descriptor per device. Startup never creates, resizes, or locks these +//! files. Logical extents still use segment-table offsets; submissions translate +//! them to the placement's physical range after checking lease bounds. Keep ranges +//! disjoint across workers, and do not change file flags or sizes while in use. +//! +//! In file mode, empty files are sparsely extended to capacity; nonempty size mismatches are //! rejected without truncation. Capacity is a logical bound, not reserved disk //! space, so later writes can fail with ENOSPC. Recycling changes metadata only: //! it does not erase, truncate, or hole-punch disk bytes. Buffer zeroization is @@ -138,7 +146,7 @@ pub use segments::{ FreezeGuard, Generation, SegmentClock, SegmentEntries, SegmentId, SegmentLease, SegmentSnapshot, SegmentState, Segments, }; -pub use slab::Slab; +pub use slab::{DevicePlacement, Slab}; use std::{ alloc::{Layout, alloc_zeroed, dealloc}, cell::RefCell, diff --git a/cmd/racer-dataplane/alloc/src/slab.rs b/cmd/racer-dataplane/alloc/src/slab.rs index 5a1832aac..cfc2d6999 100644 --- a/cmd/racer-dataplane/alloc/src/slab.rs +++ b/cmd/racer-dataplane/alloc/src/slab.rs @@ -1,4 +1,4 @@ -//! Locked sparse storage and completion-owned direct I/O. +//! Sparse file or device storage and completion-owned direct I/O. //! //! Startup is blocking and belongs outside the latency-sensitive worker path. //! Capacity bounds logical addresses, not reserved physical space: writes can @@ -12,6 +12,7 @@ use crate::{ }; use std::{ cell::{Cell, RefCell}, + collections::HashMap, ffi::CString, fs::File, future::Future, @@ -22,6 +23,7 @@ use std::{ path::{Component, Path, PathBuf}, pin::Pin, rc::Rc, + sync::Arc, task::{Context, Poll, Waker}, }; use uring_runtime::{ @@ -29,12 +31,32 @@ use uring_runtime::{ reactor::{Reactor, descriptor::Descriptor}, }; -/// One sparse direct-I/O cache file, not a durable storage transaction. +/// One logical segment's range on a caller-opened device. +/// Share one Arc per device within a worker to avoid per-segment descriptors. +#[derive(Clone, Debug)] +pub struct DevicePlacement { + /// Read/write O_DIRECT file, opened exclusively by the caller for real devices. + pub file: Arc, + + /// Physical start of this segment, in bytes. + pub offset: u64, +} + +/// Startup resources, before worker-local descriptor creation. +enum Backing { + File(PathBuf), + Devices { + placements: RefCell>, + geometry: SegmentGeometry, + }, +} + +/// Direct-I/O cache storage, not a durable storage transaction. /// `open_configured` binds submissions to one segment table; read/write require /// a successful binding, not merely an open file. #[repr(align(64))] pub struct Slab { - path: PathBuf, + backing: Backing, capacity_bytes: u64, @@ -58,7 +80,7 @@ impl Slab { max_record_bytes: usize, ) -> Self { Self { - path, + backing: Backing::File(path), capacity_bytes, segment_bytes, max_record_bytes, @@ -68,7 +90,93 @@ impl Slab { } } - /// Logical file capacity, not a reservation of physical disk blocks. + /// Describe device ranges in logical segment order, without creating, sizing, + /// or locking files. The caller must verify device capacity and supply alignment + /// that meets every device's requirements. Regular direct files are also accepted. + /// Overlapping ranges on the same device or file are rejected within this slab; + /// the caller must keep ranges in different slabs disjoint. + /// Files stay owned until `open_configured` creates worker-local descriptors. + pub fn from_devices( + placements: Vec, + segment_bytes: u64, + max_record_bytes: usize, + alignment: Alignment, + ) -> Result { + let capacity_bytes = (placements.len() as u64) + .checked_mul(segment_bytes) + .filter(|&size| size != 0 && size <= i64::MAX as u64) + .ok_or(Error::InvalidConfiguration)?; + let mut slab = Self::new( + PathBuf::new(), + capacity_bytes, + segment_bytes, + max_record_bytes, + ); + let geometry = slab.validate_layout(alignment, capacity_bytes)?; + let mut files = HashMap::new(); + let mut ranges = Vec::with_capacity(placements.len()); + for placement in &placements { + let end = placement + .offset + .checked_add(segment_bytes) + .filter(|&end| end <= i64::MAX as u64) + .ok_or(Error::InvalidConfiguration)?; + if !placement.offset.is_multiple_of(alignment.offset()) { + return Err(Error::InvalidConfiguration); + } + let key = Arc::as_ptr(&placement.file); + let metadata = match files.entry(key) { + std::collections::hash_map::Entry::Occupied(entry) => entry.into_mut(), + std::collections::hash_map::Entry::Vacant(entry) => { + // SAFETY: F_GETFL only inspects the live caller-owned descriptor. + let flags = unsafe { libc::fcntl(placement.file.as_raw_fd(), libc::F_GETFL) }; + if flags < 0 { + return Err(system_error("fcntl-getfl", std::io::Error::last_os_error())); + } + if flags & libc::O_DIRECT == 0 + || flags & libc::O_ACCMODE != libc::O_RDWR + || flags & libc::O_APPEND != 0 + { + return Err(Error::InvalidConfiguration); + } + let metadata = placement + .file + .metadata() + .map_err(|e| system_error("fstat", e))?; + if !matches!( + metadata.mode() & libc::S_IFMT, + libc::S_IFREG | libc::S_IFBLK + ) { + return Err(Error::InvalidConfiguration); + } + entry.insert(metadata) + } + }; + let identity = if metadata.mode() & libc::S_IFMT == libc::S_IFBLK { + (libc::S_IFBLK, metadata.rdev(), 0) + } else { + if end > metadata.len() { + return Err(Error::InvalidConfiguration); + } + (libc::S_IFREG, metadata.dev(), metadata.ino()) + }; + ranges.push((identity, placement.offset, end)); + } + ranges.sort_unstable(); + if ranges + .windows(2) + .any(|pair| pair[0].0 == pair[1].0 && pair[0].2 > pair[1].1) + { + return Err(Error::InvalidConfiguration); + } + slab.backing = Backing::Devices { + placements: RefCell::new(placements), + geometry, + }; + Ok(slab) + } + + /// Logical capacity, not a reservation of physical disk blocks. pub fn capacity_bytes(&self) -> u64 { self.capacity_bytes } @@ -104,7 +212,7 @@ impl Slab { }) } - /// Discovered direct-I/O requirements, or Unavailable before startup. + /// Discovered or caller-supplied direct-I/O requirements, or Unavailable before startup. pub fn alignment(&self) -> Result { self.opened .borrow() @@ -130,7 +238,7 @@ impl Slab { self.bind(segments) } - /// Attach the table identity to this file only after all dimensions agree. + /// Attach the table identity to this backing only after all dimensions agree. fn bind(&self, segments: &Segments) -> Result<()> { let mut opened = self.opened.borrow_mut(); let slab = opened.as_mut().ok_or(Error::Unavailable)?; @@ -179,11 +287,44 @@ impl Slab { self.open_file() } - /// Open and validate the physical file without granting submission authority. + /// Prepare backing descriptors without granting submission authority. fn open_file(&self) -> Result { if let Ok(a) = self.alignment() { return Ok(a); } + let path = match &self.backing { + Backing::File(path) => path, + Backing::Devices { + placements, + geometry, + } => { + let mut files = HashMap::new(); + let mut opened = Vec::with_capacity(placements.borrow().len()); + for placement in placements.borrow().iter() { + let file = match files.entry(Arc::as_ptr(&placement.file)) { + std::collections::hash_map::Entry::Occupied(entry) => entry.into_mut(), + std::collections::hash_map::Entry::Vacant(entry) => { + let file = placement + .file + .try_clone() + .map_err(|e| system_error("dup", e))?; + entry.insert(Rc::new(Descriptor::from(file))) + } + }; + opened.push(OpenPlacement { + file: file.clone(), + offset: placement.offset, + }); + } + *self.opened.borrow_mut() = Some(OpenSlab { + backing: OpenBacking::Devices(opened), + geometry: *geometry, + table: None, + }); + *placements.borrow_mut() = Vec::new(); + return Ok(geometry.alignment()); + } + }; if self.segment_bytes == 0 || self.capacity_bytes == 0 || self.max_record_bytes == 0 @@ -194,14 +335,14 @@ impl Slab { } #[cfg(feature = "simulation")] if let Some(sim) = uring_runtime::reactor::simulation::Simulation::current() { - if let Some(parent) = self.path.parent().filter(|p| !p.as_os_str().is_empty()) { + if let Some(parent) = path.parent().filter(|p| !p.as_os_str().is_empty()) { sim.create_dir_all(parent) .map_err(|e| system_error("mkdir", e))?; } let file = sim .open( None, - &self.path, + path, libc::O_CREAT | libc::O_RDWR | libc::O_DIRECT @@ -224,7 +365,7 @@ impl Slab { self.publish(file, geometry); return Ok(a); } - let file = open_private_file(&self.path)?; + let file = open_private_file(path)?; let metadata = file.metadata().map_err(|e| system_error("fstat", e))?; // SAFETY: geteuid has no arguments or borrowed memory. validate_file( @@ -272,7 +413,7 @@ impl Slab { /// Publish geometry with its owning file, initially without I/O authority. fn publish(&self, file: Descriptor, geometry: SegmentGeometry) { *self.opened.borrow_mut() = Some(OpenSlab { - file: Rc::new(file), + backing: OpenBacking::File(Rc::new(file)), geometry, table: None, }); @@ -302,7 +443,7 @@ impl Slab { extent: Extent, buffer: &AlignedBuffer, lease: &SegmentLease, - ) -> Result> { + ) -> Result<(Rc, Extent)> { let opened = self.opened.borrow(); let slab = opened.as_ref().ok_or(Error::Unavailable)?; let table = slab.table.as_ref().ok_or(Error::Unavailable)?; @@ -336,7 +477,32 @@ impl Slab { { return Err(Error::Corrupt); } - Ok(slab.file.clone()) + match &slab.backing { + OpenBacking::File(file) => Ok((file.clone(), extent)), + OpenBacking::Devices(placements) => { + let position = usize::try_from(lease.id().0).map_err(|_| Error::Corrupt)?; + let placement = placements.get(position).ok_or(Error::Corrupt)?; + let offset = placement + .offset + .checked_add(extent.offset() - start) + .ok_or(Error::Corrupt)?; + let physical = Extent::new(offset, extent.length())?; + let end = offset + .checked_add(extent.length() as u64) + .ok_or(Error::Corrupt)?; + if end + > placement + .offset + .checked_add(self.segment_bytes) + .ok_or(Error::Corrupt)? + || end > i64::MAX as u64 + { + return Err(Error::Corrupt); + } + slab.geometry.alignment().check(physical, buffer)?; + Ok((placement.file.clone(), physical)) + } + } } /// Consume a checked request so its descriptor, buffer, and lease travel together. @@ -346,7 +512,7 @@ impl Slab { buffer: AlignedBuffer, lease: SegmentLease, ) -> Result> { - let file = self.submission(extent, &buffer, &lease)?; + let (file, extent) = self.submission(extent, &buffer, &lease)?; Ok(Submission { file, extent, @@ -414,38 +580,51 @@ impl Slab { .pooled(&self.idle_buffer)) } - /// Replace the open file for fault-injection tests; geometry remains unchanged. + /// Replace file-mode backing for fault tests; device mode is rejected. #[cfg(feature = "simulation")] #[doc(hidden)] pub fn replace_file_for_test(&self, file: File) -> Result<()> { self.replace_descriptor_for_test(file.into()) } - /// Also accepts a virtual descriptor from the simulation backend. + /// Accept a virtual descriptor for file-mode fault tests; geometry stays unchanged. #[cfg(feature = "simulation")] #[doc(hidden)] pub fn replace_descriptor_for_test(&self, file: Descriptor) -> Result<()> { if self.writes_in_flight() != 0 { return Err(Error::Busy); } - self.opened - .borrow_mut() - .as_mut() - .ok_or(Error::Unavailable)? - .file = Rc::new(file); + let mut opened = self.opened.borrow_mut(); + let slab = opened.as_mut().ok_or(Error::Unavailable)?; + let OpenBacking::File(current) = &mut slab.backing else { + return Err(Error::InvalidConfiguration); + }; + *current = Rc::new(file); Ok(()) } } -/// A file and its geometry own their binding; a closed slab cannot retain authority. +/// Backing and geometry own their binding; a closed slab cannot retain authority. struct OpenSlab { - file: Rc, + backing: OpenBacking, geometry: SegmentGeometry, table: Option, } +/// Runtime owners shared by all submissions to a backing file. +enum OpenBacking { + File(Rc), + Devices(Vec), +} + +/// One logical segment's worker-local destination. +struct OpenPlacement { + file: Rc, + offset: u64, +} + /// A validated transfer owns every resource needed to keep kernel access safe. /// Only Slab::prepare constructs this capability, and submission consumes it. struct Submission { @@ -903,6 +1082,311 @@ mod tests { assert!(pool.borrow().is_none()); } + /// Open direct files without using the slab's private-file startup path. + fn device_file(path: &Path) -> Arc { + use std::os::unix::fs::OpenOptionsExt; + let file = std::fs::OpenOptions::new() + .read(true) + .write(true) + .create_new(true) + .custom_flags(libc::O_DIRECT) + .open(path) + .unwrap(); + file.set_len(16384).unwrap(); + Arc::new(file) + } + + /// Device layouts reject aliases, misalignment, overflow, and invalid file flags. + #[test] + fn device_layout_validation_is_read_only() { + use std::os::unix::fs::OpenOptionsExt; + let directory = Directory::new(); + let file = device_file(&directory.0.join("device")); + let Some(a) = real_alignment(probe(&file)) else { + return; + }; + let build = |offsets: &[u64], segment, record| { + Slab::<()>::from_devices( + offsets + .iter() + .map(|&offset| DevicePlacement { + file: file.clone(), + offset, + }) + .collect(), + segment, + record, + a, + ) + }; + for (offsets, segment, record) in [ + (vec![], 4096, 512), + (vec![0], 0, 512), + (vec![0], 4096, 0), + (vec![0], 4097, 512), + (vec![0], 4096, 4097), + (vec![0, 4096], u64::MAX, 512), + (vec![0], i64::MAX as u64 + 1, 512), + (vec![1], 4096, 512), + (vec![16384], 4096, 512), + (vec![u64::MAX], 4096, 512), + (vec![i64::MAX as u64 - 4095], 4096, 512), + (vec![0, 0], 4096, 512), + (vec![0, 2048], 4096, 512), + ] { + assert!(matches!( + build(&offsets, segment, record), + Err(Error::InvalidConfiguration) + )); + } + let duplicate = Arc::new(file.try_clone().unwrap()); + assert!(matches!( + Slab::<()>::from_devices( + vec![ + DevicePlacement { + file: file.clone(), + offset: 0 + }, + DevicePlacement { + file: duplicate, + offset: 2048 + }, + ], + 4096, + 512, + a + ), + Err(Error::InvalidConfiguration) + )); + for flags in [0, libc::O_DIRECT | libc::O_APPEND] { + let invalid = std::fs::OpenOptions::new() + .read(true) + .write(true) + .custom_flags(flags) + .open(directory.0.join("device")) + .unwrap(); + assert!(matches!( + Slab::<()>::from_devices( + vec![DevicePlacement { + file: Arc::new(invalid), + offset: 0, + }], + 4096, + 512, + a + ), + Err(Error::InvalidConfiguration) + )); + } + let readonly = std::fs::OpenOptions::new() + .read(true) + .custom_flags(libc::O_DIRECT) + .open(directory.0.join("device")) + .unwrap(); + assert!(matches!( + Slab::<()>::from_devices( + vec![DevicePlacement { + file: Arc::new(readonly), + offset: 0, + }], + 4096, + 512, + a + ), + Err(Error::InvalidConfiguration) + )); + let slab = build(&[8192, 4096], 4096, 512).unwrap(); + assert_eq!(slab.capacity_bytes(), 8192); + assert_eq!( + slab.open_configured(&Segments::new(8192)), + Err(Error::InvalidConfiguration) + ); + let segments = Segments::new(4096); + assert_eq!(slab.open_configured(&segments), Ok(a)); + assert_eq!(slab.open_configured(&segments), Ok(a)); + assert_eq!(file.metadata().unwrap().len(), 16384); + let independent = File::open(directory.0.join("device")).unwrap(); + // SAFETY: this live independent descriptor tests that startup took no flock. + assert_eq!( + unsafe { libc::flock(independent.as_raw_fd(), libc::LOCK_EX | libc::LOCK_NB) }, + 0 + ); + let opened = slab.opened.borrow(); + let OpenBacking::Devices(placements) = &opened.as_ref().unwrap().backing else { + unreachable!() + }; + assert!(Rc::ptr_eq(&placements[0].file, &placements[1].file)); + let Backing::Devices { placements, .. } = &slab.backing else { + unreachable!() + }; + assert!(placements.borrow().is_empty()); + } + + /// Translation preserves logical authority and rechecks physical bounds and alignment. + #[test] + fn device_submission_checks_logical_and_physical_extents() { + let directory = Directory::new(); + let file = device_file(&directory.0.join("device")); + let Some(a) = real_alignment(probe(&file)) else { + return; + }; + let slab = Slab::<()>::from_devices( + vec![ + DevicePlacement { + file: file.clone(), + offset: 8192, + }, + DevicePlacement { file, offset: 0 }, + ], + 4096, + 4096, + a, + ) + .unwrap(); + let segments = Segments::new(4096); + let _ = slab.open_configured(&segments).unwrap(); + drop(segments.append(4096).unwrap()); + let (lease, extent) = segments.append(4096).unwrap(); + let buffer = slab.allocate(4096, ()).unwrap(); + let (_, physical) = slab.submission(extent, &buffer, &lease).unwrap(); + assert_eq!(physical, Extent::new(0, 4096).unwrap()); + assert!(matches!( + slab.submission(Extent::new(0, 4096).unwrap(), &buffer, &lease), + Err(Error::Corrupt) + )); + assert!(matches!( + slab.submission(Extent::new(8192, 4096).unwrap(), &buffer, &lease), + Err(Error::Corrupt) + )); + assert!(matches!( + slab.submission(Extent::new(6144, 4096).unwrap(), &buffer, &lease), + Err(Error::Corrupt) + )); + assert!(matches!( + slab.submission(Extent::new(4097, 4096).unwrap(), &buffer, &lease), + Err(Error::InvalidConfiguration) + )); + let other = Segments::from_geometry(slab.geometry().unwrap()).unwrap(); + let (foreign, extent) = other.append(4096).unwrap(); + assert!(matches!( + slab.submission(extent, &buffer, &foreign), + Err(Error::Stale) + )); + let extent = Extent::new(4096, 4096).unwrap(); + for (offset, expected) in [ + (1, Error::InvalidConfiguration), + (u64::MAX - 511, Error::Corrupt), + (i64::MAX as u64 - 511, Error::Corrupt), + ] { + let mut opened = slab.opened.borrow_mut(); + let OpenBacking::Devices(placements) = &mut opened.as_mut().unwrap().backing else { + unreachable!() + }; + placements[1].offset = offset; + drop(opened); + assert!( + matches!(slab.submission(extent, &buffer, &lease), Err(error) if error == expected) + ); + } + } + + /// Device submissions keep completion error handling and release their fences. + #[cfg(feature = "simulation")] + #[test] + fn device_io_faults_release_completion_resources() { + use uring_runtime::reactor::simulation::{Fault, Simulation}; + + #[derive(Clone, Copy, Debug, PartialEq)] + enum TestError { + Alloc(Error), + Runtime(uring_runtime::Error), + } + impl From for TestError { + fn from(error: Error) -> Self { + Self::Alloc(error) + } + } + impl From for TestError { + fn from(error: uring_runtime::Error) -> Self { + Self::Runtime(error) + } + } + #[derive(Clone)] + struct TestScope; + impl Scope for TestScope { + type Error = TestError; + fn check(&self) -> std::result::Result<(), TestError> { + Ok(()) + } + } + let directory = Directory::new(); + let file = device_file(&directory.0.join("device")); + let Some(a) = real_alignment(probe(&file)) else { + return; + }; + let slab = + Slab::<()>::from_devices(vec![DevicePlacement { file, offset: 4096 }], 4096, 4096, a) + .unwrap(); + let segments = Segments::new(4096); + let _ = slab.open_configured(&segments).unwrap(); + let (lease, extent) = segments.append(4096).unwrap(); + drop(lease); + let sim = Simulation::new(); + let _environment = sim.enter(); + let descriptor = sim + .open( + None, + Path::new("/device-faults"), + libc::O_CREAT | libc::O_RDWR, + ) + .unwrap(); + descriptor.as_sim().unwrap().set_len(8192).unwrap(); + let mut opened = slab.opened.borrow_mut(); + let OpenBacking::Devices(placements) = &mut opened.as_mut().unwrap().backing else { + unreachable!() + }; + placements[0].file = Rc::new(descriptor); + drop(opened); + let reactor = Reactor::::new(16, ()); + for write in [false, true] { + for (fault, expected) in [ + (Fault::Short(512), TestError::Alloc(Error::Io)), + ( + Fault::Errno(libc::EIO), + TestError::Runtime(uring_runtime::Error::Os(libc::EIO)), + ), + ] { + sim.inject(if write { "write" } else { "read" }, fault) + .unwrap(); + let lease = segments.lease(SegmentId(0), Generation(1)).unwrap(); + let buffer = slab.allocate(4096, ()).unwrap(); + let mut operation = if write { + slab.write(&reactor, extent, buffer, lease, &TestScope) + } else { + slab.read(&reactor, extent, buffer, lease, &TestScope) + }; + let mut completed = false; + for _ in 0..100 { + if let Poll::Ready(result) = operation + .as_mut() + .poll(&mut Context::from_waker(Waker::noop())) + { + assert_eq!(result.unwrap_err(), expected); + completed = true; + break; + } + reactor.poll_budgeted(64).unwrap(); + } + assert!(completed); + drop(operation); + assert_eq!(slab.writes_in_flight(), 0); + assert_eq!(reactor.in_flight(), 0); + } + } + segments.begin_evict(SegmentId(0)).unwrap(); + segments.recycle(SegmentId(0)).unwrap(); + } + /// The live descriptor is direct, sparse, exclusively locked, and aligned. #[test] fn real_file_is_direct_aligned_and_sparse_without_reactor() { @@ -919,7 +1403,10 @@ mod tests { assert_eq!(stat.len(), slabs.capacity_bytes()); assert!(stat.blocks() * 512 < stat.len()); let opened = slabs.opened.borrow(); - let fd = opened.as_ref().unwrap().file.as_raw_fd(); + let OpenBacking::File(file) = &opened.as_ref().unwrap().backing else { + unreachable!() + }; + let fd = file.as_raw_fd(); // SAFETY: descriptor and buffers remain live throughout these synchronous calls. assert_ne!( unsafe { libc::fcntl(fd, libc::F_GETFL) } & libc::O_DIRECT, diff --git a/cmd/racer-dataplane/alloc/tests/workflows.rs b/cmd/racer-dataplane/alloc/tests/workflows.rs index 565a17a4e..cd6c43c0d 100644 --- a/cmd/racer-dataplane/alloc/tests/workflows.rs +++ b/cmd/racer-dataplane/alloc/tests/workflows.rs @@ -1,6 +1,6 @@ //! Public allocator workflows across restart, cancellation, accounting, and real I/O. -use page_alloc::{Alignment, Error, Generation, SegmentId, Segments, Slab}; +use page_alloc::{Alignment, DevicePlacement, Error, Generation, SegmentId, Segments, Slab}; #[cfg(feature = "simulation")] use page_alloc::{Charge, SegmentState}; #[cfg(feature = "simulation")] @@ -11,6 +11,7 @@ use std::{ os::fd::FromRawFd, path::PathBuf, pin::Pin, + sync::Arc, task::{Context, Poll, Waker}, }; #[cfg(feature = "simulation")] @@ -844,6 +845,129 @@ fn io_uring_roundtrip_and_completion_fence() { clock.reclaim(&EmptyEntries, 2, 4, 0).unwrap(); } +/// Device placements translate whole segments and interior extents across real files. +#[test] +fn device_placements_route_real_io_and_reject_short_reads() { + use std::os::unix::fs::{FileExt, OpenOptionsExt}; + + if !kernel_available() { + return; + } + let directory = Directory::new(); + let probe = Slab::<()>::new(directory.0.join("probe"), 16384, 4096, 4096); + let Some(alignment) = real_alignment(probe.open_configured(&Segments::new(4096))) else { + return; + }; + let files: Vec<_> = ["one", "two"] + .into_iter() + .map(|name| { + let file = std::fs::OpenOptions::new() + .create_new(true) + .read(true) + .write(true) + .custom_flags(libc::O_DIRECT) + .open(directory.0.join(name)) + .unwrap(); + file.set_len(16384).unwrap(); + Arc::new(file) + }) + .collect(); + let slab = Slab::<()>::from_devices( + vec![ + DevicePlacement { + file: files[0].clone(), + offset: 8192, + }, + DevicePlacement { + file: files[1].clone(), + offset: 4096, + }, + DevicePlacement { + file: files[0].clone(), + offset: 0, + }, + ], + 4096, + 4096, + alignment, + ) + .unwrap(); + assert_eq!(slab.capacity_bytes(), 12288); + assert_eq!(slab.alignment(), Err(Error::Unavailable)); + let segments = Segments::new(4096); + assert_eq!(slab.open_configured(&segments), Ok(alignment)); + let reactor = Reactor::::new(16, ()); + reactor.init().unwrap(); + for id in 0..3 { + for half in 0..2 { + let (lease, extent) = segments.append(2048).unwrap(); + assert_eq!(lease.id(), SegmentId(id)); + assert_eq!(extent.offset(), id * 4096 + half * 2048); + let mut buffer = slab.allocate(2048, ()).unwrap(); + buffer.as_mut_slice().fill((id * 2 + half + 1) as u8); + drop( + drive_real( + &reactor, + slab.write(&reactor, extent, buffer, lease, &TestScope), + ) + .unwrap(), + ); + let read = drive_real( + &reactor, + slab.read( + &reactor, + extent, + slab.allocate(2048, ()).unwrap(), + segments.lease(SegmentId(id), Generation(1)).unwrap(), + &TestScope, + ), + ) + .unwrap(); + assert!( + read.as_slice() + .iter() + .all(|&byte| byte == (id * 2 + half + 1) as u8) + ); + } + } + // Inspect physical addresses independently so symmetric read/write bugs cannot pass. + for (file, offset, value) in [ + (0, 0, 5), + (0, 2048, 6), + (0, 4096, 0), + (0, 8192, 1), + (0, 10240, 2), + (0, 12288, 0), + (1, 0, 0), + (1, 4096, 3), + (1, 6144, 4), + (1, 8192, 0), + ] { + let mut buffer = alignment.allocate(2048, ()).unwrap(); + assert_eq!( + files[file].read_at(buffer.as_mut_slice(), offset).unwrap(), + 2048 + ); + assert!(buffer.as_slice().iter().all(|&byte| byte == value)); + assert_eq!(files[file].metadata().unwrap().len(), 16384); + } + // External truncation violates the startup contract but must still fail closed. + files[1].set_len(4096).unwrap(); + let read = slab.read( + &reactor, + page_alloc::Extent::new(4096, 4096).unwrap(), + slab.allocate(4096, ()).unwrap(), + segments.lease(SegmentId(1), Generation(1)).unwrap(), + &TestScope, + ); + assert!(matches!( + drive_real(&reactor, read), + Err(TestError::Alloc(Error::Io)) + )); + assert_eq!(reactor.in_flight(), 0); + assert_eq!(slab.writes_in_flight(), 0); +} + /// Failed binding retains the open file but never grants admission to the reactor. #[test] fn unbound_and_failed_binding_reject_reads_and_writes_before_reactor_admission() { From 5d23a0d10a36e8862dca3f3e7a4320959193f2dd Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 17:13:53 +0000 Subject: [PATCH 23/82] test(gantry): synchronize late provider with cold-start entry --- internal/gantry/mirror/mirror_coldstart_test.go | 8 ++++---- internal/gantry/mirror/rediscover_test.go | 14 +++++++++++++- 2 files changed, 17 insertions(+), 5 deletions(-) diff --git a/internal/gantry/mirror/mirror_coldstart_test.go b/internal/gantry/mirror/mirror_coldstart_test.go index 9ff432f8f..b107534af 100644 --- a/internal/gantry/mirror/mirror_coldstart_test.go +++ b/internal/gantry/mirror/mirror_coldstart_test.go @@ -86,14 +86,14 @@ func (d *countingPeerDialer) Calls(addr string) int { func (s *stubColdStart) Resolve(_ context.Context, d digest.Digest, _ ifaces.OriginRefKind, _, _ string, _ int64) (*mirror.ColdStartResolution, error) { atomic.AddInt32(&s.calls, 1) - if s.err != nil { - return nil, s.err - } - if s.onResolve != nil { s.onResolve(d) } + if s.err != nil { + return nil, s.err + } + return &mirror.ColdStartResolution{Providers: s.providers, Outcome: "stub"}, nil } diff --git a/internal/gantry/mirror/rediscover_test.go b/internal/gantry/mirror/rediscover_test.go index e59386897..564c58b62 100644 --- a/internal/gantry/mirror/rediscover_test.go +++ b/internal/gantry/mirror/rediscover_test.go @@ -156,7 +156,12 @@ func TestMirror_Rediscover_ColdExhaustedFlushesHeadersBeforeLateProvider(t *test dialer.Put(lateAddr, d, body) dht := fakes.NewDHT() - coldStart := &stubColdStart{err: mirror.ErrColdStartExhausted} + coldStartEntered := make(chan struct{}) + signalColdStart := sync.OnceFunc(func() { close(coldStartEntered) }) + coldStart := &stubColdStart{ + err: mirror.ErrColdStartExhausted, + onResolve: func(digest.Digest) { signalColdStart() }, + } m := mirror.New(cfg, fakes.NewCache(), oc, mirror.WithLiveStreamThrough(), @@ -188,6 +193,13 @@ func TestMirror_Rediscover_ColdExhaustedFlushesHeadersBeforeLateProvider(t *test t.Fatalf("peer calls before advertise = %d, want 0", got) } + // Headers flush before round 0; wait until its empty lookup reaches cold-start. + select { + case <-coldStartEntered: + case <-time.After(2 * time.Second): + t.Fatal("cold-start was not entered before provider advertisement") + } + dht.Inject(d, ifaces.Provider{NodeID: "late-seed", Addr: lateAddr}) got, err := io.ReadAll(resp.Body) From 097bcd6f42beea7c77768ef185c85533b3dfed10 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 17:50:30 +0000 Subject: [PATCH 24/82] fix(flow): keep cohort completion results immutable --- cmd/racer-dataplane/flow/src/coalesce.rs | 70 ++++++++++++++++++++++++ 1 file changed, 70 insertions(+) diff --git a/cmd/racer-dataplane/flow/src/coalesce.rs b/cmd/racer-dataplane/flow/src/coalesce.rs index 1b54c8b5c..ae85a05ad 100644 --- a/cmd/racer-dataplane/flow/src/coalesce.rs +++ b/cmd/racer-dataplane/flow/src/coalesce.rs @@ -177,9 +177,13 @@ impl Registration { /// Broadcast a caller-owned value and close admission before waking readers. /// The operation owner must call this only after completion, not cancellation. + /// The first result is final; later calls leave it unchanged. pub fn finish(&self, result: V) { let wakers = { let mut state = self.owner.cohort.borrow_mut(); + if state.result.is_some() { + return; + } state.result = Some(result); state.leader = None; state.take_wakers() @@ -1869,6 +1873,72 @@ mod tests { } } + /// The first result survives later finishes and remains separate from replacements. + #[test] + fn first_completion_wins_for_early_and_late_readers() { + for result in [Ok(42), Err("failed")] { + let table = table(5, 2); + let leader = table.join("key", 1).unwrap(); + let early = table.join("key", 1).unwrap(); + let late = table.join("key", 1).unwrap(); + let survivor = leader.clone(); + let count = Arc::new(Counter::default()); + let waker = Waker::from(count.clone()); + assert_eq!(leader.event(Waker::noop()), Poll::Ready(Event::Lead)); + assert_eq!(early.event(&waker), Poll::Pending); + leader.finish(result); + drop(leader); + assert_eq!(count.0.load(Ordering::Relaxed), 1); + assert_eq!(early.event(&waker), Poll::Ready(Event::Complete(result))); + assert_eq!(table.active_count(), 0); + + let next = table.join("key", 1).unwrap(); + assert_eq!(next.event(Waker::noop()), Poll::Ready(Event::Lead)); + survivor.finish(Ok(99)); + early.finish(Err("late failure")); + assert_eq!(count.0.load(Ordering::Relaxed), 1); + assert_eq!(early.event(&waker), Poll::Ready(Event::Complete(result))); + assert_eq!( + late.event(Waker::noop()), + Poll::Ready(Event::Complete(result)) + ); + assert_eq!( + survivor.event(Waker::noop()), + Poll::Ready(Event::Complete(result)) + ); + assert_eq!(table.active_count(), 1); + let next_follower = table.join("key", 1).unwrap(); + assert_eq!(next_follower.event(Waker::noop()), Poll::Pending); + next.finish(Ok(7)); + assert_eq!( + next_follower.event(Waker::noop()), + Poll::Ready(Event::Complete(Ok(7))) + ); + assert_eq!( + late.event(Waker::noop()), + Poll::Ready(Event::Complete(result)) + ); + drop((survivor, early, late, next, next_follower)); + assert_eq!(table.registration_count(), 0); + assert_eq!(table.active_count(), 0); + } + } + + /// Attempt exhaustion is also a terminal result, even for unpolled followers. + #[test] + fn first_completion_from_exhaustion_cannot_be_overwritten() { + let table = table(2, 0); + let first = table.join("key", 1).unwrap(); + let late = table.join("key", 1).unwrap(); + let exhausted = Poll::Ready(Event::Complete(Err("exhausted"))); + assert_eq!(first.event(Waker::noop()), exhausted); + first.finish(Ok(1)); + late.finish(Err("late failure")); + assert_eq!(first.event(Waker::noop()), exhausted); + assert_eq!(late.event(Waker::noop()), exhausted); + assert_eq!(table.active_count(), 0); + } + /// Detached execution retains one waiter until its final handle is dropped. #[test] fn last_handle_drop_releases_leadership_and_registration() { From 6d1f7ed32043f089b659eed505c4fef6be50f84d Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 17:50:44 +0000 Subject: [PATCH 25/82] fix(flow): recheck stop after shared quota reservation --- cmd/racer-dataplane/flow/src/lib.rs | 49 +++++++++++++++++++++++++++-- 1 file changed, 47 insertions(+), 2 deletions(-) diff --git a/cmd/racer-dataplane/flow/src/lib.rs b/cmd/racer-dataplane/flow/src/lib.rs index 215f7aec9..5001a6e33 100644 --- a/cmd/racer-dataplane/flow/src/lib.rs +++ b/cmd/racer-dataplane/flow/src/lib.rs @@ -284,7 +284,7 @@ impl SharedQuotas

{ }); Error::Overloaded })?; - Ok(Charge { + let charge = Charge { class, amount, key: None, @@ -292,7 +292,13 @@ impl SharedQuotas

{ local: None, buffers: Weak::new(), stopped: self.stopped.clone(), - }) + }; + // Stop may race the policy callback or counter reservation. Dropping the + // charge rolls back usage and applies the usual release wake policy. + if self.is_stopped() && !P::allows_stopped(class) { + return Err(Error::Unavailable); + } + Ok(charge) } } @@ -1029,6 +1035,8 @@ mod quota_tests { max_keys: usize, rejected: Arc>>>, + + limit_gate: Option>, } impl TestPolicy { @@ -1038,6 +1046,7 @@ mod quota_tests { limit, max_keys, rejected: Arc::default(), + limit_gate: None, } } } @@ -1051,6 +1060,10 @@ mod quota_tests { /// All fixture classes use the same aggregate ceiling. fn limit(&self, _: Resource) -> usize { + if let Some(gate) = &self.limit_gate { + gate.0.wait(); + gate.1.wait(); + } self.limit } @@ -1403,6 +1416,38 @@ mod quota_tests { )); } + /// Stop during the limit callback rolls back only ordinary admission. + #[test] + fn shared_reservation_rechecks_stop_after_limit_callback() { + for class in [Resource::Payload, Resource::Other, Resource::Progress] { + let mut quotas = Quotas::new(TestPolicy::new(10, 1)); + let existing = quotas.reserve(None, class, 3).unwrap(); + let gate = Arc::new((std::sync::Barrier::new(2), std::sync::Barrier::new(2))); + quotas.policy.limit_gate = Some(gate.clone()); + let shared = quotas.shared(); + let reservation = std::thread::spawn(move || shared.reserve(class, 7)); + + gate.0.wait(); + assert_eq!(quotas.used(class), 3); + quotas.stop(); + gate.1.wait(); + + let result = reservation.join().unwrap(); + if TestPolicy::allows_stopped(class) { + let charge = result.unwrap(); + assert_eq!(charge.amount(), 7); + assert_eq!(quotas.used(class), 10); + drop(charge); + } else { + assert!(matches!(result, Err(Error::Unavailable))); + } + assert_eq!(quotas.used(class), 3); + assert!(quotas.policy.rejected.lock().unwrap().is_empty()); + drop(existing); + assert_eq!(quotas.used(class), 0); + } + } + /// Cross-thread release and local stop notify the registered shared waiter. #[test] fn release_and_stop_wake_shared_waiters() { From 7146be83db47db9d1bba44677ae7a905e601bf73 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 17:50:48 +0000 Subject: [PATCH 26/82] fix(flow): keep adaptive key clone panics outside admission lock --- cmd/racer-dataplane/flow/src/admission.rs | 96 +++++++++++++++++++++-- 1 file changed, 88 insertions(+), 8 deletions(-) diff --git a/cmd/racer-dataplane/flow/src/admission.rs b/cmd/racer-dataplane/flow/src/admission.rs index 5f7059ab4..a67c60198 100644 --- a/cmd/racer-dataplane/flow/src/admission.rs +++ b/cmd/racer-dataplane/flow/src/admission.rs @@ -189,6 +189,8 @@ impl Adaptive { /// Admit work or one exclusive recovery probe without waiting. pub fn acquire(self: &Arc, key: &K) -> Result>> { + let indexed = key.clone(); + let owned = key.clone(); let now = (self.now)(); let mut state = self.state.lock().map_err(|_| Error::Unavailable)?; if state.active >= state.limit { @@ -198,22 +200,19 @@ impl Adaptive { if !state.peers.contains_key(key) && state.peers.len() == self.config.capacity { let retired = state .peers - .iter() - .find(|(_, p)| { + .extract_if(.., |_, p| { p.active == 0 && !p.probe && p.retry.is_none_or(|retry| now >= retry) && now.saturating_duration_since(p.updated) >= self.config.retire_after }) - .map(|(k, _)| k.clone()); - if let Some(retired) = retired { - state.peers.remove(&retired); - } else { + .next(); + if retired.is_none() { self.observer.event(Event::Rejected); return Err(Error::Overloaded); } } - let peer = state.peers.entry(key.clone()).or_insert(Peer { + let peer = state.peers.entry(indexed).or_insert(Peer { active: 0, limit: self.config.per_key, generation: 0, @@ -241,7 +240,7 @@ impl Adaptive { } Ok(Arc::new(Permit { owner: self.clone(), - key: key.clone(), + key: owned, generation, probe, })) @@ -1107,6 +1106,87 @@ mod tests { } } + /// Clone panics leave the mutex usable and all admission capacity recoverable. + #[test] + fn adaptive_clone_panic_preserves_capacity_and_mutex() { + use std::{ + panic::{AssertUnwindSafe, catch_unwind}, + sync::atomic::{AtomicUsize, Ordering}, + }; + + static CLONES: AtomicUsize = AtomicUsize::new(0); + static PANIC_AT: AtomicUsize = AtomicUsize::new(usize::MAX); + static PANIC_KEY: AtomicUsize = AtomicUsize::new(usize::MAX); + + /// Key with selectable clone failures, including during retirement. + #[derive(Eq, Ord, PartialEq, PartialOrd)] + struct Key(usize); + + impl Clone for Key { + fn clone(&self) -> Self { + let clone = CLONES.fetch_add(1, Ordering::SeqCst) + 1; + assert_ne!(clone, PANIC_AT.load(Ordering::SeqCst), "key clone failed"); + assert_ne!( + self.0, + PANIC_KEY.load(Ordering::SeqCst), + "retired key cloned" + ); + Self(self.0) + } + } + + for existing in [false, true] { + for panic_at in [1, 2] { + let owner = Adaptive::new( + Config { + total: 2, + per_key: 2, + capacity: 1, + retire_after: Duration::ZERO, + ..config() + }, + Counts::default(), + Instant::now, + ) + .unwrap(); + let held = existing.then(|| owner.acquire(&Key(1)).unwrap()); + CLONES.store(0, Ordering::SeqCst); + PANIC_AT.store(panic_at, Ordering::SeqCst); + let result = catch_unwind(AssertUnwindSafe(|| owner.acquire(&Key(1)))); + PANIC_AT.store(usize::MAX, Ordering::SeqCst); + assert!(result.is_err()); + assert_eq!(CLONES.load(Ordering::SeqCst), panic_at); + { + let state = owner.state.lock().expect("clone panic poisoned state"); + assert_eq!(state.active, usize::from(existing)); + assert_eq!(state.peers.len(), usize::from(existing)); + if existing { + assert_eq!(state.peers[&Key(1)].active, 1); + } + } + drop(held); + let one = owner.acquire(&Key(1)).unwrap(); + let two = owner.acquire(&Key(1)).unwrap(); + assert!(matches!(owner.acquire(&Key(1)), Err(Error::Overloaded))); + drop((one, two)); + assert_eq!(owner.state.lock().unwrap().active, 0); + assert_eq!(*owner.observer.active.lock().unwrap(), 0); + + PANIC_KEY.store(1, Ordering::SeqCst); + let replacement = owner.acquire(&Key(2)).expect("retirement must not clone"); + PANIC_KEY.store(usize::MAX, Ordering::SeqCst); + { + let state = owner.state.lock().unwrap(); + assert_eq!(state.peers.len(), 1); + assert!(!state.peers.contains_key(&Key(1))); + assert_eq!(state.peers[&Key(2)].active, 1); + } + drop(replacement); + assert_eq!(owner.state.lock().unwrap().active, 0); + } + } + } + /// Stale success cannot undo failure and shared fences retain active work. #[test] fn fences_generation_exclusivity_and_local_pressure() { From fed114441f3f89755dfce2c253b361b8b7cf513b Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 17:53:40 +0000 Subject: [PATCH 27/82] fix(flow): release key borrow before rejection callback --- cmd/racer-dataplane/flow/src/lib.rs | 93 +++++++++++++++++++++++++++-- 1 file changed, 88 insertions(+), 5 deletions(-) diff --git a/cmd/racer-dataplane/flow/src/lib.rs b/cmd/racer-dataplane/flow/src/lib.rs index 5001a6e33..3bdf04481 100644 --- a/cmd/racer-dataplane/flow/src/lib.rs +++ b/cmd/racer-dataplane/flow/src/lib.rs @@ -657,6 +657,7 @@ impl Quotas

{ /// Reuse a live key record or create one after bounded retirement cleanup. fn key_counters(&self, key: &P::Key) -> Result>> { + let limit = self.policy.max_keys(); let mut keys = self.keys.borrow_mut(); let mut retired = self.retired_keys.lock().unwrap_or_else(|e| e.into_inner()); for _ in 0..256 { @@ -671,11 +672,10 @@ impl Quotas

{ } } drop(retired); - if !keys.contains_key(key) && keys.len() >= self.policy.max_keys() { - self.policy.rejected(Rejection::Keys { - used: keys.len(), - limit: self.policy.max_keys(), - }); + if !keys.contains_key(key) && keys.len() >= limit { + let used = keys.len(); + drop(keys); + self.policy.rejected(Rejection::Keys { used, limit }); return Err(Error::Overloaded); } if let Some(counts) = keys.get(key).and_then(Weak::upgrade) { @@ -1367,6 +1367,89 @@ mod quota_tests { drop(charge); } + /// Key-table rejection callbacks can inspect and admit work for a live key. + #[test] + fn key_record_rejection_allows_keyed_reentry() { + /// Keep a weak link to the authority and guard callback reentry. + struct ReentrantPolicy { + quotas: RefCell>>, + entered: std::cell::Cell, + rejected: RefCell>>, + } + + impl Policy for ReentrantPolicy { + type Class = Resource; + type Key = String; + + /// Leave room for nested admission under the existing key. + fn limit(&self, _: Resource) -> usize { + 10 + } + + /// Force a rejection for each new key while the first is live. + fn max_keys(&self) -> usize { + 1 + } + + /// This test does not register admission waiters. + fn wakes(_: Resource) -> bool { + false + } + + /// This test does not allocate page backing. + fn covers(_: Resource) -> bool { + false + } + + /// Reenter once and keep all rejection facts for assertions. + fn rejected(&self, rejection: Rejection) { + self.rejected.borrow_mut().push(rejection); + if self.entered.replace(true) { + return; + } + let quotas = self.quotas.borrow().upgrade().unwrap(); + let key = "live".to_owned(); + assert_eq!( + quotas.reclamation(&key, Resource::Payload, 10), + Some((Some(key.clone()), 1)) + ); + let nested = quotas.reserve(Some(&key), Resource::Payload, 2).unwrap(); + assert_eq!(quotas.used(Resource::Payload), 3); + drop(nested); + assert_eq!(quotas.used(Resource::Payload), 1); + } + } + + let quotas = Rc::new(Quotas::new(ReentrantPolicy { + quotas: RefCell::default(), + entered: std::cell::Cell::new(false), + rejected: RefCell::default(), + })); + *quotas.policy.quotas.borrow_mut() = Rc::downgrade("as); + let (live, other) = ("live".to_owned(), "other".to_owned()); + let charge = quotas.reserve(Some(&live), Resource::Payload, 1).unwrap(); + assert!(matches!( + quotas.reserve(Some(&other), Resource::Payload, 1), + Err(Error::Overloaded) + )); + assert!(quotas.policy.entered.get()); + assert!(matches!( + "as.policy.rejected.borrow()[..], + [ + Rejection::Keys { used: 1, limit: 1 }, + Rejection::Keys { used: 1, limit: 1 } + ] + )); + assert_eq!(quotas.used(Resource::Payload), 1); + assert_eq!(quotas.active_keys.load(Ordering::Acquire), 1); + drop(charge); + let replacement = quotas.reserve(Some(&other), Resource::Payload, 1).unwrap(); + assert!(!quotas.keys.borrow().contains_key(&live)); + drop(replacement); + assert_eq!(quotas.used(Resource::Payload), 0); + assert_eq!(quotas.active_keys.load(Ordering::Acquire), 0); + } + /// Charges and shared handles retain exact accounting across threads. #[test] fn charge_validation_and_thread_safe_shared_usage() { From 445d2d6a188c25e18758fe1be0475dc31e2eb1ad Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 17:54:18 +0000 Subject: [PATCH 28/82] fix(flow): allow reentrant circuit backoff callbacks --- cmd/racer-dataplane/flow/src/admission.rs | 123 +++++++++++++++++++++- 1 file changed, 121 insertions(+), 2 deletions(-) diff --git a/cmd/racer-dataplane/flow/src/admission.rs b/cmd/racer-dataplane/flow/src/admission.rs index a67c60198..4ac7dbc32 100644 --- a/cmd/racer-dataplane/flow/src/admission.rs +++ b/cmd/racer-dataplane/flow/src/admission.rs @@ -377,6 +377,8 @@ mod circuit { retry_at: Instant, probe_until: Option, + + pending_backoff: Option>, } impl Circuit { @@ -441,6 +443,8 @@ mod circuit { } /// Record a caller-classified failure using caller-selected retry jitter. + /// Backoff may reenter: the count is visible before the callback, and any + /// newer transition for this key takes precedence over its returned delay. pub fn failure( &self, key: &K, @@ -455,10 +459,27 @@ mod circuit { failures: 0, retry_at: now, probe_until: None, + pending_backoff: None, }); state.failures = state.failures.saturating_add(1); - state.retry_at = now + backoff(key, state.failures); - state.probe_until = None; + let failures = state.failures; + let pending = Rc::new(()); + state.pending_backoff = Some(Rc::clone(&pending)); + drop(states); + + let retry_at = now + backoff(key, failures); + let mut states = self.states.borrow_mut(); + // Never reinsert a removed record or overwrite a newer transition. + if let Some(state) = states.get_mut(key) + && state + .pending_backoff + .as_ref() + .is_some_and(|current| Rc::ptr_eq(current, &pending)) + { + state.retry_at = retry_at; + state.probe_until = None; + state.pending_backoff = None; + } Ok(()) } @@ -485,6 +506,7 @@ mod circuit { return false; } state.probe_until = Some(now + self.probe_timeout); + state.pending_backoff = None; true } @@ -509,6 +531,103 @@ mod circuit { mod tests { use super::*; + /// Backoff can inspect the circuit without borrowing conflicts. + #[test] + fn backoff_reentry_observes_failure_record() { + let now = Instant::now(); + let health = Circuits::new(1, Duration::from_secs(1)); + health + .failure(&7, now, |key, failures| { + assert_eq!((*key, failures), (7, 1)); + assert_eq!(health.len(), 1); + assert!(!health.is_empty()); + assert!(health.available(key, now)); + Duration::from_secs(2) + }) + .unwrap(); + assert!(!health.available(&7, now)); + assert!(health.available(&7, now + Duration::from_secs(2))); + } + + /// A nested success or retention change must not be undone by backoff. + #[test] + fn backoff_reentry_preserves_removal_and_capacity() { + let now = Instant::now(); + for retain in [false, true] { + let health = Circuits::new(1, Duration::ZERO); + health + .failure(&7, now, |_, _| { + if retain { + health.retain(&[]); + } else { + health.success(&7); + } + health.failure(&8, now, |_, _| Duration::ZERO).unwrap(); + Duration::from_secs(10) + }) + .unwrap(); + assert_eq!(health.len(), 1); + assert!(health.available(&7, now)); + assert_eq!( + health.failure(&9, now, |_, _| panic!("capacity is full")), + Err(Error::Overloaded) + ); + health.success(&8); + assert!(health.is_empty()); + } + } + + /// Newer failures win even when counts saturate or the key is replaced. + #[test] + fn backoff_reentry_preserves_newer_failure() { + let now = Instant::now(); + for (initial, replace) in [(0, false), (u32::MAX, false), (0, true)] { + let health = Circuits::new(1, Duration::ZERO); + if initial != 0 { + health.failure(&7, now, |_, _| Duration::ZERO).unwrap(); + health.states.borrow_mut().get_mut(&7).unwrap().failures = initial; + } + health + .failure(&7, now, |_, count| { + assert_eq!(count, initial.saturating_add(1)); + if replace { + health.success(&7); + } + health + .failure(&7, now, |_, nested_count| { + assert_eq!( + nested_count, + if replace { 1 } else { count.saturating_add(1) } + ); + Duration::from_secs(2) + }) + .unwrap(); + Duration::from_secs(10) + }) + .unwrap(); + assert!(!health.available(&7, now)); + assert!(health.available(&7, now + Duration::from_secs(2))); + assert_eq!(health.len(), 1); + } + } + + /// A probe admitted by backoff keeps its timeout after the callback. + #[test] + fn backoff_reentry_preserves_probe_timeout() { + let now = Instant::now(); + let timeout = Duration::from_secs(2); + let health = Circuits::new(1, timeout); + health.failure(&7, now, |_, _| Duration::ZERO).unwrap(); + health + .failure(&7, now, |_, _| { + drop(health.acquire(&7, now).unwrap()); + Duration::from_secs(10) + }) + .unwrap(); + assert!(!health.available(&7, now)); + assert!(health.available(&7, now + timeout)); + } + /// Failed key cloning must not leave an exclusive probe without an owner. #[test] fn review_regression_probe_clone_panic_allows_acquisition_after_timeout() { From eab17a361a40367d56d02d32ad6596c1f55eb226 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 18:51:25 +0000 Subject: [PATCH 29/82] test(alloc): derive device placement sizes from alignment --- cmd/racer-dataplane/alloc/tests/workflows.rs | 112 ++++++++++++++----- 1 file changed, 87 insertions(+), 25 deletions(-) diff --git a/cmd/racer-dataplane/alloc/tests/workflows.rs b/cmd/racer-dataplane/alloc/tests/workflows.rs index cd6c43c0d..4b7472076 100644 --- a/cmd/racer-dataplane/alloc/tests/workflows.rs +++ b/cmd/racer-dataplane/alloc/tests/workflows.rs @@ -845,6 +845,63 @@ fn io_uring_roundtrip_and_completion_fence() { clock.reclaim(&EmptyEntries, 2, 4, 0).unwrap(); } +/// Keep both halves aligned even when offset and length units differ. +fn device_placement_sizes(alignment: Alignment) -> (usize, u64) { + let half_bytes = alignment.extent(0, 2048).unwrap().length(); + let segment_bytes = 2 * half_bytes as u64; + (half_bytes, segment_bytes) +} + +/// Placement test geometry must work without relying on the host's alignment. +#[test] +fn device_placement_sizes_support_4096_and_arbitrary_units() { + for (offset, length, expected_half) in [ + (1, 1, 2048), + (512, 512, 2048), + (4096, 4096, 4096), + (4096, 512, 4096), + (512, 4096, 4096), + (768, 512, 3072), + (3, 5, 2055), + ] { + let alignment = Alignment::new(4096, offset, length).unwrap(); + let (half_bytes, segment_bytes) = device_placement_sizes(alignment); + assert_eq!(half_bytes, expected_half); + assert_eq!(segment_bytes, 2 * expected_half as u64); + let geometry = + page_alloc::SegmentGeometry::new(3 * segment_bytes, segment_bytes, 3, alignment) + .unwrap(); + let segments = Segments::from_geometry(geometry).unwrap(); + let buffer = alignment.allocate(half_bytes, ()).unwrap(); + if half_bytes != 2048 { + assert!(matches!( + segments.append(2048), + Err(Error::InvalidConfiguration) + )); + } + for (id, physical_start) in [2 * segment_bytes, segment_bytes, 0] + .into_iter() + .enumerate() + { + for half in 0..2 { + let (lease, extent) = segments.append(half_bytes).unwrap(); + assert_eq!(lease.id(), SegmentId(id as u64)); + assert_eq!(extent.length(), half_bytes); + assert_eq!( + extent.offset(), + id as u64 * segment_bytes + half * half_bytes as u64 + ); + alignment.check(extent, &buffer).unwrap(); + let physical = + page_alloc::Extent::new(physical_start + half * half_bytes as u64, half_bytes) + .unwrap(); + alignment.check(physical, &buffer).unwrap(); + } + } + assert_eq!(segments.free_count(), 0); + } +} + /// Device placements translate whole segments and interior extents across real files. #[test] fn device_placements_route_real_io_and_reject_short_reads() { @@ -858,6 +915,8 @@ fn device_placements_route_real_io_and_reject_short_reads() { let Some(alignment) = real_alignment(probe.open_configured(&Segments::new(4096))) else { return; }; + let (half_bytes, segment_bytes) = device_placement_sizes(alignment); + let file_bytes = 4 * segment_bytes; let files: Vec<_> = ["one", "two"] .into_iter() .map(|name| { @@ -868,7 +927,7 @@ fn device_placements_route_real_io_and_reject_short_reads() { .custom_flags(libc::O_DIRECT) .open(directory.0.join(name)) .unwrap(); - file.set_len(16384).unwrap(); + file.set_len(file_bytes).unwrap(); Arc::new(file) }) .collect(); @@ -876,34 +935,37 @@ fn device_placements_route_real_io_and_reject_short_reads() { vec![ DevicePlacement { file: files[0].clone(), - offset: 8192, + offset: 2 * segment_bytes, }, DevicePlacement { file: files[1].clone(), - offset: 4096, + offset: segment_bytes, }, DevicePlacement { file: files[0].clone(), offset: 0, }, ], - 4096, - 4096, + segment_bytes, + segment_bytes as usize, alignment, ) .unwrap(); - assert_eq!(slab.capacity_bytes(), 12288); + assert_eq!(slab.capacity_bytes(), 3 * segment_bytes); assert_eq!(slab.alignment(), Err(Error::Unavailable)); - let segments = Segments::new(4096); + let segments = Segments::new(segment_bytes); assert_eq!(slab.open_configured(&segments), Ok(alignment)); let reactor = Reactor::::new(16, ()); reactor.init().unwrap(); for id in 0..3 { for half in 0..2 { - let (lease, extent) = segments.append(2048).unwrap(); + let (lease, extent) = segments.append(half_bytes).unwrap(); assert_eq!(lease.id(), SegmentId(id)); - assert_eq!(extent.offset(), id * 4096 + half * 2048); - let mut buffer = slab.allocate(2048, ()).unwrap(); + assert_eq!( + extent.offset(), + id * segment_bytes + half * half_bytes as u64 + ); + let mut buffer = slab.allocate(half_bytes, ()).unwrap(); buffer.as_mut_slice().fill((id * 2 + half + 1) as u8); drop( drive_real( @@ -917,7 +979,7 @@ fn device_placements_route_real_io_and_reject_short_reads() { slab.read( &reactor, extent, - slab.allocate(2048, ()).unwrap(), + slab.allocate(half_bytes, ()).unwrap(), segments.lease(SegmentId(id), Generation(1)).unwrap(), &TestScope, ), @@ -933,30 +995,30 @@ fn device_placements_route_real_io_and_reject_short_reads() { // Inspect physical addresses independently so symmetric read/write bugs cannot pass. for (file, offset, value) in [ (0, 0, 5), - (0, 2048, 6), - (0, 4096, 0), - (0, 8192, 1), - (0, 10240, 2), - (0, 12288, 0), + (0, half_bytes as u64, 6), + (0, segment_bytes, 0), + (0, 2 * segment_bytes, 1), + (0, 2 * segment_bytes + half_bytes as u64, 2), + (0, 3 * segment_bytes, 0), (1, 0, 0), - (1, 4096, 3), - (1, 6144, 4), - (1, 8192, 0), + (1, segment_bytes, 3), + (1, segment_bytes + half_bytes as u64, 4), + (1, 2 * segment_bytes, 0), ] { - let mut buffer = alignment.allocate(2048, ()).unwrap(); + let mut buffer = alignment.allocate(half_bytes, ()).unwrap(); assert_eq!( files[file].read_at(buffer.as_mut_slice(), offset).unwrap(), - 2048 + half_bytes ); assert!(buffer.as_slice().iter().all(|&byte| byte == value)); - assert_eq!(files[file].metadata().unwrap().len(), 16384); + assert_eq!(files[file].metadata().unwrap().len(), file_bytes); } // External truncation violates the startup contract but must still fail closed. - files[1].set_len(4096).unwrap(); + files[1].set_len(segment_bytes).unwrap(); let read = slab.read( &reactor, - page_alloc::Extent::new(4096, 4096).unwrap(), - slab.allocate(4096, ()).unwrap(), + page_alloc::Extent::new(segment_bytes, segment_bytes as usize).unwrap(), + slab.allocate(segment_bytes as usize, ()).unwrap(), segments.lease(SegmentId(1), Generation(1)).unwrap(), &TestScope, ); From cea65e60b413a7e273f7d16afae898e8ce6ca4d6 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 18:51:25 +0000 Subject: [PATCH 30/82] docs(alloc): clarify device-backed startup ownership --- designs/racer-page-alloc.md | 38 +++++++++++++++++++++++++------------ 1 file changed, 26 insertions(+), 12 deletions(-) diff --git a/designs/racer-page-alloc.md b/designs/racer-page-alloc.md index 56c65cde4..4b9305b3e 100644 --- a/designs/racer-page-alloc.md +++ b/designs/racer-page-alloc.md @@ -6,9 +6,9 @@ `cmd/racer-dataplane/alloc/`. It gives one worker thread three things: 1. Memory buffers that are aligned for direct I/O and wiped before reuse. -2. A table of fixed-size segments inside one sparse cache file, with leases - that stop a segment from being reused while I/O still points at it. -3. Async read and write of that file through the `uring-runtime` reactor. +2. A table of fixed-size segments in a sparse cache file or caller-opened files + or devices, with leases that stop reuse while I/O still points at a segment. +3. Async read and write of that storage through the `uring-runtime` reactor. The crate stores bytes only. It does not know about page keys, record headers, encryption, checksums, or versions. The caller owns the page index and @@ -35,7 +35,7 @@ Non-goals: All state is worker-local. Types use `Rc`, `Cell`, and `RefCell`, so none of them are `Send` or `Sync` (`alloc/src/segments.rs:153-179`, `alloc/src/slab.rs:35-50`). -Each worker owns its own file, segment table, and buffer pool. This removes lock +Each worker owns its storage ranges, segment table, and buffer pool. This removes lock contention and makes ownership easy to reason about. The cost is that one worker cannot hand its storage to another. @@ -59,8 +59,9 @@ next request without paying for a wipe on every allocation. ## Segments -The file is split into fixed-size segments. Segment `n` starts at -`n * segment_bytes`. Each segment has a state and a generation number: +Storage is split into fixed-size logical segments. Segment `n` starts at +`n * segment_bytes`; device placements map it to a caller-supplied physical range. +Each segment has a state and a generation number: ``` Free -> Open -> Sealed -> Evicting -> Free (generation + 1) @@ -105,10 +106,10 @@ caller-provided score instead of recent reads. `reclaim_index` drops index entri (for example, an index size limit) passes. It does not free segments and does not ask `can_evict`. -## File and I/O +## Storage and I/O -`Slab` owns one cache file. Opening is blocking and is meant to run at startup -(`alloc/src/slab.rs:183-279`). It: +`Slab::new` describes one cache file. Opening is blocking and is meant to run at +startup (`alloc/src/slab.rs:368-409`). For this file-backed mode, it: - Walks the path without following symlinks or `..`. - Requires a regular file owned by the current user, mode 0600, one hard link. @@ -117,8 +118,21 @@ ask `can_evict`. - Sizes an empty file sparsely to capacity. A file with the wrong size is rejected, not truncated. -The slab is then bound to one segment table. I/O is refused until binding -succeeds, and a slab cannot be rebound to a different table. +`Slab::from_devices` instead owns caller-opened file or block-device placements, +one per logical segment, without creating, sizing, or locking them. Files must +be read/write with `O_DIRECT` and without `O_APPEND`. The constructor checks +geometry, offset alignment, regular-file bounds, and overlapping ranges within +the slab (`alloc/src/slab.rs:99-176`). The caller must open real devices +exclusively, verify device capacity, supply alignment that meets every device's +requirements, and keep ranges in different slabs disjoint +(`alloc/src/slab.rs:34-42`, `alloc/src/slab.rs:93-98`). + +For device placements, `open_configured` duplicates the owned files into +worker-local descriptors and releases the original placement references +(`alloc/src/slab.rs:295-325`). In both modes, it then binds the slab to one +segment table. I/O is refused until binding succeeds, and a slab cannot be +rebound to a different table (`alloc/src/slab.rs:241-276`, +`alloc/src/slab.rs:447-449`). `read` and `write` check the extent, alignment, and lease, then pass the buffer and lease to the reactor (`alloc/src/slab.rs:461-515`). The reactor holds both @@ -131,7 +145,7 @@ and it does not flush to disk. ## Simulation -The `simulation` feature routes file open, lock, stat, and sizing to the +For file-backed slabs, the `simulation` feature routes open, lock, stat, and sizing to the `uring-runtime` simulated filesystem. Buffers, leases, and segment rules do not change, so the same workflow tests run in both modes. From 84cd3f7410d21614ac6eeaa73a30b60a1343379d Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 19:24:28 +0000 Subject: [PATCH 31/82] fix(flow): recheck local admission stop after callbacks --- cmd/racer-dataplane/flow/src/lib.rs | 177 +++++++++++++++++++++++++++- 1 file changed, 175 insertions(+), 2 deletions(-) diff --git a/cmd/racer-dataplane/flow/src/lib.rs b/cmd/racer-dataplane/flow/src/lib.rs index 3bdf04481..b17c7819f 100644 --- a/cmd/racer-dataplane/flow/src/lib.rs +++ b/cmd/racer-dataplane/flow/src/lib.rs @@ -644,7 +644,7 @@ impl Quotas

{ if let Some(local) = &local { local.counter(class).add(amount); } - Ok(Charge { + let charge = Charge { class, amount, key, @@ -652,7 +652,13 @@ impl Quotas

{ local, buffers: Arc::downgrade(&self.buffers), stopped: self.stopped.clone(), - }) + }; + // Policy and key callbacks may stop admission. Drop the completed owner + // to roll back both counters while preserving completion and drain work. + if self.is_stopped() && mode == AdmissionMode::Ordinary && !P::allows_stopped(class) { + return Err(Error::Unavailable); + } + Ok(charge) } /// Reuse a live key record or create one after bounded retirement cleanup. @@ -1531,6 +1537,173 @@ mod quota_tests { } } + /// Local callbacks cannot admit ordinary work after stopping the authority. + #[test] + fn local_reservation_rechecks_stop_after_callbacks() { + use std::cell::Cell; + + thread_local! { + static CLONE_HOOK: RefCell>> = RefCell::new(None); + } + + /// A transferable key whose next clone can stop the local authority. + #[derive(Eq, PartialEq, Hash)] + struct Key(u8); + + impl Clone for Key { + /// Run the one-shot callback without holding its borrow. + fn clone(&self) -> Self { + let hook = CLONE_HOOK.with(|hook| hook.borrow_mut().take()); + if let Some(hook) = hook { + hook(); + } + Self(self.0) + } + } + + /// Select which application callback requests shutdown. + #[derive(Clone, Copy, Debug, Eq, PartialEq)] + enum Hook { + Limit, + + Floor, + + Clone, + } + + /// Stop through a weak authority link without retaining an ownership cycle. + struct StopPolicy { + quotas: RefCell>>, + + hook: Cell>, + + rejections: Cell, + } + + impl StopPolicy { + /// Stop only at the selected policy callback. + fn stop_at(&self, hook: Hook) { + if self.hook.get() == Some(hook) { + self.hook.set(None); + self.quotas.borrow().upgrade().unwrap().stop(); + } + } + } + + impl Policy for StopPolicy { + type Class = Resource; + + type Key = Key; + + /// Leave enough capacity for both the existing and attempted charge. + fn limit(&self, _: Resource) -> usize { + self.stop_at(Hook::Limit); + 10 + } + + /// Exercise shutdown during keyed fair-share calculation. + fn floor(&self, _: Resource) -> usize { + self.stop_at(Hook::Floor); + 1 + } + + /// Permit one live identity and verify its eventual retirement. + fn max_keys(&self) -> usize { + 1 + } + + /// Preserve the fixture's class-specific release wake policy. + fn wakes(class: Resource) -> bool { + TestPolicy::wakes(class) + } + + /// Preserve the fixture's drain-progress exception. + fn allows_stopped(class: Resource) -> bool { + TestPolicy::allows_stopped(class) + } + + /// No page backing is allocated by this test. + fn covers(_: Resource) -> bool { + false + } + + /// Stop rejection must not be reported as quota pressure. + fn rejected(&self, _: Rejection) { + self.rejections.set(self.rejections.get() + 1); + } + } + + for hook in [Hook::Limit, Hook::Floor, Hook::Clone] { + // Unkeyed, new key, and existing key exercise distinct ownership paths. + for (keyed, live_key) in [(false, false), (true, false), (true, true)] { + if !keyed && hook != Hook::Limit { + continue; + } + for (class, completion) in [ + (Resource::Payload, false), + (Resource::Other, false), + (Resource::Progress, false), + (Resource::Payload, true), + ] { + if completion && hook == Hook::Floor { + continue; + } + let quotas = Rc::new(Quotas::new(StopPolicy { + quotas: RefCell::default(), + hook: Cell::new(None), + rejections: Cell::new(0), + })); + *quotas.policy.quotas.borrow_mut() = Rc::downgrade("as); + let key = Key(1); + let existing = quotas.reserve(live_key.then_some(&key), class, 3).unwrap(); + if hook == Hook::Clone { + let weak = Rc::downgrade("as); + CLONE_HOOK.with(|hook| { + *hook.borrow_mut() = + Some(Box::new(move || weak.upgrade().unwrap().stop())); + }); + } else { + quotas.policy.hook.set(Some(hook)); + } + let result = if completion { + quotas.reserve_completion(keyed.then_some(&key), class, 7) + } else { + quotas.reserve(keyed.then_some(&key), class, 7) + }; + assert!(quotas.is_stopped(), "callback {hook:?} must run"); + if completion || StopPolicy::allows_stopped(class) { + let charge = result.unwrap(); + assert_eq!(charge.amount(), 7); + assert_eq!(quotas.used(class), 10); + if let Some(local) = &charge.local { + assert_eq!(local.counter(class).used(), if live_key { 10 } else { 7 }); + } + drop(charge); + } else { + assert!(matches!(result, Err(Error::Unavailable)), "hook {hook:?}"); + } + assert_eq!(quotas.used(class), 3); + assert_eq!( + quotas.active_keys.load(Ordering::Acquire), + usize::from(live_key) + ); + if let Some(local) = &existing.local { + assert_eq!(local.counter(class).used(), 3); + } + assert_eq!(quotas.policy.rejections.get(), 0); + drop(existing); + assert_eq!(quotas.used(class), 0); + assert_eq!(quotas.active_keys.load(Ordering::Acquire), 0); + let replacement = quotas.reserve_completion(Some(&Key(2)), class, 10).unwrap(); + assert!(!quotas.keys.borrow().contains_key(&key)); + drop(replacement); + assert_eq!(quotas.used(class), 0); + assert_eq!(quotas.active_keys.load(Ordering::Acquire), 0); + } + } + } + } + /// Cross-thread release and local stop notify the registered shared waiter. #[test] fn release_and_stop_wake_shared_waiters() { From a2bce7f23faaa8f9cfb50a6e8908f386fc2b9d04 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 19:27:02 +0000 Subject: [PATCH 32/82] fix(flow): enforce pipe waiter limit across reentry --- cmd/racer-dataplane/flow/src/pipe.rs | 142 ++++++++++++++++++++++++++- 1 file changed, 140 insertions(+), 2 deletions(-) diff --git a/cmd/racer-dataplane/flow/src/pipe.rs b/cmd/racer-dataplane/flow/src/pipe.rs index 3623b6cec..641f0eb08 100644 --- a/cmd/racer-dataplane/flow/src/pipe.rs +++ b/cmd/racer-dataplane/flow/src/pipe.rs @@ -202,7 +202,10 @@ impl PipePool

{ let reservation = self .quotas .reserve(None, self.waiter_class, self.waiter_bytes())?; - let mut register = subscribe()?; + // Reservation policy can synchronously admit another waiter. + if self.waiting.borrow().len() >= self.waiter_limit { + return Err(Error::Overloaded.into()); + } let entry = Rc::new(RefCell::new(None)); self.waiting.borrow_mut().push_back(entry.clone()); let waiting = Waiting { @@ -210,6 +213,8 @@ impl PipePool

{ entry, _reservation: reservation, }; + // Publish before reentry; Waiting rolls back subscription errors or panics. + let mut register = subscribe()?; poll_fn(|cx| { register(cx.waker()); check()?; @@ -641,6 +646,8 @@ mod tests { pipes: usize, context: usize, + + on_context_limit: RefCell>>, } impl Policy for TestPolicy { @@ -650,6 +657,12 @@ mod tests { /// Give pipes their count ceiling and other classes a byte ceiling. fn limit(&self, class: ResourceClass) -> usize { + if matches!(class, ResourceClass::RequestContext) { + let callback = self.on_context_limit.borrow_mut().take(); + if let Some(callback) = callback { + callback(); + } + } match class { ResourceClass::Pipe => self.pipes, _ => self.context, @@ -680,6 +693,7 @@ mod tests { Rc::new(Quotas::new(TestPolicy { pipes, context: 32 * 1024 * 1024, + on_context_limit: RefCell::default(), })) } @@ -759,6 +773,7 @@ mod tests { let subscribed = Cell::new(0); let subscribe = || { subscribed.set(subscribed.get() + 1); + assert_eq!(pool.waiting.borrow().len(), 1); Err::(GateError::Rejected) }; let mut cx = Context::from_waker(Waker::noop()); @@ -788,7 +803,11 @@ mod tests { assert_eq!(pool.idle_count(), 1); for (context, waiter_limit) in [(0, 8), (usize::MAX, 0)] { - let quotas = Rc::new(Quotas::new(TestPolicy { pipes: 1, context })); + let quotas = Rc::new(Quotas::new(TestPolicy { + pipes: 1, + context, + on_context_limit: RefCell::default(), + })); let pool = PipePool::new( quotas.clone(), ResourceClass::Pipe, @@ -807,6 +826,125 @@ mod tests { } } + /// Reservation callbacks cannot let an outer wait overfill a reentered queue. + #[test] + fn review_regression_waiter_limit_after_reserve_reentry() { + let quotas = admission(1); + let pool = Rc::new(PipePool::new( + quotas.clone(), + ResourceClass::Pipe, + ResourceClass::RequestContext, + 1, + )); + let held = pool.acquire().unwrap(); + let nested_pool = pool.clone(); + let nested = Rc::new(RefCell::new(Box::pin(async move { + acquire_wait(&nested_pool).await + }))); + let callback_wait = nested.clone(); + *quotas.policy.on_context_limit.borrow_mut() = Some(Box::new(move || { + let mut cx = Context::from_waker(Waker::noop()); + assert!( + callback_wait + .borrow_mut() + .as_mut() + .poll(&mut cx) + .is_pending() + ); + })); + let mut outer = Box::pin(pool.acquire_wait( + || Ok::<_, Error>(()), + || -> Result { + panic!("overloaded outer wait must not subscribe"); + }, + )); + let mut cx = Context::from_waker(Waker::noop()); + assert!(matches!( + outer.as_mut().poll(&mut cx), + Poll::Ready(Err(Error::Overloaded)) + )); + assert_eq!(pool.waiting.borrow().len(), 1); + assert_eq!( + quotas.used(ResourceClass::RequestContext), + pool.waiter_bytes() + ); + drop(held); + assert!(matches!( + nested.borrow_mut().as_mut().poll(&mut cx), + Poll::Ready(Ok(_)) + )); + assert!(pool.waiting.borrow().is_empty()); + assert_eq!(quotas.used(ResourceClass::RequestContext), 0); + } + + /// Subscription sees its own queue slot before it tries a nested acquisition. + #[test] + fn review_regression_waiter_limit_during_subscribe_reentry() { + let quotas = admission(1); + let pool = PipePool::new( + quotas.clone(), + ResourceClass::Pipe, + ResourceClass::RequestContext, + 1, + ); + let held = pool.acquire().unwrap(); + let mut nested = acquire_wait(&pool); + let mut outer = Box::pin(pool.acquire_wait( + || Ok::<_, Error>(()), + || { + let mut cx = Context::from_waker(Waker::noop()); + assert!(matches!( + nested.as_mut().poll(&mut cx), + Poll::Ready(Err(Error::Overloaded)) + )); + assert_eq!(pool.waiting.borrow().len(), 1); + assert_eq!( + quotas.used(ResourceClass::RequestContext), + pool.waiter_bytes() + ); + Ok(|_: &Waker| {}) + }, + )); + let mut cx = Context::from_waker(Waker::noop()); + assert!(outer.as_mut().poll(&mut cx).is_pending()); + assert_eq!(pool.waiting.borrow().len(), 1); + drop(held); + assert!(matches!(outer.as_mut().poll(&mut cx), Poll::Ready(Ok(_)))); + assert!(pool.waiting.borrow().is_empty()); + assert_eq!(quotas.used(ResourceClass::RequestContext), 0); + } + + /// Unwinding a subscription removes its published slot and releases its charge. + #[test] + fn review_regression_subscription_panic_rolls_back_waiting() { + let quotas = admission(1); + let pool = new_pool(quotas.clone()); + let _held = pool.acquire().unwrap(); + let published = std::cell::Cell::new(false); + let mut wait = Box::pin(pool.acquire_wait( + || Ok::<_, Error>(()), + || -> Result { + published.set(pool.waiting.borrow().len() == 1); + panic!("subscription failed"); + }, + )); + let result = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { + wait.as_mut().poll(&mut Context::from_waker(Waker::noop())) + })); + assert!(result.is_err()); + assert!(published.get()); + assert!(pool.waiting.borrow().is_empty()); + assert_eq!(pool.waiting.borrow().capacity(), 0); + assert_eq!(quotas.used(ResourceClass::RequestContext), 0); + let mut replacement = acquire_wait(&pool); + assert!( + replacement + .as_mut() + .poll(&mut Context::from_waker(Waker::noop())) + .is_pending() + ); + } + /// Gate failure, stop, and abandonment release registration and exact charges. #[test] fn gate_stop_and_abandonment_drop_registration_and_exact_charge() { From bbdb46d4c94efb7c4d44f5cba9fa51db2b0307f2 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 19:48:50 +0000 Subject: [PATCH 33/82] fix(flow): preserve handoff admission errors --- cmd/racer-dataplane/flow/src/admission.rs | 24 ++-- cmd/racer-dataplane/flow/tests/workflows.rs | 144 +++++++++++++++++++- 2 files changed, 159 insertions(+), 9 deletions(-) diff --git a/cmd/racer-dataplane/flow/src/admission.rs b/cmd/racer-dataplane/flow/src/admission.rs index 4ac7dbc32..7587deeb4 100644 --- a/cmd/racer-dataplane/flow/src/admission.rs +++ b/cmd/racer-dataplane/flow/src/admission.rs @@ -867,6 +867,7 @@ mod handoff { /// Scan open targets fairly and reserve before returning an offer. pub fn reserve(self: &Arc, waker: &Waker) -> Result> { let mut state = self.0.lock().map_err(|_| Error::Unavailable)?; + let mut rejection = None; for offset in 0..state.targets.len() { let index = (state.cursor + offset) % state.targets.len(); let (_, target) = &state.targets[index]; @@ -877,16 +878,23 @@ mod handoff { continue; }; admission.register(waker); - if let Ok(reservation) = admission.reserve() { - state.cursor = (index + 1) % state.targets.len(); - return Ok(Offer { - handoff: self.clone(), - target: index, - reservation, - }); + match admission.reserve() { + Ok(reservation) => { + state.cursor = (index + 1) % state.targets.len(); + return Ok(Offer { + handoff: self.clone(), + target: index, + reservation, + }); + } + Err(Error::Overloaded) => rejection = Some(Error::Overloaded), + Err(Error::Unavailable) => { + rejection.get_or_insert(Error::Unavailable); + } + Err(error) => return Err(error), } } - Err(Error::Overloaded) + Err(rejection.unwrap_or(Error::Overloaded)) } /// Pop at most the caller's budget while retaining each item's admission. diff --git a/cmd/racer-dataplane/flow/tests/workflows.rs b/cmd/racer-dataplane/flow/tests/workflows.rs index 8634ee5ef..5b5322513 100644 --- a/cmd/racer-dataplane/flow/tests/workflows.rs +++ b/cmd/racer-dataplane/flow/tests/workflows.rs @@ -5,7 +5,7 @@ mod handoff_tests { use flow_control::{Error, Handoff, HandoffAdmission, Result}; use std::{ sync::{ - Arc, + Arc, Mutex, atomic::{AtomicUsize, Ordering}, }, task::Waker, @@ -40,6 +40,148 @@ mod handoff_tests { } } + struct ScriptedAdmission { + key: usize, + result: Result, + registered: AtomicUsize, + calls: Arc>>, + } + + impl HandoffAdmission for ScriptedAdmission { + type Reservation = usize; + + fn register(&self, _: &Waker) { + self.registered.fetch_add(1, Ordering::SeqCst); + } + + fn reserve(&self) -> Result { + assert_eq!(self.registered.swap(0, Ordering::SeqCst), 1); + self.calls.lock().unwrap().push(self.key); + self.result + } + } + + fn scripted_handoff( + results: &[Result], + calls: &Arc>>, + ) -> Arc> { + let keys: Vec<_> = (0..results.len()).collect(); + let handoff = Arc::new(Handoff::new(&keys)); + for (key, result) in results.iter().copied().enumerate() { + handoff + .install( + &key, + ScriptedAdmission { + key, + result, + registered: AtomicUsize::new(0), + calls: calls.clone(), + }, + ) + .unwrap(); + } + handoff + } + + #[test] + fn admission_failures_preserve_classification_and_capacity_retry() { + use Error::{Overloaded, Unavailable}; + for (errors, expected) in [ + (vec![Overloaded], Overloaded), + (vec![Unavailable], Unavailable), + (vec![Unavailable, Unavailable], Unavailable), + (vec![Overloaded, Overloaded], Overloaded), + (vec![Overloaded, Unavailable], Overloaded), + (vec![Unavailable, Overloaded], Overloaded), + ] { + let calls = Arc::new(Mutex::new(Vec::new())); + let results: Vec<_> = errors.iter().copied().map(Err).collect(); + let handoff = scripted_handoff(&results, &calls); + assert_eq!(handoff.reserve(Waker::noop()).err(), Some(expected)); + assert_eq!( + *calls.lock().unwrap(), + (0..errors.len()).collect::>() + ); + } + } + + #[test] + fn fatal_admission_errors_stop_scanning_without_becoming_overload() { + for error in [Error::InvalidInput, Error::Io] { + for preceding in [None, Some(Error::Overloaded), Some(Error::Unavailable)] { + let calls = Arc::new(Mutex::new(Vec::new())); + let mut results: Vec<_> = preceding.into_iter().map(Err).collect(); + results.extend([Err(error), Ok(99)]); + let handoff = scripted_handoff(&results, &calls); + assert_eq!(handoff.reserve(Waker::noop()).err(), Some(error)); + assert_eq!( + *calls.lock().unwrap(), + (0..results.len() - 1).collect::>() + ); + } + } + } + + #[test] + fn successful_admission_skips_retryable_failures_and_keeps_round_robin() { + for failures in [ + [Error::Overloaded, Error::Unavailable], + [Error::Unavailable, Error::Overloaded], + ] { + let calls = Arc::new(Mutex::new(Vec::new())); + let handoff = scripted_handoff( + &[Err(failures[0]), Err(failures[1]), Ok(12), Ok(13)], + &calls, + ); + for target in [2, 3, 2] { + handoff + .reserve(Waker::noop()) + .unwrap() + .deliver(|| ()) + .unwrap(); + let [item] = handoff.pop_batch::<1>(&target, Waker::noop(), 1).unwrap(); + assert_eq!(item.unwrap().into_parts(), ((), target + 10)); + } + assert_eq!(*calls.lock().unwrap(), [0, 1, 2, 3, 0, 1, 2]); + } + } + + #[test] + fn closed_and_uninstalled_targets_do_not_change_admission_errors() { + let calls = Arc::new(Mutex::new(Vec::new())); + let handoff = scripted_handoff(&[Err(Error::Io), Err(Error::Unavailable)], &calls); + handoff.close(&0); + assert_eq!( + handoff.reserve(Waker::noop()).err(), + Some(Error::Unavailable) + ); + assert_eq!(*calls.lock().unwrap(), [1]); + handoff.close(&1); + assert_eq!( + handoff.reserve(Waker::noop()).err(), + Some(Error::Overloaded) + ); + assert_eq!(*calls.lock().unwrap(), [1]); + + let handoff = Arc::new(Handoff::<_, _, ()>::new(&[0, 1])); + handoff + .install( + &1, + ScriptedAdmission { + key: 1, + result: Err(Error::Unavailable), + registered: AtomicUsize::new(0), + calls: calls.clone(), + }, + ) + .unwrap(); + assert_eq!( + handoff.reserve(Waker::noop()).err(), + Some(Error::Unavailable) + ); + assert_eq!(*calls.lock().unwrap(), [1, 1]); + } + /// Offers, envelopes, and popped items retain the selected target's slot. #[test] fn round_robin_reserves_before_delivery_and_releases_after_close() { From 8aafc7046f13de890055ca1d7a12b42295dd39ec Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 19:48:22 +0000 Subject: [PATCH 34/82] fix(flow): avoid closed descriptor probes in pipe test --- cmd/racer-dataplane/flow/src/pipe.rs | 12 +++++++++--- 1 file changed, 9 insertions(+), 3 deletions(-) diff --git a/cmd/racer-dataplane/flow/src/pipe.rs b/cmd/racer-dataplane/flow/src/pipe.rs index 641f0eb08..7a21d51c7 100644 --- a/cmd/racer-dataplane/flow/src/pipe.rs +++ b/cmd/racer-dataplane/flow/src/pipe.rs @@ -1166,9 +1166,15 @@ mod tests { drop(pipe); assert!(pool.idle.borrow().is_empty()); assert_eq!(admission.used(ResourceClass::Pipe), 0); - // SAFETY: only query the closed descriptor numbers, without reusing them. - assert_eq!(unsafe { libc::fcntl(read, libc::F_GETFD) }, -1); - assert_eq!(unsafe { libc::fcntl(write, libc::F_GETFD) }, -1); + // Dropped resources close their owned descriptors. Do not probe the old + // numbers: parallel tests may already have reused them for other files. + let mut replacement = pool.acquire().unwrap(); + assert_eq!(admission.used(ResourceClass::Pipe), 1); + assert_eq!(replacement.buffered(), 0); + assert_eq!( + replacement.try_read(&mut bytes).unwrap_err().kind(), + io::ErrorKind::WouldBlock + ); } /// A lease outliving its pool continues to hold capacity until drop. From bd03958ce137bfa93fdd599312bbe41cb5e8e7c0 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 20:03:17 +0000 Subject: [PATCH 35/82] fix(flow): preserve pressure eligibility at recovery ceiling --- cmd/racer-dataplane/flow/src/admission.rs | 71 +++++++++++++++++++++++ 1 file changed, 71 insertions(+) diff --git a/cmd/racer-dataplane/flow/src/admission.rs b/cmd/racer-dataplane/flow/src/admission.rs index 7587deeb4..04ac31e14 100644 --- a/cmd/racer-dataplane/flow/src/admission.rs +++ b/cmd/racer-dataplane/flow/src/admission.rs @@ -265,6 +265,7 @@ impl Permit { return; } if outcome == Outcome::Verified + && state.limit < config.total && now.saturating_duration_since(state.updated) >= config.recovery { state.limit = state.limit.saturating_add(1).min(config.total); @@ -1345,6 +1346,76 @@ mod tests { assert_eq!(*owner.observer.limit.lock().unwrap(), 3); } + /// Success at the ceiling must not delay an eligible pressure reduction. + #[test] + fn recovery_at_ceiling_preserves_pressure_eligibility() { + fn now() -> Instant { + static NOW: OnceLock = OnceLock::new(); + *NOW.get_or_init(Instant::now) + } + + for recovery in [Duration::ZERO, config().recovery] { + let config = Config { + recovery, + ..config() + }; + let owner = Adaptive::new(config, Counts::default(), now).unwrap(); + let work = owner.acquire(&1).unwrap(); + let updated = now() - config.backoff.max(config.recovery); + owner.state.lock().unwrap().updated = updated; + + for _ in 0..2 { + work.observe(Outcome::Verified); + let state = owner.state.lock().unwrap(); + assert_eq!(state.limit, config.total); + assert_eq!(state.updated, updated); + } + work.observe(Outcome::LocalPressure); + assert_eq!(*owner.observer.limit.lock().unwrap(), config.total / 2); + assert_eq!(owner.state.lock().unwrap().updated, now()); + assert_eq!(owner.state.lock().unwrap().peers[&1].limit, config.per_key); + assert!(owner.available(&1)); + assert_eq!( + *owner.observer.events.lock().unwrap(), + [ + Event::Accepted, + Event::Verified, + Event::Verified, + Event::LocalPressure, + ] + ); + } + } + + /// Restoring the final slot still starts backoff and rate-limits pressure. + #[test] + fn recovery_restoring_slot_advances_pressure_backoff() { + fn now() -> Instant { + static NOW: OnceLock = OnceLock::new(); + *NOW.get_or_init(Instant::now) + } + + let config = config(); + let owner = Adaptive::new(config, Counts::default(), now).unwrap(); + let work = owner.acquire(&1).unwrap(); + { + let mut state = owner.state.lock().unwrap(); + state.limit = config.total - 1; + state.updated = now() - config.recovery; + } + work.observe(Outcome::Verified); + assert_eq!(*owner.observer.limit.lock().unwrap(), config.total); + assert_eq!(owner.state.lock().unwrap().updated, now()); + work.observe(Outcome::LocalPressure); + assert_eq!(*owner.observer.limit.lock().unwrap(), config.total); + + owner.state.lock().unwrap().updated = now() - config.backoff; + work.observe(Outcome::LocalPressure); + assert_eq!(*owner.observer.limit.lock().unwrap(), config.total / 2); + work.observe(Outcome::LocalPressure); + assert_eq!(*owner.observer.limit.lock().unwrap(), config.total / 2); + } + /// Only sufficiently old, idle, eligible peer records may be retired. #[test] fn capacity_preserves_live_work_and_stale_backoff_then_retires_idle() { From ebc4b84de0eb6418bf3a3d2485f7192296522aeb Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 20:02:39 +0000 Subject: [PATCH 36/82] fix(flow): clear completed hedge delay wake registration --- cmd/racer-dataplane/flow/src/admission.rs | 31 +++++++++++++++++++++++ 1 file changed, 31 insertions(+) diff --git a/cmd/racer-dataplane/flow/src/admission.rs b/cmd/racer-dataplane/flow/src/admission.rs index 04ac31e14..c0c3afbee 100644 --- a/cmd/racer-dataplane/flow/src/admission.rs +++ b/cmd/racer-dataplane/flow/src/admission.rs @@ -1063,6 +1063,7 @@ mod hedge { let mut state = self.owner.state.lock().expect("hedge alarm lock"); let alarm = state.alarms.get_mut(&self.id).expect("live hedge alarm"); if now >= alarm.due { + alarm.wake = None; Poll::Ready(()) } else { alarm.wake = Some(cx.waker().clone()); @@ -1132,6 +1133,36 @@ mod hedge { assert!(matches!(max.acquire(1, now), Err(Error::Overloaded))); } + /// Direct completion clears the waiter but retains speculative capacity. + #[test] + fn hedge_ready_delay_clears_registration_without_releasing_capacity() { + for elapsed in [Duration::ZERO, Duration::from_secs(1)] { + let now = Instant::now(); + let due = now + Duration::from_secs(1); + let owner = Hedges::new(1, 1); + let permit = owner.acquire(1, due).unwrap(); + let waiter = Arc::new(Counter::default()); + assert!( + permit + .delay(now, &mut Context::from_waker(&Waker::from(waiter.clone()))) + .is_pending() + ); + assert!( + permit + .delay(due + elapsed, &mut Context::from_waker(Waker::noop())) + .is_ready() + ); + assert_eq!(Arc::strong_count(&waiter), 1); + owner.poll(due + elapsed); + owner.poll(due + elapsed); + assert_eq!(waiter.0.load(Ordering::SeqCst), 0); + assert_eq!(Arc::strong_count(&waiter), 1); + assert!(matches!(owner.acquire(1, due), Err(Error::Overloaded))); + drop(permit); + assert!(owner.acquire(1, due).is_ok()); + } + } + /// Only the latest registered waker fires and drop cancels notification. #[test] fn hedge_alarm_replaces_waker_and_drop_removes_registration() { From 8147019fb09d121886503a656a4d3439b3ec4ce8 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 20:18:22 +0000 Subject: [PATCH 37/82] fix(flow): handle admission duration overflow --- cmd/racer-dataplane/flow/src/admission.rs | 196 ++++++++++++++++++++-- 1 file changed, 181 insertions(+), 15 deletions(-) diff --git a/cmd/racer-dataplane/flow/src/admission.rs b/cmd/racer-dataplane/flow/src/admission.rs index c0c3afbee..c539b5709 100644 --- a/cmd/racer-dataplane/flow/src/admission.rs +++ b/cmd/racer-dataplane/flow/src/admission.rs @@ -16,6 +16,26 @@ pub use circuit::{Circuits, Probe}; pub use handoff::{Admission as HandoffAdmission, Admitted, Handoff, Offer}; pub use hedge::{Hedges, Permit as HedgePermit}; +/// A retry or timeout beyond the clock's range never expires. +#[derive(Clone, Copy)] +enum Deadline { + At(Instant), + + Never, +} + +impl Deadline { + /// Keep an unrepresentable deadline distinct from an absent delay. + fn after(now: Instant, delay: Duration) -> Self { + now.checked_add(delay).map_or(Self::Never, Self::At) + } + + /// Only finite deadlines can become eligible. + fn elapsed(self, now: Instant) -> bool { + matches!(self, Self::At(at) if now >= at) + } +} + /// Limits and time intervals for shared adaptive admission. #[derive(Clone, Copy)] pub struct Config { @@ -29,6 +49,7 @@ pub struct Config { pub capacity: usize, /// Peer retry delay and minimum interval between aggregate pressure reductions. + /// A retry beyond the clock's range keeps the peer circuit open indefinitely. pub backoff: Duration, /// Minimum interval between one-slot verified-success recoveries. @@ -121,7 +142,7 @@ struct Peer { generation: u64, - retry: Option, + retry: Option, probe: bool, @@ -183,7 +204,7 @@ impl Adaptive { self.state.lock().is_ok_and(|s| { s.peers .get(key) - .is_none_or(|p| !p.probe && p.retry.is_none_or(|at| now >= at)) + .is_none_or(|p| !p.probe && p.retry.is_none_or(|at| at.elapsed(now))) }) } @@ -203,7 +224,7 @@ impl Adaptive { .extract_if(.., |_, p| { p.active == 0 && !p.probe - && p.retry.is_none_or(|retry| now >= retry) + && p.retry.is_none_or(|retry| retry.elapsed(now)) && now.saturating_duration_since(p.updated) >= self.config.retire_after }) .next(); @@ -220,7 +241,7 @@ impl Adaptive { probe: false, updated: now, }); - if peer.probe || peer.retry.is_some_and(|at| now < at) { + if peer.probe || peer.retry.is_some_and(|at| !at.elapsed(now)) { self.observer.event(Event::CircuitRejected); return Err(Error::Unavailable); } @@ -290,7 +311,7 @@ impl Permit { Outcome::PeerFailure => { peer.limit = (peer.limit / 2).max(1); peer.generation = peer.generation.saturating_add(1); - peer.retry = Some(now + config.backoff); + peer.retry = Some(Deadline::after(now, config.backoff)); peer.updated = now; } Outcome::Verified => { @@ -324,7 +345,7 @@ impl Drop for Permit { peer.probe = false; if peer.retry.is_some() { let now = now.expect("probe clock sampled before locking"); - peer.retry = Some(now + self.owner.config.backoff); + peer.retry = Some(Deadline::after(now, self.owner.config.backoff)); peer.updated = now; } } @@ -335,6 +356,7 @@ impl Drop for Permit { /// Bounded worker-local endpoint failure tracking, independent of adaptive limits. mod circuit { + use super::Deadline; use crate::{Error, Result}; use std::{ cell::RefCell, @@ -375,9 +397,9 @@ mod circuit { struct Circuit { failures: u32, - retry_at: Instant, + retry_at: Deadline, - probe_until: Option, + probe_until: Option, pending_backoff: Option>, } @@ -385,7 +407,7 @@ mod circuit { impl Circuit { /// Require both retry backoff and any abandoned probe timeout to expire. fn available(&self, now: Instant) -> bool { - now >= self.retry_at && self.probe_until.is_none_or(|until| now >= until) + self.retry_at.elapsed(now) && self.probe_until.is_none_or(|until| until.elapsed(now)) } } @@ -407,6 +429,7 @@ mod circuit { impl Circuits { /// Bound failure records and set eligibility delay for abandoned probes. + /// A timeout beyond the clock's range never expires. pub const fn new(capacity: usize, probe_timeout: Duration) -> Self { Self { capacity, @@ -446,6 +469,7 @@ mod circuit { /// Record a caller-classified failure using caller-selected retry jitter. /// Backoff may reenter: the count is visible before the callback, and any /// newer transition for this key takes precedence over its returned delay. + /// A retry beyond the clock's range never expires on its own. pub fn failure( &self, key: &K, @@ -458,7 +482,7 @@ mod circuit { } let state = states.entry(key.clone()).or_insert(Circuit { failures: 0, - retry_at: now, + retry_at: Deadline::At(now), probe_until: None, pending_backoff: None, }); @@ -468,7 +492,7 @@ mod circuit { state.pending_backoff = Some(Rc::clone(&pending)); drop(states); - let retry_at = now + backoff(key, failures); + let retry_at = Deadline::after(now, backoff(key, failures)); let mut states = self.states.borrow_mut(); // Never reinsert a removed record or overwrite a newer transition. if let Some(state) = states.get_mut(key) @@ -506,7 +530,7 @@ mod circuit { if !state.available(now) { return false; } - state.probe_until = Some(now + self.probe_timeout); + state.probe_until = Some(Deadline::after(now, self.probe_timeout)); state.pending_backoff = None; true } @@ -532,6 +556,62 @@ mod circuit { mod tests { use super::*; + /// Oversized retry delays stay closed to probes until explicit success. + #[test] + fn duration_overflow_endpoint_backoff() { + let now = Instant::now(); + assert!(now.checked_add(Duration::MAX).is_none()); + for delay in [ + Duration::ZERO, + Duration::from_secs(1), + Duration::from_secs(u64::from(u32::MAX)), + Duration::MAX, + ] { + let health = Circuits::new(1, Duration::ZERO); + health.failure(&7, now, |_, _| delay).unwrap(); + assert_eq!(health.available(&7, now), delay.is_zero()); + if let Some(due) = now.checked_add(delay) { + assert!(health.acquire(&7, due).is_ok()); + } else { + let later = now + Duration::from_secs(60); + assert!(!health.available(&7, later)); + assert!(matches!(health.acquire(&7, later), Err(Error::Unavailable))); + assert_eq!( + health.failure(&8, later, |_, _| Duration::ZERO), + Err(Error::Overloaded) + ); + } + health.success(&7); + assert!(health.acquire(&7, now).is_ok()); + } + } + + /// Oversized abandoned-probe timeouts never expire or lose ownership. + #[test] + fn duration_overflow_endpoint_probe_timeout() { + let now = Instant::now(); + for delay in [ + Duration::ZERO, + Duration::from_secs(1), + Duration::from_secs(u64::from(u32::MAX)), + Duration::MAX, + ] { + let health = Circuits::new(1, delay); + health.failure(&7, now, |_, _| Duration::ZERO).unwrap(); + let probe = health.acquire(&7, now).unwrap(); + assert!(!health.available(&7, now)); + drop(probe); + assert_eq!(health.available(&7, now), delay.is_zero()); + if let Some(due) = now.checked_add(delay) { + assert!(health.try_acquire(&7, due)); + } else { + assert!(!health.try_acquire(&7, now + Duration::from_secs(60))); + } + health.success(&7); + assert!(health.acquire(&7, now).is_ok()); + } + } + /// Backoff can inspect the circuit without borrowing conflicts. #[test] fn backoff_reentry_observes_failure_record() { @@ -1265,6 +1345,89 @@ mod tests { } } + /// Oversized peer backoff must not panic, poison state, or retire early. + #[test] + fn duration_overflow_adaptive_failure() { + let owner = Adaptive::new( + Config { + capacity: 1, + backoff: Duration::MAX, + retire_after: Duration::ZERO, + ..config() + }, + Counts::default(), + Instant::now, + ) + .unwrap(); + let stale = owner.acquire(&1).unwrap(); + let failed = owner.acquire(&1).unwrap(); + failed.observe(Outcome::PeerFailure); + stale.observe(Outcome::Verified); + assert!(!owner.available(&1)); + assert!(!owner.hedge_available(&1)); + drop((stale, failed)); + assert_eq!(owner.state.lock().unwrap().active, 0); + assert_eq!(*owner.observer.active.lock().unwrap(), 0); + assert!(matches!(owner.acquire(&1), Err(Error::Unavailable))); + assert!(matches!(owner.acquire(&2), Err(Error::Overloaded))); + } + + /// Renewing a probe can overflow even when its previous retry was finite. + #[test] + fn duration_overflow_adaptive_probe_drop() { + let owner = Adaptive::new( + Config { + backoff: Duration::MAX, + ..config() + }, + Counts::default(), + Instant::now, + ) + .unwrap(); + drop(owner.acquire(&1).unwrap()); + owner.state.lock().unwrap().peers.get_mut(&1).unwrap().retry = + Some(Deadline::At(Instant::now())); + let probe = owner.acquire(&1).unwrap(); + drop(probe); + assert_eq!(owner.state.lock().unwrap().active, 0); + assert_eq!(*owner.observer.active.lock().unwrap(), 0); + assert!(!owner.available(&1)); + assert!(matches!(owner.acquire(&1), Err(Error::Unavailable))); + assert!(owner.acquire(&2).is_ok()); + } + + /// Elapsed-time limits accept huge durations without forming deadlines. + #[test] + fn duration_overflow_elapsed_limits_remain_valid() { + let owner = Adaptive::new( + Config { + capacity: 1, + backoff: Duration::ZERO, + recovery: Duration::MAX, + retire_after: Duration::MAX, + ..config() + }, + Counts::default(), + Instant::now, + ) + .unwrap(); + let work = owner.acquire(&1).unwrap(); + work.observe(Outcome::LocalPressure); + work.observe(Outcome::PeerFailure); + drop(work); + let probe = owner.acquire(&1).unwrap(); + probe.observe(Outcome::Verified); + drop(probe); + assert!(owner.available(&1)); + assert_eq!(owner.state.lock().unwrap().limit, config().total / 2); + assert_eq!( + owner.state.lock().unwrap().peers[&1].limit, + config().per_key / 2 + ); + assert!(matches!(owner.acquire(&2), Err(Error::Overloaded))); + assert!(owner.acquire(&1).is_ok()); + } + /// Clone panics leave the mutex usable and all admission capacity recoverable. #[test] fn adaptive_clone_panic_preserves_capacity_and_mutex() { @@ -1360,7 +1523,8 @@ mod tests { drop((old, failed)); assert_eq!(*owner.observer.active.lock().unwrap(), 1); drop(fence); - owner.state.lock().unwrap().peers.get_mut(&1).unwrap().retry = Some(Instant::now()); + owner.state.lock().unwrap().peers.get_mut(&1).unwrap().retry = + Some(Deadline::At(Instant::now())); let probe = owner.acquire(&1).unwrap(); assert!(matches!(owner.acquire(&1), Err(Error::Unavailable))); probe.observe(Outcome::Verified); @@ -1466,7 +1630,8 @@ mod tests { .updated = Instant::now() - Duration::from_secs(61); assert!(matches!(owner.acquire(&3), Err(Error::Overloaded))); assert!(owner.state.lock().unwrap().peers.contains_key(&1)); - owner.state.lock().unwrap().peers.get_mut(&1).unwrap().retry = Some(Instant::now()); + owner.state.lock().unwrap().peers.get_mut(&1).unwrap().retry = + Some(Deadline::At(Instant::now())); let probe = owner.acquire(&1).unwrap(); drop(probe); assert!(!owner.available(&1)); @@ -1550,7 +1715,8 @@ mod tests { let failed = owner.acquire(&1).unwrap(); failed.observe(Outcome::PeerFailure); drop(failed); - owner.state.lock().unwrap().peers.get_mut(&1).unwrap().retry = Some(Instant::now()); + owner.state.lock().unwrap().peers.get_mut(&1).unwrap().retry = + Some(Deadline::At(Instant::now())); let probe = owner.acquire(&1).unwrap(); drop(probe); } From e911b1b1b4f13e8467689292d2fc689e1f3d8596 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 20:41:28 +0000 Subject: [PATCH 38/82] fix(alloc): recheck write fence after unregister callbacks --- cmd/racer-dataplane/alloc/src/slab.rs | 102 ++++++++++++++++++++------ 1 file changed, 81 insertions(+), 21 deletions(-) diff --git a/cmd/racer-dataplane/alloc/src/slab.rs b/cmd/racer-dataplane/alloc/src/slab.rs index cfc2d6999..5b59761ff 100644 --- a/cmd/racer-dataplane/alloc/src/slab.rs +++ b/cmd/racer-dataplane/alloc/src/slab.rs @@ -723,32 +723,45 @@ impl Future for FenceWaiter { /// Refresh the task's registration without busy-waking the worker. fn poll(mut self: Pin<&mut Self>, cx: &mut Context<'_>) -> Poll { - if self.state.count.get() == 0 { - self.unregister(); - return Poll::Ready(Ok(())); - } - let waker = Rc::new(cx.waker().clone()); - if self.state.count.get() == 0 { - self.unregister(); - return Poll::Ready(Ok(())); - } - if let Some(registration) = self.registration.clone() { - let old = registration.replace(waker); - drop(old); + loop { if self.state.count.get() == 0 { self.unregister(); - return Poll::Ready(Ok(())); + if self.state.count.get() == 0 { + return Poll::Ready(Ok(())); + } + continue; + } + let waker = Rc::new(cx.waker().clone()); + if self.state.count.get() == 0 { + drop(waker); + self.unregister(); + if self.state.count.get() == 0 { + return Poll::Ready(Ok(())); + } + continue; } - let mut waiters = self.state.waiters.borrow_mut(); - if !waiters.iter().any(|w| Rc::ptr_eq(w, ®istration)) { - waiters.push(registration); + if let Some(registration) = self.registration.clone() { + let old = registration.replace(waker); + drop(old); + if self.state.count.get() == 0 { + drop(registration); + self.unregister(); + if self.state.count.get() == 0 { + return Poll::Ready(Ok(())); + } + continue; + } + let mut waiters = self.state.waiters.borrow_mut(); + if !waiters.iter().any(|w| Rc::ptr_eq(w, ®istration)) { + waiters.push(registration); + } + } else { + let registration = Rc::new(RefCell::new(waker)); + self.state.waiters.borrow_mut().push(registration.clone()); + self.registration = Some(registration); } - } else { - let registration = Rc::new(RefCell::new(waker)); - self.state.waiters.borrow_mut().push(registration.clone()); - self.registration = Some(registration); + return Poll::Pending; } - Poll::Pending } } @@ -1625,6 +1638,53 @@ mod tests { } } + /// Cleanup callbacks can accept a write at every zero-count exit. + #[test] + fn fence_poll_rechecks_unregister_callbacks() { + for completion in [None, Some(FenceCallback::Clone), Some(FenceCallback::Drop)] { + let slab = Slab::<()>::new(PathBuf::new(), 4096, 4096, 512); + let write = slab.writes.acquire().unwrap(); + let mut waiter = slab.fence_writes(); + let waker = fence_callback_waker(); + let mut cx = Context::from_waker(&waker); + assert!(waiter.as_mut().poll(&mut cx).is_pending()); + let next = Rc::new(RefCell::new(None)); + let saved_next = next.clone(); + let state = slab.writes.clone(); + let accept: Box = Box::new(move || { + *saved_next.borrow_mut() = Some(state.acquire().unwrap()); + }); + if let Some(event) = completion { + FENCE_CALLBACK.with(|slot| { + *slot.borrow_mut() = Some(( + event, + Box::new(move || { + drop(write); + FENCE_CALLBACK.with(|slot| { + *slot.borrow_mut() = Some((FenceCallback::Drop, accept)); + }); + }), + )); + }); + } else { + drop(write); + FENCE_CALLBACK.with(|slot| { + *slot.borrow_mut() = Some((FenceCallback::Drop, accept)); + }); + } + assert!(waiter.as_mut().poll(&mut cx).is_pending()); + FENCE_CALLBACK.with(|slot| assert!(slot.borrow().is_none())); + assert_eq!(slab.writes_in_flight(), 1); + assert_eq!(slab.writes.waiters.borrow().len(), 1); + let before = FENCE_WAKES.get(); + drop(next.borrow_mut().take()); + assert_eq!(FENCE_WAKES.get(), before + 1); + assert_eq!(waiter.as_mut().poll(&mut cx), Poll::Ready(Ok(()))); + assert_eq!(slab.writes_in_flight(), 0); + assert!(slab.writes.waiters.borrow().is_empty()); + } + } + /// Completion callbacks can accept a write and repoll a drained registration. #[test] fn fence_completion_allows_reentrant_registration() { From a302c51eb7722640f2a52880a77c142cda9fec34 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 20:52:53 +0000 Subject: [PATCH 39/82] fix(alloc): drain pending segment reclamation victims --- cmd/racer-dataplane/alloc/src/segments.rs | 82 ++++++++++++++++++++++- 1 file changed, 79 insertions(+), 3 deletions(-) diff --git a/cmd/racer-dataplane/alloc/src/segments.rs b/cmd/racer-dataplane/alloc/src/segments.rs index 8079a5a7b..0ff24fa56 100644 --- a/cmd/racer-dataplane/alloc/src/segments.rs +++ b/cmd/racer-dataplane/alloc/src/segments.rs @@ -801,7 +801,7 @@ impl SegmentClock { let mut free = self.segments.free_count(); let mut entries_left = max_entries; for _ in 0..count.saturating_mul(2).min(max_visits) { - if free >= target { + if free >= target && self.segments.evicting.get() == 0 { return Ok(()); } let id = self.next(count); @@ -809,7 +809,10 @@ impl SegmentClock { if !matches!(state, SegmentState::Sealed | SegmentState::Evicting) { continue; } - if state == SegmentState::Sealed && !entries.can_evict(id) { + if state == SegmentState::Sealed + && (free.saturating_add(self.segments.evicting.get()) >= target + || !entries.can_evict(id)) + { continue; } if self.recent.borrow_mut().remove(&id) { @@ -835,7 +838,7 @@ impl SegmentClock { Err(e) => return Err(e), } } - if free >= target { + if free >= target && self.segments.evicting.get() == 0 { Ok(()) } else { Err(Error::Busy) @@ -1377,6 +1380,79 @@ mod clock_tests { } } + /// Pending victims satisfy demand without evicting more leased segments. + #[test] + fn reclaim_pending_victims_count_toward_reserve() { + let segments = segments(3); + let mut leases: Vec<_> = (0..3) + .map(|_| Some(segments.append(1024).unwrap().0)) + .collect(); + let clock = SegmentClock::new(segments.clone()); + let entries = Entries::new(vec![1; 3]); + for _ in 0..6 { + assert_eq!(clock.reclaim(&entries, 1, 1, 256), Err(Error::Busy)); + assert_eq!(*entries.counts.borrow(), [0, 1, 1]); + assert_eq!(segments.evicting.get(), 1); + assert_eq!(segments.free_count(), 0); + assert_eq!(segments.state(SegmentId(0)), Ok(SegmentState::Evicting)); + assert!(segments.lease(SegmentId(1), Generation(1)).is_ok()); + assert!(segments.lease(SegmentId(2), Generation(1)).is_ok()); + } + drop(leases[0].take()); + clock.reclaim(&entries, 1, 6, 0).unwrap(); + assert_eq!(segments.free_count(), 1); + assert_eq!(segments.evicting.get(), 0); + assert_eq!(*entries.counts.borrow(), [0, 1, 1]); + assert_eq!(segments.snapshot()[0].generation, Generation(2)); + } + + /// Lowering the target still drains existing victims within each visit budget. + #[test] + fn reclaim_pending_victims_drain_after_reserve_is_met() { + for max_visits in [1, 8] { + let segments = segments(4); + let mut leases: Vec<_> = (0..3) + .map(|_| Some(segments.append(1024).unwrap().0)) + .collect(); + drop(segments.append(1024).unwrap()); + let clock = SegmentClock::new(segments.clone()); + let entries = Entries::new(vec![1; 4]); + assert_eq!(clock.reclaim(&entries, 3, 3, 3), Err(Error::Busy)); + assert_eq!(segments.evicting.get(), 3); + assert_eq!(*entries.counts.borrow(), [0, 0, 0, 1]); + + // Wrap the cursor without starting a fourth victim. + entries.evictable.set(false); + assert_eq!(clock.reclaim(&entries, 1, 1, 1), Err(Error::Busy)); + drop(leases[0].take()); + assert_eq!(clock.reclaim(&entries, 1, max_visits, 0), Err(Error::Busy)); + assert_eq!(segments.free_count(), 1); + assert_eq!(segments.evicting.get(), 2); + assert!(matches!( + segments.lease(SegmentId(1), Generation(1)), + Err(Error::Stale) + )); + + drop(leases); + let before = segments.snapshot(); + assert_eq!(clock.reclaim(&entries, 0, 8, 8), Ok(())); + assert_eq!(clock.reclaim(&entries, 1, 0, 0), Err(Error::Busy)); + assert_eq!(segments.snapshot(), before); + for _ in 0..4 { + let _ = clock.reclaim(&entries, 1, max_visits, 0); + } + assert_eq!(clock.reclaim(&entries, 1, 0, 0), Ok(())); + assert_eq!(segments.free_count(), 3); + assert_eq!(segments.evicting.get(), 0); + assert_eq!(segments.state(SegmentId(3)), Ok(SegmentState::Sealed)); + assert_eq!(*entries.counts.borrow(), [0, 0, 0, 1]); + for image in &segments.snapshot()[..3] { + assert_eq!(image.state, SegmentState::Free); + assert_eq!(image.generation, Generation(2)); + } + } + } + /// Freeze checks precede index side effects, and leases precede physical reuse. #[test] fn busy_lease_and_frozen_table_preserve_reclaim_side_effect_order() { From 0f6474f68ce5dfde4ed7f0297b67075f7dc25fe1 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 21:02:49 +0000 Subject: [PATCH 40/82] fix(alloc): wait for scored reclamation to drain --- cmd/racer-dataplane/alloc/src/segments.rs | 100 +++++++++++++++++++++- 1 file changed, 99 insertions(+), 1 deletion(-) diff --git a/cmd/racer-dataplane/alloc/src/segments.rs b/cmd/racer-dataplane/alloc/src/segments.rs index 0ff24fa56..7f30b7f4c 100644 --- a/cmd/racer-dataplane/alloc/src/segments.rs +++ b/cmd/racer-dataplane/alloc/src/segments.rs @@ -772,7 +772,7 @@ impl SegmentClock { Err(error) => return Err(error), } } - if self.segments.free_count() >= target { + if self.segments.free_count() >= target && self.segments.evicting.get() == 0 { Ok(()) } else { Err(Error::Busy) @@ -1650,6 +1650,104 @@ mod clock_tests { assert_eq!(*entries.counts.borrow(), [0, 1, 1]); } + /// Lowering scored demand still waits for pending leases and bounded visits. + #[test] + fn scored_pending_victims_drain_after_reserve_is_met() { + for max_visits in [1, 8] { + let segments = segments(4); + let mut leases: Vec<_> = (0..3) + .map(|_| Some(segments.append(1024).unwrap().0)) + .collect(); + drop(segments.append(1024).unwrap()); + let clock = SegmentClock::new(segments.clone()); + let entries = Entries::new(vec![1; 4]); + assert_eq!( + clock.reclaim_scored(&entries, 3, 3, 3, |_| 0), + Err(Error::Busy) + ); + assert_eq!(segments.evicting.get(), 3); + assert_eq!(*entries.counts.borrow(), [0, 0, 0, 1]); + + assert_eq!( + clock.reclaim_scored(&entries, 1, 1, 1, |_| 0), + Err(Error::Busy) + ); + drop(leases[0].take()); + assert_eq!( + clock.reclaim_scored(&entries, 1, max_visits, 0, |_| 0), + Err(Error::Busy) + ); + assert_eq!(segments.free_count(), 1); + assert_eq!(segments.evicting.get(), 2); + assert!(matches!( + segments.lease(SegmentId(1), Generation(1)), + Err(Error::Stale) + )); + + drop(leases); + let before = segments.snapshot(); + assert_eq!(clock.reclaim_scored(&entries, 0, 8, 8, |_| 0), Ok(())); + assert_eq!( + clock.reclaim_scored(&entries, 1, 0, 0, |_| 0), + Err(Error::Busy) + ); + assert_eq!(segments.snapshot(), before); + for _ in 0..4 { + let _ = clock.reclaim_scored(&entries, 1, max_visits, 0, |_| 0); + } + assert_eq!(clock.reclaim_scored(&entries, 1, 0, 0, |_| 0), Ok(())); + assert_eq!(segments.free_count(), 3); + assert_eq!(segments.evicting.get(), 0); + assert_eq!(segments.state(SegmentId(3)), Ok(SegmentState::Sealed)); + assert_eq!(*entries.counts.borrow(), [0, 0, 0, 1]); + assert_eq!(entries.calls.borrow().len(), 3); + for image in &segments.snapshot()[..3] { + assert_eq!(image.state, SegmentState::Free); + assert_eq!(image.generation, Generation(2)); + } + } + } + + /// Meeting a lower reserve does not hide unfinished index removal. + #[test] + fn scored_pending_mappings_drain_after_reserve_is_met() { + let segments = segments(3); + for _ in 0..3 { + drop(segments.append(1024).unwrap()); + } + let clock = SegmentClock::new(segments.clone()); + let entries = Entries::new(vec![0, 3, 1]); + assert_eq!( + clock.reclaim_scored(&entries, 2, 3, 1, |_| 0), + Err(Error::Busy) + ); + assert_eq!(segments.free_count(), 1); + assert_eq!(segments.evicting.get(), 1); + assert_eq!(*entries.counts.borrow(), [0, 2, 1]); + + let before = segments.snapshot(); + assert_eq!( + clock.reclaim_scored(&entries, 1, 3, 0, |_| 0), + Err(Error::Busy) + ); + assert_eq!(segments.snapshot(), before); + assert_eq!(*entries.counts.borrow(), [0, 2, 1]); + assert_eq!(entries.calls.borrow().len(), 1); + assert_eq!( + clock.reclaim_scored(&entries, 1, 3, 1, |_| 0), + Err(Error::Busy) + ); + assert_eq!(*entries.counts.borrow(), [0, 1, 1]); + assert_eq!(segments.state(SegmentId(1)), Ok(SegmentState::Evicting)); + assert_eq!(clock.reclaim_scored(&entries, 1, 3, 1, |_| 0), Ok(())); + assert_eq!(segments.free_count(), 2); + assert_eq!(segments.evicting.get(), 0); + assert_eq!(segments.snapshot()[1].generation, Generation(2)); + assert_eq!(segments.state(SegmentId(2)), Ok(SegmentState::Sealed)); + assert_eq!(*entries.counts.borrow(), [0, 0, 1]); + assert_eq!(*entries.calls.borrow(), [(SegmentId(1), 1); 3]); + } + /// Over-reporting callbacks return configuration errors without unsafe reuse. #[test] fn removal_contract_violations_return_errors_without_panicking() { From 1349aaaa0bb5b0fbd8b398c02094a30645d0f794 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 21:02:49 +0000 Subject: [PATCH 41/82] test(alloc): use American English cancellation spelling --- cmd/racer-dataplane/alloc/src/slab.rs | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/cmd/racer-dataplane/alloc/src/slab.rs b/cmd/racer-dataplane/alloc/src/slab.rs index 5b59761ff..82a00af9a 100644 --- a/cmd/racer-dataplane/alloc/src/slab.rs +++ b/cmd/racer-dataplane/alloc/src/slab.rs @@ -1762,7 +1762,7 @@ mod tests { let old = Arc::new(WakeCount::default()); let current = Arc::new(WakeCount::default()); let other = Arc::new(WakeCount::default()); - let cancelled = Arc::new(WakeCount::default()); + let canceled = Arc::new(WakeCount::default()); let poll = |op: &mut Operation<'_, (), Error>, wakes: &Arc| { op.as_mut() .poll(&mut Context::from_waker(&Waker::from(wakes.clone()))) @@ -1773,18 +1773,18 @@ mod tests { assert!(poll(&mut one, &old).is_pending()); assert!(poll(&mut one, ¤t).is_pending()); assert!(poll(&mut two, &other).is_pending()); - assert!(poll(&mut abandoned, &cancelled).is_pending()); + assert!(poll(&mut abandoned, &canceled).is_pending()); assert_eq!(slab.writes.waiters.borrow().len(), 3); drop(abandoned); assert_eq!(slab.writes.waiters.borrow().len(), 2); - assert_eq!(Arc::strong_count(&cancelled), 1); + assert_eq!(Arc::strong_count(&canceled), 1); drop(first_write); - for count in [&old, ¤t, &other, &cancelled] { + for count in [&old, ¤t, &other, &canceled] { assert_eq!(count.0.load(Ordering::Relaxed), 0); } drop(last_write); assert_eq!(old.0.load(Ordering::Relaxed), 0); - assert_eq!(cancelled.0.load(Ordering::Relaxed), 0); + assert_eq!(canceled.0.load(Ordering::Relaxed), 0); assert_eq!(current.0.load(Ordering::Relaxed), 1); assert_eq!(other.0.load(Ordering::Relaxed), 1); assert!(slab.writes.waiters.borrow().is_empty()); From 9edad301bb39161cb3de78434098216d0670134d Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 21:03:09 +0000 Subject: [PATCH 42/82] docs(alloc): clarify worker-local allocation authority --- designs/racer-page-alloc.md | 16 +++++++++++----- 1 file changed, 11 insertions(+), 5 deletions(-) diff --git a/designs/racer-page-alloc.md b/designs/racer-page-alloc.md index 4b9305b3e..a4874b35f 100644 --- a/designs/racer-page-alloc.md +++ b/designs/racer-page-alloc.md @@ -33,11 +33,17 @@ Non-goals: ## Threading model -All state is worker-local. Types use `Rc`, `Cell`, and `RefCell`, so none of them -are `Send` or `Sync` (`alloc/src/segments.rs:153-179`, `alloc/src/slab.rs:35-50`). -Each worker owns its storage ranges, segment table, and buffer pool. This removes lock -contention and makes ownership easy to reason about. The cost is that one worker -cannot hand its storage to another. +Live allocation and I/O authority is worker-local. `Segments` and `Slab` use +`Rc`-owned state and are neither `Send` nor `Sync` +(`alloc/src/segments.rs:160-179`, `alloc/src/slab.rs:57-72`). Buffers, leases, +freeze guards, and the reclamation clock have the same restriction +(`alloc/src/lib.rs:545-550`, `alloc/src/segments.rs:62-76`, +`alloc/src/segments.rs:144`, `alloc/src/segments.rs:617-626`). +Value types such as `Alignment` and `SegmentId`, and startup inputs such as +`DevicePlacement`, are `Send + Sync` (`alloc/src/lib.rs:313-319`, +`alloc/src/segments.rs:13-15`, `alloc/src/slab.rs:34-43`). Each worker owns its +storage ranges, segment table, and buffer pool; live allocation and I/O authority +cannot move to another worker. ## Buffers From 2abb49bdafd8e9f90821f827c14db7aa00b6937e Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 21:03:24 +0000 Subject: [PATCH 43/82] docs(alloc): describe exact-size idle buffer reuse --- designs/racer-page-alloc.md | 10 ++++++++-- 1 file changed, 8 insertions(+), 2 deletions(-) diff --git a/designs/racer-page-alloc.md b/designs/racer-page-alloc.md index a4874b35f..0ac51da57 100644 --- a/designs/racer-page-alloc.md +++ b/designs/racer-page-alloc.md @@ -53,8 +53,14 @@ multiple of the offset and length units, so the next record also starts aligned. A single buffer is at most 1 GiB. Each buffer holds a caller-supplied `Charge`, so the caller can account for -memory against its own budget. The slab keeps at most one idle buffer of the -last size used (`alloc/src/slab.rs:398-415`). It is not a general size-class pool. +memory against its own budget. The slab keeps at most one idle buffer. +After checking charge coverage, allocation reuses it only for an exact length +match; otherwise it frees the idle buffer and allocates new storage +(`alloc/src/slab.rs:566-580`). On drop, a buffer fills the idle slot only if the +pool still exists and the slot is empty and can be mutably borrowed; otherwise +its storage is freed (`alloc/src/lib.rs:630-650`, `alloc/src/lib.rs:522-529`). +The retained size depends on return order, not necessarily the last size used. +This is not a general size-class pool. Each buffer tracks whether it is still all zeros. Any mutable access, including handing it to the kernel for a read, marks it dirty. On drop, a dirty buffer is From cdb596960ab8e158a0934ba58c40a5257f6a86ca Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 21:03:37 +0000 Subject: [PATCH 44/82] docs(alloc): use American English cancellation spelling --- designs/racer-page-alloc.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/designs/racer-page-alloc.md b/designs/racer-page-alloc.md index 0ac51da57..f23f91a85 100644 --- a/designs/racer-page-alloc.md +++ b/designs/racer-page-alloc.md @@ -165,7 +165,7 @@ change, so the same workflow tests run in both modes. - Unit tests cover padding math, wipe and reuse rules, lease and generation checks, reclaim limits, the eviction veto, restore, and file security checks. -- `alloc/tests/workflows.rs` covers restart, short I/O, dropped and cancelled +- `alloc/tests/workflows.rs` covers restart, short I/O, dropped and canceled reads and writes, startup failures, and confirms that reuse does not erase disk bytes. - CI checks and tests the production and simulation builds as separate From da5042f1e928d459e4b65246db79f6b8f48c1e44 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 21:06:08 +0000 Subject: [PATCH 45/82] fix(flow): retire keyed charge before waking admission --- cmd/racer-dataplane/flow/src/lib.rs | 93 +++++++++++++++++++++++++++++ 1 file changed, 93 insertions(+) diff --git a/cmd/racer-dataplane/flow/src/lib.rs b/cmd/racer-dataplane/flow/src/lib.rs index b17c7819f..a2cc86fb1 100644 --- a/cmd/racer-dataplane/flow/src/lib.rs +++ b/cmd/racer-dataplane/flow/src/lib.rs @@ -432,6 +432,8 @@ impl Drop for Charge

{ /// Release exactly this charge's remaining amount and apply wake policy. fn drop(&mut self) { self.release_to(0); + // Retire the final key owner before a wake callback retries admission. + drop(self.local.take()); if P::wakes(self.class) { self.totals.wake.wake(); } @@ -1456,6 +1458,97 @@ mod quota_tests { assert_eq!(quotas.active_keys.load(Ordering::Acquire), 0); } + /// A final keyed release frees its record before synchronous waiter reentry. + #[test] + fn keyed_release_retires_before_reentrant_wake() { + thread_local! { + static ON_WAKE: RefCell>> = RefCell::new(None); + } + + struct ReentrantWake(AtomicUsize); + + impl std::task::Wake for ReentrantWake { + fn wake(self: Arc) { + self.0.fetch_add(1, Ordering::Relaxed); + let callback = ON_WAKE.with(|slot| slot.borrow_mut().take()); + if let Some(callback) = callback { + callback(); + } + } + } + + let quotas = Rc::new(Quotas::new(TestPolicy::new(10, 1))); + let old = "old".to_owned(); + let mut charge = quotas.reserve(Some(&old), Resource::Other, 10).unwrap(); + let split = charge.split(4).unwrap(); + let wake = Arc::new(ReentrantWake(AtomicUsize::new(0))); + let waker = std::task::Waker::from(wake.clone()); + + let nested = quotas.clone(); + ON_WAKE.with(|slot| { + *slot.borrow_mut() = Some(Box::new(move || { + assert_eq!(nested.used(Resource::Other), 6); + assert_eq!(nested.active_keys.load(Ordering::Acquire), 1); + let local = nested.keys.borrow()["old"].upgrade().unwrap(); + assert_eq!(local.counter(Resource::Other).used(), 6); + assert!(nested.retired_keys.lock().unwrap().is_empty()); + assert!(matches!( + nested.reserve(Some(&"new".to_owned()), Resource::Other, 10), + Err(Error::Overloaded) + )); + })); + }); + quotas.shared().register(&waker); + drop(split); + assert_eq!(wake.0.load(Ordering::Relaxed), 1); + assert_eq!(charge.amount(), 6); + assert_eq!(quotas.used(Resource::Other), 6); + + quotas.policy.rejected.lock().unwrap().clear(); + let nested = quotas.clone(); + ON_WAKE.with(|slot| { + *slot.borrow_mut() = Some(Box::new(move || { + assert_eq!(nested.used(Resource::Other), 0); + let replacement = nested + .reserve(Some(&"new".to_owned()), Resource::Other, 10) + .expect("the final release must free the key slot before waking"); + assert_eq!(nested.used(Resource::Other), 10); + assert_eq!(nested.active_keys.load(Ordering::Acquire), 1); + assert_eq!( + replacement + .local + .as_ref() + .unwrap() + .counter(Resource::Other) + .used(), + 10 + ); + assert!(!nested.keys.borrow().contains_key("old")); + assert_eq!(nested.keys.borrow().len(), 1); + assert!(nested.retired_keys.lock().unwrap().is_empty()); + drop(replacement); + assert_eq!(nested.used(Resource::Other), 0); + assert_eq!(nested.active_keys.load(Ordering::Acquire), 0); + assert_eq!( + nested + .retired_keys + .lock() + .unwrap() + .iter() + .collect::>(), + vec!["new"] + ); + })); + }); + quotas.shared().register(&waker); + drop(charge); + assert_eq!(wake.0.load(Ordering::Relaxed), 2); + assert_eq!(quotas.used(Resource::Other), 0); + assert_eq!(quotas.active_keys.load(Ordering::Acquire), 0); + assert!(quotas.policy.rejected.lock().unwrap().is_empty()); + assert!(ON_WAKE.with(|slot| slot.borrow().is_none())); + } + /// Charges and shared handles retain exact accounting across threads. #[test] fn charge_validation_and_thread_safe_shared_usage() { From e505f0124972d772e5edad686349c9fa1057629b Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 21:20:00 +0000 Subject: [PATCH 46/82] fix(alloc): reject restored generation rollback --- cmd/racer-dataplane/alloc/src/segments.rs | 58 +++++++++++++++++++++++ 1 file changed, 58 insertions(+) diff --git a/cmd/racer-dataplane/alloc/src/segments.rs b/cmd/racer-dataplane/alloc/src/segments.rs index 7f30b7f4c..230ece252 100644 --- a/cmd/racer-dataplane/alloc/src/segments.rs +++ b/cmd/racer-dataplane/alloc/src/segments.rs @@ -473,6 +473,7 @@ impl Segments { for (i, s) in images.iter().enumerate() { if s.id.0 != i as u64 || s.generation.0 == 0 + || s.generation.0 < slots[i].image.generation.0 || s.used_bytes > self.segment_bytes || !s.used_bytes.is_multiple_of(alignment.offset()) || !s.used_bytes.is_multiple_of(alignment.length() as u64) @@ -496,6 +497,7 @@ impl Segments { /// Validate the complete image before publishing it, sealing its open tail. /// Frozen tables and outstanding leases return Busy without changing state. + /// A generation below the current slot generation returns Corrupt. pub fn restore(&self, images: Vec) -> Result<()> { let epoch = self.validate_restore_epoch(&images)?; let mut slots = self.slots.borrow_mut(); @@ -891,6 +893,62 @@ mod tests { assert!(s.restore(snap).is_err()); } + /// Reject rollback atomically while allowing equal or newer generations. + #[test] + fn restore_rejects_generation_rollback_without_reviving_stale_mappings() { + let s = segments(1024, 2); + drop(s.append(1024).unwrap()); + let (lease, extent) = s.append(1024).unwrap(); + let id = lease.id(); + let generation = lease.generation(); + drop(lease); + let old_sealed = s.snapshot(); + s.begin_evict(id).unwrap(); + s.recycle(id).unwrap(); + let before = s.snapshot(); + let epoch = s.restore_epoch(); + + for state in [ + SegmentState::Free, + SegmentState::Open, + SegmentState::Sealed, + SegmentState::Evicting, + ] { + let mut images = old_sealed.clone(); + images[0].generation = Generation(3); + images[1].state = state; + images[1].used_bytes = match state { + SegmentState::Free => 0, + SegmentState::Open => 512, + _ => 1024, + }; + assert_eq!(s.validate_restore(&images), Err(Error::Corrupt)); + assert_eq!(s.restore(images), Err(Error::Corrupt)); + assert_eq!(s.snapshot(), before); + assert_eq!(s.restore_epoch(), epoch); + assert_eq!(*s.free.borrow(), BTreeSet::from([1])); + assert_eq!(s.open.get(), None); + assert_eq!(s.evicting.get(), 0); + } + + let (lease, new_extent) = s.append(1024).unwrap(); + assert_eq!(lease.id(), id); + assert_eq!(lease.generation(), Generation(2)); + assert_eq!(new_extent, extent); + assert_eq!(s.validate(id, generation, &extent), Err(Error::Stale)); + assert!(matches!(s.lease(id, generation), Err(Error::Stale))); + drop(lease); + + let mut images = s.snapshot(); + assert_eq!(s.validate_restore(&images), Ok(())); + s.restore(images.clone()).unwrap(); + images[1].generation = Generation(3); + assert_eq!(s.validate_restore(&images), Ok(())); + s.restore(images).unwrap(); + assert_eq!(s.snapshot()[1].generation, Generation(3)); + assert_eq!(s.validate(id, generation, &extent), Err(Error::Stale)); + } + /// Tail rotation and generation exhaustion never wrap into stale authority. #[test] fn generation_exhaustion_never_wraps_and_small_tails_are_sealed() { From b1bee69b613399df5aaa4d1d1a3567306d1862cc Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 21:20:01 +0000 Subject: [PATCH 47/82] test(alloc): honor optional direct I/O capability skips --- cmd/racer-dataplane/alloc/src/slab.rs | 92 +++++++++++++++++++++++---- 1 file changed, 81 insertions(+), 11 deletions(-) diff --git a/cmd/racer-dataplane/alloc/src/slab.rs b/cmd/racer-dataplane/alloc/src/slab.rs index 82a00af9a..9ae1f1b3b 100644 --- a/cmd/racer-dataplane/alloc/src/slab.rs +++ b/cmd/racer-dataplane/alloc/src/slab.rs @@ -1004,13 +1004,20 @@ mod tests { } /// Skip only explicit unsupported-filesystem results unless real I/O is required. - fn real_alignment(result: Result) -> Option { + fn real_alignment(result: Result) -> Option { + real_io_result( + result, + std::env::var("PAGE_ALLOC_REQUIRE_REAL_IO").as_deref() == Ok("1"), + ) + } + + /// Apply the capability policy without changing the process environment in tests. + fn real_io_result(result: Result, required: bool) -> Option { match result { - Ok(alignment) => Some(alignment), + Ok(value) => Some(value), Err(Error::Unsupported) => { - assert_ne!( - std::env::var("PAGE_ALLOC_REQUIRE_REAL_IO").as_deref(), - Ok("1"), + assert!( + !required, "PAGE_ALLOC_REQUIRE_REAL_IO=1 forbids capability skips: filesystem does not support direct I/O geometry" ); eprintln!("SKIP real slab test: filesystem does not support direct I/O geometry"); @@ -1020,6 +1027,62 @@ mod tests { } } + /// Successful setup is preserved in both optional and required modes. + #[test] + fn real_io_success_is_preserved() { + for required in [false, true] { + assert_eq!(real_io_result(Ok(42), required), Some(42)); + } + } + + /// Unsupported direct opens skip only when the real-I/O gate is optional. + #[test] + fn real_io_unsupported_open_obeys_required_policy() { + for errno in [libc::EINVAL, libc::EOPNOTSUPP, libc::ENOSYS] { + let error = direct_error("open", std::io::Error::from_raw_os_error(errno)); + assert_eq!(real_io_result::<()>(Err(error), false), None); + let panic = + std::panic::catch_unwind(|| real_io_result::<()>(Err(error), true)).unwrap_err(); + let message = panic + .downcast_ref::() + .map(String::as_str) + .or_else(|| panic.downcast_ref::<&str>().copied()) + .unwrap(); + assert!(message.contains("PAGE_ALLOC_REQUIRE_REAL_IO=1 forbids capability skips")); + } + } + + /// Other open failures, including errors without errno, must never skip. + #[test] + fn real_io_unexpected_open_errors_fail_in_both_modes() { + for errno in [ + Some(libc::EACCES), + Some(libc::EIO), + Some(libc::EEXIST), + None, + ] { + let error = direct_error( + "open", + errno + .map(std::io::Error::from_raw_os_error) + .unwrap_or_else(|| std::io::Error::other("synthetic")), + ); + assert_eq!( + error, + Error::SystemIo { + operation: "open", + errno + } + ); + for required in [false, true] { + assert!( + std::panic::catch_unwind(|| real_io_result::<()>(Err(error), required)) + .is_err() + ); + } + } + } + /// Pooling preserves the primary charge and zeroes bytes before reuse. #[test] fn aligned_pool_reuses_only_fenced_zeroed_admitted_storage() { @@ -1096,7 +1159,7 @@ mod tests { } /// Open direct files without using the slab's private-file startup path. - fn device_file(path: &Path) -> Arc { + fn device_file(path: &Path) -> Option> { use std::os::unix::fs::OpenOptionsExt; let file = std::fs::OpenOptions::new() .read(true) @@ -1104,9 +1167,10 @@ mod tests { .create_new(true) .custom_flags(libc::O_DIRECT) .open(path) - .unwrap(); + .map_err(|error| direct_error("open", error)); + let file = real_alignment(file)?; file.set_len(16384).unwrap(); - Arc::new(file) + Some(Arc::new(file)) } /// Device layouts reject aliases, misalignment, overflow, and invalid file flags. @@ -1114,7 +1178,9 @@ mod tests { fn device_layout_validation_is_read_only() { use std::os::unix::fs::OpenOptionsExt; let directory = Directory::new(); - let file = device_file(&directory.0.join("device")); + let Some(file) = device_file(&directory.0.join("device")) else { + return; + }; let Some(a) = real_alignment(probe(&file)) else { return; }; @@ -1239,7 +1305,9 @@ mod tests { #[test] fn device_submission_checks_logical_and_physical_extents() { let directory = Directory::new(); - let file = device_file(&directory.0.join("device")); + let Some(file) = device_file(&directory.0.join("device")) else { + return; + }; let Some(a) = real_alignment(probe(&file)) else { return; }; @@ -1333,7 +1401,9 @@ mod tests { } } let directory = Directory::new(); - let file = device_file(&directory.0.join("device")); + let Some(file) = device_file(&directory.0.join("device")) else { + return; + }; let Some(a) = real_alignment(probe(&file)) else { return; }; From dc605641e92e51016d7c03ac703a1d3f1452aa18 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 21:22:38 +0000 Subject: [PATCH 48/82] fix(flow): clone recycler keys before locking --- cmd/racer-dataplane/flow/src/lib.rs | 117 ++++++++++++++++++++++++++-- 1 file changed, 111 insertions(+), 6 deletions(-) diff --git a/cmd/racer-dataplane/flow/src/lib.rs b/cmd/racer-dataplane/flow/src/lib.rs index a2cc86fb1..0cf437065 100644 --- a/cmd/racer-dataplane/flow/src/lib.rs +++ b/cmd/racer-dataplane/flow/src/lib.rs @@ -332,7 +332,8 @@ impl Charge

{ if amount == 0 || amount >= self.amount { return Err(Error::InvalidInput); } - Ok(self.transfer(amount, self.buffers.clone())) + let key = self.key.clone(); + Ok(self.transfer(amount, self.buffers.clone(), key)) } /// Caller must retain at least the capacity of every live backing allocation. @@ -387,6 +388,9 @@ impl Charge

{ let Some(pool) = self.buffers.upgrade() else { return; }; + // Application cloning may reenter stop or fill the pool. Clone unlocked, + // then check retention again before transferring any admission. + let key = self.key.clone(); let mut pool = match pool.try_lock() { Ok(pool) => pool, Err(std::sync::TryLockError::WouldBlock) => return, @@ -397,14 +401,12 @@ impl Charge

{ } // SAFETY: wipe_payload initialized the entire capacity above. unsafe { bytes.set_len(bytes.capacity()) }; - let charge = self.transfer(self.amount, Weak::new()); + let charge = self.transfer(self.amount, Weak::new(), key); pool.push((bytes, charge)); } - /// Move admission to a new owner without touching either usage counter. - fn transfer(&mut self, amount: usize, buffers: Weak>) -> Self { - // Clone the application key before mutating ownership: its Clone may panic. - let key = self.key.clone(); + /// Move admission with a precloned key without touching either usage counter. + fn transfer(&mut self, amount: usize, buffers: Weak>, key: Option) -> Self { self.amount -= amount; Self { class: self.class, @@ -1335,6 +1337,109 @@ mod quota_tests { assert_eq!(quotas.retained_buffer_bytes(), 0); } + /// Key cloning runs unlocked, and callback changes are checked before retention. + #[test] + fn recycler_key_clone_runs_before_retention_checks() { + thread_local! { + static CLONE_HOOK: RefCell>> = RefCell::new(None); + } + + #[derive(Eq, PartialEq, Hash)] + struct Key; + + impl Clone for Key { + fn clone(&self) -> Self { + let hook = CLONE_HOOK.with(|slot| slot.borrow_mut().take()); + if let Some(hook) = hook { + hook(); + } + Self + } + } + + struct ClonePolicy; + + impl Policy for ClonePolicy { + type Class = Resource; + type Key = Key; + + fn limit(&self, _: Resource) -> usize { + 4 << 20 + } + + fn max_keys(&self) -> usize { + 1 + } + + fn wakes(_: Resource) -> bool { + false + } + + fn covers(_: Resource) -> bool { + false + } + + fn rejected(&self, _: Rejection) { + panic!("unexpected rejection"); + } + } + + let size = 1 << 20; + for action in ["retain", "stop", "fill"] { + let quotas = Rc::new(Quotas::new(ClonePolicy)); + let mut idle = quotas.reserve(None, Resource::Payload, size).unwrap(); + idle.recycle(vec![0xa7; size]); + let mut donor = quotas + .reserve(Some(&Key), Resource::Payload, 2 * size) + .unwrap(); + let nested = quotas.clone(); + CLONE_HOOK.with(|slot| { + *slot.borrow_mut() = Some(Box::new(move || { + // Fail before reentry can block on the same mutex. + assert!( + nested.buffers.try_lock().is_ok(), + "key cloned under recycler lock" + ); + match action { + "stop" => nested.stop(), + "fill" => { + let mut extra = nested.reserve(None, Resource::Payload, size).unwrap(); + extra.recycle(vec![0xa7; size]); + } + _ => {} + } + })); + }); + + donor.recycle(vec![0xa7; size]); + assert!(CLONE_HOOK.with(|slot| slot.borrow().is_none())); + assert_eq!(quotas.is_stopped(), action == "stop"); + let retained = match action { + "retain" => 3 * size, + "fill" => 2 * size, + _ => 0, + }; + let owned = if action == "retain" { 0 } else { 2 * size }; + assert_eq!(donor.amount(), owned); + assert!(donor.key().is_some()); + assert_eq!( + donor + .local + .as_ref() + .unwrap() + .counter(Resource::Payload) + .used(), + 2 * size + ); + assert_eq!(quotas.retained_buffer_bytes(), retained); + assert_eq!(quotas.used(Resource::Payload), retained + owned); + drop(donor); + quotas.reclaim_buffers(); + assert_eq!(quotas.used(Resource::Payload), 0); + assert_eq!(quotas.active_keys.load(Ordering::Acquire), 0); + } + } + /// A poisoned recycler remains usable after a prior operation panicked. #[test] fn recycler_recovers_poisoned_mutex() { From 652fad887c2ecd672a0f1b81123605be0206a0e0 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 21:21:34 +0000 Subject: [PATCH 49/82] fix(flow): revalidate dynamic limits before reclamation --- cmd/racer-dataplane/flow/src/lib.rs | 148 +++++++++++++++++++++++++++- 1 file changed, 145 insertions(+), 3 deletions(-) diff --git a/cmd/racer-dataplane/flow/src/lib.rs b/cmd/racer-dataplane/flow/src/lib.rs index 0cf437065..61b577f4d 100644 --- a/cmd/racer-dataplane/flow/src/lib.rs +++ b/cmd/racer-dataplane/flow/src/lib.rs @@ -525,7 +525,8 @@ impl Quotas

{ if local_deficit != 0 { return Some((Some(key.clone()), local_deficit)); } - let deficit = self.used(class).saturating_sub(self.limit(class) - amount); + let headroom = self.limit(class).checked_sub(amount)?; + let deficit = self.used(class).saturating_sub(headroom); (deficit != 0).then_some((None, deficit)) } @@ -1037,7 +1038,7 @@ mod quota_tests { } } - /// Fixed limits and a shared rejection log for assertions. + /// Optional limit samples and a shared rejection log for assertions. #[derive(Clone)] struct TestPolicy { limit: usize, @@ -1047,6 +1048,8 @@ mod quota_tests { rejected: Arc>>>, limit_gate: Option>, + + limit_samples: Arc>>, } impl TestPolicy { @@ -1057,6 +1060,7 @@ mod quota_tests { max_keys, rejected: Arc::default(), limit_gate: None, + limit_samples: Arc::default(), } } } @@ -1074,7 +1078,11 @@ mod quota_tests { gate.0.wait(); gate.1.wait(); } - self.limit + self.limit_samples + .lock() + .unwrap() + .pop_front() + .unwrap_or(self.limit) } /// Return the fixture's key-record bound. @@ -1251,6 +1259,140 @@ mod quota_tests { assert_eq!(quotas.used(Resource::Payload), 0); } + /// A lower second limit must not wrap or suggest an impossible byte remedy. + #[test] + fn dynamic_limits_reclamation_revalidates_aggregate_headroom() { + for shared_usage in [false, true] { + for existing_key in [false, true] { + for (first, second, amount, expected) in [ + (10, 0, 5, None), + (10, 4, 5, None), + (10, 5, 5, Some((None, 3))), + (10, 7, 5, Some((None, 1))), + (10, 8, 5, None), + (10, 20, 5, None), + (usize::MAX, usize::MAX, usize::MAX, Some((None, 3))), + (4, 0, 5, None), + ] { + let quotas = Quotas::new(TestPolicy::new(100, 1)); + let shared = quotas.shared(); + let key = "a".to_owned(); + let owner = existing_key + .then(|| quotas.reserve(Some(&key), Resource::Other, 1).unwrap()); + let held = if shared_usage { + shared.reserve(Resource::Payload, 3).unwrap() + } else { + quotas.reserve(None, Resource::Payload, 3).unwrap() + }; + quotas + .policy + .limit_samples + .lock() + .unwrap() + .extend([first, second]); + assert_eq!( + quotas.reclamation(&key, Resource::Payload, amount), + expected + ); + assert_eq!(held.amount(), 3); + assert_eq!(quotas.used(Resource::Payload), 3); + assert_eq!(shared.used(Resource::Payload), 3); + assert_eq!( + quotas.active_keys.load(Ordering::Acquire), + usize::from(existing_key) + ); + assert!(quotas.policy.rejected.lock().unwrap().is_empty()); + drop((held, owner)); + assert_eq!(shared.used(Resource::Payload), 0); + assert_eq!(quotas.active_keys.load(Ordering::Acquire), 0); + } + } + } + } + + /// Lower ceilings reject without losing old charges; higher ceilings admit exactly. + #[test] + fn dynamic_limits_admission_preserves_failure_accounting_and_success_edges() { + let quotas = Quotas::new(TestPolicy::new(4, 1)); + let shared = quotas.shared(); + let key = "a".to_owned(); + quotas.policy.limit_samples.lock().unwrap().extend([10, 10]); + let held = quotas.reserve(Some(&key), Resource::Payload, 6).unwrap(); + assert!(matches!( + shared.reserve(Resource::Payload, 1), + Err(Error::Overloaded) + )); + assert!(matches!( + quotas.reserve(None, Resource::Payload, 1), + Err(Error::Overloaded) + )); + assert!(matches!( + quotas.reserve(Some(&key), Resource::Payload, 1), + Err(Error::Overloaded) + )); + assert!(matches!( + quotas.reserve_completion(Some(&key), Resource::Payload, 1), + Err(Error::Overloaded) + )); + assert!(matches!( + shared.reserve(Resource::Payload, 0), + Err(Error::InvalidInput) + )); + assert!(matches!( + quotas.reserve(Some(&key), Resource::Payload, 0), + Err(Error::InvalidInput) + )); + assert_eq!(quotas.used(Resource::Payload), 6); + assert_eq!(shared.used(Resource::Payload), 6); + assert_eq!( + held.local + .as_ref() + .unwrap() + .counter(Resource::Payload) + .used(), + 6 + ); + assert_eq!(quotas.active_keys.load(Ordering::Acquire), 1); + { + let rejected = quotas.policy.rejected.lock().unwrap(); + assert_eq!( + rejected.len(), + 7, + "local retries once; shared does not retry" + ); + for (index, rejection) in rejected.iter().enumerate() { + let (key_used, key_limit) = if matches!(index, 3 | 4) { + (Some(6), Some(4)) + } else { + (None, None) + }; + assert!(matches!(rejection, Rejection::Resource { + class: Resource::Payload, used: 6, limit: 4, requested: 1, + key_used: actual_used, key_limit: actual_limit, + } if *actual_used == key_used && *actual_limit == key_limit)); + } + } + quotas.policy.limit_samples.lock().unwrap().extend([10, 10]); + let refill = quotas.reserve(Some(&key), Resource::Payload, 4).unwrap(); + assert_eq!(shared.used(Resource::Payload), 10); + assert_eq!( + held.local + .as_ref() + .unwrap() + .counter(Resource::Payload) + .used(), + 10 + ); + drop((refill, held)); + assert_eq!(shared.used(Resource::Payload), 0); + assert_eq!(quotas.active_keys.load(Ordering::Acquire), 0); + let exact = shared.reserve(Resource::Payload, 4).unwrap(); + assert_eq!(quotas.used(Resource::Payload), 4); + drop(exact); + assert_eq!(quotas.used(Resource::Payload), 0); + assert_eq!(quotas.policy.rejected.lock().unwrap().len(), 7); + } + /// Idle backing retains two live charges and reports pressure before retry. #[test] fn recycler_retains_two_live_charges_and_reports_pressure_before_retry() { From c7d3e99c4c8fc283daf70f43a05434851ddab2f3 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 21:38:58 +0000 Subject: [PATCH 50/82] fix(flow): deduplicate handoff target keys --- cmd/racer-dataplane/flow/src/admission.rs | 98 +++++++++++++++++++---- 1 file changed, 81 insertions(+), 17 deletions(-) diff --git a/cmd/racer-dataplane/flow/src/admission.rs b/cmd/racer-dataplane/flow/src/admission.rs index c539b5709..514af843f 100644 --- a/cmd/racer-dataplane/flow/src/admission.rs +++ b/cmd/racer-dataplane/flow/src/admission.rs @@ -914,24 +914,24 @@ mod handoff { impl Handoff { /// Create stable target slots; empty target lists simply reject offers. + /// Repeated keys share one slot, in first-occurrence order. pub fn new(keys: &[K]) -> Self { - Self(Mutex::new(State { - targets: keys - .iter() - .map(|key| { - ( - key.clone(), - Target { - admission: None, - queue: VecDeque::new(), - waker: None, - closed: false, - }, - ) - }) - .collect(), - cursor: 0, - })) + let mut targets = Vec::new(); + for key in keys { + if targets.iter().any(|(candidate, _)| candidate == key) { + continue; + } + targets.push(( + key.clone(), + Target { + admission: None, + queue: VecDeque::new(), + waker: None, + closed: false, + }, + )); + } + Self(Mutex::new(State { targets, cursor: 0 })) } /// Install admission once on a known target that has not closed. @@ -1036,6 +1036,70 @@ mod handoff { Ok(()) } } + + #[cfg(test)] + mod tests { + use super::*; + + struct Ready; + + impl Admission for Ready { + type Reservation = (); + + fn register(&self, _: &Waker) {} + + fn reserve(&self) -> Result<()> { + Ok(()) + } + } + + #[test] + fn handoff_constructor_keeps_first_unique_targets() { + let cases: &[(&[u8], &[u8])] = &[ + (&[], &[]), + (&[1], &[1]), + (&[1, 1, 1], &[1]), + (&[3, 1, 3, 2, 1, 3], &[3, 1, 2]), + ]; + for &(keys, expected) in cases { + let handoff = Arc::new(Handoff::<_, Ready, u8>::new(keys)); + assert_eq!( + handoff + .0 + .lock() + .unwrap() + .targets + .iter() + .map(|(key, _)| *key) + .collect::>(), + expected + ); + for key in expected { + handoff.install(key, Ready).unwrap(); + assert_eq!(handoff.install(key, Ready), Err(Error::InvalidInput)); + } + for _ in 0..2 { + for key in expected { + handoff + .reserve(Waker::noop()) + .unwrap() + .deliver(|| *key) + .unwrap(); + let [item] = handoff.pop_batch::<1>(key, Waker::noop(), 1).unwrap(); + assert_eq!(item.unwrap().into_parts(), (*key, ())); + } + } + for key in expected { + handoff.close(key); + assert_eq!(handoff.install(key, Ready), Err(Error::InvalidInput)); + } + assert!(matches!( + handoff.reserve(Waker::noop()), + Err(Error::Overloaded) + )); + } + } + } } /// Shared speculative capacity and explicitly polled due-time registrations. From af87cf3a4241a5b85969fb442f3116a433a2bc55 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 21:39:34 +0000 Subject: [PATCH 51/82] fix(flow): release reclamation key borrow before callbacks --- cmd/racer-dataplane/flow/src/lib.rs | 185 +++++++++++++++++++++++++++- 1 file changed, 183 insertions(+), 2 deletions(-) diff --git a/cmd/racer-dataplane/flow/src/lib.rs b/cmd/racer-dataplane/flow/src/lib.rs index 61b577f4d..1aa6e056c 100644 --- a/cmd/racer-dataplane/flow/src/lib.rs +++ b/cmd/racer-dataplane/flow/src/lib.rs @@ -509,8 +509,8 @@ impl Quotas

{ class: P::Class, amount: usize, ) -> Option<(Option, usize)> { - let keys = self.keys.borrow(); - let local = keys.get(key).and_then(Weak::upgrade); + // Keep the counters alive, but release the lookup borrow before callbacks. + let local = self.keys.borrow().get(key).and_then(Weak::upgrade); let fair = self.fair_limit( class, self.active_keys.load(Ordering::Acquire) + usize::from(local.is_none()), @@ -1310,6 +1310,187 @@ mod quota_tests { } } + /// Reclamation callbacks may admit keyed work without borrowing the lookup table. + mod reclamation_callbacks { + use super::*; + + /// Application callback selected for one synchronous admission. + #[derive(Clone, Copy, Eq, PartialEq)] + enum Callback { + Limit, + + Floor, + + Clone, + } + + /// A one-shot callback, optionally delayed past the first limit sample. + struct Hook { + callback: Callback, + + skip: usize, + + action: Box, + } + + thread_local! { + static HOOK: RefCell> = const { RefCell::new(None) }; + } + + /// Take the selected callback before running it so nested calls are inert. + fn invoke(callback: Callback) { + let hook = HOOK.with(|slot| { + let mut slot = slot.borrow_mut(); + let hook = slot.as_mut()?; + if hook.callback != callback { + return None; + } + if hook.skip != 0 { + hook.skip -= 1; + return None; + } + slot.take() + }); + if let Some(hook) = hook { + (hook.action)(); + } + } + + /// A transferable identity with a worker-local clone callback. + #[derive(Debug, Eq, PartialEq, Hash)] + struct Key(u8); + + impl Clone for Key { + /// Exercise application code when reclamation returns a keyed deficit. + fn clone(&self) -> Self { + invoke(Callback::Clone); + Self(self.0) + } + } + + /// Fixed limits with one-shot application callbacks. + struct ReentrantPolicy; + + impl Policy for ReentrantPolicy { + type Class = Resource; + + type Key = Key; + + /// Permit both the initial charges and the nested admission. + fn limit(&self, _: Resource) -> usize { + invoke(Callback::Limit); + 100 + } + + /// Exercise callback reentry during fair-share calculation. + fn floor(&self, _: Resource) -> usize { + invoke(Callback::Floor); + 1 + } + + /// Leave room for nested creation as well as live-key reuse. + fn max_keys(&self) -> usize { + 3 + } + + /// No waiter is needed for this synchronous regression. + fn wakes(_: Resource) -> bool { + false + } + + /// The fixture does not allocate page backing. + fn covers(_: Resource) -> bool { + false + } + + /// Every nested admission must succeed without a retry. + fn rejected(&self, _: Rejection) { + panic!("unexpected rejection"); + } + } + + /// Check deficit results and accounting after new-key and live-key reentry. + fn check(callback: Callback, skip: usize) { + for nested_key in [0, 2] { + for (keyed, unkeyed, amount, expected) in [ + (40, 0, 11, Some((Some(Key(0)), 1))), + (0, 90, 11, Some((None, 1))), + (0, 0, 11, None), + (0, 0, 51, None), + ] { + let local_deficit = keyed != 0; + if (callback == Callback::Clone && !local_deficit) + || (skip != 0 && (local_deficit || amount > 50)) + { + continue; + } + let quotas = Rc::new(Quotas::new(ReentrantPolicy)); + let first = quotas.reserve(Some(&Key(0)), Resource::Other, 1).unwrap(); + let second = quotas.reserve(Some(&Key(1)), Resource::Other, 1).unwrap(); + let payload = (keyed + unkeyed != 0).then(|| { + quotas + .reserve( + (keyed != 0).then_some(&Key(0)), + Resource::Payload, + keyed + unkeyed, + ) + .unwrap() + }); + let nested = quotas.clone(); + HOOK.with(|slot| { + *slot.borrow_mut() = Some(Hook { + callback, + skip, + action: Box::new(move || { + let charge = nested + .reserve(Some(&Key(nested_key)), Resource::Progress, 1) + .unwrap(); + assert_eq!(nested.used(Resource::Progress), 1); + drop(charge); + assert_eq!(nested.used(Resource::Progress), 0); + }), + }); + }); + assert_eq!( + quotas.reclamation(&Key(0), Resource::Payload, amount), + expected + ); + assert!(HOOK.with(|slot| slot.borrow().is_none())); + assert_eq!(quotas.used(Resource::Payload), keyed + unkeyed); + assert_eq!(quotas.active_keys.load(Ordering::Acquire), 2); + drop((payload, first, second)); + assert_eq!(quotas.used(Resource::Payload), 0); + assert_eq!(quotas.used(Resource::Other), 0); + assert_eq!(quotas.active_keys.load(Ordering::Acquire), 0); + } + } + } + + /// The first aggregate sample runs without a key-table borrow. + #[test] + fn fair_limit_allows_keyed_reentry() { + check(Callback::Limit, 0); + } + + /// The later aggregate headroom sample also permits keyed reentry. + #[test] + fn aggregate_limit_allows_keyed_reentry() { + check(Callback::Limit, 1); + } + + /// The per-key floor callback may admit work synchronously. + #[test] + fn floor_allows_keyed_reentry() { + check(Callback::Floor, 0); + } + + /// Cloning the returned identity may admit work synchronously. + #[test] + fn key_clone_allows_keyed_reentry() { + check(Callback::Clone, 0); + } + } + /// Lower ceilings reject without losing old charges; higher ceilings admit exactly. #[test] fn dynamic_limits_admission_preserves_failure_accounting_and_success_edges() { From 4f62bac7bf81f9275f723898771f53856cba14ae Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 21:49:08 +0000 Subject: [PATCH 52/82] fix(alloc): reject oversized startup tables before opening files --- cmd/racer-dataplane/alloc/src/slab.rs | 108 ++++++++++++++++++++++++++ 1 file changed, 108 insertions(+) diff --git a/cmd/racer-dataplane/alloc/src/slab.rs b/cmd/racer-dataplane/alloc/src/slab.rs index 9ae1f1b3b..6305ac316 100644 --- a/cmd/racer-dataplane/alloc/src/slab.rs +++ b/cmd/racer-dataplane/alloc/src/slab.rs @@ -271,6 +271,15 @@ impl Slab { /// Blocking startup helper. Do not invoke on a latency-sensitive worker. pub fn open_configured(&self, segments: &Segments) -> Result { + // Automatic tables need one slot per physical segment; partial tables do not. + if !segments.is_configured() + && self + .capacity_bytes + .checked_div(self.segment_bytes) + .is_none_or(|count| count > crate::MAX_SEGMENTS) + { + return Err(Error::InvalidConfiguration); + } let alignment = self.open_file()?; self.bind(segments)?; Ok(alignment) @@ -1529,6 +1538,105 @@ mod tests { assert_eq!(conflicting.open_now(), Err(Error::Unavailable)); } + /// An oversized automatic table must fail before creating or sizing its backing. + #[test] + fn open_configured_rejects_oversized_table_without_file_side_effects() { + use std::os::unix::fs::OpenOptionsExt; + + let directory = Directory::new(); + let existing = directory.0.join("existing"); + let file = std::fs::OpenOptions::new() + .write(true) + .create_new(true) + .mode(0o600) + .open(&existing) + .unwrap(); + let paths = [ + directory.0.join("new"), + directory.0.join("missing/data"), + existing.clone(), + ]; + for path in &paths { + let segments = Segments::new(4096); + let slab = Slab::<()>::new(path.clone(), 4096 * (crate::MAX_SEGMENTS + 1), 4096, 512); + assert_eq!( + slab.open_configured(&segments), + Err(Error::InvalidConfiguration) + ); + assert!(!segments.is_configured()); + assert_eq!(segments.count(), 0); + assert_eq!(file.metadata().unwrap().len(), 0); + if path != &existing { + assert!(!path.exists()); + } + assert!(!directory.0.join("missing").exists()); + assert_eq!(slab.geometry(), Err(Error::Unavailable)); + + let retry = Slab::<()>::new(path.clone(), 8192, 4096, 512); + if real_alignment(retry.open_configured(&segments)).is_none() { + return; + } + assert_eq!(segments.count(), 2); + assert_eq!(std::fs::metadata(path).unwrap().len(), 8192); + drop(retry); + std::fs::remove_file(path).unwrap(); + if path.parent() != Some(directory.0.as_path()) { + std::fs::remove_dir(path.parent().unwrap()).unwrap(); + } + } + } + + /// Invalid arithmetic inputs fail without creating a directory or a file. + #[test] + fn open_configured_rejects_invalid_geometry_before_open() { + let directory = Directory::new(); + for (capacity, segment, record) in [ + (0, 0, 512), + (4096, 0, 512), + (0, 4096, 512), + (4097, 4096, 512), + (4096, 8192, 512), + (4096, 4096, 0), + (u64::MAX, 1, 512), + (u64::MAX, u64::MAX, 512), + ] { + let path = directory.0.join("missing/data"); + let slab = Slab::<()>::new(path.clone(), capacity, segment, record); + let segments = Segments::new(segment); + assert_eq!( + slab.open_configured(&segments), + Err(Error::InvalidConfiguration) + ); + assert!(!segments.is_configured()); + assert_eq!(slab.geometry(), Err(Error::Unavailable)); + assert!(!path.exists()); + assert!(!path.parent().unwrap().exists()); + } + } + + /// The slot limit does not cap physical backing for an already configured table. + #[test] + fn open_configured_accepts_partial_table_above_physical_segment_limit() { + let directory = Directory::new(); + let probe = Slab::<()>::new(directory.0.join("probe"), 4096, 4096, 512); + let Some(alignment) = real_alignment(probe.open_now()) else { + return; + }; + let capacity = 4096 * (crate::MAX_SEGMENTS + 1); + let segments = Segments::new(4096); + segments.configure(capacity, 2, alignment).unwrap(); + let path = directory.0.join("partial"); + let slab = Slab::<()>::new(path.clone(), capacity, 4096, 512); + assert_eq!(slab.open_configured(&segments), Ok(alignment)); + assert_eq!(segments.count(), 2); + assert_eq!(segments.capacity_bytes(), capacity); + assert_eq!( + slab.geometry().unwrap().segment_count(), + crate::MAX_SEGMENTS + 1 + ); + assert_eq!(std::fs::metadata(path).unwrap().len(), capacity); + } + /// Invalid startup dimensions never truncate a preexisting nonempty file. #[test] fn open_rejects_bad_layout_and_existing_size_without_truncating() { From 4ba02db1d59cc1d2498eb7b664a524b032784404 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 21:41:23 +0000 Subject: [PATCH 53/82] fix(alloc): check block-device placement capacity --- cmd/racer-dataplane/alloc/src/lib.rs | 12 ++- cmd/racer-dataplane/alloc/src/slab.rs | 148 ++++++++++++++++++++++++-- designs/racer-page-alloc.md | 11 +- 3 files changed, 152 insertions(+), 19 deletions(-) diff --git a/cmd/racer-dataplane/alloc/src/lib.rs b/cmd/racer-dataplane/alloc/src/lib.rs index 554d32dd5..a83543cdf 100644 --- a/cmd/racer-dataplane/alloc/src/lib.rs +++ b/cmd/racer-dataplane/alloc/src/lib.rs @@ -30,12 +30,16 @@ //! and unlink; neither traversal nor flock stops noncooperating writers. //! //! [`Slab::from_devices`] instead accepts one [`DevicePlacement`] per logical -//! segment. The caller opens devices with O_EXCL and O_DIRECT, checks capacity, -//! and supplies common alignment. Placements can share an `Arc` to use one +//! segment. The caller opens devices with O_EXCL and O_DIRECT and supplies common +//! alignment. Bounds use BLKGETSIZE64 for block devices and length for regular +//! files. Placements can share an `Arc` to use one //! runtime descriptor per device. Startup never creates, resizes, or locks these //! files. Logical extents still use segment-table offsets; submissions translate -//! them to the placement's physical range after checking lease bounds. Keep ranges -//! disjoint across workers, and do not change file flags or sizes while in use. +//! them to the placement's physical range after checking lease bounds. Overlap +//! checks only compare the same inode or device identity within one slab. They +//! cannot detect whole-disk, partition, or device-mapper aliases. The caller must +//! guarantee disjoint physical storage across aliases and slabs, keep exclusive +//! ownership, and not change file flags or sizes while in use. //! //! In file mode, empty files are sparsely extended to capacity; nonempty size mismatches are //! rejected without truncation. Capacity is a logical bound, not reserved disk diff --git a/cmd/racer-dataplane/alloc/src/slab.rs b/cmd/racer-dataplane/alloc/src/slab.rs index 6305ac316..a080743a4 100644 --- a/cmd/racer-dataplane/alloc/src/slab.rs +++ b/cmd/racer-dataplane/alloc/src/slab.rs @@ -91,16 +91,41 @@ impl Slab { } /// Describe device ranges in logical segment order, without creating, sizing, - /// or locking files. The caller must verify device capacity and supply alignment - /// that meets every device's requirements. Regular direct files are also accepted. - /// Overlapping ranges on the same device or file are rejected within this slab; - /// the caller must keep ranges in different slabs disjoint. + /// or locking files. Bounds use BLKGETSIZE64 for block devices and file length + /// for regular direct files. The caller must supply suitable alignment and keep + /// exclusive ownership, file flags, and capacity unchanged while in use. + /// Overlap checks only compare the same inode or device identity in this slab. + /// Whole-disk, partition, and device-mapper aliases are not detected. The caller + /// must guarantee disjoint physical storage across aliases and different slabs. /// Files stay owned until `open_configured` creates worker-local descriptors. pub fn from_devices( placements: Vec, segment_bytes: u64, max_record_bytes: usize, alignment: Alignment, + ) -> Result { + Self::from_devices_with_capacity( + placements, + segment_bytes, + max_record_bytes, + alignment, + |file, metadata| { + if metadata.mode() & libc::S_IFMT == libc::S_IFBLK { + block_device_capacity(file.as_raw_fd()) + } else { + Ok(metadata.len()) + } + }, + ) + } + + /// Keep capacity probing injectable for tests without block-device access. + fn from_devices_with_capacity( + placements: Vec, + segment_bytes: u64, + max_record_bytes: usize, + alignment: Alignment, + mut capacity: impl FnMut(&File, &std::fs::Metadata) -> Result, ) -> Result { let capacity_bytes = (placements.len() as u64) .checked_mul(segment_bytes) @@ -125,7 +150,7 @@ impl Slab { return Err(Error::InvalidConfiguration); } let key = Arc::as_ptr(&placement.file); - let metadata = match files.entry(key) { + let (metadata, capacity_bytes) = match files.entry(key) { std::collections::hash_map::Entry::Occupied(entry) => entry.into_mut(), std::collections::hash_map::Entry::Vacant(entry) => { // SAFETY: F_GETFL only inspects the live caller-owned descriptor. @@ -149,15 +174,16 @@ impl Slab { ) { return Err(Error::InvalidConfiguration); } - entry.insert(metadata) + let capacity_bytes = capacity(&placement.file, &metadata)?; + entry.insert((metadata, capacity_bytes)) } }; + if end > *capacity_bytes { + return Err(Error::InvalidConfiguration); + } let identity = if metadata.mode() & libc::S_IFMT == libc::S_IFBLK { (libc::S_IFBLK, metadata.rdev(), 0) } else { - if end > metadata.len() { - return Err(Error::InvalidConfiguration); - } (libc::S_IFREG, metadata.dev(), metadata.ino()) }; ranges.push((identity, placement.offset, end)); @@ -812,6 +838,22 @@ impl Drop for WriteFence { } } +/// Read the block device's byte capacity without changing its contents. +fn block_device_capacity(fd: i32) -> Result { + let mut capacity = 0u64; + // Linux fs.h encodes BLKGETSIZE64 with size_t, but the output is always u64. + let request = libc::_IOR::(0x12, 114); + // SAFETY: BLKGETSIZE64 writes one u64 to this live output pointer. Invalid + // descriptors are rejected by the kernel without accessing device storage. + if unsafe { libc::ioctl(fd, request, &mut capacity) } != 0 { + return Err(system_error( + "ioctl-blkgetsize64", + std::io::Error::last_os_error(), + )); + } + Ok(capacity) +} + /// Preserve synchronous operating-system diagnostic context. fn system_error(operation: &'static str, error: std::io::Error) -> Error { Error::SystemIo { @@ -1182,7 +1224,93 @@ mod tests { Some(Arc::new(file)) } - /// Device layouts reject aliases, misalignment, overflow, and invalid file flags. + /// Queried capacities bound every placement, including cached-file placements. + #[test] + fn device_capacity_bounds_are_checked_before_startup() { + let directory = Directory::new(); + let Some(file) = device_file(&directory.0.join("capacity")) else { + return; + }; + let a = Alignment::new(512, 512, 512).unwrap(); + for (capacity, offsets, valid) in [ + (0, vec![0], false), + (4095, vec![0], false), + (4096, vec![0], true), + (8191, vec![4096], false), + (8192, vec![4096], true), + (8192, vec![8192], false), + (8192, vec![0, 4096], true), + (8191, vec![0, 4096], false), + (u64::MAX, vec![0, 4096], true), + ] { + let mut queries = 0; + let result = Slab::<()>::from_devices_with_capacity( + offsets + .into_iter() + .map(|offset| DevicePlacement { + file: file.clone(), + offset, + }) + .collect(), + 4096, + 512, + a, + |_, _| { + queries += 1; + Ok(capacity) + }, + ); + if valid { + assert!(result.is_ok(), "capacity {capacity}: {:?}", result.err()); + } else { + assert!(matches!(result, Err(Error::InvalidConfiguration))); + } + assert_eq!(queries, 1); + assert_eq!(file.metadata().unwrap().len(), 16384); + } + } + + /// Failed capacity queries preserve syscall context rather than allowing a slab. + #[test] + fn device_capacity_errors_keep_errno() { + assert_eq!( + block_device_capacity(-1), + Err(Error::SystemIo { + operation: "ioctl-blkgetsize64", + errno: Some(libc::EBADF), + }) + ); + let directory = Directory::new(); + let Some(file) = device_file(&directory.0.join("capacity-error")) else { + return; + }; + assert_eq!( + block_device_capacity(file.as_raw_fd()), + Err(Error::SystemIo { + operation: "ioctl-blkgetsize64", + errno: Some(libc::ENOTTY), + }) + ); + for errno in [libc::EIO, libc::EACCES, libc::ENOTTY] { + let error = system_error( + "ioctl-blkgetsize64", + std::io::Error::from_raw_os_error(errno), + ); + let result = Slab::<()>::from_devices_with_capacity( + vec![DevicePlacement { + file: file.clone(), + offset: 0, + }], + 4096, + 512, + Alignment::new(512, 512, 512).unwrap(), + |_, _| Err(error), + ); + assert!(matches!(result, Err(actual) if actual == error)); + } + } + + /// Device layouts reject same-identity overlaps, misalignment, overflow, and invalid flags. #[test] fn device_layout_validation_is_read_only() { use std::os::unix::fs::OpenOptionsExt; diff --git a/designs/racer-page-alloc.md b/designs/racer-page-alloc.md index f23f91a85..3e7bc976a 100644 --- a/designs/racer-page-alloc.md +++ b/designs/racer-page-alloc.md @@ -133,11 +133,12 @@ startup (`alloc/src/slab.rs:368-409`). For this file-backed mode, it: `Slab::from_devices` instead owns caller-opened file or block-device placements, one per logical segment, without creating, sizing, or locking them. Files must be read/write with `O_DIRECT` and without `O_APPEND`. The constructor checks -geometry, offset alignment, regular-file bounds, and overlapping ranges within -the slab (`alloc/src/slab.rs:99-176`). The caller must open real devices -exclusively, verify device capacity, supply alignment that meets every device's -requirements, and keep ranges in different slabs disjoint -(`alloc/src/slab.rs:34-42`, `alloc/src/slab.rs:93-98`). +geometry, offset alignment, regular-file length, and block-device capacity from +`BLKGETSIZE64`. Overlap checks only compare the same inode or device identity +within the slab; they cannot detect whole-disk, partition, or device-mapper +aliases. The caller must guarantee disjoint physical storage across aliases and +slabs, keep exclusive ownership, supply suitable alignment, and keep file flags +and sizes unchanged while in use (see `Slab::from_devices` in `alloc/src/slab.rs`). For device placements, `open_configured` duplicates the owned files into worker-local descriptors and releases the original placement references From da579f54460669090f17350bf911f42db1a38369 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 21:43:57 +0000 Subject: [PATCH 54/82] test(alloc): exercise aligned device overlaps and inode aliases --- cmd/racer-dataplane/alloc/src/slab.rs | 244 +++++++++++++++----------- 1 file changed, 142 insertions(+), 102 deletions(-) diff --git a/cmd/racer-dataplane/alloc/src/slab.rs b/cmd/racer-dataplane/alloc/src/slab.rs index a080743a4..e2e52779b 100644 --- a/cmd/racer-dataplane/alloc/src/slab.rs +++ b/cmd/racer-dataplane/alloc/src/slab.rs @@ -1318,124 +1318,164 @@ mod tests { let Some(file) = device_file(&directory.0.join("device")) else { return; }; - let Some(a) = real_alignment(probe(&file)) else { + let Some(host_alignment) = real_alignment(probe(&file)) else { return; }; - let build = |offsets: &[u64], segment, record| { - Slab::<()>::from_devices( - offsets - .iter() - .map(|&offset| DevicePlacement { - file: file.clone(), - offset, - }) - .collect(), - segment, - record, - a, - ) - }; - for (offsets, segment, record) in [ - (vec![], 4096, 512), - (vec![0], 0, 512), - (vec![0], 4096, 0), - (vec![0], 4097, 512), - (vec![0], 4096, 4097), - (vec![0, 4096], u64::MAX, 512), - (vec![0], i64::MAX as u64 + 1, 512), - (vec![1], 4096, 512), - (vec![16384], 4096, 512), - (vec![u64::MAX], 4096, 512), - (vec![i64::MAX as u64 - 4095], 4096, 512), - (vec![0, 0], 4096, 512), - (vec![0, 2048], 4096, 512), + for a in [ + host_alignment, + Alignment::new(4096, 4096, 4096).unwrap(), + Alignment::new(8192, 8192, 8192).unwrap(), + Alignment::new(4096, 768, 512).unwrap(), ] { - assert!(matches!( - build(&offsets, segment, record), - Err(Error::InvalidConfiguration) - )); - } - let duplicate = Arc::new(file.try_clone().unwrap()); - assert!(matches!( - Slab::<()>::from_devices( - vec![ - DevicePlacement { - file: file.clone(), - offset: 0 - }, - DevicePlacement { - file: duplicate, - offset: 2048 - }, - ], - 4096, - 512, - a - ), - Err(Error::InvalidConfiguration) - )); - for flags in [0, libc::O_DIRECT | libc::O_APPEND] { - let invalid = std::fs::OpenOptions::new() + let unit = a.extent(0, 1).unwrap().length(); + let segment_bytes = 2 * unit as u64; + let file_bytes = 4 * segment_bytes; + file.set_len(file_bytes).unwrap(); + let build = |offsets: &[u64], segment, record| { + Slab::<()>::from_devices( + offsets + .iter() + .map(|&offset| DevicePlacement { + file: file.clone(), + offset, + }) + .collect(), + segment, + record, + a, + ) + }; + for (offsets, segment, record) in [ + (vec![], segment_bytes, unit), + (vec![0], 0, unit), + (vec![0], segment_bytes, 0), + (vec![0], segment_bytes + 1, unit), + (vec![0], segment_bytes, segment_bytes as usize + 1), + (vec![0, segment_bytes], u64::MAX, unit), + (vec![0], i64::MAX as u64 + 1, unit), + (vec![1], segment_bytes, unit), + (vec![file_bytes], segment_bytes, unit), + (vec![u64::MAX], segment_bytes, unit), + ( + vec![i64::MAX as u64 - segment_bytes + 1], + segment_bytes, + unit, + ), + ] { + assert!(matches!( + build(&offsets, segment, record), + Err(Error::InvalidConfiguration) + )); + } + let duplicate = Arc::new(file.try_clone().unwrap()); + assert!(!Arc::ptr_eq(&file, &duplicate)); + for second in [file.clone(), duplicate] { + let placement = |offset| DevicePlacement { + file: second.clone(), + offset, + }; + for offset in [0, unit as u64, segment_bytes] { + assert!( + Slab::<()>::from_devices(vec![placement(offset)], segment_bytes, unit, a) + .is_ok() + ); + } + assert!( + Slab::<()>::from_devices( + vec![ + DevicePlacement { + file: file.clone(), + offset: 0 + }, + placement(segment_bytes) + ], + segment_bytes, + unit, + a + ) + .is_ok() + ); + for offset in [0, unit as u64] { + assert!(matches!( + Slab::<()>::from_devices( + vec![ + DevicePlacement { + file: file.clone(), + offset: 0 + }, + placement(offset), + ], + segment_bytes, + unit, + a + ), + Err(Error::InvalidConfiguration) + )); + } + } + for flags in [0, libc::O_DIRECT | libc::O_APPEND] { + let invalid = std::fs::OpenOptions::new() + .read(true) + .write(true) + .custom_flags(flags) + .open(directory.0.join("device")) + .unwrap(); + assert!(matches!( + Slab::<()>::from_devices( + vec![DevicePlacement { + file: Arc::new(invalid), + offset: 0, + }], + segment_bytes, + unit, + a + ), + Err(Error::InvalidConfiguration) + )); + } + let readonly = std::fs::OpenOptions::new() .read(true) - .write(true) - .custom_flags(flags) + .custom_flags(libc::O_DIRECT) .open(directory.0.join("device")) .unwrap(); assert!(matches!( Slab::<()>::from_devices( vec![DevicePlacement { - file: Arc::new(invalid), + file: Arc::new(readonly), offset: 0, }], - 4096, - 512, + segment_bytes, + unit, a ), Err(Error::InvalidConfiguration) )); + let slab = build(&[2 * segment_bytes, segment_bytes], segment_bytes, unit).unwrap(); + assert_eq!(slab.capacity_bytes(), 2 * segment_bytes); + assert_eq!( + slab.open_configured(&Segments::new(2 * segment_bytes)), + Err(Error::InvalidConfiguration) + ); + let segments = Segments::new(segment_bytes); + assert_eq!(slab.open_configured(&segments), Ok(a)); + assert_eq!(slab.open_configured(&segments), Ok(a)); + assert_eq!(file.metadata().unwrap().len(), file_bytes); + let independent = File::open(directory.0.join("device")).unwrap(); + // SAFETY: this live independent descriptor tests that startup took no flock. + assert_eq!( + unsafe { libc::flock(independent.as_raw_fd(), libc::LOCK_EX | libc::LOCK_NB) }, + 0 + ); + let opened = slab.opened.borrow(); + let OpenBacking::Devices(placements) = &opened.as_ref().unwrap().backing else { + unreachable!() + }; + assert!(Rc::ptr_eq(&placements[0].file, &placements[1].file)); + let Backing::Devices { placements, .. } = &slab.backing else { + unreachable!() + }; + assert!(placements.borrow().is_empty()); } - let readonly = std::fs::OpenOptions::new() - .read(true) - .custom_flags(libc::O_DIRECT) - .open(directory.0.join("device")) - .unwrap(); - assert!(matches!( - Slab::<()>::from_devices( - vec![DevicePlacement { - file: Arc::new(readonly), - offset: 0, - }], - 4096, - 512, - a - ), - Err(Error::InvalidConfiguration) - )); - let slab = build(&[8192, 4096], 4096, 512).unwrap(); - assert_eq!(slab.capacity_bytes(), 8192); - assert_eq!( - slab.open_configured(&Segments::new(8192)), - Err(Error::InvalidConfiguration) - ); - let segments = Segments::new(4096); - assert_eq!(slab.open_configured(&segments), Ok(a)); - assert_eq!(slab.open_configured(&segments), Ok(a)); - assert_eq!(file.metadata().unwrap().len(), 16384); - let independent = File::open(directory.0.join("device")).unwrap(); - // SAFETY: this live independent descriptor tests that startup took no flock. - assert_eq!( - unsafe { libc::flock(independent.as_raw_fd(), libc::LOCK_EX | libc::LOCK_NB) }, - 0 - ); - let opened = slab.opened.borrow(); - let OpenBacking::Devices(placements) = &opened.as_ref().unwrap().backing else { - unreachable!() - }; - assert!(Rc::ptr_eq(&placements[0].file, &placements[1].file)); - let Backing::Devices { placements, .. } = &slab.backing else { - unreachable!() - }; - assert!(placements.borrow().is_empty()); } /// Translation preserves logical authority and rechecks physical bounds and alignment. From 6bbbc523e9a62dd86cdc83700cee4b55eb0cea5e Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 21:46:17 +0000 Subject: [PATCH 55/82] test(alloc): probe alignment before sizing real I/O workflows --- cmd/racer-dataplane/alloc/tests/workflows.rs | 118 +++++++++++++++---- 1 file changed, 93 insertions(+), 25 deletions(-) diff --git a/cmd/racer-dataplane/alloc/tests/workflows.rs b/cmd/racer-dataplane/alloc/tests/workflows.rs index 4b7472076..e245b1a59 100644 --- a/cmd/racer-dataplane/alloc/tests/workflows.rs +++ b/cmd/racer-dataplane/alloc/tests/workflows.rs @@ -746,6 +746,57 @@ fn real_alignment(result: page_alloc::Result) -> Option { } } +/// Discover alignment before choosing any slab dimensions. +fn probe_alignment(directory: &Directory) -> Option { + use std::os::{fd::AsRawFd, unix::fs::OpenOptionsExt}; + + let result = (|| { + let file = std::fs::OpenOptions::new() + .create_new(true) + .read(true) + .write(true) + .mode(0o600) + .custom_flags(libc::O_DIRECT) + .open(directory.0.join("alignment-probe")) + .map_err(|error| match error.raw_os_error() { + Some(libc::EINVAL | libc::EOPNOTSUPP | libc::ENOSYS) => Error::Unsupported, + errno => Error::SystemIo { + operation: "open", + errno, + }, + })?; + // SAFETY: stat is initialized and the file and empty path remain live. + let mut stat: libc::statx = unsafe { std::mem::zeroed() }; + if unsafe { + libc::statx( + file.as_raw_fd(), + c"".as_ptr(), + libc::AT_EMPTY_PATH, + libc::STATX_DIOALIGN, + &mut stat, + ) + } != 0 + { + return Err(match std::io::Error::last_os_error().raw_os_error() { + Some(libc::ENOSYS | libc::EOPNOTSUPP) => Error::Unsupported, + errno => Error::SystemIo { + operation: "statx", + errno, + }, + }); + } + if stat.stx_mask & libc::STATX_DIOALIGN == 0 { + return Err(Error::Unsupported); + } + Alignment::new( + stat.stx_dio_mem_align as usize, + stat.stx_dio_offset_align as u64, + stat.stx_dio_offset_align as usize, + ) + })(); + real_alignment(result) +} + /// Probe baseline io_uring support without hiding unexpected runtime setup failures. fn kernel_available() -> bool { let mut params = [0u64; 15]; @@ -793,14 +844,21 @@ fn io_uring_roundtrip_and_completion_fence() { return; } let directory = Directory::new(); - let slab = Slab::<()>::new(directory.0.join("uring.dat"), 8192, 4096, 512); - let segments = Segments::new(4096); - if real_alignment(slab.open_configured(&segments)).is_none() { + let Some(alignment) = probe_alignment(&directory) else { return; - } + }; + let (_, segment_bytes) = device_placement_sizes(alignment); + let slab = Slab::<()>::new( + directory.0.join("uring.dat"), + 2 * segment_bytes, + segment_bytes, + segment_bytes as usize, + ); + let segments = Segments::new(segment_bytes); + assert_eq!(slab.open_configured(&segments), Ok(alignment)); let reactor = Reactor::::new(16, ()); reactor.init().unwrap(); - let (lease, extent) = segments.append(4096).unwrap(); + let (lease, extent) = segments.append(segment_bytes as usize).unwrap(); let mut buffer = slab.allocate(extent.length(), ()).unwrap(); buffer.as_mut_slice().fill(73); drop( @@ -861,6 +919,10 @@ fn device_placement_sizes_support_4096_and_arbitrary_units() { (4096, 4096, 4096), (4096, 512, 4096), (512, 4096, 4096), + (8192, 8192, 8192), + (8192, 512, 8192), + (512, 8192, 8192), + (65536, 65536, 65536), (768, 512, 3072), (3, 5, 2055), ] { @@ -899,6 +961,14 @@ fn device_placement_sizes_support_4096_and_arbitrary_units() { } } assert_eq!(segments.free_count(), 0); + let whole_segments = Segments::from_geometry(geometry).unwrap(); + let buffer = alignment.allocate(segment_bytes as usize, ()).unwrap(); + for id in 0..3 { + let (lease, extent) = whole_segments.append(segment_bytes as usize).unwrap(); + assert_eq!(lease.id(), SegmentId(id)); + assert_eq!(extent.offset(), id * segment_bytes); + alignment.check(extent, &buffer).unwrap(); + } } } @@ -911,8 +981,7 @@ fn device_placements_route_real_io_and_reject_short_reads() { return; } let directory = Directory::new(); - let probe = Slab::<()>::new(directory.0.join("probe"), 16384, 4096, 4096); - let Some(alignment) = real_alignment(probe.open_configured(&Segments::new(4096))) else { + let Some(alignment) = probe_alignment(&directory) else { return; }; let (half_bytes, segment_bytes) = device_placement_sizes(alignment); @@ -1034,44 +1103,43 @@ fn device_placements_route_real_io_and_reject_short_reads() { #[test] fn unbound_and_failed_binding_reject_reads_and_writes_before_reactor_admission() { let directory = Directory::new(); + let Some(alignment) = probe_alignment(&directory) else { + return; + }; + let (_, segment_bytes) = device_placement_sizes(alignment); #[cfg(feature = "simulation")] let failures = ["unbound", "wrong-geometry", "frozen-table"].as_slice(); #[cfg(not(feature = "simulation"))] let failures = ["wrong-geometry", "frozen-table"].as_slice(); for &failure in failures { - let slab = Slab::<()>::new(directory.0.join(failure), 8192, 4096, 512); - let correct = Segments::new(4096); + let slab = Slab::<()>::new( + directory.0.join(failure), + 2 * segment_bytes, + segment_bytes, + segment_bytes as usize, + ); + let correct = Segments::new(segment_bytes); match failure { #[cfg(feature = "simulation")] "unbound" => { - if real_alignment(slab.open_now()).is_none() { - return; - } + assert_eq!(slab.open_now(), Ok(alignment)); } "wrong-geometry" => { - let result = slab.open_configured(&Segments::new(8192)); - if result == Err(Error::Unsupported) { - let _ = real_alignment(result); - return; - } + let result = slab.open_configured(&Segments::new(2 * segment_bytes)); assert_eq!(result, Err(Error::InvalidConfiguration)); } "frozen-table" => { let frozen = correct.freeze().unwrap(); let result = slab.open_configured(&correct); - if result == Err(Error::Unsupported) { - let _ = real_alignment(result); - return; - } assert_eq!(result, Err(Error::Busy)); drop(frozen); } _ => unreachable!(), } - let alignment = slab.alignment().unwrap(); + assert_eq!(slab.alignment(), Ok(alignment)); let correct = Segments::from_geometry(slab.geometry().unwrap()).unwrap(); - let (lease, extent) = correct.append(4096).unwrap(); - let buffer = slab.allocate(4096, ()).unwrap(); + let (lease, extent) = correct.append(segment_bytes as usize).unwrap(); + let buffer = slab.allocate(extent.length(), ()).unwrap(); let reactor = Reactor::::new(16, ()); let mut read = slab.read(&reactor, extent, buffer, lease, &TestScope); assert!(matches!( @@ -1082,7 +1150,7 @@ fn unbound_and_failed_binding_reject_reads_and_writes_before_reactor_admission() let mut write = slab.write( &reactor, extent, - slab.allocate(4096, ()).unwrap(), + slab.allocate(extent.length(), ()).unwrap(), correct.lease(SegmentId(0), Generation(1)).unwrap(), &TestScope, ); From 3f90a042d84c61d7d9ce7ef333317d2d2cabcb39 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 21:51:12 +0000 Subject: [PATCH 56/82] docs(alloc): clarify pending eviction reclaim results --- cmd/racer-dataplane/alloc/src/lib.rs | 11 ++++++++--- cmd/racer-dataplane/alloc/src/segments.rs | 9 +++++++++ designs/racer-page-alloc.md | 10 +++++++--- 3 files changed, 24 insertions(+), 6 deletions(-) diff --git a/cmd/racer-dataplane/alloc/src/lib.rs b/cmd/racer-dataplane/alloc/src/lib.rs index a83543cdf..879951544 100644 --- a/cmd/racer-dataplane/alloc/src/lib.rs +++ b/cmd/racer-dataplane/alloc/src/lib.rs @@ -122,9 +122,14 @@ //! Index reclamation forgets mappings without touching bytes or generations. //! Physical reclamation only visits Sealed/Evicting slots and recycles empty slots //! after leases drain. Each sweep visits at most two rotations, further capped by -//! the caller. Insufficient progress returns Busy, not a spin loop. A zero reserve -//! is a no-op; a zero entry budget can recycle empty slots but cannot begin -//! populated eviction. There is no compaction. [`SegmentEntries`] implementations +//! the caller. For a nonzero reserve, [`SegmentClock::reclaim`] and +//! [`SegmentClock::reclaim_scored`] succeed only when the reserve (capped at the +//! slot count) is met and no evictions remain. Busy is intentional even with enough +//! free slots while pending evictions drain. Use [`Segments::free_count`] to check +//! capacity, and retry bounded reclamation later to finish draining rather than +//! spinning. A zero reserve is a no-op, not a drain request; a zero entry budget +//! can recycle empty slots but cannot begin populated eviction. There is no +//! compaction. [`SegmentEntries`] implementations //! must compare current mappings before removal and report within budget; an //! error cannot undo callback side effects. Freeze does not block index sweeps. //! diff --git a/cmd/racer-dataplane/alloc/src/segments.rs b/cmd/racer-dataplane/alloc/src/segments.rs index 230ece252..ee35e717c 100644 --- a/cmd/racer-dataplane/alloc/src/segments.rs +++ b/cmd/racer-dataplane/alloc/src/segments.rs @@ -702,6 +702,11 @@ impl SegmentClock { /// Existing eviction work sorts first so partial removals always make progress. /// Unlike the legacy clock, logical heat is supplied by the caller, so recent /// disk completions do not override value ranking. + /// For a nonzero reserve, success requires both the reserve (capped at the slot + /// count) and no pending evictions. Busy is intentional even if the reserve is + /// met while evictions remain. Use [`Segments::free_count`] to check capacity; + /// keep retrying bounded reclamation with a nonzero reserve to drain evictions. + /// A zero reserve is a no-op, not a drain request. pub fn reclaim_scored( &self, entries: &impl SegmentEntries, @@ -784,6 +789,10 @@ impl SegmentClock { /// Reclaim a reserve with bounded visits and mapping removals, without compaction. /// Busy leases keep segments Evicting until a later sweep. A zero reserve is a /// no-op; a zero mapping budget can recycle empty but not populated segments. + /// For a nonzero reserve, success requires both the reserve (capped at the slot + /// count) and no pending evictions. Busy is intentional even if the reserve is + /// met while evictions remain. Use [`Segments::free_count`] to check capacity; + /// keep retrying bounded reclamation with a nonzero reserve to drain evictions. pub fn reclaim( &self, entries: &impl SegmentEntries, diff --git a/designs/racer-page-alloc.md b/designs/racer-page-alloc.md index 3e7bc976a..e9c72bad0 100644 --- a/designs/racer-page-alloc.md +++ b/designs/racer-page-alloc.md @@ -99,9 +99,13 @@ any segment that was open before the restart. (`alloc/src/segments.rs:785-842`). The caller calls `mark_read` when it serves a read from a segment. The sweep clears that mark once before it picks the segment. Each call has limits on -the number of segments it visits and the number of index entries it removes. If it -cannot free enough space within those limits, it returns `Busy` and the caller -tries again later. +the number of segments it visits and the number of index entries it removes. +For a nonzero reserve, `reclaim` and `reclaim_scored` succeed only when the reserve +(capped at the slot count) is met and no evictions remain. They intentionally +return `Busy` while evictions remain, even if enough slots are already free. +Callers should use `Segments::free_count` to check capacity and keep retrying +bounded reclamation later to drain pending evictions. A zero reserve is a no-op, +not a drain request. Evicting a segment happens in this order: From 37623edc1fa35b2bb68b5276b0a4a55df0f17904 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 21:51:39 +0000 Subject: [PATCH 57/82] docs(alloc): state trusted caller write contract --- cmd/racer-dataplane/alloc/src/lib.rs | 5 ++++- cmd/racer-dataplane/alloc/src/segments.rs | 3 +++ cmd/racer-dataplane/alloc/src/slab.rs | 4 ++++ designs/racer-page-alloc.md | 6 ++++++ 4 files changed, 17 insertions(+), 1 deletion(-) diff --git a/cmd/racer-dataplane/alloc/src/lib.rs b/cmd/racer-dataplane/alloc/src/lib.rs index 879951544..0799bcd32 100644 --- a/cmd/racer-dataplane/alloc/src/lib.rs +++ b/cmd/racer-dataplane/alloc/src/lib.rs @@ -72,11 +72,14 @@ //! //! For writes, round the logical size, append the padded length, allocate matching //! storage, copy bytes into the zeroed buffer, and submit [`Slab::write`]. Publish -//! the caller's mapping only after handling completion. Append reserves space and +//! the caller's mapping only after the write succeeds. Append reserves space and //! does not roll it back on failed writes. For reads, validate the stored slot, //! generation, and extent, acquire a lease, allocate storage, and submit a read. //! A lease authorizes the segment's used prefix at acquisition, not just the last //! appended record; it cannot authorize bytes appended afterward. +//! Any lease, including one from [`Segments::lease`], permits writes in that prefix. +//! Leases do not grant exclusive record ownership. The trusted caller must write +//! only reserved extents it owns and never overwrite published or readable records. //! //! Submission checks table identity, alignment, length, segment boundaries, and //! captured used bytes. Reads and writes reject short completions. Accepted I/O diff --git a/cmd/racer-dataplane/alloc/src/segments.rs b/cmd/racer-dataplane/alloc/src/segments.rs index ee35e717c..12fcf310a 100644 --- a/cmd/racer-dataplane/alloc/src/segments.rs +++ b/cmd/racer-dataplane/alloc/src/segments.rs @@ -52,6 +52,9 @@ pub struct SegmentSnapshot { /// Unique worker-local lease that prevents reuse until its completion owner drops. /// It authorizes the used prefix at acquisition, not just the latest append. +/// Any lease permits both reads and writes in that prefix; it is not exclusive +/// record ownership. The trusted caller must follow [`crate::Slab::write`]'s +/// ownership and publication rules. /// /// Lease counts cannot be duplicated by cloning a completion capability: /// diff --git a/cmd/racer-dataplane/alloc/src/slab.rs b/cmd/racer-dataplane/alloc/src/slab.rs index e2e52779b..eb801eda3 100644 --- a/cmd/racer-dataplane/alloc/src/slab.rs +++ b/cmd/racer-dataplane/alloc/src/slab.rs @@ -578,6 +578,10 @@ impl Slab { /// Write exactly one checked extent with completion-owned accounting and lease. /// Failed writes do not roll back the space reserved by append. + /// Any lease permits writes within its captured prefix, including a lease from + /// [`Segments::lease`]; it does not grant exclusive record ownership. The trusted + /// caller must write only reserved extents it owns, never overwrite published or + /// readable records, and publish a mapping only after the write succeeds. pub fn write<'a, S: Scope, B: Budget>( &'a self, reactor: &'a Reactor, diff --git a/designs/racer-page-alloc.md b/designs/racer-page-alloc.md index e9c72bad0..129564455 100644 --- a/designs/racer-page-alloc.md +++ b/designs/racer-page-alloc.md @@ -157,6 +157,12 @@ until the kernel reports completion, even if the caller drops the future or cancels. So the memory and the segment both stay reserved until the kernel is done with them. Short reads and writes are returned as errors. +Any lease, including one from `Segments::lease`, allows reads and writes within +its captured used prefix, not just the latest append. It does not grant exclusive +record ownership. The trusted caller must write only reserved extents it owns, +never overwrite published or readable records, and publish a mapping only after +the write succeeds (see `Slab::write` in `alloc/src/slab.rs`). + `fence_writes` waits until no write is in flight. It is a count, not a snapshot, and it does not flush to disk. From 50363e23172c3e12bce5a8cbe6c3fbf97f801332 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 21:52:23 +0000 Subject: [PATCH 58/82] docs(alloc): distinguish leases from initialized records --- cmd/racer-dataplane/alloc/src/lib.rs | 4 ++++ cmd/racer-dataplane/alloc/src/segments.rs | 9 ++++++++- cmd/racer-dataplane/alloc/src/slab.rs | 4 ++++ designs/racer-page-alloc.md | 12 ++++++++++-- 4 files changed, 26 insertions(+), 3 deletions(-) diff --git a/cmd/racer-dataplane/alloc/src/lib.rs b/cmd/racer-dataplane/alloc/src/lib.rs index 0799bcd32..096451066 100644 --- a/cmd/racer-dataplane/alloc/src/lib.rs +++ b/cmd/racer-dataplane/alloc/src/lib.rs @@ -80,6 +80,10 @@ //! Any lease, including one from [`Segments::lease`], permits writes in that prefix. //! Leases do not grant exclusive record ownership. The trusted caller must write //! only reserved extents it owns and never overwrite published or readable records. +//! A lease does not prove initialization in its generation: append reserves space +//! without writing it. Recycled bytes may come from an earlier generation or a +//! different cache. Before exposing them as a valid record, the caller must check +//! record integrity, authentication, and cache identity, even after a successful read. //! //! Submission checks table identity, alignment, length, segment boundaries, and //! captured used bytes. Reads and writes reject short completions. Accepted I/O diff --git a/cmd/racer-dataplane/alloc/src/segments.rs b/cmd/racer-dataplane/alloc/src/segments.rs index 12fcf310a..2d3ede54c 100644 --- a/cmd/racer-dataplane/alloc/src/segments.rs +++ b/cmd/racer-dataplane/alloc/src/segments.rs @@ -55,6 +55,10 @@ pub struct SegmentSnapshot { /// Any lease permits both reads and writes in that prefix; it is not exclusive /// record ownership. The trusted caller must follow [`crate::Slab::write`]'s /// ownership and publication rules. +/// A lease does not prove that bytes were initialized in this generation. Append +/// only reserves space; recycled bytes may belong to an earlier generation or a +/// different cache. The caller must check record integrity, authentication, and +/// cache identity before exposing bytes as a valid record. /// /// Lease counts cannot be duplicated by cloning a completion capability: /// @@ -278,6 +282,7 @@ impl Segments { /// Reserve an aligned used range. Malformed requests and lease overflow leave /// state unchanged. A valid rollover without a free slot seals the open tail /// before returning Busy, allowing reclamation to make a retry possible. + /// Reservation does not write or initialize disk bytes; see [`SegmentLease`]. pub fn append(&self, length: usize) -> Result<(SegmentLease, Extent)> { if self.frozen.get() { return Err(Error::Busy); @@ -368,7 +373,9 @@ impl Segments { usize::try_from(id.0).map_err(|_| Error::Corrupt) } - /// Acquire the current used prefix only while the generation is readable. + /// Acquire the current used prefix only while the slot is Open or Sealed and + /// the generation matches. This checks allocation state, not whether bytes + /// were initialized in this generation or form valid records; see [`SegmentLease`]. pub fn lease(&self, id: SegmentId, generation: Generation) -> Result { let slots = self.slots.borrow(); let slot = slots.get(Self::position(id)?).ok_or(Error::Corrupt)?; diff --git a/cmd/racer-dataplane/alloc/src/slab.rs b/cmd/racer-dataplane/alloc/src/slab.rs index eb801eda3..3528c8a35 100644 --- a/cmd/racer-dataplane/alloc/src/slab.rs +++ b/cmd/racer-dataplane/alloc/src/slab.rs @@ -558,6 +558,10 @@ impl Slab { /// Read exactly one checked extent, retaining its buffer and lease until completion. /// Dropping the waiting future does not release kernel-owned resources. + /// Neither a lease nor a successful read proves initialization in this + /// generation. Append only reserves space; recycled bytes may come from an + /// earlier generation or a different cache. Before exposing a valid record, + /// the caller must check its integrity, authentication, and cache identity. pub fn read<'a, S: Scope, B: Budget>( &'a self, reactor: &'a Reactor, diff --git a/designs/racer-page-alloc.md b/designs/racer-page-alloc.md index 129564455..ce8021aa4 100644 --- a/designs/racer-page-alloc.md +++ b/designs/racer-page-alloc.md @@ -86,8 +86,16 @@ Free -> Open -> Sealed -> Evicting -> Free (generation + 1) generation, and how much of the segment was in use when it was taken. While any lease exists, the segment cannot go back to `Free`. - The caller stores `(segment, generation, extent)` in its own index. A lookup - with an old generation fails with `Stale`, so a reused segment can never be - read as if it still held the old page. + with an old generation fails with `Stale`. This rejects stale mappings, not + stale bytes read through a lease for the current generation. + +Append reserves space without writing or initializing disk bytes. Neither a +`SegmentLease` nor a successful read proves that bytes were initialized in the +current generation. Recycled storage may still hold bytes from an earlier +generation or a different cache. Before exposing bytes as a valid record, the +caller must check record integrity, authentication, and cache identity (see +`SegmentLease` and `Segments::append` in `alloc/src/segments.rs`, and `Slab::read` +in `alloc/src/slab.rs`). `freeze`, `snapshot`, and `restore` support restart. Restore checks the whole image before it changes anything, requires that no leases are live, and seals From 68e957c680ad19c0c82429b41ee1da57a3d93d28 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 21:52:52 +0000 Subject: [PATCH 59/82] docs(alloc): describe scored reclamation visit limit --- cmd/racer-dataplane/alloc/src/lib.rs | 8 +++++--- cmd/racer-dataplane/alloc/src/segments.rs | 6 ++++-- designs/racer-page-alloc.md | 6 ++++-- 3 files changed, 13 insertions(+), 7 deletions(-) diff --git a/cmd/racer-dataplane/alloc/src/lib.rs b/cmd/racer-dataplane/alloc/src/lib.rs index 096451066..86b465c6d 100644 --- a/cmd/racer-dataplane/alloc/src/lib.rs +++ b/cmd/racer-dataplane/alloc/src/lib.rs @@ -127,9 +127,11 @@ //! //! [`SegmentClock`] shares a cursor and second-chance set across bounded sweeps. //! Index reclamation forgets mappings without touching bytes or generations. -//! Physical reclamation only visits Sealed/Evicting slots and recycles empty slots -//! after leases drain. Each sweep visits at most two rotations, further capped by -//! the caller. For a nonzero reserve, [`SegmentClock::reclaim`] and +//! Physical reclamation selects only Sealed/Evicting slots and recycles empty slots +//! after leases drain. Index and unscored sweeps visit at most two rotations, +//! further capped by the caller. Scored reclamation visits at most +//! min(slot count, max_visits, 64) slots, including skipped slots, then ranks +//! eligible candidates. For a nonzero reserve, [`SegmentClock::reclaim`] and //! [`SegmentClock::reclaim_scored`] succeed only when the reserve (capped at the //! slot count) is met and no evictions remain. Busy is intentional even with enough //! free slots while pending evictions drain. Use [`Segments::free_count`] to check diff --git a/cmd/racer-dataplane/alloc/src/segments.rs b/cmd/racer-dataplane/alloc/src/segments.rs index 2d3ede54c..6a071d97d 100644 --- a/cmd/racer-dataplane/alloc/src/segments.rs +++ b/cmd/racer-dataplane/alloc/src/segments.rs @@ -707,8 +707,10 @@ impl SegmentClock { Err(Error::Busy) } - /// Rank at most 64 eligible slots before entering eviction. Scores are soft - /// preferences, not pins. The callback must itself bound mapping inspection. + /// Visit at most min(slot count, max_visits, 64) slots, then rank eligible + /// candidates before entering eviction. Skipped slots count toward this limit. + /// Scores are soft preferences, not pins. The callback must itself bound + /// mapping inspection. /// Existing eviction work sorts first so partial removals always make progress. /// Unlike the legacy clock, logical heat is supplied by the caller, so recent /// disk completions do not override value ranking. diff --git a/designs/racer-page-alloc.md b/designs/racer-page-alloc.md index ce8021aa4..45f5209b4 100644 --- a/designs/racer-page-alloc.md +++ b/designs/racer-page-alloc.md @@ -125,8 +125,10 @@ Evicting a segment happens in this order: mark the segment `Free`. The key rule is: remove the index entries first, then wait for in-flight I/O, -then reuse. `reclaim_scored` is a variant that ranks a small sample by a -caller-provided score instead of recent reads. `reclaim_index` drops index entries one at a time until a caller check +then reuse. `reclaim_scored` visits at most `min(slot count, max_visits, 64)` +slots, including skipped slots. It ranks eligible candidates in that sample by a +caller-provided score instead of recent reads. The limit is on visited slots, +not eligible candidates. `reclaim_index` drops index entries one at a time until a caller check (for example, an index size limit) passes. It does not free segments and does not ask `can_evict`. From 7257f69f861656e2eaeeb62a74365bd5675cf647 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 21:53:57 +0000 Subject: [PATCH 60/82] docs(alloc): replace stale design line references --- designs/racer-page-alloc.md | 64 +++++++++++++++++++++---------------- 1 file changed, 37 insertions(+), 27 deletions(-) diff --git a/designs/racer-page-alloc.md b/designs/racer-page-alloc.md index 45f5209b4..3f588954a 100644 --- a/designs/racer-page-alloc.md +++ b/designs/racer-page-alloc.md @@ -35,37 +35,42 @@ Non-goals: Live allocation and I/O authority is worker-local. `Segments` and `Slab` use `Rc`-owned state and are neither `Send` nor `Sync` -(`alloc/src/segments.rs:160-179`, `alloc/src/slab.rs:57-72`). Buffers, leases, -freeze guards, and the reclamation clock have the same restriction -(`alloc/src/lib.rs:545-550`, `alloc/src/segments.rs:62-76`, -`alloc/src/segments.rs:144`, `alloc/src/segments.rs:617-626`). +(see `Segments` in `alloc/src/segments.rs` and `Slab` in `alloc/src/slab.rs`). +Buffers, leases, freeze guards, and the reclamation clock have the same restriction +(see `AlignedBuffer` in `alloc/src/lib.rs` and `SegmentLease`, `FreezeGuard`, and +`SegmentClock` in `alloc/src/segments.rs`). Value types such as `Alignment` and `SegmentId`, and startup inputs such as -`DevicePlacement`, are `Send + Sync` (`alloc/src/lib.rs:313-319`, -`alloc/src/segments.rs:13-15`, `alloc/src/slab.rs:34-43`). Each worker owns its +`DevicePlacement`, are `Send + Sync` (see their declarations in `alloc/src/lib.rs`, +`alloc/src/segments.rs`, and `alloc/src/slab.rs`, respectively). Each worker owns its storage ranges, segment table, and buffer pool; live allocation and I/O authority cannot move to another worker. ## Buffers `AlignedBuffer` is a heap allocation from `alloc_zeroed` with the alignment that -the file needs (`alloc/src/lib.rs:415-442`). Lengths are padded to the least common -multiple of the offset and length units, so the next record also starts aligned. -A single buffer is at most 1 GiB. +the file needs (see `Alignment::allocate` in `alloc/src/lib.rs`). Lengths are +padded to the least common multiple of the offset and length units, so the next +record also starts aligned. +A single buffer is at most 1 GiB (see `Alignment::extent` and +`Alignment::MAX_TRANSFER_LENGTH` in `alloc/src/lib.rs`). Each buffer holds a caller-supplied `Charge`, so the caller can account for memory against its own budget. The slab keeps at most one idle buffer. After checking charge coverage, allocation reuses it only for an exact length match; otherwise it frees the idle buffer and allocates new storage -(`alloc/src/slab.rs:566-580`). On drop, a buffer fills the idle slot only if the -pool still exists and the slot is empty and can be mutably borrowed; otherwise -its storage is freed (`alloc/src/lib.rs:630-650`, `alloc/src/lib.rs:522-529`). +(see `Slab::allocate` in `alloc/src/slab.rs`). On drop, a buffer fills the idle slot +only if the pool still exists and the slot is empty and can be mutably borrowed; otherwise +its storage is freed (see the `Drop` implementations for `AlignedBuffer` and +`Allocation` in `alloc/src/lib.rs`). The retained size depends on return order, not necessarily the last size used. This is not a general size-class pool. Each buffer tracks whether it is still all zeros. Any mutable access, including handing it to the kernel for a read, marks it dirty. On drop, a dirty buffer is wiped in full, including padding, with `explicit_bzero` (or `zeroize` where that -is not available) before it is pooled or freed (`alloc/src/lib.rs:483-520`). +is not available) before it is pooled or freed (see `Allocation::as_mut_slice`, +`Allocation::wipe`, and the `IoBuffer` implementation for `AlignedBuffer` in +`alloc/src/lib.rs`). Clean buffers skip the wipe. This keeps old page data from leaking into the next request without paying for a wipe on every allocation. @@ -103,11 +108,11 @@ any segment that was open before the restart. ## Reclaim -`SegmentClock` is a bounded clock (second-chance) sweep over sealed segments -(`alloc/src/segments.rs:785-842`). The caller calls `mark_read` when it serves -a read from a segment. The sweep clears that mark once before it picks the -segment. Each call has limits on -the number of segments it visits and the number of index entries it removes. +`SegmentClock::reclaim` in `alloc/src/segments.rs` is a bounded clock +(second-chance) sweep over sealed segments. The caller calls `mark_read` when it +serves a read from a segment. The sweep clears that mark once before it picks the +segment. Each call has limits on the number of segments it visits and the number +of index entries it removes. For a nonzero reserve, `reclaim` and `reclaim_scored` succeed only when the reserve (capped at the slot count) is met and no evictions remain. They intentionally return `Busy` while evictions remain, even if enough slots are already free. @@ -128,14 +133,15 @@ The key rule is: remove the index entries first, then wait for in-flight I/O, then reuse. `reclaim_scored` visits at most `min(slot count, max_visits, 64)` slots, including skipped slots. It ranks eligible candidates in that sample by a caller-provided score instead of recent reads. The limit is on visited slots, -not eligible candidates. `reclaim_index` drops index entries one at a time until a caller check -(for example, an index size limit) passes. It does not free segments and does not -ask `can_evict`. +not eligible candidates. `reclaim_index` drops index entries one at a time until a +caller check (for example, an index size limit) passes. It does not free segments +and does not ask `can_evict`. ## Storage and I/O `Slab::new` describes one cache file. Opening is blocking and is meant to run at -startup (`alloc/src/slab.rs:368-409`). For this file-backed mode, it: +startup (see `Slab::open_configured` and `Slab::open_file` in `alloc/src/slab.rs`). +For this file-backed mode, it: - Walks the path without following symlinks or `..`. - Requires a regular file owned by the current user, mode 0600, one hard link. @@ -144,6 +150,9 @@ startup (`alloc/src/slab.rs:368-409`). For this file-backed mode, it: - Sizes an empty file sparsely to capacity. A file with the wrong size is rejected, not truncated. +See `Slab::open_file`, `Slab::validate_layout`, `open_private_file`, +`validate_file`, and `probe_fd` in `alloc/src/slab.rs` for these checks. + `Slab::from_devices` instead owns caller-opened file or block-device placements, one per logical segment, without creating, sizing, or locking them. Files must be read/write with `O_DIRECT` and without `O_APPEND`. The constructor checks @@ -156,13 +165,14 @@ and sizes unchanged while in use (see `Slab::from_devices` in `alloc/src/slab.rs For device placements, `open_configured` duplicates the owned files into worker-local descriptors and releases the original placement references -(`alloc/src/slab.rs:295-325`). In both modes, it then binds the slab to one -segment table. I/O is refused until binding succeeds, and a slab cannot be -rebound to a different table (`alloc/src/slab.rs:241-276`, -`alloc/src/slab.rs:447-449`). +(see `Slab::open_file` in `alloc/src/slab.rs`). In both modes, it then binds the +slab to one segment table. I/O is refused until binding succeeds, and a slab cannot be +rebound to a different table (see `Slab::bind` and `Slab::submission` in +`alloc/src/slab.rs`). `read` and `write` check the extent, alignment, and lease, then pass the buffer -and lease to the reactor (`alloc/src/slab.rs:461-515`). The reactor holds both +and lease to the reactor (see `Slab::submission`, `Submission::read`, and +`Submission::write` in `alloc/src/slab.rs`). The reactor holds both until the kernel reports completion, even if the caller drops the future or cancels. So the memory and the segment both stay reserved until the kernel is done with them. Short reads and writes are returned as errors. From 253358a1609e867317f0a3ec509c4f90fbdf32a9 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 21:54:11 +0000 Subject: [PATCH 61/82] fix(flow): track peer idle age separately from recovery --- cmd/racer-dataplane/flow/src/admission.rs | 170 +++++++++++++++++++++- 1 file changed, 165 insertions(+), 5 deletions(-) diff --git a/cmd/racer-dataplane/flow/src/admission.rs b/cmd/racer-dataplane/flow/src/admission.rs index 514af843f..08f0751d2 100644 --- a/cmd/racer-dataplane/flow/src/admission.rs +++ b/cmd/racer-dataplane/flow/src/admission.rs @@ -147,6 +147,9 @@ struct Peer { probe: bool, updated: Instant, + + /// Start of the current idle interval, independent of recovery throttling. + idle_since: Option, } /// An admitted operation; the final shared owner releases capacity. @@ -225,7 +228,9 @@ impl Adaptive { p.active == 0 && !p.probe && p.retry.is_none_or(|retry| retry.elapsed(now)) - && now.saturating_duration_since(p.updated) >= self.config.retire_after + && p.idle_since.is_some_and(|since| { + now.saturating_duration_since(since) >= self.config.retire_after + }) }) .next(); if retired.is_none() { @@ -240,6 +245,7 @@ impl Adaptive { retry: None, probe: false, updated: now, + idle_since: None, }); if peer.probe || peer.retry.is_some_and(|at| !at.elapsed(now)) { self.observer.event(Event::CircuitRejected); @@ -252,6 +258,7 @@ impl Adaptive { let probe = peer.retry.is_some(); peer.probe = probe; peer.active += 1; + peer.idle_since = None; let generation = peer.generation; state.active += 1; self.observer.active(state.active); @@ -332,7 +339,7 @@ impl Permit { impl Drop for Permit { /// Release final ownership and renew backoff for an unsuccessful probe. fn drop(&mut self) { - let now = self.probe.then(|| (self.owner.now)()); + let now = (self.owner.now)(); let Ok(mut state) = self.owner.state.lock() else { return; }; @@ -341,10 +348,12 @@ impl Drop for Permit { .get_mut(&self.key) .expect("live permit retains key"); peer.active -= 1; + if peer.active == 0 { + peer.idle_since = Some(now); + } if self.probe { peer.probe = false; if peer.retry.is_some() { - let now = now.expect("probe clock sampled before locking"); peer.retry = Some(Deadline::after(now, self.owner.config.backoff)); peer.updated = now; } @@ -1675,6 +1684,157 @@ mod tests { assert_eq!(*owner.observer.limit.lock().unwrap(), config.total / 2); } + /// Idle age and recovery use independent clocks without wall-clock sleeps. + mod idle_retirement { + use super::*; + use std::cell::Cell; + + thread_local! { + static CLOCK: Cell = Cell::new(Instant::now()); + } + + fn now() -> Instant { + CLOCK.get() + } + + fn advance(duration: Duration) { + CLOCK.set(now() + duration); + } + + #[test] + fn waits_from_last_owner_and_restarts_after_reuse() { + for outcome in [Outcome::Neutral, Outcome::Verified] { + let config = Config { + capacity: 1, + recovery: Duration::MAX, + ..config() + }; + let owner = Adaptive::new(config, Counts::default(), now).unwrap(); + let first = owner.acquire(&1).unwrap(); + let last = owner.acquire(&1).unwrap(); + let fence = last.clone(); + advance(config.retire_after); + first.observe(outcome); + drop(first); + advance(config.retire_after); + drop(last); + assert_eq!(owner.state.lock().unwrap().peers[&1].active, 1); + assert!(matches!(owner.acquire(&2), Err(Error::Overloaded))); + drop(fence); + assert!(matches!(owner.acquire(&2), Err(Error::Overloaded))); + + advance(config.retire_after - Duration::from_nanos(1)); + assert!(matches!(owner.acquire(&2), Err(Error::Overloaded))); + let reused = owner.acquire(&1).unwrap(); + advance(config.retire_after); + assert!(matches!(owner.acquire(&2), Err(Error::Overloaded))); + drop(reused); + assert!(matches!(owner.acquire(&2), Err(Error::Overloaded))); + advance(config.retire_after - Duration::from_nanos(1)); + assert!(matches!(owner.acquire(&2), Err(Error::Overloaded))); + advance(Duration::from_nanos(1)); + let replacement = owner.acquire(&2).unwrap(); + assert!(!owner.state.lock().unwrap().peers.contains_key(&1)); + drop(replacement); + assert_eq!(*owner.observer.active.lock().unwrap(), 0); + } + } + + #[test] + fn probe_release_requires_idle_age_and_expired_backoff() { + for outcome in [Outcome::Neutral, Outcome::Verified, Outcome::PeerFailure] { + let config = Config { + capacity: 1, + backoff: Duration::from_secs(3), + retire_after: Duration::from_secs(2), + ..config() + }; + let owner = Adaptive::new(config, Counts::default(), now).unwrap(); + let failed = owner.acquire(&1).unwrap(); + failed.observe(Outcome::PeerFailure); + drop(failed); + advance(config.backoff); + let probe = owner.acquire(&1).unwrap(); + probe.observe(outcome); + advance(config.backoff + config.retire_after); + assert!(matches!(owner.acquire(&2), Err(Error::Overloaded))); + drop(probe); + assert!(matches!(owner.acquire(&2), Err(Error::Overloaded))); + advance(config.retire_after - Duration::from_nanos(1)); + assert!(matches!(owner.acquire(&2), Err(Error::Overloaded))); + advance(Duration::from_nanos(1)); + if outcome != Outcome::Verified { + assert!(matches!(owner.acquire(&2), Err(Error::Overloaded))); + assert!(matches!(owner.acquire(&1), Err(Error::Unavailable))); + advance(config.backoff - config.retire_after); + } + assert!(owner.acquire(&2).is_ok()); + } + } + + #[test] + fn acquisition_and_drop_preserve_recovery_throttle() { + let config = Config { + backoff: Duration::ZERO, + recovery: Duration::from_secs(10), + ..config() + }; + let owner = Adaptive::new(config, Counts::default(), now).unwrap(); + let failed = owner.acquire(&1).unwrap(); + failed.observe(Outcome::PeerFailure); + drop(failed); + let updated = now(); + advance(config.recovery / 2); + let probe = owner.acquire(&1).unwrap(); + probe.observe(Outcome::Verified); + drop(probe); + assert_eq!(owner.state.lock().unwrap().peers[&1].updated, updated); + assert_eq!(owner.state.lock().unwrap().peers[&1].limit, 2); + advance(config.recovery / 2); + let work = owner.acquire(&1).unwrap(); + work.observe(Outcome::Verified); + drop(work); + assert_eq!(owner.state.lock().unwrap().peers[&1].limit, 3); + let updated = now(); + advance(config.recovery - Duration::from_nanos(1)); + let early = owner.acquire(&1).unwrap(); + early.observe(Outcome::Verified); + drop(early); + assert_eq!(owner.state.lock().unwrap().peers[&1].updated, updated); + assert_eq!(owner.state.lock().unwrap().peers[&1].limit, 3); + advance(Duration::from_nanos(1)); + let ready = owner.acquire(&1).unwrap(); + ready.observe(Outcome::Verified); + assert_eq!(owner.state.lock().unwrap().peers[&1].limit, 4); + } + + #[test] + fn zero_and_maximum_idle_age_preserve_live_ownership() { + for retire_after in [Duration::ZERO, Duration::MAX] { + let owner = Adaptive::new( + Config { + capacity: 1, + retire_after, + ..config() + }, + Counts::default(), + now, + ) + .unwrap(); + let work = owner.acquire(&1).unwrap(); + advance(Duration::from_secs(120)); + assert!(matches!(owner.acquire(&2), Err(Error::Overloaded))); + drop(work); + if retire_after.is_zero() { + assert!(owner.acquire(&2).is_ok()); + } else { + assert!(matches!(owner.acquire(&2), Err(Error::Overloaded))); + assert!(owner.acquire(&1).is_ok()); + } + } + } + } + /// Only sufficiently old, idle, eligible peer records may be retired. #[test] fn capacity_preserves_live_work_and_stale_backoff_then_retires_idle() { @@ -1691,7 +1851,7 @@ mod tests { .peers .get_mut(&1) .unwrap() - .updated = Instant::now() - Duration::from_secs(61); + .idle_since = Some(Instant::now() - Duration::from_secs(61)); assert!(matches!(owner.acquire(&3), Err(Error::Overloaded))); assert!(owner.state.lock().unwrap().peers.contains_key(&1)); owner.state.lock().unwrap().peers.get_mut(&1).unwrap().retry = @@ -1708,7 +1868,7 @@ mod tests { .peers .get_mut(&2) .unwrap() - .updated = Instant::now() - Duration::from_secs(61); + .idle_since = Some(Instant::now() - Duration::from_secs(61)); assert!(owner.acquire(&3).is_ok()); assert_eq!(owner.state.lock().unwrap().peers.len(), 2); } From 253af458be7275842b92e4dae249f8d8439c48b3 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 21:55:03 +0000 Subject: [PATCH 62/82] fix(flow): release admission and wake waiters on handoff close --- cmd/racer-dataplane/flow/src/admission.rs | 176 +++++++++++++++++++++- 1 file changed, 173 insertions(+), 3 deletions(-) diff --git a/cmd/racer-dataplane/flow/src/admission.rs b/cmd/racer-dataplane/flow/src/admission.rs index 08f0751d2..9ee7f8f4f 100644 --- a/cmd/racer-dataplane/flow/src/admission.rs +++ b/cmd/racer-dataplane/flow/src/admission.rs @@ -988,6 +988,7 @@ mod handoff { } /// Pop at most the caller's budget while retaining each item's admission. + /// Closed targets reject pops without retaining the caller's waker. pub fn pop_batch( &self, key: &K, @@ -996,6 +997,9 @@ mod handoff { ) -> Result> { let mut state = self.0.lock().map_err(|_| Error::Unavailable)?; let target = state.target(key).ok_or(Error::InvalidInput)?; + if target.closed { + return Err(Error::Unavailable); + } if let Some(old) = &mut target.waker { old.clone_from(waker); } else { @@ -1010,17 +1014,25 @@ mod handoff { })) } - /// Close a target and drop queued ownership outside the shared lock. + /// Close permanently, then drop ownership and wake outside the shared lock. pub fn close(&self, key: &K) { - let queued = { + let (queued, admission, waker) = { let mut state = self.0.lock().unwrap_or_else(|e| e.into_inner()); let Some(target) = state.target(key) else { return; }; target.closed = true; - std::mem::take(&mut target.queue) + ( + std::mem::take(&mut target.queue), + target.admission.take(), + target.waker.take(), + ) }; drop(queued); + drop(admission); + if let Some(waker) = waker { + waker.wake(); + } } } @@ -1049,6 +1061,164 @@ mod handoff { #[cfg(test)] mod tests { use super::*; + use std::{ + sync::{ + Weak, + atomic::{AtomicUsize, Ordering}, + }, + task::Wake, + }; + + type ClosingHandoff = Handoff; + + #[derive(Clone)] + struct CloseDrop { + handoff: Weak, + drops: Arc, + } + + impl CloseDrop { + fn reenter(&self) { + if let Some(handoff) = self.handoff.upgrade() { + let state = handoff + .0 + .try_lock() + .expect("close must unlock before callbacks"); + assert!(state.targets[0].1.closed); + assert!(state.targets[0].1.queue.is_empty()); + assert!(state.targets[0].1.admission.is_none()); + assert!(state.targets[0].1.waker.is_none()); + drop(state); + handoff.close(&1); + assert!(matches!( + handoff.pop_batch::<1>(&1, Waker::noop(), 1), + Err(Error::Unavailable) + )); + } + } + } + + impl Drop for CloseDrop { + fn drop(&mut self) { + self.reenter(); + self.drops.fetch_add(1, Ordering::SeqCst); + } + } + + struct CloseAdmission(CloseDrop); + + impl Admission for CloseAdmission { + type Reservation = CloseDrop; + + fn register(&self, _: &Waker) {} + + fn reserve(&self) -> Result { + Ok(self.0.clone()) + } + } + + struct CloseWake { + on_drop: CloseDrop, + wakes: Arc, + } + + impl Wake for CloseWake { + fn wake(self: Arc) { + self.on_drop.reenter(); + self.wakes.fetch_add(1, Ordering::SeqCst); + } + } + + #[test] + fn close_releases_admission_and_queue_outside_lock() { + for queued in [0, 2] { + let handoff = Arc::new(ClosingHandoff::new(&[1])); + let drops = Arc::new(AtomicUsize::new(0)); + handoff + .install( + &1, + CloseAdmission(CloseDrop { + handoff: Arc::downgrade(&handoff), + drops: drops.clone(), + }), + ) + .unwrap(); + for _ in 0..queued { + handoff + .reserve(Waker::noop()) + .unwrap() + .deliver(|| CloseDrop { + handoff: Arc::downgrade(&handoff), + drops: drops.clone(), + }) + .unwrap(); + } + handoff.close(&1); + assert_eq!(drops.load(Ordering::SeqCst), 1 + 2 * queued); + handoff.close(&1); + handoff.close(&2); + assert_eq!(drops.load(Ordering::SeqCst), 1 + 2 * queued); + assert!(handoff.0.lock().unwrap().targets[0].1.admission.is_none()); + } + } + + #[test] + fn close_wakes_and_releases_waiter_outside_lock() { + let handoff = Arc::new(ClosingHandoff::new(&[1])); + let drops = Arc::new(AtomicUsize::new(0)); + let wakes = Arc::new(AtomicUsize::new(0)); + let waker = Waker::from(Arc::new(CloseWake { + on_drop: CloseDrop { + handoff: Arc::downgrade(&handoff), + drops: drops.clone(), + }, + wakes: wakes.clone(), + })); + assert!(handoff.pop_batch::<1>(&1, &waker, 1).unwrap()[0].is_none()); + drop(waker); + handoff.close(&2); + assert_eq!(wakes.load(Ordering::SeqCst), 0); + handoff.close(&1); + assert_eq!(wakes.load(Ordering::SeqCst), 1); + assert_eq!(drops.load(Ordering::SeqCst), 1); + handoff.close(&1); + assert_eq!(wakes.load(Ordering::SeqCst), 1); + assert!(handoff.0.lock().unwrap().targets[0].1.waker.is_none()); + } + + #[test] + fn close_rejects_pop_without_retaining_new_waiter() { + let handoff = Handoff::<_, Ready, ()>::new(&[1, 2]); + handoff.install(&1, Ready).unwrap(); + for key in [1, 2] { + handoff.close(&key); + for budget in [0, 1, usize::MAX] { + assert!(matches!( + handoff.pop_batch::<1>(&key, Waker::noop(), budget), + Err(Error::Unavailable) + )); + assert!(matches!( + handoff.pop_batch::<0>(&key, Waker::noop(), budget), + Err(Error::Unavailable) + )); + assert!( + handoff + .0 + .lock() + .unwrap() + .target(&key) + .unwrap() + .waker + .is_none() + ); + } + assert_eq!(handoff.install(&key, Ready), Err(Error::InvalidInput)); + } + assert!(matches!( + handoff.pop_batch::<1>(&3, Waker::noop(), 1), + Err(Error::InvalidInput) + )); + } struct Ready; From 27b820252144d6ac6ffa1e92bb3855b07a009465 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 21:56:35 +0000 Subject: [PATCH 63/82] test(alloc): derive startup geometry from probed alignment --- cmd/racer-dataplane/alloc/src/slab.rs | 43 ++++++++++++++++++++++----- 1 file changed, 35 insertions(+), 8 deletions(-) diff --git a/cmd/racer-dataplane/alloc/src/slab.rs b/cmd/racer-dataplane/alloc/src/slab.rs index 3528c8a35..d1b48eb14 100644 --- a/cmd/racer-dataplane/alloc/src/slab.rs +++ b/cmd/racer-dataplane/alloc/src/slab.rs @@ -1747,13 +1747,29 @@ mod tests { } assert!(!directory.0.join("missing").exists()); assert_eq!(slab.geometry(), Err(Error::Unavailable)); + } - let retry = Slab::<()>::new(path.clone(), 8192, 4096, 512); + let Some(alignment) = real_alignment(probe(&file)) else { + return; + }; + let segment_bytes = alignment.extent(0, 512).unwrap().length() as u64; + let capacity = segment_bytes.checked_mul(crate::MAX_SEGMENTS + 1).unwrap(); + let retry_capacity = segment_bytes.checked_mul(2).unwrap(); + for path in &paths { + let segments = Segments::new(segment_bytes); + let slab = Slab::<()>::new(path.clone(), capacity, segment_bytes, 512); + assert_eq!( + slab.open_configured(&segments), + Err(Error::InvalidConfiguration) + ); + assert!(!segments.is_configured()); + assert_eq!(slab.geometry(), Err(Error::Unavailable)); + let retry = Slab::<()>::new(path.clone(), retry_capacity, segment_bytes, 512); if real_alignment(retry.open_configured(&segments)).is_none() { return; } assert_eq!(segments.count(), 2); - assert_eq!(std::fs::metadata(path).unwrap().len(), 8192); + assert_eq!(std::fs::metadata(path).unwrap().len(), retry_capacity); drop(retry); std::fs::remove_file(path).unwrap(); if path.parent() != Some(directory.0.as_path()) { @@ -1793,17 +1809,28 @@ mod tests { /// The slot limit does not cap physical backing for an already configured table. #[test] fn open_configured_accepts_partial_table_above_physical_segment_limit() { + use std::os::unix::fs::OpenOptionsExt; + let directory = Directory::new(); - let probe = Slab::<()>::new(directory.0.join("probe"), 4096, 4096, 512); - let Some(alignment) = real_alignment(probe.open_now()) else { + let file = std::fs::OpenOptions::new() + .write(true) + .create_new(true) + .mode(0o600) + .open(directory.0.join("probe")) + .unwrap(); + let Some(alignment) = real_alignment(probe(&file)) else { return; }; - let capacity = 4096 * (crate::MAX_SEGMENTS + 1); - let segments = Segments::new(4096); + let segment_bytes = alignment.extent(0, 512).unwrap().length() as u64; + let capacity = segment_bytes.checked_mul(crate::MAX_SEGMENTS + 1).unwrap(); + let segments = Segments::new(segment_bytes); segments.configure(capacity, 2, alignment).unwrap(); let path = directory.0.join("partial"); - let slab = Slab::<()>::new(path.clone(), capacity, 4096, 512); - assert_eq!(slab.open_configured(&segments), Ok(alignment)); + let slab = Slab::<()>::new(path.clone(), capacity, segment_bytes, 512); + let Some(opened_alignment) = real_alignment(slab.open_configured(&segments)) else { + return; + }; + assert_eq!(opened_alignment, alignment); assert_eq!(segments.count(), 2); assert_eq!(segments.capacity_bytes(), capacity); assert_eq!( From 968dff0f17638461801a9003109ace229d8e56a8 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 22:09:24 +0000 Subject: [PATCH 64/82] fix(flow): fence exhausted adaptive generations --- cmd/racer-dataplane/flow/src/admission.rs | 102 ++++++++++++++++++++-- 1 file changed, 93 insertions(+), 9 deletions(-) diff --git a/cmd/racer-dataplane/flow/src/admission.rs b/cmd/racer-dataplane/flow/src/admission.rs index 9ee7f8f4f..7d8aec49d 100644 --- a/cmd/racer-dataplane/flow/src/admission.rs +++ b/cmd/racer-dataplane/flow/src/admission.rs @@ -140,7 +140,8 @@ struct Peer { limit: usize, - generation: u64, + /// Exhaustion blocks peer evidence and admission until all permits drain. + generation: Option, retry: Option, @@ -205,9 +206,9 @@ impl Adaptive { pub fn available(&self, key: &K) -> bool { let now = (self.now)(); self.state.lock().is_ok_and(|s| { - s.peers - .get(key) - .is_none_or(|p| !p.probe && p.retry.is_none_or(|at| at.elapsed(now))) + s.peers.get(key).is_none_or(|p| { + p.generation.is_some() && !p.probe && p.retry.is_none_or(|at| at.elapsed(now)) + }) }) } @@ -241,13 +242,14 @@ impl Adaptive { let peer = state.peers.entry(indexed).or_insert(Peer { active: 0, limit: self.config.per_key, - generation: 0, + generation: Some(0), retry: None, probe: false, updated: now, idle_since: None, }); - if peer.probe || peer.retry.is_some_and(|at| !at.elapsed(now)) { + if peer.generation.is_none() || peer.probe || peer.retry.is_some_and(|at| !at.elapsed(now)) + { self.observer.event(Event::CircuitRejected); return Err(Error::Unavailable); } @@ -259,7 +261,9 @@ impl Adaptive { peer.probe = probe; peer.active += 1; peer.idle_since = None; - let generation = peer.generation; + let generation = peer + .generation + .expect("admission rejects exhausted generations"); state.active += 1; self.observer.active(state.active); self.observer.event(Event::Accepted); @@ -311,13 +315,13 @@ impl Permit { Outcome::Neutral => return, }; self.owner.observer.event(event); - if peer.generation != self.generation { + if peer.generation != Some(self.generation) { return; } match outcome { Outcome::PeerFailure => { peer.limit = (peer.limit / 2).max(1); - peer.generation = peer.generation.saturating_add(1); + peer.generation = self.generation.checked_add(1); peer.retry = Some(Deadline::after(now, config.backoff)); peer.updated = now; } @@ -349,6 +353,8 @@ impl Drop for Permit { .expect("live permit retains key"); peer.active -= 1; if peer.active == 0 { + // No live permit can carry a reused generation after this point. + peer.generation.get_or_insert(0); peer.idle_since = Some(now); } if self.probe { @@ -1784,6 +1790,84 @@ mod tests { assert_eq!(*owner.observer.limit.lock().unwrap(), 3); } + /// Failed probes cannot clear retry, even at the last generation. + #[test] + fn generation_exhaustion_fences_failed_probe_success() { + for generation in [0, u64::MAX - 1, u64::MAX] { + let owner = Adaptive::new(config(), Counts::default(), Instant::now).unwrap(); + drop(owner.acquire(&1).unwrap()); + { + let mut state = owner.state.lock().unwrap(); + let peer = state.peers.get_mut(&1).unwrap(); + peer.generation = Some(generation); + peer.retry = Some(Deadline::At(Instant::now())); + } + let probe = owner.acquire(&1).unwrap(); + let fence = probe.clone(); + probe.observe(Outcome::PeerFailure); + probe.observe(Outcome::Verified); + assert!(owner.state.lock().unwrap().peers[&1].retry.is_some()); + assert!(!owner.available(&1)); + drop(probe); + assert_eq!(*owner.observer.active.lock().unwrap(), 1); + assert!(matches!(owner.acquire(&1), Err(Error::Unavailable))); + drop(fence); + assert_eq!(*owner.observer.active.lock().unwrap(), 0); + assert!(matches!(owner.acquire(&1), Err(Error::Unavailable))); + owner.state.lock().unwrap().peers.get_mut(&1).unwrap().retry = + Some(Deadline::At(Instant::now())); + let recovered = owner.acquire(&1).unwrap(); + recovered.observe(Outcome::Verified); + assert!(!owner.available(&1)); + drop(recovered); + assert!(owner.available(&1)); + } + } + + /// Exhaustion fences every old permit and resets only after final release. + #[test] + fn generation_exhaustion_waits_for_live_permits_before_reuse() { + let owner = Adaptive::new( + Config { + backoff: Duration::ZERO, + recovery: Duration::ZERO, + ..config() + }, + Counts::default(), + Instant::now, + ) + .unwrap(); + let old = owner.acquire(&1).unwrap(); + owner + .state + .lock() + .unwrap() + .peers + .get_mut(&1) + .unwrap() + .generation = Some(u64::MAX); + let failed = owner.acquire(&1).unwrap(); + let fence = failed.clone(); + failed.observe(Outcome::PeerFailure); + failed.observe(Outcome::PeerFailure); + assert_eq!(owner.state.lock().unwrap().peers[&1].limit, 2); + assert!(!owner.available(&1)); + assert!(!owner.hedge_available(&1)); + drop(failed); + drop(fence); + assert_eq!(*owner.observer.active.lock().unwrap(), 1); + assert!(matches!(owner.acquire(&1), Err(Error::Unavailable))); + assert!(owner.acquire(&2).is_ok()); + drop(old); + assert_eq!(*owner.observer.active.lock().unwrap(), 0); + assert!(owner.available(&1)); + let recovered = owner.acquire(&1).unwrap(); + recovered.observe(Outcome::Verified); + drop(recovered); + assert!(owner.available(&1)); + assert_eq!(owner.state.lock().unwrap().peers[&1].limit, 3); + } + /// Success at the ceiling must not delay an eligible pressure reduction. #[test] fn recovery_at_ceiling_preserves_pressure_eligibility() { From 4f3fb3791b618f794fe28b2ad062c1fe99483585 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 22:06:26 +0000 Subject: [PATCH 65/82] fix(flow): use state-neutral unavailable diagnostic --- cmd/racer-dataplane/flow/src/lib.rs | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/cmd/racer-dataplane/flow/src/lib.rs b/cmd/racer-dataplane/flow/src/lib.rs index 1aa6e056c..2b09fc871 100644 --- a/cmd/racer-dataplane/flow/src/lib.rs +++ b/cmd/racer-dataplane/flow/src/lib.rs @@ -75,7 +75,7 @@ impl std::fmt::Display for Error { f.write_str(match self { Self::InvalidInput => "invalid flow-control input", Self::Overloaded => "flow-control quota exhausted", - Self::Unavailable => "flow control stopped", + Self::Unavailable => "flow control unavailable", Self::Io => "flow-control I/O failed", }) } @@ -1018,6 +1018,12 @@ fn wipe_payload(bytes: &mut Vec) { mod quota_tests { use super::*; + /// Unavailability does not imply that flow control has stopped. + #[test] + fn unavailable_display_is_state_neutral() { + assert_eq!(Error::Unavailable.to_string(), "flow control unavailable"); + } + /// Distinct payload, wake-enabled, and drain-progress fixture classes. #[derive(Clone, Copy, Debug)] enum Resource { From 885549547ed50a292145d97e36e178d9dc940f47 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 22:26:55 +0000 Subject: [PATCH 66/82] fix(page-alloc): revalidate file metadata under lock --- cmd/racer-dataplane/alloc/src/slab.rs | 98 ++++++++++++++++++++++++++- 1 file changed, 95 insertions(+), 3 deletions(-) diff --git a/cmd/racer-dataplane/alloc/src/slab.rs b/cmd/racer-dataplane/alloc/src/slab.rs index d1b48eb14..58e66e871 100644 --- a/cmd/racer-dataplane/alloc/src/slab.rs +++ b/cmd/racer-dataplane/alloc/src/slab.rs @@ -324,6 +324,11 @@ impl Slab { /// Prepare backing descriptors without granting submission authority. fn open_file(&self) -> Result { + self.open_file_with_lock_handoff(|_| {}) + } + + /// Open backing storage with a hook for testing a prior lock holder's changes. + fn open_file_with_lock_handoff(&self, before_lock: impl FnOnce(&File)) -> Result { if let Ok(a) = self.alignment() { return Ok(a); } @@ -409,13 +414,21 @@ impl Slab { unsafe { libc::geteuid() }, metadata.nlink(), )?; + before_lock(&file); // SAFETY: flock synchronously borrows this live descriptor. if unsafe { libc::flock(file.as_raw_fd(), libc::LOCK_EX | libc::LOCK_NB) } != 0 { return Err(lock_error(std::io::Error::last_os_error())); } - // Refresh size under the lock: a previous lock holder may have resized - // the file between the first fstat and acquiring our lock. - let size = file.metadata().map_err(|e| system_error("fstat", e))?.len(); + // Recheck metadata under the lock: a prior holder may have changed it. + let metadata = file.metadata().map_err(|e| system_error("fstat", e))?; + // SAFETY: geteuid has no arguments or borrowed memory. + validate_file( + metadata.mode(), + metadata.uid(), + unsafe { libc::geteuid() }, + metadata.nlink(), + )?; + let size = metadata.len(); // Delay O_DIRECT until after rejecting nonregular files (including FIFOs). // SAFETY: fcntl synchronously borrows this live descriptor. let flags = unsafe { libc::fcntl(file.as_raw_fd(), libc::F_GETFL) }; @@ -2310,6 +2323,85 @@ mod tests { } } + /// A prior lock holder cannot leave unsafe metadata for startup to accept. + #[test] + fn lock_handoff_revalidates_private_file_metadata() { + use std::os::unix::fs::PermissionsExt; + + for change in ["permissions", "special-bits", "hard-link", "unlink"] { + let directory = Directory::new(); + let path = directory.0.join("data"); + let prior = open_private_file(&path).unwrap(); + let observer = File::open(&path).unwrap(); + // SAFETY: flock borrows a live descriptor owned by this test. + assert_eq!( + unsafe { libc::flock(prior.as_raw_fd(), libc::LOCK_EX | libc::LOCK_NB) }, + 0 + ); + let slab = Slab::<()>::new(path.clone(), 8192, 4096, 512); + let mut called = false; + let result = slab.open_file_with_lock_handoff(|_| { + called = true; + match change { + "permissions" => prior + .set_permissions(std::fs::Permissions::from_mode(0o644)) + .unwrap(), + "special-bits" => prior + .set_permissions(std::fs::Permissions::from_mode(0o4600)) + .unwrap(), + "hard-link" => std::fs::hard_link(&path, directory.0.join("alias")).unwrap(), + "unlink" => std::fs::remove_file(&path).unwrap(), + _ => unreachable!(), + } + drop(prior); + }); + assert!(called); + assert_eq!(result, Err(Error::InvalidConfiguration), "{change}"); + assert!(slab.opened.borrow().is_none()); + assert_eq!(observer.metadata().unwrap().len(), 0); + // SAFETY: the independent live descriptor checks lock cleanup on failure. + assert_eq!( + unsafe { libc::flock(observer.as_raw_fd(), libc::LOCK_EX | libc::LOCK_NB) }, + 0 + ); + } + } + + /// Valid handoffs use the current size and never truncate mismatched files. + #[test] + fn lock_handoff_uses_refreshed_size() { + for size in [0, 8192, 7] { + let directory = Directory::new(); + let path = directory.0.join("data"); + let prior = open_private_file(&path).unwrap(); + // SAFETY: flock borrows a live descriptor owned by this test. + assert_eq!( + unsafe { libc::flock(prior.as_raw_fd(), libc::LOCK_EX | libc::LOCK_NB) }, + 0 + ); + let slab = Slab::<()>::new(path.clone(), 8192, 4096, 512); + let result = slab.open_file_with_lock_handoff(|_| { + prior.set_len(size).unwrap(); + drop(prior); + }); + if size == 7 { + assert_eq!(result, Err(Error::InvalidConfiguration)); + assert!(slab.opened.borrow().is_none()); + assert_eq!(std::fs::metadata(&path).unwrap().len(), size); + } else { + let Some(alignment) = real_alignment(result) else { + continue; + }; + assert_eq!(slab.alignment(), Ok(alignment)); + assert_eq!(std::fs::metadata(&path).unwrap().len(), 8192); + assert_eq!( + Slab::<()>::new(path, 8192, 4096, 512).open_now(), + Err(Error::Unavailable) + ); + } + } + } + /// Descriptor traversal rejects symlinks, FIFOs, and parent-directory escapes. #[test] fn descriptor_relative_open_rejects_symlinks_and_nonregular_files() { From 1caf390263d02640e8da1db601293d3933715170 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 22:28:48 +0000 Subject: [PATCH 67/82] test(page-alloc): derive lock handoff geometry from probe --- cmd/racer-dataplane/alloc/src/slab.rs | 24 ++++++++++++++++-------- 1 file changed, 16 insertions(+), 8 deletions(-) diff --git a/cmd/racer-dataplane/alloc/src/slab.rs b/cmd/racer-dataplane/alloc/src/slab.rs index 58e66e871..73db5aab6 100644 --- a/cmd/racer-dataplane/alloc/src/slab.rs +++ b/cmd/racer-dataplane/alloc/src/slab.rs @@ -2370,32 +2370,40 @@ mod tests { /// Valid handoffs use the current size and never truncate mismatched files. #[test] fn lock_handoff_uses_refreshed_size() { - for size in [0, 8192, 7] { + for size_in_segments in [0, 2, 1] { let directory = Directory::new(); let path = directory.0.join("data"); let prior = open_private_file(&path).unwrap(); + let Some(alignment) = real_alignment(probe(&prior)) else { + continue; + }; + let segment_bytes = alignment.extent(0, 512).unwrap().length() as u64; + let capacity = segment_bytes.checked_mul(2).unwrap(); + let size = segment_bytes * size_in_segments; // SAFETY: flock borrows a live descriptor owned by this test. assert_eq!( unsafe { libc::flock(prior.as_raw_fd(), libc::LOCK_EX | libc::LOCK_NB) }, 0 ); - let slab = Slab::<()>::new(path.clone(), 8192, 4096, 512); + let slab = Slab::<()>::new(path.clone(), capacity, segment_bytes, 512); let result = slab.open_file_with_lock_handoff(|_| { prior.set_len(size).unwrap(); drop(prior); }); - if size == 7 { + if result == Err(Error::Unsupported) { + assert!(real_alignment(result).is_none()); + continue; + } + if size_in_segments == 1 { assert_eq!(result, Err(Error::InvalidConfiguration)); assert!(slab.opened.borrow().is_none()); assert_eq!(std::fs::metadata(&path).unwrap().len(), size); } else { - let Some(alignment) = real_alignment(result) else { - continue; - }; + assert_eq!(result, Ok(alignment)); assert_eq!(slab.alignment(), Ok(alignment)); - assert_eq!(std::fs::metadata(&path).unwrap().len(), 8192); + assert_eq!(std::fs::metadata(&path).unwrap().len(), capacity); assert_eq!( - Slab::<()>::new(path, 8192, 4096, 512).open_now(), + Slab::<()>::new(path, capacity, segment_bytes, 512).open_now(), Err(Error::Unavailable) ); } From 8617658fa9a87bec6c4ada7844a9d531e01ca818 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 22:32:56 +0000 Subject: [PATCH 68/82] fix(flow): run handoff waker callbacks outside the lock --- cmd/racer-dataplane/flow/src/admission.rs | 145 ++++++++++++++++++++-- 1 file changed, 134 insertions(+), 11 deletions(-) diff --git a/cmd/racer-dataplane/flow/src/admission.rs b/cmd/racer-dataplane/flow/src/admission.rs index 7d8aec49d..3bb8b45e3 100644 --- a/cmd/racer-dataplane/flow/src/admission.rs +++ b/cmd/racer-dataplane/flow/src/admission.rs @@ -893,7 +893,8 @@ mod handoff { queue: VecDeque>, - waker: Option, + // Sharing the registration avoids raw waker callbacks under the lock. + waker: Option>, closed: bool, } @@ -1001,23 +1002,23 @@ mod handoff { waker: &Waker, budget: usize, ) -> Result> { + let waker = Arc::new(waker.clone()); let mut state = self.0.lock().map_err(|_| Error::Unavailable)?; let target = state.target(key).ok_or(Error::InvalidInput)?; if target.closed { return Err(Error::Unavailable); } - if let Some(old) = &mut target.waker { - old.clone_from(waker); - } else { - target.waker = Some(waker.clone()); - } - Ok(std::array::from_fn(|index| { + let old = target.waker.replace(waker); + let batch = std::array::from_fn(|index| { if index < budget { target.queue.pop_front() } else { None } - })) + }); + drop(state); + drop(old); + Ok(batch) } /// Close permanently, then drop ownership and wake outside the shared lock. @@ -1037,7 +1038,7 @@ mod handoff { drop(queued); drop(admission); if let Some(waker) = waker { - waker.wake(); + waker.wake_by_ref(); } } } @@ -1058,7 +1059,7 @@ mod handoff { let waker = target.waker.clone(); drop(state); if let Some(waker) = waker { - waker.wake(); + waker.wake_by_ref(); } Ok(()) } @@ -1072,9 +1073,131 @@ mod handoff { Weak, atomic::{AtomicUsize, Ordering}, }, - task::Wake, + task::{RawWaker, RawWakerVTable, Wake}, }; + struct TargetWake { + handoff: Weak>, + blocked: AtomicUsize, + clones: AtomicUsize, + drops: AtomicUsize, + wakes: AtomicUsize, + } + + impl TargetWake { + fn reenter(&self) { + let Some(handoff) = self.handoff.upgrade() else { + return; + }; + if let Ok(state) = handoff.0.try_lock() { + drop(state); + handoff.close(&2); + } else { + self.blocked.fetch_add(1, Ordering::SeqCst); + } + } + + fn new(handoff: &Arc>) -> (Arc, Waker) { + let state = Arc::new(Self { + handoff: Arc::downgrade(handoff), + blocked: AtomicUsize::new(0), + clones: AtomicUsize::new(0), + drops: AtomicUsize::new(0), + wakes: AtomicUsize::new(0), + }); + // SAFETY: Each raw waker owns one Arc, managed by this vtable. + let waker = unsafe { Waker::from_raw(Self::raw(state.clone())) }; + (state, waker) + } + + fn raw(state: Arc) -> RawWaker { + RawWaker::new(Arc::into_raw(state).cast(), &Self::VTABLE) + } + + const VTABLE: RawWakerVTable = RawWakerVTable::new( + |data| { + // SAFETY: Borrow the live Arc without consuming the source waker. + let state = + std::mem::ManuallyDrop::new(unsafe { Arc::::from_raw(data.cast()) }); + state.reenter(); + state.clones.fetch_add(1, Ordering::SeqCst); + Self::raw(Arc::clone(&state)) + }, + |data| { + // SAFETY: Wake consumes this raw waker's Arc exactly once. + let state = unsafe { Arc::::from_raw(data.cast()) }; + state.reenter(); + state.wakes.fetch_add(1, Ordering::SeqCst); + }, + |data| { + // SAFETY: Borrow without consuming the source waker's Arc. + let state = + std::mem::ManuallyDrop::new(unsafe { Arc::::from_raw(data.cast()) }); + state.reenter(); + state.wakes.fetch_add(1, Ordering::SeqCst); + }, + |data| { + // SAFETY: Drop consumes this raw waker's Arc exactly once. + let state = unsafe { Arc::::from_raw(data.cast()) }; + state.reenter(); + state.drops.fetch_add(1, Ordering::SeqCst); + }, + ); + } + + #[test] + fn target_waker_clone_can_reenter_pop() { + let handoff = Arc::new(Handoff::new(&[1])); + let (state, waker) = TargetWake::new(&handoff); + for _ in 0..2 { + assert!(handoff.pop_batch::<1>(&1, &waker, 0).unwrap()[0].is_none()); + } + assert!(state.clones.load(Ordering::SeqCst) > 0); + assert_eq!(state.blocked.load(Ordering::SeqCst), 0); + } + + #[test] + fn target_waker_drop_can_reenter_replacement_and_rejection() { + let handoff = Arc::new(Handoff::new(&[1])); + let (state, waker) = TargetWake::new(&handoff); + handoff.pop_batch::<0>(&1, &waker, 0).unwrap(); + state.blocked.store(0, Ordering::SeqCst); + handoff.pop_batch::<0>(&1, Waker::noop(), 0).unwrap(); + assert_eq!(state.drops.load(Ordering::SeqCst), 1); + assert_eq!(state.blocked.load(Ordering::SeqCst), 0); + handoff.close(&1); + assert!(matches!( + handoff.pop_batch::<1>(&1, &waker, 1), + Err(Error::Unavailable) + )); + assert!(matches!( + handoff.pop_batch::<1>(&2, &waker, 1), + Err(Error::InvalidInput) + )); + assert_eq!(state.blocked.load(Ordering::SeqCst), 0); + assert!(handoff.0.lock().unwrap().targets[0].1.waker.is_none()); + } + + #[test] + fn target_waker_callbacks_can_reenter_repeated_delivery() { + let handoff = Arc::new(Handoff::new(&[1])); + handoff.install(&1, Ready).unwrap(); + let (state, waker) = TargetWake::new(&handoff); + handoff.pop_batch::<0>(&1, &waker, 0).unwrap(); + state.blocked.store(0, Ordering::SeqCst); + for item in [7, 8] { + handoff + .reserve(Waker::noop()) + .unwrap() + .deliver(|| item) + .unwrap(); + } + assert_eq!(state.wakes.load(Ordering::SeqCst), 2); + assert_eq!(state.blocked.load(Ordering::SeqCst), 0); + let items = handoff.pop_batch::<2>(&1, Waker::noop(), 2).unwrap(); + assert_eq!(items.map(|item| item.unwrap().into_parts().0), [7, 8]); + } + type ClosingHandoff = Handoff; #[derive(Clone)] From 15e9365ef17b774e5dcb526f6753ca5e92fc23b7 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 22:31:23 +0000 Subject: [PATCH 69/82] fix(flow): stage hedge delay wakers outside lock --- cmd/racer-dataplane/flow/src/admission.rs | 152 +++++++++++++++++++++- 1 file changed, 145 insertions(+), 7 deletions(-) diff --git a/cmd/racer-dataplane/flow/src/admission.rs b/cmd/racer-dataplane/flow/src/admission.rs index 3bb8b45e3..8c884c2bf 100644 --- a/cmd/racer-dataplane/flow/src/admission.rs +++ b/cmd/racer-dataplane/flow/src/admission.rs @@ -1512,15 +1512,18 @@ mod hedge { impl Permit { /// Register the latest waiter until due, without releasing capacity. pub fn delay(&self, now: Instant, cx: &mut Context<'_>) -> Poll<()> { + // Waker clone and drop callbacks may reenter the hedge registry. + let wake = cx.waker().clone(); let mut state = self.owner.state.lock().expect("hedge alarm lock"); let alarm = state.alarms.get_mut(&self.id).expect("live hedge alarm"); - if now >= alarm.due { - alarm.wake = None; - Poll::Ready(()) + let (result, previous) = if now >= alarm.due { + (Poll::Ready(()), alarm.wake.take()) } else { - alarm.wake = Some(cx.waker().clone()); - Poll::Pending - } + (Poll::Pending, alarm.wake.replace(wake)) + }; + drop(state); + drop(previous); + result } } @@ -1540,8 +1543,9 @@ mod hedge { mod tests { use super::*; use std::{ + mem::ManuallyDrop, sync::atomic::{AtomicUsize, Ordering}, - task::Wake, + task::{RawWaker, RawWakerVTable, Wake}, time::Duration, }; @@ -1556,6 +1560,140 @@ mod hedge { } } + struct DelayCallbacks { + owner: Arc, + now: Instant, + clones: AtomicUsize, + drops: AtomicUsize, + locked_clones: AtomicUsize, + locked_drops: AtomicUsize, + } + + impl DelayCallbacks { + fn new(owner: &Arc, now: Instant) -> Arc { + Arc::new(Self { + owner: owner.clone(), + now, + clones: AtomicUsize::new(0), + drops: AtomicUsize::new(0), + locked_clones: AtomicUsize::new(0), + locked_drops: AtomicUsize::new(0), + }) + } + + fn reenter(&self, calls: &AtomicUsize, locked: &AtomicUsize) { + calls.fetch_add(1, Ordering::SeqCst); + // Record a deadlock risk without hanging or poisoning the mutex. + if self.owner.state.try_lock().is_err() { + locked.fetch_add(1, Ordering::SeqCst); + return; + } + self.owner.poll(self.now); + } + + fn raw(this: Arc) -> RawWaker { + RawWaker::new(Arc::into_raw(this).cast(), &Self::VTABLE) + } + + fn waker(this: &Arc) -> Waker { + // SAFETY: The vtable owns one Arc per raw waker and uses atomic state. + unsafe { Waker::from_raw(Self::raw(this.clone())) } + } + + const VTABLE: RawWakerVTable = + RawWakerVTable::new(Self::clone_raw, Self::drop_raw, |_| {}, Self::drop_raw); + + unsafe fn clone_raw(data: *const ()) -> RawWaker { + // SAFETY: Borrow the raw waker's Arc without consuming its reference. + let this = ManuallyDrop::new(unsafe { Arc::from_raw(data.cast::()) }); + this.reenter(&this.clones, &this.locked_clones); + Self::raw(Arc::clone(&this)) + } + + unsafe fn drop_raw(data: *const ()) { + // SAFETY: Drop or consuming wake releases exactly one owned reference. + let this = unsafe { Arc::from_raw(data.cast::()) }; + this.reenter(&this.drops, &this.locked_drops); + } + } + + #[test] + fn hedge_delay_clones_waiter_outside_lock() { + let now = Instant::now(); + let due = now + Duration::from_secs(1); + let owner = Hedges::new(1, 1); + let permit = owner.acquire(1, due).unwrap(); + let callbacks = DelayCallbacks::new(&owner, now); + let waker = DelayCallbacks::waker(&callbacks); + assert!( + permit + .delay(now, &mut Context::from_waker(&waker)) + .is_pending() + ); + // Clear the registration so cleanup does not exercise Permit::drop. + assert!( + permit + .delay(due, &mut Context::from_waker(Waker::noop())) + .is_ready() + ); + assert_eq!(callbacks.clones.load(Ordering::SeqCst), 1); + assert_eq!(callbacks.locked_clones.load(Ordering::SeqCst), 0); + } + + #[test] + fn hedge_delay_replaces_waiter_outside_lock() { + let now = Instant::now(); + let due = now + Duration::from_secs(1); + let owner = Hedges::new(1, 1); + let permit = owner.acquire(1, due).unwrap(); + let callbacks = DelayCallbacks::new(&owner, now); + let waker = DelayCallbacks::waker(&callbacks); + assert!( + permit + .delay(now, &mut Context::from_waker(&waker)) + .is_pending() + ); + let latest = Arc::new(Counter::default()); + assert!( + permit + .delay(now, &mut Context::from_waker(&Waker::from(latest.clone()))) + .is_pending() + ); + assert_eq!(callbacks.drops.load(Ordering::SeqCst), 1); + assert_eq!(callbacks.locked_drops.load(Ordering::SeqCst), 0); + owner.poll(due); + owner.poll(due); + assert_eq!(latest.0.load(Ordering::SeqCst), 1); + assert!(matches!(owner.acquire(1, due), Err(Error::Overloaded))); + } + + #[test] + fn hedge_delay_clears_waiter_outside_lock() { + for elapsed in [Duration::ZERO, Duration::from_secs(1)] { + let now = Instant::now(); + let due = now + Duration::from_secs(1); + let owner = Hedges::new(1, 1); + let permit = owner.acquire(1, due).unwrap(); + let callbacks = DelayCallbacks::new(&owner, now); + let waker = DelayCallbacks::waker(&callbacks); + assert!( + permit + .delay(now, &mut Context::from_waker(&waker)) + .is_pending() + ); + assert!( + permit + .delay(due + elapsed, &mut Context::from_waker(Waker::noop())) + .is_ready() + ); + assert_eq!(callbacks.drops.load(Ordering::SeqCst), 1); + assert_eq!(callbacks.locked_drops.load(Ordering::SeqCst), 0); + owner.poll(due + elapsed); + assert_eq!(callbacks.drops.load(Ordering::SeqCst), 1); + assert!(matches!(owner.acquire(1, due), Err(Error::Overloaded))); + } + } + /// Slots and bytes stay charged after an alarm fires and across threads. #[test] fn hedge_shared_slots_costs_and_release_are_independent_of_alarm() { From 78583c58c9d9b6c25eee3c4afe14430d6eb7497d Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 22:27:50 +0000 Subject: [PATCH 70/82] fix(flow): drop hedge alarm wakers outside the lock --- cmd/racer-dataplane/flow/src/admission.rs | 68 +++++++++++++++++++++-- 1 file changed, 64 insertions(+), 4 deletions(-) diff --git a/cmd/racer-dataplane/flow/src/admission.rs b/cmd/racer-dataplane/flow/src/admission.rs index 8c884c2bf..19381a824 100644 --- a/cmd/racer-dataplane/flow/src/admission.rs +++ b/cmd/racer-dataplane/flow/src/admission.rs @@ -1530,11 +1530,12 @@ mod hedge { impl Drop for Permit { /// Remove the alarm and return its cost only at final ownership release. fn drop(&mut self) { - if let Ok(mut state) = self.owner.state.lock() - && let Some(alarm) = state.alarms.remove(&self.id) - { + let alarm = self.owner.state.lock().ok().and_then(|mut state| { + let alarm = state.alarms.remove(&self.id)?; state.used -= alarm.cost; - } + Some(alarm) + }); + drop(alarm); } } @@ -1794,6 +1795,65 @@ mod hedge { owner.poll(due); assert_eq!(current.0.load(Ordering::SeqCst), 1); } + + #[test] + fn hedge_permit_drop_releases_waker_outside_lock() { + struct ReenterOnDrop { + owner: std::sync::Weak, + drops: Arc, + wakes: Arc, + } + + impl Wake for ReenterOnDrop { + fn wake(self: Arc) { + self.wakes.fetch_add(1, Ordering::SeqCst); + } + } + + impl Drop for ReenterOnDrop { + fn drop(&mut self) { + let owner = self.owner.upgrade().unwrap(); + let state = owner + .state + .try_lock() + .expect("permit drop must unlock before dropping its waker"); + assert!(state.alarms.is_empty()); + assert_eq!(state.used, 0); + drop(state); + let now = Instant::now(); + owner.poll(now); + let permit = owner.acquire(owner.capacity, now).unwrap(); + drop(permit); + self.drops.fetch_add(1, Ordering::SeqCst); + } + } + + for cost in [0, 7] { + let now = Instant::now(); + let due = now + Duration::from_secs(1); + let owner = Hedges::new(1, 7); + let permit = owner.acquire(cost, due).unwrap(); + let drops = Arc::new(AtomicUsize::new(0)); + let wakes = Arc::new(AtomicUsize::new(0)); + let waker = Waker::from(Arc::new(ReenterOnDrop { + owner: Arc::downgrade(&owner), + drops: drops.clone(), + wakes: wakes.clone(), + })); + assert!( + permit + .delay(now, &mut Context::from_waker(&waker)) + .is_pending() + ); + drop(waker); + assert_eq!(drops.load(Ordering::SeqCst), 0); + drop(permit); + assert_eq!(drops.load(Ordering::SeqCst), 1); + owner.poll(due); + assert_eq!(wakes.load(Ordering::SeqCst), 0); + assert!(owner.acquire(7, due).is_ok()); + } + } } } From e37aef41b781cca1f3036ecd07b5f60262c6fc13 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 22:30:24 +0000 Subject: [PATCH 71/82] fix(flow): release cohort borrows before waker callbacks --- cmd/racer-dataplane/flow/src/coalesce.rs | 21 ++- cmd/racer-dataplane/flow/tests/workflows.rs | 188 +++++++++++++++++++- 2 files changed, 204 insertions(+), 5 deletions(-) diff --git a/cmd/racer-dataplane/flow/src/coalesce.rs b/cmd/racer-dataplane/flow/src/coalesce.rs index ae85a05ad..51aafbe39 100644 --- a/cmd/racer-dataplane/flow/src/coalesce.rs +++ b/cmd/racer-dataplane/flow/src/coalesce.rs @@ -196,11 +196,23 @@ impl Registration { impl Registration { /// Register the latest wake target and elect at most once per attempt. pub fn event(&self, waker: &Waker) -> Poll> { + let waker = waker.clone(); + let retired = { + let mut state = self.owner.cohort.borrow_mut(); + if state.result.is_some() { + Some(waker) + } else { + state.waiters.insert(self.owner.id, Some(waker)).flatten() + } + }; + drop(retired); + + // Clone and drop callbacks can retry, elect, or finish. Decide from the + // current state only after those callbacks have released their borrows. let mut state = self.owner.cohort.borrow_mut(); if let Some(result) = &state.result { return Poll::Ready(Event::Complete(result.clone())); } - state.waiters.insert(self.owner.id, Some(waker.clone())); if state.leader.is_none() { if state.attempts >= self.owner.table.limits.attempts_per_cohort { drop(state); @@ -232,16 +244,16 @@ impl RegistrationOwner { impl Drop for RegistrationOwner { /// Detach once, release the charge, then wake followers outside all borrows. fn drop(&mut self) { - let (empty, wakers) = { + let (empty, retired, wakers) = { let mut state = self.cohort.borrow_mut(); - state.waiters.remove(&self.id); + let retired = state.waiters.remove(&self.id); let wakers = if state.leader == Some(self.id) { state.leader = None; state.take_wakers() } else { Vec::new() }; - (state.waiters.is_empty(), wakers) + (state.waiters.is_empty(), retired, wakers) }; self.table .registrations @@ -249,6 +261,7 @@ impl Drop for RegistrationOwner { if empty { self.remove_active(); } + drop(retired); wake_all(wakers); } } diff --git a/cmd/racer-dataplane/flow/tests/workflows.rs b/cmd/racer-dataplane/flow/tests/workflows.rs index 5b5322513..998e35191 100644 --- a/cmd/racer-dataplane/flow/tests/workflows.rs +++ b/cmd/racer-dataplane/flow/tests/workflows.rs @@ -843,12 +843,198 @@ mod coalesce_tests { use std::future::Future; use std::rc::Rc; use std::sync::Arc; - use std::task::{Context, Poll, Wake, Waker}; + use std::task::{Context, Poll, RawWaker, RawWakerVTable, Wake, Waker}; use std::time::Instant; thread_local! { /// Callback invoked only by synchronous wakes on the current test worker. static ON_WAKE: RefCell>> = RefCell::new(None); + + /// One-shot hooks for raw waker ownership callbacks on this test worker. + static ON_COHORT_CLONE: RefCell>> = RefCell::new(None); + + static ON_COHORT_DROP: RefCell>> = RefCell::new(None); + } + + /// Stateless raw waker; callbacks use only the calling thread's test hooks. + fn cohort_raw_waker() -> RawWaker { + /// Run the clone hook without retaining its registry borrow. + unsafe fn clone(_: *const ()) -> RawWaker { + let callback = ON_COHORT_CLONE.with(|slot| slot.borrow_mut().take()); + if let Some(callback) = callback { + callback(); + } + cohort_raw_waker() + } + + /// Run the drop hook without retaining its registry borrow. + unsafe fn drop(_: *const ()) { + let callback = ON_COHORT_DROP.with(|slot| slot.borrow_mut().take()); + if let Some(callback) = callback { + callback(); + } + } + + /// Notifications need no action for these ownership callback tests. + unsafe fn wake(_: *const ()) {} + + RawWaker::new( + std::ptr::null(), + &RawWakerVTable::new(clone, wake, wake, drop), + ) + } + + /// Create a thread-safe stateless waker with worker-local test hooks. + fn cohort_waker() -> Waker { + // SAFETY: The vtable never dereferences data or shares thread-local hooks. + unsafe { Waker::from_raw(cohort_raw_waker()) } + } + + /// Clone and replacement callbacks may poll, retry, or finish the same cohort. + fn check_cohort_waker_event_reentry(on_clone: bool) { + for action in 0..3 { + let table = table(2); + let leader = table.join(1, 1).unwrap(); + let follower = table.join(1, 1).unwrap(); + assert_eq!(leader.event(Waker::noop()), Poll::Ready(Event::Lead)); + let waker = cohort_waker(); + if !on_clone { + assert!(follower.event(&waker).is_pending()); + } + let called = Rc::new(Cell::new(false)); + let callback: Box = Box::new({ + let leader = leader.clone(); + let follower = follower.clone(); + let called = called.clone(); + move || { + match action { + 0 => assert!(follower.event(Waker::noop()).is_pending()), + 1 => leader.retry(), + _ => leader.finish(7), + } + called.set(true); + } + }); + if on_clone { + ON_COHORT_CLONE.with(|slot| *slot.borrow_mut() = Some(callback)); + } else { + ON_COHORT_DROP.with(|slot| *slot.borrow_mut() = Some(callback)); + } + let event = follower.event(if on_clone { &waker } else { Waker::noop() }); + assert!(called.get()); + assert_eq!( + event, + match action { + 0 => Poll::Pending, + 1 => Poll::Ready(Event::Lead), + _ => Poll::Ready(Event::Complete(7)), + } + ); + assert_eq!(table.registration_count(), 2); + assert_eq!(table.active_count(), usize::from(action != 2)); + drop((leader, follower, waker)); + assert_eq!(table.registration_count(), 0); + assert_eq!(table.active_count(), 0); + } + } + + /// Cloning the incoming waker runs before borrowing or deciding the event. + #[test] + fn cohort_waker_clone_reentry() { + check_cohort_waker_event_reentry(true); + } + + /// Retiring the old waker runs before deciding the event from current state. + #[test] + fn cohort_waker_replacement_drop_reentry() { + check_cohort_waker_event_reentry(false); + } + + /// Detach updates charges and leadership before destroying the removed waker. + #[test] + fn cohort_waker_detach_drop_reentry() { + for drop_leader in [false, true] { + for action in 0..3 { + let table = table(2); + let leader = table.join(1, 1).unwrap(); + let follower = table.join(1, 1).unwrap(); + let waker = cohort_waker(); + assert_eq!(leader.event(&waker), Poll::Ready(Event::Lead)); + assert!(follower.event(&waker).is_pending()); + let (removed, remaining) = if drop_leader { + (leader, follower) + } else { + (follower, leader) + }; + let called = Rc::new(Cell::new(false)); + ON_COHORT_DROP.with(|slot| { + *slot.borrow_mut() = Some(Box::new({ + let table = table.clone(); + let remaining = remaining.clone(); + let called = called.clone(); + move || { + match action { + 0 => assert_eq!( + remaining.event(Waker::noop()), + if drop_leader { + Poll::Ready(Event::Lead) + } else { + Poll::Pending + } + ), + 1 => remaining.retry(), + _ => remaining.finish(7), + } + assert_eq!(table.registration_count(), 1); + called.set(true); + } + })); + }); + drop(removed); + assert!(called.get()); + if action == 1 { + assert_eq!(remaining.event(Waker::noop()), Poll::Ready(Event::Lead)); + } else if action == 2 { + assert_eq!( + remaining.event(Waker::noop()), + Poll::Ready(Event::Complete(7)) + ); + } + drop((remaining, waker)); + assert_eq!(table.registration_count(), 0); + assert_eq!(table.active_count(), 0); + } + } + } + + /// Last-owner waker destruction sees released capacity and can admit a new cohort. + #[test] + fn cohort_waker_last_detach_admits_replacement() { + let table = table(1); + let registration = table.join(1, 1).unwrap(); + let waker = cohort_waker(); + assert_eq!(registration.event(&waker), Poll::Ready(Event::Lead)); + let replacement = Rc::new(RefCell::new(None)); + ON_COHORT_DROP.with(|slot| { + *slot.borrow_mut() = Some(Box::new({ + let table = table.clone(); + let replacement = replacement.clone(); + move || { + assert_eq!(table.registration_count(), 0); + assert_eq!(table.active_count(), 0); + let next = table.join(1, 1).unwrap(); + assert_eq!(next.event(Waker::noop()), Poll::Ready(Event::Lead)); + replacement.replace(Some(next)); + } + })); + }); + drop(registration); + assert!(replacement.borrow().is_some()); + assert_eq!(table.registration_count(), 1); + assert_eq!(table.active_count(), 1); + drop((replacement, waker)); + assert_eq!(table.registration_count(), 0); + assert_eq!(table.active_count(), 0); } /// Safe thread-local callback dispatch; no non-Send data enters the Waker itself. From 98780357418974ed1d52c223dbd1c666e8f342bf Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 22:37:13 +0000 Subject: [PATCH 72/82] fix(flow): drop shared receiver outside table borrow --- cmd/racer-dataplane/flow/src/coalesce.rs | 65 +++++++++++++++++++++++- 1 file changed, 64 insertions(+), 1 deletion(-) diff --git a/cmd/racer-dataplane/flow/src/coalesce.rs b/cmd/racer-dataplane/flow/src/coalesce.rs index 51aafbe39..5b597cb28 100644 --- a/cmd/racer-dataplane/flow/src/coalesce.rs +++ b/cmd/racer-dataplane/flow/src/coalesce.rs @@ -388,7 +388,9 @@ pub mod shared { impl Completion { /// Real completion closes admission before notifying any old readers. pub fn finish(self, value: V) { - self.table.entries.borrow_mut().remove(&self.key); + let removed = self.table.entries.borrow_mut().remove(&self.key); + // Final receiver destruction can reenter the table. + drop(removed); let _ = self.send.send(value); } } @@ -399,6 +401,67 @@ pub mod shared { use super::*; use futures::executor::block_on; + /// Final receiver destruction can reenter through its retained closed value. + #[test] + fn completion_drops_last_receiver_outside_table_borrow() { + use std::cell::Cell; + use std::sync::Arc; + use std::task::{Context, Wake, Waker}; + + thread_local! { + static ON_DROP: RefCell>> = RefCell::new(None); + } + + struct ReentrantDrop; + + impl Wake for ReentrantDrop { + fn wake(self: Arc) { + panic!("closed result waker must only be dropped"); + } + } + + impl Drop for ReentrantDrop { + fn drop(&mut self) { + let callback = ON_DROP.with(|slot| slot.borrow_mut().take()); + if let Some(callback) = callback { + callback(); + } + } + } + + let table = Rc::new(Table::default()); + let dropped = Rc::new(Cell::new(false)); + let closed = Some(Waker::from(Arc::new(ReentrantDrop))); + let (mut receive, completion) = table.start(1, closed); + assert!( + receive + .poll_unpin(&mut Context::from_waker(Waker::noop())) + .is_pending() + ); + drop(receive); + assert_eq!(table.len(), 1); + + let nested = table.clone(); + let observed = dropped.clone(); + ON_DROP.with(|slot| { + *slot.borrow_mut() = Some(Box::new(move || { + assert!(nested.is_empty()); + assert!(nested.get(&1).is_none()); + let (replacement, completion) = nested.start(1, None); + assert_eq!(nested.len(), 1); + completion.finish(None); + assert!(block_on(replacement).is_none()); + observed.set(true); + })); + }); + + assert!(!dropped.get()); + completion.finish(None); + assert!(dropped.get()); + assert!(table.is_empty()); + assert!(ON_DROP.with(|slot| slot.borrow().is_none())); + } + /// Results remain readable after completion admits a replacement. #[test] fn success_miss_failure_broadcast_and_replacement() { From 4f819c2ff869d70b8e7b446f112217b83534e487 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 22:36:17 +0000 Subject: [PATCH 73/82] fix(flow): stage drain wakers outside owner borrows --- cmd/racer-dataplane/flow/src/coalesce.rs | 8 +- cmd/racer-dataplane/flow/tests/drain_waker.rs | 114 ++++++++++++++++++ cmd/racer-dataplane/flow/tests/workflows.rs | 4 +- 3 files changed, 122 insertions(+), 4 deletions(-) create mode 100644 cmd/racer-dataplane/flow/tests/drain_waker.rs diff --git a/cmd/racer-dataplane/flow/src/coalesce.rs b/cmd/racer-dataplane/flow/src/coalesce.rs index 5b597cb28..fca0d130d 100644 --- a/cmd/racer-dataplane/flow/src/coalesce.rs +++ b/cmd/racer-dataplane/flow/src/coalesce.rs @@ -676,9 +676,11 @@ pub mod flight { self.stopping } - /// Retain the latest drain task's wake target. - pub fn register_drain(&mut self, waker: &Waker) { - state::store_waker(&mut self.drain_waker, waker); + /// Retain the latest drain task's wake target and return the previous one. + /// Clone before borrowing the owner; drop the returned waker after releasing + /// that borrow. With `update`, return it from the transaction closure. + pub fn register_drain(&mut self, waker: Waker) -> Option { + self.drain_waker.replace(waker) } /// Enqueue a pending drain notification at most once. diff --git a/cmd/racer-dataplane/flow/tests/drain_waker.rs b/cmd/racer-dataplane/flow/tests/drain_waker.rs new file mode 100644 index 000000000..ed33813c7 --- /dev/null +++ b/cmd/racer-dataplane/flow/tests/drain_waker.rs @@ -0,0 +1,114 @@ +//! Drain registration must not invoke waker callbacks under the owner borrow. + +use flow_control::coalesce::flight::{self, Table}; +use std::cell::{Cell, RefCell}; +use std::rc::Rc; +use std::sync::Arc; +use std::sync::atomic::{AtomicUsize, Ordering}; +use std::task::{RawWaker, RawWakerVTable, Wake, Waker}; + +thread_local! { + static ON_CLONE: RefCell>> = RefCell::new(None); + static ON_DROP: RefCell>> = RefCell::new(None); +} + +/// No pointer is dereferenced or owned; callbacks belong to the current thread. +static VTABLE: RawWakerVTable = RawWakerVTable::new( + |_| { + let callback = ON_CLONE.with(|slot| slot.borrow_mut().take()); + if let Some(callback) = callback { + callback(); + } + raw_waker() + }, + |_| {}, + |_| {}, + |_| { + let callback = ON_DROP.with(|slot| slot.borrow_mut().take()); + if let Some(callback) = callback { + callback(); + } + }, +); + +fn raw_waker() -> RawWaker { + RawWaker::new(std::ptr::null(), &VTABLE) +} + +fn callback_waker() -> Waker { + // SAFETY: The vtable owns no data and only accesses thread-local callbacks. + unsafe { Waker::from_raw(raw_waker()) } +} + +fn register(owner: &RefCell>, waker: &Waker) { + let waker = waker.clone(); + drop(flight::update(owner, |table, _| { + table.register_drain(waker) + })); +} + +#[test] +fn drain_waker_clone_can_reenter_owner() { + let owner = Rc::new(RefCell::new(Table::::default())); + let called = Rc::new(Cell::new(false)); + ON_CLONE.with(|slot| { + slot.replace(Some(Box::new({ + let owner = owner.clone(); + let called = called.clone(); + move || { + flight::update(&owner, |table, _| assert!(table.is_empty())); + called.set(true); + } + }))); + }); + register(&owner, &callback_waker()); + assert!(called.get()); +} + +#[test] +fn drain_waker_replacement_drop_can_reenter_owner() { + let owner = Rc::new(RefCell::new(Table::::default())); + register(&owner, &callback_waker()); + let called = Rc::new(Cell::new(false)); + ON_DROP.with(|slot| { + slot.replace(Some(Box::new({ + let owner = owner.clone(); + let called = called.clone(); + move || { + flight::update(&owner, |table, _| assert!(table.is_empty())); + called.set(true); + } + }))); + }); + register(&owner, Waker::noop()); + assert!(called.get()); +} + +#[derive(Default)] +struct CountWake(AtomicUsize); + +impl Wake for CountWake { + fn wake(self: Arc) { + self.0.fetch_add(1, Ordering::Relaxed); + } +} + +#[test] +fn drain_waker_notifies_latest_target_once() { + let owner = RefCell::new(Table::::default()); + let old = Arc::new(CountWake::default()); + let latest = Arc::new(CountWake::default()); + let latest_waker = Waker::from(latest.clone()); + register(&owner, &Waker::from(old.clone())); + register(&owner, &latest_waker); + register(&owner, &latest_waker); + flight::update(&owner, |table, wakes| { + table.notify_drain(wakes); + table.notify_drain(wakes); + }); + assert_eq!(old.0.load(Ordering::Relaxed), 0); + assert_eq!(latest.0.load(Ordering::Relaxed), 1); + register(&owner, &latest_waker); + flight::update(&owner, |table, wakes| table.notify_drain(wakes)); + assert_eq!(latest.0.load(Ordering::Relaxed), 2); +} diff --git a/cmd/racer-dataplane/flow/tests/workflows.rs b/cmd/racer-dataplane/flow/tests/workflows.rs index 998e35191..3e90a9583 100644 --- a/cmd/racer-dataplane/flow/tests/workflows.rs +++ b/cmd/racer-dataplane/flow/tests/workflows.rs @@ -1358,8 +1358,10 @@ mod coalesce_tests { notified.set(true); } }); + drop(flight::update(&table, |table, _| { + table.register_drain(waker) + })); flight::update(&table, |table, wakes| { - table.register_drain(&waker); let operations = &mut table.get_mut(&1).unwrap().0; operations.complete(id).unwrap(); assert_eq!(operations.complete(id), Err(Stale)); From 4e2931f7ec5f1596b9d671536c73cf6bf6290604 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 22:32:30 +0000 Subject: [PATCH 74/82] fix(flow): defer removed entry destruction past owner borrows --- cmd/racer-dataplane/flow/src/coalesce.rs | 82 +++++++++------- cmd/racer-dataplane/flow/tests/workflows.rs | 102 ++++++++++++++++++-- 2 files changed, 142 insertions(+), 42 deletions(-) diff --git a/cmd/racer-dataplane/flow/src/coalesce.rs b/cmd/racer-dataplane/flow/src/coalesce.rs index fca0d130d..99884a6f4 100644 --- a/cmd/racer-dataplane/flow/src/coalesce.rs +++ b/cmd/racer-dataplane/flow/src/coalesce.rs @@ -748,13 +748,15 @@ pub mod flight { impl Table { /// Used after explicit detach or completion as well as by background sweeps. - pub fn remove_quiescent(&mut self, key: &K) -> bool { + /// Return the entry for destruction after releasing the owner borrow. + #[must_use = "drop the removed entry outside the owner borrow"] + pub fn remove_quiescent(&mut self, key: &K) -> Option { if !self.get(key).is_some_and(Entry::quiescent) { - return false; + return None; } let entry = self.entries.remove(key).expect("quiescent entry"); self.sweep.remove(&entry.sweep_id); - true + Some(entry.entry) } } @@ -775,7 +777,10 @@ pub mod flight { impl Table { /// Refresh at most the budgeted entry count and remove quiescent entries. - pub fn sweep(&mut self, budget: usize, wakes: &mut Vec) { + /// Return removed entries from `update` and drop them outside its borrow. + #[must_use = "drop the removed entries outside the owner borrow"] + pub fn sweep(&mut self, budget: usize, wakes: &mut Vec) -> Vec { + let mut removed = Vec::new(); for _ in 0..budget.min(self.sweep.len()) { let Some((_, key)) = self.cursor.next(&self.sweep) else { break; @@ -784,11 +789,14 @@ pub mod flight { if let Some(entry) = self.get_mut(&key) { entry.refresh(wakes); } - self.remove_quiescent(&key); + if let Some(entry) = self.remove_quiescent(&key) { + removed.push(entry); + } } if self.entries.is_empty() { self.notify_drain(wakes); } + removed } } @@ -1574,20 +1582,23 @@ pub mod flight { entry.operations.insert(1, Resource(live.clone())); entry.canceled = true; let mut wakes = Vec::new(); - table.sweep(1, &mut wakes); + drop(table.sweep(1, &mut wakes)); assert_eq!(table.get(&1).unwrap().waiters, 0); assert_eq!(table.get(&2).unwrap().waiters, 1); assert_eq!(live.get(), 1, "cancellation is not completion"); let resources = table.get_mut(&1).unwrap().operations.take(1).unwrap(); - assert!(!table.remove_quiescent(&1), "occupied during resource drop"); + assert!( + table.remove_quiescent(&1).is_none(), + "occupied during resource drop" + ); drop(resources); assert_eq!(live.get(), 0); assert!( - !table.remove_quiescent(&1), + table.remove_quiescent(&1).is_none(), "completion must clear the tombstone" ); table.get_mut(&1).unwrap().operations.complete(1).unwrap(); - assert!(table.remove_quiescent(&1)); + assert!(table.remove_quiescent(&1).is_some()); assert_eq!(table.len(), 1); } @@ -1601,9 +1612,10 @@ pub mod flight { impl Drop for Waiter { /// Detach this request and remove its entry only if truly quiescent. fn drop(&mut self) { - let mut table = self.table.borrow_mut(); - table.get_mut(&self.key).unwrap().waiters -= 1; - table.remove_quiescent(&self.key); + drop(update(&self.table, |table, _| { + table.get_mut(&self.key).unwrap().waiters -= 1; + table.remove_quiescent(&self.key) + })); } } @@ -1625,7 +1637,7 @@ pub mod flight { key: 1, }); drop(token); - table.borrow_mut().sweep(100, &mut Vec::new()); + drop(update(&table, |table, wakes| table.sweep(100, wakes))); assert_eq!(table.borrow().len(), 1); assert_eq!(live.get(), 1); let resources = table @@ -1643,7 +1655,7 @@ pub mod flight { .operations .complete(1) .unwrap(); - table.borrow_mut().sweep(1, &mut Vec::new()); + drop(update(&table, |table, wakes| table.sweep(1, wakes))); assert!(table.borrow().is_empty()); assert_eq!(live.get(), 0); } @@ -1681,7 +1693,7 @@ pub mod flight { wrong_owner.owner = Rc::new(()); assert_eq!(wrong_owner.validate(&old), Err(Stale)); entry.waiters = 0; - assert!(table.remove_quiescent(&1)); + assert!(table.remove_quiescent(&1).is_some()); let new = insert(&mut table, &owner, 1); assert!(!old.same_registration(&new)); assert_eq!(old.validate(&new), Err(Stale)); @@ -1696,14 +1708,14 @@ pub mod flight { for key in 1..=3 { insert(&mut table, &owner, key); } - table.sweep(0, &mut Vec::new()); + drop(table.sweep(0, &mut Vec::new())); assert!(table.values().all(|entry| entry.refreshed == 0)); for key in 1..=3 { - table.sweep(1, &mut Vec::new()); + drop(table.sweep(1, &mut Vec::new())); assert_eq!(table.get(&key).unwrap().refreshed, 1); } table.get_mut(&2).unwrap().canceled = true; - table.sweep(99, &mut Vec::new()); + drop(table.sweep(99, &mut Vec::new())); assert!(!table.contains_key(&2)); assert_eq!(table.sweep.len(), 2); assert!(table.values().all(|entry| entry.refreshed == 2)); @@ -1712,11 +1724,11 @@ pub mod flight { entry.canceled = true; }); let mut wakes = Vec::new(); - table.sweep(99, &mut wakes); + drop(table.sweep(99, &mut wakes)); assert!(table.is_empty()); assert!(table.sweep.is_empty()); assert_eq!(wakes.len(), 1); - table.sweep(99, &mut wakes); + drop(table.sweep(99, &mut wakes)); assert_eq!(wakes.len(), 1, "drain wake is taken once"); } @@ -1751,7 +1763,7 @@ pub mod flight { let mut table = TestTable::default(); assert!(table.get(&1).is_none()); assert!(table.get_mut(&1).is_none()); - assert!(!table.remove_quiescent(&1)); + assert!(table.remove_quiescent(&1).is_none()); insert(&mut table, &owner, 1); let original = table.entries[&1].sweep_id; insert(&mut table, &owner, 1); @@ -1761,13 +1773,13 @@ pub mod flight { assert_eq!(table.values().count(), 2); assert!(!table.sweep.contains_key(&original)); assert!(table.sweep.contains_key(&replacement)); - table.sweep(2, &mut Vec::new()); + drop(table.sweep(2, &mut Vec::new())); assert!(table.values().all(|entry| entry.refreshed == 1)); table.get_mut(&1).unwrap().waiters = 0; - assert!(table.remove_quiescent(&1)); + assert!(table.remove_quiescent(&1).is_some()); assert!(!table.contains_key(&1)); assert_eq!(table.sweep.len(), table.len()); - table.sweep(1, &mut Vec::new()); + drop(table.sweep(1, &mut Vec::new())); assert_eq!(table.get(&2).unwrap().refreshed, 2); } @@ -1780,12 +1792,12 @@ pub mod flight { let other = insert(&mut table, &owner, 2); table.get_mut(&1).unwrap().identity.incarnation = other.incarnation; table.get_mut(&1).unwrap().waiters = 0; - assert!(table.remove_quiescent(&1)); + assert!(table.remove_quiescent(&1).is_some()); assert_eq!(table.sweep.len(), 1); - table.sweep(1, &mut Vec::new()); + drop(table.sweep(1, &mut Vec::new())); assert_eq!(table.get(&2).unwrap().refreshed, 1); table.get_mut(&2).unwrap().waiters = 0; - table.sweep(1, &mut Vec::new()); + drop(table.sweep(1, &mut Vec::new())); assert!(table.is_empty()); assert!(table.sweep.is_empty()); } @@ -1804,12 +1816,12 @@ pub mod flight { }); insert(&mut table, &owner, 2); assert_eq!(table.sweep.len(), 3); - table.sweep(3, &mut Vec::new()); + drop(table.sweep(3, &mut Vec::new())); assert_eq!(table.len(), 1); assert_eq!(table.sweep.len(), 1); assert_eq!(table.get(&2).unwrap().refreshed, 1); table.get_mut(&2).unwrap().waiters = 0; - assert!(table.remove_quiescent(&2)); + assert!(table.remove_quiescent(&2).is_some()); assert!(table.sweep.is_empty()); } @@ -1832,10 +1844,10 @@ pub mod flight { ); } assert_eq!(table.sweep.len(), 3); - table.sweep(0, &mut Vec::new()); + drop(table.sweep(0, &mut Vec::new())); assert!(table.values().all(|entry| entry.refreshed == 0)); for key in 1..=3 { - table.sweep(1, &mut Vec::new()); + drop(table.sweep(1, &mut Vec::new())); assert_eq!(table.get(&key).unwrap().refreshed, 1); assert_eq!( table.values().map(|entry| entry.refreshed).sum::(), @@ -1843,8 +1855,8 @@ pub mod flight { ); } table.get_mut(&2).unwrap().waiters = 0; - assert!(table.remove_quiescent(&2)); - table.sweep(99, &mut Vec::new()); + assert!(table.remove_quiescent(&2).is_some()); + drop(table.sweep(99, &mut Vec::new())); assert_eq!(table.sweep.len(), 2); assert!(table.values().all(|entry| entry.refreshed == 2)); } @@ -1864,7 +1876,7 @@ pub mod flight { (first.incarnation, second.incarnation, third.incarnation), (1, 2, 3) ); - table.sweep(3, &mut Vec::new()); + drop(table.sweep(3, &mut Vec::new())); assert!(table.values().all(|entry| entry.refreshed == 1)); table.incarnation = Counter(u64::MAX); assert!(table.identity(owner).is_err()); @@ -1884,7 +1896,7 @@ pub mod flight { "insert does not require an incarnation allocation" ); table.stop(&mut Vec::new(), |entry, _| entry.canceled = true); - table.sweep(4, &mut Vec::new()); + drop(table.sweep(4, &mut Vec::new())); assert!(table.is_empty()); assert!(table.sweep.is_empty()); } diff --git a/cmd/racer-dataplane/flow/tests/workflows.rs b/cmd/racer-dataplane/flow/tests/workflows.rs index 3e90a9583..7e81c63b7 100644 --- a/cmd/racer-dataplane/flow/tests/workflows.rs +++ b/cmd/racer-dataplane/flow/tests/workflows.rs @@ -1308,6 +1308,93 @@ mod coalesce_tests { } } + /// A removable entry can still own application data with a reentrant destructor. + struct DroppableEntry { + _resource: ReentrantResource, + + quiescent: bool, + } + + impl Entry for DroppableEntry { + fn refresh(&mut self, _: &mut Vec) {} + + fn quiescent(&self) -> bool { + self.quiescent + } + } + + /// Sweeping transfers entry destruction past the owner transaction. + #[test] + fn swept_entry_destructor_can_reenter_owner() { + removed_entry_destructor_can_reenter_owner(true); + } + + /// Explicit removal has the same destruction boundary as a sweep. + #[test] + fn detached_entry_destructor_can_reenter_owner() { + removed_entry_destructor_can_reenter_owner(false); + } + + /// Check both removal paths, including ineligible entries and same-key replacement. + fn removed_entry_destructor_can_reenter_owner(sweep: bool) { + let table = Rc::new(RefCell::new(flight::Table::::default())); + let dropped = Rc::new(Cell::new(0)); + let resource = ReentrantResource(Some(Box::new({ + let table = Rc::downgrade(&table); + let dropped = dropped.clone(); + move || { + let table = table.upgrade().unwrap(); + drop(flight::update(&table, |table, wakes| { + assert!(table.is_empty()); + let removed = table.sweep(1, wakes); + assert!(removed.is_empty()); + assert_eq!(table.next_waiter_id(), Ok(1)); + table.insert( + 1, + DroppableEntry { + _resource: ReentrantResource(Some(Box::new(|| {}))), + quiescent: false, + }, + ); + removed + })); + dropped.set(dropped.get() + 1); + } + }))); + table.borrow_mut().insert( + 1, + DroppableEntry { + _resource: resource, + quiescent: false, + }, + ); + let removed = flight::update(&table, |table, wakes| { + assert!(table.remove_quiescent(&2).is_none()); + assert!(table.remove_quiescent(&1).is_none()); + assert!(table.sweep(1, wakes).is_empty()); + table.get_mut(&1).unwrap().quiescent = true; + assert!(table.sweep(0, wakes).is_empty()); + assert_eq!(table.len(), 1); + let removed = if sweep { + table.sweep(1, wakes) + } else { + table.remove_quiescent(&1).into_iter().collect() + }; + assert_eq!(removed.len(), 1); + assert!(table.is_empty()); + assert!(table.remove_quiescent(&1).is_none()); + assert_eq!(dropped.get(), 0); + removed + }); + assert_eq!(dropped.get(), 0); + drop(removed); + assert_eq!(dropped.get(), 1); + assert_eq!(table.borrow_mut().next_waiter_id(), Ok(2)); + drop(flight::update(&table, |table, wakes| table.sweep(1, wakes))); + assert_eq!(table.borrow().len(), 1, "replacement remains indexed"); + assert_eq!(dropped.get(), 1); + } + /// Resource drop reentrancy cannot erase the tombstone, and drain wakes run unlocked. #[test] fn completion_tombstone_survives_reentrant_destructor_and_shutdown() { @@ -1318,11 +1405,12 @@ mod coalesce_tests { let dropped = dropped.clone(); move || { let table = table.upgrade().unwrap(); - flight::update(&table, |table, wakes| { - table.sweep(1, wakes); + drop(flight::update(&table, |table, wakes| { + let removed = table.sweep(1, wakes); assert_eq!(table.len(), 1); - assert!(!table.remove_quiescent(&1)); - }); + assert!(table.remove_quiescent(&1).is_none()); + removed + })); dropped.set(true); } }))); @@ -1361,12 +1449,12 @@ mod coalesce_tests { drop(flight::update(&table, |table, _| { table.register_drain(waker) })); - flight::update(&table, |table, wakes| { + drop(flight::update(&table, |table, wakes| { let operations = &mut table.get_mut(&1).unwrap().0; operations.complete(id).unwrap(); assert_eq!(operations.complete(id), Err(Stale)); - table.sweep(1, wakes); - }); + table.sweep(1, wakes) + })); assert!(notified.get()); } From d0cb519716a1ce7ac8c26adb0af56d18fc1ec262 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 22:30:20 +0000 Subject: [PATCH 75/82] fix(flow): release pipe waiter borrows before waker callbacks --- cmd/racer-dataplane/flow/src/pipe.rs | 164 ++++++++++++++++++++++++++- 1 file changed, 161 insertions(+), 3 deletions(-) diff --git a/cmd/racer-dataplane/flow/src/pipe.rs b/cmd/racer-dataplane/flow/src/pipe.rs index 7a21d51c7..dc5670e2c 100644 --- a/cmd/racer-dataplane/flow/src/pipe.rs +++ b/cmd/racer-dataplane/flow/src/pipe.rs @@ -49,12 +49,12 @@ impl Drop for Notify { } } -/// Wake the FIFO head without holding a queue borrow through the callback. +/// Consume the head's notification before calling it without queue or entry borrows. fn wake_front(waiters: &Waiters) { let wake = waiters .borrow() .front() - .and_then(|entry| entry.borrow().clone()); + .and_then(|entry| entry.borrow_mut().take()); if let Some(wake) = wake { wake.wake(); } @@ -221,7 +221,10 @@ impl PipePool

{ if self.quotas.is_stopped() { return Poll::Ready(Err(Error::Unavailable.into())); } - *waiting.entry.borrow_mut() = Some(cx.waker().clone()); + // Clone and drop may reenter; publish before retiring the old waker. + let wake = cx.waker().clone(); + let old = waiting.entry.borrow_mut().replace(wake); + drop(old); if self .waiting .borrow() @@ -737,6 +740,161 @@ mod tests { } } + /// Waker callbacks use thread-local hooks without sharing worker-local state. + mod waker_callbacks { + use super::*; + use std::task::{RawWaker, RawWakerVTable}; + + type Hook = Box; + + thread_local! { + static HOOK: RefCell> = const { RefCell::new(None) }; + } + + /// Remove the hook before calling it so nested callbacks are harmless. + fn invoke(event: &str) { + let hook = HOOK.with(|slot| slot.borrow_mut().take()); + if let Some(hook) = hook { + hook(event); + } + } + + /// Install one callback on this test thread. + pub fn on_callback(hook: impl FnOnce(&str) + 'static) { + HOOK.with(|slot| *slot.borrow_mut() = Some(Box::new(hook))); + } + + /// No data pointer is owned; callbacks only access the calling thread. + fn raw() -> RawWaker { + RawWaker::new( + std::ptr::null(), + &RawWakerVTable::new( + |_| { + invoke("clone"); + raw() + }, + |_| invoke("wake"), + |_| invoke("wake_by_ref"), + |_| invoke("drop"), + ), + ) + } + + /// Build a transferable waker with no shared mutable data. + pub fn waker() -> Waker { + // SAFETY: the vtable never dereferences data and owns no allocation. + // All mutable hooks are thread-local, including on other threads. + unsafe { Waker::from_raw(raw()) } + } + } + + /// Consume the head before callbacks can borrow or notify the same queue. + #[test] + fn review_regression_pipe_waker_front_reentry() { + let entry = Rc::new(RefCell::new(Some(waker_callbacks::waker()))); + let queue: Waiters = Rc::new(RefCell::new(VecDeque::from([entry.clone()]))); + let callback_queue = queue.clone(); + let callback_entry = entry.clone(); + let called = Rc::new(std::cell::Cell::new(false)); + let callback_called = called.clone(); + waker_callbacks::on_callback(move |event| { + let queue = callback_queue.borrow_mut(); + let entry = callback_entry.borrow_mut(); + assert_eq!(event, "wake", "notification must not clone its target"); + assert!(entry.is_none()); + assert_eq!(queue.len(), 1); + drop(entry); + drop(queue); + wake_front(&callback_queue); + callback_called.set(true); + }); + wake_front(&queue); + assert!(called.get()); + assert!(entry.borrow().is_none()); + assert_eq!(queue.borrow().len(), 1); + wake_front(&queue); + } + + /// Dropping a replaced registration may synchronously notify the new target. + #[test] + fn review_regression_pipe_waker_replacement_reentry() { + let pool = new_pool(admission(1)); + let held = pool.acquire().unwrap(); + let mut wait = acquire_wait(&pool); + let old = waker_callbacks::waker(); + assert!( + wait.as_mut() + .poll(&mut Context::from_waker(&old)) + .is_pending() + ); + let queue = pool.waiting.clone(); + waker_callbacks::on_callback(move |event| { + assert_eq!(event, "drop"); + wake_front(&queue); + }); + let counter = Arc::new(WakeCounter::default()); + let new = Waker::from(counter.clone()); + let mut cx = Context::from_waker(&new); + assert!(wait.as_mut().poll(&mut cx).is_pending()); + assert_eq!(counter.count(), 1); + assert!(wait.as_mut().poll(&mut cx).is_pending()); + drop(held); + assert_eq!(counter.count(), 2); + assert!(matches!(wait.as_mut().poll(&mut cx), Poll::Ready(Ok(_)))); + assert!(pool.waiting.borrow().is_empty()); + } + + /// Consumed notifications rearm on each poll and preserve FIFO progress. + #[test] + fn review_regression_pipe_waker_rearms_and_advances_fifo() { + let quotas = admission(1); + let pool = new_pool(quotas.clone()); + let held = pool.acquire().unwrap(); + let registrations = std::cell::Cell::new(0); + let mut first = Box::pin(pool.acquire_wait( + || Ok::<_, Error>(()), + || Ok(|_: &Waker| registrations.set(registrations.get() + 1)), + )); + let mut second = acquire_wait(&pool); + let count = Arc::new(WakeCounter::default()); + let waker = Waker::from(count.clone()); + let mut cx = Context::from_waker(&waker); + assert!(first.as_mut().poll(&mut cx).is_pending()); + assert!(second.as_mut().poll(&mut cx).is_pending()); + wake_front(&pool.waiting); + wake_front(&pool.waiting); + assert_eq!(count.count(), 1); + let queue = pool.waiting.clone(); + waker_callbacks::on_callback(move |event| { + assert_eq!(event, "clone"); + wake_front(&queue); + }); + let reentrant = waker_callbacks::waker(); + assert!( + first + .as_mut() + .poll(&mut Context::from_waker(&reentrant)) + .is_pending() + ); + assert!(first.as_mut().poll(&mut cx).is_pending()); + assert_eq!(registrations.get(), 3); + assert_eq!(count.count(), 1, "pending polls must not self-wake"); + drop(held); + assert_eq!(count.count(), 2); + assert!(second.as_mut().poll(&mut cx).is_pending()); + let Poll::Ready(Ok(first_lease)) = first.as_mut().poll(&mut cx) else { + panic!("head did not acquire returned pipe"); + }; + assert_eq!(registrations.get(), 4); + assert_eq!(count.count(), 3, "head completion must notify successor"); + assert!(second.as_mut().poll(&mut cx).is_pending()); + drop(first_lease); + assert_eq!(count.count(), 4); + assert!(matches!(second.as_mut().poll(&mut cx), Poll::Ready(Ok(_)))); + assert!(pool.waiting.borrow().is_empty()); + assert_eq!(quotas.used(ResourceClass::RequestContext), 0); + } + /// Distinguish application gate rejection from flow-control exhaustion. #[derive(Debug, PartialEq)] enum GateError { From c0b2a4157194185d88990b0dd744e8175716c830 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 22:57:23 +0000 Subject: [PATCH 76/82] fix(flow): reject occupied flight insertion --- cmd/racer-dataplane/flow/src/coalesce.rs | 168 +++++++++++++++----- cmd/racer-dataplane/flow/tests/workflows.rs | 89 +++++++++-- 2 files changed, 208 insertions(+), 49 deletions(-) diff --git a/cmd/racer-dataplane/flow/src/coalesce.rs b/cmd/racer-dataplane/flow/src/coalesce.rs index 99884a6f4..9b13f0310 100644 --- a/cmd/racer-dataplane/flow/src/coalesce.rs +++ b/cmd/racer-dataplane/flow/src/coalesce.rs @@ -719,23 +719,27 @@ pub mod flight { } impl Table { - /// The adapter must have checked admission before inserting a new identity. - pub fn insert(&mut self, key: K, entry: E) { - let sweep_id = self.allocate_sweep_id(); - if let Some(previous) = self - .entries - .insert(key.clone(), Indexed { sweep_id, entry }) - { - self.sweep.remove(&previous.sweep_id); + /// Insert only into a vacant key after the adapter checks admission. + /// Occupied keys, even quiescent ones, return the rejected key and entry. + /// Remove quiescent entries explicitly before admitting replacements. + /// Return rejection from `update` and drop it outside the owner borrow. + #[must_use = "drop the rejected key and entry outside the owner borrow"] + pub fn insert(&mut self, key: K, entry: E) -> Result<(), (K, E)> { + if self.entries.contains_key(&key) { + return Err((key, entry)); } + let sweep_id = self.allocate_sweep_id(); + self.entries + .insert(key.clone(), Indexed { sweep_id, entry }); self.sweep.insert(sweep_id, key); + Ok(()) } /// Find a free internal sweep slot without consuming incarnation IDs. fn allocate_sweep_id(&mut self) -> u64 { // Unlike externally visible incarnation fences, these IDs can be reused // after removal. At most len + 1 probes find a free ID, even after wrap. - // This keeps insert infallible without coupling it to identity(). + // Vacant insertion needs no incarnation allocation or fallible counter. for _ in 0..=self.sweep.len() { self.next_sweep_id = self.next_sweep_id.wrapping_add(1); if !self.sweep.contains_key(&self.next_sweep_id) { @@ -1557,15 +1561,19 @@ pub mod flight { /// Admit an entry with one waiter and return its initial identity. fn insert(table: &mut TestTable, owner: &Rc<()>, key: u32) -> Identity { let identity = table.identity(owner.clone()).unwrap(); - table.insert( - key, - TestEntry { - identity: identity.clone(), - waiters: 1, - canceled: false, - refreshed: 0, - operations: Operations::default(), - }, + assert!( + table + .insert( + key, + TestEntry { + identity: identity.clone(), + waiters: 1, + canceled: false, + refreshed: 0, + operations: Operations::default(), + }, + ) + .is_ok() ); identity } @@ -1602,6 +1610,82 @@ pub mod flight { assert_eq!(table.len(), 1); } + /// Occupied entries retain live resources until explicit completion. + #[test] + fn occupied_live_entry_rejects_insertion() { + occupied_entry_rejects_insertion(false, false); + } + + /// Taking resources does not let insertion bypass the completion tombstone. + #[test] + fn occupied_tombstone_rejects_insertion() { + occupied_entry_rejects_insertion(true, false); + } + + /// Even quiescent entries must be explicitly removed before replacement. + #[test] + fn occupied_quiescent_entry_rejects_insertion() { + occupied_entry_rejects_insertion(true, true); + } + + /// Rejection preserves identity, resource ownership, and sweep membership. + fn occupied_entry_rejects_insertion(take: bool, complete: bool) { + let owner = Rc::new(()); + let mut table = TestTable::default(); + let original = insert(&mut table, &owner, 1); + let live = Rc::new(Cell::new(1)); + let entry = table.get_mut(&1).unwrap(); + entry.waiters = 0; + entry.operations.insert(1, Resource(live.clone())); + if take { + drop(entry.operations.take(1).unwrap()); + } + if complete { + entry.operations.complete(1).unwrap(); + } + let sweep_id = table.entries[&1].sweep_id; + let next_sweep_id = table.next_sweep_id; + let rejected_live = Rc::new(Cell::new(1)); + let mut operations = Operations::default(); + operations.insert(2, Resource(rejected_live.clone())); + let replacement = table.identity(owner).unwrap(); + let rejected = table.insert( + 1, + TestEntry { + identity: replacement, + waiters: 1, + canceled: false, + refreshed: 0, + operations, + }, + ); + assert!(rejected.is_err()); + assert!(original.same_registration(&table.get(&1).unwrap().identity)); + assert_eq!(live.get(), usize::from(!take)); + assert_eq!(rejected_live.get(), 1); + assert_eq!(table.len(), 1); + assert_eq!(table.entries[&1].sweep_id, sweep_id); + assert_eq!(table.next_sweep_id, next_sweep_id); + assert_eq!(table.sweep.len(), 1); + assert_eq!(table.sweep[&sweep_id], 1); + drop(rejected); + assert_eq!(rejected_live.get(), 0); + let removed = table.sweep(1, &mut Vec::new()); + assert_eq!(removed.len(), usize::from(complete)); + if !complete { + let entry = table.get_mut(&1).unwrap(); + assert_eq!(entry.refreshed, 1); + assert_eq!(entry.operations.len(), 1); + if !take { + drop(entry.operations.take(1).unwrap()); + } + entry.operations.complete(1).unwrap(); + assert!(table.remove_quiescent(&1).is_some()); + } + assert!(table.is_empty()); + assert!(table.sweep.is_empty()); + } + /// Request handle that detaches its waiter without completing operations. struct Waiter { table: Rc>, @@ -1766,6 +1850,8 @@ pub mod flight { assert!(table.remove_quiescent(&1).is_none()); insert(&mut table, &owner, 1); let original = table.entries[&1].sweep_id; + table.get_mut(&1).unwrap().waiters = 0; + assert!(table.remove_quiescent(&1).is_some()); insert(&mut table, &owner, 1); let replacement = table.entries[&1].sweep_id; insert(&mut table, &owner, 2); @@ -1814,6 +1900,8 @@ pub mod flight { entry.identity.incarnation = u64::MAX; entry.canceled = true; }); + table.get_mut(&2).unwrap().waiters = 0; + assert!(table.remove_quiescent(&2).is_some()); insert(&mut table, &owner, 2); assert_eq!(table.sweep.len(), 3); drop(table.sweep(3, &mut Vec::new())); @@ -1832,15 +1920,19 @@ pub mod flight { let mut table = TestTable::default(); let identity = table.identity(owner).unwrap(); for key in 1..=3 { - table.insert( - key, - TestEntry { - identity: identity.clone(), - waiters: 1, - canceled: false, - refreshed: 0, - operations: Operations::default(), - }, + assert!( + table + .insert( + key, + TestEntry { + identity: identity.clone(), + waiters: 1, + canceled: false, + refreshed: 0, + operations: Operations::default(), + }, + ) + .is_ok() ); } assert_eq!(table.sweep.len(), 3); @@ -1880,15 +1972,19 @@ pub mod flight { assert!(table.values().all(|entry| entry.refreshed == 1)); table.incarnation = Counter(u64::MAX); assert!(table.identity(owner).is_err()); - table.insert( - 4, - TestEntry { - identity: first, - waiters: 1, - canceled: false, - refreshed: 0, - operations: Operations::default(), - }, + assert!( + table + .insert( + 4, + TestEntry { + identity: first, + waiters: 1, + canceled: false, + refreshed: 0, + operations: Operations::default(), + }, + ) + .is_ok() ); assert_eq!( table.len(), diff --git a/cmd/racer-dataplane/flow/tests/workflows.rs b/cmd/racer-dataplane/flow/tests/workflows.rs index 7e81c63b7..39257389a 100644 --- a/cmd/racer-dataplane/flow/tests/workflows.rs +++ b/cmd/racer-dataplane/flow/tests/workflows.rs @@ -1323,6 +1323,60 @@ mod coalesce_tests { } } + /// Rejected entries leave the transaction before their destructors reenter it. + #[test] + fn occupied_insertion_keeps_destructors_outside_owner_borrow() { + for quiescent in [false, true] { + let table = Rc::new(RefCell::new(flight::Table::::default())); + let original_dropped = Rc::new(Cell::new(false)); + let rejected_dropped = Rc::new(Cell::new(false)); + let unlocked = Rc::new(Cell::new(false)); + let entry = |dropped: Rc>| DroppableEntry { + _resource: ReentrantResource(Some(Box::new({ + let table = Rc::downgrade(&table); + let unlocked = unlocked.clone(); + move || { + let Some(table) = table.upgrade() else { + return; + }; + if let Ok(mut table) = table.try_borrow_mut() { + assert!(table.next_waiter_id().is_ok()); + unlocked.set(true); + } + dropped.set(true); + } + }))), + quiescent, + }; + assert!( + flight::update(&table, |table, _| { + table.insert(1, entry(original_dropped.clone())) + }) + .is_ok() + ); + let rejected = flight::update(&table, |table, _| { + table.insert(1, entry(rejected_dropped.clone())) + }); + assert!(rejected.is_err()); + assert!(!original_dropped.get(), "occupied entry must stay owned"); + assert!(!rejected_dropped.get(), "rejection must return ownership"); + drop(rejected); + assert!(rejected_dropped.get()); + assert!(unlocked.get()); + assert!(!original_dropped.get()); + let removed = flight::update(&table, |table, _| { + assert_eq!(table.len(), 1); + table.get_mut(&1).unwrap().quiescent = true; + table.remove_quiescent(&1) + }); + // The original destructor also reenters after explicit removal. + unlocked.set(false); + drop(removed); + assert!(original_dropped.get()); + assert!(unlocked.get()); + } + } + /// Sweeping transfers entry destruction past the owner transaction. #[test] fn swept_entry_destructor_can_reenter_owner() { @@ -1349,24 +1403,33 @@ mod coalesce_tests { let removed = table.sweep(1, wakes); assert!(removed.is_empty()); assert_eq!(table.next_waiter_id(), Ok(1)); - table.insert( - 1, - DroppableEntry { - _resource: ReentrantResource(Some(Box::new(|| {}))), - quiescent: false, - }, + assert!( + table + .insert( + 1, + DroppableEntry { + _resource: ReentrantResource(Some(Box::new(|| {}))), + quiescent: false, + }, + ) + .is_ok() ); removed })); dropped.set(dropped.get() + 1); } }))); - table.borrow_mut().insert( - 1, - DroppableEntry { - _resource: resource, - quiescent: false, - }, + assert!( + table + .borrow_mut() + .insert( + 1, + DroppableEntry { + _resource: resource, + quiescent: false, + }, + ) + .is_ok() ); let removed = flight::update(&table, |table, wakes| { assert!(table.remove_quiescent(&2).is_none()); @@ -1418,7 +1481,7 @@ mod coalesce_tests { assert_eq!(table.next_waiter_id(), Ok(1)); assert_eq!(table.next_waiter_id(), Ok(2)); let id = table.next_operation_id().unwrap(); - table.insert(1, OwnedEntry::default()); + assert!(table.insert(1, OwnedEntry::default()).is_ok()); table.get_mut(&1).unwrap().0.insert(id, resource); assert_eq!(table.get_mut(&1).unwrap().0.complete(id), Err(Stale)); id From 3016e0a0205d185db4c439619c456242812cd0fe Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 22:56:18 +0000 Subject: [PATCH 77/82] fix(flow): clone coalesce keys outside table borrows --- cmd/racer-dataplane/flow/src/coalesce.rs | 152 ++++++++++++++++++++++- 1 file changed, 148 insertions(+), 4 deletions(-) diff --git a/cmd/racer-dataplane/flow/src/coalesce.rs b/cmd/racer-dataplane/flow/src/coalesce.rs index 9b13f0310..931d39b77 100644 --- a/cmd/racer-dataplane/flow/src/coalesce.rs +++ b/cmd/racer-dataplane/flow/src/coalesce.rs @@ -88,6 +88,22 @@ impl Table { if self.registrations.get() >= capacity.saturating_mul(self.limits.waiters_per_cohort) { return Err(CapacityError); } + let index_key = { + let active = self.active.borrow(); + if active.contains_key(&key) { + None + } else { + if active.len() >= capacity { + return Err(CapacityError); + } + drop(active); + Some(key.clone()) + } + }; + // Key cloning can admit work. Recheck all bounds and cohort membership. + if self.registrations.get() >= capacity.saturating_mul(self.limits.waiters_per_cohort) { + return Err(CapacityError); + } let mut active = self.active.borrow_mut(); let cohort = if let Some(cohort) = active.get(&key) { cohort.clone() @@ -96,7 +112,10 @@ impl Table { return Err(CapacityError); } let cohort = Rc::new(RefCell::new(Cohort::default())); - active.insert(key.clone(), cohort.clone()); + active.insert( + index_key.expect("missing cohort key was cloned"), + cohort.clone(), + ); cohort }; let id = { @@ -357,13 +376,12 @@ pub mod shared { /// Called after miss-only admission, without yielding between get and start. /// The caller must not start a replacement while an entry is present. pub fn start(self: &Rc, key: K, closed: V) -> (Receiver, Completion) { + let index_key = key.clone(); let (send, receive) = oneshot::channel(); let receive = async move { receive.await.unwrap_or(closed) } .boxed_local() .shared(); - self.entries - .borrow_mut() - .insert(key.clone(), receive.clone()); + self.entries.borrow_mut().insert(index_key, receive.clone()); ( receive, Completion { @@ -2032,6 +2050,132 @@ mod tests { )) } + thread_local! { + static ON_KEY_CLONE: RefCell>> = RefCell::new(None); + } + + /// A key whose next clone can call back into its table. + #[derive(Debug, Eq, PartialEq, Ord, PartialOrd, Hash)] + struct CloneKey(u32); + + impl Clone for CloneKey { + /// Run the one-shot callback without retaining the callback slot borrow. + fn clone(&self) -> Self { + let callback = ON_KEY_CLONE.with(|slot| slot.borrow_mut().take()); + if let Some(callback) = callback { + callback(); + } + Self(self.0) + } + } + + /// A clone callback can admit the same key before the outer join resumes. + #[test] + fn key_clone_reentry_joins_current_cohort() { + let table = Rc::new(Table::<_, u32>::new( + Limits { + waiters_per_cohort: 2, + attempts_per_cohort: 1, + }, + 0, + )); + let admitted = Rc::new(RefCell::new(None)); + let nested = table.clone(); + let retained = admitted.clone(); + ON_KEY_CLONE.with(|slot| { + *slot.borrow_mut() = Some(Box::new(move || { + assert_eq!(nested.active_count(), 0); + *retained.borrow_mut() = Some(nested.join(CloneKey(1), 1).unwrap()); + })); + }); + let outer = table.join(CloneKey(1), 1).unwrap(); + let inner = admitted.borrow_mut().take().unwrap(); + assert_eq!(table.active_count(), 1); + assert_eq!(table.registration_count(), 2); + assert_eq!(inner.event(Waker::noop()), Poll::Ready(Event::Lead)); + assert!(outer.event(Waker::noop()).is_pending()); + inner.finish(7); + assert_eq!(outer.event(Waker::noop()), Poll::Ready(Event::Complete(7))); + assert_eq!(table.active_count(), 0); + drop((inner, outer)); + assert_eq!(table.registration_count(), 0); + } + + /// Clone callbacks may fill key, waiter, or completed-reader capacity. + #[test] + fn key_clone_reentry_rechecks_all_bounds() { + for (capacity, waiters, nested_key, complete) in + [(1, 2, 2, false), (2, 1, 1, false), (1, 2, 2, true)] + { + let table = Rc::new(Table::<_, u32>::new( + Limits { + waiters_per_cohort: waiters, + attempts_per_cohort: 1, + }, + 0, + )); + let admitted = Rc::new(RefCell::new(Vec::new())); + let nested = table.clone(); + let retained = admitted.clone(); + ON_KEY_CLONE.with(|slot| { + *slot.borrow_mut() = Some(Box::new(move || { + let count = if complete { capacity * waiters } else { 1 }; + for _ in 0..count { + let registration = nested.join(CloneKey(nested_key), capacity).unwrap(); + if complete { + registration.finish(7); + } + retained.borrow_mut().push(registration); + } + })); + }); + assert!(matches!( + table.join(CloneKey(1), capacity), + Err(CapacityError) + )); + assert_eq!(table.active_count(), usize::from(!complete)); + assert_eq!(table.registration_count(), admitted.borrow().len()); + admitted.borrow_mut().clear(); + assert_eq!(table.active_count(), 0); + assert_eq!(table.registration_count(), 0); + let next = table.join(CloneKey(1), capacity).unwrap(); + assert_eq!(table.registration_count(), 1); + drop(next); + assert_eq!(table.active_count(), 0); + assert_eq!(table.registration_count(), 0); + } + } + + /// Shared key cloning can inspect the table and start unrelated work. + #[test] + fn key_clone_reentry_shared_start_preserves_both_completions() { + use futures::executor::block_on; + + let table = Rc::new(shared::Table::default()); + let admitted = Rc::new(RefCell::new(None)); + let nested = table.clone(); + let retained = admitted.clone(); + ON_KEY_CLONE.with(|slot| { + *slot.borrow_mut() = Some(Box::new(move || { + assert!(nested.is_empty()); + assert!(nested.get(&CloneKey(1)).is_none()); + *retained.borrow_mut() = Some(nested.start(CloneKey(2), 0)); + })); + }); + let (outer, completion) = table.start(CloneKey(1), 0); + let (inner, nested_completion) = admitted.borrow_mut().take().unwrap(); + assert_eq!(table.len(), 2); + completion.finish(7); + assert_eq!(block_on(outer), 7); + assert_eq!(table.len(), 1); + assert!(table.get(&CloneKey(1)).is_none()); + let follower = table.get(&CloneKey(2)).unwrap(); + nested_completion.finish(8); + assert_eq!(block_on(inner), 8); + assert_eq!(block_on(follower), 8); + assert!(table.is_empty()); + } + /// Completion reaches both parked and unpolled readers without deleting replacements. #[test] fn broadcasts_success_and_failure_before_or_after_poll() { From b6093012ba26ff071bbb9b67e1261103d9f52d77 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 23:16:34 +0000 Subject: [PATCH 78/82] fix(flow): drop retired adaptive keys after unlocking --- cmd/racer-dataplane/flow/src/admission.rs | 97 ++++++++++++++++++++++- 1 file changed, 96 insertions(+), 1 deletion(-) diff --git a/cmd/racer-dataplane/flow/src/admission.rs b/cmd/racer-dataplane/flow/src/admission.rs index 19381a824..5351d8d93 100644 --- a/cmd/racer-dataplane/flow/src/admission.rs +++ b/cmd/racer-dataplane/flow/src/admission.rs @@ -217,13 +217,15 @@ impl Adaptive { let indexed = key.clone(); let owned = key.clone(); let now = (self.now)(); + // Drop retired keys after the guard on every exit, since Drop may reenter. + let retired; let mut state = self.state.lock().map_err(|_| Error::Unavailable)?; if state.active >= state.limit { self.observer.event(Event::Rejected); return Err(Error::Overloaded); } if !state.peers.contains_key(key) && state.peers.len() == self.config.capacity { - let retired = state + retired = state .peers .extract_if(.., |_, p| { p.active == 0 @@ -2079,6 +2081,99 @@ mod tests { } } + /// Retired keys can reenter after publication; rejected attempts keep accounting. + #[test] + fn adaptive_retired_key_drop_can_reenter_after_unlock() { + use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering}; + + static OWNER: OnceLock>> = OnceLock::new(); + static WATCH: AtomicBool = AtomicBool::new(false); + static DROPS: AtomicUsize = AtomicUsize::new(0); + static BLOCKED: AtomicUsize = AtomicUsize::new(0); + + #[derive(Clone, Eq, Ord, PartialEq, PartialOrd)] + struct Key(u8); + + impl Drop for Key { + fn drop(&mut self) { + if self.0 != 1 || !WATCH.load(Ordering::SeqCst) { + return; + } + let Some(owner) = OWNER.get().and_then(std::sync::Weak::upgrade) else { + return; + }; + DROPS.fetch_add(1, Ordering::SeqCst); + let Ok(state) = owner.state.try_lock() else { + BLOCKED.fetch_add(1, Ordering::SeqCst); + return; + }; + assert_eq!(state.active, 2); + assert_eq!(state.peers.len(), 2); + assert!(state.peers.keys().all(|key| key.0 != 1)); + assert_eq!(state.peers[&Key(3)].active, 1); + drop(state); + assert!(owner.available(&Key(3))); + } + } + + let owner = Adaptive::new( + Config { + total: 2, + per_key: 1, + backoff: Duration::MAX, + ..config() + }, + Counts::default(), + Instant::now, + ) + .unwrap(); + assert!(OWNER.set(Arc::downgrade(&owner)).is_ok()); + let one = owner.acquire(&Key(1)).unwrap(); + let two = owner.acquire(&Key(2)).unwrap(); + assert!(matches!(owner.acquire(&Key(3)), Err(Error::Overloaded))); + assert_eq!(owner.state.lock().unwrap().active, 2); + drop(one); + WATCH.store(true, Ordering::SeqCst); + + assert!(matches!(owner.acquire(&Key(2)), Err(Error::Overloaded))); + two.observe(Outcome::PeerFailure); + assert!(matches!(owner.acquire(&Key(2)), Err(Error::Unavailable))); + assert!(matches!(owner.acquire(&Key(3)), Err(Error::Overloaded))); + assert_eq!(DROPS.load(Ordering::SeqCst), 0); + { + let mut state = owner.state.lock().unwrap(); + assert_eq!(state.active, 1); + assert_eq!(state.peers.len(), 2); + let idle = state.peers.values_mut().next().unwrap(); + assert_eq!(idle.active, 0); + idle.idle_since = Some(Instant::now() - config().retire_after); + } + assert_eq!(*owner.observer.active.lock().unwrap(), 1); + + let replacement = owner.acquire(&Key(3)).unwrap(); + WATCH.store(false, Ordering::SeqCst); + assert_eq!(DROPS.load(Ordering::SeqCst), 1); + assert_eq!(BLOCKED.load(Ordering::SeqCst), 0); + assert_eq!(*owner.observer.active.lock().unwrap(), 2); + assert_eq!( + *owner.observer.events.lock().unwrap(), + [ + Event::Accepted, + Event::Accepted, + Event::Rejected, + Event::Rejected, + Event::LinkFailure, + Event::CircuitRejected, + Event::Rejected, + Event::Accepted, + ] + ); + drop((two, replacement)); + assert_eq!(owner.state.lock().unwrap().active, 0); + assert_eq!(*owner.observer.active.lock().unwrap(), 0); + assert!(owner.acquire(&Key(3)).is_ok()); + } + /// Stale success cannot undo failure and shared fences retain active work. #[test] fn fences_generation_exclusivity_and_local_pressure() { From 587e513945ae4205d820b5f7213d333d68138988 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 23:14:36 +0000 Subject: [PATCH 79/82] fix(flow): join shared flights admitted during key cloning --- cmd/racer-dataplane/flow/src/coalesce.rs | 188 ++++++++++++++++++-- cmd/racer-dataplane/flow/tests/workflows.rs | 2 +- 2 files changed, 177 insertions(+), 13 deletions(-) diff --git a/cmd/racer-dataplane/flow/src/coalesce.rs b/cmd/racer-dataplane/flow/src/coalesce.rs index 931d39b77..439ce7a69 100644 --- a/cmd/racer-dataplane/flow/src/coalesce.rs +++ b/cmd/racer-dataplane/flow/src/coalesce.rs @@ -374,21 +374,38 @@ pub mod shared { impl Table { /// Called after miss-only admission, without yielding between get and start. - /// The caller must not start a replacement while an entry is present. - pub fn start(self: &Rc, key: K, closed: V) -> (Receiver, Completion) { + /// Recheck after key cloning: an occupied key joins without completion + /// authority. Start work only when a completion owner is returned. + pub fn start( + self: &Rc, + key: K, + closed: V, + ) -> (Receiver, Option>) { let index_key = key.clone(); let (send, receive) = oneshot::channel(); let receive = async move { receive.await.unwrap_or(closed) } .boxed_local() .shared(); - self.entries.borrow_mut().insert(index_key, receive.clone()); + let existing = { + let mut entries = self.entries.borrow_mut(); + // Keep the unused key and receiver outside this borrow on a join. + if let Some(existing) = entries.get(&index_key) { + Some(existing.clone()) + } else { + entries.insert(index_key, receive.clone()); + None + } + }; + if let Some(existing) = existing { + return (existing, None); + } ( receive, - Completion { + Some(Completion { table: self.clone(), key, send, - }, + }), ) } } @@ -419,6 +436,67 @@ pub mod shared { use super::*; use futures::executor::block_on; + #[test] + fn occupied_start_drops_unused_key_and_closed_value_outside_borrow() { + use std::cell::Cell; + + thread_local! { + static ON_KEY_DROP: RefCell>> = RefCell::new(None); + } + + #[derive(Clone, Eq, PartialEq, Ord, PartialOrd)] + struct Key(u32); + + impl Drop for Key { + fn drop(&mut self) { + let callback = ON_KEY_DROP.with(|slot| slot.borrow_mut().take()); + if let Some(callback) = callback { + callback(); + } + } + } + + struct Closed(Option>); + + impl Drop for Closed { + fn drop(&mut self) { + if let Some(callback) = self.0.take() { + callback(); + } + } + } + + let table = Rc::new(Table::default()); + let (first, completion) = table.start(Key(1), Rc::new(Closed(None))); + drop(first); + let key_dropped = Rc::new(Cell::new(false)); + let closed_dropped = Rc::new(Cell::new(false)); + ON_KEY_DROP.with(|slot| { + let table = table.clone(); + let observed = key_dropped.clone(); + *slot.borrow_mut() = Some(Box::new(move || { + assert_eq!(table.entries.borrow_mut().len(), 1); + observed.set(true); + })); + }); + let closed = { + let table = table.clone(); + let observed = closed_dropped.clone(); + Rc::new(Closed(Some(Box::new(move || { + assert_eq!(table.entries.borrow_mut().len(), 1); + observed.set(true); + })))) + }; + let (joined, rejected) = table.start(Key(1), closed); + assert!(rejected.is_none()); + assert!(key_dropped.get()); + assert!(closed_dropped.get()); + let result = Rc::new(Closed(None)); + completion.unwrap().finish(result.clone()); + assert!(Rc::ptr_eq(&block_on(joined), &result)); + assert!(table.is_empty()); + } + /// Final receiver destruction can reenter through its retained closed value. #[test] fn completion_drops_last_receiver_outside_table_borrow() { @@ -467,14 +545,14 @@ pub mod shared { assert!(nested.get(&1).is_none()); let (replacement, completion) = nested.start(1, None); assert_eq!(nested.len(), 1); - completion.finish(None); + completion.unwrap().finish(None); assert!(block_on(replacement).is_none()); observed.set(true); })); }); assert!(!dropped.get()); - completion.finish(None); + completion.unwrap().finish(None); assert!(dropped.get()); assert!(table.is_empty()); assert!(ON_DROP.with(|slot| slot.borrow().is_none())); @@ -488,13 +566,13 @@ pub mod shared { let (first, completion) = table.start(1, Err("closed")); let second = table.get(&1).unwrap(); assert_eq!(table.len(), 1); - completion.finish(value); + completion.unwrap().finish(value); assert!(table.is_empty()); let (next, completion) = table.start(1, Err("closed")); assert_eq!(block_on(first), value); assert_eq!(block_on(second), value); assert_eq!(table.len(), 1); - completion.finish(Ok(Some(8))); + completion.unwrap().finish(Ok(Some(8))); assert_eq!(block_on(next), Ok(Some(8))); } } @@ -507,7 +585,7 @@ pub mod shared { drop(receive); assert_eq!(table.len(), 1); let late = table.get(&1).unwrap(); - completion.finish(Ok(9)); + completion.unwrap().finish(Ok(9)); assert_eq!(block_on(late), Ok(9)); assert!(table.is_empty()); let (receive, completion) = table.start(1, Err("closed")); @@ -2165,17 +2243,103 @@ mod tests { let (outer, completion) = table.start(CloneKey(1), 0); let (inner, nested_completion) = admitted.borrow_mut().take().unwrap(); assert_eq!(table.len(), 2); - completion.finish(7); + completion.unwrap().finish(7); assert_eq!(block_on(outer), 7); assert_eq!(table.len(), 1); assert!(table.get(&CloneKey(1)).is_none()); let follower = table.get(&CloneKey(2)).unwrap(); - nested_completion.finish(8); + nested_completion.unwrap().finish(8); assert_eq!(block_on(inner), 8); assert_eq!(block_on(follower), 8); assert!(table.is_empty()); } + #[test] + fn shared_same_key_clone_reentry_joins_nested_completion() { + use futures::FutureExt; + + for result in [Ok(7), Err("failed")] { + let table = Rc::new(shared::Table::default()); + let admitted = Rc::new(RefCell::new(None)); + let nested = table.clone(); + let retained = admitted.clone(); + ON_KEY_CLONE.with(|slot| { + *slot.borrow_mut() = Some(Box::new(move || { + assert!(nested.get(&CloneKey(1)).is_none()); + *retained.borrow_mut() = Some(nested.start(CloneKey(1), Err("nested closed"))); + })); + }); + assert!(table.get(&CloneKey(1)).is_none()); + let (outer, completion) = table.start(CloneKey(1), Err("outer closed")); + let (inner, nested_completion) = admitted.borrow_mut().take().unwrap(); + let follower = table.get(&CloneKey(1)).unwrap(); + assert_eq!(table.len(), 1); + assert!(completion.is_none()); + nested_completion.unwrap().finish(result); + assert_eq!(inner.now_or_never(), Some(result)); + assert_eq!(outer.now_or_never(), Some(result)); + assert_eq!(follower.now_or_never(), Some(result)); + assert!(table.is_empty()); + let (next, next_completion) = table.start(CloneKey(1), Err("next closed")); + drop(completion); + assert_eq!(table.len(), 1); + next_completion.unwrap().finish(Ok(8)); + assert_eq!(next.now_or_never(), Some(Ok(8))); + assert!(table.is_empty()); + } + } + + #[test] + fn shared_same_key_clone_reentry_completed_before_start_keeps_new_owner() { + use futures::FutureExt; + + let table = Rc::new(shared::Table::default()); + let admitted = Rc::new(RefCell::new(None)); + let nested = table.clone(); + let retained = admitted.clone(); + ON_KEY_CLONE.with(|slot| { + *slot.borrow_mut() = Some(Box::new(move || { + let (receive, completion) = nested.start(CloneKey(1), 99); + completion.unwrap().finish(7); + *retained.borrow_mut() = Some(receive); + assert!(nested.is_empty()); + })); + }); + let (outer, completion) = table.start(CloneKey(1), 99); + assert_eq!(table.len(), 1); + let inner = admitted.borrow_mut().take().unwrap(); + assert_eq!(inner.now_or_never(), Some(7)); + let follower = table.get(&CloneKey(1)).unwrap(); + assert!(outer.clone().now_or_never().is_none()); + completion.unwrap().finish(8); + assert_eq!(outer.now_or_never(), Some(8)); + assert_eq!(follower.now_or_never(), Some(8)); + assert!(table.is_empty()); + } + + #[test] + fn shared_same_key_clone_reentry_lost_sender_stays_occupied() { + use futures::FutureExt; + + let table = Rc::new(shared::Table::default()); + let nested = table.clone(); + ON_KEY_CLONE.with(|slot| { + *slot.borrow_mut() = Some(Box::new(move || { + let (receive, completion) = nested.start(CloneKey(1), 7); + drop((receive, completion)); + assert_eq!(nested.len(), 1); + })); + }); + let (outer, completion) = table.start(CloneKey(1), 99); + assert!(completion.is_none()); + assert_eq!(outer.now_or_never(), Some(7)); + assert_eq!(table.len(), 1); + let (late, completion) = table.start(CloneKey(1), 88); + assert!(completion.is_none()); + assert_eq!(late.now_or_never(), Some(7)); + assert_eq!(table.len(), 1); + } + /// Completion reaches both parked and unpolled readers without deleting replacements. #[test] fn broadcasts_success_and_failure_before_or_after_poll() { diff --git a/cmd/racer-dataplane/flow/tests/workflows.rs b/cmd/racer-dataplane/flow/tests/workflows.rs index 39257389a..2e6986684 100644 --- a/cmd/racer-dataplane/flow/tests/workflows.rs +++ b/cmd/racer-dataplane/flow/tests/workflows.rs @@ -1244,7 +1244,7 @@ mod coalesce_tests { .poll(&mut Context::from_waker(&waker)) .is_pending() ); - complete.finish(7); + complete.unwrap().finish(7); assert_eq!(block_on(receive), 7); let (receive, complete) = replacement.borrow_mut().take().unwrap(); assert_eq!(table.len(), 1); From 18cfd7726edc5d85c3d9ecb0c290e2bba3ae8533 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 23:15:57 +0000 Subject: [PATCH 80/82] fix(flow): retire flight wakers outside owner borrows --- cmd/racer-dataplane/flow/src/coalesce.rs | 26 +++-- cmd/racer-dataplane/flow/tests/workflows.rs | 108 ++++++++++++++++++++ 2 files changed, 127 insertions(+), 7 deletions(-) diff --git a/cmd/racer-dataplane/flow/src/coalesce.rs b/cmd/racer-dataplane/flow/src/coalesce.rs index 439ce7a69..ab5df0795 100644 --- a/cmd/racer-dataplane/flow/src/coalesce.rs +++ b/cmd/racer-dataplane/flow/src/coalesce.rs @@ -603,6 +603,8 @@ pub mod flight { //! Entry hooks own result interpretation and policy. A sweep refreshes each //! selected entry once and removes it only when the hook reports quiescence. //! Callers collect wakes while borrowed and dispatch them after releasing locks. + //! Clone incoming wakers before borrowing. Return retired wakers from `update` + //! and drop them after it returns, including unused copies of the same target. use std::cell::RefCell; use std::collections::{BTreeMap, HashMap, hash_map::RandomState}; @@ -1364,10 +1366,16 @@ pub mod flight { } } - /// Replace a wake target only when it would notify a different task. - pub fn store_waker(slot: &mut Option, waker: &Waker) { - if slot.as_ref().is_none_or(|old| !old.will_wake(waker)) { - *slot = Some(waker.clone()); + /// Store an owned wake target without calling waker clone or drop callbacks. + /// Clone the input before borrowing the owner. Return the retired waker from + /// `update` and drop it after releasing the borrow. For the same target, keep + /// the stored waker and return the unused input instead. + #[must_use = "drop the retired waker outside the owner borrow"] + pub fn store_waker(slot: &mut Option, waker: Waker) -> Option { + if slot.as_ref().is_some_and(|old| old.will_wake(&waker)) { + Some(waker) + } else { + slot.replace(waker) } } @@ -1562,9 +1570,13 @@ pub mod flight { true, true, ); - let slot = &mut core.waiters.get_mut(&id).unwrap().waker; - store_waker(slot, &Waker::from(old.clone())); - store_waker(slot, &Waker::from(latest.clone())); + let old = Waker::from(old.clone()); + let latest = Waker::from(latest.clone()); + let retired = { + let slot = &mut core.waiters.get_mut(&id).unwrap().waker; + (store_waker(slot, old), store_waker(slot, latest)) + }; + drop(retired); } let mut wakes = Vec::new(); core.refresh(true, 9, 8, || now, split, &mut wakes); diff --git a/cmd/racer-dataplane/flow/tests/workflows.rs b/cmd/racer-dataplane/flow/tests/workflows.rs index 2e6986684..65a092c27 100644 --- a/cmd/racer-dataplane/flow/tests/workflows.rs +++ b/cmd/racer-dataplane/flow/tests/workflows.rs @@ -890,6 +890,114 @@ mod coalesce_tests { unsafe { Waker::from_raw(cohort_raw_waker()) } } + /// Register a flight wake target through the caller-owned borrow boundary. + fn register_flight_waker(owner: &RefCell>, waker: &Waker) { + let waker = waker.clone(); + let retired = flight::update(owner, |slot, _| flight::state::store_waker(slot, waker)); + drop(retired); + } + + /// Empty, different, and equal slots transfer ownership without callbacks. + #[test] + fn flight_store_waker_returns_retired_without_callbacks() { + let owner = RefCell::new(None); + let waker = cohort_waker(); + let cloned = Rc::new(Cell::new(false)); + let dropped = Rc::new(Cell::new(false)); + ON_COHORT_CLONE.with(|slot| { + let cloned = cloned.clone(); + *slot.borrow_mut() = Some(Box::new(move || cloned.set(true))); + }); + ON_COHORT_DROP.with(|slot| { + let dropped = dropped.clone(); + *slot.borrow_mut() = Some(Box::new(move || dropped.set(true))); + }); + let duplicate = cohort_waker(); + let retired = flight::update(&owner, |slot, _| { + assert!(flight::state::store_waker(slot, waker).is_none()); + let retired = flight::state::store_waker(slot, duplicate).unwrap(); + assert!(slot.as_ref().unwrap().will_wake(&retired)); + assert!(!cloned.get()); + assert!(!dropped.get()); + retired + }); + drop(retired); + assert!(dropped.get()); + let latest = Waker::from(Arc::new(Reenter)); + let replacement = latest.clone(); + let retired = flight::update(&owner, |slot, _| { + flight::state::store_waker(slot, replacement) + }); + assert!(retired.as_ref().unwrap().will_wake(&cohort_waker())); + assert!(owner.borrow().as_ref().unwrap().will_wake(&latest)); + assert!(!cloned.get()); + ON_COHORT_CLONE.with(|slot| slot.borrow_mut().take()); + } + + /// Clone callbacks can change the slot before registration borrows its owner. + #[test] + fn flight_store_waker_clone_reentry() { + let owner = Rc::new(RefCell::new(None)); + let called = Rc::new(Cell::new(false)); + ON_COHORT_CLONE.with(|slot| { + *slot.borrow_mut() = Some(Box::new({ + let owner = owner.clone(); + let called = called.clone(); + move || { + register_flight_waker(&owner, Waker::noop()); + called.set(true); + } + })); + }); + let waker = cohort_waker(); + register_flight_waker(&owner, &waker); + assert!(called.get()); + assert!(owner.borrow().as_ref().unwrap().will_wake(&waker)); + } + + /// Retired and redundant targets are dropped only after the owner is released. + fn check_flight_store_waker_drop_reentry(identical: bool) { + let owner = Rc::new(RefCell::new(None)); + let waker = cohort_waker(); + let latest = Waker::from(Arc::new(Reenter)); + register_flight_waker(&owner, &waker); + let called = Rc::new(Cell::new(false)); + ON_COHORT_DROP.with(|slot| { + *slot.borrow_mut() = Some(Box::new({ + let owner = owner.clone(); + let called = called.clone(); + let latest = latest.clone(); + move || { + let expected = if identical { + cohort_waker() + } else { + latest.clone() + }; + flight::update(&owner, |slot, _| { + assert!(slot.as_ref().unwrap().will_wake(&expected)); + }); + register_flight_waker(&owner, &latest); + called.set(true); + } + })); + }); + register_flight_waker(&owner, if identical { &waker } else { &latest }); + assert!(called.get()); + assert!(owner.borrow().as_ref().unwrap().will_wake(&latest)); + } + + /// Replacing a different target must return the old waker for deferred drop. + #[test] + fn flight_store_waker_replacement_drop_reentry() { + check_flight_store_waker_drop_reentry(false); + } + + /// Keeping the same target must return the unused incoming waker for deferred drop. + #[test] + fn flight_store_waker_identical_drop_reentry() { + check_flight_store_waker_drop_reentry(true); + } + /// Clone and replacement callbacks may poll, retry, or finish the same cohort. fn check_cohort_waker_event_reentry(on_clone: bool) { for action in 0..3 { From a27f9dab72deaecf4090d69f0f4ca5bc7a9144f8 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 23:33:31 +0000 Subject: [PATCH 81/82] fix(flow): clone quota keys before borrowing the map --- cmd/racer-dataplane/flow/src/lib.rs | 257 +++++++++++++++++++++++++++- 1 file changed, 251 insertions(+), 6 deletions(-) diff --git a/cmd/racer-dataplane/flow/src/lib.rs b/cmd/racer-dataplane/flow/src/lib.rs index 2b09fc871..1bf78968e 100644 --- a/cmd/racer-dataplane/flow/src/lib.rs +++ b/cmd/racer-dataplane/flow/src/lib.rs @@ -626,8 +626,11 @@ impl Quotas

{ if amount == 0 { return Err(Error::InvalidInput); } - let limit = self.limit(class); let local = key.map(|key| self.key_counters(key)).transpose()?; + // Clone before checking limits: application code may admit more work, + // change policy, stop admission, or panic without owning a charge yet. + let key = key.cloned(); + let limit = self.limit(class); if let Some(local) = local.as_ref().filter(|_| mode == AdmissionMode::Ordinary) { let fair = self.fair_limit(class, self.active_keys.load(Ordering::Acquire).max(1)); let used = local.counter(class).used(); @@ -636,9 +639,6 @@ impl Quotas

{ return Err(Error::Overloaded); } } - // Application cloning may panic. Finish it before committing admission - // so every counter increment is paired with a fully constructed owner. - let key = key.cloned(); self.totals .counter(class) .reserve(amount, limit) @@ -692,13 +692,28 @@ impl Quotas

{ if let Some(counts) = keys.get(key).and_then(Weak::upgrade) { return Ok(counts); } + drop(keys); + let retirement_key = key.clone(); + let map_key = key.clone(); + let limit = self.policy.max_keys(); + let mut keys = self.keys.borrow_mut(); + // Either clone may have installed this key or filled the table. + if let Some(counts) = keys.get(key).and_then(Weak::upgrade) { + return Ok(counts); + } + if !keys.contains_key(key) && keys.len() >= limit { + let used = keys.len(); + drop(keys); + self.policy.rejected(Rejection::Keys { used, limit }); + return Err(Error::Overloaded); + } let counts = Arc::new(Counters::keyed( P::Class::COUNT, - key.clone(), + retirement_key, self.active_keys.clone(), self.retired_keys.clone(), )); - keys.insert(key.clone(), Arc::downgrade(&counts)); + keys.insert(map_key, Arc::downgrade(&counts)); Ok(counts) } @@ -1580,6 +1595,236 @@ mod quota_tests { assert_eq!(quotas.policy.rejected.lock().unwrap().len(), 7); } + /// Check nested admission at each key clone and preserve all owned counters. + fn check_quota_key_clone(skip: usize) { + use std::cell::Cell; + + /// Select one clone callback without affecting nested admission. + struct Hook { + skip: usize, + + action: Box, + } + + thread_local! { + static HOOK: RefCell> = const { RefCell::new(None) }; + } + + /// A transferable key with a worker-local clone hook. + #[derive(Debug, Eq, PartialEq, Hash)] + struct Key(u8); + + impl Clone for Key { + /// Release the hook borrow before running application code. + fn clone(&self) -> Self { + let hook = HOOK.with(|slot| { + let mut slot = slot.borrow_mut(); + let hook = slot.as_mut()?; + if hook.skip != 0 { + hook.skip -= 1; + return None; + } + slot.take() + }); + if let Some(hook) = hook { + (hook.action)(); + } + Self(self.0) + } + } + + /// Mutable ceilings and observed rejection facts. + struct ClonePolicy { + limit: Cell, + + max_keys: Cell, + + rejected: RefCell>>, + } + + impl Policy for ClonePolicy { + type Class = Resource; + + type Key = Key; + + /// Read the current aggregate ceiling. + fn limit(&self, _: Resource) -> usize { + self.limit.get() + } + + /// Read the current record ceiling. + fn max_keys(&self) -> usize { + self.max_keys.get() + } + + /// This fixture has no admission waiter. + fn wakes(_: Resource) -> bool { + false + } + + /// This fixture allocates no page backing. + fn covers(_: Resource) -> bool { + false + } + + /// Retain the rejection facts for assertions. + fn rejected(&self, rejection: Rejection) { + self.rejected.borrow_mut().push(rejection); + } + } + + for action in [ + "same", + "keys", + "fair", + "global", + "limit", + "key_limit", + "stop", + "panic", + ] { + for completion in [false, true] { + let quotas = Rc::new(Quotas::new(ClonePolicy { + limit: Cell::new(10), + max_keys: Cell::new(2), + rejected: RefCell::default(), + })); + let held = Rc::new(RefCell::new(Vec::new())); + let nested = quotas.clone(); + let nested_held = held.clone(); + HOOK.with(|slot| { + *slot.borrow_mut() = Some(Hook { + skip, + action: Box::new(move || match action { + "same" => nested_held + .borrow_mut() + .push(nested.reserve(Some(&Key(0)), Resource::Payload, 3).unwrap()), + "keys" => { + for key in [1, 2] { + let result = + nested.reserve(Some(&Key(key)), Resource::Other, 1); + if let Ok(charge) = result { + nested_held.borrow_mut().push(charge); + } else { + assert_eq!(nested.keys.borrow().len(), 2); + } + } + } + "fair" => nested_held + .borrow_mut() + .push(nested.reserve(Some(&Key(1)), Resource::Other, 1).unwrap()), + "global" => nested_held + .borrow_mut() + .push(nested.reserve(None, Resource::Payload, 4).unwrap()), + "limit" => nested.policy.limit.set(6), + "key_limit" => nested.policy.max_keys.set(0), + "stop" => nested.stop(), + "panic" => panic!("key clone failed"), + _ => unreachable!(), + }), + }); + }); + let result = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { + if completion { + quotas.reserve_completion(Some(&Key(0)), Resource::Payload, 7) + } else { + quotas.reserve(Some(&Key(0)), Resource::Payload, 7) + } + })); + assert!(HOOK.with(|slot| slot.borrow().is_none())); + if action == "panic" { + assert!(result.is_err()); + } else { + let result = result.unwrap(); + let expected = match action { + "keys" if skip < 2 || !completion => Some(Error::Overloaded), + "key_limit" if skip < 2 => Some(Error::Overloaded), + "fair" if !completion => Some(Error::Overloaded), + "global" | "limit" => Some(Error::Overloaded), + "stop" if !completion => Some(Error::Unavailable), + _ => None, + }; + if let Some(expected) = expected { + assert!( + matches!(result, Err(error) if error == expected), + "{action}" + ); + } else { + let charge = result.unwrap(); + assert_eq!(charge.key(), Some(&Key(0))); + if action == "same" { + assert!(Arc::ptr_eq( + charge.local.as_ref().unwrap(), + held.borrow()[0].local.as_ref().unwrap(), + )); + } + held.borrow_mut().push(charge); + } + } + if action == "keys" && skip < 2 { + assert!(matches!( + quotas.policy.rejected.borrow().as_slice(), + [ + Rejection::Keys { used: 2, limit: 2 }, + Rejection::Keys { used: 2, limit: 2 } + ] + )); + } + assert!(quotas.keys.borrow().len() <= 2); + let owners = held.borrow(); + for class in [Resource::Payload, Resource::Other] { + let total: usize = owners + .iter() + .filter(|charge| charge.class.index() == class.index()) + .map(Charge::amount) + .sum(); + assert_eq!(quotas.used(class), total, "{action}"); + assert!(total <= quotas.limit(class)); + for charge in owners.iter().filter(|charge| charge.local.is_some()) { + let local = charge.local.as_ref().unwrap(); + let keyed: usize = owners + .iter() + .filter(|other| { + other.key() == charge.key() && other.class.index() == class.index() + }) + .map(Charge::amount) + .sum(); + assert_eq!(local.counter(class).used(), keyed); + } + } + let active = owners + .iter() + .filter_map(Charge::key) + .collect::>() + .len(); + assert_eq!(quotas.active_keys.load(Ordering::Acquire), active); + drop(owners); + held.borrow_mut().clear(); + assert_eq!(quotas.used(Resource::Payload), 0); + assert_eq!(quotas.used(Resource::Other), 0); + assert_eq!(quotas.active_keys.load(Ordering::Acquire), 0); + } + } + } + + /// The retirement-key clone may recursively reserve keyed work. + #[test] + fn quota_key_clone_first_reentry() { + check_quota_key_clone(0); + } + + /// The map-key clone must also run without holding the map borrow. + #[test] + fn quota_key_clone_second_reentry() { + check_quota_key_clone(1); + } + + /// The final charge-key clone must precede admission checks. + #[test] + fn quota_key_clone_charge_reentry() { + check_quota_key_clone(2); + } + /// Idle backing retains two live charges and reports pressure before retry. #[test] fn recycler_retains_two_live_charges_and_reports_pressure_before_retry() { From 3e7cb87132516973bbb8cb712e3a5594efbb3823 Mon Sep 17 00:00:00 2001 From: Jordan Olshevski Date: Wed, 7 Oct 2026 23:32:29 +0000 Subject: [PATCH 82/82] fix(flow): drain idle pipes when admission observes stop --- cmd/racer-dataplane/flow/src/pipe.rs | 152 ++++++++++++++++++++++++--- 1 file changed, 139 insertions(+), 13 deletions(-) diff --git a/cmd/racer-dataplane/flow/src/pipe.rs b/cmd/racer-dataplane/flow/src/pipe.rs index dc5670e2c..cb15864d2 100644 --- a/cmd/racer-dataplane/flow/src/pipe.rs +++ b/cmd/racer-dataplane/flow/src/pipe.rs @@ -165,6 +165,17 @@ impl PipePool

{ self.idle.borrow().len() } + /// Drain idle pipes on stop without releasing resources still owned by leases. + fn check_running(&self) -> Result<()> { + if self.quotas.is_stopped() { + let idle = std::mem::take(&mut *self.idle.borrow_mut()); + // Charge destruction may reenter the pool; release the borrow first. + drop(idle); + return Err(Error::Unavailable); + } + Ok(()) + } + /// FIFO scheduling above immediate raw admission. At most `waiter_limit` wait /// without pipes or new page acquisitions; each entry charges context bytes /// for its guard, queue slot, wake cell, and cancellation registration. @@ -193,9 +204,7 @@ impl PipePool

{ result => return result.map_err(E::from), } } - if self.quotas.is_stopped() { - return Err(Error::Unavailable.into()); - } + self.check_running()?; if self.waiting.borrow().len() >= self.waiter_limit { return Err(Error::Overloaded.into()); } @@ -218,9 +227,7 @@ impl PipePool

{ poll_fn(|cx| { register(cx.waker()); check()?; - if self.quotas.is_stopped() { - return Poll::Ready(Err(Error::Unavailable.into())); - } + self.check_running()?; // Clone and drop may reenter; publish before retiring the old waker. let wake = cx.waker().clone(); let old = waiting.entry.borrow_mut().replace(wake); @@ -243,9 +250,7 @@ impl PipePool

{ /// Reserve before creating descriptors. Exhaustion never waits for a reader. pub fn acquire(&self) -> Result> { - if self.quotas.is_stopped() { - return Err(Error::Unavailable); - } + self.check_running()?; if let Some(resources) = self.idle.borrow_mut().pop() { return Ok(self.lease(resources)); } @@ -644,7 +649,7 @@ mod tests { } } - /// Independent pipe and context limits without quota-driven wake policy. + /// Independent pipe and context limits with pipe-release notifications. struct TestPolicy { pipes: usize, @@ -677,9 +682,9 @@ mod tests { 0 } - /// Pipe return drives wakeups instead of shared quota release. - fn wakes(_: ResourceClass) -> bool { - false + /// Pipe release can notify an explicitly registered quota waiter. + fn wakes(class: ResourceClass) -> bool { + matches!(class, ResourceClass::Pipe) } /// Pipe counts do not authorize userspace page backing. @@ -764,6 +769,11 @@ mod tests { HOOK.with(|slot| *slot.borrow_mut() = Some(Box::new(hook))); } + /// Drop an unused hook before thread-local teardown, without a slot borrow. + pub fn clear() { + drop(HOOK.with(|slot| slot.borrow_mut().take())); + } + /// No data pointer is owned; callbacks only access the calling thread. fn raw() -> RawWaker { RawWaker::new( @@ -1335,6 +1345,122 @@ mod tests { ); } + /// Stop observation drains idle owners but leaves active work and waiters owned. + #[test] + fn review_regression_stop_drains_idle_on_all_acquisition_paths() { + use std::io::Read; + + for path in 0..3 { + let quotas = admission(3); + let pool = new_pool(quotas.clone()); + let idle_read = pool.acquire().unwrap(); + let idle_write = pool.acquire().unwrap(); + let mut held = pool.acquire().unwrap(); + held.try_write(b"live").unwrap(); + let mut read = std::fs::File::from( + idle_read + .resources + .as_ref() + .unwrap() + .read + .as_fd() + .try_clone_to_owned() + .unwrap(), + ); + let write = idle_write + .resources + .as_ref() + .unwrap() + .write + .as_fd() + .try_clone_to_owned() + .unwrap(); + let mut first = acquire_wait(&pool); + let mut second = acquire_wait(&pool); + let mut cx = Context::from_waker(Waker::noop()); + assert!(first.as_mut().poll(&mut cx).is_pending()); + assert!(second.as_mut().poll(&mut cx).is_pending()); + drop((idle_read, idle_write)); + assert_eq!(pool.idle_count(), 2); + assert_eq!(quotas.used(ResourceClass::Pipe), 3); + quotas.stop(); + match path { + 0 => assert!(matches!(pool.acquire(), Err(Error::Unavailable))), + 1 => assert!(matches!( + second.as_mut().poll(&mut cx), + Poll::Ready(Err(Error::Unavailable)) + )), + _ => assert!(matches!( + acquire_wait(&pool).as_mut().poll(&mut cx), + Poll::Ready(Err(Error::Unavailable)) + )), + } + assert_eq!(pool.idle_count(), 0, "path {path} retained idle pipes"); + assert_eq!(quotas.used(ResourceClass::Pipe), 1); + let pending = if path == 1 { 1 } else { 2 }; + assert_eq!(pool.waiting.borrow().len(), pending); + assert_eq!( + quotas.used(ResourceClass::RequestContext), + pending * pool.waiter_bytes() + ); + // Owned duplicates observe peer closure, never a reused descriptor number. + assert_eq!(read.read(&mut [0]).unwrap(), 0); + let mut poll = libc::pollfd { + fd: write.as_raw_fd(), + events: libc::POLLOUT, + revents: 0, + }; + // SAFETY: poll borrows one initialized entry and a live owned descriptor. + assert_eq!(unsafe { libc::poll(&mut poll, 1, 0) }, 1); + assert_ne!(poll.revents & libc::POLLERR, 0); + assert!(matches!( + first.as_mut().poll(&mut cx), + Poll::Ready(Err(Error::Unavailable)) + )); + if path != 1 { + assert!(matches!( + second.as_mut().poll(&mut cx), + Poll::Ready(Err(Error::Unavailable)) + )); + } + assert!(pool.waiting.borrow().is_empty()); + assert_eq!(quotas.used(ResourceClass::RequestContext), 0); + let mut bytes = [0; 4]; + assert_eq!(held.try_read(&mut bytes).unwrap(), 4); + assert_eq!(&bytes, b"live"); + drop(held); + assert_eq!(quotas.used(ResourceClass::Pipe), 0); + assert_eq!(pool.idle_count(), 0); + assert!(matches!(pool.acquire(), Err(Error::Unavailable))); + } + } + + /// Charge destruction may reenter a stopped pool after its idle list is detached. + #[test] + fn review_regression_stop_drain_charge_destructor_reentry() { + let quotas = admission(2); + let pool = Rc::new(new_pool(quotas.clone())); + let first = pool.acquire().unwrap(); + let second = pool.acquire().unwrap(); + drop((first, second)); + quotas.stop(); + let called = Rc::new(std::cell::Cell::new(false)); + let callback_called = called.clone(); + let callback_pool = pool.clone(); + quotas.totals.wake.register(&waker_callbacks::waker()); + waker_callbacks::on_callback(move |event| { + assert_eq!(event, "wake"); + assert_eq!(callback_pool.idle_count(), 0); + assert!(matches!(callback_pool.acquire(), Err(Error::Unavailable))); + callback_called.set(true); + }); + assert!(matches!(pool.acquire(), Err(Error::Unavailable))); + waker_callbacks::clear(); + assert!(called.get()); + assert_eq!(quotas.used(ResourceClass::Pipe), 0); + assert_eq!(pool.idle_count(), 0); + } + /// A lease outliving its pool continues to hold capacity until drop. #[test] fn exhaustion_and_drop_return_capacity_even_after_pool_drop() {