Skip to main content

mz_ore/
pool.rs

1// Copyright Materialize, Inc. and contributors. All rights reserved.
2//
3// Licensed under the Apache License, Version 2.0 (the "License");
4// you may not use this file except in compliance with the License.
5// You may obtain a copy of the License in the LICENSE file at the
6// root of this repository, or online at
7//
8//     http://www.apache.org/licenses/LICENSE-2.0
9//
10// Unless required by applicable law or agreed to in writing, software
11// distributed under the License is distributed on an "AS IS" BASIS,
12// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
13// See the License for the specific language governing permissions and
14// limitations under the License.
15
16//! Prototype buffer pool for dataflow state. See
17//! `doc/developer/design/20260610_buffer_managed_state.md`.
18//!
19//! The pool is the cache: size-class anonymous virtual-memory regions whose
20//! slots hold resident chunks. Slots are scoped to residency — eviction
21//! returns a chunk's slot to the free list along with its physical pages —
22//! so slot demand tracks the resident set (bounded by the budget), not the
23//! potentially unbounded live backlog. Reads are copy-out
24//! ([`ChunkHandle::read_into`]): a resident slot is copied and an evicted
25//! extent decompressed straight into the caller's buffer, all under the
26//! chunk's state lock, so no reference into pool memory escapes the pool and
27//! a read leaves residency untouched. The backing is the swap-backed extent
28//! store of the design's Layer 1: a slot in a pool-owned anonymous-memory
29//! extent arena holding the chunk's lz4-compressed bytes.
30//!
31//! Memory descends a ladder of tiers, each with its own ceiling and each
32//! cheaper to vacate than the one above:
33//!
34//! * **Slots** (uncompressed, free reads) — bounded by the budget; crossing
35//!   it compresses the oldest chunks into extents and releases their slots.
36//! * **Warm free slots** (pages kept for fault-free reuse) — bounded by the
37//!   warm cap.
38//! * **Compressed-resident extents** (reads decompress, no device) — bounded
39//!   by the headroom the RSS target leaves above the first two; crossing it
40//!   pushes the oldest extents to the swap device with `MADV_PAGEOUT`.
41//! * **The swap device** — overflow; reads fault and decompress.
42//!
43//! Residency is a state, not a type. It descends through eviction and
44//! ascends through exactly one transition: an admitting read
45//! ([`ChunkHandle::read_into_admit`]) lifts an evicted chunk back to
46//! `BackedResident` when a slot is free within the budget or stealable from
47//! a clean backed victim of the same class, never by evicting or
48//! compressing anything. Plain reads ([`ChunkHandle::read_into`]) leave
49//! residency untouched. Eviction I/O runs on spill threads when enabled —
50//! `WriteInFlight` marks a chunk whose compression a spill thread owns — and
51//! inline on the evicting caller otherwise. Chunks are immutable after
52//! [`Pool::insert_with`], which is what makes a `BackedResident` slot always
53//! identical to its extent and its eviction free of I/O.
54//!
55//! Freeing an `UnbackedResident` chunk is a pure memory operation — the
56//! design's "never write dead data" win, surfaced as `writes_elided` in
57//! [`PoolStats`]. Budget pressure evicts cold chunks via second-chance
58//! FIFOs banded by the caller-supplied generational depth ([`ChunkHints`]).
59
60mod extent;
61mod region;
62
63use std::collections::VecDeque;
64use std::ops::Range;
65use std::sync::atomic::{AtomicU64, Ordering};
66use std::sync::{Arc, Mutex, MutexGuard, Weak};
67
68use crate::cast::CastFrom;
69use crate::pool::extent::{ExtentArena, Scratch, SwapExtent};
70use crate::pool::region::{Region, SIZE_CLASSES};
71
72/// Virtual reservation per size class. Purely virtual: physical memory
73/// materializes only for slots in use, and slots are scoped to residency,
74/// so this must exceed the largest plausible *resident* set per class, the
75/// budget plus in-flight slack, not the backlog. It is deliberately enormous
76/// (address space costs nothing, and touched pages are bounded by peak
77/// residency) so that no realistic budget, on any machine size, reaches the
78/// heap-fallback path.
79///
80/// NOTE: Seen OoMs with Miri since it actually allocates the capacity.
81const CLASS_CAPACITY_BYTES: usize = if cfg!(miri) { 16 << 20 } else { 1 << 40 };
82
83/// A chunk-provided transform between a chunk's body bytes and the stored
84/// bytes its extent holds. The pool owns scheduling: spill threads, the
85/// residency state machine, cancellation, and the ledger. It invokes the
86/// codec on opaque bytes at the extent boundary, `encode` when backing a
87/// chunk (on a spill thread, or inline under overload) and `decode` when
88/// reading an evicted one, under the chunk's state lock. The pool itself
89/// has no opinion on the stored form: framing, compression, and validation
90/// all belong to the codec.
91///
92/// Implementations must be pure transforms: no locking, no calls back into
93/// the pool (the state lock is held at `decode` sites), and no panic on
94/// bytes their own `encode` produced. `decode` must exactly invert
95/// `encode`, and `encode`'s output must never exceed
96/// [`max_stored_len`]`(body.len())`, the bound the extent store's size
97/// classes are provisioned to.
98pub trait ExtentCodec: std::fmt::Debug + Send + Sync {
99    /// Transforms `body` into its stored form, replacing `out`'s contents.
100    /// `out`'s capacity is reused across calls; implementations size it
101    /// themselves.
102    fn encode(&self, body: &[u8], out: &mut Vec<u8>);
103
104    /// Inverts [`ExtentCodec::encode`]: reconstructs into `body` exactly
105    /// the bytes whose encoding produced `stored`. `body` is exactly the
106    /// original body's length, and implementations must panic on a length
107    /// mismatch rather than truncate or pad.
108    fn decode(&self, stored: &[u8], body: &mut [u8]);
109}
110
111/// The identity [`ExtentCodec`]: the stored form is the body. Encode and
112/// decode are copies, and range reads copy the range directly, so a chunk
113/// stored under this codec pays no compression work in either direction
114/// while remaining fully budgeted and swap-backed like any other extent.
115#[derive(Debug)]
116pub struct IdentityCodec;
117
118/// The [`IdentityCodec`] instance to pass to [`Pool::insert_with`].
119pub static IDENTITY_CODEC: IdentityCodec = IdentityCodec;
120
121impl ExtentCodec for IdentityCodec {
122    fn encode(&self, body: &[u8], out: &mut Vec<u8>) {
123        out.clear();
124        out.extend_from_slice(body);
125    }
126
127    fn decode(&self, stored: &[u8], body: &mut [u8]) {
128        assert_eq!(stored.len(), body.len(), "identity stored form is the body");
129        body.copy_from_slice(stored);
130    }
131}
132
133/// The largest stored form [`ExtentCodec::encode`] may produce for a
134/// `body_len`-byte body: an incompressible-input expansion matching lz4's
135/// worst case plus a four-byte length prefix. The extent store's size-class
136/// ladder is provisioned to this bound, so a codec that exceeds it can
137/// strand payloads with no class to hold them (they degrade to unpageable
138/// heap fallbacks).
139pub fn max_stored_len(body_len: usize) -> usize {
140    4 + body_len + body_len / 255 + 16
141}
142
143/// Advisory placement hints for a chunk, supplied at insert and immutable
144/// thereafter (merges mint new chunks, so a chunk's generation never
145/// changes). Hints steer policy — eviction order and write-behind
146/// candidacy — never correctness: a mislabeled chunk performs worse, while
147/// the budget and residency invariants hold regardless.
148#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
149pub struct ChunkHints {
150    /// Generational depth of the chunk in its producer's merge structure,
151    /// 0 for the youngest generation (and the unannotated default). Deeper
152    /// chunks are treated as colder: preferred write-behind candidates and
153    /// preferred eviction victims, cheap to evict once backed.
154    pub depth: u8,
155}
156
157/// Number of depth bands the eviction queues are split into; depths at or
158/// beyond the last band share it.
159const DEPTH_BANDS: usize = 4;
160
161/// The eviction-queue band for a chunk of `depth`.
162fn band(depth: u8) -> usize {
163    usize::from(depth).min(DEPTH_BANDS - 1)
164}
165
166/// Resident bytes insertions may reserve above `budget` while enforcement
167/// catches up. Slots can transiently hold this much past the budget, so the
168/// RSS target reserves it too.
169fn insert_slack(budget: u64) -> u64 {
170    budget / 8
171}
172
173/// Residency state of a chunk.
174#[derive(Debug, Clone, Copy, PartialEq, Eq)]
175enum Residency {
176    /// Lives only in the pool; no extent copy exists. Freeing it never
177    /// touches the backing store.
178    UnbackedResident,
179    /// Resident, and an identical extent copy exists; eviction releases
180    /// physical pages without I/O.
181    BackedResident,
182    /// Resident and readable, with compression into an extent scheduled on a
183    /// spill thread. Completion moves an evicting chunk to
184    /// [`Residency::Evicted`] and an eagerly backed one to
185    /// [`Residency::BackedResident`]; a free observed at dequeue cancels the
186    /// write instead.
187    WriteInFlight,
188    /// Extent copy only; the chunk holds no slot. The extent itself may
189    /// still be RAM-resident (the compressed tier) or paged out to the swap
190    /// device. Reads decompress the extent straight into the caller's
191    /// buffer and leave the chunk evicted, except that an admitting read
192    /// may lift it back to [`Residency::BackedResident`].
193    Evicted,
194    /// Larger than the largest size class; held as a plain heap allocation,
195    /// always resident. A prototype limitation, not a design state.
196    Oversize,
197}
198
199/// Snapshot of pool counters.
200#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
201pub struct PoolStats {
202    /// Chunks inserted.
203    pub inserts: u64,
204    /// Inserts written directly to an extent because resident admission was full.
205    pub direct_extent_inserts: u64,
206    /// Chunks freed (handle dropped).
207    pub frees: u64,
208    /// Backing writes elided: chunks dead before their compression
209    /// completed, so no extent write happened. Covers chunks freed while
210    /// `UnbackedResident` and chunks freed while queued for a spill thread
211    /// that had not yet compressed them.
212    pub writes_elided: u64,
213    /// Evictions that compressed the chunk into a new extent.
214    pub evictions_compress: u64,
215    /// Evictions of `BackedResident` chunks: pure page release, no I/O.
216    pub evictions_cheap: u64,
217    /// Compressed bytes written into extents.
218    pub extent_bytes_written: u64,
219    /// Evictions handed to spill threads.
220    pub spill_scheduled: u64,
221    /// Compressions cancelled because the chunk was freed while queued or
222    /// in flight, whatever scheduled them. Eager-backing work counts here
223    /// but never in `spill_scheduled`, so this can exceed that counter.
224    pub spill_cancelled: u64,
225    /// Entries currently queued for or being processed by spill threads.
226    pub spill_in_flight: u64,
227    /// Inserts that fell back to the heap because their size class had no
228    /// free slot (the live set outgrew the class reservation). Heap-backed
229    /// chunks behave like oversize ones: always resident, never paged.
230    pub slot_exhausted_fallbacks: u64,
231    /// Inserts whose payload exceeded the largest size class and therefore
232    /// went straight to a heap-backed oversize chunk.
233    pub oversize_payloads: u64,
234    /// Live size-classed chunks across all classes, whatever their residency.
235    /// For backlog-shaped consumers this tracks the un-drained backlog in
236    /// chunks.
237    pub live_chunks: u64,
238    /// Uncompressed bytes of currently resident chunks (including oversize).
239    pub resident_bytes: u64,
240    /// Uncompressed bytes of live oversize chunks.
241    pub oversize_bytes: u64,
242    /// Class bytes of free slots currently kept warm (pages resident for
243    /// fault-free reuse). Bounded by a fraction of the budget; RSS exceeds
244    /// `resident_bytes` by up to this amount.
245    pub warm_bytes: u64,
246    /// Slot allocations served from the warm list: reuses that faulted no
247    /// pages and skipped the kernel's page zeroing.
248    pub warm_reuses: u64,
249    /// Chunks eagerly compressed to `BackedResident` by idle spill threads
250    /// (write-behind): still readable in their slots, with eviction
251    /// pre-paid.
252    pub eager_backs: u64,
253    /// Evicted chunks re-admitted to `BackedResident` by an admitting read
254    /// out of free budget headroom.
255    pub admissions_budget: u64,
256    /// Evicted chunks re-admitted to `BackedResident` by an admitting read
257    /// stealing the slot of a clean backed victim of the same size class.
258    /// The victim becomes `Evicted` with zero I/O and its extent intact.
259    pub admissions_steal: u64,
260    /// Admitting reads of evicted chunks served as a plain decompress
261    /// instead: no budget headroom (or an exhausted size class), and no
262    /// clean victim whose growth the budget could absorb.
263    pub admissions_denied: u64,
264    /// Allocation bytes of compressed extents currently resident — the
265    /// compressed-but-resident middle tier. Bounded by the RSS target;
266    /// exceeding it pages the oldest extents out to the swap device.
267    pub extent_resident_bytes: u64,
268    /// Allocation bytes of resident extents the RSS target cannot push out:
269    /// retry-capped arena extents (the kernel declined the reclaim advice
270    /// until the retry budget ran out) and heap-fallback extents. The
271    /// compressed tier settles above its capacity by this amount.
272    pub extent_unreclaimable_bytes: u64,
273    /// Extents pushed to the swap device by RSS-target enforcement, with
274    /// the whole range observed nonresident afterwards.
275    pub extent_pageouts: u64,
276    /// Pageout passes whose observation found some of the extent's pages
277    /// still mapped: `MADV_PAGEOUT` may decline pages and still succeed, so
278    /// the page table decides. The extent keeps its full resident
279    /// accounting and is retried until its per-extent retry cap. Climbing
280    /// steadily on a loaded pool means pages cannot actually be unmapped to
281    /// the swap device (no swap, or a cgroup that cannot reclaim).
282    pub extent_pageout_incomplete: u64,
283    /// Extent writes that fell back to the heap because their extent-arena
284    /// class had no free slot. Heap-backed extents stay readable but are
285    /// never paged out, so their compressed bytes hold RAM until freed.
286    pub extent_arena_fallbacks: u64,
287}
288
289#[derive(Debug, Default)]
290struct Counters {
291    direct_extent_inserts: AtomicU64,
292    inserts: AtomicU64,
293    spill_scheduled: AtomicU64,
294    spill_cancelled: AtomicU64,
295    slot_exhausted_fallbacks: AtomicU64,
296    oversize_payloads: AtomicU64,
297    frees: AtomicU64,
298    writes_elided: AtomicU64,
299    evictions_compress: AtomicU64,
300    evictions_cheap: AtomicU64,
301    extent_bytes_written: AtomicU64,
302    resident_bytes: AtomicU64,
303    oversize_bytes: AtomicU64,
304    warm_bytes: AtomicU64,
305    warm_reuses: AtomicU64,
306    eager_backs: AtomicU64,
307    admissions_budget: AtomicU64,
308    admissions_steal: AtomicU64,
309    admissions_denied: AtomicU64,
310    extent_resident_bytes: AtomicU64,
311    extent_unreclaimable_bytes: AtomicU64,
312    extent_pageouts: AtomicU64,
313    extent_pageout_incomplete: AtomicU64,
314}
315
316/// A buffer pool over swap-backed extents. Cheap to clone; all clones share
317/// one budget and one backing store.
318#[derive(Debug, Clone)]
319pub struct Pool(Arc<PoolInner>);
320
321/// The shared state behind every [`Pool`] handle. One per process in
322/// practice; [`Pool`] clones and chunk handles share it through an `Arc`,
323/// so it lives until the last handle and spill thread release it.
324///
325/// Lock order: a chunk's `state` mutex may be held while taking any of the
326/// leaf locks — the eviction `queue`, the `extent_queue`, the spill queue,
327/// and the region slot allocators — but never the reverse. The enforcement
328/// and backing scans additionally drop the queue guard before trying a
329/// chunk's state lock (and only ever `try_lock` it), so no path holds a
330/// queue lock while waiting on chunk state. The admitting read's victim
331/// steal is the one place a chunk's state lock is held while probing
332/// another chunk's, and the victim is only ever `try_lock`ed, so two
333/// admitters stealing toward each other skip instead of deadlocking. Reads
334/// copy out under the chunk's state lock — the same lock eviction takes —
335/// so there is no reader-side count and no reader the evictor must account
336/// for.
337#[derive(Debug)]
338struct PoolInner {
339    /// Resident-bytes target, enforced against evictable bytes (resident
340    /// minus heap-backed, which no eviction can reclaim). Atomic so a
341    /// running pool can be retuned in place (operator-driven budget
342    /// changes) without orphaning live handles, which share this value
343    /// through their `Arc<PoolInner>`.
344    budget_bytes: AtomicU64,
345    /// Ceiling on the pool's *total* RSS: slots (the budget) plus warm free
346    /// slots plus compressed-resident extents. The compressed tier's
347    /// capacity derives as `max(0, rss_target - budget - warm cap)`; zero
348    /// (the default) collapses the tier, paging every extent out as soon as
349    /// it is written.
350    rss_target_bytes: AtomicU64,
351    /// One region per entry of [`SIZE_CLASSES`], same order.
352    regions: Vec<Region>,
353    /// The arena backing extents. Shared with every live [`SwapExtent`],
354    /// whose drop returns its slot.
355    extent_arena: Arc<ExtentArena>,
356    /// Second-chance FIFOs of eviction candidates, one per depth band; a
357    /// chunk joins the band of its [`ChunkHints`] depth at insert and again
358    /// on re-admission. Entries for freed chunks go stale in place and are
359    /// dropped by [`PoolInner::prune_queues`].
360    ///
361    /// Two scanners walk them with different obligations, both visiting the
362    /// deepest band first. Budget enforcement is the one that ages chunks:
363    /// it spends the touched bit (second chance) and drops entries it
364    /// evicts. Eager backing ([`PoolInner::back_one`]) rotates visited
365    /// entries to the back but never spends a touched bit, so a backing
366    /// pass shuffles FIFO order without aging any chunk toward eviction.
367    queues: [Mutex<VecDeque<Weak<ChunkMeta>>>; DEPTH_BANDS],
368    /// FIFO of chunks whose extents are resident, oldest first — the
369    /// RSS-target enforcement's victim queue. Entries go stale when an
370    /// extent pages out, is dropped, or its chunk dies; visits drop them,
371    /// and [`PoolInner::prune_extent_queue`] compacts dead-chunk entries
372    /// that under-cap operation never visits.
373    extent_queue: Mutex<VecDeque<Weak<ChunkMeta>>>,
374    /// Number of live size-classed chunks (whatever their residency), which
375    /// is the number of non-stale queue entries across all bands;
376    /// [`PoolInner::prune_queues`] compacts the queues against it.
377    live_chunks: AtomicU64,
378    /// Number of live chunks whose extent is currently resident, including
379    /// unreclaimable extents that deliberately hold no `extent_queue` entry
380    /// (heap-backed and retry-capped ones). It therefore upper-bounds the
381    /// queue's non-stale entries, and [`PoolInner::prune_extent_queue`]'s
382    /// compaction threshold is conservative by the unreclaimable count.
383    extent_residents: AtomicU64,
384    /// Single-flight claim for budget enforcement.
385    enforcing: Mutex<()>,
386    /// Set by an insert turned away from `enforcing`. The holder re-runs its
387    /// pass while it is set, so a caller turned away after the holder's final
388    /// counter read still has its bytes enforced rather than dropped.
389    enforce_pending: std::sync::atomic::AtomicBool,
390    counters: Counters,
391    spill: Spill,
392}
393
394/// Hand-off point between budget enforcement and spill threads. Eviction I/O
395/// (compression and the synchronous-reclaim `pageout`) runs on spill threads
396/// when enabled, keeping multi-millisecond work off the threads that trip the
397/// budget; with no spill threads, eviction runs inline on the caller.
398#[derive(Debug, Default)]
399struct Spill {
400    /// Chunks in `WriteInFlight`, awaiting a spill thread.
401    queue: Mutex<VecDeque<Arc<ChunkMeta>>>,
402    /// Parks idle spill threads. Notified when work lands in `queue`, when
403    /// eager backing turns on, and at shutdown; threads additionally wake
404    /// on a timeout so eager backing scans for write-behind work without a
405    /// dedicated wakeup per candidate.
406    cv: std::sync::Condvar,
407    /// Whether evictions are handed to spill threads. Set when threads are
408    /// first spawned; cleared to fall back to inline eviction.
409    enabled: std::sync::atomic::AtomicBool,
410    /// Whether idle spill threads eagerly compress unbacked chunks to
411    /// `BackedResident` (write-behind); see [`Pool::set_eager_backing`].
412    eager: std::sync::atomic::AtomicBool,
413    /// Number of spill threads spawned (spawn-once; later config changes
414    /// only toggle `enabled`).
415    threads: AtomicU64,
416    /// Queued plus currently-processing entries; `quiesce` waits on zero.
417    in_flight: AtomicU64,
418    /// Test-only lifecycle: production spill threads are immortal (the pool
419    /// is a process singleton), but Miri rejects a test binary exiting with
420    /// live threads, so tests stop and join them.
421    #[cfg(test)]
422    stop: std::sync::atomic::AtomicBool,
423    #[cfg(test)]
424    handles: Mutex<Vec<std::thread::JoinHandle<()>>>,
425}
426
427/// Beyond this many queued or in-flight spill entries, eviction degrades to
428/// inline on the caller: bounded memory overshoot under burst beats an
429/// unbounded queue of still-resident chunks.
430const SPILL_IN_FLIGHT_MAX: usize = 64;
431
432/// What a spill thread does with a chunk once compressed.
433#[derive(Clone, Copy, PartialEq, Eq)]
434enum SpillKind {
435    /// Budget-driven: release the slot, leaving the chunk `Evicted`.
436    Evict,
437    /// Eager write-behind: keep the slot, leaving the chunk
438    /// `BackedResident`.
439    Back,
440}
441
442#[derive(Debug)]
443struct ChunkMeta {
444    pool: Arc<PoolInner>,
445    /// Length in `u64` words; immutable.
446    len: usize,
447    /// Size class for slot allocations; `None` for empty chunks and payloads
448    /// beyond the largest class. Immutable: the chunk's *slot* comes and goes
449    /// with residency, but it is always drawn from this class.
450    class: Option<usize>,
451    /// The insert-time [`ChunkHints`] depth; immutable. Names the eviction
452    /// band the chunk's queue entries belong to.
453    depth: u8,
454    /// The insert-time [`ExtentCodec`]; immutable. Encodes the chunk when it
455    /// is backed and decodes its extent on reads, so it must outlive any
456    /// extent it produced, hence `'static`.
457    codec: &'static dyn ExtentCodec,
458    state: Mutex<ChunkState>,
459}
460
461#[derive(Debug)]
462struct ChunkState {
463    residency: Residency,
464    /// Second-chance bit, set on read and cleared (in lieu of eviction) when
465    /// the budget enforcer first visits the chunk.
466    touched: bool,
467    /// Set when the owning handle is dropped, so a queue entry upgraded
468    /// concurrently with the free cannot touch a recycled slot.
469    freed: bool,
470    /// The chunk's slot index within its class's region, held exactly while
471    /// the chunk occupies pool memory (the resident states and
472    /// `WriteInFlight`). Eviction returns the slot to the region free list.
473    /// Reads copy the slot out under the state lock; no pointer into the
474    /// slot outlives the lock under which it was formed.
475    slot: Option<u32>,
476    /// The backing copy; present exactly in the `BackedResident` and
477    /// `Evicted` states.
478    extent: Option<SwapExtent>,
479    /// The payload of an `Oversize` chunk.
480    oversize: Option<Vec<u64>>,
481}
482
483impl ChunkMeta {
484    /// A fresh chunk in its insert-time state.
485    fn new(
486        pool: &Arc<PoolInner>,
487        len: usize,
488        class: Option<usize>,
489        depth: u8,
490        codec: &'static dyn ExtentCodec,
491        residency: Residency,
492        slot: Option<u32>,
493        oversize: Option<Vec<u64>>,
494    ) -> ChunkMeta {
495        ChunkMeta {
496            pool: Arc::clone(pool),
497            len,
498            class,
499            depth,
500            codec,
501            state: Mutex::new(ChunkState {
502                residency,
503                touched: false,
504                freed: false,
505                slot,
506                extent: None,
507                oversize,
508            }),
509        }
510    }
511
512    fn len_bytes(&self) -> usize {
513        self.len * std::mem::size_of::<u64>()
514    }
515
516    /// Locks the chunk's state.
517    fn state(&self) -> MutexGuard<'_, ChunkState> {
518        self.state.lock().expect("chunk state poisoned")
519    }
520}
521
522/// Handle to one immutable chunk in a [`Pool`]. Dropping the handle frees the
523/// chunk: the slot (if resident) returns to the region free list with its
524/// physical pages released, and the extent (if any) is deallocated,
525/// discarding any swapped copy for free.
526#[derive(Debug)]
527pub struct ChunkHandle {
528    meta: Arc<ChunkMeta>,
529}
530
531// Test hook fired inside `PoolInner::enforce_budget`, between a pass's final
532// counter read and the release of the `enforcing` guard. A test arms it on the
533// thread whose pass it wants to freeze, to interleave a concurrent over-budget
534// insert. One-shot: the hook is taken before it runs, so a re-enforcing pass
535// does not re-arm.
536#[cfg(test)]
537thread_local! {
538    static ENFORCE_BUDGET_HOOK: std::cell::RefCell<Option<Box<dyn FnOnce()>>> =
539        const { std::cell::RefCell::new(None) };
540}
541
542#[cfg(test)]
543fn run_enforce_budget_hook() {
544    let hook = ENFORCE_BUDGET_HOOK.with(|cell| cell.borrow_mut().take());
545    if let Some(hook) = hook {
546        hook();
547    }
548}
549
550impl Pool {
551    /// Creates a pool, reserving one virtual region per size class. The
552    /// pool starts with an unlimited budget — nothing is evicted until
553    /// [`Pool::set_budget`] tunes it.
554    pub fn new() -> std::io::Result<Pool> {
555        Pool::with_class_capacity(CLASS_CAPACITY_BYTES)
556    }
557
558    /// As [`Pool::new`], with a caller-chosen virtual reservation per size
559    /// class. Small reservations let tests exercise slot exhaustion.
560    fn with_class_capacity(class_capacity_bytes: usize) -> std::io::Result<Pool> {
561        let regions = SIZE_CLASSES
562            .iter()
563            .map(|&class_size| Region::new(class_size, class_capacity_bytes))
564            .collect::<std::io::Result<Vec<_>>>()?;
565        let extent_arena = Arc::new(ExtentArena::new(class_capacity_bytes)?);
566        Ok(Pool(Arc::new(PoolInner {
567            budget_bytes: AtomicU64::new(u64::MAX),
568            rss_target_bytes: AtomicU64::new(0),
569            regions,
570            extent_arena,
571            queues: std::array::from_fn(|_| Mutex::new(VecDeque::new())),
572            extent_queue: Mutex::new(VecDeque::new()),
573            live_chunks: AtomicU64::new(0),
574            extent_residents: AtomicU64::new(0),
575            enforcing: Mutex::new(()),
576            enforce_pending: std::sync::atomic::AtomicBool::new(false),
577            counters: Counters::default(),
578            spill: Spill::default(),
579        })))
580    }
581
582    /// Allocates a chunk of `len` words and fills it in place: `fill`
583    /// receives `len` words of unspecified contents and must overwrite all
584    /// of them, so serialization writes its single copy straight into pool
585    /// memory. The returned handle starts `UnbackedResident`, or `Evicted`
586    /// when resident admission is full even after enforcement, in which case
587    /// the payload is compressed to an extent on the calling thread before
588    /// this returns. A zero `len` returns a length-0 handle
589    /// holding no slot; payloads beyond the largest size class fall back to
590    /// a plain heap allocation, always resident, a prototype limitation.
591    /// `hints` steer eviction and write-behind policy; callers without
592    /// placement knowledge pass the default. `codec` is the chunk's
593    /// [`ExtentCodec`], fixed for its lifetime: the pool invokes it whenever
594    /// the chunk moves across the extent boundary, and takes no interest in
595    /// the stored form it produces.
596    ///
597    /// Relies on abort-on-panic: a panic in `fill` that was caught would
598    /// leak the slot and its resident-bytes accounting. All hosting
599    /// binaries abort via `mz_ore::panic::install_enhanced_handler`, and
600    /// pool consumers are dataflow operators, never code hosted under a
601    /// `catch_unwind` boundary the way the optimizer is.
602    pub fn insert_with(
603        &self,
604        len: usize,
605        hints: ChunkHints,
606        codec: &'static dyn ExtentCodec,
607        fill: impl FnOnce(&mut [u64]),
608    ) -> ChunkHandle {
609        let inner = &self.0;
610        inner.counters.inserts.fetch_add(1, Ordering::Relaxed);
611        let len_bytes = len * std::mem::size_of::<u64>();
612        if len == 0 {
613            fill(&mut []);
614            let meta = ChunkMeta::new(
615                inner,
616                0,
617                None,
618                hints.depth,
619                codec,
620                Residency::UnbackedResident,
621                None,
622                None,
623            );
624            return ChunkHandle {
625                meta: Arc::new(meta),
626            };
627        }
628        let class = region::size_class_for(len_bytes);
629        if class.is_none() {
630            inner
631                .counters
632                .oversize_payloads
633                .fetch_add(1, Ordering::Relaxed);
634        }
635        if class.is_some() {
636            if !inner.reserve_insert(len_bytes) {
637                inner.enforce_budget();
638                if !inner.reserve_insert(len_bytes) {
639                    return self.insert_extent(len, class, hints, codec, fill);
640                }
641            }
642        } else {
643            inner
644                .counters
645                .resident_bytes
646                .fetch_add(u64::cast_from(len_bytes), Ordering::Relaxed);
647        }
648        // A class with no free slot degrades to the heap path below: an
649        // unpageable chunk beats a dead replica.
650        let slot = class.and_then(|class| inner.alloc_slot(class, len_bytes));
651        let meta = match (class, slot) {
652            (Some(class), Some(slot)) => {
653                let region = &inner.regions[class];
654                // SAFETY: the freshly allocated slot is at least `len_bytes`
655                // long (the class fits the payload) and is exclusively owned
656                // by this not-yet-shared chunk, so the mutable borrow is
657                // unique; region memory is mapped and writable, and `u64` has
658                // no validity requirements beyond size, so exposing the
659                // unspecified prior contents through `&mut [u64]` is sound.
660                let dst = unsafe {
661                    std::slice::from_raw_parts_mut(region.slot_ptr(slot).cast::<u64>(), len)
662                };
663                // The fill contract (overwrite all `len` words) is
664                // discipline-only. Poison in debug builds so an
665                // under-writing fill reads back as deterministic garbage
666                // instead of a previous occupant's bytes, which the heap
667                // path's zero fill would otherwise mask in tests.
668                #[cfg(debug_assertions)]
669                dst.fill(u64::from_ne_bytes([0xDE; 8]));
670                fill(dst);
671                ChunkMeta::new(
672                    inner,
673                    len,
674                    Some(class),
675                    hints.depth,
676                    codec,
677                    Residency::UnbackedResident,
678                    Some(slot),
679                    None,
680                )
681            }
682            _ => {
683                let mut payload = vec![0u64; len];
684                fill(&mut payload);
685                inner
686                    .counters
687                    .oversize_bytes
688                    .fetch_add(u64::cast_from(len_bytes), Ordering::Relaxed);
689                ChunkMeta::new(
690                    inner,
691                    len,
692                    None,
693                    hints.depth,
694                    codec,
695                    Residency::Oversize,
696                    None,
697                    Some(payload),
698                )
699            }
700        };
701        let meta = Arc::new(meta);
702        if meta.class.is_some() {
703            inner.live_chunks.fetch_add(1, Ordering::Relaxed);
704            inner
705                .queue(band(meta.depth))
706                .push_back(Arc::downgrade(&meta));
707        }
708        inner.enforce_budget();
709        ChunkHandle { meta }
710    }
711
712    /// Inserts a chunk directly as a compressed extent, with no resident slot.
713    ///
714    /// The fallback when the budget denies an insertion even after
715    /// enforcement. The payload is filled into a temporary buffer and
716    /// compressed on the calling thread, and the chunk starts evicted, so a
717    /// denied insertion adds no resident bytes while the enforcer is busy.
718    /// The temporary buffer, up to one size class, is heap memory outside the
719    /// pool's accounting. A chunk inserted here pays for compression even if
720    /// it dies young, so under sustained pressure `writes_elided` undercounts
721    /// the writes a resident insertion would have avoided.
722    fn insert_extent(
723        &self,
724        len: usize,
725        class: Option<usize>,
726        hints: ChunkHints,
727        codec: &'static dyn ExtentCodec,
728        fill: impl FnOnce(&mut [u64]),
729    ) -> ChunkHandle {
730        // Compress synchronously to keep denied insertions from queuing
731        // uncompressed payloads behind an occupied enforcer.
732        let mut words = vec![0; len];
733        // Poisoned for the same reason as the slot path in `insert_with`.
734        #[cfg(debug_assertions)]
735        words.fill(u64::from_ne_bytes([0xDE; 8]));
736        fill(&mut words);
737        let inner = &self.0;
738        let extent = SwapExtent::write(&inner.extent_arena, &words, codec, Scratch::Shrink);
739        drop(words);
740        let meta = Arc::new(ChunkMeta::new(
741            inner,
742            len,
743            class,
744            hints.depth,
745            codec,
746            Residency::Evicted,
747            None,
748            None,
749        ));
750        inner.live_chunks.fetch_add(1, Ordering::Relaxed);
751        inner
752            .counters
753            .direct_extent_inserts
754            .fetch_add(1, Ordering::Relaxed);
755        {
756            let mut state = meta.state();
757            inner.commit_extent(&meta, &mut state, extent);
758        }
759        inner.enforce_or_defer_compressed_cap();
760        ChunkHandle { meta }
761    }
762
763    /// Snapshot of the pool's counters.
764    pub fn stats(&self) -> PoolStats {
765        let c = &self.0.counters;
766        PoolStats {
767            inserts: c.inserts.load(Ordering::Relaxed),
768            direct_extent_inserts: c.direct_extent_inserts.load(Ordering::Relaxed),
769            frees: c.frees.load(Ordering::Relaxed),
770            writes_elided: c.writes_elided.load(Ordering::Relaxed),
771            evictions_compress: c.evictions_compress.load(Ordering::Relaxed),
772            evictions_cheap: c.evictions_cheap.load(Ordering::Relaxed),
773            extent_bytes_written: c.extent_bytes_written.load(Ordering::Relaxed),
774            resident_bytes: c.resident_bytes.load(Ordering::Relaxed),
775            oversize_bytes: c.oversize_bytes.load(Ordering::Relaxed),
776            warm_bytes: c.warm_bytes.load(Ordering::Relaxed),
777            warm_reuses: c.warm_reuses.load(Ordering::Relaxed),
778            eager_backs: c.eager_backs.load(Ordering::Relaxed),
779            admissions_budget: c.admissions_budget.load(Ordering::Relaxed),
780            admissions_steal: c.admissions_steal.load(Ordering::Relaxed),
781            admissions_denied: c.admissions_denied.load(Ordering::Relaxed),
782            extent_resident_bytes: c.extent_resident_bytes.load(Ordering::Relaxed),
783            extent_unreclaimable_bytes: c.extent_unreclaimable_bytes.load(Ordering::Relaxed),
784            extent_pageouts: c.extent_pageouts.load(Ordering::Relaxed),
785            extent_pageout_incomplete: c.extent_pageout_incomplete.load(Ordering::Relaxed),
786            extent_arena_fallbacks: self.0.extent_arena.fallbacks(),
787            spill_scheduled: c.spill_scheduled.load(Ordering::Relaxed),
788            spill_cancelled: c.spill_cancelled.load(Ordering::Relaxed),
789            spill_in_flight: self.0.spill.in_flight.load(Ordering::Relaxed),
790            slot_exhausted_fallbacks: c.slot_exhausted_fallbacks.load(Ordering::Relaxed),
791            oversize_payloads: c.oversize_payloads.load(Ordering::Relaxed),
792            live_chunks: self.0.live_chunks.load(Ordering::Relaxed),
793        }
794    }
795
796    /// Enables or disables off-worker eviction I/O. The first call with
797    /// `threads > 0` spawns that many spill threads (spawn-once: later calls
798    /// only toggle participation); `threads == 0` falls back to inline
799    /// eviction on the caller for subsequent victims, letting any queued
800    /// work drain.
801    pub fn set_spill_threads(&self, threads: usize) {
802        if threads == 0 {
803            self.0.spill.enabled.store(false, Ordering::Relaxed);
804            return;
805        }
806        let spawned = self.0.spill.threads.load(Ordering::Relaxed);
807        if spawned == 0 {
808            let to_spawn = u64::cast_from(threads);
809            if self
810                .0
811                .spill
812                .threads
813                .compare_exchange(0, to_spawn, Ordering::Relaxed, Ordering::Relaxed)
814                .is_ok()
815            {
816                for i in 0..threads {
817                    let inner = Arc::clone(&self.0);
818                    let handle = std::thread::Builder::new()
819                        .name(format!("pool-spill-{i}"))
820                        .spawn(move || inner.spill_worker())
821                        .expect("spawn pool spill thread");
822                    #[cfg(test)]
823                    self.0
824                        .spill
825                        .handles
826                        .lock()
827                        .expect("spill handles poisoned")
828                        .push(handle);
829                    #[cfg(not(test))]
830                    drop(handle);
831                }
832            }
833        }
834        self.0.spill.enabled.store(true, Ordering::Relaxed);
835    }
836
837    /// Enables or disables eager backing: when on, idle spill threads
838    /// compress unbacked chunks to `BackedResident` ahead of pressure, so
839    /// budget-driven eviction becomes a pure page release. Costs CPU on
840    /// chunks that die before eviction would have reached them; pays at
841    /// every pressure event. Only meaningful with spill threads spawned.
842    pub fn set_eager_backing(&self, eager: bool) {
843        self.0.spill.eager.store(eager, Ordering::Relaxed);
844        if eager {
845            self.0.spill.cv.notify_all();
846        }
847    }
848
849    /// Test hook: performs one eager-backing step on the calling thread.
850    /// Returns whether progress was made.
851    #[cfg(test)]
852    fn back_step(&self) -> bool {
853        self.0.back_one()
854    }
855
856    /// Test hook: waits until the spill queue is empty and no entry is being
857    /// processed, so tests observe deterministic post-eviction states.
858    #[cfg(test)]
859    fn quiesce_spill(&self) {
860        while self.0.spill.in_flight.load(Ordering::Relaxed) > 0 {
861            std::thread::yield_now();
862        }
863    }
864
865    /// Test hook: stops and joins the spill threads, so a test binary exits
866    /// with none alive (which Miri requires). Stopped threads process no
867    /// further queued work; call [`Pool::quiesce_spill`] first when the test
868    /// depends on the queue draining.
869    #[cfg(test)]
870    fn join_spill_threads(&self) {
871        self.0.spill.stop.store(true, Ordering::Relaxed);
872        self.0.spill.cv.notify_all();
873        let handles =
874            std::mem::take(&mut *self.0.spill.handles.lock().expect("spill handles poisoned"));
875        for handle in handles {
876            handle.join().expect("spill thread panicked");
877        }
878    }
879
880    /// Test hook: enables spill scheduling without spawning threads, so tests
881    /// drive the queue deterministically via [`Pool::spill_step`].
882    #[cfg(test)]
883    fn enable_spill_without_threads(&self) {
884        self.0.spill.enabled.store(true, Ordering::Relaxed);
885    }
886
887    /// Test hook: processes one queued spill entry on the calling thread.
888    /// Returns whether an entry was processed.
889    #[cfg(test)]
890    fn spill_step(&self) -> bool {
891        let popped = self.0.spill_queue().pop_front();
892        let Some(meta) = popped else {
893            return false;
894        };
895        self.0.spill_process(&meta, SpillKind::Evict);
896        self.0.spill.in_flight.fetch_sub(1, Ordering::Relaxed);
897        true
898    }
899
900    /// Test hook: runs one compressed-cap enforcement pass on the calling
901    /// thread.
902    #[cfg(test)]
903    fn enforce_compressed(&self) {
904        self.0.enforce_compressed_cap();
905    }
906
907    /// Test hook: evicts cold chunks until resident bytes fall to the budget
908    /// or every queued chunk has been visited once. Enforcement runs
909    /// automatically on every insert and budget shrink.
910    #[cfg(test)]
911    fn enforce_budget(&self) {
912        self.0.enforce_budget();
913    }
914
915    /// Test hook: runs one compressed-cap enforcement pass inline on the
916    /// calling thread, where the fake residency observation applies.
917    #[cfg(test)]
918    fn enforce_rss_target(&self) {
919        self.0.enforce_compressed_cap();
920    }
921
922    /// Retunes the resident-bytes budget in place and enforces it. Live
923    /// handles share the new value immediately through their `Arc<PoolInner>`;
924    /// a shrink takes effect by evicting on this call, a grow simply leaves
925    /// more headroom for future inserts.
926    pub fn set_budget(&self, budget_bytes: usize) {
927        let new = u64::cast_from(budget_bytes);
928        let prev = self.0.budget_bytes.swap(new, Ordering::Relaxed);
929        // Config application calls this per worker per tick; only a change
930        // warrants an enforcement pass (a grow needs none, and inserts
931        // enforce continuously anyway).
932        if new < prev {
933            self.0.trim_warm_pool();
934            self.0.enforce_budget();
935        }
936    }
937
938    /// Retunes the ceiling on the pool's total RSS — slots plus warm slots
939    /// plus compressed-resident extents. The compressed tier's capacity is
940    /// the gap above the budget and warm cap; zero (the default) collapses
941    /// the tier, paging extents out as soon as they are written. A shrink
942    /// takes effect by paging out the oldest extents on this call.
943    pub fn set_rss_target(&self, target_bytes: usize) {
944        let new = u64::cast_from(target_bytes);
945        let prev = self.0.rss_target_bytes.swap(new, Ordering::Relaxed);
946        if new < prev {
947            self.0.enforce_compressed_cap();
948        }
949    }
950
951    /// Test-only: the number of entries across the second-chance queues,
952    /// live and stale.
953    #[cfg(test)]
954    fn queue_len(&self) -> usize {
955        (0..DEPTH_BANDS).map(|band| self.0.queue(band).len()).sum()
956    }
957
958    /// Test-only: the number of resident-extent queue entries, live and
959    /// stale.
960    #[cfg(test)]
961    fn extent_queue_len(&self) -> usize {
962        self.0.extent_queue().len()
963    }
964
965    /// Test hook: explicitly evicts one chunk. No-op if the chunk is already
966    /// evicted, in flight, empty, or oversize. With spill threads enabled the
967    /// compression is handed off and completes asynchronously (observable via
968    /// [`Residency::WriteInFlight`]); without them it runs inline.
969    #[cfg(test)]
970    fn evict(&self, handle: &ChunkHandle) {
971        let meta = &handle.meta;
972        let mut state = meta.state();
973        if !meta.pool.spill_handoff(meta, &mut state) {
974            meta.pool.evict_locked(meta, &mut state);
975        }
976        drop(state);
977        meta.pool.enforce_or_defer_compressed_cap();
978    }
979
980    /// Test hook: overwrites every free slot's bytes with `0xDE`. The free
981    /// list keeps a freed slot's old bytes on platforms where
982    /// `MADV_DONTNEED` retains contents (macOS); poisoning lets tests prove
983    /// that reads of evicted chunks decompress from the extent rather than
984    /// passing stale slot memory through.
985    #[cfg(test)]
986    fn poison_free_slots(&self) {
987        for region in &self.0.regions {
988            region.poison_free_slots();
989        }
990    }
991}
992
993impl PoolInner {
994    /// Locks the eviction queue of one depth band.
995    fn queue(&self, band: usize) -> MutexGuard<'_, VecDeque<Weak<ChunkMeta>>> {
996        self.queues[band].lock().expect("pool queue poisoned")
997    }
998
999    /// Locks the resident-extent queue.
1000    fn extent_queue(&self) -> MutexGuard<'_, VecDeque<Weak<ChunkMeta>>> {
1001        self.extent_queue.lock().expect("extent queue poisoned")
1002    }
1003
1004    /// Locks the spill hand-off queue.
1005    fn spill_queue(&self) -> MutexGuard<'_, VecDeque<Arc<ChunkMeta>>> {
1006        self.spill.queue.lock().expect("spill queue poisoned")
1007    }
1008
1009    /// The region behind a slotted chunk's size class.
1010    fn region_of(&self, meta: &ChunkMeta) -> &Region {
1011        &self.regions[meta.class.expect("slotted chunk has a class")]
1012    }
1013
1014    /// Borrows the payload of a slotted chunk.
1015    ///
1016    /// # Safety
1017    ///
1018    /// `slot` must be `meta`'s slot, its contents must be initialized (they
1019    /// are from insert onward), and nothing may write the slot while the
1020    /// borrow lives.
1021    unsafe fn slot_data(&self, meta: &ChunkMeta, slot: u32) -> &[u64] {
1022        let ptr = self
1023            .region_of(meta)
1024            .slot_ptr(slot)
1025            .cast_const()
1026            .cast::<u64>();
1027        // SAFETY: per the function contract; `meta.len` words fit the class
1028        // by construction.
1029        unsafe { std::slice::from_raw_parts(ptr, meta.len) }
1030    }
1031
1032    /// Records a freshly written extent under the chunk's state lock: the
1033    /// compressed-bytes counter, the compressed-tier accounting, and the
1034    /// state's extent field.
1035    fn commit_extent(&self, meta: &Arc<ChunkMeta>, state: &mut ChunkState, extent: SwapExtent) {
1036        self.counters
1037            .extent_bytes_written
1038            .fetch_add(u64::cast_from(extent.comp_len()), Ordering::Relaxed);
1039        // A heap-fallback extent is born permanently capped and counts as
1040        // unreclaimable from the start.
1041        self.note_extent_resident(meta, extent.alloc_size(), !extent.pageout_capped());
1042        state.extent = Some(extent);
1043    }
1044
1045    /// Drops queue entries whose chunk has been freed, detected by their
1046    /// `Weak` no longer holding a live chunk. Each band compacts only when
1047    /// its stale entries outnumber all live chunks (plus a small floor), so
1048    /// the cost amortizes to a constant per insert and the total queue
1049    /// length stays proportional to the number of live slotted chunks even
1050    /// when the pool never comes under budget pressure.
1051    fn prune_queues(&self) {
1052        let live = usize::cast_from(self.live_chunks.load(Ordering::Relaxed));
1053        for band in 0..DEPTH_BANDS {
1054            let mut queue = self.queue(band);
1055            if queue.len() > 2 * live + 16 {
1056                queue.retain(|weak| weak.strong_count() > 0);
1057            }
1058        }
1059    }
1060
1061    /// Reserve insertion bytes before populating a slot. Enforcement can
1062    /// lag by [`insert_slack`] of the budget, or one payload for small
1063    /// budgets. Read admissions use the budget itself and cannot consume
1064    /// this slack.
1065    fn reserve_insert(&self, len_bytes: usize) -> bool {
1066        let len = u64::cast_from(len_bytes);
1067        self.counters
1068            .resident_bytes
1069            .try_update(Ordering::Relaxed, Ordering::Relaxed, |cur| {
1070                let budget = self.budget_bytes.load(Ordering::Relaxed);
1071                let ceiling = budget.saturating_add(insert_slack(budget).max(len));
1072                let next = cur.checked_add(len)?;
1073                let oversize = self.counters.oversize_bytes.load(Ordering::Relaxed);
1074                (next.saturating_sub(oversize) <= ceiling).then_some(next)
1075            })
1076            .is_ok()
1077    }
1078
1079    fn enforce_budget(&self) {
1080        // Single-flight: enforcement runs synchronously on whichever thread
1081        // trips it (every insert), and concurrent passes would
1082        // convoy on the queue mutex doing redundant scans of the same
1083        // candidates. One pass at a time reaches the budget just as well;
1084        // skipped callers hand their bytes to the in-progress pass through
1085        // `enforce_pending`. A poisoned claim means a prior pass panicked.
1086        // Recover and keep enforcing rather than silently disabling the
1087        // budget for the process's lifetime.
1088        let guard = match self.enforcing.try_lock() {
1089            Ok(guard) => guard,
1090            Err(std::sync::TryLockError::WouldBlock) => {
1091                // Release pairs with the holder's Acquire: any `resident_bytes`
1092                // bump this caller made must be visible to the re-read.
1093                self.enforce_pending.store(true, Ordering::Release);
1094                return;
1095            }
1096            Err(std::sync::TryLockError::Poisoned(poisoned)) => poisoned.into_inner(),
1097        };
1098        loop {
1099            self.enforce_budget_inner();
1100            #[cfg(test)]
1101            run_enforce_budget_hook();
1102            // A caller turned away since this pass's counter reads may have
1103            // left bytes unenforced; re-run rather than drop them. The Acquire
1104            // pairs with the turned-away Release so the re-read sees the bump,
1105            // and this swap is the only place the flag is cleared, so no set
1106            // can be lost.
1107            //
1108            // NOTE: a caller turned away between this swap and `drop(guard)`
1109            // sets the flag but finds no re-reader. That residual window is a
1110            // few instructions wide, versus the whole pass before.
1111            if !self.enforce_pending.swap(false, Ordering::Acquire) {
1112                break;
1113            }
1114        }
1115        drop(guard);
1116        // Inline evictions above may have grown the compressed tier.
1117        self.enforce_or_defer_compressed_cap();
1118    }
1119
1120    /// Bytes budget enforcement can actually reclaim: resident bytes minus
1121    /// heap-backed (oversize and class-exhaustion) chunks, which hold no
1122    /// slot and can never be evicted. Enforcing against raw resident bytes
1123    /// would, once unevictable bytes alone exceed the budget, compress
1124    /// every slotted chunk on arrival forever.
1125    fn evictable_bytes(&self) -> u64 {
1126        self.counters
1127            .resident_bytes
1128            .load(Ordering::Relaxed)
1129            .saturating_sub(self.counters.oversize_bytes.load(Ordering::Relaxed))
1130    }
1131
1132    fn enforce_budget_inner(&self) {
1133        self.prune_queues();
1134        // Deepest band first: deep chunks are the coldest, and once eager
1135        // backing has visited them (same order) their eviction is a pure
1136        // page release. The youngest band is reached only when the deeper
1137        // bands cannot satisfy the budget, keeping young data's
1138        // die-before-write chance longest.
1139        for band in (0..DEPTH_BANDS).rev() {
1140            if self.evictable_bytes() <= self.budget_bytes.load(Ordering::Relaxed) {
1141                return;
1142            }
1143            self.enforce_budget_band(band);
1144        }
1145    }
1146
1147    fn enforce_budget_band(&self, band: usize) {
1148        // The queue holds resident chunks only (entries for evicted chunks
1149        // are dropped on visit and never re-added), so a full pass is
1150        // proportional to the resident set. Visit each queued chunk at most
1151        // twice per call: a first visit may only clear the second-chance
1152        // bit, so a second is needed before an over-budget call is
1153        // guaranteed to evict every chunk it saw. The bound keeps contended
1154        // and in-flight entries from spinning this loop forever.
1155        let mut remaining = self.queue(band).len().saturating_mul(2);
1156        while remaining > 0 && self.evictable_bytes() > self.budget_bytes.load(Ordering::Relaxed) {
1157            remaining -= 1;
1158            let popped = self.queue(band).pop_front();
1159            let Some(weak) = popped else {
1160                break;
1161            };
1162            let Some(meta) = weak.upgrade() else {
1163                continue;
1164            };
1165            let requeue = {
1166                // `try_lock`: a chunk mid-eviction or mid-read holds its
1167                // lock for milliseconds; skipping it beats convoying every
1168                // budget enforcer in the process behind one chunk's I/O.
1169                let Ok(mut state) = meta.state.try_lock() else {
1170                    self.queue(band).push_back(weak);
1171                    continue;
1172                };
1173                if state.freed {
1174                    false
1175                } else if matches!(state.residency, Residency::Evicted | Residency::Oversize) {
1176                    // Nothing to evict: drop the entry. A chunk re-enters
1177                    // the queue only when it becomes resident again (insert
1178                    // or re-admission), so the queue stays proportional to
1179                    // the resident set rather than accumulating every chunk
1180                    // ever evicted.
1181                    false
1182                } else if state.touched {
1183                    state.touched = false;
1184                    true
1185                } else if self.spill_handoff(&meta, &mut state) {
1186                    // Stays queued while in flight; once the spill commits to
1187                    // `Evicted`, the next visit drops the entry.
1188                    true
1189                } else {
1190                    self.evict_locked(&meta, &mut state);
1191                    state.residency != Residency::Evicted
1192                }
1193            };
1194            if requeue {
1195                self.queue(band).push_back(weak);
1196            }
1197        }
1198    }
1199
1200    fn evict_locked(&self, meta: &Arc<ChunkMeta>, state: &mut ChunkState) {
1201        let Some(slot) = state.slot else {
1202            return;
1203        };
1204        if state.freed {
1205            return;
1206        }
1207        match state.residency {
1208            Residency::UnbackedResident => {
1209                // SAFETY: the slot belongs to this live chunk and the state
1210                // lock is held, so nothing else touches the slot while this
1211                // borrow is live (reads copy out under the same lock).
1212                let data = unsafe { self.slot_data(meta, slot) };
1213                // Inline eviction runs on whichever thread tripped the
1214                // budget, so the compression scratch must not stay parked
1215                // on it.
1216                let extent =
1217                    SwapExtent::write(&self.extent_arena, data, meta.codec, Scratch::Shrink);
1218                self.counters
1219                    .evictions_compress
1220                    .fetch_add(1, Ordering::Relaxed);
1221                self.commit_extent(meta, state, extent);
1222            }
1223            Residency::BackedResident => {
1224                self.counters
1225                    .evictions_cheap
1226                    .fetch_add(1, Ordering::Relaxed);
1227            }
1228            Residency::WriteInFlight | Residency::Evicted | Residency::Oversize => return,
1229        }
1230        // `release_slot`'s precondition holds: the state lock is held and
1231        // `!freed` was checked above under it.
1232        self.release_slot(meta, state);
1233        state.residency = Residency::Evicted;
1234    }
1235
1236    /// Whether the next eviction should be handed to spill threads: enabled,
1237    /// and the queue is below the backpressure bound (beyond it, callers
1238    /// evict inline rather than growing an unbounded queue of still-resident
1239    /// chunks).
1240    fn spill_eligible(&self) -> bool {
1241        self.spill.enabled.load(Ordering::Relaxed)
1242            && usize::cast_from(self.spill.in_flight.load(Ordering::Relaxed)) < SPILL_IN_FLIGHT_MAX
1243    }
1244
1245    /// Hands a `WriteInFlight` chunk to the spill threads.
1246    fn spill_schedule(&self, meta: Arc<ChunkMeta>) {
1247        self.counters
1248            .spill_scheduled
1249            .fetch_add(1, Ordering::Relaxed);
1250        self.spill.in_flight.fetch_add(1, Ordering::Relaxed);
1251        self.spill_queue().push_back(meta);
1252        self.spill.cv.notify_one();
1253    }
1254
1255    /// Spill-thread main loop. The thread owns an `Arc<PoolInner>`, so the
1256    /// pool (a process-wide singleton in production) lives as long as its
1257    /// threads. Queued (budget-driven) evictions take priority; with eager
1258    /// backing enabled, idle threads compress unbacked chunks to
1259    /// `BackedResident` instead of parking, and park with a timeout once
1260    /// everything reachable is backed.
1261    fn spill_worker(self: Arc<Self>) {
1262        loop {
1263            #[cfg(test)]
1264            if self.spill.stop.load(Ordering::Relaxed) {
1265                return;
1266            }
1267            // Tier-2 pageouts ride the spill threads: every pass through the
1268            // loop (job completion, condvar wakeup, park timeout) trims the
1269            // compressed tier if needed. A single atomic load when under cap.
1270            self.enforce_compressed_cap();
1271            let popped = self.spill_queue().pop_front();
1272            if let Some(meta) = popped {
1273                self.spill_process(&meta, SpillKind::Evict);
1274                self.spill.in_flight.fetch_sub(1, Ordering::Relaxed);
1275                continue;
1276            }
1277            if self.spill.eager.load(Ordering::Relaxed) && self.back_one() {
1278                continue;
1279            }
1280            // Nothing to evict or back: park. Re-checking emptiness under
1281            // the queue lock closes the lost-wakeup window (hand-offs push
1282            // under this lock before notifying); the timeout backstops
1283            // everything else (fresh inserts, tier growth, lost notifies).
1284            let queue = self.spill_queue();
1285            if queue.is_empty() {
1286                let _ = self
1287                    .spill
1288                    .cv
1289                    .wait_timeout(queue, std::time::Duration::from_millis(100))
1290                    .expect("spill queue poisoned");
1291            }
1292        }
1293    }
1294
1295    /// Eagerly compresses one unbacked chunk from the eviction queues into
1296    /// `BackedResident`, returning whether a chunk was backed — `false`
1297    /// means nothing was actionable (queues empty, or the bounded scans
1298    /// found only already-backed, in-flight, contended, or stale entries)
1299    /// and the caller should park rather than rescan. Bands are visited
1300    /// deepest first, mirroring eviction order so the chunks evicted first
1301    /// are the ones whose backing is already pre-paid.
1302    fn back_one(&self) -> bool {
1303        for band in (0..DEPTH_BANDS).rev() {
1304            if self.back_one_from(band) {
1305                return true;
1306            }
1307        }
1308        false
1309    }
1310
1311    /// One bounded backing scan over a single band's queue. Non-actionable
1312    /// entries are requeued or dropped per the same rules budget
1313    /// enforcement uses, except that the second-chance `touched` bit is
1314    /// left alone — backing is not an eviction and must not consume a
1315    /// chunk's reprieve.
1316    fn back_one_from(&self, band: usize) -> bool {
1317        for _ in 0..16 {
1318            let popped = self.queue(band).pop_front();
1319            let Some(weak) = popped else {
1320                return false;
1321            };
1322            let Some(meta) = weak.upgrade() else {
1323                continue;
1324            };
1325            {
1326                let Ok(mut state) = meta.state.try_lock() else {
1327                    self.queue(band).push_back(weak);
1328                    continue;
1329                };
1330                if state.freed {
1331                    continue;
1332                }
1333                match state.residency {
1334                    Residency::Evicted | Residency::Oversize => {
1335                        continue;
1336                    }
1337                    Residency::UnbackedResident => {
1338                        state.residency = Residency::WriteInFlight;
1339                    }
1340                    Residency::BackedResident | Residency::WriteInFlight => {
1341                        self.queue(band).push_back(weak);
1342                        continue;
1343                    }
1344                }
1345            }
1346            self.spill.in_flight.fetch_add(1, Ordering::Relaxed);
1347            self.spill_process(&meta, SpillKind::Back);
1348            self.spill.in_flight.fetch_sub(1, Ordering::Relaxed);
1349            // The chunk remains an eviction candidate (now a cheap one).
1350            self.queue(band).push_back(weak);
1351            return true;
1352        }
1353        false
1354    }
1355
1356    /// Performs (or cancels) one scheduled compression. Lock discipline: the
1357    /// chunk lock is held only to validate and to commit — never across the
1358    /// compression or the `pageout` reclaim, which are the multi-millisecond
1359    /// costs this path exists to keep off budget-enforcing threads.
1360    fn spill_process(&self, meta: &Arc<ChunkMeta>, kind: SpillKind) {
1361        // Validate under the lock, then release it for the I/O. The slot is
1362        // captured under the lock and remains owned by this chunk for the
1363        // unlocked compression: in `WriteInFlight`, eviction skips the chunk
1364        // and `ChunkHandle::drop` defers slot release to this thread.
1365        let slot;
1366        {
1367            let mut state = meta.state();
1368            if state.freed {
1369                // Freed while queued: the deferred cleanup is ours, and the
1370                // chunk dies without ever compressing — the write-behind
1371                // cancellation window. `ChunkHandle::drop` already counted
1372                // the free and the live-chunks decrement.
1373                self.counters
1374                    .spill_cancelled
1375                    .fetch_add(1, Ordering::Relaxed);
1376                self.counters.writes_elided.fetch_add(1, Ordering::Relaxed);
1377                self.release_slot(meta, &mut state);
1378                return;
1379            }
1380            if state.residency != Residency::WriteInFlight {
1381                return;
1382            }
1383            slot = state.slot.expect("write-in-flight chunk has a slot");
1384        }
1385        // SAFETY: the chunk is live (the queue holds an `Arc`) and in
1386        // `WriteInFlight`, so the slot is not recycled (`ChunkHandle::drop`
1387        // defers slot release to this thread in that state) and its contents
1388        // are immutable; concurrent copy-out reads take the state lock and
1389        // read the slot, but nothing writes it.
1390        let data = unsafe { self.slot_data(meta, slot) };
1391        // Spill threads see a steady job stream, so they keep the grown
1392        // compression scratch for the next job.
1393        let extent = SwapExtent::write(&self.extent_arena, data, meta.codec, Scratch::Retain);
1394        let mut state = meta.state();
1395        if state.freed {
1396            // Freed during compression: the extent is garbage; cleanup is
1397            // ours as above. Compression ran, so this is not an elided free.
1398            self.counters
1399                .spill_cancelled
1400                .fetch_add(1, Ordering::Relaxed);
1401            self.release_slot(meta, &mut state);
1402            return;
1403        }
1404        self.commit_extent(meta, &mut state, extent);
1405        match kind {
1406            SpillKind::Back => {
1407                // The slot stays for write-behind: the chunk remains
1408                // readable, and the extent makes a later budget eviction a
1409                // pure page release.
1410                self.counters.eager_backs.fetch_add(1, Ordering::Relaxed);
1411                state.residency = Residency::BackedResident;
1412            }
1413            SpillKind::Evict => {
1414                // `release_slot`'s precondition holds: the state lock is
1415                // held and `!freed` was observed under it.
1416                self.counters
1417                    .evictions_compress
1418                    .fetch_add(1, Ordering::Relaxed);
1419                self.release_slot(meta, &mut state);
1420                state.residency = Residency::Evicted;
1421            }
1422        };
1423        drop(state);
1424        // Counted a fresh resident extent: the tier may need trimming. Kept
1425        // here (rather than relying on the spill loop alone) so the
1426        // threadless test hooks observe deterministic post-commit states.
1427        self.enforce_compressed_cap();
1428    }
1429
1430    /// Releases `state`'s slot — slot returned to the region free list,
1431    /// physical pages discarded unless the slot joins the bounded warm pool —
1432    /// and decrements resident bytes. Releasing pages beyond the warm pool is
1433    /// what keeps RSS aligned with the `resident_bytes` gauge the budget
1434    /// enforcer trusts; the warm pool relaxes that alignment by an explicit,
1435    /// bounded amount (`warm_bytes`, capped at a fraction of the budget) so
1436    /// slot reuse faults no pages and skips the kernel's page zeroing.
1437    ///
1438    /// Precondition: the caller holds the chunk's state lock, and no
1439    /// reference into the slot exists — copy-out reads borrow the slot only
1440    /// under that same lock, and a `WriteInFlight` chunk's unlocked
1441    /// compression read belongs to the spill thread, which is the only
1442    /// caller that releases the slot in that state. This is what makes the
1443    /// `dontneed` below sound, and what makes keeping a warm slot's stale
1444    /// contents safe: the slot's next occupant fully overwrites every byte
1445    /// it reads, satisfying the contents-undefined contract either way.
1446    fn release_slot(&self, meta: &ChunkMeta, state: &mut ChunkState) {
1447        let slot = state.slot.take().expect("slotted chunk");
1448        let region = self.region_of(meta);
1449        let warm = self.try_keep_warm(region.class_size());
1450        if !warm {
1451            // SAFETY: no reference into the slot exists (the function-level
1452            // precondition, established under the held state lock).
1453            unsafe {
1454                region::dontneed(region.slot_ptr(slot), region.class_size());
1455            }
1456        }
1457        region.free(slot, warm);
1458        self.counters
1459            .resident_bytes
1460            .fetch_sub(u64::cast_from(meta.len_bytes()), Ordering::Relaxed);
1461    }
1462
1463    /// The warm pool's byte ceiling: an eighth of the budget, clamped at an
1464    /// absolute maximum. The fraction sizes fault amortization at small
1465    /// budgets; the clamp keeps large budgets from parking gigabytes of idle
1466    /// warm slots no fault rate could justify.
1467    fn warm_cap(&self) -> u64 {
1468        (self.budget_bytes.load(Ordering::Relaxed) / 8).min(1 << 30)
1469    }
1470
1471    /// Cools warm free slots until `warm_bytes` falls to the warm cap. A
1472    /// budget shrink lowers the cap, and warm capacity is checked only when
1473    /// a slot is freed, so without this pass slots parked under the old cap
1474    /// would hold their pages until same-class reuse happened to drain them,
1475    /// exactly when the shrink wanted the memory back.
1476    fn trim_warm_pool(&self) {
1477        let mut over = self
1478            .counters
1479            .warm_bytes
1480            .load(Ordering::Relaxed)
1481            .saturating_sub(self.warm_cap());
1482        for region in &self.regions {
1483            if over == 0 {
1484                return;
1485            }
1486            let cooled = u64::cast_from(region.cool_warm_slots(usize::cast_from(over)));
1487            self.counters
1488                .warm_bytes
1489                .fetch_sub(cooled, Ordering::Relaxed);
1490            over = over.saturating_sub(cooled);
1491        }
1492    }
1493
1494    /// Claims warm-pool capacity for a slot of `class_size` bytes, returning
1495    /// whether the slot may keep its pages. The RSS overshoot the warm pool
1496    /// introduces is bounded by [`PoolInner::warm_cap`] and visible as the
1497    /// `warm_bytes` stat.
1498    fn try_keep_warm(&self, class_size: usize) -> bool {
1499        let cap = self.warm_cap();
1500        let class_bytes = u64::cast_from(class_size);
1501        self.counters
1502            .warm_bytes
1503            .try_update(Ordering::Relaxed, Ordering::Relaxed, |cur| {
1504                (cur + class_bytes <= cap).then_some(cur + class_bytes)
1505            })
1506            .is_ok()
1507    }
1508
1509    /// Allocates a slot in `class` for a payload of `len_bytes` with
1510    /// warm-pool accounting (a warm allocation is counted as a reuse and
1511    /// trimmed to the payload), or `None` when the class has no free slot.
1512    fn try_alloc_slot(&self, class: usize, len_bytes: usize) -> Option<u32> {
1513        let (index, warm) = self.regions[class].alloc()?;
1514        if warm {
1515            let class_bytes = u64::cast_from(self.regions[class].class_size());
1516            self.counters
1517                .warm_bytes
1518                .fetch_sub(class_bytes, Ordering::Relaxed);
1519            self.counters.warm_reuses.fetch_add(1, Ordering::Relaxed);
1520            // A warm slot keeps the prior occupant's resident pages, which
1521            // may extend past the new payload while the ledger credits only
1522            // `len_bytes`.
1523            self.trim_slot_tail(class, index, len_bytes);
1524        }
1525        Some(index)
1526    }
1527
1528    /// Releases a slot's pages beyond the first `len_bytes` (rounded up to
1529    /// a page), so a slot reused for a smaller payload does not keep its
1530    /// prior occupant's tail pages resident with no bytes in the ledger to
1531    /// answer for them.
1532    ///
1533    /// Precondition: the caller exclusively owns the slot (freshly
1534    /// allocated, or taken from a victim under the victim's state lock)
1535    /// with no reference into it.
1536    fn trim_slot_tail(&self, class: usize, slot: u32, len_bytes: usize) {
1537        let region = &self.regions[class];
1538        // Hugepage-class slots trim at huge-page granularity: a base-page
1539        // trim would split the slot's `MADV_HUGEPAGE` folios, and khugepaged
1540        // may later re-collapse a partially trimmed range, re-instantiating
1541        // pages the ledger counts as released. Whole-folio trims leave no
1542        // partial folio to split or resurrect.
1543        let granule = if region.class_size() >= region::HUGE_PAGE {
1544            region::HUGE_PAGE
1545        } else {
1546            region::page_size()
1547        };
1548        let keep = len_bytes.next_multiple_of(granule).min(region.class_size());
1549        let tail = region.class_size() - keep;
1550        if tail == 0 {
1551            return;
1552        }
1553        // SAFETY: the caller exclusively owns the slot per the
1554        // precondition, and `keep + tail` is exactly the class size, so the
1555        // range stays within the slot.
1556        unsafe {
1557            region::dontneed(region.slot_ptr(slot).add(keep), tail);
1558        }
1559    }
1560
1561    /// Allocates a slot in `class` for an insert: as
1562    /// [`PoolInner::try_alloc_slot`], with an exhausted class counted as a
1563    /// heap fallback for a `len_bytes` payload (warned about once). `None`
1564    /// means the caller must degrade to the heap.
1565    fn alloc_slot(&self, class: usize, len_bytes: usize) -> Option<u32> {
1566        match self.try_alloc_slot(class, len_bytes) {
1567            Some(index) => Some(index),
1568            None => {
1569                self.counters
1570                    .slot_exhausted_fallbacks
1571                    .fetch_add(1, Ordering::Relaxed);
1572                static EXHAUSTED_ONCE: std::sync::Once = std::sync::Once::new();
1573                EXHAUSTED_ONCE.call_once(|| {
1574                    tracing::warn!(
1575                        len_bytes,
1576                        "buffer pool size class exhausted; falling back to heap chunks \
1577                         (raise the pool's per-class virtual reservation)",
1578                    );
1579                });
1580                None
1581            }
1582        }
1583    }
1584
1585    /// Acquires a slot for re-admitting an evicted chunk, from free budget
1586    /// headroom or by stealing a clean backed victim's slot, never by
1587    /// evicting or compressing anything. `None` counts a denied admission.
1588    /// On success the admitted chunk's resident-bytes accounting and the
1589    /// admission counter are settled, and the caller (who holds the chunk's
1590    /// state lock) owns the slot: its contents are unspecified (fresh,
1591    /// warm, or the victim's stale bytes) and must be fully overwritten.
1592    fn admit_slot(&self, meta: &ChunkMeta) -> Option<u32> {
1593        let class = meta.class.expect("evicted chunk has a class");
1594        let len_bytes = u64::cast_from(meta.len_bytes());
1595        // Free budget first: reserve the bytes, then a slot. The
1596        // reservation never pushes resident bytes past the budget, and a
1597        // class with no free slot hands the reservation back rather than
1598        // evicting anything to make room. The headroom test uses evictable
1599        // bytes, matching budget enforcement: unevictable heap-backed bytes
1600        // must not permanently veto budget-path admissions the enforcer
1601        // would never need to undo.
1602        let reserved = self
1603            .counters
1604            .resident_bytes
1605            .try_update(Ordering::Relaxed, Ordering::Relaxed, |cur| {
1606                // Loaded inside the closure so a CAS retry sees oversize
1607                // frees that landed since the last attempt.
1608                let oversize = self.counters.oversize_bytes.load(Ordering::Relaxed);
1609                let next = cur.checked_add(len_bytes)?;
1610                (next.saturating_sub(oversize) <= self.budget_bytes.load(Ordering::Relaxed))
1611                    .then_some(next)
1612            })
1613            .is_ok();
1614        if reserved {
1615            if let Some(slot) = self.try_alloc_slot(class, meta.len_bytes()) {
1616                self.counters
1617                    .admissions_budget
1618                    .fetch_add(1, Ordering::Relaxed);
1619                return Some(slot);
1620            }
1621            self.counters
1622                .resident_bytes
1623                .fetch_sub(len_bytes, Ordering::Relaxed);
1624        }
1625        if let Some(slot) = self.steal_clean_victim(class, meta.len_bytes()) {
1626            // The slot's physical pages transfer deliberately, but only up
1627            // to the admitted payload: the victim's pages past it would
1628            // stay resident with no ledger bytes to answer for them.
1629            self.trim_slot_tail(class, slot, meta.len_bytes());
1630            self.counters
1631                .admissions_steal
1632                .fetch_add(1, Ordering::Relaxed);
1633            return Some(slot);
1634        }
1635        self.counters
1636            .admissions_denied
1637            .fetch_add(1, Ordering::Relaxed);
1638        None
1639    }
1640
1641    /// Takes the slot of a clean victim in `class` for an admitted payload
1642    /// of `admitted_len_bytes`: a `BackedResident` chunk with a clear
1643    /// touched bit, whose extent already duplicates its slot, so the victim
1644    /// transitions to `Evicted` with zero I/O, its extent intact, and its
1645    /// queue entry dropped. The returned slot keeps its physical pages (no
1646    /// `dontneed`, no free-list round trip); they hold the victim's stale
1647    /// bytes. `None` when the bounded scan finds no such victim, or none
1648    /// whose growth the budget can absorb.
1649    ///
1650    /// The caller holds its own chunk's state lock. The scan follows the
1651    /// enforcement discipline (deepest band first, queue guard dropped
1652    /// before any chunk lock, victims only ever `try_lock`ed), which is
1653    /// what keeps the chunk-lock-while-probing-chunk-lock window
1654    /// deadlock-free: two admitters stealing toward each other both fail
1655    /// the `try_lock` and skip. Unlike enforcement, the scan rotates
1656    /// unsuitable entries (touched, wrong class, unbacked) to the back
1657    /// without spending touched bits, shuffling FIFO order the way the
1658    /// backing scan does.
1659    fn steal_clean_victim(&self, class: usize, admitted_len_bytes: usize) -> Option<u32> {
1660        // Bound on entries examined, per band rather than shared across the
1661        // scan: a deep band dense with touched resident chunks would
1662        // otherwise spend the whole scan on hopeless candidates and starve
1663        // the shallower bands where eager backing stocks the clean victims.
1664        // Eight visits absorb a few lock-busy or freshly touched entries
1665        // without degrading a hopeless scan into a full queue walk.
1666        const VISITS_PER_BAND: usize = 8;
1667        for band in (0..DEPTH_BANDS).rev() {
1668            let mut visits = VISITS_PER_BAND;
1669            while visits > 0 {
1670                let popped = self.queue(band).pop_front();
1671                let Some(weak) = popped else {
1672                    // Band exhausted; the next band has its own budget.
1673                    break;
1674                };
1675                let Some(meta) = weak.upgrade() else {
1676                    // Stale entries drop for free and do not spend a visit.
1677                    continue;
1678                };
1679                visits -= 1;
1680                let Ok(mut state) = meta.state.try_lock() else {
1681                    self.queue(band).push_back(weak);
1682                    continue;
1683                };
1684                if state.freed {
1685                    continue;
1686                }
1687                match state.residency {
1688                    // Entries for non-resident chunks drop, as in
1689                    // enforcement.
1690                    Residency::Evicted | Residency::Oversize => continue,
1691                    Residency::UnbackedResident | Residency::WriteInFlight => {
1692                        self.queue(band).push_back(weak);
1693                        continue;
1694                    }
1695                    Residency::BackedResident => {}
1696                }
1697                if state.touched || meta.class != Some(class) {
1698                    self.queue(band).push_back(weak);
1699                    continue;
1700                }
1701                // Settle the ledger in one step: the victim's bytes out, the
1702                // admitted payload's in. A steal that grows resident bytes
1703                // is an admission and must fit the budget (against evictable
1704                // bytes, as everywhere); a shrinking steal always may
1705                // proceed. On failure the victim is requeued untouched.
1706                let victim_len = u64::cast_from(meta.len_bytes());
1707                let admitted_len = u64::cast_from(admitted_len_bytes);
1708                let settled = self
1709                    .counters
1710                    .resident_bytes
1711                    .try_update(Ordering::Relaxed, Ordering::Relaxed, |cur| {
1712                        let next = cur.checked_add(admitted_len)?.saturating_sub(victim_len);
1713                        let oversize = self.counters.oversize_bytes.load(Ordering::Relaxed);
1714                        (next <= cur
1715                            || next.saturating_sub(oversize)
1716                                <= self.budget_bytes.load(Ordering::Relaxed))
1717                        .then_some(next)
1718                    })
1719                    .is_ok();
1720                if !settled {
1721                    self.queue(band).push_back(weak);
1722                    continue;
1723                }
1724                let slot = state.slot.take().expect("backed chunk has a slot");
1725                state.residency = Residency::Evicted;
1726                return Some(slot);
1727            }
1728        }
1729        None
1730    }
1731
1732    /// Capacity of the compressed-resident tier: the RSS target's headroom
1733    /// above the slot budget, the insertion slack currently in use, and the
1734    /// warm cap. With no target set the tier has zero capacity, so extents
1735    /// page out as soon as they are written.
1736    fn compressed_cap(&self) -> u64 {
1737        let target = self.rss_target_bytes.load(Ordering::Relaxed);
1738        let budget = self.budget_bytes.load(Ordering::Relaxed);
1739        let floor = budget
1740            .saturating_add(self.insert_slack_in_use(budget))
1741            .saturating_add(self.warm_cap());
1742        target.saturating_sub(floor)
1743    }
1744
1745    /// Slot bytes reserved above `budget` by insertions that outran
1746    /// enforcement, at most [`insert_slack`]. Oversize payloads live on the
1747    /// heap and are excluded, as in [`PoolInner::reserve_insert`].
1748    fn insert_slack_in_use(&self, budget: u64) -> u64 {
1749        let resident = self.counters.resident_bytes.load(Ordering::Relaxed);
1750        let oversize = self.counters.oversize_bytes.load(Ordering::Relaxed);
1751        resident
1752            .saturating_sub(oversize)
1753            .saturating_sub(budget)
1754            .min(insert_slack(budget))
1755    }
1756
1757    /// Counts a newly resident extent (written, or revived by a read)
1758    /// against the compressed tier. A reclaimable extent additionally
1759    /// enqueues its chunk for RSS-target enforcement; an unreclaimable one
1760    /// (a heap-fallback extent, which is never advised out) counts against
1761    /// the unreclaimable gauge instead and stays out of the queue, so
1762    /// enforcement never walks entries it cannot act on. Callers hold the
1763    /// chunk's state lock with the extent present and resident, and follow
1764    /// up with [`PoolInner::enforce_compressed_cap`] once the lock is
1765    /// released.
1766    ///
1767    /// Invariant: `extent_resident_bytes` equals the sum of `alloc_size`
1768    /// over live chunks' extents whose `is_resident()` is true, and
1769    /// `extent_residents` counts those extents; `extent_unreclaimable_bytes`
1770    /// is the subset whose `pageout_capped()` is true. This method,
1771    /// [`PoolInner::note_extent_reclaimable`],
1772    /// [`PoolInner::note_extent_released`], and the pageout arms in
1773    /// [`PoolInner::enforce_compressed_cap`] are the only adjusters; every
1774    /// flag flip pairs with one of them under the chunk's state lock.
1775    fn note_extent_resident(&self, meta: &Arc<ChunkMeta>, extent_alloc: usize, reclaimable: bool) {
1776        self.counters
1777            .extent_resident_bytes
1778            .fetch_add(u64::cast_from(extent_alloc), Ordering::Relaxed);
1779        self.extent_residents.fetch_add(1, Ordering::Relaxed);
1780        if reclaimable {
1781            self.prune_extent_queue();
1782            self.extent_queue().push_back(Arc::downgrade(meta));
1783        } else {
1784            self.counters
1785                .extent_unreclaimable_bytes
1786                .fetch_add(u64::cast_from(extent_alloc), Ordering::Relaxed);
1787        }
1788    }
1789
1790    /// Returns a retry-capped resident extent to the reclaimable set after a
1791    /// read restored its pageout budget: uncounts it from the unreclaimable
1792    /// gauge and re-enqueues its chunk for RSS-target enforcement. The
1793    /// caller holds the chunk's state lock with the extent present, resident,
1794    /// and no longer `pageout_capped()`.
1795    fn note_extent_reclaimable(&self, meta: &Arc<ChunkMeta>, extent_alloc: usize) {
1796        self.counters
1797            .extent_unreclaimable_bytes
1798            .fetch_sub(u64::cast_from(extent_alloc), Ordering::Relaxed);
1799        self.prune_extent_queue();
1800        self.extent_queue().push_back(Arc::downgrade(meta));
1801    }
1802
1803    /// Uncounts a resident extent that is being dropped (chunk freed or
1804    /// degraded). Its queue entry goes stale and is dropped on visit or by
1805    /// [`PoolInner::prune_extent_queue`].
1806    fn note_extent_released(&self, extent: &SwapExtent) {
1807        if extent.is_resident() {
1808            self.counters
1809                .extent_resident_bytes
1810                .fetch_sub(u64::cast_from(extent.alloc_size()), Ordering::Relaxed);
1811            self.extent_residents.fetch_sub(1, Ordering::Relaxed);
1812            if extent.pageout_capped() {
1813                self.counters
1814                    .extent_unreclaimable_bytes
1815                    .fetch_sub(u64::cast_from(extent.alloc_size()), Ordering::Relaxed);
1816            }
1817        }
1818    }
1819
1820    /// Drops extent-queue entries whose chunk has been freed, mirroring
1821    /// [`PoolInner::prune_queues`]: compact only when the queue outgrows
1822    /// all live resident extents (plus a small floor), so the cost
1823    /// amortizes to a constant per push. Enforcement drops stale entries
1824    /// too, but only while the tier is over capacity. A pool that stays
1825    /// under its compressed cap would otherwise accumulate an entry (and a
1826    /// pin on the dead chunk's allocation) per freed extent forever.
1827    fn prune_extent_queue(&self) {
1828        let live = usize::cast_from(self.extent_residents.load(Ordering::Relaxed));
1829        let mut queue = self.extent_queue();
1830        if queue.len() > 2 * live + 16 {
1831            queue.retain(|weak| weak.strong_count() > 0);
1832        }
1833    }
1834
1835    /// Routes compressed-cap enforcement off latency-sensitive threads: with
1836    /// spill threads spawned, wakes one to perform the pageouts
1837    /// (`MADV_PAGEOUT` is synchronous reclaim, bounded per extent but not
1838    /// free at chunk rates); without them, enforces inline. The test is for
1839    /// thread existence, not `spill.enabled`: spawned threads trim the tier
1840    /// in their loop even with eviction hand-off disabled.
1841    ///
1842    /// Deferral makes the target eventually-enforced with bounded lag, and
1843    /// the backstop below turns the lag into a bound by construction: a
1844    /// caller finding the reclaimable tier at double its capacity enforces
1845    /// inline regardless, so sustained creation can never outrun trimming
1846    /// by more than one capacity's worth.
1847    fn enforce_or_defer_compressed_cap(&self) {
1848        if self.spill.threads.load(Ordering::Relaxed) > 0 {
1849            // The inline backstop keys on the bytes enforcement can actually
1850            // reclaim. Unreclaimable extents (retry-capped, heap-backed)
1851            // would otherwise hold the backstop permanently over threshold
1852            // and put a full enforcement pass on every caller.
1853            let resident = self.counters.extent_resident_bytes.load(Ordering::Relaxed);
1854            let unreclaimable = self
1855                .counters
1856                .extent_unreclaimable_bytes
1857                .load(Ordering::Relaxed);
1858            if resident.saturating_sub(unreclaimable) > self.compressed_cap().saturating_mul(2) {
1859                self.enforce_compressed_cap();
1860            } else {
1861                self.spill.cv.notify_one();
1862            }
1863        } else {
1864            self.enforce_compressed_cap();
1865        }
1866    }
1867
1868    /// Pages out the oldest resident extents until the compressed tier falls
1869    /// to its capacity. The compression is already paid and the device write
1870    /// is the kernel's async writeback, so each pageout is one bounded
1871    /// madvise plus a page-table observation; spill threads run this between
1872    /// jobs, and other threads only when no spill threads exist (see
1873    /// [`PoolInner::enforce_or_defer_compressed_cap`]). Not single-flighted:
1874    /// concurrent passes pop disjoint victims. Visits are bounded by the
1875    /// queue's length at entry; stale entries (extent paged out, dropped, or
1876    /// chunk dead) are dropped. Incomplete extents are requeued with their
1877    /// accounting intact until their retry budget runs out, at which point
1878    /// they leave the queue with their bytes on the unreclaimable gauge, so
1879    /// the tier may settle above its capacity by the bytes the kernel
1880    /// declined to reclaim without enforcement re-walking them.
1881    fn enforce_compressed_cap(&self) {
1882        let cap = self.compressed_cap();
1883        let resident = |c: &Counters| c.extent_resident_bytes.load(Ordering::Relaxed);
1884        // Under-cap is the common case: answer it with one atomic load and
1885        // no queue lock, so frequent callers (the spill loop) stay cheap.
1886        if resident(&self.counters) <= cap {
1887            return;
1888        }
1889        let mut remaining = self.extent_queue().len();
1890        while remaining > 0 && resident(&self.counters) > cap {
1891            remaining -= 1;
1892            let popped = self.extent_queue().pop_front();
1893            let Some(weak) = popped else {
1894                break;
1895            };
1896            let Some(meta) = weak.upgrade() else {
1897                continue;
1898            };
1899            // `try_lock`: a chunk mid-read or mid-compression holds its lock
1900            // for milliseconds; requeue rather than convoy behind it.
1901            let Ok(mut state) = meta.state.try_lock() else {
1902                self.extent_queue().push_back(weak);
1903                continue;
1904            };
1905            match &mut state.extent {
1906                Some(extent) if extent.is_resident() => {
1907                    if extent.pageout_capped() {
1908                        // A leftover entry for an already-capped extent (its
1909                        // capping transition below accounted it and dropped
1910                        // its entry): drop this one too. The read that
1911                        // restores the retry budget re-enqueues the chunk.
1912                    } else if extent.pageout() {
1913                        self.counters
1914                            .extent_resident_bytes
1915                            .fetch_sub(u64::cast_from(extent.alloc_size()), Ordering::Relaxed);
1916                        self.extent_residents.fetch_sub(1, Ordering::Relaxed);
1917                        self.counters
1918                            .extent_pageouts
1919                            .fetch_add(1, Ordering::Relaxed);
1920                    } else {
1921                        // The advice left pages resident. The extent keeps
1922                        // its full accounting (the ledger may over-count
1923                        // RSS, the safe direction).
1924                        self.counters
1925                            .extent_pageout_incomplete
1926                            .fetch_add(1, Ordering::Relaxed);
1927                        if extent.pageout_capped() {
1928                            // The retry budget just ran out: the extent
1929                            // leaves the queue and its bytes move to the
1930                            // unreclaimable gauge, so enforcement and the
1931                            // inline backstop stop chasing memory the kernel
1932                            // will not give back. A read that restores the
1933                            // budget re-counts and re-enqueues it.
1934                            self.counters
1935                                .extent_unreclaimable_bytes
1936                                .fetch_add(u64::cast_from(extent.alloc_size()), Ordering::Relaxed);
1937                        } else {
1938                            // Budget remains: keep the queue slot so later
1939                            // passes retry it up to the cap.
1940                            self.extent_queue().push_back(weak);
1941                        }
1942                    }
1943                }
1944                // Paged out already or dropped: the entry is stale. A later
1945                // resident event re-enqueues.
1946                _ => {}
1947            }
1948        }
1949    }
1950
1951    /// If the chunk is a live `UnbackedResident` holding a slot and the
1952    /// spill threads have capacity, transitions it to `WriteInFlight` and
1953    /// hands it to them, returning `true`. The hand-off happens under the
1954    /// held state lock; the spill thread blocks on that lock only after this
1955    /// call returns and the caller releases it.
1956    fn spill_handoff(&self, meta: &Arc<ChunkMeta>, state: &mut ChunkState) -> bool {
1957        // The slot check excludes empty chunks, which are `UnbackedResident`
1958        // without a slot: handing one off would panic the spill thread on
1959        // the missing slot.
1960        if state.residency != Residency::UnbackedResident
1961            || state.freed
1962            || state.slot.is_none()
1963            || !self.spill_eligible()
1964        {
1965            return false;
1966        }
1967        state.residency = Residency::WriteInFlight;
1968        self.spill_schedule(Arc::clone(meta));
1969        true
1970    }
1971}
1972
1973impl ChunkHandle {
1974    /// Test hook: the chunk's current residency state.
1975    #[cfg(test)]
1976    fn residency(&self) -> Residency {
1977        self.meta.state().residency
1978    }
1979
1980    /// Copies the whole contents into `dst` (cleared first), leaving the
1981    /// chunk's residency untouched: a resident slot is copied out directly,
1982    /// and an evicted extent decompresses straight into `dst` without
1983    /// allocating a slot. A read therefore never raises resident bytes,
1984    /// never converts the chunk's state, and hands out no reference into
1985    /// pool memory.
1986    ///
1987    /// The copy runs under the chunk's state lock, which is what makes the
1988    /// no-reference contract cheap: eviction takes the same lock, so there
1989    /// is no reader it could race. The admitting variant is
1990    /// [`ChunkHandle::read_into_admit`].
1991    pub fn read_into(&self, dst: &mut Vec<u64>) {
1992        self.read_impl(0..self.meta.len, dst, false);
1993    }
1994
1995    /// As [`ChunkHandle::read_into`], restricted to the word range `range`
1996    /// of the chunk's contents, which must lie within them. `dst` receives
1997    /// exactly the range.
1998    ///
1999    /// The range narrows only the copy into `dst`: the swap backend's
2000    /// stored form is a whole compressed block, so a cold read still
2001    /// faults and decompresses the entire extent, and accounting is that
2002    /// of a whole-chunk read.
2003    pub fn read_range_into(&self, range: Range<usize>, dst: &mut Vec<u64>) {
2004        self.read_impl(range, dst, false);
2005    }
2006
2007    /// As [`ChunkHandle::read_into`], except that an evicted chunk is
2008    /// re-admitted to `BackedResident` (its extent kept, its touched bit
2009    /// set) when a slot is available from free budget headroom or by
2010    /// stealing from a clean backed victim of the same size class, never by
2011    /// evicting or compressing anything. When neither source yields a slot
2012    /// the read is served as a plain decompress and the chunk stays
2013    /// evicted.
2014    ///
2015    /// For demand reads on probe paths, where the same chunk is likely to
2016    /// be read again. Merge, drain, and other consume-once paths should use
2017    /// [`ChunkHandle::read_into`] or [`ChunkHandle::take`]: admitting there
2018    /// churns the clean-victim stock that eager backing exists to build,
2019    /// evicting probe targets to house data about to die.
2020    pub fn read_into_admit(&self, dst: &mut Vec<u64>) {
2021        self.read_impl(0..self.meta.len, dst, true);
2022    }
2023
2024    /// As [`ChunkHandle::read_into_admit`], restricted to the word range
2025    /// `range` per [`ChunkHandle::read_range_into`]. Admission is
2026    /// whole-chunk regardless of the range: the acquired slot holds the
2027    /// entire body.
2028    pub fn read_range_into_admit(&self, range: Range<usize>, dst: &mut Vec<u64>) {
2029        self.read_impl(range, dst, true);
2030    }
2031
2032    /// Shared body of the copy-out reads: fills `dst` with the word range
2033    /// `range` of the chunk's contents under the chunk's state lock,
2034    /// re-admitting an evicted chunk when `admit` is set and a slot is
2035    /// available. An empty range returns without locking or touching the
2036    /// chunk, like the whole-chunk read of an empty chunk always has.
2037    fn read_impl(&self, range: Range<usize>, dst: &mut Vec<u64>, admit: bool) {
2038        dst.clear();
2039        let meta = &*self.meta;
2040        assert!(
2041            range.start <= range.end && range.end <= meta.len,
2042            "range {range:?} exceeds the chunk's {} words",
2043            meta.len,
2044        );
2045        if range.is_empty() {
2046            return;
2047        }
2048        let mut state = meta.state();
2049        state.touched = true;
2050        let mut extent_revived = false;
2051        match state.residency {
2052            Residency::Oversize => {
2053                let payload = state.oversize.as_ref().expect("oversize chunk has payload");
2054                dst.extend_from_slice(&payload[range]);
2055            }
2056            Residency::Evicted => {
2057                let slot = if admit {
2058                    meta.pool.admit_slot(meta)
2059                } else {
2060                    None
2061                };
2062                let extent = state.extent.as_mut().expect("evicted chunk has an extent");
2063                // Reading faults the extent's pages back in either way, so
2064                // it is re-counted against the compressed tier below.
2065                let was_resident = extent.is_resident();
2066                let was_capped = extent.pageout_capped();
2067                let extent_alloc = extent.alloc_size();
2068                match slot {
2069                    Some(slot) => {
2070                        // Admission: the extent decompresses straight into
2071                        // the acquired slot, fully overwriting its
2072                        // unspecified prior contents, and the caller's
2073                        // buffer is filled from the slot.
2074                        let region = meta.pool.region_of(meta);
2075                        // SAFETY: the slot was acquired for this chunk
2076                        // under its held state lock (freshly allocated, or
2077                        // transferred from the victim under the victim's
2078                        // lock), so it is exclusively owned with no other
2079                        // reference into it, and `len_bytes` fits the
2080                        // class.
2081                        let slot_bytes = unsafe {
2082                            std::slice::from_raw_parts_mut(region.slot_ptr(slot), meta.len_bytes())
2083                        };
2084                        extent.read_into(meta.codec, slot_bytes);
2085                        state.slot = Some(slot);
2086                        state.residency = Residency::BackedResident;
2087                        // SAFETY: the slot belongs to this chunk while the
2088                        // state lock is held (eviction and free both take
2089                        // it).
2090                        let src = unsafe { meta.pool.slot_data(meta, slot) };
2091                        dst.extend_from_slice(&src[range.start..range.end]);
2092                        // Resident again: rejoin the eviction candidates.
2093                        // A leftover entry from before the chunk's eviction
2094                        // stays sound (each entry is validated against the
2095                        // chunk's state on visit) but costs policy: two live
2096                        // entries give the enforcer two chances to spend
2097                        // this chunk's single touched bit, halving its
2098                        // second chance until one entry drains.
2099                        meta.pool
2100                            .queue(band(meta.depth))
2101                            .push_back(Arc::downgrade(&self.meta));
2102                    }
2103                    None => {
2104                        // The zero-fill ahead of the decompress is deliberate
2105                        // waste (~a tenth of the decompress cost): the extent
2106                        // read takes an initialized `&mut [u8]`, so skipping
2107                        // the fill would mean exposing uninitialized memory
2108                        // through a safe reference.
2109                        dst.resize(range.end - range.start, 0);
2110                        let bytes: &mut [u8] = bytemuck::cast_slice_mut(dst.as_mut_slice());
2111                        extent.read_range_into(
2112                            meta.codec,
2113                            meta.len_bytes(),
2114                            range.start * 8,
2115                            bytes,
2116                        );
2117                    }
2118                }
2119                // TODO: a sub-range read of a rangeable stored form (file
2120                // extents, a sub-block-framed codec) revives only part of
2121                // the extent; the whole-extent accounting below would then
2122                // overcount and needs a partial-revival variant.
2123                if !was_resident {
2124                    // Revived from the device: the decompress reset any
2125                    // retry budget, so the extent re-enters reclaimable.
2126                    meta.pool
2127                        .note_extent_resident(&self.meta, extent_alloc, true);
2128                    extent_revived = true;
2129                } else if was_capped {
2130                    // The decompress faulted every page and reset the
2131                    // pageout retry budget, so a retry-capped extent is
2132                    // reclaimable again. Heap-backed extents stay
2133                    // structurally capped and stay out of the queue.
2134                    let capped = state
2135                        .extent
2136                        .as_ref()
2137                        .expect("evicted chunk has an extent")
2138                        .pageout_capped();
2139                    if !capped {
2140                        meta.pool.note_extent_reclaimable(&self.meta, extent_alloc);
2141                        extent_revived = true;
2142                    }
2143                }
2144            }
2145            Residency::UnbackedResident | Residency::BackedResident | Residency::WriteInFlight => {
2146                let slot = state.slot.expect("resident non-empty chunk has a slot");
2147                // SAFETY: the slot belongs to this chunk while the state lock
2148                // is held (eviction and free both take it).
2149                let src = unsafe { meta.pool.slot_data(meta, slot) };
2150                dst.extend_from_slice(&src[range.start..range.end]);
2151            }
2152        }
2153        drop(state);
2154        // The read revived the extent's compressed pages; the tier may need
2155        // trimming. Enforcement locks chunk states itself, so it must run
2156        // after the unlock.
2157        if extent_revived {
2158            meta.pool.enforce_or_defer_compressed_cap();
2159        }
2160    }
2161
2162    /// Copies the whole contents into `dst` (per [`ChunkHandle::read_into`],
2163    /// never admitting) and frees the chunk, cancelling any in-flight
2164    /// backing write.
2165    pub fn take(self, dst: &mut Vec<u64>) {
2166        self.read_into(dst);
2167    }
2168
2169    /// Advisory a consumer may issue before a bulk read: hints the kernel to
2170    /// swap an evicted chunk's extent back in, and is a no-op in every other
2171    /// state. Never blocks on I/O (`MADV_WILLNEED` is asynchronous).
2172    pub fn prefetch(&self) {
2173        let state = self.meta.state();
2174        if state.residency == Residency::Evicted {
2175            let extent = state.extent.as_ref().expect("evicted chunk has an extent");
2176            extent.prefetch();
2177        }
2178    }
2179
2180    /// As [`ChunkHandle::prefetch`], scoped to the word range `range` of the
2181    /// chunk's contents. The range is advisory: a backend hints at whatever
2182    /// granularity its stored form permits, and the swap backend's stored
2183    /// form is a whole compressed block, so it hints the entire extent.
2184    pub fn prefetch_range(&self, range: Range<usize>) {
2185        let _ = range;
2186        self.prefetch();
2187    }
2188
2189    /// Test hook: the byte size of the chunk's size class, or `None` for
2190    /// empty and oversize chunks.
2191    #[cfg(test)]
2192    fn size_class_bytes(&self) -> Option<usize> {
2193        self.meta.class.map(|class| SIZE_CLASSES[class])
2194    }
2195}
2196
2197impl Drop for ChunkHandle {
2198    fn drop(&mut self) {
2199        let pool = &self.meta.pool;
2200        let mut state = self.meta.state();
2201        pool.counters.frees.fetch_add(1, Ordering::Relaxed);
2202        state.freed = true;
2203        if self.meta.class.is_some() {
2204            pool.live_chunks.fetch_sub(1, Ordering::Relaxed);
2205        }
2206        let len_bytes = u64::cast_from(self.meta.len_bytes());
2207        // `release_slot`'s precondition holds in every arm below: the handle
2208        // is being dropped, so no copy-out read (which borrows the handle)
2209        // is in progress, and `freed` was set under the state lock held
2210        // here, so concurrent queue visitors skip the chunk.
2211        match state.residency {
2212            Residency::UnbackedResident => {
2213                if state.slot.is_some() {
2214                    pool.counters.writes_elided.fetch_add(1, Ordering::Relaxed);
2215                    pool.release_slot(&self.meta, &mut state);
2216                }
2217            }
2218            Residency::BackedResident => {
2219                pool.release_slot(&self.meta, &mut state);
2220                if let Some(extent) = &state.extent {
2221                    pool.note_extent_released(extent);
2222                }
2223                state.extent = None;
2224            }
2225            Residency::Evicted => {
2226                crate::soft_assert_no_log!(state.slot.is_none(), "evicted chunk holds no slot");
2227                if let Some(extent) = &state.extent {
2228                    pool.note_extent_released(extent);
2229                }
2230                state.extent = None;
2231            }
2232            Residency::WriteInFlight => {
2233                // A spill thread may be reading the slot to compress it.
2234                // `freed` (set above) tells it the chunk died; it owns the
2235                // slot release, the `resident_bytes` decrement, and the
2236                // cancellation accounting from here.
2237            }
2238            Residency::Oversize => {
2239                pool.counters
2240                    .resident_bytes
2241                    .fetch_sub(len_bytes, Ordering::Relaxed);
2242                pool.counters
2243                    .oversize_bytes
2244                    .fetch_sub(len_bytes, Ordering::Relaxed);
2245                state.oversize = None;
2246            }
2247        }
2248    }
2249}
2250
2251#[cfg(test)]
2252mod tests {
2253    use super::*;
2254    use crate::pool::extent::TEST_CODEC;
2255
2256    /// Keep test pools small: 64 MiB of virtual reservation per class.
2257    /// Under Miri the backing is real interpreter heap rather than lazy
2258    /// virtual memory, so shrink further. Classes above the capacity yield
2259    /// empty regions whose inserts degrade to the heap fallback, which is
2260    /// fine: slotted-chunk tests exercise only the smallest classes.
2261    fn test_pool(budget_bytes: usize) -> Pool {
2262        let capacity = if cfg!(miri) { 1 << 20 } else { 64 << 20 };
2263        let pool = Pool::with_class_capacity(capacity).expect("pool creation");
2264        pool.set_budget(budget_bytes);
2265        pool
2266    }
2267
2268    /// Scales an iteration count down under Miri, where one interpreted
2269    /// compression costs what thousands do natively.
2270    fn rounds(native: u64, miri: u64) -> u64 {
2271        if cfg!(miri) { miri } else { native }
2272    }
2273
2274    fn payload(words: usize, seed: u64) -> Vec<u64> {
2275        (0..u64::cast_from(words))
2276            .map(|i| seed.wrapping_mul(0x9E3779B97F4A7C15).wrapping_add(i))
2277            .collect()
2278    }
2279
2280    /// Copies `data` into the pool and clears it.
2281    fn insert(pool: &Pool, data: &mut Vec<u64>) -> ChunkHandle {
2282        insert_at_depth(pool, 0, data)
2283    }
2284
2285    /// Copies `data` into the pool at a hinted depth and clears it.
2286    fn insert_at_depth(pool: &Pool, depth: u8, data: &mut Vec<u64>) -> ChunkHandle {
2287        let hints = ChunkHints { depth };
2288        let handle = pool.insert_with(data.len(), hints, &TEST_CODEC, |dst| {
2289            dst.copy_from_slice(data.as_slice())
2290        });
2291        data.clear();
2292        handle
2293    }
2294
2295    /// Copies a chunk's contents out into a fresh buffer.
2296    fn read(handle: &ChunkHandle) -> Vec<u64> {
2297        let mut out = Vec::new();
2298        handle.read_into(&mut out);
2299        out
2300    }
2301
2302    /// Copies a chunk's contents out into a fresh buffer via the admitting
2303    /// read.
2304    fn read_admit(handle: &ChunkHandle) -> Vec<u64> {
2305        let mut out = Vec::new();
2306        handle.read_into_admit(&mut out);
2307        out
2308    }
2309
2310    /// Words that fill a 64 KiB class exactly.
2311    const SMALL: usize = (64 << 10) / 8;
2312
2313    #[allow(dead_code)]
2314    fn assert_handle_send_sync() {
2315        fn check<T: Send + Sync>() {}
2316        check::<Pool>();
2317        check::<ChunkHandle>();
2318    }
2319
2320    /// With an RSS target set, evicted chunks keep their extents resident
2321    /// (the compressed tier); shrinking the target pages the oldest extents
2322    /// out; reads revive them and re-count them.
2323    #[mz_ore::test]
2324    fn compressed_tier_round_trip() {
2325        let pool = test_pool(256 << 20);
2326        pool.set_rss_target(1 << 30);
2327        let orig = payload(SMALL, 21);
2328        let handle = insert(&pool, &mut orig.clone());
2329        pool.evict(&handle);
2330        assert_eq!(handle.residency(), Residency::Evicted);
2331        let stats = pool.stats();
2332        assert!(
2333            stats.extent_resident_bytes > 0,
2334            "under the target, the extent stays resident",
2335        );
2336        assert_eq!(stats.extent_pageouts, 0);
2337
2338        // Shrinking the target to zero pages the extent out.
2339        pool.set_rss_target(0);
2340        let stats = pool.stats();
2341        assert_eq!(stats.extent_resident_bytes, 0, "tier collapsed");
2342        assert_eq!(stats.extent_pageouts, 1);
2343
2344        // Reading revives the extent: contents round-trip, the chunk stays
2345        // evicted, and with the target restored the revived extent is
2346        // counted again.
2347        pool.set_rss_target(1 << 30);
2348        assert_eq!(read(&handle), orig);
2349        assert_eq!(handle.residency(), Residency::Evicted);
2350        assert!(
2351            pool.stats().extent_resident_bytes > 0,
2352            "revived and counted"
2353        );
2354
2355        // Dropping the handle uncounts the resident extent.
2356        drop(handle);
2357        assert_eq!(pool.stats().extent_resident_bytes, 0);
2358    }
2359
2360    /// An RSS target with 1 MiB of headroom above the budget and warm cap
2361    /// keeps an extent resident: unused insertion slack does not shrink the
2362    /// compressed tier.
2363    #[mz_ore::test]
2364    fn unused_insert_slack_leaves_compressed_tier_intact() {
2365        let budget = 64 << 20;
2366        let pool = test_pool(budget);
2367        pool.set_rss_target(budget + budget / 8 + (1 << 20));
2368        let handle = insert(&pool, &mut payload(SMALL, 22));
2369        pool.evict(&handle);
2370        let stats = pool.stats();
2371        assert!(stats.extent_resident_bytes > 0, "the extent stays resident");
2372        assert_eq!(stats.extent_pageouts, 0);
2373    }
2374
2375    /// A ranged read returns exactly the corresponding slice of a
2376    /// whole-chunk read in every residency state, and changes residency
2377    /// exactly as the equivalent whole-chunk read would.
2378    #[mz_ore::test]
2379    fn ranged_reads_match_full_read_slice() {
2380        let pool = test_pool(256 << 20);
2381        pool.set_rss_target(1 << 30);
2382        let orig = payload(SMALL, 33);
2383        let handle = insert(&pool, &mut orig.clone());
2384        let ranges = [
2385            (0usize, 7usize),
2386            (13, 100),
2387            (SMALL - 9, 9),
2388            (0, SMALL),
2389            (5, 0),
2390        ];
2391        let check = |label: &str| {
2392            for (start, len) in ranges {
2393                let mut out = Vec::new();
2394                handle.read_range_into(start..start + len, &mut out);
2395                assert_eq!(
2396                    out,
2397                    &orig[start..start + len],
2398                    "{label} range ({start}, {len})"
2399                );
2400            }
2401        };
2402        assert_eq!(handle.residency(), Residency::UnbackedResident);
2403        check("resident");
2404        pool.evict(&handle);
2405        assert_eq!(handle.residency(), Residency::Evicted);
2406        check("evicted");
2407        assert_eq!(
2408            handle.residency(),
2409            Residency::Evicted,
2410            "plain ranged reads do not admit"
2411        );
2412        // An admitting ranged read returns the range and admits the whole
2413        // chunk.
2414        let mut out = Vec::new();
2415        handle.read_range_into_admit(3..19, &mut out);
2416        assert_eq!(out, &orig[3..19]);
2417        assert_eq!(handle.residency(), Residency::BackedResident);
2418        check("backed");
2419    }
2420
2421    #[mz_ore::test]
2422    #[should_panic(expected = "exceeds the chunk's")]
2423    fn ranged_read_out_of_bounds_panics() {
2424        let pool = test_pool(256 << 20);
2425        let handle = insert(&pool, &mut payload(SMALL, 34));
2426        let mut out = Vec::new();
2427        handle.read_range_into(SMALL - 1..SMALL + 1, &mut out);
2428    }
2429
2430    #[mz_ore::test]
2431    fn default_target_pages_extents_immediately() {
2432        let pool = test_pool(256 << 20);
2433        let handle = insert(&pool, &mut payload(SMALL, 22));
2434        pool.evict(&handle);
2435        let stats = pool.stats();
2436        assert_eq!(stats.extent_resident_bytes, 0);
2437        assert_eq!(stats.extent_pageouts, 1);
2438    }
2439
2440    #[mz_ore::test]
2441    fn full_pageout_uncounts_exactly_the_extent() {
2442        let pool = test_pool(256 << 20);
2443        pool.set_rss_target(1 << 30);
2444        let handle = insert(&pool, &mut payload(SMALL, 50));
2445        pool.evict(&handle);
2446        let counted = pool.stats().extent_resident_bytes;
2447        assert!(counted > 0, "under the target, the extent stays counted");
2448        pool.set_rss_target(0);
2449        let stats = pool.stats();
2450        assert_eq!(stats.extent_resident_bytes, 0, "exactly `counted` left");
2451        assert_eq!(stats.extent_pageouts, 1);
2452        assert_eq!(stats.extent_pageout_incomplete, 0);
2453    }
2454
2455    #[mz_ore::test]
2456    fn incomplete_pageout_keeps_accounting_and_queue_position() {
2457        let pool = test_pool(256 << 20);
2458        pool.set_rss_target(1 << 30);
2459        let handle = insert(&pool, &mut payload(SMALL, 51));
2460        pool.evict(&handle);
2461        let counted = pool.stats().extent_resident_bytes;
2462        assert!(counted > 0);
2463        region::fake_residency::decline_next(1);
2464        pool.set_rss_target(0);
2465        let stats = pool.stats();
2466        assert_eq!(
2467            stats.extent_resident_bytes, counted,
2468            "full accounting stays"
2469        );
2470        assert_eq!(stats.extent_pageouts, 0);
2471        assert_eq!(stats.extent_pageout_incomplete, 1);
2472        assert_eq!(handle.residency(), Residency::Evicted);
2473        // The requeued entry is retried by the next enforcement pass.
2474        pool.enforce_rss_target();
2475        let stats = pool.stats();
2476        assert_eq!(stats.extent_resident_bytes, 0);
2477        assert_eq!(stats.extent_pageouts, 1);
2478        assert_eq!(stats.extent_pageout_incomplete, 1);
2479    }
2480
2481    /// A never-reclaimable extent stops being advised after the retry cap:
2482    /// the incomplete counter stops climbing, the bytes stay counted
2483    /// resident, and the tier keeps paging other extents out around it.
2484    #[mz_ore::test]
2485    fn pageout_retry_cap_stops_advising() {
2486        let pool = test_pool(256 << 20);
2487        let handle = insert(&pool, &mut payload(SMALL, 52));
2488        region::fake_residency::decline_next(u64::MAX);
2489        // RSS target zero: the eviction's enforcement pass advises at once.
2490        pool.evict(&handle);
2491        for _ in 0..5 {
2492            pool.enforce_rss_target();
2493        }
2494        let stats = pool.stats();
2495        assert_eq!(
2496            stats.extent_pageout_incomplete,
2497            u64::from(extent::PAGEOUT_RETRY_CAP),
2498            "advised exactly retry-cap times",
2499        );
2500        assert_eq!(stats.extent_pageouts, 0);
2501        let counted = stats.extent_resident_bytes;
2502        assert!(counted > 0, "capped extent stays counted resident");
2503        // The tier functions around the capped extent: a fresh extent still
2504        // pages out.
2505        region::fake_residency::decline_next(0);
2506        let other = insert(&pool, &mut payload(SMALL, 53));
2507        pool.evict(&other);
2508        let stats = pool.stats();
2509        assert_eq!(stats.extent_pageouts, 1);
2510        assert_eq!(
2511            stats.extent_resident_bytes, counted,
2512            "only the capped extent remains counted",
2513        );
2514        assert_eq!(read(&handle).len(), SMALL, "capped extent stays readable");
2515    }
2516
2517    #[mz_ore::test]
2518    fn read_resets_pageout_retry_budget() {
2519        let pool = test_pool(256 << 20);
2520        let orig = payload(SMALL, 54);
2521        let handle = insert(&pool, &mut orig.clone());
2522        region::fake_residency::decline_next(u64::MAX);
2523        pool.evict(&handle);
2524        for _ in 0..4 {
2525            pool.enforce_rss_target();
2526        }
2527        assert_eq!(
2528            pool.stats().extent_pageout_incomplete,
2529            u64::from(extent::PAGEOUT_RETRY_CAP),
2530            "capped",
2531        );
2532        assert!(pool.stats().extent_resident_bytes > 0);
2533        region::fake_residency::decline_next(0);
2534        assert_eq!(read(&handle), orig);
2535        pool.enforce_rss_target();
2536        let stats = pool.stats();
2537        assert_eq!(stats.extent_pageouts, 1, "the budget reset re-advised it");
2538        assert_eq!(stats.extent_resident_bytes, 0);
2539        // The paged-out extent still round-trips.
2540        assert_eq!(read(&handle), orig);
2541    }
2542
2543    /// Eager backing compresses a chunk to `BackedResident` while it stays
2544    /// readable in its slot; the later budget-driven eviction is a pure page
2545    /// release, and the contents round-trip through the extent.
2546    #[mz_ore::test]
2547    fn eager_backing_round_trip() {
2548        let pool = test_pool(256 << 20);
2549        let orig = payload(SMALL, 11);
2550        let handle = insert(&pool, &mut orig.clone());
2551        assert_eq!(handle.residency(), Residency::UnbackedResident);
2552
2553        assert!(pool.back_step(), "one chunk is backable");
2554        assert_eq!(handle.residency(), Residency::BackedResident);
2555        let stats = pool.stats();
2556        assert_eq!(stats.eager_backs, 1);
2557        assert_eq!(stats.evictions_compress, 0, "backing is not an eviction");
2558        assert!(stats.extent_bytes_written > 0);
2559
2560        // Still readable straight from the slot: the chunk is resident.
2561        assert_eq!(read(&handle), orig);
2562
2563        // The pre-paid eviction is cheap, and the extent round-trips.
2564        pool.evict(&handle);
2565        assert_eq!(handle.residency(), Residency::Evicted);
2566        assert_eq!(pool.stats().evictions_cheap, 1);
2567        pool.poison_free_slots();
2568        assert_eq!(read(&handle), orig);
2569    }
2570
2571    #[mz_ore::test]
2572    fn backing_reports_no_progress_when_all_backed() {
2573        let pool = test_pool(256 << 20);
2574        let _handle = insert(&pool, &mut payload(SMALL, 31));
2575        assert!(pool.back_step(), "one unbacked chunk is actionable");
2576        assert!(!pool.back_step(), "fully backed: no progress");
2577        assert_eq!(pool.stats().eager_backs, 1);
2578    }
2579
2580    /// Freeing under the warm cap parks the slot warm; the next insert of the
2581    /// same class reuses it fault-free and the accounting balances.
2582    #[mz_ore::test]
2583    fn warm_slot_reuse() {
2584        // Budget 8 MiB: warm cap = 1 MiB, so a 64 KiB slot fits warm.
2585        let pool = test_pool(8 << 20);
2586        let orig = payload(SMALL, 7);
2587        let handle = insert(&pool, &mut orig.clone());
2588        drop(handle);
2589        let after_free = pool.stats();
2590        assert_eq!(after_free.warm_bytes, 64 << 10, "freed slot parks warm");
2591        assert_eq!(after_free.warm_reuses, 0);
2592
2593        let handle = insert(&pool, &mut orig.clone());
2594        let after_reuse = pool.stats();
2595        assert_eq!(after_reuse.warm_reuses, 1, "second insert reuses warm slot");
2596        assert_eq!(after_reuse.warm_bytes, 0, "reuse drains the warm pool");
2597        // Contents are correct despite the skipped page release.
2598        assert_eq!(read(&handle), orig);
2599    }
2600
2601    #[mz_ore::test]
2602    fn warm_pool_respects_cap() {
2603        // Budget 1 MiB: warm cap = 128 KiB = two 64 KiB slots.
2604        let pool = test_pool(1 << 20);
2605        let handles: Vec<_> = (0..4)
2606            .map(|seed| insert(&pool, &mut payload(SMALL, seed)))
2607            .collect();
2608        drop(handles);
2609        let stats = pool.stats();
2610        assert_eq!(
2611            stats.warm_bytes,
2612            128 << 10,
2613            "warm pool stops at the budget/8 cap",
2614        );
2615    }
2616
2617    /// A kernel that keeps declining the reclaim advice caps the extent's
2618    /// retry budget: the extent leaves the enforcement queue and moves to
2619    /// the unreclaimable gauge, so enforcement stops walking it, and a read
2620    /// that restores the budget makes it reclaimable and pageable again.
2621    #[mz_ore::test]
2622    fn capped_extents_leave_the_enforcement_queue() {
2623        let pool = test_pool(256 << 20);
2624        let orig = payload(SMALL, 960);
2625        let handle = insert(&pool, &mut orig.clone());
2626        // Decline every observation: eviction's own enforcement pass plus
2627        // the passes below spend the whole retry budget.
2628        region::fake_residency::decline_next(u64::from(extent::PAGEOUT_RETRY_CAP));
2629        pool.evict(&handle);
2630        for _ in 0..extent::PAGEOUT_RETRY_CAP {
2631            pool.enforce_compressed();
2632        }
2633        let stats = pool.stats();
2634        assert_eq!(
2635            stats.extent_pageout_incomplete,
2636            u64::from(extent::PAGEOUT_RETRY_CAP),
2637        );
2638        assert!(stats.extent_unreclaimable_bytes > 0, "capped bytes counted");
2639        assert_eq!(pool.extent_queue_len(), 0, "capped extents leave the queue",);
2640        // Further enforcement is a no-op: nothing queued, no advice spent.
2641        pool.enforce_compressed();
2642        assert_eq!(
2643            pool.stats().extent_pageout_incomplete,
2644            u64::from(extent::PAGEOUT_RETRY_CAP),
2645        );
2646
2647        // A read faults everything back in and restores the retry budget:
2648        // the extent re-enters the reclaimable set and, with the kernel now
2649        // cooperating, the read's own enforcement pass pages it out.
2650        assert_eq!(read(&handle), orig);
2651        let stats = pool.stats();
2652        assert_eq!(stats.extent_unreclaimable_bytes, 0, "budget restored");
2653        assert_eq!(stats.extent_pageouts, 1, "re-enqueued extent pages out");
2654        assert_eq!(stats.extent_resident_bytes, 0);
2655        assert_eq!(pool.extent_queue_len(), 0);
2656    }
2657
2658    /// Shrinking the budget cools warm slots parked under the old, larger
2659    /// cap: their pages are released and `warm_bytes` falls to the new cap
2660    /// on the shrink itself, not on eventual same-class reuse.
2661    #[mz_ore::test]
2662    fn budget_shrink_trims_warm_pool() {
2663        // Budget 8 MiB: warm cap 1 MiB, so four 64 KiB frees all park warm.
2664        let pool = test_pool(8 << 20);
2665        let handles: Vec<_> = (0..4)
2666            .map(|seed| insert(&pool, &mut payload(SMALL, 950 + seed)))
2667            .collect();
2668        drop(handles);
2669        assert_eq!(pool.stats().warm_bytes, 4 * (64 << 10));
2670
2671        // Budget 1 MiB: warm cap 128 KiB, so two of the four slots cool.
2672        pool.set_budget(1 << 20);
2673        assert_eq!(pool.stats().warm_bytes, 128 << 10);
2674    }
2675
2676    #[mz_ore::test]
2677    fn round_trip_resident() {
2678        let pool = test_pool(256 << 20);
2679        let orig = payload(1000, 1);
2680        let mut data = orig.clone();
2681        let capacity = data.capacity();
2682        let handle = insert(&pool, &mut data);
2683        assert!(data.is_empty());
2684        assert_eq!(data.capacity(), capacity, "insert preserves capacity");
2685        assert_eq!(handle.residency(), Residency::UnbackedResident);
2686        assert_eq!(read(&handle), orig);
2687        drop(handle);
2688        let stats = pool.stats();
2689        assert_eq!(stats.inserts, 1);
2690        assert_eq!(stats.frees, 1);
2691        assert_eq!(stats.resident_bytes, 0);
2692    }
2693
2694    #[mz_ore::test]
2695    fn take_reads_and_frees() {
2696        let pool = test_pool(256 << 20);
2697        let orig = payload(SMALL, 40);
2698        let handle = insert(&pool, &mut orig.clone());
2699        let mut out = Vec::new();
2700        handle.take(&mut out);
2701        assert_eq!(out, orig);
2702        let stats = pool.stats();
2703        assert_eq!(stats.frees, 1);
2704        assert_eq!(stats.writes_elided, 1, "a resident take never writes");
2705        assert_eq!(stats.resident_bytes, 0);
2706        assert_eq!(stats.live_chunks, 0);
2707    }
2708
2709    /// `prefetch` is safe wherever it lands: on a resident chunk (a no-op),
2710    /// on an evicted chunk (whose read then round-trips), and issued with no
2711    /// read following it. It never changes residency or resident bytes.
2712    #[mz_ore::test]
2713    fn prefetch_is_safe_in_every_state() {
2714        let pool = test_pool(256 << 20);
2715        let orig = payload(SMALL, 41);
2716        let handle = insert(&pool, &mut orig.clone());
2717        handle.prefetch();
2718        assert_eq!(handle.residency(), Residency::UnbackedResident);
2719        assert_eq!(read(&handle), orig);
2720        pool.evict(&handle);
2721        handle.prefetch();
2722        assert_eq!(handle.residency(), Residency::Evicted);
2723        assert_eq!(read(&handle), orig);
2724        // An advisory with no read behind it leaves nothing to clean up.
2725        let idle = insert(&pool, &mut payload(SMALL, 42));
2726        idle.prefetch();
2727        drop(idle);
2728        drop(handle);
2729        assert_eq!(pool.stats().resident_bytes, 0);
2730    }
2731
2732    /// Reading an evicted chunk decompresses its extent straight into the
2733    /// caller's buffer and leaves the chunk evicted. Free slots are poisoned
2734    /// first, so a read passing stale slot memory through (the macOS
2735    /// `MADV_DONTNEED` hazard) would fail the content check.
2736    #[mz_ore::test]
2737    fn evict_then_read_preserves_contents() {
2738        let pool = test_pool(256 << 20);
2739        let orig = payload(SMALL, 2);
2740        let handle = insert(&pool, &mut orig.clone());
2741        pool.evict(&handle);
2742        assert_eq!(handle.residency(), Residency::Evicted);
2743        let stats = pool.stats();
2744        assert_eq!(stats.evictions_compress, 1);
2745        assert_eq!(stats.resident_bytes, 0);
2746        assert!(stats.extent_bytes_written > 0);
2747        pool.poison_free_slots();
2748        assert_eq!(read(&handle), orig);
2749        assert_eq!(handle.residency(), Residency::Evicted);
2750        assert_eq!(pool.stats().resident_bytes, 0, "reads copy out");
2751    }
2752
2753    /// An admitting read of an evicted chunk with budget headroom re-admits
2754    /// it: contents round-trip, the chunk lands `BackedResident` with its
2755    /// extent kept, and later reads serve from the slot without touching
2756    /// the extent.
2757    #[mz_ore::test]
2758    fn admit_from_free_budget_backs_the_chunk() {
2759        let pool = test_pool(256 << 20);
2760        let orig = payload(SMALL, 70);
2761        let handle = insert(&pool, &mut orig.clone());
2762        pool.evict(&handle);
2763        assert_eq!(handle.residency(), Residency::Evicted);
2764        assert_eq!(pool.stats().resident_bytes, 0);
2765
2766        pool.poison_free_slots();
2767        assert_eq!(read_admit(&handle), orig);
2768        assert_eq!(handle.residency(), Residency::BackedResident);
2769        let stats = pool.stats();
2770        assert_eq!(stats.admissions_budget, 1);
2771        assert_eq!(stats.admissions_steal, 0);
2772        assert_eq!(stats.admissions_denied, 0);
2773        assert_eq!(stats.resident_bytes, 64 << 10);
2774
2775        // Later reads serve from the slot and never touch the extent: a
2776        // decompress would revive its pages and move the revival and
2777        // pageout counters.
2778        let pageouts = stats.extent_pageouts;
2779        let extent_resident = stats.extent_resident_bytes;
2780        assert_eq!(read(&handle), orig);
2781        assert_eq!(handle.residency(), Residency::BackedResident);
2782        let stats = pool.stats();
2783        assert_eq!(stats.extent_pageouts, pageouts);
2784        assert_eq!(stats.extent_resident_bytes, extent_resident);
2785
2786        // The kept extent pre-pays the next eviction, and round-trips.
2787        pool.evict(&handle);
2788        assert_eq!(handle.residency(), Residency::Evicted);
2789        let stats = pool.stats();
2790        assert_eq!(stats.evictions_cheap, 1);
2791        assert_eq!(stats.evictions_compress, 1, "admission wrote no extent");
2792        pool.poison_free_slots();
2793        assert_eq!(read(&handle), orig);
2794        drop(handle);
2795        assert_eq!(pool.stats().resident_bytes, 0);
2796    }
2797
2798    /// With the budget pinned full and a clean backed victim of the same
2799    /// class, an admitting read steals the victim's slot: the victim is
2800    /// evicted with zero I/O and its extent intact, the admitted chunk
2801    /// lands `BackedResident`, and resident bytes, warm bytes, and the
2802    /// compression and pageout counters are all unchanged.
2803    #[mz_ore::test]
2804    fn admit_steals_clean_victim_slot() {
2805        let pool = test_pool(256 << 20);
2806        pool.set_rss_target(1 << 30);
2807        let victim_orig = payload(SMALL, 71);
2808        let target_orig = payload(SMALL, 72);
2809        let victim = insert(&pool, &mut victim_orig.clone());
2810        let target = insert(&pool, &mut target_orig.clone());
2811        pool.evict(&target);
2812        assert!(pool.back_step(), "victim is backable");
2813        assert_eq!(victim.residency(), Residency::BackedResident);
2814        // The budget now holds exactly the victim: no admission headroom.
2815        pool.set_budget(64 << 10);
2816        assert_eq!(victim.residency(), Residency::BackedResident);
2817        let before = pool.stats();
2818
2819        assert_eq!(read_admit(&target), target_orig);
2820        assert_eq!(target.residency(), Residency::BackedResident);
2821        assert_eq!(victim.residency(), Residency::Evicted);
2822        let after = pool.stats();
2823        assert_eq!(after.admissions_steal, 1);
2824        assert_eq!(after.admissions_budget, 0);
2825        assert_eq!(after.admissions_denied, 0);
2826        assert_eq!(
2827            after.resident_bytes, before.resident_bytes,
2828            "same class, same bytes",
2829        );
2830        assert_eq!(
2831            after.evictions_compress, before.evictions_compress,
2832            "no compression",
2833        );
2834        assert_eq!(
2835            after.evictions_cheap, before.evictions_cheap,
2836            "a steal is not an enforcement eviction",
2837        );
2838        assert_eq!(after.extent_bytes_written, before.extent_bytes_written);
2839        assert_eq!(after.extent_pageouts, 0, "no pageout");
2840        assert_eq!(
2841            after.warm_bytes, before.warm_bytes,
2842            "the stolen slot skipped the free list",
2843        );
2844        assert_eq!(after.warm_reuses, before.warm_reuses);
2845
2846        // The victim's extent is intact: its old slot now holds the
2847        // admitted chunk's bytes, so a correct read must come from the
2848        // extent.
2849        assert_eq!(read(&victim), victim_orig);
2850        assert_eq!(victim.residency(), Residency::Evicted);
2851
2852        drop(victim);
2853        drop(target);
2854        let stats = pool.stats();
2855        assert_eq!(stats.resident_bytes, 0);
2856        assert_eq!(stats.extent_resident_bytes, 0);
2857    }
2858
2859    /// With the budget full and every candidate touched, the admitting read
2860    /// still returns correct data, the chunk stays evicted, and the denial
2861    /// counter increments.
2862    #[mz_ore::test]
2863    fn admit_denied_when_victims_touched() {
2864        let pool = test_pool(256 << 20);
2865        let victim_orig = payload(SMALL, 73);
2866        let target_orig = payload(SMALL, 74);
2867        let victim = insert(&pool, &mut victim_orig.clone());
2868        let target = insert(&pool, &mut target_orig.clone());
2869        pool.evict(&target);
2870        assert!(pool.back_step());
2871        // Reading the victim sets its second-chance bit, disqualifying it.
2872        assert_eq!(read(&victim), victim_orig);
2873        pool.set_budget(64 << 10);
2874        let resident = pool.stats().resident_bytes;
2875
2876        assert_eq!(read_admit(&target), target_orig);
2877        assert_eq!(target.residency(), Residency::Evicted);
2878        assert_eq!(victim.residency(), Residency::BackedResident);
2879        let stats = pool.stats();
2880        assert_eq!(stats.admissions_denied, 1);
2881        assert_eq!(stats.admissions_budget, 0);
2882        assert_eq!(stats.admissions_steal, 0);
2883        assert_eq!(stats.resident_bytes, resident);
2884    }
2885
2886    /// An unbacked resident candidate is never stolen from: evicting it
2887    /// would require the compression that admission forbids.
2888    #[mz_ore::test]
2889    fn admit_denied_when_victims_unbacked() {
2890        let pool = test_pool(256 << 20);
2891        let victim = insert(&pool, &mut payload(SMALL, 75));
2892        let target_orig = payload(SMALL, 76);
2893        let target = insert(&pool, &mut target_orig.clone());
2894        pool.evict(&target);
2895        pool.set_budget(64 << 10);
2896        assert_eq!(read_admit(&target), target_orig);
2897        assert_eq!(target.residency(), Residency::Evicted);
2898        assert_eq!(victim.residency(), Residency::UnbackedResident);
2899        assert_eq!(pool.stats().admissions_denied, 1);
2900    }
2901
2902    /// A clean backed victim of a different size class is never stolen
2903    /// from: slot reuse in place requires the classes to match.
2904    #[mz_ore::test]
2905    fn admit_denied_when_victims_wrong_class() {
2906        let pool = test_pool(256 << 20);
2907        // The victim fills the 128 KiB class; the target lives in the
2908        // 64 KiB one.
2909        let victim = insert(&pool, &mut payload(2 * SMALL, 77));
2910        let target_orig = payload(SMALL, 78);
2911        let target = insert(&pool, &mut target_orig.clone());
2912        pool.evict(&target);
2913        assert!(pool.back_step());
2914        assert_eq!(victim.residency(), Residency::BackedResident);
2915        // The budget holds exactly the victim: no headroom for the target.
2916        pool.set_budget(128 << 10);
2917        assert_eq!(read_admit(&target), target_orig);
2918        assert_eq!(target.residency(), Residency::Evicted);
2919        assert_eq!(victim.residency(), Residency::BackedResident);
2920        assert_eq!(pool.stats().admissions_denied, 1);
2921    }
2922
2923    #[mz_ore::test]
2924    fn plain_read_and_take_never_admit() {
2925        let pool = test_pool(256 << 20);
2926        let orig = payload(SMALL, 79);
2927        let handle = insert(&pool, &mut orig.clone());
2928        pool.evict(&handle);
2929        assert_eq!(read(&handle), orig);
2930        assert_eq!(handle.residency(), Residency::Evicted);
2931        assert_eq!(pool.stats().resident_bytes, 0);
2932        let mut out = Vec::new();
2933        handle.take(&mut out);
2934        assert_eq!(out, orig);
2935        let stats = pool.stats();
2936        assert_eq!(stats.admissions_budget, 0);
2937        assert_eq!(stats.admissions_steal, 0);
2938        assert_eq!(stats.admissions_denied, 0);
2939        assert_eq!(stats.resident_bytes, 0);
2940        assert_eq!(stats.frees, 1);
2941    }
2942
2943    /// A steal settles the ledger with the payload difference: a shrinking
2944    /// steal always proceeds, while a steal that would grow resident bytes
2945    /// past the budget is denied and leaves the victim untouched.
2946    #[mz_ore::test]
2947    fn steal_admission_charges_the_budget() {
2948        // Shrinking steal: the victim is larger than the admitted payload,
2949        // so the steal lowers resident bytes and always may proceed.
2950        let pool = test_pool(256 << 20);
2951        let victim_orig = payload(SMALL, 84);
2952        let victim = insert(&pool, &mut victim_orig.clone());
2953        assert!(pool.back_step(), "victim backs");
2954        let small_orig = payload(SMALL / 2, 85);
2955        let handle = insert(&pool, &mut small_orig.clone());
2956        pool.evict(&handle);
2957        pool.set_budget(64 << 10);
2958        assert_eq!(read_admit(&handle), small_orig);
2959        let stats = pool.stats();
2960        assert_eq!(stats.admissions_steal, 1, "no headroom, so the read steals");
2961        assert_eq!(stats.resident_bytes, u64::cast_from(SMALL / 2 * 8));
2962        assert_eq!(victim.residency(), Residency::Evicted);
2963        assert_eq!(read(&victim), victim_orig, "victim serves from its extent");
2964
2965        // Growing steal: the admitted payload is larger than the only
2966        // victim, and the growth does not fit the budget, so the admission
2967        // is denied and the victim is left untouched.
2968        let pool = test_pool(256 << 20);
2969        let big_orig = payload(SMALL, 86);
2970        let big = insert(&pool, &mut big_orig.clone());
2971        pool.evict(&big);
2972        let small_victim = insert(&pool, &mut payload(SMALL / 2, 87));
2973        assert!(pool.back_step(), "victim backs");
2974        pool.set_budget(32 << 10);
2975        assert_eq!(read_admit(&big), big_orig);
2976        let stats = pool.stats();
2977        assert_eq!(stats.admissions_denied, 1, "growth exceeds the budget");
2978        assert_eq!(stats.admissions_steal, 0);
2979        assert_eq!(big.residency(), Residency::Evicted);
2980        assert_eq!(small_victim.residency(), Residency::BackedResident);
2981    }
2982
2983    /// Admission of a chunk whose extent was pushed to the device: the read
2984    /// revives the extent into the acquired slot and re-counts it.
2985    #[mz_ore::test]
2986    fn admission_revives_paged_out_extent() {
2987        let pool = test_pool(256 << 20);
2988        let orig = payload(SMALL, 88);
2989        let handle = insert(&pool, &mut orig.clone());
2990        pool.evict(&handle);
2991        assert_eq!(
2992            pool.stats().extent_resident_bytes,
2993            0,
2994            "zero RSS target pages the extent out on eviction",
2995        );
2996        // Raise the target so the read's own tier enforcement does not
2997        // page the revived extent straight back out.
2998        pool.set_rss_target(1 << 30);
2999        assert_eq!(read_admit(&handle), orig);
3000        assert_eq!(handle.residency(), Residency::BackedResident);
3001        let stats = pool.stats();
3002        assert_eq!(stats.admissions_budget, 1);
3003        assert!(stats.extent_resident_bytes > 0, "revived and re-counted");
3004    }
3005
3006    /// An admitting read of a chunk that is not evicted is a plain read:
3007    /// no admission counter moves and no state changes.
3008    #[mz_ore::test]
3009    fn admit_is_plain_read_on_non_evicted_chunks() {
3010        let pool = test_pool(256 << 20);
3011        let orig = payload(SMALL, 89);
3012        let resident = insert(&pool, &mut orig.clone());
3013        assert_eq!(read_admit(&resident), orig);
3014        assert_eq!(resident.residency(), Residency::UnbackedResident);
3015        let words = SIZE_CLASSES[SIZE_CLASSES.len() - 1] / 8 + 1;
3016        let oversize_orig = payload(words, 90);
3017        let oversize = insert(&pool, &mut oversize_orig.clone());
3018        assert_eq!(read_admit(&oversize), oversize_orig);
3019        let empty = insert(&pool, &mut Vec::new());
3020        assert!(read_admit(&empty).is_empty());
3021        let stats = pool.stats();
3022        assert_eq!(stats.admissions_budget, 0);
3023        assert_eq!(stats.admissions_steal, 0);
3024        assert_eq!(stats.admissions_denied, 0);
3025    }
3026
3027    /// Concurrent admitting reads with no budget headroom: every admission
3028    /// must go through the steal path, racing steals against each other on
3029    /// the same victims (the pool's only two-chunk lock edge). Contents are
3030    /// asserted on every read.
3031    #[mz_ore::test]
3032    #[cfg_attr(miri, ignore)] // too slow
3033    fn concurrent_admits_exercise_the_steal_path() {
3034        const CHUNKS: u64 = 8;
3035        let pool = test_pool(usize::MAX);
3036        let origs: Vec<_> = (0..CHUNKS).map(|seed| payload(SMALL, 900 + seed)).collect();
3037        let evicted: Arc<Vec<(Vec<u64>, ChunkHandle)>> = Arc::new(
3038            origs
3039                .iter()
3040                .map(|orig| {
3041                    let handle = insert(&pool, &mut orig.clone());
3042                    pool.evict(&handle);
3043                    (orig.clone(), handle)
3044                })
3045                .collect(),
3046        );
3047        let mut victims = Vec::new();
3048        for seed in 0..CHUNKS {
3049            victims.push(insert(&pool, &mut payload(SMALL, 950 + seed)));
3050            assert!(pool.back_step(), "victim backs");
3051        }
3052        // Exactly the victims' bytes: no free headroom, so every admission
3053        // steals or is denied.
3054        pool.set_budget(usize::cast_from(CHUNKS) * (64 << 10));
3055        let threads: Vec<_> = (0..2u64)
3056            .map(|t| {
3057                let evicted = Arc::clone(&evicted);
3058                std::thread::spawn(move || {
3059                    for round in 0..CHUNKS {
3060                        let (orig, handle) = &evicted[usize::cast_from((t + round) % CHUNKS)];
3061                        let mut out = Vec::new();
3062                        handle.read_into_admit(&mut out);
3063                        assert_eq!(&out, orig);
3064                    }
3065                })
3066            })
3067            .collect();
3068        for thread in threads {
3069            thread.join().expect("admitting thread panicked");
3070        }
3071        let stats = pool.stats();
3072        assert!(stats.admissions_steal > 0, "no headroom forces steals");
3073        assert!(
3074            stats.resident_bytes <= u64::cast_from(usize::cast_from(CHUNKS) * (64 << 10)),
3075            "steals never grow resident bytes past the budget",
3076        );
3077        // The victims were held live as steal targets; a steal leaves its
3078        // victim evicted, so at least one is evicted here.
3079        let stolen = victims
3080            .iter()
3081            .filter(|v| v.residency() == Residency::Evicted)
3082            .count();
3083        assert!(stolen > 0, "a steal evicts its victim");
3084    }
3085
3086    /// A re-admitted chunk keeps its insert-time depth: under budget
3087    /// pressure it is evicted from its own deeper band before a younger
3088    /// band-0 chunk, which a re-admission into band 0 would have inverted.
3089    #[mz_ore::test]
3090    fn admitted_chunk_keeps_its_depth() {
3091        let pool = test_pool(256 << 20);
3092        let deep_orig = payload(SMALL, 80);
3093        let deep = insert_at_depth(&pool, 2, &mut deep_orig.clone());
3094        let young = insert(&pool, &mut payload(SMALL, 81));
3095        pool.evict(&deep);
3096        assert_eq!(read_admit(&deep), deep_orig);
3097        assert_eq!(deep.residency(), Residency::BackedResident);
3098        assert_eq!(pool.stats().admissions_budget, 1);
3099
3100        // Budget of one chunk: enforcement visits the deep band first.
3101        pool.set_budget(64 << 10);
3102        assert_eq!(deep.residency(), Residency::Evicted);
3103        assert_eq!(young.residency(), Residency::UnbackedResident);
3104        assert_eq!(
3105            pool.stats().evictions_cheap,
3106            1,
3107            "the extent kept through admission pre-paid the eviction",
3108        );
3109    }
3110
3111    /// Admitting reads racing enforcement, opposing steals, and frees:
3112    /// contents stay correct, contended steals degrade to skips, and the
3113    /// accounting identity settles to zero.
3114    #[mz_ore::test]
3115    #[cfg_attr(miri, ignore)] // too slow
3116    fn concurrent_admits_race_cleanly() {
3117        let pool = test_pool(64 << 10);
3118        let per_thread = rounds(50, 3);
3119        let threads: Vec<_> = (0..4u64)
3120            .map(|t| {
3121                let pool = pool.clone();
3122                std::thread::spawn(move || {
3123                    let mut out = Vec::new();
3124                    for round in 0..per_thread {
3125                        let orig = payload(SMALL, t * 1000 + round);
3126                        let handle = insert(&pool, &mut orig.clone());
3127                        pool.evict(&handle);
3128                        handle.read_into_admit(&mut out);
3129                        assert_eq!(out, orig);
3130                        handle.read_into_admit(&mut out);
3131                        assert_eq!(out, orig);
3132                        assert_eq!(read(&handle), orig);
3133                    }
3134                })
3135            })
3136            .collect();
3137        for thread in threads {
3138            thread.join().expect("worker thread panicked");
3139        }
3140        let stats = pool.stats();
3141        assert_eq!(stats.inserts, 4 * per_thread);
3142        assert_eq!(stats.frees, 4 * per_thread);
3143        assert_eq!(stats.resident_bytes, 0);
3144        assert_eq!(stats.extent_resident_bytes, 0);
3145    }
3146
3147    /// Slots are scoped to residency: eviction releases the slot, so a
3148    /// capacity holding exactly one chunk can serve any number of chunks one
3149    /// at a time, and reads of evicted chunks need no slot at all.
3150    #[mz_ore::test]
3151    fn eviction_releases_the_slot() {
3152        // One 64 KiB slot per class.
3153        let pool = Pool::with_class_capacity(64 << 10).expect("pool creation");
3154        let a = insert(&pool, &mut payload(SMALL, 6));
3155        pool.evict(&a);
3156        // The class's only slot is free again: a second chunk fits without
3157        // falling back to the heap.
3158        let b = insert(&pool, &mut payload(SMALL, 7));
3159        assert_eq!(b.residency(), Residency::UnbackedResident);
3160        assert_eq!(pool.stats().slot_exhausted_fallbacks, 0);
3161        // Reading `a` decompresses straight from its extent while `b` holds
3162        // the class's only slot: copy-out allocates nothing.
3163        assert_eq!(read(&a), payload(SMALL, 6));
3164        assert_eq!(a.residency(), Residency::Evicted);
3165        assert_eq!(read(&b), payload(SMALL, 7));
3166    }
3167
3168    /// The eviction queue holds resident chunks only: an enforcement pass
3169    /// drops entries for evicted chunks, and reads never re-add them, so the
3170    /// scan each insert pays stays proportional to the resident set rather
3171    /// than every chunk ever evicted.
3172    #[mz_ore::test]
3173    fn queue_holds_resident_chunks_only() {
3174        let pool = test_pool(128 << 10);
3175        let mut handles = Vec::new();
3176        for seed in 0..8 {
3177            handles.push(insert(&pool, &mut payload(SMALL, 800 + seed)));
3178        }
3179        // Budget pressure evicted ~6 of 8; one more pass visits the evicted
3180        // entries and drops them (their first visit performed the eviction
3181        // and dropped them already, but second-chance survivors may linger).
3182        pool.enforce_budget();
3183        let resident = handles
3184            .iter()
3185            .filter(|h| h.residency() != Residency::Evicted)
3186            .count();
3187        assert!(
3188            pool.queue_len() <= resident + 1,
3189            "queue ({}) tracks the resident set ({resident}), not all 8 live chunks",
3190            pool.queue_len(),
3191        );
3192        // Reading an evicted chunk copies out of its extent and does not
3193        // re-enqueue it: the queue keeps tracking the resident set.
3194        let evicted = handles
3195            .iter()
3196            .find(|h| h.residency() == Residency::Evicted)
3197            .expect("something was evicted");
3198        let before = pool.queue_len();
3199        assert_eq!(read(evicted).len(), SMALL);
3200        assert_eq!(evicted.residency(), Residency::Evicted);
3201        assert_eq!(pool.queue_len(), before, "reads leave the queue alone");
3202    }
3203
3204    #[mz_ore::test]
3205    fn dead_data_is_never_written() {
3206        let pool = test_pool(256 << 20);
3207        let handle = insert(&pool, &mut payload(SMALL, 7));
3208        drop(handle);
3209        let stats = pool.stats();
3210        assert_eq!(stats.frees, 1);
3211        assert_eq!(stats.writes_elided, 1);
3212        assert_eq!(stats.extent_bytes_written, 0);
3213        assert_eq!(stats.resident_bytes, 0);
3214    }
3215
3216    #[mz_ore::test]
3217    fn budget_is_enforced_on_insert() {
3218        let budget = 128 << 10;
3219        let pool = test_pool(budget);
3220        let mut handles = Vec::new();
3221        for seed in 0..8 {
3222            handles.push(insert(&pool, &mut payload(SMALL, 100 + seed)));
3223        }
3224        let stats = pool.stats();
3225        assert!(
3226            stats.resident_bytes <= u64::cast_from(budget),
3227            "resident {} exceeds budget {}",
3228            stats.resident_bytes,
3229            budget,
3230        );
3231        assert!(stats.evictions_compress >= 6);
3232        let resident = handles
3233            .iter()
3234            .filter(|h| {
3235                matches!(
3236                    h.residency(),
3237                    Residency::UnbackedResident | Residency::BackedResident
3238                )
3239            })
3240            .count();
3241        assert_eq!(resident, 2, "budget holds exactly two small chunks");
3242    }
3243
3244    /// Budget enforcement is single-flight: an insert that trips it while a
3245    /// pass holds the `enforcing` guard bails on `WouldBlock`, trusting that
3246    /// pass. If the holder is already past its final `resident_bytes` read, the
3247    /// bailed insert's bytes are neither read by the holder nor enforced by the
3248    /// bailer, and no later insert re-trips enforcement, so the pool stays over
3249    /// budget. The fix re-runs the pass while any caller was turned away.
3250    ///
3251    /// The test hook freezes the holder's pass in that window to make the race
3252    /// deterministic: the holder parks having found the budget satisfied, the
3253    /// main thread inserts over budget and is turned away, then the holder
3254    /// resumes. The `gate` is used for both rendezvous.
3255    #[mz_ore::test]
3256    fn racing_insert_is_not_dropped_by_budget_single_flight() {
3257        // Budget for exactly one small chunk.
3258        let budget = 64 << 10;
3259        let pool = test_pool(budget);
3260        let gate = std::sync::Arc::new(std::sync::Barrier::new(2));
3261
3262        let holder = {
3263            let pool = pool.clone();
3264            let gate = std::sync::Arc::clone(&gate);
3265            std::thread::spawn(move || -> ChunkHandle {
3266                ENFORCE_BUDGET_HOOK.with(|cell| {
3267                    *cell.borrow_mut() = Some(Box::new(move || {
3268                        gate.wait(); // parked, holding the guard
3269                        gate.wait(); // resume once the race is done
3270                    }));
3271                });
3272                // At budget: the pass finds it satisfied and parks at the hook.
3273                insert(&pool, &mut payload(SMALL, 1))
3274            })
3275        };
3276
3277        gate.wait(); // holder is parked in enforcement, holding the guard
3278        // Push over budget; this insert is turned away by the held guard.
3279        let _over = insert(&pool, &mut payload(SMALL, 2));
3280        gate.wait(); // let the holder resume and release the guard
3281        // Kept alive past the assert: freeing it would drop its bytes and mask
3282        // the overshoot.
3283        let _held = holder.join().expect("holder panicked");
3284
3285        // Nothing re-trips enforcement, so the pool must not be left over budget.
3286        let resident = pool.stats().resident_bytes;
3287        assert!(
3288            resident <= u64::cast_from(budget),
3289            "resident {resident} exceeds budget {budget}: racing insert escaped enforcement",
3290        );
3291    }
3292
3293    #[mz_ore::test]
3294    fn insertion_debt_is_bounded_during_enforcement() {
3295        let budget = 2 * SMALL * 8;
3296        let pool = test_pool(budget);
3297        let guard = pool.0.enforcing.lock().expect("enforcement lock");
3298        let mut handles = Vec::new();
3299        for seed in 0..16 {
3300            handles.push(insert(&pool, &mut payload(SMALL, seed)));
3301            assert!(
3302                pool.stats().resident_bytes <= u64::cast_from(budget + SMALL * 8),
3303                "an occupied enforcer must not allow unlimited insertion debt",
3304            );
3305        }
3306        drop(guard);
3307        for (seed, handle) in handles.iter().enumerate() {
3308            assert_eq!(read(handle), payload(SMALL, u64::cast_from(seed)));
3309        }
3310        drop(handles);
3311        assert_eq!(pool.stats().resident_bytes, 0);
3312        assert_eq!(pool.stats().live_chunks, 0);
3313        assert_eq!(pool.stats().extent_resident_bytes, 0);
3314    }
3315
3316    #[mz_ore::test]
3317    fn admission_reserves_before_concurrent_fills() {
3318        let budget = 2 * SMALL * 8;
3319        let pool = test_pool(budget);
3320        let guard = pool.0.enforcing.lock().expect("enforcement lock");
3321        let gate = Arc::new(std::sync::Barrier::new(9));
3322        let threads: Vec<_> = (0..8u64)
3323            .map(|seed| {
3324                let pool = pool.clone();
3325                let gate = Arc::clone(&gate);
3326                std::thread::spawn(move || {
3327                    pool.insert_with(SMALL, ChunkHints::default(), &TEST_CODEC, |dst| {
3328                        gate.wait();
3329                        gate.wait();
3330                        dst.copy_from_slice(&payload(SMALL, seed));
3331                    })
3332                })
3333            })
3334            .collect();
3335        gate.wait();
3336        let reserved = pool.stats().resident_bytes;
3337        // Release every producer even if the assertion fails.
3338        gate.wait();
3339        let handles: Vec<_> = threads
3340            .into_iter()
3341            .map(|t| t.join().expect("producer panicked"))
3342            .collect();
3343        drop(guard);
3344        assert!(reserved <= u64::cast_from(budget + SMALL * 8));
3345        assert!(pool.stats().direct_extent_inserts > 0);
3346        for (seed, handle) in handles.iter().enumerate() {
3347            assert_eq!(read(handle), payload(SMALL, u64::cast_from(seed)));
3348        }
3349        drop(handles);
3350        assert_eq!(pool.stats().resident_bytes, 0);
3351        assert_eq!(pool.stats().live_chunks, 0);
3352    }
3353
3354    #[mz_ore::test]
3355    fn set_budget_retunes_in_place() {
3356        let pool = test_pool(usize::MAX);
3357        let mut handles = Vec::new();
3358        for seed in 0..8 {
3359            handles.push(insert(&pool, &mut payload(SMALL, 200 + seed)));
3360        }
3361        assert_eq!(pool.stats().evictions_compress, 0);
3362
3363        // Shrinking the budget evicts immediately.
3364        pool.set_budget(128 << 10);
3365        let stats = pool.stats();
3366        assert!(stats.resident_bytes <= 128 << 10);
3367        assert!(stats.evictions_compress >= 6);
3368
3369        // Growing it leaves headroom: a fresh insert stays resident.
3370        pool.set_budget(usize::MAX);
3371        let h = insert(&pool, &mut payload(SMALL, 300));
3372        assert_eq!(h.residency(), Residency::UnbackedResident);
3373        for h in &handles {
3374            assert_eq!(read(h).len(), SMALL);
3375        }
3376    }
3377
3378    #[mz_ore::test]
3379    fn second_chance_prefers_untouched_victims() {
3380        // Budget holds one and a half small chunks.
3381        let pool = test_pool((64 << 10) + (32 << 10));
3382        let orig_a = payload(SMALL, 8);
3383        let handle_a = insert(&pool, &mut orig_a.clone());
3384        assert_eq!(read(&handle_a), orig_a);
3385        // Inserting B overflows the budget; A is older but touched, so the
3386        // enforcer gives it a second chance and evicts untouched B instead.
3387        let handle_b = insert(&pool, &mut payload(SMALL, 9));
3388        assert_eq!(handle_a.residency(), Residency::UnbackedResident);
3389        assert_eq!(handle_b.residency(), Residency::Evicted);
3390    }
3391
3392    /// Depth-hinted chunks are evicted before younger ones: the deep chunk
3393    /// loses even though the young chunk is older and both are untouched
3394    /// (plain FIFO would have evicted the older, young one). Also exercises
3395    /// band clamping: depths beyond the last band share it.
3396    #[mz_ore::test]
3397    fn eviction_prefers_deeper_chunks() {
3398        // Budget of one small chunk.
3399        let pool = test_pool(64 << 10);
3400        let young = insert(&pool, &mut payload(SMALL, 900));
3401        let deep = insert_at_depth(&pool, 255, &mut payload(SMALL, 901));
3402        assert_eq!(young.residency(), Residency::UnbackedResident);
3403        assert_eq!(deep.residency(), Residency::Evicted);
3404    }
3405
3406    /// Eager backing visits deeper chunks first, mirroring eviction order,
3407    /// so the chunks evicted first are the ones already backed.
3408    #[mz_ore::test]
3409    fn backing_prefers_deeper_chunks() {
3410        let pool = test_pool(256 << 20);
3411        let young = insert(&pool, &mut payload(SMALL, 902));
3412        let deep = insert_at_depth(&pool, 2, &mut payload(SMALL, 903));
3413        assert!(pool.back_step());
3414        assert_eq!(deep.residency(), Residency::BackedResident);
3415        assert_eq!(young.residency(), Residency::UnbackedResident);
3416        assert!(pool.back_step());
3417        assert_eq!(young.residency(), Residency::BackedResident);
3418    }
3419
3420    #[mz_ore::test]
3421    fn empty_insert_consumes_no_slot() {
3422        let pool = test_pool(256 << 20);
3423        let mut data = Vec::new();
3424        let handle = insert(&pool, &mut data);
3425        assert_eq!(handle.size_class_bytes(), None);
3426        assert!(read(&handle).is_empty());
3427        // Reads clear the destination even for empty chunks.
3428        let mut out = vec![1u64, 2, 3];
3429        handle.read_into(&mut out);
3430        assert!(out.is_empty());
3431        drop(handle);
3432        let stats = pool.stats();
3433        assert_eq!(stats.resident_bytes, 0);
3434        assert_eq!(stats.writes_elided, 0);
3435    }
3436
3437    #[mz_ore::test]
3438    fn oversize_round_trips() {
3439        let pool = test_pool(256 << 20);
3440        let words = SIZE_CLASSES[SIZE_CLASSES.len() - 1] / 8 + 1;
3441        let orig = payload(words, 10);
3442        let handle = insert(&pool, &mut orig.clone());
3443        assert_eq!(handle.residency(), Residency::Oversize);
3444        assert_eq!(handle.size_class_bytes(), None);
3445        let stats = pool.stats();
3446        assert_eq!(stats.oversize_bytes, u64::cast_from(words * 8));
3447        // The payload outgrew the largest class, and no class was exhausted.
3448        assert_eq!(stats.oversize_payloads, 1);
3449        assert_eq!(stats.slot_exhausted_fallbacks, 0);
3450        // Explicit eviction and budget enforcement leave oversize chunks
3451        // resident.
3452        pool.evict(&handle);
3453        pool.enforce_budget();
3454        assert_eq!(handle.residency(), Residency::Oversize);
3455        assert_eq!(read(&handle), orig);
3456        drop(handle);
3457        let stats = pool.stats();
3458        assert_eq!(stats.oversize_bytes, 0);
3459        assert_eq!(stats.resident_bytes, 0);
3460    }
3461
3462    #[mz_ore::test]
3463    fn payload_lands_in_smallest_fitting_class() {
3464        let pool = test_pool(256 << 20);
3465        let handle = insert(&pool, &mut payload((100 << 10) / 8, 11));
3466        assert_eq!(handle.size_class_bytes(), Some(128 << 10));
3467        let exact = insert(&pool, &mut payload(SMALL, 12));
3468        assert_eq!(exact.size_class_bytes(), Some(64 << 10));
3469    }
3470
3471    #[mz_ore::test]
3472    #[cfg_attr(miri, ignore)] // too slow
3473    fn multithreaded_smoke() {
3474        // Budget of one small chunk: four inserting threads keep the pool
3475        // over budget, so every insert's enforcement pass selects victims
3476        // owned by other threads, racing cross-thread eviction against
3477        // copy-out reads and frees.
3478        let pool = test_pool(64 << 10);
3479        let per_thread = rounds(50, 3);
3480        let threads: Vec<_> = (0..4u64)
3481            .map(|t| {
3482                let pool = pool.clone();
3483                std::thread::spawn(move || {
3484                    for round in 0..per_thread {
3485                        let seed = t * 1000 + round;
3486                        let orig = payload(SMALL, seed);
3487                        let handle = insert(&pool, &mut orig.clone());
3488                        pool.evict(&handle);
3489                        assert_eq!(read(&handle), orig);
3490                        // Enforcement racing reads must never corrupt them.
3491                        pool.enforce_budget();
3492                        assert_eq!(read(&handle), orig);
3493                        drop(handle);
3494                    }
3495                })
3496            })
3497            .collect();
3498        for thread in threads {
3499            thread.join().expect("worker thread panicked");
3500        }
3501        let stats = pool.stats();
3502        assert_eq!(stats.inserts, 4 * per_thread);
3503        assert_eq!(stats.frees, 4 * per_thread);
3504        assert_eq!(stats.resident_bytes, 0);
3505    }
3506
3507    #[mz_ore::test]
3508    #[cfg_attr(miri, ignore)] // too slow
3509    fn concurrent_read_enforce_churn() {
3510        // Races the three actors that can touch one chunk's slot: readers
3511        // copying shared chunks out and verifying them, an enforcer evicting
3512        // them (the zero budget makes every chunk a victim), and a churner
3513        // whose insert/free traffic turns the queue over. Contents are
3514        // asserted on every read, so an eviction or slot recycle racing a
3515        // copy-out shows up as corruption.
3516        let pool = test_pool(0);
3517        let shared: Arc<Vec<(Vec<u64>, ChunkHandle)>> = Arc::new(
3518            (0..4u64)
3519                .map(|seed| {
3520                    let orig = payload(SMALL, 600 + seed);
3521                    let handle = insert(&pool, &mut orig.clone());
3522                    (orig, handle)
3523                })
3524                .collect(),
3525        );
3526        let churn = rounds(300, 6);
3527        let mut threads = Vec::new();
3528        for t in 0..2u64 {
3529            let shared = Arc::clone(&shared);
3530            threads.push(std::thread::spawn(move || {
3531                for round in 0..churn {
3532                    let (orig, handle) = &shared[usize::cast_from((t + round) % 4)];
3533                    assert_eq!(&read(handle), orig);
3534                }
3535            }));
3536        }
3537        {
3538            let pool = pool.clone();
3539            threads.push(std::thread::spawn(move || {
3540                for _ in 0..2 * churn {
3541                    pool.enforce_budget();
3542                }
3543            }));
3544        }
3545        {
3546            let pool = pool.clone();
3547            threads.push(std::thread::spawn(move || {
3548                for round in 0..churn {
3549                    let orig = payload(SMALL, 700 + round);
3550                    let handle = insert(&pool, &mut orig.clone());
3551                    assert_eq!(read(&handle), orig);
3552                }
3553            }));
3554        }
3555        for thread in threads {
3556            thread.join().expect("worker thread panicked");
3557        }
3558        drop(shared);
3559        assert_eq!(pool.stats().resident_bytes, 0);
3560    }
3561
3562    /// Read-only traffic never raises resident bytes: every chunk starts
3563    /// evicted and is then read once, with no inserts in between. Reads copy
3564    /// out of the extents and leave every chunk evicted, so a seek-heavy
3565    /// phase costs no pool memory at all.
3566    #[mz_ore::test]
3567    fn reads_never_raise_resident_bytes() {
3568        let pool = test_pool(128 << 10);
3569        let origs: Vec<_> = (0..8u64).map(|seed| payload(SMALL, 300 + seed)).collect();
3570        let handles: Vec<_> = origs
3571            .iter()
3572            .map(|o| insert(&pool, &mut o.clone()))
3573            .collect();
3574        for handle in &handles {
3575            pool.evict(handle);
3576        }
3577        assert_eq!(pool.stats().resident_bytes, 0);
3578        for (index, handle) in handles.iter().enumerate() {
3579            assert_eq!(read(handle), origs[index]);
3580            assert_eq!(handle.residency(), Residency::Evicted);
3581            assert_eq!(pool.stats().resident_bytes, 0);
3582        }
3583    }
3584
3585    /// Evict-then-free churn under a generous RSS target: the compressed
3586    /// tier never crosses its cap, so enforcement never visits (and never
3587    /// drops) extent-queue entries, and pruning alone must keep the queue
3588    /// proportional to the live resident extents.
3589    #[mz_ore::test]
3590    #[cfg_attr(miri, ignore)] // too slow
3591    fn extent_queue_stays_bounded_under_cap() {
3592        let pool = test_pool(256 << 20);
3593        pool.set_rss_target(1 << 40);
3594        for seed in 0..rounds(1000, 48) {
3595            let handle = insert(&pool, &mut payload(SMALL, seed));
3596            pool.evict(&handle);
3597            drop(handle);
3598        }
3599        assert_eq!(pool.stats().extent_resident_bytes, 0);
3600        let len = pool.extent_queue_len();
3601        assert!(
3602            len <= 32,
3603            "extent queue holds {len} entries for zero resident extents",
3604        );
3605    }
3606
3607    /// A warm slot reused for a smaller payload round-trips: the tail
3608    /// release past the new payload must not disturb the payload itself,
3609    /// and the ledger credits exactly the payload.
3610    #[mz_ore::test]
3611    fn warm_reuse_with_smaller_payload_round_trips() {
3612        // Budget 8 MiB: warm cap = 1 MiB, so a 64 KiB slot parks warm.
3613        let pool = test_pool(8 << 20);
3614        let full = insert(&pool, &mut payload(SMALL, 60));
3615        drop(full);
3616        assert_eq!(pool.stats().warm_bytes, 64 << 10, "freed slot parks warm");
3617        // A payload of just over a page reuses the warm slot; the slot's
3618        // pages past it are released.
3619        let words = 4096 / 8 + 1;
3620        let orig = payload(words, 61);
3621        let handle = insert(&pool, &mut orig.clone());
3622        let stats = pool.stats();
3623        assert_eq!(stats.warm_reuses, 1, "reused the warm slot");
3624        assert_eq!(stats.resident_bytes, u64::cast_from(words * 8));
3625        assert_eq!(read(&handle), orig);
3626        // Round-trips through the extent as well.
3627        pool.evict(&handle);
3628        pool.poison_free_slots();
3629        assert_eq!(read(&handle), orig);
3630        drop(handle);
3631        assert_eq!(pool.stats().resident_bytes, 0);
3632    }
3633
3634    /// Heap-backed chunks count as resident but can never be evicted, so
3635    /// the budget must not force slotted chunks out on their account: with
3636    /// unevictable bytes alone exceeding the budget, a slotted chunk that
3637    /// fits the budget stays resident.
3638    #[mz_ore::test]
3639    fn unevictable_bytes_do_not_force_eviction() {
3640        // One 64 KiB slot per class: the second and third inserts fall
3641        // back to the heap.
3642        let pool = Pool::with_class_capacity(64 << 10).expect("pool creation");
3643        pool.set_budget(64 << 10);
3644        let slotted = insert(&pool, &mut payload(SMALL, 91));
3645        let heap_a = insert(&pool, &mut payload(SMALL, 92));
3646        let heap_b = insert(&pool, &mut payload(SMALL, 93));
3647        assert_eq!(heap_a.residency(), Residency::Oversize);
3648        assert_eq!(heap_b.residency(), Residency::Oversize);
3649        let stats = pool.stats();
3650        assert!(stats.oversize_bytes > 64 << 10, "unevictable exceed budget");
3651        assert_eq!(slotted.residency(), Residency::UnbackedResident);
3652        assert_eq!(stats.evictions_compress, 0);
3653        assert_eq!(read(&slotted), payload(SMALL, 91));
3654        assert_eq!(read(&heap_a), payload(SMALL, 92));
3655    }
3656
3657    /// A slotless empty chunk survives an explicit evict with spill
3658    /// scheduling enabled: nothing is handed to the spill threads and the
3659    /// chunk stays readable.
3660    #[mz_ore::test]
3661    fn evict_of_empty_chunk_is_a_no_op() {
3662        let pool = test_pool(usize::MAX);
3663        pool.enable_spill_without_threads();
3664        let empty = insert(&pool, &mut Vec::new());
3665        pool.evict(&empty);
3666        assert_eq!(empty.residency(), Residency::UnbackedResident);
3667        assert!(!pool.spill_step(), "nothing was scheduled");
3668        assert_eq!(pool.stats().spill_scheduled, 0);
3669        assert!(read(&empty).is_empty());
3670    }
3671
3672    #[mz_ore::test]
3673    fn queue_stays_bounded_under_budget() {
3674        // Chunk churn that never exceeds the budget: the enforcer's eviction
3675        // loop never runs, so stale queue entries must be reclaimed by
3676        // pruning alone.
3677        let pool = test_pool(256 << 20);
3678        for seed in 0..rounds(1000, 48) {
3679            let handle = insert(&pool, &mut payload(SMALL, seed));
3680            drop(handle);
3681        }
3682        let len = pool.queue_len();
3683        assert!(len <= 32, "queue holds {len} entries for zero live chunks");
3684    }
3685
3686    #[mz_ore::test]
3687    fn spill_async_evict_round_trip() {
3688        let pool = test_pool(usize::MAX);
3689        pool.enable_spill_without_threads();
3690        let h = insert(&pool, &mut payload(SMALL, 400));
3691        pool.evict(&h);
3692        assert_eq!(h.residency(), Residency::WriteInFlight);
3693        // Readable while in flight: the slot is still populated, and the
3694        // copy-out coexists with the spill thread's compression read.
3695        assert_eq!(read(&h), payload(SMALL, 400));
3696        // Reads leave no trace, so the eviction commits.
3697        assert!(pool.spill_step());
3698        assert_eq!(h.residency(), Residency::Evicted);
3699        let stats = pool.stats();
3700        assert_eq!(stats.spill_scheduled, 1);
3701        assert_eq!(stats.evictions_compress, 1);
3702        pool.poison_free_slots();
3703        assert_eq!(read(&h), payload(SMALL, 400));
3704    }
3705
3706    #[mz_ore::test]
3707    fn spill_freed_while_queued_is_elided() {
3708        let pool = test_pool(usize::MAX);
3709        pool.enable_spill_without_threads();
3710        let h = insert(&pool, &mut payload(SMALL, 401));
3711        pool.evict(&h);
3712        assert_eq!(h.residency(), Residency::WriteInFlight);
3713        drop(h);
3714        assert!(pool.spill_step());
3715        let stats = pool.stats();
3716        assert_eq!(stats.spill_cancelled, 1);
3717        assert_eq!(stats.writes_elided, 1, "freed before compression: elided");
3718        assert_eq!(stats.extent_bytes_written, 0, "no extent was written");
3719        assert_eq!(stats.resident_bytes, 0, "slot accounting settled");
3720    }
3721
3722    #[mz_ore::test]
3723    fn spill_take_in_flight_cancels_write() {
3724        let pool = test_pool(usize::MAX);
3725        pool.enable_spill_without_threads();
3726        let orig = payload(SMALL, 402);
3727        let h = insert(&pool, &mut orig.clone());
3728        pool.evict(&h);
3729        assert_eq!(h.residency(), Residency::WriteInFlight);
3730        let mut out = Vec::new();
3731        h.take(&mut out);
3732        assert_eq!(out, orig);
3733        assert!(pool.spill_step());
3734        let stats = pool.stats();
3735        assert_eq!(stats.frees, 1);
3736        assert_eq!(stats.spill_cancelled, 1);
3737        assert_eq!(stats.writes_elided, 1, "taken before compression: elided");
3738        assert_eq!(stats.extent_bytes_written, 0, "no extent was written");
3739        assert_eq!(stats.resident_bytes, 0, "slot accounting settled");
3740        assert_eq!(stats.live_chunks, 0);
3741    }
3742
3743    #[mz_ore::test]
3744    #[cfg_attr(miri, ignore)] // too slow
3745    fn spill_threads_end_to_end() {
3746        let pool = test_pool(128 << 10);
3747        pool.set_spill_threads(2);
3748        let mut handles = Vec::new();
3749        for seed in 0..rounds(16, 6) {
3750            handles.push(insert(&pool, &mut payload(SMALL, 500 + seed)));
3751        }
3752        pool.quiesce_spill();
3753        let stats = pool.stats();
3754        assert!(
3755            stats.spill_scheduled > 0,
3756            "budget pressure should have scheduled spills",
3757        );
3758        for (i, h) in handles.iter().enumerate() {
3759            assert_eq!(read(h), payload(SMALL, 500 + u64::cast_from(i)));
3760        }
3761        pool.join_spill_threads();
3762    }
3763
3764    /// Races the `WriteInFlight` protocol in its true concurrent form:
3765    /// spill threads compress slots without the state lock while owner
3766    /// threads copy the same chunks out under it and drop chunks mid-flight
3767    /// (both cancellation windows). Contents are asserted on every read, so
3768    /// a compression or slot release racing a copy-out shows up as
3769    /// corruption; under Miri the aliasing itself is checked.
3770    #[mz_ore::test]
3771    fn spill_threads_race_reads_and_drops() {
3772        let pool = test_pool(usize::MAX);
3773        pool.set_spill_threads(2);
3774        let iters = rounds(50, 6);
3775        let mut threads = Vec::new();
3776        for t in 0..2u64 {
3777            let pool = pool.clone();
3778            threads.push(std::thread::spawn(move || {
3779                for round in 0..iters {
3780                    let orig = payload(SMALL, t * 10_000 + round);
3781                    let handle = insert(&pool, &mut orig.clone());
3782                    // Hands the chunk to the spill threads (`WriteInFlight`).
3783                    pool.evict(&handle);
3784                    // Copy-out read racing the unlocked compression read.
3785                    assert_eq!(read(&handle), orig);
3786                    if round % 2 == 0 {
3787                        // Free while queued or mid-compression: the
3788                        // cancellation windows own the deferred cleanup.
3789                        drop(handle);
3790                    } else {
3791                        assert_eq!(read(&handle), orig);
3792                    }
3793                }
3794            }));
3795        }
3796        for thread in threads {
3797            thread.join().expect("worker thread panicked");
3798        }
3799        pool.quiesce_spill();
3800        pool.join_spill_threads();
3801        assert_eq!(pool.stats().resident_bytes, 0);
3802    }
3803
3804    /// The identity codec stores the body verbatim: eviction and reads,
3805    /// whole and by range, reconstruct it unchanged.
3806    #[mz_ore::test]
3807    fn identity_codec_round_trips() {
3808        let pool = test_pool(usize::MAX);
3809        let want = payload(SMALL, 601);
3810        let h = pool.insert_with(SMALL, ChunkHints::default(), &IDENTITY_CODEC, |dst| {
3811            dst.copy_from_slice(&want);
3812        });
3813        assert_eq!(read(&h), want);
3814        pool.evict(&h);
3815        assert_eq!(read(&h), want, "round-trips through the extent");
3816        pool.evict(&h);
3817        let mut range = Vec::new();
3818        h.read_range_into(8..24, &mut range);
3819        assert_eq!(range, want[8..24], "range reads copy the range directly");
3820    }
3821
3822    #[mz_ore::test]
3823    fn insert_with_fills_in_place() {
3824        let pool = test_pool(usize::MAX);
3825        let want = payload(SMALL, 600);
3826        let h = pool.insert_with(SMALL, ChunkHints::default(), &TEST_CODEC, |dst| {
3827            assert_eq!(dst.len(), SMALL, "fill sees exactly the chunk length");
3828            dst.copy_from_slice(&want);
3829        });
3830        assert_eq!(h.residency(), Residency::UnbackedResident);
3831        assert_eq!(read(&h), want);
3832        pool.evict(&h);
3833        assert_eq!(read(&h), want, "round-trips through the extent");
3834
3835        // Empty and oversize take their fallback paths.
3836        let empty = pool.insert_with(0, ChunkHints::default(), &TEST_CODEC, |dst| {
3837            assert!(dst.is_empty())
3838        });
3839        assert!(read(&empty).is_empty());
3840        let big_len = (SIZE_CLASSES[SIZE_CLASSES.len() - 1] / 8) + 1;
3841        let big = pool.insert_with(big_len, ChunkHints::default(), &TEST_CODEC, |dst| {
3842            dst.fill(7)
3843        });
3844        assert_eq!(big.residency(), Residency::Oversize);
3845        assert_eq!(read(&big).len(), big_len);
3846    }
3847
3848    #[mz_ore::test]
3849    fn slot_exhaustion_degrades_to_heap() {
3850        // Two 64 KiB slots per class at this capacity; the third insert finds
3851        // no slot and must fall back to the heap rather than panic.
3852        let pool = Pool::with_class_capacity(128 << 10).expect("pool creation");
3853        let a = insert(&pool, &mut payload(SMALL, 700));
3854        let b = insert(&pool, &mut payload(SMALL, 701));
3855        let c = insert(&pool, &mut payload(SMALL, 702));
3856        assert_eq!(a.residency(), Residency::UnbackedResident);
3857        assert_eq!(b.residency(), Residency::UnbackedResident);
3858        assert_eq!(
3859            c.residency(),
3860            Residency::Oversize,
3861            "fallback is heap-backed"
3862        );
3863        assert_eq!(pool.stats().slot_exhausted_fallbacks, 1);
3864        assert_eq!(read(&c), payload(SMALL, 702));
3865        // Freeing a slotted chunk lets the next insert use the region again.
3866        drop(a);
3867        let d = insert(&pool, &mut payload(SMALL, 703));
3868        assert_eq!(d.residency(), Residency::UnbackedResident);
3869        assert_eq!(read(&d), payload(SMALL, 703));
3870    }
3871}