diff --git a/src/runtime.rs b/src/runtime.rs index 22c0476..0d6ba2a 100644 --- a/src/runtime.rs +++ b/src/runtime.rs @@ -441,6 +441,15 @@ pub const SHRINK_THRESHOLD: usize = 256 * 1024; /// rationale as [`SHRINK_THRESHOLD`]). pub const SHRINK_COOLDOWN: u32 = 64; +/// RFC 019 §6: the entry-end span (highest addresses — the frames the next +/// actor faults first) a recycled stack keeps resident; everything below it +/// is `MADV_DONTNEED`ed before the stack re-enters the pool. Ratified as a +/// constant, not Config, alongside the shrink knobs; the 64 KiB value was a +/// flagged Claude-solo call at ratification — it equals the default reserve, +/// so with an unraised Config the zap is a no-op and only Configs that raise +/// the default reserve pay it. +pub const RECYCLE_RETAIN: usize = 64 * 1024; + pub(crate) type Closure = Box; /// Lifecycle data, mutated only under the slot's cold [`RawMutex`]. Everything @@ -1472,6 +1481,12 @@ pub(crate) fn acquire_stack( /// otherwise dropped here → munmap (custom shapes and cap overflow alike). pub(crate) fn recycle_stack(inner: &RuntimeInner, stack: crate::stack::Stack) { if stack.shape() == (inner.stack_reserve, inner.stack_guard) { + // RFC 019 §6: zap the dead spike before pooling, BEFORE taking the + // pool lock — acquire_stack's invariant is that no syscall ever + // stalls another spawner under it. On the rare cap-overflow the zap + // is wasted work ahead of the munmap; harmless, and cheaper than a + // second lock round-trip to find out. + stack.recycle_zap(RECYCLE_RETAIN); let mut pool = inner.stack_pool.lock(); if pool.len() < inner.stack_pool_cap { pool.push(stack); diff --git a/src/stack.rs b/src/stack.rs index 9b4d968..2f3386f 100644 --- a/src/stack.rs +++ b/src/stack.rs @@ -94,6 +94,29 @@ impl Stack { pub fn shape(&self) -> (usize, usize) { (self.stack_size, self.guard_size) } + + /// Pool-recycle zap (RFC 019 §6): `MADV_DONTNEED` everything below the + /// retained entry end `[top − retain, top)` — the span the next actor's + /// shallow frames land in stays resident, the dead spike below it is + /// released. The stack is unowned at the call site (its actor is dead), + /// so a synchronous eager zap races nothing and the RSS drop is + /// immediate — a museum of worst-case spikes is exactly what a pool must + /// not be; DONTNEED's ~8× per-page cost vs FREE is irrelevant off the + /// hot path. Advisory like the park-path shrink: a failure degrades to + /// "the pool keeps RSS", never to incorrectness. No-op (no syscall) when + /// `retain` covers the whole usable region — i.e. always, at the 64 KiB + /// default reserve. + pub(crate) fn recycle_zap(&self, retain: usize) { + if let Some((off, len)) = retain_range(self.stack_size, retain, page_size()) { + unsafe { + libc::madvise( + self.usable_base().add(off) as *mut libc::c_void, + len, + libc::MADV_DONTNEED, + ); + } + } + } } /// Round `n` up to whole pages — the same rounding `Stack::new` applies, so @@ -144,12 +167,76 @@ pub(crate) fn shrink_range(hwm: usize, sp: usize, page: usize) -> Option<(usize, } } +/// The `(offset_from_usable_base, len)` span the pool recycle DONTNEEDs +/// (RFC 019 §6): everything below the retained entry end. "Bottom RETAIN of +/// the stack" is read stack-wise (entry frames = highest addresses of a +/// downward stack): the retained span is `[top − page_up(retain), top)`, the +/// zapped span is the rest — retaining the low-address deep end instead +/// would keep the coldest pages and release the ones the next actor faults +/// first. `retain` rounds *up* to whole pages (retain more, zap less), so +/// with `stack_size` page-rounded by `Stack::new` the result is always +/// page-aligned. Checked math: `retain ≥ stack_size` (notably the default +/// 64 KiB reserve with the 64 KiB RETAIN) and overflow collapse to `None`. +pub(crate) fn retain_range(stack_size: usize, retain: usize, page: usize) -> Option<(usize, usize)> { + debug_assert!(page.is_power_of_two()); + let retain = retain.checked_add(page - 1)? & !(page - 1); // page_up(retain) + let len = stack_size.checked_sub(retain)?; + if len == 0 { + return None; + } + Some((0, len)) +} + #[cfg(test)] mod tests { - use super::shrink_range; + use super::{retain_range, shrink_range}; const PG: usize = 4096; + #[test] + fn retain_covers_whole_stack_is_a_noop() { + // The default config: reserve == RETAIN == 64 KiB. No zap, no syscall. + assert_eq!(retain_range(16 * PG, 16 * PG, PG), None); + assert_eq!(retain_range(PG, PG, PG), None); + } + + #[test] + fn retain_larger_than_stack_is_a_noop() { + assert_eq!(retain_range(16 * PG, 17 * PG, PG), None); + assert_eq!(retain_range(PG, usize::MAX, PG), None); // page_up overflows + } + + #[test] + fn retain_zero_zaps_everything() { + assert_eq!(retain_range(16 * PG, 0, PG), Some((0, 16 * PG))); + } + + #[test] + fn retain_rounds_up_zapping_less() { + // 1 byte of retain keeps a whole page. + assert_eq!(retain_range(16 * PG, 1, PG), Some((0, 15 * PG))); + assert_eq!(retain_range(16 * PG, PG + 1, PG), Some((0, 14 * PG))); + } + + #[test] + fn retain_one_page_short_of_stack() { + assert_eq!(retain_range(2 * PG, PG, PG), Some((0, PG))); + } + + #[test] + fn retain_range_is_page_aligned() { + for size_pg in [1usize, 2, 3, 16, 1024] { + for retain in [0usize, 1, PG - 1, PG, PG + 1, 4 * PG, size_pg * PG] { + if let Some((off, len)) = retain_range(size_pg * PG, retain, PG) { + assert_eq!(off, 0); + assert_eq!(len % PG, 0); + assert!(len <= size_pg * PG); + assert!(len > 0); + } + } + } + } + #[test] fn empty_and_inverted_spans_are_none() { assert_eq!(shrink_range(0x8000_0000, 0x8000_0000, PG), None); // hwm == sp diff --git a/tests/stack_recycle.rs b/tests/stack_recycle.rs new file mode 100644 index 0000000..03b485b --- /dev/null +++ b/tests/stack_recycle.rs @@ -0,0 +1,123 @@ +//! RFC 019 commit 5 — pool recycle zaps a dead stack down to its retained +//! entry end, observed from the outside. +//! +//! A default-shaped stack that spiked deep and then died must not carry its +//! spike into the pool as resident RSS: `recycle_stack` DONTNEEDs everything +//! below the top `RECYCLE_RETAIN` bytes before pushing. The zap is +//! synchronous on the death path, so the drop is immediate — but the death +//! path itself races the observer's `join` return, hence the brief poll. +//! +//! Residency is measured with `mincore`, not smaps: a neighboring rw anon +//! mapping can land flush against the stack top and the kernel merges the +//! VMAs (observed under the full test run), so per-mapping smaps fields +//! over-count. The PROT_NONE guard below can never merge, so the usable +//! base is exactly the anchor VMA's start, and `mincore` counts pages +//! within [usable_base, usable_base + reserve) regardless of merging. + +use smarm::runtime::{Config, RECYCLE_RETAIN}; +use smarm::{channel, spawn, yield_now}; + +const RESERVE: usize = 4 * 1024 * 1024; + +/// Burn ~`frames` × 4 KiB of stack, dirtying every frame. +#[inline(never)] +fn burn_stack(frames: usize) -> u64 { + let mut local = [0u8; 4096]; + local[0] = frames as u8; + let below = if frames == 0 { 0 } else { burn_stack(frames - 1) }; + std::hint::black_box(&mut local); + below.wrapping_add(local[0] as u64) +} + +/// Resident-page count over [lo, lo + len) via mincore (len page-aligned). +fn resident_pages(lo: usize, len: usize) -> usize { + let page = 4096; + let mut vec = vec![0u8; len / page]; + let ret = unsafe { + libc::mincore(lo as *mut libc::c_void, len, vec.as_mut_ptr()) + }; + assert_eq!(ret, 0, "mincore failed: {}", std::io::Error::last_os_error()); + vec.iter().filter(|&&b| b & 1 != 0).count() +} + +/// The [start, end) of the VMA containing `addr`. +fn vma_containing(addr: usize) -> (usize, usize) { + let maps = std::fs::read_to_string("/proc/self/maps").unwrap(); + for line in maps.lines() { + if let Some((range, _)) = line.split_once(' ') { + if let Some((a, b)) = range.split_once('-') { + if let (Ok(start), Ok(end)) = + (usize::from_str_radix(a, 16), usize::from_str_radix(b, 16)) + { + if start <= addr && addr < end { + return (start, end); + } + } + } + } + } + panic!("no VMA contains {addr:#x}"); +} + +fn vma_exists(addr: usize) -> bool { + let maps = std::fs::read_to_string("/proc/self/maps").unwrap(); + for line in maps.lines() { + if let Some((range, _)) = line.split_once(' ') { + if let Some((a, b)) = range.split_once('-') { + if let (Ok(start), Ok(end)) = + (usize::from_str_radix(a, 16), usize::from_str_radix(b, 16)) + { + if start <= addr && addr < end { + return true; + } + } + } + } + } + false +} + +#[test] +fn recycle_zaps_dead_stack_down_to_retain() { + // Default reserve raised so the pool holds big stacks (default-shaped ⇒ + // pooled) and the zap has something to bite; single scheduler. + let rt = smarm::runtime::init(Config::exact(1).stack_reserve(RESERVE)); + rt.run(|| { + let (tx, rx) = channel::(); + + let h = spawn(move || { + let probe = 0u8; + let anchor = &probe as *const u8 as usize; + // The guard below is PROT_NONE and can never merge with the + // usable region, so the anchor VMA's start IS the usable base. + let (vlo, _) = vma_containing(anchor); + // Dirty ~3 MiB of the 4 MiB reserve, then die. + std::hint::black_box(burn_stack(768)); + tx.send(vlo).unwrap(); + }); + + let usable_base = rx.recv().unwrap(); + h.join().unwrap(); + + // The zap span is everything below the retained entry end. DONTNEED + // on private anon discards synchronously and unconditionally, so + // this must go to exactly zero resident pages; the poll only covers + // the death path racing join's return. + let zap_len = RESERVE - RECYCLE_RETAIN; + let mut resident = usize::MAX; + for _ in 0..10_000 { + resident = resident_pages(usable_base, zap_len); + if resident == 0 { + break; + } + yield_now(); + } + assert_eq!( + resident, 0, + "recycled stack's zap span still resident: {resident} pages in \ + [{usable_base:#x}, +{zap_len:#x})" + ); + // Pooled, not munmapped: the mapping must still be there. + assert!(vma_exists(usable_base), "default-shaped stack was unmapped instead of pooled"); + }); +}