diff --git a/docs/BENCHMARKS_AND_TUNING.md b/docs/BENCHMARKS_AND_TUNING.md index 0eeadb8..8cdd118 100644 --- a/docs/BENCHMARKS_AND_TUNING.md +++ b/docs/BENCHMARKS_AND_TUNING.md @@ -75,7 +75,7 @@ genuine advantage over tokio's task abort model. ### Spawn-heavy workloads (19–70×) -Every smarm actor `mmap`s a 64 KiB stack with a guard page. This is +Every smarm actor `mmap`s a 64 KiB stack reserve with a 64 KiB PROT_NONE guard below (both per-actor configurable since RFC 019; the reserve is demand-paged). This is a syscall. Tokio tasks are heap-allocated state machines — no stack, no syscall, ~100 bytes each. For workloads that spawn thousands of short-lived actors per second, this is a structural disadvantage. diff --git a/src/runtime.rs b/src/runtime.rs index bd5c025..d07dbd4 100644 --- a/src/runtime.rs +++ b/src/runtime.rs @@ -160,6 +160,8 @@ pub struct Config { alloc_interval: u32, timeslice_cycles: u64, stack_pool_cap: usize, + stack_reserve: usize, + stack_guard: usize, max_actors: usize, wake_slot: bool, node_id: crate::pg::NodeId, @@ -175,6 +177,8 @@ impl Config { alloc_interval: crate::preempt::DEFAULT_ALLOC_INTERVAL, timeslice_cycles: crate::preempt::DEFAULT_TIMESLICE_CYCLES, stack_pool_cap: n * 4, + stack_reserve: DEFAULT_STACK_RESERVE, + stack_guard: DEFAULT_STACK_GUARD, max_actors: DEFAULT_MAX_ACTORS, wake_slot: false, node_id: crate::pg::DEFAULT_NODE_ID, @@ -194,6 +198,8 @@ impl Config { alloc_interval: crate::preempt::DEFAULT_ALLOC_INTERVAL, timeslice_cycles: crate::preempt::DEFAULT_TIMESLICE_CYCLES, stack_pool_cap: max * 4, + stack_reserve: DEFAULT_STACK_RESERVE, + stack_guard: DEFAULT_STACK_GUARD, max_actors: DEFAULT_MAX_ACTORS, wake_slot: false, node_id: crate::pg::DEFAULT_NODE_ID, @@ -226,6 +232,30 @@ impl Config { self } + /// Default per-actor stack reserve (RFC 019). A *virtual* reservation — + /// anonymous mmap is demand-paged, so RSS follows touched pages, not + /// this number — but overflowing it hits the guard and dies. Page-rounded. + /// Per-actor override: `SpawnOpts::stack_reserve`. + /// Default: [`DEFAULT_STACK_RESERVE`] (64 KiB) — the million-cheap-actors + /// story is unchanged; big stacks are opt-in. + pub fn stack_reserve(mut self, n: usize) -> Self { + assert!(n > 0, "stack_reserve must be non-zero"); + self.stack_reserve = n; + self + } + + /// Default PROT_NONE guard below each stack (RFC 019). Address space + /// only. Page-rounded. Rust overflow is caught by any single page + /// (probestack touches pages in order); the wide default exists for + /// unprobed FFI frames, which can step over a small guard in one + /// `sub rsp`. Per-actor override: `SpawnOpts::guard_size`. + /// Default: [`DEFAULT_STACK_GUARD`] (64 KiB). + pub fn stack_guard(mut self, n: usize) -> Self { + assert!(n > 0, "stack_guard must be non-zero"); + self.stack_guard = n; + self + } + /// Capacity of the actor slot table — the maximum number of /// **simultaneously live** actors (total spawned over a run is unbounded; /// slots are recycled). The table is a fixed slab allocated once at @@ -293,6 +323,8 @@ impl Default for Config { alloc_interval: crate::preempt::DEFAULT_ALLOC_INTERVAL, timeslice_cycles: crate::preempt::DEFAULT_TIMESLICE_CYCLES, stack_pool_cap: avail * 4, + stack_reserve: DEFAULT_STACK_RESERVE, + stack_guard: DEFAULT_STACK_GUARD, max_actors: DEFAULT_MAX_ACTORS, wake_slot: false, node_id: crate::pg::DEFAULT_NODE_ID, @@ -383,7 +415,12 @@ impl RuntimeStats { // Slot — packed state word + hot atomics + cold lifecycle data // --------------------------------------------------------------------------- -pub(crate) const ACTOR_STACK_SIZE: usize = 64 * 1024; +/// Default usable stack reserve per actor (RFC 019). See [`Config::stack_reserve`]. +pub const DEFAULT_STACK_RESERVE: usize = 64 * 1024; + +/// Default PROT_NONE guard below each actor stack (RFC 019). Raised from one +/// page so unprobed C frames cannot leap it. See [`Config::stack_guard`]. +pub const DEFAULT_STACK_GUARD: usize = 64 * 1024; pub(crate) type Closure = Box; @@ -802,17 +839,23 @@ pub(crate) struct RuntimeInner { pub(crate) stack_pool: RawMutex>, /// Maximum number of stacks to retain in the pool. pub(crate) stack_pool_cap: usize, + /// Default stack shape (RFC 019), pre-page-rounded so it compares exactly + /// against `Stack::shape()`. Only stacks of exactly this shape are pooled. + pub(crate) stack_reserve: usize, + pub(crate) stack_guard: usize, } impl RuntimeInner { // Private constructor taking the parsed Config fields one-for-one; a params - // struct would only move the same 8 values across the call boundary. + // struct would only move the same 10 values across the call boundary. #[allow(clippy::too_many_arguments)] fn new( thread_count: usize, alloc_interval: u32, timeslice_cycles: u64, stack_pool_cap: usize, + stack_reserve: usize, + stack_guard: usize, max_actors: usize, wake_slot: bool, node_id: crate::pg::NodeId, @@ -854,6 +897,8 @@ impl RuntimeInner { process_groups: RawMutex::new(crate::pg::ProcessGroups::new()), stack_pool: RawMutex::new(Vec::new()), stack_pool_cap, + stack_reserve: crate::stack::round_to_pages(stack_reserve), + stack_guard: crate::stack::round_to_pages(stack_guard), }) } @@ -1052,6 +1097,8 @@ pub fn init(config: Config) -> Runtime { config.alloc_interval, config.timeslice_cycles, config.stack_pool_cap, + config.stack_reserve, + config.stack_guard, config.max_actors, config.wake_slot, config.node_id, @@ -1309,6 +1356,48 @@ pub const ROOT_PID: Pid = Pid::new(u32::MAX, u32::MAX); // Spawn-side slot installation // --------------------------------------------------------------------------- +// --------------------------------------------------------------------------- +// Stack acquisition / recycling — RFC 019 pool rule +// --------------------------------------------------------------------------- + +/// Get a stack of the requested shape (`None` ⇒ the runtime defaults). +/// +/// Pool rule (RFC 019 §1): the pool is a uniform `Vec` of +/// default-shaped stacks and stays that way. Default-shaped requests try the +/// pool first; custom shapes always mmap fresh (and `recycle_stack` never +/// admits them, so a pooled stack is default-shaped by induction). The pool +/// lock is dropped before any mmap: no syscall ever stalls another spawner. +pub(crate) fn acquire_stack( + inner: &RuntimeInner, + shape: Option<(usize, usize)>, +) -> crate::stack::Stack { + let (reserve, guard) = shape.unwrap_or((inner.stack_reserve, inner.stack_guard)); + let default_shaped = crate::stack::round_to_pages(reserve) == inner.stack_reserve + && crate::stack::round_to_pages(guard) == inner.stack_guard; + if default_shaped { + if let Some(stack) = inner.stack_pool.lock().pop() { + return stack; + } + } + match crate::stack::Stack::new(reserve, guard) { + Ok(stack) => stack, + Err(e) => panic!("stack allocation failed: {e}"), + } +} + +/// Return a dead actor's stack: pooled if default-shaped and under cap, +/// otherwise dropped here → munmap (custom shapes and cap overflow alike). +pub(crate) fn recycle_stack(inner: &RuntimeInner, stack: crate::stack::Stack) { + if stack.shape() == (inner.stack_reserve, inner.stack_guard) { + let mut pool = inner.stack_pool.lock(); + if pool.len() < inner.stack_pool_cap { + pool.push(stack); + } + // else: fall through — drop → munmap. + } + // Custom-shaped (or cap overflow): `stack` drops here → munmap. +} + /// Install a freshly spawned actor into the slot `idx` (which must have come /// from `allocate_slot`) and publish it as Queued. Returns the new `Pid`. /// Called by `scheduler::spawn_under`; lives here next to its inverse @@ -1437,13 +1526,7 @@ fn finalize_actor(inner: &Arc, pid: Pid, outcome: Outcome) { // (the trap sender can unpark its receiver — keep that outside too). let supervisor_pid = actor.supervisor; let Actor { stack, .. } = actor; - { - let mut pool = inner.stack_pool.lock(); - if pool.len() < inner.stack_pool_cap { - pool.push(stack); - } - // else: drop here → munmap, same as before - } + recycle_stack(inner, stack); // Deliver to supervisor. ROOT_PID resolves to no slot → silently absorbed. let sender = inner.slot_at(supervisor_pid).and_then(|sup| { diff --git a/src/scheduler.rs b/src/scheduler.rs index 99eb666..8074a0b 100644 --- a/src/scheduler.rs +++ b/src/scheduler.rs @@ -288,15 +288,10 @@ pub fn spawn(f: impl FnOnce() + Send + 'static) -> JoinHandle { /// rather than its true caller. pub fn spawn_under(supervisor: Pid, f: impl FnOnce() + Send + 'static) -> JoinHandle { let supervisor = supervisor.erase(); - // Stack + closure boxing happen before ANY runtime lock is taken: no - // syscall and no allocation ever stalls another scheduler thread. - let stack = with_runtime(|inner| inner.stack_pool.lock().pop()) - .unwrap_or_else(|| { - match crate::stack::Stack::new(crate::runtime::ACTOR_STACK_SIZE) { - Ok(stack) => stack, - Err(e) => panic!("stack allocation failed: {e}"), - } - }); + // Stack + closure boxing happen before the slot locks are taken; the + // pool lock inside acquire_stack is dropped before any mmap, so no + // syscall ever stalls another scheduler thread. + let stack = with_runtime(|inner| crate::runtime::acquire_stack(inner, None)); let sp = init_actor_stack(stack.top(), crate::actor::trampoline); let closure: crate::runtime::Closure = Box::new(f); diff --git a/src/stack.rs b/src/stack.rs index b742531..aca1041 100644 --- a/src/stack.rs +++ b/src/stack.rs @@ -1,32 +1,45 @@ -//! mmap-based growable stack with a guard page below. +//! mmap-based actor stack with a PROT_NONE guard region below (RFC 019). //! //! Layout (low → high address): -//! [ guard page (PROT_NONE) | stack region ] -//! ^ top() — initial stack pointer +//! [ guard region (PROT_NONE) | stack region ] +//! ^ top() — initial stack pointer //! -//! Stacks grow downward. Overflow lands in the guard page → SIGSEGV. +//! Stacks grow downward. Overflow lands in the guard region → SIGSEGV. +//! +//! Both the usable reserve and the guard are caller-chosen (page-rounded). +//! The reserve is a *virtual* reservation: anonymous mmap is demand-paged, +//! so RSS is touched-pages, not reserve × actors. The guard costs address +//! space only. A wide guard (the runtime defaults to 64 KiB) exists for +//! unprobed FFI frames: Rust frames touch pages in order (probestack), so +//! one page catches Rust overflow, but a C frame with a large local can +//! step over a single page in one `sub rsp`. use std::io; pub struct Stack { - /// Bottom of the entire mmap'd region (start of guard page). + /// Bottom of the entire mmap'd region (start of the guard). base: *mut u8, /// Total mmap'd size: guard_size + stack_size. total_size: usize, - /// Usable stack size (excluding guard page). + /// Usable stack size (excluding the guard). stack_size: usize, + /// PROT_NONE region below the usable stack. + guard_size: usize, } // Stack owns its memory; safe to send across threads. unsafe impl Send for Stack {} impl Stack { - /// Allocate a new stack. `stack_size` is the usable region; one page is - /// added below as a guard page. Both are rounded up to the page size. - pub fn new(stack_size: usize) -> io::Result { + /// Allocate a new stack. `stack_size` is the usable region; `guard_size` + /// is mapped PROT_NONE below it. Both are rounded up to the page size + /// and must be non-zero. + pub fn new(stack_size: usize, guard_size: usize) -> io::Result { + assert!(stack_size > 0, "stack_size must be non-zero"); + assert!(guard_size > 0, "guard_size must be non-zero"); let page = page_size(); let stack_size = round_up(stack_size, page); - let guard_size = page; + let guard_size = round_up(guard_size, page); let total_size = guard_size + stack_size; let base = unsafe { @@ -53,7 +66,7 @@ impl Stack { return Err(err); } - Ok(Self { base, total_size, stack_size }) + Ok(Self { base, total_size, stack_size, guard_size }) } /// 16-byte-aligned top of the usable region. @@ -62,14 +75,31 @@ impl Stack { (raw_top & !15) as *mut u8 } - /// Pointer to the bottom of the usable region (just above the guard page). + /// Pointer to the bottom of the usable region (just above the guard). pub fn usable_base(&self) -> *mut u8 { - unsafe { self.base.add(page_size()) } + unsafe { self.base.add(self.guard_size) } } pub fn stack_size(&self) -> usize { self.stack_size } + + pub fn guard_size(&self) -> usize { + self.guard_size + } + + /// `(stack_size, guard_size)` after page rounding. The pool rule + /// (RFC 019 §1) compares this against the runtime defaults: only + /// default-shaped stacks are pooled. + pub fn shape(&self) -> (usize, usize) { + (self.stack_size, self.guard_size) + } +} + +/// Round `n` up to whole pages — the same rounding `Stack::new` applies, so +/// runtime defaults stored pre-rounded compare exactly against [`Stack::shape`]. +pub(crate) fn round_to_pages(n: usize) -> usize { + round_up(n, page_size()) } impl Drop for Stack { diff --git a/tests/context.rs b/tests/context.rs index 150bcb1..dc02b23 100644 --- a/tests/context.rs +++ b/tests/context.rs @@ -23,7 +23,7 @@ extern "C-unwind" fn actor_simple() { #[test] fn actor_runs_and_returns_to_scheduler() { reset_log(); - let stack = Stack::new(64 * 1024).unwrap(); + let stack = Stack::new(64 * 1024, 4096).unwrap(); let sp = init_actor_stack(stack.top(), actor_simple); set_actor_sp(sp); unsafe { switch_to_actor() }; @@ -40,7 +40,7 @@ extern "C-unwind" fn actor_two_steps() { #[test] fn actor_yields_and_resumes() { reset_log(); - let stack = Stack::new(64 * 1024).unwrap(); + let stack = Stack::new(64 * 1024, 4096).unwrap(); let sp = init_actor_stack(stack.top(), actor_two_steps); set_actor_sp(sp); @@ -85,7 +85,7 @@ extern "C-unwind" fn actor_reg_check() { #[test] fn callee_saved_registers_survive_yield() { - let stack = Stack::new(64 * 1024).unwrap(); + let stack = Stack::new(64 * 1024, 4096).unwrap(); let sp = init_actor_stack(stack.top(), actor_reg_check); set_actor_sp(sp); unsafe { switch_to_actor(); switch_to_actor(); } @@ -117,8 +117,8 @@ extern "C-unwind" fn actor_b() { #[test] fn two_actors_dont_corrupt_each_other() { - let stack_a = Stack::new(64 * 1024).unwrap(); - let stack_b = Stack::new(64 * 1024).unwrap(); + let stack_a = Stack::new(64 * 1024, 4096).unwrap(); + let stack_b = Stack::new(64 * 1024, 4096).unwrap(); let sp_a = init_actor_stack(stack_a.top(), actor_a); let sp_b = init_actor_stack(stack_b.top(), actor_b); diff --git a/tests/runtime.rs b/tests/runtime.rs index a3c95d6..8bc20cf 100644 --- a/tests/runtime.rs +++ b/tests/runtime.rs @@ -517,3 +517,35 @@ fn runtime_reusable_after_root_panic() { r.run(move || ran_t.store(true, Ordering::Relaxed)); assert!(ran.load(Ordering::Relaxed), "runtime unusable after root panic"); } + +// --------------------------------------------------------------------------- +// RFC 019 — Config stack knobs +// --------------------------------------------------------------------------- + +/// Burn ~`frames` × 4 KiB of stack; probestack touches pages in order so +/// exceeding the reserve would hit the guard and SIGSEGV the process. +#[inline(never)] +fn burn_stack(frames: usize) -> u64 { + let mut local = [0u8; 4096]; + local[0] = frames as u8; + let below = if frames == 0 { 0 } else { burn_stack(frames - 1) }; + std::hint::black_box(&mut local); + below.wrapping_add(local[0] as u64) +} + +#[test] +fn config_stack_reserve_permits_deep_recursion() { + // ~256 KiB of frames: four times the old fixed 64 KiB reserve. With + // Config::stack_reserve raised this must complete; before RFC 019 it + // could only segfault. + let rt = smarm::runtime::init(Config::exact(1).stack_reserve(1024 * 1024)); + let done = Arc::new(AtomicBool::new(false)); + let done2 = done.clone(); + rt.run(move || { + spawn(move || { + std::hint::black_box(burn_stack(64)); + done2.store(true, Ordering::SeqCst); + }).join(); + }); + assert!(done.load(Ordering::SeqCst)); +} diff --git a/tests/stack.rs b/tests/stack.rs index cec741a..1ca44c3 100644 --- a/tests/stack.rs +++ b/tests/stack.rs @@ -7,13 +7,13 @@ use smarm::stack::Stack; #[test] fn top_is_16_byte_aligned() { - let s = Stack::new(64 * 1024).unwrap(); + let s = Stack::new(64 * 1024, 4096).unwrap(); assert_eq!(s.top() as usize % 16, 0); } #[test] fn top_is_within_allocation() { - let s = Stack::new(64 * 1024).unwrap(); + let s = Stack::new(64 * 1024, 4096).unwrap(); let top = s.top() as usize; let base = s.usable_base() as usize; assert!(top > base); @@ -22,7 +22,7 @@ fn top_is_within_allocation() { #[test] fn write_and_read_top_of_stack() { - let s = Stack::new(64 * 1024).unwrap(); + let s = Stack::new(64 * 1024, 4096).unwrap(); let sentinel: u64 = 0xDEAD_BEEF_CAFE_1234; unsafe { let ptr = s.top().sub(8) as *mut u64; @@ -33,7 +33,7 @@ fn write_and_read_top_of_stack() { #[test] fn write_and_read_bottom_of_usable_region() { - let s = Stack::new(64 * 1024).unwrap(); + let s = Stack::new(64 * 1024, 4096).unwrap(); let sentinel: u64 = 0x0102_0304_0506_0708; unsafe { let ptr = s.usable_base() as *mut u64; @@ -44,17 +44,17 @@ fn write_and_read_bottom_of_usable_region() { #[test] fn small_stack_allocates() { - assert!(Stack::new(4096).is_ok()); + assert!(Stack::new(4096, 4096).is_ok()); } #[test] fn large_stack_allocates() { - assert!(Stack::new(8 * 1024 * 1024).is_ok()); + assert!(Stack::new(8 * 1024 * 1024, 4096).is_ok()); } #[test] fn stack_size_at_least_requested() { - let s = Stack::new(64 * 1024).unwrap(); + let s = Stack::new(64 * 1024, 4096).unwrap(); assert!(s.stack_size() >= 64 * 1024); } @@ -68,15 +68,28 @@ use std::process::Command; fn run_as_child_if_requested() { match env::var("SMARM_SUBTEST").as_deref() { Ok("guard_page_direct") => { - let s = Stack::new(64 * 1024).unwrap(); + let s = Stack::new(64 * 1024, 4096).unwrap(); unsafe { let guard_ptr = s.usable_base().sub(1); guard_ptr.write_volatile(0xAB); } std::process::exit(0); } + Ok("wide_guard_top") => { + // One byte below the usable region, 64 KiB guard: must fault. + let s = Stack::new(64 * 1024, 64 * 1024).unwrap(); + unsafe { s.usable_base().sub(1).write_volatile(0xAB); } + std::process::exit(0); + } + Ok("wide_guard_bottom") => { + // The very bottom page of a 64 KiB guard: an unprobed C-style + // leap over a small guard lands here — must still fault. + let s = Stack::new(64 * 1024, 64 * 1024).unwrap(); + unsafe { s.usable_base().sub(64 * 1024).write_volatile(0xAB); } + std::process::exit(0); + } Ok("stack_overflow") => { - let s = Stack::new(64 * 1024).unwrap(); + let s = Stack::new(64 * 1024, 4096).unwrap(); unsafe { let mut ptr = s.top().sub(1); let stop = s.usable_base().sub(1); @@ -121,3 +134,58 @@ fn stack_overflow_causes_sigsegv() { assert_eq!(status.signal(), Some(11), "expected SIGSEGV, got: {:?}", status); } } + +// --------------------------------------------------------------------------- +// RFC 019 — explicit shape: rounding, guard accessor, wide-guard coverage. +// --------------------------------------------------------------------------- + +#[test] +fn sizes_round_up_to_page() { + let s = Stack::new(64 * 1024 + 1, 4096 + 1).unwrap(); + assert_eq!(s.stack_size() % 4096, 0); + assert_eq!(s.guard_size() % 4096, 0); + assert!(s.stack_size() >= 64 * 1024 + 1); + assert!(s.guard_size() >= 4096 + 1); +} + +#[test] +fn shape_reports_rounded_sizes() { + let s = Stack::new(64 * 1024, 64 * 1024).unwrap(); + assert_eq!(s.shape(), (64 * 1024, 64 * 1024)); +} + +#[test] +fn usable_base_sits_above_guard() { + let s = Stack::new(64 * 1024, 64 * 1024).unwrap(); + // The usable region must start exactly guard_size above the mapping + // base: a write at usable_base is legal, one byte below is not (the + // subprocess tests below prove the "not"). + let sentinel: u64 = 0x1111_2222_3333_4444; + unsafe { + let ptr = s.usable_base() as *mut u64; + ptr.write_volatile(sentinel); + assert_eq!(ptr.read_volatile(), sentinel); + } +} + +#[test] +fn wide_guard_faults_at_top() { + run_as_child_if_requested(); + let status = spawn_subtest("wide_guard_top"); + #[cfg(unix)] + { + use std::os::unix::process::ExitStatusExt; + assert_eq!(status.signal(), Some(11), "expected SIGSEGV, got: {:?}", status); + } +} + +#[test] +fn wide_guard_faults_at_bottom() { + run_as_child_if_requested(); + let status = spawn_subtest("wide_guard_bottom"); + #[cfg(unix)] + { + use std::os::unix::process::ExitStatusExt; + assert_eq!(status.signal(), Some(11), "expected SIGSEGV, got: {:?}", status); + } +}