Compare commits
12
Commits
8c764e9169
..
v0.6.0
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
301e3463e3 | ||
|
|
410ba33d82 | ||
|
|
5fd8aecf55 | ||
|
|
7d8b9e0310 | ||
|
|
8225716b11 | ||
|
|
3cb64eefc2 | ||
|
|
0fe052bc7e | ||
|
|
a03a7ca01e | ||
|
|
d4839f1d81 | ||
|
|
2854b560d6 | ||
|
|
7b026cfe56 | ||
|
|
006a3283e7 |
+17
-1
@@ -2,7 +2,23 @@
|
||||
# smarm pre-commit gate: clippy the library (src/) with warnings as errors.
|
||||
# unwrap_used / expect_used are denied (Cargo.toml [lints.clippy]): library
|
||||
# code must not hide a panic behind unwrap/expect. Tests/examples are not gated.
|
||||
#
|
||||
# Toolchain resolution: prefer an installed cargo-clippy; on machines whose
|
||||
# rust comes without the clippy component (e.g. NixOS home-manager), fall
|
||||
# back to an ephemeral nix-shell toolchain. The fallback uses its own target
|
||||
# dir (target/clippy) because the shell's rustc version may differ from the
|
||||
# default toolchain's — mixed-compiler artifacts in one target dir are an
|
||||
# E0514 hard error. MSRV (Cargo.toml rust-version) keeps the older shell
|
||||
# toolchain a legitimate gate.
|
||||
set -eu
|
||||
[ -f "$HOME/.cargo/env" ] && . "$HOME/.cargo/env"
|
||||
cd "$(git rev-parse --show-toplevel)"
|
||||
cargo clippy --lib -- -D warnings
|
||||
if cargo clippy --version >/dev/null 2>&1; then
|
||||
cargo clippy --lib -- -D warnings
|
||||
elif command -v nix-shell >/dev/null 2>&1; then
|
||||
nix-shell -p clippy -p cargo -p rustc \
|
||||
--run 'CARGO_TARGET_DIR=target/clippy cargo clippy --lib -- -D warnings'
|
||||
else
|
||||
echo "pre-commit: cargo clippy unavailable and no nix-shell fallback" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
+4
-1
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "smarm"
|
||||
version = "0.4.0"
|
||||
version = "0.6.0"
|
||||
edition = "2021"
|
||||
rust-version = "1.95"
|
||||
|
||||
@@ -39,6 +39,9 @@ rq-mutex = []
|
||||
rq-mpmc = []
|
||||
rq-striped = []
|
||||
|
||||
[build-dependencies]
|
||||
cc = "1"
|
||||
|
||||
[dependencies]
|
||||
libc = "0.2"
|
||||
|
||||
|
||||
@@ -0,0 +1,11 @@
|
||||
fn main() {
|
||||
// RFC 019 §7 test canary (agreed Q3): compiled without stack-clash
|
||||
// protection so its 96 KiB local is a genuine one-displacement guard
|
||||
// jumper; distro-hardened compilers would otherwise probe it page-wise
|
||||
// and defeat the test's purpose.
|
||||
cc::Build::new()
|
||||
.file("canary/canary.c")
|
||||
.flag_if_supported("-fno-stack-clash-protection")
|
||||
.compile("smarm_canary");
|
||||
println!("cargo:rerun-if-changed=canary/canary.c");
|
||||
}
|
||||
@@ -0,0 +1,14 @@
|
||||
/* RFC 019 §7 FFI canary: an honest unprobed C frame with a 96 KiB local,
|
||||
* touched from its LOW end first — the exact "one sub rsp steps over a small
|
||||
* guard" pattern the RFC's motivating incident hit (a cargo-vendored gz
|
||||
* build; cc-invoked builds do not enable -fstack-clash-protection, and this
|
||||
* file pins that off explicitly so the canary stays a canary even on
|
||||
* hardened-default toolchains). */
|
||||
void smarm_canary_burn(void) {
|
||||
volatile char buf[96 * 1024];
|
||||
buf[0] = 1; /* deepest address first */
|
||||
for (unsigned i = 0; i < sizeof buf; i += 4096) {
|
||||
buf[i] = (char)i;
|
||||
}
|
||||
buf[sizeof buf - 1] = 1;
|
||||
}
|
||||
@@ -75,7 +75,7 @@ genuine advantage over tokio's task abort model.
|
||||
|
||||
### Spawn-heavy workloads (19–70×)
|
||||
|
||||
Every smarm actor `mmap`s a 64 KiB stack with a guard page. This is
|
||||
Every smarm actor `mmap`s a 64 KiB stack reserve with a 64 KiB PROT_NONE guard below (both per-actor configurable since RFC 019; the reserve is demand-paged). This is
|
||||
a syscall. Tokio tasks are heap-allocated state machines — no stack,
|
||||
no syscall, ~100 bytes each. For workloads that spawn thousands of
|
||||
short-lived actors per second, this is a structural disadvantage.
|
||||
|
||||
+31
-5
@@ -182,7 +182,7 @@ use crate::channel::{channel, select, select_timeout, Receiver, RecvTimeoutError
|
||||
use crate::monitor::{demonitor, monitor, Down, Monitor};
|
||||
use crate::pid::Pid;
|
||||
use crate::registry::{register_with, resolve_named_sender, RegisterError};
|
||||
use crate::scheduler::{cancel_timer, request_stop, send_after_to, spawn, spawn_under};
|
||||
use crate::scheduler::{cancel_timer, request_stop, send_after_to};
|
||||
use crate::timer::TimerId;
|
||||
use std::cell::Cell;
|
||||
use std::collections::HashMap;
|
||||
@@ -643,11 +643,17 @@ pub struct GenServerBuilder<G: GenServer> {
|
||||
state: G,
|
||||
infos: Vec<Receiver<G::Info>>,
|
||||
supervisor: Option<Pid>,
|
||||
stack_opts: crate::scheduler::SpawnOpts,
|
||||
}
|
||||
|
||||
impl<G: GenServer> GenServerBuilder<G> {
|
||||
pub fn new(state: G) -> Self {
|
||||
GenServerBuilder { state, infos: Vec::new(), supervisor: None }
|
||||
GenServerBuilder {
|
||||
state,
|
||||
infos: Vec::new(),
|
||||
supervisor: None,
|
||||
stack_opts: crate::scheduler::SpawnOpts::default(),
|
||||
}
|
||||
}
|
||||
|
||||
/// Add an out-of-band channel; messages arriving on it are dispatched to
|
||||
@@ -665,6 +671,14 @@ impl<G: GenServer> GenServerBuilder<G> {
|
||||
self
|
||||
}
|
||||
|
||||
/// Stack shape for the server actor (RFC 019) — see
|
||||
/// [`SpawnOpts`](crate::SpawnOpts). Useful for servers that recurse
|
||||
/// deeply or call into FFI with large C frames.
|
||||
pub fn stack_opts(mut self, opts: crate::scheduler::SpawnOpts) -> Self {
|
||||
self.stack_opts = opts;
|
||||
self
|
||||
}
|
||||
|
||||
/// Spawn the server actor and hand back its [`GenServerRef`]. The server's
|
||||
/// lifetime is governed by its refs, not by joining, so the backing join
|
||||
/// handle is dropped.
|
||||
@@ -686,10 +700,16 @@ impl<G: GenServer> GenServerBuilder<G> {
|
||||
/// under the name before returning.
|
||||
fn spawn_server(self) -> GenServerRef<G> {
|
||||
let (tx, rx) = channel::<Envelope<G>>();
|
||||
let GenServerBuilder { state, infos, supervisor } = self;
|
||||
let GenServerBuilder { state, infos, supervisor, stack_opts } = self;
|
||||
let handle = match supervisor {
|
||||
Some(sup) => spawn_under(sup, move || server_loop::<G>(rx, state, infos)),
|
||||
None => spawn(move || server_loop::<G>(rx, state, infos)),
|
||||
Some(sup) => {
|
||||
crate::scheduler::spawn_under_with(sup, stack_opts, move || {
|
||||
server_loop::<G>(rx, state, infos)
|
||||
})
|
||||
}
|
||||
None => crate::scheduler::spawn_with(stack_opts, move || {
|
||||
server_loop::<G>(rx, state, infos)
|
||||
}),
|
||||
};
|
||||
GenServerRef { tx, pid: handle.pid() }
|
||||
}
|
||||
@@ -758,6 +778,12 @@ impl<G: GenServer> NamedGenServerBuilder<G> {
|
||||
self
|
||||
}
|
||||
|
||||
/// Stack shape for the server actor (see [`GenServerBuilder::stack_opts`]).
|
||||
pub fn stack_opts(mut self, opts: crate::scheduler::SpawnOpts) -> Self {
|
||||
self.builder = self.builder.stack_opts(opts);
|
||||
self
|
||||
}
|
||||
|
||||
/// Spawn the server and bind its name in one step. Fallible: returns
|
||||
/// [`RegisterError::NameTaken`] if the name is already held by a different
|
||||
/// live server.
|
||||
|
||||
+15
-2
@@ -71,7 +71,7 @@
|
||||
|
||||
use crate::channel::{channel, select, Receiver, Sender};
|
||||
use crate::pid::Pid;
|
||||
use crate::scheduler::{cancel_timer, send_after_to, spawn as spawn_actor};
|
||||
use crate::scheduler::{cancel_timer, send_after_to};
|
||||
use crate::timer::TimerId;
|
||||
use std::collections::{HashMap, VecDeque};
|
||||
use std::marker::PhantomData;
|
||||
@@ -434,8 +434,21 @@ impl<M: Machine> GenStatemRef<M> {
|
||||
///
|
||||
/// Panics if called outside `Runtime::run()`.
|
||||
pub fn spawn<M: Machine>(machine: M) -> GenStatemRef<M> {
|
||||
spawn_with(crate::scheduler::SpawnOpts::default(), machine)
|
||||
}
|
||||
|
||||
/// [`spawn`] with per-actor stack shape overrides (RFC 019) for the machine's
|
||||
/// actor — see [`SpawnOpts`](crate::SpawnOpts). gen_statem has no builder
|
||||
/// (its one-shot `spawn(machine)` shape predates RFC 019), so the opts ride
|
||||
/// a `_with` variant like the scheduler's own spawns.
|
||||
///
|
||||
/// Panics if called outside `Runtime::run()`.
|
||||
pub fn spawn_with<M: Machine>(
|
||||
opts: crate::scheduler::SpawnOpts,
|
||||
machine: M,
|
||||
) -> GenStatemRef<M> {
|
||||
let (tx, rx) = channel::<M::Ev>();
|
||||
let handle = spawn_actor(move || statem_loop(rx, machine));
|
||||
let handle = crate::scheduler::spawn_with(opts, move || statem_loop(rx, machine));
|
||||
GenStatemRef { tx, pid: handle.pid() }
|
||||
}
|
||||
|
||||
|
||||
@@ -179,6 +179,34 @@ pub struct ActorInfo {
|
||||
/// `budget-accounting` feature is enabled, since measuring it costs a
|
||||
/// timestamp read on every resume.
|
||||
pub budget_cycles: u64,
|
||||
/// RFC 019 §8 — this actor's stack, as the runtime sees it. All fields
|
||||
/// are lock-free atomic reads, coherent for this incarnation via the
|
||||
/// same generation check as the counters above. Exact RSS is
|
||||
/// deliberately absent: `mincore` is debug tooling, never a runtime
|
||||
/// path.
|
||||
pub stack: StackInfo,
|
||||
}
|
||||
|
||||
/// RFC 019 §8 — per-actor stack introspection. Sizes are page-rounded, as
|
||||
/// [`Stack::new`](crate::stack::Stack::new) rounds them.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub struct StackInfo {
|
||||
/// Usable stack size ([`SpawnOpts::stack_reserve`]
|
||||
/// (crate::SpawnOpts::stack_reserve) or the Config/default).
|
||||
pub reserve: usize,
|
||||
/// PROT_NONE guard below the usable region.
|
||||
pub guard: usize,
|
||||
/// Sampled high-water depth in bytes: `top − lowest saved sp`. Sampled,
|
||||
/// not exact — the context save at yields/parks/preemptions is the
|
||||
/// sampler (RFC 019 §2), so a spike the actor never yielded inside is
|
||||
/// invisible. 0 depth means "never descheduled at any depth", not
|
||||
/// "never ran".
|
||||
pub depth_high_water: usize,
|
||||
/// Parks on this incarnation since its last shrink (or since install if
|
||||
/// it has never shrunk) — the §3 cooldown counter, live.
|
||||
pub parks_since_shrink: u32,
|
||||
/// §3 shrinks performed on this incarnation.
|
||||
pub shrinks: u32,
|
||||
}
|
||||
|
||||
/// A snapshot of every actor in the runtime at (approximately) one moment.
|
||||
@@ -220,6 +248,22 @@ pub fn snapshot() -> RuntimeSnapshot {
|
||||
/// slot was reused by another), out of range, or was never a real pid at
|
||||
/// all. Unlike [`snapshot`], every field of the result describes the same
|
||||
/// instant, since there is only one actor to read.
|
||||
/// The stack shape `(reserve, guard)` of a live actor, page-rounded — the
|
||||
/// RFC 019 introspection surface's first field (depth sampling and shrink
|
||||
/// counters land with the shrink machinery). `None` if `pid` no longer names
|
||||
/// a live actor. Takes the actor's cold lock briefly; debugging/assertion
|
||||
/// use, not a hot-path call.
|
||||
pub fn stack_shape(pid: Pid) -> Option<(usize, usize)> {
|
||||
with_runtime(|inner| {
|
||||
let slot = inner.slot_at(pid)?;
|
||||
let cold = slot.cold.lock();
|
||||
if slot.generation() != pid.generation() {
|
||||
return None;
|
||||
}
|
||||
cold.actor.as_ref().map(|a| a.stack.shape())
|
||||
})
|
||||
}
|
||||
|
||||
pub fn actor_info(pid: Pid) -> Option<ActorInfo> {
|
||||
with_runtime(|inner| {
|
||||
let slot = inner.slot_at(pid)?;
|
||||
@@ -265,6 +309,14 @@ fn read_slot(slot: &Slot, idx: u32, mail: Option<&MailboxInfo>) -> Option<ActorI
|
||||
drop(cold);
|
||||
|
||||
// Counters are plain atomics, read lock-free.
|
||||
let (reserve, guard, top, hwm, parks_since_shrink, shrinks) = slot.stack_introspect();
|
||||
let stack = StackInfo {
|
||||
reserve,
|
||||
guard,
|
||||
depth_high_water: top.saturating_sub(hwm),
|
||||
parks_since_shrink,
|
||||
shrinks,
|
||||
};
|
||||
let overruns = slot.overruns();
|
||||
let messages_received = slot.messages_received();
|
||||
let budget_cycles = slot.budget_cycles();
|
||||
@@ -290,6 +342,7 @@ fn read_slot(slot: &Slot, idx: u32, mail: Option<&MailboxInfo>) -> Option<ActorI
|
||||
overruns,
|
||||
messages_received,
|
||||
budget_cycles,
|
||||
stack,
|
||||
})
|
||||
}
|
||||
|
||||
|
||||
@@ -13,44 +13,68 @@
|
||||
//! leaves the actor, no copying through an intermediary thread. Built on
|
||||
//! these are the conveniences `read(fd, &mut buf)` and `write(fd, &buf)`.
|
||||
//!
|
||||
//! Architecture
|
||||
//! ============
|
||||
//! Per `run()`, two OS threads:
|
||||
//! - **epoll thread**: owns the epollfd. Loops in `epoll_wait`. On a
|
||||
//! ready fd, pushes `Completion::FdReady { pid, fd, events }` to the
|
||||
//! shared completion queue and writes the scheduler-wake pipe. On the
|
||||
//! shutdown pipe (also registered in epollfd), exits.
|
||||
//! - **pool thread**: blocks on the request mpsc. Runs the closure
|
||||
//! inside `catch_unwind`, pushes `Completion::Blocking { pid, result }`,
|
||||
//! writes the scheduler-wake pipe.
|
||||
//! Architecture (RFC 018: driver-enqueues)
|
||||
//! =======================================
|
||||
//! Per `run()`, two OS threads, each a *producer* behind the runtime's
|
||||
//! two-call contract — make the actor runnable (`unpark_at`, whose enqueue
|
||||
//! tail wakes a parked scheduler), nothing else:
|
||||
//!
|
||||
//! Both threads share a single `completions: Arc<Mutex<VecDeque<Completion>>>`
|
||||
//! and the same scheduler-wake pipe.
|
||||
//! - **epoll thread**: owns `epoll_wait` on the epollfd. On a ready fd it
|
||||
//! removes the parked waiter from the shared `waiters` map and DELs the
|
||||
//! fd (both under the waiters lock — see below), then unparks the
|
||||
//! actor directly. On the shutdown pipe (also registered in the
|
||||
//! epollfd), exits.
|
||||
//! - **pool thread**: blocks on the request mpsc. Runs the closure inside
|
||||
//! `catch_unwind`, stashes the result in the actor's slot
|
||||
//! (`pending_io_result`, under the cold lock, generation-checked),
|
||||
//! decrements the runtime's `io_outstanding`, and unparks the actor.
|
||||
//!
|
||||
//! `epoll_ctl` (register/unregister fd interest) is called by the
|
||||
//! scheduler thread *directly* on the epollfd. That's well-defined per
|
||||
//! `epoll_ctl(2)`: a thread may be calling `epoll_wait` on the epollfd
|
||||
//! while another thread calls `epoll_ctl`. Avoids needing a second mpsc
|
||||
//! and a second wake mechanism.
|
||||
//! There is no shared completion queue and no wake pipe: each producer
|
||||
//! routes its own completion, so the whole byte-vs-completion visibility
|
||||
//! discipline of the drain era — and the stranded-completion hazards it
|
||||
//! defended against — is unrepresentable. Producers reach the runtime
|
||||
//! through a `Weak<RuntimeInner>`: upgraded per completion (the path is
|
||||
//! syscall-bound; the refcount op is noise) and avoiding an Arc cycle
|
||||
//! through `RuntimeInner::io`.
|
||||
//!
|
||||
//! `epoll_ctl` (register fd interest) is called by the scheduler thread
|
||||
//! directly on the epollfd. That's well-defined per `epoll_ctl(2)`: a
|
||||
//! thread may be calling `epoll_wait` on the epollfd while another thread
|
||||
//! calls `epoll_ctl`.
|
||||
//!
|
||||
//! Epoll mode
|
||||
//! ==========
|
||||
//! Level-triggered with EPOLLONESHOT. After a wakeup the kernel
|
||||
//! auto-disarms the fd, so we never get two wakeups for one
|
||||
//! `wait_readable` call. The scheduler explicitly `EPOLL_CTL_DEL`s the fd
|
||||
//! on completion to free the slot for re-registration. Net effect: each
|
||||
//! `wait_readable` call. The epoll thread explicitly `EPOLL_CTL_DEL`s the
|
||||
//! fd on readiness to free the slot for re-registration. Net effect: each
|
||||
//! `wait_readable(fd)` is one ADD, one wakeup, one DEL — symmetric and
|
||||
//! stateless between calls.
|
||||
//!
|
||||
//! ## The waiters lock is the ADD/DEL serialization
|
||||
//!
|
||||
//! Registration (scheduler thread: check-vacant, defensive DEL, ADD,
|
||||
//! insert) and readiness consumption (epoll thread: remove, DEL) each run
|
||||
//! entirely under the `waiters` mutex. This is what makes the
|
||||
//! oneshot-rearm race unrepresentable: a woken actor re-registering the
|
||||
//! same fd cannot interleave with the epoll thread's DEL for the *previous*
|
||||
//! registration — whichever takes the lock second sees a consistent
|
||||
//! kernel-side state. Lock order: `io` (the runtime's outer mutex, held by
|
||||
//! scheduler-side callers) → `waiters` → slot/queue leaves via `unpark_at`.
|
||||
//! The epoll thread takes `waiters` without `io` — it must never take
|
||||
//! `io`, both for lock-order hygiene and because teardown holds `io` while
|
||||
//! joining it.
|
||||
//!
|
||||
//! Fd hygiene
|
||||
//! ==========
|
||||
//! An actor stopped while waiting on an fd unwinds out of `wait_fd`'s park;
|
||||
//! a drop guard there (armed after a successful register, forgotten on a
|
||||
//! normal wake) removes the `waiters` entry iff it is still that wait's
|
||||
//! `(pid, epoch)` and only then `EPOLL_CTL_DEL`s the fd — an entry already
|
||||
//! consumed by a racing `FdReady` means the fd may carry someone else's
|
||||
//! fresh registration, which must be left alone. `epoll_register` keeps a
|
||||
//! defensive bare DEL before ADD as belt-and-braces.
|
||||
//! normal wake) calls [`IoThread::cancel_waiter`], which removes the
|
||||
//! `waiters` entry iff it is still that wait's `(pid, epoch)` and only then
|
||||
//! `EPOLL_CTL_DEL`s the fd — an entry already consumed by the epoll thread
|
||||
//! means the fd may carry someone else's fresh registration, which must be
|
||||
//! left alone. `epoll_register` keeps a defensive bare DEL before ADD as
|
||||
//! belt-and-braces.
|
||||
//!
|
||||
//! Buffers used with `read`/`write` should be on fds opened with
|
||||
//! `O_NONBLOCK`. If they aren't, the syscall may block the scheduler
|
||||
@@ -68,13 +92,14 @@
|
||||
//! they have no equivalent panic-propagation path.
|
||||
|
||||
use crate::pid::Pid;
|
||||
use crate::runtime::RuntimeInner;
|
||||
use std::any::Any;
|
||||
use std::collections::{HashMap, VecDeque};
|
||||
use std::collections::HashMap;
|
||||
use std::io;
|
||||
use std::os::fd::RawFd;
|
||||
use std::panic;
|
||||
use std::sync::mpsc;
|
||||
use std::sync::{Arc, Mutex};
|
||||
use std::sync::atomic::Ordering;
|
||||
use std::sync::{mpsc, Arc, Mutex, Weak};
|
||||
use std::thread::JoinHandle as OsJoinHandle;
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -86,42 +111,29 @@ use std::thread::JoinHandle as OsJoinHandle;
|
||||
pub type IoResult = Result<Box<dyn Any + Send>, Box<dyn Any + Send>>;
|
||||
|
||||
struct Request {
|
||||
/// The submitter's park-epoch — carried through to the `Blocking`
|
||||
/// completion so the wake is epoch-matched.
|
||||
/// The submitter's park-epoch — the eventual wake is epoch-matched.
|
||||
epoch: u32,
|
||||
pid: Pid,
|
||||
/// The work to perform. Returns the wire-form result directly.
|
||||
work: Box<dyn FnOnce() -> IoResult + Send>,
|
||||
}
|
||||
|
||||
/// Completion message from either IO thread back to the scheduler.
|
||||
pub enum Completion {
|
||||
/// A `block_on_io` closure has finished (Ok = return value, Err = panic
|
||||
/// payload).
|
||||
Blocking { pid: Pid, epoch: u32, result: IoResult },
|
||||
/// An fd registered via `wait_readable`/`wait_writable` is ready. The
|
||||
/// scheduler looks up the parked pid in `waiters`, unparks it, and
|
||||
/// removes the entry. `pid` isn't in this variant because the epoll
|
||||
/// thread doesn't have access to the `waiters` map; the scheduler
|
||||
/// thread owns that.
|
||||
FdReady { fd: RawFd, events: u32 },
|
||||
}
|
||||
/// The parked-waiter map, shared between scheduler-side registration and
|
||||
/// the epoll thread's readiness consumption. See the module docs on why
|
||||
/// this single lock is the ADD/DEL serialization.
|
||||
type Waiters = Arc<Mutex<HashMap<RawFd, (Pid, u32)>>>;
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// IoThread — created per `run()`, owned by `SchedulerState`.
|
||||
// IoThread — created per `run()`, owned by `RuntimeInner::io`.
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
pub struct IoThread {
|
||||
// ----- Channels & queues -----
|
||||
|
||||
/// Submission queue into the blocking-work pool.
|
||||
tx: mpsc::Sender<Request>,
|
||||
/// Shared completion queue, fed by both the pool and the epoll thread.
|
||||
completions: Arc<Mutex<VecDeque<Completion>>>,
|
||||
/// Pipe the scheduler polls in its idle path. Both IO threads write to
|
||||
/// `wake_write` after pushing a completion.
|
||||
wake_read: RawFd,
|
||||
wake_write: RawFd,
|
||||
/// One parked actor per registered fd. Populated by `epoll_register`,
|
||||
/// consumed by the epoll thread on readiness or `cancel_waiter` on an
|
||||
/// unwound wait.
|
||||
waiters: Waiters,
|
||||
|
||||
// ----- Epoll machinery -----
|
||||
|
||||
@@ -133,39 +145,25 @@ pub struct IoThread {
|
||||
/// shutdown.
|
||||
shutdown_read: RawFd,
|
||||
shutdown_write: RawFd,
|
||||
/// One parked actor per registered fd. Populated by `wait_readable` /
|
||||
/// `wait_writable` and drained by the scheduler when a `FdReady`
|
||||
/// completion is processed.
|
||||
pub waiters: HashMap<RawFd, (Pid, u32)>,
|
||||
|
||||
// ----- Threads -----
|
||||
|
||||
pool_thread: Option<OsJoinHandle<()>>,
|
||||
epoll_thread: Option<OsJoinHandle<()>>,
|
||||
|
||||
/// Number of `block_on_io` requests in-flight. Used by the scheduler's
|
||||
/// idle path to decide whether to wait on the pipe or exit. Fd waits
|
||||
/// are not counted here; they're counted by `waiters.len()`.
|
||||
pub outstanding: u32,
|
||||
}
|
||||
|
||||
impl IoThread {
|
||||
pub fn start() -> io::Result<Self> {
|
||||
// Scheduler-facing wake pipe.
|
||||
let (wake_read, wake_write) = make_pipe()?;
|
||||
// Pool submission channel + shared completion queue.
|
||||
/// Start the pool and epoll threads. `rt` is the producers' route back
|
||||
/// into the runtime (slot table + unpark protocol); a `Weak` so the
|
||||
/// `RuntimeInner → IoThread → RuntimeInner` cycle never forms.
|
||||
pub(crate) fn start(rt: Weak<RuntimeInner>) -> io::Result<Self> {
|
||||
// Pool submission channel.
|
||||
let (tx, rx) = mpsc::channel::<Request>();
|
||||
let completions: Arc<Mutex<VecDeque<Completion>>> =
|
||||
Arc::new(Mutex::new(VecDeque::new()));
|
||||
let waiters: Waiters = Arc::new(Mutex::new(HashMap::new()));
|
||||
|
||||
// Epoll machinery.
|
||||
let epollfd = unsafe { libc::epoll_create1(libc::EPOLL_CLOEXEC) };
|
||||
if epollfd < 0 {
|
||||
// Best-effort fd cleanup before bailing.
|
||||
unsafe {
|
||||
libc::close(wake_read);
|
||||
libc::close(wake_write);
|
||||
}
|
||||
return Err(io::Error::last_os_error());
|
||||
}
|
||||
|
||||
@@ -174,8 +172,6 @@ impl IoThread {
|
||||
Err(e) => {
|
||||
unsafe {
|
||||
libc::close(epollfd);
|
||||
libc::close(wake_read);
|
||||
libc::close(wake_write);
|
||||
}
|
||||
return Err(e);
|
||||
}
|
||||
@@ -202,42 +198,37 @@ impl IoThread {
|
||||
libc::close(epollfd);
|
||||
libc::close(shutdown_read);
|
||||
libc::close(shutdown_write);
|
||||
libc::close(wake_read);
|
||||
libc::close(wake_write);
|
||||
}
|
||||
return Err(e);
|
||||
}
|
||||
|
||||
// Spawn pool thread.
|
||||
let pool_comps = completions.clone();
|
||||
let pool_rt = rt.clone();
|
||||
let pool_thread = std::thread::Builder::new()
|
||||
.name("smarm-io-pool".into())
|
||||
.spawn(move || pool_loop(rx, pool_comps, wake_write))?;
|
||||
.spawn(move || pool_loop(rx, pool_rt))?;
|
||||
|
||||
// Spawn epoll thread.
|
||||
let epoll_comps = completions.clone();
|
||||
let epoll_waiters = waiters.clone();
|
||||
let epoll_thread = std::thread::Builder::new()
|
||||
.name("smarm-io-epoll".into())
|
||||
.spawn(move || epoll_loop(epollfd, epoll_comps, wake_write))?;
|
||||
.spawn(move || epoll_loop(epollfd, epoll_waiters, rt))?;
|
||||
|
||||
Ok(Self {
|
||||
tx,
|
||||
completions,
|
||||
wake_read,
|
||||
wake_write,
|
||||
waiters,
|
||||
epollfd,
|
||||
shutdown_read,
|
||||
shutdown_write,
|
||||
waiters: HashMap::new(),
|
||||
pool_thread: Some(pool_thread),
|
||||
epoll_thread: Some(epoll_thread),
|
||||
outstanding: 0,
|
||||
})
|
||||
}
|
||||
|
||||
/// Hand a request to the pool. Increments `outstanding`.
|
||||
/// Hand a request to the pool. The caller (scheduler.rs) increments
|
||||
/// `io_outstanding` BEFORE calling — the pool decrements on completion,
|
||||
/// and an increment that trailed the completion would underflow.
|
||||
pub fn submit(&mut self, pid: Pid, epoch: u32, work: Box<dyn FnOnce() -> IoResult + Send>) {
|
||||
self.outstanding += 1;
|
||||
// Send can only fail if the pool has hung up, which only happens
|
||||
// on shutdown. submit during shutdown is a bug.
|
||||
if self.tx.send(Request { pid, epoch, work }).is_err() {
|
||||
@@ -245,39 +236,13 @@ impl IoThread {
|
||||
}
|
||||
}
|
||||
|
||||
/// Drain every available completion. Caller (the scheduler) routes the
|
||||
/// results and updates `outstanding` / `waiters` accordingly.
|
||||
pub fn drain_completions(&mut self) -> Vec<Completion> {
|
||||
let mut q = match self.completions.lock() {
|
||||
Ok(g) => g,
|
||||
Err(e) => panic!("smarm: io completions lock poisoned (core corrupt): {e}"),
|
||||
};
|
||||
let mut out = Vec::with_capacity(q.len());
|
||||
while let Some(c) = q.pop_front() {
|
||||
out.push(c);
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
pub fn wake_fd(&self) -> RawFd {
|
||||
self.wake_read
|
||||
}
|
||||
|
||||
/// Write the wake pipe directly: rouse every scheduler thread blocked in
|
||||
/// its idle `poll_wake`. Used by the terminal (AllDone) path — an idle
|
||||
/// sibling may be blocked on a snapshot that nothing will ever refresh
|
||||
/// (an orphaned timer deadline, or `io_outstanding` from a waiter that
|
||||
/// was stop-cancelled and so never produces a completion).
|
||||
pub fn wake(&self) {
|
||||
wake_scheduler(self.wake_write);
|
||||
}
|
||||
|
||||
/// Register interest in `fd` becoming readable/writable; record `pid`
|
||||
/// as the parked waiter. The epoll thread will push a `FdReady`
|
||||
/// completion when the kernel signals.
|
||||
/// as the parked waiter. The epoll thread unparks it on readiness.
|
||||
/// The caller increments `io_fd_waiters` BEFORE calling (mirror of
|
||||
/// `submit`'s contract) and decrements it again if this errors.
|
||||
///
|
||||
/// EPOLLONESHOT: one wakeup per registration. The scheduler must
|
||||
/// `epoll_del` on completion to free the slot for re-registration.
|
||||
/// EPOLLONESHOT: one wakeup per registration; the epoll thread DELs on
|
||||
/// readiness, `cancel_waiter` DELs on an unwound wait.
|
||||
pub fn epoll_register(
|
||||
&mut self,
|
||||
fd: RawFd,
|
||||
@@ -286,20 +251,24 @@ impl IoThread {
|
||||
readable: bool,
|
||||
writable: bool,
|
||||
) -> io::Result<()> {
|
||||
let mut waiters = match self.waiters.lock() {
|
||||
Ok(g) => g,
|
||||
Err(e) => panic!("smarm: io waiters lock poisoned (core corrupt): {e}"),
|
||||
};
|
||||
// Two actors waiting on the same fd would be a misuse: the kernel
|
||||
// delivers exactly one EPOLLONESHOT wakeup, so the second waiter
|
||||
// would hang. Reject up front.
|
||||
if self.waiters.contains_key(&fd) {
|
||||
if waiters.contains_key(&fd) {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::AlreadyExists,
|
||||
"fd already has a parked waiter",
|
||||
));
|
||||
}
|
||||
|
||||
// Belt-and-braces: the unwind guard in `wait_fd` is responsible for
|
||||
// cleaning up a stopped waiter's registration, but a bare DEL is
|
||||
// harmless if the fd isn't registered (ENOENT) and removes any leak
|
||||
// a path we haven't thought of might leave behind.
|
||||
// Belt-and-braces: `cancel_waiter` is responsible for cleaning up a
|
||||
// stopped waiter's registration, but a bare DEL is harmless if the
|
||||
// fd isn't registered (ENOENT) and removes any leak a path we
|
||||
// haven't thought of might leave behind.
|
||||
unsafe {
|
||||
libc::epoll_ctl(self.epollfd, libc::EPOLL_CTL_DEL, fd, std::ptr::null_mut());
|
||||
}
|
||||
@@ -321,20 +290,30 @@ impl IoThread {
|
||||
if r < 0 {
|
||||
return Err(io::Error::last_os_error());
|
||||
}
|
||||
self.waiters.insert(fd, (pid, epoch));
|
||||
waiters.insert(fd, (pid, epoch));
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Remove `fd` from the epollfd. Called by the scheduler after a
|
||||
/// `FdReady` completion, so the next `wait_readable(fd)` can ADD again.
|
||||
///
|
||||
/// Does NOT touch `waiters` — that's the scheduler's bookkeeping; this
|
||||
/// is purely the kernel-side cleanup.
|
||||
pub fn epoll_deregister(&mut self, fd: RawFd) {
|
||||
/// Remove `fd`'s waiter iff it is still `(pid, epoch)`, DELing the fd
|
||||
/// from the epollfd in the same critical section. Returns whether the
|
||||
/// entry was removed (the caller then decrements `io_fd_waiters`).
|
||||
/// `false` means the epoll thread consumed the registration first —
|
||||
/// the fd may already carry someone else's fresh ADD; hands off.
|
||||
pub fn cancel_waiter(&mut self, fd: RawFd, pid: Pid, epoch: u32) -> bool {
|
||||
let mut waiters = match self.waiters.lock() {
|
||||
Ok(g) => g,
|
||||
Err(e) => panic!("smarm: io waiters lock poisoned (core corrupt): {e}"),
|
||||
};
|
||||
if waiters.get(&fd) == Some(&(pid, epoch)) {
|
||||
waiters.remove(&fd);
|
||||
// EPOLL_CTL_DEL of an already-removed fd returns ENOENT; ignore.
|
||||
unsafe {
|
||||
libc::epoll_ctl(self.epollfd, libc::EPOLL_CTL_DEL, fd, std::ptr::null_mut());
|
||||
}
|
||||
true
|
||||
} else {
|
||||
false
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -354,7 +333,10 @@ impl Drop for IoThread {
|
||||
let real_tx = std::mem::replace(&mut self.tx, dead_tx);
|
||||
drop(real_tx);
|
||||
|
||||
// 3. Join both threads.
|
||||
// 3. Join both threads. Safe even while the caller holds the
|
||||
// runtime's `io` mutex: neither thread ever takes it (they reach
|
||||
// the runtime through a Weak they upgrade per completion, and
|
||||
// the epoll thread's only lock is `waiters`).
|
||||
if let Some(h) = self.epoll_thread.take() {
|
||||
let _ = h.join();
|
||||
}
|
||||
@@ -367,8 +349,6 @@ impl Drop for IoThread {
|
||||
libc::close(self.epollfd);
|
||||
libc::close(self.shutdown_read);
|
||||
libc::close(self.shutdown_write);
|
||||
libc::close(self.wake_read);
|
||||
libc::close(self.wake_write);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -379,36 +359,38 @@ impl Drop for IoThread {
|
||||
const SHUTDOWN_EPOLL_TOKEN: u64 = u64::MAX;
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Pool loop
|
||||
// Pool loop (producer: Blocking completions)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
fn pool_loop(
|
||||
rx: mpsc::Receiver<Request>,
|
||||
completions: Arc<Mutex<VecDeque<Completion>>>,
|
||||
wake_write: RawFd,
|
||||
) {
|
||||
fn pool_loop(rx: mpsc::Receiver<Request>, rt: Weak<RuntimeInner>) {
|
||||
while let Ok(Request { pid, epoch, work }) = rx.recv() {
|
||||
let result: IoResult = match panic::catch_unwind(panic::AssertUnwindSafe(work)) {
|
||||
Ok(r) => r,
|
||||
Err(payload) => Err(payload),
|
||||
};
|
||||
match completions.lock() {
|
||||
Ok(mut g) => g.push_back(Completion::Blocking { pid, epoch, result }),
|
||||
Err(e) => panic!("smarm: io completions lock poisoned (core corrupt): {e}"),
|
||||
let Some(inner) = rt.upgrade() else { return };
|
||||
// Stash the result under the cold lock (generation-checked: an
|
||||
// actor stopped with the op in flight discards it), decrement the
|
||||
// in-flight count, then wake through the epoch-matched unpark. The
|
||||
// unpark's enqueue tail wakes a parked scheduler; the actor stays
|
||||
// `live` until it resumes and finalizes, so the decrement's
|
||||
// ordering against the termination verdict is not load-bearing.
|
||||
if let Some(slot) = inner.slot_at(pid) {
|
||||
let mut cold = slot.cold.lock();
|
||||
if slot.generation() == pid.generation() {
|
||||
cold.pending_io_result = Some(result);
|
||||
}
|
||||
wake_scheduler(wake_write);
|
||||
}
|
||||
inner.io_outstanding.fetch_sub(1, Ordering::AcqRel);
|
||||
inner.unpark_at(pid, epoch);
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Epoll loop
|
||||
// Epoll loop (producer: FdReady completions)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
fn epoll_loop(
|
||||
epollfd: RawFd,
|
||||
completions: Arc<Mutex<VecDeque<Completion>>>,
|
||||
wake_write: RawFd,
|
||||
) {
|
||||
fn epoll_loop(epollfd: RawFd, waiters: Waiters, rt: Weak<RuntimeInner>) {
|
||||
// Buffer for epoll_wait. 64 is plenty for our scale; if a real load
|
||||
// appears that needs more, this is a one-line change.
|
||||
const MAX_EVENTS: usize = 64;
|
||||
@@ -436,29 +418,41 @@ fn epoll_loop(
|
||||
}
|
||||
|
||||
let mut shutdown_requested = false;
|
||||
let mut pushed_any = false;
|
||||
{
|
||||
let mut q = match completions.lock() {
|
||||
Ok(g) => g,
|
||||
Err(e) => panic!("smarm: io completions lock poisoned (core corrupt): {e}"),
|
||||
};
|
||||
for ev in events.iter().take(n as usize) {
|
||||
if ev.u64 == SHUTDOWN_EPOLL_TOKEN {
|
||||
shutdown_requested = true;
|
||||
continue;
|
||||
}
|
||||
let fd = ev.u64 as RawFd;
|
||||
let evs = ev.events;
|
||||
q.push_back(Completion::FdReady {
|
||||
// Consume the registration: remove + DEL under the waiters
|
||||
// lock (the ADD/DEL serialization — see module docs). A
|
||||
// vanished entry means `cancel_waiter` beat us: the wake is
|
||||
// already moot.
|
||||
let entry = {
|
||||
let mut w = match waiters.lock() {
|
||||
Ok(g) => g,
|
||||
Err(e) => {
|
||||
panic!("smarm: io waiters lock poisoned (core corrupt): {e}")
|
||||
}
|
||||
};
|
||||
let entry = w.remove(&fd);
|
||||
if entry.is_some() {
|
||||
unsafe {
|
||||
libc::epoll_ctl(
|
||||
epollfd,
|
||||
libc::EPOLL_CTL_DEL,
|
||||
fd,
|
||||
events: evs,
|
||||
});
|
||||
pushed_any = true;
|
||||
std::ptr::null_mut(),
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
if pushed_any {
|
||||
wake_scheduler(wake_write);
|
||||
entry
|
||||
};
|
||||
if let Some((pid, epoch)) = entry {
|
||||
let Some(inner) = rt.upgrade() else { return };
|
||||
inner.io_fd_waiters.fetch_sub(1, Ordering::AcqRel);
|
||||
inner.unpark_at(pid, epoch);
|
||||
}
|
||||
}
|
||||
if shutdown_requested {
|
||||
return;
|
||||
@@ -466,27 +460,8 @@ fn epoll_loop(
|
||||
}
|
||||
}
|
||||
|
||||
/// Write one byte to the scheduler's wake pipe. Retries on EINTR; ignores
|
||||
/// EAGAIN (pipe full means there's already an outstanding wake we haven't
|
||||
/// consumed yet, which is sufficient).
|
||||
fn wake_scheduler(wake_write: RawFd) {
|
||||
let buf: [u8; 1] = [0];
|
||||
unsafe {
|
||||
loop {
|
||||
let n = libc::write(wake_write, buf.as_ptr() as *const _, 1);
|
||||
if n < 0 {
|
||||
let e = *libc::__errno_location();
|
||||
if e == libc::EINTR {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Pipe helpers (unchanged from v0.2)
|
||||
// Pipe helper
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
fn make_pipe() -> io::Result<(RawFd, RawFd)> {
|
||||
@@ -497,50 +472,3 @@ fn make_pipe() -> io::Result<(RawFd, RawFd)> {
|
||||
}
|
||||
Ok((fds[0], fds[1]))
|
||||
}
|
||||
|
||||
/// Drain pending bytes from the wake pipe. Nonblocking (pipe is O_NONBLOCK).
|
||||
///
|
||||
/// DISCIPLINE: called only by the phase-1 drain-lock winner, immediately
|
||||
/// before `drain_completions`. Bytes are the notification channel for
|
||||
/// completions; consuming one anywhere else can strand the completion it
|
||||
/// announces (see the lost-wakeup note at the call site in `schedule_loop`).
|
||||
pub fn drain_wake_pipe(fd: RawFd) {
|
||||
let mut buf = [0u8; 64];
|
||||
loop {
|
||||
let n = unsafe { libc::read(fd, buf.as_mut_ptr() as *mut _, buf.len()) };
|
||||
if n <= 0 {
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Block on `fd` for up to `timeout`, returning when either there's data
|
||||
/// to read or the timeout elapses. `None` for `timeout` means wait forever.
|
||||
pub fn poll_wake(fd: RawFd, timeout: Option<std::time::Duration>) {
|
||||
let timeout_ms: libc::c_int = match timeout {
|
||||
None => -1,
|
||||
Some(d) => {
|
||||
let ms = d.as_millis();
|
||||
if ms > i32::MAX as u128 {
|
||||
i32::MAX
|
||||
} else {
|
||||
ms as i32
|
||||
}
|
||||
}
|
||||
};
|
||||
let mut pfd = libc::pollfd {
|
||||
fd,
|
||||
events: libc::POLLIN,
|
||||
revents: 0,
|
||||
};
|
||||
loop {
|
||||
let r = unsafe { libc::poll(&mut pfd as *mut _, 1, timeout_ms) };
|
||||
if r < 0 {
|
||||
let e = unsafe { *libc::__errno_location() };
|
||||
if e == libc::EINTR {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
+6
-3
@@ -12,6 +12,7 @@
|
||||
//! See `LOOM.md` for the design intent and the deferred-for-later list.
|
||||
|
||||
pub mod stack;
|
||||
pub(crate) mod signal;
|
||||
pub mod context;
|
||||
pub mod preempt;
|
||||
pub mod pid;
|
||||
@@ -32,6 +33,7 @@ pub mod introspect;
|
||||
#[cfg(feature = "observer")]
|
||||
pub mod observer;
|
||||
pub mod runtime;
|
||||
pub(crate) mod park;
|
||||
pub(crate) mod raw_mutex;
|
||||
pub(crate) mod slot_state;
|
||||
pub(crate) mod sync_shim;
|
||||
@@ -63,7 +65,7 @@ pub use gen_statem::{
|
||||
CallError as GenStatemCallError, Cx, Machine, Reply, Resolution, SendError as GenStatemSendError,
|
||||
GenStatemRef,
|
||||
};
|
||||
pub use introspect::{
|
||||
pub use introspect::{StackInfo,
|
||||
actor_info, snapshot, tree, tree_from, ActorInfo, ActorState, RuntimeSnapshot, RuntimeTree,
|
||||
TreeNode, SNAPSHOT_FORMAT_VERSION,
|
||||
};
|
||||
@@ -82,8 +84,9 @@ pub use runtime::{init, Config, Runtime};
|
||||
pub use scheduler::{
|
||||
block_on_io, cancel_timer, request_stop, run, self_pid, send_after, send_after_named,
|
||||
send_after_named_wall, send_after_wall, sleep, sleep_wall,
|
||||
spawn, spawn_addr, spawn_under, wait_readable, wait_readable_timeout, wait_writable,
|
||||
wait_writable_timeout, yield_now, FdArm, JoinError, JoinHandle,
|
||||
spawn, spawn_addr, spawn_addr_with, spawn_under, spawn_under_with, spawn_with,
|
||||
wait_readable, wait_readable_timeout, wait_writable,
|
||||
wait_writable_timeout, yield_now, FdArm, JoinError, JoinHandle, SpawnOpts,
|
||||
};
|
||||
pub use supervisor::{ChildSpec, OneForOne, Restart, Signal, Strategy};
|
||||
pub use timer::TimerId;
|
||||
|
||||
+994
@@ -0,0 +1,994 @@
|
||||
//! Scheduler park/wake coordination layer (RFC 018).
|
||||
//!
|
||||
//! Schedulers never touch an fd to sleep: they park on a per-thread
|
||||
//! [`Parker`] and are woken through an idle-mask protocol the runtime owns
|
||||
//! outright. IO backends (epoll today, io_uring later) are *producers*
|
||||
//! behind a two-call contract — make actors runnable, then wake — which is
|
||||
//! what makes backend selection tractable (RFC 018 §step-back).
|
||||
//!
|
||||
//! Three pieces, all in [`Coordinator`]:
|
||||
//!
|
||||
//! - **Parker** (one per scheduler): permit semantics, `std::thread::park`
|
||||
//! shaped — an unpark delivered before the park sets a permit; the next
|
||||
//! park consumes it and returns immediately. This single property closes
|
||||
//! the check-then-park race. Linux: `futex(2)` `FUTEX_WAIT`/`FUTEX_WAKE`
|
||||
//! (private) with a nanosecond-precision relative `timespec` — the
|
||||
//! `as_millis` truncation defect of the retired wake pipe is
|
||||
//! unrepresentable here. Loom / non-Linux: `Mutex<bool>` + `Condvar`
|
||||
//! (the loom models run against this build).
|
||||
//! - **Idle mask**: an `AtomicU64` bitmask of parked scheduler ids
|
||||
//! (construction asserts ≤ 64 schedulers). Park protocol: set own bit,
|
||||
//! run the caller's mandatory post-publish re-check, then wait. A
|
||||
//! producer that published work before observing our bit has left us
|
||||
//! work the re-check finds; one that observes the bit wakes us.
|
||||
//!
|
||||
//! The publish/re-check pair is a store-buffer (Dekker) shape. Two sound
|
||||
//! resolutions coexist here, chosen per call-site cost profile: the
|
||||
//! **fence handshake** on the hot producer path
|
||||
//! ([`Coordinator::wake_one_if_idle`]: publish work; `fence(SeqCst)`;
|
||||
//! one *Relaxed* mask load — pairing with the consumer's `fetch_or(bit)`;
|
||||
//! `fence(SeqCst)`; re-check inside [`Coordinator::park`]), so the
|
||||
//! pure-compute hot path (everyone busy, mask 0) never takes the shared
|
||||
//! mask line exclusive — the RFC's "one relaxed load" fast path, made
|
||||
//! sound; and the **same-location-RMW read** (`fetch_or(0)`, in
|
||||
//! [`Coordinator::wake_one`] / [`Coordinator::idle_mask`]) for the rare
|
||||
//! paths (chain rule — at most one per wake) where reading the latest
|
||||
//! mask by modification-order coherence is worth an RMW. The loom models
|
||||
//! drive the fence pattern end to end (a fence-less plain-load draft
|
||||
//! would — and did — fail model 1/2 with a lost wake, as it must).
|
||||
//! - **Timekeeper**: at most one parked scheduler holds the timer
|
||||
//! deadline (RFC 018 §timers) so a timer expiry wakes one scheduler,
|
||||
//! not a herd. The role is an atomic `(holder id, armed deadline)`
|
||||
//! pair; a timer insertion with an earlier deadline wakes the holder to
|
||||
//! re-peek. Arm / insert-check MUST be serialized by the caller (the
|
||||
//! timers mutex in the runtime) — the atomics exist so the *busy-path
|
||||
//! due-check* (one Relaxed load, [`Coordinator::armed_deadline_nanos`])
|
||||
//! and the wake stay lock-free. All races are biased over-wake: a
|
||||
//! spurious permit costs one failed pop; a missed wake would cost a
|
||||
//! stranded actor, and is unrepresentable under the serialization rule.
|
||||
//!
|
||||
//! Every wake here is *at most one* futex round-trip and wakes *exactly
|
||||
//! one* scheduler by construction (`wake_one` CASes a bit clear before
|
||||
//! unparking its owner) — there is no shared level-triggered anything
|
||||
//! left to herd on.
|
||||
//!
|
||||
//! Standalone until the runtime swap (RFC 018 commit 2): nothing outside
|
||||
//! tests constructs a [`Coordinator`] yet.
|
||||
|
||||
use crate::sync_shim::{fence, AtomicU64, Ordering};
|
||||
use std::time::Instant;
|
||||
|
||||
/// Sentinel for "no timekeeper" in the holder word.
|
||||
const NO_TIMEKEEPER: u64 = u64::MAX;
|
||||
/// Sentinel for "no armed deadline" in the deadline word. Also what the
|
||||
/// busy-path due-check compares against: `now_nanos < armed` is one branch.
|
||||
pub(crate) const NO_DEADLINE: u64 = u64::MAX;
|
||||
|
||||
/// Outcome of a [`Coordinator::park`] call.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub(crate) enum ParkResult {
|
||||
/// A permit was consumed (wake delivered before or during the park).
|
||||
Woken,
|
||||
/// The deadline passed with no wake. Only possible when a deadline was
|
||||
/// supplied (and never under loom, which has no time — see `park`).
|
||||
TimedOut,
|
||||
/// The post-publish re-check found work; the thread never blocked.
|
||||
/// A permit may still be pending (a racing `wake_one` picked us after
|
||||
/// the bit was set) — it will surface as one spurious `Woken` on a
|
||||
/// later park. Benign: over-wake by design.
|
||||
WorkFound,
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Parker — permit-semantics thread parking
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Linux, non-loom: futex on a state word.
|
||||
///
|
||||
/// States: EMPTY (no permit, nobody waiting), PARKED (a thread is, or is
|
||||
/// about to be, in `futex_wait`), NOTIFIED (permit pending). The classic
|
||||
/// std-parker protocol: `unpark` swaps to NOTIFIED and futex-wakes iff it
|
||||
/// displaced PARKED; `park` CASes EMPTY→PARKED, waits, and consumes
|
||||
/// NOTIFIED on every exit path.
|
||||
#[cfg(all(target_os = "linux", not(loom)))]
|
||||
mod parker {
|
||||
use std::sync::atomic::{AtomicU32, Ordering};
|
||||
use std::time::Instant;
|
||||
|
||||
const EMPTY: u32 = 0;
|
||||
const PARKED: u32 = 1;
|
||||
const NOTIFIED: u32 = 2;
|
||||
|
||||
pub(super) struct Parker {
|
||||
state: AtomicU32,
|
||||
}
|
||||
|
||||
impl Parker {
|
||||
pub(super) fn new() -> Self {
|
||||
Self { state: AtomicU32::new(EMPTY) }
|
||||
}
|
||||
|
||||
/// Returns `true` = woken (permit consumed), `false` = timed out.
|
||||
pub(super) fn park(&self, deadline: Option<Instant>) -> bool {
|
||||
// Fast path: consume a pending permit without blocking.
|
||||
if self
|
||||
.state
|
||||
.compare_exchange(EMPTY, PARKED, Ordering::AcqRel, Ordering::Acquire)
|
||||
.is_err()
|
||||
{
|
||||
// Only NOTIFIED can be here (one thread parks at a time).
|
||||
self.state.store(EMPTY, Ordering::Release);
|
||||
return true;
|
||||
}
|
||||
loop {
|
||||
let timeout = match deadline {
|
||||
None => None,
|
||||
Some(d) => {
|
||||
let now = Instant::now();
|
||||
if d <= now {
|
||||
// Deadline passed: cancel the park. The swap
|
||||
// races a concurrent unpark — if it delivered
|
||||
// NOTIFIED first, report Woken (never lose a
|
||||
// permit).
|
||||
return self.state.swap(EMPTY, Ordering::AcqRel) == NOTIFIED;
|
||||
}
|
||||
Some(d - now)
|
||||
}
|
||||
};
|
||||
futex_wait(&self.state, PARKED, timeout);
|
||||
if self
|
||||
.state
|
||||
.compare_exchange(NOTIFIED, EMPTY, Ordering::AcqRel, Ordering::Acquire)
|
||||
.is_ok()
|
||||
{
|
||||
return true;
|
||||
}
|
||||
// Spurious wake or timeout with state still PARKED: loop —
|
||||
// the deadline check at the top decides.
|
||||
}
|
||||
}
|
||||
|
||||
pub(super) fn unpark(&self) {
|
||||
if self.state.swap(NOTIFIED, Ordering::AcqRel) == PARKED {
|
||||
futex_wake(&self.state, 1);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// `FUTEX_WAIT` with a *relative* nanosecond timeout (`CLOCK_MONOTONIC`
|
||||
/// per futex(2) for relative waits). No millisecond conversion anywhere:
|
||||
/// the timespec carries the full sub-ms remainder (RFC 018 kills the
|
||||
/// `as_millis` truncation structurally).
|
||||
fn futex_wait(word: &AtomicU32, expected: u32, timeout: Option<std::time::Duration>) {
|
||||
let ts;
|
||||
let ts_ptr: *const libc::timespec = match timeout {
|
||||
Some(d) => {
|
||||
ts = libc::timespec {
|
||||
tv_sec: d.as_secs() as libc::time_t,
|
||||
tv_nsec: d.subsec_nanos() as libc::c_long,
|
||||
};
|
||||
&ts
|
||||
}
|
||||
None => std::ptr::null(),
|
||||
};
|
||||
// Errors (EAGAIN: word changed; ETIMEDOUT; EINTR) all mean "return
|
||||
// and let the caller's state machine decide" — deliberately ignored.
|
||||
unsafe {
|
||||
libc::syscall(
|
||||
libc::SYS_futex,
|
||||
word.as_ptr(),
|
||||
libc::FUTEX_WAIT | libc::FUTEX_PRIVATE_FLAG,
|
||||
expected,
|
||||
ts_ptr,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
fn futex_wake(word: &AtomicU32, n: u32) {
|
||||
unsafe {
|
||||
libc::syscall(
|
||||
libc::SYS_futex,
|
||||
word.as_ptr(),
|
||||
libc::FUTEX_WAKE | libc::FUTEX_PRIVATE_FLAG,
|
||||
n,
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Loom / non-Linux: `Mutex<bool>` permit + `Condvar` — loom's own model
|
||||
/// of a parker, and the portable fallback. Under loom the deadline is
|
||||
/// ignored (loom has no time); the models exercise wake paths only.
|
||||
#[cfg(any(loom, not(target_os = "linux")))]
|
||||
mod parker {
|
||||
use crate::sync_shim::{Condvar, Mutex};
|
||||
use std::time::Instant;
|
||||
|
||||
pub(super) struct Parker {
|
||||
permit: Mutex<bool>,
|
||||
cv: Condvar,
|
||||
}
|
||||
|
||||
impl Parker {
|
||||
pub(super) fn new() -> Self {
|
||||
Self { permit: Mutex::new(false), cv: Condvar::new() }
|
||||
}
|
||||
|
||||
/// Returns `true` = woken (permit consumed), `false` = timed out.
|
||||
pub(super) fn park(&self, deadline: Option<Instant>) -> bool {
|
||||
let mut permit = match self.permit.lock() {
|
||||
Ok(g) => g,
|
||||
Err(_) => panic!("smarm: parker permit lock poisoned (core corrupt)"),
|
||||
};
|
||||
loop {
|
||||
if *permit {
|
||||
*permit = false;
|
||||
return true;
|
||||
}
|
||||
#[cfg(loom)]
|
||||
{
|
||||
// Loom has no clock: block until a wake. Models must
|
||||
// deliver one (a park nobody wakes is a real deadlock
|
||||
// and loom reports it as such).
|
||||
let _ = deadline;
|
||||
permit = match self.cv.wait(permit) {
|
||||
Ok(g) => g,
|
||||
Err(_) => panic!("smarm: parker cv poisoned (core corrupt)"),
|
||||
};
|
||||
}
|
||||
#[cfg(not(loom))]
|
||||
{
|
||||
match deadline {
|
||||
None => {
|
||||
permit = match self.cv.wait(permit) {
|
||||
Ok(g) => g,
|
||||
Err(_) => panic!("smarm: parker cv poisoned (core corrupt)"),
|
||||
};
|
||||
}
|
||||
Some(d) => {
|
||||
let now = Instant::now();
|
||||
if d <= now {
|
||||
return false;
|
||||
}
|
||||
permit = match self.cv.wait_timeout(permit, d - now) {
|
||||
Ok((g, _)) => g,
|
||||
Err(_) => panic!("smarm: parker cv poisoned (core corrupt)"),
|
||||
};
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub(super) fn unpark(&self) {
|
||||
let mut permit = match self.permit.lock() {
|
||||
Ok(g) => g,
|
||||
Err(_) => panic!("smarm: parker permit lock poisoned (core corrupt)"),
|
||||
};
|
||||
*permit = true;
|
||||
self.cv.notify_one();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
use parker::Parker;
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Coordinator — idle mask + wake protocol + timekeeper
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
pub(crate) struct Coordinator {
|
||||
parkers: Box<[Parker]>,
|
||||
/// Bit `i` set = scheduler `i` is parked or committed to parking (set
|
||||
/// before the re-check; cleared by `wake_one`'s CAS or by the parker
|
||||
/// itself on return). AcqRel same-location-RMW handshake — see module
|
||||
/// docs (no SeqCst needed: every producer-side read is an RMW).
|
||||
idle: AtomicU64,
|
||||
/// Timekeeper holder id, or `NO_TIMEKEEPER`. Written under the
|
||||
/// caller's timer serialization (arm/disarm/insert-check); read
|
||||
/// lock-free by the insert wake path.
|
||||
tk_holder: AtomicU64,
|
||||
/// Armed deadline as nanos since `origin`, or `NO_DEADLINE`. Written
|
||||
/// only by the timekeeper arm/disarm protocol.
|
||||
tk_armed: AtomicU64,
|
||||
/// Earliest KNOWN timer deadline (nanos since `origin`), or
|
||||
/// `NO_DEADLINE` — independent of whether any scheduler is parked,
|
||||
/// which is what the timekeeper's `tk_armed` cannot give: under
|
||||
/// saturation nobody parks and nobody arms, yet due timers must still
|
||||
/// fire (ratified design point (a)). Maintained under the caller's
|
||||
/// timers mutex (`note_deadline` on insert, `refresh_deadline` after a
|
||||
/// pop/peek); read lock-free by the busy-path due-check.
|
||||
next_deadline: AtomicU64,
|
||||
/// Time origin for the nanos encoding.
|
||||
origin: Instant,
|
||||
}
|
||||
|
||||
impl Coordinator {
|
||||
pub(crate) fn new(schedulers: usize) -> Self {
|
||||
assert!(
|
||||
(1..=64).contains(&schedulers),
|
||||
"smarm: scheduler count must be 1..=64 (idle mask is one u64); got {schedulers}"
|
||||
);
|
||||
Self {
|
||||
parkers: (0..schedulers).map(|_| Parker::new()).collect(),
|
||||
idle: AtomicU64::new(0),
|
||||
tk_holder: AtomicU64::new(NO_TIMEKEEPER),
|
||||
tk_armed: AtomicU64::new(NO_DEADLINE),
|
||||
next_deadline: AtomicU64::new(NO_DEADLINE),
|
||||
origin: Instant::now(),
|
||||
}
|
||||
}
|
||||
|
||||
/// Encode a deadline for the armed snapshot / busy-path compare.
|
||||
/// Saturating: a deadline at-or-before `origin` encodes as 0 (always
|
||||
/// due), one beyond ~584 years as `NO_DEADLINE - 1`.
|
||||
pub(crate) fn deadline_nanos(&self, deadline: Instant) -> u64 {
|
||||
let nanos = deadline
|
||||
.checked_duration_since(self.origin)
|
||||
.map(|d| d.as_nanos())
|
||||
.unwrap_or(0);
|
||||
if nanos >= NO_DEADLINE as u128 {
|
||||
NO_DEADLINE - 1
|
||||
} else {
|
||||
nanos as u64
|
||||
}
|
||||
}
|
||||
|
||||
/// Park scheduler `id` until a wake, the deadline, or a positive
|
||||
/// re-check. Protocol: (1) publish own idle bit (SeqCst), (2) run
|
||||
/// `recheck` — it MUST re-read the work source with an ordering that
|
||||
/// pairs with the producer's publish (SeqCst load, or take the mutex
|
||||
/// the producer publishes under); a `true` aborts the park, (3) block.
|
||||
pub(crate) fn park(
|
||||
&self,
|
||||
id: usize,
|
||||
deadline: Option<Instant>,
|
||||
recheck: impl FnOnce() -> bool,
|
||||
) -> ParkResult {
|
||||
debug_assert!(id < self.parkers.len(), "park: scheduler id out of range");
|
||||
let bit = 1u64 << id;
|
||||
// (1) publish. AcqRel RMW: the acquire half is the handshake — if
|
||||
// this lands after a producer's mask RMW in modification order, we
|
||||
// read-from it and the producer's earlier work publication is
|
||||
// visible to the re-check below (see module docs).
|
||||
let prev = self.idle.fetch_or(bit, Ordering::AcqRel);
|
||||
debug_assert_eq!(prev & bit, 0, "park: idle bit already set for this id");
|
||||
// Fence half of the producer handshake (see `wake_one_if_idle`):
|
||||
// orders our bit-publish before the re-check's loads, so it pairs
|
||||
// with the producer's publish→fence→mask-load — at least one side
|
||||
// must see the other's store, whichever queue backend is in play.
|
||||
fence(Ordering::SeqCst);
|
||||
// (2) the mandatory post-publish re-check.
|
||||
if recheck() {
|
||||
self.idle.fetch_and(!bit, Ordering::AcqRel);
|
||||
return ParkResult::WorkFound;
|
||||
}
|
||||
// (3) block. The permit closes the window between the re-check and
|
||||
// the futex wait: a wake_one that picked us in that window has
|
||||
// already CASed our bit clear and set the permit.
|
||||
let woken = self.parkers[id].park(deadline);
|
||||
// Clear own bit — a no-op when a waker already CASed it clear.
|
||||
self.idle.fetch_and(!bit, Ordering::AcqRel);
|
||||
if woken {
|
||||
ParkResult::Woken
|
||||
} else {
|
||||
ParkResult::TimedOut
|
||||
}
|
||||
}
|
||||
|
||||
/// Wake exactly one parked scheduler, if any: pick the highest set idle
|
||||
/// bit (LIFO-ish — warmest cache), CAS it clear, deliver a permit.
|
||||
/// Empty mask = no-op (everyone is awake and will find work by
|
||||
/// popping). Returns whether a scheduler was woken.
|
||||
pub(crate) fn wake_one(&self) -> bool {
|
||||
// RMW read, not a load: reads the latest mask by modification-order
|
||||
// coherence, closing the Dekker race with a parking consumer (see
|
||||
// module docs). The release side of the RMW is what a later-parking
|
||||
// consumer's fetch_or acquires to make its re-check sound.
|
||||
let mut mask = self.idle.fetch_or(0, Ordering::AcqRel);
|
||||
loop {
|
||||
if mask == 0 {
|
||||
return false;
|
||||
}
|
||||
let id = 63 - mask.leading_zeros() as usize; // highest set bit
|
||||
let bit = 1u64 << id;
|
||||
// The CAS is the exactly-one guarantee: whoever clears the bit
|
||||
// owns the wake; a racing wake_one retries on the observed value
|
||||
// (coherence: a failed CAS can never read older than `mask`).
|
||||
match self.idle.compare_exchange(
|
||||
mask,
|
||||
mask & !bit,
|
||||
Ordering::AcqRel,
|
||||
Ordering::Acquire,
|
||||
) {
|
||||
Ok(_) => {
|
||||
self.parkers[id].unpark();
|
||||
return true;
|
||||
}
|
||||
Err(m) => mask = m,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// The producer-side wake tail (`enqueue`'s fast path, RFC 018 "enqueue
|
||||
/// wakes"). The caller has just published work (queue push); we fence,
|
||||
/// then read the mask with ONE Relaxed load — 0 means every scheduler
|
||||
/// is awake and the pure-compute hot path pays no RMW on the shared
|
||||
/// mask line. Soundness is the fence handshake (module docs): our
|
||||
/// fence orders the caller's push before the mask load; the consumer's
|
||||
/// fence (in `park`) orders its bit-publish before its re-check — at
|
||||
/// least one side must observe the other's store, so a consumer we
|
||||
/// miss here is a consumer whose re-check finds the caller's work.
|
||||
pub(crate) fn wake_one_if_idle(&self) -> bool {
|
||||
fence(Ordering::SeqCst);
|
||||
if self.idle.load(Ordering::Relaxed) == 0 {
|
||||
return false;
|
||||
}
|
||||
self.wake_one()
|
||||
}
|
||||
|
||||
/// Terminal wake (replaces the AllDone wake-pipe byte): clear the mask
|
||||
/// and deliver a permit to *every* parker, parked or not. A permit set
|
||||
/// on a busy scheduler costs one spurious park return — nothing at the
|
||||
/// terminal boundary. Idempotent.
|
||||
pub(crate) fn wake_all(&self) {
|
||||
self.idle.store(0, Ordering::Release);
|
||||
for p in self.parkers.iter() {
|
||||
p.unpark();
|
||||
}
|
||||
}
|
||||
|
||||
/// Latest idle mask (RMW read — same handshake as `wake_one`). A
|
||||
/// test-only observer: production expresses the chain rule through
|
||||
/// `wake_one_if_idle` (fence + Relaxed load), not a mask read.
|
||||
#[cfg(test)]
|
||||
pub(crate) fn idle_mask(&self) -> u64 {
|
||||
self.idle.fetch_or(0, Ordering::AcqRel)
|
||||
}
|
||||
|
||||
// ----- timekeeper -----
|
||||
|
||||
/// Try to take the timekeeper role for scheduler `id` with `deadline`.
|
||||
/// MUST be called under the caller's timer serialization (the timers
|
||||
/// mutex), with `deadline` the heap minimum peeked under that same
|
||||
/// hold — this is what makes the insert-check race-free. Returns
|
||||
/// whether the role was taken (false = someone else holds it; park
|
||||
/// with no deadline).
|
||||
pub(crate) fn try_arm_timer(&self, id: usize, deadline: Instant) -> bool {
|
||||
debug_assert!(id < self.parkers.len(), "try_arm_timer: id out of range");
|
||||
if self
|
||||
.tk_holder
|
||||
.compare_exchange(NO_TIMEKEEPER, id as u64, Ordering::SeqCst, Ordering::SeqCst)
|
||||
.is_err()
|
||||
{
|
||||
return false;
|
||||
}
|
||||
// Holder-then-deadline order: an insert-check that sees the holder
|
||||
// with the deadline still NO_DEADLINE compares `new < MAX` = true
|
||||
// and over-wakes — the benign direction. (Under the mandated timer
|
||||
// serialization this interleaving cannot occur anyway.)
|
||||
self.tk_armed.store(self.deadline_nanos(deadline), Ordering::SeqCst);
|
||||
true
|
||||
}
|
||||
|
||||
/// Release the timekeeper role (the holder, on wake, before it
|
||||
/// re-peeks/fires). Callable without the timer serialization: a
|
||||
/// racing insert may wake a no-longer-holder — over-wake, benign.
|
||||
pub(crate) fn disarm_timer(&self, id: usize) {
|
||||
debug_assert_eq!(
|
||||
self.tk_holder.load(Ordering::SeqCst),
|
||||
id as u64,
|
||||
"disarm_timer by a non-holder"
|
||||
);
|
||||
// Deadline first: a concurrent insert-check then sees NO_DEADLINE
|
||||
// and skips the (now pointless) wake instead of waking a stale
|
||||
// holder id. Either order is correct; this one wastes less.
|
||||
self.tk_armed.store(NO_DEADLINE, Ordering::SeqCst);
|
||||
self.tk_holder.store(NO_TIMEKEEPER, Ordering::SeqCst);
|
||||
}
|
||||
|
||||
/// Insert-side re-arm check: if `deadline` is earlier than the armed
|
||||
/// snapshot, wake the timekeeper to re-peek. MUST be called under the
|
||||
/// same timer serialization as `try_arm_timer` (see there).
|
||||
pub(crate) fn timer_inserted(&self, deadline: Instant) {
|
||||
if self.deadline_nanos(deadline) < self.tk_armed.load(Ordering::SeqCst) {
|
||||
let holder = self.tk_holder.load(Ordering::SeqCst);
|
||||
if holder != NO_TIMEKEEPER {
|
||||
// Direct unpark, not wake_one: the wake targets the
|
||||
// timekeeper specifically (it must re-peek the heap). Its
|
||||
// idle bit stays set until it returns from park — a
|
||||
// concurrent wake_one may pick it too; over-wake, benign.
|
||||
self.parkers[holder as usize].unpark();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// The armed-deadline snapshot (nanos since origin; `NO_DEADLINE` =
|
||||
/// none). Test-only introspection on the timekeeper's armed value; the
|
||||
/// busy-path due-check reads `next_deadline`, not this.
|
||||
#[cfg(test)]
|
||||
pub(crate) fn armed_deadline_nanos(&self) -> u64 {
|
||||
self.tk_armed.load(Ordering::Relaxed)
|
||||
}
|
||||
|
||||
// ----- earliest-deadline snapshot (busy-path due-check) -----
|
||||
|
||||
/// Record a newly inserted timer deadline. MUST be called under the
|
||||
/// timers mutex (same serialization rule as `try_arm_timer`), which is
|
||||
/// why plain compare+store suffices for the min-maintenance. Also runs
|
||||
/// the timekeeper re-arm check (`timer_inserted`) — one call site for
|
||||
/// both consequences of an insert.
|
||||
pub(crate) fn note_deadline(&self, deadline: Instant) {
|
||||
let n = self.deadline_nanos(deadline);
|
||||
if n < self.next_deadline.load(Ordering::Relaxed) {
|
||||
self.next_deadline.store(n, Ordering::Release);
|
||||
}
|
||||
self.timer_inserted(deadline);
|
||||
}
|
||||
|
||||
/// Re-anchor the snapshot to the heap minimum (`None` = heap empty)
|
||||
/// after a `pop_due` / `clear`. MUST be called under the timers mutex.
|
||||
pub(crate) fn refresh_deadline(&self, next: Option<Instant>) {
|
||||
let n = next.map_or(NO_DEADLINE, |d| self.deadline_nanos(d));
|
||||
self.next_deadline.store(n, Ordering::Release);
|
||||
}
|
||||
|
||||
/// Busy-path due-check: is the earliest known deadline at or past
|
||||
/// `now`? One Relaxed load when no deadline is armed — the clock is
|
||||
/// read only when a timer actually exists (matching the old drain
|
||||
/// phase's is_empty guard), so the pure-compute hot path pays a load
|
||||
/// and a branch.
|
||||
pub(crate) fn deadline_due(&self) -> bool {
|
||||
let n = self.next_deadline.load(Ordering::Relaxed);
|
||||
n != NO_DEADLINE && self.deadline_nanos(Instant::now()) >= n
|
||||
}
|
||||
|
||||
/// The earliest-deadline snapshot as an `Instant` (`None` = no timer
|
||||
/// pending). Test-only: the idle path arms the timekeeper from the
|
||||
/// timer heap's own `peek_deadline` under the timers mutex.
|
||||
#[cfg(test)]
|
||||
pub(crate) fn next_deadline_instant(&self) -> Option<Instant> {
|
||||
let n = self.next_deadline.load(Ordering::Acquire);
|
||||
if n == NO_DEADLINE {
|
||||
None
|
||||
} else {
|
||||
self.origin.checked_add(std::time::Duration::from_nanos(n))
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Unit tests (std build)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
#[cfg(all(test, not(loom)))]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use std::sync::atomic::{AtomicUsize, Ordering as O};
|
||||
use std::sync::Arc;
|
||||
use std::time::{Duration, Instant};
|
||||
|
||||
#[test]
|
||||
fn permit_before_park_returns_immediately() {
|
||||
let c = Coordinator::new(1);
|
||||
// Deliver the wake first (nobody parked: wake_one no-ops on the
|
||||
// mask, so use the timekeeper-direct path? No — permit semantics
|
||||
// are the parker's own; exercise via wake_all which permits all).
|
||||
c.wake_all();
|
||||
let t0 = Instant::now();
|
||||
let r = c.park(0, None, || false);
|
||||
assert_eq!(r, ParkResult::Woken);
|
||||
assert!(t0.elapsed() < Duration::from_millis(100), "park blocked despite permit");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn submillisecond_deadline_is_honored() {
|
||||
// Regression for the retired as_millis truncation: a 500µs deadline
|
||||
// must neither busy-return instantly forever nor round to 0/∞.
|
||||
let c = Coordinator::new(1);
|
||||
let t0 = Instant::now();
|
||||
let r = c.park(0, Some(t0 + Duration::from_micros(500)), || false);
|
||||
let dt = t0.elapsed();
|
||||
assert_eq!(r, ParkResult::TimedOut);
|
||||
assert!(dt >= Duration::from_micros(400), "woke too early: {dt:?}");
|
||||
assert!(dt < Duration::from_millis(50), "overslept: {dt:?}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn recheck_true_aborts_park_and_clears_bit() {
|
||||
let c = Coordinator::new(2);
|
||||
let r = c.park(1, None, || true);
|
||||
assert_eq!(r, ParkResult::WorkFound);
|
||||
assert_eq!(c.idle_mask(), 0, "bit not cleared after WorkFound");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn recheck_observes_own_bit_published() {
|
||||
let c = Coordinator::new(2);
|
||||
let seen = std::cell::Cell::new(0u64);
|
||||
let r = c.park(1, None, || {
|
||||
seen.set(c.idle_mask());
|
||||
true
|
||||
});
|
||||
assert_eq!(r, ParkResult::WorkFound);
|
||||
assert_eq!(seen.get() & 0b10, 0b10, "bit not published before re-check");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn wake_one_wakes_exactly_one_of_n() {
|
||||
const N: usize = 4;
|
||||
let c = Arc::new(Coordinator::new(N));
|
||||
let woken = Arc::new(AtomicUsize::new(0));
|
||||
let mut ts = Vec::new();
|
||||
for id in 0..N {
|
||||
let c = c.clone();
|
||||
let woken = woken.clone();
|
||||
ts.push(std::thread::spawn(move || {
|
||||
let r = c.park(id, None, || false);
|
||||
assert_eq!(r, ParkResult::Woken);
|
||||
woken.fetch_add(1, O::SeqCst);
|
||||
}));
|
||||
}
|
||||
// Wait until all four are published idle.
|
||||
let t0 = Instant::now();
|
||||
while c.idle_mask().count_ones() != N as u32 {
|
||||
assert!(t0.elapsed() < Duration::from_secs(5), "threads never parked");
|
||||
std::thread::yield_now();
|
||||
}
|
||||
assert!(c.wake_one());
|
||||
// Exactly one wakes; give the others a beat to (incorrectly) wake.
|
||||
let t0 = Instant::now();
|
||||
while woken.load(O::SeqCst) == 0 {
|
||||
assert!(t0.elapsed() < Duration::from_secs(5), "wake_one woke nobody");
|
||||
std::thread::yield_now();
|
||||
}
|
||||
std::thread::sleep(Duration::from_millis(100));
|
||||
assert_eq!(woken.load(O::SeqCst), 1, "wake_one woke more than one");
|
||||
assert_eq!(c.idle_mask().count_ones(), (N - 1) as u32);
|
||||
c.wake_all();
|
||||
for t in ts {
|
||||
match t.join() {
|
||||
Ok(()) => {}
|
||||
Err(p) => std::panic::resume_unwind(p),
|
||||
}
|
||||
}
|
||||
assert_eq!(woken.load(O::SeqCst), N);
|
||||
assert_eq!(c.idle_mask(), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn wake_one_prefers_highest_bit() {
|
||||
let c = Arc::new(Coordinator::new(3));
|
||||
let woken_id = Arc::new(AtomicUsize::new(usize::MAX));
|
||||
let mut ts = Vec::new();
|
||||
for id in 0..3 {
|
||||
let c = c.clone();
|
||||
let woken_id = woken_id.clone();
|
||||
ts.push(std::thread::spawn(move || {
|
||||
if c.park(id, None, || false) == ParkResult::Woken {
|
||||
let _ = woken_id.compare_exchange(usize::MAX, id, O::SeqCst, O::SeqCst);
|
||||
}
|
||||
}));
|
||||
}
|
||||
let t0 = Instant::now();
|
||||
while c.idle_mask() != 0b111 {
|
||||
assert!(t0.elapsed() < Duration::from_secs(5));
|
||||
std::thread::yield_now();
|
||||
}
|
||||
assert!(c.wake_one());
|
||||
let t0 = Instant::now();
|
||||
while woken_id.load(O::SeqCst) == usize::MAX {
|
||||
assert!(t0.elapsed() < Duration::from_secs(5));
|
||||
std::thread::yield_now();
|
||||
}
|
||||
assert_eq!(woken_id.load(O::SeqCst), 2, "LIFO-ish: highest bit first");
|
||||
c.wake_all();
|
||||
for t in ts {
|
||||
let _ = t.join();
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn wake_one_on_empty_mask_is_noop() {
|
||||
let c = Coordinator::new(2);
|
||||
assert!(!c.wake_one());
|
||||
assert_eq!(c.idle_mask(), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn timekeeper_arm_is_exclusive_and_snapshot_readable() {
|
||||
let c = Coordinator::new(2);
|
||||
let d2 = Instant::now() + Duration::from_secs(10);
|
||||
let d1 = Instant::now() + Duration::from_secs(1);
|
||||
assert_eq!(c.armed_deadline_nanos(), NO_DEADLINE);
|
||||
assert!(c.try_arm_timer(0, d2));
|
||||
assert!(!c.try_arm_timer(1, d1), "second arm must fail while held");
|
||||
assert_eq!(c.armed_deadline_nanos(), c.deadline_nanos(d2));
|
||||
c.disarm_timer(0);
|
||||
assert_eq!(c.armed_deadline_nanos(), NO_DEADLINE);
|
||||
assert!(c.try_arm_timer(1, d1), "role must be re-takeable after disarm");
|
||||
c.disarm_timer(1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn earlier_insert_wakes_timekeeper() {
|
||||
let c = Coordinator::new(2);
|
||||
let far = Instant::now() + Duration::from_secs(60);
|
||||
let near = Instant::now() + Duration::from_millis(1);
|
||||
assert!(c.try_arm_timer(0, far));
|
||||
// Holder not yet parked: the wake must land as a permit.
|
||||
c.timer_inserted(near);
|
||||
let t0 = Instant::now();
|
||||
let r = c.park(0, Some(far), || false);
|
||||
assert_eq!(r, ParkResult::Woken, "re-arm wake lost");
|
||||
assert!(t0.elapsed() < Duration::from_secs(5), "slept toward the stale deadline");
|
||||
c.disarm_timer(0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn later_insert_does_not_wake_timekeeper() {
|
||||
let c = Coordinator::new(2);
|
||||
let near = Instant::now() + Duration::from_millis(20);
|
||||
let far = Instant::now() + Duration::from_secs(60);
|
||||
assert!(c.try_arm_timer(0, near));
|
||||
c.timer_inserted(far); // later than armed: no wake
|
||||
let r = c.park(0, Some(near), || false);
|
||||
assert_eq!(r, ParkResult::TimedOut, "spurious wake for a later insert");
|
||||
c.disarm_timer(0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
#[should_panic(expected = "1..=64")]
|
||||
fn more_than_64_schedulers_asserts() {
|
||||
let _ = Coordinator::new(65);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn wake_one_if_idle_noop_on_empty_and_wakes_on_parked() {
|
||||
let c = Arc::new(Coordinator::new(1));
|
||||
assert!(!c.wake_one_if_idle(), "empty mask must be a no-op");
|
||||
let c2 = c.clone();
|
||||
let t = std::thread::spawn(move || {
|
||||
assert_eq!(c2.park(0, None, || false), ParkResult::Woken);
|
||||
});
|
||||
let t0 = Instant::now();
|
||||
while c.idle_mask() == 0 {
|
||||
assert!(t0.elapsed() < Duration::from_secs(5), "never parked");
|
||||
std::thread::yield_now();
|
||||
}
|
||||
assert!(c.wake_one_if_idle());
|
||||
match t.join() {
|
||||
Ok(()) => {}
|
||||
Err(p) => std::panic::resume_unwind(p),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn deadline_snapshot_min_maintenance_and_due_check() {
|
||||
let c = Coordinator::new(1);
|
||||
assert!(!c.deadline_due(), "no deadline: never due");
|
||||
assert_eq!(c.next_deadline_instant(), None);
|
||||
let far = Instant::now() + Duration::from_secs(60);
|
||||
let near = Instant::now() + Duration::from_millis(1);
|
||||
c.note_deadline(far);
|
||||
assert!(!c.deadline_due());
|
||||
c.note_deadline(near); // min wins
|
||||
assert!(c.next_deadline_instant().is_some_and(|d| d <= near));
|
||||
c.note_deadline(far); // later insert must NOT raise the snapshot
|
||||
assert!(c.next_deadline_instant().is_some_and(|d| d <= near));
|
||||
std::thread::sleep(Duration::from_millis(2));
|
||||
assert!(c.deadline_due(), "past deadline not reported due");
|
||||
c.refresh_deadline(Some(far));
|
||||
assert!(!c.deadline_due(), "refresh did not re-anchor");
|
||||
c.refresh_deadline(None);
|
||||
assert_eq!(c.next_deadline_instant(), None);
|
||||
// A deadline at/before origin encodes as 0: always due.
|
||||
c.note_deadline(Instant::now() - Duration::from_secs(1));
|
||||
assert!(c.deadline_due());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn note_deadline_earlier_wakes_timekeeper_via_snapshot_path() {
|
||||
// note_deadline must carry the timer_inserted re-arm wake too.
|
||||
let c = Coordinator::new(2);
|
||||
let far = Instant::now() + Duration::from_secs(60);
|
||||
let near = Instant::now() + Duration::from_millis(1);
|
||||
assert!(c.try_arm_timer(0, far));
|
||||
c.note_deadline(near);
|
||||
let r = c.park(0, Some(far), || false);
|
||||
assert_eq!(r, ParkResult::Woken, "re-arm wake lost through note_deadline");
|
||||
c.disarm_timer(0);
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// loom models — RUSTFLAGS="--cfg loom" cargo test --lib --release park
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
#[cfg(all(test, loom))]
|
||||
mod loom_tests {
|
||||
use super::*;
|
||||
use loom::sync::atomic::{AtomicU64 as LAtomicU64, Ordering as O};
|
||||
use loom::sync::{Arc, Mutex as LMutex};
|
||||
use loom::thread;
|
||||
use std::time::{Duration, Instant};
|
||||
|
||||
/// RFC 018 loom model 1 — no lost wake.
|
||||
/// producer{publish item; wake_one} ∥ consumer{set bit; re-check; park}:
|
||||
/// the consumer always observes the item or a permit; it can never
|
||||
/// sleep past a published item (loom's deadlock detector is the
|
||||
/// assertion — a consumer parked forever fails the model).
|
||||
#[test]
|
||||
fn no_lost_wake() {
|
||||
loom::model(|| {
|
||||
let c = Arc::new(Coordinator::new(1));
|
||||
let item = Arc::new(LAtomicU64::new(0));
|
||||
|
||||
let prod = {
|
||||
let c = c.clone();
|
||||
let item = item.clone();
|
||||
thread::spawn(move || {
|
||||
// The enqueue shape: publish, then the fenced fast-path
|
||||
// wake tail (this is what the runtime's enqueue calls).
|
||||
item.store(1, O::SeqCst);
|
||||
c.wake_one_if_idle();
|
||||
})
|
||||
};
|
||||
|
||||
// Consumer: loop until the item is popped. Guarded load+CAS,
|
||||
// not a blind swap — a swap writes 0 even when empty, and
|
||||
// coherence allows that write to land after the producer's
|
||||
// store in modification order, destroying the item (a model
|
||||
// bug loom caught in an earlier draft of this test).
|
||||
loop {
|
||||
if item.load(O::SeqCst) == 1
|
||||
&& item.compare_exchange(1, 0, O::SeqCst, O::SeqCst).is_ok()
|
||||
{
|
||||
break;
|
||||
}
|
||||
let _ = c.park(0, None, || item.load(O::SeqCst) == 1);
|
||||
}
|
||||
prod.join().unwrap();
|
||||
});
|
||||
}
|
||||
|
||||
/// RFC 018 loom model 2 — chain propagation.
|
||||
/// Two items, two sleepers, ONE producer wake: the chain rule (a woken
|
||||
/// consumer that sees surplus work and a non-empty mask wakes again)
|
||||
/// must get both items consumed with no further producer action.
|
||||
#[test]
|
||||
fn chain_propagation() {
|
||||
loom::model(|| {
|
||||
let c = Arc::new(Coordinator::new(2));
|
||||
let items = Arc::new(LAtomicU64::new(0));
|
||||
let consumed = Arc::new(LAtomicU64::new(0));
|
||||
|
||||
let mut hs = Vec::new();
|
||||
for id in 0..2usize {
|
||||
let c = c.clone();
|
||||
let items = items.clone();
|
||||
let consumed = consumed.clone();
|
||||
hs.push(thread::spawn(move || loop {
|
||||
if consumed.load(O::SeqCst) == 2 {
|
||||
c.wake_all(); // release a sibling still parked
|
||||
break;
|
||||
}
|
||||
let cur = items.load(O::SeqCst);
|
||||
if cur > 0
|
||||
&& items
|
||||
.compare_exchange(cur, cur - 1, O::SeqCst, O::SeqCst)
|
||||
.is_ok()
|
||||
{
|
||||
consumed.fetch_add(1, O::SeqCst);
|
||||
// THE CHAIN RULE, exactly as production expresses it
|
||||
// (runtime.rs schedule_loop): surplus ⇒ the fenced
|
||||
// fast-path wake. Models the Relaxed-load chain path,
|
||||
// not just the RMW one.
|
||||
if items.load(O::SeqCst) > 0 {
|
||||
c.wake_one_if_idle();
|
||||
}
|
||||
continue;
|
||||
}
|
||||
let _ = c.park(id, None, || {
|
||||
items.load(O::SeqCst) > 0 || consumed.load(O::SeqCst) == 2
|
||||
});
|
||||
}));
|
||||
}
|
||||
|
||||
// Producer (main): two items, ONE wake, via the enqueue-shaped
|
||||
// fenced fast path.
|
||||
items.store(2, O::SeqCst);
|
||||
c.wake_one_if_idle();
|
||||
|
||||
for h in hs {
|
||||
h.join().unwrap();
|
||||
}
|
||||
assert_eq!(consumed.load(O::SeqCst), 2);
|
||||
assert_eq!(items.load(O::SeqCst), 0);
|
||||
});
|
||||
}
|
||||
|
||||
/// RFC 018 loom model 3 — timekeeper handoff.
|
||||
/// An earlier-deadline insert racing the parking timekeeper: the
|
||||
/// earlier deadline is always honored — either the timekeeper armed it
|
||||
/// directly (insert landed first under the timers lock) or the insert
|
||||
/// wakes the timekeeper to re-peek. A timekeeper sleeping toward the
|
||||
/// stale later deadline would deadlock the model (loom has no time).
|
||||
#[test]
|
||||
fn timekeeper_handoff() {
|
||||
loom::model(|| {
|
||||
let origin = Instant::now();
|
||||
let d_far = origin + Duration::from_secs(60);
|
||||
let d_near = origin + Duration::from_secs(1);
|
||||
|
||||
let c = Arc::new(Coordinator::new(1));
|
||||
// The timers-mutex stand-in: heap min under a lock.
|
||||
let heap_min = Arc::new(LMutex::new(d_far));
|
||||
|
||||
let tk = {
|
||||
let c = c.clone();
|
||||
let heap_min = heap_min.clone();
|
||||
thread::spawn(move || {
|
||||
// Peek + arm under the lock (the serialization rule).
|
||||
let armed = {
|
||||
let g = heap_min.lock().unwrap();
|
||||
let min = *g;
|
||||
assert!(c.try_arm_timer(0, min));
|
||||
min
|
||||
};
|
||||
if armed == d_far {
|
||||
// Insert hadn't landed: it MUST wake us. Parking
|
||||
// toward d_far with no wake = model deadlock.
|
||||
let r = c.park(0, Some(armed), || false);
|
||||
assert_eq!(r, ParkResult::Woken, "re-arm wake lost");
|
||||
}
|
||||
// Woken (or armed the near deadline directly): re-peek.
|
||||
c.disarm_timer(0);
|
||||
let g = heap_min.lock().unwrap();
|
||||
assert_eq!(*g, d_near, "earlier deadline not visible on re-peek");
|
||||
})
|
||||
};
|
||||
|
||||
// Inserter (main): publish the earlier deadline under the lock,
|
||||
// then the insert-check.
|
||||
{
|
||||
let mut g = heap_min.lock().unwrap();
|
||||
*g = d_near;
|
||||
c.timer_inserted(d_near);
|
||||
}
|
||||
|
||||
tk.join().unwrap();
|
||||
});
|
||||
}
|
||||
|
||||
/// RFC 018 loom model 4 — termination.
|
||||
/// The AllDone verdict: producer flips done and wake_all()s; consumers
|
||||
/// must never park forever past done (park's re-check + wake_all's
|
||||
/// permits close every interleaving; a stuck consumer = loom deadlock).
|
||||
#[test]
|
||||
fn termination_no_park_past_done() {
|
||||
loom::model(|| {
|
||||
let c = Arc::new(Coordinator::new(2));
|
||||
let done = Arc::new(LAtomicU64::new(0));
|
||||
|
||||
let mut hs = Vec::new();
|
||||
for id in 0..2usize {
|
||||
let c = c.clone();
|
||||
let done = done.clone();
|
||||
hs.push(thread::spawn(move || loop {
|
||||
if done.load(O::SeqCst) == 1 {
|
||||
break;
|
||||
}
|
||||
let _ = c.park(id, None, || done.load(O::SeqCst) == 1);
|
||||
}));
|
||||
}
|
||||
|
||||
done.store(1, O::SeqCst);
|
||||
c.wake_all();
|
||||
|
||||
for h in hs {
|
||||
h.join().unwrap();
|
||||
}
|
||||
});
|
||||
}
|
||||
}
|
||||
+7
-4
@@ -98,10 +98,13 @@ pub(crate) fn clear_current_slot() {
|
||||
CURRENT_SLOT.with(|c| c.set(std::ptr::null()));
|
||||
}
|
||||
|
||||
/// RFC 007 (`smarm-causal`) — raw pointer to the on-CPU actor's slot, null on
|
||||
/// the scheduler's own stack. Same lifetime argument as `note_overrun`: the
|
||||
/// slot is never reclaimed while its actor is on-CPU.
|
||||
#[cfg(feature = "smarm-causal")]
|
||||
/// Raw pointer to the on-CPU actor's slot, null on the scheduler's own
|
||||
/// stack. Same lifetime argument as `note_overrun`: the slot is never
|
||||
/// reclaimed while its actor is on-CPU. Consumers: the `smarm-causal`
|
||||
/// profiler (RFC 007) and — unconditionally — the SIGSEGV classifier
|
||||
/// (RFC 019 §7), which additionally relies on this being a plain load of a
|
||||
/// const-initialized TLS Cell (no lazy init, no allocation, no dtor): safe
|
||||
/// from a signal handler.
|
||||
#[inline]
|
||||
pub(crate) fn current_slot_ptr() -> *const crate::runtime::Slot {
|
||||
CURRENT_SLOT.with(|c| c.get())
|
||||
|
||||
+459
-187
@@ -65,6 +65,8 @@
|
||||
//! word stores are `Release`, loads are `Acquire`. The chain that matters:
|
||||
//! the park path stores `sp` (Relaxed) *before* its Release transition; any
|
||||
//! later Acquire transition/load of the word therefore observes that `sp`.
|
||||
//! RFC 019's `hwm` (and the shrink that reads it) piggybacks this exact
|
||||
//! pattern in the same pre-Release window and adds no edges.
|
||||
//! The run-queue mutex independently provides the same edges today; the
|
||||
//! word's own ordering is what phase 3's lock-free queue will rely on.
|
||||
//!
|
||||
@@ -86,21 +88,28 @@
|
||||
//! # Termination (counter-based)
|
||||
//!
|
||||
//! The old all-clear scanned the slot table under the big lock. Now:
|
||||
//! exit when `io_out == 0` (read *before* the queue lock, phase-1 ordering)
|
||||
//! and, under the queue lock, the queue is empty and `live_actors == 0`.
|
||||
//! `live_actors` is incremented in `spawn` before the enqueue and decremented
|
||||
//! at the very END of `finalize_actor`, strictly after every wakeup that
|
||||
//! finalize produces has been enqueued. The soundness crux: any enqueue
|
||||
//! targets a live (not-yet-finalized) actor, so `live == 0` implies no wakeup
|
||||
//! can still be in flight; combined with "spawner is itself live", observing
|
||||
//! `(queue empty, live == 0)` under the queue lock means no work can ever
|
||||
//! appear again.
|
||||
//! exit when `io_outstanding + io_fd_waiters == 0` (two Relaxed/Acquire
|
||||
//! atomic loads, read *before* the queue pop) and, under the queue lock,
|
||||
//! the queue is empty and `live_actors == 0`. `live_actors` is incremented
|
||||
//! in `spawn` before the enqueue and decremented at the very END of
|
||||
//! `finalize_actor`, strictly after every wakeup that finalize produces has
|
||||
//! been enqueued. The soundness crux: any enqueue targets a live
|
||||
//! (not-yet-finalized) actor, so `live == 0` implies no wakeup can still be
|
||||
//! in flight; combined with "spawner is itself live", observing
|
||||
//! `(queue empty, live == 0)` means no work can ever appear again.
|
||||
//!
|
||||
//! # Timer / IO drain (try-lock, one-winner)
|
||||
//! # Scheduler park/wake (RFC 018)
|
||||
//!
|
||||
//! Unchanged from phase 1: one winner per round drains due timers and IO
|
||||
//! completions from their own mutexes; wakeups go through the unpark
|
||||
//! protocol like everyone else's.
|
||||
//! Schedulers sleep on per-thread futex parkers via the coordination layer
|
||||
//! (`park.rs`), NOT on a shared wake pipe. IO backends are producers behind
|
||||
//! a two-call contract — make the actor runnable (`unpark_at`), whose
|
||||
//! `enqueue` tail wakes exactly one parked scheduler. The blocking pool and
|
||||
//! epoll thread each route their own completions (driver-enqueues); there
|
||||
//! is no shared completion queue, no drain lock, no one-winner drain phase.
|
||||
//! Timers fire two ways: a busy-path due-check every loop iteration (one
|
||||
//! Relaxed load of the earliest-deadline snapshot when no timer is armed),
|
||||
//! and the timekeeper — at most one parked scheduler holds the timer
|
||||
//! deadline, so an expiry wakes one scheduler, not a herd.
|
||||
|
||||
use crate::actor::{
|
||||
clear_current_pid, is_actor_done, reset_actor_done, set_current_actor_box,
|
||||
@@ -153,6 +162,8 @@ pub struct Config {
|
||||
alloc_interval: u32,
|
||||
timeslice_cycles: u64,
|
||||
stack_pool_cap: usize,
|
||||
stack_reserve: usize,
|
||||
stack_guard: usize,
|
||||
max_actors: usize,
|
||||
wake_slot: bool,
|
||||
node_id: crate::pg::NodeId,
|
||||
@@ -168,6 +179,8 @@ impl Config {
|
||||
alloc_interval: crate::preempt::DEFAULT_ALLOC_INTERVAL,
|
||||
timeslice_cycles: crate::preempt::DEFAULT_TIMESLICE_CYCLES,
|
||||
stack_pool_cap: n * 4,
|
||||
stack_reserve: DEFAULT_STACK_RESERVE,
|
||||
stack_guard: DEFAULT_STACK_GUARD,
|
||||
max_actors: DEFAULT_MAX_ACTORS,
|
||||
wake_slot: false,
|
||||
node_id: crate::pg::DEFAULT_NODE_ID,
|
||||
@@ -187,6 +200,8 @@ impl Config {
|
||||
alloc_interval: crate::preempt::DEFAULT_ALLOC_INTERVAL,
|
||||
timeslice_cycles: crate::preempt::DEFAULT_TIMESLICE_CYCLES,
|
||||
stack_pool_cap: max * 4,
|
||||
stack_reserve: DEFAULT_STACK_RESERVE,
|
||||
stack_guard: DEFAULT_STACK_GUARD,
|
||||
max_actors: DEFAULT_MAX_ACTORS,
|
||||
wake_slot: false,
|
||||
node_id: crate::pg::DEFAULT_NODE_ID,
|
||||
@@ -219,6 +234,31 @@ impl Config {
|
||||
self
|
||||
}
|
||||
|
||||
/// Default per-actor stack reserve (RFC 019). A *virtual* reservation —
|
||||
/// anonymous mmap is demand-paged, so RSS follows touched pages, not
|
||||
/// this number — but overflowing it hits the guard and dies. Page-rounded.
|
||||
/// Per-actor override: `SpawnOpts::stack_reserve`.
|
||||
/// Default: [`DEFAULT_STACK_RESERVE`] (64 KiB) — the million-cheap-actors
|
||||
/// story is unchanged; big stacks are opt-in.
|
||||
pub fn stack_reserve(mut self, n: usize) -> Self {
|
||||
assert!(n > 0, "stack_reserve must be non-zero");
|
||||
self.stack_reserve = n;
|
||||
self
|
||||
}
|
||||
|
||||
/// Default PROT_NONE guard below each stack (RFC 019). Address space
|
||||
/// only. Page-rounded. Rust overflow is caught by any single page
|
||||
/// (probestack touches pages in order); the wide default exists for
|
||||
/// unprobed FFI frames, which can step over a small guard in one
|
||||
/// `sub rsp`. Per-actor override: `SpawnOpts::guard_size`.
|
||||
/// Default: [`DEFAULT_STACK_GUARD`] (1 MiB — the kernel's
|
||||
/// `stack_guard_gap` convention; see its doc for why width is free).
|
||||
pub fn stack_guard(mut self, n: usize) -> Self {
|
||||
assert!(n > 0, "stack_guard must be non-zero");
|
||||
self.stack_guard = n;
|
||||
self
|
||||
}
|
||||
|
||||
/// Capacity of the actor slot table — the maximum number of
|
||||
/// **simultaneously live** actors (total spawned over a run is unbounded;
|
||||
/// slots are recycled). The table is a fixed slab allocated once at
|
||||
@@ -286,6 +326,8 @@ impl Default for Config {
|
||||
alloc_interval: crate::preempt::DEFAULT_ALLOC_INTERVAL,
|
||||
timeslice_cycles: crate::preempt::DEFAULT_TIMESLICE_CYCLES,
|
||||
stack_pool_cap: avail * 4,
|
||||
stack_reserve: DEFAULT_STACK_RESERVE,
|
||||
stack_guard: DEFAULT_STACK_GUARD,
|
||||
max_actors: DEFAULT_MAX_ACTORS,
|
||||
wake_slot: false,
|
||||
node_id: crate::pg::DEFAULT_NODE_ID,
|
||||
@@ -376,7 +418,48 @@ impl RuntimeStats {
|
||||
// Slot — packed state word + hot atomics + cold lifecycle data
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
pub(crate) const ACTOR_STACK_SIZE: usize = 64 * 1024;
|
||||
/// Default usable stack reserve per actor (RFC 019). See [`Config::stack_reserve`].
|
||||
pub const DEFAULT_STACK_RESERVE: usize = 64 * 1024;
|
||||
|
||||
/// Default PROT_NONE guard below each actor stack (RFC 019). Raised from one
|
||||
/// page so unprobed C frames cannot leap it. See [`Config::stack_guard`].
|
||||
///
|
||||
/// 1 MiB, following the kernel's own answer to the same problem: after Stack
|
||||
/// Clash (2017) the main-thread guard gap became `stack_guard_gap` = 256
|
||||
/// pages, because 4 KiB was jumpable by one honest `sub rsp` and no small
|
||||
/// constant was defensible. Guard pages are PROT_NONE: virtual address space
|
||||
/// only — zero RSS, zero page-table entries, no overcommit charge — so the
|
||||
/// wide default is free at any actor count (1 M actors ≈ 1 TiB of VA against
|
||||
/// a 128 TiB budget). A frame that jumps even this lands in the tier-2
|
||||
/// overshoot window of the SIGSEGV diagnostic (`signal.rs`) instead of
|
||||
/// silence.
|
||||
pub const DEFAULT_STACK_GUARD: usize = 1024 * 1024;
|
||||
|
||||
/// RFC 019 §3: minimum releasable span (`sp − hwm` at park) before the
|
||||
/// park-path shrink spends a syscall. A constant, not a `Config` field
|
||||
/// (ratified): nobody tunes this well and the measured stakes are low — a
|
||||
/// threshold-sized `MADV_FREE` costs ~3 µs against a ~100 ns park, paid
|
||||
/// only on spike-recovery parks, which are rare by construction and *were*
|
||||
/// the spike. Steady-state actors never reach the syscall: their check is
|
||||
/// two Relaxed loads and a compare on a line the context-save just wrote.
|
||||
pub const SHRINK_THRESHOLD: usize = 256 * 1024;
|
||||
|
||||
/// RFC 019 §3: parks between shrinks of one actor. Guards a few-µs cost, so
|
||||
/// it can be coarse (parks, not wall time); the kernel's
|
||||
/// reclaim-under-pressure-only handling of `MADV_FREE` is the real release
|
||||
/// hysteresis — re-touched-before-pressure pages cost a 0.24 µs/page
|
||||
/// cancel-write and no fault. A constant, not `Config` (ratified, same
|
||||
/// rationale as [`SHRINK_THRESHOLD`]).
|
||||
pub const SHRINK_COOLDOWN: u32 = 64;
|
||||
|
||||
/// RFC 019 §6: the entry-end span (highest addresses — the frames the next
|
||||
/// actor faults first) a recycled stack keeps resident; everything below it
|
||||
/// is `MADV_DONTNEED`ed before the stack re-enters the pool. Ratified as a
|
||||
/// constant, not Config, alongside the shrink knobs; the 64 KiB value was a
|
||||
/// flagged Claude-solo call at ratification — it equals the default reserve,
|
||||
/// so with an unraised Config the zap is a no-op and only Configs that raise
|
||||
/// the default reserve pay it.
|
||||
pub const RECYCLE_RETAIN: usize = 64 * 1024;
|
||||
|
||||
pub(crate) type Closure = Box<dyn FnOnce() + Send>;
|
||||
|
||||
@@ -419,6 +502,35 @@ pub(crate) struct Slot {
|
||||
/// Release transition out of Running; read after the Acquire transition
|
||||
/// Queued→Running. Relaxed is sufficient — ordering rides on `word`.
|
||||
sp: AtomicUsize,
|
||||
/// RFC 019: sampled stack high-water — the minimum `sp` ever stored above,
|
||||
/// i.e. the deepest excursion *observed at a switch point*. Advisory:
|
||||
/// correctness never depends on it; its one job is "is a shrink worth a
|
||||
/// syscall?". Declared adjacent to `sp` so the min-update dirties the
|
||||
/// line the context-save just wrote. Same single-writer Relaxed
|
||||
/// discipline as `sp`; reset to the fresh `sp` at install.
|
||||
hwm: AtomicUsize,
|
||||
/// RFC 019: parks since the last shrink (or install). Counted on every
|
||||
/// pass through the Park arm by the owning scheduler thread; the shrink
|
||||
/// fires only once this clears [`SHRINK_COOLDOWN`] *and* the releasable
|
||||
/// span clears [`SHRINK_THRESHOLD`]. Single-writer Relaxed.
|
||||
parks_since_shrink: AtomicU32,
|
||||
/// RFC 019: shrinks performed on this incarnation (introspection lands
|
||||
/// with the RFC's introspect surface; the counter exists from birth so
|
||||
/// tests can rely on install resetting it). Single-writer Relaxed.
|
||||
shrink_count: AtomicU32,
|
||||
/// RFC 019 §7 — stack geometry for the SIGSEGV classifier, readable
|
||||
/// without the cold lock (the `Stack` itself lives under it). Written in
|
||||
/// `install_actor` before the Release publish; consulted by the handler
|
||||
/// only while `preempt::CURRENT_SLOT` points here, i.e. while this actor
|
||||
/// is on-CPU, so the values are never stale where they are read. 0 =
|
||||
/// never installed. Usable top of the stack.
|
||||
pub(crate) diag_stack_top: AtomicUsize,
|
||||
/// See `diag_stack_top`: the reserve (usable) size.
|
||||
pub(crate) diag_stack_reserve: AtomicUsize,
|
||||
/// See `diag_stack_top`: the guard size.
|
||||
pub(crate) diag_stack_guard: AtomicUsize,
|
||||
/// See `diag_stack_top`: `(idx << 32) | generation`, for the message.
|
||||
pub(crate) diag_pid: AtomicU64,
|
||||
/// Pointer into the actor's `Arc<AtomicBool>` stop flag. Set at spawn,
|
||||
/// nulled at finalize. The box outlives every read: it is only ever read
|
||||
/// on the resume path while the actor cannot be finalized (it is on-CPU).
|
||||
@@ -490,6 +602,13 @@ impl Slot {
|
||||
Self {
|
||||
word: StateWord::new(),
|
||||
sp: AtomicUsize::new(0),
|
||||
hwm: AtomicUsize::new(0),
|
||||
parks_since_shrink: AtomicU32::new(0),
|
||||
shrink_count: AtomicU32::new(0),
|
||||
diag_stack_top: AtomicUsize::new(0),
|
||||
diag_stack_reserve: AtomicUsize::new(0),
|
||||
diag_stack_guard: AtomicUsize::new(0),
|
||||
diag_pid: AtomicU64::new(0),
|
||||
stop_ptr: AtomicPtr::new(std::ptr::null_mut()),
|
||||
closure: AtomicPtr::new(std::ptr::null_mut()),
|
||||
overruns: AtomicU64::new(0),
|
||||
@@ -543,6 +662,23 @@ impl Slot {
|
||||
|
||||
/// Read the overrun tally (Relaxed; the snapshot reads cross-thread).
|
||||
#[inline]
|
||||
/// RFC 019 §8 — the stack introspection tuple, all lock-free:
|
||||
/// `(reserve, guard, top, hwm, parks_since_shrink, shrink_count)`.
|
||||
/// Geometry from the c6 diag atomics (install-time, gen-coherent under
|
||||
/// `read_slot`'s gen check exactly like the other counters); `hwm` is the
|
||||
/// §2 sampled high-water (lowest saved sp). All zeros before first
|
||||
/// install.
|
||||
pub(crate) fn stack_introspect(&self) -> (usize, usize, usize, usize, u32, u32) {
|
||||
(
|
||||
self.diag_stack_reserve.load(Ordering::Relaxed),
|
||||
self.diag_stack_guard.load(Ordering::Relaxed),
|
||||
self.diag_stack_top.load(Ordering::Relaxed),
|
||||
self.hwm.load(Ordering::Relaxed),
|
||||
self.parks_since_shrink.load(Ordering::Relaxed),
|
||||
self.shrink_count.load(Ordering::Relaxed),
|
||||
)
|
||||
}
|
||||
|
||||
pub(crate) fn overruns(&self) -> u64 {
|
||||
self.overruns.load(Ordering::Relaxed)
|
||||
}
|
||||
@@ -750,8 +886,21 @@ pub(crate) struct RuntimeInner {
|
||||
pub(crate) io: Mutex<Option<IoThread>>,
|
||||
/// Monotonic `MonitorId` source. Never reused.
|
||||
pub(crate) next_monitor_id: AtomicU64,
|
||||
/// Try-lock: exactly one scheduler thread drains timers/IO per iteration.
|
||||
drain_lock: Mutex<()>,
|
||||
/// RFC 018: the scheduler coordination layer — per-scheduler parkers,
|
||||
/// idle mask, wake protocol, timekeeper role, earliest-deadline
|
||||
/// snapshot. Arc'd because `Timers` shares it (insert-side deadline
|
||||
/// notes run under the timers mutex).
|
||||
pub(crate) coord: Arc<crate::park::Coordinator>,
|
||||
/// `block_on_io` requests in flight. Incremented by the submitter
|
||||
/// BEFORE submit (underflow-proof), decremented by the pool thread on
|
||||
/// completion. Read lock-free by the idle path's termination verdict —
|
||||
/// the per-pop `io.lock` of the drain era is gone.
|
||||
pub(crate) io_outstanding: AtomicU32,
|
||||
/// Parked fd waiters. Incremented by the registrar BEFORE
|
||||
/// `epoll_register` (rolled back on error), decremented by whoever
|
||||
/// consumes the registration (epoll thread on readiness, canceller on
|
||||
/// an unwound wait). Same lock-free verdict read as `io_outstanding`.
|
||||
pub(crate) io_fd_waiters: AtomicU32,
|
||||
/// Per-thread stats, indexed by scheduler thread slot (0..N).
|
||||
pub(crate) stats: Vec<SchedulerStats>,
|
||||
/// Global counters for RFC 000 primitives.
|
||||
@@ -782,17 +931,23 @@ pub(crate) struct RuntimeInner {
|
||||
pub(crate) stack_pool: RawMutex<Vec<crate::stack::Stack>>,
|
||||
/// Maximum number of stacks to retain in the pool.
|
||||
pub(crate) stack_pool_cap: usize,
|
||||
/// Default stack shape (RFC 019), pre-page-rounded so it compares exactly
|
||||
/// against `Stack::shape()`. Only stacks of exactly this shape are pooled.
|
||||
pub(crate) stack_reserve: usize,
|
||||
pub(crate) stack_guard: usize,
|
||||
}
|
||||
|
||||
impl RuntimeInner {
|
||||
// Private constructor taking the parsed Config fields one-for-one; a params
|
||||
// struct would only move the same 8 values across the call boundary.
|
||||
// struct would only move the same 10 values across the call boundary.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
fn new(
|
||||
thread_count: usize,
|
||||
alloc_interval: u32,
|
||||
timeslice_cycles: u64,
|
||||
stack_pool_cap: usize,
|
||||
stack_reserve: usize,
|
||||
stack_guard: usize,
|
||||
max_actors: usize,
|
||||
wake_slot: bool,
|
||||
node_id: crate::pg::NodeId,
|
||||
@@ -802,6 +957,12 @@ impl RuntimeInner {
|
||||
let slots: Box<[Slot]> = (0..max_actors).map(|_| Slot::vacant()).collect();
|
||||
// Low indices on top of the stack so early spawns get low pids.
|
||||
let free: Vec<u32> = (0..max_actors as u32).rev().collect();
|
||||
// RFC 018: the coordination layer (asserts thread_count <= 64), and
|
||||
// the timers' hook into it — every insert under the timers mutex
|
||||
// notes its deadline (busy-path snapshot + timekeeper re-arm).
|
||||
let coord = Arc::new(crate::park::Coordinator::new(thread_count));
|
||||
let mut timers = Timers::new();
|
||||
timers.attach_coordinator(coord.clone());
|
||||
Arc::new(Self {
|
||||
run_queue: crate::run_queue::RunQueue::new(thread_count, max_actors),
|
||||
slots,
|
||||
@@ -810,10 +971,12 @@ impl RuntimeInner {
|
||||
root_bits: AtomicU64::new(u64::MAX),
|
||||
root_exited: AtomicBool::new(false),
|
||||
root_swept: AtomicBool::new(false),
|
||||
timers: Mutex::new(Timers::new()),
|
||||
timers: Mutex::new(timers),
|
||||
io: Mutex::new(None),
|
||||
next_monitor_id: AtomicU64::new(0),
|
||||
drain_lock: Mutex::new(()),
|
||||
coord,
|
||||
io_outstanding: AtomicU32::new(0),
|
||||
io_fd_waiters: AtomicU32::new(0),
|
||||
stats,
|
||||
io_parked: AtomicU32::new(0),
|
||||
sleeping: AtomicU32::new(0),
|
||||
@@ -826,6 +989,8 @@ impl RuntimeInner {
|
||||
process_groups: RawMutex::new(crate::pg::ProcessGroups::new()),
|
||||
stack_pool: RawMutex::new(Vec::new()),
|
||||
stack_pool_cap,
|
||||
stack_reserve: crate::stack::round_to_pages(stack_reserve),
|
||||
stack_guard: crate::stack::round_to_pages(stack_guard),
|
||||
})
|
||||
}
|
||||
|
||||
@@ -874,6 +1039,13 @@ impl RuntimeInner {
|
||||
);
|
||||
self.run_queue.push(pid);
|
||||
crate::te!(crate::trace::Event::Enqueue(pid));
|
||||
// RFC 018 enqueue wake (fixes the silent enqueue): if a scheduler
|
||||
// is parked, wake exactly one. The fast path when everyone is busy
|
||||
// is a fence + one Relaxed load of an unmodified line — the
|
||||
// pure-compute hot path pays (almost) nothing. Bias is over-wake:
|
||||
// a spurious wake costs one futex round-trip and a failed pop; a
|
||||
// missed wake would cost a stranded actor.
|
||||
self.coord.wake_one_if_idle();
|
||||
}
|
||||
|
||||
/// Make `pid` runnable if it is parked; coalesce or defer otherwise.
|
||||
@@ -1010,6 +1182,9 @@ pub struct Runtime {
|
||||
|
||||
/// Initialise the runtime with the given config. Returns a reusable handle.
|
||||
pub fn init(config: Config) -> Runtime {
|
||||
// RFC 019 §7: one process-global SIGSEGV handler, installed before any
|
||||
// scheduler thread (and so before any classifiable fault) can exist.
|
||||
crate::signal::install_once();
|
||||
let n = config.resolved_thread_count();
|
||||
Runtime {
|
||||
inner: RuntimeInner::new(
|
||||
@@ -1017,6 +1192,8 @@ pub fn init(config: Config) -> Runtime {
|
||||
config.alloc_interval,
|
||||
config.timeslice_cycles,
|
||||
config.stack_pool_cap,
|
||||
config.stack_reserve,
|
||||
config.stack_guard,
|
||||
config.max_actors,
|
||||
config.wake_slot,
|
||||
config.node_id,
|
||||
@@ -1076,7 +1253,16 @@ impl Runtime {
|
||||
self.inner.live_actors.load(Ordering::Acquire), 0,
|
||||
"run() called while previous run still active"
|
||||
);
|
||||
let io_thread = match IoThread::start() {
|
||||
// RFC 018: the IO producers reach the runtime (slot table + unpark)
|
||||
// through a Weak, so no RuntimeInner → IoThread → RuntimeInner cycle
|
||||
// forms. Reset the in-flight counters BEFORE the threads can touch
|
||||
// them (a prior run left them at 0 on a clean exit; the asserts pin
|
||||
// that).
|
||||
debug_assert_eq!(self.inner.io_outstanding.load(Ordering::Acquire), 0);
|
||||
debug_assert_eq!(self.inner.io_fd_waiters.load(Ordering::Acquire), 0);
|
||||
self.inner.io_outstanding.store(0, Ordering::Release);
|
||||
self.inner.io_fd_waiters.store(0, Ordering::Release);
|
||||
let io_thread = match IoThread::start(Arc::downgrade(&self.inner)) {
|
||||
Ok(io) => io,
|
||||
Err(e) => panic!("failed to start IO thread: {e}"),
|
||||
};
|
||||
@@ -1179,6 +1365,8 @@ impl Runtime {
|
||||
}
|
||||
self.inner.io_parked.store(0, Ordering::Relaxed);
|
||||
self.inner.sleeping.store(0, Ordering::Relaxed);
|
||||
self.inner.io_outstanding.store(0, Ordering::Relaxed);
|
||||
self.inner.io_fd_waiters.store(0, Ordering::Relaxed);
|
||||
|
||||
RUNTIME.with(|r| *r.borrow_mut() = None);
|
||||
|
||||
@@ -1263,6 +1451,99 @@ pub const ROOT_PID: Pid = Pid::new(u32::MAX, u32::MAX);
|
||||
// Spawn-side slot installation
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Stack shrink — RFC 019 §3 (park path only)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// The per-park shrink check. Called from the `YieldIntent::Park` arm inside
|
||||
/// the owned window (see the assert-comment at the call site). Fast path —
|
||||
/// no spike since the last shrink — is two Relaxed loads, a compare, and the
|
||||
/// park counter bump, all on the slot line the context-save just wrote.
|
||||
///
|
||||
/// On a shrink: `MADV_FREE` the whole pages of `[hwm, sp − redzone)` (the
|
||||
/// inward-rounded range from [`crate::stack::shrink_range`]), then reset
|
||||
/// `hwm = sp` and the park counter. MADV_FREE only *marks*: the kernel
|
||||
/// reclaims under pressure, skips re-dirtied pages, and refaults zero pages
|
||||
/// for writes after reclaim — so an over-eager mark costs a cancel-write,
|
||||
/// never data.
|
||||
fn maybe_shrink_stack(slot: &Slot) {
|
||||
let parks = slot.parks_since_shrink.load(Ordering::Relaxed).saturating_add(1);
|
||||
slot.parks_since_shrink.store(parks, Ordering::Relaxed);
|
||||
|
||||
let sp = slot.sp.load(Ordering::Relaxed);
|
||||
let hwm = slot.hwm.load(Ordering::Relaxed);
|
||||
if sp.wrapping_sub(hwm) < SHRINK_THRESHOLD || sp < hwm {
|
||||
return; // common case: nothing worth a syscall
|
||||
}
|
||||
if parks < SHRINK_COOLDOWN {
|
||||
return;
|
||||
}
|
||||
let page = crate::stack::page_size();
|
||||
if let Some((addr, len)) = crate::stack::shrink_range(hwm, sp, page) {
|
||||
// Advisory: on the (kernel-config) chance MADV_FREE is unsupported,
|
||||
// failing silently degrades to "never shrinks", which is correct.
|
||||
unsafe {
|
||||
libc::madvise(addr as *mut libc::c_void, len, libc::MADV_FREE);
|
||||
}
|
||||
slot.hwm.store(sp, Ordering::Relaxed);
|
||||
slot.parks_since_shrink.store(0, Ordering::Relaxed);
|
||||
slot.shrink_count.store(
|
||||
slot.shrink_count.load(Ordering::Relaxed).saturating_add(1),
|
||||
Ordering::Relaxed,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Stack acquisition / recycling — RFC 019 pool rule
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Get a stack of the shape `opts` requests (`None` fields ⇒ the runtime
|
||||
/// defaults).
|
||||
///
|
||||
/// Pool rule (RFC 019 §1): the pool is a uniform `Vec<Stack>` of
|
||||
/// default-shaped stacks and stays that way. Default-shaped requests try the
|
||||
/// pool first; custom shapes always mmap fresh (and `recycle_stack` never
|
||||
/// admits them, so a pooled stack is default-shaped by induction). The pool
|
||||
/// lock is dropped before any mmap: no syscall ever stalls another spawner.
|
||||
pub(crate) fn acquire_stack(
|
||||
inner: &RuntimeInner,
|
||||
opts: crate::scheduler::SpawnOpts,
|
||||
) -> crate::stack::Stack {
|
||||
let reserve = opts.stack_reserve.unwrap_or(inner.stack_reserve);
|
||||
let guard = opts.guard_size.unwrap_or(inner.stack_guard);
|
||||
let default_shaped = crate::stack::round_to_pages(reserve) == inner.stack_reserve
|
||||
&& crate::stack::round_to_pages(guard) == inner.stack_guard;
|
||||
if default_shaped {
|
||||
if let Some(stack) = inner.stack_pool.lock().pop() {
|
||||
return stack;
|
||||
}
|
||||
}
|
||||
match crate::stack::Stack::new(reserve, guard) {
|
||||
Ok(stack) => stack,
|
||||
Err(e) => panic!("stack allocation failed: {e}"),
|
||||
}
|
||||
}
|
||||
|
||||
/// Return a dead actor's stack: pooled if default-shaped and under cap,
|
||||
/// otherwise dropped here → munmap (custom shapes and cap overflow alike).
|
||||
pub(crate) fn recycle_stack(inner: &RuntimeInner, stack: crate::stack::Stack) {
|
||||
if stack.shape() == (inner.stack_reserve, inner.stack_guard) {
|
||||
// RFC 019 §6: zap the dead spike before pooling, BEFORE taking the
|
||||
// pool lock — acquire_stack's invariant is that no syscall ever
|
||||
// stalls another spawner under it. On the rare cap-overflow the zap
|
||||
// is wasted work ahead of the munmap; harmless, and cheaper than a
|
||||
// second lock round-trip to find out.
|
||||
stack.recycle_zap(RECYCLE_RETAIN);
|
||||
let mut pool = inner.stack_pool.lock();
|
||||
if pool.len() < inner.stack_pool_cap {
|
||||
pool.push(stack);
|
||||
}
|
||||
// else: fall through — drop → munmap.
|
||||
}
|
||||
// Custom-shaped (or cap overflow): `stack` drops here → munmap.
|
||||
}
|
||||
|
||||
/// Install a freshly spawned actor into the slot `idx` (which must have come
|
||||
/// from `allocate_slot`) and publish it as Queued. Returns the new `Pid`.
|
||||
/// Called by `scheduler::spawn_under`; lives here next to its inverse
|
||||
@@ -1280,6 +1561,11 @@ pub(crate) fn install_actor(
|
||||
let pid = Pid::new(idx, gen);
|
||||
|
||||
let stop = Arc::new(AtomicBool::new(false));
|
||||
// RFC 019 §7: geometry for the SIGSEGV classifier, captured before the
|
||||
// Stack moves under the cold lock. Ordered before readers by the
|
||||
// publish below.
|
||||
let (diag_reserve, diag_guard) = stack.shape();
|
||||
let diag_top = stack.top() as usize;
|
||||
slot.stop_ptr.store(Arc::as_ptr(&stop) as *mut _, Ordering::Release);
|
||||
{
|
||||
let mut cold = slot.cold.lock();
|
||||
@@ -1291,6 +1577,15 @@ pub(crate) fn install_actor(
|
||||
cold.pending_io_result = None;
|
||||
}
|
||||
slot.sp.store(sp, Ordering::Relaxed);
|
||||
// RFC 019: a fresh incarnation starts with its high-water at the fresh
|
||||
// top-of-stack `sp` and its shrink bookkeeping zeroed.
|
||||
slot.hwm.store(sp, Ordering::Relaxed);
|
||||
slot.parks_since_shrink.store(0, Ordering::Relaxed);
|
||||
slot.shrink_count.store(0, Ordering::Relaxed);
|
||||
slot.diag_stack_top.store(diag_top, Ordering::Relaxed);
|
||||
slot.diag_stack_reserve.store(diag_reserve, Ordering::Relaxed);
|
||||
slot.diag_stack_guard.store(diag_guard, Ordering::Relaxed);
|
||||
slot.diag_pid.store(((idx as u64) << 32) | gen as u64, Ordering::Relaxed);
|
||||
slot.store_closure(closure);
|
||||
slot.reset_counters();
|
||||
inner.live_actors.fetch_add(1, Ordering::Relaxed);
|
||||
@@ -1391,13 +1686,7 @@ fn finalize_actor(inner: &Arc<RuntimeInner>, pid: Pid, outcome: Outcome) {
|
||||
// (the trap sender can unpark its receiver — keep that outside too).
|
||||
let supervisor_pid = actor.supervisor;
|
||||
let Actor { stack, .. } = actor;
|
||||
{
|
||||
let mut pool = inner.stack_pool.lock();
|
||||
if pool.len() < inner.stack_pool_cap {
|
||||
pool.push(stack);
|
||||
}
|
||||
// else: drop here → munmap, same as before
|
||||
}
|
||||
recycle_stack(inner, stack);
|
||||
|
||||
// Deliver to supervisor. ROOT_PID resolves to no slot → silently absorbed.
|
||||
let sender = inner.slot_at(supervisor_pid).and_then(|sup| {
|
||||
@@ -1494,53 +1783,30 @@ fn stop_live_actors(inner: &Arc<RuntimeInner>) {
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// schedule_loop — runs on each scheduler OS thread
|
||||
// Timer firing — shared by the busy-path due-check and the timekeeper
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
fn schedule_loop(inner: &Arc<RuntimeInner>, slot_idx: usize) {
|
||||
crate::preempt::configure_preempt(inner.alloc_interval, inner.timeslice_cycles);
|
||||
let stats = &inner.stats[slot_idx];
|
||||
|
||||
loop {
|
||||
// ----------------------------------------------------------------
|
||||
// 1. Try to win the drain lock (timers + IO). One winner per round;
|
||||
// losers skip immediately and proceed to step 2.
|
||||
// ----------------------------------------------------------------
|
||||
if let Ok(_drain_guard) = inner.drain_lock.try_lock() {
|
||||
// Timers and IO live behind their own mutexes (phase 1), so the
|
||||
// pure-yield / pure-compute hot path never contends a global lock
|
||||
// just to discover there is nothing to drain. The clock is read
|
||||
// only when the timer heap is non-empty.
|
||||
let due = {
|
||||
let mut t = match inner.timers.lock() {
|
||||
Ok(t) => t,
|
||||
/// Pop and dispatch every due timer. `pop_due` re-anchors the
|
||||
/// earliest-deadline snapshot under the timers mutex before returning, so
|
||||
/// a caller that raced a concurrent insert simply comes back on the next
|
||||
/// due-check. Dispatch runs with the timers lock released.
|
||||
fn fire_due_timers(inner: &Arc<RuntimeInner>, try_only: bool) {
|
||||
let due = if try_only {
|
||||
// Busy path: if another scheduler is already in the timers mutex
|
||||
// (firing, inserting, or peeking) skip — the snapshot stays due
|
||||
// until someone actually pops, so the check re-fires next loop.
|
||||
match inner.timers.try_lock() {
|
||||
Ok(mut t) => t.pop_due(std::time::Instant::now()),
|
||||
Err(std::sync::TryLockError::WouldBlock) => return,
|
||||
Err(std::sync::TryLockError::Poisoned(e)) => {
|
||||
panic!("smarm: timers lock poisoned (core corrupt): {e}")
|
||||
}
|
||||
}
|
||||
} else {
|
||||
match inner.timers.lock() {
|
||||
Ok(mut t) => t.pop_due(std::time::Instant::now()),
|
||||
Err(e) => panic!("smarm: timers lock poisoned (core corrupt): {e}"),
|
||||
};
|
||||
if t.is_empty() {
|
||||
Vec::new()
|
||||
} else {
|
||||
t.pop_due(std::time::Instant::now())
|
||||
}
|
||||
};
|
||||
let completions = match inner.io.lock() {
|
||||
Ok(mut io) => io
|
||||
.as_mut()
|
||||
.map(|io| {
|
||||
// Consume wake-pipe bytes ONLY here, under the drain
|
||||
// lock and strictly before draining completions.
|
||||
// Producers push their completion before writing the
|
||||
// byte, so every byte consumed here has its completion
|
||||
// visible to the drain below. Consuming bytes anywhere
|
||||
// else — in particular after an idle poll, outside the
|
||||
// lock — loses wakeups: a try_lock loser can eat the
|
||||
// byte for a completion the winner never saw, leaving
|
||||
// it stranded (and its EPOLLONESHOT fd disarmed) until
|
||||
// an unrelated timer forces another drain pass.
|
||||
crate::io::drain_wake_pipe(io.wake_fd());
|
||||
io.drain_completions()
|
||||
})
|
||||
.unwrap_or_default(),
|
||||
Err(e) => panic!("smarm: io lock poisoned (core corrupt): {e}"),
|
||||
};
|
||||
for entry in due {
|
||||
match entry.reason {
|
||||
@@ -1549,9 +1815,7 @@ fn schedule_loop(inner: &Arc<RuntimeInner>, slot_idx: usize) {
|
||||
// actor is between `timers.insert_sleep` and
|
||||
// `park_current`; RunningNotified makes the upcoming park
|
||||
// re-queue), or gone (no-op).
|
||||
crate::timer::Reason::Sleep { epoch } => {
|
||||
inner.unpark_at(entry.pid, epoch)
|
||||
}
|
||||
crate::timer::Reason::Sleep { epoch } => inner.unpark_at(entry.pid, epoch),
|
||||
crate::timer::Reason::WaitTimeout { target, epoch } => {
|
||||
// The callback may call unpark_at itself.
|
||||
target.on_timeout(entry.pid, epoch);
|
||||
@@ -1566,57 +1830,28 @@ fn schedule_loop(inner: &Arc<RuntimeInner>, slot_idx: usize) {
|
||||
crate::timer::Reason::Send { fire } => fire(),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for completion in completions {
|
||||
match completion {
|
||||
crate::io::Completion::Blocking { pid, epoch, result } => {
|
||||
match inner.io.lock() {
|
||||
Ok(mut io) => {
|
||||
if let Some(io) = io.as_mut() {
|
||||
io.outstanding = io.outstanding.saturating_sub(1);
|
||||
// ---------------------------------------------------------------------------
|
||||
// schedule_loop — runs on each scheduler OS thread
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
fn schedule_loop(inner: &Arc<RuntimeInner>, slot_idx: usize) {
|
||||
// RFC 019 §7: a guard hit leaves no stack to handle the signal on.
|
||||
crate::signal::register_altstack();
|
||||
crate::preempt::configure_preempt(inner.alloc_interval, inner.timeslice_cycles);
|
||||
let stats = &inner.stats[slot_idx];
|
||||
|
||||
loop {
|
||||
// ----------------------------------------------------------------
|
||||
// 1. Busy-path timer due-check (RFC 018 design point (a)): under
|
||||
// saturation nobody parks, so no timekeeper exists — due timers
|
||||
// must still fire. One Relaxed load + branch when no timer is
|
||||
// armed; the clock is read only when one is.
|
||||
// ----------------------------------------------------------------
|
||||
if inner.coord.deadline_due() {
|
||||
fire_due_timers(inner, true);
|
||||
}
|
||||
}
|
||||
Err(e) => {
|
||||
panic!("smarm: io lock poisoned (core corrupt): {e}")
|
||||
}
|
||||
}
|
||||
// Stash the result under the cold lock, then unpark.
|
||||
// The protocol also covers the submit→park window
|
||||
// (RunningNotified), which the old code missed for
|
||||
// Blocking completions — a latent lost wakeup.
|
||||
if let Some(slot) = inner.slot_at(pid) {
|
||||
{
|
||||
let mut cold = slot.cold.lock();
|
||||
if slot.generation() == pid.generation() {
|
||||
cold.pending_io_result = Some(result);
|
||||
} else {
|
||||
// Actor died (stopped) with the op in
|
||||
// flight; discard the result.
|
||||
}
|
||||
}
|
||||
inner.unpark_at(pid, epoch);
|
||||
}
|
||||
}
|
||||
crate::io::Completion::FdReady { fd, events: _ } => {
|
||||
// Resolve the parked pid under the io lock, then wake
|
||||
// through the protocol. Lock order: io before all.
|
||||
let parked = match inner.io.lock() {
|
||||
Ok(mut io) => io.as_mut().and_then(|io| {
|
||||
let entry = io.waiters.remove(&fd);
|
||||
io.epoll_deregister(fd);
|
||||
entry
|
||||
}),
|
||||
Err(e) => {
|
||||
panic!("smarm: io lock poisoned (core corrupt): {e}")
|
||||
}
|
||||
};
|
||||
if let Some((pid, epoch)) = parked {
|
||||
inner.unpark_at(pid, epoch);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
} // drain_guard drops here
|
||||
|
||||
// ----------------------------------------------------------------
|
||||
// 2. Pop a runnable pid. Pop order (RFC 005): wake slot first, then
|
||||
@@ -1625,7 +1860,7 @@ fn schedule_loop(inner: &Arc<RuntimeInner>, slot_idx: usize) {
|
||||
// ----------------------------------------------------------------
|
||||
enum Pop {
|
||||
Got(Pid),
|
||||
Idle { io_outstanding: u32, wake_fd: Option<std::os::fd::RawFd> },
|
||||
Idle,
|
||||
AllDone,
|
||||
/// Root has exited and nothing is runnable: stop the parked-forever
|
||||
/// remainder, then re-pop. Fires at most once per run.
|
||||
@@ -1650,19 +1885,12 @@ fn schedule_loop(inner: &Arc<RuntimeInner>, slot_idx: usize) {
|
||||
crate::te!(crate::trace::Event::SlotPop(pid));
|
||||
pid
|
||||
} else {
|
||||
// Read IO liveness BEFORE the queue lock (phase-1 ordering: a
|
||||
// completion resurrects an actor only via the drain path, whose
|
||||
// enqueue would be visible under the queue lock we take next).
|
||||
let (io_out, io_fd) = {
|
||||
let io = match inner.io.lock() {
|
||||
Ok(io) => io,
|
||||
Err(e) => panic!("smarm: io lock poisoned (core corrupt): {e}"),
|
||||
};
|
||||
match io.as_ref() {
|
||||
Some(io) => (io.outstanding + io.waiters.len() as u32, Some(io.wake_fd())),
|
||||
None => (0, None),
|
||||
}
|
||||
};
|
||||
// Read IO liveness BEFORE the queue pop — two atomic loads now
|
||||
// (RFC 018), not a per-pop `io.lock`: a completion resurrects
|
||||
// an actor via the producer's own unpark→enqueue, whose entry
|
||||
// would be visible to the pop below.
|
||||
let io_out = inner.io_outstanding.load(Ordering::Acquire)
|
||||
+ inner.io_fd_waiters.load(Ordering::Acquire);
|
||||
|
||||
stats.run_queue_len.store(inner.run_queue.len(), Ordering::Relaxed);
|
||||
let pop = match inner.run_queue.pop() {
|
||||
@@ -1694,7 +1922,7 @@ fn schedule_loop(inner: &Arc<RuntimeInner>, slot_idx: usize) {
|
||||
// the idle wait below on the next pass.
|
||||
Pop::RootDrain
|
||||
} else {
|
||||
Pop::Idle { io_outstanding: io_out, wake_fd: io_fd }
|
||||
Pop::Idle
|
||||
}
|
||||
}
|
||||
};
|
||||
@@ -1709,22 +1937,15 @@ fn schedule_loop(inner: &Arc<RuntimeInner>, slot_idx: usize) {
|
||||
Ok(mut timers) => timers.clear(),
|
||||
Err(e) => panic!("smarm: timers lock poisoned (core corrupt): {e}"),
|
||||
}
|
||||
// Terminal wake: a sibling scheduler may be blocked in its
|
||||
// idle wait on a snapshot that is now terminally stale — an
|
||||
// orphaned long deadline (it would sleep it out in full) or
|
||||
// a stale `io_outstanding > 0` from a stop-cancelled waiter
|
||||
// (it would block in poll(-1) forever; cancellation produces
|
||||
// no completion, so nothing else writes the wake pipe).
|
||||
// One byte wakes every poller; each re-runs the verdict,
|
||||
// reaches AllDone itself, and re-wakes — idempotent.
|
||||
match inner.io.lock() {
|
||||
Ok(io) => {
|
||||
if let Some(io) = io.as_ref() {
|
||||
io.wake();
|
||||
}
|
||||
}
|
||||
Err(e) => panic!("smarm: io lock poisoned (core corrupt): {e}"),
|
||||
}
|
||||
// Terminal wake (replaces the wake-pipe byte): a sibling
|
||||
// may be parked on a snapshot that is now terminally
|
||||
// stale — an orphaned long deadline, or a stale
|
||||
// `io_fd_waiters > 0` from a stop-cancelled waiter
|
||||
// (cancellation produces no completion, so nothing else
|
||||
// will ever wake it). `wake_all` permits every parker;
|
||||
// each sibling re-runs the verdict, reaches AllDone
|
||||
// itself, and re-wakes — idempotent.
|
||||
inner.coord.wake_all();
|
||||
return;
|
||||
}
|
||||
Pop::RootDrain => {
|
||||
@@ -1734,40 +1955,51 @@ fn schedule_loop(inner: &Arc<RuntimeInner>, slot_idx: usize) {
|
||||
stop_live_actors(inner);
|
||||
continue;
|
||||
}
|
||||
Pop::Idle { io_outstanding, wake_fd } => {
|
||||
// Something is still in flight. Sleep on the appropriate
|
||||
// source to avoid hammering the queue mutex; retry on wake.
|
||||
let next_deadline = match inner.timers.lock() {
|
||||
Ok(timers) => timers.peek_deadline(),
|
||||
Err(e) => panic!("smarm: timers lock poisoned (core corrupt): {e}"),
|
||||
Pop::Idle => {
|
||||
// Something is still in flight. Park on our own futex
|
||||
// until a producer wakes us (enqueue tail), a deadline
|
||||
// passes, or the re-check finds the world changed.
|
||||
//
|
||||
// Timekeeper (RFC 018): at most one parked scheduler
|
||||
// holds the timer deadline — the first idler to arm it
|
||||
// parks with a timeout, the rest park indefinitely, so a
|
||||
// timer expiry wakes one scheduler, not a herd. Peek and
|
||||
// arm under the timers mutex (the serialization that
|
||||
// makes the insert-side re-arm race-free).
|
||||
let tk_deadline = {
|
||||
let timers = match inner.timers.lock() {
|
||||
Ok(t) => t,
|
||||
Err(e) => {
|
||||
panic!("smarm: timers lock poisoned (core corrupt): {e}")
|
||||
}
|
||||
};
|
||||
match (next_deadline, wake_fd) {
|
||||
(Some(deadline), fd_opt) => {
|
||||
let now = std::time::Instant::now();
|
||||
if deadline > now {
|
||||
let timeout = deadline - now;
|
||||
match fd_opt {
|
||||
Some(fd) => {
|
||||
// Wake only; the byte (if any) is
|
||||
// consumed by the next drain-lock
|
||||
// winner in phase 1. Level-triggered
|
||||
// poll means an unconsumed byte makes
|
||||
// this return immediately, so a loser
|
||||
// spins briefly until the winner
|
||||
// releases — never sleeps through it.
|
||||
crate::io::poll_wake(fd, Some(timeout));
|
||||
}
|
||||
None => thread::sleep(timeout),
|
||||
}
|
||||
}
|
||||
}
|
||||
(None, Some(fd)) if io_outstanding > 0 => {
|
||||
// See above: no byte consumption outside phase 1.
|
||||
crate::io::poll_wake(fd, None);
|
||||
}
|
||||
_ => {
|
||||
thread::sleep(std::time::Duration::from_micros(100));
|
||||
}
|
||||
timers
|
||||
.peek_deadline()
|
||||
.filter(|d| inner.coord.try_arm_timer(slot_idx, *d))
|
||||
};
|
||||
// The mandatory post-publish re-check: a producer that
|
||||
// enqueued (or a verdict input that flipped) before it
|
||||
// could see our idle bit has left us the evidence.
|
||||
let _ = inner.coord.park(slot_idx, tk_deadline, || {
|
||||
!inner.run_queue.is_empty()
|
||||
|| (inner.live_actors.load(Ordering::Acquire) == 0
|
||||
&& inner.io_outstanding.load(Ordering::Acquire) == 0
|
||||
&& inner.io_fd_waiters.load(Ordering::Acquire) == 0)
|
||||
|| (inner.root_exited.load(Ordering::Acquire)
|
||||
&& !inner.root_swept.load(Ordering::Acquire))
|
||||
|| inner.coord.deadline_due()
|
||||
});
|
||||
if tk_deadline.is_some() {
|
||||
// Hand the role back BEFORE firing: pop_due can run
|
||||
// `Send` thunks that insert new timers, and the
|
||||
// insert-side re-arm check must see either no
|
||||
// timekeeper (skip) or a real parked one — never us,
|
||||
// awake and about to re-peek anyway.
|
||||
inner.coord.disarm_timer(slot_idx);
|
||||
// Woken for the deadline, for work, or to re-peek
|
||||
// after an earlier insert — fire whatever is due;
|
||||
// the next idle pass re-arms with the new minimum.
|
||||
fire_due_timers(inner, false);
|
||||
}
|
||||
continue;
|
||||
}
|
||||
@@ -1780,6 +2012,21 @@ fn schedule_loop(inner: &Arc<RuntimeInner>, slot_idx: usize) {
|
||||
// by the at-most-once-enqueued invariant nothing else can have
|
||||
// changed the state of a queued actor.
|
||||
// ----------------------------------------------------------------
|
||||
|
||||
// RFC 018 chain rule: we just took one runnable; if more remain and
|
||||
// a sibling is parked, wake exactly one so the surplus runs in
|
||||
// PARALLEL rather than serially behind us (without this the surplus
|
||||
// is not stranded — we re-pop it after resuming — but it waits out
|
||||
// our whole timeslice while an idle core sits available). Cheap: the
|
||||
// queue-length check is queue-local, and `wake_one_if_idle` is a
|
||||
// fence + one Relaxed mask load when nobody is parked. A Relaxed
|
||||
// miss here is safe — the enqueue that created the surplus already
|
||||
// issued its own wake (RFC 018 no-lost-wake); this only sharpens
|
||||
// parallelism latency.
|
||||
if !inner.run_queue.is_empty() {
|
||||
inner.coord.wake_one_if_idle();
|
||||
}
|
||||
|
||||
let slot = match inner.slot_at(pid) {
|
||||
Some(s) => s,
|
||||
None => continue, // can't happen for real pids; defensive
|
||||
@@ -1839,7 +2086,15 @@ fn schedule_loop(inner: &Arc<RuntimeInner>, slot_idx: usize) {
|
||||
crate::preempt::clear_current_slot();
|
||||
|
||||
let intent = YIELD_INTENT.with(|c| c.get());
|
||||
slot.sp.store(get_actor_sp(), Ordering::Relaxed);
|
||||
let saved_sp = get_actor_sp();
|
||||
slot.sp.store(saved_sp, Ordering::Relaxed);
|
||||
// RFC 019 §2: sampled high-water — one branch + at most one store
|
||||
// into the line the store above just dirtied. Relaxed and advisory;
|
||||
// it piggybacks the existing Relaxed-store-before-Release pattern
|
||||
// (mod docs, "Memory ordering") and adds no edges.
|
||||
if saved_sp < slot.hwm.load(Ordering::Relaxed) {
|
||||
slot.hwm.store(saved_sp, Ordering::Relaxed);
|
||||
}
|
||||
|
||||
if is_actor_done() {
|
||||
crate::te!(crate::trace::Event::Done(pid));
|
||||
@@ -1861,6 +2116,23 @@ fn schedule_loop(inner: &Arc<RuntimeInner>, slot_idx: usize) {
|
||||
inner.enqueue(pid);
|
||||
}
|
||||
YieldIntent::Park => {
|
||||
// RFC 019 §3 shrink window (correctness obligation 1):
|
||||
// this site sits after the `sp` store above and before
|
||||
// the `park_return` Release transition below publishes
|
||||
// Parked — the scheduler is on its own stack and the
|
||||
// actor is saved but not yet stealable, so the madvise
|
||||
// races nothing (belt). MADV_FREE's cancel-on-write is
|
||||
// the suspenders: even a racing writer could lose
|
||||
// nothing written after the mark, and everything below
|
||||
// live `sp` is dead by definition. Runs on BOTH arms of
|
||||
// the park_return race — a consumed unpark flag means a
|
||||
// wasted-but-harmless madvise on a rare window.
|
||||
//
|
||||
// This is the ONLY shrink site: the preempt/yield path
|
||||
// deliberately never checks (§4's bounded leak under
|
||||
// saturation — syscalls must not fire when scheduler
|
||||
// cycles are scarcest).
|
||||
maybe_shrink_stack(slot);
|
||||
if slot.word.park_return(gen) {
|
||||
// RFC 007 audit: an in-site park drops its sample
|
||||
// tail (nothing flushes it; on_resume re-arms).
|
||||
|
||||
+91
-18
@@ -72,6 +72,7 @@ use crate::runtime::{
|
||||
self, RuntimeInner, YieldIntent, RUNTIME,
|
||||
};
|
||||
use crate::supervisor::Signal;
|
||||
use std::sync::atomic::Ordering;
|
||||
use std::sync::Arc;
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -258,6 +259,28 @@ impl Drop for JoinHandle {
|
||||
// spawn / spawn_under / self_pid
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Per-spawn stack shape overrides (RFC 019). `None` fields resolve to the
|
||||
/// runtime's [`Config`](crate::runtime::Config) defaults at spawn time, so
|
||||
/// struct-update syntax works anywhere without a runtime handle:
|
||||
///
|
||||
/// ```
|
||||
/// use smarm::SpawnOpts;
|
||||
/// let opts = SpawnOpts { stack_reserve: Some(8 * 1024 * 1024), ..SpawnOpts::default() };
|
||||
/// ```
|
||||
///
|
||||
/// Both sizes are page-rounded. The reserve is *virtual* (demand-paged):
|
||||
/// an 8 MiB reserve costs address space, not memory — RSS follows touched
|
||||
/// pages. The guard is PROT_NONE below the stack; raise it for FFI code
|
||||
/// with unusually large C frames. Custom-shaped stacks bypass the recycle
|
||||
/// pool: they are mmapped fresh at spawn and munmapped at death.
|
||||
#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
|
||||
pub struct SpawnOpts {
|
||||
/// Usable stack reservation. `None` ⇒ [`Config::stack_reserve`](crate::runtime::Config::stack_reserve).
|
||||
pub stack_reserve: Option<usize>,
|
||||
/// PROT_NONE guard below the stack. `None` ⇒ [`Config::stack_guard`](crate::runtime::Config::stack_guard).
|
||||
pub guard_size: Option<usize>,
|
||||
}
|
||||
|
||||
/// Start a new actor running `f`, and return a [`JoinHandle`] for it.
|
||||
///
|
||||
/// The new actor runs concurrently with its caller and with every other
|
||||
@@ -280,22 +303,34 @@ pub fn spawn(f: impl FnOnce() + Send + 'static) -> JoinHandle {
|
||||
spawn_under(parent, f)
|
||||
}
|
||||
|
||||
/// [`spawn`] with per-actor stack shape overrides (RFC 019).
|
||||
pub fn spawn_with(opts: SpawnOpts, f: impl FnOnce() + Send + 'static) -> JoinHandle {
|
||||
let parent = current_pid().unwrap_or_else(|| {
|
||||
with_runtime(|_| crate::runtime::ROOT_PID)
|
||||
});
|
||||
spawn_under_with(parent, opts, f)
|
||||
}
|
||||
|
||||
/// Like [`spawn`], but explicitly attaches the new actor to `supervisor`
|
||||
/// instead of the calling actor. Ordinary code should reach for [`spawn`];
|
||||
/// this exists for supervision trees (see [`supervisor`](crate::supervisor))
|
||||
/// and other cases that need to place a child under a specific ancestor
|
||||
/// rather than its true caller.
|
||||
pub fn spawn_under<A>(supervisor: Pid<A>, f: impl FnOnce() + Send + 'static) -> JoinHandle {
|
||||
spawn_under_with(supervisor, SpawnOpts::default(), f)
|
||||
}
|
||||
|
||||
/// [`spawn_under`] with per-actor stack shape overrides (RFC 019).
|
||||
pub fn spawn_under_with<A>(
|
||||
supervisor: Pid<A>,
|
||||
opts: SpawnOpts,
|
||||
f: impl FnOnce() + Send + 'static,
|
||||
) -> JoinHandle {
|
||||
let supervisor = supervisor.erase();
|
||||
// Stack + closure boxing happen before ANY runtime lock is taken: no
|
||||
// syscall and no allocation ever stalls another scheduler thread.
|
||||
let stack = with_runtime(|inner| inner.stack_pool.lock().pop())
|
||||
.unwrap_or_else(|| {
|
||||
match crate::stack::Stack::new(crate::runtime::ACTOR_STACK_SIZE) {
|
||||
Ok(stack) => stack,
|
||||
Err(e) => panic!("stack allocation failed: {e}"),
|
||||
}
|
||||
});
|
||||
// Stack + closure boxing happen before the slot locks are taken; the
|
||||
// pool lock inside acquire_stack is dropped before any mmap, so no
|
||||
// syscall ever stalls another scheduler thread.
|
||||
let stack = with_runtime(|inner| crate::runtime::acquire_stack(inner, opts));
|
||||
let sp = init_actor_stack(stack.top(), crate::actor::trampoline);
|
||||
let closure: crate::runtime::Closure = Box::new(f);
|
||||
|
||||
@@ -335,6 +370,18 @@ pub fn spawn_addr<A: crate::pid::Addressable>(
|
||||
crate::pid::assert_type::<A>(pid)
|
||||
}
|
||||
|
||||
/// [`spawn_addr`] with per-actor stack shape overrides (RFC 019).
|
||||
pub fn spawn_addr_with<A: crate::pid::Addressable>(
|
||||
opts: SpawnOpts,
|
||||
body: impl FnOnce(crate::channel::Receiver<A::Msg>) + Send + 'static,
|
||||
) -> Pid<A> {
|
||||
let (tx, rx) = crate::channel::channel::<A::Msg>();
|
||||
let handle = spawn_with(opts, move || body(rx));
|
||||
let pid = handle.pid();
|
||||
crate::registry::install_for::<A::Msg>(pid, tx);
|
||||
crate::pid::assert_type::<A>(pid)
|
||||
}
|
||||
|
||||
use crate::context::init_actor_stack;
|
||||
|
||||
/// The identity of the actor currently running. Use it to hand your own
|
||||
@@ -753,7 +800,14 @@ where
|
||||
Err(e) => panic!("smarm: io lock poisoned (core corrupt): {e}"),
|
||||
};
|
||||
match io.as_mut() {
|
||||
Some(io) => io.submit(me, epoch, work),
|
||||
Some(io) => {
|
||||
// RFC 018: count the op in flight BEFORE submit — the
|
||||
// pool decrements on completion, and an increment that
|
||||
// trailed the completion would underflow. Under the io
|
||||
// lock, so ordered against the same-lock submit.
|
||||
inner.io_outstanding.fetch_add(1, Ordering::AcqRel);
|
||||
io.submit(me, epoch, work);
|
||||
}
|
||||
None => panic!("io thread not started"),
|
||||
}
|
||||
});
|
||||
@@ -813,7 +867,17 @@ fn wait_fd(fd: std::os::fd::RawFd, readable: bool, writable: bool) -> std::io::R
|
||||
Err(e) => panic!("smarm: io lock poisoned (core corrupt): {e}"),
|
||||
};
|
||||
match io.as_mut() {
|
||||
Some(io) => io.epoll_register(fd, me, epoch, readable, writable),
|
||||
Some(io) => {
|
||||
// RFC 018: count the waiter BEFORE the ADD (mirror of
|
||||
// submit); roll back if the registration fails so a
|
||||
// rejected wait leaves the verdict counters clean.
|
||||
inner.io_fd_waiters.fetch_add(1, Ordering::AcqRel);
|
||||
let r = io.epoll_register(fd, me, epoch, readable, writable);
|
||||
if r.is_err() {
|
||||
inner.io_fd_waiters.fetch_sub(1, Ordering::AcqRel);
|
||||
}
|
||||
r
|
||||
}
|
||||
None => panic!("io thread not started"),
|
||||
}
|
||||
})?;
|
||||
@@ -838,9 +902,12 @@ fn wait_fd(fd: std::os::fd::RawFd, readable: bool, writable: bool) -> std::io::R
|
||||
Err(e) => panic!("smarm: io lock poisoned (core corrupt): {e}"),
|
||||
};
|
||||
if let Some(io) = io.as_mut() {
|
||||
if io.waiters.get(&self.fd) == Some(&(self.me, self.epoch)) {
|
||||
io.waiters.remove(&self.fd);
|
||||
io.epoll_deregister(self.fd);
|
||||
// `cancel_waiter` removes + DELs iff still ours, all
|
||||
// under the waiters lock (the ADD/DEL serialization);
|
||||
// decrement only when we actually removed it — a
|
||||
// FdReady that consumed it already did the decrement.
|
||||
if io.cancel_waiter(self.fd, self.me, self.epoch) {
|
||||
inner.io_fd_waiters.fetch_sub(1, Ordering::AcqRel);
|
||||
}
|
||||
}
|
||||
});
|
||||
@@ -908,7 +975,14 @@ impl crate::channel::Selectable for FdArm {
|
||||
};
|
||||
match io.as_mut() {
|
||||
Some(io) => {
|
||||
io.epoll_register(self.fd, pid, epoch, self.readable, self.writable)
|
||||
inner.io_fd_waiters.fetch_add(1, Ordering::AcqRel);
|
||||
let r = io.epoll_register(
|
||||
self.fd, pid, epoch, self.readable, self.writable,
|
||||
);
|
||||
if r.is_err() {
|
||||
inner.io_fd_waiters.fetch_sub(1, Ordering::AcqRel);
|
||||
}
|
||||
r
|
||||
}
|
||||
None => panic!("io thread not started"),
|
||||
}
|
||||
@@ -936,9 +1010,8 @@ impl crate::channel::Selectable for FdArm {
|
||||
Err(e) => panic!("smarm: io lock poisoned (core corrupt): {e}"),
|
||||
};
|
||||
if let Some(io) = io.as_mut() {
|
||||
if io.waiters.get(&self.fd) == Some(&(pid, epoch)) {
|
||||
io.waiters.remove(&self.fd);
|
||||
io.epoll_deregister(self.fd);
|
||||
if io.cancel_waiter(self.fd, pid, epoch) {
|
||||
inner.io_fd_waiters.fetch_sub(1, Ordering::AcqRel);
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
+321
@@ -0,0 +1,321 @@
|
||||
//! RFC 019 §7 — overflow diagnostics.
|
||||
//!
|
||||
//! One process-global SIGSEGV handler, installed once at [`crate::runtime::init`]
|
||||
//! (before any scheduler thread exists, so the PRIOR save is unracing), plus a
|
||||
//! per-scheduler-thread `sigaltstack` registered at `schedule_loop` entry — a
|
||||
//! guard hit means the faulting stack has no room to run anything, so the
|
||||
//! altstack is not optional.
|
||||
//!
|
||||
//! The handler classifies `si_addr` against the *current* actor only, reached
|
||||
//! through `preempt::CURRENT_SLOT` — a const-initialized `Cell<*const Slot>`
|
||||
//! whose access is a plain TLS load (no lazy init, no allocation, no dtor
|
||||
//! registration), and which every scheduler thread has materialized before an
|
||||
//! actor can run on it. The slot's diag atomics (`diag_stack_top` & co) are
|
||||
//! written in `install_actor` before the Release publish and are only consulted
|
||||
//! here while the actor is on-CPU, so they cannot be stale.
|
||||
//!
|
||||
//! Two classification tiers:
|
||||
//! - **In-guard**: definitive. Rust frames probe pages in order
|
||||
//! (`__rust_probestack`), so Rust overflow always lands here; so does any C
|
||||
//! built with `-fstack-clash-protection` (distro-packaged libraries), and —
|
||||
//! with the 1 MiB default guard — nearly every unprobed frame too.
|
||||
//! - **Overshoot**: within [`OVERSHOOT_SLOP`] *below* the guard. An unprobed
|
||||
//! frame (cargo-built C via `cc` almost never enables clash protection)
|
||||
//! large enough to step over the guard in one `sub rsp`. Attribution is
|
||||
//! "probable": the address is in unmapped VA that nothing else owns, an
|
||||
//! actor was on-CPU, and the distance fits a frame — the diagnostic says so.
|
||||
//!
|
||||
//! Classified faults print one line (async-signal-safe: stack buffer +
|
||||
//! `write(2)`, no fmt, no alloc, no locks) and re-raise with default
|
||||
//! disposition — no unwind, no resume, no fail-soft (jarred; UB-adjacent from
|
||||
//! a handler). Unclassified faults reinstate the PRIOR handler and refault, so
|
||||
//! std's own "thread ... has overflowed its stack" diagnostics for OS-thread
|
||||
//! stacks survive our presence. Reinstating deregisters us for good, which is
|
||||
//! fine: the process is dying either way.
|
||||
|
||||
use std::cell::Cell;
|
||||
use std::mem::MaybeUninit;
|
||||
use std::sync::atomic::Ordering;
|
||||
use std::sync::Once;
|
||||
|
||||
/// Tier-2 window below the guard. Matches the guard default (and the kernel's
|
||||
/// `stack_guard_gap`): a frame that out-jumps both the guard and this window
|
||||
/// in one displacement is past what a diagnostic can honestly attribute.
|
||||
pub(crate) const OVERSHOOT_SLOP: usize = 1024 * 1024;
|
||||
|
||||
/// Per-scheduler-thread signal stack. MINSIGSTKSZ is ~11 KiB on AVX-512
|
||||
/// hardware; 64 KiB leaves the formatter room without mattering to anyone.
|
||||
/// One per OS thread, never freed: scheduler threads live for the process in
|
||||
/// practice, and repeated `run()`s on reused threads re-use the registration
|
||||
/// (the TLS flag), so the leak is bounded by the OS thread count.
|
||||
const ALTSTACK_SIZE: usize = 64 * 1024;
|
||||
|
||||
static INSTALL: Once = Once::new();
|
||||
/// The handler that was installed before ours (std's, typically). Written
|
||||
/// exactly once inside INSTALL — which completes in `runtime::init` before
|
||||
/// any scheduler thread (and thus any classifiable fault) can exist — and
|
||||
/// only read from the handler afterwards.
|
||||
static mut PRIOR: MaybeUninit<libc::sigaction> = MaybeUninit::uninit();
|
||||
|
||||
thread_local! {
|
||||
/// Whether this OS thread has registered its altstack.
|
||||
static ALTSTACK_SET: Cell<bool> = const { Cell::new(false) };
|
||||
}
|
||||
|
||||
/// Where a fault landed relative to the current actor's stack.
|
||||
#[derive(Debug, PartialEq, Eq)]
|
||||
pub(crate) enum FaultClass {
|
||||
/// Inside `[top − reserve − guard, top − reserve)`: the guard region.
|
||||
Guard,
|
||||
/// Within `OVERSHOOT_SLOP` below the guard: stepped over it. Payload is
|
||||
/// the distance below `guard_lo`.
|
||||
Overshoot(usize),
|
||||
/// Not ours to explain.
|
||||
Foreign,
|
||||
}
|
||||
|
||||
/// Pure classifier — all edges unit-tested below. `top` is the stack's usable
|
||||
/// top, `reserve`/`guard` its shape; both page-rounded by `Stack::new`.
|
||||
pub(crate) fn classify(addr: usize, top: usize, reserve: usize, guard: usize) -> FaultClass {
|
||||
let guard_hi = top.wrapping_sub(reserve);
|
||||
let guard_lo = guard_hi.wrapping_sub(guard);
|
||||
if addr >= guard_lo && addr < guard_hi {
|
||||
FaultClass::Guard
|
||||
} else if addr < guard_lo && addr >= guard_lo.saturating_sub(OVERSHOOT_SLOP) {
|
||||
FaultClass::Overshoot(guard_lo - addr)
|
||||
} else {
|
||||
FaultClass::Foreign
|
||||
}
|
||||
}
|
||||
|
||||
/// Install the process-global handler. Idempotent; called from
|
||||
/// `runtime::init`.
|
||||
pub(crate) fn install_once() {
|
||||
INSTALL.call_once(|| unsafe {
|
||||
let mut sa: libc::sigaction = std::mem::zeroed();
|
||||
sa.sa_sigaction = handler as *const () as usize;
|
||||
sa.sa_flags = libc::SA_SIGINFO | libc::SA_ONSTACK;
|
||||
libc::sigemptyset(&mut sa.sa_mask);
|
||||
let prior = &mut *std::ptr::addr_of_mut!(PRIOR);
|
||||
libc::sigaction(libc::SIGSEGV, &sa, prior.as_mut_ptr());
|
||||
});
|
||||
}
|
||||
|
||||
/// Register this OS thread's altstack (idempotent per thread). Called at
|
||||
/// `schedule_loop` entry, so every thread that can run an actor has one.
|
||||
pub(crate) fn register_altstack() {
|
||||
ALTSTACK_SET.with(|set| {
|
||||
if set.get() {
|
||||
return;
|
||||
}
|
||||
unsafe {
|
||||
let sp = libc::mmap(
|
||||
std::ptr::null_mut(),
|
||||
ALTSTACK_SIZE,
|
||||
libc::PROT_READ | libc::PROT_WRITE,
|
||||
libc::MAP_PRIVATE | libc::MAP_ANONYMOUS,
|
||||
-1,
|
||||
0,
|
||||
);
|
||||
if sp == libc::MAP_FAILED {
|
||||
// Degrade: no altstack means a guard hit dies without the
|
||||
// message (handler can't run) — the pre-RFC behavior, never
|
||||
// incorrectness.
|
||||
return;
|
||||
}
|
||||
let ss = libc::stack_t {
|
||||
ss_sp: sp,
|
||||
ss_flags: 0,
|
||||
ss_size: ALTSTACK_SIZE,
|
||||
};
|
||||
libc::sigaltstack(&ss, std::ptr::null_mut());
|
||||
}
|
||||
set.set(true);
|
||||
});
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// The handler
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
unsafe extern "C" fn handler(
|
||||
_sig: libc::c_int,
|
||||
info: *mut libc::siginfo_t,
|
||||
_ctx: *mut libc::c_void,
|
||||
) {
|
||||
let slot_ptr = crate::preempt::current_slot_ptr();
|
||||
if !slot_ptr.is_null() {
|
||||
let slot = &*slot_ptr;
|
||||
let top = slot.diag_stack_top.load(Ordering::Relaxed);
|
||||
if top != 0 {
|
||||
let reserve = slot.diag_stack_reserve.load(Ordering::Relaxed);
|
||||
let guard = slot.diag_stack_guard.load(Ordering::Relaxed);
|
||||
let pid = slot.diag_pid.load(Ordering::Relaxed);
|
||||
let addr = (*info).si_addr() as usize;
|
||||
match classify(addr, top, reserve, guard) {
|
||||
FaultClass::Guard => {
|
||||
let mut b = Buf::new();
|
||||
b.s("smarm: actor ");
|
||||
b.pid(pid);
|
||||
b.s(" overflowed its stack: fault in the guard region, depth-at-fault=");
|
||||
b.u(top - addr);
|
||||
b.s(" bytes (reserve=");
|
||||
b.u(reserve);
|
||||
b.s(", guard=");
|
||||
b.u(guard);
|
||||
b.s("). Raise stack_reserve (SpawnOpts or Config).\n");
|
||||
b.emit();
|
||||
die_by_default();
|
||||
return;
|
||||
}
|
||||
FaultClass::Overshoot(below) => {
|
||||
let mut b = Buf::new();
|
||||
b.s("smarm: actor ");
|
||||
b.pid(pid);
|
||||
b.s(" probably overflowed its stack: fault ");
|
||||
b.u(below);
|
||||
b.s(" bytes below the guard - an unprobed (FFI?) frame stepped over it (reserve=");
|
||||
b.u(reserve);
|
||||
b.s(", guard=");
|
||||
b.u(guard);
|
||||
b.s("). Raise stack_guard or stack_reserve.\n");
|
||||
b.emit();
|
||||
die_by_default();
|
||||
return;
|
||||
}
|
||||
FaultClass::Foreign => {}
|
||||
}
|
||||
}
|
||||
}
|
||||
// Not ours: put back whoever was there before us and refault into them.
|
||||
let prior = &*std::ptr::addr_of!(PRIOR);
|
||||
libc::sigaction(libc::SIGSEGV, prior.as_ptr(), std::ptr::null_mut());
|
||||
}
|
||||
|
||||
/// Reset SIGSEGV to default disposition; returning from the handler then
|
||||
/// refaults at the same instruction and the process dies the normal death
|
||||
/// (core-dumpable, correct wait status), exactly as if we were never here —
|
||||
/// but with the message already on stderr.
|
||||
unsafe fn die_by_default() {
|
||||
let mut dfl: libc::sigaction = std::mem::zeroed();
|
||||
dfl.sa_sigaction = libc::SIG_DFL;
|
||||
libc::sigemptyset(&mut dfl.sa_mask);
|
||||
libc::sigaction(libc::SIGSEGV, &dfl, std::ptr::null_mut());
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Async-signal-safe formatting: fixed buffer, decimal itoa, one write(2).
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
struct Buf {
|
||||
b: [u8; 320],
|
||||
len: usize,
|
||||
}
|
||||
|
||||
impl Buf {
|
||||
fn new() -> Self {
|
||||
Buf { b: [0; 320], len: 0 }
|
||||
}
|
||||
fn s(&mut self, s: &str) {
|
||||
for &c in s.as_bytes() {
|
||||
if self.len < self.b.len() {
|
||||
self.b[self.len] = c;
|
||||
self.len += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
fn u(&mut self, mut n: usize) {
|
||||
let mut tmp = [0u8; 20];
|
||||
let mut i = tmp.len();
|
||||
loop {
|
||||
i -= 1;
|
||||
tmp[i] = b'0' + (n % 10) as u8;
|
||||
n /= 10;
|
||||
if n == 0 {
|
||||
break;
|
||||
}
|
||||
}
|
||||
for &c in &tmp[i..] {
|
||||
if self.len < self.b.len() {
|
||||
self.b[self.len] = c;
|
||||
self.len += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
/// `idx.gen`, unpacked from the install-time packing.
|
||||
fn pid(&mut self, packed: u64) {
|
||||
self.u((packed >> 32) as usize);
|
||||
self.s(".");
|
||||
self.u((packed & 0xffff_ffff) as usize);
|
||||
}
|
||||
fn emit(&self) {
|
||||
unsafe {
|
||||
libc::write(2, self.b.as_ptr() as *const libc::c_void, self.len);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Classifier units — the arithmetic edges, before anything integrates.
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::{classify, FaultClass, OVERSHOOT_SLOP};
|
||||
|
||||
const PG: usize = 4096;
|
||||
// A synthetic stack far from address-space edges: top at 1 GiB.
|
||||
const TOP: usize = 1 << 30;
|
||||
const RESERVE: usize = 16 * PG;
|
||||
const GUARD: usize = 4 * PG;
|
||||
const GUARD_HI: usize = TOP - RESERVE;
|
||||
const GUARD_LO: usize = GUARD_HI - GUARD;
|
||||
|
||||
#[test]
|
||||
fn inside_guard_both_edges() {
|
||||
assert_eq!(classify(GUARD_LO, TOP, RESERVE, GUARD), FaultClass::Guard);
|
||||
assert_eq!(classify(GUARD_HI - 1, TOP, RESERVE, GUARD), FaultClass::Guard);
|
||||
assert_eq!(classify(GUARD_LO + GUARD / 2, TOP, RESERVE, GUARD), FaultClass::Guard);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn usable_region_is_foreign() {
|
||||
// A fault inside the RW stack itself isn't a guard hit and must not
|
||||
// be explained as one.
|
||||
assert_eq!(classify(GUARD_HI, TOP, RESERVE, GUARD), FaultClass::Foreign);
|
||||
assert_eq!(classify(TOP - 1, TOP, RESERVE, GUARD), FaultClass::Foreign);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn above_top_is_foreign() {
|
||||
assert_eq!(classify(TOP, TOP, RESERVE, GUARD), FaultClass::Foreign);
|
||||
assert_eq!(classify(TOP + PG, TOP, RESERVE, GUARD), FaultClass::Foreign);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn overshoot_window_edges() {
|
||||
assert_eq!(
|
||||
classify(GUARD_LO - 1, TOP, RESERVE, GUARD),
|
||||
FaultClass::Overshoot(1)
|
||||
);
|
||||
assert_eq!(
|
||||
classify(GUARD_LO - OVERSHOOT_SLOP, TOP, RESERVE, GUARD),
|
||||
FaultClass::Overshoot(OVERSHOOT_SLOP)
|
||||
);
|
||||
assert_eq!(
|
||||
classify(GUARD_LO - OVERSHOOT_SLOP - 1, TOP, RESERVE, GUARD),
|
||||
FaultClass::Foreign
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn low_address_stack_saturates_not_wraps() {
|
||||
// A stack mapped so low that the slop window would underflow: the
|
||||
// window clips to 0 instead of wrapping around the address space.
|
||||
let top = RESERVE + GUARD + PG; // guard_lo == PG
|
||||
assert_eq!(classify(0, top, RESERVE, GUARD), FaultClass::Overshoot(PG));
|
||||
// Null-page fault still classified only because it IS within slop
|
||||
// here; with a normal-height stack it is Foreign (covered above by
|
||||
// the window-edge test at realistic addresses).
|
||||
}
|
||||
}
|
||||
+226
-13
@@ -1,32 +1,45 @@
|
||||
//! mmap-based growable stack with a guard page below.
|
||||
//! mmap-based actor stack with a PROT_NONE guard region below (RFC 019).
|
||||
//!
|
||||
//! Layout (low → high address):
|
||||
//! [ guard page (PROT_NONE) | stack region ]
|
||||
//! [ guard region (PROT_NONE) | stack region ]
|
||||
//! ^ top() — initial stack pointer
|
||||
//!
|
||||
//! Stacks grow downward. Overflow lands in the guard page → SIGSEGV.
|
||||
//! Stacks grow downward. Overflow lands in the guard region → SIGSEGV.
|
||||
//!
|
||||
//! Both the usable reserve and the guard are caller-chosen (page-rounded).
|
||||
//! The reserve is a *virtual* reservation: anonymous mmap is demand-paged,
|
||||
//! so RSS is touched-pages, not reserve × actors. The guard costs address
|
||||
//! space only. A wide guard (the runtime defaults to 64 KiB) exists for
|
||||
//! unprobed FFI frames: Rust frames touch pages in order (probestack), so
|
||||
//! one page catches Rust overflow, but a C frame with a large local can
|
||||
//! step over a single page in one `sub rsp`.
|
||||
|
||||
use std::io;
|
||||
|
||||
pub struct Stack {
|
||||
/// Bottom of the entire mmap'd region (start of guard page).
|
||||
/// Bottom of the entire mmap'd region (start of the guard).
|
||||
base: *mut u8,
|
||||
/// Total mmap'd size: guard_size + stack_size.
|
||||
total_size: usize,
|
||||
/// Usable stack size (excluding guard page).
|
||||
/// Usable stack size (excluding the guard).
|
||||
stack_size: usize,
|
||||
/// PROT_NONE region below the usable stack.
|
||||
guard_size: usize,
|
||||
}
|
||||
|
||||
// Stack owns its memory; safe to send across threads.
|
||||
unsafe impl Send for Stack {}
|
||||
|
||||
impl Stack {
|
||||
/// Allocate a new stack. `stack_size` is the usable region; one page is
|
||||
/// added below as a guard page. Both are rounded up to the page size.
|
||||
pub fn new(stack_size: usize) -> io::Result<Self> {
|
||||
/// Allocate a new stack. `stack_size` is the usable region; `guard_size`
|
||||
/// is mapped PROT_NONE below it. Both are rounded up to the page size
|
||||
/// and must be non-zero.
|
||||
pub fn new(stack_size: usize, guard_size: usize) -> io::Result<Self> {
|
||||
assert!(stack_size > 0, "stack_size must be non-zero");
|
||||
assert!(guard_size > 0, "guard_size must be non-zero");
|
||||
let page = page_size();
|
||||
let stack_size = round_up(stack_size, page);
|
||||
let guard_size = page;
|
||||
let guard_size = round_up(guard_size, page);
|
||||
let total_size = guard_size + stack_size;
|
||||
|
||||
let base = unsafe {
|
||||
@@ -53,7 +66,7 @@ impl Stack {
|
||||
return Err(err);
|
||||
}
|
||||
|
||||
Ok(Self { base, total_size, stack_size })
|
||||
Ok(Self { base, total_size, stack_size, guard_size })
|
||||
}
|
||||
|
||||
/// 16-byte-aligned top of the usable region.
|
||||
@@ -62,14 +75,54 @@ impl Stack {
|
||||
(raw_top & !15) as *mut u8
|
||||
}
|
||||
|
||||
/// Pointer to the bottom of the usable region (just above the guard page).
|
||||
/// Pointer to the bottom of the usable region (just above the guard).
|
||||
pub fn usable_base(&self) -> *mut u8 {
|
||||
unsafe { self.base.add(page_size()) }
|
||||
unsafe { self.base.add(self.guard_size) }
|
||||
}
|
||||
|
||||
pub fn stack_size(&self) -> usize {
|
||||
self.stack_size
|
||||
}
|
||||
|
||||
pub fn guard_size(&self) -> usize {
|
||||
self.guard_size
|
||||
}
|
||||
|
||||
/// `(stack_size, guard_size)` after page rounding. The pool rule
|
||||
/// (RFC 019 §1) compares this against the runtime defaults: only
|
||||
/// default-shaped stacks are pooled.
|
||||
pub fn shape(&self) -> (usize, usize) {
|
||||
(self.stack_size, self.guard_size)
|
||||
}
|
||||
|
||||
/// Pool-recycle zap (RFC 019 §6): `MADV_DONTNEED` everything below the
|
||||
/// retained entry end `[top − retain, top)` — the span the next actor's
|
||||
/// shallow frames land in stays resident, the dead spike below it is
|
||||
/// released. The stack is unowned at the call site (its actor is dead),
|
||||
/// so a synchronous eager zap races nothing and the RSS drop is
|
||||
/// immediate — a museum of worst-case spikes is exactly what a pool must
|
||||
/// not be; DONTNEED's ~8× per-page cost vs FREE is irrelevant off the
|
||||
/// hot path. Advisory like the park-path shrink: a failure degrades to
|
||||
/// "the pool keeps RSS", never to incorrectness. No-op (no syscall) when
|
||||
/// `retain` covers the whole usable region — i.e. always, at the 64 KiB
|
||||
/// default reserve.
|
||||
pub(crate) fn recycle_zap(&self, retain: usize) {
|
||||
if let Some((off, len)) = retain_range(self.stack_size, retain, page_size()) {
|
||||
unsafe {
|
||||
libc::madvise(
|
||||
self.usable_base().add(off) as *mut libc::c_void,
|
||||
len,
|
||||
libc::MADV_DONTNEED,
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Round `n` up to whole pages — the same rounding `Stack::new` applies, so
|
||||
/// runtime defaults stored pre-rounded compare exactly against [`Stack::shape`].
|
||||
pub(crate) fn round_to_pages(n: usize) -> usize {
|
||||
round_up(n, page_size())
|
||||
}
|
||||
|
||||
impl Drop for Stack {
|
||||
@@ -80,10 +133,170 @@ impl Drop for Stack {
|
||||
}
|
||||
}
|
||||
|
||||
fn page_size() -> usize {
|
||||
pub(crate) fn page_size() -> usize {
|
||||
unsafe { libc::sysconf(libc::_SC_PAGESIZE) as usize }
|
||||
}
|
||||
|
||||
fn round_up(n: usize, align: usize) -> usize {
|
||||
(n + align - 1) & !(align - 1)
|
||||
}
|
||||
|
||||
/// The whole-page span the park-path shrink may `MADV_FREE` (RFC 019 §3):
|
||||
/// `[page_up(hwm), page_down(sp − redzone))`, or `None` if no full page fits.
|
||||
///
|
||||
/// `hwm` is the sampled high-water (deepest observed `sp`); everything in
|
||||
/// `[hwm, sp)` is below the live frame and dead by definition. One page of
|
||||
/// redzone stays resident under live `sp` — it covers the SysV 128-byte red
|
||||
/// zone plus spill margin with room to spare. Rounding is inward on both
|
||||
/// ends so the result can never touch the redzone, cross `sp`, or dip below
|
||||
/// `hwm`; all arithmetic is checked so adversarial inputs (`sp < redzone`,
|
||||
/// `hwm ≥ sp`, values near the address-space edges) collapse to `None`
|
||||
/// rather than a wild or negative-length range.
|
||||
pub(crate) fn shrink_range(hwm: usize, sp: usize, page: usize) -> Option<(usize, usize)> {
|
||||
debug_assert!(page.is_power_of_two());
|
||||
if hwm >= sp {
|
||||
return None;
|
||||
}
|
||||
let redzone = page;
|
||||
let end = sp.checked_sub(redzone)? & !(page - 1); // page_down(sp − redzone)
|
||||
let start = hwm.checked_add(page - 1)? & !(page - 1); // page_up(hwm)
|
||||
if end > start {
|
||||
Some((start, end - start))
|
||||
} else {
|
||||
None
|
||||
}
|
||||
}
|
||||
|
||||
/// The `(offset_from_usable_base, len)` span the pool recycle DONTNEEDs
|
||||
/// (RFC 019 §6): everything below the retained entry end. "Bottom RETAIN of
|
||||
/// the stack" is read stack-wise (entry frames = highest addresses of a
|
||||
/// downward stack): the retained span is `[top − page_up(retain), top)`, the
|
||||
/// zapped span is the rest — retaining the low-address deep end instead
|
||||
/// would keep the coldest pages and release the ones the next actor faults
|
||||
/// first. `retain` rounds *up* to whole pages (retain more, zap less), so
|
||||
/// with `stack_size` page-rounded by `Stack::new` the result is always
|
||||
/// page-aligned. Checked math: `retain ≥ stack_size` (notably the default
|
||||
/// 64 KiB reserve with the 64 KiB RETAIN) and overflow collapse to `None`.
|
||||
pub(crate) fn retain_range(stack_size: usize, retain: usize, page: usize) -> Option<(usize, usize)> {
|
||||
debug_assert!(page.is_power_of_two());
|
||||
let retain = retain.checked_add(page - 1)? & !(page - 1); // page_up(retain)
|
||||
let len = stack_size.checked_sub(retain)?;
|
||||
if len == 0 {
|
||||
return None;
|
||||
}
|
||||
Some((0, len))
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::{retain_range, shrink_range};
|
||||
|
||||
const PG: usize = 4096;
|
||||
|
||||
#[test]
|
||||
fn retain_covers_whole_stack_is_a_noop() {
|
||||
// The default config: reserve == RETAIN == 64 KiB. No zap, no syscall.
|
||||
assert_eq!(retain_range(16 * PG, 16 * PG, PG), None);
|
||||
assert_eq!(retain_range(PG, PG, PG), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn retain_larger_than_stack_is_a_noop() {
|
||||
assert_eq!(retain_range(16 * PG, 17 * PG, PG), None);
|
||||
assert_eq!(retain_range(PG, usize::MAX, PG), None); // page_up overflows
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn retain_zero_zaps_everything() {
|
||||
assert_eq!(retain_range(16 * PG, 0, PG), Some((0, 16 * PG)));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn retain_rounds_up_zapping_less() {
|
||||
// 1 byte of retain keeps a whole page.
|
||||
assert_eq!(retain_range(16 * PG, 1, PG), Some((0, 15 * PG)));
|
||||
assert_eq!(retain_range(16 * PG, PG + 1, PG), Some((0, 14 * PG)));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn retain_one_page_short_of_stack() {
|
||||
assert_eq!(retain_range(2 * PG, PG, PG), Some((0, PG)));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn retain_range_is_page_aligned() {
|
||||
for size_pg in [1usize, 2, 3, 16, 1024] {
|
||||
for retain in [0usize, 1, PG - 1, PG, PG + 1, 4 * PG, size_pg * PG] {
|
||||
if let Some((off, len)) = retain_range(size_pg * PG, retain, PG) {
|
||||
assert_eq!(off, 0);
|
||||
assert_eq!(len % PG, 0);
|
||||
assert!(len <= size_pg * PG);
|
||||
assert!(len > 0);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn empty_and_inverted_spans_are_none() {
|
||||
assert_eq!(shrink_range(0x8000_0000, 0x8000_0000, PG), None); // hwm == sp
|
||||
assert_eq!(shrink_range(0x8000_1000, 0x8000_0000, PG), None); // hwm > sp
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn span_smaller_than_redzone_plus_page_is_none() {
|
||||
let sp = 0x8000_0000;
|
||||
// Everything within redzone+1 page of sp: no full page clears both
|
||||
// the redzone and the page_up(hwm) rounding.
|
||||
assert_eq!(shrink_range(sp - PG, sp, PG), None);
|
||||
assert_eq!(shrink_range(sp - 2 * PG + 1, sp, PG), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn exact_two_pages_frees_one() {
|
||||
let sp = 0x8000_0000;
|
||||
let hwm = sp - 2 * PG;
|
||||
// [hwm, hwm+PG) frees; [sp−PG, sp) is redzone.
|
||||
assert_eq!(shrink_range(hwm, sp, PG), Some((hwm, PG)));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn unaligned_ends_round_inward() {
|
||||
let sp = 0x8000_0123; // live sp mid-page
|
||||
let hwm = 0x7f00_0abc; // high-water mid-page
|
||||
let (start, len) = shrink_range(hwm, sp, PG).unwrap();
|
||||
assert_eq!(start % PG, 0);
|
||||
assert_eq!(len % PG, 0);
|
||||
assert!(start >= hwm); // never below the sampled high-water
|
||||
assert!(start + len <= (sp - PG) & !(PG - 1)); // never into the redzone
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn result_never_crosses_sp() {
|
||||
// Sweep hwm across every offset of the page straddling the boundary.
|
||||
let sp = 0x8000_0000 + 137;
|
||||
for hwm in (sp - 4 * PG)..(sp) {
|
||||
if let Some((start, len)) = shrink_range(hwm, sp, PG) {
|
||||
assert!(start >= hwm);
|
||||
assert!(start + len + PG <= sp + PG); // end ≤ page_down(sp − PG) < sp
|
||||
assert!(len > 0);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn underflow_near_zero_is_none() {
|
||||
assert_eq!(shrink_range(0, PG - 1, PG), None); // sp < redzone
|
||||
assert_eq!(shrink_range(0, 0, PG), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn big_span_frees_interior() {
|
||||
let sp = 0x8000_0000;
|
||||
let spike = 4 * 1024 * 1024;
|
||||
let hwm = sp - spike;
|
||||
let (start, len) = shrink_range(hwm, sp, PG).unwrap();
|
||||
assert_eq!(start, hwm); // aligned input: starts exactly at hwm
|
||||
assert_eq!(len, spike - PG); // everything but the redzone page
|
||||
}
|
||||
}
|
||||
|
||||
+11
-2
@@ -6,10 +6,19 @@
|
||||
//! Build the loom models with: `RUSTFLAGS="--cfg loom" cargo test --lib --release`
|
||||
|
||||
#[cfg(loom)]
|
||||
pub(crate) use loom::sync::atomic::{AtomicU64, AtomicUsize, Ordering};
|
||||
pub(crate) use loom::sync::atomic::{fence, AtomicU64, AtomicUsize, Ordering};
|
||||
|
||||
#[cfg(not(loom))]
|
||||
pub(crate) use std::sync::atomic::{AtomicU64, AtomicUsize, Ordering};
|
||||
pub(crate) use std::sync::atomic::{fence, AtomicU64, AtomicUsize, Ordering};
|
||||
|
||||
// park.rs condvar-parker (loom + non-Linux builds only; the Linux non-loom
|
||||
// build parks on a futex and never touches these — gating them identically
|
||||
// keeps the default build free of unused imports).
|
||||
#[cfg(loom)]
|
||||
pub(crate) use loom::sync::{Condvar, Mutex};
|
||||
|
||||
#[cfg(all(not(loom), not(target_os = "linux")))]
|
||||
pub(crate) use std::sync::{Condvar, Mutex};
|
||||
|
||||
/// `UnsafeCell` with loom's `with`/`with_mut` access API; pass-through cost
|
||||
/// is zero in normal builds (`#[inline]`, newtype over std's cell).
|
||||
|
||||
+37
-1
@@ -141,6 +141,14 @@ impl PartialOrd for Entry {
|
||||
|
||||
#[derive(Default)]
|
||||
pub struct Timers {
|
||||
/// RFC 018: the scheduler coordination layer. Attached once at
|
||||
/// `RuntimeInner::new`; every insert notes its deadline (min-maintained
|
||||
/// snapshot for the busy-path due-check + the timekeeper re-arm wake)
|
||||
/// and every pop/clear re-anchors the snapshot to the heap minimum.
|
||||
/// All calls happen under the timers mutex — the serialization the
|
||||
/// coordinator's timer protocol mandates. `None` only in unit tests
|
||||
/// that construct a bare `Timers`.
|
||||
coord: Option<std::sync::Arc<crate::park::Coordinator>>,
|
||||
/// Reverse-wrapped so the smallest deadline is at the top.
|
||||
heap: BinaryHeap<Reverse<Entry>>,
|
||||
/// Monotonic counter for the tiebreaker `seq` field (and the `TimerId` of a
|
||||
@@ -157,7 +165,18 @@ pub struct Timers {
|
||||
|
||||
impl Timers {
|
||||
pub fn new() -> Self {
|
||||
Self { heap: BinaryHeap::new(), next_seq: 0, armed: std::collections::HashSet::new() }
|
||||
Self {
|
||||
coord: None,
|
||||
heap: BinaryHeap::new(),
|
||||
next_seq: 0,
|
||||
armed: std::collections::HashSet::new(),
|
||||
}
|
||||
}
|
||||
|
||||
/// Attach the scheduler coordination layer (RFC 018). Called once, at
|
||||
/// runtime construction, before any scheduler thread exists.
|
||||
pub(crate) fn attach_coordinator(&mut self, c: std::sync::Arc<crate::park::Coordinator>) {
|
||||
self.coord = Some(c);
|
||||
}
|
||||
|
||||
/// Insert a `Sleep` timer. Convenience for the common case.
|
||||
@@ -242,6 +261,13 @@ impl Timers {
|
||||
#[cfg(feature = "smarm-causal")]
|
||||
wall,
|
||||
}));
|
||||
// RFC 018: publish the (possibly new-minimum) deadline to the
|
||||
// busy-path snapshot and wake the timekeeper if it is parked
|
||||
// toward a later one. We hold the timers mutex — the mandated
|
||||
// serialization for both.
|
||||
if let Some(c) = &self.coord {
|
||||
c.note_deadline(deadline);
|
||||
}
|
||||
seq
|
||||
}
|
||||
|
||||
@@ -255,6 +281,9 @@ impl Timers {
|
||||
pub fn clear(&mut self) {
|
||||
self.heap.clear();
|
||||
self.armed.clear();
|
||||
if let Some(c) = &self.coord {
|
||||
c.refresh_deadline(None);
|
||||
}
|
||||
}
|
||||
|
||||
/// Soonest pending deadline, or `None` if the heap is empty.
|
||||
@@ -324,6 +353,13 @@ impl Timers {
|
||||
}
|
||||
out.push(entry);
|
||||
}
|
||||
// RFC 018: re-anchor the busy-path snapshot to the new heap minimum
|
||||
// (still under the timers mutex). A causal-shift re-queue above went
|
||||
// through `heap.push` directly, so this peek is the one place the
|
||||
// snapshot is guaranteed to catch up.
|
||||
if let Some(c) = &self.coord {
|
||||
c.refresh_deadline(self.peek_deadline());
|
||||
}
|
||||
out
|
||||
}
|
||||
}
|
||||
|
||||
+5
-5
@@ -23,7 +23,7 @@ extern "C-unwind" fn actor_simple() {
|
||||
#[test]
|
||||
fn actor_runs_and_returns_to_scheduler() {
|
||||
reset_log();
|
||||
let stack = Stack::new(64 * 1024).unwrap();
|
||||
let stack = Stack::new(64 * 1024, 4096).unwrap();
|
||||
let sp = init_actor_stack(stack.top(), actor_simple);
|
||||
set_actor_sp(sp);
|
||||
unsafe { switch_to_actor() };
|
||||
@@ -40,7 +40,7 @@ extern "C-unwind" fn actor_two_steps() {
|
||||
#[test]
|
||||
fn actor_yields_and_resumes() {
|
||||
reset_log();
|
||||
let stack = Stack::new(64 * 1024).unwrap();
|
||||
let stack = Stack::new(64 * 1024, 4096).unwrap();
|
||||
let sp = init_actor_stack(stack.top(), actor_two_steps);
|
||||
set_actor_sp(sp);
|
||||
|
||||
@@ -85,7 +85,7 @@ extern "C-unwind" fn actor_reg_check() {
|
||||
|
||||
#[test]
|
||||
fn callee_saved_registers_survive_yield() {
|
||||
let stack = Stack::new(64 * 1024).unwrap();
|
||||
let stack = Stack::new(64 * 1024, 4096).unwrap();
|
||||
let sp = init_actor_stack(stack.top(), actor_reg_check);
|
||||
set_actor_sp(sp);
|
||||
unsafe { switch_to_actor(); switch_to_actor(); }
|
||||
@@ -117,8 +117,8 @@ extern "C-unwind" fn actor_b() {
|
||||
|
||||
#[test]
|
||||
fn two_actors_dont_corrupt_each_other() {
|
||||
let stack_a = Stack::new(64 * 1024).unwrap();
|
||||
let stack_b = Stack::new(64 * 1024).unwrap();
|
||||
let stack_a = Stack::new(64 * 1024, 4096).unwrap();
|
||||
let stack_b = Stack::new(64 * 1024, 4096).unwrap();
|
||||
|
||||
let sp_a = init_actor_stack(stack_a.top(), actor_a);
|
||||
let sp_b = init_actor_stack(stack_b.top(), actor_b);
|
||||
|
||||
@@ -237,6 +237,13 @@ fn tree_from_nests_children_and_reroots_orphans() {
|
||||
overruns: 0,
|
||||
messages_received: 0,
|
||||
budget_cycles: 0,
|
||||
stack: smarm::StackInfo {
|
||||
reserve: 0,
|
||||
guard: 0,
|
||||
depth_high_water: 0,
|
||||
parks_since_shrink: 0,
|
||||
shrinks: 0,
|
||||
},
|
||||
};
|
||||
|
||||
let snap = RuntimeSnapshot {
|
||||
@@ -352,3 +359,122 @@ fn budget_cycles_accumulate_when_enabled() {
|
||||
h.join().unwrap();
|
||||
});
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// RFC 019 §8 — the stack introspection surface.
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Burn ~`frames` × 4 KiB of stack with a yield at max depth, so the context
|
||||
/// save samples the high-water there (RFC 019 §2: hwm is SAMPLED at
|
||||
/// deschedule, not tracked continuously).
|
||||
#[inline(never)]
|
||||
fn burn_stack_yielding(frames: usize) -> u64 {
|
||||
let mut local = [0u8; 4096];
|
||||
local[0] = frames as u8;
|
||||
let below = if frames == 0 {
|
||||
smarm::yield_now();
|
||||
0
|
||||
} else {
|
||||
burn_stack_yielding(frames - 1)
|
||||
};
|
||||
std::hint::black_box(&mut local);
|
||||
below.wrapping_add(local[0] as u64)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn stack_info_reports_defaults_and_sampled_depth() {
|
||||
run(|| {
|
||||
let (ready_tx, ready_rx) = channel::<()>();
|
||||
let (gate_tx, gate_rx) = channel::<()>();
|
||||
|
||||
let h = spawn(move || {
|
||||
// ~32 KiB deep with a yield at the bottom: the sample point.
|
||||
std::hint::black_box(burn_stack_yielding(8));
|
||||
ready_tx.send(()).unwrap();
|
||||
gate_rx.recv().unwrap();
|
||||
});
|
||||
ready_rx.recv().unwrap();
|
||||
|
||||
let info = spin_until(h.pid(), |a| a.state == ActorState::Parked);
|
||||
let s = info.stack;
|
||||
assert_eq!(s.reserve, 64 * 1024, "default reserve");
|
||||
assert_eq!(s.guard, 1024 * 1024, "default guard (kernel stack_guard_gap convention)");
|
||||
assert!(
|
||||
s.depth_high_water >= 8 * 4096,
|
||||
"hwm sampled at the deep yield: expected ≥ 32 KiB, got {}",
|
||||
s.depth_high_water
|
||||
);
|
||||
assert!(
|
||||
s.depth_high_water < s.reserve,
|
||||
"depth {} cannot exceed the reserve {}",
|
||||
s.depth_high_water,
|
||||
s.reserve
|
||||
);
|
||||
// Parked at the gate right now, never shrunk (64 KiB reserve cannot
|
||||
// cross the shrink threshold).
|
||||
assert!(s.parks_since_shrink >= 1, "the gate park must be counted");
|
||||
assert_eq!(s.shrinks, 0);
|
||||
|
||||
gate_tx.send(()).unwrap();
|
||||
h.join().unwrap();
|
||||
});
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn stack_info_shrink_counters_are_live() {
|
||||
use smarm::runtime::{Config, SHRINK_COOLDOWN, SHRINK_THRESHOLD};
|
||||
use smarm::{spawn_with, SpawnOpts};
|
||||
|
||||
let rt = smarm::runtime::init(Config::exact(1));
|
||||
rt.run(|| {
|
||||
let (park_tx, park_rx) = channel::<()>();
|
||||
|
||||
let spike = 768 * 4096;
|
||||
assert!(spike > SHRINK_THRESHOLD);
|
||||
let worker = spawn_with(
|
||||
SpawnOpts { stack_reserve: Some(8 * 1024 * 1024), ..SpawnOpts::default() },
|
||||
move || {
|
||||
std::hint::black_box(burn_stack_yielding(768));
|
||||
for _ in 0..(SHRINK_COOLDOWN + 8) {
|
||||
park_rx.recv().unwrap();
|
||||
}
|
||||
},
|
||||
);
|
||||
|
||||
let wpid = worker.pid();
|
||||
// Before any parks complete: the spike depth is visible.
|
||||
let info = spin_until(wpid, |a| a.state == ActorState::Parked);
|
||||
assert!(
|
||||
info.stack.depth_high_water >= spike,
|
||||
"spike should be sampled: {} < {spike}",
|
||||
info.stack.depth_high_water
|
||||
);
|
||||
|
||||
// Cross the cooldown, then read the counters live while the worker
|
||||
// is parked waiting for the remaining rounds (post-join the slot is
|
||||
// reclaimed and the generation check correctly hides it).
|
||||
for _ in 0..(SHRINK_COOLDOWN + 2) {
|
||||
spin_until(wpid, |a| a.state == ActorState::Parked);
|
||||
park_tx.send(()).unwrap();
|
||||
}
|
||||
let info = spin_until(wpid, |a| a.state == ActorState::Parked && a.stack.shrinks >= 1);
|
||||
let s = info.stack;
|
||||
assert!(s.shrinks >= 1, "cooldown was crossed with a spike above threshold");
|
||||
assert!(
|
||||
s.parks_since_shrink < SHRINK_COOLDOWN,
|
||||
"counter must reset at shrink: {}",
|
||||
s.parks_since_shrink
|
||||
);
|
||||
assert!(
|
||||
s.depth_high_water < spike,
|
||||
"hwm resets to the shallow park sp at shrink; got {}",
|
||||
s.depth_high_water
|
||||
);
|
||||
|
||||
for _ in 0..6 {
|
||||
spin_until(wpid, |a| a.state == ActorState::Parked);
|
||||
park_tx.send(()).unwrap();
|
||||
}
|
||||
worker.join().unwrap();
|
||||
});
|
||||
}
|
||||
|
||||
@@ -0,0 +1,69 @@
|
||||
//! RFC 018 scheduler park/wake — observable-behavior guards.
|
||||
//!
|
||||
//! These pin the two timer-latency properties the park/wake swap must
|
||||
//! preserve or introduce:
|
||||
//!
|
||||
//! - `sleep_fires_under_saturation`: due timers fire even when every
|
||||
//! scheduler is busy (nobody parked ⇒ no timekeeper) — the busy-path
|
||||
//! due-check, ratified design point (a). The old drain phase gave this
|
||||
//! for free (timers drained every loop iteration); the new design must
|
||||
//! not lose it.
|
||||
//! - `submillisecond_sleep_is_prompt`: a sub-ms sleep completes promptly.
|
||||
//! Under the old wake pipe, `poll_wake`'s `as_millis` truncation turned
|
||||
//! sub-ms deadlines into 0ms busy-polls (correct wall time, pathological
|
||||
//! CPU); under park/wake the futex timespec carries full nanosecond
|
||||
//! precision.
|
||||
|
||||
use std::sync::atomic::{AtomicBool, Ordering};
|
||||
use std::sync::Arc;
|
||||
use std::time::{Duration, Instant};
|
||||
|
||||
#[test]
|
||||
fn sleep_fires_under_saturation() {
|
||||
let rt = smarm::runtime::init(smarm::runtime::Config::exact(4));
|
||||
rt.run(|| {
|
||||
let stop = Arc::new(AtomicBool::new(false));
|
||||
let mut spinners = Vec::new();
|
||||
// 8 spinners over 4 schedulers: the run queue never empties, so no
|
||||
// scheduler ever parks and no timekeeper exists. Only the busy-path
|
||||
// due-check can fire the sleeper's timer before the spinners quit.
|
||||
for _ in 0..8 {
|
||||
let stop = stop.clone();
|
||||
spinners.push(smarm::spawn(move || {
|
||||
let t0 = Instant::now();
|
||||
while !stop.load(Ordering::Relaxed) && t0.elapsed() < Duration::from_secs(5) {
|
||||
smarm::yield_now();
|
||||
}
|
||||
}));
|
||||
}
|
||||
let t0 = Instant::now();
|
||||
smarm::sleep(Duration::from_millis(10));
|
||||
let dt = t0.elapsed();
|
||||
stop.store(true, Ordering::Relaxed);
|
||||
for s in spinners {
|
||||
let _ = s.join();
|
||||
}
|
||||
assert!(
|
||||
dt < Duration::from_millis(500),
|
||||
"10ms sleep took {dt:?} under scheduler saturation — busy-path \
|
||||
timer firing is broken (timekeeper-only firing stalls under load)"
|
||||
);
|
||||
});
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn submillisecond_sleep_is_prompt() {
|
||||
let rt = smarm::runtime::init(smarm::runtime::Config::exact(2));
|
||||
rt.run(|| {
|
||||
// Warm one iteration, then measure.
|
||||
smarm::sleep(Duration::from_micros(500));
|
||||
let t0 = Instant::now();
|
||||
smarm::sleep(Duration::from_micros(500));
|
||||
let dt = t0.elapsed();
|
||||
assert!(dt >= Duration::from_micros(400), "woke early: {dt:?}");
|
||||
assert!(
|
||||
dt < Duration::from_millis(100),
|
||||
"500µs sleep took {dt:?} — sub-ms deadline handling is broken"
|
||||
);
|
||||
});
|
||||
}
|
||||
@@ -517,3 +517,37 @@ fn runtime_reusable_after_root_panic() {
|
||||
r.run(move || ran_t.store(true, Ordering::Relaxed));
|
||||
assert!(ran.load(Ordering::Relaxed), "runtime unusable after root panic");
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// RFC 019 — Config stack knobs
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Burn ~`frames` × 4 KiB of stack; probestack touches pages in order so
|
||||
/// exceeding the reserve would hit the guard and SIGSEGV the process.
|
||||
#[inline(never)]
|
||||
fn burn_stack(frames: usize) -> u64 {
|
||||
let mut local = [0u8; 4096];
|
||||
local[0] = frames as u8;
|
||||
let below = if frames == 0 { 0 } else { burn_stack(frames - 1) };
|
||||
std::hint::black_box(&mut local);
|
||||
below.wrapping_add(local[0] as u64)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn config_stack_reserve_permits_deep_recursion() {
|
||||
// ~256 KiB of frames: four times the old fixed 64 KiB reserve. With
|
||||
// Config::stack_reserve raised this must complete; before RFC 019 it
|
||||
// could only segfault.
|
||||
let rt = smarm::runtime::init(Config::exact(1).stack_reserve(1024 * 1024));
|
||||
let done = Arc::new(AtomicBool::new(false));
|
||||
let done2 = done.clone();
|
||||
rt.run(move || {
|
||||
spawn(move || {
|
||||
std::hint::black_box(burn_stack(64));
|
||||
done2.store(true, Ordering::SeqCst);
|
||||
})
|
||||
.join()
|
||||
.unwrap();
|
||||
});
|
||||
assert!(done.load(Ordering::SeqCst));
|
||||
}
|
||||
|
||||
@@ -0,0 +1,212 @@
|
||||
//! RFC 019 commit 2 — the `SpawnOpts` surface.
|
||||
//!
|
||||
//! Covers: per-spawn stack shape overrides on every spawn surface, the
|
||||
//! `None ⇒ Config default` resolution, the pool rule from the outside
|
||||
//! (obligation 4: a custom-shaped stack never enters the pool), and that a
|
||||
//! big reserve behaviorally takes effect (deep recursion completes).
|
||||
|
||||
use smarm::runtime::{Config, DEFAULT_STACK_GUARD, DEFAULT_STACK_RESERVE};
|
||||
use smarm::{
|
||||
self_pid, spawn, spawn_under_with, spawn_with, GenServerBuilder, SpawnOpts,
|
||||
};
|
||||
use std::sync::atomic::{AtomicBool, Ordering};
|
||||
use std::sync::Arc;
|
||||
|
||||
fn rt1() -> smarm::runtime::Runtime {
|
||||
smarm::runtime::init(Config::exact(1))
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn default_spawn_has_default_shape() {
|
||||
rt1().run(|| {
|
||||
let h = spawn(|| {
|
||||
let shape = smarm::introspect::stack_shape(self_pid()).unwrap();
|
||||
assert_eq!(shape, (DEFAULT_STACK_RESERVE, DEFAULT_STACK_GUARD));
|
||||
});
|
||||
h.join().unwrap();
|
||||
});
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn spawn_with_overrides_reserve_and_guard() {
|
||||
rt1().run(|| {
|
||||
let opts = SpawnOpts {
|
||||
stack_reserve: Some(1024 * 1024),
|
||||
guard_size: Some(256 * 1024),
|
||||
};
|
||||
let h = spawn_with(opts, || {
|
||||
let shape = smarm::introspect::stack_shape(self_pid()).unwrap();
|
||||
assert_eq!(shape, (1024 * 1024, 256 * 1024));
|
||||
});
|
||||
h.join().unwrap();
|
||||
});
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn spawn_with_partial_override_keeps_config_default_for_the_rest() {
|
||||
rt1().run(|| {
|
||||
let opts = SpawnOpts { stack_reserve: Some(1024 * 1024), ..SpawnOpts::default() };
|
||||
let h = spawn_with(opts, || {
|
||||
let shape = smarm::introspect::stack_shape(self_pid()).unwrap();
|
||||
assert_eq!(shape, (1024 * 1024, DEFAULT_STACK_GUARD));
|
||||
});
|
||||
h.join().unwrap();
|
||||
});
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn spawn_with_rounds_to_pages() {
|
||||
rt1().run(|| {
|
||||
let opts = SpawnOpts { stack_reserve: Some(64 * 1024 + 1), guard_size: Some(4097) };
|
||||
let h = spawn_with(opts, || {
|
||||
let (reserve, guard) = smarm::introspect::stack_shape(self_pid()).unwrap();
|
||||
assert_eq!(reserve % 4096, 0);
|
||||
assert_eq!(guard % 4096, 0);
|
||||
assert!(reserve >= 64 * 1024 + 1);
|
||||
assert!(guard >= 4097);
|
||||
});
|
||||
h.join().unwrap();
|
||||
});
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn spawn_under_with_takes_opts() {
|
||||
rt1().run(|| {
|
||||
let me = self_pid();
|
||||
let opts = SpawnOpts { stack_reserve: Some(128 * 1024), ..SpawnOpts::default() };
|
||||
let h = spawn_under_with(me, opts, || {
|
||||
let (reserve, _) = smarm::introspect::stack_shape(self_pid()).unwrap();
|
||||
assert_eq!(reserve, 128 * 1024);
|
||||
});
|
||||
h.join().unwrap();
|
||||
});
|
||||
}
|
||||
|
||||
/// Obligation 4, from the outside: a dead custom stack must not be handed to
|
||||
/// the next default spawn. The pool is LIFO, so if the custom stack had been
|
||||
/// (wrongly) pushed at death, the very next default-shaped spawn on this
|
||||
/// single-threaded runtime would pop it and report a custom shape.
|
||||
#[test]
|
||||
fn custom_stack_never_enters_the_pool() {
|
||||
rt1().run(|| {
|
||||
spawn_with(
|
||||
SpawnOpts { stack_reserve: Some(512 * 1024), guard_size: Some(128 * 1024) },
|
||||
|| {},
|
||||
)
|
||||
.join()
|
||||
.unwrap();
|
||||
let h = spawn(|| {
|
||||
let shape = smarm::introspect::stack_shape(self_pid()).unwrap();
|
||||
assert_eq!(shape, (DEFAULT_STACK_RESERVE, DEFAULT_STACK_GUARD));
|
||||
});
|
||||
h.join().unwrap();
|
||||
});
|
||||
}
|
||||
|
||||
/// The reverse direction of the pool rule: a default-shaped stack IS pooled
|
||||
/// and reused (cap = threads × 4 ≥ 1 here, pool empty at start).
|
||||
#[test]
|
||||
fn default_stack_is_recycled() {
|
||||
rt1().run(|| {
|
||||
spawn(|| {}).join().unwrap();
|
||||
let h = spawn(|| {
|
||||
let shape = smarm::introspect::stack_shape(self_pid()).unwrap();
|
||||
assert_eq!(shape, (DEFAULT_STACK_RESERVE, DEFAULT_STACK_GUARD));
|
||||
});
|
||||
h.join().unwrap();
|
||||
});
|
||||
}
|
||||
|
||||
/// Burn ~`frames` × 4 KiB of stack (see tests/runtime.rs twin).
|
||||
#[inline(never)]
|
||||
fn burn_stack(frames: usize) -> u64 {
|
||||
let mut local = [0u8; 4096];
|
||||
local[0] = frames as u8;
|
||||
let below = if frames == 0 { 0 } else { burn_stack(frames - 1) };
|
||||
std::hint::black_box(&mut local);
|
||||
below.wrapping_add(local[0] as u64)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn big_reserve_behaviorally_takes_effect() {
|
||||
// ~1 MiB deep on an 8 MiB per-spawn reserve, runtime default untouched.
|
||||
rt1().run(|| {
|
||||
let done = Arc::new(AtomicBool::new(false));
|
||||
let done2 = done.clone();
|
||||
spawn_with(
|
||||
SpawnOpts { stack_reserve: Some(8 * 1024 * 1024), ..SpawnOpts::default() },
|
||||
move || {
|
||||
std::hint::black_box(burn_stack(256));
|
||||
done2.store(true, Ordering::SeqCst);
|
||||
},
|
||||
)
|
||||
.join()
|
||||
.unwrap();
|
||||
assert!(done.load(Ordering::SeqCst));
|
||||
});
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Builder surfaces
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
struct Echo;
|
||||
impl smarm::GenServer for Echo {
|
||||
type Call = ();
|
||||
type Reply = (usize, usize);
|
||||
type Cast = ();
|
||||
type Info = ();
|
||||
type Timer = ();
|
||||
fn handle_call(&mut self, _c: ()) -> (usize, usize) {
|
||||
smarm::introspect::stack_shape(self_pid()).unwrap()
|
||||
}
|
||||
fn handle_cast(&mut self, _c: ()) {}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn gen_server_builder_stack_opts() {
|
||||
rt1().run(|| {
|
||||
let server = GenServerBuilder::new(Echo)
|
||||
.stack_opts(SpawnOpts { stack_reserve: Some(256 * 1024), ..SpawnOpts::default() })
|
||||
.start();
|
||||
let (reserve, guard) = server.call(()).unwrap();
|
||||
assert_eq!(reserve, 256 * 1024);
|
||||
assert_eq!(guard, DEFAULT_STACK_GUARD);
|
||||
server.shutdown();
|
||||
});
|
||||
}
|
||||
|
||||
struct Probe;
|
||||
impl smarm::Machine for Probe {
|
||||
type Ev = smarm::channel::Sender<(usize, usize)>;
|
||||
fn state_timeout_ev() -> Self::Ev {
|
||||
unreachable!("no timers in this test")
|
||||
}
|
||||
fn timeout_ev(_name: &'static str) -> Self::Ev {
|
||||
unreachable!("no timers in this test")
|
||||
}
|
||||
fn on_start(&mut self, _cx: &mut smarm::Cx<Self::Ev>) {}
|
||||
fn handle(
|
||||
&mut self,
|
||||
ev: Self::Ev,
|
||||
_cx: &mut smarm::Cx<Self::Ev>,
|
||||
) -> smarm::gen_statem::Step<Self::Ev> {
|
||||
let _ = ev.send(smarm::introspect::stack_shape(self_pid()).unwrap());
|
||||
smarm::gen_statem::Step::Stayed
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn gen_statem_spawn_with_stack_opts() {
|
||||
rt1().run(|| {
|
||||
let m = smarm::gen_statem::spawn_with(
|
||||
SpawnOpts { stack_reserve: Some(256 * 1024), ..SpawnOpts::default() },
|
||||
Probe,
|
||||
);
|
||||
let (tx, rx) = smarm::channel::channel();
|
||||
m.send(tx).unwrap();
|
||||
let (reserve, guard) = rx.recv().unwrap();
|
||||
assert_eq!(reserve, 256 * 1024);
|
||||
assert_eq!(guard, DEFAULT_STACK_GUARD);
|
||||
});
|
||||
}
|
||||
+77
-9
@@ -7,13 +7,13 @@ use smarm::stack::Stack;
|
||||
|
||||
#[test]
|
||||
fn top_is_16_byte_aligned() {
|
||||
let s = Stack::new(64 * 1024).unwrap();
|
||||
let s = Stack::new(64 * 1024, 4096).unwrap();
|
||||
assert_eq!(s.top() as usize % 16, 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn top_is_within_allocation() {
|
||||
let s = Stack::new(64 * 1024).unwrap();
|
||||
let s = Stack::new(64 * 1024, 4096).unwrap();
|
||||
let top = s.top() as usize;
|
||||
let base = s.usable_base() as usize;
|
||||
assert!(top > base);
|
||||
@@ -22,7 +22,7 @@ fn top_is_within_allocation() {
|
||||
|
||||
#[test]
|
||||
fn write_and_read_top_of_stack() {
|
||||
let s = Stack::new(64 * 1024).unwrap();
|
||||
let s = Stack::new(64 * 1024, 4096).unwrap();
|
||||
let sentinel: u64 = 0xDEAD_BEEF_CAFE_1234;
|
||||
unsafe {
|
||||
let ptr = s.top().sub(8) as *mut u64;
|
||||
@@ -33,7 +33,7 @@ fn write_and_read_top_of_stack() {
|
||||
|
||||
#[test]
|
||||
fn write_and_read_bottom_of_usable_region() {
|
||||
let s = Stack::new(64 * 1024).unwrap();
|
||||
let s = Stack::new(64 * 1024, 4096).unwrap();
|
||||
let sentinel: u64 = 0x0102_0304_0506_0708;
|
||||
unsafe {
|
||||
let ptr = s.usable_base() as *mut u64;
|
||||
@@ -44,17 +44,17 @@ fn write_and_read_bottom_of_usable_region() {
|
||||
|
||||
#[test]
|
||||
fn small_stack_allocates() {
|
||||
assert!(Stack::new(4096).is_ok());
|
||||
assert!(Stack::new(4096, 4096).is_ok());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn large_stack_allocates() {
|
||||
assert!(Stack::new(8 * 1024 * 1024).is_ok());
|
||||
assert!(Stack::new(8 * 1024 * 1024, 4096).is_ok());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn stack_size_at_least_requested() {
|
||||
let s = Stack::new(64 * 1024).unwrap();
|
||||
let s = Stack::new(64 * 1024, 4096).unwrap();
|
||||
assert!(s.stack_size() >= 64 * 1024);
|
||||
}
|
||||
|
||||
@@ -68,15 +68,28 @@ use std::process::Command;
|
||||
fn run_as_child_if_requested() {
|
||||
match env::var("SMARM_SUBTEST").as_deref() {
|
||||
Ok("guard_page_direct") => {
|
||||
let s = Stack::new(64 * 1024).unwrap();
|
||||
let s = Stack::new(64 * 1024, 4096).unwrap();
|
||||
unsafe {
|
||||
let guard_ptr = s.usable_base().sub(1);
|
||||
guard_ptr.write_volatile(0xAB);
|
||||
}
|
||||
std::process::exit(0);
|
||||
}
|
||||
Ok("wide_guard_top") => {
|
||||
// One byte below the usable region, 64 KiB guard: must fault.
|
||||
let s = Stack::new(64 * 1024, 64 * 1024).unwrap();
|
||||
unsafe { s.usable_base().sub(1).write_volatile(0xAB); }
|
||||
std::process::exit(0);
|
||||
}
|
||||
Ok("wide_guard_bottom") => {
|
||||
// The very bottom page of a 64 KiB guard: an unprobed C-style
|
||||
// leap over a small guard lands here — must still fault.
|
||||
let s = Stack::new(64 * 1024, 64 * 1024).unwrap();
|
||||
unsafe { s.usable_base().sub(64 * 1024).write_volatile(0xAB); }
|
||||
std::process::exit(0);
|
||||
}
|
||||
Ok("stack_overflow") => {
|
||||
let s = Stack::new(64 * 1024).unwrap();
|
||||
let s = Stack::new(64 * 1024, 4096).unwrap();
|
||||
unsafe {
|
||||
let mut ptr = s.top().sub(1);
|
||||
let stop = s.usable_base().sub(1);
|
||||
@@ -121,3 +134,58 @@ fn stack_overflow_causes_sigsegv() {
|
||||
assert_eq!(status.signal(), Some(11), "expected SIGSEGV, got: {:?}", status);
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// RFC 019 — explicit shape: rounding, guard accessor, wide-guard coverage.
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
#[test]
|
||||
fn sizes_round_up_to_page() {
|
||||
let s = Stack::new(64 * 1024 + 1, 4096 + 1).unwrap();
|
||||
assert_eq!(s.stack_size() % 4096, 0);
|
||||
assert_eq!(s.guard_size() % 4096, 0);
|
||||
assert!(s.stack_size() >= 64 * 1024 + 1);
|
||||
assert!(s.guard_size() >= 4096 + 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn shape_reports_rounded_sizes() {
|
||||
let s = Stack::new(64 * 1024, 64 * 1024).unwrap();
|
||||
assert_eq!(s.shape(), (64 * 1024, 64 * 1024));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn usable_base_sits_above_guard() {
|
||||
let s = Stack::new(64 * 1024, 64 * 1024).unwrap();
|
||||
// The usable region must start exactly guard_size above the mapping
|
||||
// base: a write at usable_base is legal, one byte below is not (the
|
||||
// subprocess tests below prove the "not").
|
||||
let sentinel: u64 = 0x1111_2222_3333_4444;
|
||||
unsafe {
|
||||
let ptr = s.usable_base() as *mut u64;
|
||||
ptr.write_volatile(sentinel);
|
||||
assert_eq!(ptr.read_volatile(), sentinel);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn wide_guard_faults_at_top() {
|
||||
run_as_child_if_requested();
|
||||
let status = spawn_subtest("wide_guard_top");
|
||||
#[cfg(unix)]
|
||||
{
|
||||
use std::os::unix::process::ExitStatusExt;
|
||||
assert_eq!(status.signal(), Some(11), "expected SIGSEGV, got: {:?}", status);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn wide_guard_faults_at_bottom() {
|
||||
run_as_child_if_requested();
|
||||
let status = spawn_subtest("wide_guard_bottom");
|
||||
#[cfg(unix)]
|
||||
{
|
||||
use std::os::unix::process::ExitStatusExt;
|
||||
assert_eq!(status.signal(), Some(11), "expected SIGSEGV, got: {:?}", status);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,141 @@
|
||||
//! RFC 019 §7 — overflow diagnostics, observed from outside via subprocess
|
||||
//! (mirrors tests/stack.rs's harness, plus stderr capture).
|
||||
//!
|
||||
//! Four cases:
|
||||
//! - Rust recursion at defaults: probed frames walk into the guard →
|
||||
//! tier-1 definitive message, death by SIGSEGV.
|
||||
//! - FFI canary (96 KiB unprobed C local) at defaults: first touch lands
|
||||
//! inside the 1 MiB guard → tier-1 message.
|
||||
//! - FFI canary with the guard shrunk to 4 KiB: the frame steps over it
|
||||
//! into unmapped VA below → tier-2 "stepped over" message. This is the
|
||||
//! RFC's motivating incident (cargo-vendored gz build) reproduced.
|
||||
//! - FFI canary with reserve raised to 256 KiB: fits, runs clean, exits 0 —
|
||||
//! the §1 knob is the fix, proven by the same frame.
|
||||
|
||||
use std::env;
|
||||
use std::process::Command;
|
||||
|
||||
unsafe extern "C" {
|
||||
fn smarm_canary_burn();
|
||||
}
|
||||
|
||||
/// Unbounded probed recursion; each frame dirties 4 KiB. black_box defeats
|
||||
/// tail-call elision so the walk is real.
|
||||
#[inline(never)]
|
||||
#[allow(unconditional_recursion)]
|
||||
fn recurse_forever(depth: u64) -> u64 {
|
||||
let mut local = [0u8; 4096];
|
||||
local[0] = depth as u8;
|
||||
std::hint::black_box(&mut local);
|
||||
recurse_forever(depth + 1).wrapping_add(local[0] as u64)
|
||||
}
|
||||
|
||||
fn run_as_child_if_requested() {
|
||||
let mode = match env::var("SMARM_DIAG_SUBTEST") {
|
||||
Ok(m) => m,
|
||||
Err(_) => return,
|
||||
};
|
||||
use smarm::runtime::Config;
|
||||
use smarm::{spawn_with, SpawnOpts};
|
||||
let rt = smarm::runtime::init(Config::exact(1));
|
||||
rt.run(move || {
|
||||
let opts = match mode.as_str() {
|
||||
"rust_overflow" | "ffi_tier1" => SpawnOpts::default(),
|
||||
// Small guard: the canary's 96 KiB displacement clears it.
|
||||
"ffi_tier2" => SpawnOpts { guard_size: Some(4096), ..SpawnOpts::default() },
|
||||
// Enough reserve: the same frame simply fits.
|
||||
"ffi_clean" => SpawnOpts { stack_reserve: Some(256 * 1024), ..SpawnOpts::default() },
|
||||
other => panic!("unknown subtest {other}"),
|
||||
};
|
||||
let is_rust = mode == "rust_overflow";
|
||||
spawn_with(opts, move || {
|
||||
if is_rust {
|
||||
std::hint::black_box(recurse_forever(0));
|
||||
} else {
|
||||
unsafe { smarm_canary_burn() };
|
||||
}
|
||||
})
|
||||
.join()
|
||||
.unwrap();
|
||||
});
|
||||
std::process::exit(0);
|
||||
}
|
||||
|
||||
fn spawn_subtest(name: &str) -> std::process::Output {
|
||||
let exe = env::current_exe().unwrap();
|
||||
Command::new(exe)
|
||||
.env("SMARM_DIAG_SUBTEST", name)
|
||||
.args(["--test-threads=1", "--quiet"])
|
||||
.output()
|
||||
.expect("failed to spawn subprocess")
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
fn assert_died_sigsegv(out: &std::process::Output) {
|
||||
use std::os::unix::process::ExitStatusExt;
|
||||
assert_eq!(
|
||||
out.status.signal(),
|
||||
Some(11),
|
||||
"expected death by SIGSEGV, got {:?}; stderr:\n{}",
|
||||
out.status,
|
||||
String::from_utf8_lossy(&out.stderr)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn rust_overflow_dies_with_tier1_message() {
|
||||
run_as_child_if_requested();
|
||||
let out = spawn_subtest("rust_overflow");
|
||||
assert_died_sigsegv(&out);
|
||||
let err = String::from_utf8_lossy(&out.stderr);
|
||||
assert!(
|
||||
err.contains("overflowed its stack") && err.contains("in the guard region"),
|
||||
"missing tier-1 diagnostic; stderr:\n{err}"
|
||||
);
|
||||
assert!(err.contains("reserve=65536"), "wrong reserve in message:\n{err}");
|
||||
assert!(err.contains("guard=1048576"), "wrong guard in message:\n{err}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ffi_canary_at_defaults_dies_with_tier1_message() {
|
||||
run_as_child_if_requested();
|
||||
let out = spawn_subtest("ffi_tier1");
|
||||
assert_died_sigsegv(&out);
|
||||
let err = String::from_utf8_lossy(&out.stderr);
|
||||
// 96 KiB displacement from a 64 KiB reserve lands ~32 KiB into the
|
||||
// 1 MiB guard: definitively classified.
|
||||
assert!(
|
||||
err.contains("in the guard region"),
|
||||
"wide guard should catch the unprobed frame in tier 1; stderr:\n{err}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ffi_canary_over_small_guard_dies_with_tier2_message() {
|
||||
run_as_child_if_requested();
|
||||
let out = spawn_subtest("ffi_tier2");
|
||||
assert_died_sigsegv(&out);
|
||||
let err = String::from_utf8_lossy(&out.stderr);
|
||||
assert!(
|
||||
err.contains("stepped over it") && err.contains("below the guard"),
|
||||
"expected tier-2 overshoot attribution; stderr:\n{err}"
|
||||
);
|
||||
assert!(err.contains("guard=4096"), "wrong guard in message:\n{err}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ffi_canary_with_enough_reserve_runs_clean() {
|
||||
run_as_child_if_requested();
|
||||
let out = spawn_subtest("ffi_clean");
|
||||
assert!(
|
||||
out.status.success(),
|
||||
"canary should fit in 256 KiB reserve, got {:?}; stderr:\n{}",
|
||||
out.status,
|
||||
String::from_utf8_lossy(&out.stderr)
|
||||
);
|
||||
let err = String::from_utf8_lossy(&out.stderr);
|
||||
assert!(
|
||||
!err.contains("smarm: actor"),
|
||||
"no diagnostic expected on the clean path; stderr:\n{err}"
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,123 @@
|
||||
//! RFC 019 commit 5 — pool recycle zaps a dead stack down to its retained
|
||||
//! entry end, observed from the outside.
|
||||
//!
|
||||
//! A default-shaped stack that spiked deep and then died must not carry its
|
||||
//! spike into the pool as resident RSS: `recycle_stack` DONTNEEDs everything
|
||||
//! below the top `RECYCLE_RETAIN` bytes before pushing. The zap is
|
||||
//! synchronous on the death path, so the drop is immediate — but the death
|
||||
//! path itself races the observer's `join` return, hence the brief poll.
|
||||
//!
|
||||
//! Residency is measured with `mincore`, not smaps: a neighboring rw anon
|
||||
//! mapping can land flush against the stack top and the kernel merges the
|
||||
//! VMAs (observed under the full test run), so per-mapping smaps fields
|
||||
//! over-count. The PROT_NONE guard below can never merge, so the usable
|
||||
//! base is exactly the anchor VMA's start, and `mincore` counts pages
|
||||
//! within [usable_base, usable_base + reserve) regardless of merging.
|
||||
|
||||
use smarm::runtime::{Config, RECYCLE_RETAIN};
|
||||
use smarm::{channel, spawn, yield_now};
|
||||
|
||||
const RESERVE: usize = 4 * 1024 * 1024;
|
||||
|
||||
/// Burn ~`frames` × 4 KiB of stack, dirtying every frame.
|
||||
#[inline(never)]
|
||||
fn burn_stack(frames: usize) -> u64 {
|
||||
let mut local = [0u8; 4096];
|
||||
local[0] = frames as u8;
|
||||
let below = if frames == 0 { 0 } else { burn_stack(frames - 1) };
|
||||
std::hint::black_box(&mut local);
|
||||
below.wrapping_add(local[0] as u64)
|
||||
}
|
||||
|
||||
/// Resident-page count over [lo, lo + len) via mincore (len page-aligned).
|
||||
fn resident_pages(lo: usize, len: usize) -> usize {
|
||||
let page = 4096;
|
||||
let mut vec = vec![0u8; len / page];
|
||||
let ret = unsafe {
|
||||
libc::mincore(lo as *mut libc::c_void, len, vec.as_mut_ptr())
|
||||
};
|
||||
assert_eq!(ret, 0, "mincore failed: {}", std::io::Error::last_os_error());
|
||||
vec.iter().filter(|&&b| b & 1 != 0).count()
|
||||
}
|
||||
|
||||
/// The [start, end) of the VMA containing `addr`.
|
||||
fn vma_containing(addr: usize) -> (usize, usize) {
|
||||
let maps = std::fs::read_to_string("/proc/self/maps").unwrap();
|
||||
for line in maps.lines() {
|
||||
if let Some((range, _)) = line.split_once(' ') {
|
||||
if let Some((a, b)) = range.split_once('-') {
|
||||
if let (Ok(start), Ok(end)) =
|
||||
(usize::from_str_radix(a, 16), usize::from_str_radix(b, 16))
|
||||
{
|
||||
if start <= addr && addr < end {
|
||||
return (start, end);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
panic!("no VMA contains {addr:#x}");
|
||||
}
|
||||
|
||||
fn vma_exists(addr: usize) -> bool {
|
||||
let maps = std::fs::read_to_string("/proc/self/maps").unwrap();
|
||||
for line in maps.lines() {
|
||||
if let Some((range, _)) = line.split_once(' ') {
|
||||
if let Some((a, b)) = range.split_once('-') {
|
||||
if let (Ok(start), Ok(end)) =
|
||||
(usize::from_str_radix(a, 16), usize::from_str_radix(b, 16))
|
||||
{
|
||||
if start <= addr && addr < end {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
false
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn recycle_zaps_dead_stack_down_to_retain() {
|
||||
// Default reserve raised so the pool holds big stacks (default-shaped ⇒
|
||||
// pooled) and the zap has something to bite; single scheduler.
|
||||
let rt = smarm::runtime::init(Config::exact(1).stack_reserve(RESERVE));
|
||||
rt.run(|| {
|
||||
let (tx, rx) = channel::<usize>();
|
||||
|
||||
let h = spawn(move || {
|
||||
let probe = 0u8;
|
||||
let anchor = &probe as *const u8 as usize;
|
||||
// The guard below is PROT_NONE and can never merge with the
|
||||
// usable region, so the anchor VMA's start IS the usable base.
|
||||
let (vlo, _) = vma_containing(anchor);
|
||||
// Dirty ~3 MiB of the 4 MiB reserve, then die.
|
||||
std::hint::black_box(burn_stack(768));
|
||||
tx.send(vlo).unwrap();
|
||||
});
|
||||
|
||||
let usable_base = rx.recv().unwrap();
|
||||
h.join().unwrap();
|
||||
|
||||
// The zap span is everything below the retained entry end. DONTNEED
|
||||
// on private anon discards synchronously and unconditionally, so
|
||||
// this must go to exactly zero resident pages; the poll only covers
|
||||
// the death path racing join's return.
|
||||
let zap_len = RESERVE - RECYCLE_RETAIN;
|
||||
let mut resident = usize::MAX;
|
||||
for _ in 0..10_000 {
|
||||
resident = resident_pages(usable_base, zap_len);
|
||||
if resident == 0 {
|
||||
break;
|
||||
}
|
||||
yield_now();
|
||||
}
|
||||
assert_eq!(
|
||||
resident, 0,
|
||||
"recycled stack's zap span still resident: {resident} pages in \
|
||||
[{usable_base:#x}, +{zap_len:#x})"
|
||||
);
|
||||
// Pooled, not munmapped: the mapping must still be there.
|
||||
assert!(vma_exists(usable_base), "default-shaped stack was unmapped instead of pooled");
|
||||
});
|
||||
}
|
||||
@@ -0,0 +1,152 @@
|
||||
//! RFC 019 commit 3 — park-path stack shrink, observed from the outside.
|
||||
//!
|
||||
//! The one integration-level claim of the shrink machinery: an actor that
|
||||
//! spikes deep, returns shallow, and then parks past the cooldown gets its
|
||||
//! dead span MADV_FREE'd — visible as `LazyFree` in `/proc/self/smaps`
|
||||
//! within the stack's address range — while everything live survives.
|
||||
//!
|
||||
//! The high-water mark is *sampled* at context-save, so the spike yields
|
||||
//! once at max depth to guarantee a sample there (in production, preemption
|
||||
//! provides the quasi-random samples; a test must not rely on luck).
|
||||
|
||||
use smarm::runtime::{Config, SHRINK_COOLDOWN, SHRINK_THRESHOLD};
|
||||
use smarm::{actor_info, channel, spawn, spawn_with, yield_now, ActorState, SpawnOpts};
|
||||
|
||||
/// Burn ~`frames` × 4 KiB of stack, yielding once at the bottom so the
|
||||
/// context-save samples `sp` at max depth.
|
||||
#[inline(never)]
|
||||
fn burn_stack_yielding(frames: usize) -> u64 {
|
||||
let mut local = [0u8; 4096];
|
||||
local[0] = frames as u8;
|
||||
let below = if frames == 0 {
|
||||
yield_now();
|
||||
0
|
||||
} else {
|
||||
burn_stack_yielding(frames - 1)
|
||||
};
|
||||
std::hint::black_box(&mut local);
|
||||
below.wrapping_add(local[0] as u64)
|
||||
}
|
||||
|
||||
/// Sum the `LazyFree:` kB of every smaps mapping intersecting [lo, hi).
|
||||
fn lazy_free_bytes_in(lo: usize, hi: usize) -> usize {
|
||||
let smaps = std::fs::read_to_string("/proc/self/smaps").unwrap();
|
||||
let mut total_kb = 0usize;
|
||||
let mut in_range = false;
|
||||
for line in smaps.lines() {
|
||||
if let Some((range, _)) = line.split_once(' ') {
|
||||
if let Some((a, b)) = range.split_once('-') {
|
||||
if let (Ok(start), Ok(end)) =
|
||||
(usize::from_str_radix(a, 16), usize::from_str_radix(b, 16))
|
||||
{
|
||||
in_range = start < hi && end > lo;
|
||||
continue;
|
||||
}
|
||||
}
|
||||
}
|
||||
if in_range {
|
||||
if let Some(rest) = line.strip_prefix("LazyFree:") {
|
||||
let kb: usize = rest.trim().trim_end_matches(" kB").trim().parse().unwrap();
|
||||
total_kb += kb;
|
||||
}
|
||||
}
|
||||
}
|
||||
total_kb * 1024
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn spike_then_parks_marks_lazyfree_and_keeps_live_data() {
|
||||
// Single scheduler: the controller can gate on the worker being Parked.
|
||||
let rt = smarm::runtime::init(Config::exact(1));
|
||||
rt.run(|| {
|
||||
let (park_tx, park_rx) = channel::<()>();
|
||||
let (done_tx, done_rx) = channel::<(usize, u64)>();
|
||||
|
||||
let spike = 768 * 4096; // ~3 MiB, well past SHRINK_THRESHOLD
|
||||
assert!(spike > SHRINK_THRESHOLD);
|
||||
|
||||
let worker = spawn_with(
|
||||
SpawnOpts { stack_reserve: Some(8 * 1024 * 1024), ..SpawnOpts::default() },
|
||||
move || {
|
||||
// Live data that must survive the shrink, and an anchor
|
||||
// address inside the stack for the smaps scan.
|
||||
let live = [0xA5u8; 64];
|
||||
let anchor = live.as_ptr() as usize;
|
||||
|
||||
// Spike: ~3 MiB deep, sampled at the bottom, unwound.
|
||||
std::hint::black_box(burn_stack_yielding(768));
|
||||
|
||||
// Park past the cooldown. Each recv on the drained inbox is
|
||||
// one park; the controller sends only when it sees us Parked.
|
||||
for _ in 0..(SHRINK_COOLDOWN + 8) {
|
||||
park_rx.recv().unwrap();
|
||||
}
|
||||
|
||||
// Measure from inside: the stack spans ≤ 8 MiB below anchor.
|
||||
let lazy = lazy_free_bytes_in(anchor - 8 * 1024 * 1024, anchor + 4096);
|
||||
let checksum = live.iter().map(|&b| b as u64).sum();
|
||||
done_tx.send((lazy, checksum)).unwrap();
|
||||
},
|
||||
);
|
||||
|
||||
let wpid = worker.pid();
|
||||
for _ in 0..(SHRINK_COOLDOWN + 8) {
|
||||
// Gate: send only once the worker is genuinely parked so every
|
||||
// round is a real park-on-empty-mailbox.
|
||||
loop {
|
||||
match actor_info(wpid) {
|
||||
Some(info) if info.state == ActorState::Parked => break,
|
||||
Some(_) => yield_now(),
|
||||
None => panic!("worker died early"),
|
||||
}
|
||||
}
|
||||
park_tx.send(()).unwrap();
|
||||
}
|
||||
|
||||
let (lazy, checksum) = done_rx.recv().unwrap();
|
||||
// The spike was ~3 MiB; demand at least 2 MiB marked to leave slack
|
||||
// for the redzone, rounding, and pages the unwind re-dirtied.
|
||||
assert!(
|
||||
lazy >= 2 * 1024 * 1024,
|
||||
"expected ≥ 2 MiB LazyFree in the stack range, got {} bytes",
|
||||
lazy
|
||||
);
|
||||
assert_eq!(checksum, 64 * 0xA5u64, "live stack data corrupted by shrink");
|
||||
worker.join().unwrap();
|
||||
});
|
||||
}
|
||||
|
||||
/// Steady-state actors must never pay the syscall: an actor that parks a lot
|
||||
/// but never spikes past the threshold ends with zero LazyFree in its stack.
|
||||
#[test]
|
||||
fn shallow_actor_never_shrinks() {
|
||||
let rt = smarm::runtime::init(Config::exact(1));
|
||||
rt.run(|| {
|
||||
let (park_tx, park_rx) = channel::<()>();
|
||||
let (done_tx, done_rx) = channel::<usize>();
|
||||
|
||||
let worker = spawn(move || {
|
||||
let probe = 0u8;
|
||||
let anchor = &probe as *const u8 as usize;
|
||||
for _ in 0..(SHRINK_COOLDOWN + 8) {
|
||||
park_rx.recv().unwrap();
|
||||
}
|
||||
done_tx.send(lazy_free_bytes_in(anchor - 64 * 1024, anchor + 4096)).unwrap();
|
||||
});
|
||||
|
||||
let wpid = worker.pid();
|
||||
for _ in 0..(SHRINK_COOLDOWN + 8) {
|
||||
loop {
|
||||
match actor_info(wpid) {
|
||||
Some(info) if info.state == ActorState::Parked => break,
|
||||
Some(_) => yield_now(),
|
||||
None => panic!("worker died early"),
|
||||
}
|
||||
}
|
||||
park_tx.send(()).unwrap();
|
||||
}
|
||||
|
||||
assert_eq!(done_rx.recv().unwrap(), 0, "steady-state actor was shrunk");
|
||||
worker.join().unwrap();
|
||||
});
|
||||
}
|
||||
Reference in New Issue
Block a user