Compare commits
12
Commits
8c764e9169
..
v0.6.0
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
301e3463e3 | ||
|
|
410ba33d82 | ||
|
|
5fd8aecf55 | ||
|
|
7d8b9e0310 | ||
|
|
8225716b11 | ||
|
|
3cb64eefc2 | ||
|
|
0fe052bc7e | ||
|
|
a03a7ca01e | ||
|
|
d4839f1d81 | ||
|
|
2854b560d6 | ||
|
|
7b026cfe56 | ||
|
|
006a3283e7 |
@@ -2,7 +2,23 @@
|
|||||||
# smarm pre-commit gate: clippy the library (src/) with warnings as errors.
|
# smarm pre-commit gate: clippy the library (src/) with warnings as errors.
|
||||||
# unwrap_used / expect_used are denied (Cargo.toml [lints.clippy]): library
|
# unwrap_used / expect_used are denied (Cargo.toml [lints.clippy]): library
|
||||||
# code must not hide a panic behind unwrap/expect. Tests/examples are not gated.
|
# code must not hide a panic behind unwrap/expect. Tests/examples are not gated.
|
||||||
|
#
|
||||||
|
# Toolchain resolution: prefer an installed cargo-clippy; on machines whose
|
||||||
|
# rust comes without the clippy component (e.g. NixOS home-manager), fall
|
||||||
|
# back to an ephemeral nix-shell toolchain. The fallback uses its own target
|
||||||
|
# dir (target/clippy) because the shell's rustc version may differ from the
|
||||||
|
# default toolchain's — mixed-compiler artifacts in one target dir are an
|
||||||
|
# E0514 hard error. MSRV (Cargo.toml rust-version) keeps the older shell
|
||||||
|
# toolchain a legitimate gate.
|
||||||
set -eu
|
set -eu
|
||||||
[ -f "$HOME/.cargo/env" ] && . "$HOME/.cargo/env"
|
[ -f "$HOME/.cargo/env" ] && . "$HOME/.cargo/env"
|
||||||
cd "$(git rev-parse --show-toplevel)"
|
cd "$(git rev-parse --show-toplevel)"
|
||||||
|
if cargo clippy --version >/dev/null 2>&1; then
|
||||||
cargo clippy --lib -- -D warnings
|
cargo clippy --lib -- -D warnings
|
||||||
|
elif command -v nix-shell >/dev/null 2>&1; then
|
||||||
|
nix-shell -p clippy -p cargo -p rustc \
|
||||||
|
--run 'CARGO_TARGET_DIR=target/clippy cargo clippy --lib -- -D warnings'
|
||||||
|
else
|
||||||
|
echo "pre-commit: cargo clippy unavailable and no nix-shell fallback" >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|||||||
+4
-1
@@ -1,6 +1,6 @@
|
|||||||
[package]
|
[package]
|
||||||
name = "smarm"
|
name = "smarm"
|
||||||
version = "0.4.0"
|
version = "0.6.0"
|
||||||
edition = "2021"
|
edition = "2021"
|
||||||
rust-version = "1.95"
|
rust-version = "1.95"
|
||||||
|
|
||||||
@@ -39,6 +39,9 @@ rq-mutex = []
|
|||||||
rq-mpmc = []
|
rq-mpmc = []
|
||||||
rq-striped = []
|
rq-striped = []
|
||||||
|
|
||||||
|
[build-dependencies]
|
||||||
|
cc = "1"
|
||||||
|
|
||||||
[dependencies]
|
[dependencies]
|
||||||
libc = "0.2"
|
libc = "0.2"
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,11 @@
|
|||||||
|
fn main() {
|
||||||
|
// RFC 019 §7 test canary (agreed Q3): compiled without stack-clash
|
||||||
|
// protection so its 96 KiB local is a genuine one-displacement guard
|
||||||
|
// jumper; distro-hardened compilers would otherwise probe it page-wise
|
||||||
|
// and defeat the test's purpose.
|
||||||
|
cc::Build::new()
|
||||||
|
.file("canary/canary.c")
|
||||||
|
.flag_if_supported("-fno-stack-clash-protection")
|
||||||
|
.compile("smarm_canary");
|
||||||
|
println!("cargo:rerun-if-changed=canary/canary.c");
|
||||||
|
}
|
||||||
@@ -0,0 +1,14 @@
|
|||||||
|
/* RFC 019 §7 FFI canary: an honest unprobed C frame with a 96 KiB local,
|
||||||
|
* touched from its LOW end first — the exact "one sub rsp steps over a small
|
||||||
|
* guard" pattern the RFC's motivating incident hit (a cargo-vendored gz
|
||||||
|
* build; cc-invoked builds do not enable -fstack-clash-protection, and this
|
||||||
|
* file pins that off explicitly so the canary stays a canary even on
|
||||||
|
* hardened-default toolchains). */
|
||||||
|
void smarm_canary_burn(void) {
|
||||||
|
volatile char buf[96 * 1024];
|
||||||
|
buf[0] = 1; /* deepest address first */
|
||||||
|
for (unsigned i = 0; i < sizeof buf; i += 4096) {
|
||||||
|
buf[i] = (char)i;
|
||||||
|
}
|
||||||
|
buf[sizeof buf - 1] = 1;
|
||||||
|
}
|
||||||
@@ -75,7 +75,7 @@ genuine advantage over tokio's task abort model.
|
|||||||
|
|
||||||
### Spawn-heavy workloads (19–70×)
|
### Spawn-heavy workloads (19–70×)
|
||||||
|
|
||||||
Every smarm actor `mmap`s a 64 KiB stack with a guard page. This is
|
Every smarm actor `mmap`s a 64 KiB stack reserve with a 64 KiB PROT_NONE guard below (both per-actor configurable since RFC 019; the reserve is demand-paged). This is
|
||||||
a syscall. Tokio tasks are heap-allocated state machines — no stack,
|
a syscall. Tokio tasks are heap-allocated state machines — no stack,
|
||||||
no syscall, ~100 bytes each. For workloads that spawn thousands of
|
no syscall, ~100 bytes each. For workloads that spawn thousands of
|
||||||
short-lived actors per second, this is a structural disadvantage.
|
short-lived actors per second, this is a structural disadvantage.
|
||||||
|
|||||||
+31
-5
@@ -182,7 +182,7 @@ use crate::channel::{channel, select, select_timeout, Receiver, RecvTimeoutError
|
|||||||
use crate::monitor::{demonitor, monitor, Down, Monitor};
|
use crate::monitor::{demonitor, monitor, Down, Monitor};
|
||||||
use crate::pid::Pid;
|
use crate::pid::Pid;
|
||||||
use crate::registry::{register_with, resolve_named_sender, RegisterError};
|
use crate::registry::{register_with, resolve_named_sender, RegisterError};
|
||||||
use crate::scheduler::{cancel_timer, request_stop, send_after_to, spawn, spawn_under};
|
use crate::scheduler::{cancel_timer, request_stop, send_after_to};
|
||||||
use crate::timer::TimerId;
|
use crate::timer::TimerId;
|
||||||
use std::cell::Cell;
|
use std::cell::Cell;
|
||||||
use std::collections::HashMap;
|
use std::collections::HashMap;
|
||||||
@@ -643,11 +643,17 @@ pub struct GenServerBuilder<G: GenServer> {
|
|||||||
state: G,
|
state: G,
|
||||||
infos: Vec<Receiver<G::Info>>,
|
infos: Vec<Receiver<G::Info>>,
|
||||||
supervisor: Option<Pid>,
|
supervisor: Option<Pid>,
|
||||||
|
stack_opts: crate::scheduler::SpawnOpts,
|
||||||
}
|
}
|
||||||
|
|
||||||
impl<G: GenServer> GenServerBuilder<G> {
|
impl<G: GenServer> GenServerBuilder<G> {
|
||||||
pub fn new(state: G) -> Self {
|
pub fn new(state: G) -> Self {
|
||||||
GenServerBuilder { state, infos: Vec::new(), supervisor: None }
|
GenServerBuilder {
|
||||||
|
state,
|
||||||
|
infos: Vec::new(),
|
||||||
|
supervisor: None,
|
||||||
|
stack_opts: crate::scheduler::SpawnOpts::default(),
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Add an out-of-band channel; messages arriving on it are dispatched to
|
/// Add an out-of-band channel; messages arriving on it are dispatched to
|
||||||
@@ -665,6 +671,14 @@ impl<G: GenServer> GenServerBuilder<G> {
|
|||||||
self
|
self
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Stack shape for the server actor (RFC 019) — see
|
||||||
|
/// [`SpawnOpts`](crate::SpawnOpts). Useful for servers that recurse
|
||||||
|
/// deeply or call into FFI with large C frames.
|
||||||
|
pub fn stack_opts(mut self, opts: crate::scheduler::SpawnOpts) -> Self {
|
||||||
|
self.stack_opts = opts;
|
||||||
|
self
|
||||||
|
}
|
||||||
|
|
||||||
/// Spawn the server actor and hand back its [`GenServerRef`]. The server's
|
/// Spawn the server actor and hand back its [`GenServerRef`]. The server's
|
||||||
/// lifetime is governed by its refs, not by joining, so the backing join
|
/// lifetime is governed by its refs, not by joining, so the backing join
|
||||||
/// handle is dropped.
|
/// handle is dropped.
|
||||||
@@ -686,10 +700,16 @@ impl<G: GenServer> GenServerBuilder<G> {
|
|||||||
/// under the name before returning.
|
/// under the name before returning.
|
||||||
fn spawn_server(self) -> GenServerRef<G> {
|
fn spawn_server(self) -> GenServerRef<G> {
|
||||||
let (tx, rx) = channel::<Envelope<G>>();
|
let (tx, rx) = channel::<Envelope<G>>();
|
||||||
let GenServerBuilder { state, infos, supervisor } = self;
|
let GenServerBuilder { state, infos, supervisor, stack_opts } = self;
|
||||||
let handle = match supervisor {
|
let handle = match supervisor {
|
||||||
Some(sup) => spawn_under(sup, move || server_loop::<G>(rx, state, infos)),
|
Some(sup) => {
|
||||||
None => spawn(move || server_loop::<G>(rx, state, infos)),
|
crate::scheduler::spawn_under_with(sup, stack_opts, move || {
|
||||||
|
server_loop::<G>(rx, state, infos)
|
||||||
|
})
|
||||||
|
}
|
||||||
|
None => crate::scheduler::spawn_with(stack_opts, move || {
|
||||||
|
server_loop::<G>(rx, state, infos)
|
||||||
|
}),
|
||||||
};
|
};
|
||||||
GenServerRef { tx, pid: handle.pid() }
|
GenServerRef { tx, pid: handle.pid() }
|
||||||
}
|
}
|
||||||
@@ -758,6 +778,12 @@ impl<G: GenServer> NamedGenServerBuilder<G> {
|
|||||||
self
|
self
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Stack shape for the server actor (see [`GenServerBuilder::stack_opts`]).
|
||||||
|
pub fn stack_opts(mut self, opts: crate::scheduler::SpawnOpts) -> Self {
|
||||||
|
self.builder = self.builder.stack_opts(opts);
|
||||||
|
self
|
||||||
|
}
|
||||||
|
|
||||||
/// Spawn the server and bind its name in one step. Fallible: returns
|
/// Spawn the server and bind its name in one step. Fallible: returns
|
||||||
/// [`RegisterError::NameTaken`] if the name is already held by a different
|
/// [`RegisterError::NameTaken`] if the name is already held by a different
|
||||||
/// live server.
|
/// live server.
|
||||||
|
|||||||
+15
-2
@@ -71,7 +71,7 @@
|
|||||||
|
|
||||||
use crate::channel::{channel, select, Receiver, Sender};
|
use crate::channel::{channel, select, Receiver, Sender};
|
||||||
use crate::pid::Pid;
|
use crate::pid::Pid;
|
||||||
use crate::scheduler::{cancel_timer, send_after_to, spawn as spawn_actor};
|
use crate::scheduler::{cancel_timer, send_after_to};
|
||||||
use crate::timer::TimerId;
|
use crate::timer::TimerId;
|
||||||
use std::collections::{HashMap, VecDeque};
|
use std::collections::{HashMap, VecDeque};
|
||||||
use std::marker::PhantomData;
|
use std::marker::PhantomData;
|
||||||
@@ -434,8 +434,21 @@ impl<M: Machine> GenStatemRef<M> {
|
|||||||
///
|
///
|
||||||
/// Panics if called outside `Runtime::run()`.
|
/// Panics if called outside `Runtime::run()`.
|
||||||
pub fn spawn<M: Machine>(machine: M) -> GenStatemRef<M> {
|
pub fn spawn<M: Machine>(machine: M) -> GenStatemRef<M> {
|
||||||
|
spawn_with(crate::scheduler::SpawnOpts::default(), machine)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`spawn`] with per-actor stack shape overrides (RFC 019) for the machine's
|
||||||
|
/// actor — see [`SpawnOpts`](crate::SpawnOpts). gen_statem has no builder
|
||||||
|
/// (its one-shot `spawn(machine)` shape predates RFC 019), so the opts ride
|
||||||
|
/// a `_with` variant like the scheduler's own spawns.
|
||||||
|
///
|
||||||
|
/// Panics if called outside `Runtime::run()`.
|
||||||
|
pub fn spawn_with<M: Machine>(
|
||||||
|
opts: crate::scheduler::SpawnOpts,
|
||||||
|
machine: M,
|
||||||
|
) -> GenStatemRef<M> {
|
||||||
let (tx, rx) = channel::<M::Ev>();
|
let (tx, rx) = channel::<M::Ev>();
|
||||||
let handle = spawn_actor(move || statem_loop(rx, machine));
|
let handle = crate::scheduler::spawn_with(opts, move || statem_loop(rx, machine));
|
||||||
GenStatemRef { tx, pid: handle.pid() }
|
GenStatemRef { tx, pid: handle.pid() }
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -179,6 +179,34 @@ pub struct ActorInfo {
|
|||||||
/// `budget-accounting` feature is enabled, since measuring it costs a
|
/// `budget-accounting` feature is enabled, since measuring it costs a
|
||||||
/// timestamp read on every resume.
|
/// timestamp read on every resume.
|
||||||
pub budget_cycles: u64,
|
pub budget_cycles: u64,
|
||||||
|
/// RFC 019 §8 — this actor's stack, as the runtime sees it. All fields
|
||||||
|
/// are lock-free atomic reads, coherent for this incarnation via the
|
||||||
|
/// same generation check as the counters above. Exact RSS is
|
||||||
|
/// deliberately absent: `mincore` is debug tooling, never a runtime
|
||||||
|
/// path.
|
||||||
|
pub stack: StackInfo,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// RFC 019 §8 — per-actor stack introspection. Sizes are page-rounded, as
|
||||||
|
/// [`Stack::new`](crate::stack::Stack::new) rounds them.
|
||||||
|
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||||
|
pub struct StackInfo {
|
||||||
|
/// Usable stack size ([`SpawnOpts::stack_reserve`]
|
||||||
|
/// (crate::SpawnOpts::stack_reserve) or the Config/default).
|
||||||
|
pub reserve: usize,
|
||||||
|
/// PROT_NONE guard below the usable region.
|
||||||
|
pub guard: usize,
|
||||||
|
/// Sampled high-water depth in bytes: `top − lowest saved sp`. Sampled,
|
||||||
|
/// not exact — the context save at yields/parks/preemptions is the
|
||||||
|
/// sampler (RFC 019 §2), so a spike the actor never yielded inside is
|
||||||
|
/// invisible. 0 depth means "never descheduled at any depth", not
|
||||||
|
/// "never ran".
|
||||||
|
pub depth_high_water: usize,
|
||||||
|
/// Parks on this incarnation since its last shrink (or since install if
|
||||||
|
/// it has never shrunk) — the §3 cooldown counter, live.
|
||||||
|
pub parks_since_shrink: u32,
|
||||||
|
/// §3 shrinks performed on this incarnation.
|
||||||
|
pub shrinks: u32,
|
||||||
}
|
}
|
||||||
|
|
||||||
/// A snapshot of every actor in the runtime at (approximately) one moment.
|
/// A snapshot of every actor in the runtime at (approximately) one moment.
|
||||||
@@ -220,6 +248,22 @@ pub fn snapshot() -> RuntimeSnapshot {
|
|||||||
/// slot was reused by another), out of range, or was never a real pid at
|
/// slot was reused by another), out of range, or was never a real pid at
|
||||||
/// all. Unlike [`snapshot`], every field of the result describes the same
|
/// all. Unlike [`snapshot`], every field of the result describes the same
|
||||||
/// instant, since there is only one actor to read.
|
/// instant, since there is only one actor to read.
|
||||||
|
/// The stack shape `(reserve, guard)` of a live actor, page-rounded — the
|
||||||
|
/// RFC 019 introspection surface's first field (depth sampling and shrink
|
||||||
|
/// counters land with the shrink machinery). `None` if `pid` no longer names
|
||||||
|
/// a live actor. Takes the actor's cold lock briefly; debugging/assertion
|
||||||
|
/// use, not a hot-path call.
|
||||||
|
pub fn stack_shape(pid: Pid) -> Option<(usize, usize)> {
|
||||||
|
with_runtime(|inner| {
|
||||||
|
let slot = inner.slot_at(pid)?;
|
||||||
|
let cold = slot.cold.lock();
|
||||||
|
if slot.generation() != pid.generation() {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
cold.actor.as_ref().map(|a| a.stack.shape())
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
pub fn actor_info(pid: Pid) -> Option<ActorInfo> {
|
pub fn actor_info(pid: Pid) -> Option<ActorInfo> {
|
||||||
with_runtime(|inner| {
|
with_runtime(|inner| {
|
||||||
let slot = inner.slot_at(pid)?;
|
let slot = inner.slot_at(pid)?;
|
||||||
@@ -265,6 +309,14 @@ fn read_slot(slot: &Slot, idx: u32, mail: Option<&MailboxInfo>) -> Option<ActorI
|
|||||||
drop(cold);
|
drop(cold);
|
||||||
|
|
||||||
// Counters are plain atomics, read lock-free.
|
// Counters are plain atomics, read lock-free.
|
||||||
|
let (reserve, guard, top, hwm, parks_since_shrink, shrinks) = slot.stack_introspect();
|
||||||
|
let stack = StackInfo {
|
||||||
|
reserve,
|
||||||
|
guard,
|
||||||
|
depth_high_water: top.saturating_sub(hwm),
|
||||||
|
parks_since_shrink,
|
||||||
|
shrinks,
|
||||||
|
};
|
||||||
let overruns = slot.overruns();
|
let overruns = slot.overruns();
|
||||||
let messages_received = slot.messages_received();
|
let messages_received = slot.messages_received();
|
||||||
let budget_cycles = slot.budget_cycles();
|
let budget_cycles = slot.budget_cycles();
|
||||||
@@ -290,6 +342,7 @@ fn read_slot(slot: &Slot, idx: u32, mail: Option<&MailboxInfo>) -> Option<ActorI
|
|||||||
overruns,
|
overruns,
|
||||||
messages_received,
|
messages_received,
|
||||||
budget_cycles,
|
budget_cycles,
|
||||||
|
stack,
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -13,44 +13,68 @@
|
|||||||
//! leaves the actor, no copying through an intermediary thread. Built on
|
//! leaves the actor, no copying through an intermediary thread. Built on
|
||||||
//! these are the conveniences `read(fd, &mut buf)` and `write(fd, &buf)`.
|
//! these are the conveniences `read(fd, &mut buf)` and `write(fd, &buf)`.
|
||||||
//!
|
//!
|
||||||
//! Architecture
|
//! Architecture (RFC 018: driver-enqueues)
|
||||||
//! ============
|
//! =======================================
|
||||||
//! Per `run()`, two OS threads:
|
//! Per `run()`, two OS threads, each a *producer* behind the runtime's
|
||||||
//! - **epoll thread**: owns the epollfd. Loops in `epoll_wait`. On a
|
//! two-call contract — make the actor runnable (`unpark_at`, whose enqueue
|
||||||
//! ready fd, pushes `Completion::FdReady { pid, fd, events }` to the
|
//! tail wakes a parked scheduler), nothing else:
|
||||||
//! shared completion queue and writes the scheduler-wake pipe. On the
|
|
||||||
//! shutdown pipe (also registered in epollfd), exits.
|
|
||||||
//! - **pool thread**: blocks on the request mpsc. Runs the closure
|
|
||||||
//! inside `catch_unwind`, pushes `Completion::Blocking { pid, result }`,
|
|
||||||
//! writes the scheduler-wake pipe.
|
|
||||||
//!
|
//!
|
||||||
//! Both threads share a single `completions: Arc<Mutex<VecDeque<Completion>>>`
|
//! - **epoll thread**: owns `epoll_wait` on the epollfd. On a ready fd it
|
||||||
//! and the same scheduler-wake pipe.
|
//! removes the parked waiter from the shared `waiters` map and DELs the
|
||||||
|
//! fd (both under the waiters lock — see below), then unparks the
|
||||||
|
//! actor directly. On the shutdown pipe (also registered in the
|
||||||
|
//! epollfd), exits.
|
||||||
|
//! - **pool thread**: blocks on the request mpsc. Runs the closure inside
|
||||||
|
//! `catch_unwind`, stashes the result in the actor's slot
|
||||||
|
//! (`pending_io_result`, under the cold lock, generation-checked),
|
||||||
|
//! decrements the runtime's `io_outstanding`, and unparks the actor.
|
||||||
//!
|
//!
|
||||||
//! `epoll_ctl` (register/unregister fd interest) is called by the
|
//! There is no shared completion queue and no wake pipe: each producer
|
||||||
//! scheduler thread *directly* on the epollfd. That's well-defined per
|
//! routes its own completion, so the whole byte-vs-completion visibility
|
||||||
//! `epoll_ctl(2)`: a thread may be calling `epoll_wait` on the epollfd
|
//! discipline of the drain era — and the stranded-completion hazards it
|
||||||
//! while another thread calls `epoll_ctl`. Avoids needing a second mpsc
|
//! defended against — is unrepresentable. Producers reach the runtime
|
||||||
//! and a second wake mechanism.
|
//! through a `Weak<RuntimeInner>`: upgraded per completion (the path is
|
||||||
|
//! syscall-bound; the refcount op is noise) and avoiding an Arc cycle
|
||||||
|
//! through `RuntimeInner::io`.
|
||||||
|
//!
|
||||||
|
//! `epoll_ctl` (register fd interest) is called by the scheduler thread
|
||||||
|
//! directly on the epollfd. That's well-defined per `epoll_ctl(2)`: a
|
||||||
|
//! thread may be calling `epoll_wait` on the epollfd while another thread
|
||||||
|
//! calls `epoll_ctl`.
|
||||||
//!
|
//!
|
||||||
//! Epoll mode
|
//! Epoll mode
|
||||||
//! ==========
|
//! ==========
|
||||||
//! Level-triggered with EPOLLONESHOT. After a wakeup the kernel
|
//! Level-triggered with EPOLLONESHOT. After a wakeup the kernel
|
||||||
//! auto-disarms the fd, so we never get two wakeups for one
|
//! auto-disarms the fd, so we never get two wakeups for one
|
||||||
//! `wait_readable` call. The scheduler explicitly `EPOLL_CTL_DEL`s the fd
|
//! `wait_readable` call. The epoll thread explicitly `EPOLL_CTL_DEL`s the
|
||||||
//! on completion to free the slot for re-registration. Net effect: each
|
//! fd on readiness to free the slot for re-registration. Net effect: each
|
||||||
//! `wait_readable(fd)` is one ADD, one wakeup, one DEL — symmetric and
|
//! `wait_readable(fd)` is one ADD, one wakeup, one DEL — symmetric and
|
||||||
//! stateless between calls.
|
//! stateless between calls.
|
||||||
//!
|
//!
|
||||||
|
//! ## The waiters lock is the ADD/DEL serialization
|
||||||
|
//!
|
||||||
|
//! Registration (scheduler thread: check-vacant, defensive DEL, ADD,
|
||||||
|
//! insert) and readiness consumption (epoll thread: remove, DEL) each run
|
||||||
|
//! entirely under the `waiters` mutex. This is what makes the
|
||||||
|
//! oneshot-rearm race unrepresentable: a woken actor re-registering the
|
||||||
|
//! same fd cannot interleave with the epoll thread's DEL for the *previous*
|
||||||
|
//! registration — whichever takes the lock second sees a consistent
|
||||||
|
//! kernel-side state. Lock order: `io` (the runtime's outer mutex, held by
|
||||||
|
//! scheduler-side callers) → `waiters` → slot/queue leaves via `unpark_at`.
|
||||||
|
//! The epoll thread takes `waiters` without `io` — it must never take
|
||||||
|
//! `io`, both for lock-order hygiene and because teardown holds `io` while
|
||||||
|
//! joining it.
|
||||||
|
//!
|
||||||
//! Fd hygiene
|
//! Fd hygiene
|
||||||
//! ==========
|
//! ==========
|
||||||
//! An actor stopped while waiting on an fd unwinds out of `wait_fd`'s park;
|
//! An actor stopped while waiting on an fd unwinds out of `wait_fd`'s park;
|
||||||
//! a drop guard there (armed after a successful register, forgotten on a
|
//! a drop guard there (armed after a successful register, forgotten on a
|
||||||
//! normal wake) removes the `waiters` entry iff it is still that wait's
|
//! normal wake) calls [`IoThread::cancel_waiter`], which removes the
|
||||||
//! `(pid, epoch)` and only then `EPOLL_CTL_DEL`s the fd — an entry already
|
//! `waiters` entry iff it is still that wait's `(pid, epoch)` and only then
|
||||||
//! consumed by a racing `FdReady` means the fd may carry someone else's
|
//! `EPOLL_CTL_DEL`s the fd — an entry already consumed by the epoll thread
|
||||||
//! fresh registration, which must be left alone. `epoll_register` keeps a
|
//! means the fd may carry someone else's fresh registration, which must be
|
||||||
//! defensive bare DEL before ADD as belt-and-braces.
|
//! left alone. `epoll_register` keeps a defensive bare DEL before ADD as
|
||||||
|
//! belt-and-braces.
|
||||||
//!
|
//!
|
||||||
//! Buffers used with `read`/`write` should be on fds opened with
|
//! Buffers used with `read`/`write` should be on fds opened with
|
||||||
//! `O_NONBLOCK`. If they aren't, the syscall may block the scheduler
|
//! `O_NONBLOCK`. If they aren't, the syscall may block the scheduler
|
||||||
@@ -68,13 +92,14 @@
|
|||||||
//! they have no equivalent panic-propagation path.
|
//! they have no equivalent panic-propagation path.
|
||||||
|
|
||||||
use crate::pid::Pid;
|
use crate::pid::Pid;
|
||||||
|
use crate::runtime::RuntimeInner;
|
||||||
use std::any::Any;
|
use std::any::Any;
|
||||||
use std::collections::{HashMap, VecDeque};
|
use std::collections::HashMap;
|
||||||
use std::io;
|
use std::io;
|
||||||
use std::os::fd::RawFd;
|
use std::os::fd::RawFd;
|
||||||
use std::panic;
|
use std::panic;
|
||||||
use std::sync::mpsc;
|
use std::sync::atomic::Ordering;
|
||||||
use std::sync::{Arc, Mutex};
|
use std::sync::{mpsc, Arc, Mutex, Weak};
|
||||||
use std::thread::JoinHandle as OsJoinHandle;
|
use std::thread::JoinHandle as OsJoinHandle;
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
@@ -86,42 +111,29 @@ use std::thread::JoinHandle as OsJoinHandle;
|
|||||||
pub type IoResult = Result<Box<dyn Any + Send>, Box<dyn Any + Send>>;
|
pub type IoResult = Result<Box<dyn Any + Send>, Box<dyn Any + Send>>;
|
||||||
|
|
||||||
struct Request {
|
struct Request {
|
||||||
/// The submitter's park-epoch — carried through to the `Blocking`
|
/// The submitter's park-epoch — the eventual wake is epoch-matched.
|
||||||
/// completion so the wake is epoch-matched.
|
|
||||||
epoch: u32,
|
epoch: u32,
|
||||||
pid: Pid,
|
pid: Pid,
|
||||||
/// The work to perform. Returns the wire-form result directly.
|
/// The work to perform. Returns the wire-form result directly.
|
||||||
work: Box<dyn FnOnce() -> IoResult + Send>,
|
work: Box<dyn FnOnce() -> IoResult + Send>,
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Completion message from either IO thread back to the scheduler.
|
/// The parked-waiter map, shared between scheduler-side registration and
|
||||||
pub enum Completion {
|
/// the epoll thread's readiness consumption. See the module docs on why
|
||||||
/// A `block_on_io` closure has finished (Ok = return value, Err = panic
|
/// this single lock is the ADD/DEL serialization.
|
||||||
/// payload).
|
type Waiters = Arc<Mutex<HashMap<RawFd, (Pid, u32)>>>;
|
||||||
Blocking { pid: Pid, epoch: u32, result: IoResult },
|
|
||||||
/// An fd registered via `wait_readable`/`wait_writable` is ready. The
|
|
||||||
/// scheduler looks up the parked pid in `waiters`, unparks it, and
|
|
||||||
/// removes the entry. `pid` isn't in this variant because the epoll
|
|
||||||
/// thread doesn't have access to the `waiters` map; the scheduler
|
|
||||||
/// thread owns that.
|
|
||||||
FdReady { fd: RawFd, events: u32 },
|
|
||||||
}
|
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
// IoThread — created per `run()`, owned by `SchedulerState`.
|
// IoThread — created per `run()`, owned by `RuntimeInner::io`.
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
pub struct IoThread {
|
pub struct IoThread {
|
||||||
// ----- Channels & queues -----
|
|
||||||
|
|
||||||
/// Submission queue into the blocking-work pool.
|
/// Submission queue into the blocking-work pool.
|
||||||
tx: mpsc::Sender<Request>,
|
tx: mpsc::Sender<Request>,
|
||||||
/// Shared completion queue, fed by both the pool and the epoll thread.
|
/// One parked actor per registered fd. Populated by `epoll_register`,
|
||||||
completions: Arc<Mutex<VecDeque<Completion>>>,
|
/// consumed by the epoll thread on readiness or `cancel_waiter` on an
|
||||||
/// Pipe the scheduler polls in its idle path. Both IO threads write to
|
/// unwound wait.
|
||||||
/// `wake_write` after pushing a completion.
|
waiters: Waiters,
|
||||||
wake_read: RawFd,
|
|
||||||
wake_write: RawFd,
|
|
||||||
|
|
||||||
// ----- Epoll machinery -----
|
// ----- Epoll machinery -----
|
||||||
|
|
||||||
@@ -133,39 +145,25 @@ pub struct IoThread {
|
|||||||
/// shutdown.
|
/// shutdown.
|
||||||
shutdown_read: RawFd,
|
shutdown_read: RawFd,
|
||||||
shutdown_write: RawFd,
|
shutdown_write: RawFd,
|
||||||
/// One parked actor per registered fd. Populated by `wait_readable` /
|
|
||||||
/// `wait_writable` and drained by the scheduler when a `FdReady`
|
|
||||||
/// completion is processed.
|
|
||||||
pub waiters: HashMap<RawFd, (Pid, u32)>,
|
|
||||||
|
|
||||||
// ----- Threads -----
|
// ----- Threads -----
|
||||||
|
|
||||||
pool_thread: Option<OsJoinHandle<()>>,
|
pool_thread: Option<OsJoinHandle<()>>,
|
||||||
epoll_thread: Option<OsJoinHandle<()>>,
|
epoll_thread: Option<OsJoinHandle<()>>,
|
||||||
|
|
||||||
/// Number of `block_on_io` requests in-flight. Used by the scheduler's
|
|
||||||
/// idle path to decide whether to wait on the pipe or exit. Fd waits
|
|
||||||
/// are not counted here; they're counted by `waiters.len()`.
|
|
||||||
pub outstanding: u32,
|
|
||||||
}
|
}
|
||||||
|
|
||||||
impl IoThread {
|
impl IoThread {
|
||||||
pub fn start() -> io::Result<Self> {
|
/// Start the pool and epoll threads. `rt` is the producers' route back
|
||||||
// Scheduler-facing wake pipe.
|
/// into the runtime (slot table + unpark protocol); a `Weak` so the
|
||||||
let (wake_read, wake_write) = make_pipe()?;
|
/// `RuntimeInner → IoThread → RuntimeInner` cycle never forms.
|
||||||
// Pool submission channel + shared completion queue.
|
pub(crate) fn start(rt: Weak<RuntimeInner>) -> io::Result<Self> {
|
||||||
|
// Pool submission channel.
|
||||||
let (tx, rx) = mpsc::channel::<Request>();
|
let (tx, rx) = mpsc::channel::<Request>();
|
||||||
let completions: Arc<Mutex<VecDeque<Completion>>> =
|
let waiters: Waiters = Arc::new(Mutex::new(HashMap::new()));
|
||||||
Arc::new(Mutex::new(VecDeque::new()));
|
|
||||||
|
|
||||||
// Epoll machinery.
|
// Epoll machinery.
|
||||||
let epollfd = unsafe { libc::epoll_create1(libc::EPOLL_CLOEXEC) };
|
let epollfd = unsafe { libc::epoll_create1(libc::EPOLL_CLOEXEC) };
|
||||||
if epollfd < 0 {
|
if epollfd < 0 {
|
||||||
// Best-effort fd cleanup before bailing.
|
|
||||||
unsafe {
|
|
||||||
libc::close(wake_read);
|
|
||||||
libc::close(wake_write);
|
|
||||||
}
|
|
||||||
return Err(io::Error::last_os_error());
|
return Err(io::Error::last_os_error());
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -174,8 +172,6 @@ impl IoThread {
|
|||||||
Err(e) => {
|
Err(e) => {
|
||||||
unsafe {
|
unsafe {
|
||||||
libc::close(epollfd);
|
libc::close(epollfd);
|
||||||
libc::close(wake_read);
|
|
||||||
libc::close(wake_write);
|
|
||||||
}
|
}
|
||||||
return Err(e);
|
return Err(e);
|
||||||
}
|
}
|
||||||
@@ -202,42 +198,37 @@ impl IoThread {
|
|||||||
libc::close(epollfd);
|
libc::close(epollfd);
|
||||||
libc::close(shutdown_read);
|
libc::close(shutdown_read);
|
||||||
libc::close(shutdown_write);
|
libc::close(shutdown_write);
|
||||||
libc::close(wake_read);
|
|
||||||
libc::close(wake_write);
|
|
||||||
}
|
}
|
||||||
return Err(e);
|
return Err(e);
|
||||||
}
|
}
|
||||||
|
|
||||||
// Spawn pool thread.
|
// Spawn pool thread.
|
||||||
let pool_comps = completions.clone();
|
let pool_rt = rt.clone();
|
||||||
let pool_thread = std::thread::Builder::new()
|
let pool_thread = std::thread::Builder::new()
|
||||||
.name("smarm-io-pool".into())
|
.name("smarm-io-pool".into())
|
||||||
.spawn(move || pool_loop(rx, pool_comps, wake_write))?;
|
.spawn(move || pool_loop(rx, pool_rt))?;
|
||||||
|
|
||||||
// Spawn epoll thread.
|
// Spawn epoll thread.
|
||||||
let epoll_comps = completions.clone();
|
let epoll_waiters = waiters.clone();
|
||||||
let epoll_thread = std::thread::Builder::new()
|
let epoll_thread = std::thread::Builder::new()
|
||||||
.name("smarm-io-epoll".into())
|
.name("smarm-io-epoll".into())
|
||||||
.spawn(move || epoll_loop(epollfd, epoll_comps, wake_write))?;
|
.spawn(move || epoll_loop(epollfd, epoll_waiters, rt))?;
|
||||||
|
|
||||||
Ok(Self {
|
Ok(Self {
|
||||||
tx,
|
tx,
|
||||||
completions,
|
waiters,
|
||||||
wake_read,
|
|
||||||
wake_write,
|
|
||||||
epollfd,
|
epollfd,
|
||||||
shutdown_read,
|
shutdown_read,
|
||||||
shutdown_write,
|
shutdown_write,
|
||||||
waiters: HashMap::new(),
|
|
||||||
pool_thread: Some(pool_thread),
|
pool_thread: Some(pool_thread),
|
||||||
epoll_thread: Some(epoll_thread),
|
epoll_thread: Some(epoll_thread),
|
||||||
outstanding: 0,
|
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Hand a request to the pool. Increments `outstanding`.
|
/// Hand a request to the pool. The caller (scheduler.rs) increments
|
||||||
|
/// `io_outstanding` BEFORE calling — the pool decrements on completion,
|
||||||
|
/// and an increment that trailed the completion would underflow.
|
||||||
pub fn submit(&mut self, pid: Pid, epoch: u32, work: Box<dyn FnOnce() -> IoResult + Send>) {
|
pub fn submit(&mut self, pid: Pid, epoch: u32, work: Box<dyn FnOnce() -> IoResult + Send>) {
|
||||||
self.outstanding += 1;
|
|
||||||
// Send can only fail if the pool has hung up, which only happens
|
// Send can only fail if the pool has hung up, which only happens
|
||||||
// on shutdown. submit during shutdown is a bug.
|
// on shutdown. submit during shutdown is a bug.
|
||||||
if self.tx.send(Request { pid, epoch, work }).is_err() {
|
if self.tx.send(Request { pid, epoch, work }).is_err() {
|
||||||
@@ -245,39 +236,13 @@ impl IoThread {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Drain every available completion. Caller (the scheduler) routes the
|
|
||||||
/// results and updates `outstanding` / `waiters` accordingly.
|
|
||||||
pub fn drain_completions(&mut self) -> Vec<Completion> {
|
|
||||||
let mut q = match self.completions.lock() {
|
|
||||||
Ok(g) => g,
|
|
||||||
Err(e) => panic!("smarm: io completions lock poisoned (core corrupt): {e}"),
|
|
||||||
};
|
|
||||||
let mut out = Vec::with_capacity(q.len());
|
|
||||||
while let Some(c) = q.pop_front() {
|
|
||||||
out.push(c);
|
|
||||||
}
|
|
||||||
out
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn wake_fd(&self) -> RawFd {
|
|
||||||
self.wake_read
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Write the wake pipe directly: rouse every scheduler thread blocked in
|
|
||||||
/// its idle `poll_wake`. Used by the terminal (AllDone) path — an idle
|
|
||||||
/// sibling may be blocked on a snapshot that nothing will ever refresh
|
|
||||||
/// (an orphaned timer deadline, or `io_outstanding` from a waiter that
|
|
||||||
/// was stop-cancelled and so never produces a completion).
|
|
||||||
pub fn wake(&self) {
|
|
||||||
wake_scheduler(self.wake_write);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Register interest in `fd` becoming readable/writable; record `pid`
|
/// Register interest in `fd` becoming readable/writable; record `pid`
|
||||||
/// as the parked waiter. The epoll thread will push a `FdReady`
|
/// as the parked waiter. The epoll thread unparks it on readiness.
|
||||||
/// completion when the kernel signals.
|
/// The caller increments `io_fd_waiters` BEFORE calling (mirror of
|
||||||
|
/// `submit`'s contract) and decrements it again if this errors.
|
||||||
///
|
///
|
||||||
/// EPOLLONESHOT: one wakeup per registration. The scheduler must
|
/// EPOLLONESHOT: one wakeup per registration; the epoll thread DELs on
|
||||||
/// `epoll_del` on completion to free the slot for re-registration.
|
/// readiness, `cancel_waiter` DELs on an unwound wait.
|
||||||
pub fn epoll_register(
|
pub fn epoll_register(
|
||||||
&mut self,
|
&mut self,
|
||||||
fd: RawFd,
|
fd: RawFd,
|
||||||
@@ -286,20 +251,24 @@ impl IoThread {
|
|||||||
readable: bool,
|
readable: bool,
|
||||||
writable: bool,
|
writable: bool,
|
||||||
) -> io::Result<()> {
|
) -> io::Result<()> {
|
||||||
|
let mut waiters = match self.waiters.lock() {
|
||||||
|
Ok(g) => g,
|
||||||
|
Err(e) => panic!("smarm: io waiters lock poisoned (core corrupt): {e}"),
|
||||||
|
};
|
||||||
// Two actors waiting on the same fd would be a misuse: the kernel
|
// Two actors waiting on the same fd would be a misuse: the kernel
|
||||||
// delivers exactly one EPOLLONESHOT wakeup, so the second waiter
|
// delivers exactly one EPOLLONESHOT wakeup, so the second waiter
|
||||||
// would hang. Reject up front.
|
// would hang. Reject up front.
|
||||||
if self.waiters.contains_key(&fd) {
|
if waiters.contains_key(&fd) {
|
||||||
return Err(io::Error::new(
|
return Err(io::Error::new(
|
||||||
io::ErrorKind::AlreadyExists,
|
io::ErrorKind::AlreadyExists,
|
||||||
"fd already has a parked waiter",
|
"fd already has a parked waiter",
|
||||||
));
|
));
|
||||||
}
|
}
|
||||||
|
|
||||||
// Belt-and-braces: the unwind guard in `wait_fd` is responsible for
|
// Belt-and-braces: `cancel_waiter` is responsible for cleaning up a
|
||||||
// cleaning up a stopped waiter's registration, but a bare DEL is
|
// stopped waiter's registration, but a bare DEL is harmless if the
|
||||||
// harmless if the fd isn't registered (ENOENT) and removes any leak
|
// fd isn't registered (ENOENT) and removes any leak a path we
|
||||||
// a path we haven't thought of might leave behind.
|
// haven't thought of might leave behind.
|
||||||
unsafe {
|
unsafe {
|
||||||
libc::epoll_ctl(self.epollfd, libc::EPOLL_CTL_DEL, fd, std::ptr::null_mut());
|
libc::epoll_ctl(self.epollfd, libc::EPOLL_CTL_DEL, fd, std::ptr::null_mut());
|
||||||
}
|
}
|
||||||
@@ -321,20 +290,30 @@ impl IoThread {
|
|||||||
if r < 0 {
|
if r < 0 {
|
||||||
return Err(io::Error::last_os_error());
|
return Err(io::Error::last_os_error());
|
||||||
}
|
}
|
||||||
self.waiters.insert(fd, (pid, epoch));
|
waiters.insert(fd, (pid, epoch));
|
||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Remove `fd` from the epollfd. Called by the scheduler after a
|
/// Remove `fd`'s waiter iff it is still `(pid, epoch)`, DELing the fd
|
||||||
/// `FdReady` completion, so the next `wait_readable(fd)` can ADD again.
|
/// from the epollfd in the same critical section. Returns whether the
|
||||||
///
|
/// entry was removed (the caller then decrements `io_fd_waiters`).
|
||||||
/// Does NOT touch `waiters` — that's the scheduler's bookkeeping; this
|
/// `false` means the epoll thread consumed the registration first —
|
||||||
/// is purely the kernel-side cleanup.
|
/// the fd may already carry someone else's fresh ADD; hands off.
|
||||||
pub fn epoll_deregister(&mut self, fd: RawFd) {
|
pub fn cancel_waiter(&mut self, fd: RawFd, pid: Pid, epoch: u32) -> bool {
|
||||||
|
let mut waiters = match self.waiters.lock() {
|
||||||
|
Ok(g) => g,
|
||||||
|
Err(e) => panic!("smarm: io waiters lock poisoned (core corrupt): {e}"),
|
||||||
|
};
|
||||||
|
if waiters.get(&fd) == Some(&(pid, epoch)) {
|
||||||
|
waiters.remove(&fd);
|
||||||
// EPOLL_CTL_DEL of an already-removed fd returns ENOENT; ignore.
|
// EPOLL_CTL_DEL of an already-removed fd returns ENOENT; ignore.
|
||||||
unsafe {
|
unsafe {
|
||||||
libc::epoll_ctl(self.epollfd, libc::EPOLL_CTL_DEL, fd, std::ptr::null_mut());
|
libc::epoll_ctl(self.epollfd, libc::EPOLL_CTL_DEL, fd, std::ptr::null_mut());
|
||||||
}
|
}
|
||||||
|
true
|
||||||
|
} else {
|
||||||
|
false
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -354,7 +333,10 @@ impl Drop for IoThread {
|
|||||||
let real_tx = std::mem::replace(&mut self.tx, dead_tx);
|
let real_tx = std::mem::replace(&mut self.tx, dead_tx);
|
||||||
drop(real_tx);
|
drop(real_tx);
|
||||||
|
|
||||||
// 3. Join both threads.
|
// 3. Join both threads. Safe even while the caller holds the
|
||||||
|
// runtime's `io` mutex: neither thread ever takes it (they reach
|
||||||
|
// the runtime through a Weak they upgrade per completion, and
|
||||||
|
// the epoll thread's only lock is `waiters`).
|
||||||
if let Some(h) = self.epoll_thread.take() {
|
if let Some(h) = self.epoll_thread.take() {
|
||||||
let _ = h.join();
|
let _ = h.join();
|
||||||
}
|
}
|
||||||
@@ -367,8 +349,6 @@ impl Drop for IoThread {
|
|||||||
libc::close(self.epollfd);
|
libc::close(self.epollfd);
|
||||||
libc::close(self.shutdown_read);
|
libc::close(self.shutdown_read);
|
||||||
libc::close(self.shutdown_write);
|
libc::close(self.shutdown_write);
|
||||||
libc::close(self.wake_read);
|
|
||||||
libc::close(self.wake_write);
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -379,36 +359,38 @@ impl Drop for IoThread {
|
|||||||
const SHUTDOWN_EPOLL_TOKEN: u64 = u64::MAX;
|
const SHUTDOWN_EPOLL_TOKEN: u64 = u64::MAX;
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
// Pool loop
|
// Pool loop (producer: Blocking completions)
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
fn pool_loop(
|
fn pool_loop(rx: mpsc::Receiver<Request>, rt: Weak<RuntimeInner>) {
|
||||||
rx: mpsc::Receiver<Request>,
|
|
||||||
completions: Arc<Mutex<VecDeque<Completion>>>,
|
|
||||||
wake_write: RawFd,
|
|
||||||
) {
|
|
||||||
while let Ok(Request { pid, epoch, work }) = rx.recv() {
|
while let Ok(Request { pid, epoch, work }) = rx.recv() {
|
||||||
let result: IoResult = match panic::catch_unwind(panic::AssertUnwindSafe(work)) {
|
let result: IoResult = match panic::catch_unwind(panic::AssertUnwindSafe(work)) {
|
||||||
Ok(r) => r,
|
Ok(r) => r,
|
||||||
Err(payload) => Err(payload),
|
Err(payload) => Err(payload),
|
||||||
};
|
};
|
||||||
match completions.lock() {
|
let Some(inner) = rt.upgrade() else { return };
|
||||||
Ok(mut g) => g.push_back(Completion::Blocking { pid, epoch, result }),
|
// Stash the result under the cold lock (generation-checked: an
|
||||||
Err(e) => panic!("smarm: io completions lock poisoned (core corrupt): {e}"),
|
// actor stopped with the op in flight discards it), decrement the
|
||||||
|
// in-flight count, then wake through the epoch-matched unpark. The
|
||||||
|
// unpark's enqueue tail wakes a parked scheduler; the actor stays
|
||||||
|
// `live` until it resumes and finalizes, so the decrement's
|
||||||
|
// ordering against the termination verdict is not load-bearing.
|
||||||
|
if let Some(slot) = inner.slot_at(pid) {
|
||||||
|
let mut cold = slot.cold.lock();
|
||||||
|
if slot.generation() == pid.generation() {
|
||||||
|
cold.pending_io_result = Some(result);
|
||||||
}
|
}
|
||||||
wake_scheduler(wake_write);
|
}
|
||||||
|
inner.io_outstanding.fetch_sub(1, Ordering::AcqRel);
|
||||||
|
inner.unpark_at(pid, epoch);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
// Epoll loop
|
// Epoll loop (producer: FdReady completions)
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
fn epoll_loop(
|
fn epoll_loop(epollfd: RawFd, waiters: Waiters, rt: Weak<RuntimeInner>) {
|
||||||
epollfd: RawFd,
|
|
||||||
completions: Arc<Mutex<VecDeque<Completion>>>,
|
|
||||||
wake_write: RawFd,
|
|
||||||
) {
|
|
||||||
// Buffer for epoll_wait. 64 is plenty for our scale; if a real load
|
// Buffer for epoll_wait. 64 is plenty for our scale; if a real load
|
||||||
// appears that needs more, this is a one-line change.
|
// appears that needs more, this is a one-line change.
|
||||||
const MAX_EVENTS: usize = 64;
|
const MAX_EVENTS: usize = 64;
|
||||||
@@ -436,29 +418,41 @@ fn epoll_loop(
|
|||||||
}
|
}
|
||||||
|
|
||||||
let mut shutdown_requested = false;
|
let mut shutdown_requested = false;
|
||||||
let mut pushed_any = false;
|
|
||||||
{
|
|
||||||
let mut q = match completions.lock() {
|
|
||||||
Ok(g) => g,
|
|
||||||
Err(e) => panic!("smarm: io completions lock poisoned (core corrupt): {e}"),
|
|
||||||
};
|
|
||||||
for ev in events.iter().take(n as usize) {
|
for ev in events.iter().take(n as usize) {
|
||||||
if ev.u64 == SHUTDOWN_EPOLL_TOKEN {
|
if ev.u64 == SHUTDOWN_EPOLL_TOKEN {
|
||||||
shutdown_requested = true;
|
shutdown_requested = true;
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
let fd = ev.u64 as RawFd;
|
let fd = ev.u64 as RawFd;
|
||||||
let evs = ev.events;
|
// Consume the registration: remove + DEL under the waiters
|
||||||
q.push_back(Completion::FdReady {
|
// lock (the ADD/DEL serialization — see module docs). A
|
||||||
|
// vanished entry means `cancel_waiter` beat us: the wake is
|
||||||
|
// already moot.
|
||||||
|
let entry = {
|
||||||
|
let mut w = match waiters.lock() {
|
||||||
|
Ok(g) => g,
|
||||||
|
Err(e) => {
|
||||||
|
panic!("smarm: io waiters lock poisoned (core corrupt): {e}")
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let entry = w.remove(&fd);
|
||||||
|
if entry.is_some() {
|
||||||
|
unsafe {
|
||||||
|
libc::epoll_ctl(
|
||||||
|
epollfd,
|
||||||
|
libc::EPOLL_CTL_DEL,
|
||||||
fd,
|
fd,
|
||||||
events: evs,
|
std::ptr::null_mut(),
|
||||||
});
|
);
|
||||||
pushed_any = true;
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
entry
|
||||||
if pushed_any {
|
};
|
||||||
wake_scheduler(wake_write);
|
if let Some((pid, epoch)) = entry {
|
||||||
|
let Some(inner) = rt.upgrade() else { return };
|
||||||
|
inner.io_fd_waiters.fetch_sub(1, Ordering::AcqRel);
|
||||||
|
inner.unpark_at(pid, epoch);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
if shutdown_requested {
|
if shutdown_requested {
|
||||||
return;
|
return;
|
||||||
@@ -466,27 +460,8 @@ fn epoll_loop(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Write one byte to the scheduler's wake pipe. Retries on EINTR; ignores
|
|
||||||
/// EAGAIN (pipe full means there's already an outstanding wake we haven't
|
|
||||||
/// consumed yet, which is sufficient).
|
|
||||||
fn wake_scheduler(wake_write: RawFd) {
|
|
||||||
let buf: [u8; 1] = [0];
|
|
||||||
unsafe {
|
|
||||||
loop {
|
|
||||||
let n = libc::write(wake_write, buf.as_ptr() as *const _, 1);
|
|
||||||
if n < 0 {
|
|
||||||
let e = *libc::__errno_location();
|
|
||||||
if e == libc::EINTR {
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
// Pipe helpers (unchanged from v0.2)
|
// Pipe helper
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
fn make_pipe() -> io::Result<(RawFd, RawFd)> {
|
fn make_pipe() -> io::Result<(RawFd, RawFd)> {
|
||||||
@@ -497,50 +472,3 @@ fn make_pipe() -> io::Result<(RawFd, RawFd)> {
|
|||||||
}
|
}
|
||||||
Ok((fds[0], fds[1]))
|
Ok((fds[0], fds[1]))
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Drain pending bytes from the wake pipe. Nonblocking (pipe is O_NONBLOCK).
|
|
||||||
///
|
|
||||||
/// DISCIPLINE: called only by the phase-1 drain-lock winner, immediately
|
|
||||||
/// before `drain_completions`. Bytes are the notification channel for
|
|
||||||
/// completions; consuming one anywhere else can strand the completion it
|
|
||||||
/// announces (see the lost-wakeup note at the call site in `schedule_loop`).
|
|
||||||
pub fn drain_wake_pipe(fd: RawFd) {
|
|
||||||
let mut buf = [0u8; 64];
|
|
||||||
loop {
|
|
||||||
let n = unsafe { libc::read(fd, buf.as_mut_ptr() as *mut _, buf.len()) };
|
|
||||||
if n <= 0 {
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Block on `fd` for up to `timeout`, returning when either there's data
|
|
||||||
/// to read or the timeout elapses. `None` for `timeout` means wait forever.
|
|
||||||
pub fn poll_wake(fd: RawFd, timeout: Option<std::time::Duration>) {
|
|
||||||
let timeout_ms: libc::c_int = match timeout {
|
|
||||||
None => -1,
|
|
||||||
Some(d) => {
|
|
||||||
let ms = d.as_millis();
|
|
||||||
if ms > i32::MAX as u128 {
|
|
||||||
i32::MAX
|
|
||||||
} else {
|
|
||||||
ms as i32
|
|
||||||
}
|
|
||||||
}
|
|
||||||
};
|
|
||||||
let mut pfd = libc::pollfd {
|
|
||||||
fd,
|
|
||||||
events: libc::POLLIN,
|
|
||||||
revents: 0,
|
|
||||||
};
|
|
||||||
loop {
|
|
||||||
let r = unsafe { libc::poll(&mut pfd as *mut _, 1, timeout_ms) };
|
|
||||||
if r < 0 {
|
|
||||||
let e = unsafe { *libc::__errno_location() };
|
|
||||||
if e == libc::EINTR {
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|||||||
+6
-3
@@ -12,6 +12,7 @@
|
|||||||
//! See `LOOM.md` for the design intent and the deferred-for-later list.
|
//! See `LOOM.md` for the design intent and the deferred-for-later list.
|
||||||
|
|
||||||
pub mod stack;
|
pub mod stack;
|
||||||
|
pub(crate) mod signal;
|
||||||
pub mod context;
|
pub mod context;
|
||||||
pub mod preempt;
|
pub mod preempt;
|
||||||
pub mod pid;
|
pub mod pid;
|
||||||
@@ -32,6 +33,7 @@ pub mod introspect;
|
|||||||
#[cfg(feature = "observer")]
|
#[cfg(feature = "observer")]
|
||||||
pub mod observer;
|
pub mod observer;
|
||||||
pub mod runtime;
|
pub mod runtime;
|
||||||
|
pub(crate) mod park;
|
||||||
pub(crate) mod raw_mutex;
|
pub(crate) mod raw_mutex;
|
||||||
pub(crate) mod slot_state;
|
pub(crate) mod slot_state;
|
||||||
pub(crate) mod sync_shim;
|
pub(crate) mod sync_shim;
|
||||||
@@ -63,7 +65,7 @@ pub use gen_statem::{
|
|||||||
CallError as GenStatemCallError, Cx, Machine, Reply, Resolution, SendError as GenStatemSendError,
|
CallError as GenStatemCallError, Cx, Machine, Reply, Resolution, SendError as GenStatemSendError,
|
||||||
GenStatemRef,
|
GenStatemRef,
|
||||||
};
|
};
|
||||||
pub use introspect::{
|
pub use introspect::{StackInfo,
|
||||||
actor_info, snapshot, tree, tree_from, ActorInfo, ActorState, RuntimeSnapshot, RuntimeTree,
|
actor_info, snapshot, tree, tree_from, ActorInfo, ActorState, RuntimeSnapshot, RuntimeTree,
|
||||||
TreeNode, SNAPSHOT_FORMAT_VERSION,
|
TreeNode, SNAPSHOT_FORMAT_VERSION,
|
||||||
};
|
};
|
||||||
@@ -82,8 +84,9 @@ pub use runtime::{init, Config, Runtime};
|
|||||||
pub use scheduler::{
|
pub use scheduler::{
|
||||||
block_on_io, cancel_timer, request_stop, run, self_pid, send_after, send_after_named,
|
block_on_io, cancel_timer, request_stop, run, self_pid, send_after, send_after_named,
|
||||||
send_after_named_wall, send_after_wall, sleep, sleep_wall,
|
send_after_named_wall, send_after_wall, sleep, sleep_wall,
|
||||||
spawn, spawn_addr, spawn_under, wait_readable, wait_readable_timeout, wait_writable,
|
spawn, spawn_addr, spawn_addr_with, spawn_under, spawn_under_with, spawn_with,
|
||||||
wait_writable_timeout, yield_now, FdArm, JoinError, JoinHandle,
|
wait_readable, wait_readable_timeout, wait_writable,
|
||||||
|
wait_writable_timeout, yield_now, FdArm, JoinError, JoinHandle, SpawnOpts,
|
||||||
};
|
};
|
||||||
pub use supervisor::{ChildSpec, OneForOne, Restart, Signal, Strategy};
|
pub use supervisor::{ChildSpec, OneForOne, Restart, Signal, Strategy};
|
||||||
pub use timer::TimerId;
|
pub use timer::TimerId;
|
||||||
|
|||||||
+994
@@ -0,0 +1,994 @@
|
|||||||
|
//! Scheduler park/wake coordination layer (RFC 018).
|
||||||
|
//!
|
||||||
|
//! Schedulers never touch an fd to sleep: they park on a per-thread
|
||||||
|
//! [`Parker`] and are woken through an idle-mask protocol the runtime owns
|
||||||
|
//! outright. IO backends (epoll today, io_uring later) are *producers*
|
||||||
|
//! behind a two-call contract — make actors runnable, then wake — which is
|
||||||
|
//! what makes backend selection tractable (RFC 018 §step-back).
|
||||||
|
//!
|
||||||
|
//! Three pieces, all in [`Coordinator`]:
|
||||||
|
//!
|
||||||
|
//! - **Parker** (one per scheduler): permit semantics, `std::thread::park`
|
||||||
|
//! shaped — an unpark delivered before the park sets a permit; the next
|
||||||
|
//! park consumes it and returns immediately. This single property closes
|
||||||
|
//! the check-then-park race. Linux: `futex(2)` `FUTEX_WAIT`/`FUTEX_WAKE`
|
||||||
|
//! (private) with a nanosecond-precision relative `timespec` — the
|
||||||
|
//! `as_millis` truncation defect of the retired wake pipe is
|
||||||
|
//! unrepresentable here. Loom / non-Linux: `Mutex<bool>` + `Condvar`
|
||||||
|
//! (the loom models run against this build).
|
||||||
|
//! - **Idle mask**: an `AtomicU64` bitmask of parked scheduler ids
|
||||||
|
//! (construction asserts ≤ 64 schedulers). Park protocol: set own bit,
|
||||||
|
//! run the caller's mandatory post-publish re-check, then wait. A
|
||||||
|
//! producer that published work before observing our bit has left us
|
||||||
|
//! work the re-check finds; one that observes the bit wakes us.
|
||||||
|
//!
|
||||||
|
//! The publish/re-check pair is a store-buffer (Dekker) shape. Two sound
|
||||||
|
//! resolutions coexist here, chosen per call-site cost profile: the
|
||||||
|
//! **fence handshake** on the hot producer path
|
||||||
|
//! ([`Coordinator::wake_one_if_idle`]: publish work; `fence(SeqCst)`;
|
||||||
|
//! one *Relaxed* mask load — pairing with the consumer's `fetch_or(bit)`;
|
||||||
|
//! `fence(SeqCst)`; re-check inside [`Coordinator::park`]), so the
|
||||||
|
//! pure-compute hot path (everyone busy, mask 0) never takes the shared
|
||||||
|
//! mask line exclusive — the RFC's "one relaxed load" fast path, made
|
||||||
|
//! sound; and the **same-location-RMW read** (`fetch_or(0)`, in
|
||||||
|
//! [`Coordinator::wake_one`] / [`Coordinator::idle_mask`]) for the rare
|
||||||
|
//! paths (chain rule — at most one per wake) where reading the latest
|
||||||
|
//! mask by modification-order coherence is worth an RMW. The loom models
|
||||||
|
//! drive the fence pattern end to end (a fence-less plain-load draft
|
||||||
|
//! would — and did — fail model 1/2 with a lost wake, as it must).
|
||||||
|
//! - **Timekeeper**: at most one parked scheduler holds the timer
|
||||||
|
//! deadline (RFC 018 §timers) so a timer expiry wakes one scheduler,
|
||||||
|
//! not a herd. The role is an atomic `(holder id, armed deadline)`
|
||||||
|
//! pair; a timer insertion with an earlier deadline wakes the holder to
|
||||||
|
//! re-peek. Arm / insert-check MUST be serialized by the caller (the
|
||||||
|
//! timers mutex in the runtime) — the atomics exist so the *busy-path
|
||||||
|
//! due-check* (one Relaxed load, [`Coordinator::armed_deadline_nanos`])
|
||||||
|
//! and the wake stay lock-free. All races are biased over-wake: a
|
||||||
|
//! spurious permit costs one failed pop; a missed wake would cost a
|
||||||
|
//! stranded actor, and is unrepresentable under the serialization rule.
|
||||||
|
//!
|
||||||
|
//! Every wake here is *at most one* futex round-trip and wakes *exactly
|
||||||
|
//! one* scheduler by construction (`wake_one` CASes a bit clear before
|
||||||
|
//! unparking its owner) — there is no shared level-triggered anything
|
||||||
|
//! left to herd on.
|
||||||
|
//!
|
||||||
|
//! Standalone until the runtime swap (RFC 018 commit 2): nothing outside
|
||||||
|
//! tests constructs a [`Coordinator`] yet.
|
||||||
|
|
||||||
|
use crate::sync_shim::{fence, AtomicU64, Ordering};
|
||||||
|
use std::time::Instant;
|
||||||
|
|
||||||
|
/// Sentinel for "no timekeeper" in the holder word.
|
||||||
|
const NO_TIMEKEEPER: u64 = u64::MAX;
|
||||||
|
/// Sentinel for "no armed deadline" in the deadline word. Also what the
|
||||||
|
/// busy-path due-check compares against: `now_nanos < armed` is one branch.
|
||||||
|
pub(crate) const NO_DEADLINE: u64 = u64::MAX;
|
||||||
|
|
||||||
|
/// Outcome of a [`Coordinator::park`] call.
|
||||||
|
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||||
|
pub(crate) enum ParkResult {
|
||||||
|
/// A permit was consumed (wake delivered before or during the park).
|
||||||
|
Woken,
|
||||||
|
/// The deadline passed with no wake. Only possible when a deadline was
|
||||||
|
/// supplied (and never under loom, which has no time — see `park`).
|
||||||
|
TimedOut,
|
||||||
|
/// The post-publish re-check found work; the thread never blocked.
|
||||||
|
/// A permit may still be pending (a racing `wake_one` picked us after
|
||||||
|
/// the bit was set) — it will surface as one spurious `Woken` on a
|
||||||
|
/// later park. Benign: over-wake by design.
|
||||||
|
WorkFound,
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Parker — permit-semantics thread parking
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
/// Linux, non-loom: futex on a state word.
|
||||||
|
///
|
||||||
|
/// States: EMPTY (no permit, nobody waiting), PARKED (a thread is, or is
|
||||||
|
/// about to be, in `futex_wait`), NOTIFIED (permit pending). The classic
|
||||||
|
/// std-parker protocol: `unpark` swaps to NOTIFIED and futex-wakes iff it
|
||||||
|
/// displaced PARKED; `park` CASes EMPTY→PARKED, waits, and consumes
|
||||||
|
/// NOTIFIED on every exit path.
|
||||||
|
#[cfg(all(target_os = "linux", not(loom)))]
|
||||||
|
mod parker {
|
||||||
|
use std::sync::atomic::{AtomicU32, Ordering};
|
||||||
|
use std::time::Instant;
|
||||||
|
|
||||||
|
const EMPTY: u32 = 0;
|
||||||
|
const PARKED: u32 = 1;
|
||||||
|
const NOTIFIED: u32 = 2;
|
||||||
|
|
||||||
|
pub(super) struct Parker {
|
||||||
|
state: AtomicU32,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Parker {
|
||||||
|
pub(super) fn new() -> Self {
|
||||||
|
Self { state: AtomicU32::new(EMPTY) }
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Returns `true` = woken (permit consumed), `false` = timed out.
|
||||||
|
pub(super) fn park(&self, deadline: Option<Instant>) -> bool {
|
||||||
|
// Fast path: consume a pending permit without blocking.
|
||||||
|
if self
|
||||||
|
.state
|
||||||
|
.compare_exchange(EMPTY, PARKED, Ordering::AcqRel, Ordering::Acquire)
|
||||||
|
.is_err()
|
||||||
|
{
|
||||||
|
// Only NOTIFIED can be here (one thread parks at a time).
|
||||||
|
self.state.store(EMPTY, Ordering::Release);
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
loop {
|
||||||
|
let timeout = match deadline {
|
||||||
|
None => None,
|
||||||
|
Some(d) => {
|
||||||
|
let now = Instant::now();
|
||||||
|
if d <= now {
|
||||||
|
// Deadline passed: cancel the park. The swap
|
||||||
|
// races a concurrent unpark — if it delivered
|
||||||
|
// NOTIFIED first, report Woken (never lose a
|
||||||
|
// permit).
|
||||||
|
return self.state.swap(EMPTY, Ordering::AcqRel) == NOTIFIED;
|
||||||
|
}
|
||||||
|
Some(d - now)
|
||||||
|
}
|
||||||
|
};
|
||||||
|
futex_wait(&self.state, PARKED, timeout);
|
||||||
|
if self
|
||||||
|
.state
|
||||||
|
.compare_exchange(NOTIFIED, EMPTY, Ordering::AcqRel, Ordering::Acquire)
|
||||||
|
.is_ok()
|
||||||
|
{
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
// Spurious wake or timeout with state still PARKED: loop —
|
||||||
|
// the deadline check at the top decides.
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(super) fn unpark(&self) {
|
||||||
|
if self.state.swap(NOTIFIED, Ordering::AcqRel) == PARKED {
|
||||||
|
futex_wake(&self.state, 1);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `FUTEX_WAIT` with a *relative* nanosecond timeout (`CLOCK_MONOTONIC`
|
||||||
|
/// per futex(2) for relative waits). No millisecond conversion anywhere:
|
||||||
|
/// the timespec carries the full sub-ms remainder (RFC 018 kills the
|
||||||
|
/// `as_millis` truncation structurally).
|
||||||
|
fn futex_wait(word: &AtomicU32, expected: u32, timeout: Option<std::time::Duration>) {
|
||||||
|
let ts;
|
||||||
|
let ts_ptr: *const libc::timespec = match timeout {
|
||||||
|
Some(d) => {
|
||||||
|
ts = libc::timespec {
|
||||||
|
tv_sec: d.as_secs() as libc::time_t,
|
||||||
|
tv_nsec: d.subsec_nanos() as libc::c_long,
|
||||||
|
};
|
||||||
|
&ts
|
||||||
|
}
|
||||||
|
None => std::ptr::null(),
|
||||||
|
};
|
||||||
|
// Errors (EAGAIN: word changed; ETIMEDOUT; EINTR) all mean "return
|
||||||
|
// and let the caller's state machine decide" — deliberately ignored.
|
||||||
|
unsafe {
|
||||||
|
libc::syscall(
|
||||||
|
libc::SYS_futex,
|
||||||
|
word.as_ptr(),
|
||||||
|
libc::FUTEX_WAIT | libc::FUTEX_PRIVATE_FLAG,
|
||||||
|
expected,
|
||||||
|
ts_ptr,
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn futex_wake(word: &AtomicU32, n: u32) {
|
||||||
|
unsafe {
|
||||||
|
libc::syscall(
|
||||||
|
libc::SYS_futex,
|
||||||
|
word.as_ptr(),
|
||||||
|
libc::FUTEX_WAKE | libc::FUTEX_PRIVATE_FLAG,
|
||||||
|
n,
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Loom / non-Linux: `Mutex<bool>` permit + `Condvar` — loom's own model
|
||||||
|
/// of a parker, and the portable fallback. Under loom the deadline is
|
||||||
|
/// ignored (loom has no time); the models exercise wake paths only.
|
||||||
|
#[cfg(any(loom, not(target_os = "linux")))]
|
||||||
|
mod parker {
|
||||||
|
use crate::sync_shim::{Condvar, Mutex};
|
||||||
|
use std::time::Instant;
|
||||||
|
|
||||||
|
pub(super) struct Parker {
|
||||||
|
permit: Mutex<bool>,
|
||||||
|
cv: Condvar,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Parker {
|
||||||
|
pub(super) fn new() -> Self {
|
||||||
|
Self { permit: Mutex::new(false), cv: Condvar::new() }
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Returns `true` = woken (permit consumed), `false` = timed out.
|
||||||
|
pub(super) fn park(&self, deadline: Option<Instant>) -> bool {
|
||||||
|
let mut permit = match self.permit.lock() {
|
||||||
|
Ok(g) => g,
|
||||||
|
Err(_) => panic!("smarm: parker permit lock poisoned (core corrupt)"),
|
||||||
|
};
|
||||||
|
loop {
|
||||||
|
if *permit {
|
||||||
|
*permit = false;
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
#[cfg(loom)]
|
||||||
|
{
|
||||||
|
// Loom has no clock: block until a wake. Models must
|
||||||
|
// deliver one (a park nobody wakes is a real deadlock
|
||||||
|
// and loom reports it as such).
|
||||||
|
let _ = deadline;
|
||||||
|
permit = match self.cv.wait(permit) {
|
||||||
|
Ok(g) => g,
|
||||||
|
Err(_) => panic!("smarm: parker cv poisoned (core corrupt)"),
|
||||||
|
};
|
||||||
|
}
|
||||||
|
#[cfg(not(loom))]
|
||||||
|
{
|
||||||
|
match deadline {
|
||||||
|
None => {
|
||||||
|
permit = match self.cv.wait(permit) {
|
||||||
|
Ok(g) => g,
|
||||||
|
Err(_) => panic!("smarm: parker cv poisoned (core corrupt)"),
|
||||||
|
};
|
||||||
|
}
|
||||||
|
Some(d) => {
|
||||||
|
let now = Instant::now();
|
||||||
|
if d <= now {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
permit = match self.cv.wait_timeout(permit, d - now) {
|
||||||
|
Ok((g, _)) => g,
|
||||||
|
Err(_) => panic!("smarm: parker cv poisoned (core corrupt)"),
|
||||||
|
};
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(super) fn unpark(&self) {
|
||||||
|
let mut permit = match self.permit.lock() {
|
||||||
|
Ok(g) => g,
|
||||||
|
Err(_) => panic!("smarm: parker permit lock poisoned (core corrupt)"),
|
||||||
|
};
|
||||||
|
*permit = true;
|
||||||
|
self.cv.notify_one();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
use parker::Parker;
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Coordinator — idle mask + wake protocol + timekeeper
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
pub(crate) struct Coordinator {
|
||||||
|
parkers: Box<[Parker]>,
|
||||||
|
/// Bit `i` set = scheduler `i` is parked or committed to parking (set
|
||||||
|
/// before the re-check; cleared by `wake_one`'s CAS or by the parker
|
||||||
|
/// itself on return). AcqRel same-location-RMW handshake — see module
|
||||||
|
/// docs (no SeqCst needed: every producer-side read is an RMW).
|
||||||
|
idle: AtomicU64,
|
||||||
|
/// Timekeeper holder id, or `NO_TIMEKEEPER`. Written under the
|
||||||
|
/// caller's timer serialization (arm/disarm/insert-check); read
|
||||||
|
/// lock-free by the insert wake path.
|
||||||
|
tk_holder: AtomicU64,
|
||||||
|
/// Armed deadline as nanos since `origin`, or `NO_DEADLINE`. Written
|
||||||
|
/// only by the timekeeper arm/disarm protocol.
|
||||||
|
tk_armed: AtomicU64,
|
||||||
|
/// Earliest KNOWN timer deadline (nanos since `origin`), or
|
||||||
|
/// `NO_DEADLINE` — independent of whether any scheduler is parked,
|
||||||
|
/// which is what the timekeeper's `tk_armed` cannot give: under
|
||||||
|
/// saturation nobody parks and nobody arms, yet due timers must still
|
||||||
|
/// fire (ratified design point (a)). Maintained under the caller's
|
||||||
|
/// timers mutex (`note_deadline` on insert, `refresh_deadline` after a
|
||||||
|
/// pop/peek); read lock-free by the busy-path due-check.
|
||||||
|
next_deadline: AtomicU64,
|
||||||
|
/// Time origin for the nanos encoding.
|
||||||
|
origin: Instant,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Coordinator {
|
||||||
|
pub(crate) fn new(schedulers: usize) -> Self {
|
||||||
|
assert!(
|
||||||
|
(1..=64).contains(&schedulers),
|
||||||
|
"smarm: scheduler count must be 1..=64 (idle mask is one u64); got {schedulers}"
|
||||||
|
);
|
||||||
|
Self {
|
||||||
|
parkers: (0..schedulers).map(|_| Parker::new()).collect(),
|
||||||
|
idle: AtomicU64::new(0),
|
||||||
|
tk_holder: AtomicU64::new(NO_TIMEKEEPER),
|
||||||
|
tk_armed: AtomicU64::new(NO_DEADLINE),
|
||||||
|
next_deadline: AtomicU64::new(NO_DEADLINE),
|
||||||
|
origin: Instant::now(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Encode a deadline for the armed snapshot / busy-path compare.
|
||||||
|
/// Saturating: a deadline at-or-before `origin` encodes as 0 (always
|
||||||
|
/// due), one beyond ~584 years as `NO_DEADLINE - 1`.
|
||||||
|
pub(crate) fn deadline_nanos(&self, deadline: Instant) -> u64 {
|
||||||
|
let nanos = deadline
|
||||||
|
.checked_duration_since(self.origin)
|
||||||
|
.map(|d| d.as_nanos())
|
||||||
|
.unwrap_or(0);
|
||||||
|
if nanos >= NO_DEADLINE as u128 {
|
||||||
|
NO_DEADLINE - 1
|
||||||
|
} else {
|
||||||
|
nanos as u64
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Park scheduler `id` until a wake, the deadline, or a positive
|
||||||
|
/// re-check. Protocol: (1) publish own idle bit (SeqCst), (2) run
|
||||||
|
/// `recheck` — it MUST re-read the work source with an ordering that
|
||||||
|
/// pairs with the producer's publish (SeqCst load, or take the mutex
|
||||||
|
/// the producer publishes under); a `true` aborts the park, (3) block.
|
||||||
|
pub(crate) fn park(
|
||||||
|
&self,
|
||||||
|
id: usize,
|
||||||
|
deadline: Option<Instant>,
|
||||||
|
recheck: impl FnOnce() -> bool,
|
||||||
|
) -> ParkResult {
|
||||||
|
debug_assert!(id < self.parkers.len(), "park: scheduler id out of range");
|
||||||
|
let bit = 1u64 << id;
|
||||||
|
// (1) publish. AcqRel RMW: the acquire half is the handshake — if
|
||||||
|
// this lands after a producer's mask RMW in modification order, we
|
||||||
|
// read-from it and the producer's earlier work publication is
|
||||||
|
// visible to the re-check below (see module docs).
|
||||||
|
let prev = self.idle.fetch_or(bit, Ordering::AcqRel);
|
||||||
|
debug_assert_eq!(prev & bit, 0, "park: idle bit already set for this id");
|
||||||
|
// Fence half of the producer handshake (see `wake_one_if_idle`):
|
||||||
|
// orders our bit-publish before the re-check's loads, so it pairs
|
||||||
|
// with the producer's publish→fence→mask-load — at least one side
|
||||||
|
// must see the other's store, whichever queue backend is in play.
|
||||||
|
fence(Ordering::SeqCst);
|
||||||
|
// (2) the mandatory post-publish re-check.
|
||||||
|
if recheck() {
|
||||||
|
self.idle.fetch_and(!bit, Ordering::AcqRel);
|
||||||
|
return ParkResult::WorkFound;
|
||||||
|
}
|
||||||
|
// (3) block. The permit closes the window between the re-check and
|
||||||
|
// the futex wait: a wake_one that picked us in that window has
|
||||||
|
// already CASed our bit clear and set the permit.
|
||||||
|
let woken = self.parkers[id].park(deadline);
|
||||||
|
// Clear own bit — a no-op when a waker already CASed it clear.
|
||||||
|
self.idle.fetch_and(!bit, Ordering::AcqRel);
|
||||||
|
if woken {
|
||||||
|
ParkResult::Woken
|
||||||
|
} else {
|
||||||
|
ParkResult::TimedOut
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Wake exactly one parked scheduler, if any: pick the highest set idle
|
||||||
|
/// bit (LIFO-ish — warmest cache), CAS it clear, deliver a permit.
|
||||||
|
/// Empty mask = no-op (everyone is awake and will find work by
|
||||||
|
/// popping). Returns whether a scheduler was woken.
|
||||||
|
pub(crate) fn wake_one(&self) -> bool {
|
||||||
|
// RMW read, not a load: reads the latest mask by modification-order
|
||||||
|
// coherence, closing the Dekker race with a parking consumer (see
|
||||||
|
// module docs). The release side of the RMW is what a later-parking
|
||||||
|
// consumer's fetch_or acquires to make its re-check sound.
|
||||||
|
let mut mask = self.idle.fetch_or(0, Ordering::AcqRel);
|
||||||
|
loop {
|
||||||
|
if mask == 0 {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
let id = 63 - mask.leading_zeros() as usize; // highest set bit
|
||||||
|
let bit = 1u64 << id;
|
||||||
|
// The CAS is the exactly-one guarantee: whoever clears the bit
|
||||||
|
// owns the wake; a racing wake_one retries on the observed value
|
||||||
|
// (coherence: a failed CAS can never read older than `mask`).
|
||||||
|
match self.idle.compare_exchange(
|
||||||
|
mask,
|
||||||
|
mask & !bit,
|
||||||
|
Ordering::AcqRel,
|
||||||
|
Ordering::Acquire,
|
||||||
|
) {
|
||||||
|
Ok(_) => {
|
||||||
|
self.parkers[id].unpark();
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
Err(m) => mask = m,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The producer-side wake tail (`enqueue`'s fast path, RFC 018 "enqueue
|
||||||
|
/// wakes"). The caller has just published work (queue push); we fence,
|
||||||
|
/// then read the mask with ONE Relaxed load — 0 means every scheduler
|
||||||
|
/// is awake and the pure-compute hot path pays no RMW on the shared
|
||||||
|
/// mask line. Soundness is the fence handshake (module docs): our
|
||||||
|
/// fence orders the caller's push before the mask load; the consumer's
|
||||||
|
/// fence (in `park`) orders its bit-publish before its re-check — at
|
||||||
|
/// least one side must observe the other's store, so a consumer we
|
||||||
|
/// miss here is a consumer whose re-check finds the caller's work.
|
||||||
|
pub(crate) fn wake_one_if_idle(&self) -> bool {
|
||||||
|
fence(Ordering::SeqCst);
|
||||||
|
if self.idle.load(Ordering::Relaxed) == 0 {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
self.wake_one()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Terminal wake (replaces the AllDone wake-pipe byte): clear the mask
|
||||||
|
/// and deliver a permit to *every* parker, parked or not. A permit set
|
||||||
|
/// on a busy scheduler costs one spurious park return — nothing at the
|
||||||
|
/// terminal boundary. Idempotent.
|
||||||
|
pub(crate) fn wake_all(&self) {
|
||||||
|
self.idle.store(0, Ordering::Release);
|
||||||
|
for p in self.parkers.iter() {
|
||||||
|
p.unpark();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Latest idle mask (RMW read — same handshake as `wake_one`). A
|
||||||
|
/// test-only observer: production expresses the chain rule through
|
||||||
|
/// `wake_one_if_idle` (fence + Relaxed load), not a mask read.
|
||||||
|
#[cfg(test)]
|
||||||
|
pub(crate) fn idle_mask(&self) -> u64 {
|
||||||
|
self.idle.fetch_or(0, Ordering::AcqRel)
|
||||||
|
}
|
||||||
|
|
||||||
|
// ----- timekeeper -----
|
||||||
|
|
||||||
|
/// Try to take the timekeeper role for scheduler `id` with `deadline`.
|
||||||
|
/// MUST be called under the caller's timer serialization (the timers
|
||||||
|
/// mutex), with `deadline` the heap minimum peeked under that same
|
||||||
|
/// hold — this is what makes the insert-check race-free. Returns
|
||||||
|
/// whether the role was taken (false = someone else holds it; park
|
||||||
|
/// with no deadline).
|
||||||
|
pub(crate) fn try_arm_timer(&self, id: usize, deadline: Instant) -> bool {
|
||||||
|
debug_assert!(id < self.parkers.len(), "try_arm_timer: id out of range");
|
||||||
|
if self
|
||||||
|
.tk_holder
|
||||||
|
.compare_exchange(NO_TIMEKEEPER, id as u64, Ordering::SeqCst, Ordering::SeqCst)
|
||||||
|
.is_err()
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
// Holder-then-deadline order: an insert-check that sees the holder
|
||||||
|
// with the deadline still NO_DEADLINE compares `new < MAX` = true
|
||||||
|
// and over-wakes — the benign direction. (Under the mandated timer
|
||||||
|
// serialization this interleaving cannot occur anyway.)
|
||||||
|
self.tk_armed.store(self.deadline_nanos(deadline), Ordering::SeqCst);
|
||||||
|
true
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Release the timekeeper role (the holder, on wake, before it
|
||||||
|
/// re-peeks/fires). Callable without the timer serialization: a
|
||||||
|
/// racing insert may wake a no-longer-holder — over-wake, benign.
|
||||||
|
pub(crate) fn disarm_timer(&self, id: usize) {
|
||||||
|
debug_assert_eq!(
|
||||||
|
self.tk_holder.load(Ordering::SeqCst),
|
||||||
|
id as u64,
|
||||||
|
"disarm_timer by a non-holder"
|
||||||
|
);
|
||||||
|
// Deadline first: a concurrent insert-check then sees NO_DEADLINE
|
||||||
|
// and skips the (now pointless) wake instead of waking a stale
|
||||||
|
// holder id. Either order is correct; this one wastes less.
|
||||||
|
self.tk_armed.store(NO_DEADLINE, Ordering::SeqCst);
|
||||||
|
self.tk_holder.store(NO_TIMEKEEPER, Ordering::SeqCst);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Insert-side re-arm check: if `deadline` is earlier than the armed
|
||||||
|
/// snapshot, wake the timekeeper to re-peek. MUST be called under the
|
||||||
|
/// same timer serialization as `try_arm_timer` (see there).
|
||||||
|
pub(crate) fn timer_inserted(&self, deadline: Instant) {
|
||||||
|
if self.deadline_nanos(deadline) < self.tk_armed.load(Ordering::SeqCst) {
|
||||||
|
let holder = self.tk_holder.load(Ordering::SeqCst);
|
||||||
|
if holder != NO_TIMEKEEPER {
|
||||||
|
// Direct unpark, not wake_one: the wake targets the
|
||||||
|
// timekeeper specifically (it must re-peek the heap). Its
|
||||||
|
// idle bit stays set until it returns from park — a
|
||||||
|
// concurrent wake_one may pick it too; over-wake, benign.
|
||||||
|
self.parkers[holder as usize].unpark();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The armed-deadline snapshot (nanos since origin; `NO_DEADLINE` =
|
||||||
|
/// none). Test-only introspection on the timekeeper's armed value; the
|
||||||
|
/// busy-path due-check reads `next_deadline`, not this.
|
||||||
|
#[cfg(test)]
|
||||||
|
pub(crate) fn armed_deadline_nanos(&self) -> u64 {
|
||||||
|
self.tk_armed.load(Ordering::Relaxed)
|
||||||
|
}
|
||||||
|
|
||||||
|
// ----- earliest-deadline snapshot (busy-path due-check) -----
|
||||||
|
|
||||||
|
/// Record a newly inserted timer deadline. MUST be called under the
|
||||||
|
/// timers mutex (same serialization rule as `try_arm_timer`), which is
|
||||||
|
/// why plain compare+store suffices for the min-maintenance. Also runs
|
||||||
|
/// the timekeeper re-arm check (`timer_inserted`) — one call site for
|
||||||
|
/// both consequences of an insert.
|
||||||
|
pub(crate) fn note_deadline(&self, deadline: Instant) {
|
||||||
|
let n = self.deadline_nanos(deadline);
|
||||||
|
if n < self.next_deadline.load(Ordering::Relaxed) {
|
||||||
|
self.next_deadline.store(n, Ordering::Release);
|
||||||
|
}
|
||||||
|
self.timer_inserted(deadline);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Re-anchor the snapshot to the heap minimum (`None` = heap empty)
|
||||||
|
/// after a `pop_due` / `clear`. MUST be called under the timers mutex.
|
||||||
|
pub(crate) fn refresh_deadline(&self, next: Option<Instant>) {
|
||||||
|
let n = next.map_or(NO_DEADLINE, |d| self.deadline_nanos(d));
|
||||||
|
self.next_deadline.store(n, Ordering::Release);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Busy-path due-check: is the earliest known deadline at or past
|
||||||
|
/// `now`? One Relaxed load when no deadline is armed — the clock is
|
||||||
|
/// read only when a timer actually exists (matching the old drain
|
||||||
|
/// phase's is_empty guard), so the pure-compute hot path pays a load
|
||||||
|
/// and a branch.
|
||||||
|
pub(crate) fn deadline_due(&self) -> bool {
|
||||||
|
let n = self.next_deadline.load(Ordering::Relaxed);
|
||||||
|
n != NO_DEADLINE && self.deadline_nanos(Instant::now()) >= n
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The earliest-deadline snapshot as an `Instant` (`None` = no timer
|
||||||
|
/// pending). Test-only: the idle path arms the timekeeper from the
|
||||||
|
/// timer heap's own `peek_deadline` under the timers mutex.
|
||||||
|
#[cfg(test)]
|
||||||
|
pub(crate) fn next_deadline_instant(&self) -> Option<Instant> {
|
||||||
|
let n = self.next_deadline.load(Ordering::Acquire);
|
||||||
|
if n == NO_DEADLINE {
|
||||||
|
None
|
||||||
|
} else {
|
||||||
|
self.origin.checked_add(std::time::Duration::from_nanos(n))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Unit tests (std build)
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
#[cfg(all(test, not(loom)))]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
use std::sync::atomic::{AtomicUsize, Ordering as O};
|
||||||
|
use std::sync::Arc;
|
||||||
|
use std::time::{Duration, Instant};
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn permit_before_park_returns_immediately() {
|
||||||
|
let c = Coordinator::new(1);
|
||||||
|
// Deliver the wake first (nobody parked: wake_one no-ops on the
|
||||||
|
// mask, so use the timekeeper-direct path? No — permit semantics
|
||||||
|
// are the parker's own; exercise via wake_all which permits all).
|
||||||
|
c.wake_all();
|
||||||
|
let t0 = Instant::now();
|
||||||
|
let r = c.park(0, None, || false);
|
||||||
|
assert_eq!(r, ParkResult::Woken);
|
||||||
|
assert!(t0.elapsed() < Duration::from_millis(100), "park blocked despite permit");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn submillisecond_deadline_is_honored() {
|
||||||
|
// Regression for the retired as_millis truncation: a 500µs deadline
|
||||||
|
// must neither busy-return instantly forever nor round to 0/∞.
|
||||||
|
let c = Coordinator::new(1);
|
||||||
|
let t0 = Instant::now();
|
||||||
|
let r = c.park(0, Some(t0 + Duration::from_micros(500)), || false);
|
||||||
|
let dt = t0.elapsed();
|
||||||
|
assert_eq!(r, ParkResult::TimedOut);
|
||||||
|
assert!(dt >= Duration::from_micros(400), "woke too early: {dt:?}");
|
||||||
|
assert!(dt < Duration::from_millis(50), "overslept: {dt:?}");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn recheck_true_aborts_park_and_clears_bit() {
|
||||||
|
let c = Coordinator::new(2);
|
||||||
|
let r = c.park(1, None, || true);
|
||||||
|
assert_eq!(r, ParkResult::WorkFound);
|
||||||
|
assert_eq!(c.idle_mask(), 0, "bit not cleared after WorkFound");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn recheck_observes_own_bit_published() {
|
||||||
|
let c = Coordinator::new(2);
|
||||||
|
let seen = std::cell::Cell::new(0u64);
|
||||||
|
let r = c.park(1, None, || {
|
||||||
|
seen.set(c.idle_mask());
|
||||||
|
true
|
||||||
|
});
|
||||||
|
assert_eq!(r, ParkResult::WorkFound);
|
||||||
|
assert_eq!(seen.get() & 0b10, 0b10, "bit not published before re-check");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn wake_one_wakes_exactly_one_of_n() {
|
||||||
|
const N: usize = 4;
|
||||||
|
let c = Arc::new(Coordinator::new(N));
|
||||||
|
let woken = Arc::new(AtomicUsize::new(0));
|
||||||
|
let mut ts = Vec::new();
|
||||||
|
for id in 0..N {
|
||||||
|
let c = c.clone();
|
||||||
|
let woken = woken.clone();
|
||||||
|
ts.push(std::thread::spawn(move || {
|
||||||
|
let r = c.park(id, None, || false);
|
||||||
|
assert_eq!(r, ParkResult::Woken);
|
||||||
|
woken.fetch_add(1, O::SeqCst);
|
||||||
|
}));
|
||||||
|
}
|
||||||
|
// Wait until all four are published idle.
|
||||||
|
let t0 = Instant::now();
|
||||||
|
while c.idle_mask().count_ones() != N as u32 {
|
||||||
|
assert!(t0.elapsed() < Duration::from_secs(5), "threads never parked");
|
||||||
|
std::thread::yield_now();
|
||||||
|
}
|
||||||
|
assert!(c.wake_one());
|
||||||
|
// Exactly one wakes; give the others a beat to (incorrectly) wake.
|
||||||
|
let t0 = Instant::now();
|
||||||
|
while woken.load(O::SeqCst) == 0 {
|
||||||
|
assert!(t0.elapsed() < Duration::from_secs(5), "wake_one woke nobody");
|
||||||
|
std::thread::yield_now();
|
||||||
|
}
|
||||||
|
std::thread::sleep(Duration::from_millis(100));
|
||||||
|
assert_eq!(woken.load(O::SeqCst), 1, "wake_one woke more than one");
|
||||||
|
assert_eq!(c.idle_mask().count_ones(), (N - 1) as u32);
|
||||||
|
c.wake_all();
|
||||||
|
for t in ts {
|
||||||
|
match t.join() {
|
||||||
|
Ok(()) => {}
|
||||||
|
Err(p) => std::panic::resume_unwind(p),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
assert_eq!(woken.load(O::SeqCst), N);
|
||||||
|
assert_eq!(c.idle_mask(), 0);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn wake_one_prefers_highest_bit() {
|
||||||
|
let c = Arc::new(Coordinator::new(3));
|
||||||
|
let woken_id = Arc::new(AtomicUsize::new(usize::MAX));
|
||||||
|
let mut ts = Vec::new();
|
||||||
|
for id in 0..3 {
|
||||||
|
let c = c.clone();
|
||||||
|
let woken_id = woken_id.clone();
|
||||||
|
ts.push(std::thread::spawn(move || {
|
||||||
|
if c.park(id, None, || false) == ParkResult::Woken {
|
||||||
|
let _ = woken_id.compare_exchange(usize::MAX, id, O::SeqCst, O::SeqCst);
|
||||||
|
}
|
||||||
|
}));
|
||||||
|
}
|
||||||
|
let t0 = Instant::now();
|
||||||
|
while c.idle_mask() != 0b111 {
|
||||||
|
assert!(t0.elapsed() < Duration::from_secs(5));
|
||||||
|
std::thread::yield_now();
|
||||||
|
}
|
||||||
|
assert!(c.wake_one());
|
||||||
|
let t0 = Instant::now();
|
||||||
|
while woken_id.load(O::SeqCst) == usize::MAX {
|
||||||
|
assert!(t0.elapsed() < Duration::from_secs(5));
|
||||||
|
std::thread::yield_now();
|
||||||
|
}
|
||||||
|
assert_eq!(woken_id.load(O::SeqCst), 2, "LIFO-ish: highest bit first");
|
||||||
|
c.wake_all();
|
||||||
|
for t in ts {
|
||||||
|
let _ = t.join();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn wake_one_on_empty_mask_is_noop() {
|
||||||
|
let c = Coordinator::new(2);
|
||||||
|
assert!(!c.wake_one());
|
||||||
|
assert_eq!(c.idle_mask(), 0);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn timekeeper_arm_is_exclusive_and_snapshot_readable() {
|
||||||
|
let c = Coordinator::new(2);
|
||||||
|
let d2 = Instant::now() + Duration::from_secs(10);
|
||||||
|
let d1 = Instant::now() + Duration::from_secs(1);
|
||||||
|
assert_eq!(c.armed_deadline_nanos(), NO_DEADLINE);
|
||||||
|
assert!(c.try_arm_timer(0, d2));
|
||||||
|
assert!(!c.try_arm_timer(1, d1), "second arm must fail while held");
|
||||||
|
assert_eq!(c.armed_deadline_nanos(), c.deadline_nanos(d2));
|
||||||
|
c.disarm_timer(0);
|
||||||
|
assert_eq!(c.armed_deadline_nanos(), NO_DEADLINE);
|
||||||
|
assert!(c.try_arm_timer(1, d1), "role must be re-takeable after disarm");
|
||||||
|
c.disarm_timer(1);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn earlier_insert_wakes_timekeeper() {
|
||||||
|
let c = Coordinator::new(2);
|
||||||
|
let far = Instant::now() + Duration::from_secs(60);
|
||||||
|
let near = Instant::now() + Duration::from_millis(1);
|
||||||
|
assert!(c.try_arm_timer(0, far));
|
||||||
|
// Holder not yet parked: the wake must land as a permit.
|
||||||
|
c.timer_inserted(near);
|
||||||
|
let t0 = Instant::now();
|
||||||
|
let r = c.park(0, Some(far), || false);
|
||||||
|
assert_eq!(r, ParkResult::Woken, "re-arm wake lost");
|
||||||
|
assert!(t0.elapsed() < Duration::from_secs(5), "slept toward the stale deadline");
|
||||||
|
c.disarm_timer(0);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn later_insert_does_not_wake_timekeeper() {
|
||||||
|
let c = Coordinator::new(2);
|
||||||
|
let near = Instant::now() + Duration::from_millis(20);
|
||||||
|
let far = Instant::now() + Duration::from_secs(60);
|
||||||
|
assert!(c.try_arm_timer(0, near));
|
||||||
|
c.timer_inserted(far); // later than armed: no wake
|
||||||
|
let r = c.park(0, Some(near), || false);
|
||||||
|
assert_eq!(r, ParkResult::TimedOut, "spurious wake for a later insert");
|
||||||
|
c.disarm_timer(0);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
#[should_panic(expected = "1..=64")]
|
||||||
|
fn more_than_64_schedulers_asserts() {
|
||||||
|
let _ = Coordinator::new(65);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn wake_one_if_idle_noop_on_empty_and_wakes_on_parked() {
|
||||||
|
let c = Arc::new(Coordinator::new(1));
|
||||||
|
assert!(!c.wake_one_if_idle(), "empty mask must be a no-op");
|
||||||
|
let c2 = c.clone();
|
||||||
|
let t = std::thread::spawn(move || {
|
||||||
|
assert_eq!(c2.park(0, None, || false), ParkResult::Woken);
|
||||||
|
});
|
||||||
|
let t0 = Instant::now();
|
||||||
|
while c.idle_mask() == 0 {
|
||||||
|
assert!(t0.elapsed() < Duration::from_secs(5), "never parked");
|
||||||
|
std::thread::yield_now();
|
||||||
|
}
|
||||||
|
assert!(c.wake_one_if_idle());
|
||||||
|
match t.join() {
|
||||||
|
Ok(()) => {}
|
||||||
|
Err(p) => std::panic::resume_unwind(p),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn deadline_snapshot_min_maintenance_and_due_check() {
|
||||||
|
let c = Coordinator::new(1);
|
||||||
|
assert!(!c.deadline_due(), "no deadline: never due");
|
||||||
|
assert_eq!(c.next_deadline_instant(), None);
|
||||||
|
let far = Instant::now() + Duration::from_secs(60);
|
||||||
|
let near = Instant::now() + Duration::from_millis(1);
|
||||||
|
c.note_deadline(far);
|
||||||
|
assert!(!c.deadline_due());
|
||||||
|
c.note_deadline(near); // min wins
|
||||||
|
assert!(c.next_deadline_instant().is_some_and(|d| d <= near));
|
||||||
|
c.note_deadline(far); // later insert must NOT raise the snapshot
|
||||||
|
assert!(c.next_deadline_instant().is_some_and(|d| d <= near));
|
||||||
|
std::thread::sleep(Duration::from_millis(2));
|
||||||
|
assert!(c.deadline_due(), "past deadline not reported due");
|
||||||
|
c.refresh_deadline(Some(far));
|
||||||
|
assert!(!c.deadline_due(), "refresh did not re-anchor");
|
||||||
|
c.refresh_deadline(None);
|
||||||
|
assert_eq!(c.next_deadline_instant(), None);
|
||||||
|
// A deadline at/before origin encodes as 0: always due.
|
||||||
|
c.note_deadline(Instant::now() - Duration::from_secs(1));
|
||||||
|
assert!(c.deadline_due());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn note_deadline_earlier_wakes_timekeeper_via_snapshot_path() {
|
||||||
|
// note_deadline must carry the timer_inserted re-arm wake too.
|
||||||
|
let c = Coordinator::new(2);
|
||||||
|
let far = Instant::now() + Duration::from_secs(60);
|
||||||
|
let near = Instant::now() + Duration::from_millis(1);
|
||||||
|
assert!(c.try_arm_timer(0, far));
|
||||||
|
c.note_deadline(near);
|
||||||
|
let r = c.park(0, Some(far), || false);
|
||||||
|
assert_eq!(r, ParkResult::Woken, "re-arm wake lost through note_deadline");
|
||||||
|
c.disarm_timer(0);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// loom models — RUSTFLAGS="--cfg loom" cargo test --lib --release park
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
#[cfg(all(test, loom))]
|
||||||
|
mod loom_tests {
|
||||||
|
use super::*;
|
||||||
|
use loom::sync::atomic::{AtomicU64 as LAtomicU64, Ordering as O};
|
||||||
|
use loom::sync::{Arc, Mutex as LMutex};
|
||||||
|
use loom::thread;
|
||||||
|
use std::time::{Duration, Instant};
|
||||||
|
|
||||||
|
/// RFC 018 loom model 1 — no lost wake.
|
||||||
|
/// producer{publish item; wake_one} ∥ consumer{set bit; re-check; park}:
|
||||||
|
/// the consumer always observes the item or a permit; it can never
|
||||||
|
/// sleep past a published item (loom's deadlock detector is the
|
||||||
|
/// assertion — a consumer parked forever fails the model).
|
||||||
|
#[test]
|
||||||
|
fn no_lost_wake() {
|
||||||
|
loom::model(|| {
|
||||||
|
let c = Arc::new(Coordinator::new(1));
|
||||||
|
let item = Arc::new(LAtomicU64::new(0));
|
||||||
|
|
||||||
|
let prod = {
|
||||||
|
let c = c.clone();
|
||||||
|
let item = item.clone();
|
||||||
|
thread::spawn(move || {
|
||||||
|
// The enqueue shape: publish, then the fenced fast-path
|
||||||
|
// wake tail (this is what the runtime's enqueue calls).
|
||||||
|
item.store(1, O::SeqCst);
|
||||||
|
c.wake_one_if_idle();
|
||||||
|
})
|
||||||
|
};
|
||||||
|
|
||||||
|
// Consumer: loop until the item is popped. Guarded load+CAS,
|
||||||
|
// not a blind swap — a swap writes 0 even when empty, and
|
||||||
|
// coherence allows that write to land after the producer's
|
||||||
|
// store in modification order, destroying the item (a model
|
||||||
|
// bug loom caught in an earlier draft of this test).
|
||||||
|
loop {
|
||||||
|
if item.load(O::SeqCst) == 1
|
||||||
|
&& item.compare_exchange(1, 0, O::SeqCst, O::SeqCst).is_ok()
|
||||||
|
{
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
let _ = c.park(0, None, || item.load(O::SeqCst) == 1);
|
||||||
|
}
|
||||||
|
prod.join().unwrap();
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// RFC 018 loom model 2 — chain propagation.
|
||||||
|
/// Two items, two sleepers, ONE producer wake: the chain rule (a woken
|
||||||
|
/// consumer that sees surplus work and a non-empty mask wakes again)
|
||||||
|
/// must get both items consumed with no further producer action.
|
||||||
|
#[test]
|
||||||
|
fn chain_propagation() {
|
||||||
|
loom::model(|| {
|
||||||
|
let c = Arc::new(Coordinator::new(2));
|
||||||
|
let items = Arc::new(LAtomicU64::new(0));
|
||||||
|
let consumed = Arc::new(LAtomicU64::new(0));
|
||||||
|
|
||||||
|
let mut hs = Vec::new();
|
||||||
|
for id in 0..2usize {
|
||||||
|
let c = c.clone();
|
||||||
|
let items = items.clone();
|
||||||
|
let consumed = consumed.clone();
|
||||||
|
hs.push(thread::spawn(move || loop {
|
||||||
|
if consumed.load(O::SeqCst) == 2 {
|
||||||
|
c.wake_all(); // release a sibling still parked
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
let cur = items.load(O::SeqCst);
|
||||||
|
if cur > 0
|
||||||
|
&& items
|
||||||
|
.compare_exchange(cur, cur - 1, O::SeqCst, O::SeqCst)
|
||||||
|
.is_ok()
|
||||||
|
{
|
||||||
|
consumed.fetch_add(1, O::SeqCst);
|
||||||
|
// THE CHAIN RULE, exactly as production expresses it
|
||||||
|
// (runtime.rs schedule_loop): surplus ⇒ the fenced
|
||||||
|
// fast-path wake. Models the Relaxed-load chain path,
|
||||||
|
// not just the RMW one.
|
||||||
|
if items.load(O::SeqCst) > 0 {
|
||||||
|
c.wake_one_if_idle();
|
||||||
|
}
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
let _ = c.park(id, None, || {
|
||||||
|
items.load(O::SeqCst) > 0 || consumed.load(O::SeqCst) == 2
|
||||||
|
});
|
||||||
|
}));
|
||||||
|
}
|
||||||
|
|
||||||
|
// Producer (main): two items, ONE wake, via the enqueue-shaped
|
||||||
|
// fenced fast path.
|
||||||
|
items.store(2, O::SeqCst);
|
||||||
|
c.wake_one_if_idle();
|
||||||
|
|
||||||
|
for h in hs {
|
||||||
|
h.join().unwrap();
|
||||||
|
}
|
||||||
|
assert_eq!(consumed.load(O::SeqCst), 2);
|
||||||
|
assert_eq!(items.load(O::SeqCst), 0);
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// RFC 018 loom model 3 — timekeeper handoff.
|
||||||
|
/// An earlier-deadline insert racing the parking timekeeper: the
|
||||||
|
/// earlier deadline is always honored — either the timekeeper armed it
|
||||||
|
/// directly (insert landed first under the timers lock) or the insert
|
||||||
|
/// wakes the timekeeper to re-peek. A timekeeper sleeping toward the
|
||||||
|
/// stale later deadline would deadlock the model (loom has no time).
|
||||||
|
#[test]
|
||||||
|
fn timekeeper_handoff() {
|
||||||
|
loom::model(|| {
|
||||||
|
let origin = Instant::now();
|
||||||
|
let d_far = origin + Duration::from_secs(60);
|
||||||
|
let d_near = origin + Duration::from_secs(1);
|
||||||
|
|
||||||
|
let c = Arc::new(Coordinator::new(1));
|
||||||
|
// The timers-mutex stand-in: heap min under a lock.
|
||||||
|
let heap_min = Arc::new(LMutex::new(d_far));
|
||||||
|
|
||||||
|
let tk = {
|
||||||
|
let c = c.clone();
|
||||||
|
let heap_min = heap_min.clone();
|
||||||
|
thread::spawn(move || {
|
||||||
|
// Peek + arm under the lock (the serialization rule).
|
||||||
|
let armed = {
|
||||||
|
let g = heap_min.lock().unwrap();
|
||||||
|
let min = *g;
|
||||||
|
assert!(c.try_arm_timer(0, min));
|
||||||
|
min
|
||||||
|
};
|
||||||
|
if armed == d_far {
|
||||||
|
// Insert hadn't landed: it MUST wake us. Parking
|
||||||
|
// toward d_far with no wake = model deadlock.
|
||||||
|
let r = c.park(0, Some(armed), || false);
|
||||||
|
assert_eq!(r, ParkResult::Woken, "re-arm wake lost");
|
||||||
|
}
|
||||||
|
// Woken (or armed the near deadline directly): re-peek.
|
||||||
|
c.disarm_timer(0);
|
||||||
|
let g = heap_min.lock().unwrap();
|
||||||
|
assert_eq!(*g, d_near, "earlier deadline not visible on re-peek");
|
||||||
|
})
|
||||||
|
};
|
||||||
|
|
||||||
|
// Inserter (main): publish the earlier deadline under the lock,
|
||||||
|
// then the insert-check.
|
||||||
|
{
|
||||||
|
let mut g = heap_min.lock().unwrap();
|
||||||
|
*g = d_near;
|
||||||
|
c.timer_inserted(d_near);
|
||||||
|
}
|
||||||
|
|
||||||
|
tk.join().unwrap();
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// RFC 018 loom model 4 — termination.
|
||||||
|
/// The AllDone verdict: producer flips done and wake_all()s; consumers
|
||||||
|
/// must never park forever past done (park's re-check + wake_all's
|
||||||
|
/// permits close every interleaving; a stuck consumer = loom deadlock).
|
||||||
|
#[test]
|
||||||
|
fn termination_no_park_past_done() {
|
||||||
|
loom::model(|| {
|
||||||
|
let c = Arc::new(Coordinator::new(2));
|
||||||
|
let done = Arc::new(LAtomicU64::new(0));
|
||||||
|
|
||||||
|
let mut hs = Vec::new();
|
||||||
|
for id in 0..2usize {
|
||||||
|
let c = c.clone();
|
||||||
|
let done = done.clone();
|
||||||
|
hs.push(thread::spawn(move || loop {
|
||||||
|
if done.load(O::SeqCst) == 1 {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
let _ = c.park(id, None, || done.load(O::SeqCst) == 1);
|
||||||
|
}));
|
||||||
|
}
|
||||||
|
|
||||||
|
done.store(1, O::SeqCst);
|
||||||
|
c.wake_all();
|
||||||
|
|
||||||
|
for h in hs {
|
||||||
|
h.join().unwrap();
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
}
|
||||||
+7
-4
@@ -98,10 +98,13 @@ pub(crate) fn clear_current_slot() {
|
|||||||
CURRENT_SLOT.with(|c| c.set(std::ptr::null()));
|
CURRENT_SLOT.with(|c| c.set(std::ptr::null()));
|
||||||
}
|
}
|
||||||
|
|
||||||
/// RFC 007 (`smarm-causal`) — raw pointer to the on-CPU actor's slot, null on
|
/// Raw pointer to the on-CPU actor's slot, null on the scheduler's own
|
||||||
/// the scheduler's own stack. Same lifetime argument as `note_overrun`: the
|
/// stack. Same lifetime argument as `note_overrun`: the slot is never
|
||||||
/// slot is never reclaimed while its actor is on-CPU.
|
/// reclaimed while its actor is on-CPU. Consumers: the `smarm-causal`
|
||||||
#[cfg(feature = "smarm-causal")]
|
/// profiler (RFC 007) and — unconditionally — the SIGSEGV classifier
|
||||||
|
/// (RFC 019 §7), which additionally relies on this being a plain load of a
|
||||||
|
/// const-initialized TLS Cell (no lazy init, no allocation, no dtor): safe
|
||||||
|
/// from a signal handler.
|
||||||
#[inline]
|
#[inline]
|
||||||
pub(crate) fn current_slot_ptr() -> *const crate::runtime::Slot {
|
pub(crate) fn current_slot_ptr() -> *const crate::runtime::Slot {
|
||||||
CURRENT_SLOT.with(|c| c.get())
|
CURRENT_SLOT.with(|c| c.get())
|
||||||
|
|||||||
+459
-187
@@ -65,6 +65,8 @@
|
|||||||
//! word stores are `Release`, loads are `Acquire`. The chain that matters:
|
//! word stores are `Release`, loads are `Acquire`. The chain that matters:
|
||||||
//! the park path stores `sp` (Relaxed) *before* its Release transition; any
|
//! the park path stores `sp` (Relaxed) *before* its Release transition; any
|
||||||
//! later Acquire transition/load of the word therefore observes that `sp`.
|
//! later Acquire transition/load of the word therefore observes that `sp`.
|
||||||
|
//! RFC 019's `hwm` (and the shrink that reads it) piggybacks this exact
|
||||||
|
//! pattern in the same pre-Release window and adds no edges.
|
||||||
//! The run-queue mutex independently provides the same edges today; the
|
//! The run-queue mutex independently provides the same edges today; the
|
||||||
//! word's own ordering is what phase 3's lock-free queue will rely on.
|
//! word's own ordering is what phase 3's lock-free queue will rely on.
|
||||||
//!
|
//!
|
||||||
@@ -86,21 +88,28 @@
|
|||||||
//! # Termination (counter-based)
|
//! # Termination (counter-based)
|
||||||
//!
|
//!
|
||||||
//! The old all-clear scanned the slot table under the big lock. Now:
|
//! The old all-clear scanned the slot table under the big lock. Now:
|
||||||
//! exit when `io_out == 0` (read *before* the queue lock, phase-1 ordering)
|
//! exit when `io_outstanding + io_fd_waiters == 0` (two Relaxed/Acquire
|
||||||
//! and, under the queue lock, the queue is empty and `live_actors == 0`.
|
//! atomic loads, read *before* the queue pop) and, under the queue lock,
|
||||||
//! `live_actors` is incremented in `spawn` before the enqueue and decremented
|
//! the queue is empty and `live_actors == 0`. `live_actors` is incremented
|
||||||
//! at the very END of `finalize_actor`, strictly after every wakeup that
|
//! in `spawn` before the enqueue and decremented at the very END of
|
||||||
//! finalize produces has been enqueued. The soundness crux: any enqueue
|
//! `finalize_actor`, strictly after every wakeup that finalize produces has
|
||||||
//! targets a live (not-yet-finalized) actor, so `live == 0` implies no wakeup
|
//! been enqueued. The soundness crux: any enqueue targets a live
|
||||||
//! can still be in flight; combined with "spawner is itself live", observing
|
//! (not-yet-finalized) actor, so `live == 0` implies no wakeup can still be
|
||||||
//! `(queue empty, live == 0)` under the queue lock means no work can ever
|
//! in flight; combined with "spawner is itself live", observing
|
||||||
//! appear again.
|
//! `(queue empty, live == 0)` means no work can ever appear again.
|
||||||
//!
|
//!
|
||||||
//! # Timer / IO drain (try-lock, one-winner)
|
//! # Scheduler park/wake (RFC 018)
|
||||||
//!
|
//!
|
||||||
//! Unchanged from phase 1: one winner per round drains due timers and IO
|
//! Schedulers sleep on per-thread futex parkers via the coordination layer
|
||||||
//! completions from their own mutexes; wakeups go through the unpark
|
//! (`park.rs`), NOT on a shared wake pipe. IO backends are producers behind
|
||||||
//! protocol like everyone else's.
|
//! a two-call contract — make the actor runnable (`unpark_at`), whose
|
||||||
|
//! `enqueue` tail wakes exactly one parked scheduler. The blocking pool and
|
||||||
|
//! epoll thread each route their own completions (driver-enqueues); there
|
||||||
|
//! is no shared completion queue, no drain lock, no one-winner drain phase.
|
||||||
|
//! Timers fire two ways: a busy-path due-check every loop iteration (one
|
||||||
|
//! Relaxed load of the earliest-deadline snapshot when no timer is armed),
|
||||||
|
//! and the timekeeper — at most one parked scheduler holds the timer
|
||||||
|
//! deadline, so an expiry wakes one scheduler, not a herd.
|
||||||
|
|
||||||
use crate::actor::{
|
use crate::actor::{
|
||||||
clear_current_pid, is_actor_done, reset_actor_done, set_current_actor_box,
|
clear_current_pid, is_actor_done, reset_actor_done, set_current_actor_box,
|
||||||
@@ -153,6 +162,8 @@ pub struct Config {
|
|||||||
alloc_interval: u32,
|
alloc_interval: u32,
|
||||||
timeslice_cycles: u64,
|
timeslice_cycles: u64,
|
||||||
stack_pool_cap: usize,
|
stack_pool_cap: usize,
|
||||||
|
stack_reserve: usize,
|
||||||
|
stack_guard: usize,
|
||||||
max_actors: usize,
|
max_actors: usize,
|
||||||
wake_slot: bool,
|
wake_slot: bool,
|
||||||
node_id: crate::pg::NodeId,
|
node_id: crate::pg::NodeId,
|
||||||
@@ -168,6 +179,8 @@ impl Config {
|
|||||||
alloc_interval: crate::preempt::DEFAULT_ALLOC_INTERVAL,
|
alloc_interval: crate::preempt::DEFAULT_ALLOC_INTERVAL,
|
||||||
timeslice_cycles: crate::preempt::DEFAULT_TIMESLICE_CYCLES,
|
timeslice_cycles: crate::preempt::DEFAULT_TIMESLICE_CYCLES,
|
||||||
stack_pool_cap: n * 4,
|
stack_pool_cap: n * 4,
|
||||||
|
stack_reserve: DEFAULT_STACK_RESERVE,
|
||||||
|
stack_guard: DEFAULT_STACK_GUARD,
|
||||||
max_actors: DEFAULT_MAX_ACTORS,
|
max_actors: DEFAULT_MAX_ACTORS,
|
||||||
wake_slot: false,
|
wake_slot: false,
|
||||||
node_id: crate::pg::DEFAULT_NODE_ID,
|
node_id: crate::pg::DEFAULT_NODE_ID,
|
||||||
@@ -187,6 +200,8 @@ impl Config {
|
|||||||
alloc_interval: crate::preempt::DEFAULT_ALLOC_INTERVAL,
|
alloc_interval: crate::preempt::DEFAULT_ALLOC_INTERVAL,
|
||||||
timeslice_cycles: crate::preempt::DEFAULT_TIMESLICE_CYCLES,
|
timeslice_cycles: crate::preempt::DEFAULT_TIMESLICE_CYCLES,
|
||||||
stack_pool_cap: max * 4,
|
stack_pool_cap: max * 4,
|
||||||
|
stack_reserve: DEFAULT_STACK_RESERVE,
|
||||||
|
stack_guard: DEFAULT_STACK_GUARD,
|
||||||
max_actors: DEFAULT_MAX_ACTORS,
|
max_actors: DEFAULT_MAX_ACTORS,
|
||||||
wake_slot: false,
|
wake_slot: false,
|
||||||
node_id: crate::pg::DEFAULT_NODE_ID,
|
node_id: crate::pg::DEFAULT_NODE_ID,
|
||||||
@@ -219,6 +234,31 @@ impl Config {
|
|||||||
self
|
self
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Default per-actor stack reserve (RFC 019). A *virtual* reservation —
|
||||||
|
/// anonymous mmap is demand-paged, so RSS follows touched pages, not
|
||||||
|
/// this number — but overflowing it hits the guard and dies. Page-rounded.
|
||||||
|
/// Per-actor override: `SpawnOpts::stack_reserve`.
|
||||||
|
/// Default: [`DEFAULT_STACK_RESERVE`] (64 KiB) — the million-cheap-actors
|
||||||
|
/// story is unchanged; big stacks are opt-in.
|
||||||
|
pub fn stack_reserve(mut self, n: usize) -> Self {
|
||||||
|
assert!(n > 0, "stack_reserve must be non-zero");
|
||||||
|
self.stack_reserve = n;
|
||||||
|
self
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Default PROT_NONE guard below each stack (RFC 019). Address space
|
||||||
|
/// only. Page-rounded. Rust overflow is caught by any single page
|
||||||
|
/// (probestack touches pages in order); the wide default exists for
|
||||||
|
/// unprobed FFI frames, which can step over a small guard in one
|
||||||
|
/// `sub rsp`. Per-actor override: `SpawnOpts::guard_size`.
|
||||||
|
/// Default: [`DEFAULT_STACK_GUARD`] (1 MiB — the kernel's
|
||||||
|
/// `stack_guard_gap` convention; see its doc for why width is free).
|
||||||
|
pub fn stack_guard(mut self, n: usize) -> Self {
|
||||||
|
assert!(n > 0, "stack_guard must be non-zero");
|
||||||
|
self.stack_guard = n;
|
||||||
|
self
|
||||||
|
}
|
||||||
|
|
||||||
/// Capacity of the actor slot table — the maximum number of
|
/// Capacity of the actor slot table — the maximum number of
|
||||||
/// **simultaneously live** actors (total spawned over a run is unbounded;
|
/// **simultaneously live** actors (total spawned over a run is unbounded;
|
||||||
/// slots are recycled). The table is a fixed slab allocated once at
|
/// slots are recycled). The table is a fixed slab allocated once at
|
||||||
@@ -286,6 +326,8 @@ impl Default for Config {
|
|||||||
alloc_interval: crate::preempt::DEFAULT_ALLOC_INTERVAL,
|
alloc_interval: crate::preempt::DEFAULT_ALLOC_INTERVAL,
|
||||||
timeslice_cycles: crate::preempt::DEFAULT_TIMESLICE_CYCLES,
|
timeslice_cycles: crate::preempt::DEFAULT_TIMESLICE_CYCLES,
|
||||||
stack_pool_cap: avail * 4,
|
stack_pool_cap: avail * 4,
|
||||||
|
stack_reserve: DEFAULT_STACK_RESERVE,
|
||||||
|
stack_guard: DEFAULT_STACK_GUARD,
|
||||||
max_actors: DEFAULT_MAX_ACTORS,
|
max_actors: DEFAULT_MAX_ACTORS,
|
||||||
wake_slot: false,
|
wake_slot: false,
|
||||||
node_id: crate::pg::DEFAULT_NODE_ID,
|
node_id: crate::pg::DEFAULT_NODE_ID,
|
||||||
@@ -376,7 +418,48 @@ impl RuntimeStats {
|
|||||||
// Slot — packed state word + hot atomics + cold lifecycle data
|
// Slot — packed state word + hot atomics + cold lifecycle data
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
pub(crate) const ACTOR_STACK_SIZE: usize = 64 * 1024;
|
/// Default usable stack reserve per actor (RFC 019). See [`Config::stack_reserve`].
|
||||||
|
pub const DEFAULT_STACK_RESERVE: usize = 64 * 1024;
|
||||||
|
|
||||||
|
/// Default PROT_NONE guard below each actor stack (RFC 019). Raised from one
|
||||||
|
/// page so unprobed C frames cannot leap it. See [`Config::stack_guard`].
|
||||||
|
///
|
||||||
|
/// 1 MiB, following the kernel's own answer to the same problem: after Stack
|
||||||
|
/// Clash (2017) the main-thread guard gap became `stack_guard_gap` = 256
|
||||||
|
/// pages, because 4 KiB was jumpable by one honest `sub rsp` and no small
|
||||||
|
/// constant was defensible. Guard pages are PROT_NONE: virtual address space
|
||||||
|
/// only — zero RSS, zero page-table entries, no overcommit charge — so the
|
||||||
|
/// wide default is free at any actor count (1 M actors ≈ 1 TiB of VA against
|
||||||
|
/// a 128 TiB budget). A frame that jumps even this lands in the tier-2
|
||||||
|
/// overshoot window of the SIGSEGV diagnostic (`signal.rs`) instead of
|
||||||
|
/// silence.
|
||||||
|
pub const DEFAULT_STACK_GUARD: usize = 1024 * 1024;
|
||||||
|
|
||||||
|
/// RFC 019 §3: minimum releasable span (`sp − hwm` at park) before the
|
||||||
|
/// park-path shrink spends a syscall. A constant, not a `Config` field
|
||||||
|
/// (ratified): nobody tunes this well and the measured stakes are low — a
|
||||||
|
/// threshold-sized `MADV_FREE` costs ~3 µs against a ~100 ns park, paid
|
||||||
|
/// only on spike-recovery parks, which are rare by construction and *were*
|
||||||
|
/// the spike. Steady-state actors never reach the syscall: their check is
|
||||||
|
/// two Relaxed loads and a compare on a line the context-save just wrote.
|
||||||
|
pub const SHRINK_THRESHOLD: usize = 256 * 1024;
|
||||||
|
|
||||||
|
/// RFC 019 §3: parks between shrinks of one actor. Guards a few-µs cost, so
|
||||||
|
/// it can be coarse (parks, not wall time); the kernel's
|
||||||
|
/// reclaim-under-pressure-only handling of `MADV_FREE` is the real release
|
||||||
|
/// hysteresis — re-touched-before-pressure pages cost a 0.24 µs/page
|
||||||
|
/// cancel-write and no fault. A constant, not `Config` (ratified, same
|
||||||
|
/// rationale as [`SHRINK_THRESHOLD`]).
|
||||||
|
pub const SHRINK_COOLDOWN: u32 = 64;
|
||||||
|
|
||||||
|
/// RFC 019 §6: the entry-end span (highest addresses — the frames the next
|
||||||
|
/// actor faults first) a recycled stack keeps resident; everything below it
|
||||||
|
/// is `MADV_DONTNEED`ed before the stack re-enters the pool. Ratified as a
|
||||||
|
/// constant, not Config, alongside the shrink knobs; the 64 KiB value was a
|
||||||
|
/// flagged Claude-solo call at ratification — it equals the default reserve,
|
||||||
|
/// so with an unraised Config the zap is a no-op and only Configs that raise
|
||||||
|
/// the default reserve pay it.
|
||||||
|
pub const RECYCLE_RETAIN: usize = 64 * 1024;
|
||||||
|
|
||||||
pub(crate) type Closure = Box<dyn FnOnce() + Send>;
|
pub(crate) type Closure = Box<dyn FnOnce() + Send>;
|
||||||
|
|
||||||
@@ -419,6 +502,35 @@ pub(crate) struct Slot {
|
|||||||
/// Release transition out of Running; read after the Acquire transition
|
/// Release transition out of Running; read after the Acquire transition
|
||||||
/// Queued→Running. Relaxed is sufficient — ordering rides on `word`.
|
/// Queued→Running. Relaxed is sufficient — ordering rides on `word`.
|
||||||
sp: AtomicUsize,
|
sp: AtomicUsize,
|
||||||
|
/// RFC 019: sampled stack high-water — the minimum `sp` ever stored above,
|
||||||
|
/// i.e. the deepest excursion *observed at a switch point*. Advisory:
|
||||||
|
/// correctness never depends on it; its one job is "is a shrink worth a
|
||||||
|
/// syscall?". Declared adjacent to `sp` so the min-update dirties the
|
||||||
|
/// line the context-save just wrote. Same single-writer Relaxed
|
||||||
|
/// discipline as `sp`; reset to the fresh `sp` at install.
|
||||||
|
hwm: AtomicUsize,
|
||||||
|
/// RFC 019: parks since the last shrink (or install). Counted on every
|
||||||
|
/// pass through the Park arm by the owning scheduler thread; the shrink
|
||||||
|
/// fires only once this clears [`SHRINK_COOLDOWN`] *and* the releasable
|
||||||
|
/// span clears [`SHRINK_THRESHOLD`]. Single-writer Relaxed.
|
||||||
|
parks_since_shrink: AtomicU32,
|
||||||
|
/// RFC 019: shrinks performed on this incarnation (introspection lands
|
||||||
|
/// with the RFC's introspect surface; the counter exists from birth so
|
||||||
|
/// tests can rely on install resetting it). Single-writer Relaxed.
|
||||||
|
shrink_count: AtomicU32,
|
||||||
|
/// RFC 019 §7 — stack geometry for the SIGSEGV classifier, readable
|
||||||
|
/// without the cold lock (the `Stack` itself lives under it). Written in
|
||||||
|
/// `install_actor` before the Release publish; consulted by the handler
|
||||||
|
/// only while `preempt::CURRENT_SLOT` points here, i.e. while this actor
|
||||||
|
/// is on-CPU, so the values are never stale where they are read. 0 =
|
||||||
|
/// never installed. Usable top of the stack.
|
||||||
|
pub(crate) diag_stack_top: AtomicUsize,
|
||||||
|
/// See `diag_stack_top`: the reserve (usable) size.
|
||||||
|
pub(crate) diag_stack_reserve: AtomicUsize,
|
||||||
|
/// See `diag_stack_top`: the guard size.
|
||||||
|
pub(crate) diag_stack_guard: AtomicUsize,
|
||||||
|
/// See `diag_stack_top`: `(idx << 32) | generation`, for the message.
|
||||||
|
pub(crate) diag_pid: AtomicU64,
|
||||||
/// Pointer into the actor's `Arc<AtomicBool>` stop flag. Set at spawn,
|
/// Pointer into the actor's `Arc<AtomicBool>` stop flag. Set at spawn,
|
||||||
/// nulled at finalize. The box outlives every read: it is only ever read
|
/// nulled at finalize. The box outlives every read: it is only ever read
|
||||||
/// on the resume path while the actor cannot be finalized (it is on-CPU).
|
/// on the resume path while the actor cannot be finalized (it is on-CPU).
|
||||||
@@ -490,6 +602,13 @@ impl Slot {
|
|||||||
Self {
|
Self {
|
||||||
word: StateWord::new(),
|
word: StateWord::new(),
|
||||||
sp: AtomicUsize::new(0),
|
sp: AtomicUsize::new(0),
|
||||||
|
hwm: AtomicUsize::new(0),
|
||||||
|
parks_since_shrink: AtomicU32::new(0),
|
||||||
|
shrink_count: AtomicU32::new(0),
|
||||||
|
diag_stack_top: AtomicUsize::new(0),
|
||||||
|
diag_stack_reserve: AtomicUsize::new(0),
|
||||||
|
diag_stack_guard: AtomicUsize::new(0),
|
||||||
|
diag_pid: AtomicU64::new(0),
|
||||||
stop_ptr: AtomicPtr::new(std::ptr::null_mut()),
|
stop_ptr: AtomicPtr::new(std::ptr::null_mut()),
|
||||||
closure: AtomicPtr::new(std::ptr::null_mut()),
|
closure: AtomicPtr::new(std::ptr::null_mut()),
|
||||||
overruns: AtomicU64::new(0),
|
overruns: AtomicU64::new(0),
|
||||||
@@ -543,6 +662,23 @@ impl Slot {
|
|||||||
|
|
||||||
/// Read the overrun tally (Relaxed; the snapshot reads cross-thread).
|
/// Read the overrun tally (Relaxed; the snapshot reads cross-thread).
|
||||||
#[inline]
|
#[inline]
|
||||||
|
/// RFC 019 §8 — the stack introspection tuple, all lock-free:
|
||||||
|
/// `(reserve, guard, top, hwm, parks_since_shrink, shrink_count)`.
|
||||||
|
/// Geometry from the c6 diag atomics (install-time, gen-coherent under
|
||||||
|
/// `read_slot`'s gen check exactly like the other counters); `hwm` is the
|
||||||
|
/// §2 sampled high-water (lowest saved sp). All zeros before first
|
||||||
|
/// install.
|
||||||
|
pub(crate) fn stack_introspect(&self) -> (usize, usize, usize, usize, u32, u32) {
|
||||||
|
(
|
||||||
|
self.diag_stack_reserve.load(Ordering::Relaxed),
|
||||||
|
self.diag_stack_guard.load(Ordering::Relaxed),
|
||||||
|
self.diag_stack_top.load(Ordering::Relaxed),
|
||||||
|
self.hwm.load(Ordering::Relaxed),
|
||||||
|
self.parks_since_shrink.load(Ordering::Relaxed),
|
||||||
|
self.shrink_count.load(Ordering::Relaxed),
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
pub(crate) fn overruns(&self) -> u64 {
|
pub(crate) fn overruns(&self) -> u64 {
|
||||||
self.overruns.load(Ordering::Relaxed)
|
self.overruns.load(Ordering::Relaxed)
|
||||||
}
|
}
|
||||||
@@ -750,8 +886,21 @@ pub(crate) struct RuntimeInner {
|
|||||||
pub(crate) io: Mutex<Option<IoThread>>,
|
pub(crate) io: Mutex<Option<IoThread>>,
|
||||||
/// Monotonic `MonitorId` source. Never reused.
|
/// Monotonic `MonitorId` source. Never reused.
|
||||||
pub(crate) next_monitor_id: AtomicU64,
|
pub(crate) next_monitor_id: AtomicU64,
|
||||||
/// Try-lock: exactly one scheduler thread drains timers/IO per iteration.
|
/// RFC 018: the scheduler coordination layer — per-scheduler parkers,
|
||||||
drain_lock: Mutex<()>,
|
/// idle mask, wake protocol, timekeeper role, earliest-deadline
|
||||||
|
/// snapshot. Arc'd because `Timers` shares it (insert-side deadline
|
||||||
|
/// notes run under the timers mutex).
|
||||||
|
pub(crate) coord: Arc<crate::park::Coordinator>,
|
||||||
|
/// `block_on_io` requests in flight. Incremented by the submitter
|
||||||
|
/// BEFORE submit (underflow-proof), decremented by the pool thread on
|
||||||
|
/// completion. Read lock-free by the idle path's termination verdict —
|
||||||
|
/// the per-pop `io.lock` of the drain era is gone.
|
||||||
|
pub(crate) io_outstanding: AtomicU32,
|
||||||
|
/// Parked fd waiters. Incremented by the registrar BEFORE
|
||||||
|
/// `epoll_register` (rolled back on error), decremented by whoever
|
||||||
|
/// consumes the registration (epoll thread on readiness, canceller on
|
||||||
|
/// an unwound wait). Same lock-free verdict read as `io_outstanding`.
|
||||||
|
pub(crate) io_fd_waiters: AtomicU32,
|
||||||
/// Per-thread stats, indexed by scheduler thread slot (0..N).
|
/// Per-thread stats, indexed by scheduler thread slot (0..N).
|
||||||
pub(crate) stats: Vec<SchedulerStats>,
|
pub(crate) stats: Vec<SchedulerStats>,
|
||||||
/// Global counters for RFC 000 primitives.
|
/// Global counters for RFC 000 primitives.
|
||||||
@@ -782,17 +931,23 @@ pub(crate) struct RuntimeInner {
|
|||||||
pub(crate) stack_pool: RawMutex<Vec<crate::stack::Stack>>,
|
pub(crate) stack_pool: RawMutex<Vec<crate::stack::Stack>>,
|
||||||
/// Maximum number of stacks to retain in the pool.
|
/// Maximum number of stacks to retain in the pool.
|
||||||
pub(crate) stack_pool_cap: usize,
|
pub(crate) stack_pool_cap: usize,
|
||||||
|
/// Default stack shape (RFC 019), pre-page-rounded so it compares exactly
|
||||||
|
/// against `Stack::shape()`. Only stacks of exactly this shape are pooled.
|
||||||
|
pub(crate) stack_reserve: usize,
|
||||||
|
pub(crate) stack_guard: usize,
|
||||||
}
|
}
|
||||||
|
|
||||||
impl RuntimeInner {
|
impl RuntimeInner {
|
||||||
// Private constructor taking the parsed Config fields one-for-one; a params
|
// Private constructor taking the parsed Config fields one-for-one; a params
|
||||||
// struct would only move the same 8 values across the call boundary.
|
// struct would only move the same 10 values across the call boundary.
|
||||||
#[allow(clippy::too_many_arguments)]
|
#[allow(clippy::too_many_arguments)]
|
||||||
fn new(
|
fn new(
|
||||||
thread_count: usize,
|
thread_count: usize,
|
||||||
alloc_interval: u32,
|
alloc_interval: u32,
|
||||||
timeslice_cycles: u64,
|
timeslice_cycles: u64,
|
||||||
stack_pool_cap: usize,
|
stack_pool_cap: usize,
|
||||||
|
stack_reserve: usize,
|
||||||
|
stack_guard: usize,
|
||||||
max_actors: usize,
|
max_actors: usize,
|
||||||
wake_slot: bool,
|
wake_slot: bool,
|
||||||
node_id: crate::pg::NodeId,
|
node_id: crate::pg::NodeId,
|
||||||
@@ -802,6 +957,12 @@ impl RuntimeInner {
|
|||||||
let slots: Box<[Slot]> = (0..max_actors).map(|_| Slot::vacant()).collect();
|
let slots: Box<[Slot]> = (0..max_actors).map(|_| Slot::vacant()).collect();
|
||||||
// Low indices on top of the stack so early spawns get low pids.
|
// Low indices on top of the stack so early spawns get low pids.
|
||||||
let free: Vec<u32> = (0..max_actors as u32).rev().collect();
|
let free: Vec<u32> = (0..max_actors as u32).rev().collect();
|
||||||
|
// RFC 018: the coordination layer (asserts thread_count <= 64), and
|
||||||
|
// the timers' hook into it — every insert under the timers mutex
|
||||||
|
// notes its deadline (busy-path snapshot + timekeeper re-arm).
|
||||||
|
let coord = Arc::new(crate::park::Coordinator::new(thread_count));
|
||||||
|
let mut timers = Timers::new();
|
||||||
|
timers.attach_coordinator(coord.clone());
|
||||||
Arc::new(Self {
|
Arc::new(Self {
|
||||||
run_queue: crate::run_queue::RunQueue::new(thread_count, max_actors),
|
run_queue: crate::run_queue::RunQueue::new(thread_count, max_actors),
|
||||||
slots,
|
slots,
|
||||||
@@ -810,10 +971,12 @@ impl RuntimeInner {
|
|||||||
root_bits: AtomicU64::new(u64::MAX),
|
root_bits: AtomicU64::new(u64::MAX),
|
||||||
root_exited: AtomicBool::new(false),
|
root_exited: AtomicBool::new(false),
|
||||||
root_swept: AtomicBool::new(false),
|
root_swept: AtomicBool::new(false),
|
||||||
timers: Mutex::new(Timers::new()),
|
timers: Mutex::new(timers),
|
||||||
io: Mutex::new(None),
|
io: Mutex::new(None),
|
||||||
next_monitor_id: AtomicU64::new(0),
|
next_monitor_id: AtomicU64::new(0),
|
||||||
drain_lock: Mutex::new(()),
|
coord,
|
||||||
|
io_outstanding: AtomicU32::new(0),
|
||||||
|
io_fd_waiters: AtomicU32::new(0),
|
||||||
stats,
|
stats,
|
||||||
io_parked: AtomicU32::new(0),
|
io_parked: AtomicU32::new(0),
|
||||||
sleeping: AtomicU32::new(0),
|
sleeping: AtomicU32::new(0),
|
||||||
@@ -826,6 +989,8 @@ impl RuntimeInner {
|
|||||||
process_groups: RawMutex::new(crate::pg::ProcessGroups::new()),
|
process_groups: RawMutex::new(crate::pg::ProcessGroups::new()),
|
||||||
stack_pool: RawMutex::new(Vec::new()),
|
stack_pool: RawMutex::new(Vec::new()),
|
||||||
stack_pool_cap,
|
stack_pool_cap,
|
||||||
|
stack_reserve: crate::stack::round_to_pages(stack_reserve),
|
||||||
|
stack_guard: crate::stack::round_to_pages(stack_guard),
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -874,6 +1039,13 @@ impl RuntimeInner {
|
|||||||
);
|
);
|
||||||
self.run_queue.push(pid);
|
self.run_queue.push(pid);
|
||||||
crate::te!(crate::trace::Event::Enqueue(pid));
|
crate::te!(crate::trace::Event::Enqueue(pid));
|
||||||
|
// RFC 018 enqueue wake (fixes the silent enqueue): if a scheduler
|
||||||
|
// is parked, wake exactly one. The fast path when everyone is busy
|
||||||
|
// is a fence + one Relaxed load of an unmodified line — the
|
||||||
|
// pure-compute hot path pays (almost) nothing. Bias is over-wake:
|
||||||
|
// a spurious wake costs one futex round-trip and a failed pop; a
|
||||||
|
// missed wake would cost a stranded actor.
|
||||||
|
self.coord.wake_one_if_idle();
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Make `pid` runnable if it is parked; coalesce or defer otherwise.
|
/// Make `pid` runnable if it is parked; coalesce or defer otherwise.
|
||||||
@@ -1010,6 +1182,9 @@ pub struct Runtime {
|
|||||||
|
|
||||||
/// Initialise the runtime with the given config. Returns a reusable handle.
|
/// Initialise the runtime with the given config. Returns a reusable handle.
|
||||||
pub fn init(config: Config) -> Runtime {
|
pub fn init(config: Config) -> Runtime {
|
||||||
|
// RFC 019 §7: one process-global SIGSEGV handler, installed before any
|
||||||
|
// scheduler thread (and so before any classifiable fault) can exist.
|
||||||
|
crate::signal::install_once();
|
||||||
let n = config.resolved_thread_count();
|
let n = config.resolved_thread_count();
|
||||||
Runtime {
|
Runtime {
|
||||||
inner: RuntimeInner::new(
|
inner: RuntimeInner::new(
|
||||||
@@ -1017,6 +1192,8 @@ pub fn init(config: Config) -> Runtime {
|
|||||||
config.alloc_interval,
|
config.alloc_interval,
|
||||||
config.timeslice_cycles,
|
config.timeslice_cycles,
|
||||||
config.stack_pool_cap,
|
config.stack_pool_cap,
|
||||||
|
config.stack_reserve,
|
||||||
|
config.stack_guard,
|
||||||
config.max_actors,
|
config.max_actors,
|
||||||
config.wake_slot,
|
config.wake_slot,
|
||||||
config.node_id,
|
config.node_id,
|
||||||
@@ -1076,7 +1253,16 @@ impl Runtime {
|
|||||||
self.inner.live_actors.load(Ordering::Acquire), 0,
|
self.inner.live_actors.load(Ordering::Acquire), 0,
|
||||||
"run() called while previous run still active"
|
"run() called while previous run still active"
|
||||||
);
|
);
|
||||||
let io_thread = match IoThread::start() {
|
// RFC 018: the IO producers reach the runtime (slot table + unpark)
|
||||||
|
// through a Weak, so no RuntimeInner → IoThread → RuntimeInner cycle
|
||||||
|
// forms. Reset the in-flight counters BEFORE the threads can touch
|
||||||
|
// them (a prior run left them at 0 on a clean exit; the asserts pin
|
||||||
|
// that).
|
||||||
|
debug_assert_eq!(self.inner.io_outstanding.load(Ordering::Acquire), 0);
|
||||||
|
debug_assert_eq!(self.inner.io_fd_waiters.load(Ordering::Acquire), 0);
|
||||||
|
self.inner.io_outstanding.store(0, Ordering::Release);
|
||||||
|
self.inner.io_fd_waiters.store(0, Ordering::Release);
|
||||||
|
let io_thread = match IoThread::start(Arc::downgrade(&self.inner)) {
|
||||||
Ok(io) => io,
|
Ok(io) => io,
|
||||||
Err(e) => panic!("failed to start IO thread: {e}"),
|
Err(e) => panic!("failed to start IO thread: {e}"),
|
||||||
};
|
};
|
||||||
@@ -1179,6 +1365,8 @@ impl Runtime {
|
|||||||
}
|
}
|
||||||
self.inner.io_parked.store(0, Ordering::Relaxed);
|
self.inner.io_parked.store(0, Ordering::Relaxed);
|
||||||
self.inner.sleeping.store(0, Ordering::Relaxed);
|
self.inner.sleeping.store(0, Ordering::Relaxed);
|
||||||
|
self.inner.io_outstanding.store(0, Ordering::Relaxed);
|
||||||
|
self.inner.io_fd_waiters.store(0, Ordering::Relaxed);
|
||||||
|
|
||||||
RUNTIME.with(|r| *r.borrow_mut() = None);
|
RUNTIME.with(|r| *r.borrow_mut() = None);
|
||||||
|
|
||||||
@@ -1263,6 +1451,99 @@ pub const ROOT_PID: Pid = Pid::new(u32::MAX, u32::MAX);
|
|||||||
// Spawn-side slot installation
|
// Spawn-side slot installation
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Stack shrink — RFC 019 §3 (park path only)
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
/// The per-park shrink check. Called from the `YieldIntent::Park` arm inside
|
||||||
|
/// the owned window (see the assert-comment at the call site). Fast path —
|
||||||
|
/// no spike since the last shrink — is two Relaxed loads, a compare, and the
|
||||||
|
/// park counter bump, all on the slot line the context-save just wrote.
|
||||||
|
///
|
||||||
|
/// On a shrink: `MADV_FREE` the whole pages of `[hwm, sp − redzone)` (the
|
||||||
|
/// inward-rounded range from [`crate::stack::shrink_range`]), then reset
|
||||||
|
/// `hwm = sp` and the park counter. MADV_FREE only *marks*: the kernel
|
||||||
|
/// reclaims under pressure, skips re-dirtied pages, and refaults zero pages
|
||||||
|
/// for writes after reclaim — so an over-eager mark costs a cancel-write,
|
||||||
|
/// never data.
|
||||||
|
fn maybe_shrink_stack(slot: &Slot) {
|
||||||
|
let parks = slot.parks_since_shrink.load(Ordering::Relaxed).saturating_add(1);
|
||||||
|
slot.parks_since_shrink.store(parks, Ordering::Relaxed);
|
||||||
|
|
||||||
|
let sp = slot.sp.load(Ordering::Relaxed);
|
||||||
|
let hwm = slot.hwm.load(Ordering::Relaxed);
|
||||||
|
if sp.wrapping_sub(hwm) < SHRINK_THRESHOLD || sp < hwm {
|
||||||
|
return; // common case: nothing worth a syscall
|
||||||
|
}
|
||||||
|
if parks < SHRINK_COOLDOWN {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
let page = crate::stack::page_size();
|
||||||
|
if let Some((addr, len)) = crate::stack::shrink_range(hwm, sp, page) {
|
||||||
|
// Advisory: on the (kernel-config) chance MADV_FREE is unsupported,
|
||||||
|
// failing silently degrades to "never shrinks", which is correct.
|
||||||
|
unsafe {
|
||||||
|
libc::madvise(addr as *mut libc::c_void, len, libc::MADV_FREE);
|
||||||
|
}
|
||||||
|
slot.hwm.store(sp, Ordering::Relaxed);
|
||||||
|
slot.parks_since_shrink.store(0, Ordering::Relaxed);
|
||||||
|
slot.shrink_count.store(
|
||||||
|
slot.shrink_count.load(Ordering::Relaxed).saturating_add(1),
|
||||||
|
Ordering::Relaxed,
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Stack acquisition / recycling — RFC 019 pool rule
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
/// Get a stack of the shape `opts` requests (`None` fields ⇒ the runtime
|
||||||
|
/// defaults).
|
||||||
|
///
|
||||||
|
/// Pool rule (RFC 019 §1): the pool is a uniform `Vec<Stack>` of
|
||||||
|
/// default-shaped stacks and stays that way. Default-shaped requests try the
|
||||||
|
/// pool first; custom shapes always mmap fresh (and `recycle_stack` never
|
||||||
|
/// admits them, so a pooled stack is default-shaped by induction). The pool
|
||||||
|
/// lock is dropped before any mmap: no syscall ever stalls another spawner.
|
||||||
|
pub(crate) fn acquire_stack(
|
||||||
|
inner: &RuntimeInner,
|
||||||
|
opts: crate::scheduler::SpawnOpts,
|
||||||
|
) -> crate::stack::Stack {
|
||||||
|
let reserve = opts.stack_reserve.unwrap_or(inner.stack_reserve);
|
||||||
|
let guard = opts.guard_size.unwrap_or(inner.stack_guard);
|
||||||
|
let default_shaped = crate::stack::round_to_pages(reserve) == inner.stack_reserve
|
||||||
|
&& crate::stack::round_to_pages(guard) == inner.stack_guard;
|
||||||
|
if default_shaped {
|
||||||
|
if let Some(stack) = inner.stack_pool.lock().pop() {
|
||||||
|
return stack;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
match crate::stack::Stack::new(reserve, guard) {
|
||||||
|
Ok(stack) => stack,
|
||||||
|
Err(e) => panic!("stack allocation failed: {e}"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Return a dead actor's stack: pooled if default-shaped and under cap,
|
||||||
|
/// otherwise dropped here → munmap (custom shapes and cap overflow alike).
|
||||||
|
pub(crate) fn recycle_stack(inner: &RuntimeInner, stack: crate::stack::Stack) {
|
||||||
|
if stack.shape() == (inner.stack_reserve, inner.stack_guard) {
|
||||||
|
// RFC 019 §6: zap the dead spike before pooling, BEFORE taking the
|
||||||
|
// pool lock — acquire_stack's invariant is that no syscall ever
|
||||||
|
// stalls another spawner under it. On the rare cap-overflow the zap
|
||||||
|
// is wasted work ahead of the munmap; harmless, and cheaper than a
|
||||||
|
// second lock round-trip to find out.
|
||||||
|
stack.recycle_zap(RECYCLE_RETAIN);
|
||||||
|
let mut pool = inner.stack_pool.lock();
|
||||||
|
if pool.len() < inner.stack_pool_cap {
|
||||||
|
pool.push(stack);
|
||||||
|
}
|
||||||
|
// else: fall through — drop → munmap.
|
||||||
|
}
|
||||||
|
// Custom-shaped (or cap overflow): `stack` drops here → munmap.
|
||||||
|
}
|
||||||
|
|
||||||
/// Install a freshly spawned actor into the slot `idx` (which must have come
|
/// Install a freshly spawned actor into the slot `idx` (which must have come
|
||||||
/// from `allocate_slot`) and publish it as Queued. Returns the new `Pid`.
|
/// from `allocate_slot`) and publish it as Queued. Returns the new `Pid`.
|
||||||
/// Called by `scheduler::spawn_under`; lives here next to its inverse
|
/// Called by `scheduler::spawn_under`; lives here next to its inverse
|
||||||
@@ -1280,6 +1561,11 @@ pub(crate) fn install_actor(
|
|||||||
let pid = Pid::new(idx, gen);
|
let pid = Pid::new(idx, gen);
|
||||||
|
|
||||||
let stop = Arc::new(AtomicBool::new(false));
|
let stop = Arc::new(AtomicBool::new(false));
|
||||||
|
// RFC 019 §7: geometry for the SIGSEGV classifier, captured before the
|
||||||
|
// Stack moves under the cold lock. Ordered before readers by the
|
||||||
|
// publish below.
|
||||||
|
let (diag_reserve, diag_guard) = stack.shape();
|
||||||
|
let diag_top = stack.top() as usize;
|
||||||
slot.stop_ptr.store(Arc::as_ptr(&stop) as *mut _, Ordering::Release);
|
slot.stop_ptr.store(Arc::as_ptr(&stop) as *mut _, Ordering::Release);
|
||||||
{
|
{
|
||||||
let mut cold = slot.cold.lock();
|
let mut cold = slot.cold.lock();
|
||||||
@@ -1291,6 +1577,15 @@ pub(crate) fn install_actor(
|
|||||||
cold.pending_io_result = None;
|
cold.pending_io_result = None;
|
||||||
}
|
}
|
||||||
slot.sp.store(sp, Ordering::Relaxed);
|
slot.sp.store(sp, Ordering::Relaxed);
|
||||||
|
// RFC 019: a fresh incarnation starts with its high-water at the fresh
|
||||||
|
// top-of-stack `sp` and its shrink bookkeeping zeroed.
|
||||||
|
slot.hwm.store(sp, Ordering::Relaxed);
|
||||||
|
slot.parks_since_shrink.store(0, Ordering::Relaxed);
|
||||||
|
slot.shrink_count.store(0, Ordering::Relaxed);
|
||||||
|
slot.diag_stack_top.store(diag_top, Ordering::Relaxed);
|
||||||
|
slot.diag_stack_reserve.store(diag_reserve, Ordering::Relaxed);
|
||||||
|
slot.diag_stack_guard.store(diag_guard, Ordering::Relaxed);
|
||||||
|
slot.diag_pid.store(((idx as u64) << 32) | gen as u64, Ordering::Relaxed);
|
||||||
slot.store_closure(closure);
|
slot.store_closure(closure);
|
||||||
slot.reset_counters();
|
slot.reset_counters();
|
||||||
inner.live_actors.fetch_add(1, Ordering::Relaxed);
|
inner.live_actors.fetch_add(1, Ordering::Relaxed);
|
||||||
@@ -1391,13 +1686,7 @@ fn finalize_actor(inner: &Arc<RuntimeInner>, pid: Pid, outcome: Outcome) {
|
|||||||
// (the trap sender can unpark its receiver — keep that outside too).
|
// (the trap sender can unpark its receiver — keep that outside too).
|
||||||
let supervisor_pid = actor.supervisor;
|
let supervisor_pid = actor.supervisor;
|
||||||
let Actor { stack, .. } = actor;
|
let Actor { stack, .. } = actor;
|
||||||
{
|
recycle_stack(inner, stack);
|
||||||
let mut pool = inner.stack_pool.lock();
|
|
||||||
if pool.len() < inner.stack_pool_cap {
|
|
||||||
pool.push(stack);
|
|
||||||
}
|
|
||||||
// else: drop here → munmap, same as before
|
|
||||||
}
|
|
||||||
|
|
||||||
// Deliver to supervisor. ROOT_PID resolves to no slot → silently absorbed.
|
// Deliver to supervisor. ROOT_PID resolves to no slot → silently absorbed.
|
||||||
let sender = inner.slot_at(supervisor_pid).and_then(|sup| {
|
let sender = inner.slot_at(supervisor_pid).and_then(|sup| {
|
||||||
@@ -1494,53 +1783,30 @@ fn stop_live_actors(inner: &Arc<RuntimeInner>) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
// schedule_loop — runs on each scheduler OS thread
|
// Timer firing — shared by the busy-path due-check and the timekeeper
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
fn schedule_loop(inner: &Arc<RuntimeInner>, slot_idx: usize) {
|
/// Pop and dispatch every due timer. `pop_due` re-anchors the
|
||||||
crate::preempt::configure_preempt(inner.alloc_interval, inner.timeslice_cycles);
|
/// earliest-deadline snapshot under the timers mutex before returning, so
|
||||||
let stats = &inner.stats[slot_idx];
|
/// a caller that raced a concurrent insert simply comes back on the next
|
||||||
|
/// due-check. Dispatch runs with the timers lock released.
|
||||||
loop {
|
fn fire_due_timers(inner: &Arc<RuntimeInner>, try_only: bool) {
|
||||||
// ----------------------------------------------------------------
|
let due = if try_only {
|
||||||
// 1. Try to win the drain lock (timers + IO). One winner per round;
|
// Busy path: if another scheduler is already in the timers mutex
|
||||||
// losers skip immediately and proceed to step 2.
|
// (firing, inserting, or peeking) skip — the snapshot stays due
|
||||||
// ----------------------------------------------------------------
|
// until someone actually pops, so the check re-fires next loop.
|
||||||
if let Ok(_drain_guard) = inner.drain_lock.try_lock() {
|
match inner.timers.try_lock() {
|
||||||
// Timers and IO live behind their own mutexes (phase 1), so the
|
Ok(mut t) => t.pop_due(std::time::Instant::now()),
|
||||||
// pure-yield / pure-compute hot path never contends a global lock
|
Err(std::sync::TryLockError::WouldBlock) => return,
|
||||||
// just to discover there is nothing to drain. The clock is read
|
Err(std::sync::TryLockError::Poisoned(e)) => {
|
||||||
// only when the timer heap is non-empty.
|
panic!("smarm: timers lock poisoned (core corrupt): {e}")
|
||||||
let due = {
|
}
|
||||||
let mut t = match inner.timers.lock() {
|
}
|
||||||
Ok(t) => t,
|
} else {
|
||||||
|
match inner.timers.lock() {
|
||||||
|
Ok(mut t) => t.pop_due(std::time::Instant::now()),
|
||||||
Err(e) => panic!("smarm: timers lock poisoned (core corrupt): {e}"),
|
Err(e) => panic!("smarm: timers lock poisoned (core corrupt): {e}"),
|
||||||
};
|
|
||||||
if t.is_empty() {
|
|
||||||
Vec::new()
|
|
||||||
} else {
|
|
||||||
t.pop_due(std::time::Instant::now())
|
|
||||||
}
|
}
|
||||||
};
|
|
||||||
let completions = match inner.io.lock() {
|
|
||||||
Ok(mut io) => io
|
|
||||||
.as_mut()
|
|
||||||
.map(|io| {
|
|
||||||
// Consume wake-pipe bytes ONLY here, under the drain
|
|
||||||
// lock and strictly before draining completions.
|
|
||||||
// Producers push their completion before writing the
|
|
||||||
// byte, so every byte consumed here has its completion
|
|
||||||
// visible to the drain below. Consuming bytes anywhere
|
|
||||||
// else — in particular after an idle poll, outside the
|
|
||||||
// lock — loses wakeups: a try_lock loser can eat the
|
|
||||||
// byte for a completion the winner never saw, leaving
|
|
||||||
// it stranded (and its EPOLLONESHOT fd disarmed) until
|
|
||||||
// an unrelated timer forces another drain pass.
|
|
||||||
crate::io::drain_wake_pipe(io.wake_fd());
|
|
||||||
io.drain_completions()
|
|
||||||
})
|
|
||||||
.unwrap_or_default(),
|
|
||||||
Err(e) => panic!("smarm: io lock poisoned (core corrupt): {e}"),
|
|
||||||
};
|
};
|
||||||
for entry in due {
|
for entry in due {
|
||||||
match entry.reason {
|
match entry.reason {
|
||||||
@@ -1549,9 +1815,7 @@ fn schedule_loop(inner: &Arc<RuntimeInner>, slot_idx: usize) {
|
|||||||
// actor is between `timers.insert_sleep` and
|
// actor is between `timers.insert_sleep` and
|
||||||
// `park_current`; RunningNotified makes the upcoming park
|
// `park_current`; RunningNotified makes the upcoming park
|
||||||
// re-queue), or gone (no-op).
|
// re-queue), or gone (no-op).
|
||||||
crate::timer::Reason::Sleep { epoch } => {
|
crate::timer::Reason::Sleep { epoch } => inner.unpark_at(entry.pid, epoch),
|
||||||
inner.unpark_at(entry.pid, epoch)
|
|
||||||
}
|
|
||||||
crate::timer::Reason::WaitTimeout { target, epoch } => {
|
crate::timer::Reason::WaitTimeout { target, epoch } => {
|
||||||
// The callback may call unpark_at itself.
|
// The callback may call unpark_at itself.
|
||||||
target.on_timeout(entry.pid, epoch);
|
target.on_timeout(entry.pid, epoch);
|
||||||
@@ -1566,57 +1830,28 @@ fn schedule_loop(inner: &Arc<RuntimeInner>, slot_idx: usize) {
|
|||||||
crate::timer::Reason::Send { fire } => fire(),
|
crate::timer::Reason::Send { fire } => fire(),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
}
|
||||||
|
|
||||||
for completion in completions {
|
// ---------------------------------------------------------------------------
|
||||||
match completion {
|
// schedule_loop — runs on each scheduler OS thread
|
||||||
crate::io::Completion::Blocking { pid, epoch, result } => {
|
// ---------------------------------------------------------------------------
|
||||||
match inner.io.lock() {
|
|
||||||
Ok(mut io) => {
|
fn schedule_loop(inner: &Arc<RuntimeInner>, slot_idx: usize) {
|
||||||
if let Some(io) = io.as_mut() {
|
// RFC 019 §7: a guard hit leaves no stack to handle the signal on.
|
||||||
io.outstanding = io.outstanding.saturating_sub(1);
|
crate::signal::register_altstack();
|
||||||
|
crate::preempt::configure_preempt(inner.alloc_interval, inner.timeslice_cycles);
|
||||||
|
let stats = &inner.stats[slot_idx];
|
||||||
|
|
||||||
|
loop {
|
||||||
|
// ----------------------------------------------------------------
|
||||||
|
// 1. Busy-path timer due-check (RFC 018 design point (a)): under
|
||||||
|
// saturation nobody parks, so no timekeeper exists — due timers
|
||||||
|
// must still fire. One Relaxed load + branch when no timer is
|
||||||
|
// armed; the clock is read only when one is.
|
||||||
|
// ----------------------------------------------------------------
|
||||||
|
if inner.coord.deadline_due() {
|
||||||
|
fire_due_timers(inner, true);
|
||||||
}
|
}
|
||||||
}
|
|
||||||
Err(e) => {
|
|
||||||
panic!("smarm: io lock poisoned (core corrupt): {e}")
|
|
||||||
}
|
|
||||||
}
|
|
||||||
// Stash the result under the cold lock, then unpark.
|
|
||||||
// The protocol also covers the submit→park window
|
|
||||||
// (RunningNotified), which the old code missed for
|
|
||||||
// Blocking completions — a latent lost wakeup.
|
|
||||||
if let Some(slot) = inner.slot_at(pid) {
|
|
||||||
{
|
|
||||||
let mut cold = slot.cold.lock();
|
|
||||||
if slot.generation() == pid.generation() {
|
|
||||||
cold.pending_io_result = Some(result);
|
|
||||||
} else {
|
|
||||||
// Actor died (stopped) with the op in
|
|
||||||
// flight; discard the result.
|
|
||||||
}
|
|
||||||
}
|
|
||||||
inner.unpark_at(pid, epoch);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
crate::io::Completion::FdReady { fd, events: _ } => {
|
|
||||||
// Resolve the parked pid under the io lock, then wake
|
|
||||||
// through the protocol. Lock order: io before all.
|
|
||||||
let parked = match inner.io.lock() {
|
|
||||||
Ok(mut io) => io.as_mut().and_then(|io| {
|
|
||||||
let entry = io.waiters.remove(&fd);
|
|
||||||
io.epoll_deregister(fd);
|
|
||||||
entry
|
|
||||||
}),
|
|
||||||
Err(e) => {
|
|
||||||
panic!("smarm: io lock poisoned (core corrupt): {e}")
|
|
||||||
}
|
|
||||||
};
|
|
||||||
if let Some((pid, epoch)) = parked {
|
|
||||||
inner.unpark_at(pid, epoch);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
} // drain_guard drops here
|
|
||||||
|
|
||||||
// ----------------------------------------------------------------
|
// ----------------------------------------------------------------
|
||||||
// 2. Pop a runnable pid. Pop order (RFC 005): wake slot first, then
|
// 2. Pop a runnable pid. Pop order (RFC 005): wake slot first, then
|
||||||
@@ -1625,7 +1860,7 @@ fn schedule_loop(inner: &Arc<RuntimeInner>, slot_idx: usize) {
|
|||||||
// ----------------------------------------------------------------
|
// ----------------------------------------------------------------
|
||||||
enum Pop {
|
enum Pop {
|
||||||
Got(Pid),
|
Got(Pid),
|
||||||
Idle { io_outstanding: u32, wake_fd: Option<std::os::fd::RawFd> },
|
Idle,
|
||||||
AllDone,
|
AllDone,
|
||||||
/// Root has exited and nothing is runnable: stop the parked-forever
|
/// Root has exited and nothing is runnable: stop the parked-forever
|
||||||
/// remainder, then re-pop. Fires at most once per run.
|
/// remainder, then re-pop. Fires at most once per run.
|
||||||
@@ -1650,19 +1885,12 @@ fn schedule_loop(inner: &Arc<RuntimeInner>, slot_idx: usize) {
|
|||||||
crate::te!(crate::trace::Event::SlotPop(pid));
|
crate::te!(crate::trace::Event::SlotPop(pid));
|
||||||
pid
|
pid
|
||||||
} else {
|
} else {
|
||||||
// Read IO liveness BEFORE the queue lock (phase-1 ordering: a
|
// Read IO liveness BEFORE the queue pop — two atomic loads now
|
||||||
// completion resurrects an actor only via the drain path, whose
|
// (RFC 018), not a per-pop `io.lock`: a completion resurrects
|
||||||
// enqueue would be visible under the queue lock we take next).
|
// an actor via the producer's own unpark→enqueue, whose entry
|
||||||
let (io_out, io_fd) = {
|
// would be visible to the pop below.
|
||||||
let io = match inner.io.lock() {
|
let io_out = inner.io_outstanding.load(Ordering::Acquire)
|
||||||
Ok(io) => io,
|
+ inner.io_fd_waiters.load(Ordering::Acquire);
|
||||||
Err(e) => panic!("smarm: io lock poisoned (core corrupt): {e}"),
|
|
||||||
};
|
|
||||||
match io.as_ref() {
|
|
||||||
Some(io) => (io.outstanding + io.waiters.len() as u32, Some(io.wake_fd())),
|
|
||||||
None => (0, None),
|
|
||||||
}
|
|
||||||
};
|
|
||||||
|
|
||||||
stats.run_queue_len.store(inner.run_queue.len(), Ordering::Relaxed);
|
stats.run_queue_len.store(inner.run_queue.len(), Ordering::Relaxed);
|
||||||
let pop = match inner.run_queue.pop() {
|
let pop = match inner.run_queue.pop() {
|
||||||
@@ -1694,7 +1922,7 @@ fn schedule_loop(inner: &Arc<RuntimeInner>, slot_idx: usize) {
|
|||||||
// the idle wait below on the next pass.
|
// the idle wait below on the next pass.
|
||||||
Pop::RootDrain
|
Pop::RootDrain
|
||||||
} else {
|
} else {
|
||||||
Pop::Idle { io_outstanding: io_out, wake_fd: io_fd }
|
Pop::Idle
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
@@ -1709,22 +1937,15 @@ fn schedule_loop(inner: &Arc<RuntimeInner>, slot_idx: usize) {
|
|||||||
Ok(mut timers) => timers.clear(),
|
Ok(mut timers) => timers.clear(),
|
||||||
Err(e) => panic!("smarm: timers lock poisoned (core corrupt): {e}"),
|
Err(e) => panic!("smarm: timers lock poisoned (core corrupt): {e}"),
|
||||||
}
|
}
|
||||||
// Terminal wake: a sibling scheduler may be blocked in its
|
// Terminal wake (replaces the wake-pipe byte): a sibling
|
||||||
// idle wait on a snapshot that is now terminally stale — an
|
// may be parked on a snapshot that is now terminally
|
||||||
// orphaned long deadline (it would sleep it out in full) or
|
// stale — an orphaned long deadline, or a stale
|
||||||
// a stale `io_outstanding > 0` from a stop-cancelled waiter
|
// `io_fd_waiters > 0` from a stop-cancelled waiter
|
||||||
// (it would block in poll(-1) forever; cancellation produces
|
// (cancellation produces no completion, so nothing else
|
||||||
// no completion, so nothing else writes the wake pipe).
|
// will ever wake it). `wake_all` permits every parker;
|
||||||
// One byte wakes every poller; each re-runs the verdict,
|
// each sibling re-runs the verdict, reaches AllDone
|
||||||
// reaches AllDone itself, and re-wakes — idempotent.
|
// itself, and re-wakes — idempotent.
|
||||||
match inner.io.lock() {
|
inner.coord.wake_all();
|
||||||
Ok(io) => {
|
|
||||||
if let Some(io) = io.as_ref() {
|
|
||||||
io.wake();
|
|
||||||
}
|
|
||||||
}
|
|
||||||
Err(e) => panic!("smarm: io lock poisoned (core corrupt): {e}"),
|
|
||||||
}
|
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
Pop::RootDrain => {
|
Pop::RootDrain => {
|
||||||
@@ -1734,40 +1955,51 @@ fn schedule_loop(inner: &Arc<RuntimeInner>, slot_idx: usize) {
|
|||||||
stop_live_actors(inner);
|
stop_live_actors(inner);
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
Pop::Idle { io_outstanding, wake_fd } => {
|
Pop::Idle => {
|
||||||
// Something is still in flight. Sleep on the appropriate
|
// Something is still in flight. Park on our own futex
|
||||||
// source to avoid hammering the queue mutex; retry on wake.
|
// until a producer wakes us (enqueue tail), a deadline
|
||||||
let next_deadline = match inner.timers.lock() {
|
// passes, or the re-check finds the world changed.
|
||||||
Ok(timers) => timers.peek_deadline(),
|
//
|
||||||
Err(e) => panic!("smarm: timers lock poisoned (core corrupt): {e}"),
|
// Timekeeper (RFC 018): at most one parked scheduler
|
||||||
|
// holds the timer deadline — the first idler to arm it
|
||||||
|
// parks with a timeout, the rest park indefinitely, so a
|
||||||
|
// timer expiry wakes one scheduler, not a herd. Peek and
|
||||||
|
// arm under the timers mutex (the serialization that
|
||||||
|
// makes the insert-side re-arm race-free).
|
||||||
|
let tk_deadline = {
|
||||||
|
let timers = match inner.timers.lock() {
|
||||||
|
Ok(t) => t,
|
||||||
|
Err(e) => {
|
||||||
|
panic!("smarm: timers lock poisoned (core corrupt): {e}")
|
||||||
|
}
|
||||||
};
|
};
|
||||||
match (next_deadline, wake_fd) {
|
timers
|
||||||
(Some(deadline), fd_opt) => {
|
.peek_deadline()
|
||||||
let now = std::time::Instant::now();
|
.filter(|d| inner.coord.try_arm_timer(slot_idx, *d))
|
||||||
if deadline > now {
|
};
|
||||||
let timeout = deadline - now;
|
// The mandatory post-publish re-check: a producer that
|
||||||
match fd_opt {
|
// enqueued (or a verdict input that flipped) before it
|
||||||
Some(fd) => {
|
// could see our idle bit has left us the evidence.
|
||||||
// Wake only; the byte (if any) is
|
let _ = inner.coord.park(slot_idx, tk_deadline, || {
|
||||||
// consumed by the next drain-lock
|
!inner.run_queue.is_empty()
|
||||||
// winner in phase 1. Level-triggered
|
|| (inner.live_actors.load(Ordering::Acquire) == 0
|
||||||
// poll means an unconsumed byte makes
|
&& inner.io_outstanding.load(Ordering::Acquire) == 0
|
||||||
// this return immediately, so a loser
|
&& inner.io_fd_waiters.load(Ordering::Acquire) == 0)
|
||||||
// spins briefly until the winner
|
|| (inner.root_exited.load(Ordering::Acquire)
|
||||||
// releases — never sleeps through it.
|
&& !inner.root_swept.load(Ordering::Acquire))
|
||||||
crate::io::poll_wake(fd, Some(timeout));
|
|| inner.coord.deadline_due()
|
||||||
}
|
});
|
||||||
None => thread::sleep(timeout),
|
if tk_deadline.is_some() {
|
||||||
}
|
// Hand the role back BEFORE firing: pop_due can run
|
||||||
}
|
// `Send` thunks that insert new timers, and the
|
||||||
}
|
// insert-side re-arm check must see either no
|
||||||
(None, Some(fd)) if io_outstanding > 0 => {
|
// timekeeper (skip) or a real parked one — never us,
|
||||||
// See above: no byte consumption outside phase 1.
|
// awake and about to re-peek anyway.
|
||||||
crate::io::poll_wake(fd, None);
|
inner.coord.disarm_timer(slot_idx);
|
||||||
}
|
// Woken for the deadline, for work, or to re-peek
|
||||||
_ => {
|
// after an earlier insert — fire whatever is due;
|
||||||
thread::sleep(std::time::Duration::from_micros(100));
|
// the next idle pass re-arms with the new minimum.
|
||||||
}
|
fire_due_timers(inner, false);
|
||||||
}
|
}
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
@@ -1780,6 +2012,21 @@ fn schedule_loop(inner: &Arc<RuntimeInner>, slot_idx: usize) {
|
|||||||
// by the at-most-once-enqueued invariant nothing else can have
|
// by the at-most-once-enqueued invariant nothing else can have
|
||||||
// changed the state of a queued actor.
|
// changed the state of a queued actor.
|
||||||
// ----------------------------------------------------------------
|
// ----------------------------------------------------------------
|
||||||
|
|
||||||
|
// RFC 018 chain rule: we just took one runnable; if more remain and
|
||||||
|
// a sibling is parked, wake exactly one so the surplus runs in
|
||||||
|
// PARALLEL rather than serially behind us (without this the surplus
|
||||||
|
// is not stranded — we re-pop it after resuming — but it waits out
|
||||||
|
// our whole timeslice while an idle core sits available). Cheap: the
|
||||||
|
// queue-length check is queue-local, and `wake_one_if_idle` is a
|
||||||
|
// fence + one Relaxed mask load when nobody is parked. A Relaxed
|
||||||
|
// miss here is safe — the enqueue that created the surplus already
|
||||||
|
// issued its own wake (RFC 018 no-lost-wake); this only sharpens
|
||||||
|
// parallelism latency.
|
||||||
|
if !inner.run_queue.is_empty() {
|
||||||
|
inner.coord.wake_one_if_idle();
|
||||||
|
}
|
||||||
|
|
||||||
let slot = match inner.slot_at(pid) {
|
let slot = match inner.slot_at(pid) {
|
||||||
Some(s) => s,
|
Some(s) => s,
|
||||||
None => continue, // can't happen for real pids; defensive
|
None => continue, // can't happen for real pids; defensive
|
||||||
@@ -1839,7 +2086,15 @@ fn schedule_loop(inner: &Arc<RuntimeInner>, slot_idx: usize) {
|
|||||||
crate::preempt::clear_current_slot();
|
crate::preempt::clear_current_slot();
|
||||||
|
|
||||||
let intent = YIELD_INTENT.with(|c| c.get());
|
let intent = YIELD_INTENT.with(|c| c.get());
|
||||||
slot.sp.store(get_actor_sp(), Ordering::Relaxed);
|
let saved_sp = get_actor_sp();
|
||||||
|
slot.sp.store(saved_sp, Ordering::Relaxed);
|
||||||
|
// RFC 019 §2: sampled high-water — one branch + at most one store
|
||||||
|
// into the line the store above just dirtied. Relaxed and advisory;
|
||||||
|
// it piggybacks the existing Relaxed-store-before-Release pattern
|
||||||
|
// (mod docs, "Memory ordering") and adds no edges.
|
||||||
|
if saved_sp < slot.hwm.load(Ordering::Relaxed) {
|
||||||
|
slot.hwm.store(saved_sp, Ordering::Relaxed);
|
||||||
|
}
|
||||||
|
|
||||||
if is_actor_done() {
|
if is_actor_done() {
|
||||||
crate::te!(crate::trace::Event::Done(pid));
|
crate::te!(crate::trace::Event::Done(pid));
|
||||||
@@ -1861,6 +2116,23 @@ fn schedule_loop(inner: &Arc<RuntimeInner>, slot_idx: usize) {
|
|||||||
inner.enqueue(pid);
|
inner.enqueue(pid);
|
||||||
}
|
}
|
||||||
YieldIntent::Park => {
|
YieldIntent::Park => {
|
||||||
|
// RFC 019 §3 shrink window (correctness obligation 1):
|
||||||
|
// this site sits after the `sp` store above and before
|
||||||
|
// the `park_return` Release transition below publishes
|
||||||
|
// Parked — the scheduler is on its own stack and the
|
||||||
|
// actor is saved but not yet stealable, so the madvise
|
||||||
|
// races nothing (belt). MADV_FREE's cancel-on-write is
|
||||||
|
// the suspenders: even a racing writer could lose
|
||||||
|
// nothing written after the mark, and everything below
|
||||||
|
// live `sp` is dead by definition. Runs on BOTH arms of
|
||||||
|
// the park_return race — a consumed unpark flag means a
|
||||||
|
// wasted-but-harmless madvise on a rare window.
|
||||||
|
//
|
||||||
|
// This is the ONLY shrink site: the preempt/yield path
|
||||||
|
// deliberately never checks (§4's bounded leak under
|
||||||
|
// saturation — syscalls must not fire when scheduler
|
||||||
|
// cycles are scarcest).
|
||||||
|
maybe_shrink_stack(slot);
|
||||||
if slot.word.park_return(gen) {
|
if slot.word.park_return(gen) {
|
||||||
// RFC 007 audit: an in-site park drops its sample
|
// RFC 007 audit: an in-site park drops its sample
|
||||||
// tail (nothing flushes it; on_resume re-arms).
|
// tail (nothing flushes it; on_resume re-arms).
|
||||||
|
|||||||
+91
-18
@@ -72,6 +72,7 @@ use crate::runtime::{
|
|||||||
self, RuntimeInner, YieldIntent, RUNTIME,
|
self, RuntimeInner, YieldIntent, RUNTIME,
|
||||||
};
|
};
|
||||||
use crate::supervisor::Signal;
|
use crate::supervisor::Signal;
|
||||||
|
use std::sync::atomic::Ordering;
|
||||||
use std::sync::Arc;
|
use std::sync::Arc;
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
@@ -258,6 +259,28 @@ impl Drop for JoinHandle {
|
|||||||
// spawn / spawn_under / self_pid
|
// spawn / spawn_under / self_pid
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
/// Per-spawn stack shape overrides (RFC 019). `None` fields resolve to the
|
||||||
|
/// runtime's [`Config`](crate::runtime::Config) defaults at spawn time, so
|
||||||
|
/// struct-update syntax works anywhere without a runtime handle:
|
||||||
|
///
|
||||||
|
/// ```
|
||||||
|
/// use smarm::SpawnOpts;
|
||||||
|
/// let opts = SpawnOpts { stack_reserve: Some(8 * 1024 * 1024), ..SpawnOpts::default() };
|
||||||
|
/// ```
|
||||||
|
///
|
||||||
|
/// Both sizes are page-rounded. The reserve is *virtual* (demand-paged):
|
||||||
|
/// an 8 MiB reserve costs address space, not memory — RSS follows touched
|
||||||
|
/// pages. The guard is PROT_NONE below the stack; raise it for FFI code
|
||||||
|
/// with unusually large C frames. Custom-shaped stacks bypass the recycle
|
||||||
|
/// pool: they are mmapped fresh at spawn and munmapped at death.
|
||||||
|
#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
|
||||||
|
pub struct SpawnOpts {
|
||||||
|
/// Usable stack reservation. `None` ⇒ [`Config::stack_reserve`](crate::runtime::Config::stack_reserve).
|
||||||
|
pub stack_reserve: Option<usize>,
|
||||||
|
/// PROT_NONE guard below the stack. `None` ⇒ [`Config::stack_guard`](crate::runtime::Config::stack_guard).
|
||||||
|
pub guard_size: Option<usize>,
|
||||||
|
}
|
||||||
|
|
||||||
/// Start a new actor running `f`, and return a [`JoinHandle`] for it.
|
/// Start a new actor running `f`, and return a [`JoinHandle`] for it.
|
||||||
///
|
///
|
||||||
/// The new actor runs concurrently with its caller and with every other
|
/// The new actor runs concurrently with its caller and with every other
|
||||||
@@ -280,22 +303,34 @@ pub fn spawn(f: impl FnOnce() + Send + 'static) -> JoinHandle {
|
|||||||
spawn_under(parent, f)
|
spawn_under(parent, f)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// [`spawn`] with per-actor stack shape overrides (RFC 019).
|
||||||
|
pub fn spawn_with(opts: SpawnOpts, f: impl FnOnce() + Send + 'static) -> JoinHandle {
|
||||||
|
let parent = current_pid().unwrap_or_else(|| {
|
||||||
|
with_runtime(|_| crate::runtime::ROOT_PID)
|
||||||
|
});
|
||||||
|
spawn_under_with(parent, opts, f)
|
||||||
|
}
|
||||||
|
|
||||||
/// Like [`spawn`], but explicitly attaches the new actor to `supervisor`
|
/// Like [`spawn`], but explicitly attaches the new actor to `supervisor`
|
||||||
/// instead of the calling actor. Ordinary code should reach for [`spawn`];
|
/// instead of the calling actor. Ordinary code should reach for [`spawn`];
|
||||||
/// this exists for supervision trees (see [`supervisor`](crate::supervisor))
|
/// this exists for supervision trees (see [`supervisor`](crate::supervisor))
|
||||||
/// and other cases that need to place a child under a specific ancestor
|
/// and other cases that need to place a child under a specific ancestor
|
||||||
/// rather than its true caller.
|
/// rather than its true caller.
|
||||||
pub fn spawn_under<A>(supervisor: Pid<A>, f: impl FnOnce() + Send + 'static) -> JoinHandle {
|
pub fn spawn_under<A>(supervisor: Pid<A>, f: impl FnOnce() + Send + 'static) -> JoinHandle {
|
||||||
let supervisor = supervisor.erase();
|
spawn_under_with(supervisor, SpawnOpts::default(), f)
|
||||||
// Stack + closure boxing happen before ANY runtime lock is taken: no
|
|
||||||
// syscall and no allocation ever stalls another scheduler thread.
|
|
||||||
let stack = with_runtime(|inner| inner.stack_pool.lock().pop())
|
|
||||||
.unwrap_or_else(|| {
|
|
||||||
match crate::stack::Stack::new(crate::runtime::ACTOR_STACK_SIZE) {
|
|
||||||
Ok(stack) => stack,
|
|
||||||
Err(e) => panic!("stack allocation failed: {e}"),
|
|
||||||
}
|
}
|
||||||
});
|
|
||||||
|
/// [`spawn_under`] with per-actor stack shape overrides (RFC 019).
|
||||||
|
pub fn spawn_under_with<A>(
|
||||||
|
supervisor: Pid<A>,
|
||||||
|
opts: SpawnOpts,
|
||||||
|
f: impl FnOnce() + Send + 'static,
|
||||||
|
) -> JoinHandle {
|
||||||
|
let supervisor = supervisor.erase();
|
||||||
|
// Stack + closure boxing happen before the slot locks are taken; the
|
||||||
|
// pool lock inside acquire_stack is dropped before any mmap, so no
|
||||||
|
// syscall ever stalls another scheduler thread.
|
||||||
|
let stack = with_runtime(|inner| crate::runtime::acquire_stack(inner, opts));
|
||||||
let sp = init_actor_stack(stack.top(), crate::actor::trampoline);
|
let sp = init_actor_stack(stack.top(), crate::actor::trampoline);
|
||||||
let closure: crate::runtime::Closure = Box::new(f);
|
let closure: crate::runtime::Closure = Box::new(f);
|
||||||
|
|
||||||
@@ -335,6 +370,18 @@ pub fn spawn_addr<A: crate::pid::Addressable>(
|
|||||||
crate::pid::assert_type::<A>(pid)
|
crate::pid::assert_type::<A>(pid)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// [`spawn_addr`] with per-actor stack shape overrides (RFC 019).
|
||||||
|
pub fn spawn_addr_with<A: crate::pid::Addressable>(
|
||||||
|
opts: SpawnOpts,
|
||||||
|
body: impl FnOnce(crate::channel::Receiver<A::Msg>) + Send + 'static,
|
||||||
|
) -> Pid<A> {
|
||||||
|
let (tx, rx) = crate::channel::channel::<A::Msg>();
|
||||||
|
let handle = spawn_with(opts, move || body(rx));
|
||||||
|
let pid = handle.pid();
|
||||||
|
crate::registry::install_for::<A::Msg>(pid, tx);
|
||||||
|
crate::pid::assert_type::<A>(pid)
|
||||||
|
}
|
||||||
|
|
||||||
use crate::context::init_actor_stack;
|
use crate::context::init_actor_stack;
|
||||||
|
|
||||||
/// The identity of the actor currently running. Use it to hand your own
|
/// The identity of the actor currently running. Use it to hand your own
|
||||||
@@ -753,7 +800,14 @@ where
|
|||||||
Err(e) => panic!("smarm: io lock poisoned (core corrupt): {e}"),
|
Err(e) => panic!("smarm: io lock poisoned (core corrupt): {e}"),
|
||||||
};
|
};
|
||||||
match io.as_mut() {
|
match io.as_mut() {
|
||||||
Some(io) => io.submit(me, epoch, work),
|
Some(io) => {
|
||||||
|
// RFC 018: count the op in flight BEFORE submit — the
|
||||||
|
// pool decrements on completion, and an increment that
|
||||||
|
// trailed the completion would underflow. Under the io
|
||||||
|
// lock, so ordered against the same-lock submit.
|
||||||
|
inner.io_outstanding.fetch_add(1, Ordering::AcqRel);
|
||||||
|
io.submit(me, epoch, work);
|
||||||
|
}
|
||||||
None => panic!("io thread not started"),
|
None => panic!("io thread not started"),
|
||||||
}
|
}
|
||||||
});
|
});
|
||||||
@@ -813,7 +867,17 @@ fn wait_fd(fd: std::os::fd::RawFd, readable: bool, writable: bool) -> std::io::R
|
|||||||
Err(e) => panic!("smarm: io lock poisoned (core corrupt): {e}"),
|
Err(e) => panic!("smarm: io lock poisoned (core corrupt): {e}"),
|
||||||
};
|
};
|
||||||
match io.as_mut() {
|
match io.as_mut() {
|
||||||
Some(io) => io.epoll_register(fd, me, epoch, readable, writable),
|
Some(io) => {
|
||||||
|
// RFC 018: count the waiter BEFORE the ADD (mirror of
|
||||||
|
// submit); roll back if the registration fails so a
|
||||||
|
// rejected wait leaves the verdict counters clean.
|
||||||
|
inner.io_fd_waiters.fetch_add(1, Ordering::AcqRel);
|
||||||
|
let r = io.epoll_register(fd, me, epoch, readable, writable);
|
||||||
|
if r.is_err() {
|
||||||
|
inner.io_fd_waiters.fetch_sub(1, Ordering::AcqRel);
|
||||||
|
}
|
||||||
|
r
|
||||||
|
}
|
||||||
None => panic!("io thread not started"),
|
None => panic!("io thread not started"),
|
||||||
}
|
}
|
||||||
})?;
|
})?;
|
||||||
@@ -838,9 +902,12 @@ fn wait_fd(fd: std::os::fd::RawFd, readable: bool, writable: bool) -> std::io::R
|
|||||||
Err(e) => panic!("smarm: io lock poisoned (core corrupt): {e}"),
|
Err(e) => panic!("smarm: io lock poisoned (core corrupt): {e}"),
|
||||||
};
|
};
|
||||||
if let Some(io) = io.as_mut() {
|
if let Some(io) = io.as_mut() {
|
||||||
if io.waiters.get(&self.fd) == Some(&(self.me, self.epoch)) {
|
// `cancel_waiter` removes + DELs iff still ours, all
|
||||||
io.waiters.remove(&self.fd);
|
// under the waiters lock (the ADD/DEL serialization);
|
||||||
io.epoll_deregister(self.fd);
|
// decrement only when we actually removed it — a
|
||||||
|
// FdReady that consumed it already did the decrement.
|
||||||
|
if io.cancel_waiter(self.fd, self.me, self.epoch) {
|
||||||
|
inner.io_fd_waiters.fetch_sub(1, Ordering::AcqRel);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
});
|
});
|
||||||
@@ -908,7 +975,14 @@ impl crate::channel::Selectable for FdArm {
|
|||||||
};
|
};
|
||||||
match io.as_mut() {
|
match io.as_mut() {
|
||||||
Some(io) => {
|
Some(io) => {
|
||||||
io.epoll_register(self.fd, pid, epoch, self.readable, self.writable)
|
inner.io_fd_waiters.fetch_add(1, Ordering::AcqRel);
|
||||||
|
let r = io.epoll_register(
|
||||||
|
self.fd, pid, epoch, self.readable, self.writable,
|
||||||
|
);
|
||||||
|
if r.is_err() {
|
||||||
|
inner.io_fd_waiters.fetch_sub(1, Ordering::AcqRel);
|
||||||
|
}
|
||||||
|
r
|
||||||
}
|
}
|
||||||
None => panic!("io thread not started"),
|
None => panic!("io thread not started"),
|
||||||
}
|
}
|
||||||
@@ -936,9 +1010,8 @@ impl crate::channel::Selectable for FdArm {
|
|||||||
Err(e) => panic!("smarm: io lock poisoned (core corrupt): {e}"),
|
Err(e) => panic!("smarm: io lock poisoned (core corrupt): {e}"),
|
||||||
};
|
};
|
||||||
if let Some(io) = io.as_mut() {
|
if let Some(io) = io.as_mut() {
|
||||||
if io.waiters.get(&self.fd) == Some(&(pid, epoch)) {
|
if io.cancel_waiter(self.fd, pid, epoch) {
|
||||||
io.waiters.remove(&self.fd);
|
inner.io_fd_waiters.fetch_sub(1, Ordering::AcqRel);
|
||||||
io.epoll_deregister(self.fd);
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
});
|
});
|
||||||
|
|||||||
+321
@@ -0,0 +1,321 @@
|
|||||||
|
//! RFC 019 §7 — overflow diagnostics.
|
||||||
|
//!
|
||||||
|
//! One process-global SIGSEGV handler, installed once at [`crate::runtime::init`]
|
||||||
|
//! (before any scheduler thread exists, so the PRIOR save is unracing), plus a
|
||||||
|
//! per-scheduler-thread `sigaltstack` registered at `schedule_loop` entry — a
|
||||||
|
//! guard hit means the faulting stack has no room to run anything, so the
|
||||||
|
//! altstack is not optional.
|
||||||
|
//!
|
||||||
|
//! The handler classifies `si_addr` against the *current* actor only, reached
|
||||||
|
//! through `preempt::CURRENT_SLOT` — a const-initialized `Cell<*const Slot>`
|
||||||
|
//! whose access is a plain TLS load (no lazy init, no allocation, no dtor
|
||||||
|
//! registration), and which every scheduler thread has materialized before an
|
||||||
|
//! actor can run on it. The slot's diag atomics (`diag_stack_top` & co) are
|
||||||
|
//! written in `install_actor` before the Release publish and are only consulted
|
||||||
|
//! here while the actor is on-CPU, so they cannot be stale.
|
||||||
|
//!
|
||||||
|
//! Two classification tiers:
|
||||||
|
//! - **In-guard**: definitive. Rust frames probe pages in order
|
||||||
|
//! (`__rust_probestack`), so Rust overflow always lands here; so does any C
|
||||||
|
//! built with `-fstack-clash-protection` (distro-packaged libraries), and —
|
||||||
|
//! with the 1 MiB default guard — nearly every unprobed frame too.
|
||||||
|
//! - **Overshoot**: within [`OVERSHOOT_SLOP`] *below* the guard. An unprobed
|
||||||
|
//! frame (cargo-built C via `cc` almost never enables clash protection)
|
||||||
|
//! large enough to step over the guard in one `sub rsp`. Attribution is
|
||||||
|
//! "probable": the address is in unmapped VA that nothing else owns, an
|
||||||
|
//! actor was on-CPU, and the distance fits a frame — the diagnostic says so.
|
||||||
|
//!
|
||||||
|
//! Classified faults print one line (async-signal-safe: stack buffer +
|
||||||
|
//! `write(2)`, no fmt, no alloc, no locks) and re-raise with default
|
||||||
|
//! disposition — no unwind, no resume, no fail-soft (jarred; UB-adjacent from
|
||||||
|
//! a handler). Unclassified faults reinstate the PRIOR handler and refault, so
|
||||||
|
//! std's own "thread ... has overflowed its stack" diagnostics for OS-thread
|
||||||
|
//! stacks survive our presence. Reinstating deregisters us for good, which is
|
||||||
|
//! fine: the process is dying either way.
|
||||||
|
|
||||||
|
use std::cell::Cell;
|
||||||
|
use std::mem::MaybeUninit;
|
||||||
|
use std::sync::atomic::Ordering;
|
||||||
|
use std::sync::Once;
|
||||||
|
|
||||||
|
/// Tier-2 window below the guard. Matches the guard default (and the kernel's
|
||||||
|
/// `stack_guard_gap`): a frame that out-jumps both the guard and this window
|
||||||
|
/// in one displacement is past what a diagnostic can honestly attribute.
|
||||||
|
pub(crate) const OVERSHOOT_SLOP: usize = 1024 * 1024;
|
||||||
|
|
||||||
|
/// Per-scheduler-thread signal stack. MINSIGSTKSZ is ~11 KiB on AVX-512
|
||||||
|
/// hardware; 64 KiB leaves the formatter room without mattering to anyone.
|
||||||
|
/// One per OS thread, never freed: scheduler threads live for the process in
|
||||||
|
/// practice, and repeated `run()`s on reused threads re-use the registration
|
||||||
|
/// (the TLS flag), so the leak is bounded by the OS thread count.
|
||||||
|
const ALTSTACK_SIZE: usize = 64 * 1024;
|
||||||
|
|
||||||
|
static INSTALL: Once = Once::new();
|
||||||
|
/// The handler that was installed before ours (std's, typically). Written
|
||||||
|
/// exactly once inside INSTALL — which completes in `runtime::init` before
|
||||||
|
/// any scheduler thread (and thus any classifiable fault) can exist — and
|
||||||
|
/// only read from the handler afterwards.
|
||||||
|
static mut PRIOR: MaybeUninit<libc::sigaction> = MaybeUninit::uninit();
|
||||||
|
|
||||||
|
thread_local! {
|
||||||
|
/// Whether this OS thread has registered its altstack.
|
||||||
|
static ALTSTACK_SET: Cell<bool> = const { Cell::new(false) };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Where a fault landed relative to the current actor's stack.
|
||||||
|
#[derive(Debug, PartialEq, Eq)]
|
||||||
|
pub(crate) enum FaultClass {
|
||||||
|
/// Inside `[top − reserve − guard, top − reserve)`: the guard region.
|
||||||
|
Guard,
|
||||||
|
/// Within `OVERSHOOT_SLOP` below the guard: stepped over it. Payload is
|
||||||
|
/// the distance below `guard_lo`.
|
||||||
|
Overshoot(usize),
|
||||||
|
/// Not ours to explain.
|
||||||
|
Foreign,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Pure classifier — all edges unit-tested below. `top` is the stack's usable
|
||||||
|
/// top, `reserve`/`guard` its shape; both page-rounded by `Stack::new`.
|
||||||
|
pub(crate) fn classify(addr: usize, top: usize, reserve: usize, guard: usize) -> FaultClass {
|
||||||
|
let guard_hi = top.wrapping_sub(reserve);
|
||||||
|
let guard_lo = guard_hi.wrapping_sub(guard);
|
||||||
|
if addr >= guard_lo && addr < guard_hi {
|
||||||
|
FaultClass::Guard
|
||||||
|
} else if addr < guard_lo && addr >= guard_lo.saturating_sub(OVERSHOOT_SLOP) {
|
||||||
|
FaultClass::Overshoot(guard_lo - addr)
|
||||||
|
} else {
|
||||||
|
FaultClass::Foreign
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Install the process-global handler. Idempotent; called from
|
||||||
|
/// `runtime::init`.
|
||||||
|
pub(crate) fn install_once() {
|
||||||
|
INSTALL.call_once(|| unsafe {
|
||||||
|
let mut sa: libc::sigaction = std::mem::zeroed();
|
||||||
|
sa.sa_sigaction = handler as *const () as usize;
|
||||||
|
sa.sa_flags = libc::SA_SIGINFO | libc::SA_ONSTACK;
|
||||||
|
libc::sigemptyset(&mut sa.sa_mask);
|
||||||
|
let prior = &mut *std::ptr::addr_of_mut!(PRIOR);
|
||||||
|
libc::sigaction(libc::SIGSEGV, &sa, prior.as_mut_ptr());
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Register this OS thread's altstack (idempotent per thread). Called at
|
||||||
|
/// `schedule_loop` entry, so every thread that can run an actor has one.
|
||||||
|
pub(crate) fn register_altstack() {
|
||||||
|
ALTSTACK_SET.with(|set| {
|
||||||
|
if set.get() {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
unsafe {
|
||||||
|
let sp = libc::mmap(
|
||||||
|
std::ptr::null_mut(),
|
||||||
|
ALTSTACK_SIZE,
|
||||||
|
libc::PROT_READ | libc::PROT_WRITE,
|
||||||
|
libc::MAP_PRIVATE | libc::MAP_ANONYMOUS,
|
||||||
|
-1,
|
||||||
|
0,
|
||||||
|
);
|
||||||
|
if sp == libc::MAP_FAILED {
|
||||||
|
// Degrade: no altstack means a guard hit dies without the
|
||||||
|
// message (handler can't run) — the pre-RFC behavior, never
|
||||||
|
// incorrectness.
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
let ss = libc::stack_t {
|
||||||
|
ss_sp: sp,
|
||||||
|
ss_flags: 0,
|
||||||
|
ss_size: ALTSTACK_SIZE,
|
||||||
|
};
|
||||||
|
libc::sigaltstack(&ss, std::ptr::null_mut());
|
||||||
|
}
|
||||||
|
set.set(true);
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// The handler
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
unsafe extern "C" fn handler(
|
||||||
|
_sig: libc::c_int,
|
||||||
|
info: *mut libc::siginfo_t,
|
||||||
|
_ctx: *mut libc::c_void,
|
||||||
|
) {
|
||||||
|
let slot_ptr = crate::preempt::current_slot_ptr();
|
||||||
|
if !slot_ptr.is_null() {
|
||||||
|
let slot = &*slot_ptr;
|
||||||
|
let top = slot.diag_stack_top.load(Ordering::Relaxed);
|
||||||
|
if top != 0 {
|
||||||
|
let reserve = slot.diag_stack_reserve.load(Ordering::Relaxed);
|
||||||
|
let guard = slot.diag_stack_guard.load(Ordering::Relaxed);
|
||||||
|
let pid = slot.diag_pid.load(Ordering::Relaxed);
|
||||||
|
let addr = (*info).si_addr() as usize;
|
||||||
|
match classify(addr, top, reserve, guard) {
|
||||||
|
FaultClass::Guard => {
|
||||||
|
let mut b = Buf::new();
|
||||||
|
b.s("smarm: actor ");
|
||||||
|
b.pid(pid);
|
||||||
|
b.s(" overflowed its stack: fault in the guard region, depth-at-fault=");
|
||||||
|
b.u(top - addr);
|
||||||
|
b.s(" bytes (reserve=");
|
||||||
|
b.u(reserve);
|
||||||
|
b.s(", guard=");
|
||||||
|
b.u(guard);
|
||||||
|
b.s("). Raise stack_reserve (SpawnOpts or Config).\n");
|
||||||
|
b.emit();
|
||||||
|
die_by_default();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
FaultClass::Overshoot(below) => {
|
||||||
|
let mut b = Buf::new();
|
||||||
|
b.s("smarm: actor ");
|
||||||
|
b.pid(pid);
|
||||||
|
b.s(" probably overflowed its stack: fault ");
|
||||||
|
b.u(below);
|
||||||
|
b.s(" bytes below the guard - an unprobed (FFI?) frame stepped over it (reserve=");
|
||||||
|
b.u(reserve);
|
||||||
|
b.s(", guard=");
|
||||||
|
b.u(guard);
|
||||||
|
b.s("). Raise stack_guard or stack_reserve.\n");
|
||||||
|
b.emit();
|
||||||
|
die_by_default();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
FaultClass::Foreign => {}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// Not ours: put back whoever was there before us and refault into them.
|
||||||
|
let prior = &*std::ptr::addr_of!(PRIOR);
|
||||||
|
libc::sigaction(libc::SIGSEGV, prior.as_ptr(), std::ptr::null_mut());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Reset SIGSEGV to default disposition; returning from the handler then
|
||||||
|
/// refaults at the same instruction and the process dies the normal death
|
||||||
|
/// (core-dumpable, correct wait status), exactly as if we were never here —
|
||||||
|
/// but with the message already on stderr.
|
||||||
|
unsafe fn die_by_default() {
|
||||||
|
let mut dfl: libc::sigaction = std::mem::zeroed();
|
||||||
|
dfl.sa_sigaction = libc::SIG_DFL;
|
||||||
|
libc::sigemptyset(&mut dfl.sa_mask);
|
||||||
|
libc::sigaction(libc::SIGSEGV, &dfl, std::ptr::null_mut());
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Async-signal-safe formatting: fixed buffer, decimal itoa, one write(2).
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
struct Buf {
|
||||||
|
b: [u8; 320],
|
||||||
|
len: usize,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Buf {
|
||||||
|
fn new() -> Self {
|
||||||
|
Buf { b: [0; 320], len: 0 }
|
||||||
|
}
|
||||||
|
fn s(&mut self, s: &str) {
|
||||||
|
for &c in s.as_bytes() {
|
||||||
|
if self.len < self.b.len() {
|
||||||
|
self.b[self.len] = c;
|
||||||
|
self.len += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
fn u(&mut self, mut n: usize) {
|
||||||
|
let mut tmp = [0u8; 20];
|
||||||
|
let mut i = tmp.len();
|
||||||
|
loop {
|
||||||
|
i -= 1;
|
||||||
|
tmp[i] = b'0' + (n % 10) as u8;
|
||||||
|
n /= 10;
|
||||||
|
if n == 0 {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for &c in &tmp[i..] {
|
||||||
|
if self.len < self.b.len() {
|
||||||
|
self.b[self.len] = c;
|
||||||
|
self.len += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
/// `idx.gen`, unpacked from the install-time packing.
|
||||||
|
fn pid(&mut self, packed: u64) {
|
||||||
|
self.u((packed >> 32) as usize);
|
||||||
|
self.s(".");
|
||||||
|
self.u((packed & 0xffff_ffff) as usize);
|
||||||
|
}
|
||||||
|
fn emit(&self) {
|
||||||
|
unsafe {
|
||||||
|
libc::write(2, self.b.as_ptr() as *const libc::c_void, self.len);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Classifier units — the arithmetic edges, before anything integrates.
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::{classify, FaultClass, OVERSHOOT_SLOP};
|
||||||
|
|
||||||
|
const PG: usize = 4096;
|
||||||
|
// A synthetic stack far from address-space edges: top at 1 GiB.
|
||||||
|
const TOP: usize = 1 << 30;
|
||||||
|
const RESERVE: usize = 16 * PG;
|
||||||
|
const GUARD: usize = 4 * PG;
|
||||||
|
const GUARD_HI: usize = TOP - RESERVE;
|
||||||
|
const GUARD_LO: usize = GUARD_HI - GUARD;
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn inside_guard_both_edges() {
|
||||||
|
assert_eq!(classify(GUARD_LO, TOP, RESERVE, GUARD), FaultClass::Guard);
|
||||||
|
assert_eq!(classify(GUARD_HI - 1, TOP, RESERVE, GUARD), FaultClass::Guard);
|
||||||
|
assert_eq!(classify(GUARD_LO + GUARD / 2, TOP, RESERVE, GUARD), FaultClass::Guard);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn usable_region_is_foreign() {
|
||||||
|
// A fault inside the RW stack itself isn't a guard hit and must not
|
||||||
|
// be explained as one.
|
||||||
|
assert_eq!(classify(GUARD_HI, TOP, RESERVE, GUARD), FaultClass::Foreign);
|
||||||
|
assert_eq!(classify(TOP - 1, TOP, RESERVE, GUARD), FaultClass::Foreign);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn above_top_is_foreign() {
|
||||||
|
assert_eq!(classify(TOP, TOP, RESERVE, GUARD), FaultClass::Foreign);
|
||||||
|
assert_eq!(classify(TOP + PG, TOP, RESERVE, GUARD), FaultClass::Foreign);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn overshoot_window_edges() {
|
||||||
|
assert_eq!(
|
||||||
|
classify(GUARD_LO - 1, TOP, RESERVE, GUARD),
|
||||||
|
FaultClass::Overshoot(1)
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
classify(GUARD_LO - OVERSHOOT_SLOP, TOP, RESERVE, GUARD),
|
||||||
|
FaultClass::Overshoot(OVERSHOOT_SLOP)
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
classify(GUARD_LO - OVERSHOOT_SLOP - 1, TOP, RESERVE, GUARD),
|
||||||
|
FaultClass::Foreign
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn low_address_stack_saturates_not_wraps() {
|
||||||
|
// A stack mapped so low that the slop window would underflow: the
|
||||||
|
// window clips to 0 instead of wrapping around the address space.
|
||||||
|
let top = RESERVE + GUARD + PG; // guard_lo == PG
|
||||||
|
assert_eq!(classify(0, top, RESERVE, GUARD), FaultClass::Overshoot(PG));
|
||||||
|
// Null-page fault still classified only because it IS within slop
|
||||||
|
// here; with a normal-height stack it is Foreign (covered above by
|
||||||
|
// the window-edge test at realistic addresses).
|
||||||
|
}
|
||||||
|
}
|
||||||
+226
-13
@@ -1,32 +1,45 @@
|
|||||||
//! mmap-based growable stack with a guard page below.
|
//! mmap-based actor stack with a PROT_NONE guard region below (RFC 019).
|
||||||
//!
|
//!
|
||||||
//! Layout (low → high address):
|
//! Layout (low → high address):
|
||||||
//! [ guard page (PROT_NONE) | stack region ]
|
//! [ guard region (PROT_NONE) | stack region ]
|
||||||
//! ^ top() — initial stack pointer
|
//! ^ top() — initial stack pointer
|
||||||
//!
|
//!
|
||||||
//! Stacks grow downward. Overflow lands in the guard page → SIGSEGV.
|
//! Stacks grow downward. Overflow lands in the guard region → SIGSEGV.
|
||||||
|
//!
|
||||||
|
//! Both the usable reserve and the guard are caller-chosen (page-rounded).
|
||||||
|
//! The reserve is a *virtual* reservation: anonymous mmap is demand-paged,
|
||||||
|
//! so RSS is touched-pages, not reserve × actors. The guard costs address
|
||||||
|
//! space only. A wide guard (the runtime defaults to 64 KiB) exists for
|
||||||
|
//! unprobed FFI frames: Rust frames touch pages in order (probestack), so
|
||||||
|
//! one page catches Rust overflow, but a C frame with a large local can
|
||||||
|
//! step over a single page in one `sub rsp`.
|
||||||
|
|
||||||
use std::io;
|
use std::io;
|
||||||
|
|
||||||
pub struct Stack {
|
pub struct Stack {
|
||||||
/// Bottom of the entire mmap'd region (start of guard page).
|
/// Bottom of the entire mmap'd region (start of the guard).
|
||||||
base: *mut u8,
|
base: *mut u8,
|
||||||
/// Total mmap'd size: guard_size + stack_size.
|
/// Total mmap'd size: guard_size + stack_size.
|
||||||
total_size: usize,
|
total_size: usize,
|
||||||
/// Usable stack size (excluding guard page).
|
/// Usable stack size (excluding the guard).
|
||||||
stack_size: usize,
|
stack_size: usize,
|
||||||
|
/// PROT_NONE region below the usable stack.
|
||||||
|
guard_size: usize,
|
||||||
}
|
}
|
||||||
|
|
||||||
// Stack owns its memory; safe to send across threads.
|
// Stack owns its memory; safe to send across threads.
|
||||||
unsafe impl Send for Stack {}
|
unsafe impl Send for Stack {}
|
||||||
|
|
||||||
impl Stack {
|
impl Stack {
|
||||||
/// Allocate a new stack. `stack_size` is the usable region; one page is
|
/// Allocate a new stack. `stack_size` is the usable region; `guard_size`
|
||||||
/// added below as a guard page. Both are rounded up to the page size.
|
/// is mapped PROT_NONE below it. Both are rounded up to the page size
|
||||||
pub fn new(stack_size: usize) -> io::Result<Self> {
|
/// and must be non-zero.
|
||||||
|
pub fn new(stack_size: usize, guard_size: usize) -> io::Result<Self> {
|
||||||
|
assert!(stack_size > 0, "stack_size must be non-zero");
|
||||||
|
assert!(guard_size > 0, "guard_size must be non-zero");
|
||||||
let page = page_size();
|
let page = page_size();
|
||||||
let stack_size = round_up(stack_size, page);
|
let stack_size = round_up(stack_size, page);
|
||||||
let guard_size = page;
|
let guard_size = round_up(guard_size, page);
|
||||||
let total_size = guard_size + stack_size;
|
let total_size = guard_size + stack_size;
|
||||||
|
|
||||||
let base = unsafe {
|
let base = unsafe {
|
||||||
@@ -53,7 +66,7 @@ impl Stack {
|
|||||||
return Err(err);
|
return Err(err);
|
||||||
}
|
}
|
||||||
|
|
||||||
Ok(Self { base, total_size, stack_size })
|
Ok(Self { base, total_size, stack_size, guard_size })
|
||||||
}
|
}
|
||||||
|
|
||||||
/// 16-byte-aligned top of the usable region.
|
/// 16-byte-aligned top of the usable region.
|
||||||
@@ -62,14 +75,54 @@ impl Stack {
|
|||||||
(raw_top & !15) as *mut u8
|
(raw_top & !15) as *mut u8
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Pointer to the bottom of the usable region (just above the guard page).
|
/// Pointer to the bottom of the usable region (just above the guard).
|
||||||
pub fn usable_base(&self) -> *mut u8 {
|
pub fn usable_base(&self) -> *mut u8 {
|
||||||
unsafe { self.base.add(page_size()) }
|
unsafe { self.base.add(self.guard_size) }
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn stack_size(&self) -> usize {
|
pub fn stack_size(&self) -> usize {
|
||||||
self.stack_size
|
self.stack_size
|
||||||
}
|
}
|
||||||
|
|
||||||
|
pub fn guard_size(&self) -> usize {
|
||||||
|
self.guard_size
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `(stack_size, guard_size)` after page rounding. The pool rule
|
||||||
|
/// (RFC 019 §1) compares this against the runtime defaults: only
|
||||||
|
/// default-shaped stacks are pooled.
|
||||||
|
pub fn shape(&self) -> (usize, usize) {
|
||||||
|
(self.stack_size, self.guard_size)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Pool-recycle zap (RFC 019 §6): `MADV_DONTNEED` everything below the
|
||||||
|
/// retained entry end `[top − retain, top)` — the span the next actor's
|
||||||
|
/// shallow frames land in stays resident, the dead spike below it is
|
||||||
|
/// released. The stack is unowned at the call site (its actor is dead),
|
||||||
|
/// so a synchronous eager zap races nothing and the RSS drop is
|
||||||
|
/// immediate — a museum of worst-case spikes is exactly what a pool must
|
||||||
|
/// not be; DONTNEED's ~8× per-page cost vs FREE is irrelevant off the
|
||||||
|
/// hot path. Advisory like the park-path shrink: a failure degrades to
|
||||||
|
/// "the pool keeps RSS", never to incorrectness. No-op (no syscall) when
|
||||||
|
/// `retain` covers the whole usable region — i.e. always, at the 64 KiB
|
||||||
|
/// default reserve.
|
||||||
|
pub(crate) fn recycle_zap(&self, retain: usize) {
|
||||||
|
if let Some((off, len)) = retain_range(self.stack_size, retain, page_size()) {
|
||||||
|
unsafe {
|
||||||
|
libc::madvise(
|
||||||
|
self.usable_base().add(off) as *mut libc::c_void,
|
||||||
|
len,
|
||||||
|
libc::MADV_DONTNEED,
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Round `n` up to whole pages — the same rounding `Stack::new` applies, so
|
||||||
|
/// runtime defaults stored pre-rounded compare exactly against [`Stack::shape`].
|
||||||
|
pub(crate) fn round_to_pages(n: usize) -> usize {
|
||||||
|
round_up(n, page_size())
|
||||||
}
|
}
|
||||||
|
|
||||||
impl Drop for Stack {
|
impl Drop for Stack {
|
||||||
@@ -80,10 +133,170 @@ impl Drop for Stack {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
fn page_size() -> usize {
|
pub(crate) fn page_size() -> usize {
|
||||||
unsafe { libc::sysconf(libc::_SC_PAGESIZE) as usize }
|
unsafe { libc::sysconf(libc::_SC_PAGESIZE) as usize }
|
||||||
}
|
}
|
||||||
|
|
||||||
fn round_up(n: usize, align: usize) -> usize {
|
fn round_up(n: usize, align: usize) -> usize {
|
||||||
(n + align - 1) & !(align - 1)
|
(n + align - 1) & !(align - 1)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The whole-page span the park-path shrink may `MADV_FREE` (RFC 019 §3):
|
||||||
|
/// `[page_up(hwm), page_down(sp − redzone))`, or `None` if no full page fits.
|
||||||
|
///
|
||||||
|
/// `hwm` is the sampled high-water (deepest observed `sp`); everything in
|
||||||
|
/// `[hwm, sp)` is below the live frame and dead by definition. One page of
|
||||||
|
/// redzone stays resident under live `sp` — it covers the SysV 128-byte red
|
||||||
|
/// zone plus spill margin with room to spare. Rounding is inward on both
|
||||||
|
/// ends so the result can never touch the redzone, cross `sp`, or dip below
|
||||||
|
/// `hwm`; all arithmetic is checked so adversarial inputs (`sp < redzone`,
|
||||||
|
/// `hwm ≥ sp`, values near the address-space edges) collapse to `None`
|
||||||
|
/// rather than a wild or negative-length range.
|
||||||
|
pub(crate) fn shrink_range(hwm: usize, sp: usize, page: usize) -> Option<(usize, usize)> {
|
||||||
|
debug_assert!(page.is_power_of_two());
|
||||||
|
if hwm >= sp {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
let redzone = page;
|
||||||
|
let end = sp.checked_sub(redzone)? & !(page - 1); // page_down(sp − redzone)
|
||||||
|
let start = hwm.checked_add(page - 1)? & !(page - 1); // page_up(hwm)
|
||||||
|
if end > start {
|
||||||
|
Some((start, end - start))
|
||||||
|
} else {
|
||||||
|
None
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The `(offset_from_usable_base, len)` span the pool recycle DONTNEEDs
|
||||||
|
/// (RFC 019 §6): everything below the retained entry end. "Bottom RETAIN of
|
||||||
|
/// the stack" is read stack-wise (entry frames = highest addresses of a
|
||||||
|
/// downward stack): the retained span is `[top − page_up(retain), top)`, the
|
||||||
|
/// zapped span is the rest — retaining the low-address deep end instead
|
||||||
|
/// would keep the coldest pages and release the ones the next actor faults
|
||||||
|
/// first. `retain` rounds *up* to whole pages (retain more, zap less), so
|
||||||
|
/// with `stack_size` page-rounded by `Stack::new` the result is always
|
||||||
|
/// page-aligned. Checked math: `retain ≥ stack_size` (notably the default
|
||||||
|
/// 64 KiB reserve with the 64 KiB RETAIN) and overflow collapse to `None`.
|
||||||
|
pub(crate) fn retain_range(stack_size: usize, retain: usize, page: usize) -> Option<(usize, usize)> {
|
||||||
|
debug_assert!(page.is_power_of_two());
|
||||||
|
let retain = retain.checked_add(page - 1)? & !(page - 1); // page_up(retain)
|
||||||
|
let len = stack_size.checked_sub(retain)?;
|
||||||
|
if len == 0 {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
Some((0, len))
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::{retain_range, shrink_range};
|
||||||
|
|
||||||
|
const PG: usize = 4096;
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn retain_covers_whole_stack_is_a_noop() {
|
||||||
|
// The default config: reserve == RETAIN == 64 KiB. No zap, no syscall.
|
||||||
|
assert_eq!(retain_range(16 * PG, 16 * PG, PG), None);
|
||||||
|
assert_eq!(retain_range(PG, PG, PG), None);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn retain_larger_than_stack_is_a_noop() {
|
||||||
|
assert_eq!(retain_range(16 * PG, 17 * PG, PG), None);
|
||||||
|
assert_eq!(retain_range(PG, usize::MAX, PG), None); // page_up overflows
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn retain_zero_zaps_everything() {
|
||||||
|
assert_eq!(retain_range(16 * PG, 0, PG), Some((0, 16 * PG)));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn retain_rounds_up_zapping_less() {
|
||||||
|
// 1 byte of retain keeps a whole page.
|
||||||
|
assert_eq!(retain_range(16 * PG, 1, PG), Some((0, 15 * PG)));
|
||||||
|
assert_eq!(retain_range(16 * PG, PG + 1, PG), Some((0, 14 * PG)));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn retain_one_page_short_of_stack() {
|
||||||
|
assert_eq!(retain_range(2 * PG, PG, PG), Some((0, PG)));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn retain_range_is_page_aligned() {
|
||||||
|
for size_pg in [1usize, 2, 3, 16, 1024] {
|
||||||
|
for retain in [0usize, 1, PG - 1, PG, PG + 1, 4 * PG, size_pg * PG] {
|
||||||
|
if let Some((off, len)) = retain_range(size_pg * PG, retain, PG) {
|
||||||
|
assert_eq!(off, 0);
|
||||||
|
assert_eq!(len % PG, 0);
|
||||||
|
assert!(len <= size_pg * PG);
|
||||||
|
assert!(len > 0);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn empty_and_inverted_spans_are_none() {
|
||||||
|
assert_eq!(shrink_range(0x8000_0000, 0x8000_0000, PG), None); // hwm == sp
|
||||||
|
assert_eq!(shrink_range(0x8000_1000, 0x8000_0000, PG), None); // hwm > sp
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn span_smaller_than_redzone_plus_page_is_none() {
|
||||||
|
let sp = 0x8000_0000;
|
||||||
|
// Everything within redzone+1 page of sp: no full page clears both
|
||||||
|
// the redzone and the page_up(hwm) rounding.
|
||||||
|
assert_eq!(shrink_range(sp - PG, sp, PG), None);
|
||||||
|
assert_eq!(shrink_range(sp - 2 * PG + 1, sp, PG), None);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn exact_two_pages_frees_one() {
|
||||||
|
let sp = 0x8000_0000;
|
||||||
|
let hwm = sp - 2 * PG;
|
||||||
|
// [hwm, hwm+PG) frees; [sp−PG, sp) is redzone.
|
||||||
|
assert_eq!(shrink_range(hwm, sp, PG), Some((hwm, PG)));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn unaligned_ends_round_inward() {
|
||||||
|
let sp = 0x8000_0123; // live sp mid-page
|
||||||
|
let hwm = 0x7f00_0abc; // high-water mid-page
|
||||||
|
let (start, len) = shrink_range(hwm, sp, PG).unwrap();
|
||||||
|
assert_eq!(start % PG, 0);
|
||||||
|
assert_eq!(len % PG, 0);
|
||||||
|
assert!(start >= hwm); // never below the sampled high-water
|
||||||
|
assert!(start + len <= (sp - PG) & !(PG - 1)); // never into the redzone
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn result_never_crosses_sp() {
|
||||||
|
// Sweep hwm across every offset of the page straddling the boundary.
|
||||||
|
let sp = 0x8000_0000 + 137;
|
||||||
|
for hwm in (sp - 4 * PG)..(sp) {
|
||||||
|
if let Some((start, len)) = shrink_range(hwm, sp, PG) {
|
||||||
|
assert!(start >= hwm);
|
||||||
|
assert!(start + len + PG <= sp + PG); // end ≤ page_down(sp − PG) < sp
|
||||||
|
assert!(len > 0);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn underflow_near_zero_is_none() {
|
||||||
|
assert_eq!(shrink_range(0, PG - 1, PG), None); // sp < redzone
|
||||||
|
assert_eq!(shrink_range(0, 0, PG), None);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn big_span_frees_interior() {
|
||||||
|
let sp = 0x8000_0000;
|
||||||
|
let spike = 4 * 1024 * 1024;
|
||||||
|
let hwm = sp - spike;
|
||||||
|
let (start, len) = shrink_range(hwm, sp, PG).unwrap();
|
||||||
|
assert_eq!(start, hwm); // aligned input: starts exactly at hwm
|
||||||
|
assert_eq!(len, spike - PG); // everything but the redzone page
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
+11
-2
@@ -6,10 +6,19 @@
|
|||||||
//! Build the loom models with: `RUSTFLAGS="--cfg loom" cargo test --lib --release`
|
//! Build the loom models with: `RUSTFLAGS="--cfg loom" cargo test --lib --release`
|
||||||
|
|
||||||
#[cfg(loom)]
|
#[cfg(loom)]
|
||||||
pub(crate) use loom::sync::atomic::{AtomicU64, AtomicUsize, Ordering};
|
pub(crate) use loom::sync::atomic::{fence, AtomicU64, AtomicUsize, Ordering};
|
||||||
|
|
||||||
#[cfg(not(loom))]
|
#[cfg(not(loom))]
|
||||||
pub(crate) use std::sync::atomic::{AtomicU64, AtomicUsize, Ordering};
|
pub(crate) use std::sync::atomic::{fence, AtomicU64, AtomicUsize, Ordering};
|
||||||
|
|
||||||
|
// park.rs condvar-parker (loom + non-Linux builds only; the Linux non-loom
|
||||||
|
// build parks on a futex and never touches these — gating them identically
|
||||||
|
// keeps the default build free of unused imports).
|
||||||
|
#[cfg(loom)]
|
||||||
|
pub(crate) use loom::sync::{Condvar, Mutex};
|
||||||
|
|
||||||
|
#[cfg(all(not(loom), not(target_os = "linux")))]
|
||||||
|
pub(crate) use std::sync::{Condvar, Mutex};
|
||||||
|
|
||||||
/// `UnsafeCell` with loom's `with`/`with_mut` access API; pass-through cost
|
/// `UnsafeCell` with loom's `with`/`with_mut` access API; pass-through cost
|
||||||
/// is zero in normal builds (`#[inline]`, newtype over std's cell).
|
/// is zero in normal builds (`#[inline]`, newtype over std's cell).
|
||||||
|
|||||||
+37
-1
@@ -141,6 +141,14 @@ impl PartialOrd for Entry {
|
|||||||
|
|
||||||
#[derive(Default)]
|
#[derive(Default)]
|
||||||
pub struct Timers {
|
pub struct Timers {
|
||||||
|
/// RFC 018: the scheduler coordination layer. Attached once at
|
||||||
|
/// `RuntimeInner::new`; every insert notes its deadline (min-maintained
|
||||||
|
/// snapshot for the busy-path due-check + the timekeeper re-arm wake)
|
||||||
|
/// and every pop/clear re-anchors the snapshot to the heap minimum.
|
||||||
|
/// All calls happen under the timers mutex — the serialization the
|
||||||
|
/// coordinator's timer protocol mandates. `None` only in unit tests
|
||||||
|
/// that construct a bare `Timers`.
|
||||||
|
coord: Option<std::sync::Arc<crate::park::Coordinator>>,
|
||||||
/// Reverse-wrapped so the smallest deadline is at the top.
|
/// Reverse-wrapped so the smallest deadline is at the top.
|
||||||
heap: BinaryHeap<Reverse<Entry>>,
|
heap: BinaryHeap<Reverse<Entry>>,
|
||||||
/// Monotonic counter for the tiebreaker `seq` field (and the `TimerId` of a
|
/// Monotonic counter for the tiebreaker `seq` field (and the `TimerId` of a
|
||||||
@@ -157,7 +165,18 @@ pub struct Timers {
|
|||||||
|
|
||||||
impl Timers {
|
impl Timers {
|
||||||
pub fn new() -> Self {
|
pub fn new() -> Self {
|
||||||
Self { heap: BinaryHeap::new(), next_seq: 0, armed: std::collections::HashSet::new() }
|
Self {
|
||||||
|
coord: None,
|
||||||
|
heap: BinaryHeap::new(),
|
||||||
|
next_seq: 0,
|
||||||
|
armed: std::collections::HashSet::new(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Attach the scheduler coordination layer (RFC 018). Called once, at
|
||||||
|
/// runtime construction, before any scheduler thread exists.
|
||||||
|
pub(crate) fn attach_coordinator(&mut self, c: std::sync::Arc<crate::park::Coordinator>) {
|
||||||
|
self.coord = Some(c);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Insert a `Sleep` timer. Convenience for the common case.
|
/// Insert a `Sleep` timer. Convenience for the common case.
|
||||||
@@ -242,6 +261,13 @@ impl Timers {
|
|||||||
#[cfg(feature = "smarm-causal")]
|
#[cfg(feature = "smarm-causal")]
|
||||||
wall,
|
wall,
|
||||||
}));
|
}));
|
||||||
|
// RFC 018: publish the (possibly new-minimum) deadline to the
|
||||||
|
// busy-path snapshot and wake the timekeeper if it is parked
|
||||||
|
// toward a later one. We hold the timers mutex — the mandated
|
||||||
|
// serialization for both.
|
||||||
|
if let Some(c) = &self.coord {
|
||||||
|
c.note_deadline(deadline);
|
||||||
|
}
|
||||||
seq
|
seq
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -255,6 +281,9 @@ impl Timers {
|
|||||||
pub fn clear(&mut self) {
|
pub fn clear(&mut self) {
|
||||||
self.heap.clear();
|
self.heap.clear();
|
||||||
self.armed.clear();
|
self.armed.clear();
|
||||||
|
if let Some(c) = &self.coord {
|
||||||
|
c.refresh_deadline(None);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Soonest pending deadline, or `None` if the heap is empty.
|
/// Soonest pending deadline, or `None` if the heap is empty.
|
||||||
@@ -324,6 +353,13 @@ impl Timers {
|
|||||||
}
|
}
|
||||||
out.push(entry);
|
out.push(entry);
|
||||||
}
|
}
|
||||||
|
// RFC 018: re-anchor the busy-path snapshot to the new heap minimum
|
||||||
|
// (still under the timers mutex). A causal-shift re-queue above went
|
||||||
|
// through `heap.push` directly, so this peek is the one place the
|
||||||
|
// snapshot is guaranteed to catch up.
|
||||||
|
if let Some(c) = &self.coord {
|
||||||
|
c.refresh_deadline(self.peek_deadline());
|
||||||
|
}
|
||||||
out
|
out
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
+5
-5
@@ -23,7 +23,7 @@ extern "C-unwind" fn actor_simple() {
|
|||||||
#[test]
|
#[test]
|
||||||
fn actor_runs_and_returns_to_scheduler() {
|
fn actor_runs_and_returns_to_scheduler() {
|
||||||
reset_log();
|
reset_log();
|
||||||
let stack = Stack::new(64 * 1024).unwrap();
|
let stack = Stack::new(64 * 1024, 4096).unwrap();
|
||||||
let sp = init_actor_stack(stack.top(), actor_simple);
|
let sp = init_actor_stack(stack.top(), actor_simple);
|
||||||
set_actor_sp(sp);
|
set_actor_sp(sp);
|
||||||
unsafe { switch_to_actor() };
|
unsafe { switch_to_actor() };
|
||||||
@@ -40,7 +40,7 @@ extern "C-unwind" fn actor_two_steps() {
|
|||||||
#[test]
|
#[test]
|
||||||
fn actor_yields_and_resumes() {
|
fn actor_yields_and_resumes() {
|
||||||
reset_log();
|
reset_log();
|
||||||
let stack = Stack::new(64 * 1024).unwrap();
|
let stack = Stack::new(64 * 1024, 4096).unwrap();
|
||||||
let sp = init_actor_stack(stack.top(), actor_two_steps);
|
let sp = init_actor_stack(stack.top(), actor_two_steps);
|
||||||
set_actor_sp(sp);
|
set_actor_sp(sp);
|
||||||
|
|
||||||
@@ -85,7 +85,7 @@ extern "C-unwind" fn actor_reg_check() {
|
|||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn callee_saved_registers_survive_yield() {
|
fn callee_saved_registers_survive_yield() {
|
||||||
let stack = Stack::new(64 * 1024).unwrap();
|
let stack = Stack::new(64 * 1024, 4096).unwrap();
|
||||||
let sp = init_actor_stack(stack.top(), actor_reg_check);
|
let sp = init_actor_stack(stack.top(), actor_reg_check);
|
||||||
set_actor_sp(sp);
|
set_actor_sp(sp);
|
||||||
unsafe { switch_to_actor(); switch_to_actor(); }
|
unsafe { switch_to_actor(); switch_to_actor(); }
|
||||||
@@ -117,8 +117,8 @@ extern "C-unwind" fn actor_b() {
|
|||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn two_actors_dont_corrupt_each_other() {
|
fn two_actors_dont_corrupt_each_other() {
|
||||||
let stack_a = Stack::new(64 * 1024).unwrap();
|
let stack_a = Stack::new(64 * 1024, 4096).unwrap();
|
||||||
let stack_b = Stack::new(64 * 1024).unwrap();
|
let stack_b = Stack::new(64 * 1024, 4096).unwrap();
|
||||||
|
|
||||||
let sp_a = init_actor_stack(stack_a.top(), actor_a);
|
let sp_a = init_actor_stack(stack_a.top(), actor_a);
|
||||||
let sp_b = init_actor_stack(stack_b.top(), actor_b);
|
let sp_b = init_actor_stack(stack_b.top(), actor_b);
|
||||||
|
|||||||
@@ -237,6 +237,13 @@ fn tree_from_nests_children_and_reroots_orphans() {
|
|||||||
overruns: 0,
|
overruns: 0,
|
||||||
messages_received: 0,
|
messages_received: 0,
|
||||||
budget_cycles: 0,
|
budget_cycles: 0,
|
||||||
|
stack: smarm::StackInfo {
|
||||||
|
reserve: 0,
|
||||||
|
guard: 0,
|
||||||
|
depth_high_water: 0,
|
||||||
|
parks_since_shrink: 0,
|
||||||
|
shrinks: 0,
|
||||||
|
},
|
||||||
};
|
};
|
||||||
|
|
||||||
let snap = RuntimeSnapshot {
|
let snap = RuntimeSnapshot {
|
||||||
@@ -352,3 +359,122 @@ fn budget_cycles_accumulate_when_enabled() {
|
|||||||
h.join().unwrap();
|
h.join().unwrap();
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// RFC 019 §8 — the stack introspection surface.
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
/// Burn ~`frames` × 4 KiB of stack with a yield at max depth, so the context
|
||||||
|
/// save samples the high-water there (RFC 019 §2: hwm is SAMPLED at
|
||||||
|
/// deschedule, not tracked continuously).
|
||||||
|
#[inline(never)]
|
||||||
|
fn burn_stack_yielding(frames: usize) -> u64 {
|
||||||
|
let mut local = [0u8; 4096];
|
||||||
|
local[0] = frames as u8;
|
||||||
|
let below = if frames == 0 {
|
||||||
|
smarm::yield_now();
|
||||||
|
0
|
||||||
|
} else {
|
||||||
|
burn_stack_yielding(frames - 1)
|
||||||
|
};
|
||||||
|
std::hint::black_box(&mut local);
|
||||||
|
below.wrapping_add(local[0] as u64)
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn stack_info_reports_defaults_and_sampled_depth() {
|
||||||
|
run(|| {
|
||||||
|
let (ready_tx, ready_rx) = channel::<()>();
|
||||||
|
let (gate_tx, gate_rx) = channel::<()>();
|
||||||
|
|
||||||
|
let h = spawn(move || {
|
||||||
|
// ~32 KiB deep with a yield at the bottom: the sample point.
|
||||||
|
std::hint::black_box(burn_stack_yielding(8));
|
||||||
|
ready_tx.send(()).unwrap();
|
||||||
|
gate_rx.recv().unwrap();
|
||||||
|
});
|
||||||
|
ready_rx.recv().unwrap();
|
||||||
|
|
||||||
|
let info = spin_until(h.pid(), |a| a.state == ActorState::Parked);
|
||||||
|
let s = info.stack;
|
||||||
|
assert_eq!(s.reserve, 64 * 1024, "default reserve");
|
||||||
|
assert_eq!(s.guard, 1024 * 1024, "default guard (kernel stack_guard_gap convention)");
|
||||||
|
assert!(
|
||||||
|
s.depth_high_water >= 8 * 4096,
|
||||||
|
"hwm sampled at the deep yield: expected ≥ 32 KiB, got {}",
|
||||||
|
s.depth_high_water
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
s.depth_high_water < s.reserve,
|
||||||
|
"depth {} cannot exceed the reserve {}",
|
||||||
|
s.depth_high_water,
|
||||||
|
s.reserve
|
||||||
|
);
|
||||||
|
// Parked at the gate right now, never shrunk (64 KiB reserve cannot
|
||||||
|
// cross the shrink threshold).
|
||||||
|
assert!(s.parks_since_shrink >= 1, "the gate park must be counted");
|
||||||
|
assert_eq!(s.shrinks, 0);
|
||||||
|
|
||||||
|
gate_tx.send(()).unwrap();
|
||||||
|
h.join().unwrap();
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn stack_info_shrink_counters_are_live() {
|
||||||
|
use smarm::runtime::{Config, SHRINK_COOLDOWN, SHRINK_THRESHOLD};
|
||||||
|
use smarm::{spawn_with, SpawnOpts};
|
||||||
|
|
||||||
|
let rt = smarm::runtime::init(Config::exact(1));
|
||||||
|
rt.run(|| {
|
||||||
|
let (park_tx, park_rx) = channel::<()>();
|
||||||
|
|
||||||
|
let spike = 768 * 4096;
|
||||||
|
assert!(spike > SHRINK_THRESHOLD);
|
||||||
|
let worker = spawn_with(
|
||||||
|
SpawnOpts { stack_reserve: Some(8 * 1024 * 1024), ..SpawnOpts::default() },
|
||||||
|
move || {
|
||||||
|
std::hint::black_box(burn_stack_yielding(768));
|
||||||
|
for _ in 0..(SHRINK_COOLDOWN + 8) {
|
||||||
|
park_rx.recv().unwrap();
|
||||||
|
}
|
||||||
|
},
|
||||||
|
);
|
||||||
|
|
||||||
|
let wpid = worker.pid();
|
||||||
|
// Before any parks complete: the spike depth is visible.
|
||||||
|
let info = spin_until(wpid, |a| a.state == ActorState::Parked);
|
||||||
|
assert!(
|
||||||
|
info.stack.depth_high_water >= spike,
|
||||||
|
"spike should be sampled: {} < {spike}",
|
||||||
|
info.stack.depth_high_water
|
||||||
|
);
|
||||||
|
|
||||||
|
// Cross the cooldown, then read the counters live while the worker
|
||||||
|
// is parked waiting for the remaining rounds (post-join the slot is
|
||||||
|
// reclaimed and the generation check correctly hides it).
|
||||||
|
for _ in 0..(SHRINK_COOLDOWN + 2) {
|
||||||
|
spin_until(wpid, |a| a.state == ActorState::Parked);
|
||||||
|
park_tx.send(()).unwrap();
|
||||||
|
}
|
||||||
|
let info = spin_until(wpid, |a| a.state == ActorState::Parked && a.stack.shrinks >= 1);
|
||||||
|
let s = info.stack;
|
||||||
|
assert!(s.shrinks >= 1, "cooldown was crossed with a spike above threshold");
|
||||||
|
assert!(
|
||||||
|
s.parks_since_shrink < SHRINK_COOLDOWN,
|
||||||
|
"counter must reset at shrink: {}",
|
||||||
|
s.parks_since_shrink
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
s.depth_high_water < spike,
|
||||||
|
"hwm resets to the shallow park sp at shrink; got {}",
|
||||||
|
s.depth_high_water
|
||||||
|
);
|
||||||
|
|
||||||
|
for _ in 0..6 {
|
||||||
|
spin_until(wpid, |a| a.state == ActorState::Parked);
|
||||||
|
park_tx.send(()).unwrap();
|
||||||
|
}
|
||||||
|
worker.join().unwrap();
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|||||||
@@ -0,0 +1,69 @@
|
|||||||
|
//! RFC 018 scheduler park/wake — observable-behavior guards.
|
||||||
|
//!
|
||||||
|
//! These pin the two timer-latency properties the park/wake swap must
|
||||||
|
//! preserve or introduce:
|
||||||
|
//!
|
||||||
|
//! - `sleep_fires_under_saturation`: due timers fire even when every
|
||||||
|
//! scheduler is busy (nobody parked ⇒ no timekeeper) — the busy-path
|
||||||
|
//! due-check, ratified design point (a). The old drain phase gave this
|
||||||
|
//! for free (timers drained every loop iteration); the new design must
|
||||||
|
//! not lose it.
|
||||||
|
//! - `submillisecond_sleep_is_prompt`: a sub-ms sleep completes promptly.
|
||||||
|
//! Under the old wake pipe, `poll_wake`'s `as_millis` truncation turned
|
||||||
|
//! sub-ms deadlines into 0ms busy-polls (correct wall time, pathological
|
||||||
|
//! CPU); under park/wake the futex timespec carries full nanosecond
|
||||||
|
//! precision.
|
||||||
|
|
||||||
|
use std::sync::atomic::{AtomicBool, Ordering};
|
||||||
|
use std::sync::Arc;
|
||||||
|
use std::time::{Duration, Instant};
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn sleep_fires_under_saturation() {
|
||||||
|
let rt = smarm::runtime::init(smarm::runtime::Config::exact(4));
|
||||||
|
rt.run(|| {
|
||||||
|
let stop = Arc::new(AtomicBool::new(false));
|
||||||
|
let mut spinners = Vec::new();
|
||||||
|
// 8 spinners over 4 schedulers: the run queue never empties, so no
|
||||||
|
// scheduler ever parks and no timekeeper exists. Only the busy-path
|
||||||
|
// due-check can fire the sleeper's timer before the spinners quit.
|
||||||
|
for _ in 0..8 {
|
||||||
|
let stop = stop.clone();
|
||||||
|
spinners.push(smarm::spawn(move || {
|
||||||
|
let t0 = Instant::now();
|
||||||
|
while !stop.load(Ordering::Relaxed) && t0.elapsed() < Duration::from_secs(5) {
|
||||||
|
smarm::yield_now();
|
||||||
|
}
|
||||||
|
}));
|
||||||
|
}
|
||||||
|
let t0 = Instant::now();
|
||||||
|
smarm::sleep(Duration::from_millis(10));
|
||||||
|
let dt = t0.elapsed();
|
||||||
|
stop.store(true, Ordering::Relaxed);
|
||||||
|
for s in spinners {
|
||||||
|
let _ = s.join();
|
||||||
|
}
|
||||||
|
assert!(
|
||||||
|
dt < Duration::from_millis(500),
|
||||||
|
"10ms sleep took {dt:?} under scheduler saturation — busy-path \
|
||||||
|
timer firing is broken (timekeeper-only firing stalls under load)"
|
||||||
|
);
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn submillisecond_sleep_is_prompt() {
|
||||||
|
let rt = smarm::runtime::init(smarm::runtime::Config::exact(2));
|
||||||
|
rt.run(|| {
|
||||||
|
// Warm one iteration, then measure.
|
||||||
|
smarm::sleep(Duration::from_micros(500));
|
||||||
|
let t0 = Instant::now();
|
||||||
|
smarm::sleep(Duration::from_micros(500));
|
||||||
|
let dt = t0.elapsed();
|
||||||
|
assert!(dt >= Duration::from_micros(400), "woke early: {dt:?}");
|
||||||
|
assert!(
|
||||||
|
dt < Duration::from_millis(100),
|
||||||
|
"500µs sleep took {dt:?} — sub-ms deadline handling is broken"
|
||||||
|
);
|
||||||
|
});
|
||||||
|
}
|
||||||
@@ -517,3 +517,37 @@ fn runtime_reusable_after_root_panic() {
|
|||||||
r.run(move || ran_t.store(true, Ordering::Relaxed));
|
r.run(move || ran_t.store(true, Ordering::Relaxed));
|
||||||
assert!(ran.load(Ordering::Relaxed), "runtime unusable after root panic");
|
assert!(ran.load(Ordering::Relaxed), "runtime unusable after root panic");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// RFC 019 — Config stack knobs
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
/// Burn ~`frames` × 4 KiB of stack; probestack touches pages in order so
|
||||||
|
/// exceeding the reserve would hit the guard and SIGSEGV the process.
|
||||||
|
#[inline(never)]
|
||||||
|
fn burn_stack(frames: usize) -> u64 {
|
||||||
|
let mut local = [0u8; 4096];
|
||||||
|
local[0] = frames as u8;
|
||||||
|
let below = if frames == 0 { 0 } else { burn_stack(frames - 1) };
|
||||||
|
std::hint::black_box(&mut local);
|
||||||
|
below.wrapping_add(local[0] as u64)
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn config_stack_reserve_permits_deep_recursion() {
|
||||||
|
// ~256 KiB of frames: four times the old fixed 64 KiB reserve. With
|
||||||
|
// Config::stack_reserve raised this must complete; before RFC 019 it
|
||||||
|
// could only segfault.
|
||||||
|
let rt = smarm::runtime::init(Config::exact(1).stack_reserve(1024 * 1024));
|
||||||
|
let done = Arc::new(AtomicBool::new(false));
|
||||||
|
let done2 = done.clone();
|
||||||
|
rt.run(move || {
|
||||||
|
spawn(move || {
|
||||||
|
std::hint::black_box(burn_stack(64));
|
||||||
|
done2.store(true, Ordering::SeqCst);
|
||||||
|
})
|
||||||
|
.join()
|
||||||
|
.unwrap();
|
||||||
|
});
|
||||||
|
assert!(done.load(Ordering::SeqCst));
|
||||||
|
}
|
||||||
|
|||||||
@@ -0,0 +1,212 @@
|
|||||||
|
//! RFC 019 commit 2 — the `SpawnOpts` surface.
|
||||||
|
//!
|
||||||
|
//! Covers: per-spawn stack shape overrides on every spawn surface, the
|
||||||
|
//! `None ⇒ Config default` resolution, the pool rule from the outside
|
||||||
|
//! (obligation 4: a custom-shaped stack never enters the pool), and that a
|
||||||
|
//! big reserve behaviorally takes effect (deep recursion completes).
|
||||||
|
|
||||||
|
use smarm::runtime::{Config, DEFAULT_STACK_GUARD, DEFAULT_STACK_RESERVE};
|
||||||
|
use smarm::{
|
||||||
|
self_pid, spawn, spawn_under_with, spawn_with, GenServerBuilder, SpawnOpts,
|
||||||
|
};
|
||||||
|
use std::sync::atomic::{AtomicBool, Ordering};
|
||||||
|
use std::sync::Arc;
|
||||||
|
|
||||||
|
fn rt1() -> smarm::runtime::Runtime {
|
||||||
|
smarm::runtime::init(Config::exact(1))
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn default_spawn_has_default_shape() {
|
||||||
|
rt1().run(|| {
|
||||||
|
let h = spawn(|| {
|
||||||
|
let shape = smarm::introspect::stack_shape(self_pid()).unwrap();
|
||||||
|
assert_eq!(shape, (DEFAULT_STACK_RESERVE, DEFAULT_STACK_GUARD));
|
||||||
|
});
|
||||||
|
h.join().unwrap();
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn spawn_with_overrides_reserve_and_guard() {
|
||||||
|
rt1().run(|| {
|
||||||
|
let opts = SpawnOpts {
|
||||||
|
stack_reserve: Some(1024 * 1024),
|
||||||
|
guard_size: Some(256 * 1024),
|
||||||
|
};
|
||||||
|
let h = spawn_with(opts, || {
|
||||||
|
let shape = smarm::introspect::stack_shape(self_pid()).unwrap();
|
||||||
|
assert_eq!(shape, (1024 * 1024, 256 * 1024));
|
||||||
|
});
|
||||||
|
h.join().unwrap();
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn spawn_with_partial_override_keeps_config_default_for_the_rest() {
|
||||||
|
rt1().run(|| {
|
||||||
|
let opts = SpawnOpts { stack_reserve: Some(1024 * 1024), ..SpawnOpts::default() };
|
||||||
|
let h = spawn_with(opts, || {
|
||||||
|
let shape = smarm::introspect::stack_shape(self_pid()).unwrap();
|
||||||
|
assert_eq!(shape, (1024 * 1024, DEFAULT_STACK_GUARD));
|
||||||
|
});
|
||||||
|
h.join().unwrap();
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn spawn_with_rounds_to_pages() {
|
||||||
|
rt1().run(|| {
|
||||||
|
let opts = SpawnOpts { stack_reserve: Some(64 * 1024 + 1), guard_size: Some(4097) };
|
||||||
|
let h = spawn_with(opts, || {
|
||||||
|
let (reserve, guard) = smarm::introspect::stack_shape(self_pid()).unwrap();
|
||||||
|
assert_eq!(reserve % 4096, 0);
|
||||||
|
assert_eq!(guard % 4096, 0);
|
||||||
|
assert!(reserve >= 64 * 1024 + 1);
|
||||||
|
assert!(guard >= 4097);
|
||||||
|
});
|
||||||
|
h.join().unwrap();
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn spawn_under_with_takes_opts() {
|
||||||
|
rt1().run(|| {
|
||||||
|
let me = self_pid();
|
||||||
|
let opts = SpawnOpts { stack_reserve: Some(128 * 1024), ..SpawnOpts::default() };
|
||||||
|
let h = spawn_under_with(me, opts, || {
|
||||||
|
let (reserve, _) = smarm::introspect::stack_shape(self_pid()).unwrap();
|
||||||
|
assert_eq!(reserve, 128 * 1024);
|
||||||
|
});
|
||||||
|
h.join().unwrap();
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Obligation 4, from the outside: a dead custom stack must not be handed to
|
||||||
|
/// the next default spawn. The pool is LIFO, so if the custom stack had been
|
||||||
|
/// (wrongly) pushed at death, the very next default-shaped spawn on this
|
||||||
|
/// single-threaded runtime would pop it and report a custom shape.
|
||||||
|
#[test]
|
||||||
|
fn custom_stack_never_enters_the_pool() {
|
||||||
|
rt1().run(|| {
|
||||||
|
spawn_with(
|
||||||
|
SpawnOpts { stack_reserve: Some(512 * 1024), guard_size: Some(128 * 1024) },
|
||||||
|
|| {},
|
||||||
|
)
|
||||||
|
.join()
|
||||||
|
.unwrap();
|
||||||
|
let h = spawn(|| {
|
||||||
|
let shape = smarm::introspect::stack_shape(self_pid()).unwrap();
|
||||||
|
assert_eq!(shape, (DEFAULT_STACK_RESERVE, DEFAULT_STACK_GUARD));
|
||||||
|
});
|
||||||
|
h.join().unwrap();
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The reverse direction of the pool rule: a default-shaped stack IS pooled
|
||||||
|
/// and reused (cap = threads × 4 ≥ 1 here, pool empty at start).
|
||||||
|
#[test]
|
||||||
|
fn default_stack_is_recycled() {
|
||||||
|
rt1().run(|| {
|
||||||
|
spawn(|| {}).join().unwrap();
|
||||||
|
let h = spawn(|| {
|
||||||
|
let shape = smarm::introspect::stack_shape(self_pid()).unwrap();
|
||||||
|
assert_eq!(shape, (DEFAULT_STACK_RESERVE, DEFAULT_STACK_GUARD));
|
||||||
|
});
|
||||||
|
h.join().unwrap();
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Burn ~`frames` × 4 KiB of stack (see tests/runtime.rs twin).
|
||||||
|
#[inline(never)]
|
||||||
|
fn burn_stack(frames: usize) -> u64 {
|
||||||
|
let mut local = [0u8; 4096];
|
||||||
|
local[0] = frames as u8;
|
||||||
|
let below = if frames == 0 { 0 } else { burn_stack(frames - 1) };
|
||||||
|
std::hint::black_box(&mut local);
|
||||||
|
below.wrapping_add(local[0] as u64)
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn big_reserve_behaviorally_takes_effect() {
|
||||||
|
// ~1 MiB deep on an 8 MiB per-spawn reserve, runtime default untouched.
|
||||||
|
rt1().run(|| {
|
||||||
|
let done = Arc::new(AtomicBool::new(false));
|
||||||
|
let done2 = done.clone();
|
||||||
|
spawn_with(
|
||||||
|
SpawnOpts { stack_reserve: Some(8 * 1024 * 1024), ..SpawnOpts::default() },
|
||||||
|
move || {
|
||||||
|
std::hint::black_box(burn_stack(256));
|
||||||
|
done2.store(true, Ordering::SeqCst);
|
||||||
|
},
|
||||||
|
)
|
||||||
|
.join()
|
||||||
|
.unwrap();
|
||||||
|
assert!(done.load(Ordering::SeqCst));
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Builder surfaces
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
struct Echo;
|
||||||
|
impl smarm::GenServer for Echo {
|
||||||
|
type Call = ();
|
||||||
|
type Reply = (usize, usize);
|
||||||
|
type Cast = ();
|
||||||
|
type Info = ();
|
||||||
|
type Timer = ();
|
||||||
|
fn handle_call(&mut self, _c: ()) -> (usize, usize) {
|
||||||
|
smarm::introspect::stack_shape(self_pid()).unwrap()
|
||||||
|
}
|
||||||
|
fn handle_cast(&mut self, _c: ()) {}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn gen_server_builder_stack_opts() {
|
||||||
|
rt1().run(|| {
|
||||||
|
let server = GenServerBuilder::new(Echo)
|
||||||
|
.stack_opts(SpawnOpts { stack_reserve: Some(256 * 1024), ..SpawnOpts::default() })
|
||||||
|
.start();
|
||||||
|
let (reserve, guard) = server.call(()).unwrap();
|
||||||
|
assert_eq!(reserve, 256 * 1024);
|
||||||
|
assert_eq!(guard, DEFAULT_STACK_GUARD);
|
||||||
|
server.shutdown();
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
struct Probe;
|
||||||
|
impl smarm::Machine for Probe {
|
||||||
|
type Ev = smarm::channel::Sender<(usize, usize)>;
|
||||||
|
fn state_timeout_ev() -> Self::Ev {
|
||||||
|
unreachable!("no timers in this test")
|
||||||
|
}
|
||||||
|
fn timeout_ev(_name: &'static str) -> Self::Ev {
|
||||||
|
unreachable!("no timers in this test")
|
||||||
|
}
|
||||||
|
fn on_start(&mut self, _cx: &mut smarm::Cx<Self::Ev>) {}
|
||||||
|
fn handle(
|
||||||
|
&mut self,
|
||||||
|
ev: Self::Ev,
|
||||||
|
_cx: &mut smarm::Cx<Self::Ev>,
|
||||||
|
) -> smarm::gen_statem::Step<Self::Ev> {
|
||||||
|
let _ = ev.send(smarm::introspect::stack_shape(self_pid()).unwrap());
|
||||||
|
smarm::gen_statem::Step::Stayed
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn gen_statem_spawn_with_stack_opts() {
|
||||||
|
rt1().run(|| {
|
||||||
|
let m = smarm::gen_statem::spawn_with(
|
||||||
|
SpawnOpts { stack_reserve: Some(256 * 1024), ..SpawnOpts::default() },
|
||||||
|
Probe,
|
||||||
|
);
|
||||||
|
let (tx, rx) = smarm::channel::channel();
|
||||||
|
m.send(tx).unwrap();
|
||||||
|
let (reserve, guard) = rx.recv().unwrap();
|
||||||
|
assert_eq!(reserve, 256 * 1024);
|
||||||
|
assert_eq!(guard, DEFAULT_STACK_GUARD);
|
||||||
|
});
|
||||||
|
}
|
||||||
+77
-9
@@ -7,13 +7,13 @@ use smarm::stack::Stack;
|
|||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn top_is_16_byte_aligned() {
|
fn top_is_16_byte_aligned() {
|
||||||
let s = Stack::new(64 * 1024).unwrap();
|
let s = Stack::new(64 * 1024, 4096).unwrap();
|
||||||
assert_eq!(s.top() as usize % 16, 0);
|
assert_eq!(s.top() as usize % 16, 0);
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn top_is_within_allocation() {
|
fn top_is_within_allocation() {
|
||||||
let s = Stack::new(64 * 1024).unwrap();
|
let s = Stack::new(64 * 1024, 4096).unwrap();
|
||||||
let top = s.top() as usize;
|
let top = s.top() as usize;
|
||||||
let base = s.usable_base() as usize;
|
let base = s.usable_base() as usize;
|
||||||
assert!(top > base);
|
assert!(top > base);
|
||||||
@@ -22,7 +22,7 @@ fn top_is_within_allocation() {
|
|||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn write_and_read_top_of_stack() {
|
fn write_and_read_top_of_stack() {
|
||||||
let s = Stack::new(64 * 1024).unwrap();
|
let s = Stack::new(64 * 1024, 4096).unwrap();
|
||||||
let sentinel: u64 = 0xDEAD_BEEF_CAFE_1234;
|
let sentinel: u64 = 0xDEAD_BEEF_CAFE_1234;
|
||||||
unsafe {
|
unsafe {
|
||||||
let ptr = s.top().sub(8) as *mut u64;
|
let ptr = s.top().sub(8) as *mut u64;
|
||||||
@@ -33,7 +33,7 @@ fn write_and_read_top_of_stack() {
|
|||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn write_and_read_bottom_of_usable_region() {
|
fn write_and_read_bottom_of_usable_region() {
|
||||||
let s = Stack::new(64 * 1024).unwrap();
|
let s = Stack::new(64 * 1024, 4096).unwrap();
|
||||||
let sentinel: u64 = 0x0102_0304_0506_0708;
|
let sentinel: u64 = 0x0102_0304_0506_0708;
|
||||||
unsafe {
|
unsafe {
|
||||||
let ptr = s.usable_base() as *mut u64;
|
let ptr = s.usable_base() as *mut u64;
|
||||||
@@ -44,17 +44,17 @@ fn write_and_read_bottom_of_usable_region() {
|
|||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn small_stack_allocates() {
|
fn small_stack_allocates() {
|
||||||
assert!(Stack::new(4096).is_ok());
|
assert!(Stack::new(4096, 4096).is_ok());
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn large_stack_allocates() {
|
fn large_stack_allocates() {
|
||||||
assert!(Stack::new(8 * 1024 * 1024).is_ok());
|
assert!(Stack::new(8 * 1024 * 1024, 4096).is_ok());
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn stack_size_at_least_requested() {
|
fn stack_size_at_least_requested() {
|
||||||
let s = Stack::new(64 * 1024).unwrap();
|
let s = Stack::new(64 * 1024, 4096).unwrap();
|
||||||
assert!(s.stack_size() >= 64 * 1024);
|
assert!(s.stack_size() >= 64 * 1024);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -68,15 +68,28 @@ use std::process::Command;
|
|||||||
fn run_as_child_if_requested() {
|
fn run_as_child_if_requested() {
|
||||||
match env::var("SMARM_SUBTEST").as_deref() {
|
match env::var("SMARM_SUBTEST").as_deref() {
|
||||||
Ok("guard_page_direct") => {
|
Ok("guard_page_direct") => {
|
||||||
let s = Stack::new(64 * 1024).unwrap();
|
let s = Stack::new(64 * 1024, 4096).unwrap();
|
||||||
unsafe {
|
unsafe {
|
||||||
let guard_ptr = s.usable_base().sub(1);
|
let guard_ptr = s.usable_base().sub(1);
|
||||||
guard_ptr.write_volatile(0xAB);
|
guard_ptr.write_volatile(0xAB);
|
||||||
}
|
}
|
||||||
std::process::exit(0);
|
std::process::exit(0);
|
||||||
}
|
}
|
||||||
|
Ok("wide_guard_top") => {
|
||||||
|
// One byte below the usable region, 64 KiB guard: must fault.
|
||||||
|
let s = Stack::new(64 * 1024, 64 * 1024).unwrap();
|
||||||
|
unsafe { s.usable_base().sub(1).write_volatile(0xAB); }
|
||||||
|
std::process::exit(0);
|
||||||
|
}
|
||||||
|
Ok("wide_guard_bottom") => {
|
||||||
|
// The very bottom page of a 64 KiB guard: an unprobed C-style
|
||||||
|
// leap over a small guard lands here — must still fault.
|
||||||
|
let s = Stack::new(64 * 1024, 64 * 1024).unwrap();
|
||||||
|
unsafe { s.usable_base().sub(64 * 1024).write_volatile(0xAB); }
|
||||||
|
std::process::exit(0);
|
||||||
|
}
|
||||||
Ok("stack_overflow") => {
|
Ok("stack_overflow") => {
|
||||||
let s = Stack::new(64 * 1024).unwrap();
|
let s = Stack::new(64 * 1024, 4096).unwrap();
|
||||||
unsafe {
|
unsafe {
|
||||||
let mut ptr = s.top().sub(1);
|
let mut ptr = s.top().sub(1);
|
||||||
let stop = s.usable_base().sub(1);
|
let stop = s.usable_base().sub(1);
|
||||||
@@ -121,3 +134,58 @@ fn stack_overflow_causes_sigsegv() {
|
|||||||
assert_eq!(status.signal(), Some(11), "expected SIGSEGV, got: {:?}", status);
|
assert_eq!(status.signal(), Some(11), "expected SIGSEGV, got: {:?}", status);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// RFC 019 — explicit shape: rounding, guard accessor, wide-guard coverage.
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn sizes_round_up_to_page() {
|
||||||
|
let s = Stack::new(64 * 1024 + 1, 4096 + 1).unwrap();
|
||||||
|
assert_eq!(s.stack_size() % 4096, 0);
|
||||||
|
assert_eq!(s.guard_size() % 4096, 0);
|
||||||
|
assert!(s.stack_size() >= 64 * 1024 + 1);
|
||||||
|
assert!(s.guard_size() >= 4096 + 1);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn shape_reports_rounded_sizes() {
|
||||||
|
let s = Stack::new(64 * 1024, 64 * 1024).unwrap();
|
||||||
|
assert_eq!(s.shape(), (64 * 1024, 64 * 1024));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn usable_base_sits_above_guard() {
|
||||||
|
let s = Stack::new(64 * 1024, 64 * 1024).unwrap();
|
||||||
|
// The usable region must start exactly guard_size above the mapping
|
||||||
|
// base: a write at usable_base is legal, one byte below is not (the
|
||||||
|
// subprocess tests below prove the "not").
|
||||||
|
let sentinel: u64 = 0x1111_2222_3333_4444;
|
||||||
|
unsafe {
|
||||||
|
let ptr = s.usable_base() as *mut u64;
|
||||||
|
ptr.write_volatile(sentinel);
|
||||||
|
assert_eq!(ptr.read_volatile(), sentinel);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn wide_guard_faults_at_top() {
|
||||||
|
run_as_child_if_requested();
|
||||||
|
let status = spawn_subtest("wide_guard_top");
|
||||||
|
#[cfg(unix)]
|
||||||
|
{
|
||||||
|
use std::os::unix::process::ExitStatusExt;
|
||||||
|
assert_eq!(status.signal(), Some(11), "expected SIGSEGV, got: {:?}", status);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn wide_guard_faults_at_bottom() {
|
||||||
|
run_as_child_if_requested();
|
||||||
|
let status = spawn_subtest("wide_guard_bottom");
|
||||||
|
#[cfg(unix)]
|
||||||
|
{
|
||||||
|
use std::os::unix::process::ExitStatusExt;
|
||||||
|
assert_eq!(status.signal(), Some(11), "expected SIGSEGV, got: {:?}", status);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -0,0 +1,141 @@
|
|||||||
|
//! RFC 019 §7 — overflow diagnostics, observed from outside via subprocess
|
||||||
|
//! (mirrors tests/stack.rs's harness, plus stderr capture).
|
||||||
|
//!
|
||||||
|
//! Four cases:
|
||||||
|
//! - Rust recursion at defaults: probed frames walk into the guard →
|
||||||
|
//! tier-1 definitive message, death by SIGSEGV.
|
||||||
|
//! - FFI canary (96 KiB unprobed C local) at defaults: first touch lands
|
||||||
|
//! inside the 1 MiB guard → tier-1 message.
|
||||||
|
//! - FFI canary with the guard shrunk to 4 KiB: the frame steps over it
|
||||||
|
//! into unmapped VA below → tier-2 "stepped over" message. This is the
|
||||||
|
//! RFC's motivating incident (cargo-vendored gz build) reproduced.
|
||||||
|
//! - FFI canary with reserve raised to 256 KiB: fits, runs clean, exits 0 —
|
||||||
|
//! the §1 knob is the fix, proven by the same frame.
|
||||||
|
|
||||||
|
use std::env;
|
||||||
|
use std::process::Command;
|
||||||
|
|
||||||
|
unsafe extern "C" {
|
||||||
|
fn smarm_canary_burn();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Unbounded probed recursion; each frame dirties 4 KiB. black_box defeats
|
||||||
|
/// tail-call elision so the walk is real.
|
||||||
|
#[inline(never)]
|
||||||
|
#[allow(unconditional_recursion)]
|
||||||
|
fn recurse_forever(depth: u64) -> u64 {
|
||||||
|
let mut local = [0u8; 4096];
|
||||||
|
local[0] = depth as u8;
|
||||||
|
std::hint::black_box(&mut local);
|
||||||
|
recurse_forever(depth + 1).wrapping_add(local[0] as u64)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn run_as_child_if_requested() {
|
||||||
|
let mode = match env::var("SMARM_DIAG_SUBTEST") {
|
||||||
|
Ok(m) => m,
|
||||||
|
Err(_) => return,
|
||||||
|
};
|
||||||
|
use smarm::runtime::Config;
|
||||||
|
use smarm::{spawn_with, SpawnOpts};
|
||||||
|
let rt = smarm::runtime::init(Config::exact(1));
|
||||||
|
rt.run(move || {
|
||||||
|
let opts = match mode.as_str() {
|
||||||
|
"rust_overflow" | "ffi_tier1" => SpawnOpts::default(),
|
||||||
|
// Small guard: the canary's 96 KiB displacement clears it.
|
||||||
|
"ffi_tier2" => SpawnOpts { guard_size: Some(4096), ..SpawnOpts::default() },
|
||||||
|
// Enough reserve: the same frame simply fits.
|
||||||
|
"ffi_clean" => SpawnOpts { stack_reserve: Some(256 * 1024), ..SpawnOpts::default() },
|
||||||
|
other => panic!("unknown subtest {other}"),
|
||||||
|
};
|
||||||
|
let is_rust = mode == "rust_overflow";
|
||||||
|
spawn_with(opts, move || {
|
||||||
|
if is_rust {
|
||||||
|
std::hint::black_box(recurse_forever(0));
|
||||||
|
} else {
|
||||||
|
unsafe { smarm_canary_burn() };
|
||||||
|
}
|
||||||
|
})
|
||||||
|
.join()
|
||||||
|
.unwrap();
|
||||||
|
});
|
||||||
|
std::process::exit(0);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn spawn_subtest(name: &str) -> std::process::Output {
|
||||||
|
let exe = env::current_exe().unwrap();
|
||||||
|
Command::new(exe)
|
||||||
|
.env("SMARM_DIAG_SUBTEST", name)
|
||||||
|
.args(["--test-threads=1", "--quiet"])
|
||||||
|
.output()
|
||||||
|
.expect("failed to spawn subprocess")
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(unix)]
|
||||||
|
fn assert_died_sigsegv(out: &std::process::Output) {
|
||||||
|
use std::os::unix::process::ExitStatusExt;
|
||||||
|
assert_eq!(
|
||||||
|
out.status.signal(),
|
||||||
|
Some(11),
|
||||||
|
"expected death by SIGSEGV, got {:?}; stderr:\n{}",
|
||||||
|
out.status,
|
||||||
|
String::from_utf8_lossy(&out.stderr)
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn rust_overflow_dies_with_tier1_message() {
|
||||||
|
run_as_child_if_requested();
|
||||||
|
let out = spawn_subtest("rust_overflow");
|
||||||
|
assert_died_sigsegv(&out);
|
||||||
|
let err = String::from_utf8_lossy(&out.stderr);
|
||||||
|
assert!(
|
||||||
|
err.contains("overflowed its stack") && err.contains("in the guard region"),
|
||||||
|
"missing tier-1 diagnostic; stderr:\n{err}"
|
||||||
|
);
|
||||||
|
assert!(err.contains("reserve=65536"), "wrong reserve in message:\n{err}");
|
||||||
|
assert!(err.contains("guard=1048576"), "wrong guard in message:\n{err}");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn ffi_canary_at_defaults_dies_with_tier1_message() {
|
||||||
|
run_as_child_if_requested();
|
||||||
|
let out = spawn_subtest("ffi_tier1");
|
||||||
|
assert_died_sigsegv(&out);
|
||||||
|
let err = String::from_utf8_lossy(&out.stderr);
|
||||||
|
// 96 KiB displacement from a 64 KiB reserve lands ~32 KiB into the
|
||||||
|
// 1 MiB guard: definitively classified.
|
||||||
|
assert!(
|
||||||
|
err.contains("in the guard region"),
|
||||||
|
"wide guard should catch the unprobed frame in tier 1; stderr:\n{err}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn ffi_canary_over_small_guard_dies_with_tier2_message() {
|
||||||
|
run_as_child_if_requested();
|
||||||
|
let out = spawn_subtest("ffi_tier2");
|
||||||
|
assert_died_sigsegv(&out);
|
||||||
|
let err = String::from_utf8_lossy(&out.stderr);
|
||||||
|
assert!(
|
||||||
|
err.contains("stepped over it") && err.contains("below the guard"),
|
||||||
|
"expected tier-2 overshoot attribution; stderr:\n{err}"
|
||||||
|
);
|
||||||
|
assert!(err.contains("guard=4096"), "wrong guard in message:\n{err}");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn ffi_canary_with_enough_reserve_runs_clean() {
|
||||||
|
run_as_child_if_requested();
|
||||||
|
let out = spawn_subtest("ffi_clean");
|
||||||
|
assert!(
|
||||||
|
out.status.success(),
|
||||||
|
"canary should fit in 256 KiB reserve, got {:?}; stderr:\n{}",
|
||||||
|
out.status,
|
||||||
|
String::from_utf8_lossy(&out.stderr)
|
||||||
|
);
|
||||||
|
let err = String::from_utf8_lossy(&out.stderr);
|
||||||
|
assert!(
|
||||||
|
!err.contains("smarm: actor"),
|
||||||
|
"no diagnostic expected on the clean path; stderr:\n{err}"
|
||||||
|
);
|
||||||
|
}
|
||||||
@@ -0,0 +1,123 @@
|
|||||||
|
//! RFC 019 commit 5 — pool recycle zaps a dead stack down to its retained
|
||||||
|
//! entry end, observed from the outside.
|
||||||
|
//!
|
||||||
|
//! A default-shaped stack that spiked deep and then died must not carry its
|
||||||
|
//! spike into the pool as resident RSS: `recycle_stack` DONTNEEDs everything
|
||||||
|
//! below the top `RECYCLE_RETAIN` bytes before pushing. The zap is
|
||||||
|
//! synchronous on the death path, so the drop is immediate — but the death
|
||||||
|
//! path itself races the observer's `join` return, hence the brief poll.
|
||||||
|
//!
|
||||||
|
//! Residency is measured with `mincore`, not smaps: a neighboring rw anon
|
||||||
|
//! mapping can land flush against the stack top and the kernel merges the
|
||||||
|
//! VMAs (observed under the full test run), so per-mapping smaps fields
|
||||||
|
//! over-count. The PROT_NONE guard below can never merge, so the usable
|
||||||
|
//! base is exactly the anchor VMA's start, and `mincore` counts pages
|
||||||
|
//! within [usable_base, usable_base + reserve) regardless of merging.
|
||||||
|
|
||||||
|
use smarm::runtime::{Config, RECYCLE_RETAIN};
|
||||||
|
use smarm::{channel, spawn, yield_now};
|
||||||
|
|
||||||
|
const RESERVE: usize = 4 * 1024 * 1024;
|
||||||
|
|
||||||
|
/// Burn ~`frames` × 4 KiB of stack, dirtying every frame.
|
||||||
|
#[inline(never)]
|
||||||
|
fn burn_stack(frames: usize) -> u64 {
|
||||||
|
let mut local = [0u8; 4096];
|
||||||
|
local[0] = frames as u8;
|
||||||
|
let below = if frames == 0 { 0 } else { burn_stack(frames - 1) };
|
||||||
|
std::hint::black_box(&mut local);
|
||||||
|
below.wrapping_add(local[0] as u64)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Resident-page count over [lo, lo + len) via mincore (len page-aligned).
|
||||||
|
fn resident_pages(lo: usize, len: usize) -> usize {
|
||||||
|
let page = 4096;
|
||||||
|
let mut vec = vec![0u8; len / page];
|
||||||
|
let ret = unsafe {
|
||||||
|
libc::mincore(lo as *mut libc::c_void, len, vec.as_mut_ptr())
|
||||||
|
};
|
||||||
|
assert_eq!(ret, 0, "mincore failed: {}", std::io::Error::last_os_error());
|
||||||
|
vec.iter().filter(|&&b| b & 1 != 0).count()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The [start, end) of the VMA containing `addr`.
|
||||||
|
fn vma_containing(addr: usize) -> (usize, usize) {
|
||||||
|
let maps = std::fs::read_to_string("/proc/self/maps").unwrap();
|
||||||
|
for line in maps.lines() {
|
||||||
|
if let Some((range, _)) = line.split_once(' ') {
|
||||||
|
if let Some((a, b)) = range.split_once('-') {
|
||||||
|
if let (Ok(start), Ok(end)) =
|
||||||
|
(usize::from_str_radix(a, 16), usize::from_str_radix(b, 16))
|
||||||
|
{
|
||||||
|
if start <= addr && addr < end {
|
||||||
|
return (start, end);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
panic!("no VMA contains {addr:#x}");
|
||||||
|
}
|
||||||
|
|
||||||
|
fn vma_exists(addr: usize) -> bool {
|
||||||
|
let maps = std::fs::read_to_string("/proc/self/maps").unwrap();
|
||||||
|
for line in maps.lines() {
|
||||||
|
if let Some((range, _)) = line.split_once(' ') {
|
||||||
|
if let Some((a, b)) = range.split_once('-') {
|
||||||
|
if let (Ok(start), Ok(end)) =
|
||||||
|
(usize::from_str_radix(a, 16), usize::from_str_radix(b, 16))
|
||||||
|
{
|
||||||
|
if start <= addr && addr < end {
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
false
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn recycle_zaps_dead_stack_down_to_retain() {
|
||||||
|
// Default reserve raised so the pool holds big stacks (default-shaped ⇒
|
||||||
|
// pooled) and the zap has something to bite; single scheduler.
|
||||||
|
let rt = smarm::runtime::init(Config::exact(1).stack_reserve(RESERVE));
|
||||||
|
rt.run(|| {
|
||||||
|
let (tx, rx) = channel::<usize>();
|
||||||
|
|
||||||
|
let h = spawn(move || {
|
||||||
|
let probe = 0u8;
|
||||||
|
let anchor = &probe as *const u8 as usize;
|
||||||
|
// The guard below is PROT_NONE and can never merge with the
|
||||||
|
// usable region, so the anchor VMA's start IS the usable base.
|
||||||
|
let (vlo, _) = vma_containing(anchor);
|
||||||
|
// Dirty ~3 MiB of the 4 MiB reserve, then die.
|
||||||
|
std::hint::black_box(burn_stack(768));
|
||||||
|
tx.send(vlo).unwrap();
|
||||||
|
});
|
||||||
|
|
||||||
|
let usable_base = rx.recv().unwrap();
|
||||||
|
h.join().unwrap();
|
||||||
|
|
||||||
|
// The zap span is everything below the retained entry end. DONTNEED
|
||||||
|
// on private anon discards synchronously and unconditionally, so
|
||||||
|
// this must go to exactly zero resident pages; the poll only covers
|
||||||
|
// the death path racing join's return.
|
||||||
|
let zap_len = RESERVE - RECYCLE_RETAIN;
|
||||||
|
let mut resident = usize::MAX;
|
||||||
|
for _ in 0..10_000 {
|
||||||
|
resident = resident_pages(usable_base, zap_len);
|
||||||
|
if resident == 0 {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
yield_now();
|
||||||
|
}
|
||||||
|
assert_eq!(
|
||||||
|
resident, 0,
|
||||||
|
"recycled stack's zap span still resident: {resident} pages in \
|
||||||
|
[{usable_base:#x}, +{zap_len:#x})"
|
||||||
|
);
|
||||||
|
// Pooled, not munmapped: the mapping must still be there.
|
||||||
|
assert!(vma_exists(usable_base), "default-shaped stack was unmapped instead of pooled");
|
||||||
|
});
|
||||||
|
}
|
||||||
@@ -0,0 +1,152 @@
|
|||||||
|
//! RFC 019 commit 3 — park-path stack shrink, observed from the outside.
|
||||||
|
//!
|
||||||
|
//! The one integration-level claim of the shrink machinery: an actor that
|
||||||
|
//! spikes deep, returns shallow, and then parks past the cooldown gets its
|
||||||
|
//! dead span MADV_FREE'd — visible as `LazyFree` in `/proc/self/smaps`
|
||||||
|
//! within the stack's address range — while everything live survives.
|
||||||
|
//!
|
||||||
|
//! The high-water mark is *sampled* at context-save, so the spike yields
|
||||||
|
//! once at max depth to guarantee a sample there (in production, preemption
|
||||||
|
//! provides the quasi-random samples; a test must not rely on luck).
|
||||||
|
|
||||||
|
use smarm::runtime::{Config, SHRINK_COOLDOWN, SHRINK_THRESHOLD};
|
||||||
|
use smarm::{actor_info, channel, spawn, spawn_with, yield_now, ActorState, SpawnOpts};
|
||||||
|
|
||||||
|
/// Burn ~`frames` × 4 KiB of stack, yielding once at the bottom so the
|
||||||
|
/// context-save samples `sp` at max depth.
|
||||||
|
#[inline(never)]
|
||||||
|
fn burn_stack_yielding(frames: usize) -> u64 {
|
||||||
|
let mut local = [0u8; 4096];
|
||||||
|
local[0] = frames as u8;
|
||||||
|
let below = if frames == 0 {
|
||||||
|
yield_now();
|
||||||
|
0
|
||||||
|
} else {
|
||||||
|
burn_stack_yielding(frames - 1)
|
||||||
|
};
|
||||||
|
std::hint::black_box(&mut local);
|
||||||
|
below.wrapping_add(local[0] as u64)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Sum the `LazyFree:` kB of every smaps mapping intersecting [lo, hi).
|
||||||
|
fn lazy_free_bytes_in(lo: usize, hi: usize) -> usize {
|
||||||
|
let smaps = std::fs::read_to_string("/proc/self/smaps").unwrap();
|
||||||
|
let mut total_kb = 0usize;
|
||||||
|
let mut in_range = false;
|
||||||
|
for line in smaps.lines() {
|
||||||
|
if let Some((range, _)) = line.split_once(' ') {
|
||||||
|
if let Some((a, b)) = range.split_once('-') {
|
||||||
|
if let (Ok(start), Ok(end)) =
|
||||||
|
(usize::from_str_radix(a, 16), usize::from_str_radix(b, 16))
|
||||||
|
{
|
||||||
|
in_range = start < hi && end > lo;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if in_range {
|
||||||
|
if let Some(rest) = line.strip_prefix("LazyFree:") {
|
||||||
|
let kb: usize = rest.trim().trim_end_matches(" kB").trim().parse().unwrap();
|
||||||
|
total_kb += kb;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
total_kb * 1024
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn spike_then_parks_marks_lazyfree_and_keeps_live_data() {
|
||||||
|
// Single scheduler: the controller can gate on the worker being Parked.
|
||||||
|
let rt = smarm::runtime::init(Config::exact(1));
|
||||||
|
rt.run(|| {
|
||||||
|
let (park_tx, park_rx) = channel::<()>();
|
||||||
|
let (done_tx, done_rx) = channel::<(usize, u64)>();
|
||||||
|
|
||||||
|
let spike = 768 * 4096; // ~3 MiB, well past SHRINK_THRESHOLD
|
||||||
|
assert!(spike > SHRINK_THRESHOLD);
|
||||||
|
|
||||||
|
let worker = spawn_with(
|
||||||
|
SpawnOpts { stack_reserve: Some(8 * 1024 * 1024), ..SpawnOpts::default() },
|
||||||
|
move || {
|
||||||
|
// Live data that must survive the shrink, and an anchor
|
||||||
|
// address inside the stack for the smaps scan.
|
||||||
|
let live = [0xA5u8; 64];
|
||||||
|
let anchor = live.as_ptr() as usize;
|
||||||
|
|
||||||
|
// Spike: ~3 MiB deep, sampled at the bottom, unwound.
|
||||||
|
std::hint::black_box(burn_stack_yielding(768));
|
||||||
|
|
||||||
|
// Park past the cooldown. Each recv on the drained inbox is
|
||||||
|
// one park; the controller sends only when it sees us Parked.
|
||||||
|
for _ in 0..(SHRINK_COOLDOWN + 8) {
|
||||||
|
park_rx.recv().unwrap();
|
||||||
|
}
|
||||||
|
|
||||||
|
// Measure from inside: the stack spans ≤ 8 MiB below anchor.
|
||||||
|
let lazy = lazy_free_bytes_in(anchor - 8 * 1024 * 1024, anchor + 4096);
|
||||||
|
let checksum = live.iter().map(|&b| b as u64).sum();
|
||||||
|
done_tx.send((lazy, checksum)).unwrap();
|
||||||
|
},
|
||||||
|
);
|
||||||
|
|
||||||
|
let wpid = worker.pid();
|
||||||
|
for _ in 0..(SHRINK_COOLDOWN + 8) {
|
||||||
|
// Gate: send only once the worker is genuinely parked so every
|
||||||
|
// round is a real park-on-empty-mailbox.
|
||||||
|
loop {
|
||||||
|
match actor_info(wpid) {
|
||||||
|
Some(info) if info.state == ActorState::Parked => break,
|
||||||
|
Some(_) => yield_now(),
|
||||||
|
None => panic!("worker died early"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
park_tx.send(()).unwrap();
|
||||||
|
}
|
||||||
|
|
||||||
|
let (lazy, checksum) = done_rx.recv().unwrap();
|
||||||
|
// The spike was ~3 MiB; demand at least 2 MiB marked to leave slack
|
||||||
|
// for the redzone, rounding, and pages the unwind re-dirtied.
|
||||||
|
assert!(
|
||||||
|
lazy >= 2 * 1024 * 1024,
|
||||||
|
"expected ≥ 2 MiB LazyFree in the stack range, got {} bytes",
|
||||||
|
lazy
|
||||||
|
);
|
||||||
|
assert_eq!(checksum, 64 * 0xA5u64, "live stack data corrupted by shrink");
|
||||||
|
worker.join().unwrap();
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Steady-state actors must never pay the syscall: an actor that parks a lot
|
||||||
|
/// but never spikes past the threshold ends with zero LazyFree in its stack.
|
||||||
|
#[test]
|
||||||
|
fn shallow_actor_never_shrinks() {
|
||||||
|
let rt = smarm::runtime::init(Config::exact(1));
|
||||||
|
rt.run(|| {
|
||||||
|
let (park_tx, park_rx) = channel::<()>();
|
||||||
|
let (done_tx, done_rx) = channel::<usize>();
|
||||||
|
|
||||||
|
let worker = spawn(move || {
|
||||||
|
let probe = 0u8;
|
||||||
|
let anchor = &probe as *const u8 as usize;
|
||||||
|
for _ in 0..(SHRINK_COOLDOWN + 8) {
|
||||||
|
park_rx.recv().unwrap();
|
||||||
|
}
|
||||||
|
done_tx.send(lazy_free_bytes_in(anchor - 64 * 1024, anchor + 4096)).unwrap();
|
||||||
|
});
|
||||||
|
|
||||||
|
let wpid = worker.pid();
|
||||||
|
for _ in 0..(SHRINK_COOLDOWN + 8) {
|
||||||
|
loop {
|
||||||
|
match actor_info(wpid) {
|
||||||
|
Some(info) if info.state == ActorState::Parked => break,
|
||||||
|
Some(_) => yield_now(),
|
||||||
|
None => panic!("worker died early"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
park_tx.send(()).unwrap();
|
||||||
|
}
|
||||||
|
|
||||||
|
assert_eq!(done_rx.recv().unwrap(), 0, "steady-state actor was shrunk");
|
||||||
|
worker.join().unwrap();
|
||||||
|
});
|
||||||
|
}
|
||||||
Reference in New Issue
Block a user