375 lines
16 KiB
Rust
375 lines
16 KiB
Rust
//! Sleep + wait-with-timeout timers.
|
|
//!
|
|
//! A min-heap of `(deadline, seq, reason)` entries lives on `SchedulerState`.
|
|
//! When an actor sleeps or starts a bounded wait (e.g. `mutex.lock()` with a
|
|
//! timeout), the runtime inserts an entry, marks the actor parked, and yields.
|
|
//! On every scheduler loop iteration the runtime pops all entries whose
|
|
//! deadline has passed and dispatches each according to its `Reason`:
|
|
//!
|
|
//! - `Sleep`: unpark the actor.
|
|
//! - `WaitTimeout`: call `on_timeout` on the registered target. The target
|
|
//! (e.g. a `Mutex`) decides whether the actor was actually still waiting
|
|
//! (timer fires first → unpark with error) or had already been granted
|
|
//! what it was waiting for (lock granted first → no-op).
|
|
//!
|
|
//! `BinaryHeap` is a max-heap; entries are wrapped in `Reverse` to get
|
|
//! min-heap behaviour.
|
|
//!
|
|
//! Cancellation is selective. A `Sleep` / `WaitTimeout` entry is left in the
|
|
//! heap on a non-timer wakeup (lock granted before timeout): it is popped
|
|
//! eventually and no-ops because a stale unpark fails its epoch CAS — cheap
|
|
//! (~32 bytes per stale entry plus a few cycles on pop), bounded by one entry
|
|
//! per parked actor. A `Send` entry is different: running its thunk delivers a
|
|
//! real message, so a stale one is *not* inert. `send_after` therefore carries
|
|
//! true cancellation via the `armed` set keyed on the entry's `seq`; `pop_due`
|
|
//! fires a `Send` only while it is still armed, and `cancel` removes the arm.
|
|
//!
|
|
//! Stale pids (slot reused since the timer was inserted) are filtered on
|
|
//! pop by the scheduler — same convention as the run queue.
|
|
|
|
use crate::pid::Pid;
|
|
use std::cmp::Reverse;
|
|
use std::collections::BinaryHeap;
|
|
use std::sync::Arc;
|
|
use std::time::{Duration, Instant};
|
|
|
|
/// What to do when a timer entry's deadline arrives.
|
|
///
|
|
/// Held inside `Entry`, dispatched by the scheduler in `pop_due`.
|
|
pub enum Reason {
|
|
/// `sleep(d)`. Wake `pid` via the epoch-matched unpark: if anything
|
|
/// else (necessarily a terminal wake) already consumed the wait, the
|
|
/// entry is stale and no-ops at the CAS.
|
|
Sleep { epoch: u32 },
|
|
/// A bounded wait (`Mutex::lock_timeout`, `Receiver::recv_timeout`,
|
|
/// `select_timeout`). On expiry the scheduler calls
|
|
/// `target.on_timeout(pid, epoch)`. The target then decides whether
|
|
/// `pid` was actually still waiting (registration still present under
|
|
/// its lock), and if so takes the registration and unparks via
|
|
/// `unpark_at`. The epoch is the slot-word park-epoch — the runtime-wide
|
|
/// wait identity — so a stale entry is doubly inert: the registration
|
|
/// check misses, and even a racing unpark fails the word's epoch CAS.
|
|
WaitTimeout {
|
|
target: Arc<dyn TimerTarget>,
|
|
epoch: u32,
|
|
},
|
|
/// `send_after`: deliver a message to an address at the deadline,
|
|
/// cancellable. The destination (a `Pid<A>` / `Name<M>`) and the message
|
|
/// are captured inside `fire`, which resolves the address through the
|
|
/// registry and sends *when run* — so a target that died or, for a name,
|
|
/// was restarted is observed at fire time, not arm time. A failed resolve
|
|
/// or send is dropped (Erlang `erlang:send_after` semantics).
|
|
///
|
|
/// Unlike `Sleep` / `WaitTimeout`, a stale `Send` is **not** inert — running
|
|
/// the thunk delivers a real message — so these are the only timers that
|
|
/// carry true cancellation (the `armed` set on [`Timers`], keyed by the
|
|
/// entry's `seq`). `pop_due` fires the thunk only for an entry still armed.
|
|
Send { fire: Box<dyn FnOnce() + Send> },
|
|
}
|
|
|
|
/// Opaque handle to an armed `send_after` timer, returned by
|
|
/// [`Timers::insert_send`] and consumed by [`Timers::cancel`]. The inner value
|
|
/// is the entry's insertion `seq`; callers must treat it as opaque so the
|
|
/// backing structure can change (e.g. a future hierarchical timing wheel) with
|
|
/// no API churn.
|
|
#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
|
|
pub struct TimerId(u64);
|
|
|
|
impl TimerId {
|
|
/// Wrap a raw value. Crate-internal: the gen_server timer layer mints its
|
|
/// own loop-local `TimerId`s (the public ids it hands out, decoupled from
|
|
/// the per-re-arm substrate `seq`) and maps them to live substrate ids.
|
|
/// These local ids are only ever resolved through that layer's registry —
|
|
/// never passed back to [`Timers::cancel`] — so the two id roles do not mix.
|
|
pub(crate) fn from_raw(v: u64) -> Self {
|
|
TimerId(v)
|
|
}
|
|
}
|
|
|
|
/// Callback the scheduler invokes when a `WaitTimeout` entry pops.
|
|
///
|
|
/// Implementors: do not touch `SchedulerState` other than via the public
|
|
/// `unpark` / channel APIs. The scheduler is mid-iteration when this fires.
|
|
pub trait TimerTarget: Send + Sync {
|
|
fn on_timeout(&self, pid: Pid, epoch: u32);
|
|
}
|
|
|
|
pub struct Entry {
|
|
pub deadline: Instant,
|
|
/// Insertion order, used purely as a tiebreaker so `Entry: Ord` works
|
|
/// without having to compare the `Reason` payload (which contains an
|
|
/// `Rc<dyn TimerTarget>` and isn't `Ord`).
|
|
seq: u64,
|
|
pub pid: Pid,
|
|
pub reason: Reason,
|
|
/// RFC 007 virtual time: the global delay ledger reading when this entry
|
|
/// was (re-)queued. `pop_due` shifts the effective deadline by any delay
|
|
/// injected since, so timers dilate together with the causally-delayed
|
|
/// workload instead of firing early in virtual terms.
|
|
#[cfg(feature = "smarm-causal")]
|
|
delay_stamp: u64,
|
|
/// RFC 007: a wall-anchored entry opts out of the virtual-time shift —
|
|
/// its deadline is honoured in wall time regardless of injected delay.
|
|
/// Used by the causal controller's own measurement/cooldown sleeps so
|
|
/// experiment windows keep a fixed wall length; ordinary workload timers
|
|
/// stay virtual (`false`).
|
|
#[cfg(feature = "smarm-causal")]
|
|
wall: bool,
|
|
}
|
|
|
|
impl PartialEq for Entry {
|
|
fn eq(&self, other: &Self) -> bool {
|
|
self.deadline == other.deadline && self.seq == other.seq
|
|
}
|
|
}
|
|
impl Eq for Entry {}
|
|
|
|
impl Ord for Entry {
|
|
fn cmp(&self, other: &Self) -> std::cmp::Ordering {
|
|
// Earlier deadline first; ties broken by insertion order so the
|
|
// ordering is total. `Reason` and `Pid` deliberately don't
|
|
// participate.
|
|
self.deadline
|
|
.cmp(&other.deadline)
|
|
.then_with(|| self.seq.cmp(&other.seq))
|
|
}
|
|
}
|
|
|
|
impl PartialOrd for Entry {
|
|
fn partial_cmp(&self, other: &Self) -> Option<std::cmp::Ordering> {
|
|
Some(self.cmp(other))
|
|
}
|
|
}
|
|
|
|
#[derive(Default)]
|
|
pub struct Timers {
|
|
/// RFC 018: the scheduler coordination layer. Attached once at
|
|
/// `RuntimeInner::new`; every insert notes its deadline (min-maintained
|
|
/// snapshot for the busy-path due-check + the timekeeper re-arm wake)
|
|
/// and every pop/clear re-anchors the snapshot to the heap minimum.
|
|
/// All calls happen under the timers mutex — the serialization the
|
|
/// coordinator's timer protocol mandates. `None` only in unit tests
|
|
/// that construct a bare `Timers`.
|
|
coord: Option<std::sync::Arc<crate::park::Coordinator>>,
|
|
/// Reverse-wrapped so the smallest deadline is at the top.
|
|
heap: BinaryHeap<Reverse<Entry>>,
|
|
/// Monotonic counter for the tiebreaker `seq` field (and the `TimerId` of a
|
|
/// `Send` timer — the two are the same value).
|
|
next_seq: u64,
|
|
/// Presence set of *live* `Send` timers, keyed by `seq`. Populated on
|
|
/// `insert_send`, removed on fire (in `pop_due`) and on `cancel`. A `Send`
|
|
/// entry fires only while present, so a `cancel` that lands before the
|
|
/// entry pops prevents delivery; a `cancel` after it has fired finds
|
|
/// nothing (the race signal). Bounded by armed-but-not-yet-resolved timers
|
|
/// and self-collecting — no sweep. `Sleep` / `WaitTimeout` never touch it.
|
|
armed: std::collections::HashSet<u64>,
|
|
}
|
|
|
|
impl Timers {
|
|
pub fn new() -> Self {
|
|
Self {
|
|
coord: None,
|
|
heap: BinaryHeap::new(),
|
|
next_seq: 0,
|
|
armed: std::collections::HashSet::new(),
|
|
}
|
|
}
|
|
|
|
/// Attach the scheduler coordination layer (RFC 018). Called once, at
|
|
/// runtime construction, before any scheduler thread exists.
|
|
pub(crate) fn attach_coordinator(&mut self, c: std::sync::Arc<crate::park::Coordinator>) {
|
|
self.coord = Some(c);
|
|
}
|
|
|
|
/// Insert a `Sleep` timer. Convenience for the common case.
|
|
pub fn insert_sleep(&mut self, deadline: Instant, pid: Pid, epoch: u32) {
|
|
self.insert(deadline, pid, Reason::Sleep { epoch });
|
|
}
|
|
|
|
/// Insert a *wall-anchored* `Sleep` timer: fires at `deadline` in wall
|
|
/// time even while causal profiling (feature `smarm-causal`) is injecting
|
|
/// virtual delay — it never chases the delay ledger. Without the feature
|
|
/// this is identical to [`insert_sleep`](Self::insert_sleep).
|
|
///
|
|
/// Intended for measurement machinery (the causal controller's window and
|
|
/// cooldown sleeps, TSC calibration) whose durations *define* wall time
|
|
/// rather than participate in the workload. Workload code should use the
|
|
/// ordinary virtual-anchored timers.
|
|
pub fn insert_sleep_wall(&mut self, deadline: Instant, pid: Pid, epoch: u32) {
|
|
self.push(deadline, pid, Reason::Sleep { epoch }, true);
|
|
}
|
|
|
|
/// Arm a cancellable `send_after` timer: run `fire` at `deadline` unless
|
|
/// [`cancel`](Self::cancel)led first. `pid` is informational only (the
|
|
/// destination, or who armed it — useful for introspection); it is *not*
|
|
/// used to wake anyone, the delivery lives entirely inside `fire`. Returns
|
|
/// a [`TimerId`] for cancellation.
|
|
pub fn insert_send(
|
|
&mut self,
|
|
deadline: Instant,
|
|
pid: Pid,
|
|
fire: Box<dyn FnOnce() + Send>,
|
|
) -> TimerId {
|
|
self.armed.insert(self.next_seq);
|
|
TimerId(self.push(deadline, pid, Reason::Send { fire }, false))
|
|
}
|
|
|
|
/// Arm a *wall-anchored* cancellable `send_after` timer (RFC 007): the
|
|
/// same contract as [`insert_send`](Self::insert_send), but the entry
|
|
/// opts out of the virtual-time shift and fires at its raw deadline
|
|
/// regardless of injected delay — the `Send`-reason sibling of
|
|
/// [`insert_sleep_wall`](Self::insert_sleep_wall). Without the
|
|
/// `smarm-causal` feature this is identical to `insert_send`.
|
|
pub fn insert_send_wall(
|
|
&mut self,
|
|
deadline: Instant,
|
|
pid: Pid,
|
|
fire: Box<dyn FnOnce() + Send>,
|
|
) -> TimerId {
|
|
self.armed.insert(self.next_seq);
|
|
TimerId(self.push(deadline, pid, Reason::Send { fire }, true))
|
|
}
|
|
|
|
/// Cancel an armed `send_after` timer. Returns `true` if the timer was
|
|
/// still armed (delivery is now prevented), `false` if it had already
|
|
/// fired or been cancelled. The heap entry, if still pending, is left to be
|
|
/// discarded when its deadline passes — `pop_due` drops any `Send` entry
|
|
/// whose `seq` is no longer armed.
|
|
pub fn cancel(&mut self, id: TimerId) -> bool {
|
|
self.armed.remove(&id.0)
|
|
}
|
|
|
|
/// Insert an arbitrary (virtual-anchored) timer entry.
|
|
pub fn insert(&mut self, deadline: Instant, pid: Pid, reason: Reason) {
|
|
self.push(deadline, pid, reason, false);
|
|
}
|
|
|
|
/// Common insertion path. `wall` selects the RFC 007 anchor (see
|
|
/// [`insert_sleep_wall`](Self::insert_sleep_wall)); it is accepted — and
|
|
/// ignored — without the `smarm-causal` feature so callers don't fork.
|
|
/// Returns the entry's `seq`.
|
|
fn push(&mut self, deadline: Instant, pid: Pid, reason: Reason, wall: bool) -> u64 {
|
|
#[cfg(not(feature = "smarm-causal"))]
|
|
let _ = wall;
|
|
let seq = self.next_seq;
|
|
self.next_seq = self.next_seq.wrapping_add(1);
|
|
self.heap.push(Reverse(Entry {
|
|
deadline,
|
|
seq,
|
|
pid,
|
|
reason,
|
|
#[cfg(feature = "smarm-causal")]
|
|
delay_stamp: crate::causal::global_delay_cycles(),
|
|
#[cfg(feature = "smarm-causal")]
|
|
wall,
|
|
}));
|
|
// RFC 018: publish the (possibly new-minimum) deadline to the
|
|
// busy-path snapshot and wake the timekeeper if it is parked
|
|
// toward a later one. We hold the timers mutex — the mandated
|
|
// serialization for both.
|
|
if let Some(c) = &self.coord {
|
|
c.note_deadline(deadline);
|
|
}
|
|
seq
|
|
}
|
|
|
|
pub fn is_empty(&self) -> bool {
|
|
self.heap.is_empty()
|
|
}
|
|
|
|
/// Drop all pending entries. Called by the scheduler when it has decided
|
|
/// no actor is live: any remaining timer is orphaned and exists only to be
|
|
/// discarded so it can't keep the runtime alive.
|
|
pub fn clear(&mut self) {
|
|
self.heap.clear();
|
|
self.armed.clear();
|
|
if let Some(c) = &self.coord {
|
|
c.refresh_deadline(None);
|
|
}
|
|
}
|
|
|
|
/// Soonest pending deadline, or `None` if the heap is empty.
|
|
pub fn peek_deadline(&self) -> Option<Instant> {
|
|
self.heap.peek().map(|r| r.0.deadline)
|
|
}
|
|
|
|
/// Pop every entry whose deadline is ≤ `now`, in deadline order.
|
|
/// The scheduler dispatches each entry by inspecting `entry.reason`.
|
|
///
|
|
/// A due `Send` entry is returned only if it is still armed; a cancelled
|
|
/// one is silently dropped here (its `seq` was already removed from
|
|
/// `armed` by [`cancel`](Self::cancel)). Returning it removes it from
|
|
/// `armed`, so a later `cancel` of a fired timer reports `false`.
|
|
///
|
|
/// RFC 007 virtual time (feature `smarm-causal`): before an entry fires,
|
|
/// any global delay injected since it was (re-)queued is added to its
|
|
/// deadline; an entry whose *effective* deadline hasn't passed is pushed
|
|
/// back with the shifted deadline and a fresh stamp, so it keeps chasing
|
|
/// delay injected while it waits. Consequences, both benign:
|
|
/// [`peek_deadline`](Self::peek_deadline) may under-report (raw deadline
|
|
/// earlier than effective), costing at most one spurious scheduler wake
|
|
/// per injected chunk; and a shift never converts wall time — with zero
|
|
/// debt the path is byte-identical to the featureless one. Wall-anchored
|
|
/// entries ([`insert_sleep_wall`](Self::insert_sleep_wall)) are exempt
|
|
/// from the shift and always fire at their raw deadline.
|
|
pub fn pop_due(&mut self, now: Instant) -> Vec<Entry> {
|
|
let mut out = Vec::new();
|
|
#[cfg(feature = "smarm-causal")]
|
|
let global = crate::causal::global_delay_cycles();
|
|
while let Some(r) = self.heap.peek() {
|
|
if r.0.deadline > now {
|
|
break;
|
|
}
|
|
#[allow(unused_mut)]
|
|
let mut entry = match self.heap.pop() {
|
|
Some(e) => e.0,
|
|
None => panic!("smarm: timer heap pop after peek returned None (core corrupt)"),
|
|
};
|
|
if matches!(entry.reason, Reason::Send { .. }) && !self.armed.contains(&entry.seq) {
|
|
// Cancelled before it came due: discard, do not deliver.
|
|
// (Checked before any shift so a cancelled entry is never
|
|
// re-queued just to be discarded later.)
|
|
continue;
|
|
}
|
|
#[cfg(feature = "smarm-causal")]
|
|
if !entry.wall {
|
|
let debt = global.saturating_sub(entry.delay_stamp);
|
|
if debt > 0 {
|
|
let shifted = entry
|
|
.deadline
|
|
.checked_add(crate::causal::cycles_to_duration(debt))
|
|
.unwrap_or(entry.deadline);
|
|
if shifted > now {
|
|
// Not due in virtual time: re-queue at the shifted
|
|
// deadline, stamped, keeping `seq` (and thus `Send`
|
|
// cancellation identity) intact.
|
|
entry.deadline = shifted;
|
|
entry.delay_stamp = global;
|
|
self.heap.push(Reverse(entry));
|
|
continue;
|
|
}
|
|
}
|
|
}
|
|
if matches!(entry.reason, Reason::Send { .. }) {
|
|
self.armed.remove(&entry.seq);
|
|
}
|
|
out.push(entry);
|
|
}
|
|
// RFC 018: re-anchor the busy-path snapshot to the new heap minimum
|
|
// (still under the timers mutex). A causal-shift re-queue above went
|
|
// through `heap.push` directly, so this peek is the one place the
|
|
// snapshot is guaranteed to catch up.
|
|
if let Some(c) = &self.coord {
|
|
c.refresh_deadline(self.peek_deadline());
|
|
}
|
|
out
|
|
}
|
|
}
|
|
|
|
/// Wall-clock duration helper exposed for `sleep` and `lock_timeout`.
|
|
pub fn deadline_from_now(duration: Duration) -> Instant {
|
|
Instant::now()
|
|
.checked_add(duration)
|
|
.unwrap_or_else(Instant::now)
|
|
}
|