feat(channel,runtime): off-runtime cross-thread wake for parked receivers and stop
Wakes issued from a non-scheduler OS thread were silent no-ops. Every off-runtime wake primitive (unpark, unpark_at, request_stop) reaches the runtime through the RUNTIME thread-local, which is unset on any foreign thread — so a send from a plain std::thread enqueued its message but never woke the parked receiver, and there was no way to drive a stop into a runtime from an application thread (e.g. an OS-signal handler). The former strands a parked recv forever; the latter is why a downstream server must poll a shutdown flag instead of parking on it. Generalize RFC 018's rule — a producer reaches the runtime through a Weak it holds — from the IO backend to channel senders and to a new handle: - A receiver captures a Weak<RuntimeInner> (provably live at that moment) alongside its (pid, epoch) when it parks. send() and the last-sender drop wake through scheduler::unpark_at_via, which takes the thread-local path when on a scheduler thread (preemption-gated, slot-eligible) and the captured Weak otherwise — the same cross-context wake the IO threads do. - Runtime::handle() returns a Send + Sync RuntimeHandle carrying that Weak; RuntimeHandle::request_stop drives a cooperative stop from any thread and is a no-op once the runtime is dropped. The in-runtime wake paths (recv/select timers) are unchanged; only the sites reachable from a foreign thread route through the Weak. RuntimeHandle exposes request_stop only: send-wake needs no user-facing handle, and off-runtime unpark is covered because request_stop drives unpark on the upgraded inner. tests/cross_thread_wake.rs: a foreign-thread send wakes a parked receiver; a foreign-thread request_stop wakes and stops a parked actor; a RuntimeHandle held across and beyond run never blocks all-done and degrades to a no-op.
This commit is contained in:
+34
-21
@@ -90,8 +90,9 @@
|
||||
|
||||
use crate::pid::Pid;
|
||||
use crate::raw_mutex::RawMutex;
|
||||
use crate::runtime::RuntimeInner;
|
||||
use std::collections::VecDeque;
|
||||
use std::sync::Arc;
|
||||
use std::sync::{Arc, Weak};
|
||||
|
||||
/// Create a new channel and return its `(Sender, Receiver)` halves.
|
||||
///
|
||||
@@ -114,12 +115,15 @@ pub fn channel<T>() -> (Sender<T>, Receiver<T>) {
|
||||
|
||||
struct Inner<T> {
|
||||
queue: VecDeque<T>,
|
||||
/// The parked receiver's `(pid, park-epoch)`, if one is currently
|
||||
/// The parked receiver's `(pid, park-epoch, runtime)`, if one is currently
|
||||
/// waiting. The epoch identifies exactly which wait this is, so a waker
|
||||
/// left over from a wait that already ended (a losing `select` arm, a
|
||||
/// `recv_timeout` whose timer fired after it was already satisfied) is
|
||||
/// inert and does nothing when it fires.
|
||||
parked_receiver: Option<(Pid, u32)>,
|
||||
/// inert and does nothing when it fires. The `Weak<RuntimeInner>` is the
|
||||
/// receiver's runtime, captured while it parked (so provably alive then);
|
||||
/// it lets a sender on a foreign OS thread wake the receiver without the
|
||||
/// `RUNTIME` thread-local, which is unset off a scheduler thread.
|
||||
parked_receiver: Option<(Pid, u32, Weak<RuntimeInner>)>,
|
||||
senders: usize,
|
||||
receiver_alive: bool,
|
||||
}
|
||||
@@ -206,8 +210,8 @@ impl<T> Drop for Sender<T> {
|
||||
None
|
||||
}
|
||||
};
|
||||
if let Some((pid, epoch)) = unpark {
|
||||
crate::scheduler::unpark_at(pid, epoch);
|
||||
if let Some((pid, epoch, rt)) = unpark {
|
||||
crate::scheduler::unpark_at_via(pid, epoch, &rt);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -254,13 +258,13 @@ impl<T> Sender<T> {
|
||||
g.queue.push_back(value);
|
||||
g.parked_receiver.take()
|
||||
};
|
||||
if let Some((pid, epoch)) = unpark {
|
||||
if let Some((pid, epoch, rt)) = unpark {
|
||||
crate::te!(crate::trace::Event::Send {
|
||||
sender: crate::actor::current_pid()
|
||||
.unwrap_or(crate::pid::Pid::new(u32::MAX, u32::MAX)),
|
||||
receiver: Some(pid)
|
||||
});
|
||||
crate::scheduler::unpark_at(pid, epoch);
|
||||
crate::scheduler::unpark_at_via(pid, epoch, &rt);
|
||||
} else {
|
||||
crate::te!(crate::trace::Event::Send {
|
||||
sender: crate::actor::current_pid()
|
||||
@@ -293,13 +297,17 @@ impl<T> Receiver<T> {
|
||||
None => panic!("smarm: recv() called outside an actor"),
|
||||
};
|
||||
debug_assert!(
|
||||
g.parked_receiver.is_none_or(|(p, _)| p == me),
|
||||
g.parked_receiver.as_ref().is_none_or(|(p, _, _)| *p == me),
|
||||
"channel has more than one receiver"
|
||||
);
|
||||
// begin_wait is lock-free, so it's legal under the Channel lock;
|
||||
// registering in the same critical section makes the epoch
|
||||
// atomic with the senders' view of the registration.
|
||||
g.parked_receiver = Some((me, crate::scheduler::begin_wait()));
|
||||
g.parked_receiver = Some((
|
||||
me,
|
||||
crate::scheduler::begin_wait(),
|
||||
crate::scheduler::runtime_weak(),
|
||||
));
|
||||
crate::te!(crate::trace::Event::RecvPark(me));
|
||||
}
|
||||
// Release the lock before parking: the unparker will need it.
|
||||
@@ -347,11 +355,11 @@ impl<T> Receiver<T> {
|
||||
return Err(RecvTimeoutError::Disconnected);
|
||||
}
|
||||
debug_assert!(
|
||||
g.parked_receiver.is_none_or(|(p, _)| p == me),
|
||||
g.parked_receiver.as_ref().is_none_or(|(p, _, _)| *p == me),
|
||||
"channel has more than one receiver"
|
||||
);
|
||||
epoch = crate::scheduler::begin_wait();
|
||||
g.parked_receiver = Some((me, epoch));
|
||||
g.parked_receiver = Some((me, epoch, crate::scheduler::runtime_weak()));
|
||||
crate::te!(crate::trace::Event::RecvPark(me));
|
||||
}
|
||||
|
||||
@@ -423,10 +431,14 @@ impl<T> Receiver<T> {
|
||||
None => panic!("smarm: recv_match() called outside an actor"),
|
||||
};
|
||||
debug_assert!(
|
||||
g.parked_receiver.is_none_or(|(p, _)| p == me),
|
||||
g.parked_receiver.as_ref().is_none_or(|(p, _, _)| *p == me),
|
||||
"channel has more than one receiver"
|
||||
);
|
||||
g.parked_receiver = Some((me, crate::scheduler::begin_wait()));
|
||||
g.parked_receiver = Some((
|
||||
me,
|
||||
crate::scheduler::begin_wait(),
|
||||
crate::scheduler::runtime_weak(),
|
||||
));
|
||||
crate::te!(crate::trace::Event::RecvPark(me));
|
||||
}
|
||||
// Release the lock before parking: the unparker will need it.
|
||||
@@ -497,11 +509,12 @@ impl<T: Send + 'static> crate::timer::TimerTarget for RawMutex<Inner<T>> {
|
||||
// keeps the registration bookkeeping exact.)
|
||||
let unpark = {
|
||||
let mut g = self.lock();
|
||||
if g.parked_receiver == Some((pid, epoch)) {
|
||||
g.parked_receiver = None;
|
||||
true
|
||||
} else {
|
||||
false
|
||||
match g.parked_receiver {
|
||||
Some((p, e, _)) if p == pid && e == epoch => {
|
||||
g.parked_receiver = None;
|
||||
true
|
||||
}
|
||||
_ => false,
|
||||
}
|
||||
};
|
||||
// Unpark outside the channel lock: it may take the run-queue lock;
|
||||
@@ -562,10 +575,10 @@ impl<T> Selectable for Receiver<T> {
|
||||
return Ok(false);
|
||||
}
|
||||
debug_assert!(
|
||||
g.parked_receiver.is_none_or(|(p, _)| p == pid),
|
||||
g.parked_receiver.as_ref().is_none_or(|(p, _, _)| *p == pid),
|
||||
"channel has more than one receiver"
|
||||
);
|
||||
g.parked_receiver = Some((pid, epoch));
|
||||
g.parked_receiver = Some((pid, epoch, crate::scheduler::runtime_weak()));
|
||||
Ok(true)
|
||||
}
|
||||
|
||||
|
||||
+1
-1
@@ -85,7 +85,7 @@ pub use registry::{
|
||||
install, lookup_as, register, resolve_name, send, send_dyn, send_to, unregister, whereis,
|
||||
NameResolution, RegisterError, SendError,
|
||||
};
|
||||
pub use runtime::{init, Config, Runtime};
|
||||
pub use runtime::{init, Config, Runtime, RuntimeHandle};
|
||||
pub use scheduler::{
|
||||
block_on_io, cancel_timer, request_stop, run, self_pid, send_after, send_after_named,
|
||||
send_after_named_wall, send_after_wall, sleep, sleep_wall, spawn, spawn_addr, spawn_addr_with,
|
||||
|
||||
+54
-1
@@ -127,7 +127,7 @@ use crate::supervisor::Signal;
|
||||
use crate::timer::Timers;
|
||||
|
||||
use std::sync::atomic::{AtomicBool, AtomicPtr, AtomicU32, AtomicU64, AtomicUsize, Ordering};
|
||||
use std::sync::{Arc, Mutex};
|
||||
use std::sync::{Arc, Mutex, Weak};
|
||||
use std::thread;
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -1457,6 +1457,59 @@ impl Runtime {
|
||||
inner: self.inner.clone(),
|
||||
}
|
||||
}
|
||||
|
||||
/// A `Send + Sync` handle to this runtime, usable from any thread —
|
||||
/// including threads that are not smarm schedulers (an OS-signal handler
|
||||
/// thread, an external event source). Grab it *before* [`run`](Self::run)
|
||||
/// and hand it to e.g. a signal thread; that thread can then
|
||||
/// [`request_stop`](RuntimeHandle::request_stop) the runtime's top
|
||||
/// supervisor to drive an ordered shutdown from outside the runtime.
|
||||
///
|
||||
/// The in-runtime primitives ([`scheduler::request_stop`](crate::request_stop)
|
||||
/// and friends) reach the runtime through a thread-local that is unset on
|
||||
/// any non-scheduler thread, so they are silent no-ops off-runtime; this
|
||||
/// handle carries its own reference and closes that gap.
|
||||
pub fn handle(&self) -> RuntimeHandle {
|
||||
RuntimeHandle {
|
||||
inner: Arc::downgrade(&self.inner),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// RuntimeHandle — off-runtime wake/stop
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// A `Send + Sync` handle to a [`Runtime`], obtained from
|
||||
/// [`Runtime::handle`]. Lets a thread that is *not* a smarm scheduler thread
|
||||
/// drive a cooperative stop into the runtime — the off-runtime counterpart to
|
||||
/// [`scheduler::request_stop`](crate::request_stop).
|
||||
///
|
||||
/// Holds a [`Weak`] to the runtime, for the same reason the IO backend does
|
||||
/// (RFC 018): a lingering handle can never keep the runtime's slot table alive
|
||||
/// and can never block [`Runtime::run`] from finishing. Once the `Runtime` is
|
||||
/// dropped every method is a harmless no-op — the same end state as calling
|
||||
/// `request_stop` on an actor that has already exited.
|
||||
#[derive(Clone)]
|
||||
pub struct RuntimeHandle {
|
||||
inner: Weak<RuntimeInner>,
|
||||
}
|
||||
|
||||
impl RuntimeHandle {
|
||||
/// Ask `pid` to stop cooperatively, from any thread. The off-runtime
|
||||
/// equivalent of [`scheduler::request_stop`](crate::request_stop): it sets
|
||||
/// the target's stop flag and wakes it, so a parked actor unwinds at its
|
||||
/// next checkpoint exactly as it would for an in-runtime stop. A no-op if
|
||||
/// the runtime has been dropped, or if the actor has already exited.
|
||||
pub fn request_stop<A>(&self, pid: Pid<A>) {
|
||||
let pid = pid.erase();
|
||||
// Upgrade the Weak per call, like the IO backend does (io.rs): a live
|
||||
// runtime yields the inner and we drive the same stop the in-runtime
|
||||
// path would; a dropped runtime makes this a no-op.
|
||||
if let Some(inner) = self.inner.upgrade() {
|
||||
crate::scheduler::request_stop_inner(&inner, pid);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
+24
-1
@@ -71,7 +71,7 @@ use crate::pid::{Name, Pid};
|
||||
use crate::runtime::{self, RuntimeInner, YieldIntent, RUNTIME};
|
||||
use crate::supervisor::Signal;
|
||||
use std::sync::atomic::Ordering;
|
||||
use std::sync::Arc;
|
||||
use std::sync::{Arc, Weak};
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// with_runtime / try_with_runtime
|
||||
@@ -544,6 +544,29 @@ pub(crate) fn unpark_at(pid: Pid, epoch: u32) {
|
||||
let _ = try_with_runtime(|inner| inner.unpark_at(pid, epoch));
|
||||
}
|
||||
|
||||
// The current actor's runtime as a `Weak`, for a waker that must reach the
|
||||
// runtime from a foreign thread later. A channel captures this when its
|
||||
// receiver parks, so a cross-thread `send` can wake without the `RUNTIME`
|
||||
// thread-local (unset off a scheduler thread). Panics outside `Runtime::run()`,
|
||||
// the same contract as `begin_wait`.
|
||||
pub(crate) fn runtime_weak() -> Weak<RuntimeInner> {
|
||||
with_runtime(Arc::downgrade)
|
||||
}
|
||||
|
||||
// Epoch-matched wake of `pid` from a waker that may or may not be on a
|
||||
// scheduler thread. On a scheduler thread we take the thread-local path
|
||||
// (preemption-gated, slot-eligible); off one that path is a silent no-op, so
|
||||
// we reach the runtime through `rt` — the `Weak` the waker captured while it
|
||||
// was in-runtime. Mirrors the IO backend's cross-context wake (io.rs, RFC 018).
|
||||
pub(crate) fn unpark_at_via(pid: Pid, epoch: u32, rt: &Weak<RuntimeInner>) {
|
||||
if try_with_runtime(|inner| inner.unpark_at(pid, epoch)).is_some() {
|
||||
return;
|
||||
}
|
||||
if let Some(inner) = rt.upgrade() {
|
||||
inner.unpark_at(pid, epoch);
|
||||
}
|
||||
}
|
||||
|
||||
// Open a new wait for the current actor and return its wait identity
|
||||
// ("epoch"). Call once per wait, before registering with any waker. Lock-free,
|
||||
// so it's legal to call while already holding another internal lock.
|
||||
|
||||
Reference in New Issue
Block a user