diff --git a/Cargo.toml b/Cargo.toml index 2ba0dd8..029989e 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -33,6 +33,11 @@ budget-accounting = [] # and unflagged; only the optional gen_server transport sits behind this, so a # release build pays nothing for an observer it never starts. observer = [] +# RFC 010 c1: clustering. Off by default — the default build stays libc-only, +# byte-for-byte (gate checked per phase). serde is the payload contract, +# postcard the payload codec; both minimal (no default features). Everything +# cluster-shaped lives behind this flag. +cluster = ["dep:serde", "dep:postcard"] # Run-queue selection: exactly one, compile-time (see src/run_queue.rs). # Non-default variants need --no-default-features (features are additive). rq-mutex = [] @@ -44,12 +49,18 @@ cc = "1" [dependencies] libc = "0.2" +# RFC 010 §2 — only compiled under `--features cluster`. +serde = { version = "1", default-features = false, optional = true } +# `alloc` (not `std`): the seam serializes to Vec; postcard stays no_std-aligned. +postcard = { version = "1", default-features = false, features = ["alloc"], optional = true } [target.'cfg(loom)'.dependencies] loom = "0.7" [dev-dependencies] libc = "0.2" +# derive + std for cluster envelope tests only; the lib itself never needs them +serde = { version = "1", features = ["derive"] } tokio = { version = "1", features = ["rt", "rt-multi-thread", "macros", "sync", "time"] } [profile.dev] diff --git a/build.rs b/build.rs index f00ab7f..1e200ed 100644 --- a/build.rs +++ b/build.rs @@ -8,4 +8,29 @@ fn main() { .flag_if_supported("-fno-stack-clash-protection") .compile("smarm_canary"); println!("cargo:rerun-if-changed=canary/canary.c"); + + // RFC 010 c6d — build_hash inputs. The compile-time facts a peer must + // share for a mesh link: the exact toolchain and the declared (enabled) + // feature set. Emitted as a plain string; the hashing (FNV-1a folded + // with PROTO_VERSION) happens in src/cluster.rs where the protocol + // version actually lives — parsing it out of a source file here would + // be a second, fragile copy. Always emitted, even for non-cluster + // builds: one env var costs the default build nothing. + let rustc = std::env::var("RUSTC").unwrap_or_else(|_| "rustc".to_string()); + let version = std::process::Command::new(&rustc) + .arg("-V") + .output() + .ok() + .map(|o| String::from_utf8_lossy(&o.stdout).trim().to_string()) + .filter(|v| !v.is_empty()) + .unwrap_or_else(|| "rustc-unknown".to_string()); + let mut feats: Vec = std::env::vars() + .filter_map(|(k, _)| k.strip_prefix("CARGO_FEATURE_").map(str::to_string)) + .collect(); + feats.sort(); + println!( + "cargo:rustc-env=SMARM_BUILD_HASH_INPUTS={version};features={}", + feats.join(",") + ); + println!("cargo:rerun-if-env-changed=RUSTC"); } diff --git a/src/channel.rs b/src/channel.rs index 54e8f6a..24c1fc9 100644 --- a/src/channel.rs +++ b/src/channel.rs @@ -263,6 +263,16 @@ impl Sender { self.inner.lock().queue.len() } + /// Whether the [`Receiver`] is still alive (a send would be accepted). + pub(crate) fn receiver_alive(&self) -> bool { + self.inner.lock().receiver_alive + } + + /// Whether `other` is a sender of this very channel (a clone). + pub(crate) fn same_channel(&self, other: &Sender) -> bool { + Arc::ptr_eq(&self.inner, &other.inner) + } + /// Push `value` onto the channel. Succeeds unconditionally as long as /// the [`Receiver`] is still alive: the queue has no capacity limit, so /// this never blocks and never fails except when the channel is closed, diff --git a/src/cluster.rs b/src/cluster.rs new file mode 100644 index 0000000..635c611 --- /dev/null +++ b/src/cluster.rs @@ -0,0 +1,282 @@ +//! RFC 010 — clustering (smarm⇄smarm, explicit remote boundary). +//! +//! c1: feature flag + optional deps. c2: the owned envelope. c3: the +//! transport trait (control connection), framed codec, and the TCP + +//! loopback impls. c5: the handshake state machine. c6: the connection +//! [`manager`] (registry) and per-peer connection actors ([`conn`]), started +//! as an explicit supervision subtree, plus the handshake on the +//! accept/connect path ([`connect`]). Everything above them lands in later +//! chunks. + +pub mod conn; +pub mod connect; +pub mod connector; +pub mod discovery; +pub mod envelope; +pub mod expose; +pub mod handshake; +pub mod manager; +pub mod membership; +pub mod pg; +pub mod remote; +pub mod transport; + +use std::io; +use std::time::{Duration, SystemTime, UNIX_EPOCH}; + +use crate::gen_server::{self, GenServerBuilder}; +use crate::monitor::monitor; +use crate::pg::Incarnation; +use crate::scheduler::{sleep, spawn, JoinHandle}; +use crate::supervisor::{ChildSpec, OneForOne, Restart}; + +use envelope::NodeMeta; +use handshake::Local; +use transport::tcp::TcpTransport; +use transport::Transport; + +pub use conn::{spawn_established, ConnHandle}; +pub use connect::{dial, spawn_acceptor, AcceptorHandle}; +pub use connector::{spawn_connector, ConnectorHandle}; +pub use discovery::{Discovery, StaticSeeds, Strategy}; +pub use envelope::RemoteDownReason; +pub use expose::{expose, expose_type, type_hash, DeliverError}; +pub use manager::{Manager, MANAGER}; +pub use membership::{subscribe, view, MembershipEvents, NodeEvent, NodeInfo}; +pub use pg::{dispatch_any, members_all, pick_any, DispatchAnyError, GroupMember, PgMsg, PG_NAME}; +pub use remote::{ + demonitor_remote, monitor_remote, send_to_remote, NotConnected, RemoteDown, RemoteMonitor, + RemoteName, RemotePid, RemoteSendError, ToRemoteError, +}; + +/// c6d — the derived build hash for [`handshake::LocalNode::build_hash`]: +/// two builds may mesh only when this matches, and it is a pure function of +/// the compile-time inputs that define wire compatibility today — the exact +/// toolchain (`rustc -V`), the declared feature set, and +/// [`envelope::PROTO_VERSION`]. FNV-1a 64 over the build-script string, then +/// the proto version folded byte-wise, so a proto bump moves the hash even +/// on an identical toolchain. The domain is deliberately lean and +/// tightenable later without a wire change — it is just a `u64`. +pub const BUILD_HASH: u64 = fold_u32( + fnv1a64(env!("SMARM_BUILD_HASH_INPUTS").as_bytes()), + envelope::PROTO_VERSION, +); + +/// FNV-1a 64 (const so [`BUILD_HASH`] is a compile-time fact). +const fn fnv1a64(bytes: &[u8]) -> u64 { + let mut h: u64 = 0xcbf2_9ce4_8422_2325; + let mut i = 0; + while i < bytes.len() { + h ^= bytes[i] as u64; + h = h.wrapping_mul(0x0000_0100_0000_01b3); + i += 1; + } + h +} + +/// Continue an FNV-1a state over a `u32`'s little-endian bytes. +const fn fold_u32(mut h: u64, v: u32) -> u64 { + let b = v.to_le_bytes(); + let mut i = 0; + while i < b.len() { + h ^= b[i] as u64; + h = h.wrapping_mul(0x0000_0100_0000_01b3); + i += 1; + } + h +} + +/// The control-plane timing knobs, all with today's fixed values as +/// defaults ([`Timing::default`]). One struct threaded explicitly to the +/// acceptor, the dial path, every connection actor and the connector — no +/// ambient state, so a test can run a fast mesh without touching globals. +/// Every node in a mesh should agree on `heartbeat_interval` < +/// `liveness_timeout`; nothing enforces it. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct Timing { + /// Idle-connection heartbeat pace. Default [`conn::HEARTBEAT_INTERVAL`]. + pub heartbeat_interval: Duration, + /// Inbound silence that tears a connection down. Default + /// [`conn::LIVENESS_TIMEOUT`]. + pub liveness_timeout: Duration, + /// Per-frame handshake deadline on the accept/dial path. Default + /// [`connect::HANDSHAKE_TIMEOUT`]. + pub handshake_timeout: Duration, + /// Connector redial delay after the first failure. Default + /// [`connector::INITIAL_BACKOFF`]. + pub initial_backoff: Duration, + /// Connector redial delay cap. Default [`connector::MAX_BACKOFF`]. + pub max_backoff: Duration, +} + +impl Default for Timing { + fn default() -> Self { + Timing { + heartbeat_interval: conn::HEARTBEAT_INTERVAL, + liveness_timeout: conn::LIVENESS_TIMEOUT, + handshake_timeout: connect::HANDSHAKE_TIMEOUT, + initial_backoff: connector::INITIAL_BACKOFF, + max_backoff: connector::MAX_BACKOFF, + } + } +} + +/// How to run this node: its identity and how it finds peers. +pub struct Config { + /// This node's claimed name — the mesh-wide identity peers dial by and + /// the tie-break input. Must be unique across the mesh. + pub node_name: String, + /// Metadata offered in this node's `Hello`. + pub meta: NodeMeta, + /// The control-connection listen address (e.g. `"127.0.0.1:0"`; the + /// concrete bound address is [`Cluster::local_addr`]). + pub listen_addr: String, + /// The peer-discovery strategy — [`StaticSeeds`] until richer ones land. + pub strategy: Box, + /// Heartbeat / liveness / handshake / backoff knobs; [`Timing::default`] + /// is the shipping configuration. + pub timing: Timing, +} + +/// A running cluster node: the supervised [`Manager`], the acceptor over the +/// bound listener, and the connector driving its [`Strategy`]. Roles will +/// eventually mount this; until the role mechanism lands it is started by +/// hand (RFC 010 §7). +/// +/// Dropping the handle stops the acceptor and connector loops (no new +/// connections in either direction) but detaches the manager subtree, which +/// — with every established connection — keeps running for the life of the +/// runtime, the same split as [`AcceptorHandle`] alone. +pub struct Cluster { + _sup: JoinHandle, + acceptor: AcceptorHandle, + connector: ConnectorHandle, + local: Local, +} + +impl Cluster { + /// The concrete bound listen address, dialable as-is. + pub fn local_addr(&self) -> &str { + self.acceptor.local_addr() + } + + /// This node's handshake identity (name, incarnation, build hash, meta). + pub fn local(&self) -> &Local { + &self.local + } + + /// Stop accepting and dialing. Established connections stay up (they + /// belong to the manager); tear those down via the manager. + pub fn shutdown(&self) { + self.acceptor.shutdown(); + self.connector.shutdown(); + } +} + +/// Start a cluster node: the supervised manager (blocking until it is +/// registered and ready to answer), the acceptor bound per +/// [`Config::listen_addr`], and the connector running [`Config::strategy`]. +/// The node's identity is completed here: `incarnation` is +/// [`self_incarnation`] and `build_hash` is [`BUILD_HASH`] — c7 is its first +/// consumer. Errs only if the listener cannot bind. +/// +/// The manager is a supervised child (restarted on crash); per-peer +/// connection actors are dynamic and monitored by the manager rather than +/// statically supervised — a lost connection is re-established by the +/// connector's dial loop, never resurrected onto a stale socket. +pub fn start(config: Config) -> io::Result { + let sup = spawn(|| { + OneForOne::new() + .child(ChildSpec::new(Restart::Permanent, manager_child)) + .run() + }); + while gen_server::whereis_server(MANAGER).is_none() { + sleep(Duration::from_millis(1)); + } + let local = Local { + node_name: config.node_name, + incarnation: self_incarnation(), + build_hash: BUILD_HASH, + meta: config.meta, + }; + // The wire identity serialized pids are stamped with (c10). + remote::set_local_identity(&local.node_name, local.incarnation); + // The pg actor (Phase 5): subscribes membership, owns the "pg" name. + pg::attach_cluster(); + let listener = TcpTransport.listen(&config.listen_addr)?; + let acceptor = spawn_acceptor(listener, local.clone(), config.timing); + let connector = spawn_connector( + Box::new(TcpTransport), + local.clone(), + config.strategy, + config.timing, + ); + Ok(Cluster { + _sup: sup, + acceptor, + connector, + local, + }) +} + +/// This process's incarnation epoch: milliseconds since the Unix epoch, +/// truncated to `u32`. Not a clock — its one job is separating a node from +/// its own restart (two starts of the same name land on the same value only +/// if they happen within the same millisecond modulo ~49.7 days). Seconds +/// would be too coarse: a crash-and-restart inside one second is routine +/// under supervision. +pub fn self_incarnation() -> Incarnation { + let ms = SystemTime::now() + .duration_since(UNIX_EPOCH) + .map(|d| d.as_millis()) + .unwrap_or(0); + Incarnation::new(ms as u32) +} + +/// The supervised manager child body. It *is* the child actor: it starts the +/// named manager, then parks on the manager's own termination so this actor's +/// lifetime tracks the manager's — the supervisor's restart accounting keys off +/// this actor exiting. +fn manager_child() { + let m = match GenServerBuilder::new(Manager::new()).named(MANAGER).start() { + Ok(m) => m, + // Name still held by a not-yet-reaped prior instance: return and let + // the supervisor retry under its restart policy. + Err(_) => return, + }; + let _ = monitor(m.pid()).rx.recv(); +} + +#[cfg(test)] +mod tests { + use super::*; + + /// The hash core against the published FNV-1a 64 test vectors — the + /// contract is "this is FNV-1a", not "whatever the fn does". + #[test] + fn fnv1a64_known_vectors() { + assert_eq!(fnv1a64(b""), 0xcbf2_9ce4_8422_2325); + assert_eq!(fnv1a64(b"a"), 0xaf63_dc4c_8601_ec8c); + assert_eq!(fnv1a64(b"foobar"), 0x85944171f73967e8); + } + + /// Folding the proto version continues the same FNV state: identical + /// inputs with a different version must land on a different hash. + #[test] + fn proto_version_moves_the_hash() { + let base = fnv1a64(b"same-toolchain;features=CLUSTER"); + assert_ne!(fold_u32(base, 1), fold_u32(base, 2)); + // And it equals hashing the bytes in one pass — the fold is a + // continuation, not a second construction. + let mut all = b"same-toolchain;features=CLUSTER".to_vec(); + all.extend_from_slice(&1u32.to_le_bytes()); + assert_eq!(fold_u32(base, 1), fnv1a64(&all)); + } + + /// The derived constant exists, is compile-time, and is not degenerate. + #[test] + fn build_hash_is_nonzero() { + const H: u64 = BUILD_HASH; + assert_ne!(H, 0); + } +} diff --git a/src/cluster/conn.rs b/src/cluster/conn.rs new file mode 100644 index 0000000..76a33ba --- /dev/null +++ b/src/cluster/conn.rs @@ -0,0 +1,630 @@ +//! RFC 010 c6 — the per-peer connection actor. +//! +//! One actor per established control connection. It owns the whole +//! [`FramedConn`] and, in a single [`select`](crate::select), waits on two +//! things at once: its command inbox and the connection becoming readable (the +//! [`FdArm`](crate::scheduler::FdArm) the transport hands back). That is why it +//! is a plain select-loop actor rather than a `gen_server` or `gen_statem` — +//! neither of those can fold fd-readiness into its wait, and folding it in is +//! the whole job. The single owner sends and receives on the one `FramedConn`, +//! so no read/write split is needed. +//! +//! The handshake completes *before* this actor exists (on the accept/connect +//! path — c6b) and produces the [`Peer`]; the *path* then registers the +//! connection with the [`manager`](crate::cluster::manager), which takes +//! ownership of its [`ConnHandle`] and monitors the actor, so any exit +//! deregisters the connection. The actor itself holds no authority over its +//! own lifetime: it runs until the manager drops its handle (deregistration, +//! `Disconnect`, or manager shutdown), the connection ends, or liveness +//! expires. Heartbeat send and fixed-timeout liveness are the timeout arm of +//! the same `select` (c6c): [`HEARTBEAT_INTERVAL`] paces outbound +//! [`Frame::Heartbeat`](crate::cluster::envelope::Frame::Heartbeat)s, and a +//! [`LIVENESS_TIMEOUT`] window — reset by any inbound frame — tears the +//! connection down when it empties. +//! +//! c9 adds the third arm — the connection's dedicated **outbound inbox** +//! (`Sender` bound in the manager-maintained outbound table, D13), +//! drained onto the wire in the same loop — and inbound *interpretation*: +//! `SendNamed` goes to the one resolution seam, +//! [`remote::deliver_named`](crate::cluster::remote::deliver_named). +//! `Send` goes to the pid seam (c10). The outbound +//! sender is a separate channel from `cmd_tx` on purpose: closing it is not +//! a stop signal — lifetime authority stays with the [`ConnHandle`] (D9). +//! +//! c12 adds the monitor plane, and it lives *here* on purpose. Two tables, +//! both owned by this actor and dying with the connection: +//! +//! - **outstanding** — monitors *this* node holds on actors at the peer: +//! `monitor_id → (target, Sender)`. Fed by +//! [`MonCmd`](crate::cluster::remote::MonCmd) from `monitor_remote`; the +//! actor records the id and *then* emits the `Monitor` frame, so a `Down` +//! frame can never race an entry that isn't there yet. An inbound `Down` +//! removes the entry and delivers. +//! - **watched** — monitors the *peer* holds on actors here: `monitor_id → +//! local Monitor`. An inbound `Monitor` is admitted only for a pid that +//! was exposed or crossed the wire (`is_watchable`, D12): a corpse answers +//! with its recorded terminal reason (RFC §6), an unwatchable or unknown +//! pid with `NoProc` — indistinguishable from dead, so nothing leaks. A +//! live watchable pid gets a local monitor whose `rx` is one more arm of +//! the select; its `Down` goes back as a frame. +//! +//! Because both tables are actor state, connection loss (c13) needs no +//! second bookkeeping owner: this actor's exit is the one place that knows +//! every monitor the link was carrying. `Monitors::teardown` runs on every +//! exit path and answers each outstanding monitor with `Disconnected` — +//! the roadmap's "partition vs. death" contrast: an actor that dies sends +//! its true reason over the link, a link that dies says only that. + +use std::collections::HashMap; + +use std::time::{Duration, Instant}; + +use crate::channel::{channel, try_select_timeout, Receiver, Selectable, Sender}; +use crate::cluster::envelope::{Frame, RemoteDownReason}; +use crate::cluster::handshake::Peer; +use crate::cluster::manager::{Call, Registered, Reply, MANAGER}; +use crate::cluster::remote::{ + deliver_named, deliver_to_pid, InboundVerdict, MonCmd, RemoteDown, RemotePid, +}; +use crate::cluster::transport::FramedConn; +use crate::cluster::Timing; +use crate::gen_server; +use crate::monitor::{ + demonitor, is_watchable, monitor, terminal_reason, DownReason, Monitor, MonitorId, +}; +use crate::pid::{Erased, Pid}; +use crate::scheduler::spawn; + +/// Commands to a running connection actor. +enum Cmd { + Shutdown, +} + +/// The manager's authority over one connection actor: while this handle +/// lives the connection lives, and dropping it stops the actor and closes +/// the socket. Only the [`manager`](crate::cluster::manager) holds one — +/// callers of [`spawn_established`] get a [`Pid`] and no lifetime authority, +/// so a connection can never outlive, or die with, whichever actor happened +/// to establish it. +pub struct ConnHandle { + cmd_tx: Sender, + /// The connection's dedicated outbound inboxes — frames and monitor + /// commands. The manager moves them into the outbound table on + /// `Register` (see [`take_outbound`](ConnHandle::take_outbound)); a + /// `Duplicate` verdict drops them with the handle. + out_tx: Option<(Sender, Sender)>, +} + +impl std::fmt::Debug for ConnHandle { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.write_str("ConnHandle") + } +} + +impl ConnHandle { + /// Ask the connection to close and exit. Idempotent, and a no-op if the + /// actor has already gone. Dropping the handle does the same thing; this + /// exists for the manager's explicit `Disconnect` path. + pub fn shutdown(&self) { + let _ = self.cmd_tx.send(Cmd::Shutdown); + } + + /// Manager-only: take the outbound senders to bind into the outbound + /// table. Once, at registration. + pub(crate) fn take_outbound(&mut self) -> Option<(Sender, Sender)> { + self.out_tx.take() + } +} + +/// The name was already claimed by a live connection, so this one was +/// refused; its actor has been stopped and its socket closed. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct RegisterRefused; + +/// Spawn a connection actor for an **already-established** connection (the +/// handshake completed on the path and produced `peer`) and register it with +/// the manager, synchronously, before returning. The manager takes the +/// actor's [`ConnHandle`]; the caller gets only the [`Pid`], because +/// connection lifetime belongs to the table and not to the establishing +/// actor. A refusal has already stopped the actor and closed the socket. +pub fn spawn_established( + framed: FramedConn, + peer: Peer, + timing: Timing, +) -> Result { + let (cmd_tx, cmd_rx) = channel(); + let (out_tx, out_rx) = channel(); + let (mon_tx, mon_rx) = channel(); + let reg_peer = peer.clone(); + let pid = spawn(move || run(framed, peer, timing, cmd_rx, out_rx, mon_rx)).pid(); + match gen_server::call( + MANAGER, + Call::Register { + peer: reg_peer, + pid, + handle: ConnHandle { + cmd_tx, + out_tx: Some((out_tx, mon_tx)), + }, + }, + ) { + Ok(Reply::Registered(Registered::Ok)) => Ok(pid), + // Duplicate name, or the manager is unreachable. Either way the + // handle went with the call and is dropped there (or never arrived + // and dropped with it), which stops the actor and closes the socket. + _ => Err(RegisterRefused), + } +} + +/// Default for [`Timing::heartbeat_interval`]: how often this end emits +/// [`Frame::Heartbeat`] on an idle connection. The first one goes out +/// immediately at spawn, so the peer's liveness window starts fed. +pub const HEARTBEAT_INTERVAL: Duration = Duration::from_secs(1); + +/// Default for [`Timing::liveness_timeout`]: how long the connection may go +/// without a single inbound frame before it is declared dead and torn down. Any inbound frame resets the window — +/// heartbeats keep an idle connection alive, and real traffic (c8+) counts +/// for free. Fixed by design (RFC v2 §5): this is the control connection, a +/// heartbeat can never queue behind bulk traffic, so a fixed timeout is an +/// honest detector. +pub const LIVENESS_TIMEOUT: Duration = Duration::from_secs(4); + +fn run( + mut framed: FramedConn, + _peer: Peer, + timing: Timing, + cmd_rx: Receiver, + out_rx: Receiver, + mon_rx: Receiver, +) { + let mut mons = Monitors::default(); + match framed.readable_arm() { + Some(arm) => run_live( + &mut framed, + arm, + timing, + &cmd_rx, + &out_rx, + &mon_rx, + &mut mons, + ), + None => run_inert(&cmd_rx), + } + framed.close(); + mons.teardown(&mon_rx); +} + +/// The monitor plane's two tables (module docs). Owned by the actor. +#[derive(Default)] +struct Monitors { + /// Monitors this node holds on peer actors: id → (target, delivery). + outstanding: HashMap, Sender)>, + /// Monitors the peer holds on local actors: id → the local monitor. + watched: HashMap, +} + +impl Monitors { + /// The connection is gone, whatever the exit path (liveness expiry, + /// EOF, wire failure, commanded stop): release the peer's local + /// monitors, and answer every one of ours with `Disconnected` — nothing + /// more can be known about those actors. Commands still sitting in the + /// inbox are folded in first (a `Monitor` handed to us but never + /// processed gets its notice too; a `Demonitor` still cancels), so the + /// only registration that can miss this is one that lands after the + /// drain and before the inbox drops — the reader side backstops that + /// (`RemoteMonitor`). Entries leave the table as they are answered, and + /// this runs once per actor, so no monitor sees two notices. + fn teardown(&mut self, mon_rx: &Receiver) { + for (_, m) in self.watched.drain() { + let _ = demonitor(&m); + } + while let Ok(Some(cmd)) = mon_rx.try_recv() { + match cmd { + MonCmd::Monitor { id, target, tx } => { + self.outstanding.insert(id, (target, tx)); + } + MonCmd::Demonitor { id } => { + self.outstanding.remove(&id); + } + } + } + for (_, (pid, tx)) in self.outstanding.drain() { + let _ = tx.send(RemoteDown { + pid, + reason: RemoteDownReason::Disconnected, + }); + } + } + + /// Admit a peer's `Monitor` for local `(index, generation)`. Returns + /// the reason to answer with at once, or `None` if a live monitor was + /// installed. Corpse → recorded terminal reason (RFC §6, and only + /// watchable deaths are recorded); live watchable → monitor; anything + /// else → `NoProc`. The check-then-monitor race (dies in between) is + /// closed on the read side: a `NoProc` from a monitor we installed on a + /// live pid is upgraded through `terminal_reason` in `sweep_watched`. + fn admit(&mut self, id: MonitorId, index: u32, generation: u32) -> Option { + let pid = Pid::new(index, generation); + if let Some(reason) = terminal_reason(pid) { + return Some(reason); + } + if !is_watchable(pid) { + return Some(DownReason::NoProc); + } + let m = monitor(pid); + self.watched.insert(id, m); + None + } + + fn cancel(&mut self, id: MonitorId) { + if let Some(m) = self.watched.remove(&id) { + let _ = demonitor(&m); + } + } + + /// Collect every local `Down` that has arrived for a peer-held monitor. + fn sweep_watched(&mut self) -> Vec<(MonitorId, DownReason)> { + let mut fired = Vec::new(); + for (id, m) in self.watched.iter() { + if let Ok(Some(down)) = m.rx.try_recv() { + let reason = match down.reason { + DownReason::NoProc => terminal_reason(m.target).unwrap_or(DownReason::NoProc), + r => r, + }; + fired.push((*id, reason)); + } + } + for (id, _) in &fired { + self.watched.remove(id); + } + fired + } + + /// The peer reports a monitored actor down: deliver locally. + fn down(&mut self, id: MonitorId, reason: RemoteDownReason) { + if let Some((pid, tx)) = self.outstanding.remove(&id) { + let _ = tx.send(RemoteDown { pid, reason }); + } + } +} + +/// The steady-state loop over an fd-backed connection: one +/// `select_timeout` folds the command inbox, the outbound inbox, socket +/// readability, and the nearer of the two deadlines (`hb_send`, +/// `liveness`) into a single wait. +fn run_live( + framed: &mut FramedConn, + arm: crate::scheduler::FdArm, + timing: Timing, + cmd_rx: &Receiver, + out_rx: &Receiver, + mon_rx: &Receiver, + mons: &mut Monitors, +) { + let mut next_hb = Instant::now(); + let mut live_until = Instant::now() + timing.liveness_timeout; + // The outbound senders live in the manager's table and are dropped on + // unbind; after that these arms would wake forever, so they drop out of + // the select (not a stop signal — see the module docs). + let mut out_open = true; + let mut mon_open = true; + // Which wait each select arm stands for. Built in lockstep with the + // `Selectable` vector each iteration, so a wake is decoded by name and + // never by position. + enum Arm { + Cmd, + Fd, + Out, + Mon, + /// A peer-held local monitor (any of them: firing sweeps them all). + Watched, + } + fn push<'s>( + arms: &mut Vec<&'s dyn Selectable>, + what: &mut Vec, + s: &'s dyn Selectable, + a: Arm, + ) { + arms.push(s); + what.push(a); + } + loop { + let now = Instant::now(); + if now >= live_until { + break; // liveness expired: the peer is dead to us + } + if now >= next_hb { + if framed.send(&Frame::Heartbeat).is_err() { + break; + } + next_hb = now + timing.heartbeat_interval; + } + let wait = next_hb.min(live_until).saturating_duration_since(now); + let mut arms: Vec<&dyn Selectable> = Vec::new(); + let mut what: Vec = Vec::new(); + push(&mut arms, &mut what, cmd_rx, Arm::Cmd); + push(&mut arms, &mut what, &arm, Arm::Fd); + if out_open { + push(&mut arms, &mut what, out_rx, Arm::Out); + } + if mon_open { + push(&mut arms, &mut what, mon_rx, Arm::Mon); + } + for m in mons.watched.values() { + push(&mut arms, &mut what, &m.rx, Arm::Watched); + } + match try_select_timeout(&arms, wait).map(|i| i.map(|i| &what[i])) { + Ok(Some(Arm::Cmd)) => { + if should_stop(cmd_rx) { + break; + } + } + Ok(Some(Arm::Fd)) => match pump_readable(framed, mons) { + Pump::Ended => break, + Pump::Frames(n) => { + if n > 0 { + live_until = Instant::now() + timing.liveness_timeout; + } + } + }, + Ok(Some(Arm::Out)) => match pump_outbound(framed, out_rx) { + Outbound::Drained => {} + Outbound::Closed => out_open = false, + Outbound::WireFailed => break, + }, + Ok(Some(Arm::Mon)) => match pump_moncmds(framed, mon_rx, mons) { + Outbound::Drained => {} + Outbound::Closed => mon_open = false, + Outbound::WireFailed => break, + }, + Ok(Some(Arm::Watched)) => { + for (id, reason) in mons.sweep_watched() { + let frame = Frame::Down { + monitor_id: id.0, + reason: reason.into(), + }; + if framed.send(&frame).is_err() { + return; + } + } + } + // A deadline passed; the top of the loop acts on whichever. + Ok(None) => {} + // The fd arm failed to register — the connection is gone. + Err(_) => break, + } + } +} + +/// Drain the monitor-command inbox: record, then emit (module docs). +fn pump_moncmds( + framed: &mut FramedConn, + mon_rx: &Receiver, + mons: &mut Monitors, +) -> Outbound { + loop { + match mon_rx.try_recv() { + Ok(Some(MonCmd::Monitor { id, target, tx })) => { + let frame = Frame::Monitor { + monitor_id: id.0, + index: target.index(), + generation: target.generation(), + }; + mons.outstanding.insert(id, (target, tx)); + if framed.send(&frame).is_err() { + return Outbound::WireFailed; + } + } + Ok(Some(MonCmd::Demonitor { id })) => { + if mons.outstanding.remove(&id).is_some() + && framed.send(&Frame::Demonitor { monitor_id: id.0 }).is_err() + { + return Outbound::WireFailed; + } + } + Ok(None) => return Outbound::Drained, + Err(_) => return Outbound::Closed, + } + } +} + +/// What one outbound-side wake (frames or monitor commands) yielded. +enum Outbound { + /// Everything queued went onto the wire; the inbox is open and empty. + Drained, + /// The manager unbound this connection's sender; nothing more will come. + Closed, + /// The socket refused a write: the connection is gone. + WireFailed, +} + +/// Drain every queued outbound frame onto the wire. +fn pump_outbound(framed: &mut FramedConn, out_rx: &Receiver) -> Outbound { + loop { + match out_rx.try_recv() { + Ok(Some(frame)) => { + if framed.send(&frame).is_err() { + return Outbound::WireFailed; + } + } + Ok(None) => return Outbound::Drained, + Err(_) => return Outbound::Closed, + } + } +} + +/// No fd to select on (loopback): only a command can end the wait, and +/// neither heartbeats nor liveness run — a transport that can't report +/// readiness can't be timed either (same caveat as +/// [`FramedConn::recv_deadline`]). Loopback is a test transport; every real +/// connection is fd-backed. +fn run_inert(cmd_rx: &Receiver) { + loop { + let arms: [&dyn Selectable; 1] = [cmd_rx]; + let _ = crate::channel::select(&arms); + if should_stop(cmd_rx) { + break; + } + } +} + +/// Drain the command arm. Returns `true` when the actor should exit — a +/// shutdown was requested, or the last handle was dropped. +fn should_stop(cmd_rx: &Receiver) -> bool { + match cmd_rx.try_recv() { + Ok(Some(Cmd::Shutdown)) => true, + Ok(None) => false, // spurious wake + Err(_) => true, // all senders dropped + } +} + +/// What one readable wake yielded. +enum Pump { + /// The connection has ended: EOF (clean or mid-frame) or an + /// unrecoverable stream error. + Ended, + /// Still up; this many complete frames were consumed (possibly zero, if + /// the wake delivered only part of a frame). Any nonzero count resets + /// the liveness window. + Frames(usize), +} + +/// Surface an inbound verdict: one `smarm-trace` event, nothing else — it +/// is local knowledge (RFC §3). A no-op without the feature. +fn note_verdict(verdict: InboundVerdict) { + #[cfg(feature = "smarm-trace")] + crate::te!(crate::trace::Event::ClusterInbound(verdict.label())); + #[cfg(not(feature = "smarm-trace"))] + drop(verdict); +} + +/// Consume one readable wake: exactly one socket read (which cannot block +/// after a level-triggered readable indication), then drain every complete +/// frame the buffer now holds. A blocking `recv` here would park the actor +/// past its heartbeat and liveness deadlines whenever a frame arrives split. +/// Every consumed frame counts for liveness; `SendNamed` goes to the one +/// name-resolution seam and `Send` to the pid seam. Verdicts are local +/// knowledge only — nothing goes back on the wire (RFC §3) — and surface +/// as one `smarm-trace` `ClusterInbound` event each (zero cost off). +/// `Monitor`/`Demonitor`/`Down` go to the [`Monitors`] tables; a `Monitor` +/// that can be answered at once is answered inline. +fn pump_readable(framed: &mut FramedConn, mons: &mut Monitors) -> Pump { + let eof = match framed.read_once() { + Ok(n) => n == 0, + Err(_) => return Pump::Ended, + }; + let mut got = 0; + loop { + match framed.next_buffered() { + Ok(Some(frame)) => { + got += 1; + match frame { + Frame::SendNamed { + name, + type_hash, + payload, + } => { + note_verdict(deliver_named(&name, type_hash, &payload)); + } + Frame::Send { + index, + generation, + type_hash, + payload, + } => { + note_verdict(deliver_to_pid(index, generation, type_hash, &payload)); + } + Frame::Monitor { + monitor_id, + index, + generation, + } => { + let id = MonitorId(monitor_id); + if let Some(reason) = mons.admit(id, index, generation) { + let frame = Frame::Down { + monitor_id, + reason: reason.into(), + }; + if framed.send(&frame).is_err() { + return Pump::Ended; + } + } + } + Frame::Demonitor { monitor_id } => mons.cancel(MonitorId(monitor_id)), + Frame::Down { monitor_id, reason } => mons.down(MonitorId(monitor_id), reason), + // Heartbeat: liveness only. Handshake frames after + // establishment: ignored. + _ => {} + } + } + Ok(None) => break, + Err(_) => return Pump::Ended, // corrupt stream + } + } + if eof { + Pump::Ended + } else { + Pump::Frames(got) + } +} + +#[cfg(test)] +mod tests { + //! `Monitors::teardown` in isolation: the actor-side half of c13, pinned + //! separately because from the outside it is indistinguishable from the + //! read-side backstop in `RemoteMonitor` (both yield `Disconnected`). + use super::*; + use crate::pg::Incarnation; + + fn pid(index: u32) -> RemotePid { + RemotePid::from_parts("peer", Incarnation::new(1), index, 1) + } + + #[test] + fn teardown_answers_every_outstanding_and_unread_monitor_once() { + crate::run(|| { + let mut mons = Monitors::default(); + let (mon_tx, mon_rx) = channel::(); + + // Already registered. + let (tx1, rx1) = channel::(); + mons.outstanding.insert(MonitorId(1), (pid(1), tx1)); + // In the inbox, never processed. + let (tx2, rx2) = channel::(); + mon_tx + .send(MonCmd::Monitor { + id: MonitorId(2), + target: pid(2), + tx: tx2, + }) + .ok() + .unwrap(); + // Registered, then cancelled in the inbox: silence. + let (tx3, rx3) = channel::(); + mons.outstanding.insert(MonitorId(3), (pid(3), tx3)); + mon_tx + .send(MonCmd::Demonitor { id: MonitorId(3) }) + .ok() + .unwrap(); + + mons.teardown(&mon_rx); + + let d1 = rx1.recv().unwrap(); + assert_eq!( + (d1.pid, d1.reason), + (pid(1), RemoteDownReason::Disconnected) + ); + let d2 = rx2.recv().unwrap(); + assert_eq!( + (d2.pid, d2.reason), + (pid(2), RemoteDownReason::Disconnected) + ); + // Cancelled: no notice was sent (its sender is dropped, channel + // closed-empty), and nobody got a second one. + assert!(rx3.try_recv().is_err()); + assert!(rx1.try_recv().is_err()); + assert!(rx2.try_recv().is_err()); + assert!(mons.outstanding.is_empty()); + }); + } +} diff --git a/src/cluster/connect.rs b/src/cluster/connect.rs new file mode 100644 index 0000000..cf3d1a1 --- /dev/null +++ b/src/cluster/connect.rs @@ -0,0 +1,365 @@ +//! RFC 010 c6b — the handshake on the accept/connect path. +//! +//! Per D8 (re-amended): the c5 machines are driven by **straight-line code +//! on the path**, not by an actor. The dial side runs [`Initiator`]; the +//! acceptor loop runs [`Responder`]. A connection actor is spawned only +//! *after* a successful handshake ([`spawn_established`]); every reject, +//! protocol failure, timeout, and tie-break loss is resolved right here, +//! on the path, by closing — no actor ever exists for a connection that +//! didn't establish. +//! +//! Buffer trap (binding): the path reader and the steady-state actor share +//! ONE [`FramedConn`]. Its decode buffer may hold read-ahead past the +//! handshake frames, so the *whole* `FramedConn` travels into +//! [`spawn_established`] — never a fresh codec over the same socket. +//! +//! Layering: [`dial_handshake`] and [`accept_handshake`] are the bare path +//! steps — IO on a `FramedConn`, no manager, no actors — testable over the +//! loopback transport on plain threads. [`dial`] and [`spawn_acceptor`] are +//! the manager-integrated layer (actor context required): they keep the +//! [`manager`](crate::cluster::manager)'s dial-intent set honest and spawn +//! the connection actor on success. + +use std::io; +use std::time::{Duration, Instant}; + +use crate::channel::{channel, Receiver, Selectable, Sender}; +use crate::cluster::conn::spawn_established; +use crate::cluster::envelope::{Frame, RejectReason}; +use crate::cluster::handshake::{ + Initiator, InitiatorOutcome, Local, Peer, PeerStanding, Responder, ResponderOutcome, +}; +use crate::cluster::manager::{Call, Reply, MANAGER}; +use crate::cluster::transport::{FramedConn, Listener, RecvError, SendError, Transport}; +use crate::cluster::Timing; +use crate::gen_server; +use crate::pid::Pid; +use crate::scheduler::{self, spawn}; + +/// Default for [`Timing::handshake_timeout`]: how long either side waits for +/// the peer's handshake frame before giving up and closing. Enforced on the path via [`FramedConn::recv_deadline`], +/// so a peer that connects and goes silent cannot wedge the acceptor. +pub const HANDSHAKE_TIMEOUT: Duration = Duration::from_secs(5); + +/// Why a handshake did not establish. In every case the connection has +/// already been closed on the path by the time this is returned. +#[derive(Debug)] +pub enum HandshakeError { + /// A `HelloReject` travelled — sent by us (accept side) or received by + /// us (dial side). + Rejected(RejectReason), + /// Accept side only: the inbound dial lost the simultaneous-connect + /// tie-break (D7) and was closed silently, no frame sent. + TieBreakLoss, + /// The peer spoke a valid frame that is wrong here (non-`Hello` first + /// frame; non-response to our `Hello`), or an undecodable byte stream. + Protocol, + /// EOF before the handshake resolved. On the dial side this is also + /// what losing the tie-break looks like: the peer closes silently. + Closed, + /// [`HANDSHAKE_TIMEOUT`] (or the caller's deadline) passed first. + TimedOut, + /// The transport failed mid-handshake. + Transport(io::Error), +} + +fn from_send(e: SendError) -> HandshakeError { + match e { + // Handshake frames are small and self-made; an encode failure is a + // protocol-level impossibility, not a transport fault. + SendError::Encode(_) => HandshakeError::Protocol, + SendError::Io(e) => HandshakeError::Transport(e), + } +} + +fn from_recv(e: RecvError) -> HandshakeError { + match e { + RecvError::Corrupt(_) => HandshakeError::Protocol, + RecvError::TruncatedByPeer => HandshakeError::Closed, + RecvError::Io(e) => HandshakeError::Transport(e), + RecvError::TimedOut => HandshakeError::TimedOut, + } +} + +/// Dial-side path step: send our `Hello`, interpret the one response. On +/// `Ok` the connection is established and `framed` is live (with any +/// read-ahead intact in its buffer); on `Err` the connection is closed. +pub fn dial_handshake( + framed: &mut FramedConn, + local: &Local, + deadline: Instant, +) -> Result { + let (initiator, hello) = Initiator::new(local); + if let Err(e) = framed.send(&hello) { + framed.close(); + return Err(from_send(e)); + } + let outcome = match framed.recv_deadline(deadline) { + Ok(Some(frame)) => initiator.on_frame(frame), + Ok(None) => { + framed.close(); + return Err(HandshakeError::Closed); + } + Err(e) => { + framed.close(); + return Err(from_recv(e)); + } + }; + match outcome { + InitiatorOutcome::Established(peer) => Ok(peer), + InitiatorOutcome::Rejected(reason) => { + framed.close(); + Err(HandshakeError::Rejected(reason)) + } + InitiatorOutcome::Failed(_) => { + framed.close(); + Err(HandshakeError::Protocol) + } + } +} + +/// Accept-side path step: read the first frame, judge it, answer or close. +/// +/// `standing_of` supplies the [`PeerStanding`] of the *offered* name — knowledge +/// only the frame reveals, which is why it is a callback and not a value +/// (the integrated acceptor asks the manager; loopback tests fabricate). +/// It is not called when the first frame is not a `Hello`. +/// +/// On `Ok` the ack has been sent and `framed` is live (read-ahead intact); +/// on `Err` any owed reject has been sent and the connection is closed. +pub fn accept_handshake( + framed: &mut FramedConn, + local: Local, + standing_of: impl FnOnce(&str) -> PeerStanding, + deadline: Instant, +) -> Result { + let frame = match framed.recv_deadline(deadline) { + Ok(Some(frame)) => frame, + Ok(None) => { + framed.close(); + return Err(HandshakeError::Closed); + } + Err(e) => { + framed.close(); + return Err(from_recv(e)); + } + }; + let standing = match &frame { + Frame::Hello { node_name, .. } => standing_of(node_name), + _ => PeerStanding::Free, + }; + match Responder::new(local).on_frame(frame, standing) { + ResponderOutcome::Accepted { reply, peer } => { + if let Err(e) = framed.send(&reply) { + framed.close(); + return Err(from_send(e)); + } + Ok(peer) + } + ResponderOutcome::Rejected { reply, reason } => { + // Best effort: the reject is the cross-version compatibility + // anchor, but if the write fails the peer sees a bare close, + // which it must survive anyway. + let _ = framed.send(&reply); + framed.close(); + Err(HandshakeError::Rejected(reason)) + } + ResponderOutcome::TieBreakLoss => { + // D7: close silently — the peer computes the same verdict. + framed.close(); + Err(HandshakeError::TieBreakLoss) + } + ResponderOutcome::Failed(_) => { + framed.close(); + Err(HandshakeError::Protocol) + } + } +} + +// --------------------------------------------------------------------------- +// Manager-integrated layer +// --------------------------------------------------------------------------- + +/// Why an integrated [`dial`] did not produce a connection. +#[derive(Debug)] +pub enum DialError { + /// Another dial to this peer name is already in flight. + AlreadyDialing, + /// The manager is not running (or answered nonsense). + ManagerUnavailable, + /// The transport could not connect. + Connect(io::Error), + /// Connected, but the handshake did not establish. + Handshake(HandshakeError), + /// The peer at `addr` established, but answered as a different name + /// than the one we dialed — the tie-break bookkeeping (keyed by the + /// dialed name) would be unsound, so the connection is closed. + PeerNameMismatch { expected: String, got: String }, + /// The handshake established, but the manager refused the registration: + /// a connection to this peer already exists. The loser has been closed. + Duplicate, +} + +impl DialError { + /// A short static label per kind, for the `smarm-trace` `ClusterDial` + /// event; the payload (io error, names) is not carried. + pub fn label(&self) -> &'static str { + match self { + DialError::AlreadyDialing => "already_dialing", + DialError::ManagerUnavailable => "manager_unavailable", + DialError::Connect(_) => "connect", + DialError::Handshake(_) => "handshake", + DialError::PeerNameMismatch { .. } => "peer_name_mismatch", + DialError::Duplicate => "duplicate", + } + } +} + +/// Dial `peer_name` at `addr` and run the handshake, keeping the manager's +/// dial-intent set honest around it: the intent is registered *before* +/// connecting (so a crossing inbound `Hello` sees it) and cleared the +/// moment the handshake resolves, before the connection actor is spawned. +/// Must run inside an actor. Retrying is the caller's business (c7's dial +/// loop); a lost tie-break surfaces as `Handshake(Closed)` — the peer's +/// accepted connection is already on its way. +pub fn dial( + transport: &dyn Transport, + addr: &str, + peer_name: &str, + local: &Local, + timing: Timing, +) -> Result { + let me = scheduler::self_pid(); + match gen_server::call( + MANAGER, + Call::DialBegin { + name: peer_name.to_string(), + pid: me, + }, + ) { + Ok(Reply::DialBegan(true)) => {} + Ok(Reply::DialBegan(false)) => return Err(DialError::AlreadyDialing), + _ => return Err(DialError::ManagerUnavailable), + } + let result = connect_and_shake(transport, addr, local, timing); + // Cleared immediately on outcome — a stale intent during the established + // window would corrupt later tie-breaks. Synchronous (a call): the + // intent is provably gone before anything else happens. + let _ = gen_server::call( + MANAGER, + Call::DialEnd { + name: peer_name.to_string(), + }, + ); + let (mut framed, peer) = result?; + if peer.node_name != peer_name { + framed.close(); + return Err(DialError::PeerNameMismatch { + expected: peer_name.to_string(), + got: peer.node_name, + }); + } + spawn_established(framed, peer, timing).map_err(|_| DialError::Duplicate) +} + +fn connect_and_shake( + transport: &dyn Transport, + addr: &str, + local: &Local, + timing: Timing, +) -> Result<(FramedConn, Peer), DialError> { + let conn = transport.dial(addr).map_err(DialError::Connect)?; + let mut framed = FramedConn::new(conn); + let deadline = Instant::now() + timing.handshake_timeout; + let peer = dial_handshake(&mut framed, local, deadline).map_err(DialError::Handshake)?; + Ok((framed, peer)) +} + +/// A running acceptor. [`shutdown`](AcceptorHandle::shutdown) (or dropping +/// the last handle) stops the accept loop only: connections it established +/// belong to the [`manager`](crate::cluster::manager) and keep running, to +/// be torn down through the table (`Disconnect`, a peer close, or manager +/// shutdown). +pub struct AcceptorHandle { + cmd_tx: Sender<()>, + addr: String, +} + +impl AcceptorHandle { + /// Ask the acceptor to stop. Idempotent; a no-op if it already has. + pub fn shutdown(&self) { + let _ = self.cmd_tx.send(()); + } + + /// The concrete bound address, dialable as-is. + pub fn local_addr(&self) -> &str { + &self.addr + } +} + +/// Spawn the acceptor actor over a bound listener. Each inbound connection +/// is handshaken **inline in the loop** (a deliberate serialization: the +/// per-frame deadline bounds how long any one peer can hold the line, and +/// nothing concurrent exists to be starved before c7). The listener must be +/// fd-backed ([`Listener::readable_arm`]); the loopback listener is not, +/// and its acceptor exits immediately — loopback handshakes are driven +/// synchronously through the path fns instead, per D8. +pub fn spawn_acceptor(listener: Box, local: Local, timing: Timing) -> AcceptorHandle { + let addr = listener.local_addr(); + let (cmd_tx, cmd_rx) = channel(); + spawn(move || accept_loop(listener, local, timing, cmd_rx)); + AcceptorHandle { cmd_tx, addr } +} + +fn accept_loop( + mut listener: Box, + local: Local, + timing: Timing, + cmd_rx: Receiver<()>, +) { + loop { + let Some(arm) = listener.readable_arm() else { + return; + }; + let arms: [&dyn Selectable; 2] = [&cmd_rx, &arm]; + match crate::channel::try_select(&arms) { + Ok(0) => match cmd_rx.try_recv() { + Ok(Some(())) => return, + Ok(None) => continue, // spurious wake + Err(_) => return, // all handles dropped + }, + Ok(_) => { + // The listener is readable: accept completes without parking. + let conn = match listener.accept() { + Ok(conn) => conn, + Err(_) => return, // listener itself is broken + }; + handle_inbound(FramedConn::new(conn), &local, timing); + } + Err(_) => return, // fd arm failed to register: listener is gone + } + } +} + +/// Run the accept-side handshake for one inbound connection, asking the +/// manager for the [`PeerStanding`], and hand the established connection to the +/// manager. Every failure was already resolved on the path (reject sent / +/// closed, or the registration refused and the actor stopped), so there is +/// nothing for the acceptor to carry forward. +fn handle_inbound(mut framed: FramedConn, local: &Local, timing: Timing) { + let deadline = Instant::now() + timing.handshake_timeout; + let standing_of = |name: &str| match gen_server::call( + MANAGER, + Call::Standing { + peer_name: name.to_string(), + }, + ) { + Ok(Reply::Standing(s)) => s, + // Manager unreachable: nobody could register this connection anyway, + // so claim the name taken and reject rather than accept an orphan. + _ => PeerStanding::Claimed, + }; + if let Ok(peer) = accept_handshake(&mut framed, local.clone(), standing_of, deadline) { + let _ = spawn_established(framed, peer, timing); + } +} diff --git a/src/cluster/connector.rs b/src/cluster/connector.rs new file mode 100644 index 0000000..2d9eaa0 --- /dev/null +++ b/src/cluster/connector.rs @@ -0,0 +1,316 @@ +//! RFC 010 c7b — the connector: the dial loop that turns discovered +//! candidates into a full mesh. +//! +//! A plain select-loop actor (the c6 shape). It spawns its [`Strategy`] as a +//! child actor and receives [`Discovery`] events from it; it tracks which +//! peers are up by **subscribing to membership like any other consumer** — +//! no privileged channel into the manager, the same snapshot-then-stream +//! surface c8 will use. One `select` folds the command inbox, the discovery +//! stream, the membership stream, and the earliest retry deadline into a +//! single wait. +//! +//! Per-candidate state: dial on arrival; on failure retry with capped +//! exponential backoff ([`INITIAL_BACKOFF`] doubling to [`MAX_BACKOFF`]); +//! on the peer's `node_up` stop dialing and reset the backoff; on its +//! `node_down` resume immediately (a fresh sequence — the reconnect case is +//! the one backoff exists to pace, but the *first* retry after a death +//! should be prompt). A candidate bearing our own name is parked permanently +//! — that seed is us; so is one whose address answers as a different name +//! (`PeerNameMismatch`: a misconfigured or stale seed — each retry would +//! only blip the peer's membership). Every other failure retries: in +//! particular a `NameTaken` reject can be our own ghost at the peer, not +//! yet reaped by its liveness timer, so it must not park. Each attempt's +//! outcome is one `smarm-trace` `ClusterDial` event. A [`Discovery::Withdrawn`] +//! drops its `(name, addr)` from the dial set — only that: a live +//! connection is membership's, and a re-announce re-adds it fresh. +//! +//! Dials run **inline in the loop** — the same deliberate serialization as +//! the acceptor (c6b): each attempt is bounded by the connect + handshake +//! deadlines, and nothing concurrent exists to be starved. A wall of slow +//! unreachable seeds would stretch the loop's latency; revisit if a real +//! deployment ever hits that shape. + +use std::collections::HashSet; +use std::time::{Duration, Instant}; + +use crate::channel::{channel, select, select_timeout, Receiver, Selectable, Sender}; +use crate::cluster::connect::{dial, DialError}; +use crate::cluster::discovery::{Discovery, Strategy}; +use crate::cluster::handshake::Local; +use crate::cluster::membership::{subscribe, NodeEvent}; +use crate::cluster::transport::Transport; +use crate::cluster::Timing; +use crate::scheduler::spawn; + +/// Default for [`Timing::initial_backoff`]: first retry delay after a failed +/// dial attempt. +pub const INITIAL_BACKOFF: Duration = Duration::from_millis(250); +/// Default for [`Timing::max_backoff`]: an unreachable seed is retried this +/// often, forever. +pub const MAX_BACKOFF: Duration = Duration::from_secs(5); + +enum Cmd { + Shutdown, +} + +/// A running connector. `shutdown` (or dropping the last handle) stops the +/// dial loop and its strategy only — established connections belong to the +/// manager, exactly as with the acceptor. +pub struct ConnectorHandle { + cmd_tx: Sender, +} + +impl ConnectorHandle { + /// Ask the connector to stop. Idempotent; a no-op if it already has. + pub fn shutdown(&self) { + let _ = self.cmd_tx.send(Cmd::Shutdown); + } +} + +/// One discovered `(name, addr)` and our dial intent towards it. +struct Candidate { + name: String, + addr: String, + state: State, +} + +/// The connector's *intent* for a candidate. Whether the peer is currently +/// up is a separate, name-keyed membership fact (`up` in [`run`]): a +/// candidate can arrive after its peer's `node_up` (the snapshot lands +/// before the strategy has said anything), so "up" cannot live on the +/// candidate alone — it is a filter over dialing, not a candidate state. +enum State { + /// Never dialed: this seed is the local node itself, or the address + /// answered as a *different* name than the one seeded + /// (`DialError::PeerNameMismatch` — a misconfigured or stale seed; + /// redialing would only blip the peer's membership forever). The way + /// back is the strategy's: `Withdrawn` then a fresh `Candidate`. + Parked, + /// Dial when due; on failure, back off. + Dialing { + /// Delay to apply after the *next* failure. + backoff: Duration, + next_attempt: Instant, + }, +} + +impl State { + fn fresh(timing: &Timing) -> Self { + State::Dialing { + backoff: timing.initial_backoff, + next_attempt: Instant::now(), + } + } +} + +impl Candidate { + /// The retry deadline, if this candidate is dialing at all. + fn due(&self) -> Option { + match self.state { + State::Parked => None, + State::Dialing { next_attempt, .. } => Some(next_attempt), + } + } + /// A dial attempt was made: schedule the retry, grow the backoff. + fn attempted(&mut self, timing: &Timing) { + if let State::Dialing { + backoff, + next_attempt, + } = &mut self.state + { + *next_attempt = Instant::now() + *backoff; + *backoff = (*backoff * 2).min(timing.max_backoff); + } + } + /// The peer came up: the next sequence (after a later `node_down`) + /// starts from the initial delay again. + fn peer_up(&mut self, timing: &Timing) { + if let State::Dialing { backoff, .. } = &mut self.state { + *backoff = timing.initial_backoff; + } + } + /// The peer went down: redial promptly, fresh sequence. + fn peer_down(&mut self, timing: &Timing) { + if matches!(self.state, State::Dialing { .. }) { + self.state = State::fresh(timing); + } + } +} + +/// Spawn the connector actor. The strategy is spawned as its child; the +/// membership subscription is taken inside the actor. Must be called from +/// inside an actor (the same requirement as `dial`). +pub fn spawn_connector( + transport: Box, + local: Local, + strategy: Box, + timing: Timing, +) -> ConnectorHandle { + let (cmd_tx, cmd_rx) = channel(); + spawn(move || run(transport, local, strategy, timing, cmd_rx)); + ConnectorHandle { cmd_tx } +} + +fn run( + transport: Box, + local: Local, + strategy: Box, + timing: Timing, + cmd_rx: Receiver, +) { + // Membership is the connector's source of truth for "who is up" — the + // snapshot seeds `up` before any candidate arrives. + let Some(events) = subscribe() else { + return; // no manager, no cluster to connect + }; + let (disc_tx, disc_rx) = channel(); + spawn(move || strategy.run(disc_tx)); + + let mut cands: Vec = Vec::new(); + let mut up: HashSet = HashSet::new(); + let mut strategy_done = false; + + loop { + // Drain every input, then act. Order does not matter: acting is + // idempotent against the resulting state. + match drain_cmd(&cmd_rx) { + Drained::Stop => return, + Drained::Open => {} + } + if !strategy_done { + strategy_done = drain_discoveries(&disc_rx, &local, &timing, &mut cands); + } + match drain_events(&events.rx, &timing, &mut up, &mut cands) { + Drained::Stop => return, // manager gone: the cluster is tearing down + Drained::Open => {} + } + + // Dial everything due, inline (see the module docs on serialization). + let now = Instant::now(); + for c in cands + .iter_mut() + .filter(|c| !up.contains(&c.name) && c.due().is_some_and(|d| d <= now)) + { + // On success the manager's node_up is on its way and lands in + // `up` (backing off meanwhile keeps a racing re-attempt from + // spinning); every failure retries — see the module docs — + // except a peer-name mismatch, which parks the candidate. + let outcome = dial(&*transport, &c.addr, &c.name, &local, timing); + note_dial(&outcome); + match outcome { + Err(DialError::PeerNameMismatch { .. }) => c.state = State::Parked, + _ => c.attempted(&timing), + } + } + + // Wait: until the earliest retry deadline among actionable + // candidates, or indefinitely if none is pending. + let deadline = cands + .iter() + .filter(|c| !up.contains(&c.name)) + .filter_map(Candidate::due) + .min(); + let mut arms: Vec<&dyn Selectable> = vec![&cmd_rx, &events.rx]; + if !strategy_done { + arms.push(&disc_rx); + } + match deadline { + Some(d) => { + let wait = d.saturating_duration_since(Instant::now()); + let _ = select_timeout(&arms, wait); + } + None => { + let _ = select(&arms); + } + } + } +} + +enum Drained { + Open, + Stop, +} + +fn drain_cmd(rx: &Receiver) -> Drained { + match rx.try_recv() { + Ok(Some(Cmd::Shutdown)) => Drained::Stop, + Ok(None) => Drained::Open, + Err(_) => Drained::Stop, // all handles dropped + } +} + +/// Pull every pending discovery into the candidate set (deduplicated by +/// `(name, addr)`; a candidate bearing the local name is parked; a +/// `Withdrawn` removes its pair from the dial set and nothing else — see +/// [`Discovery::Withdrawn`]). Returns `true` once the strategy's channel +/// closes — it has said all it will. +fn drain_discoveries( + rx: &Receiver, + local: &Local, + timing: &Timing, + cands: &mut Vec, +) -> bool { + loop { + match rx.try_recv() { + Ok(Some(Discovery::Withdrawn { name, addr })) => { + cands.retain(|c| !(c.name == name && c.addr == addr)); + } + Ok(Some(Discovery::Candidate { name, addr })) => { + if cands.iter().any(|c| c.name == name && c.addr == addr) { + continue; + } + let state = if name == local.node_name { + State::Parked + } else { + State::fresh(timing) + }; + cands.push(Candidate { name, addr, state }); + } + Ok(None) => return false, + Err(_) => return true, // strategy done; its candidates live on here + } + } +} + +/// Fold pending membership events into `up`; each is also a transition on +/// that peer's candidates (see [`Candidate::peer_up`] / [`peer_down`]). +/// +/// [`peer_down`]: Candidate::peer_down +fn drain_events( + rx: &Receiver, + timing: &Timing, + up: &mut HashSet, + cands: &mut [Candidate], +) -> Drained { + loop { + match rx.try_recv() { + Ok(Some(NodeEvent::NodeUp(info))) => { + cands + .iter_mut() + .filter(|c| c.name == info.name) + .for_each(|c| c.peer_up(timing)); + up.insert(info.name); + } + Ok(Some(NodeEvent::NodeDown(info))) => { + up.remove(&info.name); + cands + .iter_mut() + .filter(|c| c.name == info.name) + .for_each(|c| c.peer_down(timing)); + } + Ok(None) => return Drained::Open, + Err(_) => return Drained::Stop, + } + } +} + +/// Surface a dial outcome: one `smarm-trace` event, nothing else. The +/// connector's bookkeeping is decided by the caller. +fn note_dial(outcome: &Result) { + #[cfg(feature = "smarm-trace")] + crate::te!(crate::trace::Event::ClusterDial( + outcome.as_ref().map_or_else(DialError::label, |_| "ok") + )); + #[cfg(not(feature = "smarm-trace"))] + let _ = outcome; +} diff --git a/src/cluster/discovery.rs b/src/cluster/discovery.rs new file mode 100644 index 0000000..edd37a7 --- /dev/null +++ b/src/cluster/discovery.rs @@ -0,0 +1,76 @@ +//! RFC 010 c7b — peer discovery: the [`Strategy`] seam and the static-seeds +//! implementation. +//! +//! A strategy is **push-based and runs as its own actor**: the +//! [`connector`](crate::cluster::connector) spawns it with the sending end of +//! a channel, and the strategy emits [`Discovery`] events whenever it learns +//! something — once at startup for a static list, continuously for a future +//! mDNS/DNS strategy — for as long as it cares to run. Returning ends the +//! strategy actor; the candidates it pushed live on in the connector (the +//! connector owns all retry/backoff state, so a strategy never re-announces). +//! +//! A candidate is a **`(node_name, addr)` pair**, not a bare address: the +//! dial path and the D7 tie-break are keyed by peer *name* (the dial intent +//! must be registered before connecting so a crossing inbound `Hello` sees +//! it), so an anonymous dial would reintroduce exactly the +//! simultaneous-connect flap D7 exists to prevent. Discovery mechanisms know +//! names — that is what they discover. + +use crate::channel::Sender; + +/// A discovery event, as pushed by a [`Strategy`]. +/// +/// `Candidate` announces, `Withdrawn` retracts — the primitive pair. A +/// strategy that wants TTL semantics builds them on top (track its own +/// last-seen times, emit `Withdrawn` on expiry); the connector deliberately +/// has no clock of its own for candidates (D11: strategies never +/// re-announce, the connector owns retry). `#[non_exhaustive]` so more can +/// land without breaking strategies. +#[derive(Debug, Clone, PartialEq, Eq)] +#[non_exhaustive] +pub enum Discovery { + /// A peer worth dialing: its claimed node name and a dialable address. + Candidate { name: String, addr: String }, + /// Stop dialing this `(name, addr)`. Dial-set only: a connection that + /// is already up is membership's business and is left alone; an + /// attempt in flight completes on its own; a later `Candidate` for the + /// same pair re-adds it with fresh backoff. Unknown pairs are ignored. + Withdrawn { name: String, addr: String }, +} + +/// A source of peers to dial. Implementations are spawned as actors by the +/// connector — see the module docs for the contract. +pub trait Strategy: Send + 'static { + /// Run the strategy: push [`Discovery`] events into `out` as they are + /// learned; return when done discovering (or when `out` reports closed — + /// the connector is gone). Runs inside an actor, so blocking + /// cooperatively is fine. + fn run(self: Box, out: Sender); +} + +/// The static-seeds strategy: a fixed `(name, addr)` list, announced once. +#[derive(Debug, Clone, Default)] +pub struct StaticSeeds { + seeds: Vec<(String, String)>, +} + +impl StaticSeeds { + pub fn new(seeds: impl IntoIterator, impl Into)>) -> Self { + StaticSeeds { + seeds: seeds + .into_iter() + .map(|(n, a)| (n.into(), a.into())) + .collect(), + } + } +} + +impl Strategy for StaticSeeds { + fn run(self: Box, out: Sender) { + for (name, addr) in self.seeds { + if out.send(Discovery::Candidate { name, addr }).is_err() { + return; // connector gone; nobody to discover for + } + } + } +} diff --git a/src/cluster/envelope.rs b/src/cluster/envelope.rs new file mode 100644 index 0000000..2048021 --- /dev/null +++ b/src/cluster/envelope.rs @@ -0,0 +1,549 @@ +//! RFC 010 c2 — the owned wire envelope. +//! +//! Every control-plane frame is `u32` little-endian length prefix (of tag + +//! body), `u8` tag, hand-encoded body. postcard appears in exactly one place: +//! the payload blob inside `Send`/`SendNamed`, via [`encode_payload`] / +//! [`decode_payload`] — the seam where a codec swap would land (RFC 010 §2). +//! Everything else is hand-rolled and wholly owned. +//! +//! Integers are little-endian. Strings are `u16` length + UTF-8 bytes. +//! Payload blobs are `u32` length + bytes. Enum-shaped fields +//! ([`RejectReason`], [`DownReason`]) are a single tag byte. + +use crate::monitor::DownReason; +use crate::pg::Incarnation; + +/// Wire protocol version, checked in the handshake (c5). +pub const PROTO_VERSION: u32 = 1; + +/// Hard cap on the length prefix. The control plane never carries bulk data +/// (RFC 010 §5 — that is the jarred rkyv plane), so anything larger is +/// corruption or an attack, not a legitimate frame. +pub const MAX_FRAME_LEN: usize = 16 * 1024 * 1024; + +/// Per-node metadata exchanged in the handshake (RFC 010 §1: not identity). +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct NodeMeta { + pub role: String, + pub region: String, +} + +/// Why a `Hello` was rejected. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum RejectReason { + /// Build hashes differ — not the same binary. + HashMismatch, + /// The offered node name is already claimed by a live peer. + NameTaken, + /// Wire protocol version mismatch. + ProtoVersion, +} + +/// The control-plane frame inventory (RFC 010, *Implementation details*). +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum Frame { + Hello { + proto_version: u32, + build_hash: u64, + node_name: String, + incarnation: Incarnation, + meta: NodeMeta, + }, + HelloAck { + node_name: String, + incarnation: Incarnation, + meta: NodeMeta, + }, + HelloReject { + reason: RejectReason, + }, + Heartbeat, + Send { + /// Target slot index (node is implicit in the connection, incarnation + /// is bound at handshake — RFC 010 §3). + index: u32, + generation: u32, + type_hash: u64, + payload: Vec, + }, + SendNamed { + name: String, + type_hash: u64, + payload: Vec, + }, + Monitor { + monitor_id: u64, + index: u32, + generation: u32, + }, + Demonitor { + monitor_id: u64, + }, + Down { + monitor_id: u64, + reason: RemoteDownReason, + }, +} + +/// Why a remotely-monitored actor is reported down: either the target's own +/// terminal [`DownReason`] as its node recorded it, or the *link* to that +/// node was lost (or absent) — which says nothing about the actor itself. +/// +/// This is the cluster-side widening of `DownReason` (p5): `Disconnected` +/// is a fact about a connection, never about a local actor, so it lives +/// here rather than in the core enum — a local `Down` can never carry it, +/// and matches on `DownReason` stay exhaustive over actor outcomes only. +/// On the wire `Local(r)` uses `r`'s tag and `Disconnected` is tag 5, +/// bound since c11; no peer emits it today (a lost link is synthesized +/// locally), but the codec honours it both ways. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum RemoteDownReason { + /// The target itself terminated; the peer reported this reason. + Local(DownReason), + /// The link to the target's node was lost or was never up. + Disconnected, +} + +impl RemoteDownReason { + /// The actor's own reason, if this was not a link loss. + pub fn local(self) -> Option { + match self { + RemoteDownReason::Local(r) => Some(r), + RemoteDownReason::Disconnected => None, + } + } +} + +impl From for RemoteDownReason { + fn from(r: DownReason) -> Self { + RemoteDownReason::Local(r) + } +} + +// Frame tags. 0 is deliberately unassigned so an all-zero buffer never parses. +const TAG_HELLO: u8 = 1; +const TAG_HELLO_ACK: u8 = 2; +const TAG_HELLO_REJECT: u8 = 3; +const TAG_HEARTBEAT: u8 = 4; +const TAG_SEND: u8 = 5; +const TAG_SEND_NAMED: u8 = 6; +const TAG_MONITOR: u8 = 7; +const TAG_DEMONITOR: u8 = 8; +const TAG_DOWN: u8 = 9; + +// RejectReason tags. +const REJ_HASH_MISMATCH: u8 = 1; +const REJ_NAME_TAKEN: u8 = 2; +const REJ_PROTO_VERSION: u8 = 3; + +// DownReason tags. Do not reuse tags. +const DR_EXIT: u8 = 1; +const DR_PANIC: u8 = 2; +const DR_STOPPED: u8 = 3; +const DR_NOPROC: u8 = 4; +const DR_DISCONNECTED: u8 = 5; +// `Shutdown` never rides in a `Down` by contract (a target that honours the +// request exits normally) — the tag exists so the codec stays total. +const DR_SHUTDOWN: u8 = 6; + +/// Frame could not be encoded. The output buffer is left exactly as it was. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum EncodeError { + /// tag + body exceed [`MAX_FRAME_LEN`]. + FrameTooLarge { len: usize }, + /// A string field exceeds `u16::MAX` bytes. + StringTooLong { len: usize }, +} + +impl core::fmt::Display for EncodeError { + fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { + match self { + Self::FrameTooLarge { len } => { + write!(f, "frame body of {len} bytes exceeds MAX_FRAME_LEN") + } + Self::StringTooLong { len } => { + write!(f, "string field of {len} bytes exceeds u16::MAX") + } + } + } +} + +impl std::error::Error for EncodeError {} + +/// Frame could not be decoded. Everything here is *corruption* — "not enough +/// bytes yet" is the `Ok(None)` streaming case, never an error. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum DecodeError { + /// The length prefix exceeds [`MAX_FRAME_LEN`]. + FrameTooLarge { declared: usize }, + /// The length prefix is zero — there is no tag byte. + EmptyFrame, + /// Unknown frame tag. + UnknownTag(u8), + /// Unknown tag for an enum-shaped field. + UnknownEnumTag { what: &'static str, tag: u8 }, + /// A field ran past the declared frame end (the length prefix lied long, + /// or a length-carrying field inside the body lied). + Truncated, + /// Bytes were left over after the body (the length prefix lied short). + Trailing { extra: usize }, + /// A string field was not valid UTF-8. + Utf8, +} + +impl core::fmt::Display for DecodeError { + fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { + match self { + Self::FrameTooLarge { declared } => { + write!(f, "declared frame length {declared} exceeds MAX_FRAME_LEN") + } + Self::EmptyFrame => write!(f, "zero-length frame (no tag byte)"), + Self::UnknownTag(t) => write!(f, "unknown frame tag {t}"), + Self::UnknownEnumTag { what, tag } => write!(f, "unknown {what} tag {tag}"), + Self::Truncated => write!(f, "frame body truncated mid-field"), + Self::Trailing { extra } => write!(f, "{extra} trailing bytes after frame body"), + Self::Utf8 => write!(f, "string field is not valid UTF-8"), + } + } +} + +impl std::error::Error for DecodeError {} + +impl Frame { + /// Append this frame, length-prefixed, to `out`. + /// + /// On error `out` is left untouched. + pub fn encode(&self, out: &mut Vec) -> Result<(), EncodeError> { + let start = out.len(); + out.extend_from_slice(&[0u8; 4]); // length placeholder, patched below + let result = self.encode_body(out); + match result { + Ok(()) => { + let frame_len = out.len() - start - 4; + if frame_len > MAX_FRAME_LEN { + out.truncate(start); + return Err(EncodeError::FrameTooLarge { len: frame_len }); + } + // Cast is lossless: MAX_FRAME_LEN < u32::MAX, checked above. + let len32 = frame_len as u32; + out[start..start + 4].copy_from_slice(&len32.to_le_bytes()); + Ok(()) + } + Err(e) => { + out.truncate(start); + Err(e) + } + } + } + + fn encode_body(&self, out: &mut Vec) -> Result<(), EncodeError> { + match self { + Frame::Hello { + proto_version, + build_hash, + node_name, + incarnation, + meta, + } => { + out.push(TAG_HELLO); + put_u32(out, *proto_version); + put_u64(out, *build_hash); + put_str(out, node_name)?; + put_u32(out, incarnation.get()); + put_meta(out, meta)?; + } + Frame::HelloAck { + node_name, + incarnation, + meta, + } => { + out.push(TAG_HELLO_ACK); + put_str(out, node_name)?; + put_u32(out, incarnation.get()); + put_meta(out, meta)?; + } + Frame::HelloReject { reason } => { + out.push(TAG_HELLO_REJECT); + out.push(match reason { + RejectReason::HashMismatch => REJ_HASH_MISMATCH, + RejectReason::NameTaken => REJ_NAME_TAKEN, + RejectReason::ProtoVersion => REJ_PROTO_VERSION, + }); + } + Frame::Heartbeat => out.push(TAG_HEARTBEAT), + Frame::Send { + index, + generation, + type_hash, + payload, + } => { + out.push(TAG_SEND); + put_u32(out, *index); + put_u32(out, *generation); + put_u64(out, *type_hash); + put_blob(out, payload)?; + } + Frame::SendNamed { + name, + type_hash, + payload, + } => { + out.push(TAG_SEND_NAMED); + put_str(out, name)?; + put_u64(out, *type_hash); + put_blob(out, payload)?; + } + Frame::Monitor { + monitor_id, + index, + generation, + } => { + out.push(TAG_MONITOR); + put_u64(out, *monitor_id); + put_u32(out, *index); + put_u32(out, *generation); + } + Frame::Demonitor { monitor_id } => { + out.push(TAG_DEMONITOR); + put_u64(out, *monitor_id); + } + Frame::Down { monitor_id, reason } => { + out.push(TAG_DOWN); + put_u64(out, *monitor_id); + out.push(match reason { + RemoteDownReason::Local(DownReason::Exit) => DR_EXIT, + RemoteDownReason::Local(DownReason::Panic) => DR_PANIC, + RemoteDownReason::Local(DownReason::Stopped) => DR_STOPPED, + RemoteDownReason::Local(DownReason::NoProc) => DR_NOPROC, + RemoteDownReason::Local(DownReason::Shutdown) => DR_SHUTDOWN, + RemoteDownReason::Disconnected => DR_DISCONNECTED, + }); + } + } + Ok(()) + } + + /// Try to decode one frame from the start of `buf`. + /// + /// `Ok(Some((frame, consumed)))` — a full frame; the caller advances by + /// `consumed`. `Ok(None)` — not enough bytes yet (streaming); read more + /// and retry. `Err(_)` — the bytes are corrupt; the connection is dead. + pub fn decode(buf: &[u8]) -> Result, DecodeError> { + let Some(prefix) = buf.get(0..4) else { + return Ok(None); + }; + let mut len4 = [0u8; 4]; + len4.copy_from_slice(prefix); + let declared = u32::from_le_bytes(len4) as usize; + if declared > MAX_FRAME_LEN { + return Err(DecodeError::FrameTooLarge { declared }); + } + if declared == 0 { + return Err(DecodeError::EmptyFrame); + } + let Some(body) = buf.get(4..4 + declared) else { + return Ok(None); + }; + let mut r = Reader { buf: body, pos: 0 }; + let frame = Self::decode_body(&mut r)?; + if r.pos != body.len() { + return Err(DecodeError::Trailing { + extra: body.len() - r.pos, + }); + } + Ok(Some((frame, 4 + declared))) + } + + fn decode_body(r: &mut Reader<'_>) -> Result { + let tag = r.u8()?; + let frame = match tag { + TAG_HELLO => Frame::Hello { + proto_version: r.u32()?, + build_hash: r.u64()?, + node_name: r.string()?, + incarnation: Incarnation::new(r.u32()?), + meta: r.meta()?, + }, + TAG_HELLO_ACK => Frame::HelloAck { + node_name: r.string()?, + incarnation: Incarnation::new(r.u32()?), + meta: r.meta()?, + }, + TAG_HELLO_REJECT => Frame::HelloReject { + reason: match r.u8()? { + REJ_HASH_MISMATCH => RejectReason::HashMismatch, + REJ_NAME_TAKEN => RejectReason::NameTaken, + REJ_PROTO_VERSION => RejectReason::ProtoVersion, + t => { + return Err(DecodeError::UnknownEnumTag { + what: "RejectReason", + tag: t, + }) + } + }, + }, + TAG_HEARTBEAT => Frame::Heartbeat, + TAG_SEND => Frame::Send { + index: r.u32()?, + generation: r.u32()?, + type_hash: r.u64()?, + payload: r.blob()?, + }, + TAG_SEND_NAMED => Frame::SendNamed { + name: r.string()?, + type_hash: r.u64()?, + payload: r.blob()?, + }, + TAG_MONITOR => Frame::Monitor { + monitor_id: r.u64()?, + index: r.u32()?, + generation: r.u32()?, + }, + TAG_DEMONITOR => Frame::Demonitor { + monitor_id: r.u64()?, + }, + TAG_DOWN => Frame::Down { + monitor_id: r.u64()?, + reason: match r.u8()? { + DR_EXIT => RemoteDownReason::Local(DownReason::Exit), + DR_PANIC => RemoteDownReason::Local(DownReason::Panic), + DR_STOPPED => RemoteDownReason::Local(DownReason::Stopped), + DR_NOPROC => RemoteDownReason::Local(DownReason::NoProc), + DR_SHUTDOWN => RemoteDownReason::Local(DownReason::Shutdown), + DR_DISCONNECTED => RemoteDownReason::Disconnected, + t => { + return Err(DecodeError::UnknownEnumTag { + what: "RemoteDownReason", + tag: t, + }) + } + }, + }, + t => return Err(DecodeError::UnknownTag(t)), + }; + Ok(frame) + } +} + +// --------------------------------------------------------------------------- +// Body writers +// --------------------------------------------------------------------------- + +fn put_u32(out: &mut Vec, v: u32) { + out.extend_from_slice(&v.to_le_bytes()); +} + +fn put_u64(out: &mut Vec, v: u64) { + out.extend_from_slice(&v.to_le_bytes()); +} + +fn put_str(out: &mut Vec, s: &str) -> Result<(), EncodeError> { + let Ok(len) = u16::try_from(s.len()) else { + return Err(EncodeError::StringTooLong { len: s.len() }); + }; + out.extend_from_slice(&len.to_le_bytes()); + out.extend_from_slice(s.as_bytes()); + Ok(()) +} + +fn put_blob(out: &mut Vec, b: &[u8]) -> Result<(), EncodeError> { + let Ok(len) = u32::try_from(b.len()) else { + return Err(EncodeError::FrameTooLarge { len: b.len() }); + }; + out.extend_from_slice(&len.to_le_bytes()); + out.extend_from_slice(b); + Ok(()) +} + +fn put_meta(out: &mut Vec, m: &NodeMeta) -> Result<(), EncodeError> { + put_str(out, &m.role)?; + put_str(out, &m.region) +} + +// --------------------------------------------------------------------------- +// Body reader +// --------------------------------------------------------------------------- + +struct Reader<'a> { + buf: &'a [u8], + pos: usize, +} + +impl Reader<'_> { + fn take(&mut self, n: usize) -> Result<&[u8], DecodeError> { + let end = self.pos.checked_add(n).ok_or(DecodeError::Truncated)?; + let s = self.buf.get(self.pos..end).ok_or(DecodeError::Truncated)?; + self.pos = end; + Ok(s) + } + + fn u8(&mut self) -> Result { + Ok(self.take(1)?[0]) + } + + fn u16(&mut self) -> Result { + let mut b = [0u8; 2]; + b.copy_from_slice(self.take(2)?); + Ok(u16::from_le_bytes(b)) + } + + fn u32(&mut self) -> Result { + let mut b = [0u8; 4]; + b.copy_from_slice(self.take(4)?); + Ok(u32::from_le_bytes(b)) + } + + fn u64(&mut self) -> Result { + let mut b = [0u8; 8]; + b.copy_from_slice(self.take(8)?); + Ok(u64::from_le_bytes(b)) + } + + fn string(&mut self) -> Result { + let len = self.u16()? as usize; + let bytes = self.take(len)?; + match core::str::from_utf8(bytes) { + Ok(s) => Ok(s.to_owned()), + Err(_) => Err(DecodeError::Utf8), + } + } + + fn blob(&mut self) -> Result, DecodeError> { + let len = self.u32()? as usize; + Ok(self.take(len)?.to_vec()) + } + + fn meta(&mut self) -> Result { + Ok(NodeMeta { + role: self.string()?, + region: self.string()?, + }) + } +} + +// --------------------------------------------------------------------------- +// The postcard seam (RFC 010 §2) — the ONLY place payload bytes are produced +// or consumed. A codec swap lands here and nowhere else. +// --------------------------------------------------------------------------- + +/// Payload (de)serialization failed at the codec seam. +#[derive(Debug)] +pub struct PayloadError(String); + +impl core::fmt::Display for PayloadError { + fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { + write!(f, "payload codec: {}", self.0) + } +} + +impl std::error::Error for PayloadError {} + +/// Serialize a payload value to the wire blob. +pub fn encode_payload(value: &T) -> Result, PayloadError> { + postcard::to_allocvec(value).map_err(|e| PayloadError(e.to_string())) +} + +/// Deserialize a payload value from the wire blob. +pub fn decode_payload(bytes: &[u8]) -> Result { + postcard::from_bytes(bytes).map_err(|e| PayloadError(e.to_string())) +} diff --git a/src/cluster/expose.rs b/src/cluster/expose.rs new file mode 100644 index 0000000..e35de5a --- /dev/null +++ b/src/cluster/expose.rs @@ -0,0 +1,238 @@ +//! RFC 010 c8 — explicit exposure: the node's remote surface, and the +//! fixed-seed type hash. +//! +//! Nothing local is remotely reachable by default (RFC §4 — "a gun needs a +//! safety"). [`expose`] marks a registered name remotely addressable and +//! registers `M`'s decoder under [`type_hash::()`](type_hash); +//! [`expose_type`] registers only the decoder (the reply-to path: a +//! `RemotePid` received in a message is sendable only if `A::Msg`'s +//! decoder was explicitly registered). The exposed set is the node's +//! visible, auditable remote surface ([`exposed_names`]). +//! +//! ## Where the state lives +//! +//! On `RuntimeInner`, the [`pg`](crate::pg) pattern: a leaf-locked table, +//! cfg-gated behind the `cluster` feature (zero-cost-when-off, per c1). +//! Chosen over manager-held state because c9's inbound decode consults it +//! per frame — a hot path that must not serialize every remote delivery +//! through one gen_server. The state resets with the runtime, like every +//! registry. +//! +//! ## The watchable fold (D3), against the code as it stands +//! +//! RFC §4: the exposed set is not a new registry — it folds into the +//! existing `watchable` machinery, one set, two set-sites (a pid crossing +//! the membrane, and expose). Reading the code: `register` **already +//! stamps every named holder watchable** ("no successfully-registered actor +//! can die unflagged", registry.rs), so an exposed *name*'s holder needs no +//! extra mark here — the guarantee holds by registration, and re-registration +//! after a holder's death re-stamps the new holder for free (a per-tenancy +//! mark taken at expose time could not do that). The cluster's own +//! `mark_watchable` set-site is therefore the **pid crossing the wire** — +//! serialization of a pid into a frame, c10 — the exact analog of the +//! membrane crossing. What lives here is only the name/type-level state +//! neither the registry nor the slot bits can carry: which names are +//! exposed, and how to decode each type hash. +//! +//! ## The hash +//! +//! [`type_hash`] is FNV-1a 64 (fixed seed: the FNV offset basis) over +//! `TypeId`, so it is a constant of the binary: stable across runs of the +//! same build — exactly the scope the build-hash handshake reduces the mesh +//! to — and deliberately *not* stable across builds (scope guard: no +//! cross-version wire compatibility). A collision between two exposed types +//! degrades to a decode error or a refused channel, never a misroute — the +//! local `SendError::NoChannel` guarantee survives the network (RFC §3). +//! +//! ## The decoder contract +//! +//! A decoder is **decode-and-deliver-to-pid**: it captures `M` (the one +//! typed site), decodes the payload, and hands the value to the target's +//! published channel via the registry's own dynamic send. Wire-name → +//! local-pid resolution deliberately stays *outside* — that is c9's single +//! resolution seam, and it calls [`decode_deliver`]. + +use std::any::TypeId; +use std::collections::HashMap; +use std::hash::{Hash, Hasher}; + +use crate::cluster::envelope::{decode_payload, PayloadError}; +use crate::pid::{Name, Pid}; +use crate::registry::{send_dyn, SendError}; +use crate::scheduler::with_runtime; + +/// The fixed-seed `TypeId` → `u64` hash: FNV-1a 64 over the `TypeId`'s hash +/// bytes, seeded with the FNV offset basis. A constant of the binary — see +/// the module docs for scope. +pub fn type_hash() -> u64 { + let mut h = Fnv1a64::new(); + TypeId::of::().hash(&mut h); + h.finish() +} + +/// FNV-1a 64 as a `Hasher`, so `TypeId` (opaque, `Hash`-only) can feed it. +/// Same constants as the const fns in [`crate::cluster`] (BUILD_HASH). +struct Fnv1a64(u64); + +impl Fnv1a64 { + fn new() -> Self { + Fnv1a64(0xcbf2_9ce4_8422_2325) + } +} + +impl Hasher for Fnv1a64 { + fn write(&mut self, bytes: &[u8]) { + for &b in bytes { + self.0 ^= b as u64; + self.0 = self.0.wrapping_mul(0x0000_0100_0000_01b3); + } + } + fn finish(&self) -> u64 { + self.0 + } +} + +/// Why a [`decode_deliver`] did not deliver. Payload-free mirror of the +/// registry's `SendError` where relevant — the caller (c9's inbound path) +/// has only bytes to give back, not a typed message. +#[derive(Debug)] +pub enum DeliverError { + /// No decoder is registered under this hash — the type was never + /// exposed here. + UnknownType, + /// The bytes did not decode as the registered type. + Decode(PayloadError), + /// The target actor is dead (or was never alive). + Dead, + /// The target is live but has no channel for this message type, or that + /// channel is closed — the `NoChannel` guarantee: a decoded value is + /// refused, never misrouted. + WrongChannel, +} + +/// A registered decoder: decode `bytes` as the captured type and deliver to +/// `pid`'s published channel. `Arc`, so [`decode_deliver`] can clone it out +/// from under the exposure lock and call it lock-free — the decoder's +/// `send_dyn` takes the registry lock, and the two are mutual Leaves that +/// must never nest. +type Decoder = std::sync::Arc Result<(), DeliverError> + Send + Sync>; + +/// The exposure state, one per runtime (a `RuntimeInner` field, pg-style). +pub(crate) struct ExposureState { + /// The exposed names: registry key → the type hash it expects. + exposed: HashMap<&'static str, u64>, + /// The decoders: type hash → decode-and-deliver. + decoders: HashMap, +} + +impl ExposureState { + pub(crate) fn new() -> Self { + ExposureState { + exposed: HashMap::new(), + decoders: HashMap::new(), + } + } +} + +/// Mark `name` remotely addressable and register `M`'s decoder under its +/// type hash (so both name-sends and pid-sends of `M` work — RFC §4). +/// Returns the hash. +/// +/// Exposure is a **name-level fact**, independent of who currently holds the +/// name (names late-bind: the registry re-resolves on every send, and c9's +/// seam resolves per delivery). Exposing an unregistered name is therefore +/// valid — deliveries fail with "unresolved" until someone registers it. +/// Idempotent. Must run inside [`run`](crate::run). +pub fn expose(name: Name) -> u64 +where + M: serde::de::DeserializeOwned + Send + 'static, +{ + let h = ensure_decoder::(); + with_runtime(|inner| { + inner.exposure.lock().exposed.insert(name.as_str(), h); + }); + h +} + +/// Register only `M`'s decoder (no name): the reply-to path. Returns the +/// hash. Idempotent. Must run inside [`run`](crate::run). +pub fn expose_type() -> u64 +where + M: serde::de::DeserializeOwned + Send + 'static, +{ + ensure_decoder::() +} + +fn ensure_decoder() -> u64 +where + M: serde::de::DeserializeOwned + Send + 'static, +{ + let h = type_hash::(); + with_runtime(|inner| { + inner + .exposure + .lock() + .decoders + .entry(h) + .or_insert_with(decoder::); + }); + h +} + +/// The one typed site: decode as `M`, deliver via the registry's dynamic +/// send. See the module docs for the error mapping. +fn decoder() -> Decoder +where + M: serde::de::DeserializeOwned + Send + 'static, +{ + std::sync::Arc::new(|pid, bytes| { + let m: M = decode_payload(bytes).map_err(DeliverError::Decode)?; + send_dyn(pid, m).map_err(|e| match e { + SendError::Dead(_) | SendError::Unresolved(_) | SendError::NoMember(_) => { + DeliverError::Dead + } + SendError::NoChannel(_) | SendError::Closed(_) => DeliverError::WrongChannel, + }) + }) +} + +/// The type hash `name` was exposed with, or `None` if it is not exposed. +/// Must run inside [`run`](crate::run). +pub fn exposed_hash(name: &str) -> Option { + with_runtime(|inner| inner.exposure.lock().exposed.get(name).copied()) +} + +/// Whether a decoder is registered under `hash`. Must run inside +/// [`run`](crate::run). +pub fn decoder_registered(hash: u64) -> bool { + with_runtime(|inner| inner.exposure.lock().decoders.contains_key(&hash)) +} + +/// Decode `bytes` under `hash`'s registered decoder and deliver to `pid`. +/// This is the delivery half c9's single resolution seam calls after it has +/// resolved a wire name to a local pid. Must run inside [`run`](crate::run). +pub fn decode_deliver(hash: u64, to: Pid, bytes: &[u8]) -> Result<(), DeliverError> { + // Clone the Arc under the lock, call outside it: the decoder's + // `send_dyn` takes the registry lock — a mutual Leaf with the exposure + // lock (the runtime asserts if Leaves nest). This also keeps unrelated + // deliveries uncoupled from a slow decode. + let d = with_runtime(|inner| inner.exposure.lock().decoders.get(&hash).cloned()); + match d { + Some(d) => d(to, bytes), + None => Err(DeliverError::UnknownType), + } +} + +/// The auditable remote surface: every exposed name and its type hash, +/// unordered. Must run inside [`run`](crate::run). +pub fn exposed_names() -> Vec<(&'static str, u64)> { + with_runtime(|inner| { + inner + .exposure + .lock() + .exposed + .iter() + .map(|(&n, &h)| (n, h)) + .collect() + }) +} diff --git a/src/cluster/handshake.rs b/src/cluster/handshake.rs new file mode 100644 index 0000000..36f8408 --- /dev/null +++ b/src/cluster/handshake.rs @@ -0,0 +1,178 @@ +//! RFC 010 c5 — the handshake as a pure state machine. +//! +//! Frames in, actions out — no IO, no clocks, no actors. The c6 connection +//! actor drives these machines and executes their actions; everything +//! time-shaped (handshake deadline, heartbeats) lives there. + +use crate::cluster::envelope::{Frame, NodeMeta, RejectReason, PROTO_VERSION}; +use crate::pg::Incarnation; + +/// This node's identity and metadata, as offered in (or checked against) a +/// `Hello`. +#[derive(Debug, Clone)] +pub struct Local { + pub node_name: String, + pub incarnation: Incarnation, + pub build_hash: u64, + pub meta: NodeMeta, +} + +/// The peer identity a successful handshake yields (what c7 feeds `node_up`). +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct Peer { + pub node_name: String, + pub incarnation: Incarnation, + pub meta: NodeMeta, +} + +/// Driver-supplied standing of the *offered* name at this node — knowledge +/// the pure machine cannot have (c6 owns the connection table and dial +/// set). One answer, in the responder's own precedence: an established +/// peer under that name outranks an in-flight dial to it. +#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] +pub enum PeerStanding { + /// Neither connected to nor dialing that name. + #[default] + Free, + /// An established peer already holds that name. + Claimed, + /// We have our own dial in flight to that name. + Dialing, +} + +/// Simultaneous-connect tie-break: does the connection dialed by +/// `dialer_name` survive against the reverse dial? +/// The rule (ratified 2026-08-14, a wire-protocol fact): the connection +/// dialed by the lexicographically **smaller** name survives. Both ends know +/// both names, so both compute the same verdict — which is why the losing +/// side may close silently instead of sending a reject. +pub fn dial_wins(dialer_name: &str, acceptor_name: &str) -> bool { + dialer_name < acceptor_name +} + +/// Dial side: emits `Hello` at construction, interprets the single response. +#[must_use] +#[derive(Debug)] +pub struct Initiator(()); + +/// What the dial side's response frame meant. +#[must_use] +#[derive(Debug, PartialEq, Eq)] +pub enum InitiatorOutcome { + Established(Peer), + Rejected(RejectReason), + /// Protocol violation before the ack — close. Carries the offending frame. + Failed(Frame), +} + +impl Initiator { + /// Start a dial-side handshake: the returned frame is the `Hello` to + /// send; the returned machine is the right to interpret the response. + pub fn new(local: &Local) -> (Self, Frame) { + let hello = Frame::Hello { + proto_version: PROTO_VERSION, + build_hash: local.build_hash, + node_name: local.node_name.clone(), + incarnation: local.incarnation, + meta: local.meta.clone(), + }; + (Initiator(()), hello) + } + + /// Interpret the response. The `HelloAck` carries no hash or version — + /// the responder already checked ours against its own, and equality is + /// symmetric, so a one-sided check is sound. + pub fn on_frame(self, frame: Frame) -> InitiatorOutcome { + match frame { + Frame::HelloAck { + node_name, + incarnation, + meta, + } => InitiatorOutcome::Established(Peer { + node_name, + incarnation, + meta, + }), + Frame::HelloReject { reason } => InitiatorOutcome::Rejected(reason), + other => InitiatorOutcome::Failed(other), + } + } +} + +/// Accept side: awaits exactly one `Hello`, answers or closes. +#[must_use] +#[derive(Debug)] +pub struct Responder { + local: Local, +} + +/// What to do with an inbound connection's first frame. +#[must_use] +#[derive(Debug, PartialEq, Eq)] +pub enum ResponderOutcome { + /// Send the ack; the connection is established. + Accepted { reply: Frame, peer: Peer }, + /// Send the reject, then close. + Rejected { reply: Frame, reason: RejectReason }, + /// Lost the simultaneous-connect tie-break: close silently, no frame. + TieBreakLoss, + /// Protocol violation before Hello — close, no reply. Carries the frame. + Failed(Frame), +} + +impl Responder { + pub fn new(local: Local) -> Self { + Responder { local } + } + + /// Judge the connection's first frame. Check order is proto → hash → + /// name → tie-break: validity before identity. `HelloReject` is the + /// cross-version compatibility anchor, so a version-mismatched peer + /// still gets one. + pub fn on_frame(self, frame: Frame, standing: PeerStanding) -> ResponderOutcome { + let Frame::Hello { + proto_version, + build_hash, + node_name, + incarnation, + meta, + } = frame + else { + return ResponderOutcome::Failed(frame); + }; + + let reject = |reason| ResponderOutcome::Rejected { + reply: Frame::HelloReject { reason }, + reason, + }; + + if proto_version != PROTO_VERSION { + return reject(RejectReason::ProtoVersion); + } + if build_hash != self.local.build_hash { + return reject(RejectReason::HashMismatch); + } + if node_name == self.local.node_name || standing == PeerStanding::Claimed { + return reject(RejectReason::NameTaken); + } + // Simultaneous connect: the inbound frame is the peer's dial. If our + // own in-flight dial wins instead, drop this one silently — the peer + // computes the same verdict (see `dial_wins`). + if standing == PeerStanding::Dialing && !dial_wins(&node_name, &self.local.node_name) { + return ResponderOutcome::TieBreakLoss; + } + + ResponderOutcome::Accepted { + reply: Frame::HelloAck { + node_name: self.local.node_name, + incarnation: self.local.incarnation, + meta: self.local.meta, + }, + peer: Peer { + node_name, + incarnation, + meta, + }, + } + } +} diff --git a/src/cluster/manager.rs b/src/cluster/manager.rs new file mode 100644 index 0000000..f92482d --- /dev/null +++ b/src/cluster/manager.rs @@ -0,0 +1,295 @@ +//! RFC 010 c6 — the cluster connection manager. +//! +//! One manager per runtime: the single registry of live peer connections and +//! the source of truth for whether a peer name is already claimed. The +//! accept/connect path registers each established connection here, handing +//! over its [`ConnHandle`] — **the manager owns connection lifetime**. A +//! connection lives as long as its table entry, so it neither outlives nor +//! dies with whichever actor happened to establish it. Registered actors are +//! also *monitored*, so the table self-heals on any exit path — a connection +//! that panics, is cancelled, or closes cleanly is removed without +//! cooperation from the dying actor. +//! +//! The manager also holds the **membership state** (c7a): `node_up` fires on +//! a successful registration and `node_down` on removal — they are derived +//! facts of the exact events this table already owns, so holding the view +//! here means no cross-actor race between "connection exists" and "node is +//! up". The consumer surface (event types, [`subscribe`], [`view`], +//! semantics) is [`membership`](crate::cluster::membership); no consumer +//! ever touches the table itself. +//! +//! The connector dial loop is c7b, built on top of both. + +use std::collections::HashMap; + +use crate::channel::Sender; +use crate::cluster::conn::ConnHandle; +use crate::cluster::handshake::{Peer, PeerStanding}; +use crate::cluster::membership::{NodeEvent, NodeInfo}; +use crate::cluster::remote::{bind_outbound, unbind_outbound}; +use crate::gen_server::{GenServer, GenServerCtx, GenServerName, Watcher}; +use crate::monitor::{monitor, Down}; +use crate::pg::NodeId; +use crate::pid::Pid; + +/// Well-known name of the singleton manager within a runtime. Connection +/// actors reach it by name rather than by a passed-around ref, so a restarted +/// manager is always found at the same key. +pub const MANAGER: GenServerName = GenServerName::new("smarm.cluster.manager"); + +/// One live connection's entry: the actor running it, the handle whose +/// lifetime *is* the connection's (see the module docs), and the peer's +/// membership identity (what `node_up` announced and `node_down` will name). +struct ConnEntry { + pid: Pid, + info: NodeInfo, + _handle: ConnHandle, +} + +/// The connection registry: peer name → the connection actor that owns that +/// peer's control connection. Plus the membership state layered on it (c7a): +/// subscribers, and the `(name, incarnation)` → [`NodeId`] memo. +pub struct Manager { + conns: HashMap, + /// In-flight dial intents: peer name -> the actor performing the dial. + /// Registered *before* connecting so a crossing inbound `Hello` sees it + /// ([`PeerStanding::Dialing`]); cleared the moment + /// the dial resolves, and — because the dialer is monitored — on the + /// dialer's death, so a panicking dial can never wedge the tie-break. + dials: HashMap, + /// Membership subscribers; a closed channel is pruned on the next emit. + subscribers: Vec>, + /// The [`NodeId`] memo: a reconnect at the same incarnation keeps its id, + /// a restart (new incarnation) allocates a fresh one. Grows one entry per + /// distinct `(name, incarnation)` ever seen — unbounded in principle, + /// bounded in practice by restarts actually happening. + ids: HashMap<(String, u32), NodeId>, + /// Next id to allocate. Starts at 1: id 0 is + /// [`DEFAULT_NODE_ID`](crate::pg::DEFAULT_NODE_ID), the local node. + next_id: u32, + watcher: Option>, +} + +impl Manager { + pub fn new() -> Self { + Manager { + conns: HashMap::new(), + dials: HashMap::new(), + subscribers: Vec::new(), + ids: HashMap::new(), + next_id: 1, + watcher: None, + } + } + + /// The memoized id for `(name, incarnation)` — see the field docs. + fn node_id(&mut self, name: &str, incarnation: u32) -> NodeId { + *self + .ids + .entry((name.to_string(), incarnation)) + .or_insert_with(|| { + let id = NodeId::new(self.next_id); + self.next_id += 1; + id + }) + } + + /// Deliver `event` to every live subscriber, pruning the dead: a closed + /// channel means the subscriber dropped its [`MembershipEvents`] + /// (crate::cluster::membership::MembershipEvents). + fn emit(&mut self, event: &NodeEvent) { + self.subscribers.retain(|tx| tx.send(event.clone()).is_ok()); + } +} + +impl Default for Manager { + fn default() -> Self { + Manager::new() + } +} + +/// Outcome of a [`Call::Register`]. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum Registered { + /// The name was free; this connection is now the peer of record. + Ok, + /// Another live connection already holds this name — the caller lost the + /// race (or is a duplicate) and must not run. + Duplicate, +} + +/// Requests to the manager. +pub enum Call { + /// The path claims its peer's name for a freshly-established connection, + /// handing the manager the actor's [`ConnHandle`] and the handshake's + /// [`Peer`] (the membership identity `node_up` announces). The manager + /// monitors `pid` and holds the handle for as long as the entry lives; a + /// [`Registered::Duplicate`] verdict drops the handle here, which stops + /// the refused actor. + Register { + peer: Peer, + pid: Pid, + handle: ConnHandle, + }, + /// Tear down the connection to `name`: the manager drops its handle, the + /// actor stops, and the monitor removes the entry. A no-op if no such + /// connection is live. + Disconnect { name: String }, + /// The current peer names, sorted. For observation and tests. + Peers, + /// A dialer declares an in-flight dial to `name` before connecting. The + /// pid is the dialing actor, monitored so the intent dies with it. + DialBegin { name: String, pid: Pid }, + /// The dial to `name` resolved (either way): drop the intent. A call, + /// not a cast, so the intent is provably gone before the dialer moves on. + DialEnd { name: String }, + /// The [`PeerStanding`] of an inbound `Hello` offering `peer_name` — the + /// accept path asks this between reading the frame and judging it. + Standing { peer_name: String }, + /// Subscribe `tx` to membership events, snapshot-then-stream: one + /// [`NodeEvent::NodeUp`] per live peer is queued into `tx` before this + /// call answers, so the stream is exact from its first event (handlers + /// are serialized — nothing interleaves with the snapshot). Use + /// [`subscribe`](crate::cluster::membership::subscribe). + Subscribe { tx: Sender }, + /// The current view: every live peer's [`NodeInfo`], unordered. Use + /// [`view`](crate::cluster::membership::view). + View, +} + +/// Replies from the manager. +#[derive(Debug)] +pub enum Reply { + Registered(Registered), + Disconnected, + Peers(Vec), + /// `false`: another dial to this name is already in flight — do not dial. + DialBegan(bool), + DialEnded, + Standing(PeerStanding), + Subscribed, + View(Vec), +} + +impl GenServer for Manager { + type Call = Call; + type Reply = Reply; + type Cast = (); + type Info = (); + type Timer = (); + + fn init(&mut self, ctx: &GenServerCtx) { + self.watcher = Some(ctx.watcher()); + } + + /// Manager shutdown drops every entry (and with it every ConnHandle); + /// the outbound table must not outlive the connections it names. + fn terminate(&mut self) { + for name in self.conns.keys() { + unbind_outbound(name); + } + } + + fn handle_call(&mut self, request: Call) -> Reply { + match request { + Call::Register { + peer, + pid, + mut handle, + } => { + if self.conns.contains_key(&peer.node_name) { + // `handle` drops here: the refused actor stops itself. + return Reply::Registered(Registered::Duplicate); + } + if let Some(w) = &self.watcher { + w.watch(monitor(pid)); + } + // The outbound table (c9) is maintained here, inside the same + // serialized handlers that own the connection's lifetime. + if let Some((frames, monitors)) = handle.take_outbound() { + bind_outbound(&peer.node_name, peer.incarnation, frames, monitors); + } + let info = NodeInfo { + node: self.node_id(&peer.node_name, peer.incarnation.get()), + name: peer.node_name.clone(), + incarnation: peer.incarnation, + meta: peer.meta, + }; + self.conns.insert( + peer.node_name, + ConnEntry { + pid, + info: info.clone(), + _handle: handle, + }, + ); + self.emit(&NodeEvent::NodeUp(info)); + Reply::Registered(Registered::Ok) + } + Call::Disconnect { name } => { + // Dropping the entry drops the handle, which stops the actor. + if let Some(entry) = self.conns.remove(&name) { + unbind_outbound(&name); + self.emit(&NodeEvent::NodeDown(entry.info)); + } + Reply::Disconnected + } + Call::Peers => { + let mut names: Vec = self.conns.keys().cloned().collect(); + names.sort(); + Reply::Peers(names) + } + Call::DialBegin { name, pid } => { + if self.dials.contains_key(&name) { + return Reply::DialBegan(false); + } + if let Some(w) = &self.watcher { + w.watch(monitor(pid)); + } + self.dials.insert(name, pid); + Reply::DialBegan(true) + } + Call::DialEnd { name } => { + self.dials.remove(&name); + Reply::DialEnded + } + Call::Standing { peer_name } => { + Reply::Standing(if self.conns.contains_key(&peer_name) { + PeerStanding::Claimed + } else if self.dials.contains_key(&peer_name) { + PeerStanding::Dialing + } else { + PeerStanding::Free + }) + } + Call::Subscribe { tx } => { + // The snapshot: queued before `tx` joins the list, and — the + // handlers being serialized — before any later event. + for entry in self.conns.values() { + let _ = tx.send(NodeEvent::NodeUp(entry.info.clone())); + } + self.subscribers.push(tx); + Reply::Subscribed + } + Call::View => Reply::View(self.conns.values().map(|e| e.info.clone()).collect()), + } + } + + fn handle_cast(&mut self, _request: ()) {} + + fn handle_down(&mut self, down: Down) { + let mut downs = Vec::new(); + self.conns.retain(|name, entry| { + let dead = entry.pid == down.pid; + if dead { + downs.push((name.clone(), entry.info.clone())); + } + !dead + }); + for (name, info) in downs { + unbind_outbound(&name); + self.emit(&NodeEvent::NodeDown(info)); + } + self.dials.retain(|_, pid| *pid != down.pid); + } +} diff --git a/src/cluster/membership.rs b/src/cluster/membership.rs new file mode 100644 index 0000000..49e178a --- /dev/null +++ b/src/cluster/membership.rs @@ -0,0 +1,97 @@ +//! RFC 010 c7a — membership: `node_up`/`node_down` events and the view. +//! +//! The membership *state* lives inside the [`manager`](crate::cluster::manager) +//! — `node_up` and `node_down` are derived facts of the exact events the +//! manager already owns (a successful registration; a reap or `Disconnect`), +//! so holding the view anywhere else would only add a cross-actor ordering +//! seam. This module is the consumer surface: the event and view types, and +//! the [`subscribe`]/[`view`] entry points. No consumer ever touches the +//! connection table (roadmap-binding, enforced by module privacy: the table +//! is a private field, and nothing here exposes names→pids). +//! +//! ## Subscription semantics (ratified 2026-08-15) +//! +//! [`subscribe`] is **snapshot-then-stream**: the returned receiver first +//! yields one [`NodeEvent::NodeUp`] per currently-live peer, then live events +//! as they happen. Because the manager is a `gen_server` (handlers are +//! serialized), the snapshot is exact — no event can interleave with it, and +//! per-subscriber ordering matches the manager's processing order. There is +//! no join-race for late subscribers and no separate "get, then diff" dance; +//! [`view`] exists for observation, not for synchronization. +//! +//! A dropped subscriber is pruned on the next emission (its channel reports +//! closed) — no monitor needed, the sender itself tells us. +//! +//! ## NodeId identity +//! +//! A [`NodeId`] is a compact **local alias for the wire identity** +//! `(node_name, incarnation)`, memoized by the manager: a reconnect blip at +//! the same incarnation keeps its id (down, then up, same id), while a +//! restart — a new incarnation — gets a fresh one, so a node's ghost and its +//! successor are always distinguishable. Ids are allocated from 1; +//! [`DEFAULT_NODE_ID`](crate::pg::DEFAULT_NODE_ID) (0) remains the local +//! node, per [`pg`](crate::pg)'s framing. + +use crate::channel::{channel, Receiver}; +use crate::cluster::envelope::NodeMeta; +use crate::cluster::manager::{Call, Reply, MANAGER}; +use crate::gen_server; +use crate::pg::{Incarnation, NodeId}; + +/// One live remote node, as the view and [`NodeEvent::NodeUp`] describe it. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct NodeInfo { + /// The local alias for `(name, incarnation)` — see the module docs. + pub node: NodeId, + /// The peer's claimed node name (handshake-verified). + pub name: String, + /// The peer's incarnation epoch, as offered in its `Hello`. + pub incarnation: Incarnation, + /// The peer's `Hello` metadata. + pub meta: NodeMeta, +} + +/// A membership change, as delivered to subscribers. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum NodeEvent { + /// A peer's control connection established and registered. + NodeUp(NodeInfo), + /// That peer's connection ended — reaped, commanded down, or the manager + /// itself shut down. Carries the same [`NodeInfo`] the corresponding + /// `NodeUp` delivered, so consumers need no id→name reverse map. + NodeDown(NodeInfo), +} + +/// A live membership subscription: the receiving end of the event stream +/// (the [`Monitor`](crate::monitor::Monitor) shape — read from [`rx`], drop +/// to unsubscribe). +/// +/// [rx]: MembershipEvents::rx +pub struct MembershipEvents { + /// The event stream: the snapshot's `NodeUp`s first, then live events. + /// Fold it into a `select` from a plain actor, or pipe it into a + /// `gen_server` via `with_info`. + pub rx: Receiver, +} + +/// Subscribe to membership events (snapshot-then-stream — see the module +/// docs). `None`: the manager is not running. Must be called from inside an +/// actor. +pub fn subscribe() -> Option { + let (tx, rx) = channel(); + match gen_server::call(MANAGER, Call::Subscribe { tx }) { + Ok(Reply::Subscribed) => Some(MembershipEvents { rx }), + _ => None, + } +} + +/// The current view: every live peer's [`NodeInfo`], unordered. For +/// observation and tests; consumers that need to *track* the view should +/// [`subscribe`] instead (the snapshot makes the stream self-sufficient). +/// `None`: the manager is not running. Must be called from inside an actor. +pub fn view() -> Option> { + match gen_server::call(MANAGER, Call::View) { + Ok(Reply::View(v)) => Some(v), + _ => None, + } +} diff --git a/src/cluster/pg.rs b/src/cluster/pg.rs new file mode 100644 index 0000000..6996237 --- /dev/null +++ b/src/cluster/pg.rs @@ -0,0 +1,546 @@ +//! RFC 010 c15 — distributed process groups (Phase 5). +//! +//! The Erlang `pg` shape (D18): every node's group store is the union of +//! its own local members and each peer's *announced* local members. There +//! is one **pg actor** per node — the c14 reaper grown up — and it is the +//! only writer of remote entries and the only sender of announcements: +//! +//! - **Origin owns its members.** Joins are local (`pg::join`), the eager +//! reaper is the liveness authority, and the origin announces every +//! change: `Join`/`Leave` incrementally to every up node, and a full +//! `Sync` of its local groups to a peer the moment that peer comes up +//! (`NodeUp`). Nobody monitors a remote member; a peer's `NodeDown` sweeps +//! every member it announced. +//! - **Transport is a pure consumer** of Phase 3/4: the exposed name +//! [`PG_NAME`] (`"pg"`) carrying [`PgMsg`] over postcard, sent with +//! [`remote::send`]. No new frame, no manager change. +//! - **No anti-entropy.** Per-origin ordering rides the single TCP link: +//! the actor sends `Sync` to a peer *before* it can send that peer any +//! `Join`/`Leave` (both from the same loop, over the same connection), and +//! a reconnect is a fresh `NodeUp` ⇒ fresh `Sync` replacing that peer's +//! set wholesale. +//! - **Local API unchanged.** `members`/`pick`/`dispatch` stay local-only +//! (`get_local_members`); a remote entry in the store carries the peer's +//! `NodeId` and never surfaces there. Cluster-wide reads are the new, +//! additive [`members_all`] over [`GroupMember`] (c16 adds `pick_any` / +//! `dispatch_any`). +//! +//! ## Ordering inside the node +//! +//! `pg::join`/`pg::leave` mutate the store on the caller's thread and then +//! *announce* to the actor's control inbox. Because the store op precedes the +//! announcement and the actor re-reads the store before broadcasting a +//! `Joined`, an announcement that has been overtaken (the member left or died +//! before the actor got to it) is dropped rather than advertised: the wire +//! never sees a `Join` for a member the origin no longer holds. `Leave` +//! broadcasts unconditionally — a spurious `Leave` is a no-op at the peer. +//! +//! Inbound: `NodeUp` is emitted by the manager on the accept/connect path, +//! *before* the peer's connection actor exists, so it is queued on the +//! membership stream before any frame from that peer can reach this inbox. +//! The actor still drains membership before it interprets a `PgMsg` whose +//! sender it does not know, and drops the message if the sender is still not +//! up (a ghost — its next `NodeUp` brings a `Sync`). + +use std::collections::HashMap; + +use crate::channel::{channel, select, Receiver, Selectable}; +use crate::cluster::expose::expose; +use crate::cluster::membership::{subscribe, MembershipEvents, NodeEvent, NodeInfo}; +use crate::cluster::remote::{ + self, local_identity, send_to_remote, RemoteName, RemotePid, ToRemoteError, +}; +use crate::monitor::Down; +use crate::pg::{ + live, member_for, reaper_inboxes, sweep_local_death, Incarnation, Member, Membership, PgEvent, +}; +use crate::pid::{assert_type, Addressable, Erased, Pid}; +use crate::registry::{register, send_to, SendError}; +use crate::scheduler::with_runtime; +use crate::Name; + +/// The exposed name every node's pg actor answers under. +pub const PG_NAME: Name = Name::new("pg"); + +/// The pg wire protocol. Every variant is origin-authored: `from` / the +/// pid's node is the node whose local members are being described. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum PgMsg { + /// The origin's complete local membership, sent to a peer on `NodeUp`. + /// Replaces whatever the receiver held for that origin. + Sync { + from: String, + groups: Vec<(String, Vec>)>, + }, + /// The origin added `pid` (its own) to `group`. + Join { + group: String, + pid: RemotePid, + }, + /// The origin removed `pid` from `group` — voluntary leave or death. + Leave { + group: String, + pid: RemotePid, + }, +} + +// Hand-rolled serde (the crate carries no serde-derive), as a 3-tuple with a +// leading tag: (0, from, groups) | (1, group, pid) | (2, group, pid). +impl serde::Serialize for PgMsg { + fn serialize(&self, s: S) -> Result { + use serde::ser::SerializeTuple; + let mut t = s.serialize_tuple(3)?; + match self { + PgMsg::Sync { from, groups } => { + t.serialize_element(&0u8)?; + t.serialize_element(from)?; + t.serialize_element(groups)?; + } + PgMsg::Join { group, pid } => { + t.serialize_element(&1u8)?; + t.serialize_element(group)?; + t.serialize_element(pid)?; + } + PgMsg::Leave { group, pid } => { + t.serialize_element(&2u8)?; + t.serialize_element(group)?; + t.serialize_element(pid)?; + } + } + t.end() + } +} + +impl<'de> serde::Deserialize<'de> for PgMsg { + fn deserialize>(d: D) -> Result { + struct V; + impl<'de> serde::de::Visitor<'de> for V { + type Value = PgMsg; + fn expecting(&self, f: &mut std::fmt::Formatter) -> std::fmt::Result { + f.write_str("a pg message tuple") + } + fn visit_seq>( + self, + mut seq: A, + ) -> Result { + use serde::de::Error; + let tag: u8 = seq + .next_element()? + .ok_or_else(|| A::Error::custom("pg: missing tag"))?; + let text: String = seq + .next_element()? + .ok_or_else(|| A::Error::custom("pg: missing name"))?; + match tag { + 0 => { + let groups = seq + .next_element()? + .ok_or_else(|| A::Error::custom("pg: missing groups"))?; + Ok(PgMsg::Sync { from: text, groups }) + } + 1 | 2 => { + let pid = seq + .next_element()? + .ok_or_else(|| A::Error::custom("pg: missing pid"))?; + Ok(if tag == 1 { + PgMsg::Join { group: text, pid } + } else { + PgMsg::Leave { group: text, pid } + }) + } + t => Err(A::Error::custom(format!("pg: unknown tag {t}"))), + } + } + } + d.deserialize_tuple(3, V) + } +} + +/// A member of a group as the cluster sees it: on this node (a plain +/// [`Pid`], sendable locally) or on a peer (a [`RemotePid`], sendable via +/// [`send_to_remote`](remote::send_to_remote)). `Pid` cannot hold a remote +/// (D14), hence the two-variant shape. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum GroupMember { + Local(Pid), + Remote(RemotePid), +} + +/// Every member of `group` cluster-wide, in the store's order: local members +/// filtered by the same liveness backstop as [`members`](crate::pg::members), +/// remote members exactly as their origins last announced them. Must run +/// inside [`run`](crate::run). +pub fn members_all(group: &str) -> Vec { + with_runtime(|inner| { + let me = inner.node_id; + let pg = inner.process_groups.lock(); + pg.all_of(group) + .into_iter() + .filter_map(|m| { + if m.node == me { + live(inner, m.pid).then_some(GroupMember::Local(m.pid)) + } else { + // A remote entry always has its node's name recorded + // (they land under the same lock); a missing one is a + // node already swept, so it hides rather than misnames. + pg.node_name(m.node).map(|name| { + GroupMember::Remote(RemotePid::from_parts( + name, + m.incarnation, + m.pid.index(), + m.pid.generation(), + )) + }) + } + }) + .collect() + }) +} + +/// One member of `group` cluster-wide, or `None` if it has none: the first +/// entry in the store's order (this node's members in join order first when +/// they joined first — the same stateless first-live scan as +/// [`pick`](crate::pg::pick), extended over the peers' announced members). +/// Must run inside [`run`](crate::run). +pub fn pick_any(group: &str) -> Option { + members_all(group).into_iter().next() +} + +/// Why [`dispatch_any`] handed `msg` back. +#[derive(Debug)] +pub enum DispatchAnyError { + /// The group has no member anywhere. + NoMember(M), + /// The pick was local and the local typed send failed. + Local(SendError), + /// The pick was remote and the remote send failed at this node. + Remote(ToRemoteError), +} + +impl DispatchAnyError { + /// The undelivered message. + pub fn into_inner(self) -> M { + match self { + DispatchAnyError::NoMember(m) => m, + DispatchAnyError::Local(e) => e.into_inner(), + DispatchAnyError::Remote(e) => e.into_inner(), + } + } +} + +/// [`pick_any`] and send in one step, returning the member reached: a local +/// pick goes through [`send_to`], a remote one through [`send_to_remote`] +/// (so `Ok` for a remote member means "handed to the connection", RFC 010 +/// §3). Homogeneous pool assumed, as for [`dispatch`](crate::pg::dispatch); +/// a wrong `A` degrades to a clean error at the target, never a misroute. +/// Must run inside [`run`](crate::run). +pub fn dispatch_any(group: &str, msg: A::Msg) -> Result> +where + A: Addressable, + A::Msg: serde::Serialize, +{ + match pick_any(group) { + None => Err(DispatchAnyError::NoMember(msg)), + Some(GroupMember::Local(pid)) => send_to(assert_type::(pid), msg) + .map(|()| GroupMember::Local(pid)) + .map_err(DispatchAnyError::Local), + Some(GroupMember::Remote(rp)) => send_to_remote(rp.clone().assert_type::(), msg) + .map(|()| GroupMember::Remote(rp)) + .map_err(DispatchAnyError::Remote), + } +} + +/// Attach the pg actor to the running cluster. Called once by +/// `cluster::start` after the manager is up and the local identity is set; +/// spawns the actor if this run has not joined anything yet. +pub(crate) fn attach_cluster() { + let _ = reaper_inboxes().ctl.send(PgEvent::Attach); +} + +/// The attached half of the actor's state: who is up (by name) and the +/// membership stream. +struct Attached { + events: MembershipEvents, + peers: HashMap, + /// This node's wire identity: `Sync`'s `from`, and the stamp on every + /// pid we ship (attach requires it, so no `None` path exists here). + me: String, + incarnation: Incarnation, +} + +/// The pg actor: the c14 reaper (`deaths`), the local API's announcements +/// (`ctl`), and — once attached — the membership stream and the exposed +/// `"pg"` inbox, all in one drain-then-select loop. `deaths`/`ctl` closing +/// is the run tearing down; the membership stream closing is the manager +/// gone (detach, keep reaping). +pub(crate) fn actor(deaths: Receiver, ctl: Receiver) { + let (pg_tx, pg_rx) = channel::(); + let mut cl: Option = None; + loop { + loop { + match deaths.try_recv() { + Ok(Some(down)) => on_death(cl.as_ref(), down.pid), + Ok(None) => break, + Err(_) => return, + } + } + loop { + match ctl.try_recv() { + Ok(Some(PgEvent::Attach)) => { + // Own the name BEFORE subscribing (which yields to the + // manager): a peer's first frame must find "pg" exposed + // and resolvable, or it is dropped. Idempotent for a + // re-attach: same actor, same channel (the registry + // refuses a *second* live one). + let _ = register(PG_NAME, pg_tx.clone()); + expose(PG_NAME); + if let Some(a) = attach() { + cl = Some(a); + } + } + Ok(Some(PgEvent::Joined { group, pid })) => on_joined(cl.as_ref(), &group, pid), + Ok(Some(PgEvent::Left { group, pid })) => on_left(cl.as_ref(), &group, pid), + Ok(None) => break, + Err(_) => return, + } + } + if let Some(a) = cl.as_mut() { + if !drain_events(a) { + cl = None; + continue; + } + loop { + match pg_rx.try_recv() { + Ok(Some(msg)) => on_msg(a, msg), + Ok(None) => break, + Err(_) => return, // our own inbox: only on teardown + } + } + } + // Wait. Control first (attach/teardown must be prompt), then deaths, + // then the cluster arms. + let mut arms: Vec<&dyn Selectable> = vec![&ctl, &deaths]; + if let Some(a) = cl.as_ref() { + arms.push(&a.events.rx); + arms.push(&pg_rx); + } + let _ = select(&arms); + } +} + +fn attach() -> Option { + let events = subscribe()?; + let (me, incarnation) = local_identity()?; + Some(Attached { + events, + peers: HashMap::new(), + me, + incarnation, + }) +} + +/// Fold pending membership events: `NodeUp` ⇒ record + `Sync` that peer; +/// `NodeDown` ⇒ sweep every member it announced. `false` when the stream +/// has closed. +fn drain_events(a: &mut Attached) -> bool { + loop { + match a.events.rx.try_recv() { + Ok(Some(NodeEvent::NodeUp(info))) => { + let name = info.name.clone(); + a.peers.insert(name.clone(), info); + // Snapshot under the store lock, then stamp wire pids + // outside it (`from_local` marks watchable under the slot's + // cold lock — Leaf-on-Leaf nesting is asserted). + let local: Vec<(String, Vec)> = + with_runtime(|inner| inner.process_groups.lock().groups_on(inner.node_id)); + let groups = local + .into_iter() + .map(|(g, pids)| (g, pids.into_iter().map(|p| wire(a, p)).collect())) + .collect(); + let msg = PgMsg::Sync { + from: a.me.clone(), + groups, + }; + let _ = remote::send(RemoteName::new(name, PG_NAME), msg); + } + Ok(Some(NodeEvent::NodeDown(info))) => { + a.peers.remove(&info.name); + with_runtime(|inner| { + let mut pg = inner.process_groups.lock(); + pg.remove_where(|m| m.node == info.node); + pg.forget_node_name(info.node); + }); + } + Ok(None) => return true, + Err(_) => return false, + } + } +} + +/// Send `msg` to every up peer. `NotConnected` is ignored: that peer's +/// `NodeDown` is on its way and its next `NodeUp` gets a `Sync`. +fn broadcast(a: &Attached, msg: PgMsg) { + for name in a.peers.keys() { + let _ = remote::send(RemoteName::new(name.clone(), PG_NAME), msg.clone()); + } +} + +fn on_death(a: Option<&Attached>, pid: Pid) { + let evicted = sweep_local_death(pid); + if let Some(a) = a { + for (group, ms) in evicted { + broadcast( + a, + PgMsg::Leave { + group, + pid: wire(a, ms.member.pid), + }, + ); + } + } +} + +fn on_joined(a: Option<&Attached>, group: &str, pid: Pid) { + let Some(a) = a else { return }; + // Re-check: a leave/death may have overtaken the announcement. + let still = with_runtime(|inner| { + let m = member_for(inner, pid); + inner.process_groups.lock().contains(group, &m) + }); + if still { + broadcast( + a, + PgMsg::Join { + group: group.to_owned(), + pid: wire(a, pid), + }, + ); + } +} + +fn on_left(a: Option<&Attached>, group: &str, pid: Pid) { + let Some(a) = a else { return }; + broadcast( + a, + PgMsg::Leave { + group: group.to_owned(), + pid: wire(a, pid), + }, + ); +} + +/// The wire form of a local member pid, stamped with the identity the +/// actor was attached with (marks watchable, like `from_local`). +fn wire(a: &Attached, pid: Pid) -> RemotePid { + RemotePid::from_local_at(pid, a.me.clone(), a.incarnation) +} + +/// The named origin's `NodeInfo`, if it is up. A second look at the +/// membership stream covers a `NodeUp` that landed after this loop +/// iteration's drain; anything still unknown is a ghost and is dropped. +fn origin(a: &mut Attached, name: &str) -> Option { + if let Some(i) = a.peers.get(name) { + return Some(i.clone()); + } + drain_events(a); + a.peers.get(name).cloned() +} + +/// `origin`, additionally requiring `pid` to be stamped with the origin's +/// current incarnation — a pid from a previous life of that node is a ghost. +fn origin_of(a: &mut Attached, pid: &RemotePid) -> Option { + origin(a, pid.node()).filter(|i| i.incarnation == pid.incarnation()) +} + +fn remote_membership(origin: &NodeInfo, pid: &RemotePid) -> Membership { + Membership { + member: Member { + node: origin.node, + incarnation: origin.incarnation, + pid: Pid::new(pid.index(), pid.generation()), + }, + monitor: None, + } +} + +fn on_msg(a: &mut Attached, msg: PgMsg) { + match msg { + PgMsg::Sync { from, groups } => { + let Some(info) = origin(a, &from) else { return }; + with_runtime(|inner| { + let mut pg = inner.process_groups.lock(); + pg.remove_where(|m| m.node == info.node); + pg.set_node_name(info.node, info.name.clone()); + for (group, pids) in &groups { + // Origin-authored: only its own current-incarnation pids. + for p in pids + .iter() + .filter(|p| p.node() == from && p.incarnation() == info.incarnation) + { + pg.join(group, remote_membership(&info, p)); + } + } + }); + } + PgMsg::Join { group, pid } => { + let Some(info) = origin_of(a, &pid) else { + return; + }; + with_runtime(|inner| { + let mut pg = inner.process_groups.lock(); + pg.set_node_name(info.node, info.name.clone()); + pg.join(&group, remote_membership(&info, &pid)); + }); + } + PgMsg::Leave { group, pid } => { + let Some(info) = origin_of(a, &pid) else { + return; + }; + with_runtime(|inner| { + let ms = remote_membership(&info, &pid); + inner.process_groups.lock().leave(&group, ms.member); + }); + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::cluster::envelope::{decode_payload, encode_payload}; + + #[test] + fn pg_msg_roundtrips_every_variant() { + let p = RemotePid::::from_parts("a", Incarnation::new(9), 3, 1); + for m in [ + PgMsg::Sync { + from: "a".into(), + groups: vec![ + ("g".into(), vec![p.clone(), p.clone()]), + ("h".into(), vec![]), + ], + }, + PgMsg::Sync { + from: "a".into(), + groups: vec![], + }, + PgMsg::Join { + group: "g".into(), + pid: p.clone(), + }, + PgMsg::Leave { + group: "g".into(), + pid: p.clone(), + }, + ] { + let bytes = encode_payload(&m).unwrap(); + let back: PgMsg = decode_payload(&bytes).unwrap(); + assert_eq!(back, m); + } + } + + #[test] + fn pg_msg_rejects_unknown_tag() { + let bytes = encode_payload(&(7u8, "x", 0u32)).unwrap(); + assert!(decode_payload::(&bytes).is_err()); + } +} diff --git a/src/cluster/remote.rs b/src/cluster/remote.rs new file mode 100644 index 0000000..8b74c43 --- /dev/null +++ b/src/cluster/remote.rs @@ -0,0 +1,835 @@ +//! RFC 010 c9 — remote `Name` sends: the outbound path and the single +//! inbound name-resolution seam. +//! +//! ## Outbound (D13, ratified 2026-08-15) +//! +//! A module-private table `node name → Sender` — one dedicated +//! outbound channel per live connection, populated and torn down by the +//! manager inside the same serialized handlers that own the connection's +//! lifetime (Register / Disconnect / reap), living on `RuntimeInner` beside +//! the exposure state. [`send`] is one leaf-lock lookup + one channel send: +//! no gen_server on the data plane, no published channel anyone holding a +//! pid could inject raw frames into. `Ok(())` means **handed to the +//! connection's inbox** — local knowledge only, exactly the BEAM contract +//! (RFC §3): a missing entry or a closed channel is +//! [`RemoteSendError::NotConnected`]; delivery confirmation is the monitor's +//! job (c11). The entry-present/actor-dying-mid-send window is *honest* +//! under that contract, not a bug. +//! +//! The outbound sender is deliberately **separate from the conn actor's +//! command channel**: if it were a clone of `cmd_tx`, the manager dropping +//! its `ConnHandle` would no longer close that channel and connection +//! lifetime would leak to whoever holds a sender — a D9 violation. +//! +//! Buffering is unbounded toward a slow peer (the BEAM `busy_dist_port` +//! shape); backpressure is out of c9's scope and noted here rather than +//! silently absent. +//! +//! ## Inbound — the ONE resolution seam (RFC v2) +//! +//! Every wire-name → local-pid resolution goes through [`deliver_named`], +//! and nothing else: the conn actor hands it the three fields of a +//! `SendNamed` and gets back a verdict. It checks the exposed set first (an +//! unexposed name is unreachable — the gun's safety), then the type hash +//! against what the name was exposed with, then resolves the name through +//! the registry and delivers via c8's [`decode_deliver`]. When an owned-name +//! table lands beside the `&'static str` registry, it slots in here without +//! touching call sites. Module privacy enforces the funnel: the exposed and +//! outbound tables are `pub(crate)`, and no other module resolves names for +//! the wire. +//! +//! Refusals are silent to the sender by design (§3: send failure reflects +//! local knowledge only); they are observable locally as the returned +//! [`InboundVerdict`], which the conn actor may log or count. +//! +//! ## Pids (c10, D14) +//! +//! [`RemotePid`] = `(node_name, incarnation, index, generation)` + +//! phantom — identity-bound, dead when that incarnation dies, never +//! redirects. The node travels as its **name** (a global identifier, so a pid +//! forwarded through a third node needs no re-mapping); NodeId is a local +//! alias and never crosses. A local `Pid` serializes *as* a `RemotePid` +//! stamped from the ambient [local identity](set_local_identity); a +//! `RemotePid` deserializes into `Pid` only when it names this node (the +//! collapse), else it is a decode error — fields that may hold a pid from +//! anywhere are typed `RemotePid`. +//! +//! [`send_to_remote`] is the pid-targeted send. A self-node pid short- +//! circuits to the local typed send with the message object itself — no +//! encode, no frame (zero-copy-equivalent). Otherwise the outbound table +//! (widened to carry each node's **current incarnation**) does the RFC v2 §3 +//! check at the send site: a pid of a dead incarnation is +//! [`ToRemoteError::DeadIncarnation`] and no frame is emitted. Inbound +//! `Send` frames are delivered by index/generation through c8's +//! [`decode_deliver`]: the target actor's published channel for the exposed +//! type is the only route (the reply-to path requires +//! [`expose_type`](crate::cluster::expose::expose_type) at the receiver). + +use std::cell::Cell; +use std::collections::HashMap; +use std::marker::PhantomData; + +use crate::channel::{channel, Receiver, RecvError, Selectable, Sender}; +use crate::cluster::envelope::{encode_payload, Frame, PayloadError, RemoteDownReason}; +use crate::cluster::expose::{decode_deliver, exposed_hash, type_hash, DeliverError}; +use crate::monitor::{demonitor, monitor, Monitor, MonitorId}; +use crate::pg::Incarnation; +use crate::pid::{Addressable, Erased, Name, Pid}; +use crate::registry::{send_to, whereis, SendError}; +use crate::scheduler::with_runtime; + +/// A name on a specific remote node: `(node_name, Name)`. Sendable via +/// [`send`]; typed, so the payload is `M` and the wire hash is +/// [`type_hash::()`](type_hash). +pub struct RemoteName { + node: String, + name: Name, + _marker: PhantomData M>, +} + +impl RemoteName { + pub fn new(node: impl Into, name: Name) -> Self { + RemoteName { + node: node.into(), + name, + _marker: PhantomData, + } + } + pub fn node(&self) -> &str { + &self.node + } + pub fn name(&self) -> Name { + self.name + } +} + +impl Clone for RemoteName { + fn clone(&self) -> Self { + RemoteName { + node: self.node.clone(), + name: self.name, + _marker: PhantomData, + } + } +} + +impl std::fmt::Debug for RemoteName { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "{}@{}", self.name.as_str(), self.node) + } +} + +/// Why a remote send did not leave this node. Local knowledge only. +#[derive(Debug)] +pub enum RemoteSendError { + /// No live connection to that node right now (never connected, or gone + /// and not yet re-dialed). The message is handed back. + NotConnected(M), + /// The payload did not serialize. + Encode(M, PayloadError), +} + +impl RemoteSendError { + pub fn into_inner(self) -> M { + match self { + RemoteSendError::NotConnected(m) | RemoteSendError::Encode(m, _) => m, + } + } +} + +/// The outbound table, one per runtime (a `RuntimeInner` field): per live +/// node, its current incarnation (the RFC v2 §3 send-site check) and the +/// connection's dedicated outbound sender. Plus this node's own wire +/// identity, which pid serialization stamps. +pub(crate) struct Outbound { + by_node: HashMap, + local: Option<(String, Incarnation)>, +} + +/// One live connection as the outbound path sees it: the peer's current +/// incarnation and the two inboxes of its connection actor — frames (c9) +/// and monitor bookkeeping (c12, [`MonCmd`]). +pub(crate) struct Route { + incarnation: Incarnation, + frames: Sender, + monitors: Sender, +} + +impl Outbound { + pub(crate) fn new() -> Self { + Outbound { + by_node: HashMap::new(), + local: None, + } + } +} + +/// Set this node's wire identity — what serialized pids are stamped with +/// and what a `RemotePid` must name to collapse. `cluster::start` sets it; +/// exposed for local tests. Must run inside [`run`](crate::run). +pub fn set_local_identity(node: &str, incarnation: Incarnation) { + with_runtime(|inner| { + inner.outbound.lock().local = Some((node.to_string(), incarnation)); + }); +} + +/// This node's wire identity, if set. Must run inside [`run`](crate::run). +pub fn local_identity() -> Option<(String, Incarnation)> { + with_runtime(|inner| inner.outbound.lock().local.clone()) +} + +/// Manager-only: bind `node`'s outbound channels at `incarnation`. Called +/// inside `Register`. +pub(crate) fn bind_outbound( + node: &str, + incarnation: Incarnation, + frames: Sender, + monitors: Sender, +) { + with_runtime(|inner| { + inner.outbound.lock().by_node.insert( + node.to_string(), + Route { + incarnation, + frames, + monitors, + }, + ); + }); +} + +/// Test probe: bind an arbitrary sender as `node`'s outbound so a test can +/// assert what frames leave — or don't. Same table, same lookup as the real +/// path (this is how "no frame emitted" is asserted at the frame level). +/// Frames only: there is no connection actor behind a probe, so a +/// [`monitor_remote`] against a probed node reports `Disconnected`. +pub fn bind_outbound_probe(node: &str, incarnation: Incarnation, tx: Sender) { + drop(bind_outbound_probe_with_monitors(node, incarnation, tx)); +} + +/// The monitor half of a probed node's inbox: opaque, held only to be +/// dropped. See [`bind_outbound_probe_with_monitors`]. +pub struct MonitorInbox { + _rx: Receiver, +} + +/// Test probe: like [`bind_outbound_probe`], but the monitor-command +/// receiver is handed back instead of dropped, so a test can stage the +/// c13 drain gap — a `Monitor` command that reached the connection's inbox +/// and dies unread when the inbox is dropped. While the inbox lives, +/// [`monitor_remote`] against the probed node is simply in flight. +pub fn bind_outbound_probe_with_monitors( + node: &str, + incarnation: Incarnation, + tx: Sender, +) -> MonitorInbox { + let (mon_tx, mon_rx) = channel(); + bind_outbound(node, incarnation, tx, mon_tx); + MonitorInbox { _rx: mon_rx } +} + +/// Manager-only: unbind `node`'s outbound channel. Called on `Disconnect`, +/// reap, and manager shutdown. Dropping the sender is what closes the conn +/// actor's outbound arm — but that arm's closure is NOT a stop signal (the +/// cmd channel is, per D9); the actor simply stops selecting on it. +pub(crate) fn unbind_outbound(node: &str) { + with_runtime(|inner| { + inner.outbound.lock().by_node.remove(node); + }); +} + +/// Send `msg` to `target`. `Ok(())` = handed to the connection's inbox, and +/// nothing more — see the module docs. Must run inside +/// [`run`](crate::run). +pub fn send(target: RemoteName, msg: M) -> Result<(), RemoteSendError> +where + M: serde::Serialize + Send + 'static, +{ + let payload = match encode_payload(&msg) { + Ok(p) => p, + Err(e) => return Err(RemoteSendError::Encode(msg, e)), + }; + let frame = Frame::SendNamed { + name: target.name.as_str().to_string(), + type_hash: type_hash::(), + payload, + }; + match hand_to_connection(&target.node, frame) { + Ok(()) => Ok(()), + Err(NotConnected) => Err(RemoteSendError::NotConnected(msg)), + } +} + +/// No live connection to the named node — the payload-free form of +/// [`RemoteSendError::NotConnected`], for the raw path. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct NotConnected; + +/// The untyped escape hatch: send pre-encoded `payload` under an explicit +/// `type_hash`. Exists so tests (and future codecs) can put deliberately +/// wrong frames on the wire; the typed [`send`] cannot express a hash/type +/// mismatch, by design. Same `Ok` semantics as [`send`]. +pub fn send_remote_raw( + node: &str, + name: &str, + type_hash: u64, + payload: &[u8], +) -> Result<(), NotConnected> { + hand_to_connection( + node, + Frame::SendNamed { + name: name.to_string(), + type_hash, + payload: payload.to_vec(), + }, + ) +} + +/// One lookup, one send. Clone the sender out under the lock and send +/// outside it (a channel send can unpark the conn actor). +fn hand_to_connection(node: &str, frame: Frame) -> Result<(), NotConnected> { + let tx = with_runtime(|inner| { + inner + .outbound + .lock() + .by_node + .get(node) + .map(|r| r.frames.clone()) + }); + match tx { + Some(tx) => tx.send(frame).map_err(|_| NotConnected), + None => Err(NotConnected), + } +} + +// ---- pids --------------------------------------------------------------- + +/// A pid on some node: `(node_name, incarnation, index, generation)` plus +/// the actor type. See the module docs. Serializes as a 4-tuple. +pub struct RemotePid { + node: String, + incarnation: Incarnation, + index: u32, + generation: u32, + _marker: PhantomData A>, +} + +impl RemotePid { + /// Build from raw parts (tests, and codecs re-hydrating a pid). + pub fn from_parts( + node: impl Into, + incarnation: Incarnation, + index: u32, + generation: u32, + ) -> Self { + RemotePid { + node: node.into(), + incarnation, + index, + generation, + _marker: PhantomData, + } + } + + /// The wire form of a local pid, stamped with this node's identity, and + /// **marked watchable** — asking for the wire form *is* the intent to + /// ship the pid, so this is the same D12 set-site as `Pid::serialize` + /// (c12 made it explicit: a peer may monitor exactly the pids that + /// crossed, and a pid handed out via `from_local` in a hand-built reply + /// has crossed). Must run inside [`run`](crate::run). + /// + /// `None` when this runtime has no wire identity (no `cluster::start`, + /// no [`set_local_identity`]): such a pid cannot name a node, and a + /// stamped `("", 0)` would be dropped by every peer with no signal. + /// The pid is not marked watchable in that case either. + pub fn from_local(pid: Pid) -> Option { + let (node, incarnation) = local_identity()?; + Some(Self::from_local_at(pid, node, incarnation)) + } + + /// `from_local` with the identity supplied by the caller — for a holder + /// that already carries the node's identity (the pg actor) and must not + /// have a `None` path. Marks watchable like `from_local`. + pub(crate) fn from_local_at(pid: Pid, node: String, incarnation: Incarnation) -> Self { + crate::monitor::mark_watchable(pid); + RemotePid::from_parts(node, incarnation, pid.index(), pid.generation()) + } + + /// The collapse: `Some(local pid)` iff this pid names this very node + /// (name and incarnation). Must run inside [`run`](crate::run). + pub fn local(&self) -> Option> { + let (n, i) = local_identity()?; + (n == self.node && i == self.incarnation) + .then(|| crate::pid::assert_type::(Pid::new(self.index, self.generation))) + } + + /// Drop the actor type: the untyped `RemotePid`, the form + /// [`RemoteDown`] and [`RemoteMonitor`] carry (mirrors [`Pid::erase`]). + pub fn erase(self) -> RemotePid { + RemotePid::from_parts(self.node, self.incarnation, self.index, self.generation) + } + + /// Re-type an erased pid as `RemotePid` — the unchecked mirror of + /// `pid::assert_type`, with the same degradation: a wrong `B` means the + /// target refuses the payload's hash (never a misroute). + pub(crate) fn assert_type(self) -> RemotePid { + RemotePid::from_parts(self.node, self.incarnation, self.index, self.generation) + } + + pub fn node(&self) -> &str { + &self.node + } + pub fn incarnation(&self) -> Incarnation { + self.incarnation + } + pub fn index(&self) -> u32 { + self.index + } + pub fn generation(&self) -> u32 { + self.generation + } +} + +impl Clone for RemotePid { + fn clone(&self) -> Self { + RemotePid::from_parts( + self.node.clone(), + self.incarnation, + self.index, + self.generation, + ) + } +} +impl PartialEq for RemotePid { + fn eq(&self, o: &Self) -> bool { + self.node == o.node + && self.incarnation == o.incarnation + && self.index == o.index + && self.generation == o.generation + } +} +impl Eq for RemotePid {} +impl std::fmt::Debug for RemotePid { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!( + f, + "<{}.{}@{}#{}>", + self.index, + self.generation, + self.node, + self.incarnation.get() + ) + } +} + +impl serde::Serialize for RemotePid { + fn serialize(&self, s: S) -> Result { + ( + self.node.as_str(), + self.incarnation.get(), + self.index, + self.generation, + ) + .serialize(s) + } +} +impl<'de, A> serde::Deserialize<'de> for RemotePid { + fn deserialize>(d: D) -> Result { + let (node, inc, index, generation) = <(String, u32, u32, u32)>::deserialize(d)?; + Ok(RemotePid::from_parts( + node, + Incarnation::new(inc), + index, + generation, + )) + } +} + +/// Why a pid-targeted send did not leave this node. Local knowledge only. +#[derive(Debug)] +pub enum ToRemoteError { + /// No live connection to the pid's node. + NotConnected(M), + /// The pid's incarnation is not that node's current one (RFC v2 §3): the + /// actor died with its incarnation. Detected at the send site; no frame. + DeadIncarnation(M), + /// The payload did not serialize. + Encode(M, PayloadError), + /// The pid collapsed to a local one and the local typed send failed. + Local(SendError), +} + +impl ToRemoteError { + /// The undelivered message. + pub fn into_inner(self) -> M { + match self { + ToRemoteError::NotConnected(m) + | ToRemoteError::DeadIncarnation(m) + | ToRemoteError::Encode(m, _) => m, + ToRemoteError::Local(e) => e.into_inner(), + } + } +} + +/// Send `msg` to a pid, wherever it lives. Self-node pids short-circuit to +/// the local typed send with `msg` itself (no encode, no frame); others go +/// out as a `Send` frame after the incarnation check. `Ok(())` for a remote +/// target = handed to the connection's inbox. Must run inside +/// [`run`](crate::run). +pub fn send_to_remote(target: RemotePid, msg: A::Msg) -> Result<(), ToRemoteError> +where + A: Addressable, + A::Msg: serde::Serialize, +{ + if let Some(local) = target.local() { + return send_to(local, msg).map_err(ToRemoteError::Local); + } + let route = with_runtime(|inner| { + inner + .outbound + .lock() + .by_node + .get(&target.node) + .map(|r| (r.incarnation, r.frames.clone())) + }); + let (current, tx) = match route { + Some(r) => r, + None => return Err(ToRemoteError::NotConnected(msg)), + }; + if current != target.incarnation { + return Err(ToRemoteError::DeadIncarnation(msg)); + } + let payload = match encode_payload(&msg) { + Ok(p) => p, + Err(e) => return Err(ToRemoteError::Encode(msg, e)), + }; + let frame = Frame::Send { + index: target.index, + generation: target.generation, + type_hash: type_hash::(), + payload, + }; + tx.send(frame).map_err(|_| ToRemoteError::NotConnected(msg)) +} + +/// The inbound `Send` seam: deliver `payload` under `type_hash` to the local +/// actor `(index, generation)`. Node and incarnation are implicit in the +/// connection (bound at handshake) — the frame carries only the slot +/// identity. Delivery goes through c8's decoder table, so only types the +/// receiver has [`expose_type`](crate::cluster::expose::expose_type)d (or +/// exposed by name) can land; anything else is refused, never misrouted. +pub fn deliver_to_pid( + index: u32, + generation: u32, + type_hash: u64, + payload: &[u8], +) -> InboundVerdict { + let pid = Pid::new(index, generation); + match decode_deliver(type_hash, pid, payload) { + Ok(()) => InboundVerdict::Delivered, + Err(e) => InboundVerdict::Refused(e), + } +} + +/// What the inbound seam did with a `SendNamed`. Local observability only; +/// nothing goes back on the wire (RFC §3). +#[derive(Debug)] +pub enum InboundVerdict { + /// Decoded and handed to the name's holder. + Delivered, + /// The name is not in this node's exposed set. + NotExposed, + /// The frame's hash is not the hash the name was exposed with. + HashMismatch { expected: u64, got: u64 }, + /// Exposed, but no live holder right now (unbound, or its holder died + /// and the binding is being pruned). + Unresolved, + /// Resolved, but the delivery half refused it (decode failure, or the + /// holder's channel does not accept the exposed type — a local + /// re-registration under a different type; never a misroute). + Refused(DeliverError), +} + +impl InboundVerdict { + /// A short static label for tracing/counting (`smarm-trace` records one + /// `ClusterInbound` event per frame with it). + pub fn label(&self) -> &'static str { + match self { + InboundVerdict::Delivered => "delivered", + InboundVerdict::NotExposed => "not_exposed", + InboundVerdict::HashMismatch { .. } => "hash_mismatch", + InboundVerdict::Unresolved => "unresolved", + InboundVerdict::Refused(_) => "refused", + } + } +} + +/// THE inbound resolution seam: exposed-set check → hash check → registry +/// resolution → c8 delivery. See the module docs. Must run inside +/// [`run`](crate::run) — the conn actor's context. +pub fn deliver_named(name: &str, type_hash: u64, payload: &[u8]) -> InboundVerdict { + let Some(expected) = exposed_hash(name) else { + return InboundVerdict::NotExposed; + }; + if expected != type_hash { + return InboundVerdict::HashMismatch { + expected, + got: type_hash, + }; + } + let Some(pid) = whereis(name) else { + return InboundVerdict::Unresolved; + }; + match decode_deliver(type_hash, pid, payload) { + Ok(()) => InboundVerdict::Delivered, + Err(e) => InboundVerdict::Refused(e), + } +} + +// ---- monitors (c12) ----------------------------------------------------- + +/// Bookkeeping commands from [`monitor_remote`]/[`demonitor_remote`] to the +/// connection actor that owns the link to the target's node. The actor +/// records the registration and *then* emits the `Monitor` frame itself, so +/// a `Down` can never arrive at a table that does not yet know the id. It +/// lives in the actor (not on `RuntimeInner`) so the bookkeeping dies with +/// the connection — exactly what c13 needs to synthesize `Disconnected`. +pub(crate) enum MonCmd { + Monitor { + id: MonitorId, + target: RemotePid, + tx: Sender, + }, + Demonitor { + id: MonitorId, + }, +} + +/// A remotely-monitored actor's termination notice — the cluster analog of +/// [`Down`](crate::monitor::Down), with the pid in its wire form because it +/// may name any node. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct RemoteDown { + /// The pid that was being monitored. + pub pid: RemotePid, + /// How it went down. `Disconnected` means the *link* to its node was + /// lost (or absent) — nothing is known about the actor itself. + pub reason: RemoteDownReason, +} + +enum Watch { + /// The target collapsed to this node: an ordinary local monitor, + /// translated on read. + Local(Monitor), + /// The target is elsewhere: the connection actor for its node holds the + /// registration and forwards the peer's `Down` frame here. The + /// `RemoteState` is the read-side backstop (c13): a channel that closes + /// while `Live` — the connection died with our command unread — reads + /// as `Disconnected` once; afterwards, and after a cancel, closed is + /// just closed. + Remote(Receiver, Cell), +} + +/// Where a remote-watch stands from the reader's side. +#[derive(Clone, Copy, PartialEq, Eq)] +enum RemoteState { + /// No notice yet, not cancelled: a closed channel means `Disconnected`. + Live, + /// The one notice has been read (or synthesized): nothing more is due. + Done, + /// `demonitor_remote` ran: never synthesize. + Cancelled, +} + +/// A live remote monitor: read its one [`RemoteDown`] with +/// [`recv`](RemoteMonitor::recv)/[`try_recv`](RemoteMonitor::try_recv), or +/// fold it into a `select` via [`arm`](RemoteMonitor::arm). Distinct from +/// [`Monitor`] on purpose: its target is a [`RemotePid`], its notice a +/// [`RemoteDown`], and it can report `Disconnected` — none of which a local +/// monitor can express. Dropping it discards an unread notice, like the +/// local one; after [`demonitor_remote`] the channel is closed and empty, so +/// `recv` errs rather than parking — also like the local one. +/// +/// Exactly one notice is guaranteed even if the connection actor dies with +/// the registration unread (the c13 drain gap): a channel that closes +/// before any notice — and before any cancel — reads as `Disconnected`, +/// once. The next read is the ordinary closed-channel `Err`. +pub struct RemoteMonitor { + /// This registration's process-unique id — minted here, echoed by the + /// peer in its `Down` frame. + pub id: MonitorId, + /// The pid being monitored. + pub target: RemotePid, + watch: Watch, +} + +impl RemoteMonitor { + /// Block (cooperatively) for the notice. + pub fn recv(&self) -> Result { + match &self.watch { + Watch::Local(m) => m.rx.recv().map(|d| RemoteDown { + pid: self.target.clone(), + reason: d.reason.into(), + }), + Watch::Remote(rx, st) => match rx.recv() { + Ok(d) => { + st.set(RemoteState::Done); + Ok(d) + } + Err(e) => self.closed(st).ok_or(e), + }, + } + } + + /// The notice if it has arrived; `Ok(None)` if not yet. + pub fn try_recv(&self) -> Result, RecvError> { + match &self.watch { + Watch::Local(m) => m.rx.try_recv().map(|o| { + o.map(|d| RemoteDown { + pid: self.target.clone(), + reason: d.reason.into(), + }) + }), + Watch::Remote(rx, st) => match rx.try_recv() { + Ok(Some(d)) => { + st.set(RemoteState::Done); + Ok(Some(d)) + } + Ok(None) => Ok(None), + Err(e) => self.closed(st).map(Some).ok_or(e), + }, + } + } + + /// The channel closed. While `Live` — no notice yet, no cancel — that + /// is the connection having died with our registration unread, so + /// synthesize the one `Disconnected` and mark `Done`; otherwise closed + /// is just closed. + fn closed(&self, st: &Cell) -> Option { + if st.get() != RemoteState::Live { + return None; + } + st.set(RemoteState::Done); + Some(RemoteDown { + pid: self.target.clone(), + reason: RemoteDownReason::Disconnected, + }) + } + + /// The selectable arm: readiness means [`try_recv`](Self::try_recv) + /// will yield the notice. + pub fn arm(&self) -> &dyn Selectable { + match &self.watch { + Watch::Local(m) => &m.rx, + Watch::Remote(rx, _) => rx, + } + } +} + +impl std::fmt::Debug for RemoteMonitor { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("RemoteMonitor") + .field("id", &self.id) + .field("target", &self.target) + .finish_non_exhaustive() + } +} + +/// Monitor `target`, wherever it lives. Exactly one [`RemoteDown`] arrives: +/// +/// - self-node pid ⇒ an ordinary local monitor underneath (same NoProc rule); +/// - no live connection to the pid's node ⇒ `Disconnected`, queued at once +/// (the remote analog of NoProc: nothing can be known); +/// - the pid's incarnation is not the node's current one ⇒ `NoProc`, queued +/// at once — the node is *known* to have restarted, so its actor is a +/// corpse, not a partition (RFC v2 §3); +/// - otherwise the connection actor registers the id and sends `Monitor`; +/// the peer answers with the true terminal reason on exit, or immediately +/// with the recorded reason for a corpse (`terminal_reason`, RFC §6) or +/// `NoProc` for a pid it never exposed and never shipped. +/// +/// The connection dropping while the monitor is outstanding delivers +/// `Disconnected` (c13): the connection actor synthesizes it on teardown, +/// and the monitor's own read path backstops the case where the actor died +/// with the registration still unread. Must run inside [`run`](crate::run). +pub fn monitor_remote(target: RemotePid) -> RemoteMonitor { + if let Some(local) = target.local() { + let m = monitor(local); + return RemoteMonitor { + id: m.id, + target: target.erase(), + watch: Watch::Local(m), + }; + } + let target = target.erase(); + let (id, route) = with_runtime(|inner| { + let id = inner.alloc_monitor_id(); + let route = inner + .outbound + .lock() + .by_node + .get(&target.node) + .map(|r| (r.incarnation, r.monitors.clone())); + (id, route) + }); + let (tx, rx) = channel::(); + let immediate = match route { + None => Some(RemoteDownReason::Disconnected), + Some((current, _)) if current != target.incarnation => { + Some(RemoteDownReason::Local(crate::monitor::DownReason::NoProc)) + } + Some((_, mon_tx)) => { + let cmd = MonCmd::Monitor { + id, + target: target.clone(), + tx: tx.clone(), + }; + match mon_tx.send(cmd) { + Ok(()) => None, + Err(_) => Some(RemoteDownReason::Disconnected), // actor already gone + } + } + }; + if let Some(reason) = immediate { + let _ = tx.send(RemoteDown { + pid: target.clone(), + reason, + }); + } + RemoteMonitor { + id, + target, + watch: Watch::Remote(rx, Cell::new(RemoteState::Live)), + } +} + +/// Cancel `m`. No future notice will be *sent* for it; a notice already in +/// flight from the peer is dropped on arrival, and one already sitting in +/// `m` is discarded when `m` is dropped (same contract as +/// [`demonitor`]). Unlike the local form this returns nothing: the +/// registration is owned by the connection actor, so whether the `Down` +/// beat the cancel is not local knowledge. Must run inside +/// [`run`](crate::run). +pub fn demonitor_remote(m: &RemoteMonitor) { + match &m.watch { + Watch::Local(local) => { + let _ = demonitor(local); + } + Watch::Remote(_, st) => { + // Cancel first: a channel closing after this is closed, not a + // Disconnected notice — the caller asked for silence. + st.set(RemoteState::Cancelled); + let mon_tx = with_runtime(|inner| { + inner + .outbound + .lock() + .by_node + .get(&m.target.node) + .map(|r| r.monitors.clone()) + }); + if let Some(mon_tx) = mon_tx { + let _ = mon_tx.send(MonCmd::Demonitor { id: m.id }); + } + } + } +} diff --git a/src/cluster/transport.rs b/src/cluster/transport.rs new file mode 100644 index 0000000..8f0e9fc --- /dev/null +++ b/src/cluster/transport.rs @@ -0,0 +1,298 @@ +//! RFC 010 c3 — transport abstraction for the **control** connection. +//! +//! Scope, per RFC 010 v2 §5 and D2: +//! +//! - A "connection" here is the *control* connection: the one carrying this +//! RFC's frame inventory ([`crate::cluster::envelope::Frame`]), whose +//! heartbeats feed failure detection. The trait deliberately says nothing +//! about how many connections a peer pair may hold — the jarred rkyv bulk +//! plane opens **additional per-peer connections** outside this trait, and +//! nothing here may foreclose that. +//! - Homogeneous smarm⇄smarm only. The BEAM membrane is *not* a transport +//! impl and the trait does not accommodate it (D2). +//! - Addresses are opaque, **pre-resolved** strings. Name resolution is a +//! single separate seam (roadmap c9); impls reject unresolved names rather +//! than resolving them. +//! +//! Blocking model: [`Conn`] calls block the caller. The TCP impl parks the +//! calling *actor* (fd readiness via the scheduler); the loopback impl blocks +//! the calling *OS thread* and is a test transport — do not drive it from a +//! scheduler thread. +//! +//! Framing is not part of the trait: [`FramedConn`] is the single shared +//! codec that turns any byte-stream [`Conn`] into a frame pipe, feeding +//! [`Frame::decode`]'s incremental contract. Impls never re-implement +//! framing, and the conformance suite exercises the same codec over every +//! impl. + +use std::io; + +use crate::cluster::envelope::{DecodeError, EncodeError, Frame}; + +pub mod loopback; +pub mod tcp; + +/// An established control connection: a bidirectional byte stream. +pub trait Conn: Send { + /// Read at least one byte, blocking the caller until data is available, + /// EOF, or error. `Ok(0)` means EOF: the peer closed and all bytes it + /// wrote before closing have been consumed. + fn read(&mut self, buf: &mut [u8]) -> io::Result; + + /// Write the whole buffer, blocking the caller as needed. + fn write_all(&mut self, buf: &[u8]) -> io::Result<()>; + + /// Close both directions. Idempotent. Bytes already written remain + /// readable at the peer, which then observes EOF; peer writes after this + /// fail. + fn close(&mut self); + + /// Diagnostic label for logs only. Mesh identity comes from the + /// handshake (`Hello`/`HelloAck`), never from the transport. + fn peer_addr(&self) -> String; + + /// Readiness as a [`select`](crate::select) arm, for transports backed by + /// a file descriptor. `Some` lets a driver wait on "this connection is + /// readable" alongside an ordinary command inbox in a single `select`, so + /// one actor can interleave reading with control messages without a + /// second thread. The default is `None`: a transport with no fd (the + /// in-memory loopback) cannot be selected on and must be driven another + /// way. + fn readable_arm(&self) -> Option { + None + } +} + +/// A bound listen point producing inbound [`Conn`]s. +pub trait Listener: Send { + /// Accept the next inbound connection, blocking the caller. + fn accept(&mut self) -> io::Result>; + + /// The concrete bound address, dialable as-is (e.g. the real port when + /// bound with port 0). + fn local_addr(&self) -> String; + + /// Readiness as a [`select`](crate::select) arm, mirroring + /// [`Conn::readable_arm`]: `Some` lets an acceptor wait on "an inbound + /// connection is pending" alongside a command inbox in one `select`, so + /// it can be told to stop without a poll loop. Default `None` (the + /// loopback listener has no fd and must be driven synchronously). + fn readable_arm(&self) -> Option { + None + } +} + +/// A way of establishing control connections. Object-safe on purpose: the +/// connector and membership layers hold `&dyn Transport` / boxed conns +/// rather than growing a generic parameter. +pub trait Transport: Send + Sync { + /// Connect to a peer's listen address. Blocks the caller until + /// established or failed. + fn dial(&self, addr: &str) -> io::Result>; + + /// Bind a listen point. + fn listen(&self, addr: &str) -> io::Result>; +} + +impl std::fmt::Debug for dyn Conn { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "Conn({})", self.peer_addr()) + } +} + +impl std::fmt::Debug for dyn Listener { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "Listener({})", self.local_addr()) + } +} + +/// Error surface of [`FramedConn::send`]. +#[derive(Debug)] +pub enum SendError { + /// The frame could not be encoded (e.g. a field over its wire limit). + Encode(EncodeError), + /// The transport failed mid-write. + Io(io::Error), +} + +impl std::fmt::Display for SendError { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + SendError::Encode(e) => write!(f, "frame encode failed: {e:?}"), + SendError::Io(e) => write!(f, "transport write failed: {e}"), + } + } +} + +impl std::error::Error for SendError {} + +/// Error surface of [`FramedConn::recv`]. +#[derive(Debug)] +pub enum RecvError { + /// The byte stream is not a valid frame stream (bad tag, lying length, + /// oversized frame, …). The connection is unusable. + Corrupt(DecodeError), + /// The peer closed mid-frame: EOF arrived with a partial frame buffered. + /// Distinct from a clean close, which is `Ok(None)`. + TruncatedByPeer, + /// The transport failed mid-read. + Io(io::Error), + /// The deadline passed before a full frame arrived + /// ([`FramedConn::recv_deadline`] only; plain [`recv`](FramedConn::recv) + /// never returns this). + TimedOut, +} + +impl std::fmt::Display for RecvError { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + RecvError::Corrupt(e) => write!(f, "frame stream corrupt: {e:?}"), + RecvError::TruncatedByPeer => write!(f, "peer closed mid-frame"), + RecvError::Io(e) => write!(f, "transport read failed: {e}"), + RecvError::TimedOut => write!(f, "deadline passed mid-receive"), + } + } +} + +impl std::error::Error for RecvError {} + +/// How many bytes each blocking read asks the transport for. +const READ_CHUNK: usize = 8 * 1024; + +/// The shared framed codec: one of these per control connection, owning the +/// [`Conn`] and the reassembly buffer. Frames may arrive split or coalesced +/// arbitrarily; [`recv`](FramedConn::recv) reassembles either way. +pub struct FramedConn { + conn: Box, + rbuf: Vec, +} + +impl FramedConn { + pub fn new(conn: Box) -> Self { + FramedConn { + conn, + rbuf: Vec::new(), + } + } + + /// Encode and write one frame. + pub fn send(&mut self, frame: &Frame) -> Result<(), SendError> { + let mut out = Vec::new(); + frame.encode(&mut out).map_err(SendError::Encode)?; + self.conn.write_all(&out).map_err(SendError::Io) + } + + /// Receive the next frame. `Ok(None)` is a clean close: EOF at a frame + /// boundary. EOF mid-frame is [`RecvError::TruncatedByPeer`]. + pub fn recv(&mut self) -> Result, RecvError> { + loop { + match Frame::decode(&self.rbuf) { + Ok(Some((frame, consumed))) => { + self.rbuf.drain(..consumed); + return Ok(Some(frame)); + } + Ok(None) => {} + Err(e) => return Err(RecvError::Corrupt(e)), + } + let mut chunk = [0u8; READ_CHUNK]; + let n = self.conn.read(&mut chunk).map_err(RecvError::Io)?; + if n == 0 { + return if self.rbuf.is_empty() { + Ok(None) + } else { + Err(RecvError::TruncatedByPeer) + }; + } + self.rbuf.extend_from_slice(&chunk[..n]); + } + } + + /// Like [`recv`](FramedConn::recv), but gives up with + /// [`RecvError::TimedOut`] once `deadline` passes without a full frame. + /// The deadline is enforced between reads via the connection's fd arm + /// (so the caller must be an actor); a transport with no fd (loopback) + /// cannot be timed out and this degrades to a plain blocking `recv` — + /// the same caveat as liveness. + pub fn recv_deadline( + &mut self, + deadline: std::time::Instant, + ) -> Result, RecvError> { + loop { + match Frame::decode(&self.rbuf) { + Ok(Some((frame, consumed))) => { + self.rbuf.drain(..consumed); + return Ok(Some(frame)); + } + Ok(None) => {} + Err(e) => return Err(RecvError::Corrupt(e)), + } + if let Some(arm) = self.conn.readable_arm() { + let left = deadline.saturating_duration_since(std::time::Instant::now()); + if left.is_zero() { + return Err(RecvError::TimedOut); + } + match crate::channel::try_select_timeout(&[&arm], left) { + Ok(Some(_)) => {} + Ok(None) => return Err(RecvError::TimedOut), + Err(e) => return Err(RecvError::Io(e)), + } + } + let mut chunk = [0u8; READ_CHUNK]; + let n = self.conn.read(&mut chunk).map_err(RecvError::Io)?; + if n == 0 { + return if self.rbuf.is_empty() { + Ok(None) + } else { + Err(RecvError::TruncatedByPeer) + }; + } + self.rbuf.extend_from_slice(&chunk[..n]); + } + } + + /// One socket read, appended to the reassembly buffer. Returns the byte + /// count (`0` = EOF). For select-loop callers that were just told the fd + /// is readable: under the level-triggered IO thread exactly one read per + /// readable wake never blocks and never loses data — leftover socket + /// bytes re-signal on the next select, and complete frames already + /// reassembled are drained with [`next_buffered`](FramedConn::next_buffered). + /// (A plain [`recv`](FramedConn::recv) can block into the socket while + /// the buffer holds a partial frame, which a loop with deadlines to keep + /// cannot afford.) + pub fn read_once(&mut self) -> std::io::Result { + let mut chunk = [0u8; READ_CHUNK]; + let n = self.conn.read(&mut chunk)?; + self.rbuf.extend_from_slice(&chunk[..n]); + Ok(n) + } + + /// Decode the next complete frame already sitting in the reassembly + /// buffer, without touching the socket. `Ok(None)` means the buffer + /// holds no complete frame (empty, or a partial awaiting more bytes). + pub fn next_buffered(&mut self) -> Result, DecodeError> { + match Frame::decode(&self.rbuf) { + Ok(Some((frame, consumed))) => { + self.rbuf.drain(..consumed); + Ok(Some(frame)) + } + Ok(None) => Ok(None), + Err(e) => Err(e), + } + } + + /// Close the underlying connection (idempotent, see [`Conn::close`]). + pub fn close(&mut self) { + self.conn.close(); + } + + /// Diagnostic label of the underlying connection. + pub fn peer_addr(&self) -> String { + self.conn.peer_addr() + } + + /// The underlying connection's readiness arm, if it is fd-backed (see + /// [`Conn::readable_arm`]). + pub fn readable_arm(&self) -> Option { + self.conn.readable_arm() + } +} diff --git a/src/cluster/transport/loopback.rs b/src/cluster/transport/loopback.rs new file mode 100644 index 0000000..61705f9 --- /dev/null +++ b/src/cluster/transport/loopback.rs @@ -0,0 +1,274 @@ +//! In-memory loopback transport — a shipped **test** transport. +//! +//! Lets Phases 2–4 exercise protocol logic (connector, membership, +//! monitors) through the real transport trait and the real framed codec +//! without sockets or timing flake. +//! +//! Blocking model: calls block the **OS thread** on a condvar. That is the +//! right shape for plain `#[test]`s driving protocol state machines; it is +//! the wrong shape for scheduler threads. Do not drive a loopback conn from +//! inside an actor — use the TCP impl there. +//! +//! Semantics mirror TCP shutdown where it matters for the codec: bytes +//! written before `close` remain readable at the peer, which then sees EOF; +//! writes toward a closed peer fail with `BrokenPipe`. Write buffers are +//! unbounded, so writes never block — backpressure is not simulated. + +use std::collections::{HashMap, VecDeque}; +use std::io; +use std::sync::{Arc, Condvar, Mutex, MutexGuard}; + +use super::{Conn, Listener, Transport}; + +/// Poison-tolerant lock: a panicked holder in a *test* transport must not +/// cascade; the byte-queue state stays consistent under every early return. +fn lock(m: &Mutex) -> MutexGuard<'_, T> { + match m.lock() { + Ok(g) => g, + Err(poisoned) => poisoned.into_inner(), + } +} + +// --------------------------------------------------------------------------- +// One direction of a duplex: a byte queue with close flags for both ends +// --------------------------------------------------------------------------- + +#[derive(Default)] +struct PipeState { + bytes: VecDeque, + /// The writing end closed: readers drain remaining bytes, then EOF. + write_closed: bool, + /// The reading end closed: writers fail with `BrokenPipe`. + read_closed: bool, +} + +#[derive(Default)] +struct Pipe { + state: Mutex, + cv: Condvar, +} + +impl Pipe { + fn write_all(&self, buf: &[u8]) -> io::Result<()> { + let mut st = lock(&self.state); + if st.write_closed { + return Err(io::Error::new( + io::ErrorKind::NotConnected, + "loopback conn closed locally", + )); + } + if st.read_closed { + return Err(io::Error::new( + io::ErrorKind::BrokenPipe, + "loopback peer closed", + )); + } + st.bytes.extend(buf); + self.cv.notify_all(); + Ok(()) + } + + fn read(&self, buf: &mut [u8]) -> io::Result { + if buf.is_empty() { + return Ok(0); + } + let mut st = lock(&self.state); + loop { + if !st.bytes.is_empty() { + let n = st.bytes.len().min(buf.len()); + for (slot, byte) in buf.iter_mut().zip(st.bytes.drain(..n)) { + *slot = byte; + } + return Ok(n); + } + if st.write_closed || st.read_closed { + return Ok(0); // EOF: peer closed, or our own end closed. + } + st = match self.cv.wait(st) { + Ok(g) => g, + Err(poisoned) => poisoned.into_inner(), + }; + } + } + + /// Close from the writer side: remaining bytes stay readable, then EOF. + fn close_write(&self) { + lock(&self.state).write_closed = true; + self.cv.notify_all(); + } + + /// Close from the reader side: peer writes fail from now on. + fn close_read(&self) { + lock(&self.state).read_closed = true; + self.cv.notify_all(); + } +} + +// --------------------------------------------------------------------------- +// Conn: two pipes, one per direction +// --------------------------------------------------------------------------- + +/// One end of an established loopback connection. +pub struct LoopbackConn { + tx: Arc, + rx: Arc, + peer: String, +} + +impl Conn for LoopbackConn { + fn read(&mut self, buf: &mut [u8]) -> io::Result { + self.rx.read(buf) + } + + fn write_all(&mut self, buf: &[u8]) -> io::Result<()> { + self.tx.write_all(buf) + } + + fn close(&mut self) { + self.tx.close_write(); + self.rx.close_read(); + } + + fn peer_addr(&self) -> String { + self.peer.clone() + } +} + +impl Drop for LoopbackConn { + fn drop(&mut self) { + self.close(); + } +} + +fn conn_pair(listen_addr: &str, conn_no: u64) -> (LoopbackConn, LoopbackConn) { + let a_to_b = Arc::new(Pipe::default()); + let b_to_a = Arc::new(Pipe::default()); + let dialer = LoopbackConn { + tx: a_to_b.clone(), + rx: b_to_a.clone(), + peer: listen_addr.to_string(), + }; + let accepted = LoopbackConn { + tx: b_to_a, + rx: a_to_b, + peer: format!("{listen_addr}#dialer-{conn_no}"), + }; + (dialer, accepted) +} + +// --------------------------------------------------------------------------- +// Listener + registry +// --------------------------------------------------------------------------- + +#[derive(Default)] +struct AcceptState { + pending: VecDeque, + closed: bool, +} + +#[derive(Default)] +struct AcceptQueue { + state: Mutex, + cv: Condvar, +} + +/// A bound loopback listen point. +pub struct LoopbackListener { + addr: String, + queue: Arc, + registry: Arc>, +} + +impl Listener for LoopbackListener { + fn accept(&mut self) -> io::Result> { + let mut st = lock(&self.queue.state); + loop { + if let Some(conn) = st.pending.pop_front() { + return Ok(Box::new(conn)); + } + if st.closed { + return Err(io::Error::new( + io::ErrorKind::NotConnected, + "loopback listener closed", + )); + } + st = match self.queue.cv.wait(st) { + Ok(g) => g, + Err(poisoned) => poisoned.into_inner(), + }; + } + } + + fn local_addr(&self) -> String { + self.addr.clone() + } +} + +impl Drop for LoopbackListener { + fn drop(&mut self) { + lock(&self.registry).listeners.remove(&self.addr); + let mut st = lock(&self.queue.state); + st.closed = true; + self.queue.cv.notify_all(); + } +} + +#[derive(Default)] +struct Registry { + listeners: HashMap>, + dial_count: u64, +} + +/// The loopback transport. Addresses are arbitrary strings scoped to one +/// transport instance; distinct instances never see each other's listeners. +#[derive(Default)] +pub struct LoopbackTransport { + registry: Arc>, +} + +impl Transport for LoopbackTransport { + fn dial(&self, addr: &str) -> io::Result> { + let (queue, conn_no) = { + let mut reg = lock(&self.registry); + reg.dial_count += 1; + let no = reg.dial_count; + match reg.listeners.get(addr) { + Some(q) => (q.clone(), no), + None => { + return Err(io::Error::new( + io::ErrorKind::ConnectionRefused, + format!("no loopback listener at {addr:?}"), + )); + } + } + }; + let (dialer, accepted) = conn_pair(addr, conn_no); + let mut st = lock(&queue.state); + if st.closed { + return Err(io::Error::new( + io::ErrorKind::ConnectionRefused, + format!("loopback listener at {addr:?} closed"), + )); + } + st.pending.push_back(accepted); + queue.cv.notify_all(); + Ok(Box::new(dialer)) + } + + fn listen(&self, addr: &str) -> io::Result> { + let queue = Arc::new(AcceptQueue::default()); + let mut reg = lock(&self.registry); + if reg.listeners.contains_key(addr) { + return Err(io::Error::new( + io::ErrorKind::AddrInUse, + format!("loopback listener already bound at {addr:?}"), + )); + } + reg.listeners.insert(addr.to_string(), queue.clone()); + Ok(Box::new(LoopbackListener { + addr: addr.to_string(), + queue, + registry: self.registry.clone(), + })) + } +} diff --git a/src/cluster/transport/tcp.rs b/src/cluster/transport/tcp.rs new file mode 100644 index 0000000..2fdc421 --- /dev/null +++ b/src/cluster/transport/tcp.rs @@ -0,0 +1,285 @@ +//! TCP transport — the production control-plane transport. +//! +//! Blocking model: every blocking point parks the **calling actor** on fd +//! readiness ([`crate::scheduler::wait_readable`] / `wait_writable`); the +//! scheduler thread is never blocked. All conn/listener methods must +//! therefore run inside an actor. `listen` itself only binds (no waiting) +//! and is callable anywhere. +//! +//! Addresses are pre-resolved `ip:port` strings (`SocketAddr` syntax, IPv4 +//! or IPv6). Hostnames are rejected with `InvalidInput`: name resolution is +//! the single c9 seam, not something each transport does on the side. +//! +//! Writes use `send(2)` with `MSG_NOSIGNAL` — a peer reset must surface as +//! `BrokenPipe`/`ConnectionReset`, not `SIGPIPE`. + +use std::io; +use std::net::{SocketAddr, TcpListener as StdListener, TcpStream}; +use std::os::fd::{AsRawFd, RawFd}; + +use crate::scheduler::{wait_readable, wait_writable}; + +use super::{Conn, Listener, Transport}; + +// --------------------------------------------------------------------------- +// sockaddr plumbing +// --------------------------------------------------------------------------- + +/// A `sockaddr_in`/`sockaddr_in6` built from a parsed `SocketAddr`, plus its +/// length, ready for `connect(2)`. +union SockAddrUnion { + v4: libc::sockaddr_in, + v6: libc::sockaddr_in6, +} + +fn to_sockaddr(sa: &SocketAddr) -> (SockAddrUnion, libc::socklen_t) { + match sa { + SocketAddr::V4(v4) => { + let raw = libc::sockaddr_in { + sin_family: libc::AF_INET as libc::sa_family_t, + sin_port: v4.port().to_be(), + sin_addr: libc::in_addr { + s_addr: u32::from_be_bytes(v4.ip().octets()).to_be(), + }, + sin_zero: [0; 8], + }; + ( + SockAddrUnion { v4: raw }, + std::mem::size_of::() as libc::socklen_t, + ) + } + SocketAddr::V6(v6) => { + let raw = libc::sockaddr_in6 { + sin6_family: libc::AF_INET6 as libc::sa_family_t, + sin6_port: v6.port().to_be(), + sin6_flowinfo: v6.flowinfo(), + sin6_addr: libc::in6_addr { + s6_addr: v6.ip().octets(), + }, + sin6_scope_id: v6.scope_id(), + }; + ( + SockAddrUnion { v6: raw }, + std::mem::size_of::() as libc::socklen_t, + ) + } + } +} + +fn parse_addr(addr: &str) -> io::Result { + addr.parse().map_err(|_| { + io::Error::new( + io::ErrorKind::InvalidInput, + format!("{addr:?} is not a resolved ip:port — resolution is the c9 seam"), + ) + }) +} + +fn so_error(fd: RawFd) -> io::Result<()> { + let mut err: libc::c_int = 0; + let mut len = std::mem::size_of::() as libc::socklen_t; + let rc = unsafe { + libc::getsockopt( + fd, + libc::SOL_SOCKET, + libc::SO_ERROR, + (&mut err) as *mut _ as *mut libc::c_void, + &mut len, + ) + }; + if rc != 0 { + return Err(io::Error::last_os_error()); + } + if err != 0 { + return Err(io::Error::from_raw_os_error(err)); + } + Ok(()) +} + +// --------------------------------------------------------------------------- +// Conn +// --------------------------------------------------------------------------- + +/// One established TCP control connection. Owns the socket; drop closes it. +pub struct TcpConn { + stream: TcpStream, + closed: bool, +} + +impl TcpConn { + fn fd(&self) -> RawFd { + self.stream.as_raw_fd() + } +} + +impl Conn for TcpConn { + fn read(&mut self, buf: &mut [u8]) -> io::Result { + if self.closed { + return Ok(0); + } + if buf.is_empty() { + return Ok(0); + } + loop { + wait_readable(self.fd())?; + let n = unsafe { libc::read(self.fd(), buf.as_mut_ptr() as *mut _, buf.len()) }; + if n >= 0 { + return Ok(n as usize); + } + let e = io::Error::last_os_error(); + match e.kind() { + // Spurious readiness or signal: park again. + io::ErrorKind::WouldBlock | io::ErrorKind::Interrupted => continue, + _ => return Err(e), + } + } + } + + fn write_all(&mut self, mut buf: &[u8]) -> io::Result<()> { + if self.closed { + return Err(io::Error::new( + io::ErrorKind::NotConnected, + "tcp conn closed locally", + )); + } + while !buf.is_empty() { + wait_writable(self.fd())?; + let n = unsafe { + libc::send( + self.fd(), + buf.as_ptr() as *const _, + buf.len(), + libc::MSG_NOSIGNAL, + ) + }; + if n >= 0 { + buf = &buf[n as usize..]; + continue; + } + let e = io::Error::last_os_error(); + match e.kind() { + io::ErrorKind::WouldBlock | io::ErrorKind::Interrupted => continue, + _ => return Err(e), + } + } + Ok(()) + } + + fn close(&mut self) { + if !self.closed { + self.closed = true; + // Best-effort: the peer sees EOF after draining. The fd itself + // is released when the owning stream drops. + let _ = self.stream.shutdown(std::net::Shutdown::Both); + } + } + + fn peer_addr(&self) -> String { + match self.stream.peer_addr() { + Ok(sa) => sa.to_string(), + Err(_) => "".to_string(), + } + } + + fn readable_arm(&self) -> Option { + Some(crate::scheduler::FdArm::readable(self.fd())) + } +} + +// --------------------------------------------------------------------------- +// Listener +// --------------------------------------------------------------------------- + +/// A bound TCP listen point (non-blocking socket; accept parks the actor). +pub struct TcpListener { + inner: StdListener, + local: SocketAddr, +} + +impl Listener for TcpListener { + fn readable_arm(&self) -> Option { + Some(crate::scheduler::FdArm::readable(self.inner.as_raw_fd())) + } + + fn accept(&mut self) -> io::Result> { + loop { + wait_readable(self.inner.as_raw_fd())?; + match self.inner.accept() { + Ok((stream, _peer)) => { + stream.set_nonblocking(true)?; + return Ok(Box::new(TcpConn { + stream, + closed: false, + })); + } + Err(e) + if e.kind() == io::ErrorKind::WouldBlock + || e.kind() == io::ErrorKind::Interrupted => + { + continue; + } + Err(e) => return Err(e), + } + } + } + + fn local_addr(&self) -> String { + self.local.to_string() + } +} + +// --------------------------------------------------------------------------- +// Transport +// --------------------------------------------------------------------------- + +/// The TCP transport. Stateless; every call stands alone. +pub struct TcpTransport; + +impl Transport for TcpTransport { + fn dial(&self, addr: &str) -> io::Result> { + let sa = parse_addr(addr)?; + let family = match sa { + SocketAddr::V4(_) => libc::AF_INET, + SocketAddr::V6(_) => libc::AF_INET6, + }; + let fd = unsafe { + libc::socket( + family, + libc::SOCK_STREAM | libc::SOCK_NONBLOCK | libc::SOCK_CLOEXEC, + 0, + ) + }; + if fd < 0 { + return Err(io::Error::last_os_error()); + } + // From here the fd is owned by `stream`; any early return drops it. + let stream = unsafe { + use std::os::fd::FromRawFd; + TcpStream::from_raw_fd(fd) + }; + let (raw, len) = to_sockaddr(&sa); + let rc = unsafe { libc::connect(fd, (&raw) as *const _ as *const libc::sockaddr, len) }; + if rc != 0 { + let e = io::Error::last_os_error(); + if e.raw_os_error() != Some(libc::EINPROGRESS) { + return Err(e); + } + // Connect in flight: park until the socket is writable, then the + // verdict is in SO_ERROR. + wait_writable(fd)?; + so_error(fd)?; + } + Ok(Box::new(TcpConn { + stream, + closed: false, + })) + } + + fn listen(&self, addr: &str) -> io::Result> { + let sa = parse_addr(addr)?; + let inner = StdListener::bind(sa)?; + inner.set_nonblocking(true)?; + let local = inner.local_addr()?; + Ok(Box::new(TcpListener { inner, local })) + } +} diff --git a/src/lib.rs b/src/lib.rs index 7e54c70..48c4ff6 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -23,6 +23,8 @@ pub mod actor; pub mod causal; pub mod channel; +#[cfg(feature = "cluster")] +pub mod cluster; pub mod context; pub mod gen_server; pub mod gen_statem; diff --git a/src/monitor.rs b/src/monitor.rs index e6b16a3..650bdb1 100644 --- a/src/monitor.rs +++ b/src/monitor.rs @@ -156,40 +156,68 @@ pub struct Monitor { pub fn monitor(target: Pid) -> Monitor { let target = target.erase(); let (tx, rx) = channel::(); - - // Implementation note: registration happens under the target's cold - // lock. `tx.clone()` takes the channel's own lock, a Channel-class - // RawMutex, which is explicitly permitted under a Leaf (cold) lock by - // the lock order documented in raw_mutex.rs. We must still not *send* - // under the lock, since `Sender::send` can unpark a parked receiver, - // and there's no reason to nest that. - let (id, registered) = with_runtime(|inner| { - let id = inner.alloc_monitor_id(); - let registered = match inner.slot_at(target) { - Some(slot) => { - let mut cold = slot.cold.lock(); - if slot.is_live_for(target) { - cold.monitors.push((id, tx.clone())); - true - } else { - false - } - } - None => false, - }; - (id, registered) - }); - - if !registered { + let id = with_runtime(|inner| inner.alloc_monitor_id()); + if !register_monitor(target, id, &tx) { let _ = tx.send(Down { pid: target, reason: DownReason::NoProc, }); } - Monitor { id, target, rx } } +/// Register a monitor `id` on `target` that delivers its `Down` to `tx` — the +/// primitive under [`monitor`], split out so a caller can fan many monitors +/// into ONE channel (process groups: every membership's death lands on the +/// reaper's single inbox). Returns `false` if `target` is already gone, in +/// which case nothing is registered and the caller decides what to queue +/// (`monitor` sends `NoProc`). The caller allocates `id` up front so it can +/// record the registration *before* arming it. +/// +/// Implementation note: registration happens under the target's cold lock. +/// `tx.clone()` takes the channel's own lock, a Channel-class RawMutex, which +/// is explicitly permitted under a Leaf (cold) lock by the lock order +/// documented in raw_mutex.rs. We must still not *send* under the lock, since +/// `Sender::send` can unpark a parked receiver, and there's no reason to nest +/// that. +pub(crate) fn register_monitor(target: Pid, id: MonitorId, tx: &Sender) -> bool { + with_runtime(|inner| match inner.slot_at(target) { + Some(slot) => { + let mut cold = slot.cold.lock(); + if slot.is_live_for(target) { + cold.monitors.push((id, tx.clone())); + true + } else { + false + } + } + None => false, + }) +} + +/// Remove registration `id` from `target` — the primitive under +/// [`demonitor`], for callers that hold only the id (see +/// [`register_monitor`]). `None` if the registration is not there: already +/// fired, already removed, or the slot has moved on to a new tenant. +/// +/// The registration is removed under the target's cold lock, but the +/// `Sender` is moved *out* and dropped only after the lock is released. +/// Dropping the last sender runs `Sender::drop`, which may unpark a parked +/// receiver; legal under a cold lock, but pointless to nest. +pub(crate) fn unregister_monitor(target: Pid, id: MonitorId) -> Option { + let removed: Option<(MonitorId, Sender)> = with_runtime(|inner| { + let slot = inner.slot_at(target)?; + let mut cold = slot.cold.lock(); + if slot.generation() != target.generation() { + return None; // slot reused; the Down already fired + } + let pos = cold.monitors.iter().position(|(mid, _)| *mid == id)?; + Some(cold.monitors.remove(pos)) + }); + // `removed`'s sender drops here, outside the lock. + removed.map(|(id, _sender)| id) +} + /// Flag `target`'s tenancy as watchable: its death will stamp the slot's /// terminal record (see [`terminal_reason`]), exactly as registering a name /// does. The bridge calls this wherever a smarm pid is *encoded across the @@ -220,6 +248,23 @@ pub fn mark_watchable(target: Pid) { }); } +/// Whether `target` is live *and* its tenancy is watchable. The cluster's +/// remote-monitor admission check (RFC 010 c12): a peer may monitor a pid only +/// if that pid was exposed or crossed the wire (the D12 set-sites), and a +/// live-but-unwatchable pid answers exactly like a dead one — no liveness leak +/// beyond what `watchable` already grants. Same context contract as +/// [`monitor`]. +#[cfg(feature = "cluster")] +pub(crate) fn is_watchable(target: Pid) -> bool { + let target = target.erase(); + with_runtime(|inner| { + inner.slot_at(target).is_some_and(|slot| { + let cold = slot.cold.lock(); + slot.is_live_for(target) && cold.watchable + }) + }) +} + /// The terminal [`DownReason`] of the tenancy `target` names, if that tenancy /// ever registered a name and is the *most recent named* death of its slot: /// finalize stamps the slot with `(generation, reason)` for once-registered @@ -259,20 +304,5 @@ pub fn terminal_reason(target: Pid) -> Option { /// instead of, or in addition to, calling this: dropping the [`Monitor`] /// closes its receiver and any queued notice is discarded with it. pub fn demonitor(m: &Monitor) -> Option { - // Implementation note: the registration is removed under the target's - // cold lock, but the `Sender` is moved *out* and dropped only after the - // lock is released. Dropping the last sender runs `Sender::drop`, which - // may unpark a parked receiver; legal under a cold lock, but pointless - // to nest. - let removed: Option<(MonitorId, Sender)> = with_runtime(|inner| { - let slot = inner.slot_at(m.target)?; - let mut cold = slot.cold.lock(); - if slot.generation() != m.target.generation() { - return None; // slot reused; the Down already fired - } - let pos = cold.monitors.iter().position(|(mid, _)| *mid == m.id)?; - Some(cold.monitors.remove(pos)) - }); - // `removed`'s sender drops here, outside the lock. - removed.map(|(id, _sender)| id) + unregister_monitor(m.target, m.id) } diff --git a/src/pg.rs b/src/pg.rs index 2b75f6e..03f39e6 100644 --- a/src/pg.rs +++ b/src/pg.rs @@ -99,10 +99,13 @@ //! ## Identity and clustering //! //! A group member is described by a [`Member`] — a [`Pid`] plus a [`NodeId`] and -//! an [`Incarnation`]. Today everything is single-node, those two fields are -//! fixed defaults, and you only ever pass and receive a plain [`Pid`]: the extra -//! identity is carried so this API will not have to change when groups learn to -//! span a cluster. +//! an [`Incarnation`]. Everything on this page is **local**: you pass and +//! receive plain [`Pid`]s, and [`members`] / [`pick`] / [`dispatch`] only ever +//! name actors on this node (Erlang's `get_local_members`). With the `cluster` +//! feature a group also holds the members other nodes have announced, carried +//! under their [`NodeId`]; those never surface here — the cluster-wide reads +//! live in [`cluster::pg`](crate::cluster::pg) (`members_all` and friends) and +//! return a `Local | Remote` member type, since a [`Pid`] cannot hold a remote. //! //! ## Running context //! @@ -110,10 +113,11 @@ //! from inside [`run`](crate::run) (that is, on an actor thread). Calling one //! from outside a running runtime panics. -use crate::monitor::{demonitor, monitor, Monitor}; +use crate::channel::{channel, Sender}; +use crate::monitor::{register_monitor, unregister_monitor, Down, DownReason, MonitorId}; use crate::pid::{assert_type, Addressable, Pid}; use crate::registry::{send_to, SendError}; -use crate::scheduler::with_runtime; +use crate::scheduler::{spawn_under, with_runtime}; use std::collections::HashMap; /// A cluster node handle. A `u32` integer handle, *not* an interned atom — the @@ -186,13 +190,15 @@ pub struct Member { pub pid: Pid, } -/// One membership: a [`Member`] and the [`Monitor`] that watches its liveness. -/// The monitor lives *alongside* the group entry so a group is -/// self-contained: draining the membership tells us whether the member is -/// still alive, and dropping the membership drops its monitor. -struct Membership { - member: Member, - monitor: Monitor, +/// One membership: a [`Member`] and the id of the monitor that watches its +/// liveness. The monitor's `Down` is delivered to the group reaper's single +/// inbox (see [`ProcessGroups::deaths`]), so the membership carries only what +/// [`leave`] needs to tear the registration down: the id. +pub(crate) struct Membership { + pub(crate) member: Member, + /// `None` for a remote member (cluster): the origin node is its liveness + /// authority; nothing here watches it. + pub(crate) monitor: Option, } /// The store: `name → multiset`. Within a single group a `Member` @@ -202,46 +208,60 @@ struct Membership { /// /// Locking discipline. Held under one Leaf-class `RawMutex` on `RuntimeInner`, /// mirroring the registry, and never held together with another Leaf lock (it -/// never touches the registry or a slot's cold lock). The two operations that -/// do need another lock are kept off the group-lock path: +/// never touches the registry or a slot's cold lock). Monitor registration and +/// removal take the target's cold lock (also Leaf), so they run *before* / +/// *after* the group lock, never under it — see [`join`] for the ordering that +/// makes that safe. Nothing under this lock ever touches a channel. /// -/// - `monitor()` / `demonitor()` take the target's cold lock (also Leaf), so -/// they run *before* / *after* the group lock, never under it. -/// - draining a monitor with `try_recv` takes the channel's Channel-class -/// lock, which the lock order permits *under* a Leaf; a channel critical -/// section only does the lock-free unpark protocol, so no Leaf ever nests -/// under it. -/// -/// Evicted and rejected [`Monitor`]s are therefore dropped only *after* the -/// group lock is released, so a receiver-drop never runs a wakeup under the -/// lock — the same discipline as `demonitor`. +/// Eviction is *eager*: every membership's monitor delivers to the one +/// `deaths` channel, drained by a per-run reaper actor that sweeps the dead +/// pid out of every group the moment its `Down` is scheduled. The read path +/// keeps a slot-liveness backstop for the window between a death and the +/// reaper's turn. pub(crate) struct ProcessGroups { groups: HashMap>, + /// The reaper's inboxes: every membership monitor is registered against + /// a clone of `deaths`. `None` until the first `join` of a run spawns + /// the reaper; a stale one (receiver gone with the previous run's + /// teardown) is detected via `receiver_alive` and replaced. + reaper: Option, + /// `NodeId → node name` for every peer with members in the store, kept + /// by the pg actor under this lock, so a stored remote member can be + /// rendered back to its wire identity without asking anyone. + #[cfg(feature = "cluster")] + node_names: HashMap, } impl ProcessGroups { pub(crate) fn new() -> Self { Self { groups: HashMap::new(), + reaper: None, + #[cfg(feature = "cluster")] + node_names: HashMap::new(), } } - /// Insert `ms` into `group`. Idempotent on the *member*: if the member is - /// already present the new membership is handed back (`Some`) so the caller - /// can tear its now-redundant monitor down outside the lock; `None` means - /// it was inserted. - fn join(&mut self, group: &str, ms: Membership) -> Option { + /// Forget the reaper. Called at the start of every `run()` so a stopped + /// reaper from a previous run is never sent to; `join` respawns. + pub(crate) fn reset_reaper(&mut self) { + self.reaper = None; + } + + /// Insert `ms` into `group`. Idempotent on the *member*: `false` means the + /// member was already present and nothing changed; `true` means inserted. + pub(crate) fn join(&mut self, group: &str, ms: Membership) -> bool { let v = self.groups.entry(group.to_owned()).or_default(); if v.iter().any(|e| e.member == ms.member) { - return Some(ms); + return false; } v.push(ms); - None + true } /// Remove `member`'s membership from `group`, returning it (so the caller - /// can `demonitor` it outside the lock). An emptied group is pruned. - fn leave(&mut self, group: &str, member: Member) -> Option { + /// can unregister its monitor outside the lock). An emptied group is pruned. + pub(crate) fn leave(&mut self, group: &str, member: Member) -> Option { let v = self.groups.get_mut(group)?; let pos = v.iter().position(|e| e.member == member)?; let removed = v.remove(pos); @@ -252,20 +272,23 @@ impl ProcessGroups { } /// The one dumb eviction primitive: drop every member matching `pred` from - /// every group, pruning emptied groups, and return the evicted memberships' - /// monitors for the caller to drop outside the lock. The primitive does not - /// know *why* a member leaves; that is the caller's concern. Its callers are - /// the death hook (`reap_group`) and, once clustering lands, an - /// incarnation-eviction sweep — both over this same predicate path, which is - /// the whole reason to shape eviction as a predicate. Insertion order within - /// a group is preserved (`members` / `pick` are order-stable). - fn remove_where(&mut self, mut pred: impl FnMut(&Member) -> bool) -> Vec { + /// every group, pruning emptied groups, and return the evicted + /// memberships with the group each was in. The primitive does not know + /// *why* a member leaves; that is the caller's concern. Its callers are + /// the reaper (a local death) and the cluster's node-down / re-sync + /// sweeps — all over this same predicate path, which is the whole reason + /// to shape eviction as a predicate. Insertion order within a group is + /// preserved (`members` / `pick` are order-stable). + pub(crate) fn remove_where( + &mut self, + mut pred: impl FnMut(&Member) -> bool, + ) -> Vec<(String, Membership)> { let mut evicted = Vec::new(); - self.groups.retain(|_, v| { + self.groups.retain(|g, v| { let mut i = 0; while i < v.len() { if pred(&v[i].member) { - evicted.push(v.remove(i).monitor); + evicted.push((g.clone(), v.remove(i))); } else { i += 1; } @@ -275,38 +298,6 @@ impl ProcessGroups { evicted } - /// Drain-on-contact death hook. The registry can prune a stale binding - /// lazily, on contact, because it only ever resolves one binding at a time; - /// a group is *iterated* — `members` fans out to everyone — so it must not - /// carry a dead member across a broadcast. Every group operation reaps the - /// group it touches first. - /// - /// Drains every membership monitor in `group` with a non-blocking - /// `try_recv`: a delivered `Down` (any reason) or a closed channel means - /// that member is dead. On the first death detected, sweep *all* of the - /// dead pids out of *every* group via [`remove_where`] — a death is removed - /// from each group it joined, not just the one being touched. Returns the - /// evicted monitors to drop outside the lock. - fn reap_group(&mut self, group: &str) -> Vec { - let dead: Vec = { - let Some(v) = self.groups.get(group) else { - return Vec::new(); - }; - v.iter() - .filter_map(|e| match e.monitor.rx.try_recv() { - // A Down arrived, or the channel closed and drained: dead. - Ok(Some(_)) | Err(_) => Some(e.member.pid), - // Empty but open — the sender still lives in the slot: alive. - Ok(None) => None, - }) - .collect() - }; - if dead.is_empty() { - return Vec::new(); - } - self.remove_where(|m| dead.contains(&m.pid)) - } - /// Raw enumeration of a group's members — no liveness filtering. Used by /// tests to assert storage state independently of the read-path backstop. #[cfg(test)] @@ -317,16 +308,24 @@ impl ProcessGroups { .unwrap_or_default() } - /// Live members of `group`, in insertion order. The `is_live` oracle is the - /// read-path backstop: a member whose slot is already dead is - /// dropped from the *result* even if its `Down` has not been drained yet. - /// Backstop only — the entry stays in storage; eviction is the monitor's - /// job (`reap_group`). - fn members_where(&self, group: &str, mut is_live: impl FnMut(Pid) -> bool) -> Vec { + /// Live members of `group` **on `node`**, in insertion order. The + /// `is_live` oracle is the read-path backstop: a member whose slot is + /// already dead is dropped from the *result* even if the reaper has not + /// swept it yet. Backstop only — the entry stays in storage; eviction is + /// the reaper's job. The node filter is what keeps the local API local: + /// a remote member's `pid` is another node's slot bits, meaningless to + /// `is_live` and to any local send. + fn members_where( + &self, + group: &str, + node: NodeId, + mut is_live: impl FnMut(Pid) -> bool, + ) -> Vec { self.groups .get(group) .map(|v| { v.iter() + .filter(|e| e.member.node == node) .map(|e| e.member.pid) .filter(|&p| is_live(p)) .collect() @@ -334,19 +333,210 @@ impl ProcessGroups { .unwrap_or_default() } - /// The first live member of `group` in insertion order — stateless - /// first-live `pick`, with the same read-path backstop as `members_where`. - fn first_member_where(&self, group: &str, mut is_live: impl FnMut(Pid) -> bool) -> Option { + /// The first live member of `group` on `node` in insertion order — + /// stateless first-live `pick`, with the same read-path backstop and node + /// filter as `members_where`. + fn first_member_where( + &self, + group: &str, + node: NodeId, + mut is_live: impl FnMut(Pid) -> bool, + ) -> Option { self.groups .get(group)? .iter() + .filter(|e| e.member.node == node) .map(|e| e.member.pid) .find(|&p| is_live(p)) } } +/// The store's cluster-side surface: raw reads the pg actor needs to speak +/// for this node (`Sync`, membership checks) and the peer-name memo. One +/// `cfg` block: everything here exists only when there is a mesh. +#[cfg(feature = "cluster")] +impl ProcessGroups { + /// Does `group` hold `member` right now? (Raw storage, no liveness.) + pub(crate) fn contains(&self, group: &str, member: &Member) -> bool { + self.groups + .get(group) + .is_some_and(|v| v.iter().any(|e| e.member == *member)) + } + + /// Every stored member of `group`, any node, insertion order. Raw storage. + pub(crate) fn all_of(&self, group: &str) -> Vec { + self.groups + .get(group) + .map(|v| v.iter().map(|e| e.member).collect()) + .unwrap_or_default() + } + + /// `(group, [pid])` for every group with a member on `node` — the + /// `Sync` payload. Raw storage; groups with no such member are omitted. + pub(crate) fn groups_on(&self, node: NodeId) -> Vec<(String, Vec)> { + let mut out: Vec<(String, Vec)> = self + .groups + .iter() + .filter_map(|(g, v)| { + let pids: Vec = v + .iter() + .filter(|e| e.member.node == node) + .map(|e| e.member.pid) + .collect(); + (!pids.is_empty()).then(|| (g.clone(), pids)) + }) + .collect(); + out.sort_by(|a, b| a.0.cmp(&b.0)); + out + } + + /// Record / forget the name behind a peer's `NodeId`. + pub(crate) fn set_node_name(&mut self, node: NodeId, name: String) { + self.node_names.insert(node, name); + } + pub(crate) fn forget_node_name(&mut self, node: NodeId) { + self.node_names.remove(&node); + } + pub(crate) fn node_name(&self, node: NodeId) -> Option<&str> { + self.node_names.get(&node).map(String::as_str) + } +} + +/// The group reaper: one detached actor per run, spawned by the first `join`, +/// parked on the shared `deaths` inbox. Every local membership's monitor +/// delivers here, so a death is swept out of *every* group it joined as soon +/// as the reaper is scheduled — no group operation has to happen first. +/// Sweeps by `(node, pid)`: only local members, since a remote member's pid +/// bits are meaningless here. Exits when the last sender is gone, i.e. never +/// during a run (the store holds one); the run's teardown stops it like any +/// other parked actor. Spawned under `ROOT_PID` so its exit signal is absorbed +/// rather than delivered to whichever supervisor's child happened to join +/// first. +/// +/// Under `cluster` the same actor is the node's **pg actor** (RFC 010 Phase +/// 5, c15): it also drains a control inbox of local join/leave announcements, +/// the membership stream and the exposed `"pg"` inbox — see +/// [`crate::cluster::pg`]. Its store-side sweep is unchanged. +#[cfg(not(feature = "cluster"))] +fn reaper(rx: crate::channel::Receiver, ctl: crate::channel::Receiver) { + // No mesh: nothing to tell about joins/leaves. Drop the control inbox + // so announcements are refused at the sender rather than queued. + drop(ctl); + while let Ok(down) = rx.recv() { + sweep_local_death(down.pid); + // Evicted memberships hold only ids; their monitors have fired. + } +} + +/// What the local API tells the reaper besides deaths (which arrive as +/// [`Down`] on their own inbox — that channel's type is fixed by the monitor +/// primitive, so the two cannot be one enum). The default reaper has no use +/// for these; the cluster's pg actor broadcasts them (RFC 010 Phase 5). +// The default reaper never looks inside — that is the point, not a bug. +#[cfg_attr(not(feature = "cluster"), allow(dead_code))] +pub(crate) enum PgEvent { + /// `join` inserted `pid` into `group`. The consumer re-checks the store + /// before acting on it. + Joined { group: String, pid: Pid }, + /// `leave` removed `pid` from `group`. + Left { group: String, pid: Pid }, + /// `cluster::start` has the manager up and the local identity set: take + /// a membership subscription, register + expose the `"pg"` name, and + /// start speaking to peers. + #[cfg(feature = "cluster")] + Attach, +} + +/// Evict the local member `pid` from every group. The reaper's one store +/// operation; returns what was evicted with its group (the cluster's +/// `Leave` broadcast wants both). +pub(crate) fn sweep_local_death(pid: Pid) -> Vec<(String, Membership)> { + with_runtime(|inner| { + let node = inner.node_id; + inner + .process_groups + .lock() + .remove_where(|m| m.node == node && m.pid == pid) + }) +} + +/// The reaper's inboxes. `deaths` is the liveness authority for the set +/// (`ctl` is created and dropped with it, on the same actor). +#[derive(Clone)] +pub(crate) struct ReaperInboxes { + pub(crate) deaths: Sender, + /// The control inbox: local `join`/`leave` announce here (see + /// [`PgEvent`]). The default reaper closes it on entry. + pub(crate) ctl: Sender, +} + +impl ReaperInboxes { + fn alive(&self) -> bool { + self.deaths.receiver_alive() + } +} + +/// Live senders for the reaper's inboxes, spawning the reaper if this run has +/// none yet. Two racing first-spawns may both spawn; the loser's senders drop +/// on return, its spare reaper sees a closed inbox and exits. +pub(crate) fn reaper_inboxes() -> ReaperInboxes { + let existing = with_runtime(|inner| { + let pg = inner.process_groups.lock(); + pg.reaper.clone().filter(ReaperInboxes::alive) + }); + if let Some(r) = existing { + return r; + } + let (tx, rx) = channel::(); + let (ctl_tx, ctl_rx) = channel::(); + // Detached: the handle drops here. The reaper's lifetime is the run's. + // The ONE seam between the local store and the cluster: same inboxes, + // different body. + #[cfg(not(feature = "cluster"))] + let _ = spawn_under(crate::runtime::ROOT_PID, move || reaper(rx, ctl_rx)); + #[cfg(feature = "cluster")] + let _ = spawn_under(crate::runtime::ROOT_PID, move || { + crate::cluster::pg::actor(rx, ctl_rx) + }); + let fresh = ReaperInboxes { + deaths: tx, + ctl: ctl_tx, + }; + with_runtime(|inner| { + let mut pg = inner.process_groups.lock(); + match &pg.reaper { + Some(r) if r.alive() => r.clone(), + _ => { + pg.reaper = Some(fresh.clone()); + fresh + } + } + }) +} + +/// A live sender for the reaper's `deaths` inbox (spawning it if needed). +fn deaths_sender() -> Sender { + reaper_inboxes().deaths +} + +/// Announce a local group change to the reaper, if this run has one. A +/// closed inbox is the default reaper (uninterested) or a run tearing down. +fn announce(msg: PgEvent) { + let ctl = with_runtime(|inner| { + inner + .process_groups + .lock() + .reaper + .as_ref() + .map(|r| r.ctl.clone()) + }); + if let Some(ctl) = ctl { + let _ = ctl.send(msg); + } +} + /// Build the full member identity for `pid` from runtime identity. -fn member_for(inner: &crate::runtime::RuntimeInner, pid: Pid) -> Member { +pub(crate) fn member_for(inner: &crate::runtime::RuntimeInner, pid: Pid) -> Member { Member { node: inner.node_id, incarnation: inner.incarnation, @@ -358,7 +548,7 @@ fn member_for(inner: &crate::runtime::RuntimeInner, pid: Pid) -> Member { /// no lock — identical to the registry's guard. The read-path backstop: a /// generation is never reused, so a dead member is detectable independently of /// whether its monitor `Down` has been drained yet. -fn live(inner: &crate::runtime::RuntimeInner, pid: Pid) -> bool { +pub(crate) fn live(inner: &crate::runtime::RuntimeInner, pid: Pid) -> bool { inner.slot_at(pid).is_some_and(|s| s.is_live_for(pid)) } @@ -367,61 +557,76 @@ fn live(inner: &crate::runtime::RuntimeInner, pid: Pid) -> bool { /// added the membership, `false` if it was already a member. /// /// Installs a monitor on `pid` so the actor's death evicts it from the group -/// automatically — you never have to remove a dead member yourself. A redundant -/// (idempotent) join tears its extra monitor back down. +/// automatically — you never have to remove a dead member yourself. Joining a +/// pid that is already dead is accepted and evicted the same way (via a +/// `NoProc` notice), so it never shows up in a read. /// /// Panics if called outside `Runtime::run()`. pub fn join(group: impl Into, pid: Pid) -> bool { let group = group.into(); let pid = pid.erase(); - // Install the monitor BEFORE taking the group lock: monitor() acquires the - // target's cold lock (Leaf), and two Leaf locks are never held at once. The - // registration races `finalize_actor` under that cold lock exactly as every - // other monitor does, so no death can slip between the join and the monitor - // being in place. - let mon = monitor(pid); - - let (rejected, reaped) = with_runtime(|inner| { + let deaths = deaths_sender(); + // Record the membership BEFORE arming its monitor: the reaper sweeps by + // pid on the first `Down`, so a `Down` that could precede the entry would + // leave a corpse in storage forever (visible to no read — the backstop + // hides it — but a leak, and once groups are clustered a member that + // would be announced). Arming after insertion means every `Down` finds + // its entry. The monitor id is allocated up front so `leave` can tear the + // registration down even if it lands in the tiny window before arming (an + // orphaned registration is harmless: its `Down` names a pid whose + // membership is gone, and the sweep finds nothing). + let id = with_runtime(|inner| inner.alloc_monitor_id()); + let inserted = with_runtime(|inner| { let ms = Membership { member: member_for(inner, pid), - monitor: mon, + monitor: Some(id), }; - let mut pg = inner.process_groups.lock(); - let reaped = pg.reap_group(&group); - let rejected = pg.join(&group, ms); - (rejected, reaped) + inner.process_groups.lock().join(&group, ms) + }); + if !inserted { + return false; + } + // Tell the reaper (the cluster's pg actor re-checks the store before it + // broadcasts, so a `leave`/death that overtakes this announcement is + // never advertised as a join). + announce(PgEvent::Joined { + group: group.clone(), + pid, }); - // Outside the group lock: drop the reaped (dead) monitors, and if this join - // was redundant, demonitor + drop the extra monitor we just installed. - drop(reaped); - match rejected { - Some(dup) => { - demonitor(&dup.monitor); - false - } - None => true, + // Outside the group lock: registration takes the target's cold lock (Leaf). + // The registration races `finalize_actor` under that cold lock exactly as + // every other monitor does, so no death can slip between the join and the + // monitor being in place. + if !register_monitor(pid, id, &deaths) { + // Already gone: queue the notice ourselves, exactly as `monitor` does. + let _ = deaths.send(Down { + pid, + reason: DownReason::NoProc, + }); } + true } /// Drop `pid`'s membership of `group`. Returns whether a membership was -/// removed. The membership's monitor is demonitored and dropped. +/// removed. The membership's monitor registration is torn down. /// /// Panics if called outside `Runtime::run()`. pub fn leave(group: &str, pid: Pid) -> bool { let pid = pid.erase(); - let (removed, reaped) = with_runtime(|inner| { + let removed = with_runtime(|inner| { let member = member_for(inner, pid); - let mut pg = inner.process_groups.lock(); - let reaped = pg.reap_group(group); - let removed = pg.leave(group, member); - (removed, reaped) + inner.process_groups.lock().leave(group, member) }); - - drop(reaped); match removed { Some(ms) => { - demonitor(&ms.monitor); + if let Some(id) = ms.monitor { + unregister_monitor(pid, id); + } + announce(PgEvent::Left { + group: group.to_owned(), + pid, + }); true } None => false, @@ -431,21 +636,19 @@ pub fn leave(group: &str, pid: Pid) -> bool { /// Every live member of `group`, in the order they joined. Returns an empty /// vector if the group does not exist or has no live members. /// -/// Dead members are never returned: the group is pruned of anything that has -/// died before the read, and as a backstop a member whose slot is already dead -/// is dropped from the result even in the brief window before its death has -/// been fully processed. +/// Dead members are never returned: the reaper evicts a member as soon as its +/// death is processed, and as a backstop a member whose slot is already dead +/// is dropped from the result even in the brief window before the reaper's +/// turn. /// /// Panics if called outside `Runtime::run()`. pub fn members(group: &str) -> Vec { - let (pids, reaped) = with_runtime(|inner| { - let mut pg = inner.process_groups.lock(); - let reaped = pg.reap_group(group); - let pids = pg.members_where(group, |pid| live(inner, pid)); - (pids, reaped) - }); - drop(reaped); - pids + with_runtime(|inner| { + inner + .process_groups + .lock() + .members_where(group, inner.node_id, |pid| live(inner, pid)) + }) } /// One live member of `group`, or `None` if the group is empty (or every @@ -455,14 +658,12 @@ pub fn members(group: &str) -> Vec { /// /// Panics if called outside `Runtime::run()`. pub fn pick(group: &str) -> Option { - let (picked, reaped) = with_runtime(|inner| { - let mut pg = inner.process_groups.lock(); - let reaped = pg.reap_group(group); - let picked = pg.first_member_where(group, |pid| live(inner, pid)); - (picked, reaped) - }); - drop(reaped); - picked + with_runtime(|inner| { + inner + .process_groups + .lock() + .first_member_where(group, inner.node_id, |pid| live(inner, pid)) + }) } /// Typed [`pick`]: one live member of `group` as a [`Pid`](Pid). @@ -505,8 +706,8 @@ pub fn dispatch(group: &str, msg: A::Msg) -> Result, Send #[cfg(test)] mod tests { use super::*; - use crate::channel::{channel, Sender}; - use crate::monitor::{Down, DownReason, MonitorId}; + use crate::scheduler::spawn; + use std::time::{Duration, Instant}; fn member(index: u32, generation: u32) -> Member { Member { @@ -516,33 +717,21 @@ mod tests { } } - /// A synthetic membership with a real (but slot-less) monitor channel. The - /// returned `Sender` stands in for the slot's `Down` sender: hold it to - /// keep the member "alive" (`try_recv` → `Ok(None)`), `send` a `Down` to - /// simulate death, or `drop` it to simulate a drained/closed channel. - fn synth(index: u32, generation: u32) -> (Membership, Sender) { - let pid = Pid::new(index, generation); - let (tx, rx) = channel::(); - let ms = Membership { + /// A synthetic membership: the store never looks at the id. + fn synth(index: u32, generation: u32) -> Membership { + Membership { member: member(index, generation), - monitor: Monitor { - id: MonitorId(0), - target: pid, - rx, - }, - }; - (ms, tx) + monitor: Some(MonitorId(0)), + } } #[test] fn join_is_idempotent_within_a_group() { let mut pg = ProcessGroups::new(); - let (a, _ta) = synth(1, 0); - let (b, _tb) = synth(1, 0); - assert!(pg.join("workers", a).is_none(), "first join inserts"); + assert!(pg.join("workers", synth(1, 0)), "first join inserts"); assert!( - pg.join("workers", b).is_some(), - "second identical join is handed back" + !pg.join("workers", synth(1, 0)), + "second identical join is refused" ); assert_eq!(pg.members_of("workers"), vec![member(1, 0)]); } @@ -550,12 +739,9 @@ mod tests { #[test] fn same_pid_in_many_groups_is_independent() { let mut pg = ProcessGroups::new(); - let (a, _ta) = synth(1, 0); - let (b, _tb) = synth(1, 0); - let (c, _tc) = synth(2, 0); - pg.join("a", a); - pg.join("b", b); - pg.join("b", c); + pg.join("a", synth(1, 0)); + pg.join("b", synth(1, 0)); + pg.join("b", synth(2, 0)); assert_eq!(pg.members_of("a"), vec![member(1, 0)]); assert_eq!(pg.members_of("b"), vec![member(1, 0), member(2, 0)]); } @@ -564,11 +750,9 @@ mod tests { fn distinct_generations_are_distinct_members() { // ABA guard: same slot index, different generation = different actor. let mut pg = ProcessGroups::new(); - let (a, _ta) = synth(1, 0); - let (b, _tb) = synth(1, 1); - assert!(pg.join("g", a).is_none()); + assert!(pg.join("g", synth(1, 0))); assert!( - pg.join("g", b).is_none(), + pg.join("g", synth(1, 1)), "different generation is a distinct member" ); assert_eq!(pg.members_of("g"), vec![member(1, 0), member(1, 1)]); @@ -577,10 +761,8 @@ mod tests { #[test] fn leave_removes_one_membership_and_prunes_empty_groups() { let mut pg = ProcessGroups::new(); - let (a, _ta) = synth(1, 0); - let (b, _tb) = synth(2, 0); - pg.join("g", a); - pg.join("g", b); + pg.join("g", synth(1, 0)); + pg.join("g", synth(2, 0)); assert!(pg.leave("g", member(1, 0)).is_some()); assert_eq!(pg.members_of("g"), vec![member(2, 0)]); assert!( @@ -598,7 +780,7 @@ mod tests { #[test] fn remove_where_sweeps_every_group() { let mut pg = ProcessGroups::new(); - for (g, (m, _t)) in [ + for (g, m) in [ ("a", synth(1, 0)), ("a", synth(2, 0)), ("b", synth(1, 0)), @@ -616,104 +798,137 @@ mod tests { #[test] fn remove_where_can_match_an_incarnation_sweep() { - // Shape check for the later evict_incarnation(node, inc) caller. + // Shape check for the node-down / incarnation sweep caller. let mut pg = ProcessGroups::new(); - let pid = Pid::new(1, 0); - let (tx, rx) = channel::(); - let dead = Membership { + let stale = Membership { member: Member { node: DEFAULT_NODE_ID, incarnation: Incarnation::new(7), - pid, - }, - monitor: Monitor { - id: MonitorId(0), - target: pid, - rx, + pid: Pid::new(1, 0), }, + monitor: Some(MonitorId(0)), }; - let _keep = tx; - let (live, _tl) = synth(2, 0); - pg.join("g", dead); - pg.join("g", live); + pg.join("g", stale); + pg.join("g", synth(2, 0)); let evicted = pg.remove_where(|mem| mem.incarnation == Incarnation::new(7)); assert_eq!(evicted.len(), 1); assert_eq!(pg.members_of("g"), vec![member(2, 0)]); } #[test] - fn reap_keeps_live_members() { + fn read_backstop_hides_a_member_the_reaper_has_not_yet_swept() { let mut pg = ProcessGroups::new(); - let (a, _ta) = synth(1, 0); // sender held: member stays alive - pg.join("a", a); - assert!(pg.reap_group("a").is_empty(), "no deaths"); - assert_eq!(pg.members_of("a"), vec![member(1, 0)]); - } - - #[test] - fn reap_evicts_a_dead_member_and_sweeps_all_its_groups() { - let mut pg = ProcessGroups::new(); - let (a1, ta1) = synth(1, 0); // pid 1 in group a - let (a2, _ta2) = synth(2, 0); // pid 2 in group a (stays alive) - let (b1, _tb1) = synth(1, 0); // pid 1 in group b - pg.join("a", a1); - pg.join("a", a2); - pg.join("b", b1); - // pid 1 dies: its group-a monitor receives a Down. Its group-b monitor - // has not — reap must still sweep pid 1 out of b by the pid predicate. - ta1.send(Down { - pid: Pid::new(1, 0), - reason: DownReason::Exit, - }) - .unwrap(); - let evicted = pg.reap_group("a"); - assert_eq!( - evicted.len(), - 2, - "pid 1's memberships in both a and b are evicted" - ); - assert_eq!(pg.members_of("a"), vec![member(2, 0)]); - assert!(pg.members_of("b").is_empty(), "swept from b too; pruned"); - } - - #[test] - fn reap_treats_a_closed_channel_as_dead() { - let mut pg = ProcessGroups::new(); - let (a, ta) = synth(1, 0); - pg.join("a", a); - drop(ta); // sender gone, queue empty → try_recv = Err(RecvError) = dead - let evicted = pg.reap_group("a"); - assert_eq!(evicted.len(), 1); - assert!(pg.members_of("a").is_empty()); - } - - #[test] - fn read_backstop_hides_a_member_the_monitor_has_not_yet_reaped() { - let mut pg = ProcessGroups::new(); - // Both senders held: reap_group would see Ok(None) and evict neither. - let (a, _ta) = synth(1, 0); - let (b, _tb) = synth(2, 0); - pg.join("g", a); - pg.join("g", b); + pg.join("g", synth(1, 0)); + pg.join("g", synth(2, 0)); // The slot-word oracle already reports pid 1 dead (finalize window), - // ahead of any Down delivery. + // ahead of the reaper's turn. let dead = Pid::new(1, 0); let oracle = |pid: Pid| pid != dead; assert_eq!( - pg.members_where("g", oracle), + pg.members_where("g", DEFAULT_NODE_ID, oracle), vec![Pid::new(2, 0)], "dead pid filtered from read" ); assert_eq!( - pg.first_member_where("g", oracle), + pg.first_member_where("g", DEFAULT_NODE_ID, oracle), Some(Pid::new(2, 0)), "pick skips the dead first member" ); - // Backstop does not evict — that stays the monitor's job; raw storage - // still holds both until reap runs. + // Backstop does not evict — that stays the reaper's job; raw storage + // still holds both until it runs. assert_eq!(pg.members_of("g"), vec![member(1, 0), member(2, 0)]); } + + // ---- reaper: eager eviction against a live runtime ---- + + /// Raw storage view for a group, bypassing the read-path backstop. + fn stored(group: &str) -> Vec { + with_runtime(|inner| inner.process_groups.lock().members_of(group)) + } + + /// Cooperative wait (`smarm::sleep`, never an OS block) until `pred`. + fn wait_until(what: &str, mut pred: impl FnMut() -> bool) { + let deadline = Instant::now() + Duration::from_secs(2); + while !pred() { + assert!(Instant::now() < deadline, "timed out waiting for: {what}"); + crate::sleep(Duration::from_millis(1)); + } + } + + #[test] + fn a_death_is_swept_from_storage_without_any_group_operation() { + crate::run(|| { + let (tx, rx) = channel::<()>(); + let w = spawn(move || { + rx.recv().unwrap(); + }); + let pid = w.pid(); + join("a", pid); + join("b", pid); + assert_eq!(stored("a"), vec![member_for_test(pid)]); + + tx.send(()).unwrap(); + w.join().unwrap(); + // No members()/pick()/join() on a or b from here on: the reaper + // alone must clear both. + wait_until("reaper sweeps a and b", || { + stored("a").is_empty() && stored("b").is_empty() + }); + }); + } + + #[test] + fn a_dead_at_join_pid_is_swept_from_storage() { + crate::run(|| { + let h = spawn(|| {}); + let pid = h.pid(); + h.join().unwrap(); + assert!(join("late", pid), "join is accepted; eviction is uniform"); + wait_until("reaper sweeps the NoProc member", || { + stored("late").is_empty() + }); + }); + } + + #[test] + fn leave_then_death_does_not_disturb_a_rejoined_group() { + // A monitor unregistered by `leave` must not fire later; the pid's + // fresh membership after re-join is swept exactly once, by its own + // monitor, on death. + crate::run(|| { + let (tx, rx) = channel::<()>(); + let w = spawn(move || { + rx.recv().unwrap(); + }); + let pid = w.pid(); + join("g", pid); + assert!(leave("g", pid)); + assert!(join("g", pid)); + assert_eq!(members("g"), vec![pid]); + tx.send(()).unwrap(); + w.join().unwrap(); + wait_until("reaper sweeps g", || stored("g").is_empty()); + }); + } + + #[test] + fn reaper_is_respawned_for_a_second_run_of_the_same_runtime() { + let rt = crate::runtime::init(crate::runtime::Config::exact(1)); + let body = || { + let h = spawn(|| {}); + let pid = h.pid(); + h.join().unwrap(); + join("g", pid); + wait_until("reaper sweeps g", || stored("g").is_empty()); + }; + rt.run(body); + rt.run(body); + } + + fn member_for_test(pid: Pid) -> Member { + with_runtime(|inner| member_for(inner, pid)) + } } diff --git a/src/pid.rs b/src/pid.rs index 729d67e..564363a 100644 --- a/src/pid.rs +++ b/src/pid.rs @@ -292,3 +292,44 @@ mod typed_pid_tests { assert_send_sync::>(); } } + +// ---- RFC 010 c10: pids auto-serialize (cluster feature) --------------------- + +/// A local `Pid` serializes as a +/// [`RemotePid`](crate::cluster::remote::RemotePid): the wire form stamps +/// this node's name and incarnation from the ambient runtime, so a pid can +/// sit inside any message field and reply-to needs no ceremony (RFC 010 §3, +/// "sugar not a bear trap"). Serializing a pid also marks it **watchable** +/// — the wire crossing is the cluster's `mark_watchable` set-site (D12), the +/// exact analog of the membrane crossing. +/// +/// Must run inside `run()` (the ambient identity lives on the runtime); a +/// runtime without a cluster identity cannot serialize a pid at all — it is +/// a serialize error, surfacing as the send's `Encode` failure — rather than +/// a `("", 0)` stamp that every peer would silently drop. +#[cfg(feature = "cluster")] +impl serde::Serialize for Pid { + fn serialize(&self, s: S) -> Result { + crate::cluster::remote::RemotePid::::from_local(*self) + .ok_or_else(|| serde::ser::Error::custom("pid serialized with no local node identity"))? + .serialize(s) + } +} + +/// Deserializing into a `Pid` is the **collapse**: it succeeds only when +/// the wire pid names this very node (name and incarnation both), and is a +/// decode error otherwise — a foreign pid cannot become a local `Pid`. +/// Fields that may hold a pid from anywhere are `RemotePid`. +#[cfg(feature = "cluster")] +impl<'de, A: 'static> serde::Deserialize<'de> for Pid { + fn deserialize>(d: D) -> Result { + let rp = crate::cluster::remote::RemotePid::::deserialize(d)?; + rp.local().ok_or_else(|| { + serde::de::Error::custom(format!( + "pid {}@{} is not local to this node", + rp.index(), + rp.node() + )) + }) + } +} diff --git a/src/registry.rs b/src/registry.rs index 24b1496..a98de0d 100644 --- a/src/registry.rs +++ b/src/registry.rs @@ -215,6 +215,7 @@ impl std::error::Error for SendError {} trait ErasedSender: Send { fn as_any(&self) -> &dyn Any; fn queued_len(&self) -> usize; + fn receiver_alive(&self) -> bool; } impl ErasedSender for Sender { @@ -224,6 +225,9 @@ impl ErasedSender for Sender { fn queued_len(&self) -> usize { Sender::queued_len(self) } + fn receiver_alive(&self) -> bool { + Sender::receiver_alive(self) + } } /// One typed channel of an actor, type-erased. Concretely a `Sender` filed @@ -442,6 +446,18 @@ pub(crate) fn register_with( /// name) and [`install`] (which does not). A leftover mailbox at this slot /// index from a dead prior incarnation (pid mismatch) is replaced wholesale. /// Caller holds the registry lock and has established that `me` is live. +/// +/// **One channel per message type per actor.** Publishing a second `M` +/// channel on the same live actor replaces the first — and if the first's +/// receiver is still alive, that replacement drops its last sender, closing +/// it, and any `recv`/`select` on it then returns "closed" immediately and +/// forever: a silent hot loop that starves the scheduler. That is never +/// intended, so it panics here (found the hard way in RFC 010 c9, where two +/// `Name`s registered on one actor did exactly this). Replacing a +/// channel whose receiver is already gone is fine (an actor re-registering +/// after dropping its old inbox) and stays silent. To hold two names of the +/// same type, register them from two actors, or bind both names to one +/// cloned sender. fn publish_channel(reg: &mut Registry, me: Pid, tx: Sender) { let mb = reg .by_index @@ -450,6 +466,16 @@ fn publish_channel(reg: &mut Registry, me: Pid, tx: Sender if mb.pid != me { *mb = Mailbox::new(me); } + if let Some(existing) = mb.channels.get(&TypeId::of::()) { + assert!( + !existing.sender.receiver_alive() || same_channel::(existing, &tx), + "smarm: actor {me:?} already publishes a live channel for message type `{}`; \ + a second one would replace and CLOSE the first (its receiver would then \ + read as closed forever). Register the second name from another actor, or \ + bind both names to a clone of the same sender.", + type_name::() + ); + } mb.channels.insert( TypeId::of::(), Channel { @@ -459,6 +485,17 @@ fn publish_channel(reg: &mut Registry, me: Pid, tx: Sender ); } +/// True if `existing` and `tx` are senders of the very same channel (a +/// cloned sender bound under a second name is the sanctioned way to hold two +/// names of one type on one actor). +fn same_channel(existing: &Channel, tx: &Sender) -> bool { + existing + .sender + .as_any() + .downcast_ref::>() + .is_some_and(|old| old.same_channel(tx)) +} + /// Publish the current actor's `Sender` into its mailbox **without** /// binding a name, and hand back the typed [`Pid`] that addresses this /// actor directly. diff --git a/src/runtime.rs b/src/runtime.rs index 41ae326..6a470bd 100644 --- a/src/runtime.rs +++ b/src/runtime.rs @@ -1041,6 +1041,17 @@ pub(crate) struct RuntimeInner { /// checks under it read only the atomic slot word, and the eviction path /// keeps it off the send path. pub(crate) process_groups: RawMutex, + /// RFC 010 c8: the exposure registry (exposed names + type-hash decoders). + /// RawMutex Leaf, same discipline as `process_groups`; decoders run under + /// it and are leaf-only by contract (they decode and send — `send_dyn` + /// takes `registry`, never this). cfg-gated: zero-cost-when-off (c1). + #[cfg(feature = "cluster")] + pub(crate) exposure: RawMutex, + /// RFC 010 c9: the outbound table (node name → the connection's + /// dedicated `Sender`), manager-maintained. Leaf; the send happens + /// outside the lock. cfg-gated like `exposure`. + #[cfg(feature = "cluster")] + pub(crate) outbound: RawMutex, /// Recycled stacks waiting to be reused by the next spawn. pub(crate) stack_pool: RawMutex>, /// Maximum number of stacks to retain in the pool. @@ -1099,6 +1110,10 @@ impl RuntimeInner { node_id, incarnation, process_groups: RawMutex::new(crate::pg::ProcessGroups::new()), + #[cfg(feature = "cluster")] + exposure: RawMutex::new(crate::cluster::expose::ExposureState::new()), + #[cfg(feature = "cluster")] + outbound: RawMutex::new(crate::cluster::remote::Outbound::new()), stack_pool: RawMutex::new(Vec::new()), stack_pool_cap, stack_reserve: crate::stack::round_to_pages(stack_reserve), @@ -1423,6 +1438,9 @@ impl Runtime { // done" — every remaining top-level actor is asked to shut down (see // `finalize_actor` / `shutdown_forest_roots`). self.inner.set_root(initial_handle.pid()); + // A previous run's group reaper was stopped with that run; forget it + // so the first `join` of this run spawns a fresh one. + self.inner.process_groups.lock().reset_reaper(); // Launch N-1 extra scheduler threads, named `smarm-sched-{slot}` so // they are identifiable in `/proc//task/*/comm`, stack dumps and diff --git a/src/trace.rs b/src/trace.rs index 6a48270..66fe69b 100644 --- a/src/trace.rs +++ b/src/trace.rs @@ -80,6 +80,13 @@ mod inner { // RFC 005 wake slot SlotPush(Pid), // actor-context wake parked in the waking thread's slot SlotPop(Pid), // scheduler resumed a pid from its own slot + // Cluster (RFC 010): the conn actor's verdict on one inbound frame — + // local knowledge only, never on the wire; the label is + // `InboundVerdict::label()`. No pid: a refused frame has none. + ClusterInbound(&'static str), + // Cluster (RFC 010): the connector's verdict on one dial attempt — + // `"ok"` or `DialError::label()`. No pid. + ClusterDial(&'static str), } // ----------------------------------------------------------------------- @@ -291,6 +298,8 @@ mod inner { Event::Dequeue(p) => ("dequeue".into(), p.index()), Event::SlotPush(p) => ("slot_push".into(), p.index()), Event::SlotPop(p) => ("slot_pop".into(), p.index()), + Event::ClusterInbound(v) => (format!("cluster_inbound {v}"), 0), + Event::ClusterDial(v) => (format!("cluster_dial {v}"), 0), } } diff --git a/tests/cluster_conn_lifecycle.rs b/tests/cluster_conn_lifecycle.rs new file mode 100644 index 0000000..f989ddc --- /dev/null +++ b/tests/cluster_conn_lifecycle.rs @@ -0,0 +1,117 @@ +//! RFC 010 c6a — connection-actor lifecycle against the manager table. +//! +//! The handshake is bypassed here (c6b wires it): each connection is +//! constructed already-established over a real localhost TCP pair, handed a +//! fabricated `Peer`, and spawned. `spawn_established` registers it with the +//! manager, which takes its handle and monitors it, so the table reflects the +//! connection while it lives and reaps it on any exit path. This proves three +//! things at once: a live connection shows up, a commanded `Disconnect` +//! removes exactly that one, and a peer close (EOF, no command) removes the +//! other. +//! +//! TCP parks the calling actor, so everything runs inside `smarm::run`; the +//! single-threaded runtime is fine because every wait is a cooperative fd park. +#![cfg(feature = "cluster")] + +use std::time::Duration; + +use smarm::cluster::envelope::NodeMeta; +use smarm::cluster::handshake::Peer; +use smarm::cluster::manager::{Call, Manager, Reply, MANAGER}; +use smarm::cluster::spawn_established; +use smarm::cluster::transport::tcp::TcpTransport; +use smarm::cluster::transport::{Conn, FramedConn, Transport}; +use smarm::cluster::Timing; +use smarm::gen_server::{self, GenServerBuilder}; +use smarm::pg::Incarnation; +use smarm::{run, sleep}; + +/// A fabricated post-handshake peer identity. Only `node_name` matters to the +/// manager table; the rest is filler until c7 consumes it. +fn peer(name: &str) -> Peer { + Peer { + node_name: name.to_string(), + incarnation: Incarnation::new(1), + meta: NodeMeta { + role: "test".to_string(), + region: "test".to_string(), + }, + } +} + +/// One established transport pair over localhost. Relies on TCP backlog so the +/// sequential dial-then-accept needs no concurrent acceptor (same assumption as +/// the c3 conformance suite). +fn pair(t: &dyn Transport) -> (Box, Box) { + let mut l = t.listen("127.0.0.1:0").unwrap(); + let a = t.dial(&l.local_addr()).unwrap(); + let b = l.accept().unwrap(); + (a, b) +} + +/// Poll the manager until its peer set matches `expected` (sorted), or fail. +/// The bound is generous against a sub-millisecond real cost. +fn wait_peers(expected: &[&str]) { + let want: Vec = expected.iter().map(|s| s.to_string()).collect(); + for _ in 0..2000 { + if let Ok(Reply::Peers(got)) = gen_server::call(MANAGER, Call::Peers) { + if got == want { + return; + } + } + sleep(Duration::from_millis(1)); + } + let got = gen_server::call(MANAGER, Call::Peers); + panic!("timed out waiting for peers == {want:?}; last = {got:?}"); +} + +#[test] +fn connection_up_commanded_shutdown_and_eof_all_reflected_in_table() { + run(|| { + // The manager, started plainly and reachable at its well-known name. + // (The supervised subtree in `cluster::start` is permanent by design; + // a plainly-started manager lets this test terminate cleanly.) + let mgr = GenServerBuilder::new(Manager::new()) + .named(MANAGER) + .start() + .expect("manager name is free"); + + let t = TcpTransport; + let (a1, b1) = pair(&t); + let (a2, b2) = pair(&t); + + // Manage the `a` ends as peers node-b and node-c; keep the `b` far ends + // open so neither socket is closed from the far side yet. + spawn_established(FramedConn::new(a1), peer("node-b"), Timing::default()) + .expect("node-b registers"); + spawn_established(FramedConn::new(a2), peer("node-c"), Timing::default()) + .expect("node-c registers"); + + // Up: both connections register and the table shows them. + wait_peers(&["node-b", "node-c"]); + + // A commanded disconnect reaps exactly its own connection: the + // manager drops that entry's handle and the actor stops. + assert!(matches!( + gen_server::call( + MANAGER, + Call::Disconnect { + name: "node-b".to_string() + } + ), + Ok(Reply::Disconnected) + )); + wait_peers(&["node-c"]); + + // A peer close (EOF) reaps the other with no command at all. + drop(b2); + wait_peers(&[]); + + // node-b's far end stayed open until here, so its removal above was the + // disconnect command and not an EOF. + drop(b1); + + // All connection actors have exited; stop the manager so `run` returns. + mgr.shutdown(); + }); +} diff --git a/tests/cluster_conn_liveness.rs b/tests/cluster_conn_liveness.rs new file mode 100644 index 0000000..5104efc --- /dev/null +++ b/tests/cluster_conn_liveness.rs @@ -0,0 +1,180 @@ +//! RFC 010 c6c — heartbeat send + fixed-timeout liveness + teardown. +//! +//! Each case runs one real connection actor over an in-process localhost TCP +//! pair, with the far end held as a raw `FramedConn` (no actor) so the test +//! controls exactly what — if anything — the peer says. That gives the three +//! protocol-visible facts direct handles: heartbeats appear on the wire +//! unprompted; a mute peer is torn down (and reaped from the manager table) +//! once `LIVENESS_TIMEOUT` empties; and a peer that does nothing but send +//! heartbeats keeps the connection alive past that same window. +//! +//! Loopback has no fd and cannot drive liveness (documented on the actor), +//! so everything here is TCP. TCP parks the calling actor, so everything +//! runs inside `smarm::run`. +#![cfg(feature = "cluster")] + +use std::time::{Duration, Instant}; + +use smarm::cluster::conn::{HEARTBEAT_INTERVAL, LIVENESS_TIMEOUT}; +use smarm::cluster::envelope::{Frame, NodeMeta}; +use smarm::cluster::handshake::Peer; +use smarm::cluster::manager::{Call, Manager, Reply, MANAGER}; +use smarm::cluster::spawn_established; +use smarm::cluster::transport::tcp::TcpTransport; +use smarm::cluster::transport::{Conn, FramedConn, Transport}; +use smarm::cluster::Timing; +use smarm::gen_server::{self, GenServerBuilder}; +use smarm::pg::Incarnation; +use smarm::{run, sleep, spawn}; + +/// A fabricated post-handshake peer identity (same shape as the c6a suite). +fn peer(name: &str) -> Peer { + Peer { + node_name: name.to_string(), + incarnation: Incarnation::new(1), + meta: NodeMeta { + role: "test".to_string(), + region: "test".to_string(), + }, + } +} + +/// One established transport pair over localhost (TCP backlog covers the +/// sequential dial-then-accept, as in the c3 conformance suite). +fn pair(t: &dyn Transport) -> (Box, Box) { + let mut l = t.listen("127.0.0.1:0").unwrap(); + let a = t.dial(&l.local_addr()).unwrap(); + let b = l.accept().unwrap(); + (a, b) +} + +fn peers() -> Vec { + match gen_server::call(MANAGER, Call::Peers) { + Ok(Reply::Peers(p)) => p, + other => panic!("manager unreachable: {other:?}"), + } +} + +/// Poll until the manager's peer set matches `expected` (sorted) or `budget` +/// runs out. +fn wait_peers(expected: &[&str], budget: Duration) { + let want: Vec = expected.iter().map(|s| s.to_string()).collect(); + let deadline = Instant::now() + budget; + while Instant::now() < deadline { + if peers() == want { + return; + } + sleep(Duration::from_millis(10)); + } + panic!( + "timed out waiting for peers == {want:?}; last = {:?}", + peers() + ); +} + +/// The actor emits heartbeats unprompted: the raw far end, saying nothing, +/// sees a `Frame::Heartbeat` well within one interval (the first goes out at +/// spawn). +#[test] +fn heartbeats_are_sent_unprompted() { + run(|| { + let mgr = GenServerBuilder::new(Manager::new()) + .named(MANAGER) + .start() + .expect("manager name is free"); + + let (a, b) = pair(&TcpTransport); + spawn_established(FramedConn::new(a), peer("hb-send"), Timing::default()) + .expect("register"); + let mut far = FramedConn::new(b); + + let frame = far + .recv_deadline(Instant::now() + HEARTBEAT_INTERVAL) + .expect("a heartbeat before one interval elapses"); + assert_eq!(frame, Some(Frame::Heartbeat)); + + // Teardown: closing the far end is an EOF at the actor. + far.close(); + wait_peers(&[], Duration::from_secs(2)); + mgr.shutdown(); + }); +} + +/// A mute peer is dead: no inbound frame for `LIVENESS_TIMEOUT` tears the +/// connection down and the manager's monitor reaps the table entry. The +/// entry is still present well inside the window — the teardown is the +/// timer, not an accident of setup. +#[test] +fn mute_peer_is_torn_down_after_liveness_timeout() { + run(|| { + let mgr = GenServerBuilder::new(Manager::new()) + .named(MANAGER) + .start() + .expect("manager name is free"); + + let (a, b) = pair(&TcpTransport); + spawn_established(FramedConn::new(a), peer("mute"), Timing::default()).expect("register"); + // Held open and silent: no frames, no EOF. (Unread inbound + // heartbeats sit in kernel buffers; they are 5 bytes each.) + let _far = FramedConn::new(b); + + // Well inside the window the connection is still up. + sleep(LIVENESS_TIMEOUT / 2); + assert_eq!(peers(), vec!["mute".to_string()], "torn down too early"); + + // ...and once the window empties it is gone. Generous budget over + // the remaining half-window. + wait_peers(&[], LIVENESS_TIMEOUT); + mgr.shutdown(); + }); +} + +/// Heartbeats alone keep a connection alive past `LIVENESS_TIMEOUT`: a far +/// end that sends `Frame::Heartbeat` at the interval (and nothing else) +/// holds the entry; when it goes quiet, liveness finally fires. +#[test] +fn heartbeats_keep_the_connection_alive() { + run(|| { + let mgr = GenServerBuilder::new(Manager::new()) + .named(MANAGER) + .start() + .expect("manager name is free"); + + let (a, b) = pair(&TcpTransport); + spawn_established(FramedConn::new(a), peer("kept"), Timing::default()).expect("register"); + + // The far heartbeat pump: interval-paced sends until told to stop, + // then holds the socket open, silent, so the eventual teardown is + // liveness — not EOF. + let (ctl_tx, ctl_rx) = smarm::channel::channel::<()>(); + spawn(move || { + let mut far = FramedConn::new(b); + // Phase 1: heartbeat at the interval until the first signal. + while matches!(ctl_rx.try_recv(), Ok(None)) { + far.send(&Frame::Heartbeat).expect("far send"); + sleep(HEARTBEAT_INTERVAL); + } + // Phase 2: silent but with the socket held open — dropping + // `far` here would EOF the actor and mask the liveness path. + // Exits when the test's closure ends and drops `ctl_tx` (an + // eternal park would stop `run` from ever returning). + while matches!(ctl_rx.try_recv(), Ok(None)) { + sleep(Duration::from_millis(20)); + } + }); + + // Past the liveness window with margin: still up. + sleep(LIVENESS_TIMEOUT + LIVENESS_TIMEOUT / 2); + assert_eq!( + peers(), + vec!["kept".to_string()], + "liveness fired despite heartbeats" + ); + + // Silence the pump; liveness now empties and the entry goes. + ctl_tx.send(()).expect("pump alive"); + wait_peers(&[], LIVENESS_TIMEOUT * 2); + mgr.shutdown(); + // `ctl_tx` drops here, releasing the pump's phase-2 wait. + }); +} diff --git a/tests/cluster_connect.rs b/tests/cluster_connect.rs new file mode 100644 index 0000000..20286da --- /dev/null +++ b/tests/cluster_connect.rs @@ -0,0 +1,482 @@ +//! RFC 010 c6b — the handshake on the accept/connect path. +//! +//! Path-level tests drive [`dial_handshake`]/[`accept_handshake`] over the +//! loopback transport on plain threads (its intended use — synchronous, no +//! runtime). Integration tests run the manager-backed [`dial`] and +//! [`spawn_acceptor`] over real localhost TCP inside `smarm::run`, and the +//! two-node case as subprocesses via the c4 harness. Flake budget: see +//! tests/common/mod.rs. +#![cfg(feature = "cluster")] + +mod common; + +use std::sync::mpsc; +use std::time::{Duration, Instant}; + +use common::{maybe_child, spawn_node, WAIT}; +use smarm::cluster::connect::{ + accept_handshake, dial, dial_handshake, spawn_acceptor, DialError, HandshakeError, + HANDSHAKE_TIMEOUT, +}; +use smarm::cluster::envelope::{Frame, NodeMeta, RejectReason}; +use smarm::cluster::handshake::{Local, PeerStanding}; +use smarm::cluster::manager::{Call, Manager, Reply, MANAGER}; +use smarm::cluster::transport::loopback::LoopbackTransport; +use smarm::cluster::transport::tcp::TcpTransport; +use smarm::cluster::transport::{FramedConn, Transport}; +use smarm::cluster::Timing; +use smarm::gen_server::{self, GenServerBuilder}; +use smarm::pg::Incarnation; +use smarm::{run, sleep}; + +const ROLES: &[(&str, fn())] = &[ + ("hs_listener", role_hs_listener), + ("hs_dialer", role_hs_dialer), +]; + +const HASH: u64 = 0xC6B0_C6B0_C6B0_C6B0; + +fn local(name: &str) -> Local { + Local { + node_name: name.into(), + incarnation: Incarnation::new(3), + build_hash: HASH, + meta: NodeMeta { + role: "test".into(), + region: "test".into(), + }, + } +} + +/// A loopback conn pair as `FramedConn`s, ready for a threaded handshake. +fn loopback_pair() -> (FramedConn, FramedConn) { + let t = LoopbackTransport::default(); + let mut l = t.listen("hs").unwrap(); + let dialer = FramedConn::new(t.dial("hs").unwrap()); + let accepted = FramedConn::new(l.accept().unwrap()); + (dialer, accepted) +} + +/// Far-future deadline for loopback paths, where it cannot fire anyway. +fn no_deadline() -> Instant { + Instant::now() + Duration::from_secs(3600) +} + +/// Park the node forever: it has announced everything the parent asserts on, +/// and must now hold its connection open until SIGKILLed. +fn park() -> ! { + loop { + sleep(Duration::from_secs(1)); + } +} + +/// Cooperative bounded receive across the closure/actor boundary. A blocking +/// `std::mpsc` wait would park the OS thread and starve the single-threaded +/// scheduler, so every wait inside `run` polls with [`sleep`] instead. +fn poll_recv(rx: &mpsc::Receiver, what: &str) -> T { + let deadline = Instant::now() + WAIT; + loop { + match rx.try_recv() { + Ok(v) => return v, + Err(mpsc::TryRecvError::Empty) => { + assert!(Instant::now() < deadline, "timed out waiting for {what}"); + sleep(Duration::from_millis(1)); + } + Err(mpsc::TryRecvError::Disconnected) => panic!("channel closed waiting for {what}"), + } + } +} + +// --------------------------------------------------------------------------- +// Path level, over loopback on plain threads +// --------------------------------------------------------------------------- + +#[test] +fn loopback_happy_path_establishes_both_ends() { + maybe_child(ROLES); + let (mut dialer, mut accepted) = loopback_pair(); + let responder = std::thread::spawn(move || { + accept_handshake( + &mut accepted, + local("node-b"), + |name| { + assert_eq!(name, "node-a"); + PeerStanding::Free + }, + no_deadline(), + ) + }); + let peer_of_dialer = dial_handshake(&mut dialer, &local("node-a"), no_deadline()).unwrap(); + let peer_of_acceptor = responder.join().unwrap().unwrap(); + assert_eq!(peer_of_dialer.node_name, "node-b"); + assert_eq!(peer_of_acceptor.node_name, "node-a"); +} + +#[test] +fn loopback_hash_mismatch_rejected_with_frame_then_eof() { + maybe_child(ROLES); + let (mut dialer, mut accepted) = loopback_pair(); + let mut wrong = local("node-b"); + wrong.build_hash ^= 1; + let responder = std::thread::spawn(move || { + accept_handshake(&mut accepted, wrong, |_| PeerStanding::Free, no_deadline()) + }); + // The dial side receives the reject frame — the compatibility anchor. + match dial_handshake(&mut dialer, &local("node-a"), no_deadline()) { + Err(HandshakeError::Rejected(RejectReason::HashMismatch)) => {} + other => panic!("expected HashMismatch reject, got {other:?}"), + } + match responder.join().unwrap() { + Err(HandshakeError::Rejected(RejectReason::HashMismatch)) => {} + other => panic!("expected accept side to report the reject, got {other:?}"), + } +} + +#[test] +fn loopback_tie_break_loser_closed_silently() { + maybe_child(ROLES); + // The inbound dial is from "node-z"; we are "node-a" with our own dial to + // node-z in flight. dial_wins("node-z", "node-a") is false, so the + // inbound loses: closed with no frame at all. + let (mut dialer, mut accepted) = loopback_pair(); + let responder = std::thread::spawn(move || { + accept_handshake( + &mut accepted, + local("node-a"), + |_| PeerStanding::Dialing, + no_deadline(), + ) + }); + // Silent close: the dial side sees EOF, never a frame. + match dial_handshake(&mut dialer, &local("node-z"), no_deadline()) { + Err(HandshakeError::Closed) => {} + other => panic!("expected silent close (Closed), got {other:?}"), + } + match responder.join().unwrap() { + Err(HandshakeError::TieBreakLoss) => {} + other => panic!("expected TieBreakLoss on the accept side, got {other:?}"), + } +} + +#[test] +fn loopback_read_ahead_past_hello_survives_into_established_conn() { + maybe_child(ROLES); + // The buffer trap, proven: the dialer coalesces Hello + Heartbeat before + // the responder's first read, so the Heartbeat lands in the shared + // FramedConn's decode buffer during the handshake. The dialer sends + // nothing afterwards — the post-handshake recv can only succeed if the + // read-ahead travelled with the FramedConn. + let (mut dialer, mut accepted) = loopback_pair(); + let (_init, hello) = smarm::cluster::handshake::Initiator::new(&local("node-a")); + dialer.send(&hello).unwrap(); + dialer.send(&Frame::Heartbeat).unwrap(); + // Both frames are buffered before the responder reads at all. + let (tx, rx) = mpsc::channel(); + std::thread::spawn(move || { + let peer = accept_handshake( + &mut accepted, + local("node-b"), + |_| PeerStanding::Free, + no_deadline(), + ) + .unwrap(); + let next = accepted.recv(); + let _ = tx.send((peer, next)); + }); + // A bounded wait: if the Heartbeat were NOT carried in the buffer, the + // recv above would block forever (the dialer stays open and silent). + let (peer, next) = rx + .recv_timeout(Duration::from_secs(5)) + .expect("read-ahead lost: post-handshake recv blocked"); + assert_eq!(peer.node_name, "node-a"); + match next { + Ok(Some(Frame::Heartbeat)) => {} + other => panic!("expected the read-ahead Heartbeat, got {other:?}"), + } + drop(dialer); +} + +// --------------------------------------------------------------------------- +// Deadline + manager integration, over TCP inside the runtime +// --------------------------------------------------------------------------- + +#[test] +fn tcp_silent_peer_times_out_on_the_accept_path() { + maybe_child(ROLES); + run(|| { + let t = TcpTransport; + let mut l = t.listen("127.0.0.1:0").unwrap(); + // Connect and then say nothing at all. + let silent = t.dial(&l.local_addr()).unwrap(); + let mut accepted = FramedConn::new(l.accept().unwrap()); + let (tx, rx) = mpsc::channel(); + smarm::spawn(move || { + let r = accept_handshake( + &mut accepted, + local("node-b"), + |_| PeerStanding::Free, + Instant::now() + Duration::from_millis(200), + ); + let _ = tx.send(r); + }); + match poll_recv(&rx, "accept-path outcome") { + Err(HandshakeError::TimedOut) => {} + other => panic!("expected TimedOut, got {other:?}"), + } + drop(silent); + }); +} + +/// Poll the manager until its peer set matches `expected` (sorted), or fail. +fn wait_peers(expected: &[&str]) { + let want: Vec = expected.iter().map(|s| s.to_string()).collect(); + for _ in 0..5000 { + if let Ok(Reply::Peers(got)) = gen_server::call(MANAGER, Call::Peers) { + if got == want { + return; + } + } + sleep(Duration::from_millis(1)); + } + let got = gen_server::call(MANAGER, Call::Peers); + panic!("timed out waiting for peers == {want:?}; last = {got:?}"); +} + +#[test] +fn tcp_duplicate_name_rejected_by_acceptor() { + maybe_child(ROLES); + run(|| { + let mgr = GenServerBuilder::new(Manager::new()) + .named(MANAGER) + .start() + .expect("manager name is free"); + let listener = TcpTransport.listen("127.0.0.1:0").unwrap(); + let acceptor = spawn_acceptor(listener, local("node-b"), Timing::default()); + let addr = acceptor.local_addr().to_string(); + + // First dial offering "dup-node": establishes and registers. + let mut first = FramedConn::new(TcpTransport.dial(&addr).unwrap()); + let peer = dial_handshake( + &mut first, + &local("dup-node"), + Instant::now() + HANDSHAKE_TIMEOUT, + ) + .unwrap(); + assert_eq!(peer.node_name, "node-b"); + wait_peers(&["dup-node"]); + + // Second dial offering the same name: deterministic NameTaken. + let mut second = FramedConn::new(TcpTransport.dial(&addr).unwrap()); + match dial_handshake( + &mut second, + &local("dup-node"), + Instant::now() + HANDSHAKE_TIMEOUT, + ) { + Err(HandshakeError::Rejected(RejectReason::NameTaken)) => {} + other => panic!("expected NameTaken, got {other:?}"), + } + // The established connection was untouched by the rejected one. + wait_peers(&["dup-node"]); + + // Teardown: the acceptor owns no connections, so the established one + // is torn down through the table. + acceptor.shutdown(); + assert!(matches!( + gen_server::call( + MANAGER, + Call::Disconnect { + name: "dup-node".to_string() + } + ), + Ok(Reply::Disconnected) + )); + wait_peers(&[]); + first.close(); + mgr.shutdown(); + }); +} + +#[test] +fn dial_intent_cleared_when_dialer_dies() { + maybe_child(ROLES); + run(|| { + let mgr = GenServerBuilder::new(Manager::new()) + .named(MANAGER) + .start() + .expect("manager name is free"); + let (begun_tx, begun_rx) = mpsc::channel(); + let (go_tx, go_rx) = mpsc::channel::<()>(); + smarm::spawn(move || { + let me = smarm::self_pid(); + match gen_server::call( + MANAGER, + Call::DialBegin { + name: "ghost".into(), + pid: me, + }, + ) { + Ok(Reply::DialBegan(true)) => {} + other => panic!("DialBegin failed: {other:?}"), + } + let _ = begun_tx.send(()); + let () = poll_recv(&go_rx, "go signal"); + panic!("dialer dies mid-dial"); + }); + poll_recv(&begun_rx, "DialBegin done"); + // While the dialer lives, the intent is visible. + match gen_server::call( + MANAGER, + Call::Standing { + peer_name: "ghost".into(), + }, + ) { + Ok(Reply::Standing(s)) => assert_eq!(s, PeerStanding::Dialing), + other => panic!("PeerStanding failed: {other:?}"), + } + // Kill it; the monitor must clear the intent without cooperation. + go_tx.send(()).unwrap(); + let deadline = Instant::now() + WAIT; + loop { + match gen_server::call( + MANAGER, + Call::Standing { + peer_name: "ghost".into(), + }, + ) { + Ok(Reply::Standing(s)) if s != PeerStanding::Dialing => break, + _ if Instant::now() > deadline => { + panic!("dial intent not cleared after dialer death") + } + _ => sleep(Duration::from_millis(1)), + } + } + mgr.shutdown(); + }); +} + +// --------------------------------------------------------------------------- +// Two nodes, two processes: the integrated dial against a real acceptor +// --------------------------------------------------------------------------- + +/// Announce, then park forever. Neither role ever tears its connection +/// down: a table entry only exists while the *peer* holds its side open, so +/// any teardown here would retract the other node's observation before it +/// had made it. The parent reaps both with SIGKILL once it has both +/// announcements (see [`common::Node`]'s `Drop`). +fn role_hs_listener() { + run(|| { + let _mgr = GenServerBuilder::new(Manager::new()) + .named(MANAGER) + .start() + .expect("manager name is free"); + let listener = TcpTransport.listen("127.0.0.1:0").unwrap(); + let acceptor = spawn_acceptor(listener, local("node-b"), Timing::default()); + println!("LISTENING {}", acceptor.local_addr()); + wait_peers(&["node-a"]); + println!("PEERS node-a"); + park(); + }); +} + +fn role_hs_dialer() { + let addr = std::env::var("SMARM_PEER_ADDR").expect("SMARM_PEER_ADDR not set"); + run(move || { + let _mgr = GenServerBuilder::new(Manager::new()) + .named(MANAGER) + .start() + .expect("manager name is free"); + let (tx, rx) = mpsc::channel(); + smarm::spawn(move || { + let r = dial( + &TcpTransport, + &addr, + "node-b", + &local("node-a"), + Timing::default(), + ); + let _ = tx.send(r); + }); + if let Err(e) = poll_recv(&rx, "dial outcome") { + println!("DIAL failed: {e:?}"); + std::process::exit(3); + } + wait_peers(&["node-b"]); + println!("PEERS node-b"); + park(); + }); +} + +#[test] +fn two_node_integrated_handshake_over_tcp() { + maybe_child(ROLES); + let mut listener = spawn_node("hs_listener", &[]); + let addr = listener.wait_listening(); + let mut dialer = spawn_node("hs_dialer", &[("SMARM_PEER_ADDR", &addr)]); + // Each node reports its own table naming the other: a real dial against a + // real acceptor established in both directions. Both nodes then park — + // clean-exit behaviour is the c4 harness's own smoke test, and demanding + // it here would mean a teardown, which is exactly what cannot be ordered + // safely across two processes. Dropping the nodes SIGKILLs them. + dialer.wait_line("PEERS node-b", |l| l == "PEERS node-b"); + listener.wait_line("PEERS node-a", |l| l == "PEERS node-a"); +} + +// --------------------------------------------------------------------------- +// Integrated-dial guardrails (no acceptor involved) +// --------------------------------------------------------------------------- + +#[test] +fn concurrent_dial_to_same_name_refused() { + maybe_child(ROLES); + run(|| { + let mgr = GenServerBuilder::new(Manager::new()) + .named(MANAGER) + .start() + .expect("manager name is free"); + let (begun_tx, begun_rx) = mpsc::channel(); + let (go_tx, go_rx) = mpsc::channel::<()>(); + // First dialer parks with the intent held (it never connects — + // 'holding the intent' is all this test needs from it). + smarm::spawn(move || { + let me = smarm::self_pid(); + assert!(matches!( + gen_server::call( + MANAGER, + Call::DialBegin { + name: "node-x".into(), + pid: me, + } + ), + Ok(Reply::DialBegan(true)) + )); + let _ = begun_tx.send(()); + let () = poll_recv(&go_rx, "go signal"); + let _ = gen_server::call( + MANAGER, + Call::DialEnd { + name: "node-x".into(), + }, + ); + }); + poll_recv(&begun_rx, "DialBegin done"); + // Second integrated dial to the same name: refused before connecting + // (the addr is unroutable on purpose — it must never be dialed). + let (tx, rx) = mpsc::channel(); + smarm::spawn(move || { + let r = dial( + &TcpTransport, + "127.0.0.1:1", + "node-x", + &local("node-a"), + Timing::default(), + ); + let _ = tx.send(r); + }); + match poll_recv(&rx, "second dial outcome") { + Err(DialError::AlreadyDialing) => {} + other => panic!("expected AlreadyDialing, got {other:?}"), + } + go_tx.send(()).unwrap(); + mgr.shutdown(); + }); +} diff --git a/tests/cluster_dial_mismatch.rs b/tests/cluster_dial_mismatch.rs new file mode 100644 index 0000000..03cc754 --- /dev/null +++ b/tests/cluster_dial_mismatch.rs @@ -0,0 +1,115 @@ +//! RFC 010 — a seed whose address answers as a *different* name +//! (`DialError::PeerNameMismatch`) is dialed once and then parked: the +//! connector must not redial it on backoff forever. +//! +//! Observed from the misdialed peer: each such dial establishes at the +//! responder (it registers, `node_up`), then the dialer closes on the name +//! check (`node_down`) — one membership blip per attempt. Cross-process: a +//! *server* named `server` subscribes and reports; a *client* on fast +//! timing (50–500ms backoff) seeds `("wrongname", server_addr)`. After the +//! first blip the server counts further `NodeUp`s across 2s — several +//! backoff periods. Parked ⇒ zero. Negative-control-verified: with the park +//! stubbed out the count is ≥ 1 in the same window. +#![cfg(feature = "cluster")] + +mod common; + +use common::{maybe_child, spawn_node}; +use smarm::cluster::envelope::NodeMeta; +use smarm::cluster::membership::{subscribe, NodeEvent}; +use smarm::cluster::{start, Config, StaticSeeds, Timing}; +use std::time::{Duration, Instant}; + +const ROLES: &[(&str, fn())] = &[("server", role_server), ("client", role_client)]; + +fn meta() -> NodeMeta { + NodeMeta { + role: "mismatch".into(), + region: "local".into(), + } +} + +fn timing() -> Timing { + Timing { + initial_backoff: Duration::from_millis(50), + max_backoff: Duration::from_millis(500), + ..Timing::default() + } +} + +fn role_server() { + smarm::run(|| { + let cluster = start(Config { + node_name: "server".into(), + meta: meta(), + listen_addr: std::env::var("SMARM_LISTEN_ADDR") + .unwrap_or_else(|_| "127.0.0.1:0".into()), + strategy: Box::new(StaticSeeds::new(Vec::<(String, String)>::new())), + timing: timing(), + }) + .expect("binds"); + let ev = subscribe().unwrap(); + println!("LISTENING {}", cluster.local_addr()); + // First blip: the misdialed client establishes, then closes on us. + loop { + match ev.rx.recv() { + Ok(NodeEvent::NodeDown(i)) if i.name == "client" => break, + Ok(_) => continue, + Err(_) => panic!("manager gone"), + } + } + println!("BLIP"); + // Now count further NodeUps across several backoff periods. + let mut more = 0usize; + let t0 = Instant::now(); + while t0.elapsed() < Duration::from_millis(2000) { + match ev.rx.try_recv() { + Ok(Some(NodeEvent::NodeUp(i))) if i.name == "client" => more += 1, + Ok(_) => {} + Err(_) => panic!("manager gone"), + } + smarm::sleep(Duration::from_millis(50)); + } + println!("MORE {more}"); + loop { + smarm::sleep(Duration::from_secs(3600)); + } + }); +} + +fn role_client() { + let server_addr = std::env::var("SMARM_SERVER_ADDR").expect("SMARM_SERVER_ADDR"); + smarm::run(move || { + let _cluster = start(Config { + node_name: "client".into(), + meta: meta(), + listen_addr: "127.0.0.1:0".into(), + strategy: Box::new(StaticSeeds::new(vec![( + "wrongname".to_string(), + server_addr, + )])), + timing: timing(), + }) + .expect("binds"); + println!("CLIENT UP"); + loop { + smarm::sleep(Duration::from_secs(3600)); + } + }); +} + +#[test] +fn mismatched_seed_is_dialed_once_then_parked() { + maybe_child(ROLES); + let mut server = spawn_node("server", &[]); + let saddr = server.wait_listening(); + let mut client = spawn_node("client", &[("SMARM_SERVER_ADDR", &saddr)]); + client.wait_line("CLIENT UP", |l| l == "CLIENT UP"); + server.wait_line("BLIP", |l| l == "BLIP"); + let line = server.wait_line("MORE", |l| l.starts_with("MORE ")); + let more: usize = line.split_whitespace().nth(1).unwrap().parse().unwrap(); + assert_eq!( + more, 0, + "mismatched seed was redialed {more}× after being parked" + ); +} diff --git a/tests/cluster_disconnect.rs b/tests/cluster_disconnect.rs new file mode 100644 index 0000000..a7ae186 --- /dev/null +++ b/tests/cluster_disconnect.rs @@ -0,0 +1,379 @@ +//! RFC 010 c13 — connection-loss synthesis. +//! +//! Local suite (`run()`, no network): the read-side backstop. A +//! `RemoteMonitor` whose channel closes without a notice reads as +//! `Disconnected` exactly once (a `Monitor` command that reached the conn +//! actor's inbox but was never processed — the drain gap); after +//! `demonitor_remote` a closed channel stays a plain `Err`, never a notice. +//! +//! Cross-process: the headline contrast — an actor's own death gives its +//! TRUE reason, loss of the LINK gives `Disconnected` (both a commanded +//! `Disconnect` and a SIGKILLed peer process are `Disconnected` from the +//! monitor's view: nobody is left to say otherwise). Reconnect does not +//! resurrect: the old monitor yields nothing more, proven by stream ORDER +//! (a fresh monitor over the new link delivers first). The ignored test +//! trips liveness by SIGSTOP and then drops the link too, asserting exactly +//! one notice for one monitor. +#![cfg(feature = "cluster")] + +mod common; + +use common::{maybe_child, spawn_node}; +use smarm::cluster::envelope::NodeMeta; +use smarm::cluster::expose::{expose, expose_type}; +use smarm::cluster::manager::{Call, Reply, MANAGER}; +use smarm::cluster::membership::{subscribe, NodeEvent}; +use smarm::cluster::remote::{ + self, demonitor_remote, monitor_remote, send_to_remote, RemoteName, RemotePid, +}; +use smarm::cluster::{start, Config, RemoteDownReason, StaticSeeds, Timing}; +use smarm::pg::Incarnation; +use smarm::{ + channel, gen_server, install, register, run, spawn, Addressable, DownReason, Erased, Name, Pid, +}; +use std::collections::HashMap; +use std::time::Duration; + +// ---- message types (hand-rolled serde; the crate is derive-less) --------- + +#[derive(Debug)] +struct Ctl { + cmd: String, + reply_to: RemotePid, +} +#[derive(Debug)] +struct Answer { + text: String, + pid: Option>, +} +struct Client; +impl Addressable for Client { + type Msg = Answer; +} + +impl serde::Serialize for Ctl { + fn serialize(&self, s: S) -> Result { + use serde::ser::SerializeTuple; + let mut t = s.serialize_tuple(2)?; + t.serialize_element(&self.cmd)?; + t.serialize_element(&self.reply_to)?; + t.end() + } +} +impl<'de> serde::Deserialize<'de> for Ctl { + fn deserialize>(d: D) -> Result { + let (cmd, reply_to) = <(String, RemotePid)>::deserialize(d)?; + Ok(Ctl { cmd, reply_to }) + } +} +impl serde::Serialize for Answer { + fn serialize(&self, s: S) -> Result { + use serde::ser::SerializeTuple; + let mut t = s.serialize_tuple(2)?; + t.serialize_element(&self.text)?; + t.serialize_element(&self.pid)?; + t.end() + } +} +impl<'de> serde::Deserialize<'de> for Answer { + fn deserialize>(d: D) -> Result { + let (text, pid) = <(String, Option>)>::deserialize(d)?; + Ok(Answer { text, pid }) + } +} + +// ================= local suite ========================================= + +/// A `Monitor` command handed to the connection but never processed (its +/// receiver dropped unread) reads as `Disconnected` — once. A second read +/// is the ordinary closed-channel `Err`, so "exactly one notice" holds. +#[test] +fn unread_command_reads_as_disconnected_once() { + maybe_child(ROLES); + run(|| { + remote::set_local_identity("me", Incarnation::new(7)); + let (probe_tx, _probe_rx) = channel(); + let inbox = + remote::bind_outbound_probe_with_monitors("peer", Incarnation::new(5), probe_tx); + let target = RemotePid::::from_parts("peer", Incarnation::new(5), 9, 1); + let m = monitor_remote(target.clone()); + assert!( + matches!(m.try_recv(), Ok(None)), + "command is in flight, no notice yet" + ); + drop(inbox); // the conn actor died with the command unread + let d = m.recv().unwrap(); + assert_eq!(d.pid, target); + assert_eq!(d.reason, RemoteDownReason::Disconnected); + assert!( + m.recv().is_err(), + "second read is closed, not a second notice" + ); + assert!(m.try_recv().is_err()); + }); +} + +/// After `demonitor_remote`, a closed channel is a closed channel: no +/// notice is synthesized for a monitor the caller cancelled. +#[test] +fn cancelled_monitor_never_synthesizes() { + maybe_child(ROLES); + run(|| { + remote::set_local_identity("me", Incarnation::new(7)); + let (probe_tx, _probe_rx) = channel(); + let inbox = + remote::bind_outbound_probe_with_monitors("peer", Incarnation::new(5), probe_tx); + let target = RemotePid::::from_parts("peer", Incarnation::new(5), 9, 1); + let m = monitor_remote(target); + demonitor_remote(&m); + drop(inbox); + assert!(m.recv().is_err()); + assert!(m.try_recv().is_err()); + }); +} + +// ================= cross-process ====================================== + +const ROLES: &[(&str, fn())] = &[ + ("server", role_server), + ("client", role_client), + ("client_stop", role_client_stop), +]; + +const CTL: Name = Name::new("c13.ctl"); + +fn cfg(name: &str, seeds: Vec<(String, String)>) -> Config { + Config { + node_name: name.to_string(), + meta: NodeMeta { + role: "c13".into(), + region: "local".into(), + }, + listen_addr: std::env::var("SMARM_LISTEN_ADDR").unwrap_or_else(|_| "127.0.0.1:0".into()), + strategy: Box::new(StaticSeeds::new(seeds)), + timing: timing(), + } +} + +/// The p11 knobs make the liveness test fast: both roles of that test are +/// spawned with `SMARM_FAST_TIMING=1` and agree on a 100ms heartbeat / +/// 500ms liveness window. Everything else runs the shipping defaults. +fn timing() -> Timing { + if std::env::var_os("SMARM_FAST_TIMING").is_some() { + Timing { + heartbeat_interval: Duration::from_millis(100), + liveness_timeout: Duration::from_millis(500), + initial_backoff: Duration::from_millis(50), + max_backoff: Duration::from_millis(500), + ..Timing::default() + } + } else { + Timing::default() + } +} + +fn wait_up(events: &smarm::cluster::membership::MembershipEvents, who: &str) { + loop { + match events.rx.recv() { + Ok(NodeEvent::NodeUp(i)) if i.name == who => return, + Ok(_) => continue, + Err(_) => panic!("manager gone"), + } + } +} + +fn disconnect(name: &str) { + assert!(matches!( + gen_server::call( + MANAGER, + Call::Disconnect { + name: name.to_string() + } + ), + Ok(Reply::Disconnected) + )); +} + +/// Server: `spawn` ⇒ a parked worker (answer carries its pid); +/// `kill:` releases it, whereupon it returns (Exit). +fn role_server() { + smarm::run(move || { + let cluster = start(cfg("server", vec![])).expect("binds"); + println!("LISTENING {}", cluster.local_addr()); + let (tx, rx) = channel::(); + register(CTL, tx).unwrap(); + expose(CTL); + println!("READY"); + let mut workers: HashMap> = HashMap::new(); + loop { + let ctl = rx.recv().unwrap(); + println!("CTL {}", ctl.cmd); + let (text, pid): (String, Option>) = match ctl.cmd.as_str() { + "spawn" => { + let (go_tx, go_rx) = channel::<()>(); + let p: Pid = spawn(move || { + let _ = go_rx.recv(); + }) + .pid(); + workers.insert(p.index(), go_tx); + ( + "ok".into(), + Some(RemotePid::from_local(p).expect("identity set")), + ) + } + other => { + let idx: u32 = other.strip_prefix("kill:").unwrap().parse().unwrap(); + if let Some(go) = workers.remove(&idx) { + let _ = go.send(()); + } + ("killed".into(), None) + } + }; + send_to_remote(ctl.reply_to, Answer { text, pid }).unwrap(); + } + }); +} + +/// Client-side setup shared by both client roles: join, expose the reply +/// path, hand back an `ask` closure and the membership stream. +fn client_setup() -> ( + smarm::cluster::Cluster, + smarm::cluster::membership::MembershipEvents, + impl Fn(&str) -> Answer, +) { + let server_addr = std::env::var("SMARM_SERVER_ADDR").expect("SMARM_SERVER_ADDR"); + let cluster = start(cfg("client", vec![("server".into(), server_addr)])).expect("binds"); + let ev = subscribe().unwrap(); + wait_up(&ev, "server"); + let (tx, rx) = channel::(); + let me: Pid = install::(tx); + expose_type::(); + let ask = move |cmd: &str| -> Answer { + remote::send( + RemoteName::new("server", CTL), + Ctl { + cmd: cmd.into(), + reply_to: RemotePid::from_local(me).expect("identity set"), + }, + ) + .unwrap(); + rx.recv().unwrap() + }; + (cluster, ev, ask) +} + +fn role_client() { + smarm::run(move || { + let (_cluster, ev, ask) = client_setup(); + + // 1. Headline: actor death ⇒ TRUE reason; link cut ⇒ Disconnected. + let a = ask("spawn").pid.unwrap(); + let b = ask("spawn").pid.unwrap(); + let ma = monitor_remote(a.clone()); + let mb = monitor_remote(b.clone()); + ask(&format!("kill:{}", a.index())); + let d = ma.recv().unwrap(); + assert_eq!(d.pid, a); + println!("DOWN actor {:?}", d.reason); + disconnect("server"); + let d = mb.recv().unwrap(); + assert_eq!(d.pid, b); + println!("DOWN link {:?}", d.reason); + + // 2. Reconnect does not resurrect. The connector redials on + // node_down; over the NEW link a fresh monitor delivers, while + // the old one (already answered) yields nothing further — order + // proves it, and `b` is even still alive on the server. + wait_up(&ev, "server"); + println!("RECONNECTED"); + let c = ask("spawn").pid.unwrap(); + let mc = monitor_remote(c.clone()); + ask(&format!("kill:{}", b.index())); + ask(&format!("kill:{}", c.index())); + assert_eq!(mc.recv().unwrap().reason, DownReason::Exit.into()); + let stray = matches!(mb.try_recv(), Ok(Some(_))); + println!("RESURRECT stray={stray}"); + + // 3. Peer PROCESS killed ⇒ Disconnected too (nobody is left to send + // Down): the parent SIGKILLs the server once it sees the marker. + let e = ask("spawn").pid.unwrap(); + let me_ = monitor_remote(e.clone()); + println!("KILL SERVER NOW"); + let d = me_.recv().unwrap(); + assert_eq!(d.pid, e); + println!("DOWN procdeath {:?}", d.reason); + + println!("CLIENT DONE"); + loop { + smarm::sleep(Duration::from_secs(3600)); + } + }); +} + +/// The slow role: liveness expiry (peer SIGSTOPped) followed by the link +/// dropping for real (peer SIGKILLed) — one monitor, exactly one notice. +fn role_client_stop() { + smarm::run(move || { + let (_cluster, _ev, ask) = client_setup(); + let a = ask("spawn").pid.unwrap(); + let ma = monitor_remote(a.clone()); + println!("STOP SERVER NOW"); + let d = ma.recv().unwrap(); // liveness expiry, ~liveness_timeout + assert_eq!(d.pid, a); + println!("DOWN stopped {:?}", d.reason); + println!("KILL SERVER NOW"); + // Give the drop every chance to produce a second notice, then look. + smarm::sleep(Duration::from_secs(1)); + let dup = matches!(ma.try_recv(), Ok(Some(_))); + println!("DUPLICATE dup={dup}"); + println!("CLIENT DONE"); + loop { + smarm::sleep(Duration::from_secs(3600)); + } + }); +} + +/// The Phase 4 c13 gate: partition vs. death distinguishable; nothing +/// survives reconnect; a dead peer process is a Disconnected too. +#[test] +fn link_loss_is_disconnected_and_does_not_survive_reconnect() { + maybe_child(ROLES); + let mut server = spawn_node("server", &[]); + let saddr = server.wait_listening(); + server.wait_line("READY", |l| l == "READY"); + let mut client = spawn_node("client", &[("SMARM_SERVER_ADDR", &saddr)]); + client.wait_line("DOWN actor Local(Exit)", |l| l == "DOWN actor Local(Exit)"); + client.wait_line("DOWN link Disconnected", |l| l == "DOWN link Disconnected"); + client.wait_line("RECONNECTED", |l| l == "RECONNECTED"); + client.wait_line("RESURRECT stray=false", |l| l == "RESURRECT stray=false"); + client.wait_line("KILL SERVER NOW", |l| l == "KILL SERVER NOW"); + server.kill(); + client.wait_line("DOWN procdeath Disconnected", |l| { + l == "DOWN procdeath Disconnected" + }); + client.wait_line("CLIENT DONE", |l| l == "CLIENT DONE"); +} + +/// Covers the invariant the headline test cannot: liveness expiry and the +/// transport drop both firing for the same connection yield ONE notice. +/// Runs on the fast [`timing`] (both roles) — was `#[ignore]`d at the 4s +/// default until the p11 knobs landed. +#[test] +fn timeout_then_drop_yields_one_notice() { + maybe_child(ROLES); + let fast = ("SMARM_FAST_TIMING", "1"); + let mut server = spawn_node("server", &[fast]); + let saddr = server.wait_listening(); + server.wait_line("READY", |l| l == "READY"); + let mut client = spawn_node("client_stop", &[("SMARM_SERVER_ADDR", &saddr), fast]); + client.wait_line("STOP SERVER NOW", |l| l == "STOP SERVER NOW"); + let spid = server.pid().expect("server alive") as libc::pid_t; + assert_eq!(unsafe { libc::kill(spid, libc::SIGSTOP) }, 0); + client.wait_line("DOWN stopped Disconnected", |l| { + l == "DOWN stopped Disconnected" + }); + client.wait_line("KILL SERVER NOW", |l| l == "KILL SERVER NOW"); + server.kill(); // SIGKILL works on a stopped process; Drop would too + client.wait_line("DUPLICATE dup=false", |l| l == "DUPLICATE dup=false"); + client.wait_line("CLIENT DONE", |l| l == "CLIENT DONE"); +} diff --git a/tests/cluster_discovery_withdraw.rs b/tests/cluster_discovery_withdraw.rs new file mode 100644 index 0000000..f261451 --- /dev/null +++ b/tests/cluster_discovery_withdraw.rs @@ -0,0 +1,161 @@ +//! RFC 010 — `Discovery::Withdrawn`: a strategy retracts a candidate and the +//! connector stops dialing it. +//! +//! Cross-process: a plain *server* node, and a *client* whose strategy is a +//! script: announce a decoy `(ghost, addr)` where `addr` is a raw +//! `TcpListener` the client itself holds (an OS thread accepts and +//! immediately closes, so every dial fails at handshake and the connector +//! keeps retrying on backoff — the accept count is the dial count); after a +//! beat, withdraw the decoy and announce the real server. The client waits +//! for the server's `node_up` — which is *after* the withdrawal in the +//! strategy's own stream — then watches the decoy's accept count stay flat +//! across a window longer than the pending backoff. Before withdrawal it +//! must have been climbing (≥ 1), or the negative proves nothing. +#![cfg(feature = "cluster")] + +mod common; + +use common::{maybe_child, spawn_node}; +use smarm::channel::Sender; +use smarm::cluster::discovery::{Discovery, Strategy}; +use smarm::cluster::envelope::NodeMeta; +use smarm::cluster::membership::{subscribe, NodeEvent}; +use smarm::cluster::{start, Config, StaticSeeds, Timing}; +use std::net::TcpListener; +use std::sync::atomic::{AtomicUsize, Ordering}; +use std::sync::Arc; +use std::time::{Duration, Instant}; + +const ROLES: &[(&str, fn())] = &[("server", role_server), ("client", role_client)]; + +fn meta() -> NodeMeta { + NodeMeta { + role: "withdraw".into(), + region: "local".into(), + } +} + +fn role_server() { + smarm::run(|| { + let cluster = start(Config { + node_name: "server".into(), + meta: meta(), + listen_addr: std::env::var("SMARM_LISTEN_ADDR") + .unwrap_or_else(|_| "127.0.0.1:0".into()), + strategy: Box::new(StaticSeeds::new(Vec::<(String, String)>::new())), + timing: Timing::default(), + }) + .expect("binds"); + println!("LISTENING {}", cluster.local_addr()); + loop { + smarm::sleep(Duration::from_secs(3600)); + } + }); +} + +/// Scripted strategy: decoy, pause, withdraw decoy, real server, done. +struct Script { + decoy: String, + server: String, +} + +impl Strategy for Script { + fn run(self: Box, out: Sender) { + let _ = out.send(Discovery::Candidate { + name: "ghost".into(), + addr: self.decoy.clone(), + }); + // Long enough for the 250ms/500ms retries to land: ≥ 3 dials. + smarm::sleep(Duration::from_millis(1100)); + let _ = out.send(Discovery::Withdrawn { + name: "ghost".into(), + addr: self.decoy, + }); + let _ = out.send(Discovery::Candidate { + name: "server".into(), + addr: self.server, + }); + } +} + +fn role_client() { + let server_addr = std::env::var("SMARM_SERVER_ADDR").expect("SMARM_SERVER_ADDR"); + // The decoy: accept-and-close on an OS thread; count every accept. + let decoy = TcpListener::bind("127.0.0.1:0").unwrap(); + let decoy_addr = decoy.local_addr().unwrap().to_string(); + let dials = Arc::new(AtomicUsize::new(0)); + let counter = dials.clone(); + std::thread::spawn(move || { + for conn in decoy.incoming() { + counter.fetch_add(1, Ordering::SeqCst); + drop(conn); + } + }); + + smarm::run(move || { + let _cluster = start(Config { + node_name: "client".into(), + meta: meta(), + listen_addr: "127.0.0.1:0".into(), + strategy: Box::new(Script { + decoy: decoy_addr, + server: server_addr, + }), + timing: Timing::default(), + }) + .expect("binds"); + let ev = subscribe().unwrap(); + loop { + match ev.rx.recv() { + Ok(NodeEvent::NodeUp(i)) if i.name == "server" => break, + Ok(_) => continue, + Err(_) => panic!("manager gone"), + } + } + // The withdrawal preceded the server candidate in the strategy's + // stream, so it has been applied. Any dial that started before it + // is bounded by the connect+handshake deadlines; let it drain, then + // hold the count flat across a window longer than the pending + // backoff would be (1s at this point, 2s next). + let before = dials.load(Ordering::SeqCst); + smarm::sleep(Duration::from_millis(500)); + let settled = dials.load(Ordering::SeqCst); + let t0 = Instant::now(); + while t0.elapsed() < Duration::from_millis(3000) { + smarm::sleep(Duration::from_millis(100)); + } + let after = dials.load(Ordering::SeqCst); + println!("WITHDRAWN before={before} settled={settled} after={after}"); + println!("CLIENT DONE"); + loop { + smarm::sleep(Duration::from_secs(3600)); + } + }); +} + +#[test] +fn withdrawn_candidate_is_no_longer_dialed() { + maybe_child(ROLES); + let mut server = spawn_node("server", &[]); + let saddr = server.wait_listening(); + let mut client = spawn_node("client", &[("SMARM_SERVER_ADDR", &saddr)]); + let line = client.wait_line("WITHDRAWN", |l| l.starts_with("WITHDRAWN ")); + let mut nums = line + .split_whitespace() + .skip(1) + .map(|kv| kv.split_once('=').unwrap().1.parse::().unwrap()); + let (before, settled, after) = ( + nums.next().unwrap(), + nums.next().unwrap(), + nums.next().unwrap(), + ); + assert!( + before >= 1, + "decoy was never dialed; the negative proves nothing: {line}" + ); + assert_eq!( + settled, after, + "connector kept dialing a withdrawn candidate: {line}" + ); + client.wait_line("CLIENT DONE", |l| l == "CLIENT DONE"); +} diff --git a/tests/cluster_envelope.rs b/tests/cluster_envelope.rs new file mode 100644 index 0000000..f81dad3 --- /dev/null +++ b/tests/cluster_envelope.rs @@ -0,0 +1,276 @@ +//! RFC 010 c2 — owned envelope tests (roadmap: per-frame roundtrip, +//! truncation mid-field, unknown tag, length prefix lying long and short, +//! zero-length payload, adversarial lengths). +#![cfg(feature = "cluster")] + +use serde::{Deserialize, Serialize}; +use smarm::cluster::envelope::{ + decode_payload, encode_payload, DecodeError, Frame, NodeMeta, RejectReason, MAX_FRAME_LEN, + PROTO_VERSION, +}; +use smarm::cluster::RemoteDownReason; +use smarm::monitor::DownReason; +use smarm::pg::Incarnation; + +fn meta() -> NodeMeta { + NodeMeta { + role: "worker".into(), + region: "eu-west".into(), + } +} + +fn all_frames() -> Vec { + vec![ + Frame::Hello { + proto_version: PROTO_VERSION, + build_hash: 0xDEAD_BEEF_CAFE_F00D, + node_name: "alpha".into(), + incarnation: Incarnation::new(7), + meta: meta(), + }, + Frame::HelloAck { + node_name: "beta".into(), + incarnation: Incarnation::new(9), + meta: meta(), + }, + Frame::HelloReject { + reason: RejectReason::NameTaken, + }, + Frame::Heartbeat, + Frame::Send { + index: 42, + generation: 3, + type_hash: 0x1234_5678_9ABC_DEF0, + payload: vec![1, 2, 3, 4, 5], + }, + Frame::SendNamed { + name: "the_counter".into(), + type_hash: 0xFFFF_0000_FFFF_0000, + payload: vec![], + }, + Frame::Monitor { + monitor_id: 77, + index: 42, + generation: 3, + }, + Frame::Demonitor { monitor_id: 77 }, + Frame::Down { + monitor_id: 77, + reason: RemoteDownReason::Local(DownReason::Panic), + }, + Frame::Down { + monitor_id: 78, + reason: RemoteDownReason::Disconnected, + }, + ] +} + +fn encode_one(f: &Frame) -> Vec { + let mut buf = Vec::new(); + f.encode(&mut buf).unwrap(); + buf +} + +#[test] +fn per_frame_roundtrip() { + for f in all_frames() { + let buf = encode_one(&f); + let (decoded, consumed) = Frame::decode(&buf).unwrap().unwrap(); + assert_eq!(decoded, f, "roundtrip mismatch"); + assert_eq!(consumed, buf.len(), "consumed != buffer length for {f:?}"); + } +} + +#[test] +fn back_to_back_frames_decode_sequentially() { + let mut buf = Vec::new(); + for f in all_frames() { + f.encode(&mut buf).unwrap(); + } + let mut off = 0; + let mut decoded = Vec::new(); + while off < buf.len() { + let (f, n) = Frame::decode(&buf[off..]).unwrap().unwrap(); + decoded.push(f); + off += n; + } + assert_eq!(decoded, all_frames()); + assert_eq!(off, buf.len()); +} + +#[test] +fn heartbeat_golden_bytes() { + // Locks the layout: u32 LE length prefix, then the tag byte. + let buf = encode_one(&Frame::Heartbeat); + assert_eq!(buf, vec![1, 0, 0, 0, 4]); +} + +#[test] +fn zero_length_payload_roundtrips() { + let f = Frame::Send { + index: 0, + generation: 0, + type_hash: 0, + payload: vec![], + }; + let buf = encode_one(&f); + let (decoded, consumed) = Frame::decode(&buf).unwrap().unwrap(); + assert_eq!(decoded, f); + assert_eq!(consumed, buf.len()); +} + +#[test] +fn incomplete_is_none_not_error() { + let buf = encode_one(&all_frames()[0]); + // Every strict prefix short of the full frame must report "need more". + for cut in 0..buf.len() { + assert_eq!( + Frame::decode(&buf[..cut]).unwrap(), + None, + "cut at {cut} should be incomplete" + ); + } +} + +#[test] +fn unknown_frame_tag() { + let buf = vec![1, 0, 0, 0, 250]; + assert_eq!(Frame::decode(&buf), Err(DecodeError::UnknownTag(250))); +} + +#[test] +fn unknown_enum_tags() { + // HelloReject with a bogus reason tag. + let buf = vec![2, 0, 0, 0, 3, 99]; + assert_eq!( + Frame::decode(&buf), + Err(DecodeError::UnknownEnumTag { + what: "RejectReason", + tag: 99 + }) + ); + // Down with a bogus reason tag (id = 0u64). + let mut buf = vec![10, 0, 0, 0, 9]; + buf.extend_from_slice(&0u64.to_le_bytes()); + buf.push(200); + assert_eq!( + Frame::decode(&buf), + Err(DecodeError::UnknownEnumTag { + what: "RemoteDownReason", + tag: 200 + }) + ); +} + +#[test] +fn length_prefix_lying_long_with_bytes_present_is_trailing() { + let mut buf = encode_one(&Frame::Heartbeat); + // Declare 3 extra body bytes and actually supply them. + let declared = u32::from_le_bytes([buf[0], buf[1], buf[2], buf[3]]) + 3; + buf[0..4].copy_from_slice(&declared.to_le_bytes()); + buf.extend_from_slice(&[0xAA, 0xBB, 0xCC]); + assert_eq!(Frame::decode(&buf), Err(DecodeError::Trailing { extra: 3 })); +} + +#[test] +fn length_prefix_lying_long_without_bytes_is_incomplete() { + // Indistinguishable from a partial read — must be None, not an error. + let mut buf = encode_one(&Frame::Heartbeat); + let declared = u32::from_le_bytes([buf[0], buf[1], buf[2], buf[3]]) + 3; + buf[0..4].copy_from_slice(&declared.to_le_bytes()); + assert_eq!(Frame::decode(&buf).unwrap(), None); +} + +#[test] +fn length_prefix_lying_short_truncates_a_field() { + let f = &all_frames()[0]; // Hello: plenty of fields to cut into + let mut buf = encode_one(f); + let declared = u32::from_le_bytes([buf[0], buf[1], buf[2], buf[3]]); + let lie = declared - 4; // cut mid-field + buf[0..4].copy_from_slice(&lie.to_le_bytes()); + assert_eq!(Frame::decode(&buf), Err(DecodeError::Truncated)); +} + +#[test] +fn truncation_mid_string_field() { + // A frame whose declared length is intact but whose inner string length + // runs past the body: SendNamed claiming a 1000-byte name in a tiny body. + let mut body = vec![6u8]; // TAG_SEND_NAMED + body.extend_from_slice(&1000u16.to_le_bytes()); + body.extend_from_slice(b"short"); + let mut buf = (body.len() as u32).to_le_bytes().to_vec(); + buf.extend_from_slice(&body); + assert_eq!(Frame::decode(&buf), Err(DecodeError::Truncated)); +} + +#[test] +fn adversarial_lengths() { + // Length prefix of u32::MAX: reject as oversized, do not wait for 4 GiB. + let buf = [0xFF, 0xFF, 0xFF, 0xFF, 0]; + assert_eq!( + Frame::decode(&buf), + Err(DecodeError::FrameTooLarge { + declared: u32::MAX as usize + }) + ); + // Just over the cap: also rejected. + let over = (MAX_FRAME_LEN as u32 + 1).to_le_bytes(); + assert!(matches!( + Frame::decode(&over), + Err(DecodeError::FrameTooLarge { .. }) + )); + // Zero-length frame: there is no tag byte; corrupt, not incomplete. + let buf = [0, 0, 0, 0]; + assert_eq!(Frame::decode(&buf), Err(DecodeError::EmptyFrame)); +} + +#[test] +fn invalid_utf8_in_string_field() { + let mut buf = encode_one(&Frame::SendNamed { + name: "abcd".into(), + type_hash: 0, + payload: vec![], + }); + // name bytes start after: 4 (len) + 1 (tag) + 2 (str len) = offset 7 + buf[7] = 0xFF; + assert_eq!(Frame::decode(&buf), Err(DecodeError::Utf8)); +} + +#[derive(Debug, PartialEq, Serialize, Deserialize)] +struct Ping { + seq: u64, + label: String, +} + +#[test] +fn payload_seam_roundtrip() { + let ping = Ping { + seq: 31337, + label: "hello".into(), + }; + let blob = encode_payload(&ping).unwrap(); + // Carry it through a real frame, as it will travel in c9. + let f = Frame::Send { + index: 1, + generation: 1, + type_hash: 0xABCD, + payload: blob, + }; + let buf = encode_one(&f); + let (decoded, _) = Frame::decode(&buf).unwrap().unwrap(); + let Frame::Send { payload, .. } = decoded else { + panic!("wrong frame"); + }; + let back: Ping = decode_payload(&payload).unwrap(); + assert_eq!(back, ping); +} + +#[test] +fn payload_seam_rejects_truncated_blob() { + let blob = encode_payload(&Ping { + seq: 1, + label: "x".into(), + }) + .unwrap(); + assert!(decode_payload::(&blob[..blob.len() - 1]).is_err()); +} diff --git a/tests/cluster_expose.rs b/tests/cluster_expose.rs new file mode 100644 index 0000000..9aad3af --- /dev/null +++ b/tests/cluster_expose.rs @@ -0,0 +1,161 @@ +//! RFC 010 c8 — exposure registry + type hashing. Purely local, no network. +//! +//! Payload types are std types (`String`, `u64`) because the crate's serde is +//! deliberately derive-less (`default-features = false`) — user crates bring +//! their own derive; the contract here is `DeserializeOwned`. +//! +//! The hash-stability test re-execs the current binary (the c4 harness): the +//! guarantee under test is "stable across runs in the SAME binary" — exactly +//! what the build-hash handshake reduces the mesh to — not stability across +//! builds, which the scope guard explicitly rejects. +#![cfg(feature = "cluster")] + +mod common; + +use common::{maybe_child, spawn_node}; +use smarm::cluster::envelope::encode_payload; +use smarm::cluster::expose::{ + decode_deliver, decoder_registered, expose, expose_type, exposed_hash, exposed_names, + type_hash, DeliverError, +}; +use smarm::monitor::{monitor, terminal_reason, DownReason}; +use smarm::{channel, register, run, spawn, Name}; + +const ROLES: &[(&str, fn())] = &[("hasher", role_hasher)]; + +/// Print the hashes this process computes; the parent (a different run of +/// the same binary) compares against its own. +fn role_hasher() { + println!("HASH-STRING {}", type_hash::()); + println!("HASH-U64 {}", type_hash::()); +} + +const GREETER: Name = Name::new("expose-test.greeter"); + +/// Exposed and unexposed lookup, the returned hash, and the audit listing. +#[test] +fn exposed_and_unexposed_lookup() { + maybe_child(ROLES); + run(|| { + let h = expose(GREETER); + assert_eq!(h, type_hash::()); + assert_eq!(exposed_hash("expose-test.greeter"), Some(h)); + assert_eq!(exposed_hash("never-exposed"), None); + assert!(exposed_names().contains(&("expose-test.greeter", h))); + }); +} + +/// Distinct types land on distinct hashes (FNV over distinct TypeIds — a +/// smoke assertion; a collision would degrade to a decode error, never a +/// misroute, per RFC §3). +#[test] +fn distinct_types_distinct_hashes() { + maybe_child(ROLES); + run(|| { + assert_ne!(type_hash::(), type_hash::()); + assert_ne!(type_hash::(), type_hash::>()); + }); +} + +/// The decode-and-deliver contract: a registered hash decodes into the +/// target's typed channel; an unknown hash, corrupt bytes, and a missing +/// channel each fail without delivering — `WrongChannel`, never a misroute. +#[test] +fn decoder_registration_and_delivery() { + maybe_child(ROLES); + run(|| { + let h_string = expose_type::(); + let h_u64 = expose_type::(); + assert!(decoder_registered(h_string)); + assert!(!decoder_registered(h_string.wrapping_add(1))); + + // A live actor with a String channel (registered from its own body, + // announced via a ready signal — the tests/registry.rs idiom). + let (ready_tx, ready_rx) = channel::<()>(); + let (stop_tx, stop_rx) = channel::<()>(); + let (msg_tx, msg_rx) = channel::(); + let pid = spawn(move || { + register(Name::::new("expose-test.sink"), msg_tx).unwrap(); + ready_tx.send(()).unwrap(); + let _ = stop_rx.recv(); + }) + .pid(); + ready_rx.recv().unwrap(); + + // Happy path: decode + deliver through the published channel. + let bytes = encode_payload("hello across the seam").unwrap(); + decode_deliver(h_string, pid, &bytes).unwrap(); + assert_eq!(msg_rx.recv().unwrap(), "hello across the seam"); + + // Unknown hash: nothing was registered under it. + assert!(matches!( + decode_deliver(h_string.wrapping_add(1), pid, &bytes), + Err(DeliverError::UnknownType) + )); + + // Corrupt bytes: the decoder fails before any send. + assert!(matches!( + decode_deliver(h_string, pid, &[0xff; 3]), + Err(DeliverError::Decode(_)) + )); + + // Right decoder, wrong channel: the actor has no u64 channel, so the + // decoded value is refused — the NoChannel guarantee. + let u64_bytes = encode_payload(&7u64).unwrap(); + assert!(matches!( + decode_deliver(h_u64, pid, &u64_bytes), + Err(DeliverError::WrongChannel) + )); + + stop_tx.send(()).unwrap(); + }); +} + +/// `expose` and the bridge crossing agree on the resulting set: both funnel +/// the pid-boundary mark through the watchable machinery, so an exposed +/// name's holder dies with a terminal record — the exact observable +/// `mark_watchable` guarantees the membrane. (For named holders the mark is +/// already stamped by `register` itself; this pins the shared contract.) +#[test] +fn expose_and_bridge_crossing_agree_on_the_set() { + maybe_child(ROLES); + run(|| { + let (ready_tx, ready_rx) = channel::<()>(); + let (stop_tx, stop_rx) = channel::<()>(); + let (msg_tx, _msg_rx) = channel::(); + let pid = spawn(move || { + register(GREETER, msg_tx).unwrap(); + ready_tx.send(()).unwrap(); + let _ = stop_rx.recv(); + }) + .pid(); + ready_rx.recv().unwrap(); + + expose(GREETER); + let m = monitor(pid); + stop_tx.send(()).unwrap(); + assert_eq!(m.rx.recv().unwrap().reason, DownReason::Exit); + assert_eq!(terminal_reason(pid), Some(DownReason::Exit)); + }); +} + +/// Hash stability across runs in the same binary: a re-exec of this binary +/// computes the same hashes this process does. +#[test] +fn hash_stable_across_runs_in_same_binary() { + maybe_child(ROLES); + let (mine_string, mine_u64) = { + // Computing a TypeId hash needs no runtime, but keep the contract + // uniform with real call sites. + (type_hash::(), type_hash::()) + }; + let mut child = spawn_node("hasher", &[]); + let line = child.wait_line("HASH-STRING", |l| l.starts_with("HASH-STRING ")); + assert_eq!( + line["HASH-STRING ".len()..].parse::().unwrap(), + mine_string + ); + let line = child.wait_line("HASH-U64", |l| l.starts_with("HASH-U64 ")); + assert_eq!(line["HASH-U64 ".len()..].parse::().unwrap(), mine_u64); + child.wait_exit(); +} diff --git a/tests/cluster_handshake.rs b/tests/cluster_handshake.rs new file mode 100644 index 0000000..fa4a755 --- /dev/null +++ b/tests/cluster_handshake.rs @@ -0,0 +1,240 @@ +//! RFC 010 c5 — handshake state-machine tests (roadmap: happy path; hash +//! mismatch; proto-version mismatch; name already claimed; simultaneous-connect +//! tie-break; garbage before Hello). Pure — no IO, no actors, no runtime. +#![cfg(feature = "cluster")] + +use smarm::cluster::envelope::{Frame, NodeMeta, RejectReason, PROTO_VERSION}; +use smarm::cluster::handshake::{ + dial_wins, Initiator, InitiatorOutcome, Local, PeerStanding, Responder, ResponderOutcome, +}; +use smarm::pg::Incarnation; + +const HASH: u64 = 0xDEAD_BEEF_CAFE_F00D; + +fn local(name: &str) -> Local { + Local { + node_name: name.into(), + incarnation: Incarnation::new(7), + build_hash: HASH, + meta: NodeMeta { + role: "worker".into(), + region: "eu-west".into(), + }, + } +} + +/// The Hello that `Initiator::new(&local(name))` emits, built by hand. +fn hello_from(name: &str) -> Frame { + let l = local(name); + Frame::Hello { + proto_version: PROTO_VERSION, + build_hash: l.build_hash, + node_name: l.node_name, + incarnation: l.incarnation, + meta: l.meta, + } +} + +#[test] +fn happy_path_establishes_both_ends() { + // alpha dials beta. + let (initiator, hello) = Initiator::new(&local("alpha")); + assert_eq!(hello, hello_from("alpha"), "initiator emits its identity"); + + let responder = Responder::new(local("beta")); + let (reply, peer) = match responder.on_frame(hello, PeerStanding::Free) { + ResponderOutcome::Accepted { reply, peer } => (reply, peer), + other => panic!("expected Accepted, got {other:?}"), + }; + assert_eq!(peer.node_name, "alpha"); + assert_eq!(peer.incarnation, Incarnation::new(7)); + assert_eq!(peer.meta.role, "worker"); + + // The ack carries the responder's identity, no hash/version (one-sided + // check — sound because equality is symmetric). + let l = local("beta"); + assert_eq!( + reply, + Frame::HelloAck { + node_name: l.node_name, + incarnation: l.incarnation, + meta: l.meta, + } + ); + + match initiator.on_frame(reply) { + InitiatorOutcome::Established(peer) => { + assert_eq!(peer.node_name, "beta"); + assert_eq!(peer.incarnation, Incarnation::new(7)); + assert_eq!(peer.meta.region, "eu-west"); + } + other => panic!("expected Established, got {other:?}"), + } +} + +#[test] +fn hash_mismatch_rejected() { + let responder = Responder::new(local("beta")); + let hello = Frame::Hello { + proto_version: PROTO_VERSION, + build_hash: HASH ^ 1, + node_name: "alpha".into(), + incarnation: Incarnation::new(7), + meta: local("alpha").meta, + }; + match responder.on_frame(hello, PeerStanding::Free) { + ResponderOutcome::Rejected { reply, reason } => { + assert_eq!(reason, RejectReason::HashMismatch); + assert_eq!(reply, Frame::HelloReject { reason }); + } + other => panic!("expected Rejected, got {other:?}"), + } + + // The dialer side of the same story: a reject frame comes back. + let (initiator, _hello) = Initiator::new(&local("alpha")); + match initiator.on_frame(Frame::HelloReject { + reason: RejectReason::HashMismatch, + }) { + InitiatorOutcome::Rejected(RejectReason::HashMismatch) => {} + other => panic!("expected Rejected(HashMismatch), got {other:?}"), + } +} + +#[test] +fn proto_version_mismatch_rejected_and_checked_first() { + // Both proto and hash wrong: proto wins — nothing after the version can + // be trusted, and HelloReject is the cross-version compatibility anchor. + let responder = Responder::new(local("beta")); + let hello = Frame::Hello { + proto_version: PROTO_VERSION + 1, + build_hash: HASH ^ 1, + node_name: "alpha".into(), + incarnation: Incarnation::new(7), + meta: local("alpha").meta, + }; + match responder.on_frame(hello, PeerStanding::Free) { + ResponderOutcome::Rejected { reason, .. } => { + assert_eq!(reason, RejectReason::ProtoVersion); + } + other => panic!("expected Rejected, got {other:?}"), + } +} + +#[test] +fn claimed_name_rejected() { + let responder = Responder::new(local("beta")); + let ctx = PeerStanding::Claimed; + match responder.on_frame(hello_from("alpha"), ctx) { + ResponderOutcome::Rejected { reason, .. } => { + assert_eq!(reason, RejectReason::NameTaken); + } + other => panic!("expected Rejected, got {other:?}"), + } +} + +#[test] +fn own_name_offered_rejected_as_name_taken() { + // Self-connect or genuine collision: the responder's own name arrives. + let responder = Responder::new(local("beta")); + match responder.on_frame(hello_from("beta"), PeerStanding::Free) { + ResponderOutcome::Rejected { reason, .. } => { + assert_eq!(reason, RejectReason::NameTaken); + } + other => panic!("expected Rejected, got {other:?}"), + } +} + +#[test] +fn hash_checked_before_name() { + // Wrong hash AND claimed name: hash wins (validity before identity). + let responder = Responder::new(local("beta")); + let hello = Frame::Hello { + proto_version: PROTO_VERSION, + build_hash: HASH ^ 1, + node_name: "alpha".into(), + incarnation: Incarnation::new(7), + meta: local("alpha").meta, + }; + let ctx = PeerStanding::Claimed; + match responder.on_frame(hello, ctx) { + ResponderOutcome::Rejected { reason, .. } => { + assert_eq!(reason, RejectReason::HashMismatch); + } + other => panic!("expected Rejected, got {other:?}"), + } +} + +#[test] +fn dial_wins_is_deterministic_and_antisymmetric() { + // The smaller name's dial survives; both ends compute the same verdict. + assert!(dial_wins("alpha", "beta")); + assert!(!dial_wins("beta", "alpha")); + for (a, b) in [("a", "b"), ("node-1", "node-2"), ("x", "xx")] { + assert_ne!(dial_wins(a, b), dial_wins(b, a), "({a}, {b})"); + } +} + +#[test] +fn simultaneous_connect_exactly_one_side_accepts() { + // alpha and beta dial each other at once. Each responder sees the peer's + // Hello while its own dial is in flight. + let ctx = PeerStanding::Dialing; + + // On beta: inbound is alpha's dial; alpha < beta, so the inbound wins. + let on_beta = Responder::new(local("beta")).on_frame(hello_from("alpha"), ctx); + assert!( + matches!(on_beta, ResponderOutcome::Accepted { .. }), + "beta must accept alpha's dial, got {on_beta:?}" + ); + + // On alpha: inbound is beta's dial; it loses — close silently, no frame + // (ratified: both ends can compute the outcome, a reject adds nothing). + let on_alpha = Responder::new(local("alpha")).on_frame(hello_from("beta"), ctx); + assert!( + matches!(on_alpha, ResponderOutcome::TieBreakLoss), + "alpha must silently drop beta's dial, got {on_alpha:?}" + ); +} + +#[test] +fn tiebreak_loss_only_applies_when_dialing() { + // Same inbound Hello, no dial in flight: plain accept. + let on_alpha = Responder::new(local("alpha")).on_frame(hello_from("beta"), PeerStanding::Free); + assert!(matches!(on_alpha, ResponderOutcome::Accepted { .. })); +} + +#[test] +fn garbage_before_hello_fails_without_reply() { + // Any valid-but-wrong frame before Hello is a protocol violation: close, + // no reject frame. (Undecodable bytes are the codec's Err, not ours.) + for frame in [ + Frame::Heartbeat, + Frame::HelloAck { + node_name: "alpha".into(), + incarnation: Incarnation::new(7), + meta: local("alpha").meta, + }, + Frame::Demonitor { monitor_id: 3 }, + ] { + let out = Responder::new(local("beta")).on_frame(frame.clone(), PeerStanding::Free); + match out { + ResponderOutcome::Failed(f) => assert_eq!(f, frame), + other => panic!("expected Failed({frame:?}), got {other:?}"), + } + } +} + +#[test] +fn garbage_before_ack_fails_the_initiator() { + for frame in [ + Frame::Heartbeat, + hello_from("beta"), + Frame::Demonitor { monitor_id: 3 }, + ] { + let (initiator, _hello) = Initiator::new(&local("alpha")); + match initiator.on_frame(frame.clone()) { + InitiatorOutcome::Failed(f) => assert_eq!(f, frame), + other => panic!("expected Failed({frame:?}), got {other:?}"), + } + } +} diff --git a/tests/cluster_membership.rs b/tests/cluster_membership.rs new file mode 100644 index 0000000..6a597be --- /dev/null +++ b/tests/cluster_membership.rs @@ -0,0 +1,270 @@ +//! RFC 010 c7a — membership events and the view, at the manager. +//! +//! Same construction as the c6a lifecycle suite: the handshake is bypassed, +//! connections are built already-established over localhost TCP pairs with +//! fabricated `Peer`s, and the manager is started plainly so the test can +//! terminate. What is under test is the membership layer that c7 adds to the +//! manager: `node_up`/`node_down` events to subscribers (snapshot-then-stream), +//! the view, and NodeId identity — memoized per `(name, incarnation)`, so a +//! reconnect blip keeps its id and a restart (new incarnation) gets a fresh one. +#![cfg(feature = "cluster")] + +use std::time::Duration; + +use smarm::cluster::envelope::NodeMeta; +use smarm::cluster::handshake::Peer; +use smarm::cluster::manager::{Call, Manager, Reply, MANAGER}; +use smarm::cluster::membership::{subscribe, view, MembershipEvents, NodeEvent}; +use smarm::cluster::spawn_established; +use smarm::cluster::transport::tcp::TcpTransport; +use smarm::cluster::transport::{Conn, FramedConn, Transport}; +use smarm::cluster::Timing; +use smarm::gen_server::{self, GenServerBuilder}; +use smarm::pg::{Incarnation, NodeId}; +use smarm::run; + +/// A fabricated post-handshake peer identity, with the incarnation under the +/// test's control (it is identity-bearing here, unlike in the c6a suite). +fn peer(name: &str, inc: u32) -> Peer { + Peer { + node_name: name.to_string(), + incarnation: Incarnation::new(inc), + meta: NodeMeta { + role: "test".to_string(), + region: "test".to_string(), + }, + } +} + +/// One established transport pair over localhost (TCP backlog covers the +/// sequential dial-then-accept, as in the c3 conformance suite). +fn pair(t: &dyn Transport) -> (Box, Box) { + let mut l = t.listen("127.0.0.1:0").unwrap(); + let a = t.dial(&l.local_addr()).unwrap(); + let b = l.accept().unwrap(); + (a, b) +} + +/// The next event, or a panic naming the wait. The bound is generous against +/// a sub-millisecond real cost. +fn next_event(ev: &MembershipEvents, waiting_for: &str) -> NodeEvent { + ev.rx + .recv_timeout(Duration::from_secs(5)) + .unwrap_or_else(|e| panic!("timed out waiting for {waiting_for}: {e:?}")) +} + +/// Assert the subscription is drained: no event is pending. +fn assert_quiet(ev: &MembershipEvents) { + assert!(matches!(ev.rx.try_recv(), Ok(None))); +} + +fn disconnect(name: &str) { + assert!(matches!( + gen_server::call( + MANAGER, + Call::Disconnect { + name: name.to_string() + } + ), + Ok(Reply::Disconnected) + )); +} + +/// Live subscription: an empty snapshot, then `NodeUp` on registration and +/// `NodeDown` (same id) on commanded disconnect and on peer EOF alike. +#[test] +fn subscriber_sees_up_and_down() { + run(|| { + let mgr = GenServerBuilder::new(Manager::new()) + .named(MANAGER) + .start() + .expect("manager name is free"); + + let ev = subscribe().expect("manager is up"); + assert_quiet(&ev); // nothing live: the snapshot is empty + + let t = TcpTransport; + let (a1, b1) = pair(&t); + let (a2, b2) = pair(&t); + spawn_established(FramedConn::new(a1), peer("node-b", 1), Timing::default()) + .expect("node-b registers"); + spawn_established(FramedConn::new(a2), peer("node-c", 1), Timing::default()) + .expect("node-c registers"); + + let up_b = match next_event(&ev, "node_up(node-b)") { + NodeEvent::NodeUp(info) => { + assert_eq!(info.name, "node-b"); + assert_eq!(info.incarnation, Incarnation::new(1)); + assert_eq!(info.meta.role, "test"); + info + } + other => panic!("expected node_up(node-b), got {other:?}"), + }; + let up_c = match next_event(&ev, "node_up(node-c)") { + NodeEvent::NodeUp(info) => { + assert_eq!(info.name, "node-c"); + info + } + other => panic!("expected node_up(node-c), got {other:?}"), + }; + assert_ne!(up_b.node, up_c.node, "distinct peers get distinct ids"); + + // Commanded disconnect: down with node-b's id. + disconnect("node-b"); + assert_eq!( + next_event(&ev, "node_down(node-b)"), + NodeEvent::NodeDown(up_b.clone()) + ); + + // Peer EOF, no command: down with node-c's id. + drop(b2); + assert_eq!( + next_event(&ev, "node_down(node-c)"), + NodeEvent::NodeDown(up_c.clone()) + ); + assert_quiet(&ev); + + drop(b1); + mgr.shutdown(); + }); +} + +/// Snapshot-then-stream: a subscriber arriving after connections established +/// receives one `NodeUp` per live peer before anything else, and the view +/// call agrees with it. +#[test] +fn late_subscriber_gets_snapshot_and_view_agrees() { + run(|| { + let mgr = GenServerBuilder::new(Manager::new()) + .named(MANAGER) + .start() + .expect("manager name is free"); + + let t = TcpTransport; + let (a1, b1) = pair(&t); + let (a2, b2) = pair(&t); + spawn_established(FramedConn::new(a1), peer("node-b", 1), Timing::default()) + .expect("node-b registers"); + spawn_established(FramedConn::new(a2), peer("node-c", 1), Timing::default()) + .expect("node-c registers"); + + let ev = subscribe().expect("manager is up"); + let mut names = Vec::new(); + for _ in 0..2 { + match next_event(&ev, "a snapshot node_up") { + NodeEvent::NodeUp(info) => names.push(info.name), + other => panic!("expected a snapshot node_up, got {other:?}"), + } + } + names.sort(); + assert_eq!(names, ["node-b", "node-c"]); + assert_quiet(&ev); // the snapshot is exactly the live set + + let mut v = view().expect("manager is up"); + v.sort_by(|a, b| a.name.cmp(&b.name)); + assert_eq!(v.len(), 2); + assert_eq!(v[0].name, "node-b"); + assert_eq!(v[1].name, "node-c"); + + disconnect("node-b"); + disconnect("node-c"); + drop((b1, b2)); + // Drain the two downs so the subscription ends quiet. + let _ = next_event(&ev, "node_down"); + let _ = next_event(&ev, "node_down"); + mgr.shutdown(); + }); +} + +/// NodeId identity: a restart (same name, new incarnation) is a NEW id — the +/// ghost and its successor are distinguishable — while a reconnect blip (same +/// name, same incarnation) keeps its id. +#[test] +fn restart_gets_new_id_blip_keeps_id() { + run(|| { + let mgr = GenServerBuilder::new(Manager::new()) + .named(MANAGER) + .start() + .expect("manager name is free"); + let ev = subscribe().expect("manager is up"); + let t = TcpTransport; + + let id = |e: NodeEvent, what: &str| -> NodeId { + match e { + NodeEvent::NodeUp(info) => info.node, + other => panic!("expected node_up ({what}), got {other:?}"), + } + }; + let down_id = |e: NodeEvent, what: &str| -> NodeId { + match e { + NodeEvent::NodeDown(info) => info.node, + other => panic!("expected node_down ({what}), got {other:?}"), + } + }; + + // Up at incarnation 1, then the peer dies (EOF). + let (a1, b1) = pair(&t); + spawn_established(FramedConn::new(a1), peer("node-b", 1), Timing::default()) + .expect("registers"); + let id1 = id(next_event(&ev, "node_up inc 1"), "inc 1"); + drop(b1); + assert_eq!(down_id(next_event(&ev, "node_down inc 1"), "inc 1"), id1); + + // Restart: new incarnation, new id — the ghost's id is not reused. + let (a2, b2) = pair(&t); + spawn_established(FramedConn::new(a2), peer("node-b", 2), Timing::default()) + .expect("registers"); + let id2 = id(next_event(&ev, "node_up inc 2"), "inc 2"); + assert_ne!( + id1, id2, + "a restarted node must be distinguishable from its ghost" + ); + + // Blip: the same incarnation reconnects and keeps its id. + disconnect("node-b"); + assert_eq!(down_id(next_event(&ev, "node_down inc 2"), "inc 2"), id2); + let (a3, b3) = pair(&t); + spawn_established(FramedConn::new(a3), peer("node-b", 2), Timing::default()) + .expect("registers"); + let id3 = id(next_event(&ev, "node_up after blip"), "blip"); + assert_eq!( + id2, id3, + "a reconnect at the same incarnation is the same node" + ); + + disconnect("node-b"); + let _ = next_event(&ev, "final node_down"); + drop((b2, b3)); + mgr.shutdown(); + }); +} + +/// A dropped subscriber is pruned on the next emit and never disturbs the +/// manager or a live subscriber. +#[test] +fn dead_subscriber_is_pruned() { + run(|| { + let mgr = GenServerBuilder::new(Manager::new()) + .named(MANAGER) + .start() + .expect("manager name is free"); + + let dead = subscribe().expect("manager is up"); + drop(dead); + let live = subscribe().expect("manager is up"); + + let t = TcpTransport; + let (a1, b1) = pair(&t); + spawn_established(FramedConn::new(a1), peer("node-b", 1), Timing::default()) + .expect("registers"); + match next_event(&live, "node_up despite a dead co-subscriber") { + NodeEvent::NodeUp(info) => assert_eq!(info.name, "node-b"), + other => panic!("expected node_up, got {other:?}"), + } + + disconnect("node-b"); + let _ = next_event(&live, "node_down"); + drop(b1); + mgr.shutdown(); + }); +} diff --git a/tests/cluster_mesh.rs b/tests/cluster_mesh.rs new file mode 100644 index 0000000..08451d4 --- /dev/null +++ b/tests/cluster_mesh.rs @@ -0,0 +1,175 @@ +//! RFC 010 c7 — the Phase 2 gate: a 3-node mesh under the subprocess +//! harness, repeatable. +//! +//! Each node process runs the integrated `cluster::start` (manager + +//! acceptor + connector + static seeds), subscribes to membership like any +//! consumer, and announces protocol-visible facts as lines: +//! `LISTENING `, `MEMBER-UP inc=`, `MEMBER-DOWN `. +//! Then it **parks forever** — cross-process teardown is retractable state +//! (binding trap), so the parent SIGKILLs via `Node`'s `Drop` and clean exit +//! stays the c4 harness's own smoke test. +//! +//! Ports: nodes bind `:0` and report, so the mesh is built by seeding each +//! node with the previously-reported addresses (n1: no seeds; n2: n1; +//! n3: n1+n2 — inbound covers the reverse edges). The late-seed test is the +//! one exception: the parent pre-reserves a port by binding-and-closing it, +//! seeds one node with it, then starts the second node on that exact +//! address. In principle another process could steal the port in the gap; +//! in practice the window is microseconds on a local runner — accepted, and +//! confined to that one test. +#![cfg(feature = "cluster")] + +mod common; + +use common::{maybe_child, spawn_node, Node}; +use smarm::cluster::envelope::NodeMeta; +use smarm::cluster::membership::{subscribe, NodeEvent}; +use smarm::cluster::{start, Config, StaticSeeds, Timing}; +use std::time::Duration; + +const ROLES: &[(&str, fn())] = &[("node", role_node)]; + +/// A mesh node: identity and seeds from env, membership events to stdout, +/// park forever (the parent reaps). +fn role_node() { + let name = std::env::var("SMARM_NODE_NAME").expect("SMARM_NODE_NAME not set"); + let listen = std::env::var("SMARM_LISTEN_ADDR").unwrap_or_else(|_| "127.0.0.1:0".to_string()); + // Seeds: comma-separated `name=addr` pairs; empty or unset means none. + let seeds: Vec<(String, String)> = std::env::var("SMARM_SEEDS") + .unwrap_or_default() + .split(',') + .filter(|s| !s.is_empty()) + .map(|s| { + let (n, a) = s.split_once('=').expect("seed must be name=addr"); + (n.to_string(), a.to_string()) + }) + .collect(); + + smarm::run(move || { + let cluster = start(Config { + node_name: name, + meta: NodeMeta { + role: "mesh-test".to_string(), + region: "local".to_string(), + }, + listen_addr: listen, + strategy: Box::new(StaticSeeds::new(seeds)), + timing: Timing::default(), + }) + .expect("listener binds"); + println!("LISTENING {}", cluster.local_addr()); + + let events = subscribe().expect("manager is up"); + loop { + match events.rx.recv() { + Ok(NodeEvent::NodeUp(info)) => { + println!("MEMBER-UP {} inc={}", info.name, info.incarnation.get()); + } + Ok(NodeEvent::NodeDown(info)) => { + println!("MEMBER-DOWN {}", info.name); + } + Err(_) => break, // manager gone; park below regardless + } + } + loop { + smarm::sleep(Duration::from_secs(3600)); + } + }); +} + +fn spawn_mesh_node(name: &str, seeds: &str, listen: Option<&str>) -> Node { + let mut env: Vec<(&str, &str)> = vec![("SMARM_NODE_NAME", name), ("SMARM_SEEDS", seeds)]; + if let Some(addr) = listen { + env.push(("SMARM_LISTEN_ADDR", addr)); + } + spawn_node("node", &env) +} + +/// Wait for `MEMBER-UP inc=` and return the incarnation. +fn wait_member_up(node: &mut Node, peer: &str) -> u32 { + let prefix = format!("MEMBER-UP {peer} inc="); + let line = node.wait_line(&format!("MEMBER-UP {peer}"), |l| l.starts_with(&prefix)); + line[prefix.len()..].parse().expect("incarnation parses") +} + +fn wait_member_down(node: &mut Node, peer: &str) { + let want = format!("MEMBER-DOWN {peer}"); + node.wait_line(&want, |l| l == want); +} + +/// The gate, plus the kill and restart facts, as one mesh's life: three +/// nodes form a full mesh (every node sees both others up); killing one +/// yields `node_down` at both survivors; its restart under the same name +/// arrives as a NEW incarnation — the ghost and its successor are +/// distinguishable at every observer. +#[test] +fn three_node_mesh_forms_then_kill_then_restart_distinguishable() { + maybe_child(ROLES); + + let mut n1 = spawn_mesh_node("node-1", "", None); + let a1 = n1.wait_listening(); + let mut n2 = spawn_mesh_node("node-2", &format!("node-1={a1}"), None); + let a2 = n2.wait_listening(); + let mut n3 = spawn_mesh_node("node-3", &format!("node-1={a1},node-2={a2}"), None); + let _a3 = n3.wait_listening(); + + // Full mesh: each node reports both peers up (dialed or inbound alike). + wait_member_up(&mut n1, "node-2"); + let inc3_at_n1 = wait_member_up(&mut n1, "node-3"); + wait_member_up(&mut n2, "node-1"); + let inc3_at_n2 = wait_member_up(&mut n2, "node-3"); + wait_member_up(&mut n3, "node-1"); + wait_member_up(&mut n3, "node-2"); + assert_eq!( + inc3_at_n1, inc3_at_n2, + "one node, one incarnation, all observers" + ); + + // Kill node-3 (SIGKILL via Drop): node_down at both survivors. + drop(n3); + wait_member_down(&mut n1, "node-3"); + wait_member_down(&mut n2, "node-3"); + + // Restart node-3 under the same name: it re-dials its seeds and comes + // up everywhere as a new incarnation — never the ghost's. + let mut n3b = spawn_mesh_node("node-3", &format!("node-1={a1},node-2={a2}"), None); + let _ = n3b.wait_listening(); + let inc3b_at_n1 = wait_member_up(&mut n1, "node-3"); + let inc3b_at_n2 = wait_member_up(&mut n2, "node-3"); + assert_eq!(inc3b_at_n1, inc3b_at_n2); + assert_ne!( + inc3_at_n1, inc3b_at_n1, + "a restarted node must be distinguishable from its ghost" + ); + wait_member_up(&mut n3b, "node-1"); + wait_member_up(&mut n3b, "node-2"); +} + +/// A seed that is unreachable at start is not fatal: the connector retries +/// on backoff, and when a node finally appears at that address, the mesh +/// edge forms. +#[test] +fn seed_unreachable_at_start_then_arriving_later() { + maybe_child(ROLES); + + // Pre-reserve an address by binding and immediately closing it (see the + // module docs for the accepted steal window). Dials to it are refused + // until node-b starts there. + let reserved = { + let l = std::net::TcpListener::bind("127.0.0.1:0").expect("bind"); + l.local_addr().expect("addr").to_string() + }; + + let mut a = spawn_mesh_node("node-a", &format!("node-b={reserved}"), None); + let _ = a.wait_listening(); + + // Let a few refused attempts happen before the seed comes up, so the + // retry path is what forms the edge (backoff cap 5s < harness WAIT 10s). + std::thread::sleep(Duration::from_millis(600)); + + let mut b = spawn_mesh_node("node-b", "", Some(&reserved)); + let _ = b.wait_listening(); + + wait_member_up(&mut a, "node-b"); + wait_member_up(&mut b, "node-a"); +} diff --git a/tests/cluster_monitor.rs b/tests/cluster_monitor.rs new file mode 100644 index 0000000..8b141ee --- /dev/null +++ b/tests/cluster_monitor.rs @@ -0,0 +1,359 @@ +//! RFC 010 c12 — remote monitors. +//! +//! Local suite (`run()`, no network): the immediate answers — no connection +//! ⇒ `Disconnected`, dead incarnation ⇒ `NoProc` — and the self-node +//! collapse (a plain local monitor underneath, incl. `demonitor_remote`). +//! +//! Cross-process: a *server* exposes a control name and spawns workers on +//! request, replying with each worker's pid (via `RemotePid::from_local`, +//! the D12 set-site) or, for the deliberately unshipped one, only its raw +//! slot numbers. The *client* monitors them and asserts: kill ⇒ the true +//! reason (Exit / Panic); a corpse ⇒ its recorded terminal reason, not +//! NoProc; a live pid that never crossed the wire ⇒ NoProc (no liveness +//! leak); a demonitor racing the kill ⇒ no notice, proven by stream ORDER +//! (a later notice on the same connection arrives while the earlier slot +//! is still empty), not by sleeping. +#![cfg(feature = "cluster")] + +mod common; + +use common::{maybe_child, spawn_node}; +use smarm::cluster::envelope::NodeMeta; +use smarm::cluster::expose::{expose, expose_type}; +use smarm::cluster::membership::{subscribe, NodeEvent}; +use smarm::cluster::remote::{ + self, demonitor_remote, monitor_remote, send_to_remote, RemoteName, RemotePid, +}; +use smarm::cluster::{start, Config, RemoteDownReason, StaticSeeds, Timing}; +use smarm::pg::Incarnation; +use smarm::{channel, install, register, run, spawn, Addressable, DownReason, Erased, Name, Pid}; +use std::collections::HashMap; +use std::time::Duration; + +// ---- message types (hand-rolled serde; the crate is derive-less) --------- + +#[derive(Debug)] +struct Ctl { + cmd: String, + reply_to: RemotePid, +} +#[derive(Debug)] +struct Answer { + text: String, + pid: Option>, +} +struct Client; +impl Addressable for Client { + type Msg = Answer; +} + +impl serde::Serialize for Ctl { + fn serialize(&self, s: S) -> Result { + use serde::ser::SerializeTuple; + let mut t = s.serialize_tuple(2)?; + t.serialize_element(&self.cmd)?; + t.serialize_element(&self.reply_to)?; + t.end() + } +} +impl<'de> serde::Deserialize<'de> for Ctl { + fn deserialize>(d: D) -> Result { + let (cmd, reply_to) = <(String, RemotePid)>::deserialize(d)?; + Ok(Ctl { cmd, reply_to }) + } +} +impl serde::Serialize for Answer { + fn serialize(&self, s: S) -> Result { + use serde::ser::SerializeTuple; + let mut t = s.serialize_tuple(2)?; + t.serialize_element(&self.text)?; + t.serialize_element(&self.pid)?; + t.end() + } +} +impl<'de> serde::Deserialize<'de> for Answer { + fn deserialize>(d: D) -> Result { + let (text, pid) = <(String, Option>)>::deserialize(d)?; + Ok(Answer { text, pid }) + } +} + +// ================= local suite ========================================= + +/// No connection to the pid's node: `Disconnected` at once — the remote +/// analog of NoProc, and the first thing c11's variant is for. +#[test] +fn unconnected_node_is_disconnected_immediately() { + maybe_child(ROLES); + run(|| { + remote::set_local_identity("me", Incarnation::new(7)); + let ghost = RemotePid::::from_parts("nowhere", Incarnation::new(1), 3, 1); + let m = monitor_remote(ghost.clone()); + let d = m.recv().unwrap(); + assert_eq!(d.pid, ghost); + assert_eq!(d.reason, RemoteDownReason::Disconnected); + }); +} + +/// The node is connected but the pid names an earlier incarnation: the +/// actor is a known corpse (RFC v2 §3), so `NoProc` at once — never +/// `Disconnected`, nothing on the wire. +#[test] +fn dead_incarnation_is_noproc_immediately() { + maybe_child(ROLES); + run(|| { + remote::set_local_identity("me", Incarnation::new(7)); + let (probe_tx, probe_rx) = channel(); + remote::bind_outbound_probe("peer", Incarnation::new(5), probe_tx); + let stale = RemotePid::::from_parts("peer", Incarnation::new(4), 9, 1); + let m = monitor_remote(stale); + assert_eq!(m.recv().unwrap().reason, DownReason::NoProc.into()); + assert!(probe_rx.try_recv().unwrap().is_none(), "no frame emitted"); + }); +} + +/// A self-node pid collapses to an ordinary local monitor: the true reason +/// on exit, and `demonitor_remote` cancels it. +#[test] +fn self_node_pid_collapses_to_local_monitor() { + maybe_child(ROLES); + run(|| { + remote::set_local_identity("me", Incarnation::new(7)); + let (go_tx, go_rx) = channel::<()>(); + let (go2_tx, go2_rx) = channel::<()>(); + let a = spawn(move || { + let _ = go_rx.recv(); + }) + .pid(); + let b = spawn(move || { + let _ = go2_rx.recv(); + }) + .pid(); + let ma = monitor_remote(RemotePid::from_local(a).expect("identity set")); + let mb = monitor_remote(RemotePid::from_local(b).expect("identity set")); + assert_ne!(ma.id, mb.id); + assert!(ma.target.local() == Some(a)); + + demonitor_remote(&mb); + go2_tx.send(()).unwrap(); + go_tx.send(()).unwrap(); + let d = ma.recv().unwrap(); + assert_eq!(d.reason, DownReason::Exit.into()); + assert_eq!(d.pid.local(), Some(a)); + // `a` is down (its notice arrived), and `b` was killed first on the + // same scheduler — a notice for `b` would be here by now. After a + // demonitor the channel is closed-empty (`Err`), like the local one. + assert!(matches!(mb.try_recv(), Ok(None) | Err(_))); + }); +} + +// ================= cross-process ====================================== + +const ROLES: &[(&str, fn())] = &[("server", role_server), ("client", role_client)]; + +const CTL: Name = Name::new("c12.ctl"); + +fn cfg(name: &str, seeds: Vec<(String, String)>) -> Config { + Config { + node_name: name.to_string(), + meta: NodeMeta { + role: "c12".into(), + region: "local".into(), + }, + listen_addr: std::env::var("SMARM_LISTEN_ADDR").unwrap_or_else(|_| "127.0.0.1:0".into()), + strategy: Box::new(StaticSeeds::new(seeds)), + timing: Timing::default(), + } +} + +fn wait_up(events: &smarm::cluster::membership::MembershipEvents, who: &str) { + loop { + match events.rx.recv() { + Ok(NodeEvent::NodeUp(i)) if i.name == who => return, + Ok(_) => continue, + Err(_) => panic!("manager gone"), + } + } +} + +/// Server commands (all answered to `reply_to`): +/// - `spawn:exit` / `spawn:panic` — a parked worker; `kill:` releases +/// it, whereupon it returns / panics. Answer carries its pid. +/// - `spawn:corpse` — a worker that has already exited when the answer is +/// sent; the pid was shipped (watchable) before it died. +/// - `spawn:unwatched` — a parked worker whose pid is NEVER shipped; the +/// answer carries only `text = "slot::"`. +fn role_server() { + smarm::run(move || { + let cluster = start(cfg("server", vec![])).expect("binds"); + println!("LISTENING {}", cluster.local_addr()); + let (tx, rx) = channel::(); + register(CTL, tx).unwrap(); + expose(CTL); + println!("READY"); + let mut workers: HashMap> = HashMap::new(); + loop { + let ctl = rx.recv().unwrap(); + println!("CTL {}", ctl.cmd); + let (text, pid): (String, Option>) = match ctl.cmd.as_str() { + "spawn:exit" | "spawn:panic" => { + let panic = ctl.cmd == "spawn:panic"; + let (go_tx, go_rx) = channel::<()>(); + let p: Pid = spawn(move || { + let _ = go_rx.recv(); + if panic { + panic!("worker asked to panic"); + } + }) + .pid(); + workers.insert(p.index(), go_tx); + ( + "ok".into(), + Some(RemotePid::from_local(p).expect("identity set")), + ) + } + "spawn:corpse" => { + let p: Pid = spawn(|| {}).pid(); + let rp = RemotePid::from_local(p).expect("identity set"); // shipped ⇒ watchable + let m = smarm::monitor(p); + let _ = m.rx.recv(); // dead before the answer goes out + ("ok".into(), Some(rp)) + } + "spawn:unwatched" => { + let (go_tx, go_rx) = channel::<()>(); + let p: Pid = spawn(move || { + let _ = go_rx.recv(); + }) + .pid(); + workers.insert(p.index(), go_tx); + (format!("slot:{}:{}", p.index(), p.generation()), None) + } + other => { + let idx: u32 = other.strip_prefix("kill:").unwrap().parse().unwrap(); + if let Some(go) = workers.remove(&idx) { + let _ = go.send(()); + } + ("killed".into(), None) + } + }; + send_to_remote(ctl.reply_to, Answer { text, pid }).unwrap(); + } + }); +} + +fn role_client() { + let server_addr = std::env::var("SMARM_SERVER_ADDR").expect("SMARM_SERVER_ADDR"); + smarm::run(move || { + let _cluster = start(cfg("client", vec![("server".into(), server_addr)])).expect("binds"); + let ev = subscribe().unwrap(); + wait_up(&ev, "server"); + let (tx, rx) = channel::(); + let me: Pid = install::(tx); + expose_type::(); + let ask = |cmd: &str| -> Answer { + remote::send( + RemoteName::new("server", CTL), + Ctl { + cmd: cmd.into(), + reply_to: RemotePid::from_local(me).expect("identity set"), + }, + ) + .unwrap(); + rx.recv().unwrap() + }; + let server_inc = ev_incarnation(); + + // 1. kill ⇒ true reason (Exit). + let a = ask("spawn:exit").pid.unwrap(); + let ma = monitor_remote(a.clone()); + ask(&format!("kill:{}", a.index())); + let d = ma.recv().unwrap(); + assert_eq!(d.pid, a); + println!("DOWN exit {:?}", d.reason); + + // 2. kill ⇒ true reason (Panic). + let b = ask("spawn:panic").pid.unwrap(); + let mb = monitor_remote(b.clone()); + ask(&format!("kill:{}", b.index())); + println!("DOWN panic {:?}", mb.recv().unwrap().reason); + + // 3. corpse ⇒ recorded terminal reason, not NoProc. + let c = ask("spawn:corpse").pid.unwrap(); + println!("DOWN corpse {:?}", monitor_remote(c).recv().unwrap().reason); + + // 4. live but never shipped/exposed ⇒ NoProc (no leak); a made-up + // slot on the same node ⇒ NoProc too, indistinguishably. + let ans = ask("spawn:unwatched"); + let mut it = ans.text.strip_prefix("slot:").unwrap().split(':'); + let (idx, gen): (u32, u32) = ( + it.next().unwrap().parse().unwrap(), + it.next().unwrap().parse().unwrap(), + ); + let hidden = RemotePid::::from_parts("server", server_inc, idx, gen); + println!( + "DOWN hidden {:?}", + monitor_remote(hidden).recv().unwrap().reason + ); + let bogus = RemotePid::::from_parts("server", server_inc, 100_000, 1); + println!( + "DOWN bogus {:?}", + monitor_remote(bogus).recv().unwrap().reason + ); + + // 5. demonitor races the kill: no notice for `d1`, proven by order — + // `d2`'s notice (same connection, later) arrives while `d1`'s + // slot is still empty. + let d1 = ask("spawn:exit").pid.unwrap(); + let m1 = monitor_remote(d1.clone()); + demonitor_remote(&m1); + ask(&format!("kill:{}", d1.index())); + let d2 = ask("spawn:exit").pid.unwrap(); + let m2 = monitor_remote(d2.clone()); + ask(&format!("kill:{}", d2.index())); + assert_eq!(m2.recv().unwrap().reason, DownReason::Exit.into()); + // Closed-empty (`Err`) or open-empty (`Ok(None)`) both mean no notice. + let stray = matches!(m1.try_recv(), Ok(Some(_))); + println!("DEMONITOR stray={stray}"); + + println!("CLIENT DONE"); + loop { + smarm::sleep(Duration::from_secs(3600)); + } + }); +} + +/// The server's incarnation as this node sees it — for building pids by hand. +fn ev_incarnation() -> Incarnation { + smarm::cluster::membership::view() + .expect("manager up") + .into_iter() + .find(|i| i.name == "server") + .map(|i| i.incarnation) + .expect("server in view") +} + +/// The Phase 4 c12 gate: remote monitors report the true reason, honour +/// corpses, leak nothing for unshipped pids, and cancel cleanly. +#[test] +fn remote_monitors_report_true_reasons() { + maybe_child(ROLES); + let mut server = spawn_node("server", &[]); + let saddr = server.wait_listening(); + server.wait_line("READY", |l| l == "READY"); + let mut client = spawn_node("client", &[("SMARM_SERVER_ADDR", &saddr)]); + client.wait_line("DOWN exit Local(Exit)", |l| l == "DOWN exit Local(Exit)"); + client.wait_line("DOWN panic Local(Panic)", |l| { + l == "DOWN panic Local(Panic)" + }); + client.wait_line("DOWN corpse Local(Exit)", |l| { + l == "DOWN corpse Local(Exit)" + }); + client.wait_line("DOWN hidden Local(NoProc)", |l| { + l == "DOWN hidden Local(NoProc)" + }); + client.wait_line("DOWN bogus Local(NoProc)", |l| { + l == "DOWN bogus Local(NoProc)" + }); + client.wait_line("DEMONITOR stray=false", |l| l == "DEMONITOR stray=false"); + client.wait_line("CLIENT DONE", |l| l == "CLIENT DONE"); +} diff --git a/tests/cluster_pg.rs b/tests/cluster_pg.rs new file mode 100644 index 0000000..626b4b8 --- /dev/null +++ b/tests/cluster_pg.rs @@ -0,0 +1,254 @@ +//! RFC 010 c15 — distributed pg: sync on `NodeUp`, incremental +//! `Join`/`Leave`, eager eviction announced, `NodeDown` sweep. +//! +//! Two nodes. The *origin* joins two local workers to `"pool"` before the +//! *observer* connects (so the observer's view comes from `Sync`), exposes a +//! `"go"` command inbox and then does exactly what the observer tells it: +//! kill one worker, join a third, leave with the second. The observer drives +//! that script through the cluster itself and asserts every step from +//! `members_all` — never touching the group on its own side, except once to +//! prove a mixed local+remote group reads correctly and that `members` stays +//! local. `dispatch_any` is exercised both ways: into the origin's worker +//! (remote pick, `send_to_remote`) and, once the origin is gone, into the +//! observer's own (local pick, `send_to`). Finally the parent SIGKILLs the +//! origin: the observer must sweep every remote member on `NodeDown`. +#![cfg(feature = "cluster")] + +mod common; + +use common::{maybe_child, spawn_node}; +use smarm::cluster::envelope::NodeMeta; +use smarm::cluster::expose::{expose, expose_type}; +use smarm::cluster::membership::{subscribe, NodeEvent}; +use smarm::cluster::remote::{self, RemoteName}; +use smarm::cluster::{ + dispatch_any, members_all, pick_any, start, Config, DispatchAnyError, GroupMember, StaticSeeds, + Timing, +}; +use smarm::{channel, join, leave, members, register, send_to, spawn_addr, Addressable, Name, Pid}; +use std::time::{Duration, Instant}; + +const GO: Name = Name::new("go"); +const POOL: &str = "pool"; + +/// A pool worker's message: `"die"` stops it, anything else is printed. +#[derive(Debug, PartialEq)] +struct Job(String); +struct Worker; +impl Addressable for Worker { + type Msg = Job; +} +impl serde::Serialize for Job { + fn serialize(&self, s: S) -> Result { + self.0.serialize(s) + } +} +impl<'de> serde::Deserialize<'de> for Job { + fn deserialize>(d: D) -> Result { + String::deserialize(d).map(Job) + } +} + +const ROLES: &[(&str, fn())] = &[("origin", role_origin), ("observer", role_observer)]; + +fn cfg(name: &str, seeds: Vec<(String, String)>) -> Config { + Config { + node_name: name.into(), + meta: NodeMeta { + role: "c15".into(), + region: "local".into(), + }, + listen_addr: "127.0.0.1:0".into(), + strategy: Box::new(StaticSeeds::new(seeds)), + timing: Timing::default(), + } +} + +/// A pool worker: prints every job it is handed, exits on `"die"`. +fn worker() -> Pid { + spawn_addr::(|rx| { + while let Ok(Job(s)) = rx.recv() { + if s == "die" { + return; + } + println!("JOB {s}"); + } + }) +} + +fn role_origin() { + smarm::run(|| { + let cluster = start(cfg("origin", vec![])).expect("binds"); + // Remote dispatch lands here only for a type this node accepts. + expose_type::(); + let w1 = worker(); + let w2 = worker(); + assert!(join(POOL, w1)); + assert!(join(POOL, w2)); + let (go_tx, go_rx) = channel::(); + register(GO, go_tx).unwrap(); + expose(GO); + println!("LISTENING {}", cluster.local_addr()); + println!("JOINED 2"); + loop { + match go_rx.recv().unwrap() { + 1 => { + send_to(w1, Job("die".into())).unwrap(); + println!("KILLED w1"); + } + 2 => { + assert!(leave(POOL, w2)); + println!("LEFT w2"); + } + 3 => { + let w3 = worker(); + assert!(join(POOL, w3)); + println!("JOINED w3"); + } + n => panic!("unknown command {n}"), + } + } + }); +} + +fn remote_count(group: &str) -> usize { + members_all(group) + .iter() + .filter(|m| matches!(m, GroupMember::Remote(_))) + .count() +} + +/// Cooperative poll until `pred`; panics (with the last view) on timeout. +fn wait_view(what: &str, group: &str, pred: impl Fn(&[GroupMember]) -> bool) { + let deadline = Instant::now() + Duration::from_secs(5); + loop { + let v = members_all(group); + if pred(&v) { + return; + } + assert!( + Instant::now() < deadline, + "timed out waiting for {what}; view = {v:?}" + ); + smarm::sleep(Duration::from_millis(5)); + } +} + +fn role_observer() { + let origin_addr = std::env::var("SMARM_ORIGIN_ADDR").expect("SMARM_ORIGIN_ADDR"); + smarm::run(move || { + let _cluster = start(cfg("observer", vec![("origin".into(), origin_addr)])).expect("binds"); + let ev = subscribe().unwrap(); + loop { + match ev.rx.recv().unwrap() { + NodeEvent::NodeUp(i) if i.name == "origin" => break, + _ => {} + } + } + let go = |n: u8| remote::send(RemoteName::new("origin", GO), n).unwrap(); + + // Sync: both pre-existing members arrive with no join on this side. + wait_view("sync of 2 remote members", POOL, |v| { + v.len() == 2 && v.iter().all(|m| matches!(m, GroupMember::Remote(_))) + }); + let synced = members_all(POOL); + assert!(synced.iter().all(|m| match m { + GroupMember::Remote(p) => p.node() == "origin", + GroupMember::Local(_) => false, + })); + println!("SEES 2"); + + // Origin-side death: the origin's reaper announces the leave. + go(1); + wait_view("death evicted on observer", POOL, |v| v.len() == 1); + println!("SEES 1 after death"); + + // Incremental Join. + go(3); + wait_view("incremental join", POOL, |v| v.len() == 2); + println!("SEES 2 after join"); + + // Voluntary Leave. + go(2); + wait_view("incremental leave", POOL, |v| v.len() == 1); + println!("SEES 1 after leave"); + + // Mixed group: our own member sits beside the remote one in + // `members_all`; `members` stays local-only. + let me = worker(); + assert!(join(POOL, me)); + wait_view("mixed local+remote", POOL, |v| { + v.len() == 2 && v.contains(&GroupMember::Local(me.erase())) + }); + assert_eq!( + members(POOL), + vec![me.erase()], + "local API never shows remotes" + ); + assert_eq!(remote_count(POOL), 1); + println!("MIXED ok"); + + // dispatch_any: the store's first entry is the origin's w3 (it was + // announced before we joined), so the pick is remote and the job + // crosses the wire — the origin's worker prints it. + let picked = pick_any(POOL).expect("pool has members"); + assert!( + matches!(picked, GroupMember::Remote(_)), + "first entry is remote: {picked:?}" + ); + let reached = dispatch_any::(POOL, Job("from-observer".into())).unwrap(); + assert_eq!(reached, picked); + println!("DISPATCHED remote"); + + println!("PARK"); + // Parent SIGKILLs the origin now: NodeDown must sweep its member, + // ours must survive. + wait_view("node_down sweep", POOL, |v| { + v == [GroupMember::Local(me.erase())] + }); + assert_eq!(members(POOL), vec![me.erase()]); + println!("SWEPT"); + + // Now the only member is ours: a local pick, a local send. + let reached = dispatch_any::(POOL, Job("local".into())).unwrap(); + assert_eq!(reached, GroupMember::Local(me.erase())); + // And an empty group hands the message back. + match dispatch_any::("nobody", Job("lost".into())) { + Err(DispatchAnyError::NoMember(Job(s))) => assert_eq!(s, "lost"), + other => panic!("expected NoMember, got {other:?}"), + } + println!("DISPATCHED local"); + loop { + smarm::sleep(Duration::from_secs(3600)); + } + }); +} + +/// The Phase 5 gate: sync, join, leave, death, node_down — all observed from +/// the peer, none of them a group operation on the peer — plus dispatch_any +/// reaching a remote member and a local one. +#[test] +fn groups_span_two_nodes() { + maybe_child(ROLES); + let mut origin = spawn_node("origin", &[]); + let addr = origin.wait_listening(); + origin.wait_line("JOINED 2", |l| l == "JOINED 2"); + let mut observer = spawn_node("observer", &[("SMARM_ORIGIN_ADDR", &addr)]); + observer.wait_line("SEES 2", |l| l == "SEES 2"); + origin.wait_line("KILLED w1", |l| l == "KILLED w1"); + observer.wait_line("SEES 1 after death", |l| l == "SEES 1 after death"); + origin.wait_line("JOINED w3", |l| l == "JOINED w3"); + observer.wait_line("SEES 2 after join", |l| l == "SEES 2 after join"); + origin.wait_line("LEFT w2", |l| l == "LEFT w2"); + observer.wait_line("SEES 1 after leave", |l| l == "SEES 1 after leave"); + observer.wait_line("MIXED ok", |l| l == "MIXED ok"); + observer.wait_line("DISPATCHED remote", |l| l == "DISPATCHED remote"); + origin.wait_line("JOB from-observer", |l| l == "JOB from-observer"); + observer.wait_line("PARK", |l| l == "PARK"); + origin.kill(); + observer.wait_line("SWEPT", |l| l == "SWEPT"); + // Order between the root's line and the worker's is scheduling; wait + // for the later one to be certain both happened. + observer.wait_line("DISPATCHED local", |l| l == "DISPATCHED local"); + observer.wait_line("JOB local", |l| l == "JOB local"); +} diff --git a/tests/cluster_pid_send.rs b/tests/cluster_pid_send.rs new file mode 100644 index 0000000..9c1a074 --- /dev/null +++ b/tests/cluster_pid_send.rs @@ -0,0 +1,356 @@ +//! RFC 010 c10 — pid targeting + auto-serialization. The Phase 3 gate: +//! cross-node call/reply with no ceremony, under the subprocess harness. +//! +//! Local suite (`run()`, no network): serialize/deserialize shapes, +//! self-collapse, the outside-runtime contract, the local send-site +//! incarnation check with a probe proving **no frame is emitted**. +//! +//! Cross-process: two nodes. The *server* exposes a `Name`; the +//! *client* sends a `Req` carrying its own `Pid` (auto-serialized to +//! a `RemotePid` on the wire); the server replies via `send_to_remote` +//! straight back to that pid — no name at the client end, no ceremony. A +//! third-node roundtrip: the client's pid travels client→server→relay→ +//! server→client, and still delivers. +#![cfg(feature = "cluster")] + +mod common; + +use common::{maybe_child, spawn_node}; +use smarm::cluster::envelope::{encode_payload, Frame, NodeMeta}; +use smarm::cluster::expose::{expose, type_hash}; +use smarm::cluster::membership::{subscribe, NodeEvent}; +use smarm::cluster::remote::{self, send_to_remote, RemoteName, RemotePid, ToRemoteError}; +use smarm::cluster::{start, Config, StaticSeeds, Timing}; +use smarm::pg::Incarnation; +use smarm::{channel, install, register, run, Addressable, Name, Pid}; +use std::time::Duration; + +// ---- message types (std-only payloads; the crate's serde is derive-less, +// so wire types are hand-rolled with serde's tuple/seq API via `serde::ser` +// impls below — the same thing a user's derive would generate) ------------ + +/// A request carrying a reply-to. Serialize/Deserialize are written by hand +/// here for exactly one reason: this crate deliberately does not pull in +/// serde-derive. Field 1 is the auto-serializing pid. +#[derive(Debug, PartialEq)] +struct Req { + text: String, + reply_to: RemotePid, +} + +#[derive(Debug, PartialEq)] +struct Reply(String); + +struct Replier; +impl Addressable for Replier { + type Msg = Reply; +} + +impl serde::Serialize for Req { + fn serialize(&self, s: S) -> Result { + use serde::ser::SerializeTuple; + let mut t = s.serialize_tuple(2)?; + t.serialize_element(&self.text)?; + t.serialize_element(&self.reply_to)?; + t.end() + } +} +impl<'de> serde::Deserialize<'de> for Req { + fn deserialize>(d: D) -> Result { + let (text, reply_to) = <(String, RemotePid)>::deserialize(d)?; + Ok(Req { text, reply_to }) + } +} +impl serde::Serialize for Reply { + fn serialize(&self, s: S) -> Result { + self.0.serialize(s) + } +} +impl<'de> serde::Deserialize<'de> for Reply { + fn deserialize>(d: D) -> Result { + String::deserialize(d).map(Reply) + } +} + +// ================= local suite ========================================= + +/// A local `Pid` serializes as a `RemotePid` stamped with this node's +/// identity; deserializing it back on the same node collapses to the same +/// local pid (`local()` is `Some`, `Pid` round-trips). +#[test] +fn local_pid_serializes_and_collapses_on_self() { + maybe_child(ROLES); + run(|| { + // The local identity is set by cluster::start; the local suite sets + // it directly. + remote::set_local_identity("me", Incarnation::new(7)); + let (tx, _rx) = channel::(); + let me: Pid = install::(tx); + + let bytes = encode_payload(&me).unwrap(); + let rp: RemotePid = smarm::cluster::envelope::decode_payload(&bytes).unwrap(); + assert_eq!(rp.node(), "me"); + assert_eq!(rp.incarnation(), Incarnation::new(7)); + assert_eq!( + rp.local(), + Some(me), + "self-node pid collapses to the local pid" + ); + + // Deserializing straight into Pid works for a self-node pid... + let back: Pid = smarm::cluster::envelope::decode_payload(&bytes).unwrap(); + assert_eq!(back, me); + + // ...and FAILS for a foreign one (collapse is literal: node == self). + let foreign = RemotePid::::from_parts("elsewhere", Incarnation::new(1), 3, 1); + let fbytes = encode_payload(&foreign).unwrap(); + assert!(smarm::cluster::envelope::decode_payload::>(&fbytes).is_err()); + assert_eq!(foreign.local(), None); + }); +} + +/// `send_to_remote` short-circuits locally for a self-node pid — the +/// zero-copy-equivalent collapse: the message object itself lands in the +/// local channel, no encode, no frame. +#[test] +fn send_to_remote_collapses_locally_for_self() { + maybe_child(ROLES); + run(|| { + remote::set_local_identity("me", Incarnation::new(7)); + let (tx, rx) = channel::(); + let me: Pid = install::(tx); + let rp = RemotePid::from_local(me).expect("identity set"); + // Probe the outbound path: nothing must be handed to any connection. + let (probe_tx, probe_rx) = channel::(); + remote::bind_outbound_probe("me", Incarnation::new(7), probe_tx); + + send_to_remote(rp, Reply("hi".into())).unwrap(); + assert_eq!(rx.recv().unwrap(), Reply("hi".into())); + assert!( + matches!(probe_rx.try_recv(), Ok(None)), + "no frame for a local collapse" + ); + }); +} + +/// RFC v2 §3: a `RemotePid` whose incarnation is not the current one for its +/// node fails at the local send site with `DeadIncarnation`, and NO frame +/// is emitted — asserted on a probe sender bound as that node's outbound. +#[test] +fn stale_incarnation_rejected_locally_no_frame() { + maybe_child(ROLES); + run(|| { + remote::set_local_identity("me", Incarnation::new(7)); + let (probe_tx, probe_rx) = channel::(); + remote::bind_outbound_probe("peer", Incarnation::new(5), probe_tx); + + let stale = RemotePid::::from_parts("peer", Incarnation::new(4), 9, 1); + match send_to_remote(stale, Reply("late".into())) { + Err(ToRemoteError::DeadIncarnation(Reply(s))) => assert_eq!(s, "late"), + other => panic!("expected DeadIncarnation, got {other:?}"), + } + assert!( + matches!(probe_rx.try_recv(), Ok(None)), + "stale pid must emit no frame" + ); + + // The current incarnation goes through: a Send frame with the pid's + // (index, generation) and Reply's hash lands on the probe. + let live = RemotePid::::from_parts("peer", Incarnation::new(5), 9, 1); + send_to_remote(live, Reply("now".into())).unwrap(); + match probe_rx.recv().unwrap() { + Frame::Send { + index, + generation, + type_hash: h, + payload, + } => { + assert_eq!((index, generation), (9, 1)); + assert_eq!(h, type_hash::()); + let r: Reply = smarm::cluster::envelope::decode_payload(&payload).unwrap(); + assert_eq!(r, Reply("now".into())); + } + f => panic!("expected Send, got {f:?}"), + } + + // Unknown node: NotConnected, no frame anywhere. + let nowhere = RemotePid::::from_parts("nowhere", Incarnation::new(1), 1, 1); + assert!(matches!( + send_to_remote(nowhere, Reply("x".into())), + Err(ToRemoteError::NotConnected(_)) + )); + }); +} + +// ================= cross-process gate ================================== + +const ROLES: &[(&str, fn())] = &[ + ("server", role_server), + ("client", role_client), + ("relay", role_relay), +]; + +const ECHO: Name = Name::new("c10.echo"); +const RELAY: Name = Name::new("c10.relay"); + +fn cfg(name: &str, seeds: Vec<(String, String)>) -> Config { + Config { + node_name: name.to_string(), + meta: NodeMeta { + role: "c10".into(), + region: "local".into(), + }, + listen_addr: std::env::var("SMARM_LISTEN_ADDR").unwrap_or_else(|_| "127.0.0.1:0".into()), + strategy: Box::new(StaticSeeds::new(seeds)), + timing: Timing::default(), + } +} + +fn wait_up(events: &smarm::cluster::membership::MembershipEvents, who: &str) { + loop { + match events.rx.recv() { + Ok(NodeEvent::NodeUp(i)) if i.name == who => return, + Ok(_) => continue, + Err(_) => panic!("manager gone"), + } + } +} + +/// Server: exposes ECHO; each Req is answered by `send_to_remote` to its +/// reply_to — the server never learns a name for the client. If the Req text +/// starts with "via-relay:", it forwards the whole Req (reply_to and all) to +/// the relay node instead, which sends it back here; the second arrival is +/// answered normally. That is the pid's third-node roundtrip. +fn role_server() { + let relay_addr = std::env::var("SMARM_RELAY_ADDR").ok(); + smarm::run(move || { + let seeds = relay_addr + .map(|a| vec![("relay".to_string(), a)]) + .unwrap_or_default(); + let cluster = start(cfg("server", seeds)).expect("binds"); + println!("LISTENING {}", cluster.local_addr()); + let (tx, rx) = channel::(); + register(ECHO, tx).unwrap(); + expose(ECHO); + println!("READY"); + loop { + let req = rx.recv().unwrap(); + if let Some(rest) = req.text.strip_prefix("via-relay:") { + let fwd = Req { + text: format!("relayed:{rest}"), + reply_to: req.reply_to, + }; + remote::send(RemoteName::new("relay", RELAY), fwd).unwrap(); + println!("FORWARDED"); + continue; + } + println!("REQ {}", req.text); + send_to_remote(req.reply_to, Reply(format!("echo:{}", req.text))).unwrap(); + } + }); +} + +/// Relay: exposes RELAY; bounces every Req straight back to the server's +/// ECHO, untouched. The client's pid inside it now crosses relay→server. +fn role_relay() { + let server_addr = std::env::var("SMARM_SERVER_ADDR").expect("SMARM_SERVER_ADDR"); + smarm::run(move || { + let cluster = start(cfg("relay", vec![("server".into(), server_addr)])).expect("binds"); + println!("LISTENING {}", cluster.local_addr()); + let (tx, rx) = channel::(); + register(RELAY, tx).unwrap(); + expose(RELAY); + let ev = subscribe().unwrap(); + wait_up(&ev, "server"); + println!("READY"); + loop { + let req = rx.recv().unwrap(); + println!("RELAYING {}", req.text); + remote::send(RemoteName::new("server", ECHO), req).unwrap(); + } + }); +} + +/// Client: connects to server, installs a Reply inbox on its own pid, +/// declares it accepts `Reply` (`expose_type` — the RFC's one kept piece of +/// ceremony: nothing is remotely deliverable by default), sends a Req with +/// `reply_to = my pid` (auto-serialized), awaits the reply. +fn role_client() { + let server_addr = std::env::var("SMARM_SERVER_ADDR").expect("SMARM_SERVER_ADDR"); + let via_relay = std::env::var("SMARM_VIA_RELAY").is_ok(); + smarm::run(move || { + let _cluster = start(cfg("client", vec![("server".into(), server_addr)])).expect("binds"); + let ev = subscribe().unwrap(); + wait_up(&ev, "server"); + println!("MEMBER-UP server"); + + let (tx, rx) = channel::(); + let me: Pid = install::(tx); + // The one deliberate line: a pid-targeted inbound is deliverable only + // for types this node has said it accepts (RFC §4, the safety). + smarm::cluster::expose::expose_type::(); + let text = if via_relay { "via-relay:ping" } else { "ping" }; + remote::send( + RemoteName::new("server", ECHO), + Req { + text: text.into(), + reply_to: RemotePid::from_local(me).expect("identity set"), + }, + ) + .unwrap(); + println!("SENT"); + let Reply(s) = rx.recv().unwrap(); + println!("REPLY {s}"); + loop { + smarm::sleep(Duration::from_secs(3600)); + } + }); +} + +/// The gate: cross-node call/reply with no ceremony. +#[test] +fn cross_node_call_reply_no_ceremony() { + maybe_child(ROLES); + let mut server = spawn_node("server", &[]); + let saddr = server.wait_listening(); + server.wait_line("READY", |l| l == "READY"); + let mut client = spawn_node("client", &[("SMARM_SERVER_ADDR", &saddr)]); + client.wait_line("SENT", |l| l == "SENT"); + server.wait_line("REQ ping", |l| l == "REQ ping"); + client.wait_line("REPLY echo:ping", |l| l == "REPLY echo:ping"); +} + +/// The client's pid, round-tripped through a third node, still delivers. +#[test] +fn pid_roundtrips_through_third_node() { + maybe_child(ROLES); + // Relay needs the server address; server needs the relay address — + // pre-reserve the relay port (same accepted micro-window as cluster_mesh). + let relay_addr = { + let l = std::net::TcpListener::bind("127.0.0.1:0").unwrap(); + l.local_addr().unwrap().to_string() + }; + let mut server = spawn_node("server", &[("SMARM_RELAY_ADDR", &relay_addr)]); + let saddr = server.wait_listening(); + server.wait_line("READY", |l| l == "READY"); + let mut relay = spawn_node( + "relay", + &[ + ("SMARM_SERVER_ADDR", &saddr), + ("SMARM_LISTEN_ADDR", &relay_addr), + ], + ); + let _ = relay.wait_listening(); + relay.wait_line("READY", |l| l == "READY"); + let mut client = spawn_node( + "client", + &[("SMARM_SERVER_ADDR", &saddr), ("SMARM_VIA_RELAY", "1")], + ); + client.wait_line("SENT", |l| l == "SENT"); + server.wait_line("FORWARDED", |l| l == "FORWARDED"); + relay.wait_line("RELAYING", |l| l.starts_with("RELAYING")); + server.wait_line("REQ relayed:ping", |l| l == "REQ relayed:ping"); + client.wait_line("REPLY echo:relayed:ping", |l| { + l == "REPLY echo:relayed:ping" + }); +} diff --git a/tests/cluster_remote_send.rs b/tests/cluster_remote_send.rs new file mode 100644 index 0000000..fc93486 --- /dev/null +++ b/tests/cluster_remote_send.rs @@ -0,0 +1,221 @@ +//! RFC 010 c9 — remote `Name` sends: the outbound seam and the single +//! inbound name-resolution seam, cross-process. +//! +//! Two node processes each run the integrated `cluster::start`. The +//! *receiver* registers a `String` inbox under a name and exposes it (and +//! registers a second name it does NOT expose); the *sender* waits for +//! `node_up`, then sends. Facts cross as stdout lines: `LISTENING `, +//! `MEMBER-UP `, `GOT `, `SEND-RESULT `. +//! Roles park forever afterwards (retractable-state trap); the parent +//! SIGKILLs via `Drop`. +//! +//! What is asserted at each end (roadmap-binding): +//! - cross-node name-send delivers the payload; +//! - an unexposed name is unreachable — the receiver's inbox stays empty +//! even though the name IS registered locally; +//! - a wrong type hash is a decode failure at the receiver, never a +//! misroute — the `String` inbox does not see a `u64` delivered under a +//! made-up hash, nor a `u64` under `u64`'s hash; +//! - a send to a disconnected (never-connected) node fails locally with +//! `NotConnected`, and `Ok(())` means only "handed to the transport". +//! +//! Timing note for the "stays empty" assertions: they are proven by +//! ORDERING, not by waiting — the sender emits the negative-case frames +//! BEFORE the positive one on the same connection (in-order stream), so when +//! the receiver has seen the positive payload, the negatives have already +//! been processed and refused. No sleep-and-hope. +#![cfg(feature = "cluster")] + +mod common; + +use common::{maybe_child, spawn_node, Node}; +use smarm::cluster::envelope::NodeMeta; +use smarm::cluster::expose::expose; +use smarm::cluster::membership::{subscribe, NodeEvent}; +use smarm::cluster::remote::{send_remote_raw, RemoteName, RemoteSendError}; +use smarm::cluster::{start, Config, StaticSeeds, Timing}; +use smarm::{channel, register, Name}; +use std::time::Duration; + +const ROLES: &[(&str, fn())] = &[("receiver", role_receiver), ("sender", role_sender)]; + +const INBOX: Name = Name::new("c9.inbox"); +const HIDDEN: Name = Name::new("c9.hidden"); + +fn base_config(name: &str, seeds: Vec<(String, String)>) -> Config { + Config { + node_name: name.to_string(), + meta: NodeMeta { + role: "c9".to_string(), + region: "local".to_string(), + }, + listen_addr: "127.0.0.1:0".to_string(), + strategy: Box::new(StaticSeeds::new(seeds)), + timing: Timing::default(), + } +} + +/// Receiver: register + expose INBOX; register HIDDEN unexposed **in a +/// separate actor** (one actor holds one channel per message type — a +/// second `register` of the same `M` on one actor silently replaces the +/// first, closing it); print every payload that lands in either. +fn role_receiver() { + smarm::run(|| { + let cluster = start(base_config("recv", vec![])).expect("listener binds"); + println!("LISTENING {}", cluster.local_addr()); + + // HIDDEN's holder: its own actor, so its String channel does not + // displace INBOX's on the root actor. + let (hidden_ready_tx, hidden_ready_rx) = channel::<()>(); + smarm::spawn(move || { + let (hid_tx, hid_rx) = channel::(); + register(HIDDEN, hid_tx).unwrap(); + hidden_ready_tx.send(()).unwrap(); + loop { + match hid_rx.recv() { + Ok(s) => println!("GOT-HIDDEN {s}"), + Err(_) => break, + } + } + }); + hidden_ready_rx.recv().unwrap(); + + let (in_tx, in_rx) = channel::(); + register(INBOX, in_tx).unwrap(); + expose(INBOX); + println!("READY"); + loop { + match in_rx.recv() { + Ok(s) => println!("GOT {s}"), + Err(_) => break, + } + } + loop { + smarm::sleep(Duration::from_secs(3600)); + } + }); +} + +/// Sender: connect to recv, wait for node_up, then in this ORDER on the one +/// connection: hidden-name send, wrong-hash sends (two flavours), then the +/// positive send. Plus a send to a node that is not connected at all. +fn role_sender() { + let recv_addr = std::env::var("SMARM_RECV_ADDR").expect("SMARM_RECV_ADDR"); + smarm::run(move || { + let _cluster = start(base_config("send", vec![("recv".to_string(), recv_addr)])) + .expect("listener binds"); + let events = subscribe().expect("manager is up"); + loop { + match events.rx.recv() { + Ok(NodeEvent::NodeUp(info)) if info.name == "recv" => break, + Ok(_) => continue, + Err(_) => panic!("manager gone"), + } + } + println!("MEMBER-UP recv"); + + // Not connected: purely local knowledge, no frame leaves. + let ghost: RemoteName = RemoteName::new("nowhere", INBOX); + let r = smarm::cluster::remote::send(ghost, "lost".to_string()); + println!( + "SEND-RESULT not-connected {}", + match r { + Err(RemoteSendError::NotConnected(_)) => "NotConnected", + Ok(()) => "Ok", + Err(_) => "OtherErr", + } + ); + + // Unexposed name at the peer: the frame goes (local knowledge can't + // know the peer's exposed set) and the peer refuses it. + let hidden: RemoteName = RemoteName::new("recv", HIDDEN); + let r = smarm::cluster::remote::send(hidden, "should not land".to_string()); + println!( + "SEND-RESULT hidden {}", + if r.is_ok() { "Ok" } else { "Err" } + ); + + // Wrong hash, two flavours: (a) a u64 payload under a made-up hash + // (unknown type at the peer); (b) a u64 payload under u64's real + // hash against a String-typed name (decoder known, wrong channel). + // Both are raw sends — the typed API cannot express them, by design. + let bogus = 0xdead_beef_u64; + let r = send_remote_raw( + "recv", + "c9.inbox", + bogus, + &smarm::cluster::envelope::encode_payload(&7u64).unwrap(), + ); + println!( + "SEND-RESULT wrong-hash-unknown {}", + if r.is_ok() { "Ok" } else { "Err" } + ); + let r = send_remote_raw( + "recv", + "c9.inbox", + smarm::cluster::expose::type_hash::(), + &smarm::cluster::envelope::encode_payload(&7u64).unwrap(), + ); + println!( + "SEND-RESULT wrong-hash-known {}", + if r.is_ok() { "Ok" } else { "Err" } + ); + + // Positive: last on the stream, so its arrival proves the negatives + // were already processed. + let inbox: RemoteName = RemoteName::new("recv", INBOX); + let r = smarm::cluster::remote::send(inbox, "hello from send".to_string()); + println!( + "SEND-RESULT positive {}", + if r.is_ok() { "Ok" } else { "Err" } + ); + + loop { + smarm::sleep(Duration::from_secs(3600)); + } + }); +} + +fn wait_send_result(node: &mut Node, case: &str) -> String { + let prefix = format!("SEND-RESULT {case} "); + let line = node.wait_line(&prefix, |l| l.starts_with(&prefix)); + line[prefix.len()..].to_string() +} + +#[test] +fn remote_name_send_delivers_and_refusals_never_misroute() { + maybe_child(ROLES); + + let mut recv = spawn_node("receiver", &[]); + let addr = recv.wait_listening(); + recv.wait_line("READY", |l| l == "READY"); + + let mut send = spawn_node("sender", &[("SMARM_RECV_ADDR", &addr)]); + send.wait_line("MEMBER-UP recv", |l| l == "MEMBER-UP recv"); + + // Local-knowledge-only failure for an unknown node. + assert_eq!(wait_send_result(&mut send, "not-connected"), "NotConnected"); + // Every frame-bearing send is Ok — Ok means "handed to the transport", + // nothing about what the peer does with it (RFC §3, documented here). + assert_eq!(wait_send_result(&mut send, "hidden"), "Ok"); + assert_eq!(wait_send_result(&mut send, "wrong-hash-unknown"), "Ok"); + assert_eq!(wait_send_result(&mut send, "wrong-hash-known"), "Ok"); + assert_eq!(wait_send_result(&mut send, "positive"), "Ok"); + + // The positive payload lands... + recv.wait_line("GOT hello from send", |l| l == "GOT hello from send"); + // ...and, by stream ordering, every negative before it was refused: no + // GOT for the wrong-hash frames, no GOT-HIDDEN at all. The transcript + // up to this point is the proof. + let transcript = recv.transcript(); + let gots: Vec<&str> = transcript + .iter() + .map(|s| s.as_str()) + .filter(|l| l.starts_with("GOT")) + .collect(); + assert_eq!( + gots, + ["GOT hello from send"], + "exactly one delivery, the exposed one" + ); +} diff --git a/tests/cluster_transport.rs b/tests/cluster_transport.rs new file mode 100644 index 0000000..f9df0ae --- /dev/null +++ b/tests/cluster_transport.rs @@ -0,0 +1,272 @@ +//! RFC 010 c3 — transport conformance suite, run against both shipped impls +//! (TCP and in-memory loopback), plus impl-specific cases. +//! +//! Shared suite (roadmap): frame roundtrips through the framed codec, framing +//! across a split write, coalesced frames in one write, peer-close mid-frame +//! (must error, not EOF), clean close at a frame boundary (EOF as `Ok(None)`). +//! +//! The TCP impl parks the calling actor, so its runs live inside `smarm::run`; +//! loopback blocks the OS thread and runs as plain tests. +#![cfg(feature = "cluster")] + +use smarm::cluster::envelope::Frame; +use smarm::cluster::transport::loopback::LoopbackTransport; +use smarm::cluster::transport::tcp::TcpTransport; +use smarm::cluster::transport::{Conn, FramedConn, RecvError, Transport}; + +// --------------------------------------------------------------------------- +// Helpers +// --------------------------------------------------------------------------- + +/// Listener + dial + accept against one transport, both conns returned. +/// Relies on dial not requiring a concurrent accept (TCP backlog / loopback +/// queue), so a single thread or actor can hold both ends. +fn pair(t: &dyn Transport, addr: &str) -> (Box, Box) { + let mut l = t.listen(addr).unwrap(); + let a = t.dial(&l.local_addr()).unwrap(); + let b = l.accept().unwrap(); + (a, b) +} + +fn frames() -> Vec { + vec![ + Frame::Heartbeat, + Frame::Send { + index: 42, + generation: 3, + type_hash: 0x1234_5678_9ABC_DEF0, + payload: vec![1, 2, 3, 4, 5], + }, + Frame::SendNamed { + name: "the_counter".into(), + type_hash: 0xFFFF_0000_FFFF_0000, + payload: vec![], + }, + Frame::Demonitor { monitor_id: 77 }, + ] +} + +fn encode(f: &Frame) -> Vec { + let mut out = Vec::new(); + f.encode(&mut out).unwrap(); + out +} + +// --------------------------------------------------------------------------- +// Shared conformance suite — generic over an established pair +// --------------------------------------------------------------------------- + +fn suite_roundtrip(a: Box, b: Box) { + let mut fa = FramedConn::new(a); + let mut fb = FramedConn::new(b); + // a -> b, then b -> a: both directions carry every frame shape. + for f in frames() { + fa.send(&f).unwrap(); + assert_eq!(fb.recv().unwrap().unwrap(), f); + } + for f in frames() { + fb.send(&f).unwrap(); + assert_eq!(fa.recv().unwrap().unwrap(), f); + } +} + +fn suite_split_write(mut a: Box, b: Box) { + let f = Frame::Send { + index: 7, + generation: 1, + type_hash: 0xAB, + payload: vec![9; 64], + }; + let bytes = encode(&f); + // Split inside the length prefix, then inside the body: the reader must + // reassemble regardless of where the boundary falls. + a.write_all(&bytes[..2]).unwrap(); + a.write_all(&bytes[2..10]).unwrap(); + a.write_all(&bytes[10..]).unwrap(); + let mut fb = FramedConn::new(b); + assert_eq!(fb.recv().unwrap().unwrap(), f); +} + +fn suite_coalesced(mut a: Box, b: Box) { + let f1 = Frame::Heartbeat; + let f2 = Frame::Demonitor { monitor_id: 5 }; + let mut bytes = encode(&f1); + bytes.extend_from_slice(&encode(&f2)); + a.write_all(&bytes).unwrap(); + let mut fb = FramedConn::new(b); + assert_eq!(fb.recv().unwrap().unwrap(), f1); + assert_eq!(fb.recv().unwrap().unwrap(), f2); +} + +fn suite_close_mid_frame(mut a: Box, b: Box) { + let bytes = encode(&Frame::Send { + index: 1, + generation: 1, + type_hash: 1, + payload: vec![0; 128], + }); + a.write_all(&bytes[..bytes.len() / 2]).unwrap(); + a.close(); + let mut fb = FramedConn::new(b); + match fb.recv() { + Err(RecvError::TruncatedByPeer) => {} + other => panic!("expected TruncatedByPeer, got {other:?}"), + } +} + +fn suite_clean_close(mut a: Box, b: Box) { + let f = Frame::Heartbeat; + a.write_all(&encode(&f)).unwrap(); + a.close(); + let mut fb = FramedConn::new(b); + // The buffered frame is still delivered, then EOF at the boundary. + assert_eq!(fb.recv().unwrap().unwrap(), f); + assert!(fb.recv().unwrap().is_none()); +} + +fn run_suite(t: &dyn Transport, addr: &str) { + let (a, b) = pair(t, addr); + suite_roundtrip(a, b); + let (a, b) = pair(t, addr); + suite_split_write(a, b); + let (a, b) = pair(t, addr); + suite_coalesced(a, b); + let (a, b) = pair(t, addr); + suite_close_mid_frame(a, b); + let (a, b) = pair(t, addr); + suite_clean_close(a, b); +} + +// --------------------------------------------------------------------------- +// Loopback — plain tests, no runtime +// --------------------------------------------------------------------------- + +#[test] +fn loopback_conformance() { + // Fresh transport per pair() call is fine, but one instance must also + // support sequential re-listen on distinct addresses. + let t = LoopbackTransport::default(); + run_suite(&t, "alpha"); +} + +#[test] +fn loopback_dial_unknown_addr_refused() { + let t = LoopbackTransport::default(); + let err = t.dial("nobody-home").unwrap_err(); + assert_eq!(err.kind(), std::io::ErrorKind::ConnectionRefused); +} + +#[test] +fn loopback_addr_in_use() { + let t = LoopbackTransport::default(); + let _l = t.listen("alpha").unwrap(); + let err = t.listen("alpha").unwrap_err(); + assert_eq!(err.kind(), std::io::ErrorKind::AddrInUse); +} + +#[test] +fn loopback_listener_drop_frees_addr_and_refuses_dial() { + let t = LoopbackTransport::default(); + let l = t.listen("alpha").unwrap(); + drop(l); + let err = t.dial("alpha").unwrap_err(); + assert_eq!(err.kind(), std::io::ErrorKind::ConnectionRefused); + // Address is reusable after the listener is gone. + let _l2 = t.listen("alpha").unwrap(); +} + +#[test] +fn loopback_write_after_peer_close_broken_pipe() { + let t = LoopbackTransport::default(); + let (mut a, mut b) = pair(&t, "alpha"); + b.close(); + let err = a.write_all(&[1, 2, 3]).unwrap_err(); + assert_eq!(err.kind(), std::io::ErrorKind::BrokenPipe); +} + +#[test] +fn loopback_cross_thread_blocking_read() { + // Reader blocks on an empty pipe until the writer thread delivers. + let t = LoopbackTransport::default(); + let (a, b) = pair(&t, "alpha"); + let mut fb = FramedConn::new(b); + let writer = std::thread::spawn(move || { + let mut a = a; + std::thread::sleep(std::time::Duration::from_millis(30)); + a.write_all(&encode(&Frame::Heartbeat)).unwrap(); + }); + assert_eq!(fb.recv().unwrap().unwrap(), Frame::Heartbeat); + writer.join().unwrap(); +} + +// --------------------------------------------------------------------------- +// TCP — inside the runtime (read/write park the calling actor) +// --------------------------------------------------------------------------- + +#[test] +fn tcp_conformance() { + smarm::run(|| { + run_suite(&TcpTransport, "127.0.0.1:0"); + }); +} + +#[test] +fn tcp_dial_refused() { + smarm::run(|| { + // Bind to an OS-assigned port, learn it, close the listener, dial it. + let addr = { + let l = TcpTransport.listen("127.0.0.1:0").unwrap(); + l.local_addr() + }; + let err = TcpTransport.dial(&addr).unwrap_err(); + assert_eq!(err.kind(), std::io::ErrorKind::ConnectionRefused); + }); +} + +#[test] +fn tcp_bad_addr_rejected_without_resolution() { + // Addresses are opaque pre-resolved strings; the c9 seam resolves names. + // A hostname is therefore invalid input here, not something to resolve. + let err = TcpTransport.dial("localhost:1234").unwrap_err(); + assert_eq!(err.kind(), std::io::ErrorKind::InvalidInput); +} + +#[test] +fn tcp_local_addr_reports_real_port() { + let l = TcpTransport.listen("127.0.0.1:0").unwrap(); + let addr = l.local_addr(); + let port: u16 = addr.rsplit(':').next().unwrap().parse().unwrap(); + assert_ne!(port, 0); +} + +#[test] +fn tcp_big_frame_across_socket_buffers() { + // A payload far beyond socket buffer sizes forces genuine fragmentation + // and write backpressure: writer and reader must run concurrently. + smarm::run(|| { + let (tx, rx) = smarm::channel::(); + let payload = vec![0xA5u8; 4 * 1024 * 1024]; + let f = Frame::Send { + index: 9, + generation: 2, + type_hash: 0xC0FFEE, + payload, + }; + let mut l = TcpTransport.listen("127.0.0.1:0").unwrap(); + let addr = l.local_addr(); + let fw = f.clone(); + let writer = smarm::spawn(move || { + let mut fa = FramedConn::new(TcpTransport.dial(&addr).unwrap()); + fa.send(&fw).unwrap(); + }); + let reader = smarm::spawn(move || { + let mut fb = FramedConn::new(l.accept().unwrap()); + let got = fb.recv().unwrap().unwrap(); + tx.send(got).unwrap(); + }); + let got = rx.recv().unwrap(); + assert_eq!(got, f); + writer.join().unwrap(); + reader.join().unwrap(); + }); +} diff --git a/tests/cluster_two_node.rs b/tests/cluster_two_node.rs new file mode 100644 index 0000000..712b370 --- /dev/null +++ b/tests/cluster_two_node.rs @@ -0,0 +1,108 @@ +//! RFC 010 c4 — two-node harness smoke tests. +//! +//! Roadmap: "spawn two, handshake-less connect, both exit clean." The +//! listener node binds port 0 and announces its concrete address; the +//! dialer connects raw (no Hello — c5 doesn't exist yet), pushes one +//! Heartbeat through the real framed codec, and closes. Assertions are on +//! protocol-visible lines only. Flake budget: see tests/common/mod.rs. +#![cfg(feature = "cluster")] + +mod common; + +use common::{maybe_child, spawn_node}; +use smarm::cluster::envelope::Frame; +use smarm::cluster::transport::tcp::TcpTransport; +use smarm::cluster::transport::{FramedConn, Transport}; + +const ROLES: &[(&str, fn())] = &[ + ("listener", role_listener), + ("dialer", role_dialer), + ("hang", role_hang), + ("fail", role_fail), +]; + +fn role_listener() { + smarm::run(|| { + let mut l = TcpTransport.listen("127.0.0.1:0").unwrap(); + println!("LISTENING {}", l.local_addr()); + let mut fc = FramedConn::new(l.accept().unwrap()); + match fc.recv() { + Ok(Some(Frame::Heartbeat)) => println!("RECV heartbeat"), + other => { + println!("RECV unexpected: {other:?}"); + std::process::exit(3); + } + } + match fc.recv() { + Ok(None) => println!("PEER-CLOSED clean"), + other => { + println!("PEER-CLOSED unexpected: {other:?}"); + std::process::exit(3); + } + } + }); + println!("EXIT ok"); +} + +fn role_dialer() { + let addr = std::env::var("SMARM_PEER_ADDR").expect("SMARM_PEER_ADDR not set"); + smarm::run(move || { + let mut fc = FramedConn::new(TcpTransport.dial(&addr).unwrap()); + fc.send(&Frame::Heartbeat).unwrap(); + fc.close(); + println!("SENT heartbeat"); + }); + println!("EXIT ok"); +} + +fn role_hang() { + println!("HANGING"); + loop { + std::thread::sleep(std::time::Duration::from_secs(3600)); + } +} + +fn role_fail() { + std::process::exit(7); +} + +/// The roadmap smoke test: two real processes, raw transport connect, one +/// frame across, clean close observed on both sides, both exit 0. +#[test] +fn two_nodes_connect_and_exit_clean() { + maybe_child(ROLES); + let mut listener = spawn_node("listener", &[]); + let addr = listener.wait_listening(); + let mut dialer = spawn_node("dialer", &[("SMARM_PEER_ADDR", &addr)]); + dialer.wait_line("SENT heartbeat", |l| l == "SENT heartbeat"); + listener.wait_line("RECV heartbeat", |l| l == "RECV heartbeat"); + listener.wait_line("clean peer close", |l| l == "PEER-CLOSED clean"); + dialer.wait_exit_ok(); + listener.wait_exit_ok(); +} + +/// Reap guarantee: dropping a Node kills a hung child — no orphan survives +/// a panicking test. +#[test] +fn drop_reaps_hung_node() { + maybe_child(ROLES); + let mut node = spawn_node("hang", &[]); + node.wait_line("HANGING", |l| l == "HANGING"); + let pid = node.pid().expect("live child has a pid") as libc::pid_t; + drop(node); + // After Drop's kill+wait the pid is fully reaped: signalling it fails + // with ESRCH (pid-reuse in this instant is not a realistic race). + let rc = unsafe { libc::kill(pid, 0) }; + assert_eq!(rc, -1, "process still signallable after Drop"); + let errno = std::io::Error::last_os_error().raw_os_error(); + assert_eq!(errno, Some(libc::ESRCH), "expected ESRCH, got {errno:?}"); +} + +/// Nonzero child exits surface as statuses, not hangs or panics. +#[test] +fn nonzero_exit_is_reported() { + maybe_child(ROLES); + let mut node = spawn_node("fail", &[]); + let status = node.wait_exit(); + assert_eq!(status.code(), Some(7)); +} diff --git a/tests/common/mod.rs b/tests/common/mod.rs new file mode 100644 index 0000000..36ae39f --- /dev/null +++ b/tests/common/mod.rs @@ -0,0 +1,250 @@ +//! RFC 010 c4 — subprocess multi-node test harness. +//! +//! The runtime is a process singleton, so two real nodes means two +//! processes. This harness re-execs the *current test binary* as node +//! processes (precedent: tests/stack_diag.rs), tails their output live, +//! waits on protocol-visible lines, and reaps reliably no matter how the +//! test dies. +//! +//! Usage, per test file: +//! +//! - Declare roles as plain `fn()`s. A role prints protocol-visible facts +//! as single lines (Rust's piped stdout is line-buffered, so `println!` +//! is enough) and exits. +//! - **Every** `#[test]` in the file starts with +//! [`maybe_child`]`(ROLES)` — in the child re-exec, whichever test +//! libtest runs first performs the role and exits before the rest of the +//! suite runs (children are spawned with `--test-threads=1 --quiet`). +//! - The parent side spawns nodes with [`spawn_node`], waits on lines with +//! [`Node::wait_line`], and on exits with [`Node::wait_exit`]. +//! +//! Port assignment: children bind port 0 and *report* the concrete address +//! (e.g. `LISTENING 127.0.0.1:41733`) rather than the parent pre-picking a +//! port — no bind/steal race by construction. +//! +//! Reaping: [`Node`]'s `Drop` SIGKILLs and `wait(2)`s the child, so a +//! panicking test (including a `wait_line` timeout) leaves no orphan and +//! no zombie. Tail threads exit on pipe EOF. +//! +//! Flake budget (explicit, per roadmap): every wait is bounded by +//! [`WAIT`] (10 s) against a typical cost of well under 1 s; the smoke +//! suite ran 10/10 clean at authoring time. Treat >1 failure in 100 runs +//! as a harness or runtime regression, not weather. On timeout the panic +//! message carries the node's full transcript so far. + +#![allow(dead_code)] // Reusable surface: later phases use more of it than any one file. + +use std::env; +use std::io::{BufRead, BufReader}; +use std::process::{Child, Command, ExitStatus, Stdio}; +use std::sync::mpsc::{Receiver, RecvTimeoutError}; +use std::time::{Duration, Instant}; + +/// Env var selecting the child role in a re-exec. +const ROLE_ENV: &str = "SMARM_TWO_NODE_ROLE"; + +/// Upper bound for every wait in the harness. See the flake budget above. +pub const WAIT: Duration = Duration::from_secs(10); + +/// In the child re-exec: run the matching role and exit. In the parent (no +/// role env set): return immediately. Call this first in every `#[test]` of +/// any file using the harness, passing the file's full role table. +pub fn maybe_child(roles: &[(&str, fn())]) { + let role = match env::var(ROLE_ENV) { + Ok(r) => r, + Err(_) => return, + }; + for (name, f) in roles { + if *name == role { + f(); + std::process::exit(0); + } + } + eprintln!("two_node harness: unknown role {role:?}"); + std::process::exit(2); +} + +/// One spawned node process with live-tailed output. +pub struct Node { + /// Role name, for panic messages. + pub role: String, + child: Option, + stdout_rx: Receiver, + stderr_rx: Receiver, + /// Every line consumed from stdout/stderr so far, for failure dumps. + transcript: Vec, +} + +fn tail(stream: impl std::io::Read + Send + 'static, prefix: &'static str) -> Receiver { + let (tx, rx) = std::sync::mpsc::channel(); + std::thread::spawn(move || { + for line in BufReader::new(stream).lines() { + let line = match line { + Ok(l) => l, + Err(_) => break, + }; + // Receiver gone (Node dropped): stop tailing. + if tx.send(format!("{prefix}{line}")).is_err() { + break; + } + } + }); + rx +} + +/// Re-exec the current test binary as `role`, with any extra env vars. +pub fn spawn_node(role: &str, extra_env: &[(&str, &str)]) -> Node { + let exe = env::current_exe().expect("current_exe"); + let mut cmd = Command::new(exe); + cmd.env(ROLE_ENV, role) + // --test-threads=1: exactly one test fn starts, hits maybe_child, + // and becomes the role. --nocapture: libtest must not swallow the + // role's println! lines — the parent tails them live. + .args(["--test-threads=1", "--quiet", "--nocapture"]) + .stdout(Stdio::piped()) + .stderr(Stdio::piped()); + for (k, v) in extra_env { + cmd.env(k, v); + } + let mut child = cmd.spawn().expect("failed to spawn node process"); + let stdout_rx = tail(child.stdout.take().expect("piped stdout"), ""); + let stderr_rx = tail(child.stderr.take().expect("piped stderr"), "[stderr] "); + Node { + role: role.to_string(), + child: Some(child), + stdout_rx, + stderr_rx, + transcript: Vec::new(), + } +} + +impl Node { + fn drain_stderr(&mut self) { + while let Ok(l) = self.stderr_rx.try_recv() { + self.transcript.push(l); + } + } + + fn dump(&self) -> String { + if self.transcript.is_empty() { + "".to_string() + } else { + self.transcript.join("\n") + } + } + + /// Every stdout/stderr line seen so far, in arrival order. For + /// ordering-proof assertions ("by the time X arrived, Y had not"). + #[allow(dead_code)] + pub fn transcript(&self) -> &[String] { + &self.transcript + } + + /// Wait until a stdout line satisfies `pred`; return it. Panics with the + /// full transcript after [`WAIT`]. `what` names the expectation in the + /// panic message. + pub fn wait_line(&mut self, what: &str, pred: impl Fn(&str) -> bool) -> String { + let deadline = Instant::now() + WAIT; + loop { + self.drain_stderr(); + let left = deadline.saturating_duration_since(Instant::now()); + match self.stdout_rx.recv_timeout(left) { + Ok(line) => { + self.transcript.push(line.clone()); + if pred(&line) { + return line; + } + } + Err(RecvTimeoutError::Timeout) => { + // Pull in whatever stderr arrived since the last drain, + // so a role's eprintln! diagnostics survive into the dump. + self.drain_stderr(); + panic!( + "node {:?}: timed out waiting for {what} after {WAIT:?}; transcript:\n{}", + self.role, + self.dump() + ); + } + Err(RecvTimeoutError::Disconnected) => { + self.drain_stderr(); + panic!( + "node {:?}: output closed while waiting for {what}; transcript:\n{}", + self.role, + self.dump() + ); + } + } + } + } + + /// Shorthand: wait for a `LISTENING ` announcement, return the addr. + pub fn wait_listening(&mut self) -> String { + let line = self.wait_line("LISTENING announcement", |l| l.starts_with("LISTENING ")); + line["LISTENING ".len()..].to_string() + } + + /// Wait for the process to exit; panics with the transcript on timeout. + pub fn wait_exit(&mut self) -> ExitStatus { + let deadline = Instant::now() + WAIT; + loop { + let polled = match self.child.as_mut() { + Some(c) => c.try_wait(), + None => panic!("node {:?}: already reaped", self.role), + }; + match polled { + Ok(Some(status)) => { + // Drain remaining output into the transcript for dumps. + self.drain_stderr(); + while let Ok(l) = self.stdout_rx.try_recv() { + self.transcript.push(l); + } + self.child = None; + return status; + } + Ok(None) => { + if Instant::now() >= deadline { + self.drain_stderr(); + self.kill(); + panic!( + "node {:?}: did not exit within {WAIT:?}; transcript:\n{}", + self.role, + self.dump() + ); + } + std::thread::sleep(Duration::from_millis(10)); + } + Err(e) => panic!("node {:?}: try_wait failed: {e}", self.role), + } + } + } + + /// Wait for exit and require success, dumping the transcript otherwise. + pub fn wait_exit_ok(&mut self) { + let status = self.wait_exit(); + assert!( + status.success(), + "node {:?}: exited with {status}; transcript:\n{}", + self.role, + self.dump() + ); + } + + /// The child's OS pid, if not yet reaped. + pub fn pid(&self) -> Option { + self.child.as_ref().map(Child::id) + } + + /// SIGKILL + reap now (idempotent). + pub fn kill(&mut self) { + if let Some(mut child) = self.child.take() { + let _ = child.kill(); + let _ = child.wait(); + } + } +} + +impl Drop for Node { + fn drop(&mut self) { + self.kill(); + } +} diff --git a/tests/pg.rs b/tests/pg.rs index 180ae10..3b06bbe 100644 --- a/tests/pg.rs +++ b/tests/pg.rs @@ -1,5 +1,6 @@ //! Process-group tests that run under the scheduler: `join` installs a real -//! monitor on a live actor, and a real death drives eviction on next contact. +//! monitor on a live actor, and a real death drives eviction (the reaper +//! actor sweeps it; the read path hides it in the meantime). //! (Pure structural invariants live in the `pg` unit tests.) use smarm::{channel, members, pick, run, spawn}; @@ -35,19 +36,15 @@ fn a_dead_actor_vanishes_from_every_group_it_joined() { assert_eq!(members("g1"), vec![pid]); assert_eq!(members("g2"), vec![pid]); - // Release and reap the actor. finalize_actor queues the Down to our - // monitors before unparking joiners, so by the time join() returns the - // Down is already waiting in the membership channel. + // Release the actor. finalize_actor queues the Down to the reaper and + // marks the slot dead before unparking joiners, so by the time join() + // returns every read hides the pid whether or not the reaper has run. tx.send(()).unwrap(); h.join().unwrap(); - // Drain-on-contact: touching g1 detects the death and sweeps the pid - // out of every group (g2 included), not just g1. - assert!(members("g1").is_empty(), "evicted from the touched group"); - assert!( - members("g2").is_empty(), - "and swept from the untouched group" - ); + // Gone from every group it joined, not just one. + assert!(members("g1").is_empty(), "gone from g1"); + assert!(members("g2").is_empty(), "and from g2"); assert_eq!(pick("g1"), None); }); } @@ -86,11 +83,7 @@ fn live_members_survive_a_peers_death() { tx_a.send(()).unwrap(); a.join().unwrap(); - assert_eq!( - members("svc"), - vec![b.pid()], - "only the dead peer is reaped" - ); + assert_eq!(members("svc"), vec![b.pid()], "only the dead peer is gone"); assert_eq!(pick("svc"), Some(b.pid())); tx_b.send(()).unwrap(); @@ -123,18 +116,18 @@ fn leave_drops_a_membership_without_affecting_others() { } #[test] -fn joining_an_already_dead_pid_is_evicted_on_next_contact() { +fn joining_an_already_dead_pid_never_shows_in_a_read() { run(|| { let h = spawn(|| {}); let pid = h.pid(); h.join().unwrap(); // actor is finalized before we join it to anything - // monitor() on a gone pid queues a NoProc Down immediately, so the - // membership is reaped the next time the group is touched. + // join() on a gone pid queues a NoProc Down to the reaper immediately; + // reads never show it either way (slot-liveness backstop). join("late", pid); assert!( members("late").is_empty(), - "dead-at-join member is reaped on read" + "dead-at-join member never reads as live" ); assert_eq!(pick("late"), None); }); diff --git a/tests/registry.rs b/tests/registry.rs index d1e23f3..2ca251b 100644 --- a/tests/registry.rs +++ b/tests/registry.rs @@ -258,3 +258,46 @@ fn send_dyn_to_dead_pid_is_dead() { assert!(matches!(send_dyn::(p, 1u64), Err(SendError::Dead(_)))); }); } + +// --- one channel per message type per actor ----------------------------------- + +/// Registering a second name of the same message type on one actor, with a +/// *fresh* channel, would silently replace and close the first — so it +/// panics (found in RFC 010 c9). The sanctioned shapes stay quiet: bind both +/// names to a clone of one sender, or use two actors. +#[test] +#[should_panic(expected = "already publishes a live channel")] +fn second_live_channel_of_same_type_on_one_actor_panics() { + run(|| { + let (tx1, _rx1) = channel::(); + let (tx2, _rx2) = channel::(); + register(Name::::new("dup-a"), tx1).unwrap(); + register(Name::::new("dup-b"), tx2).unwrap(); // panics + }); +} + +#[test] +fn two_names_on_one_cloned_sender_is_fine() { + run(|| { + let (tx, rx) = channel::(); + register(Name::::new("twin-a"), tx.clone()).unwrap(); + register(Name::::new("twin-b"), tx).unwrap(); + send(Name::::new("twin-a"), 1).unwrap(); + send(Name::::new("twin-b"), 2).unwrap(); + assert_eq!(rx.recv().unwrap(), 1); + assert_eq!(rx.recv().unwrap(), 2); + }); +} + +#[test] +fn replacing_a_channel_whose_receiver_is_gone_is_fine() { + run(|| { + let (tx1, rx1) = channel::(); + register(Name::::new("reborn"), tx1).unwrap(); + drop(rx1); // old inbox gone: replacement is the honest thing to do + let (tx2, rx2) = channel::(); + register(Name::::new("reborn-2"), tx2).unwrap(); + send(Name::::new("reborn-2"), 9).unwrap(); + assert_eq!(rx2.recv().unwrap(), 9); + }); +}