diff --git a/src/cluster.rs b/src/cluster.rs index a56cc6c..635c611 100644 --- a/src/cluster.rs +++ b/src/cluster.rs @@ -17,6 +17,7 @@ pub mod expose; pub mod handshake; pub mod manager; pub mod membership; +pub mod pg; pub mod remote; pub mod transport; @@ -38,10 +39,15 @@ pub use conn::{spawn_established, ConnHandle}; pub use connect::{dial, spawn_acceptor, AcceptorHandle}; pub use connector::{spawn_connector, ConnectorHandle}; pub use discovery::{Discovery, StaticSeeds, Strategy}; +pub use envelope::RemoteDownReason; pub use expose::{expose, expose_type, type_hash, DeliverError}; pub use manager::{Manager, MANAGER}; pub use membership::{subscribe, view, MembershipEvents, NodeEvent, NodeInfo}; -pub use remote::{NotConnected, RemoteName, RemoteSendError}; +pub use pg::{dispatch_any, members_all, pick_any, DispatchAnyError, GroupMember, PgMsg, PG_NAME}; +pub use remote::{ + demonitor_remote, monitor_remote, send_to_remote, NotConnected, RemoteDown, RemoteMonitor, + RemoteName, RemotePid, RemoteSendError, ToRemoteError, +}; /// c6d — the derived build hash for [`handshake::LocalNode::build_hash`]: /// two builds may mesh only when this matches, and it is a pure function of @@ -80,6 +86,41 @@ const fn fold_u32(mut h: u64, v: u32) -> u64 { h } +/// The control-plane timing knobs, all with today's fixed values as +/// defaults ([`Timing::default`]). One struct threaded explicitly to the +/// acceptor, the dial path, every connection actor and the connector — no +/// ambient state, so a test can run a fast mesh without touching globals. +/// Every node in a mesh should agree on `heartbeat_interval` < +/// `liveness_timeout`; nothing enforces it. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct Timing { + /// Idle-connection heartbeat pace. Default [`conn::HEARTBEAT_INTERVAL`]. + pub heartbeat_interval: Duration, + /// Inbound silence that tears a connection down. Default + /// [`conn::LIVENESS_TIMEOUT`]. + pub liveness_timeout: Duration, + /// Per-frame handshake deadline on the accept/dial path. Default + /// [`connect::HANDSHAKE_TIMEOUT`]. + pub handshake_timeout: Duration, + /// Connector redial delay after the first failure. Default + /// [`connector::INITIAL_BACKOFF`]. + pub initial_backoff: Duration, + /// Connector redial delay cap. Default [`connector::MAX_BACKOFF`]. + pub max_backoff: Duration, +} + +impl Default for Timing { + fn default() -> Self { + Timing { + heartbeat_interval: conn::HEARTBEAT_INTERVAL, + liveness_timeout: conn::LIVENESS_TIMEOUT, + handshake_timeout: connect::HANDSHAKE_TIMEOUT, + initial_backoff: connector::INITIAL_BACKOFF, + max_backoff: connector::MAX_BACKOFF, + } + } +} + /// How to run this node: its identity and how it finds peers. pub struct Config { /// This node's claimed name — the mesh-wide identity peers dial by and @@ -92,6 +133,9 @@ pub struct Config { pub listen_addr: String, /// The peer-discovery strategy — [`StaticSeeds`] until richer ones land. pub strategy: Box, + /// Heartbeat / liveness / handshake / backoff knobs; [`Timing::default`] + /// is the shipping configuration. + pub timing: Timing, } /// A running cluster node: the supervised [`Manager`], the acceptor over the @@ -155,9 +199,18 @@ pub fn start(config: Config) -> io::Result { build_hash: BUILD_HASH, meta: config.meta, }; + // The wire identity serialized pids are stamped with (c10). + remote::set_local_identity(&local.node_name, local.incarnation); + // The pg actor (Phase 5): subscribes membership, owns the "pg" name. + pg::attach_cluster(); let listener = TcpTransport.listen(&config.listen_addr)?; - let acceptor = spawn_acceptor(listener, local.clone()); - let connector = spawn_connector(Box::new(TcpTransport), local.clone(), config.strategy); + let acceptor = spawn_acceptor(listener, local.clone(), config.timing); + let connector = spawn_connector( + Box::new(TcpTransport), + local.clone(), + config.strategy, + config.timing, + ); Ok(Cluster { _sup: sup, acceptor, diff --git a/src/cluster/conn.rs b/src/cluster/conn.rs index a9abad4..76a33ba 100644 --- a/src/cluster/conn.rs +++ b/src/cluster/conn.rs @@ -27,21 +27,52 @@ //! drained onto the wire in the same loop — and inbound *interpretation*: //! `SendNamed` goes to the one resolution seam, //! [`remote::deliver_named`](crate::cluster::remote::deliver_named). -//! Frames the connection actor has no business with yet (`Send`, `Monitor`, -//! …: c10/c11) are consumed for liveness and otherwise ignored. The outbound +//! `Send` goes to the pid seam (c10). The outbound //! sender is a separate channel from `cmd_tx` on purpose: closing it is not //! a stop signal — lifetime authority stays with the [`ConnHandle`] (D9). +//! +//! c12 adds the monitor plane, and it lives *here* on purpose. Two tables, +//! both owned by this actor and dying with the connection: +//! +//! - **outstanding** — monitors *this* node holds on actors at the peer: +//! `monitor_id → (target, Sender)`. Fed by +//! [`MonCmd`](crate::cluster::remote::MonCmd) from `monitor_remote`; the +//! actor records the id and *then* emits the `Monitor` frame, so a `Down` +//! frame can never race an entry that isn't there yet. An inbound `Down` +//! removes the entry and delivers. +//! - **watched** — monitors the *peer* holds on actors here: `monitor_id → +//! local Monitor`. An inbound `Monitor` is admitted only for a pid that +//! was exposed or crossed the wire (`is_watchable`, D12): a corpse answers +//! with its recorded terminal reason (RFC §6), an unwatchable or unknown +//! pid with `NoProc` — indistinguishable from dead, so nothing leaks. A +//! live watchable pid gets a local monitor whose `rx` is one more arm of +//! the select; its `Down` goes back as a frame. +//! +//! Because both tables are actor state, connection loss (c13) needs no +//! second bookkeeping owner: this actor's exit is the one place that knows +//! every monitor the link was carrying. `Monitors::teardown` runs on every +//! exit path and answers each outstanding monitor with `Disconnected` — +//! the roadmap's "partition vs. death" contrast: an actor that dies sends +//! its true reason over the link, a link that dies says only that. + +use std::collections::HashMap; use std::time::{Duration, Instant}; use crate::channel::{channel, try_select_timeout, Receiver, Selectable, Sender}; -use crate::cluster::envelope::Frame; +use crate::cluster::envelope::{Frame, RemoteDownReason}; use crate::cluster::handshake::Peer; use crate::cluster::manager::{Call, Registered, Reply, MANAGER}; -use crate::cluster::remote::deliver_named; +use crate::cluster::remote::{ + deliver_named, deliver_to_pid, InboundVerdict, MonCmd, RemoteDown, RemotePid, +}; use crate::cluster::transport::FramedConn; +use crate::cluster::Timing; use crate::gen_server; -use crate::pid::Pid; +use crate::monitor::{ + demonitor, is_watchable, monitor, terminal_reason, DownReason, Monitor, MonitorId, +}; +use crate::pid::{Erased, Pid}; use crate::scheduler::spawn; /// Commands to a running connection actor. @@ -57,11 +88,11 @@ enum Cmd { /// to establish it. pub struct ConnHandle { cmd_tx: Sender, - /// The connection's dedicated outbound inbox. The manager moves this - /// into the outbound table on `Register` (see - /// [`take_outbound`](ConnHandle::take_outbound)); a `Duplicate` verdict - /// drops it with the handle. - out_tx: Option>, + /// The connection's dedicated outbound inboxes — frames and monitor + /// commands. The manager moves them into the outbound table on + /// `Register` (see [`take_outbound`](ConnHandle::take_outbound)); a + /// `Duplicate` verdict drops them with the handle. + out_tx: Option<(Sender, Sender)>, } impl std::fmt::Debug for ConnHandle { @@ -78,9 +109,9 @@ impl ConnHandle { let _ = self.cmd_tx.send(Cmd::Shutdown); } - /// Manager-only: take the outbound sender to bind into the outbound + /// Manager-only: take the outbound senders to bind into the outbound /// table. Once, at registration. - pub(crate) fn take_outbound(&mut self) -> Option> { + pub(crate) fn take_outbound(&mut self) -> Option<(Sender, Sender)> { self.out_tx.take() } } @@ -96,11 +127,16 @@ pub struct RegisterRefused; /// actor's [`ConnHandle`]; the caller gets only the [`Pid`], because /// connection lifetime belongs to the table and not to the establishing /// actor. A refusal has already stopped the actor and closed the socket. -pub fn spawn_established(framed: FramedConn, peer: Peer) -> Result { +pub fn spawn_established( + framed: FramedConn, + peer: Peer, + timing: Timing, +) -> Result { let (cmd_tx, cmd_rx) = channel(); let (out_tx, out_rx) = channel(); + let (mon_tx, mon_rx) = channel(); let reg_peer = peer.clone(); - let pid = spawn(move || run(framed, peer, cmd_rx, out_rx)).pid(); + let pid = spawn(move || run(framed, peer, timing, cmd_rx, out_rx, mon_rx)).pid(); match gen_server::call( MANAGER, Call::Register { @@ -108,7 +144,7 @@ pub fn spawn_established(framed: FramedConn, peer: Peer) -> Result Result, out_rx: Receiver) { +fn run( + mut framed: FramedConn, + _peer: Peer, + timing: Timing, + cmd_rx: Receiver, + out_rx: Receiver, + mon_rx: Receiver, +) { + let mut mons = Monitors::default(); match framed.readable_arm() { - Some(arm) => run_live(&mut framed, arm, &cmd_rx, &out_rx), + Some(arm) => run_live( + &mut framed, + arm, + timing, + &cmd_rx, + &out_rx, + &mon_rx, + &mut mons, + ), None => run_inert(&cmd_rx), } framed.close(); + mons.teardown(&mon_rx); +} + +/// The monitor plane's two tables (module docs). Owned by the actor. +#[derive(Default)] +struct Monitors { + /// Monitors this node holds on peer actors: id → (target, delivery). + outstanding: HashMap, Sender)>, + /// Monitors the peer holds on local actors: id → the local monitor. + watched: HashMap, +} + +impl Monitors { + /// The connection is gone, whatever the exit path (liveness expiry, + /// EOF, wire failure, commanded stop): release the peer's local + /// monitors, and answer every one of ours with `Disconnected` — nothing + /// more can be known about those actors. Commands still sitting in the + /// inbox are folded in first (a `Monitor` handed to us but never + /// processed gets its notice too; a `Demonitor` still cancels), so the + /// only registration that can miss this is one that lands after the + /// drain and before the inbox drops — the reader side backstops that + /// (`RemoteMonitor`). Entries leave the table as they are answered, and + /// this runs once per actor, so no monitor sees two notices. + fn teardown(&mut self, mon_rx: &Receiver) { + for (_, m) in self.watched.drain() { + let _ = demonitor(&m); + } + while let Ok(Some(cmd)) = mon_rx.try_recv() { + match cmd { + MonCmd::Monitor { id, target, tx } => { + self.outstanding.insert(id, (target, tx)); + } + MonCmd::Demonitor { id } => { + self.outstanding.remove(&id); + } + } + } + for (_, (pid, tx)) in self.outstanding.drain() { + let _ = tx.send(RemoteDown { + pid, + reason: RemoteDownReason::Disconnected, + }); + } + } + + /// Admit a peer's `Monitor` for local `(index, generation)`. Returns + /// the reason to answer with at once, or `None` if a live monitor was + /// installed. Corpse → recorded terminal reason (RFC §6, and only + /// watchable deaths are recorded); live watchable → monitor; anything + /// else → `NoProc`. The check-then-monitor race (dies in between) is + /// closed on the read side: a `NoProc` from a monitor we installed on a + /// live pid is upgraded through `terminal_reason` in `sweep_watched`. + fn admit(&mut self, id: MonitorId, index: u32, generation: u32) -> Option { + let pid = Pid::new(index, generation); + if let Some(reason) = terminal_reason(pid) { + return Some(reason); + } + if !is_watchable(pid) { + return Some(DownReason::NoProc); + } + let m = monitor(pid); + self.watched.insert(id, m); + None + } + + fn cancel(&mut self, id: MonitorId) { + if let Some(m) = self.watched.remove(&id) { + let _ = demonitor(&m); + } + } + + /// Collect every local `Down` that has arrived for a peer-held monitor. + fn sweep_watched(&mut self) -> Vec<(MonitorId, DownReason)> { + let mut fired = Vec::new(); + for (id, m) in self.watched.iter() { + if let Ok(Some(down)) = m.rx.try_recv() { + let reason = match down.reason { + DownReason::NoProc => terminal_reason(m.target).unwrap_or(DownReason::NoProc), + r => r, + }; + fired.push((*id, reason)); + } + } + for (id, _) in &fired { + self.watched.remove(id); + } + fired + } + + /// The peer reports a monitored actor down: deliver locally. + fn down(&mut self, id: MonitorId, reason: RemoteDownReason) { + if let Some((pid, tx)) = self.outstanding.remove(&id) { + let _ = tx.send(RemoteDown { pid, reason }); + } + } } /// The steady-state loop over an fd-backed connection: one @@ -148,15 +295,39 @@ fn run(mut framed: FramedConn, _peer: Peer, cmd_rx: Receiver, out_rx: Recei fn run_live( framed: &mut FramedConn, arm: crate::scheduler::FdArm, + timing: Timing, cmd_rx: &Receiver, out_rx: &Receiver, + mon_rx: &Receiver, + mons: &mut Monitors, ) { let mut next_hb = Instant::now(); - let mut live_until = Instant::now() + LIVENESS_TIMEOUT; - // The outbound sender lives in the manager's table and is dropped on - // unbind; after that this arm would wake forever, so it drops out of + let mut live_until = Instant::now() + timing.liveness_timeout; + // The outbound senders live in the manager's table and are dropped on + // unbind; after that these arms would wake forever, so they drop out of // the select (not a stop signal — see the module docs). let mut out_open = true; + let mut mon_open = true; + // Which wait each select arm stands for. Built in lockstep with the + // `Selectable` vector each iteration, so a wake is decoded by name and + // never by position. + enum Arm { + Cmd, + Fd, + Out, + Mon, + /// A peer-held local monitor (any of them: firing sweeps them all). + Watched, + } + fn push<'s>( + arms: &mut Vec<&'s dyn Selectable>, + what: &mut Vec, + s: &'s dyn Selectable, + a: Arm, + ) { + arms.push(s); + what.push(a); + } loop { let now = Instant::now(); if now >= live_until { @@ -166,33 +337,57 @@ fn run_live( if framed.send(&Frame::Heartbeat).is_err() { break; } - next_hb = now + HEARTBEAT_INTERVAL; + next_hb = now + timing.heartbeat_interval; } let wait = next_hb.min(live_until).saturating_duration_since(now); - // Arm indices: 0 cmd, 1 fd, 2 outbound (when open). - let mut arms: Vec<&dyn Selectable> = vec![cmd_rx, &arm]; + let mut arms: Vec<&dyn Selectable> = Vec::new(); + let mut what: Vec = Vec::new(); + push(&mut arms, &mut what, cmd_rx, Arm::Cmd); + push(&mut arms, &mut what, &arm, Arm::Fd); if out_open { - arms.push(out_rx); + push(&mut arms, &mut what, out_rx, Arm::Out); } - match try_select_timeout(&arms, wait) { - Ok(Some(0)) => { + if mon_open { + push(&mut arms, &mut what, mon_rx, Arm::Mon); + } + for m in mons.watched.values() { + push(&mut arms, &mut what, &m.rx, Arm::Watched); + } + match try_select_timeout(&arms, wait).map(|i| i.map(|i| &what[i])) { + Ok(Some(Arm::Cmd)) => { if should_stop(cmd_rx) { break; } } - Ok(Some(1)) => match pump_readable(framed) { + Ok(Some(Arm::Fd)) => match pump_readable(framed, mons) { Pump::Ended => break, Pump::Frames(n) => { if n > 0 { - live_until = Instant::now() + LIVENESS_TIMEOUT; + live_until = Instant::now() + timing.liveness_timeout; } } }, - Ok(Some(_)) => match pump_outbound(framed, out_rx) { - Outbound::Sent => {} + Ok(Some(Arm::Out)) => match pump_outbound(framed, out_rx) { + Outbound::Drained => {} Outbound::Closed => out_open = false, Outbound::WireFailed => break, }, + Ok(Some(Arm::Mon)) => match pump_moncmds(framed, mon_rx, mons) { + Outbound::Drained => {} + Outbound::Closed => mon_open = false, + Outbound::WireFailed => break, + }, + Ok(Some(Arm::Watched)) => { + for (id, reason) in mons.sweep_watched() { + let frame = Frame::Down { + monitor_id: id.0, + reason: reason.into(), + }; + if framed.send(&frame).is_err() { + return; + } + } + } // A deadline passed; the top of the loop acts on whichever. Ok(None) => {} // The fd arm failed to register — the connection is gone. @@ -201,9 +396,42 @@ fn run_live( } } -/// What one outbound wake yielded. +/// Drain the monitor-command inbox: record, then emit (module docs). +fn pump_moncmds( + framed: &mut FramedConn, + mon_rx: &Receiver, + mons: &mut Monitors, +) -> Outbound { + loop { + match mon_rx.try_recv() { + Ok(Some(MonCmd::Monitor { id, target, tx })) => { + let frame = Frame::Monitor { + monitor_id: id.0, + index: target.index(), + generation: target.generation(), + }; + mons.outstanding.insert(id, (target, tx)); + if framed.send(&frame).is_err() { + return Outbound::WireFailed; + } + } + Ok(Some(MonCmd::Demonitor { id })) => { + if mons.outstanding.remove(&id).is_some() + && framed.send(&Frame::Demonitor { monitor_id: id.0 }).is_err() + { + return Outbound::WireFailed; + } + } + Ok(None) => return Outbound::Drained, + Err(_) => return Outbound::Closed, + } + } +} + +/// What one outbound-side wake (frames or monitor commands) yielded. enum Outbound { - Sent, + /// Everything queued went onto the wire; the inbox is open and empty. + Drained, /// The manager unbound this connection's sender; nothing more will come. Closed, /// The socket refused a write: the connection is gone. @@ -219,7 +447,7 @@ fn pump_outbound(framed: &mut FramedConn, out_rx: &Receiver) -> Outbound return Outbound::WireFailed; } } - Ok(None) => return Outbound::Sent, + Ok(None) => return Outbound::Drained, Err(_) => return Outbound::Closed, } } @@ -261,16 +489,26 @@ enum Pump { Frames(usize), } +/// Surface an inbound verdict: one `smarm-trace` event, nothing else — it +/// is local knowledge (RFC §3). A no-op without the feature. +fn note_verdict(verdict: InboundVerdict) { + #[cfg(feature = "smarm-trace")] + crate::te!(crate::trace::Event::ClusterInbound(verdict.label())); + #[cfg(not(feature = "smarm-trace"))] + drop(verdict); +} + /// Consume one readable wake: exactly one socket read (which cannot block /// after a level-triggered readable indication), then drain every complete /// frame the buffer now holds. A blocking `recv` here would park the actor /// past its heartbeat and liveness deadlines whenever a frame arrives split. -/// Every consumed frame counts for liveness; `SendNamed` additionally goes -/// to the one inbound resolution seam. Its verdict is local knowledge only -/// — nothing goes back on the wire (RFC §3) — and is currently discarded -/// (a future trace hook is the place to surface it). Frames for later -/// chunks (`Send` c10, `Monitor`/`Down` c11) are consumed and ignored. -fn pump_readable(framed: &mut FramedConn) -> Pump { +/// Every consumed frame counts for liveness; `SendNamed` goes to the one +/// name-resolution seam and `Send` to the pid seam. Verdicts are local +/// knowledge only — nothing goes back on the wire (RFC §3) — and surface +/// as one `smarm-trace` `ClusterInbound` event each (zero cost off). +/// `Monitor`/`Demonitor`/`Down` go to the [`Monitors`] tables; a `Monitor` +/// that can be answered at once is answered inline. +fn pump_readable(framed: &mut FramedConn, mons: &mut Monitors) -> Pump { let eof = match framed.read_once() { Ok(n) => n == 0, Err(_) => return Pump::Ended, @@ -280,13 +518,43 @@ fn pump_readable(framed: &mut FramedConn) -> Pump { match framed.next_buffered() { Ok(Some(frame)) => { got += 1; - if let Frame::SendNamed { - name, - type_hash, - payload, - } = frame - { - let _verdict = deliver_named(&name, type_hash, &payload); + match frame { + Frame::SendNamed { + name, + type_hash, + payload, + } => { + note_verdict(deliver_named(&name, type_hash, &payload)); + } + Frame::Send { + index, + generation, + type_hash, + payload, + } => { + note_verdict(deliver_to_pid(index, generation, type_hash, &payload)); + } + Frame::Monitor { + monitor_id, + index, + generation, + } => { + let id = MonitorId(monitor_id); + if let Some(reason) = mons.admit(id, index, generation) { + let frame = Frame::Down { + monitor_id, + reason: reason.into(), + }; + if framed.send(&frame).is_err() { + return Pump::Ended; + } + } + } + Frame::Demonitor { monitor_id } => mons.cancel(MonitorId(monitor_id)), + Frame::Down { monitor_id, reason } => mons.down(MonitorId(monitor_id), reason), + // Heartbeat: liveness only. Handshake frames after + // establishment: ignored. + _ => {} } } Ok(None) => break, @@ -299,3 +567,64 @@ fn pump_readable(framed: &mut FramedConn) -> Pump { Pump::Frames(got) } } + +#[cfg(test)] +mod tests { + //! `Monitors::teardown` in isolation: the actor-side half of c13, pinned + //! separately because from the outside it is indistinguishable from the + //! read-side backstop in `RemoteMonitor` (both yield `Disconnected`). + use super::*; + use crate::pg::Incarnation; + + fn pid(index: u32) -> RemotePid { + RemotePid::from_parts("peer", Incarnation::new(1), index, 1) + } + + #[test] + fn teardown_answers_every_outstanding_and_unread_monitor_once() { + crate::run(|| { + let mut mons = Monitors::default(); + let (mon_tx, mon_rx) = channel::(); + + // Already registered. + let (tx1, rx1) = channel::(); + mons.outstanding.insert(MonitorId(1), (pid(1), tx1)); + // In the inbox, never processed. + let (tx2, rx2) = channel::(); + mon_tx + .send(MonCmd::Monitor { + id: MonitorId(2), + target: pid(2), + tx: tx2, + }) + .ok() + .unwrap(); + // Registered, then cancelled in the inbox: silence. + let (tx3, rx3) = channel::(); + mons.outstanding.insert(MonitorId(3), (pid(3), tx3)); + mon_tx + .send(MonCmd::Demonitor { id: MonitorId(3) }) + .ok() + .unwrap(); + + mons.teardown(&mon_rx); + + let d1 = rx1.recv().unwrap(); + assert_eq!( + (d1.pid, d1.reason), + (pid(1), RemoteDownReason::Disconnected) + ); + let d2 = rx2.recv().unwrap(); + assert_eq!( + (d2.pid, d2.reason), + (pid(2), RemoteDownReason::Disconnected) + ); + // Cancelled: no notice was sent (its sender is dropped, channel + // closed-empty), and nobody got a second one. + assert!(rx3.try_recv().is_err()); + assert!(rx1.try_recv().is_err()); + assert!(rx2.try_recv().is_err()); + assert!(mons.outstanding.is_empty()); + }); + } +} diff --git a/src/cluster/connect.rs b/src/cluster/connect.rs index 503b37c..cf3d1a1 100644 --- a/src/cluster/connect.rs +++ b/src/cluster/connect.rs @@ -27,16 +27,17 @@ use crate::channel::{channel, Receiver, Selectable, Sender}; use crate::cluster::conn::spawn_established; use crate::cluster::envelope::{Frame, RejectReason}; use crate::cluster::handshake::{ - HelloCtx, Initiator, InitiatorOutcome, Local, Peer, Responder, ResponderOutcome, + Initiator, InitiatorOutcome, Local, Peer, PeerStanding, Responder, ResponderOutcome, }; use crate::cluster::manager::{Call, Reply, MANAGER}; use crate::cluster::transport::{FramedConn, Listener, RecvError, SendError, Transport}; +use crate::cluster::Timing; use crate::gen_server; use crate::pid::Pid; use crate::scheduler::{self, spawn}; -/// How long either side waits for the peer's handshake frame before giving -/// up and closing. Enforced on the path via [`FramedConn::recv_deadline`], +/// Default for [`Timing::handshake_timeout`]: how long either side waits for +/// the peer's handshake frame before giving up and closing. Enforced on the path via [`FramedConn::recv_deadline`], /// so a peer that connects and goes silent cannot wedge the acceptor. pub const HANDSHAKE_TIMEOUT: Duration = Duration::from_secs(5); @@ -119,7 +120,7 @@ pub fn dial_handshake( /// Accept-side path step: read the first frame, judge it, answer or close. /// -/// `ctx_for` supplies the [`HelloCtx`] for the *offered* name — knowledge +/// `standing_of` supplies the [`PeerStanding`] of the *offered* name — knowledge /// only the frame reveals, which is why it is a callback and not a value /// (the integrated acceptor asks the manager; loopback tests fabricate). /// It is not called when the first frame is not a `Hello`. @@ -129,7 +130,7 @@ pub fn dial_handshake( pub fn accept_handshake( framed: &mut FramedConn, local: Local, - ctx_for: impl FnOnce(&str) -> HelloCtx, + standing_of: impl FnOnce(&str) -> PeerStanding, deadline: Instant, ) -> Result { let frame = match framed.recv_deadline(deadline) { @@ -143,11 +144,11 @@ pub fn accept_handshake( return Err(from_recv(e)); } }; - let ctx = match &frame { - Frame::Hello { node_name, .. } => ctx_for(node_name), - _ => HelloCtx::default(), + let standing = match &frame { + Frame::Hello { node_name, .. } => standing_of(node_name), + _ => PeerStanding::Free, }; - match Responder::new(local).on_frame(frame, ctx) { + match Responder::new(local).on_frame(frame, standing) { ResponderOutcome::Accepted { reply, peer } => { if let Err(e) = framed.send(&reply) { framed.close(); @@ -199,6 +200,21 @@ pub enum DialError { Duplicate, } +impl DialError { + /// A short static label per kind, for the `smarm-trace` `ClusterDial` + /// event; the payload (io error, names) is not carried. + pub fn label(&self) -> &'static str { + match self { + DialError::AlreadyDialing => "already_dialing", + DialError::ManagerUnavailable => "manager_unavailable", + DialError::Connect(_) => "connect", + DialError::Handshake(_) => "handshake", + DialError::PeerNameMismatch { .. } => "peer_name_mismatch", + DialError::Duplicate => "duplicate", + } + } +} + /// Dial `peer_name` at `addr` and run the handshake, keeping the manager's /// dial-intent set honest around it: the intent is registered *before* /// connecting (so a crossing inbound `Hello` sees it) and cleared the @@ -211,6 +227,7 @@ pub fn dial( addr: &str, peer_name: &str, local: &Local, + timing: Timing, ) -> Result { let me = scheduler::self_pid(); match gen_server::call( @@ -224,7 +241,7 @@ pub fn dial( Ok(Reply::DialBegan(false)) => return Err(DialError::AlreadyDialing), _ => return Err(DialError::ManagerUnavailable), } - let result = connect_and_shake(transport, addr, local); + let result = connect_and_shake(transport, addr, local, timing); // Cleared immediately on outcome — a stale intent during the established // window would corrupt later tie-breaks. Synchronous (a call): the // intent is provably gone before anything else happens. @@ -242,17 +259,18 @@ pub fn dial( got: peer.node_name, }); } - spawn_established(framed, peer).map_err(|_| DialError::Duplicate) + spawn_established(framed, peer, timing).map_err(|_| DialError::Duplicate) } fn connect_and_shake( transport: &dyn Transport, addr: &str, local: &Local, + timing: Timing, ) -> Result<(FramedConn, Peer), DialError> { let conn = transport.dial(addr).map_err(DialError::Connect)?; let mut framed = FramedConn::new(conn); - let deadline = Instant::now() + HANDSHAKE_TIMEOUT; + let deadline = Instant::now() + timing.handshake_timeout; let peer = dial_handshake(&mut framed, local, deadline).map_err(DialError::Handshake)?; Ok((framed, peer)) } @@ -286,14 +304,19 @@ impl AcceptorHandle { /// fd-backed ([`Listener::readable_arm`]); the loopback listener is not, /// and its acceptor exits immediately — loopback handshakes are driven /// synchronously through the path fns instead, per D8. -pub fn spawn_acceptor(listener: Box, local: Local) -> AcceptorHandle { +pub fn spawn_acceptor(listener: Box, local: Local, timing: Timing) -> AcceptorHandle { let addr = listener.local_addr(); let (cmd_tx, cmd_rx) = channel(); - spawn(move || accept_loop(listener, local, cmd_rx)); + spawn(move || accept_loop(listener, local, timing, cmd_rx)); AcceptorHandle { cmd_tx, addr } } -fn accept_loop(mut listener: Box, local: Local, cmd_rx: Receiver<()>) { +fn accept_loop( + mut listener: Box, + local: Local, + timing: Timing, + cmd_rx: Receiver<()>, +) { loop { let Some(arm) = listener.readable_arm() else { return; @@ -311,7 +334,7 @@ fn accept_loop(mut listener: Box, local: Local, cmd_rx: Receiver<( Ok(conn) => conn, Err(_) => return, // listener itself is broken }; - handle_inbound(FramedConn::new(conn), &local); + handle_inbound(FramedConn::new(conn), &local, timing); } Err(_) => return, // fd arm failed to register: listener is gone } @@ -319,27 +342,24 @@ fn accept_loop(mut listener: Box, local: Local, cmd_rx: Receiver<( } /// Run the accept-side handshake for one inbound connection, asking the -/// manager for the [`HelloCtx`], and hand the established connection to the +/// manager for the [`PeerStanding`], and hand the established connection to the /// manager. Every failure was already resolved on the path (reject sent / /// closed, or the registration refused and the actor stopped), so there is /// nothing for the acceptor to carry forward. -fn handle_inbound(mut framed: FramedConn, local: &Local) { - let deadline = Instant::now() + HANDSHAKE_TIMEOUT; - let ctx_for = |name: &str| match gen_server::call( +fn handle_inbound(mut framed: FramedConn, local: &Local, timing: Timing) { + let deadline = Instant::now() + timing.handshake_timeout; + let standing_of = |name: &str| match gen_server::call( MANAGER, - Call::HelloCtx { + Call::Standing { peer_name: name.to_string(), }, ) { - Ok(Reply::HelloCtx(ctx)) => ctx, + Ok(Reply::Standing(s)) => s, // Manager unreachable: nobody could register this connection anyway, // so claim the name taken and reject rather than accept an orphan. - _ => HelloCtx { - name_claimed: true, - dialing_this_peer: false, - }, + _ => PeerStanding::Claimed, }; - if let Ok(peer) = accept_handshake(&mut framed, local.clone(), ctx_for, deadline) { - let _ = spawn_established(framed, peer); + if let Ok(peer) = accept_handshake(&mut framed, local.clone(), standing_of, deadline) { + let _ = spawn_established(framed, peer, timing); } } diff --git a/src/cluster/connector.rs b/src/cluster/connector.rs index b4633ea..2d9eaa0 100644 --- a/src/cluster/connector.rs +++ b/src/cluster/connector.rs @@ -15,9 +15,14 @@ //! `node_down` resume immediately (a fresh sequence — the reconnect case is //! the one backoff exists to pace, but the *first* retry after a death //! should be prompt). A candidate bearing our own name is parked permanently -//! — that seed is us. Every other failure retries: in particular a -//! `NameTaken` reject can be our own ghost at the peer, not yet reaped by -//! its liveness timer, so it must not park. +//! — that seed is us; so is one whose address answers as a different name +//! (`PeerNameMismatch`: a misconfigured or stale seed — each retry would +//! only blip the peer's membership). Every other failure retries: in +//! particular a `NameTaken` reject can be our own ghost at the peer, not +//! yet reaped by its liveness timer, so it must not park. Each attempt's +//! outcome is one `smarm-trace` `ClusterDial` event. A [`Discovery::Withdrawn`] +//! drops its `(name, addr)` from the dial set — only that: a live +//! connection is membership's, and a re-announce re-adds it fresh. //! //! Dials run **inline in the loop** — the same deliberate serialization as //! the acceptor (c6b): each attempt is bounded by the connect + handshake @@ -25,21 +30,23 @@ //! unreachable seeds would stretch the loop's latency; revisit if a real //! deployment ever hits that shape. -use std::collections::{HashMap, HashSet}; +use std::collections::HashSet; use std::time::{Duration, Instant}; use crate::channel::{channel, select, select_timeout, Receiver, Selectable, Sender}; -use crate::cluster::connect::dial; +use crate::cluster::connect::{dial, DialError}; use crate::cluster::discovery::{Discovery, Strategy}; use crate::cluster::handshake::Local; use crate::cluster::membership::{subscribe, NodeEvent}; use crate::cluster::transport::Transport; -use crate::pg::NodeId; +use crate::cluster::Timing; use crate::scheduler::spawn; -/// First retry delay after a failed dial attempt. +/// Default for [`Timing::initial_backoff`]: first retry delay after a failed +/// dial attempt. pub const INITIAL_BACKOFF: Duration = Duration::from_millis(250); -/// Backoff ceiling: an unreachable seed is retried this often, forever. +/// Default for [`Timing::max_backoff`]: an unreachable seed is retried this +/// often, forever. pub const MAX_BACKOFF: Duration = Duration::from_secs(5); enum Cmd { @@ -60,15 +67,74 @@ impl ConnectorHandle { } } -/// One discovered `(name, addr)` and its dial state. +/// One discovered `(name, addr)` and our dial intent towards it. struct Candidate { name: String, addr: String, - /// This seed is the local node itself: never dialed. - parked: bool, - /// Delay to apply after the *next* failure. - backoff: Duration, - next_attempt: Instant, + state: State, +} + +/// The connector's *intent* for a candidate. Whether the peer is currently +/// up is a separate, name-keyed membership fact (`up` in [`run`]): a +/// candidate can arrive after its peer's `node_up` (the snapshot lands +/// before the strategy has said anything), so "up" cannot live on the +/// candidate alone — it is a filter over dialing, not a candidate state. +enum State { + /// Never dialed: this seed is the local node itself, or the address + /// answered as a *different* name than the one seeded + /// (`DialError::PeerNameMismatch` — a misconfigured or stale seed; + /// redialing would only blip the peer's membership forever). The way + /// back is the strategy's: `Withdrawn` then a fresh `Candidate`. + Parked, + /// Dial when due; on failure, back off. + Dialing { + /// Delay to apply after the *next* failure. + backoff: Duration, + next_attempt: Instant, + }, +} + +impl State { + fn fresh(timing: &Timing) -> Self { + State::Dialing { + backoff: timing.initial_backoff, + next_attempt: Instant::now(), + } + } +} + +impl Candidate { + /// The retry deadline, if this candidate is dialing at all. + fn due(&self) -> Option { + match self.state { + State::Parked => None, + State::Dialing { next_attempt, .. } => Some(next_attempt), + } + } + /// A dial attempt was made: schedule the retry, grow the backoff. + fn attempted(&mut self, timing: &Timing) { + if let State::Dialing { + backoff, + next_attempt, + } = &mut self.state + { + *next_attempt = Instant::now() + *backoff; + *backoff = (*backoff * 2).min(timing.max_backoff); + } + } + /// The peer came up: the next sequence (after a later `node_down`) + /// starts from the initial delay again. + fn peer_up(&mut self, timing: &Timing) { + if let State::Dialing { backoff, .. } = &mut self.state { + *backoff = timing.initial_backoff; + } + } + /// The peer went down: redial promptly, fresh sequence. + fn peer_down(&mut self, timing: &Timing) { + if matches!(self.state, State::Dialing { .. }) { + self.state = State::fresh(timing); + } + } } /// Spawn the connector actor. The strategy is spawned as its child; the @@ -78,9 +144,10 @@ pub fn spawn_connector( transport: Box, local: Local, strategy: Box, + timing: Timing, ) -> ConnectorHandle { let (cmd_tx, cmd_rx) = channel(); - spawn(move || run(transport, local, strategy, cmd_rx)); + spawn(move || run(transport, local, strategy, timing, cmd_rx)); ConnectorHandle { cmd_tx } } @@ -88,10 +155,11 @@ fn run( transport: Box, local: Local, strategy: Box, + timing: Timing, cmd_rx: Receiver, ) { // Membership is the connector's source of truth for "who is up" — the - // snapshot seeds `connected` before any candidate arrives. + // snapshot seeds `up` before any candidate arrives. let Some(events) = subscribe() else { return; // no manager, no cluster to connect }; @@ -99,8 +167,7 @@ fn run( spawn(move || strategy.run(disc_tx)); let mut cands: Vec = Vec::new(); - let mut connected: HashSet = HashSet::new(); - let mut names: HashMap = HashMap::new(); // NodeDown carries only the id + let mut up: HashSet = HashSet::new(); let mut strategy_done = false; loop { @@ -111,9 +178,9 @@ fn run( Drained::Open => {} } if !strategy_done { - strategy_done = drain_discoveries(&disc_rx, &local, &mut cands); + strategy_done = drain_discoveries(&disc_rx, &local, &timing, &mut cands); } - match drain_events(&events.rx, &mut connected, &mut names, &mut cands) { + match drain_events(&events.rx, &timing, &mut up, &mut cands) { Drained::Stop => return, // manager gone: the cluster is tearing down Drained::Open => {} } @@ -122,23 +189,26 @@ fn run( let now = Instant::now(); for c in cands .iter_mut() - .filter(|c| !c.parked && !connected.contains(&c.name) && c.next_attempt <= now) + .filter(|c| !up.contains(&c.name) && c.due().is_some_and(|d| d <= now)) { - // The outcome does not branch the bookkeeping: on success the - // manager's node_up is on its way and flips `connected` (backing - // off meanwhile keeps a racing re-attempt from spinning); every - // failure retries — see the module docs. - let _ = dial(&*transport, &c.addr, &c.name, &local); - c.next_attempt = Instant::now() + c.backoff; - c.backoff = (c.backoff * 2).min(MAX_BACKOFF); + // On success the manager's node_up is on its way and lands in + // `up` (backing off meanwhile keeps a racing re-attempt from + // spinning); every failure retries — see the module docs — + // except a peer-name mismatch, which parks the candidate. + let outcome = dial(&*transport, &c.addr, &c.name, &local, timing); + note_dial(&outcome); + match outcome { + Err(DialError::PeerNameMismatch { .. }) => c.state = State::Parked, + _ => c.attempted(&timing), + } } // Wait: until the earliest retry deadline among actionable // candidates, or indefinitely if none is pending. let deadline = cands .iter() - .filter(|c| !c.parked && !connected.contains(&c.name)) - .map(|c| c.next_attempt) + .filter(|c| !up.contains(&c.name)) + .filter_map(Candidate::due) .min(); let mut arms: Vec<&dyn Selectable> = vec![&cmd_rx, &events.rx]; if !strategy_done { @@ -170,23 +240,31 @@ fn drain_cmd(rx: &Receiver) -> Drained { } /// Pull every pending discovery into the candidate set (deduplicated by -/// `(name, addr)`; a candidate bearing the local name is parked). Returns -/// `true` once the strategy's channel closes — it has said all it will. -fn drain_discoveries(rx: &Receiver, local: &Local, cands: &mut Vec) -> bool { +/// `(name, addr)`; a candidate bearing the local name is parked; a +/// `Withdrawn` removes its pair from the dial set and nothing else — see +/// [`Discovery::Withdrawn`]). Returns `true` once the strategy's channel +/// closes — it has said all it will. +fn drain_discoveries( + rx: &Receiver, + local: &Local, + timing: &Timing, + cands: &mut Vec, +) -> bool { loop { match rx.try_recv() { + Ok(Some(Discovery::Withdrawn { name, addr })) => { + cands.retain(|c| !(c.name == name && c.addr == addr)); + } Ok(Some(Discovery::Candidate { name, addr })) => { if cands.iter().any(|c| c.name == name && c.addr == addr) { continue; } - let parked = name == local.node_name; - cands.push(Candidate { - name, - addr, - parked, - backoff: INITIAL_BACKOFF, - next_attempt: Instant::now(), - }); + let state = if name == local.node_name { + State::Parked + } else { + State::fresh(timing) + }; + cands.push(Candidate { name, addr, state }); } Ok(None) => return false, Err(_) => return true, // strategy done; its candidates live on here @@ -194,36 +272,45 @@ fn drain_discoveries(rx: &Receiver, local: &Local, cands: &mut Vec, - connected: &mut HashSet, - names: &mut HashMap, + timing: &Timing, + up: &mut HashSet, cands: &mut [Candidate], ) -> Drained { loop { match rx.try_recv() { Ok(Some(NodeEvent::NodeUp(info))) => { - names.insert(info.node, info.name.clone()); - for c in cands.iter_mut().filter(|c| c.name == info.name) { - c.backoff = INITIAL_BACKOFF; - } - connected.insert(info.name); + cands + .iter_mut() + .filter(|c| c.name == info.name) + .for_each(|c| c.peer_up(timing)); + up.insert(info.name); } - Ok(Some(NodeEvent::NodeDown { node })) => { - if let Some(name) = names.remove(&node) { - connected.remove(&name); - let now = Instant::now(); - for c in cands.iter_mut().filter(|c| c.name == name) { - c.backoff = INITIAL_BACKOFF; - c.next_attempt = now; - } - } + Ok(Some(NodeEvent::NodeDown(info))) => { + up.remove(&info.name); + cands + .iter_mut() + .filter(|c| c.name == info.name) + .for_each(|c| c.peer_down(timing)); } Ok(None) => return Drained::Open, Err(_) => return Drained::Stop, } } } + +/// Surface a dial outcome: one `smarm-trace` event, nothing else. The +/// connector's bookkeeping is decided by the caller. +fn note_dial(outcome: &Result) { + #[cfg(feature = "smarm-trace")] + crate::te!(crate::trace::Event::ClusterDial( + outcome.as_ref().map_or_else(DialError::label, |_| "ok") + )); + #[cfg(not(feature = "smarm-trace"))] + let _ = outcome; +} diff --git a/src/cluster/discovery.rs b/src/cluster/discovery.rs index b7c0e29..edd37a7 100644 --- a/src/cluster/discovery.rs +++ b/src/cluster/discovery.rs @@ -20,13 +20,22 @@ use crate::channel::Sender; /// A discovery event, as pushed by a [`Strategy`]. /// -/// Additive-only for now (candidates are announced, never withdrawn); -/// `#[non_exhaustive]` so expiry can land later without breaking strategies. +/// `Candidate` announces, `Withdrawn` retracts — the primitive pair. A +/// strategy that wants TTL semantics builds them on top (track its own +/// last-seen times, emit `Withdrawn` on expiry); the connector deliberately +/// has no clock of its own for candidates (D11: strategies never +/// re-announce, the connector owns retry). `#[non_exhaustive]` so more can +/// land without breaking strategies. #[derive(Debug, Clone, PartialEq, Eq)] #[non_exhaustive] pub enum Discovery { /// A peer worth dialing: its claimed node name and a dialable address. Candidate { name: String, addr: String }, + /// Stop dialing this `(name, addr)`. Dial-set only: a connection that + /// is already up is membership's business and is left alone; an + /// attempt in flight completes on its own; a later `Candidate` for the + /// same pair re-adds it with fresh backoff. Unknown pairs are ignored. + Withdrawn { name: String, addr: String }, } /// A source of peers to dial. Implementations are spawned as actors by the diff --git a/src/cluster/envelope.rs b/src/cluster/envelope.rs index 445374c..b7d749c 100644 --- a/src/cluster/envelope.rs +++ b/src/cluster/envelope.rs @@ -81,10 +81,45 @@ pub enum Frame { }, Down { monitor_id: u64, - reason: DownReason, + reason: RemoteDownReason, }, } +/// Why a remotely-monitored actor is reported down: either the target's own +/// terminal [`DownReason`] as its node recorded it, or the *link* to that +/// node was lost (or absent) — which says nothing about the actor itself. +/// +/// This is the cluster-side widening of `DownReason` (p5): `Disconnected` +/// is a fact about a connection, never about a local actor, so it lives +/// here rather than in the core enum — a local `Down` can never carry it, +/// and matches on `DownReason` stay exhaustive over actor outcomes only. +/// On the wire `Local(r)` uses `r`'s tag and `Disconnected` is tag 5, +/// bound since c11; no peer emits it today (a lost link is synthesized +/// locally), but the codec honours it both ways. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum RemoteDownReason { + /// The target itself terminated; the peer reported this reason. + Local(DownReason), + /// The link to the target's node was lost or was never up. + Disconnected, +} + +impl RemoteDownReason { + /// The actor's own reason, if this was not a link loss. + pub fn local(self) -> Option { + match self { + RemoteDownReason::Local(r) => Some(r), + RemoteDownReason::Disconnected => None, + } + } +} + +impl From for RemoteDownReason { + fn from(r: DownReason) -> Self { + RemoteDownReason::Local(r) + } +} + // Frame tags. 0 is deliberately unassigned so an all-zero buffer never parses. const TAG_HELLO: u8 = 1; const TAG_HELLO_ACK: u8 = 2; @@ -101,11 +136,12 @@ const REJ_HASH_MISMATCH: u8 = 1; const REJ_NAME_TAKEN: u8 = 2; const REJ_PROTO_VERSION: u8 = 3; -// DownReason tags. c11 adds `Disconnected = 5`; do not reuse tags. +// DownReason tags. Do not reuse tags. const DR_EXIT: u8 = 1; const DR_PANIC: u8 = 2; const DR_STOPPED: u8 = 3; const DR_NOPROC: u8 = 4; +const DR_DISCONNECTED: u8 = 5; /// Frame could not be encoded. The output buffer is left exactly as it was. #[derive(Debug, Clone, Copy, PartialEq, Eq)] @@ -272,10 +308,11 @@ impl Frame { out.push(TAG_DOWN); put_u64(out, *monitor_id); out.push(match reason { - DownReason::Exit => DR_EXIT, - DownReason::Panic => DR_PANIC, - DownReason::Stopped => DR_STOPPED, - DownReason::NoProc => DR_NOPROC, + RemoteDownReason::Local(DownReason::Exit) => DR_EXIT, + RemoteDownReason::Local(DownReason::Panic) => DR_PANIC, + RemoteDownReason::Local(DownReason::Stopped) => DR_STOPPED, + RemoteDownReason::Local(DownReason::NoProc) => DR_NOPROC, + RemoteDownReason::Disconnected => DR_DISCONNECTED, }); } } @@ -364,13 +401,14 @@ impl Frame { TAG_DOWN => Frame::Down { monitor_id: r.u64()?, reason: match r.u8()? { - DR_EXIT => DownReason::Exit, - DR_PANIC => DownReason::Panic, - DR_STOPPED => DownReason::Stopped, - DR_NOPROC => DownReason::NoProc, + DR_EXIT => RemoteDownReason::Local(DownReason::Exit), + DR_PANIC => RemoteDownReason::Local(DownReason::Panic), + DR_STOPPED => RemoteDownReason::Local(DownReason::Stopped), + DR_NOPROC => RemoteDownReason::Local(DownReason::NoProc), + DR_DISCONNECTED => RemoteDownReason::Disconnected, t => { return Err(DecodeError::UnknownEnumTag { - what: "DownReason", + what: "RemoteDownReason", tag: t, }) } diff --git a/src/cluster/handshake.rs b/src/cluster/handshake.rs index 9e05f32..36f8408 100644 --- a/src/cluster/handshake.rs +++ b/src/cluster/handshake.rs @@ -25,14 +25,19 @@ pub struct Peer { pub meta: NodeMeta, } -/// Driver-supplied context for an inbound `Hello` — knowledge the pure -/// machine cannot have (c6 owns the connection table and dial set). -#[derive(Debug, Clone, Copy, Default)] -pub struct HelloCtx { - /// The offered name is already claimed by an established peer. - pub name_claimed: bool, - /// We have our own dial in flight to this peer name. - pub dialing_this_peer: bool, +/// Driver-supplied standing of the *offered* name at this node — knowledge +/// the pure machine cannot have (c6 owns the connection table and dial +/// set). One answer, in the responder's own precedence: an established +/// peer under that name outranks an in-flight dial to it. +#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] +pub enum PeerStanding { + /// Neither connected to nor dialing that name. + #[default] + Free, + /// An established peer already holds that name. + Claimed, + /// We have our own dial in flight to that name. + Dialing, } /// Simultaneous-connect tie-break: does the connection dialed by @@ -124,7 +129,7 @@ impl Responder { /// name → tie-break: validity before identity. `HelloReject` is the /// cross-version compatibility anchor, so a version-mismatched peer /// still gets one. - pub fn on_frame(self, frame: Frame, ctx: HelloCtx) -> ResponderOutcome { + pub fn on_frame(self, frame: Frame, standing: PeerStanding) -> ResponderOutcome { let Frame::Hello { proto_version, build_hash, @@ -147,13 +152,13 @@ impl Responder { if build_hash != self.local.build_hash { return reject(RejectReason::HashMismatch); } - if node_name == self.local.node_name || ctx.name_claimed { + if node_name == self.local.node_name || standing == PeerStanding::Claimed { return reject(RejectReason::NameTaken); } // Simultaneous connect: the inbound frame is the peer's dial. If our // own in-flight dial wins instead, drop this one silently — the peer // computes the same verdict (see `dial_wins`). - if ctx.dialing_this_peer && !dial_wins(&node_name, &self.local.node_name) { + if standing == PeerStanding::Dialing && !dial_wins(&node_name, &self.local.node_name) { return ResponderOutcome::TieBreakLoss; } diff --git a/src/cluster/manager.rs b/src/cluster/manager.rs index 2588023..f92482d 100644 --- a/src/cluster/manager.rs +++ b/src/cluster/manager.rs @@ -24,7 +24,7 @@ use std::collections::HashMap; use crate::channel::Sender; use crate::cluster::conn::ConnHandle; -use crate::cluster::handshake::{HelloCtx, Peer}; +use crate::cluster::handshake::{Peer, PeerStanding}; use crate::cluster::membership::{NodeEvent, NodeInfo}; use crate::cluster::remote::{bind_outbound, unbind_outbound}; use crate::gen_server::{GenServer, GenServerCtx, GenServerName, Watcher}; @@ -53,7 +53,7 @@ pub struct Manager { conns: HashMap, /// In-flight dial intents: peer name -> the actor performing the dial. /// Registered *before* connecting so a crossing inbound `Hello` sees it - /// (the `dialing_this_peer` half of [`HelloCtx`]); cleared the moment + /// ([`PeerStanding::Dialing`]); cleared the moment /// the dial resolves, and — because the dialer is monitored — on the /// dialer's death, so a panicking dial can never wedge the tie-break. dials: HashMap, @@ -143,9 +143,9 @@ pub enum Call { /// The dial to `name` resolved (either way): drop the intent. A call, /// not a cast, so the intent is provably gone before the dialer moves on. DialEnd { name: String }, - /// The [`HelloCtx`] for an inbound `Hello` offering `peer_name` — the + /// The [`PeerStanding`] of an inbound `Hello` offering `peer_name` — the /// accept path asks this between reading the frame and judging it. - HelloCtx { peer_name: String }, + Standing { peer_name: String }, /// Subscribe `tx` to membership events, snapshot-then-stream: one /// [`NodeEvent::NodeUp`] per live peer is queued into `tx` before this /// call answers, so the stream is exact from its first event (handlers @@ -166,7 +166,7 @@ pub enum Reply { /// `false`: another dial to this name is already in flight — do not dial. DialBegan(bool), DialEnded, - HelloCtx(HelloCtx), + Standing(PeerStanding), Subscribed, View(Vec), } @@ -206,8 +206,8 @@ impl GenServer for Manager { } // The outbound table (c9) is maintained here, inside the same // serialized handlers that own the connection's lifetime. - if let Some(out) = handle.take_outbound() { - bind_outbound(&peer.node_name, out); + if let Some((frames, monitors)) = handle.take_outbound() { + bind_outbound(&peer.node_name, peer.incarnation, frames, monitors); } let info = NodeInfo { node: self.node_id(&peer.node_name, peer.incarnation.get()), @@ -230,9 +230,7 @@ impl GenServer for Manager { // Dropping the entry drops the handle, which stops the actor. if let Some(entry) = self.conns.remove(&name) { unbind_outbound(&name); - self.emit(&NodeEvent::NodeDown { - node: entry.info.node, - }); + self.emit(&NodeEvent::NodeDown(entry.info)); } Reply::Disconnected } @@ -255,10 +253,15 @@ impl GenServer for Manager { self.dials.remove(&name); Reply::DialEnded } - Call::HelloCtx { peer_name } => Reply::HelloCtx(HelloCtx { - name_claimed: self.conns.contains_key(&peer_name), - dialing_this_peer: self.dials.contains_key(&peer_name), - }), + Call::Standing { peer_name } => { + Reply::Standing(if self.conns.contains_key(&peer_name) { + PeerStanding::Claimed + } else if self.dials.contains_key(&peer_name) { + PeerStanding::Dialing + } else { + PeerStanding::Free + }) + } Call::Subscribe { tx } => { // The snapshot: queued before `tx` joins the list, and — the // handlers being serialized — before any later event. @@ -279,13 +282,13 @@ impl GenServer for Manager { self.conns.retain(|name, entry| { let dead = entry.pid == down.pid; if dead { - downs.push((name.clone(), entry.info.node)); + downs.push((name.clone(), entry.info.clone())); } !dead }); - for (name, node) in downs { + for (name, info) in downs { unbind_outbound(&name); - self.emit(&NodeEvent::NodeDown { node }); + self.emit(&NodeEvent::NodeDown(info)); } self.dials.retain(|_, pid| *pid != down.pid); } diff --git a/src/cluster/membership.rs b/src/cluster/membership.rs index 4a97390..49e178a 100644 --- a/src/cluster/membership.rs +++ b/src/cluster/membership.rs @@ -57,9 +57,9 @@ pub enum NodeEvent { /// A peer's control connection established and registered. NodeUp(NodeInfo), /// That peer's connection ended — reaped, commanded down, or the manager - /// itself shut down. Which [`NodeInfo`] this id named was delivered in - /// the corresponding `NodeUp`. - NodeDown { node: NodeId }, + /// itself shut down. Carries the same [`NodeInfo`] the corresponding + /// `NodeUp` delivered, so consumers need no id→name reverse map. + NodeDown(NodeInfo), } /// A live membership subscription: the receiving end of the event stream diff --git a/src/cluster/pg.rs b/src/cluster/pg.rs new file mode 100644 index 0000000..6996237 --- /dev/null +++ b/src/cluster/pg.rs @@ -0,0 +1,546 @@ +//! RFC 010 c15 — distributed process groups (Phase 5). +//! +//! The Erlang `pg` shape (D18): every node's group store is the union of +//! its own local members and each peer's *announced* local members. There +//! is one **pg actor** per node — the c14 reaper grown up — and it is the +//! only writer of remote entries and the only sender of announcements: +//! +//! - **Origin owns its members.** Joins are local (`pg::join`), the eager +//! reaper is the liveness authority, and the origin announces every +//! change: `Join`/`Leave` incrementally to every up node, and a full +//! `Sync` of its local groups to a peer the moment that peer comes up +//! (`NodeUp`). Nobody monitors a remote member; a peer's `NodeDown` sweeps +//! every member it announced. +//! - **Transport is a pure consumer** of Phase 3/4: the exposed name +//! [`PG_NAME`] (`"pg"`) carrying [`PgMsg`] over postcard, sent with +//! [`remote::send`]. No new frame, no manager change. +//! - **No anti-entropy.** Per-origin ordering rides the single TCP link: +//! the actor sends `Sync` to a peer *before* it can send that peer any +//! `Join`/`Leave` (both from the same loop, over the same connection), and +//! a reconnect is a fresh `NodeUp` ⇒ fresh `Sync` replacing that peer's +//! set wholesale. +//! - **Local API unchanged.** `members`/`pick`/`dispatch` stay local-only +//! (`get_local_members`); a remote entry in the store carries the peer's +//! `NodeId` and never surfaces there. Cluster-wide reads are the new, +//! additive [`members_all`] over [`GroupMember`] (c16 adds `pick_any` / +//! `dispatch_any`). +//! +//! ## Ordering inside the node +//! +//! `pg::join`/`pg::leave` mutate the store on the caller's thread and then +//! *announce* to the actor's control inbox. Because the store op precedes the +//! announcement and the actor re-reads the store before broadcasting a +//! `Joined`, an announcement that has been overtaken (the member left or died +//! before the actor got to it) is dropped rather than advertised: the wire +//! never sees a `Join` for a member the origin no longer holds. `Leave` +//! broadcasts unconditionally — a spurious `Leave` is a no-op at the peer. +//! +//! Inbound: `NodeUp` is emitted by the manager on the accept/connect path, +//! *before* the peer's connection actor exists, so it is queued on the +//! membership stream before any frame from that peer can reach this inbox. +//! The actor still drains membership before it interprets a `PgMsg` whose +//! sender it does not know, and drops the message if the sender is still not +//! up (a ghost — its next `NodeUp` brings a `Sync`). + +use std::collections::HashMap; + +use crate::channel::{channel, select, Receiver, Selectable}; +use crate::cluster::expose::expose; +use crate::cluster::membership::{subscribe, MembershipEvents, NodeEvent, NodeInfo}; +use crate::cluster::remote::{ + self, local_identity, send_to_remote, RemoteName, RemotePid, ToRemoteError, +}; +use crate::monitor::Down; +use crate::pg::{ + live, member_for, reaper_inboxes, sweep_local_death, Incarnation, Member, Membership, PgEvent, +}; +use crate::pid::{assert_type, Addressable, Erased, Pid}; +use crate::registry::{register, send_to, SendError}; +use crate::scheduler::with_runtime; +use crate::Name; + +/// The exposed name every node's pg actor answers under. +pub const PG_NAME: Name = Name::new("pg"); + +/// The pg wire protocol. Every variant is origin-authored: `from` / the +/// pid's node is the node whose local members are being described. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum PgMsg { + /// The origin's complete local membership, sent to a peer on `NodeUp`. + /// Replaces whatever the receiver held for that origin. + Sync { + from: String, + groups: Vec<(String, Vec>)>, + }, + /// The origin added `pid` (its own) to `group`. + Join { + group: String, + pid: RemotePid, + }, + /// The origin removed `pid` from `group` — voluntary leave or death. + Leave { + group: String, + pid: RemotePid, + }, +} + +// Hand-rolled serde (the crate carries no serde-derive), as a 3-tuple with a +// leading tag: (0, from, groups) | (1, group, pid) | (2, group, pid). +impl serde::Serialize for PgMsg { + fn serialize(&self, s: S) -> Result { + use serde::ser::SerializeTuple; + let mut t = s.serialize_tuple(3)?; + match self { + PgMsg::Sync { from, groups } => { + t.serialize_element(&0u8)?; + t.serialize_element(from)?; + t.serialize_element(groups)?; + } + PgMsg::Join { group, pid } => { + t.serialize_element(&1u8)?; + t.serialize_element(group)?; + t.serialize_element(pid)?; + } + PgMsg::Leave { group, pid } => { + t.serialize_element(&2u8)?; + t.serialize_element(group)?; + t.serialize_element(pid)?; + } + } + t.end() + } +} + +impl<'de> serde::Deserialize<'de> for PgMsg { + fn deserialize>(d: D) -> Result { + struct V; + impl<'de> serde::de::Visitor<'de> for V { + type Value = PgMsg; + fn expecting(&self, f: &mut std::fmt::Formatter) -> std::fmt::Result { + f.write_str("a pg message tuple") + } + fn visit_seq>( + self, + mut seq: A, + ) -> Result { + use serde::de::Error; + let tag: u8 = seq + .next_element()? + .ok_or_else(|| A::Error::custom("pg: missing tag"))?; + let text: String = seq + .next_element()? + .ok_or_else(|| A::Error::custom("pg: missing name"))?; + match tag { + 0 => { + let groups = seq + .next_element()? + .ok_or_else(|| A::Error::custom("pg: missing groups"))?; + Ok(PgMsg::Sync { from: text, groups }) + } + 1 | 2 => { + let pid = seq + .next_element()? + .ok_or_else(|| A::Error::custom("pg: missing pid"))?; + Ok(if tag == 1 { + PgMsg::Join { group: text, pid } + } else { + PgMsg::Leave { group: text, pid } + }) + } + t => Err(A::Error::custom(format!("pg: unknown tag {t}"))), + } + } + } + d.deserialize_tuple(3, V) + } +} + +/// A member of a group as the cluster sees it: on this node (a plain +/// [`Pid`], sendable locally) or on a peer (a [`RemotePid`], sendable via +/// [`send_to_remote`](remote::send_to_remote)). `Pid` cannot hold a remote +/// (D14), hence the two-variant shape. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum GroupMember { + Local(Pid), + Remote(RemotePid), +} + +/// Every member of `group` cluster-wide, in the store's order: local members +/// filtered by the same liveness backstop as [`members`](crate::pg::members), +/// remote members exactly as their origins last announced them. Must run +/// inside [`run`](crate::run). +pub fn members_all(group: &str) -> Vec { + with_runtime(|inner| { + let me = inner.node_id; + let pg = inner.process_groups.lock(); + pg.all_of(group) + .into_iter() + .filter_map(|m| { + if m.node == me { + live(inner, m.pid).then_some(GroupMember::Local(m.pid)) + } else { + // A remote entry always has its node's name recorded + // (they land under the same lock); a missing one is a + // node already swept, so it hides rather than misnames. + pg.node_name(m.node).map(|name| { + GroupMember::Remote(RemotePid::from_parts( + name, + m.incarnation, + m.pid.index(), + m.pid.generation(), + )) + }) + } + }) + .collect() + }) +} + +/// One member of `group` cluster-wide, or `None` if it has none: the first +/// entry in the store's order (this node's members in join order first when +/// they joined first — the same stateless first-live scan as +/// [`pick`](crate::pg::pick), extended over the peers' announced members). +/// Must run inside [`run`](crate::run). +pub fn pick_any(group: &str) -> Option { + members_all(group).into_iter().next() +} + +/// Why [`dispatch_any`] handed `msg` back. +#[derive(Debug)] +pub enum DispatchAnyError { + /// The group has no member anywhere. + NoMember(M), + /// The pick was local and the local typed send failed. + Local(SendError), + /// The pick was remote and the remote send failed at this node. + Remote(ToRemoteError), +} + +impl DispatchAnyError { + /// The undelivered message. + pub fn into_inner(self) -> M { + match self { + DispatchAnyError::NoMember(m) => m, + DispatchAnyError::Local(e) => e.into_inner(), + DispatchAnyError::Remote(e) => e.into_inner(), + } + } +} + +/// [`pick_any`] and send in one step, returning the member reached: a local +/// pick goes through [`send_to`], a remote one through [`send_to_remote`] +/// (so `Ok` for a remote member means "handed to the connection", RFC 010 +/// §3). Homogeneous pool assumed, as for [`dispatch`](crate::pg::dispatch); +/// a wrong `A` degrades to a clean error at the target, never a misroute. +/// Must run inside [`run`](crate::run). +pub fn dispatch_any(group: &str, msg: A::Msg) -> Result> +where + A: Addressable, + A::Msg: serde::Serialize, +{ + match pick_any(group) { + None => Err(DispatchAnyError::NoMember(msg)), + Some(GroupMember::Local(pid)) => send_to(assert_type::(pid), msg) + .map(|()| GroupMember::Local(pid)) + .map_err(DispatchAnyError::Local), + Some(GroupMember::Remote(rp)) => send_to_remote(rp.clone().assert_type::(), msg) + .map(|()| GroupMember::Remote(rp)) + .map_err(DispatchAnyError::Remote), + } +} + +/// Attach the pg actor to the running cluster. Called once by +/// `cluster::start` after the manager is up and the local identity is set; +/// spawns the actor if this run has not joined anything yet. +pub(crate) fn attach_cluster() { + let _ = reaper_inboxes().ctl.send(PgEvent::Attach); +} + +/// The attached half of the actor's state: who is up (by name) and the +/// membership stream. +struct Attached { + events: MembershipEvents, + peers: HashMap, + /// This node's wire identity: `Sync`'s `from`, and the stamp on every + /// pid we ship (attach requires it, so no `None` path exists here). + me: String, + incarnation: Incarnation, +} + +/// The pg actor: the c14 reaper (`deaths`), the local API's announcements +/// (`ctl`), and — once attached — the membership stream and the exposed +/// `"pg"` inbox, all in one drain-then-select loop. `deaths`/`ctl` closing +/// is the run tearing down; the membership stream closing is the manager +/// gone (detach, keep reaping). +pub(crate) fn actor(deaths: Receiver, ctl: Receiver) { + let (pg_tx, pg_rx) = channel::(); + let mut cl: Option = None; + loop { + loop { + match deaths.try_recv() { + Ok(Some(down)) => on_death(cl.as_ref(), down.pid), + Ok(None) => break, + Err(_) => return, + } + } + loop { + match ctl.try_recv() { + Ok(Some(PgEvent::Attach)) => { + // Own the name BEFORE subscribing (which yields to the + // manager): a peer's first frame must find "pg" exposed + // and resolvable, or it is dropped. Idempotent for a + // re-attach: same actor, same channel (the registry + // refuses a *second* live one). + let _ = register(PG_NAME, pg_tx.clone()); + expose(PG_NAME); + if let Some(a) = attach() { + cl = Some(a); + } + } + Ok(Some(PgEvent::Joined { group, pid })) => on_joined(cl.as_ref(), &group, pid), + Ok(Some(PgEvent::Left { group, pid })) => on_left(cl.as_ref(), &group, pid), + Ok(None) => break, + Err(_) => return, + } + } + if let Some(a) = cl.as_mut() { + if !drain_events(a) { + cl = None; + continue; + } + loop { + match pg_rx.try_recv() { + Ok(Some(msg)) => on_msg(a, msg), + Ok(None) => break, + Err(_) => return, // our own inbox: only on teardown + } + } + } + // Wait. Control first (attach/teardown must be prompt), then deaths, + // then the cluster arms. + let mut arms: Vec<&dyn Selectable> = vec![&ctl, &deaths]; + if let Some(a) = cl.as_ref() { + arms.push(&a.events.rx); + arms.push(&pg_rx); + } + let _ = select(&arms); + } +} + +fn attach() -> Option { + let events = subscribe()?; + let (me, incarnation) = local_identity()?; + Some(Attached { + events, + peers: HashMap::new(), + me, + incarnation, + }) +} + +/// Fold pending membership events: `NodeUp` ⇒ record + `Sync` that peer; +/// `NodeDown` ⇒ sweep every member it announced. `false` when the stream +/// has closed. +fn drain_events(a: &mut Attached) -> bool { + loop { + match a.events.rx.try_recv() { + Ok(Some(NodeEvent::NodeUp(info))) => { + let name = info.name.clone(); + a.peers.insert(name.clone(), info); + // Snapshot under the store lock, then stamp wire pids + // outside it (`from_local` marks watchable under the slot's + // cold lock — Leaf-on-Leaf nesting is asserted). + let local: Vec<(String, Vec)> = + with_runtime(|inner| inner.process_groups.lock().groups_on(inner.node_id)); + let groups = local + .into_iter() + .map(|(g, pids)| (g, pids.into_iter().map(|p| wire(a, p)).collect())) + .collect(); + let msg = PgMsg::Sync { + from: a.me.clone(), + groups, + }; + let _ = remote::send(RemoteName::new(name, PG_NAME), msg); + } + Ok(Some(NodeEvent::NodeDown(info))) => { + a.peers.remove(&info.name); + with_runtime(|inner| { + let mut pg = inner.process_groups.lock(); + pg.remove_where(|m| m.node == info.node); + pg.forget_node_name(info.node); + }); + } + Ok(None) => return true, + Err(_) => return false, + } + } +} + +/// Send `msg` to every up peer. `NotConnected` is ignored: that peer's +/// `NodeDown` is on its way and its next `NodeUp` gets a `Sync`. +fn broadcast(a: &Attached, msg: PgMsg) { + for name in a.peers.keys() { + let _ = remote::send(RemoteName::new(name.clone(), PG_NAME), msg.clone()); + } +} + +fn on_death(a: Option<&Attached>, pid: Pid) { + let evicted = sweep_local_death(pid); + if let Some(a) = a { + for (group, ms) in evicted { + broadcast( + a, + PgMsg::Leave { + group, + pid: wire(a, ms.member.pid), + }, + ); + } + } +} + +fn on_joined(a: Option<&Attached>, group: &str, pid: Pid) { + let Some(a) = a else { return }; + // Re-check: a leave/death may have overtaken the announcement. + let still = with_runtime(|inner| { + let m = member_for(inner, pid); + inner.process_groups.lock().contains(group, &m) + }); + if still { + broadcast( + a, + PgMsg::Join { + group: group.to_owned(), + pid: wire(a, pid), + }, + ); + } +} + +fn on_left(a: Option<&Attached>, group: &str, pid: Pid) { + let Some(a) = a else { return }; + broadcast( + a, + PgMsg::Leave { + group: group.to_owned(), + pid: wire(a, pid), + }, + ); +} + +/// The wire form of a local member pid, stamped with the identity the +/// actor was attached with (marks watchable, like `from_local`). +fn wire(a: &Attached, pid: Pid) -> RemotePid { + RemotePid::from_local_at(pid, a.me.clone(), a.incarnation) +} + +/// The named origin's `NodeInfo`, if it is up. A second look at the +/// membership stream covers a `NodeUp` that landed after this loop +/// iteration's drain; anything still unknown is a ghost and is dropped. +fn origin(a: &mut Attached, name: &str) -> Option { + if let Some(i) = a.peers.get(name) { + return Some(i.clone()); + } + drain_events(a); + a.peers.get(name).cloned() +} + +/// `origin`, additionally requiring `pid` to be stamped with the origin's +/// current incarnation — a pid from a previous life of that node is a ghost. +fn origin_of(a: &mut Attached, pid: &RemotePid) -> Option { + origin(a, pid.node()).filter(|i| i.incarnation == pid.incarnation()) +} + +fn remote_membership(origin: &NodeInfo, pid: &RemotePid) -> Membership { + Membership { + member: Member { + node: origin.node, + incarnation: origin.incarnation, + pid: Pid::new(pid.index(), pid.generation()), + }, + monitor: None, + } +} + +fn on_msg(a: &mut Attached, msg: PgMsg) { + match msg { + PgMsg::Sync { from, groups } => { + let Some(info) = origin(a, &from) else { return }; + with_runtime(|inner| { + let mut pg = inner.process_groups.lock(); + pg.remove_where(|m| m.node == info.node); + pg.set_node_name(info.node, info.name.clone()); + for (group, pids) in &groups { + // Origin-authored: only its own current-incarnation pids. + for p in pids + .iter() + .filter(|p| p.node() == from && p.incarnation() == info.incarnation) + { + pg.join(group, remote_membership(&info, p)); + } + } + }); + } + PgMsg::Join { group, pid } => { + let Some(info) = origin_of(a, &pid) else { + return; + }; + with_runtime(|inner| { + let mut pg = inner.process_groups.lock(); + pg.set_node_name(info.node, info.name.clone()); + pg.join(&group, remote_membership(&info, &pid)); + }); + } + PgMsg::Leave { group, pid } => { + let Some(info) = origin_of(a, &pid) else { + return; + }; + with_runtime(|inner| { + let ms = remote_membership(&info, &pid); + inner.process_groups.lock().leave(&group, ms.member); + }); + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::cluster::envelope::{decode_payload, encode_payload}; + + #[test] + fn pg_msg_roundtrips_every_variant() { + let p = RemotePid::::from_parts("a", Incarnation::new(9), 3, 1); + for m in [ + PgMsg::Sync { + from: "a".into(), + groups: vec![ + ("g".into(), vec![p.clone(), p.clone()]), + ("h".into(), vec![]), + ], + }, + PgMsg::Sync { + from: "a".into(), + groups: vec![], + }, + PgMsg::Join { + group: "g".into(), + pid: p.clone(), + }, + PgMsg::Leave { + group: "g".into(), + pid: p.clone(), + }, + ] { + let bytes = encode_payload(&m).unwrap(); + let back: PgMsg = decode_payload(&bytes).unwrap(); + assert_eq!(back, m); + } + } + + #[test] + fn pg_msg_rejects_unknown_tag() { + let bytes = encode_payload(&(7u8, "x", 0u32)).unwrap(); + assert!(decode_payload::(&bytes).is_err()); + } +} diff --git a/src/cluster/remote.rs b/src/cluster/remote.rs index 33e4191..8b74c43 100644 --- a/src/cluster/remote.rs +++ b/src/cluster/remote.rs @@ -41,15 +41,41 @@ //! Refusals are silent to the sender by design (§3: send failure reflects //! local knowledge only); they are observable locally as the returned //! [`InboundVerdict`], which the conn actor may log or count. +//! +//! ## Pids (c10, D14) +//! +//! [`RemotePid`] = `(node_name, incarnation, index, generation)` + +//! phantom — identity-bound, dead when that incarnation dies, never +//! redirects. The node travels as its **name** (a global identifier, so a pid +//! forwarded through a third node needs no re-mapping); NodeId is a local +//! alias and never crosses. A local `Pid` serializes *as* a `RemotePid` +//! stamped from the ambient [local identity](set_local_identity); a +//! `RemotePid` deserializes into `Pid` only when it names this node (the +//! collapse), else it is a decode error — fields that may hold a pid from +//! anywhere are typed `RemotePid`. +//! +//! [`send_to_remote`] is the pid-targeted send. A self-node pid short- +//! circuits to the local typed send with the message object itself — no +//! encode, no frame (zero-copy-equivalent). Otherwise the outbound table +//! (widened to carry each node's **current incarnation**) does the RFC v2 §3 +//! check at the send site: a pid of a dead incarnation is +//! [`ToRemoteError::DeadIncarnation`] and no frame is emitted. Inbound +//! `Send` frames are delivered by index/generation through c8's +//! [`decode_deliver`]: the target actor's published channel for the exposed +//! type is the only route (the reply-to path requires +//! [`expose_type`](crate::cluster::expose::expose_type) at the receiver). +use std::cell::Cell; use std::collections::HashMap; use std::marker::PhantomData; -use crate::channel::Sender; -use crate::cluster::envelope::{encode_payload, Frame, PayloadError}; +use crate::channel::{channel, Receiver, RecvError, Selectable, Sender}; +use crate::cluster::envelope::{encode_payload, Frame, PayloadError, RemoteDownReason}; use crate::cluster::expose::{decode_deliver, exposed_hash, type_hash, DeliverError}; -use crate::pid::Name; -use crate::registry::whereis; +use crate::monitor::{demonitor, monitor, Monitor, MonitorId}; +use crate::pg::Incarnation; +use crate::pid::{Addressable, Erased, Name, Pid}; +use crate::registry::{send_to, whereis, SendError}; use crate::scheduler::with_runtime; /// A name on a specific remote node: `(node_name, Name)`. Sendable via @@ -111,26 +137,97 @@ impl RemoteSendError { } } -/// The outbound table, one per runtime (a `RuntimeInner` field). +/// The outbound table, one per runtime (a `RuntimeInner` field): per live +/// node, its current incarnation (the RFC v2 §3 send-site check) and the +/// connection's dedicated outbound sender. Plus this node's own wire +/// identity, which pid serialization stamps. pub(crate) struct Outbound { - by_node: HashMap>, + by_node: HashMap, + local: Option<(String, Incarnation)>, +} + +/// One live connection as the outbound path sees it: the peer's current +/// incarnation and the two inboxes of its connection actor — frames (c9) +/// and monitor bookkeeping (c12, [`MonCmd`]). +pub(crate) struct Route { + incarnation: Incarnation, + frames: Sender, + monitors: Sender, } impl Outbound { pub(crate) fn new() -> Self { Outbound { by_node: HashMap::new(), + local: None, } } } -/// Manager-only: bind `node`'s outbound channel. Called inside `Register`. -pub(crate) fn bind_outbound(node: &str, tx: Sender) { +/// Set this node's wire identity — what serialized pids are stamped with +/// and what a `RemotePid` must name to collapse. `cluster::start` sets it; +/// exposed for local tests. Must run inside [`run`](crate::run). +pub fn set_local_identity(node: &str, incarnation: Incarnation) { with_runtime(|inner| { - inner.outbound.lock().by_node.insert(node.to_string(), tx); + inner.outbound.lock().local = Some((node.to_string(), incarnation)); }); } +/// This node's wire identity, if set. Must run inside [`run`](crate::run). +pub fn local_identity() -> Option<(String, Incarnation)> { + with_runtime(|inner| inner.outbound.lock().local.clone()) +} + +/// Manager-only: bind `node`'s outbound channels at `incarnation`. Called +/// inside `Register`. +pub(crate) fn bind_outbound( + node: &str, + incarnation: Incarnation, + frames: Sender, + monitors: Sender, +) { + with_runtime(|inner| { + inner.outbound.lock().by_node.insert( + node.to_string(), + Route { + incarnation, + frames, + monitors, + }, + ); + }); +} + +/// Test probe: bind an arbitrary sender as `node`'s outbound so a test can +/// assert what frames leave — or don't. Same table, same lookup as the real +/// path (this is how "no frame emitted" is asserted at the frame level). +/// Frames only: there is no connection actor behind a probe, so a +/// [`monitor_remote`] against a probed node reports `Disconnected`. +pub fn bind_outbound_probe(node: &str, incarnation: Incarnation, tx: Sender) { + drop(bind_outbound_probe_with_monitors(node, incarnation, tx)); +} + +/// The monitor half of a probed node's inbox: opaque, held only to be +/// dropped. See [`bind_outbound_probe_with_monitors`]. +pub struct MonitorInbox { + _rx: Receiver, +} + +/// Test probe: like [`bind_outbound_probe`], but the monitor-command +/// receiver is handed back instead of dropped, so a test can stage the +/// c13 drain gap — a `Monitor` command that reached the connection's inbox +/// and dies unread when the inbox is dropped. While the inbox lives, +/// [`monitor_remote`] against the probed node is simply in flight. +pub fn bind_outbound_probe_with_monitors( + node: &str, + incarnation: Incarnation, + tx: Sender, +) -> MonitorInbox { + let (mon_tx, mon_rx) = channel(); + bind_outbound(node, incarnation, tx, mon_tx); + MonitorInbox { _rx: mon_rx } +} + /// Manager-only: unbind `node`'s outbound channel. Called on `Disconnect`, /// reap, and manager shutdown. Dropping the sender is what closes the conn /// actor's outbound arm — but that arm's closure is NOT a stop signal (the @@ -191,13 +288,249 @@ pub fn send_remote_raw( /// One lookup, one send. Clone the sender out under the lock and send /// outside it (a channel send can unpark the conn actor). fn hand_to_connection(node: &str, frame: Frame) -> Result<(), NotConnected> { - let tx = with_runtime(|inner| inner.outbound.lock().by_node.get(node).cloned()); + let tx = with_runtime(|inner| { + inner + .outbound + .lock() + .by_node + .get(node) + .map(|r| r.frames.clone()) + }); match tx { Some(tx) => tx.send(frame).map_err(|_| NotConnected), None => Err(NotConnected), } } +// ---- pids --------------------------------------------------------------- + +/// A pid on some node: `(node_name, incarnation, index, generation)` plus +/// the actor type. See the module docs. Serializes as a 4-tuple. +pub struct RemotePid { + node: String, + incarnation: Incarnation, + index: u32, + generation: u32, + _marker: PhantomData A>, +} + +impl RemotePid { + /// Build from raw parts (tests, and codecs re-hydrating a pid). + pub fn from_parts( + node: impl Into, + incarnation: Incarnation, + index: u32, + generation: u32, + ) -> Self { + RemotePid { + node: node.into(), + incarnation, + index, + generation, + _marker: PhantomData, + } + } + + /// The wire form of a local pid, stamped with this node's identity, and + /// **marked watchable** — asking for the wire form *is* the intent to + /// ship the pid, so this is the same D12 set-site as `Pid::serialize` + /// (c12 made it explicit: a peer may monitor exactly the pids that + /// crossed, and a pid handed out via `from_local` in a hand-built reply + /// has crossed). Must run inside [`run`](crate::run). + /// + /// `None` when this runtime has no wire identity (no `cluster::start`, + /// no [`set_local_identity`]): such a pid cannot name a node, and a + /// stamped `("", 0)` would be dropped by every peer with no signal. + /// The pid is not marked watchable in that case either. + pub fn from_local(pid: Pid) -> Option { + let (node, incarnation) = local_identity()?; + Some(Self::from_local_at(pid, node, incarnation)) + } + + /// `from_local` with the identity supplied by the caller — for a holder + /// that already carries the node's identity (the pg actor) and must not + /// have a `None` path. Marks watchable like `from_local`. + pub(crate) fn from_local_at(pid: Pid, node: String, incarnation: Incarnation) -> Self { + crate::monitor::mark_watchable(pid); + RemotePid::from_parts(node, incarnation, pid.index(), pid.generation()) + } + + /// The collapse: `Some(local pid)` iff this pid names this very node + /// (name and incarnation). Must run inside [`run`](crate::run). + pub fn local(&self) -> Option> { + let (n, i) = local_identity()?; + (n == self.node && i == self.incarnation) + .then(|| crate::pid::assert_type::(Pid::new(self.index, self.generation))) + } + + /// Drop the actor type: the untyped `RemotePid`, the form + /// [`RemoteDown`] and [`RemoteMonitor`] carry (mirrors [`Pid::erase`]). + pub fn erase(self) -> RemotePid { + RemotePid::from_parts(self.node, self.incarnation, self.index, self.generation) + } + + /// Re-type an erased pid as `RemotePid` — the unchecked mirror of + /// `pid::assert_type`, with the same degradation: a wrong `B` means the + /// target refuses the payload's hash (never a misroute). + pub(crate) fn assert_type(self) -> RemotePid { + RemotePid::from_parts(self.node, self.incarnation, self.index, self.generation) + } + + pub fn node(&self) -> &str { + &self.node + } + pub fn incarnation(&self) -> Incarnation { + self.incarnation + } + pub fn index(&self) -> u32 { + self.index + } + pub fn generation(&self) -> u32 { + self.generation + } +} + +impl Clone for RemotePid { + fn clone(&self) -> Self { + RemotePid::from_parts( + self.node.clone(), + self.incarnation, + self.index, + self.generation, + ) + } +} +impl PartialEq for RemotePid { + fn eq(&self, o: &Self) -> bool { + self.node == o.node + && self.incarnation == o.incarnation + && self.index == o.index + && self.generation == o.generation + } +} +impl Eq for RemotePid {} +impl std::fmt::Debug for RemotePid { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!( + f, + "<{}.{}@{}#{}>", + self.index, + self.generation, + self.node, + self.incarnation.get() + ) + } +} + +impl serde::Serialize for RemotePid { + fn serialize(&self, s: S) -> Result { + ( + self.node.as_str(), + self.incarnation.get(), + self.index, + self.generation, + ) + .serialize(s) + } +} +impl<'de, A> serde::Deserialize<'de> for RemotePid { + fn deserialize>(d: D) -> Result { + let (node, inc, index, generation) = <(String, u32, u32, u32)>::deserialize(d)?; + Ok(RemotePid::from_parts( + node, + Incarnation::new(inc), + index, + generation, + )) + } +} + +/// Why a pid-targeted send did not leave this node. Local knowledge only. +#[derive(Debug)] +pub enum ToRemoteError { + /// No live connection to the pid's node. + NotConnected(M), + /// The pid's incarnation is not that node's current one (RFC v2 §3): the + /// actor died with its incarnation. Detected at the send site; no frame. + DeadIncarnation(M), + /// The payload did not serialize. + Encode(M, PayloadError), + /// The pid collapsed to a local one and the local typed send failed. + Local(SendError), +} + +impl ToRemoteError { + /// The undelivered message. + pub fn into_inner(self) -> M { + match self { + ToRemoteError::NotConnected(m) + | ToRemoteError::DeadIncarnation(m) + | ToRemoteError::Encode(m, _) => m, + ToRemoteError::Local(e) => e.into_inner(), + } + } +} + +/// Send `msg` to a pid, wherever it lives. Self-node pids short-circuit to +/// the local typed send with `msg` itself (no encode, no frame); others go +/// out as a `Send` frame after the incarnation check. `Ok(())` for a remote +/// target = handed to the connection's inbox. Must run inside +/// [`run`](crate::run). +pub fn send_to_remote(target: RemotePid, msg: A::Msg) -> Result<(), ToRemoteError> +where + A: Addressable, + A::Msg: serde::Serialize, +{ + if let Some(local) = target.local() { + return send_to(local, msg).map_err(ToRemoteError::Local); + } + let route = with_runtime(|inner| { + inner + .outbound + .lock() + .by_node + .get(&target.node) + .map(|r| (r.incarnation, r.frames.clone())) + }); + let (current, tx) = match route { + Some(r) => r, + None => return Err(ToRemoteError::NotConnected(msg)), + }; + if current != target.incarnation { + return Err(ToRemoteError::DeadIncarnation(msg)); + } + let payload = match encode_payload(&msg) { + Ok(p) => p, + Err(e) => return Err(ToRemoteError::Encode(msg, e)), + }; + let frame = Frame::Send { + index: target.index, + generation: target.generation, + type_hash: type_hash::(), + payload, + }; + tx.send(frame).map_err(|_| ToRemoteError::NotConnected(msg)) +} + +/// The inbound `Send` seam: deliver `payload` under `type_hash` to the local +/// actor `(index, generation)`. Node and incarnation are implicit in the +/// connection (bound at handshake) — the frame carries only the slot +/// identity. Delivery goes through c8's decoder table, so only types the +/// receiver has [`expose_type`](crate::cluster::expose::expose_type)d (or +/// exposed by name) can land; anything else is refused, never misrouted. +pub fn deliver_to_pid( + index: u32, + generation: u32, + type_hash: u64, + payload: &[u8], +) -> InboundVerdict { + let pid = Pid::new(index, generation); + match decode_deliver(type_hash, pid, payload) { + Ok(()) => InboundVerdict::Delivered, + Err(e) => InboundVerdict::Refused(e), + } +} + /// What the inbound seam did with a `SendNamed`. Local observability only; /// nothing goes back on the wire (RFC §3). #[derive(Debug)] @@ -217,6 +550,20 @@ pub enum InboundVerdict { Refused(DeliverError), } +impl InboundVerdict { + /// A short static label for tracing/counting (`smarm-trace` records one + /// `ClusterInbound` event per frame with it). + pub fn label(&self) -> &'static str { + match self { + InboundVerdict::Delivered => "delivered", + InboundVerdict::NotExposed => "not_exposed", + InboundVerdict::HashMismatch { .. } => "hash_mismatch", + InboundVerdict::Unresolved => "unresolved", + InboundVerdict::Refused(_) => "refused", + } + } +} + /// THE inbound resolution seam: exposed-set check → hash check → registry /// resolution → c8 delivery. See the module docs. Must run inside /// [`run`](crate::run) — the conn actor's context. @@ -238,3 +585,251 @@ pub fn deliver_named(name: &str, type_hash: u64, payload: &[u8]) -> InboundVerdi Err(e) => InboundVerdict::Refused(e), } } + +// ---- monitors (c12) ----------------------------------------------------- + +/// Bookkeeping commands from [`monitor_remote`]/[`demonitor_remote`] to the +/// connection actor that owns the link to the target's node. The actor +/// records the registration and *then* emits the `Monitor` frame itself, so +/// a `Down` can never arrive at a table that does not yet know the id. It +/// lives in the actor (not on `RuntimeInner`) so the bookkeeping dies with +/// the connection — exactly what c13 needs to synthesize `Disconnected`. +pub(crate) enum MonCmd { + Monitor { + id: MonitorId, + target: RemotePid, + tx: Sender, + }, + Demonitor { + id: MonitorId, + }, +} + +/// A remotely-monitored actor's termination notice — the cluster analog of +/// [`Down`](crate::monitor::Down), with the pid in its wire form because it +/// may name any node. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct RemoteDown { + /// The pid that was being monitored. + pub pid: RemotePid, + /// How it went down. `Disconnected` means the *link* to its node was + /// lost (or absent) — nothing is known about the actor itself. + pub reason: RemoteDownReason, +} + +enum Watch { + /// The target collapsed to this node: an ordinary local monitor, + /// translated on read. + Local(Monitor), + /// The target is elsewhere: the connection actor for its node holds the + /// registration and forwards the peer's `Down` frame here. The + /// `RemoteState` is the read-side backstop (c13): a channel that closes + /// while `Live` — the connection died with our command unread — reads + /// as `Disconnected` once; afterwards, and after a cancel, closed is + /// just closed. + Remote(Receiver, Cell), +} + +/// Where a remote-watch stands from the reader's side. +#[derive(Clone, Copy, PartialEq, Eq)] +enum RemoteState { + /// No notice yet, not cancelled: a closed channel means `Disconnected`. + Live, + /// The one notice has been read (or synthesized): nothing more is due. + Done, + /// `demonitor_remote` ran: never synthesize. + Cancelled, +} + +/// A live remote monitor: read its one [`RemoteDown`] with +/// [`recv`](RemoteMonitor::recv)/[`try_recv`](RemoteMonitor::try_recv), or +/// fold it into a `select` via [`arm`](RemoteMonitor::arm). Distinct from +/// [`Monitor`] on purpose: its target is a [`RemotePid`], its notice a +/// [`RemoteDown`], and it can report `Disconnected` — none of which a local +/// monitor can express. Dropping it discards an unread notice, like the +/// local one; after [`demonitor_remote`] the channel is closed and empty, so +/// `recv` errs rather than parking — also like the local one. +/// +/// Exactly one notice is guaranteed even if the connection actor dies with +/// the registration unread (the c13 drain gap): a channel that closes +/// before any notice — and before any cancel — reads as `Disconnected`, +/// once. The next read is the ordinary closed-channel `Err`. +pub struct RemoteMonitor { + /// This registration's process-unique id — minted here, echoed by the + /// peer in its `Down` frame. + pub id: MonitorId, + /// The pid being monitored. + pub target: RemotePid, + watch: Watch, +} + +impl RemoteMonitor { + /// Block (cooperatively) for the notice. + pub fn recv(&self) -> Result { + match &self.watch { + Watch::Local(m) => m.rx.recv().map(|d| RemoteDown { + pid: self.target.clone(), + reason: d.reason.into(), + }), + Watch::Remote(rx, st) => match rx.recv() { + Ok(d) => { + st.set(RemoteState::Done); + Ok(d) + } + Err(e) => self.closed(st).ok_or(e), + }, + } + } + + /// The notice if it has arrived; `Ok(None)` if not yet. + pub fn try_recv(&self) -> Result, RecvError> { + match &self.watch { + Watch::Local(m) => m.rx.try_recv().map(|o| { + o.map(|d| RemoteDown { + pid: self.target.clone(), + reason: d.reason.into(), + }) + }), + Watch::Remote(rx, st) => match rx.try_recv() { + Ok(Some(d)) => { + st.set(RemoteState::Done); + Ok(Some(d)) + } + Ok(None) => Ok(None), + Err(e) => self.closed(st).map(Some).ok_or(e), + }, + } + } + + /// The channel closed. While `Live` — no notice yet, no cancel — that + /// is the connection having died with our registration unread, so + /// synthesize the one `Disconnected` and mark `Done`; otherwise closed + /// is just closed. + fn closed(&self, st: &Cell) -> Option { + if st.get() != RemoteState::Live { + return None; + } + st.set(RemoteState::Done); + Some(RemoteDown { + pid: self.target.clone(), + reason: RemoteDownReason::Disconnected, + }) + } + + /// The selectable arm: readiness means [`try_recv`](Self::try_recv) + /// will yield the notice. + pub fn arm(&self) -> &dyn Selectable { + match &self.watch { + Watch::Local(m) => &m.rx, + Watch::Remote(rx, _) => rx, + } + } +} + +impl std::fmt::Debug for RemoteMonitor { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("RemoteMonitor") + .field("id", &self.id) + .field("target", &self.target) + .finish_non_exhaustive() + } +} + +/// Monitor `target`, wherever it lives. Exactly one [`RemoteDown`] arrives: +/// +/// - self-node pid ⇒ an ordinary local monitor underneath (same NoProc rule); +/// - no live connection to the pid's node ⇒ `Disconnected`, queued at once +/// (the remote analog of NoProc: nothing can be known); +/// - the pid's incarnation is not the node's current one ⇒ `NoProc`, queued +/// at once — the node is *known* to have restarted, so its actor is a +/// corpse, not a partition (RFC v2 §3); +/// - otherwise the connection actor registers the id and sends `Monitor`; +/// the peer answers with the true terminal reason on exit, or immediately +/// with the recorded reason for a corpse (`terminal_reason`, RFC §6) or +/// `NoProc` for a pid it never exposed and never shipped. +/// +/// The connection dropping while the monitor is outstanding delivers +/// `Disconnected` (c13): the connection actor synthesizes it on teardown, +/// and the monitor's own read path backstops the case where the actor died +/// with the registration still unread. Must run inside [`run`](crate::run). +pub fn monitor_remote(target: RemotePid) -> RemoteMonitor { + if let Some(local) = target.local() { + let m = monitor(local); + return RemoteMonitor { + id: m.id, + target: target.erase(), + watch: Watch::Local(m), + }; + } + let target = target.erase(); + let (id, route) = with_runtime(|inner| { + let id = inner.alloc_monitor_id(); + let route = inner + .outbound + .lock() + .by_node + .get(&target.node) + .map(|r| (r.incarnation, r.monitors.clone())); + (id, route) + }); + let (tx, rx) = channel::(); + let immediate = match route { + None => Some(RemoteDownReason::Disconnected), + Some((current, _)) if current != target.incarnation => { + Some(RemoteDownReason::Local(crate::monitor::DownReason::NoProc)) + } + Some((_, mon_tx)) => { + let cmd = MonCmd::Monitor { + id, + target: target.clone(), + tx: tx.clone(), + }; + match mon_tx.send(cmd) { + Ok(()) => None, + Err(_) => Some(RemoteDownReason::Disconnected), // actor already gone + } + } + }; + if let Some(reason) = immediate { + let _ = tx.send(RemoteDown { + pid: target.clone(), + reason, + }); + } + RemoteMonitor { + id, + target, + watch: Watch::Remote(rx, Cell::new(RemoteState::Live)), + } +} + +/// Cancel `m`. No future notice will be *sent* for it; a notice already in +/// flight from the peer is dropped on arrival, and one already sitting in +/// `m` is discarded when `m` is dropped (same contract as +/// [`demonitor`]). Unlike the local form this returns nothing: the +/// registration is owned by the connection actor, so whether the `Down` +/// beat the cancel is not local knowledge. Must run inside +/// [`run`](crate::run). +pub fn demonitor_remote(m: &RemoteMonitor) { + match &m.watch { + Watch::Local(local) => { + let _ = demonitor(local); + } + Watch::Remote(_, st) => { + // Cancel first: a channel closing after this is closed, not a + // Disconnected notice — the caller asked for silence. + st.set(RemoteState::Cancelled); + let mon_tx = with_runtime(|inner| { + inner + .outbound + .lock() + .by_node + .get(&m.target.node) + .map(|r| r.monitors.clone()) + }); + if let Some(mon_tx) = mon_tx { + let _ = mon_tx.send(MonCmd::Demonitor { id: m.id }); + } + } + } +} diff --git a/src/monitor.rs b/src/monitor.rs index ac233d3..7aa8ae0 100644 --- a/src/monitor.rs +++ b/src/monitor.rs @@ -146,40 +146,68 @@ pub struct Monitor { pub fn monitor(target: Pid) -> Monitor { let target = target.erase(); let (tx, rx) = channel::(); - - // Implementation note: registration happens under the target's cold - // lock. `tx.clone()` takes the channel's own lock, a Channel-class - // RawMutex, which is explicitly permitted under a Leaf (cold) lock by - // the lock order documented in raw_mutex.rs. We must still not *send* - // under the lock, since `Sender::send` can unpark a parked receiver, - // and there's no reason to nest that. - let (id, registered) = with_runtime(|inner| { - let id = inner.alloc_monitor_id(); - let registered = match inner.slot_at(target) { - Some(slot) => { - let mut cold = slot.cold.lock(); - if slot.is_live_for(target) { - cold.monitors.push((id, tx.clone())); - true - } else { - false - } - } - None => false, - }; - (id, registered) - }); - - if !registered { + let id = with_runtime(|inner| inner.alloc_monitor_id()); + if !register_monitor(target, id, &tx) { let _ = tx.send(Down { pid: target, reason: DownReason::NoProc, }); } - Monitor { id, target, rx } } +/// Register a monitor `id` on `target` that delivers its `Down` to `tx` — the +/// primitive under [`monitor`], split out so a caller can fan many monitors +/// into ONE channel (process groups: every membership's death lands on the +/// reaper's single inbox). Returns `false` if `target` is already gone, in +/// which case nothing is registered and the caller decides what to queue +/// (`monitor` sends `NoProc`). The caller allocates `id` up front so it can +/// record the registration *before* arming it. +/// +/// Implementation note: registration happens under the target's cold lock. +/// `tx.clone()` takes the channel's own lock, a Channel-class RawMutex, which +/// is explicitly permitted under a Leaf (cold) lock by the lock order +/// documented in raw_mutex.rs. We must still not *send* under the lock, since +/// `Sender::send` can unpark a parked receiver, and there's no reason to nest +/// that. +pub(crate) fn register_monitor(target: Pid, id: MonitorId, tx: &Sender) -> bool { + with_runtime(|inner| match inner.slot_at(target) { + Some(slot) => { + let mut cold = slot.cold.lock(); + if slot.is_live_for(target) { + cold.monitors.push((id, tx.clone())); + true + } else { + false + } + } + None => false, + }) +} + +/// Remove registration `id` from `target` — the primitive under +/// [`demonitor`], for callers that hold only the id (see +/// [`register_monitor`]). `None` if the registration is not there: already +/// fired, already removed, or the slot has moved on to a new tenant. +/// +/// The registration is removed under the target's cold lock, but the +/// `Sender` is moved *out* and dropped only after the lock is released. +/// Dropping the last sender runs `Sender::drop`, which may unpark a parked +/// receiver; legal under a cold lock, but pointless to nest. +pub(crate) fn unregister_monitor(target: Pid, id: MonitorId) -> Option { + let removed: Option<(MonitorId, Sender)> = with_runtime(|inner| { + let slot = inner.slot_at(target)?; + let mut cold = slot.cold.lock(); + if slot.generation() != target.generation() { + return None; // slot reused; the Down already fired + } + let pos = cold.monitors.iter().position(|(mid, _)| *mid == id)?; + Some(cold.monitors.remove(pos)) + }); + // `removed`'s sender drops here, outside the lock. + removed.map(|(id, _sender)| id) +} + /// Flag `target`'s tenancy as watchable: its death will stamp the slot's /// terminal record (see [`terminal_reason`]), exactly as registering a name /// does. The bridge calls this wherever a smarm pid is *encoded across the @@ -210,6 +238,23 @@ pub fn mark_watchable(target: Pid) { }); } +/// Whether `target` is live *and* its tenancy is watchable. The cluster's +/// remote-monitor admission check (RFC 010 c12): a peer may monitor a pid only +/// if that pid was exposed or crossed the wire (the D12 set-sites), and a +/// live-but-unwatchable pid answers exactly like a dead one — no liveness leak +/// beyond what `watchable` already grants. Same context contract as +/// [`monitor`]. +#[cfg(feature = "cluster")] +pub(crate) fn is_watchable(target: Pid) -> bool { + let target = target.erase(); + with_runtime(|inner| { + inner.slot_at(target).is_some_and(|slot| { + let cold = slot.cold.lock(); + slot.is_live_for(target) && cold.watchable + }) + }) +} + /// The terminal [`DownReason`] of the tenancy `target` names, if that tenancy /// ever registered a name and is the *most recent named* death of its slot: /// finalize stamps the slot with `(generation, reason)` for once-registered @@ -249,20 +294,5 @@ pub fn terminal_reason(target: Pid) -> Option { /// instead of, or in addition to, calling this: dropping the [`Monitor`] /// closes its receiver and any queued notice is discarded with it. pub fn demonitor(m: &Monitor) -> Option { - // Implementation note: the registration is removed under the target's - // cold lock, but the `Sender` is moved *out* and dropped only after the - // lock is released. Dropping the last sender runs `Sender::drop`, which - // may unpark a parked receiver; legal under a cold lock, but pointless - // to nest. - let removed: Option<(MonitorId, Sender)> = with_runtime(|inner| { - let slot = inner.slot_at(m.target)?; - let mut cold = slot.cold.lock(); - if slot.generation() != m.target.generation() { - return None; // slot reused; the Down already fired - } - let pos = cold.monitors.iter().position(|(mid, _)| *mid == m.id)?; - Some(cold.monitors.remove(pos)) - }); - // `removed`'s sender drops here, outside the lock. - removed.map(|(id, _sender)| id) + unregister_monitor(m.target, m.id) } diff --git a/src/pg.rs b/src/pg.rs index 2b75f6e..03f39e6 100644 --- a/src/pg.rs +++ b/src/pg.rs @@ -99,10 +99,13 @@ //! ## Identity and clustering //! //! A group member is described by a [`Member`] — a [`Pid`] plus a [`NodeId`] and -//! an [`Incarnation`]. Today everything is single-node, those two fields are -//! fixed defaults, and you only ever pass and receive a plain [`Pid`]: the extra -//! identity is carried so this API will not have to change when groups learn to -//! span a cluster. +//! an [`Incarnation`]. Everything on this page is **local**: you pass and +//! receive plain [`Pid`]s, and [`members`] / [`pick`] / [`dispatch`] only ever +//! name actors on this node (Erlang's `get_local_members`). With the `cluster` +//! feature a group also holds the members other nodes have announced, carried +//! under their [`NodeId`]; those never surface here — the cluster-wide reads +//! live in [`cluster::pg`](crate::cluster::pg) (`members_all` and friends) and +//! return a `Local | Remote` member type, since a [`Pid`] cannot hold a remote. //! //! ## Running context //! @@ -110,10 +113,11 @@ //! from inside [`run`](crate::run) (that is, on an actor thread). Calling one //! from outside a running runtime panics. -use crate::monitor::{demonitor, monitor, Monitor}; +use crate::channel::{channel, Sender}; +use crate::monitor::{register_monitor, unregister_monitor, Down, DownReason, MonitorId}; use crate::pid::{assert_type, Addressable, Pid}; use crate::registry::{send_to, SendError}; -use crate::scheduler::with_runtime; +use crate::scheduler::{spawn_under, with_runtime}; use std::collections::HashMap; /// A cluster node handle. A `u32` integer handle, *not* an interned atom — the @@ -186,13 +190,15 @@ pub struct Member { pub pid: Pid, } -/// One membership: a [`Member`] and the [`Monitor`] that watches its liveness. -/// The monitor lives *alongside* the group entry so a group is -/// self-contained: draining the membership tells us whether the member is -/// still alive, and dropping the membership drops its monitor. -struct Membership { - member: Member, - monitor: Monitor, +/// One membership: a [`Member`] and the id of the monitor that watches its +/// liveness. The monitor's `Down` is delivered to the group reaper's single +/// inbox (see [`ProcessGroups::deaths`]), so the membership carries only what +/// [`leave`] needs to tear the registration down: the id. +pub(crate) struct Membership { + pub(crate) member: Member, + /// `None` for a remote member (cluster): the origin node is its liveness + /// authority; nothing here watches it. + pub(crate) monitor: Option, } /// The store: `name → multiset`. Within a single group a `Member` @@ -202,46 +208,60 @@ struct Membership { /// /// Locking discipline. Held under one Leaf-class `RawMutex` on `RuntimeInner`, /// mirroring the registry, and never held together with another Leaf lock (it -/// never touches the registry or a slot's cold lock). The two operations that -/// do need another lock are kept off the group-lock path: +/// never touches the registry or a slot's cold lock). Monitor registration and +/// removal take the target's cold lock (also Leaf), so they run *before* / +/// *after* the group lock, never under it — see [`join`] for the ordering that +/// makes that safe. Nothing under this lock ever touches a channel. /// -/// - `monitor()` / `demonitor()` take the target's cold lock (also Leaf), so -/// they run *before* / *after* the group lock, never under it. -/// - draining a monitor with `try_recv` takes the channel's Channel-class -/// lock, which the lock order permits *under* a Leaf; a channel critical -/// section only does the lock-free unpark protocol, so no Leaf ever nests -/// under it. -/// -/// Evicted and rejected [`Monitor`]s are therefore dropped only *after* the -/// group lock is released, so a receiver-drop never runs a wakeup under the -/// lock — the same discipline as `demonitor`. +/// Eviction is *eager*: every membership's monitor delivers to the one +/// `deaths` channel, drained by a per-run reaper actor that sweeps the dead +/// pid out of every group the moment its `Down` is scheduled. The read path +/// keeps a slot-liveness backstop for the window between a death and the +/// reaper's turn. pub(crate) struct ProcessGroups { groups: HashMap>, + /// The reaper's inboxes: every membership monitor is registered against + /// a clone of `deaths`. `None` until the first `join` of a run spawns + /// the reaper; a stale one (receiver gone with the previous run's + /// teardown) is detected via `receiver_alive` and replaced. + reaper: Option, + /// `NodeId → node name` for every peer with members in the store, kept + /// by the pg actor under this lock, so a stored remote member can be + /// rendered back to its wire identity without asking anyone. + #[cfg(feature = "cluster")] + node_names: HashMap, } impl ProcessGroups { pub(crate) fn new() -> Self { Self { groups: HashMap::new(), + reaper: None, + #[cfg(feature = "cluster")] + node_names: HashMap::new(), } } - /// Insert `ms` into `group`. Idempotent on the *member*: if the member is - /// already present the new membership is handed back (`Some`) so the caller - /// can tear its now-redundant monitor down outside the lock; `None` means - /// it was inserted. - fn join(&mut self, group: &str, ms: Membership) -> Option { + /// Forget the reaper. Called at the start of every `run()` so a stopped + /// reaper from a previous run is never sent to; `join` respawns. + pub(crate) fn reset_reaper(&mut self) { + self.reaper = None; + } + + /// Insert `ms` into `group`. Idempotent on the *member*: `false` means the + /// member was already present and nothing changed; `true` means inserted. + pub(crate) fn join(&mut self, group: &str, ms: Membership) -> bool { let v = self.groups.entry(group.to_owned()).or_default(); if v.iter().any(|e| e.member == ms.member) { - return Some(ms); + return false; } v.push(ms); - None + true } /// Remove `member`'s membership from `group`, returning it (so the caller - /// can `demonitor` it outside the lock). An emptied group is pruned. - fn leave(&mut self, group: &str, member: Member) -> Option { + /// can unregister its monitor outside the lock). An emptied group is pruned. + pub(crate) fn leave(&mut self, group: &str, member: Member) -> Option { let v = self.groups.get_mut(group)?; let pos = v.iter().position(|e| e.member == member)?; let removed = v.remove(pos); @@ -252,20 +272,23 @@ impl ProcessGroups { } /// The one dumb eviction primitive: drop every member matching `pred` from - /// every group, pruning emptied groups, and return the evicted memberships' - /// monitors for the caller to drop outside the lock. The primitive does not - /// know *why* a member leaves; that is the caller's concern. Its callers are - /// the death hook (`reap_group`) and, once clustering lands, an - /// incarnation-eviction sweep — both over this same predicate path, which is - /// the whole reason to shape eviction as a predicate. Insertion order within - /// a group is preserved (`members` / `pick` are order-stable). - fn remove_where(&mut self, mut pred: impl FnMut(&Member) -> bool) -> Vec { + /// every group, pruning emptied groups, and return the evicted + /// memberships with the group each was in. The primitive does not know + /// *why* a member leaves; that is the caller's concern. Its callers are + /// the reaper (a local death) and the cluster's node-down / re-sync + /// sweeps — all over this same predicate path, which is the whole reason + /// to shape eviction as a predicate. Insertion order within a group is + /// preserved (`members` / `pick` are order-stable). + pub(crate) fn remove_where( + &mut self, + mut pred: impl FnMut(&Member) -> bool, + ) -> Vec<(String, Membership)> { let mut evicted = Vec::new(); - self.groups.retain(|_, v| { + self.groups.retain(|g, v| { let mut i = 0; while i < v.len() { if pred(&v[i].member) { - evicted.push(v.remove(i).monitor); + evicted.push((g.clone(), v.remove(i))); } else { i += 1; } @@ -275,38 +298,6 @@ impl ProcessGroups { evicted } - /// Drain-on-contact death hook. The registry can prune a stale binding - /// lazily, on contact, because it only ever resolves one binding at a time; - /// a group is *iterated* — `members` fans out to everyone — so it must not - /// carry a dead member across a broadcast. Every group operation reaps the - /// group it touches first. - /// - /// Drains every membership monitor in `group` with a non-blocking - /// `try_recv`: a delivered `Down` (any reason) or a closed channel means - /// that member is dead. On the first death detected, sweep *all* of the - /// dead pids out of *every* group via [`remove_where`] — a death is removed - /// from each group it joined, not just the one being touched. Returns the - /// evicted monitors to drop outside the lock. - fn reap_group(&mut self, group: &str) -> Vec { - let dead: Vec = { - let Some(v) = self.groups.get(group) else { - return Vec::new(); - }; - v.iter() - .filter_map(|e| match e.monitor.rx.try_recv() { - // A Down arrived, or the channel closed and drained: dead. - Ok(Some(_)) | Err(_) => Some(e.member.pid), - // Empty but open — the sender still lives in the slot: alive. - Ok(None) => None, - }) - .collect() - }; - if dead.is_empty() { - return Vec::new(); - } - self.remove_where(|m| dead.contains(&m.pid)) - } - /// Raw enumeration of a group's members — no liveness filtering. Used by /// tests to assert storage state independently of the read-path backstop. #[cfg(test)] @@ -317,16 +308,24 @@ impl ProcessGroups { .unwrap_or_default() } - /// Live members of `group`, in insertion order. The `is_live` oracle is the - /// read-path backstop: a member whose slot is already dead is - /// dropped from the *result* even if its `Down` has not been drained yet. - /// Backstop only — the entry stays in storage; eviction is the monitor's - /// job (`reap_group`). - fn members_where(&self, group: &str, mut is_live: impl FnMut(Pid) -> bool) -> Vec { + /// Live members of `group` **on `node`**, in insertion order. The + /// `is_live` oracle is the read-path backstop: a member whose slot is + /// already dead is dropped from the *result* even if the reaper has not + /// swept it yet. Backstop only — the entry stays in storage; eviction is + /// the reaper's job. The node filter is what keeps the local API local: + /// a remote member's `pid` is another node's slot bits, meaningless to + /// `is_live` and to any local send. + fn members_where( + &self, + group: &str, + node: NodeId, + mut is_live: impl FnMut(Pid) -> bool, + ) -> Vec { self.groups .get(group) .map(|v| { v.iter() + .filter(|e| e.member.node == node) .map(|e| e.member.pid) .filter(|&p| is_live(p)) .collect() @@ -334,19 +333,210 @@ impl ProcessGroups { .unwrap_or_default() } - /// The first live member of `group` in insertion order — stateless - /// first-live `pick`, with the same read-path backstop as `members_where`. - fn first_member_where(&self, group: &str, mut is_live: impl FnMut(Pid) -> bool) -> Option { + /// The first live member of `group` on `node` in insertion order — + /// stateless first-live `pick`, with the same read-path backstop and node + /// filter as `members_where`. + fn first_member_where( + &self, + group: &str, + node: NodeId, + mut is_live: impl FnMut(Pid) -> bool, + ) -> Option { self.groups .get(group)? .iter() + .filter(|e| e.member.node == node) .map(|e| e.member.pid) .find(|&p| is_live(p)) } } +/// The store's cluster-side surface: raw reads the pg actor needs to speak +/// for this node (`Sync`, membership checks) and the peer-name memo. One +/// `cfg` block: everything here exists only when there is a mesh. +#[cfg(feature = "cluster")] +impl ProcessGroups { + /// Does `group` hold `member` right now? (Raw storage, no liveness.) + pub(crate) fn contains(&self, group: &str, member: &Member) -> bool { + self.groups + .get(group) + .is_some_and(|v| v.iter().any(|e| e.member == *member)) + } + + /// Every stored member of `group`, any node, insertion order. Raw storage. + pub(crate) fn all_of(&self, group: &str) -> Vec { + self.groups + .get(group) + .map(|v| v.iter().map(|e| e.member).collect()) + .unwrap_or_default() + } + + /// `(group, [pid])` for every group with a member on `node` — the + /// `Sync` payload. Raw storage; groups with no such member are omitted. + pub(crate) fn groups_on(&self, node: NodeId) -> Vec<(String, Vec)> { + let mut out: Vec<(String, Vec)> = self + .groups + .iter() + .filter_map(|(g, v)| { + let pids: Vec = v + .iter() + .filter(|e| e.member.node == node) + .map(|e| e.member.pid) + .collect(); + (!pids.is_empty()).then(|| (g.clone(), pids)) + }) + .collect(); + out.sort_by(|a, b| a.0.cmp(&b.0)); + out + } + + /// Record / forget the name behind a peer's `NodeId`. + pub(crate) fn set_node_name(&mut self, node: NodeId, name: String) { + self.node_names.insert(node, name); + } + pub(crate) fn forget_node_name(&mut self, node: NodeId) { + self.node_names.remove(&node); + } + pub(crate) fn node_name(&self, node: NodeId) -> Option<&str> { + self.node_names.get(&node).map(String::as_str) + } +} + +/// The group reaper: one detached actor per run, spawned by the first `join`, +/// parked on the shared `deaths` inbox. Every local membership's monitor +/// delivers here, so a death is swept out of *every* group it joined as soon +/// as the reaper is scheduled — no group operation has to happen first. +/// Sweeps by `(node, pid)`: only local members, since a remote member's pid +/// bits are meaningless here. Exits when the last sender is gone, i.e. never +/// during a run (the store holds one); the run's teardown stops it like any +/// other parked actor. Spawned under `ROOT_PID` so its exit signal is absorbed +/// rather than delivered to whichever supervisor's child happened to join +/// first. +/// +/// Under `cluster` the same actor is the node's **pg actor** (RFC 010 Phase +/// 5, c15): it also drains a control inbox of local join/leave announcements, +/// the membership stream and the exposed `"pg"` inbox — see +/// [`crate::cluster::pg`]. Its store-side sweep is unchanged. +#[cfg(not(feature = "cluster"))] +fn reaper(rx: crate::channel::Receiver, ctl: crate::channel::Receiver) { + // No mesh: nothing to tell about joins/leaves. Drop the control inbox + // so announcements are refused at the sender rather than queued. + drop(ctl); + while let Ok(down) = rx.recv() { + sweep_local_death(down.pid); + // Evicted memberships hold only ids; their monitors have fired. + } +} + +/// What the local API tells the reaper besides deaths (which arrive as +/// [`Down`] on their own inbox — that channel's type is fixed by the monitor +/// primitive, so the two cannot be one enum). The default reaper has no use +/// for these; the cluster's pg actor broadcasts them (RFC 010 Phase 5). +// The default reaper never looks inside — that is the point, not a bug. +#[cfg_attr(not(feature = "cluster"), allow(dead_code))] +pub(crate) enum PgEvent { + /// `join` inserted `pid` into `group`. The consumer re-checks the store + /// before acting on it. + Joined { group: String, pid: Pid }, + /// `leave` removed `pid` from `group`. + Left { group: String, pid: Pid }, + /// `cluster::start` has the manager up and the local identity set: take + /// a membership subscription, register + expose the `"pg"` name, and + /// start speaking to peers. + #[cfg(feature = "cluster")] + Attach, +} + +/// Evict the local member `pid` from every group. The reaper's one store +/// operation; returns what was evicted with its group (the cluster's +/// `Leave` broadcast wants both). +pub(crate) fn sweep_local_death(pid: Pid) -> Vec<(String, Membership)> { + with_runtime(|inner| { + let node = inner.node_id; + inner + .process_groups + .lock() + .remove_where(|m| m.node == node && m.pid == pid) + }) +} + +/// The reaper's inboxes. `deaths` is the liveness authority for the set +/// (`ctl` is created and dropped with it, on the same actor). +#[derive(Clone)] +pub(crate) struct ReaperInboxes { + pub(crate) deaths: Sender, + /// The control inbox: local `join`/`leave` announce here (see + /// [`PgEvent`]). The default reaper closes it on entry. + pub(crate) ctl: Sender, +} + +impl ReaperInboxes { + fn alive(&self) -> bool { + self.deaths.receiver_alive() + } +} + +/// Live senders for the reaper's inboxes, spawning the reaper if this run has +/// none yet. Two racing first-spawns may both spawn; the loser's senders drop +/// on return, its spare reaper sees a closed inbox and exits. +pub(crate) fn reaper_inboxes() -> ReaperInboxes { + let existing = with_runtime(|inner| { + let pg = inner.process_groups.lock(); + pg.reaper.clone().filter(ReaperInboxes::alive) + }); + if let Some(r) = existing { + return r; + } + let (tx, rx) = channel::(); + let (ctl_tx, ctl_rx) = channel::(); + // Detached: the handle drops here. The reaper's lifetime is the run's. + // The ONE seam between the local store and the cluster: same inboxes, + // different body. + #[cfg(not(feature = "cluster"))] + let _ = spawn_under(crate::runtime::ROOT_PID, move || reaper(rx, ctl_rx)); + #[cfg(feature = "cluster")] + let _ = spawn_under(crate::runtime::ROOT_PID, move || { + crate::cluster::pg::actor(rx, ctl_rx) + }); + let fresh = ReaperInboxes { + deaths: tx, + ctl: ctl_tx, + }; + with_runtime(|inner| { + let mut pg = inner.process_groups.lock(); + match &pg.reaper { + Some(r) if r.alive() => r.clone(), + _ => { + pg.reaper = Some(fresh.clone()); + fresh + } + } + }) +} + +/// A live sender for the reaper's `deaths` inbox (spawning it if needed). +fn deaths_sender() -> Sender { + reaper_inboxes().deaths +} + +/// Announce a local group change to the reaper, if this run has one. A +/// closed inbox is the default reaper (uninterested) or a run tearing down. +fn announce(msg: PgEvent) { + let ctl = with_runtime(|inner| { + inner + .process_groups + .lock() + .reaper + .as_ref() + .map(|r| r.ctl.clone()) + }); + if let Some(ctl) = ctl { + let _ = ctl.send(msg); + } +} + /// Build the full member identity for `pid` from runtime identity. -fn member_for(inner: &crate::runtime::RuntimeInner, pid: Pid) -> Member { +pub(crate) fn member_for(inner: &crate::runtime::RuntimeInner, pid: Pid) -> Member { Member { node: inner.node_id, incarnation: inner.incarnation, @@ -358,7 +548,7 @@ fn member_for(inner: &crate::runtime::RuntimeInner, pid: Pid) -> Member { /// no lock — identical to the registry's guard. The read-path backstop: a /// generation is never reused, so a dead member is detectable independently of /// whether its monitor `Down` has been drained yet. -fn live(inner: &crate::runtime::RuntimeInner, pid: Pid) -> bool { +pub(crate) fn live(inner: &crate::runtime::RuntimeInner, pid: Pid) -> bool { inner.slot_at(pid).is_some_and(|s| s.is_live_for(pid)) } @@ -367,61 +557,76 @@ fn live(inner: &crate::runtime::RuntimeInner, pid: Pid) -> bool { /// added the membership, `false` if it was already a member. /// /// Installs a monitor on `pid` so the actor's death evicts it from the group -/// automatically — you never have to remove a dead member yourself. A redundant -/// (idempotent) join tears its extra monitor back down. +/// automatically — you never have to remove a dead member yourself. Joining a +/// pid that is already dead is accepted and evicted the same way (via a +/// `NoProc` notice), so it never shows up in a read. /// /// Panics if called outside `Runtime::run()`. pub fn join(group: impl Into, pid: Pid) -> bool { let group = group.into(); let pid = pid.erase(); - // Install the monitor BEFORE taking the group lock: monitor() acquires the - // target's cold lock (Leaf), and two Leaf locks are never held at once. The - // registration races `finalize_actor` under that cold lock exactly as every - // other monitor does, so no death can slip between the join and the monitor - // being in place. - let mon = monitor(pid); - - let (rejected, reaped) = with_runtime(|inner| { + let deaths = deaths_sender(); + // Record the membership BEFORE arming its monitor: the reaper sweeps by + // pid on the first `Down`, so a `Down` that could precede the entry would + // leave a corpse in storage forever (visible to no read — the backstop + // hides it — but a leak, and once groups are clustered a member that + // would be announced). Arming after insertion means every `Down` finds + // its entry. The monitor id is allocated up front so `leave` can tear the + // registration down even if it lands in the tiny window before arming (an + // orphaned registration is harmless: its `Down` names a pid whose + // membership is gone, and the sweep finds nothing). + let id = with_runtime(|inner| inner.alloc_monitor_id()); + let inserted = with_runtime(|inner| { let ms = Membership { member: member_for(inner, pid), - monitor: mon, + monitor: Some(id), }; - let mut pg = inner.process_groups.lock(); - let reaped = pg.reap_group(&group); - let rejected = pg.join(&group, ms); - (rejected, reaped) + inner.process_groups.lock().join(&group, ms) + }); + if !inserted { + return false; + } + // Tell the reaper (the cluster's pg actor re-checks the store before it + // broadcasts, so a `leave`/death that overtakes this announcement is + // never advertised as a join). + announce(PgEvent::Joined { + group: group.clone(), + pid, }); - // Outside the group lock: drop the reaped (dead) monitors, and if this join - // was redundant, demonitor + drop the extra monitor we just installed. - drop(reaped); - match rejected { - Some(dup) => { - demonitor(&dup.monitor); - false - } - None => true, + // Outside the group lock: registration takes the target's cold lock (Leaf). + // The registration races `finalize_actor` under that cold lock exactly as + // every other monitor does, so no death can slip between the join and the + // monitor being in place. + if !register_monitor(pid, id, &deaths) { + // Already gone: queue the notice ourselves, exactly as `monitor` does. + let _ = deaths.send(Down { + pid, + reason: DownReason::NoProc, + }); } + true } /// Drop `pid`'s membership of `group`. Returns whether a membership was -/// removed. The membership's monitor is demonitored and dropped. +/// removed. The membership's monitor registration is torn down. /// /// Panics if called outside `Runtime::run()`. pub fn leave(group: &str, pid: Pid) -> bool { let pid = pid.erase(); - let (removed, reaped) = with_runtime(|inner| { + let removed = with_runtime(|inner| { let member = member_for(inner, pid); - let mut pg = inner.process_groups.lock(); - let reaped = pg.reap_group(group); - let removed = pg.leave(group, member); - (removed, reaped) + inner.process_groups.lock().leave(group, member) }); - - drop(reaped); match removed { Some(ms) => { - demonitor(&ms.monitor); + if let Some(id) = ms.monitor { + unregister_monitor(pid, id); + } + announce(PgEvent::Left { + group: group.to_owned(), + pid, + }); true } None => false, @@ -431,21 +636,19 @@ pub fn leave(group: &str, pid: Pid) -> bool { /// Every live member of `group`, in the order they joined. Returns an empty /// vector if the group does not exist or has no live members. /// -/// Dead members are never returned: the group is pruned of anything that has -/// died before the read, and as a backstop a member whose slot is already dead -/// is dropped from the result even in the brief window before its death has -/// been fully processed. +/// Dead members are never returned: the reaper evicts a member as soon as its +/// death is processed, and as a backstop a member whose slot is already dead +/// is dropped from the result even in the brief window before the reaper's +/// turn. /// /// Panics if called outside `Runtime::run()`. pub fn members(group: &str) -> Vec { - let (pids, reaped) = with_runtime(|inner| { - let mut pg = inner.process_groups.lock(); - let reaped = pg.reap_group(group); - let pids = pg.members_where(group, |pid| live(inner, pid)); - (pids, reaped) - }); - drop(reaped); - pids + with_runtime(|inner| { + inner + .process_groups + .lock() + .members_where(group, inner.node_id, |pid| live(inner, pid)) + }) } /// One live member of `group`, or `None` if the group is empty (or every @@ -455,14 +658,12 @@ pub fn members(group: &str) -> Vec { /// /// Panics if called outside `Runtime::run()`. pub fn pick(group: &str) -> Option { - let (picked, reaped) = with_runtime(|inner| { - let mut pg = inner.process_groups.lock(); - let reaped = pg.reap_group(group); - let picked = pg.first_member_where(group, |pid| live(inner, pid)); - (picked, reaped) - }); - drop(reaped); - picked + with_runtime(|inner| { + inner + .process_groups + .lock() + .first_member_where(group, inner.node_id, |pid| live(inner, pid)) + }) } /// Typed [`pick`]: one live member of `group` as a [`Pid`](Pid). @@ -505,8 +706,8 @@ pub fn dispatch(group: &str, msg: A::Msg) -> Result, Send #[cfg(test)] mod tests { use super::*; - use crate::channel::{channel, Sender}; - use crate::monitor::{Down, DownReason, MonitorId}; + use crate::scheduler::spawn; + use std::time::{Duration, Instant}; fn member(index: u32, generation: u32) -> Member { Member { @@ -516,33 +717,21 @@ mod tests { } } - /// A synthetic membership with a real (but slot-less) monitor channel. The - /// returned `Sender` stands in for the slot's `Down` sender: hold it to - /// keep the member "alive" (`try_recv` → `Ok(None)`), `send` a `Down` to - /// simulate death, or `drop` it to simulate a drained/closed channel. - fn synth(index: u32, generation: u32) -> (Membership, Sender) { - let pid = Pid::new(index, generation); - let (tx, rx) = channel::(); - let ms = Membership { + /// A synthetic membership: the store never looks at the id. + fn synth(index: u32, generation: u32) -> Membership { + Membership { member: member(index, generation), - monitor: Monitor { - id: MonitorId(0), - target: pid, - rx, - }, - }; - (ms, tx) + monitor: Some(MonitorId(0)), + } } #[test] fn join_is_idempotent_within_a_group() { let mut pg = ProcessGroups::new(); - let (a, _ta) = synth(1, 0); - let (b, _tb) = synth(1, 0); - assert!(pg.join("workers", a).is_none(), "first join inserts"); + assert!(pg.join("workers", synth(1, 0)), "first join inserts"); assert!( - pg.join("workers", b).is_some(), - "second identical join is handed back" + !pg.join("workers", synth(1, 0)), + "second identical join is refused" ); assert_eq!(pg.members_of("workers"), vec![member(1, 0)]); } @@ -550,12 +739,9 @@ mod tests { #[test] fn same_pid_in_many_groups_is_independent() { let mut pg = ProcessGroups::new(); - let (a, _ta) = synth(1, 0); - let (b, _tb) = synth(1, 0); - let (c, _tc) = synth(2, 0); - pg.join("a", a); - pg.join("b", b); - pg.join("b", c); + pg.join("a", synth(1, 0)); + pg.join("b", synth(1, 0)); + pg.join("b", synth(2, 0)); assert_eq!(pg.members_of("a"), vec![member(1, 0)]); assert_eq!(pg.members_of("b"), vec![member(1, 0), member(2, 0)]); } @@ -564,11 +750,9 @@ mod tests { fn distinct_generations_are_distinct_members() { // ABA guard: same slot index, different generation = different actor. let mut pg = ProcessGroups::new(); - let (a, _ta) = synth(1, 0); - let (b, _tb) = synth(1, 1); - assert!(pg.join("g", a).is_none()); + assert!(pg.join("g", synth(1, 0))); assert!( - pg.join("g", b).is_none(), + pg.join("g", synth(1, 1)), "different generation is a distinct member" ); assert_eq!(pg.members_of("g"), vec![member(1, 0), member(1, 1)]); @@ -577,10 +761,8 @@ mod tests { #[test] fn leave_removes_one_membership_and_prunes_empty_groups() { let mut pg = ProcessGroups::new(); - let (a, _ta) = synth(1, 0); - let (b, _tb) = synth(2, 0); - pg.join("g", a); - pg.join("g", b); + pg.join("g", synth(1, 0)); + pg.join("g", synth(2, 0)); assert!(pg.leave("g", member(1, 0)).is_some()); assert_eq!(pg.members_of("g"), vec![member(2, 0)]); assert!( @@ -598,7 +780,7 @@ mod tests { #[test] fn remove_where_sweeps_every_group() { let mut pg = ProcessGroups::new(); - for (g, (m, _t)) in [ + for (g, m) in [ ("a", synth(1, 0)), ("a", synth(2, 0)), ("b", synth(1, 0)), @@ -616,104 +798,137 @@ mod tests { #[test] fn remove_where_can_match_an_incarnation_sweep() { - // Shape check for the later evict_incarnation(node, inc) caller. + // Shape check for the node-down / incarnation sweep caller. let mut pg = ProcessGroups::new(); - let pid = Pid::new(1, 0); - let (tx, rx) = channel::(); - let dead = Membership { + let stale = Membership { member: Member { node: DEFAULT_NODE_ID, incarnation: Incarnation::new(7), - pid, - }, - monitor: Monitor { - id: MonitorId(0), - target: pid, - rx, + pid: Pid::new(1, 0), }, + monitor: Some(MonitorId(0)), }; - let _keep = tx; - let (live, _tl) = synth(2, 0); - pg.join("g", dead); - pg.join("g", live); + pg.join("g", stale); + pg.join("g", synth(2, 0)); let evicted = pg.remove_where(|mem| mem.incarnation == Incarnation::new(7)); assert_eq!(evicted.len(), 1); assert_eq!(pg.members_of("g"), vec![member(2, 0)]); } #[test] - fn reap_keeps_live_members() { + fn read_backstop_hides_a_member_the_reaper_has_not_yet_swept() { let mut pg = ProcessGroups::new(); - let (a, _ta) = synth(1, 0); // sender held: member stays alive - pg.join("a", a); - assert!(pg.reap_group("a").is_empty(), "no deaths"); - assert_eq!(pg.members_of("a"), vec![member(1, 0)]); - } - - #[test] - fn reap_evicts_a_dead_member_and_sweeps_all_its_groups() { - let mut pg = ProcessGroups::new(); - let (a1, ta1) = synth(1, 0); // pid 1 in group a - let (a2, _ta2) = synth(2, 0); // pid 2 in group a (stays alive) - let (b1, _tb1) = synth(1, 0); // pid 1 in group b - pg.join("a", a1); - pg.join("a", a2); - pg.join("b", b1); - // pid 1 dies: its group-a monitor receives a Down. Its group-b monitor - // has not — reap must still sweep pid 1 out of b by the pid predicate. - ta1.send(Down { - pid: Pid::new(1, 0), - reason: DownReason::Exit, - }) - .unwrap(); - let evicted = pg.reap_group("a"); - assert_eq!( - evicted.len(), - 2, - "pid 1's memberships in both a and b are evicted" - ); - assert_eq!(pg.members_of("a"), vec![member(2, 0)]); - assert!(pg.members_of("b").is_empty(), "swept from b too; pruned"); - } - - #[test] - fn reap_treats_a_closed_channel_as_dead() { - let mut pg = ProcessGroups::new(); - let (a, ta) = synth(1, 0); - pg.join("a", a); - drop(ta); // sender gone, queue empty → try_recv = Err(RecvError) = dead - let evicted = pg.reap_group("a"); - assert_eq!(evicted.len(), 1); - assert!(pg.members_of("a").is_empty()); - } - - #[test] - fn read_backstop_hides_a_member_the_monitor_has_not_yet_reaped() { - let mut pg = ProcessGroups::new(); - // Both senders held: reap_group would see Ok(None) and evict neither. - let (a, _ta) = synth(1, 0); - let (b, _tb) = synth(2, 0); - pg.join("g", a); - pg.join("g", b); + pg.join("g", synth(1, 0)); + pg.join("g", synth(2, 0)); // The slot-word oracle already reports pid 1 dead (finalize window), - // ahead of any Down delivery. + // ahead of the reaper's turn. let dead = Pid::new(1, 0); let oracle = |pid: Pid| pid != dead; assert_eq!( - pg.members_where("g", oracle), + pg.members_where("g", DEFAULT_NODE_ID, oracle), vec![Pid::new(2, 0)], "dead pid filtered from read" ); assert_eq!( - pg.first_member_where("g", oracle), + pg.first_member_where("g", DEFAULT_NODE_ID, oracle), Some(Pid::new(2, 0)), "pick skips the dead first member" ); - // Backstop does not evict — that stays the monitor's job; raw storage - // still holds both until reap runs. + // Backstop does not evict — that stays the reaper's job; raw storage + // still holds both until it runs. assert_eq!(pg.members_of("g"), vec![member(1, 0), member(2, 0)]); } + + // ---- reaper: eager eviction against a live runtime ---- + + /// Raw storage view for a group, bypassing the read-path backstop. + fn stored(group: &str) -> Vec { + with_runtime(|inner| inner.process_groups.lock().members_of(group)) + } + + /// Cooperative wait (`smarm::sleep`, never an OS block) until `pred`. + fn wait_until(what: &str, mut pred: impl FnMut() -> bool) { + let deadline = Instant::now() + Duration::from_secs(2); + while !pred() { + assert!(Instant::now() < deadline, "timed out waiting for: {what}"); + crate::sleep(Duration::from_millis(1)); + } + } + + #[test] + fn a_death_is_swept_from_storage_without_any_group_operation() { + crate::run(|| { + let (tx, rx) = channel::<()>(); + let w = spawn(move || { + rx.recv().unwrap(); + }); + let pid = w.pid(); + join("a", pid); + join("b", pid); + assert_eq!(stored("a"), vec![member_for_test(pid)]); + + tx.send(()).unwrap(); + w.join().unwrap(); + // No members()/pick()/join() on a or b from here on: the reaper + // alone must clear both. + wait_until("reaper sweeps a and b", || { + stored("a").is_empty() && stored("b").is_empty() + }); + }); + } + + #[test] + fn a_dead_at_join_pid_is_swept_from_storage() { + crate::run(|| { + let h = spawn(|| {}); + let pid = h.pid(); + h.join().unwrap(); + assert!(join("late", pid), "join is accepted; eviction is uniform"); + wait_until("reaper sweeps the NoProc member", || { + stored("late").is_empty() + }); + }); + } + + #[test] + fn leave_then_death_does_not_disturb_a_rejoined_group() { + // A monitor unregistered by `leave` must not fire later; the pid's + // fresh membership after re-join is swept exactly once, by its own + // monitor, on death. + crate::run(|| { + let (tx, rx) = channel::<()>(); + let w = spawn(move || { + rx.recv().unwrap(); + }); + let pid = w.pid(); + join("g", pid); + assert!(leave("g", pid)); + assert!(join("g", pid)); + assert_eq!(members("g"), vec![pid]); + tx.send(()).unwrap(); + w.join().unwrap(); + wait_until("reaper sweeps g", || stored("g").is_empty()); + }); + } + + #[test] + fn reaper_is_respawned_for_a_second_run_of_the_same_runtime() { + let rt = crate::runtime::init(crate::runtime::Config::exact(1)); + let body = || { + let h = spawn(|| {}); + let pid = h.pid(); + h.join().unwrap(); + join("g", pid); + wait_until("reaper sweeps g", || stored("g").is_empty()); + }; + rt.run(body); + rt.run(body); + } + + fn member_for_test(pid: Pid) -> Member { + with_runtime(|inner| member_for(inner, pid)) + } } diff --git a/src/pid.rs b/src/pid.rs index 96b3f21..2f5c0fb 100644 --- a/src/pid.rs +++ b/src/pid.rs @@ -292,3 +292,44 @@ mod typed_pid_tests { assert_send_sync::>(); } } + +// ---- RFC 010 c10: pids auto-serialize (cluster feature) --------------------- + +/// A local `Pid` serializes as a +/// [`RemotePid`](crate::cluster::remote::RemotePid): the wire form stamps +/// this node's name and incarnation from the ambient runtime, so a pid can +/// sit inside any message field and reply-to needs no ceremony (RFC 010 §3, +/// "sugar not a bear trap"). Serializing a pid also marks it **watchable** +/// — the wire crossing is the cluster's `mark_watchable` set-site (D12), the +/// exact analog of the membrane crossing. +/// +/// Must run inside `run()` (the ambient identity lives on the runtime); a +/// runtime without a cluster identity cannot serialize a pid at all — it is +/// a serialize error, surfacing as the send's `Encode` failure — rather than +/// a `("", 0)` stamp that every peer would silently drop. +#[cfg(feature = "cluster")] +impl serde::Serialize for Pid { + fn serialize(&self, s: S) -> Result { + crate::cluster::remote::RemotePid::::from_local(*self) + .ok_or_else(|| serde::ser::Error::custom("pid serialized with no local node identity"))? + .serialize(s) + } +} + +/// Deserializing into a `Pid` is the **collapse**: it succeeds only when +/// the wire pid names this very node (name and incarnation both), and is a +/// decode error otherwise — a foreign pid cannot become a local `Pid`. +/// Fields that may hold a pid from anywhere are `RemotePid`. +#[cfg(feature = "cluster")] +impl<'de, A: 'static> serde::Deserialize<'de> for Pid { + fn deserialize>(d: D) -> Result { + let rp = crate::cluster::remote::RemotePid::::deserialize(d)?; + rp.local().ok_or_else(|| { + serde::de::Error::custom(format!( + "pid {}@{} is not local to this node", + rp.index(), + rp.node() + )) + }) + } +} diff --git a/src/runtime.rs b/src/runtime.rs index 911114c..6cd4f82 100644 --- a/src/runtime.rs +++ b/src/runtime.rs @@ -1358,6 +1358,9 @@ impl Runtime { self.inner.root_exited.store(false, Ordering::Relaxed); self.inner.root_swept.store(false, Ordering::Relaxed); self.inner.set_root(initial_handle.pid()); + // A previous run's group reaper was stopped with that run; forget it + // so the first `join` of this run spawns a fresh one. + self.inner.process_groups.lock().reset_reaper(); // Launch N-1 extra scheduler threads, named `smarm-sched-{slot}` so // they are identifiable in `/proc//task/*/comm`, stack dumps and diff --git a/src/trace.rs b/src/trace.rs index 923cbbd..180828e 100644 --- a/src/trace.rs +++ b/src/trace.rs @@ -65,6 +65,13 @@ mod inner { // RFC 005 wake slot SlotPush(Pid), // actor-context wake parked in the waking thread's slot SlotPop(Pid), // scheduler resumed a pid from its own slot + // Cluster (RFC 010): the conn actor's verdict on one inbound frame — + // local knowledge only, never on the wire; the label is + // `InboundVerdict::label()`. No pid: a refused frame has none. + ClusterInbound(&'static str), + // Cluster (RFC 010): the connector's verdict on one dial attempt — + // `"ok"` or `DialError::label()`. No pid. + ClusterDial(&'static str), } // ----------------------------------------------------------------------- @@ -271,6 +278,8 @@ mod inner { Event::Dequeue(p) => ("dequeue".into(), p.index()), Event::SlotPush(p) => ("slot_push".into(), p.index()), Event::SlotPop(p) => ("slot_pop".into(), p.index()), + Event::ClusterInbound(v) => (format!("cluster_inbound {v}"), 0), + Event::ClusterDial(v) => (format!("cluster_dial {v}"), 0), } } diff --git a/tests/channel.rs b/tests/channel.rs index cc92e95..e3bc1d4 100644 --- a/tests/channel.rs +++ b/tests/channel.rs @@ -137,11 +137,18 @@ fn channel_ops_interleaved_with_monitor_churn_multi_thread() { for i in 0..32i64 { let tx = tx.clone(); handles.push(spawn(move || { - // Short-lived target whose death fires the monitor below. + // Short-lived target whose death fires the monitor below. It + // is gated: on a multi-thread scheduler it could otherwise + // run and exit before `monitor` registers, and monitoring a + // corpse queues `NoProc` by contract — the point here is a + // `Down` sent from finalize, so register first, then release. + let (go_tx, go_rx) = channel::<()>(); let t = spawn(move || { + let _ = go_rx.recv(); tx.send(i).unwrap(); }); let m = smarm::monitor(t.pid()); + go_tx.send(()).unwrap(); t.join().unwrap(); // Down delivery exercises send-from-finalize. let d = m.rx.recv().unwrap(); diff --git a/tests/cluster_conn_lifecycle.rs b/tests/cluster_conn_lifecycle.rs index eff1b93..f989ddc 100644 --- a/tests/cluster_conn_lifecycle.rs +++ b/tests/cluster_conn_lifecycle.rs @@ -21,6 +21,7 @@ use smarm::cluster::manager::{Call, Manager, Reply, MANAGER}; use smarm::cluster::spawn_established; use smarm::cluster::transport::tcp::TcpTransport; use smarm::cluster::transport::{Conn, FramedConn, Transport}; +use smarm::cluster::Timing; use smarm::gen_server::{self, GenServerBuilder}; use smarm::pg::Incarnation; use smarm::{run, sleep}; @@ -81,8 +82,10 @@ fn connection_up_commanded_shutdown_and_eof_all_reflected_in_table() { // Manage the `a` ends as peers node-b and node-c; keep the `b` far ends // open so neither socket is closed from the far side yet. - spawn_established(FramedConn::new(a1), peer("node-b")).expect("node-b registers"); - spawn_established(FramedConn::new(a2), peer("node-c")).expect("node-c registers"); + spawn_established(FramedConn::new(a1), peer("node-b"), Timing::default()) + .expect("node-b registers"); + spawn_established(FramedConn::new(a2), peer("node-c"), Timing::default()) + .expect("node-c registers"); // Up: both connections register and the table shows them. wait_peers(&["node-b", "node-c"]); diff --git a/tests/cluster_conn_liveness.rs b/tests/cluster_conn_liveness.rs index ba2aea0..5104efc 100644 --- a/tests/cluster_conn_liveness.rs +++ b/tests/cluster_conn_liveness.rs @@ -22,6 +22,7 @@ use smarm::cluster::manager::{Call, Manager, Reply, MANAGER}; use smarm::cluster::spawn_established; use smarm::cluster::transport::tcp::TcpTransport; use smarm::cluster::transport::{Conn, FramedConn, Transport}; +use smarm::cluster::Timing; use smarm::gen_server::{self, GenServerBuilder}; use smarm::pg::Incarnation; use smarm::{run, sleep, spawn}; @@ -83,7 +84,8 @@ fn heartbeats_are_sent_unprompted() { .expect("manager name is free"); let (a, b) = pair(&TcpTransport); - spawn_established(FramedConn::new(a), peer("hb-send")).expect("register"); + spawn_established(FramedConn::new(a), peer("hb-send"), Timing::default()) + .expect("register"); let mut far = FramedConn::new(b); let frame = far @@ -111,7 +113,7 @@ fn mute_peer_is_torn_down_after_liveness_timeout() { .expect("manager name is free"); let (a, b) = pair(&TcpTransport); - spawn_established(FramedConn::new(a), peer("mute")).expect("register"); + spawn_established(FramedConn::new(a), peer("mute"), Timing::default()).expect("register"); // Held open and silent: no frames, no EOF. (Unread inbound // heartbeats sit in kernel buffers; they are 5 bytes each.) let _far = FramedConn::new(b); @@ -139,7 +141,7 @@ fn heartbeats_keep_the_connection_alive() { .expect("manager name is free"); let (a, b) = pair(&TcpTransport); - spawn_established(FramedConn::new(a), peer("kept")).expect("register"); + spawn_established(FramedConn::new(a), peer("kept"), Timing::default()).expect("register"); // The far heartbeat pump: interval-paced sends until told to stop, // then holds the socket open, silent, so the eventual teardown is diff --git a/tests/cluster_connect.rs b/tests/cluster_connect.rs index 405ddc4..20286da 100644 --- a/tests/cluster_connect.rs +++ b/tests/cluster_connect.rs @@ -19,11 +19,12 @@ use smarm::cluster::connect::{ HANDSHAKE_TIMEOUT, }; use smarm::cluster::envelope::{Frame, NodeMeta, RejectReason}; -use smarm::cluster::handshake::{HelloCtx, Local}; +use smarm::cluster::handshake::{Local, PeerStanding}; use smarm::cluster::manager::{Call, Manager, Reply, MANAGER}; use smarm::cluster::transport::loopback::LoopbackTransport; use smarm::cluster::transport::tcp::TcpTransport; use smarm::cluster::transport::{FramedConn, Transport}; +use smarm::cluster::Timing; use smarm::gen_server::{self, GenServerBuilder}; use smarm::pg::Incarnation; use smarm::{run, sleep}; @@ -100,7 +101,7 @@ fn loopback_happy_path_establishes_both_ends() { local("node-b"), |name| { assert_eq!(name, "node-a"); - HelloCtx::default() + PeerStanding::Free }, no_deadline(), ) @@ -118,7 +119,7 @@ fn loopback_hash_mismatch_rejected_with_frame_then_eof() { let mut wrong = local("node-b"); wrong.build_hash ^= 1; let responder = std::thread::spawn(move || { - accept_handshake(&mut accepted, wrong, |_| HelloCtx::default(), no_deadline()) + accept_handshake(&mut accepted, wrong, |_| PeerStanding::Free, no_deadline()) }); // The dial side receives the reject frame — the compatibility anchor. match dial_handshake(&mut dialer, &local("node-a"), no_deadline()) { @@ -142,10 +143,7 @@ fn loopback_tie_break_loser_closed_silently() { accept_handshake( &mut accepted, local("node-a"), - |_| HelloCtx { - name_claimed: false, - dialing_this_peer: true, - }, + |_| PeerStanding::Dialing, no_deadline(), ) }); @@ -178,7 +176,7 @@ fn loopback_read_ahead_past_hello_survives_into_established_conn() { let peer = accept_handshake( &mut accepted, local("node-b"), - |_| HelloCtx::default(), + |_| PeerStanding::Free, no_deadline(), ) .unwrap(); @@ -216,7 +214,7 @@ fn tcp_silent_peer_times_out_on_the_accept_path() { let r = accept_handshake( &mut accepted, local("node-b"), - |_| HelloCtx::default(), + |_| PeerStanding::Free, Instant::now() + Duration::from_millis(200), ); let _ = tx.send(r); @@ -253,7 +251,7 @@ fn tcp_duplicate_name_rejected_by_acceptor() { .start() .expect("manager name is free"); let listener = TcpTransport.listen("127.0.0.1:0").unwrap(); - let acceptor = spawn_acceptor(listener, local("node-b")); + let acceptor = spawn_acceptor(listener, local("node-b"), Timing::default()); let addr = acceptor.local_addr().to_string(); // First dial offering "dup-node": establishes and registers. @@ -328,12 +326,12 @@ fn dial_intent_cleared_when_dialer_dies() { // While the dialer lives, the intent is visible. match gen_server::call( MANAGER, - Call::HelloCtx { + Call::Standing { peer_name: "ghost".into(), }, ) { - Ok(Reply::HelloCtx(ctx)) => assert!(ctx.dialing_this_peer), - other => panic!("HelloCtx failed: {other:?}"), + Ok(Reply::Standing(s)) => assert_eq!(s, PeerStanding::Dialing), + other => panic!("PeerStanding failed: {other:?}"), } // Kill it; the monitor must clear the intent without cooperation. go_tx.send(()).unwrap(); @@ -341,11 +339,11 @@ fn dial_intent_cleared_when_dialer_dies() { loop { match gen_server::call( MANAGER, - Call::HelloCtx { + Call::Standing { peer_name: "ghost".into(), }, ) { - Ok(Reply::HelloCtx(ctx)) if !ctx.dialing_this_peer => break, + Ok(Reply::Standing(s)) if s != PeerStanding::Dialing => break, _ if Instant::now() > deadline => { panic!("dial intent not cleared after dialer death") } @@ -372,7 +370,7 @@ fn role_hs_listener() { .start() .expect("manager name is free"); let listener = TcpTransport.listen("127.0.0.1:0").unwrap(); - let acceptor = spawn_acceptor(listener, local("node-b")); + let acceptor = spawn_acceptor(listener, local("node-b"), Timing::default()); println!("LISTENING {}", acceptor.local_addr()); wait_peers(&["node-a"]); println!("PEERS node-a"); @@ -389,7 +387,13 @@ fn role_hs_dialer() { .expect("manager name is free"); let (tx, rx) = mpsc::channel(); smarm::spawn(move || { - let r = dial(&TcpTransport, &addr, "node-b", &local("node-a")); + let r = dial( + &TcpTransport, + &addr, + "node-b", + &local("node-a"), + Timing::default(), + ); let _ = tx.send(r); }); if let Err(e) = poll_recv(&rx, "dial outcome") { @@ -459,7 +463,13 @@ fn concurrent_dial_to_same_name_refused() { // (the addr is unroutable on purpose — it must never be dialed). let (tx, rx) = mpsc::channel(); smarm::spawn(move || { - let r = dial(&TcpTransport, "127.0.0.1:1", "node-x", &local("node-a")); + let r = dial( + &TcpTransport, + "127.0.0.1:1", + "node-x", + &local("node-a"), + Timing::default(), + ); let _ = tx.send(r); }); match poll_recv(&rx, "second dial outcome") { diff --git a/tests/cluster_dial_mismatch.rs b/tests/cluster_dial_mismatch.rs new file mode 100644 index 0000000..03cc754 --- /dev/null +++ b/tests/cluster_dial_mismatch.rs @@ -0,0 +1,115 @@ +//! RFC 010 — a seed whose address answers as a *different* name +//! (`DialError::PeerNameMismatch`) is dialed once and then parked: the +//! connector must not redial it on backoff forever. +//! +//! Observed from the misdialed peer: each such dial establishes at the +//! responder (it registers, `node_up`), then the dialer closes on the name +//! check (`node_down`) — one membership blip per attempt. Cross-process: a +//! *server* named `server` subscribes and reports; a *client* on fast +//! timing (50–500ms backoff) seeds `("wrongname", server_addr)`. After the +//! first blip the server counts further `NodeUp`s across 2s — several +//! backoff periods. Parked ⇒ zero. Negative-control-verified: with the park +//! stubbed out the count is ≥ 1 in the same window. +#![cfg(feature = "cluster")] + +mod common; + +use common::{maybe_child, spawn_node}; +use smarm::cluster::envelope::NodeMeta; +use smarm::cluster::membership::{subscribe, NodeEvent}; +use smarm::cluster::{start, Config, StaticSeeds, Timing}; +use std::time::{Duration, Instant}; + +const ROLES: &[(&str, fn())] = &[("server", role_server), ("client", role_client)]; + +fn meta() -> NodeMeta { + NodeMeta { + role: "mismatch".into(), + region: "local".into(), + } +} + +fn timing() -> Timing { + Timing { + initial_backoff: Duration::from_millis(50), + max_backoff: Duration::from_millis(500), + ..Timing::default() + } +} + +fn role_server() { + smarm::run(|| { + let cluster = start(Config { + node_name: "server".into(), + meta: meta(), + listen_addr: std::env::var("SMARM_LISTEN_ADDR") + .unwrap_or_else(|_| "127.0.0.1:0".into()), + strategy: Box::new(StaticSeeds::new(Vec::<(String, String)>::new())), + timing: timing(), + }) + .expect("binds"); + let ev = subscribe().unwrap(); + println!("LISTENING {}", cluster.local_addr()); + // First blip: the misdialed client establishes, then closes on us. + loop { + match ev.rx.recv() { + Ok(NodeEvent::NodeDown(i)) if i.name == "client" => break, + Ok(_) => continue, + Err(_) => panic!("manager gone"), + } + } + println!("BLIP"); + // Now count further NodeUps across several backoff periods. + let mut more = 0usize; + let t0 = Instant::now(); + while t0.elapsed() < Duration::from_millis(2000) { + match ev.rx.try_recv() { + Ok(Some(NodeEvent::NodeUp(i))) if i.name == "client" => more += 1, + Ok(_) => {} + Err(_) => panic!("manager gone"), + } + smarm::sleep(Duration::from_millis(50)); + } + println!("MORE {more}"); + loop { + smarm::sleep(Duration::from_secs(3600)); + } + }); +} + +fn role_client() { + let server_addr = std::env::var("SMARM_SERVER_ADDR").expect("SMARM_SERVER_ADDR"); + smarm::run(move || { + let _cluster = start(Config { + node_name: "client".into(), + meta: meta(), + listen_addr: "127.0.0.1:0".into(), + strategy: Box::new(StaticSeeds::new(vec![( + "wrongname".to_string(), + server_addr, + )])), + timing: timing(), + }) + .expect("binds"); + println!("CLIENT UP"); + loop { + smarm::sleep(Duration::from_secs(3600)); + } + }); +} + +#[test] +fn mismatched_seed_is_dialed_once_then_parked() { + maybe_child(ROLES); + let mut server = spawn_node("server", &[]); + let saddr = server.wait_listening(); + let mut client = spawn_node("client", &[("SMARM_SERVER_ADDR", &saddr)]); + client.wait_line("CLIENT UP", |l| l == "CLIENT UP"); + server.wait_line("BLIP", |l| l == "BLIP"); + let line = server.wait_line("MORE", |l| l.starts_with("MORE ")); + let more: usize = line.split_whitespace().nth(1).unwrap().parse().unwrap(); + assert_eq!( + more, 0, + "mismatched seed was redialed {more}× after being parked" + ); +} diff --git a/tests/cluster_disconnect.rs b/tests/cluster_disconnect.rs new file mode 100644 index 0000000..a7ae186 --- /dev/null +++ b/tests/cluster_disconnect.rs @@ -0,0 +1,379 @@ +//! RFC 010 c13 — connection-loss synthesis. +//! +//! Local suite (`run()`, no network): the read-side backstop. A +//! `RemoteMonitor` whose channel closes without a notice reads as +//! `Disconnected` exactly once (a `Monitor` command that reached the conn +//! actor's inbox but was never processed — the drain gap); after +//! `demonitor_remote` a closed channel stays a plain `Err`, never a notice. +//! +//! Cross-process: the headline contrast — an actor's own death gives its +//! TRUE reason, loss of the LINK gives `Disconnected` (both a commanded +//! `Disconnect` and a SIGKILLed peer process are `Disconnected` from the +//! monitor's view: nobody is left to say otherwise). Reconnect does not +//! resurrect: the old monitor yields nothing more, proven by stream ORDER +//! (a fresh monitor over the new link delivers first). The ignored test +//! trips liveness by SIGSTOP and then drops the link too, asserting exactly +//! one notice for one monitor. +#![cfg(feature = "cluster")] + +mod common; + +use common::{maybe_child, spawn_node}; +use smarm::cluster::envelope::NodeMeta; +use smarm::cluster::expose::{expose, expose_type}; +use smarm::cluster::manager::{Call, Reply, MANAGER}; +use smarm::cluster::membership::{subscribe, NodeEvent}; +use smarm::cluster::remote::{ + self, demonitor_remote, monitor_remote, send_to_remote, RemoteName, RemotePid, +}; +use smarm::cluster::{start, Config, RemoteDownReason, StaticSeeds, Timing}; +use smarm::pg::Incarnation; +use smarm::{ + channel, gen_server, install, register, run, spawn, Addressable, DownReason, Erased, Name, Pid, +}; +use std::collections::HashMap; +use std::time::Duration; + +// ---- message types (hand-rolled serde; the crate is derive-less) --------- + +#[derive(Debug)] +struct Ctl { + cmd: String, + reply_to: RemotePid, +} +#[derive(Debug)] +struct Answer { + text: String, + pid: Option>, +} +struct Client; +impl Addressable for Client { + type Msg = Answer; +} + +impl serde::Serialize for Ctl { + fn serialize(&self, s: S) -> Result { + use serde::ser::SerializeTuple; + let mut t = s.serialize_tuple(2)?; + t.serialize_element(&self.cmd)?; + t.serialize_element(&self.reply_to)?; + t.end() + } +} +impl<'de> serde::Deserialize<'de> for Ctl { + fn deserialize>(d: D) -> Result { + let (cmd, reply_to) = <(String, RemotePid)>::deserialize(d)?; + Ok(Ctl { cmd, reply_to }) + } +} +impl serde::Serialize for Answer { + fn serialize(&self, s: S) -> Result { + use serde::ser::SerializeTuple; + let mut t = s.serialize_tuple(2)?; + t.serialize_element(&self.text)?; + t.serialize_element(&self.pid)?; + t.end() + } +} +impl<'de> serde::Deserialize<'de> for Answer { + fn deserialize>(d: D) -> Result { + let (text, pid) = <(String, Option>)>::deserialize(d)?; + Ok(Answer { text, pid }) + } +} + +// ================= local suite ========================================= + +/// A `Monitor` command handed to the connection but never processed (its +/// receiver dropped unread) reads as `Disconnected` — once. A second read +/// is the ordinary closed-channel `Err`, so "exactly one notice" holds. +#[test] +fn unread_command_reads_as_disconnected_once() { + maybe_child(ROLES); + run(|| { + remote::set_local_identity("me", Incarnation::new(7)); + let (probe_tx, _probe_rx) = channel(); + let inbox = + remote::bind_outbound_probe_with_monitors("peer", Incarnation::new(5), probe_tx); + let target = RemotePid::::from_parts("peer", Incarnation::new(5), 9, 1); + let m = monitor_remote(target.clone()); + assert!( + matches!(m.try_recv(), Ok(None)), + "command is in flight, no notice yet" + ); + drop(inbox); // the conn actor died with the command unread + let d = m.recv().unwrap(); + assert_eq!(d.pid, target); + assert_eq!(d.reason, RemoteDownReason::Disconnected); + assert!( + m.recv().is_err(), + "second read is closed, not a second notice" + ); + assert!(m.try_recv().is_err()); + }); +} + +/// After `demonitor_remote`, a closed channel is a closed channel: no +/// notice is synthesized for a monitor the caller cancelled. +#[test] +fn cancelled_monitor_never_synthesizes() { + maybe_child(ROLES); + run(|| { + remote::set_local_identity("me", Incarnation::new(7)); + let (probe_tx, _probe_rx) = channel(); + let inbox = + remote::bind_outbound_probe_with_monitors("peer", Incarnation::new(5), probe_tx); + let target = RemotePid::::from_parts("peer", Incarnation::new(5), 9, 1); + let m = monitor_remote(target); + demonitor_remote(&m); + drop(inbox); + assert!(m.recv().is_err()); + assert!(m.try_recv().is_err()); + }); +} + +// ================= cross-process ====================================== + +const ROLES: &[(&str, fn())] = &[ + ("server", role_server), + ("client", role_client), + ("client_stop", role_client_stop), +]; + +const CTL: Name = Name::new("c13.ctl"); + +fn cfg(name: &str, seeds: Vec<(String, String)>) -> Config { + Config { + node_name: name.to_string(), + meta: NodeMeta { + role: "c13".into(), + region: "local".into(), + }, + listen_addr: std::env::var("SMARM_LISTEN_ADDR").unwrap_or_else(|_| "127.0.0.1:0".into()), + strategy: Box::new(StaticSeeds::new(seeds)), + timing: timing(), + } +} + +/// The p11 knobs make the liveness test fast: both roles of that test are +/// spawned with `SMARM_FAST_TIMING=1` and agree on a 100ms heartbeat / +/// 500ms liveness window. Everything else runs the shipping defaults. +fn timing() -> Timing { + if std::env::var_os("SMARM_FAST_TIMING").is_some() { + Timing { + heartbeat_interval: Duration::from_millis(100), + liveness_timeout: Duration::from_millis(500), + initial_backoff: Duration::from_millis(50), + max_backoff: Duration::from_millis(500), + ..Timing::default() + } + } else { + Timing::default() + } +} + +fn wait_up(events: &smarm::cluster::membership::MembershipEvents, who: &str) { + loop { + match events.rx.recv() { + Ok(NodeEvent::NodeUp(i)) if i.name == who => return, + Ok(_) => continue, + Err(_) => panic!("manager gone"), + } + } +} + +fn disconnect(name: &str) { + assert!(matches!( + gen_server::call( + MANAGER, + Call::Disconnect { + name: name.to_string() + } + ), + Ok(Reply::Disconnected) + )); +} + +/// Server: `spawn` ⇒ a parked worker (answer carries its pid); +/// `kill:` releases it, whereupon it returns (Exit). +fn role_server() { + smarm::run(move || { + let cluster = start(cfg("server", vec![])).expect("binds"); + println!("LISTENING {}", cluster.local_addr()); + let (tx, rx) = channel::(); + register(CTL, tx).unwrap(); + expose(CTL); + println!("READY"); + let mut workers: HashMap> = HashMap::new(); + loop { + let ctl = rx.recv().unwrap(); + println!("CTL {}", ctl.cmd); + let (text, pid): (String, Option>) = match ctl.cmd.as_str() { + "spawn" => { + let (go_tx, go_rx) = channel::<()>(); + let p: Pid = spawn(move || { + let _ = go_rx.recv(); + }) + .pid(); + workers.insert(p.index(), go_tx); + ( + "ok".into(), + Some(RemotePid::from_local(p).expect("identity set")), + ) + } + other => { + let idx: u32 = other.strip_prefix("kill:").unwrap().parse().unwrap(); + if let Some(go) = workers.remove(&idx) { + let _ = go.send(()); + } + ("killed".into(), None) + } + }; + send_to_remote(ctl.reply_to, Answer { text, pid }).unwrap(); + } + }); +} + +/// Client-side setup shared by both client roles: join, expose the reply +/// path, hand back an `ask` closure and the membership stream. +fn client_setup() -> ( + smarm::cluster::Cluster, + smarm::cluster::membership::MembershipEvents, + impl Fn(&str) -> Answer, +) { + let server_addr = std::env::var("SMARM_SERVER_ADDR").expect("SMARM_SERVER_ADDR"); + let cluster = start(cfg("client", vec![("server".into(), server_addr)])).expect("binds"); + let ev = subscribe().unwrap(); + wait_up(&ev, "server"); + let (tx, rx) = channel::(); + let me: Pid = install::(tx); + expose_type::(); + let ask = move |cmd: &str| -> Answer { + remote::send( + RemoteName::new("server", CTL), + Ctl { + cmd: cmd.into(), + reply_to: RemotePid::from_local(me).expect("identity set"), + }, + ) + .unwrap(); + rx.recv().unwrap() + }; + (cluster, ev, ask) +} + +fn role_client() { + smarm::run(move || { + let (_cluster, ev, ask) = client_setup(); + + // 1. Headline: actor death ⇒ TRUE reason; link cut ⇒ Disconnected. + let a = ask("spawn").pid.unwrap(); + let b = ask("spawn").pid.unwrap(); + let ma = monitor_remote(a.clone()); + let mb = monitor_remote(b.clone()); + ask(&format!("kill:{}", a.index())); + let d = ma.recv().unwrap(); + assert_eq!(d.pid, a); + println!("DOWN actor {:?}", d.reason); + disconnect("server"); + let d = mb.recv().unwrap(); + assert_eq!(d.pid, b); + println!("DOWN link {:?}", d.reason); + + // 2. Reconnect does not resurrect. The connector redials on + // node_down; over the NEW link a fresh monitor delivers, while + // the old one (already answered) yields nothing further — order + // proves it, and `b` is even still alive on the server. + wait_up(&ev, "server"); + println!("RECONNECTED"); + let c = ask("spawn").pid.unwrap(); + let mc = monitor_remote(c.clone()); + ask(&format!("kill:{}", b.index())); + ask(&format!("kill:{}", c.index())); + assert_eq!(mc.recv().unwrap().reason, DownReason::Exit.into()); + let stray = matches!(mb.try_recv(), Ok(Some(_))); + println!("RESURRECT stray={stray}"); + + // 3. Peer PROCESS killed ⇒ Disconnected too (nobody is left to send + // Down): the parent SIGKILLs the server once it sees the marker. + let e = ask("spawn").pid.unwrap(); + let me_ = monitor_remote(e.clone()); + println!("KILL SERVER NOW"); + let d = me_.recv().unwrap(); + assert_eq!(d.pid, e); + println!("DOWN procdeath {:?}", d.reason); + + println!("CLIENT DONE"); + loop { + smarm::sleep(Duration::from_secs(3600)); + } + }); +} + +/// The slow role: liveness expiry (peer SIGSTOPped) followed by the link +/// dropping for real (peer SIGKILLed) — one monitor, exactly one notice. +fn role_client_stop() { + smarm::run(move || { + let (_cluster, _ev, ask) = client_setup(); + let a = ask("spawn").pid.unwrap(); + let ma = monitor_remote(a.clone()); + println!("STOP SERVER NOW"); + let d = ma.recv().unwrap(); // liveness expiry, ~liveness_timeout + assert_eq!(d.pid, a); + println!("DOWN stopped {:?}", d.reason); + println!("KILL SERVER NOW"); + // Give the drop every chance to produce a second notice, then look. + smarm::sleep(Duration::from_secs(1)); + let dup = matches!(ma.try_recv(), Ok(Some(_))); + println!("DUPLICATE dup={dup}"); + println!("CLIENT DONE"); + loop { + smarm::sleep(Duration::from_secs(3600)); + } + }); +} + +/// The Phase 4 c13 gate: partition vs. death distinguishable; nothing +/// survives reconnect; a dead peer process is a Disconnected too. +#[test] +fn link_loss_is_disconnected_and_does_not_survive_reconnect() { + maybe_child(ROLES); + let mut server = spawn_node("server", &[]); + let saddr = server.wait_listening(); + server.wait_line("READY", |l| l == "READY"); + let mut client = spawn_node("client", &[("SMARM_SERVER_ADDR", &saddr)]); + client.wait_line("DOWN actor Local(Exit)", |l| l == "DOWN actor Local(Exit)"); + client.wait_line("DOWN link Disconnected", |l| l == "DOWN link Disconnected"); + client.wait_line("RECONNECTED", |l| l == "RECONNECTED"); + client.wait_line("RESURRECT stray=false", |l| l == "RESURRECT stray=false"); + client.wait_line("KILL SERVER NOW", |l| l == "KILL SERVER NOW"); + server.kill(); + client.wait_line("DOWN procdeath Disconnected", |l| { + l == "DOWN procdeath Disconnected" + }); + client.wait_line("CLIENT DONE", |l| l == "CLIENT DONE"); +} + +/// Covers the invariant the headline test cannot: liveness expiry and the +/// transport drop both firing for the same connection yield ONE notice. +/// Runs on the fast [`timing`] (both roles) — was `#[ignore]`d at the 4s +/// default until the p11 knobs landed. +#[test] +fn timeout_then_drop_yields_one_notice() { + maybe_child(ROLES); + let fast = ("SMARM_FAST_TIMING", "1"); + let mut server = spawn_node("server", &[fast]); + let saddr = server.wait_listening(); + server.wait_line("READY", |l| l == "READY"); + let mut client = spawn_node("client_stop", &[("SMARM_SERVER_ADDR", &saddr), fast]); + client.wait_line("STOP SERVER NOW", |l| l == "STOP SERVER NOW"); + let spid = server.pid().expect("server alive") as libc::pid_t; + assert_eq!(unsafe { libc::kill(spid, libc::SIGSTOP) }, 0); + client.wait_line("DOWN stopped Disconnected", |l| { + l == "DOWN stopped Disconnected" + }); + client.wait_line("KILL SERVER NOW", |l| l == "KILL SERVER NOW"); + server.kill(); // SIGKILL works on a stopped process; Drop would too + client.wait_line("DUPLICATE dup=false", |l| l == "DUPLICATE dup=false"); + client.wait_line("CLIENT DONE", |l| l == "CLIENT DONE"); +} diff --git a/tests/cluster_discovery_withdraw.rs b/tests/cluster_discovery_withdraw.rs new file mode 100644 index 0000000..f261451 --- /dev/null +++ b/tests/cluster_discovery_withdraw.rs @@ -0,0 +1,161 @@ +//! RFC 010 — `Discovery::Withdrawn`: a strategy retracts a candidate and the +//! connector stops dialing it. +//! +//! Cross-process: a plain *server* node, and a *client* whose strategy is a +//! script: announce a decoy `(ghost, addr)` where `addr` is a raw +//! `TcpListener` the client itself holds (an OS thread accepts and +//! immediately closes, so every dial fails at handshake and the connector +//! keeps retrying on backoff — the accept count is the dial count); after a +//! beat, withdraw the decoy and announce the real server. The client waits +//! for the server's `node_up` — which is *after* the withdrawal in the +//! strategy's own stream — then watches the decoy's accept count stay flat +//! across a window longer than the pending backoff. Before withdrawal it +//! must have been climbing (≥ 1), or the negative proves nothing. +#![cfg(feature = "cluster")] + +mod common; + +use common::{maybe_child, spawn_node}; +use smarm::channel::Sender; +use smarm::cluster::discovery::{Discovery, Strategy}; +use smarm::cluster::envelope::NodeMeta; +use smarm::cluster::membership::{subscribe, NodeEvent}; +use smarm::cluster::{start, Config, StaticSeeds, Timing}; +use std::net::TcpListener; +use std::sync::atomic::{AtomicUsize, Ordering}; +use std::sync::Arc; +use std::time::{Duration, Instant}; + +const ROLES: &[(&str, fn())] = &[("server", role_server), ("client", role_client)]; + +fn meta() -> NodeMeta { + NodeMeta { + role: "withdraw".into(), + region: "local".into(), + } +} + +fn role_server() { + smarm::run(|| { + let cluster = start(Config { + node_name: "server".into(), + meta: meta(), + listen_addr: std::env::var("SMARM_LISTEN_ADDR") + .unwrap_or_else(|_| "127.0.0.1:0".into()), + strategy: Box::new(StaticSeeds::new(Vec::<(String, String)>::new())), + timing: Timing::default(), + }) + .expect("binds"); + println!("LISTENING {}", cluster.local_addr()); + loop { + smarm::sleep(Duration::from_secs(3600)); + } + }); +} + +/// Scripted strategy: decoy, pause, withdraw decoy, real server, done. +struct Script { + decoy: String, + server: String, +} + +impl Strategy for Script { + fn run(self: Box, out: Sender) { + let _ = out.send(Discovery::Candidate { + name: "ghost".into(), + addr: self.decoy.clone(), + }); + // Long enough for the 250ms/500ms retries to land: ≥ 3 dials. + smarm::sleep(Duration::from_millis(1100)); + let _ = out.send(Discovery::Withdrawn { + name: "ghost".into(), + addr: self.decoy, + }); + let _ = out.send(Discovery::Candidate { + name: "server".into(), + addr: self.server, + }); + } +} + +fn role_client() { + let server_addr = std::env::var("SMARM_SERVER_ADDR").expect("SMARM_SERVER_ADDR"); + // The decoy: accept-and-close on an OS thread; count every accept. + let decoy = TcpListener::bind("127.0.0.1:0").unwrap(); + let decoy_addr = decoy.local_addr().unwrap().to_string(); + let dials = Arc::new(AtomicUsize::new(0)); + let counter = dials.clone(); + std::thread::spawn(move || { + for conn in decoy.incoming() { + counter.fetch_add(1, Ordering::SeqCst); + drop(conn); + } + }); + + smarm::run(move || { + let _cluster = start(Config { + node_name: "client".into(), + meta: meta(), + listen_addr: "127.0.0.1:0".into(), + strategy: Box::new(Script { + decoy: decoy_addr, + server: server_addr, + }), + timing: Timing::default(), + }) + .expect("binds"); + let ev = subscribe().unwrap(); + loop { + match ev.rx.recv() { + Ok(NodeEvent::NodeUp(i)) if i.name == "server" => break, + Ok(_) => continue, + Err(_) => panic!("manager gone"), + } + } + // The withdrawal preceded the server candidate in the strategy's + // stream, so it has been applied. Any dial that started before it + // is bounded by the connect+handshake deadlines; let it drain, then + // hold the count flat across a window longer than the pending + // backoff would be (1s at this point, 2s next). + let before = dials.load(Ordering::SeqCst); + smarm::sleep(Duration::from_millis(500)); + let settled = dials.load(Ordering::SeqCst); + let t0 = Instant::now(); + while t0.elapsed() < Duration::from_millis(3000) { + smarm::sleep(Duration::from_millis(100)); + } + let after = dials.load(Ordering::SeqCst); + println!("WITHDRAWN before={before} settled={settled} after={after}"); + println!("CLIENT DONE"); + loop { + smarm::sleep(Duration::from_secs(3600)); + } + }); +} + +#[test] +fn withdrawn_candidate_is_no_longer_dialed() { + maybe_child(ROLES); + let mut server = spawn_node("server", &[]); + let saddr = server.wait_listening(); + let mut client = spawn_node("client", &[("SMARM_SERVER_ADDR", &saddr)]); + let line = client.wait_line("WITHDRAWN", |l| l.starts_with("WITHDRAWN ")); + let mut nums = line + .split_whitespace() + .skip(1) + .map(|kv| kv.split_once('=').unwrap().1.parse::().unwrap()); + let (before, settled, after) = ( + nums.next().unwrap(), + nums.next().unwrap(), + nums.next().unwrap(), + ); + assert!( + before >= 1, + "decoy was never dialed; the negative proves nothing: {line}" + ); + assert_eq!( + settled, after, + "connector kept dialing a withdrawn candidate: {line}" + ); + client.wait_line("CLIENT DONE", |l| l == "CLIENT DONE"); +} diff --git a/tests/cluster_envelope.rs b/tests/cluster_envelope.rs index aba1773..f81dad3 100644 --- a/tests/cluster_envelope.rs +++ b/tests/cluster_envelope.rs @@ -8,6 +8,7 @@ use smarm::cluster::envelope::{ decode_payload, encode_payload, DecodeError, Frame, NodeMeta, RejectReason, MAX_FRAME_LEN, PROTO_VERSION, }; +use smarm::cluster::RemoteDownReason; use smarm::monitor::DownReason; use smarm::pg::Incarnation; @@ -55,7 +56,11 @@ fn all_frames() -> Vec { Frame::Demonitor { monitor_id: 77 }, Frame::Down { monitor_id: 77, - reason: DownReason::Panic, + reason: RemoteDownReason::Local(DownReason::Panic), + }, + Frame::Down { + monitor_id: 78, + reason: RemoteDownReason::Disconnected, }, ] } @@ -151,7 +156,7 @@ fn unknown_enum_tags() { assert_eq!( Frame::decode(&buf), Err(DecodeError::UnknownEnumTag { - what: "DownReason", + what: "RemoteDownReason", tag: 200 }) ); diff --git a/tests/cluster_handshake.rs b/tests/cluster_handshake.rs index 0eede87..fa4a755 100644 --- a/tests/cluster_handshake.rs +++ b/tests/cluster_handshake.rs @@ -5,7 +5,7 @@ use smarm::cluster::envelope::{Frame, NodeMeta, RejectReason, PROTO_VERSION}; use smarm::cluster::handshake::{ - dial_wins, HelloCtx, Initiator, InitiatorOutcome, Local, Responder, ResponderOutcome, + dial_wins, Initiator, InitiatorOutcome, Local, PeerStanding, Responder, ResponderOutcome, }; use smarm::pg::Incarnation; @@ -42,7 +42,7 @@ fn happy_path_establishes_both_ends() { assert_eq!(hello, hello_from("alpha"), "initiator emits its identity"); let responder = Responder::new(local("beta")); - let (reply, peer) = match responder.on_frame(hello, HelloCtx::default()) { + let (reply, peer) = match responder.on_frame(hello, PeerStanding::Free) { ResponderOutcome::Accepted { reply, peer } => (reply, peer), other => panic!("expected Accepted, got {other:?}"), }; @@ -82,7 +82,7 @@ fn hash_mismatch_rejected() { incarnation: Incarnation::new(7), meta: local("alpha").meta, }; - match responder.on_frame(hello, HelloCtx::default()) { + match responder.on_frame(hello, PeerStanding::Free) { ResponderOutcome::Rejected { reply, reason } => { assert_eq!(reason, RejectReason::HashMismatch); assert_eq!(reply, Frame::HelloReject { reason }); @@ -112,7 +112,7 @@ fn proto_version_mismatch_rejected_and_checked_first() { incarnation: Incarnation::new(7), meta: local("alpha").meta, }; - match responder.on_frame(hello, HelloCtx::default()) { + match responder.on_frame(hello, PeerStanding::Free) { ResponderOutcome::Rejected { reason, .. } => { assert_eq!(reason, RejectReason::ProtoVersion); } @@ -123,10 +123,7 @@ fn proto_version_mismatch_rejected_and_checked_first() { #[test] fn claimed_name_rejected() { let responder = Responder::new(local("beta")); - let ctx = HelloCtx { - name_claimed: true, - dialing_this_peer: false, - }; + let ctx = PeerStanding::Claimed; match responder.on_frame(hello_from("alpha"), ctx) { ResponderOutcome::Rejected { reason, .. } => { assert_eq!(reason, RejectReason::NameTaken); @@ -139,7 +136,7 @@ fn claimed_name_rejected() { fn own_name_offered_rejected_as_name_taken() { // Self-connect or genuine collision: the responder's own name arrives. let responder = Responder::new(local("beta")); - match responder.on_frame(hello_from("beta"), HelloCtx::default()) { + match responder.on_frame(hello_from("beta"), PeerStanding::Free) { ResponderOutcome::Rejected { reason, .. } => { assert_eq!(reason, RejectReason::NameTaken); } @@ -158,10 +155,7 @@ fn hash_checked_before_name() { incarnation: Incarnation::new(7), meta: local("alpha").meta, }; - let ctx = HelloCtx { - name_claimed: true, - dialing_this_peer: false, - }; + let ctx = PeerStanding::Claimed; match responder.on_frame(hello, ctx) { ResponderOutcome::Rejected { reason, .. } => { assert_eq!(reason, RejectReason::HashMismatch); @@ -184,10 +178,7 @@ fn dial_wins_is_deterministic_and_antisymmetric() { fn simultaneous_connect_exactly_one_side_accepts() { // alpha and beta dial each other at once. Each responder sees the peer's // Hello while its own dial is in flight. - let ctx = HelloCtx { - name_claimed: false, - dialing_this_peer: true, - }; + let ctx = PeerStanding::Dialing; // On beta: inbound is alpha's dial; alpha < beta, so the inbound wins. let on_beta = Responder::new(local("beta")).on_frame(hello_from("alpha"), ctx); @@ -208,7 +199,7 @@ fn simultaneous_connect_exactly_one_side_accepts() { #[test] fn tiebreak_loss_only_applies_when_dialing() { // Same inbound Hello, no dial in flight: plain accept. - let on_alpha = Responder::new(local("alpha")).on_frame(hello_from("beta"), HelloCtx::default()); + let on_alpha = Responder::new(local("alpha")).on_frame(hello_from("beta"), PeerStanding::Free); assert!(matches!(on_alpha, ResponderOutcome::Accepted { .. })); } @@ -225,7 +216,7 @@ fn garbage_before_hello_fails_without_reply() { }, Frame::Demonitor { monitor_id: 3 }, ] { - let out = Responder::new(local("beta")).on_frame(frame.clone(), HelloCtx::default()); + let out = Responder::new(local("beta")).on_frame(frame.clone(), PeerStanding::Free); match out { ResponderOutcome::Failed(f) => assert_eq!(f, frame), other => panic!("expected Failed({frame:?}), got {other:?}"), diff --git a/tests/cluster_membership.rs b/tests/cluster_membership.rs index cf2eb67..6a597be 100644 --- a/tests/cluster_membership.rs +++ b/tests/cluster_membership.rs @@ -18,6 +18,7 @@ use smarm::cluster::membership::{subscribe, view, MembershipEvents, NodeEvent}; use smarm::cluster::spawn_established; use smarm::cluster::transport::tcp::TcpTransport; use smarm::cluster::transport::{Conn, FramedConn, Transport}; +use smarm::cluster::Timing; use smarm::gen_server::{self, GenServerBuilder}; use smarm::pg::{Incarnation, NodeId}; use smarm::run; @@ -85,8 +86,10 @@ fn subscriber_sees_up_and_down() { let t = TcpTransport; let (a1, b1) = pair(&t); let (a2, b2) = pair(&t); - spawn_established(FramedConn::new(a1), peer("node-b", 1)).expect("node-b registers"); - spawn_established(FramedConn::new(a2), peer("node-c", 1)).expect("node-c registers"); + spawn_established(FramedConn::new(a1), peer("node-b", 1), Timing::default()) + .expect("node-b registers"); + spawn_established(FramedConn::new(a2), peer("node-c", 1), Timing::default()) + .expect("node-c registers"); let up_b = match next_event(&ev, "node_up(node-b)") { NodeEvent::NodeUp(info) => { @@ -110,14 +113,14 @@ fn subscriber_sees_up_and_down() { disconnect("node-b"); assert_eq!( next_event(&ev, "node_down(node-b)"), - NodeEvent::NodeDown { node: up_b.node } + NodeEvent::NodeDown(up_b.clone()) ); // Peer EOF, no command: down with node-c's id. drop(b2); assert_eq!( next_event(&ev, "node_down(node-c)"), - NodeEvent::NodeDown { node: up_c.node } + NodeEvent::NodeDown(up_c.clone()) ); assert_quiet(&ev); @@ -140,8 +143,10 @@ fn late_subscriber_gets_snapshot_and_view_agrees() { let t = TcpTransport; let (a1, b1) = pair(&t); let (a2, b2) = pair(&t); - spawn_established(FramedConn::new(a1), peer("node-b", 1)).expect("node-b registers"); - spawn_established(FramedConn::new(a2), peer("node-c", 1)).expect("node-c registers"); + spawn_established(FramedConn::new(a1), peer("node-b", 1), Timing::default()) + .expect("node-b registers"); + spawn_established(FramedConn::new(a2), peer("node-c", 1), Timing::default()) + .expect("node-c registers"); let ev = subscribe().expect("manager is up"); let mut names = Vec::new(); @@ -190,20 +195,25 @@ fn restart_gets_new_id_blip_keeps_id() { other => panic!("expected node_up ({what}), got {other:?}"), } }; + let down_id = |e: NodeEvent, what: &str| -> NodeId { + match e { + NodeEvent::NodeDown(info) => info.node, + other => panic!("expected node_down ({what}), got {other:?}"), + } + }; // Up at incarnation 1, then the peer dies (EOF). let (a1, b1) = pair(&t); - spawn_established(FramedConn::new(a1), peer("node-b", 1)).expect("registers"); + spawn_established(FramedConn::new(a1), peer("node-b", 1), Timing::default()) + .expect("registers"); let id1 = id(next_event(&ev, "node_up inc 1"), "inc 1"); drop(b1); - assert_eq!( - next_event(&ev, "node_down inc 1"), - NodeEvent::NodeDown { node: id1 } - ); + assert_eq!(down_id(next_event(&ev, "node_down inc 1"), "inc 1"), id1); // Restart: new incarnation, new id — the ghost's id is not reused. let (a2, b2) = pair(&t); - spawn_established(FramedConn::new(a2), peer("node-b", 2)).expect("registers"); + spawn_established(FramedConn::new(a2), peer("node-b", 2), Timing::default()) + .expect("registers"); let id2 = id(next_event(&ev, "node_up inc 2"), "inc 2"); assert_ne!( id1, id2, @@ -212,12 +222,10 @@ fn restart_gets_new_id_blip_keeps_id() { // Blip: the same incarnation reconnects and keeps its id. disconnect("node-b"); - assert_eq!( - next_event(&ev, "node_down inc 2"), - NodeEvent::NodeDown { node: id2 } - ); + assert_eq!(down_id(next_event(&ev, "node_down inc 2"), "inc 2"), id2); let (a3, b3) = pair(&t); - spawn_established(FramedConn::new(a3), peer("node-b", 2)).expect("registers"); + spawn_established(FramedConn::new(a3), peer("node-b", 2), Timing::default()) + .expect("registers"); let id3 = id(next_event(&ev, "node_up after blip"), "blip"); assert_eq!( id2, id3, @@ -247,7 +255,8 @@ fn dead_subscriber_is_pruned() { let t = TcpTransport; let (a1, b1) = pair(&t); - spawn_established(FramedConn::new(a1), peer("node-b", 1)).expect("registers"); + spawn_established(FramedConn::new(a1), peer("node-b", 1), Timing::default()) + .expect("registers"); match next_event(&live, "node_up despite a dead co-subscriber") { NodeEvent::NodeUp(info) => assert_eq!(info.name, "node-b"), other => panic!("expected node_up, got {other:?}"), diff --git a/tests/cluster_mesh.rs b/tests/cluster_mesh.rs index 6b6fb18..08451d4 100644 --- a/tests/cluster_mesh.rs +++ b/tests/cluster_mesh.rs @@ -24,9 +24,7 @@ mod common; use common::{maybe_child, spawn_node, Node}; use smarm::cluster::envelope::NodeMeta; use smarm::cluster::membership::{subscribe, NodeEvent}; -use smarm::cluster::{start, Config, StaticSeeds}; -use smarm::pg::NodeId; -use std::collections::HashMap; +use smarm::cluster::{start, Config, StaticSeeds, Timing}; use std::time::Duration; const ROLES: &[(&str, fn())] = &[("node", role_node)]; @@ -56,21 +54,19 @@ fn role_node() { }, listen_addr: listen, strategy: Box::new(StaticSeeds::new(seeds)), + timing: Timing::default(), }) .expect("listener binds"); println!("LISTENING {}", cluster.local_addr()); let events = subscribe().expect("manager is up"); - let mut names: HashMap = HashMap::new(); loop { match events.rx.recv() { Ok(NodeEvent::NodeUp(info)) => { - names.insert(info.node, info.name.clone()); println!("MEMBER-UP {} inc={}", info.name, info.incarnation.get()); } - Ok(NodeEvent::NodeDown { node }) => { - let name = names.remove(&node).unwrap_or_else(|| "?".to_string()); - println!("MEMBER-DOWN {name}"); + Ok(NodeEvent::NodeDown(info)) => { + println!("MEMBER-DOWN {}", info.name); } Err(_) => break, // manager gone; park below regardless } diff --git a/tests/cluster_monitor.rs b/tests/cluster_monitor.rs new file mode 100644 index 0000000..8b141ee --- /dev/null +++ b/tests/cluster_monitor.rs @@ -0,0 +1,359 @@ +//! RFC 010 c12 — remote monitors. +//! +//! Local suite (`run()`, no network): the immediate answers — no connection +//! ⇒ `Disconnected`, dead incarnation ⇒ `NoProc` — and the self-node +//! collapse (a plain local monitor underneath, incl. `demonitor_remote`). +//! +//! Cross-process: a *server* exposes a control name and spawns workers on +//! request, replying with each worker's pid (via `RemotePid::from_local`, +//! the D12 set-site) or, for the deliberately unshipped one, only its raw +//! slot numbers. The *client* monitors them and asserts: kill ⇒ the true +//! reason (Exit / Panic); a corpse ⇒ its recorded terminal reason, not +//! NoProc; a live pid that never crossed the wire ⇒ NoProc (no liveness +//! leak); a demonitor racing the kill ⇒ no notice, proven by stream ORDER +//! (a later notice on the same connection arrives while the earlier slot +//! is still empty), not by sleeping. +#![cfg(feature = "cluster")] + +mod common; + +use common::{maybe_child, spawn_node}; +use smarm::cluster::envelope::NodeMeta; +use smarm::cluster::expose::{expose, expose_type}; +use smarm::cluster::membership::{subscribe, NodeEvent}; +use smarm::cluster::remote::{ + self, demonitor_remote, monitor_remote, send_to_remote, RemoteName, RemotePid, +}; +use smarm::cluster::{start, Config, RemoteDownReason, StaticSeeds, Timing}; +use smarm::pg::Incarnation; +use smarm::{channel, install, register, run, spawn, Addressable, DownReason, Erased, Name, Pid}; +use std::collections::HashMap; +use std::time::Duration; + +// ---- message types (hand-rolled serde; the crate is derive-less) --------- + +#[derive(Debug)] +struct Ctl { + cmd: String, + reply_to: RemotePid, +} +#[derive(Debug)] +struct Answer { + text: String, + pid: Option>, +} +struct Client; +impl Addressable for Client { + type Msg = Answer; +} + +impl serde::Serialize for Ctl { + fn serialize(&self, s: S) -> Result { + use serde::ser::SerializeTuple; + let mut t = s.serialize_tuple(2)?; + t.serialize_element(&self.cmd)?; + t.serialize_element(&self.reply_to)?; + t.end() + } +} +impl<'de> serde::Deserialize<'de> for Ctl { + fn deserialize>(d: D) -> Result { + let (cmd, reply_to) = <(String, RemotePid)>::deserialize(d)?; + Ok(Ctl { cmd, reply_to }) + } +} +impl serde::Serialize for Answer { + fn serialize(&self, s: S) -> Result { + use serde::ser::SerializeTuple; + let mut t = s.serialize_tuple(2)?; + t.serialize_element(&self.text)?; + t.serialize_element(&self.pid)?; + t.end() + } +} +impl<'de> serde::Deserialize<'de> for Answer { + fn deserialize>(d: D) -> Result { + let (text, pid) = <(String, Option>)>::deserialize(d)?; + Ok(Answer { text, pid }) + } +} + +// ================= local suite ========================================= + +/// No connection to the pid's node: `Disconnected` at once — the remote +/// analog of NoProc, and the first thing c11's variant is for. +#[test] +fn unconnected_node_is_disconnected_immediately() { + maybe_child(ROLES); + run(|| { + remote::set_local_identity("me", Incarnation::new(7)); + let ghost = RemotePid::::from_parts("nowhere", Incarnation::new(1), 3, 1); + let m = monitor_remote(ghost.clone()); + let d = m.recv().unwrap(); + assert_eq!(d.pid, ghost); + assert_eq!(d.reason, RemoteDownReason::Disconnected); + }); +} + +/// The node is connected but the pid names an earlier incarnation: the +/// actor is a known corpse (RFC v2 §3), so `NoProc` at once — never +/// `Disconnected`, nothing on the wire. +#[test] +fn dead_incarnation_is_noproc_immediately() { + maybe_child(ROLES); + run(|| { + remote::set_local_identity("me", Incarnation::new(7)); + let (probe_tx, probe_rx) = channel(); + remote::bind_outbound_probe("peer", Incarnation::new(5), probe_tx); + let stale = RemotePid::::from_parts("peer", Incarnation::new(4), 9, 1); + let m = monitor_remote(stale); + assert_eq!(m.recv().unwrap().reason, DownReason::NoProc.into()); + assert!(probe_rx.try_recv().unwrap().is_none(), "no frame emitted"); + }); +} + +/// A self-node pid collapses to an ordinary local monitor: the true reason +/// on exit, and `demonitor_remote` cancels it. +#[test] +fn self_node_pid_collapses_to_local_monitor() { + maybe_child(ROLES); + run(|| { + remote::set_local_identity("me", Incarnation::new(7)); + let (go_tx, go_rx) = channel::<()>(); + let (go2_tx, go2_rx) = channel::<()>(); + let a = spawn(move || { + let _ = go_rx.recv(); + }) + .pid(); + let b = spawn(move || { + let _ = go2_rx.recv(); + }) + .pid(); + let ma = monitor_remote(RemotePid::from_local(a).expect("identity set")); + let mb = monitor_remote(RemotePid::from_local(b).expect("identity set")); + assert_ne!(ma.id, mb.id); + assert!(ma.target.local() == Some(a)); + + demonitor_remote(&mb); + go2_tx.send(()).unwrap(); + go_tx.send(()).unwrap(); + let d = ma.recv().unwrap(); + assert_eq!(d.reason, DownReason::Exit.into()); + assert_eq!(d.pid.local(), Some(a)); + // `a` is down (its notice arrived), and `b` was killed first on the + // same scheduler — a notice for `b` would be here by now. After a + // demonitor the channel is closed-empty (`Err`), like the local one. + assert!(matches!(mb.try_recv(), Ok(None) | Err(_))); + }); +} + +// ================= cross-process ====================================== + +const ROLES: &[(&str, fn())] = &[("server", role_server), ("client", role_client)]; + +const CTL: Name = Name::new("c12.ctl"); + +fn cfg(name: &str, seeds: Vec<(String, String)>) -> Config { + Config { + node_name: name.to_string(), + meta: NodeMeta { + role: "c12".into(), + region: "local".into(), + }, + listen_addr: std::env::var("SMARM_LISTEN_ADDR").unwrap_or_else(|_| "127.0.0.1:0".into()), + strategy: Box::new(StaticSeeds::new(seeds)), + timing: Timing::default(), + } +} + +fn wait_up(events: &smarm::cluster::membership::MembershipEvents, who: &str) { + loop { + match events.rx.recv() { + Ok(NodeEvent::NodeUp(i)) if i.name == who => return, + Ok(_) => continue, + Err(_) => panic!("manager gone"), + } + } +} + +/// Server commands (all answered to `reply_to`): +/// - `spawn:exit` / `spawn:panic` — a parked worker; `kill:` releases +/// it, whereupon it returns / panics. Answer carries its pid. +/// - `spawn:corpse` — a worker that has already exited when the answer is +/// sent; the pid was shipped (watchable) before it died. +/// - `spawn:unwatched` — a parked worker whose pid is NEVER shipped; the +/// answer carries only `text = "slot::"`. +fn role_server() { + smarm::run(move || { + let cluster = start(cfg("server", vec![])).expect("binds"); + println!("LISTENING {}", cluster.local_addr()); + let (tx, rx) = channel::(); + register(CTL, tx).unwrap(); + expose(CTL); + println!("READY"); + let mut workers: HashMap> = HashMap::new(); + loop { + let ctl = rx.recv().unwrap(); + println!("CTL {}", ctl.cmd); + let (text, pid): (String, Option>) = match ctl.cmd.as_str() { + "spawn:exit" | "spawn:panic" => { + let panic = ctl.cmd == "spawn:panic"; + let (go_tx, go_rx) = channel::<()>(); + let p: Pid = spawn(move || { + let _ = go_rx.recv(); + if panic { + panic!("worker asked to panic"); + } + }) + .pid(); + workers.insert(p.index(), go_tx); + ( + "ok".into(), + Some(RemotePid::from_local(p).expect("identity set")), + ) + } + "spawn:corpse" => { + let p: Pid = spawn(|| {}).pid(); + let rp = RemotePid::from_local(p).expect("identity set"); // shipped ⇒ watchable + let m = smarm::monitor(p); + let _ = m.rx.recv(); // dead before the answer goes out + ("ok".into(), Some(rp)) + } + "spawn:unwatched" => { + let (go_tx, go_rx) = channel::<()>(); + let p: Pid = spawn(move || { + let _ = go_rx.recv(); + }) + .pid(); + workers.insert(p.index(), go_tx); + (format!("slot:{}:{}", p.index(), p.generation()), None) + } + other => { + let idx: u32 = other.strip_prefix("kill:").unwrap().parse().unwrap(); + if let Some(go) = workers.remove(&idx) { + let _ = go.send(()); + } + ("killed".into(), None) + } + }; + send_to_remote(ctl.reply_to, Answer { text, pid }).unwrap(); + } + }); +} + +fn role_client() { + let server_addr = std::env::var("SMARM_SERVER_ADDR").expect("SMARM_SERVER_ADDR"); + smarm::run(move || { + let _cluster = start(cfg("client", vec![("server".into(), server_addr)])).expect("binds"); + let ev = subscribe().unwrap(); + wait_up(&ev, "server"); + let (tx, rx) = channel::(); + let me: Pid = install::(tx); + expose_type::(); + let ask = |cmd: &str| -> Answer { + remote::send( + RemoteName::new("server", CTL), + Ctl { + cmd: cmd.into(), + reply_to: RemotePid::from_local(me).expect("identity set"), + }, + ) + .unwrap(); + rx.recv().unwrap() + }; + let server_inc = ev_incarnation(); + + // 1. kill ⇒ true reason (Exit). + let a = ask("spawn:exit").pid.unwrap(); + let ma = monitor_remote(a.clone()); + ask(&format!("kill:{}", a.index())); + let d = ma.recv().unwrap(); + assert_eq!(d.pid, a); + println!("DOWN exit {:?}", d.reason); + + // 2. kill ⇒ true reason (Panic). + let b = ask("spawn:panic").pid.unwrap(); + let mb = monitor_remote(b.clone()); + ask(&format!("kill:{}", b.index())); + println!("DOWN panic {:?}", mb.recv().unwrap().reason); + + // 3. corpse ⇒ recorded terminal reason, not NoProc. + let c = ask("spawn:corpse").pid.unwrap(); + println!("DOWN corpse {:?}", monitor_remote(c).recv().unwrap().reason); + + // 4. live but never shipped/exposed ⇒ NoProc (no leak); a made-up + // slot on the same node ⇒ NoProc too, indistinguishably. + let ans = ask("spawn:unwatched"); + let mut it = ans.text.strip_prefix("slot:").unwrap().split(':'); + let (idx, gen): (u32, u32) = ( + it.next().unwrap().parse().unwrap(), + it.next().unwrap().parse().unwrap(), + ); + let hidden = RemotePid::::from_parts("server", server_inc, idx, gen); + println!( + "DOWN hidden {:?}", + monitor_remote(hidden).recv().unwrap().reason + ); + let bogus = RemotePid::::from_parts("server", server_inc, 100_000, 1); + println!( + "DOWN bogus {:?}", + monitor_remote(bogus).recv().unwrap().reason + ); + + // 5. demonitor races the kill: no notice for `d1`, proven by order — + // `d2`'s notice (same connection, later) arrives while `d1`'s + // slot is still empty. + let d1 = ask("spawn:exit").pid.unwrap(); + let m1 = monitor_remote(d1.clone()); + demonitor_remote(&m1); + ask(&format!("kill:{}", d1.index())); + let d2 = ask("spawn:exit").pid.unwrap(); + let m2 = monitor_remote(d2.clone()); + ask(&format!("kill:{}", d2.index())); + assert_eq!(m2.recv().unwrap().reason, DownReason::Exit.into()); + // Closed-empty (`Err`) or open-empty (`Ok(None)`) both mean no notice. + let stray = matches!(m1.try_recv(), Ok(Some(_))); + println!("DEMONITOR stray={stray}"); + + println!("CLIENT DONE"); + loop { + smarm::sleep(Duration::from_secs(3600)); + } + }); +} + +/// The server's incarnation as this node sees it — for building pids by hand. +fn ev_incarnation() -> Incarnation { + smarm::cluster::membership::view() + .expect("manager up") + .into_iter() + .find(|i| i.name == "server") + .map(|i| i.incarnation) + .expect("server in view") +} + +/// The Phase 4 c12 gate: remote monitors report the true reason, honour +/// corpses, leak nothing for unshipped pids, and cancel cleanly. +#[test] +fn remote_monitors_report_true_reasons() { + maybe_child(ROLES); + let mut server = spawn_node("server", &[]); + let saddr = server.wait_listening(); + server.wait_line("READY", |l| l == "READY"); + let mut client = spawn_node("client", &[("SMARM_SERVER_ADDR", &saddr)]); + client.wait_line("DOWN exit Local(Exit)", |l| l == "DOWN exit Local(Exit)"); + client.wait_line("DOWN panic Local(Panic)", |l| { + l == "DOWN panic Local(Panic)" + }); + client.wait_line("DOWN corpse Local(Exit)", |l| { + l == "DOWN corpse Local(Exit)" + }); + client.wait_line("DOWN hidden Local(NoProc)", |l| { + l == "DOWN hidden Local(NoProc)" + }); + client.wait_line("DOWN bogus Local(NoProc)", |l| { + l == "DOWN bogus Local(NoProc)" + }); + client.wait_line("DEMONITOR stray=false", |l| l == "DEMONITOR stray=false"); + client.wait_line("CLIENT DONE", |l| l == "CLIENT DONE"); +} diff --git a/tests/cluster_pg.rs b/tests/cluster_pg.rs new file mode 100644 index 0000000..626b4b8 --- /dev/null +++ b/tests/cluster_pg.rs @@ -0,0 +1,254 @@ +//! RFC 010 c15 — distributed pg: sync on `NodeUp`, incremental +//! `Join`/`Leave`, eager eviction announced, `NodeDown` sweep. +//! +//! Two nodes. The *origin* joins two local workers to `"pool"` before the +//! *observer* connects (so the observer's view comes from `Sync`), exposes a +//! `"go"` command inbox and then does exactly what the observer tells it: +//! kill one worker, join a third, leave with the second. The observer drives +//! that script through the cluster itself and asserts every step from +//! `members_all` — never touching the group on its own side, except once to +//! prove a mixed local+remote group reads correctly and that `members` stays +//! local. `dispatch_any` is exercised both ways: into the origin's worker +//! (remote pick, `send_to_remote`) and, once the origin is gone, into the +//! observer's own (local pick, `send_to`). Finally the parent SIGKILLs the +//! origin: the observer must sweep every remote member on `NodeDown`. +#![cfg(feature = "cluster")] + +mod common; + +use common::{maybe_child, spawn_node}; +use smarm::cluster::envelope::NodeMeta; +use smarm::cluster::expose::{expose, expose_type}; +use smarm::cluster::membership::{subscribe, NodeEvent}; +use smarm::cluster::remote::{self, RemoteName}; +use smarm::cluster::{ + dispatch_any, members_all, pick_any, start, Config, DispatchAnyError, GroupMember, StaticSeeds, + Timing, +}; +use smarm::{channel, join, leave, members, register, send_to, spawn_addr, Addressable, Name, Pid}; +use std::time::{Duration, Instant}; + +const GO: Name = Name::new("go"); +const POOL: &str = "pool"; + +/// A pool worker's message: `"die"` stops it, anything else is printed. +#[derive(Debug, PartialEq)] +struct Job(String); +struct Worker; +impl Addressable for Worker { + type Msg = Job; +} +impl serde::Serialize for Job { + fn serialize(&self, s: S) -> Result { + self.0.serialize(s) + } +} +impl<'de> serde::Deserialize<'de> for Job { + fn deserialize>(d: D) -> Result { + String::deserialize(d).map(Job) + } +} + +const ROLES: &[(&str, fn())] = &[("origin", role_origin), ("observer", role_observer)]; + +fn cfg(name: &str, seeds: Vec<(String, String)>) -> Config { + Config { + node_name: name.into(), + meta: NodeMeta { + role: "c15".into(), + region: "local".into(), + }, + listen_addr: "127.0.0.1:0".into(), + strategy: Box::new(StaticSeeds::new(seeds)), + timing: Timing::default(), + } +} + +/// A pool worker: prints every job it is handed, exits on `"die"`. +fn worker() -> Pid { + spawn_addr::(|rx| { + while let Ok(Job(s)) = rx.recv() { + if s == "die" { + return; + } + println!("JOB {s}"); + } + }) +} + +fn role_origin() { + smarm::run(|| { + let cluster = start(cfg("origin", vec![])).expect("binds"); + // Remote dispatch lands here only for a type this node accepts. + expose_type::(); + let w1 = worker(); + let w2 = worker(); + assert!(join(POOL, w1)); + assert!(join(POOL, w2)); + let (go_tx, go_rx) = channel::(); + register(GO, go_tx).unwrap(); + expose(GO); + println!("LISTENING {}", cluster.local_addr()); + println!("JOINED 2"); + loop { + match go_rx.recv().unwrap() { + 1 => { + send_to(w1, Job("die".into())).unwrap(); + println!("KILLED w1"); + } + 2 => { + assert!(leave(POOL, w2)); + println!("LEFT w2"); + } + 3 => { + let w3 = worker(); + assert!(join(POOL, w3)); + println!("JOINED w3"); + } + n => panic!("unknown command {n}"), + } + } + }); +} + +fn remote_count(group: &str) -> usize { + members_all(group) + .iter() + .filter(|m| matches!(m, GroupMember::Remote(_))) + .count() +} + +/// Cooperative poll until `pred`; panics (with the last view) on timeout. +fn wait_view(what: &str, group: &str, pred: impl Fn(&[GroupMember]) -> bool) { + let deadline = Instant::now() + Duration::from_secs(5); + loop { + let v = members_all(group); + if pred(&v) { + return; + } + assert!( + Instant::now() < deadline, + "timed out waiting for {what}; view = {v:?}" + ); + smarm::sleep(Duration::from_millis(5)); + } +} + +fn role_observer() { + let origin_addr = std::env::var("SMARM_ORIGIN_ADDR").expect("SMARM_ORIGIN_ADDR"); + smarm::run(move || { + let _cluster = start(cfg("observer", vec![("origin".into(), origin_addr)])).expect("binds"); + let ev = subscribe().unwrap(); + loop { + match ev.rx.recv().unwrap() { + NodeEvent::NodeUp(i) if i.name == "origin" => break, + _ => {} + } + } + let go = |n: u8| remote::send(RemoteName::new("origin", GO), n).unwrap(); + + // Sync: both pre-existing members arrive with no join on this side. + wait_view("sync of 2 remote members", POOL, |v| { + v.len() == 2 && v.iter().all(|m| matches!(m, GroupMember::Remote(_))) + }); + let synced = members_all(POOL); + assert!(synced.iter().all(|m| match m { + GroupMember::Remote(p) => p.node() == "origin", + GroupMember::Local(_) => false, + })); + println!("SEES 2"); + + // Origin-side death: the origin's reaper announces the leave. + go(1); + wait_view("death evicted on observer", POOL, |v| v.len() == 1); + println!("SEES 1 after death"); + + // Incremental Join. + go(3); + wait_view("incremental join", POOL, |v| v.len() == 2); + println!("SEES 2 after join"); + + // Voluntary Leave. + go(2); + wait_view("incremental leave", POOL, |v| v.len() == 1); + println!("SEES 1 after leave"); + + // Mixed group: our own member sits beside the remote one in + // `members_all`; `members` stays local-only. + let me = worker(); + assert!(join(POOL, me)); + wait_view("mixed local+remote", POOL, |v| { + v.len() == 2 && v.contains(&GroupMember::Local(me.erase())) + }); + assert_eq!( + members(POOL), + vec![me.erase()], + "local API never shows remotes" + ); + assert_eq!(remote_count(POOL), 1); + println!("MIXED ok"); + + // dispatch_any: the store's first entry is the origin's w3 (it was + // announced before we joined), so the pick is remote and the job + // crosses the wire — the origin's worker prints it. + let picked = pick_any(POOL).expect("pool has members"); + assert!( + matches!(picked, GroupMember::Remote(_)), + "first entry is remote: {picked:?}" + ); + let reached = dispatch_any::(POOL, Job("from-observer".into())).unwrap(); + assert_eq!(reached, picked); + println!("DISPATCHED remote"); + + println!("PARK"); + // Parent SIGKILLs the origin now: NodeDown must sweep its member, + // ours must survive. + wait_view("node_down sweep", POOL, |v| { + v == [GroupMember::Local(me.erase())] + }); + assert_eq!(members(POOL), vec![me.erase()]); + println!("SWEPT"); + + // Now the only member is ours: a local pick, a local send. + let reached = dispatch_any::(POOL, Job("local".into())).unwrap(); + assert_eq!(reached, GroupMember::Local(me.erase())); + // And an empty group hands the message back. + match dispatch_any::("nobody", Job("lost".into())) { + Err(DispatchAnyError::NoMember(Job(s))) => assert_eq!(s, "lost"), + other => panic!("expected NoMember, got {other:?}"), + } + println!("DISPATCHED local"); + loop { + smarm::sleep(Duration::from_secs(3600)); + } + }); +} + +/// The Phase 5 gate: sync, join, leave, death, node_down — all observed from +/// the peer, none of them a group operation on the peer — plus dispatch_any +/// reaching a remote member and a local one. +#[test] +fn groups_span_two_nodes() { + maybe_child(ROLES); + let mut origin = spawn_node("origin", &[]); + let addr = origin.wait_listening(); + origin.wait_line("JOINED 2", |l| l == "JOINED 2"); + let mut observer = spawn_node("observer", &[("SMARM_ORIGIN_ADDR", &addr)]); + observer.wait_line("SEES 2", |l| l == "SEES 2"); + origin.wait_line("KILLED w1", |l| l == "KILLED w1"); + observer.wait_line("SEES 1 after death", |l| l == "SEES 1 after death"); + origin.wait_line("JOINED w3", |l| l == "JOINED w3"); + observer.wait_line("SEES 2 after join", |l| l == "SEES 2 after join"); + origin.wait_line("LEFT w2", |l| l == "LEFT w2"); + observer.wait_line("SEES 1 after leave", |l| l == "SEES 1 after leave"); + observer.wait_line("MIXED ok", |l| l == "MIXED ok"); + observer.wait_line("DISPATCHED remote", |l| l == "DISPATCHED remote"); + origin.wait_line("JOB from-observer", |l| l == "JOB from-observer"); + observer.wait_line("PARK", |l| l == "PARK"); + origin.kill(); + observer.wait_line("SWEPT", |l| l == "SWEPT"); + // Order between the root's line and the worker's is scheduling; wait + // for the later one to be certain both happened. + observer.wait_line("DISPATCHED local", |l| l == "DISPATCHED local"); + observer.wait_line("JOB local", |l| l == "JOB local"); +} diff --git a/tests/cluster_pid_send.rs b/tests/cluster_pid_send.rs new file mode 100644 index 0000000..9c1a074 --- /dev/null +++ b/tests/cluster_pid_send.rs @@ -0,0 +1,356 @@ +//! RFC 010 c10 — pid targeting + auto-serialization. The Phase 3 gate: +//! cross-node call/reply with no ceremony, under the subprocess harness. +//! +//! Local suite (`run()`, no network): serialize/deserialize shapes, +//! self-collapse, the outside-runtime contract, the local send-site +//! incarnation check with a probe proving **no frame is emitted**. +//! +//! Cross-process: two nodes. The *server* exposes a `Name`; the +//! *client* sends a `Req` carrying its own `Pid` (auto-serialized to +//! a `RemotePid` on the wire); the server replies via `send_to_remote` +//! straight back to that pid — no name at the client end, no ceremony. A +//! third-node roundtrip: the client's pid travels client→server→relay→ +//! server→client, and still delivers. +#![cfg(feature = "cluster")] + +mod common; + +use common::{maybe_child, spawn_node}; +use smarm::cluster::envelope::{encode_payload, Frame, NodeMeta}; +use smarm::cluster::expose::{expose, type_hash}; +use smarm::cluster::membership::{subscribe, NodeEvent}; +use smarm::cluster::remote::{self, send_to_remote, RemoteName, RemotePid, ToRemoteError}; +use smarm::cluster::{start, Config, StaticSeeds, Timing}; +use smarm::pg::Incarnation; +use smarm::{channel, install, register, run, Addressable, Name, Pid}; +use std::time::Duration; + +// ---- message types (std-only payloads; the crate's serde is derive-less, +// so wire types are hand-rolled with serde's tuple/seq API via `serde::ser` +// impls below — the same thing a user's derive would generate) ------------ + +/// A request carrying a reply-to. Serialize/Deserialize are written by hand +/// here for exactly one reason: this crate deliberately does not pull in +/// serde-derive. Field 1 is the auto-serializing pid. +#[derive(Debug, PartialEq)] +struct Req { + text: String, + reply_to: RemotePid, +} + +#[derive(Debug, PartialEq)] +struct Reply(String); + +struct Replier; +impl Addressable for Replier { + type Msg = Reply; +} + +impl serde::Serialize for Req { + fn serialize(&self, s: S) -> Result { + use serde::ser::SerializeTuple; + let mut t = s.serialize_tuple(2)?; + t.serialize_element(&self.text)?; + t.serialize_element(&self.reply_to)?; + t.end() + } +} +impl<'de> serde::Deserialize<'de> for Req { + fn deserialize>(d: D) -> Result { + let (text, reply_to) = <(String, RemotePid)>::deserialize(d)?; + Ok(Req { text, reply_to }) + } +} +impl serde::Serialize for Reply { + fn serialize(&self, s: S) -> Result { + self.0.serialize(s) + } +} +impl<'de> serde::Deserialize<'de> for Reply { + fn deserialize>(d: D) -> Result { + String::deserialize(d).map(Reply) + } +} + +// ================= local suite ========================================= + +/// A local `Pid` serializes as a `RemotePid` stamped with this node's +/// identity; deserializing it back on the same node collapses to the same +/// local pid (`local()` is `Some`, `Pid` round-trips). +#[test] +fn local_pid_serializes_and_collapses_on_self() { + maybe_child(ROLES); + run(|| { + // The local identity is set by cluster::start; the local suite sets + // it directly. + remote::set_local_identity("me", Incarnation::new(7)); + let (tx, _rx) = channel::(); + let me: Pid = install::(tx); + + let bytes = encode_payload(&me).unwrap(); + let rp: RemotePid = smarm::cluster::envelope::decode_payload(&bytes).unwrap(); + assert_eq!(rp.node(), "me"); + assert_eq!(rp.incarnation(), Incarnation::new(7)); + assert_eq!( + rp.local(), + Some(me), + "self-node pid collapses to the local pid" + ); + + // Deserializing straight into Pid works for a self-node pid... + let back: Pid = smarm::cluster::envelope::decode_payload(&bytes).unwrap(); + assert_eq!(back, me); + + // ...and FAILS for a foreign one (collapse is literal: node == self). + let foreign = RemotePid::::from_parts("elsewhere", Incarnation::new(1), 3, 1); + let fbytes = encode_payload(&foreign).unwrap(); + assert!(smarm::cluster::envelope::decode_payload::>(&fbytes).is_err()); + assert_eq!(foreign.local(), None); + }); +} + +/// `send_to_remote` short-circuits locally for a self-node pid — the +/// zero-copy-equivalent collapse: the message object itself lands in the +/// local channel, no encode, no frame. +#[test] +fn send_to_remote_collapses_locally_for_self() { + maybe_child(ROLES); + run(|| { + remote::set_local_identity("me", Incarnation::new(7)); + let (tx, rx) = channel::(); + let me: Pid = install::(tx); + let rp = RemotePid::from_local(me).expect("identity set"); + // Probe the outbound path: nothing must be handed to any connection. + let (probe_tx, probe_rx) = channel::(); + remote::bind_outbound_probe("me", Incarnation::new(7), probe_tx); + + send_to_remote(rp, Reply("hi".into())).unwrap(); + assert_eq!(rx.recv().unwrap(), Reply("hi".into())); + assert!( + matches!(probe_rx.try_recv(), Ok(None)), + "no frame for a local collapse" + ); + }); +} + +/// RFC v2 §3: a `RemotePid` whose incarnation is not the current one for its +/// node fails at the local send site with `DeadIncarnation`, and NO frame +/// is emitted — asserted on a probe sender bound as that node's outbound. +#[test] +fn stale_incarnation_rejected_locally_no_frame() { + maybe_child(ROLES); + run(|| { + remote::set_local_identity("me", Incarnation::new(7)); + let (probe_tx, probe_rx) = channel::(); + remote::bind_outbound_probe("peer", Incarnation::new(5), probe_tx); + + let stale = RemotePid::::from_parts("peer", Incarnation::new(4), 9, 1); + match send_to_remote(stale, Reply("late".into())) { + Err(ToRemoteError::DeadIncarnation(Reply(s))) => assert_eq!(s, "late"), + other => panic!("expected DeadIncarnation, got {other:?}"), + } + assert!( + matches!(probe_rx.try_recv(), Ok(None)), + "stale pid must emit no frame" + ); + + // The current incarnation goes through: a Send frame with the pid's + // (index, generation) and Reply's hash lands on the probe. + let live = RemotePid::::from_parts("peer", Incarnation::new(5), 9, 1); + send_to_remote(live, Reply("now".into())).unwrap(); + match probe_rx.recv().unwrap() { + Frame::Send { + index, + generation, + type_hash: h, + payload, + } => { + assert_eq!((index, generation), (9, 1)); + assert_eq!(h, type_hash::()); + let r: Reply = smarm::cluster::envelope::decode_payload(&payload).unwrap(); + assert_eq!(r, Reply("now".into())); + } + f => panic!("expected Send, got {f:?}"), + } + + // Unknown node: NotConnected, no frame anywhere. + let nowhere = RemotePid::::from_parts("nowhere", Incarnation::new(1), 1, 1); + assert!(matches!( + send_to_remote(nowhere, Reply("x".into())), + Err(ToRemoteError::NotConnected(_)) + )); + }); +} + +// ================= cross-process gate ================================== + +const ROLES: &[(&str, fn())] = &[ + ("server", role_server), + ("client", role_client), + ("relay", role_relay), +]; + +const ECHO: Name = Name::new("c10.echo"); +const RELAY: Name = Name::new("c10.relay"); + +fn cfg(name: &str, seeds: Vec<(String, String)>) -> Config { + Config { + node_name: name.to_string(), + meta: NodeMeta { + role: "c10".into(), + region: "local".into(), + }, + listen_addr: std::env::var("SMARM_LISTEN_ADDR").unwrap_or_else(|_| "127.0.0.1:0".into()), + strategy: Box::new(StaticSeeds::new(seeds)), + timing: Timing::default(), + } +} + +fn wait_up(events: &smarm::cluster::membership::MembershipEvents, who: &str) { + loop { + match events.rx.recv() { + Ok(NodeEvent::NodeUp(i)) if i.name == who => return, + Ok(_) => continue, + Err(_) => panic!("manager gone"), + } + } +} + +/// Server: exposes ECHO; each Req is answered by `send_to_remote` to its +/// reply_to — the server never learns a name for the client. If the Req text +/// starts with "via-relay:", it forwards the whole Req (reply_to and all) to +/// the relay node instead, which sends it back here; the second arrival is +/// answered normally. That is the pid's third-node roundtrip. +fn role_server() { + let relay_addr = std::env::var("SMARM_RELAY_ADDR").ok(); + smarm::run(move || { + let seeds = relay_addr + .map(|a| vec![("relay".to_string(), a)]) + .unwrap_or_default(); + let cluster = start(cfg("server", seeds)).expect("binds"); + println!("LISTENING {}", cluster.local_addr()); + let (tx, rx) = channel::(); + register(ECHO, tx).unwrap(); + expose(ECHO); + println!("READY"); + loop { + let req = rx.recv().unwrap(); + if let Some(rest) = req.text.strip_prefix("via-relay:") { + let fwd = Req { + text: format!("relayed:{rest}"), + reply_to: req.reply_to, + }; + remote::send(RemoteName::new("relay", RELAY), fwd).unwrap(); + println!("FORWARDED"); + continue; + } + println!("REQ {}", req.text); + send_to_remote(req.reply_to, Reply(format!("echo:{}", req.text))).unwrap(); + } + }); +} + +/// Relay: exposes RELAY; bounces every Req straight back to the server's +/// ECHO, untouched. The client's pid inside it now crosses relay→server. +fn role_relay() { + let server_addr = std::env::var("SMARM_SERVER_ADDR").expect("SMARM_SERVER_ADDR"); + smarm::run(move || { + let cluster = start(cfg("relay", vec![("server".into(), server_addr)])).expect("binds"); + println!("LISTENING {}", cluster.local_addr()); + let (tx, rx) = channel::(); + register(RELAY, tx).unwrap(); + expose(RELAY); + let ev = subscribe().unwrap(); + wait_up(&ev, "server"); + println!("READY"); + loop { + let req = rx.recv().unwrap(); + println!("RELAYING {}", req.text); + remote::send(RemoteName::new("server", ECHO), req).unwrap(); + } + }); +} + +/// Client: connects to server, installs a Reply inbox on its own pid, +/// declares it accepts `Reply` (`expose_type` — the RFC's one kept piece of +/// ceremony: nothing is remotely deliverable by default), sends a Req with +/// `reply_to = my pid` (auto-serialized), awaits the reply. +fn role_client() { + let server_addr = std::env::var("SMARM_SERVER_ADDR").expect("SMARM_SERVER_ADDR"); + let via_relay = std::env::var("SMARM_VIA_RELAY").is_ok(); + smarm::run(move || { + let _cluster = start(cfg("client", vec![("server".into(), server_addr)])).expect("binds"); + let ev = subscribe().unwrap(); + wait_up(&ev, "server"); + println!("MEMBER-UP server"); + + let (tx, rx) = channel::(); + let me: Pid = install::(tx); + // The one deliberate line: a pid-targeted inbound is deliverable only + // for types this node has said it accepts (RFC §4, the safety). + smarm::cluster::expose::expose_type::(); + let text = if via_relay { "via-relay:ping" } else { "ping" }; + remote::send( + RemoteName::new("server", ECHO), + Req { + text: text.into(), + reply_to: RemotePid::from_local(me).expect("identity set"), + }, + ) + .unwrap(); + println!("SENT"); + let Reply(s) = rx.recv().unwrap(); + println!("REPLY {s}"); + loop { + smarm::sleep(Duration::from_secs(3600)); + } + }); +} + +/// The gate: cross-node call/reply with no ceremony. +#[test] +fn cross_node_call_reply_no_ceremony() { + maybe_child(ROLES); + let mut server = spawn_node("server", &[]); + let saddr = server.wait_listening(); + server.wait_line("READY", |l| l == "READY"); + let mut client = spawn_node("client", &[("SMARM_SERVER_ADDR", &saddr)]); + client.wait_line("SENT", |l| l == "SENT"); + server.wait_line("REQ ping", |l| l == "REQ ping"); + client.wait_line("REPLY echo:ping", |l| l == "REPLY echo:ping"); +} + +/// The client's pid, round-tripped through a third node, still delivers. +#[test] +fn pid_roundtrips_through_third_node() { + maybe_child(ROLES); + // Relay needs the server address; server needs the relay address — + // pre-reserve the relay port (same accepted micro-window as cluster_mesh). + let relay_addr = { + let l = std::net::TcpListener::bind("127.0.0.1:0").unwrap(); + l.local_addr().unwrap().to_string() + }; + let mut server = spawn_node("server", &[("SMARM_RELAY_ADDR", &relay_addr)]); + let saddr = server.wait_listening(); + server.wait_line("READY", |l| l == "READY"); + let mut relay = spawn_node( + "relay", + &[ + ("SMARM_SERVER_ADDR", &saddr), + ("SMARM_LISTEN_ADDR", &relay_addr), + ], + ); + let _ = relay.wait_listening(); + relay.wait_line("READY", |l| l == "READY"); + let mut client = spawn_node( + "client", + &[("SMARM_SERVER_ADDR", &saddr), ("SMARM_VIA_RELAY", "1")], + ); + client.wait_line("SENT", |l| l == "SENT"); + server.wait_line("FORWARDED", |l| l == "FORWARDED"); + relay.wait_line("RELAYING", |l| l.starts_with("RELAYING")); + server.wait_line("REQ relayed:ping", |l| l == "REQ relayed:ping"); + client.wait_line("REPLY echo:relayed:ping", |l| { + l == "REPLY echo:relayed:ping" + }); +} diff --git a/tests/cluster_remote_send.rs b/tests/cluster_remote_send.rs index 2f6e1e1..fc93486 100644 --- a/tests/cluster_remote_send.rs +++ b/tests/cluster_remote_send.rs @@ -33,7 +33,7 @@ use smarm::cluster::envelope::NodeMeta; use smarm::cluster::expose::expose; use smarm::cluster::membership::{subscribe, NodeEvent}; use smarm::cluster::remote::{send_remote_raw, RemoteName, RemoteSendError}; -use smarm::cluster::{start, Config, StaticSeeds}; +use smarm::cluster::{start, Config, StaticSeeds, Timing}; use smarm::{channel, register, Name}; use std::time::Duration; @@ -51,6 +51,7 @@ fn base_config(name: &str, seeds: Vec<(String, String)>) -> Config { }, listen_addr: "127.0.0.1:0".to_string(), strategy: Box::new(StaticSeeds::new(seeds)), + timing: Timing::default(), } } diff --git a/tests/pg.rs b/tests/pg.rs index 180ae10..3b06bbe 100644 --- a/tests/pg.rs +++ b/tests/pg.rs @@ -1,5 +1,6 @@ //! Process-group tests that run under the scheduler: `join` installs a real -//! monitor on a live actor, and a real death drives eviction on next contact. +//! monitor on a live actor, and a real death drives eviction (the reaper +//! actor sweeps it; the read path hides it in the meantime). //! (Pure structural invariants live in the `pg` unit tests.) use smarm::{channel, members, pick, run, spawn}; @@ -35,19 +36,15 @@ fn a_dead_actor_vanishes_from_every_group_it_joined() { assert_eq!(members("g1"), vec![pid]); assert_eq!(members("g2"), vec![pid]); - // Release and reap the actor. finalize_actor queues the Down to our - // monitors before unparking joiners, so by the time join() returns the - // Down is already waiting in the membership channel. + // Release the actor. finalize_actor queues the Down to the reaper and + // marks the slot dead before unparking joiners, so by the time join() + // returns every read hides the pid whether or not the reaper has run. tx.send(()).unwrap(); h.join().unwrap(); - // Drain-on-contact: touching g1 detects the death and sweeps the pid - // out of every group (g2 included), not just g1. - assert!(members("g1").is_empty(), "evicted from the touched group"); - assert!( - members("g2").is_empty(), - "and swept from the untouched group" - ); + // Gone from every group it joined, not just one. + assert!(members("g1").is_empty(), "gone from g1"); + assert!(members("g2").is_empty(), "and from g2"); assert_eq!(pick("g1"), None); }); } @@ -86,11 +83,7 @@ fn live_members_survive_a_peers_death() { tx_a.send(()).unwrap(); a.join().unwrap(); - assert_eq!( - members("svc"), - vec![b.pid()], - "only the dead peer is reaped" - ); + assert_eq!(members("svc"), vec![b.pid()], "only the dead peer is gone"); assert_eq!(pick("svc"), Some(b.pid())); tx_b.send(()).unwrap(); @@ -123,18 +116,18 @@ fn leave_drops_a_membership_without_affecting_others() { } #[test] -fn joining_an_already_dead_pid_is_evicted_on_next_contact() { +fn joining_an_already_dead_pid_never_shows_in_a_read() { run(|| { let h = spawn(|| {}); let pid = h.pid(); h.join().unwrap(); // actor is finalized before we join it to anything - // monitor() on a gone pid queues a NoProc Down immediately, so the - // membership is reaped the next time the group is touched. + // join() on a gone pid queues a NoProc Down to the reaper immediately; + // reads never show it either way (slot-liveness backstop). join("late", pid); assert!( members("late").is_empty(), - "dead-at-join member is reaped on read" + "dead-at-join member never reads as live" ); assert_eq!(pick("late"), None); });