From 58a2fe3046725a48ad9cb0448e658c8978864a0a Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 14 Aug 2026 13:20:20 +0000 Subject: [PATCH 01/14] =?UTF-8?q?feat(cluster):=20RFC=20010=20c1=20?= =?UTF-8?q?=E2=80=94=20cluster=20feature=20flag=20+=20optional=20serde/pos?= =?UTF-8?q?tcard=20deps?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Off by default; the default build stays libc-only. serde (payload contract) and postcard (payload codec) are optional, no default features. src/cluster.rs is an intentionally empty cfg-gated stub so the flag's default-build invariance is reviewable in isolation. --- Cargo.toml | 8 ++++++++ src/cluster.rs | 6 ++++++ src/lib.rs | 2 ++ 3 files changed, 16 insertions(+) create mode 100644 src/cluster.rs diff --git a/Cargo.toml b/Cargo.toml index 8b101b6..362b77d 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -33,6 +33,11 @@ budget-accounting = [] # and unflagged; only the optional gen_server transport sits behind this, so a # release build pays nothing for an observer it never starts. observer = [] +# RFC 010 c1: clustering. Off by default — the default build stays libc-only, +# byte-for-byte (gate checked per phase). serde is the payload contract, +# postcard the payload codec; both minimal (no default features). Everything +# cluster-shaped lives behind this flag. +cluster = ["dep:serde", "dep:postcard"] # Run-queue selection: exactly one, compile-time (see src/run_queue.rs). # Non-default variants need --no-default-features (features are additive). rq-mutex = [] @@ -44,6 +49,9 @@ cc = "1" [dependencies] libc = "0.2" +# RFC 010 §2 — only compiled under `--features cluster`. +serde = { version = "1", default-features = false, optional = true } +postcard = { version = "1", default-features = false, optional = true } [target.'cfg(loom)'.dependencies] loom = "0.7" diff --git a/src/cluster.rs b/src/cluster.rs new file mode 100644 index 0000000..9326694 --- /dev/null +++ b/src/cluster.rs @@ -0,0 +1,6 @@ +//! RFC 010 — clustering (smarm⇄smarm, explicit remote boundary). +//! +//! c1: feature flag + optional deps only. The owned envelope (c2), transport +//! trait (c3), and everything above them land in later chunks. This module is +//! intentionally empty so the `cluster` feature's default-build invariance is +//! reviewable in isolation. diff --git a/src/lib.rs b/src/lib.rs index 7dfb3f4..3a6ee80 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -14,6 +14,8 @@ pub mod actor; pub mod causal; pub mod channel; +#[cfg(feature = "cluster")] +pub mod cluster; pub mod context; pub mod gen_server; pub mod gen_statem; From 3850f6099bc58a7bf5767cfd83ef9a30345d7bab Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 14 Aug 2026 13:23:50 +0000 Subject: [PATCH 02/14] =?UTF-8?q?feat(cluster):=20RFC=20010=20c2=20?= =?UTF-8?q?=E2=80=94=20owned=20wire=20envelope?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Frame enum per the RFC inventory; hand-rolled encode/decode with u32 LE length prefix + u8 tag; strings u16-prefixed, payload blobs u32-prefixed; MAX_FRAME_LEN cap (control plane never carries bulk, §5). Streaming decode: Ok(None) = need more bytes, every Err = corruption. postcard confined to encode_payload/decode_payload — the single codec seam (§2); postcard gains the alloc feature for to_allocvec (still no_std-aligned, no default features). DownReason/RejectReason travel as single tag bytes; Down tag 5 is reserved for c11's Disconnected. Tests: per-frame roundtrip, back-to-back frames, golden heartbeat bytes, zero-length payload, every-prefix incomplete, unknown frame/enum tags, length prefix lying long (with and without bytes present) and short, truncation mid-string, adversarial lengths (u32::MAX, cap+1, zero), UTF-8 corruption, payload seam roundtrip through a real Send frame. --- Cargo.toml | 5 +- src/cluster.rs | 2 + src/cluster/envelope.rs | 506 ++++++++++++++++++++++++++++++++++++++ tests/cluster_envelope.rs | 271 ++++++++++++++++++++ 4 files changed, 783 insertions(+), 1 deletion(-) create mode 100644 src/cluster/envelope.rs create mode 100644 tests/cluster_envelope.rs diff --git a/Cargo.toml b/Cargo.toml index 362b77d..1fe9d06 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -51,13 +51,16 @@ cc = "1" libc = "0.2" # RFC 010 §2 — only compiled under `--features cluster`. serde = { version = "1", default-features = false, optional = true } -postcard = { version = "1", default-features = false, optional = true } +# `alloc` (not `std`): the seam serializes to Vec; postcard stays no_std-aligned. +postcard = { version = "1", default-features = false, features = ["alloc"], optional = true } [target.'cfg(loom)'.dependencies] loom = "0.7" [dev-dependencies] libc = "0.2" +# derive + std for cluster envelope tests only; the lib itself never needs them +serde = { version = "1", features = ["derive"] } tokio = { version = "1", features = ["rt", "rt-multi-thread", "macros", "sync", "time"] } [profile.dev] diff --git a/src/cluster.rs b/src/cluster.rs index 9326694..b5953ad 100644 --- a/src/cluster.rs +++ b/src/cluster.rs @@ -4,3 +4,5 @@ //! trait (c3), and everything above them land in later chunks. This module is //! intentionally empty so the `cluster` feature's default-build invariance is //! reviewable in isolation. + +pub mod envelope; diff --git a/src/cluster/envelope.rs b/src/cluster/envelope.rs new file mode 100644 index 0000000..445374c --- /dev/null +++ b/src/cluster/envelope.rs @@ -0,0 +1,506 @@ +//! RFC 010 c2 — the owned wire envelope. +//! +//! Every control-plane frame is `u32` little-endian length prefix (of tag + +//! body), `u8` tag, hand-encoded body. postcard appears in exactly one place: +//! the payload blob inside `Send`/`SendNamed`, via [`encode_payload`] / +//! [`decode_payload`] — the seam where a codec swap would land (RFC 010 §2). +//! Everything else is hand-rolled and wholly owned. +//! +//! Integers are little-endian. Strings are `u16` length + UTF-8 bytes. +//! Payload blobs are `u32` length + bytes. Enum-shaped fields +//! ([`RejectReason`], [`DownReason`]) are a single tag byte. + +use crate::monitor::DownReason; +use crate::pg::Incarnation; + +/// Wire protocol version, checked in the handshake (c5). +pub const PROTO_VERSION: u32 = 1; + +/// Hard cap on the length prefix. The control plane never carries bulk data +/// (RFC 010 §5 — that is the jarred rkyv plane), so anything larger is +/// corruption or an attack, not a legitimate frame. +pub const MAX_FRAME_LEN: usize = 16 * 1024 * 1024; + +/// Per-node metadata exchanged in the handshake (RFC 010 §1: not identity). +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct NodeMeta { + pub role: String, + pub region: String, +} + +/// Why a `Hello` was rejected. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum RejectReason { + /// Build hashes differ — not the same binary. + HashMismatch, + /// The offered node name is already claimed by a live peer. + NameTaken, + /// Wire protocol version mismatch. + ProtoVersion, +} + +/// The control-plane frame inventory (RFC 010, *Implementation details*). +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum Frame { + Hello { + proto_version: u32, + build_hash: u64, + node_name: String, + incarnation: Incarnation, + meta: NodeMeta, + }, + HelloAck { + node_name: String, + incarnation: Incarnation, + meta: NodeMeta, + }, + HelloReject { + reason: RejectReason, + }, + Heartbeat, + Send { + /// Target slot index (node is implicit in the connection, incarnation + /// is bound at handshake — RFC 010 §3). + index: u32, + generation: u32, + type_hash: u64, + payload: Vec, + }, + SendNamed { + name: String, + type_hash: u64, + payload: Vec, + }, + Monitor { + monitor_id: u64, + index: u32, + generation: u32, + }, + Demonitor { + monitor_id: u64, + }, + Down { + monitor_id: u64, + reason: DownReason, + }, +} + +// Frame tags. 0 is deliberately unassigned so an all-zero buffer never parses. +const TAG_HELLO: u8 = 1; +const TAG_HELLO_ACK: u8 = 2; +const TAG_HELLO_REJECT: u8 = 3; +const TAG_HEARTBEAT: u8 = 4; +const TAG_SEND: u8 = 5; +const TAG_SEND_NAMED: u8 = 6; +const TAG_MONITOR: u8 = 7; +const TAG_DEMONITOR: u8 = 8; +const TAG_DOWN: u8 = 9; + +// RejectReason tags. +const REJ_HASH_MISMATCH: u8 = 1; +const REJ_NAME_TAKEN: u8 = 2; +const REJ_PROTO_VERSION: u8 = 3; + +// DownReason tags. c11 adds `Disconnected = 5`; do not reuse tags. +const DR_EXIT: u8 = 1; +const DR_PANIC: u8 = 2; +const DR_STOPPED: u8 = 3; +const DR_NOPROC: u8 = 4; + +/// Frame could not be encoded. The output buffer is left exactly as it was. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum EncodeError { + /// tag + body exceed [`MAX_FRAME_LEN`]. + FrameTooLarge { len: usize }, + /// A string field exceeds `u16::MAX` bytes. + StringTooLong { len: usize }, +} + +impl core::fmt::Display for EncodeError { + fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { + match self { + Self::FrameTooLarge { len } => { + write!(f, "frame body of {len} bytes exceeds MAX_FRAME_LEN") + } + Self::StringTooLong { len } => { + write!(f, "string field of {len} bytes exceeds u16::MAX") + } + } + } +} + +impl std::error::Error for EncodeError {} + +/// Frame could not be decoded. Everything here is *corruption* — "not enough +/// bytes yet" is the `Ok(None)` streaming case, never an error. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum DecodeError { + /// The length prefix exceeds [`MAX_FRAME_LEN`]. + FrameTooLarge { declared: usize }, + /// The length prefix is zero — there is no tag byte. + EmptyFrame, + /// Unknown frame tag. + UnknownTag(u8), + /// Unknown tag for an enum-shaped field. + UnknownEnumTag { what: &'static str, tag: u8 }, + /// A field ran past the declared frame end (the length prefix lied long, + /// or a length-carrying field inside the body lied). + Truncated, + /// Bytes were left over after the body (the length prefix lied short). + Trailing { extra: usize }, + /// A string field was not valid UTF-8. + Utf8, +} + +impl core::fmt::Display for DecodeError { + fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { + match self { + Self::FrameTooLarge { declared } => { + write!(f, "declared frame length {declared} exceeds MAX_FRAME_LEN") + } + Self::EmptyFrame => write!(f, "zero-length frame (no tag byte)"), + Self::UnknownTag(t) => write!(f, "unknown frame tag {t}"), + Self::UnknownEnumTag { what, tag } => write!(f, "unknown {what} tag {tag}"), + Self::Truncated => write!(f, "frame body truncated mid-field"), + Self::Trailing { extra } => write!(f, "{extra} trailing bytes after frame body"), + Self::Utf8 => write!(f, "string field is not valid UTF-8"), + } + } +} + +impl std::error::Error for DecodeError {} + +impl Frame { + /// Append this frame, length-prefixed, to `out`. + /// + /// On error `out` is left untouched. + pub fn encode(&self, out: &mut Vec) -> Result<(), EncodeError> { + let start = out.len(); + out.extend_from_slice(&[0u8; 4]); // length placeholder, patched below + let result = self.encode_body(out); + match result { + Ok(()) => { + let frame_len = out.len() - start - 4; + if frame_len > MAX_FRAME_LEN { + out.truncate(start); + return Err(EncodeError::FrameTooLarge { len: frame_len }); + } + // Cast is lossless: MAX_FRAME_LEN < u32::MAX, checked above. + let len32 = frame_len as u32; + out[start..start + 4].copy_from_slice(&len32.to_le_bytes()); + Ok(()) + } + Err(e) => { + out.truncate(start); + Err(e) + } + } + } + + fn encode_body(&self, out: &mut Vec) -> Result<(), EncodeError> { + match self { + Frame::Hello { + proto_version, + build_hash, + node_name, + incarnation, + meta, + } => { + out.push(TAG_HELLO); + put_u32(out, *proto_version); + put_u64(out, *build_hash); + put_str(out, node_name)?; + put_u32(out, incarnation.get()); + put_meta(out, meta)?; + } + Frame::HelloAck { + node_name, + incarnation, + meta, + } => { + out.push(TAG_HELLO_ACK); + put_str(out, node_name)?; + put_u32(out, incarnation.get()); + put_meta(out, meta)?; + } + Frame::HelloReject { reason } => { + out.push(TAG_HELLO_REJECT); + out.push(match reason { + RejectReason::HashMismatch => REJ_HASH_MISMATCH, + RejectReason::NameTaken => REJ_NAME_TAKEN, + RejectReason::ProtoVersion => REJ_PROTO_VERSION, + }); + } + Frame::Heartbeat => out.push(TAG_HEARTBEAT), + Frame::Send { + index, + generation, + type_hash, + payload, + } => { + out.push(TAG_SEND); + put_u32(out, *index); + put_u32(out, *generation); + put_u64(out, *type_hash); + put_blob(out, payload)?; + } + Frame::SendNamed { + name, + type_hash, + payload, + } => { + out.push(TAG_SEND_NAMED); + put_str(out, name)?; + put_u64(out, *type_hash); + put_blob(out, payload)?; + } + Frame::Monitor { + monitor_id, + index, + generation, + } => { + out.push(TAG_MONITOR); + put_u64(out, *monitor_id); + put_u32(out, *index); + put_u32(out, *generation); + } + Frame::Demonitor { monitor_id } => { + out.push(TAG_DEMONITOR); + put_u64(out, *monitor_id); + } + Frame::Down { monitor_id, reason } => { + out.push(TAG_DOWN); + put_u64(out, *monitor_id); + out.push(match reason { + DownReason::Exit => DR_EXIT, + DownReason::Panic => DR_PANIC, + DownReason::Stopped => DR_STOPPED, + DownReason::NoProc => DR_NOPROC, + }); + } + } + Ok(()) + } + + /// Try to decode one frame from the start of `buf`. + /// + /// `Ok(Some((frame, consumed)))` — a full frame; the caller advances by + /// `consumed`. `Ok(None)` — not enough bytes yet (streaming); read more + /// and retry. `Err(_)` — the bytes are corrupt; the connection is dead. + pub fn decode(buf: &[u8]) -> Result, DecodeError> { + let Some(prefix) = buf.get(0..4) else { + return Ok(None); + }; + let mut len4 = [0u8; 4]; + len4.copy_from_slice(prefix); + let declared = u32::from_le_bytes(len4) as usize; + if declared > MAX_FRAME_LEN { + return Err(DecodeError::FrameTooLarge { declared }); + } + if declared == 0 { + return Err(DecodeError::EmptyFrame); + } + let Some(body) = buf.get(4..4 + declared) else { + return Ok(None); + }; + let mut r = Reader { buf: body, pos: 0 }; + let frame = Self::decode_body(&mut r)?; + if r.pos != body.len() { + return Err(DecodeError::Trailing { + extra: body.len() - r.pos, + }); + } + Ok(Some((frame, 4 + declared))) + } + + fn decode_body(r: &mut Reader<'_>) -> Result { + let tag = r.u8()?; + let frame = match tag { + TAG_HELLO => Frame::Hello { + proto_version: r.u32()?, + build_hash: r.u64()?, + node_name: r.string()?, + incarnation: Incarnation::new(r.u32()?), + meta: r.meta()?, + }, + TAG_HELLO_ACK => Frame::HelloAck { + node_name: r.string()?, + incarnation: Incarnation::new(r.u32()?), + meta: r.meta()?, + }, + TAG_HELLO_REJECT => Frame::HelloReject { + reason: match r.u8()? { + REJ_HASH_MISMATCH => RejectReason::HashMismatch, + REJ_NAME_TAKEN => RejectReason::NameTaken, + REJ_PROTO_VERSION => RejectReason::ProtoVersion, + t => { + return Err(DecodeError::UnknownEnumTag { + what: "RejectReason", + tag: t, + }) + } + }, + }, + TAG_HEARTBEAT => Frame::Heartbeat, + TAG_SEND => Frame::Send { + index: r.u32()?, + generation: r.u32()?, + type_hash: r.u64()?, + payload: r.blob()?, + }, + TAG_SEND_NAMED => Frame::SendNamed { + name: r.string()?, + type_hash: r.u64()?, + payload: r.blob()?, + }, + TAG_MONITOR => Frame::Monitor { + monitor_id: r.u64()?, + index: r.u32()?, + generation: r.u32()?, + }, + TAG_DEMONITOR => Frame::Demonitor { + monitor_id: r.u64()?, + }, + TAG_DOWN => Frame::Down { + monitor_id: r.u64()?, + reason: match r.u8()? { + DR_EXIT => DownReason::Exit, + DR_PANIC => DownReason::Panic, + DR_STOPPED => DownReason::Stopped, + DR_NOPROC => DownReason::NoProc, + t => { + return Err(DecodeError::UnknownEnumTag { + what: "DownReason", + tag: t, + }) + } + }, + }, + t => return Err(DecodeError::UnknownTag(t)), + }; + Ok(frame) + } +} + +// --------------------------------------------------------------------------- +// Body writers +// --------------------------------------------------------------------------- + +fn put_u32(out: &mut Vec, v: u32) { + out.extend_from_slice(&v.to_le_bytes()); +} + +fn put_u64(out: &mut Vec, v: u64) { + out.extend_from_slice(&v.to_le_bytes()); +} + +fn put_str(out: &mut Vec, s: &str) -> Result<(), EncodeError> { + let Ok(len) = u16::try_from(s.len()) else { + return Err(EncodeError::StringTooLong { len: s.len() }); + }; + out.extend_from_slice(&len.to_le_bytes()); + out.extend_from_slice(s.as_bytes()); + Ok(()) +} + +fn put_blob(out: &mut Vec, b: &[u8]) -> Result<(), EncodeError> { + let Ok(len) = u32::try_from(b.len()) else { + return Err(EncodeError::FrameTooLarge { len: b.len() }); + }; + out.extend_from_slice(&len.to_le_bytes()); + out.extend_from_slice(b); + Ok(()) +} + +fn put_meta(out: &mut Vec, m: &NodeMeta) -> Result<(), EncodeError> { + put_str(out, &m.role)?; + put_str(out, &m.region) +} + +// --------------------------------------------------------------------------- +// Body reader +// --------------------------------------------------------------------------- + +struct Reader<'a> { + buf: &'a [u8], + pos: usize, +} + +impl Reader<'_> { + fn take(&mut self, n: usize) -> Result<&[u8], DecodeError> { + let end = self.pos.checked_add(n).ok_or(DecodeError::Truncated)?; + let s = self.buf.get(self.pos..end).ok_or(DecodeError::Truncated)?; + self.pos = end; + Ok(s) + } + + fn u8(&mut self) -> Result { + Ok(self.take(1)?[0]) + } + + fn u16(&mut self) -> Result { + let mut b = [0u8; 2]; + b.copy_from_slice(self.take(2)?); + Ok(u16::from_le_bytes(b)) + } + + fn u32(&mut self) -> Result { + let mut b = [0u8; 4]; + b.copy_from_slice(self.take(4)?); + Ok(u32::from_le_bytes(b)) + } + + fn u64(&mut self) -> Result { + let mut b = [0u8; 8]; + b.copy_from_slice(self.take(8)?); + Ok(u64::from_le_bytes(b)) + } + + fn string(&mut self) -> Result { + let len = self.u16()? as usize; + let bytes = self.take(len)?; + match core::str::from_utf8(bytes) { + Ok(s) => Ok(s.to_owned()), + Err(_) => Err(DecodeError::Utf8), + } + } + + fn blob(&mut self) -> Result, DecodeError> { + let len = self.u32()? as usize; + Ok(self.take(len)?.to_vec()) + } + + fn meta(&mut self) -> Result { + Ok(NodeMeta { + role: self.string()?, + region: self.string()?, + }) + } +} + +// --------------------------------------------------------------------------- +// The postcard seam (RFC 010 §2) — the ONLY place payload bytes are produced +// or consumed. A codec swap lands here and nowhere else. +// --------------------------------------------------------------------------- + +/// Payload (de)serialization failed at the codec seam. +#[derive(Debug)] +pub struct PayloadError(String); + +impl core::fmt::Display for PayloadError { + fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { + write!(f, "payload codec: {}", self.0) + } +} + +impl std::error::Error for PayloadError {} + +/// Serialize a payload value to the wire blob. +pub fn encode_payload(value: &T) -> Result, PayloadError> { + postcard::to_allocvec(value).map_err(|e| PayloadError(e.to_string())) +} + +/// Deserialize a payload value from the wire blob. +pub fn decode_payload(bytes: &[u8]) -> Result { + postcard::from_bytes(bytes).map_err(|e| PayloadError(e.to_string())) +} diff --git a/tests/cluster_envelope.rs b/tests/cluster_envelope.rs new file mode 100644 index 0000000..aba1773 --- /dev/null +++ b/tests/cluster_envelope.rs @@ -0,0 +1,271 @@ +//! RFC 010 c2 — owned envelope tests (roadmap: per-frame roundtrip, +//! truncation mid-field, unknown tag, length prefix lying long and short, +//! zero-length payload, adversarial lengths). +#![cfg(feature = "cluster")] + +use serde::{Deserialize, Serialize}; +use smarm::cluster::envelope::{ + decode_payload, encode_payload, DecodeError, Frame, NodeMeta, RejectReason, MAX_FRAME_LEN, + PROTO_VERSION, +}; +use smarm::monitor::DownReason; +use smarm::pg::Incarnation; + +fn meta() -> NodeMeta { + NodeMeta { + role: "worker".into(), + region: "eu-west".into(), + } +} + +fn all_frames() -> Vec { + vec![ + Frame::Hello { + proto_version: PROTO_VERSION, + build_hash: 0xDEAD_BEEF_CAFE_F00D, + node_name: "alpha".into(), + incarnation: Incarnation::new(7), + meta: meta(), + }, + Frame::HelloAck { + node_name: "beta".into(), + incarnation: Incarnation::new(9), + meta: meta(), + }, + Frame::HelloReject { + reason: RejectReason::NameTaken, + }, + Frame::Heartbeat, + Frame::Send { + index: 42, + generation: 3, + type_hash: 0x1234_5678_9ABC_DEF0, + payload: vec![1, 2, 3, 4, 5], + }, + Frame::SendNamed { + name: "the_counter".into(), + type_hash: 0xFFFF_0000_FFFF_0000, + payload: vec![], + }, + Frame::Monitor { + monitor_id: 77, + index: 42, + generation: 3, + }, + Frame::Demonitor { monitor_id: 77 }, + Frame::Down { + monitor_id: 77, + reason: DownReason::Panic, + }, + ] +} + +fn encode_one(f: &Frame) -> Vec { + let mut buf = Vec::new(); + f.encode(&mut buf).unwrap(); + buf +} + +#[test] +fn per_frame_roundtrip() { + for f in all_frames() { + let buf = encode_one(&f); + let (decoded, consumed) = Frame::decode(&buf).unwrap().unwrap(); + assert_eq!(decoded, f, "roundtrip mismatch"); + assert_eq!(consumed, buf.len(), "consumed != buffer length for {f:?}"); + } +} + +#[test] +fn back_to_back_frames_decode_sequentially() { + let mut buf = Vec::new(); + for f in all_frames() { + f.encode(&mut buf).unwrap(); + } + let mut off = 0; + let mut decoded = Vec::new(); + while off < buf.len() { + let (f, n) = Frame::decode(&buf[off..]).unwrap().unwrap(); + decoded.push(f); + off += n; + } + assert_eq!(decoded, all_frames()); + assert_eq!(off, buf.len()); +} + +#[test] +fn heartbeat_golden_bytes() { + // Locks the layout: u32 LE length prefix, then the tag byte. + let buf = encode_one(&Frame::Heartbeat); + assert_eq!(buf, vec![1, 0, 0, 0, 4]); +} + +#[test] +fn zero_length_payload_roundtrips() { + let f = Frame::Send { + index: 0, + generation: 0, + type_hash: 0, + payload: vec![], + }; + let buf = encode_one(&f); + let (decoded, consumed) = Frame::decode(&buf).unwrap().unwrap(); + assert_eq!(decoded, f); + assert_eq!(consumed, buf.len()); +} + +#[test] +fn incomplete_is_none_not_error() { + let buf = encode_one(&all_frames()[0]); + // Every strict prefix short of the full frame must report "need more". + for cut in 0..buf.len() { + assert_eq!( + Frame::decode(&buf[..cut]).unwrap(), + None, + "cut at {cut} should be incomplete" + ); + } +} + +#[test] +fn unknown_frame_tag() { + let buf = vec![1, 0, 0, 0, 250]; + assert_eq!(Frame::decode(&buf), Err(DecodeError::UnknownTag(250))); +} + +#[test] +fn unknown_enum_tags() { + // HelloReject with a bogus reason tag. + let buf = vec![2, 0, 0, 0, 3, 99]; + assert_eq!( + Frame::decode(&buf), + Err(DecodeError::UnknownEnumTag { + what: "RejectReason", + tag: 99 + }) + ); + // Down with a bogus reason tag (id = 0u64). + let mut buf = vec![10, 0, 0, 0, 9]; + buf.extend_from_slice(&0u64.to_le_bytes()); + buf.push(200); + assert_eq!( + Frame::decode(&buf), + Err(DecodeError::UnknownEnumTag { + what: "DownReason", + tag: 200 + }) + ); +} + +#[test] +fn length_prefix_lying_long_with_bytes_present_is_trailing() { + let mut buf = encode_one(&Frame::Heartbeat); + // Declare 3 extra body bytes and actually supply them. + let declared = u32::from_le_bytes([buf[0], buf[1], buf[2], buf[3]]) + 3; + buf[0..4].copy_from_slice(&declared.to_le_bytes()); + buf.extend_from_slice(&[0xAA, 0xBB, 0xCC]); + assert_eq!(Frame::decode(&buf), Err(DecodeError::Trailing { extra: 3 })); +} + +#[test] +fn length_prefix_lying_long_without_bytes_is_incomplete() { + // Indistinguishable from a partial read — must be None, not an error. + let mut buf = encode_one(&Frame::Heartbeat); + let declared = u32::from_le_bytes([buf[0], buf[1], buf[2], buf[3]]) + 3; + buf[0..4].copy_from_slice(&declared.to_le_bytes()); + assert_eq!(Frame::decode(&buf).unwrap(), None); +} + +#[test] +fn length_prefix_lying_short_truncates_a_field() { + let f = &all_frames()[0]; // Hello: plenty of fields to cut into + let mut buf = encode_one(f); + let declared = u32::from_le_bytes([buf[0], buf[1], buf[2], buf[3]]); + let lie = declared - 4; // cut mid-field + buf[0..4].copy_from_slice(&lie.to_le_bytes()); + assert_eq!(Frame::decode(&buf), Err(DecodeError::Truncated)); +} + +#[test] +fn truncation_mid_string_field() { + // A frame whose declared length is intact but whose inner string length + // runs past the body: SendNamed claiming a 1000-byte name in a tiny body. + let mut body = vec![6u8]; // TAG_SEND_NAMED + body.extend_from_slice(&1000u16.to_le_bytes()); + body.extend_from_slice(b"short"); + let mut buf = (body.len() as u32).to_le_bytes().to_vec(); + buf.extend_from_slice(&body); + assert_eq!(Frame::decode(&buf), Err(DecodeError::Truncated)); +} + +#[test] +fn adversarial_lengths() { + // Length prefix of u32::MAX: reject as oversized, do not wait for 4 GiB. + let buf = [0xFF, 0xFF, 0xFF, 0xFF, 0]; + assert_eq!( + Frame::decode(&buf), + Err(DecodeError::FrameTooLarge { + declared: u32::MAX as usize + }) + ); + // Just over the cap: also rejected. + let over = (MAX_FRAME_LEN as u32 + 1).to_le_bytes(); + assert!(matches!( + Frame::decode(&over), + Err(DecodeError::FrameTooLarge { .. }) + )); + // Zero-length frame: there is no tag byte; corrupt, not incomplete. + let buf = [0, 0, 0, 0]; + assert_eq!(Frame::decode(&buf), Err(DecodeError::EmptyFrame)); +} + +#[test] +fn invalid_utf8_in_string_field() { + let mut buf = encode_one(&Frame::SendNamed { + name: "abcd".into(), + type_hash: 0, + payload: vec![], + }); + // name bytes start after: 4 (len) + 1 (tag) + 2 (str len) = offset 7 + buf[7] = 0xFF; + assert_eq!(Frame::decode(&buf), Err(DecodeError::Utf8)); +} + +#[derive(Debug, PartialEq, Serialize, Deserialize)] +struct Ping { + seq: u64, + label: String, +} + +#[test] +fn payload_seam_roundtrip() { + let ping = Ping { + seq: 31337, + label: "hello".into(), + }; + let blob = encode_payload(&ping).unwrap(); + // Carry it through a real frame, as it will travel in c9. + let f = Frame::Send { + index: 1, + generation: 1, + type_hash: 0xABCD, + payload: blob, + }; + let buf = encode_one(&f); + let (decoded, _) = Frame::decode(&buf).unwrap().unwrap(); + let Frame::Send { payload, .. } = decoded else { + panic!("wrong frame"); + }; + let back: Ping = decode_payload(&payload).unwrap(); + assert_eq!(back, ping); +} + +#[test] +fn payload_seam_rejects_truncated_blob() { + let blob = encode_payload(&Ping { + seq: 1, + label: "x".into(), + }) + .unwrap(); + assert!(decode_payload::(&blob[..blob.len() - 1]).is_err()); +} From 39ab92871e4364dab17d2717488f49e2873d729e Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 14 Aug 2026 14:31:39 +0000 Subject: [PATCH 03/14] =?UTF-8?q?feat(cluster):=20RFC=20010=20c3=20?= =?UTF-8?q?=E2=80=94=20transport=20trait,=20framed=20codec,=20TCP=20+=20lo?= =?UTF-8?q?opback=20impls?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The control-connection abstraction (RFC v2 §5): object-safe Transport/ Listener/Conn over opaque pre-resolved addresses (resolution stays the c9 seam), with FramedConn as the single shared byte->Frame codec feeding Frame::decode's incremental contract. Nothing forecloses additional per-peer connections for the jarred bulk plane; the membrane is not a transport (D2). TCP parks the calling actor via scheduler fd readiness (MSG_NOSIGNAL writes, EINPROGRESS dial resolved through SO_ERROR). Loopback is the shipped in-memory test transport: OS-thread-blocking condvar pipes with TCP-shaped close semantics, per-instance address registry. Conformance suite runs the same codec over both impls: roundtrips both directions, framing across split writes, coalesced frames, peer-close mid-frame as TruncatedByPeer (not EOF), clean close as Ok(None). Plus impl-specific establishment/error cases and a 4 MiB cross-buffer TCP frame under real backpressure. --- src/cluster.rs | 8 +- src/cluster/transport.rs | 194 +++++++++++++++++++++ src/cluster/transport/loopback.rs | 274 +++++++++++++++++++++++++++++ src/cluster/transport/tcp.rs | 277 ++++++++++++++++++++++++++++++ tests/cluster_transport.rs | 272 +++++++++++++++++++++++++++++ 5 files changed, 1021 insertions(+), 4 deletions(-) create mode 100644 src/cluster/transport.rs create mode 100644 src/cluster/transport/loopback.rs create mode 100644 src/cluster/transport/tcp.rs create mode 100644 tests/cluster_transport.rs diff --git a/src/cluster.rs b/src/cluster.rs index b5953ad..950900c 100644 --- a/src/cluster.rs +++ b/src/cluster.rs @@ -1,8 +1,8 @@ //! RFC 010 — clustering (smarm⇄smarm, explicit remote boundary). //! -//! c1: feature flag + optional deps only. The owned envelope (c2), transport -//! trait (c3), and everything above them land in later chunks. This module is -//! intentionally empty so the `cluster` feature's default-build invariance is -//! reviewable in isolation. +//! c1: feature flag + optional deps. c2: the owned envelope. c3: the +//! transport trait (control connection), framed codec, and the TCP + +//! loopback impls. Everything above them lands in later chunks. pub mod envelope; +pub mod transport; diff --git a/src/cluster/transport.rs b/src/cluster/transport.rs new file mode 100644 index 0000000..f16057d --- /dev/null +++ b/src/cluster/transport.rs @@ -0,0 +1,194 @@ +//! RFC 010 c3 — transport abstraction for the **control** connection. +//! +//! Scope, per RFC 010 v2 §5 and D2: +//! +//! - A "connection" here is the *control* connection: the one carrying this +//! RFC's frame inventory ([`crate::cluster::envelope::Frame`]), whose +//! heartbeats feed failure detection. The trait deliberately says nothing +//! about how many connections a peer pair may hold — the jarred rkyv bulk +//! plane opens **additional per-peer connections** outside this trait, and +//! nothing here may foreclose that. +//! - Homogeneous smarm⇄smarm only. The BEAM membrane is *not* a transport +//! impl and the trait does not accommodate it (D2). +//! - Addresses are opaque, **pre-resolved** strings. Name resolution is a +//! single separate seam (roadmap c9); impls reject unresolved names rather +//! than resolving them. +//! +//! Blocking model: [`Conn`] calls block the caller. The TCP impl parks the +//! calling *actor* (fd readiness via the scheduler); the loopback impl blocks +//! the calling *OS thread* and is a test transport — do not drive it from a +//! scheduler thread. +//! +//! Framing is not part of the trait: [`FramedConn`] is the single shared +//! codec that turns any byte-stream [`Conn`] into a frame pipe, feeding +//! [`Frame::decode`]'s incremental contract. Impls never re-implement +//! framing, and the conformance suite exercises the same codec over every +//! impl. + +use std::io; + +use crate::cluster::envelope::{DecodeError, EncodeError, Frame}; + +pub mod loopback; +pub mod tcp; + +/// An established control connection: a bidirectional byte stream. +pub trait Conn: Send { + /// Read at least one byte, blocking the caller until data is available, + /// EOF, or error. `Ok(0)` means EOF: the peer closed and all bytes it + /// wrote before closing have been consumed. + fn read(&mut self, buf: &mut [u8]) -> io::Result; + + /// Write the whole buffer, blocking the caller as needed. + fn write_all(&mut self, buf: &[u8]) -> io::Result<()>; + + /// Close both directions. Idempotent. Bytes already written remain + /// readable at the peer, which then observes EOF; peer writes after this + /// fail. + fn close(&mut self); + + /// Diagnostic label for logs only. Mesh identity comes from the + /// handshake (`Hello`/`HelloAck`), never from the transport. + fn peer_addr(&self) -> String; +} + +/// A bound listen point producing inbound [`Conn`]s. +pub trait Listener: Send { + /// Accept the next inbound connection, blocking the caller. + fn accept(&mut self) -> io::Result>; + + /// The concrete bound address, dialable as-is (e.g. the real port when + /// bound with port 0). + fn local_addr(&self) -> String; +} + +/// A way of establishing control connections. Object-safe on purpose: the +/// connector and membership layers hold `&dyn Transport` / boxed conns +/// rather than growing a generic parameter. +pub trait Transport: Send + Sync { + /// Connect to a peer's listen address. Blocks the caller until + /// established or failed. + fn dial(&self, addr: &str) -> io::Result>; + + /// Bind a listen point. + fn listen(&self, addr: &str) -> io::Result>; +} + +impl std::fmt::Debug for dyn Conn { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "Conn({})", self.peer_addr()) + } +} + +impl std::fmt::Debug for dyn Listener { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "Listener({})", self.local_addr()) + } +} + +/// Error surface of [`FramedConn::send`]. +#[derive(Debug)] +pub enum SendError { + /// The frame could not be encoded (e.g. a field over its wire limit). + Encode(EncodeError), + /// The transport failed mid-write. + Io(io::Error), +} + +impl std::fmt::Display for SendError { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + SendError::Encode(e) => write!(f, "frame encode failed: {e:?}"), + SendError::Io(e) => write!(f, "transport write failed: {e}"), + } + } +} + +impl std::error::Error for SendError {} + +/// Error surface of [`FramedConn::recv`]. +#[derive(Debug)] +pub enum RecvError { + /// The byte stream is not a valid frame stream (bad tag, lying length, + /// oversized frame, …). The connection is unusable. + Corrupt(DecodeError), + /// The peer closed mid-frame: EOF arrived with a partial frame buffered. + /// Distinct from a clean close, which is `Ok(None)`. + TruncatedByPeer, + /// The transport failed mid-read. + Io(io::Error), +} + +impl std::fmt::Display for RecvError { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + RecvError::Corrupt(e) => write!(f, "frame stream corrupt: {e:?}"), + RecvError::TruncatedByPeer => write!(f, "peer closed mid-frame"), + RecvError::Io(e) => write!(f, "transport read failed: {e}"), + } + } +} + +impl std::error::Error for RecvError {} + +/// How many bytes each blocking read asks the transport for. +const READ_CHUNK: usize = 8 * 1024; + +/// The shared framed codec: one of these per control connection, owning the +/// [`Conn`] and the reassembly buffer. Frames may arrive split or coalesced +/// arbitrarily; [`recv`](FramedConn::recv) reassembles either way. +pub struct FramedConn { + conn: Box, + rbuf: Vec, +} + +impl FramedConn { + pub fn new(conn: Box) -> Self { + FramedConn { + conn, + rbuf: Vec::new(), + } + } + + /// Encode and write one frame. + pub fn send(&mut self, frame: &Frame) -> Result<(), SendError> { + let mut out = Vec::new(); + frame.encode(&mut out).map_err(SendError::Encode)?; + self.conn.write_all(&out).map_err(SendError::Io) + } + + /// Receive the next frame. `Ok(None)` is a clean close: EOF at a frame + /// boundary. EOF mid-frame is [`RecvError::TruncatedByPeer`]. + pub fn recv(&mut self) -> Result, RecvError> { + loop { + match Frame::decode(&self.rbuf) { + Ok(Some((frame, consumed))) => { + self.rbuf.drain(..consumed); + return Ok(Some(frame)); + } + Ok(None) => {} + Err(e) => return Err(RecvError::Corrupt(e)), + } + let mut chunk = [0u8; READ_CHUNK]; + let n = self.conn.read(&mut chunk).map_err(RecvError::Io)?; + if n == 0 { + return if self.rbuf.is_empty() { + Ok(None) + } else { + Err(RecvError::TruncatedByPeer) + }; + } + self.rbuf.extend_from_slice(&chunk[..n]); + } + } + + /// Close the underlying connection (idempotent, see [`Conn::close`]). + pub fn close(&mut self) { + self.conn.close(); + } + + /// Diagnostic label of the underlying connection. + pub fn peer_addr(&self) -> String { + self.conn.peer_addr() + } +} diff --git a/src/cluster/transport/loopback.rs b/src/cluster/transport/loopback.rs new file mode 100644 index 0000000..61705f9 --- /dev/null +++ b/src/cluster/transport/loopback.rs @@ -0,0 +1,274 @@ +//! In-memory loopback transport — a shipped **test** transport. +//! +//! Lets Phases 2–4 exercise protocol logic (connector, membership, +//! monitors) through the real transport trait and the real framed codec +//! without sockets or timing flake. +//! +//! Blocking model: calls block the **OS thread** on a condvar. That is the +//! right shape for plain `#[test]`s driving protocol state machines; it is +//! the wrong shape for scheduler threads. Do not drive a loopback conn from +//! inside an actor — use the TCP impl there. +//! +//! Semantics mirror TCP shutdown where it matters for the codec: bytes +//! written before `close` remain readable at the peer, which then sees EOF; +//! writes toward a closed peer fail with `BrokenPipe`. Write buffers are +//! unbounded, so writes never block — backpressure is not simulated. + +use std::collections::{HashMap, VecDeque}; +use std::io; +use std::sync::{Arc, Condvar, Mutex, MutexGuard}; + +use super::{Conn, Listener, Transport}; + +/// Poison-tolerant lock: a panicked holder in a *test* transport must not +/// cascade; the byte-queue state stays consistent under every early return. +fn lock(m: &Mutex) -> MutexGuard<'_, T> { + match m.lock() { + Ok(g) => g, + Err(poisoned) => poisoned.into_inner(), + } +} + +// --------------------------------------------------------------------------- +// One direction of a duplex: a byte queue with close flags for both ends +// --------------------------------------------------------------------------- + +#[derive(Default)] +struct PipeState { + bytes: VecDeque, + /// The writing end closed: readers drain remaining bytes, then EOF. + write_closed: bool, + /// The reading end closed: writers fail with `BrokenPipe`. + read_closed: bool, +} + +#[derive(Default)] +struct Pipe { + state: Mutex, + cv: Condvar, +} + +impl Pipe { + fn write_all(&self, buf: &[u8]) -> io::Result<()> { + let mut st = lock(&self.state); + if st.write_closed { + return Err(io::Error::new( + io::ErrorKind::NotConnected, + "loopback conn closed locally", + )); + } + if st.read_closed { + return Err(io::Error::new( + io::ErrorKind::BrokenPipe, + "loopback peer closed", + )); + } + st.bytes.extend(buf); + self.cv.notify_all(); + Ok(()) + } + + fn read(&self, buf: &mut [u8]) -> io::Result { + if buf.is_empty() { + return Ok(0); + } + let mut st = lock(&self.state); + loop { + if !st.bytes.is_empty() { + let n = st.bytes.len().min(buf.len()); + for (slot, byte) in buf.iter_mut().zip(st.bytes.drain(..n)) { + *slot = byte; + } + return Ok(n); + } + if st.write_closed || st.read_closed { + return Ok(0); // EOF: peer closed, or our own end closed. + } + st = match self.cv.wait(st) { + Ok(g) => g, + Err(poisoned) => poisoned.into_inner(), + }; + } + } + + /// Close from the writer side: remaining bytes stay readable, then EOF. + fn close_write(&self) { + lock(&self.state).write_closed = true; + self.cv.notify_all(); + } + + /// Close from the reader side: peer writes fail from now on. + fn close_read(&self) { + lock(&self.state).read_closed = true; + self.cv.notify_all(); + } +} + +// --------------------------------------------------------------------------- +// Conn: two pipes, one per direction +// --------------------------------------------------------------------------- + +/// One end of an established loopback connection. +pub struct LoopbackConn { + tx: Arc, + rx: Arc, + peer: String, +} + +impl Conn for LoopbackConn { + fn read(&mut self, buf: &mut [u8]) -> io::Result { + self.rx.read(buf) + } + + fn write_all(&mut self, buf: &[u8]) -> io::Result<()> { + self.tx.write_all(buf) + } + + fn close(&mut self) { + self.tx.close_write(); + self.rx.close_read(); + } + + fn peer_addr(&self) -> String { + self.peer.clone() + } +} + +impl Drop for LoopbackConn { + fn drop(&mut self) { + self.close(); + } +} + +fn conn_pair(listen_addr: &str, conn_no: u64) -> (LoopbackConn, LoopbackConn) { + let a_to_b = Arc::new(Pipe::default()); + let b_to_a = Arc::new(Pipe::default()); + let dialer = LoopbackConn { + tx: a_to_b.clone(), + rx: b_to_a.clone(), + peer: listen_addr.to_string(), + }; + let accepted = LoopbackConn { + tx: b_to_a, + rx: a_to_b, + peer: format!("{listen_addr}#dialer-{conn_no}"), + }; + (dialer, accepted) +} + +// --------------------------------------------------------------------------- +// Listener + registry +// --------------------------------------------------------------------------- + +#[derive(Default)] +struct AcceptState { + pending: VecDeque, + closed: bool, +} + +#[derive(Default)] +struct AcceptQueue { + state: Mutex, + cv: Condvar, +} + +/// A bound loopback listen point. +pub struct LoopbackListener { + addr: String, + queue: Arc, + registry: Arc>, +} + +impl Listener for LoopbackListener { + fn accept(&mut self) -> io::Result> { + let mut st = lock(&self.queue.state); + loop { + if let Some(conn) = st.pending.pop_front() { + return Ok(Box::new(conn)); + } + if st.closed { + return Err(io::Error::new( + io::ErrorKind::NotConnected, + "loopback listener closed", + )); + } + st = match self.queue.cv.wait(st) { + Ok(g) => g, + Err(poisoned) => poisoned.into_inner(), + }; + } + } + + fn local_addr(&self) -> String { + self.addr.clone() + } +} + +impl Drop for LoopbackListener { + fn drop(&mut self) { + lock(&self.registry).listeners.remove(&self.addr); + let mut st = lock(&self.queue.state); + st.closed = true; + self.queue.cv.notify_all(); + } +} + +#[derive(Default)] +struct Registry { + listeners: HashMap>, + dial_count: u64, +} + +/// The loopback transport. Addresses are arbitrary strings scoped to one +/// transport instance; distinct instances never see each other's listeners. +#[derive(Default)] +pub struct LoopbackTransport { + registry: Arc>, +} + +impl Transport for LoopbackTransport { + fn dial(&self, addr: &str) -> io::Result> { + let (queue, conn_no) = { + let mut reg = lock(&self.registry); + reg.dial_count += 1; + let no = reg.dial_count; + match reg.listeners.get(addr) { + Some(q) => (q.clone(), no), + None => { + return Err(io::Error::new( + io::ErrorKind::ConnectionRefused, + format!("no loopback listener at {addr:?}"), + )); + } + } + }; + let (dialer, accepted) = conn_pair(addr, conn_no); + let mut st = lock(&queue.state); + if st.closed { + return Err(io::Error::new( + io::ErrorKind::ConnectionRefused, + format!("loopback listener at {addr:?} closed"), + )); + } + st.pending.push_back(accepted); + queue.cv.notify_all(); + Ok(Box::new(dialer)) + } + + fn listen(&self, addr: &str) -> io::Result> { + let queue = Arc::new(AcceptQueue::default()); + let mut reg = lock(&self.registry); + if reg.listeners.contains_key(addr) { + return Err(io::Error::new( + io::ErrorKind::AddrInUse, + format!("loopback listener already bound at {addr:?}"), + )); + } + reg.listeners.insert(addr.to_string(), queue.clone()); + Ok(Box::new(LoopbackListener { + addr: addr.to_string(), + queue, + registry: self.registry.clone(), + })) + } +} diff --git a/src/cluster/transport/tcp.rs b/src/cluster/transport/tcp.rs new file mode 100644 index 0000000..c725088 --- /dev/null +++ b/src/cluster/transport/tcp.rs @@ -0,0 +1,277 @@ +//! TCP transport — the production control-plane transport. +//! +//! Blocking model: every blocking point parks the **calling actor** on fd +//! readiness ([`crate::scheduler::wait_readable`] / `wait_writable`); the +//! scheduler thread is never blocked. All conn/listener methods must +//! therefore run inside an actor. `listen` itself only binds (no waiting) +//! and is callable anywhere. +//! +//! Addresses are pre-resolved `ip:port` strings (`SocketAddr` syntax, IPv4 +//! or IPv6). Hostnames are rejected with `InvalidInput`: name resolution is +//! the single c9 seam, not something each transport does on the side. +//! +//! Writes use `send(2)` with `MSG_NOSIGNAL` — a peer reset must surface as +//! `BrokenPipe`/`ConnectionReset`, not `SIGPIPE`. + +use std::io; +use std::net::{SocketAddr, TcpListener as StdListener, TcpStream}; +use std::os::fd::{AsRawFd, RawFd}; + +use crate::scheduler::{wait_readable, wait_writable}; + +use super::{Conn, Listener, Transport}; + +// --------------------------------------------------------------------------- +// sockaddr plumbing +// --------------------------------------------------------------------------- + +/// A `sockaddr_in`/`sockaddr_in6` built from a parsed `SocketAddr`, plus its +/// length, ready for `connect(2)`. +union SockAddrUnion { + v4: libc::sockaddr_in, + v6: libc::sockaddr_in6, +} + +fn to_sockaddr(sa: &SocketAddr) -> (SockAddrUnion, libc::socklen_t) { + match sa { + SocketAddr::V4(v4) => { + let raw = libc::sockaddr_in { + sin_family: libc::AF_INET as libc::sa_family_t, + sin_port: v4.port().to_be(), + sin_addr: libc::in_addr { + s_addr: u32::from_be_bytes(v4.ip().octets()).to_be(), + }, + sin_zero: [0; 8], + }; + ( + SockAddrUnion { v4: raw }, + std::mem::size_of::() as libc::socklen_t, + ) + } + SocketAddr::V6(v6) => { + let raw = libc::sockaddr_in6 { + sin6_family: libc::AF_INET6 as libc::sa_family_t, + sin6_port: v6.port().to_be(), + sin6_flowinfo: v6.flowinfo(), + sin6_addr: libc::in6_addr { + s6_addr: v6.ip().octets(), + }, + sin6_scope_id: v6.scope_id(), + }; + ( + SockAddrUnion { v6: raw }, + std::mem::size_of::() as libc::socklen_t, + ) + } + } +} + +fn parse_addr(addr: &str) -> io::Result { + addr.parse().map_err(|_| { + io::Error::new( + io::ErrorKind::InvalidInput, + format!("{addr:?} is not a resolved ip:port — resolution is the c9 seam"), + ) + }) +} + +fn so_error(fd: RawFd) -> io::Result<()> { + let mut err: libc::c_int = 0; + let mut len = std::mem::size_of::() as libc::socklen_t; + let rc = unsafe { + libc::getsockopt( + fd, + libc::SOL_SOCKET, + libc::SO_ERROR, + (&mut err) as *mut _ as *mut libc::c_void, + &mut len, + ) + }; + if rc != 0 { + return Err(io::Error::last_os_error()); + } + if err != 0 { + return Err(io::Error::from_raw_os_error(err)); + } + Ok(()) +} + +// --------------------------------------------------------------------------- +// Conn +// --------------------------------------------------------------------------- + +/// One established TCP control connection. Owns the socket; drop closes it. +pub struct TcpConn { + stream: TcpStream, + closed: bool, +} + +impl TcpConn { + fn fd(&self) -> RawFd { + self.stream.as_raw_fd() + } +} + +impl Conn for TcpConn { + fn read(&mut self, buf: &mut [u8]) -> io::Result { + if self.closed { + return Ok(0); + } + if buf.is_empty() { + return Ok(0); + } + loop { + wait_readable(self.fd())?; + let n = unsafe { libc::read(self.fd(), buf.as_mut_ptr() as *mut _, buf.len()) }; + if n >= 0 { + return Ok(n as usize); + } + let e = io::Error::last_os_error(); + match e.kind() { + // Spurious readiness or signal: park again. + io::ErrorKind::WouldBlock | io::ErrorKind::Interrupted => continue, + _ => return Err(e), + } + } + } + + fn write_all(&mut self, mut buf: &[u8]) -> io::Result<()> { + if self.closed { + return Err(io::Error::new( + io::ErrorKind::NotConnected, + "tcp conn closed locally", + )); + } + while !buf.is_empty() { + wait_writable(self.fd())?; + let n = unsafe { + libc::send( + self.fd(), + buf.as_ptr() as *const _, + buf.len(), + libc::MSG_NOSIGNAL, + ) + }; + if n >= 0 { + buf = &buf[n as usize..]; + continue; + } + let e = io::Error::last_os_error(); + match e.kind() { + io::ErrorKind::WouldBlock | io::ErrorKind::Interrupted => continue, + _ => return Err(e), + } + } + Ok(()) + } + + fn close(&mut self) { + if !self.closed { + self.closed = true; + // Best-effort: the peer sees EOF after draining. The fd itself + // is released when the owning stream drops. + let _ = self.stream.shutdown(std::net::Shutdown::Both); + } + } + + fn peer_addr(&self) -> String { + match self.stream.peer_addr() { + Ok(sa) => sa.to_string(), + Err(_) => "".to_string(), + } + } +} + +// --------------------------------------------------------------------------- +// Listener +// --------------------------------------------------------------------------- + +/// A bound TCP listen point (non-blocking socket; accept parks the actor). +pub struct TcpListener { + inner: StdListener, + local: SocketAddr, +} + +impl Listener for TcpListener { + fn accept(&mut self) -> io::Result> { + loop { + wait_readable(self.inner.as_raw_fd())?; + match self.inner.accept() { + Ok((stream, _peer)) => { + stream.set_nonblocking(true)?; + return Ok(Box::new(TcpConn { + stream, + closed: false, + })); + } + Err(e) + if e.kind() == io::ErrorKind::WouldBlock + || e.kind() == io::ErrorKind::Interrupted => + { + continue; + } + Err(e) => return Err(e), + } + } + } + + fn local_addr(&self) -> String { + self.local.to_string() + } +} + +// --------------------------------------------------------------------------- +// Transport +// --------------------------------------------------------------------------- + +/// The TCP transport. Stateless; every call stands alone. +pub struct TcpTransport; + +impl Transport for TcpTransport { + fn dial(&self, addr: &str) -> io::Result> { + let sa = parse_addr(addr)?; + let family = match sa { + SocketAddr::V4(_) => libc::AF_INET, + SocketAddr::V6(_) => libc::AF_INET6, + }; + let fd = unsafe { + libc::socket( + family, + libc::SOCK_STREAM | libc::SOCK_NONBLOCK | libc::SOCK_CLOEXEC, + 0, + ) + }; + if fd < 0 { + return Err(io::Error::last_os_error()); + } + // From here the fd is owned by `stream`; any early return drops it. + let stream = unsafe { + use std::os::fd::FromRawFd; + TcpStream::from_raw_fd(fd) + }; + let (raw, len) = to_sockaddr(&sa); + let rc = unsafe { libc::connect(fd, (&raw) as *const _ as *const libc::sockaddr, len) }; + if rc != 0 { + let e = io::Error::last_os_error(); + if e.raw_os_error() != Some(libc::EINPROGRESS) { + return Err(e); + } + // Connect in flight: park until the socket is writable, then the + // verdict is in SO_ERROR. + wait_writable(fd)?; + so_error(fd)?; + } + Ok(Box::new(TcpConn { + stream, + closed: false, + })) + } + + fn listen(&self, addr: &str) -> io::Result> { + let sa = parse_addr(addr)?; + let inner = StdListener::bind(sa)?; + inner.set_nonblocking(true)?; + let local = inner.local_addr()?; + Ok(Box::new(TcpListener { inner, local })) + } +} diff --git a/tests/cluster_transport.rs b/tests/cluster_transport.rs new file mode 100644 index 0000000..f9df0ae --- /dev/null +++ b/tests/cluster_transport.rs @@ -0,0 +1,272 @@ +//! RFC 010 c3 — transport conformance suite, run against both shipped impls +//! (TCP and in-memory loopback), plus impl-specific cases. +//! +//! Shared suite (roadmap): frame roundtrips through the framed codec, framing +//! across a split write, coalesced frames in one write, peer-close mid-frame +//! (must error, not EOF), clean close at a frame boundary (EOF as `Ok(None)`). +//! +//! The TCP impl parks the calling actor, so its runs live inside `smarm::run`; +//! loopback blocks the OS thread and runs as plain tests. +#![cfg(feature = "cluster")] + +use smarm::cluster::envelope::Frame; +use smarm::cluster::transport::loopback::LoopbackTransport; +use smarm::cluster::transport::tcp::TcpTransport; +use smarm::cluster::transport::{Conn, FramedConn, RecvError, Transport}; + +// --------------------------------------------------------------------------- +// Helpers +// --------------------------------------------------------------------------- + +/// Listener + dial + accept against one transport, both conns returned. +/// Relies on dial not requiring a concurrent accept (TCP backlog / loopback +/// queue), so a single thread or actor can hold both ends. +fn pair(t: &dyn Transport, addr: &str) -> (Box, Box) { + let mut l = t.listen(addr).unwrap(); + let a = t.dial(&l.local_addr()).unwrap(); + let b = l.accept().unwrap(); + (a, b) +} + +fn frames() -> Vec { + vec![ + Frame::Heartbeat, + Frame::Send { + index: 42, + generation: 3, + type_hash: 0x1234_5678_9ABC_DEF0, + payload: vec![1, 2, 3, 4, 5], + }, + Frame::SendNamed { + name: "the_counter".into(), + type_hash: 0xFFFF_0000_FFFF_0000, + payload: vec![], + }, + Frame::Demonitor { monitor_id: 77 }, + ] +} + +fn encode(f: &Frame) -> Vec { + let mut out = Vec::new(); + f.encode(&mut out).unwrap(); + out +} + +// --------------------------------------------------------------------------- +// Shared conformance suite — generic over an established pair +// --------------------------------------------------------------------------- + +fn suite_roundtrip(a: Box, b: Box) { + let mut fa = FramedConn::new(a); + let mut fb = FramedConn::new(b); + // a -> b, then b -> a: both directions carry every frame shape. + for f in frames() { + fa.send(&f).unwrap(); + assert_eq!(fb.recv().unwrap().unwrap(), f); + } + for f in frames() { + fb.send(&f).unwrap(); + assert_eq!(fa.recv().unwrap().unwrap(), f); + } +} + +fn suite_split_write(mut a: Box, b: Box) { + let f = Frame::Send { + index: 7, + generation: 1, + type_hash: 0xAB, + payload: vec![9; 64], + }; + let bytes = encode(&f); + // Split inside the length prefix, then inside the body: the reader must + // reassemble regardless of where the boundary falls. + a.write_all(&bytes[..2]).unwrap(); + a.write_all(&bytes[2..10]).unwrap(); + a.write_all(&bytes[10..]).unwrap(); + let mut fb = FramedConn::new(b); + assert_eq!(fb.recv().unwrap().unwrap(), f); +} + +fn suite_coalesced(mut a: Box, b: Box) { + let f1 = Frame::Heartbeat; + let f2 = Frame::Demonitor { monitor_id: 5 }; + let mut bytes = encode(&f1); + bytes.extend_from_slice(&encode(&f2)); + a.write_all(&bytes).unwrap(); + let mut fb = FramedConn::new(b); + assert_eq!(fb.recv().unwrap().unwrap(), f1); + assert_eq!(fb.recv().unwrap().unwrap(), f2); +} + +fn suite_close_mid_frame(mut a: Box, b: Box) { + let bytes = encode(&Frame::Send { + index: 1, + generation: 1, + type_hash: 1, + payload: vec![0; 128], + }); + a.write_all(&bytes[..bytes.len() / 2]).unwrap(); + a.close(); + let mut fb = FramedConn::new(b); + match fb.recv() { + Err(RecvError::TruncatedByPeer) => {} + other => panic!("expected TruncatedByPeer, got {other:?}"), + } +} + +fn suite_clean_close(mut a: Box, b: Box) { + let f = Frame::Heartbeat; + a.write_all(&encode(&f)).unwrap(); + a.close(); + let mut fb = FramedConn::new(b); + // The buffered frame is still delivered, then EOF at the boundary. + assert_eq!(fb.recv().unwrap().unwrap(), f); + assert!(fb.recv().unwrap().is_none()); +} + +fn run_suite(t: &dyn Transport, addr: &str) { + let (a, b) = pair(t, addr); + suite_roundtrip(a, b); + let (a, b) = pair(t, addr); + suite_split_write(a, b); + let (a, b) = pair(t, addr); + suite_coalesced(a, b); + let (a, b) = pair(t, addr); + suite_close_mid_frame(a, b); + let (a, b) = pair(t, addr); + suite_clean_close(a, b); +} + +// --------------------------------------------------------------------------- +// Loopback — plain tests, no runtime +// --------------------------------------------------------------------------- + +#[test] +fn loopback_conformance() { + // Fresh transport per pair() call is fine, but one instance must also + // support sequential re-listen on distinct addresses. + let t = LoopbackTransport::default(); + run_suite(&t, "alpha"); +} + +#[test] +fn loopback_dial_unknown_addr_refused() { + let t = LoopbackTransport::default(); + let err = t.dial("nobody-home").unwrap_err(); + assert_eq!(err.kind(), std::io::ErrorKind::ConnectionRefused); +} + +#[test] +fn loopback_addr_in_use() { + let t = LoopbackTransport::default(); + let _l = t.listen("alpha").unwrap(); + let err = t.listen("alpha").unwrap_err(); + assert_eq!(err.kind(), std::io::ErrorKind::AddrInUse); +} + +#[test] +fn loopback_listener_drop_frees_addr_and_refuses_dial() { + let t = LoopbackTransport::default(); + let l = t.listen("alpha").unwrap(); + drop(l); + let err = t.dial("alpha").unwrap_err(); + assert_eq!(err.kind(), std::io::ErrorKind::ConnectionRefused); + // Address is reusable after the listener is gone. + let _l2 = t.listen("alpha").unwrap(); +} + +#[test] +fn loopback_write_after_peer_close_broken_pipe() { + let t = LoopbackTransport::default(); + let (mut a, mut b) = pair(&t, "alpha"); + b.close(); + let err = a.write_all(&[1, 2, 3]).unwrap_err(); + assert_eq!(err.kind(), std::io::ErrorKind::BrokenPipe); +} + +#[test] +fn loopback_cross_thread_blocking_read() { + // Reader blocks on an empty pipe until the writer thread delivers. + let t = LoopbackTransport::default(); + let (a, b) = pair(&t, "alpha"); + let mut fb = FramedConn::new(b); + let writer = std::thread::spawn(move || { + let mut a = a; + std::thread::sleep(std::time::Duration::from_millis(30)); + a.write_all(&encode(&Frame::Heartbeat)).unwrap(); + }); + assert_eq!(fb.recv().unwrap().unwrap(), Frame::Heartbeat); + writer.join().unwrap(); +} + +// --------------------------------------------------------------------------- +// TCP — inside the runtime (read/write park the calling actor) +// --------------------------------------------------------------------------- + +#[test] +fn tcp_conformance() { + smarm::run(|| { + run_suite(&TcpTransport, "127.0.0.1:0"); + }); +} + +#[test] +fn tcp_dial_refused() { + smarm::run(|| { + // Bind to an OS-assigned port, learn it, close the listener, dial it. + let addr = { + let l = TcpTransport.listen("127.0.0.1:0").unwrap(); + l.local_addr() + }; + let err = TcpTransport.dial(&addr).unwrap_err(); + assert_eq!(err.kind(), std::io::ErrorKind::ConnectionRefused); + }); +} + +#[test] +fn tcp_bad_addr_rejected_without_resolution() { + // Addresses are opaque pre-resolved strings; the c9 seam resolves names. + // A hostname is therefore invalid input here, not something to resolve. + let err = TcpTransport.dial("localhost:1234").unwrap_err(); + assert_eq!(err.kind(), std::io::ErrorKind::InvalidInput); +} + +#[test] +fn tcp_local_addr_reports_real_port() { + let l = TcpTransport.listen("127.0.0.1:0").unwrap(); + let addr = l.local_addr(); + let port: u16 = addr.rsplit(':').next().unwrap().parse().unwrap(); + assert_ne!(port, 0); +} + +#[test] +fn tcp_big_frame_across_socket_buffers() { + // A payload far beyond socket buffer sizes forces genuine fragmentation + // and write backpressure: writer and reader must run concurrently. + smarm::run(|| { + let (tx, rx) = smarm::channel::(); + let payload = vec![0xA5u8; 4 * 1024 * 1024]; + let f = Frame::Send { + index: 9, + generation: 2, + type_hash: 0xC0FFEE, + payload, + }; + let mut l = TcpTransport.listen("127.0.0.1:0").unwrap(); + let addr = l.local_addr(); + let fw = f.clone(); + let writer = smarm::spawn(move || { + let mut fa = FramedConn::new(TcpTransport.dial(&addr).unwrap()); + fa.send(&fw).unwrap(); + }); + let reader = smarm::spawn(move || { + let mut fb = FramedConn::new(l.accept().unwrap()); + let got = fb.recv().unwrap().unwrap(); + tx.send(got).unwrap(); + }); + let got = rx.recv().unwrap(); + assert_eq!(got, f); + writer.join().unwrap(); + reader.join().unwrap(); + }); +} From 8a9e2b81b1939585e9a522ad45d88b7a652e6d7f Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 14 Aug 2026 14:46:28 +0000 Subject: [PATCH 04/14] =?UTF-8?q?test(cluster):=20RFC=20010=20c4=20?= =?UTF-8?q?=E2=80=94=20subprocess=20two-node=20harness?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The runtime is a process singleton, so multi-node tests mean multiple processes. tests/common/mod.rs is the reusable harness (precedent: RFC 019 c6 / tests/stack_diag.rs self-re-exec, extended to live tailing): re-execs the current test binary as named roles, tails stdout/stderr on reader threads, waits on protocol-visible lines with bounded timeouts (panic dumps carry the full transcript), and Drop SIGKILLs+reaps so a panicking test leaves no orphan or zombie. Children run with --test-threads=1 --quiet --nocapture; the last flag is load-bearing — libtest's capture would otherwise swallow role output. Port assignment is race-free by construction: children bind port 0 and announce the concrete address (LISTENING ); the parent never pre-picks. Smoke suite per roadmap: two real nodes, handshake-less TCP connect through the real framed codec (one Heartbeat across, clean close seen on both sides, both exit 0), plus reap-on-drop proven via ESRCH and nonzero-exit surfacing. Flake budget stated in the module doc: 10 s bound per wait, 10/10 clean at authoring, >1/100 failures = regression. --- tests/cluster_two_node.rs | 108 +++++++++++++++++ tests/common/mod.rs | 239 ++++++++++++++++++++++++++++++++++++++ 2 files changed, 347 insertions(+) create mode 100644 tests/cluster_two_node.rs create mode 100644 tests/common/mod.rs diff --git a/tests/cluster_two_node.rs b/tests/cluster_two_node.rs new file mode 100644 index 0000000..712b370 --- /dev/null +++ b/tests/cluster_two_node.rs @@ -0,0 +1,108 @@ +//! RFC 010 c4 — two-node harness smoke tests. +//! +//! Roadmap: "spawn two, handshake-less connect, both exit clean." The +//! listener node binds port 0 and announces its concrete address; the +//! dialer connects raw (no Hello — c5 doesn't exist yet), pushes one +//! Heartbeat through the real framed codec, and closes. Assertions are on +//! protocol-visible lines only. Flake budget: see tests/common/mod.rs. +#![cfg(feature = "cluster")] + +mod common; + +use common::{maybe_child, spawn_node}; +use smarm::cluster::envelope::Frame; +use smarm::cluster::transport::tcp::TcpTransport; +use smarm::cluster::transport::{FramedConn, Transport}; + +const ROLES: &[(&str, fn())] = &[ + ("listener", role_listener), + ("dialer", role_dialer), + ("hang", role_hang), + ("fail", role_fail), +]; + +fn role_listener() { + smarm::run(|| { + let mut l = TcpTransport.listen("127.0.0.1:0").unwrap(); + println!("LISTENING {}", l.local_addr()); + let mut fc = FramedConn::new(l.accept().unwrap()); + match fc.recv() { + Ok(Some(Frame::Heartbeat)) => println!("RECV heartbeat"), + other => { + println!("RECV unexpected: {other:?}"); + std::process::exit(3); + } + } + match fc.recv() { + Ok(None) => println!("PEER-CLOSED clean"), + other => { + println!("PEER-CLOSED unexpected: {other:?}"); + std::process::exit(3); + } + } + }); + println!("EXIT ok"); +} + +fn role_dialer() { + let addr = std::env::var("SMARM_PEER_ADDR").expect("SMARM_PEER_ADDR not set"); + smarm::run(move || { + let mut fc = FramedConn::new(TcpTransport.dial(&addr).unwrap()); + fc.send(&Frame::Heartbeat).unwrap(); + fc.close(); + println!("SENT heartbeat"); + }); + println!("EXIT ok"); +} + +fn role_hang() { + println!("HANGING"); + loop { + std::thread::sleep(std::time::Duration::from_secs(3600)); + } +} + +fn role_fail() { + std::process::exit(7); +} + +/// The roadmap smoke test: two real processes, raw transport connect, one +/// frame across, clean close observed on both sides, both exit 0. +#[test] +fn two_nodes_connect_and_exit_clean() { + maybe_child(ROLES); + let mut listener = spawn_node("listener", &[]); + let addr = listener.wait_listening(); + let mut dialer = spawn_node("dialer", &[("SMARM_PEER_ADDR", &addr)]); + dialer.wait_line("SENT heartbeat", |l| l == "SENT heartbeat"); + listener.wait_line("RECV heartbeat", |l| l == "RECV heartbeat"); + listener.wait_line("clean peer close", |l| l == "PEER-CLOSED clean"); + dialer.wait_exit_ok(); + listener.wait_exit_ok(); +} + +/// Reap guarantee: dropping a Node kills a hung child — no orphan survives +/// a panicking test. +#[test] +fn drop_reaps_hung_node() { + maybe_child(ROLES); + let mut node = spawn_node("hang", &[]); + node.wait_line("HANGING", |l| l == "HANGING"); + let pid = node.pid().expect("live child has a pid") as libc::pid_t; + drop(node); + // After Drop's kill+wait the pid is fully reaped: signalling it fails + // with ESRCH (pid-reuse in this instant is not a realistic race). + let rc = unsafe { libc::kill(pid, 0) }; + assert_eq!(rc, -1, "process still signallable after Drop"); + let errno = std::io::Error::last_os_error().raw_os_error(); + assert_eq!(errno, Some(libc::ESRCH), "expected ESRCH, got {errno:?}"); +} + +/// Nonzero child exits surface as statuses, not hangs or panics. +#[test] +fn nonzero_exit_is_reported() { + maybe_child(ROLES); + let mut node = spawn_node("fail", &[]); + let status = node.wait_exit(); + assert_eq!(status.code(), Some(7)); +} diff --git a/tests/common/mod.rs b/tests/common/mod.rs new file mode 100644 index 0000000..ba522e7 --- /dev/null +++ b/tests/common/mod.rs @@ -0,0 +1,239 @@ +//! RFC 010 c4 — subprocess multi-node test harness. +//! +//! The runtime is a process singleton, so two real nodes means two +//! processes. This harness re-execs the *current test binary* as node +//! processes (precedent: tests/stack_diag.rs), tails their output live, +//! waits on protocol-visible lines, and reaps reliably no matter how the +//! test dies. +//! +//! Usage, per test file: +//! +//! - Declare roles as plain `fn()`s. A role prints protocol-visible facts +//! as single lines (Rust's piped stdout is line-buffered, so `println!` +//! is enough) and exits. +//! - **Every** `#[test]` in the file starts with +//! [`maybe_child`]`(ROLES)` — in the child re-exec, whichever test +//! libtest runs first performs the role and exits before the rest of the +//! suite runs (children are spawned with `--test-threads=1 --quiet`). +//! - The parent side spawns nodes with [`spawn_node`], waits on lines with +//! [`Node::wait_line`], and on exits with [`Node::wait_exit`]. +//! +//! Port assignment: children bind port 0 and *report* the concrete address +//! (e.g. `LISTENING 127.0.0.1:41733`) rather than the parent pre-picking a +//! port — no bind/steal race by construction. +//! +//! Reaping: [`Node`]'s `Drop` SIGKILLs and `wait(2)`s the child, so a +//! panicking test (including a `wait_line` timeout) leaves no orphan and +//! no zombie. Tail threads exit on pipe EOF. +//! +//! Flake budget (explicit, per roadmap): every wait is bounded by +//! [`WAIT`] (10 s) against a typical cost of well under 1 s; the smoke +//! suite ran 10/10 clean at authoring time. Treat >1 failure in 100 runs +//! as a harness or runtime regression, not weather. On timeout the panic +//! message carries the node's full transcript so far. + +#![allow(dead_code)] // Reusable surface: later phases use more of it than any one file. + +use std::env; +use std::io::{BufRead, BufReader}; +use std::process::{Child, Command, ExitStatus, Stdio}; +use std::sync::mpsc::{Receiver, RecvTimeoutError}; +use std::time::{Duration, Instant}; + +/// Env var selecting the child role in a re-exec. +const ROLE_ENV: &str = "SMARM_TWO_NODE_ROLE"; + +/// Upper bound for every wait in the harness. See the flake budget above. +pub const WAIT: Duration = Duration::from_secs(10); + +/// In the child re-exec: run the matching role and exit. In the parent (no +/// role env set): return immediately. Call this first in every `#[test]` of +/// any file using the harness, passing the file's full role table. +pub fn maybe_child(roles: &[(&str, fn())]) { + let role = match env::var(ROLE_ENV) { + Ok(r) => r, + Err(_) => return, + }; + for (name, f) in roles { + if *name == role { + f(); + std::process::exit(0); + } + } + eprintln!("two_node harness: unknown role {role:?}"); + std::process::exit(2); +} + +/// One spawned node process with live-tailed output. +pub struct Node { + /// Role name, for panic messages. + pub role: String, + child: Option, + stdout_rx: Receiver, + stderr_rx: Receiver, + /// Every line consumed from stdout/stderr so far, for failure dumps. + transcript: Vec, +} + +fn tail(stream: impl std::io::Read + Send + 'static, prefix: &'static str) -> Receiver { + let (tx, rx) = std::sync::mpsc::channel(); + std::thread::spawn(move || { + for line in BufReader::new(stream).lines() { + let line = match line { + Ok(l) => l, + Err(_) => break, + }; + // Receiver gone (Node dropped): stop tailing. + if tx.send(format!("{prefix}{line}")).is_err() { + break; + } + } + }); + rx +} + +/// Re-exec the current test binary as `role`, with any extra env vars. +pub fn spawn_node(role: &str, extra_env: &[(&str, &str)]) -> Node { + let exe = env::current_exe().expect("current_exe"); + let mut cmd = Command::new(exe); + cmd.env(ROLE_ENV, role) + // --test-threads=1: exactly one test fn starts, hits maybe_child, + // and becomes the role. --nocapture: libtest must not swallow the + // role's println! lines — the parent tails them live. + .args(["--test-threads=1", "--quiet", "--nocapture"]) + .stdout(Stdio::piped()) + .stderr(Stdio::piped()); + for (k, v) in extra_env { + cmd.env(k, v); + } + let mut child = cmd.spawn().expect("failed to spawn node process"); + let stdout_rx = tail(child.stdout.take().expect("piped stdout"), ""); + let stderr_rx = tail(child.stderr.take().expect("piped stderr"), "[stderr] "); + Node { + role: role.to_string(), + child: Some(child), + stdout_rx, + stderr_rx, + transcript: Vec::new(), + } +} + +impl Node { + fn drain_stderr(&mut self) { + while let Ok(l) = self.stderr_rx.try_recv() { + self.transcript.push(l); + } + } + + fn dump(&self) -> String { + if self.transcript.is_empty() { + "".to_string() + } else { + self.transcript.join("\n") + } + } + + /// Wait until a stdout line satisfies `pred`; return it. Panics with the + /// full transcript after [`WAIT`]. `what` names the expectation in the + /// panic message. + pub fn wait_line(&mut self, what: &str, pred: impl Fn(&str) -> bool) -> String { + let deadline = Instant::now() + WAIT; + loop { + self.drain_stderr(); + let left = deadline.saturating_duration_since(Instant::now()); + match self.stdout_rx.recv_timeout(left) { + Ok(line) => { + self.transcript.push(line.clone()); + if pred(&line) { + return line; + } + } + Err(RecvTimeoutError::Timeout) => { + panic!( + "node {:?}: timed out waiting for {what} after {WAIT:?}; transcript:\n{}", + self.role, + self.dump() + ); + } + Err(RecvTimeoutError::Disconnected) => { + panic!( + "node {:?}: output closed while waiting for {what}; transcript:\n{}", + self.role, + self.dump() + ); + } + } + } + } + + /// Shorthand: wait for a `LISTENING ` announcement, return the addr. + pub fn wait_listening(&mut self) -> String { + let line = self.wait_line("LISTENING announcement", |l| l.starts_with("LISTENING ")); + line["LISTENING ".len()..].to_string() + } + + /// Wait for the process to exit; panics with the transcript on timeout. + pub fn wait_exit(&mut self) -> ExitStatus { + let deadline = Instant::now() + WAIT; + loop { + let polled = match self.child.as_mut() { + Some(c) => c.try_wait(), + None => panic!("node {:?}: already reaped", self.role), + }; + match polled { + Ok(Some(status)) => { + // Drain remaining output into the transcript for dumps. + self.drain_stderr(); + while let Ok(l) = self.stdout_rx.try_recv() { + self.transcript.push(l); + } + self.child = None; + return status; + } + Ok(None) => { + if Instant::now() >= deadline { + self.drain_stderr(); + self.kill(); + panic!( + "node {:?}: did not exit within {WAIT:?}; transcript:\n{}", + self.role, + self.dump() + ); + } + std::thread::sleep(Duration::from_millis(10)); + } + Err(e) => panic!("node {:?}: try_wait failed: {e}", self.role), + } + } + } + + /// Wait for exit and require success, dumping the transcript otherwise. + pub fn wait_exit_ok(&mut self) { + let status = self.wait_exit(); + assert!( + status.success(), + "node {:?}: exited with {status}; transcript:\n{}", + self.role, + self.dump() + ); + } + + /// The child's OS pid, if not yet reaped. + pub fn pid(&self) -> Option { + self.child.as_ref().map(Child::id) + } + + /// SIGKILL + reap now (idempotent). + pub fn kill(&mut self) { + if let Some(mut child) = self.child.take() { + let _ = child.kill(); + let _ = child.wait(); + } + } +} + +impl Drop for Node { + fn drop(&mut self) { + self.kill(); + } +} From 9e490384740c4a051ff986a6859ea3b8c3fe2280 Mon Sep 17 00:00:00 2001 From: claude Date: Fri, 14 Aug 2026 17:17:14 +0000 Subject: [PATCH 05/14] =?UTF-8?q?feat(cluster):=20RFC=20010=20c5=20?= =?UTF-8?q?=E2=80=94=20handshake=20as=20a=20pure=20state=20machine?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Frames in, actions out — no IO, no clocks, no actors; the c6 connection actor will drive it. Initiator (dial: emit Hello, interpret the single response) and Responder (accept: judge the first frame) as consuming-self machines; check order proto -> hash -> name -> tie-break. Driver-supplied HelloCtx carries the two facts the pure machine cannot know (name claimed, own dial in flight). Tie-break ratified as a wire fact: the smaller name's dial survives; the losing inbound closes silently (both ends compute the same verdict, no reject frame needed). Peer's own name offered => NameTaken. build_hash is config-supplied; derivation lands with c6. --- src/cluster.rs | 4 +- src/cluster/handshake.rs | 173 ++++++++++++++++++++++++++ tests/cluster_handshake.rs | 249 +++++++++++++++++++++++++++++++++++++ 3 files changed, 425 insertions(+), 1 deletion(-) create mode 100644 src/cluster/handshake.rs create mode 100644 tests/cluster_handshake.rs diff --git a/src/cluster.rs b/src/cluster.rs index 950900c..16b78b8 100644 --- a/src/cluster.rs +++ b/src/cluster.rs @@ -2,7 +2,9 @@ //! //! c1: feature flag + optional deps. c2: the owned envelope. c3: the //! transport trait (control connection), framed codec, and the TCP + -//! loopback impls. Everything above them lands in later chunks. +//! loopback impls. c5: the handshake state machine. Everything above them +//! lands in later chunks. pub mod envelope; +pub mod handshake; pub mod transport; diff --git a/src/cluster/handshake.rs b/src/cluster/handshake.rs new file mode 100644 index 0000000..9e05f32 --- /dev/null +++ b/src/cluster/handshake.rs @@ -0,0 +1,173 @@ +//! RFC 010 c5 — the handshake as a pure state machine. +//! +//! Frames in, actions out — no IO, no clocks, no actors. The c6 connection +//! actor drives these machines and executes their actions; everything +//! time-shaped (handshake deadline, heartbeats) lives there. + +use crate::cluster::envelope::{Frame, NodeMeta, RejectReason, PROTO_VERSION}; +use crate::pg::Incarnation; + +/// This node's identity and metadata, as offered in (or checked against) a +/// `Hello`. +#[derive(Debug, Clone)] +pub struct Local { + pub node_name: String, + pub incarnation: Incarnation, + pub build_hash: u64, + pub meta: NodeMeta, +} + +/// The peer identity a successful handshake yields (what c7 feeds `node_up`). +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct Peer { + pub node_name: String, + pub incarnation: Incarnation, + pub meta: NodeMeta, +} + +/// Driver-supplied context for an inbound `Hello` — knowledge the pure +/// machine cannot have (c6 owns the connection table and dial set). +#[derive(Debug, Clone, Copy, Default)] +pub struct HelloCtx { + /// The offered name is already claimed by an established peer. + pub name_claimed: bool, + /// We have our own dial in flight to this peer name. + pub dialing_this_peer: bool, +} + +/// Simultaneous-connect tie-break: does the connection dialed by +/// `dialer_name` survive against the reverse dial? +/// The rule (ratified 2026-08-14, a wire-protocol fact): the connection +/// dialed by the lexicographically **smaller** name survives. Both ends know +/// both names, so both compute the same verdict — which is why the losing +/// side may close silently instead of sending a reject. +pub fn dial_wins(dialer_name: &str, acceptor_name: &str) -> bool { + dialer_name < acceptor_name +} + +/// Dial side: emits `Hello` at construction, interprets the single response. +#[must_use] +#[derive(Debug)] +pub struct Initiator(()); + +/// What the dial side's response frame meant. +#[must_use] +#[derive(Debug, PartialEq, Eq)] +pub enum InitiatorOutcome { + Established(Peer), + Rejected(RejectReason), + /// Protocol violation before the ack — close. Carries the offending frame. + Failed(Frame), +} + +impl Initiator { + /// Start a dial-side handshake: the returned frame is the `Hello` to + /// send; the returned machine is the right to interpret the response. + pub fn new(local: &Local) -> (Self, Frame) { + let hello = Frame::Hello { + proto_version: PROTO_VERSION, + build_hash: local.build_hash, + node_name: local.node_name.clone(), + incarnation: local.incarnation, + meta: local.meta.clone(), + }; + (Initiator(()), hello) + } + + /// Interpret the response. The `HelloAck` carries no hash or version — + /// the responder already checked ours against its own, and equality is + /// symmetric, so a one-sided check is sound. + pub fn on_frame(self, frame: Frame) -> InitiatorOutcome { + match frame { + Frame::HelloAck { + node_name, + incarnation, + meta, + } => InitiatorOutcome::Established(Peer { + node_name, + incarnation, + meta, + }), + Frame::HelloReject { reason } => InitiatorOutcome::Rejected(reason), + other => InitiatorOutcome::Failed(other), + } + } +} + +/// Accept side: awaits exactly one `Hello`, answers or closes. +#[must_use] +#[derive(Debug)] +pub struct Responder { + local: Local, +} + +/// What to do with an inbound connection's first frame. +#[must_use] +#[derive(Debug, PartialEq, Eq)] +pub enum ResponderOutcome { + /// Send the ack; the connection is established. + Accepted { reply: Frame, peer: Peer }, + /// Send the reject, then close. + Rejected { reply: Frame, reason: RejectReason }, + /// Lost the simultaneous-connect tie-break: close silently, no frame. + TieBreakLoss, + /// Protocol violation before Hello — close, no reply. Carries the frame. + Failed(Frame), +} + +impl Responder { + pub fn new(local: Local) -> Self { + Responder { local } + } + + /// Judge the connection's first frame. Check order is proto → hash → + /// name → tie-break: validity before identity. `HelloReject` is the + /// cross-version compatibility anchor, so a version-mismatched peer + /// still gets one. + pub fn on_frame(self, frame: Frame, ctx: HelloCtx) -> ResponderOutcome { + let Frame::Hello { + proto_version, + build_hash, + node_name, + incarnation, + meta, + } = frame + else { + return ResponderOutcome::Failed(frame); + }; + + let reject = |reason| ResponderOutcome::Rejected { + reply: Frame::HelloReject { reason }, + reason, + }; + + if proto_version != PROTO_VERSION { + return reject(RejectReason::ProtoVersion); + } + if build_hash != self.local.build_hash { + return reject(RejectReason::HashMismatch); + } + if node_name == self.local.node_name || ctx.name_claimed { + return reject(RejectReason::NameTaken); + } + // Simultaneous connect: the inbound frame is the peer's dial. If our + // own in-flight dial wins instead, drop this one silently — the peer + // computes the same verdict (see `dial_wins`). + if ctx.dialing_this_peer && !dial_wins(&node_name, &self.local.node_name) { + return ResponderOutcome::TieBreakLoss; + } + + ResponderOutcome::Accepted { + reply: Frame::HelloAck { + node_name: self.local.node_name, + incarnation: self.local.incarnation, + meta: self.local.meta, + }, + peer: Peer { + node_name, + incarnation, + meta, + }, + } + } +} diff --git a/tests/cluster_handshake.rs b/tests/cluster_handshake.rs new file mode 100644 index 0000000..0eede87 --- /dev/null +++ b/tests/cluster_handshake.rs @@ -0,0 +1,249 @@ +//! RFC 010 c5 — handshake state-machine tests (roadmap: happy path; hash +//! mismatch; proto-version mismatch; name already claimed; simultaneous-connect +//! tie-break; garbage before Hello). Pure — no IO, no actors, no runtime. +#![cfg(feature = "cluster")] + +use smarm::cluster::envelope::{Frame, NodeMeta, RejectReason, PROTO_VERSION}; +use smarm::cluster::handshake::{ + dial_wins, HelloCtx, Initiator, InitiatorOutcome, Local, Responder, ResponderOutcome, +}; +use smarm::pg::Incarnation; + +const HASH: u64 = 0xDEAD_BEEF_CAFE_F00D; + +fn local(name: &str) -> Local { + Local { + node_name: name.into(), + incarnation: Incarnation::new(7), + build_hash: HASH, + meta: NodeMeta { + role: "worker".into(), + region: "eu-west".into(), + }, + } +} + +/// The Hello that `Initiator::new(&local(name))` emits, built by hand. +fn hello_from(name: &str) -> Frame { + let l = local(name); + Frame::Hello { + proto_version: PROTO_VERSION, + build_hash: l.build_hash, + node_name: l.node_name, + incarnation: l.incarnation, + meta: l.meta, + } +} + +#[test] +fn happy_path_establishes_both_ends() { + // alpha dials beta. + let (initiator, hello) = Initiator::new(&local("alpha")); + assert_eq!(hello, hello_from("alpha"), "initiator emits its identity"); + + let responder = Responder::new(local("beta")); + let (reply, peer) = match responder.on_frame(hello, HelloCtx::default()) { + ResponderOutcome::Accepted { reply, peer } => (reply, peer), + other => panic!("expected Accepted, got {other:?}"), + }; + assert_eq!(peer.node_name, "alpha"); + assert_eq!(peer.incarnation, Incarnation::new(7)); + assert_eq!(peer.meta.role, "worker"); + + // The ack carries the responder's identity, no hash/version (one-sided + // check — sound because equality is symmetric). + let l = local("beta"); + assert_eq!( + reply, + Frame::HelloAck { + node_name: l.node_name, + incarnation: l.incarnation, + meta: l.meta, + } + ); + + match initiator.on_frame(reply) { + InitiatorOutcome::Established(peer) => { + assert_eq!(peer.node_name, "beta"); + assert_eq!(peer.incarnation, Incarnation::new(7)); + assert_eq!(peer.meta.region, "eu-west"); + } + other => panic!("expected Established, got {other:?}"), + } +} + +#[test] +fn hash_mismatch_rejected() { + let responder = Responder::new(local("beta")); + let hello = Frame::Hello { + proto_version: PROTO_VERSION, + build_hash: HASH ^ 1, + node_name: "alpha".into(), + incarnation: Incarnation::new(7), + meta: local("alpha").meta, + }; + match responder.on_frame(hello, HelloCtx::default()) { + ResponderOutcome::Rejected { reply, reason } => { + assert_eq!(reason, RejectReason::HashMismatch); + assert_eq!(reply, Frame::HelloReject { reason }); + } + other => panic!("expected Rejected, got {other:?}"), + } + + // The dialer side of the same story: a reject frame comes back. + let (initiator, _hello) = Initiator::new(&local("alpha")); + match initiator.on_frame(Frame::HelloReject { + reason: RejectReason::HashMismatch, + }) { + InitiatorOutcome::Rejected(RejectReason::HashMismatch) => {} + other => panic!("expected Rejected(HashMismatch), got {other:?}"), + } +} + +#[test] +fn proto_version_mismatch_rejected_and_checked_first() { + // Both proto and hash wrong: proto wins — nothing after the version can + // be trusted, and HelloReject is the cross-version compatibility anchor. + let responder = Responder::new(local("beta")); + let hello = Frame::Hello { + proto_version: PROTO_VERSION + 1, + build_hash: HASH ^ 1, + node_name: "alpha".into(), + incarnation: Incarnation::new(7), + meta: local("alpha").meta, + }; + match responder.on_frame(hello, HelloCtx::default()) { + ResponderOutcome::Rejected { reason, .. } => { + assert_eq!(reason, RejectReason::ProtoVersion); + } + other => panic!("expected Rejected, got {other:?}"), + } +} + +#[test] +fn claimed_name_rejected() { + let responder = Responder::new(local("beta")); + let ctx = HelloCtx { + name_claimed: true, + dialing_this_peer: false, + }; + match responder.on_frame(hello_from("alpha"), ctx) { + ResponderOutcome::Rejected { reason, .. } => { + assert_eq!(reason, RejectReason::NameTaken); + } + other => panic!("expected Rejected, got {other:?}"), + } +} + +#[test] +fn own_name_offered_rejected_as_name_taken() { + // Self-connect or genuine collision: the responder's own name arrives. + let responder = Responder::new(local("beta")); + match responder.on_frame(hello_from("beta"), HelloCtx::default()) { + ResponderOutcome::Rejected { reason, .. } => { + assert_eq!(reason, RejectReason::NameTaken); + } + other => panic!("expected Rejected, got {other:?}"), + } +} + +#[test] +fn hash_checked_before_name() { + // Wrong hash AND claimed name: hash wins (validity before identity). + let responder = Responder::new(local("beta")); + let hello = Frame::Hello { + proto_version: PROTO_VERSION, + build_hash: HASH ^ 1, + node_name: "alpha".into(), + incarnation: Incarnation::new(7), + meta: local("alpha").meta, + }; + let ctx = HelloCtx { + name_claimed: true, + dialing_this_peer: false, + }; + match responder.on_frame(hello, ctx) { + ResponderOutcome::Rejected { reason, .. } => { + assert_eq!(reason, RejectReason::HashMismatch); + } + other => panic!("expected Rejected, got {other:?}"), + } +} + +#[test] +fn dial_wins_is_deterministic_and_antisymmetric() { + // The smaller name's dial survives; both ends compute the same verdict. + assert!(dial_wins("alpha", "beta")); + assert!(!dial_wins("beta", "alpha")); + for (a, b) in [("a", "b"), ("node-1", "node-2"), ("x", "xx")] { + assert_ne!(dial_wins(a, b), dial_wins(b, a), "({a}, {b})"); + } +} + +#[test] +fn simultaneous_connect_exactly_one_side_accepts() { + // alpha and beta dial each other at once. Each responder sees the peer's + // Hello while its own dial is in flight. + let ctx = HelloCtx { + name_claimed: false, + dialing_this_peer: true, + }; + + // On beta: inbound is alpha's dial; alpha < beta, so the inbound wins. + let on_beta = Responder::new(local("beta")).on_frame(hello_from("alpha"), ctx); + assert!( + matches!(on_beta, ResponderOutcome::Accepted { .. }), + "beta must accept alpha's dial, got {on_beta:?}" + ); + + // On alpha: inbound is beta's dial; it loses — close silently, no frame + // (ratified: both ends can compute the outcome, a reject adds nothing). + let on_alpha = Responder::new(local("alpha")).on_frame(hello_from("beta"), ctx); + assert!( + matches!(on_alpha, ResponderOutcome::TieBreakLoss), + "alpha must silently drop beta's dial, got {on_alpha:?}" + ); +} + +#[test] +fn tiebreak_loss_only_applies_when_dialing() { + // Same inbound Hello, no dial in flight: plain accept. + let on_alpha = Responder::new(local("alpha")).on_frame(hello_from("beta"), HelloCtx::default()); + assert!(matches!(on_alpha, ResponderOutcome::Accepted { .. })); +} + +#[test] +fn garbage_before_hello_fails_without_reply() { + // Any valid-but-wrong frame before Hello is a protocol violation: close, + // no reject frame. (Undecodable bytes are the codec's Err, not ours.) + for frame in [ + Frame::Heartbeat, + Frame::HelloAck { + node_name: "alpha".into(), + incarnation: Incarnation::new(7), + meta: local("alpha").meta, + }, + Frame::Demonitor { monitor_id: 3 }, + ] { + let out = Responder::new(local("beta")).on_frame(frame.clone(), HelloCtx::default()); + match out { + ResponderOutcome::Failed(f) => assert_eq!(f, frame), + other => panic!("expected Failed({frame:?}), got {other:?}"), + } + } +} + +#[test] +fn garbage_before_ack_fails_the_initiator() { + for frame in [ + Frame::Heartbeat, + hello_from("beta"), + Frame::Demonitor { monitor_id: 3 }, + ] { + let (initiator, _hello) = Initiator::new(&local("alpha")); + match initiator.on_frame(frame.clone()) { + InitiatorOutcome::Failed(f) => assert_eq!(f, frame), + other => panic!("expected Failed({frame:?}), got {other:?}"), + } + } +} From bbaaa062e31dce9bcf0e258257cd0f0e9bf7f1a4 Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 14 Aug 2026 19:23:39 +0000 Subject: [PATCH 06/14] =?UTF-8?q?feat(cluster):=20RFC=20010=20c6a=20?= =?UTF-8?q?=E2=80=94=20connection=20actor=20+=20manager=20subtree?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Per-peer connection actor as a single select-loop plain actor owning the whole FramedConn: one select folds its command inbox and the transport's readable arm, so reads and control share one execution context — no reader thread, no read/write split. The handshake is bypassed here (c6b wires it); the actor is spawned already-established and self-registers with the manager. Manager gen_server: the peer-name -> conn-pid registry and the uniqueness source the handshake's NameTaken depends on. It monitors each connection, so the table self-heals on any exit path. Explicit supervision subtree keeps the manager up; connections are dynamic and monitored, never restarted (c7 re-dials). Transport gains an additive Conn::readable_arm -> Option (default None; TCP returns its fd's arm, loopback stays None). Existing c3 transport tests unchanged. Lifecycle test over localhost TCP: up reflected in the table, commanded shutdown reaps exactly one, peer EOF reaps the other. --- src/cluster.rs | 57 ++++++++++++++- src/cluster/conn.rs | 120 ++++++++++++++++++++++++++++++++ src/cluster/manager.rs | 110 +++++++++++++++++++++++++++++ src/cluster/transport.rs | 17 +++++ src/cluster/transport/tcp.rs | 4 ++ tests/cluster_conn_lifecycle.rs | 104 +++++++++++++++++++++++++++ 6 files changed, 410 insertions(+), 2 deletions(-) create mode 100644 src/cluster/conn.rs create mode 100644 src/cluster/manager.rs create mode 100644 tests/cluster_conn_lifecycle.rs diff --git a/src/cluster.rs b/src/cluster.rs index 16b78b8..f5bfffc 100644 --- a/src/cluster.rs +++ b/src/cluster.rs @@ -2,9 +2,62 @@ //! //! c1: feature flag + optional deps. c2: the owned envelope. c3: the //! transport trait (control connection), framed codec, and the TCP + -//! loopback impls. c5: the handshake state machine. Everything above them -//! lands in later chunks. +//! loopback impls. c5: the handshake state machine. c6: the connection +//! [`manager`] (registry) and per-peer connection actors ([`conn`]), started +//! as an explicit supervision subtree. Everything above them lands in later +//! chunks. +pub mod conn; pub mod envelope; pub mod handshake; +pub mod manager; pub mod transport; + +use std::time::Duration; + +use crate::gen_server::{self, GenServerBuilder}; +use crate::monitor::monitor; +use crate::scheduler::{sleep, spawn, JoinHandle}; +use crate::supervisor::{ChildSpec, OneForOne, Restart}; + +pub use conn::{spawn_established, ConnHandle}; +pub use manager::{Manager, MANAGER}; + +/// A running cluster subtree: an explicitly-started supervisor over the +/// connection [`Manager`]. Roles will eventually mount this subtree; until the +/// role mechanism lands it is started by hand (RFC 010 §7). Dropping the handle +/// detaches the subtree, which keeps running for the life of the runtime. +pub struct Cluster { + _sup: JoinHandle, +} + +/// Start the cluster subtree and block until the manager is registered and +/// ready to answer. The manager is a supervised child (restarted on crash); +/// per-peer connection actors are dynamic and monitored by the manager rather +/// than statically supervised — a lost connection is re-established by dialing +/// (c7), never resurrected onto a stale socket. +pub fn start() -> Cluster { + let sup = spawn(|| { + OneForOne::new() + .child(ChildSpec::new(Restart::Permanent, manager_child)) + .run() + }); + while gen_server::whereis_server(MANAGER).is_none() { + sleep(Duration::from_millis(1)); + } + Cluster { _sup: sup } +} + +/// The supervised manager child body. It *is* the child actor: it starts the +/// named manager, then parks on the manager's own termination so this actor's +/// lifetime tracks the manager's — the supervisor's restart accounting keys off +/// this actor exiting. +fn manager_child() { + let m = match GenServerBuilder::new(Manager::new()).named(MANAGER).start() { + Ok(m) => m, + // Name still held by a not-yet-reaped prior instance: return and let + // the supervisor retry under its restart policy. + Err(_) => return, + }; + let _ = monitor(m.pid()).rx.recv(); +} diff --git a/src/cluster/conn.rs b/src/cluster/conn.rs new file mode 100644 index 0000000..78122e7 --- /dev/null +++ b/src/cluster/conn.rs @@ -0,0 +1,120 @@ +//! RFC 010 c6 — the per-peer connection actor. +//! +//! One actor per established control connection. It owns the whole +//! [`FramedConn`] and, in a single [`select`](crate::select), waits on two +//! things at once: its command inbox and the connection becoming readable (the +//! [`FdArm`](crate::scheduler::FdArm) the transport hands back). That is why it +//! is a plain select-loop actor rather than a `gen_server` or `gen_statem` — +//! neither of those can fold fd-readiness into its wait, and folding it in is +//! the whole job. The single owner sends and receives on the one `FramedConn`, +//! so no read/write split is needed. +//! +//! The handshake completes *before* this actor exists (on the accept/connect +//! path — c6b) and produces the [`Peer`]; the actor registers that peer with +//! the [`manager`](crate::cluster::manager), which monitors it so any exit +//! deregisters the connection. Heartbeat send and fixed-timeout liveness join +//! the loop in c6c (the timeout arm of the same `select`). + +use crate::channel::{channel, Receiver, Selectable, Sender}; +use crate::cluster::handshake::Peer; +use crate::cluster::manager::{Call, Registered, Reply, MANAGER}; +use crate::cluster::transport::FramedConn; +use crate::gen_server; +use crate::scheduler::{self, spawn}; + +/// Commands to a running connection actor. +enum Cmd { + Shutdown, +} + +/// A handle to a running connection actor. +pub struct ConnHandle { + cmd_tx: Sender, +} + +impl ConnHandle { + /// Ask the connection to close and exit. Idempotent, and a no-op if the + /// actor has already gone. + pub fn shutdown(&self) { + let _ = self.cmd_tx.send(Cmd::Shutdown); + } +} + +/// Spawn a connection actor for an **already-established** connection: the +/// handshake has completed elsewhere and produced `peer`. Returns as soon as +/// the actor is spawned; the actor's first act is to register with the manager. +pub fn spawn_established(framed: FramedConn, peer: Peer) -> ConnHandle { + let (cmd_tx, cmd_rx) = channel(); + spawn(move || run(framed, peer, cmd_rx)); + ConnHandle { cmd_tx } +} + +fn run(mut framed: FramedConn, peer: Peer, cmd_rx: Receiver) { + let me = scheduler::self_pid(); + match gen_server::call( + MANAGER, + Call::Register { + name: peer.node_name.clone(), + pid: me, + }, + ) { + Ok(Reply::Registered(Registered::Ok)) => {} + // Duplicate name, or the manager is unreachable: do not run. The + // connection drops (closing the socket) as `framed` falls out of scope. + _ => return, + } + + loop { + match framed.readable_arm() { + Some(arm) => { + let arms: [&dyn Selectable; 2] = [&cmd_rx, &arm]; + match crate::channel::try_select(&arms) { + Ok(0) => { + if should_stop(&cmd_rx) { + break; + } + } + Ok(_) => { + if pump_readable(&mut framed) { + break; + } + } + // The fd arm failed to register — the connection is gone. + Err(_) => break, + } + } + None => { + // No fd to select on (loopback): only a command can end the + // wait. Liveness over such a transport is out of scope. + let arms: [&dyn Selectable; 1] = [&cmd_rx]; + let _ = crate::channel::select(&arms); + if should_stop(&cmd_rx) { + break; + } + } + } + } + + framed.close(); +} + +/// Drain the command arm. Returns `true` when the actor should exit — a +/// shutdown was requested, or the last handle was dropped. +fn should_stop(cmd_rx: &Receiver) -> bool { + match cmd_rx.try_recv() { + Ok(Some(Cmd::Shutdown)) => true, + Ok(None) => false, // spurious wake + Err(_) => true, // all senders dropped + } +} + +/// Consume whatever is readable now. Returns `true` when the connection has +/// ended (clean EOF or an unrecoverable stream error). This chunk does not +/// interpret frames; c6c handles heartbeats and resets the liveness timer here. +fn pump_readable(framed: &mut FramedConn) -> bool { + match framed.recv() { + Ok(Some(_frame)) => false, + Ok(None) => true, // clean EOF at a frame boundary + Err(_) => true, // corrupt / truncated / io + } +} diff --git a/src/cluster/manager.rs b/src/cluster/manager.rs new file mode 100644 index 0000000..9ab3218 --- /dev/null +++ b/src/cluster/manager.rs @@ -0,0 +1,110 @@ +//! RFC 010 c6 — the cluster connection manager. +//! +//! One manager per runtime: the single registry of live peer connections and +//! the source of truth for whether a peer name is already claimed. Every +//! connection actor registers here as its first act and is *monitored* by the +//! manager, so the table self-heals on any exit path — a connection that +//! panics, is cancelled, or closes cleanly is removed without cooperation from +//! the dying actor. +//! +//! Membership as consumers will see it (the `node_up`/`node_down` interface and +//! the view) and the connector dial loop are c7, built on top of this table. +//! What lives here is only the table itself and the uniqueness rule the +//! handshake's `NameTaken` verdict depends on. + +use std::collections::HashMap; + +use crate::gen_server::{GenServer, GenServerCtx, GenServerName, Watcher}; +use crate::monitor::{monitor, Down}; +use crate::pid::Pid; + +/// Well-known name of the singleton manager within a runtime. Connection +/// actors reach it by name rather than by a passed-around ref, so a restarted +/// manager is always found at the same key. +pub const MANAGER: GenServerName = GenServerName::new("smarm.cluster.manager"); + +/// The connection registry: peer name → the connection actor that owns that +/// peer's control connection. +pub struct Manager { + conns: HashMap, + watcher: Option>, +} + +impl Manager { + pub fn new() -> Self { + Manager { + conns: HashMap::new(), + watcher: None, + } + } +} + +impl Default for Manager { + fn default() -> Self { + Manager::new() + } +} + +/// Outcome of a [`Call::Register`]. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum Registered { + /// The name was free; this connection is now the peer of record. + Ok, + /// Another live connection already holds this name — the caller lost the + /// race (or is a duplicate) and must not run. + Duplicate, +} + +/// Requests to the manager. +pub enum Call { + /// A freshly-established connection actor claims its peer's name. The pid + /// is the calling connection actor, which the manager then monitors. + Register { name: String, pid: Pid }, + /// The current peer names, sorted. For observation and tests. + Peers, +} + +/// Replies from the manager. +#[derive(Debug)] +pub enum Reply { + Registered(Registered), + Peers(Vec), +} + +impl GenServer for Manager { + type Call = Call; + type Reply = Reply; + type Cast = (); + type Info = (); + type Timer = (); + + fn init(&mut self, ctx: &GenServerCtx) { + self.watcher = Some(ctx.watcher()); + } + + fn handle_call(&mut self, request: Call) -> Reply { + match request { + Call::Register { name, pid } => { + if self.conns.contains_key(&name) { + return Reply::Registered(Registered::Duplicate); + } + if let Some(w) = &self.watcher { + w.watch(monitor(pid)); + } + self.conns.insert(name, pid); + Reply::Registered(Registered::Ok) + } + Call::Peers => { + let mut names: Vec = self.conns.keys().cloned().collect(); + names.sort(); + Reply::Peers(names) + } + } + } + + fn handle_cast(&mut self, _request: ()) {} + + fn handle_down(&mut self, down: Down) { + self.conns.retain(|_, pid| *pid != down.pid); + } +} diff --git a/src/cluster/transport.rs b/src/cluster/transport.rs index f16057d..d62d786 100644 --- a/src/cluster/transport.rs +++ b/src/cluster/transport.rs @@ -50,6 +50,17 @@ pub trait Conn: Send { /// Diagnostic label for logs only. Mesh identity comes from the /// handshake (`Hello`/`HelloAck`), never from the transport. fn peer_addr(&self) -> String; + + /// Readiness as a [`select`](crate::select) arm, for transports backed by + /// a file descriptor. `Some` lets a driver wait on "this connection is + /// readable" alongside an ordinary command inbox in a single `select`, so + /// one actor can interleave reading with control messages without a + /// second thread. The default is `None`: a transport with no fd (the + /// in-memory loopback) cannot be selected on and must be driven another + /// way. + fn readable_arm(&self) -> Option { + None + } } /// A bound listen point producing inbound [`Conn`]s. @@ -191,4 +202,10 @@ impl FramedConn { pub fn peer_addr(&self) -> String { self.conn.peer_addr() } + + /// The underlying connection's readiness arm, if it is fd-backed (see + /// [`Conn::readable_arm`]). + pub fn readable_arm(&self) -> Option { + self.conn.readable_arm() + } } diff --git a/src/cluster/transport/tcp.rs b/src/cluster/transport/tcp.rs index c725088..d3764e6 100644 --- a/src/cluster/transport/tcp.rs +++ b/src/cluster/transport/tcp.rs @@ -180,6 +180,10 @@ impl Conn for TcpConn { Err(_) => "".to_string(), } } + + fn readable_arm(&self) -> Option { + Some(crate::scheduler::FdArm::readable(self.fd())) + } } // --------------------------------------------------------------------------- diff --git a/tests/cluster_conn_lifecycle.rs b/tests/cluster_conn_lifecycle.rs new file mode 100644 index 0000000..040afbc --- /dev/null +++ b/tests/cluster_conn_lifecycle.rs @@ -0,0 +1,104 @@ +//! RFC 010 c6a — connection-actor lifecycle against the manager table. +//! +//! The handshake is bypassed here (c6b wires it): each connection is +//! constructed already-established over a real localhost TCP pair, handed a +//! fabricated `Peer`, and spawned. The actor registers with the manager, which +//! monitors it, so the table reflects the connection while it lives and reaps +//! it on any exit path. This proves three things at once: a live connection +//! shows up, a commanded shutdown removes exactly that one, and a peer close +//! (EOF, no command) removes the other. +//! +//! TCP parks the calling actor, so everything runs inside `smarm::run`; the +//! single-threaded runtime is fine because every wait is a cooperative fd park. +#![cfg(feature = "cluster")] + +use std::time::Duration; + +use smarm::cluster::envelope::NodeMeta; +use smarm::cluster::handshake::Peer; +use smarm::cluster::manager::{Call, Manager, Reply, MANAGER}; +use smarm::cluster::spawn_established; +use smarm::cluster::transport::tcp::TcpTransport; +use smarm::cluster::transport::{Conn, FramedConn, Transport}; +use smarm::gen_server::{self, GenServerBuilder}; +use smarm::pg::Incarnation; +use smarm::{run, sleep}; + +/// A fabricated post-handshake peer identity. Only `node_name` matters to the +/// manager table; the rest is filler until c7 consumes it. +fn peer(name: &str) -> Peer { + Peer { + node_name: name.to_string(), + incarnation: Incarnation::new(1), + meta: NodeMeta { + role: "test".to_string(), + region: "test".to_string(), + }, + } +} + +/// One established transport pair over localhost. Relies on TCP backlog so the +/// sequential dial-then-accept needs no concurrent acceptor (same assumption as +/// the c3 conformance suite). +fn pair(t: &dyn Transport) -> (Box, Box) { + let mut l = t.listen("127.0.0.1:0").unwrap(); + let a = t.dial(&l.local_addr()).unwrap(); + let b = l.accept().unwrap(); + (a, b) +} + +/// Poll the manager until its peer set matches `expected` (sorted), or fail. +/// The bound is generous against a sub-millisecond real cost. +fn wait_peers(expected: &[&str]) { + let want: Vec = expected.iter().map(|s| s.to_string()).collect(); + for _ in 0..2000 { + if let Ok(Reply::Peers(got)) = gen_server::call(MANAGER, Call::Peers) { + if got == want { + return; + } + } + sleep(Duration::from_millis(1)); + } + let got = gen_server::call(MANAGER, Call::Peers); + panic!("timed out waiting for peers == {want:?}; last = {got:?}"); +} + +#[test] +fn connection_up_commanded_shutdown_and_eof_all_reflected_in_table() { + run(|| { + // The manager, started plainly and reachable at its well-known name. + // (The supervised subtree in `cluster::start` is permanent by design; + // a plainly-started manager lets this test terminate cleanly.) + let mgr = GenServerBuilder::new(Manager::new()) + .named(MANAGER) + .start() + .expect("manager name is free"); + + let t = TcpTransport; + let (a1, b1) = pair(&t); + let (a2, b2) = pair(&t); + + // Manage the `a` ends as peers node-b and node-c; keep the `b` far ends + // open so neither socket is closed from the far side yet. + let h1 = spawn_established(FramedConn::new(a1), peer("node-b")); + let _h2 = spawn_established(FramedConn::new(a2), peer("node-c")); + + // Up: both connections register and the table shows them. + wait_peers(&["node-b", "node-c"]); + + // A commanded shutdown reaps exactly its own connection. + h1.shutdown(); + wait_peers(&["node-c"]); + + // A peer close (EOF) reaps the other with no command at all. + drop(b2); + wait_peers(&[]); + + // node-b's far end stayed open until here, so its removal above was the + // shutdown command and not an EOF. + drop(b1); + + // All connection actors have exited; stop the manager so `run` returns. + mgr.shutdown(); + }); +} From c8ed858e4c590f4efcdf7b083284600a8313c976 Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 14 Aug 2026 21:09:08 +0000 Subject: [PATCH 07/14] =?UTF-8?q?feat(cluster):=20RFC=20010=20c6b=20?= =?UTF-8?q?=E2=80=94=20handshake=20on=20the=20accept/connect=20path?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Drive the c5 machines as straight-line code on the path (D8): dial_handshake and accept_handshake do the IO on a shared FramedConn, and a connection actor is spawned only after a successful handshake. Rejects, tie-break losses (D7), protocol faults and timeouts are all resolved on the path by closing, so no actor ever exists for a connection that did not establish. The whole FramedConn travels into spawn_established, carrying any read-ahead past the handshake frames. Handshake deadlines land here rather than in c6c: FramedConn::recv_deadline enforces them between reads via the connection's fd arm, so a peer that connects and goes silent cannot wedge the acceptor. Connection lifetime moves to the manager (pulled forward from c7). The path registers each established connection and hands over its ConnHandle; the manager owns it, monitors the actor, and tears the connection down on Disconnect, on peer close, or at manager shutdown. spawn_established returns a Pid, so a connection neither outlives nor dies with whichever actor established it — the ownership that made two-node teardown unorderable. The manager also tracks in-flight dial intents, monitored so a panicking dial cannot wedge the tie-break, and answers HelloCtx for the accept path. --- src/cluster.rs | 5 +- src/cluster/conn.rs | 71 +++-- src/cluster/connect.rs | 345 +++++++++++++++++++++++ src/cluster/manager.rs | 99 ++++++- src/cluster/transport.rs | 57 ++++ src/cluster/transport/tcp.rs | 4 + tests/cluster_conn_lifecycle.rs | 30 +- tests/cluster_connect.rs | 472 ++++++++++++++++++++++++++++++++ 8 files changed, 1037 insertions(+), 46 deletions(-) create mode 100644 src/cluster/connect.rs create mode 100644 tests/cluster_connect.rs diff --git a/src/cluster.rs b/src/cluster.rs index f5bfffc..b3cea4f 100644 --- a/src/cluster.rs +++ b/src/cluster.rs @@ -4,10 +4,12 @@ //! transport trait (control connection), framed codec, and the TCP + //! loopback impls. c5: the handshake state machine. c6: the connection //! [`manager`] (registry) and per-peer connection actors ([`conn`]), started -//! as an explicit supervision subtree. Everything above them lands in later +//! as an explicit supervision subtree, plus the handshake on the +//! accept/connect path ([`connect`]). Everything above them lands in later //! chunks. pub mod conn; +pub mod connect; pub mod envelope; pub mod handshake; pub mod manager; @@ -21,6 +23,7 @@ use crate::scheduler::{sleep, spawn, JoinHandle}; use crate::supervisor::{ChildSpec, OneForOne, Restart}; pub use conn::{spawn_established, ConnHandle}; +pub use connect::{dial, spawn_acceptor, AcceptorHandle}; pub use manager::{Manager, MANAGER}; /// A running cluster subtree: an explicitly-started supervisor over the diff --git a/src/cluster/conn.rs b/src/cluster/conn.rs index 78122e7..43d99cb 100644 --- a/src/cluster/conn.rs +++ b/src/cluster/conn.rs @@ -10,60 +10,85 @@ //! so no read/write split is needed. //! //! The handshake completes *before* this actor exists (on the accept/connect -//! path — c6b) and produces the [`Peer`]; the actor registers that peer with -//! the [`manager`](crate::cluster::manager), which monitors it so any exit -//! deregisters the connection. Heartbeat send and fixed-timeout liveness join -//! the loop in c6c (the timeout arm of the same `select`). +//! path — c6b) and produces the [`Peer`]; the *path* then registers the +//! connection with the [`manager`](crate::cluster::manager), which takes +//! ownership of its [`ConnHandle`] and monitors the actor, so any exit +//! deregisters the connection. The actor itself holds no authority over its +//! own lifetime: it runs until the manager drops its handle (deregistration, +//! `Disconnect`, or manager shutdown) or the connection ends. Heartbeat send +//! and fixed-timeout liveness join the loop in c6c (the timeout arm of the +//! same `select`). use crate::channel::{channel, Receiver, Selectable, Sender}; use crate::cluster::handshake::Peer; use crate::cluster::manager::{Call, Registered, Reply, MANAGER}; use crate::cluster::transport::FramedConn; use crate::gen_server; -use crate::scheduler::{self, spawn}; +use crate::pid::Pid; +use crate::scheduler::spawn; /// Commands to a running connection actor. enum Cmd { Shutdown, } -/// A handle to a running connection actor. +/// The manager's authority over one connection actor: while this handle +/// lives the connection lives, and dropping it stops the actor and closes +/// the socket. Only the [`manager`](crate::cluster::manager) holds one — +/// callers of [`spawn_established`] get a [`Pid`] and no lifetime authority, +/// so a connection can never outlive, or die with, whichever actor happened +/// to establish it. pub struct ConnHandle { cmd_tx: Sender, } +impl std::fmt::Debug for ConnHandle { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.write_str("ConnHandle") + } +} + impl ConnHandle { /// Ask the connection to close and exit. Idempotent, and a no-op if the - /// actor has already gone. + /// actor has already gone. Dropping the handle does the same thing; this + /// exists for the manager's explicit `Disconnect` path. pub fn shutdown(&self) { let _ = self.cmd_tx.send(Cmd::Shutdown); } } -/// Spawn a connection actor for an **already-established** connection: the -/// handshake has completed elsewhere and produced `peer`. Returns as soon as -/// the actor is spawned; the actor's first act is to register with the manager. -pub fn spawn_established(framed: FramedConn, peer: Peer) -> ConnHandle { - let (cmd_tx, cmd_rx) = channel(); - spawn(move || run(framed, peer, cmd_rx)); - ConnHandle { cmd_tx } -} +/// The name was already claimed by a live connection, so this one was +/// refused; its actor has been stopped and its socket closed. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct RegisterRefused; -fn run(mut framed: FramedConn, peer: Peer, cmd_rx: Receiver) { - let me = scheduler::self_pid(); +/// Spawn a connection actor for an **already-established** connection (the +/// handshake completed on the path and produced `peer`) and register it with +/// the manager, synchronously, before returning. The manager takes the +/// actor's [`ConnHandle`]; the caller gets only the [`Pid`], because +/// connection lifetime belongs to the table and not to the establishing +/// actor. A refusal has already stopped the actor and closed the socket. +pub fn spawn_established(framed: FramedConn, peer: Peer) -> Result { + let (cmd_tx, cmd_rx) = channel(); + let name = peer.node_name.clone(); + let pid = spawn(move || run(framed, peer, cmd_rx)).pid(); match gen_server::call( MANAGER, Call::Register { - name: peer.node_name.clone(), - pid: me, + name, + pid, + handle: ConnHandle { cmd_tx }, }, ) { - Ok(Reply::Registered(Registered::Ok)) => {} - // Duplicate name, or the manager is unreachable: do not run. The - // connection drops (closing the socket) as `framed` falls out of scope. - _ => return, + Ok(Reply::Registered(Registered::Ok)) => Ok(pid), + // Duplicate name, or the manager is unreachable. Either way the + // handle went with the call and is dropped there (or never arrived + // and dropped with it), which stops the actor and closes the socket. + _ => Err(RegisterRefused), } +} +fn run(mut framed: FramedConn, _peer: Peer, cmd_rx: Receiver) { loop { match framed.readable_arm() { Some(arm) => { diff --git a/src/cluster/connect.rs b/src/cluster/connect.rs new file mode 100644 index 0000000..503b37c --- /dev/null +++ b/src/cluster/connect.rs @@ -0,0 +1,345 @@ +//! RFC 010 c6b — the handshake on the accept/connect path. +//! +//! Per D8 (re-amended): the c5 machines are driven by **straight-line code +//! on the path**, not by an actor. The dial side runs [`Initiator`]; the +//! acceptor loop runs [`Responder`]. A connection actor is spawned only +//! *after* a successful handshake ([`spawn_established`]); every reject, +//! protocol failure, timeout, and tie-break loss is resolved right here, +//! on the path, by closing — no actor ever exists for a connection that +//! didn't establish. +//! +//! Buffer trap (binding): the path reader and the steady-state actor share +//! ONE [`FramedConn`]. Its decode buffer may hold read-ahead past the +//! handshake frames, so the *whole* `FramedConn` travels into +//! [`spawn_established`] — never a fresh codec over the same socket. +//! +//! Layering: [`dial_handshake`] and [`accept_handshake`] are the bare path +//! steps — IO on a `FramedConn`, no manager, no actors — testable over the +//! loopback transport on plain threads. [`dial`] and [`spawn_acceptor`] are +//! the manager-integrated layer (actor context required): they keep the +//! [`manager`](crate::cluster::manager)'s dial-intent set honest and spawn +//! the connection actor on success. + +use std::io; +use std::time::{Duration, Instant}; + +use crate::channel::{channel, Receiver, Selectable, Sender}; +use crate::cluster::conn::spawn_established; +use crate::cluster::envelope::{Frame, RejectReason}; +use crate::cluster::handshake::{ + HelloCtx, Initiator, InitiatorOutcome, Local, Peer, Responder, ResponderOutcome, +}; +use crate::cluster::manager::{Call, Reply, MANAGER}; +use crate::cluster::transport::{FramedConn, Listener, RecvError, SendError, Transport}; +use crate::gen_server; +use crate::pid::Pid; +use crate::scheduler::{self, spawn}; + +/// How long either side waits for the peer's handshake frame before giving +/// up and closing. Enforced on the path via [`FramedConn::recv_deadline`], +/// so a peer that connects and goes silent cannot wedge the acceptor. +pub const HANDSHAKE_TIMEOUT: Duration = Duration::from_secs(5); + +/// Why a handshake did not establish. In every case the connection has +/// already been closed on the path by the time this is returned. +#[derive(Debug)] +pub enum HandshakeError { + /// A `HelloReject` travelled — sent by us (accept side) or received by + /// us (dial side). + Rejected(RejectReason), + /// Accept side only: the inbound dial lost the simultaneous-connect + /// tie-break (D7) and was closed silently, no frame sent. + TieBreakLoss, + /// The peer spoke a valid frame that is wrong here (non-`Hello` first + /// frame; non-response to our `Hello`), or an undecodable byte stream. + Protocol, + /// EOF before the handshake resolved. On the dial side this is also + /// what losing the tie-break looks like: the peer closes silently. + Closed, + /// [`HANDSHAKE_TIMEOUT`] (or the caller's deadline) passed first. + TimedOut, + /// The transport failed mid-handshake. + Transport(io::Error), +} + +fn from_send(e: SendError) -> HandshakeError { + match e { + // Handshake frames are small and self-made; an encode failure is a + // protocol-level impossibility, not a transport fault. + SendError::Encode(_) => HandshakeError::Protocol, + SendError::Io(e) => HandshakeError::Transport(e), + } +} + +fn from_recv(e: RecvError) -> HandshakeError { + match e { + RecvError::Corrupt(_) => HandshakeError::Protocol, + RecvError::TruncatedByPeer => HandshakeError::Closed, + RecvError::Io(e) => HandshakeError::Transport(e), + RecvError::TimedOut => HandshakeError::TimedOut, + } +} + +/// Dial-side path step: send our `Hello`, interpret the one response. On +/// `Ok` the connection is established and `framed` is live (with any +/// read-ahead intact in its buffer); on `Err` the connection is closed. +pub fn dial_handshake( + framed: &mut FramedConn, + local: &Local, + deadline: Instant, +) -> Result { + let (initiator, hello) = Initiator::new(local); + if let Err(e) = framed.send(&hello) { + framed.close(); + return Err(from_send(e)); + } + let outcome = match framed.recv_deadline(deadline) { + Ok(Some(frame)) => initiator.on_frame(frame), + Ok(None) => { + framed.close(); + return Err(HandshakeError::Closed); + } + Err(e) => { + framed.close(); + return Err(from_recv(e)); + } + }; + match outcome { + InitiatorOutcome::Established(peer) => Ok(peer), + InitiatorOutcome::Rejected(reason) => { + framed.close(); + Err(HandshakeError::Rejected(reason)) + } + InitiatorOutcome::Failed(_) => { + framed.close(); + Err(HandshakeError::Protocol) + } + } +} + +/// Accept-side path step: read the first frame, judge it, answer or close. +/// +/// `ctx_for` supplies the [`HelloCtx`] for the *offered* name — knowledge +/// only the frame reveals, which is why it is a callback and not a value +/// (the integrated acceptor asks the manager; loopback tests fabricate). +/// It is not called when the first frame is not a `Hello`. +/// +/// On `Ok` the ack has been sent and `framed` is live (read-ahead intact); +/// on `Err` any owed reject has been sent and the connection is closed. +pub fn accept_handshake( + framed: &mut FramedConn, + local: Local, + ctx_for: impl FnOnce(&str) -> HelloCtx, + deadline: Instant, +) -> Result { + let frame = match framed.recv_deadline(deadline) { + Ok(Some(frame)) => frame, + Ok(None) => { + framed.close(); + return Err(HandshakeError::Closed); + } + Err(e) => { + framed.close(); + return Err(from_recv(e)); + } + }; + let ctx = match &frame { + Frame::Hello { node_name, .. } => ctx_for(node_name), + _ => HelloCtx::default(), + }; + match Responder::new(local).on_frame(frame, ctx) { + ResponderOutcome::Accepted { reply, peer } => { + if let Err(e) = framed.send(&reply) { + framed.close(); + return Err(from_send(e)); + } + Ok(peer) + } + ResponderOutcome::Rejected { reply, reason } => { + // Best effort: the reject is the cross-version compatibility + // anchor, but if the write fails the peer sees a bare close, + // which it must survive anyway. + let _ = framed.send(&reply); + framed.close(); + Err(HandshakeError::Rejected(reason)) + } + ResponderOutcome::TieBreakLoss => { + // D7: close silently — the peer computes the same verdict. + framed.close(); + Err(HandshakeError::TieBreakLoss) + } + ResponderOutcome::Failed(_) => { + framed.close(); + Err(HandshakeError::Protocol) + } + } +} + +// --------------------------------------------------------------------------- +// Manager-integrated layer +// --------------------------------------------------------------------------- + +/// Why an integrated [`dial`] did not produce a connection. +#[derive(Debug)] +pub enum DialError { + /// Another dial to this peer name is already in flight. + AlreadyDialing, + /// The manager is not running (or answered nonsense). + ManagerUnavailable, + /// The transport could not connect. + Connect(io::Error), + /// Connected, but the handshake did not establish. + Handshake(HandshakeError), + /// The peer at `addr` established, but answered as a different name + /// than the one we dialed — the tie-break bookkeeping (keyed by the + /// dialed name) would be unsound, so the connection is closed. + PeerNameMismatch { expected: String, got: String }, + /// The handshake established, but the manager refused the registration: + /// a connection to this peer already exists. The loser has been closed. + Duplicate, +} + +/// Dial `peer_name` at `addr` and run the handshake, keeping the manager's +/// dial-intent set honest around it: the intent is registered *before* +/// connecting (so a crossing inbound `Hello` sees it) and cleared the +/// moment the handshake resolves, before the connection actor is spawned. +/// Must run inside an actor. Retrying is the caller's business (c7's dial +/// loop); a lost tie-break surfaces as `Handshake(Closed)` — the peer's +/// accepted connection is already on its way. +pub fn dial( + transport: &dyn Transport, + addr: &str, + peer_name: &str, + local: &Local, +) -> Result { + let me = scheduler::self_pid(); + match gen_server::call( + MANAGER, + Call::DialBegin { + name: peer_name.to_string(), + pid: me, + }, + ) { + Ok(Reply::DialBegan(true)) => {} + Ok(Reply::DialBegan(false)) => return Err(DialError::AlreadyDialing), + _ => return Err(DialError::ManagerUnavailable), + } + let result = connect_and_shake(transport, addr, local); + // Cleared immediately on outcome — a stale intent during the established + // window would corrupt later tie-breaks. Synchronous (a call): the + // intent is provably gone before anything else happens. + let _ = gen_server::call( + MANAGER, + Call::DialEnd { + name: peer_name.to_string(), + }, + ); + let (mut framed, peer) = result?; + if peer.node_name != peer_name { + framed.close(); + return Err(DialError::PeerNameMismatch { + expected: peer_name.to_string(), + got: peer.node_name, + }); + } + spawn_established(framed, peer).map_err(|_| DialError::Duplicate) +} + +fn connect_and_shake( + transport: &dyn Transport, + addr: &str, + local: &Local, +) -> Result<(FramedConn, Peer), DialError> { + let conn = transport.dial(addr).map_err(DialError::Connect)?; + let mut framed = FramedConn::new(conn); + let deadline = Instant::now() + HANDSHAKE_TIMEOUT; + let peer = dial_handshake(&mut framed, local, deadline).map_err(DialError::Handshake)?; + Ok((framed, peer)) +} + +/// A running acceptor. [`shutdown`](AcceptorHandle::shutdown) (or dropping +/// the last handle) stops the accept loop only: connections it established +/// belong to the [`manager`](crate::cluster::manager) and keep running, to +/// be torn down through the table (`Disconnect`, a peer close, or manager +/// shutdown). +pub struct AcceptorHandle { + cmd_tx: Sender<()>, + addr: String, +} + +impl AcceptorHandle { + /// Ask the acceptor to stop. Idempotent; a no-op if it already has. + pub fn shutdown(&self) { + let _ = self.cmd_tx.send(()); + } + + /// The concrete bound address, dialable as-is. + pub fn local_addr(&self) -> &str { + &self.addr + } +} + +/// Spawn the acceptor actor over a bound listener. Each inbound connection +/// is handshaken **inline in the loop** (a deliberate serialization: the +/// per-frame deadline bounds how long any one peer can hold the line, and +/// nothing concurrent exists to be starved before c7). The listener must be +/// fd-backed ([`Listener::readable_arm`]); the loopback listener is not, +/// and its acceptor exits immediately — loopback handshakes are driven +/// synchronously through the path fns instead, per D8. +pub fn spawn_acceptor(listener: Box, local: Local) -> AcceptorHandle { + let addr = listener.local_addr(); + let (cmd_tx, cmd_rx) = channel(); + spawn(move || accept_loop(listener, local, cmd_rx)); + AcceptorHandle { cmd_tx, addr } +} + +fn accept_loop(mut listener: Box, local: Local, cmd_rx: Receiver<()>) { + loop { + let Some(arm) = listener.readable_arm() else { + return; + }; + let arms: [&dyn Selectable; 2] = [&cmd_rx, &arm]; + match crate::channel::try_select(&arms) { + Ok(0) => match cmd_rx.try_recv() { + Ok(Some(())) => return, + Ok(None) => continue, // spurious wake + Err(_) => return, // all handles dropped + }, + Ok(_) => { + // The listener is readable: accept completes without parking. + let conn = match listener.accept() { + Ok(conn) => conn, + Err(_) => return, // listener itself is broken + }; + handle_inbound(FramedConn::new(conn), &local); + } + Err(_) => return, // fd arm failed to register: listener is gone + } + } +} + +/// Run the accept-side handshake for one inbound connection, asking the +/// manager for the [`HelloCtx`], and hand the established connection to the +/// manager. Every failure was already resolved on the path (reject sent / +/// closed, or the registration refused and the actor stopped), so there is +/// nothing for the acceptor to carry forward. +fn handle_inbound(mut framed: FramedConn, local: &Local) { + let deadline = Instant::now() + HANDSHAKE_TIMEOUT; + let ctx_for = |name: &str| match gen_server::call( + MANAGER, + Call::HelloCtx { + peer_name: name.to_string(), + }, + ) { + Ok(Reply::HelloCtx(ctx)) => ctx, + // Manager unreachable: nobody could register this connection anyway, + // so claim the name taken and reject rather than accept an orphan. + _ => HelloCtx { + name_claimed: true, + dialing_this_peer: false, + }, + }; + if let Ok(peer) = accept_handshake(&mut framed, local.clone(), ctx_for, deadline) { + let _ = spawn_established(framed, peer); + } +} diff --git a/src/cluster/manager.rs b/src/cluster/manager.rs index 9ab3218..b5153b1 100644 --- a/src/cluster/manager.rs +++ b/src/cluster/manager.rs @@ -1,11 +1,14 @@ //! RFC 010 c6 — the cluster connection manager. //! //! One manager per runtime: the single registry of live peer connections and -//! the source of truth for whether a peer name is already claimed. Every -//! connection actor registers here as its first act and is *monitored* by the -//! manager, so the table self-heals on any exit path — a connection that -//! panics, is cancelled, or closes cleanly is removed without cooperation from -//! the dying actor. +//! the source of truth for whether a peer name is already claimed. The +//! accept/connect path registers each established connection here, handing +//! over its [`ConnHandle`] — **the manager owns connection lifetime**. A +//! connection lives as long as its table entry, so it neither outlives nor +//! dies with whichever actor happened to establish it. Registered actors are +//! also *monitored*, so the table self-heals on any exit path — a connection +//! that panics, is cancelled, or closes cleanly is removed without +//! cooperation from the dying actor. //! //! Membership as consumers will see it (the `node_up`/`node_down` interface and //! the view) and the connector dial loop are c7, built on top of this table. @@ -14,6 +17,8 @@ use std::collections::HashMap; +use crate::cluster::conn::ConnHandle; +use crate::cluster::handshake::HelloCtx; use crate::gen_server::{GenServer, GenServerCtx, GenServerName, Watcher}; use crate::monitor::{monitor, Down}; use crate::pid::Pid; @@ -23,10 +28,23 @@ use crate::pid::Pid; /// manager is always found at the same key. pub const MANAGER: GenServerName = GenServerName::new("smarm.cluster.manager"); +/// One live connection's entry: the actor running it, and the handle whose +/// lifetime *is* the connection's (see the module docs). +struct ConnEntry { + pid: Pid, + _handle: ConnHandle, +} + /// The connection registry: peer name → the connection actor that owns that /// peer's control connection. pub struct Manager { - conns: HashMap, + conns: HashMap, + /// In-flight dial intents: peer name -> the actor performing the dial. + /// Registered *before* connecting so a crossing inbound `Hello` sees it + /// (the `dialing_this_peer` half of [`HelloCtx`]); cleared the moment + /// the dial resolves, and — because the dialer is monitored — on the + /// dialer's death, so a panicking dial can never wedge the tie-break. + dials: HashMap, watcher: Option>, } @@ -34,6 +52,7 @@ impl Manager { pub fn new() -> Self { Manager { conns: HashMap::new(), + dials: HashMap::new(), watcher: None, } } @@ -57,18 +76,43 @@ pub enum Registered { /// Requests to the manager. pub enum Call { - /// A freshly-established connection actor claims its peer's name. The pid - /// is the calling connection actor, which the manager then monitors. - Register { name: String, pid: Pid }, + /// The path claims its peer's name for a freshly-established connection, + /// handing the manager the actor's [`ConnHandle`]. The manager monitors + /// `pid` and holds the handle for as long as the entry lives; a + /// [`Registered::Duplicate`] verdict drops the handle here, which stops + /// the refused actor. + Register { + name: String, + pid: Pid, + handle: ConnHandle, + }, + /// Tear down the connection to `name`: the manager drops its handle, the + /// actor stops, and the monitor removes the entry. A no-op if no such + /// connection is live. + Disconnect { name: String }, /// The current peer names, sorted. For observation and tests. Peers, + /// A dialer declares an in-flight dial to `name` before connecting. The + /// pid is the dialing actor, monitored so the intent dies with it. + DialBegin { name: String, pid: Pid }, + /// The dial to `name` resolved (either way): drop the intent. A call, + /// not a cast, so the intent is provably gone before the dialer moves on. + DialEnd { name: String }, + /// The [`HelloCtx`] for an inbound `Hello` offering `peer_name` — the + /// accept path asks this between reading the frame and judging it. + HelloCtx { peer_name: String }, } /// Replies from the manager. #[derive(Debug)] pub enum Reply { Registered(Registered), + Disconnected, Peers(Vec), + /// `false`: another dial to this name is already in flight — do not dial. + DialBegan(bool), + DialEnded, + HelloCtx(HelloCtx), } impl GenServer for Manager { @@ -84,27 +128,58 @@ impl GenServer for Manager { fn handle_call(&mut self, request: Call) -> Reply { match request { - Call::Register { name, pid } => { + Call::Register { name, pid, handle } => { if self.conns.contains_key(&name) { + // `handle` drops here: the refused actor stops itself. return Reply::Registered(Registered::Duplicate); } if let Some(w) = &self.watcher { w.watch(monitor(pid)); } - self.conns.insert(name, pid); + self.conns.insert( + name, + ConnEntry { + pid, + _handle: handle, + }, + ); Reply::Registered(Registered::Ok) } + Call::Disconnect { name } => { + // Dropping the entry drops the handle, which stops the actor. + self.conns.remove(&name); + Reply::Disconnected + } Call::Peers => { let mut names: Vec = self.conns.keys().cloned().collect(); names.sort(); Reply::Peers(names) } + Call::DialBegin { name, pid } => { + if self.dials.contains_key(&name) { + return Reply::DialBegan(false); + } + if let Some(w) = &self.watcher { + w.watch(monitor(pid)); + } + self.dials.insert(name, pid); + Reply::DialBegan(true) + } + Call::DialEnd { name } => { + self.dials.remove(&name); + Reply::DialEnded + } + Call::HelloCtx { peer_name } => Reply::HelloCtx(HelloCtx { + name_claimed: self.conns.contains_key(&peer_name), + dialing_this_peer: self.dials.contains_key(&peer_name), + }), } } fn handle_cast(&mut self, _request: ()) {} fn handle_down(&mut self, down: Down) { - self.conns.retain(|_, pid| *pid != down.pid); + self.conns.retain(|_, entry| entry.pid != down.pid); + self.dials.retain(|_, pid| *pid != down.pid); } } diff --git a/src/cluster/transport.rs b/src/cluster/transport.rs index d62d786..5564c47 100644 --- a/src/cluster/transport.rs +++ b/src/cluster/transport.rs @@ -71,6 +71,15 @@ pub trait Listener: Send { /// The concrete bound address, dialable as-is (e.g. the real port when /// bound with port 0). fn local_addr(&self) -> String; + + /// Readiness as a [`select`](crate::select) arm, mirroring + /// [`Conn::readable_arm`]: `Some` lets an acceptor wait on "an inbound + /// connection is pending" alongside a command inbox in one `select`, so + /// it can be told to stop without a poll loop. Default `None` (the + /// loopback listener has no fd and must be driven synchronously). + fn readable_arm(&self) -> Option { + None + } } /// A way of establishing control connections. Object-safe on purpose: the @@ -128,6 +137,10 @@ pub enum RecvError { TruncatedByPeer, /// The transport failed mid-read. Io(io::Error), + /// The deadline passed before a full frame arrived + /// ([`FramedConn::recv_deadline`] only; plain [`recv`](FramedConn::recv) + /// never returns this). + TimedOut, } impl std::fmt::Display for RecvError { @@ -136,6 +149,7 @@ impl std::fmt::Display for RecvError { RecvError::Corrupt(e) => write!(f, "frame stream corrupt: {e:?}"), RecvError::TruncatedByPeer => write!(f, "peer closed mid-frame"), RecvError::Io(e) => write!(f, "transport read failed: {e}"), + RecvError::TimedOut => write!(f, "deadline passed mid-receive"), } } } @@ -193,6 +207,49 @@ impl FramedConn { } } + /// Like [`recv`](FramedConn::recv), but gives up with + /// [`RecvError::TimedOut`] once `deadline` passes without a full frame. + /// The deadline is enforced between reads via the connection's fd arm + /// (so the caller must be an actor); a transport with no fd (loopback) + /// cannot be timed out and this degrades to a plain blocking `recv` — + /// the same caveat as liveness. + pub fn recv_deadline( + &mut self, + deadline: std::time::Instant, + ) -> Result, RecvError> { + loop { + match Frame::decode(&self.rbuf) { + Ok(Some((frame, consumed))) => { + self.rbuf.drain(..consumed); + return Ok(Some(frame)); + } + Ok(None) => {} + Err(e) => return Err(RecvError::Corrupt(e)), + } + if let Some(arm) = self.conn.readable_arm() { + let left = deadline.saturating_duration_since(std::time::Instant::now()); + if left.is_zero() { + return Err(RecvError::TimedOut); + } + match crate::channel::try_select_timeout(&[&arm], left) { + Ok(Some(_)) => {} + Ok(None) => return Err(RecvError::TimedOut), + Err(e) => return Err(RecvError::Io(e)), + } + } + let mut chunk = [0u8; READ_CHUNK]; + let n = self.conn.read(&mut chunk).map_err(RecvError::Io)?; + if n == 0 { + return if self.rbuf.is_empty() { + Ok(None) + } else { + Err(RecvError::TruncatedByPeer) + }; + } + self.rbuf.extend_from_slice(&chunk[..n]); + } + } + /// Close the underlying connection (idempotent, see [`Conn::close`]). pub fn close(&mut self) { self.conn.close(); diff --git a/src/cluster/transport/tcp.rs b/src/cluster/transport/tcp.rs index d3764e6..2fdc421 100644 --- a/src/cluster/transport/tcp.rs +++ b/src/cluster/transport/tcp.rs @@ -197,6 +197,10 @@ pub struct TcpListener { } impl Listener for TcpListener { + fn readable_arm(&self) -> Option { + Some(crate::scheduler::FdArm::readable(self.inner.as_raw_fd())) + } + fn accept(&mut self) -> io::Result> { loop { wait_readable(self.inner.as_raw_fd())?; diff --git a/tests/cluster_conn_lifecycle.rs b/tests/cluster_conn_lifecycle.rs index 040afbc..eff1b93 100644 --- a/tests/cluster_conn_lifecycle.rs +++ b/tests/cluster_conn_lifecycle.rs @@ -2,11 +2,12 @@ //! //! The handshake is bypassed here (c6b wires it): each connection is //! constructed already-established over a real localhost TCP pair, handed a -//! fabricated `Peer`, and spawned. The actor registers with the manager, which -//! monitors it, so the table reflects the connection while it lives and reaps -//! it on any exit path. This proves three things at once: a live connection -//! shows up, a commanded shutdown removes exactly that one, and a peer close -//! (EOF, no command) removes the other. +//! fabricated `Peer`, and spawned. `spawn_established` registers it with the +//! manager, which takes its handle and monitors it, so the table reflects the +//! connection while it lives and reaps it on any exit path. This proves three +//! things at once: a live connection shows up, a commanded `Disconnect` +//! removes exactly that one, and a peer close (EOF, no command) removes the +//! other. //! //! TCP parks the calling actor, so everything runs inside `smarm::run`; the //! single-threaded runtime is fine because every wait is a cooperative fd park. @@ -80,14 +81,23 @@ fn connection_up_commanded_shutdown_and_eof_all_reflected_in_table() { // Manage the `a` ends as peers node-b and node-c; keep the `b` far ends // open so neither socket is closed from the far side yet. - let h1 = spawn_established(FramedConn::new(a1), peer("node-b")); - let _h2 = spawn_established(FramedConn::new(a2), peer("node-c")); + spawn_established(FramedConn::new(a1), peer("node-b")).expect("node-b registers"); + spawn_established(FramedConn::new(a2), peer("node-c")).expect("node-c registers"); // Up: both connections register and the table shows them. wait_peers(&["node-b", "node-c"]); - // A commanded shutdown reaps exactly its own connection. - h1.shutdown(); + // A commanded disconnect reaps exactly its own connection: the + // manager drops that entry's handle and the actor stops. + assert!(matches!( + gen_server::call( + MANAGER, + Call::Disconnect { + name: "node-b".to_string() + } + ), + Ok(Reply::Disconnected) + )); wait_peers(&["node-c"]); // A peer close (EOF) reaps the other with no command at all. @@ -95,7 +105,7 @@ fn connection_up_commanded_shutdown_and_eof_all_reflected_in_table() { wait_peers(&[]); // node-b's far end stayed open until here, so its removal above was the - // shutdown command and not an EOF. + // disconnect command and not an EOF. drop(b1); // All connection actors have exited; stop the manager so `run` returns. diff --git a/tests/cluster_connect.rs b/tests/cluster_connect.rs new file mode 100644 index 0000000..405ddc4 --- /dev/null +++ b/tests/cluster_connect.rs @@ -0,0 +1,472 @@ +//! RFC 010 c6b — the handshake on the accept/connect path. +//! +//! Path-level tests drive [`dial_handshake`]/[`accept_handshake`] over the +//! loopback transport on plain threads (its intended use — synchronous, no +//! runtime). Integration tests run the manager-backed [`dial`] and +//! [`spawn_acceptor`] over real localhost TCP inside `smarm::run`, and the +//! two-node case as subprocesses via the c4 harness. Flake budget: see +//! tests/common/mod.rs. +#![cfg(feature = "cluster")] + +mod common; + +use std::sync::mpsc; +use std::time::{Duration, Instant}; + +use common::{maybe_child, spawn_node, WAIT}; +use smarm::cluster::connect::{ + accept_handshake, dial, dial_handshake, spawn_acceptor, DialError, HandshakeError, + HANDSHAKE_TIMEOUT, +}; +use smarm::cluster::envelope::{Frame, NodeMeta, RejectReason}; +use smarm::cluster::handshake::{HelloCtx, Local}; +use smarm::cluster::manager::{Call, Manager, Reply, MANAGER}; +use smarm::cluster::transport::loopback::LoopbackTransport; +use smarm::cluster::transport::tcp::TcpTransport; +use smarm::cluster::transport::{FramedConn, Transport}; +use smarm::gen_server::{self, GenServerBuilder}; +use smarm::pg::Incarnation; +use smarm::{run, sleep}; + +const ROLES: &[(&str, fn())] = &[ + ("hs_listener", role_hs_listener), + ("hs_dialer", role_hs_dialer), +]; + +const HASH: u64 = 0xC6B0_C6B0_C6B0_C6B0; + +fn local(name: &str) -> Local { + Local { + node_name: name.into(), + incarnation: Incarnation::new(3), + build_hash: HASH, + meta: NodeMeta { + role: "test".into(), + region: "test".into(), + }, + } +} + +/// A loopback conn pair as `FramedConn`s, ready for a threaded handshake. +fn loopback_pair() -> (FramedConn, FramedConn) { + let t = LoopbackTransport::default(); + let mut l = t.listen("hs").unwrap(); + let dialer = FramedConn::new(t.dial("hs").unwrap()); + let accepted = FramedConn::new(l.accept().unwrap()); + (dialer, accepted) +} + +/// Far-future deadline for loopback paths, where it cannot fire anyway. +fn no_deadline() -> Instant { + Instant::now() + Duration::from_secs(3600) +} + +/// Park the node forever: it has announced everything the parent asserts on, +/// and must now hold its connection open until SIGKILLed. +fn park() -> ! { + loop { + sleep(Duration::from_secs(1)); + } +} + +/// Cooperative bounded receive across the closure/actor boundary. A blocking +/// `std::mpsc` wait would park the OS thread and starve the single-threaded +/// scheduler, so every wait inside `run` polls with [`sleep`] instead. +fn poll_recv(rx: &mpsc::Receiver, what: &str) -> T { + let deadline = Instant::now() + WAIT; + loop { + match rx.try_recv() { + Ok(v) => return v, + Err(mpsc::TryRecvError::Empty) => { + assert!(Instant::now() < deadline, "timed out waiting for {what}"); + sleep(Duration::from_millis(1)); + } + Err(mpsc::TryRecvError::Disconnected) => panic!("channel closed waiting for {what}"), + } + } +} + +// --------------------------------------------------------------------------- +// Path level, over loopback on plain threads +// --------------------------------------------------------------------------- + +#[test] +fn loopback_happy_path_establishes_both_ends() { + maybe_child(ROLES); + let (mut dialer, mut accepted) = loopback_pair(); + let responder = std::thread::spawn(move || { + accept_handshake( + &mut accepted, + local("node-b"), + |name| { + assert_eq!(name, "node-a"); + HelloCtx::default() + }, + no_deadline(), + ) + }); + let peer_of_dialer = dial_handshake(&mut dialer, &local("node-a"), no_deadline()).unwrap(); + let peer_of_acceptor = responder.join().unwrap().unwrap(); + assert_eq!(peer_of_dialer.node_name, "node-b"); + assert_eq!(peer_of_acceptor.node_name, "node-a"); +} + +#[test] +fn loopback_hash_mismatch_rejected_with_frame_then_eof() { + maybe_child(ROLES); + let (mut dialer, mut accepted) = loopback_pair(); + let mut wrong = local("node-b"); + wrong.build_hash ^= 1; + let responder = std::thread::spawn(move || { + accept_handshake(&mut accepted, wrong, |_| HelloCtx::default(), no_deadline()) + }); + // The dial side receives the reject frame — the compatibility anchor. + match dial_handshake(&mut dialer, &local("node-a"), no_deadline()) { + Err(HandshakeError::Rejected(RejectReason::HashMismatch)) => {} + other => panic!("expected HashMismatch reject, got {other:?}"), + } + match responder.join().unwrap() { + Err(HandshakeError::Rejected(RejectReason::HashMismatch)) => {} + other => panic!("expected accept side to report the reject, got {other:?}"), + } +} + +#[test] +fn loopback_tie_break_loser_closed_silently() { + maybe_child(ROLES); + // The inbound dial is from "node-z"; we are "node-a" with our own dial to + // node-z in flight. dial_wins("node-z", "node-a") is false, so the + // inbound loses: closed with no frame at all. + let (mut dialer, mut accepted) = loopback_pair(); + let responder = std::thread::spawn(move || { + accept_handshake( + &mut accepted, + local("node-a"), + |_| HelloCtx { + name_claimed: false, + dialing_this_peer: true, + }, + no_deadline(), + ) + }); + // Silent close: the dial side sees EOF, never a frame. + match dial_handshake(&mut dialer, &local("node-z"), no_deadline()) { + Err(HandshakeError::Closed) => {} + other => panic!("expected silent close (Closed), got {other:?}"), + } + match responder.join().unwrap() { + Err(HandshakeError::TieBreakLoss) => {} + other => panic!("expected TieBreakLoss on the accept side, got {other:?}"), + } +} + +#[test] +fn loopback_read_ahead_past_hello_survives_into_established_conn() { + maybe_child(ROLES); + // The buffer trap, proven: the dialer coalesces Hello + Heartbeat before + // the responder's first read, so the Heartbeat lands in the shared + // FramedConn's decode buffer during the handshake. The dialer sends + // nothing afterwards — the post-handshake recv can only succeed if the + // read-ahead travelled with the FramedConn. + let (mut dialer, mut accepted) = loopback_pair(); + let (_init, hello) = smarm::cluster::handshake::Initiator::new(&local("node-a")); + dialer.send(&hello).unwrap(); + dialer.send(&Frame::Heartbeat).unwrap(); + // Both frames are buffered before the responder reads at all. + let (tx, rx) = mpsc::channel(); + std::thread::spawn(move || { + let peer = accept_handshake( + &mut accepted, + local("node-b"), + |_| HelloCtx::default(), + no_deadline(), + ) + .unwrap(); + let next = accepted.recv(); + let _ = tx.send((peer, next)); + }); + // A bounded wait: if the Heartbeat were NOT carried in the buffer, the + // recv above would block forever (the dialer stays open and silent). + let (peer, next) = rx + .recv_timeout(Duration::from_secs(5)) + .expect("read-ahead lost: post-handshake recv blocked"); + assert_eq!(peer.node_name, "node-a"); + match next { + Ok(Some(Frame::Heartbeat)) => {} + other => panic!("expected the read-ahead Heartbeat, got {other:?}"), + } + drop(dialer); +} + +// --------------------------------------------------------------------------- +// Deadline + manager integration, over TCP inside the runtime +// --------------------------------------------------------------------------- + +#[test] +fn tcp_silent_peer_times_out_on_the_accept_path() { + maybe_child(ROLES); + run(|| { + let t = TcpTransport; + let mut l = t.listen("127.0.0.1:0").unwrap(); + // Connect and then say nothing at all. + let silent = t.dial(&l.local_addr()).unwrap(); + let mut accepted = FramedConn::new(l.accept().unwrap()); + let (tx, rx) = mpsc::channel(); + smarm::spawn(move || { + let r = accept_handshake( + &mut accepted, + local("node-b"), + |_| HelloCtx::default(), + Instant::now() + Duration::from_millis(200), + ); + let _ = tx.send(r); + }); + match poll_recv(&rx, "accept-path outcome") { + Err(HandshakeError::TimedOut) => {} + other => panic!("expected TimedOut, got {other:?}"), + } + drop(silent); + }); +} + +/// Poll the manager until its peer set matches `expected` (sorted), or fail. +fn wait_peers(expected: &[&str]) { + let want: Vec = expected.iter().map(|s| s.to_string()).collect(); + for _ in 0..5000 { + if let Ok(Reply::Peers(got)) = gen_server::call(MANAGER, Call::Peers) { + if got == want { + return; + } + } + sleep(Duration::from_millis(1)); + } + let got = gen_server::call(MANAGER, Call::Peers); + panic!("timed out waiting for peers == {want:?}; last = {got:?}"); +} + +#[test] +fn tcp_duplicate_name_rejected_by_acceptor() { + maybe_child(ROLES); + run(|| { + let mgr = GenServerBuilder::new(Manager::new()) + .named(MANAGER) + .start() + .expect("manager name is free"); + let listener = TcpTransport.listen("127.0.0.1:0").unwrap(); + let acceptor = spawn_acceptor(listener, local("node-b")); + let addr = acceptor.local_addr().to_string(); + + // First dial offering "dup-node": establishes and registers. + let mut first = FramedConn::new(TcpTransport.dial(&addr).unwrap()); + let peer = dial_handshake( + &mut first, + &local("dup-node"), + Instant::now() + HANDSHAKE_TIMEOUT, + ) + .unwrap(); + assert_eq!(peer.node_name, "node-b"); + wait_peers(&["dup-node"]); + + // Second dial offering the same name: deterministic NameTaken. + let mut second = FramedConn::new(TcpTransport.dial(&addr).unwrap()); + match dial_handshake( + &mut second, + &local("dup-node"), + Instant::now() + HANDSHAKE_TIMEOUT, + ) { + Err(HandshakeError::Rejected(RejectReason::NameTaken)) => {} + other => panic!("expected NameTaken, got {other:?}"), + } + // The established connection was untouched by the rejected one. + wait_peers(&["dup-node"]); + + // Teardown: the acceptor owns no connections, so the established one + // is torn down through the table. + acceptor.shutdown(); + assert!(matches!( + gen_server::call( + MANAGER, + Call::Disconnect { + name: "dup-node".to_string() + } + ), + Ok(Reply::Disconnected) + )); + wait_peers(&[]); + first.close(); + mgr.shutdown(); + }); +} + +#[test] +fn dial_intent_cleared_when_dialer_dies() { + maybe_child(ROLES); + run(|| { + let mgr = GenServerBuilder::new(Manager::new()) + .named(MANAGER) + .start() + .expect("manager name is free"); + let (begun_tx, begun_rx) = mpsc::channel(); + let (go_tx, go_rx) = mpsc::channel::<()>(); + smarm::spawn(move || { + let me = smarm::self_pid(); + match gen_server::call( + MANAGER, + Call::DialBegin { + name: "ghost".into(), + pid: me, + }, + ) { + Ok(Reply::DialBegan(true)) => {} + other => panic!("DialBegin failed: {other:?}"), + } + let _ = begun_tx.send(()); + let () = poll_recv(&go_rx, "go signal"); + panic!("dialer dies mid-dial"); + }); + poll_recv(&begun_rx, "DialBegin done"); + // While the dialer lives, the intent is visible. + match gen_server::call( + MANAGER, + Call::HelloCtx { + peer_name: "ghost".into(), + }, + ) { + Ok(Reply::HelloCtx(ctx)) => assert!(ctx.dialing_this_peer), + other => panic!("HelloCtx failed: {other:?}"), + } + // Kill it; the monitor must clear the intent without cooperation. + go_tx.send(()).unwrap(); + let deadline = Instant::now() + WAIT; + loop { + match gen_server::call( + MANAGER, + Call::HelloCtx { + peer_name: "ghost".into(), + }, + ) { + Ok(Reply::HelloCtx(ctx)) if !ctx.dialing_this_peer => break, + _ if Instant::now() > deadline => { + panic!("dial intent not cleared after dialer death") + } + _ => sleep(Duration::from_millis(1)), + } + } + mgr.shutdown(); + }); +} + +// --------------------------------------------------------------------------- +// Two nodes, two processes: the integrated dial against a real acceptor +// --------------------------------------------------------------------------- + +/// Announce, then park forever. Neither role ever tears its connection +/// down: a table entry only exists while the *peer* holds its side open, so +/// any teardown here would retract the other node's observation before it +/// had made it. The parent reaps both with SIGKILL once it has both +/// announcements (see [`common::Node`]'s `Drop`). +fn role_hs_listener() { + run(|| { + let _mgr = GenServerBuilder::new(Manager::new()) + .named(MANAGER) + .start() + .expect("manager name is free"); + let listener = TcpTransport.listen("127.0.0.1:0").unwrap(); + let acceptor = spawn_acceptor(listener, local("node-b")); + println!("LISTENING {}", acceptor.local_addr()); + wait_peers(&["node-a"]); + println!("PEERS node-a"); + park(); + }); +} + +fn role_hs_dialer() { + let addr = std::env::var("SMARM_PEER_ADDR").expect("SMARM_PEER_ADDR not set"); + run(move || { + let _mgr = GenServerBuilder::new(Manager::new()) + .named(MANAGER) + .start() + .expect("manager name is free"); + let (tx, rx) = mpsc::channel(); + smarm::spawn(move || { + let r = dial(&TcpTransport, &addr, "node-b", &local("node-a")); + let _ = tx.send(r); + }); + if let Err(e) = poll_recv(&rx, "dial outcome") { + println!("DIAL failed: {e:?}"); + std::process::exit(3); + } + wait_peers(&["node-b"]); + println!("PEERS node-b"); + park(); + }); +} + +#[test] +fn two_node_integrated_handshake_over_tcp() { + maybe_child(ROLES); + let mut listener = spawn_node("hs_listener", &[]); + let addr = listener.wait_listening(); + let mut dialer = spawn_node("hs_dialer", &[("SMARM_PEER_ADDR", &addr)]); + // Each node reports its own table naming the other: a real dial against a + // real acceptor established in both directions. Both nodes then park — + // clean-exit behaviour is the c4 harness's own smoke test, and demanding + // it here would mean a teardown, which is exactly what cannot be ordered + // safely across two processes. Dropping the nodes SIGKILLs them. + dialer.wait_line("PEERS node-b", |l| l == "PEERS node-b"); + listener.wait_line("PEERS node-a", |l| l == "PEERS node-a"); +} + +// --------------------------------------------------------------------------- +// Integrated-dial guardrails (no acceptor involved) +// --------------------------------------------------------------------------- + +#[test] +fn concurrent_dial_to_same_name_refused() { + maybe_child(ROLES); + run(|| { + let mgr = GenServerBuilder::new(Manager::new()) + .named(MANAGER) + .start() + .expect("manager name is free"); + let (begun_tx, begun_rx) = mpsc::channel(); + let (go_tx, go_rx) = mpsc::channel::<()>(); + // First dialer parks with the intent held (it never connects — + // 'holding the intent' is all this test needs from it). + smarm::spawn(move || { + let me = smarm::self_pid(); + assert!(matches!( + gen_server::call( + MANAGER, + Call::DialBegin { + name: "node-x".into(), + pid: me, + } + ), + Ok(Reply::DialBegan(true)) + )); + let _ = begun_tx.send(()); + let () = poll_recv(&go_rx, "go signal"); + let _ = gen_server::call( + MANAGER, + Call::DialEnd { + name: "node-x".into(), + }, + ); + }); + poll_recv(&begun_rx, "DialBegin done"); + // Second integrated dial to the same name: refused before connecting + // (the addr is unroutable on purpose — it must never be dialed). + let (tx, rx) = mpsc::channel(); + smarm::spawn(move || { + let r = dial(&TcpTransport, "127.0.0.1:1", "node-x", &local("node-a")); + let _ = tx.send(r); + }); + match poll_recv(&rx, "second dial outcome") { + Err(DialError::AlreadyDialing) => {} + other => panic!("expected AlreadyDialing, got {other:?}"), + } + go_tx.send(()).unwrap(); + mgr.shutdown(); + }); +} From ad4958421f8b6b12dc2996ab36ed681b2003f42f Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 15 Aug 2026 06:36:07 +0000 Subject: [PATCH 08/14] =?UTF-8?q?feat(cluster):=20RFC=20010=20c6c=20?= =?UTF-8?q?=E2=80=94=20heartbeat=20send=20+=20fixed-timeout=20liveness=20+?= =?UTF-8?q?=20teardown?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The timeout arm of the connection actor's select: HEARTBEAT_INTERVAL (1s) paces outbound Frame::Heartbeat (first at spawn, so the peer's window starts fed) and LIVENESS_TIMEOUT (4s = 4 intervals) declares the peer dead when no inbound frame arrives inside it — any frame resets the window, so heartbeats keep an idle connection alive and real traffic (c8+) counts for free. Fire => close + exit; the manager's monitor reaps the table entry as on every other exit path. Fixed timeout per RFC v2 §5 (control connection, heartbeats can't queue behind bulk). Intervals are the one-viable-answer call flagged for veto at diff review. The pump was made non-blocking to keep the deadlines honest: a plain recv() blocks into the socket while the buffer holds a partial frame, parking the actor past its timers. Two additive FramedConn methods (read_once, next_buffered): exactly one socket read per level-triggered readable wake (cannot block, cannot strand — leftovers re-signal), then drain every complete buffered frame. Liveness resets only on complete frames. No-fd transports (loopback) still get the command-only loop: no readiness means no timers, same caveat as recv_deadline. tests/cluster_conn_liveness.rs 3/0, stable over 5 runs (raw far end over localhost TCP: heartbeats appear unprompted; mute peer still up at half the window, gone after it; heartbeat-only peer survives 1.5x the window, then reaped once silenced). Cluster suites regression-clean; clippy --lib green both configs; fmt clean. --- src/cluster/conn.rs | 146 ++++++++++++++++++++------- src/cluster/transport.rs | 30 ++++++ tests/cluster_conn_liveness.rs | 178 +++++++++++++++++++++++++++++++++ 3 files changed, 318 insertions(+), 36 deletions(-) create mode 100644 tests/cluster_conn_liveness.rs diff --git a/src/cluster/conn.rs b/src/cluster/conn.rs index 43d99cb..3eee6c5 100644 --- a/src/cluster/conn.rs +++ b/src/cluster/conn.rs @@ -15,11 +15,17 @@ //! ownership of its [`ConnHandle`] and monitors the actor, so any exit //! deregisters the connection. The actor itself holds no authority over its //! own lifetime: it runs until the manager drops its handle (deregistration, -//! `Disconnect`, or manager shutdown) or the connection ends. Heartbeat send -//! and fixed-timeout liveness join the loop in c6c (the timeout arm of the -//! same `select`). +//! `Disconnect`, or manager shutdown), the connection ends, or liveness +//! expires. Heartbeat send and fixed-timeout liveness are the timeout arm of +//! the same `select` (c6c): [`HEARTBEAT_INTERVAL`] paces outbound +//! [`Frame::Heartbeat`](crate::cluster::envelope::Frame::Heartbeat)s, and a +//! [`LIVENESS_TIMEOUT`] window — reset by any inbound frame — tears the +//! connection down when it empties. -use crate::channel::{channel, Receiver, Selectable, Sender}; +use std::time::{Duration, Instant}; + +use crate::channel::{channel, try_select_timeout, Receiver, Selectable, Sender}; +use crate::cluster::envelope::Frame; use crate::cluster::handshake::Peer; use crate::cluster::manager::{Call, Registered, Reply, MANAGER}; use crate::cluster::transport::FramedConn; @@ -88,39 +94,81 @@ pub fn spawn_established(framed: FramedConn, peer: Peer) -> Result) { + match framed.readable_arm() { + Some(arm) => run_live(&mut framed, arm, &cmd_rx), + None => run_inert(&cmd_rx), + } + framed.close(); +} + +/// The steady-state loop over an fd-backed connection: one +/// `select_timeout` folds the command inbox, socket readability, and the +/// nearer of the two deadlines (`hb_send`, `liveness`) into a single wait. +fn run_live(framed: &mut FramedConn, arm: crate::scheduler::FdArm, cmd_rx: &Receiver) { + let mut next_hb = Instant::now(); + let mut live_until = Instant::now() + LIVENESS_TIMEOUT; loop { - match framed.readable_arm() { - Some(arm) => { - let arms: [&dyn Selectable; 2] = [&cmd_rx, &arm]; - match crate::channel::try_select(&arms) { - Ok(0) => { - if should_stop(&cmd_rx) { - break; - } - } - Ok(_) => { - if pump_readable(&mut framed) { - break; - } - } - // The fd arm failed to register — the connection is gone. - Err(_) => break, - } + let now = Instant::now(); + if now >= live_until { + break; // liveness expired: the peer is dead to us + } + if now >= next_hb { + if framed.send(&Frame::Heartbeat).is_err() { + break; } - None => { - // No fd to select on (loopback): only a command can end the - // wait. Liveness over such a transport is out of scope. - let arms: [&dyn Selectable; 1] = [&cmd_rx]; - let _ = crate::channel::select(&arms); - if should_stop(&cmd_rx) { + next_hb = now + HEARTBEAT_INTERVAL; + } + let wait = next_hb.min(live_until).saturating_duration_since(now); + let arms: [&dyn Selectable; 2] = [cmd_rx, &arm]; + match try_select_timeout(&arms, wait) { + Ok(Some(0)) => { + if should_stop(cmd_rx) { break; } } + Ok(Some(_)) => match pump_readable(framed) { + Pump::Ended => break, + Pump::Frames(n) => { + if n > 0 { + live_until = Instant::now() + LIVENESS_TIMEOUT; + } + } + }, + // A deadline passed; the top of the loop acts on whichever. + Ok(None) => {} + // The fd arm failed to register — the connection is gone. + Err(_) => break, } } +} - framed.close(); +/// No fd to select on (loopback): only a command can end the wait, and +/// neither heartbeats nor liveness run — a transport that can't report +/// readiness can't be timed either (same caveat as +/// [`FramedConn::recv_deadline`]). Loopback is a test transport; every real +/// connection is fd-backed. +fn run_inert(cmd_rx: &Receiver) { + loop { + let arms: [&dyn Selectable; 1] = [cmd_rx]; + let _ = crate::channel::select(&arms); + if should_stop(cmd_rx) { + break; + } + } } /// Drain the command arm. Returns `true` when the actor should exit — a @@ -133,13 +181,39 @@ fn should_stop(cmd_rx: &Receiver) -> bool { } } -/// Consume whatever is readable now. Returns `true` when the connection has -/// ended (clean EOF or an unrecoverable stream error). This chunk does not -/// interpret frames; c6c handles heartbeats and resets the liveness timer here. -fn pump_readable(framed: &mut FramedConn) -> bool { - match framed.recv() { - Ok(Some(_frame)) => false, - Ok(None) => true, // clean EOF at a frame boundary - Err(_) => true, // corrupt / truncated / io +/// What one readable wake yielded. +enum Pump { + /// The connection has ended: EOF (clean or mid-frame) or an + /// unrecoverable stream error. + Ended, + /// Still up; this many complete frames were consumed (possibly zero, if + /// the wake delivered only part of a frame). Any nonzero count resets + /// the liveness window. + Frames(usize), +} + +/// Consume one readable wake: exactly one socket read (which cannot block +/// after a level-triggered readable indication), then drain every complete +/// frame the buffer now holds. A blocking `recv` here would park the actor +/// past its heartbeat and liveness deadlines whenever a frame arrives split. +/// Frames are not interpreted yet — a heartbeat's entire job is the liveness +/// reset, and everything else waits for c8. +fn pump_readable(framed: &mut FramedConn) -> Pump { + let eof = match framed.read_once() { + Ok(n) => n == 0, + Err(_) => return Pump::Ended, + }; + let mut got = 0; + loop { + match framed.next_buffered() { + Ok(Some(_frame)) => got += 1, + Ok(None) => break, + Err(_) => return Pump::Ended, // corrupt stream + } + } + if eof { + Pump::Ended + } else { + Pump::Frames(got) } } diff --git a/src/cluster/transport.rs b/src/cluster/transport.rs index 5564c47..8f0e9fc 100644 --- a/src/cluster/transport.rs +++ b/src/cluster/transport.rs @@ -250,6 +250,36 @@ impl FramedConn { } } + /// One socket read, appended to the reassembly buffer. Returns the byte + /// count (`0` = EOF). For select-loop callers that were just told the fd + /// is readable: under the level-triggered IO thread exactly one read per + /// readable wake never blocks and never loses data — leftover socket + /// bytes re-signal on the next select, and complete frames already + /// reassembled are drained with [`next_buffered`](FramedConn::next_buffered). + /// (A plain [`recv`](FramedConn::recv) can block into the socket while + /// the buffer holds a partial frame, which a loop with deadlines to keep + /// cannot afford.) + pub fn read_once(&mut self) -> std::io::Result { + let mut chunk = [0u8; READ_CHUNK]; + let n = self.conn.read(&mut chunk)?; + self.rbuf.extend_from_slice(&chunk[..n]); + Ok(n) + } + + /// Decode the next complete frame already sitting in the reassembly + /// buffer, without touching the socket. `Ok(None)` means the buffer + /// holds no complete frame (empty, or a partial awaiting more bytes). + pub fn next_buffered(&mut self) -> Result, DecodeError> { + match Frame::decode(&self.rbuf) { + Ok(Some((frame, consumed))) => { + self.rbuf.drain(..consumed); + Ok(Some(frame)) + } + Ok(None) => Ok(None), + Err(e) => Err(e), + } + } + /// Close the underlying connection (idempotent, see [`Conn::close`]). pub fn close(&mut self) { self.conn.close(); diff --git a/tests/cluster_conn_liveness.rs b/tests/cluster_conn_liveness.rs new file mode 100644 index 0000000..ba2aea0 --- /dev/null +++ b/tests/cluster_conn_liveness.rs @@ -0,0 +1,178 @@ +//! RFC 010 c6c — heartbeat send + fixed-timeout liveness + teardown. +//! +//! Each case runs one real connection actor over an in-process localhost TCP +//! pair, with the far end held as a raw `FramedConn` (no actor) so the test +//! controls exactly what — if anything — the peer says. That gives the three +//! protocol-visible facts direct handles: heartbeats appear on the wire +//! unprompted; a mute peer is torn down (and reaped from the manager table) +//! once `LIVENESS_TIMEOUT` empties; and a peer that does nothing but send +//! heartbeats keeps the connection alive past that same window. +//! +//! Loopback has no fd and cannot drive liveness (documented on the actor), +//! so everything here is TCP. TCP parks the calling actor, so everything +//! runs inside `smarm::run`. +#![cfg(feature = "cluster")] + +use std::time::{Duration, Instant}; + +use smarm::cluster::conn::{HEARTBEAT_INTERVAL, LIVENESS_TIMEOUT}; +use smarm::cluster::envelope::{Frame, NodeMeta}; +use smarm::cluster::handshake::Peer; +use smarm::cluster::manager::{Call, Manager, Reply, MANAGER}; +use smarm::cluster::spawn_established; +use smarm::cluster::transport::tcp::TcpTransport; +use smarm::cluster::transport::{Conn, FramedConn, Transport}; +use smarm::gen_server::{self, GenServerBuilder}; +use smarm::pg::Incarnation; +use smarm::{run, sleep, spawn}; + +/// A fabricated post-handshake peer identity (same shape as the c6a suite). +fn peer(name: &str) -> Peer { + Peer { + node_name: name.to_string(), + incarnation: Incarnation::new(1), + meta: NodeMeta { + role: "test".to_string(), + region: "test".to_string(), + }, + } +} + +/// One established transport pair over localhost (TCP backlog covers the +/// sequential dial-then-accept, as in the c3 conformance suite). +fn pair(t: &dyn Transport) -> (Box, Box) { + let mut l = t.listen("127.0.0.1:0").unwrap(); + let a = t.dial(&l.local_addr()).unwrap(); + let b = l.accept().unwrap(); + (a, b) +} + +fn peers() -> Vec { + match gen_server::call(MANAGER, Call::Peers) { + Ok(Reply::Peers(p)) => p, + other => panic!("manager unreachable: {other:?}"), + } +} + +/// Poll until the manager's peer set matches `expected` (sorted) or `budget` +/// runs out. +fn wait_peers(expected: &[&str], budget: Duration) { + let want: Vec = expected.iter().map(|s| s.to_string()).collect(); + let deadline = Instant::now() + budget; + while Instant::now() < deadline { + if peers() == want { + return; + } + sleep(Duration::from_millis(10)); + } + panic!( + "timed out waiting for peers == {want:?}; last = {:?}", + peers() + ); +} + +/// The actor emits heartbeats unprompted: the raw far end, saying nothing, +/// sees a `Frame::Heartbeat` well within one interval (the first goes out at +/// spawn). +#[test] +fn heartbeats_are_sent_unprompted() { + run(|| { + let mgr = GenServerBuilder::new(Manager::new()) + .named(MANAGER) + .start() + .expect("manager name is free"); + + let (a, b) = pair(&TcpTransport); + spawn_established(FramedConn::new(a), peer("hb-send")).expect("register"); + let mut far = FramedConn::new(b); + + let frame = far + .recv_deadline(Instant::now() + HEARTBEAT_INTERVAL) + .expect("a heartbeat before one interval elapses"); + assert_eq!(frame, Some(Frame::Heartbeat)); + + // Teardown: closing the far end is an EOF at the actor. + far.close(); + wait_peers(&[], Duration::from_secs(2)); + mgr.shutdown(); + }); +} + +/// A mute peer is dead: no inbound frame for `LIVENESS_TIMEOUT` tears the +/// connection down and the manager's monitor reaps the table entry. The +/// entry is still present well inside the window — the teardown is the +/// timer, not an accident of setup. +#[test] +fn mute_peer_is_torn_down_after_liveness_timeout() { + run(|| { + let mgr = GenServerBuilder::new(Manager::new()) + .named(MANAGER) + .start() + .expect("manager name is free"); + + let (a, b) = pair(&TcpTransport); + spawn_established(FramedConn::new(a), peer("mute")).expect("register"); + // Held open and silent: no frames, no EOF. (Unread inbound + // heartbeats sit in kernel buffers; they are 5 bytes each.) + let _far = FramedConn::new(b); + + // Well inside the window the connection is still up. + sleep(LIVENESS_TIMEOUT / 2); + assert_eq!(peers(), vec!["mute".to_string()], "torn down too early"); + + // ...and once the window empties it is gone. Generous budget over + // the remaining half-window. + wait_peers(&[], LIVENESS_TIMEOUT); + mgr.shutdown(); + }); +} + +/// Heartbeats alone keep a connection alive past `LIVENESS_TIMEOUT`: a far +/// end that sends `Frame::Heartbeat` at the interval (and nothing else) +/// holds the entry; when it goes quiet, liveness finally fires. +#[test] +fn heartbeats_keep_the_connection_alive() { + run(|| { + let mgr = GenServerBuilder::new(Manager::new()) + .named(MANAGER) + .start() + .expect("manager name is free"); + + let (a, b) = pair(&TcpTransport); + spawn_established(FramedConn::new(a), peer("kept")).expect("register"); + + // The far heartbeat pump: interval-paced sends until told to stop, + // then holds the socket open, silent, so the eventual teardown is + // liveness — not EOF. + let (ctl_tx, ctl_rx) = smarm::channel::channel::<()>(); + spawn(move || { + let mut far = FramedConn::new(b); + // Phase 1: heartbeat at the interval until the first signal. + while matches!(ctl_rx.try_recv(), Ok(None)) { + far.send(&Frame::Heartbeat).expect("far send"); + sleep(HEARTBEAT_INTERVAL); + } + // Phase 2: silent but with the socket held open — dropping + // `far` here would EOF the actor and mask the liveness path. + // Exits when the test's closure ends and drops `ctl_tx` (an + // eternal park would stop `run` from ever returning). + while matches!(ctl_rx.try_recv(), Ok(None)) { + sleep(Duration::from_millis(20)); + } + }); + + // Past the liveness window with margin: still up. + sleep(LIVENESS_TIMEOUT + LIVENESS_TIMEOUT / 2); + assert_eq!( + peers(), + vec!["kept".to_string()], + "liveness fired despite heartbeats" + ); + + // Silence the pump; liveness now empties and the entry goes. + ctl_tx.send(()).expect("pump alive"); + wait_peers(&[], LIVENESS_TIMEOUT * 2); + mgr.shutdown(); + // `ctl_tx` drops here, releasing the pump's phase-2 wait. + }); +} From 112d6b2e655f8478e68c3365bcc2c2c8d77e2fcc Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 15 Aug 2026 06:37:57 +0000 Subject: [PATCH 09/14] =?UTF-8?q?feat(cluster):=20RFC=20010=20c6d=20?= =?UTF-8?q?=E2=80=94=20build=5Fhash=20derivation?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit cluster::BUILD_HASH, the derived value for LocalNode::build_hash: a compile-time const, FNV-1a 64 over the build script's input string (rustc -V + sorted enabled CARGO_FEATURE_* set) with PROTO_VERSION folded in as a continuation of the same state — a proto bump moves the hash even on an identical toolchain. build.rs emits only the raw inputs (SMARM_BUILD_HASH_INPUTS); the hashing lives in src/cluster.rs next to PROTO_VERSION rather than build.rs parsing it out of a source file. The env var is emitted unconditionally — one string costs the default build nothing, and the const itself is behind the cluster feature with the rest of the module. Domain is the flagged lean (one-viable-answer, veto at diff review): toolchain + declared features + proto version. Tightenable later without a wire change — it is just a u64 on the Hello. Unit tests: FNV-1a 64 published vectors (empty/'a'/'foobar'); the proto fold is a continuation of the same FNV state and moves the hash; BUILD_HASH is const-evaluable and nonzero. All cluster suites regression-clean (7 files); clippy --lib green both configs; fmt clean; default build compiles. --- build.rs | 25 ++++++++++++++++++ src/cluster.rs | 71 ++++++++++++++++++++++++++++++++++++++++++++++++++ 2 files changed, 96 insertions(+) diff --git a/build.rs b/build.rs index f00ab7f..1e200ed 100644 --- a/build.rs +++ b/build.rs @@ -8,4 +8,29 @@ fn main() { .flag_if_supported("-fno-stack-clash-protection") .compile("smarm_canary"); println!("cargo:rerun-if-changed=canary/canary.c"); + + // RFC 010 c6d — build_hash inputs. The compile-time facts a peer must + // share for a mesh link: the exact toolchain and the declared (enabled) + // feature set. Emitted as a plain string; the hashing (FNV-1a folded + // with PROTO_VERSION) happens in src/cluster.rs where the protocol + // version actually lives — parsing it out of a source file here would + // be a second, fragile copy. Always emitted, even for non-cluster + // builds: one env var costs the default build nothing. + let rustc = std::env::var("RUSTC").unwrap_or_else(|_| "rustc".to_string()); + let version = std::process::Command::new(&rustc) + .arg("-V") + .output() + .ok() + .map(|o| String::from_utf8_lossy(&o.stdout).trim().to_string()) + .filter(|v| !v.is_empty()) + .unwrap_or_else(|| "rustc-unknown".to_string()); + let mut feats: Vec = std::env::vars() + .filter_map(|(k, _)| k.strip_prefix("CARGO_FEATURE_").map(str::to_string)) + .collect(); + feats.sort(); + println!( + "cargo:rustc-env=SMARM_BUILD_HASH_INPUTS={version};features={}", + feats.join(",") + ); + println!("cargo:rerun-if-env-changed=RUSTC"); } diff --git a/src/cluster.rs b/src/cluster.rs index b3cea4f..aea18d8 100644 --- a/src/cluster.rs +++ b/src/cluster.rs @@ -26,6 +26,43 @@ pub use conn::{spawn_established, ConnHandle}; pub use connect::{dial, spawn_acceptor, AcceptorHandle}; pub use manager::{Manager, MANAGER}; +/// c6d — the derived build hash for [`handshake::LocalNode::build_hash`]: +/// two builds may mesh only when this matches, and it is a pure function of +/// the compile-time inputs that define wire compatibility today — the exact +/// toolchain (`rustc -V`), the declared feature set, and +/// [`envelope::PROTO_VERSION`]. FNV-1a 64 over the build-script string, then +/// the proto version folded byte-wise, so a proto bump moves the hash even +/// on an identical toolchain. The domain is deliberately lean and +/// tightenable later without a wire change — it is just a `u64`. +pub const BUILD_HASH: u64 = fold_u32( + fnv1a64(env!("SMARM_BUILD_HASH_INPUTS").as_bytes()), + envelope::PROTO_VERSION, +); + +/// FNV-1a 64 (const so [`BUILD_HASH`] is a compile-time fact). +const fn fnv1a64(bytes: &[u8]) -> u64 { + let mut h: u64 = 0xcbf2_9ce4_8422_2325; + let mut i = 0; + while i < bytes.len() { + h ^= bytes[i] as u64; + h = h.wrapping_mul(0x0000_0100_0000_01b3); + i += 1; + } + h +} + +/// Continue an FNV-1a state over a `u32`'s little-endian bytes. +const fn fold_u32(mut h: u64, v: u32) -> u64 { + let b = v.to_le_bytes(); + let mut i = 0; + while i < b.len() { + h ^= b[i] as u64; + h = h.wrapping_mul(0x0000_0100_0000_01b3); + i += 1; + } + h +} + /// A running cluster subtree: an explicitly-started supervisor over the /// connection [`Manager`]. Roles will eventually mount this subtree; until the /// role mechanism lands it is started by hand (RFC 010 §7). Dropping the handle @@ -64,3 +101,37 @@ fn manager_child() { }; let _ = monitor(m.pid()).rx.recv(); } + +#[cfg(test)] +mod tests { + use super::*; + + /// The hash core against the published FNV-1a 64 test vectors — the + /// contract is "this is FNV-1a", not "whatever the fn does". + #[test] + fn fnv1a64_known_vectors() { + assert_eq!(fnv1a64(b""), 0xcbf2_9ce4_8422_2325); + assert_eq!(fnv1a64(b"a"), 0xaf63_dc4c_8601_ec8c); + assert_eq!(fnv1a64(b"foobar"), 0x85944171f73967e8); + } + + /// Folding the proto version continues the same FNV state: identical + /// inputs with a different version must land on a different hash. + #[test] + fn proto_version_moves_the_hash() { + let base = fnv1a64(b"same-toolchain;features=CLUSTER"); + assert_ne!(fold_u32(base, 1), fold_u32(base, 2)); + // And it equals hashing the bytes in one pass — the fold is a + // continuation, not a second construction. + let mut all = b"same-toolchain;features=CLUSTER".to_vec(); + all.extend_from_slice(&1u32.to_le_bytes()); + assert_eq!(fold_u32(base, 1), fnv1a64(&all)); + } + + /// The derived constant exists, is compile-time, and is not degenerate. + #[test] + fn build_hash_is_nonzero() { + const H: u64 = BUILD_HASH; + assert_ne!(H, 0); + } +} From 160967939b51bbd742f77d05b1cb63cb9bf46cfa Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 15 Aug 2026 07:07:14 +0000 Subject: [PATCH 10/14] =?UTF-8?q?feat(cluster):=20RFC=20010=20c7a=20?= =?UTF-8?q?=E2=80=94=20membership=20events=20+=20view=20at=20the=20manager?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit node_up/node_down are derived facts of the manager's own register/remove events, so the membership state lives in the manager (no cross-actor race between 'connection exists' and 'node is up'); src/cluster/membership.rs is the consumer surface: NodeEvent/NodeInfo, subscribe(), view(). The conn table stays private — no consumer touches it (roadmap-binding). Ratified semantics: subscribe() is snapshot-then-stream — one NodeUp per live peer is queued before the subscription joins the list, exact because gen_server handlers are serialized. Dropped subscribers are pruned on the next emit (closed channel), no monitor needed. One-viable call, flagged: NodeId is memoized per (name, incarnation) — a compact local alias for the wire identity, per pg.rs's framing. A reconnect blip at the same incarnation keeps its id; a restart (new incarnation) gets a fresh one, so a ghost and its successor are always distinguishable. Allocation starts at 1; NodeId(0) stays pg::DEFAULT_NODE_ID (self). Call::Register now carries the whole handshake Peer (the path already has it; node_up needs incarnation + meta). tests/cluster_membership.rs 4/0 stable over 5 runs (live up/down over localhost TCP with commanded and EOF teardown; late-subscriber snapshot + view agreement; restart-vs-blip id identity; dead-subscriber pruning). All cluster suites regression-clean; clippy --lib green both configs; fmt clean; default build compiles. --- src/cluster.rs | 2 + src/cluster/conn.rs | 4 +- src/cluster/manager.rs | 119 +++++++++++++--- src/cluster/membership.rs | 97 ++++++++++++++ tests/cluster_membership.rs | 261 ++++++++++++++++++++++++++++++++++++ 5 files changed, 465 insertions(+), 18 deletions(-) create mode 100644 src/cluster/membership.rs create mode 100644 tests/cluster_membership.rs diff --git a/src/cluster.rs b/src/cluster.rs index aea18d8..286ed0c 100644 --- a/src/cluster.rs +++ b/src/cluster.rs @@ -13,6 +13,7 @@ pub mod connect; pub mod envelope; pub mod handshake; pub mod manager; +pub mod membership; pub mod transport; use std::time::Duration; @@ -25,6 +26,7 @@ use crate::supervisor::{ChildSpec, OneForOne, Restart}; pub use conn::{spawn_established, ConnHandle}; pub use connect::{dial, spawn_acceptor, AcceptorHandle}; pub use manager::{Manager, MANAGER}; +pub use membership::{subscribe, view, MembershipEvents, NodeEvent, NodeInfo}; /// c6d — the derived build hash for [`handshake::LocalNode::build_hash`]: /// two builds may mesh only when this matches, and it is a pure function of diff --git a/src/cluster/conn.rs b/src/cluster/conn.rs index 3eee6c5..b2953b5 100644 --- a/src/cluster/conn.rs +++ b/src/cluster/conn.rs @@ -76,12 +76,12 @@ pub struct RegisterRefused; /// actor. A refusal has already stopped the actor and closed the socket. pub fn spawn_established(framed: FramedConn, peer: Peer) -> Result { let (cmd_tx, cmd_rx) = channel(); - let name = peer.node_name.clone(); + let reg_peer = peer.clone(); let pid = spawn(move || run(framed, peer, cmd_rx)).pid(); match gen_server::call( MANAGER, Call::Register { - name, + peer: reg_peer, pid, handle: ConnHandle { cmd_tx }, }, diff --git a/src/cluster/manager.rs b/src/cluster/manager.rs index b5153b1..4a40152 100644 --- a/src/cluster/manager.rs +++ b/src/cluster/manager.rs @@ -10,17 +10,25 @@ //! that panics, is cancelled, or closes cleanly is removed without //! cooperation from the dying actor. //! -//! Membership as consumers will see it (the `node_up`/`node_down` interface and -//! the view) and the connector dial loop are c7, built on top of this table. -//! What lives here is only the table itself and the uniqueness rule the -//! handshake's `NameTaken` verdict depends on. +//! The manager also holds the **membership state** (c7a): `node_up` fires on +//! a successful registration and `node_down` on removal — they are derived +//! facts of the exact events this table already owns, so holding the view +//! here means no cross-actor race between "connection exists" and "node is +//! up". The consumer surface (event types, [`subscribe`], [`view`], +//! semantics) is [`membership`](crate::cluster::membership); no consumer +//! ever touches the table itself. +//! +//! The connector dial loop is c7b, built on top of both. use std::collections::HashMap; +use crate::channel::Sender; use crate::cluster::conn::ConnHandle; -use crate::cluster::handshake::HelloCtx; +use crate::cluster::handshake::{HelloCtx, Peer}; +use crate::cluster::membership::{NodeEvent, NodeInfo}; use crate::gen_server::{GenServer, GenServerCtx, GenServerName, Watcher}; use crate::monitor::{monitor, Down}; +use crate::pg::NodeId; use crate::pid::Pid; /// Well-known name of the singleton manager within a runtime. Connection @@ -28,15 +36,18 @@ use crate::pid::Pid; /// manager is always found at the same key. pub const MANAGER: GenServerName = GenServerName::new("smarm.cluster.manager"); -/// One live connection's entry: the actor running it, and the handle whose -/// lifetime *is* the connection's (see the module docs). +/// One live connection's entry: the actor running it, the handle whose +/// lifetime *is* the connection's (see the module docs), and the peer's +/// membership identity (what `node_up` announced and `node_down` will name). struct ConnEntry { pid: Pid, + info: NodeInfo, _handle: ConnHandle, } /// The connection registry: peer name → the connection actor that owns that -/// peer's control connection. +/// peer's control connection. Plus the membership state layered on it (c7a): +/// subscribers, and the `(name, incarnation)` → [`NodeId`] memo. pub struct Manager { conns: HashMap, /// In-flight dial intents: peer name -> the actor performing the dial. @@ -45,6 +56,16 @@ pub struct Manager { /// the dial resolves, and — because the dialer is monitored — on the /// dialer's death, so a panicking dial can never wedge the tie-break. dials: HashMap, + /// Membership subscribers; a closed channel is pruned on the next emit. + subscribers: Vec>, + /// The [`NodeId`] memo: a reconnect at the same incarnation keeps its id, + /// a restart (new incarnation) allocates a fresh one. Grows one entry per + /// distinct `(name, incarnation)` ever seen — unbounded in principle, + /// bounded in practice by restarts actually happening. + ids: HashMap<(String, u32), NodeId>, + /// Next id to allocate. Starts at 1: id 0 is + /// [`DEFAULT_NODE_ID`](crate::pg::DEFAULT_NODE_ID), the local node. + next_id: u32, watcher: Option>, } @@ -53,9 +74,31 @@ impl Manager { Manager { conns: HashMap::new(), dials: HashMap::new(), + subscribers: Vec::new(), + ids: HashMap::new(), + next_id: 1, watcher: None, } } + + /// The memoized id for `(name, incarnation)` — see the field docs. + fn node_id(&mut self, name: &str, incarnation: u32) -> NodeId { + *self + .ids + .entry((name.to_string(), incarnation)) + .or_insert_with(|| { + let id = NodeId::new(self.next_id); + self.next_id += 1; + id + }) + } + + /// Deliver `event` to every live subscriber, pruning the dead: a closed + /// channel means the subscriber dropped its [`MembershipEvents`] + /// (crate::cluster::membership::MembershipEvents). + fn emit(&mut self, event: &NodeEvent) { + self.subscribers.retain(|tx| tx.send(event.clone()).is_ok()); + } } impl Default for Manager { @@ -77,12 +120,13 @@ pub enum Registered { /// Requests to the manager. pub enum Call { /// The path claims its peer's name for a freshly-established connection, - /// handing the manager the actor's [`ConnHandle`]. The manager monitors - /// `pid` and holds the handle for as long as the entry lives; a + /// handing the manager the actor's [`ConnHandle`] and the handshake's + /// [`Peer`] (the membership identity `node_up` announces). The manager + /// monitors `pid` and holds the handle for as long as the entry lives; a /// [`Registered::Duplicate`] verdict drops the handle here, which stops /// the refused actor. Register { - name: String, + peer: Peer, pid: Pid, handle: ConnHandle, }, @@ -101,6 +145,15 @@ pub enum Call { /// The [`HelloCtx`] for an inbound `Hello` offering `peer_name` — the /// accept path asks this between reading the frame and judging it. HelloCtx { peer_name: String }, + /// Subscribe `tx` to membership events, snapshot-then-stream: one + /// [`NodeEvent::NodeUp`] per live peer is queued into `tx` before this + /// call answers, so the stream is exact from its first event (handlers + /// are serialized — nothing interleaves with the snapshot). Use + /// [`subscribe`](crate::cluster::membership::subscribe). + Subscribe { tx: Sender }, + /// The current view: every live peer's [`NodeInfo`], unordered. Use + /// [`view`](crate::cluster::membership::view). + View, } /// Replies from the manager. @@ -113,6 +166,8 @@ pub enum Reply { DialBegan(bool), DialEnded, HelloCtx(HelloCtx), + Subscribed, + View(Vec), } impl GenServer for Manager { @@ -128,26 +183,38 @@ impl GenServer for Manager { fn handle_call(&mut self, request: Call) -> Reply { match request { - Call::Register { name, pid, handle } => { - if self.conns.contains_key(&name) { + Call::Register { peer, pid, handle } => { + if self.conns.contains_key(&peer.node_name) { // `handle` drops here: the refused actor stops itself. return Reply::Registered(Registered::Duplicate); } if let Some(w) = &self.watcher { w.watch(monitor(pid)); } + let info = NodeInfo { + node: self.node_id(&peer.node_name, peer.incarnation.get()), + name: peer.node_name.clone(), + incarnation: peer.incarnation, + meta: peer.meta, + }; self.conns.insert( - name, + peer.node_name, ConnEntry { pid, + info: info.clone(), _handle: handle, }, ); + self.emit(&NodeEvent::NodeUp(info)); Reply::Registered(Registered::Ok) } Call::Disconnect { name } => { // Dropping the entry drops the handle, which stops the actor. - self.conns.remove(&name); + if let Some(entry) = self.conns.remove(&name) { + self.emit(&NodeEvent::NodeDown { + node: entry.info.node, + }); + } Reply::Disconnected } Call::Peers => { @@ -173,13 +240,33 @@ impl GenServer for Manager { name_claimed: self.conns.contains_key(&peer_name), dialing_this_peer: self.dials.contains_key(&peer_name), }), + Call::Subscribe { tx } => { + // The snapshot: queued before `tx` joins the list, and — the + // handlers being serialized — before any later event. + for entry in self.conns.values() { + let _ = tx.send(NodeEvent::NodeUp(entry.info.clone())); + } + self.subscribers.push(tx); + Reply::Subscribed + } + Call::View => Reply::View(self.conns.values().map(|e| e.info.clone()).collect()), } } fn handle_cast(&mut self, _request: ()) {} fn handle_down(&mut self, down: Down) { - self.conns.retain(|_, entry| entry.pid != down.pid); + let mut downs = Vec::new(); + self.conns.retain(|_, entry| { + let dead = entry.pid == down.pid; + if dead { + downs.push(entry.info.node); + } + !dead + }); + for node in downs { + self.emit(&NodeEvent::NodeDown { node }); + } self.dials.retain(|_, pid| *pid != down.pid); } } diff --git a/src/cluster/membership.rs b/src/cluster/membership.rs new file mode 100644 index 0000000..4a97390 --- /dev/null +++ b/src/cluster/membership.rs @@ -0,0 +1,97 @@ +//! RFC 010 c7a — membership: `node_up`/`node_down` events and the view. +//! +//! The membership *state* lives inside the [`manager`](crate::cluster::manager) +//! — `node_up` and `node_down` are derived facts of the exact events the +//! manager already owns (a successful registration; a reap or `Disconnect`), +//! so holding the view anywhere else would only add a cross-actor ordering +//! seam. This module is the consumer surface: the event and view types, and +//! the [`subscribe`]/[`view`] entry points. No consumer ever touches the +//! connection table (roadmap-binding, enforced by module privacy: the table +//! is a private field, and nothing here exposes names→pids). +//! +//! ## Subscription semantics (ratified 2026-08-15) +//! +//! [`subscribe`] is **snapshot-then-stream**: the returned receiver first +//! yields one [`NodeEvent::NodeUp`] per currently-live peer, then live events +//! as they happen. Because the manager is a `gen_server` (handlers are +//! serialized), the snapshot is exact — no event can interleave with it, and +//! per-subscriber ordering matches the manager's processing order. There is +//! no join-race for late subscribers and no separate "get, then diff" dance; +//! [`view`] exists for observation, not for synchronization. +//! +//! A dropped subscriber is pruned on the next emission (its channel reports +//! closed) — no monitor needed, the sender itself tells us. +//! +//! ## NodeId identity +//! +//! A [`NodeId`] is a compact **local alias for the wire identity** +//! `(node_name, incarnation)`, memoized by the manager: a reconnect blip at +//! the same incarnation keeps its id (down, then up, same id), while a +//! restart — a new incarnation — gets a fresh one, so a node's ghost and its +//! successor are always distinguishable. Ids are allocated from 1; +//! [`DEFAULT_NODE_ID`](crate::pg::DEFAULT_NODE_ID) (0) remains the local +//! node, per [`pg`](crate::pg)'s framing. + +use crate::channel::{channel, Receiver}; +use crate::cluster::envelope::NodeMeta; +use crate::cluster::manager::{Call, Reply, MANAGER}; +use crate::gen_server; +use crate::pg::{Incarnation, NodeId}; + +/// One live remote node, as the view and [`NodeEvent::NodeUp`] describe it. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct NodeInfo { + /// The local alias for `(name, incarnation)` — see the module docs. + pub node: NodeId, + /// The peer's claimed node name (handshake-verified). + pub name: String, + /// The peer's incarnation epoch, as offered in its `Hello`. + pub incarnation: Incarnation, + /// The peer's `Hello` metadata. + pub meta: NodeMeta, +} + +/// A membership change, as delivered to subscribers. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum NodeEvent { + /// A peer's control connection established and registered. + NodeUp(NodeInfo), + /// That peer's connection ended — reaped, commanded down, or the manager + /// itself shut down. Which [`NodeInfo`] this id named was delivered in + /// the corresponding `NodeUp`. + NodeDown { node: NodeId }, +} + +/// A live membership subscription: the receiving end of the event stream +/// (the [`Monitor`](crate::monitor::Monitor) shape — read from [`rx`], drop +/// to unsubscribe). +/// +/// [rx]: MembershipEvents::rx +pub struct MembershipEvents { + /// The event stream: the snapshot's `NodeUp`s first, then live events. + /// Fold it into a `select` from a plain actor, or pipe it into a + /// `gen_server` via `with_info`. + pub rx: Receiver, +} + +/// Subscribe to membership events (snapshot-then-stream — see the module +/// docs). `None`: the manager is not running. Must be called from inside an +/// actor. +pub fn subscribe() -> Option { + let (tx, rx) = channel(); + match gen_server::call(MANAGER, Call::Subscribe { tx }) { + Ok(Reply::Subscribed) => Some(MembershipEvents { rx }), + _ => None, + } +} + +/// The current view: every live peer's [`NodeInfo`], unordered. For +/// observation and tests; consumers that need to *track* the view should +/// [`subscribe`] instead (the snapshot makes the stream self-sufficient). +/// `None`: the manager is not running. Must be called from inside an actor. +pub fn view() -> Option> { + match gen_server::call(MANAGER, Call::View) { + Ok(Reply::View(v)) => Some(v), + _ => None, + } +} diff --git a/tests/cluster_membership.rs b/tests/cluster_membership.rs new file mode 100644 index 0000000..cf2eb67 --- /dev/null +++ b/tests/cluster_membership.rs @@ -0,0 +1,261 @@ +//! RFC 010 c7a — membership events and the view, at the manager. +//! +//! Same construction as the c6a lifecycle suite: the handshake is bypassed, +//! connections are built already-established over localhost TCP pairs with +//! fabricated `Peer`s, and the manager is started plainly so the test can +//! terminate. What is under test is the membership layer that c7 adds to the +//! manager: `node_up`/`node_down` events to subscribers (snapshot-then-stream), +//! the view, and NodeId identity — memoized per `(name, incarnation)`, so a +//! reconnect blip keeps its id and a restart (new incarnation) gets a fresh one. +#![cfg(feature = "cluster")] + +use std::time::Duration; + +use smarm::cluster::envelope::NodeMeta; +use smarm::cluster::handshake::Peer; +use smarm::cluster::manager::{Call, Manager, Reply, MANAGER}; +use smarm::cluster::membership::{subscribe, view, MembershipEvents, NodeEvent}; +use smarm::cluster::spawn_established; +use smarm::cluster::transport::tcp::TcpTransport; +use smarm::cluster::transport::{Conn, FramedConn, Transport}; +use smarm::gen_server::{self, GenServerBuilder}; +use smarm::pg::{Incarnation, NodeId}; +use smarm::run; + +/// A fabricated post-handshake peer identity, with the incarnation under the +/// test's control (it is identity-bearing here, unlike in the c6a suite). +fn peer(name: &str, inc: u32) -> Peer { + Peer { + node_name: name.to_string(), + incarnation: Incarnation::new(inc), + meta: NodeMeta { + role: "test".to_string(), + region: "test".to_string(), + }, + } +} + +/// One established transport pair over localhost (TCP backlog covers the +/// sequential dial-then-accept, as in the c3 conformance suite). +fn pair(t: &dyn Transport) -> (Box, Box) { + let mut l = t.listen("127.0.0.1:0").unwrap(); + let a = t.dial(&l.local_addr()).unwrap(); + let b = l.accept().unwrap(); + (a, b) +} + +/// The next event, or a panic naming the wait. The bound is generous against +/// a sub-millisecond real cost. +fn next_event(ev: &MembershipEvents, waiting_for: &str) -> NodeEvent { + ev.rx + .recv_timeout(Duration::from_secs(5)) + .unwrap_or_else(|e| panic!("timed out waiting for {waiting_for}: {e:?}")) +} + +/// Assert the subscription is drained: no event is pending. +fn assert_quiet(ev: &MembershipEvents) { + assert!(matches!(ev.rx.try_recv(), Ok(None))); +} + +fn disconnect(name: &str) { + assert!(matches!( + gen_server::call( + MANAGER, + Call::Disconnect { + name: name.to_string() + } + ), + Ok(Reply::Disconnected) + )); +} + +/// Live subscription: an empty snapshot, then `NodeUp` on registration and +/// `NodeDown` (same id) on commanded disconnect and on peer EOF alike. +#[test] +fn subscriber_sees_up_and_down() { + run(|| { + let mgr = GenServerBuilder::new(Manager::new()) + .named(MANAGER) + .start() + .expect("manager name is free"); + + let ev = subscribe().expect("manager is up"); + assert_quiet(&ev); // nothing live: the snapshot is empty + + let t = TcpTransport; + let (a1, b1) = pair(&t); + let (a2, b2) = pair(&t); + spawn_established(FramedConn::new(a1), peer("node-b", 1)).expect("node-b registers"); + spawn_established(FramedConn::new(a2), peer("node-c", 1)).expect("node-c registers"); + + let up_b = match next_event(&ev, "node_up(node-b)") { + NodeEvent::NodeUp(info) => { + assert_eq!(info.name, "node-b"); + assert_eq!(info.incarnation, Incarnation::new(1)); + assert_eq!(info.meta.role, "test"); + info + } + other => panic!("expected node_up(node-b), got {other:?}"), + }; + let up_c = match next_event(&ev, "node_up(node-c)") { + NodeEvent::NodeUp(info) => { + assert_eq!(info.name, "node-c"); + info + } + other => panic!("expected node_up(node-c), got {other:?}"), + }; + assert_ne!(up_b.node, up_c.node, "distinct peers get distinct ids"); + + // Commanded disconnect: down with node-b's id. + disconnect("node-b"); + assert_eq!( + next_event(&ev, "node_down(node-b)"), + NodeEvent::NodeDown { node: up_b.node } + ); + + // Peer EOF, no command: down with node-c's id. + drop(b2); + assert_eq!( + next_event(&ev, "node_down(node-c)"), + NodeEvent::NodeDown { node: up_c.node } + ); + assert_quiet(&ev); + + drop(b1); + mgr.shutdown(); + }); +} + +/// Snapshot-then-stream: a subscriber arriving after connections established +/// receives one `NodeUp` per live peer before anything else, and the view +/// call agrees with it. +#[test] +fn late_subscriber_gets_snapshot_and_view_agrees() { + run(|| { + let mgr = GenServerBuilder::new(Manager::new()) + .named(MANAGER) + .start() + .expect("manager name is free"); + + let t = TcpTransport; + let (a1, b1) = pair(&t); + let (a2, b2) = pair(&t); + spawn_established(FramedConn::new(a1), peer("node-b", 1)).expect("node-b registers"); + spawn_established(FramedConn::new(a2), peer("node-c", 1)).expect("node-c registers"); + + let ev = subscribe().expect("manager is up"); + let mut names = Vec::new(); + for _ in 0..2 { + match next_event(&ev, "a snapshot node_up") { + NodeEvent::NodeUp(info) => names.push(info.name), + other => panic!("expected a snapshot node_up, got {other:?}"), + } + } + names.sort(); + assert_eq!(names, ["node-b", "node-c"]); + assert_quiet(&ev); // the snapshot is exactly the live set + + let mut v = view().expect("manager is up"); + v.sort_by(|a, b| a.name.cmp(&b.name)); + assert_eq!(v.len(), 2); + assert_eq!(v[0].name, "node-b"); + assert_eq!(v[1].name, "node-c"); + + disconnect("node-b"); + disconnect("node-c"); + drop((b1, b2)); + // Drain the two downs so the subscription ends quiet. + let _ = next_event(&ev, "node_down"); + let _ = next_event(&ev, "node_down"); + mgr.shutdown(); + }); +} + +/// NodeId identity: a restart (same name, new incarnation) is a NEW id — the +/// ghost and its successor are distinguishable — while a reconnect blip (same +/// name, same incarnation) keeps its id. +#[test] +fn restart_gets_new_id_blip_keeps_id() { + run(|| { + let mgr = GenServerBuilder::new(Manager::new()) + .named(MANAGER) + .start() + .expect("manager name is free"); + let ev = subscribe().expect("manager is up"); + let t = TcpTransport; + + let id = |e: NodeEvent, what: &str| -> NodeId { + match e { + NodeEvent::NodeUp(info) => info.node, + other => panic!("expected node_up ({what}), got {other:?}"), + } + }; + + // Up at incarnation 1, then the peer dies (EOF). + let (a1, b1) = pair(&t); + spawn_established(FramedConn::new(a1), peer("node-b", 1)).expect("registers"); + let id1 = id(next_event(&ev, "node_up inc 1"), "inc 1"); + drop(b1); + assert_eq!( + next_event(&ev, "node_down inc 1"), + NodeEvent::NodeDown { node: id1 } + ); + + // Restart: new incarnation, new id — the ghost's id is not reused. + let (a2, b2) = pair(&t); + spawn_established(FramedConn::new(a2), peer("node-b", 2)).expect("registers"); + let id2 = id(next_event(&ev, "node_up inc 2"), "inc 2"); + assert_ne!( + id1, id2, + "a restarted node must be distinguishable from its ghost" + ); + + // Blip: the same incarnation reconnects and keeps its id. + disconnect("node-b"); + assert_eq!( + next_event(&ev, "node_down inc 2"), + NodeEvent::NodeDown { node: id2 } + ); + let (a3, b3) = pair(&t); + spawn_established(FramedConn::new(a3), peer("node-b", 2)).expect("registers"); + let id3 = id(next_event(&ev, "node_up after blip"), "blip"); + assert_eq!( + id2, id3, + "a reconnect at the same incarnation is the same node" + ); + + disconnect("node-b"); + let _ = next_event(&ev, "final node_down"); + drop((b2, b3)); + mgr.shutdown(); + }); +} + +/// A dropped subscriber is pruned on the next emit and never disturbs the +/// manager or a live subscriber. +#[test] +fn dead_subscriber_is_pruned() { + run(|| { + let mgr = GenServerBuilder::new(Manager::new()) + .named(MANAGER) + .start() + .expect("manager name is free"); + + let dead = subscribe().expect("manager is up"); + drop(dead); + let live = subscribe().expect("manager is up"); + + let t = TcpTransport; + let (a1, b1) = pair(&t); + spawn_established(FramedConn::new(a1), peer("node-b", 1)).expect("registers"); + match next_event(&live, "node_up despite a dead co-subscriber") { + NodeEvent::NodeUp(info) => assert_eq!(info.name, "node-b"), + other => panic!("expected node_up, got {other:?}"), + } + + disconnect("node-b"); + let _ = next_event(&live, "node_down"); + drop(b1); + mgr.shutdown(); + }); +} From 1282c3a08d08f10b6754c116906fcc8d17a290a3 Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 15 Aug 2026 07:14:13 +0000 Subject: [PATCH 11/14] =?UTF-8?q?feat(cluster):=20RFC=20010=20c7b=20?= =?UTF-8?q?=E2=80=94=20discovery=20Strategy,=20static=20seeds,=20connector?= =?UTF-8?q?=20dial=20loop?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Phase 2 gate: 3-node mesh under the subprocess harness, repeatable (10/10). Strategy (ratified): push-based, spawned as its own actor by the connector — it emits Discovery events into a channel whenever it learns something and may run forever; the connector owns all retry/backoff state. StaticSeeds announces its list once and exits. Discovery is #[non_exhaustive] and additive-only (candidates announced, never withdrawn) so expiry can land later without breaking strategies. One-viable correction to the ratified Discovery shape, flagged: a candidate is a (name, addr) PAIR, not a bare address. The dial path and the D7 tie-break are keyed by peer name (the dial intent must be registered before connecting so a crossing inbound Hello sees it), so an anonymous dial would reintroduce exactly the simultaneous-connect flap D7 exists to prevent. Discovery mechanisms know names — that is what they discover. Connector: plain select-loop actor (the c6 shape) folding cmd inbox, discovery stream, membership stream, and the earliest retry deadline into one wait. It tracks who is up by SUBSCRIBING TO MEMBERSHIP like any consumer — first consumer of c7a's snapshot-then-stream surface, no privileged channel into the manager. Backoff: 250ms doubling to a 5s cap (the c6c class of one-viable constants), reset on node_up; node_down schedules a prompt redial with a fresh sequence. A candidate bearing the local name is parked (that seed is us); every other failure retries — in particular NameTaken can be our own ghost at the peer, not yet reaped by its liveness timer, so it must not park. Dials run inline in the loop, the acceptor's deliberate serialization (each attempt bounded by the connect + handshake deadlines). cluster::start(Config {node_name, meta, listen_addr, strategy}) is now the integrated node start: supervised manager + acceptor + connector. It completes the node identity: build_hash = cluster::BUILD_HASH (first consumer, closing the c6d loose end) and incarnation = self_incarnation() — unix-epoch MILLIS truncated to u32, not seconds: a supervised crash-and-restart inside one second is routine, and seconds would collide the ghost with its successor. Cluster handle: local_addr()/local()/ shutdown(); drop stops acceptor+connector loops, manager subtree detaches (same split as AcceptorHandle alone). Roadmap-binding, asserted in review: no consumer touches the connection table — Manager.conns and ConnEntry stay private; the only exposures are Call::Peers (sorted names, pre-existing) and the membership surface. tests/cluster_mesh.rs 2/0, 10/10 flake runs: (1) 3-node mesh forms; kill one (SIGKILL via Drop, per the retractable-state trap: roles park forever) => node_down at both survivors; restart same name => new incarnation at every observer, distinguishable from the ghost; (2) seed unreachable at start (pre-reserved closed port; accepted micro steal-window, documented) then arriving later => edge forms via the retry path. All cluster suites regression-clean (envelope 15, handshake 11, transport 11, lifecycle 1, liveness 3, connect 9, two_node 3, membership 4); clippy --lib green both configs; fmt clean; default build compiles. --- src/cluster.rs | 114 ++++++++++++++++--- src/cluster/connector.rs | 229 +++++++++++++++++++++++++++++++++++++++ src/cluster/discovery.rs | 67 ++++++++++++ tests/cluster_mesh.rs | 179 ++++++++++++++++++++++++++++++ 4 files changed, 575 insertions(+), 14 deletions(-) create mode 100644 src/cluster/connector.rs create mode 100644 src/cluster/discovery.rs create mode 100644 tests/cluster_mesh.rs diff --git a/src/cluster.rs b/src/cluster.rs index 286ed0c..851a011 100644 --- a/src/cluster.rs +++ b/src/cluster.rs @@ -10,21 +10,32 @@ pub mod conn; pub mod connect; +pub mod connector; +pub mod discovery; pub mod envelope; pub mod handshake; pub mod manager; pub mod membership; pub mod transport; -use std::time::Duration; +use std::io; +use std::time::{Duration, SystemTime, UNIX_EPOCH}; use crate::gen_server::{self, GenServerBuilder}; use crate::monitor::monitor; +use crate::pg::Incarnation; use crate::scheduler::{sleep, spawn, JoinHandle}; use crate::supervisor::{ChildSpec, OneForOne, Restart}; +use envelope::NodeMeta; +use handshake::Local; +use transport::tcp::TcpTransport; +use transport::Transport; + pub use conn::{spawn_established, ConnHandle}; pub use connect::{dial, spawn_acceptor, AcceptorHandle}; +pub use connector::{spawn_connector, ConnectorHandle}; +pub use discovery::{Discovery, StaticSeeds, Strategy}; pub use manager::{Manager, MANAGER}; pub use membership::{subscribe, view, MembershipEvents, NodeEvent, NodeInfo}; @@ -65,20 +76,67 @@ const fn fold_u32(mut h: u64, v: u32) -> u64 { h } -/// A running cluster subtree: an explicitly-started supervisor over the -/// connection [`Manager`]. Roles will eventually mount this subtree; until the -/// role mechanism lands it is started by hand (RFC 010 §7). Dropping the handle -/// detaches the subtree, which keeps running for the life of the runtime. -pub struct Cluster { - _sup: JoinHandle, +/// How to run this node: its identity and how it finds peers. +pub struct Config { + /// This node's claimed name — the mesh-wide identity peers dial by and + /// the tie-break input. Must be unique across the mesh. + pub node_name: String, + /// Metadata offered in this node's `Hello`. + pub meta: NodeMeta, + /// The control-connection listen address (e.g. `"127.0.0.1:0"`; the + /// concrete bound address is [`Cluster::local_addr`]). + pub listen_addr: String, + /// The peer-discovery strategy — [`StaticSeeds`] until richer ones land. + pub strategy: Box, } -/// Start the cluster subtree and block until the manager is registered and -/// ready to answer. The manager is a supervised child (restarted on crash); -/// per-peer connection actors are dynamic and monitored by the manager rather -/// than statically supervised — a lost connection is re-established by dialing -/// (c7), never resurrected onto a stale socket. -pub fn start() -> Cluster { +/// A running cluster node: the supervised [`Manager`], the acceptor over the +/// bound listener, and the connector driving its [`Strategy`]. Roles will +/// eventually mount this; until the role mechanism lands it is started by +/// hand (RFC 010 §7). +/// +/// Dropping the handle stops the acceptor and connector loops (no new +/// connections in either direction) but detaches the manager subtree, which +/// — with every established connection — keeps running for the life of the +/// runtime, the same split as [`AcceptorHandle`] alone. +pub struct Cluster { + _sup: JoinHandle, + acceptor: AcceptorHandle, + connector: ConnectorHandle, + local: Local, +} + +impl Cluster { + /// The concrete bound listen address, dialable as-is. + pub fn local_addr(&self) -> &str { + self.acceptor.local_addr() + } + + /// This node's handshake identity (name, incarnation, build hash, meta). + pub fn local(&self) -> &Local { + &self.local + } + + /// Stop accepting and dialing. Established connections stay up (they + /// belong to the manager); tear those down via the manager. + pub fn shutdown(&self) { + self.acceptor.shutdown(); + self.connector.shutdown(); + } +} + +/// Start a cluster node: the supervised manager (blocking until it is +/// registered and ready to answer), the acceptor bound per +/// [`Config::listen_addr`], and the connector running [`Config::strategy`]. +/// The node's identity is completed here: `incarnation` is +/// [`self_incarnation`] and `build_hash` is [`BUILD_HASH`] — c7 is its first +/// consumer. Errs only if the listener cannot bind. +/// +/// The manager is a supervised child (restarted on crash); per-peer +/// connection actors are dynamic and monitored by the manager rather than +/// statically supervised — a lost connection is re-established by the +/// connector's dial loop, never resurrected onto a stale socket. +pub fn start(config: Config) -> io::Result { let sup = spawn(|| { OneForOne::new() .child(ChildSpec::new(Restart::Permanent, manager_child)) @@ -87,7 +145,35 @@ pub fn start() -> Cluster { while gen_server::whereis_server(MANAGER).is_none() { sleep(Duration::from_millis(1)); } - Cluster { _sup: sup } + let local = Local { + node_name: config.node_name, + incarnation: self_incarnation(), + build_hash: BUILD_HASH, + meta: config.meta, + }; + let listener = TcpTransport.listen(&config.listen_addr)?; + let acceptor = spawn_acceptor(listener, local.clone()); + let connector = spawn_connector(Box::new(TcpTransport), local.clone(), config.strategy); + Ok(Cluster { + _sup: sup, + acceptor, + connector, + local, + }) +} + +/// This process's incarnation epoch: milliseconds since the Unix epoch, +/// truncated to `u32`. Not a clock — its one job is separating a node from +/// its own restart (two starts of the same name land on the same value only +/// if they happen within the same millisecond modulo ~49.7 days). Seconds +/// would be too coarse: a crash-and-restart inside one second is routine +/// under supervision. +pub fn self_incarnation() -> Incarnation { + let ms = SystemTime::now() + .duration_since(UNIX_EPOCH) + .map(|d| d.as_millis()) + .unwrap_or(0); + Incarnation::new(ms as u32) } /// The supervised manager child body. It *is* the child actor: it starts the diff --git a/src/cluster/connector.rs b/src/cluster/connector.rs new file mode 100644 index 0000000..b4633ea --- /dev/null +++ b/src/cluster/connector.rs @@ -0,0 +1,229 @@ +//! RFC 010 c7b — the connector: the dial loop that turns discovered +//! candidates into a full mesh. +//! +//! A plain select-loop actor (the c6 shape). It spawns its [`Strategy`] as a +//! child actor and receives [`Discovery`] events from it; it tracks which +//! peers are up by **subscribing to membership like any other consumer** — +//! no privileged channel into the manager, the same snapshot-then-stream +//! surface c8 will use. One `select` folds the command inbox, the discovery +//! stream, the membership stream, and the earliest retry deadline into a +//! single wait. +//! +//! Per-candidate state: dial on arrival; on failure retry with capped +//! exponential backoff ([`INITIAL_BACKOFF`] doubling to [`MAX_BACKOFF`]); +//! on the peer's `node_up` stop dialing and reset the backoff; on its +//! `node_down` resume immediately (a fresh sequence — the reconnect case is +//! the one backoff exists to pace, but the *first* retry after a death +//! should be prompt). A candidate bearing our own name is parked permanently +//! — that seed is us. Every other failure retries: in particular a +//! `NameTaken` reject can be our own ghost at the peer, not yet reaped by +//! its liveness timer, so it must not park. +//! +//! Dials run **inline in the loop** — the same deliberate serialization as +//! the acceptor (c6b): each attempt is bounded by the connect + handshake +//! deadlines, and nothing concurrent exists to be starved. A wall of slow +//! unreachable seeds would stretch the loop's latency; revisit if a real +//! deployment ever hits that shape. + +use std::collections::{HashMap, HashSet}; +use std::time::{Duration, Instant}; + +use crate::channel::{channel, select, select_timeout, Receiver, Selectable, Sender}; +use crate::cluster::connect::dial; +use crate::cluster::discovery::{Discovery, Strategy}; +use crate::cluster::handshake::Local; +use crate::cluster::membership::{subscribe, NodeEvent}; +use crate::cluster::transport::Transport; +use crate::pg::NodeId; +use crate::scheduler::spawn; + +/// First retry delay after a failed dial attempt. +pub const INITIAL_BACKOFF: Duration = Duration::from_millis(250); +/// Backoff ceiling: an unreachable seed is retried this often, forever. +pub const MAX_BACKOFF: Duration = Duration::from_secs(5); + +enum Cmd { + Shutdown, +} + +/// A running connector. `shutdown` (or dropping the last handle) stops the +/// dial loop and its strategy only — established connections belong to the +/// manager, exactly as with the acceptor. +pub struct ConnectorHandle { + cmd_tx: Sender, +} + +impl ConnectorHandle { + /// Ask the connector to stop. Idempotent; a no-op if it already has. + pub fn shutdown(&self) { + let _ = self.cmd_tx.send(Cmd::Shutdown); + } +} + +/// One discovered `(name, addr)` and its dial state. +struct Candidate { + name: String, + addr: String, + /// This seed is the local node itself: never dialed. + parked: bool, + /// Delay to apply after the *next* failure. + backoff: Duration, + next_attempt: Instant, +} + +/// Spawn the connector actor. The strategy is spawned as its child; the +/// membership subscription is taken inside the actor. Must be called from +/// inside an actor (the same requirement as `dial`). +pub fn spawn_connector( + transport: Box, + local: Local, + strategy: Box, +) -> ConnectorHandle { + let (cmd_tx, cmd_rx) = channel(); + spawn(move || run(transport, local, strategy, cmd_rx)); + ConnectorHandle { cmd_tx } +} + +fn run( + transport: Box, + local: Local, + strategy: Box, + cmd_rx: Receiver, +) { + // Membership is the connector's source of truth for "who is up" — the + // snapshot seeds `connected` before any candidate arrives. + let Some(events) = subscribe() else { + return; // no manager, no cluster to connect + }; + let (disc_tx, disc_rx) = channel(); + spawn(move || strategy.run(disc_tx)); + + let mut cands: Vec = Vec::new(); + let mut connected: HashSet = HashSet::new(); + let mut names: HashMap = HashMap::new(); // NodeDown carries only the id + let mut strategy_done = false; + + loop { + // Drain every input, then act. Order does not matter: acting is + // idempotent against the resulting state. + match drain_cmd(&cmd_rx) { + Drained::Stop => return, + Drained::Open => {} + } + if !strategy_done { + strategy_done = drain_discoveries(&disc_rx, &local, &mut cands); + } + match drain_events(&events.rx, &mut connected, &mut names, &mut cands) { + Drained::Stop => return, // manager gone: the cluster is tearing down + Drained::Open => {} + } + + // Dial everything due, inline (see the module docs on serialization). + let now = Instant::now(); + for c in cands + .iter_mut() + .filter(|c| !c.parked && !connected.contains(&c.name) && c.next_attempt <= now) + { + // The outcome does not branch the bookkeeping: on success the + // manager's node_up is on its way and flips `connected` (backing + // off meanwhile keeps a racing re-attempt from spinning); every + // failure retries — see the module docs. + let _ = dial(&*transport, &c.addr, &c.name, &local); + c.next_attempt = Instant::now() + c.backoff; + c.backoff = (c.backoff * 2).min(MAX_BACKOFF); + } + + // Wait: until the earliest retry deadline among actionable + // candidates, or indefinitely if none is pending. + let deadline = cands + .iter() + .filter(|c| !c.parked && !connected.contains(&c.name)) + .map(|c| c.next_attempt) + .min(); + let mut arms: Vec<&dyn Selectable> = vec![&cmd_rx, &events.rx]; + if !strategy_done { + arms.push(&disc_rx); + } + match deadline { + Some(d) => { + let wait = d.saturating_duration_since(Instant::now()); + let _ = select_timeout(&arms, wait); + } + None => { + let _ = select(&arms); + } + } + } +} + +enum Drained { + Open, + Stop, +} + +fn drain_cmd(rx: &Receiver) -> Drained { + match rx.try_recv() { + Ok(Some(Cmd::Shutdown)) => Drained::Stop, + Ok(None) => Drained::Open, + Err(_) => Drained::Stop, // all handles dropped + } +} + +/// Pull every pending discovery into the candidate set (deduplicated by +/// `(name, addr)`; a candidate bearing the local name is parked). Returns +/// `true` once the strategy's channel closes — it has said all it will. +fn drain_discoveries(rx: &Receiver, local: &Local, cands: &mut Vec) -> bool { + loop { + match rx.try_recv() { + Ok(Some(Discovery::Candidate { name, addr })) => { + if cands.iter().any(|c| c.name == name && c.addr == addr) { + continue; + } + let parked = name == local.node_name; + cands.push(Candidate { + name, + addr, + parked, + backoff: INITIAL_BACKOFF, + next_attempt: Instant::now(), + }); + } + Ok(None) => return false, + Err(_) => return true, // strategy done; its candidates live on here + } + } +} + +/// Fold pending membership events into `connected` (and the id→name map). +/// `node_up` resets its candidates' backoff; `node_down` schedules a prompt +/// redial with a fresh sequence. +fn drain_events( + rx: &Receiver, + connected: &mut HashSet, + names: &mut HashMap, + cands: &mut [Candidate], +) -> Drained { + loop { + match rx.try_recv() { + Ok(Some(NodeEvent::NodeUp(info))) => { + names.insert(info.node, info.name.clone()); + for c in cands.iter_mut().filter(|c| c.name == info.name) { + c.backoff = INITIAL_BACKOFF; + } + connected.insert(info.name); + } + Ok(Some(NodeEvent::NodeDown { node })) => { + if let Some(name) = names.remove(&node) { + connected.remove(&name); + let now = Instant::now(); + for c in cands.iter_mut().filter(|c| c.name == name) { + c.backoff = INITIAL_BACKOFF; + c.next_attempt = now; + } + } + } + Ok(None) => return Drained::Open, + Err(_) => return Drained::Stop, + } + } +} diff --git a/src/cluster/discovery.rs b/src/cluster/discovery.rs new file mode 100644 index 0000000..b7c0e29 --- /dev/null +++ b/src/cluster/discovery.rs @@ -0,0 +1,67 @@ +//! RFC 010 c7b — peer discovery: the [`Strategy`] seam and the static-seeds +//! implementation. +//! +//! A strategy is **push-based and runs as its own actor**: the +//! [`connector`](crate::cluster::connector) spawns it with the sending end of +//! a channel, and the strategy emits [`Discovery`] events whenever it learns +//! something — once at startup for a static list, continuously for a future +//! mDNS/DNS strategy — for as long as it cares to run. Returning ends the +//! strategy actor; the candidates it pushed live on in the connector (the +//! connector owns all retry/backoff state, so a strategy never re-announces). +//! +//! A candidate is a **`(node_name, addr)` pair**, not a bare address: the +//! dial path and the D7 tie-break are keyed by peer *name* (the dial intent +//! must be registered before connecting so a crossing inbound `Hello` sees +//! it), so an anonymous dial would reintroduce exactly the +//! simultaneous-connect flap D7 exists to prevent. Discovery mechanisms know +//! names — that is what they discover. + +use crate::channel::Sender; + +/// A discovery event, as pushed by a [`Strategy`]. +/// +/// Additive-only for now (candidates are announced, never withdrawn); +/// `#[non_exhaustive]` so expiry can land later without breaking strategies. +#[derive(Debug, Clone, PartialEq, Eq)] +#[non_exhaustive] +pub enum Discovery { + /// A peer worth dialing: its claimed node name and a dialable address. + Candidate { name: String, addr: String }, +} + +/// A source of peers to dial. Implementations are spawned as actors by the +/// connector — see the module docs for the contract. +pub trait Strategy: Send + 'static { + /// Run the strategy: push [`Discovery`] events into `out` as they are + /// learned; return when done discovering (or when `out` reports closed — + /// the connector is gone). Runs inside an actor, so blocking + /// cooperatively is fine. + fn run(self: Box, out: Sender); +} + +/// The static-seeds strategy: a fixed `(name, addr)` list, announced once. +#[derive(Debug, Clone, Default)] +pub struct StaticSeeds { + seeds: Vec<(String, String)>, +} + +impl StaticSeeds { + pub fn new(seeds: impl IntoIterator, impl Into)>) -> Self { + StaticSeeds { + seeds: seeds + .into_iter() + .map(|(n, a)| (n.into(), a.into())) + .collect(), + } + } +} + +impl Strategy for StaticSeeds { + fn run(self: Box, out: Sender) { + for (name, addr) in self.seeds { + if out.send(Discovery::Candidate { name, addr }).is_err() { + return; // connector gone; nobody to discover for + } + } + } +} diff --git a/tests/cluster_mesh.rs b/tests/cluster_mesh.rs new file mode 100644 index 0000000..6b6fb18 --- /dev/null +++ b/tests/cluster_mesh.rs @@ -0,0 +1,179 @@ +//! RFC 010 c7 — the Phase 2 gate: a 3-node mesh under the subprocess +//! harness, repeatable. +//! +//! Each node process runs the integrated `cluster::start` (manager + +//! acceptor + connector + static seeds), subscribes to membership like any +//! consumer, and announces protocol-visible facts as lines: +//! `LISTENING `, `MEMBER-UP inc=`, `MEMBER-DOWN `. +//! Then it **parks forever** — cross-process teardown is retractable state +//! (binding trap), so the parent SIGKILLs via `Node`'s `Drop` and clean exit +//! stays the c4 harness's own smoke test. +//! +//! Ports: nodes bind `:0` and report, so the mesh is built by seeding each +//! node with the previously-reported addresses (n1: no seeds; n2: n1; +//! n3: n1+n2 — inbound covers the reverse edges). The late-seed test is the +//! one exception: the parent pre-reserves a port by binding-and-closing it, +//! seeds one node with it, then starts the second node on that exact +//! address. In principle another process could steal the port in the gap; +//! in practice the window is microseconds on a local runner — accepted, and +//! confined to that one test. +#![cfg(feature = "cluster")] + +mod common; + +use common::{maybe_child, spawn_node, Node}; +use smarm::cluster::envelope::NodeMeta; +use smarm::cluster::membership::{subscribe, NodeEvent}; +use smarm::cluster::{start, Config, StaticSeeds}; +use smarm::pg::NodeId; +use std::collections::HashMap; +use std::time::Duration; + +const ROLES: &[(&str, fn())] = &[("node", role_node)]; + +/// A mesh node: identity and seeds from env, membership events to stdout, +/// park forever (the parent reaps). +fn role_node() { + let name = std::env::var("SMARM_NODE_NAME").expect("SMARM_NODE_NAME not set"); + let listen = std::env::var("SMARM_LISTEN_ADDR").unwrap_or_else(|_| "127.0.0.1:0".to_string()); + // Seeds: comma-separated `name=addr` pairs; empty or unset means none. + let seeds: Vec<(String, String)> = std::env::var("SMARM_SEEDS") + .unwrap_or_default() + .split(',') + .filter(|s| !s.is_empty()) + .map(|s| { + let (n, a) = s.split_once('=').expect("seed must be name=addr"); + (n.to_string(), a.to_string()) + }) + .collect(); + + smarm::run(move || { + let cluster = start(Config { + node_name: name, + meta: NodeMeta { + role: "mesh-test".to_string(), + region: "local".to_string(), + }, + listen_addr: listen, + strategy: Box::new(StaticSeeds::new(seeds)), + }) + .expect("listener binds"); + println!("LISTENING {}", cluster.local_addr()); + + let events = subscribe().expect("manager is up"); + let mut names: HashMap = HashMap::new(); + loop { + match events.rx.recv() { + Ok(NodeEvent::NodeUp(info)) => { + names.insert(info.node, info.name.clone()); + println!("MEMBER-UP {} inc={}", info.name, info.incarnation.get()); + } + Ok(NodeEvent::NodeDown { node }) => { + let name = names.remove(&node).unwrap_or_else(|| "?".to_string()); + println!("MEMBER-DOWN {name}"); + } + Err(_) => break, // manager gone; park below regardless + } + } + loop { + smarm::sleep(Duration::from_secs(3600)); + } + }); +} + +fn spawn_mesh_node(name: &str, seeds: &str, listen: Option<&str>) -> Node { + let mut env: Vec<(&str, &str)> = vec![("SMARM_NODE_NAME", name), ("SMARM_SEEDS", seeds)]; + if let Some(addr) = listen { + env.push(("SMARM_LISTEN_ADDR", addr)); + } + spawn_node("node", &env) +} + +/// Wait for `MEMBER-UP inc=` and return the incarnation. +fn wait_member_up(node: &mut Node, peer: &str) -> u32 { + let prefix = format!("MEMBER-UP {peer} inc="); + let line = node.wait_line(&format!("MEMBER-UP {peer}"), |l| l.starts_with(&prefix)); + line[prefix.len()..].parse().expect("incarnation parses") +} + +fn wait_member_down(node: &mut Node, peer: &str) { + let want = format!("MEMBER-DOWN {peer}"); + node.wait_line(&want, |l| l == want); +} + +/// The gate, plus the kill and restart facts, as one mesh's life: three +/// nodes form a full mesh (every node sees both others up); killing one +/// yields `node_down` at both survivors; its restart under the same name +/// arrives as a NEW incarnation — the ghost and its successor are +/// distinguishable at every observer. +#[test] +fn three_node_mesh_forms_then_kill_then_restart_distinguishable() { + maybe_child(ROLES); + + let mut n1 = spawn_mesh_node("node-1", "", None); + let a1 = n1.wait_listening(); + let mut n2 = spawn_mesh_node("node-2", &format!("node-1={a1}"), None); + let a2 = n2.wait_listening(); + let mut n3 = spawn_mesh_node("node-3", &format!("node-1={a1},node-2={a2}"), None); + let _a3 = n3.wait_listening(); + + // Full mesh: each node reports both peers up (dialed or inbound alike). + wait_member_up(&mut n1, "node-2"); + let inc3_at_n1 = wait_member_up(&mut n1, "node-3"); + wait_member_up(&mut n2, "node-1"); + let inc3_at_n2 = wait_member_up(&mut n2, "node-3"); + wait_member_up(&mut n3, "node-1"); + wait_member_up(&mut n3, "node-2"); + assert_eq!( + inc3_at_n1, inc3_at_n2, + "one node, one incarnation, all observers" + ); + + // Kill node-3 (SIGKILL via Drop): node_down at both survivors. + drop(n3); + wait_member_down(&mut n1, "node-3"); + wait_member_down(&mut n2, "node-3"); + + // Restart node-3 under the same name: it re-dials its seeds and comes + // up everywhere as a new incarnation — never the ghost's. + let mut n3b = spawn_mesh_node("node-3", &format!("node-1={a1},node-2={a2}"), None); + let _ = n3b.wait_listening(); + let inc3b_at_n1 = wait_member_up(&mut n1, "node-3"); + let inc3b_at_n2 = wait_member_up(&mut n2, "node-3"); + assert_eq!(inc3b_at_n1, inc3b_at_n2); + assert_ne!( + inc3_at_n1, inc3b_at_n1, + "a restarted node must be distinguishable from its ghost" + ); + wait_member_up(&mut n3b, "node-1"); + wait_member_up(&mut n3b, "node-2"); +} + +/// A seed that is unreachable at start is not fatal: the connector retries +/// on backoff, and when a node finally appears at that address, the mesh +/// edge forms. +#[test] +fn seed_unreachable_at_start_then_arriving_later() { + maybe_child(ROLES); + + // Pre-reserve an address by binding and immediately closing it (see the + // module docs for the accepted steal window). Dials to it are refused + // until node-b starts there. + let reserved = { + let l = std::net::TcpListener::bind("127.0.0.1:0").expect("bind"); + l.local_addr().expect("addr").to_string() + }; + + let mut a = spawn_mesh_node("node-a", &format!("node-b={reserved}"), None); + let _ = a.wait_listening(); + + // Let a few refused attempts happen before the seed comes up, so the + // retry path is what forms the edge (backoff cap 5s < harness WAIT 10s). + std::thread::sleep(Duration::from_millis(600)); + + let mut b = spawn_mesh_node("node-b", "", Some(&reserved)); + let _ = b.wait_listening(); + + wait_member_up(&mut a, "node-b"); + wait_member_up(&mut b, "node-a"); +} From a7f98f8d48394fd640621df84218fc549ad37eac Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 15 Aug 2026 07:25:42 +0000 Subject: [PATCH 12/14] =?UTF-8?q?feat(cluster):=20RFC=20010=20c8=20?= =?UTF-8?q?=E2=80=94=20exposure=20registry=20+=20fixed-seed=20type=20hashi?= =?UTF-8?q?ng?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Nothing local is remotely reachable by default (RFC §4). expose(Name) marks a name remotely addressable and registers M's decoder under type_hash::(); expose_type::() registers only the decoder (the reply-to path). exposed_names() is the auditable remote surface. D3's watchable fold, resolved against the code as it stands (stated in the module docs): register() ALREADY stamps every named holder watchable ('no successfully-registered actor can die unflagged', registry.rs), so an exposed name's holder needs no extra mark — and re-registration after a holder's death re-stamps the new holder for free, which a per-tenancy mark taken at expose time could not do. The cluster's own mark_watchable set-site is therefore the pid crossing the wire (frame serialization, c10) — the exact analog of the membrane crossing. c8 adds only the name/type state neither the registry nor slot bits can carry. No new pid registry; RFC §4 honored. One-viable calls, flagged: - State lives on RuntimeInner (the pg pattern: leaf RawMutex field, cfg-gated behind cluster, zero-cost-when-off per c1) — c9's inbound decode consults it per frame; manager-held state would serialize every remote delivery through one gen_server. - type_hash = FNV-1a 64 (fixed seed: the offset basis) over TypeId: a constant of the binary — stable across runs of the same build (the scope the build-hash handshake reduces the mesh to), deliberately not across builds. Collisions degrade to decode error / refused channel, never a misroute (the NoChannel guarantee, RFC §3). - Decoder = decode-and-deliver-to-pid Arc closure capturing M (the one typed site): decode_payload then send_dyn. Wire-name → pid resolution stays OUTSIDE — that is c9's single seam, which calls decode_deliver. Arc so the call happens with the exposure lock RELEASED: send_dyn takes the registry lock, a mutual Leaf (the runtime asserts on nesting — caught live by the first test run). - expose is a name-level fact, valid for an unregistered name (names late-bind; c9 resolves per delivery). tests/cluster_expose.rs 5/0 stable x5, purely local per roadmap: exposed/unexposed lookup + audit listing; decoder registration and the delivery contract (happy path into a registered String channel; unknown hash; corrupt bytes; wrong channel refused — never misrouted); distinct types distinct hashes; expose/bridge-crossing agreement via the shared watchable observable (terminal_reason after holder death); hash stability across runs in the same binary via a c4-harness re-exec. Payload types are std types — the crate's serde is derive-less by design, user crates bring their own derive. All cluster suites regression-clean (envelope 15, handshake 11, transport 11, lifecycle 1, liveness 3, connect 9, two_node 3, membership 4, mesh 2); clippy --lib green both configs; fmt clean; default build compiles. --- src/cluster.rs | 2 + src/cluster/expose.rs | 238 ++++++++++++++++++++++++++++++++++++++++ src/runtime.rs | 8 ++ tests/cluster_expose.rs | 161 +++++++++++++++++++++++++++ 4 files changed, 409 insertions(+) create mode 100644 src/cluster/expose.rs create mode 100644 tests/cluster_expose.rs diff --git a/src/cluster.rs b/src/cluster.rs index 851a011..e2b2225 100644 --- a/src/cluster.rs +++ b/src/cluster.rs @@ -13,6 +13,7 @@ pub mod connect; pub mod connector; pub mod discovery; pub mod envelope; +pub mod expose; pub mod handshake; pub mod manager; pub mod membership; @@ -36,6 +37,7 @@ pub use conn::{spawn_established, ConnHandle}; pub use connect::{dial, spawn_acceptor, AcceptorHandle}; pub use connector::{spawn_connector, ConnectorHandle}; pub use discovery::{Discovery, StaticSeeds, Strategy}; +pub use expose::{expose, expose_type, type_hash, DeliverError}; pub use manager::{Manager, MANAGER}; pub use membership::{subscribe, view, MembershipEvents, NodeEvent, NodeInfo}; diff --git a/src/cluster/expose.rs b/src/cluster/expose.rs new file mode 100644 index 0000000..e35de5a --- /dev/null +++ b/src/cluster/expose.rs @@ -0,0 +1,238 @@ +//! RFC 010 c8 — explicit exposure: the node's remote surface, and the +//! fixed-seed type hash. +//! +//! Nothing local is remotely reachable by default (RFC §4 — "a gun needs a +//! safety"). [`expose`] marks a registered name remotely addressable and +//! registers `M`'s decoder under [`type_hash::()`](type_hash); +//! [`expose_type`] registers only the decoder (the reply-to path: a +//! `RemotePid` received in a message is sendable only if `A::Msg`'s +//! decoder was explicitly registered). The exposed set is the node's +//! visible, auditable remote surface ([`exposed_names`]). +//! +//! ## Where the state lives +//! +//! On `RuntimeInner`, the [`pg`](crate::pg) pattern: a leaf-locked table, +//! cfg-gated behind the `cluster` feature (zero-cost-when-off, per c1). +//! Chosen over manager-held state because c9's inbound decode consults it +//! per frame — a hot path that must not serialize every remote delivery +//! through one gen_server. The state resets with the runtime, like every +//! registry. +//! +//! ## The watchable fold (D3), against the code as it stands +//! +//! RFC §4: the exposed set is not a new registry — it folds into the +//! existing `watchable` machinery, one set, two set-sites (a pid crossing +//! the membrane, and expose). Reading the code: `register` **already +//! stamps every named holder watchable** ("no successfully-registered actor +//! can die unflagged", registry.rs), so an exposed *name*'s holder needs no +//! extra mark here — the guarantee holds by registration, and re-registration +//! after a holder's death re-stamps the new holder for free (a per-tenancy +//! mark taken at expose time could not do that). The cluster's own +//! `mark_watchable` set-site is therefore the **pid crossing the wire** — +//! serialization of a pid into a frame, c10 — the exact analog of the +//! membrane crossing. What lives here is only the name/type-level state +//! neither the registry nor the slot bits can carry: which names are +//! exposed, and how to decode each type hash. +//! +//! ## The hash +//! +//! [`type_hash`] is FNV-1a 64 (fixed seed: the FNV offset basis) over +//! `TypeId`, so it is a constant of the binary: stable across runs of the +//! same build — exactly the scope the build-hash handshake reduces the mesh +//! to — and deliberately *not* stable across builds (scope guard: no +//! cross-version wire compatibility). A collision between two exposed types +//! degrades to a decode error or a refused channel, never a misroute — the +//! local `SendError::NoChannel` guarantee survives the network (RFC §3). +//! +//! ## The decoder contract +//! +//! A decoder is **decode-and-deliver-to-pid**: it captures `M` (the one +//! typed site), decodes the payload, and hands the value to the target's +//! published channel via the registry's own dynamic send. Wire-name → +//! local-pid resolution deliberately stays *outside* — that is c9's single +//! resolution seam, and it calls [`decode_deliver`]. + +use std::any::TypeId; +use std::collections::HashMap; +use std::hash::{Hash, Hasher}; + +use crate::cluster::envelope::{decode_payload, PayloadError}; +use crate::pid::{Name, Pid}; +use crate::registry::{send_dyn, SendError}; +use crate::scheduler::with_runtime; + +/// The fixed-seed `TypeId` → `u64` hash: FNV-1a 64 over the `TypeId`'s hash +/// bytes, seeded with the FNV offset basis. A constant of the binary — see +/// the module docs for scope. +pub fn type_hash() -> u64 { + let mut h = Fnv1a64::new(); + TypeId::of::().hash(&mut h); + h.finish() +} + +/// FNV-1a 64 as a `Hasher`, so `TypeId` (opaque, `Hash`-only) can feed it. +/// Same constants as the const fns in [`crate::cluster`] (BUILD_HASH). +struct Fnv1a64(u64); + +impl Fnv1a64 { + fn new() -> Self { + Fnv1a64(0xcbf2_9ce4_8422_2325) + } +} + +impl Hasher for Fnv1a64 { + fn write(&mut self, bytes: &[u8]) { + for &b in bytes { + self.0 ^= b as u64; + self.0 = self.0.wrapping_mul(0x0000_0100_0000_01b3); + } + } + fn finish(&self) -> u64 { + self.0 + } +} + +/// Why a [`decode_deliver`] did not deliver. Payload-free mirror of the +/// registry's `SendError` where relevant — the caller (c9's inbound path) +/// has only bytes to give back, not a typed message. +#[derive(Debug)] +pub enum DeliverError { + /// No decoder is registered under this hash — the type was never + /// exposed here. + UnknownType, + /// The bytes did not decode as the registered type. + Decode(PayloadError), + /// The target actor is dead (or was never alive). + Dead, + /// The target is live but has no channel for this message type, or that + /// channel is closed — the `NoChannel` guarantee: a decoded value is + /// refused, never misrouted. + WrongChannel, +} + +/// A registered decoder: decode `bytes` as the captured type and deliver to +/// `pid`'s published channel. `Arc`, so [`decode_deliver`] can clone it out +/// from under the exposure lock and call it lock-free — the decoder's +/// `send_dyn` takes the registry lock, and the two are mutual Leaves that +/// must never nest. +type Decoder = std::sync::Arc Result<(), DeliverError> + Send + Sync>; + +/// The exposure state, one per runtime (a `RuntimeInner` field, pg-style). +pub(crate) struct ExposureState { + /// The exposed names: registry key → the type hash it expects. + exposed: HashMap<&'static str, u64>, + /// The decoders: type hash → decode-and-deliver. + decoders: HashMap, +} + +impl ExposureState { + pub(crate) fn new() -> Self { + ExposureState { + exposed: HashMap::new(), + decoders: HashMap::new(), + } + } +} + +/// Mark `name` remotely addressable and register `M`'s decoder under its +/// type hash (so both name-sends and pid-sends of `M` work — RFC §4). +/// Returns the hash. +/// +/// Exposure is a **name-level fact**, independent of who currently holds the +/// name (names late-bind: the registry re-resolves on every send, and c9's +/// seam resolves per delivery). Exposing an unregistered name is therefore +/// valid — deliveries fail with "unresolved" until someone registers it. +/// Idempotent. Must run inside [`run`](crate::run). +pub fn expose(name: Name) -> u64 +where + M: serde::de::DeserializeOwned + Send + 'static, +{ + let h = ensure_decoder::(); + with_runtime(|inner| { + inner.exposure.lock().exposed.insert(name.as_str(), h); + }); + h +} + +/// Register only `M`'s decoder (no name): the reply-to path. Returns the +/// hash. Idempotent. Must run inside [`run`](crate::run). +pub fn expose_type() -> u64 +where + M: serde::de::DeserializeOwned + Send + 'static, +{ + ensure_decoder::() +} + +fn ensure_decoder() -> u64 +where + M: serde::de::DeserializeOwned + Send + 'static, +{ + let h = type_hash::(); + with_runtime(|inner| { + inner + .exposure + .lock() + .decoders + .entry(h) + .or_insert_with(decoder::); + }); + h +} + +/// The one typed site: decode as `M`, deliver via the registry's dynamic +/// send. See the module docs for the error mapping. +fn decoder() -> Decoder +where + M: serde::de::DeserializeOwned + Send + 'static, +{ + std::sync::Arc::new(|pid, bytes| { + let m: M = decode_payload(bytes).map_err(DeliverError::Decode)?; + send_dyn(pid, m).map_err(|e| match e { + SendError::Dead(_) | SendError::Unresolved(_) | SendError::NoMember(_) => { + DeliverError::Dead + } + SendError::NoChannel(_) | SendError::Closed(_) => DeliverError::WrongChannel, + }) + }) +} + +/// The type hash `name` was exposed with, or `None` if it is not exposed. +/// Must run inside [`run`](crate::run). +pub fn exposed_hash(name: &str) -> Option { + with_runtime(|inner| inner.exposure.lock().exposed.get(name).copied()) +} + +/// Whether a decoder is registered under `hash`. Must run inside +/// [`run`](crate::run). +pub fn decoder_registered(hash: u64) -> bool { + with_runtime(|inner| inner.exposure.lock().decoders.contains_key(&hash)) +} + +/// Decode `bytes` under `hash`'s registered decoder and deliver to `pid`. +/// This is the delivery half c9's single resolution seam calls after it has +/// resolved a wire name to a local pid. Must run inside [`run`](crate::run). +pub fn decode_deliver(hash: u64, to: Pid, bytes: &[u8]) -> Result<(), DeliverError> { + // Clone the Arc under the lock, call outside it: the decoder's + // `send_dyn` takes the registry lock — a mutual Leaf with the exposure + // lock (the runtime asserts if Leaves nest). This also keeps unrelated + // deliveries uncoupled from a slow decode. + let d = with_runtime(|inner| inner.exposure.lock().decoders.get(&hash).cloned()); + match d { + Some(d) => d(to, bytes), + None => Err(DeliverError::UnknownType), + } +} + +/// The auditable remote surface: every exposed name and its type hash, +/// unordered. Must run inside [`run`](crate::run). +pub fn exposed_names() -> Vec<(&'static str, u64)> { + with_runtime(|inner| { + inner + .exposure + .lock() + .exposed + .iter() + .map(|(&n, &h)| (n, h)) + .collect() + }) +} diff --git a/src/runtime.rs b/src/runtime.rs index be56e5e..d25e167 100644 --- a/src/runtime.rs +++ b/src/runtime.rs @@ -962,6 +962,12 @@ pub(crate) struct RuntimeInner { /// checks under it read only the atomic slot word, and the eviction path /// keeps it off the send path. pub(crate) process_groups: RawMutex, + /// RFC 010 c8: the exposure registry (exposed names + type-hash decoders). + /// RawMutex Leaf, same discipline as `process_groups`; decoders run under + /// it and are leaf-only by contract (they decode and send — `send_dyn` + /// takes `registry`, never this). cfg-gated: zero-cost-when-off (c1). + #[cfg(feature = "cluster")] + pub(crate) exposure: RawMutex, /// Recycled stacks waiting to be reused by the next spawn. pub(crate) stack_pool: RawMutex>, /// Maximum number of stacks to retain in the pool. @@ -1022,6 +1028,8 @@ impl RuntimeInner { node_id, incarnation, process_groups: RawMutex::new(crate::pg::ProcessGroups::new()), + #[cfg(feature = "cluster")] + exposure: RawMutex::new(crate::cluster::expose::ExposureState::new()), stack_pool: RawMutex::new(Vec::new()), stack_pool_cap, stack_reserve: crate::stack::round_to_pages(stack_reserve), diff --git a/tests/cluster_expose.rs b/tests/cluster_expose.rs new file mode 100644 index 0000000..9aad3af --- /dev/null +++ b/tests/cluster_expose.rs @@ -0,0 +1,161 @@ +//! RFC 010 c8 — exposure registry + type hashing. Purely local, no network. +//! +//! Payload types are std types (`String`, `u64`) because the crate's serde is +//! deliberately derive-less (`default-features = false`) — user crates bring +//! their own derive; the contract here is `DeserializeOwned`. +//! +//! The hash-stability test re-execs the current binary (the c4 harness): the +//! guarantee under test is "stable across runs in the SAME binary" — exactly +//! what the build-hash handshake reduces the mesh to — not stability across +//! builds, which the scope guard explicitly rejects. +#![cfg(feature = "cluster")] + +mod common; + +use common::{maybe_child, spawn_node}; +use smarm::cluster::envelope::encode_payload; +use smarm::cluster::expose::{ + decode_deliver, decoder_registered, expose, expose_type, exposed_hash, exposed_names, + type_hash, DeliverError, +}; +use smarm::monitor::{monitor, terminal_reason, DownReason}; +use smarm::{channel, register, run, spawn, Name}; + +const ROLES: &[(&str, fn())] = &[("hasher", role_hasher)]; + +/// Print the hashes this process computes; the parent (a different run of +/// the same binary) compares against its own. +fn role_hasher() { + println!("HASH-STRING {}", type_hash::()); + println!("HASH-U64 {}", type_hash::()); +} + +const GREETER: Name = Name::new("expose-test.greeter"); + +/// Exposed and unexposed lookup, the returned hash, and the audit listing. +#[test] +fn exposed_and_unexposed_lookup() { + maybe_child(ROLES); + run(|| { + let h = expose(GREETER); + assert_eq!(h, type_hash::()); + assert_eq!(exposed_hash("expose-test.greeter"), Some(h)); + assert_eq!(exposed_hash("never-exposed"), None); + assert!(exposed_names().contains(&("expose-test.greeter", h))); + }); +} + +/// Distinct types land on distinct hashes (FNV over distinct TypeIds — a +/// smoke assertion; a collision would degrade to a decode error, never a +/// misroute, per RFC §3). +#[test] +fn distinct_types_distinct_hashes() { + maybe_child(ROLES); + run(|| { + assert_ne!(type_hash::(), type_hash::()); + assert_ne!(type_hash::(), type_hash::>()); + }); +} + +/// The decode-and-deliver contract: a registered hash decodes into the +/// target's typed channel; an unknown hash, corrupt bytes, and a missing +/// channel each fail without delivering — `WrongChannel`, never a misroute. +#[test] +fn decoder_registration_and_delivery() { + maybe_child(ROLES); + run(|| { + let h_string = expose_type::(); + let h_u64 = expose_type::(); + assert!(decoder_registered(h_string)); + assert!(!decoder_registered(h_string.wrapping_add(1))); + + // A live actor with a String channel (registered from its own body, + // announced via a ready signal — the tests/registry.rs idiom). + let (ready_tx, ready_rx) = channel::<()>(); + let (stop_tx, stop_rx) = channel::<()>(); + let (msg_tx, msg_rx) = channel::(); + let pid = spawn(move || { + register(Name::::new("expose-test.sink"), msg_tx).unwrap(); + ready_tx.send(()).unwrap(); + let _ = stop_rx.recv(); + }) + .pid(); + ready_rx.recv().unwrap(); + + // Happy path: decode + deliver through the published channel. + let bytes = encode_payload("hello across the seam").unwrap(); + decode_deliver(h_string, pid, &bytes).unwrap(); + assert_eq!(msg_rx.recv().unwrap(), "hello across the seam"); + + // Unknown hash: nothing was registered under it. + assert!(matches!( + decode_deliver(h_string.wrapping_add(1), pid, &bytes), + Err(DeliverError::UnknownType) + )); + + // Corrupt bytes: the decoder fails before any send. + assert!(matches!( + decode_deliver(h_string, pid, &[0xff; 3]), + Err(DeliverError::Decode(_)) + )); + + // Right decoder, wrong channel: the actor has no u64 channel, so the + // decoded value is refused — the NoChannel guarantee. + let u64_bytes = encode_payload(&7u64).unwrap(); + assert!(matches!( + decode_deliver(h_u64, pid, &u64_bytes), + Err(DeliverError::WrongChannel) + )); + + stop_tx.send(()).unwrap(); + }); +} + +/// `expose` and the bridge crossing agree on the resulting set: both funnel +/// the pid-boundary mark through the watchable machinery, so an exposed +/// name's holder dies with a terminal record — the exact observable +/// `mark_watchable` guarantees the membrane. (For named holders the mark is +/// already stamped by `register` itself; this pins the shared contract.) +#[test] +fn expose_and_bridge_crossing_agree_on_the_set() { + maybe_child(ROLES); + run(|| { + let (ready_tx, ready_rx) = channel::<()>(); + let (stop_tx, stop_rx) = channel::<()>(); + let (msg_tx, _msg_rx) = channel::(); + let pid = spawn(move || { + register(GREETER, msg_tx).unwrap(); + ready_tx.send(()).unwrap(); + let _ = stop_rx.recv(); + }) + .pid(); + ready_rx.recv().unwrap(); + + expose(GREETER); + let m = monitor(pid); + stop_tx.send(()).unwrap(); + assert_eq!(m.rx.recv().unwrap().reason, DownReason::Exit); + assert_eq!(terminal_reason(pid), Some(DownReason::Exit)); + }); +} + +/// Hash stability across runs in the same binary: a re-exec of this binary +/// computes the same hashes this process does. +#[test] +fn hash_stable_across_runs_in_same_binary() { + maybe_child(ROLES); + let (mine_string, mine_u64) = { + // Computing a TypeId hash needs no runtime, but keep the contract + // uniform with real call sites. + (type_hash::(), type_hash::()) + }; + let mut child = spawn_node("hasher", &[]); + let line = child.wait_line("HASH-STRING", |l| l.starts_with("HASH-STRING ")); + assert_eq!( + line["HASH-STRING ".len()..].parse::().unwrap(), + mine_string + ); + let line = child.wait_line("HASH-U64", |l| l.starts_with("HASH-U64 ")); + assert_eq!(line["HASH-U64 ".len()..].parse::().unwrap(), mine_u64); + child.wait_exit(); +} From 16ef583455d2d12998037fba1d6e30a33272477b Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 15 Aug 2026 21:22:56 +0000 Subject: [PATCH 13/14] =?UTF-8?q?feat(cluster):=20RFC=20010=20c9=20?= =?UTF-8?q?=E2=80=94=20remote=20Name=20sends:=20outbound=20table=20+=20the?= =?UTF-8?q?=20one=20inbound=20seam?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Outbound (D13, ratified 2026-08-15): a module-private table node → dedicated Sender, populated/torn down by the manager inside the same serialized handlers that own connection lifetime (Register / Disconnect / reap / terminate), on RuntimeInner beside the exposure state. remote::send is one leaf-lock lookup + one channel send: no gen_server on the data plane (rejected send-through-manager — worse than the c8 argument, it is the data plane), no published channel a leaked conn pid could inject raw frames into (rejected send_dyn-at-conn-pid — module privacy cannot fence a published channel). Ok(()) = handed to the connection's inbox, local knowledge only (RFC §3); missing entry / closed channel = NotConnected. The outbound sender is a SEPARATE channel from cmd_tx on purpose: a clone would let the table hold lifetime authority (D9 violation); its closure is not a stop signal. Buffering toward a slow peer is unbounded (BEAM busy_dist_port shape); backpressure out of scope, documented not silently absent. Inbound: remote::deliver_named is THE resolution seam (RFC v2) — exposed-set check (unexposed = unreachable, the safety) → hash check against what the name was exposed with → registry whereis → c8's decode_deliver. Verdicts are local-only (InboundVerdict, discarded by the conn actor for now; a trace hook is the place). The conn actor gains a third select arm (outbound inbox → wire) and interprets SendNamed; Send/Monitor/Down are consumed for liveness and ignored until c10/c11. Public surface: RemoteName (node, Name), RemoteSendError, NotConnected, send, send_remote_raw (untyped escape hatch so tests can put deliberately wrong frames on the wire — the typed API cannot express a hash mismatch). CORE (found the hard way this chunk): publishing a second channel of the same message type on one live actor silently replaced and CLOSED the first, so a select/recv on it returned 'closed' immediately forever — a hot loop starving the single-threaded scheduler. publish_channel now asserts when an existing same-type channel's receiver is still alive AND the new sender is not a clone of it (Sender::same_channel via Arc::ptr_eq); replacing a dead-receiver channel stays silent (an actor re-registering after dropping its inbox is legit). Message names the sanctioned shapes. Three registry tests pin panic / cloned-sender-ok / dead-receiver-ok. Full default suite clean. Harness: wait_line drains stderr before dumping on timeout/EOF (eprintln! diagnostics no longer vanish); Node::transcript() accessor for ordering-proof assertions. tests/cluster_remote_send.rs 1/0, 10/10 flake runs, subprocess harness: cross-node name-send delivers; unexposed name unreachable (registered locally, never delivered); wrong hash never misroutes (unknown-hash and known-hash-wrong-channel flavours); send to an unconnected node = NotConnected locally; every frame-bearing send is Ok — 'handed to transport', asserted and documented at the test. Negatives are proven by STREAM ORDERING (they precede the positive on one in-order connection), not by sleeping. All cluster suites regression-clean (envelope 15, handshake 11, transport 11, lifecycle 1, liveness 3, connect 9, two_node 3, membership 4, mesh 2, expose 5); clippy --lib green both configs; fmt clean; default build compiles. --- src/channel.rs | 10 ++ src/cluster.rs | 2 + src/cluster/conn.rs | 106 ++++++++++++++-- src/cluster/manager.rs | 28 +++- src/cluster/remote.rs | 240 +++++++++++++++++++++++++++++++++++ src/registry.rs | 37 ++++++ src/runtime.rs | 7 + tests/cluster_remote_send.rs | 220 ++++++++++++++++++++++++++++++++ tests/common/mod.rs | 11 ++ tests/registry.rs | 43 +++++++ 10 files changed, 688 insertions(+), 16 deletions(-) create mode 100644 src/cluster/remote.rs create mode 100644 tests/cluster_remote_send.rs diff --git a/src/channel.rs b/src/channel.rs index a2af72a..9c50081 100644 --- a/src/channel.rs +++ b/src/channel.rs @@ -241,6 +241,16 @@ impl Sender { self.inner.lock().queue.len() } + /// Whether the [`Receiver`] is still alive (a send would be accepted). + pub(crate) fn receiver_alive(&self) -> bool { + self.inner.lock().receiver_alive + } + + /// Whether `other` is a sender of this very channel (a clone). + pub(crate) fn same_channel(&self, other: &Sender) -> bool { + Arc::ptr_eq(&self.inner, &other.inner) + } + /// Push `value` onto the channel. Succeeds unconditionally as long as /// the [`Receiver`] is still alive: the queue has no capacity limit, so /// this never blocks and never fails except when the channel is closed, diff --git a/src/cluster.rs b/src/cluster.rs index e2b2225..a56cc6c 100644 --- a/src/cluster.rs +++ b/src/cluster.rs @@ -17,6 +17,7 @@ pub mod expose; pub mod handshake; pub mod manager; pub mod membership; +pub mod remote; pub mod transport; use std::io; @@ -40,6 +41,7 @@ pub use discovery::{Discovery, StaticSeeds, Strategy}; pub use expose::{expose, expose_type, type_hash, DeliverError}; pub use manager::{Manager, MANAGER}; pub use membership::{subscribe, view, MembershipEvents, NodeEvent, NodeInfo}; +pub use remote::{NotConnected, RemoteName, RemoteSendError}; /// c6d — the derived build hash for [`handshake::LocalNode::build_hash`]: /// two builds may mesh only when this matches, and it is a pure function of diff --git a/src/cluster/conn.rs b/src/cluster/conn.rs index b2953b5..a9abad4 100644 --- a/src/cluster/conn.rs +++ b/src/cluster/conn.rs @@ -21,6 +21,16 @@ //! [`Frame::Heartbeat`](crate::cluster::envelope::Frame::Heartbeat)s, and a //! [`LIVENESS_TIMEOUT`] window — reset by any inbound frame — tears the //! connection down when it empties. +//! +//! c9 adds the third arm — the connection's dedicated **outbound inbox** +//! (`Sender` bound in the manager-maintained outbound table, D13), +//! drained onto the wire in the same loop — and inbound *interpretation*: +//! `SendNamed` goes to the one resolution seam, +//! [`remote::deliver_named`](crate::cluster::remote::deliver_named). +//! Frames the connection actor has no business with yet (`Send`, `Monitor`, +//! …: c10/c11) are consumed for liveness and otherwise ignored. The outbound +//! sender is a separate channel from `cmd_tx` on purpose: closing it is not +//! a stop signal — lifetime authority stays with the [`ConnHandle`] (D9). use std::time::{Duration, Instant}; @@ -28,6 +38,7 @@ use crate::channel::{channel, try_select_timeout, Receiver, Selectable, Sender}; use crate::cluster::envelope::Frame; use crate::cluster::handshake::Peer; use crate::cluster::manager::{Call, Registered, Reply, MANAGER}; +use crate::cluster::remote::deliver_named; use crate::cluster::transport::FramedConn; use crate::gen_server; use crate::pid::Pid; @@ -46,6 +57,11 @@ enum Cmd { /// to establish it. pub struct ConnHandle { cmd_tx: Sender, + /// The connection's dedicated outbound inbox. The manager moves this + /// into the outbound table on `Register` (see + /// [`take_outbound`](ConnHandle::take_outbound)); a `Duplicate` verdict + /// drops it with the handle. + out_tx: Option>, } impl std::fmt::Debug for ConnHandle { @@ -61,6 +77,12 @@ impl ConnHandle { pub fn shutdown(&self) { let _ = self.cmd_tx.send(Cmd::Shutdown); } + + /// Manager-only: take the outbound sender to bind into the outbound + /// table. Once, at registration. + pub(crate) fn take_outbound(&mut self) -> Option> { + self.out_tx.take() + } } /// The name was already claimed by a live connection, so this one was @@ -76,14 +98,18 @@ pub struct RegisterRefused; /// actor. A refusal has already stopped the actor and closed the socket. pub fn spawn_established(framed: FramedConn, peer: Peer) -> Result { let (cmd_tx, cmd_rx) = channel(); + let (out_tx, out_rx) = channel(); let reg_peer = peer.clone(); - let pid = spawn(move || run(framed, peer, cmd_rx)).pid(); + let pid = spawn(move || run(framed, peer, cmd_rx, out_rx)).pid(); match gen_server::call( MANAGER, Call::Register { peer: reg_peer, pid, - handle: ConnHandle { cmd_tx }, + handle: ConnHandle { + cmd_tx, + out_tx: Some(out_tx), + }, }, ) { Ok(Reply::Registered(Registered::Ok)) => Ok(pid), @@ -107,20 +133,30 @@ pub const HEARTBEAT_INTERVAL: Duration = Duration::from_secs(1); /// honest detector. pub const LIVENESS_TIMEOUT: Duration = Duration::from_secs(4); -fn run(mut framed: FramedConn, _peer: Peer, cmd_rx: Receiver) { +fn run(mut framed: FramedConn, _peer: Peer, cmd_rx: Receiver, out_rx: Receiver) { match framed.readable_arm() { - Some(arm) => run_live(&mut framed, arm, &cmd_rx), + Some(arm) => run_live(&mut framed, arm, &cmd_rx, &out_rx), None => run_inert(&cmd_rx), } framed.close(); } /// The steady-state loop over an fd-backed connection: one -/// `select_timeout` folds the command inbox, socket readability, and the -/// nearer of the two deadlines (`hb_send`, `liveness`) into a single wait. -fn run_live(framed: &mut FramedConn, arm: crate::scheduler::FdArm, cmd_rx: &Receiver) { +/// `select_timeout` folds the command inbox, the outbound inbox, socket +/// readability, and the nearer of the two deadlines (`hb_send`, +/// `liveness`) into a single wait. +fn run_live( + framed: &mut FramedConn, + arm: crate::scheduler::FdArm, + cmd_rx: &Receiver, + out_rx: &Receiver, +) { let mut next_hb = Instant::now(); let mut live_until = Instant::now() + LIVENESS_TIMEOUT; + // The outbound sender lives in the manager's table and is dropped on + // unbind; after that this arm would wake forever, so it drops out of + // the select (not a stop signal — see the module docs). + let mut out_open = true; loop { let now = Instant::now(); if now >= live_until { @@ -133,14 +169,18 @@ fn run_live(framed: &mut FramedConn, arm: crate::scheduler::FdArm, cmd_rx: &Rece next_hb = now + HEARTBEAT_INTERVAL; } let wait = next_hb.min(live_until).saturating_duration_since(now); - let arms: [&dyn Selectable; 2] = [cmd_rx, &arm]; + // Arm indices: 0 cmd, 1 fd, 2 outbound (when open). + let mut arms: Vec<&dyn Selectable> = vec![cmd_rx, &arm]; + if out_open { + arms.push(out_rx); + } match try_select_timeout(&arms, wait) { Ok(Some(0)) => { if should_stop(cmd_rx) { break; } } - Ok(Some(_)) => match pump_readable(framed) { + Ok(Some(1)) => match pump_readable(framed) { Pump::Ended => break, Pump::Frames(n) => { if n > 0 { @@ -148,6 +188,11 @@ fn run_live(framed: &mut FramedConn, arm: crate::scheduler::FdArm, cmd_rx: &Rece } } }, + Ok(Some(_)) => match pump_outbound(framed, out_rx) { + Outbound::Sent => {} + Outbound::Closed => out_open = false, + Outbound::WireFailed => break, + }, // A deadline passed; the top of the loop acts on whichever. Ok(None) => {} // The fd arm failed to register — the connection is gone. @@ -156,6 +201,30 @@ fn run_live(framed: &mut FramedConn, arm: crate::scheduler::FdArm, cmd_rx: &Rece } } +/// What one outbound wake yielded. +enum Outbound { + Sent, + /// The manager unbound this connection's sender; nothing more will come. + Closed, + /// The socket refused a write: the connection is gone. + WireFailed, +} + +/// Drain every queued outbound frame onto the wire. +fn pump_outbound(framed: &mut FramedConn, out_rx: &Receiver) -> Outbound { + loop { + match out_rx.try_recv() { + Ok(Some(frame)) => { + if framed.send(&frame).is_err() { + return Outbound::WireFailed; + } + } + Ok(None) => return Outbound::Sent, + Err(_) => return Outbound::Closed, + } + } +} + /// No fd to select on (loopback): only a command can end the wait, and /// neither heartbeats nor liveness run — a transport that can't report /// readiness can't be timed either (same caveat as @@ -196,8 +265,11 @@ enum Pump { /// after a level-triggered readable indication), then drain every complete /// frame the buffer now holds. A blocking `recv` here would park the actor /// past its heartbeat and liveness deadlines whenever a frame arrives split. -/// Frames are not interpreted yet — a heartbeat's entire job is the liveness -/// reset, and everything else waits for c8. +/// Every consumed frame counts for liveness; `SendNamed` additionally goes +/// to the one inbound resolution seam. Its verdict is local knowledge only +/// — nothing goes back on the wire (RFC §3) — and is currently discarded +/// (a future trace hook is the place to surface it). Frames for later +/// chunks (`Send` c10, `Monitor`/`Down` c11) are consumed and ignored. fn pump_readable(framed: &mut FramedConn) -> Pump { let eof = match framed.read_once() { Ok(n) => n == 0, @@ -206,7 +278,17 @@ fn pump_readable(framed: &mut FramedConn) -> Pump { let mut got = 0; loop { match framed.next_buffered() { - Ok(Some(_frame)) => got += 1, + Ok(Some(frame)) => { + got += 1; + if let Frame::SendNamed { + name, + type_hash, + payload, + } = frame + { + let _verdict = deliver_named(&name, type_hash, &payload); + } + } Ok(None) => break, Err(_) => return Pump::Ended, // corrupt stream } diff --git a/src/cluster/manager.rs b/src/cluster/manager.rs index 4a40152..2588023 100644 --- a/src/cluster/manager.rs +++ b/src/cluster/manager.rs @@ -26,6 +26,7 @@ use crate::channel::Sender; use crate::cluster::conn::ConnHandle; use crate::cluster::handshake::{HelloCtx, Peer}; use crate::cluster::membership::{NodeEvent, NodeInfo}; +use crate::cluster::remote::{bind_outbound, unbind_outbound}; use crate::gen_server::{GenServer, GenServerCtx, GenServerName, Watcher}; use crate::monitor::{monitor, Down}; use crate::pg::NodeId; @@ -181,9 +182,21 @@ impl GenServer for Manager { self.watcher = Some(ctx.watcher()); } + /// Manager shutdown drops every entry (and with it every ConnHandle); + /// the outbound table must not outlive the connections it names. + fn terminate(&mut self) { + for name in self.conns.keys() { + unbind_outbound(name); + } + } + fn handle_call(&mut self, request: Call) -> Reply { match request { - Call::Register { peer, pid, handle } => { + Call::Register { + peer, + pid, + mut handle, + } => { if self.conns.contains_key(&peer.node_name) { // `handle` drops here: the refused actor stops itself. return Reply::Registered(Registered::Duplicate); @@ -191,6 +204,11 @@ impl GenServer for Manager { if let Some(w) = &self.watcher { w.watch(monitor(pid)); } + // The outbound table (c9) is maintained here, inside the same + // serialized handlers that own the connection's lifetime. + if let Some(out) = handle.take_outbound() { + bind_outbound(&peer.node_name, out); + } let info = NodeInfo { node: self.node_id(&peer.node_name, peer.incarnation.get()), name: peer.node_name.clone(), @@ -211,6 +229,7 @@ impl GenServer for Manager { Call::Disconnect { name } => { // Dropping the entry drops the handle, which stops the actor. if let Some(entry) = self.conns.remove(&name) { + unbind_outbound(&name); self.emit(&NodeEvent::NodeDown { node: entry.info.node, }); @@ -257,14 +276,15 @@ impl GenServer for Manager { fn handle_down(&mut self, down: Down) { let mut downs = Vec::new(); - self.conns.retain(|_, entry| { + self.conns.retain(|name, entry| { let dead = entry.pid == down.pid; if dead { - downs.push(entry.info.node); + downs.push((name.clone(), entry.info.node)); } !dead }); - for node in downs { + for (name, node) in downs { + unbind_outbound(&name); self.emit(&NodeEvent::NodeDown { node }); } self.dials.retain(|_, pid| *pid != down.pid); diff --git a/src/cluster/remote.rs b/src/cluster/remote.rs new file mode 100644 index 0000000..33e4191 --- /dev/null +++ b/src/cluster/remote.rs @@ -0,0 +1,240 @@ +//! RFC 010 c9 — remote `Name` sends: the outbound path and the single +//! inbound name-resolution seam. +//! +//! ## Outbound (D13, ratified 2026-08-15) +//! +//! A module-private table `node name → Sender` — one dedicated +//! outbound channel per live connection, populated and torn down by the +//! manager inside the same serialized handlers that own the connection's +//! lifetime (Register / Disconnect / reap), living on `RuntimeInner` beside +//! the exposure state. [`send`] is one leaf-lock lookup + one channel send: +//! no gen_server on the data plane, no published channel anyone holding a +//! pid could inject raw frames into. `Ok(())` means **handed to the +//! connection's inbox** — local knowledge only, exactly the BEAM contract +//! (RFC §3): a missing entry or a closed channel is +//! [`RemoteSendError::NotConnected`]; delivery confirmation is the monitor's +//! job (c11). The entry-present/actor-dying-mid-send window is *honest* +//! under that contract, not a bug. +//! +//! The outbound sender is deliberately **separate from the conn actor's +//! command channel**: if it were a clone of `cmd_tx`, the manager dropping +//! its `ConnHandle` would no longer close that channel and connection +//! lifetime would leak to whoever holds a sender — a D9 violation. +//! +//! Buffering is unbounded toward a slow peer (the BEAM `busy_dist_port` +//! shape); backpressure is out of c9's scope and noted here rather than +//! silently absent. +//! +//! ## Inbound — the ONE resolution seam (RFC v2) +//! +//! Every wire-name → local-pid resolution goes through [`deliver_named`], +//! and nothing else: the conn actor hands it the three fields of a +//! `SendNamed` and gets back a verdict. It checks the exposed set first (an +//! unexposed name is unreachable — the gun's safety), then the type hash +//! against what the name was exposed with, then resolves the name through +//! the registry and delivers via c8's [`decode_deliver`]. When an owned-name +//! table lands beside the `&'static str` registry, it slots in here without +//! touching call sites. Module privacy enforces the funnel: the exposed and +//! outbound tables are `pub(crate)`, and no other module resolves names for +//! the wire. +//! +//! Refusals are silent to the sender by design (§3: send failure reflects +//! local knowledge only); they are observable locally as the returned +//! [`InboundVerdict`], which the conn actor may log or count. + +use std::collections::HashMap; +use std::marker::PhantomData; + +use crate::channel::Sender; +use crate::cluster::envelope::{encode_payload, Frame, PayloadError}; +use crate::cluster::expose::{decode_deliver, exposed_hash, type_hash, DeliverError}; +use crate::pid::Name; +use crate::registry::whereis; +use crate::scheduler::with_runtime; + +/// A name on a specific remote node: `(node_name, Name)`. Sendable via +/// [`send`]; typed, so the payload is `M` and the wire hash is +/// [`type_hash::()`](type_hash). +pub struct RemoteName { + node: String, + name: Name, + _marker: PhantomData M>, +} + +impl RemoteName { + pub fn new(node: impl Into, name: Name) -> Self { + RemoteName { + node: node.into(), + name, + _marker: PhantomData, + } + } + pub fn node(&self) -> &str { + &self.node + } + pub fn name(&self) -> Name { + self.name + } +} + +impl Clone for RemoteName { + fn clone(&self) -> Self { + RemoteName { + node: self.node.clone(), + name: self.name, + _marker: PhantomData, + } + } +} + +impl std::fmt::Debug for RemoteName { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "{}@{}", self.name.as_str(), self.node) + } +} + +/// Why a remote send did not leave this node. Local knowledge only. +#[derive(Debug)] +pub enum RemoteSendError { + /// No live connection to that node right now (never connected, or gone + /// and not yet re-dialed). The message is handed back. + NotConnected(M), + /// The payload did not serialize. + Encode(M, PayloadError), +} + +impl RemoteSendError { + pub fn into_inner(self) -> M { + match self { + RemoteSendError::NotConnected(m) | RemoteSendError::Encode(m, _) => m, + } + } +} + +/// The outbound table, one per runtime (a `RuntimeInner` field). +pub(crate) struct Outbound { + by_node: HashMap>, +} + +impl Outbound { + pub(crate) fn new() -> Self { + Outbound { + by_node: HashMap::new(), + } + } +} + +/// Manager-only: bind `node`'s outbound channel. Called inside `Register`. +pub(crate) fn bind_outbound(node: &str, tx: Sender) { + with_runtime(|inner| { + inner.outbound.lock().by_node.insert(node.to_string(), tx); + }); +} + +/// Manager-only: unbind `node`'s outbound channel. Called on `Disconnect`, +/// reap, and manager shutdown. Dropping the sender is what closes the conn +/// actor's outbound arm — but that arm's closure is NOT a stop signal (the +/// cmd channel is, per D9); the actor simply stops selecting on it. +pub(crate) fn unbind_outbound(node: &str) { + with_runtime(|inner| { + inner.outbound.lock().by_node.remove(node); + }); +} + +/// Send `msg` to `target`. `Ok(())` = handed to the connection's inbox, and +/// nothing more — see the module docs. Must run inside +/// [`run`](crate::run). +pub fn send(target: RemoteName, msg: M) -> Result<(), RemoteSendError> +where + M: serde::Serialize + Send + 'static, +{ + let payload = match encode_payload(&msg) { + Ok(p) => p, + Err(e) => return Err(RemoteSendError::Encode(msg, e)), + }; + let frame = Frame::SendNamed { + name: target.name.as_str().to_string(), + type_hash: type_hash::(), + payload, + }; + match hand_to_connection(&target.node, frame) { + Ok(()) => Ok(()), + Err(NotConnected) => Err(RemoteSendError::NotConnected(msg)), + } +} + +/// No live connection to the named node — the payload-free form of +/// [`RemoteSendError::NotConnected`], for the raw path. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct NotConnected; + +/// The untyped escape hatch: send pre-encoded `payload` under an explicit +/// `type_hash`. Exists so tests (and future codecs) can put deliberately +/// wrong frames on the wire; the typed [`send`] cannot express a hash/type +/// mismatch, by design. Same `Ok` semantics as [`send`]. +pub fn send_remote_raw( + node: &str, + name: &str, + type_hash: u64, + payload: &[u8], +) -> Result<(), NotConnected> { + hand_to_connection( + node, + Frame::SendNamed { + name: name.to_string(), + type_hash, + payload: payload.to_vec(), + }, + ) +} + +/// One lookup, one send. Clone the sender out under the lock and send +/// outside it (a channel send can unpark the conn actor). +fn hand_to_connection(node: &str, frame: Frame) -> Result<(), NotConnected> { + let tx = with_runtime(|inner| inner.outbound.lock().by_node.get(node).cloned()); + match tx { + Some(tx) => tx.send(frame).map_err(|_| NotConnected), + None => Err(NotConnected), + } +} + +/// What the inbound seam did with a `SendNamed`. Local observability only; +/// nothing goes back on the wire (RFC §3). +#[derive(Debug)] +pub enum InboundVerdict { + /// Decoded and handed to the name's holder. + Delivered, + /// The name is not in this node's exposed set. + NotExposed, + /// The frame's hash is not the hash the name was exposed with. + HashMismatch { expected: u64, got: u64 }, + /// Exposed, but no live holder right now (unbound, or its holder died + /// and the binding is being pruned). + Unresolved, + /// Resolved, but the delivery half refused it (decode failure, or the + /// holder's channel does not accept the exposed type — a local + /// re-registration under a different type; never a misroute). + Refused(DeliverError), +} + +/// THE inbound resolution seam: exposed-set check → hash check → registry +/// resolution → c8 delivery. See the module docs. Must run inside +/// [`run`](crate::run) — the conn actor's context. +pub fn deliver_named(name: &str, type_hash: u64, payload: &[u8]) -> InboundVerdict { + let Some(expected) = exposed_hash(name) else { + return InboundVerdict::NotExposed; + }; + if expected != type_hash { + return InboundVerdict::HashMismatch { + expected, + got: type_hash, + }; + } + let Some(pid) = whereis(name) else { + return InboundVerdict::Unresolved; + }; + match decode_deliver(type_hash, pid, payload) { + Ok(()) => InboundVerdict::Delivered, + Err(e) => InboundVerdict::Refused(e), + } +} diff --git a/src/registry.rs b/src/registry.rs index 7c20b1d..76be1db 100644 --- a/src/registry.rs +++ b/src/registry.rs @@ -215,6 +215,7 @@ impl std::error::Error for SendError {} trait ErasedSender: Send { fn as_any(&self) -> &dyn Any; fn queued_len(&self) -> usize; + fn receiver_alive(&self) -> bool; } impl ErasedSender for Sender { @@ -224,6 +225,9 @@ impl ErasedSender for Sender { fn queued_len(&self) -> usize { Sender::queued_len(self) } + fn receiver_alive(&self) -> bool { + Sender::receiver_alive(self) + } } /// One typed channel of an actor, type-erased. Concretely a `Sender` filed @@ -442,6 +446,18 @@ pub(crate) fn register_with( /// name) and [`install`] (which does not). A leftover mailbox at this slot /// index from a dead prior incarnation (pid mismatch) is replaced wholesale. /// Caller holds the registry lock and has established that `me` is live. +/// +/// **One channel per message type per actor.** Publishing a second `M` +/// channel on the same live actor replaces the first — and if the first's +/// receiver is still alive, that replacement drops its last sender, closing +/// it, and any `recv`/`select` on it then returns "closed" immediately and +/// forever: a silent hot loop that starves the scheduler. That is never +/// intended, so it panics here (found the hard way in RFC 010 c9, where two +/// `Name`s registered on one actor did exactly this). Replacing a +/// channel whose receiver is already gone is fine (an actor re-registering +/// after dropping its old inbox) and stays silent. To hold two names of the +/// same type, register them from two actors, or bind both names to one +/// cloned sender. fn publish_channel(reg: &mut Registry, me: Pid, tx: Sender) { let mb = reg .by_index @@ -450,6 +466,16 @@ fn publish_channel(reg: &mut Registry, me: Pid, tx: Sender if mb.pid != me { *mb = Mailbox::new(me); } + if let Some(existing) = mb.channels.get(&TypeId::of::()) { + assert!( + !existing.sender.receiver_alive() || same_channel::(existing, &tx), + "smarm: actor {me:?} already publishes a live channel for message type `{}`; \ + a second one would replace and CLOSE the first (its receiver would then \ + read as closed forever). Register the second name from another actor, or \ + bind both names to a clone of the same sender.", + type_name::() + ); + } mb.channels.insert( TypeId::of::(), Channel { @@ -459,6 +485,17 @@ fn publish_channel(reg: &mut Registry, me: Pid, tx: Sender ); } +/// True if `existing` and `tx` are senders of the very same channel (a +/// cloned sender bound under a second name is the sanctioned way to hold two +/// names of one type on one actor). +fn same_channel(existing: &Channel, tx: &Sender) -> bool { + existing + .sender + .as_any() + .downcast_ref::>() + .is_some_and(|old| old.same_channel(tx)) +} + /// Publish the current actor's `Sender` into its mailbox **without** /// binding a name, and hand back the typed [`Pid`] that addresses this /// actor directly. diff --git a/src/runtime.rs b/src/runtime.rs index d25e167..911114c 100644 --- a/src/runtime.rs +++ b/src/runtime.rs @@ -968,6 +968,11 @@ pub(crate) struct RuntimeInner { /// takes `registry`, never this). cfg-gated: zero-cost-when-off (c1). #[cfg(feature = "cluster")] pub(crate) exposure: RawMutex, + /// RFC 010 c9: the outbound table (node name → the connection's + /// dedicated `Sender`), manager-maintained. Leaf; the send happens + /// outside the lock. cfg-gated like `exposure`. + #[cfg(feature = "cluster")] + pub(crate) outbound: RawMutex, /// Recycled stacks waiting to be reused by the next spawn. pub(crate) stack_pool: RawMutex>, /// Maximum number of stacks to retain in the pool. @@ -1030,6 +1035,8 @@ impl RuntimeInner { process_groups: RawMutex::new(crate::pg::ProcessGroups::new()), #[cfg(feature = "cluster")] exposure: RawMutex::new(crate::cluster::expose::ExposureState::new()), + #[cfg(feature = "cluster")] + outbound: RawMutex::new(crate::cluster::remote::Outbound::new()), stack_pool: RawMutex::new(Vec::new()), stack_pool_cap, stack_reserve: crate::stack::round_to_pages(stack_reserve), diff --git a/tests/cluster_remote_send.rs b/tests/cluster_remote_send.rs new file mode 100644 index 0000000..2f6e1e1 --- /dev/null +++ b/tests/cluster_remote_send.rs @@ -0,0 +1,220 @@ +//! RFC 010 c9 — remote `Name` sends: the outbound seam and the single +//! inbound name-resolution seam, cross-process. +//! +//! Two node processes each run the integrated `cluster::start`. The +//! *receiver* registers a `String` inbox under a name and exposes it (and +//! registers a second name it does NOT expose); the *sender* waits for +//! `node_up`, then sends. Facts cross as stdout lines: `LISTENING `, +//! `MEMBER-UP `, `GOT `, `SEND-RESULT `. +//! Roles park forever afterwards (retractable-state trap); the parent +//! SIGKILLs via `Drop`. +//! +//! What is asserted at each end (roadmap-binding): +//! - cross-node name-send delivers the payload; +//! - an unexposed name is unreachable — the receiver's inbox stays empty +//! even though the name IS registered locally; +//! - a wrong type hash is a decode failure at the receiver, never a +//! misroute — the `String` inbox does not see a `u64` delivered under a +//! made-up hash, nor a `u64` under `u64`'s hash; +//! - a send to a disconnected (never-connected) node fails locally with +//! `NotConnected`, and `Ok(())` means only "handed to the transport". +//! +//! Timing note for the "stays empty" assertions: they are proven by +//! ORDERING, not by waiting — the sender emits the negative-case frames +//! BEFORE the positive one on the same connection (in-order stream), so when +//! the receiver has seen the positive payload, the negatives have already +//! been processed and refused. No sleep-and-hope. +#![cfg(feature = "cluster")] + +mod common; + +use common::{maybe_child, spawn_node, Node}; +use smarm::cluster::envelope::NodeMeta; +use smarm::cluster::expose::expose; +use smarm::cluster::membership::{subscribe, NodeEvent}; +use smarm::cluster::remote::{send_remote_raw, RemoteName, RemoteSendError}; +use smarm::cluster::{start, Config, StaticSeeds}; +use smarm::{channel, register, Name}; +use std::time::Duration; + +const ROLES: &[(&str, fn())] = &[("receiver", role_receiver), ("sender", role_sender)]; + +const INBOX: Name = Name::new("c9.inbox"); +const HIDDEN: Name = Name::new("c9.hidden"); + +fn base_config(name: &str, seeds: Vec<(String, String)>) -> Config { + Config { + node_name: name.to_string(), + meta: NodeMeta { + role: "c9".to_string(), + region: "local".to_string(), + }, + listen_addr: "127.0.0.1:0".to_string(), + strategy: Box::new(StaticSeeds::new(seeds)), + } +} + +/// Receiver: register + expose INBOX; register HIDDEN unexposed **in a +/// separate actor** (one actor holds one channel per message type — a +/// second `register` of the same `M` on one actor silently replaces the +/// first, closing it); print every payload that lands in either. +fn role_receiver() { + smarm::run(|| { + let cluster = start(base_config("recv", vec![])).expect("listener binds"); + println!("LISTENING {}", cluster.local_addr()); + + // HIDDEN's holder: its own actor, so its String channel does not + // displace INBOX's on the root actor. + let (hidden_ready_tx, hidden_ready_rx) = channel::<()>(); + smarm::spawn(move || { + let (hid_tx, hid_rx) = channel::(); + register(HIDDEN, hid_tx).unwrap(); + hidden_ready_tx.send(()).unwrap(); + loop { + match hid_rx.recv() { + Ok(s) => println!("GOT-HIDDEN {s}"), + Err(_) => break, + } + } + }); + hidden_ready_rx.recv().unwrap(); + + let (in_tx, in_rx) = channel::(); + register(INBOX, in_tx).unwrap(); + expose(INBOX); + println!("READY"); + loop { + match in_rx.recv() { + Ok(s) => println!("GOT {s}"), + Err(_) => break, + } + } + loop { + smarm::sleep(Duration::from_secs(3600)); + } + }); +} + +/// Sender: connect to recv, wait for node_up, then in this ORDER on the one +/// connection: hidden-name send, wrong-hash sends (two flavours), then the +/// positive send. Plus a send to a node that is not connected at all. +fn role_sender() { + let recv_addr = std::env::var("SMARM_RECV_ADDR").expect("SMARM_RECV_ADDR"); + smarm::run(move || { + let _cluster = start(base_config("send", vec![("recv".to_string(), recv_addr)])) + .expect("listener binds"); + let events = subscribe().expect("manager is up"); + loop { + match events.rx.recv() { + Ok(NodeEvent::NodeUp(info)) if info.name == "recv" => break, + Ok(_) => continue, + Err(_) => panic!("manager gone"), + } + } + println!("MEMBER-UP recv"); + + // Not connected: purely local knowledge, no frame leaves. + let ghost: RemoteName = RemoteName::new("nowhere", INBOX); + let r = smarm::cluster::remote::send(ghost, "lost".to_string()); + println!( + "SEND-RESULT not-connected {}", + match r { + Err(RemoteSendError::NotConnected(_)) => "NotConnected", + Ok(()) => "Ok", + Err(_) => "OtherErr", + } + ); + + // Unexposed name at the peer: the frame goes (local knowledge can't + // know the peer's exposed set) and the peer refuses it. + let hidden: RemoteName = RemoteName::new("recv", HIDDEN); + let r = smarm::cluster::remote::send(hidden, "should not land".to_string()); + println!( + "SEND-RESULT hidden {}", + if r.is_ok() { "Ok" } else { "Err" } + ); + + // Wrong hash, two flavours: (a) a u64 payload under a made-up hash + // (unknown type at the peer); (b) a u64 payload under u64's real + // hash against a String-typed name (decoder known, wrong channel). + // Both are raw sends — the typed API cannot express them, by design. + let bogus = 0xdead_beef_u64; + let r = send_remote_raw( + "recv", + "c9.inbox", + bogus, + &smarm::cluster::envelope::encode_payload(&7u64).unwrap(), + ); + println!( + "SEND-RESULT wrong-hash-unknown {}", + if r.is_ok() { "Ok" } else { "Err" } + ); + let r = send_remote_raw( + "recv", + "c9.inbox", + smarm::cluster::expose::type_hash::(), + &smarm::cluster::envelope::encode_payload(&7u64).unwrap(), + ); + println!( + "SEND-RESULT wrong-hash-known {}", + if r.is_ok() { "Ok" } else { "Err" } + ); + + // Positive: last on the stream, so its arrival proves the negatives + // were already processed. + let inbox: RemoteName = RemoteName::new("recv", INBOX); + let r = smarm::cluster::remote::send(inbox, "hello from send".to_string()); + println!( + "SEND-RESULT positive {}", + if r.is_ok() { "Ok" } else { "Err" } + ); + + loop { + smarm::sleep(Duration::from_secs(3600)); + } + }); +} + +fn wait_send_result(node: &mut Node, case: &str) -> String { + let prefix = format!("SEND-RESULT {case} "); + let line = node.wait_line(&prefix, |l| l.starts_with(&prefix)); + line[prefix.len()..].to_string() +} + +#[test] +fn remote_name_send_delivers_and_refusals_never_misroute() { + maybe_child(ROLES); + + let mut recv = spawn_node("receiver", &[]); + let addr = recv.wait_listening(); + recv.wait_line("READY", |l| l == "READY"); + + let mut send = spawn_node("sender", &[("SMARM_RECV_ADDR", &addr)]); + send.wait_line("MEMBER-UP recv", |l| l == "MEMBER-UP recv"); + + // Local-knowledge-only failure for an unknown node. + assert_eq!(wait_send_result(&mut send, "not-connected"), "NotConnected"); + // Every frame-bearing send is Ok — Ok means "handed to the transport", + // nothing about what the peer does with it (RFC §3, documented here). + assert_eq!(wait_send_result(&mut send, "hidden"), "Ok"); + assert_eq!(wait_send_result(&mut send, "wrong-hash-unknown"), "Ok"); + assert_eq!(wait_send_result(&mut send, "wrong-hash-known"), "Ok"); + assert_eq!(wait_send_result(&mut send, "positive"), "Ok"); + + // The positive payload lands... + recv.wait_line("GOT hello from send", |l| l == "GOT hello from send"); + // ...and, by stream ordering, every negative before it was refused: no + // GOT for the wrong-hash frames, no GOT-HIDDEN at all. The transcript + // up to this point is the proof. + let transcript = recv.transcript(); + let gots: Vec<&str> = transcript + .iter() + .map(|s| s.as_str()) + .filter(|l| l.starts_with("GOT")) + .collect(); + assert_eq!( + gots, + ["GOT hello from send"], + "exactly one delivery, the exposed one" + ); +} diff --git a/tests/common/mod.rs b/tests/common/mod.rs index ba522e7..36ae39f 100644 --- a/tests/common/mod.rs +++ b/tests/common/mod.rs @@ -133,6 +133,13 @@ impl Node { } } + /// Every stdout/stderr line seen so far, in arrival order. For + /// ordering-proof assertions ("by the time X arrived, Y had not"). + #[allow(dead_code)] + pub fn transcript(&self) -> &[String] { + &self.transcript + } + /// Wait until a stdout line satisfies `pred`; return it. Panics with the /// full transcript after [`WAIT`]. `what` names the expectation in the /// panic message. @@ -149,6 +156,9 @@ impl Node { } } Err(RecvTimeoutError::Timeout) => { + // Pull in whatever stderr arrived since the last drain, + // so a role's eprintln! diagnostics survive into the dump. + self.drain_stderr(); panic!( "node {:?}: timed out waiting for {what} after {WAIT:?}; transcript:\n{}", self.role, @@ -156,6 +166,7 @@ impl Node { ); } Err(RecvTimeoutError::Disconnected) => { + self.drain_stderr(); panic!( "node {:?}: output closed while waiting for {what}; transcript:\n{}", self.role, diff --git a/tests/registry.rs b/tests/registry.rs index d1e23f3..2ca251b 100644 --- a/tests/registry.rs +++ b/tests/registry.rs @@ -258,3 +258,46 @@ fn send_dyn_to_dead_pid_is_dead() { assert!(matches!(send_dyn::(p, 1u64), Err(SendError::Dead(_)))); }); } + +// --- one channel per message type per actor ----------------------------------- + +/// Registering a second name of the same message type on one actor, with a +/// *fresh* channel, would silently replace and close the first — so it +/// panics (found in RFC 010 c9). The sanctioned shapes stay quiet: bind both +/// names to a clone of one sender, or use two actors. +#[test] +#[should_panic(expected = "already publishes a live channel")] +fn second_live_channel_of_same_type_on_one_actor_panics() { + run(|| { + let (tx1, _rx1) = channel::(); + let (tx2, _rx2) = channel::(); + register(Name::::new("dup-a"), tx1).unwrap(); + register(Name::::new("dup-b"), tx2).unwrap(); // panics + }); +} + +#[test] +fn two_names_on_one_cloned_sender_is_fine() { + run(|| { + let (tx, rx) = channel::(); + register(Name::::new("twin-a"), tx.clone()).unwrap(); + register(Name::::new("twin-b"), tx).unwrap(); + send(Name::::new("twin-a"), 1).unwrap(); + send(Name::::new("twin-b"), 2).unwrap(); + assert_eq!(rx.recv().unwrap(), 1); + assert_eq!(rx.recv().unwrap(), 2); + }); +} + +#[test] +fn replacing_a_channel_whose_receiver_is_gone_is_fine() { + run(|| { + let (tx1, rx1) = channel::(); + register(Name::::new("reborn"), tx1).unwrap(); + drop(rx1); // old inbox gone: replacement is the honest thing to do + let (tx2, rx2) = channel::(); + register(Name::::new("reborn-2"), tx2).unwrap(); + send(Name::::new("reborn-2"), 9).unwrap(); + assert_eq!(rx2.recv().unwrap(), 9); + }); +} From 4c0e42152fc136508d2b5bac6f79dde640d51238 Mon Sep 17 00:00:00 2001 From: claude-asm-audit Date: Tue, 18 Aug 2026 16:00:00 +0200 Subject: [PATCH 14/14] =?UTF-8?q?feat(cluster):=20RFC=20010=20c10=E2=80=93?= =?UTF-8?q?c16,=20follow-ups=20and=20Phase=206=20(squash=20of=2016ef583..d?= =?UTF-8?q?9c62a8)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Tree snapshot of d9c62a8 (2026-08-18). The 20 source commits between 16ef583 (c9) and d9c62a8 were never pushed and the clone that held them was lost; this commit carries their combined tree verbatim so the build history stays auditable from the c1–c9 commits below it. Original hashes as recorded in the session handoff: c10 f03e94d pid targeting + auto-serialization (RemotePid, D14 name on the wire); Phase 3 gate c11 7ef4bad DownReason::Disconnected, wire tag 5 c12 d124162 remote monitors (Monitor/Demonitor/Down frames) c13 9de967b connection-loss synthesis (A+B: Monitors::teardown + unread-command Disconnected); Phase 4 gate c14 7e822b7 eager pg eviction (reaper actor, ReaperInboxes) dbe1a22 InboundVerdict::label(), trace::Event::ClusterInbound 31a9877 tests/channel.rs monitor-churn target gated on `go` 653559e Discovery::Withdrawn{name, addr} c15 b41d76e distributed pg: Sync on NodeUp, Join/Leave broadcast, NodeDown sweep, members_all; PgMsg wire type c16 fafa881 pick_any / dispatch_any; Phase 5 complete Phase 6 Tier A: 195c73e p4 NodeEvent::NodeDown(NodeInfo) 48fd766 p1 connector Candidate{name, addr, state} ce8cf99 p2+p7 conn.rs select arms as Vec; Outbound::Drained bf24988 p6 RemotePid::from_local -> Option 9ae0380 p3 PeerStanding{Free, Claimed, Dialing} c7d62a1 p11 cluster::Timing knobs, threaded by value 46f171d p11 cluster_disconnect un-ignored on SMARM_FAST_TIMING Phase 6 Tier B: 391a9ae p5 cluster::RemoteDownReason{Local, Disconnected}; DownReason::Disconnected removed from core 7ddd908 p9 pg ctl channel unconditional, one cfg seam at spawn d9c62a8 PeerNameMismatch parks the candidate; ClusterDial trace Verified at d9c62a8: default 361/0, cluster 448/0, clippy --lib on default / cluster / cluster+smarm-trace, fmt, 10x flake on cluster_dial_mismatch, 5x on cluster_pg. --- src/cluster.rs | 59 ++- src/cluster/conn.rs | 431 +++++++++++++++-- src/cluster/connect.rs | 76 +-- src/cluster/connector.rs | 207 +++++--- src/cluster/discovery.rs | 13 +- src/cluster/envelope.rs | 60 ++- src/cluster/handshake.rs | 27 +- src/cluster/manager.rs | 37 +- src/cluster/membership.rs | 6 +- src/cluster/pg.rs | 546 +++++++++++++++++++++ src/cluster/remote.rs | 615 +++++++++++++++++++++++- src/monitor.rs | 114 +++-- src/pg.rs | 711 ++++++++++++++++++---------- src/pid.rs | 41 ++ src/runtime.rs | 3 + src/trace.rs | 9 + tests/channel.rs | 9 +- tests/cluster_conn_lifecycle.rs | 7 +- tests/cluster_conn_liveness.rs | 8 +- tests/cluster_connect.rs | 46 +- tests/cluster_dial_mismatch.rs | 115 +++++ tests/cluster_disconnect.rs | 379 +++++++++++++++ tests/cluster_discovery_withdraw.rs | 161 +++++++ tests/cluster_envelope.rs | 9 +- tests/cluster_handshake.rs | 29 +- tests/cluster_membership.rs | 45 +- tests/cluster_mesh.rs | 12 +- tests/cluster_monitor.rs | 359 ++++++++++++++ tests/cluster_pg.rs | 254 ++++++++++ tests/cluster_pid_send.rs | 356 ++++++++++++++ tests/cluster_remote_send.rs | 3 +- tests/pg.rs | 33 +- 32 files changed, 4202 insertions(+), 578 deletions(-) create mode 100644 src/cluster/pg.rs create mode 100644 tests/cluster_dial_mismatch.rs create mode 100644 tests/cluster_disconnect.rs create mode 100644 tests/cluster_discovery_withdraw.rs create mode 100644 tests/cluster_monitor.rs create mode 100644 tests/cluster_pg.rs create mode 100644 tests/cluster_pid_send.rs diff --git a/src/cluster.rs b/src/cluster.rs index a56cc6c..635c611 100644 --- a/src/cluster.rs +++ b/src/cluster.rs @@ -17,6 +17,7 @@ pub mod expose; pub mod handshake; pub mod manager; pub mod membership; +pub mod pg; pub mod remote; pub mod transport; @@ -38,10 +39,15 @@ pub use conn::{spawn_established, ConnHandle}; pub use connect::{dial, spawn_acceptor, AcceptorHandle}; pub use connector::{spawn_connector, ConnectorHandle}; pub use discovery::{Discovery, StaticSeeds, Strategy}; +pub use envelope::RemoteDownReason; pub use expose::{expose, expose_type, type_hash, DeliverError}; pub use manager::{Manager, MANAGER}; pub use membership::{subscribe, view, MembershipEvents, NodeEvent, NodeInfo}; -pub use remote::{NotConnected, RemoteName, RemoteSendError}; +pub use pg::{dispatch_any, members_all, pick_any, DispatchAnyError, GroupMember, PgMsg, PG_NAME}; +pub use remote::{ + demonitor_remote, monitor_remote, send_to_remote, NotConnected, RemoteDown, RemoteMonitor, + RemoteName, RemotePid, RemoteSendError, ToRemoteError, +}; /// c6d — the derived build hash for [`handshake::LocalNode::build_hash`]: /// two builds may mesh only when this matches, and it is a pure function of @@ -80,6 +86,41 @@ const fn fold_u32(mut h: u64, v: u32) -> u64 { h } +/// The control-plane timing knobs, all with today's fixed values as +/// defaults ([`Timing::default`]). One struct threaded explicitly to the +/// acceptor, the dial path, every connection actor and the connector — no +/// ambient state, so a test can run a fast mesh without touching globals. +/// Every node in a mesh should agree on `heartbeat_interval` < +/// `liveness_timeout`; nothing enforces it. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct Timing { + /// Idle-connection heartbeat pace. Default [`conn::HEARTBEAT_INTERVAL`]. + pub heartbeat_interval: Duration, + /// Inbound silence that tears a connection down. Default + /// [`conn::LIVENESS_TIMEOUT`]. + pub liveness_timeout: Duration, + /// Per-frame handshake deadline on the accept/dial path. Default + /// [`connect::HANDSHAKE_TIMEOUT`]. + pub handshake_timeout: Duration, + /// Connector redial delay after the first failure. Default + /// [`connector::INITIAL_BACKOFF`]. + pub initial_backoff: Duration, + /// Connector redial delay cap. Default [`connector::MAX_BACKOFF`]. + pub max_backoff: Duration, +} + +impl Default for Timing { + fn default() -> Self { + Timing { + heartbeat_interval: conn::HEARTBEAT_INTERVAL, + liveness_timeout: conn::LIVENESS_TIMEOUT, + handshake_timeout: connect::HANDSHAKE_TIMEOUT, + initial_backoff: connector::INITIAL_BACKOFF, + max_backoff: connector::MAX_BACKOFF, + } + } +} + /// How to run this node: its identity and how it finds peers. pub struct Config { /// This node's claimed name — the mesh-wide identity peers dial by and @@ -92,6 +133,9 @@ pub struct Config { pub listen_addr: String, /// The peer-discovery strategy — [`StaticSeeds`] until richer ones land. pub strategy: Box, + /// Heartbeat / liveness / handshake / backoff knobs; [`Timing::default`] + /// is the shipping configuration. + pub timing: Timing, } /// A running cluster node: the supervised [`Manager`], the acceptor over the @@ -155,9 +199,18 @@ pub fn start(config: Config) -> io::Result { build_hash: BUILD_HASH, meta: config.meta, }; + // The wire identity serialized pids are stamped with (c10). + remote::set_local_identity(&local.node_name, local.incarnation); + // The pg actor (Phase 5): subscribes membership, owns the "pg" name. + pg::attach_cluster(); let listener = TcpTransport.listen(&config.listen_addr)?; - let acceptor = spawn_acceptor(listener, local.clone()); - let connector = spawn_connector(Box::new(TcpTransport), local.clone(), config.strategy); + let acceptor = spawn_acceptor(listener, local.clone(), config.timing); + let connector = spawn_connector( + Box::new(TcpTransport), + local.clone(), + config.strategy, + config.timing, + ); Ok(Cluster { _sup: sup, acceptor, diff --git a/src/cluster/conn.rs b/src/cluster/conn.rs index a9abad4..76a33ba 100644 --- a/src/cluster/conn.rs +++ b/src/cluster/conn.rs @@ -27,21 +27,52 @@ //! drained onto the wire in the same loop — and inbound *interpretation*: //! `SendNamed` goes to the one resolution seam, //! [`remote::deliver_named`](crate::cluster::remote::deliver_named). -//! Frames the connection actor has no business with yet (`Send`, `Monitor`, -//! …: c10/c11) are consumed for liveness and otherwise ignored. The outbound +//! `Send` goes to the pid seam (c10). The outbound //! sender is a separate channel from `cmd_tx` on purpose: closing it is not //! a stop signal — lifetime authority stays with the [`ConnHandle`] (D9). +//! +//! c12 adds the monitor plane, and it lives *here* on purpose. Two tables, +//! both owned by this actor and dying with the connection: +//! +//! - **outstanding** — monitors *this* node holds on actors at the peer: +//! `monitor_id → (target, Sender)`. Fed by +//! [`MonCmd`](crate::cluster::remote::MonCmd) from `monitor_remote`; the +//! actor records the id and *then* emits the `Monitor` frame, so a `Down` +//! frame can never race an entry that isn't there yet. An inbound `Down` +//! removes the entry and delivers. +//! - **watched** — monitors the *peer* holds on actors here: `monitor_id → +//! local Monitor`. An inbound `Monitor` is admitted only for a pid that +//! was exposed or crossed the wire (`is_watchable`, D12): a corpse answers +//! with its recorded terminal reason (RFC §6), an unwatchable or unknown +//! pid with `NoProc` — indistinguishable from dead, so nothing leaks. A +//! live watchable pid gets a local monitor whose `rx` is one more arm of +//! the select; its `Down` goes back as a frame. +//! +//! Because both tables are actor state, connection loss (c13) needs no +//! second bookkeeping owner: this actor's exit is the one place that knows +//! every monitor the link was carrying. `Monitors::teardown` runs on every +//! exit path and answers each outstanding monitor with `Disconnected` — +//! the roadmap's "partition vs. death" contrast: an actor that dies sends +//! its true reason over the link, a link that dies says only that. + +use std::collections::HashMap; use std::time::{Duration, Instant}; use crate::channel::{channel, try_select_timeout, Receiver, Selectable, Sender}; -use crate::cluster::envelope::Frame; +use crate::cluster::envelope::{Frame, RemoteDownReason}; use crate::cluster::handshake::Peer; use crate::cluster::manager::{Call, Registered, Reply, MANAGER}; -use crate::cluster::remote::deliver_named; +use crate::cluster::remote::{ + deliver_named, deliver_to_pid, InboundVerdict, MonCmd, RemoteDown, RemotePid, +}; use crate::cluster::transport::FramedConn; +use crate::cluster::Timing; use crate::gen_server; -use crate::pid::Pid; +use crate::monitor::{ + demonitor, is_watchable, monitor, terminal_reason, DownReason, Monitor, MonitorId, +}; +use crate::pid::{Erased, Pid}; use crate::scheduler::spawn; /// Commands to a running connection actor. @@ -57,11 +88,11 @@ enum Cmd { /// to establish it. pub struct ConnHandle { cmd_tx: Sender, - /// The connection's dedicated outbound inbox. The manager moves this - /// into the outbound table on `Register` (see - /// [`take_outbound`](ConnHandle::take_outbound)); a `Duplicate` verdict - /// drops it with the handle. - out_tx: Option>, + /// The connection's dedicated outbound inboxes — frames and monitor + /// commands. The manager moves them into the outbound table on + /// `Register` (see [`take_outbound`](ConnHandle::take_outbound)); a + /// `Duplicate` verdict drops them with the handle. + out_tx: Option<(Sender, Sender)>, } impl std::fmt::Debug for ConnHandle { @@ -78,9 +109,9 @@ impl ConnHandle { let _ = self.cmd_tx.send(Cmd::Shutdown); } - /// Manager-only: take the outbound sender to bind into the outbound + /// Manager-only: take the outbound senders to bind into the outbound /// table. Once, at registration. - pub(crate) fn take_outbound(&mut self) -> Option> { + pub(crate) fn take_outbound(&mut self) -> Option<(Sender, Sender)> { self.out_tx.take() } } @@ -96,11 +127,16 @@ pub struct RegisterRefused; /// actor's [`ConnHandle`]; the caller gets only the [`Pid`], because /// connection lifetime belongs to the table and not to the establishing /// actor. A refusal has already stopped the actor and closed the socket. -pub fn spawn_established(framed: FramedConn, peer: Peer) -> Result { +pub fn spawn_established( + framed: FramedConn, + peer: Peer, + timing: Timing, +) -> Result { let (cmd_tx, cmd_rx) = channel(); let (out_tx, out_rx) = channel(); + let (mon_tx, mon_rx) = channel(); let reg_peer = peer.clone(); - let pid = spawn(move || run(framed, peer, cmd_rx, out_rx)).pid(); + let pid = spawn(move || run(framed, peer, timing, cmd_rx, out_rx, mon_rx)).pid(); match gen_server::call( MANAGER, Call::Register { @@ -108,7 +144,7 @@ pub fn spawn_established(framed: FramedConn, peer: Peer) -> Result Result, out_rx: Receiver) { +fn run( + mut framed: FramedConn, + _peer: Peer, + timing: Timing, + cmd_rx: Receiver, + out_rx: Receiver, + mon_rx: Receiver, +) { + let mut mons = Monitors::default(); match framed.readable_arm() { - Some(arm) => run_live(&mut framed, arm, &cmd_rx, &out_rx), + Some(arm) => run_live( + &mut framed, + arm, + timing, + &cmd_rx, + &out_rx, + &mon_rx, + &mut mons, + ), None => run_inert(&cmd_rx), } framed.close(); + mons.teardown(&mon_rx); +} + +/// The monitor plane's two tables (module docs). Owned by the actor. +#[derive(Default)] +struct Monitors { + /// Monitors this node holds on peer actors: id → (target, delivery). + outstanding: HashMap, Sender)>, + /// Monitors the peer holds on local actors: id → the local monitor. + watched: HashMap, +} + +impl Monitors { + /// The connection is gone, whatever the exit path (liveness expiry, + /// EOF, wire failure, commanded stop): release the peer's local + /// monitors, and answer every one of ours with `Disconnected` — nothing + /// more can be known about those actors. Commands still sitting in the + /// inbox are folded in first (a `Monitor` handed to us but never + /// processed gets its notice too; a `Demonitor` still cancels), so the + /// only registration that can miss this is one that lands after the + /// drain and before the inbox drops — the reader side backstops that + /// (`RemoteMonitor`). Entries leave the table as they are answered, and + /// this runs once per actor, so no monitor sees two notices. + fn teardown(&mut self, mon_rx: &Receiver) { + for (_, m) in self.watched.drain() { + let _ = demonitor(&m); + } + while let Ok(Some(cmd)) = mon_rx.try_recv() { + match cmd { + MonCmd::Monitor { id, target, tx } => { + self.outstanding.insert(id, (target, tx)); + } + MonCmd::Demonitor { id } => { + self.outstanding.remove(&id); + } + } + } + for (_, (pid, tx)) in self.outstanding.drain() { + let _ = tx.send(RemoteDown { + pid, + reason: RemoteDownReason::Disconnected, + }); + } + } + + /// Admit a peer's `Monitor` for local `(index, generation)`. Returns + /// the reason to answer with at once, or `None` if a live monitor was + /// installed. Corpse → recorded terminal reason (RFC §6, and only + /// watchable deaths are recorded); live watchable → monitor; anything + /// else → `NoProc`. The check-then-monitor race (dies in between) is + /// closed on the read side: a `NoProc` from a monitor we installed on a + /// live pid is upgraded through `terminal_reason` in `sweep_watched`. + fn admit(&mut self, id: MonitorId, index: u32, generation: u32) -> Option { + let pid = Pid::new(index, generation); + if let Some(reason) = terminal_reason(pid) { + return Some(reason); + } + if !is_watchable(pid) { + return Some(DownReason::NoProc); + } + let m = monitor(pid); + self.watched.insert(id, m); + None + } + + fn cancel(&mut self, id: MonitorId) { + if let Some(m) = self.watched.remove(&id) { + let _ = demonitor(&m); + } + } + + /// Collect every local `Down` that has arrived for a peer-held monitor. + fn sweep_watched(&mut self) -> Vec<(MonitorId, DownReason)> { + let mut fired = Vec::new(); + for (id, m) in self.watched.iter() { + if let Ok(Some(down)) = m.rx.try_recv() { + let reason = match down.reason { + DownReason::NoProc => terminal_reason(m.target).unwrap_or(DownReason::NoProc), + r => r, + }; + fired.push((*id, reason)); + } + } + for (id, _) in &fired { + self.watched.remove(id); + } + fired + } + + /// The peer reports a monitored actor down: deliver locally. + fn down(&mut self, id: MonitorId, reason: RemoteDownReason) { + if let Some((pid, tx)) = self.outstanding.remove(&id) { + let _ = tx.send(RemoteDown { pid, reason }); + } + } } /// The steady-state loop over an fd-backed connection: one @@ -148,15 +295,39 @@ fn run(mut framed: FramedConn, _peer: Peer, cmd_rx: Receiver, out_rx: Recei fn run_live( framed: &mut FramedConn, arm: crate::scheduler::FdArm, + timing: Timing, cmd_rx: &Receiver, out_rx: &Receiver, + mon_rx: &Receiver, + mons: &mut Monitors, ) { let mut next_hb = Instant::now(); - let mut live_until = Instant::now() + LIVENESS_TIMEOUT; - // The outbound sender lives in the manager's table and is dropped on - // unbind; after that this arm would wake forever, so it drops out of + let mut live_until = Instant::now() + timing.liveness_timeout; + // The outbound senders live in the manager's table and are dropped on + // unbind; after that these arms would wake forever, so they drop out of // the select (not a stop signal — see the module docs). let mut out_open = true; + let mut mon_open = true; + // Which wait each select arm stands for. Built in lockstep with the + // `Selectable` vector each iteration, so a wake is decoded by name and + // never by position. + enum Arm { + Cmd, + Fd, + Out, + Mon, + /// A peer-held local monitor (any of them: firing sweeps them all). + Watched, + } + fn push<'s>( + arms: &mut Vec<&'s dyn Selectable>, + what: &mut Vec, + s: &'s dyn Selectable, + a: Arm, + ) { + arms.push(s); + what.push(a); + } loop { let now = Instant::now(); if now >= live_until { @@ -166,33 +337,57 @@ fn run_live( if framed.send(&Frame::Heartbeat).is_err() { break; } - next_hb = now + HEARTBEAT_INTERVAL; + next_hb = now + timing.heartbeat_interval; } let wait = next_hb.min(live_until).saturating_duration_since(now); - // Arm indices: 0 cmd, 1 fd, 2 outbound (when open). - let mut arms: Vec<&dyn Selectable> = vec![cmd_rx, &arm]; + let mut arms: Vec<&dyn Selectable> = Vec::new(); + let mut what: Vec = Vec::new(); + push(&mut arms, &mut what, cmd_rx, Arm::Cmd); + push(&mut arms, &mut what, &arm, Arm::Fd); if out_open { - arms.push(out_rx); + push(&mut arms, &mut what, out_rx, Arm::Out); } - match try_select_timeout(&arms, wait) { - Ok(Some(0)) => { + if mon_open { + push(&mut arms, &mut what, mon_rx, Arm::Mon); + } + for m in mons.watched.values() { + push(&mut arms, &mut what, &m.rx, Arm::Watched); + } + match try_select_timeout(&arms, wait).map(|i| i.map(|i| &what[i])) { + Ok(Some(Arm::Cmd)) => { if should_stop(cmd_rx) { break; } } - Ok(Some(1)) => match pump_readable(framed) { + Ok(Some(Arm::Fd)) => match pump_readable(framed, mons) { Pump::Ended => break, Pump::Frames(n) => { if n > 0 { - live_until = Instant::now() + LIVENESS_TIMEOUT; + live_until = Instant::now() + timing.liveness_timeout; } } }, - Ok(Some(_)) => match pump_outbound(framed, out_rx) { - Outbound::Sent => {} + Ok(Some(Arm::Out)) => match pump_outbound(framed, out_rx) { + Outbound::Drained => {} Outbound::Closed => out_open = false, Outbound::WireFailed => break, }, + Ok(Some(Arm::Mon)) => match pump_moncmds(framed, mon_rx, mons) { + Outbound::Drained => {} + Outbound::Closed => mon_open = false, + Outbound::WireFailed => break, + }, + Ok(Some(Arm::Watched)) => { + for (id, reason) in mons.sweep_watched() { + let frame = Frame::Down { + monitor_id: id.0, + reason: reason.into(), + }; + if framed.send(&frame).is_err() { + return; + } + } + } // A deadline passed; the top of the loop acts on whichever. Ok(None) => {} // The fd arm failed to register — the connection is gone. @@ -201,9 +396,42 @@ fn run_live( } } -/// What one outbound wake yielded. +/// Drain the monitor-command inbox: record, then emit (module docs). +fn pump_moncmds( + framed: &mut FramedConn, + mon_rx: &Receiver, + mons: &mut Monitors, +) -> Outbound { + loop { + match mon_rx.try_recv() { + Ok(Some(MonCmd::Monitor { id, target, tx })) => { + let frame = Frame::Monitor { + monitor_id: id.0, + index: target.index(), + generation: target.generation(), + }; + mons.outstanding.insert(id, (target, tx)); + if framed.send(&frame).is_err() { + return Outbound::WireFailed; + } + } + Ok(Some(MonCmd::Demonitor { id })) => { + if mons.outstanding.remove(&id).is_some() + && framed.send(&Frame::Demonitor { monitor_id: id.0 }).is_err() + { + return Outbound::WireFailed; + } + } + Ok(None) => return Outbound::Drained, + Err(_) => return Outbound::Closed, + } + } +} + +/// What one outbound-side wake (frames or monitor commands) yielded. enum Outbound { - Sent, + /// Everything queued went onto the wire; the inbox is open and empty. + Drained, /// The manager unbound this connection's sender; nothing more will come. Closed, /// The socket refused a write: the connection is gone. @@ -219,7 +447,7 @@ fn pump_outbound(framed: &mut FramedConn, out_rx: &Receiver) -> Outbound return Outbound::WireFailed; } } - Ok(None) => return Outbound::Sent, + Ok(None) => return Outbound::Drained, Err(_) => return Outbound::Closed, } } @@ -261,16 +489,26 @@ enum Pump { Frames(usize), } +/// Surface an inbound verdict: one `smarm-trace` event, nothing else — it +/// is local knowledge (RFC §3). A no-op without the feature. +fn note_verdict(verdict: InboundVerdict) { + #[cfg(feature = "smarm-trace")] + crate::te!(crate::trace::Event::ClusterInbound(verdict.label())); + #[cfg(not(feature = "smarm-trace"))] + drop(verdict); +} + /// Consume one readable wake: exactly one socket read (which cannot block /// after a level-triggered readable indication), then drain every complete /// frame the buffer now holds. A blocking `recv` here would park the actor /// past its heartbeat and liveness deadlines whenever a frame arrives split. -/// Every consumed frame counts for liveness; `SendNamed` additionally goes -/// to the one inbound resolution seam. Its verdict is local knowledge only -/// — nothing goes back on the wire (RFC §3) — and is currently discarded -/// (a future trace hook is the place to surface it). Frames for later -/// chunks (`Send` c10, `Monitor`/`Down` c11) are consumed and ignored. -fn pump_readable(framed: &mut FramedConn) -> Pump { +/// Every consumed frame counts for liveness; `SendNamed` goes to the one +/// name-resolution seam and `Send` to the pid seam. Verdicts are local +/// knowledge only — nothing goes back on the wire (RFC §3) — and surface +/// as one `smarm-trace` `ClusterInbound` event each (zero cost off). +/// `Monitor`/`Demonitor`/`Down` go to the [`Monitors`] tables; a `Monitor` +/// that can be answered at once is answered inline. +fn pump_readable(framed: &mut FramedConn, mons: &mut Monitors) -> Pump { let eof = match framed.read_once() { Ok(n) => n == 0, Err(_) => return Pump::Ended, @@ -280,13 +518,43 @@ fn pump_readable(framed: &mut FramedConn) -> Pump { match framed.next_buffered() { Ok(Some(frame)) => { got += 1; - if let Frame::SendNamed { - name, - type_hash, - payload, - } = frame - { - let _verdict = deliver_named(&name, type_hash, &payload); + match frame { + Frame::SendNamed { + name, + type_hash, + payload, + } => { + note_verdict(deliver_named(&name, type_hash, &payload)); + } + Frame::Send { + index, + generation, + type_hash, + payload, + } => { + note_verdict(deliver_to_pid(index, generation, type_hash, &payload)); + } + Frame::Monitor { + monitor_id, + index, + generation, + } => { + let id = MonitorId(monitor_id); + if let Some(reason) = mons.admit(id, index, generation) { + let frame = Frame::Down { + monitor_id, + reason: reason.into(), + }; + if framed.send(&frame).is_err() { + return Pump::Ended; + } + } + } + Frame::Demonitor { monitor_id } => mons.cancel(MonitorId(monitor_id)), + Frame::Down { monitor_id, reason } => mons.down(MonitorId(monitor_id), reason), + // Heartbeat: liveness only. Handshake frames after + // establishment: ignored. + _ => {} } } Ok(None) => break, @@ -299,3 +567,64 @@ fn pump_readable(framed: &mut FramedConn) -> Pump { Pump::Frames(got) } } + +#[cfg(test)] +mod tests { + //! `Monitors::teardown` in isolation: the actor-side half of c13, pinned + //! separately because from the outside it is indistinguishable from the + //! read-side backstop in `RemoteMonitor` (both yield `Disconnected`). + use super::*; + use crate::pg::Incarnation; + + fn pid(index: u32) -> RemotePid { + RemotePid::from_parts("peer", Incarnation::new(1), index, 1) + } + + #[test] + fn teardown_answers_every_outstanding_and_unread_monitor_once() { + crate::run(|| { + let mut mons = Monitors::default(); + let (mon_tx, mon_rx) = channel::(); + + // Already registered. + let (tx1, rx1) = channel::(); + mons.outstanding.insert(MonitorId(1), (pid(1), tx1)); + // In the inbox, never processed. + let (tx2, rx2) = channel::(); + mon_tx + .send(MonCmd::Monitor { + id: MonitorId(2), + target: pid(2), + tx: tx2, + }) + .ok() + .unwrap(); + // Registered, then cancelled in the inbox: silence. + let (tx3, rx3) = channel::(); + mons.outstanding.insert(MonitorId(3), (pid(3), tx3)); + mon_tx + .send(MonCmd::Demonitor { id: MonitorId(3) }) + .ok() + .unwrap(); + + mons.teardown(&mon_rx); + + let d1 = rx1.recv().unwrap(); + assert_eq!( + (d1.pid, d1.reason), + (pid(1), RemoteDownReason::Disconnected) + ); + let d2 = rx2.recv().unwrap(); + assert_eq!( + (d2.pid, d2.reason), + (pid(2), RemoteDownReason::Disconnected) + ); + // Cancelled: no notice was sent (its sender is dropped, channel + // closed-empty), and nobody got a second one. + assert!(rx3.try_recv().is_err()); + assert!(rx1.try_recv().is_err()); + assert!(rx2.try_recv().is_err()); + assert!(mons.outstanding.is_empty()); + }); + } +} diff --git a/src/cluster/connect.rs b/src/cluster/connect.rs index 503b37c..cf3d1a1 100644 --- a/src/cluster/connect.rs +++ b/src/cluster/connect.rs @@ -27,16 +27,17 @@ use crate::channel::{channel, Receiver, Selectable, Sender}; use crate::cluster::conn::spawn_established; use crate::cluster::envelope::{Frame, RejectReason}; use crate::cluster::handshake::{ - HelloCtx, Initiator, InitiatorOutcome, Local, Peer, Responder, ResponderOutcome, + Initiator, InitiatorOutcome, Local, Peer, PeerStanding, Responder, ResponderOutcome, }; use crate::cluster::manager::{Call, Reply, MANAGER}; use crate::cluster::transport::{FramedConn, Listener, RecvError, SendError, Transport}; +use crate::cluster::Timing; use crate::gen_server; use crate::pid::Pid; use crate::scheduler::{self, spawn}; -/// How long either side waits for the peer's handshake frame before giving -/// up and closing. Enforced on the path via [`FramedConn::recv_deadline`], +/// Default for [`Timing::handshake_timeout`]: how long either side waits for +/// the peer's handshake frame before giving up and closing. Enforced on the path via [`FramedConn::recv_deadline`], /// so a peer that connects and goes silent cannot wedge the acceptor. pub const HANDSHAKE_TIMEOUT: Duration = Duration::from_secs(5); @@ -119,7 +120,7 @@ pub fn dial_handshake( /// Accept-side path step: read the first frame, judge it, answer or close. /// -/// `ctx_for` supplies the [`HelloCtx`] for the *offered* name — knowledge +/// `standing_of` supplies the [`PeerStanding`] of the *offered* name — knowledge /// only the frame reveals, which is why it is a callback and not a value /// (the integrated acceptor asks the manager; loopback tests fabricate). /// It is not called when the first frame is not a `Hello`. @@ -129,7 +130,7 @@ pub fn dial_handshake( pub fn accept_handshake( framed: &mut FramedConn, local: Local, - ctx_for: impl FnOnce(&str) -> HelloCtx, + standing_of: impl FnOnce(&str) -> PeerStanding, deadline: Instant, ) -> Result { let frame = match framed.recv_deadline(deadline) { @@ -143,11 +144,11 @@ pub fn accept_handshake( return Err(from_recv(e)); } }; - let ctx = match &frame { - Frame::Hello { node_name, .. } => ctx_for(node_name), - _ => HelloCtx::default(), + let standing = match &frame { + Frame::Hello { node_name, .. } => standing_of(node_name), + _ => PeerStanding::Free, }; - match Responder::new(local).on_frame(frame, ctx) { + match Responder::new(local).on_frame(frame, standing) { ResponderOutcome::Accepted { reply, peer } => { if let Err(e) = framed.send(&reply) { framed.close(); @@ -199,6 +200,21 @@ pub enum DialError { Duplicate, } +impl DialError { + /// A short static label per kind, for the `smarm-trace` `ClusterDial` + /// event; the payload (io error, names) is not carried. + pub fn label(&self) -> &'static str { + match self { + DialError::AlreadyDialing => "already_dialing", + DialError::ManagerUnavailable => "manager_unavailable", + DialError::Connect(_) => "connect", + DialError::Handshake(_) => "handshake", + DialError::PeerNameMismatch { .. } => "peer_name_mismatch", + DialError::Duplicate => "duplicate", + } + } +} + /// Dial `peer_name` at `addr` and run the handshake, keeping the manager's /// dial-intent set honest around it: the intent is registered *before* /// connecting (so a crossing inbound `Hello` sees it) and cleared the @@ -211,6 +227,7 @@ pub fn dial( addr: &str, peer_name: &str, local: &Local, + timing: Timing, ) -> Result { let me = scheduler::self_pid(); match gen_server::call( @@ -224,7 +241,7 @@ pub fn dial( Ok(Reply::DialBegan(false)) => return Err(DialError::AlreadyDialing), _ => return Err(DialError::ManagerUnavailable), } - let result = connect_and_shake(transport, addr, local); + let result = connect_and_shake(transport, addr, local, timing); // Cleared immediately on outcome — a stale intent during the established // window would corrupt later tie-breaks. Synchronous (a call): the // intent is provably gone before anything else happens. @@ -242,17 +259,18 @@ pub fn dial( got: peer.node_name, }); } - spawn_established(framed, peer).map_err(|_| DialError::Duplicate) + spawn_established(framed, peer, timing).map_err(|_| DialError::Duplicate) } fn connect_and_shake( transport: &dyn Transport, addr: &str, local: &Local, + timing: Timing, ) -> Result<(FramedConn, Peer), DialError> { let conn = transport.dial(addr).map_err(DialError::Connect)?; let mut framed = FramedConn::new(conn); - let deadline = Instant::now() + HANDSHAKE_TIMEOUT; + let deadline = Instant::now() + timing.handshake_timeout; let peer = dial_handshake(&mut framed, local, deadline).map_err(DialError::Handshake)?; Ok((framed, peer)) } @@ -286,14 +304,19 @@ impl AcceptorHandle { /// fd-backed ([`Listener::readable_arm`]); the loopback listener is not, /// and its acceptor exits immediately — loopback handshakes are driven /// synchronously through the path fns instead, per D8. -pub fn spawn_acceptor(listener: Box, local: Local) -> AcceptorHandle { +pub fn spawn_acceptor(listener: Box, local: Local, timing: Timing) -> AcceptorHandle { let addr = listener.local_addr(); let (cmd_tx, cmd_rx) = channel(); - spawn(move || accept_loop(listener, local, cmd_rx)); + spawn(move || accept_loop(listener, local, timing, cmd_rx)); AcceptorHandle { cmd_tx, addr } } -fn accept_loop(mut listener: Box, local: Local, cmd_rx: Receiver<()>) { +fn accept_loop( + mut listener: Box, + local: Local, + timing: Timing, + cmd_rx: Receiver<()>, +) { loop { let Some(arm) = listener.readable_arm() else { return; @@ -311,7 +334,7 @@ fn accept_loop(mut listener: Box, local: Local, cmd_rx: Receiver<( Ok(conn) => conn, Err(_) => return, // listener itself is broken }; - handle_inbound(FramedConn::new(conn), &local); + handle_inbound(FramedConn::new(conn), &local, timing); } Err(_) => return, // fd arm failed to register: listener is gone } @@ -319,27 +342,24 @@ fn accept_loop(mut listener: Box, local: Local, cmd_rx: Receiver<( } /// Run the accept-side handshake for one inbound connection, asking the -/// manager for the [`HelloCtx`], and hand the established connection to the +/// manager for the [`PeerStanding`], and hand the established connection to the /// manager. Every failure was already resolved on the path (reject sent / /// closed, or the registration refused and the actor stopped), so there is /// nothing for the acceptor to carry forward. -fn handle_inbound(mut framed: FramedConn, local: &Local) { - let deadline = Instant::now() + HANDSHAKE_TIMEOUT; - let ctx_for = |name: &str| match gen_server::call( +fn handle_inbound(mut framed: FramedConn, local: &Local, timing: Timing) { + let deadline = Instant::now() + timing.handshake_timeout; + let standing_of = |name: &str| match gen_server::call( MANAGER, - Call::HelloCtx { + Call::Standing { peer_name: name.to_string(), }, ) { - Ok(Reply::HelloCtx(ctx)) => ctx, + Ok(Reply::Standing(s)) => s, // Manager unreachable: nobody could register this connection anyway, // so claim the name taken and reject rather than accept an orphan. - _ => HelloCtx { - name_claimed: true, - dialing_this_peer: false, - }, + _ => PeerStanding::Claimed, }; - if let Ok(peer) = accept_handshake(&mut framed, local.clone(), ctx_for, deadline) { - let _ = spawn_established(framed, peer); + if let Ok(peer) = accept_handshake(&mut framed, local.clone(), standing_of, deadline) { + let _ = spawn_established(framed, peer, timing); } } diff --git a/src/cluster/connector.rs b/src/cluster/connector.rs index b4633ea..2d9eaa0 100644 --- a/src/cluster/connector.rs +++ b/src/cluster/connector.rs @@ -15,9 +15,14 @@ //! `node_down` resume immediately (a fresh sequence — the reconnect case is //! the one backoff exists to pace, but the *first* retry after a death //! should be prompt). A candidate bearing our own name is parked permanently -//! — that seed is us. Every other failure retries: in particular a -//! `NameTaken` reject can be our own ghost at the peer, not yet reaped by -//! its liveness timer, so it must not park. +//! — that seed is us; so is one whose address answers as a different name +//! (`PeerNameMismatch`: a misconfigured or stale seed — each retry would +//! only blip the peer's membership). Every other failure retries: in +//! particular a `NameTaken` reject can be our own ghost at the peer, not +//! yet reaped by its liveness timer, so it must not park. Each attempt's +//! outcome is one `smarm-trace` `ClusterDial` event. A [`Discovery::Withdrawn`] +//! drops its `(name, addr)` from the dial set — only that: a live +//! connection is membership's, and a re-announce re-adds it fresh. //! //! Dials run **inline in the loop** — the same deliberate serialization as //! the acceptor (c6b): each attempt is bounded by the connect + handshake @@ -25,21 +30,23 @@ //! unreachable seeds would stretch the loop's latency; revisit if a real //! deployment ever hits that shape. -use std::collections::{HashMap, HashSet}; +use std::collections::HashSet; use std::time::{Duration, Instant}; use crate::channel::{channel, select, select_timeout, Receiver, Selectable, Sender}; -use crate::cluster::connect::dial; +use crate::cluster::connect::{dial, DialError}; use crate::cluster::discovery::{Discovery, Strategy}; use crate::cluster::handshake::Local; use crate::cluster::membership::{subscribe, NodeEvent}; use crate::cluster::transport::Transport; -use crate::pg::NodeId; +use crate::cluster::Timing; use crate::scheduler::spawn; -/// First retry delay after a failed dial attempt. +/// Default for [`Timing::initial_backoff`]: first retry delay after a failed +/// dial attempt. pub const INITIAL_BACKOFF: Duration = Duration::from_millis(250); -/// Backoff ceiling: an unreachable seed is retried this often, forever. +/// Default for [`Timing::max_backoff`]: an unreachable seed is retried this +/// often, forever. pub const MAX_BACKOFF: Duration = Duration::from_secs(5); enum Cmd { @@ -60,15 +67,74 @@ impl ConnectorHandle { } } -/// One discovered `(name, addr)` and its dial state. +/// One discovered `(name, addr)` and our dial intent towards it. struct Candidate { name: String, addr: String, - /// This seed is the local node itself: never dialed. - parked: bool, - /// Delay to apply after the *next* failure. - backoff: Duration, - next_attempt: Instant, + state: State, +} + +/// The connector's *intent* for a candidate. Whether the peer is currently +/// up is a separate, name-keyed membership fact (`up` in [`run`]): a +/// candidate can arrive after its peer's `node_up` (the snapshot lands +/// before the strategy has said anything), so "up" cannot live on the +/// candidate alone — it is a filter over dialing, not a candidate state. +enum State { + /// Never dialed: this seed is the local node itself, or the address + /// answered as a *different* name than the one seeded + /// (`DialError::PeerNameMismatch` — a misconfigured or stale seed; + /// redialing would only blip the peer's membership forever). The way + /// back is the strategy's: `Withdrawn` then a fresh `Candidate`. + Parked, + /// Dial when due; on failure, back off. + Dialing { + /// Delay to apply after the *next* failure. + backoff: Duration, + next_attempt: Instant, + }, +} + +impl State { + fn fresh(timing: &Timing) -> Self { + State::Dialing { + backoff: timing.initial_backoff, + next_attempt: Instant::now(), + } + } +} + +impl Candidate { + /// The retry deadline, if this candidate is dialing at all. + fn due(&self) -> Option { + match self.state { + State::Parked => None, + State::Dialing { next_attempt, .. } => Some(next_attempt), + } + } + /// A dial attempt was made: schedule the retry, grow the backoff. + fn attempted(&mut self, timing: &Timing) { + if let State::Dialing { + backoff, + next_attempt, + } = &mut self.state + { + *next_attempt = Instant::now() + *backoff; + *backoff = (*backoff * 2).min(timing.max_backoff); + } + } + /// The peer came up: the next sequence (after a later `node_down`) + /// starts from the initial delay again. + fn peer_up(&mut self, timing: &Timing) { + if let State::Dialing { backoff, .. } = &mut self.state { + *backoff = timing.initial_backoff; + } + } + /// The peer went down: redial promptly, fresh sequence. + fn peer_down(&mut self, timing: &Timing) { + if matches!(self.state, State::Dialing { .. }) { + self.state = State::fresh(timing); + } + } } /// Spawn the connector actor. The strategy is spawned as its child; the @@ -78,9 +144,10 @@ pub fn spawn_connector( transport: Box, local: Local, strategy: Box, + timing: Timing, ) -> ConnectorHandle { let (cmd_tx, cmd_rx) = channel(); - spawn(move || run(transport, local, strategy, cmd_rx)); + spawn(move || run(transport, local, strategy, timing, cmd_rx)); ConnectorHandle { cmd_tx } } @@ -88,10 +155,11 @@ fn run( transport: Box, local: Local, strategy: Box, + timing: Timing, cmd_rx: Receiver, ) { // Membership is the connector's source of truth for "who is up" — the - // snapshot seeds `connected` before any candidate arrives. + // snapshot seeds `up` before any candidate arrives. let Some(events) = subscribe() else { return; // no manager, no cluster to connect }; @@ -99,8 +167,7 @@ fn run( spawn(move || strategy.run(disc_tx)); let mut cands: Vec = Vec::new(); - let mut connected: HashSet = HashSet::new(); - let mut names: HashMap = HashMap::new(); // NodeDown carries only the id + let mut up: HashSet = HashSet::new(); let mut strategy_done = false; loop { @@ -111,9 +178,9 @@ fn run( Drained::Open => {} } if !strategy_done { - strategy_done = drain_discoveries(&disc_rx, &local, &mut cands); + strategy_done = drain_discoveries(&disc_rx, &local, &timing, &mut cands); } - match drain_events(&events.rx, &mut connected, &mut names, &mut cands) { + match drain_events(&events.rx, &timing, &mut up, &mut cands) { Drained::Stop => return, // manager gone: the cluster is tearing down Drained::Open => {} } @@ -122,23 +189,26 @@ fn run( let now = Instant::now(); for c in cands .iter_mut() - .filter(|c| !c.parked && !connected.contains(&c.name) && c.next_attempt <= now) + .filter(|c| !up.contains(&c.name) && c.due().is_some_and(|d| d <= now)) { - // The outcome does not branch the bookkeeping: on success the - // manager's node_up is on its way and flips `connected` (backing - // off meanwhile keeps a racing re-attempt from spinning); every - // failure retries — see the module docs. - let _ = dial(&*transport, &c.addr, &c.name, &local); - c.next_attempt = Instant::now() + c.backoff; - c.backoff = (c.backoff * 2).min(MAX_BACKOFF); + // On success the manager's node_up is on its way and lands in + // `up` (backing off meanwhile keeps a racing re-attempt from + // spinning); every failure retries — see the module docs — + // except a peer-name mismatch, which parks the candidate. + let outcome = dial(&*transport, &c.addr, &c.name, &local, timing); + note_dial(&outcome); + match outcome { + Err(DialError::PeerNameMismatch { .. }) => c.state = State::Parked, + _ => c.attempted(&timing), + } } // Wait: until the earliest retry deadline among actionable // candidates, or indefinitely if none is pending. let deadline = cands .iter() - .filter(|c| !c.parked && !connected.contains(&c.name)) - .map(|c| c.next_attempt) + .filter(|c| !up.contains(&c.name)) + .filter_map(Candidate::due) .min(); let mut arms: Vec<&dyn Selectable> = vec![&cmd_rx, &events.rx]; if !strategy_done { @@ -170,23 +240,31 @@ fn drain_cmd(rx: &Receiver) -> Drained { } /// Pull every pending discovery into the candidate set (deduplicated by -/// `(name, addr)`; a candidate bearing the local name is parked). Returns -/// `true` once the strategy's channel closes — it has said all it will. -fn drain_discoveries(rx: &Receiver, local: &Local, cands: &mut Vec) -> bool { +/// `(name, addr)`; a candidate bearing the local name is parked; a +/// `Withdrawn` removes its pair from the dial set and nothing else — see +/// [`Discovery::Withdrawn`]). Returns `true` once the strategy's channel +/// closes — it has said all it will. +fn drain_discoveries( + rx: &Receiver, + local: &Local, + timing: &Timing, + cands: &mut Vec, +) -> bool { loop { match rx.try_recv() { + Ok(Some(Discovery::Withdrawn { name, addr })) => { + cands.retain(|c| !(c.name == name && c.addr == addr)); + } Ok(Some(Discovery::Candidate { name, addr })) => { if cands.iter().any(|c| c.name == name && c.addr == addr) { continue; } - let parked = name == local.node_name; - cands.push(Candidate { - name, - addr, - parked, - backoff: INITIAL_BACKOFF, - next_attempt: Instant::now(), - }); + let state = if name == local.node_name { + State::Parked + } else { + State::fresh(timing) + }; + cands.push(Candidate { name, addr, state }); } Ok(None) => return false, Err(_) => return true, // strategy done; its candidates live on here @@ -194,36 +272,45 @@ fn drain_discoveries(rx: &Receiver, local: &Local, cands: &mut Vec, - connected: &mut HashSet, - names: &mut HashMap, + timing: &Timing, + up: &mut HashSet, cands: &mut [Candidate], ) -> Drained { loop { match rx.try_recv() { Ok(Some(NodeEvent::NodeUp(info))) => { - names.insert(info.node, info.name.clone()); - for c in cands.iter_mut().filter(|c| c.name == info.name) { - c.backoff = INITIAL_BACKOFF; - } - connected.insert(info.name); + cands + .iter_mut() + .filter(|c| c.name == info.name) + .for_each(|c| c.peer_up(timing)); + up.insert(info.name); } - Ok(Some(NodeEvent::NodeDown { node })) => { - if let Some(name) = names.remove(&node) { - connected.remove(&name); - let now = Instant::now(); - for c in cands.iter_mut().filter(|c| c.name == name) { - c.backoff = INITIAL_BACKOFF; - c.next_attempt = now; - } - } + Ok(Some(NodeEvent::NodeDown(info))) => { + up.remove(&info.name); + cands + .iter_mut() + .filter(|c| c.name == info.name) + .for_each(|c| c.peer_down(timing)); } Ok(None) => return Drained::Open, Err(_) => return Drained::Stop, } } } + +/// Surface a dial outcome: one `smarm-trace` event, nothing else. The +/// connector's bookkeeping is decided by the caller. +fn note_dial(outcome: &Result) { + #[cfg(feature = "smarm-trace")] + crate::te!(crate::trace::Event::ClusterDial( + outcome.as_ref().map_or_else(DialError::label, |_| "ok") + )); + #[cfg(not(feature = "smarm-trace"))] + let _ = outcome; +} diff --git a/src/cluster/discovery.rs b/src/cluster/discovery.rs index b7c0e29..edd37a7 100644 --- a/src/cluster/discovery.rs +++ b/src/cluster/discovery.rs @@ -20,13 +20,22 @@ use crate::channel::Sender; /// A discovery event, as pushed by a [`Strategy`]. /// -/// Additive-only for now (candidates are announced, never withdrawn); -/// `#[non_exhaustive]` so expiry can land later without breaking strategies. +/// `Candidate` announces, `Withdrawn` retracts — the primitive pair. A +/// strategy that wants TTL semantics builds them on top (track its own +/// last-seen times, emit `Withdrawn` on expiry); the connector deliberately +/// has no clock of its own for candidates (D11: strategies never +/// re-announce, the connector owns retry). `#[non_exhaustive]` so more can +/// land without breaking strategies. #[derive(Debug, Clone, PartialEq, Eq)] #[non_exhaustive] pub enum Discovery { /// A peer worth dialing: its claimed node name and a dialable address. Candidate { name: String, addr: String }, + /// Stop dialing this `(name, addr)`. Dial-set only: a connection that + /// is already up is membership's business and is left alone; an + /// attempt in flight completes on its own; a later `Candidate` for the + /// same pair re-adds it with fresh backoff. Unknown pairs are ignored. + Withdrawn { name: String, addr: String }, } /// A source of peers to dial. Implementations are spawned as actors by the diff --git a/src/cluster/envelope.rs b/src/cluster/envelope.rs index 445374c..b7d749c 100644 --- a/src/cluster/envelope.rs +++ b/src/cluster/envelope.rs @@ -81,10 +81,45 @@ pub enum Frame { }, Down { monitor_id: u64, - reason: DownReason, + reason: RemoteDownReason, }, } +/// Why a remotely-monitored actor is reported down: either the target's own +/// terminal [`DownReason`] as its node recorded it, or the *link* to that +/// node was lost (or absent) — which says nothing about the actor itself. +/// +/// This is the cluster-side widening of `DownReason` (p5): `Disconnected` +/// is a fact about a connection, never about a local actor, so it lives +/// here rather than in the core enum — a local `Down` can never carry it, +/// and matches on `DownReason` stay exhaustive over actor outcomes only. +/// On the wire `Local(r)` uses `r`'s tag and `Disconnected` is tag 5, +/// bound since c11; no peer emits it today (a lost link is synthesized +/// locally), but the codec honours it both ways. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum RemoteDownReason { + /// The target itself terminated; the peer reported this reason. + Local(DownReason), + /// The link to the target's node was lost or was never up. + Disconnected, +} + +impl RemoteDownReason { + /// The actor's own reason, if this was not a link loss. + pub fn local(self) -> Option { + match self { + RemoteDownReason::Local(r) => Some(r), + RemoteDownReason::Disconnected => None, + } + } +} + +impl From for RemoteDownReason { + fn from(r: DownReason) -> Self { + RemoteDownReason::Local(r) + } +} + // Frame tags. 0 is deliberately unassigned so an all-zero buffer never parses. const TAG_HELLO: u8 = 1; const TAG_HELLO_ACK: u8 = 2; @@ -101,11 +136,12 @@ const REJ_HASH_MISMATCH: u8 = 1; const REJ_NAME_TAKEN: u8 = 2; const REJ_PROTO_VERSION: u8 = 3; -// DownReason tags. c11 adds `Disconnected = 5`; do not reuse tags. +// DownReason tags. Do not reuse tags. const DR_EXIT: u8 = 1; const DR_PANIC: u8 = 2; const DR_STOPPED: u8 = 3; const DR_NOPROC: u8 = 4; +const DR_DISCONNECTED: u8 = 5; /// Frame could not be encoded. The output buffer is left exactly as it was. #[derive(Debug, Clone, Copy, PartialEq, Eq)] @@ -272,10 +308,11 @@ impl Frame { out.push(TAG_DOWN); put_u64(out, *monitor_id); out.push(match reason { - DownReason::Exit => DR_EXIT, - DownReason::Panic => DR_PANIC, - DownReason::Stopped => DR_STOPPED, - DownReason::NoProc => DR_NOPROC, + RemoteDownReason::Local(DownReason::Exit) => DR_EXIT, + RemoteDownReason::Local(DownReason::Panic) => DR_PANIC, + RemoteDownReason::Local(DownReason::Stopped) => DR_STOPPED, + RemoteDownReason::Local(DownReason::NoProc) => DR_NOPROC, + RemoteDownReason::Disconnected => DR_DISCONNECTED, }); } } @@ -364,13 +401,14 @@ impl Frame { TAG_DOWN => Frame::Down { monitor_id: r.u64()?, reason: match r.u8()? { - DR_EXIT => DownReason::Exit, - DR_PANIC => DownReason::Panic, - DR_STOPPED => DownReason::Stopped, - DR_NOPROC => DownReason::NoProc, + DR_EXIT => RemoteDownReason::Local(DownReason::Exit), + DR_PANIC => RemoteDownReason::Local(DownReason::Panic), + DR_STOPPED => RemoteDownReason::Local(DownReason::Stopped), + DR_NOPROC => RemoteDownReason::Local(DownReason::NoProc), + DR_DISCONNECTED => RemoteDownReason::Disconnected, t => { return Err(DecodeError::UnknownEnumTag { - what: "DownReason", + what: "RemoteDownReason", tag: t, }) } diff --git a/src/cluster/handshake.rs b/src/cluster/handshake.rs index 9e05f32..36f8408 100644 --- a/src/cluster/handshake.rs +++ b/src/cluster/handshake.rs @@ -25,14 +25,19 @@ pub struct Peer { pub meta: NodeMeta, } -/// Driver-supplied context for an inbound `Hello` — knowledge the pure -/// machine cannot have (c6 owns the connection table and dial set). -#[derive(Debug, Clone, Copy, Default)] -pub struct HelloCtx { - /// The offered name is already claimed by an established peer. - pub name_claimed: bool, - /// We have our own dial in flight to this peer name. - pub dialing_this_peer: bool, +/// Driver-supplied standing of the *offered* name at this node — knowledge +/// the pure machine cannot have (c6 owns the connection table and dial +/// set). One answer, in the responder's own precedence: an established +/// peer under that name outranks an in-flight dial to it. +#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] +pub enum PeerStanding { + /// Neither connected to nor dialing that name. + #[default] + Free, + /// An established peer already holds that name. + Claimed, + /// We have our own dial in flight to that name. + Dialing, } /// Simultaneous-connect tie-break: does the connection dialed by @@ -124,7 +129,7 @@ impl Responder { /// name → tie-break: validity before identity. `HelloReject` is the /// cross-version compatibility anchor, so a version-mismatched peer /// still gets one. - pub fn on_frame(self, frame: Frame, ctx: HelloCtx) -> ResponderOutcome { + pub fn on_frame(self, frame: Frame, standing: PeerStanding) -> ResponderOutcome { let Frame::Hello { proto_version, build_hash, @@ -147,13 +152,13 @@ impl Responder { if build_hash != self.local.build_hash { return reject(RejectReason::HashMismatch); } - if node_name == self.local.node_name || ctx.name_claimed { + if node_name == self.local.node_name || standing == PeerStanding::Claimed { return reject(RejectReason::NameTaken); } // Simultaneous connect: the inbound frame is the peer's dial. If our // own in-flight dial wins instead, drop this one silently — the peer // computes the same verdict (see `dial_wins`). - if ctx.dialing_this_peer && !dial_wins(&node_name, &self.local.node_name) { + if standing == PeerStanding::Dialing && !dial_wins(&node_name, &self.local.node_name) { return ResponderOutcome::TieBreakLoss; } diff --git a/src/cluster/manager.rs b/src/cluster/manager.rs index 2588023..f92482d 100644 --- a/src/cluster/manager.rs +++ b/src/cluster/manager.rs @@ -24,7 +24,7 @@ use std::collections::HashMap; use crate::channel::Sender; use crate::cluster::conn::ConnHandle; -use crate::cluster::handshake::{HelloCtx, Peer}; +use crate::cluster::handshake::{Peer, PeerStanding}; use crate::cluster::membership::{NodeEvent, NodeInfo}; use crate::cluster::remote::{bind_outbound, unbind_outbound}; use crate::gen_server::{GenServer, GenServerCtx, GenServerName, Watcher}; @@ -53,7 +53,7 @@ pub struct Manager { conns: HashMap, /// In-flight dial intents: peer name -> the actor performing the dial. /// Registered *before* connecting so a crossing inbound `Hello` sees it - /// (the `dialing_this_peer` half of [`HelloCtx`]); cleared the moment + /// ([`PeerStanding::Dialing`]); cleared the moment /// the dial resolves, and — because the dialer is monitored — on the /// dialer's death, so a panicking dial can never wedge the tie-break. dials: HashMap, @@ -143,9 +143,9 @@ pub enum Call { /// The dial to `name` resolved (either way): drop the intent. A call, /// not a cast, so the intent is provably gone before the dialer moves on. DialEnd { name: String }, - /// The [`HelloCtx`] for an inbound `Hello` offering `peer_name` — the + /// The [`PeerStanding`] of an inbound `Hello` offering `peer_name` — the /// accept path asks this between reading the frame and judging it. - HelloCtx { peer_name: String }, + Standing { peer_name: String }, /// Subscribe `tx` to membership events, snapshot-then-stream: one /// [`NodeEvent::NodeUp`] per live peer is queued into `tx` before this /// call answers, so the stream is exact from its first event (handlers @@ -166,7 +166,7 @@ pub enum Reply { /// `false`: another dial to this name is already in flight — do not dial. DialBegan(bool), DialEnded, - HelloCtx(HelloCtx), + Standing(PeerStanding), Subscribed, View(Vec), } @@ -206,8 +206,8 @@ impl GenServer for Manager { } // The outbound table (c9) is maintained here, inside the same // serialized handlers that own the connection's lifetime. - if let Some(out) = handle.take_outbound() { - bind_outbound(&peer.node_name, out); + if let Some((frames, monitors)) = handle.take_outbound() { + bind_outbound(&peer.node_name, peer.incarnation, frames, monitors); } let info = NodeInfo { node: self.node_id(&peer.node_name, peer.incarnation.get()), @@ -230,9 +230,7 @@ impl GenServer for Manager { // Dropping the entry drops the handle, which stops the actor. if let Some(entry) = self.conns.remove(&name) { unbind_outbound(&name); - self.emit(&NodeEvent::NodeDown { - node: entry.info.node, - }); + self.emit(&NodeEvent::NodeDown(entry.info)); } Reply::Disconnected } @@ -255,10 +253,15 @@ impl GenServer for Manager { self.dials.remove(&name); Reply::DialEnded } - Call::HelloCtx { peer_name } => Reply::HelloCtx(HelloCtx { - name_claimed: self.conns.contains_key(&peer_name), - dialing_this_peer: self.dials.contains_key(&peer_name), - }), + Call::Standing { peer_name } => { + Reply::Standing(if self.conns.contains_key(&peer_name) { + PeerStanding::Claimed + } else if self.dials.contains_key(&peer_name) { + PeerStanding::Dialing + } else { + PeerStanding::Free + }) + } Call::Subscribe { tx } => { // The snapshot: queued before `tx` joins the list, and — the // handlers being serialized — before any later event. @@ -279,13 +282,13 @@ impl GenServer for Manager { self.conns.retain(|name, entry| { let dead = entry.pid == down.pid; if dead { - downs.push((name.clone(), entry.info.node)); + downs.push((name.clone(), entry.info.clone())); } !dead }); - for (name, node) in downs { + for (name, info) in downs { unbind_outbound(&name); - self.emit(&NodeEvent::NodeDown { node }); + self.emit(&NodeEvent::NodeDown(info)); } self.dials.retain(|_, pid| *pid != down.pid); } diff --git a/src/cluster/membership.rs b/src/cluster/membership.rs index 4a97390..49e178a 100644 --- a/src/cluster/membership.rs +++ b/src/cluster/membership.rs @@ -57,9 +57,9 @@ pub enum NodeEvent { /// A peer's control connection established and registered. NodeUp(NodeInfo), /// That peer's connection ended — reaped, commanded down, or the manager - /// itself shut down. Which [`NodeInfo`] this id named was delivered in - /// the corresponding `NodeUp`. - NodeDown { node: NodeId }, + /// itself shut down. Carries the same [`NodeInfo`] the corresponding + /// `NodeUp` delivered, so consumers need no id→name reverse map. + NodeDown(NodeInfo), } /// A live membership subscription: the receiving end of the event stream diff --git a/src/cluster/pg.rs b/src/cluster/pg.rs new file mode 100644 index 0000000..6996237 --- /dev/null +++ b/src/cluster/pg.rs @@ -0,0 +1,546 @@ +//! RFC 010 c15 — distributed process groups (Phase 5). +//! +//! The Erlang `pg` shape (D18): every node's group store is the union of +//! its own local members and each peer's *announced* local members. There +//! is one **pg actor** per node — the c14 reaper grown up — and it is the +//! only writer of remote entries and the only sender of announcements: +//! +//! - **Origin owns its members.** Joins are local (`pg::join`), the eager +//! reaper is the liveness authority, and the origin announces every +//! change: `Join`/`Leave` incrementally to every up node, and a full +//! `Sync` of its local groups to a peer the moment that peer comes up +//! (`NodeUp`). Nobody monitors a remote member; a peer's `NodeDown` sweeps +//! every member it announced. +//! - **Transport is a pure consumer** of Phase 3/4: the exposed name +//! [`PG_NAME`] (`"pg"`) carrying [`PgMsg`] over postcard, sent with +//! [`remote::send`]. No new frame, no manager change. +//! - **No anti-entropy.** Per-origin ordering rides the single TCP link: +//! the actor sends `Sync` to a peer *before* it can send that peer any +//! `Join`/`Leave` (both from the same loop, over the same connection), and +//! a reconnect is a fresh `NodeUp` ⇒ fresh `Sync` replacing that peer's +//! set wholesale. +//! - **Local API unchanged.** `members`/`pick`/`dispatch` stay local-only +//! (`get_local_members`); a remote entry in the store carries the peer's +//! `NodeId` and never surfaces there. Cluster-wide reads are the new, +//! additive [`members_all`] over [`GroupMember`] (c16 adds `pick_any` / +//! `dispatch_any`). +//! +//! ## Ordering inside the node +//! +//! `pg::join`/`pg::leave` mutate the store on the caller's thread and then +//! *announce* to the actor's control inbox. Because the store op precedes the +//! announcement and the actor re-reads the store before broadcasting a +//! `Joined`, an announcement that has been overtaken (the member left or died +//! before the actor got to it) is dropped rather than advertised: the wire +//! never sees a `Join` for a member the origin no longer holds. `Leave` +//! broadcasts unconditionally — a spurious `Leave` is a no-op at the peer. +//! +//! Inbound: `NodeUp` is emitted by the manager on the accept/connect path, +//! *before* the peer's connection actor exists, so it is queued on the +//! membership stream before any frame from that peer can reach this inbox. +//! The actor still drains membership before it interprets a `PgMsg` whose +//! sender it does not know, and drops the message if the sender is still not +//! up (a ghost — its next `NodeUp` brings a `Sync`). + +use std::collections::HashMap; + +use crate::channel::{channel, select, Receiver, Selectable}; +use crate::cluster::expose::expose; +use crate::cluster::membership::{subscribe, MembershipEvents, NodeEvent, NodeInfo}; +use crate::cluster::remote::{ + self, local_identity, send_to_remote, RemoteName, RemotePid, ToRemoteError, +}; +use crate::monitor::Down; +use crate::pg::{ + live, member_for, reaper_inboxes, sweep_local_death, Incarnation, Member, Membership, PgEvent, +}; +use crate::pid::{assert_type, Addressable, Erased, Pid}; +use crate::registry::{register, send_to, SendError}; +use crate::scheduler::with_runtime; +use crate::Name; + +/// The exposed name every node's pg actor answers under. +pub const PG_NAME: Name = Name::new("pg"); + +/// The pg wire protocol. Every variant is origin-authored: `from` / the +/// pid's node is the node whose local members are being described. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum PgMsg { + /// The origin's complete local membership, sent to a peer on `NodeUp`. + /// Replaces whatever the receiver held for that origin. + Sync { + from: String, + groups: Vec<(String, Vec>)>, + }, + /// The origin added `pid` (its own) to `group`. + Join { + group: String, + pid: RemotePid, + }, + /// The origin removed `pid` from `group` — voluntary leave or death. + Leave { + group: String, + pid: RemotePid, + }, +} + +// Hand-rolled serde (the crate carries no serde-derive), as a 3-tuple with a +// leading tag: (0, from, groups) | (1, group, pid) | (2, group, pid). +impl serde::Serialize for PgMsg { + fn serialize(&self, s: S) -> Result { + use serde::ser::SerializeTuple; + let mut t = s.serialize_tuple(3)?; + match self { + PgMsg::Sync { from, groups } => { + t.serialize_element(&0u8)?; + t.serialize_element(from)?; + t.serialize_element(groups)?; + } + PgMsg::Join { group, pid } => { + t.serialize_element(&1u8)?; + t.serialize_element(group)?; + t.serialize_element(pid)?; + } + PgMsg::Leave { group, pid } => { + t.serialize_element(&2u8)?; + t.serialize_element(group)?; + t.serialize_element(pid)?; + } + } + t.end() + } +} + +impl<'de> serde::Deserialize<'de> for PgMsg { + fn deserialize>(d: D) -> Result { + struct V; + impl<'de> serde::de::Visitor<'de> for V { + type Value = PgMsg; + fn expecting(&self, f: &mut std::fmt::Formatter) -> std::fmt::Result { + f.write_str("a pg message tuple") + } + fn visit_seq>( + self, + mut seq: A, + ) -> Result { + use serde::de::Error; + let tag: u8 = seq + .next_element()? + .ok_or_else(|| A::Error::custom("pg: missing tag"))?; + let text: String = seq + .next_element()? + .ok_or_else(|| A::Error::custom("pg: missing name"))?; + match tag { + 0 => { + let groups = seq + .next_element()? + .ok_or_else(|| A::Error::custom("pg: missing groups"))?; + Ok(PgMsg::Sync { from: text, groups }) + } + 1 | 2 => { + let pid = seq + .next_element()? + .ok_or_else(|| A::Error::custom("pg: missing pid"))?; + Ok(if tag == 1 { + PgMsg::Join { group: text, pid } + } else { + PgMsg::Leave { group: text, pid } + }) + } + t => Err(A::Error::custom(format!("pg: unknown tag {t}"))), + } + } + } + d.deserialize_tuple(3, V) + } +} + +/// A member of a group as the cluster sees it: on this node (a plain +/// [`Pid`], sendable locally) or on a peer (a [`RemotePid`], sendable via +/// [`send_to_remote`](remote::send_to_remote)). `Pid` cannot hold a remote +/// (D14), hence the two-variant shape. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum GroupMember { + Local(Pid), + Remote(RemotePid), +} + +/// Every member of `group` cluster-wide, in the store's order: local members +/// filtered by the same liveness backstop as [`members`](crate::pg::members), +/// remote members exactly as their origins last announced them. Must run +/// inside [`run`](crate::run). +pub fn members_all(group: &str) -> Vec { + with_runtime(|inner| { + let me = inner.node_id; + let pg = inner.process_groups.lock(); + pg.all_of(group) + .into_iter() + .filter_map(|m| { + if m.node == me { + live(inner, m.pid).then_some(GroupMember::Local(m.pid)) + } else { + // A remote entry always has its node's name recorded + // (they land under the same lock); a missing one is a + // node already swept, so it hides rather than misnames. + pg.node_name(m.node).map(|name| { + GroupMember::Remote(RemotePid::from_parts( + name, + m.incarnation, + m.pid.index(), + m.pid.generation(), + )) + }) + } + }) + .collect() + }) +} + +/// One member of `group` cluster-wide, or `None` if it has none: the first +/// entry in the store's order (this node's members in join order first when +/// they joined first — the same stateless first-live scan as +/// [`pick`](crate::pg::pick), extended over the peers' announced members). +/// Must run inside [`run`](crate::run). +pub fn pick_any(group: &str) -> Option { + members_all(group).into_iter().next() +} + +/// Why [`dispatch_any`] handed `msg` back. +#[derive(Debug)] +pub enum DispatchAnyError { + /// The group has no member anywhere. + NoMember(M), + /// The pick was local and the local typed send failed. + Local(SendError), + /// The pick was remote and the remote send failed at this node. + Remote(ToRemoteError), +} + +impl DispatchAnyError { + /// The undelivered message. + pub fn into_inner(self) -> M { + match self { + DispatchAnyError::NoMember(m) => m, + DispatchAnyError::Local(e) => e.into_inner(), + DispatchAnyError::Remote(e) => e.into_inner(), + } + } +} + +/// [`pick_any`] and send in one step, returning the member reached: a local +/// pick goes through [`send_to`], a remote one through [`send_to_remote`] +/// (so `Ok` for a remote member means "handed to the connection", RFC 010 +/// §3). Homogeneous pool assumed, as for [`dispatch`](crate::pg::dispatch); +/// a wrong `A` degrades to a clean error at the target, never a misroute. +/// Must run inside [`run`](crate::run). +pub fn dispatch_any(group: &str, msg: A::Msg) -> Result> +where + A: Addressable, + A::Msg: serde::Serialize, +{ + match pick_any(group) { + None => Err(DispatchAnyError::NoMember(msg)), + Some(GroupMember::Local(pid)) => send_to(assert_type::(pid), msg) + .map(|()| GroupMember::Local(pid)) + .map_err(DispatchAnyError::Local), + Some(GroupMember::Remote(rp)) => send_to_remote(rp.clone().assert_type::(), msg) + .map(|()| GroupMember::Remote(rp)) + .map_err(DispatchAnyError::Remote), + } +} + +/// Attach the pg actor to the running cluster. Called once by +/// `cluster::start` after the manager is up and the local identity is set; +/// spawns the actor if this run has not joined anything yet. +pub(crate) fn attach_cluster() { + let _ = reaper_inboxes().ctl.send(PgEvent::Attach); +} + +/// The attached half of the actor's state: who is up (by name) and the +/// membership stream. +struct Attached { + events: MembershipEvents, + peers: HashMap, + /// This node's wire identity: `Sync`'s `from`, and the stamp on every + /// pid we ship (attach requires it, so no `None` path exists here). + me: String, + incarnation: Incarnation, +} + +/// The pg actor: the c14 reaper (`deaths`), the local API's announcements +/// (`ctl`), and — once attached — the membership stream and the exposed +/// `"pg"` inbox, all in one drain-then-select loop. `deaths`/`ctl` closing +/// is the run tearing down; the membership stream closing is the manager +/// gone (detach, keep reaping). +pub(crate) fn actor(deaths: Receiver, ctl: Receiver) { + let (pg_tx, pg_rx) = channel::(); + let mut cl: Option = None; + loop { + loop { + match deaths.try_recv() { + Ok(Some(down)) => on_death(cl.as_ref(), down.pid), + Ok(None) => break, + Err(_) => return, + } + } + loop { + match ctl.try_recv() { + Ok(Some(PgEvent::Attach)) => { + // Own the name BEFORE subscribing (which yields to the + // manager): a peer's first frame must find "pg" exposed + // and resolvable, or it is dropped. Idempotent for a + // re-attach: same actor, same channel (the registry + // refuses a *second* live one). + let _ = register(PG_NAME, pg_tx.clone()); + expose(PG_NAME); + if let Some(a) = attach() { + cl = Some(a); + } + } + Ok(Some(PgEvent::Joined { group, pid })) => on_joined(cl.as_ref(), &group, pid), + Ok(Some(PgEvent::Left { group, pid })) => on_left(cl.as_ref(), &group, pid), + Ok(None) => break, + Err(_) => return, + } + } + if let Some(a) = cl.as_mut() { + if !drain_events(a) { + cl = None; + continue; + } + loop { + match pg_rx.try_recv() { + Ok(Some(msg)) => on_msg(a, msg), + Ok(None) => break, + Err(_) => return, // our own inbox: only on teardown + } + } + } + // Wait. Control first (attach/teardown must be prompt), then deaths, + // then the cluster arms. + let mut arms: Vec<&dyn Selectable> = vec![&ctl, &deaths]; + if let Some(a) = cl.as_ref() { + arms.push(&a.events.rx); + arms.push(&pg_rx); + } + let _ = select(&arms); + } +} + +fn attach() -> Option { + let events = subscribe()?; + let (me, incarnation) = local_identity()?; + Some(Attached { + events, + peers: HashMap::new(), + me, + incarnation, + }) +} + +/// Fold pending membership events: `NodeUp` ⇒ record + `Sync` that peer; +/// `NodeDown` ⇒ sweep every member it announced. `false` when the stream +/// has closed. +fn drain_events(a: &mut Attached) -> bool { + loop { + match a.events.rx.try_recv() { + Ok(Some(NodeEvent::NodeUp(info))) => { + let name = info.name.clone(); + a.peers.insert(name.clone(), info); + // Snapshot under the store lock, then stamp wire pids + // outside it (`from_local` marks watchable under the slot's + // cold lock — Leaf-on-Leaf nesting is asserted). + let local: Vec<(String, Vec)> = + with_runtime(|inner| inner.process_groups.lock().groups_on(inner.node_id)); + let groups = local + .into_iter() + .map(|(g, pids)| (g, pids.into_iter().map(|p| wire(a, p)).collect())) + .collect(); + let msg = PgMsg::Sync { + from: a.me.clone(), + groups, + }; + let _ = remote::send(RemoteName::new(name, PG_NAME), msg); + } + Ok(Some(NodeEvent::NodeDown(info))) => { + a.peers.remove(&info.name); + with_runtime(|inner| { + let mut pg = inner.process_groups.lock(); + pg.remove_where(|m| m.node == info.node); + pg.forget_node_name(info.node); + }); + } + Ok(None) => return true, + Err(_) => return false, + } + } +} + +/// Send `msg` to every up peer. `NotConnected` is ignored: that peer's +/// `NodeDown` is on its way and its next `NodeUp` gets a `Sync`. +fn broadcast(a: &Attached, msg: PgMsg) { + for name in a.peers.keys() { + let _ = remote::send(RemoteName::new(name.clone(), PG_NAME), msg.clone()); + } +} + +fn on_death(a: Option<&Attached>, pid: Pid) { + let evicted = sweep_local_death(pid); + if let Some(a) = a { + for (group, ms) in evicted { + broadcast( + a, + PgMsg::Leave { + group, + pid: wire(a, ms.member.pid), + }, + ); + } + } +} + +fn on_joined(a: Option<&Attached>, group: &str, pid: Pid) { + let Some(a) = a else { return }; + // Re-check: a leave/death may have overtaken the announcement. + let still = with_runtime(|inner| { + let m = member_for(inner, pid); + inner.process_groups.lock().contains(group, &m) + }); + if still { + broadcast( + a, + PgMsg::Join { + group: group.to_owned(), + pid: wire(a, pid), + }, + ); + } +} + +fn on_left(a: Option<&Attached>, group: &str, pid: Pid) { + let Some(a) = a else { return }; + broadcast( + a, + PgMsg::Leave { + group: group.to_owned(), + pid: wire(a, pid), + }, + ); +} + +/// The wire form of a local member pid, stamped with the identity the +/// actor was attached with (marks watchable, like `from_local`). +fn wire(a: &Attached, pid: Pid) -> RemotePid { + RemotePid::from_local_at(pid, a.me.clone(), a.incarnation) +} + +/// The named origin's `NodeInfo`, if it is up. A second look at the +/// membership stream covers a `NodeUp` that landed after this loop +/// iteration's drain; anything still unknown is a ghost and is dropped. +fn origin(a: &mut Attached, name: &str) -> Option { + if let Some(i) = a.peers.get(name) { + return Some(i.clone()); + } + drain_events(a); + a.peers.get(name).cloned() +} + +/// `origin`, additionally requiring `pid` to be stamped with the origin's +/// current incarnation — a pid from a previous life of that node is a ghost. +fn origin_of(a: &mut Attached, pid: &RemotePid) -> Option { + origin(a, pid.node()).filter(|i| i.incarnation == pid.incarnation()) +} + +fn remote_membership(origin: &NodeInfo, pid: &RemotePid) -> Membership { + Membership { + member: Member { + node: origin.node, + incarnation: origin.incarnation, + pid: Pid::new(pid.index(), pid.generation()), + }, + monitor: None, + } +} + +fn on_msg(a: &mut Attached, msg: PgMsg) { + match msg { + PgMsg::Sync { from, groups } => { + let Some(info) = origin(a, &from) else { return }; + with_runtime(|inner| { + let mut pg = inner.process_groups.lock(); + pg.remove_where(|m| m.node == info.node); + pg.set_node_name(info.node, info.name.clone()); + for (group, pids) in &groups { + // Origin-authored: only its own current-incarnation pids. + for p in pids + .iter() + .filter(|p| p.node() == from && p.incarnation() == info.incarnation) + { + pg.join(group, remote_membership(&info, p)); + } + } + }); + } + PgMsg::Join { group, pid } => { + let Some(info) = origin_of(a, &pid) else { + return; + }; + with_runtime(|inner| { + let mut pg = inner.process_groups.lock(); + pg.set_node_name(info.node, info.name.clone()); + pg.join(&group, remote_membership(&info, &pid)); + }); + } + PgMsg::Leave { group, pid } => { + let Some(info) = origin_of(a, &pid) else { + return; + }; + with_runtime(|inner| { + let ms = remote_membership(&info, &pid); + inner.process_groups.lock().leave(&group, ms.member); + }); + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::cluster::envelope::{decode_payload, encode_payload}; + + #[test] + fn pg_msg_roundtrips_every_variant() { + let p = RemotePid::::from_parts("a", Incarnation::new(9), 3, 1); + for m in [ + PgMsg::Sync { + from: "a".into(), + groups: vec![ + ("g".into(), vec![p.clone(), p.clone()]), + ("h".into(), vec![]), + ], + }, + PgMsg::Sync { + from: "a".into(), + groups: vec![], + }, + PgMsg::Join { + group: "g".into(), + pid: p.clone(), + }, + PgMsg::Leave { + group: "g".into(), + pid: p.clone(), + }, + ] { + let bytes = encode_payload(&m).unwrap(); + let back: PgMsg = decode_payload(&bytes).unwrap(); + assert_eq!(back, m); + } + } + + #[test] + fn pg_msg_rejects_unknown_tag() { + let bytes = encode_payload(&(7u8, "x", 0u32)).unwrap(); + assert!(decode_payload::(&bytes).is_err()); + } +} diff --git a/src/cluster/remote.rs b/src/cluster/remote.rs index 33e4191..8b74c43 100644 --- a/src/cluster/remote.rs +++ b/src/cluster/remote.rs @@ -41,15 +41,41 @@ //! Refusals are silent to the sender by design (§3: send failure reflects //! local knowledge only); they are observable locally as the returned //! [`InboundVerdict`], which the conn actor may log or count. +//! +//! ## Pids (c10, D14) +//! +//! [`RemotePid`] = `(node_name, incarnation, index, generation)` + +//! phantom — identity-bound, dead when that incarnation dies, never +//! redirects. The node travels as its **name** (a global identifier, so a pid +//! forwarded through a third node needs no re-mapping); NodeId is a local +//! alias and never crosses. A local `Pid` serializes *as* a `RemotePid` +//! stamped from the ambient [local identity](set_local_identity); a +//! `RemotePid` deserializes into `Pid` only when it names this node (the +//! collapse), else it is a decode error — fields that may hold a pid from +//! anywhere are typed `RemotePid`. +//! +//! [`send_to_remote`] is the pid-targeted send. A self-node pid short- +//! circuits to the local typed send with the message object itself — no +//! encode, no frame (zero-copy-equivalent). Otherwise the outbound table +//! (widened to carry each node's **current incarnation**) does the RFC v2 §3 +//! check at the send site: a pid of a dead incarnation is +//! [`ToRemoteError::DeadIncarnation`] and no frame is emitted. Inbound +//! `Send` frames are delivered by index/generation through c8's +//! [`decode_deliver`]: the target actor's published channel for the exposed +//! type is the only route (the reply-to path requires +//! [`expose_type`](crate::cluster::expose::expose_type) at the receiver). +use std::cell::Cell; use std::collections::HashMap; use std::marker::PhantomData; -use crate::channel::Sender; -use crate::cluster::envelope::{encode_payload, Frame, PayloadError}; +use crate::channel::{channel, Receiver, RecvError, Selectable, Sender}; +use crate::cluster::envelope::{encode_payload, Frame, PayloadError, RemoteDownReason}; use crate::cluster::expose::{decode_deliver, exposed_hash, type_hash, DeliverError}; -use crate::pid::Name; -use crate::registry::whereis; +use crate::monitor::{demonitor, monitor, Monitor, MonitorId}; +use crate::pg::Incarnation; +use crate::pid::{Addressable, Erased, Name, Pid}; +use crate::registry::{send_to, whereis, SendError}; use crate::scheduler::with_runtime; /// A name on a specific remote node: `(node_name, Name)`. Sendable via @@ -111,26 +137,97 @@ impl RemoteSendError { } } -/// The outbound table, one per runtime (a `RuntimeInner` field). +/// The outbound table, one per runtime (a `RuntimeInner` field): per live +/// node, its current incarnation (the RFC v2 §3 send-site check) and the +/// connection's dedicated outbound sender. Plus this node's own wire +/// identity, which pid serialization stamps. pub(crate) struct Outbound { - by_node: HashMap>, + by_node: HashMap, + local: Option<(String, Incarnation)>, +} + +/// One live connection as the outbound path sees it: the peer's current +/// incarnation and the two inboxes of its connection actor — frames (c9) +/// and monitor bookkeeping (c12, [`MonCmd`]). +pub(crate) struct Route { + incarnation: Incarnation, + frames: Sender, + monitors: Sender, } impl Outbound { pub(crate) fn new() -> Self { Outbound { by_node: HashMap::new(), + local: None, } } } -/// Manager-only: bind `node`'s outbound channel. Called inside `Register`. -pub(crate) fn bind_outbound(node: &str, tx: Sender) { +/// Set this node's wire identity — what serialized pids are stamped with +/// and what a `RemotePid` must name to collapse. `cluster::start` sets it; +/// exposed for local tests. Must run inside [`run`](crate::run). +pub fn set_local_identity(node: &str, incarnation: Incarnation) { with_runtime(|inner| { - inner.outbound.lock().by_node.insert(node.to_string(), tx); + inner.outbound.lock().local = Some((node.to_string(), incarnation)); }); } +/// This node's wire identity, if set. Must run inside [`run`](crate::run). +pub fn local_identity() -> Option<(String, Incarnation)> { + with_runtime(|inner| inner.outbound.lock().local.clone()) +} + +/// Manager-only: bind `node`'s outbound channels at `incarnation`. Called +/// inside `Register`. +pub(crate) fn bind_outbound( + node: &str, + incarnation: Incarnation, + frames: Sender, + monitors: Sender, +) { + with_runtime(|inner| { + inner.outbound.lock().by_node.insert( + node.to_string(), + Route { + incarnation, + frames, + monitors, + }, + ); + }); +} + +/// Test probe: bind an arbitrary sender as `node`'s outbound so a test can +/// assert what frames leave — or don't. Same table, same lookup as the real +/// path (this is how "no frame emitted" is asserted at the frame level). +/// Frames only: there is no connection actor behind a probe, so a +/// [`monitor_remote`] against a probed node reports `Disconnected`. +pub fn bind_outbound_probe(node: &str, incarnation: Incarnation, tx: Sender) { + drop(bind_outbound_probe_with_monitors(node, incarnation, tx)); +} + +/// The monitor half of a probed node's inbox: opaque, held only to be +/// dropped. See [`bind_outbound_probe_with_monitors`]. +pub struct MonitorInbox { + _rx: Receiver, +} + +/// Test probe: like [`bind_outbound_probe`], but the monitor-command +/// receiver is handed back instead of dropped, so a test can stage the +/// c13 drain gap — a `Monitor` command that reached the connection's inbox +/// and dies unread when the inbox is dropped. While the inbox lives, +/// [`monitor_remote`] against the probed node is simply in flight. +pub fn bind_outbound_probe_with_monitors( + node: &str, + incarnation: Incarnation, + tx: Sender, +) -> MonitorInbox { + let (mon_tx, mon_rx) = channel(); + bind_outbound(node, incarnation, tx, mon_tx); + MonitorInbox { _rx: mon_rx } +} + /// Manager-only: unbind `node`'s outbound channel. Called on `Disconnect`, /// reap, and manager shutdown. Dropping the sender is what closes the conn /// actor's outbound arm — but that arm's closure is NOT a stop signal (the @@ -191,13 +288,249 @@ pub fn send_remote_raw( /// One lookup, one send. Clone the sender out under the lock and send /// outside it (a channel send can unpark the conn actor). fn hand_to_connection(node: &str, frame: Frame) -> Result<(), NotConnected> { - let tx = with_runtime(|inner| inner.outbound.lock().by_node.get(node).cloned()); + let tx = with_runtime(|inner| { + inner + .outbound + .lock() + .by_node + .get(node) + .map(|r| r.frames.clone()) + }); match tx { Some(tx) => tx.send(frame).map_err(|_| NotConnected), None => Err(NotConnected), } } +// ---- pids --------------------------------------------------------------- + +/// A pid on some node: `(node_name, incarnation, index, generation)` plus +/// the actor type. See the module docs. Serializes as a 4-tuple. +pub struct RemotePid { + node: String, + incarnation: Incarnation, + index: u32, + generation: u32, + _marker: PhantomData A>, +} + +impl RemotePid { + /// Build from raw parts (tests, and codecs re-hydrating a pid). + pub fn from_parts( + node: impl Into, + incarnation: Incarnation, + index: u32, + generation: u32, + ) -> Self { + RemotePid { + node: node.into(), + incarnation, + index, + generation, + _marker: PhantomData, + } + } + + /// The wire form of a local pid, stamped with this node's identity, and + /// **marked watchable** — asking for the wire form *is* the intent to + /// ship the pid, so this is the same D12 set-site as `Pid::serialize` + /// (c12 made it explicit: a peer may monitor exactly the pids that + /// crossed, and a pid handed out via `from_local` in a hand-built reply + /// has crossed). Must run inside [`run`](crate::run). + /// + /// `None` when this runtime has no wire identity (no `cluster::start`, + /// no [`set_local_identity`]): such a pid cannot name a node, and a + /// stamped `("", 0)` would be dropped by every peer with no signal. + /// The pid is not marked watchable in that case either. + pub fn from_local(pid: Pid) -> Option { + let (node, incarnation) = local_identity()?; + Some(Self::from_local_at(pid, node, incarnation)) + } + + /// `from_local` with the identity supplied by the caller — for a holder + /// that already carries the node's identity (the pg actor) and must not + /// have a `None` path. Marks watchable like `from_local`. + pub(crate) fn from_local_at(pid: Pid, node: String, incarnation: Incarnation) -> Self { + crate::monitor::mark_watchable(pid); + RemotePid::from_parts(node, incarnation, pid.index(), pid.generation()) + } + + /// The collapse: `Some(local pid)` iff this pid names this very node + /// (name and incarnation). Must run inside [`run`](crate::run). + pub fn local(&self) -> Option> { + let (n, i) = local_identity()?; + (n == self.node && i == self.incarnation) + .then(|| crate::pid::assert_type::(Pid::new(self.index, self.generation))) + } + + /// Drop the actor type: the untyped `RemotePid`, the form + /// [`RemoteDown`] and [`RemoteMonitor`] carry (mirrors [`Pid::erase`]). + pub fn erase(self) -> RemotePid { + RemotePid::from_parts(self.node, self.incarnation, self.index, self.generation) + } + + /// Re-type an erased pid as `RemotePid` — the unchecked mirror of + /// `pid::assert_type`, with the same degradation: a wrong `B` means the + /// target refuses the payload's hash (never a misroute). + pub(crate) fn assert_type(self) -> RemotePid { + RemotePid::from_parts(self.node, self.incarnation, self.index, self.generation) + } + + pub fn node(&self) -> &str { + &self.node + } + pub fn incarnation(&self) -> Incarnation { + self.incarnation + } + pub fn index(&self) -> u32 { + self.index + } + pub fn generation(&self) -> u32 { + self.generation + } +} + +impl Clone for RemotePid { + fn clone(&self) -> Self { + RemotePid::from_parts( + self.node.clone(), + self.incarnation, + self.index, + self.generation, + ) + } +} +impl PartialEq for RemotePid { + fn eq(&self, o: &Self) -> bool { + self.node == o.node + && self.incarnation == o.incarnation + && self.index == o.index + && self.generation == o.generation + } +} +impl Eq for RemotePid {} +impl std::fmt::Debug for RemotePid { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!( + f, + "<{}.{}@{}#{}>", + self.index, + self.generation, + self.node, + self.incarnation.get() + ) + } +} + +impl serde::Serialize for RemotePid { + fn serialize(&self, s: S) -> Result { + ( + self.node.as_str(), + self.incarnation.get(), + self.index, + self.generation, + ) + .serialize(s) + } +} +impl<'de, A> serde::Deserialize<'de> for RemotePid { + fn deserialize>(d: D) -> Result { + let (node, inc, index, generation) = <(String, u32, u32, u32)>::deserialize(d)?; + Ok(RemotePid::from_parts( + node, + Incarnation::new(inc), + index, + generation, + )) + } +} + +/// Why a pid-targeted send did not leave this node. Local knowledge only. +#[derive(Debug)] +pub enum ToRemoteError { + /// No live connection to the pid's node. + NotConnected(M), + /// The pid's incarnation is not that node's current one (RFC v2 §3): the + /// actor died with its incarnation. Detected at the send site; no frame. + DeadIncarnation(M), + /// The payload did not serialize. + Encode(M, PayloadError), + /// The pid collapsed to a local one and the local typed send failed. + Local(SendError), +} + +impl ToRemoteError { + /// The undelivered message. + pub fn into_inner(self) -> M { + match self { + ToRemoteError::NotConnected(m) + | ToRemoteError::DeadIncarnation(m) + | ToRemoteError::Encode(m, _) => m, + ToRemoteError::Local(e) => e.into_inner(), + } + } +} + +/// Send `msg` to a pid, wherever it lives. Self-node pids short-circuit to +/// the local typed send with `msg` itself (no encode, no frame); others go +/// out as a `Send` frame after the incarnation check. `Ok(())` for a remote +/// target = handed to the connection's inbox. Must run inside +/// [`run`](crate::run). +pub fn send_to_remote(target: RemotePid, msg: A::Msg) -> Result<(), ToRemoteError> +where + A: Addressable, + A::Msg: serde::Serialize, +{ + if let Some(local) = target.local() { + return send_to(local, msg).map_err(ToRemoteError::Local); + } + let route = with_runtime(|inner| { + inner + .outbound + .lock() + .by_node + .get(&target.node) + .map(|r| (r.incarnation, r.frames.clone())) + }); + let (current, tx) = match route { + Some(r) => r, + None => return Err(ToRemoteError::NotConnected(msg)), + }; + if current != target.incarnation { + return Err(ToRemoteError::DeadIncarnation(msg)); + } + let payload = match encode_payload(&msg) { + Ok(p) => p, + Err(e) => return Err(ToRemoteError::Encode(msg, e)), + }; + let frame = Frame::Send { + index: target.index, + generation: target.generation, + type_hash: type_hash::(), + payload, + }; + tx.send(frame).map_err(|_| ToRemoteError::NotConnected(msg)) +} + +/// The inbound `Send` seam: deliver `payload` under `type_hash` to the local +/// actor `(index, generation)`. Node and incarnation are implicit in the +/// connection (bound at handshake) — the frame carries only the slot +/// identity. Delivery goes through c8's decoder table, so only types the +/// receiver has [`expose_type`](crate::cluster::expose::expose_type)d (or +/// exposed by name) can land; anything else is refused, never misrouted. +pub fn deliver_to_pid( + index: u32, + generation: u32, + type_hash: u64, + payload: &[u8], +) -> InboundVerdict { + let pid = Pid::new(index, generation); + match decode_deliver(type_hash, pid, payload) { + Ok(()) => InboundVerdict::Delivered, + Err(e) => InboundVerdict::Refused(e), + } +} + /// What the inbound seam did with a `SendNamed`. Local observability only; /// nothing goes back on the wire (RFC §3). #[derive(Debug)] @@ -217,6 +550,20 @@ pub enum InboundVerdict { Refused(DeliverError), } +impl InboundVerdict { + /// A short static label for tracing/counting (`smarm-trace` records one + /// `ClusterInbound` event per frame with it). + pub fn label(&self) -> &'static str { + match self { + InboundVerdict::Delivered => "delivered", + InboundVerdict::NotExposed => "not_exposed", + InboundVerdict::HashMismatch { .. } => "hash_mismatch", + InboundVerdict::Unresolved => "unresolved", + InboundVerdict::Refused(_) => "refused", + } + } +} + /// THE inbound resolution seam: exposed-set check → hash check → registry /// resolution → c8 delivery. See the module docs. Must run inside /// [`run`](crate::run) — the conn actor's context. @@ -238,3 +585,251 @@ pub fn deliver_named(name: &str, type_hash: u64, payload: &[u8]) -> InboundVerdi Err(e) => InboundVerdict::Refused(e), } } + +// ---- monitors (c12) ----------------------------------------------------- + +/// Bookkeeping commands from [`monitor_remote`]/[`demonitor_remote`] to the +/// connection actor that owns the link to the target's node. The actor +/// records the registration and *then* emits the `Monitor` frame itself, so +/// a `Down` can never arrive at a table that does not yet know the id. It +/// lives in the actor (not on `RuntimeInner`) so the bookkeeping dies with +/// the connection — exactly what c13 needs to synthesize `Disconnected`. +pub(crate) enum MonCmd { + Monitor { + id: MonitorId, + target: RemotePid, + tx: Sender, + }, + Demonitor { + id: MonitorId, + }, +} + +/// A remotely-monitored actor's termination notice — the cluster analog of +/// [`Down`](crate::monitor::Down), with the pid in its wire form because it +/// may name any node. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct RemoteDown { + /// The pid that was being monitored. + pub pid: RemotePid, + /// How it went down. `Disconnected` means the *link* to its node was + /// lost (or absent) — nothing is known about the actor itself. + pub reason: RemoteDownReason, +} + +enum Watch { + /// The target collapsed to this node: an ordinary local monitor, + /// translated on read. + Local(Monitor), + /// The target is elsewhere: the connection actor for its node holds the + /// registration and forwards the peer's `Down` frame here. The + /// `RemoteState` is the read-side backstop (c13): a channel that closes + /// while `Live` — the connection died with our command unread — reads + /// as `Disconnected` once; afterwards, and after a cancel, closed is + /// just closed. + Remote(Receiver, Cell), +} + +/// Where a remote-watch stands from the reader's side. +#[derive(Clone, Copy, PartialEq, Eq)] +enum RemoteState { + /// No notice yet, not cancelled: a closed channel means `Disconnected`. + Live, + /// The one notice has been read (or synthesized): nothing more is due. + Done, + /// `demonitor_remote` ran: never synthesize. + Cancelled, +} + +/// A live remote monitor: read its one [`RemoteDown`] with +/// [`recv`](RemoteMonitor::recv)/[`try_recv`](RemoteMonitor::try_recv), or +/// fold it into a `select` via [`arm`](RemoteMonitor::arm). Distinct from +/// [`Monitor`] on purpose: its target is a [`RemotePid`], its notice a +/// [`RemoteDown`], and it can report `Disconnected` — none of which a local +/// monitor can express. Dropping it discards an unread notice, like the +/// local one; after [`demonitor_remote`] the channel is closed and empty, so +/// `recv` errs rather than parking — also like the local one. +/// +/// Exactly one notice is guaranteed even if the connection actor dies with +/// the registration unread (the c13 drain gap): a channel that closes +/// before any notice — and before any cancel — reads as `Disconnected`, +/// once. The next read is the ordinary closed-channel `Err`. +pub struct RemoteMonitor { + /// This registration's process-unique id — minted here, echoed by the + /// peer in its `Down` frame. + pub id: MonitorId, + /// The pid being monitored. + pub target: RemotePid, + watch: Watch, +} + +impl RemoteMonitor { + /// Block (cooperatively) for the notice. + pub fn recv(&self) -> Result { + match &self.watch { + Watch::Local(m) => m.rx.recv().map(|d| RemoteDown { + pid: self.target.clone(), + reason: d.reason.into(), + }), + Watch::Remote(rx, st) => match rx.recv() { + Ok(d) => { + st.set(RemoteState::Done); + Ok(d) + } + Err(e) => self.closed(st).ok_or(e), + }, + } + } + + /// The notice if it has arrived; `Ok(None)` if not yet. + pub fn try_recv(&self) -> Result, RecvError> { + match &self.watch { + Watch::Local(m) => m.rx.try_recv().map(|o| { + o.map(|d| RemoteDown { + pid: self.target.clone(), + reason: d.reason.into(), + }) + }), + Watch::Remote(rx, st) => match rx.try_recv() { + Ok(Some(d)) => { + st.set(RemoteState::Done); + Ok(Some(d)) + } + Ok(None) => Ok(None), + Err(e) => self.closed(st).map(Some).ok_or(e), + }, + } + } + + /// The channel closed. While `Live` — no notice yet, no cancel — that + /// is the connection having died with our registration unread, so + /// synthesize the one `Disconnected` and mark `Done`; otherwise closed + /// is just closed. + fn closed(&self, st: &Cell) -> Option { + if st.get() != RemoteState::Live { + return None; + } + st.set(RemoteState::Done); + Some(RemoteDown { + pid: self.target.clone(), + reason: RemoteDownReason::Disconnected, + }) + } + + /// The selectable arm: readiness means [`try_recv`](Self::try_recv) + /// will yield the notice. + pub fn arm(&self) -> &dyn Selectable { + match &self.watch { + Watch::Local(m) => &m.rx, + Watch::Remote(rx, _) => rx, + } + } +} + +impl std::fmt::Debug for RemoteMonitor { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("RemoteMonitor") + .field("id", &self.id) + .field("target", &self.target) + .finish_non_exhaustive() + } +} + +/// Monitor `target`, wherever it lives. Exactly one [`RemoteDown`] arrives: +/// +/// - self-node pid ⇒ an ordinary local monitor underneath (same NoProc rule); +/// - no live connection to the pid's node ⇒ `Disconnected`, queued at once +/// (the remote analog of NoProc: nothing can be known); +/// - the pid's incarnation is not the node's current one ⇒ `NoProc`, queued +/// at once — the node is *known* to have restarted, so its actor is a +/// corpse, not a partition (RFC v2 §3); +/// - otherwise the connection actor registers the id and sends `Monitor`; +/// the peer answers with the true terminal reason on exit, or immediately +/// with the recorded reason for a corpse (`terminal_reason`, RFC §6) or +/// `NoProc` for a pid it never exposed and never shipped. +/// +/// The connection dropping while the monitor is outstanding delivers +/// `Disconnected` (c13): the connection actor synthesizes it on teardown, +/// and the monitor's own read path backstops the case where the actor died +/// with the registration still unread. Must run inside [`run`](crate::run). +pub fn monitor_remote(target: RemotePid) -> RemoteMonitor { + if let Some(local) = target.local() { + let m = monitor(local); + return RemoteMonitor { + id: m.id, + target: target.erase(), + watch: Watch::Local(m), + }; + } + let target = target.erase(); + let (id, route) = with_runtime(|inner| { + let id = inner.alloc_monitor_id(); + let route = inner + .outbound + .lock() + .by_node + .get(&target.node) + .map(|r| (r.incarnation, r.monitors.clone())); + (id, route) + }); + let (tx, rx) = channel::(); + let immediate = match route { + None => Some(RemoteDownReason::Disconnected), + Some((current, _)) if current != target.incarnation => { + Some(RemoteDownReason::Local(crate::monitor::DownReason::NoProc)) + } + Some((_, mon_tx)) => { + let cmd = MonCmd::Monitor { + id, + target: target.clone(), + tx: tx.clone(), + }; + match mon_tx.send(cmd) { + Ok(()) => None, + Err(_) => Some(RemoteDownReason::Disconnected), // actor already gone + } + } + }; + if let Some(reason) = immediate { + let _ = tx.send(RemoteDown { + pid: target.clone(), + reason, + }); + } + RemoteMonitor { + id, + target, + watch: Watch::Remote(rx, Cell::new(RemoteState::Live)), + } +} + +/// Cancel `m`. No future notice will be *sent* for it; a notice already in +/// flight from the peer is dropped on arrival, and one already sitting in +/// `m` is discarded when `m` is dropped (same contract as +/// [`demonitor`]). Unlike the local form this returns nothing: the +/// registration is owned by the connection actor, so whether the `Down` +/// beat the cancel is not local knowledge. Must run inside +/// [`run`](crate::run). +pub fn demonitor_remote(m: &RemoteMonitor) { + match &m.watch { + Watch::Local(local) => { + let _ = demonitor(local); + } + Watch::Remote(_, st) => { + // Cancel first: a channel closing after this is closed, not a + // Disconnected notice — the caller asked for silence. + st.set(RemoteState::Cancelled); + let mon_tx = with_runtime(|inner| { + inner + .outbound + .lock() + .by_node + .get(&m.target.node) + .map(|r| r.monitors.clone()) + }); + if let Some(mon_tx) = mon_tx { + let _ = mon_tx.send(MonCmd::Demonitor { id: m.id }); + } + } + } +} diff --git a/src/monitor.rs b/src/monitor.rs index ac233d3..7aa8ae0 100644 --- a/src/monitor.rs +++ b/src/monitor.rs @@ -146,40 +146,68 @@ pub struct Monitor { pub fn monitor(target: Pid) -> Monitor { let target = target.erase(); let (tx, rx) = channel::(); - - // Implementation note: registration happens under the target's cold - // lock. `tx.clone()` takes the channel's own lock, a Channel-class - // RawMutex, which is explicitly permitted under a Leaf (cold) lock by - // the lock order documented in raw_mutex.rs. We must still not *send* - // under the lock, since `Sender::send` can unpark a parked receiver, - // and there's no reason to nest that. - let (id, registered) = with_runtime(|inner| { - let id = inner.alloc_monitor_id(); - let registered = match inner.slot_at(target) { - Some(slot) => { - let mut cold = slot.cold.lock(); - if slot.is_live_for(target) { - cold.monitors.push((id, tx.clone())); - true - } else { - false - } - } - None => false, - }; - (id, registered) - }); - - if !registered { + let id = with_runtime(|inner| inner.alloc_monitor_id()); + if !register_monitor(target, id, &tx) { let _ = tx.send(Down { pid: target, reason: DownReason::NoProc, }); } - Monitor { id, target, rx } } +/// Register a monitor `id` on `target` that delivers its `Down` to `tx` — the +/// primitive under [`monitor`], split out so a caller can fan many monitors +/// into ONE channel (process groups: every membership's death lands on the +/// reaper's single inbox). Returns `false` if `target` is already gone, in +/// which case nothing is registered and the caller decides what to queue +/// (`monitor` sends `NoProc`). The caller allocates `id` up front so it can +/// record the registration *before* arming it. +/// +/// Implementation note: registration happens under the target's cold lock. +/// `tx.clone()` takes the channel's own lock, a Channel-class RawMutex, which +/// is explicitly permitted under a Leaf (cold) lock by the lock order +/// documented in raw_mutex.rs. We must still not *send* under the lock, since +/// `Sender::send` can unpark a parked receiver, and there's no reason to nest +/// that. +pub(crate) fn register_monitor(target: Pid, id: MonitorId, tx: &Sender) -> bool { + with_runtime(|inner| match inner.slot_at(target) { + Some(slot) => { + let mut cold = slot.cold.lock(); + if slot.is_live_for(target) { + cold.monitors.push((id, tx.clone())); + true + } else { + false + } + } + None => false, + }) +} + +/// Remove registration `id` from `target` — the primitive under +/// [`demonitor`], for callers that hold only the id (see +/// [`register_monitor`]). `None` if the registration is not there: already +/// fired, already removed, or the slot has moved on to a new tenant. +/// +/// The registration is removed under the target's cold lock, but the +/// `Sender` is moved *out* and dropped only after the lock is released. +/// Dropping the last sender runs `Sender::drop`, which may unpark a parked +/// receiver; legal under a cold lock, but pointless to nest. +pub(crate) fn unregister_monitor(target: Pid, id: MonitorId) -> Option { + let removed: Option<(MonitorId, Sender)> = with_runtime(|inner| { + let slot = inner.slot_at(target)?; + let mut cold = slot.cold.lock(); + if slot.generation() != target.generation() { + return None; // slot reused; the Down already fired + } + let pos = cold.monitors.iter().position(|(mid, _)| *mid == id)?; + Some(cold.monitors.remove(pos)) + }); + // `removed`'s sender drops here, outside the lock. + removed.map(|(id, _sender)| id) +} + /// Flag `target`'s tenancy as watchable: its death will stamp the slot's /// terminal record (see [`terminal_reason`]), exactly as registering a name /// does. The bridge calls this wherever a smarm pid is *encoded across the @@ -210,6 +238,23 @@ pub fn mark_watchable(target: Pid) { }); } +/// Whether `target` is live *and* its tenancy is watchable. The cluster's +/// remote-monitor admission check (RFC 010 c12): a peer may monitor a pid only +/// if that pid was exposed or crossed the wire (the D12 set-sites), and a +/// live-but-unwatchable pid answers exactly like a dead one — no liveness leak +/// beyond what `watchable` already grants. Same context contract as +/// [`monitor`]. +#[cfg(feature = "cluster")] +pub(crate) fn is_watchable(target: Pid) -> bool { + let target = target.erase(); + with_runtime(|inner| { + inner.slot_at(target).is_some_and(|slot| { + let cold = slot.cold.lock(); + slot.is_live_for(target) && cold.watchable + }) + }) +} + /// The terminal [`DownReason`] of the tenancy `target` names, if that tenancy /// ever registered a name and is the *most recent named* death of its slot: /// finalize stamps the slot with `(generation, reason)` for once-registered @@ -249,20 +294,5 @@ pub fn terminal_reason(target: Pid) -> Option { /// instead of, or in addition to, calling this: dropping the [`Monitor`] /// closes its receiver and any queued notice is discarded with it. pub fn demonitor(m: &Monitor) -> Option { - // Implementation note: the registration is removed under the target's - // cold lock, but the `Sender` is moved *out* and dropped only after the - // lock is released. Dropping the last sender runs `Sender::drop`, which - // may unpark a parked receiver; legal under a cold lock, but pointless - // to nest. - let removed: Option<(MonitorId, Sender)> = with_runtime(|inner| { - let slot = inner.slot_at(m.target)?; - let mut cold = slot.cold.lock(); - if slot.generation() != m.target.generation() { - return None; // slot reused; the Down already fired - } - let pos = cold.monitors.iter().position(|(mid, _)| *mid == m.id)?; - Some(cold.monitors.remove(pos)) - }); - // `removed`'s sender drops here, outside the lock. - removed.map(|(id, _sender)| id) + unregister_monitor(m.target, m.id) } diff --git a/src/pg.rs b/src/pg.rs index 2b75f6e..03f39e6 100644 --- a/src/pg.rs +++ b/src/pg.rs @@ -99,10 +99,13 @@ //! ## Identity and clustering //! //! A group member is described by a [`Member`] — a [`Pid`] plus a [`NodeId`] and -//! an [`Incarnation`]. Today everything is single-node, those two fields are -//! fixed defaults, and you only ever pass and receive a plain [`Pid`]: the extra -//! identity is carried so this API will not have to change when groups learn to -//! span a cluster. +//! an [`Incarnation`]. Everything on this page is **local**: you pass and +//! receive plain [`Pid`]s, and [`members`] / [`pick`] / [`dispatch`] only ever +//! name actors on this node (Erlang's `get_local_members`). With the `cluster` +//! feature a group also holds the members other nodes have announced, carried +//! under their [`NodeId`]; those never surface here — the cluster-wide reads +//! live in [`cluster::pg`](crate::cluster::pg) (`members_all` and friends) and +//! return a `Local | Remote` member type, since a [`Pid`] cannot hold a remote. //! //! ## Running context //! @@ -110,10 +113,11 @@ //! from inside [`run`](crate::run) (that is, on an actor thread). Calling one //! from outside a running runtime panics. -use crate::monitor::{demonitor, monitor, Monitor}; +use crate::channel::{channel, Sender}; +use crate::monitor::{register_monitor, unregister_monitor, Down, DownReason, MonitorId}; use crate::pid::{assert_type, Addressable, Pid}; use crate::registry::{send_to, SendError}; -use crate::scheduler::with_runtime; +use crate::scheduler::{spawn_under, with_runtime}; use std::collections::HashMap; /// A cluster node handle. A `u32` integer handle, *not* an interned atom — the @@ -186,13 +190,15 @@ pub struct Member { pub pid: Pid, } -/// One membership: a [`Member`] and the [`Monitor`] that watches its liveness. -/// The monitor lives *alongside* the group entry so a group is -/// self-contained: draining the membership tells us whether the member is -/// still alive, and dropping the membership drops its monitor. -struct Membership { - member: Member, - monitor: Monitor, +/// One membership: a [`Member`] and the id of the monitor that watches its +/// liveness. The monitor's `Down` is delivered to the group reaper's single +/// inbox (see [`ProcessGroups::deaths`]), so the membership carries only what +/// [`leave`] needs to tear the registration down: the id. +pub(crate) struct Membership { + pub(crate) member: Member, + /// `None` for a remote member (cluster): the origin node is its liveness + /// authority; nothing here watches it. + pub(crate) monitor: Option, } /// The store: `name → multiset`. Within a single group a `Member` @@ -202,46 +208,60 @@ struct Membership { /// /// Locking discipline. Held under one Leaf-class `RawMutex` on `RuntimeInner`, /// mirroring the registry, and never held together with another Leaf lock (it -/// never touches the registry or a slot's cold lock). The two operations that -/// do need another lock are kept off the group-lock path: +/// never touches the registry or a slot's cold lock). Monitor registration and +/// removal take the target's cold lock (also Leaf), so they run *before* / +/// *after* the group lock, never under it — see [`join`] for the ordering that +/// makes that safe. Nothing under this lock ever touches a channel. /// -/// - `monitor()` / `demonitor()` take the target's cold lock (also Leaf), so -/// they run *before* / *after* the group lock, never under it. -/// - draining a monitor with `try_recv` takes the channel's Channel-class -/// lock, which the lock order permits *under* a Leaf; a channel critical -/// section only does the lock-free unpark protocol, so no Leaf ever nests -/// under it. -/// -/// Evicted and rejected [`Monitor`]s are therefore dropped only *after* the -/// group lock is released, so a receiver-drop never runs a wakeup under the -/// lock — the same discipline as `demonitor`. +/// Eviction is *eager*: every membership's monitor delivers to the one +/// `deaths` channel, drained by a per-run reaper actor that sweeps the dead +/// pid out of every group the moment its `Down` is scheduled. The read path +/// keeps a slot-liveness backstop for the window between a death and the +/// reaper's turn. pub(crate) struct ProcessGroups { groups: HashMap>, + /// The reaper's inboxes: every membership monitor is registered against + /// a clone of `deaths`. `None` until the first `join` of a run spawns + /// the reaper; a stale one (receiver gone with the previous run's + /// teardown) is detected via `receiver_alive` and replaced. + reaper: Option, + /// `NodeId → node name` for every peer with members in the store, kept + /// by the pg actor under this lock, so a stored remote member can be + /// rendered back to its wire identity without asking anyone. + #[cfg(feature = "cluster")] + node_names: HashMap, } impl ProcessGroups { pub(crate) fn new() -> Self { Self { groups: HashMap::new(), + reaper: None, + #[cfg(feature = "cluster")] + node_names: HashMap::new(), } } - /// Insert `ms` into `group`. Idempotent on the *member*: if the member is - /// already present the new membership is handed back (`Some`) so the caller - /// can tear its now-redundant monitor down outside the lock; `None` means - /// it was inserted. - fn join(&mut self, group: &str, ms: Membership) -> Option { + /// Forget the reaper. Called at the start of every `run()` so a stopped + /// reaper from a previous run is never sent to; `join` respawns. + pub(crate) fn reset_reaper(&mut self) { + self.reaper = None; + } + + /// Insert `ms` into `group`. Idempotent on the *member*: `false` means the + /// member was already present and nothing changed; `true` means inserted. + pub(crate) fn join(&mut self, group: &str, ms: Membership) -> bool { let v = self.groups.entry(group.to_owned()).or_default(); if v.iter().any(|e| e.member == ms.member) { - return Some(ms); + return false; } v.push(ms); - None + true } /// Remove `member`'s membership from `group`, returning it (so the caller - /// can `demonitor` it outside the lock). An emptied group is pruned. - fn leave(&mut self, group: &str, member: Member) -> Option { + /// can unregister its monitor outside the lock). An emptied group is pruned. + pub(crate) fn leave(&mut self, group: &str, member: Member) -> Option { let v = self.groups.get_mut(group)?; let pos = v.iter().position(|e| e.member == member)?; let removed = v.remove(pos); @@ -252,20 +272,23 @@ impl ProcessGroups { } /// The one dumb eviction primitive: drop every member matching `pred` from - /// every group, pruning emptied groups, and return the evicted memberships' - /// monitors for the caller to drop outside the lock. The primitive does not - /// know *why* a member leaves; that is the caller's concern. Its callers are - /// the death hook (`reap_group`) and, once clustering lands, an - /// incarnation-eviction sweep — both over this same predicate path, which is - /// the whole reason to shape eviction as a predicate. Insertion order within - /// a group is preserved (`members` / `pick` are order-stable). - fn remove_where(&mut self, mut pred: impl FnMut(&Member) -> bool) -> Vec { + /// every group, pruning emptied groups, and return the evicted + /// memberships with the group each was in. The primitive does not know + /// *why* a member leaves; that is the caller's concern. Its callers are + /// the reaper (a local death) and the cluster's node-down / re-sync + /// sweeps — all over this same predicate path, which is the whole reason + /// to shape eviction as a predicate. Insertion order within a group is + /// preserved (`members` / `pick` are order-stable). + pub(crate) fn remove_where( + &mut self, + mut pred: impl FnMut(&Member) -> bool, + ) -> Vec<(String, Membership)> { let mut evicted = Vec::new(); - self.groups.retain(|_, v| { + self.groups.retain(|g, v| { let mut i = 0; while i < v.len() { if pred(&v[i].member) { - evicted.push(v.remove(i).monitor); + evicted.push((g.clone(), v.remove(i))); } else { i += 1; } @@ -275,38 +298,6 @@ impl ProcessGroups { evicted } - /// Drain-on-contact death hook. The registry can prune a stale binding - /// lazily, on contact, because it only ever resolves one binding at a time; - /// a group is *iterated* — `members` fans out to everyone — so it must not - /// carry a dead member across a broadcast. Every group operation reaps the - /// group it touches first. - /// - /// Drains every membership monitor in `group` with a non-blocking - /// `try_recv`: a delivered `Down` (any reason) or a closed channel means - /// that member is dead. On the first death detected, sweep *all* of the - /// dead pids out of *every* group via [`remove_where`] — a death is removed - /// from each group it joined, not just the one being touched. Returns the - /// evicted monitors to drop outside the lock. - fn reap_group(&mut self, group: &str) -> Vec { - let dead: Vec = { - let Some(v) = self.groups.get(group) else { - return Vec::new(); - }; - v.iter() - .filter_map(|e| match e.monitor.rx.try_recv() { - // A Down arrived, or the channel closed and drained: dead. - Ok(Some(_)) | Err(_) => Some(e.member.pid), - // Empty but open — the sender still lives in the slot: alive. - Ok(None) => None, - }) - .collect() - }; - if dead.is_empty() { - return Vec::new(); - } - self.remove_where(|m| dead.contains(&m.pid)) - } - /// Raw enumeration of a group's members — no liveness filtering. Used by /// tests to assert storage state independently of the read-path backstop. #[cfg(test)] @@ -317,16 +308,24 @@ impl ProcessGroups { .unwrap_or_default() } - /// Live members of `group`, in insertion order. The `is_live` oracle is the - /// read-path backstop: a member whose slot is already dead is - /// dropped from the *result* even if its `Down` has not been drained yet. - /// Backstop only — the entry stays in storage; eviction is the monitor's - /// job (`reap_group`). - fn members_where(&self, group: &str, mut is_live: impl FnMut(Pid) -> bool) -> Vec { + /// Live members of `group` **on `node`**, in insertion order. The + /// `is_live` oracle is the read-path backstop: a member whose slot is + /// already dead is dropped from the *result* even if the reaper has not + /// swept it yet. Backstop only — the entry stays in storage; eviction is + /// the reaper's job. The node filter is what keeps the local API local: + /// a remote member's `pid` is another node's slot bits, meaningless to + /// `is_live` and to any local send. + fn members_where( + &self, + group: &str, + node: NodeId, + mut is_live: impl FnMut(Pid) -> bool, + ) -> Vec { self.groups .get(group) .map(|v| { v.iter() + .filter(|e| e.member.node == node) .map(|e| e.member.pid) .filter(|&p| is_live(p)) .collect() @@ -334,19 +333,210 @@ impl ProcessGroups { .unwrap_or_default() } - /// The first live member of `group` in insertion order — stateless - /// first-live `pick`, with the same read-path backstop as `members_where`. - fn first_member_where(&self, group: &str, mut is_live: impl FnMut(Pid) -> bool) -> Option { + /// The first live member of `group` on `node` in insertion order — + /// stateless first-live `pick`, with the same read-path backstop and node + /// filter as `members_where`. + fn first_member_where( + &self, + group: &str, + node: NodeId, + mut is_live: impl FnMut(Pid) -> bool, + ) -> Option { self.groups .get(group)? .iter() + .filter(|e| e.member.node == node) .map(|e| e.member.pid) .find(|&p| is_live(p)) } } +/// The store's cluster-side surface: raw reads the pg actor needs to speak +/// for this node (`Sync`, membership checks) and the peer-name memo. One +/// `cfg` block: everything here exists only when there is a mesh. +#[cfg(feature = "cluster")] +impl ProcessGroups { + /// Does `group` hold `member` right now? (Raw storage, no liveness.) + pub(crate) fn contains(&self, group: &str, member: &Member) -> bool { + self.groups + .get(group) + .is_some_and(|v| v.iter().any(|e| e.member == *member)) + } + + /// Every stored member of `group`, any node, insertion order. Raw storage. + pub(crate) fn all_of(&self, group: &str) -> Vec { + self.groups + .get(group) + .map(|v| v.iter().map(|e| e.member).collect()) + .unwrap_or_default() + } + + /// `(group, [pid])` for every group with a member on `node` — the + /// `Sync` payload. Raw storage; groups with no such member are omitted. + pub(crate) fn groups_on(&self, node: NodeId) -> Vec<(String, Vec)> { + let mut out: Vec<(String, Vec)> = self + .groups + .iter() + .filter_map(|(g, v)| { + let pids: Vec = v + .iter() + .filter(|e| e.member.node == node) + .map(|e| e.member.pid) + .collect(); + (!pids.is_empty()).then(|| (g.clone(), pids)) + }) + .collect(); + out.sort_by(|a, b| a.0.cmp(&b.0)); + out + } + + /// Record / forget the name behind a peer's `NodeId`. + pub(crate) fn set_node_name(&mut self, node: NodeId, name: String) { + self.node_names.insert(node, name); + } + pub(crate) fn forget_node_name(&mut self, node: NodeId) { + self.node_names.remove(&node); + } + pub(crate) fn node_name(&self, node: NodeId) -> Option<&str> { + self.node_names.get(&node).map(String::as_str) + } +} + +/// The group reaper: one detached actor per run, spawned by the first `join`, +/// parked on the shared `deaths` inbox. Every local membership's monitor +/// delivers here, so a death is swept out of *every* group it joined as soon +/// as the reaper is scheduled — no group operation has to happen first. +/// Sweeps by `(node, pid)`: only local members, since a remote member's pid +/// bits are meaningless here. Exits when the last sender is gone, i.e. never +/// during a run (the store holds one); the run's teardown stops it like any +/// other parked actor. Spawned under `ROOT_PID` so its exit signal is absorbed +/// rather than delivered to whichever supervisor's child happened to join +/// first. +/// +/// Under `cluster` the same actor is the node's **pg actor** (RFC 010 Phase +/// 5, c15): it also drains a control inbox of local join/leave announcements, +/// the membership stream and the exposed `"pg"` inbox — see +/// [`crate::cluster::pg`]. Its store-side sweep is unchanged. +#[cfg(not(feature = "cluster"))] +fn reaper(rx: crate::channel::Receiver, ctl: crate::channel::Receiver) { + // No mesh: nothing to tell about joins/leaves. Drop the control inbox + // so announcements are refused at the sender rather than queued. + drop(ctl); + while let Ok(down) = rx.recv() { + sweep_local_death(down.pid); + // Evicted memberships hold only ids; their monitors have fired. + } +} + +/// What the local API tells the reaper besides deaths (which arrive as +/// [`Down`] on their own inbox — that channel's type is fixed by the monitor +/// primitive, so the two cannot be one enum). The default reaper has no use +/// for these; the cluster's pg actor broadcasts them (RFC 010 Phase 5). +// The default reaper never looks inside — that is the point, not a bug. +#[cfg_attr(not(feature = "cluster"), allow(dead_code))] +pub(crate) enum PgEvent { + /// `join` inserted `pid` into `group`. The consumer re-checks the store + /// before acting on it. + Joined { group: String, pid: Pid }, + /// `leave` removed `pid` from `group`. + Left { group: String, pid: Pid }, + /// `cluster::start` has the manager up and the local identity set: take + /// a membership subscription, register + expose the `"pg"` name, and + /// start speaking to peers. + #[cfg(feature = "cluster")] + Attach, +} + +/// Evict the local member `pid` from every group. The reaper's one store +/// operation; returns what was evicted with its group (the cluster's +/// `Leave` broadcast wants both). +pub(crate) fn sweep_local_death(pid: Pid) -> Vec<(String, Membership)> { + with_runtime(|inner| { + let node = inner.node_id; + inner + .process_groups + .lock() + .remove_where(|m| m.node == node && m.pid == pid) + }) +} + +/// The reaper's inboxes. `deaths` is the liveness authority for the set +/// (`ctl` is created and dropped with it, on the same actor). +#[derive(Clone)] +pub(crate) struct ReaperInboxes { + pub(crate) deaths: Sender, + /// The control inbox: local `join`/`leave` announce here (see + /// [`PgEvent`]). The default reaper closes it on entry. + pub(crate) ctl: Sender, +} + +impl ReaperInboxes { + fn alive(&self) -> bool { + self.deaths.receiver_alive() + } +} + +/// Live senders for the reaper's inboxes, spawning the reaper if this run has +/// none yet. Two racing first-spawns may both spawn; the loser's senders drop +/// on return, its spare reaper sees a closed inbox and exits. +pub(crate) fn reaper_inboxes() -> ReaperInboxes { + let existing = with_runtime(|inner| { + let pg = inner.process_groups.lock(); + pg.reaper.clone().filter(ReaperInboxes::alive) + }); + if let Some(r) = existing { + return r; + } + let (tx, rx) = channel::(); + let (ctl_tx, ctl_rx) = channel::(); + // Detached: the handle drops here. The reaper's lifetime is the run's. + // The ONE seam between the local store and the cluster: same inboxes, + // different body. + #[cfg(not(feature = "cluster"))] + let _ = spawn_under(crate::runtime::ROOT_PID, move || reaper(rx, ctl_rx)); + #[cfg(feature = "cluster")] + let _ = spawn_under(crate::runtime::ROOT_PID, move || { + crate::cluster::pg::actor(rx, ctl_rx) + }); + let fresh = ReaperInboxes { + deaths: tx, + ctl: ctl_tx, + }; + with_runtime(|inner| { + let mut pg = inner.process_groups.lock(); + match &pg.reaper { + Some(r) if r.alive() => r.clone(), + _ => { + pg.reaper = Some(fresh.clone()); + fresh + } + } + }) +} + +/// A live sender for the reaper's `deaths` inbox (spawning it if needed). +fn deaths_sender() -> Sender { + reaper_inboxes().deaths +} + +/// Announce a local group change to the reaper, if this run has one. A +/// closed inbox is the default reaper (uninterested) or a run tearing down. +fn announce(msg: PgEvent) { + let ctl = with_runtime(|inner| { + inner + .process_groups + .lock() + .reaper + .as_ref() + .map(|r| r.ctl.clone()) + }); + if let Some(ctl) = ctl { + let _ = ctl.send(msg); + } +} + /// Build the full member identity for `pid` from runtime identity. -fn member_for(inner: &crate::runtime::RuntimeInner, pid: Pid) -> Member { +pub(crate) fn member_for(inner: &crate::runtime::RuntimeInner, pid: Pid) -> Member { Member { node: inner.node_id, incarnation: inner.incarnation, @@ -358,7 +548,7 @@ fn member_for(inner: &crate::runtime::RuntimeInner, pid: Pid) -> Member { /// no lock — identical to the registry's guard. The read-path backstop: a /// generation is never reused, so a dead member is detectable independently of /// whether its monitor `Down` has been drained yet. -fn live(inner: &crate::runtime::RuntimeInner, pid: Pid) -> bool { +pub(crate) fn live(inner: &crate::runtime::RuntimeInner, pid: Pid) -> bool { inner.slot_at(pid).is_some_and(|s| s.is_live_for(pid)) } @@ -367,61 +557,76 @@ fn live(inner: &crate::runtime::RuntimeInner, pid: Pid) -> bool { /// added the membership, `false` if it was already a member. /// /// Installs a monitor on `pid` so the actor's death evicts it from the group -/// automatically — you never have to remove a dead member yourself. A redundant -/// (idempotent) join tears its extra monitor back down. +/// automatically — you never have to remove a dead member yourself. Joining a +/// pid that is already dead is accepted and evicted the same way (via a +/// `NoProc` notice), so it never shows up in a read. /// /// Panics if called outside `Runtime::run()`. pub fn join(group: impl Into, pid: Pid) -> bool { let group = group.into(); let pid = pid.erase(); - // Install the monitor BEFORE taking the group lock: monitor() acquires the - // target's cold lock (Leaf), and two Leaf locks are never held at once. The - // registration races `finalize_actor` under that cold lock exactly as every - // other monitor does, so no death can slip between the join and the monitor - // being in place. - let mon = monitor(pid); - - let (rejected, reaped) = with_runtime(|inner| { + let deaths = deaths_sender(); + // Record the membership BEFORE arming its monitor: the reaper sweeps by + // pid on the first `Down`, so a `Down` that could precede the entry would + // leave a corpse in storage forever (visible to no read — the backstop + // hides it — but a leak, and once groups are clustered a member that + // would be announced). Arming after insertion means every `Down` finds + // its entry. The monitor id is allocated up front so `leave` can tear the + // registration down even if it lands in the tiny window before arming (an + // orphaned registration is harmless: its `Down` names a pid whose + // membership is gone, and the sweep finds nothing). + let id = with_runtime(|inner| inner.alloc_monitor_id()); + let inserted = with_runtime(|inner| { let ms = Membership { member: member_for(inner, pid), - monitor: mon, + monitor: Some(id), }; - let mut pg = inner.process_groups.lock(); - let reaped = pg.reap_group(&group); - let rejected = pg.join(&group, ms); - (rejected, reaped) + inner.process_groups.lock().join(&group, ms) + }); + if !inserted { + return false; + } + // Tell the reaper (the cluster's pg actor re-checks the store before it + // broadcasts, so a `leave`/death that overtakes this announcement is + // never advertised as a join). + announce(PgEvent::Joined { + group: group.clone(), + pid, }); - // Outside the group lock: drop the reaped (dead) monitors, and if this join - // was redundant, demonitor + drop the extra monitor we just installed. - drop(reaped); - match rejected { - Some(dup) => { - demonitor(&dup.monitor); - false - } - None => true, + // Outside the group lock: registration takes the target's cold lock (Leaf). + // The registration races `finalize_actor` under that cold lock exactly as + // every other monitor does, so no death can slip between the join and the + // monitor being in place. + if !register_monitor(pid, id, &deaths) { + // Already gone: queue the notice ourselves, exactly as `monitor` does. + let _ = deaths.send(Down { + pid, + reason: DownReason::NoProc, + }); } + true } /// Drop `pid`'s membership of `group`. Returns whether a membership was -/// removed. The membership's monitor is demonitored and dropped. +/// removed. The membership's monitor registration is torn down. /// /// Panics if called outside `Runtime::run()`. pub fn leave(group: &str, pid: Pid) -> bool { let pid = pid.erase(); - let (removed, reaped) = with_runtime(|inner| { + let removed = with_runtime(|inner| { let member = member_for(inner, pid); - let mut pg = inner.process_groups.lock(); - let reaped = pg.reap_group(group); - let removed = pg.leave(group, member); - (removed, reaped) + inner.process_groups.lock().leave(group, member) }); - - drop(reaped); match removed { Some(ms) => { - demonitor(&ms.monitor); + if let Some(id) = ms.monitor { + unregister_monitor(pid, id); + } + announce(PgEvent::Left { + group: group.to_owned(), + pid, + }); true } None => false, @@ -431,21 +636,19 @@ pub fn leave(group: &str, pid: Pid) -> bool { /// Every live member of `group`, in the order they joined. Returns an empty /// vector if the group does not exist or has no live members. /// -/// Dead members are never returned: the group is pruned of anything that has -/// died before the read, and as a backstop a member whose slot is already dead -/// is dropped from the result even in the brief window before its death has -/// been fully processed. +/// Dead members are never returned: the reaper evicts a member as soon as its +/// death is processed, and as a backstop a member whose slot is already dead +/// is dropped from the result even in the brief window before the reaper's +/// turn. /// /// Panics if called outside `Runtime::run()`. pub fn members(group: &str) -> Vec { - let (pids, reaped) = with_runtime(|inner| { - let mut pg = inner.process_groups.lock(); - let reaped = pg.reap_group(group); - let pids = pg.members_where(group, |pid| live(inner, pid)); - (pids, reaped) - }); - drop(reaped); - pids + with_runtime(|inner| { + inner + .process_groups + .lock() + .members_where(group, inner.node_id, |pid| live(inner, pid)) + }) } /// One live member of `group`, or `None` if the group is empty (or every @@ -455,14 +658,12 @@ pub fn members(group: &str) -> Vec { /// /// Panics if called outside `Runtime::run()`. pub fn pick(group: &str) -> Option { - let (picked, reaped) = with_runtime(|inner| { - let mut pg = inner.process_groups.lock(); - let reaped = pg.reap_group(group); - let picked = pg.first_member_where(group, |pid| live(inner, pid)); - (picked, reaped) - }); - drop(reaped); - picked + with_runtime(|inner| { + inner + .process_groups + .lock() + .first_member_where(group, inner.node_id, |pid| live(inner, pid)) + }) } /// Typed [`pick`]: one live member of `group` as a [`Pid`](Pid). @@ -505,8 +706,8 @@ pub fn dispatch(group: &str, msg: A::Msg) -> Result, Send #[cfg(test)] mod tests { use super::*; - use crate::channel::{channel, Sender}; - use crate::monitor::{Down, DownReason, MonitorId}; + use crate::scheduler::spawn; + use std::time::{Duration, Instant}; fn member(index: u32, generation: u32) -> Member { Member { @@ -516,33 +717,21 @@ mod tests { } } - /// A synthetic membership with a real (but slot-less) monitor channel. The - /// returned `Sender` stands in for the slot's `Down` sender: hold it to - /// keep the member "alive" (`try_recv` → `Ok(None)`), `send` a `Down` to - /// simulate death, or `drop` it to simulate a drained/closed channel. - fn synth(index: u32, generation: u32) -> (Membership, Sender) { - let pid = Pid::new(index, generation); - let (tx, rx) = channel::(); - let ms = Membership { + /// A synthetic membership: the store never looks at the id. + fn synth(index: u32, generation: u32) -> Membership { + Membership { member: member(index, generation), - monitor: Monitor { - id: MonitorId(0), - target: pid, - rx, - }, - }; - (ms, tx) + monitor: Some(MonitorId(0)), + } } #[test] fn join_is_idempotent_within_a_group() { let mut pg = ProcessGroups::new(); - let (a, _ta) = synth(1, 0); - let (b, _tb) = synth(1, 0); - assert!(pg.join("workers", a).is_none(), "first join inserts"); + assert!(pg.join("workers", synth(1, 0)), "first join inserts"); assert!( - pg.join("workers", b).is_some(), - "second identical join is handed back" + !pg.join("workers", synth(1, 0)), + "second identical join is refused" ); assert_eq!(pg.members_of("workers"), vec![member(1, 0)]); } @@ -550,12 +739,9 @@ mod tests { #[test] fn same_pid_in_many_groups_is_independent() { let mut pg = ProcessGroups::new(); - let (a, _ta) = synth(1, 0); - let (b, _tb) = synth(1, 0); - let (c, _tc) = synth(2, 0); - pg.join("a", a); - pg.join("b", b); - pg.join("b", c); + pg.join("a", synth(1, 0)); + pg.join("b", synth(1, 0)); + pg.join("b", synth(2, 0)); assert_eq!(pg.members_of("a"), vec![member(1, 0)]); assert_eq!(pg.members_of("b"), vec![member(1, 0), member(2, 0)]); } @@ -564,11 +750,9 @@ mod tests { fn distinct_generations_are_distinct_members() { // ABA guard: same slot index, different generation = different actor. let mut pg = ProcessGroups::new(); - let (a, _ta) = synth(1, 0); - let (b, _tb) = synth(1, 1); - assert!(pg.join("g", a).is_none()); + assert!(pg.join("g", synth(1, 0))); assert!( - pg.join("g", b).is_none(), + pg.join("g", synth(1, 1)), "different generation is a distinct member" ); assert_eq!(pg.members_of("g"), vec![member(1, 0), member(1, 1)]); @@ -577,10 +761,8 @@ mod tests { #[test] fn leave_removes_one_membership_and_prunes_empty_groups() { let mut pg = ProcessGroups::new(); - let (a, _ta) = synth(1, 0); - let (b, _tb) = synth(2, 0); - pg.join("g", a); - pg.join("g", b); + pg.join("g", synth(1, 0)); + pg.join("g", synth(2, 0)); assert!(pg.leave("g", member(1, 0)).is_some()); assert_eq!(pg.members_of("g"), vec![member(2, 0)]); assert!( @@ -598,7 +780,7 @@ mod tests { #[test] fn remove_where_sweeps_every_group() { let mut pg = ProcessGroups::new(); - for (g, (m, _t)) in [ + for (g, m) in [ ("a", synth(1, 0)), ("a", synth(2, 0)), ("b", synth(1, 0)), @@ -616,104 +798,137 @@ mod tests { #[test] fn remove_where_can_match_an_incarnation_sweep() { - // Shape check for the later evict_incarnation(node, inc) caller. + // Shape check for the node-down / incarnation sweep caller. let mut pg = ProcessGroups::new(); - let pid = Pid::new(1, 0); - let (tx, rx) = channel::(); - let dead = Membership { + let stale = Membership { member: Member { node: DEFAULT_NODE_ID, incarnation: Incarnation::new(7), - pid, - }, - monitor: Monitor { - id: MonitorId(0), - target: pid, - rx, + pid: Pid::new(1, 0), }, + monitor: Some(MonitorId(0)), }; - let _keep = tx; - let (live, _tl) = synth(2, 0); - pg.join("g", dead); - pg.join("g", live); + pg.join("g", stale); + pg.join("g", synth(2, 0)); let evicted = pg.remove_where(|mem| mem.incarnation == Incarnation::new(7)); assert_eq!(evicted.len(), 1); assert_eq!(pg.members_of("g"), vec![member(2, 0)]); } #[test] - fn reap_keeps_live_members() { + fn read_backstop_hides_a_member_the_reaper_has_not_yet_swept() { let mut pg = ProcessGroups::new(); - let (a, _ta) = synth(1, 0); // sender held: member stays alive - pg.join("a", a); - assert!(pg.reap_group("a").is_empty(), "no deaths"); - assert_eq!(pg.members_of("a"), vec![member(1, 0)]); - } - - #[test] - fn reap_evicts_a_dead_member_and_sweeps_all_its_groups() { - let mut pg = ProcessGroups::new(); - let (a1, ta1) = synth(1, 0); // pid 1 in group a - let (a2, _ta2) = synth(2, 0); // pid 2 in group a (stays alive) - let (b1, _tb1) = synth(1, 0); // pid 1 in group b - pg.join("a", a1); - pg.join("a", a2); - pg.join("b", b1); - // pid 1 dies: its group-a monitor receives a Down. Its group-b monitor - // has not — reap must still sweep pid 1 out of b by the pid predicate. - ta1.send(Down { - pid: Pid::new(1, 0), - reason: DownReason::Exit, - }) - .unwrap(); - let evicted = pg.reap_group("a"); - assert_eq!( - evicted.len(), - 2, - "pid 1's memberships in both a and b are evicted" - ); - assert_eq!(pg.members_of("a"), vec![member(2, 0)]); - assert!(pg.members_of("b").is_empty(), "swept from b too; pruned"); - } - - #[test] - fn reap_treats_a_closed_channel_as_dead() { - let mut pg = ProcessGroups::new(); - let (a, ta) = synth(1, 0); - pg.join("a", a); - drop(ta); // sender gone, queue empty → try_recv = Err(RecvError) = dead - let evicted = pg.reap_group("a"); - assert_eq!(evicted.len(), 1); - assert!(pg.members_of("a").is_empty()); - } - - #[test] - fn read_backstop_hides_a_member_the_monitor_has_not_yet_reaped() { - let mut pg = ProcessGroups::new(); - // Both senders held: reap_group would see Ok(None) and evict neither. - let (a, _ta) = synth(1, 0); - let (b, _tb) = synth(2, 0); - pg.join("g", a); - pg.join("g", b); + pg.join("g", synth(1, 0)); + pg.join("g", synth(2, 0)); // The slot-word oracle already reports pid 1 dead (finalize window), - // ahead of any Down delivery. + // ahead of the reaper's turn. let dead = Pid::new(1, 0); let oracle = |pid: Pid| pid != dead; assert_eq!( - pg.members_where("g", oracle), + pg.members_where("g", DEFAULT_NODE_ID, oracle), vec![Pid::new(2, 0)], "dead pid filtered from read" ); assert_eq!( - pg.first_member_where("g", oracle), + pg.first_member_where("g", DEFAULT_NODE_ID, oracle), Some(Pid::new(2, 0)), "pick skips the dead first member" ); - // Backstop does not evict — that stays the monitor's job; raw storage - // still holds both until reap runs. + // Backstop does not evict — that stays the reaper's job; raw storage + // still holds both until it runs. assert_eq!(pg.members_of("g"), vec![member(1, 0), member(2, 0)]); } + + // ---- reaper: eager eviction against a live runtime ---- + + /// Raw storage view for a group, bypassing the read-path backstop. + fn stored(group: &str) -> Vec { + with_runtime(|inner| inner.process_groups.lock().members_of(group)) + } + + /// Cooperative wait (`smarm::sleep`, never an OS block) until `pred`. + fn wait_until(what: &str, mut pred: impl FnMut() -> bool) { + let deadline = Instant::now() + Duration::from_secs(2); + while !pred() { + assert!(Instant::now() < deadline, "timed out waiting for: {what}"); + crate::sleep(Duration::from_millis(1)); + } + } + + #[test] + fn a_death_is_swept_from_storage_without_any_group_operation() { + crate::run(|| { + let (tx, rx) = channel::<()>(); + let w = spawn(move || { + rx.recv().unwrap(); + }); + let pid = w.pid(); + join("a", pid); + join("b", pid); + assert_eq!(stored("a"), vec![member_for_test(pid)]); + + tx.send(()).unwrap(); + w.join().unwrap(); + // No members()/pick()/join() on a or b from here on: the reaper + // alone must clear both. + wait_until("reaper sweeps a and b", || { + stored("a").is_empty() && stored("b").is_empty() + }); + }); + } + + #[test] + fn a_dead_at_join_pid_is_swept_from_storage() { + crate::run(|| { + let h = spawn(|| {}); + let pid = h.pid(); + h.join().unwrap(); + assert!(join("late", pid), "join is accepted; eviction is uniform"); + wait_until("reaper sweeps the NoProc member", || { + stored("late").is_empty() + }); + }); + } + + #[test] + fn leave_then_death_does_not_disturb_a_rejoined_group() { + // A monitor unregistered by `leave` must not fire later; the pid's + // fresh membership after re-join is swept exactly once, by its own + // monitor, on death. + crate::run(|| { + let (tx, rx) = channel::<()>(); + let w = spawn(move || { + rx.recv().unwrap(); + }); + let pid = w.pid(); + join("g", pid); + assert!(leave("g", pid)); + assert!(join("g", pid)); + assert_eq!(members("g"), vec![pid]); + tx.send(()).unwrap(); + w.join().unwrap(); + wait_until("reaper sweeps g", || stored("g").is_empty()); + }); + } + + #[test] + fn reaper_is_respawned_for_a_second_run_of_the_same_runtime() { + let rt = crate::runtime::init(crate::runtime::Config::exact(1)); + let body = || { + let h = spawn(|| {}); + let pid = h.pid(); + h.join().unwrap(); + join("g", pid); + wait_until("reaper sweeps g", || stored("g").is_empty()); + }; + rt.run(body); + rt.run(body); + } + + fn member_for_test(pid: Pid) -> Member { + with_runtime(|inner| member_for(inner, pid)) + } } diff --git a/src/pid.rs b/src/pid.rs index 96b3f21..2f5c0fb 100644 --- a/src/pid.rs +++ b/src/pid.rs @@ -292,3 +292,44 @@ mod typed_pid_tests { assert_send_sync::>(); } } + +// ---- RFC 010 c10: pids auto-serialize (cluster feature) --------------------- + +/// A local `Pid` serializes as a +/// [`RemotePid`](crate::cluster::remote::RemotePid): the wire form stamps +/// this node's name and incarnation from the ambient runtime, so a pid can +/// sit inside any message field and reply-to needs no ceremony (RFC 010 §3, +/// "sugar not a bear trap"). Serializing a pid also marks it **watchable** +/// — the wire crossing is the cluster's `mark_watchable` set-site (D12), the +/// exact analog of the membrane crossing. +/// +/// Must run inside `run()` (the ambient identity lives on the runtime); a +/// runtime without a cluster identity cannot serialize a pid at all — it is +/// a serialize error, surfacing as the send's `Encode` failure — rather than +/// a `("", 0)` stamp that every peer would silently drop. +#[cfg(feature = "cluster")] +impl serde::Serialize for Pid { + fn serialize(&self, s: S) -> Result { + crate::cluster::remote::RemotePid::::from_local(*self) + .ok_or_else(|| serde::ser::Error::custom("pid serialized with no local node identity"))? + .serialize(s) + } +} + +/// Deserializing into a `Pid` is the **collapse**: it succeeds only when +/// the wire pid names this very node (name and incarnation both), and is a +/// decode error otherwise — a foreign pid cannot become a local `Pid`. +/// Fields that may hold a pid from anywhere are `RemotePid`. +#[cfg(feature = "cluster")] +impl<'de, A: 'static> serde::Deserialize<'de> for Pid { + fn deserialize>(d: D) -> Result { + let rp = crate::cluster::remote::RemotePid::::deserialize(d)?; + rp.local().ok_or_else(|| { + serde::de::Error::custom(format!( + "pid {}@{} is not local to this node", + rp.index(), + rp.node() + )) + }) + } +} diff --git a/src/runtime.rs b/src/runtime.rs index 911114c..6cd4f82 100644 --- a/src/runtime.rs +++ b/src/runtime.rs @@ -1358,6 +1358,9 @@ impl Runtime { self.inner.root_exited.store(false, Ordering::Relaxed); self.inner.root_swept.store(false, Ordering::Relaxed); self.inner.set_root(initial_handle.pid()); + // A previous run's group reaper was stopped with that run; forget it + // so the first `join` of this run spawns a fresh one. + self.inner.process_groups.lock().reset_reaper(); // Launch N-1 extra scheduler threads, named `smarm-sched-{slot}` so // they are identifiable in `/proc//task/*/comm`, stack dumps and diff --git a/src/trace.rs b/src/trace.rs index 923cbbd..180828e 100644 --- a/src/trace.rs +++ b/src/trace.rs @@ -65,6 +65,13 @@ mod inner { // RFC 005 wake slot SlotPush(Pid), // actor-context wake parked in the waking thread's slot SlotPop(Pid), // scheduler resumed a pid from its own slot + // Cluster (RFC 010): the conn actor's verdict on one inbound frame — + // local knowledge only, never on the wire; the label is + // `InboundVerdict::label()`. No pid: a refused frame has none. + ClusterInbound(&'static str), + // Cluster (RFC 010): the connector's verdict on one dial attempt — + // `"ok"` or `DialError::label()`. No pid. + ClusterDial(&'static str), } // ----------------------------------------------------------------------- @@ -271,6 +278,8 @@ mod inner { Event::Dequeue(p) => ("dequeue".into(), p.index()), Event::SlotPush(p) => ("slot_push".into(), p.index()), Event::SlotPop(p) => ("slot_pop".into(), p.index()), + Event::ClusterInbound(v) => (format!("cluster_inbound {v}"), 0), + Event::ClusterDial(v) => (format!("cluster_dial {v}"), 0), } } diff --git a/tests/channel.rs b/tests/channel.rs index cc92e95..e3bc1d4 100644 --- a/tests/channel.rs +++ b/tests/channel.rs @@ -137,11 +137,18 @@ fn channel_ops_interleaved_with_monitor_churn_multi_thread() { for i in 0..32i64 { let tx = tx.clone(); handles.push(spawn(move || { - // Short-lived target whose death fires the monitor below. + // Short-lived target whose death fires the monitor below. It + // is gated: on a multi-thread scheduler it could otherwise + // run and exit before `monitor` registers, and monitoring a + // corpse queues `NoProc` by contract — the point here is a + // `Down` sent from finalize, so register first, then release. + let (go_tx, go_rx) = channel::<()>(); let t = spawn(move || { + let _ = go_rx.recv(); tx.send(i).unwrap(); }); let m = smarm::monitor(t.pid()); + go_tx.send(()).unwrap(); t.join().unwrap(); // Down delivery exercises send-from-finalize. let d = m.rx.recv().unwrap(); diff --git a/tests/cluster_conn_lifecycle.rs b/tests/cluster_conn_lifecycle.rs index eff1b93..f989ddc 100644 --- a/tests/cluster_conn_lifecycle.rs +++ b/tests/cluster_conn_lifecycle.rs @@ -21,6 +21,7 @@ use smarm::cluster::manager::{Call, Manager, Reply, MANAGER}; use smarm::cluster::spawn_established; use smarm::cluster::transport::tcp::TcpTransport; use smarm::cluster::transport::{Conn, FramedConn, Transport}; +use smarm::cluster::Timing; use smarm::gen_server::{self, GenServerBuilder}; use smarm::pg::Incarnation; use smarm::{run, sleep}; @@ -81,8 +82,10 @@ fn connection_up_commanded_shutdown_and_eof_all_reflected_in_table() { // Manage the `a` ends as peers node-b and node-c; keep the `b` far ends // open so neither socket is closed from the far side yet. - spawn_established(FramedConn::new(a1), peer("node-b")).expect("node-b registers"); - spawn_established(FramedConn::new(a2), peer("node-c")).expect("node-c registers"); + spawn_established(FramedConn::new(a1), peer("node-b"), Timing::default()) + .expect("node-b registers"); + spawn_established(FramedConn::new(a2), peer("node-c"), Timing::default()) + .expect("node-c registers"); // Up: both connections register and the table shows them. wait_peers(&["node-b", "node-c"]); diff --git a/tests/cluster_conn_liveness.rs b/tests/cluster_conn_liveness.rs index ba2aea0..5104efc 100644 --- a/tests/cluster_conn_liveness.rs +++ b/tests/cluster_conn_liveness.rs @@ -22,6 +22,7 @@ use smarm::cluster::manager::{Call, Manager, Reply, MANAGER}; use smarm::cluster::spawn_established; use smarm::cluster::transport::tcp::TcpTransport; use smarm::cluster::transport::{Conn, FramedConn, Transport}; +use smarm::cluster::Timing; use smarm::gen_server::{self, GenServerBuilder}; use smarm::pg::Incarnation; use smarm::{run, sleep, spawn}; @@ -83,7 +84,8 @@ fn heartbeats_are_sent_unprompted() { .expect("manager name is free"); let (a, b) = pair(&TcpTransport); - spawn_established(FramedConn::new(a), peer("hb-send")).expect("register"); + spawn_established(FramedConn::new(a), peer("hb-send"), Timing::default()) + .expect("register"); let mut far = FramedConn::new(b); let frame = far @@ -111,7 +113,7 @@ fn mute_peer_is_torn_down_after_liveness_timeout() { .expect("manager name is free"); let (a, b) = pair(&TcpTransport); - spawn_established(FramedConn::new(a), peer("mute")).expect("register"); + spawn_established(FramedConn::new(a), peer("mute"), Timing::default()).expect("register"); // Held open and silent: no frames, no EOF. (Unread inbound // heartbeats sit in kernel buffers; they are 5 bytes each.) let _far = FramedConn::new(b); @@ -139,7 +141,7 @@ fn heartbeats_keep_the_connection_alive() { .expect("manager name is free"); let (a, b) = pair(&TcpTransport); - spawn_established(FramedConn::new(a), peer("kept")).expect("register"); + spawn_established(FramedConn::new(a), peer("kept"), Timing::default()).expect("register"); // The far heartbeat pump: interval-paced sends until told to stop, // then holds the socket open, silent, so the eventual teardown is diff --git a/tests/cluster_connect.rs b/tests/cluster_connect.rs index 405ddc4..20286da 100644 --- a/tests/cluster_connect.rs +++ b/tests/cluster_connect.rs @@ -19,11 +19,12 @@ use smarm::cluster::connect::{ HANDSHAKE_TIMEOUT, }; use smarm::cluster::envelope::{Frame, NodeMeta, RejectReason}; -use smarm::cluster::handshake::{HelloCtx, Local}; +use smarm::cluster::handshake::{Local, PeerStanding}; use smarm::cluster::manager::{Call, Manager, Reply, MANAGER}; use smarm::cluster::transport::loopback::LoopbackTransport; use smarm::cluster::transport::tcp::TcpTransport; use smarm::cluster::transport::{FramedConn, Transport}; +use smarm::cluster::Timing; use smarm::gen_server::{self, GenServerBuilder}; use smarm::pg::Incarnation; use smarm::{run, sleep}; @@ -100,7 +101,7 @@ fn loopback_happy_path_establishes_both_ends() { local("node-b"), |name| { assert_eq!(name, "node-a"); - HelloCtx::default() + PeerStanding::Free }, no_deadline(), ) @@ -118,7 +119,7 @@ fn loopback_hash_mismatch_rejected_with_frame_then_eof() { let mut wrong = local("node-b"); wrong.build_hash ^= 1; let responder = std::thread::spawn(move || { - accept_handshake(&mut accepted, wrong, |_| HelloCtx::default(), no_deadline()) + accept_handshake(&mut accepted, wrong, |_| PeerStanding::Free, no_deadline()) }); // The dial side receives the reject frame — the compatibility anchor. match dial_handshake(&mut dialer, &local("node-a"), no_deadline()) { @@ -142,10 +143,7 @@ fn loopback_tie_break_loser_closed_silently() { accept_handshake( &mut accepted, local("node-a"), - |_| HelloCtx { - name_claimed: false, - dialing_this_peer: true, - }, + |_| PeerStanding::Dialing, no_deadline(), ) }); @@ -178,7 +176,7 @@ fn loopback_read_ahead_past_hello_survives_into_established_conn() { let peer = accept_handshake( &mut accepted, local("node-b"), - |_| HelloCtx::default(), + |_| PeerStanding::Free, no_deadline(), ) .unwrap(); @@ -216,7 +214,7 @@ fn tcp_silent_peer_times_out_on_the_accept_path() { let r = accept_handshake( &mut accepted, local("node-b"), - |_| HelloCtx::default(), + |_| PeerStanding::Free, Instant::now() + Duration::from_millis(200), ); let _ = tx.send(r); @@ -253,7 +251,7 @@ fn tcp_duplicate_name_rejected_by_acceptor() { .start() .expect("manager name is free"); let listener = TcpTransport.listen("127.0.0.1:0").unwrap(); - let acceptor = spawn_acceptor(listener, local("node-b")); + let acceptor = spawn_acceptor(listener, local("node-b"), Timing::default()); let addr = acceptor.local_addr().to_string(); // First dial offering "dup-node": establishes and registers. @@ -328,12 +326,12 @@ fn dial_intent_cleared_when_dialer_dies() { // While the dialer lives, the intent is visible. match gen_server::call( MANAGER, - Call::HelloCtx { + Call::Standing { peer_name: "ghost".into(), }, ) { - Ok(Reply::HelloCtx(ctx)) => assert!(ctx.dialing_this_peer), - other => panic!("HelloCtx failed: {other:?}"), + Ok(Reply::Standing(s)) => assert_eq!(s, PeerStanding::Dialing), + other => panic!("PeerStanding failed: {other:?}"), } // Kill it; the monitor must clear the intent without cooperation. go_tx.send(()).unwrap(); @@ -341,11 +339,11 @@ fn dial_intent_cleared_when_dialer_dies() { loop { match gen_server::call( MANAGER, - Call::HelloCtx { + Call::Standing { peer_name: "ghost".into(), }, ) { - Ok(Reply::HelloCtx(ctx)) if !ctx.dialing_this_peer => break, + Ok(Reply::Standing(s)) if s != PeerStanding::Dialing => break, _ if Instant::now() > deadline => { panic!("dial intent not cleared after dialer death") } @@ -372,7 +370,7 @@ fn role_hs_listener() { .start() .expect("manager name is free"); let listener = TcpTransport.listen("127.0.0.1:0").unwrap(); - let acceptor = spawn_acceptor(listener, local("node-b")); + let acceptor = spawn_acceptor(listener, local("node-b"), Timing::default()); println!("LISTENING {}", acceptor.local_addr()); wait_peers(&["node-a"]); println!("PEERS node-a"); @@ -389,7 +387,13 @@ fn role_hs_dialer() { .expect("manager name is free"); let (tx, rx) = mpsc::channel(); smarm::spawn(move || { - let r = dial(&TcpTransport, &addr, "node-b", &local("node-a")); + let r = dial( + &TcpTransport, + &addr, + "node-b", + &local("node-a"), + Timing::default(), + ); let _ = tx.send(r); }); if let Err(e) = poll_recv(&rx, "dial outcome") { @@ -459,7 +463,13 @@ fn concurrent_dial_to_same_name_refused() { // (the addr is unroutable on purpose — it must never be dialed). let (tx, rx) = mpsc::channel(); smarm::spawn(move || { - let r = dial(&TcpTransport, "127.0.0.1:1", "node-x", &local("node-a")); + let r = dial( + &TcpTransport, + "127.0.0.1:1", + "node-x", + &local("node-a"), + Timing::default(), + ); let _ = tx.send(r); }); match poll_recv(&rx, "second dial outcome") { diff --git a/tests/cluster_dial_mismatch.rs b/tests/cluster_dial_mismatch.rs new file mode 100644 index 0000000..03cc754 --- /dev/null +++ b/tests/cluster_dial_mismatch.rs @@ -0,0 +1,115 @@ +//! RFC 010 — a seed whose address answers as a *different* name +//! (`DialError::PeerNameMismatch`) is dialed once and then parked: the +//! connector must not redial it on backoff forever. +//! +//! Observed from the misdialed peer: each such dial establishes at the +//! responder (it registers, `node_up`), then the dialer closes on the name +//! check (`node_down`) — one membership blip per attempt. Cross-process: a +//! *server* named `server` subscribes and reports; a *client* on fast +//! timing (50–500ms backoff) seeds `("wrongname", server_addr)`. After the +//! first blip the server counts further `NodeUp`s across 2s — several +//! backoff periods. Parked ⇒ zero. Negative-control-verified: with the park +//! stubbed out the count is ≥ 1 in the same window. +#![cfg(feature = "cluster")] + +mod common; + +use common::{maybe_child, spawn_node}; +use smarm::cluster::envelope::NodeMeta; +use smarm::cluster::membership::{subscribe, NodeEvent}; +use smarm::cluster::{start, Config, StaticSeeds, Timing}; +use std::time::{Duration, Instant}; + +const ROLES: &[(&str, fn())] = &[("server", role_server), ("client", role_client)]; + +fn meta() -> NodeMeta { + NodeMeta { + role: "mismatch".into(), + region: "local".into(), + } +} + +fn timing() -> Timing { + Timing { + initial_backoff: Duration::from_millis(50), + max_backoff: Duration::from_millis(500), + ..Timing::default() + } +} + +fn role_server() { + smarm::run(|| { + let cluster = start(Config { + node_name: "server".into(), + meta: meta(), + listen_addr: std::env::var("SMARM_LISTEN_ADDR") + .unwrap_or_else(|_| "127.0.0.1:0".into()), + strategy: Box::new(StaticSeeds::new(Vec::<(String, String)>::new())), + timing: timing(), + }) + .expect("binds"); + let ev = subscribe().unwrap(); + println!("LISTENING {}", cluster.local_addr()); + // First blip: the misdialed client establishes, then closes on us. + loop { + match ev.rx.recv() { + Ok(NodeEvent::NodeDown(i)) if i.name == "client" => break, + Ok(_) => continue, + Err(_) => panic!("manager gone"), + } + } + println!("BLIP"); + // Now count further NodeUps across several backoff periods. + let mut more = 0usize; + let t0 = Instant::now(); + while t0.elapsed() < Duration::from_millis(2000) { + match ev.rx.try_recv() { + Ok(Some(NodeEvent::NodeUp(i))) if i.name == "client" => more += 1, + Ok(_) => {} + Err(_) => panic!("manager gone"), + } + smarm::sleep(Duration::from_millis(50)); + } + println!("MORE {more}"); + loop { + smarm::sleep(Duration::from_secs(3600)); + } + }); +} + +fn role_client() { + let server_addr = std::env::var("SMARM_SERVER_ADDR").expect("SMARM_SERVER_ADDR"); + smarm::run(move || { + let _cluster = start(Config { + node_name: "client".into(), + meta: meta(), + listen_addr: "127.0.0.1:0".into(), + strategy: Box::new(StaticSeeds::new(vec![( + "wrongname".to_string(), + server_addr, + )])), + timing: timing(), + }) + .expect("binds"); + println!("CLIENT UP"); + loop { + smarm::sleep(Duration::from_secs(3600)); + } + }); +} + +#[test] +fn mismatched_seed_is_dialed_once_then_parked() { + maybe_child(ROLES); + let mut server = spawn_node("server", &[]); + let saddr = server.wait_listening(); + let mut client = spawn_node("client", &[("SMARM_SERVER_ADDR", &saddr)]); + client.wait_line("CLIENT UP", |l| l == "CLIENT UP"); + server.wait_line("BLIP", |l| l == "BLIP"); + let line = server.wait_line("MORE", |l| l.starts_with("MORE ")); + let more: usize = line.split_whitespace().nth(1).unwrap().parse().unwrap(); + assert_eq!( + more, 0, + "mismatched seed was redialed {more}× after being parked" + ); +} diff --git a/tests/cluster_disconnect.rs b/tests/cluster_disconnect.rs new file mode 100644 index 0000000..a7ae186 --- /dev/null +++ b/tests/cluster_disconnect.rs @@ -0,0 +1,379 @@ +//! RFC 010 c13 — connection-loss synthesis. +//! +//! Local suite (`run()`, no network): the read-side backstop. A +//! `RemoteMonitor` whose channel closes without a notice reads as +//! `Disconnected` exactly once (a `Monitor` command that reached the conn +//! actor's inbox but was never processed — the drain gap); after +//! `demonitor_remote` a closed channel stays a plain `Err`, never a notice. +//! +//! Cross-process: the headline contrast — an actor's own death gives its +//! TRUE reason, loss of the LINK gives `Disconnected` (both a commanded +//! `Disconnect` and a SIGKILLed peer process are `Disconnected` from the +//! monitor's view: nobody is left to say otherwise). Reconnect does not +//! resurrect: the old monitor yields nothing more, proven by stream ORDER +//! (a fresh monitor over the new link delivers first). The ignored test +//! trips liveness by SIGSTOP and then drops the link too, asserting exactly +//! one notice for one monitor. +#![cfg(feature = "cluster")] + +mod common; + +use common::{maybe_child, spawn_node}; +use smarm::cluster::envelope::NodeMeta; +use smarm::cluster::expose::{expose, expose_type}; +use smarm::cluster::manager::{Call, Reply, MANAGER}; +use smarm::cluster::membership::{subscribe, NodeEvent}; +use smarm::cluster::remote::{ + self, demonitor_remote, monitor_remote, send_to_remote, RemoteName, RemotePid, +}; +use smarm::cluster::{start, Config, RemoteDownReason, StaticSeeds, Timing}; +use smarm::pg::Incarnation; +use smarm::{ + channel, gen_server, install, register, run, spawn, Addressable, DownReason, Erased, Name, Pid, +}; +use std::collections::HashMap; +use std::time::Duration; + +// ---- message types (hand-rolled serde; the crate is derive-less) --------- + +#[derive(Debug)] +struct Ctl { + cmd: String, + reply_to: RemotePid, +} +#[derive(Debug)] +struct Answer { + text: String, + pid: Option>, +} +struct Client; +impl Addressable for Client { + type Msg = Answer; +} + +impl serde::Serialize for Ctl { + fn serialize(&self, s: S) -> Result { + use serde::ser::SerializeTuple; + let mut t = s.serialize_tuple(2)?; + t.serialize_element(&self.cmd)?; + t.serialize_element(&self.reply_to)?; + t.end() + } +} +impl<'de> serde::Deserialize<'de> for Ctl { + fn deserialize>(d: D) -> Result { + let (cmd, reply_to) = <(String, RemotePid)>::deserialize(d)?; + Ok(Ctl { cmd, reply_to }) + } +} +impl serde::Serialize for Answer { + fn serialize(&self, s: S) -> Result { + use serde::ser::SerializeTuple; + let mut t = s.serialize_tuple(2)?; + t.serialize_element(&self.text)?; + t.serialize_element(&self.pid)?; + t.end() + } +} +impl<'de> serde::Deserialize<'de> for Answer { + fn deserialize>(d: D) -> Result { + let (text, pid) = <(String, Option>)>::deserialize(d)?; + Ok(Answer { text, pid }) + } +} + +// ================= local suite ========================================= + +/// A `Monitor` command handed to the connection but never processed (its +/// receiver dropped unread) reads as `Disconnected` — once. A second read +/// is the ordinary closed-channel `Err`, so "exactly one notice" holds. +#[test] +fn unread_command_reads_as_disconnected_once() { + maybe_child(ROLES); + run(|| { + remote::set_local_identity("me", Incarnation::new(7)); + let (probe_tx, _probe_rx) = channel(); + let inbox = + remote::bind_outbound_probe_with_monitors("peer", Incarnation::new(5), probe_tx); + let target = RemotePid::::from_parts("peer", Incarnation::new(5), 9, 1); + let m = monitor_remote(target.clone()); + assert!( + matches!(m.try_recv(), Ok(None)), + "command is in flight, no notice yet" + ); + drop(inbox); // the conn actor died with the command unread + let d = m.recv().unwrap(); + assert_eq!(d.pid, target); + assert_eq!(d.reason, RemoteDownReason::Disconnected); + assert!( + m.recv().is_err(), + "second read is closed, not a second notice" + ); + assert!(m.try_recv().is_err()); + }); +} + +/// After `demonitor_remote`, a closed channel is a closed channel: no +/// notice is synthesized for a monitor the caller cancelled. +#[test] +fn cancelled_monitor_never_synthesizes() { + maybe_child(ROLES); + run(|| { + remote::set_local_identity("me", Incarnation::new(7)); + let (probe_tx, _probe_rx) = channel(); + let inbox = + remote::bind_outbound_probe_with_monitors("peer", Incarnation::new(5), probe_tx); + let target = RemotePid::::from_parts("peer", Incarnation::new(5), 9, 1); + let m = monitor_remote(target); + demonitor_remote(&m); + drop(inbox); + assert!(m.recv().is_err()); + assert!(m.try_recv().is_err()); + }); +} + +// ================= cross-process ====================================== + +const ROLES: &[(&str, fn())] = &[ + ("server", role_server), + ("client", role_client), + ("client_stop", role_client_stop), +]; + +const CTL: Name = Name::new("c13.ctl"); + +fn cfg(name: &str, seeds: Vec<(String, String)>) -> Config { + Config { + node_name: name.to_string(), + meta: NodeMeta { + role: "c13".into(), + region: "local".into(), + }, + listen_addr: std::env::var("SMARM_LISTEN_ADDR").unwrap_or_else(|_| "127.0.0.1:0".into()), + strategy: Box::new(StaticSeeds::new(seeds)), + timing: timing(), + } +} + +/// The p11 knobs make the liveness test fast: both roles of that test are +/// spawned with `SMARM_FAST_TIMING=1` and agree on a 100ms heartbeat / +/// 500ms liveness window. Everything else runs the shipping defaults. +fn timing() -> Timing { + if std::env::var_os("SMARM_FAST_TIMING").is_some() { + Timing { + heartbeat_interval: Duration::from_millis(100), + liveness_timeout: Duration::from_millis(500), + initial_backoff: Duration::from_millis(50), + max_backoff: Duration::from_millis(500), + ..Timing::default() + } + } else { + Timing::default() + } +} + +fn wait_up(events: &smarm::cluster::membership::MembershipEvents, who: &str) { + loop { + match events.rx.recv() { + Ok(NodeEvent::NodeUp(i)) if i.name == who => return, + Ok(_) => continue, + Err(_) => panic!("manager gone"), + } + } +} + +fn disconnect(name: &str) { + assert!(matches!( + gen_server::call( + MANAGER, + Call::Disconnect { + name: name.to_string() + } + ), + Ok(Reply::Disconnected) + )); +} + +/// Server: `spawn` ⇒ a parked worker (answer carries its pid); +/// `kill:` releases it, whereupon it returns (Exit). +fn role_server() { + smarm::run(move || { + let cluster = start(cfg("server", vec![])).expect("binds"); + println!("LISTENING {}", cluster.local_addr()); + let (tx, rx) = channel::(); + register(CTL, tx).unwrap(); + expose(CTL); + println!("READY"); + let mut workers: HashMap> = HashMap::new(); + loop { + let ctl = rx.recv().unwrap(); + println!("CTL {}", ctl.cmd); + let (text, pid): (String, Option>) = match ctl.cmd.as_str() { + "spawn" => { + let (go_tx, go_rx) = channel::<()>(); + let p: Pid = spawn(move || { + let _ = go_rx.recv(); + }) + .pid(); + workers.insert(p.index(), go_tx); + ( + "ok".into(), + Some(RemotePid::from_local(p).expect("identity set")), + ) + } + other => { + let idx: u32 = other.strip_prefix("kill:").unwrap().parse().unwrap(); + if let Some(go) = workers.remove(&idx) { + let _ = go.send(()); + } + ("killed".into(), None) + } + }; + send_to_remote(ctl.reply_to, Answer { text, pid }).unwrap(); + } + }); +} + +/// Client-side setup shared by both client roles: join, expose the reply +/// path, hand back an `ask` closure and the membership stream. +fn client_setup() -> ( + smarm::cluster::Cluster, + smarm::cluster::membership::MembershipEvents, + impl Fn(&str) -> Answer, +) { + let server_addr = std::env::var("SMARM_SERVER_ADDR").expect("SMARM_SERVER_ADDR"); + let cluster = start(cfg("client", vec![("server".into(), server_addr)])).expect("binds"); + let ev = subscribe().unwrap(); + wait_up(&ev, "server"); + let (tx, rx) = channel::(); + let me: Pid = install::(tx); + expose_type::(); + let ask = move |cmd: &str| -> Answer { + remote::send( + RemoteName::new("server", CTL), + Ctl { + cmd: cmd.into(), + reply_to: RemotePid::from_local(me).expect("identity set"), + }, + ) + .unwrap(); + rx.recv().unwrap() + }; + (cluster, ev, ask) +} + +fn role_client() { + smarm::run(move || { + let (_cluster, ev, ask) = client_setup(); + + // 1. Headline: actor death ⇒ TRUE reason; link cut ⇒ Disconnected. + let a = ask("spawn").pid.unwrap(); + let b = ask("spawn").pid.unwrap(); + let ma = monitor_remote(a.clone()); + let mb = monitor_remote(b.clone()); + ask(&format!("kill:{}", a.index())); + let d = ma.recv().unwrap(); + assert_eq!(d.pid, a); + println!("DOWN actor {:?}", d.reason); + disconnect("server"); + let d = mb.recv().unwrap(); + assert_eq!(d.pid, b); + println!("DOWN link {:?}", d.reason); + + // 2. Reconnect does not resurrect. The connector redials on + // node_down; over the NEW link a fresh monitor delivers, while + // the old one (already answered) yields nothing further — order + // proves it, and `b` is even still alive on the server. + wait_up(&ev, "server"); + println!("RECONNECTED"); + let c = ask("spawn").pid.unwrap(); + let mc = monitor_remote(c.clone()); + ask(&format!("kill:{}", b.index())); + ask(&format!("kill:{}", c.index())); + assert_eq!(mc.recv().unwrap().reason, DownReason::Exit.into()); + let stray = matches!(mb.try_recv(), Ok(Some(_))); + println!("RESURRECT stray={stray}"); + + // 3. Peer PROCESS killed ⇒ Disconnected too (nobody is left to send + // Down): the parent SIGKILLs the server once it sees the marker. + let e = ask("spawn").pid.unwrap(); + let me_ = monitor_remote(e.clone()); + println!("KILL SERVER NOW"); + let d = me_.recv().unwrap(); + assert_eq!(d.pid, e); + println!("DOWN procdeath {:?}", d.reason); + + println!("CLIENT DONE"); + loop { + smarm::sleep(Duration::from_secs(3600)); + } + }); +} + +/// The slow role: liveness expiry (peer SIGSTOPped) followed by the link +/// dropping for real (peer SIGKILLed) — one monitor, exactly one notice. +fn role_client_stop() { + smarm::run(move || { + let (_cluster, _ev, ask) = client_setup(); + let a = ask("spawn").pid.unwrap(); + let ma = monitor_remote(a.clone()); + println!("STOP SERVER NOW"); + let d = ma.recv().unwrap(); // liveness expiry, ~liveness_timeout + assert_eq!(d.pid, a); + println!("DOWN stopped {:?}", d.reason); + println!("KILL SERVER NOW"); + // Give the drop every chance to produce a second notice, then look. + smarm::sleep(Duration::from_secs(1)); + let dup = matches!(ma.try_recv(), Ok(Some(_))); + println!("DUPLICATE dup={dup}"); + println!("CLIENT DONE"); + loop { + smarm::sleep(Duration::from_secs(3600)); + } + }); +} + +/// The Phase 4 c13 gate: partition vs. death distinguishable; nothing +/// survives reconnect; a dead peer process is a Disconnected too. +#[test] +fn link_loss_is_disconnected_and_does_not_survive_reconnect() { + maybe_child(ROLES); + let mut server = spawn_node("server", &[]); + let saddr = server.wait_listening(); + server.wait_line("READY", |l| l == "READY"); + let mut client = spawn_node("client", &[("SMARM_SERVER_ADDR", &saddr)]); + client.wait_line("DOWN actor Local(Exit)", |l| l == "DOWN actor Local(Exit)"); + client.wait_line("DOWN link Disconnected", |l| l == "DOWN link Disconnected"); + client.wait_line("RECONNECTED", |l| l == "RECONNECTED"); + client.wait_line("RESURRECT stray=false", |l| l == "RESURRECT stray=false"); + client.wait_line("KILL SERVER NOW", |l| l == "KILL SERVER NOW"); + server.kill(); + client.wait_line("DOWN procdeath Disconnected", |l| { + l == "DOWN procdeath Disconnected" + }); + client.wait_line("CLIENT DONE", |l| l == "CLIENT DONE"); +} + +/// Covers the invariant the headline test cannot: liveness expiry and the +/// transport drop both firing for the same connection yield ONE notice. +/// Runs on the fast [`timing`] (both roles) — was `#[ignore]`d at the 4s +/// default until the p11 knobs landed. +#[test] +fn timeout_then_drop_yields_one_notice() { + maybe_child(ROLES); + let fast = ("SMARM_FAST_TIMING", "1"); + let mut server = spawn_node("server", &[fast]); + let saddr = server.wait_listening(); + server.wait_line("READY", |l| l == "READY"); + let mut client = spawn_node("client_stop", &[("SMARM_SERVER_ADDR", &saddr), fast]); + client.wait_line("STOP SERVER NOW", |l| l == "STOP SERVER NOW"); + let spid = server.pid().expect("server alive") as libc::pid_t; + assert_eq!(unsafe { libc::kill(spid, libc::SIGSTOP) }, 0); + client.wait_line("DOWN stopped Disconnected", |l| { + l == "DOWN stopped Disconnected" + }); + client.wait_line("KILL SERVER NOW", |l| l == "KILL SERVER NOW"); + server.kill(); // SIGKILL works on a stopped process; Drop would too + client.wait_line("DUPLICATE dup=false", |l| l == "DUPLICATE dup=false"); + client.wait_line("CLIENT DONE", |l| l == "CLIENT DONE"); +} diff --git a/tests/cluster_discovery_withdraw.rs b/tests/cluster_discovery_withdraw.rs new file mode 100644 index 0000000..f261451 --- /dev/null +++ b/tests/cluster_discovery_withdraw.rs @@ -0,0 +1,161 @@ +//! RFC 010 — `Discovery::Withdrawn`: a strategy retracts a candidate and the +//! connector stops dialing it. +//! +//! Cross-process: a plain *server* node, and a *client* whose strategy is a +//! script: announce a decoy `(ghost, addr)` where `addr` is a raw +//! `TcpListener` the client itself holds (an OS thread accepts and +//! immediately closes, so every dial fails at handshake and the connector +//! keeps retrying on backoff — the accept count is the dial count); after a +//! beat, withdraw the decoy and announce the real server. The client waits +//! for the server's `node_up` — which is *after* the withdrawal in the +//! strategy's own stream — then watches the decoy's accept count stay flat +//! across a window longer than the pending backoff. Before withdrawal it +//! must have been climbing (≥ 1), or the negative proves nothing. +#![cfg(feature = "cluster")] + +mod common; + +use common::{maybe_child, spawn_node}; +use smarm::channel::Sender; +use smarm::cluster::discovery::{Discovery, Strategy}; +use smarm::cluster::envelope::NodeMeta; +use smarm::cluster::membership::{subscribe, NodeEvent}; +use smarm::cluster::{start, Config, StaticSeeds, Timing}; +use std::net::TcpListener; +use std::sync::atomic::{AtomicUsize, Ordering}; +use std::sync::Arc; +use std::time::{Duration, Instant}; + +const ROLES: &[(&str, fn())] = &[("server", role_server), ("client", role_client)]; + +fn meta() -> NodeMeta { + NodeMeta { + role: "withdraw".into(), + region: "local".into(), + } +} + +fn role_server() { + smarm::run(|| { + let cluster = start(Config { + node_name: "server".into(), + meta: meta(), + listen_addr: std::env::var("SMARM_LISTEN_ADDR") + .unwrap_or_else(|_| "127.0.0.1:0".into()), + strategy: Box::new(StaticSeeds::new(Vec::<(String, String)>::new())), + timing: Timing::default(), + }) + .expect("binds"); + println!("LISTENING {}", cluster.local_addr()); + loop { + smarm::sleep(Duration::from_secs(3600)); + } + }); +} + +/// Scripted strategy: decoy, pause, withdraw decoy, real server, done. +struct Script { + decoy: String, + server: String, +} + +impl Strategy for Script { + fn run(self: Box, out: Sender) { + let _ = out.send(Discovery::Candidate { + name: "ghost".into(), + addr: self.decoy.clone(), + }); + // Long enough for the 250ms/500ms retries to land: ≥ 3 dials. + smarm::sleep(Duration::from_millis(1100)); + let _ = out.send(Discovery::Withdrawn { + name: "ghost".into(), + addr: self.decoy, + }); + let _ = out.send(Discovery::Candidate { + name: "server".into(), + addr: self.server, + }); + } +} + +fn role_client() { + let server_addr = std::env::var("SMARM_SERVER_ADDR").expect("SMARM_SERVER_ADDR"); + // The decoy: accept-and-close on an OS thread; count every accept. + let decoy = TcpListener::bind("127.0.0.1:0").unwrap(); + let decoy_addr = decoy.local_addr().unwrap().to_string(); + let dials = Arc::new(AtomicUsize::new(0)); + let counter = dials.clone(); + std::thread::spawn(move || { + for conn in decoy.incoming() { + counter.fetch_add(1, Ordering::SeqCst); + drop(conn); + } + }); + + smarm::run(move || { + let _cluster = start(Config { + node_name: "client".into(), + meta: meta(), + listen_addr: "127.0.0.1:0".into(), + strategy: Box::new(Script { + decoy: decoy_addr, + server: server_addr, + }), + timing: Timing::default(), + }) + .expect("binds"); + let ev = subscribe().unwrap(); + loop { + match ev.rx.recv() { + Ok(NodeEvent::NodeUp(i)) if i.name == "server" => break, + Ok(_) => continue, + Err(_) => panic!("manager gone"), + } + } + // The withdrawal preceded the server candidate in the strategy's + // stream, so it has been applied. Any dial that started before it + // is bounded by the connect+handshake deadlines; let it drain, then + // hold the count flat across a window longer than the pending + // backoff would be (1s at this point, 2s next). + let before = dials.load(Ordering::SeqCst); + smarm::sleep(Duration::from_millis(500)); + let settled = dials.load(Ordering::SeqCst); + let t0 = Instant::now(); + while t0.elapsed() < Duration::from_millis(3000) { + smarm::sleep(Duration::from_millis(100)); + } + let after = dials.load(Ordering::SeqCst); + println!("WITHDRAWN before={before} settled={settled} after={after}"); + println!("CLIENT DONE"); + loop { + smarm::sleep(Duration::from_secs(3600)); + } + }); +} + +#[test] +fn withdrawn_candidate_is_no_longer_dialed() { + maybe_child(ROLES); + let mut server = spawn_node("server", &[]); + let saddr = server.wait_listening(); + let mut client = spawn_node("client", &[("SMARM_SERVER_ADDR", &saddr)]); + let line = client.wait_line("WITHDRAWN", |l| l.starts_with("WITHDRAWN ")); + let mut nums = line + .split_whitespace() + .skip(1) + .map(|kv| kv.split_once('=').unwrap().1.parse::().unwrap()); + let (before, settled, after) = ( + nums.next().unwrap(), + nums.next().unwrap(), + nums.next().unwrap(), + ); + assert!( + before >= 1, + "decoy was never dialed; the negative proves nothing: {line}" + ); + assert_eq!( + settled, after, + "connector kept dialing a withdrawn candidate: {line}" + ); + client.wait_line("CLIENT DONE", |l| l == "CLIENT DONE"); +} diff --git a/tests/cluster_envelope.rs b/tests/cluster_envelope.rs index aba1773..f81dad3 100644 --- a/tests/cluster_envelope.rs +++ b/tests/cluster_envelope.rs @@ -8,6 +8,7 @@ use smarm::cluster::envelope::{ decode_payload, encode_payload, DecodeError, Frame, NodeMeta, RejectReason, MAX_FRAME_LEN, PROTO_VERSION, }; +use smarm::cluster::RemoteDownReason; use smarm::monitor::DownReason; use smarm::pg::Incarnation; @@ -55,7 +56,11 @@ fn all_frames() -> Vec { Frame::Demonitor { monitor_id: 77 }, Frame::Down { monitor_id: 77, - reason: DownReason::Panic, + reason: RemoteDownReason::Local(DownReason::Panic), + }, + Frame::Down { + monitor_id: 78, + reason: RemoteDownReason::Disconnected, }, ] } @@ -151,7 +156,7 @@ fn unknown_enum_tags() { assert_eq!( Frame::decode(&buf), Err(DecodeError::UnknownEnumTag { - what: "DownReason", + what: "RemoteDownReason", tag: 200 }) ); diff --git a/tests/cluster_handshake.rs b/tests/cluster_handshake.rs index 0eede87..fa4a755 100644 --- a/tests/cluster_handshake.rs +++ b/tests/cluster_handshake.rs @@ -5,7 +5,7 @@ use smarm::cluster::envelope::{Frame, NodeMeta, RejectReason, PROTO_VERSION}; use smarm::cluster::handshake::{ - dial_wins, HelloCtx, Initiator, InitiatorOutcome, Local, Responder, ResponderOutcome, + dial_wins, Initiator, InitiatorOutcome, Local, PeerStanding, Responder, ResponderOutcome, }; use smarm::pg::Incarnation; @@ -42,7 +42,7 @@ fn happy_path_establishes_both_ends() { assert_eq!(hello, hello_from("alpha"), "initiator emits its identity"); let responder = Responder::new(local("beta")); - let (reply, peer) = match responder.on_frame(hello, HelloCtx::default()) { + let (reply, peer) = match responder.on_frame(hello, PeerStanding::Free) { ResponderOutcome::Accepted { reply, peer } => (reply, peer), other => panic!("expected Accepted, got {other:?}"), }; @@ -82,7 +82,7 @@ fn hash_mismatch_rejected() { incarnation: Incarnation::new(7), meta: local("alpha").meta, }; - match responder.on_frame(hello, HelloCtx::default()) { + match responder.on_frame(hello, PeerStanding::Free) { ResponderOutcome::Rejected { reply, reason } => { assert_eq!(reason, RejectReason::HashMismatch); assert_eq!(reply, Frame::HelloReject { reason }); @@ -112,7 +112,7 @@ fn proto_version_mismatch_rejected_and_checked_first() { incarnation: Incarnation::new(7), meta: local("alpha").meta, }; - match responder.on_frame(hello, HelloCtx::default()) { + match responder.on_frame(hello, PeerStanding::Free) { ResponderOutcome::Rejected { reason, .. } => { assert_eq!(reason, RejectReason::ProtoVersion); } @@ -123,10 +123,7 @@ fn proto_version_mismatch_rejected_and_checked_first() { #[test] fn claimed_name_rejected() { let responder = Responder::new(local("beta")); - let ctx = HelloCtx { - name_claimed: true, - dialing_this_peer: false, - }; + let ctx = PeerStanding::Claimed; match responder.on_frame(hello_from("alpha"), ctx) { ResponderOutcome::Rejected { reason, .. } => { assert_eq!(reason, RejectReason::NameTaken); @@ -139,7 +136,7 @@ fn claimed_name_rejected() { fn own_name_offered_rejected_as_name_taken() { // Self-connect or genuine collision: the responder's own name arrives. let responder = Responder::new(local("beta")); - match responder.on_frame(hello_from("beta"), HelloCtx::default()) { + match responder.on_frame(hello_from("beta"), PeerStanding::Free) { ResponderOutcome::Rejected { reason, .. } => { assert_eq!(reason, RejectReason::NameTaken); } @@ -158,10 +155,7 @@ fn hash_checked_before_name() { incarnation: Incarnation::new(7), meta: local("alpha").meta, }; - let ctx = HelloCtx { - name_claimed: true, - dialing_this_peer: false, - }; + let ctx = PeerStanding::Claimed; match responder.on_frame(hello, ctx) { ResponderOutcome::Rejected { reason, .. } => { assert_eq!(reason, RejectReason::HashMismatch); @@ -184,10 +178,7 @@ fn dial_wins_is_deterministic_and_antisymmetric() { fn simultaneous_connect_exactly_one_side_accepts() { // alpha and beta dial each other at once. Each responder sees the peer's // Hello while its own dial is in flight. - let ctx = HelloCtx { - name_claimed: false, - dialing_this_peer: true, - }; + let ctx = PeerStanding::Dialing; // On beta: inbound is alpha's dial; alpha < beta, so the inbound wins. let on_beta = Responder::new(local("beta")).on_frame(hello_from("alpha"), ctx); @@ -208,7 +199,7 @@ fn simultaneous_connect_exactly_one_side_accepts() { #[test] fn tiebreak_loss_only_applies_when_dialing() { // Same inbound Hello, no dial in flight: plain accept. - let on_alpha = Responder::new(local("alpha")).on_frame(hello_from("beta"), HelloCtx::default()); + let on_alpha = Responder::new(local("alpha")).on_frame(hello_from("beta"), PeerStanding::Free); assert!(matches!(on_alpha, ResponderOutcome::Accepted { .. })); } @@ -225,7 +216,7 @@ fn garbage_before_hello_fails_without_reply() { }, Frame::Demonitor { monitor_id: 3 }, ] { - let out = Responder::new(local("beta")).on_frame(frame.clone(), HelloCtx::default()); + let out = Responder::new(local("beta")).on_frame(frame.clone(), PeerStanding::Free); match out { ResponderOutcome::Failed(f) => assert_eq!(f, frame), other => panic!("expected Failed({frame:?}), got {other:?}"), diff --git a/tests/cluster_membership.rs b/tests/cluster_membership.rs index cf2eb67..6a597be 100644 --- a/tests/cluster_membership.rs +++ b/tests/cluster_membership.rs @@ -18,6 +18,7 @@ use smarm::cluster::membership::{subscribe, view, MembershipEvents, NodeEvent}; use smarm::cluster::spawn_established; use smarm::cluster::transport::tcp::TcpTransport; use smarm::cluster::transport::{Conn, FramedConn, Transport}; +use smarm::cluster::Timing; use smarm::gen_server::{self, GenServerBuilder}; use smarm::pg::{Incarnation, NodeId}; use smarm::run; @@ -85,8 +86,10 @@ fn subscriber_sees_up_and_down() { let t = TcpTransport; let (a1, b1) = pair(&t); let (a2, b2) = pair(&t); - spawn_established(FramedConn::new(a1), peer("node-b", 1)).expect("node-b registers"); - spawn_established(FramedConn::new(a2), peer("node-c", 1)).expect("node-c registers"); + spawn_established(FramedConn::new(a1), peer("node-b", 1), Timing::default()) + .expect("node-b registers"); + spawn_established(FramedConn::new(a2), peer("node-c", 1), Timing::default()) + .expect("node-c registers"); let up_b = match next_event(&ev, "node_up(node-b)") { NodeEvent::NodeUp(info) => { @@ -110,14 +113,14 @@ fn subscriber_sees_up_and_down() { disconnect("node-b"); assert_eq!( next_event(&ev, "node_down(node-b)"), - NodeEvent::NodeDown { node: up_b.node } + NodeEvent::NodeDown(up_b.clone()) ); // Peer EOF, no command: down with node-c's id. drop(b2); assert_eq!( next_event(&ev, "node_down(node-c)"), - NodeEvent::NodeDown { node: up_c.node } + NodeEvent::NodeDown(up_c.clone()) ); assert_quiet(&ev); @@ -140,8 +143,10 @@ fn late_subscriber_gets_snapshot_and_view_agrees() { let t = TcpTransport; let (a1, b1) = pair(&t); let (a2, b2) = pair(&t); - spawn_established(FramedConn::new(a1), peer("node-b", 1)).expect("node-b registers"); - spawn_established(FramedConn::new(a2), peer("node-c", 1)).expect("node-c registers"); + spawn_established(FramedConn::new(a1), peer("node-b", 1), Timing::default()) + .expect("node-b registers"); + spawn_established(FramedConn::new(a2), peer("node-c", 1), Timing::default()) + .expect("node-c registers"); let ev = subscribe().expect("manager is up"); let mut names = Vec::new(); @@ -190,20 +195,25 @@ fn restart_gets_new_id_blip_keeps_id() { other => panic!("expected node_up ({what}), got {other:?}"), } }; + let down_id = |e: NodeEvent, what: &str| -> NodeId { + match e { + NodeEvent::NodeDown(info) => info.node, + other => panic!("expected node_down ({what}), got {other:?}"), + } + }; // Up at incarnation 1, then the peer dies (EOF). let (a1, b1) = pair(&t); - spawn_established(FramedConn::new(a1), peer("node-b", 1)).expect("registers"); + spawn_established(FramedConn::new(a1), peer("node-b", 1), Timing::default()) + .expect("registers"); let id1 = id(next_event(&ev, "node_up inc 1"), "inc 1"); drop(b1); - assert_eq!( - next_event(&ev, "node_down inc 1"), - NodeEvent::NodeDown { node: id1 } - ); + assert_eq!(down_id(next_event(&ev, "node_down inc 1"), "inc 1"), id1); // Restart: new incarnation, new id — the ghost's id is not reused. let (a2, b2) = pair(&t); - spawn_established(FramedConn::new(a2), peer("node-b", 2)).expect("registers"); + spawn_established(FramedConn::new(a2), peer("node-b", 2), Timing::default()) + .expect("registers"); let id2 = id(next_event(&ev, "node_up inc 2"), "inc 2"); assert_ne!( id1, id2, @@ -212,12 +222,10 @@ fn restart_gets_new_id_blip_keeps_id() { // Blip: the same incarnation reconnects and keeps its id. disconnect("node-b"); - assert_eq!( - next_event(&ev, "node_down inc 2"), - NodeEvent::NodeDown { node: id2 } - ); + assert_eq!(down_id(next_event(&ev, "node_down inc 2"), "inc 2"), id2); let (a3, b3) = pair(&t); - spawn_established(FramedConn::new(a3), peer("node-b", 2)).expect("registers"); + spawn_established(FramedConn::new(a3), peer("node-b", 2), Timing::default()) + .expect("registers"); let id3 = id(next_event(&ev, "node_up after blip"), "blip"); assert_eq!( id2, id3, @@ -247,7 +255,8 @@ fn dead_subscriber_is_pruned() { let t = TcpTransport; let (a1, b1) = pair(&t); - spawn_established(FramedConn::new(a1), peer("node-b", 1)).expect("registers"); + spawn_established(FramedConn::new(a1), peer("node-b", 1), Timing::default()) + .expect("registers"); match next_event(&live, "node_up despite a dead co-subscriber") { NodeEvent::NodeUp(info) => assert_eq!(info.name, "node-b"), other => panic!("expected node_up, got {other:?}"), diff --git a/tests/cluster_mesh.rs b/tests/cluster_mesh.rs index 6b6fb18..08451d4 100644 --- a/tests/cluster_mesh.rs +++ b/tests/cluster_mesh.rs @@ -24,9 +24,7 @@ mod common; use common::{maybe_child, spawn_node, Node}; use smarm::cluster::envelope::NodeMeta; use smarm::cluster::membership::{subscribe, NodeEvent}; -use smarm::cluster::{start, Config, StaticSeeds}; -use smarm::pg::NodeId; -use std::collections::HashMap; +use smarm::cluster::{start, Config, StaticSeeds, Timing}; use std::time::Duration; const ROLES: &[(&str, fn())] = &[("node", role_node)]; @@ -56,21 +54,19 @@ fn role_node() { }, listen_addr: listen, strategy: Box::new(StaticSeeds::new(seeds)), + timing: Timing::default(), }) .expect("listener binds"); println!("LISTENING {}", cluster.local_addr()); let events = subscribe().expect("manager is up"); - let mut names: HashMap = HashMap::new(); loop { match events.rx.recv() { Ok(NodeEvent::NodeUp(info)) => { - names.insert(info.node, info.name.clone()); println!("MEMBER-UP {} inc={}", info.name, info.incarnation.get()); } - Ok(NodeEvent::NodeDown { node }) => { - let name = names.remove(&node).unwrap_or_else(|| "?".to_string()); - println!("MEMBER-DOWN {name}"); + Ok(NodeEvent::NodeDown(info)) => { + println!("MEMBER-DOWN {}", info.name); } Err(_) => break, // manager gone; park below regardless } diff --git a/tests/cluster_monitor.rs b/tests/cluster_monitor.rs new file mode 100644 index 0000000..8b141ee --- /dev/null +++ b/tests/cluster_monitor.rs @@ -0,0 +1,359 @@ +//! RFC 010 c12 — remote monitors. +//! +//! Local suite (`run()`, no network): the immediate answers — no connection +//! ⇒ `Disconnected`, dead incarnation ⇒ `NoProc` — and the self-node +//! collapse (a plain local monitor underneath, incl. `demonitor_remote`). +//! +//! Cross-process: a *server* exposes a control name and spawns workers on +//! request, replying with each worker's pid (via `RemotePid::from_local`, +//! the D12 set-site) or, for the deliberately unshipped one, only its raw +//! slot numbers. The *client* monitors them and asserts: kill ⇒ the true +//! reason (Exit / Panic); a corpse ⇒ its recorded terminal reason, not +//! NoProc; a live pid that never crossed the wire ⇒ NoProc (no liveness +//! leak); a demonitor racing the kill ⇒ no notice, proven by stream ORDER +//! (a later notice on the same connection arrives while the earlier slot +//! is still empty), not by sleeping. +#![cfg(feature = "cluster")] + +mod common; + +use common::{maybe_child, spawn_node}; +use smarm::cluster::envelope::NodeMeta; +use smarm::cluster::expose::{expose, expose_type}; +use smarm::cluster::membership::{subscribe, NodeEvent}; +use smarm::cluster::remote::{ + self, demonitor_remote, monitor_remote, send_to_remote, RemoteName, RemotePid, +}; +use smarm::cluster::{start, Config, RemoteDownReason, StaticSeeds, Timing}; +use smarm::pg::Incarnation; +use smarm::{channel, install, register, run, spawn, Addressable, DownReason, Erased, Name, Pid}; +use std::collections::HashMap; +use std::time::Duration; + +// ---- message types (hand-rolled serde; the crate is derive-less) --------- + +#[derive(Debug)] +struct Ctl { + cmd: String, + reply_to: RemotePid, +} +#[derive(Debug)] +struct Answer { + text: String, + pid: Option>, +} +struct Client; +impl Addressable for Client { + type Msg = Answer; +} + +impl serde::Serialize for Ctl { + fn serialize(&self, s: S) -> Result { + use serde::ser::SerializeTuple; + let mut t = s.serialize_tuple(2)?; + t.serialize_element(&self.cmd)?; + t.serialize_element(&self.reply_to)?; + t.end() + } +} +impl<'de> serde::Deserialize<'de> for Ctl { + fn deserialize>(d: D) -> Result { + let (cmd, reply_to) = <(String, RemotePid)>::deserialize(d)?; + Ok(Ctl { cmd, reply_to }) + } +} +impl serde::Serialize for Answer { + fn serialize(&self, s: S) -> Result { + use serde::ser::SerializeTuple; + let mut t = s.serialize_tuple(2)?; + t.serialize_element(&self.text)?; + t.serialize_element(&self.pid)?; + t.end() + } +} +impl<'de> serde::Deserialize<'de> for Answer { + fn deserialize>(d: D) -> Result { + let (text, pid) = <(String, Option>)>::deserialize(d)?; + Ok(Answer { text, pid }) + } +} + +// ================= local suite ========================================= + +/// No connection to the pid's node: `Disconnected` at once — the remote +/// analog of NoProc, and the first thing c11's variant is for. +#[test] +fn unconnected_node_is_disconnected_immediately() { + maybe_child(ROLES); + run(|| { + remote::set_local_identity("me", Incarnation::new(7)); + let ghost = RemotePid::::from_parts("nowhere", Incarnation::new(1), 3, 1); + let m = monitor_remote(ghost.clone()); + let d = m.recv().unwrap(); + assert_eq!(d.pid, ghost); + assert_eq!(d.reason, RemoteDownReason::Disconnected); + }); +} + +/// The node is connected but the pid names an earlier incarnation: the +/// actor is a known corpse (RFC v2 §3), so `NoProc` at once — never +/// `Disconnected`, nothing on the wire. +#[test] +fn dead_incarnation_is_noproc_immediately() { + maybe_child(ROLES); + run(|| { + remote::set_local_identity("me", Incarnation::new(7)); + let (probe_tx, probe_rx) = channel(); + remote::bind_outbound_probe("peer", Incarnation::new(5), probe_tx); + let stale = RemotePid::::from_parts("peer", Incarnation::new(4), 9, 1); + let m = monitor_remote(stale); + assert_eq!(m.recv().unwrap().reason, DownReason::NoProc.into()); + assert!(probe_rx.try_recv().unwrap().is_none(), "no frame emitted"); + }); +} + +/// A self-node pid collapses to an ordinary local monitor: the true reason +/// on exit, and `demonitor_remote` cancels it. +#[test] +fn self_node_pid_collapses_to_local_monitor() { + maybe_child(ROLES); + run(|| { + remote::set_local_identity("me", Incarnation::new(7)); + let (go_tx, go_rx) = channel::<()>(); + let (go2_tx, go2_rx) = channel::<()>(); + let a = spawn(move || { + let _ = go_rx.recv(); + }) + .pid(); + let b = spawn(move || { + let _ = go2_rx.recv(); + }) + .pid(); + let ma = monitor_remote(RemotePid::from_local(a).expect("identity set")); + let mb = monitor_remote(RemotePid::from_local(b).expect("identity set")); + assert_ne!(ma.id, mb.id); + assert!(ma.target.local() == Some(a)); + + demonitor_remote(&mb); + go2_tx.send(()).unwrap(); + go_tx.send(()).unwrap(); + let d = ma.recv().unwrap(); + assert_eq!(d.reason, DownReason::Exit.into()); + assert_eq!(d.pid.local(), Some(a)); + // `a` is down (its notice arrived), and `b` was killed first on the + // same scheduler — a notice for `b` would be here by now. After a + // demonitor the channel is closed-empty (`Err`), like the local one. + assert!(matches!(mb.try_recv(), Ok(None) | Err(_))); + }); +} + +// ================= cross-process ====================================== + +const ROLES: &[(&str, fn())] = &[("server", role_server), ("client", role_client)]; + +const CTL: Name = Name::new("c12.ctl"); + +fn cfg(name: &str, seeds: Vec<(String, String)>) -> Config { + Config { + node_name: name.to_string(), + meta: NodeMeta { + role: "c12".into(), + region: "local".into(), + }, + listen_addr: std::env::var("SMARM_LISTEN_ADDR").unwrap_or_else(|_| "127.0.0.1:0".into()), + strategy: Box::new(StaticSeeds::new(seeds)), + timing: Timing::default(), + } +} + +fn wait_up(events: &smarm::cluster::membership::MembershipEvents, who: &str) { + loop { + match events.rx.recv() { + Ok(NodeEvent::NodeUp(i)) if i.name == who => return, + Ok(_) => continue, + Err(_) => panic!("manager gone"), + } + } +} + +/// Server commands (all answered to `reply_to`): +/// - `spawn:exit` / `spawn:panic` — a parked worker; `kill:` releases +/// it, whereupon it returns / panics. Answer carries its pid. +/// - `spawn:corpse` — a worker that has already exited when the answer is +/// sent; the pid was shipped (watchable) before it died. +/// - `spawn:unwatched` — a parked worker whose pid is NEVER shipped; the +/// answer carries only `text = "slot::"`. +fn role_server() { + smarm::run(move || { + let cluster = start(cfg("server", vec![])).expect("binds"); + println!("LISTENING {}", cluster.local_addr()); + let (tx, rx) = channel::(); + register(CTL, tx).unwrap(); + expose(CTL); + println!("READY"); + let mut workers: HashMap> = HashMap::new(); + loop { + let ctl = rx.recv().unwrap(); + println!("CTL {}", ctl.cmd); + let (text, pid): (String, Option>) = match ctl.cmd.as_str() { + "spawn:exit" | "spawn:panic" => { + let panic = ctl.cmd == "spawn:panic"; + let (go_tx, go_rx) = channel::<()>(); + let p: Pid = spawn(move || { + let _ = go_rx.recv(); + if panic { + panic!("worker asked to panic"); + } + }) + .pid(); + workers.insert(p.index(), go_tx); + ( + "ok".into(), + Some(RemotePid::from_local(p).expect("identity set")), + ) + } + "spawn:corpse" => { + let p: Pid = spawn(|| {}).pid(); + let rp = RemotePid::from_local(p).expect("identity set"); // shipped ⇒ watchable + let m = smarm::monitor(p); + let _ = m.rx.recv(); // dead before the answer goes out + ("ok".into(), Some(rp)) + } + "spawn:unwatched" => { + let (go_tx, go_rx) = channel::<()>(); + let p: Pid = spawn(move || { + let _ = go_rx.recv(); + }) + .pid(); + workers.insert(p.index(), go_tx); + (format!("slot:{}:{}", p.index(), p.generation()), None) + } + other => { + let idx: u32 = other.strip_prefix("kill:").unwrap().parse().unwrap(); + if let Some(go) = workers.remove(&idx) { + let _ = go.send(()); + } + ("killed".into(), None) + } + }; + send_to_remote(ctl.reply_to, Answer { text, pid }).unwrap(); + } + }); +} + +fn role_client() { + let server_addr = std::env::var("SMARM_SERVER_ADDR").expect("SMARM_SERVER_ADDR"); + smarm::run(move || { + let _cluster = start(cfg("client", vec![("server".into(), server_addr)])).expect("binds"); + let ev = subscribe().unwrap(); + wait_up(&ev, "server"); + let (tx, rx) = channel::(); + let me: Pid = install::(tx); + expose_type::(); + let ask = |cmd: &str| -> Answer { + remote::send( + RemoteName::new("server", CTL), + Ctl { + cmd: cmd.into(), + reply_to: RemotePid::from_local(me).expect("identity set"), + }, + ) + .unwrap(); + rx.recv().unwrap() + }; + let server_inc = ev_incarnation(); + + // 1. kill ⇒ true reason (Exit). + let a = ask("spawn:exit").pid.unwrap(); + let ma = monitor_remote(a.clone()); + ask(&format!("kill:{}", a.index())); + let d = ma.recv().unwrap(); + assert_eq!(d.pid, a); + println!("DOWN exit {:?}", d.reason); + + // 2. kill ⇒ true reason (Panic). + let b = ask("spawn:panic").pid.unwrap(); + let mb = monitor_remote(b.clone()); + ask(&format!("kill:{}", b.index())); + println!("DOWN panic {:?}", mb.recv().unwrap().reason); + + // 3. corpse ⇒ recorded terminal reason, not NoProc. + let c = ask("spawn:corpse").pid.unwrap(); + println!("DOWN corpse {:?}", monitor_remote(c).recv().unwrap().reason); + + // 4. live but never shipped/exposed ⇒ NoProc (no leak); a made-up + // slot on the same node ⇒ NoProc too, indistinguishably. + let ans = ask("spawn:unwatched"); + let mut it = ans.text.strip_prefix("slot:").unwrap().split(':'); + let (idx, gen): (u32, u32) = ( + it.next().unwrap().parse().unwrap(), + it.next().unwrap().parse().unwrap(), + ); + let hidden = RemotePid::::from_parts("server", server_inc, idx, gen); + println!( + "DOWN hidden {:?}", + monitor_remote(hidden).recv().unwrap().reason + ); + let bogus = RemotePid::::from_parts("server", server_inc, 100_000, 1); + println!( + "DOWN bogus {:?}", + monitor_remote(bogus).recv().unwrap().reason + ); + + // 5. demonitor races the kill: no notice for `d1`, proven by order — + // `d2`'s notice (same connection, later) arrives while `d1`'s + // slot is still empty. + let d1 = ask("spawn:exit").pid.unwrap(); + let m1 = monitor_remote(d1.clone()); + demonitor_remote(&m1); + ask(&format!("kill:{}", d1.index())); + let d2 = ask("spawn:exit").pid.unwrap(); + let m2 = monitor_remote(d2.clone()); + ask(&format!("kill:{}", d2.index())); + assert_eq!(m2.recv().unwrap().reason, DownReason::Exit.into()); + // Closed-empty (`Err`) or open-empty (`Ok(None)`) both mean no notice. + let stray = matches!(m1.try_recv(), Ok(Some(_))); + println!("DEMONITOR stray={stray}"); + + println!("CLIENT DONE"); + loop { + smarm::sleep(Duration::from_secs(3600)); + } + }); +} + +/// The server's incarnation as this node sees it — for building pids by hand. +fn ev_incarnation() -> Incarnation { + smarm::cluster::membership::view() + .expect("manager up") + .into_iter() + .find(|i| i.name == "server") + .map(|i| i.incarnation) + .expect("server in view") +} + +/// The Phase 4 c12 gate: remote monitors report the true reason, honour +/// corpses, leak nothing for unshipped pids, and cancel cleanly. +#[test] +fn remote_monitors_report_true_reasons() { + maybe_child(ROLES); + let mut server = spawn_node("server", &[]); + let saddr = server.wait_listening(); + server.wait_line("READY", |l| l == "READY"); + let mut client = spawn_node("client", &[("SMARM_SERVER_ADDR", &saddr)]); + client.wait_line("DOWN exit Local(Exit)", |l| l == "DOWN exit Local(Exit)"); + client.wait_line("DOWN panic Local(Panic)", |l| { + l == "DOWN panic Local(Panic)" + }); + client.wait_line("DOWN corpse Local(Exit)", |l| { + l == "DOWN corpse Local(Exit)" + }); + client.wait_line("DOWN hidden Local(NoProc)", |l| { + l == "DOWN hidden Local(NoProc)" + }); + client.wait_line("DOWN bogus Local(NoProc)", |l| { + l == "DOWN bogus Local(NoProc)" + }); + client.wait_line("DEMONITOR stray=false", |l| l == "DEMONITOR stray=false"); + client.wait_line("CLIENT DONE", |l| l == "CLIENT DONE"); +} diff --git a/tests/cluster_pg.rs b/tests/cluster_pg.rs new file mode 100644 index 0000000..626b4b8 --- /dev/null +++ b/tests/cluster_pg.rs @@ -0,0 +1,254 @@ +//! RFC 010 c15 — distributed pg: sync on `NodeUp`, incremental +//! `Join`/`Leave`, eager eviction announced, `NodeDown` sweep. +//! +//! Two nodes. The *origin* joins two local workers to `"pool"` before the +//! *observer* connects (so the observer's view comes from `Sync`), exposes a +//! `"go"` command inbox and then does exactly what the observer tells it: +//! kill one worker, join a third, leave with the second. The observer drives +//! that script through the cluster itself and asserts every step from +//! `members_all` — never touching the group on its own side, except once to +//! prove a mixed local+remote group reads correctly and that `members` stays +//! local. `dispatch_any` is exercised both ways: into the origin's worker +//! (remote pick, `send_to_remote`) and, once the origin is gone, into the +//! observer's own (local pick, `send_to`). Finally the parent SIGKILLs the +//! origin: the observer must sweep every remote member on `NodeDown`. +#![cfg(feature = "cluster")] + +mod common; + +use common::{maybe_child, spawn_node}; +use smarm::cluster::envelope::NodeMeta; +use smarm::cluster::expose::{expose, expose_type}; +use smarm::cluster::membership::{subscribe, NodeEvent}; +use smarm::cluster::remote::{self, RemoteName}; +use smarm::cluster::{ + dispatch_any, members_all, pick_any, start, Config, DispatchAnyError, GroupMember, StaticSeeds, + Timing, +}; +use smarm::{channel, join, leave, members, register, send_to, spawn_addr, Addressable, Name, Pid}; +use std::time::{Duration, Instant}; + +const GO: Name = Name::new("go"); +const POOL: &str = "pool"; + +/// A pool worker's message: `"die"` stops it, anything else is printed. +#[derive(Debug, PartialEq)] +struct Job(String); +struct Worker; +impl Addressable for Worker { + type Msg = Job; +} +impl serde::Serialize for Job { + fn serialize(&self, s: S) -> Result { + self.0.serialize(s) + } +} +impl<'de> serde::Deserialize<'de> for Job { + fn deserialize>(d: D) -> Result { + String::deserialize(d).map(Job) + } +} + +const ROLES: &[(&str, fn())] = &[("origin", role_origin), ("observer", role_observer)]; + +fn cfg(name: &str, seeds: Vec<(String, String)>) -> Config { + Config { + node_name: name.into(), + meta: NodeMeta { + role: "c15".into(), + region: "local".into(), + }, + listen_addr: "127.0.0.1:0".into(), + strategy: Box::new(StaticSeeds::new(seeds)), + timing: Timing::default(), + } +} + +/// A pool worker: prints every job it is handed, exits on `"die"`. +fn worker() -> Pid { + spawn_addr::(|rx| { + while let Ok(Job(s)) = rx.recv() { + if s == "die" { + return; + } + println!("JOB {s}"); + } + }) +} + +fn role_origin() { + smarm::run(|| { + let cluster = start(cfg("origin", vec![])).expect("binds"); + // Remote dispatch lands here only for a type this node accepts. + expose_type::(); + let w1 = worker(); + let w2 = worker(); + assert!(join(POOL, w1)); + assert!(join(POOL, w2)); + let (go_tx, go_rx) = channel::(); + register(GO, go_tx).unwrap(); + expose(GO); + println!("LISTENING {}", cluster.local_addr()); + println!("JOINED 2"); + loop { + match go_rx.recv().unwrap() { + 1 => { + send_to(w1, Job("die".into())).unwrap(); + println!("KILLED w1"); + } + 2 => { + assert!(leave(POOL, w2)); + println!("LEFT w2"); + } + 3 => { + let w3 = worker(); + assert!(join(POOL, w3)); + println!("JOINED w3"); + } + n => panic!("unknown command {n}"), + } + } + }); +} + +fn remote_count(group: &str) -> usize { + members_all(group) + .iter() + .filter(|m| matches!(m, GroupMember::Remote(_))) + .count() +} + +/// Cooperative poll until `pred`; panics (with the last view) on timeout. +fn wait_view(what: &str, group: &str, pred: impl Fn(&[GroupMember]) -> bool) { + let deadline = Instant::now() + Duration::from_secs(5); + loop { + let v = members_all(group); + if pred(&v) { + return; + } + assert!( + Instant::now() < deadline, + "timed out waiting for {what}; view = {v:?}" + ); + smarm::sleep(Duration::from_millis(5)); + } +} + +fn role_observer() { + let origin_addr = std::env::var("SMARM_ORIGIN_ADDR").expect("SMARM_ORIGIN_ADDR"); + smarm::run(move || { + let _cluster = start(cfg("observer", vec![("origin".into(), origin_addr)])).expect("binds"); + let ev = subscribe().unwrap(); + loop { + match ev.rx.recv().unwrap() { + NodeEvent::NodeUp(i) if i.name == "origin" => break, + _ => {} + } + } + let go = |n: u8| remote::send(RemoteName::new("origin", GO), n).unwrap(); + + // Sync: both pre-existing members arrive with no join on this side. + wait_view("sync of 2 remote members", POOL, |v| { + v.len() == 2 && v.iter().all(|m| matches!(m, GroupMember::Remote(_))) + }); + let synced = members_all(POOL); + assert!(synced.iter().all(|m| match m { + GroupMember::Remote(p) => p.node() == "origin", + GroupMember::Local(_) => false, + })); + println!("SEES 2"); + + // Origin-side death: the origin's reaper announces the leave. + go(1); + wait_view("death evicted on observer", POOL, |v| v.len() == 1); + println!("SEES 1 after death"); + + // Incremental Join. + go(3); + wait_view("incremental join", POOL, |v| v.len() == 2); + println!("SEES 2 after join"); + + // Voluntary Leave. + go(2); + wait_view("incremental leave", POOL, |v| v.len() == 1); + println!("SEES 1 after leave"); + + // Mixed group: our own member sits beside the remote one in + // `members_all`; `members` stays local-only. + let me = worker(); + assert!(join(POOL, me)); + wait_view("mixed local+remote", POOL, |v| { + v.len() == 2 && v.contains(&GroupMember::Local(me.erase())) + }); + assert_eq!( + members(POOL), + vec![me.erase()], + "local API never shows remotes" + ); + assert_eq!(remote_count(POOL), 1); + println!("MIXED ok"); + + // dispatch_any: the store's first entry is the origin's w3 (it was + // announced before we joined), so the pick is remote and the job + // crosses the wire — the origin's worker prints it. + let picked = pick_any(POOL).expect("pool has members"); + assert!( + matches!(picked, GroupMember::Remote(_)), + "first entry is remote: {picked:?}" + ); + let reached = dispatch_any::(POOL, Job("from-observer".into())).unwrap(); + assert_eq!(reached, picked); + println!("DISPATCHED remote"); + + println!("PARK"); + // Parent SIGKILLs the origin now: NodeDown must sweep its member, + // ours must survive. + wait_view("node_down sweep", POOL, |v| { + v == [GroupMember::Local(me.erase())] + }); + assert_eq!(members(POOL), vec![me.erase()]); + println!("SWEPT"); + + // Now the only member is ours: a local pick, a local send. + let reached = dispatch_any::(POOL, Job("local".into())).unwrap(); + assert_eq!(reached, GroupMember::Local(me.erase())); + // And an empty group hands the message back. + match dispatch_any::("nobody", Job("lost".into())) { + Err(DispatchAnyError::NoMember(Job(s))) => assert_eq!(s, "lost"), + other => panic!("expected NoMember, got {other:?}"), + } + println!("DISPATCHED local"); + loop { + smarm::sleep(Duration::from_secs(3600)); + } + }); +} + +/// The Phase 5 gate: sync, join, leave, death, node_down — all observed from +/// the peer, none of them a group operation on the peer — plus dispatch_any +/// reaching a remote member and a local one. +#[test] +fn groups_span_two_nodes() { + maybe_child(ROLES); + let mut origin = spawn_node("origin", &[]); + let addr = origin.wait_listening(); + origin.wait_line("JOINED 2", |l| l == "JOINED 2"); + let mut observer = spawn_node("observer", &[("SMARM_ORIGIN_ADDR", &addr)]); + observer.wait_line("SEES 2", |l| l == "SEES 2"); + origin.wait_line("KILLED w1", |l| l == "KILLED w1"); + observer.wait_line("SEES 1 after death", |l| l == "SEES 1 after death"); + origin.wait_line("JOINED w3", |l| l == "JOINED w3"); + observer.wait_line("SEES 2 after join", |l| l == "SEES 2 after join"); + origin.wait_line("LEFT w2", |l| l == "LEFT w2"); + observer.wait_line("SEES 1 after leave", |l| l == "SEES 1 after leave"); + observer.wait_line("MIXED ok", |l| l == "MIXED ok"); + observer.wait_line("DISPATCHED remote", |l| l == "DISPATCHED remote"); + origin.wait_line("JOB from-observer", |l| l == "JOB from-observer"); + observer.wait_line("PARK", |l| l == "PARK"); + origin.kill(); + observer.wait_line("SWEPT", |l| l == "SWEPT"); + // Order between the root's line and the worker's is scheduling; wait + // for the later one to be certain both happened. + observer.wait_line("DISPATCHED local", |l| l == "DISPATCHED local"); + observer.wait_line("JOB local", |l| l == "JOB local"); +} diff --git a/tests/cluster_pid_send.rs b/tests/cluster_pid_send.rs new file mode 100644 index 0000000..9c1a074 --- /dev/null +++ b/tests/cluster_pid_send.rs @@ -0,0 +1,356 @@ +//! RFC 010 c10 — pid targeting + auto-serialization. The Phase 3 gate: +//! cross-node call/reply with no ceremony, under the subprocess harness. +//! +//! Local suite (`run()`, no network): serialize/deserialize shapes, +//! self-collapse, the outside-runtime contract, the local send-site +//! incarnation check with a probe proving **no frame is emitted**. +//! +//! Cross-process: two nodes. The *server* exposes a `Name`; the +//! *client* sends a `Req` carrying its own `Pid` (auto-serialized to +//! a `RemotePid` on the wire); the server replies via `send_to_remote` +//! straight back to that pid — no name at the client end, no ceremony. A +//! third-node roundtrip: the client's pid travels client→server→relay→ +//! server→client, and still delivers. +#![cfg(feature = "cluster")] + +mod common; + +use common::{maybe_child, spawn_node}; +use smarm::cluster::envelope::{encode_payload, Frame, NodeMeta}; +use smarm::cluster::expose::{expose, type_hash}; +use smarm::cluster::membership::{subscribe, NodeEvent}; +use smarm::cluster::remote::{self, send_to_remote, RemoteName, RemotePid, ToRemoteError}; +use smarm::cluster::{start, Config, StaticSeeds, Timing}; +use smarm::pg::Incarnation; +use smarm::{channel, install, register, run, Addressable, Name, Pid}; +use std::time::Duration; + +// ---- message types (std-only payloads; the crate's serde is derive-less, +// so wire types are hand-rolled with serde's tuple/seq API via `serde::ser` +// impls below — the same thing a user's derive would generate) ------------ + +/// A request carrying a reply-to. Serialize/Deserialize are written by hand +/// here for exactly one reason: this crate deliberately does not pull in +/// serde-derive. Field 1 is the auto-serializing pid. +#[derive(Debug, PartialEq)] +struct Req { + text: String, + reply_to: RemotePid, +} + +#[derive(Debug, PartialEq)] +struct Reply(String); + +struct Replier; +impl Addressable for Replier { + type Msg = Reply; +} + +impl serde::Serialize for Req { + fn serialize(&self, s: S) -> Result { + use serde::ser::SerializeTuple; + let mut t = s.serialize_tuple(2)?; + t.serialize_element(&self.text)?; + t.serialize_element(&self.reply_to)?; + t.end() + } +} +impl<'de> serde::Deserialize<'de> for Req { + fn deserialize>(d: D) -> Result { + let (text, reply_to) = <(String, RemotePid)>::deserialize(d)?; + Ok(Req { text, reply_to }) + } +} +impl serde::Serialize for Reply { + fn serialize(&self, s: S) -> Result { + self.0.serialize(s) + } +} +impl<'de> serde::Deserialize<'de> for Reply { + fn deserialize>(d: D) -> Result { + String::deserialize(d).map(Reply) + } +} + +// ================= local suite ========================================= + +/// A local `Pid` serializes as a `RemotePid` stamped with this node's +/// identity; deserializing it back on the same node collapses to the same +/// local pid (`local()` is `Some`, `Pid` round-trips). +#[test] +fn local_pid_serializes_and_collapses_on_self() { + maybe_child(ROLES); + run(|| { + // The local identity is set by cluster::start; the local suite sets + // it directly. + remote::set_local_identity("me", Incarnation::new(7)); + let (tx, _rx) = channel::(); + let me: Pid = install::(tx); + + let bytes = encode_payload(&me).unwrap(); + let rp: RemotePid = smarm::cluster::envelope::decode_payload(&bytes).unwrap(); + assert_eq!(rp.node(), "me"); + assert_eq!(rp.incarnation(), Incarnation::new(7)); + assert_eq!( + rp.local(), + Some(me), + "self-node pid collapses to the local pid" + ); + + // Deserializing straight into Pid works for a self-node pid... + let back: Pid = smarm::cluster::envelope::decode_payload(&bytes).unwrap(); + assert_eq!(back, me); + + // ...and FAILS for a foreign one (collapse is literal: node == self). + let foreign = RemotePid::::from_parts("elsewhere", Incarnation::new(1), 3, 1); + let fbytes = encode_payload(&foreign).unwrap(); + assert!(smarm::cluster::envelope::decode_payload::>(&fbytes).is_err()); + assert_eq!(foreign.local(), None); + }); +} + +/// `send_to_remote` short-circuits locally for a self-node pid — the +/// zero-copy-equivalent collapse: the message object itself lands in the +/// local channel, no encode, no frame. +#[test] +fn send_to_remote_collapses_locally_for_self() { + maybe_child(ROLES); + run(|| { + remote::set_local_identity("me", Incarnation::new(7)); + let (tx, rx) = channel::(); + let me: Pid = install::(tx); + let rp = RemotePid::from_local(me).expect("identity set"); + // Probe the outbound path: nothing must be handed to any connection. + let (probe_tx, probe_rx) = channel::(); + remote::bind_outbound_probe("me", Incarnation::new(7), probe_tx); + + send_to_remote(rp, Reply("hi".into())).unwrap(); + assert_eq!(rx.recv().unwrap(), Reply("hi".into())); + assert!( + matches!(probe_rx.try_recv(), Ok(None)), + "no frame for a local collapse" + ); + }); +} + +/// RFC v2 §3: a `RemotePid` whose incarnation is not the current one for its +/// node fails at the local send site with `DeadIncarnation`, and NO frame +/// is emitted — asserted on a probe sender bound as that node's outbound. +#[test] +fn stale_incarnation_rejected_locally_no_frame() { + maybe_child(ROLES); + run(|| { + remote::set_local_identity("me", Incarnation::new(7)); + let (probe_tx, probe_rx) = channel::(); + remote::bind_outbound_probe("peer", Incarnation::new(5), probe_tx); + + let stale = RemotePid::::from_parts("peer", Incarnation::new(4), 9, 1); + match send_to_remote(stale, Reply("late".into())) { + Err(ToRemoteError::DeadIncarnation(Reply(s))) => assert_eq!(s, "late"), + other => panic!("expected DeadIncarnation, got {other:?}"), + } + assert!( + matches!(probe_rx.try_recv(), Ok(None)), + "stale pid must emit no frame" + ); + + // The current incarnation goes through: a Send frame with the pid's + // (index, generation) and Reply's hash lands on the probe. + let live = RemotePid::::from_parts("peer", Incarnation::new(5), 9, 1); + send_to_remote(live, Reply("now".into())).unwrap(); + match probe_rx.recv().unwrap() { + Frame::Send { + index, + generation, + type_hash: h, + payload, + } => { + assert_eq!((index, generation), (9, 1)); + assert_eq!(h, type_hash::()); + let r: Reply = smarm::cluster::envelope::decode_payload(&payload).unwrap(); + assert_eq!(r, Reply("now".into())); + } + f => panic!("expected Send, got {f:?}"), + } + + // Unknown node: NotConnected, no frame anywhere. + let nowhere = RemotePid::::from_parts("nowhere", Incarnation::new(1), 1, 1); + assert!(matches!( + send_to_remote(nowhere, Reply("x".into())), + Err(ToRemoteError::NotConnected(_)) + )); + }); +} + +// ================= cross-process gate ================================== + +const ROLES: &[(&str, fn())] = &[ + ("server", role_server), + ("client", role_client), + ("relay", role_relay), +]; + +const ECHO: Name = Name::new("c10.echo"); +const RELAY: Name = Name::new("c10.relay"); + +fn cfg(name: &str, seeds: Vec<(String, String)>) -> Config { + Config { + node_name: name.to_string(), + meta: NodeMeta { + role: "c10".into(), + region: "local".into(), + }, + listen_addr: std::env::var("SMARM_LISTEN_ADDR").unwrap_or_else(|_| "127.0.0.1:0".into()), + strategy: Box::new(StaticSeeds::new(seeds)), + timing: Timing::default(), + } +} + +fn wait_up(events: &smarm::cluster::membership::MembershipEvents, who: &str) { + loop { + match events.rx.recv() { + Ok(NodeEvent::NodeUp(i)) if i.name == who => return, + Ok(_) => continue, + Err(_) => panic!("manager gone"), + } + } +} + +/// Server: exposes ECHO; each Req is answered by `send_to_remote` to its +/// reply_to — the server never learns a name for the client. If the Req text +/// starts with "via-relay:", it forwards the whole Req (reply_to and all) to +/// the relay node instead, which sends it back here; the second arrival is +/// answered normally. That is the pid's third-node roundtrip. +fn role_server() { + let relay_addr = std::env::var("SMARM_RELAY_ADDR").ok(); + smarm::run(move || { + let seeds = relay_addr + .map(|a| vec![("relay".to_string(), a)]) + .unwrap_or_default(); + let cluster = start(cfg("server", seeds)).expect("binds"); + println!("LISTENING {}", cluster.local_addr()); + let (tx, rx) = channel::(); + register(ECHO, tx).unwrap(); + expose(ECHO); + println!("READY"); + loop { + let req = rx.recv().unwrap(); + if let Some(rest) = req.text.strip_prefix("via-relay:") { + let fwd = Req { + text: format!("relayed:{rest}"), + reply_to: req.reply_to, + }; + remote::send(RemoteName::new("relay", RELAY), fwd).unwrap(); + println!("FORWARDED"); + continue; + } + println!("REQ {}", req.text); + send_to_remote(req.reply_to, Reply(format!("echo:{}", req.text))).unwrap(); + } + }); +} + +/// Relay: exposes RELAY; bounces every Req straight back to the server's +/// ECHO, untouched. The client's pid inside it now crosses relay→server. +fn role_relay() { + let server_addr = std::env::var("SMARM_SERVER_ADDR").expect("SMARM_SERVER_ADDR"); + smarm::run(move || { + let cluster = start(cfg("relay", vec![("server".into(), server_addr)])).expect("binds"); + println!("LISTENING {}", cluster.local_addr()); + let (tx, rx) = channel::(); + register(RELAY, tx).unwrap(); + expose(RELAY); + let ev = subscribe().unwrap(); + wait_up(&ev, "server"); + println!("READY"); + loop { + let req = rx.recv().unwrap(); + println!("RELAYING {}", req.text); + remote::send(RemoteName::new("server", ECHO), req).unwrap(); + } + }); +} + +/// Client: connects to server, installs a Reply inbox on its own pid, +/// declares it accepts `Reply` (`expose_type` — the RFC's one kept piece of +/// ceremony: nothing is remotely deliverable by default), sends a Req with +/// `reply_to = my pid` (auto-serialized), awaits the reply. +fn role_client() { + let server_addr = std::env::var("SMARM_SERVER_ADDR").expect("SMARM_SERVER_ADDR"); + let via_relay = std::env::var("SMARM_VIA_RELAY").is_ok(); + smarm::run(move || { + let _cluster = start(cfg("client", vec![("server".into(), server_addr)])).expect("binds"); + let ev = subscribe().unwrap(); + wait_up(&ev, "server"); + println!("MEMBER-UP server"); + + let (tx, rx) = channel::(); + let me: Pid = install::(tx); + // The one deliberate line: a pid-targeted inbound is deliverable only + // for types this node has said it accepts (RFC §4, the safety). + smarm::cluster::expose::expose_type::(); + let text = if via_relay { "via-relay:ping" } else { "ping" }; + remote::send( + RemoteName::new("server", ECHO), + Req { + text: text.into(), + reply_to: RemotePid::from_local(me).expect("identity set"), + }, + ) + .unwrap(); + println!("SENT"); + let Reply(s) = rx.recv().unwrap(); + println!("REPLY {s}"); + loop { + smarm::sleep(Duration::from_secs(3600)); + } + }); +} + +/// The gate: cross-node call/reply with no ceremony. +#[test] +fn cross_node_call_reply_no_ceremony() { + maybe_child(ROLES); + let mut server = spawn_node("server", &[]); + let saddr = server.wait_listening(); + server.wait_line("READY", |l| l == "READY"); + let mut client = spawn_node("client", &[("SMARM_SERVER_ADDR", &saddr)]); + client.wait_line("SENT", |l| l == "SENT"); + server.wait_line("REQ ping", |l| l == "REQ ping"); + client.wait_line("REPLY echo:ping", |l| l == "REPLY echo:ping"); +} + +/// The client's pid, round-tripped through a third node, still delivers. +#[test] +fn pid_roundtrips_through_third_node() { + maybe_child(ROLES); + // Relay needs the server address; server needs the relay address — + // pre-reserve the relay port (same accepted micro-window as cluster_mesh). + let relay_addr = { + let l = std::net::TcpListener::bind("127.0.0.1:0").unwrap(); + l.local_addr().unwrap().to_string() + }; + let mut server = spawn_node("server", &[("SMARM_RELAY_ADDR", &relay_addr)]); + let saddr = server.wait_listening(); + server.wait_line("READY", |l| l == "READY"); + let mut relay = spawn_node( + "relay", + &[ + ("SMARM_SERVER_ADDR", &saddr), + ("SMARM_LISTEN_ADDR", &relay_addr), + ], + ); + let _ = relay.wait_listening(); + relay.wait_line("READY", |l| l == "READY"); + let mut client = spawn_node( + "client", + &[("SMARM_SERVER_ADDR", &saddr), ("SMARM_VIA_RELAY", "1")], + ); + client.wait_line("SENT", |l| l == "SENT"); + server.wait_line("FORWARDED", |l| l == "FORWARDED"); + relay.wait_line("RELAYING", |l| l.starts_with("RELAYING")); + server.wait_line("REQ relayed:ping", |l| l == "REQ relayed:ping"); + client.wait_line("REPLY echo:relayed:ping", |l| { + l == "REPLY echo:relayed:ping" + }); +} diff --git a/tests/cluster_remote_send.rs b/tests/cluster_remote_send.rs index 2f6e1e1..fc93486 100644 --- a/tests/cluster_remote_send.rs +++ b/tests/cluster_remote_send.rs @@ -33,7 +33,7 @@ use smarm::cluster::envelope::NodeMeta; use smarm::cluster::expose::expose; use smarm::cluster::membership::{subscribe, NodeEvent}; use smarm::cluster::remote::{send_remote_raw, RemoteName, RemoteSendError}; -use smarm::cluster::{start, Config, StaticSeeds}; +use smarm::cluster::{start, Config, StaticSeeds, Timing}; use smarm::{channel, register, Name}; use std::time::Duration; @@ -51,6 +51,7 @@ fn base_config(name: &str, seeds: Vec<(String, String)>) -> Config { }, listen_addr: "127.0.0.1:0".to_string(), strategy: Box::new(StaticSeeds::new(seeds)), + timing: Timing::default(), } } diff --git a/tests/pg.rs b/tests/pg.rs index 180ae10..3b06bbe 100644 --- a/tests/pg.rs +++ b/tests/pg.rs @@ -1,5 +1,6 @@ //! Process-group tests that run under the scheduler: `join` installs a real -//! monitor on a live actor, and a real death drives eviction on next contact. +//! monitor on a live actor, and a real death drives eviction (the reaper +//! actor sweeps it; the read path hides it in the meantime). //! (Pure structural invariants live in the `pg` unit tests.) use smarm::{channel, members, pick, run, spawn}; @@ -35,19 +36,15 @@ fn a_dead_actor_vanishes_from_every_group_it_joined() { assert_eq!(members("g1"), vec![pid]); assert_eq!(members("g2"), vec![pid]); - // Release and reap the actor. finalize_actor queues the Down to our - // monitors before unparking joiners, so by the time join() returns the - // Down is already waiting in the membership channel. + // Release the actor. finalize_actor queues the Down to the reaper and + // marks the slot dead before unparking joiners, so by the time join() + // returns every read hides the pid whether or not the reaper has run. tx.send(()).unwrap(); h.join().unwrap(); - // Drain-on-contact: touching g1 detects the death and sweeps the pid - // out of every group (g2 included), not just g1. - assert!(members("g1").is_empty(), "evicted from the touched group"); - assert!( - members("g2").is_empty(), - "and swept from the untouched group" - ); + // Gone from every group it joined, not just one. + assert!(members("g1").is_empty(), "gone from g1"); + assert!(members("g2").is_empty(), "and from g2"); assert_eq!(pick("g1"), None); }); } @@ -86,11 +83,7 @@ fn live_members_survive_a_peers_death() { tx_a.send(()).unwrap(); a.join().unwrap(); - assert_eq!( - members("svc"), - vec![b.pid()], - "only the dead peer is reaped" - ); + assert_eq!(members("svc"), vec![b.pid()], "only the dead peer is gone"); assert_eq!(pick("svc"), Some(b.pid())); tx_b.send(()).unwrap(); @@ -123,18 +116,18 @@ fn leave_drops_a_membership_without_affecting_others() { } #[test] -fn joining_an_already_dead_pid_is_evicted_on_next_contact() { +fn joining_an_already_dead_pid_never_shows_in_a_read() { run(|| { let h = spawn(|| {}); let pid = h.pid(); h.join().unwrap(); // actor is finalized before we join it to anything - // monitor() on a gone pid queues a NoProc Down immediately, so the - // membership is reaped the next time the group is touched. + // join() on a gone pid queues a NoProc Down to the reaper immediately; + // reads never show it either way (slot-liveness backstop). join("late", pid); assert!( members("late").is_empty(), - "dead-at-join member is reaped on read" + "dead-at-join member never reads as live" ); assert_eq!(pick("late"), None); });