Merge branch 'rfc010-cluster': RFC 010 clustering (c1–c16 + Phase 6) onto the v0.7.0 + perf-audit tree

Both lines branched from ca1c983 (v0.6.1 + try_spawn). master carried
graceful shutdown, gen_server/gen_statem lifetime, v0.7.0 and the 16
perf-audit commits; rfc010-cluster carried clustering behind
`--features cluster`. Two resolutions beyond the automatic merge:

- tests/channel.rs: both sides deflaked the spawn-then-monitor race in
  channel_ops_interleaved_with_monitor_churn_multi_thread. Kept master's
  `spawn_monitor` (monitor registered before publish) over the cluster
  side's `go`-gated spawn; same intent, API-level fix.
- src/cluster/envelope.rs: master added `DownReason::Shutdown`, which
  made the `Frame::Down` reason codec non-exhaustive. New wire tag
  DR_SHUTDOWN = 6, encoded and decoded symmetrically. `Shutdown` never
  rides in a `Down` by contract (a target that honours the request
  exits normally); the tag exists so the codec stays total. Tags 1–5
  unchanged.

Gates on the merged tree (rustc 1.98.1): default 405/0, cluster 492/0
(0 ignored beyond the 11 pre-existing `ignore` doctests), clippy --lib
-D warnings on both configs, doctests. cluster_disconnect still ~1.8s
(SMARM_FAST_TIMING plumbing intact).
This commit is contained in:
claude-asm-audit
2026-09-12 05:25:50 +00:00
45 changed files with 10384 additions and 310 deletions
+117
View File
@@ -0,0 +1,117 @@
//! RFC 010 c6a — connection-actor lifecycle against the manager table.
//!
//! The handshake is bypassed here (c6b wires it): each connection is
//! constructed already-established over a real localhost TCP pair, handed a
//! fabricated `Peer`, and spawned. `spawn_established` registers it with the
//! manager, which takes its handle and monitors it, so the table reflects the
//! connection while it lives and reaps it on any exit path. This proves three
//! things at once: a live connection shows up, a commanded `Disconnect`
//! removes exactly that one, and a peer close (EOF, no command) removes the
//! other.
//!
//! TCP parks the calling actor, so everything runs inside `smarm::run`; the
//! single-threaded runtime is fine because every wait is a cooperative fd park.
#![cfg(feature = "cluster")]
use std::time::Duration;
use smarm::cluster::envelope::NodeMeta;
use smarm::cluster::handshake::Peer;
use smarm::cluster::manager::{Call, Manager, Reply, MANAGER};
use smarm::cluster::spawn_established;
use smarm::cluster::transport::tcp::TcpTransport;
use smarm::cluster::transport::{Conn, FramedConn, Transport};
use smarm::cluster::Timing;
use smarm::gen_server::{self, GenServerBuilder};
use smarm::pg::Incarnation;
use smarm::{run, sleep};
/// A fabricated post-handshake peer identity. Only `node_name` matters to the
/// manager table; the rest is filler until c7 consumes it.
fn peer(name: &str) -> Peer {
Peer {
node_name: name.to_string(),
incarnation: Incarnation::new(1),
meta: NodeMeta {
role: "test".to_string(),
region: "test".to_string(),
},
}
}
/// One established transport pair over localhost. Relies on TCP backlog so the
/// sequential dial-then-accept needs no concurrent acceptor (same assumption as
/// the c3 conformance suite).
fn pair(t: &dyn Transport) -> (Box<dyn Conn>, Box<dyn Conn>) {
let mut l = t.listen("127.0.0.1:0").unwrap();
let a = t.dial(&l.local_addr()).unwrap();
let b = l.accept().unwrap();
(a, b)
}
/// Poll the manager until its peer set matches `expected` (sorted), or fail.
/// The bound is generous against a sub-millisecond real cost.
fn wait_peers(expected: &[&str]) {
let want: Vec<String> = expected.iter().map(|s| s.to_string()).collect();
for _ in 0..2000 {
if let Ok(Reply::Peers(got)) = gen_server::call(MANAGER, Call::Peers) {
if got == want {
return;
}
}
sleep(Duration::from_millis(1));
}
let got = gen_server::call(MANAGER, Call::Peers);
panic!("timed out waiting for peers == {want:?}; last = {got:?}");
}
#[test]
fn connection_up_commanded_shutdown_and_eof_all_reflected_in_table() {
run(|| {
// The manager, started plainly and reachable at its well-known name.
// (The supervised subtree in `cluster::start` is permanent by design;
// a plainly-started manager lets this test terminate cleanly.)
let mgr = GenServerBuilder::new(Manager::new())
.named(MANAGER)
.start()
.expect("manager name is free");
let t = TcpTransport;
let (a1, b1) = pair(&t);
let (a2, b2) = pair(&t);
// Manage the `a` ends as peers node-b and node-c; keep the `b` far ends
// open so neither socket is closed from the far side yet.
spawn_established(FramedConn::new(a1), peer("node-b"), Timing::default())
.expect("node-b registers");
spawn_established(FramedConn::new(a2), peer("node-c"), Timing::default())
.expect("node-c registers");
// Up: both connections register and the table shows them.
wait_peers(&["node-b", "node-c"]);
// A commanded disconnect reaps exactly its own connection: the
// manager drops that entry's handle and the actor stops.
assert!(matches!(
gen_server::call(
MANAGER,
Call::Disconnect {
name: "node-b".to_string()
}
),
Ok(Reply::Disconnected)
));
wait_peers(&["node-c"]);
// A peer close (EOF) reaps the other with no command at all.
drop(b2);
wait_peers(&[]);
// node-b's far end stayed open until here, so its removal above was the
// disconnect command and not an EOF.
drop(b1);
// All connection actors have exited; stop the manager so `run` returns.
mgr.shutdown();
});
}
+180
View File
@@ -0,0 +1,180 @@
//! RFC 010 c6c — heartbeat send + fixed-timeout liveness + teardown.
//!
//! Each case runs one real connection actor over an in-process localhost TCP
//! pair, with the far end held as a raw `FramedConn` (no actor) so the test
//! controls exactly what — if anything — the peer says. That gives the three
//! protocol-visible facts direct handles: heartbeats appear on the wire
//! unprompted; a mute peer is torn down (and reaped from the manager table)
//! once `LIVENESS_TIMEOUT` empties; and a peer that does nothing but send
//! heartbeats keeps the connection alive past that same window.
//!
//! Loopback has no fd and cannot drive liveness (documented on the actor),
//! so everything here is TCP. TCP parks the calling actor, so everything
//! runs inside `smarm::run`.
#![cfg(feature = "cluster")]
use std::time::{Duration, Instant};
use smarm::cluster::conn::{HEARTBEAT_INTERVAL, LIVENESS_TIMEOUT};
use smarm::cluster::envelope::{Frame, NodeMeta};
use smarm::cluster::handshake::Peer;
use smarm::cluster::manager::{Call, Manager, Reply, MANAGER};
use smarm::cluster::spawn_established;
use smarm::cluster::transport::tcp::TcpTransport;
use smarm::cluster::transport::{Conn, FramedConn, Transport};
use smarm::cluster::Timing;
use smarm::gen_server::{self, GenServerBuilder};
use smarm::pg::Incarnation;
use smarm::{run, sleep, spawn};
/// A fabricated post-handshake peer identity (same shape as the c6a suite).
fn peer(name: &str) -> Peer {
Peer {
node_name: name.to_string(),
incarnation: Incarnation::new(1),
meta: NodeMeta {
role: "test".to_string(),
region: "test".to_string(),
},
}
}
/// One established transport pair over localhost (TCP backlog covers the
/// sequential dial-then-accept, as in the c3 conformance suite).
fn pair(t: &dyn Transport) -> (Box<dyn Conn>, Box<dyn Conn>) {
let mut l = t.listen("127.0.0.1:0").unwrap();
let a = t.dial(&l.local_addr()).unwrap();
let b = l.accept().unwrap();
(a, b)
}
fn peers() -> Vec<String> {
match gen_server::call(MANAGER, Call::Peers) {
Ok(Reply::Peers(p)) => p,
other => panic!("manager unreachable: {other:?}"),
}
}
/// Poll until the manager's peer set matches `expected` (sorted) or `budget`
/// runs out.
fn wait_peers(expected: &[&str], budget: Duration) {
let want: Vec<String> = expected.iter().map(|s| s.to_string()).collect();
let deadline = Instant::now() + budget;
while Instant::now() < deadline {
if peers() == want {
return;
}
sleep(Duration::from_millis(10));
}
panic!(
"timed out waiting for peers == {want:?}; last = {:?}",
peers()
);
}
/// The actor emits heartbeats unprompted: the raw far end, saying nothing,
/// sees a `Frame::Heartbeat` well within one interval (the first goes out at
/// spawn).
#[test]
fn heartbeats_are_sent_unprompted() {
run(|| {
let mgr = GenServerBuilder::new(Manager::new())
.named(MANAGER)
.start()
.expect("manager name is free");
let (a, b) = pair(&TcpTransport);
spawn_established(FramedConn::new(a), peer("hb-send"), Timing::default())
.expect("register");
let mut far = FramedConn::new(b);
let frame = far
.recv_deadline(Instant::now() + HEARTBEAT_INTERVAL)
.expect("a heartbeat before one interval elapses");
assert_eq!(frame, Some(Frame::Heartbeat));
// Teardown: closing the far end is an EOF at the actor.
far.close();
wait_peers(&[], Duration::from_secs(2));
mgr.shutdown();
});
}
/// A mute peer is dead: no inbound frame for `LIVENESS_TIMEOUT` tears the
/// connection down and the manager's monitor reaps the table entry. The
/// entry is still present well inside the window — the teardown is the
/// timer, not an accident of setup.
#[test]
fn mute_peer_is_torn_down_after_liveness_timeout() {
run(|| {
let mgr = GenServerBuilder::new(Manager::new())
.named(MANAGER)
.start()
.expect("manager name is free");
let (a, b) = pair(&TcpTransport);
spawn_established(FramedConn::new(a), peer("mute"), Timing::default()).expect("register");
// Held open and silent: no frames, no EOF. (Unread inbound
// heartbeats sit in kernel buffers; they are 5 bytes each.)
let _far = FramedConn::new(b);
// Well inside the window the connection is still up.
sleep(LIVENESS_TIMEOUT / 2);
assert_eq!(peers(), vec!["mute".to_string()], "torn down too early");
// ...and once the window empties it is gone. Generous budget over
// the remaining half-window.
wait_peers(&[], LIVENESS_TIMEOUT);
mgr.shutdown();
});
}
/// Heartbeats alone keep a connection alive past `LIVENESS_TIMEOUT`: a far
/// end that sends `Frame::Heartbeat` at the interval (and nothing else)
/// holds the entry; when it goes quiet, liveness finally fires.
#[test]
fn heartbeats_keep_the_connection_alive() {
run(|| {
let mgr = GenServerBuilder::new(Manager::new())
.named(MANAGER)
.start()
.expect("manager name is free");
let (a, b) = pair(&TcpTransport);
spawn_established(FramedConn::new(a), peer("kept"), Timing::default()).expect("register");
// The far heartbeat pump: interval-paced sends until told to stop,
// then holds the socket open, silent, so the eventual teardown is
// liveness — not EOF.
let (ctl_tx, ctl_rx) = smarm::channel::channel::<()>();
spawn(move || {
let mut far = FramedConn::new(b);
// Phase 1: heartbeat at the interval until the first signal.
while matches!(ctl_rx.try_recv(), Ok(None)) {
far.send(&Frame::Heartbeat).expect("far send");
sleep(HEARTBEAT_INTERVAL);
}
// Phase 2: silent but with the socket held open — dropping
// `far` here would EOF the actor and mask the liveness path.
// Exits when the test's closure ends and drops `ctl_tx` (an
// eternal park would stop `run` from ever returning).
while matches!(ctl_rx.try_recv(), Ok(None)) {
sleep(Duration::from_millis(20));
}
});
// Past the liveness window with margin: still up.
sleep(LIVENESS_TIMEOUT + LIVENESS_TIMEOUT / 2);
assert_eq!(
peers(),
vec!["kept".to_string()],
"liveness fired despite heartbeats"
);
// Silence the pump; liveness now empties and the entry goes.
ctl_tx.send(()).expect("pump alive");
wait_peers(&[], LIVENESS_TIMEOUT * 2);
mgr.shutdown();
// `ctl_tx` drops here, releasing the pump's phase-2 wait.
});
}
+482
View File
@@ -0,0 +1,482 @@
//! RFC 010 c6b — the handshake on the accept/connect path.
//!
//! Path-level tests drive [`dial_handshake`]/[`accept_handshake`] over the
//! loopback transport on plain threads (its intended use — synchronous, no
//! runtime). Integration tests run the manager-backed [`dial`] and
//! [`spawn_acceptor`] over real localhost TCP inside `smarm::run`, and the
//! two-node case as subprocesses via the c4 harness. Flake budget: see
//! tests/common/mod.rs.
#![cfg(feature = "cluster")]
mod common;
use std::sync::mpsc;
use std::time::{Duration, Instant};
use common::{maybe_child, spawn_node, WAIT};
use smarm::cluster::connect::{
accept_handshake, dial, dial_handshake, spawn_acceptor, DialError, HandshakeError,
HANDSHAKE_TIMEOUT,
};
use smarm::cluster::envelope::{Frame, NodeMeta, RejectReason};
use smarm::cluster::handshake::{Local, PeerStanding};
use smarm::cluster::manager::{Call, Manager, Reply, MANAGER};
use smarm::cluster::transport::loopback::LoopbackTransport;
use smarm::cluster::transport::tcp::TcpTransport;
use smarm::cluster::transport::{FramedConn, Transport};
use smarm::cluster::Timing;
use smarm::gen_server::{self, GenServerBuilder};
use smarm::pg::Incarnation;
use smarm::{run, sleep};
const ROLES: &[(&str, fn())] = &[
("hs_listener", role_hs_listener),
("hs_dialer", role_hs_dialer),
];
const HASH: u64 = 0xC6B0_C6B0_C6B0_C6B0;
fn local(name: &str) -> Local {
Local {
node_name: name.into(),
incarnation: Incarnation::new(3),
build_hash: HASH,
meta: NodeMeta {
role: "test".into(),
region: "test".into(),
},
}
}
/// A loopback conn pair as `FramedConn`s, ready for a threaded handshake.
fn loopback_pair() -> (FramedConn, FramedConn) {
let t = LoopbackTransport::default();
let mut l = t.listen("hs").unwrap();
let dialer = FramedConn::new(t.dial("hs").unwrap());
let accepted = FramedConn::new(l.accept().unwrap());
(dialer, accepted)
}
/// Far-future deadline for loopback paths, where it cannot fire anyway.
fn no_deadline() -> Instant {
Instant::now() + Duration::from_secs(3600)
}
/// Park the node forever: it has announced everything the parent asserts on,
/// and must now hold its connection open until SIGKILLed.
fn park() -> ! {
loop {
sleep(Duration::from_secs(1));
}
}
/// Cooperative bounded receive across the closure/actor boundary. A blocking
/// `std::mpsc` wait would park the OS thread and starve the single-threaded
/// scheduler, so every wait inside `run` polls with [`sleep`] instead.
fn poll_recv<T>(rx: &mpsc::Receiver<T>, what: &str) -> T {
let deadline = Instant::now() + WAIT;
loop {
match rx.try_recv() {
Ok(v) => return v,
Err(mpsc::TryRecvError::Empty) => {
assert!(Instant::now() < deadline, "timed out waiting for {what}");
sleep(Duration::from_millis(1));
}
Err(mpsc::TryRecvError::Disconnected) => panic!("channel closed waiting for {what}"),
}
}
}
// ---------------------------------------------------------------------------
// Path level, over loopback on plain threads
// ---------------------------------------------------------------------------
#[test]
fn loopback_happy_path_establishes_both_ends() {
maybe_child(ROLES);
let (mut dialer, mut accepted) = loopback_pair();
let responder = std::thread::spawn(move || {
accept_handshake(
&mut accepted,
local("node-b"),
|name| {
assert_eq!(name, "node-a");
PeerStanding::Free
},
no_deadline(),
)
});
let peer_of_dialer = dial_handshake(&mut dialer, &local("node-a"), no_deadline()).unwrap();
let peer_of_acceptor = responder.join().unwrap().unwrap();
assert_eq!(peer_of_dialer.node_name, "node-b");
assert_eq!(peer_of_acceptor.node_name, "node-a");
}
#[test]
fn loopback_hash_mismatch_rejected_with_frame_then_eof() {
maybe_child(ROLES);
let (mut dialer, mut accepted) = loopback_pair();
let mut wrong = local("node-b");
wrong.build_hash ^= 1;
let responder = std::thread::spawn(move || {
accept_handshake(&mut accepted, wrong, |_| PeerStanding::Free, no_deadline())
});
// The dial side receives the reject frame — the compatibility anchor.
match dial_handshake(&mut dialer, &local("node-a"), no_deadline()) {
Err(HandshakeError::Rejected(RejectReason::HashMismatch)) => {}
other => panic!("expected HashMismatch reject, got {other:?}"),
}
match responder.join().unwrap() {
Err(HandshakeError::Rejected(RejectReason::HashMismatch)) => {}
other => panic!("expected accept side to report the reject, got {other:?}"),
}
}
#[test]
fn loopback_tie_break_loser_closed_silently() {
maybe_child(ROLES);
// The inbound dial is from "node-z"; we are "node-a" with our own dial to
// node-z in flight. dial_wins("node-z", "node-a") is false, so the
// inbound loses: closed with no frame at all.
let (mut dialer, mut accepted) = loopback_pair();
let responder = std::thread::spawn(move || {
accept_handshake(
&mut accepted,
local("node-a"),
|_| PeerStanding::Dialing,
no_deadline(),
)
});
// Silent close: the dial side sees EOF, never a frame.
match dial_handshake(&mut dialer, &local("node-z"), no_deadline()) {
Err(HandshakeError::Closed) => {}
other => panic!("expected silent close (Closed), got {other:?}"),
}
match responder.join().unwrap() {
Err(HandshakeError::TieBreakLoss) => {}
other => panic!("expected TieBreakLoss on the accept side, got {other:?}"),
}
}
#[test]
fn loopback_read_ahead_past_hello_survives_into_established_conn() {
maybe_child(ROLES);
// The buffer trap, proven: the dialer coalesces Hello + Heartbeat before
// the responder's first read, so the Heartbeat lands in the shared
// FramedConn's decode buffer during the handshake. The dialer sends
// nothing afterwards — the post-handshake recv can only succeed if the
// read-ahead travelled with the FramedConn.
let (mut dialer, mut accepted) = loopback_pair();
let (_init, hello) = smarm::cluster::handshake::Initiator::new(&local("node-a"));
dialer.send(&hello).unwrap();
dialer.send(&Frame::Heartbeat).unwrap();
// Both frames are buffered before the responder reads at all.
let (tx, rx) = mpsc::channel();
std::thread::spawn(move || {
let peer = accept_handshake(
&mut accepted,
local("node-b"),
|_| PeerStanding::Free,
no_deadline(),
)
.unwrap();
let next = accepted.recv();
let _ = tx.send((peer, next));
});
// A bounded wait: if the Heartbeat were NOT carried in the buffer, the
// recv above would block forever (the dialer stays open and silent).
let (peer, next) = rx
.recv_timeout(Duration::from_secs(5))
.expect("read-ahead lost: post-handshake recv blocked");
assert_eq!(peer.node_name, "node-a");
match next {
Ok(Some(Frame::Heartbeat)) => {}
other => panic!("expected the read-ahead Heartbeat, got {other:?}"),
}
drop(dialer);
}
// ---------------------------------------------------------------------------
// Deadline + manager integration, over TCP inside the runtime
// ---------------------------------------------------------------------------
#[test]
fn tcp_silent_peer_times_out_on_the_accept_path() {
maybe_child(ROLES);
run(|| {
let t = TcpTransport;
let mut l = t.listen("127.0.0.1:0").unwrap();
// Connect and then say nothing at all.
let silent = t.dial(&l.local_addr()).unwrap();
let mut accepted = FramedConn::new(l.accept().unwrap());
let (tx, rx) = mpsc::channel();
smarm::spawn(move || {
let r = accept_handshake(
&mut accepted,
local("node-b"),
|_| PeerStanding::Free,
Instant::now() + Duration::from_millis(200),
);
let _ = tx.send(r);
});
match poll_recv(&rx, "accept-path outcome") {
Err(HandshakeError::TimedOut) => {}
other => panic!("expected TimedOut, got {other:?}"),
}
drop(silent);
});
}
/// Poll the manager until its peer set matches `expected` (sorted), or fail.
fn wait_peers(expected: &[&str]) {
let want: Vec<String> = expected.iter().map(|s| s.to_string()).collect();
for _ in 0..5000 {
if let Ok(Reply::Peers(got)) = gen_server::call(MANAGER, Call::Peers) {
if got == want {
return;
}
}
sleep(Duration::from_millis(1));
}
let got = gen_server::call(MANAGER, Call::Peers);
panic!("timed out waiting for peers == {want:?}; last = {got:?}");
}
#[test]
fn tcp_duplicate_name_rejected_by_acceptor() {
maybe_child(ROLES);
run(|| {
let mgr = GenServerBuilder::new(Manager::new())
.named(MANAGER)
.start()
.expect("manager name is free");
let listener = TcpTransport.listen("127.0.0.1:0").unwrap();
let acceptor = spawn_acceptor(listener, local("node-b"), Timing::default());
let addr = acceptor.local_addr().to_string();
// First dial offering "dup-node": establishes and registers.
let mut first = FramedConn::new(TcpTransport.dial(&addr).unwrap());
let peer = dial_handshake(
&mut first,
&local("dup-node"),
Instant::now() + HANDSHAKE_TIMEOUT,
)
.unwrap();
assert_eq!(peer.node_name, "node-b");
wait_peers(&["dup-node"]);
// Second dial offering the same name: deterministic NameTaken.
let mut second = FramedConn::new(TcpTransport.dial(&addr).unwrap());
match dial_handshake(
&mut second,
&local("dup-node"),
Instant::now() + HANDSHAKE_TIMEOUT,
) {
Err(HandshakeError::Rejected(RejectReason::NameTaken)) => {}
other => panic!("expected NameTaken, got {other:?}"),
}
// The established connection was untouched by the rejected one.
wait_peers(&["dup-node"]);
// Teardown: the acceptor owns no connections, so the established one
// is torn down through the table.
acceptor.shutdown();
assert!(matches!(
gen_server::call(
MANAGER,
Call::Disconnect {
name: "dup-node".to_string()
}
),
Ok(Reply::Disconnected)
));
wait_peers(&[]);
first.close();
mgr.shutdown();
});
}
#[test]
fn dial_intent_cleared_when_dialer_dies() {
maybe_child(ROLES);
run(|| {
let mgr = GenServerBuilder::new(Manager::new())
.named(MANAGER)
.start()
.expect("manager name is free");
let (begun_tx, begun_rx) = mpsc::channel();
let (go_tx, go_rx) = mpsc::channel::<()>();
smarm::spawn(move || {
let me = smarm::self_pid();
match gen_server::call(
MANAGER,
Call::DialBegin {
name: "ghost".into(),
pid: me,
},
) {
Ok(Reply::DialBegan(true)) => {}
other => panic!("DialBegin failed: {other:?}"),
}
let _ = begun_tx.send(());
let () = poll_recv(&go_rx, "go signal");
panic!("dialer dies mid-dial");
});
poll_recv(&begun_rx, "DialBegin done");
// While the dialer lives, the intent is visible.
match gen_server::call(
MANAGER,
Call::Standing {
peer_name: "ghost".into(),
},
) {
Ok(Reply::Standing(s)) => assert_eq!(s, PeerStanding::Dialing),
other => panic!("PeerStanding failed: {other:?}"),
}
// Kill it; the monitor must clear the intent without cooperation.
go_tx.send(()).unwrap();
let deadline = Instant::now() + WAIT;
loop {
match gen_server::call(
MANAGER,
Call::Standing {
peer_name: "ghost".into(),
},
) {
Ok(Reply::Standing(s)) if s != PeerStanding::Dialing => break,
_ if Instant::now() > deadline => {
panic!("dial intent not cleared after dialer death")
}
_ => sleep(Duration::from_millis(1)),
}
}
mgr.shutdown();
});
}
// ---------------------------------------------------------------------------
// Two nodes, two processes: the integrated dial against a real acceptor
// ---------------------------------------------------------------------------
/// Announce, then park forever. Neither role ever tears its connection
/// down: a table entry only exists while the *peer* holds its side open, so
/// any teardown here would retract the other node's observation before it
/// had made it. The parent reaps both with SIGKILL once it has both
/// announcements (see [`common::Node`]'s `Drop`).
fn role_hs_listener() {
run(|| {
let _mgr = GenServerBuilder::new(Manager::new())
.named(MANAGER)
.start()
.expect("manager name is free");
let listener = TcpTransport.listen("127.0.0.1:0").unwrap();
let acceptor = spawn_acceptor(listener, local("node-b"), Timing::default());
println!("LISTENING {}", acceptor.local_addr());
wait_peers(&["node-a"]);
println!("PEERS node-a");
park();
});
}
fn role_hs_dialer() {
let addr = std::env::var("SMARM_PEER_ADDR").expect("SMARM_PEER_ADDR not set");
run(move || {
let _mgr = GenServerBuilder::new(Manager::new())
.named(MANAGER)
.start()
.expect("manager name is free");
let (tx, rx) = mpsc::channel();
smarm::spawn(move || {
let r = dial(
&TcpTransport,
&addr,
"node-b",
&local("node-a"),
Timing::default(),
);
let _ = tx.send(r);
});
if let Err(e) = poll_recv(&rx, "dial outcome") {
println!("DIAL failed: {e:?}");
std::process::exit(3);
}
wait_peers(&["node-b"]);
println!("PEERS node-b");
park();
});
}
#[test]
fn two_node_integrated_handshake_over_tcp() {
maybe_child(ROLES);
let mut listener = spawn_node("hs_listener", &[]);
let addr = listener.wait_listening();
let mut dialer = spawn_node("hs_dialer", &[("SMARM_PEER_ADDR", &addr)]);
// Each node reports its own table naming the other: a real dial against a
// real acceptor established in both directions. Both nodes then park —
// clean-exit behaviour is the c4 harness's own smoke test, and demanding
// it here would mean a teardown, which is exactly what cannot be ordered
// safely across two processes. Dropping the nodes SIGKILLs them.
dialer.wait_line("PEERS node-b", |l| l == "PEERS node-b");
listener.wait_line("PEERS node-a", |l| l == "PEERS node-a");
}
// ---------------------------------------------------------------------------
// Integrated-dial guardrails (no acceptor involved)
// ---------------------------------------------------------------------------
#[test]
fn concurrent_dial_to_same_name_refused() {
maybe_child(ROLES);
run(|| {
let mgr = GenServerBuilder::new(Manager::new())
.named(MANAGER)
.start()
.expect("manager name is free");
let (begun_tx, begun_rx) = mpsc::channel();
let (go_tx, go_rx) = mpsc::channel::<()>();
// First dialer parks with the intent held (it never connects —
// 'holding the intent' is all this test needs from it).
smarm::spawn(move || {
let me = smarm::self_pid();
assert!(matches!(
gen_server::call(
MANAGER,
Call::DialBegin {
name: "node-x".into(),
pid: me,
}
),
Ok(Reply::DialBegan(true))
));
let _ = begun_tx.send(());
let () = poll_recv(&go_rx, "go signal");
let _ = gen_server::call(
MANAGER,
Call::DialEnd {
name: "node-x".into(),
},
);
});
poll_recv(&begun_rx, "DialBegin done");
// Second integrated dial to the same name: refused before connecting
// (the addr is unroutable on purpose — it must never be dialed).
let (tx, rx) = mpsc::channel();
smarm::spawn(move || {
let r = dial(
&TcpTransport,
"127.0.0.1:1",
"node-x",
&local("node-a"),
Timing::default(),
);
let _ = tx.send(r);
});
match poll_recv(&rx, "second dial outcome") {
Err(DialError::AlreadyDialing) => {}
other => panic!("expected AlreadyDialing, got {other:?}"),
}
go_tx.send(()).unwrap();
mgr.shutdown();
});
}
+115
View File
@@ -0,0 +1,115 @@
//! RFC 010 — a seed whose address answers as a *different* name
//! (`DialError::PeerNameMismatch`) is dialed once and then parked: the
//! connector must not redial it on backoff forever.
//!
//! Observed from the misdialed peer: each such dial establishes at the
//! responder (it registers, `node_up`), then the dialer closes on the name
//! check (`node_down`) — one membership blip per attempt. Cross-process: a
//! *server* named `server` subscribes and reports; a *client* on fast
//! timing (50–500ms backoff) seeds `("wrongname", server_addr)`. After the
//! first blip the server counts further `NodeUp`s across 2s — several
//! backoff periods. Parked ⇒ zero. Negative-control-verified: with the park
//! stubbed out the count is ≥ 1 in the same window.
#![cfg(feature = "cluster")]
mod common;
use common::{maybe_child, spawn_node};
use smarm::cluster::envelope::NodeMeta;
use smarm::cluster::membership::{subscribe, NodeEvent};
use smarm::cluster::{start, Config, StaticSeeds, Timing};
use std::time::{Duration, Instant};
const ROLES: &[(&str, fn())] = &[("server", role_server), ("client", role_client)];
fn meta() -> NodeMeta {
NodeMeta {
role: "mismatch".into(),
region: "local".into(),
}
}
fn timing() -> Timing {
Timing {
initial_backoff: Duration::from_millis(50),
max_backoff: Duration::from_millis(500),
..Timing::default()
}
}
fn role_server() {
smarm::run(|| {
let cluster = start(Config {
node_name: "server".into(),
meta: meta(),
listen_addr: std::env::var("SMARM_LISTEN_ADDR")
.unwrap_or_else(|_| "127.0.0.1:0".into()),
strategy: Box::new(StaticSeeds::new(Vec::<(String, String)>::new())),
timing: timing(),
})
.expect("binds");
let ev = subscribe().unwrap();
println!("LISTENING {}", cluster.local_addr());
// First blip: the misdialed client establishes, then closes on us.
loop {
match ev.rx.recv() {
Ok(NodeEvent::NodeDown(i)) if i.name == "client" => break,
Ok(_) => continue,
Err(_) => panic!("manager gone"),
}
}
println!("BLIP");
// Now count further NodeUps across several backoff periods.
let mut more = 0usize;
let t0 = Instant::now();
while t0.elapsed() < Duration::from_millis(2000) {
match ev.rx.try_recv() {
Ok(Some(NodeEvent::NodeUp(i))) if i.name == "client" => more += 1,
Ok(_) => {}
Err(_) => panic!("manager gone"),
}
smarm::sleep(Duration::from_millis(50));
}
println!("MORE {more}");
loop {
smarm::sleep(Duration::from_secs(3600));
}
});
}
fn role_client() {
let server_addr = std::env::var("SMARM_SERVER_ADDR").expect("SMARM_SERVER_ADDR");
smarm::run(move || {
let _cluster = start(Config {
node_name: "client".into(),
meta: meta(),
listen_addr: "127.0.0.1:0".into(),
strategy: Box::new(StaticSeeds::new(vec![(
"wrongname".to_string(),
server_addr,
)])),
timing: timing(),
})
.expect("binds");
println!("CLIENT UP");
loop {
smarm::sleep(Duration::from_secs(3600));
}
});
}
#[test]
fn mismatched_seed_is_dialed_once_then_parked() {
maybe_child(ROLES);
let mut server = spawn_node("server", &[]);
let saddr = server.wait_listening();
let mut client = spawn_node("client", &[("SMARM_SERVER_ADDR", &saddr)]);
client.wait_line("CLIENT UP", |l| l == "CLIENT UP");
server.wait_line("BLIP", |l| l == "BLIP");
let line = server.wait_line("MORE", |l| l.starts_with("MORE "));
let more: usize = line.split_whitespace().nth(1).unwrap().parse().unwrap();
assert_eq!(
more, 0,
"mismatched seed was redialed {more}× after being parked"
);
}
+379
View File
@@ -0,0 +1,379 @@
//! RFC 010 c13 — connection-loss synthesis.
//!
//! Local suite (`run()`, no network): the read-side backstop. A
//! `RemoteMonitor` whose channel closes without a notice reads as
//! `Disconnected` exactly once (a `Monitor` command that reached the conn
//! actor's inbox but was never processed — the drain gap); after
//! `demonitor_remote` a closed channel stays a plain `Err`, never a notice.
//!
//! Cross-process: the headline contrast — an actor's own death gives its
//! TRUE reason, loss of the LINK gives `Disconnected` (both a commanded
//! `Disconnect` and a SIGKILLed peer process are `Disconnected` from the
//! monitor's view: nobody is left to say otherwise). Reconnect does not
//! resurrect: the old monitor yields nothing more, proven by stream ORDER
//! (a fresh monitor over the new link delivers first). The ignored test
//! trips liveness by SIGSTOP and then drops the link too, asserting exactly
//! one notice for one monitor.
#![cfg(feature = "cluster")]
mod common;
use common::{maybe_child, spawn_node};
use smarm::cluster::envelope::NodeMeta;
use smarm::cluster::expose::{expose, expose_type};
use smarm::cluster::manager::{Call, Reply, MANAGER};
use smarm::cluster::membership::{subscribe, NodeEvent};
use smarm::cluster::remote::{
self, demonitor_remote, monitor_remote, send_to_remote, RemoteName, RemotePid,
};
use smarm::cluster::{start, Config, RemoteDownReason, StaticSeeds, Timing};
use smarm::pg::Incarnation;
use smarm::{
channel, gen_server, install, register, run, spawn, Addressable, DownReason, Erased, Name, Pid,
};
use std::collections::HashMap;
use std::time::Duration;
// ---- message types (hand-rolled serde; the crate is derive-less) ---------
#[derive(Debug)]
struct Ctl {
cmd: String,
reply_to: RemotePid<Client>,
}
#[derive(Debug)]
struct Answer {
text: String,
pid: Option<RemotePid<Erased>>,
}
struct Client;
impl Addressable for Client {
type Msg = Answer;
}
impl serde::Serialize for Ctl {
fn serialize<S: serde::Serializer>(&self, s: S) -> Result<S::Ok, S::Error> {
use serde::ser::SerializeTuple;
let mut t = s.serialize_tuple(2)?;
t.serialize_element(&self.cmd)?;
t.serialize_element(&self.reply_to)?;
t.end()
}
}
impl<'de> serde::Deserialize<'de> for Ctl {
fn deserialize<D: serde::Deserializer<'de>>(d: D) -> Result<Self, D::Error> {
let (cmd, reply_to) = <(String, RemotePid<Client>)>::deserialize(d)?;
Ok(Ctl { cmd, reply_to })
}
}
impl serde::Serialize for Answer {
fn serialize<S: serde::Serializer>(&self, s: S) -> Result<S::Ok, S::Error> {
use serde::ser::SerializeTuple;
let mut t = s.serialize_tuple(2)?;
t.serialize_element(&self.text)?;
t.serialize_element(&self.pid)?;
t.end()
}
}
impl<'de> serde::Deserialize<'de> for Answer {
fn deserialize<D: serde::Deserializer<'de>>(d: D) -> Result<Self, D::Error> {
let (text, pid) = <(String, Option<RemotePid<Erased>>)>::deserialize(d)?;
Ok(Answer { text, pid })
}
}
// ================= local suite =========================================
/// A `Monitor` command handed to the connection but never processed (its
/// receiver dropped unread) reads as `Disconnected` — once. A second read
/// is the ordinary closed-channel `Err`, so "exactly one notice" holds.
#[test]
fn unread_command_reads_as_disconnected_once() {
maybe_child(ROLES);
run(|| {
remote::set_local_identity("me", Incarnation::new(7));
let (probe_tx, _probe_rx) = channel();
let inbox =
remote::bind_outbound_probe_with_monitors("peer", Incarnation::new(5), probe_tx);
let target = RemotePid::<Erased>::from_parts("peer", Incarnation::new(5), 9, 1);
let m = monitor_remote(target.clone());
assert!(
matches!(m.try_recv(), Ok(None)),
"command is in flight, no notice yet"
);
drop(inbox); // the conn actor died with the command unread
let d = m.recv().unwrap();
assert_eq!(d.pid, target);
assert_eq!(d.reason, RemoteDownReason::Disconnected);
assert!(
m.recv().is_err(),
"second read is closed, not a second notice"
);
assert!(m.try_recv().is_err());
});
}
/// After `demonitor_remote`, a closed channel is a closed channel: no
/// notice is synthesized for a monitor the caller cancelled.
#[test]
fn cancelled_monitor_never_synthesizes() {
maybe_child(ROLES);
run(|| {
remote::set_local_identity("me", Incarnation::new(7));
let (probe_tx, _probe_rx) = channel();
let inbox =
remote::bind_outbound_probe_with_monitors("peer", Incarnation::new(5), probe_tx);
let target = RemotePid::<Erased>::from_parts("peer", Incarnation::new(5), 9, 1);
let m = monitor_remote(target);
demonitor_remote(&m);
drop(inbox);
assert!(m.recv().is_err());
assert!(m.try_recv().is_err());
});
}
// ================= cross-process ======================================
const ROLES: &[(&str, fn())] = &[
("server", role_server),
("client", role_client),
("client_stop", role_client_stop),
];
const CTL: Name<Ctl> = Name::new("c13.ctl");
fn cfg(name: &str, seeds: Vec<(String, String)>) -> Config {
Config {
node_name: name.to_string(),
meta: NodeMeta {
role: "c13".into(),
region: "local".into(),
},
listen_addr: std::env::var("SMARM_LISTEN_ADDR").unwrap_or_else(|_| "127.0.0.1:0".into()),
strategy: Box::new(StaticSeeds::new(seeds)),
timing: timing(),
}
}
/// The p11 knobs make the liveness test fast: both roles of that test are
/// spawned with `SMARM_FAST_TIMING=1` and agree on a 100ms heartbeat /
/// 500ms liveness window. Everything else runs the shipping defaults.
fn timing() -> Timing {
if std::env::var_os("SMARM_FAST_TIMING").is_some() {
Timing {
heartbeat_interval: Duration::from_millis(100),
liveness_timeout: Duration::from_millis(500),
initial_backoff: Duration::from_millis(50),
max_backoff: Duration::from_millis(500),
..Timing::default()
}
} else {
Timing::default()
}
}
fn wait_up(events: &smarm::cluster::membership::MembershipEvents, who: &str) {
loop {
match events.rx.recv() {
Ok(NodeEvent::NodeUp(i)) if i.name == who => return,
Ok(_) => continue,
Err(_) => panic!("manager gone"),
}
}
}
fn disconnect(name: &str) {
assert!(matches!(
gen_server::call(
MANAGER,
Call::Disconnect {
name: name.to_string()
}
),
Ok(Reply::Disconnected)
));
}
/// Server: `spawn` ⇒ a parked worker (answer carries its pid);
/// `kill:<index>` releases it, whereupon it returns (Exit).
fn role_server() {
smarm::run(move || {
let cluster = start(cfg("server", vec![])).expect("binds");
println!("LISTENING {}", cluster.local_addr());
let (tx, rx) = channel::<Ctl>();
register(CTL, tx).unwrap();
expose(CTL);
println!("READY");
let mut workers: HashMap<u32, smarm::channel::Sender<()>> = HashMap::new();
loop {
let ctl = rx.recv().unwrap();
println!("CTL {}", ctl.cmd);
let (text, pid): (String, Option<RemotePid<Erased>>) = match ctl.cmd.as_str() {
"spawn" => {
let (go_tx, go_rx) = channel::<()>();
let p: Pid = spawn(move || {
let _ = go_rx.recv();
})
.pid();
workers.insert(p.index(), go_tx);
(
"ok".into(),
Some(RemotePid::from_local(p).expect("identity set")),
)
}
other => {
let idx: u32 = other.strip_prefix("kill:").unwrap().parse().unwrap();
if let Some(go) = workers.remove(&idx) {
let _ = go.send(());
}
("killed".into(), None)
}
};
send_to_remote(ctl.reply_to, Answer { text, pid }).unwrap();
}
});
}
/// Client-side setup shared by both client roles: join, expose the reply
/// path, hand back an `ask` closure and the membership stream.
fn client_setup() -> (
smarm::cluster::Cluster,
smarm::cluster::membership::MembershipEvents,
impl Fn(&str) -> Answer,
) {
let server_addr = std::env::var("SMARM_SERVER_ADDR").expect("SMARM_SERVER_ADDR");
let cluster = start(cfg("client", vec![("server".into(), server_addr)])).expect("binds");
let ev = subscribe().unwrap();
wait_up(&ev, "server");
let (tx, rx) = channel::<Answer>();
let me: Pid<Client> = install::<Client>(tx);
expose_type::<Answer>();
let ask = move |cmd: &str| -> Answer {
remote::send(
RemoteName::new("server", CTL),
Ctl {
cmd: cmd.into(),
reply_to: RemotePid::from_local(me).expect("identity set"),
},
)
.unwrap();
rx.recv().unwrap()
};
(cluster, ev, ask)
}
fn role_client() {
smarm::run(move || {
let (_cluster, ev, ask) = client_setup();
// 1. Headline: actor death ⇒ TRUE reason; link cut ⇒ Disconnected.
let a = ask("spawn").pid.unwrap();
let b = ask("spawn").pid.unwrap();
let ma = monitor_remote(a.clone());
let mb = monitor_remote(b.clone());
ask(&format!("kill:{}", a.index()));
let d = ma.recv().unwrap();
assert_eq!(d.pid, a);
println!("DOWN actor {:?}", d.reason);
disconnect("server");
let d = mb.recv().unwrap();
assert_eq!(d.pid, b);
println!("DOWN link {:?}", d.reason);
// 2. Reconnect does not resurrect. The connector redials on
// node_down; over the NEW link a fresh monitor delivers, while
// the old one (already answered) yields nothing further — order
// proves it, and `b` is even still alive on the server.
wait_up(&ev, "server");
println!("RECONNECTED");
let c = ask("spawn").pid.unwrap();
let mc = monitor_remote(c.clone());
ask(&format!("kill:{}", b.index()));
ask(&format!("kill:{}", c.index()));
assert_eq!(mc.recv().unwrap().reason, DownReason::Exit.into());
let stray = matches!(mb.try_recv(), Ok(Some(_)));
println!("RESURRECT stray={stray}");
// 3. Peer PROCESS killed ⇒ Disconnected too (nobody is left to send
// Down): the parent SIGKILLs the server once it sees the marker.
let e = ask("spawn").pid.unwrap();
let me_ = monitor_remote(e.clone());
println!("KILL SERVER NOW");
let d = me_.recv().unwrap();
assert_eq!(d.pid, e);
println!("DOWN procdeath {:?}", d.reason);
println!("CLIENT DONE");
loop {
smarm::sleep(Duration::from_secs(3600));
}
});
}
/// The slow role: liveness expiry (peer SIGSTOPped) followed by the link
/// dropping for real (peer SIGKILLed) — one monitor, exactly one notice.
fn role_client_stop() {
smarm::run(move || {
let (_cluster, _ev, ask) = client_setup();
let a = ask("spawn").pid.unwrap();
let ma = monitor_remote(a.clone());
println!("STOP SERVER NOW");
let d = ma.recv().unwrap(); // liveness expiry, ~liveness_timeout
assert_eq!(d.pid, a);
println!("DOWN stopped {:?}", d.reason);
println!("KILL SERVER NOW");
// Give the drop every chance to produce a second notice, then look.
smarm::sleep(Duration::from_secs(1));
let dup = matches!(ma.try_recv(), Ok(Some(_)));
println!("DUPLICATE dup={dup}");
println!("CLIENT DONE");
loop {
smarm::sleep(Duration::from_secs(3600));
}
});
}
/// The Phase 4 c13 gate: partition vs. death distinguishable; nothing
/// survives reconnect; a dead peer process is a Disconnected too.
#[test]
fn link_loss_is_disconnected_and_does_not_survive_reconnect() {
maybe_child(ROLES);
let mut server = spawn_node("server", &[]);
let saddr = server.wait_listening();
server.wait_line("READY", |l| l == "READY");
let mut client = spawn_node("client", &[("SMARM_SERVER_ADDR", &saddr)]);
client.wait_line("DOWN actor Local(Exit)", |l| l == "DOWN actor Local(Exit)");
client.wait_line("DOWN link Disconnected", |l| l == "DOWN link Disconnected");
client.wait_line("RECONNECTED", |l| l == "RECONNECTED");
client.wait_line("RESURRECT stray=false", |l| l == "RESURRECT stray=false");
client.wait_line("KILL SERVER NOW", |l| l == "KILL SERVER NOW");
server.kill();
client.wait_line("DOWN procdeath Disconnected", |l| {
l == "DOWN procdeath Disconnected"
});
client.wait_line("CLIENT DONE", |l| l == "CLIENT DONE");
}
/// Covers the invariant the headline test cannot: liveness expiry and the
/// transport drop both firing for the same connection yield ONE notice.
/// Runs on the fast [`timing`] (both roles) — was `#[ignore]`d at the 4s
/// default until the p11 knobs landed.
#[test]
fn timeout_then_drop_yields_one_notice() {
maybe_child(ROLES);
let fast = ("SMARM_FAST_TIMING", "1");
let mut server = spawn_node("server", &[fast]);
let saddr = server.wait_listening();
server.wait_line("READY", |l| l == "READY");
let mut client = spawn_node("client_stop", &[("SMARM_SERVER_ADDR", &saddr), fast]);
client.wait_line("STOP SERVER NOW", |l| l == "STOP SERVER NOW");
let spid = server.pid().expect("server alive") as libc::pid_t;
assert_eq!(unsafe { libc::kill(spid, libc::SIGSTOP) }, 0);
client.wait_line("DOWN stopped Disconnected", |l| {
l == "DOWN stopped Disconnected"
});
client.wait_line("KILL SERVER NOW", |l| l == "KILL SERVER NOW");
server.kill(); // SIGKILL works on a stopped process; Drop would too
client.wait_line("DUPLICATE dup=false", |l| l == "DUPLICATE dup=false");
client.wait_line("CLIENT DONE", |l| l == "CLIENT DONE");
}
+161
View File
@@ -0,0 +1,161 @@
//! RFC 010 — `Discovery::Withdrawn`: a strategy retracts a candidate and the
//! connector stops dialing it.
//!
//! Cross-process: a plain *server* node, and a *client* whose strategy is a
//! script: announce a decoy `(ghost, addr)` where `addr` is a raw
//! `TcpListener` the client itself holds (an OS thread accepts and
//! immediately closes, so every dial fails at handshake and the connector
//! keeps retrying on backoff — the accept count is the dial count); after a
//! beat, withdraw the decoy and announce the real server. The client waits
//! for the server's `node_up` — which is *after* the withdrawal in the
//! strategy's own stream — then watches the decoy's accept count stay flat
//! across a window longer than the pending backoff. Before withdrawal it
//! must have been climbing (≥ 1), or the negative proves nothing.
#![cfg(feature = "cluster")]
mod common;
use common::{maybe_child, spawn_node};
use smarm::channel::Sender;
use smarm::cluster::discovery::{Discovery, Strategy};
use smarm::cluster::envelope::NodeMeta;
use smarm::cluster::membership::{subscribe, NodeEvent};
use smarm::cluster::{start, Config, StaticSeeds, Timing};
use std::net::TcpListener;
use std::sync::atomic::{AtomicUsize, Ordering};
use std::sync::Arc;
use std::time::{Duration, Instant};
const ROLES: &[(&str, fn())] = &[("server", role_server), ("client", role_client)];
fn meta() -> NodeMeta {
NodeMeta {
role: "withdraw".into(),
region: "local".into(),
}
}
fn role_server() {
smarm::run(|| {
let cluster = start(Config {
node_name: "server".into(),
meta: meta(),
listen_addr: std::env::var("SMARM_LISTEN_ADDR")
.unwrap_or_else(|_| "127.0.0.1:0".into()),
strategy: Box::new(StaticSeeds::new(Vec::<(String, String)>::new())),
timing: Timing::default(),
})
.expect("binds");
println!("LISTENING {}", cluster.local_addr());
loop {
smarm::sleep(Duration::from_secs(3600));
}
});
}
/// Scripted strategy: decoy, pause, withdraw decoy, real server, done.
struct Script {
decoy: String,
server: String,
}
impl Strategy for Script {
fn run(self: Box<Self>, out: Sender<Discovery>) {
let _ = out.send(Discovery::Candidate {
name: "ghost".into(),
addr: self.decoy.clone(),
});
// Long enough for the 250ms/500ms retries to land: ≥ 3 dials.
smarm::sleep(Duration::from_millis(1100));
let _ = out.send(Discovery::Withdrawn {
name: "ghost".into(),
addr: self.decoy,
});
let _ = out.send(Discovery::Candidate {
name: "server".into(),
addr: self.server,
});
}
}
fn role_client() {
let server_addr = std::env::var("SMARM_SERVER_ADDR").expect("SMARM_SERVER_ADDR");
// The decoy: accept-and-close on an OS thread; count every accept.
let decoy = TcpListener::bind("127.0.0.1:0").unwrap();
let decoy_addr = decoy.local_addr().unwrap().to_string();
let dials = Arc::new(AtomicUsize::new(0));
let counter = dials.clone();
std::thread::spawn(move || {
for conn in decoy.incoming() {
counter.fetch_add(1, Ordering::SeqCst);
drop(conn);
}
});
smarm::run(move || {
let _cluster = start(Config {
node_name: "client".into(),
meta: meta(),
listen_addr: "127.0.0.1:0".into(),
strategy: Box::new(Script {
decoy: decoy_addr,
server: server_addr,
}),
timing: Timing::default(),
})
.expect("binds");
let ev = subscribe().unwrap();
loop {
match ev.rx.recv() {
Ok(NodeEvent::NodeUp(i)) if i.name == "server" => break,
Ok(_) => continue,
Err(_) => panic!("manager gone"),
}
}
// The withdrawal preceded the server candidate in the strategy's
// stream, so it has been applied. Any dial that started before it
// is bounded by the connect+handshake deadlines; let it drain, then
// hold the count flat across a window longer than the pending
// backoff would be (1s at this point, 2s next).
let before = dials.load(Ordering::SeqCst);
smarm::sleep(Duration::from_millis(500));
let settled = dials.load(Ordering::SeqCst);
let t0 = Instant::now();
while t0.elapsed() < Duration::from_millis(3000) {
smarm::sleep(Duration::from_millis(100));
}
let after = dials.load(Ordering::SeqCst);
println!("WITHDRAWN before={before} settled={settled} after={after}");
println!("CLIENT DONE");
loop {
smarm::sleep(Duration::from_secs(3600));
}
});
}
#[test]
fn withdrawn_candidate_is_no_longer_dialed() {
maybe_child(ROLES);
let mut server = spawn_node("server", &[]);
let saddr = server.wait_listening();
let mut client = spawn_node("client", &[("SMARM_SERVER_ADDR", &saddr)]);
let line = client.wait_line("WITHDRAWN", |l| l.starts_with("WITHDRAWN "));
let mut nums = line
.split_whitespace()
.skip(1)
.map(|kv| kv.split_once('=').unwrap().1.parse::<usize>().unwrap());
let (before, settled, after) = (
nums.next().unwrap(),
nums.next().unwrap(),
nums.next().unwrap(),
);
assert!(
before >= 1,
"decoy was never dialed; the negative proves nothing: {line}"
);
assert_eq!(
settled, after,
"connector kept dialing a withdrawn candidate: {line}"
);
client.wait_line("CLIENT DONE", |l| l == "CLIENT DONE");
}
+276
View File
@@ -0,0 +1,276 @@
//! RFC 010 c2 — owned envelope tests (roadmap: per-frame roundtrip,
//! truncation mid-field, unknown tag, length prefix lying long and short,
//! zero-length payload, adversarial lengths).
#![cfg(feature = "cluster")]
use serde::{Deserialize, Serialize};
use smarm::cluster::envelope::{
decode_payload, encode_payload, DecodeError, Frame, NodeMeta, RejectReason, MAX_FRAME_LEN,
PROTO_VERSION,
};
use smarm::cluster::RemoteDownReason;
use smarm::monitor::DownReason;
use smarm::pg::Incarnation;
fn meta() -> NodeMeta {
NodeMeta {
role: "worker".into(),
region: "eu-west".into(),
}
}
fn all_frames() -> Vec<Frame> {
vec![
Frame::Hello {
proto_version: PROTO_VERSION,
build_hash: 0xDEAD_BEEF_CAFE_F00D,
node_name: "alpha".into(),
incarnation: Incarnation::new(7),
meta: meta(),
},
Frame::HelloAck {
node_name: "beta".into(),
incarnation: Incarnation::new(9),
meta: meta(),
},
Frame::HelloReject {
reason: RejectReason::NameTaken,
},
Frame::Heartbeat,
Frame::Send {
index: 42,
generation: 3,
type_hash: 0x1234_5678_9ABC_DEF0,
payload: vec![1, 2, 3, 4, 5],
},
Frame::SendNamed {
name: "the_counter".into(),
type_hash: 0xFFFF_0000_FFFF_0000,
payload: vec![],
},
Frame::Monitor {
monitor_id: 77,
index: 42,
generation: 3,
},
Frame::Demonitor { monitor_id: 77 },
Frame::Down {
monitor_id: 77,
reason: RemoteDownReason::Local(DownReason::Panic),
},
Frame::Down {
monitor_id: 78,
reason: RemoteDownReason::Disconnected,
},
]
}
fn encode_one(f: &Frame) -> Vec<u8> {
let mut buf = Vec::new();
f.encode(&mut buf).unwrap();
buf
}
#[test]
fn per_frame_roundtrip() {
for f in all_frames() {
let buf = encode_one(&f);
let (decoded, consumed) = Frame::decode(&buf).unwrap().unwrap();
assert_eq!(decoded, f, "roundtrip mismatch");
assert_eq!(consumed, buf.len(), "consumed != buffer length for {f:?}");
}
}
#[test]
fn back_to_back_frames_decode_sequentially() {
let mut buf = Vec::new();
for f in all_frames() {
f.encode(&mut buf).unwrap();
}
let mut off = 0;
let mut decoded = Vec::new();
while off < buf.len() {
let (f, n) = Frame::decode(&buf[off..]).unwrap().unwrap();
decoded.push(f);
off += n;
}
assert_eq!(decoded, all_frames());
assert_eq!(off, buf.len());
}
#[test]
fn heartbeat_golden_bytes() {
// Locks the layout: u32 LE length prefix, then the tag byte.
let buf = encode_one(&Frame::Heartbeat);
assert_eq!(buf, vec![1, 0, 0, 0, 4]);
}
#[test]
fn zero_length_payload_roundtrips() {
let f = Frame::Send {
index: 0,
generation: 0,
type_hash: 0,
payload: vec![],
};
let buf = encode_one(&f);
let (decoded, consumed) = Frame::decode(&buf).unwrap().unwrap();
assert_eq!(decoded, f);
assert_eq!(consumed, buf.len());
}
#[test]
fn incomplete_is_none_not_error() {
let buf = encode_one(&all_frames()[0]);
// Every strict prefix short of the full frame must report "need more".
for cut in 0..buf.len() {
assert_eq!(
Frame::decode(&buf[..cut]).unwrap(),
None,
"cut at {cut} should be incomplete"
);
}
}
#[test]
fn unknown_frame_tag() {
let buf = vec![1, 0, 0, 0, 250];
assert_eq!(Frame::decode(&buf), Err(DecodeError::UnknownTag(250)));
}
#[test]
fn unknown_enum_tags() {
// HelloReject with a bogus reason tag.
let buf = vec![2, 0, 0, 0, 3, 99];
assert_eq!(
Frame::decode(&buf),
Err(DecodeError::UnknownEnumTag {
what: "RejectReason",
tag: 99
})
);
// Down with a bogus reason tag (id = 0u64).
let mut buf = vec![10, 0, 0, 0, 9];
buf.extend_from_slice(&0u64.to_le_bytes());
buf.push(200);
assert_eq!(
Frame::decode(&buf),
Err(DecodeError::UnknownEnumTag {
what: "RemoteDownReason",
tag: 200
})
);
}
#[test]
fn length_prefix_lying_long_with_bytes_present_is_trailing() {
let mut buf = encode_one(&Frame::Heartbeat);
// Declare 3 extra body bytes and actually supply them.
let declared = u32::from_le_bytes([buf[0], buf[1], buf[2], buf[3]]) + 3;
buf[0..4].copy_from_slice(&declared.to_le_bytes());
buf.extend_from_slice(&[0xAA, 0xBB, 0xCC]);
assert_eq!(Frame::decode(&buf), Err(DecodeError::Trailing { extra: 3 }));
}
#[test]
fn length_prefix_lying_long_without_bytes_is_incomplete() {
// Indistinguishable from a partial read — must be None, not an error.
let mut buf = encode_one(&Frame::Heartbeat);
let declared = u32::from_le_bytes([buf[0], buf[1], buf[2], buf[3]]) + 3;
buf[0..4].copy_from_slice(&declared.to_le_bytes());
assert_eq!(Frame::decode(&buf).unwrap(), None);
}
#[test]
fn length_prefix_lying_short_truncates_a_field() {
let f = &all_frames()[0]; // Hello: plenty of fields to cut into
let mut buf = encode_one(f);
let declared = u32::from_le_bytes([buf[0], buf[1], buf[2], buf[3]]);
let lie = declared - 4; // cut mid-field
buf[0..4].copy_from_slice(&lie.to_le_bytes());
assert_eq!(Frame::decode(&buf), Err(DecodeError::Truncated));
}
#[test]
fn truncation_mid_string_field() {
// A frame whose declared length is intact but whose inner string length
// runs past the body: SendNamed claiming a 1000-byte name in a tiny body.
let mut body = vec![6u8]; // TAG_SEND_NAMED
body.extend_from_slice(&1000u16.to_le_bytes());
body.extend_from_slice(b"short");
let mut buf = (body.len() as u32).to_le_bytes().to_vec();
buf.extend_from_slice(&body);
assert_eq!(Frame::decode(&buf), Err(DecodeError::Truncated));
}
#[test]
fn adversarial_lengths() {
// Length prefix of u32::MAX: reject as oversized, do not wait for 4 GiB.
let buf = [0xFF, 0xFF, 0xFF, 0xFF, 0];
assert_eq!(
Frame::decode(&buf),
Err(DecodeError::FrameTooLarge {
declared: u32::MAX as usize
})
);
// Just over the cap: also rejected.
let over = (MAX_FRAME_LEN as u32 + 1).to_le_bytes();
assert!(matches!(
Frame::decode(&over),
Err(DecodeError::FrameTooLarge { .. })
));
// Zero-length frame: there is no tag byte; corrupt, not incomplete.
let buf = [0, 0, 0, 0];
assert_eq!(Frame::decode(&buf), Err(DecodeError::EmptyFrame));
}
#[test]
fn invalid_utf8_in_string_field() {
let mut buf = encode_one(&Frame::SendNamed {
name: "abcd".into(),
type_hash: 0,
payload: vec![],
});
// name bytes start after: 4 (len) + 1 (tag) + 2 (str len) = offset 7
buf[7] = 0xFF;
assert_eq!(Frame::decode(&buf), Err(DecodeError::Utf8));
}
#[derive(Debug, PartialEq, Serialize, Deserialize)]
struct Ping {
seq: u64,
label: String,
}
#[test]
fn payload_seam_roundtrip() {
let ping = Ping {
seq: 31337,
label: "hello".into(),
};
let blob = encode_payload(&ping).unwrap();
// Carry it through a real frame, as it will travel in c9.
let f = Frame::Send {
index: 1,
generation: 1,
type_hash: 0xABCD,
payload: blob,
};
let buf = encode_one(&f);
let (decoded, _) = Frame::decode(&buf).unwrap().unwrap();
let Frame::Send { payload, .. } = decoded else {
panic!("wrong frame");
};
let back: Ping = decode_payload(&payload).unwrap();
assert_eq!(back, ping);
}
#[test]
fn payload_seam_rejects_truncated_blob() {
let blob = encode_payload(&Ping {
seq: 1,
label: "x".into(),
})
.unwrap();
assert!(decode_payload::<Ping>(&blob[..blob.len() - 1]).is_err());
}
+161
View File
@@ -0,0 +1,161 @@
//! RFC 010 c8 — exposure registry + type hashing. Purely local, no network.
//!
//! Payload types are std types (`String`, `u64`) because the crate's serde is
//! deliberately derive-less (`default-features = false`) — user crates bring
//! their own derive; the contract here is `DeserializeOwned`.
//!
//! The hash-stability test re-execs the current binary (the c4 harness): the
//! guarantee under test is "stable across runs in the SAME binary" — exactly
//! what the build-hash handshake reduces the mesh to — not stability across
//! builds, which the scope guard explicitly rejects.
#![cfg(feature = "cluster")]
mod common;
use common::{maybe_child, spawn_node};
use smarm::cluster::envelope::encode_payload;
use smarm::cluster::expose::{
decode_deliver, decoder_registered, expose, expose_type, exposed_hash, exposed_names,
type_hash, DeliverError,
};
use smarm::monitor::{monitor, terminal_reason, DownReason};
use smarm::{channel, register, run, spawn, Name};
const ROLES: &[(&str, fn())] = &[("hasher", role_hasher)];
/// Print the hashes this process computes; the parent (a different run of
/// the same binary) compares against its own.
fn role_hasher() {
println!("HASH-STRING {}", type_hash::<String>());
println!("HASH-U64 {}", type_hash::<u64>());
}
const GREETER: Name<String> = Name::new("expose-test.greeter");
/// Exposed and unexposed lookup, the returned hash, and the audit listing.
#[test]
fn exposed_and_unexposed_lookup() {
maybe_child(ROLES);
run(|| {
let h = expose(GREETER);
assert_eq!(h, type_hash::<String>());
assert_eq!(exposed_hash("expose-test.greeter"), Some(h));
assert_eq!(exposed_hash("never-exposed"), None);
assert!(exposed_names().contains(&("expose-test.greeter", h)));
});
}
/// Distinct types land on distinct hashes (FNV over distinct TypeIds — a
/// smoke assertion; a collision would degrade to a decode error, never a
/// misroute, per RFC §3).
#[test]
fn distinct_types_distinct_hashes() {
maybe_child(ROLES);
run(|| {
assert_ne!(type_hash::<String>(), type_hash::<u64>());
assert_ne!(type_hash::<String>(), type_hash::<Vec<u8>>());
});
}
/// The decode-and-deliver contract: a registered hash decodes into the
/// target's typed channel; an unknown hash, corrupt bytes, and a missing
/// channel each fail without delivering — `WrongChannel`, never a misroute.
#[test]
fn decoder_registration_and_delivery() {
maybe_child(ROLES);
run(|| {
let h_string = expose_type::<String>();
let h_u64 = expose_type::<u64>();
assert!(decoder_registered(h_string));
assert!(!decoder_registered(h_string.wrapping_add(1)));
// A live actor with a String channel (registered from its own body,
// announced via a ready signal — the tests/registry.rs idiom).
let (ready_tx, ready_rx) = channel::<()>();
let (stop_tx, stop_rx) = channel::<()>();
let (msg_tx, msg_rx) = channel::<String>();
let pid = spawn(move || {
register(Name::<String>::new("expose-test.sink"), msg_tx).unwrap();
ready_tx.send(()).unwrap();
let _ = stop_rx.recv();
})
.pid();
ready_rx.recv().unwrap();
// Happy path: decode + deliver through the published channel.
let bytes = encode_payload("hello across the seam").unwrap();
decode_deliver(h_string, pid, &bytes).unwrap();
assert_eq!(msg_rx.recv().unwrap(), "hello across the seam");
// Unknown hash: nothing was registered under it.
assert!(matches!(
decode_deliver(h_string.wrapping_add(1), pid, &bytes),
Err(DeliverError::UnknownType)
));
// Corrupt bytes: the decoder fails before any send.
assert!(matches!(
decode_deliver(h_string, pid, &[0xff; 3]),
Err(DeliverError::Decode(_))
));
// Right decoder, wrong channel: the actor has no u64 channel, so the
// decoded value is refused — the NoChannel guarantee.
let u64_bytes = encode_payload(&7u64).unwrap();
assert!(matches!(
decode_deliver(h_u64, pid, &u64_bytes),
Err(DeliverError::WrongChannel)
));
stop_tx.send(()).unwrap();
});
}
/// `expose` and the bridge crossing agree on the resulting set: both funnel
/// the pid-boundary mark through the watchable machinery, so an exposed
/// name's holder dies with a terminal record — the exact observable
/// `mark_watchable` guarantees the membrane. (For named holders the mark is
/// already stamped by `register` itself; this pins the shared contract.)
#[test]
fn expose_and_bridge_crossing_agree_on_the_set() {
maybe_child(ROLES);
run(|| {
let (ready_tx, ready_rx) = channel::<()>();
let (stop_tx, stop_rx) = channel::<()>();
let (msg_tx, _msg_rx) = channel::<String>();
let pid = spawn(move || {
register(GREETER, msg_tx).unwrap();
ready_tx.send(()).unwrap();
let _ = stop_rx.recv();
})
.pid();
ready_rx.recv().unwrap();
expose(GREETER);
let m = monitor(pid);
stop_tx.send(()).unwrap();
assert_eq!(m.rx.recv().unwrap().reason, DownReason::Exit);
assert_eq!(terminal_reason(pid), Some(DownReason::Exit));
});
}
/// Hash stability across runs in the same binary: a re-exec of this binary
/// computes the same hashes this process does.
#[test]
fn hash_stable_across_runs_in_same_binary() {
maybe_child(ROLES);
let (mine_string, mine_u64) = {
// Computing a TypeId hash needs no runtime, but keep the contract
// uniform with real call sites.
(type_hash::<String>(), type_hash::<u64>())
};
let mut child = spawn_node("hasher", &[]);
let line = child.wait_line("HASH-STRING", |l| l.starts_with("HASH-STRING "));
assert_eq!(
line["HASH-STRING ".len()..].parse::<u64>().unwrap(),
mine_string
);
let line = child.wait_line("HASH-U64", |l| l.starts_with("HASH-U64 "));
assert_eq!(line["HASH-U64 ".len()..].parse::<u64>().unwrap(), mine_u64);
child.wait_exit();
}
+240
View File
@@ -0,0 +1,240 @@
//! RFC 010 c5 — handshake state-machine tests (roadmap: happy path; hash
//! mismatch; proto-version mismatch; name already claimed; simultaneous-connect
//! tie-break; garbage before Hello). Pure — no IO, no actors, no runtime.
#![cfg(feature = "cluster")]
use smarm::cluster::envelope::{Frame, NodeMeta, RejectReason, PROTO_VERSION};
use smarm::cluster::handshake::{
dial_wins, Initiator, InitiatorOutcome, Local, PeerStanding, Responder, ResponderOutcome,
};
use smarm::pg::Incarnation;
const HASH: u64 = 0xDEAD_BEEF_CAFE_F00D;
fn local(name: &str) -> Local {
Local {
node_name: name.into(),
incarnation: Incarnation::new(7),
build_hash: HASH,
meta: NodeMeta {
role: "worker".into(),
region: "eu-west".into(),
},
}
}
/// The Hello that `Initiator::new(&local(name))` emits, built by hand.
fn hello_from(name: &str) -> Frame {
let l = local(name);
Frame::Hello {
proto_version: PROTO_VERSION,
build_hash: l.build_hash,
node_name: l.node_name,
incarnation: l.incarnation,
meta: l.meta,
}
}
#[test]
fn happy_path_establishes_both_ends() {
// alpha dials beta.
let (initiator, hello) = Initiator::new(&local("alpha"));
assert_eq!(hello, hello_from("alpha"), "initiator emits its identity");
let responder = Responder::new(local("beta"));
let (reply, peer) = match responder.on_frame(hello, PeerStanding::Free) {
ResponderOutcome::Accepted { reply, peer } => (reply, peer),
other => panic!("expected Accepted, got {other:?}"),
};
assert_eq!(peer.node_name, "alpha");
assert_eq!(peer.incarnation, Incarnation::new(7));
assert_eq!(peer.meta.role, "worker");
// The ack carries the responder's identity, no hash/version (one-sided
// check — sound because equality is symmetric).
let l = local("beta");
assert_eq!(
reply,
Frame::HelloAck {
node_name: l.node_name,
incarnation: l.incarnation,
meta: l.meta,
}
);
match initiator.on_frame(reply) {
InitiatorOutcome::Established(peer) => {
assert_eq!(peer.node_name, "beta");
assert_eq!(peer.incarnation, Incarnation::new(7));
assert_eq!(peer.meta.region, "eu-west");
}
other => panic!("expected Established, got {other:?}"),
}
}
#[test]
fn hash_mismatch_rejected() {
let responder = Responder::new(local("beta"));
let hello = Frame::Hello {
proto_version: PROTO_VERSION,
build_hash: HASH ^ 1,
node_name: "alpha".into(),
incarnation: Incarnation::new(7),
meta: local("alpha").meta,
};
match responder.on_frame(hello, PeerStanding::Free) {
ResponderOutcome::Rejected { reply, reason } => {
assert_eq!(reason, RejectReason::HashMismatch);
assert_eq!(reply, Frame::HelloReject { reason });
}
other => panic!("expected Rejected, got {other:?}"),
}
// The dialer side of the same story: a reject frame comes back.
let (initiator, _hello) = Initiator::new(&local("alpha"));
match initiator.on_frame(Frame::HelloReject {
reason: RejectReason::HashMismatch,
}) {
InitiatorOutcome::Rejected(RejectReason::HashMismatch) => {}
other => panic!("expected Rejected(HashMismatch), got {other:?}"),
}
}
#[test]
fn proto_version_mismatch_rejected_and_checked_first() {
// Both proto and hash wrong: proto wins — nothing after the version can
// be trusted, and HelloReject is the cross-version compatibility anchor.
let responder = Responder::new(local("beta"));
let hello = Frame::Hello {
proto_version: PROTO_VERSION + 1,
build_hash: HASH ^ 1,
node_name: "alpha".into(),
incarnation: Incarnation::new(7),
meta: local("alpha").meta,
};
match responder.on_frame(hello, PeerStanding::Free) {
ResponderOutcome::Rejected { reason, .. } => {
assert_eq!(reason, RejectReason::ProtoVersion);
}
other => panic!("expected Rejected, got {other:?}"),
}
}
#[test]
fn claimed_name_rejected() {
let responder = Responder::new(local("beta"));
let ctx = PeerStanding::Claimed;
match responder.on_frame(hello_from("alpha"), ctx) {
ResponderOutcome::Rejected { reason, .. } => {
assert_eq!(reason, RejectReason::NameTaken);
}
other => panic!("expected Rejected, got {other:?}"),
}
}
#[test]
fn own_name_offered_rejected_as_name_taken() {
// Self-connect or genuine collision: the responder's own name arrives.
let responder = Responder::new(local("beta"));
match responder.on_frame(hello_from("beta"), PeerStanding::Free) {
ResponderOutcome::Rejected { reason, .. } => {
assert_eq!(reason, RejectReason::NameTaken);
}
other => panic!("expected Rejected, got {other:?}"),
}
}
#[test]
fn hash_checked_before_name() {
// Wrong hash AND claimed name: hash wins (validity before identity).
let responder = Responder::new(local("beta"));
let hello = Frame::Hello {
proto_version: PROTO_VERSION,
build_hash: HASH ^ 1,
node_name: "alpha".into(),
incarnation: Incarnation::new(7),
meta: local("alpha").meta,
};
let ctx = PeerStanding::Claimed;
match responder.on_frame(hello, ctx) {
ResponderOutcome::Rejected { reason, .. } => {
assert_eq!(reason, RejectReason::HashMismatch);
}
other => panic!("expected Rejected, got {other:?}"),
}
}
#[test]
fn dial_wins_is_deterministic_and_antisymmetric() {
// The smaller name's dial survives; both ends compute the same verdict.
assert!(dial_wins("alpha", "beta"));
assert!(!dial_wins("beta", "alpha"));
for (a, b) in [("a", "b"), ("node-1", "node-2"), ("x", "xx")] {
assert_ne!(dial_wins(a, b), dial_wins(b, a), "({a}, {b})");
}
}
#[test]
fn simultaneous_connect_exactly_one_side_accepts() {
// alpha and beta dial each other at once. Each responder sees the peer's
// Hello while its own dial is in flight.
let ctx = PeerStanding::Dialing;
// On beta: inbound is alpha's dial; alpha < beta, so the inbound wins.
let on_beta = Responder::new(local("beta")).on_frame(hello_from("alpha"), ctx);
assert!(
matches!(on_beta, ResponderOutcome::Accepted { .. }),
"beta must accept alpha's dial, got {on_beta:?}"
);
// On alpha: inbound is beta's dial; it loses — close silently, no frame
// (ratified: both ends can compute the outcome, a reject adds nothing).
let on_alpha = Responder::new(local("alpha")).on_frame(hello_from("beta"), ctx);
assert!(
matches!(on_alpha, ResponderOutcome::TieBreakLoss),
"alpha must silently drop beta's dial, got {on_alpha:?}"
);
}
#[test]
fn tiebreak_loss_only_applies_when_dialing() {
// Same inbound Hello, no dial in flight: plain accept.
let on_alpha = Responder::new(local("alpha")).on_frame(hello_from("beta"), PeerStanding::Free);
assert!(matches!(on_alpha, ResponderOutcome::Accepted { .. }));
}
#[test]
fn garbage_before_hello_fails_without_reply() {
// Any valid-but-wrong frame before Hello is a protocol violation: close,
// no reject frame. (Undecodable bytes are the codec's Err, not ours.)
for frame in [
Frame::Heartbeat,
Frame::HelloAck {
node_name: "alpha".into(),
incarnation: Incarnation::new(7),
meta: local("alpha").meta,
},
Frame::Demonitor { monitor_id: 3 },
] {
let out = Responder::new(local("beta")).on_frame(frame.clone(), PeerStanding::Free);
match out {
ResponderOutcome::Failed(f) => assert_eq!(f, frame),
other => panic!("expected Failed({frame:?}), got {other:?}"),
}
}
}
#[test]
fn garbage_before_ack_fails_the_initiator() {
for frame in [
Frame::Heartbeat,
hello_from("beta"),
Frame::Demonitor { monitor_id: 3 },
] {
let (initiator, _hello) = Initiator::new(&local("alpha"));
match initiator.on_frame(frame.clone()) {
InitiatorOutcome::Failed(f) => assert_eq!(f, frame),
other => panic!("expected Failed({frame:?}), got {other:?}"),
}
}
}
+270
View File
@@ -0,0 +1,270 @@
//! RFC 010 c7a — membership events and the view, at the manager.
//!
//! Same construction as the c6a lifecycle suite: the handshake is bypassed,
//! connections are built already-established over localhost TCP pairs with
//! fabricated `Peer`s, and the manager is started plainly so the test can
//! terminate. What is under test is the membership layer that c7 adds to the
//! manager: `node_up`/`node_down` events to subscribers (snapshot-then-stream),
//! the view, and NodeId identity — memoized per `(name, incarnation)`, so a
//! reconnect blip keeps its id and a restart (new incarnation) gets a fresh one.
#![cfg(feature = "cluster")]
use std::time::Duration;
use smarm::cluster::envelope::NodeMeta;
use smarm::cluster::handshake::Peer;
use smarm::cluster::manager::{Call, Manager, Reply, MANAGER};
use smarm::cluster::membership::{subscribe, view, MembershipEvents, NodeEvent};
use smarm::cluster::spawn_established;
use smarm::cluster::transport::tcp::TcpTransport;
use smarm::cluster::transport::{Conn, FramedConn, Transport};
use smarm::cluster::Timing;
use smarm::gen_server::{self, GenServerBuilder};
use smarm::pg::{Incarnation, NodeId};
use smarm::run;
/// A fabricated post-handshake peer identity, with the incarnation under the
/// test's control (it is identity-bearing here, unlike in the c6a suite).
fn peer(name: &str, inc: u32) -> Peer {
Peer {
node_name: name.to_string(),
incarnation: Incarnation::new(inc),
meta: NodeMeta {
role: "test".to_string(),
region: "test".to_string(),
},
}
}
/// One established transport pair over localhost (TCP backlog covers the
/// sequential dial-then-accept, as in the c3 conformance suite).
fn pair(t: &dyn Transport) -> (Box<dyn Conn>, Box<dyn Conn>) {
let mut l = t.listen("127.0.0.1:0").unwrap();
let a = t.dial(&l.local_addr()).unwrap();
let b = l.accept().unwrap();
(a, b)
}
/// The next event, or a panic naming the wait. The bound is generous against
/// a sub-millisecond real cost.
fn next_event(ev: &MembershipEvents, waiting_for: &str) -> NodeEvent {
ev.rx
.recv_timeout(Duration::from_secs(5))
.unwrap_or_else(|e| panic!("timed out waiting for {waiting_for}: {e:?}"))
}
/// Assert the subscription is drained: no event is pending.
fn assert_quiet(ev: &MembershipEvents) {
assert!(matches!(ev.rx.try_recv(), Ok(None)));
}
fn disconnect(name: &str) {
assert!(matches!(
gen_server::call(
MANAGER,
Call::Disconnect {
name: name.to_string()
}
),
Ok(Reply::Disconnected)
));
}
/// Live subscription: an empty snapshot, then `NodeUp` on registration and
/// `NodeDown` (same id) on commanded disconnect and on peer EOF alike.
#[test]
fn subscriber_sees_up_and_down() {
run(|| {
let mgr = GenServerBuilder::new(Manager::new())
.named(MANAGER)
.start()
.expect("manager name is free");
let ev = subscribe().expect("manager is up");
assert_quiet(&ev); // nothing live: the snapshot is empty
let t = TcpTransport;
let (a1, b1) = pair(&t);
let (a2, b2) = pair(&t);
spawn_established(FramedConn::new(a1), peer("node-b", 1), Timing::default())
.expect("node-b registers");
spawn_established(FramedConn::new(a2), peer("node-c", 1), Timing::default())
.expect("node-c registers");
let up_b = match next_event(&ev, "node_up(node-b)") {
NodeEvent::NodeUp(info) => {
assert_eq!(info.name, "node-b");
assert_eq!(info.incarnation, Incarnation::new(1));
assert_eq!(info.meta.role, "test");
info
}
other => panic!("expected node_up(node-b), got {other:?}"),
};
let up_c = match next_event(&ev, "node_up(node-c)") {
NodeEvent::NodeUp(info) => {
assert_eq!(info.name, "node-c");
info
}
other => panic!("expected node_up(node-c), got {other:?}"),
};
assert_ne!(up_b.node, up_c.node, "distinct peers get distinct ids");
// Commanded disconnect: down with node-b's id.
disconnect("node-b");
assert_eq!(
next_event(&ev, "node_down(node-b)"),
NodeEvent::NodeDown(up_b.clone())
);
// Peer EOF, no command: down with node-c's id.
drop(b2);
assert_eq!(
next_event(&ev, "node_down(node-c)"),
NodeEvent::NodeDown(up_c.clone())
);
assert_quiet(&ev);
drop(b1);
mgr.shutdown();
});
}
/// Snapshot-then-stream: a subscriber arriving after connections established
/// receives one `NodeUp` per live peer before anything else, and the view
/// call agrees with it.
#[test]
fn late_subscriber_gets_snapshot_and_view_agrees() {
run(|| {
let mgr = GenServerBuilder::new(Manager::new())
.named(MANAGER)
.start()
.expect("manager name is free");
let t = TcpTransport;
let (a1, b1) = pair(&t);
let (a2, b2) = pair(&t);
spawn_established(FramedConn::new(a1), peer("node-b", 1), Timing::default())
.expect("node-b registers");
spawn_established(FramedConn::new(a2), peer("node-c", 1), Timing::default())
.expect("node-c registers");
let ev = subscribe().expect("manager is up");
let mut names = Vec::new();
for _ in 0..2 {
match next_event(&ev, "a snapshot node_up") {
NodeEvent::NodeUp(info) => names.push(info.name),
other => panic!("expected a snapshot node_up, got {other:?}"),
}
}
names.sort();
assert_eq!(names, ["node-b", "node-c"]);
assert_quiet(&ev); // the snapshot is exactly the live set
let mut v = view().expect("manager is up");
v.sort_by(|a, b| a.name.cmp(&b.name));
assert_eq!(v.len(), 2);
assert_eq!(v[0].name, "node-b");
assert_eq!(v[1].name, "node-c");
disconnect("node-b");
disconnect("node-c");
drop((b1, b2));
// Drain the two downs so the subscription ends quiet.
let _ = next_event(&ev, "node_down");
let _ = next_event(&ev, "node_down");
mgr.shutdown();
});
}
/// NodeId identity: a restart (same name, new incarnation) is a NEW id — the
/// ghost and its successor are distinguishable — while a reconnect blip (same
/// name, same incarnation) keeps its id.
#[test]
fn restart_gets_new_id_blip_keeps_id() {
run(|| {
let mgr = GenServerBuilder::new(Manager::new())
.named(MANAGER)
.start()
.expect("manager name is free");
let ev = subscribe().expect("manager is up");
let t = TcpTransport;
let id = |e: NodeEvent, what: &str| -> NodeId {
match e {
NodeEvent::NodeUp(info) => info.node,
other => panic!("expected node_up ({what}), got {other:?}"),
}
};
let down_id = |e: NodeEvent, what: &str| -> NodeId {
match e {
NodeEvent::NodeDown(info) => info.node,
other => panic!("expected node_down ({what}), got {other:?}"),
}
};
// Up at incarnation 1, then the peer dies (EOF).
let (a1, b1) = pair(&t);
spawn_established(FramedConn::new(a1), peer("node-b", 1), Timing::default())
.expect("registers");
let id1 = id(next_event(&ev, "node_up inc 1"), "inc 1");
drop(b1);
assert_eq!(down_id(next_event(&ev, "node_down inc 1"), "inc 1"), id1);
// Restart: new incarnation, new id — the ghost's id is not reused.
let (a2, b2) = pair(&t);
spawn_established(FramedConn::new(a2), peer("node-b", 2), Timing::default())
.expect("registers");
let id2 = id(next_event(&ev, "node_up inc 2"), "inc 2");
assert_ne!(
id1, id2,
"a restarted node must be distinguishable from its ghost"
);
// Blip: the same incarnation reconnects and keeps its id.
disconnect("node-b");
assert_eq!(down_id(next_event(&ev, "node_down inc 2"), "inc 2"), id2);
let (a3, b3) = pair(&t);
spawn_established(FramedConn::new(a3), peer("node-b", 2), Timing::default())
.expect("registers");
let id3 = id(next_event(&ev, "node_up after blip"), "blip");
assert_eq!(
id2, id3,
"a reconnect at the same incarnation is the same node"
);
disconnect("node-b");
let _ = next_event(&ev, "final node_down");
drop((b2, b3));
mgr.shutdown();
});
}
/// A dropped subscriber is pruned on the next emit and never disturbs the
/// manager or a live subscriber.
#[test]
fn dead_subscriber_is_pruned() {
run(|| {
let mgr = GenServerBuilder::new(Manager::new())
.named(MANAGER)
.start()
.expect("manager name is free");
let dead = subscribe().expect("manager is up");
drop(dead);
let live = subscribe().expect("manager is up");
let t = TcpTransport;
let (a1, b1) = pair(&t);
spawn_established(FramedConn::new(a1), peer("node-b", 1), Timing::default())
.expect("registers");
match next_event(&live, "node_up despite a dead co-subscriber") {
NodeEvent::NodeUp(info) => assert_eq!(info.name, "node-b"),
other => panic!("expected node_up, got {other:?}"),
}
disconnect("node-b");
let _ = next_event(&live, "node_down");
drop(b1);
mgr.shutdown();
});
}
+175
View File
@@ -0,0 +1,175 @@
//! RFC 010 c7 — the Phase 2 gate: a 3-node mesh under the subprocess
//! harness, repeatable.
//!
//! Each node process runs the integrated `cluster::start` (manager +
//! acceptor + connector + static seeds), subscribes to membership like any
//! consumer, and announces protocol-visible facts as lines:
//! `LISTENING <addr>`, `MEMBER-UP <name> inc=<n>`, `MEMBER-DOWN <name>`.
//! Then it **parks forever** — cross-process teardown is retractable state
//! (binding trap), so the parent SIGKILLs via `Node`'s `Drop` and clean exit
//! stays the c4 harness's own smoke test.
//!
//! Ports: nodes bind `:0` and report, so the mesh is built by seeding each
//! node with the previously-reported addresses (n1: no seeds; n2: n1;
//! n3: n1+n2 — inbound covers the reverse edges). The late-seed test is the
//! one exception: the parent pre-reserves a port by binding-and-closing it,
//! seeds one node with it, then starts the second node on that exact
//! address. In principle another process could steal the port in the gap;
//! in practice the window is microseconds on a local runner — accepted, and
//! confined to that one test.
#![cfg(feature = "cluster")]
mod common;
use common::{maybe_child, spawn_node, Node};
use smarm::cluster::envelope::NodeMeta;
use smarm::cluster::membership::{subscribe, NodeEvent};
use smarm::cluster::{start, Config, StaticSeeds, Timing};
use std::time::Duration;
const ROLES: &[(&str, fn())] = &[("node", role_node)];
/// A mesh node: identity and seeds from env, membership events to stdout,
/// park forever (the parent reaps).
fn role_node() {
let name = std::env::var("SMARM_NODE_NAME").expect("SMARM_NODE_NAME not set");
let listen = std::env::var("SMARM_LISTEN_ADDR").unwrap_or_else(|_| "127.0.0.1:0".to_string());
// Seeds: comma-separated `name=addr` pairs; empty or unset means none.
let seeds: Vec<(String, String)> = std::env::var("SMARM_SEEDS")
.unwrap_or_default()
.split(',')
.filter(|s| !s.is_empty())
.map(|s| {
let (n, a) = s.split_once('=').expect("seed must be name=addr");
(n.to_string(), a.to_string())
})
.collect();
smarm::run(move || {
let cluster = start(Config {
node_name: name,
meta: NodeMeta {
role: "mesh-test".to_string(),
region: "local".to_string(),
},
listen_addr: listen,
strategy: Box::new(StaticSeeds::new(seeds)),
timing: Timing::default(),
})
.expect("listener binds");
println!("LISTENING {}", cluster.local_addr());
let events = subscribe().expect("manager is up");
loop {
match events.rx.recv() {
Ok(NodeEvent::NodeUp(info)) => {
println!("MEMBER-UP {} inc={}", info.name, info.incarnation.get());
}
Ok(NodeEvent::NodeDown(info)) => {
println!("MEMBER-DOWN {}", info.name);
}
Err(_) => break, // manager gone; park below regardless
}
}
loop {
smarm::sleep(Duration::from_secs(3600));
}
});
}
fn spawn_mesh_node(name: &str, seeds: &str, listen: Option<&str>) -> Node {
let mut env: Vec<(&str, &str)> = vec![("SMARM_NODE_NAME", name), ("SMARM_SEEDS", seeds)];
if let Some(addr) = listen {
env.push(("SMARM_LISTEN_ADDR", addr));
}
spawn_node("node", &env)
}
/// Wait for `MEMBER-UP <peer> inc=<n>` and return the incarnation.
fn wait_member_up(node: &mut Node, peer: &str) -> u32 {
let prefix = format!("MEMBER-UP {peer} inc=");
let line = node.wait_line(&format!("MEMBER-UP {peer}"), |l| l.starts_with(&prefix));
line[prefix.len()..].parse().expect("incarnation parses")
}
fn wait_member_down(node: &mut Node, peer: &str) {
let want = format!("MEMBER-DOWN {peer}");
node.wait_line(&want, |l| l == want);
}
/// The gate, plus the kill and restart facts, as one mesh's life: three
/// nodes form a full mesh (every node sees both others up); killing one
/// yields `node_down` at both survivors; its restart under the same name
/// arrives as a NEW incarnation — the ghost and its successor are
/// distinguishable at every observer.
#[test]
fn three_node_mesh_forms_then_kill_then_restart_distinguishable() {
maybe_child(ROLES);
let mut n1 = spawn_mesh_node("node-1", "", None);
let a1 = n1.wait_listening();
let mut n2 = spawn_mesh_node("node-2", &format!("node-1={a1}"), None);
let a2 = n2.wait_listening();
let mut n3 = spawn_mesh_node("node-3", &format!("node-1={a1},node-2={a2}"), None);
let _a3 = n3.wait_listening();
// Full mesh: each node reports both peers up (dialed or inbound alike).
wait_member_up(&mut n1, "node-2");
let inc3_at_n1 = wait_member_up(&mut n1, "node-3");
wait_member_up(&mut n2, "node-1");
let inc3_at_n2 = wait_member_up(&mut n2, "node-3");
wait_member_up(&mut n3, "node-1");
wait_member_up(&mut n3, "node-2");
assert_eq!(
inc3_at_n1, inc3_at_n2,
"one node, one incarnation, all observers"
);
// Kill node-3 (SIGKILL via Drop): node_down at both survivors.
drop(n3);
wait_member_down(&mut n1, "node-3");
wait_member_down(&mut n2, "node-3");
// Restart node-3 under the same name: it re-dials its seeds and comes
// up everywhere as a new incarnation — never the ghost's.
let mut n3b = spawn_mesh_node("node-3", &format!("node-1={a1},node-2={a2}"), None);
let _ = n3b.wait_listening();
let inc3b_at_n1 = wait_member_up(&mut n1, "node-3");
let inc3b_at_n2 = wait_member_up(&mut n2, "node-3");
assert_eq!(inc3b_at_n1, inc3b_at_n2);
assert_ne!(
inc3_at_n1, inc3b_at_n1,
"a restarted node must be distinguishable from its ghost"
);
wait_member_up(&mut n3b, "node-1");
wait_member_up(&mut n3b, "node-2");
}
/// A seed that is unreachable at start is not fatal: the connector retries
/// on backoff, and when a node finally appears at that address, the mesh
/// edge forms.
#[test]
fn seed_unreachable_at_start_then_arriving_later() {
maybe_child(ROLES);
// Pre-reserve an address by binding and immediately closing it (see the
// module docs for the accepted steal window). Dials to it are refused
// until node-b starts there.
let reserved = {
let l = std::net::TcpListener::bind("127.0.0.1:0").expect("bind");
l.local_addr().expect("addr").to_string()
};
let mut a = spawn_mesh_node("node-a", &format!("node-b={reserved}"), None);
let _ = a.wait_listening();
// Let a few refused attempts happen before the seed comes up, so the
// retry path is what forms the edge (backoff cap 5s < harness WAIT 10s).
std::thread::sleep(Duration::from_millis(600));
let mut b = spawn_mesh_node("node-b", "", Some(&reserved));
let _ = b.wait_listening();
wait_member_up(&mut a, "node-b");
wait_member_up(&mut b, "node-a");
}
+359
View File
@@ -0,0 +1,359 @@
//! RFC 010 c12 — remote monitors.
//!
//! Local suite (`run()`, no network): the immediate answers — no connection
//! ⇒ `Disconnected`, dead incarnation ⇒ `NoProc` — and the self-node
//! collapse (a plain local monitor underneath, incl. `demonitor_remote`).
//!
//! Cross-process: a *server* exposes a control name and spawns workers on
//! request, replying with each worker's pid (via `RemotePid::from_local`,
//! the D12 set-site) or, for the deliberately unshipped one, only its raw
//! slot numbers. The *client* monitors them and asserts: kill ⇒ the true
//! reason (Exit / Panic); a corpse ⇒ its recorded terminal reason, not
//! NoProc; a live pid that never crossed the wire ⇒ NoProc (no liveness
//! leak); a demonitor racing the kill ⇒ no notice, proven by stream ORDER
//! (a later notice on the same connection arrives while the earlier slot
//! is still empty), not by sleeping.
#![cfg(feature = "cluster")]
mod common;
use common::{maybe_child, spawn_node};
use smarm::cluster::envelope::NodeMeta;
use smarm::cluster::expose::{expose, expose_type};
use smarm::cluster::membership::{subscribe, NodeEvent};
use smarm::cluster::remote::{
self, demonitor_remote, monitor_remote, send_to_remote, RemoteName, RemotePid,
};
use smarm::cluster::{start, Config, RemoteDownReason, StaticSeeds, Timing};
use smarm::pg::Incarnation;
use smarm::{channel, install, register, run, spawn, Addressable, DownReason, Erased, Name, Pid};
use std::collections::HashMap;
use std::time::Duration;
// ---- message types (hand-rolled serde; the crate is derive-less) ---------
#[derive(Debug)]
struct Ctl {
cmd: String,
reply_to: RemotePid<Client>,
}
#[derive(Debug)]
struct Answer {
text: String,
pid: Option<RemotePid<Erased>>,
}
struct Client;
impl Addressable for Client {
type Msg = Answer;
}
impl serde::Serialize for Ctl {
fn serialize<S: serde::Serializer>(&self, s: S) -> Result<S::Ok, S::Error> {
use serde::ser::SerializeTuple;
let mut t = s.serialize_tuple(2)?;
t.serialize_element(&self.cmd)?;
t.serialize_element(&self.reply_to)?;
t.end()
}
}
impl<'de> serde::Deserialize<'de> for Ctl {
fn deserialize<D: serde::Deserializer<'de>>(d: D) -> Result<Self, D::Error> {
let (cmd, reply_to) = <(String, RemotePid<Client>)>::deserialize(d)?;
Ok(Ctl { cmd, reply_to })
}
}
impl serde::Serialize for Answer {
fn serialize<S: serde::Serializer>(&self, s: S) -> Result<S::Ok, S::Error> {
use serde::ser::SerializeTuple;
let mut t = s.serialize_tuple(2)?;
t.serialize_element(&self.text)?;
t.serialize_element(&self.pid)?;
t.end()
}
}
impl<'de> serde::Deserialize<'de> for Answer {
fn deserialize<D: serde::Deserializer<'de>>(d: D) -> Result<Self, D::Error> {
let (text, pid) = <(String, Option<RemotePid<Erased>>)>::deserialize(d)?;
Ok(Answer { text, pid })
}
}
// ================= local suite =========================================
/// No connection to the pid's node: `Disconnected` at once — the remote
/// analog of NoProc, and the first thing c11's variant is for.
#[test]
fn unconnected_node_is_disconnected_immediately() {
maybe_child(ROLES);
run(|| {
remote::set_local_identity("me", Incarnation::new(7));
let ghost = RemotePid::<Erased>::from_parts("nowhere", Incarnation::new(1), 3, 1);
let m = monitor_remote(ghost.clone());
let d = m.recv().unwrap();
assert_eq!(d.pid, ghost);
assert_eq!(d.reason, RemoteDownReason::Disconnected);
});
}
/// The node is connected but the pid names an earlier incarnation: the
/// actor is a known corpse (RFC v2 §3), so `NoProc` at once — never
/// `Disconnected`, nothing on the wire.
#[test]
fn dead_incarnation_is_noproc_immediately() {
maybe_child(ROLES);
run(|| {
remote::set_local_identity("me", Incarnation::new(7));
let (probe_tx, probe_rx) = channel();
remote::bind_outbound_probe("peer", Incarnation::new(5), probe_tx);
let stale = RemotePid::<Erased>::from_parts("peer", Incarnation::new(4), 9, 1);
let m = monitor_remote(stale);
assert_eq!(m.recv().unwrap().reason, DownReason::NoProc.into());
assert!(probe_rx.try_recv().unwrap().is_none(), "no frame emitted");
});
}
/// A self-node pid collapses to an ordinary local monitor: the true reason
/// on exit, and `demonitor_remote` cancels it.
#[test]
fn self_node_pid_collapses_to_local_monitor() {
maybe_child(ROLES);
run(|| {
remote::set_local_identity("me", Incarnation::new(7));
let (go_tx, go_rx) = channel::<()>();
let (go2_tx, go2_rx) = channel::<()>();
let a = spawn(move || {
let _ = go_rx.recv();
})
.pid();
let b = spawn(move || {
let _ = go2_rx.recv();
})
.pid();
let ma = monitor_remote(RemotePid::from_local(a).expect("identity set"));
let mb = monitor_remote(RemotePid::from_local(b).expect("identity set"));
assert_ne!(ma.id, mb.id);
assert!(ma.target.local() == Some(a));
demonitor_remote(&mb);
go2_tx.send(()).unwrap();
go_tx.send(()).unwrap();
let d = ma.recv().unwrap();
assert_eq!(d.reason, DownReason::Exit.into());
assert_eq!(d.pid.local(), Some(a));
// `a` is down (its notice arrived), and `b` was killed first on the
// same scheduler — a notice for `b` would be here by now. After a
// demonitor the channel is closed-empty (`Err`), like the local one.
assert!(matches!(mb.try_recv(), Ok(None) | Err(_)));
});
}
// ================= cross-process ======================================
const ROLES: &[(&str, fn())] = &[("server", role_server), ("client", role_client)];
const CTL: Name<Ctl> = Name::new("c12.ctl");
fn cfg(name: &str, seeds: Vec<(String, String)>) -> Config {
Config {
node_name: name.to_string(),
meta: NodeMeta {
role: "c12".into(),
region: "local".into(),
},
listen_addr: std::env::var("SMARM_LISTEN_ADDR").unwrap_or_else(|_| "127.0.0.1:0".into()),
strategy: Box::new(StaticSeeds::new(seeds)),
timing: Timing::default(),
}
}
fn wait_up(events: &smarm::cluster::membership::MembershipEvents, who: &str) {
loop {
match events.rx.recv() {
Ok(NodeEvent::NodeUp(i)) if i.name == who => return,
Ok(_) => continue,
Err(_) => panic!("manager gone"),
}
}
}
/// Server commands (all answered to `reply_to`):
/// - `spawn:exit` / `spawn:panic` — a parked worker; `kill:<index>` releases
/// it, whereupon it returns / panics. Answer carries its pid.
/// - `spawn:corpse` — a worker that has already exited when the answer is
/// sent; the pid was shipped (watchable) before it died.
/// - `spawn:unwatched` — a parked worker whose pid is NEVER shipped; the
/// answer carries only `text = "slot:<index>:<generation>"`.
fn role_server() {
smarm::run(move || {
let cluster = start(cfg("server", vec![])).expect("binds");
println!("LISTENING {}", cluster.local_addr());
let (tx, rx) = channel::<Ctl>();
register(CTL, tx).unwrap();
expose(CTL);
println!("READY");
let mut workers: HashMap<u32, smarm::channel::Sender<()>> = HashMap::new();
loop {
let ctl = rx.recv().unwrap();
println!("CTL {}", ctl.cmd);
let (text, pid): (String, Option<RemotePid<Erased>>) = match ctl.cmd.as_str() {
"spawn:exit" | "spawn:panic" => {
let panic = ctl.cmd == "spawn:panic";
let (go_tx, go_rx) = channel::<()>();
let p: Pid = spawn(move || {
let _ = go_rx.recv();
if panic {
panic!("worker asked to panic");
}
})
.pid();
workers.insert(p.index(), go_tx);
(
"ok".into(),
Some(RemotePid::from_local(p).expect("identity set")),
)
}
"spawn:corpse" => {
let p: Pid = spawn(|| {}).pid();
let rp = RemotePid::from_local(p).expect("identity set"); // shipped ⇒ watchable
let m = smarm::monitor(p);
let _ = m.rx.recv(); // dead before the answer goes out
("ok".into(), Some(rp))
}
"spawn:unwatched" => {
let (go_tx, go_rx) = channel::<()>();
let p: Pid = spawn(move || {
let _ = go_rx.recv();
})
.pid();
workers.insert(p.index(), go_tx);
(format!("slot:{}:{}", p.index(), p.generation()), None)
}
other => {
let idx: u32 = other.strip_prefix("kill:").unwrap().parse().unwrap();
if let Some(go) = workers.remove(&idx) {
let _ = go.send(());
}
("killed".into(), None)
}
};
send_to_remote(ctl.reply_to, Answer { text, pid }).unwrap();
}
});
}
fn role_client() {
let server_addr = std::env::var("SMARM_SERVER_ADDR").expect("SMARM_SERVER_ADDR");
smarm::run(move || {
let _cluster = start(cfg("client", vec![("server".into(), server_addr)])).expect("binds");
let ev = subscribe().unwrap();
wait_up(&ev, "server");
let (tx, rx) = channel::<Answer>();
let me: Pid<Client> = install::<Client>(tx);
expose_type::<Answer>();
let ask = |cmd: &str| -> Answer {
remote::send(
RemoteName::new("server", CTL),
Ctl {
cmd: cmd.into(),
reply_to: RemotePid::from_local(me).expect("identity set"),
},
)
.unwrap();
rx.recv().unwrap()
};
let server_inc = ev_incarnation();
// 1. kill ⇒ true reason (Exit).
let a = ask("spawn:exit").pid.unwrap();
let ma = monitor_remote(a.clone());
ask(&format!("kill:{}", a.index()));
let d = ma.recv().unwrap();
assert_eq!(d.pid, a);
println!("DOWN exit {:?}", d.reason);
// 2. kill ⇒ true reason (Panic).
let b = ask("spawn:panic").pid.unwrap();
let mb = monitor_remote(b.clone());
ask(&format!("kill:{}", b.index()));
println!("DOWN panic {:?}", mb.recv().unwrap().reason);
// 3. corpse ⇒ recorded terminal reason, not NoProc.
let c = ask("spawn:corpse").pid.unwrap();
println!("DOWN corpse {:?}", monitor_remote(c).recv().unwrap().reason);
// 4. live but never shipped/exposed ⇒ NoProc (no leak); a made-up
// slot on the same node ⇒ NoProc too, indistinguishably.
let ans = ask("spawn:unwatched");
let mut it = ans.text.strip_prefix("slot:").unwrap().split(':');
let (idx, gen): (u32, u32) = (
it.next().unwrap().parse().unwrap(),
it.next().unwrap().parse().unwrap(),
);
let hidden = RemotePid::<Erased>::from_parts("server", server_inc, idx, gen);
println!(
"DOWN hidden {:?}",
monitor_remote(hidden).recv().unwrap().reason
);
let bogus = RemotePid::<Erased>::from_parts("server", server_inc, 100_000, 1);
println!(
"DOWN bogus {:?}",
monitor_remote(bogus).recv().unwrap().reason
);
// 5. demonitor races the kill: no notice for `d1`, proven by order —
// `d2`'s notice (same connection, later) arrives while `d1`'s
// slot is still empty.
let d1 = ask("spawn:exit").pid.unwrap();
let m1 = monitor_remote(d1.clone());
demonitor_remote(&m1);
ask(&format!("kill:{}", d1.index()));
let d2 = ask("spawn:exit").pid.unwrap();
let m2 = monitor_remote(d2.clone());
ask(&format!("kill:{}", d2.index()));
assert_eq!(m2.recv().unwrap().reason, DownReason::Exit.into());
// Closed-empty (`Err`) or open-empty (`Ok(None)`) both mean no notice.
let stray = matches!(m1.try_recv(), Ok(Some(_)));
println!("DEMONITOR stray={stray}");
println!("CLIENT DONE");
loop {
smarm::sleep(Duration::from_secs(3600));
}
});
}
/// The server's incarnation as this node sees it — for building pids by hand.
fn ev_incarnation() -> Incarnation {
smarm::cluster::membership::view()
.expect("manager up")
.into_iter()
.find(|i| i.name == "server")
.map(|i| i.incarnation)
.expect("server in view")
}
/// The Phase 4 c12 gate: remote monitors report the true reason, honour
/// corpses, leak nothing for unshipped pids, and cancel cleanly.
#[test]
fn remote_monitors_report_true_reasons() {
maybe_child(ROLES);
let mut server = spawn_node("server", &[]);
let saddr = server.wait_listening();
server.wait_line("READY", |l| l == "READY");
let mut client = spawn_node("client", &[("SMARM_SERVER_ADDR", &saddr)]);
client.wait_line("DOWN exit Local(Exit)", |l| l == "DOWN exit Local(Exit)");
client.wait_line("DOWN panic Local(Panic)", |l| {
l == "DOWN panic Local(Panic)"
});
client.wait_line("DOWN corpse Local(Exit)", |l| {
l == "DOWN corpse Local(Exit)"
});
client.wait_line("DOWN hidden Local(NoProc)", |l| {
l == "DOWN hidden Local(NoProc)"
});
client.wait_line("DOWN bogus Local(NoProc)", |l| {
l == "DOWN bogus Local(NoProc)"
});
client.wait_line("DEMONITOR stray=false", |l| l == "DEMONITOR stray=false");
client.wait_line("CLIENT DONE", |l| l == "CLIENT DONE");
}
+254
View File
@@ -0,0 +1,254 @@
//! RFC 010 c15 — distributed pg: sync on `NodeUp`, incremental
//! `Join`/`Leave`, eager eviction announced, `NodeDown` sweep.
//!
//! Two nodes. The *origin* joins two local workers to `"pool"` before the
//! *observer* connects (so the observer's view comes from `Sync`), exposes a
//! `"go"` command inbox and then does exactly what the observer tells it:
//! kill one worker, join a third, leave with the second. The observer drives
//! that script through the cluster itself and asserts every step from
//! `members_all` — never touching the group on its own side, except once to
//! prove a mixed local+remote group reads correctly and that `members` stays
//! local. `dispatch_any` is exercised both ways: into the origin's worker
//! (remote pick, `send_to_remote`) and, once the origin is gone, into the
//! observer's own (local pick, `send_to`). Finally the parent SIGKILLs the
//! origin: the observer must sweep every remote member on `NodeDown`.
#![cfg(feature = "cluster")]
mod common;
use common::{maybe_child, spawn_node};
use smarm::cluster::envelope::NodeMeta;
use smarm::cluster::expose::{expose, expose_type};
use smarm::cluster::membership::{subscribe, NodeEvent};
use smarm::cluster::remote::{self, RemoteName};
use smarm::cluster::{
dispatch_any, members_all, pick_any, start, Config, DispatchAnyError, GroupMember, StaticSeeds,
Timing,
};
use smarm::{channel, join, leave, members, register, send_to, spawn_addr, Addressable, Name, Pid};
use std::time::{Duration, Instant};
const GO: Name<u8> = Name::new("go");
const POOL: &str = "pool";
/// A pool worker's message: `"die"` stops it, anything else is printed.
#[derive(Debug, PartialEq)]
struct Job(String);
struct Worker;
impl Addressable for Worker {
type Msg = Job;
}
impl serde::Serialize for Job {
fn serialize<S: serde::Serializer>(&self, s: S) -> Result<S::Ok, S::Error> {
self.0.serialize(s)
}
}
impl<'de> serde::Deserialize<'de> for Job {
fn deserialize<D: serde::Deserializer<'de>>(d: D) -> Result<Self, D::Error> {
String::deserialize(d).map(Job)
}
}
const ROLES: &[(&str, fn())] = &[("origin", role_origin), ("observer", role_observer)];
fn cfg(name: &str, seeds: Vec<(String, String)>) -> Config {
Config {
node_name: name.into(),
meta: NodeMeta {
role: "c15".into(),
region: "local".into(),
},
listen_addr: "127.0.0.1:0".into(),
strategy: Box::new(StaticSeeds::new(seeds)),
timing: Timing::default(),
}
}
/// A pool worker: prints every job it is handed, exits on `"die"`.
fn worker() -> Pid<Worker> {
spawn_addr::<Worker>(|rx| {
while let Ok(Job(s)) = rx.recv() {
if s == "die" {
return;
}
println!("JOB {s}");
}
})
}
fn role_origin() {
smarm::run(|| {
let cluster = start(cfg("origin", vec![])).expect("binds");
// Remote dispatch lands here only for a type this node accepts.
expose_type::<Job>();
let w1 = worker();
let w2 = worker();
assert!(join(POOL, w1));
assert!(join(POOL, w2));
let (go_tx, go_rx) = channel::<u8>();
register(GO, go_tx).unwrap();
expose(GO);
println!("LISTENING {}", cluster.local_addr());
println!("JOINED 2");
loop {
match go_rx.recv().unwrap() {
1 => {
send_to(w1, Job("die".into())).unwrap();
println!("KILLED w1");
}
2 => {
assert!(leave(POOL, w2));
println!("LEFT w2");
}
3 => {
let w3 = worker();
assert!(join(POOL, w3));
println!("JOINED w3");
}
n => panic!("unknown command {n}"),
}
}
});
}
fn remote_count(group: &str) -> usize {
members_all(group)
.iter()
.filter(|m| matches!(m, GroupMember::Remote(_)))
.count()
}
/// Cooperative poll until `pred`; panics (with the last view) on timeout.
fn wait_view(what: &str, group: &str, pred: impl Fn(&[GroupMember]) -> bool) {
let deadline = Instant::now() + Duration::from_secs(5);
loop {
let v = members_all(group);
if pred(&v) {
return;
}
assert!(
Instant::now() < deadline,
"timed out waiting for {what}; view = {v:?}"
);
smarm::sleep(Duration::from_millis(5));
}
}
fn role_observer() {
let origin_addr = std::env::var("SMARM_ORIGIN_ADDR").expect("SMARM_ORIGIN_ADDR");
smarm::run(move || {
let _cluster = start(cfg("observer", vec![("origin".into(), origin_addr)])).expect("binds");
let ev = subscribe().unwrap();
loop {
match ev.rx.recv().unwrap() {
NodeEvent::NodeUp(i) if i.name == "origin" => break,
_ => {}
}
}
let go = |n: u8| remote::send(RemoteName::new("origin", GO), n).unwrap();
// Sync: both pre-existing members arrive with no join on this side.
wait_view("sync of 2 remote members", POOL, |v| {
v.len() == 2 && v.iter().all(|m| matches!(m, GroupMember::Remote(_)))
});
let synced = members_all(POOL);
assert!(synced.iter().all(|m| match m {
GroupMember::Remote(p) => p.node() == "origin",
GroupMember::Local(_) => false,
}));
println!("SEES 2");
// Origin-side death: the origin's reaper announces the leave.
go(1);
wait_view("death evicted on observer", POOL, |v| v.len() == 1);
println!("SEES 1 after death");
// Incremental Join.
go(3);
wait_view("incremental join", POOL, |v| v.len() == 2);
println!("SEES 2 after join");
// Voluntary Leave.
go(2);
wait_view("incremental leave", POOL, |v| v.len() == 1);
println!("SEES 1 after leave");
// Mixed group: our own member sits beside the remote one in
// `members_all`; `members` stays local-only.
let me = worker();
assert!(join(POOL, me));
wait_view("mixed local+remote", POOL, |v| {
v.len() == 2 && v.contains(&GroupMember::Local(me.erase()))
});
assert_eq!(
members(POOL),
vec![me.erase()],
"local API never shows remotes"
);
assert_eq!(remote_count(POOL), 1);
println!("MIXED ok");
// dispatch_any: the store's first entry is the origin's w3 (it was
// announced before we joined), so the pick is remote and the job
// crosses the wire — the origin's worker prints it.
let picked = pick_any(POOL).expect("pool has members");
assert!(
matches!(picked, GroupMember::Remote(_)),
"first entry is remote: {picked:?}"
);
let reached = dispatch_any::<Worker>(POOL, Job("from-observer".into())).unwrap();
assert_eq!(reached, picked);
println!("DISPATCHED remote");
println!("PARK");
// Parent SIGKILLs the origin now: NodeDown must sweep its member,
// ours must survive.
wait_view("node_down sweep", POOL, |v| {
v == [GroupMember::Local(me.erase())]
});
assert_eq!(members(POOL), vec![me.erase()]);
println!("SWEPT");
// Now the only member is ours: a local pick, a local send.
let reached = dispatch_any::<Worker>(POOL, Job("local".into())).unwrap();
assert_eq!(reached, GroupMember::Local(me.erase()));
// And an empty group hands the message back.
match dispatch_any::<Worker>("nobody", Job("lost".into())) {
Err(DispatchAnyError::NoMember(Job(s))) => assert_eq!(s, "lost"),
other => panic!("expected NoMember, got {other:?}"),
}
println!("DISPATCHED local");
loop {
smarm::sleep(Duration::from_secs(3600));
}
});
}
/// The Phase 5 gate: sync, join, leave, death, node_down — all observed from
/// the peer, none of them a group operation on the peer — plus dispatch_any
/// reaching a remote member and a local one.
#[test]
fn groups_span_two_nodes() {
maybe_child(ROLES);
let mut origin = spawn_node("origin", &[]);
let addr = origin.wait_listening();
origin.wait_line("JOINED 2", |l| l == "JOINED 2");
let mut observer = spawn_node("observer", &[("SMARM_ORIGIN_ADDR", &addr)]);
observer.wait_line("SEES 2", |l| l == "SEES 2");
origin.wait_line("KILLED w1", |l| l == "KILLED w1");
observer.wait_line("SEES 1 after death", |l| l == "SEES 1 after death");
origin.wait_line("JOINED w3", |l| l == "JOINED w3");
observer.wait_line("SEES 2 after join", |l| l == "SEES 2 after join");
origin.wait_line("LEFT w2", |l| l == "LEFT w2");
observer.wait_line("SEES 1 after leave", |l| l == "SEES 1 after leave");
observer.wait_line("MIXED ok", |l| l == "MIXED ok");
observer.wait_line("DISPATCHED remote", |l| l == "DISPATCHED remote");
origin.wait_line("JOB from-observer", |l| l == "JOB from-observer");
observer.wait_line("PARK", |l| l == "PARK");
origin.kill();
observer.wait_line("SWEPT", |l| l == "SWEPT");
// Order between the root's line and the worker's is scheduling; wait
// for the later one to be certain both happened.
observer.wait_line("DISPATCHED local", |l| l == "DISPATCHED local");
observer.wait_line("JOB local", |l| l == "JOB local");
}
+356
View File
@@ -0,0 +1,356 @@
//! RFC 010 c10 — pid targeting + auto-serialization. The Phase 3 gate:
//! cross-node call/reply with no ceremony, under the subprocess harness.
//!
//! Local suite (`run()`, no network): serialize/deserialize shapes,
//! self-collapse, the outside-runtime contract, the local send-site
//! incarnation check with a probe proving **no frame is emitted**.
//!
//! Cross-process: two nodes. The *server* exposes a `Name<Req>`; the
//! *client* sends a `Req` carrying its own `Pid<Reply>` (auto-serialized to
//! a `RemotePid` on the wire); the server replies via `send_to_remote`
//! straight back to that pid — no name at the client end, no ceremony. A
//! third-node roundtrip: the client's pid travels client→server→relay→
//! server→client, and still delivers.
#![cfg(feature = "cluster")]
mod common;
use common::{maybe_child, spawn_node};
use smarm::cluster::envelope::{encode_payload, Frame, NodeMeta};
use smarm::cluster::expose::{expose, type_hash};
use smarm::cluster::membership::{subscribe, NodeEvent};
use smarm::cluster::remote::{self, send_to_remote, RemoteName, RemotePid, ToRemoteError};
use smarm::cluster::{start, Config, StaticSeeds, Timing};
use smarm::pg::Incarnation;
use smarm::{channel, install, register, run, Addressable, Name, Pid};
use std::time::Duration;
// ---- message types (std-only payloads; the crate's serde is derive-less,
// so wire types are hand-rolled with serde's tuple/seq API via `serde::ser`
// impls below — the same thing a user's derive would generate) ------------
/// A request carrying a reply-to. Serialize/Deserialize are written by hand
/// here for exactly one reason: this crate deliberately does not pull in
/// serde-derive. Field 1 is the auto-serializing pid.
#[derive(Debug, PartialEq)]
struct Req {
text: String,
reply_to: RemotePid<Replier>,
}
#[derive(Debug, PartialEq)]
struct Reply(String);
struct Replier;
impl Addressable for Replier {
type Msg = Reply;
}
impl serde::Serialize for Req {
fn serialize<S: serde::Serializer>(&self, s: S) -> Result<S::Ok, S::Error> {
use serde::ser::SerializeTuple;
let mut t = s.serialize_tuple(2)?;
t.serialize_element(&self.text)?;
t.serialize_element(&self.reply_to)?;
t.end()
}
}
impl<'de> serde::Deserialize<'de> for Req {
fn deserialize<D: serde::Deserializer<'de>>(d: D) -> Result<Self, D::Error> {
let (text, reply_to) = <(String, RemotePid<Replier>)>::deserialize(d)?;
Ok(Req { text, reply_to })
}
}
impl serde::Serialize for Reply {
fn serialize<S: serde::Serializer>(&self, s: S) -> Result<S::Ok, S::Error> {
self.0.serialize(s)
}
}
impl<'de> serde::Deserialize<'de> for Reply {
fn deserialize<D: serde::Deserializer<'de>>(d: D) -> Result<Self, D::Error> {
String::deserialize(d).map(Reply)
}
}
// ================= local suite =========================================
/// A local `Pid<A>` serializes as a `RemotePid<A>` stamped with this node's
/// identity; deserializing it back on the same node collapses to the same
/// local pid (`local()` is `Some`, `Pid` round-trips).
#[test]
fn local_pid_serializes_and_collapses_on_self() {
maybe_child(ROLES);
run(|| {
// The local identity is set by cluster::start; the local suite sets
// it directly.
remote::set_local_identity("me", Incarnation::new(7));
let (tx, _rx) = channel::<Reply>();
let me: Pid<Replier> = install::<Replier>(tx);
let bytes = encode_payload(&me).unwrap();
let rp: RemotePid<Replier> = smarm::cluster::envelope::decode_payload(&bytes).unwrap();
assert_eq!(rp.node(), "me");
assert_eq!(rp.incarnation(), Incarnation::new(7));
assert_eq!(
rp.local(),
Some(me),
"self-node pid collapses to the local pid"
);
// Deserializing straight into Pid<A> works for a self-node pid...
let back: Pid<Replier> = smarm::cluster::envelope::decode_payload(&bytes).unwrap();
assert_eq!(back, me);
// ...and FAILS for a foreign one (collapse is literal: node == self).
let foreign = RemotePid::<Replier>::from_parts("elsewhere", Incarnation::new(1), 3, 1);
let fbytes = encode_payload(&foreign).unwrap();
assert!(smarm::cluster::envelope::decode_payload::<Pid<Replier>>(&fbytes).is_err());
assert_eq!(foreign.local(), None);
});
}
/// `send_to_remote` short-circuits locally for a self-node pid — the
/// zero-copy-equivalent collapse: the message object itself lands in the
/// local channel, no encode, no frame.
#[test]
fn send_to_remote_collapses_locally_for_self() {
maybe_child(ROLES);
run(|| {
remote::set_local_identity("me", Incarnation::new(7));
let (tx, rx) = channel::<Reply>();
let me: Pid<Replier> = install::<Replier>(tx);
let rp = RemotePid::from_local(me).expect("identity set");
// Probe the outbound path: nothing must be handed to any connection.
let (probe_tx, probe_rx) = channel::<Frame>();
remote::bind_outbound_probe("me", Incarnation::new(7), probe_tx);
send_to_remote(rp, Reply("hi".into())).unwrap();
assert_eq!(rx.recv().unwrap(), Reply("hi".into()));
assert!(
matches!(probe_rx.try_recv(), Ok(None)),
"no frame for a local collapse"
);
});
}
/// RFC v2 §3: a `RemotePid` whose incarnation is not the current one for its
/// node fails at the local send site with `DeadIncarnation`, and NO frame
/// is emitted — asserted on a probe sender bound as that node's outbound.
#[test]
fn stale_incarnation_rejected_locally_no_frame() {
maybe_child(ROLES);
run(|| {
remote::set_local_identity("me", Incarnation::new(7));
let (probe_tx, probe_rx) = channel::<Frame>();
remote::bind_outbound_probe("peer", Incarnation::new(5), probe_tx);
let stale = RemotePid::<Replier>::from_parts("peer", Incarnation::new(4), 9, 1);
match send_to_remote(stale, Reply("late".into())) {
Err(ToRemoteError::DeadIncarnation(Reply(s))) => assert_eq!(s, "late"),
other => panic!("expected DeadIncarnation, got {other:?}"),
}
assert!(
matches!(probe_rx.try_recv(), Ok(None)),
"stale pid must emit no frame"
);
// The current incarnation goes through: a Send frame with the pid's
// (index, generation) and Reply's hash lands on the probe.
let live = RemotePid::<Replier>::from_parts("peer", Incarnation::new(5), 9, 1);
send_to_remote(live, Reply("now".into())).unwrap();
match probe_rx.recv().unwrap() {
Frame::Send {
index,
generation,
type_hash: h,
payload,
} => {
assert_eq!((index, generation), (9, 1));
assert_eq!(h, type_hash::<Reply>());
let r: Reply = smarm::cluster::envelope::decode_payload(&payload).unwrap();
assert_eq!(r, Reply("now".into()));
}
f => panic!("expected Send, got {f:?}"),
}
// Unknown node: NotConnected, no frame anywhere.
let nowhere = RemotePid::<Replier>::from_parts("nowhere", Incarnation::new(1), 1, 1);
assert!(matches!(
send_to_remote(nowhere, Reply("x".into())),
Err(ToRemoteError::NotConnected(_))
));
});
}
// ================= cross-process gate ==================================
const ROLES: &[(&str, fn())] = &[
("server", role_server),
("client", role_client),
("relay", role_relay),
];
const ECHO: Name<Req> = Name::new("c10.echo");
const RELAY: Name<Req> = Name::new("c10.relay");
fn cfg(name: &str, seeds: Vec<(String, String)>) -> Config {
Config {
node_name: name.to_string(),
meta: NodeMeta {
role: "c10".into(),
region: "local".into(),
},
listen_addr: std::env::var("SMARM_LISTEN_ADDR").unwrap_or_else(|_| "127.0.0.1:0".into()),
strategy: Box::new(StaticSeeds::new(seeds)),
timing: Timing::default(),
}
}
fn wait_up(events: &smarm::cluster::membership::MembershipEvents, who: &str) {
loop {
match events.rx.recv() {
Ok(NodeEvent::NodeUp(i)) if i.name == who => return,
Ok(_) => continue,
Err(_) => panic!("manager gone"),
}
}
}
/// Server: exposes ECHO; each Req is answered by `send_to_remote` to its
/// reply_to — the server never learns a name for the client. If the Req text
/// starts with "via-relay:", it forwards the whole Req (reply_to and all) to
/// the relay node instead, which sends it back here; the second arrival is
/// answered normally. That is the pid's third-node roundtrip.
fn role_server() {
let relay_addr = std::env::var("SMARM_RELAY_ADDR").ok();
smarm::run(move || {
let seeds = relay_addr
.map(|a| vec![("relay".to_string(), a)])
.unwrap_or_default();
let cluster = start(cfg("server", seeds)).expect("binds");
println!("LISTENING {}", cluster.local_addr());
let (tx, rx) = channel::<Req>();
register(ECHO, tx).unwrap();
expose(ECHO);
println!("READY");
loop {
let req = rx.recv().unwrap();
if let Some(rest) = req.text.strip_prefix("via-relay:") {
let fwd = Req {
text: format!("relayed:{rest}"),
reply_to: req.reply_to,
};
remote::send(RemoteName::new("relay", RELAY), fwd).unwrap();
println!("FORWARDED");
continue;
}
println!("REQ {}", req.text);
send_to_remote(req.reply_to, Reply(format!("echo:{}", req.text))).unwrap();
}
});
}
/// Relay: exposes RELAY; bounces every Req straight back to the server's
/// ECHO, untouched. The client's pid inside it now crosses relay→server.
fn role_relay() {
let server_addr = std::env::var("SMARM_SERVER_ADDR").expect("SMARM_SERVER_ADDR");
smarm::run(move || {
let cluster = start(cfg("relay", vec![("server".into(), server_addr)])).expect("binds");
println!("LISTENING {}", cluster.local_addr());
let (tx, rx) = channel::<Req>();
register(RELAY, tx).unwrap();
expose(RELAY);
let ev = subscribe().unwrap();
wait_up(&ev, "server");
println!("READY");
loop {
let req = rx.recv().unwrap();
println!("RELAYING {}", req.text);
remote::send(RemoteName::new("server", ECHO), req).unwrap();
}
});
}
/// Client: connects to server, installs a Reply inbox on its own pid,
/// declares it accepts `Reply` (`expose_type` — the RFC's one kept piece of
/// ceremony: nothing is remotely deliverable by default), sends a Req with
/// `reply_to = my pid` (auto-serialized), awaits the reply.
fn role_client() {
let server_addr = std::env::var("SMARM_SERVER_ADDR").expect("SMARM_SERVER_ADDR");
let via_relay = std::env::var("SMARM_VIA_RELAY").is_ok();
smarm::run(move || {
let _cluster = start(cfg("client", vec![("server".into(), server_addr)])).expect("binds");
let ev = subscribe().unwrap();
wait_up(&ev, "server");
println!("MEMBER-UP server");
let (tx, rx) = channel::<Reply>();
let me: Pid<Replier> = install::<Replier>(tx);
// The one deliberate line: a pid-targeted inbound is deliverable only
// for types this node has said it accepts (RFC §4, the safety).
smarm::cluster::expose::expose_type::<Reply>();
let text = if via_relay { "via-relay:ping" } else { "ping" };
remote::send(
RemoteName::new("server", ECHO),
Req {
text: text.into(),
reply_to: RemotePid::from_local(me).expect("identity set"),
},
)
.unwrap();
println!("SENT");
let Reply(s) = rx.recv().unwrap();
println!("REPLY {s}");
loop {
smarm::sleep(Duration::from_secs(3600));
}
});
}
/// The gate: cross-node call/reply with no ceremony.
#[test]
fn cross_node_call_reply_no_ceremony() {
maybe_child(ROLES);
let mut server = spawn_node("server", &[]);
let saddr = server.wait_listening();
server.wait_line("READY", |l| l == "READY");
let mut client = spawn_node("client", &[("SMARM_SERVER_ADDR", &saddr)]);
client.wait_line("SENT", |l| l == "SENT");
server.wait_line("REQ ping", |l| l == "REQ ping");
client.wait_line("REPLY echo:ping", |l| l == "REPLY echo:ping");
}
/// The client's pid, round-tripped through a third node, still delivers.
#[test]
fn pid_roundtrips_through_third_node() {
maybe_child(ROLES);
// Relay needs the server address; server needs the relay address —
// pre-reserve the relay port (same accepted micro-window as cluster_mesh).
let relay_addr = {
let l = std::net::TcpListener::bind("127.0.0.1:0").unwrap();
l.local_addr().unwrap().to_string()
};
let mut server = spawn_node("server", &[("SMARM_RELAY_ADDR", &relay_addr)]);
let saddr = server.wait_listening();
server.wait_line("READY", |l| l == "READY");
let mut relay = spawn_node(
"relay",
&[
("SMARM_SERVER_ADDR", &saddr),
("SMARM_LISTEN_ADDR", &relay_addr),
],
);
let _ = relay.wait_listening();
relay.wait_line("READY", |l| l == "READY");
let mut client = spawn_node(
"client",
&[("SMARM_SERVER_ADDR", &saddr), ("SMARM_VIA_RELAY", "1")],
);
client.wait_line("SENT", |l| l == "SENT");
server.wait_line("FORWARDED", |l| l == "FORWARDED");
relay.wait_line("RELAYING", |l| l.starts_with("RELAYING"));
server.wait_line("REQ relayed:ping", |l| l == "REQ relayed:ping");
client.wait_line("REPLY echo:relayed:ping", |l| {
l == "REPLY echo:relayed:ping"
});
}
+221
View File
@@ -0,0 +1,221 @@
//! RFC 010 c9 — remote `Name` sends: the outbound seam and the single
//! inbound name-resolution seam, cross-process.
//!
//! Two node processes each run the integrated `cluster::start`. The
//! *receiver* registers a `String` inbox under a name and exposes it (and
//! registers a second name it does NOT expose); the *sender* waits for
//! `node_up`, then sends. Facts cross as stdout lines: `LISTENING <addr>`,
//! `MEMBER-UP <name>`, `GOT <payload>`, `SEND-RESULT <case> <verdict>`.
//! Roles park forever afterwards (retractable-state trap); the parent
//! SIGKILLs via `Drop`.
//!
//! What is asserted at each end (roadmap-binding):
//! - cross-node name-send delivers the payload;
//! - an unexposed name is unreachable — the receiver's inbox stays empty
//! even though the name IS registered locally;
//! - a wrong type hash is a decode failure at the receiver, never a
//! misroute — the `String` inbox does not see a `u64` delivered under a
//! made-up hash, nor a `u64` under `u64`'s hash;
//! - a send to a disconnected (never-connected) node fails locally with
//! `NotConnected`, and `Ok(())` means only "handed to the transport".
//!
//! Timing note for the "stays empty" assertions: they are proven by
//! ORDERING, not by waiting — the sender emits the negative-case frames
//! BEFORE the positive one on the same connection (in-order stream), so when
//! the receiver has seen the positive payload, the negatives have already
//! been processed and refused. No sleep-and-hope.
#![cfg(feature = "cluster")]
mod common;
use common::{maybe_child, spawn_node, Node};
use smarm::cluster::envelope::NodeMeta;
use smarm::cluster::expose::expose;
use smarm::cluster::membership::{subscribe, NodeEvent};
use smarm::cluster::remote::{send_remote_raw, RemoteName, RemoteSendError};
use smarm::cluster::{start, Config, StaticSeeds, Timing};
use smarm::{channel, register, Name};
use std::time::Duration;
const ROLES: &[(&str, fn())] = &[("receiver", role_receiver), ("sender", role_sender)];
const INBOX: Name<String> = Name::new("c9.inbox");
const HIDDEN: Name<String> = Name::new("c9.hidden");
fn base_config(name: &str, seeds: Vec<(String, String)>) -> Config {
Config {
node_name: name.to_string(),
meta: NodeMeta {
role: "c9".to_string(),
region: "local".to_string(),
},
listen_addr: "127.0.0.1:0".to_string(),
strategy: Box::new(StaticSeeds::new(seeds)),
timing: Timing::default(),
}
}
/// Receiver: register + expose INBOX; register HIDDEN unexposed **in a
/// separate actor** (one actor holds one channel per message type — a
/// second `register` of the same `M` on one actor silently replaces the
/// first, closing it); print every payload that lands in either.
fn role_receiver() {
smarm::run(|| {
let cluster = start(base_config("recv", vec![])).expect("listener binds");
println!("LISTENING {}", cluster.local_addr());
// HIDDEN's holder: its own actor, so its String channel does not
// displace INBOX's on the root actor.
let (hidden_ready_tx, hidden_ready_rx) = channel::<()>();
smarm::spawn(move || {
let (hid_tx, hid_rx) = channel::<String>();
register(HIDDEN, hid_tx).unwrap();
hidden_ready_tx.send(()).unwrap();
loop {
match hid_rx.recv() {
Ok(s) => println!("GOT-HIDDEN {s}"),
Err(_) => break,
}
}
});
hidden_ready_rx.recv().unwrap();
let (in_tx, in_rx) = channel::<String>();
register(INBOX, in_tx).unwrap();
expose(INBOX);
println!("READY");
loop {
match in_rx.recv() {
Ok(s) => println!("GOT {s}"),
Err(_) => break,
}
}
loop {
smarm::sleep(Duration::from_secs(3600));
}
});
}
/// Sender: connect to recv, wait for node_up, then in this ORDER on the one
/// connection: hidden-name send, wrong-hash sends (two flavours), then the
/// positive send. Plus a send to a node that is not connected at all.
fn role_sender() {
let recv_addr = std::env::var("SMARM_RECV_ADDR").expect("SMARM_RECV_ADDR");
smarm::run(move || {
let _cluster = start(base_config("send", vec![("recv".to_string(), recv_addr)]))
.expect("listener binds");
let events = subscribe().expect("manager is up");
loop {
match events.rx.recv() {
Ok(NodeEvent::NodeUp(info)) if info.name == "recv" => break,
Ok(_) => continue,
Err(_) => panic!("manager gone"),
}
}
println!("MEMBER-UP recv");
// Not connected: purely local knowledge, no frame leaves.
let ghost: RemoteName<String> = RemoteName::new("nowhere", INBOX);
let r = smarm::cluster::remote::send(ghost, "lost".to_string());
println!(
"SEND-RESULT not-connected {}",
match r {
Err(RemoteSendError::NotConnected(_)) => "NotConnected",
Ok(()) => "Ok",
Err(_) => "OtherErr",
}
);
// Unexposed name at the peer: the frame goes (local knowledge can't
// know the peer's exposed set) and the peer refuses it.
let hidden: RemoteName<String> = RemoteName::new("recv", HIDDEN);
let r = smarm::cluster::remote::send(hidden, "should not land".to_string());
println!(
"SEND-RESULT hidden {}",
if r.is_ok() { "Ok" } else { "Err" }
);
// Wrong hash, two flavours: (a) a u64 payload under a made-up hash
// (unknown type at the peer); (b) a u64 payload under u64's real
// hash against a String-typed name (decoder known, wrong channel).
// Both are raw sends — the typed API cannot express them, by design.
let bogus = 0xdead_beef_u64;
let r = send_remote_raw(
"recv",
"c9.inbox",
bogus,
&smarm::cluster::envelope::encode_payload(&7u64).unwrap(),
);
println!(
"SEND-RESULT wrong-hash-unknown {}",
if r.is_ok() { "Ok" } else { "Err" }
);
let r = send_remote_raw(
"recv",
"c9.inbox",
smarm::cluster::expose::type_hash::<u64>(),
&smarm::cluster::envelope::encode_payload(&7u64).unwrap(),
);
println!(
"SEND-RESULT wrong-hash-known {}",
if r.is_ok() { "Ok" } else { "Err" }
);
// Positive: last on the stream, so its arrival proves the negatives
// were already processed.
let inbox: RemoteName<String> = RemoteName::new("recv", INBOX);
let r = smarm::cluster::remote::send(inbox, "hello from send".to_string());
println!(
"SEND-RESULT positive {}",
if r.is_ok() { "Ok" } else { "Err" }
);
loop {
smarm::sleep(Duration::from_secs(3600));
}
});
}
fn wait_send_result(node: &mut Node, case: &str) -> String {
let prefix = format!("SEND-RESULT {case} ");
let line = node.wait_line(&prefix, |l| l.starts_with(&prefix));
line[prefix.len()..].to_string()
}
#[test]
fn remote_name_send_delivers_and_refusals_never_misroute() {
maybe_child(ROLES);
let mut recv = spawn_node("receiver", &[]);
let addr = recv.wait_listening();
recv.wait_line("READY", |l| l == "READY");
let mut send = spawn_node("sender", &[("SMARM_RECV_ADDR", &addr)]);
send.wait_line("MEMBER-UP recv", |l| l == "MEMBER-UP recv");
// Local-knowledge-only failure for an unknown node.
assert_eq!(wait_send_result(&mut send, "not-connected"), "NotConnected");
// Every frame-bearing send is Ok — Ok means "handed to the transport",
// nothing about what the peer does with it (RFC §3, documented here).
assert_eq!(wait_send_result(&mut send, "hidden"), "Ok");
assert_eq!(wait_send_result(&mut send, "wrong-hash-unknown"), "Ok");
assert_eq!(wait_send_result(&mut send, "wrong-hash-known"), "Ok");
assert_eq!(wait_send_result(&mut send, "positive"), "Ok");
// The positive payload lands...
recv.wait_line("GOT hello from send", |l| l == "GOT hello from send");
// ...and, by stream ordering, every negative before it was refused: no
// GOT for the wrong-hash frames, no GOT-HIDDEN at all. The transcript
// up to this point is the proof.
let transcript = recv.transcript();
let gots: Vec<&str> = transcript
.iter()
.map(|s| s.as_str())
.filter(|l| l.starts_with("GOT"))
.collect();
assert_eq!(
gots,
["GOT hello from send"],
"exactly one delivery, the exposed one"
);
}
+272
View File
@@ -0,0 +1,272 @@
//! RFC 010 c3 — transport conformance suite, run against both shipped impls
//! (TCP and in-memory loopback), plus impl-specific cases.
//!
//! Shared suite (roadmap): frame roundtrips through the framed codec, framing
//! across a split write, coalesced frames in one write, peer-close mid-frame
//! (must error, not EOF), clean close at a frame boundary (EOF as `Ok(None)`).
//!
//! The TCP impl parks the calling actor, so its runs live inside `smarm::run`;
//! loopback blocks the OS thread and runs as plain tests.
#![cfg(feature = "cluster")]
use smarm::cluster::envelope::Frame;
use smarm::cluster::transport::loopback::LoopbackTransport;
use smarm::cluster::transport::tcp::TcpTransport;
use smarm::cluster::transport::{Conn, FramedConn, RecvError, Transport};
// ---------------------------------------------------------------------------
// Helpers
// ---------------------------------------------------------------------------
/// Listener + dial + accept against one transport, both conns returned.
/// Relies on dial not requiring a concurrent accept (TCP backlog / loopback
/// queue), so a single thread or actor can hold both ends.
fn pair(t: &dyn Transport, addr: &str) -> (Box<dyn Conn>, Box<dyn Conn>) {
let mut l = t.listen(addr).unwrap();
let a = t.dial(&l.local_addr()).unwrap();
let b = l.accept().unwrap();
(a, b)
}
fn frames() -> Vec<Frame> {
vec![
Frame::Heartbeat,
Frame::Send {
index: 42,
generation: 3,
type_hash: 0x1234_5678_9ABC_DEF0,
payload: vec![1, 2, 3, 4, 5],
},
Frame::SendNamed {
name: "the_counter".into(),
type_hash: 0xFFFF_0000_FFFF_0000,
payload: vec![],
},
Frame::Demonitor { monitor_id: 77 },
]
}
fn encode(f: &Frame) -> Vec<u8> {
let mut out = Vec::new();
f.encode(&mut out).unwrap();
out
}
// ---------------------------------------------------------------------------
// Shared conformance suite — generic over an established pair
// ---------------------------------------------------------------------------
fn suite_roundtrip(a: Box<dyn Conn>, b: Box<dyn Conn>) {
let mut fa = FramedConn::new(a);
let mut fb = FramedConn::new(b);
// a -> b, then b -> a: both directions carry every frame shape.
for f in frames() {
fa.send(&f).unwrap();
assert_eq!(fb.recv().unwrap().unwrap(), f);
}
for f in frames() {
fb.send(&f).unwrap();
assert_eq!(fa.recv().unwrap().unwrap(), f);
}
}
fn suite_split_write(mut a: Box<dyn Conn>, b: Box<dyn Conn>) {
let f = Frame::Send {
index: 7,
generation: 1,
type_hash: 0xAB,
payload: vec![9; 64],
};
let bytes = encode(&f);
// Split inside the length prefix, then inside the body: the reader must
// reassemble regardless of where the boundary falls.
a.write_all(&bytes[..2]).unwrap();
a.write_all(&bytes[2..10]).unwrap();
a.write_all(&bytes[10..]).unwrap();
let mut fb = FramedConn::new(b);
assert_eq!(fb.recv().unwrap().unwrap(), f);
}
fn suite_coalesced(mut a: Box<dyn Conn>, b: Box<dyn Conn>) {
let f1 = Frame::Heartbeat;
let f2 = Frame::Demonitor { monitor_id: 5 };
let mut bytes = encode(&f1);
bytes.extend_from_slice(&encode(&f2));
a.write_all(&bytes).unwrap();
let mut fb = FramedConn::new(b);
assert_eq!(fb.recv().unwrap().unwrap(), f1);
assert_eq!(fb.recv().unwrap().unwrap(), f2);
}
fn suite_close_mid_frame(mut a: Box<dyn Conn>, b: Box<dyn Conn>) {
let bytes = encode(&Frame::Send {
index: 1,
generation: 1,
type_hash: 1,
payload: vec![0; 128],
});
a.write_all(&bytes[..bytes.len() / 2]).unwrap();
a.close();
let mut fb = FramedConn::new(b);
match fb.recv() {
Err(RecvError::TruncatedByPeer) => {}
other => panic!("expected TruncatedByPeer, got {other:?}"),
}
}
fn suite_clean_close(mut a: Box<dyn Conn>, b: Box<dyn Conn>) {
let f = Frame::Heartbeat;
a.write_all(&encode(&f)).unwrap();
a.close();
let mut fb = FramedConn::new(b);
// The buffered frame is still delivered, then EOF at the boundary.
assert_eq!(fb.recv().unwrap().unwrap(), f);
assert!(fb.recv().unwrap().is_none());
}
fn run_suite(t: &dyn Transport, addr: &str) {
let (a, b) = pair(t, addr);
suite_roundtrip(a, b);
let (a, b) = pair(t, addr);
suite_split_write(a, b);
let (a, b) = pair(t, addr);
suite_coalesced(a, b);
let (a, b) = pair(t, addr);
suite_close_mid_frame(a, b);
let (a, b) = pair(t, addr);
suite_clean_close(a, b);
}
// ---------------------------------------------------------------------------
// Loopback — plain tests, no runtime
// ---------------------------------------------------------------------------
#[test]
fn loopback_conformance() {
// Fresh transport per pair() call is fine, but one instance must also
// support sequential re-listen on distinct addresses.
let t = LoopbackTransport::default();
run_suite(&t, "alpha");
}
#[test]
fn loopback_dial_unknown_addr_refused() {
let t = LoopbackTransport::default();
let err = t.dial("nobody-home").unwrap_err();
assert_eq!(err.kind(), std::io::ErrorKind::ConnectionRefused);
}
#[test]
fn loopback_addr_in_use() {
let t = LoopbackTransport::default();
let _l = t.listen("alpha").unwrap();
let err = t.listen("alpha").unwrap_err();
assert_eq!(err.kind(), std::io::ErrorKind::AddrInUse);
}
#[test]
fn loopback_listener_drop_frees_addr_and_refuses_dial() {
let t = LoopbackTransport::default();
let l = t.listen("alpha").unwrap();
drop(l);
let err = t.dial("alpha").unwrap_err();
assert_eq!(err.kind(), std::io::ErrorKind::ConnectionRefused);
// Address is reusable after the listener is gone.
let _l2 = t.listen("alpha").unwrap();
}
#[test]
fn loopback_write_after_peer_close_broken_pipe() {
let t = LoopbackTransport::default();
let (mut a, mut b) = pair(&t, "alpha");
b.close();
let err = a.write_all(&[1, 2, 3]).unwrap_err();
assert_eq!(err.kind(), std::io::ErrorKind::BrokenPipe);
}
#[test]
fn loopback_cross_thread_blocking_read() {
// Reader blocks on an empty pipe until the writer thread delivers.
let t = LoopbackTransport::default();
let (a, b) = pair(&t, "alpha");
let mut fb = FramedConn::new(b);
let writer = std::thread::spawn(move || {
let mut a = a;
std::thread::sleep(std::time::Duration::from_millis(30));
a.write_all(&encode(&Frame::Heartbeat)).unwrap();
});
assert_eq!(fb.recv().unwrap().unwrap(), Frame::Heartbeat);
writer.join().unwrap();
}
// ---------------------------------------------------------------------------
// TCP — inside the runtime (read/write park the calling actor)
// ---------------------------------------------------------------------------
#[test]
fn tcp_conformance() {
smarm::run(|| {
run_suite(&TcpTransport, "127.0.0.1:0");
});
}
#[test]
fn tcp_dial_refused() {
smarm::run(|| {
// Bind to an OS-assigned port, learn it, close the listener, dial it.
let addr = {
let l = TcpTransport.listen("127.0.0.1:0").unwrap();
l.local_addr()
};
let err = TcpTransport.dial(&addr).unwrap_err();
assert_eq!(err.kind(), std::io::ErrorKind::ConnectionRefused);
});
}
#[test]
fn tcp_bad_addr_rejected_without_resolution() {
// Addresses are opaque pre-resolved strings; the c9 seam resolves names.
// A hostname is therefore invalid input here, not something to resolve.
let err = TcpTransport.dial("localhost:1234").unwrap_err();
assert_eq!(err.kind(), std::io::ErrorKind::InvalidInput);
}
#[test]
fn tcp_local_addr_reports_real_port() {
let l = TcpTransport.listen("127.0.0.1:0").unwrap();
let addr = l.local_addr();
let port: u16 = addr.rsplit(':').next().unwrap().parse().unwrap();
assert_ne!(port, 0);
}
#[test]
fn tcp_big_frame_across_socket_buffers() {
// A payload far beyond socket buffer sizes forces genuine fragmentation
// and write backpressure: writer and reader must run concurrently.
smarm::run(|| {
let (tx, rx) = smarm::channel::<Frame>();
let payload = vec![0xA5u8; 4 * 1024 * 1024];
let f = Frame::Send {
index: 9,
generation: 2,
type_hash: 0xC0FFEE,
payload,
};
let mut l = TcpTransport.listen("127.0.0.1:0").unwrap();
let addr = l.local_addr();
let fw = f.clone();
let writer = smarm::spawn(move || {
let mut fa = FramedConn::new(TcpTransport.dial(&addr).unwrap());
fa.send(&fw).unwrap();
});
let reader = smarm::spawn(move || {
let mut fb = FramedConn::new(l.accept().unwrap());
let got = fb.recv().unwrap().unwrap();
tx.send(got).unwrap();
});
let got = rx.recv().unwrap();
assert_eq!(got, f);
writer.join().unwrap();
reader.join().unwrap();
});
}
+108
View File
@@ -0,0 +1,108 @@
//! RFC 010 c4 — two-node harness smoke tests.
//!
//! Roadmap: "spawn two, handshake-less connect, both exit clean." The
//! listener node binds port 0 and announces its concrete address; the
//! dialer connects raw (no Hello — c5 doesn't exist yet), pushes one
//! Heartbeat through the real framed codec, and closes. Assertions are on
//! protocol-visible lines only. Flake budget: see tests/common/mod.rs.
#![cfg(feature = "cluster")]
mod common;
use common::{maybe_child, spawn_node};
use smarm::cluster::envelope::Frame;
use smarm::cluster::transport::tcp::TcpTransport;
use smarm::cluster::transport::{FramedConn, Transport};
const ROLES: &[(&str, fn())] = &[
("listener", role_listener),
("dialer", role_dialer),
("hang", role_hang),
("fail", role_fail),
];
fn role_listener() {
smarm::run(|| {
let mut l = TcpTransport.listen("127.0.0.1:0").unwrap();
println!("LISTENING {}", l.local_addr());
let mut fc = FramedConn::new(l.accept().unwrap());
match fc.recv() {
Ok(Some(Frame::Heartbeat)) => println!("RECV heartbeat"),
other => {
println!("RECV unexpected: {other:?}");
std::process::exit(3);
}
}
match fc.recv() {
Ok(None) => println!("PEER-CLOSED clean"),
other => {
println!("PEER-CLOSED unexpected: {other:?}");
std::process::exit(3);
}
}
});
println!("EXIT ok");
}
fn role_dialer() {
let addr = std::env::var("SMARM_PEER_ADDR").expect("SMARM_PEER_ADDR not set");
smarm::run(move || {
let mut fc = FramedConn::new(TcpTransport.dial(&addr).unwrap());
fc.send(&Frame::Heartbeat).unwrap();
fc.close();
println!("SENT heartbeat");
});
println!("EXIT ok");
}
fn role_hang() {
println!("HANGING");
loop {
std::thread::sleep(std::time::Duration::from_secs(3600));
}
}
fn role_fail() {
std::process::exit(7);
}
/// The roadmap smoke test: two real processes, raw transport connect, one
/// frame across, clean close observed on both sides, both exit 0.
#[test]
fn two_nodes_connect_and_exit_clean() {
maybe_child(ROLES);
let mut listener = spawn_node("listener", &[]);
let addr = listener.wait_listening();
let mut dialer = spawn_node("dialer", &[("SMARM_PEER_ADDR", &addr)]);
dialer.wait_line("SENT heartbeat", |l| l == "SENT heartbeat");
listener.wait_line("RECV heartbeat", |l| l == "RECV heartbeat");
listener.wait_line("clean peer close", |l| l == "PEER-CLOSED clean");
dialer.wait_exit_ok();
listener.wait_exit_ok();
}
/// Reap guarantee: dropping a Node kills a hung child — no orphan survives
/// a panicking test.
#[test]
fn drop_reaps_hung_node() {
maybe_child(ROLES);
let mut node = spawn_node("hang", &[]);
node.wait_line("HANGING", |l| l == "HANGING");
let pid = node.pid().expect("live child has a pid") as libc::pid_t;
drop(node);
// After Drop's kill+wait the pid is fully reaped: signalling it fails
// with ESRCH (pid-reuse in this instant is not a realistic race).
let rc = unsafe { libc::kill(pid, 0) };
assert_eq!(rc, -1, "process still signallable after Drop");
let errno = std::io::Error::last_os_error().raw_os_error();
assert_eq!(errno, Some(libc::ESRCH), "expected ESRCH, got {errno:?}");
}
/// Nonzero child exits surface as statuses, not hangs or panics.
#[test]
fn nonzero_exit_is_reported() {
maybe_child(ROLES);
let mut node = spawn_node("fail", &[]);
let status = node.wait_exit();
assert_eq!(status.code(), Some(7));
}
+250
View File
@@ -0,0 +1,250 @@
//! RFC 010 c4 — subprocess multi-node test harness.
//!
//! The runtime is a process singleton, so two real nodes means two
//! processes. This harness re-execs the *current test binary* as node
//! processes (precedent: tests/stack_diag.rs), tails their output live,
//! waits on protocol-visible lines, and reaps reliably no matter how the
//! test dies.
//!
//! Usage, per test file:
//!
//! - Declare roles as plain `fn()`s. A role prints protocol-visible facts
//! as single lines (Rust's piped stdout is line-buffered, so `println!`
//! is enough) and exits.
//! - **Every** `#[test]` in the file starts with
//! [`maybe_child`]`(ROLES)` — in the child re-exec, whichever test
//! libtest runs first performs the role and exits before the rest of the
//! suite runs (children are spawned with `--test-threads=1 --quiet`).
//! - The parent side spawns nodes with [`spawn_node`], waits on lines with
//! [`Node::wait_line`], and on exits with [`Node::wait_exit`].
//!
//! Port assignment: children bind port 0 and *report* the concrete address
//! (e.g. `LISTENING 127.0.0.1:41733`) rather than the parent pre-picking a
//! port — no bind/steal race by construction.
//!
//! Reaping: [`Node`]'s `Drop` SIGKILLs and `wait(2)`s the child, so a
//! panicking test (including a `wait_line` timeout) leaves no orphan and
//! no zombie. Tail threads exit on pipe EOF.
//!
//! Flake budget (explicit, per roadmap): every wait is bounded by
//! [`WAIT`] (10 s) against a typical cost of well under 1 s; the smoke
//! suite ran 10/10 clean at authoring time. Treat >1 failure in 100 runs
//! as a harness or runtime regression, not weather. On timeout the panic
//! message carries the node's full transcript so far.
#![allow(dead_code)] // Reusable surface: later phases use more of it than any one file.
use std::env;
use std::io::{BufRead, BufReader};
use std::process::{Child, Command, ExitStatus, Stdio};
use std::sync::mpsc::{Receiver, RecvTimeoutError};
use std::time::{Duration, Instant};
/// Env var selecting the child role in a re-exec.
const ROLE_ENV: &str = "SMARM_TWO_NODE_ROLE";
/// Upper bound for every wait in the harness. See the flake budget above.
pub const WAIT: Duration = Duration::from_secs(10);
/// In the child re-exec: run the matching role and exit. In the parent (no
/// role env set): return immediately. Call this first in every `#[test]` of
/// any file using the harness, passing the file's full role table.
pub fn maybe_child(roles: &[(&str, fn())]) {
let role = match env::var(ROLE_ENV) {
Ok(r) => r,
Err(_) => return,
};
for (name, f) in roles {
if *name == role {
f();
std::process::exit(0);
}
}
eprintln!("two_node harness: unknown role {role:?}");
std::process::exit(2);
}
/// One spawned node process with live-tailed output.
pub struct Node {
/// Role name, for panic messages.
pub role: String,
child: Option<Child>,
stdout_rx: Receiver<String>,
stderr_rx: Receiver<String>,
/// Every line consumed from stdout/stderr so far, for failure dumps.
transcript: Vec<String>,
}
fn tail(stream: impl std::io::Read + Send + 'static, prefix: &'static str) -> Receiver<String> {
let (tx, rx) = std::sync::mpsc::channel();
std::thread::spawn(move || {
for line in BufReader::new(stream).lines() {
let line = match line {
Ok(l) => l,
Err(_) => break,
};
// Receiver gone (Node dropped): stop tailing.
if tx.send(format!("{prefix}{line}")).is_err() {
break;
}
}
});
rx
}
/// Re-exec the current test binary as `role`, with any extra env vars.
pub fn spawn_node(role: &str, extra_env: &[(&str, &str)]) -> Node {
let exe = env::current_exe().expect("current_exe");
let mut cmd = Command::new(exe);
cmd.env(ROLE_ENV, role)
// --test-threads=1: exactly one test fn starts, hits maybe_child,
// and becomes the role. --nocapture: libtest must not swallow the
// role's println! lines — the parent tails them live.
.args(["--test-threads=1", "--quiet", "--nocapture"])
.stdout(Stdio::piped())
.stderr(Stdio::piped());
for (k, v) in extra_env {
cmd.env(k, v);
}
let mut child = cmd.spawn().expect("failed to spawn node process");
let stdout_rx = tail(child.stdout.take().expect("piped stdout"), "");
let stderr_rx = tail(child.stderr.take().expect("piped stderr"), "[stderr] ");
Node {
role: role.to_string(),
child: Some(child),
stdout_rx,
stderr_rx,
transcript: Vec::new(),
}
}
impl Node {
fn drain_stderr(&mut self) {
while let Ok(l) = self.stderr_rx.try_recv() {
self.transcript.push(l);
}
}
fn dump(&self) -> String {
if self.transcript.is_empty() {
"<no output>".to_string()
} else {
self.transcript.join("\n")
}
}
/// Every stdout/stderr line seen so far, in arrival order. For
/// ordering-proof assertions ("by the time X arrived, Y had not").
#[allow(dead_code)]
pub fn transcript(&self) -> &[String] {
&self.transcript
}
/// Wait until a stdout line satisfies `pred`; return it. Panics with the
/// full transcript after [`WAIT`]. `what` names the expectation in the
/// panic message.
pub fn wait_line(&mut self, what: &str, pred: impl Fn(&str) -> bool) -> String {
let deadline = Instant::now() + WAIT;
loop {
self.drain_stderr();
let left = deadline.saturating_duration_since(Instant::now());
match self.stdout_rx.recv_timeout(left) {
Ok(line) => {
self.transcript.push(line.clone());
if pred(&line) {
return line;
}
}
Err(RecvTimeoutError::Timeout) => {
// Pull in whatever stderr arrived since the last drain,
// so a role's eprintln! diagnostics survive into the dump.
self.drain_stderr();
panic!(
"node {:?}: timed out waiting for {what} after {WAIT:?}; transcript:\n{}",
self.role,
self.dump()
);
}
Err(RecvTimeoutError::Disconnected) => {
self.drain_stderr();
panic!(
"node {:?}: output closed while waiting for {what}; transcript:\n{}",
self.role,
self.dump()
);
}
}
}
}
/// Shorthand: wait for a `LISTENING <addr>` announcement, return the addr.
pub fn wait_listening(&mut self) -> String {
let line = self.wait_line("LISTENING announcement", |l| l.starts_with("LISTENING "));
line["LISTENING ".len()..].to_string()
}
/// Wait for the process to exit; panics with the transcript on timeout.
pub fn wait_exit(&mut self) -> ExitStatus {
let deadline = Instant::now() + WAIT;
loop {
let polled = match self.child.as_mut() {
Some(c) => c.try_wait(),
None => panic!("node {:?}: already reaped", self.role),
};
match polled {
Ok(Some(status)) => {
// Drain remaining output into the transcript for dumps.
self.drain_stderr();
while let Ok(l) = self.stdout_rx.try_recv() {
self.transcript.push(l);
}
self.child = None;
return status;
}
Ok(None) => {
if Instant::now() >= deadline {
self.drain_stderr();
self.kill();
panic!(
"node {:?}: did not exit within {WAIT:?}; transcript:\n{}",
self.role,
self.dump()
);
}
std::thread::sleep(Duration::from_millis(10));
}
Err(e) => panic!("node {:?}: try_wait failed: {e}", self.role),
}
}
}
/// Wait for exit and require success, dumping the transcript otherwise.
pub fn wait_exit_ok(&mut self) {
let status = self.wait_exit();
assert!(
status.success(),
"node {:?}: exited with {status}; transcript:\n{}",
self.role,
self.dump()
);
}
/// The child's OS pid, if not yet reaped.
pub fn pid(&self) -> Option<u32> {
self.child.as_ref().map(Child::id)
}
/// SIGKILL + reap now (idempotent).
pub fn kill(&mut self) {
if let Some(mut child) = self.child.take() {
let _ = child.kill();
let _ = child.wait();
}
}
}
impl Drop for Node {
fn drop(&mut self) {
self.kill();
}
}
+13 -20
View File
@@ -1,5 +1,6 @@
//! Process-group tests that run under the scheduler: `join` installs a real
//! monitor on a live actor, and a real death drives eviction on next contact.
//! monitor on a live actor, and a real death drives eviction (the reaper
//! actor sweeps it; the read path hides it in the meantime).
//! (Pure structural invariants live in the `pg` unit tests.)
use smarm::{channel, members, pick, run, spawn};
@@ -35,19 +36,15 @@ fn a_dead_actor_vanishes_from_every_group_it_joined() {
assert_eq!(members("g1"), vec![pid]);
assert_eq!(members("g2"), vec![pid]);
// Release and reap the actor. finalize_actor queues the Down to our
// monitors before unparking joiners, so by the time join() returns the
// Down is already waiting in the membership channel.
// Release the actor. finalize_actor queues the Down to the reaper and
// marks the slot dead before unparking joiners, so by the time join()
// returns every read hides the pid whether or not the reaper has run.
tx.send(()).unwrap();
h.join().unwrap();
// Drain-on-contact: touching g1 detects the death and sweeps the pid
// out of every group (g2 included), not just g1.
assert!(members("g1").is_empty(), "evicted from the touched group");
assert!(
members("g2").is_empty(),
"and swept from the untouched group"
);
// Gone from every group it joined, not just one.
assert!(members("g1").is_empty(), "gone from g1");
assert!(members("g2").is_empty(), "and from g2");
assert_eq!(pick("g1"), None);
});
}
@@ -86,11 +83,7 @@ fn live_members_survive_a_peers_death() {
tx_a.send(()).unwrap();
a.join().unwrap();
assert_eq!(
members("svc"),
vec![b.pid()],
"only the dead peer is reaped"
);
assert_eq!(members("svc"), vec![b.pid()], "only the dead peer is gone");
assert_eq!(pick("svc"), Some(b.pid()));
tx_b.send(()).unwrap();
@@ -123,18 +116,18 @@ fn leave_drops_a_membership_without_affecting_others() {
}
#[test]
fn joining_an_already_dead_pid_is_evicted_on_next_contact() {
fn joining_an_already_dead_pid_never_shows_in_a_read() {
run(|| {
let h = spawn(|| {});
let pid = h.pid();
h.join().unwrap(); // actor is finalized before we join it to anything
// monitor() on a gone pid queues a NoProc Down immediately, so the
// membership is reaped the next time the group is touched.
// join() on a gone pid queues a NoProc Down to the reaper immediately;
// reads never show it either way (slot-liveness backstop).
join("late", pid);
assert!(
members("late").is_empty(),
"dead-at-join member is reaped on read"
"dead-at-join member never reads as live"
);
assert_eq!(pick("late"), None);
});
+43
View File
@@ -258,3 +258,46 @@ fn send_dyn_to_dead_pid_is_dead() {
assert!(matches!(send_dyn::<u64>(p, 1u64), Err(SendError::Dead(_))));
});
}
// --- one channel per message type per actor -----------------------------------
/// Registering a second name of the same message type on one actor, with a
/// *fresh* channel, would silently replace and close the first — so it
/// panics (found in RFC 010 c9). The sanctioned shapes stay quiet: bind both
/// names to a clone of one sender, or use two actors.
#[test]
#[should_panic(expected = "already publishes a live channel")]
fn second_live_channel_of_same_type_on_one_actor_panics() {
run(|| {
let (tx1, _rx1) = channel::<u64>();
let (tx2, _rx2) = channel::<u64>();
register(Name::<u64>::new("dup-a"), tx1).unwrap();
register(Name::<u64>::new("dup-b"), tx2).unwrap(); // panics
});
}
#[test]
fn two_names_on_one_cloned_sender_is_fine() {
run(|| {
let (tx, rx) = channel::<u64>();
register(Name::<u64>::new("twin-a"), tx.clone()).unwrap();
register(Name::<u64>::new("twin-b"), tx).unwrap();
send(Name::<u64>::new("twin-a"), 1).unwrap();
send(Name::<u64>::new("twin-b"), 2).unwrap();
assert_eq!(rx.recv().unwrap(), 1);
assert_eq!(rx.recv().unwrap(), 2);
});
}
#[test]
fn replacing_a_channel_whose_receiver_is_gone_is_fine() {
run(|| {
let (tx1, rx1) = channel::<u64>();
register(Name::<u64>::new("reborn"), tx1).unwrap();
drop(rx1); // old inbox gone: replacement is the honest thing to do
let (tx2, rx2) = channel::<u64>();
register(Name::<u64>::new("reborn-2"), tx2).unwrap();
send(Name::<u64>::new("reborn-2"), 9).unwrap();
assert_eq!(rx2.recv().unwrap(), 9);
});
}