feat(cluster): RFC 010 c10–c16, follow-ups and Phase 6 (squash of 16ef583..d9c62a8)
Tree snapshot of d9c62a8 (2026-08-18). The 20 source commits between
16ef583 (c9) and d9c62a8 were never pushed and the clone that held them
was lost; this commit carries their combined tree verbatim so the build
history stays auditable from the c1–c9 commits below it. Original
hashes as recorded in the session handoff:
c10 f03e94d pid targeting + auto-serialization (RemotePid, D14 name
on the wire); Phase 3 gate
c11 7ef4bad DownReason::Disconnected, wire tag 5
c12 d124162 remote monitors (Monitor/Demonitor/Down frames)
c13 9de967b connection-loss synthesis (A+B: Monitors::teardown +
unread-command Disconnected); Phase 4 gate
c14 7e822b7 eager pg eviction (reaper actor, ReaperInboxes)
dbe1a22 InboundVerdict::label(), trace::Event::ClusterInbound
31a9877 tests/channel.rs monitor-churn target gated on `go`
653559e Discovery::Withdrawn{name, addr}
c15 b41d76e distributed pg: Sync on NodeUp, Join/Leave broadcast,
NodeDown sweep, members_all; PgMsg wire type
c16 fafa881 pick_any / dispatch_any; Phase 5 complete
Phase 6 Tier A:
195c73e p4 NodeEvent::NodeDown(NodeInfo)
48fd766 p1 connector Candidate{name, addr, state}
ce8cf99 p2+p7 conn.rs select arms as Vec<Arm>; Outbound::Drained
bf24988 p6 RemotePid::from_local -> Option
9ae0380 p3 PeerStanding{Free, Claimed, Dialing}
c7d62a1 p11 cluster::Timing knobs, threaded by value
46f171d p11 cluster_disconnect un-ignored on SMARM_FAST_TIMING
Phase 6 Tier B:
391a9ae p5 cluster::RemoteDownReason{Local, Disconnected};
DownReason::Disconnected removed from core
7ddd908 p9 pg ctl channel unconditional, one cfg seam at spawn
d9c62a8 PeerNameMismatch parks the candidate; ClusterDial trace
Verified at d9c62a8: default 361/0, cluster 448/0, clippy --lib on
default / cluster / cluster+smarm-trace, fmt, 10x flake on
cluster_dial_mismatch, 5x on cluster_pg.
This commit is contained in:
+8
-1
@@ -137,11 +137,18 @@ fn channel_ops_interleaved_with_monitor_churn_multi_thread() {
|
||||
for i in 0..32i64 {
|
||||
let tx = tx.clone();
|
||||
handles.push(spawn(move || {
|
||||
// Short-lived target whose death fires the monitor below.
|
||||
// Short-lived target whose death fires the monitor below. It
|
||||
// is gated: on a multi-thread scheduler it could otherwise
|
||||
// run and exit before `monitor` registers, and monitoring a
|
||||
// corpse queues `NoProc` by contract — the point here is a
|
||||
// `Down` sent from finalize, so register first, then release.
|
||||
let (go_tx, go_rx) = channel::<()>();
|
||||
let t = spawn(move || {
|
||||
let _ = go_rx.recv();
|
||||
tx.send(i).unwrap();
|
||||
});
|
||||
let m = smarm::monitor(t.pid());
|
||||
go_tx.send(()).unwrap();
|
||||
t.join().unwrap();
|
||||
// Down delivery exercises send-from-finalize.
|
||||
let d = m.rx.recv().unwrap();
|
||||
|
||||
@@ -21,6 +21,7 @@ use smarm::cluster::manager::{Call, Manager, Reply, MANAGER};
|
||||
use smarm::cluster::spawn_established;
|
||||
use smarm::cluster::transport::tcp::TcpTransport;
|
||||
use smarm::cluster::transport::{Conn, FramedConn, Transport};
|
||||
use smarm::cluster::Timing;
|
||||
use smarm::gen_server::{self, GenServerBuilder};
|
||||
use smarm::pg::Incarnation;
|
||||
use smarm::{run, sleep};
|
||||
@@ -81,8 +82,10 @@ fn connection_up_commanded_shutdown_and_eof_all_reflected_in_table() {
|
||||
|
||||
// Manage the `a` ends as peers node-b and node-c; keep the `b` far ends
|
||||
// open so neither socket is closed from the far side yet.
|
||||
spawn_established(FramedConn::new(a1), peer("node-b")).expect("node-b registers");
|
||||
spawn_established(FramedConn::new(a2), peer("node-c")).expect("node-c registers");
|
||||
spawn_established(FramedConn::new(a1), peer("node-b"), Timing::default())
|
||||
.expect("node-b registers");
|
||||
spawn_established(FramedConn::new(a2), peer("node-c"), Timing::default())
|
||||
.expect("node-c registers");
|
||||
|
||||
// Up: both connections register and the table shows them.
|
||||
wait_peers(&["node-b", "node-c"]);
|
||||
|
||||
@@ -22,6 +22,7 @@ use smarm::cluster::manager::{Call, Manager, Reply, MANAGER};
|
||||
use smarm::cluster::spawn_established;
|
||||
use smarm::cluster::transport::tcp::TcpTransport;
|
||||
use smarm::cluster::transport::{Conn, FramedConn, Transport};
|
||||
use smarm::cluster::Timing;
|
||||
use smarm::gen_server::{self, GenServerBuilder};
|
||||
use smarm::pg::Incarnation;
|
||||
use smarm::{run, sleep, spawn};
|
||||
@@ -83,7 +84,8 @@ fn heartbeats_are_sent_unprompted() {
|
||||
.expect("manager name is free");
|
||||
|
||||
let (a, b) = pair(&TcpTransport);
|
||||
spawn_established(FramedConn::new(a), peer("hb-send")).expect("register");
|
||||
spawn_established(FramedConn::new(a), peer("hb-send"), Timing::default())
|
||||
.expect("register");
|
||||
let mut far = FramedConn::new(b);
|
||||
|
||||
let frame = far
|
||||
@@ -111,7 +113,7 @@ fn mute_peer_is_torn_down_after_liveness_timeout() {
|
||||
.expect("manager name is free");
|
||||
|
||||
let (a, b) = pair(&TcpTransport);
|
||||
spawn_established(FramedConn::new(a), peer("mute")).expect("register");
|
||||
spawn_established(FramedConn::new(a), peer("mute"), Timing::default()).expect("register");
|
||||
// Held open and silent: no frames, no EOF. (Unread inbound
|
||||
// heartbeats sit in kernel buffers; they are 5 bytes each.)
|
||||
let _far = FramedConn::new(b);
|
||||
@@ -139,7 +141,7 @@ fn heartbeats_keep_the_connection_alive() {
|
||||
.expect("manager name is free");
|
||||
|
||||
let (a, b) = pair(&TcpTransport);
|
||||
spawn_established(FramedConn::new(a), peer("kept")).expect("register");
|
||||
spawn_established(FramedConn::new(a), peer("kept"), Timing::default()).expect("register");
|
||||
|
||||
// The far heartbeat pump: interval-paced sends until told to stop,
|
||||
// then holds the socket open, silent, so the eventual teardown is
|
||||
|
||||
+28
-18
@@ -19,11 +19,12 @@ use smarm::cluster::connect::{
|
||||
HANDSHAKE_TIMEOUT,
|
||||
};
|
||||
use smarm::cluster::envelope::{Frame, NodeMeta, RejectReason};
|
||||
use smarm::cluster::handshake::{HelloCtx, Local};
|
||||
use smarm::cluster::handshake::{Local, PeerStanding};
|
||||
use smarm::cluster::manager::{Call, Manager, Reply, MANAGER};
|
||||
use smarm::cluster::transport::loopback::LoopbackTransport;
|
||||
use smarm::cluster::transport::tcp::TcpTransport;
|
||||
use smarm::cluster::transport::{FramedConn, Transport};
|
||||
use smarm::cluster::Timing;
|
||||
use smarm::gen_server::{self, GenServerBuilder};
|
||||
use smarm::pg::Incarnation;
|
||||
use smarm::{run, sleep};
|
||||
@@ -100,7 +101,7 @@ fn loopback_happy_path_establishes_both_ends() {
|
||||
local("node-b"),
|
||||
|name| {
|
||||
assert_eq!(name, "node-a");
|
||||
HelloCtx::default()
|
||||
PeerStanding::Free
|
||||
},
|
||||
no_deadline(),
|
||||
)
|
||||
@@ -118,7 +119,7 @@ fn loopback_hash_mismatch_rejected_with_frame_then_eof() {
|
||||
let mut wrong = local("node-b");
|
||||
wrong.build_hash ^= 1;
|
||||
let responder = std::thread::spawn(move || {
|
||||
accept_handshake(&mut accepted, wrong, |_| HelloCtx::default(), no_deadline())
|
||||
accept_handshake(&mut accepted, wrong, |_| PeerStanding::Free, no_deadline())
|
||||
});
|
||||
// The dial side receives the reject frame — the compatibility anchor.
|
||||
match dial_handshake(&mut dialer, &local("node-a"), no_deadline()) {
|
||||
@@ -142,10 +143,7 @@ fn loopback_tie_break_loser_closed_silently() {
|
||||
accept_handshake(
|
||||
&mut accepted,
|
||||
local("node-a"),
|
||||
|_| HelloCtx {
|
||||
name_claimed: false,
|
||||
dialing_this_peer: true,
|
||||
},
|
||||
|_| PeerStanding::Dialing,
|
||||
no_deadline(),
|
||||
)
|
||||
});
|
||||
@@ -178,7 +176,7 @@ fn loopback_read_ahead_past_hello_survives_into_established_conn() {
|
||||
let peer = accept_handshake(
|
||||
&mut accepted,
|
||||
local("node-b"),
|
||||
|_| HelloCtx::default(),
|
||||
|_| PeerStanding::Free,
|
||||
no_deadline(),
|
||||
)
|
||||
.unwrap();
|
||||
@@ -216,7 +214,7 @@ fn tcp_silent_peer_times_out_on_the_accept_path() {
|
||||
let r = accept_handshake(
|
||||
&mut accepted,
|
||||
local("node-b"),
|
||||
|_| HelloCtx::default(),
|
||||
|_| PeerStanding::Free,
|
||||
Instant::now() + Duration::from_millis(200),
|
||||
);
|
||||
let _ = tx.send(r);
|
||||
@@ -253,7 +251,7 @@ fn tcp_duplicate_name_rejected_by_acceptor() {
|
||||
.start()
|
||||
.expect("manager name is free");
|
||||
let listener = TcpTransport.listen("127.0.0.1:0").unwrap();
|
||||
let acceptor = spawn_acceptor(listener, local("node-b"));
|
||||
let acceptor = spawn_acceptor(listener, local("node-b"), Timing::default());
|
||||
let addr = acceptor.local_addr().to_string();
|
||||
|
||||
// First dial offering "dup-node": establishes and registers.
|
||||
@@ -328,12 +326,12 @@ fn dial_intent_cleared_when_dialer_dies() {
|
||||
// While the dialer lives, the intent is visible.
|
||||
match gen_server::call(
|
||||
MANAGER,
|
||||
Call::HelloCtx {
|
||||
Call::Standing {
|
||||
peer_name: "ghost".into(),
|
||||
},
|
||||
) {
|
||||
Ok(Reply::HelloCtx(ctx)) => assert!(ctx.dialing_this_peer),
|
||||
other => panic!("HelloCtx failed: {other:?}"),
|
||||
Ok(Reply::Standing(s)) => assert_eq!(s, PeerStanding::Dialing),
|
||||
other => panic!("PeerStanding failed: {other:?}"),
|
||||
}
|
||||
// Kill it; the monitor must clear the intent without cooperation.
|
||||
go_tx.send(()).unwrap();
|
||||
@@ -341,11 +339,11 @@ fn dial_intent_cleared_when_dialer_dies() {
|
||||
loop {
|
||||
match gen_server::call(
|
||||
MANAGER,
|
||||
Call::HelloCtx {
|
||||
Call::Standing {
|
||||
peer_name: "ghost".into(),
|
||||
},
|
||||
) {
|
||||
Ok(Reply::HelloCtx(ctx)) if !ctx.dialing_this_peer => break,
|
||||
Ok(Reply::Standing(s)) if s != PeerStanding::Dialing => break,
|
||||
_ if Instant::now() > deadline => {
|
||||
panic!("dial intent not cleared after dialer death")
|
||||
}
|
||||
@@ -372,7 +370,7 @@ fn role_hs_listener() {
|
||||
.start()
|
||||
.expect("manager name is free");
|
||||
let listener = TcpTransport.listen("127.0.0.1:0").unwrap();
|
||||
let acceptor = spawn_acceptor(listener, local("node-b"));
|
||||
let acceptor = spawn_acceptor(listener, local("node-b"), Timing::default());
|
||||
println!("LISTENING {}", acceptor.local_addr());
|
||||
wait_peers(&["node-a"]);
|
||||
println!("PEERS node-a");
|
||||
@@ -389,7 +387,13 @@ fn role_hs_dialer() {
|
||||
.expect("manager name is free");
|
||||
let (tx, rx) = mpsc::channel();
|
||||
smarm::spawn(move || {
|
||||
let r = dial(&TcpTransport, &addr, "node-b", &local("node-a"));
|
||||
let r = dial(
|
||||
&TcpTransport,
|
||||
&addr,
|
||||
"node-b",
|
||||
&local("node-a"),
|
||||
Timing::default(),
|
||||
);
|
||||
let _ = tx.send(r);
|
||||
});
|
||||
if let Err(e) = poll_recv(&rx, "dial outcome") {
|
||||
@@ -459,7 +463,13 @@ fn concurrent_dial_to_same_name_refused() {
|
||||
// (the addr is unroutable on purpose — it must never be dialed).
|
||||
let (tx, rx) = mpsc::channel();
|
||||
smarm::spawn(move || {
|
||||
let r = dial(&TcpTransport, "127.0.0.1:1", "node-x", &local("node-a"));
|
||||
let r = dial(
|
||||
&TcpTransport,
|
||||
"127.0.0.1:1",
|
||||
"node-x",
|
||||
&local("node-a"),
|
||||
Timing::default(),
|
||||
);
|
||||
let _ = tx.send(r);
|
||||
});
|
||||
match poll_recv(&rx, "second dial outcome") {
|
||||
|
||||
@@ -0,0 +1,115 @@
|
||||
//! RFC 010 — a seed whose address answers as a *different* name
|
||||
//! (`DialError::PeerNameMismatch`) is dialed once and then parked: the
|
||||
//! connector must not redial it on backoff forever.
|
||||
//!
|
||||
//! Observed from the misdialed peer: each such dial establishes at the
|
||||
//! responder (it registers, `node_up`), then the dialer closes on the name
|
||||
//! check (`node_down`) — one membership blip per attempt. Cross-process: a
|
||||
//! *server* named `server` subscribes and reports; a *client* on fast
|
||||
//! timing (50–500ms backoff) seeds `("wrongname", server_addr)`. After the
|
||||
//! first blip the server counts further `NodeUp`s across 2s — several
|
||||
//! backoff periods. Parked ⇒ zero. Negative-control-verified: with the park
|
||||
//! stubbed out the count is ≥ 1 in the same window.
|
||||
#![cfg(feature = "cluster")]
|
||||
|
||||
mod common;
|
||||
|
||||
use common::{maybe_child, spawn_node};
|
||||
use smarm::cluster::envelope::NodeMeta;
|
||||
use smarm::cluster::membership::{subscribe, NodeEvent};
|
||||
use smarm::cluster::{start, Config, StaticSeeds, Timing};
|
||||
use std::time::{Duration, Instant};
|
||||
|
||||
const ROLES: &[(&str, fn())] = &[("server", role_server), ("client", role_client)];
|
||||
|
||||
fn meta() -> NodeMeta {
|
||||
NodeMeta {
|
||||
role: "mismatch".into(),
|
||||
region: "local".into(),
|
||||
}
|
||||
}
|
||||
|
||||
fn timing() -> Timing {
|
||||
Timing {
|
||||
initial_backoff: Duration::from_millis(50),
|
||||
max_backoff: Duration::from_millis(500),
|
||||
..Timing::default()
|
||||
}
|
||||
}
|
||||
|
||||
fn role_server() {
|
||||
smarm::run(|| {
|
||||
let cluster = start(Config {
|
||||
node_name: "server".into(),
|
||||
meta: meta(),
|
||||
listen_addr: std::env::var("SMARM_LISTEN_ADDR")
|
||||
.unwrap_or_else(|_| "127.0.0.1:0".into()),
|
||||
strategy: Box::new(StaticSeeds::new(Vec::<(String, String)>::new())),
|
||||
timing: timing(),
|
||||
})
|
||||
.expect("binds");
|
||||
let ev = subscribe().unwrap();
|
||||
println!("LISTENING {}", cluster.local_addr());
|
||||
// First blip: the misdialed client establishes, then closes on us.
|
||||
loop {
|
||||
match ev.rx.recv() {
|
||||
Ok(NodeEvent::NodeDown(i)) if i.name == "client" => break,
|
||||
Ok(_) => continue,
|
||||
Err(_) => panic!("manager gone"),
|
||||
}
|
||||
}
|
||||
println!("BLIP");
|
||||
// Now count further NodeUps across several backoff periods.
|
||||
let mut more = 0usize;
|
||||
let t0 = Instant::now();
|
||||
while t0.elapsed() < Duration::from_millis(2000) {
|
||||
match ev.rx.try_recv() {
|
||||
Ok(Some(NodeEvent::NodeUp(i))) if i.name == "client" => more += 1,
|
||||
Ok(_) => {}
|
||||
Err(_) => panic!("manager gone"),
|
||||
}
|
||||
smarm::sleep(Duration::from_millis(50));
|
||||
}
|
||||
println!("MORE {more}");
|
||||
loop {
|
||||
smarm::sleep(Duration::from_secs(3600));
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
fn role_client() {
|
||||
let server_addr = std::env::var("SMARM_SERVER_ADDR").expect("SMARM_SERVER_ADDR");
|
||||
smarm::run(move || {
|
||||
let _cluster = start(Config {
|
||||
node_name: "client".into(),
|
||||
meta: meta(),
|
||||
listen_addr: "127.0.0.1:0".into(),
|
||||
strategy: Box::new(StaticSeeds::new(vec![(
|
||||
"wrongname".to_string(),
|
||||
server_addr,
|
||||
)])),
|
||||
timing: timing(),
|
||||
})
|
||||
.expect("binds");
|
||||
println!("CLIENT UP");
|
||||
loop {
|
||||
smarm::sleep(Duration::from_secs(3600));
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn mismatched_seed_is_dialed_once_then_parked() {
|
||||
maybe_child(ROLES);
|
||||
let mut server = spawn_node("server", &[]);
|
||||
let saddr = server.wait_listening();
|
||||
let mut client = spawn_node("client", &[("SMARM_SERVER_ADDR", &saddr)]);
|
||||
client.wait_line("CLIENT UP", |l| l == "CLIENT UP");
|
||||
server.wait_line("BLIP", |l| l == "BLIP");
|
||||
let line = server.wait_line("MORE", |l| l.starts_with("MORE "));
|
||||
let more: usize = line.split_whitespace().nth(1).unwrap().parse().unwrap();
|
||||
assert_eq!(
|
||||
more, 0,
|
||||
"mismatched seed was redialed {more}× after being parked"
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,379 @@
|
||||
//! RFC 010 c13 — connection-loss synthesis.
|
||||
//!
|
||||
//! Local suite (`run()`, no network): the read-side backstop. A
|
||||
//! `RemoteMonitor` whose channel closes without a notice reads as
|
||||
//! `Disconnected` exactly once (a `Monitor` command that reached the conn
|
||||
//! actor's inbox but was never processed — the drain gap); after
|
||||
//! `demonitor_remote` a closed channel stays a plain `Err`, never a notice.
|
||||
//!
|
||||
//! Cross-process: the headline contrast — an actor's own death gives its
|
||||
//! TRUE reason, loss of the LINK gives `Disconnected` (both a commanded
|
||||
//! `Disconnect` and a SIGKILLed peer process are `Disconnected` from the
|
||||
//! monitor's view: nobody is left to say otherwise). Reconnect does not
|
||||
//! resurrect: the old monitor yields nothing more, proven by stream ORDER
|
||||
//! (a fresh monitor over the new link delivers first). The ignored test
|
||||
//! trips liveness by SIGSTOP and then drops the link too, asserting exactly
|
||||
//! one notice for one monitor.
|
||||
#![cfg(feature = "cluster")]
|
||||
|
||||
mod common;
|
||||
|
||||
use common::{maybe_child, spawn_node};
|
||||
use smarm::cluster::envelope::NodeMeta;
|
||||
use smarm::cluster::expose::{expose, expose_type};
|
||||
use smarm::cluster::manager::{Call, Reply, MANAGER};
|
||||
use smarm::cluster::membership::{subscribe, NodeEvent};
|
||||
use smarm::cluster::remote::{
|
||||
self, demonitor_remote, monitor_remote, send_to_remote, RemoteName, RemotePid,
|
||||
};
|
||||
use smarm::cluster::{start, Config, RemoteDownReason, StaticSeeds, Timing};
|
||||
use smarm::pg::Incarnation;
|
||||
use smarm::{
|
||||
channel, gen_server, install, register, run, spawn, Addressable, DownReason, Erased, Name, Pid,
|
||||
};
|
||||
use std::collections::HashMap;
|
||||
use std::time::Duration;
|
||||
|
||||
// ---- message types (hand-rolled serde; the crate is derive-less) ---------
|
||||
|
||||
#[derive(Debug)]
|
||||
struct Ctl {
|
||||
cmd: String,
|
||||
reply_to: RemotePid<Client>,
|
||||
}
|
||||
#[derive(Debug)]
|
||||
struct Answer {
|
||||
text: String,
|
||||
pid: Option<RemotePid<Erased>>,
|
||||
}
|
||||
struct Client;
|
||||
impl Addressable for Client {
|
||||
type Msg = Answer;
|
||||
}
|
||||
|
||||
impl serde::Serialize for Ctl {
|
||||
fn serialize<S: serde::Serializer>(&self, s: S) -> Result<S::Ok, S::Error> {
|
||||
use serde::ser::SerializeTuple;
|
||||
let mut t = s.serialize_tuple(2)?;
|
||||
t.serialize_element(&self.cmd)?;
|
||||
t.serialize_element(&self.reply_to)?;
|
||||
t.end()
|
||||
}
|
||||
}
|
||||
impl<'de> serde::Deserialize<'de> for Ctl {
|
||||
fn deserialize<D: serde::Deserializer<'de>>(d: D) -> Result<Self, D::Error> {
|
||||
let (cmd, reply_to) = <(String, RemotePid<Client>)>::deserialize(d)?;
|
||||
Ok(Ctl { cmd, reply_to })
|
||||
}
|
||||
}
|
||||
impl serde::Serialize for Answer {
|
||||
fn serialize<S: serde::Serializer>(&self, s: S) -> Result<S::Ok, S::Error> {
|
||||
use serde::ser::SerializeTuple;
|
||||
let mut t = s.serialize_tuple(2)?;
|
||||
t.serialize_element(&self.text)?;
|
||||
t.serialize_element(&self.pid)?;
|
||||
t.end()
|
||||
}
|
||||
}
|
||||
impl<'de> serde::Deserialize<'de> for Answer {
|
||||
fn deserialize<D: serde::Deserializer<'de>>(d: D) -> Result<Self, D::Error> {
|
||||
let (text, pid) = <(String, Option<RemotePid<Erased>>)>::deserialize(d)?;
|
||||
Ok(Answer { text, pid })
|
||||
}
|
||||
}
|
||||
|
||||
// ================= local suite =========================================
|
||||
|
||||
/// A `Monitor` command handed to the connection but never processed (its
|
||||
/// receiver dropped unread) reads as `Disconnected` — once. A second read
|
||||
/// is the ordinary closed-channel `Err`, so "exactly one notice" holds.
|
||||
#[test]
|
||||
fn unread_command_reads_as_disconnected_once() {
|
||||
maybe_child(ROLES);
|
||||
run(|| {
|
||||
remote::set_local_identity("me", Incarnation::new(7));
|
||||
let (probe_tx, _probe_rx) = channel();
|
||||
let inbox =
|
||||
remote::bind_outbound_probe_with_monitors("peer", Incarnation::new(5), probe_tx);
|
||||
let target = RemotePid::<Erased>::from_parts("peer", Incarnation::new(5), 9, 1);
|
||||
let m = monitor_remote(target.clone());
|
||||
assert!(
|
||||
matches!(m.try_recv(), Ok(None)),
|
||||
"command is in flight, no notice yet"
|
||||
);
|
||||
drop(inbox); // the conn actor died with the command unread
|
||||
let d = m.recv().unwrap();
|
||||
assert_eq!(d.pid, target);
|
||||
assert_eq!(d.reason, RemoteDownReason::Disconnected);
|
||||
assert!(
|
||||
m.recv().is_err(),
|
||||
"second read is closed, not a second notice"
|
||||
);
|
||||
assert!(m.try_recv().is_err());
|
||||
});
|
||||
}
|
||||
|
||||
/// After `demonitor_remote`, a closed channel is a closed channel: no
|
||||
/// notice is synthesized for a monitor the caller cancelled.
|
||||
#[test]
|
||||
fn cancelled_monitor_never_synthesizes() {
|
||||
maybe_child(ROLES);
|
||||
run(|| {
|
||||
remote::set_local_identity("me", Incarnation::new(7));
|
||||
let (probe_tx, _probe_rx) = channel();
|
||||
let inbox =
|
||||
remote::bind_outbound_probe_with_monitors("peer", Incarnation::new(5), probe_tx);
|
||||
let target = RemotePid::<Erased>::from_parts("peer", Incarnation::new(5), 9, 1);
|
||||
let m = monitor_remote(target);
|
||||
demonitor_remote(&m);
|
||||
drop(inbox);
|
||||
assert!(m.recv().is_err());
|
||||
assert!(m.try_recv().is_err());
|
||||
});
|
||||
}
|
||||
|
||||
// ================= cross-process ======================================
|
||||
|
||||
const ROLES: &[(&str, fn())] = &[
|
||||
("server", role_server),
|
||||
("client", role_client),
|
||||
("client_stop", role_client_stop),
|
||||
];
|
||||
|
||||
const CTL: Name<Ctl> = Name::new("c13.ctl");
|
||||
|
||||
fn cfg(name: &str, seeds: Vec<(String, String)>) -> Config {
|
||||
Config {
|
||||
node_name: name.to_string(),
|
||||
meta: NodeMeta {
|
||||
role: "c13".into(),
|
||||
region: "local".into(),
|
||||
},
|
||||
listen_addr: std::env::var("SMARM_LISTEN_ADDR").unwrap_or_else(|_| "127.0.0.1:0".into()),
|
||||
strategy: Box::new(StaticSeeds::new(seeds)),
|
||||
timing: timing(),
|
||||
}
|
||||
}
|
||||
|
||||
/// The p11 knobs make the liveness test fast: both roles of that test are
|
||||
/// spawned with `SMARM_FAST_TIMING=1` and agree on a 100ms heartbeat /
|
||||
/// 500ms liveness window. Everything else runs the shipping defaults.
|
||||
fn timing() -> Timing {
|
||||
if std::env::var_os("SMARM_FAST_TIMING").is_some() {
|
||||
Timing {
|
||||
heartbeat_interval: Duration::from_millis(100),
|
||||
liveness_timeout: Duration::from_millis(500),
|
||||
initial_backoff: Duration::from_millis(50),
|
||||
max_backoff: Duration::from_millis(500),
|
||||
..Timing::default()
|
||||
}
|
||||
} else {
|
||||
Timing::default()
|
||||
}
|
||||
}
|
||||
|
||||
fn wait_up(events: &smarm::cluster::membership::MembershipEvents, who: &str) {
|
||||
loop {
|
||||
match events.rx.recv() {
|
||||
Ok(NodeEvent::NodeUp(i)) if i.name == who => return,
|
||||
Ok(_) => continue,
|
||||
Err(_) => panic!("manager gone"),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn disconnect(name: &str) {
|
||||
assert!(matches!(
|
||||
gen_server::call(
|
||||
MANAGER,
|
||||
Call::Disconnect {
|
||||
name: name.to_string()
|
||||
}
|
||||
),
|
||||
Ok(Reply::Disconnected)
|
||||
));
|
||||
}
|
||||
|
||||
/// Server: `spawn` ⇒ a parked worker (answer carries its pid);
|
||||
/// `kill:<index>` releases it, whereupon it returns (Exit).
|
||||
fn role_server() {
|
||||
smarm::run(move || {
|
||||
let cluster = start(cfg("server", vec![])).expect("binds");
|
||||
println!("LISTENING {}", cluster.local_addr());
|
||||
let (tx, rx) = channel::<Ctl>();
|
||||
register(CTL, tx).unwrap();
|
||||
expose(CTL);
|
||||
println!("READY");
|
||||
let mut workers: HashMap<u32, smarm::channel::Sender<()>> = HashMap::new();
|
||||
loop {
|
||||
let ctl = rx.recv().unwrap();
|
||||
println!("CTL {}", ctl.cmd);
|
||||
let (text, pid): (String, Option<RemotePid<Erased>>) = match ctl.cmd.as_str() {
|
||||
"spawn" => {
|
||||
let (go_tx, go_rx) = channel::<()>();
|
||||
let p: Pid = spawn(move || {
|
||||
let _ = go_rx.recv();
|
||||
})
|
||||
.pid();
|
||||
workers.insert(p.index(), go_tx);
|
||||
(
|
||||
"ok".into(),
|
||||
Some(RemotePid::from_local(p).expect("identity set")),
|
||||
)
|
||||
}
|
||||
other => {
|
||||
let idx: u32 = other.strip_prefix("kill:").unwrap().parse().unwrap();
|
||||
if let Some(go) = workers.remove(&idx) {
|
||||
let _ = go.send(());
|
||||
}
|
||||
("killed".into(), None)
|
||||
}
|
||||
};
|
||||
send_to_remote(ctl.reply_to, Answer { text, pid }).unwrap();
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/// Client-side setup shared by both client roles: join, expose the reply
|
||||
/// path, hand back an `ask` closure and the membership stream.
|
||||
fn client_setup() -> (
|
||||
smarm::cluster::Cluster,
|
||||
smarm::cluster::membership::MembershipEvents,
|
||||
impl Fn(&str) -> Answer,
|
||||
) {
|
||||
let server_addr = std::env::var("SMARM_SERVER_ADDR").expect("SMARM_SERVER_ADDR");
|
||||
let cluster = start(cfg("client", vec![("server".into(), server_addr)])).expect("binds");
|
||||
let ev = subscribe().unwrap();
|
||||
wait_up(&ev, "server");
|
||||
let (tx, rx) = channel::<Answer>();
|
||||
let me: Pid<Client> = install::<Client>(tx);
|
||||
expose_type::<Answer>();
|
||||
let ask = move |cmd: &str| -> Answer {
|
||||
remote::send(
|
||||
RemoteName::new("server", CTL),
|
||||
Ctl {
|
||||
cmd: cmd.into(),
|
||||
reply_to: RemotePid::from_local(me).expect("identity set"),
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
rx.recv().unwrap()
|
||||
};
|
||||
(cluster, ev, ask)
|
||||
}
|
||||
|
||||
fn role_client() {
|
||||
smarm::run(move || {
|
||||
let (_cluster, ev, ask) = client_setup();
|
||||
|
||||
// 1. Headline: actor death ⇒ TRUE reason; link cut ⇒ Disconnected.
|
||||
let a = ask("spawn").pid.unwrap();
|
||||
let b = ask("spawn").pid.unwrap();
|
||||
let ma = monitor_remote(a.clone());
|
||||
let mb = monitor_remote(b.clone());
|
||||
ask(&format!("kill:{}", a.index()));
|
||||
let d = ma.recv().unwrap();
|
||||
assert_eq!(d.pid, a);
|
||||
println!("DOWN actor {:?}", d.reason);
|
||||
disconnect("server");
|
||||
let d = mb.recv().unwrap();
|
||||
assert_eq!(d.pid, b);
|
||||
println!("DOWN link {:?}", d.reason);
|
||||
|
||||
// 2. Reconnect does not resurrect. The connector redials on
|
||||
// node_down; over the NEW link a fresh monitor delivers, while
|
||||
// the old one (already answered) yields nothing further — order
|
||||
// proves it, and `b` is even still alive on the server.
|
||||
wait_up(&ev, "server");
|
||||
println!("RECONNECTED");
|
||||
let c = ask("spawn").pid.unwrap();
|
||||
let mc = monitor_remote(c.clone());
|
||||
ask(&format!("kill:{}", b.index()));
|
||||
ask(&format!("kill:{}", c.index()));
|
||||
assert_eq!(mc.recv().unwrap().reason, DownReason::Exit.into());
|
||||
let stray = matches!(mb.try_recv(), Ok(Some(_)));
|
||||
println!("RESURRECT stray={stray}");
|
||||
|
||||
// 3. Peer PROCESS killed ⇒ Disconnected too (nobody is left to send
|
||||
// Down): the parent SIGKILLs the server once it sees the marker.
|
||||
let e = ask("spawn").pid.unwrap();
|
||||
let me_ = monitor_remote(e.clone());
|
||||
println!("KILL SERVER NOW");
|
||||
let d = me_.recv().unwrap();
|
||||
assert_eq!(d.pid, e);
|
||||
println!("DOWN procdeath {:?}", d.reason);
|
||||
|
||||
println!("CLIENT DONE");
|
||||
loop {
|
||||
smarm::sleep(Duration::from_secs(3600));
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/// The slow role: liveness expiry (peer SIGSTOPped) followed by the link
|
||||
/// dropping for real (peer SIGKILLed) — one monitor, exactly one notice.
|
||||
fn role_client_stop() {
|
||||
smarm::run(move || {
|
||||
let (_cluster, _ev, ask) = client_setup();
|
||||
let a = ask("spawn").pid.unwrap();
|
||||
let ma = monitor_remote(a.clone());
|
||||
println!("STOP SERVER NOW");
|
||||
let d = ma.recv().unwrap(); // liveness expiry, ~liveness_timeout
|
||||
assert_eq!(d.pid, a);
|
||||
println!("DOWN stopped {:?}", d.reason);
|
||||
println!("KILL SERVER NOW");
|
||||
// Give the drop every chance to produce a second notice, then look.
|
||||
smarm::sleep(Duration::from_secs(1));
|
||||
let dup = matches!(ma.try_recv(), Ok(Some(_)));
|
||||
println!("DUPLICATE dup={dup}");
|
||||
println!("CLIENT DONE");
|
||||
loop {
|
||||
smarm::sleep(Duration::from_secs(3600));
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/// The Phase 4 c13 gate: partition vs. death distinguishable; nothing
|
||||
/// survives reconnect; a dead peer process is a Disconnected too.
|
||||
#[test]
|
||||
fn link_loss_is_disconnected_and_does_not_survive_reconnect() {
|
||||
maybe_child(ROLES);
|
||||
let mut server = spawn_node("server", &[]);
|
||||
let saddr = server.wait_listening();
|
||||
server.wait_line("READY", |l| l == "READY");
|
||||
let mut client = spawn_node("client", &[("SMARM_SERVER_ADDR", &saddr)]);
|
||||
client.wait_line("DOWN actor Local(Exit)", |l| l == "DOWN actor Local(Exit)");
|
||||
client.wait_line("DOWN link Disconnected", |l| l == "DOWN link Disconnected");
|
||||
client.wait_line("RECONNECTED", |l| l == "RECONNECTED");
|
||||
client.wait_line("RESURRECT stray=false", |l| l == "RESURRECT stray=false");
|
||||
client.wait_line("KILL SERVER NOW", |l| l == "KILL SERVER NOW");
|
||||
server.kill();
|
||||
client.wait_line("DOWN procdeath Disconnected", |l| {
|
||||
l == "DOWN procdeath Disconnected"
|
||||
});
|
||||
client.wait_line("CLIENT DONE", |l| l == "CLIENT DONE");
|
||||
}
|
||||
|
||||
/// Covers the invariant the headline test cannot: liveness expiry and the
|
||||
/// transport drop both firing for the same connection yield ONE notice.
|
||||
/// Runs on the fast [`timing`] (both roles) — was `#[ignore]`d at the 4s
|
||||
/// default until the p11 knobs landed.
|
||||
#[test]
|
||||
fn timeout_then_drop_yields_one_notice() {
|
||||
maybe_child(ROLES);
|
||||
let fast = ("SMARM_FAST_TIMING", "1");
|
||||
let mut server = spawn_node("server", &[fast]);
|
||||
let saddr = server.wait_listening();
|
||||
server.wait_line("READY", |l| l == "READY");
|
||||
let mut client = spawn_node("client_stop", &[("SMARM_SERVER_ADDR", &saddr), fast]);
|
||||
client.wait_line("STOP SERVER NOW", |l| l == "STOP SERVER NOW");
|
||||
let spid = server.pid().expect("server alive") as libc::pid_t;
|
||||
assert_eq!(unsafe { libc::kill(spid, libc::SIGSTOP) }, 0);
|
||||
client.wait_line("DOWN stopped Disconnected", |l| {
|
||||
l == "DOWN stopped Disconnected"
|
||||
});
|
||||
client.wait_line("KILL SERVER NOW", |l| l == "KILL SERVER NOW");
|
||||
server.kill(); // SIGKILL works on a stopped process; Drop would too
|
||||
client.wait_line("DUPLICATE dup=false", |l| l == "DUPLICATE dup=false");
|
||||
client.wait_line("CLIENT DONE", |l| l == "CLIENT DONE");
|
||||
}
|
||||
@@ -0,0 +1,161 @@
|
||||
//! RFC 010 — `Discovery::Withdrawn`: a strategy retracts a candidate and the
|
||||
//! connector stops dialing it.
|
||||
//!
|
||||
//! Cross-process: a plain *server* node, and a *client* whose strategy is a
|
||||
//! script: announce a decoy `(ghost, addr)` where `addr` is a raw
|
||||
//! `TcpListener` the client itself holds (an OS thread accepts and
|
||||
//! immediately closes, so every dial fails at handshake and the connector
|
||||
//! keeps retrying on backoff — the accept count is the dial count); after a
|
||||
//! beat, withdraw the decoy and announce the real server. The client waits
|
||||
//! for the server's `node_up` — which is *after* the withdrawal in the
|
||||
//! strategy's own stream — then watches the decoy's accept count stay flat
|
||||
//! across a window longer than the pending backoff. Before withdrawal it
|
||||
//! must have been climbing (≥ 1), or the negative proves nothing.
|
||||
#![cfg(feature = "cluster")]
|
||||
|
||||
mod common;
|
||||
|
||||
use common::{maybe_child, spawn_node};
|
||||
use smarm::channel::Sender;
|
||||
use smarm::cluster::discovery::{Discovery, Strategy};
|
||||
use smarm::cluster::envelope::NodeMeta;
|
||||
use smarm::cluster::membership::{subscribe, NodeEvent};
|
||||
use smarm::cluster::{start, Config, StaticSeeds, Timing};
|
||||
use std::net::TcpListener;
|
||||
use std::sync::atomic::{AtomicUsize, Ordering};
|
||||
use std::sync::Arc;
|
||||
use std::time::{Duration, Instant};
|
||||
|
||||
const ROLES: &[(&str, fn())] = &[("server", role_server), ("client", role_client)];
|
||||
|
||||
fn meta() -> NodeMeta {
|
||||
NodeMeta {
|
||||
role: "withdraw".into(),
|
||||
region: "local".into(),
|
||||
}
|
||||
}
|
||||
|
||||
fn role_server() {
|
||||
smarm::run(|| {
|
||||
let cluster = start(Config {
|
||||
node_name: "server".into(),
|
||||
meta: meta(),
|
||||
listen_addr: std::env::var("SMARM_LISTEN_ADDR")
|
||||
.unwrap_or_else(|_| "127.0.0.1:0".into()),
|
||||
strategy: Box::new(StaticSeeds::new(Vec::<(String, String)>::new())),
|
||||
timing: Timing::default(),
|
||||
})
|
||||
.expect("binds");
|
||||
println!("LISTENING {}", cluster.local_addr());
|
||||
loop {
|
||||
smarm::sleep(Duration::from_secs(3600));
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/// Scripted strategy: decoy, pause, withdraw decoy, real server, done.
|
||||
struct Script {
|
||||
decoy: String,
|
||||
server: String,
|
||||
}
|
||||
|
||||
impl Strategy for Script {
|
||||
fn run(self: Box<Self>, out: Sender<Discovery>) {
|
||||
let _ = out.send(Discovery::Candidate {
|
||||
name: "ghost".into(),
|
||||
addr: self.decoy.clone(),
|
||||
});
|
||||
// Long enough for the 250ms/500ms retries to land: ≥ 3 dials.
|
||||
smarm::sleep(Duration::from_millis(1100));
|
||||
let _ = out.send(Discovery::Withdrawn {
|
||||
name: "ghost".into(),
|
||||
addr: self.decoy,
|
||||
});
|
||||
let _ = out.send(Discovery::Candidate {
|
||||
name: "server".into(),
|
||||
addr: self.server,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
fn role_client() {
|
||||
let server_addr = std::env::var("SMARM_SERVER_ADDR").expect("SMARM_SERVER_ADDR");
|
||||
// The decoy: accept-and-close on an OS thread; count every accept.
|
||||
let decoy = TcpListener::bind("127.0.0.1:0").unwrap();
|
||||
let decoy_addr = decoy.local_addr().unwrap().to_string();
|
||||
let dials = Arc::new(AtomicUsize::new(0));
|
||||
let counter = dials.clone();
|
||||
std::thread::spawn(move || {
|
||||
for conn in decoy.incoming() {
|
||||
counter.fetch_add(1, Ordering::SeqCst);
|
||||
drop(conn);
|
||||
}
|
||||
});
|
||||
|
||||
smarm::run(move || {
|
||||
let _cluster = start(Config {
|
||||
node_name: "client".into(),
|
||||
meta: meta(),
|
||||
listen_addr: "127.0.0.1:0".into(),
|
||||
strategy: Box::new(Script {
|
||||
decoy: decoy_addr,
|
||||
server: server_addr,
|
||||
}),
|
||||
timing: Timing::default(),
|
||||
})
|
||||
.expect("binds");
|
||||
let ev = subscribe().unwrap();
|
||||
loop {
|
||||
match ev.rx.recv() {
|
||||
Ok(NodeEvent::NodeUp(i)) if i.name == "server" => break,
|
||||
Ok(_) => continue,
|
||||
Err(_) => panic!("manager gone"),
|
||||
}
|
||||
}
|
||||
// The withdrawal preceded the server candidate in the strategy's
|
||||
// stream, so it has been applied. Any dial that started before it
|
||||
// is bounded by the connect+handshake deadlines; let it drain, then
|
||||
// hold the count flat across a window longer than the pending
|
||||
// backoff would be (1s at this point, 2s next).
|
||||
let before = dials.load(Ordering::SeqCst);
|
||||
smarm::sleep(Duration::from_millis(500));
|
||||
let settled = dials.load(Ordering::SeqCst);
|
||||
let t0 = Instant::now();
|
||||
while t0.elapsed() < Duration::from_millis(3000) {
|
||||
smarm::sleep(Duration::from_millis(100));
|
||||
}
|
||||
let after = dials.load(Ordering::SeqCst);
|
||||
println!("WITHDRAWN before={before} settled={settled} after={after}");
|
||||
println!("CLIENT DONE");
|
||||
loop {
|
||||
smarm::sleep(Duration::from_secs(3600));
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn withdrawn_candidate_is_no_longer_dialed() {
|
||||
maybe_child(ROLES);
|
||||
let mut server = spawn_node("server", &[]);
|
||||
let saddr = server.wait_listening();
|
||||
let mut client = spawn_node("client", &[("SMARM_SERVER_ADDR", &saddr)]);
|
||||
let line = client.wait_line("WITHDRAWN", |l| l.starts_with("WITHDRAWN "));
|
||||
let mut nums = line
|
||||
.split_whitespace()
|
||||
.skip(1)
|
||||
.map(|kv| kv.split_once('=').unwrap().1.parse::<usize>().unwrap());
|
||||
let (before, settled, after) = (
|
||||
nums.next().unwrap(),
|
||||
nums.next().unwrap(),
|
||||
nums.next().unwrap(),
|
||||
);
|
||||
assert!(
|
||||
before >= 1,
|
||||
"decoy was never dialed; the negative proves nothing: {line}"
|
||||
);
|
||||
assert_eq!(
|
||||
settled, after,
|
||||
"connector kept dialing a withdrawn candidate: {line}"
|
||||
);
|
||||
client.wait_line("CLIENT DONE", |l| l == "CLIENT DONE");
|
||||
}
|
||||
@@ -8,6 +8,7 @@ use smarm::cluster::envelope::{
|
||||
decode_payload, encode_payload, DecodeError, Frame, NodeMeta, RejectReason, MAX_FRAME_LEN,
|
||||
PROTO_VERSION,
|
||||
};
|
||||
use smarm::cluster::RemoteDownReason;
|
||||
use smarm::monitor::DownReason;
|
||||
use smarm::pg::Incarnation;
|
||||
|
||||
@@ -55,7 +56,11 @@ fn all_frames() -> Vec<Frame> {
|
||||
Frame::Demonitor { monitor_id: 77 },
|
||||
Frame::Down {
|
||||
monitor_id: 77,
|
||||
reason: DownReason::Panic,
|
||||
reason: RemoteDownReason::Local(DownReason::Panic),
|
||||
},
|
||||
Frame::Down {
|
||||
monitor_id: 78,
|
||||
reason: RemoteDownReason::Disconnected,
|
||||
},
|
||||
]
|
||||
}
|
||||
@@ -151,7 +156,7 @@ fn unknown_enum_tags() {
|
||||
assert_eq!(
|
||||
Frame::decode(&buf),
|
||||
Err(DecodeError::UnknownEnumTag {
|
||||
what: "DownReason",
|
||||
what: "RemoteDownReason",
|
||||
tag: 200
|
||||
})
|
||||
);
|
||||
|
||||
+10
-19
@@ -5,7 +5,7 @@
|
||||
|
||||
use smarm::cluster::envelope::{Frame, NodeMeta, RejectReason, PROTO_VERSION};
|
||||
use smarm::cluster::handshake::{
|
||||
dial_wins, HelloCtx, Initiator, InitiatorOutcome, Local, Responder, ResponderOutcome,
|
||||
dial_wins, Initiator, InitiatorOutcome, Local, PeerStanding, Responder, ResponderOutcome,
|
||||
};
|
||||
use smarm::pg::Incarnation;
|
||||
|
||||
@@ -42,7 +42,7 @@ fn happy_path_establishes_both_ends() {
|
||||
assert_eq!(hello, hello_from("alpha"), "initiator emits its identity");
|
||||
|
||||
let responder = Responder::new(local("beta"));
|
||||
let (reply, peer) = match responder.on_frame(hello, HelloCtx::default()) {
|
||||
let (reply, peer) = match responder.on_frame(hello, PeerStanding::Free) {
|
||||
ResponderOutcome::Accepted { reply, peer } => (reply, peer),
|
||||
other => panic!("expected Accepted, got {other:?}"),
|
||||
};
|
||||
@@ -82,7 +82,7 @@ fn hash_mismatch_rejected() {
|
||||
incarnation: Incarnation::new(7),
|
||||
meta: local("alpha").meta,
|
||||
};
|
||||
match responder.on_frame(hello, HelloCtx::default()) {
|
||||
match responder.on_frame(hello, PeerStanding::Free) {
|
||||
ResponderOutcome::Rejected { reply, reason } => {
|
||||
assert_eq!(reason, RejectReason::HashMismatch);
|
||||
assert_eq!(reply, Frame::HelloReject { reason });
|
||||
@@ -112,7 +112,7 @@ fn proto_version_mismatch_rejected_and_checked_first() {
|
||||
incarnation: Incarnation::new(7),
|
||||
meta: local("alpha").meta,
|
||||
};
|
||||
match responder.on_frame(hello, HelloCtx::default()) {
|
||||
match responder.on_frame(hello, PeerStanding::Free) {
|
||||
ResponderOutcome::Rejected { reason, .. } => {
|
||||
assert_eq!(reason, RejectReason::ProtoVersion);
|
||||
}
|
||||
@@ -123,10 +123,7 @@ fn proto_version_mismatch_rejected_and_checked_first() {
|
||||
#[test]
|
||||
fn claimed_name_rejected() {
|
||||
let responder = Responder::new(local("beta"));
|
||||
let ctx = HelloCtx {
|
||||
name_claimed: true,
|
||||
dialing_this_peer: false,
|
||||
};
|
||||
let ctx = PeerStanding::Claimed;
|
||||
match responder.on_frame(hello_from("alpha"), ctx) {
|
||||
ResponderOutcome::Rejected { reason, .. } => {
|
||||
assert_eq!(reason, RejectReason::NameTaken);
|
||||
@@ -139,7 +136,7 @@ fn claimed_name_rejected() {
|
||||
fn own_name_offered_rejected_as_name_taken() {
|
||||
// Self-connect or genuine collision: the responder's own name arrives.
|
||||
let responder = Responder::new(local("beta"));
|
||||
match responder.on_frame(hello_from("beta"), HelloCtx::default()) {
|
||||
match responder.on_frame(hello_from("beta"), PeerStanding::Free) {
|
||||
ResponderOutcome::Rejected { reason, .. } => {
|
||||
assert_eq!(reason, RejectReason::NameTaken);
|
||||
}
|
||||
@@ -158,10 +155,7 @@ fn hash_checked_before_name() {
|
||||
incarnation: Incarnation::new(7),
|
||||
meta: local("alpha").meta,
|
||||
};
|
||||
let ctx = HelloCtx {
|
||||
name_claimed: true,
|
||||
dialing_this_peer: false,
|
||||
};
|
||||
let ctx = PeerStanding::Claimed;
|
||||
match responder.on_frame(hello, ctx) {
|
||||
ResponderOutcome::Rejected { reason, .. } => {
|
||||
assert_eq!(reason, RejectReason::HashMismatch);
|
||||
@@ -184,10 +178,7 @@ fn dial_wins_is_deterministic_and_antisymmetric() {
|
||||
fn simultaneous_connect_exactly_one_side_accepts() {
|
||||
// alpha and beta dial each other at once. Each responder sees the peer's
|
||||
// Hello while its own dial is in flight.
|
||||
let ctx = HelloCtx {
|
||||
name_claimed: false,
|
||||
dialing_this_peer: true,
|
||||
};
|
||||
let ctx = PeerStanding::Dialing;
|
||||
|
||||
// On beta: inbound is alpha's dial; alpha < beta, so the inbound wins.
|
||||
let on_beta = Responder::new(local("beta")).on_frame(hello_from("alpha"), ctx);
|
||||
@@ -208,7 +199,7 @@ fn simultaneous_connect_exactly_one_side_accepts() {
|
||||
#[test]
|
||||
fn tiebreak_loss_only_applies_when_dialing() {
|
||||
// Same inbound Hello, no dial in flight: plain accept.
|
||||
let on_alpha = Responder::new(local("alpha")).on_frame(hello_from("beta"), HelloCtx::default());
|
||||
let on_alpha = Responder::new(local("alpha")).on_frame(hello_from("beta"), PeerStanding::Free);
|
||||
assert!(matches!(on_alpha, ResponderOutcome::Accepted { .. }));
|
||||
}
|
||||
|
||||
@@ -225,7 +216,7 @@ fn garbage_before_hello_fails_without_reply() {
|
||||
},
|
||||
Frame::Demonitor { monitor_id: 3 },
|
||||
] {
|
||||
let out = Responder::new(local("beta")).on_frame(frame.clone(), HelloCtx::default());
|
||||
let out = Responder::new(local("beta")).on_frame(frame.clone(), PeerStanding::Free);
|
||||
match out {
|
||||
ResponderOutcome::Failed(f) => assert_eq!(f, frame),
|
||||
other => panic!("expected Failed({frame:?}), got {other:?}"),
|
||||
|
||||
+27
-18
@@ -18,6 +18,7 @@ use smarm::cluster::membership::{subscribe, view, MembershipEvents, NodeEvent};
|
||||
use smarm::cluster::spawn_established;
|
||||
use smarm::cluster::transport::tcp::TcpTransport;
|
||||
use smarm::cluster::transport::{Conn, FramedConn, Transport};
|
||||
use smarm::cluster::Timing;
|
||||
use smarm::gen_server::{self, GenServerBuilder};
|
||||
use smarm::pg::{Incarnation, NodeId};
|
||||
use smarm::run;
|
||||
@@ -85,8 +86,10 @@ fn subscriber_sees_up_and_down() {
|
||||
let t = TcpTransport;
|
||||
let (a1, b1) = pair(&t);
|
||||
let (a2, b2) = pair(&t);
|
||||
spawn_established(FramedConn::new(a1), peer("node-b", 1)).expect("node-b registers");
|
||||
spawn_established(FramedConn::new(a2), peer("node-c", 1)).expect("node-c registers");
|
||||
spawn_established(FramedConn::new(a1), peer("node-b", 1), Timing::default())
|
||||
.expect("node-b registers");
|
||||
spawn_established(FramedConn::new(a2), peer("node-c", 1), Timing::default())
|
||||
.expect("node-c registers");
|
||||
|
||||
let up_b = match next_event(&ev, "node_up(node-b)") {
|
||||
NodeEvent::NodeUp(info) => {
|
||||
@@ -110,14 +113,14 @@ fn subscriber_sees_up_and_down() {
|
||||
disconnect("node-b");
|
||||
assert_eq!(
|
||||
next_event(&ev, "node_down(node-b)"),
|
||||
NodeEvent::NodeDown { node: up_b.node }
|
||||
NodeEvent::NodeDown(up_b.clone())
|
||||
);
|
||||
|
||||
// Peer EOF, no command: down with node-c's id.
|
||||
drop(b2);
|
||||
assert_eq!(
|
||||
next_event(&ev, "node_down(node-c)"),
|
||||
NodeEvent::NodeDown { node: up_c.node }
|
||||
NodeEvent::NodeDown(up_c.clone())
|
||||
);
|
||||
assert_quiet(&ev);
|
||||
|
||||
@@ -140,8 +143,10 @@ fn late_subscriber_gets_snapshot_and_view_agrees() {
|
||||
let t = TcpTransport;
|
||||
let (a1, b1) = pair(&t);
|
||||
let (a2, b2) = pair(&t);
|
||||
spawn_established(FramedConn::new(a1), peer("node-b", 1)).expect("node-b registers");
|
||||
spawn_established(FramedConn::new(a2), peer("node-c", 1)).expect("node-c registers");
|
||||
spawn_established(FramedConn::new(a1), peer("node-b", 1), Timing::default())
|
||||
.expect("node-b registers");
|
||||
spawn_established(FramedConn::new(a2), peer("node-c", 1), Timing::default())
|
||||
.expect("node-c registers");
|
||||
|
||||
let ev = subscribe().expect("manager is up");
|
||||
let mut names = Vec::new();
|
||||
@@ -190,20 +195,25 @@ fn restart_gets_new_id_blip_keeps_id() {
|
||||
other => panic!("expected node_up ({what}), got {other:?}"),
|
||||
}
|
||||
};
|
||||
let down_id = |e: NodeEvent, what: &str| -> NodeId {
|
||||
match e {
|
||||
NodeEvent::NodeDown(info) => info.node,
|
||||
other => panic!("expected node_down ({what}), got {other:?}"),
|
||||
}
|
||||
};
|
||||
|
||||
// Up at incarnation 1, then the peer dies (EOF).
|
||||
let (a1, b1) = pair(&t);
|
||||
spawn_established(FramedConn::new(a1), peer("node-b", 1)).expect("registers");
|
||||
spawn_established(FramedConn::new(a1), peer("node-b", 1), Timing::default())
|
||||
.expect("registers");
|
||||
let id1 = id(next_event(&ev, "node_up inc 1"), "inc 1");
|
||||
drop(b1);
|
||||
assert_eq!(
|
||||
next_event(&ev, "node_down inc 1"),
|
||||
NodeEvent::NodeDown { node: id1 }
|
||||
);
|
||||
assert_eq!(down_id(next_event(&ev, "node_down inc 1"), "inc 1"), id1);
|
||||
|
||||
// Restart: new incarnation, new id — the ghost's id is not reused.
|
||||
let (a2, b2) = pair(&t);
|
||||
spawn_established(FramedConn::new(a2), peer("node-b", 2)).expect("registers");
|
||||
spawn_established(FramedConn::new(a2), peer("node-b", 2), Timing::default())
|
||||
.expect("registers");
|
||||
let id2 = id(next_event(&ev, "node_up inc 2"), "inc 2");
|
||||
assert_ne!(
|
||||
id1, id2,
|
||||
@@ -212,12 +222,10 @@ fn restart_gets_new_id_blip_keeps_id() {
|
||||
|
||||
// Blip: the same incarnation reconnects and keeps its id.
|
||||
disconnect("node-b");
|
||||
assert_eq!(
|
||||
next_event(&ev, "node_down inc 2"),
|
||||
NodeEvent::NodeDown { node: id2 }
|
||||
);
|
||||
assert_eq!(down_id(next_event(&ev, "node_down inc 2"), "inc 2"), id2);
|
||||
let (a3, b3) = pair(&t);
|
||||
spawn_established(FramedConn::new(a3), peer("node-b", 2)).expect("registers");
|
||||
spawn_established(FramedConn::new(a3), peer("node-b", 2), Timing::default())
|
||||
.expect("registers");
|
||||
let id3 = id(next_event(&ev, "node_up after blip"), "blip");
|
||||
assert_eq!(
|
||||
id2, id3,
|
||||
@@ -247,7 +255,8 @@ fn dead_subscriber_is_pruned() {
|
||||
|
||||
let t = TcpTransport;
|
||||
let (a1, b1) = pair(&t);
|
||||
spawn_established(FramedConn::new(a1), peer("node-b", 1)).expect("registers");
|
||||
spawn_established(FramedConn::new(a1), peer("node-b", 1), Timing::default())
|
||||
.expect("registers");
|
||||
match next_event(&live, "node_up despite a dead co-subscriber") {
|
||||
NodeEvent::NodeUp(info) => assert_eq!(info.name, "node-b"),
|
||||
other => panic!("expected node_up, got {other:?}"),
|
||||
|
||||
@@ -24,9 +24,7 @@ mod common;
|
||||
use common::{maybe_child, spawn_node, Node};
|
||||
use smarm::cluster::envelope::NodeMeta;
|
||||
use smarm::cluster::membership::{subscribe, NodeEvent};
|
||||
use smarm::cluster::{start, Config, StaticSeeds};
|
||||
use smarm::pg::NodeId;
|
||||
use std::collections::HashMap;
|
||||
use smarm::cluster::{start, Config, StaticSeeds, Timing};
|
||||
use std::time::Duration;
|
||||
|
||||
const ROLES: &[(&str, fn())] = &[("node", role_node)];
|
||||
@@ -56,21 +54,19 @@ fn role_node() {
|
||||
},
|
||||
listen_addr: listen,
|
||||
strategy: Box::new(StaticSeeds::new(seeds)),
|
||||
timing: Timing::default(),
|
||||
})
|
||||
.expect("listener binds");
|
||||
println!("LISTENING {}", cluster.local_addr());
|
||||
|
||||
let events = subscribe().expect("manager is up");
|
||||
let mut names: HashMap<NodeId, String> = HashMap::new();
|
||||
loop {
|
||||
match events.rx.recv() {
|
||||
Ok(NodeEvent::NodeUp(info)) => {
|
||||
names.insert(info.node, info.name.clone());
|
||||
println!("MEMBER-UP {} inc={}", info.name, info.incarnation.get());
|
||||
}
|
||||
Ok(NodeEvent::NodeDown { node }) => {
|
||||
let name = names.remove(&node).unwrap_or_else(|| "?".to_string());
|
||||
println!("MEMBER-DOWN {name}");
|
||||
Ok(NodeEvent::NodeDown(info)) => {
|
||||
println!("MEMBER-DOWN {}", info.name);
|
||||
}
|
||||
Err(_) => break, // manager gone; park below regardless
|
||||
}
|
||||
|
||||
@@ -0,0 +1,359 @@
|
||||
//! RFC 010 c12 — remote monitors.
|
||||
//!
|
||||
//! Local suite (`run()`, no network): the immediate answers — no connection
|
||||
//! ⇒ `Disconnected`, dead incarnation ⇒ `NoProc` — and the self-node
|
||||
//! collapse (a plain local monitor underneath, incl. `demonitor_remote`).
|
||||
//!
|
||||
//! Cross-process: a *server* exposes a control name and spawns workers on
|
||||
//! request, replying with each worker's pid (via `RemotePid::from_local`,
|
||||
//! the D12 set-site) or, for the deliberately unshipped one, only its raw
|
||||
//! slot numbers. The *client* monitors them and asserts: kill ⇒ the true
|
||||
//! reason (Exit / Panic); a corpse ⇒ its recorded terminal reason, not
|
||||
//! NoProc; a live pid that never crossed the wire ⇒ NoProc (no liveness
|
||||
//! leak); a demonitor racing the kill ⇒ no notice, proven by stream ORDER
|
||||
//! (a later notice on the same connection arrives while the earlier slot
|
||||
//! is still empty), not by sleeping.
|
||||
#![cfg(feature = "cluster")]
|
||||
|
||||
mod common;
|
||||
|
||||
use common::{maybe_child, spawn_node};
|
||||
use smarm::cluster::envelope::NodeMeta;
|
||||
use smarm::cluster::expose::{expose, expose_type};
|
||||
use smarm::cluster::membership::{subscribe, NodeEvent};
|
||||
use smarm::cluster::remote::{
|
||||
self, demonitor_remote, monitor_remote, send_to_remote, RemoteName, RemotePid,
|
||||
};
|
||||
use smarm::cluster::{start, Config, RemoteDownReason, StaticSeeds, Timing};
|
||||
use smarm::pg::Incarnation;
|
||||
use smarm::{channel, install, register, run, spawn, Addressable, DownReason, Erased, Name, Pid};
|
||||
use std::collections::HashMap;
|
||||
use std::time::Duration;
|
||||
|
||||
// ---- message types (hand-rolled serde; the crate is derive-less) ---------
|
||||
|
||||
#[derive(Debug)]
|
||||
struct Ctl {
|
||||
cmd: String,
|
||||
reply_to: RemotePid<Client>,
|
||||
}
|
||||
#[derive(Debug)]
|
||||
struct Answer {
|
||||
text: String,
|
||||
pid: Option<RemotePid<Erased>>,
|
||||
}
|
||||
struct Client;
|
||||
impl Addressable for Client {
|
||||
type Msg = Answer;
|
||||
}
|
||||
|
||||
impl serde::Serialize for Ctl {
|
||||
fn serialize<S: serde::Serializer>(&self, s: S) -> Result<S::Ok, S::Error> {
|
||||
use serde::ser::SerializeTuple;
|
||||
let mut t = s.serialize_tuple(2)?;
|
||||
t.serialize_element(&self.cmd)?;
|
||||
t.serialize_element(&self.reply_to)?;
|
||||
t.end()
|
||||
}
|
||||
}
|
||||
impl<'de> serde::Deserialize<'de> for Ctl {
|
||||
fn deserialize<D: serde::Deserializer<'de>>(d: D) -> Result<Self, D::Error> {
|
||||
let (cmd, reply_to) = <(String, RemotePid<Client>)>::deserialize(d)?;
|
||||
Ok(Ctl { cmd, reply_to })
|
||||
}
|
||||
}
|
||||
impl serde::Serialize for Answer {
|
||||
fn serialize<S: serde::Serializer>(&self, s: S) -> Result<S::Ok, S::Error> {
|
||||
use serde::ser::SerializeTuple;
|
||||
let mut t = s.serialize_tuple(2)?;
|
||||
t.serialize_element(&self.text)?;
|
||||
t.serialize_element(&self.pid)?;
|
||||
t.end()
|
||||
}
|
||||
}
|
||||
impl<'de> serde::Deserialize<'de> for Answer {
|
||||
fn deserialize<D: serde::Deserializer<'de>>(d: D) -> Result<Self, D::Error> {
|
||||
let (text, pid) = <(String, Option<RemotePid<Erased>>)>::deserialize(d)?;
|
||||
Ok(Answer { text, pid })
|
||||
}
|
||||
}
|
||||
|
||||
// ================= local suite =========================================
|
||||
|
||||
/// No connection to the pid's node: `Disconnected` at once — the remote
|
||||
/// analog of NoProc, and the first thing c11's variant is for.
|
||||
#[test]
|
||||
fn unconnected_node_is_disconnected_immediately() {
|
||||
maybe_child(ROLES);
|
||||
run(|| {
|
||||
remote::set_local_identity("me", Incarnation::new(7));
|
||||
let ghost = RemotePid::<Erased>::from_parts("nowhere", Incarnation::new(1), 3, 1);
|
||||
let m = monitor_remote(ghost.clone());
|
||||
let d = m.recv().unwrap();
|
||||
assert_eq!(d.pid, ghost);
|
||||
assert_eq!(d.reason, RemoteDownReason::Disconnected);
|
||||
});
|
||||
}
|
||||
|
||||
/// The node is connected but the pid names an earlier incarnation: the
|
||||
/// actor is a known corpse (RFC v2 §3), so `NoProc` at once — never
|
||||
/// `Disconnected`, nothing on the wire.
|
||||
#[test]
|
||||
fn dead_incarnation_is_noproc_immediately() {
|
||||
maybe_child(ROLES);
|
||||
run(|| {
|
||||
remote::set_local_identity("me", Incarnation::new(7));
|
||||
let (probe_tx, probe_rx) = channel();
|
||||
remote::bind_outbound_probe("peer", Incarnation::new(5), probe_tx);
|
||||
let stale = RemotePid::<Erased>::from_parts("peer", Incarnation::new(4), 9, 1);
|
||||
let m = monitor_remote(stale);
|
||||
assert_eq!(m.recv().unwrap().reason, DownReason::NoProc.into());
|
||||
assert!(probe_rx.try_recv().unwrap().is_none(), "no frame emitted");
|
||||
});
|
||||
}
|
||||
|
||||
/// A self-node pid collapses to an ordinary local monitor: the true reason
|
||||
/// on exit, and `demonitor_remote` cancels it.
|
||||
#[test]
|
||||
fn self_node_pid_collapses_to_local_monitor() {
|
||||
maybe_child(ROLES);
|
||||
run(|| {
|
||||
remote::set_local_identity("me", Incarnation::new(7));
|
||||
let (go_tx, go_rx) = channel::<()>();
|
||||
let (go2_tx, go2_rx) = channel::<()>();
|
||||
let a = spawn(move || {
|
||||
let _ = go_rx.recv();
|
||||
})
|
||||
.pid();
|
||||
let b = spawn(move || {
|
||||
let _ = go2_rx.recv();
|
||||
})
|
||||
.pid();
|
||||
let ma = monitor_remote(RemotePid::from_local(a).expect("identity set"));
|
||||
let mb = monitor_remote(RemotePid::from_local(b).expect("identity set"));
|
||||
assert_ne!(ma.id, mb.id);
|
||||
assert!(ma.target.local() == Some(a));
|
||||
|
||||
demonitor_remote(&mb);
|
||||
go2_tx.send(()).unwrap();
|
||||
go_tx.send(()).unwrap();
|
||||
let d = ma.recv().unwrap();
|
||||
assert_eq!(d.reason, DownReason::Exit.into());
|
||||
assert_eq!(d.pid.local(), Some(a));
|
||||
// `a` is down (its notice arrived), and `b` was killed first on the
|
||||
// same scheduler — a notice for `b` would be here by now. After a
|
||||
// demonitor the channel is closed-empty (`Err`), like the local one.
|
||||
assert!(matches!(mb.try_recv(), Ok(None) | Err(_)));
|
||||
});
|
||||
}
|
||||
|
||||
// ================= cross-process ======================================
|
||||
|
||||
const ROLES: &[(&str, fn())] = &[("server", role_server), ("client", role_client)];
|
||||
|
||||
const CTL: Name<Ctl> = Name::new("c12.ctl");
|
||||
|
||||
fn cfg(name: &str, seeds: Vec<(String, String)>) -> Config {
|
||||
Config {
|
||||
node_name: name.to_string(),
|
||||
meta: NodeMeta {
|
||||
role: "c12".into(),
|
||||
region: "local".into(),
|
||||
},
|
||||
listen_addr: std::env::var("SMARM_LISTEN_ADDR").unwrap_or_else(|_| "127.0.0.1:0".into()),
|
||||
strategy: Box::new(StaticSeeds::new(seeds)),
|
||||
timing: Timing::default(),
|
||||
}
|
||||
}
|
||||
|
||||
fn wait_up(events: &smarm::cluster::membership::MembershipEvents, who: &str) {
|
||||
loop {
|
||||
match events.rx.recv() {
|
||||
Ok(NodeEvent::NodeUp(i)) if i.name == who => return,
|
||||
Ok(_) => continue,
|
||||
Err(_) => panic!("manager gone"),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Server commands (all answered to `reply_to`):
|
||||
/// - `spawn:exit` / `spawn:panic` — a parked worker; `kill:<index>` releases
|
||||
/// it, whereupon it returns / panics. Answer carries its pid.
|
||||
/// - `spawn:corpse` — a worker that has already exited when the answer is
|
||||
/// sent; the pid was shipped (watchable) before it died.
|
||||
/// - `spawn:unwatched` — a parked worker whose pid is NEVER shipped; the
|
||||
/// answer carries only `text = "slot:<index>:<generation>"`.
|
||||
fn role_server() {
|
||||
smarm::run(move || {
|
||||
let cluster = start(cfg("server", vec![])).expect("binds");
|
||||
println!("LISTENING {}", cluster.local_addr());
|
||||
let (tx, rx) = channel::<Ctl>();
|
||||
register(CTL, tx).unwrap();
|
||||
expose(CTL);
|
||||
println!("READY");
|
||||
let mut workers: HashMap<u32, smarm::channel::Sender<()>> = HashMap::new();
|
||||
loop {
|
||||
let ctl = rx.recv().unwrap();
|
||||
println!("CTL {}", ctl.cmd);
|
||||
let (text, pid): (String, Option<RemotePid<Erased>>) = match ctl.cmd.as_str() {
|
||||
"spawn:exit" | "spawn:panic" => {
|
||||
let panic = ctl.cmd == "spawn:panic";
|
||||
let (go_tx, go_rx) = channel::<()>();
|
||||
let p: Pid = spawn(move || {
|
||||
let _ = go_rx.recv();
|
||||
if panic {
|
||||
panic!("worker asked to panic");
|
||||
}
|
||||
})
|
||||
.pid();
|
||||
workers.insert(p.index(), go_tx);
|
||||
(
|
||||
"ok".into(),
|
||||
Some(RemotePid::from_local(p).expect("identity set")),
|
||||
)
|
||||
}
|
||||
"spawn:corpse" => {
|
||||
let p: Pid = spawn(|| {}).pid();
|
||||
let rp = RemotePid::from_local(p).expect("identity set"); // shipped ⇒ watchable
|
||||
let m = smarm::monitor(p);
|
||||
let _ = m.rx.recv(); // dead before the answer goes out
|
||||
("ok".into(), Some(rp))
|
||||
}
|
||||
"spawn:unwatched" => {
|
||||
let (go_tx, go_rx) = channel::<()>();
|
||||
let p: Pid = spawn(move || {
|
||||
let _ = go_rx.recv();
|
||||
})
|
||||
.pid();
|
||||
workers.insert(p.index(), go_tx);
|
||||
(format!("slot:{}:{}", p.index(), p.generation()), None)
|
||||
}
|
||||
other => {
|
||||
let idx: u32 = other.strip_prefix("kill:").unwrap().parse().unwrap();
|
||||
if let Some(go) = workers.remove(&idx) {
|
||||
let _ = go.send(());
|
||||
}
|
||||
("killed".into(), None)
|
||||
}
|
||||
};
|
||||
send_to_remote(ctl.reply_to, Answer { text, pid }).unwrap();
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
fn role_client() {
|
||||
let server_addr = std::env::var("SMARM_SERVER_ADDR").expect("SMARM_SERVER_ADDR");
|
||||
smarm::run(move || {
|
||||
let _cluster = start(cfg("client", vec![("server".into(), server_addr)])).expect("binds");
|
||||
let ev = subscribe().unwrap();
|
||||
wait_up(&ev, "server");
|
||||
let (tx, rx) = channel::<Answer>();
|
||||
let me: Pid<Client> = install::<Client>(tx);
|
||||
expose_type::<Answer>();
|
||||
let ask = |cmd: &str| -> Answer {
|
||||
remote::send(
|
||||
RemoteName::new("server", CTL),
|
||||
Ctl {
|
||||
cmd: cmd.into(),
|
||||
reply_to: RemotePid::from_local(me).expect("identity set"),
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
rx.recv().unwrap()
|
||||
};
|
||||
let server_inc = ev_incarnation();
|
||||
|
||||
// 1. kill ⇒ true reason (Exit).
|
||||
let a = ask("spawn:exit").pid.unwrap();
|
||||
let ma = monitor_remote(a.clone());
|
||||
ask(&format!("kill:{}", a.index()));
|
||||
let d = ma.recv().unwrap();
|
||||
assert_eq!(d.pid, a);
|
||||
println!("DOWN exit {:?}", d.reason);
|
||||
|
||||
// 2. kill ⇒ true reason (Panic).
|
||||
let b = ask("spawn:panic").pid.unwrap();
|
||||
let mb = monitor_remote(b.clone());
|
||||
ask(&format!("kill:{}", b.index()));
|
||||
println!("DOWN panic {:?}", mb.recv().unwrap().reason);
|
||||
|
||||
// 3. corpse ⇒ recorded terminal reason, not NoProc.
|
||||
let c = ask("spawn:corpse").pid.unwrap();
|
||||
println!("DOWN corpse {:?}", monitor_remote(c).recv().unwrap().reason);
|
||||
|
||||
// 4. live but never shipped/exposed ⇒ NoProc (no leak); a made-up
|
||||
// slot on the same node ⇒ NoProc too, indistinguishably.
|
||||
let ans = ask("spawn:unwatched");
|
||||
let mut it = ans.text.strip_prefix("slot:").unwrap().split(':');
|
||||
let (idx, gen): (u32, u32) = (
|
||||
it.next().unwrap().parse().unwrap(),
|
||||
it.next().unwrap().parse().unwrap(),
|
||||
);
|
||||
let hidden = RemotePid::<Erased>::from_parts("server", server_inc, idx, gen);
|
||||
println!(
|
||||
"DOWN hidden {:?}",
|
||||
monitor_remote(hidden).recv().unwrap().reason
|
||||
);
|
||||
let bogus = RemotePid::<Erased>::from_parts("server", server_inc, 100_000, 1);
|
||||
println!(
|
||||
"DOWN bogus {:?}",
|
||||
monitor_remote(bogus).recv().unwrap().reason
|
||||
);
|
||||
|
||||
// 5. demonitor races the kill: no notice for `d1`, proven by order —
|
||||
// `d2`'s notice (same connection, later) arrives while `d1`'s
|
||||
// slot is still empty.
|
||||
let d1 = ask("spawn:exit").pid.unwrap();
|
||||
let m1 = monitor_remote(d1.clone());
|
||||
demonitor_remote(&m1);
|
||||
ask(&format!("kill:{}", d1.index()));
|
||||
let d2 = ask("spawn:exit").pid.unwrap();
|
||||
let m2 = monitor_remote(d2.clone());
|
||||
ask(&format!("kill:{}", d2.index()));
|
||||
assert_eq!(m2.recv().unwrap().reason, DownReason::Exit.into());
|
||||
// Closed-empty (`Err`) or open-empty (`Ok(None)`) both mean no notice.
|
||||
let stray = matches!(m1.try_recv(), Ok(Some(_)));
|
||||
println!("DEMONITOR stray={stray}");
|
||||
|
||||
println!("CLIENT DONE");
|
||||
loop {
|
||||
smarm::sleep(Duration::from_secs(3600));
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/// The server's incarnation as this node sees it — for building pids by hand.
|
||||
fn ev_incarnation() -> Incarnation {
|
||||
smarm::cluster::membership::view()
|
||||
.expect("manager up")
|
||||
.into_iter()
|
||||
.find(|i| i.name == "server")
|
||||
.map(|i| i.incarnation)
|
||||
.expect("server in view")
|
||||
}
|
||||
|
||||
/// The Phase 4 c12 gate: remote monitors report the true reason, honour
|
||||
/// corpses, leak nothing for unshipped pids, and cancel cleanly.
|
||||
#[test]
|
||||
fn remote_monitors_report_true_reasons() {
|
||||
maybe_child(ROLES);
|
||||
let mut server = spawn_node("server", &[]);
|
||||
let saddr = server.wait_listening();
|
||||
server.wait_line("READY", |l| l == "READY");
|
||||
let mut client = spawn_node("client", &[("SMARM_SERVER_ADDR", &saddr)]);
|
||||
client.wait_line("DOWN exit Local(Exit)", |l| l == "DOWN exit Local(Exit)");
|
||||
client.wait_line("DOWN panic Local(Panic)", |l| {
|
||||
l == "DOWN panic Local(Panic)"
|
||||
});
|
||||
client.wait_line("DOWN corpse Local(Exit)", |l| {
|
||||
l == "DOWN corpse Local(Exit)"
|
||||
});
|
||||
client.wait_line("DOWN hidden Local(NoProc)", |l| {
|
||||
l == "DOWN hidden Local(NoProc)"
|
||||
});
|
||||
client.wait_line("DOWN bogus Local(NoProc)", |l| {
|
||||
l == "DOWN bogus Local(NoProc)"
|
||||
});
|
||||
client.wait_line("DEMONITOR stray=false", |l| l == "DEMONITOR stray=false");
|
||||
client.wait_line("CLIENT DONE", |l| l == "CLIENT DONE");
|
||||
}
|
||||
@@ -0,0 +1,254 @@
|
||||
//! RFC 010 c15 — distributed pg: sync on `NodeUp`, incremental
|
||||
//! `Join`/`Leave`, eager eviction announced, `NodeDown` sweep.
|
||||
//!
|
||||
//! Two nodes. The *origin* joins two local workers to `"pool"` before the
|
||||
//! *observer* connects (so the observer's view comes from `Sync`), exposes a
|
||||
//! `"go"` command inbox and then does exactly what the observer tells it:
|
||||
//! kill one worker, join a third, leave with the second. The observer drives
|
||||
//! that script through the cluster itself and asserts every step from
|
||||
//! `members_all` — never touching the group on its own side, except once to
|
||||
//! prove a mixed local+remote group reads correctly and that `members` stays
|
||||
//! local. `dispatch_any` is exercised both ways: into the origin's worker
|
||||
//! (remote pick, `send_to_remote`) and, once the origin is gone, into the
|
||||
//! observer's own (local pick, `send_to`). Finally the parent SIGKILLs the
|
||||
//! origin: the observer must sweep every remote member on `NodeDown`.
|
||||
#![cfg(feature = "cluster")]
|
||||
|
||||
mod common;
|
||||
|
||||
use common::{maybe_child, spawn_node};
|
||||
use smarm::cluster::envelope::NodeMeta;
|
||||
use smarm::cluster::expose::{expose, expose_type};
|
||||
use smarm::cluster::membership::{subscribe, NodeEvent};
|
||||
use smarm::cluster::remote::{self, RemoteName};
|
||||
use smarm::cluster::{
|
||||
dispatch_any, members_all, pick_any, start, Config, DispatchAnyError, GroupMember, StaticSeeds,
|
||||
Timing,
|
||||
};
|
||||
use smarm::{channel, join, leave, members, register, send_to, spawn_addr, Addressable, Name, Pid};
|
||||
use std::time::{Duration, Instant};
|
||||
|
||||
const GO: Name<u8> = Name::new("go");
|
||||
const POOL: &str = "pool";
|
||||
|
||||
/// A pool worker's message: `"die"` stops it, anything else is printed.
|
||||
#[derive(Debug, PartialEq)]
|
||||
struct Job(String);
|
||||
struct Worker;
|
||||
impl Addressable for Worker {
|
||||
type Msg = Job;
|
||||
}
|
||||
impl serde::Serialize for Job {
|
||||
fn serialize<S: serde::Serializer>(&self, s: S) -> Result<S::Ok, S::Error> {
|
||||
self.0.serialize(s)
|
||||
}
|
||||
}
|
||||
impl<'de> serde::Deserialize<'de> for Job {
|
||||
fn deserialize<D: serde::Deserializer<'de>>(d: D) -> Result<Self, D::Error> {
|
||||
String::deserialize(d).map(Job)
|
||||
}
|
||||
}
|
||||
|
||||
const ROLES: &[(&str, fn())] = &[("origin", role_origin), ("observer", role_observer)];
|
||||
|
||||
fn cfg(name: &str, seeds: Vec<(String, String)>) -> Config {
|
||||
Config {
|
||||
node_name: name.into(),
|
||||
meta: NodeMeta {
|
||||
role: "c15".into(),
|
||||
region: "local".into(),
|
||||
},
|
||||
listen_addr: "127.0.0.1:0".into(),
|
||||
strategy: Box::new(StaticSeeds::new(seeds)),
|
||||
timing: Timing::default(),
|
||||
}
|
||||
}
|
||||
|
||||
/// A pool worker: prints every job it is handed, exits on `"die"`.
|
||||
fn worker() -> Pid<Worker> {
|
||||
spawn_addr::<Worker>(|rx| {
|
||||
while let Ok(Job(s)) = rx.recv() {
|
||||
if s == "die" {
|
||||
return;
|
||||
}
|
||||
println!("JOB {s}");
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
fn role_origin() {
|
||||
smarm::run(|| {
|
||||
let cluster = start(cfg("origin", vec![])).expect("binds");
|
||||
// Remote dispatch lands here only for a type this node accepts.
|
||||
expose_type::<Job>();
|
||||
let w1 = worker();
|
||||
let w2 = worker();
|
||||
assert!(join(POOL, w1));
|
||||
assert!(join(POOL, w2));
|
||||
let (go_tx, go_rx) = channel::<u8>();
|
||||
register(GO, go_tx).unwrap();
|
||||
expose(GO);
|
||||
println!("LISTENING {}", cluster.local_addr());
|
||||
println!("JOINED 2");
|
||||
loop {
|
||||
match go_rx.recv().unwrap() {
|
||||
1 => {
|
||||
send_to(w1, Job("die".into())).unwrap();
|
||||
println!("KILLED w1");
|
||||
}
|
||||
2 => {
|
||||
assert!(leave(POOL, w2));
|
||||
println!("LEFT w2");
|
||||
}
|
||||
3 => {
|
||||
let w3 = worker();
|
||||
assert!(join(POOL, w3));
|
||||
println!("JOINED w3");
|
||||
}
|
||||
n => panic!("unknown command {n}"),
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
fn remote_count(group: &str) -> usize {
|
||||
members_all(group)
|
||||
.iter()
|
||||
.filter(|m| matches!(m, GroupMember::Remote(_)))
|
||||
.count()
|
||||
}
|
||||
|
||||
/// Cooperative poll until `pred`; panics (with the last view) on timeout.
|
||||
fn wait_view(what: &str, group: &str, pred: impl Fn(&[GroupMember]) -> bool) {
|
||||
let deadline = Instant::now() + Duration::from_secs(5);
|
||||
loop {
|
||||
let v = members_all(group);
|
||||
if pred(&v) {
|
||||
return;
|
||||
}
|
||||
assert!(
|
||||
Instant::now() < deadline,
|
||||
"timed out waiting for {what}; view = {v:?}"
|
||||
);
|
||||
smarm::sleep(Duration::from_millis(5));
|
||||
}
|
||||
}
|
||||
|
||||
fn role_observer() {
|
||||
let origin_addr = std::env::var("SMARM_ORIGIN_ADDR").expect("SMARM_ORIGIN_ADDR");
|
||||
smarm::run(move || {
|
||||
let _cluster = start(cfg("observer", vec![("origin".into(), origin_addr)])).expect("binds");
|
||||
let ev = subscribe().unwrap();
|
||||
loop {
|
||||
match ev.rx.recv().unwrap() {
|
||||
NodeEvent::NodeUp(i) if i.name == "origin" => break,
|
||||
_ => {}
|
||||
}
|
||||
}
|
||||
let go = |n: u8| remote::send(RemoteName::new("origin", GO), n).unwrap();
|
||||
|
||||
// Sync: both pre-existing members arrive with no join on this side.
|
||||
wait_view("sync of 2 remote members", POOL, |v| {
|
||||
v.len() == 2 && v.iter().all(|m| matches!(m, GroupMember::Remote(_)))
|
||||
});
|
||||
let synced = members_all(POOL);
|
||||
assert!(synced.iter().all(|m| match m {
|
||||
GroupMember::Remote(p) => p.node() == "origin",
|
||||
GroupMember::Local(_) => false,
|
||||
}));
|
||||
println!("SEES 2");
|
||||
|
||||
// Origin-side death: the origin's reaper announces the leave.
|
||||
go(1);
|
||||
wait_view("death evicted on observer", POOL, |v| v.len() == 1);
|
||||
println!("SEES 1 after death");
|
||||
|
||||
// Incremental Join.
|
||||
go(3);
|
||||
wait_view("incremental join", POOL, |v| v.len() == 2);
|
||||
println!("SEES 2 after join");
|
||||
|
||||
// Voluntary Leave.
|
||||
go(2);
|
||||
wait_view("incremental leave", POOL, |v| v.len() == 1);
|
||||
println!("SEES 1 after leave");
|
||||
|
||||
// Mixed group: our own member sits beside the remote one in
|
||||
// `members_all`; `members` stays local-only.
|
||||
let me = worker();
|
||||
assert!(join(POOL, me));
|
||||
wait_view("mixed local+remote", POOL, |v| {
|
||||
v.len() == 2 && v.contains(&GroupMember::Local(me.erase()))
|
||||
});
|
||||
assert_eq!(
|
||||
members(POOL),
|
||||
vec![me.erase()],
|
||||
"local API never shows remotes"
|
||||
);
|
||||
assert_eq!(remote_count(POOL), 1);
|
||||
println!("MIXED ok");
|
||||
|
||||
// dispatch_any: the store's first entry is the origin's w3 (it was
|
||||
// announced before we joined), so the pick is remote and the job
|
||||
// crosses the wire — the origin's worker prints it.
|
||||
let picked = pick_any(POOL).expect("pool has members");
|
||||
assert!(
|
||||
matches!(picked, GroupMember::Remote(_)),
|
||||
"first entry is remote: {picked:?}"
|
||||
);
|
||||
let reached = dispatch_any::<Worker>(POOL, Job("from-observer".into())).unwrap();
|
||||
assert_eq!(reached, picked);
|
||||
println!("DISPATCHED remote");
|
||||
|
||||
println!("PARK");
|
||||
// Parent SIGKILLs the origin now: NodeDown must sweep its member,
|
||||
// ours must survive.
|
||||
wait_view("node_down sweep", POOL, |v| {
|
||||
v == [GroupMember::Local(me.erase())]
|
||||
});
|
||||
assert_eq!(members(POOL), vec![me.erase()]);
|
||||
println!("SWEPT");
|
||||
|
||||
// Now the only member is ours: a local pick, a local send.
|
||||
let reached = dispatch_any::<Worker>(POOL, Job("local".into())).unwrap();
|
||||
assert_eq!(reached, GroupMember::Local(me.erase()));
|
||||
// And an empty group hands the message back.
|
||||
match dispatch_any::<Worker>("nobody", Job("lost".into())) {
|
||||
Err(DispatchAnyError::NoMember(Job(s))) => assert_eq!(s, "lost"),
|
||||
other => panic!("expected NoMember, got {other:?}"),
|
||||
}
|
||||
println!("DISPATCHED local");
|
||||
loop {
|
||||
smarm::sleep(Duration::from_secs(3600));
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/// The Phase 5 gate: sync, join, leave, death, node_down — all observed from
|
||||
/// the peer, none of them a group operation on the peer — plus dispatch_any
|
||||
/// reaching a remote member and a local one.
|
||||
#[test]
|
||||
fn groups_span_two_nodes() {
|
||||
maybe_child(ROLES);
|
||||
let mut origin = spawn_node("origin", &[]);
|
||||
let addr = origin.wait_listening();
|
||||
origin.wait_line("JOINED 2", |l| l == "JOINED 2");
|
||||
let mut observer = spawn_node("observer", &[("SMARM_ORIGIN_ADDR", &addr)]);
|
||||
observer.wait_line("SEES 2", |l| l == "SEES 2");
|
||||
origin.wait_line("KILLED w1", |l| l == "KILLED w1");
|
||||
observer.wait_line("SEES 1 after death", |l| l == "SEES 1 after death");
|
||||
origin.wait_line("JOINED w3", |l| l == "JOINED w3");
|
||||
observer.wait_line("SEES 2 after join", |l| l == "SEES 2 after join");
|
||||
origin.wait_line("LEFT w2", |l| l == "LEFT w2");
|
||||
observer.wait_line("SEES 1 after leave", |l| l == "SEES 1 after leave");
|
||||
observer.wait_line("MIXED ok", |l| l == "MIXED ok");
|
||||
observer.wait_line("DISPATCHED remote", |l| l == "DISPATCHED remote");
|
||||
origin.wait_line("JOB from-observer", |l| l == "JOB from-observer");
|
||||
observer.wait_line("PARK", |l| l == "PARK");
|
||||
origin.kill();
|
||||
observer.wait_line("SWEPT", |l| l == "SWEPT");
|
||||
// Order between the root's line and the worker's is scheduling; wait
|
||||
// for the later one to be certain both happened.
|
||||
observer.wait_line("DISPATCHED local", |l| l == "DISPATCHED local");
|
||||
observer.wait_line("JOB local", |l| l == "JOB local");
|
||||
}
|
||||
@@ -0,0 +1,356 @@
|
||||
//! RFC 010 c10 — pid targeting + auto-serialization. The Phase 3 gate:
|
||||
//! cross-node call/reply with no ceremony, under the subprocess harness.
|
||||
//!
|
||||
//! Local suite (`run()`, no network): serialize/deserialize shapes,
|
||||
//! self-collapse, the outside-runtime contract, the local send-site
|
||||
//! incarnation check with a probe proving **no frame is emitted**.
|
||||
//!
|
||||
//! Cross-process: two nodes. The *server* exposes a `Name<Req>`; the
|
||||
//! *client* sends a `Req` carrying its own `Pid<Reply>` (auto-serialized to
|
||||
//! a `RemotePid` on the wire); the server replies via `send_to_remote`
|
||||
//! straight back to that pid — no name at the client end, no ceremony. A
|
||||
//! third-node roundtrip: the client's pid travels client→server→relay→
|
||||
//! server→client, and still delivers.
|
||||
#![cfg(feature = "cluster")]
|
||||
|
||||
mod common;
|
||||
|
||||
use common::{maybe_child, spawn_node};
|
||||
use smarm::cluster::envelope::{encode_payload, Frame, NodeMeta};
|
||||
use smarm::cluster::expose::{expose, type_hash};
|
||||
use smarm::cluster::membership::{subscribe, NodeEvent};
|
||||
use smarm::cluster::remote::{self, send_to_remote, RemoteName, RemotePid, ToRemoteError};
|
||||
use smarm::cluster::{start, Config, StaticSeeds, Timing};
|
||||
use smarm::pg::Incarnation;
|
||||
use smarm::{channel, install, register, run, Addressable, Name, Pid};
|
||||
use std::time::Duration;
|
||||
|
||||
// ---- message types (std-only payloads; the crate's serde is derive-less,
|
||||
// so wire types are hand-rolled with serde's tuple/seq API via `serde::ser`
|
||||
// impls below — the same thing a user's derive would generate) ------------
|
||||
|
||||
/// A request carrying a reply-to. Serialize/Deserialize are written by hand
|
||||
/// here for exactly one reason: this crate deliberately does not pull in
|
||||
/// serde-derive. Field 1 is the auto-serializing pid.
|
||||
#[derive(Debug, PartialEq)]
|
||||
struct Req {
|
||||
text: String,
|
||||
reply_to: RemotePid<Replier>,
|
||||
}
|
||||
|
||||
#[derive(Debug, PartialEq)]
|
||||
struct Reply(String);
|
||||
|
||||
struct Replier;
|
||||
impl Addressable for Replier {
|
||||
type Msg = Reply;
|
||||
}
|
||||
|
||||
impl serde::Serialize for Req {
|
||||
fn serialize<S: serde::Serializer>(&self, s: S) -> Result<S::Ok, S::Error> {
|
||||
use serde::ser::SerializeTuple;
|
||||
let mut t = s.serialize_tuple(2)?;
|
||||
t.serialize_element(&self.text)?;
|
||||
t.serialize_element(&self.reply_to)?;
|
||||
t.end()
|
||||
}
|
||||
}
|
||||
impl<'de> serde::Deserialize<'de> for Req {
|
||||
fn deserialize<D: serde::Deserializer<'de>>(d: D) -> Result<Self, D::Error> {
|
||||
let (text, reply_to) = <(String, RemotePid<Replier>)>::deserialize(d)?;
|
||||
Ok(Req { text, reply_to })
|
||||
}
|
||||
}
|
||||
impl serde::Serialize for Reply {
|
||||
fn serialize<S: serde::Serializer>(&self, s: S) -> Result<S::Ok, S::Error> {
|
||||
self.0.serialize(s)
|
||||
}
|
||||
}
|
||||
impl<'de> serde::Deserialize<'de> for Reply {
|
||||
fn deserialize<D: serde::Deserializer<'de>>(d: D) -> Result<Self, D::Error> {
|
||||
String::deserialize(d).map(Reply)
|
||||
}
|
||||
}
|
||||
|
||||
// ================= local suite =========================================
|
||||
|
||||
/// A local `Pid<A>` serializes as a `RemotePid<A>` stamped with this node's
|
||||
/// identity; deserializing it back on the same node collapses to the same
|
||||
/// local pid (`local()` is `Some`, `Pid` round-trips).
|
||||
#[test]
|
||||
fn local_pid_serializes_and_collapses_on_self() {
|
||||
maybe_child(ROLES);
|
||||
run(|| {
|
||||
// The local identity is set by cluster::start; the local suite sets
|
||||
// it directly.
|
||||
remote::set_local_identity("me", Incarnation::new(7));
|
||||
let (tx, _rx) = channel::<Reply>();
|
||||
let me: Pid<Replier> = install::<Replier>(tx);
|
||||
|
||||
let bytes = encode_payload(&me).unwrap();
|
||||
let rp: RemotePid<Replier> = smarm::cluster::envelope::decode_payload(&bytes).unwrap();
|
||||
assert_eq!(rp.node(), "me");
|
||||
assert_eq!(rp.incarnation(), Incarnation::new(7));
|
||||
assert_eq!(
|
||||
rp.local(),
|
||||
Some(me),
|
||||
"self-node pid collapses to the local pid"
|
||||
);
|
||||
|
||||
// Deserializing straight into Pid<A> works for a self-node pid...
|
||||
let back: Pid<Replier> = smarm::cluster::envelope::decode_payload(&bytes).unwrap();
|
||||
assert_eq!(back, me);
|
||||
|
||||
// ...and FAILS for a foreign one (collapse is literal: node == self).
|
||||
let foreign = RemotePid::<Replier>::from_parts("elsewhere", Incarnation::new(1), 3, 1);
|
||||
let fbytes = encode_payload(&foreign).unwrap();
|
||||
assert!(smarm::cluster::envelope::decode_payload::<Pid<Replier>>(&fbytes).is_err());
|
||||
assert_eq!(foreign.local(), None);
|
||||
});
|
||||
}
|
||||
|
||||
/// `send_to_remote` short-circuits locally for a self-node pid — the
|
||||
/// zero-copy-equivalent collapse: the message object itself lands in the
|
||||
/// local channel, no encode, no frame.
|
||||
#[test]
|
||||
fn send_to_remote_collapses_locally_for_self() {
|
||||
maybe_child(ROLES);
|
||||
run(|| {
|
||||
remote::set_local_identity("me", Incarnation::new(7));
|
||||
let (tx, rx) = channel::<Reply>();
|
||||
let me: Pid<Replier> = install::<Replier>(tx);
|
||||
let rp = RemotePid::from_local(me).expect("identity set");
|
||||
// Probe the outbound path: nothing must be handed to any connection.
|
||||
let (probe_tx, probe_rx) = channel::<Frame>();
|
||||
remote::bind_outbound_probe("me", Incarnation::new(7), probe_tx);
|
||||
|
||||
send_to_remote(rp, Reply("hi".into())).unwrap();
|
||||
assert_eq!(rx.recv().unwrap(), Reply("hi".into()));
|
||||
assert!(
|
||||
matches!(probe_rx.try_recv(), Ok(None)),
|
||||
"no frame for a local collapse"
|
||||
);
|
||||
});
|
||||
}
|
||||
|
||||
/// RFC v2 §3: a `RemotePid` whose incarnation is not the current one for its
|
||||
/// node fails at the local send site with `DeadIncarnation`, and NO frame
|
||||
/// is emitted — asserted on a probe sender bound as that node's outbound.
|
||||
#[test]
|
||||
fn stale_incarnation_rejected_locally_no_frame() {
|
||||
maybe_child(ROLES);
|
||||
run(|| {
|
||||
remote::set_local_identity("me", Incarnation::new(7));
|
||||
let (probe_tx, probe_rx) = channel::<Frame>();
|
||||
remote::bind_outbound_probe("peer", Incarnation::new(5), probe_tx);
|
||||
|
||||
let stale = RemotePid::<Replier>::from_parts("peer", Incarnation::new(4), 9, 1);
|
||||
match send_to_remote(stale, Reply("late".into())) {
|
||||
Err(ToRemoteError::DeadIncarnation(Reply(s))) => assert_eq!(s, "late"),
|
||||
other => panic!("expected DeadIncarnation, got {other:?}"),
|
||||
}
|
||||
assert!(
|
||||
matches!(probe_rx.try_recv(), Ok(None)),
|
||||
"stale pid must emit no frame"
|
||||
);
|
||||
|
||||
// The current incarnation goes through: a Send frame with the pid's
|
||||
// (index, generation) and Reply's hash lands on the probe.
|
||||
let live = RemotePid::<Replier>::from_parts("peer", Incarnation::new(5), 9, 1);
|
||||
send_to_remote(live, Reply("now".into())).unwrap();
|
||||
match probe_rx.recv().unwrap() {
|
||||
Frame::Send {
|
||||
index,
|
||||
generation,
|
||||
type_hash: h,
|
||||
payload,
|
||||
} => {
|
||||
assert_eq!((index, generation), (9, 1));
|
||||
assert_eq!(h, type_hash::<Reply>());
|
||||
let r: Reply = smarm::cluster::envelope::decode_payload(&payload).unwrap();
|
||||
assert_eq!(r, Reply("now".into()));
|
||||
}
|
||||
f => panic!("expected Send, got {f:?}"),
|
||||
}
|
||||
|
||||
// Unknown node: NotConnected, no frame anywhere.
|
||||
let nowhere = RemotePid::<Replier>::from_parts("nowhere", Incarnation::new(1), 1, 1);
|
||||
assert!(matches!(
|
||||
send_to_remote(nowhere, Reply("x".into())),
|
||||
Err(ToRemoteError::NotConnected(_))
|
||||
));
|
||||
});
|
||||
}
|
||||
|
||||
// ================= cross-process gate ==================================
|
||||
|
||||
const ROLES: &[(&str, fn())] = &[
|
||||
("server", role_server),
|
||||
("client", role_client),
|
||||
("relay", role_relay),
|
||||
];
|
||||
|
||||
const ECHO: Name<Req> = Name::new("c10.echo");
|
||||
const RELAY: Name<Req> = Name::new("c10.relay");
|
||||
|
||||
fn cfg(name: &str, seeds: Vec<(String, String)>) -> Config {
|
||||
Config {
|
||||
node_name: name.to_string(),
|
||||
meta: NodeMeta {
|
||||
role: "c10".into(),
|
||||
region: "local".into(),
|
||||
},
|
||||
listen_addr: std::env::var("SMARM_LISTEN_ADDR").unwrap_or_else(|_| "127.0.0.1:0".into()),
|
||||
strategy: Box::new(StaticSeeds::new(seeds)),
|
||||
timing: Timing::default(),
|
||||
}
|
||||
}
|
||||
|
||||
fn wait_up(events: &smarm::cluster::membership::MembershipEvents, who: &str) {
|
||||
loop {
|
||||
match events.rx.recv() {
|
||||
Ok(NodeEvent::NodeUp(i)) if i.name == who => return,
|
||||
Ok(_) => continue,
|
||||
Err(_) => panic!("manager gone"),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Server: exposes ECHO; each Req is answered by `send_to_remote` to its
|
||||
/// reply_to — the server never learns a name for the client. If the Req text
|
||||
/// starts with "via-relay:", it forwards the whole Req (reply_to and all) to
|
||||
/// the relay node instead, which sends it back here; the second arrival is
|
||||
/// answered normally. That is the pid's third-node roundtrip.
|
||||
fn role_server() {
|
||||
let relay_addr = std::env::var("SMARM_RELAY_ADDR").ok();
|
||||
smarm::run(move || {
|
||||
let seeds = relay_addr
|
||||
.map(|a| vec![("relay".to_string(), a)])
|
||||
.unwrap_or_default();
|
||||
let cluster = start(cfg("server", seeds)).expect("binds");
|
||||
println!("LISTENING {}", cluster.local_addr());
|
||||
let (tx, rx) = channel::<Req>();
|
||||
register(ECHO, tx).unwrap();
|
||||
expose(ECHO);
|
||||
println!("READY");
|
||||
loop {
|
||||
let req = rx.recv().unwrap();
|
||||
if let Some(rest) = req.text.strip_prefix("via-relay:") {
|
||||
let fwd = Req {
|
||||
text: format!("relayed:{rest}"),
|
||||
reply_to: req.reply_to,
|
||||
};
|
||||
remote::send(RemoteName::new("relay", RELAY), fwd).unwrap();
|
||||
println!("FORWARDED");
|
||||
continue;
|
||||
}
|
||||
println!("REQ {}", req.text);
|
||||
send_to_remote(req.reply_to, Reply(format!("echo:{}", req.text))).unwrap();
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/// Relay: exposes RELAY; bounces every Req straight back to the server's
|
||||
/// ECHO, untouched. The client's pid inside it now crosses relay→server.
|
||||
fn role_relay() {
|
||||
let server_addr = std::env::var("SMARM_SERVER_ADDR").expect("SMARM_SERVER_ADDR");
|
||||
smarm::run(move || {
|
||||
let cluster = start(cfg("relay", vec![("server".into(), server_addr)])).expect("binds");
|
||||
println!("LISTENING {}", cluster.local_addr());
|
||||
let (tx, rx) = channel::<Req>();
|
||||
register(RELAY, tx).unwrap();
|
||||
expose(RELAY);
|
||||
let ev = subscribe().unwrap();
|
||||
wait_up(&ev, "server");
|
||||
println!("READY");
|
||||
loop {
|
||||
let req = rx.recv().unwrap();
|
||||
println!("RELAYING {}", req.text);
|
||||
remote::send(RemoteName::new("server", ECHO), req).unwrap();
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/// Client: connects to server, installs a Reply inbox on its own pid,
|
||||
/// declares it accepts `Reply` (`expose_type` — the RFC's one kept piece of
|
||||
/// ceremony: nothing is remotely deliverable by default), sends a Req with
|
||||
/// `reply_to = my pid` (auto-serialized), awaits the reply.
|
||||
fn role_client() {
|
||||
let server_addr = std::env::var("SMARM_SERVER_ADDR").expect("SMARM_SERVER_ADDR");
|
||||
let via_relay = std::env::var("SMARM_VIA_RELAY").is_ok();
|
||||
smarm::run(move || {
|
||||
let _cluster = start(cfg("client", vec![("server".into(), server_addr)])).expect("binds");
|
||||
let ev = subscribe().unwrap();
|
||||
wait_up(&ev, "server");
|
||||
println!("MEMBER-UP server");
|
||||
|
||||
let (tx, rx) = channel::<Reply>();
|
||||
let me: Pid<Replier> = install::<Replier>(tx);
|
||||
// The one deliberate line: a pid-targeted inbound is deliverable only
|
||||
// for types this node has said it accepts (RFC §4, the safety).
|
||||
smarm::cluster::expose::expose_type::<Reply>();
|
||||
let text = if via_relay { "via-relay:ping" } else { "ping" };
|
||||
remote::send(
|
||||
RemoteName::new("server", ECHO),
|
||||
Req {
|
||||
text: text.into(),
|
||||
reply_to: RemotePid::from_local(me).expect("identity set"),
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
println!("SENT");
|
||||
let Reply(s) = rx.recv().unwrap();
|
||||
println!("REPLY {s}");
|
||||
loop {
|
||||
smarm::sleep(Duration::from_secs(3600));
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/// The gate: cross-node call/reply with no ceremony.
|
||||
#[test]
|
||||
fn cross_node_call_reply_no_ceremony() {
|
||||
maybe_child(ROLES);
|
||||
let mut server = spawn_node("server", &[]);
|
||||
let saddr = server.wait_listening();
|
||||
server.wait_line("READY", |l| l == "READY");
|
||||
let mut client = spawn_node("client", &[("SMARM_SERVER_ADDR", &saddr)]);
|
||||
client.wait_line("SENT", |l| l == "SENT");
|
||||
server.wait_line("REQ ping", |l| l == "REQ ping");
|
||||
client.wait_line("REPLY echo:ping", |l| l == "REPLY echo:ping");
|
||||
}
|
||||
|
||||
/// The client's pid, round-tripped through a third node, still delivers.
|
||||
#[test]
|
||||
fn pid_roundtrips_through_third_node() {
|
||||
maybe_child(ROLES);
|
||||
// Relay needs the server address; server needs the relay address —
|
||||
// pre-reserve the relay port (same accepted micro-window as cluster_mesh).
|
||||
let relay_addr = {
|
||||
let l = std::net::TcpListener::bind("127.0.0.1:0").unwrap();
|
||||
l.local_addr().unwrap().to_string()
|
||||
};
|
||||
let mut server = spawn_node("server", &[("SMARM_RELAY_ADDR", &relay_addr)]);
|
||||
let saddr = server.wait_listening();
|
||||
server.wait_line("READY", |l| l == "READY");
|
||||
let mut relay = spawn_node(
|
||||
"relay",
|
||||
&[
|
||||
("SMARM_SERVER_ADDR", &saddr),
|
||||
("SMARM_LISTEN_ADDR", &relay_addr),
|
||||
],
|
||||
);
|
||||
let _ = relay.wait_listening();
|
||||
relay.wait_line("READY", |l| l == "READY");
|
||||
let mut client = spawn_node(
|
||||
"client",
|
||||
&[("SMARM_SERVER_ADDR", &saddr), ("SMARM_VIA_RELAY", "1")],
|
||||
);
|
||||
client.wait_line("SENT", |l| l == "SENT");
|
||||
server.wait_line("FORWARDED", |l| l == "FORWARDED");
|
||||
relay.wait_line("RELAYING", |l| l.starts_with("RELAYING"));
|
||||
server.wait_line("REQ relayed:ping", |l| l == "REQ relayed:ping");
|
||||
client.wait_line("REPLY echo:relayed:ping", |l| {
|
||||
l == "REPLY echo:relayed:ping"
|
||||
});
|
||||
}
|
||||
@@ -33,7 +33,7 @@ use smarm::cluster::envelope::NodeMeta;
|
||||
use smarm::cluster::expose::expose;
|
||||
use smarm::cluster::membership::{subscribe, NodeEvent};
|
||||
use smarm::cluster::remote::{send_remote_raw, RemoteName, RemoteSendError};
|
||||
use smarm::cluster::{start, Config, StaticSeeds};
|
||||
use smarm::cluster::{start, Config, StaticSeeds, Timing};
|
||||
use smarm::{channel, register, Name};
|
||||
use std::time::Duration;
|
||||
|
||||
@@ -51,6 +51,7 @@ fn base_config(name: &str, seeds: Vec<(String, String)>) -> Config {
|
||||
},
|
||||
listen_addr: "127.0.0.1:0".to_string(),
|
||||
strategy: Box::new(StaticSeeds::new(seeds)),
|
||||
timing: Timing::default(),
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+13
-20
@@ -1,5 +1,6 @@
|
||||
//! Process-group tests that run under the scheduler: `join` installs a real
|
||||
//! monitor on a live actor, and a real death drives eviction on next contact.
|
||||
//! monitor on a live actor, and a real death drives eviction (the reaper
|
||||
//! actor sweeps it; the read path hides it in the meantime).
|
||||
//! (Pure structural invariants live in the `pg` unit tests.)
|
||||
|
||||
use smarm::{channel, members, pick, run, spawn};
|
||||
@@ -35,19 +36,15 @@ fn a_dead_actor_vanishes_from_every_group_it_joined() {
|
||||
assert_eq!(members("g1"), vec![pid]);
|
||||
assert_eq!(members("g2"), vec![pid]);
|
||||
|
||||
// Release and reap the actor. finalize_actor queues the Down to our
|
||||
// monitors before unparking joiners, so by the time join() returns the
|
||||
// Down is already waiting in the membership channel.
|
||||
// Release the actor. finalize_actor queues the Down to the reaper and
|
||||
// marks the slot dead before unparking joiners, so by the time join()
|
||||
// returns every read hides the pid whether or not the reaper has run.
|
||||
tx.send(()).unwrap();
|
||||
h.join().unwrap();
|
||||
|
||||
// Drain-on-contact: touching g1 detects the death and sweeps the pid
|
||||
// out of every group (g2 included), not just g1.
|
||||
assert!(members("g1").is_empty(), "evicted from the touched group");
|
||||
assert!(
|
||||
members("g2").is_empty(),
|
||||
"and swept from the untouched group"
|
||||
);
|
||||
// Gone from every group it joined, not just one.
|
||||
assert!(members("g1").is_empty(), "gone from g1");
|
||||
assert!(members("g2").is_empty(), "and from g2");
|
||||
assert_eq!(pick("g1"), None);
|
||||
});
|
||||
}
|
||||
@@ -86,11 +83,7 @@ fn live_members_survive_a_peers_death() {
|
||||
tx_a.send(()).unwrap();
|
||||
a.join().unwrap();
|
||||
|
||||
assert_eq!(
|
||||
members("svc"),
|
||||
vec![b.pid()],
|
||||
"only the dead peer is reaped"
|
||||
);
|
||||
assert_eq!(members("svc"), vec![b.pid()], "only the dead peer is gone");
|
||||
assert_eq!(pick("svc"), Some(b.pid()));
|
||||
|
||||
tx_b.send(()).unwrap();
|
||||
@@ -123,18 +116,18 @@ fn leave_drops_a_membership_without_affecting_others() {
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn joining_an_already_dead_pid_is_evicted_on_next_contact() {
|
||||
fn joining_an_already_dead_pid_never_shows_in_a_read() {
|
||||
run(|| {
|
||||
let h = spawn(|| {});
|
||||
let pid = h.pid();
|
||||
h.join().unwrap(); // actor is finalized before we join it to anything
|
||||
|
||||
// monitor() on a gone pid queues a NoProc Down immediately, so the
|
||||
// membership is reaped the next time the group is touched.
|
||||
// join() on a gone pid queues a NoProc Down to the reaper immediately;
|
||||
// reads never show it either way (slot-liveness backstop).
|
||||
join("late", pid);
|
||||
assert!(
|
||||
members("late").is_empty(),
|
||||
"dead-at-join member is reaped on read"
|
||||
"dead-at-join member never reads as live"
|
||||
);
|
||||
assert_eq!(pick("late"), None);
|
||||
});
|
||||
|
||||
Reference in New Issue
Block a user