The timeout arm of the connection actor's select: HEARTBEAT_INTERVAL (1s) paces outbound Frame::Heartbeat (first at spawn, so the peer's window starts fed) and LIVENESS_TIMEOUT (4s = 4 intervals) declares the peer dead when no inbound frame arrives inside it — any frame resets the window, so heartbeats keep an idle connection alive and real traffic (c8+) counts for free. Fire => close + exit; the manager's monitor reaps the table entry as on every other exit path. Fixed timeout per RFC v2 §5 (control connection, heartbeats can't queue behind bulk). Intervals are the one-viable-answer call flagged for veto at diff review. The pump was made non-blocking to keep the deadlines honest: a plain recv() blocks into the socket while the buffer holds a partial frame, parking the actor past its timers. Two additive FramedConn methods (read_once, next_buffered): exactly one socket read per level-triggered readable wake (cannot block, cannot strand — leftovers re-signal), then drain every complete buffered frame. Liveness resets only on complete frames. No-fd transports (loopback) still get the command-only loop: no readiness means no timers, same caveat as recv_deadline. tests/cluster_conn_liveness.rs 3/0, stable over 5 runs (raw far end over localhost TCP: heartbeats appear unprompted; mute peer still up at half the window, gone after it; heartbeat-only peer survives 1.5x the window, then reaped once silenced). Cluster suites regression-clean; clippy --lib green both configs; fmt clean.
179 lines
6.7 KiB
Rust
179 lines
6.7 KiB
Rust
//! RFC 010 c6c — heartbeat send + fixed-timeout liveness + teardown.
|
|
//!
|
|
//! Each case runs one real connection actor over an in-process localhost TCP
|
|
//! pair, with the far end held as a raw `FramedConn` (no actor) so the test
|
|
//! controls exactly what — if anything — the peer says. That gives the three
|
|
//! protocol-visible facts direct handles: heartbeats appear on the wire
|
|
//! unprompted; a mute peer is torn down (and reaped from the manager table)
|
|
//! once `LIVENESS_TIMEOUT` empties; and a peer that does nothing but send
|
|
//! heartbeats keeps the connection alive past that same window.
|
|
//!
|
|
//! Loopback has no fd and cannot drive liveness (documented on the actor),
|
|
//! so everything here is TCP. TCP parks the calling actor, so everything
|
|
//! runs inside `smarm::run`.
|
|
#![cfg(feature = "cluster")]
|
|
|
|
use std::time::{Duration, Instant};
|
|
|
|
use smarm::cluster::conn::{HEARTBEAT_INTERVAL, LIVENESS_TIMEOUT};
|
|
use smarm::cluster::envelope::{Frame, NodeMeta};
|
|
use smarm::cluster::handshake::Peer;
|
|
use smarm::cluster::manager::{Call, Manager, Reply, MANAGER};
|
|
use smarm::cluster::spawn_established;
|
|
use smarm::cluster::transport::tcp::TcpTransport;
|
|
use smarm::cluster::transport::{Conn, FramedConn, Transport};
|
|
use smarm::gen_server::{self, GenServerBuilder};
|
|
use smarm::pg::Incarnation;
|
|
use smarm::{run, sleep, spawn};
|
|
|
|
/// A fabricated post-handshake peer identity (same shape as the c6a suite).
|
|
fn peer(name: &str) -> Peer {
|
|
Peer {
|
|
node_name: name.to_string(),
|
|
incarnation: Incarnation::new(1),
|
|
meta: NodeMeta {
|
|
role: "test".to_string(),
|
|
region: "test".to_string(),
|
|
},
|
|
}
|
|
}
|
|
|
|
/// One established transport pair over localhost (TCP backlog covers the
|
|
/// sequential dial-then-accept, as in the c3 conformance suite).
|
|
fn pair(t: &dyn Transport) -> (Box<dyn Conn>, Box<dyn Conn>) {
|
|
let mut l = t.listen("127.0.0.1:0").unwrap();
|
|
let a = t.dial(&l.local_addr()).unwrap();
|
|
let b = l.accept().unwrap();
|
|
(a, b)
|
|
}
|
|
|
|
fn peers() -> Vec<String> {
|
|
match gen_server::call(MANAGER, Call::Peers) {
|
|
Ok(Reply::Peers(p)) => p,
|
|
other => panic!("manager unreachable: {other:?}"),
|
|
}
|
|
}
|
|
|
|
/// Poll until the manager's peer set matches `expected` (sorted) or `budget`
|
|
/// runs out.
|
|
fn wait_peers(expected: &[&str], budget: Duration) {
|
|
let want: Vec<String> = expected.iter().map(|s| s.to_string()).collect();
|
|
let deadline = Instant::now() + budget;
|
|
while Instant::now() < deadline {
|
|
if peers() == want {
|
|
return;
|
|
}
|
|
sleep(Duration::from_millis(10));
|
|
}
|
|
panic!(
|
|
"timed out waiting for peers == {want:?}; last = {:?}",
|
|
peers()
|
|
);
|
|
}
|
|
|
|
/// The actor emits heartbeats unprompted: the raw far end, saying nothing,
|
|
/// sees a `Frame::Heartbeat` well within one interval (the first goes out at
|
|
/// spawn).
|
|
#[test]
|
|
fn heartbeats_are_sent_unprompted() {
|
|
run(|| {
|
|
let mgr = GenServerBuilder::new(Manager::new())
|
|
.named(MANAGER)
|
|
.start()
|
|
.expect("manager name is free");
|
|
|
|
let (a, b) = pair(&TcpTransport);
|
|
spawn_established(FramedConn::new(a), peer("hb-send")).expect("register");
|
|
let mut far = FramedConn::new(b);
|
|
|
|
let frame = far
|
|
.recv_deadline(Instant::now() + HEARTBEAT_INTERVAL)
|
|
.expect("a heartbeat before one interval elapses");
|
|
assert_eq!(frame, Some(Frame::Heartbeat));
|
|
|
|
// Teardown: closing the far end is an EOF at the actor.
|
|
far.close();
|
|
wait_peers(&[], Duration::from_secs(2));
|
|
mgr.shutdown();
|
|
});
|
|
}
|
|
|
|
/// A mute peer is dead: no inbound frame for `LIVENESS_TIMEOUT` tears the
|
|
/// connection down and the manager's monitor reaps the table entry. The
|
|
/// entry is still present well inside the window — the teardown is the
|
|
/// timer, not an accident of setup.
|
|
#[test]
|
|
fn mute_peer_is_torn_down_after_liveness_timeout() {
|
|
run(|| {
|
|
let mgr = GenServerBuilder::new(Manager::new())
|
|
.named(MANAGER)
|
|
.start()
|
|
.expect("manager name is free");
|
|
|
|
let (a, b) = pair(&TcpTransport);
|
|
spawn_established(FramedConn::new(a), peer("mute")).expect("register");
|
|
// Held open and silent: no frames, no EOF. (Unread inbound
|
|
// heartbeats sit in kernel buffers; they are 5 bytes each.)
|
|
let _far = FramedConn::new(b);
|
|
|
|
// Well inside the window the connection is still up.
|
|
sleep(LIVENESS_TIMEOUT / 2);
|
|
assert_eq!(peers(), vec!["mute".to_string()], "torn down too early");
|
|
|
|
// ...and once the window empties it is gone. Generous budget over
|
|
// the remaining half-window.
|
|
wait_peers(&[], LIVENESS_TIMEOUT);
|
|
mgr.shutdown();
|
|
});
|
|
}
|
|
|
|
/// Heartbeats alone keep a connection alive past `LIVENESS_TIMEOUT`: a far
|
|
/// end that sends `Frame::Heartbeat` at the interval (and nothing else)
|
|
/// holds the entry; when it goes quiet, liveness finally fires.
|
|
#[test]
|
|
fn heartbeats_keep_the_connection_alive() {
|
|
run(|| {
|
|
let mgr = GenServerBuilder::new(Manager::new())
|
|
.named(MANAGER)
|
|
.start()
|
|
.expect("manager name is free");
|
|
|
|
let (a, b) = pair(&TcpTransport);
|
|
spawn_established(FramedConn::new(a), peer("kept")).expect("register");
|
|
|
|
// The far heartbeat pump: interval-paced sends until told to stop,
|
|
// then holds the socket open, silent, so the eventual teardown is
|
|
// liveness — not EOF.
|
|
let (ctl_tx, ctl_rx) = smarm::channel::channel::<()>();
|
|
spawn(move || {
|
|
let mut far = FramedConn::new(b);
|
|
// Phase 1: heartbeat at the interval until the first signal.
|
|
while matches!(ctl_rx.try_recv(), Ok(None)) {
|
|
far.send(&Frame::Heartbeat).expect("far send");
|
|
sleep(HEARTBEAT_INTERVAL);
|
|
}
|
|
// Phase 2: silent but with the socket held open — dropping
|
|
// `far` here would EOF the actor and mask the liveness path.
|
|
// Exits when the test's closure ends and drops `ctl_tx` (an
|
|
// eternal park would stop `run` from ever returning).
|
|
while matches!(ctl_rx.try_recv(), Ok(None)) {
|
|
sleep(Duration::from_millis(20));
|
|
}
|
|
});
|
|
|
|
// Past the liveness window with margin: still up.
|
|
sleep(LIVENESS_TIMEOUT + LIVENESS_TIMEOUT / 2);
|
|
assert_eq!(
|
|
peers(),
|
|
vec!["kept".to_string()],
|
|
"liveness fired despite heartbeats"
|
|
);
|
|
|
|
// Silence the pump; liveness now empties and the entry goes.
|
|
ctl_tx.send(()).expect("pump alive");
|
|
wait_peers(&[], LIVENESS_TIMEOUT * 2);
|
|
mgr.shutdown();
|
|
// `ctl_tx` drops here, releasing the pump's phase-2 wait.
|
|
});
|
|
}
|