Compare commits
33
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
8d2fed10d6 | ||
|
|
0f10415fb3 | ||
|
|
69a52a5578 | ||
|
|
8fd724a6b6 | ||
|
|
3b0e06ef13 | ||
|
|
93bc83a5b3 | ||
|
|
a9341e2d82 | ||
|
|
d300a9d536 | ||
|
|
bc0a5e8656 | ||
|
|
ea2b222cff | ||
|
|
7001f04b65 | ||
|
|
77938cd31d | ||
|
|
5ddd122711 | ||
|
|
343e53e17b | ||
|
|
9b215573de | ||
|
|
bf55cef3e3 | ||
|
|
2f88264426 | ||
|
|
6172f4231d | ||
|
|
d7082eb266 | ||
|
|
4c0e42152f | ||
|
|
16ef583455 | ||
|
|
a7f98f8d48 | ||
|
|
1282c3a08d | ||
|
|
160967939b | ||
|
|
112d6b2e65 | ||
|
|
ad4958421f | ||
|
|
c8ed858e4c | ||
|
|
bbaaa062e3 | ||
|
|
9e49038474 | ||
|
|
8a9e2b81b1 | ||
|
|
39ab92871e | ||
|
|
3850f6099b | ||
|
|
58a2fe3046 |
+21
-2
@@ -1,6 +1,6 @@
|
|||||||
[package]
|
[package]
|
||||||
name = "smarm"
|
name = "smarm"
|
||||||
version = "0.7.0"
|
version = "0.8.0"
|
||||||
edition = "2021"
|
edition = "2021"
|
||||||
rust-version = "1.95"
|
rust-version = "1.95"
|
||||||
|
|
||||||
@@ -17,7 +17,7 @@ unwrap_used = "deny"
|
|||||||
expect_used = "deny"
|
expect_used = "deny"
|
||||||
|
|
||||||
[features]
|
[features]
|
||||||
default = ["rq-mutex"]
|
default = ["rq-mpmc"]
|
||||||
smarm-trace = []
|
smarm-trace = []
|
||||||
# RFC 007: native causal profiling. Zero cost when off (cf. smarm-trace): the
|
# RFC 007: native causal profiling. Zero cost when off (cf. smarm-trace): the
|
||||||
# hook in `maybe_preempt` and the resume-path fast-forward compile away; the
|
# hook in `maybe_preempt` and the resume-path fast-forward compile away; the
|
||||||
@@ -33,6 +33,11 @@ budget-accounting = []
|
|||||||
# and unflagged; only the optional gen_server transport sits behind this, so a
|
# and unflagged; only the optional gen_server transport sits behind this, so a
|
||||||
# release build pays nothing for an observer it never starts.
|
# release build pays nothing for an observer it never starts.
|
||||||
observer = []
|
observer = []
|
||||||
|
# RFC 010 c1: clustering. Off by default — the default build stays libc-only,
|
||||||
|
# byte-for-byte (gate checked per phase). serde is the payload contract,
|
||||||
|
# postcard the payload codec; both minimal (no default features). Everything
|
||||||
|
# cluster-shaped lives behind this flag.
|
||||||
|
cluster = ["dep:serde", "dep:postcard"]
|
||||||
# Run-queue selection: exactly one, compile-time (see src/run_queue.rs).
|
# Run-queue selection: exactly one, compile-time (see src/run_queue.rs).
|
||||||
# Non-default variants need --no-default-features (features are additive).
|
# Non-default variants need --no-default-features (features are additive).
|
||||||
rq-mutex = []
|
rq-mutex = []
|
||||||
@@ -44,12 +49,18 @@ cc = "1"
|
|||||||
|
|
||||||
[dependencies]
|
[dependencies]
|
||||||
libc = "0.2"
|
libc = "0.2"
|
||||||
|
# RFC 010 §2 — only compiled under `--features cluster`.
|
||||||
|
serde = { version = "1", default-features = false, optional = true }
|
||||||
|
# `alloc` (not `std`): the seam serializes to Vec; postcard stays no_std-aligned.
|
||||||
|
postcard = { version = "1", default-features = false, features = ["alloc"], optional = true }
|
||||||
|
|
||||||
[target.'cfg(loom)'.dependencies]
|
[target.'cfg(loom)'.dependencies]
|
||||||
loom = "0.7"
|
loom = "0.7"
|
||||||
|
|
||||||
[dev-dependencies]
|
[dev-dependencies]
|
||||||
libc = "0.2"
|
libc = "0.2"
|
||||||
|
# derive + std for cluster envelope tests only; the lib itself never needs them
|
||||||
|
serde = { version = "1", features = ["derive"] }
|
||||||
tokio = { version = "1", features = ["rt", "rt-multi-thread", "macros", "sync", "time"] }
|
tokio = { version = "1", features = ["rt", "rt-multi-thread", "macros", "sync", "time"] }
|
||||||
|
|
||||||
[profile.dev]
|
[profile.dev]
|
||||||
@@ -60,6 +71,14 @@ panic = "unwind"
|
|||||||
lto = "thin"
|
lto = "thin"
|
||||||
codegen-units = 1
|
codegen-units = 1
|
||||||
|
|
||||||
|
# `cargo test --profile reltest`: release codegen for the crate (same opt-level,
|
||||||
|
# same panic strategy) but no LTO at the final link. Thin LTO is what makes each
|
||||||
|
# of the ~40 test binaries cost ~12 s to link instead of ~2 s; the tests don't
|
||||||
|
# need cross-crate LTO, the benches do (they keep using `release`).
|
||||||
|
[profile.reltest]
|
||||||
|
inherits = "release"
|
||||||
|
lto = false
|
||||||
|
|
||||||
[[bench]]
|
[[bench]]
|
||||||
name = "primes"
|
name = "primes"
|
||||||
harness = false
|
harness = false
|
||||||
|
|||||||
+14
-1
@@ -28,6 +28,15 @@ and **excised** (not worth the code cost; preserved on branch
|
|||||||
`rfc-004-spinning`). Also a false-sharing fix (`align(64)` on `SchedulerStats`)
|
`rfc-004-spinning`). Also a false-sharing fix (`align(64)` on `SchedulerStats`)
|
||||||
and a termination wake for idle siblings.
|
and a termination wake for idle siblings.
|
||||||
Commits `2708042`, `37d9319`, `eddf3fe`.
|
Commits `2708042`, `37d9319`, `eddf3fe`.
|
||||||
|
**Default flipped ON 2026-08-18** (history.md findings 17/18): the slot had
|
||||||
|
shipped default-off "until the shootout accepts it" and the flip was never
|
||||||
|
made, so every general.rs number since was slot-off. Acceptance sweep
|
||||||
|
(rq_runtime, 1/2/4/8/20 schedulers, rq-mpmc, 7 runs): ping-pong-pairs
|
||||||
|
−7% at 1T, 3.6×/5.3×/15.6×/10× faster at 2/4/8/20T, 100% slot hits, 0
|
||||||
|
displaced; yield-storm and spawn-storm within ±10% noise both ways.
|
||||||
|
general.rs on the flip: ping_pong_steady 20T 18485→1549 µs, 1T −15%,
|
||||||
|
mpsc_contention 1T −51%, ping_pong_oneshot 20T −18%; no smarm regressions.
|
||||||
|
`benches/baseline.json` regenerated slot-on.
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
@@ -146,7 +155,11 @@ Needs an RFC.
|
|||||||
#### Per-switch cost (context shims, epoch protocol)
|
#### Per-switch cost (context shims, epoch protocol)
|
||||||
The shootout's residual: per-wake latency is 0.16–0.18 µs at N=1 and
|
The shootout's residual: per-wake latency is 0.16–0.18 µs at N=1 and
|
||||||
0.8–1.2 µs at N=8+, dominated by the context-switch shims and the epoch
|
0.8–1.2 µs at N=8+, dominated by the context-switch shims and the epoch
|
||||||
protocol, not the queue. On current evidence this is the larger constant —
|
protocol, not the queue. **Premise corrected 2026-08-18 (finding 17): the
|
||||||
|
N=8+ figure was the slot-off futex path (one futex_wake per landed wake,
|
||||||
|
woken worker steals the pair); slot-on it is ~1.3× N=1. The shims are ~5
|
||||||
|
cycles (finding 3) and the epoch CASes ~117 cycles/roundtrip (finding 15).
|
||||||
|
Re-measure slot-on before spending anything here.** On current evidence this is the larger constant —
|
||||||
"the whole game" alongside the v0.9 work — but there is no spec yet. Needs a
|
"the whole game" alongside the v0.9 work — but there is no spec yet. Needs a
|
||||||
profiling spike (where do the cycles actually go per park/unpark round-trip)
|
profiling spike (where do the cycles actually go per park/unpark round-trip)
|
||||||
and then an RFC before it can be scheduled.
|
and then an RFC before it can be scheduled.
|
||||||
|
|||||||
+317
-265
@@ -1,314 +1,366 @@
|
|||||||
{
|
{
|
||||||
|
"catch_unwind_panics": {
|
||||||
|
"smarm 1-thread": {
|
||||||
|
"result": 10000,
|
||||||
|
"median": 122512,
|
||||||
|
"min": 113735,
|
||||||
|
"max": 124564
|
||||||
|
},
|
||||||
|
"smarm 20-thread": {
|
||||||
|
"result": 10000,
|
||||||
|
"median": 20569,
|
||||||
|
"min": 19671,
|
||||||
|
"max": 21174
|
||||||
|
},
|
||||||
|
"tokio current_thread": {
|
||||||
|
"result": 10000,
|
||||||
|
"median": 10134,
|
||||||
|
"min": 9081,
|
||||||
|
"max": 10524
|
||||||
|
},
|
||||||
|
"tokio multi-thread": {
|
||||||
|
"result": 10000,
|
||||||
|
"median": 2461,
|
||||||
|
"min": 2351,
|
||||||
|
"max": 2715
|
||||||
|
}
|
||||||
|
},
|
||||||
"chained_spawn": {
|
"chained_spawn": {
|
||||||
"smarm 1-thread": {
|
"smarm 1-thread": {
|
||||||
"result": 1000,
|
"result": 1000,
|
||||||
"median": 413,
|
"median": 498,
|
||||||
"min": 410,
|
"min": 491,
|
||||||
"max": 439
|
"max": 503
|
||||||
},
|
},
|
||||||
"smarm 24-thread": {
|
"smarm 20-thread": {
|
||||||
"result": 1000,
|
"result": 1000,
|
||||||
"median": 909,
|
"median": 2135,
|
||||||
"min": 888,
|
"min": 1881,
|
||||||
"max": 951
|
"max": 2284
|
||||||
},
|
},
|
||||||
"tokio current_thread": {
|
"tokio current_thread": {
|
||||||
"result": 1000,
|
"result": 1000,
|
||||||
"median": 62,
|
"median": 114,
|
||||||
"min": 61,
|
"min": 106,
|
||||||
"max": 62
|
"max": 119
|
||||||
},
|
},
|
||||||
"tokio multi-thread": {
|
"tokio multi-thread": {
|
||||||
"result": 1000,
|
"result": 1000,
|
||||||
"median": 197,
|
"median": 164,
|
||||||
"min": 194,
|
"min": 160,
|
||||||
"max": 210
|
"max": 186
|
||||||
}
|
|
||||||
},
|
|
||||||
"yield_many": {
|
|
||||||
"smarm 1-thread": {
|
|
||||||
"result": 200000,
|
|
||||||
"median": 16475,
|
|
||||||
"min": 16393,
|
|
||||||
"max": 16732
|
|
||||||
},
|
|
||||||
"smarm 24-thread": {
|
|
||||||
"result": 200000,
|
|
||||||
"median": 148708,
|
|
||||||
"min": 111213,
|
|
||||||
"max": 156462
|
|
||||||
},
|
|
||||||
"tokio current_thread": {
|
|
||||||
"result": 200000,
|
|
||||||
"median": 4751,
|
|
||||||
"min": 4740,
|
|
||||||
"max": 5259
|
|
||||||
},
|
|
||||||
"tokio multi-thread": {
|
|
||||||
"result": 200000,
|
|
||||||
"median": 8320,
|
|
||||||
"min": 7862,
|
|
||||||
"max": 8882
|
|
||||||
}
|
|
||||||
},
|
|
||||||
"fan_out_compute": {
|
|
||||||
"smarm 1-thread": {
|
|
||||||
"result": 33860,
|
|
||||||
"median": 13453,
|
|
||||||
"min": 13305,
|
|
||||||
"max": 15077
|
|
||||||
},
|
|
||||||
"smarm 24-thread": {
|
|
||||||
"result": 33860,
|
|
||||||
"median": 2451,
|
|
||||||
"min": 2330,
|
|
||||||
"max": 2520
|
|
||||||
},
|
|
||||||
"tokio current_thread": {
|
|
||||||
"result": 33860,
|
|
||||||
"median": 14019,
|
|
||||||
"min": 12339,
|
|
||||||
"max": 14045
|
|
||||||
},
|
|
||||||
"tokio multi-thread": {
|
|
||||||
"result": 33860,
|
|
||||||
"median": 1500,
|
|
||||||
"min": 1426,
|
|
||||||
"max": 1600
|
|
||||||
}
|
|
||||||
},
|
|
||||||
"ping_pong_oneshot": {
|
|
||||||
"smarm 1-thread": {
|
|
||||||
"result": 1000,
|
|
||||||
"median": 898,
|
|
||||||
"min": 782,
|
|
||||||
"max": 920
|
|
||||||
},
|
|
||||||
"smarm 24-thread": {
|
|
||||||
"result": 1000,
|
|
||||||
"median": 1494,
|
|
||||||
"min": 1489,
|
|
||||||
"max": 1546
|
|
||||||
},
|
|
||||||
"tokio current_thread": {
|
|
||||||
"result": 1000,
|
|
||||||
"median": 396,
|
|
||||||
"min": 389,
|
|
||||||
"max": 409
|
|
||||||
},
|
|
||||||
"tokio multi-thread": {
|
|
||||||
"result": 1000,
|
|
||||||
"median": 10382,
|
|
||||||
"min": 9559,
|
|
||||||
"max": 10924
|
|
||||||
}
|
|
||||||
},
|
|
||||||
"spawn_storm_busy": {
|
|
||||||
"smarm 1-thread": {
|
|
||||||
"result": 10000,
|
|
||||||
"median": 106232,
|
|
||||||
"min": 105627,
|
|
||||||
"max": 107693
|
|
||||||
},
|
|
||||||
"smarm 24-thread": {
|
|
||||||
"result": 10000,
|
|
||||||
"median": 49198,
|
|
||||||
"min": 48894,
|
|
||||||
"max": 52606
|
|
||||||
},
|
|
||||||
"tokio current_thread": {
|
|
||||||
"result": 10000,
|
|
||||||
"median": 1140,
|
|
||||||
"min": 1018,
|
|
||||||
"max": 1169
|
|
||||||
},
|
|
||||||
"tokio multi-thread": {
|
|
||||||
"result": 10000,
|
|
||||||
"median": 18650,
|
|
||||||
"min": 16486,
|
|
||||||
"max": 19396
|
|
||||||
}
|
|
||||||
},
|
|
||||||
"mpsc_contention": {
|
|
||||||
"smarm 1-thread": {
|
|
||||||
"result": 320000,
|
|
||||||
"median": 6173,
|
|
||||||
"min": 5446,
|
|
||||||
"max": 6226
|
|
||||||
},
|
|
||||||
"smarm 24-thread": {
|
|
||||||
"result": 320000,
|
|
||||||
"median": 36329,
|
|
||||||
"min": 35121,
|
|
||||||
"max": 37590
|
|
||||||
},
|
|
||||||
"tokio current_thread": {
|
|
||||||
"result": 320000,
|
|
||||||
"median": 5563,
|
|
||||||
"min": 5527,
|
|
||||||
"max": 6239
|
|
||||||
},
|
|
||||||
"tokio multi-thread": {
|
|
||||||
"result": 320000,
|
|
||||||
"median": 63966,
|
|
||||||
"min": 59972,
|
|
||||||
"max": 67534
|
|
||||||
}
|
|
||||||
},
|
|
||||||
"many_timers": {
|
|
||||||
"smarm 1-thread": {
|
|
||||||
"result": 10000,
|
|
||||||
"median": 107826,
|
|
||||||
"min": 107093,
|
|
||||||
"max": 119034
|
|
||||||
},
|
|
||||||
"smarm 24-thread": {
|
|
||||||
"result": 10000,
|
|
||||||
"median": 96539,
|
|
||||||
"min": 95413,
|
|
||||||
"max": 97804
|
|
||||||
},
|
|
||||||
"tokio current_thread": {
|
|
||||||
"result": 10000,
|
|
||||||
"median": 12584,
|
|
||||||
"min": 12539,
|
|
||||||
"max": 12627
|
|
||||||
},
|
|
||||||
"tokio multi-thread": {
|
|
||||||
"result": 10000,
|
|
||||||
"median": 16183,
|
|
||||||
"min": 16024,
|
|
||||||
"max": 16541
|
|
||||||
}
|
|
||||||
},
|
|
||||||
"multi_thread_scaling": {
|
|
||||||
"smarm 1-thread": {
|
|
||||||
"result": 33860,
|
|
||||||
"median": 15083,
|
|
||||||
"min": 15071,
|
|
||||||
"max": 15283
|
|
||||||
},
|
|
||||||
"smarm 2-thread": {
|
|
||||||
"result": 33860,
|
|
||||||
"median": 8070,
|
|
||||||
"min": 8003,
|
|
||||||
"max": 8096
|
|
||||||
},
|
|
||||||
"smarm 4-thread": {
|
|
||||||
"result": 33860,
|
|
||||||
"median": 4460,
|
|
||||||
"min": 4454,
|
|
||||||
"max": 4516
|
|
||||||
},
|
|
||||||
"smarm 24-thread": {
|
|
||||||
"result": 33860,
|
|
||||||
"median": 2333,
|
|
||||||
"min": 2294,
|
|
||||||
"max": 2348
|
|
||||||
},
|
|
||||||
"tokio multi 1-thread": {
|
|
||||||
"result": 33860,
|
|
||||||
"median": 14504,
|
|
||||||
"min": 14193,
|
|
||||||
"max": 14562
|
|
||||||
},
|
|
||||||
"tokio multi 2-thread": {
|
|
||||||
"result": 33860,
|
|
||||||
"median": 7297,
|
|
||||||
"min": 7289,
|
|
||||||
"max": 7398
|
|
||||||
},
|
|
||||||
"tokio multi 4-thread": {
|
|
||||||
"result": 33860,
|
|
||||||
"median": 3795,
|
|
||||||
"min": 3756,
|
|
||||||
"max": 3799
|
|
||||||
},
|
|
||||||
"tokio multi 24-thread": {
|
|
||||||
"result": 33860,
|
|
||||||
"median": 1544,
|
|
||||||
"min": 1520,
|
|
||||||
"max": 1610
|
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
"deep_recursion": {
|
"deep_recursion": {
|
||||||
"smarm 1-thread": {
|
"smarm 1-thread": {
|
||||||
"result": 1,
|
"result": 1,
|
||||||
"median": 226,
|
"median": 316,
|
||||||
"min": 222,
|
"min": 314,
|
||||||
"max": 243
|
"max": 322
|
||||||
},
|
},
|
||||||
"smarm 24-thread": {
|
"smarm 20-thread": {
|
||||||
"result": 1,
|
"result": 1,
|
||||||
"median": 745,
|
"median": 758,
|
||||||
"min": 744,
|
"min": 725,
|
||||||
"max": 776
|
"max": 1222
|
||||||
},
|
},
|
||||||
"tokio current_thread": {
|
"tokio current_thread": {
|
||||||
"result": 1,
|
"result": 1,
|
||||||
"median": 11,
|
"median": 11,
|
||||||
"min": 10,
|
"min": 11,
|
||||||
"max": 13
|
"max": 14
|
||||||
},
|
},
|
||||||
"tokio multi-thread": {
|
"tokio multi-thread": {
|
||||||
"result": 1,
|
"result": 1,
|
||||||
"median": 53,
|
"median": 52,
|
||||||
"min": 53,
|
"min": 51,
|
||||||
"max": 57
|
"max": 55
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
"yield_in_hot_loop": {
|
"fan_out_compute": {
|
||||||
"smarm 1-thread": {
|
"smarm 1-thread": {
|
||||||
"result": 1000000,
|
"result": 33860,
|
||||||
"median": 64849,
|
"median": 15074,
|
||||||
"min": 64396,
|
"min": 15064,
|
||||||
"max": 65283
|
"max": 15081
|
||||||
|
},
|
||||||
|
"smarm 20-thread": {
|
||||||
|
"result": 33860,
|
||||||
|
"median": 2786,
|
||||||
|
"min": 2539,
|
||||||
|
"max": 2913
|
||||||
},
|
},
|
||||||
"tokio current_thread": {
|
"tokio current_thread": {
|
||||||
"result": 1000000,
|
"result": 33860,
|
||||||
"median": 68507,
|
"median": 14040,
|
||||||
"min": 62018,
|
"min": 13749,
|
||||||
"max": 72341
|
"max": 14056
|
||||||
|
},
|
||||||
|
"tokio multi-thread": {
|
||||||
|
"result": 33860,
|
||||||
|
"median": 2404,
|
||||||
|
"min": 2136,
|
||||||
|
"max": 2436
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"many_timers": {
|
||||||
|
"smarm 1-thread": {
|
||||||
|
"result": 10000,
|
||||||
|
"median": 117779,
|
||||||
|
"min": 107641,
|
||||||
|
"max": 120785
|
||||||
|
},
|
||||||
|
"smarm 20-thread": {
|
||||||
|
"result": 10000,
|
||||||
|
"median": 54031,
|
||||||
|
"min": 53646,
|
||||||
|
"max": 54841
|
||||||
|
},
|
||||||
|
"tokio current_thread": {
|
||||||
|
"result": 10000,
|
||||||
|
"median": 12417,
|
||||||
|
"min": 12383,
|
||||||
|
"max": 12479
|
||||||
|
},
|
||||||
|
"tokio multi-thread": {
|
||||||
|
"result": 10000,
|
||||||
|
"median": 13880,
|
||||||
|
"min": 13405,
|
||||||
|
"max": 13953
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"mpsc_contention": {
|
||||||
|
"smarm 1-thread": {
|
||||||
|
"result": 320000,
|
||||||
|
"median": 3260,
|
||||||
|
"min": 2896,
|
||||||
|
"max": 3427
|
||||||
|
},
|
||||||
|
"smarm 20-thread": {
|
||||||
|
"result": 320000,
|
||||||
|
"median": 41206,
|
||||||
|
"min": 40562,
|
||||||
|
"max": 41754
|
||||||
|
},
|
||||||
|
"tokio current_thread": {
|
||||||
|
"result": 320000,
|
||||||
|
"median": 5895,
|
||||||
|
"min": 5272,
|
||||||
|
"max": 5902
|
||||||
|
},
|
||||||
|
"tokio multi-thread": {
|
||||||
|
"result": 320000,
|
||||||
|
"median": 71069,
|
||||||
|
"min": 60076,
|
||||||
|
"max": 80026
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"multi_thread_scaling": {
|
||||||
|
"smarm 1-thread": {
|
||||||
|
"result": 33860,
|
||||||
|
"median": 15163,
|
||||||
|
"min": 15125,
|
||||||
|
"max": 15172
|
||||||
|
},
|
||||||
|
"smarm 2-thread": {
|
||||||
|
"result": 33860,
|
||||||
|
"median": 7980,
|
||||||
|
"min": 7947,
|
||||||
|
"max": 8103
|
||||||
|
},
|
||||||
|
"smarm 4-thread": {
|
||||||
|
"result": 33860,
|
||||||
|
"median": 4324,
|
||||||
|
"min": 4300,
|
||||||
|
"max": 4364
|
||||||
|
},
|
||||||
|
"smarm 20-thread": {
|
||||||
|
"result": 33860,
|
||||||
|
"median": 2704,
|
||||||
|
"min": 2391,
|
||||||
|
"max": 2841
|
||||||
|
},
|
||||||
|
"tokio multi 1-thread": {
|
||||||
|
"result": 33860,
|
||||||
|
"median": 14271,
|
||||||
|
"min": 14251,
|
||||||
|
"max": 14459
|
||||||
|
},
|
||||||
|
"tokio multi 2-thread": {
|
||||||
|
"result": 33860,
|
||||||
|
"median": 7395,
|
||||||
|
"min": 7195,
|
||||||
|
"max": 7443
|
||||||
|
},
|
||||||
|
"tokio multi 4-thread": {
|
||||||
|
"result": 33860,
|
||||||
|
"median": 3740,
|
||||||
|
"min": 3709,
|
||||||
|
"max": 3752
|
||||||
|
},
|
||||||
|
"tokio multi 20-thread": {
|
||||||
|
"result": 33860,
|
||||||
|
"median": 2373,
|
||||||
|
"min": 2176,
|
||||||
|
"max": 2380
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"ping_pong_oneshot": {
|
||||||
|
"smarm 1-thread": {
|
||||||
|
"result": 1000,
|
||||||
|
"median": 935,
|
||||||
|
"min": 910,
|
||||||
|
"max": 941
|
||||||
|
},
|
||||||
|
"smarm 20-thread": {
|
||||||
|
"result": 1000,
|
||||||
|
"median": 6168,
|
||||||
|
"min": 6029,
|
||||||
|
"max": 6550
|
||||||
|
},
|
||||||
|
"tokio current_thread": {
|
||||||
|
"result": 1000,
|
||||||
|
"median": 407,
|
||||||
|
"min": 401,
|
||||||
|
"max": 430
|
||||||
|
},
|
||||||
|
"tokio multi-thread": {
|
||||||
|
"result": 1000,
|
||||||
|
"median": 9395,
|
||||||
|
"min": 8795,
|
||||||
|
"max": 9812
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"ping_pong_steady": {
|
||||||
|
"smarm 1-thread": {
|
||||||
|
"result": 10000,
|
||||||
|
"median": 1517,
|
||||||
|
"min": 1501,
|
||||||
|
"max": 1527
|
||||||
|
},
|
||||||
|
"smarm 20-thread": {
|
||||||
|
"result": 10000,
|
||||||
|
"median": 1910,
|
||||||
|
"min": 1848,
|
||||||
|
"max": 1947
|
||||||
|
},
|
||||||
|
"tokio current_thread": {
|
||||||
|
"result": 10000,
|
||||||
|
"median": 1284,
|
||||||
|
"min": 1278,
|
||||||
|
"max": 1296
|
||||||
|
},
|
||||||
|
"tokio multi-thread": {
|
||||||
|
"result": 10000,
|
||||||
|
"median": 82015,
|
||||||
|
"min": 81833,
|
||||||
|
"max": 82993
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"spawn_pair_control": {
|
||||||
|
"smarm 1-thread": {
|
||||||
|
"result": 1000,
|
||||||
|
"median": 637,
|
||||||
|
"min": 632,
|
||||||
|
"max": 661
|
||||||
|
},
|
||||||
|
"smarm 20-thread": {
|
||||||
|
"result": 1000,
|
||||||
|
"median": 5118,
|
||||||
|
"min": 5089,
|
||||||
|
"max": 5157
|
||||||
|
},
|
||||||
|
"tokio current_thread": {
|
||||||
|
"result": 1000,
|
||||||
|
"median": 288,
|
||||||
|
"min": 249,
|
||||||
|
"max": 300
|
||||||
|
},
|
||||||
|
"tokio multi-thread": {
|
||||||
|
"result": 1000,
|
||||||
|
"median": 8617,
|
||||||
|
"min": 8534,
|
||||||
|
"max": 8808
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"spawn_storm_busy": {
|
||||||
|
"smarm 1-thread": {
|
||||||
|
"result": 10000,
|
||||||
|
"median": 105937,
|
||||||
|
"min": 101927,
|
||||||
|
"max": 107104
|
||||||
|
},
|
||||||
|
"smarm 20-thread": {
|
||||||
|
"result": 10000,
|
||||||
|
"median": 15928,
|
||||||
|
"min": 15918,
|
||||||
|
"max": 16581
|
||||||
|
},
|
||||||
|
"tokio current_thread": {
|
||||||
|
"result": 10000,
|
||||||
|
"median": 1248,
|
||||||
|
"min": 1121,
|
||||||
|
"max": 1253
|
||||||
|
},
|
||||||
|
"tokio multi-thread": {
|
||||||
|
"result": 10000,
|
||||||
|
"median": 12313,
|
||||||
|
"min": 11641,
|
||||||
|
"max": 16848
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
"uncontended_channel": {
|
"uncontended_channel": {
|
||||||
"smarm 1-thread": {
|
"smarm 1-thread": {
|
||||||
"result": 1000000,
|
"result": 1000000,
|
||||||
"median": 11949,
|
"median": 12452,
|
||||||
"min": 11928,
|
"min": 12315,
|
||||||
"max": 13596
|
"max": 13700
|
||||||
},
|
},
|
||||||
"tokio current_thread": {
|
"tokio current_thread": {
|
||||||
"result": 1000000,
|
"result": 1000000,
|
||||||
"median": 15083,
|
"median": 15091,
|
||||||
"min": 15038,
|
"min": 15086,
|
||||||
"max": 16994
|
"max": 16944
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
"catch_unwind_panics": {
|
"yield_in_hot_loop": {
|
||||||
"smarm 1-thread": {
|
"smarm 1-thread": {
|
||||||
"result": 10000,
|
"result": 1000000,
|
||||||
"median": 110932,
|
"median": 33187,
|
||||||
"min": 110182,
|
"min": 33000,
|
||||||
"max": 124147
|
"max": 33325
|
||||||
},
|
|
||||||
"smarm 24-thread": {
|
|
||||||
"result": 10000,
|
|
||||||
"median": 13665,
|
|
||||||
"min": 13172,
|
|
||||||
"max": 13784
|
|
||||||
},
|
},
|
||||||
"tokio current_thread": {
|
"tokio current_thread": {
|
||||||
"result": 10000,
|
"result": 1000000,
|
||||||
"median": 10431,
|
"median": 74444,
|
||||||
"min": 9346,
|
"min": 67009,
|
||||||
"max": 10915
|
"max": 75753
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"yield_many": {
|
||||||
|
"smarm 1-thread": {
|
||||||
|
"result": 200000,
|
||||||
|
"median": 10921,
|
||||||
|
"min": 10828,
|
||||||
|
"max": 11041
|
||||||
|
},
|
||||||
|
"smarm 20-thread": {
|
||||||
|
"result": 200000,
|
||||||
|
"median": 44422,
|
||||||
|
"min": 43715,
|
||||||
|
"max": 44595
|
||||||
|
},
|
||||||
|
"tokio current_thread": {
|
||||||
|
"result": 200000,
|
||||||
|
"median": 5355,
|
||||||
|
"min": 5327,
|
||||||
|
"max": 5426
|
||||||
},
|
},
|
||||||
"tokio multi-thread": {
|
"tokio multi-thread": {
|
||||||
"result": 10000,
|
"result": 200000,
|
||||||
"median": 6171,
|
"median": 6170,
|
||||||
"min": 5626,
|
"min": 5982,
|
||||||
"max": 6555
|
"max": 6898
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
+163
-2
@@ -13,7 +13,18 @@
|
|||||||
//! completeness.
|
//! completeness.
|
||||||
//! 4. ping_pong_oneshot — N rounds of (spawn pair, send oneshot, await).
|
//! 4. ping_pong_oneshot — N rounds of (spawn pair, send oneshot, await).
|
||||||
//! Closer to a request/response workload than channel
|
//! Closer to a request/response workload than channel
|
||||||
//! ping-pong.
|
//! ping-pong. NOTE: 2 spawns + 2 channel allocs + 2
|
||||||
|
//! joins per round; the message path is a minority
|
||||||
|
//! of it. Sections 5 and 6 split it apart.
|
||||||
|
//! 5. spawn_pair_control — section 4 with the messages removed: same
|
||||||
|
//! spawn/join shape, actors return immediately.
|
||||||
|
//! (4 − 5) ≈ per-round message-path cost.
|
||||||
|
//! 6. ping_pong_steady — ONE persistent pair, N roundtrips over unbounded
|
||||||
|
//! MPSC channels (smarm::channel vs
|
||||||
|
//! tokio::sync::mpsc::unbounded_channel — both
|
||||||
|
//! unbounded, non-blocking send). Steady-state
|
||||||
|
//! park/unpark cost per roundtrip, no spawn in the
|
||||||
|
//! loop.
|
||||||
|
|
||||||
use std::sync::atomic::{AtomicU64, Ordering};
|
use std::sync::atomic::{AtomicU64, Ordering};
|
||||||
use std::sync::Arc;
|
use std::sync::Arc;
|
||||||
@@ -418,6 +429,135 @@ fn bench_pp_tokio_multi() -> (u64, u128) {
|
|||||||
(PP_ROUNDS, start.elapsed().as_micros())
|
(PP_ROUNDS, start.elapsed().as_micros())
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// 5. spawn_pair_control — section 4 minus the messages
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
/// Target-5 instrumentation: emit the runtime's wake-path counters for one
|
||||||
|
/// run when SMARM_WAKE_DIAG is set. Off by default so sweep.py output is
|
||||||
|
/// unchanged.
|
||||||
|
fn wake_diag(section: &str, threads: usize, rt: &smarm::runtime::Runtime, us: u128) {
|
||||||
|
if std::env::var_os("SMARM_WAKE_DIAG").is_some() {
|
||||||
|
println!("DIAG,{section},{threads},{us},{}", rt.stats().wake_diag());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn bench_ctl_smarm(threads: usize) -> (u64, u128) {
|
||||||
|
let start = Instant::now();
|
||||||
|
let rt = smarm::runtime::init(bench_cfg(threads));
|
||||||
|
rt.run(|| {
|
||||||
|
for _ in 0..PP_ROUNDS {
|
||||||
|
let hb = smarm::spawn(|| {});
|
||||||
|
let ha = smarm::spawn(|| {});
|
||||||
|
ha.join().unwrap();
|
||||||
|
hb.join().unwrap();
|
||||||
|
}
|
||||||
|
});
|
||||||
|
let us = start.elapsed().as_micros();
|
||||||
|
wake_diag("spawn_pair_control", threads, &rt, us);
|
||||||
|
(PP_ROUNDS, us)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn bench_ctl_tokio_current() -> (u64, u128) {
|
||||||
|
let rt = tokio::runtime::Builder::new_current_thread()
|
||||||
|
.build()
|
||||||
|
.unwrap();
|
||||||
|
let start = Instant::now();
|
||||||
|
let local = tokio::task::LocalSet::new();
|
||||||
|
local.block_on(&rt, async move {
|
||||||
|
for _ in 0..PP_ROUNDS {
|
||||||
|
let hb = tokio::task::spawn_local(async {});
|
||||||
|
let ha = tokio::task::spawn_local(async {});
|
||||||
|
let _ = ha.await;
|
||||||
|
let _ = hb.await;
|
||||||
|
}
|
||||||
|
});
|
||||||
|
(PP_ROUNDS, start.elapsed().as_micros())
|
||||||
|
}
|
||||||
|
|
||||||
|
fn bench_ctl_tokio_multi() -> (u64, u128) {
|
||||||
|
let rt = tokio::runtime::Builder::new_multi_thread()
|
||||||
|
.worker_threads(available_threads())
|
||||||
|
.build()
|
||||||
|
.unwrap();
|
||||||
|
let start = Instant::now();
|
||||||
|
rt.block_on(async move {
|
||||||
|
for _ in 0..PP_ROUNDS {
|
||||||
|
let hb = tokio::spawn(async {});
|
||||||
|
let ha = tokio::spawn(async {});
|
||||||
|
let _ = ha.await;
|
||||||
|
let _ = hb.await;
|
||||||
|
}
|
||||||
|
});
|
||||||
|
(PP_ROUNDS, start.elapsed().as_micros())
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// 6. ping_pong_steady — one persistent pair, PP_STEADY roundtrips
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
const PP_STEADY: u64 = 10_000;
|
||||||
|
|
||||||
|
fn bench_steady_smarm(threads: usize) -> (u64, u128) {
|
||||||
|
let start = Instant::now();
|
||||||
|
let rt = smarm::runtime::init(bench_cfg(threads));
|
||||||
|
rt.run(|| {
|
||||||
|
let (tx_ab, rx_ab) = smarm::channel::<u64>();
|
||||||
|
let (tx_ba, rx_ba) = smarm::channel::<u64>();
|
||||||
|
let echo = smarm::spawn(move || {
|
||||||
|
for _ in 0..PP_STEADY {
|
||||||
|
let v = rx_ab.recv().unwrap();
|
||||||
|
tx_ba.send(v + 1).unwrap();
|
||||||
|
}
|
||||||
|
});
|
||||||
|
for i in 0..PP_STEADY {
|
||||||
|
tx_ab.send(i).unwrap();
|
||||||
|
let v = rx_ba.recv().unwrap();
|
||||||
|
assert_eq!(v, i + 1);
|
||||||
|
}
|
||||||
|
echo.join().unwrap();
|
||||||
|
});
|
||||||
|
let us = start.elapsed().as_micros();
|
||||||
|
wake_diag("ping_pong_steady", threads, &rt, us);
|
||||||
|
(PP_STEADY, us)
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn steady_tokio_body() {
|
||||||
|
let (tx_ab, mut rx_ab) = tokio::sync::mpsc::unbounded_channel::<u64>();
|
||||||
|
let (tx_ba, mut rx_ba) = tokio::sync::mpsc::unbounded_channel::<u64>();
|
||||||
|
let echo = tokio::spawn(async move {
|
||||||
|
for _ in 0..PP_STEADY {
|
||||||
|
let v = rx_ab.recv().await.unwrap();
|
||||||
|
tx_ba.send(v + 1).unwrap();
|
||||||
|
}
|
||||||
|
});
|
||||||
|
for i in 0..PP_STEADY {
|
||||||
|
tx_ab.send(i).unwrap();
|
||||||
|
let v = rx_ba.recv().await.unwrap();
|
||||||
|
assert_eq!(v, i + 1);
|
||||||
|
}
|
||||||
|
echo.await.unwrap();
|
||||||
|
}
|
||||||
|
|
||||||
|
fn bench_steady_tokio_current() -> (u64, u128) {
|
||||||
|
let rt = tokio::runtime::Builder::new_current_thread()
|
||||||
|
.build()
|
||||||
|
.unwrap();
|
||||||
|
let start = Instant::now();
|
||||||
|
rt.block_on(steady_tokio_body());
|
||||||
|
(PP_STEADY, start.elapsed().as_micros())
|
||||||
|
}
|
||||||
|
|
||||||
|
fn bench_steady_tokio_multi() -> (u64, u128) {
|
||||||
|
let rt = tokio::runtime::Builder::new_multi_thread()
|
||||||
|
.worker_threads(available_threads())
|
||||||
|
.build()
|
||||||
|
.unwrap();
|
||||||
|
let start = Instant::now();
|
||||||
|
rt.block_on(steady_tokio_body());
|
||||||
|
(PP_STEADY, start.elapsed().as_micros())
|
||||||
|
}
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
// main
|
// main
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
@@ -453,7 +593,8 @@ fn main() {
|
|||||||
);
|
);
|
||||||
println!(
|
println!(
|
||||||
"CHAIN_DEPTH={CHAIN_DEPTH}, YIELD_TASKS={YIELD_TASKS}×{YIELD_ROUNDS}, \
|
"CHAIN_DEPTH={CHAIN_DEPTH}, YIELD_TASKS={YIELD_TASKS}×{YIELD_ROUNDS}, \
|
||||||
PRIME_N={PRIME_N}/{PRIME_WORKERS} workers, PP_ROUNDS={PP_ROUNDS}"
|
PRIME_N={PRIME_N}/{PRIME_WORKERS} workers, PP_ROUNDS={PP_ROUNDS}, \
|
||||||
|
PP_STEADY={PP_STEADY}"
|
||||||
);
|
);
|
||||||
|
|
||||||
// ---- 1. chained_spawn ----
|
// ---- 1. chained_spawn ----
|
||||||
@@ -491,4 +632,24 @@ fn main() {
|
|||||||
run_n(&format!("smarm {n}-thread"), ITERS, || bench_pp_smarm(n));
|
run_n(&format!("smarm {n}-thread"), ITERS, || bench_pp_smarm(n));
|
||||||
run_n("tokio current_thread", ITERS, bench_pp_tokio_current);
|
run_n("tokio current_thread", ITERS, bench_pp_tokio_current);
|
||||||
run_n("tokio multi-thread", ITERS, bench_pp_tokio_multi);
|
run_n("tokio multi-thread", ITERS, bench_pp_tokio_multi);
|
||||||
|
|
||||||
|
// ---- 5. spawn_pair_control ----
|
||||||
|
print_header(&format!(
|
||||||
|
"spawn_pair_control: {PP_ROUNDS} rounds, no messages"
|
||||||
|
));
|
||||||
|
run_n("smarm 1-thread", ITERS, || bench_ctl_smarm(1));
|
||||||
|
run_n(&format!("smarm {n}-thread"), ITERS, || bench_ctl_smarm(n));
|
||||||
|
run_n("tokio current_thread", ITERS, bench_ctl_tokio_current);
|
||||||
|
run_n("tokio multi-thread", ITERS, bench_ctl_tokio_multi);
|
||||||
|
|
||||||
|
// ---- 6. ping_pong_steady ----
|
||||||
|
print_header(&format!(
|
||||||
|
"ping_pong_steady: 1 pair × {PP_STEADY} roundtrips"
|
||||||
|
));
|
||||||
|
run_n("smarm 1-thread", ITERS, || bench_steady_smarm(1));
|
||||||
|
run_n(&format!("smarm {n}-thread"), ITERS, || {
|
||||||
|
bench_steady_smarm(n)
|
||||||
|
});
|
||||||
|
run_n("tokio current_thread", ITERS, bench_steady_tokio_current);
|
||||||
|
run_n("tokio multi-thread", ITERS, bench_steady_tokio_multi);
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -90,6 +90,7 @@ struct Sample {
|
|||||||
us: u128,
|
us: u128,
|
||||||
hits: u64,
|
hits: u64,
|
||||||
displacements: u64,
|
displacements: u64,
|
||||||
|
diag: String,
|
||||||
}
|
}
|
||||||
|
|
||||||
fn yield_storm(threads: usize, slot: bool, actors: usize, yields: usize) -> Sample {
|
fn yield_storm(threads: usize, slot: bool, actors: usize, yields: usize) -> Sample {
|
||||||
@@ -116,6 +117,7 @@ fn yield_storm(threads: usize, slot: bool, actors: usize, yields: usize) -> Samp
|
|||||||
us,
|
us,
|
||||||
hits: stats.slot_hits(),
|
hits: stats.slot_hits(),
|
||||||
displacements: stats.slot_displacements(),
|
displacements: stats.slot_displacements(),
|
||||||
|
diag: stats.wake_diag(),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -159,6 +161,7 @@ fn ping_pong_pairs(threads: usize, slot: bool, pairs: usize, roundtrips: usize)
|
|||||||
us,
|
us,
|
||||||
hits: stats.slot_hits(),
|
hits: stats.slot_hits(),
|
||||||
displacements: stats.slot_displacements(),
|
displacements: stats.slot_displacements(),
|
||||||
|
diag: stats.wake_diag(),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -185,6 +188,7 @@ fn spawn_storm(threads: usize, slot: bool, spawns: usize) -> Sample {
|
|||||||
us,
|
us,
|
||||||
hits: stats.slot_hits(),
|
hits: stats.slot_hits(),
|
||||||
displacements: stats.slot_displacements(),
|
displacements: stats.slot_displacements(),
|
||||||
|
diag: stats.wake_diag(),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -253,6 +257,14 @@ fn main() {
|
|||||||
mid.us,
|
mid.us,
|
||||||
per_s
|
per_s
|
||||||
);
|
);
|
||||||
|
println!(
|
||||||
|
"RQDIAG,{},{},{},{},{}",
|
||||||
|
variant(),
|
||||||
|
slot_str,
|
||||||
|
name,
|
||||||
|
t,
|
||||||
|
mid.diag
|
||||||
|
);
|
||||||
if slot {
|
if slot {
|
||||||
println!(
|
println!(
|
||||||
"RQSLOT,{},{},{},{},{}",
|
"RQSLOT,{},{},{},{},{}",
|
||||||
|
|||||||
@@ -8,4 +8,29 @@ fn main() {
|
|||||||
.flag_if_supported("-fno-stack-clash-protection")
|
.flag_if_supported("-fno-stack-clash-protection")
|
||||||
.compile("smarm_canary");
|
.compile("smarm_canary");
|
||||||
println!("cargo:rerun-if-changed=canary/canary.c");
|
println!("cargo:rerun-if-changed=canary/canary.c");
|
||||||
|
|
||||||
|
// RFC 010 c6d — build_hash inputs. The compile-time facts a peer must
|
||||||
|
// share for a mesh link: the exact toolchain and the declared (enabled)
|
||||||
|
// feature set. Emitted as a plain string; the hashing (FNV-1a folded
|
||||||
|
// with PROTO_VERSION) happens in src/cluster.rs where the protocol
|
||||||
|
// version actually lives — parsing it out of a source file here would
|
||||||
|
// be a second, fragile copy. Always emitted, even for non-cluster
|
||||||
|
// builds: one env var costs the default build nothing.
|
||||||
|
let rustc = std::env::var("RUSTC").unwrap_or_else(|_| "rustc".to_string());
|
||||||
|
let version = std::process::Command::new(&rustc)
|
||||||
|
.arg("-V")
|
||||||
|
.output()
|
||||||
|
.ok()
|
||||||
|
.map(|o| String::from_utf8_lossy(&o.stdout).trim().to_string())
|
||||||
|
.filter(|v| !v.is_empty())
|
||||||
|
.unwrap_or_else(|| "rustc-unknown".to_string());
|
||||||
|
let mut feats: Vec<String> = std::env::vars()
|
||||||
|
.filter_map(|(k, _)| k.strip_prefix("CARGO_FEATURE_").map(str::to_string))
|
||||||
|
.collect();
|
||||||
|
feats.sort();
|
||||||
|
println!(
|
||||||
|
"cargo:rustc-env=SMARM_BUILD_HASH_INPUTS={version};features={}",
|
||||||
|
feats.join(",")
|
||||||
|
);
|
||||||
|
println!("cargo:rerun-if-env-changed=RUSTC");
|
||||||
}
|
}
|
||||||
|
|||||||
+15
-2
@@ -72,7 +72,10 @@ pub fn clear_current_pid() {
|
|||||||
CURRENT_PID.with(|c| c.set(None));
|
CURRENT_PID.with(|c| c.set(None));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Actor-side TLS accessor: `#[inline(never)]` + fence, see `context` docs.
|
||||||
|
#[inline(never)]
|
||||||
pub fn current_pid() -> Option<Pid> {
|
pub fn current_pid() -> Option<Pid> {
|
||||||
|
crate::context::tls_fence();
|
||||||
CURRENT_PID.with(|c| c.get())
|
CURRENT_PID.with(|c| c.get())
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -109,8 +112,7 @@ pub extern "C-unwind" fn trampoline() {
|
|||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
LAST_OUTCOME.with(|r| *r.borrow_mut() = Some(outcome));
|
publish_outcome(outcome);
|
||||||
ACTOR_DONE.with(|c| c.set(true));
|
|
||||||
|
|
||||||
// Hand control back. The scheduler will tear down our slot and never
|
// Hand control back. The scheduler will tear down our slot and never
|
||||||
// resume us again.
|
// resume us again.
|
||||||
@@ -119,6 +121,17 @@ pub extern "C-unwind" fn trampoline() {
|
|||||||
unreachable!("scheduler resumed a done actor");
|
unreachable!("scheduler resumed a done actor");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Record the outcome for the scheduler that is about to be switched to. Kept
|
||||||
|
/// out of line: the actor may have migrated threads while its closure ran, so
|
||||||
|
/// the TLS base `trampoline` computed at entry must not be reused here (see
|
||||||
|
/// `context` module docs on thread-locals and migration).
|
||||||
|
#[inline(never)]
|
||||||
|
fn publish_outcome(outcome: Outcome) {
|
||||||
|
crate::context::tls_fence();
|
||||||
|
LAST_OUTCOME.with(|r| *r.borrow_mut() = Some(outcome));
|
||||||
|
ACTOR_DONE.with(|c| c.set(true));
|
||||||
|
}
|
||||||
|
|
||||||
/// One actor's worth of state. Owned by the scheduler's slot table.
|
/// One actor's worth of state. Owned by the scheduler's slot table.
|
||||||
pub struct Actor {
|
pub struct Actor {
|
||||||
/// The PID this actor was assigned at spawn time.
|
/// The PID this actor was assigned at spawn time.
|
||||||
|
|||||||
+12
-4
@@ -274,12 +274,16 @@ mod inner {
|
|||||||
|
|
||||||
/// The experiment-active path, kept out of the inlined fast path.
|
/// The experiment-active path, kept out of the inlined fast path.
|
||||||
#[cold]
|
#[cold]
|
||||||
|
#[inline(never)]
|
||||||
fn cold_check(exp: u64) {
|
fn cold_check(exp: u64) {
|
||||||
|
crate::context::tls_fence();
|
||||||
let slot = preempt::current_slot_ptr();
|
let slot = preempt::current_slot_ptr();
|
||||||
if slot.is_null() {
|
if slot.is_null() {
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
let now = preempt::rdtsc();
|
// Serialised: `now` closes an interval attributed to a code site; a
|
||||||
|
// speculative early read would drop that site's tail (RFC 007).
|
||||||
|
let now = preempt::rdtsc_serialising();
|
||||||
let last = LAST_SAMPLE_TSC.with(|c| c.replace(now));
|
let last = LAST_SAMPLE_TSC.with(|c| c.replace(now));
|
||||||
let target_site = (exp >> 32) as u32;
|
let target_site = (exp >> 32) as u32;
|
||||||
let pct = exp & 0xffff_ffff;
|
let pct = exp & 0xffff_ffff;
|
||||||
@@ -365,8 +369,9 @@ mod inner {
|
|||||||
/// - Entering the target site: re-arm the sample clock, so time spent
|
/// - Entering the target site: re-arm the sample clock, so time spent
|
||||||
/// *before* the site can never be attributed to it by the first
|
/// *before* the site can never be attributed to it by the first
|
||||||
/// in-site check (the symmetric over-attribution).
|
/// in-site check (the symmetric over-attribution).
|
||||||
#[inline]
|
#[inline(never)]
|
||||||
fn site_transition(slot: *const crate::runtime::Slot, old: u32, new: u32) {
|
fn site_transition(slot: *const crate::runtime::Slot, old: u32, new: u32) {
|
||||||
|
crate::context::tls_fence();
|
||||||
let exp = EXPERIMENT.load(Ordering::Relaxed);
|
let exp = EXPERIMENT.load(Ordering::Relaxed);
|
||||||
if exp == 0 || old == new {
|
if exp == 0 || old == new {
|
||||||
return;
|
return;
|
||||||
@@ -374,7 +379,8 @@ mod inner {
|
|||||||
let target = (exp >> 32) as u32;
|
let target = (exp >> 32) as u32;
|
||||||
let pct = exp & 0xffff_ffff;
|
let pct = exp & 0xffff_ffff;
|
||||||
if old == target && new != target {
|
if old == target && new != target {
|
||||||
let now = preempt::rdtsc();
|
// Serialised: guard exit bounds the site's interval exactly.
|
||||||
|
let now = preempt::rdtsc_serialising();
|
||||||
let last = LAST_SAMPLE_TSC.with(|c| c.replace(now));
|
let last = LAST_SAMPLE_TSC.with(|c| c.replace(now));
|
||||||
if pct > 0 {
|
if pct > 0 {
|
||||||
if last != 0 {
|
if last != 0 {
|
||||||
@@ -386,7 +392,9 @@ mod inner {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
} else if new == target && old != target {
|
} else if new == target && old != target {
|
||||||
LAST_SAMPLE_TSC.with(|c| c.set(preempt::rdtsc()));
|
// Serialised: an early arm would let pre-site work leak into the
|
||||||
|
// first in-site interval — the over-attribution this guards.
|
||||||
|
LAST_SAMPLE_TSC.with(|c| c.set(preempt::rdtsc_serialising()));
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+55
-32
@@ -102,6 +102,7 @@ pub fn channel<T>() -> (Sender<T>, Receiver<T>) {
|
|||||||
let inner = Arc::new(RawMutex::new_channel(Inner {
|
let inner = Arc::new(RawMutex::new_channel(Inner {
|
||||||
queue: VecDeque::new(),
|
queue: VecDeque::new(),
|
||||||
parked_receiver: None,
|
parked_receiver: None,
|
||||||
|
rt: None,
|
||||||
senders: 1,
|
senders: 1,
|
||||||
receiver_alive: true,
|
receiver_alive: true,
|
||||||
}));
|
}));
|
||||||
@@ -115,19 +116,36 @@ pub fn channel<T>() -> (Sender<T>, Receiver<T>) {
|
|||||||
|
|
||||||
struct Inner<T> {
|
struct Inner<T> {
|
||||||
queue: VecDeque<T>,
|
queue: VecDeque<T>,
|
||||||
/// The parked receiver's `(pid, park-epoch, runtime)`, if one is currently
|
/// The parked receiver's `(pid, park-epoch)`, if one is currently
|
||||||
/// waiting. The epoch identifies exactly which wait this is, so a waker
|
/// waiting. The epoch identifies exactly which wait this is, so a waker
|
||||||
/// left over from a wait that already ended (a losing `select` arm, a
|
/// left over from a wait that already ended (a losing `select` arm, a
|
||||||
/// `recv_timeout` whose timer fired after it was already satisfied) is
|
/// `recv_timeout` whose timer fired after it was already satisfied) is
|
||||||
/// inert and does nothing when it fires. The `Weak<RuntimeInner>` is the
|
/// inert and does nothing when it fires.
|
||||||
/// receiver's runtime, captured while it parked (so provably alive then);
|
parked_receiver: Option<(Pid, u32)>,
|
||||||
/// it lets a sender on a foreign OS thread wake the receiver without the
|
/// The receiver's runtime, captured the first time it parks (so provably
|
||||||
/// `RUNTIME` thread-local, which is unset off a scheduler thread.
|
/// alive then) and kept for the life of the channel: it lets a sender on a
|
||||||
parked_receiver: Option<(Pid, u32, Weak<RuntimeInner>)>,
|
/// foreign OS thread wake the receiver without the `RUNTIME` thread-local,
|
||||||
|
/// which is unset off a scheduler thread. Captured once rather than per
|
||||||
|
/// park because `Arc::downgrade` + drop is a locked RMW pair on a shared
|
||||||
|
/// counter, and parking is the channel hot path. A `Receiver` never
|
||||||
|
/// migrates between runtimes — it is pinned to its actor — so one capture
|
||||||
|
/// stays correct for every later park.
|
||||||
|
rt: Option<Weak<RuntimeInner>>,
|
||||||
senders: usize,
|
senders: usize,
|
||||||
receiver_alive: bool,
|
receiver_alive: bool,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
impl<T> Inner<T> {
|
||||||
|
/// Capture the receiver's runtime if we have not already. Called under the
|
||||||
|
/// channel lock at every park site; after the first park it is one branch
|
||||||
|
/// on an `Option`, no atomics.
|
||||||
|
fn note_runtime(&mut self) {
|
||||||
|
if self.rt.is_none() {
|
||||||
|
self.rt = crate::scheduler::runtime_weak();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// The sending half of a channel, created by [`channel`]. Clonable: every
|
/// The sending half of a channel, created by [`channel`]. Clonable: every
|
||||||
/// clone pushes onto the same queue, and the channel stays open as long as
|
/// clone pushes onto the same queue, and the channel stays open as long as
|
||||||
/// any clone is alive. Dropping the last `Sender` closes the channel, which
|
/// any clone is alive. Dropping the last `Sender` closes the channel, which
|
||||||
@@ -210,8 +228,8 @@ impl<T> Drop for Sender<T> {
|
|||||||
None
|
None
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
if let Some((pid, epoch, rt)) = unpark {
|
if let Some((pid, epoch)) = unpark {
|
||||||
crate::scheduler::unpark_at_via(pid, epoch, &rt);
|
crate::scheduler::unpark_at_via(pid, epoch, || self.inner.lock().rt.clone());
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -245,6 +263,16 @@ impl<T> Sender<T> {
|
|||||||
self.inner.lock().queue.len()
|
self.inner.lock().queue.len()
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Whether the [`Receiver`] is still alive (a send would be accepted).
|
||||||
|
pub(crate) fn receiver_alive(&self) -> bool {
|
||||||
|
self.inner.lock().receiver_alive
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether `other` is a sender of this very channel (a clone).
|
||||||
|
pub(crate) fn same_channel(&self, other: &Sender<T>) -> bool {
|
||||||
|
Arc::ptr_eq(&self.inner, &other.inner)
|
||||||
|
}
|
||||||
|
|
||||||
/// Push `value` onto the channel. Succeeds unconditionally as long as
|
/// Push `value` onto the channel. Succeeds unconditionally as long as
|
||||||
/// the [`Receiver`] is still alive: the queue has no capacity limit, so
|
/// the [`Receiver`] is still alive: the queue has no capacity limit, so
|
||||||
/// this never blocks and never fails except when the channel is closed,
|
/// this never blocks and never fails except when the channel is closed,
|
||||||
@@ -258,13 +286,13 @@ impl<T> Sender<T> {
|
|||||||
g.queue.push_back(value);
|
g.queue.push_back(value);
|
||||||
g.parked_receiver.take()
|
g.parked_receiver.take()
|
||||||
};
|
};
|
||||||
if let Some((pid, epoch, rt)) = unpark {
|
if let Some((pid, epoch)) = unpark {
|
||||||
crate::te!(crate::trace::Event::Send {
|
crate::te!(crate::trace::Event::Send {
|
||||||
sender: crate::actor::current_pid()
|
sender: crate::actor::current_pid()
|
||||||
.unwrap_or(crate::pid::Pid::new(u32::MAX, u32::MAX)),
|
.unwrap_or(crate::pid::Pid::new(u32::MAX, u32::MAX)),
|
||||||
receiver: Some(pid)
|
receiver: Some(pid)
|
||||||
});
|
});
|
||||||
crate::scheduler::unpark_at_via(pid, epoch, &rt);
|
crate::scheduler::unpark_at_via(pid, epoch, || self.inner.lock().rt.clone());
|
||||||
} else {
|
} else {
|
||||||
crate::te!(crate::trace::Event::Send {
|
crate::te!(crate::trace::Event::Send {
|
||||||
sender: crate::actor::current_pid()
|
sender: crate::actor::current_pid()
|
||||||
@@ -297,17 +325,14 @@ impl<T> Receiver<T> {
|
|||||||
None => panic!("smarm: recv() called outside an actor"),
|
None => panic!("smarm: recv() called outside an actor"),
|
||||||
};
|
};
|
||||||
debug_assert!(
|
debug_assert!(
|
||||||
g.parked_receiver.as_ref().is_none_or(|(p, _, _)| *p == me),
|
g.parked_receiver.is_none_or(|(p, _)| p == me),
|
||||||
"channel has more than one receiver"
|
"channel has more than one receiver"
|
||||||
);
|
);
|
||||||
// begin_wait is lock-free, so it's legal under the Channel lock;
|
// begin_wait is lock-free, so it's legal under the Channel lock;
|
||||||
// registering in the same critical section makes the epoch
|
// registering in the same critical section makes the epoch
|
||||||
// atomic with the senders' view of the registration.
|
// atomic with the senders' view of the registration.
|
||||||
g.parked_receiver = Some((
|
g.note_runtime();
|
||||||
me,
|
g.parked_receiver = Some((me, crate::scheduler::begin_wait()));
|
||||||
crate::scheduler::begin_wait(),
|
|
||||||
crate::scheduler::runtime_weak(),
|
|
||||||
));
|
|
||||||
crate::te!(crate::trace::Event::RecvPark(me));
|
crate::te!(crate::trace::Event::RecvPark(me));
|
||||||
}
|
}
|
||||||
// Release the lock before parking: the unparker will need it.
|
// Release the lock before parking: the unparker will need it.
|
||||||
@@ -355,11 +380,12 @@ impl<T> Receiver<T> {
|
|||||||
return Err(RecvTimeoutError::Disconnected);
|
return Err(RecvTimeoutError::Disconnected);
|
||||||
}
|
}
|
||||||
debug_assert!(
|
debug_assert!(
|
||||||
g.parked_receiver.as_ref().is_none_or(|(p, _, _)| *p == me),
|
g.parked_receiver.is_none_or(|(p, _)| p == me),
|
||||||
"channel has more than one receiver"
|
"channel has more than one receiver"
|
||||||
);
|
);
|
||||||
epoch = crate::scheduler::begin_wait();
|
epoch = crate::scheduler::begin_wait();
|
||||||
g.parked_receiver = Some((me, epoch, crate::scheduler::runtime_weak()));
|
g.note_runtime();
|
||||||
|
g.parked_receiver = Some((me, epoch));
|
||||||
crate::te!(crate::trace::Event::RecvPark(me));
|
crate::te!(crate::trace::Event::RecvPark(me));
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -431,14 +457,11 @@ impl<T> Receiver<T> {
|
|||||||
None => panic!("smarm: recv_match() called outside an actor"),
|
None => panic!("smarm: recv_match() called outside an actor"),
|
||||||
};
|
};
|
||||||
debug_assert!(
|
debug_assert!(
|
||||||
g.parked_receiver.as_ref().is_none_or(|(p, _, _)| *p == me),
|
g.parked_receiver.is_none_or(|(p, _)| p == me),
|
||||||
"channel has more than one receiver"
|
"channel has more than one receiver"
|
||||||
);
|
);
|
||||||
g.parked_receiver = Some((
|
g.note_runtime();
|
||||||
me,
|
g.parked_receiver = Some((me, crate::scheduler::begin_wait()));
|
||||||
crate::scheduler::begin_wait(),
|
|
||||||
crate::scheduler::runtime_weak(),
|
|
||||||
));
|
|
||||||
crate::te!(crate::trace::Event::RecvPark(me));
|
crate::te!(crate::trace::Event::RecvPark(me));
|
||||||
}
|
}
|
||||||
// Release the lock before parking: the unparker will need it.
|
// Release the lock before parking: the unparker will need it.
|
||||||
@@ -509,12 +532,11 @@ impl<T: Send + 'static> crate::timer::TimerTarget for RawMutex<Inner<T>> {
|
|||||||
// keeps the registration bookkeeping exact.)
|
// keeps the registration bookkeeping exact.)
|
||||||
let unpark = {
|
let unpark = {
|
||||||
let mut g = self.lock();
|
let mut g = self.lock();
|
||||||
match g.parked_receiver {
|
if g.parked_receiver == Some((pid, epoch)) {
|
||||||
Some((p, e, _)) if p == pid && e == epoch => {
|
g.parked_receiver = None;
|
||||||
g.parked_receiver = None;
|
true
|
||||||
true
|
} else {
|
||||||
}
|
false
|
||||||
_ => false,
|
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
// Unpark outside the channel lock: it may take the run-queue lock;
|
// Unpark outside the channel lock: it may take the run-queue lock;
|
||||||
@@ -575,10 +597,11 @@ impl<T> Selectable for Receiver<T> {
|
|||||||
return Ok(false);
|
return Ok(false);
|
||||||
}
|
}
|
||||||
debug_assert!(
|
debug_assert!(
|
||||||
g.parked_receiver.as_ref().is_none_or(|(p, _, _)| *p == pid),
|
g.parked_receiver.is_none_or(|(p, _)| p == pid),
|
||||||
"channel has more than one receiver"
|
"channel has more than one receiver"
|
||||||
);
|
);
|
||||||
g.parked_receiver = Some((pid, epoch, crate::scheduler::runtime_weak()));
|
g.note_runtime();
|
||||||
|
g.parked_receiver = Some((pid, epoch));
|
||||||
Ok(true)
|
Ok(true)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+282
@@ -0,0 +1,282 @@
|
|||||||
|
//! RFC 010 — clustering (smarm⇄smarm, explicit remote boundary).
|
||||||
|
//!
|
||||||
|
//! c1: feature flag + optional deps. c2: the owned envelope. c3: the
|
||||||
|
//! transport trait (control connection), framed codec, and the TCP +
|
||||||
|
//! loopback impls. c5: the handshake state machine. c6: the connection
|
||||||
|
//! [`manager`] (registry) and per-peer connection actors ([`conn`]), started
|
||||||
|
//! as an explicit supervision subtree, plus the handshake on the
|
||||||
|
//! accept/connect path ([`connect`]). Everything above them lands in later
|
||||||
|
//! chunks.
|
||||||
|
|
||||||
|
pub mod conn;
|
||||||
|
pub mod connect;
|
||||||
|
pub mod connector;
|
||||||
|
pub mod discovery;
|
||||||
|
pub mod envelope;
|
||||||
|
pub mod expose;
|
||||||
|
pub mod handshake;
|
||||||
|
pub mod manager;
|
||||||
|
pub mod membership;
|
||||||
|
pub mod pg;
|
||||||
|
pub mod remote;
|
||||||
|
pub mod transport;
|
||||||
|
|
||||||
|
use std::io;
|
||||||
|
use std::time::{Duration, SystemTime, UNIX_EPOCH};
|
||||||
|
|
||||||
|
use crate::gen_server::{self, GenServerBuilder};
|
||||||
|
use crate::monitor::monitor;
|
||||||
|
use crate::pg::Incarnation;
|
||||||
|
use crate::scheduler::{sleep, spawn, JoinHandle};
|
||||||
|
use crate::supervisor::{ChildSpec, OneForOne, Restart};
|
||||||
|
|
||||||
|
use envelope::NodeMeta;
|
||||||
|
use handshake::Local;
|
||||||
|
use transport::tcp::TcpTransport;
|
||||||
|
use transport::Transport;
|
||||||
|
|
||||||
|
pub use conn::{spawn_established, ConnHandle};
|
||||||
|
pub use connect::{dial, spawn_acceptor, AcceptorHandle};
|
||||||
|
pub use connector::{spawn_connector, ConnectorHandle};
|
||||||
|
pub use discovery::{Discovery, StaticSeeds, Strategy};
|
||||||
|
pub use envelope::RemoteDownReason;
|
||||||
|
pub use expose::{expose, expose_type, type_hash, DeliverError};
|
||||||
|
pub use manager::{Manager, MANAGER};
|
||||||
|
pub use membership::{subscribe, view, MembershipEvents, NodeEvent, NodeInfo};
|
||||||
|
pub use pg::{dispatch_any, members_all, pick_any, DispatchAnyError, GroupMember, PgMsg, PG_NAME};
|
||||||
|
pub use remote::{
|
||||||
|
demonitor_remote, monitor_remote, send_to_remote, NotConnected, RemoteDown, RemoteMonitor,
|
||||||
|
RemoteName, RemotePid, RemoteSendError, ToRemoteError,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// c6d — the derived build hash for [`handshake::LocalNode::build_hash`]:
|
||||||
|
/// two builds may mesh only when this matches, and it is a pure function of
|
||||||
|
/// the compile-time inputs that define wire compatibility today — the exact
|
||||||
|
/// toolchain (`rustc -V`), the declared feature set, and
|
||||||
|
/// [`envelope::PROTO_VERSION`]. FNV-1a 64 over the build-script string, then
|
||||||
|
/// the proto version folded byte-wise, so a proto bump moves the hash even
|
||||||
|
/// on an identical toolchain. The domain is deliberately lean and
|
||||||
|
/// tightenable later without a wire change — it is just a `u64`.
|
||||||
|
pub const BUILD_HASH: u64 = fold_u32(
|
||||||
|
fnv1a64(env!("SMARM_BUILD_HASH_INPUTS").as_bytes()),
|
||||||
|
envelope::PROTO_VERSION,
|
||||||
|
);
|
||||||
|
|
||||||
|
/// FNV-1a 64 (const so [`BUILD_HASH`] is a compile-time fact).
|
||||||
|
const fn fnv1a64(bytes: &[u8]) -> u64 {
|
||||||
|
let mut h: u64 = 0xcbf2_9ce4_8422_2325;
|
||||||
|
let mut i = 0;
|
||||||
|
while i < bytes.len() {
|
||||||
|
h ^= bytes[i] as u64;
|
||||||
|
h = h.wrapping_mul(0x0000_0100_0000_01b3);
|
||||||
|
i += 1;
|
||||||
|
}
|
||||||
|
h
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Continue an FNV-1a state over a `u32`'s little-endian bytes.
|
||||||
|
const fn fold_u32(mut h: u64, v: u32) -> u64 {
|
||||||
|
let b = v.to_le_bytes();
|
||||||
|
let mut i = 0;
|
||||||
|
while i < b.len() {
|
||||||
|
h ^= b[i] as u64;
|
||||||
|
h = h.wrapping_mul(0x0000_0100_0000_01b3);
|
||||||
|
i += 1;
|
||||||
|
}
|
||||||
|
h
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The control-plane timing knobs, all with today's fixed values as
|
||||||
|
/// defaults ([`Timing::default`]). One struct threaded explicitly to the
|
||||||
|
/// acceptor, the dial path, every connection actor and the connector — no
|
||||||
|
/// ambient state, so a test can run a fast mesh without touching globals.
|
||||||
|
/// Every node in a mesh should agree on `heartbeat_interval` <
|
||||||
|
/// `liveness_timeout`; nothing enforces it.
|
||||||
|
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||||
|
pub struct Timing {
|
||||||
|
/// Idle-connection heartbeat pace. Default [`conn::HEARTBEAT_INTERVAL`].
|
||||||
|
pub heartbeat_interval: Duration,
|
||||||
|
/// Inbound silence that tears a connection down. Default
|
||||||
|
/// [`conn::LIVENESS_TIMEOUT`].
|
||||||
|
pub liveness_timeout: Duration,
|
||||||
|
/// Per-frame handshake deadline on the accept/dial path. Default
|
||||||
|
/// [`connect::HANDSHAKE_TIMEOUT`].
|
||||||
|
pub handshake_timeout: Duration,
|
||||||
|
/// Connector redial delay after the first failure. Default
|
||||||
|
/// [`connector::INITIAL_BACKOFF`].
|
||||||
|
pub initial_backoff: Duration,
|
||||||
|
/// Connector redial delay cap. Default [`connector::MAX_BACKOFF`].
|
||||||
|
pub max_backoff: Duration,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Default for Timing {
|
||||||
|
fn default() -> Self {
|
||||||
|
Timing {
|
||||||
|
heartbeat_interval: conn::HEARTBEAT_INTERVAL,
|
||||||
|
liveness_timeout: conn::LIVENESS_TIMEOUT,
|
||||||
|
handshake_timeout: connect::HANDSHAKE_TIMEOUT,
|
||||||
|
initial_backoff: connector::INITIAL_BACKOFF,
|
||||||
|
max_backoff: connector::MAX_BACKOFF,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// How to run this node: its identity and how it finds peers.
|
||||||
|
pub struct Config {
|
||||||
|
/// This node's claimed name — the mesh-wide identity peers dial by and
|
||||||
|
/// the tie-break input. Must be unique across the mesh.
|
||||||
|
pub node_name: String,
|
||||||
|
/// Metadata offered in this node's `Hello`.
|
||||||
|
pub meta: NodeMeta,
|
||||||
|
/// The control-connection listen address (e.g. `"127.0.0.1:0"`; the
|
||||||
|
/// concrete bound address is [`Cluster::local_addr`]).
|
||||||
|
pub listen_addr: String,
|
||||||
|
/// The peer-discovery strategy — [`StaticSeeds`] until richer ones land.
|
||||||
|
pub strategy: Box<dyn Strategy>,
|
||||||
|
/// Heartbeat / liveness / handshake / backoff knobs; [`Timing::default`]
|
||||||
|
/// is the shipping configuration.
|
||||||
|
pub timing: Timing,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A running cluster node: the supervised [`Manager`], the acceptor over the
|
||||||
|
/// bound listener, and the connector driving its [`Strategy`]. Roles will
|
||||||
|
/// eventually mount this; until the role mechanism lands it is started by
|
||||||
|
/// hand (RFC 010 §7).
|
||||||
|
///
|
||||||
|
/// Dropping the handle stops the acceptor and connector loops (no new
|
||||||
|
/// connections in either direction) but detaches the manager subtree, which
|
||||||
|
/// — with every established connection — keeps running for the life of the
|
||||||
|
/// runtime, the same split as [`AcceptorHandle`] alone.
|
||||||
|
pub struct Cluster {
|
||||||
|
_sup: JoinHandle,
|
||||||
|
acceptor: AcceptorHandle,
|
||||||
|
connector: ConnectorHandle,
|
||||||
|
local: Local,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Cluster {
|
||||||
|
/// The concrete bound listen address, dialable as-is.
|
||||||
|
pub fn local_addr(&self) -> &str {
|
||||||
|
self.acceptor.local_addr()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// This node's handshake identity (name, incarnation, build hash, meta).
|
||||||
|
pub fn local(&self) -> &Local {
|
||||||
|
&self.local
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Stop accepting and dialing. Established connections stay up (they
|
||||||
|
/// belong to the manager); tear those down via the manager.
|
||||||
|
pub fn shutdown(&self) {
|
||||||
|
self.acceptor.shutdown();
|
||||||
|
self.connector.shutdown();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Start a cluster node: the supervised manager (blocking until it is
|
||||||
|
/// registered and ready to answer), the acceptor bound per
|
||||||
|
/// [`Config::listen_addr`], and the connector running [`Config::strategy`].
|
||||||
|
/// The node's identity is completed here: `incarnation` is
|
||||||
|
/// [`self_incarnation`] and `build_hash` is [`BUILD_HASH`] — c7 is its first
|
||||||
|
/// consumer. Errs only if the listener cannot bind.
|
||||||
|
///
|
||||||
|
/// The manager is a supervised child (restarted on crash); per-peer
|
||||||
|
/// connection actors are dynamic and monitored by the manager rather than
|
||||||
|
/// statically supervised — a lost connection is re-established by the
|
||||||
|
/// connector's dial loop, never resurrected onto a stale socket.
|
||||||
|
pub fn start(config: Config) -> io::Result<Cluster> {
|
||||||
|
let sup = spawn(|| {
|
||||||
|
OneForOne::new()
|
||||||
|
.child(ChildSpec::new(Restart::Permanent, manager_child))
|
||||||
|
.run()
|
||||||
|
});
|
||||||
|
while gen_server::whereis_server(MANAGER).is_none() {
|
||||||
|
sleep(Duration::from_millis(1));
|
||||||
|
}
|
||||||
|
let local = Local {
|
||||||
|
node_name: config.node_name,
|
||||||
|
incarnation: self_incarnation(),
|
||||||
|
build_hash: BUILD_HASH,
|
||||||
|
meta: config.meta,
|
||||||
|
};
|
||||||
|
// The wire identity serialized pids are stamped with (c10).
|
||||||
|
remote::set_local_identity(&local.node_name, local.incarnation);
|
||||||
|
// The pg actor (Phase 5): subscribes membership, owns the "pg" name.
|
||||||
|
pg::attach_cluster();
|
||||||
|
let listener = TcpTransport.listen(&config.listen_addr)?;
|
||||||
|
let acceptor = spawn_acceptor(listener, local.clone(), config.timing);
|
||||||
|
let connector = spawn_connector(
|
||||||
|
Box::new(TcpTransport),
|
||||||
|
local.clone(),
|
||||||
|
config.strategy,
|
||||||
|
config.timing,
|
||||||
|
);
|
||||||
|
Ok(Cluster {
|
||||||
|
_sup: sup,
|
||||||
|
acceptor,
|
||||||
|
connector,
|
||||||
|
local,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/// This process's incarnation epoch: milliseconds since the Unix epoch,
|
||||||
|
/// truncated to `u32`. Not a clock — its one job is separating a node from
|
||||||
|
/// its own restart (two starts of the same name land on the same value only
|
||||||
|
/// if they happen within the same millisecond modulo ~49.7 days). Seconds
|
||||||
|
/// would be too coarse: a crash-and-restart inside one second is routine
|
||||||
|
/// under supervision.
|
||||||
|
pub fn self_incarnation() -> Incarnation {
|
||||||
|
let ms = SystemTime::now()
|
||||||
|
.duration_since(UNIX_EPOCH)
|
||||||
|
.map(|d| d.as_millis())
|
||||||
|
.unwrap_or(0);
|
||||||
|
Incarnation::new(ms as u32)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The supervised manager child body. It *is* the child actor: it starts the
|
||||||
|
/// named manager, then parks on the manager's own termination so this actor's
|
||||||
|
/// lifetime tracks the manager's — the supervisor's restart accounting keys off
|
||||||
|
/// this actor exiting.
|
||||||
|
fn manager_child() {
|
||||||
|
let m = match GenServerBuilder::new(Manager::new()).named(MANAGER).start() {
|
||||||
|
Ok(m) => m,
|
||||||
|
// Name still held by a not-yet-reaped prior instance: return and let
|
||||||
|
// the supervisor retry under its restart policy.
|
||||||
|
Err(_) => return,
|
||||||
|
};
|
||||||
|
let _ = monitor(m.pid()).rx.recv();
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
/// The hash core against the published FNV-1a 64 test vectors — the
|
||||||
|
/// contract is "this is FNV-1a", not "whatever the fn does".
|
||||||
|
#[test]
|
||||||
|
fn fnv1a64_known_vectors() {
|
||||||
|
assert_eq!(fnv1a64(b""), 0xcbf2_9ce4_8422_2325);
|
||||||
|
assert_eq!(fnv1a64(b"a"), 0xaf63_dc4c_8601_ec8c);
|
||||||
|
assert_eq!(fnv1a64(b"foobar"), 0x85944171f73967e8);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Folding the proto version continues the same FNV state: identical
|
||||||
|
/// inputs with a different version must land on a different hash.
|
||||||
|
#[test]
|
||||||
|
fn proto_version_moves_the_hash() {
|
||||||
|
let base = fnv1a64(b"same-toolchain;features=CLUSTER");
|
||||||
|
assert_ne!(fold_u32(base, 1), fold_u32(base, 2));
|
||||||
|
// And it equals hashing the bytes in one pass — the fold is a
|
||||||
|
// continuation, not a second construction.
|
||||||
|
let mut all = b"same-toolchain;features=CLUSTER".to_vec();
|
||||||
|
all.extend_from_slice(&1u32.to_le_bytes());
|
||||||
|
assert_eq!(fold_u32(base, 1), fnv1a64(&all));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The derived constant exists, is compile-time, and is not degenerate.
|
||||||
|
#[test]
|
||||||
|
fn build_hash_is_nonzero() {
|
||||||
|
const H: u64 = BUILD_HASH;
|
||||||
|
assert_ne!(H, 0);
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,630 @@
|
|||||||
|
//! RFC 010 c6 — the per-peer connection actor.
|
||||||
|
//!
|
||||||
|
//! One actor per established control connection. It owns the whole
|
||||||
|
//! [`FramedConn`] and, in a single [`select`](crate::select), waits on two
|
||||||
|
//! things at once: its command inbox and the connection becoming readable (the
|
||||||
|
//! [`FdArm`](crate::scheduler::FdArm) the transport hands back). That is why it
|
||||||
|
//! is a plain select-loop actor rather than a `gen_server` or `gen_statem` —
|
||||||
|
//! neither of those can fold fd-readiness into its wait, and folding it in is
|
||||||
|
//! the whole job. The single owner sends and receives on the one `FramedConn`,
|
||||||
|
//! so no read/write split is needed.
|
||||||
|
//!
|
||||||
|
//! The handshake completes *before* this actor exists (on the accept/connect
|
||||||
|
//! path — c6b) and produces the [`Peer`]; the *path* then registers the
|
||||||
|
//! connection with the [`manager`](crate::cluster::manager), which takes
|
||||||
|
//! ownership of its [`ConnHandle`] and monitors the actor, so any exit
|
||||||
|
//! deregisters the connection. The actor itself holds no authority over its
|
||||||
|
//! own lifetime: it runs until the manager drops its handle (deregistration,
|
||||||
|
//! `Disconnect`, or manager shutdown), the connection ends, or liveness
|
||||||
|
//! expires. Heartbeat send and fixed-timeout liveness are the timeout arm of
|
||||||
|
//! the same `select` (c6c): [`HEARTBEAT_INTERVAL`] paces outbound
|
||||||
|
//! [`Frame::Heartbeat`](crate::cluster::envelope::Frame::Heartbeat)s, and a
|
||||||
|
//! [`LIVENESS_TIMEOUT`] window — reset by any inbound frame — tears the
|
||||||
|
//! connection down when it empties.
|
||||||
|
//!
|
||||||
|
//! c9 adds the third arm — the connection's dedicated **outbound inbox**
|
||||||
|
//! (`Sender<Frame>` bound in the manager-maintained outbound table, D13),
|
||||||
|
//! drained onto the wire in the same loop — and inbound *interpretation*:
|
||||||
|
//! `SendNamed` goes to the one resolution seam,
|
||||||
|
//! [`remote::deliver_named`](crate::cluster::remote::deliver_named).
|
||||||
|
//! `Send` goes to the pid seam (c10). The outbound
|
||||||
|
//! sender is a separate channel from `cmd_tx` on purpose: closing it is not
|
||||||
|
//! a stop signal — lifetime authority stays with the [`ConnHandle`] (D9).
|
||||||
|
//!
|
||||||
|
//! c12 adds the monitor plane, and it lives *here* on purpose. Two tables,
|
||||||
|
//! both owned by this actor and dying with the connection:
|
||||||
|
//!
|
||||||
|
//! - **outstanding** — monitors *this* node holds on actors at the peer:
|
||||||
|
//! `monitor_id → (target, Sender<RemoteDown>)`. Fed by
|
||||||
|
//! [`MonCmd`](crate::cluster::remote::MonCmd) from `monitor_remote`; the
|
||||||
|
//! actor records the id and *then* emits the `Monitor` frame, so a `Down`
|
||||||
|
//! frame can never race an entry that isn't there yet. An inbound `Down`
|
||||||
|
//! removes the entry and delivers.
|
||||||
|
//! - **watched** — monitors the *peer* holds on actors here: `monitor_id →
|
||||||
|
//! local Monitor`. An inbound `Monitor` is admitted only for a pid that
|
||||||
|
//! was exposed or crossed the wire (`is_watchable`, D12): a corpse answers
|
||||||
|
//! with its recorded terminal reason (RFC §6), an unwatchable or unknown
|
||||||
|
//! pid with `NoProc` — indistinguishable from dead, so nothing leaks. A
|
||||||
|
//! live watchable pid gets a local monitor whose `rx` is one more arm of
|
||||||
|
//! the select; its `Down` goes back as a frame.
|
||||||
|
//!
|
||||||
|
//! Because both tables are actor state, connection loss (c13) needs no
|
||||||
|
//! second bookkeeping owner: this actor's exit is the one place that knows
|
||||||
|
//! every monitor the link was carrying. `Monitors::teardown` runs on every
|
||||||
|
//! exit path and answers each outstanding monitor with `Disconnected` —
|
||||||
|
//! the roadmap's "partition vs. death" contrast: an actor that dies sends
|
||||||
|
//! its true reason over the link, a link that dies says only that.
|
||||||
|
|
||||||
|
use std::collections::HashMap;
|
||||||
|
|
||||||
|
use std::time::{Duration, Instant};
|
||||||
|
|
||||||
|
use crate::channel::{channel, try_select_timeout, Receiver, Selectable, Sender};
|
||||||
|
use crate::cluster::envelope::{Frame, RemoteDownReason};
|
||||||
|
use crate::cluster::handshake::Peer;
|
||||||
|
use crate::cluster::manager::{Call, Registered, Reply, MANAGER};
|
||||||
|
use crate::cluster::remote::{
|
||||||
|
deliver_named, deliver_to_pid, InboundVerdict, MonCmd, RemoteDown, RemotePid,
|
||||||
|
};
|
||||||
|
use crate::cluster::transport::FramedConn;
|
||||||
|
use crate::cluster::Timing;
|
||||||
|
use crate::gen_server;
|
||||||
|
use crate::monitor::{
|
||||||
|
demonitor, is_watchable, monitor, terminal_reason, DownReason, Monitor, MonitorId,
|
||||||
|
};
|
||||||
|
use crate::pid::{Erased, Pid};
|
||||||
|
use crate::scheduler::spawn;
|
||||||
|
|
||||||
|
/// Commands to a running connection actor.
|
||||||
|
enum Cmd {
|
||||||
|
Shutdown,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The manager's authority over one connection actor: while this handle
|
||||||
|
/// lives the connection lives, and dropping it stops the actor and closes
|
||||||
|
/// the socket. Only the [`manager`](crate::cluster::manager) holds one —
|
||||||
|
/// callers of [`spawn_established`] get a [`Pid`] and no lifetime authority,
|
||||||
|
/// so a connection can never outlive, or die with, whichever actor happened
|
||||||
|
/// to establish it.
|
||||||
|
pub struct ConnHandle {
|
||||||
|
cmd_tx: Sender<Cmd>,
|
||||||
|
/// The connection's dedicated outbound inboxes — frames and monitor
|
||||||
|
/// commands. The manager moves them into the outbound table on
|
||||||
|
/// `Register` (see [`take_outbound`](ConnHandle::take_outbound)); a
|
||||||
|
/// `Duplicate` verdict drops them with the handle.
|
||||||
|
out_tx: Option<(Sender<Frame>, Sender<MonCmd>)>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl std::fmt::Debug for ConnHandle {
|
||||||
|
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||||
|
f.write_str("ConnHandle")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl ConnHandle {
|
||||||
|
/// Ask the connection to close and exit. Idempotent, and a no-op if the
|
||||||
|
/// actor has already gone. Dropping the handle does the same thing; this
|
||||||
|
/// exists for the manager's explicit `Disconnect` path.
|
||||||
|
pub fn shutdown(&self) {
|
||||||
|
let _ = self.cmd_tx.send(Cmd::Shutdown);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Manager-only: take the outbound senders to bind into the outbound
|
||||||
|
/// table. Once, at registration.
|
||||||
|
pub(crate) fn take_outbound(&mut self) -> Option<(Sender<Frame>, Sender<MonCmd>)> {
|
||||||
|
self.out_tx.take()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The name was already claimed by a live connection, so this one was
|
||||||
|
/// refused; its actor has been stopped and its socket closed.
|
||||||
|
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||||
|
pub struct RegisterRefused;
|
||||||
|
|
||||||
|
/// Spawn a connection actor for an **already-established** connection (the
|
||||||
|
/// handshake completed on the path and produced `peer`) and register it with
|
||||||
|
/// the manager, synchronously, before returning. The manager takes the
|
||||||
|
/// actor's [`ConnHandle`]; the caller gets only the [`Pid`], because
|
||||||
|
/// connection lifetime belongs to the table and not to the establishing
|
||||||
|
/// actor. A refusal has already stopped the actor and closed the socket.
|
||||||
|
pub fn spawn_established(
|
||||||
|
framed: FramedConn,
|
||||||
|
peer: Peer,
|
||||||
|
timing: Timing,
|
||||||
|
) -> Result<Pid, RegisterRefused> {
|
||||||
|
let (cmd_tx, cmd_rx) = channel();
|
||||||
|
let (out_tx, out_rx) = channel();
|
||||||
|
let (mon_tx, mon_rx) = channel();
|
||||||
|
let reg_peer = peer.clone();
|
||||||
|
let pid = spawn(move || run(framed, peer, timing, cmd_rx, out_rx, mon_rx)).pid();
|
||||||
|
match gen_server::call(
|
||||||
|
MANAGER,
|
||||||
|
Call::Register {
|
||||||
|
peer: reg_peer,
|
||||||
|
pid,
|
||||||
|
handle: ConnHandle {
|
||||||
|
cmd_tx,
|
||||||
|
out_tx: Some((out_tx, mon_tx)),
|
||||||
|
},
|
||||||
|
},
|
||||||
|
) {
|
||||||
|
Ok(Reply::Registered(Registered::Ok)) => Ok(pid),
|
||||||
|
// Duplicate name, or the manager is unreachable. Either way the
|
||||||
|
// handle went with the call and is dropped there (or never arrived
|
||||||
|
// and dropped with it), which stops the actor and closes the socket.
|
||||||
|
_ => Err(RegisterRefused),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Default for [`Timing::heartbeat_interval`]: how often this end emits
|
||||||
|
/// [`Frame::Heartbeat`] on an idle connection. The first one goes out
|
||||||
|
/// immediately at spawn, so the peer's liveness window starts fed.
|
||||||
|
pub const HEARTBEAT_INTERVAL: Duration = Duration::from_secs(1);
|
||||||
|
|
||||||
|
/// Default for [`Timing::liveness_timeout`]: how long the connection may go
|
||||||
|
/// without a single inbound frame before it is declared dead and torn down. Any inbound frame resets the window —
|
||||||
|
/// heartbeats keep an idle connection alive, and real traffic (c8+) counts
|
||||||
|
/// for free. Fixed by design (RFC v2 §5): this is the control connection, a
|
||||||
|
/// heartbeat can never queue behind bulk traffic, so a fixed timeout is an
|
||||||
|
/// honest detector.
|
||||||
|
pub const LIVENESS_TIMEOUT: Duration = Duration::from_secs(4);
|
||||||
|
|
||||||
|
fn run(
|
||||||
|
mut framed: FramedConn,
|
||||||
|
_peer: Peer,
|
||||||
|
timing: Timing,
|
||||||
|
cmd_rx: Receiver<Cmd>,
|
||||||
|
out_rx: Receiver<Frame>,
|
||||||
|
mon_rx: Receiver<MonCmd>,
|
||||||
|
) {
|
||||||
|
let mut mons = Monitors::default();
|
||||||
|
match framed.readable_arm() {
|
||||||
|
Some(arm) => run_live(
|
||||||
|
&mut framed,
|
||||||
|
arm,
|
||||||
|
timing,
|
||||||
|
&cmd_rx,
|
||||||
|
&out_rx,
|
||||||
|
&mon_rx,
|
||||||
|
&mut mons,
|
||||||
|
),
|
||||||
|
None => run_inert(&cmd_rx),
|
||||||
|
}
|
||||||
|
framed.close();
|
||||||
|
mons.teardown(&mon_rx);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The monitor plane's two tables (module docs). Owned by the actor.
|
||||||
|
#[derive(Default)]
|
||||||
|
struct Monitors {
|
||||||
|
/// Monitors this node holds on peer actors: id → (target, delivery).
|
||||||
|
outstanding: HashMap<MonitorId, (RemotePid<Erased>, Sender<RemoteDown>)>,
|
||||||
|
/// Monitors the peer holds on local actors: id → the local monitor.
|
||||||
|
watched: HashMap<MonitorId, Monitor>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Monitors {
|
||||||
|
/// The connection is gone, whatever the exit path (liveness expiry,
|
||||||
|
/// EOF, wire failure, commanded stop): release the peer's local
|
||||||
|
/// monitors, and answer every one of ours with `Disconnected` — nothing
|
||||||
|
/// more can be known about those actors. Commands still sitting in the
|
||||||
|
/// inbox are folded in first (a `Monitor` handed to us but never
|
||||||
|
/// processed gets its notice too; a `Demonitor` still cancels), so the
|
||||||
|
/// only registration that can miss this is one that lands after the
|
||||||
|
/// drain and before the inbox drops — the reader side backstops that
|
||||||
|
/// (`RemoteMonitor`). Entries leave the table as they are answered, and
|
||||||
|
/// this runs once per actor, so no monitor sees two notices.
|
||||||
|
fn teardown(&mut self, mon_rx: &Receiver<MonCmd>) {
|
||||||
|
for (_, m) in self.watched.drain() {
|
||||||
|
let _ = demonitor(&m);
|
||||||
|
}
|
||||||
|
while let Ok(Some(cmd)) = mon_rx.try_recv() {
|
||||||
|
match cmd {
|
||||||
|
MonCmd::Monitor { id, target, tx } => {
|
||||||
|
self.outstanding.insert(id, (target, tx));
|
||||||
|
}
|
||||||
|
MonCmd::Demonitor { id } => {
|
||||||
|
self.outstanding.remove(&id);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for (_, (pid, tx)) in self.outstanding.drain() {
|
||||||
|
let _ = tx.send(RemoteDown {
|
||||||
|
pid,
|
||||||
|
reason: RemoteDownReason::Disconnected,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Admit a peer's `Monitor` for local `(index, generation)`. Returns
|
||||||
|
/// the reason to answer with at once, or `None` if a live monitor was
|
||||||
|
/// installed. Corpse → recorded terminal reason (RFC §6, and only
|
||||||
|
/// watchable deaths are recorded); live watchable → monitor; anything
|
||||||
|
/// else → `NoProc`. The check-then-monitor race (dies in between) is
|
||||||
|
/// closed on the read side: a `NoProc` from a monitor we installed on a
|
||||||
|
/// live pid is upgraded through `terminal_reason` in `sweep_watched`.
|
||||||
|
fn admit(&mut self, id: MonitorId, index: u32, generation: u32) -> Option<DownReason> {
|
||||||
|
let pid = Pid::new(index, generation);
|
||||||
|
if let Some(reason) = terminal_reason(pid) {
|
||||||
|
return Some(reason);
|
||||||
|
}
|
||||||
|
if !is_watchable(pid) {
|
||||||
|
return Some(DownReason::NoProc);
|
||||||
|
}
|
||||||
|
let m = monitor(pid);
|
||||||
|
self.watched.insert(id, m);
|
||||||
|
None
|
||||||
|
}
|
||||||
|
|
||||||
|
fn cancel(&mut self, id: MonitorId) {
|
||||||
|
if let Some(m) = self.watched.remove(&id) {
|
||||||
|
let _ = demonitor(&m);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Collect every local `Down` that has arrived for a peer-held monitor.
|
||||||
|
fn sweep_watched(&mut self) -> Vec<(MonitorId, DownReason)> {
|
||||||
|
let mut fired = Vec::new();
|
||||||
|
for (id, m) in self.watched.iter() {
|
||||||
|
if let Ok(Some(down)) = m.rx.try_recv() {
|
||||||
|
let reason = match down.reason {
|
||||||
|
DownReason::NoProc => terminal_reason(m.target).unwrap_or(DownReason::NoProc),
|
||||||
|
r => r,
|
||||||
|
};
|
||||||
|
fired.push((*id, reason));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for (id, _) in &fired {
|
||||||
|
self.watched.remove(id);
|
||||||
|
}
|
||||||
|
fired
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The peer reports a monitored actor down: deliver locally.
|
||||||
|
fn down(&mut self, id: MonitorId, reason: RemoteDownReason) {
|
||||||
|
if let Some((pid, tx)) = self.outstanding.remove(&id) {
|
||||||
|
let _ = tx.send(RemoteDown { pid, reason });
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The steady-state loop over an fd-backed connection: one
|
||||||
|
/// `select_timeout` folds the command inbox, the outbound inbox, socket
|
||||||
|
/// readability, and the nearer of the two deadlines (`hb_send`,
|
||||||
|
/// `liveness`) into a single wait.
|
||||||
|
fn run_live(
|
||||||
|
framed: &mut FramedConn,
|
||||||
|
arm: crate::scheduler::FdArm,
|
||||||
|
timing: Timing,
|
||||||
|
cmd_rx: &Receiver<Cmd>,
|
||||||
|
out_rx: &Receiver<Frame>,
|
||||||
|
mon_rx: &Receiver<MonCmd>,
|
||||||
|
mons: &mut Monitors,
|
||||||
|
) {
|
||||||
|
let mut next_hb = Instant::now();
|
||||||
|
let mut live_until = Instant::now() + timing.liveness_timeout;
|
||||||
|
// The outbound senders live in the manager's table and are dropped on
|
||||||
|
// unbind; after that these arms would wake forever, so they drop out of
|
||||||
|
// the select (not a stop signal — see the module docs).
|
||||||
|
let mut out_open = true;
|
||||||
|
let mut mon_open = true;
|
||||||
|
// Which wait each select arm stands for. Built in lockstep with the
|
||||||
|
// `Selectable` vector each iteration, so a wake is decoded by name and
|
||||||
|
// never by position.
|
||||||
|
enum Arm {
|
||||||
|
Cmd,
|
||||||
|
Fd,
|
||||||
|
Out,
|
||||||
|
Mon,
|
||||||
|
/// A peer-held local monitor (any of them: firing sweeps them all).
|
||||||
|
Watched,
|
||||||
|
}
|
||||||
|
fn push<'s>(
|
||||||
|
arms: &mut Vec<&'s dyn Selectable>,
|
||||||
|
what: &mut Vec<Arm>,
|
||||||
|
s: &'s dyn Selectable,
|
||||||
|
a: Arm,
|
||||||
|
) {
|
||||||
|
arms.push(s);
|
||||||
|
what.push(a);
|
||||||
|
}
|
||||||
|
loop {
|
||||||
|
let now = Instant::now();
|
||||||
|
if now >= live_until {
|
||||||
|
break; // liveness expired: the peer is dead to us
|
||||||
|
}
|
||||||
|
if now >= next_hb {
|
||||||
|
if framed.send(&Frame::Heartbeat).is_err() {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
next_hb = now + timing.heartbeat_interval;
|
||||||
|
}
|
||||||
|
let wait = next_hb.min(live_until).saturating_duration_since(now);
|
||||||
|
let mut arms: Vec<&dyn Selectable> = Vec::new();
|
||||||
|
let mut what: Vec<Arm> = Vec::new();
|
||||||
|
push(&mut arms, &mut what, cmd_rx, Arm::Cmd);
|
||||||
|
push(&mut arms, &mut what, &arm, Arm::Fd);
|
||||||
|
if out_open {
|
||||||
|
push(&mut arms, &mut what, out_rx, Arm::Out);
|
||||||
|
}
|
||||||
|
if mon_open {
|
||||||
|
push(&mut arms, &mut what, mon_rx, Arm::Mon);
|
||||||
|
}
|
||||||
|
for m in mons.watched.values() {
|
||||||
|
push(&mut arms, &mut what, &m.rx, Arm::Watched);
|
||||||
|
}
|
||||||
|
match try_select_timeout(&arms, wait).map(|i| i.map(|i| &what[i])) {
|
||||||
|
Ok(Some(Arm::Cmd)) => {
|
||||||
|
if should_stop(cmd_rx) {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(Some(Arm::Fd)) => match pump_readable(framed, mons) {
|
||||||
|
Pump::Ended => break,
|
||||||
|
Pump::Frames(n) => {
|
||||||
|
if n > 0 {
|
||||||
|
live_until = Instant::now() + timing.liveness_timeout;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
},
|
||||||
|
Ok(Some(Arm::Out)) => match pump_outbound(framed, out_rx) {
|
||||||
|
Outbound::Drained => {}
|
||||||
|
Outbound::Closed => out_open = false,
|
||||||
|
Outbound::WireFailed => break,
|
||||||
|
},
|
||||||
|
Ok(Some(Arm::Mon)) => match pump_moncmds(framed, mon_rx, mons) {
|
||||||
|
Outbound::Drained => {}
|
||||||
|
Outbound::Closed => mon_open = false,
|
||||||
|
Outbound::WireFailed => break,
|
||||||
|
},
|
||||||
|
Ok(Some(Arm::Watched)) => {
|
||||||
|
for (id, reason) in mons.sweep_watched() {
|
||||||
|
let frame = Frame::Down {
|
||||||
|
monitor_id: id.0,
|
||||||
|
reason: reason.into(),
|
||||||
|
};
|
||||||
|
if framed.send(&frame).is_err() {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// A deadline passed; the top of the loop acts on whichever.
|
||||||
|
Ok(None) => {}
|
||||||
|
// The fd arm failed to register — the connection is gone.
|
||||||
|
Err(_) => break,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Drain the monitor-command inbox: record, then emit (module docs).
|
||||||
|
fn pump_moncmds(
|
||||||
|
framed: &mut FramedConn,
|
||||||
|
mon_rx: &Receiver<MonCmd>,
|
||||||
|
mons: &mut Monitors,
|
||||||
|
) -> Outbound {
|
||||||
|
loop {
|
||||||
|
match mon_rx.try_recv() {
|
||||||
|
Ok(Some(MonCmd::Monitor { id, target, tx })) => {
|
||||||
|
let frame = Frame::Monitor {
|
||||||
|
monitor_id: id.0,
|
||||||
|
index: target.index(),
|
||||||
|
generation: target.generation(),
|
||||||
|
};
|
||||||
|
mons.outstanding.insert(id, (target, tx));
|
||||||
|
if framed.send(&frame).is_err() {
|
||||||
|
return Outbound::WireFailed;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(Some(MonCmd::Demonitor { id })) => {
|
||||||
|
if mons.outstanding.remove(&id).is_some()
|
||||||
|
&& framed.send(&Frame::Demonitor { monitor_id: id.0 }).is_err()
|
||||||
|
{
|
||||||
|
return Outbound::WireFailed;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(None) => return Outbound::Drained,
|
||||||
|
Err(_) => return Outbound::Closed,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// What one outbound-side wake (frames or monitor commands) yielded.
|
||||||
|
enum Outbound {
|
||||||
|
/// Everything queued went onto the wire; the inbox is open and empty.
|
||||||
|
Drained,
|
||||||
|
/// The manager unbound this connection's sender; nothing more will come.
|
||||||
|
Closed,
|
||||||
|
/// The socket refused a write: the connection is gone.
|
||||||
|
WireFailed,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Drain every queued outbound frame onto the wire.
|
||||||
|
fn pump_outbound(framed: &mut FramedConn, out_rx: &Receiver<Frame>) -> Outbound {
|
||||||
|
loop {
|
||||||
|
match out_rx.try_recv() {
|
||||||
|
Ok(Some(frame)) => {
|
||||||
|
if framed.send(&frame).is_err() {
|
||||||
|
return Outbound::WireFailed;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(None) => return Outbound::Drained,
|
||||||
|
Err(_) => return Outbound::Closed,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// No fd to select on (loopback): only a command can end the wait, and
|
||||||
|
/// neither heartbeats nor liveness run — a transport that can't report
|
||||||
|
/// readiness can't be timed either (same caveat as
|
||||||
|
/// [`FramedConn::recv_deadline`]). Loopback is a test transport; every real
|
||||||
|
/// connection is fd-backed.
|
||||||
|
fn run_inert(cmd_rx: &Receiver<Cmd>) {
|
||||||
|
loop {
|
||||||
|
let arms: [&dyn Selectable; 1] = [cmd_rx];
|
||||||
|
let _ = crate::channel::select(&arms);
|
||||||
|
if should_stop(cmd_rx) {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Drain the command arm. Returns `true` when the actor should exit — a
|
||||||
|
/// shutdown was requested, or the last handle was dropped.
|
||||||
|
fn should_stop(cmd_rx: &Receiver<Cmd>) -> bool {
|
||||||
|
match cmd_rx.try_recv() {
|
||||||
|
Ok(Some(Cmd::Shutdown)) => true,
|
||||||
|
Ok(None) => false, // spurious wake
|
||||||
|
Err(_) => true, // all senders dropped
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// What one readable wake yielded.
|
||||||
|
enum Pump {
|
||||||
|
/// The connection has ended: EOF (clean or mid-frame) or an
|
||||||
|
/// unrecoverable stream error.
|
||||||
|
Ended,
|
||||||
|
/// Still up; this many complete frames were consumed (possibly zero, if
|
||||||
|
/// the wake delivered only part of a frame). Any nonzero count resets
|
||||||
|
/// the liveness window.
|
||||||
|
Frames(usize),
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Surface an inbound verdict: one `smarm-trace` event, nothing else — it
|
||||||
|
/// is local knowledge (RFC §3). A no-op without the feature.
|
||||||
|
fn note_verdict(verdict: InboundVerdict) {
|
||||||
|
#[cfg(feature = "smarm-trace")]
|
||||||
|
crate::te!(crate::trace::Event::ClusterInbound(verdict.label()));
|
||||||
|
#[cfg(not(feature = "smarm-trace"))]
|
||||||
|
drop(verdict);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Consume one readable wake: exactly one socket read (which cannot block
|
||||||
|
/// after a level-triggered readable indication), then drain every complete
|
||||||
|
/// frame the buffer now holds. A blocking `recv` here would park the actor
|
||||||
|
/// past its heartbeat and liveness deadlines whenever a frame arrives split.
|
||||||
|
/// Every consumed frame counts for liveness; `SendNamed` goes to the one
|
||||||
|
/// name-resolution seam and `Send` to the pid seam. Verdicts are local
|
||||||
|
/// knowledge only — nothing goes back on the wire (RFC §3) — and surface
|
||||||
|
/// as one `smarm-trace` `ClusterInbound` event each (zero cost off).
|
||||||
|
/// `Monitor`/`Demonitor`/`Down` go to the [`Monitors`] tables; a `Monitor`
|
||||||
|
/// that can be answered at once is answered inline.
|
||||||
|
fn pump_readable(framed: &mut FramedConn, mons: &mut Monitors) -> Pump {
|
||||||
|
let eof = match framed.read_once() {
|
||||||
|
Ok(n) => n == 0,
|
||||||
|
Err(_) => return Pump::Ended,
|
||||||
|
};
|
||||||
|
let mut got = 0;
|
||||||
|
loop {
|
||||||
|
match framed.next_buffered() {
|
||||||
|
Ok(Some(frame)) => {
|
||||||
|
got += 1;
|
||||||
|
match frame {
|
||||||
|
Frame::SendNamed {
|
||||||
|
name,
|
||||||
|
type_hash,
|
||||||
|
payload,
|
||||||
|
} => {
|
||||||
|
note_verdict(deliver_named(&name, type_hash, &payload));
|
||||||
|
}
|
||||||
|
Frame::Send {
|
||||||
|
index,
|
||||||
|
generation,
|
||||||
|
type_hash,
|
||||||
|
payload,
|
||||||
|
} => {
|
||||||
|
note_verdict(deliver_to_pid(index, generation, type_hash, &payload));
|
||||||
|
}
|
||||||
|
Frame::Monitor {
|
||||||
|
monitor_id,
|
||||||
|
index,
|
||||||
|
generation,
|
||||||
|
} => {
|
||||||
|
let id = MonitorId(monitor_id);
|
||||||
|
if let Some(reason) = mons.admit(id, index, generation) {
|
||||||
|
let frame = Frame::Down {
|
||||||
|
monitor_id,
|
||||||
|
reason: reason.into(),
|
||||||
|
};
|
||||||
|
if framed.send(&frame).is_err() {
|
||||||
|
return Pump::Ended;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Frame::Demonitor { monitor_id } => mons.cancel(MonitorId(monitor_id)),
|
||||||
|
Frame::Down { monitor_id, reason } => mons.down(MonitorId(monitor_id), reason),
|
||||||
|
// Heartbeat: liveness only. Handshake frames after
|
||||||
|
// establishment: ignored.
|
||||||
|
_ => {}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(None) => break,
|
||||||
|
Err(_) => return Pump::Ended, // corrupt stream
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if eof {
|
||||||
|
Pump::Ended
|
||||||
|
} else {
|
||||||
|
Pump::Frames(got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
//! `Monitors::teardown` in isolation: the actor-side half of c13, pinned
|
||||||
|
//! separately because from the outside it is indistinguishable from the
|
||||||
|
//! read-side backstop in `RemoteMonitor` (both yield `Disconnected`).
|
||||||
|
use super::*;
|
||||||
|
use crate::pg::Incarnation;
|
||||||
|
|
||||||
|
fn pid(index: u32) -> RemotePid<Erased> {
|
||||||
|
RemotePid::from_parts("peer", Incarnation::new(1), index, 1)
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn teardown_answers_every_outstanding_and_unread_monitor_once() {
|
||||||
|
crate::run(|| {
|
||||||
|
let mut mons = Monitors::default();
|
||||||
|
let (mon_tx, mon_rx) = channel::<MonCmd>();
|
||||||
|
|
||||||
|
// Already registered.
|
||||||
|
let (tx1, rx1) = channel::<RemoteDown>();
|
||||||
|
mons.outstanding.insert(MonitorId(1), (pid(1), tx1));
|
||||||
|
// In the inbox, never processed.
|
||||||
|
let (tx2, rx2) = channel::<RemoteDown>();
|
||||||
|
mon_tx
|
||||||
|
.send(MonCmd::Monitor {
|
||||||
|
id: MonitorId(2),
|
||||||
|
target: pid(2),
|
||||||
|
tx: tx2,
|
||||||
|
})
|
||||||
|
.ok()
|
||||||
|
.unwrap();
|
||||||
|
// Registered, then cancelled in the inbox: silence.
|
||||||
|
let (tx3, rx3) = channel::<RemoteDown>();
|
||||||
|
mons.outstanding.insert(MonitorId(3), (pid(3), tx3));
|
||||||
|
mon_tx
|
||||||
|
.send(MonCmd::Demonitor { id: MonitorId(3) })
|
||||||
|
.ok()
|
||||||
|
.unwrap();
|
||||||
|
|
||||||
|
mons.teardown(&mon_rx);
|
||||||
|
|
||||||
|
let d1 = rx1.recv().unwrap();
|
||||||
|
assert_eq!(
|
||||||
|
(d1.pid, d1.reason),
|
||||||
|
(pid(1), RemoteDownReason::Disconnected)
|
||||||
|
);
|
||||||
|
let d2 = rx2.recv().unwrap();
|
||||||
|
assert_eq!(
|
||||||
|
(d2.pid, d2.reason),
|
||||||
|
(pid(2), RemoteDownReason::Disconnected)
|
||||||
|
);
|
||||||
|
// Cancelled: no notice was sent (its sender is dropped, channel
|
||||||
|
// closed-empty), and nobody got a second one.
|
||||||
|
assert!(rx3.try_recv().is_err());
|
||||||
|
assert!(rx1.try_recv().is_err());
|
||||||
|
assert!(rx2.try_recv().is_err());
|
||||||
|
assert!(mons.outstanding.is_empty());
|
||||||
|
});
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,365 @@
|
|||||||
|
//! RFC 010 c6b — the handshake on the accept/connect path.
|
||||||
|
//!
|
||||||
|
//! Per D8 (re-amended): the c5 machines are driven by **straight-line code
|
||||||
|
//! on the path**, not by an actor. The dial side runs [`Initiator`]; the
|
||||||
|
//! acceptor loop runs [`Responder`]. A connection actor is spawned only
|
||||||
|
//! *after* a successful handshake ([`spawn_established`]); every reject,
|
||||||
|
//! protocol failure, timeout, and tie-break loss is resolved right here,
|
||||||
|
//! on the path, by closing — no actor ever exists for a connection that
|
||||||
|
//! didn't establish.
|
||||||
|
//!
|
||||||
|
//! Buffer trap (binding): the path reader and the steady-state actor share
|
||||||
|
//! ONE [`FramedConn`]. Its decode buffer may hold read-ahead past the
|
||||||
|
//! handshake frames, so the *whole* `FramedConn` travels into
|
||||||
|
//! [`spawn_established`] — never a fresh codec over the same socket.
|
||||||
|
//!
|
||||||
|
//! Layering: [`dial_handshake`] and [`accept_handshake`] are the bare path
|
||||||
|
//! steps — IO on a `FramedConn`, no manager, no actors — testable over the
|
||||||
|
//! loopback transport on plain threads. [`dial`] and [`spawn_acceptor`] are
|
||||||
|
//! the manager-integrated layer (actor context required): they keep the
|
||||||
|
//! [`manager`](crate::cluster::manager)'s dial-intent set honest and spawn
|
||||||
|
//! the connection actor on success.
|
||||||
|
|
||||||
|
use std::io;
|
||||||
|
use std::time::{Duration, Instant};
|
||||||
|
|
||||||
|
use crate::channel::{channel, Receiver, Selectable, Sender};
|
||||||
|
use crate::cluster::conn::spawn_established;
|
||||||
|
use crate::cluster::envelope::{Frame, RejectReason};
|
||||||
|
use crate::cluster::handshake::{
|
||||||
|
Initiator, InitiatorOutcome, Local, Peer, PeerStanding, Responder, ResponderOutcome,
|
||||||
|
};
|
||||||
|
use crate::cluster::manager::{Call, Reply, MANAGER};
|
||||||
|
use crate::cluster::transport::{FramedConn, Listener, RecvError, SendError, Transport};
|
||||||
|
use crate::cluster::Timing;
|
||||||
|
use crate::gen_server;
|
||||||
|
use crate::pid::Pid;
|
||||||
|
use crate::scheduler::{self, spawn};
|
||||||
|
|
||||||
|
/// Default for [`Timing::handshake_timeout`]: how long either side waits for
|
||||||
|
/// the peer's handshake frame before giving up and closing. Enforced on the path via [`FramedConn::recv_deadline`],
|
||||||
|
/// so a peer that connects and goes silent cannot wedge the acceptor.
|
||||||
|
pub const HANDSHAKE_TIMEOUT: Duration = Duration::from_secs(5);
|
||||||
|
|
||||||
|
/// Why a handshake did not establish. In every case the connection has
|
||||||
|
/// already been closed on the path by the time this is returned.
|
||||||
|
#[derive(Debug)]
|
||||||
|
pub enum HandshakeError {
|
||||||
|
/// A `HelloReject` travelled — sent by us (accept side) or received by
|
||||||
|
/// us (dial side).
|
||||||
|
Rejected(RejectReason),
|
||||||
|
/// Accept side only: the inbound dial lost the simultaneous-connect
|
||||||
|
/// tie-break (D7) and was closed silently, no frame sent.
|
||||||
|
TieBreakLoss,
|
||||||
|
/// The peer spoke a valid frame that is wrong here (non-`Hello` first
|
||||||
|
/// frame; non-response to our `Hello`), or an undecodable byte stream.
|
||||||
|
Protocol,
|
||||||
|
/// EOF before the handshake resolved. On the dial side this is also
|
||||||
|
/// what losing the tie-break looks like: the peer closes silently.
|
||||||
|
Closed,
|
||||||
|
/// [`HANDSHAKE_TIMEOUT`] (or the caller's deadline) passed first.
|
||||||
|
TimedOut,
|
||||||
|
/// The transport failed mid-handshake.
|
||||||
|
Transport(io::Error),
|
||||||
|
}
|
||||||
|
|
||||||
|
fn from_send(e: SendError) -> HandshakeError {
|
||||||
|
match e {
|
||||||
|
// Handshake frames are small and self-made; an encode failure is a
|
||||||
|
// protocol-level impossibility, not a transport fault.
|
||||||
|
SendError::Encode(_) => HandshakeError::Protocol,
|
||||||
|
SendError::Io(e) => HandshakeError::Transport(e),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn from_recv(e: RecvError) -> HandshakeError {
|
||||||
|
match e {
|
||||||
|
RecvError::Corrupt(_) => HandshakeError::Protocol,
|
||||||
|
RecvError::TruncatedByPeer => HandshakeError::Closed,
|
||||||
|
RecvError::Io(e) => HandshakeError::Transport(e),
|
||||||
|
RecvError::TimedOut => HandshakeError::TimedOut,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Dial-side path step: send our `Hello`, interpret the one response. On
|
||||||
|
/// `Ok` the connection is established and `framed` is live (with any
|
||||||
|
/// read-ahead intact in its buffer); on `Err` the connection is closed.
|
||||||
|
pub fn dial_handshake(
|
||||||
|
framed: &mut FramedConn,
|
||||||
|
local: &Local,
|
||||||
|
deadline: Instant,
|
||||||
|
) -> Result<Peer, HandshakeError> {
|
||||||
|
let (initiator, hello) = Initiator::new(local);
|
||||||
|
if let Err(e) = framed.send(&hello) {
|
||||||
|
framed.close();
|
||||||
|
return Err(from_send(e));
|
||||||
|
}
|
||||||
|
let outcome = match framed.recv_deadline(deadline) {
|
||||||
|
Ok(Some(frame)) => initiator.on_frame(frame),
|
||||||
|
Ok(None) => {
|
||||||
|
framed.close();
|
||||||
|
return Err(HandshakeError::Closed);
|
||||||
|
}
|
||||||
|
Err(e) => {
|
||||||
|
framed.close();
|
||||||
|
return Err(from_recv(e));
|
||||||
|
}
|
||||||
|
};
|
||||||
|
match outcome {
|
||||||
|
InitiatorOutcome::Established(peer) => Ok(peer),
|
||||||
|
InitiatorOutcome::Rejected(reason) => {
|
||||||
|
framed.close();
|
||||||
|
Err(HandshakeError::Rejected(reason))
|
||||||
|
}
|
||||||
|
InitiatorOutcome::Failed(_) => {
|
||||||
|
framed.close();
|
||||||
|
Err(HandshakeError::Protocol)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Accept-side path step: read the first frame, judge it, answer or close.
|
||||||
|
///
|
||||||
|
/// `standing_of` supplies the [`PeerStanding`] of the *offered* name — knowledge
|
||||||
|
/// only the frame reveals, which is why it is a callback and not a value
|
||||||
|
/// (the integrated acceptor asks the manager; loopback tests fabricate).
|
||||||
|
/// It is not called when the first frame is not a `Hello`.
|
||||||
|
///
|
||||||
|
/// On `Ok` the ack has been sent and `framed` is live (read-ahead intact);
|
||||||
|
/// on `Err` any owed reject has been sent and the connection is closed.
|
||||||
|
pub fn accept_handshake(
|
||||||
|
framed: &mut FramedConn,
|
||||||
|
local: Local,
|
||||||
|
standing_of: impl FnOnce(&str) -> PeerStanding,
|
||||||
|
deadline: Instant,
|
||||||
|
) -> Result<Peer, HandshakeError> {
|
||||||
|
let frame = match framed.recv_deadline(deadline) {
|
||||||
|
Ok(Some(frame)) => frame,
|
||||||
|
Ok(None) => {
|
||||||
|
framed.close();
|
||||||
|
return Err(HandshakeError::Closed);
|
||||||
|
}
|
||||||
|
Err(e) => {
|
||||||
|
framed.close();
|
||||||
|
return Err(from_recv(e));
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let standing = match &frame {
|
||||||
|
Frame::Hello { node_name, .. } => standing_of(node_name),
|
||||||
|
_ => PeerStanding::Free,
|
||||||
|
};
|
||||||
|
match Responder::new(local).on_frame(frame, standing) {
|
||||||
|
ResponderOutcome::Accepted { reply, peer } => {
|
||||||
|
if let Err(e) = framed.send(&reply) {
|
||||||
|
framed.close();
|
||||||
|
return Err(from_send(e));
|
||||||
|
}
|
||||||
|
Ok(peer)
|
||||||
|
}
|
||||||
|
ResponderOutcome::Rejected { reply, reason } => {
|
||||||
|
// Best effort: the reject is the cross-version compatibility
|
||||||
|
// anchor, but if the write fails the peer sees a bare close,
|
||||||
|
// which it must survive anyway.
|
||||||
|
let _ = framed.send(&reply);
|
||||||
|
framed.close();
|
||||||
|
Err(HandshakeError::Rejected(reason))
|
||||||
|
}
|
||||||
|
ResponderOutcome::TieBreakLoss => {
|
||||||
|
// D7: close silently — the peer computes the same verdict.
|
||||||
|
framed.close();
|
||||||
|
Err(HandshakeError::TieBreakLoss)
|
||||||
|
}
|
||||||
|
ResponderOutcome::Failed(_) => {
|
||||||
|
framed.close();
|
||||||
|
Err(HandshakeError::Protocol)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Manager-integrated layer
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
/// Why an integrated [`dial`] did not produce a connection.
|
||||||
|
#[derive(Debug)]
|
||||||
|
pub enum DialError {
|
||||||
|
/// Another dial to this peer name is already in flight.
|
||||||
|
AlreadyDialing,
|
||||||
|
/// The manager is not running (or answered nonsense).
|
||||||
|
ManagerUnavailable,
|
||||||
|
/// The transport could not connect.
|
||||||
|
Connect(io::Error),
|
||||||
|
/// Connected, but the handshake did not establish.
|
||||||
|
Handshake(HandshakeError),
|
||||||
|
/// The peer at `addr` established, but answered as a different name
|
||||||
|
/// than the one we dialed — the tie-break bookkeeping (keyed by the
|
||||||
|
/// dialed name) would be unsound, so the connection is closed.
|
||||||
|
PeerNameMismatch { expected: String, got: String },
|
||||||
|
/// The handshake established, but the manager refused the registration:
|
||||||
|
/// a connection to this peer already exists. The loser has been closed.
|
||||||
|
Duplicate,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl DialError {
|
||||||
|
/// A short static label per kind, for the `smarm-trace` `ClusterDial`
|
||||||
|
/// event; the payload (io error, names) is not carried.
|
||||||
|
pub fn label(&self) -> &'static str {
|
||||||
|
match self {
|
||||||
|
DialError::AlreadyDialing => "already_dialing",
|
||||||
|
DialError::ManagerUnavailable => "manager_unavailable",
|
||||||
|
DialError::Connect(_) => "connect",
|
||||||
|
DialError::Handshake(_) => "handshake",
|
||||||
|
DialError::PeerNameMismatch { .. } => "peer_name_mismatch",
|
||||||
|
DialError::Duplicate => "duplicate",
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Dial `peer_name` at `addr` and run the handshake, keeping the manager's
|
||||||
|
/// dial-intent set honest around it: the intent is registered *before*
|
||||||
|
/// connecting (so a crossing inbound `Hello` sees it) and cleared the
|
||||||
|
/// moment the handshake resolves, before the connection actor is spawned.
|
||||||
|
/// Must run inside an actor. Retrying is the caller's business (c7's dial
|
||||||
|
/// loop); a lost tie-break surfaces as `Handshake(Closed)` — the peer's
|
||||||
|
/// accepted connection is already on its way.
|
||||||
|
pub fn dial(
|
||||||
|
transport: &dyn Transport,
|
||||||
|
addr: &str,
|
||||||
|
peer_name: &str,
|
||||||
|
local: &Local,
|
||||||
|
timing: Timing,
|
||||||
|
) -> Result<Pid, DialError> {
|
||||||
|
let me = scheduler::self_pid();
|
||||||
|
match gen_server::call(
|
||||||
|
MANAGER,
|
||||||
|
Call::DialBegin {
|
||||||
|
name: peer_name.to_string(),
|
||||||
|
pid: me,
|
||||||
|
},
|
||||||
|
) {
|
||||||
|
Ok(Reply::DialBegan(true)) => {}
|
||||||
|
Ok(Reply::DialBegan(false)) => return Err(DialError::AlreadyDialing),
|
||||||
|
_ => return Err(DialError::ManagerUnavailable),
|
||||||
|
}
|
||||||
|
let result = connect_and_shake(transport, addr, local, timing);
|
||||||
|
// Cleared immediately on outcome — a stale intent during the established
|
||||||
|
// window would corrupt later tie-breaks. Synchronous (a call): the
|
||||||
|
// intent is provably gone before anything else happens.
|
||||||
|
let _ = gen_server::call(
|
||||||
|
MANAGER,
|
||||||
|
Call::DialEnd {
|
||||||
|
name: peer_name.to_string(),
|
||||||
|
},
|
||||||
|
);
|
||||||
|
let (mut framed, peer) = result?;
|
||||||
|
if peer.node_name != peer_name {
|
||||||
|
framed.close();
|
||||||
|
return Err(DialError::PeerNameMismatch {
|
||||||
|
expected: peer_name.to_string(),
|
||||||
|
got: peer.node_name,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
spawn_established(framed, peer, timing).map_err(|_| DialError::Duplicate)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn connect_and_shake(
|
||||||
|
transport: &dyn Transport,
|
||||||
|
addr: &str,
|
||||||
|
local: &Local,
|
||||||
|
timing: Timing,
|
||||||
|
) -> Result<(FramedConn, Peer), DialError> {
|
||||||
|
let conn = transport.dial(addr).map_err(DialError::Connect)?;
|
||||||
|
let mut framed = FramedConn::new(conn);
|
||||||
|
let deadline = Instant::now() + timing.handshake_timeout;
|
||||||
|
let peer = dial_handshake(&mut framed, local, deadline).map_err(DialError::Handshake)?;
|
||||||
|
Ok((framed, peer))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A running acceptor. [`shutdown`](AcceptorHandle::shutdown) (or dropping
|
||||||
|
/// the last handle) stops the accept loop only: connections it established
|
||||||
|
/// belong to the [`manager`](crate::cluster::manager) and keep running, to
|
||||||
|
/// be torn down through the table (`Disconnect`, a peer close, or manager
|
||||||
|
/// shutdown).
|
||||||
|
pub struct AcceptorHandle {
|
||||||
|
cmd_tx: Sender<()>,
|
||||||
|
addr: String,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl AcceptorHandle {
|
||||||
|
/// Ask the acceptor to stop. Idempotent; a no-op if it already has.
|
||||||
|
pub fn shutdown(&self) {
|
||||||
|
let _ = self.cmd_tx.send(());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The concrete bound address, dialable as-is.
|
||||||
|
pub fn local_addr(&self) -> &str {
|
||||||
|
&self.addr
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Spawn the acceptor actor over a bound listener. Each inbound connection
|
||||||
|
/// is handshaken **inline in the loop** (a deliberate serialization: the
|
||||||
|
/// per-frame deadline bounds how long any one peer can hold the line, and
|
||||||
|
/// nothing concurrent exists to be starved before c7). The listener must be
|
||||||
|
/// fd-backed ([`Listener::readable_arm`]); the loopback listener is not,
|
||||||
|
/// and its acceptor exits immediately — loopback handshakes are driven
|
||||||
|
/// synchronously through the path fns instead, per D8.
|
||||||
|
pub fn spawn_acceptor(listener: Box<dyn Listener>, local: Local, timing: Timing) -> AcceptorHandle {
|
||||||
|
let addr = listener.local_addr();
|
||||||
|
let (cmd_tx, cmd_rx) = channel();
|
||||||
|
spawn(move || accept_loop(listener, local, timing, cmd_rx));
|
||||||
|
AcceptorHandle { cmd_tx, addr }
|
||||||
|
}
|
||||||
|
|
||||||
|
fn accept_loop(
|
||||||
|
mut listener: Box<dyn Listener>,
|
||||||
|
local: Local,
|
||||||
|
timing: Timing,
|
||||||
|
cmd_rx: Receiver<()>,
|
||||||
|
) {
|
||||||
|
loop {
|
||||||
|
let Some(arm) = listener.readable_arm() else {
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
let arms: [&dyn Selectable; 2] = [&cmd_rx, &arm];
|
||||||
|
match crate::channel::try_select(&arms) {
|
||||||
|
Ok(0) => match cmd_rx.try_recv() {
|
||||||
|
Ok(Some(())) => return,
|
||||||
|
Ok(None) => continue, // spurious wake
|
||||||
|
Err(_) => return, // all handles dropped
|
||||||
|
},
|
||||||
|
Ok(_) => {
|
||||||
|
// The listener is readable: accept completes without parking.
|
||||||
|
let conn = match listener.accept() {
|
||||||
|
Ok(conn) => conn,
|
||||||
|
Err(_) => return, // listener itself is broken
|
||||||
|
};
|
||||||
|
handle_inbound(FramedConn::new(conn), &local, timing);
|
||||||
|
}
|
||||||
|
Err(_) => return, // fd arm failed to register: listener is gone
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Run the accept-side handshake for one inbound connection, asking the
|
||||||
|
/// manager for the [`PeerStanding`], and hand the established connection to the
|
||||||
|
/// manager. Every failure was already resolved on the path (reject sent /
|
||||||
|
/// closed, or the registration refused and the actor stopped), so there is
|
||||||
|
/// nothing for the acceptor to carry forward.
|
||||||
|
fn handle_inbound(mut framed: FramedConn, local: &Local, timing: Timing) {
|
||||||
|
let deadline = Instant::now() + timing.handshake_timeout;
|
||||||
|
let standing_of = |name: &str| match gen_server::call(
|
||||||
|
MANAGER,
|
||||||
|
Call::Standing {
|
||||||
|
peer_name: name.to_string(),
|
||||||
|
},
|
||||||
|
) {
|
||||||
|
Ok(Reply::Standing(s)) => s,
|
||||||
|
// Manager unreachable: nobody could register this connection anyway,
|
||||||
|
// so claim the name taken and reject rather than accept an orphan.
|
||||||
|
_ => PeerStanding::Claimed,
|
||||||
|
};
|
||||||
|
if let Ok(peer) = accept_handshake(&mut framed, local.clone(), standing_of, deadline) {
|
||||||
|
let _ = spawn_established(framed, peer, timing);
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,316 @@
|
|||||||
|
//! RFC 010 c7b — the connector: the dial loop that turns discovered
|
||||||
|
//! candidates into a full mesh.
|
||||||
|
//!
|
||||||
|
//! A plain select-loop actor (the c6 shape). It spawns its [`Strategy`] as a
|
||||||
|
//! child actor and receives [`Discovery`] events from it; it tracks which
|
||||||
|
//! peers are up by **subscribing to membership like any other consumer** —
|
||||||
|
//! no privileged channel into the manager, the same snapshot-then-stream
|
||||||
|
//! surface c8 will use. One `select` folds the command inbox, the discovery
|
||||||
|
//! stream, the membership stream, and the earliest retry deadline into a
|
||||||
|
//! single wait.
|
||||||
|
//!
|
||||||
|
//! Per-candidate state: dial on arrival; on failure retry with capped
|
||||||
|
//! exponential backoff ([`INITIAL_BACKOFF`] doubling to [`MAX_BACKOFF`]);
|
||||||
|
//! on the peer's `node_up` stop dialing and reset the backoff; on its
|
||||||
|
//! `node_down` resume immediately (a fresh sequence — the reconnect case is
|
||||||
|
//! the one backoff exists to pace, but the *first* retry after a death
|
||||||
|
//! should be prompt). A candidate bearing our own name is parked permanently
|
||||||
|
//! — that seed is us; so is one whose address answers as a different name
|
||||||
|
//! (`PeerNameMismatch`: a misconfigured or stale seed — each retry would
|
||||||
|
//! only blip the peer's membership). Every other failure retries: in
|
||||||
|
//! particular a `NameTaken` reject can be our own ghost at the peer, not
|
||||||
|
//! yet reaped by its liveness timer, so it must not park. Each attempt's
|
||||||
|
//! outcome is one `smarm-trace` `ClusterDial` event. A [`Discovery::Withdrawn`]
|
||||||
|
//! drops its `(name, addr)` from the dial set — only that: a live
|
||||||
|
//! connection is membership's, and a re-announce re-adds it fresh.
|
||||||
|
//!
|
||||||
|
//! Dials run **inline in the loop** — the same deliberate serialization as
|
||||||
|
//! the acceptor (c6b): each attempt is bounded by the connect + handshake
|
||||||
|
//! deadlines, and nothing concurrent exists to be starved. A wall of slow
|
||||||
|
//! unreachable seeds would stretch the loop's latency; revisit if a real
|
||||||
|
//! deployment ever hits that shape.
|
||||||
|
|
||||||
|
use std::collections::HashSet;
|
||||||
|
use std::time::{Duration, Instant};
|
||||||
|
|
||||||
|
use crate::channel::{channel, select, select_timeout, Receiver, Selectable, Sender};
|
||||||
|
use crate::cluster::connect::{dial, DialError};
|
||||||
|
use crate::cluster::discovery::{Discovery, Strategy};
|
||||||
|
use crate::cluster::handshake::Local;
|
||||||
|
use crate::cluster::membership::{subscribe, NodeEvent};
|
||||||
|
use crate::cluster::transport::Transport;
|
||||||
|
use crate::cluster::Timing;
|
||||||
|
use crate::scheduler::spawn;
|
||||||
|
|
||||||
|
/// Default for [`Timing::initial_backoff`]: first retry delay after a failed
|
||||||
|
/// dial attempt.
|
||||||
|
pub const INITIAL_BACKOFF: Duration = Duration::from_millis(250);
|
||||||
|
/// Default for [`Timing::max_backoff`]: an unreachable seed is retried this
|
||||||
|
/// often, forever.
|
||||||
|
pub const MAX_BACKOFF: Duration = Duration::from_secs(5);
|
||||||
|
|
||||||
|
enum Cmd {
|
||||||
|
Shutdown,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A running connector. `shutdown` (or dropping the last handle) stops the
|
||||||
|
/// dial loop and its strategy only — established connections belong to the
|
||||||
|
/// manager, exactly as with the acceptor.
|
||||||
|
pub struct ConnectorHandle {
|
||||||
|
cmd_tx: Sender<Cmd>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl ConnectorHandle {
|
||||||
|
/// Ask the connector to stop. Idempotent; a no-op if it already has.
|
||||||
|
pub fn shutdown(&self) {
|
||||||
|
let _ = self.cmd_tx.send(Cmd::Shutdown);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One discovered `(name, addr)` and our dial intent towards it.
|
||||||
|
struct Candidate {
|
||||||
|
name: String,
|
||||||
|
addr: String,
|
||||||
|
state: State,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The connector's *intent* for a candidate. Whether the peer is currently
|
||||||
|
/// up is a separate, name-keyed membership fact (`up` in [`run`]): a
|
||||||
|
/// candidate can arrive after its peer's `node_up` (the snapshot lands
|
||||||
|
/// before the strategy has said anything), so "up" cannot live on the
|
||||||
|
/// candidate alone — it is a filter over dialing, not a candidate state.
|
||||||
|
enum State {
|
||||||
|
/// Never dialed: this seed is the local node itself, or the address
|
||||||
|
/// answered as a *different* name than the one seeded
|
||||||
|
/// (`DialError::PeerNameMismatch` — a misconfigured or stale seed;
|
||||||
|
/// redialing would only blip the peer's membership forever). The way
|
||||||
|
/// back is the strategy's: `Withdrawn` then a fresh `Candidate`.
|
||||||
|
Parked,
|
||||||
|
/// Dial when due; on failure, back off.
|
||||||
|
Dialing {
|
||||||
|
/// Delay to apply after the *next* failure.
|
||||||
|
backoff: Duration,
|
||||||
|
next_attempt: Instant,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
impl State {
|
||||||
|
fn fresh(timing: &Timing) -> Self {
|
||||||
|
State::Dialing {
|
||||||
|
backoff: timing.initial_backoff,
|
||||||
|
next_attempt: Instant::now(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Candidate {
|
||||||
|
/// The retry deadline, if this candidate is dialing at all.
|
||||||
|
fn due(&self) -> Option<Instant> {
|
||||||
|
match self.state {
|
||||||
|
State::Parked => None,
|
||||||
|
State::Dialing { next_attempt, .. } => Some(next_attempt),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
/// A dial attempt was made: schedule the retry, grow the backoff.
|
||||||
|
fn attempted(&mut self, timing: &Timing) {
|
||||||
|
if let State::Dialing {
|
||||||
|
backoff,
|
||||||
|
next_attempt,
|
||||||
|
} = &mut self.state
|
||||||
|
{
|
||||||
|
*next_attempt = Instant::now() + *backoff;
|
||||||
|
*backoff = (*backoff * 2).min(timing.max_backoff);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
/// The peer came up: the next sequence (after a later `node_down`)
|
||||||
|
/// starts from the initial delay again.
|
||||||
|
fn peer_up(&mut self, timing: &Timing) {
|
||||||
|
if let State::Dialing { backoff, .. } = &mut self.state {
|
||||||
|
*backoff = timing.initial_backoff;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
/// The peer went down: redial promptly, fresh sequence.
|
||||||
|
fn peer_down(&mut self, timing: &Timing) {
|
||||||
|
if matches!(self.state, State::Dialing { .. }) {
|
||||||
|
self.state = State::fresh(timing);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Spawn the connector actor. The strategy is spawned as its child; the
|
||||||
|
/// membership subscription is taken inside the actor. Must be called from
|
||||||
|
/// inside an actor (the same requirement as `dial`).
|
||||||
|
pub fn spawn_connector(
|
||||||
|
transport: Box<dyn Transport>,
|
||||||
|
local: Local,
|
||||||
|
strategy: Box<dyn Strategy>,
|
||||||
|
timing: Timing,
|
||||||
|
) -> ConnectorHandle {
|
||||||
|
let (cmd_tx, cmd_rx) = channel();
|
||||||
|
spawn(move || run(transport, local, strategy, timing, cmd_rx));
|
||||||
|
ConnectorHandle { cmd_tx }
|
||||||
|
}
|
||||||
|
|
||||||
|
fn run(
|
||||||
|
transport: Box<dyn Transport>,
|
||||||
|
local: Local,
|
||||||
|
strategy: Box<dyn Strategy>,
|
||||||
|
timing: Timing,
|
||||||
|
cmd_rx: Receiver<Cmd>,
|
||||||
|
) {
|
||||||
|
// Membership is the connector's source of truth for "who is up" — the
|
||||||
|
// snapshot seeds `up` before any candidate arrives.
|
||||||
|
let Some(events) = subscribe() else {
|
||||||
|
return; // no manager, no cluster to connect
|
||||||
|
};
|
||||||
|
let (disc_tx, disc_rx) = channel();
|
||||||
|
spawn(move || strategy.run(disc_tx));
|
||||||
|
|
||||||
|
let mut cands: Vec<Candidate> = Vec::new();
|
||||||
|
let mut up: HashSet<String> = HashSet::new();
|
||||||
|
let mut strategy_done = false;
|
||||||
|
|
||||||
|
loop {
|
||||||
|
// Drain every input, then act. Order does not matter: acting is
|
||||||
|
// idempotent against the resulting state.
|
||||||
|
match drain_cmd(&cmd_rx) {
|
||||||
|
Drained::Stop => return,
|
||||||
|
Drained::Open => {}
|
||||||
|
}
|
||||||
|
if !strategy_done {
|
||||||
|
strategy_done = drain_discoveries(&disc_rx, &local, &timing, &mut cands);
|
||||||
|
}
|
||||||
|
match drain_events(&events.rx, &timing, &mut up, &mut cands) {
|
||||||
|
Drained::Stop => return, // manager gone: the cluster is tearing down
|
||||||
|
Drained::Open => {}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Dial everything due, inline (see the module docs on serialization).
|
||||||
|
let now = Instant::now();
|
||||||
|
for c in cands
|
||||||
|
.iter_mut()
|
||||||
|
.filter(|c| !up.contains(&c.name) && c.due().is_some_and(|d| d <= now))
|
||||||
|
{
|
||||||
|
// On success the manager's node_up is on its way and lands in
|
||||||
|
// `up` (backing off meanwhile keeps a racing re-attempt from
|
||||||
|
// spinning); every failure retries — see the module docs —
|
||||||
|
// except a peer-name mismatch, which parks the candidate.
|
||||||
|
let outcome = dial(&*transport, &c.addr, &c.name, &local, timing);
|
||||||
|
note_dial(&outcome);
|
||||||
|
match outcome {
|
||||||
|
Err(DialError::PeerNameMismatch { .. }) => c.state = State::Parked,
|
||||||
|
_ => c.attempted(&timing),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Wait: until the earliest retry deadline among actionable
|
||||||
|
// candidates, or indefinitely if none is pending.
|
||||||
|
let deadline = cands
|
||||||
|
.iter()
|
||||||
|
.filter(|c| !up.contains(&c.name))
|
||||||
|
.filter_map(Candidate::due)
|
||||||
|
.min();
|
||||||
|
let mut arms: Vec<&dyn Selectable> = vec![&cmd_rx, &events.rx];
|
||||||
|
if !strategy_done {
|
||||||
|
arms.push(&disc_rx);
|
||||||
|
}
|
||||||
|
match deadline {
|
||||||
|
Some(d) => {
|
||||||
|
let wait = d.saturating_duration_since(Instant::now());
|
||||||
|
let _ = select_timeout(&arms, wait);
|
||||||
|
}
|
||||||
|
None => {
|
||||||
|
let _ = select(&arms);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
enum Drained {
|
||||||
|
Open,
|
||||||
|
Stop,
|
||||||
|
}
|
||||||
|
|
||||||
|
fn drain_cmd(rx: &Receiver<Cmd>) -> Drained {
|
||||||
|
match rx.try_recv() {
|
||||||
|
Ok(Some(Cmd::Shutdown)) => Drained::Stop,
|
||||||
|
Ok(None) => Drained::Open,
|
||||||
|
Err(_) => Drained::Stop, // all handles dropped
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Pull every pending discovery into the candidate set (deduplicated by
|
||||||
|
/// `(name, addr)`; a candidate bearing the local name is parked; a
|
||||||
|
/// `Withdrawn` removes its pair from the dial set and nothing else — see
|
||||||
|
/// [`Discovery::Withdrawn`]). Returns `true` once the strategy's channel
|
||||||
|
/// closes — it has said all it will.
|
||||||
|
fn drain_discoveries(
|
||||||
|
rx: &Receiver<Discovery>,
|
||||||
|
local: &Local,
|
||||||
|
timing: &Timing,
|
||||||
|
cands: &mut Vec<Candidate>,
|
||||||
|
) -> bool {
|
||||||
|
loop {
|
||||||
|
match rx.try_recv() {
|
||||||
|
Ok(Some(Discovery::Withdrawn { name, addr })) => {
|
||||||
|
cands.retain(|c| !(c.name == name && c.addr == addr));
|
||||||
|
}
|
||||||
|
Ok(Some(Discovery::Candidate { name, addr })) => {
|
||||||
|
if cands.iter().any(|c| c.name == name && c.addr == addr) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
let state = if name == local.node_name {
|
||||||
|
State::Parked
|
||||||
|
} else {
|
||||||
|
State::fresh(timing)
|
||||||
|
};
|
||||||
|
cands.push(Candidate { name, addr, state });
|
||||||
|
}
|
||||||
|
Ok(None) => return false,
|
||||||
|
Err(_) => return true, // strategy done; its candidates live on here
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Fold pending membership events into `up`; each is also a transition on
|
||||||
|
/// that peer's candidates (see [`Candidate::peer_up`] / [`peer_down`]).
|
||||||
|
///
|
||||||
|
/// [`peer_down`]: Candidate::peer_down
|
||||||
|
fn drain_events(
|
||||||
|
rx: &Receiver<NodeEvent>,
|
||||||
|
timing: &Timing,
|
||||||
|
up: &mut HashSet<String>,
|
||||||
|
cands: &mut [Candidate],
|
||||||
|
) -> Drained {
|
||||||
|
loop {
|
||||||
|
match rx.try_recv() {
|
||||||
|
Ok(Some(NodeEvent::NodeUp(info))) => {
|
||||||
|
cands
|
||||||
|
.iter_mut()
|
||||||
|
.filter(|c| c.name == info.name)
|
||||||
|
.for_each(|c| c.peer_up(timing));
|
||||||
|
up.insert(info.name);
|
||||||
|
}
|
||||||
|
Ok(Some(NodeEvent::NodeDown(info))) => {
|
||||||
|
up.remove(&info.name);
|
||||||
|
cands
|
||||||
|
.iter_mut()
|
||||||
|
.filter(|c| c.name == info.name)
|
||||||
|
.for_each(|c| c.peer_down(timing));
|
||||||
|
}
|
||||||
|
Ok(None) => return Drained::Open,
|
||||||
|
Err(_) => return Drained::Stop,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Surface a dial outcome: one `smarm-trace` event, nothing else. The
|
||||||
|
/// connector's bookkeeping is decided by the caller.
|
||||||
|
fn note_dial(outcome: &Result<crate::pid::Pid, DialError>) {
|
||||||
|
#[cfg(feature = "smarm-trace")]
|
||||||
|
crate::te!(crate::trace::Event::ClusterDial(
|
||||||
|
outcome.as_ref().map_or_else(DialError::label, |_| "ok")
|
||||||
|
));
|
||||||
|
#[cfg(not(feature = "smarm-trace"))]
|
||||||
|
let _ = outcome;
|
||||||
|
}
|
||||||
@@ -0,0 +1,76 @@
|
|||||||
|
//! RFC 010 c7b — peer discovery: the [`Strategy`] seam and the static-seeds
|
||||||
|
//! implementation.
|
||||||
|
//!
|
||||||
|
//! A strategy is **push-based and runs as its own actor**: the
|
||||||
|
//! [`connector`](crate::cluster::connector) spawns it with the sending end of
|
||||||
|
//! a channel, and the strategy emits [`Discovery`] events whenever it learns
|
||||||
|
//! something — once at startup for a static list, continuously for a future
|
||||||
|
//! mDNS/DNS strategy — for as long as it cares to run. Returning ends the
|
||||||
|
//! strategy actor; the candidates it pushed live on in the connector (the
|
||||||
|
//! connector owns all retry/backoff state, so a strategy never re-announces).
|
||||||
|
//!
|
||||||
|
//! A candidate is a **`(node_name, addr)` pair**, not a bare address: the
|
||||||
|
//! dial path and the D7 tie-break are keyed by peer *name* (the dial intent
|
||||||
|
//! must be registered before connecting so a crossing inbound `Hello` sees
|
||||||
|
//! it), so an anonymous dial would reintroduce exactly the
|
||||||
|
//! simultaneous-connect flap D7 exists to prevent. Discovery mechanisms know
|
||||||
|
//! names — that is what they discover.
|
||||||
|
|
||||||
|
use crate::channel::Sender;
|
||||||
|
|
||||||
|
/// A discovery event, as pushed by a [`Strategy`].
|
||||||
|
///
|
||||||
|
/// `Candidate` announces, `Withdrawn` retracts — the primitive pair. A
|
||||||
|
/// strategy that wants TTL semantics builds them on top (track its own
|
||||||
|
/// last-seen times, emit `Withdrawn` on expiry); the connector deliberately
|
||||||
|
/// has no clock of its own for candidates (D11: strategies never
|
||||||
|
/// re-announce, the connector owns retry). `#[non_exhaustive]` so more can
|
||||||
|
/// land without breaking strategies.
|
||||||
|
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||||
|
#[non_exhaustive]
|
||||||
|
pub enum Discovery {
|
||||||
|
/// A peer worth dialing: its claimed node name and a dialable address.
|
||||||
|
Candidate { name: String, addr: String },
|
||||||
|
/// Stop dialing this `(name, addr)`. Dial-set only: a connection that
|
||||||
|
/// is already up is membership's business and is left alone; an
|
||||||
|
/// attempt in flight completes on its own; a later `Candidate` for the
|
||||||
|
/// same pair re-adds it with fresh backoff. Unknown pairs are ignored.
|
||||||
|
Withdrawn { name: String, addr: String },
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A source of peers to dial. Implementations are spawned as actors by the
|
||||||
|
/// connector — see the module docs for the contract.
|
||||||
|
pub trait Strategy: Send + 'static {
|
||||||
|
/// Run the strategy: push [`Discovery`] events into `out` as they are
|
||||||
|
/// learned; return when done discovering (or when `out` reports closed —
|
||||||
|
/// the connector is gone). Runs inside an actor, so blocking
|
||||||
|
/// cooperatively is fine.
|
||||||
|
fn run(self: Box<Self>, out: Sender<Discovery>);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The static-seeds strategy: a fixed `(name, addr)` list, announced once.
|
||||||
|
#[derive(Debug, Clone, Default)]
|
||||||
|
pub struct StaticSeeds {
|
||||||
|
seeds: Vec<(String, String)>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl StaticSeeds {
|
||||||
|
pub fn new(seeds: impl IntoIterator<Item = (impl Into<String>, impl Into<String>)>) -> Self {
|
||||||
|
StaticSeeds {
|
||||||
|
seeds: seeds
|
||||||
|
.into_iter()
|
||||||
|
.map(|(n, a)| (n.into(), a.into()))
|
||||||
|
.collect(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Strategy for StaticSeeds {
|
||||||
|
fn run(self: Box<Self>, out: Sender<Discovery>) {
|
||||||
|
for (name, addr) in self.seeds {
|
||||||
|
if out.send(Discovery::Candidate { name, addr }).is_err() {
|
||||||
|
return; // connector gone; nobody to discover for
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,549 @@
|
|||||||
|
//! RFC 010 c2 — the owned wire envelope.
|
||||||
|
//!
|
||||||
|
//! Every control-plane frame is `u32` little-endian length prefix (of tag +
|
||||||
|
//! body), `u8` tag, hand-encoded body. postcard appears in exactly one place:
|
||||||
|
//! the payload blob inside `Send`/`SendNamed`, via [`encode_payload`] /
|
||||||
|
//! [`decode_payload`] — the seam where a codec swap would land (RFC 010 §2).
|
||||||
|
//! Everything else is hand-rolled and wholly owned.
|
||||||
|
//!
|
||||||
|
//! Integers are little-endian. Strings are `u16` length + UTF-8 bytes.
|
||||||
|
//! Payload blobs are `u32` length + bytes. Enum-shaped fields
|
||||||
|
//! ([`RejectReason`], [`DownReason`]) are a single tag byte.
|
||||||
|
|
||||||
|
use crate::monitor::DownReason;
|
||||||
|
use crate::pg::Incarnation;
|
||||||
|
|
||||||
|
/// Wire protocol version, checked in the handshake (c5).
|
||||||
|
pub const PROTO_VERSION: u32 = 1;
|
||||||
|
|
||||||
|
/// Hard cap on the length prefix. The control plane never carries bulk data
|
||||||
|
/// (RFC 010 §5 — that is the jarred rkyv plane), so anything larger is
|
||||||
|
/// corruption or an attack, not a legitimate frame.
|
||||||
|
pub const MAX_FRAME_LEN: usize = 16 * 1024 * 1024;
|
||||||
|
|
||||||
|
/// Per-node metadata exchanged in the handshake (RFC 010 §1: not identity).
|
||||||
|
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||||
|
pub struct NodeMeta {
|
||||||
|
pub role: String,
|
||||||
|
pub region: String,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Why a `Hello` was rejected.
|
||||||
|
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||||
|
pub enum RejectReason {
|
||||||
|
/// Build hashes differ — not the same binary.
|
||||||
|
HashMismatch,
|
||||||
|
/// The offered node name is already claimed by a live peer.
|
||||||
|
NameTaken,
|
||||||
|
/// Wire protocol version mismatch.
|
||||||
|
ProtoVersion,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The control-plane frame inventory (RFC 010, *Implementation details*).
|
||||||
|
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||||
|
pub enum Frame {
|
||||||
|
Hello {
|
||||||
|
proto_version: u32,
|
||||||
|
build_hash: u64,
|
||||||
|
node_name: String,
|
||||||
|
incarnation: Incarnation,
|
||||||
|
meta: NodeMeta,
|
||||||
|
},
|
||||||
|
HelloAck {
|
||||||
|
node_name: String,
|
||||||
|
incarnation: Incarnation,
|
||||||
|
meta: NodeMeta,
|
||||||
|
},
|
||||||
|
HelloReject {
|
||||||
|
reason: RejectReason,
|
||||||
|
},
|
||||||
|
Heartbeat,
|
||||||
|
Send {
|
||||||
|
/// Target slot index (node is implicit in the connection, incarnation
|
||||||
|
/// is bound at handshake — RFC 010 §3).
|
||||||
|
index: u32,
|
||||||
|
generation: u32,
|
||||||
|
type_hash: u64,
|
||||||
|
payload: Vec<u8>,
|
||||||
|
},
|
||||||
|
SendNamed {
|
||||||
|
name: String,
|
||||||
|
type_hash: u64,
|
||||||
|
payload: Vec<u8>,
|
||||||
|
},
|
||||||
|
Monitor {
|
||||||
|
monitor_id: u64,
|
||||||
|
index: u32,
|
||||||
|
generation: u32,
|
||||||
|
},
|
||||||
|
Demonitor {
|
||||||
|
monitor_id: u64,
|
||||||
|
},
|
||||||
|
Down {
|
||||||
|
monitor_id: u64,
|
||||||
|
reason: RemoteDownReason,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Why a remotely-monitored actor is reported down: either the target's own
|
||||||
|
/// terminal [`DownReason`] as its node recorded it, or the *link* to that
|
||||||
|
/// node was lost (or absent) — which says nothing about the actor itself.
|
||||||
|
///
|
||||||
|
/// This is the cluster-side widening of `DownReason` (p5): `Disconnected`
|
||||||
|
/// is a fact about a connection, never about a local actor, so it lives
|
||||||
|
/// here rather than in the core enum — a local `Down` can never carry it,
|
||||||
|
/// and matches on `DownReason` stay exhaustive over actor outcomes only.
|
||||||
|
/// On the wire `Local(r)` uses `r`'s tag and `Disconnected` is tag 5,
|
||||||
|
/// bound since c11; no peer emits it today (a lost link is synthesized
|
||||||
|
/// locally), but the codec honours it both ways.
|
||||||
|
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||||
|
pub enum RemoteDownReason {
|
||||||
|
/// The target itself terminated; the peer reported this reason.
|
||||||
|
Local(DownReason),
|
||||||
|
/// The link to the target's node was lost or was never up.
|
||||||
|
Disconnected,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl RemoteDownReason {
|
||||||
|
/// The actor's own reason, if this was not a link loss.
|
||||||
|
pub fn local(self) -> Option<DownReason> {
|
||||||
|
match self {
|
||||||
|
RemoteDownReason::Local(r) => Some(r),
|
||||||
|
RemoteDownReason::Disconnected => None,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<DownReason> for RemoteDownReason {
|
||||||
|
fn from(r: DownReason) -> Self {
|
||||||
|
RemoteDownReason::Local(r)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Frame tags. 0 is deliberately unassigned so an all-zero buffer never parses.
|
||||||
|
const TAG_HELLO: u8 = 1;
|
||||||
|
const TAG_HELLO_ACK: u8 = 2;
|
||||||
|
const TAG_HELLO_REJECT: u8 = 3;
|
||||||
|
const TAG_HEARTBEAT: u8 = 4;
|
||||||
|
const TAG_SEND: u8 = 5;
|
||||||
|
const TAG_SEND_NAMED: u8 = 6;
|
||||||
|
const TAG_MONITOR: u8 = 7;
|
||||||
|
const TAG_DEMONITOR: u8 = 8;
|
||||||
|
const TAG_DOWN: u8 = 9;
|
||||||
|
|
||||||
|
// RejectReason tags.
|
||||||
|
const REJ_HASH_MISMATCH: u8 = 1;
|
||||||
|
const REJ_NAME_TAKEN: u8 = 2;
|
||||||
|
const REJ_PROTO_VERSION: u8 = 3;
|
||||||
|
|
||||||
|
// DownReason tags. Do not reuse tags.
|
||||||
|
const DR_EXIT: u8 = 1;
|
||||||
|
const DR_PANIC: u8 = 2;
|
||||||
|
const DR_STOPPED: u8 = 3;
|
||||||
|
const DR_NOPROC: u8 = 4;
|
||||||
|
const DR_DISCONNECTED: u8 = 5;
|
||||||
|
// `Shutdown` never rides in a `Down` by contract (a target that honours the
|
||||||
|
// request exits normally) — the tag exists so the codec stays total.
|
||||||
|
const DR_SHUTDOWN: u8 = 6;
|
||||||
|
|
||||||
|
/// Frame could not be encoded. The output buffer is left exactly as it was.
|
||||||
|
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||||
|
pub enum EncodeError {
|
||||||
|
/// tag + body exceed [`MAX_FRAME_LEN`].
|
||||||
|
FrameTooLarge { len: usize },
|
||||||
|
/// A string field exceeds `u16::MAX` bytes.
|
||||||
|
StringTooLong { len: usize },
|
||||||
|
}
|
||||||
|
|
||||||
|
impl core::fmt::Display for EncodeError {
|
||||||
|
fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result {
|
||||||
|
match self {
|
||||||
|
Self::FrameTooLarge { len } => {
|
||||||
|
write!(f, "frame body of {len} bytes exceeds MAX_FRAME_LEN")
|
||||||
|
}
|
||||||
|
Self::StringTooLong { len } => {
|
||||||
|
write!(f, "string field of {len} bytes exceeds u16::MAX")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl std::error::Error for EncodeError {}
|
||||||
|
|
||||||
|
/// Frame could not be decoded. Everything here is *corruption* — "not enough
|
||||||
|
/// bytes yet" is the `Ok(None)` streaming case, never an error.
|
||||||
|
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||||
|
pub enum DecodeError {
|
||||||
|
/// The length prefix exceeds [`MAX_FRAME_LEN`].
|
||||||
|
FrameTooLarge { declared: usize },
|
||||||
|
/// The length prefix is zero — there is no tag byte.
|
||||||
|
EmptyFrame,
|
||||||
|
/// Unknown frame tag.
|
||||||
|
UnknownTag(u8),
|
||||||
|
/// Unknown tag for an enum-shaped field.
|
||||||
|
UnknownEnumTag { what: &'static str, tag: u8 },
|
||||||
|
/// A field ran past the declared frame end (the length prefix lied long,
|
||||||
|
/// or a length-carrying field inside the body lied).
|
||||||
|
Truncated,
|
||||||
|
/// Bytes were left over after the body (the length prefix lied short).
|
||||||
|
Trailing { extra: usize },
|
||||||
|
/// A string field was not valid UTF-8.
|
||||||
|
Utf8,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl core::fmt::Display for DecodeError {
|
||||||
|
fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result {
|
||||||
|
match self {
|
||||||
|
Self::FrameTooLarge { declared } => {
|
||||||
|
write!(f, "declared frame length {declared} exceeds MAX_FRAME_LEN")
|
||||||
|
}
|
||||||
|
Self::EmptyFrame => write!(f, "zero-length frame (no tag byte)"),
|
||||||
|
Self::UnknownTag(t) => write!(f, "unknown frame tag {t}"),
|
||||||
|
Self::UnknownEnumTag { what, tag } => write!(f, "unknown {what} tag {tag}"),
|
||||||
|
Self::Truncated => write!(f, "frame body truncated mid-field"),
|
||||||
|
Self::Trailing { extra } => write!(f, "{extra} trailing bytes after frame body"),
|
||||||
|
Self::Utf8 => write!(f, "string field is not valid UTF-8"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl std::error::Error for DecodeError {}
|
||||||
|
|
||||||
|
impl Frame {
|
||||||
|
/// Append this frame, length-prefixed, to `out`.
|
||||||
|
///
|
||||||
|
/// On error `out` is left untouched.
|
||||||
|
pub fn encode(&self, out: &mut Vec<u8>) -> Result<(), EncodeError> {
|
||||||
|
let start = out.len();
|
||||||
|
out.extend_from_slice(&[0u8; 4]); // length placeholder, patched below
|
||||||
|
let result = self.encode_body(out);
|
||||||
|
match result {
|
||||||
|
Ok(()) => {
|
||||||
|
let frame_len = out.len() - start - 4;
|
||||||
|
if frame_len > MAX_FRAME_LEN {
|
||||||
|
out.truncate(start);
|
||||||
|
return Err(EncodeError::FrameTooLarge { len: frame_len });
|
||||||
|
}
|
||||||
|
// Cast is lossless: MAX_FRAME_LEN < u32::MAX, checked above.
|
||||||
|
let len32 = frame_len as u32;
|
||||||
|
out[start..start + 4].copy_from_slice(&len32.to_le_bytes());
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
Err(e) => {
|
||||||
|
out.truncate(start);
|
||||||
|
Err(e)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn encode_body(&self, out: &mut Vec<u8>) -> Result<(), EncodeError> {
|
||||||
|
match self {
|
||||||
|
Frame::Hello {
|
||||||
|
proto_version,
|
||||||
|
build_hash,
|
||||||
|
node_name,
|
||||||
|
incarnation,
|
||||||
|
meta,
|
||||||
|
} => {
|
||||||
|
out.push(TAG_HELLO);
|
||||||
|
put_u32(out, *proto_version);
|
||||||
|
put_u64(out, *build_hash);
|
||||||
|
put_str(out, node_name)?;
|
||||||
|
put_u32(out, incarnation.get());
|
||||||
|
put_meta(out, meta)?;
|
||||||
|
}
|
||||||
|
Frame::HelloAck {
|
||||||
|
node_name,
|
||||||
|
incarnation,
|
||||||
|
meta,
|
||||||
|
} => {
|
||||||
|
out.push(TAG_HELLO_ACK);
|
||||||
|
put_str(out, node_name)?;
|
||||||
|
put_u32(out, incarnation.get());
|
||||||
|
put_meta(out, meta)?;
|
||||||
|
}
|
||||||
|
Frame::HelloReject { reason } => {
|
||||||
|
out.push(TAG_HELLO_REJECT);
|
||||||
|
out.push(match reason {
|
||||||
|
RejectReason::HashMismatch => REJ_HASH_MISMATCH,
|
||||||
|
RejectReason::NameTaken => REJ_NAME_TAKEN,
|
||||||
|
RejectReason::ProtoVersion => REJ_PROTO_VERSION,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
Frame::Heartbeat => out.push(TAG_HEARTBEAT),
|
||||||
|
Frame::Send {
|
||||||
|
index,
|
||||||
|
generation,
|
||||||
|
type_hash,
|
||||||
|
payload,
|
||||||
|
} => {
|
||||||
|
out.push(TAG_SEND);
|
||||||
|
put_u32(out, *index);
|
||||||
|
put_u32(out, *generation);
|
||||||
|
put_u64(out, *type_hash);
|
||||||
|
put_blob(out, payload)?;
|
||||||
|
}
|
||||||
|
Frame::SendNamed {
|
||||||
|
name,
|
||||||
|
type_hash,
|
||||||
|
payload,
|
||||||
|
} => {
|
||||||
|
out.push(TAG_SEND_NAMED);
|
||||||
|
put_str(out, name)?;
|
||||||
|
put_u64(out, *type_hash);
|
||||||
|
put_blob(out, payload)?;
|
||||||
|
}
|
||||||
|
Frame::Monitor {
|
||||||
|
monitor_id,
|
||||||
|
index,
|
||||||
|
generation,
|
||||||
|
} => {
|
||||||
|
out.push(TAG_MONITOR);
|
||||||
|
put_u64(out, *monitor_id);
|
||||||
|
put_u32(out, *index);
|
||||||
|
put_u32(out, *generation);
|
||||||
|
}
|
||||||
|
Frame::Demonitor { monitor_id } => {
|
||||||
|
out.push(TAG_DEMONITOR);
|
||||||
|
put_u64(out, *monitor_id);
|
||||||
|
}
|
||||||
|
Frame::Down { monitor_id, reason } => {
|
||||||
|
out.push(TAG_DOWN);
|
||||||
|
put_u64(out, *monitor_id);
|
||||||
|
out.push(match reason {
|
||||||
|
RemoteDownReason::Local(DownReason::Exit) => DR_EXIT,
|
||||||
|
RemoteDownReason::Local(DownReason::Panic) => DR_PANIC,
|
||||||
|
RemoteDownReason::Local(DownReason::Stopped) => DR_STOPPED,
|
||||||
|
RemoteDownReason::Local(DownReason::NoProc) => DR_NOPROC,
|
||||||
|
RemoteDownReason::Local(DownReason::Shutdown) => DR_SHUTDOWN,
|
||||||
|
RemoteDownReason::Disconnected => DR_DISCONNECTED,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Try to decode one frame from the start of `buf`.
|
||||||
|
///
|
||||||
|
/// `Ok(Some((frame, consumed)))` — a full frame; the caller advances by
|
||||||
|
/// `consumed`. `Ok(None)` — not enough bytes yet (streaming); read more
|
||||||
|
/// and retry. `Err(_)` — the bytes are corrupt; the connection is dead.
|
||||||
|
pub fn decode(buf: &[u8]) -> Result<Option<(Frame, usize)>, DecodeError> {
|
||||||
|
let Some(prefix) = buf.get(0..4) else {
|
||||||
|
return Ok(None);
|
||||||
|
};
|
||||||
|
let mut len4 = [0u8; 4];
|
||||||
|
len4.copy_from_slice(prefix);
|
||||||
|
let declared = u32::from_le_bytes(len4) as usize;
|
||||||
|
if declared > MAX_FRAME_LEN {
|
||||||
|
return Err(DecodeError::FrameTooLarge { declared });
|
||||||
|
}
|
||||||
|
if declared == 0 {
|
||||||
|
return Err(DecodeError::EmptyFrame);
|
||||||
|
}
|
||||||
|
let Some(body) = buf.get(4..4 + declared) else {
|
||||||
|
return Ok(None);
|
||||||
|
};
|
||||||
|
let mut r = Reader { buf: body, pos: 0 };
|
||||||
|
let frame = Self::decode_body(&mut r)?;
|
||||||
|
if r.pos != body.len() {
|
||||||
|
return Err(DecodeError::Trailing {
|
||||||
|
extra: body.len() - r.pos,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
Ok(Some((frame, 4 + declared)))
|
||||||
|
}
|
||||||
|
|
||||||
|
fn decode_body(r: &mut Reader<'_>) -> Result<Frame, DecodeError> {
|
||||||
|
let tag = r.u8()?;
|
||||||
|
let frame = match tag {
|
||||||
|
TAG_HELLO => Frame::Hello {
|
||||||
|
proto_version: r.u32()?,
|
||||||
|
build_hash: r.u64()?,
|
||||||
|
node_name: r.string()?,
|
||||||
|
incarnation: Incarnation::new(r.u32()?),
|
||||||
|
meta: r.meta()?,
|
||||||
|
},
|
||||||
|
TAG_HELLO_ACK => Frame::HelloAck {
|
||||||
|
node_name: r.string()?,
|
||||||
|
incarnation: Incarnation::new(r.u32()?),
|
||||||
|
meta: r.meta()?,
|
||||||
|
},
|
||||||
|
TAG_HELLO_REJECT => Frame::HelloReject {
|
||||||
|
reason: match r.u8()? {
|
||||||
|
REJ_HASH_MISMATCH => RejectReason::HashMismatch,
|
||||||
|
REJ_NAME_TAKEN => RejectReason::NameTaken,
|
||||||
|
REJ_PROTO_VERSION => RejectReason::ProtoVersion,
|
||||||
|
t => {
|
||||||
|
return Err(DecodeError::UnknownEnumTag {
|
||||||
|
what: "RejectReason",
|
||||||
|
tag: t,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
},
|
||||||
|
},
|
||||||
|
TAG_HEARTBEAT => Frame::Heartbeat,
|
||||||
|
TAG_SEND => Frame::Send {
|
||||||
|
index: r.u32()?,
|
||||||
|
generation: r.u32()?,
|
||||||
|
type_hash: r.u64()?,
|
||||||
|
payload: r.blob()?,
|
||||||
|
},
|
||||||
|
TAG_SEND_NAMED => Frame::SendNamed {
|
||||||
|
name: r.string()?,
|
||||||
|
type_hash: r.u64()?,
|
||||||
|
payload: r.blob()?,
|
||||||
|
},
|
||||||
|
TAG_MONITOR => Frame::Monitor {
|
||||||
|
monitor_id: r.u64()?,
|
||||||
|
index: r.u32()?,
|
||||||
|
generation: r.u32()?,
|
||||||
|
},
|
||||||
|
TAG_DEMONITOR => Frame::Demonitor {
|
||||||
|
monitor_id: r.u64()?,
|
||||||
|
},
|
||||||
|
TAG_DOWN => Frame::Down {
|
||||||
|
monitor_id: r.u64()?,
|
||||||
|
reason: match r.u8()? {
|
||||||
|
DR_EXIT => RemoteDownReason::Local(DownReason::Exit),
|
||||||
|
DR_PANIC => RemoteDownReason::Local(DownReason::Panic),
|
||||||
|
DR_STOPPED => RemoteDownReason::Local(DownReason::Stopped),
|
||||||
|
DR_NOPROC => RemoteDownReason::Local(DownReason::NoProc),
|
||||||
|
DR_SHUTDOWN => RemoteDownReason::Local(DownReason::Shutdown),
|
||||||
|
DR_DISCONNECTED => RemoteDownReason::Disconnected,
|
||||||
|
t => {
|
||||||
|
return Err(DecodeError::UnknownEnumTag {
|
||||||
|
what: "RemoteDownReason",
|
||||||
|
tag: t,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
},
|
||||||
|
},
|
||||||
|
t => return Err(DecodeError::UnknownTag(t)),
|
||||||
|
};
|
||||||
|
Ok(frame)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Body writers
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
fn put_u32(out: &mut Vec<u8>, v: u32) {
|
||||||
|
out.extend_from_slice(&v.to_le_bytes());
|
||||||
|
}
|
||||||
|
|
||||||
|
fn put_u64(out: &mut Vec<u8>, v: u64) {
|
||||||
|
out.extend_from_slice(&v.to_le_bytes());
|
||||||
|
}
|
||||||
|
|
||||||
|
fn put_str(out: &mut Vec<u8>, s: &str) -> Result<(), EncodeError> {
|
||||||
|
let Ok(len) = u16::try_from(s.len()) else {
|
||||||
|
return Err(EncodeError::StringTooLong { len: s.len() });
|
||||||
|
};
|
||||||
|
out.extend_from_slice(&len.to_le_bytes());
|
||||||
|
out.extend_from_slice(s.as_bytes());
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
fn put_blob(out: &mut Vec<u8>, b: &[u8]) -> Result<(), EncodeError> {
|
||||||
|
let Ok(len) = u32::try_from(b.len()) else {
|
||||||
|
return Err(EncodeError::FrameTooLarge { len: b.len() });
|
||||||
|
};
|
||||||
|
out.extend_from_slice(&len.to_le_bytes());
|
||||||
|
out.extend_from_slice(b);
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
fn put_meta(out: &mut Vec<u8>, m: &NodeMeta) -> Result<(), EncodeError> {
|
||||||
|
put_str(out, &m.role)?;
|
||||||
|
put_str(out, &m.region)
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Body reader
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
struct Reader<'a> {
|
||||||
|
buf: &'a [u8],
|
||||||
|
pos: usize,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Reader<'_> {
|
||||||
|
fn take(&mut self, n: usize) -> Result<&[u8], DecodeError> {
|
||||||
|
let end = self.pos.checked_add(n).ok_or(DecodeError::Truncated)?;
|
||||||
|
let s = self.buf.get(self.pos..end).ok_or(DecodeError::Truncated)?;
|
||||||
|
self.pos = end;
|
||||||
|
Ok(s)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn u8(&mut self) -> Result<u8, DecodeError> {
|
||||||
|
Ok(self.take(1)?[0])
|
||||||
|
}
|
||||||
|
|
||||||
|
fn u16(&mut self) -> Result<u16, DecodeError> {
|
||||||
|
let mut b = [0u8; 2];
|
||||||
|
b.copy_from_slice(self.take(2)?);
|
||||||
|
Ok(u16::from_le_bytes(b))
|
||||||
|
}
|
||||||
|
|
||||||
|
fn u32(&mut self) -> Result<u32, DecodeError> {
|
||||||
|
let mut b = [0u8; 4];
|
||||||
|
b.copy_from_slice(self.take(4)?);
|
||||||
|
Ok(u32::from_le_bytes(b))
|
||||||
|
}
|
||||||
|
|
||||||
|
fn u64(&mut self) -> Result<u64, DecodeError> {
|
||||||
|
let mut b = [0u8; 8];
|
||||||
|
b.copy_from_slice(self.take(8)?);
|
||||||
|
Ok(u64::from_le_bytes(b))
|
||||||
|
}
|
||||||
|
|
||||||
|
fn string(&mut self) -> Result<String, DecodeError> {
|
||||||
|
let len = self.u16()? as usize;
|
||||||
|
let bytes = self.take(len)?;
|
||||||
|
match core::str::from_utf8(bytes) {
|
||||||
|
Ok(s) => Ok(s.to_owned()),
|
||||||
|
Err(_) => Err(DecodeError::Utf8),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn blob(&mut self) -> Result<Vec<u8>, DecodeError> {
|
||||||
|
let len = self.u32()? as usize;
|
||||||
|
Ok(self.take(len)?.to_vec())
|
||||||
|
}
|
||||||
|
|
||||||
|
fn meta(&mut self) -> Result<NodeMeta, DecodeError> {
|
||||||
|
Ok(NodeMeta {
|
||||||
|
role: self.string()?,
|
||||||
|
region: self.string()?,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// The postcard seam (RFC 010 §2) — the ONLY place payload bytes are produced
|
||||||
|
// or consumed. A codec swap lands here and nowhere else.
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
/// Payload (de)serialization failed at the codec seam.
|
||||||
|
#[derive(Debug)]
|
||||||
|
pub struct PayloadError(String);
|
||||||
|
|
||||||
|
impl core::fmt::Display for PayloadError {
|
||||||
|
fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result {
|
||||||
|
write!(f, "payload codec: {}", self.0)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl std::error::Error for PayloadError {}
|
||||||
|
|
||||||
|
/// Serialize a payload value to the wire blob.
|
||||||
|
pub fn encode_payload<T: serde::Serialize + ?Sized>(value: &T) -> Result<Vec<u8>, PayloadError> {
|
||||||
|
postcard::to_allocvec(value).map_err(|e| PayloadError(e.to_string()))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Deserialize a payload value from the wire blob.
|
||||||
|
pub fn decode_payload<T: serde::de::DeserializeOwned>(bytes: &[u8]) -> Result<T, PayloadError> {
|
||||||
|
postcard::from_bytes(bytes).map_err(|e| PayloadError(e.to_string()))
|
||||||
|
}
|
||||||
@@ -0,0 +1,238 @@
|
|||||||
|
//! RFC 010 c8 — explicit exposure: the node's remote surface, and the
|
||||||
|
//! fixed-seed type hash.
|
||||||
|
//!
|
||||||
|
//! Nothing local is remotely reachable by default (RFC §4 — "a gun needs a
|
||||||
|
//! safety"). [`expose`] marks a registered name remotely addressable and
|
||||||
|
//! registers `M`'s decoder under [`type_hash::<M>()`](type_hash);
|
||||||
|
//! [`expose_type`] registers only the decoder (the reply-to path: a
|
||||||
|
//! `RemotePid<A>` received in a message is sendable only if `A::Msg`'s
|
||||||
|
//! decoder was explicitly registered). The exposed set is the node's
|
||||||
|
//! visible, auditable remote surface ([`exposed_names`]).
|
||||||
|
//!
|
||||||
|
//! ## Where the state lives
|
||||||
|
//!
|
||||||
|
//! On `RuntimeInner`, the [`pg`](crate::pg) pattern: a leaf-locked table,
|
||||||
|
//! cfg-gated behind the `cluster` feature (zero-cost-when-off, per c1).
|
||||||
|
//! Chosen over manager-held state because c9's inbound decode consults it
|
||||||
|
//! per frame — a hot path that must not serialize every remote delivery
|
||||||
|
//! through one gen_server. The state resets with the runtime, like every
|
||||||
|
//! registry.
|
||||||
|
//!
|
||||||
|
//! ## The watchable fold (D3), against the code as it stands
|
||||||
|
//!
|
||||||
|
//! RFC §4: the exposed set is not a new registry — it folds into the
|
||||||
|
//! existing `watchable` machinery, one set, two set-sites (a pid crossing
|
||||||
|
//! the membrane, and expose). Reading the code: `register` **already
|
||||||
|
//! stamps every named holder watchable** ("no successfully-registered actor
|
||||||
|
//! can die unflagged", registry.rs), so an exposed *name*'s holder needs no
|
||||||
|
//! extra mark here — the guarantee holds by registration, and re-registration
|
||||||
|
//! after a holder's death re-stamps the new holder for free (a per-tenancy
|
||||||
|
//! mark taken at expose time could not do that). The cluster's own
|
||||||
|
//! `mark_watchable` set-site is therefore the **pid crossing the wire** —
|
||||||
|
//! serialization of a pid into a frame, c10 — the exact analog of the
|
||||||
|
//! membrane crossing. What lives here is only the name/type-level state
|
||||||
|
//! neither the registry nor the slot bits can carry: which names are
|
||||||
|
//! exposed, and how to decode each type hash.
|
||||||
|
//!
|
||||||
|
//! ## The hash
|
||||||
|
//!
|
||||||
|
//! [`type_hash`] is FNV-1a 64 (fixed seed: the FNV offset basis) over
|
||||||
|
//! `TypeId`, so it is a constant of the binary: stable across runs of the
|
||||||
|
//! same build — exactly the scope the build-hash handshake reduces the mesh
|
||||||
|
//! to — and deliberately *not* stable across builds (scope guard: no
|
||||||
|
//! cross-version wire compatibility). A collision between two exposed types
|
||||||
|
//! degrades to a decode error or a refused channel, never a misroute — the
|
||||||
|
//! local `SendError::NoChannel` guarantee survives the network (RFC §3).
|
||||||
|
//!
|
||||||
|
//! ## The decoder contract
|
||||||
|
//!
|
||||||
|
//! A decoder is **decode-and-deliver-to-pid**: it captures `M` (the one
|
||||||
|
//! typed site), decodes the payload, and hands the value to the target's
|
||||||
|
//! published channel via the registry's own dynamic send. Wire-name →
|
||||||
|
//! local-pid resolution deliberately stays *outside* — that is c9's single
|
||||||
|
//! resolution seam, and it calls [`decode_deliver`].
|
||||||
|
|
||||||
|
use std::any::TypeId;
|
||||||
|
use std::collections::HashMap;
|
||||||
|
use std::hash::{Hash, Hasher};
|
||||||
|
|
||||||
|
use crate::cluster::envelope::{decode_payload, PayloadError};
|
||||||
|
use crate::pid::{Name, Pid};
|
||||||
|
use crate::registry::{send_dyn, SendError};
|
||||||
|
use crate::scheduler::with_runtime;
|
||||||
|
|
||||||
|
/// The fixed-seed `TypeId` → `u64` hash: FNV-1a 64 over the `TypeId`'s hash
|
||||||
|
/// bytes, seeded with the FNV offset basis. A constant of the binary — see
|
||||||
|
/// the module docs for scope.
|
||||||
|
pub fn type_hash<M: 'static>() -> u64 {
|
||||||
|
let mut h = Fnv1a64::new();
|
||||||
|
TypeId::of::<M>().hash(&mut h);
|
||||||
|
h.finish()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// FNV-1a 64 as a `Hasher`, so `TypeId` (opaque, `Hash`-only) can feed it.
|
||||||
|
/// Same constants as the const fns in [`crate::cluster`] (BUILD_HASH).
|
||||||
|
struct Fnv1a64(u64);
|
||||||
|
|
||||||
|
impl Fnv1a64 {
|
||||||
|
fn new() -> Self {
|
||||||
|
Fnv1a64(0xcbf2_9ce4_8422_2325)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Hasher for Fnv1a64 {
|
||||||
|
fn write(&mut self, bytes: &[u8]) {
|
||||||
|
for &b in bytes {
|
||||||
|
self.0 ^= b as u64;
|
||||||
|
self.0 = self.0.wrapping_mul(0x0000_0100_0000_01b3);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
fn finish(&self) -> u64 {
|
||||||
|
self.0
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Why a [`decode_deliver`] did not deliver. Payload-free mirror of the
|
||||||
|
/// registry's `SendError` where relevant — the caller (c9's inbound path)
|
||||||
|
/// has only bytes to give back, not a typed message.
|
||||||
|
#[derive(Debug)]
|
||||||
|
pub enum DeliverError {
|
||||||
|
/// No decoder is registered under this hash — the type was never
|
||||||
|
/// exposed here.
|
||||||
|
UnknownType,
|
||||||
|
/// The bytes did not decode as the registered type.
|
||||||
|
Decode(PayloadError),
|
||||||
|
/// The target actor is dead (or was never alive).
|
||||||
|
Dead,
|
||||||
|
/// The target is live but has no channel for this message type, or that
|
||||||
|
/// channel is closed — the `NoChannel` guarantee: a decoded value is
|
||||||
|
/// refused, never misrouted.
|
||||||
|
WrongChannel,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A registered decoder: decode `bytes` as the captured type and deliver to
|
||||||
|
/// `pid`'s published channel. `Arc`, so [`decode_deliver`] can clone it out
|
||||||
|
/// from under the exposure lock and call it lock-free — the decoder's
|
||||||
|
/// `send_dyn` takes the registry lock, and the two are mutual Leaves that
|
||||||
|
/// must never nest.
|
||||||
|
type Decoder = std::sync::Arc<dyn Fn(Pid, &[u8]) -> Result<(), DeliverError> + Send + Sync>;
|
||||||
|
|
||||||
|
/// The exposure state, one per runtime (a `RuntimeInner` field, pg-style).
|
||||||
|
pub(crate) struct ExposureState {
|
||||||
|
/// The exposed names: registry key → the type hash it expects.
|
||||||
|
exposed: HashMap<&'static str, u64>,
|
||||||
|
/// The decoders: type hash → decode-and-deliver.
|
||||||
|
decoders: HashMap<u64, Decoder>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl ExposureState {
|
||||||
|
pub(crate) fn new() -> Self {
|
||||||
|
ExposureState {
|
||||||
|
exposed: HashMap::new(),
|
||||||
|
decoders: HashMap::new(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Mark `name` remotely addressable and register `M`'s decoder under its
|
||||||
|
/// type hash (so both name-sends and pid-sends of `M` work — RFC §4).
|
||||||
|
/// Returns the hash.
|
||||||
|
///
|
||||||
|
/// Exposure is a **name-level fact**, independent of who currently holds the
|
||||||
|
/// name (names late-bind: the registry re-resolves on every send, and c9's
|
||||||
|
/// seam resolves per delivery). Exposing an unregistered name is therefore
|
||||||
|
/// valid — deliveries fail with "unresolved" until someone registers it.
|
||||||
|
/// Idempotent. Must run inside [`run`](crate::run).
|
||||||
|
pub fn expose<M>(name: Name<M>) -> u64
|
||||||
|
where
|
||||||
|
M: serde::de::DeserializeOwned + Send + 'static,
|
||||||
|
{
|
||||||
|
let h = ensure_decoder::<M>();
|
||||||
|
with_runtime(|inner| {
|
||||||
|
inner.exposure.lock().exposed.insert(name.as_str(), h);
|
||||||
|
});
|
||||||
|
h
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Register only `M`'s decoder (no name): the reply-to path. Returns the
|
||||||
|
/// hash. Idempotent. Must run inside [`run`](crate::run).
|
||||||
|
pub fn expose_type<M>() -> u64
|
||||||
|
where
|
||||||
|
M: serde::de::DeserializeOwned + Send + 'static,
|
||||||
|
{
|
||||||
|
ensure_decoder::<M>()
|
||||||
|
}
|
||||||
|
|
||||||
|
fn ensure_decoder<M>() -> u64
|
||||||
|
where
|
||||||
|
M: serde::de::DeserializeOwned + Send + 'static,
|
||||||
|
{
|
||||||
|
let h = type_hash::<M>();
|
||||||
|
with_runtime(|inner| {
|
||||||
|
inner
|
||||||
|
.exposure
|
||||||
|
.lock()
|
||||||
|
.decoders
|
||||||
|
.entry(h)
|
||||||
|
.or_insert_with(decoder::<M>);
|
||||||
|
});
|
||||||
|
h
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The one typed site: decode as `M`, deliver via the registry's dynamic
|
||||||
|
/// send. See the module docs for the error mapping.
|
||||||
|
fn decoder<M>() -> Decoder
|
||||||
|
where
|
||||||
|
M: serde::de::DeserializeOwned + Send + 'static,
|
||||||
|
{
|
||||||
|
std::sync::Arc::new(|pid, bytes| {
|
||||||
|
let m: M = decode_payload(bytes).map_err(DeliverError::Decode)?;
|
||||||
|
send_dyn(pid, m).map_err(|e| match e {
|
||||||
|
SendError::Dead(_) | SendError::Unresolved(_) | SendError::NoMember(_) => {
|
||||||
|
DeliverError::Dead
|
||||||
|
}
|
||||||
|
SendError::NoChannel(_) | SendError::Closed(_) => DeliverError::WrongChannel,
|
||||||
|
})
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The type hash `name` was exposed with, or `None` if it is not exposed.
|
||||||
|
/// Must run inside [`run`](crate::run).
|
||||||
|
pub fn exposed_hash(name: &str) -> Option<u64> {
|
||||||
|
with_runtime(|inner| inner.exposure.lock().exposed.get(name).copied())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether a decoder is registered under `hash`. Must run inside
|
||||||
|
/// [`run`](crate::run).
|
||||||
|
pub fn decoder_registered(hash: u64) -> bool {
|
||||||
|
with_runtime(|inner| inner.exposure.lock().decoders.contains_key(&hash))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Decode `bytes` under `hash`'s registered decoder and deliver to `pid`.
|
||||||
|
/// This is the delivery half c9's single resolution seam calls after it has
|
||||||
|
/// resolved a wire name to a local pid. Must run inside [`run`](crate::run).
|
||||||
|
pub fn decode_deliver(hash: u64, to: Pid, bytes: &[u8]) -> Result<(), DeliverError> {
|
||||||
|
// Clone the Arc under the lock, call outside it: the decoder's
|
||||||
|
// `send_dyn` takes the registry lock — a mutual Leaf with the exposure
|
||||||
|
// lock (the runtime asserts if Leaves nest). This also keeps unrelated
|
||||||
|
// deliveries uncoupled from a slow decode.
|
||||||
|
let d = with_runtime(|inner| inner.exposure.lock().decoders.get(&hash).cloned());
|
||||||
|
match d {
|
||||||
|
Some(d) => d(to, bytes),
|
||||||
|
None => Err(DeliverError::UnknownType),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The auditable remote surface: every exposed name and its type hash,
|
||||||
|
/// unordered. Must run inside [`run`](crate::run).
|
||||||
|
pub fn exposed_names() -> Vec<(&'static str, u64)> {
|
||||||
|
with_runtime(|inner| {
|
||||||
|
inner
|
||||||
|
.exposure
|
||||||
|
.lock()
|
||||||
|
.exposed
|
||||||
|
.iter()
|
||||||
|
.map(|(&n, &h)| (n, h))
|
||||||
|
.collect()
|
||||||
|
})
|
||||||
|
}
|
||||||
@@ -0,0 +1,178 @@
|
|||||||
|
//! RFC 010 c5 — the handshake as a pure state machine.
|
||||||
|
//!
|
||||||
|
//! Frames in, actions out — no IO, no clocks, no actors. The c6 connection
|
||||||
|
//! actor drives these machines and executes their actions; everything
|
||||||
|
//! time-shaped (handshake deadline, heartbeats) lives there.
|
||||||
|
|
||||||
|
use crate::cluster::envelope::{Frame, NodeMeta, RejectReason, PROTO_VERSION};
|
||||||
|
use crate::pg::Incarnation;
|
||||||
|
|
||||||
|
/// This node's identity and metadata, as offered in (or checked against) a
|
||||||
|
/// `Hello`.
|
||||||
|
#[derive(Debug, Clone)]
|
||||||
|
pub struct Local {
|
||||||
|
pub node_name: String,
|
||||||
|
pub incarnation: Incarnation,
|
||||||
|
pub build_hash: u64,
|
||||||
|
pub meta: NodeMeta,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The peer identity a successful handshake yields (what c7 feeds `node_up`).
|
||||||
|
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||||
|
pub struct Peer {
|
||||||
|
pub node_name: String,
|
||||||
|
pub incarnation: Incarnation,
|
||||||
|
pub meta: NodeMeta,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Driver-supplied standing of the *offered* name at this node — knowledge
|
||||||
|
/// the pure machine cannot have (c6 owns the connection table and dial
|
||||||
|
/// set). One answer, in the responder's own precedence: an established
|
||||||
|
/// peer under that name outranks an in-flight dial to it.
|
||||||
|
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
|
||||||
|
pub enum PeerStanding {
|
||||||
|
/// Neither connected to nor dialing that name.
|
||||||
|
#[default]
|
||||||
|
Free,
|
||||||
|
/// An established peer already holds that name.
|
||||||
|
Claimed,
|
||||||
|
/// We have our own dial in flight to that name.
|
||||||
|
Dialing,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Simultaneous-connect tie-break: does the connection dialed by
|
||||||
|
/// `dialer_name` survive against the reverse dial?
|
||||||
|
/// The rule (ratified 2026-08-14, a wire-protocol fact): the connection
|
||||||
|
/// dialed by the lexicographically **smaller** name survives. Both ends know
|
||||||
|
/// both names, so both compute the same verdict — which is why the losing
|
||||||
|
/// side may close silently instead of sending a reject.
|
||||||
|
pub fn dial_wins(dialer_name: &str, acceptor_name: &str) -> bool {
|
||||||
|
dialer_name < acceptor_name
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Dial side: emits `Hello` at construction, interprets the single response.
|
||||||
|
#[must_use]
|
||||||
|
#[derive(Debug)]
|
||||||
|
pub struct Initiator(());
|
||||||
|
|
||||||
|
/// What the dial side's response frame meant.
|
||||||
|
#[must_use]
|
||||||
|
#[derive(Debug, PartialEq, Eq)]
|
||||||
|
pub enum InitiatorOutcome {
|
||||||
|
Established(Peer),
|
||||||
|
Rejected(RejectReason),
|
||||||
|
/// Protocol violation before the ack — close. Carries the offending frame.
|
||||||
|
Failed(Frame),
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Initiator {
|
||||||
|
/// Start a dial-side handshake: the returned frame is the `Hello` to
|
||||||
|
/// send; the returned machine is the right to interpret the response.
|
||||||
|
pub fn new(local: &Local) -> (Self, Frame) {
|
||||||
|
let hello = Frame::Hello {
|
||||||
|
proto_version: PROTO_VERSION,
|
||||||
|
build_hash: local.build_hash,
|
||||||
|
node_name: local.node_name.clone(),
|
||||||
|
incarnation: local.incarnation,
|
||||||
|
meta: local.meta.clone(),
|
||||||
|
};
|
||||||
|
(Initiator(()), hello)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Interpret the response. The `HelloAck` carries no hash or version —
|
||||||
|
/// the responder already checked ours against its own, and equality is
|
||||||
|
/// symmetric, so a one-sided check is sound.
|
||||||
|
pub fn on_frame(self, frame: Frame) -> InitiatorOutcome {
|
||||||
|
match frame {
|
||||||
|
Frame::HelloAck {
|
||||||
|
node_name,
|
||||||
|
incarnation,
|
||||||
|
meta,
|
||||||
|
} => InitiatorOutcome::Established(Peer {
|
||||||
|
node_name,
|
||||||
|
incarnation,
|
||||||
|
meta,
|
||||||
|
}),
|
||||||
|
Frame::HelloReject { reason } => InitiatorOutcome::Rejected(reason),
|
||||||
|
other => InitiatorOutcome::Failed(other),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Accept side: awaits exactly one `Hello`, answers or closes.
|
||||||
|
#[must_use]
|
||||||
|
#[derive(Debug)]
|
||||||
|
pub struct Responder {
|
||||||
|
local: Local,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// What to do with an inbound connection's first frame.
|
||||||
|
#[must_use]
|
||||||
|
#[derive(Debug, PartialEq, Eq)]
|
||||||
|
pub enum ResponderOutcome {
|
||||||
|
/// Send the ack; the connection is established.
|
||||||
|
Accepted { reply: Frame, peer: Peer },
|
||||||
|
/// Send the reject, then close.
|
||||||
|
Rejected { reply: Frame, reason: RejectReason },
|
||||||
|
/// Lost the simultaneous-connect tie-break: close silently, no frame.
|
||||||
|
TieBreakLoss,
|
||||||
|
/// Protocol violation before Hello — close, no reply. Carries the frame.
|
||||||
|
Failed(Frame),
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Responder {
|
||||||
|
pub fn new(local: Local) -> Self {
|
||||||
|
Responder { local }
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Judge the connection's first frame. Check order is proto → hash →
|
||||||
|
/// name → tie-break: validity before identity. `HelloReject` is the
|
||||||
|
/// cross-version compatibility anchor, so a version-mismatched peer
|
||||||
|
/// still gets one.
|
||||||
|
pub fn on_frame(self, frame: Frame, standing: PeerStanding) -> ResponderOutcome {
|
||||||
|
let Frame::Hello {
|
||||||
|
proto_version,
|
||||||
|
build_hash,
|
||||||
|
node_name,
|
||||||
|
incarnation,
|
||||||
|
meta,
|
||||||
|
} = frame
|
||||||
|
else {
|
||||||
|
return ResponderOutcome::Failed(frame);
|
||||||
|
};
|
||||||
|
|
||||||
|
let reject = |reason| ResponderOutcome::Rejected {
|
||||||
|
reply: Frame::HelloReject { reason },
|
||||||
|
reason,
|
||||||
|
};
|
||||||
|
|
||||||
|
if proto_version != PROTO_VERSION {
|
||||||
|
return reject(RejectReason::ProtoVersion);
|
||||||
|
}
|
||||||
|
if build_hash != self.local.build_hash {
|
||||||
|
return reject(RejectReason::HashMismatch);
|
||||||
|
}
|
||||||
|
if node_name == self.local.node_name || standing == PeerStanding::Claimed {
|
||||||
|
return reject(RejectReason::NameTaken);
|
||||||
|
}
|
||||||
|
// Simultaneous connect: the inbound frame is the peer's dial. If our
|
||||||
|
// own in-flight dial wins instead, drop this one silently — the peer
|
||||||
|
// computes the same verdict (see `dial_wins`).
|
||||||
|
if standing == PeerStanding::Dialing && !dial_wins(&node_name, &self.local.node_name) {
|
||||||
|
return ResponderOutcome::TieBreakLoss;
|
||||||
|
}
|
||||||
|
|
||||||
|
ResponderOutcome::Accepted {
|
||||||
|
reply: Frame::HelloAck {
|
||||||
|
node_name: self.local.node_name,
|
||||||
|
incarnation: self.local.incarnation,
|
||||||
|
meta: self.local.meta,
|
||||||
|
},
|
||||||
|
peer: Peer {
|
||||||
|
node_name,
|
||||||
|
incarnation,
|
||||||
|
meta,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,295 @@
|
|||||||
|
//! RFC 010 c6 — the cluster connection manager.
|
||||||
|
//!
|
||||||
|
//! One manager per runtime: the single registry of live peer connections and
|
||||||
|
//! the source of truth for whether a peer name is already claimed. The
|
||||||
|
//! accept/connect path registers each established connection here, handing
|
||||||
|
//! over its [`ConnHandle`] — **the manager owns connection lifetime**. A
|
||||||
|
//! connection lives as long as its table entry, so it neither outlives nor
|
||||||
|
//! dies with whichever actor happened to establish it. Registered actors are
|
||||||
|
//! also *monitored*, so the table self-heals on any exit path — a connection
|
||||||
|
//! that panics, is cancelled, or closes cleanly is removed without
|
||||||
|
//! cooperation from the dying actor.
|
||||||
|
//!
|
||||||
|
//! The manager also holds the **membership state** (c7a): `node_up` fires on
|
||||||
|
//! a successful registration and `node_down` on removal — they are derived
|
||||||
|
//! facts of the exact events this table already owns, so holding the view
|
||||||
|
//! here means no cross-actor race between "connection exists" and "node is
|
||||||
|
//! up". The consumer surface (event types, [`subscribe`], [`view`],
|
||||||
|
//! semantics) is [`membership`](crate::cluster::membership); no consumer
|
||||||
|
//! ever touches the table itself.
|
||||||
|
//!
|
||||||
|
//! The connector dial loop is c7b, built on top of both.
|
||||||
|
|
||||||
|
use std::collections::HashMap;
|
||||||
|
|
||||||
|
use crate::channel::Sender;
|
||||||
|
use crate::cluster::conn::ConnHandle;
|
||||||
|
use crate::cluster::handshake::{Peer, PeerStanding};
|
||||||
|
use crate::cluster::membership::{NodeEvent, NodeInfo};
|
||||||
|
use crate::cluster::remote::{bind_outbound, unbind_outbound};
|
||||||
|
use crate::gen_server::{GenServer, GenServerCtx, GenServerName, Watcher};
|
||||||
|
use crate::monitor::{monitor, Down};
|
||||||
|
use crate::pg::NodeId;
|
||||||
|
use crate::pid::Pid;
|
||||||
|
|
||||||
|
/// Well-known name of the singleton manager within a runtime. Connection
|
||||||
|
/// actors reach it by name rather than by a passed-around ref, so a restarted
|
||||||
|
/// manager is always found at the same key.
|
||||||
|
pub const MANAGER: GenServerName<Manager> = GenServerName::new("smarm.cluster.manager");
|
||||||
|
|
||||||
|
/// One live connection's entry: the actor running it, the handle whose
|
||||||
|
/// lifetime *is* the connection's (see the module docs), and the peer's
|
||||||
|
/// membership identity (what `node_up` announced and `node_down` will name).
|
||||||
|
struct ConnEntry {
|
||||||
|
pid: Pid,
|
||||||
|
info: NodeInfo,
|
||||||
|
_handle: ConnHandle,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The connection registry: peer name → the connection actor that owns that
|
||||||
|
/// peer's control connection. Plus the membership state layered on it (c7a):
|
||||||
|
/// subscribers, and the `(name, incarnation)` → [`NodeId`] memo.
|
||||||
|
pub struct Manager {
|
||||||
|
conns: HashMap<String, ConnEntry>,
|
||||||
|
/// In-flight dial intents: peer name -> the actor performing the dial.
|
||||||
|
/// Registered *before* connecting so a crossing inbound `Hello` sees it
|
||||||
|
/// ([`PeerStanding::Dialing`]); cleared the moment
|
||||||
|
/// the dial resolves, and — because the dialer is monitored — on the
|
||||||
|
/// dialer's death, so a panicking dial can never wedge the tie-break.
|
||||||
|
dials: HashMap<String, Pid>,
|
||||||
|
/// Membership subscribers; a closed channel is pruned on the next emit.
|
||||||
|
subscribers: Vec<Sender<NodeEvent>>,
|
||||||
|
/// The [`NodeId`] memo: a reconnect at the same incarnation keeps its id,
|
||||||
|
/// a restart (new incarnation) allocates a fresh one. Grows one entry per
|
||||||
|
/// distinct `(name, incarnation)` ever seen — unbounded in principle,
|
||||||
|
/// bounded in practice by restarts actually happening.
|
||||||
|
ids: HashMap<(String, u32), NodeId>,
|
||||||
|
/// Next id to allocate. Starts at 1: id 0 is
|
||||||
|
/// [`DEFAULT_NODE_ID`](crate::pg::DEFAULT_NODE_ID), the local node.
|
||||||
|
next_id: u32,
|
||||||
|
watcher: Option<Watcher<Manager>>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Manager {
|
||||||
|
pub fn new() -> Self {
|
||||||
|
Manager {
|
||||||
|
conns: HashMap::new(),
|
||||||
|
dials: HashMap::new(),
|
||||||
|
subscribers: Vec::new(),
|
||||||
|
ids: HashMap::new(),
|
||||||
|
next_id: 1,
|
||||||
|
watcher: None,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The memoized id for `(name, incarnation)` — see the field docs.
|
||||||
|
fn node_id(&mut self, name: &str, incarnation: u32) -> NodeId {
|
||||||
|
*self
|
||||||
|
.ids
|
||||||
|
.entry((name.to_string(), incarnation))
|
||||||
|
.or_insert_with(|| {
|
||||||
|
let id = NodeId::new(self.next_id);
|
||||||
|
self.next_id += 1;
|
||||||
|
id
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Deliver `event` to every live subscriber, pruning the dead: a closed
|
||||||
|
/// channel means the subscriber dropped its [`MembershipEvents`]
|
||||||
|
/// (crate::cluster::membership::MembershipEvents).
|
||||||
|
fn emit(&mut self, event: &NodeEvent) {
|
||||||
|
self.subscribers.retain(|tx| tx.send(event.clone()).is_ok());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Default for Manager {
|
||||||
|
fn default() -> Self {
|
||||||
|
Manager::new()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Outcome of a [`Call::Register`].
|
||||||
|
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||||
|
pub enum Registered {
|
||||||
|
/// The name was free; this connection is now the peer of record.
|
||||||
|
Ok,
|
||||||
|
/// Another live connection already holds this name — the caller lost the
|
||||||
|
/// race (or is a duplicate) and must not run.
|
||||||
|
Duplicate,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Requests to the manager.
|
||||||
|
pub enum Call {
|
||||||
|
/// The path claims its peer's name for a freshly-established connection,
|
||||||
|
/// handing the manager the actor's [`ConnHandle`] and the handshake's
|
||||||
|
/// [`Peer`] (the membership identity `node_up` announces). The manager
|
||||||
|
/// monitors `pid` and holds the handle for as long as the entry lives; a
|
||||||
|
/// [`Registered::Duplicate`] verdict drops the handle here, which stops
|
||||||
|
/// the refused actor.
|
||||||
|
Register {
|
||||||
|
peer: Peer,
|
||||||
|
pid: Pid,
|
||||||
|
handle: ConnHandle,
|
||||||
|
},
|
||||||
|
/// Tear down the connection to `name`: the manager drops its handle, the
|
||||||
|
/// actor stops, and the monitor removes the entry. A no-op if no such
|
||||||
|
/// connection is live.
|
||||||
|
Disconnect { name: String },
|
||||||
|
/// The current peer names, sorted. For observation and tests.
|
||||||
|
Peers,
|
||||||
|
/// A dialer declares an in-flight dial to `name` before connecting. The
|
||||||
|
/// pid is the dialing actor, monitored so the intent dies with it.
|
||||||
|
DialBegin { name: String, pid: Pid },
|
||||||
|
/// The dial to `name` resolved (either way): drop the intent. A call,
|
||||||
|
/// not a cast, so the intent is provably gone before the dialer moves on.
|
||||||
|
DialEnd { name: String },
|
||||||
|
/// The [`PeerStanding`] of an inbound `Hello` offering `peer_name` — the
|
||||||
|
/// accept path asks this between reading the frame and judging it.
|
||||||
|
Standing { peer_name: String },
|
||||||
|
/// Subscribe `tx` to membership events, snapshot-then-stream: one
|
||||||
|
/// [`NodeEvent::NodeUp`] per live peer is queued into `tx` before this
|
||||||
|
/// call answers, so the stream is exact from its first event (handlers
|
||||||
|
/// are serialized — nothing interleaves with the snapshot). Use
|
||||||
|
/// [`subscribe`](crate::cluster::membership::subscribe).
|
||||||
|
Subscribe { tx: Sender<NodeEvent> },
|
||||||
|
/// The current view: every live peer's [`NodeInfo`], unordered. Use
|
||||||
|
/// [`view`](crate::cluster::membership::view).
|
||||||
|
View,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Replies from the manager.
|
||||||
|
#[derive(Debug)]
|
||||||
|
pub enum Reply {
|
||||||
|
Registered(Registered),
|
||||||
|
Disconnected,
|
||||||
|
Peers(Vec<String>),
|
||||||
|
/// `false`: another dial to this name is already in flight — do not dial.
|
||||||
|
DialBegan(bool),
|
||||||
|
DialEnded,
|
||||||
|
Standing(PeerStanding),
|
||||||
|
Subscribed,
|
||||||
|
View(Vec<NodeInfo>),
|
||||||
|
}
|
||||||
|
|
||||||
|
impl GenServer for Manager {
|
||||||
|
type Call = Call;
|
||||||
|
type Reply = Reply;
|
||||||
|
type Cast = ();
|
||||||
|
type Info = ();
|
||||||
|
type Timer = ();
|
||||||
|
|
||||||
|
fn init(&mut self, ctx: &GenServerCtx<Self>) {
|
||||||
|
self.watcher = Some(ctx.watcher());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Manager shutdown drops every entry (and with it every ConnHandle);
|
||||||
|
/// the outbound table must not outlive the connections it names.
|
||||||
|
fn terminate(&mut self) {
|
||||||
|
for name in self.conns.keys() {
|
||||||
|
unbind_outbound(name);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn handle_call(&mut self, request: Call) -> Reply {
|
||||||
|
match request {
|
||||||
|
Call::Register {
|
||||||
|
peer,
|
||||||
|
pid,
|
||||||
|
mut handle,
|
||||||
|
} => {
|
||||||
|
if self.conns.contains_key(&peer.node_name) {
|
||||||
|
// `handle` drops here: the refused actor stops itself.
|
||||||
|
return Reply::Registered(Registered::Duplicate);
|
||||||
|
}
|
||||||
|
if let Some(w) = &self.watcher {
|
||||||
|
w.watch(monitor(pid));
|
||||||
|
}
|
||||||
|
// The outbound table (c9) is maintained here, inside the same
|
||||||
|
// serialized handlers that own the connection's lifetime.
|
||||||
|
if let Some((frames, monitors)) = handle.take_outbound() {
|
||||||
|
bind_outbound(&peer.node_name, peer.incarnation, frames, monitors);
|
||||||
|
}
|
||||||
|
let info = NodeInfo {
|
||||||
|
node: self.node_id(&peer.node_name, peer.incarnation.get()),
|
||||||
|
name: peer.node_name.clone(),
|
||||||
|
incarnation: peer.incarnation,
|
||||||
|
meta: peer.meta,
|
||||||
|
};
|
||||||
|
self.conns.insert(
|
||||||
|
peer.node_name,
|
||||||
|
ConnEntry {
|
||||||
|
pid,
|
||||||
|
info: info.clone(),
|
||||||
|
_handle: handle,
|
||||||
|
},
|
||||||
|
);
|
||||||
|
self.emit(&NodeEvent::NodeUp(info));
|
||||||
|
Reply::Registered(Registered::Ok)
|
||||||
|
}
|
||||||
|
Call::Disconnect { name } => {
|
||||||
|
// Dropping the entry drops the handle, which stops the actor.
|
||||||
|
if let Some(entry) = self.conns.remove(&name) {
|
||||||
|
unbind_outbound(&name);
|
||||||
|
self.emit(&NodeEvent::NodeDown(entry.info));
|
||||||
|
}
|
||||||
|
Reply::Disconnected
|
||||||
|
}
|
||||||
|
Call::Peers => {
|
||||||
|
let mut names: Vec<String> = self.conns.keys().cloned().collect();
|
||||||
|
names.sort();
|
||||||
|
Reply::Peers(names)
|
||||||
|
}
|
||||||
|
Call::DialBegin { name, pid } => {
|
||||||
|
if self.dials.contains_key(&name) {
|
||||||
|
return Reply::DialBegan(false);
|
||||||
|
}
|
||||||
|
if let Some(w) = &self.watcher {
|
||||||
|
w.watch(monitor(pid));
|
||||||
|
}
|
||||||
|
self.dials.insert(name, pid);
|
||||||
|
Reply::DialBegan(true)
|
||||||
|
}
|
||||||
|
Call::DialEnd { name } => {
|
||||||
|
self.dials.remove(&name);
|
||||||
|
Reply::DialEnded
|
||||||
|
}
|
||||||
|
Call::Standing { peer_name } => {
|
||||||
|
Reply::Standing(if self.conns.contains_key(&peer_name) {
|
||||||
|
PeerStanding::Claimed
|
||||||
|
} else if self.dials.contains_key(&peer_name) {
|
||||||
|
PeerStanding::Dialing
|
||||||
|
} else {
|
||||||
|
PeerStanding::Free
|
||||||
|
})
|
||||||
|
}
|
||||||
|
Call::Subscribe { tx } => {
|
||||||
|
// The snapshot: queued before `tx` joins the list, and — the
|
||||||
|
// handlers being serialized — before any later event.
|
||||||
|
for entry in self.conns.values() {
|
||||||
|
let _ = tx.send(NodeEvent::NodeUp(entry.info.clone()));
|
||||||
|
}
|
||||||
|
self.subscribers.push(tx);
|
||||||
|
Reply::Subscribed
|
||||||
|
}
|
||||||
|
Call::View => Reply::View(self.conns.values().map(|e| e.info.clone()).collect()),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn handle_cast(&mut self, _request: ()) {}
|
||||||
|
|
||||||
|
fn handle_down(&mut self, down: Down) {
|
||||||
|
let mut downs = Vec::new();
|
||||||
|
self.conns.retain(|name, entry| {
|
||||||
|
let dead = entry.pid == down.pid;
|
||||||
|
if dead {
|
||||||
|
downs.push((name.clone(), entry.info.clone()));
|
||||||
|
}
|
||||||
|
!dead
|
||||||
|
});
|
||||||
|
for (name, info) in downs {
|
||||||
|
unbind_outbound(&name);
|
||||||
|
self.emit(&NodeEvent::NodeDown(info));
|
||||||
|
}
|
||||||
|
self.dials.retain(|_, pid| *pid != down.pid);
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,97 @@
|
|||||||
|
//! RFC 010 c7a — membership: `node_up`/`node_down` events and the view.
|
||||||
|
//!
|
||||||
|
//! The membership *state* lives inside the [`manager`](crate::cluster::manager)
|
||||||
|
//! — `node_up` and `node_down` are derived facts of the exact events the
|
||||||
|
//! manager already owns (a successful registration; a reap or `Disconnect`),
|
||||||
|
//! so holding the view anywhere else would only add a cross-actor ordering
|
||||||
|
//! seam. This module is the consumer surface: the event and view types, and
|
||||||
|
//! the [`subscribe`]/[`view`] entry points. No consumer ever touches the
|
||||||
|
//! connection table (roadmap-binding, enforced by module privacy: the table
|
||||||
|
//! is a private field, and nothing here exposes names→pids).
|
||||||
|
//!
|
||||||
|
//! ## Subscription semantics (ratified 2026-08-15)
|
||||||
|
//!
|
||||||
|
//! [`subscribe`] is **snapshot-then-stream**: the returned receiver first
|
||||||
|
//! yields one [`NodeEvent::NodeUp`] per currently-live peer, then live events
|
||||||
|
//! as they happen. Because the manager is a `gen_server` (handlers are
|
||||||
|
//! serialized), the snapshot is exact — no event can interleave with it, and
|
||||||
|
//! per-subscriber ordering matches the manager's processing order. There is
|
||||||
|
//! no join-race for late subscribers and no separate "get, then diff" dance;
|
||||||
|
//! [`view`] exists for observation, not for synchronization.
|
||||||
|
//!
|
||||||
|
//! A dropped subscriber is pruned on the next emission (its channel reports
|
||||||
|
//! closed) — no monitor needed, the sender itself tells us.
|
||||||
|
//!
|
||||||
|
//! ## NodeId identity
|
||||||
|
//!
|
||||||
|
//! A [`NodeId`] is a compact **local alias for the wire identity**
|
||||||
|
//! `(node_name, incarnation)`, memoized by the manager: a reconnect blip at
|
||||||
|
//! the same incarnation keeps its id (down, then up, same id), while a
|
||||||
|
//! restart — a new incarnation — gets a fresh one, so a node's ghost and its
|
||||||
|
//! successor are always distinguishable. Ids are allocated from 1;
|
||||||
|
//! [`DEFAULT_NODE_ID`](crate::pg::DEFAULT_NODE_ID) (0) remains the local
|
||||||
|
//! node, per [`pg`](crate::pg)'s framing.
|
||||||
|
|
||||||
|
use crate::channel::{channel, Receiver};
|
||||||
|
use crate::cluster::envelope::NodeMeta;
|
||||||
|
use crate::cluster::manager::{Call, Reply, MANAGER};
|
||||||
|
use crate::gen_server;
|
||||||
|
use crate::pg::{Incarnation, NodeId};
|
||||||
|
|
||||||
|
/// One live remote node, as the view and [`NodeEvent::NodeUp`] describe it.
|
||||||
|
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||||
|
pub struct NodeInfo {
|
||||||
|
/// The local alias for `(name, incarnation)` — see the module docs.
|
||||||
|
pub node: NodeId,
|
||||||
|
/// The peer's claimed node name (handshake-verified).
|
||||||
|
pub name: String,
|
||||||
|
/// The peer's incarnation epoch, as offered in its `Hello`.
|
||||||
|
pub incarnation: Incarnation,
|
||||||
|
/// The peer's `Hello` metadata.
|
||||||
|
pub meta: NodeMeta,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A membership change, as delivered to subscribers.
|
||||||
|
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||||
|
pub enum NodeEvent {
|
||||||
|
/// A peer's control connection established and registered.
|
||||||
|
NodeUp(NodeInfo),
|
||||||
|
/// That peer's connection ended — reaped, commanded down, or the manager
|
||||||
|
/// itself shut down. Carries the same [`NodeInfo`] the corresponding
|
||||||
|
/// `NodeUp` delivered, so consumers need no id→name reverse map.
|
||||||
|
NodeDown(NodeInfo),
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A live membership subscription: the receiving end of the event stream
|
||||||
|
/// (the [`Monitor`](crate::monitor::Monitor) shape — read from [`rx`], drop
|
||||||
|
/// to unsubscribe).
|
||||||
|
///
|
||||||
|
/// [rx]: MembershipEvents::rx
|
||||||
|
pub struct MembershipEvents {
|
||||||
|
/// The event stream: the snapshot's `NodeUp`s first, then live events.
|
||||||
|
/// Fold it into a `select` from a plain actor, or pipe it into a
|
||||||
|
/// `gen_server` via `with_info`.
|
||||||
|
pub rx: Receiver<NodeEvent>,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Subscribe to membership events (snapshot-then-stream — see the module
|
||||||
|
/// docs). `None`: the manager is not running. Must be called from inside an
|
||||||
|
/// actor.
|
||||||
|
pub fn subscribe() -> Option<MembershipEvents> {
|
||||||
|
let (tx, rx) = channel();
|
||||||
|
match gen_server::call(MANAGER, Call::Subscribe { tx }) {
|
||||||
|
Ok(Reply::Subscribed) => Some(MembershipEvents { rx }),
|
||||||
|
_ => None,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The current view: every live peer's [`NodeInfo`], unordered. For
|
||||||
|
/// observation and tests; consumers that need to *track* the view should
|
||||||
|
/// [`subscribe`] instead (the snapshot makes the stream self-sufficient).
|
||||||
|
/// `None`: the manager is not running. Must be called from inside an actor.
|
||||||
|
pub fn view() -> Option<Vec<NodeInfo>> {
|
||||||
|
match gen_server::call(MANAGER, Call::View) {
|
||||||
|
Ok(Reply::View(v)) => Some(v),
|
||||||
|
_ => None,
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,546 @@
|
|||||||
|
//! RFC 010 c15 — distributed process groups (Phase 5).
|
||||||
|
//!
|
||||||
|
//! The Erlang `pg` shape (D18): every node's group store is the union of
|
||||||
|
//! its own local members and each peer's *announced* local members. There
|
||||||
|
//! is one **pg actor** per node — the c14 reaper grown up — and it is the
|
||||||
|
//! only writer of remote entries and the only sender of announcements:
|
||||||
|
//!
|
||||||
|
//! - **Origin owns its members.** Joins are local (`pg::join`), the eager
|
||||||
|
//! reaper is the liveness authority, and the origin announces every
|
||||||
|
//! change: `Join`/`Leave` incrementally to every up node, and a full
|
||||||
|
//! `Sync` of its local groups to a peer the moment that peer comes up
|
||||||
|
//! (`NodeUp`). Nobody monitors a remote member; a peer's `NodeDown` sweeps
|
||||||
|
//! every member it announced.
|
||||||
|
//! - **Transport is a pure consumer** of Phase 3/4: the exposed name
|
||||||
|
//! [`PG_NAME`] (`"pg"`) carrying [`PgMsg`] over postcard, sent with
|
||||||
|
//! [`remote::send`]. No new frame, no manager change.
|
||||||
|
//! - **No anti-entropy.** Per-origin ordering rides the single TCP link:
|
||||||
|
//! the actor sends `Sync` to a peer *before* it can send that peer any
|
||||||
|
//! `Join`/`Leave` (both from the same loop, over the same connection), and
|
||||||
|
//! a reconnect is a fresh `NodeUp` ⇒ fresh `Sync` replacing that peer's
|
||||||
|
//! set wholesale.
|
||||||
|
//! - **Local API unchanged.** `members`/`pick`/`dispatch` stay local-only
|
||||||
|
//! (`get_local_members`); a remote entry in the store carries the peer's
|
||||||
|
//! `NodeId` and never surfaces there. Cluster-wide reads are the new,
|
||||||
|
//! additive [`members_all`] over [`GroupMember`] (c16 adds `pick_any` /
|
||||||
|
//! `dispatch_any`).
|
||||||
|
//!
|
||||||
|
//! ## Ordering inside the node
|
||||||
|
//!
|
||||||
|
//! `pg::join`/`pg::leave` mutate the store on the caller's thread and then
|
||||||
|
//! *announce* to the actor's control inbox. Because the store op precedes the
|
||||||
|
//! announcement and the actor re-reads the store before broadcasting a
|
||||||
|
//! `Joined`, an announcement that has been overtaken (the member left or died
|
||||||
|
//! before the actor got to it) is dropped rather than advertised: the wire
|
||||||
|
//! never sees a `Join` for a member the origin no longer holds. `Leave`
|
||||||
|
//! broadcasts unconditionally — a spurious `Leave` is a no-op at the peer.
|
||||||
|
//!
|
||||||
|
//! Inbound: `NodeUp` is emitted by the manager on the accept/connect path,
|
||||||
|
//! *before* the peer's connection actor exists, so it is queued on the
|
||||||
|
//! membership stream before any frame from that peer can reach this inbox.
|
||||||
|
//! The actor still drains membership before it interprets a `PgMsg` whose
|
||||||
|
//! sender it does not know, and drops the message if the sender is still not
|
||||||
|
//! up (a ghost — its next `NodeUp` brings a `Sync`).
|
||||||
|
|
||||||
|
use std::collections::HashMap;
|
||||||
|
|
||||||
|
use crate::channel::{channel, select, Receiver, Selectable};
|
||||||
|
use crate::cluster::expose::expose;
|
||||||
|
use crate::cluster::membership::{subscribe, MembershipEvents, NodeEvent, NodeInfo};
|
||||||
|
use crate::cluster::remote::{
|
||||||
|
self, local_identity, send_to_remote, RemoteName, RemotePid, ToRemoteError,
|
||||||
|
};
|
||||||
|
use crate::monitor::Down;
|
||||||
|
use crate::pg::{
|
||||||
|
live, member_for, reaper_inboxes, sweep_local_death, Incarnation, Member, Membership, PgEvent,
|
||||||
|
};
|
||||||
|
use crate::pid::{assert_type, Addressable, Erased, Pid};
|
||||||
|
use crate::registry::{register, send_to, SendError};
|
||||||
|
use crate::scheduler::with_runtime;
|
||||||
|
use crate::Name;
|
||||||
|
|
||||||
|
/// The exposed name every node's pg actor answers under.
|
||||||
|
pub const PG_NAME: Name<PgMsg> = Name::new("pg");
|
||||||
|
|
||||||
|
/// The pg wire protocol. Every variant is origin-authored: `from` / the
|
||||||
|
/// pid's node is the node whose local members are being described.
|
||||||
|
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||||
|
pub enum PgMsg {
|
||||||
|
/// The origin's complete local membership, sent to a peer on `NodeUp`.
|
||||||
|
/// Replaces whatever the receiver held for that origin.
|
||||||
|
Sync {
|
||||||
|
from: String,
|
||||||
|
groups: Vec<(String, Vec<RemotePid<Erased>>)>,
|
||||||
|
},
|
||||||
|
/// The origin added `pid` (its own) to `group`.
|
||||||
|
Join {
|
||||||
|
group: String,
|
||||||
|
pid: RemotePid<Erased>,
|
||||||
|
},
|
||||||
|
/// The origin removed `pid` from `group` — voluntary leave or death.
|
||||||
|
Leave {
|
||||||
|
group: String,
|
||||||
|
pid: RemotePid<Erased>,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
// Hand-rolled serde (the crate carries no serde-derive), as a 3-tuple with a
|
||||||
|
// leading tag: (0, from, groups) | (1, group, pid) | (2, group, pid).
|
||||||
|
impl serde::Serialize for PgMsg {
|
||||||
|
fn serialize<S: serde::Serializer>(&self, s: S) -> Result<S::Ok, S::Error> {
|
||||||
|
use serde::ser::SerializeTuple;
|
||||||
|
let mut t = s.serialize_tuple(3)?;
|
||||||
|
match self {
|
||||||
|
PgMsg::Sync { from, groups } => {
|
||||||
|
t.serialize_element(&0u8)?;
|
||||||
|
t.serialize_element(from)?;
|
||||||
|
t.serialize_element(groups)?;
|
||||||
|
}
|
||||||
|
PgMsg::Join { group, pid } => {
|
||||||
|
t.serialize_element(&1u8)?;
|
||||||
|
t.serialize_element(group)?;
|
||||||
|
t.serialize_element(pid)?;
|
||||||
|
}
|
||||||
|
PgMsg::Leave { group, pid } => {
|
||||||
|
t.serialize_element(&2u8)?;
|
||||||
|
t.serialize_element(group)?;
|
||||||
|
t.serialize_element(pid)?;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
t.end()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl<'de> serde::Deserialize<'de> for PgMsg {
|
||||||
|
fn deserialize<D: serde::Deserializer<'de>>(d: D) -> Result<Self, D::Error> {
|
||||||
|
struct V;
|
||||||
|
impl<'de> serde::de::Visitor<'de> for V {
|
||||||
|
type Value = PgMsg;
|
||||||
|
fn expecting(&self, f: &mut std::fmt::Formatter) -> std::fmt::Result {
|
||||||
|
f.write_str("a pg message tuple")
|
||||||
|
}
|
||||||
|
fn visit_seq<A: serde::de::SeqAccess<'de>>(
|
||||||
|
self,
|
||||||
|
mut seq: A,
|
||||||
|
) -> Result<PgMsg, A::Error> {
|
||||||
|
use serde::de::Error;
|
||||||
|
let tag: u8 = seq
|
||||||
|
.next_element()?
|
||||||
|
.ok_or_else(|| A::Error::custom("pg: missing tag"))?;
|
||||||
|
let text: String = seq
|
||||||
|
.next_element()?
|
||||||
|
.ok_or_else(|| A::Error::custom("pg: missing name"))?;
|
||||||
|
match tag {
|
||||||
|
0 => {
|
||||||
|
let groups = seq
|
||||||
|
.next_element()?
|
||||||
|
.ok_or_else(|| A::Error::custom("pg: missing groups"))?;
|
||||||
|
Ok(PgMsg::Sync { from: text, groups })
|
||||||
|
}
|
||||||
|
1 | 2 => {
|
||||||
|
let pid = seq
|
||||||
|
.next_element()?
|
||||||
|
.ok_or_else(|| A::Error::custom("pg: missing pid"))?;
|
||||||
|
Ok(if tag == 1 {
|
||||||
|
PgMsg::Join { group: text, pid }
|
||||||
|
} else {
|
||||||
|
PgMsg::Leave { group: text, pid }
|
||||||
|
})
|
||||||
|
}
|
||||||
|
t => Err(A::Error::custom(format!("pg: unknown tag {t}"))),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
d.deserialize_tuple(3, V)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A member of a group as the cluster sees it: on this node (a plain
|
||||||
|
/// [`Pid`], sendable locally) or on a peer (a [`RemotePid`], sendable via
|
||||||
|
/// [`send_to_remote`](remote::send_to_remote)). `Pid` cannot hold a remote
|
||||||
|
/// (D14), hence the two-variant shape.
|
||||||
|
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||||
|
pub enum GroupMember {
|
||||||
|
Local(Pid),
|
||||||
|
Remote(RemotePid<Erased>),
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Every member of `group` cluster-wide, in the store's order: local members
|
||||||
|
/// filtered by the same liveness backstop as [`members`](crate::pg::members),
|
||||||
|
/// remote members exactly as their origins last announced them. Must run
|
||||||
|
/// inside [`run`](crate::run).
|
||||||
|
pub fn members_all(group: &str) -> Vec<GroupMember> {
|
||||||
|
with_runtime(|inner| {
|
||||||
|
let me = inner.node_id;
|
||||||
|
let pg = inner.process_groups.lock();
|
||||||
|
pg.all_of(group)
|
||||||
|
.into_iter()
|
||||||
|
.filter_map(|m| {
|
||||||
|
if m.node == me {
|
||||||
|
live(inner, m.pid).then_some(GroupMember::Local(m.pid))
|
||||||
|
} else {
|
||||||
|
// A remote entry always has its node's name recorded
|
||||||
|
// (they land under the same lock); a missing one is a
|
||||||
|
// node already swept, so it hides rather than misnames.
|
||||||
|
pg.node_name(m.node).map(|name| {
|
||||||
|
GroupMember::Remote(RemotePid::from_parts(
|
||||||
|
name,
|
||||||
|
m.incarnation,
|
||||||
|
m.pid.index(),
|
||||||
|
m.pid.generation(),
|
||||||
|
))
|
||||||
|
})
|
||||||
|
}
|
||||||
|
})
|
||||||
|
.collect()
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One member of `group` cluster-wide, or `None` if it has none: the first
|
||||||
|
/// entry in the store's order (this node's members in join order first when
|
||||||
|
/// they joined first — the same stateless first-live scan as
|
||||||
|
/// [`pick`](crate::pg::pick), extended over the peers' announced members).
|
||||||
|
/// Must run inside [`run`](crate::run).
|
||||||
|
pub fn pick_any(group: &str) -> Option<GroupMember> {
|
||||||
|
members_all(group).into_iter().next()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Why [`dispatch_any`] handed `msg` back.
|
||||||
|
#[derive(Debug)]
|
||||||
|
pub enum DispatchAnyError<M> {
|
||||||
|
/// The group has no member anywhere.
|
||||||
|
NoMember(M),
|
||||||
|
/// The pick was local and the local typed send failed.
|
||||||
|
Local(SendError<M>),
|
||||||
|
/// The pick was remote and the remote send failed at this node.
|
||||||
|
Remote(ToRemoteError<M>),
|
||||||
|
}
|
||||||
|
|
||||||
|
impl<M> DispatchAnyError<M> {
|
||||||
|
/// The undelivered message.
|
||||||
|
pub fn into_inner(self) -> M {
|
||||||
|
match self {
|
||||||
|
DispatchAnyError::NoMember(m) => m,
|
||||||
|
DispatchAnyError::Local(e) => e.into_inner(),
|
||||||
|
DispatchAnyError::Remote(e) => e.into_inner(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`pick_any`] and send in one step, returning the member reached: a local
|
||||||
|
/// pick goes through [`send_to`], a remote one through [`send_to_remote`]
|
||||||
|
/// (so `Ok` for a remote member means "handed to the connection", RFC 010
|
||||||
|
/// §3). Homogeneous pool assumed, as for [`dispatch`](crate::pg::dispatch);
|
||||||
|
/// a wrong `A` degrades to a clean error at the target, never a misroute.
|
||||||
|
/// Must run inside [`run`](crate::run).
|
||||||
|
pub fn dispatch_any<A>(group: &str, msg: A::Msg) -> Result<GroupMember, DispatchAnyError<A::Msg>>
|
||||||
|
where
|
||||||
|
A: Addressable,
|
||||||
|
A::Msg: serde::Serialize,
|
||||||
|
{
|
||||||
|
match pick_any(group) {
|
||||||
|
None => Err(DispatchAnyError::NoMember(msg)),
|
||||||
|
Some(GroupMember::Local(pid)) => send_to(assert_type::<A>(pid), msg)
|
||||||
|
.map(|()| GroupMember::Local(pid))
|
||||||
|
.map_err(DispatchAnyError::Local),
|
||||||
|
Some(GroupMember::Remote(rp)) => send_to_remote(rp.clone().assert_type::<A>(), msg)
|
||||||
|
.map(|()| GroupMember::Remote(rp))
|
||||||
|
.map_err(DispatchAnyError::Remote),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Attach the pg actor to the running cluster. Called once by
|
||||||
|
/// `cluster::start` after the manager is up and the local identity is set;
|
||||||
|
/// spawns the actor if this run has not joined anything yet.
|
||||||
|
pub(crate) fn attach_cluster() {
|
||||||
|
let _ = reaper_inboxes().ctl.send(PgEvent::Attach);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The attached half of the actor's state: who is up (by name) and the
|
||||||
|
/// membership stream.
|
||||||
|
struct Attached {
|
||||||
|
events: MembershipEvents,
|
||||||
|
peers: HashMap<String, NodeInfo>,
|
||||||
|
/// This node's wire identity: `Sync`'s `from`, and the stamp on every
|
||||||
|
/// pid we ship (attach requires it, so no `None` path exists here).
|
||||||
|
me: String,
|
||||||
|
incarnation: Incarnation,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The pg actor: the c14 reaper (`deaths`), the local API's announcements
|
||||||
|
/// (`ctl`), and — once attached — the membership stream and the exposed
|
||||||
|
/// `"pg"` inbox, all in one drain-then-select loop. `deaths`/`ctl` closing
|
||||||
|
/// is the run tearing down; the membership stream closing is the manager
|
||||||
|
/// gone (detach, keep reaping).
|
||||||
|
pub(crate) fn actor(deaths: Receiver<Down>, ctl: Receiver<PgEvent>) {
|
||||||
|
let (pg_tx, pg_rx) = channel::<PgMsg>();
|
||||||
|
let mut cl: Option<Attached> = None;
|
||||||
|
loop {
|
||||||
|
loop {
|
||||||
|
match deaths.try_recv() {
|
||||||
|
Ok(Some(down)) => on_death(cl.as_ref(), down.pid),
|
||||||
|
Ok(None) => break,
|
||||||
|
Err(_) => return,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
loop {
|
||||||
|
match ctl.try_recv() {
|
||||||
|
Ok(Some(PgEvent::Attach)) => {
|
||||||
|
// Own the name BEFORE subscribing (which yields to the
|
||||||
|
// manager): a peer's first frame must find "pg" exposed
|
||||||
|
// and resolvable, or it is dropped. Idempotent for a
|
||||||
|
// re-attach: same actor, same channel (the registry
|
||||||
|
// refuses a *second* live one).
|
||||||
|
let _ = register(PG_NAME, pg_tx.clone());
|
||||||
|
expose(PG_NAME);
|
||||||
|
if let Some(a) = attach() {
|
||||||
|
cl = Some(a);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(Some(PgEvent::Joined { group, pid })) => on_joined(cl.as_ref(), &group, pid),
|
||||||
|
Ok(Some(PgEvent::Left { group, pid })) => on_left(cl.as_ref(), &group, pid),
|
||||||
|
Ok(None) => break,
|
||||||
|
Err(_) => return,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if let Some(a) = cl.as_mut() {
|
||||||
|
if !drain_events(a) {
|
||||||
|
cl = None;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
loop {
|
||||||
|
match pg_rx.try_recv() {
|
||||||
|
Ok(Some(msg)) => on_msg(a, msg),
|
||||||
|
Ok(None) => break,
|
||||||
|
Err(_) => return, // our own inbox: only on teardown
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// Wait. Control first (attach/teardown must be prompt), then deaths,
|
||||||
|
// then the cluster arms.
|
||||||
|
let mut arms: Vec<&dyn Selectable> = vec![&ctl, &deaths];
|
||||||
|
if let Some(a) = cl.as_ref() {
|
||||||
|
arms.push(&a.events.rx);
|
||||||
|
arms.push(&pg_rx);
|
||||||
|
}
|
||||||
|
let _ = select(&arms);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn attach() -> Option<Attached> {
|
||||||
|
let events = subscribe()?;
|
||||||
|
let (me, incarnation) = local_identity()?;
|
||||||
|
Some(Attached {
|
||||||
|
events,
|
||||||
|
peers: HashMap::new(),
|
||||||
|
me,
|
||||||
|
incarnation,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Fold pending membership events: `NodeUp` ⇒ record + `Sync` that peer;
|
||||||
|
/// `NodeDown` ⇒ sweep every member it announced. `false` when the stream
|
||||||
|
/// has closed.
|
||||||
|
fn drain_events(a: &mut Attached) -> bool {
|
||||||
|
loop {
|
||||||
|
match a.events.rx.try_recv() {
|
||||||
|
Ok(Some(NodeEvent::NodeUp(info))) => {
|
||||||
|
let name = info.name.clone();
|
||||||
|
a.peers.insert(name.clone(), info);
|
||||||
|
// Snapshot under the store lock, then stamp wire pids
|
||||||
|
// outside it (`from_local` marks watchable under the slot's
|
||||||
|
// cold lock — Leaf-on-Leaf nesting is asserted).
|
||||||
|
let local: Vec<(String, Vec<Pid>)> =
|
||||||
|
with_runtime(|inner| inner.process_groups.lock().groups_on(inner.node_id));
|
||||||
|
let groups = local
|
||||||
|
.into_iter()
|
||||||
|
.map(|(g, pids)| (g, pids.into_iter().map(|p| wire(a, p)).collect()))
|
||||||
|
.collect();
|
||||||
|
let msg = PgMsg::Sync {
|
||||||
|
from: a.me.clone(),
|
||||||
|
groups,
|
||||||
|
};
|
||||||
|
let _ = remote::send(RemoteName::new(name, PG_NAME), msg);
|
||||||
|
}
|
||||||
|
Ok(Some(NodeEvent::NodeDown(info))) => {
|
||||||
|
a.peers.remove(&info.name);
|
||||||
|
with_runtime(|inner| {
|
||||||
|
let mut pg = inner.process_groups.lock();
|
||||||
|
pg.remove_where(|m| m.node == info.node);
|
||||||
|
pg.forget_node_name(info.node);
|
||||||
|
});
|
||||||
|
}
|
||||||
|
Ok(None) => return true,
|
||||||
|
Err(_) => return false,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Send `msg` to every up peer. `NotConnected` is ignored: that peer's
|
||||||
|
/// `NodeDown` is on its way and its next `NodeUp` gets a `Sync`.
|
||||||
|
fn broadcast(a: &Attached, msg: PgMsg) {
|
||||||
|
for name in a.peers.keys() {
|
||||||
|
let _ = remote::send(RemoteName::new(name.clone(), PG_NAME), msg.clone());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn on_death(a: Option<&Attached>, pid: Pid) {
|
||||||
|
let evicted = sweep_local_death(pid);
|
||||||
|
if let Some(a) = a {
|
||||||
|
for (group, ms) in evicted {
|
||||||
|
broadcast(
|
||||||
|
a,
|
||||||
|
PgMsg::Leave {
|
||||||
|
group,
|
||||||
|
pid: wire(a, ms.member.pid),
|
||||||
|
},
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn on_joined(a: Option<&Attached>, group: &str, pid: Pid) {
|
||||||
|
let Some(a) = a else { return };
|
||||||
|
// Re-check: a leave/death may have overtaken the announcement.
|
||||||
|
let still = with_runtime(|inner| {
|
||||||
|
let m = member_for(inner, pid);
|
||||||
|
inner.process_groups.lock().contains(group, &m)
|
||||||
|
});
|
||||||
|
if still {
|
||||||
|
broadcast(
|
||||||
|
a,
|
||||||
|
PgMsg::Join {
|
||||||
|
group: group.to_owned(),
|
||||||
|
pid: wire(a, pid),
|
||||||
|
},
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn on_left(a: Option<&Attached>, group: &str, pid: Pid) {
|
||||||
|
let Some(a) = a else { return };
|
||||||
|
broadcast(
|
||||||
|
a,
|
||||||
|
PgMsg::Leave {
|
||||||
|
group: group.to_owned(),
|
||||||
|
pid: wire(a, pid),
|
||||||
|
},
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The wire form of a local member pid, stamped with the identity the
|
||||||
|
/// actor was attached with (marks watchable, like `from_local`).
|
||||||
|
fn wire(a: &Attached, pid: Pid) -> RemotePid<Erased> {
|
||||||
|
RemotePid::from_local_at(pid, a.me.clone(), a.incarnation)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The named origin's `NodeInfo`, if it is up. A second look at the
|
||||||
|
/// membership stream covers a `NodeUp` that landed after this loop
|
||||||
|
/// iteration's drain; anything still unknown is a ghost and is dropped.
|
||||||
|
fn origin(a: &mut Attached, name: &str) -> Option<NodeInfo> {
|
||||||
|
if let Some(i) = a.peers.get(name) {
|
||||||
|
return Some(i.clone());
|
||||||
|
}
|
||||||
|
drain_events(a);
|
||||||
|
a.peers.get(name).cloned()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `origin`, additionally requiring `pid` to be stamped with the origin's
|
||||||
|
/// current incarnation — a pid from a previous life of that node is a ghost.
|
||||||
|
fn origin_of(a: &mut Attached, pid: &RemotePid<Erased>) -> Option<NodeInfo> {
|
||||||
|
origin(a, pid.node()).filter(|i| i.incarnation == pid.incarnation())
|
||||||
|
}
|
||||||
|
|
||||||
|
fn remote_membership(origin: &NodeInfo, pid: &RemotePid<Erased>) -> Membership {
|
||||||
|
Membership {
|
||||||
|
member: Member {
|
||||||
|
node: origin.node,
|
||||||
|
incarnation: origin.incarnation,
|
||||||
|
pid: Pid::new(pid.index(), pid.generation()),
|
||||||
|
},
|
||||||
|
monitor: None,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn on_msg(a: &mut Attached, msg: PgMsg) {
|
||||||
|
match msg {
|
||||||
|
PgMsg::Sync { from, groups } => {
|
||||||
|
let Some(info) = origin(a, &from) else { return };
|
||||||
|
with_runtime(|inner| {
|
||||||
|
let mut pg = inner.process_groups.lock();
|
||||||
|
pg.remove_where(|m| m.node == info.node);
|
||||||
|
pg.set_node_name(info.node, info.name.clone());
|
||||||
|
for (group, pids) in &groups {
|
||||||
|
// Origin-authored: only its own current-incarnation pids.
|
||||||
|
for p in pids
|
||||||
|
.iter()
|
||||||
|
.filter(|p| p.node() == from && p.incarnation() == info.incarnation)
|
||||||
|
{
|
||||||
|
pg.join(group, remote_membership(&info, p));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
PgMsg::Join { group, pid } => {
|
||||||
|
let Some(info) = origin_of(a, &pid) else {
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
with_runtime(|inner| {
|
||||||
|
let mut pg = inner.process_groups.lock();
|
||||||
|
pg.set_node_name(info.node, info.name.clone());
|
||||||
|
pg.join(&group, remote_membership(&info, &pid));
|
||||||
|
});
|
||||||
|
}
|
||||||
|
PgMsg::Leave { group, pid } => {
|
||||||
|
let Some(info) = origin_of(a, &pid) else {
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
with_runtime(|inner| {
|
||||||
|
let ms = remote_membership(&info, &pid);
|
||||||
|
inner.process_groups.lock().leave(&group, ms.member);
|
||||||
|
});
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
use crate::cluster::envelope::{decode_payload, encode_payload};
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn pg_msg_roundtrips_every_variant() {
|
||||||
|
let p = RemotePid::<Erased>::from_parts("a", Incarnation::new(9), 3, 1);
|
||||||
|
for m in [
|
||||||
|
PgMsg::Sync {
|
||||||
|
from: "a".into(),
|
||||||
|
groups: vec![
|
||||||
|
("g".into(), vec![p.clone(), p.clone()]),
|
||||||
|
("h".into(), vec![]),
|
||||||
|
],
|
||||||
|
},
|
||||||
|
PgMsg::Sync {
|
||||||
|
from: "a".into(),
|
||||||
|
groups: vec![],
|
||||||
|
},
|
||||||
|
PgMsg::Join {
|
||||||
|
group: "g".into(),
|
||||||
|
pid: p.clone(),
|
||||||
|
},
|
||||||
|
PgMsg::Leave {
|
||||||
|
group: "g".into(),
|
||||||
|
pid: p.clone(),
|
||||||
|
},
|
||||||
|
] {
|
||||||
|
let bytes = encode_payload(&m).unwrap();
|
||||||
|
let back: PgMsg = decode_payload(&bytes).unwrap();
|
||||||
|
assert_eq!(back, m);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn pg_msg_rejects_unknown_tag() {
|
||||||
|
let bytes = encode_payload(&(7u8, "x", 0u32)).unwrap();
|
||||||
|
assert!(decode_payload::<PgMsg>(&bytes).is_err());
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,835 @@
|
|||||||
|
//! RFC 010 c9 — remote `Name` sends: the outbound path and the single
|
||||||
|
//! inbound name-resolution seam.
|
||||||
|
//!
|
||||||
|
//! ## Outbound (D13, ratified 2026-08-15)
|
||||||
|
//!
|
||||||
|
//! A module-private table `node name → Sender<Frame>` — one dedicated
|
||||||
|
//! outbound channel per live connection, populated and torn down by the
|
||||||
|
//! manager inside the same serialized handlers that own the connection's
|
||||||
|
//! lifetime (Register / Disconnect / reap), living on `RuntimeInner` beside
|
||||||
|
//! the exposure state. [`send`] is one leaf-lock lookup + one channel send:
|
||||||
|
//! no gen_server on the data plane, no published channel anyone holding a
|
||||||
|
//! pid could inject raw frames into. `Ok(())` means **handed to the
|
||||||
|
//! connection's inbox** — local knowledge only, exactly the BEAM contract
|
||||||
|
//! (RFC §3): a missing entry or a closed channel is
|
||||||
|
//! [`RemoteSendError::NotConnected`]; delivery confirmation is the monitor's
|
||||||
|
//! job (c11). The entry-present/actor-dying-mid-send window is *honest*
|
||||||
|
//! under that contract, not a bug.
|
||||||
|
//!
|
||||||
|
//! The outbound sender is deliberately **separate from the conn actor's
|
||||||
|
//! command channel**: if it were a clone of `cmd_tx`, the manager dropping
|
||||||
|
//! its `ConnHandle` would no longer close that channel and connection
|
||||||
|
//! lifetime would leak to whoever holds a sender — a D9 violation.
|
||||||
|
//!
|
||||||
|
//! Buffering is unbounded toward a slow peer (the BEAM `busy_dist_port`
|
||||||
|
//! shape); backpressure is out of c9's scope and noted here rather than
|
||||||
|
//! silently absent.
|
||||||
|
//!
|
||||||
|
//! ## Inbound — the ONE resolution seam (RFC v2)
|
||||||
|
//!
|
||||||
|
//! Every wire-name → local-pid resolution goes through [`deliver_named`],
|
||||||
|
//! and nothing else: the conn actor hands it the three fields of a
|
||||||
|
//! `SendNamed` and gets back a verdict. It checks the exposed set first (an
|
||||||
|
//! unexposed name is unreachable — the gun's safety), then the type hash
|
||||||
|
//! against what the name was exposed with, then resolves the name through
|
||||||
|
//! the registry and delivers via c8's [`decode_deliver`]. When an owned-name
|
||||||
|
//! table lands beside the `&'static str` registry, it slots in here without
|
||||||
|
//! touching call sites. Module privacy enforces the funnel: the exposed and
|
||||||
|
//! outbound tables are `pub(crate)`, and no other module resolves names for
|
||||||
|
//! the wire.
|
||||||
|
//!
|
||||||
|
//! Refusals are silent to the sender by design (§3: send failure reflects
|
||||||
|
//! local knowledge only); they are observable locally as the returned
|
||||||
|
//! [`InboundVerdict`], which the conn actor may log or count.
|
||||||
|
//!
|
||||||
|
//! ## Pids (c10, D14)
|
||||||
|
//!
|
||||||
|
//! [`RemotePid<A>`] = `(node_name, incarnation, index, generation)` +
|
||||||
|
//! phantom — identity-bound, dead when that incarnation dies, never
|
||||||
|
//! redirects. The node travels as its **name** (a global identifier, so a pid
|
||||||
|
//! forwarded through a third node needs no re-mapping); NodeId is a local
|
||||||
|
//! alias and never crosses. A local `Pid<A>` serializes *as* a `RemotePid`
|
||||||
|
//! stamped from the ambient [local identity](set_local_identity); a
|
||||||
|
//! `RemotePid` deserializes into `Pid<A>` only when it names this node (the
|
||||||
|
//! collapse), else it is a decode error — fields that may hold a pid from
|
||||||
|
//! anywhere are typed `RemotePid<A>`.
|
||||||
|
//!
|
||||||
|
//! [`send_to_remote`] is the pid-targeted send. A self-node pid short-
|
||||||
|
//! circuits to the local typed send with the message object itself — no
|
||||||
|
//! encode, no frame (zero-copy-equivalent). Otherwise the outbound table
|
||||||
|
//! (widened to carry each node's **current incarnation**) does the RFC v2 §3
|
||||||
|
//! check at the send site: a pid of a dead incarnation is
|
||||||
|
//! [`ToRemoteError::DeadIncarnation`] and no frame is emitted. Inbound
|
||||||
|
//! `Send` frames are delivered by index/generation through c8's
|
||||||
|
//! [`decode_deliver`]: the target actor's published channel for the exposed
|
||||||
|
//! type is the only route (the reply-to path requires
|
||||||
|
//! [`expose_type`](crate::cluster::expose::expose_type) at the receiver).
|
||||||
|
|
||||||
|
use std::cell::Cell;
|
||||||
|
use std::collections::HashMap;
|
||||||
|
use std::marker::PhantomData;
|
||||||
|
|
||||||
|
use crate::channel::{channel, Receiver, RecvError, Selectable, Sender};
|
||||||
|
use crate::cluster::envelope::{encode_payload, Frame, PayloadError, RemoteDownReason};
|
||||||
|
use crate::cluster::expose::{decode_deliver, exposed_hash, type_hash, DeliverError};
|
||||||
|
use crate::monitor::{demonitor, monitor, Monitor, MonitorId};
|
||||||
|
use crate::pg::Incarnation;
|
||||||
|
use crate::pid::{Addressable, Erased, Name, Pid};
|
||||||
|
use crate::registry::{send_to, whereis, SendError};
|
||||||
|
use crate::scheduler::with_runtime;
|
||||||
|
|
||||||
|
/// A name on a specific remote node: `(node_name, Name<M>)`. Sendable via
|
||||||
|
/// [`send`]; typed, so the payload is `M` and the wire hash is
|
||||||
|
/// [`type_hash::<M>()`](type_hash).
|
||||||
|
pub struct RemoteName<M> {
|
||||||
|
node: String,
|
||||||
|
name: Name<M>,
|
||||||
|
_marker: PhantomData<fn() -> M>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl<M> RemoteName<M> {
|
||||||
|
pub fn new(node: impl Into<String>, name: Name<M>) -> Self {
|
||||||
|
RemoteName {
|
||||||
|
node: node.into(),
|
||||||
|
name,
|
||||||
|
_marker: PhantomData,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
pub fn node(&self) -> &str {
|
||||||
|
&self.node
|
||||||
|
}
|
||||||
|
pub fn name(&self) -> Name<M> {
|
||||||
|
self.name
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl<M> Clone for RemoteName<M> {
|
||||||
|
fn clone(&self) -> Self {
|
||||||
|
RemoteName {
|
||||||
|
node: self.node.clone(),
|
||||||
|
name: self.name,
|
||||||
|
_marker: PhantomData,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl<M> std::fmt::Debug for RemoteName<M> {
|
||||||
|
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||||
|
write!(f, "{}@{}", self.name.as_str(), self.node)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Why a remote send did not leave this node. Local knowledge only.
|
||||||
|
#[derive(Debug)]
|
||||||
|
pub enum RemoteSendError<M> {
|
||||||
|
/// No live connection to that node right now (never connected, or gone
|
||||||
|
/// and not yet re-dialed). The message is handed back.
|
||||||
|
NotConnected(M),
|
||||||
|
/// The payload did not serialize.
|
||||||
|
Encode(M, PayloadError),
|
||||||
|
}
|
||||||
|
|
||||||
|
impl<M> RemoteSendError<M> {
|
||||||
|
pub fn into_inner(self) -> M {
|
||||||
|
match self {
|
||||||
|
RemoteSendError::NotConnected(m) | RemoteSendError::Encode(m, _) => m,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The outbound table, one per runtime (a `RuntimeInner` field): per live
|
||||||
|
/// node, its current incarnation (the RFC v2 §3 send-site check) and the
|
||||||
|
/// connection's dedicated outbound sender. Plus this node's own wire
|
||||||
|
/// identity, which pid serialization stamps.
|
||||||
|
pub(crate) struct Outbound {
|
||||||
|
by_node: HashMap<String, Route>,
|
||||||
|
local: Option<(String, Incarnation)>,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One live connection as the outbound path sees it: the peer's current
|
||||||
|
/// incarnation and the two inboxes of its connection actor — frames (c9)
|
||||||
|
/// and monitor bookkeeping (c12, [`MonCmd`]).
|
||||||
|
pub(crate) struct Route {
|
||||||
|
incarnation: Incarnation,
|
||||||
|
frames: Sender<Frame>,
|
||||||
|
monitors: Sender<MonCmd>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Outbound {
|
||||||
|
pub(crate) fn new() -> Self {
|
||||||
|
Outbound {
|
||||||
|
by_node: HashMap::new(),
|
||||||
|
local: None,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Set this node's wire identity — what serialized pids are stamped with
|
||||||
|
/// and what a `RemotePid` must name to collapse. `cluster::start` sets it;
|
||||||
|
/// exposed for local tests. Must run inside [`run`](crate::run).
|
||||||
|
pub fn set_local_identity(node: &str, incarnation: Incarnation) {
|
||||||
|
with_runtime(|inner| {
|
||||||
|
inner.outbound.lock().local = Some((node.to_string(), incarnation));
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// This node's wire identity, if set. Must run inside [`run`](crate::run).
|
||||||
|
pub fn local_identity() -> Option<(String, Incarnation)> {
|
||||||
|
with_runtime(|inner| inner.outbound.lock().local.clone())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Manager-only: bind `node`'s outbound channels at `incarnation`. Called
|
||||||
|
/// inside `Register`.
|
||||||
|
pub(crate) fn bind_outbound(
|
||||||
|
node: &str,
|
||||||
|
incarnation: Incarnation,
|
||||||
|
frames: Sender<Frame>,
|
||||||
|
monitors: Sender<MonCmd>,
|
||||||
|
) {
|
||||||
|
with_runtime(|inner| {
|
||||||
|
inner.outbound.lock().by_node.insert(
|
||||||
|
node.to_string(),
|
||||||
|
Route {
|
||||||
|
incarnation,
|
||||||
|
frames,
|
||||||
|
monitors,
|
||||||
|
},
|
||||||
|
);
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Test probe: bind an arbitrary sender as `node`'s outbound so a test can
|
||||||
|
/// assert what frames leave — or don't. Same table, same lookup as the real
|
||||||
|
/// path (this is how "no frame emitted" is asserted at the frame level).
|
||||||
|
/// Frames only: there is no connection actor behind a probe, so a
|
||||||
|
/// [`monitor_remote`] against a probed node reports `Disconnected`.
|
||||||
|
pub fn bind_outbound_probe(node: &str, incarnation: Incarnation, tx: Sender<Frame>) {
|
||||||
|
drop(bind_outbound_probe_with_monitors(node, incarnation, tx));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The monitor half of a probed node's inbox: opaque, held only to be
|
||||||
|
/// dropped. See [`bind_outbound_probe_with_monitors`].
|
||||||
|
pub struct MonitorInbox {
|
||||||
|
_rx: Receiver<MonCmd>,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Test probe: like [`bind_outbound_probe`], but the monitor-command
|
||||||
|
/// receiver is handed back instead of dropped, so a test can stage the
|
||||||
|
/// c13 drain gap — a `Monitor` command that reached the connection's inbox
|
||||||
|
/// and dies unread when the inbox is dropped. While the inbox lives,
|
||||||
|
/// [`monitor_remote`] against the probed node is simply in flight.
|
||||||
|
pub fn bind_outbound_probe_with_monitors(
|
||||||
|
node: &str,
|
||||||
|
incarnation: Incarnation,
|
||||||
|
tx: Sender<Frame>,
|
||||||
|
) -> MonitorInbox {
|
||||||
|
let (mon_tx, mon_rx) = channel();
|
||||||
|
bind_outbound(node, incarnation, tx, mon_tx);
|
||||||
|
MonitorInbox { _rx: mon_rx }
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Manager-only: unbind `node`'s outbound channel. Called on `Disconnect`,
|
||||||
|
/// reap, and manager shutdown. Dropping the sender is what closes the conn
|
||||||
|
/// actor's outbound arm — but that arm's closure is NOT a stop signal (the
|
||||||
|
/// cmd channel is, per D9); the actor simply stops selecting on it.
|
||||||
|
pub(crate) fn unbind_outbound(node: &str) {
|
||||||
|
with_runtime(|inner| {
|
||||||
|
inner.outbound.lock().by_node.remove(node);
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Send `msg` to `target`. `Ok(())` = handed to the connection's inbox, and
|
||||||
|
/// nothing more — see the module docs. Must run inside
|
||||||
|
/// [`run`](crate::run).
|
||||||
|
pub fn send<M>(target: RemoteName<M>, msg: M) -> Result<(), RemoteSendError<M>>
|
||||||
|
where
|
||||||
|
M: serde::Serialize + Send + 'static,
|
||||||
|
{
|
||||||
|
let payload = match encode_payload(&msg) {
|
||||||
|
Ok(p) => p,
|
||||||
|
Err(e) => return Err(RemoteSendError::Encode(msg, e)),
|
||||||
|
};
|
||||||
|
let frame = Frame::SendNamed {
|
||||||
|
name: target.name.as_str().to_string(),
|
||||||
|
type_hash: type_hash::<M>(),
|
||||||
|
payload,
|
||||||
|
};
|
||||||
|
match hand_to_connection(&target.node, frame) {
|
||||||
|
Ok(()) => Ok(()),
|
||||||
|
Err(NotConnected) => Err(RemoteSendError::NotConnected(msg)),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// No live connection to the named node — the payload-free form of
|
||||||
|
/// [`RemoteSendError::NotConnected`], for the raw path.
|
||||||
|
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||||
|
pub struct NotConnected;
|
||||||
|
|
||||||
|
/// The untyped escape hatch: send pre-encoded `payload` under an explicit
|
||||||
|
/// `type_hash`. Exists so tests (and future codecs) can put deliberately
|
||||||
|
/// wrong frames on the wire; the typed [`send`] cannot express a hash/type
|
||||||
|
/// mismatch, by design. Same `Ok` semantics as [`send`].
|
||||||
|
pub fn send_remote_raw(
|
||||||
|
node: &str,
|
||||||
|
name: &str,
|
||||||
|
type_hash: u64,
|
||||||
|
payload: &[u8],
|
||||||
|
) -> Result<(), NotConnected> {
|
||||||
|
hand_to_connection(
|
||||||
|
node,
|
||||||
|
Frame::SendNamed {
|
||||||
|
name: name.to_string(),
|
||||||
|
type_hash,
|
||||||
|
payload: payload.to_vec(),
|
||||||
|
},
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One lookup, one send. Clone the sender out under the lock and send
|
||||||
|
/// outside it (a channel send can unpark the conn actor).
|
||||||
|
fn hand_to_connection(node: &str, frame: Frame) -> Result<(), NotConnected> {
|
||||||
|
let tx = with_runtime(|inner| {
|
||||||
|
inner
|
||||||
|
.outbound
|
||||||
|
.lock()
|
||||||
|
.by_node
|
||||||
|
.get(node)
|
||||||
|
.map(|r| r.frames.clone())
|
||||||
|
});
|
||||||
|
match tx {
|
||||||
|
Some(tx) => tx.send(frame).map_err(|_| NotConnected),
|
||||||
|
None => Err(NotConnected),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---- pids ---------------------------------------------------------------
|
||||||
|
|
||||||
|
/// A pid on some node: `(node_name, incarnation, index, generation)` plus
|
||||||
|
/// the actor type. See the module docs. Serializes as a 4-tuple.
|
||||||
|
pub struct RemotePid<A> {
|
||||||
|
node: String,
|
||||||
|
incarnation: Incarnation,
|
||||||
|
index: u32,
|
||||||
|
generation: u32,
|
||||||
|
_marker: PhantomData<fn() -> A>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl<A> RemotePid<A> {
|
||||||
|
/// Build from raw parts (tests, and codecs re-hydrating a pid).
|
||||||
|
pub fn from_parts(
|
||||||
|
node: impl Into<String>,
|
||||||
|
incarnation: Incarnation,
|
||||||
|
index: u32,
|
||||||
|
generation: u32,
|
||||||
|
) -> Self {
|
||||||
|
RemotePid {
|
||||||
|
node: node.into(),
|
||||||
|
incarnation,
|
||||||
|
index,
|
||||||
|
generation,
|
||||||
|
_marker: PhantomData,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The wire form of a local pid, stamped with this node's identity, and
|
||||||
|
/// **marked watchable** — asking for the wire form *is* the intent to
|
||||||
|
/// ship the pid, so this is the same D12 set-site as `Pid::serialize`
|
||||||
|
/// (c12 made it explicit: a peer may monitor exactly the pids that
|
||||||
|
/// crossed, and a pid handed out via `from_local` in a hand-built reply
|
||||||
|
/// has crossed). Must run inside [`run`](crate::run).
|
||||||
|
///
|
||||||
|
/// `None` when this runtime has no wire identity (no `cluster::start`,
|
||||||
|
/// no [`set_local_identity`]): such a pid cannot name a node, and a
|
||||||
|
/// stamped `("", 0)` would be dropped by every peer with no signal.
|
||||||
|
/// The pid is not marked watchable in that case either.
|
||||||
|
pub fn from_local(pid: Pid<A>) -> Option<Self> {
|
||||||
|
let (node, incarnation) = local_identity()?;
|
||||||
|
Some(Self::from_local_at(pid, node, incarnation))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `from_local` with the identity supplied by the caller — for a holder
|
||||||
|
/// that already carries the node's identity (the pg actor) and must not
|
||||||
|
/// have a `None` path. Marks watchable like `from_local`.
|
||||||
|
pub(crate) fn from_local_at(pid: Pid<A>, node: String, incarnation: Incarnation) -> Self {
|
||||||
|
crate::monitor::mark_watchable(pid);
|
||||||
|
RemotePid::from_parts(node, incarnation, pid.index(), pid.generation())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The collapse: `Some(local pid)` iff this pid names this very node
|
||||||
|
/// (name and incarnation). Must run inside [`run`](crate::run).
|
||||||
|
pub fn local(&self) -> Option<Pid<A>> {
|
||||||
|
let (n, i) = local_identity()?;
|
||||||
|
(n == self.node && i == self.incarnation)
|
||||||
|
.then(|| crate::pid::assert_type::<A>(Pid::new(self.index, self.generation)))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Drop the actor type: the untyped `RemotePid<Erased>`, the form
|
||||||
|
/// [`RemoteDown`] and [`RemoteMonitor`] carry (mirrors [`Pid::erase`]).
|
||||||
|
pub fn erase(self) -> RemotePid<Erased> {
|
||||||
|
RemotePid::from_parts(self.node, self.incarnation, self.index, self.generation)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Re-type an erased pid as `RemotePid<B>` — the unchecked mirror of
|
||||||
|
/// `pid::assert_type`, with the same degradation: a wrong `B` means the
|
||||||
|
/// target refuses the payload's hash (never a misroute).
|
||||||
|
pub(crate) fn assert_type<B>(self) -> RemotePid<B> {
|
||||||
|
RemotePid::from_parts(self.node, self.incarnation, self.index, self.generation)
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn node(&self) -> &str {
|
||||||
|
&self.node
|
||||||
|
}
|
||||||
|
pub fn incarnation(&self) -> Incarnation {
|
||||||
|
self.incarnation
|
||||||
|
}
|
||||||
|
pub fn index(&self) -> u32 {
|
||||||
|
self.index
|
||||||
|
}
|
||||||
|
pub fn generation(&self) -> u32 {
|
||||||
|
self.generation
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl<A> Clone for RemotePid<A> {
|
||||||
|
fn clone(&self) -> Self {
|
||||||
|
RemotePid::from_parts(
|
||||||
|
self.node.clone(),
|
||||||
|
self.incarnation,
|
||||||
|
self.index,
|
||||||
|
self.generation,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
impl<A> PartialEq for RemotePid<A> {
|
||||||
|
fn eq(&self, o: &Self) -> bool {
|
||||||
|
self.node == o.node
|
||||||
|
&& self.incarnation == o.incarnation
|
||||||
|
&& self.index == o.index
|
||||||
|
&& self.generation == o.generation
|
||||||
|
}
|
||||||
|
}
|
||||||
|
impl<A> Eq for RemotePid<A> {}
|
||||||
|
impl<A> std::fmt::Debug for RemotePid<A> {
|
||||||
|
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||||
|
write!(
|
||||||
|
f,
|
||||||
|
"<{}.{}@{}#{}>",
|
||||||
|
self.index,
|
||||||
|
self.generation,
|
||||||
|
self.node,
|
||||||
|
self.incarnation.get()
|
||||||
|
)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl<A> serde::Serialize for RemotePid<A> {
|
||||||
|
fn serialize<S: serde::Serializer>(&self, s: S) -> Result<S::Ok, S::Error> {
|
||||||
|
(
|
||||||
|
self.node.as_str(),
|
||||||
|
self.incarnation.get(),
|
||||||
|
self.index,
|
||||||
|
self.generation,
|
||||||
|
)
|
||||||
|
.serialize(s)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
impl<'de, A> serde::Deserialize<'de> for RemotePid<A> {
|
||||||
|
fn deserialize<D: serde::Deserializer<'de>>(d: D) -> Result<Self, D::Error> {
|
||||||
|
let (node, inc, index, generation) = <(String, u32, u32, u32)>::deserialize(d)?;
|
||||||
|
Ok(RemotePid::from_parts(
|
||||||
|
node,
|
||||||
|
Incarnation::new(inc),
|
||||||
|
index,
|
||||||
|
generation,
|
||||||
|
))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Why a pid-targeted send did not leave this node. Local knowledge only.
|
||||||
|
#[derive(Debug)]
|
||||||
|
pub enum ToRemoteError<M> {
|
||||||
|
/// No live connection to the pid's node.
|
||||||
|
NotConnected(M),
|
||||||
|
/// The pid's incarnation is not that node's current one (RFC v2 §3): the
|
||||||
|
/// actor died with its incarnation. Detected at the send site; no frame.
|
||||||
|
DeadIncarnation(M),
|
||||||
|
/// The payload did not serialize.
|
||||||
|
Encode(M, PayloadError),
|
||||||
|
/// The pid collapsed to a local one and the local typed send failed.
|
||||||
|
Local(SendError<M>),
|
||||||
|
}
|
||||||
|
|
||||||
|
impl<M> ToRemoteError<M> {
|
||||||
|
/// The undelivered message.
|
||||||
|
pub fn into_inner(self) -> M {
|
||||||
|
match self {
|
||||||
|
ToRemoteError::NotConnected(m)
|
||||||
|
| ToRemoteError::DeadIncarnation(m)
|
||||||
|
| ToRemoteError::Encode(m, _) => m,
|
||||||
|
ToRemoteError::Local(e) => e.into_inner(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Send `msg` to a pid, wherever it lives. Self-node pids short-circuit to
|
||||||
|
/// the local typed send with `msg` itself (no encode, no frame); others go
|
||||||
|
/// out as a `Send` frame after the incarnation check. `Ok(())` for a remote
|
||||||
|
/// target = handed to the connection's inbox. Must run inside
|
||||||
|
/// [`run`](crate::run).
|
||||||
|
pub fn send_to_remote<A>(target: RemotePid<A>, msg: A::Msg) -> Result<(), ToRemoteError<A::Msg>>
|
||||||
|
where
|
||||||
|
A: Addressable,
|
||||||
|
A::Msg: serde::Serialize,
|
||||||
|
{
|
||||||
|
if let Some(local) = target.local() {
|
||||||
|
return send_to(local, msg).map_err(ToRemoteError::Local);
|
||||||
|
}
|
||||||
|
let route = with_runtime(|inner| {
|
||||||
|
inner
|
||||||
|
.outbound
|
||||||
|
.lock()
|
||||||
|
.by_node
|
||||||
|
.get(&target.node)
|
||||||
|
.map(|r| (r.incarnation, r.frames.clone()))
|
||||||
|
});
|
||||||
|
let (current, tx) = match route {
|
||||||
|
Some(r) => r,
|
||||||
|
None => return Err(ToRemoteError::NotConnected(msg)),
|
||||||
|
};
|
||||||
|
if current != target.incarnation {
|
||||||
|
return Err(ToRemoteError::DeadIncarnation(msg));
|
||||||
|
}
|
||||||
|
let payload = match encode_payload(&msg) {
|
||||||
|
Ok(p) => p,
|
||||||
|
Err(e) => return Err(ToRemoteError::Encode(msg, e)),
|
||||||
|
};
|
||||||
|
let frame = Frame::Send {
|
||||||
|
index: target.index,
|
||||||
|
generation: target.generation,
|
||||||
|
type_hash: type_hash::<A::Msg>(),
|
||||||
|
payload,
|
||||||
|
};
|
||||||
|
tx.send(frame).map_err(|_| ToRemoteError::NotConnected(msg))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The inbound `Send` seam: deliver `payload` under `type_hash` to the local
|
||||||
|
/// actor `(index, generation)`. Node and incarnation are implicit in the
|
||||||
|
/// connection (bound at handshake) — the frame carries only the slot
|
||||||
|
/// identity. Delivery goes through c8's decoder table, so only types the
|
||||||
|
/// receiver has [`expose_type`](crate::cluster::expose::expose_type)d (or
|
||||||
|
/// exposed by name) can land; anything else is refused, never misrouted.
|
||||||
|
pub fn deliver_to_pid(
|
||||||
|
index: u32,
|
||||||
|
generation: u32,
|
||||||
|
type_hash: u64,
|
||||||
|
payload: &[u8],
|
||||||
|
) -> InboundVerdict {
|
||||||
|
let pid = Pid::new(index, generation);
|
||||||
|
match decode_deliver(type_hash, pid, payload) {
|
||||||
|
Ok(()) => InboundVerdict::Delivered,
|
||||||
|
Err(e) => InboundVerdict::Refused(e),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// What the inbound seam did with a `SendNamed`. Local observability only;
|
||||||
|
/// nothing goes back on the wire (RFC §3).
|
||||||
|
#[derive(Debug)]
|
||||||
|
pub enum InboundVerdict {
|
||||||
|
/// Decoded and handed to the name's holder.
|
||||||
|
Delivered,
|
||||||
|
/// The name is not in this node's exposed set.
|
||||||
|
NotExposed,
|
||||||
|
/// The frame's hash is not the hash the name was exposed with.
|
||||||
|
HashMismatch { expected: u64, got: u64 },
|
||||||
|
/// Exposed, but no live holder right now (unbound, or its holder died
|
||||||
|
/// and the binding is being pruned).
|
||||||
|
Unresolved,
|
||||||
|
/// Resolved, but the delivery half refused it (decode failure, or the
|
||||||
|
/// holder's channel does not accept the exposed type — a local
|
||||||
|
/// re-registration under a different type; never a misroute).
|
||||||
|
Refused(DeliverError),
|
||||||
|
}
|
||||||
|
|
||||||
|
impl InboundVerdict {
|
||||||
|
/// A short static label for tracing/counting (`smarm-trace` records one
|
||||||
|
/// `ClusterInbound` event per frame with it).
|
||||||
|
pub fn label(&self) -> &'static str {
|
||||||
|
match self {
|
||||||
|
InboundVerdict::Delivered => "delivered",
|
||||||
|
InboundVerdict::NotExposed => "not_exposed",
|
||||||
|
InboundVerdict::HashMismatch { .. } => "hash_mismatch",
|
||||||
|
InboundVerdict::Unresolved => "unresolved",
|
||||||
|
InboundVerdict::Refused(_) => "refused",
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// THE inbound resolution seam: exposed-set check → hash check → registry
|
||||||
|
/// resolution → c8 delivery. See the module docs. Must run inside
|
||||||
|
/// [`run`](crate::run) — the conn actor's context.
|
||||||
|
pub fn deliver_named(name: &str, type_hash: u64, payload: &[u8]) -> InboundVerdict {
|
||||||
|
let Some(expected) = exposed_hash(name) else {
|
||||||
|
return InboundVerdict::NotExposed;
|
||||||
|
};
|
||||||
|
if expected != type_hash {
|
||||||
|
return InboundVerdict::HashMismatch {
|
||||||
|
expected,
|
||||||
|
got: type_hash,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
let Some(pid) = whereis(name) else {
|
||||||
|
return InboundVerdict::Unresolved;
|
||||||
|
};
|
||||||
|
match decode_deliver(type_hash, pid, payload) {
|
||||||
|
Ok(()) => InboundVerdict::Delivered,
|
||||||
|
Err(e) => InboundVerdict::Refused(e),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---- monitors (c12) -----------------------------------------------------
|
||||||
|
|
||||||
|
/// Bookkeeping commands from [`monitor_remote`]/[`demonitor_remote`] to the
|
||||||
|
/// connection actor that owns the link to the target's node. The actor
|
||||||
|
/// records the registration and *then* emits the `Monitor` frame itself, so
|
||||||
|
/// a `Down` can never arrive at a table that does not yet know the id. It
|
||||||
|
/// lives in the actor (not on `RuntimeInner`) so the bookkeeping dies with
|
||||||
|
/// the connection — exactly what c13 needs to synthesize `Disconnected`.
|
||||||
|
pub(crate) enum MonCmd {
|
||||||
|
Monitor {
|
||||||
|
id: MonitorId,
|
||||||
|
target: RemotePid<Erased>,
|
||||||
|
tx: Sender<RemoteDown>,
|
||||||
|
},
|
||||||
|
Demonitor {
|
||||||
|
id: MonitorId,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A remotely-monitored actor's termination notice — the cluster analog of
|
||||||
|
/// [`Down`](crate::monitor::Down), with the pid in its wire form because it
|
||||||
|
/// may name any node.
|
||||||
|
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||||
|
pub struct RemoteDown {
|
||||||
|
/// The pid that was being monitored.
|
||||||
|
pub pid: RemotePid<Erased>,
|
||||||
|
/// How it went down. `Disconnected` means the *link* to its node was
|
||||||
|
/// lost (or absent) — nothing is known about the actor itself.
|
||||||
|
pub reason: RemoteDownReason,
|
||||||
|
}
|
||||||
|
|
||||||
|
enum Watch {
|
||||||
|
/// The target collapsed to this node: an ordinary local monitor,
|
||||||
|
/// translated on read.
|
||||||
|
Local(Monitor),
|
||||||
|
/// The target is elsewhere: the connection actor for its node holds the
|
||||||
|
/// registration and forwards the peer's `Down` frame here. The
|
||||||
|
/// `RemoteState` is the read-side backstop (c13): a channel that closes
|
||||||
|
/// while `Live` — the connection died with our command unread — reads
|
||||||
|
/// as `Disconnected` once; afterwards, and after a cancel, closed is
|
||||||
|
/// just closed.
|
||||||
|
Remote(Receiver<RemoteDown>, Cell<RemoteState>),
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Where a remote-watch stands from the reader's side.
|
||||||
|
#[derive(Clone, Copy, PartialEq, Eq)]
|
||||||
|
enum RemoteState {
|
||||||
|
/// No notice yet, not cancelled: a closed channel means `Disconnected`.
|
||||||
|
Live,
|
||||||
|
/// The one notice has been read (or synthesized): nothing more is due.
|
||||||
|
Done,
|
||||||
|
/// `demonitor_remote` ran: never synthesize.
|
||||||
|
Cancelled,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A live remote monitor: read its one [`RemoteDown`] with
|
||||||
|
/// [`recv`](RemoteMonitor::recv)/[`try_recv`](RemoteMonitor::try_recv), or
|
||||||
|
/// fold it into a `select` via [`arm`](RemoteMonitor::arm). Distinct from
|
||||||
|
/// [`Monitor`] on purpose: its target is a [`RemotePid`], its notice a
|
||||||
|
/// [`RemoteDown`], and it can report `Disconnected` — none of which a local
|
||||||
|
/// monitor can express. Dropping it discards an unread notice, like the
|
||||||
|
/// local one; after [`demonitor_remote`] the channel is closed and empty, so
|
||||||
|
/// `recv` errs rather than parking — also like the local one.
|
||||||
|
///
|
||||||
|
/// Exactly one notice is guaranteed even if the connection actor dies with
|
||||||
|
/// the registration unread (the c13 drain gap): a channel that closes
|
||||||
|
/// before any notice — and before any cancel — reads as `Disconnected`,
|
||||||
|
/// once. The next read is the ordinary closed-channel `Err`.
|
||||||
|
pub struct RemoteMonitor {
|
||||||
|
/// This registration's process-unique id — minted here, echoed by the
|
||||||
|
/// peer in its `Down` frame.
|
||||||
|
pub id: MonitorId,
|
||||||
|
/// The pid being monitored.
|
||||||
|
pub target: RemotePid<Erased>,
|
||||||
|
watch: Watch,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl RemoteMonitor {
|
||||||
|
/// Block (cooperatively) for the notice.
|
||||||
|
pub fn recv(&self) -> Result<RemoteDown, RecvError> {
|
||||||
|
match &self.watch {
|
||||||
|
Watch::Local(m) => m.rx.recv().map(|d| RemoteDown {
|
||||||
|
pid: self.target.clone(),
|
||||||
|
reason: d.reason.into(),
|
||||||
|
}),
|
||||||
|
Watch::Remote(rx, st) => match rx.recv() {
|
||||||
|
Ok(d) => {
|
||||||
|
st.set(RemoteState::Done);
|
||||||
|
Ok(d)
|
||||||
|
}
|
||||||
|
Err(e) => self.closed(st).ok_or(e),
|
||||||
|
},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The notice if it has arrived; `Ok(None)` if not yet.
|
||||||
|
pub fn try_recv(&self) -> Result<Option<RemoteDown>, RecvError> {
|
||||||
|
match &self.watch {
|
||||||
|
Watch::Local(m) => m.rx.try_recv().map(|o| {
|
||||||
|
o.map(|d| RemoteDown {
|
||||||
|
pid: self.target.clone(),
|
||||||
|
reason: d.reason.into(),
|
||||||
|
})
|
||||||
|
}),
|
||||||
|
Watch::Remote(rx, st) => match rx.try_recv() {
|
||||||
|
Ok(Some(d)) => {
|
||||||
|
st.set(RemoteState::Done);
|
||||||
|
Ok(Some(d))
|
||||||
|
}
|
||||||
|
Ok(None) => Ok(None),
|
||||||
|
Err(e) => self.closed(st).map(Some).ok_or(e),
|
||||||
|
},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The channel closed. While `Live` — no notice yet, no cancel — that
|
||||||
|
/// is the connection having died with our registration unread, so
|
||||||
|
/// synthesize the one `Disconnected` and mark `Done`; otherwise closed
|
||||||
|
/// is just closed.
|
||||||
|
fn closed(&self, st: &Cell<RemoteState>) -> Option<RemoteDown> {
|
||||||
|
if st.get() != RemoteState::Live {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
st.set(RemoteState::Done);
|
||||||
|
Some(RemoteDown {
|
||||||
|
pid: self.target.clone(),
|
||||||
|
reason: RemoteDownReason::Disconnected,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The selectable arm: readiness means [`try_recv`](Self::try_recv)
|
||||||
|
/// will yield the notice.
|
||||||
|
pub fn arm(&self) -> &dyn Selectable {
|
||||||
|
match &self.watch {
|
||||||
|
Watch::Local(m) => &m.rx,
|
||||||
|
Watch::Remote(rx, _) => rx,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl std::fmt::Debug for RemoteMonitor {
|
||||||
|
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||||
|
f.debug_struct("RemoteMonitor")
|
||||||
|
.field("id", &self.id)
|
||||||
|
.field("target", &self.target)
|
||||||
|
.finish_non_exhaustive()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Monitor `target`, wherever it lives. Exactly one [`RemoteDown`] arrives:
|
||||||
|
///
|
||||||
|
/// - self-node pid ⇒ an ordinary local monitor underneath (same NoProc rule);
|
||||||
|
/// - no live connection to the pid's node ⇒ `Disconnected`, queued at once
|
||||||
|
/// (the remote analog of NoProc: nothing can be known);
|
||||||
|
/// - the pid's incarnation is not the node's current one ⇒ `NoProc`, queued
|
||||||
|
/// at once — the node is *known* to have restarted, so its actor is a
|
||||||
|
/// corpse, not a partition (RFC v2 §3);
|
||||||
|
/// - otherwise the connection actor registers the id and sends `Monitor`;
|
||||||
|
/// the peer answers with the true terminal reason on exit, or immediately
|
||||||
|
/// with the recorded reason for a corpse (`terminal_reason`, RFC §6) or
|
||||||
|
/// `NoProc` for a pid it never exposed and never shipped.
|
||||||
|
///
|
||||||
|
/// The connection dropping while the monitor is outstanding delivers
|
||||||
|
/// `Disconnected` (c13): the connection actor synthesizes it on teardown,
|
||||||
|
/// and the monitor's own read path backstops the case where the actor died
|
||||||
|
/// with the registration still unread. Must run inside [`run`](crate::run).
|
||||||
|
pub fn monitor_remote<A>(target: RemotePid<A>) -> RemoteMonitor {
|
||||||
|
if let Some(local) = target.local() {
|
||||||
|
let m = monitor(local);
|
||||||
|
return RemoteMonitor {
|
||||||
|
id: m.id,
|
||||||
|
target: target.erase(),
|
||||||
|
watch: Watch::Local(m),
|
||||||
|
};
|
||||||
|
}
|
||||||
|
let target = target.erase();
|
||||||
|
let (id, route) = with_runtime(|inner| {
|
||||||
|
let id = inner.alloc_monitor_id();
|
||||||
|
let route = inner
|
||||||
|
.outbound
|
||||||
|
.lock()
|
||||||
|
.by_node
|
||||||
|
.get(&target.node)
|
||||||
|
.map(|r| (r.incarnation, r.monitors.clone()));
|
||||||
|
(id, route)
|
||||||
|
});
|
||||||
|
let (tx, rx) = channel::<RemoteDown>();
|
||||||
|
let immediate = match route {
|
||||||
|
None => Some(RemoteDownReason::Disconnected),
|
||||||
|
Some((current, _)) if current != target.incarnation => {
|
||||||
|
Some(RemoteDownReason::Local(crate::monitor::DownReason::NoProc))
|
||||||
|
}
|
||||||
|
Some((_, mon_tx)) => {
|
||||||
|
let cmd = MonCmd::Monitor {
|
||||||
|
id,
|
||||||
|
target: target.clone(),
|
||||||
|
tx: tx.clone(),
|
||||||
|
};
|
||||||
|
match mon_tx.send(cmd) {
|
||||||
|
Ok(()) => None,
|
||||||
|
Err(_) => Some(RemoteDownReason::Disconnected), // actor already gone
|
||||||
|
}
|
||||||
|
}
|
||||||
|
};
|
||||||
|
if let Some(reason) = immediate {
|
||||||
|
let _ = tx.send(RemoteDown {
|
||||||
|
pid: target.clone(),
|
||||||
|
reason,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
RemoteMonitor {
|
||||||
|
id,
|
||||||
|
target,
|
||||||
|
watch: Watch::Remote(rx, Cell::new(RemoteState::Live)),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Cancel `m`. No future notice will be *sent* for it; a notice already in
|
||||||
|
/// flight from the peer is dropped on arrival, and one already sitting in
|
||||||
|
/// `m` is discarded when `m` is dropped (same contract as
|
||||||
|
/// [`demonitor`]). Unlike the local form this returns nothing: the
|
||||||
|
/// registration is owned by the connection actor, so whether the `Down`
|
||||||
|
/// beat the cancel is not local knowledge. Must run inside
|
||||||
|
/// [`run`](crate::run).
|
||||||
|
pub fn demonitor_remote(m: &RemoteMonitor) {
|
||||||
|
match &m.watch {
|
||||||
|
Watch::Local(local) => {
|
||||||
|
let _ = demonitor(local);
|
||||||
|
}
|
||||||
|
Watch::Remote(_, st) => {
|
||||||
|
// Cancel first: a channel closing after this is closed, not a
|
||||||
|
// Disconnected notice — the caller asked for silence.
|
||||||
|
st.set(RemoteState::Cancelled);
|
||||||
|
let mon_tx = with_runtime(|inner| {
|
||||||
|
inner
|
||||||
|
.outbound
|
||||||
|
.lock()
|
||||||
|
.by_node
|
||||||
|
.get(&m.target.node)
|
||||||
|
.map(|r| r.monitors.clone())
|
||||||
|
});
|
||||||
|
if let Some(mon_tx) = mon_tx {
|
||||||
|
let _ = mon_tx.send(MonCmd::Demonitor { id: m.id });
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,298 @@
|
|||||||
|
//! RFC 010 c3 — transport abstraction for the **control** connection.
|
||||||
|
//!
|
||||||
|
//! Scope, per RFC 010 v2 §5 and D2:
|
||||||
|
//!
|
||||||
|
//! - A "connection" here is the *control* connection: the one carrying this
|
||||||
|
//! RFC's frame inventory ([`crate::cluster::envelope::Frame`]), whose
|
||||||
|
//! heartbeats feed failure detection. The trait deliberately says nothing
|
||||||
|
//! about how many connections a peer pair may hold — the jarred rkyv bulk
|
||||||
|
//! plane opens **additional per-peer connections** outside this trait, and
|
||||||
|
//! nothing here may foreclose that.
|
||||||
|
//! - Homogeneous smarm⇄smarm only. The BEAM membrane is *not* a transport
|
||||||
|
//! impl and the trait does not accommodate it (D2).
|
||||||
|
//! - Addresses are opaque, **pre-resolved** strings. Name resolution is a
|
||||||
|
//! single separate seam (roadmap c9); impls reject unresolved names rather
|
||||||
|
//! than resolving them.
|
||||||
|
//!
|
||||||
|
//! Blocking model: [`Conn`] calls block the caller. The TCP impl parks the
|
||||||
|
//! calling *actor* (fd readiness via the scheduler); the loopback impl blocks
|
||||||
|
//! the calling *OS thread* and is a test transport — do not drive it from a
|
||||||
|
//! scheduler thread.
|
||||||
|
//!
|
||||||
|
//! Framing is not part of the trait: [`FramedConn`] is the single shared
|
||||||
|
//! codec that turns any byte-stream [`Conn`] into a frame pipe, feeding
|
||||||
|
//! [`Frame::decode`]'s incremental contract. Impls never re-implement
|
||||||
|
//! framing, and the conformance suite exercises the same codec over every
|
||||||
|
//! impl.
|
||||||
|
|
||||||
|
use std::io;
|
||||||
|
|
||||||
|
use crate::cluster::envelope::{DecodeError, EncodeError, Frame};
|
||||||
|
|
||||||
|
pub mod loopback;
|
||||||
|
pub mod tcp;
|
||||||
|
|
||||||
|
/// An established control connection: a bidirectional byte stream.
|
||||||
|
pub trait Conn: Send {
|
||||||
|
/// Read at least one byte, blocking the caller until data is available,
|
||||||
|
/// EOF, or error. `Ok(0)` means EOF: the peer closed and all bytes it
|
||||||
|
/// wrote before closing have been consumed.
|
||||||
|
fn read(&mut self, buf: &mut [u8]) -> io::Result<usize>;
|
||||||
|
|
||||||
|
/// Write the whole buffer, blocking the caller as needed.
|
||||||
|
fn write_all(&mut self, buf: &[u8]) -> io::Result<()>;
|
||||||
|
|
||||||
|
/// Close both directions. Idempotent. Bytes already written remain
|
||||||
|
/// readable at the peer, which then observes EOF; peer writes after this
|
||||||
|
/// fail.
|
||||||
|
fn close(&mut self);
|
||||||
|
|
||||||
|
/// Diagnostic label for logs only. Mesh identity comes from the
|
||||||
|
/// handshake (`Hello`/`HelloAck`), never from the transport.
|
||||||
|
fn peer_addr(&self) -> String;
|
||||||
|
|
||||||
|
/// Readiness as a [`select`](crate::select) arm, for transports backed by
|
||||||
|
/// a file descriptor. `Some` lets a driver wait on "this connection is
|
||||||
|
/// readable" alongside an ordinary command inbox in a single `select`, so
|
||||||
|
/// one actor can interleave reading with control messages without a
|
||||||
|
/// second thread. The default is `None`: a transport with no fd (the
|
||||||
|
/// in-memory loopback) cannot be selected on and must be driven another
|
||||||
|
/// way.
|
||||||
|
fn readable_arm(&self) -> Option<crate::scheduler::FdArm> {
|
||||||
|
None
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A bound listen point producing inbound [`Conn`]s.
|
||||||
|
pub trait Listener: Send {
|
||||||
|
/// Accept the next inbound connection, blocking the caller.
|
||||||
|
fn accept(&mut self) -> io::Result<Box<dyn Conn>>;
|
||||||
|
|
||||||
|
/// The concrete bound address, dialable as-is (e.g. the real port when
|
||||||
|
/// bound with port 0).
|
||||||
|
fn local_addr(&self) -> String;
|
||||||
|
|
||||||
|
/// Readiness as a [`select`](crate::select) arm, mirroring
|
||||||
|
/// [`Conn::readable_arm`]: `Some` lets an acceptor wait on "an inbound
|
||||||
|
/// connection is pending" alongside a command inbox in one `select`, so
|
||||||
|
/// it can be told to stop without a poll loop. Default `None` (the
|
||||||
|
/// loopback listener has no fd and must be driven synchronously).
|
||||||
|
fn readable_arm(&self) -> Option<crate::scheduler::FdArm> {
|
||||||
|
None
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A way of establishing control connections. Object-safe on purpose: the
|
||||||
|
/// connector and membership layers hold `&dyn Transport` / boxed conns
|
||||||
|
/// rather than growing a generic parameter.
|
||||||
|
pub trait Transport: Send + Sync {
|
||||||
|
/// Connect to a peer's listen address. Blocks the caller until
|
||||||
|
/// established or failed.
|
||||||
|
fn dial(&self, addr: &str) -> io::Result<Box<dyn Conn>>;
|
||||||
|
|
||||||
|
/// Bind a listen point.
|
||||||
|
fn listen(&self, addr: &str) -> io::Result<Box<dyn Listener>>;
|
||||||
|
}
|
||||||
|
|
||||||
|
impl std::fmt::Debug for dyn Conn {
|
||||||
|
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||||
|
write!(f, "Conn({})", self.peer_addr())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl std::fmt::Debug for dyn Listener {
|
||||||
|
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||||
|
write!(f, "Listener({})", self.local_addr())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Error surface of [`FramedConn::send`].
|
||||||
|
#[derive(Debug)]
|
||||||
|
pub enum SendError {
|
||||||
|
/// The frame could not be encoded (e.g. a field over its wire limit).
|
||||||
|
Encode(EncodeError),
|
||||||
|
/// The transport failed mid-write.
|
||||||
|
Io(io::Error),
|
||||||
|
}
|
||||||
|
|
||||||
|
impl std::fmt::Display for SendError {
|
||||||
|
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||||
|
match self {
|
||||||
|
SendError::Encode(e) => write!(f, "frame encode failed: {e:?}"),
|
||||||
|
SendError::Io(e) => write!(f, "transport write failed: {e}"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl std::error::Error for SendError {}
|
||||||
|
|
||||||
|
/// Error surface of [`FramedConn::recv`].
|
||||||
|
#[derive(Debug)]
|
||||||
|
pub enum RecvError {
|
||||||
|
/// The byte stream is not a valid frame stream (bad tag, lying length,
|
||||||
|
/// oversized frame, …). The connection is unusable.
|
||||||
|
Corrupt(DecodeError),
|
||||||
|
/// The peer closed mid-frame: EOF arrived with a partial frame buffered.
|
||||||
|
/// Distinct from a clean close, which is `Ok(None)`.
|
||||||
|
TruncatedByPeer,
|
||||||
|
/// The transport failed mid-read.
|
||||||
|
Io(io::Error),
|
||||||
|
/// The deadline passed before a full frame arrived
|
||||||
|
/// ([`FramedConn::recv_deadline`] only; plain [`recv`](FramedConn::recv)
|
||||||
|
/// never returns this).
|
||||||
|
TimedOut,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl std::fmt::Display for RecvError {
|
||||||
|
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||||
|
match self {
|
||||||
|
RecvError::Corrupt(e) => write!(f, "frame stream corrupt: {e:?}"),
|
||||||
|
RecvError::TruncatedByPeer => write!(f, "peer closed mid-frame"),
|
||||||
|
RecvError::Io(e) => write!(f, "transport read failed: {e}"),
|
||||||
|
RecvError::TimedOut => write!(f, "deadline passed mid-receive"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl std::error::Error for RecvError {}
|
||||||
|
|
||||||
|
/// How many bytes each blocking read asks the transport for.
|
||||||
|
const READ_CHUNK: usize = 8 * 1024;
|
||||||
|
|
||||||
|
/// The shared framed codec: one of these per control connection, owning the
|
||||||
|
/// [`Conn`] and the reassembly buffer. Frames may arrive split or coalesced
|
||||||
|
/// arbitrarily; [`recv`](FramedConn::recv) reassembles either way.
|
||||||
|
pub struct FramedConn {
|
||||||
|
conn: Box<dyn Conn>,
|
||||||
|
rbuf: Vec<u8>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl FramedConn {
|
||||||
|
pub fn new(conn: Box<dyn Conn>) -> Self {
|
||||||
|
FramedConn {
|
||||||
|
conn,
|
||||||
|
rbuf: Vec::new(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Encode and write one frame.
|
||||||
|
pub fn send(&mut self, frame: &Frame) -> Result<(), SendError> {
|
||||||
|
let mut out = Vec::new();
|
||||||
|
frame.encode(&mut out).map_err(SendError::Encode)?;
|
||||||
|
self.conn.write_all(&out).map_err(SendError::Io)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Receive the next frame. `Ok(None)` is a clean close: EOF at a frame
|
||||||
|
/// boundary. EOF mid-frame is [`RecvError::TruncatedByPeer`].
|
||||||
|
pub fn recv(&mut self) -> Result<Option<Frame>, RecvError> {
|
||||||
|
loop {
|
||||||
|
match Frame::decode(&self.rbuf) {
|
||||||
|
Ok(Some((frame, consumed))) => {
|
||||||
|
self.rbuf.drain(..consumed);
|
||||||
|
return Ok(Some(frame));
|
||||||
|
}
|
||||||
|
Ok(None) => {}
|
||||||
|
Err(e) => return Err(RecvError::Corrupt(e)),
|
||||||
|
}
|
||||||
|
let mut chunk = [0u8; READ_CHUNK];
|
||||||
|
let n = self.conn.read(&mut chunk).map_err(RecvError::Io)?;
|
||||||
|
if n == 0 {
|
||||||
|
return if self.rbuf.is_empty() {
|
||||||
|
Ok(None)
|
||||||
|
} else {
|
||||||
|
Err(RecvError::TruncatedByPeer)
|
||||||
|
};
|
||||||
|
}
|
||||||
|
self.rbuf.extend_from_slice(&chunk[..n]);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Like [`recv`](FramedConn::recv), but gives up with
|
||||||
|
/// [`RecvError::TimedOut`] once `deadline` passes without a full frame.
|
||||||
|
/// The deadline is enforced between reads via the connection's fd arm
|
||||||
|
/// (so the caller must be an actor); a transport with no fd (loopback)
|
||||||
|
/// cannot be timed out and this degrades to a plain blocking `recv` —
|
||||||
|
/// the same caveat as liveness.
|
||||||
|
pub fn recv_deadline(
|
||||||
|
&mut self,
|
||||||
|
deadline: std::time::Instant,
|
||||||
|
) -> Result<Option<Frame>, RecvError> {
|
||||||
|
loop {
|
||||||
|
match Frame::decode(&self.rbuf) {
|
||||||
|
Ok(Some((frame, consumed))) => {
|
||||||
|
self.rbuf.drain(..consumed);
|
||||||
|
return Ok(Some(frame));
|
||||||
|
}
|
||||||
|
Ok(None) => {}
|
||||||
|
Err(e) => return Err(RecvError::Corrupt(e)),
|
||||||
|
}
|
||||||
|
if let Some(arm) = self.conn.readable_arm() {
|
||||||
|
let left = deadline.saturating_duration_since(std::time::Instant::now());
|
||||||
|
if left.is_zero() {
|
||||||
|
return Err(RecvError::TimedOut);
|
||||||
|
}
|
||||||
|
match crate::channel::try_select_timeout(&[&arm], left) {
|
||||||
|
Ok(Some(_)) => {}
|
||||||
|
Ok(None) => return Err(RecvError::TimedOut),
|
||||||
|
Err(e) => return Err(RecvError::Io(e)),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let mut chunk = [0u8; READ_CHUNK];
|
||||||
|
let n = self.conn.read(&mut chunk).map_err(RecvError::Io)?;
|
||||||
|
if n == 0 {
|
||||||
|
return if self.rbuf.is_empty() {
|
||||||
|
Ok(None)
|
||||||
|
} else {
|
||||||
|
Err(RecvError::TruncatedByPeer)
|
||||||
|
};
|
||||||
|
}
|
||||||
|
self.rbuf.extend_from_slice(&chunk[..n]);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One socket read, appended to the reassembly buffer. Returns the byte
|
||||||
|
/// count (`0` = EOF). For select-loop callers that were just told the fd
|
||||||
|
/// is readable: under the level-triggered IO thread exactly one read per
|
||||||
|
/// readable wake never blocks and never loses data — leftover socket
|
||||||
|
/// bytes re-signal on the next select, and complete frames already
|
||||||
|
/// reassembled are drained with [`next_buffered`](FramedConn::next_buffered).
|
||||||
|
/// (A plain [`recv`](FramedConn::recv) can block into the socket while
|
||||||
|
/// the buffer holds a partial frame, which a loop with deadlines to keep
|
||||||
|
/// cannot afford.)
|
||||||
|
pub fn read_once(&mut self) -> std::io::Result<usize> {
|
||||||
|
let mut chunk = [0u8; READ_CHUNK];
|
||||||
|
let n = self.conn.read(&mut chunk)?;
|
||||||
|
self.rbuf.extend_from_slice(&chunk[..n]);
|
||||||
|
Ok(n)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Decode the next complete frame already sitting in the reassembly
|
||||||
|
/// buffer, without touching the socket. `Ok(None)` means the buffer
|
||||||
|
/// holds no complete frame (empty, or a partial awaiting more bytes).
|
||||||
|
pub fn next_buffered(&mut self) -> Result<Option<Frame>, DecodeError> {
|
||||||
|
match Frame::decode(&self.rbuf) {
|
||||||
|
Ok(Some((frame, consumed))) => {
|
||||||
|
self.rbuf.drain(..consumed);
|
||||||
|
Ok(Some(frame))
|
||||||
|
}
|
||||||
|
Ok(None) => Ok(None),
|
||||||
|
Err(e) => Err(e),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Close the underlying connection (idempotent, see [`Conn::close`]).
|
||||||
|
pub fn close(&mut self) {
|
||||||
|
self.conn.close();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Diagnostic label of the underlying connection.
|
||||||
|
pub fn peer_addr(&self) -> String {
|
||||||
|
self.conn.peer_addr()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The underlying connection's readiness arm, if it is fd-backed (see
|
||||||
|
/// [`Conn::readable_arm`]).
|
||||||
|
pub fn readable_arm(&self) -> Option<crate::scheduler::FdArm> {
|
||||||
|
self.conn.readable_arm()
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,274 @@
|
|||||||
|
//! In-memory loopback transport — a shipped **test** transport.
|
||||||
|
//!
|
||||||
|
//! Lets Phases 2–4 exercise protocol logic (connector, membership,
|
||||||
|
//! monitors) through the real transport trait and the real framed codec
|
||||||
|
//! without sockets or timing flake.
|
||||||
|
//!
|
||||||
|
//! Blocking model: calls block the **OS thread** on a condvar. That is the
|
||||||
|
//! right shape for plain `#[test]`s driving protocol state machines; it is
|
||||||
|
//! the wrong shape for scheduler threads. Do not drive a loopback conn from
|
||||||
|
//! inside an actor — use the TCP impl there.
|
||||||
|
//!
|
||||||
|
//! Semantics mirror TCP shutdown where it matters for the codec: bytes
|
||||||
|
//! written before `close` remain readable at the peer, which then sees EOF;
|
||||||
|
//! writes toward a closed peer fail with `BrokenPipe`. Write buffers are
|
||||||
|
//! unbounded, so writes never block — backpressure is not simulated.
|
||||||
|
|
||||||
|
use std::collections::{HashMap, VecDeque};
|
||||||
|
use std::io;
|
||||||
|
use std::sync::{Arc, Condvar, Mutex, MutexGuard};
|
||||||
|
|
||||||
|
use super::{Conn, Listener, Transport};
|
||||||
|
|
||||||
|
/// Poison-tolerant lock: a panicked holder in a *test* transport must not
|
||||||
|
/// cascade; the byte-queue state stays consistent under every early return.
|
||||||
|
fn lock<T>(m: &Mutex<T>) -> MutexGuard<'_, T> {
|
||||||
|
match m.lock() {
|
||||||
|
Ok(g) => g,
|
||||||
|
Err(poisoned) => poisoned.into_inner(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// One direction of a duplex: a byte queue with close flags for both ends
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
#[derive(Default)]
|
||||||
|
struct PipeState {
|
||||||
|
bytes: VecDeque<u8>,
|
||||||
|
/// The writing end closed: readers drain remaining bytes, then EOF.
|
||||||
|
write_closed: bool,
|
||||||
|
/// The reading end closed: writers fail with `BrokenPipe`.
|
||||||
|
read_closed: bool,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Default)]
|
||||||
|
struct Pipe {
|
||||||
|
state: Mutex<PipeState>,
|
||||||
|
cv: Condvar,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Pipe {
|
||||||
|
fn write_all(&self, buf: &[u8]) -> io::Result<()> {
|
||||||
|
let mut st = lock(&self.state);
|
||||||
|
if st.write_closed {
|
||||||
|
return Err(io::Error::new(
|
||||||
|
io::ErrorKind::NotConnected,
|
||||||
|
"loopback conn closed locally",
|
||||||
|
));
|
||||||
|
}
|
||||||
|
if st.read_closed {
|
||||||
|
return Err(io::Error::new(
|
||||||
|
io::ErrorKind::BrokenPipe,
|
||||||
|
"loopback peer closed",
|
||||||
|
));
|
||||||
|
}
|
||||||
|
st.bytes.extend(buf);
|
||||||
|
self.cv.notify_all();
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
fn read(&self, buf: &mut [u8]) -> io::Result<usize> {
|
||||||
|
if buf.is_empty() {
|
||||||
|
return Ok(0);
|
||||||
|
}
|
||||||
|
let mut st = lock(&self.state);
|
||||||
|
loop {
|
||||||
|
if !st.bytes.is_empty() {
|
||||||
|
let n = st.bytes.len().min(buf.len());
|
||||||
|
for (slot, byte) in buf.iter_mut().zip(st.bytes.drain(..n)) {
|
||||||
|
*slot = byte;
|
||||||
|
}
|
||||||
|
return Ok(n);
|
||||||
|
}
|
||||||
|
if st.write_closed || st.read_closed {
|
||||||
|
return Ok(0); // EOF: peer closed, or our own end closed.
|
||||||
|
}
|
||||||
|
st = match self.cv.wait(st) {
|
||||||
|
Ok(g) => g,
|
||||||
|
Err(poisoned) => poisoned.into_inner(),
|
||||||
|
};
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Close from the writer side: remaining bytes stay readable, then EOF.
|
||||||
|
fn close_write(&self) {
|
||||||
|
lock(&self.state).write_closed = true;
|
||||||
|
self.cv.notify_all();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Close from the reader side: peer writes fail from now on.
|
||||||
|
fn close_read(&self) {
|
||||||
|
lock(&self.state).read_closed = true;
|
||||||
|
self.cv.notify_all();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Conn: two pipes, one per direction
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
/// One end of an established loopback connection.
|
||||||
|
pub struct LoopbackConn {
|
||||||
|
tx: Arc<Pipe>,
|
||||||
|
rx: Arc<Pipe>,
|
||||||
|
peer: String,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Conn for LoopbackConn {
|
||||||
|
fn read(&mut self, buf: &mut [u8]) -> io::Result<usize> {
|
||||||
|
self.rx.read(buf)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn write_all(&mut self, buf: &[u8]) -> io::Result<()> {
|
||||||
|
self.tx.write_all(buf)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn close(&mut self) {
|
||||||
|
self.tx.close_write();
|
||||||
|
self.rx.close_read();
|
||||||
|
}
|
||||||
|
|
||||||
|
fn peer_addr(&self) -> String {
|
||||||
|
self.peer.clone()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Drop for LoopbackConn {
|
||||||
|
fn drop(&mut self) {
|
||||||
|
self.close();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn conn_pair(listen_addr: &str, conn_no: u64) -> (LoopbackConn, LoopbackConn) {
|
||||||
|
let a_to_b = Arc::new(Pipe::default());
|
||||||
|
let b_to_a = Arc::new(Pipe::default());
|
||||||
|
let dialer = LoopbackConn {
|
||||||
|
tx: a_to_b.clone(),
|
||||||
|
rx: b_to_a.clone(),
|
||||||
|
peer: listen_addr.to_string(),
|
||||||
|
};
|
||||||
|
let accepted = LoopbackConn {
|
||||||
|
tx: b_to_a,
|
||||||
|
rx: a_to_b,
|
||||||
|
peer: format!("{listen_addr}#dialer-{conn_no}"),
|
||||||
|
};
|
||||||
|
(dialer, accepted)
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Listener + registry
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
#[derive(Default)]
|
||||||
|
struct AcceptState {
|
||||||
|
pending: VecDeque<LoopbackConn>,
|
||||||
|
closed: bool,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Default)]
|
||||||
|
struct AcceptQueue {
|
||||||
|
state: Mutex<AcceptState>,
|
||||||
|
cv: Condvar,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A bound loopback listen point.
|
||||||
|
pub struct LoopbackListener {
|
||||||
|
addr: String,
|
||||||
|
queue: Arc<AcceptQueue>,
|
||||||
|
registry: Arc<Mutex<Registry>>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Listener for LoopbackListener {
|
||||||
|
fn accept(&mut self) -> io::Result<Box<dyn Conn>> {
|
||||||
|
let mut st = lock(&self.queue.state);
|
||||||
|
loop {
|
||||||
|
if let Some(conn) = st.pending.pop_front() {
|
||||||
|
return Ok(Box::new(conn));
|
||||||
|
}
|
||||||
|
if st.closed {
|
||||||
|
return Err(io::Error::new(
|
||||||
|
io::ErrorKind::NotConnected,
|
||||||
|
"loopback listener closed",
|
||||||
|
));
|
||||||
|
}
|
||||||
|
st = match self.queue.cv.wait(st) {
|
||||||
|
Ok(g) => g,
|
||||||
|
Err(poisoned) => poisoned.into_inner(),
|
||||||
|
};
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn local_addr(&self) -> String {
|
||||||
|
self.addr.clone()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Drop for LoopbackListener {
|
||||||
|
fn drop(&mut self) {
|
||||||
|
lock(&self.registry).listeners.remove(&self.addr);
|
||||||
|
let mut st = lock(&self.queue.state);
|
||||||
|
st.closed = true;
|
||||||
|
self.queue.cv.notify_all();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Default)]
|
||||||
|
struct Registry {
|
||||||
|
listeners: HashMap<String, Arc<AcceptQueue>>,
|
||||||
|
dial_count: u64,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The loopback transport. Addresses are arbitrary strings scoped to one
|
||||||
|
/// transport instance; distinct instances never see each other's listeners.
|
||||||
|
#[derive(Default)]
|
||||||
|
pub struct LoopbackTransport {
|
||||||
|
registry: Arc<Mutex<Registry>>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Transport for LoopbackTransport {
|
||||||
|
fn dial(&self, addr: &str) -> io::Result<Box<dyn Conn>> {
|
||||||
|
let (queue, conn_no) = {
|
||||||
|
let mut reg = lock(&self.registry);
|
||||||
|
reg.dial_count += 1;
|
||||||
|
let no = reg.dial_count;
|
||||||
|
match reg.listeners.get(addr) {
|
||||||
|
Some(q) => (q.clone(), no),
|
||||||
|
None => {
|
||||||
|
return Err(io::Error::new(
|
||||||
|
io::ErrorKind::ConnectionRefused,
|
||||||
|
format!("no loopback listener at {addr:?}"),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let (dialer, accepted) = conn_pair(addr, conn_no);
|
||||||
|
let mut st = lock(&queue.state);
|
||||||
|
if st.closed {
|
||||||
|
return Err(io::Error::new(
|
||||||
|
io::ErrorKind::ConnectionRefused,
|
||||||
|
format!("loopback listener at {addr:?} closed"),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
st.pending.push_back(accepted);
|
||||||
|
queue.cv.notify_all();
|
||||||
|
Ok(Box::new(dialer))
|
||||||
|
}
|
||||||
|
|
||||||
|
fn listen(&self, addr: &str) -> io::Result<Box<dyn Listener>> {
|
||||||
|
let queue = Arc::new(AcceptQueue::default());
|
||||||
|
let mut reg = lock(&self.registry);
|
||||||
|
if reg.listeners.contains_key(addr) {
|
||||||
|
return Err(io::Error::new(
|
||||||
|
io::ErrorKind::AddrInUse,
|
||||||
|
format!("loopback listener already bound at {addr:?}"),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
reg.listeners.insert(addr.to_string(), queue.clone());
|
||||||
|
Ok(Box::new(LoopbackListener {
|
||||||
|
addr: addr.to_string(),
|
||||||
|
queue,
|
||||||
|
registry: self.registry.clone(),
|
||||||
|
}))
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,285 @@
|
|||||||
|
//! TCP transport — the production control-plane transport.
|
||||||
|
//!
|
||||||
|
//! Blocking model: every blocking point parks the **calling actor** on fd
|
||||||
|
//! readiness ([`crate::scheduler::wait_readable`] / `wait_writable`); the
|
||||||
|
//! scheduler thread is never blocked. All conn/listener methods must
|
||||||
|
//! therefore run inside an actor. `listen` itself only binds (no waiting)
|
||||||
|
//! and is callable anywhere.
|
||||||
|
//!
|
||||||
|
//! Addresses are pre-resolved `ip:port` strings (`SocketAddr` syntax, IPv4
|
||||||
|
//! or IPv6). Hostnames are rejected with `InvalidInput`: name resolution is
|
||||||
|
//! the single c9 seam, not something each transport does on the side.
|
||||||
|
//!
|
||||||
|
//! Writes use `send(2)` with `MSG_NOSIGNAL` — a peer reset must surface as
|
||||||
|
//! `BrokenPipe`/`ConnectionReset`, not `SIGPIPE`.
|
||||||
|
|
||||||
|
use std::io;
|
||||||
|
use std::net::{SocketAddr, TcpListener as StdListener, TcpStream};
|
||||||
|
use std::os::fd::{AsRawFd, RawFd};
|
||||||
|
|
||||||
|
use crate::scheduler::{wait_readable, wait_writable};
|
||||||
|
|
||||||
|
use super::{Conn, Listener, Transport};
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// sockaddr plumbing
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
/// A `sockaddr_in`/`sockaddr_in6` built from a parsed `SocketAddr`, plus its
|
||||||
|
/// length, ready for `connect(2)`.
|
||||||
|
union SockAddrUnion {
|
||||||
|
v4: libc::sockaddr_in,
|
||||||
|
v6: libc::sockaddr_in6,
|
||||||
|
}
|
||||||
|
|
||||||
|
fn to_sockaddr(sa: &SocketAddr) -> (SockAddrUnion, libc::socklen_t) {
|
||||||
|
match sa {
|
||||||
|
SocketAddr::V4(v4) => {
|
||||||
|
let raw = libc::sockaddr_in {
|
||||||
|
sin_family: libc::AF_INET as libc::sa_family_t,
|
||||||
|
sin_port: v4.port().to_be(),
|
||||||
|
sin_addr: libc::in_addr {
|
||||||
|
s_addr: u32::from_be_bytes(v4.ip().octets()).to_be(),
|
||||||
|
},
|
||||||
|
sin_zero: [0; 8],
|
||||||
|
};
|
||||||
|
(
|
||||||
|
SockAddrUnion { v4: raw },
|
||||||
|
std::mem::size_of::<libc::sockaddr_in>() as libc::socklen_t,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
SocketAddr::V6(v6) => {
|
||||||
|
let raw = libc::sockaddr_in6 {
|
||||||
|
sin6_family: libc::AF_INET6 as libc::sa_family_t,
|
||||||
|
sin6_port: v6.port().to_be(),
|
||||||
|
sin6_flowinfo: v6.flowinfo(),
|
||||||
|
sin6_addr: libc::in6_addr {
|
||||||
|
s6_addr: v6.ip().octets(),
|
||||||
|
},
|
||||||
|
sin6_scope_id: v6.scope_id(),
|
||||||
|
};
|
||||||
|
(
|
||||||
|
SockAddrUnion { v6: raw },
|
||||||
|
std::mem::size_of::<libc::sockaddr_in6>() as libc::socklen_t,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parse_addr(addr: &str) -> io::Result<SocketAddr> {
|
||||||
|
addr.parse().map_err(|_| {
|
||||||
|
io::Error::new(
|
||||||
|
io::ErrorKind::InvalidInput,
|
||||||
|
format!("{addr:?} is not a resolved ip:port — resolution is the c9 seam"),
|
||||||
|
)
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
fn so_error(fd: RawFd) -> io::Result<()> {
|
||||||
|
let mut err: libc::c_int = 0;
|
||||||
|
let mut len = std::mem::size_of::<libc::c_int>() as libc::socklen_t;
|
||||||
|
let rc = unsafe {
|
||||||
|
libc::getsockopt(
|
||||||
|
fd,
|
||||||
|
libc::SOL_SOCKET,
|
||||||
|
libc::SO_ERROR,
|
||||||
|
(&mut err) as *mut _ as *mut libc::c_void,
|
||||||
|
&mut len,
|
||||||
|
)
|
||||||
|
};
|
||||||
|
if rc != 0 {
|
||||||
|
return Err(io::Error::last_os_error());
|
||||||
|
}
|
||||||
|
if err != 0 {
|
||||||
|
return Err(io::Error::from_raw_os_error(err));
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Conn
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
/// One established TCP control connection. Owns the socket; drop closes it.
|
||||||
|
pub struct TcpConn {
|
||||||
|
stream: TcpStream,
|
||||||
|
closed: bool,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl TcpConn {
|
||||||
|
fn fd(&self) -> RawFd {
|
||||||
|
self.stream.as_raw_fd()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Conn for TcpConn {
|
||||||
|
fn read(&mut self, buf: &mut [u8]) -> io::Result<usize> {
|
||||||
|
if self.closed {
|
||||||
|
return Ok(0);
|
||||||
|
}
|
||||||
|
if buf.is_empty() {
|
||||||
|
return Ok(0);
|
||||||
|
}
|
||||||
|
loop {
|
||||||
|
wait_readable(self.fd())?;
|
||||||
|
let n = unsafe { libc::read(self.fd(), buf.as_mut_ptr() as *mut _, buf.len()) };
|
||||||
|
if n >= 0 {
|
||||||
|
return Ok(n as usize);
|
||||||
|
}
|
||||||
|
let e = io::Error::last_os_error();
|
||||||
|
match e.kind() {
|
||||||
|
// Spurious readiness or signal: park again.
|
||||||
|
io::ErrorKind::WouldBlock | io::ErrorKind::Interrupted => continue,
|
||||||
|
_ => return Err(e),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn write_all(&mut self, mut buf: &[u8]) -> io::Result<()> {
|
||||||
|
if self.closed {
|
||||||
|
return Err(io::Error::new(
|
||||||
|
io::ErrorKind::NotConnected,
|
||||||
|
"tcp conn closed locally",
|
||||||
|
));
|
||||||
|
}
|
||||||
|
while !buf.is_empty() {
|
||||||
|
wait_writable(self.fd())?;
|
||||||
|
let n = unsafe {
|
||||||
|
libc::send(
|
||||||
|
self.fd(),
|
||||||
|
buf.as_ptr() as *const _,
|
||||||
|
buf.len(),
|
||||||
|
libc::MSG_NOSIGNAL,
|
||||||
|
)
|
||||||
|
};
|
||||||
|
if n >= 0 {
|
||||||
|
buf = &buf[n as usize..];
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
let e = io::Error::last_os_error();
|
||||||
|
match e.kind() {
|
||||||
|
io::ErrorKind::WouldBlock | io::ErrorKind::Interrupted => continue,
|
||||||
|
_ => return Err(e),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
fn close(&mut self) {
|
||||||
|
if !self.closed {
|
||||||
|
self.closed = true;
|
||||||
|
// Best-effort: the peer sees EOF after draining. The fd itself
|
||||||
|
// is released when the owning stream drops.
|
||||||
|
let _ = self.stream.shutdown(std::net::Shutdown::Both);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn peer_addr(&self) -> String {
|
||||||
|
match self.stream.peer_addr() {
|
||||||
|
Ok(sa) => sa.to_string(),
|
||||||
|
Err(_) => "<disconnected>".to_string(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn readable_arm(&self) -> Option<crate::scheduler::FdArm> {
|
||||||
|
Some(crate::scheduler::FdArm::readable(self.fd()))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Listener
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
/// A bound TCP listen point (non-blocking socket; accept parks the actor).
|
||||||
|
pub struct TcpListener {
|
||||||
|
inner: StdListener,
|
||||||
|
local: SocketAddr,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Listener for TcpListener {
|
||||||
|
fn readable_arm(&self) -> Option<crate::scheduler::FdArm> {
|
||||||
|
Some(crate::scheduler::FdArm::readable(self.inner.as_raw_fd()))
|
||||||
|
}
|
||||||
|
|
||||||
|
fn accept(&mut self) -> io::Result<Box<dyn Conn>> {
|
||||||
|
loop {
|
||||||
|
wait_readable(self.inner.as_raw_fd())?;
|
||||||
|
match self.inner.accept() {
|
||||||
|
Ok((stream, _peer)) => {
|
||||||
|
stream.set_nonblocking(true)?;
|
||||||
|
return Ok(Box::new(TcpConn {
|
||||||
|
stream,
|
||||||
|
closed: false,
|
||||||
|
}));
|
||||||
|
}
|
||||||
|
Err(e)
|
||||||
|
if e.kind() == io::ErrorKind::WouldBlock
|
||||||
|
|| e.kind() == io::ErrorKind::Interrupted =>
|
||||||
|
{
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
Err(e) => return Err(e),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn local_addr(&self) -> String {
|
||||||
|
self.local.to_string()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Transport
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
/// The TCP transport. Stateless; every call stands alone.
|
||||||
|
pub struct TcpTransport;
|
||||||
|
|
||||||
|
impl Transport for TcpTransport {
|
||||||
|
fn dial(&self, addr: &str) -> io::Result<Box<dyn Conn>> {
|
||||||
|
let sa = parse_addr(addr)?;
|
||||||
|
let family = match sa {
|
||||||
|
SocketAddr::V4(_) => libc::AF_INET,
|
||||||
|
SocketAddr::V6(_) => libc::AF_INET6,
|
||||||
|
};
|
||||||
|
let fd = unsafe {
|
||||||
|
libc::socket(
|
||||||
|
family,
|
||||||
|
libc::SOCK_STREAM | libc::SOCK_NONBLOCK | libc::SOCK_CLOEXEC,
|
||||||
|
0,
|
||||||
|
)
|
||||||
|
};
|
||||||
|
if fd < 0 {
|
||||||
|
return Err(io::Error::last_os_error());
|
||||||
|
}
|
||||||
|
// From here the fd is owned by `stream`; any early return drops it.
|
||||||
|
let stream = unsafe {
|
||||||
|
use std::os::fd::FromRawFd;
|
||||||
|
TcpStream::from_raw_fd(fd)
|
||||||
|
};
|
||||||
|
let (raw, len) = to_sockaddr(&sa);
|
||||||
|
let rc = unsafe { libc::connect(fd, (&raw) as *const _ as *const libc::sockaddr, len) };
|
||||||
|
if rc != 0 {
|
||||||
|
let e = io::Error::last_os_error();
|
||||||
|
if e.raw_os_error() != Some(libc::EINPROGRESS) {
|
||||||
|
return Err(e);
|
||||||
|
}
|
||||||
|
// Connect in flight: park until the socket is writable, then the
|
||||||
|
// verdict is in SO_ERROR.
|
||||||
|
wait_writable(fd)?;
|
||||||
|
so_error(fd)?;
|
||||||
|
}
|
||||||
|
Ok(Box::new(TcpConn {
|
||||||
|
stream,
|
||||||
|
closed: false,
|
||||||
|
}))
|
||||||
|
}
|
||||||
|
|
||||||
|
fn listen(&self, addr: &str) -> io::Result<Box<dyn Listener>> {
|
||||||
|
let sa = parse_addr(addr)?;
|
||||||
|
let inner = StdListener::bind(sa)?;
|
||||||
|
inner.set_nonblocking(true)?;
|
||||||
|
let local = inner.local_addr()?;
|
||||||
|
Ok(Box::new(TcpListener { inner, local }))
|
||||||
|
}
|
||||||
|
}
|
||||||
+82
-34
@@ -4,30 +4,66 @@
|
|||||||
//! actor running on its own mmap'd stack. The compiler cannot do this; the
|
//! actor running on its own mmap'd stack. The compiler cannot do this; the
|
||||||
//! whole point of `#[unsafe(naked)]` is that we control every instruction.
|
//! whole point of `#[unsafe(naked)]` is that we control every instruction.
|
||||||
//!
|
//!
|
||||||
//! `SCHEDULER_SP` and `ACTOR_SP` are thread-locals holding each side's saved
|
//! The actor's stack pointer travels in registers: `switch_to_actor` takes
|
||||||
//! stack pointer. `init_actor_stack` builds the initial stack so that the
|
//! the target sp as its argument and returns the sp the actor saved when it
|
||||||
//! first `switch_to_actor` lands inside the entry function with `rsp % 16 == 8`
|
//! next yielded (handed back in `rax` by `switch_to_scheduler`'s shim). Only
|
||||||
//! (the x86-64 ABI requirement at function entry).
|
//! the *scheduler* sp lives in a thread-local — the yielding actor sits at
|
||||||
|
//! arbitrary call depth with no argument channel back to the scheduler, so
|
||||||
|
//! TLS is the one place it can find the way home. `init_actor_stack` builds
|
||||||
|
//! the initial stack so that the first `switch_to_actor` lands inside the
|
||||||
|
//! entry function with `rsp % 16 == 8` (the x86-64 ABI requirement at
|
||||||
|
//! function entry).
|
||||||
|
//!
|
||||||
|
//! # Thread-locals and migration (read before touching any `thread_local!`)
|
||||||
|
//!
|
||||||
|
//! An actor may park on scheduler thread A and be resumed on thread B. LLVM
|
||||||
|
//! treats the address of a thread-local as a loop-invariant, side-effect-free
|
||||||
|
//! value: it computes `%fs:0 + offset` once per function and happily keeps it
|
||||||
|
//! in a callee-saved register across calls — including across
|
||||||
|
//! `switch_to_scheduler`. Any function that touches a scheduler thread-local
|
||||||
|
//! both before and after a switch (or that gets *inlined* into one that does)
|
||||||
|
//! therefore reads and writes the *old thread's* TLS after migration. Nothing
|
||||||
|
//! at the switch can prevent this: it is not a memory clobber problem, the
|
||||||
|
//! address is not memory-derived in LLVM's model. Under thin LTO the code
|
||||||
|
//! happened to use the local-exec model (`%fs:imm` operands, nothing to
|
||||||
|
//! cache) so it worked by luck; a plain `cargo build --release` of a
|
||||||
|
//! downstream crate broke multi-thread runs (`ACTOR_DONE` written to the wrong
|
||||||
|
//! thread → "scheduler resumed a done actor").
|
||||||
|
//!
|
||||||
|
//! Rule: every function that touches a thread-local and can execute on an
|
||||||
|
//! actor stack must be `#[inline(never)]` and call [`tls_fence`] first, so the
|
||||||
|
//! TLS base is recomputed inside a callee that cannot be inlined into a frame
|
||||||
|
//! spanning a switch, and so LLVM cannot infer the accessor is pure and merge
|
||||||
|
//! two calls to it. Scheduler-side code (`schedule_loop` and what it calls
|
||||||
|
//! before/after `switch_to_actor`) never migrates and is exempt. Do not return
|
||||||
|
//! `&Cell`/pointers into TLS from these accessors; return values.
|
||||||
|
//! `cargo test --profile reltest` (no LTO) is the regression oracle.
|
||||||
|
|
||||||
use std::cell::Cell;
|
use std::cell::Cell;
|
||||||
|
|
||||||
thread_local! {
|
/// Compiler barrier for TLS accessors, see the module docs. Emits no code; a
|
||||||
static SCHEDULER_SP: Cell<usize> = const { Cell::new(0) };
|
/// side-effecting empty asm keeps LLVM from marking the enclosing
|
||||||
static ACTOR_SP: Cell<usize> = const { Cell::new(0) };
|
/// `#[inline(never)]` accessor `memory(none)` and merging calls to it.
|
||||||
|
#[inline(always)]
|
||||||
|
pub(crate) fn tls_fence() {
|
||||||
|
// SAFETY: empty asm, no operands, no stack, no flags.
|
||||||
|
unsafe { core::arch::asm!("", options(nostack, preserves_flags)) }
|
||||||
}
|
}
|
||||||
|
|
||||||
|
thread_local! {
|
||||||
|
static SCHEDULER_SP: Cell<usize> = const { Cell::new(0) };
|
||||||
|
}
|
||||||
|
|
||||||
|
#[inline(never)]
|
||||||
fn get_scheduler_sp() -> usize {
|
fn get_scheduler_sp() -> usize {
|
||||||
|
tls_fence();
|
||||||
SCHEDULER_SP.with(|c| c.get())
|
SCHEDULER_SP.with(|c| c.get())
|
||||||
}
|
}
|
||||||
|
#[inline(never)]
|
||||||
fn set_scheduler_sp(v: usize) {
|
fn set_scheduler_sp(v: usize) {
|
||||||
|
tls_fence();
|
||||||
SCHEDULER_SP.with(|c| c.set(v))
|
SCHEDULER_SP.with(|c| c.set(v))
|
||||||
}
|
}
|
||||||
pub fn get_actor_sp() -> usize {
|
|
||||||
ACTOR_SP.with(|c| c.get())
|
|
||||||
}
|
|
||||||
pub fn set_actor_sp(v: usize) {
|
|
||||||
ACTOR_SP.with(|c| c.set(v))
|
|
||||||
}
|
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
// Initial stack layout
|
// Initial stack layout
|
||||||
@@ -78,12 +114,24 @@ pub fn init_actor_stack(top: *mut u8, entry: extern "C-unwind" fn()) -> usize {
|
|||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
// Context switch shims
|
// Context switch shims
|
||||||
//
|
//
|
||||||
// Each shim:
|
// switch_to_actor_asm (rdi = target actor sp, returns rax = the sp the actor
|
||||||
// 1. Pushes the six callee-saved integer registers.
|
// saved when it next yielded):
|
||||||
// 2. Snaps rsp into rdi and calls the Rust helper that stores it.
|
// 1. Pushes the six callee-saved integer registers (scheduler side).
|
||||||
// 3. Calls the Rust helper that returns the *other* side's saved rsp.
|
// 2. Stashes the target sp in rbx — free scratch: the register's live
|
||||||
// 4. Moves that into rsp.
|
// value is on the stack we just pushed to, and the pops below load the
|
||||||
// 5. Pops the six registers and rets.
|
// *other* side's values anyway — then snaps rsp into rdi and calls the
|
||||||
|
// Rust helper that stores it in SCHEDULER_SP.
|
||||||
|
// 3. Installs the target sp and pops the actor's registers; `ret` lands
|
||||||
|
// where the actor yielded (or in `entry` on first resume).
|
||||||
|
//
|
||||||
|
// switch_to_scheduler_asm (no args; its "return value" materialises on the
|
||||||
|
// OTHER stack, as switch_to_actor's rax):
|
||||||
|
// 1. Pushes the six callee-saved integer registers (actor side).
|
||||||
|
// 2. Stashes its own rsp in rbx (same free-scratch argument), asks the
|
||||||
|
// Rust helper for SCHEDULER_SP.
|
||||||
|
// 3. Installs the scheduler sp, moves the saved actor sp into rax, pops
|
||||||
|
// the scheduler's registers and rets — completing the scheduler's
|
||||||
|
// `switch_to_actor(sp)` call with the actor's new sp as its result.
|
||||||
//
|
//
|
||||||
// XMM registers are NOT saved here. We rely on every yield happening through
|
// XMM registers are NOT saved here. We rely on every yield happening through
|
||||||
// a Rust call site, which means the compiler has spilled any live XMM state
|
// a Rust call site, which means the compiler has spilled any live XMM state
|
||||||
@@ -94,31 +142,32 @@ pub fn init_actor_stack(top: *mut u8, entry: extern "C-unwind" fn()) -> usize {
|
|||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
#[unsafe(naked)]
|
#[unsafe(naked)]
|
||||||
unsafe extern "C" fn switch_to_actor_asm() {
|
unsafe extern "C" fn switch_to_actor_asm(actor_sp: usize) -> usize {
|
||||||
core::arch::naked_asm!(
|
core::arch::naked_asm!(
|
||||||
"push rbx", "push rbp", "push r12", "push r13", "push r14", "push r15",
|
"push rbx", "push rbp", "push r12", "push r13", "push r14", "push r15",
|
||||||
|
"mov rbx, rdi",
|
||||||
"mov rdi, rsp",
|
"mov rdi, rsp",
|
||||||
"call {set_sched_sp}",
|
"call {set_sched_sp}",
|
||||||
"call {get_actor_sp}",
|
"mov rsp, rbx",
|
||||||
"mov rsp, rax",
|
|
||||||
"pop r15", "pop r14", "pop r13", "pop r12", "pop rbp", "pop rbx",
|
"pop r15", "pop r14", "pop r13", "pop r12", "pop rbp", "pop rbx",
|
||||||
"ret",
|
"ret",
|
||||||
set_sched_sp = sym set_scheduler_sp,
|
set_sched_sp = sym set_scheduler_sp,
|
||||||
get_actor_sp = sym get_actor_sp,
|
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Resume the actor whose sp is in `ACTOR_SP`. Returns when the actor yields.
|
/// Resume the actor whose saved stack pointer is `actor_sp`. Returns when the
|
||||||
|
/// actor yields, with the stack pointer the actor saved as it did — store it
|
||||||
|
/// back into the slot for the next resume.
|
||||||
///
|
///
|
||||||
/// # Safety
|
/// # Safety
|
||||||
///
|
///
|
||||||
/// The caller must be running on a scheduler thread with a valid actor stack
|
/// `actor_sp` must be a valid saved actor stack pointer — either from
|
||||||
/// pointer installed in `ACTOR_SP` — either by `init_actor_stack` (first
|
/// `init_actor_stack` (first resume) or the value a prior `switch_to_actor`
|
||||||
/// resume) or by a prior `switch_to_scheduler` (subsequent resumes). Resuming
|
/// returned for this actor (subsequent resumes). Resuming with a stale or
|
||||||
/// with an unset or stale `ACTOR_SP` transfers control to an arbitrary address.
|
/// forged sp transfers control to an arbitrary address. Must not be called
|
||||||
/// Must not be called from within an actor (only the scheduler side may resume).
|
/// from within an actor (only the scheduler side may resume).
|
||||||
pub unsafe fn switch_to_actor() {
|
pub unsafe fn switch_to_actor(actor_sp: usize) -> usize {
|
||||||
unsafe { switch_to_actor_asm() };
|
unsafe { switch_to_actor_asm(actor_sp) }
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Yield from the running actor back to its scheduler thread. Returns when the
|
/// Yield from the running actor back to its scheduler thread. Returns when the
|
||||||
@@ -135,13 +184,12 @@ pub unsafe fn switch_to_actor() {
|
|||||||
pub unsafe extern "C" fn switch_to_scheduler() {
|
pub unsafe extern "C" fn switch_to_scheduler() {
|
||||||
core::arch::naked_asm!(
|
core::arch::naked_asm!(
|
||||||
"push rbx", "push rbp", "push r12", "push r13", "push r14", "push r15",
|
"push rbx", "push rbp", "push r12", "push r13", "push r14", "push r15",
|
||||||
"mov rdi, rsp",
|
"mov rbx, rsp",
|
||||||
"call {set_actor_sp}",
|
|
||||||
"call {get_sched_sp}",
|
"call {get_sched_sp}",
|
||||||
"mov rsp, rax",
|
"mov rsp, rax",
|
||||||
|
"mov rax, rbx",
|
||||||
"pop r15", "pop r14", "pop r13", "pop r12", "pop rbp", "pop rbx",
|
"pop r15", "pop r14", "pop r13", "pop r12", "pop rbp", "pop rbx",
|
||||||
"ret",
|
"ret",
|
||||||
set_actor_sp = sym set_actor_sp,
|
|
||||||
get_sched_sp = sym get_scheduler_sp,
|
get_sched_sp = sym get_scheduler_sp,
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|||||||
+4
-4
@@ -301,7 +301,7 @@ pub trait GenServer: Send + 'static {
|
|||||||
/// full idle window (set via [`GenServerCtx::idle_after`] in `init`) without
|
/// full idle window (set via [`GenServerCtx::idle_after`] in `init`) without
|
||||||
/// dispatching any message. The window resets automatically after this
|
/// dispatching any message. The window resets automatically after this
|
||||||
/// fires, so it acts as a steady idle detector. To shut down after one idle
|
/// fires, so it acts as a steady idle detector. To shut down after one idle
|
||||||
/// period, call [`request_stop`](crate::scheduler::request_stop) here.
|
/// period, call [`request_stop`] here.
|
||||||
/// Default: no-op.
|
/// Default: no-op.
|
||||||
fn handle_idle(&mut self) {}
|
fn handle_idle(&mut self) {}
|
||||||
|
|
||||||
@@ -814,7 +814,7 @@ impl<G: GenServer> GenServerBuilder<G> {
|
|||||||
self
|
self
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Spawn the server under an explicit supervisor pid (via [`spawn_under`])
|
/// Spawn the server under an explicit supervisor pid (via [`spawn_under`](crate::scheduler::spawn_under))
|
||||||
/// so it slots into the supervision tree.
|
/// so it slots into the supervision tree.
|
||||||
pub fn under(mut self, supervisor: Pid) -> Self {
|
pub fn under(mut self, supervisor: Pid) -> Self {
|
||||||
self.supervisor = Some(supervisor);
|
self.supervisor = Some(supervisor);
|
||||||
@@ -1032,14 +1032,14 @@ pub fn shutdown<G: GenServer>(name: GenServerName<G>) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Spawn `state` as a server under the current actor (via [`spawn`]). Returns a
|
/// Spawn `state` as a server under the current actor (via [`spawn`](crate::scheduler::spawn)). Returns a
|
||||||
/// [`GenServerRef`]. Shorthand for `GenServerBuilder::new(state).start()`.
|
/// [`GenServerRef`]. Shorthand for `GenServerBuilder::new(state).start()`.
|
||||||
pub fn start<G: GenServer>(state: G) -> GenServerRef<G> {
|
pub fn start<G: GenServer>(state: G) -> GenServerRef<G> {
|
||||||
GenServerBuilder::new(state).start()
|
GenServerBuilder::new(state).start()
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Like [`start`], but spawns the server under an explicit supervisor pid (via
|
/// Like [`start`], but spawns the server under an explicit supervisor pid (via
|
||||||
/// [`spawn_under`]) so it slots into the supervision tree.
|
/// [`spawn_under`](crate::scheduler::spawn_under)) so it slots into the supervision tree.
|
||||||
pub fn start_under<G: GenServer>(supervisor: Pid, state: G) -> GenServerRef<G> {
|
pub fn start_under<G: GenServer>(supervisor: Pid, state: G) -> GenServerRef<G> {
|
||||||
GenServerBuilder::new(state).under(supervisor).start()
|
GenServerBuilder::new(state).under(supervisor).start()
|
||||||
}
|
}
|
||||||
|
|||||||
+2
-2
@@ -191,8 +191,8 @@ pub struct ActorInfo {
|
|||||||
/// [`Stack::new`](crate::stack::Stack::new) rounds them.
|
/// [`Stack::new`](crate::stack::Stack::new) rounds them.
|
||||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||||
pub struct StackInfo {
|
pub struct StackInfo {
|
||||||
/// Usable stack size ([`SpawnOpts::stack_reserve`]
|
/// Usable stack size ([`SpawnOpts::stack_reserve`](crate::SpawnOpts::stack_reserve)
|
||||||
/// (crate::SpawnOpts::stack_reserve) or the Config/default).
|
/// or the Config/default).
|
||||||
pub reserve: usize,
|
pub reserve: usize,
|
||||||
/// PROT_NONE guard below the usable region.
|
/// PROT_NONE guard below the usable region.
|
||||||
pub guard: usize,
|
pub guard: usize,
|
||||||
|
|||||||
+14
-3
@@ -11,9 +11,20 @@
|
|||||||
//!
|
//!
|
||||||
//! See `LOOM.md` for the design intent and the deferred-for-later list.
|
//! See `LOOM.md` for the design intent and the deferred-for-later list.
|
||||||
|
|
||||||
|
// Docs are part of the contract: broken/private intra-doc links fail `cargo doc`,
|
||||||
|
// and every doctest is compiled with `deny(warnings)` under `cargo test --doc`.
|
||||||
|
#![deny(rustdoc::broken_intra_doc_links)]
|
||||||
|
#![deny(rustdoc::private_intra_doc_links)]
|
||||||
|
#![deny(rustdoc::redundant_explicit_links)]
|
||||||
|
#![deny(rustdoc::invalid_codeblock_attributes)]
|
||||||
|
#![deny(rustdoc::invalid_rust_codeblocks)]
|
||||||
|
#![doc(test(attr(deny(warnings))))]
|
||||||
|
|
||||||
pub mod actor;
|
pub mod actor;
|
||||||
pub mod causal;
|
pub mod causal;
|
||||||
pub mod channel;
|
pub mod channel;
|
||||||
|
#[cfg(feature = "cluster")]
|
||||||
|
pub mod cluster;
|
||||||
pub mod context;
|
pub mod context;
|
||||||
pub mod gen_server;
|
pub mod gen_server;
|
||||||
pub mod gen_statem;
|
pub mod gen_statem;
|
||||||
@@ -89,9 +100,9 @@ pub use runtime::{init, Config, Runtime, RuntimeHandle};
|
|||||||
pub use scheduler::{
|
pub use scheduler::{
|
||||||
block_on_io, cancel_timer, request_shutdown, request_stop, run, self_pid, send_after,
|
block_on_io, cancel_timer, request_shutdown, request_stop, run, self_pid, send_after,
|
||||||
send_after_named, send_after_named_wall, send_after_wall, sleep, sleep_wall, spawn, spawn_addr,
|
send_after_named, send_after_named_wall, send_after_wall, sleep, sleep_wall, spawn, spawn_addr,
|
||||||
spawn_addr_with, spawn_under, spawn_under_with, spawn_with, try_spawn, try_spawn_under_with,
|
spawn_addr_with, spawn_monitor, spawn_monitor_with, spawn_under, spawn_under_with, spawn_with,
|
||||||
wait_readable, wait_readable_timeout, wait_writable, wait_writable_timeout, yield_now, FdArm,
|
try_spawn, try_spawn_under_with, wait_readable, wait_readable_timeout, wait_writable,
|
||||||
JoinError, JoinHandle, SpawnError, SpawnOpts,
|
wait_writable_timeout, yield_now, FdArm, JoinError, JoinHandle, SpawnError, SpawnOpts,
|
||||||
};
|
};
|
||||||
pub use supervisor::{ChildSpec, OneForOne, Restart, Shutdown, Signal, Strategy};
|
pub use supervisor::{ChildSpec, OneForOne, Restart, Shutdown, Signal, Strategy};
|
||||||
pub use timer::TimerId;
|
pub use timer::TimerId;
|
||||||
|
|||||||
+77
-42
@@ -145,46 +145,79 @@ pub struct Monitor {
|
|||||||
/// Monitor `target`. Returns a [`Monitor`] whose `rx` receives exactly one
|
/// Monitor `target`. Returns a [`Monitor`] whose `rx` receives exactly one
|
||||||
/// [`Down`].
|
/// [`Down`].
|
||||||
///
|
///
|
||||||
|
/// To monitor a child you are spawning yourself, prefer
|
||||||
|
/// [`spawn_monitor`](crate::spawn_monitor): `spawn` followed by `monitor` on
|
||||||
|
/// the returned pid can race the child's death and observe `NoProc` instead
|
||||||
|
/// of its real reason.
|
||||||
|
///
|
||||||
/// If `target` is still live, the `Down` arrives when it terminates. If
|
/// If `target` is still live, the `Down` arrives when it terminates. If
|
||||||
/// `target` is already gone, a [`DownReason::NoProc`] `Down` is queued
|
/// `target` is already gone, a [`DownReason::NoProc`] `Down` is queued
|
||||||
/// immediately so the caller's `rx.recv()` returns without parking.
|
/// immediately so the caller's `rx.recv()` returns without parking.
|
||||||
pub fn monitor<A>(target: Pid<A>) -> Monitor {
|
pub fn monitor<A>(target: Pid<A>) -> Monitor {
|
||||||
let target = target.erase();
|
let target = target.erase();
|
||||||
let (tx, rx) = channel::<Down>();
|
let (tx, rx) = channel::<Down>();
|
||||||
|
let id = with_runtime(|inner| inner.alloc_monitor_id());
|
||||||
// Implementation note: registration happens under the target's cold
|
if !register_monitor(target, id, &tx) {
|
||||||
// lock. `tx.clone()` takes the channel's own lock, a Channel-class
|
|
||||||
// RawMutex, which is explicitly permitted under a Leaf (cold) lock by
|
|
||||||
// the lock order documented in raw_mutex.rs. We must still not *send*
|
|
||||||
// under the lock, since `Sender::send` can unpark a parked receiver,
|
|
||||||
// and there's no reason to nest that.
|
|
||||||
let (id, registered) = with_runtime(|inner| {
|
|
||||||
let id = inner.alloc_monitor_id();
|
|
||||||
let registered = match inner.slot_at(target) {
|
|
||||||
Some(slot) => {
|
|
||||||
let mut cold = slot.cold.lock();
|
|
||||||
if slot.is_live_for(target) {
|
|
||||||
cold.monitors.push((id, tx.clone()));
|
|
||||||
true
|
|
||||||
} else {
|
|
||||||
false
|
|
||||||
}
|
|
||||||
}
|
|
||||||
None => false,
|
|
||||||
};
|
|
||||||
(id, registered)
|
|
||||||
});
|
|
||||||
|
|
||||||
if !registered {
|
|
||||||
let _ = tx.send(Down {
|
let _ = tx.send(Down {
|
||||||
pid: target,
|
pid: target,
|
||||||
reason: DownReason::NoProc,
|
reason: DownReason::NoProc,
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
Monitor { id, target, rx }
|
Monitor { id, target, rx }
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Register a monitor `id` on `target` that delivers its `Down` to `tx` — the
|
||||||
|
/// primitive under [`monitor`], split out so a caller can fan many monitors
|
||||||
|
/// into ONE channel (process groups: every membership's death lands on the
|
||||||
|
/// reaper's single inbox). Returns `false` if `target` is already gone, in
|
||||||
|
/// which case nothing is registered and the caller decides what to queue
|
||||||
|
/// (`monitor` sends `NoProc`). The caller allocates `id` up front so it can
|
||||||
|
/// record the registration *before* arming it.
|
||||||
|
///
|
||||||
|
/// Implementation note: registration happens under the target's cold lock.
|
||||||
|
/// `tx.clone()` takes the channel's own lock, a Channel-class RawMutex, which
|
||||||
|
/// is explicitly permitted under a Leaf (cold) lock by the lock order
|
||||||
|
/// documented in raw_mutex.rs. We must still not *send* under the lock, since
|
||||||
|
/// `Sender::send` can unpark a parked receiver, and there's no reason to nest
|
||||||
|
/// that.
|
||||||
|
pub(crate) fn register_monitor(target: Pid, id: MonitorId, tx: &Sender<Down>) -> bool {
|
||||||
|
with_runtime(|inner| match inner.slot_at(target) {
|
||||||
|
Some(slot) => {
|
||||||
|
let mut cold = slot.cold.lock();
|
||||||
|
if slot.is_live_for(target) {
|
||||||
|
cold.monitors.push((id, tx.clone()));
|
||||||
|
true
|
||||||
|
} else {
|
||||||
|
false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
None => false,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Remove registration `id` from `target` — the primitive under
|
||||||
|
/// [`demonitor`], for callers that hold only the id (see
|
||||||
|
/// [`register_monitor`]). `None` if the registration is not there: already
|
||||||
|
/// fired, already removed, or the slot has moved on to a new tenant.
|
||||||
|
///
|
||||||
|
/// The registration is removed under the target's cold lock, but the
|
||||||
|
/// `Sender` is moved *out* and dropped only after the lock is released.
|
||||||
|
/// Dropping the last sender runs `Sender::drop`, which may unpark a parked
|
||||||
|
/// receiver; legal under a cold lock, but pointless to nest.
|
||||||
|
pub(crate) fn unregister_monitor(target: Pid, id: MonitorId) -> Option<MonitorId> {
|
||||||
|
let removed: Option<(MonitorId, Sender<Down>)> = with_runtime(|inner| {
|
||||||
|
let slot = inner.slot_at(target)?;
|
||||||
|
let mut cold = slot.cold.lock();
|
||||||
|
if slot.generation() != target.generation() {
|
||||||
|
return None; // slot reused; the Down already fired
|
||||||
|
}
|
||||||
|
let pos = cold.monitors.iter().position(|(mid, _)| *mid == id)?;
|
||||||
|
Some(cold.monitors.remove(pos))
|
||||||
|
});
|
||||||
|
// `removed`'s sender drops here, outside the lock.
|
||||||
|
removed.map(|(id, _sender)| id)
|
||||||
|
}
|
||||||
|
|
||||||
/// Flag `target`'s tenancy as watchable: its death will stamp the slot's
|
/// Flag `target`'s tenancy as watchable: its death will stamp the slot's
|
||||||
/// terminal record (see [`terminal_reason`]), exactly as registering a name
|
/// terminal record (see [`terminal_reason`]), exactly as registering a name
|
||||||
/// does. The bridge calls this wherever a smarm pid is *encoded across the
|
/// does. The bridge calls this wherever a smarm pid is *encoded across the
|
||||||
@@ -215,6 +248,23 @@ pub fn mark_watchable<A>(target: Pid<A>) {
|
|||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Whether `target` is live *and* its tenancy is watchable. The cluster's
|
||||||
|
/// remote-monitor admission check (RFC 010 c12): a peer may monitor a pid only
|
||||||
|
/// if that pid was exposed or crossed the wire (the D12 set-sites), and a
|
||||||
|
/// live-but-unwatchable pid answers exactly like a dead one — no liveness leak
|
||||||
|
/// beyond what `watchable` already grants. Same context contract as
|
||||||
|
/// [`monitor`].
|
||||||
|
#[cfg(feature = "cluster")]
|
||||||
|
pub(crate) fn is_watchable<A>(target: Pid<A>) -> bool {
|
||||||
|
let target = target.erase();
|
||||||
|
with_runtime(|inner| {
|
||||||
|
inner.slot_at(target).is_some_and(|slot| {
|
||||||
|
let cold = slot.cold.lock();
|
||||||
|
slot.is_live_for(target) && cold.watchable
|
||||||
|
})
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
/// The terminal [`DownReason`] of the tenancy `target` names, if that tenancy
|
/// The terminal [`DownReason`] of the tenancy `target` names, if that tenancy
|
||||||
/// ever registered a name and is the *most recent named* death of its slot:
|
/// ever registered a name and is the *most recent named* death of its slot:
|
||||||
/// finalize stamps the slot with `(generation, reason)` for once-registered
|
/// finalize stamps the slot with `(generation, reason)` for once-registered
|
||||||
@@ -254,20 +304,5 @@ pub fn terminal_reason<A>(target: Pid<A>) -> Option<DownReason> {
|
|||||||
/// instead of, or in addition to, calling this: dropping the [`Monitor`]
|
/// instead of, or in addition to, calling this: dropping the [`Monitor`]
|
||||||
/// closes its receiver and any queued notice is discarded with it.
|
/// closes its receiver and any queued notice is discarded with it.
|
||||||
pub fn demonitor(m: &Monitor) -> Option<MonitorId> {
|
pub fn demonitor(m: &Monitor) -> Option<MonitorId> {
|
||||||
// Implementation note: the registration is removed under the target's
|
unregister_monitor(m.target, m.id)
|
||||||
// cold lock, but the `Sender` is moved *out* and dropped only after the
|
|
||||||
// lock is released. Dropping the last sender runs `Sender::drop`, which
|
|
||||||
// may unpark a parked receiver; legal under a cold lock, but pointless
|
|
||||||
// to nest.
|
|
||||||
let removed: Option<(MonitorId, Sender<Down>)> = with_runtime(|inner| {
|
|
||||||
let slot = inner.slot_at(m.target)?;
|
|
||||||
let mut cold = slot.cold.lock();
|
|
||||||
if slot.generation() != m.target.generation() {
|
|
||||||
return None; // slot reused; the Down already fired
|
|
||||||
}
|
|
||||||
let pos = cold.monitors.iter().position(|(mid, _)| *mid == m.id)?;
|
|
||||||
Some(cold.monitors.remove(pos))
|
|
||||||
});
|
|
||||||
// `removed`'s sender drops here, outside the lock.
|
|
||||||
removed.map(|(id, _sender)| id)
|
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -99,10 +99,13 @@
|
|||||||
//! ## Identity and clustering
|
//! ## Identity and clustering
|
||||||
//!
|
//!
|
||||||
//! A group member is described by a [`Member`] — a [`Pid`] plus a [`NodeId`] and
|
//! A group member is described by a [`Member`] — a [`Pid`] plus a [`NodeId`] and
|
||||||
//! an [`Incarnation`]. Today everything is single-node, those two fields are
|
//! an [`Incarnation`]. Everything on this page is **local**: you pass and
|
||||||
//! fixed defaults, and you only ever pass and receive a plain [`Pid`]: the extra
|
//! receive plain [`Pid`]s, and [`members`] / [`pick`] / [`dispatch`] only ever
|
||||||
//! identity is carried so this API will not have to change when groups learn to
|
//! name actors on this node (Erlang's `get_local_members`). With the `cluster`
|
||||||
//! span a cluster.
|
//! feature a group also holds the members other nodes have announced, carried
|
||||||
|
//! under their [`NodeId`]; those never surface here — the cluster-wide reads
|
||||||
|
//! live in [`cluster::pg`](crate::cluster::pg) (`members_all` and friends) and
|
||||||
|
//! return a `Local | Remote` member type, since a [`Pid`] cannot hold a remote.
|
||||||
//!
|
//!
|
||||||
//! ## Running context
|
//! ## Running context
|
||||||
//!
|
//!
|
||||||
@@ -110,10 +113,11 @@
|
|||||||
//! from inside [`run`](crate::run) (that is, on an actor thread). Calling one
|
//! from inside [`run`](crate::run) (that is, on an actor thread). Calling one
|
||||||
//! from outside a running runtime panics.
|
//! from outside a running runtime panics.
|
||||||
|
|
||||||
use crate::monitor::{demonitor, monitor, Monitor};
|
use crate::channel::{channel, Sender};
|
||||||
|
use crate::monitor::{register_monitor, unregister_monitor, Down, DownReason, MonitorId};
|
||||||
use crate::pid::{assert_type, Addressable, Pid};
|
use crate::pid::{assert_type, Addressable, Pid};
|
||||||
use crate::registry::{send_to, SendError};
|
use crate::registry::{send_to, SendError};
|
||||||
use crate::scheduler::with_runtime;
|
use crate::scheduler::{spawn_under, with_runtime};
|
||||||
use std::collections::HashMap;
|
use std::collections::HashMap;
|
||||||
|
|
||||||
/// A cluster node handle. A `u32` integer handle, *not* an interned atom — the
|
/// A cluster node handle. A `u32` integer handle, *not* an interned atom — the
|
||||||
@@ -186,13 +190,15 @@ pub struct Member {
|
|||||||
pub pid: Pid,
|
pub pid: Pid,
|
||||||
}
|
}
|
||||||
|
|
||||||
/// One membership: a [`Member`] and the [`Monitor`] that watches its liveness.
|
/// One membership: a [`Member`] and the id of the monitor that watches its
|
||||||
/// The monitor lives *alongside* the group entry so a group is
|
/// liveness. The monitor's `Down` is delivered to the group reaper's single
|
||||||
/// self-contained: draining the membership tells us whether the member is
|
/// inbox (see [`ProcessGroups::deaths`]), so the membership carries only what
|
||||||
/// still alive, and dropping the membership drops its monitor.
|
/// [`leave`] needs to tear the registration down: the id.
|
||||||
struct Membership {
|
pub(crate) struct Membership {
|
||||||
member: Member,
|
pub(crate) member: Member,
|
||||||
monitor: Monitor,
|
/// `None` for a remote member (cluster): the origin node is its liveness
|
||||||
|
/// authority; nothing here watches it.
|
||||||
|
pub(crate) monitor: Option<MonitorId>,
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The store: `name → multiset<Member>`. Within a single group a `Member`
|
/// The store: `name → multiset<Member>`. Within a single group a `Member`
|
||||||
@@ -202,46 +208,60 @@ struct Membership {
|
|||||||
///
|
///
|
||||||
/// Locking discipline. Held under one Leaf-class `RawMutex` on `RuntimeInner`,
|
/// Locking discipline. Held under one Leaf-class `RawMutex` on `RuntimeInner`,
|
||||||
/// mirroring the registry, and never held together with another Leaf lock (it
|
/// mirroring the registry, and never held together with another Leaf lock (it
|
||||||
/// never touches the registry or a slot's cold lock). The two operations that
|
/// never touches the registry or a slot's cold lock). Monitor registration and
|
||||||
/// do need another lock are kept off the group-lock path:
|
/// removal take the target's cold lock (also Leaf), so they run *before* /
|
||||||
|
/// *after* the group lock, never under it — see [`join`] for the ordering that
|
||||||
|
/// makes that safe. Nothing under this lock ever touches a channel.
|
||||||
///
|
///
|
||||||
/// - `monitor()` / `demonitor()` take the target's cold lock (also Leaf), so
|
/// Eviction is *eager*: every membership's monitor delivers to the one
|
||||||
/// they run *before* / *after* the group lock, never under it.
|
/// `deaths` channel, drained by a per-run reaper actor that sweeps the dead
|
||||||
/// - draining a monitor with `try_recv` takes the channel's Channel-class
|
/// pid out of every group the moment its `Down` is scheduled. The read path
|
||||||
/// lock, which the lock order permits *under* a Leaf; a channel critical
|
/// keeps a slot-liveness backstop for the window between a death and the
|
||||||
/// section only does the lock-free unpark protocol, so no Leaf ever nests
|
/// reaper's turn.
|
||||||
/// under it.
|
|
||||||
///
|
|
||||||
/// Evicted and rejected [`Monitor`]s are therefore dropped only *after* the
|
|
||||||
/// group lock is released, so a receiver-drop never runs a wakeup under the
|
|
||||||
/// lock — the same discipline as `demonitor`.
|
|
||||||
pub(crate) struct ProcessGroups {
|
pub(crate) struct ProcessGroups {
|
||||||
groups: HashMap<String, Vec<Membership>>,
|
groups: HashMap<String, Vec<Membership>>,
|
||||||
|
/// The reaper's inboxes: every membership monitor is registered against
|
||||||
|
/// a clone of `deaths`. `None` until the first `join` of a run spawns
|
||||||
|
/// the reaper; a stale one (receiver gone with the previous run's
|
||||||
|
/// teardown) is detected via `receiver_alive` and replaced.
|
||||||
|
reaper: Option<ReaperInboxes>,
|
||||||
|
/// `NodeId → node name` for every peer with members in the store, kept
|
||||||
|
/// by the pg actor under this lock, so a stored remote member can be
|
||||||
|
/// rendered back to its wire identity without asking anyone.
|
||||||
|
#[cfg(feature = "cluster")]
|
||||||
|
node_names: HashMap<NodeId, String>,
|
||||||
}
|
}
|
||||||
|
|
||||||
impl ProcessGroups {
|
impl ProcessGroups {
|
||||||
pub(crate) fn new() -> Self {
|
pub(crate) fn new() -> Self {
|
||||||
Self {
|
Self {
|
||||||
groups: HashMap::new(),
|
groups: HashMap::new(),
|
||||||
|
reaper: None,
|
||||||
|
#[cfg(feature = "cluster")]
|
||||||
|
node_names: HashMap::new(),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Insert `ms` into `group`. Idempotent on the *member*: if the member is
|
/// Forget the reaper. Called at the start of every `run()` so a stopped
|
||||||
/// already present the new membership is handed back (`Some`) so the caller
|
/// reaper from a previous run is never sent to; `join` respawns.
|
||||||
/// can tear its now-redundant monitor down outside the lock; `None` means
|
pub(crate) fn reset_reaper(&mut self) {
|
||||||
/// it was inserted.
|
self.reaper = None;
|
||||||
fn join(&mut self, group: &str, ms: Membership) -> Option<Membership> {
|
}
|
||||||
|
|
||||||
|
/// Insert `ms` into `group`. Idempotent on the *member*: `false` means the
|
||||||
|
/// member was already present and nothing changed; `true` means inserted.
|
||||||
|
pub(crate) fn join(&mut self, group: &str, ms: Membership) -> bool {
|
||||||
let v = self.groups.entry(group.to_owned()).or_default();
|
let v = self.groups.entry(group.to_owned()).or_default();
|
||||||
if v.iter().any(|e| e.member == ms.member) {
|
if v.iter().any(|e| e.member == ms.member) {
|
||||||
return Some(ms);
|
return false;
|
||||||
}
|
}
|
||||||
v.push(ms);
|
v.push(ms);
|
||||||
None
|
true
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Remove `member`'s membership from `group`, returning it (so the caller
|
/// Remove `member`'s membership from `group`, returning it (so the caller
|
||||||
/// can `demonitor` it outside the lock). An emptied group is pruned.
|
/// can unregister its monitor outside the lock). An emptied group is pruned.
|
||||||
fn leave(&mut self, group: &str, member: Member) -> Option<Membership> {
|
pub(crate) fn leave(&mut self, group: &str, member: Member) -> Option<Membership> {
|
||||||
let v = self.groups.get_mut(group)?;
|
let v = self.groups.get_mut(group)?;
|
||||||
let pos = v.iter().position(|e| e.member == member)?;
|
let pos = v.iter().position(|e| e.member == member)?;
|
||||||
let removed = v.remove(pos);
|
let removed = v.remove(pos);
|
||||||
@@ -252,20 +272,23 @@ impl ProcessGroups {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// The one dumb eviction primitive: drop every member matching `pred` from
|
/// The one dumb eviction primitive: drop every member matching `pred` from
|
||||||
/// every group, pruning emptied groups, and return the evicted memberships'
|
/// every group, pruning emptied groups, and return the evicted
|
||||||
/// monitors for the caller to drop outside the lock. The primitive does not
|
/// memberships with the group each was in. The primitive does not know
|
||||||
/// know *why* a member leaves; that is the caller's concern. Its callers are
|
/// *why* a member leaves; that is the caller's concern. Its callers are
|
||||||
/// the death hook (`reap_group`) and, once clustering lands, an
|
/// the reaper (a local death) and the cluster's node-down / re-sync
|
||||||
/// incarnation-eviction sweep — both over this same predicate path, which is
|
/// sweeps — all over this same predicate path, which is the whole reason
|
||||||
/// the whole reason to shape eviction as a predicate. Insertion order within
|
/// to shape eviction as a predicate. Insertion order within a group is
|
||||||
/// a group is preserved (`members` / `pick` are order-stable).
|
/// preserved (`members` / `pick` are order-stable).
|
||||||
fn remove_where(&mut self, mut pred: impl FnMut(&Member) -> bool) -> Vec<Monitor> {
|
pub(crate) fn remove_where(
|
||||||
|
&mut self,
|
||||||
|
mut pred: impl FnMut(&Member) -> bool,
|
||||||
|
) -> Vec<(String, Membership)> {
|
||||||
let mut evicted = Vec::new();
|
let mut evicted = Vec::new();
|
||||||
self.groups.retain(|_, v| {
|
self.groups.retain(|g, v| {
|
||||||
let mut i = 0;
|
let mut i = 0;
|
||||||
while i < v.len() {
|
while i < v.len() {
|
||||||
if pred(&v[i].member) {
|
if pred(&v[i].member) {
|
||||||
evicted.push(v.remove(i).monitor);
|
evicted.push((g.clone(), v.remove(i)));
|
||||||
} else {
|
} else {
|
||||||
i += 1;
|
i += 1;
|
||||||
}
|
}
|
||||||
@@ -275,38 +298,6 @@ impl ProcessGroups {
|
|||||||
evicted
|
evicted
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Drain-on-contact death hook. The registry can prune a stale binding
|
|
||||||
/// lazily, on contact, because it only ever resolves one binding at a time;
|
|
||||||
/// a group is *iterated* — `members` fans out to everyone — so it must not
|
|
||||||
/// carry a dead member across a broadcast. Every group operation reaps the
|
|
||||||
/// group it touches first.
|
|
||||||
///
|
|
||||||
/// Drains every membership monitor in `group` with a non-blocking
|
|
||||||
/// `try_recv`: a delivered `Down` (any reason) or a closed channel means
|
|
||||||
/// that member is dead. On the first death detected, sweep *all* of the
|
|
||||||
/// dead pids out of *every* group via [`remove_where`] — a death is removed
|
|
||||||
/// from each group it joined, not just the one being touched. Returns the
|
|
||||||
/// evicted monitors to drop outside the lock.
|
|
||||||
fn reap_group(&mut self, group: &str) -> Vec<Monitor> {
|
|
||||||
let dead: Vec<Pid> = {
|
|
||||||
let Some(v) = self.groups.get(group) else {
|
|
||||||
return Vec::new();
|
|
||||||
};
|
|
||||||
v.iter()
|
|
||||||
.filter_map(|e| match e.monitor.rx.try_recv() {
|
|
||||||
// A Down arrived, or the channel closed and drained: dead.
|
|
||||||
Ok(Some(_)) | Err(_) => Some(e.member.pid),
|
|
||||||
// Empty but open — the sender still lives in the slot: alive.
|
|
||||||
Ok(None) => None,
|
|
||||||
})
|
|
||||||
.collect()
|
|
||||||
};
|
|
||||||
if dead.is_empty() {
|
|
||||||
return Vec::new();
|
|
||||||
}
|
|
||||||
self.remove_where(|m| dead.contains(&m.pid))
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Raw enumeration of a group's members — no liveness filtering. Used by
|
/// Raw enumeration of a group's members — no liveness filtering. Used by
|
||||||
/// tests to assert storage state independently of the read-path backstop.
|
/// tests to assert storage state independently of the read-path backstop.
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
@@ -317,16 +308,24 @@ impl ProcessGroups {
|
|||||||
.unwrap_or_default()
|
.unwrap_or_default()
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Live members of `group`, in insertion order. The `is_live` oracle is the
|
/// Live members of `group` **on `node`**, in insertion order. The
|
||||||
/// read-path backstop: a member whose slot is already dead is
|
/// `is_live` oracle is the read-path backstop: a member whose slot is
|
||||||
/// dropped from the *result* even if its `Down` has not been drained yet.
|
/// already dead is dropped from the *result* even if the reaper has not
|
||||||
/// Backstop only — the entry stays in storage; eviction is the monitor's
|
/// swept it yet. Backstop only — the entry stays in storage; eviction is
|
||||||
/// job (`reap_group`).
|
/// the reaper's job. The node filter is what keeps the local API local:
|
||||||
fn members_where(&self, group: &str, mut is_live: impl FnMut(Pid) -> bool) -> Vec<Pid> {
|
/// a remote member's `pid` is another node's slot bits, meaningless to
|
||||||
|
/// `is_live` and to any local send.
|
||||||
|
fn members_where(
|
||||||
|
&self,
|
||||||
|
group: &str,
|
||||||
|
node: NodeId,
|
||||||
|
mut is_live: impl FnMut(Pid) -> bool,
|
||||||
|
) -> Vec<Pid> {
|
||||||
self.groups
|
self.groups
|
||||||
.get(group)
|
.get(group)
|
||||||
.map(|v| {
|
.map(|v| {
|
||||||
v.iter()
|
v.iter()
|
||||||
|
.filter(|e| e.member.node == node)
|
||||||
.map(|e| e.member.pid)
|
.map(|e| e.member.pid)
|
||||||
.filter(|&p| is_live(p))
|
.filter(|&p| is_live(p))
|
||||||
.collect()
|
.collect()
|
||||||
@@ -334,19 +333,210 @@ impl ProcessGroups {
|
|||||||
.unwrap_or_default()
|
.unwrap_or_default()
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The first live member of `group` in insertion order — stateless
|
/// The first live member of `group` on `node` in insertion order —
|
||||||
/// first-live `pick`, with the same read-path backstop as `members_where`.
|
/// stateless first-live `pick`, with the same read-path backstop and node
|
||||||
fn first_member_where(&self, group: &str, mut is_live: impl FnMut(Pid) -> bool) -> Option<Pid> {
|
/// filter as `members_where`.
|
||||||
|
fn first_member_where(
|
||||||
|
&self,
|
||||||
|
group: &str,
|
||||||
|
node: NodeId,
|
||||||
|
mut is_live: impl FnMut(Pid) -> bool,
|
||||||
|
) -> Option<Pid> {
|
||||||
self.groups
|
self.groups
|
||||||
.get(group)?
|
.get(group)?
|
||||||
.iter()
|
.iter()
|
||||||
|
.filter(|e| e.member.node == node)
|
||||||
.map(|e| e.member.pid)
|
.map(|e| e.member.pid)
|
||||||
.find(|&p| is_live(p))
|
.find(|&p| is_live(p))
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The store's cluster-side surface: raw reads the pg actor needs to speak
|
||||||
|
/// for this node (`Sync`, membership checks) and the peer-name memo. One
|
||||||
|
/// `cfg` block: everything here exists only when there is a mesh.
|
||||||
|
#[cfg(feature = "cluster")]
|
||||||
|
impl ProcessGroups {
|
||||||
|
/// Does `group` hold `member` right now? (Raw storage, no liveness.)
|
||||||
|
pub(crate) fn contains(&self, group: &str, member: &Member) -> bool {
|
||||||
|
self.groups
|
||||||
|
.get(group)
|
||||||
|
.is_some_and(|v| v.iter().any(|e| e.member == *member))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Every stored member of `group`, any node, insertion order. Raw storage.
|
||||||
|
pub(crate) fn all_of(&self, group: &str) -> Vec<Member> {
|
||||||
|
self.groups
|
||||||
|
.get(group)
|
||||||
|
.map(|v| v.iter().map(|e| e.member).collect())
|
||||||
|
.unwrap_or_default()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `(group, [pid])` for every group with a member on `node` — the
|
||||||
|
/// `Sync` payload. Raw storage; groups with no such member are omitted.
|
||||||
|
pub(crate) fn groups_on(&self, node: NodeId) -> Vec<(String, Vec<Pid>)> {
|
||||||
|
let mut out: Vec<(String, Vec<Pid>)> = self
|
||||||
|
.groups
|
||||||
|
.iter()
|
||||||
|
.filter_map(|(g, v)| {
|
||||||
|
let pids: Vec<Pid> = v
|
||||||
|
.iter()
|
||||||
|
.filter(|e| e.member.node == node)
|
||||||
|
.map(|e| e.member.pid)
|
||||||
|
.collect();
|
||||||
|
(!pids.is_empty()).then(|| (g.clone(), pids))
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
out.sort_by(|a, b| a.0.cmp(&b.0));
|
||||||
|
out
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Record / forget the name behind a peer's `NodeId`.
|
||||||
|
pub(crate) fn set_node_name(&mut self, node: NodeId, name: String) {
|
||||||
|
self.node_names.insert(node, name);
|
||||||
|
}
|
||||||
|
pub(crate) fn forget_node_name(&mut self, node: NodeId) {
|
||||||
|
self.node_names.remove(&node);
|
||||||
|
}
|
||||||
|
pub(crate) fn node_name(&self, node: NodeId) -> Option<&str> {
|
||||||
|
self.node_names.get(&node).map(String::as_str)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The group reaper: one detached actor per run, spawned by the first `join`,
|
||||||
|
/// parked on the shared `deaths` inbox. Every local membership's monitor
|
||||||
|
/// delivers here, so a death is swept out of *every* group it joined as soon
|
||||||
|
/// as the reaper is scheduled — no group operation has to happen first.
|
||||||
|
/// Sweeps by `(node, pid)`: only local members, since a remote member's pid
|
||||||
|
/// bits are meaningless here. Exits when the last sender is gone, i.e. never
|
||||||
|
/// during a run (the store holds one); the run's teardown stops it like any
|
||||||
|
/// other parked actor. Spawned under `ROOT_PID` so its exit signal is absorbed
|
||||||
|
/// rather than delivered to whichever supervisor's child happened to join
|
||||||
|
/// first.
|
||||||
|
///
|
||||||
|
/// Under `cluster` the same actor is the node's **pg actor** (RFC 010 Phase
|
||||||
|
/// 5, c15): it also drains a control inbox of local join/leave announcements,
|
||||||
|
/// the membership stream and the exposed `"pg"` inbox — see
|
||||||
|
/// [`crate::cluster::pg`]. Its store-side sweep is unchanged.
|
||||||
|
#[cfg(not(feature = "cluster"))]
|
||||||
|
fn reaper(rx: crate::channel::Receiver<Down>, ctl: crate::channel::Receiver<PgEvent>) {
|
||||||
|
// No mesh: nothing to tell about joins/leaves. Drop the control inbox
|
||||||
|
// so announcements are refused at the sender rather than queued.
|
||||||
|
drop(ctl);
|
||||||
|
while let Ok(down) = rx.recv() {
|
||||||
|
sweep_local_death(down.pid);
|
||||||
|
// Evicted memberships hold only ids; their monitors have fired.
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// What the local API tells the reaper besides deaths (which arrive as
|
||||||
|
/// [`Down`] on their own inbox — that channel's type is fixed by the monitor
|
||||||
|
/// primitive, so the two cannot be one enum). The default reaper has no use
|
||||||
|
/// for these; the cluster's pg actor broadcasts them (RFC 010 Phase 5).
|
||||||
|
// The default reaper never looks inside — that is the point, not a bug.
|
||||||
|
#[cfg_attr(not(feature = "cluster"), allow(dead_code))]
|
||||||
|
pub(crate) enum PgEvent {
|
||||||
|
/// `join` inserted `pid` into `group`. The consumer re-checks the store
|
||||||
|
/// before acting on it.
|
||||||
|
Joined { group: String, pid: Pid },
|
||||||
|
/// `leave` removed `pid` from `group`.
|
||||||
|
Left { group: String, pid: Pid },
|
||||||
|
/// `cluster::start` has the manager up and the local identity set: take
|
||||||
|
/// a membership subscription, register + expose the `"pg"` name, and
|
||||||
|
/// start speaking to peers.
|
||||||
|
#[cfg(feature = "cluster")]
|
||||||
|
Attach,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Evict the local member `pid` from every group. The reaper's one store
|
||||||
|
/// operation; returns what was evicted with its group (the cluster's
|
||||||
|
/// `Leave` broadcast wants both).
|
||||||
|
pub(crate) fn sweep_local_death(pid: Pid) -> Vec<(String, Membership)> {
|
||||||
|
with_runtime(|inner| {
|
||||||
|
let node = inner.node_id;
|
||||||
|
inner
|
||||||
|
.process_groups
|
||||||
|
.lock()
|
||||||
|
.remove_where(|m| m.node == node && m.pid == pid)
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The reaper's inboxes. `deaths` is the liveness authority for the set
|
||||||
|
/// (`ctl` is created and dropped with it, on the same actor).
|
||||||
|
#[derive(Clone)]
|
||||||
|
pub(crate) struct ReaperInboxes {
|
||||||
|
pub(crate) deaths: Sender<Down>,
|
||||||
|
/// The control inbox: local `join`/`leave` announce here (see
|
||||||
|
/// [`PgEvent`]). The default reaper closes it on entry.
|
||||||
|
pub(crate) ctl: Sender<PgEvent>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl ReaperInboxes {
|
||||||
|
fn alive(&self) -> bool {
|
||||||
|
self.deaths.receiver_alive()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Live senders for the reaper's inboxes, spawning the reaper if this run has
|
||||||
|
/// none yet. Two racing first-spawns may both spawn; the loser's senders drop
|
||||||
|
/// on return, its spare reaper sees a closed inbox and exits.
|
||||||
|
pub(crate) fn reaper_inboxes() -> ReaperInboxes {
|
||||||
|
let existing = with_runtime(|inner| {
|
||||||
|
let pg = inner.process_groups.lock();
|
||||||
|
pg.reaper.clone().filter(ReaperInboxes::alive)
|
||||||
|
});
|
||||||
|
if let Some(r) = existing {
|
||||||
|
return r;
|
||||||
|
}
|
||||||
|
let (tx, rx) = channel::<Down>();
|
||||||
|
let (ctl_tx, ctl_rx) = channel::<PgEvent>();
|
||||||
|
// Detached: the handle drops here. The reaper's lifetime is the run's.
|
||||||
|
// The ONE seam between the local store and the cluster: same inboxes,
|
||||||
|
// different body.
|
||||||
|
#[cfg(not(feature = "cluster"))]
|
||||||
|
let _ = spawn_under(crate::runtime::ROOT_PID, move || reaper(rx, ctl_rx));
|
||||||
|
#[cfg(feature = "cluster")]
|
||||||
|
let _ = spawn_under(crate::runtime::ROOT_PID, move || {
|
||||||
|
crate::cluster::pg::actor(rx, ctl_rx)
|
||||||
|
});
|
||||||
|
let fresh = ReaperInboxes {
|
||||||
|
deaths: tx,
|
||||||
|
ctl: ctl_tx,
|
||||||
|
};
|
||||||
|
with_runtime(|inner| {
|
||||||
|
let mut pg = inner.process_groups.lock();
|
||||||
|
match &pg.reaper {
|
||||||
|
Some(r) if r.alive() => r.clone(),
|
||||||
|
_ => {
|
||||||
|
pg.reaper = Some(fresh.clone());
|
||||||
|
fresh
|
||||||
|
}
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A live sender for the reaper's `deaths` inbox (spawning it if needed).
|
||||||
|
fn deaths_sender() -> Sender<Down> {
|
||||||
|
reaper_inboxes().deaths
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Announce a local group change to the reaper, if this run has one. A
|
||||||
|
/// closed inbox is the default reaper (uninterested) or a run tearing down.
|
||||||
|
fn announce(msg: PgEvent) {
|
||||||
|
let ctl = with_runtime(|inner| {
|
||||||
|
inner
|
||||||
|
.process_groups
|
||||||
|
.lock()
|
||||||
|
.reaper
|
||||||
|
.as_ref()
|
||||||
|
.map(|r| r.ctl.clone())
|
||||||
|
});
|
||||||
|
if let Some(ctl) = ctl {
|
||||||
|
let _ = ctl.send(msg);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// Build the full member identity for `pid` from runtime identity.
|
/// Build the full member identity for `pid` from runtime identity.
|
||||||
fn member_for(inner: &crate::runtime::RuntimeInner, pid: Pid) -> Member {
|
pub(crate) fn member_for(inner: &crate::runtime::RuntimeInner, pid: Pid) -> Member {
|
||||||
Member {
|
Member {
|
||||||
node: inner.node_id,
|
node: inner.node_id,
|
||||||
incarnation: inner.incarnation,
|
incarnation: inner.incarnation,
|
||||||
@@ -358,7 +548,7 @@ fn member_for(inner: &crate::runtime::RuntimeInner, pid: Pid) -> Member {
|
|||||||
/// no lock — identical to the registry's guard. The read-path backstop: a
|
/// no lock — identical to the registry's guard. The read-path backstop: a
|
||||||
/// generation is never reused, so a dead member is detectable independently of
|
/// generation is never reused, so a dead member is detectable independently of
|
||||||
/// whether its monitor `Down` has been drained yet.
|
/// whether its monitor `Down` has been drained yet.
|
||||||
fn live(inner: &crate::runtime::RuntimeInner, pid: Pid) -> bool {
|
pub(crate) fn live(inner: &crate::runtime::RuntimeInner, pid: Pid) -> bool {
|
||||||
inner.slot_at(pid).is_some_and(|s| s.is_live_for(pid))
|
inner.slot_at(pid).is_some_and(|s| s.is_live_for(pid))
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -367,61 +557,76 @@ fn live(inner: &crate::runtime::RuntimeInner, pid: Pid) -> bool {
|
|||||||
/// added the membership, `false` if it was already a member.
|
/// added the membership, `false` if it was already a member.
|
||||||
///
|
///
|
||||||
/// Installs a monitor on `pid` so the actor's death evicts it from the group
|
/// Installs a monitor on `pid` so the actor's death evicts it from the group
|
||||||
/// automatically — you never have to remove a dead member yourself. A redundant
|
/// automatically — you never have to remove a dead member yourself. Joining a
|
||||||
/// (idempotent) join tears its extra monitor back down.
|
/// pid that is already dead is accepted and evicted the same way (via a
|
||||||
|
/// `NoProc` notice), so it never shows up in a read.
|
||||||
///
|
///
|
||||||
/// Panics if called outside `Runtime::run()`.
|
/// Panics if called outside `Runtime::run()`.
|
||||||
pub fn join<A>(group: impl Into<String>, pid: Pid<A>) -> bool {
|
pub fn join<A>(group: impl Into<String>, pid: Pid<A>) -> bool {
|
||||||
let group = group.into();
|
let group = group.into();
|
||||||
let pid = pid.erase();
|
let pid = pid.erase();
|
||||||
// Install the monitor BEFORE taking the group lock: monitor() acquires the
|
let deaths = deaths_sender();
|
||||||
// target's cold lock (Leaf), and two Leaf locks are never held at once. The
|
// Record the membership BEFORE arming its monitor: the reaper sweeps by
|
||||||
// registration races `finalize_actor` under that cold lock exactly as every
|
// pid on the first `Down`, so a `Down` that could precede the entry would
|
||||||
// other monitor does, so no death can slip between the join and the monitor
|
// leave a corpse in storage forever (visible to no read — the backstop
|
||||||
// being in place.
|
// hides it — but a leak, and once groups are clustered a member that
|
||||||
let mon = monitor(pid);
|
// would be announced). Arming after insertion means every `Down` finds
|
||||||
|
// its entry. The monitor id is allocated up front so `leave` can tear the
|
||||||
let (rejected, reaped) = with_runtime(|inner| {
|
// registration down even if it lands in the tiny window before arming (an
|
||||||
|
// orphaned registration is harmless: its `Down` names a pid whose
|
||||||
|
// membership is gone, and the sweep finds nothing).
|
||||||
|
let id = with_runtime(|inner| inner.alloc_monitor_id());
|
||||||
|
let inserted = with_runtime(|inner| {
|
||||||
let ms = Membership {
|
let ms = Membership {
|
||||||
member: member_for(inner, pid),
|
member: member_for(inner, pid),
|
||||||
monitor: mon,
|
monitor: Some(id),
|
||||||
};
|
};
|
||||||
let mut pg = inner.process_groups.lock();
|
inner.process_groups.lock().join(&group, ms)
|
||||||
let reaped = pg.reap_group(&group);
|
});
|
||||||
let rejected = pg.join(&group, ms);
|
if !inserted {
|
||||||
(rejected, reaped)
|
return false;
|
||||||
|
}
|
||||||
|
// Tell the reaper (the cluster's pg actor re-checks the store before it
|
||||||
|
// broadcasts, so a `leave`/death that overtakes this announcement is
|
||||||
|
// never advertised as a join).
|
||||||
|
announce(PgEvent::Joined {
|
||||||
|
group: group.clone(),
|
||||||
|
pid,
|
||||||
});
|
});
|
||||||
|
|
||||||
// Outside the group lock: drop the reaped (dead) monitors, and if this join
|
// Outside the group lock: registration takes the target's cold lock (Leaf).
|
||||||
// was redundant, demonitor + drop the extra monitor we just installed.
|
// The registration races `finalize_actor` under that cold lock exactly as
|
||||||
drop(reaped);
|
// every other monitor does, so no death can slip between the join and the
|
||||||
match rejected {
|
// monitor being in place.
|
||||||
Some(dup) => {
|
if !register_monitor(pid, id, &deaths) {
|
||||||
demonitor(&dup.monitor);
|
// Already gone: queue the notice ourselves, exactly as `monitor` does.
|
||||||
false
|
let _ = deaths.send(Down {
|
||||||
}
|
pid,
|
||||||
None => true,
|
reason: DownReason::NoProc,
|
||||||
|
});
|
||||||
}
|
}
|
||||||
|
true
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Drop `pid`'s membership of `group`. Returns whether a membership was
|
/// Drop `pid`'s membership of `group`. Returns whether a membership was
|
||||||
/// removed. The membership's monitor is demonitored and dropped.
|
/// removed. The membership's monitor registration is torn down.
|
||||||
///
|
///
|
||||||
/// Panics if called outside `Runtime::run()`.
|
/// Panics if called outside `Runtime::run()`.
|
||||||
pub fn leave<A>(group: &str, pid: Pid<A>) -> bool {
|
pub fn leave<A>(group: &str, pid: Pid<A>) -> bool {
|
||||||
let pid = pid.erase();
|
let pid = pid.erase();
|
||||||
let (removed, reaped) = with_runtime(|inner| {
|
let removed = with_runtime(|inner| {
|
||||||
let member = member_for(inner, pid);
|
let member = member_for(inner, pid);
|
||||||
let mut pg = inner.process_groups.lock();
|
inner.process_groups.lock().leave(group, member)
|
||||||
let reaped = pg.reap_group(group);
|
|
||||||
let removed = pg.leave(group, member);
|
|
||||||
(removed, reaped)
|
|
||||||
});
|
});
|
||||||
|
|
||||||
drop(reaped);
|
|
||||||
match removed {
|
match removed {
|
||||||
Some(ms) => {
|
Some(ms) => {
|
||||||
demonitor(&ms.monitor);
|
if let Some(id) = ms.monitor {
|
||||||
|
unregister_monitor(pid, id);
|
||||||
|
}
|
||||||
|
announce(PgEvent::Left {
|
||||||
|
group: group.to_owned(),
|
||||||
|
pid,
|
||||||
|
});
|
||||||
true
|
true
|
||||||
}
|
}
|
||||||
None => false,
|
None => false,
|
||||||
@@ -431,21 +636,19 @@ pub fn leave<A>(group: &str, pid: Pid<A>) -> bool {
|
|||||||
/// Every live member of `group`, in the order they joined. Returns an empty
|
/// Every live member of `group`, in the order they joined. Returns an empty
|
||||||
/// vector if the group does not exist or has no live members.
|
/// vector if the group does not exist or has no live members.
|
||||||
///
|
///
|
||||||
/// Dead members are never returned: the group is pruned of anything that has
|
/// Dead members are never returned: the reaper evicts a member as soon as its
|
||||||
/// died before the read, and as a backstop a member whose slot is already dead
|
/// death is processed, and as a backstop a member whose slot is already dead
|
||||||
/// is dropped from the result even in the brief window before its death has
|
/// is dropped from the result even in the brief window before the reaper's
|
||||||
/// been fully processed.
|
/// turn.
|
||||||
///
|
///
|
||||||
/// Panics if called outside `Runtime::run()`.
|
/// Panics if called outside `Runtime::run()`.
|
||||||
pub fn members(group: &str) -> Vec<Pid> {
|
pub fn members(group: &str) -> Vec<Pid> {
|
||||||
let (pids, reaped) = with_runtime(|inner| {
|
with_runtime(|inner| {
|
||||||
let mut pg = inner.process_groups.lock();
|
inner
|
||||||
let reaped = pg.reap_group(group);
|
.process_groups
|
||||||
let pids = pg.members_where(group, |pid| live(inner, pid));
|
.lock()
|
||||||
(pids, reaped)
|
.members_where(group, inner.node_id, |pid| live(inner, pid))
|
||||||
});
|
})
|
||||||
drop(reaped);
|
|
||||||
pids
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// One live member of `group`, or `None` if the group is empty (or every
|
/// One live member of `group`, or `None` if the group is empty (or every
|
||||||
@@ -455,14 +658,12 @@ pub fn members(group: &str) -> Vec<Pid> {
|
|||||||
///
|
///
|
||||||
/// Panics if called outside `Runtime::run()`.
|
/// Panics if called outside `Runtime::run()`.
|
||||||
pub fn pick(group: &str) -> Option<Pid> {
|
pub fn pick(group: &str) -> Option<Pid> {
|
||||||
let (picked, reaped) = with_runtime(|inner| {
|
with_runtime(|inner| {
|
||||||
let mut pg = inner.process_groups.lock();
|
inner
|
||||||
let reaped = pg.reap_group(group);
|
.process_groups
|
||||||
let picked = pg.first_member_where(group, |pid| live(inner, pid));
|
.lock()
|
||||||
(picked, reaped)
|
.first_member_where(group, inner.node_id, |pid| live(inner, pid))
|
||||||
});
|
})
|
||||||
drop(reaped);
|
|
||||||
picked
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Typed [`pick`]: one live member of `group` as a [`Pid<A>`](Pid).
|
/// Typed [`pick`]: one live member of `group` as a [`Pid<A>`](Pid).
|
||||||
@@ -505,8 +706,8 @@ pub fn dispatch<A: Addressable>(group: &str, msg: A::Msg) -> Result<Pid<A>, Send
|
|||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
mod tests {
|
mod tests {
|
||||||
use super::*;
|
use super::*;
|
||||||
use crate::channel::{channel, Sender};
|
use crate::scheduler::spawn;
|
||||||
use crate::monitor::{Down, DownReason, MonitorId};
|
use std::time::{Duration, Instant};
|
||||||
|
|
||||||
fn member(index: u32, generation: u32) -> Member {
|
fn member(index: u32, generation: u32) -> Member {
|
||||||
Member {
|
Member {
|
||||||
@@ -516,33 +717,21 @@ mod tests {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// A synthetic membership with a real (but slot-less) monitor channel. The
|
/// A synthetic membership: the store never looks at the id.
|
||||||
/// returned `Sender` stands in for the slot's `Down` sender: hold it to
|
fn synth(index: u32, generation: u32) -> Membership {
|
||||||
/// keep the member "alive" (`try_recv` → `Ok(None)`), `send` a `Down` to
|
Membership {
|
||||||
/// simulate death, or `drop` it to simulate a drained/closed channel.
|
|
||||||
fn synth(index: u32, generation: u32) -> (Membership, Sender<Down>) {
|
|
||||||
let pid = Pid::new(index, generation);
|
|
||||||
let (tx, rx) = channel::<Down>();
|
|
||||||
let ms = Membership {
|
|
||||||
member: member(index, generation),
|
member: member(index, generation),
|
||||||
monitor: Monitor {
|
monitor: Some(MonitorId(0)),
|
||||||
id: MonitorId(0),
|
}
|
||||||
target: pid,
|
|
||||||
rx,
|
|
||||||
},
|
|
||||||
};
|
|
||||||
(ms, tx)
|
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn join_is_idempotent_within_a_group() {
|
fn join_is_idempotent_within_a_group() {
|
||||||
let mut pg = ProcessGroups::new();
|
let mut pg = ProcessGroups::new();
|
||||||
let (a, _ta) = synth(1, 0);
|
assert!(pg.join("workers", synth(1, 0)), "first join inserts");
|
||||||
let (b, _tb) = synth(1, 0);
|
|
||||||
assert!(pg.join("workers", a).is_none(), "first join inserts");
|
|
||||||
assert!(
|
assert!(
|
||||||
pg.join("workers", b).is_some(),
|
!pg.join("workers", synth(1, 0)),
|
||||||
"second identical join is handed back"
|
"second identical join is refused"
|
||||||
);
|
);
|
||||||
assert_eq!(pg.members_of("workers"), vec![member(1, 0)]);
|
assert_eq!(pg.members_of("workers"), vec![member(1, 0)]);
|
||||||
}
|
}
|
||||||
@@ -550,12 +739,9 @@ mod tests {
|
|||||||
#[test]
|
#[test]
|
||||||
fn same_pid_in_many_groups_is_independent() {
|
fn same_pid_in_many_groups_is_independent() {
|
||||||
let mut pg = ProcessGroups::new();
|
let mut pg = ProcessGroups::new();
|
||||||
let (a, _ta) = synth(1, 0);
|
pg.join("a", synth(1, 0));
|
||||||
let (b, _tb) = synth(1, 0);
|
pg.join("b", synth(1, 0));
|
||||||
let (c, _tc) = synth(2, 0);
|
pg.join("b", synth(2, 0));
|
||||||
pg.join("a", a);
|
|
||||||
pg.join("b", b);
|
|
||||||
pg.join("b", c);
|
|
||||||
assert_eq!(pg.members_of("a"), vec![member(1, 0)]);
|
assert_eq!(pg.members_of("a"), vec![member(1, 0)]);
|
||||||
assert_eq!(pg.members_of("b"), vec![member(1, 0), member(2, 0)]);
|
assert_eq!(pg.members_of("b"), vec![member(1, 0), member(2, 0)]);
|
||||||
}
|
}
|
||||||
@@ -564,11 +750,9 @@ mod tests {
|
|||||||
fn distinct_generations_are_distinct_members() {
|
fn distinct_generations_are_distinct_members() {
|
||||||
// ABA guard: same slot index, different generation = different actor.
|
// ABA guard: same slot index, different generation = different actor.
|
||||||
let mut pg = ProcessGroups::new();
|
let mut pg = ProcessGroups::new();
|
||||||
let (a, _ta) = synth(1, 0);
|
assert!(pg.join("g", synth(1, 0)));
|
||||||
let (b, _tb) = synth(1, 1);
|
|
||||||
assert!(pg.join("g", a).is_none());
|
|
||||||
assert!(
|
assert!(
|
||||||
pg.join("g", b).is_none(),
|
pg.join("g", synth(1, 1)),
|
||||||
"different generation is a distinct member"
|
"different generation is a distinct member"
|
||||||
);
|
);
|
||||||
assert_eq!(pg.members_of("g"), vec![member(1, 0), member(1, 1)]);
|
assert_eq!(pg.members_of("g"), vec![member(1, 0), member(1, 1)]);
|
||||||
@@ -577,10 +761,8 @@ mod tests {
|
|||||||
#[test]
|
#[test]
|
||||||
fn leave_removes_one_membership_and_prunes_empty_groups() {
|
fn leave_removes_one_membership_and_prunes_empty_groups() {
|
||||||
let mut pg = ProcessGroups::new();
|
let mut pg = ProcessGroups::new();
|
||||||
let (a, _ta) = synth(1, 0);
|
pg.join("g", synth(1, 0));
|
||||||
let (b, _tb) = synth(2, 0);
|
pg.join("g", synth(2, 0));
|
||||||
pg.join("g", a);
|
|
||||||
pg.join("g", b);
|
|
||||||
assert!(pg.leave("g", member(1, 0)).is_some());
|
assert!(pg.leave("g", member(1, 0)).is_some());
|
||||||
assert_eq!(pg.members_of("g"), vec![member(2, 0)]);
|
assert_eq!(pg.members_of("g"), vec![member(2, 0)]);
|
||||||
assert!(
|
assert!(
|
||||||
@@ -598,7 +780,7 @@ mod tests {
|
|||||||
#[test]
|
#[test]
|
||||||
fn remove_where_sweeps_every_group() {
|
fn remove_where_sweeps_every_group() {
|
||||||
let mut pg = ProcessGroups::new();
|
let mut pg = ProcessGroups::new();
|
||||||
for (g, (m, _t)) in [
|
for (g, m) in [
|
||||||
("a", synth(1, 0)),
|
("a", synth(1, 0)),
|
||||||
("a", synth(2, 0)),
|
("a", synth(2, 0)),
|
||||||
("b", synth(1, 0)),
|
("b", synth(1, 0)),
|
||||||
@@ -616,104 +798,137 @@ mod tests {
|
|||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn remove_where_can_match_an_incarnation_sweep() {
|
fn remove_where_can_match_an_incarnation_sweep() {
|
||||||
// Shape check for the later evict_incarnation(node, inc) caller.
|
// Shape check for the node-down / incarnation sweep caller.
|
||||||
let mut pg = ProcessGroups::new();
|
let mut pg = ProcessGroups::new();
|
||||||
let pid = Pid::new(1, 0);
|
let stale = Membership {
|
||||||
let (tx, rx) = channel::<Down>();
|
|
||||||
let dead = Membership {
|
|
||||||
member: Member {
|
member: Member {
|
||||||
node: DEFAULT_NODE_ID,
|
node: DEFAULT_NODE_ID,
|
||||||
incarnation: Incarnation::new(7),
|
incarnation: Incarnation::new(7),
|
||||||
pid,
|
pid: Pid::new(1, 0),
|
||||||
},
|
|
||||||
monitor: Monitor {
|
|
||||||
id: MonitorId(0),
|
|
||||||
target: pid,
|
|
||||||
rx,
|
|
||||||
},
|
},
|
||||||
|
monitor: Some(MonitorId(0)),
|
||||||
};
|
};
|
||||||
let _keep = tx;
|
pg.join("g", stale);
|
||||||
let (live, _tl) = synth(2, 0);
|
pg.join("g", synth(2, 0));
|
||||||
pg.join("g", dead);
|
|
||||||
pg.join("g", live);
|
|
||||||
let evicted = pg.remove_where(|mem| mem.incarnation == Incarnation::new(7));
|
let evicted = pg.remove_where(|mem| mem.incarnation == Incarnation::new(7));
|
||||||
assert_eq!(evicted.len(), 1);
|
assert_eq!(evicted.len(), 1);
|
||||||
assert_eq!(pg.members_of("g"), vec![member(2, 0)]);
|
assert_eq!(pg.members_of("g"), vec![member(2, 0)]);
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn reap_keeps_live_members() {
|
fn read_backstop_hides_a_member_the_reaper_has_not_yet_swept() {
|
||||||
let mut pg = ProcessGroups::new();
|
let mut pg = ProcessGroups::new();
|
||||||
let (a, _ta) = synth(1, 0); // sender held: member stays alive
|
pg.join("g", synth(1, 0));
|
||||||
pg.join("a", a);
|
pg.join("g", synth(2, 0));
|
||||||
assert!(pg.reap_group("a").is_empty(), "no deaths");
|
|
||||||
assert_eq!(pg.members_of("a"), vec![member(1, 0)]);
|
|
||||||
}
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn reap_evicts_a_dead_member_and_sweeps_all_its_groups() {
|
|
||||||
let mut pg = ProcessGroups::new();
|
|
||||||
let (a1, ta1) = synth(1, 0); // pid 1 in group a
|
|
||||||
let (a2, _ta2) = synth(2, 0); // pid 2 in group a (stays alive)
|
|
||||||
let (b1, _tb1) = synth(1, 0); // pid 1 in group b
|
|
||||||
pg.join("a", a1);
|
|
||||||
pg.join("a", a2);
|
|
||||||
pg.join("b", b1);
|
|
||||||
// pid 1 dies: its group-a monitor receives a Down. Its group-b monitor
|
|
||||||
// has not — reap must still sweep pid 1 out of b by the pid predicate.
|
|
||||||
ta1.send(Down {
|
|
||||||
pid: Pid::new(1, 0),
|
|
||||||
reason: DownReason::Exit,
|
|
||||||
})
|
|
||||||
.unwrap();
|
|
||||||
let evicted = pg.reap_group("a");
|
|
||||||
assert_eq!(
|
|
||||||
evicted.len(),
|
|
||||||
2,
|
|
||||||
"pid 1's memberships in both a and b are evicted"
|
|
||||||
);
|
|
||||||
assert_eq!(pg.members_of("a"), vec![member(2, 0)]);
|
|
||||||
assert!(pg.members_of("b").is_empty(), "swept from b too; pruned");
|
|
||||||
}
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn reap_treats_a_closed_channel_as_dead() {
|
|
||||||
let mut pg = ProcessGroups::new();
|
|
||||||
let (a, ta) = synth(1, 0);
|
|
||||||
pg.join("a", a);
|
|
||||||
drop(ta); // sender gone, queue empty → try_recv = Err(RecvError) = dead
|
|
||||||
let evicted = pg.reap_group("a");
|
|
||||||
assert_eq!(evicted.len(), 1);
|
|
||||||
assert!(pg.members_of("a").is_empty());
|
|
||||||
}
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn read_backstop_hides_a_member_the_monitor_has_not_yet_reaped() {
|
|
||||||
let mut pg = ProcessGroups::new();
|
|
||||||
// Both senders held: reap_group would see Ok(None) and evict neither.
|
|
||||||
let (a, _ta) = synth(1, 0);
|
|
||||||
let (b, _tb) = synth(2, 0);
|
|
||||||
pg.join("g", a);
|
|
||||||
pg.join("g", b);
|
|
||||||
|
|
||||||
// The slot-word oracle already reports pid 1 dead (finalize window),
|
// The slot-word oracle already reports pid 1 dead (finalize window),
|
||||||
// ahead of any Down delivery.
|
// ahead of the reaper's turn.
|
||||||
let dead = Pid::new(1, 0);
|
let dead = Pid::new(1, 0);
|
||||||
let oracle = |pid: Pid| pid != dead;
|
let oracle = |pid: Pid| pid != dead;
|
||||||
|
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
pg.members_where("g", oracle),
|
pg.members_where("g", DEFAULT_NODE_ID, oracle),
|
||||||
vec![Pid::new(2, 0)],
|
vec![Pid::new(2, 0)],
|
||||||
"dead pid filtered from read"
|
"dead pid filtered from read"
|
||||||
);
|
);
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
pg.first_member_where("g", oracle),
|
pg.first_member_where("g", DEFAULT_NODE_ID, oracle),
|
||||||
Some(Pid::new(2, 0)),
|
Some(Pid::new(2, 0)),
|
||||||
"pick skips the dead first member"
|
"pick skips the dead first member"
|
||||||
);
|
);
|
||||||
|
|
||||||
// Backstop does not evict — that stays the monitor's job; raw storage
|
// Backstop does not evict — that stays the reaper's job; raw storage
|
||||||
// still holds both until reap runs.
|
// still holds both until it runs.
|
||||||
assert_eq!(pg.members_of("g"), vec![member(1, 0), member(2, 0)]);
|
assert_eq!(pg.members_of("g"), vec![member(1, 0), member(2, 0)]);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ---- reaper: eager eviction against a live runtime ----
|
||||||
|
|
||||||
|
/// Raw storage view for a group, bypassing the read-path backstop.
|
||||||
|
fn stored(group: &str) -> Vec<Member> {
|
||||||
|
with_runtime(|inner| inner.process_groups.lock().members_of(group))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Cooperative wait (`smarm::sleep`, never an OS block) until `pred`.
|
||||||
|
fn wait_until(what: &str, mut pred: impl FnMut() -> bool) {
|
||||||
|
let deadline = Instant::now() + Duration::from_secs(2);
|
||||||
|
while !pred() {
|
||||||
|
assert!(Instant::now() < deadline, "timed out waiting for: {what}");
|
||||||
|
crate::sleep(Duration::from_millis(1));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn a_death_is_swept_from_storage_without_any_group_operation() {
|
||||||
|
crate::run(|| {
|
||||||
|
let (tx, rx) = channel::<()>();
|
||||||
|
let w = spawn(move || {
|
||||||
|
rx.recv().unwrap();
|
||||||
|
});
|
||||||
|
let pid = w.pid();
|
||||||
|
join("a", pid);
|
||||||
|
join("b", pid);
|
||||||
|
assert_eq!(stored("a"), vec![member_for_test(pid)]);
|
||||||
|
|
||||||
|
tx.send(()).unwrap();
|
||||||
|
w.join().unwrap();
|
||||||
|
// No members()/pick()/join() on a or b from here on: the reaper
|
||||||
|
// alone must clear both.
|
||||||
|
wait_until("reaper sweeps a and b", || {
|
||||||
|
stored("a").is_empty() && stored("b").is_empty()
|
||||||
|
});
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn a_dead_at_join_pid_is_swept_from_storage() {
|
||||||
|
crate::run(|| {
|
||||||
|
let h = spawn(|| {});
|
||||||
|
let pid = h.pid();
|
||||||
|
h.join().unwrap();
|
||||||
|
assert!(join("late", pid), "join is accepted; eviction is uniform");
|
||||||
|
wait_until("reaper sweeps the NoProc member", || {
|
||||||
|
stored("late").is_empty()
|
||||||
|
});
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn leave_then_death_does_not_disturb_a_rejoined_group() {
|
||||||
|
// A monitor unregistered by `leave` must not fire later; the pid's
|
||||||
|
// fresh membership after re-join is swept exactly once, by its own
|
||||||
|
// monitor, on death.
|
||||||
|
crate::run(|| {
|
||||||
|
let (tx, rx) = channel::<()>();
|
||||||
|
let w = spawn(move || {
|
||||||
|
rx.recv().unwrap();
|
||||||
|
});
|
||||||
|
let pid = w.pid();
|
||||||
|
join("g", pid);
|
||||||
|
assert!(leave("g", pid));
|
||||||
|
assert!(join("g", pid));
|
||||||
|
assert_eq!(members("g"), vec![pid]);
|
||||||
|
tx.send(()).unwrap();
|
||||||
|
w.join().unwrap();
|
||||||
|
wait_until("reaper sweeps g", || stored("g").is_empty());
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn reaper_is_respawned_for_a_second_run_of_the_same_runtime() {
|
||||||
|
let rt = crate::runtime::init(crate::runtime::Config::exact(1));
|
||||||
|
let body = || {
|
||||||
|
let h = spawn(|| {});
|
||||||
|
let pid = h.pid();
|
||||||
|
h.join().unwrap();
|
||||||
|
join("g", pid);
|
||||||
|
wait_until("reaper sweeps g", || stored("g").is_empty());
|
||||||
|
};
|
||||||
|
rt.run(body);
|
||||||
|
rt.run(body);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn member_for_test(pid: Pid) -> Member {
|
||||||
|
with_runtime(|inner| member_for(inner, pid))
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
+42
-1
@@ -76,7 +76,7 @@ pub struct Pid<A = Erased> {
|
|||||||
|
|
||||||
impl Pid<Erased> {
|
impl Pid<Erased> {
|
||||||
/// Build an untyped pid from raw numbers. The runtime mints identities
|
/// Build an untyped pid from raw numbers. The runtime mints identities
|
||||||
/// here; typing happens at typed-actor boundaries via [`Pid::from_raw`].
|
/// here; typing happens at typed-actor boundaries via `Pid::from_raw`.
|
||||||
#[inline]
|
#[inline]
|
||||||
pub const fn new(index: u32, generation: u32) -> Self {
|
pub const fn new(index: u32, generation: u32) -> Self {
|
||||||
Self {
|
Self {
|
||||||
@@ -292,3 +292,44 @@ mod typed_pid_tests {
|
|||||||
assert_send_sync::<Name<CounterMsg>>();
|
assert_send_sync::<Name<CounterMsg>>();
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ---- RFC 010 c10: pids auto-serialize (cluster feature) ---------------------
|
||||||
|
|
||||||
|
/// A local `Pid<A>` serializes as a
|
||||||
|
/// [`RemotePid<A>`](crate::cluster::remote::RemotePid): the wire form stamps
|
||||||
|
/// this node's name and incarnation from the ambient runtime, so a pid can
|
||||||
|
/// sit inside any message field and reply-to needs no ceremony (RFC 010 §3,
|
||||||
|
/// "sugar not a bear trap"). Serializing a pid also marks it **watchable**
|
||||||
|
/// — the wire crossing is the cluster's `mark_watchable` set-site (D12), the
|
||||||
|
/// exact analog of the membrane crossing.
|
||||||
|
///
|
||||||
|
/// Must run inside `run()` (the ambient identity lives on the runtime); a
|
||||||
|
/// runtime without a cluster identity cannot serialize a pid at all — it is
|
||||||
|
/// a serialize error, surfacing as the send's `Encode` failure — rather than
|
||||||
|
/// a `("", 0)` stamp that every peer would silently drop.
|
||||||
|
#[cfg(feature = "cluster")]
|
||||||
|
impl<A: 'static> serde::Serialize for Pid<A> {
|
||||||
|
fn serialize<S: serde::Serializer>(&self, s: S) -> Result<S::Ok, S::Error> {
|
||||||
|
crate::cluster::remote::RemotePid::<A>::from_local(*self)
|
||||||
|
.ok_or_else(|| serde::ser::Error::custom("pid serialized with no local node identity"))?
|
||||||
|
.serialize(s)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Deserializing into a `Pid<A>` is the **collapse**: it succeeds only when
|
||||||
|
/// the wire pid names this very node (name and incarnation both), and is a
|
||||||
|
/// decode error otherwise — a foreign pid cannot become a local `Pid`.
|
||||||
|
/// Fields that may hold a pid from anywhere are `RemotePid<A>`.
|
||||||
|
#[cfg(feature = "cluster")]
|
||||||
|
impl<'de, A: 'static> serde::Deserialize<'de> for Pid<A> {
|
||||||
|
fn deserialize<D: serde::Deserializer<'de>>(d: D) -> Result<Self, D::Error> {
|
||||||
|
let rp = crate::cluster::remote::RemotePid::<A>::deserialize(d)?;
|
||||||
|
rp.local().ok_or_else(|| {
|
||||||
|
serde::de::Error::custom(format!(
|
||||||
|
"pid {}@{} is not local to this node",
|
||||||
|
rp.index(),
|
||||||
|
rp.node()
|
||||||
|
))
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
+53
-6
@@ -105,18 +105,36 @@ pub(crate) fn clear_current_slot() {
|
|||||||
/// (RFC 019 §7), which additionally relies on this being a plain load of a
|
/// (RFC 019 §7), which additionally relies on this being a plain load of a
|
||||||
/// const-initialized TLS Cell (no lazy init, no allocation, no dtor): safe
|
/// const-initialized TLS Cell (no lazy init, no allocation, no dtor): safe
|
||||||
/// from a signal handler.
|
/// from a signal handler.
|
||||||
#[inline]
|
#[inline(never)]
|
||||||
pub(crate) fn current_slot_ptr() -> *const crate::runtime::Slot {
|
pub(crate) fn current_slot_ptr() -> *const crate::runtime::Slot {
|
||||||
|
crate::context::tls_fence();
|
||||||
CURRENT_SLOT.with(|c| c.get())
|
CURRENT_SLOT.with(|c| c.get())
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Swap the preemption gate, returning the previous value. The one accessor
|
||||||
|
/// for `PREEMPTION_ENABLED` from actor context (`NoPreempt`, `RawMutex`,
|
||||||
|
/// `with_runtime`, trace): `#[inline(never)]` + fence, see `context` docs.
|
||||||
|
#[inline(never)]
|
||||||
|
pub(crate) fn preemption_swap(enabled: bool) -> bool {
|
||||||
|
crate::context::tls_fence();
|
||||||
|
PREEMPTION_ENABLED.with(|c| c.replace(enabled))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Read the preemption gate (debug assertions on the queue paths).
|
||||||
|
#[inline(never)]
|
||||||
|
pub(crate) fn preemption_enabled() -> bool {
|
||||||
|
crate::context::tls_fence();
|
||||||
|
PREEMPTION_ENABLED.with(|c| c.get())
|
||||||
|
}
|
||||||
|
|
||||||
/// RFC 007 (`smarm-causal`) — push the slice start forward by `cycles`, so
|
/// RFC 007 (`smarm-causal`) — push the slice start forward by `cycles`, so
|
||||||
/// virtually-injected delay spun inside `maybe_preempt` does not count against
|
/// virtually-injected delay spun inside `maybe_preempt` does not count against
|
||||||
/// the actor's timeslice (the clock-correction half of the RFC: the runtime
|
/// the actor's timeslice (the clock-correction half of the RFC: the runtime
|
||||||
/// owns this clock, so it can subtract its own perturbation).
|
/// owns this clock, so it can subtract its own perturbation).
|
||||||
#[cfg(feature = "smarm-causal")]
|
#[cfg(feature = "smarm-causal")]
|
||||||
#[inline]
|
#[inline(never)]
|
||||||
pub(crate) fn extend_timeslice(cycles: u64) {
|
pub(crate) fn extend_timeslice(cycles: u64) {
|
||||||
|
crate::context::tls_fence();
|
||||||
TIMESLICE_START.with(|c| c.set(c.get().wrapping_add(cycles)));
|
TIMESLICE_START.with(|c| c.set(c.get().wrapping_add(cycles)));
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -124,8 +142,9 @@ pub(crate) fn extend_timeslice(cycles: u64) {
|
|||||||
/// no-op if no actor is bound (the scheduler's own stack). Reached only from
|
/// no-op if no actor is bound (the scheduler's own stack). Reached only from
|
||||||
/// the slice-expiry branch, which is already the yield path, so its cost is
|
/// the slice-expiry branch, which is already the yield path, so its cost is
|
||||||
/// irrelevant.
|
/// irrelevant.
|
||||||
#[inline]
|
#[inline(never)]
|
||||||
fn note_overrun() {
|
fn note_overrun() {
|
||||||
|
crate::context::tls_fence();
|
||||||
let p = CURRENT_SLOT.with(|c| c.get());
|
let p = CURRENT_SLOT.with(|c| c.get());
|
||||||
// SAFETY: `p` is null (no actor on-CPU) or a pointer to the on-CPU actor's
|
// SAFETY: `p` is null (no actor on-CPU) or a pointer to the on-CPU actor's
|
||||||
// slot in the fixed slab. The slot is not reclaimed while the actor runs
|
// slot in the fixed slab. The slot is not reclaimed while the actor runs
|
||||||
@@ -141,8 +160,9 @@ fn note_overrun() {
|
|||||||
/// outside an actor (null slot). One TLS load + one Relaxed load/store on a
|
/// outside an actor (null slot). One TLS load + one Relaxed load/store on a
|
||||||
/// cache line the receiving thread already owns — no atomic RMW, no lock. Same
|
/// cache line the receiving thread already owns — no atomic RMW, no lock. Same
|
||||||
/// slot-lifetime safety argument as `note_overrun`.
|
/// slot-lifetime safety argument as `note_overrun`.
|
||||||
#[inline]
|
#[inline(never)]
|
||||||
pub(crate) fn note_message_received() {
|
pub(crate) fn note_message_received() {
|
||||||
|
crate::context::tls_fence();
|
||||||
let p = CURRENT_SLOT.with(|c| c.get());
|
let p = CURRENT_SLOT.with(|c| c.get());
|
||||||
if !p.is_null() {
|
if !p.is_null() {
|
||||||
unsafe { (*p).record_message() };
|
unsafe { (*p).record_message() };
|
||||||
@@ -158,8 +178,9 @@ pub(crate) fn note_message_received() {
|
|||||||
/// Called from `maybe_preempt` (amortised, the `check!()`/alloc path) and from
|
/// Called from `maybe_preempt` (amortised, the `check!()`/alloc path) and from
|
||||||
/// the wakeup side of every blocking park (`park_current`/`yield_now`), which
|
/// the wakeup side of every blocking park (`park_current`/`yield_now`), which
|
||||||
/// is past the prep-to-park window — so it can never lose a wakeup.
|
/// is past the prep-to-park window — so it can never lose a wakeup.
|
||||||
#[inline]
|
#[inline(never)]
|
||||||
pub fn check_cancelled() {
|
pub fn check_cancelled() {
|
||||||
|
crate::context::tls_fence();
|
||||||
let p = CURRENT_STOP.with(|c| c.get());
|
let p = CURRENT_STOP.with(|c| c.get());
|
||||||
// SAFETY: `p` is either null (no actor on-CPU — the scheduler clears it on
|
// SAFETY: `p` is either null (no actor on-CPU — the scheduler clears it on
|
||||||
// every return) or a pointer into the on-CPU actor's `Arc<AtomicBool>`
|
// every return) or a pointer into the on-CPU actor's `Arc<AtomicBool>`
|
||||||
@@ -198,8 +219,27 @@ pub(crate) fn elapsed_slice_cycles() -> u64 {
|
|||||||
rdtsc().saturating_sub(TIMESLICE_START.with(|c| c.get()))
|
rdtsc().saturating_sub(TIMESLICE_START.with(|c| c.get()))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Read the TSC, unserialised. The core may execute this before earlier
|
||||||
|
/// instructions retire (or after later ones start), so a single stamp can
|
||||||
|
/// land tens of cycles — worst case a stalled load's worth — early or late.
|
||||||
|
/// That is negligible against every consumer in this module: the timeslice
|
||||||
|
/// arm/expiry compare against a ~10^5-cycle slice, and an early stamp only
|
||||||
|
/// makes a slice look *more* used (expires marginally sooner, never later).
|
||||||
|
/// The `lfence` this used to carry was a pipeline drain paid on every
|
||||||
|
/// resume; the one place that needs it is causal-site attribution, which
|
||||||
|
/// opts in via [`rdtsc_serialising`].
|
||||||
#[inline(always)]
|
#[inline(always)]
|
||||||
pub fn rdtsc() -> u64 {
|
pub fn rdtsc() -> u64 {
|
||||||
|
// SAFETY: x86-64 only (this crate is x86-64 Linux only).
|
||||||
|
unsafe { core::arch::x86_64::_rdtsc() }
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Read the TSC after all prior instructions have completed locally.
|
||||||
|
/// Use where a stamp bounds an interval attributed to *code* — a speculative
|
||||||
|
/// early read would credit the tail of that code to whatever comes next.
|
||||||
|
/// Costs a pipeline drain; keep it off the per-resume path.
|
||||||
|
#[inline(always)]
|
||||||
|
pub fn rdtsc_serialising() -> u64 {
|
||||||
unsafe {
|
unsafe {
|
||||||
// SAFETY: x86-64 only. `lfence` serialises the instruction stream so
|
// SAFETY: x86-64 only. `lfence` serialises the instruction stream so
|
||||||
// we don't measure time before prior instructions retire.
|
// we don't measure time before prior instructions retire.
|
||||||
@@ -247,8 +287,13 @@ unsafe impl GlobalAlloc for PreemptingAllocator {
|
|||||||
/// the actor would then park, and the wakeup would be lost. Library
|
/// the actor would then park, and the wakeup would be lost. Library
|
||||||
/// code that touches the parking primitives must keep its prep-to-park
|
/// code that touches the parking primitives must keep its prep-to-park
|
||||||
/// regions allocation-free and check!()-free.
|
/// regions allocation-free and check!()-free.
|
||||||
#[inline(always)]
|
///
|
||||||
|
/// `#[inline(never)]`: this touches thread-locals and can switch threads in
|
||||||
|
/// the middle; inlined into a caller's loop the TLS base would be hoisted
|
||||||
|
/// across the switch (see `context` module docs). The call is the price.
|
||||||
|
#[inline(never)]
|
||||||
pub fn maybe_preempt() {
|
pub fn maybe_preempt() {
|
||||||
|
crate::context::tls_fence();
|
||||||
ALLOC_COUNT.with(|c| {
|
ALLOC_COUNT.with(|c| {
|
||||||
let n = c.get();
|
let n = c.get();
|
||||||
if n == 0 {
|
if n == 0 {
|
||||||
@@ -297,7 +342,9 @@ pub fn maybe_preempt() {
|
|||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
/// Force-expire the timeslice so the next RDTSC check preempts.
|
/// Force-expire the timeslice so the next RDTSC check preempts.
|
||||||
|
#[inline(never)]
|
||||||
pub fn expire_timeslice_for_test() {
|
pub fn expire_timeslice_for_test() {
|
||||||
|
crate::context::tls_fence();
|
||||||
TIMESLICE_START.with(|c| c.set(0));
|
TIMESLICE_START.with(|c| c.set(0));
|
||||||
ALLOC_COUNT.with(|c| c.set(0));
|
ALLOC_COUNT.with(|c| c.set(0));
|
||||||
}
|
}
|
||||||
|
|||||||
+12
-4
@@ -75,8 +75,13 @@ thread_local! {
|
|||||||
static CHANNELS_HELD: std::cell::Cell<u32> = const { std::cell::Cell::new(0) };
|
static CHANNELS_HELD: std::cell::Cell<u32> = const { std::cell::Cell::new(0) };
|
||||||
}
|
}
|
||||||
|
|
||||||
#[inline]
|
// Debug-only TLS bookkeeping: out of line only when it has a body (release
|
||||||
|
// builds must not pay a call for an empty function).
|
||||||
|
#[cfg_attr(debug_assertions, inline(never))]
|
||||||
|
#[cfg_attr(not(debug_assertions), inline(always))]
|
||||||
fn order_check_acquire(class: LockClass) {
|
fn order_check_acquire(class: LockClass) {
|
||||||
|
#[cfg(debug_assertions)]
|
||||||
|
crate::context::tls_fence();
|
||||||
#[cfg(debug_assertions)]
|
#[cfg(debug_assertions)]
|
||||||
match class {
|
match class {
|
||||||
LockClass::Leaf => LEAVES_HELD.with(|l| {
|
LockClass::Leaf => LEAVES_HELD.with(|l| {
|
||||||
@@ -111,8 +116,11 @@ fn order_check_acquire(class: LockClass) {
|
|||||||
let _ = class;
|
let _ = class;
|
||||||
}
|
}
|
||||||
|
|
||||||
#[inline]
|
#[cfg_attr(debug_assertions, inline(never))]
|
||||||
|
#[cfg_attr(not(debug_assertions), inline(always))]
|
||||||
fn order_check_release(class: LockClass) {
|
fn order_check_release(class: LockClass) {
|
||||||
|
#[cfg(debug_assertions)]
|
||||||
|
crate::context::tls_fence();
|
||||||
#[cfg(debug_assertions)]
|
#[cfg(debug_assertions)]
|
||||||
match class {
|
match class {
|
||||||
LockClass::Leaf => LEAVES_HELD.with(|c| c.set(c.get() - 1)),
|
LockClass::Leaf => LEAVES_HELD.with(|c| c.set(c.get() - 1)),
|
||||||
@@ -157,7 +165,7 @@ impl<T> RawMutex<T> {
|
|||||||
pub(crate) fn lock(&self) -> RawMutexGuard<'_, T> {
|
pub(crate) fn lock(&self) -> RawMutexGuard<'_, T> {
|
||||||
// Enter NoPreempt *before* acquiring, so a preemption can't fire
|
// Enter NoPreempt *before* acquiring, so a preemption can't fire
|
||||||
// between acquisition and guard construction.
|
// between acquisition and guard construction.
|
||||||
let prev_preempt = crate::preempt::PREEMPTION_ENABLED.with(|c| c.replace(false));
|
let prev_preempt = crate::preempt::preemption_swap(false);
|
||||||
order_check_acquire(self.class);
|
order_check_acquire(self.class);
|
||||||
if self
|
if self
|
||||||
.state
|
.state
|
||||||
@@ -237,7 +245,7 @@ impl<T> Drop for RawMutexGuard<'_, T> {
|
|||||||
self.m.unlock();
|
self.m.unlock();
|
||||||
order_check_release(self.m.class);
|
order_check_release(self.m.class);
|
||||||
// Restore preemption only after the lock is released.
|
// Restore preemption only after the lock is released.
|
||||||
crate::preempt::PREEMPTION_ENABLED.with(|c| c.set(self.prev_preempt));
|
crate::preempt::preemption_swap(self.prev_preempt);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+38
-1
@@ -10,7 +10,7 @@
|
|||||||
//! directly, without ever having been handed a `Pid`.
|
//! directly, without ever having been handed a `Pid`.
|
||||||
//!
|
//!
|
||||||
//! ```
|
//! ```
|
||||||
//! use smarm::{channel, register, run, send, spawn, unregister, whereis, Name};
|
//! use smarm::{channel, register, run, send, spawn, whereis, Name};
|
||||||
//!
|
//!
|
||||||
//! const COUNTER: Name<u64> = Name::new("counter");
|
//! const COUNTER: Name<u64> = Name::new("counter");
|
||||||
//!
|
//!
|
||||||
@@ -215,6 +215,7 @@ impl<M> std::error::Error for SendError<M> {}
|
|||||||
trait ErasedSender: Send {
|
trait ErasedSender: Send {
|
||||||
fn as_any(&self) -> &dyn Any;
|
fn as_any(&self) -> &dyn Any;
|
||||||
fn queued_len(&self) -> usize;
|
fn queued_len(&self) -> usize;
|
||||||
|
fn receiver_alive(&self) -> bool;
|
||||||
}
|
}
|
||||||
|
|
||||||
impl<M: Send + 'static> ErasedSender for Sender<M> {
|
impl<M: Send + 'static> ErasedSender for Sender<M> {
|
||||||
@@ -224,6 +225,9 @@ impl<M: Send + 'static> ErasedSender for Sender<M> {
|
|||||||
fn queued_len(&self) -> usize {
|
fn queued_len(&self) -> usize {
|
||||||
Sender::queued_len(self)
|
Sender::queued_len(self)
|
||||||
}
|
}
|
||||||
|
fn receiver_alive(&self) -> bool {
|
||||||
|
Sender::receiver_alive(self)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// One typed channel of an actor, type-erased. Concretely a `Sender<M>` filed
|
/// One typed channel of an actor, type-erased. Concretely a `Sender<M>` filed
|
||||||
@@ -442,6 +446,18 @@ pub(crate) fn register_with<M: Send + 'static>(
|
|||||||
/// name) and [`install`] (which does not). A leftover mailbox at this slot
|
/// name) and [`install`] (which does not). A leftover mailbox at this slot
|
||||||
/// index from a dead prior incarnation (pid mismatch) is replaced wholesale.
|
/// index from a dead prior incarnation (pid mismatch) is replaced wholesale.
|
||||||
/// Caller holds the registry lock and has established that `me` is live.
|
/// Caller holds the registry lock and has established that `me` is live.
|
||||||
|
///
|
||||||
|
/// **One channel per message type per actor.** Publishing a second `M`
|
||||||
|
/// channel on the same live actor replaces the first — and if the first's
|
||||||
|
/// receiver is still alive, that replacement drops its last sender, closing
|
||||||
|
/// it, and any `recv`/`select` on it then returns "closed" immediately and
|
||||||
|
/// forever: a silent hot loop that starves the scheduler. That is never
|
||||||
|
/// intended, so it panics here (found the hard way in RFC 010 c9, where two
|
||||||
|
/// `Name<String>`s registered on one actor did exactly this). Replacing a
|
||||||
|
/// channel whose receiver is already gone is fine (an actor re-registering
|
||||||
|
/// after dropping its old inbox) and stays silent. To hold two names of the
|
||||||
|
/// same type, register them from two actors, or bind both names to one
|
||||||
|
/// cloned sender.
|
||||||
fn publish_channel<M: Send + 'static>(reg: &mut Registry, me: Pid, tx: Sender<M>) {
|
fn publish_channel<M: Send + 'static>(reg: &mut Registry, me: Pid, tx: Sender<M>) {
|
||||||
let mb = reg
|
let mb = reg
|
||||||
.by_index
|
.by_index
|
||||||
@@ -450,6 +466,16 @@ fn publish_channel<M: Send + 'static>(reg: &mut Registry, me: Pid, tx: Sender<M>
|
|||||||
if mb.pid != me {
|
if mb.pid != me {
|
||||||
*mb = Mailbox::new(me);
|
*mb = Mailbox::new(me);
|
||||||
}
|
}
|
||||||
|
if let Some(existing) = mb.channels.get(&TypeId::of::<M>()) {
|
||||||
|
assert!(
|
||||||
|
!existing.sender.receiver_alive() || same_channel::<M>(existing, &tx),
|
||||||
|
"smarm: actor {me:?} already publishes a live channel for message type `{}`; \
|
||||||
|
a second one would replace and CLOSE the first (its receiver would then \
|
||||||
|
read as closed forever). Register the second name from another actor, or \
|
||||||
|
bind both names to a clone of the same sender.",
|
||||||
|
type_name::<M>()
|
||||||
|
);
|
||||||
|
}
|
||||||
mb.channels.insert(
|
mb.channels.insert(
|
||||||
TypeId::of::<M>(),
|
TypeId::of::<M>(),
|
||||||
Channel {
|
Channel {
|
||||||
@@ -459,6 +485,17 @@ fn publish_channel<M: Send + 'static>(reg: &mut Registry, me: Pid, tx: Sender<M>
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// True if `existing` and `tx` are senders of the very same channel (a
|
||||||
|
/// cloned sender bound under a second name is the sanctioned way to hold two
|
||||||
|
/// names of one type on one actor).
|
||||||
|
fn same_channel<M: Send + 'static>(existing: &Channel, tx: &Sender<M>) -> bool {
|
||||||
|
existing
|
||||||
|
.sender
|
||||||
|
.as_any()
|
||||||
|
.downcast_ref::<Sender<M>>()
|
||||||
|
.is_some_and(|old| old.same_channel(tx))
|
||||||
|
}
|
||||||
|
|
||||||
/// Publish the current actor's `Sender<A::Msg>` into its mailbox **without**
|
/// Publish the current actor's `Sender<A::Msg>` into its mailbox **without**
|
||||||
/// binding a name, and hand back the typed [`Pid<A>`] that addresses this
|
/// binding a name, and hand back the typed [`Pid<A>`] that addresses this
|
||||||
/// actor directly.
|
/// actor directly.
|
||||||
|
|||||||
+162
-23
@@ -2,9 +2,9 @@
|
|||||||
//! features (no runtime dispatch — the scheduler's pop loop is the hottest
|
//! features (no runtime dispatch — the scheduler's pop loop is the hottest
|
||||||
//! code in the runtime):
|
//! code in the runtime):
|
||||||
//!
|
//!
|
||||||
//! - `rq-mutex` (default) — `Mutex<VecDeque>`. The control/baseline:
|
//! - `rq-mutex` — `Mutex<VecDeque>`. The control/baseline:
|
||||||
//! strictly FIFO, trivially correct, one global lock.
|
//! strictly FIFO, trivially correct, one global lock.
|
||||||
//! - `rq-mpmc` — a single hand-rolled Vyukov bounded MPMC ring (per-cell
|
//! - `rq-mpmc` (default) — a single hand-rolled Vyukov bounded MPMC ring (per-cell
|
||||||
//! sequence numbers). Strict FIFO, lock-free, one hot
|
//! sequence numbers). Strict FIFO, lock-free, one hot
|
||||||
//! enqueue/dequeue cache-line pair.
|
//! enqueue/dequeue cache-line pair.
|
||||||
//! - `rq-striped` — M Vyukov rings with fetch-add ticket distribution.
|
//! - `rq-striped` — M Vyukov rings with fetch-add ticket distribution.
|
||||||
@@ -18,7 +18,7 @@
|
|||||||
//!
|
//!
|
||||||
//! All variants are compiled unconditionally (so every build runs every
|
//! All variants are compiled unconditionally (so every build runs every
|
||||||
//! variant's unit tests); the feature only picks which one the runtime uses
|
//! variant's unit tests); the feature only picks which one the runtime uses
|
||||||
//! via the [`RunQueue`] alias.
|
//! via the `RunQueue` alias.
|
||||||
//!
|
//!
|
||||||
//! # Contract (shared by all variants)
|
//! # Contract (shared by all variants)
|
||||||
//!
|
//!
|
||||||
@@ -51,7 +51,7 @@
|
|||||||
//! - `len()` is approximate (stats only).
|
//! - `len()` is approximate (stats only).
|
||||||
|
|
||||||
use crate::pid::Pid;
|
use crate::pid::Pid;
|
||||||
use crate::sync_shim::{AtomicUsize, Ordering, UnsafeCell};
|
use crate::sync_shim::{fence, AtomicUsize, Ordering, UnsafeCell};
|
||||||
use std::mem::MaybeUninit;
|
use std::mem::MaybeUninit;
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
@@ -61,20 +61,20 @@ use std::mem::MaybeUninit;
|
|||||||
#[cfg(not(any(feature = "rq-mutex", feature = "rq-mpmc", feature = "rq-striped")))]
|
#[cfg(not(any(feature = "rq-mutex", feature = "rq-mpmc", feature = "rq-striped")))]
|
||||||
compile_error!(
|
compile_error!(
|
||||||
"smarm: no run queue selected. Enable exactly one of the features \
|
"smarm: no run queue selected. Enable exactly one of the features \
|
||||||
`rq-mutex` (default), `rq-mpmc`, `rq-striped`."
|
`rq-mpmc` (default), `rq-mutex`, `rq-striped`."
|
||||||
);
|
);
|
||||||
#[cfg(all(feature = "rq-mutex", feature = "rq-mpmc"))]
|
#[cfg(all(feature = "rq-mutex", feature = "rq-mpmc"))]
|
||||||
compile_error!(
|
compile_error!(
|
||||||
"smarm: features `rq-mutex` and `rq-mpmc` are mutually exclusive \
|
"smarm: features `rq-mutex` and `rq-mpmc` are mutually exclusive \
|
||||||
(use --no-default-features to drop the default `rq-mutex`)."
|
(use --no-default-features to drop the default `rq-mpmc`)."
|
||||||
);
|
);
|
||||||
#[cfg(all(feature = "rq-mutex", feature = "rq-striped"))]
|
#[cfg(all(feature = "rq-mutex", feature = "rq-striped"))]
|
||||||
compile_error!(
|
compile_error!("smarm: features `rq-mutex` and `rq-striped` are mutually exclusive.");
|
||||||
"smarm: features `rq-mutex` and `rq-striped` are mutually exclusive \
|
|
||||||
(use --no-default-features to drop the default `rq-mutex`)."
|
|
||||||
);
|
|
||||||
#[cfg(all(feature = "rq-mpmc", feature = "rq-striped"))]
|
#[cfg(all(feature = "rq-mpmc", feature = "rq-striped"))]
|
||||||
compile_error!("smarm: features `rq-mpmc` and `rq-striped` are mutually exclusive.");
|
compile_error!(
|
||||||
|
"smarm: features `rq-mpmc` and `rq-striped` are mutually exclusive \
|
||||||
|
(use --no-default-features to drop the default `rq-mpmc`)."
|
||||||
|
);
|
||||||
|
|
||||||
#[cfg(feature = "rq-mutex")]
|
#[cfg(feature = "rq-mutex")]
|
||||||
pub(crate) type RunQueue = MutexQueue;
|
pub(crate) type RunQueue = MutexQueue;
|
||||||
@@ -86,12 +86,52 @@ pub(crate) type RunQueue = StripedRing;
|
|||||||
#[inline]
|
#[inline]
|
||||||
fn assert_no_preempt() {
|
fn assert_no_preempt() {
|
||||||
debug_assert!(
|
debug_assert!(
|
||||||
!crate::preempt::PREEMPTION_ENABLED.with(|c| c.get()),
|
!crate::preempt::preemption_enabled(),
|
||||||
"run-queue op with preemption enabled — a switch mid-op stalls or \
|
"run-queue op with preemption enabled — a switch mid-op stalls or \
|
||||||
corrupts the queue; route through with_runtime or scheduler context"
|
corrupts the queue; route through with_runtime or scheduler context"
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Escalating wait for transient ring stalls: a peer preempted by the OS
|
||||||
|
/// inside its ~100ns claim→publish window (finding 13). Spin first (the
|
||||||
|
/// common stall is a peer that is merely slow, gone within a few hundred
|
||||||
|
/// cycles), then donate the timeslice — under oversubscription pure
|
||||||
|
/// spinning STARVES the descheduled peer of the CPU it needs to publish
|
||||||
|
/// (measured in the finding-13 soak: 10⁶ pure spins can outlast the very
|
||||||
|
/// stall they prolong). OS-level yielding is orthogonal to
|
||||||
|
/// `assert_no_preempt`, which guards smarm signal preemption only.
|
||||||
|
struct Backoff(u32);
|
||||||
|
|
||||||
|
impl Backoff {
|
||||||
|
const SPIN_LIMIT: u32 = 6;
|
||||||
|
|
||||||
|
fn new() -> Self {
|
||||||
|
Self(0)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Total waits so far — lets bounded callers cap the yield phase.
|
||||||
|
fn steps(&self) -> u32 {
|
||||||
|
self.0
|
||||||
|
}
|
||||||
|
|
||||||
|
#[inline]
|
||||||
|
fn wait(&mut self) {
|
||||||
|
#[cfg(loom)]
|
||||||
|
// Loom has no notion of spinning time; every wait is a scheduling
|
||||||
|
// point so the model explores the stalled peer's progress.
|
||||||
|
loom::thread::yield_now();
|
||||||
|
#[cfg(not(loom))]
|
||||||
|
if self.0 <= Self::SPIN_LIMIT {
|
||||||
|
for _ in 0..1u32 << self.0 {
|
||||||
|
std::hint::spin_loop();
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
std::thread::yield_now();
|
||||||
|
}
|
||||||
|
self.0 += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
// rq-mutex — the baseline
|
// rq-mutex — the baseline
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
@@ -204,15 +244,57 @@ impl MpmcRing {
|
|||||||
|
|
||||||
pub fn push(&self, pid: Pid) {
|
pub fn push(&self, pid: Pid) {
|
||||||
assert_no_preempt();
|
assert_no_preempt();
|
||||||
assert!(
|
if self.try_push(pid) {
|
||||||
self.try_push(pid),
|
return;
|
||||||
"smarm: run queue overflow — occupancy exceeded the slab bound, \
|
}
|
||||||
which the at-most-once-enqueued invariant forbids. This is a \
|
self.push_slow(pid);
|
||||||
runtime bug (double enqueue), not a capacity tuning problem."
|
|
||||||
);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// One full claim attempt; `false` only if the ring is full.
|
/// Cold path: the cell at `enqueue_pos` is still a lap behind. Two
|
||||||
|
/// worlds are indistinguishable at the cell (finding 13): a consumer
|
||||||
|
/// preempted between its `dequeue_pos` claim and its seq release while
|
||||||
|
/// the ring lapped onto that cell (transient), or a genuine occupancy
|
||||||
|
/// overflow (a runtime bug). The COUNTERS discriminate — the same shape
|
||||||
|
/// as crossbeam `ArrayQueue::push`'s fence + opposite-counter check:
|
||||||
|
/// occupancy < capacity ⇒ transient ⇒ wait for the stalled peer;
|
||||||
|
/// occupancy ≥ capacity ⇒ the at-most-once-enqueued invariant really is
|
||||||
|
/// broken ⇒ panic (a legal push starts from occupancy ≤ max_actors − 1).
|
||||||
|
#[cold]
|
||||||
|
fn push_slow(&self, pid: Pid) {
|
||||||
|
let mut backoff = Backoff::new();
|
||||||
|
loop {
|
||||||
|
// Order the counter reads after the failed cell read.
|
||||||
|
// `enqueue_pos` is loaded BEFORE `dequeue_pos`, so a pop racing
|
||||||
|
// us can only make the computed occupancy an UNDERestimate —
|
||||||
|
// conservative in the safe direction (never a spurious panic;
|
||||||
|
// a genuine violation is persistent and caught next lap).
|
||||||
|
fence(Ordering::SeqCst);
|
||||||
|
let enq = self.enqueue_pos.0.load(Ordering::SeqCst);
|
||||||
|
let deq = self.dequeue_pos.0.load(Ordering::SeqCst);
|
||||||
|
let occ = enq.wrapping_sub(deq);
|
||||||
|
assert!(
|
||||||
|
occ <= self.mask,
|
||||||
|
"smarm: run queue occupancy {} reached capacity {} — a pid \
|
||||||
|
was enqueued more than once (or more pids exist than \
|
||||||
|
max_actors); the at-most-once-enqueued invariant is broken \
|
||||||
|
(enq={} deq={})",
|
||||||
|
occ,
|
||||||
|
self.mask + 1,
|
||||||
|
enq,
|
||||||
|
deq
|
||||||
|
);
|
||||||
|
backoff.wait();
|
||||||
|
if self.try_push(pid) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One full claim attempt; `false` when the cell at `enqueue_pos` is
|
||||||
|
/// still a lap behind — which means EITHER genuinely full OR a
|
||||||
|
/// lap-stalled consumer (finding 13). The cell cannot tell the two
|
||||||
|
/// apart; callers must disambiguate via the counters (`push_slow`) or
|
||||||
|
/// tolerate refusal (`StripedRing`'s probe).
|
||||||
fn try_push(&self, pid: Pid) -> bool {
|
fn try_push(&self, pid: Pid) -> bool {
|
||||||
let mut pos = self.enqueue_pos.0.load(Ordering::Relaxed);
|
let mut pos = self.enqueue_pos.0.load(Ordering::Relaxed);
|
||||||
loop {
|
loop {
|
||||||
@@ -246,6 +328,7 @@ impl MpmcRing {
|
|||||||
|
|
||||||
pub fn pop(&self) -> Option<Pid> {
|
pub fn pop(&self) -> Option<Pid> {
|
||||||
assert_no_preempt();
|
assert_no_preempt();
|
||||||
|
let mut backoff = Backoff::new();
|
||||||
let mut pos = self.dequeue_pos.0.load(Ordering::Relaxed);
|
let mut pos = self.dequeue_pos.0.load(Ordering::Relaxed);
|
||||||
loop {
|
loop {
|
||||||
let cell = &self.buf[pos & self.mask];
|
let cell = &self.buf[pos & self.mask];
|
||||||
@@ -270,9 +353,31 @@ impl MpmcRing {
|
|||||||
Err(actual) => pos = actual,
|
Err(actual) => pos = actual,
|
||||||
}
|
}
|
||||||
} else if diff < 0 {
|
} else if diff < 0 {
|
||||||
// Empty (or the producer at this cell hasn't published yet —
|
// The cell at `pos` is unpublished: either the queue is
|
||||||
// a snapshot miss the caller's idle-retry loop absorbs).
|
// empty, or the producer that claimed it was preempted
|
||||||
return None;
|
// inside its claim→publish window (finding 13). The cell
|
||||||
|
// cannot tell the two apart; the counters can.
|
||||||
|
fence(Ordering::SeqCst);
|
||||||
|
let deq = self.dequeue_pos.0.load(Ordering::SeqCst);
|
||||||
|
if deq != pos {
|
||||||
|
// Stale head — no verdict; re-probe at the real head.
|
||||||
|
pos = deq;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if self.enqueue_pos.0.load(Ordering::SeqCst) == pos {
|
||||||
|
return None; // counters agree: genuinely empty
|
||||||
|
}
|
||||||
|
// Producer mid-publish. Wait briefly, then report None
|
||||||
|
// anyway — a DELIBERATE bounded deviation from crossbeam's
|
||||||
|
// unbounded retry: a spurious None is correctness-benign
|
||||||
|
// here (the stalled push completes and its RFC 018
|
||||||
|
// enqueue-wake re-wakes a parked scheduler), a parked
|
||||||
|
// scheduler beats a yielding one, and StripedRing's pop
|
||||||
|
// probe must not hang on one stripe.
|
||||||
|
if backoff.steps() > Backoff::SPIN_LIMIT + 8 {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
backoff.wait();
|
||||||
} else {
|
} else {
|
||||||
pos = self.dequeue_pos.0.load(Ordering::Relaxed);
|
pos = self.dequeue_pos.0.load(Ordering::Relaxed);
|
||||||
}
|
}
|
||||||
@@ -342,6 +447,7 @@ impl StripedRing {
|
|||||||
// guarantees a free stripe exists, so the outer loop terminates.
|
// guarantees a free stripe exists, so the outer loop terminates.
|
||||||
// The retry-from-home lap handles the racy case where every stripe
|
// The retry-from-home lap handles the racy case where every stripe
|
||||||
// momentarily refused us.
|
// momentarily refused us.
|
||||||
|
let mut backoff = Backoff::new();
|
||||||
loop {
|
loop {
|
||||||
for i in 0..=self.stripe_mask {
|
for i in 0..=self.stripe_mask {
|
||||||
let s = &self.stripes[(home + i) & self.stripe_mask];
|
let s = &self.stripes[(home + i) & self.stripe_mask];
|
||||||
@@ -349,7 +455,13 @@ impl StripedRing {
|
|||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
std::hint::spin_loop();
|
// Every stripe refused this lap: either transiently full (the
|
||||||
|
// headroom argument above guarantees a genuinely free stripe
|
||||||
|
// exists) or the probes landed on lap-stalled cells (finding
|
||||||
|
// 13 — try_push cannot tell the two apart). Waiting is correct
|
||||||
|
// either way; the backoff escalates to an OS yield so stalled
|
||||||
|
// consumers get the CPU they need to release their cells.
|
||||||
|
backoff.wait();
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -560,6 +672,33 @@ mod loom_tests {
|
|||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Finding 13 regression: a consumer stalled between its dequeue_pos
|
||||||
|
/// claim and its seq release must NOT make a lapping producer conclude
|
||||||
|
/// "full" (the old code asserted here — occupancy never exceeded the
|
||||||
|
/// bound; the cell just hadn't been recycled). Capacity 2: fill, pop
|
||||||
|
/// once on each thread, then push a third element — in the
|
||||||
|
/// interleavings where a pop's release is still pending, that push
|
||||||
|
/// laps onto the stalled cell and must wait, not die.
|
||||||
|
#[test]
|
||||||
|
fn mpmc_lap_onto_stalled_consumer_completes() {
|
||||||
|
loom::model(|| {
|
||||||
|
let q = Arc::new(MpmcRing::with_capacity(2));
|
||||||
|
q.push(pid(0));
|
||||||
|
q.push(pid(1));
|
||||||
|
let q2 = q.clone();
|
||||||
|
let h = thread::spawn(move || q2.pop().expect("ring has two elements"));
|
||||||
|
let a = q.pop().expect("ring has two elements");
|
||||||
|
// enqueue_pos = 2 → cell 0: laps onto the other thread's cell
|
||||||
|
// whenever its release is delayed.
|
||||||
|
q.push(pid(2));
|
||||||
|
let b = h.join().unwrap();
|
||||||
|
let c = q.pop().expect("the lapping push must have landed");
|
||||||
|
let mut got = vec![a.index(), b.index(), c.index()];
|
||||||
|
got.sort_unstable();
|
||||||
|
assert_eq!(got, vec![0, 1, 2]);
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
/// Producer races a consumer on a single element: the consumer either
|
/// Producer races a consumer on a single element: the consumer either
|
||||||
/// gets it or sees a clean None — never a torn/duplicated element.
|
/// gets it or sees a clean None — never a torn/duplicated element.
|
||||||
#[test]
|
#[test]
|
||||||
|
|||||||
+178
-24
@@ -76,7 +76,7 @@
|
|||||||
//! time (`rq-mutex` / `rq-mpmc` / `rq-striped`). Queue ops require
|
//! time (`rq-mutex` / `rq-mpmc` / `rq-striped`). Queue ops require
|
||||||
//! preemption disabled (debug-asserted there); when the mutex variant is in
|
//! preemption disabled (debug-asserted there); when the mutex variant is in
|
||||||
//! play it is the innermost lock — nothing else is acquired under it.
|
//! play it is the innermost lock — nothing else is acquired under it.
|
||||||
//! - Per-slot `cold` locks ([`RawMutex`], non-poisoning, guard enters
|
//! - Per-slot `cold` locks (`RawMutex`, non-poisoning, guard enters
|
||||||
//! `NoPreempt`) guard the lifecycle collections. **Leaf rule: never hold
|
//! `NoPreempt`) guard the lifecycle collections. **Leaf rule: never hold
|
||||||
//! two cold locks at once** — `finalize_actor`'s link cascade and `link()`
|
//! two cold locks at once** — `finalize_actor`'s link cascade and `link()`
|
||||||
//! lock peers one at a time (correctness arguments at the call sites).
|
//! lock peers one at a time (correctness arguments at the call sites).
|
||||||
@@ -116,7 +116,7 @@ use crate::actor::{
|
|||||||
take_last_outcome, Actor, Outcome,
|
take_last_outcome, Actor, Outcome,
|
||||||
};
|
};
|
||||||
use crate::channel::Sender;
|
use crate::channel::Sender;
|
||||||
use crate::context::{get_actor_sp, set_actor_sp, switch_to_actor};
|
use crate::context::switch_to_actor;
|
||||||
use crate::io::IoThread;
|
use crate::io::IoThread;
|
||||||
use crate::monitor::{Down, DownReason, MonitorId};
|
use crate::monitor::{Down, DownReason, MonitorId};
|
||||||
use crate::pid::Pid;
|
use crate::pid::Pid;
|
||||||
@@ -144,13 +144,13 @@ pub const DEFAULT_MAX_ACTORS: usize = 16_384;
|
|||||||
/// use smarm::runtime::Config;
|
/// use smarm::runtime::Config;
|
||||||
///
|
///
|
||||||
/// // Use all available CPUs (default):
|
/// // Use all available CPUs (default):
|
||||||
/// let c = Config::default();
|
/// let _all = Config::default();
|
||||||
///
|
///
|
||||||
/// // Exactly 4 scheduler threads:
|
/// // Exactly 4 scheduler threads:
|
||||||
/// let c = Config::exact(4);
|
/// let _four = Config::exact(4);
|
||||||
///
|
///
|
||||||
/// // Between 2 and 8, clamped to available parallelism:
|
/// // Between 2 and 8, clamped to available parallelism:
|
||||||
/// let c = Config::new(2, 8, None);
|
/// let _clamped = Config::new(2, 8, None);
|
||||||
/// ```
|
/// ```
|
||||||
#[derive(Clone, Debug)]
|
#[derive(Clone, Debug)]
|
||||||
pub struct Config {
|
pub struct Config {
|
||||||
@@ -182,7 +182,7 @@ impl Config {
|
|||||||
stack_reserve: DEFAULT_STACK_RESERVE,
|
stack_reserve: DEFAULT_STACK_RESERVE,
|
||||||
stack_guard: DEFAULT_STACK_GUARD,
|
stack_guard: DEFAULT_STACK_GUARD,
|
||||||
max_actors: DEFAULT_MAX_ACTORS,
|
max_actors: DEFAULT_MAX_ACTORS,
|
||||||
wake_slot: false,
|
wake_slot: true,
|
||||||
node_id: crate::pg::DEFAULT_NODE_ID,
|
node_id: crate::pg::DEFAULT_NODE_ID,
|
||||||
incarnation: crate::pg::DEFAULT_INCARNATION,
|
incarnation: crate::pg::DEFAULT_INCARNATION,
|
||||||
}
|
}
|
||||||
@@ -205,7 +205,7 @@ impl Config {
|
|||||||
stack_reserve: DEFAULT_STACK_RESERVE,
|
stack_reserve: DEFAULT_STACK_RESERVE,
|
||||||
stack_guard: DEFAULT_STACK_GUARD,
|
stack_guard: DEFAULT_STACK_GUARD,
|
||||||
max_actors: DEFAULT_MAX_ACTORS,
|
max_actors: DEFAULT_MAX_ACTORS,
|
||||||
wake_slot: false,
|
wake_slot: true,
|
||||||
node_id: crate::pg::DEFAULT_NODE_ID,
|
node_id: crate::pg::DEFAULT_NODE_ID,
|
||||||
incarnation: crate::pg::DEFAULT_INCARNATION,
|
incarnation: crate::pg::DEFAULT_INCARNATION,
|
||||||
}
|
}
|
||||||
@@ -283,7 +283,7 @@ impl Config {
|
|||||||
/// thread's slot; it is resumed next on that core and inherits the
|
/// thread's slot; it is resumed next on that core and inherits the
|
||||||
/// remainder of the waker's timeslice. Scheduler-context wakes
|
/// remainder of the waker's timeslice. Scheduler-context wakes
|
||||||
/// (timer/IO drain) and spawns always go to the shared queue.
|
/// (timer/IO drain) and spawns always go to the shared queue.
|
||||||
/// Default: `false` (off until the slot shootout accepts it).
|
/// Default: `true` (accepted 2026-08-18, history.md finding 17/18).
|
||||||
pub fn wake_slot(mut self, on: bool) -> Self {
|
pub fn wake_slot(mut self, on: bool) -> Self {
|
||||||
self.wake_slot = on;
|
self.wake_slot = on;
|
||||||
self
|
self
|
||||||
@@ -333,7 +333,7 @@ impl Default for Config {
|
|||||||
stack_reserve: DEFAULT_STACK_RESERVE,
|
stack_reserve: DEFAULT_STACK_RESERVE,
|
||||||
stack_guard: DEFAULT_STACK_GUARD,
|
stack_guard: DEFAULT_STACK_GUARD,
|
||||||
max_actors: DEFAULT_MAX_ACTORS,
|
max_actors: DEFAULT_MAX_ACTORS,
|
||||||
wake_slot: false,
|
wake_slot: true,
|
||||||
node_id: crate::pg::DEFAULT_NODE_ID,
|
node_id: crate::pg::DEFAULT_NODE_ID,
|
||||||
incarnation: crate::pg::DEFAULT_INCARNATION,
|
incarnation: crate::pg::DEFAULT_INCARNATION,
|
||||||
}
|
}
|
||||||
@@ -356,6 +356,25 @@ pub struct SchedulerStats {
|
|||||||
pub slot_hits: AtomicU64,
|
pub slot_hits: AtomicU64,
|
||||||
/// RFC 005: slot occupants displaced to the shared queue by a newer wake.
|
/// RFC 005: slot occupants displaced to the shared queue by a newer wake.
|
||||||
pub slot_displacements: AtomicU64,
|
pub slot_displacements: AtomicU64,
|
||||||
|
// --- wake-path diagnostics (target 5). Relaxed counters, cheap. ---
|
||||||
|
/// unpark hit Parked from actor context → slot_push.
|
||||||
|
pub unpark_slot: AtomicU64,
|
||||||
|
/// unpark hit Parked from scheduler/foreign context → shared enqueue.
|
||||||
|
pub unpark_queue: AtomicU64,
|
||||||
|
/// unpark hit Running → RunningNotified (peer had not parked yet).
|
||||||
|
pub unpark_notified: AtomicU64,
|
||||||
|
/// park_return found the flag consumed → shared re-enqueue.
|
||||||
|
pub park_flag_consumed: AtomicU64,
|
||||||
|
/// Yield-intent re-enqueues.
|
||||||
|
pub yield_requeues: AtomicU64,
|
||||||
|
/// `enqueue` tail wake actually delivered a futex permit.
|
||||||
|
pub enqueue_wakes: AtomicU64,
|
||||||
|
/// Chain-rule wake actually delivered a permit.
|
||||||
|
pub chain_wakes: AtomicU64,
|
||||||
|
/// Times this scheduler entered Pop::Idle (about to futex-park).
|
||||||
|
pub idle_parks: AtomicU64,
|
||||||
|
/// Idle parks whose recheck aborted (WorkFound) — no futex.
|
||||||
|
pub idle_recheck_hits: AtomicU64,
|
||||||
}
|
}
|
||||||
|
|
||||||
impl SchedulerStats {
|
impl SchedulerStats {
|
||||||
@@ -365,8 +384,44 @@ impl SchedulerStats {
|
|||||||
run_queue_len: AtomicU64::new(0),
|
run_queue_len: AtomicU64::new(0),
|
||||||
slot_hits: AtomicU64::new(0),
|
slot_hits: AtomicU64::new(0),
|
||||||
slot_displacements: AtomicU64::new(0),
|
slot_displacements: AtomicU64::new(0),
|
||||||
|
unpark_slot: AtomicU64::new(0),
|
||||||
|
unpark_queue: AtomicU64::new(0),
|
||||||
|
unpark_notified: AtomicU64::new(0),
|
||||||
|
park_flag_consumed: AtomicU64::new(0),
|
||||||
|
yield_requeues: AtomicU64::new(0),
|
||||||
|
enqueue_wakes: AtomicU64::new(0),
|
||||||
|
chain_wakes: AtomicU64::new(0),
|
||||||
|
idle_parks: AtomicU64::new(0),
|
||||||
|
idle_recheck_hits: AtomicU64::new(0),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
fn reset(&self) {
|
||||||
|
for c in [
|
||||||
|
&self.slot_hits,
|
||||||
|
&self.slot_displacements,
|
||||||
|
&self.unpark_slot,
|
||||||
|
&self.unpark_queue,
|
||||||
|
&self.unpark_notified,
|
||||||
|
&self.park_flag_consumed,
|
||||||
|
&self.yield_requeues,
|
||||||
|
&self.enqueue_wakes,
|
||||||
|
&self.chain_wakes,
|
||||||
|
&self.idle_parks,
|
||||||
|
&self.idle_recheck_hits,
|
||||||
|
] {
|
||||||
|
c.store(0, Ordering::Relaxed);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Bump a per-thread diagnostic counter on the calling scheduler thread.
|
||||||
|
macro_rules! diag {
|
||||||
|
($inner:expr, $field:ident) => {
|
||||||
|
$inner.stats[sched_slot()]
|
||||||
|
.$field
|
||||||
|
.fetch_add(1, Ordering::Relaxed)
|
||||||
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
@@ -422,6 +477,34 @@ impl RuntimeStats {
|
|||||||
.map(|s| s.slot_displacements.load(Ordering::Relaxed))
|
.map(|s| s.slot_displacements.load(Ordering::Relaxed))
|
||||||
.sum()
|
.sum()
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Wake-path diagnostics summed across threads, as `name=value` pairs
|
||||||
|
/// (target 5 instrumentation). Reset at the start of each `run()`.
|
||||||
|
pub fn wake_diag(&self) -> String {
|
||||||
|
let sum = |f: fn(&SchedulerStats) -> &AtomicU64| -> u64 {
|
||||||
|
self.inner
|
||||||
|
.stats
|
||||||
|
.iter()
|
||||||
|
.map(|s| f(s).load(Ordering::Relaxed))
|
||||||
|
.sum()
|
||||||
|
};
|
||||||
|
format!(
|
||||||
|
"slot_hits={} displaced={} unpark_slot={} unpark_queue={} unpark_notified={} \
|
||||||
|
flag_consumed={} yield_requeues={} enqueue_wakes={} chain_wakes={} \
|
||||||
|
idle_parks={} idle_recheck_hits={}",
|
||||||
|
sum(|s| &s.slot_hits),
|
||||||
|
sum(|s| &s.slot_displacements),
|
||||||
|
sum(|s| &s.unpark_slot),
|
||||||
|
sum(|s| &s.unpark_queue),
|
||||||
|
sum(|s| &s.unpark_notified),
|
||||||
|
sum(|s| &s.park_flag_consumed),
|
||||||
|
sum(|s| &s.yield_requeues),
|
||||||
|
sum(|s| &s.enqueue_wakes),
|
||||||
|
sum(|s| &s.chain_wakes),
|
||||||
|
sum(|s| &s.idle_parks),
|
||||||
|
sum(|s| &s.idle_recheck_hits),
|
||||||
|
)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
@@ -873,6 +956,15 @@ impl Slot {
|
|||||||
}
|
}
|
||||||
|
|
||||||
fn take_closure(&self) -> Option<Closure> {
|
fn take_closure(&self) -> Option<Closure> {
|
||||||
|
// Fast path: every resume after the first (the overwhelming case)
|
||||||
|
// finds null. A plain load suffices to prove it — `store_closure`
|
||||||
|
// runs only before `publish_queued`, whose Release/Acquire pairing
|
||||||
|
// with the claimer's `try_claim` orders it before this call, so no
|
||||||
|
// writer can race the load within an occupancy. This keeps the
|
||||||
|
// locked RMW (full barrier, ~20+ cycles) off the per-resume path.
|
||||||
|
if self.closure.load(Ordering::Relaxed).is_null() {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
let raw = self.closure.swap(std::ptr::null_mut(), Ordering::Acquire);
|
let raw = self.closure.swap(std::ptr::null_mut(), Ordering::Acquire);
|
||||||
if raw.is_null() {
|
if raw.is_null() {
|
||||||
None
|
None
|
||||||
@@ -955,6 +1047,17 @@ pub(crate) struct RuntimeInner {
|
|||||||
/// checks under it read only the atomic slot word, and the eviction path
|
/// checks under it read only the atomic slot word, and the eviction path
|
||||||
/// keeps it off the send path.
|
/// keeps it off the send path.
|
||||||
pub(crate) process_groups: RawMutex<crate::pg::ProcessGroups>,
|
pub(crate) process_groups: RawMutex<crate::pg::ProcessGroups>,
|
||||||
|
/// RFC 010 c8: the exposure registry (exposed names + type-hash decoders).
|
||||||
|
/// RawMutex Leaf, same discipline as `process_groups`; decoders run under
|
||||||
|
/// it and are leaf-only by contract (they decode and send — `send_dyn`
|
||||||
|
/// takes `registry`, never this). cfg-gated: zero-cost-when-off (c1).
|
||||||
|
#[cfg(feature = "cluster")]
|
||||||
|
pub(crate) exposure: RawMutex<crate::cluster::expose::ExposureState>,
|
||||||
|
/// RFC 010 c9: the outbound table (node name → the connection's
|
||||||
|
/// dedicated `Sender<Frame>`), manager-maintained. Leaf; the send happens
|
||||||
|
/// outside the lock. cfg-gated like `exposure`.
|
||||||
|
#[cfg(feature = "cluster")]
|
||||||
|
pub(crate) outbound: RawMutex<crate::cluster::remote::Outbound>,
|
||||||
/// Recycled stacks waiting to be reused by the next spawn.
|
/// Recycled stacks waiting to be reused by the next spawn.
|
||||||
pub(crate) stack_pool: RawMutex<Vec<crate::stack::Stack>>,
|
pub(crate) stack_pool: RawMutex<Vec<crate::stack::Stack>>,
|
||||||
/// Maximum number of stacks to retain in the pool.
|
/// Maximum number of stacks to retain in the pool.
|
||||||
@@ -1013,6 +1116,10 @@ impl RuntimeInner {
|
|||||||
node_id,
|
node_id,
|
||||||
incarnation,
|
incarnation,
|
||||||
process_groups: RawMutex::new(crate::pg::ProcessGroups::new()),
|
process_groups: RawMutex::new(crate::pg::ProcessGroups::new()),
|
||||||
|
#[cfg(feature = "cluster")]
|
||||||
|
exposure: RawMutex::new(crate::cluster::expose::ExposureState::new()),
|
||||||
|
#[cfg(feature = "cluster")]
|
||||||
|
outbound: RawMutex::new(crate::cluster::remote::Outbound::new()),
|
||||||
stack_pool: RawMutex::new(Vec::new()),
|
stack_pool: RawMutex::new(Vec::new()),
|
||||||
stack_pool_cap,
|
stack_pool_cap,
|
||||||
stack_reserve: crate::stack::round_to_pages(stack_reserve),
|
stack_reserve: crate::stack::round_to_pages(stack_reserve),
|
||||||
@@ -1071,7 +1178,9 @@ impl RuntimeInner {
|
|||||||
// pure-compute hot path pays (almost) nothing. Bias is over-wake:
|
// pure-compute hot path pays (almost) nothing. Bias is over-wake:
|
||||||
// a spurious wake costs one futex round-trip and a failed pop; a
|
// a spurious wake costs one futex round-trip and a failed pop; a
|
||||||
// missed wake would cost a stranded actor.
|
// missed wake would cost a stranded actor.
|
||||||
self.coord.wake_one_if_idle();
|
if self.coord.wake_one_if_idle() {
|
||||||
|
diag!(self, enqueue_wakes);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Make `pid` runnable if it is parked; coalesce or defer otherwise.
|
/// Make `pid` runnable if it is parked; coalesce or defer otherwise.
|
||||||
@@ -1128,12 +1237,15 @@ impl RuntimeInner {
|
|||||||
// (install_actor enqueues directly), so they bypass the
|
// (install_actor enqueues directly), so they bypass the
|
||||||
// slot by construction.
|
// slot by construction.
|
||||||
if self.wake_slot && crate::actor::current_pid().is_some() {
|
if self.wake_slot && crate::actor::current_pid().is_some() {
|
||||||
|
diag!(self, unpark_slot);
|
||||||
self.slot_push(pid);
|
self.slot_push(pid);
|
||||||
} else {
|
} else {
|
||||||
|
diag!(self, unpark_queue);
|
||||||
self.enqueue(pid);
|
self.enqueue(pid);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
Unpark::Notified => {
|
Unpark::Notified => {
|
||||||
|
diag!(self, unpark_notified);
|
||||||
crate::te!(crate::trace::Event::UnparkDeferred(pid));
|
crate::te!(crate::trace::Event::UnparkDeferred(pid));
|
||||||
}
|
}
|
||||||
Unpark::Noop => {}
|
Unpark::Noop => {}
|
||||||
@@ -1148,9 +1260,11 @@ impl RuntimeInner {
|
|||||||
/// Displacement: the NEW wake takes the slot (newest is hottest; the old
|
/// Displacement: the NEW wake takes the slot (newest is hottest; the old
|
||||||
/// occupant was about to lose its locality window anyway) and the old
|
/// occupant was about to lose its locality window anyway) and the old
|
||||||
/// occupant is pushed to the shared queue.
|
/// occupant is pushed to the shared queue.
|
||||||
|
#[inline(never)]
|
||||||
fn slot_push(&self, pid: Pid) {
|
fn slot_push(&self, pid: Pid) {
|
||||||
|
crate::context::tls_fence();
|
||||||
debug_assert!(
|
debug_assert!(
|
||||||
!crate::preempt::PREEMPTION_ENABLED.with(|c| c.get()),
|
!crate::preempt::preemption_enabled(),
|
||||||
"slot_push with preemption enabled — a switch mid-op could \
|
"slot_push with preemption enabled — a switch mid-op could \
|
||||||
migrate the actor and split the slot access across threads"
|
migrate the actor and split the slot access across threads"
|
||||||
);
|
);
|
||||||
@@ -1164,11 +1278,9 @@ impl RuntimeInner {
|
|||||||
let displaced = WAKE_SLOT.with(|s| s.replace(Some(pid)));
|
let displaced = WAKE_SLOT.with(|s| s.replace(Some(pid)));
|
||||||
crate::te!(crate::trace::Event::SlotPush(pid));
|
crate::te!(crate::trace::Event::SlotPush(pid));
|
||||||
if let Some(old) = displaced {
|
if let Some(old) = displaced {
|
||||||
SCHED_SLOT.with(|s| {
|
self.stats[sched_slot()]
|
||||||
self.stats[s.get()]
|
.slot_displacements
|
||||||
.slot_displacements
|
.fetch_add(1, Ordering::Relaxed);
|
||||||
.fetch_add(1, Ordering::Relaxed)
|
|
||||||
});
|
|
||||||
self.enqueue(old);
|
self.enqueue(old);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -1321,8 +1433,7 @@ impl Runtime {
|
|||||||
// RFC 005: slot counters reset at the START of a run (not the end),
|
// RFC 005: slot counters reset at the START of a run (not the end),
|
||||||
// so `stats()` read after `run()` returns reports that run's totals.
|
// so `stats()` read after `run()` returns reports that run's totals.
|
||||||
for stat in &self.inner.stats {
|
for stat in &self.inner.stats {
|
||||||
stat.slot_hits.store(0, Ordering::Relaxed);
|
stat.reset();
|
||||||
stat.slot_displacements.store(0, Ordering::Relaxed);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// Spawn the initial actor through the public spawn path (which
|
// Spawn the initial actor through the public spawn path (which
|
||||||
@@ -1333,6 +1444,9 @@ impl Runtime {
|
|||||||
// done" — every remaining top-level actor is asked to shut down (see
|
// done" — every remaining top-level actor is asked to shut down (see
|
||||||
// `finalize_actor` / `shutdown_forest_roots`).
|
// `finalize_actor` / `shutdown_forest_roots`).
|
||||||
self.inner.set_root(initial_handle.pid());
|
self.inner.set_root(initial_handle.pid());
|
||||||
|
// A previous run's group reaper was stopped with that run; forget it
|
||||||
|
// so the first `join` of this run spawns a fresh one.
|
||||||
|
self.inner.process_groups.lock().reset_reaper();
|
||||||
|
|
||||||
// Launch N-1 extra scheduler threads, named `smarm-sched-{slot}` so
|
// Launch N-1 extra scheduler threads, named `smarm-sched-{slot}` so
|
||||||
// they are identifiable in `/proc/<pid>/task/*/comm`, stack dumps and
|
// they are identifiable in `/proc/<pid>/task/*/comm`, stack dumps and
|
||||||
@@ -1550,7 +1664,17 @@ pub(crate) enum YieldIntent {
|
|||||||
Park,
|
Park,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// This scheduler thread's stats index. Read from actor context by every
|
||||||
|
/// `stat!()` on the enqueue/wake paths, hence out of line (`context` docs).
|
||||||
|
#[inline(never)]
|
||||||
|
fn sched_slot() -> usize {
|
||||||
|
crate::context::tls_fence();
|
||||||
|
SCHED_SLOT.with(|s| s.get())
|
||||||
|
}
|
||||||
|
|
||||||
|
#[inline(never)]
|
||||||
pub(crate) fn set_yield_intent(i: YieldIntent) {
|
pub(crate) fn set_yield_intent(i: YieldIntent) {
|
||||||
|
crate::context::tls_fence();
|
||||||
YIELD_INTENT.with(|c| c.set(i));
|
YIELD_INTENT.with(|c| c.set(i));
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1674,6 +1798,10 @@ pub(crate) fn install_actor(
|
|||||||
stack: crate::stack::Stack,
|
stack: crate::stack::Stack,
|
||||||
supervisor: Pid,
|
supervisor: Pid,
|
||||||
closure: Closure,
|
closure: Closure,
|
||||||
|
monitor: Option<(
|
||||||
|
crate::monitor::MonitorId,
|
||||||
|
crate::channel::Sender<crate::monitor::Down>,
|
||||||
|
)>,
|
||||||
) -> Pid {
|
) -> Pid {
|
||||||
let slot = &inner.slots[idx as usize];
|
let slot = &inner.slots[idx as usize];
|
||||||
let gen = slot.generation(); // stable: we own the vacant slot via the free list
|
let gen = slot.generation(); // stable: we own the vacant slot via the free list
|
||||||
@@ -1701,6 +1829,12 @@ pub(crate) fn install_actor(
|
|||||||
cold.outstanding_handles = 1;
|
cold.outstanding_handles = 1;
|
||||||
cold.outcome = None;
|
cold.outcome = None;
|
||||||
cold.pending_io_result = None;
|
cold.pending_io_result = None;
|
||||||
|
// `spawn_monitor`: register before publish, so no scheduler can run
|
||||||
|
// (and finalize) the child before its monitor exists. Same slot as a
|
||||||
|
// `monitor()` registration; the send-from-finalize path is unchanged.
|
||||||
|
if let Some(m) = monitor {
|
||||||
|
cold.monitors.push(m);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
slot.sp.store(sp, Ordering::Relaxed);
|
slot.sp.store(sp, Ordering::Relaxed);
|
||||||
// RFC 019: a fresh incarnation starts with its high-water at the fresh
|
// RFC 019: a fresh incarnation starts with its high-water at the fresh
|
||||||
@@ -1949,6 +2083,22 @@ fn shutdown_forest_roots(inner: &Arc<RuntimeInner>, root: Pid) {
|
|||||||
if pid == root {
|
if pid == root {
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
|
// Lock-free reject first. The slab is `max_actors` entries (16_384 by
|
||||||
|
// default) and is almost entirely vacant at root exit, so locking every
|
||||||
|
// slot's `cold` to discover `actor == None` made this scan cost one
|
||||||
|
// uncontended mutex round-trip per slot — a fixed ~150 µs per run on a
|
||||||
|
// 5900X, and `general.rs` times `init` + `run` together, so it landed
|
||||||
|
// in every smarm bench number. A non-Live slot has no actor to shut
|
||||||
|
// down, and the lock is taken again below for the ones that do.
|
||||||
|
//
|
||||||
|
// This does not weaken the sweep. `is_live_for` is a snapshot, so a
|
||||||
|
// slot can go Live just after we pass it — but that was already true
|
||||||
|
// of a spawn landing after the scan finished, and there is no second
|
||||||
|
// sweep either way (an actor that outlives this scan is its spawner's
|
||||||
|
// business, per the rule below).
|
||||||
|
if !slot.is_live_for(pid) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
// Read the parent under the cold lock (generation-verified); act
|
// Read the parent under the cold lock (generation-verified); act
|
||||||
// outside it — `request_shutdown_inner` sends and may unpark.
|
// outside it — `request_shutdown_inner` sends and may unpark.
|
||||||
let parent = {
|
let parent = {
|
||||||
@@ -2162,13 +2312,17 @@ fn schedule_loop(inner: &Arc<RuntimeInner>, slot_idx: usize) {
|
|||||||
// The mandatory post-publish re-check: a producer that
|
// The mandatory post-publish re-check: a producer that
|
||||||
// enqueued (or a verdict input that flipped) before it
|
// enqueued (or a verdict input that flipped) before it
|
||||||
// could see our idle bit has left us the evidence.
|
// could see our idle bit has left us the evidence.
|
||||||
let _ = inner.coord.park(slot_idx, tk_deadline, || {
|
stats.idle_parks.fetch_add(1, Ordering::Relaxed);
|
||||||
|
let pr = inner.coord.park(slot_idx, tk_deadline, || {
|
||||||
!inner.run_queue.is_empty()
|
!inner.run_queue.is_empty()
|
||||||
|| (inner.live_actors.load(Ordering::Acquire) == 0
|
|| (inner.live_actors.load(Ordering::Acquire) == 0
|
||||||
&& inner.io_outstanding.load(Ordering::Acquire) == 0
|
&& inner.io_outstanding.load(Ordering::Acquire) == 0
|
||||||
&& inner.io_fd_waiters.load(Ordering::Acquire) == 0)
|
&& inner.io_fd_waiters.load(Ordering::Acquire) == 0)
|
||||||
|| inner.coord.deadline_due()
|
|| inner.coord.deadline_due()
|
||||||
});
|
});
|
||||||
|
if matches!(pr, crate::park::ParkResult::WorkFound) {
|
||||||
|
stats.idle_recheck_hits.fetch_add(1, Ordering::Relaxed);
|
||||||
|
}
|
||||||
if tk_deadline.is_some() {
|
if tk_deadline.is_some() {
|
||||||
// Hand the role back BEFORE firing: pop_due can run
|
// Hand the role back BEFORE firing: pop_due can run
|
||||||
// `Send` thunks that insert new timers, and the
|
// `Send` thunks that insert new timers, and the
|
||||||
@@ -2203,8 +2357,8 @@ fn schedule_loop(inner: &Arc<RuntimeInner>, slot_idx: usize) {
|
|||||||
// miss here is safe — the enqueue that created the surplus already
|
// miss here is safe — the enqueue that created the surplus already
|
||||||
// issued its own wake (RFC 018 no-lost-wake); this only sharpens
|
// issued its own wake (RFC 018 no-lost-wake); this only sharpens
|
||||||
// parallelism latency.
|
// parallelism latency.
|
||||||
if !inner.run_queue.is_empty() {
|
if !inner.run_queue.is_empty() && inner.coord.wake_one_if_idle() {
|
||||||
inner.coord.wake_one_if_idle();
|
stats.chain_wakes.fetch_add(1, Ordering::Relaxed);
|
||||||
}
|
}
|
||||||
|
|
||||||
let slot = match inner.slot_at(pid) {
|
let slot = match inner.slot_at(pid) {
|
||||||
@@ -2228,7 +2382,6 @@ fn schedule_loop(inner: &Arc<RuntimeInner>, slot_idx: usize) {
|
|||||||
.current_pid_index
|
.current_pid_index
|
||||||
.store(pid.index(), Ordering::Relaxed);
|
.store(pid.index(), Ordering::Relaxed);
|
||||||
|
|
||||||
set_actor_sp(sp);
|
|
||||||
set_current_pid(pid);
|
set_current_pid(pid);
|
||||||
crate::preempt::set_current_stop(stop_flag);
|
crate::preempt::set_current_stop(stop_flag);
|
||||||
crate::preempt::set_current_slot(slot as *const Slot);
|
crate::preempt::set_current_slot(slot as *const Slot);
|
||||||
@@ -2254,7 +2407,7 @@ fn schedule_loop(inner: &Arc<RuntimeInner>, slot_idx: usize) {
|
|||||||
crate::causal::on_resume(slot);
|
crate::causal::on_resume(slot);
|
||||||
|
|
||||||
crate::te!(crate::trace::Event::Resume(pid));
|
crate::te!(crate::trace::Event::Resume(pid));
|
||||||
unsafe { switch_to_actor() };
|
let saved_sp = unsafe { switch_to_actor(sp) };
|
||||||
|
|
||||||
PREEMPTION_ENABLED.with(|c| c.set(false));
|
PREEMPTION_ENABLED.with(|c| c.set(false));
|
||||||
// RFC 016 Chunk 2: charge the cycles this resume consumed to the actor
|
// RFC 016 Chunk 2: charge the cycles this resume consumed to the actor
|
||||||
@@ -2268,7 +2421,6 @@ fn schedule_loop(inner: &Arc<RuntimeInner>, slot_idx: usize) {
|
|||||||
crate::preempt::clear_current_slot();
|
crate::preempt::clear_current_slot();
|
||||||
|
|
||||||
let intent = YIELD_INTENT.with(|c| c.get());
|
let intent = YIELD_INTENT.with(|c| c.get());
|
||||||
let saved_sp = get_actor_sp();
|
|
||||||
slot.sp.store(saved_sp, Ordering::Relaxed);
|
slot.sp.store(saved_sp, Ordering::Relaxed);
|
||||||
// RFC 019 §2: sampled high-water — one branch + at most one store
|
// RFC 019 §2: sampled high-water — one branch + at most one store
|
||||||
// into the line the store above just dirtied. Relaxed and advisory;
|
// into the line the store above just dirtied. Relaxed and advisory;
|
||||||
@@ -2295,6 +2447,7 @@ fn schedule_loop(inner: &Arc<RuntimeInner>, slot_idx: usize) {
|
|||||||
// arriving mid-run coalesces into the re-queue.
|
// arriving mid-run coalesces into the re-queue.
|
||||||
crate::te!(crate::trace::Event::Yield(pid));
|
crate::te!(crate::trace::Event::Yield(pid));
|
||||||
slot.word.yield_return(gen);
|
slot.word.yield_return(gen);
|
||||||
|
diag!(inner, yield_requeues);
|
||||||
inner.enqueue(pid);
|
inner.enqueue(pid);
|
||||||
}
|
}
|
||||||
YieldIntent::Park => {
|
YieldIntent::Park => {
|
||||||
@@ -2335,6 +2488,7 @@ fn schedule_loop(inner: &Arc<RuntimeInner>, slot_idx: usize) {
|
|||||||
#[cfg(feature = "smarm-causal")]
|
#[cfg(feature = "smarm-causal")]
|
||||||
crate::causal::on_deschedule(slot, false);
|
crate::causal::on_deschedule(slot, false);
|
||||||
crate::te!(crate::trace::Event::UnparkFlagConsumed(pid));
|
crate::te!(crate::trace::Event::UnparkFlagConsumed(pid));
|
||||||
|
diag!(inner, park_flag_consumed);
|
||||||
inner.enqueue(pid);
|
inner.enqueue(pid);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
+76
-19
@@ -85,8 +85,10 @@ use std::sync::{Arc, Weak};
|
|||||||
// released on the wrong thread's copy of the thread-local, corrupting its
|
// released on the wrong thread's copy of the thread-local, corrupting its
|
||||||
// borrow count. `f` is also always runtime bookkeeping that should run to
|
// borrow count. `f` is also always runtime bookkeeping that should run to
|
||||||
// completion without the actor being suspended or unwound partway through.
|
// completion without the actor being suspended or unwound partway through.
|
||||||
|
#[inline(never)]
|
||||||
pub(crate) fn with_runtime<R>(f: impl FnOnce(&Arc<RuntimeInner>) -> R) -> R {
|
pub(crate) fn with_runtime<R>(f: impl FnOnce(&Arc<RuntimeInner>) -> R) -> R {
|
||||||
let prev = crate::preempt::PREEMPTION_ENABLED.with(|c| c.replace(false));
|
crate::context::tls_fence();
|
||||||
|
let prev = crate::preempt::preemption_swap(false);
|
||||||
let result = RUNTIME.with(|r| {
|
let result = RUNTIME.with(|r| {
|
||||||
let b = r.borrow();
|
let b = r.borrow();
|
||||||
let inner = match b.as_ref() {
|
let inner = match b.as_ref() {
|
||||||
@@ -95,17 +97,19 @@ pub(crate) fn with_runtime<R>(f: impl FnOnce(&Arc<RuntimeInner>) -> R) -> R {
|
|||||||
};
|
};
|
||||||
f(inner)
|
f(inner)
|
||||||
});
|
});
|
||||||
crate::preempt::PREEMPTION_ENABLED.with(|c| c.set(prev));
|
crate::preempt::preemption_swap(prev);
|
||||||
result
|
result
|
||||||
}
|
}
|
||||||
|
|
||||||
// Borrow the runtime if present, otherwise `None`. Used on cleanup paths
|
// Borrow the runtime if present, otherwise `None`. Used on cleanup paths
|
||||||
// (e.g. a channel's Drop impl during teardown) that may run after the
|
// (e.g. a channel's Drop impl during teardown) that may run after the
|
||||||
// runtime has already gone away. Same preemption gate as `with_runtime`.
|
// runtime has already gone away. Same preemption gate as `with_runtime`.
|
||||||
|
#[inline(never)]
|
||||||
pub(crate) fn try_with_runtime<R>(f: impl FnOnce(&Arc<RuntimeInner>) -> R) -> Option<R> {
|
pub(crate) fn try_with_runtime<R>(f: impl FnOnce(&Arc<RuntimeInner>) -> R) -> Option<R> {
|
||||||
let prev = crate::preempt::PREEMPTION_ENABLED.with(|c| c.replace(false));
|
crate::context::tls_fence();
|
||||||
|
let prev = crate::preempt::preemption_swap(false);
|
||||||
let result = RUNTIME.with(|r| r.borrow().as_ref().map(f));
|
let result = RUNTIME.with(|r| r.borrow().as_ref().map(f));
|
||||||
crate::preempt::PREEMPTION_ENABLED.with(|c| c.set(prev));
|
crate::preempt::preemption_swap(prev);
|
||||||
result
|
result
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -263,6 +267,7 @@ impl Drop for JoinHandle {
|
|||||||
/// ```
|
/// ```
|
||||||
/// use smarm::SpawnOpts;
|
/// use smarm::SpawnOpts;
|
||||||
/// let opts = SpawnOpts { stack_reserve: Some(8 * 1024 * 1024), ..SpawnOpts::default() };
|
/// let opts = SpawnOpts { stack_reserve: Some(8 * 1024 * 1024), ..SpawnOpts::default() };
|
||||||
|
/// assert_eq!(opts.stack_reserve, Some(8 * 1024 * 1024));
|
||||||
/// ```
|
/// ```
|
||||||
///
|
///
|
||||||
/// Both sizes are page-rounded. The reserve is *virtual* (demand-paged):
|
/// Both sizes are page-rounded. The reserve is *virtual* (demand-paged):
|
||||||
@@ -286,8 +291,8 @@ pub struct SpawnOpts {
|
|||||||
#[non_exhaustive]
|
#[non_exhaustive]
|
||||||
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
|
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
|
||||||
pub enum SpawnError {
|
pub enum SpawnError {
|
||||||
/// The fixed actor slab ([`Config::max_actors`]
|
/// The fixed actor slab ([`Config::max_actors`](crate::runtime::Config::max_actors))
|
||||||
/// (crate::runtime::Config::max_actors)) is full: every slot is claimed
|
/// is full: every slot is claimed
|
||||||
/// by a live actor. This is a routine overload condition, not an
|
/// by a live actor. This is a routine overload condition, not an
|
||||||
/// invariant violation — shed the unit of work (close the socket,
|
/// invariant violation — shed the unit of work (close the socket,
|
||||||
/// return a 503) and try again once actors have died.
|
/// return a 503) and try again once actors have died.
|
||||||
@@ -362,7 +367,7 @@ pub fn spawn_under_with<A>(
|
|||||||
|
|
||||||
let pid = with_runtime(|inner| {
|
let pid = with_runtime(|inner| {
|
||||||
let idx = inner.allocate_slot(); // panics loudly on slab exhaustion
|
let idx = inner.allocate_slot(); // panics loudly on slab exhaustion
|
||||||
crate::runtime::install_actor(inner, idx, sp, stack, supervisor, closure)
|
crate::runtime::install_actor(inner, idx, sp, stack, supervisor, closure, None)
|
||||||
});
|
});
|
||||||
|
|
||||||
JoinHandle {
|
JoinHandle {
|
||||||
@@ -371,6 +376,52 @@ pub fn spawn_under_with<A>(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// [`spawn`] and [`monitor`](crate::monitor()) the child in one step, with no
|
||||||
|
/// window in which the child can die unobserved.
|
||||||
|
///
|
||||||
|
/// `spawn` followed by `monitor(h.pid())` races: on a multi-scheduler
|
||||||
|
/// runtime the child can run to completion before the monitor registers, and
|
||||||
|
/// a monitor on a dead pid delivers [`DownReason::NoProc`](crate::DownReason::NoProc)
|
||||||
|
/// — the real reason (Exit vs Panic) is lost. Here the monitor is registered
|
||||||
|
/// on the child's slot *before* the child is published to any run queue, so
|
||||||
|
/// the `Down` always carries the child's actual termination reason. This is
|
||||||
|
/// Erlang's `spawn_monitor/1`.
|
||||||
|
pub fn spawn_monitor(f: impl FnOnce() + Send + 'static) -> (JoinHandle, crate::monitor::Monitor) {
|
||||||
|
spawn_monitor_with(SpawnOpts::default(), f)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`spawn_monitor`] with per-actor stack shape overrides (RFC 019).
|
||||||
|
pub fn spawn_monitor_with(
|
||||||
|
opts: SpawnOpts,
|
||||||
|
f: impl FnOnce() + Send + 'static,
|
||||||
|
) -> (JoinHandle, crate::monitor::Monitor) {
|
||||||
|
let parent = current_pid().unwrap_or_else(|| with_runtime(|_| crate::runtime::ROOT_PID));
|
||||||
|
let (tx, rx) = crate::channel::channel::<crate::monitor::Down>();
|
||||||
|
let stack = with_runtime(|inner| crate::runtime::acquire_stack(inner, opts));
|
||||||
|
let sp = init_actor_stack(stack.top(), crate::actor::trampoline);
|
||||||
|
let closure: crate::runtime::Closure = Box::new(f);
|
||||||
|
|
||||||
|
let (pid, id) = with_runtime(|inner| {
|
||||||
|
let idx = inner.allocate_slot(); // panics loudly on slab exhaustion
|
||||||
|
let id = inner.alloc_monitor_id();
|
||||||
|
let pid =
|
||||||
|
crate::runtime::install_actor(inner, idx, sp, stack, parent, closure, Some((id, tx)));
|
||||||
|
(pid, id)
|
||||||
|
});
|
||||||
|
|
||||||
|
(
|
||||||
|
JoinHandle {
|
||||||
|
pid,
|
||||||
|
consumed: false,
|
||||||
|
},
|
||||||
|
crate::monitor::Monitor {
|
||||||
|
id,
|
||||||
|
target: pid,
|
||||||
|
rx,
|
||||||
|
},
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
/// [`spawn`] that reports a full actor slab instead of panicking.
|
/// [`spawn`] that reports a full actor slab instead of panicking.
|
||||||
///
|
///
|
||||||
/// Behaviour parity with [`spawn`] in every case except one: when the fixed
|
/// Behaviour parity with [`spawn`] in every case except one: when the fixed
|
||||||
@@ -429,7 +480,7 @@ pub fn try_spawn_under_with<A>(
|
|||||||
|
|
||||||
claimed.0 = None; // install_actor takes ownership of the slot from here
|
claimed.0 = None; // install_actor takes ownership of the slot from here
|
||||||
let pid = with_runtime(|inner| {
|
let pid = with_runtime(|inner| {
|
||||||
crate::runtime::install_actor(inner, idx, sp, stack, supervisor, closure)
|
crate::runtime::install_actor(inner, idx, sp, stack, supervisor, closure, None)
|
||||||
});
|
});
|
||||||
|
|
||||||
Ok(JoinHandle {
|
Ok(JoinHandle {
|
||||||
@@ -545,12 +596,16 @@ pub(crate) fn unpark_at(pid: Pid, epoch: u32) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// The current actor's runtime as a `Weak`, for a waker that must reach the
|
// The current actor's runtime as a `Weak`, for a waker that must reach the
|
||||||
// runtime from a foreign thread later. A channel captures this when its
|
// runtime from a foreign thread later. A channel captures this ONCE, the first
|
||||||
// receiver parks, so a cross-thread `send` can wake without the `RUNTIME`
|
// time its receiver parks, so a cross-thread `send` can wake without the
|
||||||
// thread-local (unset off a scheduler thread). Panics outside `Runtime::run()`,
|
// `RUNTIME` thread-local (unset off a scheduler thread). `None` off a
|
||||||
// the same contract as `begin_wait`.
|
// scheduler thread, where there is nothing to capture.
|
||||||
pub(crate) fn runtime_weak() -> Weak<RuntimeInner> {
|
//
|
||||||
with_runtime(Arc::downgrade)
|
// Deliberately not called per park: `Arc::downgrade` plus the matching drop is
|
||||||
|
// a locked RMW pair on one globally shared counter, and the park/unpark
|
||||||
|
// round-trip is the hot path of every channel workload.
|
||||||
|
pub(crate) fn runtime_weak() -> Option<Weak<RuntimeInner>> {
|
||||||
|
try_with_runtime(Arc::downgrade)
|
||||||
}
|
}
|
||||||
|
|
||||||
// Epoch-matched wake of `pid` from a waker that may or may not be on a
|
// Epoch-matched wake of `pid` from a waker that may or may not be on a
|
||||||
@@ -558,11 +613,14 @@ pub(crate) fn runtime_weak() -> Weak<RuntimeInner> {
|
|||||||
// (preemption-gated, slot-eligible); off one that path is a silent no-op, so
|
// (preemption-gated, slot-eligible); off one that path is a silent no-op, so
|
||||||
// we reach the runtime through `rt` — the `Weak` the waker captured while it
|
// we reach the runtime through `rt` — the `Weak` the waker captured while it
|
||||||
// was in-runtime. Mirrors the IO backend's cross-context wake (io.rs, RFC 018).
|
// was in-runtime. Mirrors the IO backend's cross-context wake (io.rs, RFC 018).
|
||||||
pub(crate) fn unpark_at_via(pid: Pid, epoch: u32, rt: &Weak<RuntimeInner>) {
|
pub(crate) fn unpark_at_via(pid: Pid, epoch: u32, rt: impl FnOnce() -> Option<Weak<RuntimeInner>>) {
|
||||||
if try_with_runtime(|inner| inner.unpark_at(pid, epoch)).is_some() {
|
if try_with_runtime(|inner| inner.unpark_at(pid, epoch)).is_some() {
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
if let Some(inner) = rt.upgrade() {
|
// Off a scheduler thread only: `rt` re-takes the waker's lock to read the
|
||||||
|
// captured `Weak`, which is why it is a closure and not a value — the
|
||||||
|
// in-runtime path above must not pay for it.
|
||||||
|
if let Some(inner) = rt().as_ref().and_then(Weak::upgrade) {
|
||||||
inner.unpark_at(pid, epoch);
|
inner.unpark_at(pid, epoch);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -712,14 +770,13 @@ pub struct NoPreempt(bool);
|
|||||||
|
|
||||||
impl NoPreempt {
|
impl NoPreempt {
|
||||||
pub fn enter() -> Self {
|
pub fn enter() -> Self {
|
||||||
let prev = crate::preempt::PREEMPTION_ENABLED.with(|c| c.replace(false));
|
NoPreempt(crate::preempt::preemption_swap(false))
|
||||||
NoPreempt(prev)
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
impl Drop for NoPreempt {
|
impl Drop for NoPreempt {
|
||||||
fn drop(&mut self) {
|
fn drop(&mut self) {
|
||||||
crate::preempt::PREEMPTION_ENABLED.with(|c| c.set(self.0));
|
crate::preempt::preemption_swap(self.0);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+13
-6
@@ -80,6 +80,13 @@ mod inner {
|
|||||||
// RFC 005 wake slot
|
// RFC 005 wake slot
|
||||||
SlotPush(Pid), // actor-context wake parked in the waking thread's slot
|
SlotPush(Pid), // actor-context wake parked in the waking thread's slot
|
||||||
SlotPop(Pid), // scheduler resumed a pid from its own slot
|
SlotPop(Pid), // scheduler resumed a pid from its own slot
|
||||||
|
// Cluster (RFC 010): the conn actor's verdict on one inbound frame —
|
||||||
|
// local knowledge only, never on the wire; the label is
|
||||||
|
// `InboundVerdict::label()`. No pid: a refused frame has none.
|
||||||
|
ClusterInbound(&'static str),
|
||||||
|
// Cluster (RFC 010): the connector's verdict on one dial attempt —
|
||||||
|
// `"ok"` or `DialError::label()`. No pid.
|
||||||
|
ClusterDial(&'static str),
|
||||||
}
|
}
|
||||||
|
|
||||||
// -----------------------------------------------------------------------
|
// -----------------------------------------------------------------------
|
||||||
@@ -176,18 +183,16 @@ mod inner {
|
|||||||
// Hot path
|
// Hot path
|
||||||
// -----------------------------------------------------------------------
|
// -----------------------------------------------------------------------
|
||||||
|
|
||||||
|
#[inline(never)]
|
||||||
pub fn record(event: Event) {
|
pub fn record(event: Event) {
|
||||||
|
crate::context::tls_fence();
|
||||||
// Disable preemption for the entire duration of record(). Any
|
// Disable preemption for the entire duration of record(). Any
|
||||||
// allocation here (mutex internals, channel send, lazy init) would
|
// allocation here (mutex internals, channel send, lazy init) would
|
||||||
// trigger PreemptingAllocator -> maybe_preempt -> switch_to_scheduler,
|
// trigger PreemptingAllocator -> maybe_preempt -> switch_to_scheduler,
|
||||||
// which would try to re-acquire inner.shared (already held at many
|
// which would try to re-acquire inner.shared (already held at many
|
||||||
// te!() call sites) -> deadlock. Guard at the very top, before any
|
// te!() call sites) -> deadlock. Guard at the very top, before any
|
||||||
// allocation-capable call.
|
// allocation-capable call.
|
||||||
let was_enabled = crate::preempt::PREEMPTION_ENABLED.with(|e| {
|
let was_enabled = crate::preempt::preemption_swap(false);
|
||||||
let v = e.get();
|
|
||||||
e.set(false);
|
|
||||||
v
|
|
||||||
});
|
|
||||||
|
|
||||||
LOCAL_STATE.with(|cell| {
|
LOCAL_STATE.with(|cell| {
|
||||||
let mut opt = cell.borrow_mut();
|
let mut opt = cell.borrow_mut();
|
||||||
@@ -209,7 +214,7 @@ mod inner {
|
|||||||
}
|
}
|
||||||
});
|
});
|
||||||
|
|
||||||
crate::preempt::PREEMPTION_ENABLED.with(|e| e.set(was_enabled));
|
crate::preempt::preemption_swap(was_enabled);
|
||||||
}
|
}
|
||||||
|
|
||||||
// -----------------------------------------------------------------------
|
// -----------------------------------------------------------------------
|
||||||
@@ -293,6 +298,8 @@ mod inner {
|
|||||||
Event::Dequeue(p) => ("dequeue".into(), p.index()),
|
Event::Dequeue(p) => ("dequeue".into(), p.index()),
|
||||||
Event::SlotPush(p) => ("slot_push".into(), p.index()),
|
Event::SlotPush(p) => ("slot_push".into(), p.index()),
|
||||||
Event::SlotPop(p) => ("slot_pop".into(), p.index()),
|
Event::SlotPop(p) => ("slot_pop".into(), p.index()),
|
||||||
|
Event::ClusterInbound(v) => (format!("cluster_inbound {v}"), 0),
|
||||||
|
Event::ClusterDial(v) => (format!("cluster_dial {v}"), 0),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+4
-2
@@ -138,10 +138,12 @@ fn channel_ops_interleaved_with_monitor_churn_multi_thread() {
|
|||||||
let tx = tx.clone();
|
let tx = tx.clone();
|
||||||
handles.push(spawn(move || {
|
handles.push(spawn(move || {
|
||||||
// Short-lived target whose death fires the monitor below.
|
// Short-lived target whose death fires the monitor below.
|
||||||
let t = spawn(move || {
|
// spawn_monitor: registered before publish, so the Down is
|
||||||
|
// the finalize-sent Exit this test is about, never NoProc
|
||||||
|
// (spawn-then-monitor raced ~8% at 4 threads).
|
||||||
|
let (t, m) = smarm::spawn_monitor(move || {
|
||||||
tx.send(i).unwrap();
|
tx.send(i).unwrap();
|
||||||
});
|
});
|
||||||
let m = smarm::monitor(t.pid());
|
|
||||||
t.join().unwrap();
|
t.join().unwrap();
|
||||||
// Down delivery exercises send-from-finalize.
|
// Down delivery exercises send-from-finalize.
|
||||||
let d = m.rx.recv().unwrap();
|
let d = m.rx.recv().unwrap();
|
||||||
|
|||||||
@@ -0,0 +1,117 @@
|
|||||||
|
//! RFC 010 c6a — connection-actor lifecycle against the manager table.
|
||||||
|
//!
|
||||||
|
//! The handshake is bypassed here (c6b wires it): each connection is
|
||||||
|
//! constructed already-established over a real localhost TCP pair, handed a
|
||||||
|
//! fabricated `Peer`, and spawned. `spawn_established` registers it with the
|
||||||
|
//! manager, which takes its handle and monitors it, so the table reflects the
|
||||||
|
//! connection while it lives and reaps it on any exit path. This proves three
|
||||||
|
//! things at once: a live connection shows up, a commanded `Disconnect`
|
||||||
|
//! removes exactly that one, and a peer close (EOF, no command) removes the
|
||||||
|
//! other.
|
||||||
|
//!
|
||||||
|
//! TCP parks the calling actor, so everything runs inside `smarm::run`; the
|
||||||
|
//! single-threaded runtime is fine because every wait is a cooperative fd park.
|
||||||
|
#![cfg(feature = "cluster")]
|
||||||
|
|
||||||
|
use std::time::Duration;
|
||||||
|
|
||||||
|
use smarm::cluster::envelope::NodeMeta;
|
||||||
|
use smarm::cluster::handshake::Peer;
|
||||||
|
use smarm::cluster::manager::{Call, Manager, Reply, MANAGER};
|
||||||
|
use smarm::cluster::spawn_established;
|
||||||
|
use smarm::cluster::transport::tcp::TcpTransport;
|
||||||
|
use smarm::cluster::transport::{Conn, FramedConn, Transport};
|
||||||
|
use smarm::cluster::Timing;
|
||||||
|
use smarm::gen_server::{self, GenServerBuilder};
|
||||||
|
use smarm::pg::Incarnation;
|
||||||
|
use smarm::{run, sleep};
|
||||||
|
|
||||||
|
/// A fabricated post-handshake peer identity. Only `node_name` matters to the
|
||||||
|
/// manager table; the rest is filler until c7 consumes it.
|
||||||
|
fn peer(name: &str) -> Peer {
|
||||||
|
Peer {
|
||||||
|
node_name: name.to_string(),
|
||||||
|
incarnation: Incarnation::new(1),
|
||||||
|
meta: NodeMeta {
|
||||||
|
role: "test".to_string(),
|
||||||
|
region: "test".to_string(),
|
||||||
|
},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One established transport pair over localhost. Relies on TCP backlog so the
|
||||||
|
/// sequential dial-then-accept needs no concurrent acceptor (same assumption as
|
||||||
|
/// the c3 conformance suite).
|
||||||
|
fn pair(t: &dyn Transport) -> (Box<dyn Conn>, Box<dyn Conn>) {
|
||||||
|
let mut l = t.listen("127.0.0.1:0").unwrap();
|
||||||
|
let a = t.dial(&l.local_addr()).unwrap();
|
||||||
|
let b = l.accept().unwrap();
|
||||||
|
(a, b)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Poll the manager until its peer set matches `expected` (sorted), or fail.
|
||||||
|
/// The bound is generous against a sub-millisecond real cost.
|
||||||
|
fn wait_peers(expected: &[&str]) {
|
||||||
|
let want: Vec<String> = expected.iter().map(|s| s.to_string()).collect();
|
||||||
|
for _ in 0..2000 {
|
||||||
|
if let Ok(Reply::Peers(got)) = gen_server::call(MANAGER, Call::Peers) {
|
||||||
|
if got == want {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
sleep(Duration::from_millis(1));
|
||||||
|
}
|
||||||
|
let got = gen_server::call(MANAGER, Call::Peers);
|
||||||
|
panic!("timed out waiting for peers == {want:?}; last = {got:?}");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn connection_up_commanded_shutdown_and_eof_all_reflected_in_table() {
|
||||||
|
run(|| {
|
||||||
|
// The manager, started plainly and reachable at its well-known name.
|
||||||
|
// (The supervised subtree in `cluster::start` is permanent by design;
|
||||||
|
// a plainly-started manager lets this test terminate cleanly.)
|
||||||
|
let mgr = GenServerBuilder::new(Manager::new())
|
||||||
|
.named(MANAGER)
|
||||||
|
.start()
|
||||||
|
.expect("manager name is free");
|
||||||
|
|
||||||
|
let t = TcpTransport;
|
||||||
|
let (a1, b1) = pair(&t);
|
||||||
|
let (a2, b2) = pair(&t);
|
||||||
|
|
||||||
|
// Manage the `a` ends as peers node-b and node-c; keep the `b` far ends
|
||||||
|
// open so neither socket is closed from the far side yet.
|
||||||
|
spawn_established(FramedConn::new(a1), peer("node-b"), Timing::default())
|
||||||
|
.expect("node-b registers");
|
||||||
|
spawn_established(FramedConn::new(a2), peer("node-c"), Timing::default())
|
||||||
|
.expect("node-c registers");
|
||||||
|
|
||||||
|
// Up: both connections register and the table shows them.
|
||||||
|
wait_peers(&["node-b", "node-c"]);
|
||||||
|
|
||||||
|
// A commanded disconnect reaps exactly its own connection: the
|
||||||
|
// manager drops that entry's handle and the actor stops.
|
||||||
|
assert!(matches!(
|
||||||
|
gen_server::call(
|
||||||
|
MANAGER,
|
||||||
|
Call::Disconnect {
|
||||||
|
name: "node-b".to_string()
|
||||||
|
}
|
||||||
|
),
|
||||||
|
Ok(Reply::Disconnected)
|
||||||
|
));
|
||||||
|
wait_peers(&["node-c"]);
|
||||||
|
|
||||||
|
// A peer close (EOF) reaps the other with no command at all.
|
||||||
|
drop(b2);
|
||||||
|
wait_peers(&[]);
|
||||||
|
|
||||||
|
// node-b's far end stayed open until here, so its removal above was the
|
||||||
|
// disconnect command and not an EOF.
|
||||||
|
drop(b1);
|
||||||
|
|
||||||
|
// All connection actors have exited; stop the manager so `run` returns.
|
||||||
|
mgr.shutdown();
|
||||||
|
});
|
||||||
|
}
|
||||||
@@ -0,0 +1,180 @@
|
|||||||
|
//! RFC 010 c6c — heartbeat send + fixed-timeout liveness + teardown.
|
||||||
|
//!
|
||||||
|
//! Each case runs one real connection actor over an in-process localhost TCP
|
||||||
|
//! pair, with the far end held as a raw `FramedConn` (no actor) so the test
|
||||||
|
//! controls exactly what — if anything — the peer says. That gives the three
|
||||||
|
//! protocol-visible facts direct handles: heartbeats appear on the wire
|
||||||
|
//! unprompted; a mute peer is torn down (and reaped from the manager table)
|
||||||
|
//! once `LIVENESS_TIMEOUT` empties; and a peer that does nothing but send
|
||||||
|
//! heartbeats keeps the connection alive past that same window.
|
||||||
|
//!
|
||||||
|
//! Loopback has no fd and cannot drive liveness (documented on the actor),
|
||||||
|
//! so everything here is TCP. TCP parks the calling actor, so everything
|
||||||
|
//! runs inside `smarm::run`.
|
||||||
|
#![cfg(feature = "cluster")]
|
||||||
|
|
||||||
|
use std::time::{Duration, Instant};
|
||||||
|
|
||||||
|
use smarm::cluster::conn::{HEARTBEAT_INTERVAL, LIVENESS_TIMEOUT};
|
||||||
|
use smarm::cluster::envelope::{Frame, NodeMeta};
|
||||||
|
use smarm::cluster::handshake::Peer;
|
||||||
|
use smarm::cluster::manager::{Call, Manager, Reply, MANAGER};
|
||||||
|
use smarm::cluster::spawn_established;
|
||||||
|
use smarm::cluster::transport::tcp::TcpTransport;
|
||||||
|
use smarm::cluster::transport::{Conn, FramedConn, Transport};
|
||||||
|
use smarm::cluster::Timing;
|
||||||
|
use smarm::gen_server::{self, GenServerBuilder};
|
||||||
|
use smarm::pg::Incarnation;
|
||||||
|
use smarm::{run, sleep, spawn};
|
||||||
|
|
||||||
|
/// A fabricated post-handshake peer identity (same shape as the c6a suite).
|
||||||
|
fn peer(name: &str) -> Peer {
|
||||||
|
Peer {
|
||||||
|
node_name: name.to_string(),
|
||||||
|
incarnation: Incarnation::new(1),
|
||||||
|
meta: NodeMeta {
|
||||||
|
role: "test".to_string(),
|
||||||
|
region: "test".to_string(),
|
||||||
|
},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One established transport pair over localhost (TCP backlog covers the
|
||||||
|
/// sequential dial-then-accept, as in the c3 conformance suite).
|
||||||
|
fn pair(t: &dyn Transport) -> (Box<dyn Conn>, Box<dyn Conn>) {
|
||||||
|
let mut l = t.listen("127.0.0.1:0").unwrap();
|
||||||
|
let a = t.dial(&l.local_addr()).unwrap();
|
||||||
|
let b = l.accept().unwrap();
|
||||||
|
(a, b)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn peers() -> Vec<String> {
|
||||||
|
match gen_server::call(MANAGER, Call::Peers) {
|
||||||
|
Ok(Reply::Peers(p)) => p,
|
||||||
|
other => panic!("manager unreachable: {other:?}"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Poll until the manager's peer set matches `expected` (sorted) or `budget`
|
||||||
|
/// runs out.
|
||||||
|
fn wait_peers(expected: &[&str], budget: Duration) {
|
||||||
|
let want: Vec<String> = expected.iter().map(|s| s.to_string()).collect();
|
||||||
|
let deadline = Instant::now() + budget;
|
||||||
|
while Instant::now() < deadline {
|
||||||
|
if peers() == want {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
sleep(Duration::from_millis(10));
|
||||||
|
}
|
||||||
|
panic!(
|
||||||
|
"timed out waiting for peers == {want:?}; last = {:?}",
|
||||||
|
peers()
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The actor emits heartbeats unprompted: the raw far end, saying nothing,
|
||||||
|
/// sees a `Frame::Heartbeat` well within one interval (the first goes out at
|
||||||
|
/// spawn).
|
||||||
|
#[test]
|
||||||
|
fn heartbeats_are_sent_unprompted() {
|
||||||
|
run(|| {
|
||||||
|
let mgr = GenServerBuilder::new(Manager::new())
|
||||||
|
.named(MANAGER)
|
||||||
|
.start()
|
||||||
|
.expect("manager name is free");
|
||||||
|
|
||||||
|
let (a, b) = pair(&TcpTransport);
|
||||||
|
spawn_established(FramedConn::new(a), peer("hb-send"), Timing::default())
|
||||||
|
.expect("register");
|
||||||
|
let mut far = FramedConn::new(b);
|
||||||
|
|
||||||
|
let frame = far
|
||||||
|
.recv_deadline(Instant::now() + HEARTBEAT_INTERVAL)
|
||||||
|
.expect("a heartbeat before one interval elapses");
|
||||||
|
assert_eq!(frame, Some(Frame::Heartbeat));
|
||||||
|
|
||||||
|
// Teardown: closing the far end is an EOF at the actor.
|
||||||
|
far.close();
|
||||||
|
wait_peers(&[], Duration::from_secs(2));
|
||||||
|
mgr.shutdown();
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A mute peer is dead: no inbound frame for `LIVENESS_TIMEOUT` tears the
|
||||||
|
/// connection down and the manager's monitor reaps the table entry. The
|
||||||
|
/// entry is still present well inside the window — the teardown is the
|
||||||
|
/// timer, not an accident of setup.
|
||||||
|
#[test]
|
||||||
|
fn mute_peer_is_torn_down_after_liveness_timeout() {
|
||||||
|
run(|| {
|
||||||
|
let mgr = GenServerBuilder::new(Manager::new())
|
||||||
|
.named(MANAGER)
|
||||||
|
.start()
|
||||||
|
.expect("manager name is free");
|
||||||
|
|
||||||
|
let (a, b) = pair(&TcpTransport);
|
||||||
|
spawn_established(FramedConn::new(a), peer("mute"), Timing::default()).expect("register");
|
||||||
|
// Held open and silent: no frames, no EOF. (Unread inbound
|
||||||
|
// heartbeats sit in kernel buffers; they are 5 bytes each.)
|
||||||
|
let _far = FramedConn::new(b);
|
||||||
|
|
||||||
|
// Well inside the window the connection is still up.
|
||||||
|
sleep(LIVENESS_TIMEOUT / 2);
|
||||||
|
assert_eq!(peers(), vec!["mute".to_string()], "torn down too early");
|
||||||
|
|
||||||
|
// ...and once the window empties it is gone. Generous budget over
|
||||||
|
// the remaining half-window.
|
||||||
|
wait_peers(&[], LIVENESS_TIMEOUT);
|
||||||
|
mgr.shutdown();
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Heartbeats alone keep a connection alive past `LIVENESS_TIMEOUT`: a far
|
||||||
|
/// end that sends `Frame::Heartbeat` at the interval (and nothing else)
|
||||||
|
/// holds the entry; when it goes quiet, liveness finally fires.
|
||||||
|
#[test]
|
||||||
|
fn heartbeats_keep_the_connection_alive() {
|
||||||
|
run(|| {
|
||||||
|
let mgr = GenServerBuilder::new(Manager::new())
|
||||||
|
.named(MANAGER)
|
||||||
|
.start()
|
||||||
|
.expect("manager name is free");
|
||||||
|
|
||||||
|
let (a, b) = pair(&TcpTransport);
|
||||||
|
spawn_established(FramedConn::new(a), peer("kept"), Timing::default()).expect("register");
|
||||||
|
|
||||||
|
// The far heartbeat pump: interval-paced sends until told to stop,
|
||||||
|
// then holds the socket open, silent, so the eventual teardown is
|
||||||
|
// liveness — not EOF.
|
||||||
|
let (ctl_tx, ctl_rx) = smarm::channel::channel::<()>();
|
||||||
|
spawn(move || {
|
||||||
|
let mut far = FramedConn::new(b);
|
||||||
|
// Phase 1: heartbeat at the interval until the first signal.
|
||||||
|
while matches!(ctl_rx.try_recv(), Ok(None)) {
|
||||||
|
far.send(&Frame::Heartbeat).expect("far send");
|
||||||
|
sleep(HEARTBEAT_INTERVAL);
|
||||||
|
}
|
||||||
|
// Phase 2: silent but with the socket held open — dropping
|
||||||
|
// `far` here would EOF the actor and mask the liveness path.
|
||||||
|
// Exits when the test's closure ends and drops `ctl_tx` (an
|
||||||
|
// eternal park would stop `run` from ever returning).
|
||||||
|
while matches!(ctl_rx.try_recv(), Ok(None)) {
|
||||||
|
sleep(Duration::from_millis(20));
|
||||||
|
}
|
||||||
|
});
|
||||||
|
|
||||||
|
// Past the liveness window with margin: still up.
|
||||||
|
sleep(LIVENESS_TIMEOUT + LIVENESS_TIMEOUT / 2);
|
||||||
|
assert_eq!(
|
||||||
|
peers(),
|
||||||
|
vec!["kept".to_string()],
|
||||||
|
"liveness fired despite heartbeats"
|
||||||
|
);
|
||||||
|
|
||||||
|
// Silence the pump; liveness now empties and the entry goes.
|
||||||
|
ctl_tx.send(()).expect("pump alive");
|
||||||
|
wait_peers(&[], LIVENESS_TIMEOUT * 2);
|
||||||
|
mgr.shutdown();
|
||||||
|
// `ctl_tx` drops here, releasing the pump's phase-2 wait.
|
||||||
|
});
|
||||||
|
}
|
||||||
@@ -0,0 +1,482 @@
|
|||||||
|
//! RFC 010 c6b — the handshake on the accept/connect path.
|
||||||
|
//!
|
||||||
|
//! Path-level tests drive [`dial_handshake`]/[`accept_handshake`] over the
|
||||||
|
//! loopback transport on plain threads (its intended use — synchronous, no
|
||||||
|
//! runtime). Integration tests run the manager-backed [`dial`] and
|
||||||
|
//! [`spawn_acceptor`] over real localhost TCP inside `smarm::run`, and the
|
||||||
|
//! two-node case as subprocesses via the c4 harness. Flake budget: see
|
||||||
|
//! tests/common/mod.rs.
|
||||||
|
#![cfg(feature = "cluster")]
|
||||||
|
|
||||||
|
mod common;
|
||||||
|
|
||||||
|
use std::sync::mpsc;
|
||||||
|
use std::time::{Duration, Instant};
|
||||||
|
|
||||||
|
use common::{maybe_child, spawn_node, WAIT};
|
||||||
|
use smarm::cluster::connect::{
|
||||||
|
accept_handshake, dial, dial_handshake, spawn_acceptor, DialError, HandshakeError,
|
||||||
|
HANDSHAKE_TIMEOUT,
|
||||||
|
};
|
||||||
|
use smarm::cluster::envelope::{Frame, NodeMeta, RejectReason};
|
||||||
|
use smarm::cluster::handshake::{Local, PeerStanding};
|
||||||
|
use smarm::cluster::manager::{Call, Manager, Reply, MANAGER};
|
||||||
|
use smarm::cluster::transport::loopback::LoopbackTransport;
|
||||||
|
use smarm::cluster::transport::tcp::TcpTransport;
|
||||||
|
use smarm::cluster::transport::{FramedConn, Transport};
|
||||||
|
use smarm::cluster::Timing;
|
||||||
|
use smarm::gen_server::{self, GenServerBuilder};
|
||||||
|
use smarm::pg::Incarnation;
|
||||||
|
use smarm::{run, sleep};
|
||||||
|
|
||||||
|
const ROLES: &[(&str, fn())] = &[
|
||||||
|
("hs_listener", role_hs_listener),
|
||||||
|
("hs_dialer", role_hs_dialer),
|
||||||
|
];
|
||||||
|
|
||||||
|
const HASH: u64 = 0xC6B0_C6B0_C6B0_C6B0;
|
||||||
|
|
||||||
|
fn local(name: &str) -> Local {
|
||||||
|
Local {
|
||||||
|
node_name: name.into(),
|
||||||
|
incarnation: Incarnation::new(3),
|
||||||
|
build_hash: HASH,
|
||||||
|
meta: NodeMeta {
|
||||||
|
role: "test".into(),
|
||||||
|
region: "test".into(),
|
||||||
|
},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A loopback conn pair as `FramedConn`s, ready for a threaded handshake.
|
||||||
|
fn loopback_pair() -> (FramedConn, FramedConn) {
|
||||||
|
let t = LoopbackTransport::default();
|
||||||
|
let mut l = t.listen("hs").unwrap();
|
||||||
|
let dialer = FramedConn::new(t.dial("hs").unwrap());
|
||||||
|
let accepted = FramedConn::new(l.accept().unwrap());
|
||||||
|
(dialer, accepted)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Far-future deadline for loopback paths, where it cannot fire anyway.
|
||||||
|
fn no_deadline() -> Instant {
|
||||||
|
Instant::now() + Duration::from_secs(3600)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Park the node forever: it has announced everything the parent asserts on,
|
||||||
|
/// and must now hold its connection open until SIGKILLed.
|
||||||
|
fn park() -> ! {
|
||||||
|
loop {
|
||||||
|
sleep(Duration::from_secs(1));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Cooperative bounded receive across the closure/actor boundary. A blocking
|
||||||
|
/// `std::mpsc` wait would park the OS thread and starve the single-threaded
|
||||||
|
/// scheduler, so every wait inside `run` polls with [`sleep`] instead.
|
||||||
|
fn poll_recv<T>(rx: &mpsc::Receiver<T>, what: &str) -> T {
|
||||||
|
let deadline = Instant::now() + WAIT;
|
||||||
|
loop {
|
||||||
|
match rx.try_recv() {
|
||||||
|
Ok(v) => return v,
|
||||||
|
Err(mpsc::TryRecvError::Empty) => {
|
||||||
|
assert!(Instant::now() < deadline, "timed out waiting for {what}");
|
||||||
|
sleep(Duration::from_millis(1));
|
||||||
|
}
|
||||||
|
Err(mpsc::TryRecvError::Disconnected) => panic!("channel closed waiting for {what}"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Path level, over loopback on plain threads
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn loopback_happy_path_establishes_both_ends() {
|
||||||
|
maybe_child(ROLES);
|
||||||
|
let (mut dialer, mut accepted) = loopback_pair();
|
||||||
|
let responder = std::thread::spawn(move || {
|
||||||
|
accept_handshake(
|
||||||
|
&mut accepted,
|
||||||
|
local("node-b"),
|
||||||
|
|name| {
|
||||||
|
assert_eq!(name, "node-a");
|
||||||
|
PeerStanding::Free
|
||||||
|
},
|
||||||
|
no_deadline(),
|
||||||
|
)
|
||||||
|
});
|
||||||
|
let peer_of_dialer = dial_handshake(&mut dialer, &local("node-a"), no_deadline()).unwrap();
|
||||||
|
let peer_of_acceptor = responder.join().unwrap().unwrap();
|
||||||
|
assert_eq!(peer_of_dialer.node_name, "node-b");
|
||||||
|
assert_eq!(peer_of_acceptor.node_name, "node-a");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn loopback_hash_mismatch_rejected_with_frame_then_eof() {
|
||||||
|
maybe_child(ROLES);
|
||||||
|
let (mut dialer, mut accepted) = loopback_pair();
|
||||||
|
let mut wrong = local("node-b");
|
||||||
|
wrong.build_hash ^= 1;
|
||||||
|
let responder = std::thread::spawn(move || {
|
||||||
|
accept_handshake(&mut accepted, wrong, |_| PeerStanding::Free, no_deadline())
|
||||||
|
});
|
||||||
|
// The dial side receives the reject frame — the compatibility anchor.
|
||||||
|
match dial_handshake(&mut dialer, &local("node-a"), no_deadline()) {
|
||||||
|
Err(HandshakeError::Rejected(RejectReason::HashMismatch)) => {}
|
||||||
|
other => panic!("expected HashMismatch reject, got {other:?}"),
|
||||||
|
}
|
||||||
|
match responder.join().unwrap() {
|
||||||
|
Err(HandshakeError::Rejected(RejectReason::HashMismatch)) => {}
|
||||||
|
other => panic!("expected accept side to report the reject, got {other:?}"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn loopback_tie_break_loser_closed_silently() {
|
||||||
|
maybe_child(ROLES);
|
||||||
|
// The inbound dial is from "node-z"; we are "node-a" with our own dial to
|
||||||
|
// node-z in flight. dial_wins("node-z", "node-a") is false, so the
|
||||||
|
// inbound loses: closed with no frame at all.
|
||||||
|
let (mut dialer, mut accepted) = loopback_pair();
|
||||||
|
let responder = std::thread::spawn(move || {
|
||||||
|
accept_handshake(
|
||||||
|
&mut accepted,
|
||||||
|
local("node-a"),
|
||||||
|
|_| PeerStanding::Dialing,
|
||||||
|
no_deadline(),
|
||||||
|
)
|
||||||
|
});
|
||||||
|
// Silent close: the dial side sees EOF, never a frame.
|
||||||
|
match dial_handshake(&mut dialer, &local("node-z"), no_deadline()) {
|
||||||
|
Err(HandshakeError::Closed) => {}
|
||||||
|
other => panic!("expected silent close (Closed), got {other:?}"),
|
||||||
|
}
|
||||||
|
match responder.join().unwrap() {
|
||||||
|
Err(HandshakeError::TieBreakLoss) => {}
|
||||||
|
other => panic!("expected TieBreakLoss on the accept side, got {other:?}"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn loopback_read_ahead_past_hello_survives_into_established_conn() {
|
||||||
|
maybe_child(ROLES);
|
||||||
|
// The buffer trap, proven: the dialer coalesces Hello + Heartbeat before
|
||||||
|
// the responder's first read, so the Heartbeat lands in the shared
|
||||||
|
// FramedConn's decode buffer during the handshake. The dialer sends
|
||||||
|
// nothing afterwards — the post-handshake recv can only succeed if the
|
||||||
|
// read-ahead travelled with the FramedConn.
|
||||||
|
let (mut dialer, mut accepted) = loopback_pair();
|
||||||
|
let (_init, hello) = smarm::cluster::handshake::Initiator::new(&local("node-a"));
|
||||||
|
dialer.send(&hello).unwrap();
|
||||||
|
dialer.send(&Frame::Heartbeat).unwrap();
|
||||||
|
// Both frames are buffered before the responder reads at all.
|
||||||
|
let (tx, rx) = mpsc::channel();
|
||||||
|
std::thread::spawn(move || {
|
||||||
|
let peer = accept_handshake(
|
||||||
|
&mut accepted,
|
||||||
|
local("node-b"),
|
||||||
|
|_| PeerStanding::Free,
|
||||||
|
no_deadline(),
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
let next = accepted.recv();
|
||||||
|
let _ = tx.send((peer, next));
|
||||||
|
});
|
||||||
|
// A bounded wait: if the Heartbeat were NOT carried in the buffer, the
|
||||||
|
// recv above would block forever (the dialer stays open and silent).
|
||||||
|
let (peer, next) = rx
|
||||||
|
.recv_timeout(Duration::from_secs(5))
|
||||||
|
.expect("read-ahead lost: post-handshake recv blocked");
|
||||||
|
assert_eq!(peer.node_name, "node-a");
|
||||||
|
match next {
|
||||||
|
Ok(Some(Frame::Heartbeat)) => {}
|
||||||
|
other => panic!("expected the read-ahead Heartbeat, got {other:?}"),
|
||||||
|
}
|
||||||
|
drop(dialer);
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Deadline + manager integration, over TCP inside the runtime
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn tcp_silent_peer_times_out_on_the_accept_path() {
|
||||||
|
maybe_child(ROLES);
|
||||||
|
run(|| {
|
||||||
|
let t = TcpTransport;
|
||||||
|
let mut l = t.listen("127.0.0.1:0").unwrap();
|
||||||
|
// Connect and then say nothing at all.
|
||||||
|
let silent = t.dial(&l.local_addr()).unwrap();
|
||||||
|
let mut accepted = FramedConn::new(l.accept().unwrap());
|
||||||
|
let (tx, rx) = mpsc::channel();
|
||||||
|
smarm::spawn(move || {
|
||||||
|
let r = accept_handshake(
|
||||||
|
&mut accepted,
|
||||||
|
local("node-b"),
|
||||||
|
|_| PeerStanding::Free,
|
||||||
|
Instant::now() + Duration::from_millis(200),
|
||||||
|
);
|
||||||
|
let _ = tx.send(r);
|
||||||
|
});
|
||||||
|
match poll_recv(&rx, "accept-path outcome") {
|
||||||
|
Err(HandshakeError::TimedOut) => {}
|
||||||
|
other => panic!("expected TimedOut, got {other:?}"),
|
||||||
|
}
|
||||||
|
drop(silent);
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Poll the manager until its peer set matches `expected` (sorted), or fail.
|
||||||
|
fn wait_peers(expected: &[&str]) {
|
||||||
|
let want: Vec<String> = expected.iter().map(|s| s.to_string()).collect();
|
||||||
|
for _ in 0..5000 {
|
||||||
|
if let Ok(Reply::Peers(got)) = gen_server::call(MANAGER, Call::Peers) {
|
||||||
|
if got == want {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
sleep(Duration::from_millis(1));
|
||||||
|
}
|
||||||
|
let got = gen_server::call(MANAGER, Call::Peers);
|
||||||
|
panic!("timed out waiting for peers == {want:?}; last = {got:?}");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn tcp_duplicate_name_rejected_by_acceptor() {
|
||||||
|
maybe_child(ROLES);
|
||||||
|
run(|| {
|
||||||
|
let mgr = GenServerBuilder::new(Manager::new())
|
||||||
|
.named(MANAGER)
|
||||||
|
.start()
|
||||||
|
.expect("manager name is free");
|
||||||
|
let listener = TcpTransport.listen("127.0.0.1:0").unwrap();
|
||||||
|
let acceptor = spawn_acceptor(listener, local("node-b"), Timing::default());
|
||||||
|
let addr = acceptor.local_addr().to_string();
|
||||||
|
|
||||||
|
// First dial offering "dup-node": establishes and registers.
|
||||||
|
let mut first = FramedConn::new(TcpTransport.dial(&addr).unwrap());
|
||||||
|
let peer = dial_handshake(
|
||||||
|
&mut first,
|
||||||
|
&local("dup-node"),
|
||||||
|
Instant::now() + HANDSHAKE_TIMEOUT,
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
assert_eq!(peer.node_name, "node-b");
|
||||||
|
wait_peers(&["dup-node"]);
|
||||||
|
|
||||||
|
// Second dial offering the same name: deterministic NameTaken.
|
||||||
|
let mut second = FramedConn::new(TcpTransport.dial(&addr).unwrap());
|
||||||
|
match dial_handshake(
|
||||||
|
&mut second,
|
||||||
|
&local("dup-node"),
|
||||||
|
Instant::now() + HANDSHAKE_TIMEOUT,
|
||||||
|
) {
|
||||||
|
Err(HandshakeError::Rejected(RejectReason::NameTaken)) => {}
|
||||||
|
other => panic!("expected NameTaken, got {other:?}"),
|
||||||
|
}
|
||||||
|
// The established connection was untouched by the rejected one.
|
||||||
|
wait_peers(&["dup-node"]);
|
||||||
|
|
||||||
|
// Teardown: the acceptor owns no connections, so the established one
|
||||||
|
// is torn down through the table.
|
||||||
|
acceptor.shutdown();
|
||||||
|
assert!(matches!(
|
||||||
|
gen_server::call(
|
||||||
|
MANAGER,
|
||||||
|
Call::Disconnect {
|
||||||
|
name: "dup-node".to_string()
|
||||||
|
}
|
||||||
|
),
|
||||||
|
Ok(Reply::Disconnected)
|
||||||
|
));
|
||||||
|
wait_peers(&[]);
|
||||||
|
first.close();
|
||||||
|
mgr.shutdown();
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn dial_intent_cleared_when_dialer_dies() {
|
||||||
|
maybe_child(ROLES);
|
||||||
|
run(|| {
|
||||||
|
let mgr = GenServerBuilder::new(Manager::new())
|
||||||
|
.named(MANAGER)
|
||||||
|
.start()
|
||||||
|
.expect("manager name is free");
|
||||||
|
let (begun_tx, begun_rx) = mpsc::channel();
|
||||||
|
let (go_tx, go_rx) = mpsc::channel::<()>();
|
||||||
|
smarm::spawn(move || {
|
||||||
|
let me = smarm::self_pid();
|
||||||
|
match gen_server::call(
|
||||||
|
MANAGER,
|
||||||
|
Call::DialBegin {
|
||||||
|
name: "ghost".into(),
|
||||||
|
pid: me,
|
||||||
|
},
|
||||||
|
) {
|
||||||
|
Ok(Reply::DialBegan(true)) => {}
|
||||||
|
other => panic!("DialBegin failed: {other:?}"),
|
||||||
|
}
|
||||||
|
let _ = begun_tx.send(());
|
||||||
|
let () = poll_recv(&go_rx, "go signal");
|
||||||
|
panic!("dialer dies mid-dial");
|
||||||
|
});
|
||||||
|
poll_recv(&begun_rx, "DialBegin done");
|
||||||
|
// While the dialer lives, the intent is visible.
|
||||||
|
match gen_server::call(
|
||||||
|
MANAGER,
|
||||||
|
Call::Standing {
|
||||||
|
peer_name: "ghost".into(),
|
||||||
|
},
|
||||||
|
) {
|
||||||
|
Ok(Reply::Standing(s)) => assert_eq!(s, PeerStanding::Dialing),
|
||||||
|
other => panic!("PeerStanding failed: {other:?}"),
|
||||||
|
}
|
||||||
|
// Kill it; the monitor must clear the intent without cooperation.
|
||||||
|
go_tx.send(()).unwrap();
|
||||||
|
let deadline = Instant::now() + WAIT;
|
||||||
|
loop {
|
||||||
|
match gen_server::call(
|
||||||
|
MANAGER,
|
||||||
|
Call::Standing {
|
||||||
|
peer_name: "ghost".into(),
|
||||||
|
},
|
||||||
|
) {
|
||||||
|
Ok(Reply::Standing(s)) if s != PeerStanding::Dialing => break,
|
||||||
|
_ if Instant::now() > deadline => {
|
||||||
|
panic!("dial intent not cleared after dialer death")
|
||||||
|
}
|
||||||
|
_ => sleep(Duration::from_millis(1)),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
mgr.shutdown();
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Two nodes, two processes: the integrated dial against a real acceptor
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
/// Announce, then park forever. Neither role ever tears its connection
|
||||||
|
/// down: a table entry only exists while the *peer* holds its side open, so
|
||||||
|
/// any teardown here would retract the other node's observation before it
|
||||||
|
/// had made it. The parent reaps both with SIGKILL once it has both
|
||||||
|
/// announcements (see [`common::Node`]'s `Drop`).
|
||||||
|
fn role_hs_listener() {
|
||||||
|
run(|| {
|
||||||
|
let _mgr = GenServerBuilder::new(Manager::new())
|
||||||
|
.named(MANAGER)
|
||||||
|
.start()
|
||||||
|
.expect("manager name is free");
|
||||||
|
let listener = TcpTransport.listen("127.0.0.1:0").unwrap();
|
||||||
|
let acceptor = spawn_acceptor(listener, local("node-b"), Timing::default());
|
||||||
|
println!("LISTENING {}", acceptor.local_addr());
|
||||||
|
wait_peers(&["node-a"]);
|
||||||
|
println!("PEERS node-a");
|
||||||
|
park();
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
fn role_hs_dialer() {
|
||||||
|
let addr = std::env::var("SMARM_PEER_ADDR").expect("SMARM_PEER_ADDR not set");
|
||||||
|
run(move || {
|
||||||
|
let _mgr = GenServerBuilder::new(Manager::new())
|
||||||
|
.named(MANAGER)
|
||||||
|
.start()
|
||||||
|
.expect("manager name is free");
|
||||||
|
let (tx, rx) = mpsc::channel();
|
||||||
|
smarm::spawn(move || {
|
||||||
|
let r = dial(
|
||||||
|
&TcpTransport,
|
||||||
|
&addr,
|
||||||
|
"node-b",
|
||||||
|
&local("node-a"),
|
||||||
|
Timing::default(),
|
||||||
|
);
|
||||||
|
let _ = tx.send(r);
|
||||||
|
});
|
||||||
|
if let Err(e) = poll_recv(&rx, "dial outcome") {
|
||||||
|
println!("DIAL failed: {e:?}");
|
||||||
|
std::process::exit(3);
|
||||||
|
}
|
||||||
|
wait_peers(&["node-b"]);
|
||||||
|
println!("PEERS node-b");
|
||||||
|
park();
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn two_node_integrated_handshake_over_tcp() {
|
||||||
|
maybe_child(ROLES);
|
||||||
|
let mut listener = spawn_node("hs_listener", &[]);
|
||||||
|
let addr = listener.wait_listening();
|
||||||
|
let mut dialer = spawn_node("hs_dialer", &[("SMARM_PEER_ADDR", &addr)]);
|
||||||
|
// Each node reports its own table naming the other: a real dial against a
|
||||||
|
// real acceptor established in both directions. Both nodes then park —
|
||||||
|
// clean-exit behaviour is the c4 harness's own smoke test, and demanding
|
||||||
|
// it here would mean a teardown, which is exactly what cannot be ordered
|
||||||
|
// safely across two processes. Dropping the nodes SIGKILLs them.
|
||||||
|
dialer.wait_line("PEERS node-b", |l| l == "PEERS node-b");
|
||||||
|
listener.wait_line("PEERS node-a", |l| l == "PEERS node-a");
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Integrated-dial guardrails (no acceptor involved)
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn concurrent_dial_to_same_name_refused() {
|
||||||
|
maybe_child(ROLES);
|
||||||
|
run(|| {
|
||||||
|
let mgr = GenServerBuilder::new(Manager::new())
|
||||||
|
.named(MANAGER)
|
||||||
|
.start()
|
||||||
|
.expect("manager name is free");
|
||||||
|
let (begun_tx, begun_rx) = mpsc::channel();
|
||||||
|
let (go_tx, go_rx) = mpsc::channel::<()>();
|
||||||
|
// First dialer parks with the intent held (it never connects —
|
||||||
|
// 'holding the intent' is all this test needs from it).
|
||||||
|
smarm::spawn(move || {
|
||||||
|
let me = smarm::self_pid();
|
||||||
|
assert!(matches!(
|
||||||
|
gen_server::call(
|
||||||
|
MANAGER,
|
||||||
|
Call::DialBegin {
|
||||||
|
name: "node-x".into(),
|
||||||
|
pid: me,
|
||||||
|
}
|
||||||
|
),
|
||||||
|
Ok(Reply::DialBegan(true))
|
||||||
|
));
|
||||||
|
let _ = begun_tx.send(());
|
||||||
|
let () = poll_recv(&go_rx, "go signal");
|
||||||
|
let _ = gen_server::call(
|
||||||
|
MANAGER,
|
||||||
|
Call::DialEnd {
|
||||||
|
name: "node-x".into(),
|
||||||
|
},
|
||||||
|
);
|
||||||
|
});
|
||||||
|
poll_recv(&begun_rx, "DialBegin done");
|
||||||
|
// Second integrated dial to the same name: refused before connecting
|
||||||
|
// (the addr is unroutable on purpose — it must never be dialed).
|
||||||
|
let (tx, rx) = mpsc::channel();
|
||||||
|
smarm::spawn(move || {
|
||||||
|
let r = dial(
|
||||||
|
&TcpTransport,
|
||||||
|
"127.0.0.1:1",
|
||||||
|
"node-x",
|
||||||
|
&local("node-a"),
|
||||||
|
Timing::default(),
|
||||||
|
);
|
||||||
|
let _ = tx.send(r);
|
||||||
|
});
|
||||||
|
match poll_recv(&rx, "second dial outcome") {
|
||||||
|
Err(DialError::AlreadyDialing) => {}
|
||||||
|
other => panic!("expected AlreadyDialing, got {other:?}"),
|
||||||
|
}
|
||||||
|
go_tx.send(()).unwrap();
|
||||||
|
mgr.shutdown();
|
||||||
|
});
|
||||||
|
}
|
||||||
@@ -0,0 +1,115 @@
|
|||||||
|
//! RFC 010 — a seed whose address answers as a *different* name
|
||||||
|
//! (`DialError::PeerNameMismatch`) is dialed once and then parked: the
|
||||||
|
//! connector must not redial it on backoff forever.
|
||||||
|
//!
|
||||||
|
//! Observed from the misdialed peer: each such dial establishes at the
|
||||||
|
//! responder (it registers, `node_up`), then the dialer closes on the name
|
||||||
|
//! check (`node_down`) — one membership blip per attempt. Cross-process: a
|
||||||
|
//! *server* named `server` subscribes and reports; a *client* on fast
|
||||||
|
//! timing (50–500ms backoff) seeds `("wrongname", server_addr)`. After the
|
||||||
|
//! first blip the server counts further `NodeUp`s across 2s — several
|
||||||
|
//! backoff periods. Parked ⇒ zero. Negative-control-verified: with the park
|
||||||
|
//! stubbed out the count is ≥ 1 in the same window.
|
||||||
|
#![cfg(feature = "cluster")]
|
||||||
|
|
||||||
|
mod common;
|
||||||
|
|
||||||
|
use common::{maybe_child, spawn_node};
|
||||||
|
use smarm::cluster::envelope::NodeMeta;
|
||||||
|
use smarm::cluster::membership::{subscribe, NodeEvent};
|
||||||
|
use smarm::cluster::{start, Config, StaticSeeds, Timing};
|
||||||
|
use std::time::{Duration, Instant};
|
||||||
|
|
||||||
|
const ROLES: &[(&str, fn())] = &[("server", role_server), ("client", role_client)];
|
||||||
|
|
||||||
|
fn meta() -> NodeMeta {
|
||||||
|
NodeMeta {
|
||||||
|
role: "mismatch".into(),
|
||||||
|
region: "local".into(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn timing() -> Timing {
|
||||||
|
Timing {
|
||||||
|
initial_backoff: Duration::from_millis(50),
|
||||||
|
max_backoff: Duration::from_millis(500),
|
||||||
|
..Timing::default()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn role_server() {
|
||||||
|
smarm::run(|| {
|
||||||
|
let cluster = start(Config {
|
||||||
|
node_name: "server".into(),
|
||||||
|
meta: meta(),
|
||||||
|
listen_addr: std::env::var("SMARM_LISTEN_ADDR")
|
||||||
|
.unwrap_or_else(|_| "127.0.0.1:0".into()),
|
||||||
|
strategy: Box::new(StaticSeeds::new(Vec::<(String, String)>::new())),
|
||||||
|
timing: timing(),
|
||||||
|
})
|
||||||
|
.expect("binds");
|
||||||
|
let ev = subscribe().unwrap();
|
||||||
|
println!("LISTENING {}", cluster.local_addr());
|
||||||
|
// First blip: the misdialed client establishes, then closes on us.
|
||||||
|
loop {
|
||||||
|
match ev.rx.recv() {
|
||||||
|
Ok(NodeEvent::NodeDown(i)) if i.name == "client" => break,
|
||||||
|
Ok(_) => continue,
|
||||||
|
Err(_) => panic!("manager gone"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
println!("BLIP");
|
||||||
|
// Now count further NodeUps across several backoff periods.
|
||||||
|
let mut more = 0usize;
|
||||||
|
let t0 = Instant::now();
|
||||||
|
while t0.elapsed() < Duration::from_millis(2000) {
|
||||||
|
match ev.rx.try_recv() {
|
||||||
|
Ok(Some(NodeEvent::NodeUp(i))) if i.name == "client" => more += 1,
|
||||||
|
Ok(_) => {}
|
||||||
|
Err(_) => panic!("manager gone"),
|
||||||
|
}
|
||||||
|
smarm::sleep(Duration::from_millis(50));
|
||||||
|
}
|
||||||
|
println!("MORE {more}");
|
||||||
|
loop {
|
||||||
|
smarm::sleep(Duration::from_secs(3600));
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
fn role_client() {
|
||||||
|
let server_addr = std::env::var("SMARM_SERVER_ADDR").expect("SMARM_SERVER_ADDR");
|
||||||
|
smarm::run(move || {
|
||||||
|
let _cluster = start(Config {
|
||||||
|
node_name: "client".into(),
|
||||||
|
meta: meta(),
|
||||||
|
listen_addr: "127.0.0.1:0".into(),
|
||||||
|
strategy: Box::new(StaticSeeds::new(vec![(
|
||||||
|
"wrongname".to_string(),
|
||||||
|
server_addr,
|
||||||
|
)])),
|
||||||
|
timing: timing(),
|
||||||
|
})
|
||||||
|
.expect("binds");
|
||||||
|
println!("CLIENT UP");
|
||||||
|
loop {
|
||||||
|
smarm::sleep(Duration::from_secs(3600));
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn mismatched_seed_is_dialed_once_then_parked() {
|
||||||
|
maybe_child(ROLES);
|
||||||
|
let mut server = spawn_node("server", &[]);
|
||||||
|
let saddr = server.wait_listening();
|
||||||
|
let mut client = spawn_node("client", &[("SMARM_SERVER_ADDR", &saddr)]);
|
||||||
|
client.wait_line("CLIENT UP", |l| l == "CLIENT UP");
|
||||||
|
server.wait_line("BLIP", |l| l == "BLIP");
|
||||||
|
let line = server.wait_line("MORE", |l| l.starts_with("MORE "));
|
||||||
|
let more: usize = line.split_whitespace().nth(1).unwrap().parse().unwrap();
|
||||||
|
assert_eq!(
|
||||||
|
more, 0,
|
||||||
|
"mismatched seed was redialed {more}× after being parked"
|
||||||
|
);
|
||||||
|
}
|
||||||
@@ -0,0 +1,379 @@
|
|||||||
|
//! RFC 010 c13 — connection-loss synthesis.
|
||||||
|
//!
|
||||||
|
//! Local suite (`run()`, no network): the read-side backstop. A
|
||||||
|
//! `RemoteMonitor` whose channel closes without a notice reads as
|
||||||
|
//! `Disconnected` exactly once (a `Monitor` command that reached the conn
|
||||||
|
//! actor's inbox but was never processed — the drain gap); after
|
||||||
|
//! `demonitor_remote` a closed channel stays a plain `Err`, never a notice.
|
||||||
|
//!
|
||||||
|
//! Cross-process: the headline contrast — an actor's own death gives its
|
||||||
|
//! TRUE reason, loss of the LINK gives `Disconnected` (both a commanded
|
||||||
|
//! `Disconnect` and a SIGKILLed peer process are `Disconnected` from the
|
||||||
|
//! monitor's view: nobody is left to say otherwise). Reconnect does not
|
||||||
|
//! resurrect: the old monitor yields nothing more, proven by stream ORDER
|
||||||
|
//! (a fresh monitor over the new link delivers first). The ignored test
|
||||||
|
//! trips liveness by SIGSTOP and then drops the link too, asserting exactly
|
||||||
|
//! one notice for one monitor.
|
||||||
|
#![cfg(feature = "cluster")]
|
||||||
|
|
||||||
|
mod common;
|
||||||
|
|
||||||
|
use common::{maybe_child, spawn_node};
|
||||||
|
use smarm::cluster::envelope::NodeMeta;
|
||||||
|
use smarm::cluster::expose::{expose, expose_type};
|
||||||
|
use smarm::cluster::manager::{Call, Reply, MANAGER};
|
||||||
|
use smarm::cluster::membership::{subscribe, NodeEvent};
|
||||||
|
use smarm::cluster::remote::{
|
||||||
|
self, demonitor_remote, monitor_remote, send_to_remote, RemoteName, RemotePid,
|
||||||
|
};
|
||||||
|
use smarm::cluster::{start, Config, RemoteDownReason, StaticSeeds, Timing};
|
||||||
|
use smarm::pg::Incarnation;
|
||||||
|
use smarm::{
|
||||||
|
channel, gen_server, install, register, run, spawn, Addressable, DownReason, Erased, Name, Pid,
|
||||||
|
};
|
||||||
|
use std::collections::HashMap;
|
||||||
|
use std::time::Duration;
|
||||||
|
|
||||||
|
// ---- message types (hand-rolled serde; the crate is derive-less) ---------
|
||||||
|
|
||||||
|
#[derive(Debug)]
|
||||||
|
struct Ctl {
|
||||||
|
cmd: String,
|
||||||
|
reply_to: RemotePid<Client>,
|
||||||
|
}
|
||||||
|
#[derive(Debug)]
|
||||||
|
struct Answer {
|
||||||
|
text: String,
|
||||||
|
pid: Option<RemotePid<Erased>>,
|
||||||
|
}
|
||||||
|
struct Client;
|
||||||
|
impl Addressable for Client {
|
||||||
|
type Msg = Answer;
|
||||||
|
}
|
||||||
|
|
||||||
|
impl serde::Serialize for Ctl {
|
||||||
|
fn serialize<S: serde::Serializer>(&self, s: S) -> Result<S::Ok, S::Error> {
|
||||||
|
use serde::ser::SerializeTuple;
|
||||||
|
let mut t = s.serialize_tuple(2)?;
|
||||||
|
t.serialize_element(&self.cmd)?;
|
||||||
|
t.serialize_element(&self.reply_to)?;
|
||||||
|
t.end()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
impl<'de> serde::Deserialize<'de> for Ctl {
|
||||||
|
fn deserialize<D: serde::Deserializer<'de>>(d: D) -> Result<Self, D::Error> {
|
||||||
|
let (cmd, reply_to) = <(String, RemotePid<Client>)>::deserialize(d)?;
|
||||||
|
Ok(Ctl { cmd, reply_to })
|
||||||
|
}
|
||||||
|
}
|
||||||
|
impl serde::Serialize for Answer {
|
||||||
|
fn serialize<S: serde::Serializer>(&self, s: S) -> Result<S::Ok, S::Error> {
|
||||||
|
use serde::ser::SerializeTuple;
|
||||||
|
let mut t = s.serialize_tuple(2)?;
|
||||||
|
t.serialize_element(&self.text)?;
|
||||||
|
t.serialize_element(&self.pid)?;
|
||||||
|
t.end()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
impl<'de> serde::Deserialize<'de> for Answer {
|
||||||
|
fn deserialize<D: serde::Deserializer<'de>>(d: D) -> Result<Self, D::Error> {
|
||||||
|
let (text, pid) = <(String, Option<RemotePid<Erased>>)>::deserialize(d)?;
|
||||||
|
Ok(Answer { text, pid })
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ================= local suite =========================================
|
||||||
|
|
||||||
|
/// A `Monitor` command handed to the connection but never processed (its
|
||||||
|
/// receiver dropped unread) reads as `Disconnected` — once. A second read
|
||||||
|
/// is the ordinary closed-channel `Err`, so "exactly one notice" holds.
|
||||||
|
#[test]
|
||||||
|
fn unread_command_reads_as_disconnected_once() {
|
||||||
|
maybe_child(ROLES);
|
||||||
|
run(|| {
|
||||||
|
remote::set_local_identity("me", Incarnation::new(7));
|
||||||
|
let (probe_tx, _probe_rx) = channel();
|
||||||
|
let inbox =
|
||||||
|
remote::bind_outbound_probe_with_monitors("peer", Incarnation::new(5), probe_tx);
|
||||||
|
let target = RemotePid::<Erased>::from_parts("peer", Incarnation::new(5), 9, 1);
|
||||||
|
let m = monitor_remote(target.clone());
|
||||||
|
assert!(
|
||||||
|
matches!(m.try_recv(), Ok(None)),
|
||||||
|
"command is in flight, no notice yet"
|
||||||
|
);
|
||||||
|
drop(inbox); // the conn actor died with the command unread
|
||||||
|
let d = m.recv().unwrap();
|
||||||
|
assert_eq!(d.pid, target);
|
||||||
|
assert_eq!(d.reason, RemoteDownReason::Disconnected);
|
||||||
|
assert!(
|
||||||
|
m.recv().is_err(),
|
||||||
|
"second read is closed, not a second notice"
|
||||||
|
);
|
||||||
|
assert!(m.try_recv().is_err());
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// After `demonitor_remote`, a closed channel is a closed channel: no
|
||||||
|
/// notice is synthesized for a monitor the caller cancelled.
|
||||||
|
#[test]
|
||||||
|
fn cancelled_monitor_never_synthesizes() {
|
||||||
|
maybe_child(ROLES);
|
||||||
|
run(|| {
|
||||||
|
remote::set_local_identity("me", Incarnation::new(7));
|
||||||
|
let (probe_tx, _probe_rx) = channel();
|
||||||
|
let inbox =
|
||||||
|
remote::bind_outbound_probe_with_monitors("peer", Incarnation::new(5), probe_tx);
|
||||||
|
let target = RemotePid::<Erased>::from_parts("peer", Incarnation::new(5), 9, 1);
|
||||||
|
let m = monitor_remote(target);
|
||||||
|
demonitor_remote(&m);
|
||||||
|
drop(inbox);
|
||||||
|
assert!(m.recv().is_err());
|
||||||
|
assert!(m.try_recv().is_err());
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
// ================= cross-process ======================================
|
||||||
|
|
||||||
|
const ROLES: &[(&str, fn())] = &[
|
||||||
|
("server", role_server),
|
||||||
|
("client", role_client),
|
||||||
|
("client_stop", role_client_stop),
|
||||||
|
];
|
||||||
|
|
||||||
|
const CTL: Name<Ctl> = Name::new("c13.ctl");
|
||||||
|
|
||||||
|
fn cfg(name: &str, seeds: Vec<(String, String)>) -> Config {
|
||||||
|
Config {
|
||||||
|
node_name: name.to_string(),
|
||||||
|
meta: NodeMeta {
|
||||||
|
role: "c13".into(),
|
||||||
|
region: "local".into(),
|
||||||
|
},
|
||||||
|
listen_addr: std::env::var("SMARM_LISTEN_ADDR").unwrap_or_else(|_| "127.0.0.1:0".into()),
|
||||||
|
strategy: Box::new(StaticSeeds::new(seeds)),
|
||||||
|
timing: timing(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The p11 knobs make the liveness test fast: both roles of that test are
|
||||||
|
/// spawned with `SMARM_FAST_TIMING=1` and agree on a 100ms heartbeat /
|
||||||
|
/// 500ms liveness window. Everything else runs the shipping defaults.
|
||||||
|
fn timing() -> Timing {
|
||||||
|
if std::env::var_os("SMARM_FAST_TIMING").is_some() {
|
||||||
|
Timing {
|
||||||
|
heartbeat_interval: Duration::from_millis(100),
|
||||||
|
liveness_timeout: Duration::from_millis(500),
|
||||||
|
initial_backoff: Duration::from_millis(50),
|
||||||
|
max_backoff: Duration::from_millis(500),
|
||||||
|
..Timing::default()
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
Timing::default()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn wait_up(events: &smarm::cluster::membership::MembershipEvents, who: &str) {
|
||||||
|
loop {
|
||||||
|
match events.rx.recv() {
|
||||||
|
Ok(NodeEvent::NodeUp(i)) if i.name == who => return,
|
||||||
|
Ok(_) => continue,
|
||||||
|
Err(_) => panic!("manager gone"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn disconnect(name: &str) {
|
||||||
|
assert!(matches!(
|
||||||
|
gen_server::call(
|
||||||
|
MANAGER,
|
||||||
|
Call::Disconnect {
|
||||||
|
name: name.to_string()
|
||||||
|
}
|
||||||
|
),
|
||||||
|
Ok(Reply::Disconnected)
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Server: `spawn` ⇒ a parked worker (answer carries its pid);
|
||||||
|
/// `kill:<index>` releases it, whereupon it returns (Exit).
|
||||||
|
fn role_server() {
|
||||||
|
smarm::run(move || {
|
||||||
|
let cluster = start(cfg("server", vec![])).expect("binds");
|
||||||
|
println!("LISTENING {}", cluster.local_addr());
|
||||||
|
let (tx, rx) = channel::<Ctl>();
|
||||||
|
register(CTL, tx).unwrap();
|
||||||
|
expose(CTL);
|
||||||
|
println!("READY");
|
||||||
|
let mut workers: HashMap<u32, smarm::channel::Sender<()>> = HashMap::new();
|
||||||
|
loop {
|
||||||
|
let ctl = rx.recv().unwrap();
|
||||||
|
println!("CTL {}", ctl.cmd);
|
||||||
|
let (text, pid): (String, Option<RemotePid<Erased>>) = match ctl.cmd.as_str() {
|
||||||
|
"spawn" => {
|
||||||
|
let (go_tx, go_rx) = channel::<()>();
|
||||||
|
let p: Pid = spawn(move || {
|
||||||
|
let _ = go_rx.recv();
|
||||||
|
})
|
||||||
|
.pid();
|
||||||
|
workers.insert(p.index(), go_tx);
|
||||||
|
(
|
||||||
|
"ok".into(),
|
||||||
|
Some(RemotePid::from_local(p).expect("identity set")),
|
||||||
|
)
|
||||||
|
}
|
||||||
|
other => {
|
||||||
|
let idx: u32 = other.strip_prefix("kill:").unwrap().parse().unwrap();
|
||||||
|
if let Some(go) = workers.remove(&idx) {
|
||||||
|
let _ = go.send(());
|
||||||
|
}
|
||||||
|
("killed".into(), None)
|
||||||
|
}
|
||||||
|
};
|
||||||
|
send_to_remote(ctl.reply_to, Answer { text, pid }).unwrap();
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Client-side setup shared by both client roles: join, expose the reply
|
||||||
|
/// path, hand back an `ask` closure and the membership stream.
|
||||||
|
fn client_setup() -> (
|
||||||
|
smarm::cluster::Cluster,
|
||||||
|
smarm::cluster::membership::MembershipEvents,
|
||||||
|
impl Fn(&str) -> Answer,
|
||||||
|
) {
|
||||||
|
let server_addr = std::env::var("SMARM_SERVER_ADDR").expect("SMARM_SERVER_ADDR");
|
||||||
|
let cluster = start(cfg("client", vec![("server".into(), server_addr)])).expect("binds");
|
||||||
|
let ev = subscribe().unwrap();
|
||||||
|
wait_up(&ev, "server");
|
||||||
|
let (tx, rx) = channel::<Answer>();
|
||||||
|
let me: Pid<Client> = install::<Client>(tx);
|
||||||
|
expose_type::<Answer>();
|
||||||
|
let ask = move |cmd: &str| -> Answer {
|
||||||
|
remote::send(
|
||||||
|
RemoteName::new("server", CTL),
|
||||||
|
Ctl {
|
||||||
|
cmd: cmd.into(),
|
||||||
|
reply_to: RemotePid::from_local(me).expect("identity set"),
|
||||||
|
},
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
rx.recv().unwrap()
|
||||||
|
};
|
||||||
|
(cluster, ev, ask)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn role_client() {
|
||||||
|
smarm::run(move || {
|
||||||
|
let (_cluster, ev, ask) = client_setup();
|
||||||
|
|
||||||
|
// 1. Headline: actor death ⇒ TRUE reason; link cut ⇒ Disconnected.
|
||||||
|
let a = ask("spawn").pid.unwrap();
|
||||||
|
let b = ask("spawn").pid.unwrap();
|
||||||
|
let ma = monitor_remote(a.clone());
|
||||||
|
let mb = monitor_remote(b.clone());
|
||||||
|
ask(&format!("kill:{}", a.index()));
|
||||||
|
let d = ma.recv().unwrap();
|
||||||
|
assert_eq!(d.pid, a);
|
||||||
|
println!("DOWN actor {:?}", d.reason);
|
||||||
|
disconnect("server");
|
||||||
|
let d = mb.recv().unwrap();
|
||||||
|
assert_eq!(d.pid, b);
|
||||||
|
println!("DOWN link {:?}", d.reason);
|
||||||
|
|
||||||
|
// 2. Reconnect does not resurrect. The connector redials on
|
||||||
|
// node_down; over the NEW link a fresh monitor delivers, while
|
||||||
|
// the old one (already answered) yields nothing further — order
|
||||||
|
// proves it, and `b` is even still alive on the server.
|
||||||
|
wait_up(&ev, "server");
|
||||||
|
println!("RECONNECTED");
|
||||||
|
let c = ask("spawn").pid.unwrap();
|
||||||
|
let mc = monitor_remote(c.clone());
|
||||||
|
ask(&format!("kill:{}", b.index()));
|
||||||
|
ask(&format!("kill:{}", c.index()));
|
||||||
|
assert_eq!(mc.recv().unwrap().reason, DownReason::Exit.into());
|
||||||
|
let stray = matches!(mb.try_recv(), Ok(Some(_)));
|
||||||
|
println!("RESURRECT stray={stray}");
|
||||||
|
|
||||||
|
// 3. Peer PROCESS killed ⇒ Disconnected too (nobody is left to send
|
||||||
|
// Down): the parent SIGKILLs the server once it sees the marker.
|
||||||
|
let e = ask("spawn").pid.unwrap();
|
||||||
|
let me_ = monitor_remote(e.clone());
|
||||||
|
println!("KILL SERVER NOW");
|
||||||
|
let d = me_.recv().unwrap();
|
||||||
|
assert_eq!(d.pid, e);
|
||||||
|
println!("DOWN procdeath {:?}", d.reason);
|
||||||
|
|
||||||
|
println!("CLIENT DONE");
|
||||||
|
loop {
|
||||||
|
smarm::sleep(Duration::from_secs(3600));
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The slow role: liveness expiry (peer SIGSTOPped) followed by the link
|
||||||
|
/// dropping for real (peer SIGKILLed) — one monitor, exactly one notice.
|
||||||
|
fn role_client_stop() {
|
||||||
|
smarm::run(move || {
|
||||||
|
let (_cluster, _ev, ask) = client_setup();
|
||||||
|
let a = ask("spawn").pid.unwrap();
|
||||||
|
let ma = monitor_remote(a.clone());
|
||||||
|
println!("STOP SERVER NOW");
|
||||||
|
let d = ma.recv().unwrap(); // liveness expiry, ~liveness_timeout
|
||||||
|
assert_eq!(d.pid, a);
|
||||||
|
println!("DOWN stopped {:?}", d.reason);
|
||||||
|
println!("KILL SERVER NOW");
|
||||||
|
// Give the drop every chance to produce a second notice, then look.
|
||||||
|
smarm::sleep(Duration::from_secs(1));
|
||||||
|
let dup = matches!(ma.try_recv(), Ok(Some(_)));
|
||||||
|
println!("DUPLICATE dup={dup}");
|
||||||
|
println!("CLIENT DONE");
|
||||||
|
loop {
|
||||||
|
smarm::sleep(Duration::from_secs(3600));
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The Phase 4 c13 gate: partition vs. death distinguishable; nothing
|
||||||
|
/// survives reconnect; a dead peer process is a Disconnected too.
|
||||||
|
#[test]
|
||||||
|
fn link_loss_is_disconnected_and_does_not_survive_reconnect() {
|
||||||
|
maybe_child(ROLES);
|
||||||
|
let mut server = spawn_node("server", &[]);
|
||||||
|
let saddr = server.wait_listening();
|
||||||
|
server.wait_line("READY", |l| l == "READY");
|
||||||
|
let mut client = spawn_node("client", &[("SMARM_SERVER_ADDR", &saddr)]);
|
||||||
|
client.wait_line("DOWN actor Local(Exit)", |l| l == "DOWN actor Local(Exit)");
|
||||||
|
client.wait_line("DOWN link Disconnected", |l| l == "DOWN link Disconnected");
|
||||||
|
client.wait_line("RECONNECTED", |l| l == "RECONNECTED");
|
||||||
|
client.wait_line("RESURRECT stray=false", |l| l == "RESURRECT stray=false");
|
||||||
|
client.wait_line("KILL SERVER NOW", |l| l == "KILL SERVER NOW");
|
||||||
|
server.kill();
|
||||||
|
client.wait_line("DOWN procdeath Disconnected", |l| {
|
||||||
|
l == "DOWN procdeath Disconnected"
|
||||||
|
});
|
||||||
|
client.wait_line("CLIENT DONE", |l| l == "CLIENT DONE");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Covers the invariant the headline test cannot: liveness expiry and the
|
||||||
|
/// transport drop both firing for the same connection yield ONE notice.
|
||||||
|
/// Runs on the fast [`timing`] (both roles) — was `#[ignore]`d at the 4s
|
||||||
|
/// default until the p11 knobs landed.
|
||||||
|
#[test]
|
||||||
|
fn timeout_then_drop_yields_one_notice() {
|
||||||
|
maybe_child(ROLES);
|
||||||
|
let fast = ("SMARM_FAST_TIMING", "1");
|
||||||
|
let mut server = spawn_node("server", &[fast]);
|
||||||
|
let saddr = server.wait_listening();
|
||||||
|
server.wait_line("READY", |l| l == "READY");
|
||||||
|
let mut client = spawn_node("client_stop", &[("SMARM_SERVER_ADDR", &saddr), fast]);
|
||||||
|
client.wait_line("STOP SERVER NOW", |l| l == "STOP SERVER NOW");
|
||||||
|
let spid = server.pid().expect("server alive") as libc::pid_t;
|
||||||
|
assert_eq!(unsafe { libc::kill(spid, libc::SIGSTOP) }, 0);
|
||||||
|
client.wait_line("DOWN stopped Disconnected", |l| {
|
||||||
|
l == "DOWN stopped Disconnected"
|
||||||
|
});
|
||||||
|
client.wait_line("KILL SERVER NOW", |l| l == "KILL SERVER NOW");
|
||||||
|
server.kill(); // SIGKILL works on a stopped process; Drop would too
|
||||||
|
client.wait_line("DUPLICATE dup=false", |l| l == "DUPLICATE dup=false");
|
||||||
|
client.wait_line("CLIENT DONE", |l| l == "CLIENT DONE");
|
||||||
|
}
|
||||||
@@ -0,0 +1,161 @@
|
|||||||
|
//! RFC 010 — `Discovery::Withdrawn`: a strategy retracts a candidate and the
|
||||||
|
//! connector stops dialing it.
|
||||||
|
//!
|
||||||
|
//! Cross-process: a plain *server* node, and a *client* whose strategy is a
|
||||||
|
//! script: announce a decoy `(ghost, addr)` where `addr` is a raw
|
||||||
|
//! `TcpListener` the client itself holds (an OS thread accepts and
|
||||||
|
//! immediately closes, so every dial fails at handshake and the connector
|
||||||
|
//! keeps retrying on backoff — the accept count is the dial count); after a
|
||||||
|
//! beat, withdraw the decoy and announce the real server. The client waits
|
||||||
|
//! for the server's `node_up` — which is *after* the withdrawal in the
|
||||||
|
//! strategy's own stream — then watches the decoy's accept count stay flat
|
||||||
|
//! across a window longer than the pending backoff. Before withdrawal it
|
||||||
|
//! must have been climbing (≥ 1), or the negative proves nothing.
|
||||||
|
#![cfg(feature = "cluster")]
|
||||||
|
|
||||||
|
mod common;
|
||||||
|
|
||||||
|
use common::{maybe_child, spawn_node};
|
||||||
|
use smarm::channel::Sender;
|
||||||
|
use smarm::cluster::discovery::{Discovery, Strategy};
|
||||||
|
use smarm::cluster::envelope::NodeMeta;
|
||||||
|
use smarm::cluster::membership::{subscribe, NodeEvent};
|
||||||
|
use smarm::cluster::{start, Config, StaticSeeds, Timing};
|
||||||
|
use std::net::TcpListener;
|
||||||
|
use std::sync::atomic::{AtomicUsize, Ordering};
|
||||||
|
use std::sync::Arc;
|
||||||
|
use std::time::{Duration, Instant};
|
||||||
|
|
||||||
|
const ROLES: &[(&str, fn())] = &[("server", role_server), ("client", role_client)];
|
||||||
|
|
||||||
|
fn meta() -> NodeMeta {
|
||||||
|
NodeMeta {
|
||||||
|
role: "withdraw".into(),
|
||||||
|
region: "local".into(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn role_server() {
|
||||||
|
smarm::run(|| {
|
||||||
|
let cluster = start(Config {
|
||||||
|
node_name: "server".into(),
|
||||||
|
meta: meta(),
|
||||||
|
listen_addr: std::env::var("SMARM_LISTEN_ADDR")
|
||||||
|
.unwrap_or_else(|_| "127.0.0.1:0".into()),
|
||||||
|
strategy: Box::new(StaticSeeds::new(Vec::<(String, String)>::new())),
|
||||||
|
timing: Timing::default(),
|
||||||
|
})
|
||||||
|
.expect("binds");
|
||||||
|
println!("LISTENING {}", cluster.local_addr());
|
||||||
|
loop {
|
||||||
|
smarm::sleep(Duration::from_secs(3600));
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Scripted strategy: decoy, pause, withdraw decoy, real server, done.
|
||||||
|
struct Script {
|
||||||
|
decoy: String,
|
||||||
|
server: String,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Strategy for Script {
|
||||||
|
fn run(self: Box<Self>, out: Sender<Discovery>) {
|
||||||
|
let _ = out.send(Discovery::Candidate {
|
||||||
|
name: "ghost".into(),
|
||||||
|
addr: self.decoy.clone(),
|
||||||
|
});
|
||||||
|
// Long enough for the 250ms/500ms retries to land: ≥ 3 dials.
|
||||||
|
smarm::sleep(Duration::from_millis(1100));
|
||||||
|
let _ = out.send(Discovery::Withdrawn {
|
||||||
|
name: "ghost".into(),
|
||||||
|
addr: self.decoy,
|
||||||
|
});
|
||||||
|
let _ = out.send(Discovery::Candidate {
|
||||||
|
name: "server".into(),
|
||||||
|
addr: self.server,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn role_client() {
|
||||||
|
let server_addr = std::env::var("SMARM_SERVER_ADDR").expect("SMARM_SERVER_ADDR");
|
||||||
|
// The decoy: accept-and-close on an OS thread; count every accept.
|
||||||
|
let decoy = TcpListener::bind("127.0.0.1:0").unwrap();
|
||||||
|
let decoy_addr = decoy.local_addr().unwrap().to_string();
|
||||||
|
let dials = Arc::new(AtomicUsize::new(0));
|
||||||
|
let counter = dials.clone();
|
||||||
|
std::thread::spawn(move || {
|
||||||
|
for conn in decoy.incoming() {
|
||||||
|
counter.fetch_add(1, Ordering::SeqCst);
|
||||||
|
drop(conn);
|
||||||
|
}
|
||||||
|
});
|
||||||
|
|
||||||
|
smarm::run(move || {
|
||||||
|
let _cluster = start(Config {
|
||||||
|
node_name: "client".into(),
|
||||||
|
meta: meta(),
|
||||||
|
listen_addr: "127.0.0.1:0".into(),
|
||||||
|
strategy: Box::new(Script {
|
||||||
|
decoy: decoy_addr,
|
||||||
|
server: server_addr,
|
||||||
|
}),
|
||||||
|
timing: Timing::default(),
|
||||||
|
})
|
||||||
|
.expect("binds");
|
||||||
|
let ev = subscribe().unwrap();
|
||||||
|
loop {
|
||||||
|
match ev.rx.recv() {
|
||||||
|
Ok(NodeEvent::NodeUp(i)) if i.name == "server" => break,
|
||||||
|
Ok(_) => continue,
|
||||||
|
Err(_) => panic!("manager gone"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// The withdrawal preceded the server candidate in the strategy's
|
||||||
|
// stream, so it has been applied. Any dial that started before it
|
||||||
|
// is bounded by the connect+handshake deadlines; let it drain, then
|
||||||
|
// hold the count flat across a window longer than the pending
|
||||||
|
// backoff would be (1s at this point, 2s next).
|
||||||
|
let before = dials.load(Ordering::SeqCst);
|
||||||
|
smarm::sleep(Duration::from_millis(500));
|
||||||
|
let settled = dials.load(Ordering::SeqCst);
|
||||||
|
let t0 = Instant::now();
|
||||||
|
while t0.elapsed() < Duration::from_millis(3000) {
|
||||||
|
smarm::sleep(Duration::from_millis(100));
|
||||||
|
}
|
||||||
|
let after = dials.load(Ordering::SeqCst);
|
||||||
|
println!("WITHDRAWN before={before} settled={settled} after={after}");
|
||||||
|
println!("CLIENT DONE");
|
||||||
|
loop {
|
||||||
|
smarm::sleep(Duration::from_secs(3600));
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn withdrawn_candidate_is_no_longer_dialed() {
|
||||||
|
maybe_child(ROLES);
|
||||||
|
let mut server = spawn_node("server", &[]);
|
||||||
|
let saddr = server.wait_listening();
|
||||||
|
let mut client = spawn_node("client", &[("SMARM_SERVER_ADDR", &saddr)]);
|
||||||
|
let line = client.wait_line("WITHDRAWN", |l| l.starts_with("WITHDRAWN "));
|
||||||
|
let mut nums = line
|
||||||
|
.split_whitespace()
|
||||||
|
.skip(1)
|
||||||
|
.map(|kv| kv.split_once('=').unwrap().1.parse::<usize>().unwrap());
|
||||||
|
let (before, settled, after) = (
|
||||||
|
nums.next().unwrap(),
|
||||||
|
nums.next().unwrap(),
|
||||||
|
nums.next().unwrap(),
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
before >= 1,
|
||||||
|
"decoy was never dialed; the negative proves nothing: {line}"
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
settled, after,
|
||||||
|
"connector kept dialing a withdrawn candidate: {line}"
|
||||||
|
);
|
||||||
|
client.wait_line("CLIENT DONE", |l| l == "CLIENT DONE");
|
||||||
|
}
|
||||||
@@ -0,0 +1,276 @@
|
|||||||
|
//! RFC 010 c2 — owned envelope tests (roadmap: per-frame roundtrip,
|
||||||
|
//! truncation mid-field, unknown tag, length prefix lying long and short,
|
||||||
|
//! zero-length payload, adversarial lengths).
|
||||||
|
#![cfg(feature = "cluster")]
|
||||||
|
|
||||||
|
use serde::{Deserialize, Serialize};
|
||||||
|
use smarm::cluster::envelope::{
|
||||||
|
decode_payload, encode_payload, DecodeError, Frame, NodeMeta, RejectReason, MAX_FRAME_LEN,
|
||||||
|
PROTO_VERSION,
|
||||||
|
};
|
||||||
|
use smarm::cluster::RemoteDownReason;
|
||||||
|
use smarm::monitor::DownReason;
|
||||||
|
use smarm::pg::Incarnation;
|
||||||
|
|
||||||
|
fn meta() -> NodeMeta {
|
||||||
|
NodeMeta {
|
||||||
|
role: "worker".into(),
|
||||||
|
region: "eu-west".into(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn all_frames() -> Vec<Frame> {
|
||||||
|
vec![
|
||||||
|
Frame::Hello {
|
||||||
|
proto_version: PROTO_VERSION,
|
||||||
|
build_hash: 0xDEAD_BEEF_CAFE_F00D,
|
||||||
|
node_name: "alpha".into(),
|
||||||
|
incarnation: Incarnation::new(7),
|
||||||
|
meta: meta(),
|
||||||
|
},
|
||||||
|
Frame::HelloAck {
|
||||||
|
node_name: "beta".into(),
|
||||||
|
incarnation: Incarnation::new(9),
|
||||||
|
meta: meta(),
|
||||||
|
},
|
||||||
|
Frame::HelloReject {
|
||||||
|
reason: RejectReason::NameTaken,
|
||||||
|
},
|
||||||
|
Frame::Heartbeat,
|
||||||
|
Frame::Send {
|
||||||
|
index: 42,
|
||||||
|
generation: 3,
|
||||||
|
type_hash: 0x1234_5678_9ABC_DEF0,
|
||||||
|
payload: vec![1, 2, 3, 4, 5],
|
||||||
|
},
|
||||||
|
Frame::SendNamed {
|
||||||
|
name: "the_counter".into(),
|
||||||
|
type_hash: 0xFFFF_0000_FFFF_0000,
|
||||||
|
payload: vec![],
|
||||||
|
},
|
||||||
|
Frame::Monitor {
|
||||||
|
monitor_id: 77,
|
||||||
|
index: 42,
|
||||||
|
generation: 3,
|
||||||
|
},
|
||||||
|
Frame::Demonitor { monitor_id: 77 },
|
||||||
|
Frame::Down {
|
||||||
|
monitor_id: 77,
|
||||||
|
reason: RemoteDownReason::Local(DownReason::Panic),
|
||||||
|
},
|
||||||
|
Frame::Down {
|
||||||
|
monitor_id: 78,
|
||||||
|
reason: RemoteDownReason::Disconnected,
|
||||||
|
},
|
||||||
|
]
|
||||||
|
}
|
||||||
|
|
||||||
|
fn encode_one(f: &Frame) -> Vec<u8> {
|
||||||
|
let mut buf = Vec::new();
|
||||||
|
f.encode(&mut buf).unwrap();
|
||||||
|
buf
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn per_frame_roundtrip() {
|
||||||
|
for f in all_frames() {
|
||||||
|
let buf = encode_one(&f);
|
||||||
|
let (decoded, consumed) = Frame::decode(&buf).unwrap().unwrap();
|
||||||
|
assert_eq!(decoded, f, "roundtrip mismatch");
|
||||||
|
assert_eq!(consumed, buf.len(), "consumed != buffer length for {f:?}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn back_to_back_frames_decode_sequentially() {
|
||||||
|
let mut buf = Vec::new();
|
||||||
|
for f in all_frames() {
|
||||||
|
f.encode(&mut buf).unwrap();
|
||||||
|
}
|
||||||
|
let mut off = 0;
|
||||||
|
let mut decoded = Vec::new();
|
||||||
|
while off < buf.len() {
|
||||||
|
let (f, n) = Frame::decode(&buf[off..]).unwrap().unwrap();
|
||||||
|
decoded.push(f);
|
||||||
|
off += n;
|
||||||
|
}
|
||||||
|
assert_eq!(decoded, all_frames());
|
||||||
|
assert_eq!(off, buf.len());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn heartbeat_golden_bytes() {
|
||||||
|
// Locks the layout: u32 LE length prefix, then the tag byte.
|
||||||
|
let buf = encode_one(&Frame::Heartbeat);
|
||||||
|
assert_eq!(buf, vec![1, 0, 0, 0, 4]);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn zero_length_payload_roundtrips() {
|
||||||
|
let f = Frame::Send {
|
||||||
|
index: 0,
|
||||||
|
generation: 0,
|
||||||
|
type_hash: 0,
|
||||||
|
payload: vec![],
|
||||||
|
};
|
||||||
|
let buf = encode_one(&f);
|
||||||
|
let (decoded, consumed) = Frame::decode(&buf).unwrap().unwrap();
|
||||||
|
assert_eq!(decoded, f);
|
||||||
|
assert_eq!(consumed, buf.len());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn incomplete_is_none_not_error() {
|
||||||
|
let buf = encode_one(&all_frames()[0]);
|
||||||
|
// Every strict prefix short of the full frame must report "need more".
|
||||||
|
for cut in 0..buf.len() {
|
||||||
|
assert_eq!(
|
||||||
|
Frame::decode(&buf[..cut]).unwrap(),
|
||||||
|
None,
|
||||||
|
"cut at {cut} should be incomplete"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn unknown_frame_tag() {
|
||||||
|
let buf = vec![1, 0, 0, 0, 250];
|
||||||
|
assert_eq!(Frame::decode(&buf), Err(DecodeError::UnknownTag(250)));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn unknown_enum_tags() {
|
||||||
|
// HelloReject with a bogus reason tag.
|
||||||
|
let buf = vec![2, 0, 0, 0, 3, 99];
|
||||||
|
assert_eq!(
|
||||||
|
Frame::decode(&buf),
|
||||||
|
Err(DecodeError::UnknownEnumTag {
|
||||||
|
what: "RejectReason",
|
||||||
|
tag: 99
|
||||||
|
})
|
||||||
|
);
|
||||||
|
// Down with a bogus reason tag (id = 0u64).
|
||||||
|
let mut buf = vec![10, 0, 0, 0, 9];
|
||||||
|
buf.extend_from_slice(&0u64.to_le_bytes());
|
||||||
|
buf.push(200);
|
||||||
|
assert_eq!(
|
||||||
|
Frame::decode(&buf),
|
||||||
|
Err(DecodeError::UnknownEnumTag {
|
||||||
|
what: "RemoteDownReason",
|
||||||
|
tag: 200
|
||||||
|
})
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn length_prefix_lying_long_with_bytes_present_is_trailing() {
|
||||||
|
let mut buf = encode_one(&Frame::Heartbeat);
|
||||||
|
// Declare 3 extra body bytes and actually supply them.
|
||||||
|
let declared = u32::from_le_bytes([buf[0], buf[1], buf[2], buf[3]]) + 3;
|
||||||
|
buf[0..4].copy_from_slice(&declared.to_le_bytes());
|
||||||
|
buf.extend_from_slice(&[0xAA, 0xBB, 0xCC]);
|
||||||
|
assert_eq!(Frame::decode(&buf), Err(DecodeError::Trailing { extra: 3 }));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn length_prefix_lying_long_without_bytes_is_incomplete() {
|
||||||
|
// Indistinguishable from a partial read — must be None, not an error.
|
||||||
|
let mut buf = encode_one(&Frame::Heartbeat);
|
||||||
|
let declared = u32::from_le_bytes([buf[0], buf[1], buf[2], buf[3]]) + 3;
|
||||||
|
buf[0..4].copy_from_slice(&declared.to_le_bytes());
|
||||||
|
assert_eq!(Frame::decode(&buf).unwrap(), None);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn length_prefix_lying_short_truncates_a_field() {
|
||||||
|
let f = &all_frames()[0]; // Hello: plenty of fields to cut into
|
||||||
|
let mut buf = encode_one(f);
|
||||||
|
let declared = u32::from_le_bytes([buf[0], buf[1], buf[2], buf[3]]);
|
||||||
|
let lie = declared - 4; // cut mid-field
|
||||||
|
buf[0..4].copy_from_slice(&lie.to_le_bytes());
|
||||||
|
assert_eq!(Frame::decode(&buf), Err(DecodeError::Truncated));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn truncation_mid_string_field() {
|
||||||
|
// A frame whose declared length is intact but whose inner string length
|
||||||
|
// runs past the body: SendNamed claiming a 1000-byte name in a tiny body.
|
||||||
|
let mut body = vec![6u8]; // TAG_SEND_NAMED
|
||||||
|
body.extend_from_slice(&1000u16.to_le_bytes());
|
||||||
|
body.extend_from_slice(b"short");
|
||||||
|
let mut buf = (body.len() as u32).to_le_bytes().to_vec();
|
||||||
|
buf.extend_from_slice(&body);
|
||||||
|
assert_eq!(Frame::decode(&buf), Err(DecodeError::Truncated));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn adversarial_lengths() {
|
||||||
|
// Length prefix of u32::MAX: reject as oversized, do not wait for 4 GiB.
|
||||||
|
let buf = [0xFF, 0xFF, 0xFF, 0xFF, 0];
|
||||||
|
assert_eq!(
|
||||||
|
Frame::decode(&buf),
|
||||||
|
Err(DecodeError::FrameTooLarge {
|
||||||
|
declared: u32::MAX as usize
|
||||||
|
})
|
||||||
|
);
|
||||||
|
// Just over the cap: also rejected.
|
||||||
|
let over = (MAX_FRAME_LEN as u32 + 1).to_le_bytes();
|
||||||
|
assert!(matches!(
|
||||||
|
Frame::decode(&over),
|
||||||
|
Err(DecodeError::FrameTooLarge { .. })
|
||||||
|
));
|
||||||
|
// Zero-length frame: there is no tag byte; corrupt, not incomplete.
|
||||||
|
let buf = [0, 0, 0, 0];
|
||||||
|
assert_eq!(Frame::decode(&buf), Err(DecodeError::EmptyFrame));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn invalid_utf8_in_string_field() {
|
||||||
|
let mut buf = encode_one(&Frame::SendNamed {
|
||||||
|
name: "abcd".into(),
|
||||||
|
type_hash: 0,
|
||||||
|
payload: vec![],
|
||||||
|
});
|
||||||
|
// name bytes start after: 4 (len) + 1 (tag) + 2 (str len) = offset 7
|
||||||
|
buf[7] = 0xFF;
|
||||||
|
assert_eq!(Frame::decode(&buf), Err(DecodeError::Utf8));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, PartialEq, Serialize, Deserialize)]
|
||||||
|
struct Ping {
|
||||||
|
seq: u64,
|
||||||
|
label: String,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn payload_seam_roundtrip() {
|
||||||
|
let ping = Ping {
|
||||||
|
seq: 31337,
|
||||||
|
label: "hello".into(),
|
||||||
|
};
|
||||||
|
let blob = encode_payload(&ping).unwrap();
|
||||||
|
// Carry it through a real frame, as it will travel in c9.
|
||||||
|
let f = Frame::Send {
|
||||||
|
index: 1,
|
||||||
|
generation: 1,
|
||||||
|
type_hash: 0xABCD,
|
||||||
|
payload: blob,
|
||||||
|
};
|
||||||
|
let buf = encode_one(&f);
|
||||||
|
let (decoded, _) = Frame::decode(&buf).unwrap().unwrap();
|
||||||
|
let Frame::Send { payload, .. } = decoded else {
|
||||||
|
panic!("wrong frame");
|
||||||
|
};
|
||||||
|
let back: Ping = decode_payload(&payload).unwrap();
|
||||||
|
assert_eq!(back, ping);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn payload_seam_rejects_truncated_blob() {
|
||||||
|
let blob = encode_payload(&Ping {
|
||||||
|
seq: 1,
|
||||||
|
label: "x".into(),
|
||||||
|
})
|
||||||
|
.unwrap();
|
||||||
|
assert!(decode_payload::<Ping>(&blob[..blob.len() - 1]).is_err());
|
||||||
|
}
|
||||||
@@ -0,0 +1,161 @@
|
|||||||
|
//! RFC 010 c8 — exposure registry + type hashing. Purely local, no network.
|
||||||
|
//!
|
||||||
|
//! Payload types are std types (`String`, `u64`) because the crate's serde is
|
||||||
|
//! deliberately derive-less (`default-features = false`) — user crates bring
|
||||||
|
//! their own derive; the contract here is `DeserializeOwned`.
|
||||||
|
//!
|
||||||
|
//! The hash-stability test re-execs the current binary (the c4 harness): the
|
||||||
|
//! guarantee under test is "stable across runs in the SAME binary" — exactly
|
||||||
|
//! what the build-hash handshake reduces the mesh to — not stability across
|
||||||
|
//! builds, which the scope guard explicitly rejects.
|
||||||
|
#![cfg(feature = "cluster")]
|
||||||
|
|
||||||
|
mod common;
|
||||||
|
|
||||||
|
use common::{maybe_child, spawn_node};
|
||||||
|
use smarm::cluster::envelope::encode_payload;
|
||||||
|
use smarm::cluster::expose::{
|
||||||
|
decode_deliver, decoder_registered, expose, expose_type, exposed_hash, exposed_names,
|
||||||
|
type_hash, DeliverError,
|
||||||
|
};
|
||||||
|
use smarm::monitor::{monitor, terminal_reason, DownReason};
|
||||||
|
use smarm::{channel, register, run, spawn, Name};
|
||||||
|
|
||||||
|
const ROLES: &[(&str, fn())] = &[("hasher", role_hasher)];
|
||||||
|
|
||||||
|
/// Print the hashes this process computes; the parent (a different run of
|
||||||
|
/// the same binary) compares against its own.
|
||||||
|
fn role_hasher() {
|
||||||
|
println!("HASH-STRING {}", type_hash::<String>());
|
||||||
|
println!("HASH-U64 {}", type_hash::<u64>());
|
||||||
|
}
|
||||||
|
|
||||||
|
const GREETER: Name<String> = Name::new("expose-test.greeter");
|
||||||
|
|
||||||
|
/// Exposed and unexposed lookup, the returned hash, and the audit listing.
|
||||||
|
#[test]
|
||||||
|
fn exposed_and_unexposed_lookup() {
|
||||||
|
maybe_child(ROLES);
|
||||||
|
run(|| {
|
||||||
|
let h = expose(GREETER);
|
||||||
|
assert_eq!(h, type_hash::<String>());
|
||||||
|
assert_eq!(exposed_hash("expose-test.greeter"), Some(h));
|
||||||
|
assert_eq!(exposed_hash("never-exposed"), None);
|
||||||
|
assert!(exposed_names().contains(&("expose-test.greeter", h)));
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Distinct types land on distinct hashes (FNV over distinct TypeIds — a
|
||||||
|
/// smoke assertion; a collision would degrade to a decode error, never a
|
||||||
|
/// misroute, per RFC §3).
|
||||||
|
#[test]
|
||||||
|
fn distinct_types_distinct_hashes() {
|
||||||
|
maybe_child(ROLES);
|
||||||
|
run(|| {
|
||||||
|
assert_ne!(type_hash::<String>(), type_hash::<u64>());
|
||||||
|
assert_ne!(type_hash::<String>(), type_hash::<Vec<u8>>());
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The decode-and-deliver contract: a registered hash decodes into the
|
||||||
|
/// target's typed channel; an unknown hash, corrupt bytes, and a missing
|
||||||
|
/// channel each fail without delivering — `WrongChannel`, never a misroute.
|
||||||
|
#[test]
|
||||||
|
fn decoder_registration_and_delivery() {
|
||||||
|
maybe_child(ROLES);
|
||||||
|
run(|| {
|
||||||
|
let h_string = expose_type::<String>();
|
||||||
|
let h_u64 = expose_type::<u64>();
|
||||||
|
assert!(decoder_registered(h_string));
|
||||||
|
assert!(!decoder_registered(h_string.wrapping_add(1)));
|
||||||
|
|
||||||
|
// A live actor with a String channel (registered from its own body,
|
||||||
|
// announced via a ready signal — the tests/registry.rs idiom).
|
||||||
|
let (ready_tx, ready_rx) = channel::<()>();
|
||||||
|
let (stop_tx, stop_rx) = channel::<()>();
|
||||||
|
let (msg_tx, msg_rx) = channel::<String>();
|
||||||
|
let pid = spawn(move || {
|
||||||
|
register(Name::<String>::new("expose-test.sink"), msg_tx).unwrap();
|
||||||
|
ready_tx.send(()).unwrap();
|
||||||
|
let _ = stop_rx.recv();
|
||||||
|
})
|
||||||
|
.pid();
|
||||||
|
ready_rx.recv().unwrap();
|
||||||
|
|
||||||
|
// Happy path: decode + deliver through the published channel.
|
||||||
|
let bytes = encode_payload("hello across the seam").unwrap();
|
||||||
|
decode_deliver(h_string, pid, &bytes).unwrap();
|
||||||
|
assert_eq!(msg_rx.recv().unwrap(), "hello across the seam");
|
||||||
|
|
||||||
|
// Unknown hash: nothing was registered under it.
|
||||||
|
assert!(matches!(
|
||||||
|
decode_deliver(h_string.wrapping_add(1), pid, &bytes),
|
||||||
|
Err(DeliverError::UnknownType)
|
||||||
|
));
|
||||||
|
|
||||||
|
// Corrupt bytes: the decoder fails before any send.
|
||||||
|
assert!(matches!(
|
||||||
|
decode_deliver(h_string, pid, &[0xff; 3]),
|
||||||
|
Err(DeliverError::Decode(_))
|
||||||
|
));
|
||||||
|
|
||||||
|
// Right decoder, wrong channel: the actor has no u64 channel, so the
|
||||||
|
// decoded value is refused — the NoChannel guarantee.
|
||||||
|
let u64_bytes = encode_payload(&7u64).unwrap();
|
||||||
|
assert!(matches!(
|
||||||
|
decode_deliver(h_u64, pid, &u64_bytes),
|
||||||
|
Err(DeliverError::WrongChannel)
|
||||||
|
));
|
||||||
|
|
||||||
|
stop_tx.send(()).unwrap();
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `expose` and the bridge crossing agree on the resulting set: both funnel
|
||||||
|
/// the pid-boundary mark through the watchable machinery, so an exposed
|
||||||
|
/// name's holder dies with a terminal record — the exact observable
|
||||||
|
/// `mark_watchable` guarantees the membrane. (For named holders the mark is
|
||||||
|
/// already stamped by `register` itself; this pins the shared contract.)
|
||||||
|
#[test]
|
||||||
|
fn expose_and_bridge_crossing_agree_on_the_set() {
|
||||||
|
maybe_child(ROLES);
|
||||||
|
run(|| {
|
||||||
|
let (ready_tx, ready_rx) = channel::<()>();
|
||||||
|
let (stop_tx, stop_rx) = channel::<()>();
|
||||||
|
let (msg_tx, _msg_rx) = channel::<String>();
|
||||||
|
let pid = spawn(move || {
|
||||||
|
register(GREETER, msg_tx).unwrap();
|
||||||
|
ready_tx.send(()).unwrap();
|
||||||
|
let _ = stop_rx.recv();
|
||||||
|
})
|
||||||
|
.pid();
|
||||||
|
ready_rx.recv().unwrap();
|
||||||
|
|
||||||
|
expose(GREETER);
|
||||||
|
let m = monitor(pid);
|
||||||
|
stop_tx.send(()).unwrap();
|
||||||
|
assert_eq!(m.rx.recv().unwrap().reason, DownReason::Exit);
|
||||||
|
assert_eq!(terminal_reason(pid), Some(DownReason::Exit));
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Hash stability across runs in the same binary: a re-exec of this binary
|
||||||
|
/// computes the same hashes this process does.
|
||||||
|
#[test]
|
||||||
|
fn hash_stable_across_runs_in_same_binary() {
|
||||||
|
maybe_child(ROLES);
|
||||||
|
let (mine_string, mine_u64) = {
|
||||||
|
// Computing a TypeId hash needs no runtime, but keep the contract
|
||||||
|
// uniform with real call sites.
|
||||||
|
(type_hash::<String>(), type_hash::<u64>())
|
||||||
|
};
|
||||||
|
let mut child = spawn_node("hasher", &[]);
|
||||||
|
let line = child.wait_line("HASH-STRING", |l| l.starts_with("HASH-STRING "));
|
||||||
|
assert_eq!(
|
||||||
|
line["HASH-STRING ".len()..].parse::<u64>().unwrap(),
|
||||||
|
mine_string
|
||||||
|
);
|
||||||
|
let line = child.wait_line("HASH-U64", |l| l.starts_with("HASH-U64 "));
|
||||||
|
assert_eq!(line["HASH-U64 ".len()..].parse::<u64>().unwrap(), mine_u64);
|
||||||
|
child.wait_exit();
|
||||||
|
}
|
||||||
@@ -0,0 +1,240 @@
|
|||||||
|
//! RFC 010 c5 — handshake state-machine tests (roadmap: happy path; hash
|
||||||
|
//! mismatch; proto-version mismatch; name already claimed; simultaneous-connect
|
||||||
|
//! tie-break; garbage before Hello). Pure — no IO, no actors, no runtime.
|
||||||
|
#![cfg(feature = "cluster")]
|
||||||
|
|
||||||
|
use smarm::cluster::envelope::{Frame, NodeMeta, RejectReason, PROTO_VERSION};
|
||||||
|
use smarm::cluster::handshake::{
|
||||||
|
dial_wins, Initiator, InitiatorOutcome, Local, PeerStanding, Responder, ResponderOutcome,
|
||||||
|
};
|
||||||
|
use smarm::pg::Incarnation;
|
||||||
|
|
||||||
|
const HASH: u64 = 0xDEAD_BEEF_CAFE_F00D;
|
||||||
|
|
||||||
|
fn local(name: &str) -> Local {
|
||||||
|
Local {
|
||||||
|
node_name: name.into(),
|
||||||
|
incarnation: Incarnation::new(7),
|
||||||
|
build_hash: HASH,
|
||||||
|
meta: NodeMeta {
|
||||||
|
role: "worker".into(),
|
||||||
|
region: "eu-west".into(),
|
||||||
|
},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The Hello that `Initiator::new(&local(name))` emits, built by hand.
|
||||||
|
fn hello_from(name: &str) -> Frame {
|
||||||
|
let l = local(name);
|
||||||
|
Frame::Hello {
|
||||||
|
proto_version: PROTO_VERSION,
|
||||||
|
build_hash: l.build_hash,
|
||||||
|
node_name: l.node_name,
|
||||||
|
incarnation: l.incarnation,
|
||||||
|
meta: l.meta,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn happy_path_establishes_both_ends() {
|
||||||
|
// alpha dials beta.
|
||||||
|
let (initiator, hello) = Initiator::new(&local("alpha"));
|
||||||
|
assert_eq!(hello, hello_from("alpha"), "initiator emits its identity");
|
||||||
|
|
||||||
|
let responder = Responder::new(local("beta"));
|
||||||
|
let (reply, peer) = match responder.on_frame(hello, PeerStanding::Free) {
|
||||||
|
ResponderOutcome::Accepted { reply, peer } => (reply, peer),
|
||||||
|
other => panic!("expected Accepted, got {other:?}"),
|
||||||
|
};
|
||||||
|
assert_eq!(peer.node_name, "alpha");
|
||||||
|
assert_eq!(peer.incarnation, Incarnation::new(7));
|
||||||
|
assert_eq!(peer.meta.role, "worker");
|
||||||
|
|
||||||
|
// The ack carries the responder's identity, no hash/version (one-sided
|
||||||
|
// check — sound because equality is symmetric).
|
||||||
|
let l = local("beta");
|
||||||
|
assert_eq!(
|
||||||
|
reply,
|
||||||
|
Frame::HelloAck {
|
||||||
|
node_name: l.node_name,
|
||||||
|
incarnation: l.incarnation,
|
||||||
|
meta: l.meta,
|
||||||
|
}
|
||||||
|
);
|
||||||
|
|
||||||
|
match initiator.on_frame(reply) {
|
||||||
|
InitiatorOutcome::Established(peer) => {
|
||||||
|
assert_eq!(peer.node_name, "beta");
|
||||||
|
assert_eq!(peer.incarnation, Incarnation::new(7));
|
||||||
|
assert_eq!(peer.meta.region, "eu-west");
|
||||||
|
}
|
||||||
|
other => panic!("expected Established, got {other:?}"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn hash_mismatch_rejected() {
|
||||||
|
let responder = Responder::new(local("beta"));
|
||||||
|
let hello = Frame::Hello {
|
||||||
|
proto_version: PROTO_VERSION,
|
||||||
|
build_hash: HASH ^ 1,
|
||||||
|
node_name: "alpha".into(),
|
||||||
|
incarnation: Incarnation::new(7),
|
||||||
|
meta: local("alpha").meta,
|
||||||
|
};
|
||||||
|
match responder.on_frame(hello, PeerStanding::Free) {
|
||||||
|
ResponderOutcome::Rejected { reply, reason } => {
|
||||||
|
assert_eq!(reason, RejectReason::HashMismatch);
|
||||||
|
assert_eq!(reply, Frame::HelloReject { reason });
|
||||||
|
}
|
||||||
|
other => panic!("expected Rejected, got {other:?}"),
|
||||||
|
}
|
||||||
|
|
||||||
|
// The dialer side of the same story: a reject frame comes back.
|
||||||
|
let (initiator, _hello) = Initiator::new(&local("alpha"));
|
||||||
|
match initiator.on_frame(Frame::HelloReject {
|
||||||
|
reason: RejectReason::HashMismatch,
|
||||||
|
}) {
|
||||||
|
InitiatorOutcome::Rejected(RejectReason::HashMismatch) => {}
|
||||||
|
other => panic!("expected Rejected(HashMismatch), got {other:?}"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn proto_version_mismatch_rejected_and_checked_first() {
|
||||||
|
// Both proto and hash wrong: proto wins — nothing after the version can
|
||||||
|
// be trusted, and HelloReject is the cross-version compatibility anchor.
|
||||||
|
let responder = Responder::new(local("beta"));
|
||||||
|
let hello = Frame::Hello {
|
||||||
|
proto_version: PROTO_VERSION + 1,
|
||||||
|
build_hash: HASH ^ 1,
|
||||||
|
node_name: "alpha".into(),
|
||||||
|
incarnation: Incarnation::new(7),
|
||||||
|
meta: local("alpha").meta,
|
||||||
|
};
|
||||||
|
match responder.on_frame(hello, PeerStanding::Free) {
|
||||||
|
ResponderOutcome::Rejected { reason, .. } => {
|
||||||
|
assert_eq!(reason, RejectReason::ProtoVersion);
|
||||||
|
}
|
||||||
|
other => panic!("expected Rejected, got {other:?}"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn claimed_name_rejected() {
|
||||||
|
let responder = Responder::new(local("beta"));
|
||||||
|
let ctx = PeerStanding::Claimed;
|
||||||
|
match responder.on_frame(hello_from("alpha"), ctx) {
|
||||||
|
ResponderOutcome::Rejected { reason, .. } => {
|
||||||
|
assert_eq!(reason, RejectReason::NameTaken);
|
||||||
|
}
|
||||||
|
other => panic!("expected Rejected, got {other:?}"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn own_name_offered_rejected_as_name_taken() {
|
||||||
|
// Self-connect or genuine collision: the responder's own name arrives.
|
||||||
|
let responder = Responder::new(local("beta"));
|
||||||
|
match responder.on_frame(hello_from("beta"), PeerStanding::Free) {
|
||||||
|
ResponderOutcome::Rejected { reason, .. } => {
|
||||||
|
assert_eq!(reason, RejectReason::NameTaken);
|
||||||
|
}
|
||||||
|
other => panic!("expected Rejected, got {other:?}"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn hash_checked_before_name() {
|
||||||
|
// Wrong hash AND claimed name: hash wins (validity before identity).
|
||||||
|
let responder = Responder::new(local("beta"));
|
||||||
|
let hello = Frame::Hello {
|
||||||
|
proto_version: PROTO_VERSION,
|
||||||
|
build_hash: HASH ^ 1,
|
||||||
|
node_name: "alpha".into(),
|
||||||
|
incarnation: Incarnation::new(7),
|
||||||
|
meta: local("alpha").meta,
|
||||||
|
};
|
||||||
|
let ctx = PeerStanding::Claimed;
|
||||||
|
match responder.on_frame(hello, ctx) {
|
||||||
|
ResponderOutcome::Rejected { reason, .. } => {
|
||||||
|
assert_eq!(reason, RejectReason::HashMismatch);
|
||||||
|
}
|
||||||
|
other => panic!("expected Rejected, got {other:?}"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn dial_wins_is_deterministic_and_antisymmetric() {
|
||||||
|
// The smaller name's dial survives; both ends compute the same verdict.
|
||||||
|
assert!(dial_wins("alpha", "beta"));
|
||||||
|
assert!(!dial_wins("beta", "alpha"));
|
||||||
|
for (a, b) in [("a", "b"), ("node-1", "node-2"), ("x", "xx")] {
|
||||||
|
assert_ne!(dial_wins(a, b), dial_wins(b, a), "({a}, {b})");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn simultaneous_connect_exactly_one_side_accepts() {
|
||||||
|
// alpha and beta dial each other at once. Each responder sees the peer's
|
||||||
|
// Hello while its own dial is in flight.
|
||||||
|
let ctx = PeerStanding::Dialing;
|
||||||
|
|
||||||
|
// On beta: inbound is alpha's dial; alpha < beta, so the inbound wins.
|
||||||
|
let on_beta = Responder::new(local("beta")).on_frame(hello_from("alpha"), ctx);
|
||||||
|
assert!(
|
||||||
|
matches!(on_beta, ResponderOutcome::Accepted { .. }),
|
||||||
|
"beta must accept alpha's dial, got {on_beta:?}"
|
||||||
|
);
|
||||||
|
|
||||||
|
// On alpha: inbound is beta's dial; it loses — close silently, no frame
|
||||||
|
// (ratified: both ends can compute the outcome, a reject adds nothing).
|
||||||
|
let on_alpha = Responder::new(local("alpha")).on_frame(hello_from("beta"), ctx);
|
||||||
|
assert!(
|
||||||
|
matches!(on_alpha, ResponderOutcome::TieBreakLoss),
|
||||||
|
"alpha must silently drop beta's dial, got {on_alpha:?}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn tiebreak_loss_only_applies_when_dialing() {
|
||||||
|
// Same inbound Hello, no dial in flight: plain accept.
|
||||||
|
let on_alpha = Responder::new(local("alpha")).on_frame(hello_from("beta"), PeerStanding::Free);
|
||||||
|
assert!(matches!(on_alpha, ResponderOutcome::Accepted { .. }));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn garbage_before_hello_fails_without_reply() {
|
||||||
|
// Any valid-but-wrong frame before Hello is a protocol violation: close,
|
||||||
|
// no reject frame. (Undecodable bytes are the codec's Err, not ours.)
|
||||||
|
for frame in [
|
||||||
|
Frame::Heartbeat,
|
||||||
|
Frame::HelloAck {
|
||||||
|
node_name: "alpha".into(),
|
||||||
|
incarnation: Incarnation::new(7),
|
||||||
|
meta: local("alpha").meta,
|
||||||
|
},
|
||||||
|
Frame::Demonitor { monitor_id: 3 },
|
||||||
|
] {
|
||||||
|
let out = Responder::new(local("beta")).on_frame(frame.clone(), PeerStanding::Free);
|
||||||
|
match out {
|
||||||
|
ResponderOutcome::Failed(f) => assert_eq!(f, frame),
|
||||||
|
other => panic!("expected Failed({frame:?}), got {other:?}"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn garbage_before_ack_fails_the_initiator() {
|
||||||
|
for frame in [
|
||||||
|
Frame::Heartbeat,
|
||||||
|
hello_from("beta"),
|
||||||
|
Frame::Demonitor { monitor_id: 3 },
|
||||||
|
] {
|
||||||
|
let (initiator, _hello) = Initiator::new(&local("alpha"));
|
||||||
|
match initiator.on_frame(frame.clone()) {
|
||||||
|
InitiatorOutcome::Failed(f) => assert_eq!(f, frame),
|
||||||
|
other => panic!("expected Failed({frame:?}), got {other:?}"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,270 @@
|
|||||||
|
//! RFC 010 c7a — membership events and the view, at the manager.
|
||||||
|
//!
|
||||||
|
//! Same construction as the c6a lifecycle suite: the handshake is bypassed,
|
||||||
|
//! connections are built already-established over localhost TCP pairs with
|
||||||
|
//! fabricated `Peer`s, and the manager is started plainly so the test can
|
||||||
|
//! terminate. What is under test is the membership layer that c7 adds to the
|
||||||
|
//! manager: `node_up`/`node_down` events to subscribers (snapshot-then-stream),
|
||||||
|
//! the view, and NodeId identity — memoized per `(name, incarnation)`, so a
|
||||||
|
//! reconnect blip keeps its id and a restart (new incarnation) gets a fresh one.
|
||||||
|
#![cfg(feature = "cluster")]
|
||||||
|
|
||||||
|
use std::time::Duration;
|
||||||
|
|
||||||
|
use smarm::cluster::envelope::NodeMeta;
|
||||||
|
use smarm::cluster::handshake::Peer;
|
||||||
|
use smarm::cluster::manager::{Call, Manager, Reply, MANAGER};
|
||||||
|
use smarm::cluster::membership::{subscribe, view, MembershipEvents, NodeEvent};
|
||||||
|
use smarm::cluster::spawn_established;
|
||||||
|
use smarm::cluster::transport::tcp::TcpTransport;
|
||||||
|
use smarm::cluster::transport::{Conn, FramedConn, Transport};
|
||||||
|
use smarm::cluster::Timing;
|
||||||
|
use smarm::gen_server::{self, GenServerBuilder};
|
||||||
|
use smarm::pg::{Incarnation, NodeId};
|
||||||
|
use smarm::run;
|
||||||
|
|
||||||
|
/// A fabricated post-handshake peer identity, with the incarnation under the
|
||||||
|
/// test's control (it is identity-bearing here, unlike in the c6a suite).
|
||||||
|
fn peer(name: &str, inc: u32) -> Peer {
|
||||||
|
Peer {
|
||||||
|
node_name: name.to_string(),
|
||||||
|
incarnation: Incarnation::new(inc),
|
||||||
|
meta: NodeMeta {
|
||||||
|
role: "test".to_string(),
|
||||||
|
region: "test".to_string(),
|
||||||
|
},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One established transport pair over localhost (TCP backlog covers the
|
||||||
|
/// sequential dial-then-accept, as in the c3 conformance suite).
|
||||||
|
fn pair(t: &dyn Transport) -> (Box<dyn Conn>, Box<dyn Conn>) {
|
||||||
|
let mut l = t.listen("127.0.0.1:0").unwrap();
|
||||||
|
let a = t.dial(&l.local_addr()).unwrap();
|
||||||
|
let b = l.accept().unwrap();
|
||||||
|
(a, b)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The next event, or a panic naming the wait. The bound is generous against
|
||||||
|
/// a sub-millisecond real cost.
|
||||||
|
fn next_event(ev: &MembershipEvents, waiting_for: &str) -> NodeEvent {
|
||||||
|
ev.rx
|
||||||
|
.recv_timeout(Duration::from_secs(5))
|
||||||
|
.unwrap_or_else(|e| panic!("timed out waiting for {waiting_for}: {e:?}"))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Assert the subscription is drained: no event is pending.
|
||||||
|
fn assert_quiet(ev: &MembershipEvents) {
|
||||||
|
assert!(matches!(ev.rx.try_recv(), Ok(None)));
|
||||||
|
}
|
||||||
|
|
||||||
|
fn disconnect(name: &str) {
|
||||||
|
assert!(matches!(
|
||||||
|
gen_server::call(
|
||||||
|
MANAGER,
|
||||||
|
Call::Disconnect {
|
||||||
|
name: name.to_string()
|
||||||
|
}
|
||||||
|
),
|
||||||
|
Ok(Reply::Disconnected)
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Live subscription: an empty snapshot, then `NodeUp` on registration and
|
||||||
|
/// `NodeDown` (same id) on commanded disconnect and on peer EOF alike.
|
||||||
|
#[test]
|
||||||
|
fn subscriber_sees_up_and_down() {
|
||||||
|
run(|| {
|
||||||
|
let mgr = GenServerBuilder::new(Manager::new())
|
||||||
|
.named(MANAGER)
|
||||||
|
.start()
|
||||||
|
.expect("manager name is free");
|
||||||
|
|
||||||
|
let ev = subscribe().expect("manager is up");
|
||||||
|
assert_quiet(&ev); // nothing live: the snapshot is empty
|
||||||
|
|
||||||
|
let t = TcpTransport;
|
||||||
|
let (a1, b1) = pair(&t);
|
||||||
|
let (a2, b2) = pair(&t);
|
||||||
|
spawn_established(FramedConn::new(a1), peer("node-b", 1), Timing::default())
|
||||||
|
.expect("node-b registers");
|
||||||
|
spawn_established(FramedConn::new(a2), peer("node-c", 1), Timing::default())
|
||||||
|
.expect("node-c registers");
|
||||||
|
|
||||||
|
let up_b = match next_event(&ev, "node_up(node-b)") {
|
||||||
|
NodeEvent::NodeUp(info) => {
|
||||||
|
assert_eq!(info.name, "node-b");
|
||||||
|
assert_eq!(info.incarnation, Incarnation::new(1));
|
||||||
|
assert_eq!(info.meta.role, "test");
|
||||||
|
info
|
||||||
|
}
|
||||||
|
other => panic!("expected node_up(node-b), got {other:?}"),
|
||||||
|
};
|
||||||
|
let up_c = match next_event(&ev, "node_up(node-c)") {
|
||||||
|
NodeEvent::NodeUp(info) => {
|
||||||
|
assert_eq!(info.name, "node-c");
|
||||||
|
info
|
||||||
|
}
|
||||||
|
other => panic!("expected node_up(node-c), got {other:?}"),
|
||||||
|
};
|
||||||
|
assert_ne!(up_b.node, up_c.node, "distinct peers get distinct ids");
|
||||||
|
|
||||||
|
// Commanded disconnect: down with node-b's id.
|
||||||
|
disconnect("node-b");
|
||||||
|
assert_eq!(
|
||||||
|
next_event(&ev, "node_down(node-b)"),
|
||||||
|
NodeEvent::NodeDown(up_b.clone())
|
||||||
|
);
|
||||||
|
|
||||||
|
// Peer EOF, no command: down with node-c's id.
|
||||||
|
drop(b2);
|
||||||
|
assert_eq!(
|
||||||
|
next_event(&ev, "node_down(node-c)"),
|
||||||
|
NodeEvent::NodeDown(up_c.clone())
|
||||||
|
);
|
||||||
|
assert_quiet(&ev);
|
||||||
|
|
||||||
|
drop(b1);
|
||||||
|
mgr.shutdown();
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Snapshot-then-stream: a subscriber arriving after connections established
|
||||||
|
/// receives one `NodeUp` per live peer before anything else, and the view
|
||||||
|
/// call agrees with it.
|
||||||
|
#[test]
|
||||||
|
fn late_subscriber_gets_snapshot_and_view_agrees() {
|
||||||
|
run(|| {
|
||||||
|
let mgr = GenServerBuilder::new(Manager::new())
|
||||||
|
.named(MANAGER)
|
||||||
|
.start()
|
||||||
|
.expect("manager name is free");
|
||||||
|
|
||||||
|
let t = TcpTransport;
|
||||||
|
let (a1, b1) = pair(&t);
|
||||||
|
let (a2, b2) = pair(&t);
|
||||||
|
spawn_established(FramedConn::new(a1), peer("node-b", 1), Timing::default())
|
||||||
|
.expect("node-b registers");
|
||||||
|
spawn_established(FramedConn::new(a2), peer("node-c", 1), Timing::default())
|
||||||
|
.expect("node-c registers");
|
||||||
|
|
||||||
|
let ev = subscribe().expect("manager is up");
|
||||||
|
let mut names = Vec::new();
|
||||||
|
for _ in 0..2 {
|
||||||
|
match next_event(&ev, "a snapshot node_up") {
|
||||||
|
NodeEvent::NodeUp(info) => names.push(info.name),
|
||||||
|
other => panic!("expected a snapshot node_up, got {other:?}"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
names.sort();
|
||||||
|
assert_eq!(names, ["node-b", "node-c"]);
|
||||||
|
assert_quiet(&ev); // the snapshot is exactly the live set
|
||||||
|
|
||||||
|
let mut v = view().expect("manager is up");
|
||||||
|
v.sort_by(|a, b| a.name.cmp(&b.name));
|
||||||
|
assert_eq!(v.len(), 2);
|
||||||
|
assert_eq!(v[0].name, "node-b");
|
||||||
|
assert_eq!(v[1].name, "node-c");
|
||||||
|
|
||||||
|
disconnect("node-b");
|
||||||
|
disconnect("node-c");
|
||||||
|
drop((b1, b2));
|
||||||
|
// Drain the two downs so the subscription ends quiet.
|
||||||
|
let _ = next_event(&ev, "node_down");
|
||||||
|
let _ = next_event(&ev, "node_down");
|
||||||
|
mgr.shutdown();
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// NodeId identity: a restart (same name, new incarnation) is a NEW id — the
|
||||||
|
/// ghost and its successor are distinguishable — while a reconnect blip (same
|
||||||
|
/// name, same incarnation) keeps its id.
|
||||||
|
#[test]
|
||||||
|
fn restart_gets_new_id_blip_keeps_id() {
|
||||||
|
run(|| {
|
||||||
|
let mgr = GenServerBuilder::new(Manager::new())
|
||||||
|
.named(MANAGER)
|
||||||
|
.start()
|
||||||
|
.expect("manager name is free");
|
||||||
|
let ev = subscribe().expect("manager is up");
|
||||||
|
let t = TcpTransport;
|
||||||
|
|
||||||
|
let id = |e: NodeEvent, what: &str| -> NodeId {
|
||||||
|
match e {
|
||||||
|
NodeEvent::NodeUp(info) => info.node,
|
||||||
|
other => panic!("expected node_up ({what}), got {other:?}"),
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let down_id = |e: NodeEvent, what: &str| -> NodeId {
|
||||||
|
match e {
|
||||||
|
NodeEvent::NodeDown(info) => info.node,
|
||||||
|
other => panic!("expected node_down ({what}), got {other:?}"),
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
// Up at incarnation 1, then the peer dies (EOF).
|
||||||
|
let (a1, b1) = pair(&t);
|
||||||
|
spawn_established(FramedConn::new(a1), peer("node-b", 1), Timing::default())
|
||||||
|
.expect("registers");
|
||||||
|
let id1 = id(next_event(&ev, "node_up inc 1"), "inc 1");
|
||||||
|
drop(b1);
|
||||||
|
assert_eq!(down_id(next_event(&ev, "node_down inc 1"), "inc 1"), id1);
|
||||||
|
|
||||||
|
// Restart: new incarnation, new id — the ghost's id is not reused.
|
||||||
|
let (a2, b2) = pair(&t);
|
||||||
|
spawn_established(FramedConn::new(a2), peer("node-b", 2), Timing::default())
|
||||||
|
.expect("registers");
|
||||||
|
let id2 = id(next_event(&ev, "node_up inc 2"), "inc 2");
|
||||||
|
assert_ne!(
|
||||||
|
id1, id2,
|
||||||
|
"a restarted node must be distinguishable from its ghost"
|
||||||
|
);
|
||||||
|
|
||||||
|
// Blip: the same incarnation reconnects and keeps its id.
|
||||||
|
disconnect("node-b");
|
||||||
|
assert_eq!(down_id(next_event(&ev, "node_down inc 2"), "inc 2"), id2);
|
||||||
|
let (a3, b3) = pair(&t);
|
||||||
|
spawn_established(FramedConn::new(a3), peer("node-b", 2), Timing::default())
|
||||||
|
.expect("registers");
|
||||||
|
let id3 = id(next_event(&ev, "node_up after blip"), "blip");
|
||||||
|
assert_eq!(
|
||||||
|
id2, id3,
|
||||||
|
"a reconnect at the same incarnation is the same node"
|
||||||
|
);
|
||||||
|
|
||||||
|
disconnect("node-b");
|
||||||
|
let _ = next_event(&ev, "final node_down");
|
||||||
|
drop((b2, b3));
|
||||||
|
mgr.shutdown();
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A dropped subscriber is pruned on the next emit and never disturbs the
|
||||||
|
/// manager or a live subscriber.
|
||||||
|
#[test]
|
||||||
|
fn dead_subscriber_is_pruned() {
|
||||||
|
run(|| {
|
||||||
|
let mgr = GenServerBuilder::new(Manager::new())
|
||||||
|
.named(MANAGER)
|
||||||
|
.start()
|
||||||
|
.expect("manager name is free");
|
||||||
|
|
||||||
|
let dead = subscribe().expect("manager is up");
|
||||||
|
drop(dead);
|
||||||
|
let live = subscribe().expect("manager is up");
|
||||||
|
|
||||||
|
let t = TcpTransport;
|
||||||
|
let (a1, b1) = pair(&t);
|
||||||
|
spawn_established(FramedConn::new(a1), peer("node-b", 1), Timing::default())
|
||||||
|
.expect("registers");
|
||||||
|
match next_event(&live, "node_up despite a dead co-subscriber") {
|
||||||
|
NodeEvent::NodeUp(info) => assert_eq!(info.name, "node-b"),
|
||||||
|
other => panic!("expected node_up, got {other:?}"),
|
||||||
|
}
|
||||||
|
|
||||||
|
disconnect("node-b");
|
||||||
|
let _ = next_event(&live, "node_down");
|
||||||
|
drop(b1);
|
||||||
|
mgr.shutdown();
|
||||||
|
});
|
||||||
|
}
|
||||||
@@ -0,0 +1,175 @@
|
|||||||
|
//! RFC 010 c7 — the Phase 2 gate: a 3-node mesh under the subprocess
|
||||||
|
//! harness, repeatable.
|
||||||
|
//!
|
||||||
|
//! Each node process runs the integrated `cluster::start` (manager +
|
||||||
|
//! acceptor + connector + static seeds), subscribes to membership like any
|
||||||
|
//! consumer, and announces protocol-visible facts as lines:
|
||||||
|
//! `LISTENING <addr>`, `MEMBER-UP <name> inc=<n>`, `MEMBER-DOWN <name>`.
|
||||||
|
//! Then it **parks forever** — cross-process teardown is retractable state
|
||||||
|
//! (binding trap), so the parent SIGKILLs via `Node`'s `Drop` and clean exit
|
||||||
|
//! stays the c4 harness's own smoke test.
|
||||||
|
//!
|
||||||
|
//! Ports: nodes bind `:0` and report, so the mesh is built by seeding each
|
||||||
|
//! node with the previously-reported addresses (n1: no seeds; n2: n1;
|
||||||
|
//! n3: n1+n2 — inbound covers the reverse edges). The late-seed test is the
|
||||||
|
//! one exception: the parent pre-reserves a port by binding-and-closing it,
|
||||||
|
//! seeds one node with it, then starts the second node on that exact
|
||||||
|
//! address. In principle another process could steal the port in the gap;
|
||||||
|
//! in practice the window is microseconds on a local runner — accepted, and
|
||||||
|
//! confined to that one test.
|
||||||
|
#![cfg(feature = "cluster")]
|
||||||
|
|
||||||
|
mod common;
|
||||||
|
|
||||||
|
use common::{maybe_child, spawn_node, Node};
|
||||||
|
use smarm::cluster::envelope::NodeMeta;
|
||||||
|
use smarm::cluster::membership::{subscribe, NodeEvent};
|
||||||
|
use smarm::cluster::{start, Config, StaticSeeds, Timing};
|
||||||
|
use std::time::Duration;
|
||||||
|
|
||||||
|
const ROLES: &[(&str, fn())] = &[("node", role_node)];
|
||||||
|
|
||||||
|
/// A mesh node: identity and seeds from env, membership events to stdout,
|
||||||
|
/// park forever (the parent reaps).
|
||||||
|
fn role_node() {
|
||||||
|
let name = std::env::var("SMARM_NODE_NAME").expect("SMARM_NODE_NAME not set");
|
||||||
|
let listen = std::env::var("SMARM_LISTEN_ADDR").unwrap_or_else(|_| "127.0.0.1:0".to_string());
|
||||||
|
// Seeds: comma-separated `name=addr` pairs; empty or unset means none.
|
||||||
|
let seeds: Vec<(String, String)> = std::env::var("SMARM_SEEDS")
|
||||||
|
.unwrap_or_default()
|
||||||
|
.split(',')
|
||||||
|
.filter(|s| !s.is_empty())
|
||||||
|
.map(|s| {
|
||||||
|
let (n, a) = s.split_once('=').expect("seed must be name=addr");
|
||||||
|
(n.to_string(), a.to_string())
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
|
||||||
|
smarm::run(move || {
|
||||||
|
let cluster = start(Config {
|
||||||
|
node_name: name,
|
||||||
|
meta: NodeMeta {
|
||||||
|
role: "mesh-test".to_string(),
|
||||||
|
region: "local".to_string(),
|
||||||
|
},
|
||||||
|
listen_addr: listen,
|
||||||
|
strategy: Box::new(StaticSeeds::new(seeds)),
|
||||||
|
timing: Timing::default(),
|
||||||
|
})
|
||||||
|
.expect("listener binds");
|
||||||
|
println!("LISTENING {}", cluster.local_addr());
|
||||||
|
|
||||||
|
let events = subscribe().expect("manager is up");
|
||||||
|
loop {
|
||||||
|
match events.rx.recv() {
|
||||||
|
Ok(NodeEvent::NodeUp(info)) => {
|
||||||
|
println!("MEMBER-UP {} inc={}", info.name, info.incarnation.get());
|
||||||
|
}
|
||||||
|
Ok(NodeEvent::NodeDown(info)) => {
|
||||||
|
println!("MEMBER-DOWN {}", info.name);
|
||||||
|
}
|
||||||
|
Err(_) => break, // manager gone; park below regardless
|
||||||
|
}
|
||||||
|
}
|
||||||
|
loop {
|
||||||
|
smarm::sleep(Duration::from_secs(3600));
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
fn spawn_mesh_node(name: &str, seeds: &str, listen: Option<&str>) -> Node {
|
||||||
|
let mut env: Vec<(&str, &str)> = vec![("SMARM_NODE_NAME", name), ("SMARM_SEEDS", seeds)];
|
||||||
|
if let Some(addr) = listen {
|
||||||
|
env.push(("SMARM_LISTEN_ADDR", addr));
|
||||||
|
}
|
||||||
|
spawn_node("node", &env)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Wait for `MEMBER-UP <peer> inc=<n>` and return the incarnation.
|
||||||
|
fn wait_member_up(node: &mut Node, peer: &str) -> u32 {
|
||||||
|
let prefix = format!("MEMBER-UP {peer} inc=");
|
||||||
|
let line = node.wait_line(&format!("MEMBER-UP {peer}"), |l| l.starts_with(&prefix));
|
||||||
|
line[prefix.len()..].parse().expect("incarnation parses")
|
||||||
|
}
|
||||||
|
|
||||||
|
fn wait_member_down(node: &mut Node, peer: &str) {
|
||||||
|
let want = format!("MEMBER-DOWN {peer}");
|
||||||
|
node.wait_line(&want, |l| l == want);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The gate, plus the kill and restart facts, as one mesh's life: three
|
||||||
|
/// nodes form a full mesh (every node sees both others up); killing one
|
||||||
|
/// yields `node_down` at both survivors; its restart under the same name
|
||||||
|
/// arrives as a NEW incarnation — the ghost and its successor are
|
||||||
|
/// distinguishable at every observer.
|
||||||
|
#[test]
|
||||||
|
fn three_node_mesh_forms_then_kill_then_restart_distinguishable() {
|
||||||
|
maybe_child(ROLES);
|
||||||
|
|
||||||
|
let mut n1 = spawn_mesh_node("node-1", "", None);
|
||||||
|
let a1 = n1.wait_listening();
|
||||||
|
let mut n2 = spawn_mesh_node("node-2", &format!("node-1={a1}"), None);
|
||||||
|
let a2 = n2.wait_listening();
|
||||||
|
let mut n3 = spawn_mesh_node("node-3", &format!("node-1={a1},node-2={a2}"), None);
|
||||||
|
let _a3 = n3.wait_listening();
|
||||||
|
|
||||||
|
// Full mesh: each node reports both peers up (dialed or inbound alike).
|
||||||
|
wait_member_up(&mut n1, "node-2");
|
||||||
|
let inc3_at_n1 = wait_member_up(&mut n1, "node-3");
|
||||||
|
wait_member_up(&mut n2, "node-1");
|
||||||
|
let inc3_at_n2 = wait_member_up(&mut n2, "node-3");
|
||||||
|
wait_member_up(&mut n3, "node-1");
|
||||||
|
wait_member_up(&mut n3, "node-2");
|
||||||
|
assert_eq!(
|
||||||
|
inc3_at_n1, inc3_at_n2,
|
||||||
|
"one node, one incarnation, all observers"
|
||||||
|
);
|
||||||
|
|
||||||
|
// Kill node-3 (SIGKILL via Drop): node_down at both survivors.
|
||||||
|
drop(n3);
|
||||||
|
wait_member_down(&mut n1, "node-3");
|
||||||
|
wait_member_down(&mut n2, "node-3");
|
||||||
|
|
||||||
|
// Restart node-3 under the same name: it re-dials its seeds and comes
|
||||||
|
// up everywhere as a new incarnation — never the ghost's.
|
||||||
|
let mut n3b = spawn_mesh_node("node-3", &format!("node-1={a1},node-2={a2}"), None);
|
||||||
|
let _ = n3b.wait_listening();
|
||||||
|
let inc3b_at_n1 = wait_member_up(&mut n1, "node-3");
|
||||||
|
let inc3b_at_n2 = wait_member_up(&mut n2, "node-3");
|
||||||
|
assert_eq!(inc3b_at_n1, inc3b_at_n2);
|
||||||
|
assert_ne!(
|
||||||
|
inc3_at_n1, inc3b_at_n1,
|
||||||
|
"a restarted node must be distinguishable from its ghost"
|
||||||
|
);
|
||||||
|
wait_member_up(&mut n3b, "node-1");
|
||||||
|
wait_member_up(&mut n3b, "node-2");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A seed that is unreachable at start is not fatal: the connector retries
|
||||||
|
/// on backoff, and when a node finally appears at that address, the mesh
|
||||||
|
/// edge forms.
|
||||||
|
#[test]
|
||||||
|
fn seed_unreachable_at_start_then_arriving_later() {
|
||||||
|
maybe_child(ROLES);
|
||||||
|
|
||||||
|
// Pre-reserve an address by binding and immediately closing it (see the
|
||||||
|
// module docs for the accepted steal window). Dials to it are refused
|
||||||
|
// until node-b starts there.
|
||||||
|
let reserved = {
|
||||||
|
let l = std::net::TcpListener::bind("127.0.0.1:0").expect("bind");
|
||||||
|
l.local_addr().expect("addr").to_string()
|
||||||
|
};
|
||||||
|
|
||||||
|
let mut a = spawn_mesh_node("node-a", &format!("node-b={reserved}"), None);
|
||||||
|
let _ = a.wait_listening();
|
||||||
|
|
||||||
|
// Let a few refused attempts happen before the seed comes up, so the
|
||||||
|
// retry path is what forms the edge (backoff cap 5s < harness WAIT 10s).
|
||||||
|
std::thread::sleep(Duration::from_millis(600));
|
||||||
|
|
||||||
|
let mut b = spawn_mesh_node("node-b", "", Some(&reserved));
|
||||||
|
let _ = b.wait_listening();
|
||||||
|
|
||||||
|
wait_member_up(&mut a, "node-b");
|
||||||
|
wait_member_up(&mut b, "node-a");
|
||||||
|
}
|
||||||
@@ -0,0 +1,359 @@
|
|||||||
|
//! RFC 010 c12 — remote monitors.
|
||||||
|
//!
|
||||||
|
//! Local suite (`run()`, no network): the immediate answers — no connection
|
||||||
|
//! ⇒ `Disconnected`, dead incarnation ⇒ `NoProc` — and the self-node
|
||||||
|
//! collapse (a plain local monitor underneath, incl. `demonitor_remote`).
|
||||||
|
//!
|
||||||
|
//! Cross-process: a *server* exposes a control name and spawns workers on
|
||||||
|
//! request, replying with each worker's pid (via `RemotePid::from_local`,
|
||||||
|
//! the D12 set-site) or, for the deliberately unshipped one, only its raw
|
||||||
|
//! slot numbers. The *client* monitors them and asserts: kill ⇒ the true
|
||||||
|
//! reason (Exit / Panic); a corpse ⇒ its recorded terminal reason, not
|
||||||
|
//! NoProc; a live pid that never crossed the wire ⇒ NoProc (no liveness
|
||||||
|
//! leak); a demonitor racing the kill ⇒ no notice, proven by stream ORDER
|
||||||
|
//! (a later notice on the same connection arrives while the earlier slot
|
||||||
|
//! is still empty), not by sleeping.
|
||||||
|
#![cfg(feature = "cluster")]
|
||||||
|
|
||||||
|
mod common;
|
||||||
|
|
||||||
|
use common::{maybe_child, spawn_node};
|
||||||
|
use smarm::cluster::envelope::NodeMeta;
|
||||||
|
use smarm::cluster::expose::{expose, expose_type};
|
||||||
|
use smarm::cluster::membership::{subscribe, NodeEvent};
|
||||||
|
use smarm::cluster::remote::{
|
||||||
|
self, demonitor_remote, monitor_remote, send_to_remote, RemoteName, RemotePid,
|
||||||
|
};
|
||||||
|
use smarm::cluster::{start, Config, RemoteDownReason, StaticSeeds, Timing};
|
||||||
|
use smarm::pg::Incarnation;
|
||||||
|
use smarm::{channel, install, register, run, spawn, Addressable, DownReason, Erased, Name, Pid};
|
||||||
|
use std::collections::HashMap;
|
||||||
|
use std::time::Duration;
|
||||||
|
|
||||||
|
// ---- message types (hand-rolled serde; the crate is derive-less) ---------
|
||||||
|
|
||||||
|
#[derive(Debug)]
|
||||||
|
struct Ctl {
|
||||||
|
cmd: String,
|
||||||
|
reply_to: RemotePid<Client>,
|
||||||
|
}
|
||||||
|
#[derive(Debug)]
|
||||||
|
struct Answer {
|
||||||
|
text: String,
|
||||||
|
pid: Option<RemotePid<Erased>>,
|
||||||
|
}
|
||||||
|
struct Client;
|
||||||
|
impl Addressable for Client {
|
||||||
|
type Msg = Answer;
|
||||||
|
}
|
||||||
|
|
||||||
|
impl serde::Serialize for Ctl {
|
||||||
|
fn serialize<S: serde::Serializer>(&self, s: S) -> Result<S::Ok, S::Error> {
|
||||||
|
use serde::ser::SerializeTuple;
|
||||||
|
let mut t = s.serialize_tuple(2)?;
|
||||||
|
t.serialize_element(&self.cmd)?;
|
||||||
|
t.serialize_element(&self.reply_to)?;
|
||||||
|
t.end()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
impl<'de> serde::Deserialize<'de> for Ctl {
|
||||||
|
fn deserialize<D: serde::Deserializer<'de>>(d: D) -> Result<Self, D::Error> {
|
||||||
|
let (cmd, reply_to) = <(String, RemotePid<Client>)>::deserialize(d)?;
|
||||||
|
Ok(Ctl { cmd, reply_to })
|
||||||
|
}
|
||||||
|
}
|
||||||
|
impl serde::Serialize for Answer {
|
||||||
|
fn serialize<S: serde::Serializer>(&self, s: S) -> Result<S::Ok, S::Error> {
|
||||||
|
use serde::ser::SerializeTuple;
|
||||||
|
let mut t = s.serialize_tuple(2)?;
|
||||||
|
t.serialize_element(&self.text)?;
|
||||||
|
t.serialize_element(&self.pid)?;
|
||||||
|
t.end()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
impl<'de> serde::Deserialize<'de> for Answer {
|
||||||
|
fn deserialize<D: serde::Deserializer<'de>>(d: D) -> Result<Self, D::Error> {
|
||||||
|
let (text, pid) = <(String, Option<RemotePid<Erased>>)>::deserialize(d)?;
|
||||||
|
Ok(Answer { text, pid })
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ================= local suite =========================================
|
||||||
|
|
||||||
|
/// No connection to the pid's node: `Disconnected` at once — the remote
|
||||||
|
/// analog of NoProc, and the first thing c11's variant is for.
|
||||||
|
#[test]
|
||||||
|
fn unconnected_node_is_disconnected_immediately() {
|
||||||
|
maybe_child(ROLES);
|
||||||
|
run(|| {
|
||||||
|
remote::set_local_identity("me", Incarnation::new(7));
|
||||||
|
let ghost = RemotePid::<Erased>::from_parts("nowhere", Incarnation::new(1), 3, 1);
|
||||||
|
let m = monitor_remote(ghost.clone());
|
||||||
|
let d = m.recv().unwrap();
|
||||||
|
assert_eq!(d.pid, ghost);
|
||||||
|
assert_eq!(d.reason, RemoteDownReason::Disconnected);
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The node is connected but the pid names an earlier incarnation: the
|
||||||
|
/// actor is a known corpse (RFC v2 §3), so `NoProc` at once — never
|
||||||
|
/// `Disconnected`, nothing on the wire.
|
||||||
|
#[test]
|
||||||
|
fn dead_incarnation_is_noproc_immediately() {
|
||||||
|
maybe_child(ROLES);
|
||||||
|
run(|| {
|
||||||
|
remote::set_local_identity("me", Incarnation::new(7));
|
||||||
|
let (probe_tx, probe_rx) = channel();
|
||||||
|
remote::bind_outbound_probe("peer", Incarnation::new(5), probe_tx);
|
||||||
|
let stale = RemotePid::<Erased>::from_parts("peer", Incarnation::new(4), 9, 1);
|
||||||
|
let m = monitor_remote(stale);
|
||||||
|
assert_eq!(m.recv().unwrap().reason, DownReason::NoProc.into());
|
||||||
|
assert!(probe_rx.try_recv().unwrap().is_none(), "no frame emitted");
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A self-node pid collapses to an ordinary local monitor: the true reason
|
||||||
|
/// on exit, and `demonitor_remote` cancels it.
|
||||||
|
#[test]
|
||||||
|
fn self_node_pid_collapses_to_local_monitor() {
|
||||||
|
maybe_child(ROLES);
|
||||||
|
run(|| {
|
||||||
|
remote::set_local_identity("me", Incarnation::new(7));
|
||||||
|
let (go_tx, go_rx) = channel::<()>();
|
||||||
|
let (go2_tx, go2_rx) = channel::<()>();
|
||||||
|
let a = spawn(move || {
|
||||||
|
let _ = go_rx.recv();
|
||||||
|
})
|
||||||
|
.pid();
|
||||||
|
let b = spawn(move || {
|
||||||
|
let _ = go2_rx.recv();
|
||||||
|
})
|
||||||
|
.pid();
|
||||||
|
let ma = monitor_remote(RemotePid::from_local(a).expect("identity set"));
|
||||||
|
let mb = monitor_remote(RemotePid::from_local(b).expect("identity set"));
|
||||||
|
assert_ne!(ma.id, mb.id);
|
||||||
|
assert!(ma.target.local() == Some(a));
|
||||||
|
|
||||||
|
demonitor_remote(&mb);
|
||||||
|
go2_tx.send(()).unwrap();
|
||||||
|
go_tx.send(()).unwrap();
|
||||||
|
let d = ma.recv().unwrap();
|
||||||
|
assert_eq!(d.reason, DownReason::Exit.into());
|
||||||
|
assert_eq!(d.pid.local(), Some(a));
|
||||||
|
// `a` is down (its notice arrived), and `b` was killed first on the
|
||||||
|
// same scheduler — a notice for `b` would be here by now. After a
|
||||||
|
// demonitor the channel is closed-empty (`Err`), like the local one.
|
||||||
|
assert!(matches!(mb.try_recv(), Ok(None) | Err(_)));
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
// ================= cross-process ======================================
|
||||||
|
|
||||||
|
const ROLES: &[(&str, fn())] = &[("server", role_server), ("client", role_client)];
|
||||||
|
|
||||||
|
const CTL: Name<Ctl> = Name::new("c12.ctl");
|
||||||
|
|
||||||
|
fn cfg(name: &str, seeds: Vec<(String, String)>) -> Config {
|
||||||
|
Config {
|
||||||
|
node_name: name.to_string(),
|
||||||
|
meta: NodeMeta {
|
||||||
|
role: "c12".into(),
|
||||||
|
region: "local".into(),
|
||||||
|
},
|
||||||
|
listen_addr: std::env::var("SMARM_LISTEN_ADDR").unwrap_or_else(|_| "127.0.0.1:0".into()),
|
||||||
|
strategy: Box::new(StaticSeeds::new(seeds)),
|
||||||
|
timing: Timing::default(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn wait_up(events: &smarm::cluster::membership::MembershipEvents, who: &str) {
|
||||||
|
loop {
|
||||||
|
match events.rx.recv() {
|
||||||
|
Ok(NodeEvent::NodeUp(i)) if i.name == who => return,
|
||||||
|
Ok(_) => continue,
|
||||||
|
Err(_) => panic!("manager gone"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Server commands (all answered to `reply_to`):
|
||||||
|
/// - `spawn:exit` / `spawn:panic` — a parked worker; `kill:<index>` releases
|
||||||
|
/// it, whereupon it returns / panics. Answer carries its pid.
|
||||||
|
/// - `spawn:corpse` — a worker that has already exited when the answer is
|
||||||
|
/// sent; the pid was shipped (watchable) before it died.
|
||||||
|
/// - `spawn:unwatched` — a parked worker whose pid is NEVER shipped; the
|
||||||
|
/// answer carries only `text = "slot:<index>:<generation>"`.
|
||||||
|
fn role_server() {
|
||||||
|
smarm::run(move || {
|
||||||
|
let cluster = start(cfg("server", vec![])).expect("binds");
|
||||||
|
println!("LISTENING {}", cluster.local_addr());
|
||||||
|
let (tx, rx) = channel::<Ctl>();
|
||||||
|
register(CTL, tx).unwrap();
|
||||||
|
expose(CTL);
|
||||||
|
println!("READY");
|
||||||
|
let mut workers: HashMap<u32, smarm::channel::Sender<()>> = HashMap::new();
|
||||||
|
loop {
|
||||||
|
let ctl = rx.recv().unwrap();
|
||||||
|
println!("CTL {}", ctl.cmd);
|
||||||
|
let (text, pid): (String, Option<RemotePid<Erased>>) = match ctl.cmd.as_str() {
|
||||||
|
"spawn:exit" | "spawn:panic" => {
|
||||||
|
let panic = ctl.cmd == "spawn:panic";
|
||||||
|
let (go_tx, go_rx) = channel::<()>();
|
||||||
|
let p: Pid = spawn(move || {
|
||||||
|
let _ = go_rx.recv();
|
||||||
|
if panic {
|
||||||
|
panic!("worker asked to panic");
|
||||||
|
}
|
||||||
|
})
|
||||||
|
.pid();
|
||||||
|
workers.insert(p.index(), go_tx);
|
||||||
|
(
|
||||||
|
"ok".into(),
|
||||||
|
Some(RemotePid::from_local(p).expect("identity set")),
|
||||||
|
)
|
||||||
|
}
|
||||||
|
"spawn:corpse" => {
|
||||||
|
let p: Pid = spawn(|| {}).pid();
|
||||||
|
let rp = RemotePid::from_local(p).expect("identity set"); // shipped ⇒ watchable
|
||||||
|
let m = smarm::monitor(p);
|
||||||
|
let _ = m.rx.recv(); // dead before the answer goes out
|
||||||
|
("ok".into(), Some(rp))
|
||||||
|
}
|
||||||
|
"spawn:unwatched" => {
|
||||||
|
let (go_tx, go_rx) = channel::<()>();
|
||||||
|
let p: Pid = spawn(move || {
|
||||||
|
let _ = go_rx.recv();
|
||||||
|
})
|
||||||
|
.pid();
|
||||||
|
workers.insert(p.index(), go_tx);
|
||||||
|
(format!("slot:{}:{}", p.index(), p.generation()), None)
|
||||||
|
}
|
||||||
|
other => {
|
||||||
|
let idx: u32 = other.strip_prefix("kill:").unwrap().parse().unwrap();
|
||||||
|
if let Some(go) = workers.remove(&idx) {
|
||||||
|
let _ = go.send(());
|
||||||
|
}
|
||||||
|
("killed".into(), None)
|
||||||
|
}
|
||||||
|
};
|
||||||
|
send_to_remote(ctl.reply_to, Answer { text, pid }).unwrap();
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
fn role_client() {
|
||||||
|
let server_addr = std::env::var("SMARM_SERVER_ADDR").expect("SMARM_SERVER_ADDR");
|
||||||
|
smarm::run(move || {
|
||||||
|
let _cluster = start(cfg("client", vec![("server".into(), server_addr)])).expect("binds");
|
||||||
|
let ev = subscribe().unwrap();
|
||||||
|
wait_up(&ev, "server");
|
||||||
|
let (tx, rx) = channel::<Answer>();
|
||||||
|
let me: Pid<Client> = install::<Client>(tx);
|
||||||
|
expose_type::<Answer>();
|
||||||
|
let ask = |cmd: &str| -> Answer {
|
||||||
|
remote::send(
|
||||||
|
RemoteName::new("server", CTL),
|
||||||
|
Ctl {
|
||||||
|
cmd: cmd.into(),
|
||||||
|
reply_to: RemotePid::from_local(me).expect("identity set"),
|
||||||
|
},
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
rx.recv().unwrap()
|
||||||
|
};
|
||||||
|
let server_inc = ev_incarnation();
|
||||||
|
|
||||||
|
// 1. kill ⇒ true reason (Exit).
|
||||||
|
let a = ask("spawn:exit").pid.unwrap();
|
||||||
|
let ma = monitor_remote(a.clone());
|
||||||
|
ask(&format!("kill:{}", a.index()));
|
||||||
|
let d = ma.recv().unwrap();
|
||||||
|
assert_eq!(d.pid, a);
|
||||||
|
println!("DOWN exit {:?}", d.reason);
|
||||||
|
|
||||||
|
// 2. kill ⇒ true reason (Panic).
|
||||||
|
let b = ask("spawn:panic").pid.unwrap();
|
||||||
|
let mb = monitor_remote(b.clone());
|
||||||
|
ask(&format!("kill:{}", b.index()));
|
||||||
|
println!("DOWN panic {:?}", mb.recv().unwrap().reason);
|
||||||
|
|
||||||
|
// 3. corpse ⇒ recorded terminal reason, not NoProc.
|
||||||
|
let c = ask("spawn:corpse").pid.unwrap();
|
||||||
|
println!("DOWN corpse {:?}", monitor_remote(c).recv().unwrap().reason);
|
||||||
|
|
||||||
|
// 4. live but never shipped/exposed ⇒ NoProc (no leak); a made-up
|
||||||
|
// slot on the same node ⇒ NoProc too, indistinguishably.
|
||||||
|
let ans = ask("spawn:unwatched");
|
||||||
|
let mut it = ans.text.strip_prefix("slot:").unwrap().split(':');
|
||||||
|
let (idx, gen): (u32, u32) = (
|
||||||
|
it.next().unwrap().parse().unwrap(),
|
||||||
|
it.next().unwrap().parse().unwrap(),
|
||||||
|
);
|
||||||
|
let hidden = RemotePid::<Erased>::from_parts("server", server_inc, idx, gen);
|
||||||
|
println!(
|
||||||
|
"DOWN hidden {:?}",
|
||||||
|
monitor_remote(hidden).recv().unwrap().reason
|
||||||
|
);
|
||||||
|
let bogus = RemotePid::<Erased>::from_parts("server", server_inc, 100_000, 1);
|
||||||
|
println!(
|
||||||
|
"DOWN bogus {:?}",
|
||||||
|
monitor_remote(bogus).recv().unwrap().reason
|
||||||
|
);
|
||||||
|
|
||||||
|
// 5. demonitor races the kill: no notice for `d1`, proven by order —
|
||||||
|
// `d2`'s notice (same connection, later) arrives while `d1`'s
|
||||||
|
// slot is still empty.
|
||||||
|
let d1 = ask("spawn:exit").pid.unwrap();
|
||||||
|
let m1 = monitor_remote(d1.clone());
|
||||||
|
demonitor_remote(&m1);
|
||||||
|
ask(&format!("kill:{}", d1.index()));
|
||||||
|
let d2 = ask("spawn:exit").pid.unwrap();
|
||||||
|
let m2 = monitor_remote(d2.clone());
|
||||||
|
ask(&format!("kill:{}", d2.index()));
|
||||||
|
assert_eq!(m2.recv().unwrap().reason, DownReason::Exit.into());
|
||||||
|
// Closed-empty (`Err`) or open-empty (`Ok(None)`) both mean no notice.
|
||||||
|
let stray = matches!(m1.try_recv(), Ok(Some(_)));
|
||||||
|
println!("DEMONITOR stray={stray}");
|
||||||
|
|
||||||
|
println!("CLIENT DONE");
|
||||||
|
loop {
|
||||||
|
smarm::sleep(Duration::from_secs(3600));
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The server's incarnation as this node sees it — for building pids by hand.
|
||||||
|
fn ev_incarnation() -> Incarnation {
|
||||||
|
smarm::cluster::membership::view()
|
||||||
|
.expect("manager up")
|
||||||
|
.into_iter()
|
||||||
|
.find(|i| i.name == "server")
|
||||||
|
.map(|i| i.incarnation)
|
||||||
|
.expect("server in view")
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The Phase 4 c12 gate: remote monitors report the true reason, honour
|
||||||
|
/// corpses, leak nothing for unshipped pids, and cancel cleanly.
|
||||||
|
#[test]
|
||||||
|
fn remote_monitors_report_true_reasons() {
|
||||||
|
maybe_child(ROLES);
|
||||||
|
let mut server = spawn_node("server", &[]);
|
||||||
|
let saddr = server.wait_listening();
|
||||||
|
server.wait_line("READY", |l| l == "READY");
|
||||||
|
let mut client = spawn_node("client", &[("SMARM_SERVER_ADDR", &saddr)]);
|
||||||
|
client.wait_line("DOWN exit Local(Exit)", |l| l == "DOWN exit Local(Exit)");
|
||||||
|
client.wait_line("DOWN panic Local(Panic)", |l| {
|
||||||
|
l == "DOWN panic Local(Panic)"
|
||||||
|
});
|
||||||
|
client.wait_line("DOWN corpse Local(Exit)", |l| {
|
||||||
|
l == "DOWN corpse Local(Exit)"
|
||||||
|
});
|
||||||
|
client.wait_line("DOWN hidden Local(NoProc)", |l| {
|
||||||
|
l == "DOWN hidden Local(NoProc)"
|
||||||
|
});
|
||||||
|
client.wait_line("DOWN bogus Local(NoProc)", |l| {
|
||||||
|
l == "DOWN bogus Local(NoProc)"
|
||||||
|
});
|
||||||
|
client.wait_line("DEMONITOR stray=false", |l| l == "DEMONITOR stray=false");
|
||||||
|
client.wait_line("CLIENT DONE", |l| l == "CLIENT DONE");
|
||||||
|
}
|
||||||
@@ -0,0 +1,254 @@
|
|||||||
|
//! RFC 010 c15 — distributed pg: sync on `NodeUp`, incremental
|
||||||
|
//! `Join`/`Leave`, eager eviction announced, `NodeDown` sweep.
|
||||||
|
//!
|
||||||
|
//! Two nodes. The *origin* joins two local workers to `"pool"` before the
|
||||||
|
//! *observer* connects (so the observer's view comes from `Sync`), exposes a
|
||||||
|
//! `"go"` command inbox and then does exactly what the observer tells it:
|
||||||
|
//! kill one worker, join a third, leave with the second. The observer drives
|
||||||
|
//! that script through the cluster itself and asserts every step from
|
||||||
|
//! `members_all` — never touching the group on its own side, except once to
|
||||||
|
//! prove a mixed local+remote group reads correctly and that `members` stays
|
||||||
|
//! local. `dispatch_any` is exercised both ways: into the origin's worker
|
||||||
|
//! (remote pick, `send_to_remote`) and, once the origin is gone, into the
|
||||||
|
//! observer's own (local pick, `send_to`). Finally the parent SIGKILLs the
|
||||||
|
//! origin: the observer must sweep every remote member on `NodeDown`.
|
||||||
|
#![cfg(feature = "cluster")]
|
||||||
|
|
||||||
|
mod common;
|
||||||
|
|
||||||
|
use common::{maybe_child, spawn_node};
|
||||||
|
use smarm::cluster::envelope::NodeMeta;
|
||||||
|
use smarm::cluster::expose::{expose, expose_type};
|
||||||
|
use smarm::cluster::membership::{subscribe, NodeEvent};
|
||||||
|
use smarm::cluster::remote::{self, RemoteName};
|
||||||
|
use smarm::cluster::{
|
||||||
|
dispatch_any, members_all, pick_any, start, Config, DispatchAnyError, GroupMember, StaticSeeds,
|
||||||
|
Timing,
|
||||||
|
};
|
||||||
|
use smarm::{channel, join, leave, members, register, send_to, spawn_addr, Addressable, Name, Pid};
|
||||||
|
use std::time::{Duration, Instant};
|
||||||
|
|
||||||
|
const GO: Name<u8> = Name::new("go");
|
||||||
|
const POOL: &str = "pool";
|
||||||
|
|
||||||
|
/// A pool worker's message: `"die"` stops it, anything else is printed.
|
||||||
|
#[derive(Debug, PartialEq)]
|
||||||
|
struct Job(String);
|
||||||
|
struct Worker;
|
||||||
|
impl Addressable for Worker {
|
||||||
|
type Msg = Job;
|
||||||
|
}
|
||||||
|
impl serde::Serialize for Job {
|
||||||
|
fn serialize<S: serde::Serializer>(&self, s: S) -> Result<S::Ok, S::Error> {
|
||||||
|
self.0.serialize(s)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
impl<'de> serde::Deserialize<'de> for Job {
|
||||||
|
fn deserialize<D: serde::Deserializer<'de>>(d: D) -> Result<Self, D::Error> {
|
||||||
|
String::deserialize(d).map(Job)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
const ROLES: &[(&str, fn())] = &[("origin", role_origin), ("observer", role_observer)];
|
||||||
|
|
||||||
|
fn cfg(name: &str, seeds: Vec<(String, String)>) -> Config {
|
||||||
|
Config {
|
||||||
|
node_name: name.into(),
|
||||||
|
meta: NodeMeta {
|
||||||
|
role: "c15".into(),
|
||||||
|
region: "local".into(),
|
||||||
|
},
|
||||||
|
listen_addr: "127.0.0.1:0".into(),
|
||||||
|
strategy: Box::new(StaticSeeds::new(seeds)),
|
||||||
|
timing: Timing::default(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A pool worker: prints every job it is handed, exits on `"die"`.
|
||||||
|
fn worker() -> Pid<Worker> {
|
||||||
|
spawn_addr::<Worker>(|rx| {
|
||||||
|
while let Ok(Job(s)) = rx.recv() {
|
||||||
|
if s == "die" {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
println!("JOB {s}");
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
fn role_origin() {
|
||||||
|
smarm::run(|| {
|
||||||
|
let cluster = start(cfg("origin", vec![])).expect("binds");
|
||||||
|
// Remote dispatch lands here only for a type this node accepts.
|
||||||
|
expose_type::<Job>();
|
||||||
|
let w1 = worker();
|
||||||
|
let w2 = worker();
|
||||||
|
assert!(join(POOL, w1));
|
||||||
|
assert!(join(POOL, w2));
|
||||||
|
let (go_tx, go_rx) = channel::<u8>();
|
||||||
|
register(GO, go_tx).unwrap();
|
||||||
|
expose(GO);
|
||||||
|
println!("LISTENING {}", cluster.local_addr());
|
||||||
|
println!("JOINED 2");
|
||||||
|
loop {
|
||||||
|
match go_rx.recv().unwrap() {
|
||||||
|
1 => {
|
||||||
|
send_to(w1, Job("die".into())).unwrap();
|
||||||
|
println!("KILLED w1");
|
||||||
|
}
|
||||||
|
2 => {
|
||||||
|
assert!(leave(POOL, w2));
|
||||||
|
println!("LEFT w2");
|
||||||
|
}
|
||||||
|
3 => {
|
||||||
|
let w3 = worker();
|
||||||
|
assert!(join(POOL, w3));
|
||||||
|
println!("JOINED w3");
|
||||||
|
}
|
||||||
|
n => panic!("unknown command {n}"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
fn remote_count(group: &str) -> usize {
|
||||||
|
members_all(group)
|
||||||
|
.iter()
|
||||||
|
.filter(|m| matches!(m, GroupMember::Remote(_)))
|
||||||
|
.count()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Cooperative poll until `pred`; panics (with the last view) on timeout.
|
||||||
|
fn wait_view(what: &str, group: &str, pred: impl Fn(&[GroupMember]) -> bool) {
|
||||||
|
let deadline = Instant::now() + Duration::from_secs(5);
|
||||||
|
loop {
|
||||||
|
let v = members_all(group);
|
||||||
|
if pred(&v) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
assert!(
|
||||||
|
Instant::now() < deadline,
|
||||||
|
"timed out waiting for {what}; view = {v:?}"
|
||||||
|
);
|
||||||
|
smarm::sleep(Duration::from_millis(5));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn role_observer() {
|
||||||
|
let origin_addr = std::env::var("SMARM_ORIGIN_ADDR").expect("SMARM_ORIGIN_ADDR");
|
||||||
|
smarm::run(move || {
|
||||||
|
let _cluster = start(cfg("observer", vec![("origin".into(), origin_addr)])).expect("binds");
|
||||||
|
let ev = subscribe().unwrap();
|
||||||
|
loop {
|
||||||
|
match ev.rx.recv().unwrap() {
|
||||||
|
NodeEvent::NodeUp(i) if i.name == "origin" => break,
|
||||||
|
_ => {}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let go = |n: u8| remote::send(RemoteName::new("origin", GO), n).unwrap();
|
||||||
|
|
||||||
|
// Sync: both pre-existing members arrive with no join on this side.
|
||||||
|
wait_view("sync of 2 remote members", POOL, |v| {
|
||||||
|
v.len() == 2 && v.iter().all(|m| matches!(m, GroupMember::Remote(_)))
|
||||||
|
});
|
||||||
|
let synced = members_all(POOL);
|
||||||
|
assert!(synced.iter().all(|m| match m {
|
||||||
|
GroupMember::Remote(p) => p.node() == "origin",
|
||||||
|
GroupMember::Local(_) => false,
|
||||||
|
}));
|
||||||
|
println!("SEES 2");
|
||||||
|
|
||||||
|
// Origin-side death: the origin's reaper announces the leave.
|
||||||
|
go(1);
|
||||||
|
wait_view("death evicted on observer", POOL, |v| v.len() == 1);
|
||||||
|
println!("SEES 1 after death");
|
||||||
|
|
||||||
|
// Incremental Join.
|
||||||
|
go(3);
|
||||||
|
wait_view("incremental join", POOL, |v| v.len() == 2);
|
||||||
|
println!("SEES 2 after join");
|
||||||
|
|
||||||
|
// Voluntary Leave.
|
||||||
|
go(2);
|
||||||
|
wait_view("incremental leave", POOL, |v| v.len() == 1);
|
||||||
|
println!("SEES 1 after leave");
|
||||||
|
|
||||||
|
// Mixed group: our own member sits beside the remote one in
|
||||||
|
// `members_all`; `members` stays local-only.
|
||||||
|
let me = worker();
|
||||||
|
assert!(join(POOL, me));
|
||||||
|
wait_view("mixed local+remote", POOL, |v| {
|
||||||
|
v.len() == 2 && v.contains(&GroupMember::Local(me.erase()))
|
||||||
|
});
|
||||||
|
assert_eq!(
|
||||||
|
members(POOL),
|
||||||
|
vec![me.erase()],
|
||||||
|
"local API never shows remotes"
|
||||||
|
);
|
||||||
|
assert_eq!(remote_count(POOL), 1);
|
||||||
|
println!("MIXED ok");
|
||||||
|
|
||||||
|
// dispatch_any: the store's first entry is the origin's w3 (it was
|
||||||
|
// announced before we joined), so the pick is remote and the job
|
||||||
|
// crosses the wire — the origin's worker prints it.
|
||||||
|
let picked = pick_any(POOL).expect("pool has members");
|
||||||
|
assert!(
|
||||||
|
matches!(picked, GroupMember::Remote(_)),
|
||||||
|
"first entry is remote: {picked:?}"
|
||||||
|
);
|
||||||
|
let reached = dispatch_any::<Worker>(POOL, Job("from-observer".into())).unwrap();
|
||||||
|
assert_eq!(reached, picked);
|
||||||
|
println!("DISPATCHED remote");
|
||||||
|
|
||||||
|
println!("PARK");
|
||||||
|
// Parent SIGKILLs the origin now: NodeDown must sweep its member,
|
||||||
|
// ours must survive.
|
||||||
|
wait_view("node_down sweep", POOL, |v| {
|
||||||
|
v == [GroupMember::Local(me.erase())]
|
||||||
|
});
|
||||||
|
assert_eq!(members(POOL), vec![me.erase()]);
|
||||||
|
println!("SWEPT");
|
||||||
|
|
||||||
|
// Now the only member is ours: a local pick, a local send.
|
||||||
|
let reached = dispatch_any::<Worker>(POOL, Job("local".into())).unwrap();
|
||||||
|
assert_eq!(reached, GroupMember::Local(me.erase()));
|
||||||
|
// And an empty group hands the message back.
|
||||||
|
match dispatch_any::<Worker>("nobody", Job("lost".into())) {
|
||||||
|
Err(DispatchAnyError::NoMember(Job(s))) => assert_eq!(s, "lost"),
|
||||||
|
other => panic!("expected NoMember, got {other:?}"),
|
||||||
|
}
|
||||||
|
println!("DISPATCHED local");
|
||||||
|
loop {
|
||||||
|
smarm::sleep(Duration::from_secs(3600));
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The Phase 5 gate: sync, join, leave, death, node_down — all observed from
|
||||||
|
/// the peer, none of them a group operation on the peer — plus dispatch_any
|
||||||
|
/// reaching a remote member and a local one.
|
||||||
|
#[test]
|
||||||
|
fn groups_span_two_nodes() {
|
||||||
|
maybe_child(ROLES);
|
||||||
|
let mut origin = spawn_node("origin", &[]);
|
||||||
|
let addr = origin.wait_listening();
|
||||||
|
origin.wait_line("JOINED 2", |l| l == "JOINED 2");
|
||||||
|
let mut observer = spawn_node("observer", &[("SMARM_ORIGIN_ADDR", &addr)]);
|
||||||
|
observer.wait_line("SEES 2", |l| l == "SEES 2");
|
||||||
|
origin.wait_line("KILLED w1", |l| l == "KILLED w1");
|
||||||
|
observer.wait_line("SEES 1 after death", |l| l == "SEES 1 after death");
|
||||||
|
origin.wait_line("JOINED w3", |l| l == "JOINED w3");
|
||||||
|
observer.wait_line("SEES 2 after join", |l| l == "SEES 2 after join");
|
||||||
|
origin.wait_line("LEFT w2", |l| l == "LEFT w2");
|
||||||
|
observer.wait_line("SEES 1 after leave", |l| l == "SEES 1 after leave");
|
||||||
|
observer.wait_line("MIXED ok", |l| l == "MIXED ok");
|
||||||
|
observer.wait_line("DISPATCHED remote", |l| l == "DISPATCHED remote");
|
||||||
|
origin.wait_line("JOB from-observer", |l| l == "JOB from-observer");
|
||||||
|
observer.wait_line("PARK", |l| l == "PARK");
|
||||||
|
origin.kill();
|
||||||
|
observer.wait_line("SWEPT", |l| l == "SWEPT");
|
||||||
|
// Order between the root's line and the worker's is scheduling; wait
|
||||||
|
// for the later one to be certain both happened.
|
||||||
|
observer.wait_line("DISPATCHED local", |l| l == "DISPATCHED local");
|
||||||
|
observer.wait_line("JOB local", |l| l == "JOB local");
|
||||||
|
}
|
||||||
@@ -0,0 +1,356 @@
|
|||||||
|
//! RFC 010 c10 — pid targeting + auto-serialization. The Phase 3 gate:
|
||||||
|
//! cross-node call/reply with no ceremony, under the subprocess harness.
|
||||||
|
//!
|
||||||
|
//! Local suite (`run()`, no network): serialize/deserialize shapes,
|
||||||
|
//! self-collapse, the outside-runtime contract, the local send-site
|
||||||
|
//! incarnation check with a probe proving **no frame is emitted**.
|
||||||
|
//!
|
||||||
|
//! Cross-process: two nodes. The *server* exposes a `Name<Req>`; the
|
||||||
|
//! *client* sends a `Req` carrying its own `Pid<Reply>` (auto-serialized to
|
||||||
|
//! a `RemotePid` on the wire); the server replies via `send_to_remote`
|
||||||
|
//! straight back to that pid — no name at the client end, no ceremony. A
|
||||||
|
//! third-node roundtrip: the client's pid travels client→server→relay→
|
||||||
|
//! server→client, and still delivers.
|
||||||
|
#![cfg(feature = "cluster")]
|
||||||
|
|
||||||
|
mod common;
|
||||||
|
|
||||||
|
use common::{maybe_child, spawn_node};
|
||||||
|
use smarm::cluster::envelope::{encode_payload, Frame, NodeMeta};
|
||||||
|
use smarm::cluster::expose::{expose, type_hash};
|
||||||
|
use smarm::cluster::membership::{subscribe, NodeEvent};
|
||||||
|
use smarm::cluster::remote::{self, send_to_remote, RemoteName, RemotePid, ToRemoteError};
|
||||||
|
use smarm::cluster::{start, Config, StaticSeeds, Timing};
|
||||||
|
use smarm::pg::Incarnation;
|
||||||
|
use smarm::{channel, install, register, run, Addressable, Name, Pid};
|
||||||
|
use std::time::Duration;
|
||||||
|
|
||||||
|
// ---- message types (std-only payloads; the crate's serde is derive-less,
|
||||||
|
// so wire types are hand-rolled with serde's tuple/seq API via `serde::ser`
|
||||||
|
// impls below — the same thing a user's derive would generate) ------------
|
||||||
|
|
||||||
|
/// A request carrying a reply-to. Serialize/Deserialize are written by hand
|
||||||
|
/// here for exactly one reason: this crate deliberately does not pull in
|
||||||
|
/// serde-derive. Field 1 is the auto-serializing pid.
|
||||||
|
#[derive(Debug, PartialEq)]
|
||||||
|
struct Req {
|
||||||
|
text: String,
|
||||||
|
reply_to: RemotePid<Replier>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, PartialEq)]
|
||||||
|
struct Reply(String);
|
||||||
|
|
||||||
|
struct Replier;
|
||||||
|
impl Addressable for Replier {
|
||||||
|
type Msg = Reply;
|
||||||
|
}
|
||||||
|
|
||||||
|
impl serde::Serialize for Req {
|
||||||
|
fn serialize<S: serde::Serializer>(&self, s: S) -> Result<S::Ok, S::Error> {
|
||||||
|
use serde::ser::SerializeTuple;
|
||||||
|
let mut t = s.serialize_tuple(2)?;
|
||||||
|
t.serialize_element(&self.text)?;
|
||||||
|
t.serialize_element(&self.reply_to)?;
|
||||||
|
t.end()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
impl<'de> serde::Deserialize<'de> for Req {
|
||||||
|
fn deserialize<D: serde::Deserializer<'de>>(d: D) -> Result<Self, D::Error> {
|
||||||
|
let (text, reply_to) = <(String, RemotePid<Replier>)>::deserialize(d)?;
|
||||||
|
Ok(Req { text, reply_to })
|
||||||
|
}
|
||||||
|
}
|
||||||
|
impl serde::Serialize for Reply {
|
||||||
|
fn serialize<S: serde::Serializer>(&self, s: S) -> Result<S::Ok, S::Error> {
|
||||||
|
self.0.serialize(s)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
impl<'de> serde::Deserialize<'de> for Reply {
|
||||||
|
fn deserialize<D: serde::Deserializer<'de>>(d: D) -> Result<Self, D::Error> {
|
||||||
|
String::deserialize(d).map(Reply)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ================= local suite =========================================
|
||||||
|
|
||||||
|
/// A local `Pid<A>` serializes as a `RemotePid<A>` stamped with this node's
|
||||||
|
/// identity; deserializing it back on the same node collapses to the same
|
||||||
|
/// local pid (`local()` is `Some`, `Pid` round-trips).
|
||||||
|
#[test]
|
||||||
|
fn local_pid_serializes_and_collapses_on_self() {
|
||||||
|
maybe_child(ROLES);
|
||||||
|
run(|| {
|
||||||
|
// The local identity is set by cluster::start; the local suite sets
|
||||||
|
// it directly.
|
||||||
|
remote::set_local_identity("me", Incarnation::new(7));
|
||||||
|
let (tx, _rx) = channel::<Reply>();
|
||||||
|
let me: Pid<Replier> = install::<Replier>(tx);
|
||||||
|
|
||||||
|
let bytes = encode_payload(&me).unwrap();
|
||||||
|
let rp: RemotePid<Replier> = smarm::cluster::envelope::decode_payload(&bytes).unwrap();
|
||||||
|
assert_eq!(rp.node(), "me");
|
||||||
|
assert_eq!(rp.incarnation(), Incarnation::new(7));
|
||||||
|
assert_eq!(
|
||||||
|
rp.local(),
|
||||||
|
Some(me),
|
||||||
|
"self-node pid collapses to the local pid"
|
||||||
|
);
|
||||||
|
|
||||||
|
// Deserializing straight into Pid<A> works for a self-node pid...
|
||||||
|
let back: Pid<Replier> = smarm::cluster::envelope::decode_payload(&bytes).unwrap();
|
||||||
|
assert_eq!(back, me);
|
||||||
|
|
||||||
|
// ...and FAILS for a foreign one (collapse is literal: node == self).
|
||||||
|
let foreign = RemotePid::<Replier>::from_parts("elsewhere", Incarnation::new(1), 3, 1);
|
||||||
|
let fbytes = encode_payload(&foreign).unwrap();
|
||||||
|
assert!(smarm::cluster::envelope::decode_payload::<Pid<Replier>>(&fbytes).is_err());
|
||||||
|
assert_eq!(foreign.local(), None);
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `send_to_remote` short-circuits locally for a self-node pid — the
|
||||||
|
/// zero-copy-equivalent collapse: the message object itself lands in the
|
||||||
|
/// local channel, no encode, no frame.
|
||||||
|
#[test]
|
||||||
|
fn send_to_remote_collapses_locally_for_self() {
|
||||||
|
maybe_child(ROLES);
|
||||||
|
run(|| {
|
||||||
|
remote::set_local_identity("me", Incarnation::new(7));
|
||||||
|
let (tx, rx) = channel::<Reply>();
|
||||||
|
let me: Pid<Replier> = install::<Replier>(tx);
|
||||||
|
let rp = RemotePid::from_local(me).expect("identity set");
|
||||||
|
// Probe the outbound path: nothing must be handed to any connection.
|
||||||
|
let (probe_tx, probe_rx) = channel::<Frame>();
|
||||||
|
remote::bind_outbound_probe("me", Incarnation::new(7), probe_tx);
|
||||||
|
|
||||||
|
send_to_remote(rp, Reply("hi".into())).unwrap();
|
||||||
|
assert_eq!(rx.recv().unwrap(), Reply("hi".into()));
|
||||||
|
assert!(
|
||||||
|
matches!(probe_rx.try_recv(), Ok(None)),
|
||||||
|
"no frame for a local collapse"
|
||||||
|
);
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// RFC v2 §3: a `RemotePid` whose incarnation is not the current one for its
|
||||||
|
/// node fails at the local send site with `DeadIncarnation`, and NO frame
|
||||||
|
/// is emitted — asserted on a probe sender bound as that node's outbound.
|
||||||
|
#[test]
|
||||||
|
fn stale_incarnation_rejected_locally_no_frame() {
|
||||||
|
maybe_child(ROLES);
|
||||||
|
run(|| {
|
||||||
|
remote::set_local_identity("me", Incarnation::new(7));
|
||||||
|
let (probe_tx, probe_rx) = channel::<Frame>();
|
||||||
|
remote::bind_outbound_probe("peer", Incarnation::new(5), probe_tx);
|
||||||
|
|
||||||
|
let stale = RemotePid::<Replier>::from_parts("peer", Incarnation::new(4), 9, 1);
|
||||||
|
match send_to_remote(stale, Reply("late".into())) {
|
||||||
|
Err(ToRemoteError::DeadIncarnation(Reply(s))) => assert_eq!(s, "late"),
|
||||||
|
other => panic!("expected DeadIncarnation, got {other:?}"),
|
||||||
|
}
|
||||||
|
assert!(
|
||||||
|
matches!(probe_rx.try_recv(), Ok(None)),
|
||||||
|
"stale pid must emit no frame"
|
||||||
|
);
|
||||||
|
|
||||||
|
// The current incarnation goes through: a Send frame with the pid's
|
||||||
|
// (index, generation) and Reply's hash lands on the probe.
|
||||||
|
let live = RemotePid::<Replier>::from_parts("peer", Incarnation::new(5), 9, 1);
|
||||||
|
send_to_remote(live, Reply("now".into())).unwrap();
|
||||||
|
match probe_rx.recv().unwrap() {
|
||||||
|
Frame::Send {
|
||||||
|
index,
|
||||||
|
generation,
|
||||||
|
type_hash: h,
|
||||||
|
payload,
|
||||||
|
} => {
|
||||||
|
assert_eq!((index, generation), (9, 1));
|
||||||
|
assert_eq!(h, type_hash::<Reply>());
|
||||||
|
let r: Reply = smarm::cluster::envelope::decode_payload(&payload).unwrap();
|
||||||
|
assert_eq!(r, Reply("now".into()));
|
||||||
|
}
|
||||||
|
f => panic!("expected Send, got {f:?}"),
|
||||||
|
}
|
||||||
|
|
||||||
|
// Unknown node: NotConnected, no frame anywhere.
|
||||||
|
let nowhere = RemotePid::<Replier>::from_parts("nowhere", Incarnation::new(1), 1, 1);
|
||||||
|
assert!(matches!(
|
||||||
|
send_to_remote(nowhere, Reply("x".into())),
|
||||||
|
Err(ToRemoteError::NotConnected(_))
|
||||||
|
));
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
// ================= cross-process gate ==================================
|
||||||
|
|
||||||
|
const ROLES: &[(&str, fn())] = &[
|
||||||
|
("server", role_server),
|
||||||
|
("client", role_client),
|
||||||
|
("relay", role_relay),
|
||||||
|
];
|
||||||
|
|
||||||
|
const ECHO: Name<Req> = Name::new("c10.echo");
|
||||||
|
const RELAY: Name<Req> = Name::new("c10.relay");
|
||||||
|
|
||||||
|
fn cfg(name: &str, seeds: Vec<(String, String)>) -> Config {
|
||||||
|
Config {
|
||||||
|
node_name: name.to_string(),
|
||||||
|
meta: NodeMeta {
|
||||||
|
role: "c10".into(),
|
||||||
|
region: "local".into(),
|
||||||
|
},
|
||||||
|
listen_addr: std::env::var("SMARM_LISTEN_ADDR").unwrap_or_else(|_| "127.0.0.1:0".into()),
|
||||||
|
strategy: Box::new(StaticSeeds::new(seeds)),
|
||||||
|
timing: Timing::default(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn wait_up(events: &smarm::cluster::membership::MembershipEvents, who: &str) {
|
||||||
|
loop {
|
||||||
|
match events.rx.recv() {
|
||||||
|
Ok(NodeEvent::NodeUp(i)) if i.name == who => return,
|
||||||
|
Ok(_) => continue,
|
||||||
|
Err(_) => panic!("manager gone"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Server: exposes ECHO; each Req is answered by `send_to_remote` to its
|
||||||
|
/// reply_to — the server never learns a name for the client. If the Req text
|
||||||
|
/// starts with "via-relay:", it forwards the whole Req (reply_to and all) to
|
||||||
|
/// the relay node instead, which sends it back here; the second arrival is
|
||||||
|
/// answered normally. That is the pid's third-node roundtrip.
|
||||||
|
fn role_server() {
|
||||||
|
let relay_addr = std::env::var("SMARM_RELAY_ADDR").ok();
|
||||||
|
smarm::run(move || {
|
||||||
|
let seeds = relay_addr
|
||||||
|
.map(|a| vec![("relay".to_string(), a)])
|
||||||
|
.unwrap_or_default();
|
||||||
|
let cluster = start(cfg("server", seeds)).expect("binds");
|
||||||
|
println!("LISTENING {}", cluster.local_addr());
|
||||||
|
let (tx, rx) = channel::<Req>();
|
||||||
|
register(ECHO, tx).unwrap();
|
||||||
|
expose(ECHO);
|
||||||
|
println!("READY");
|
||||||
|
loop {
|
||||||
|
let req = rx.recv().unwrap();
|
||||||
|
if let Some(rest) = req.text.strip_prefix("via-relay:") {
|
||||||
|
let fwd = Req {
|
||||||
|
text: format!("relayed:{rest}"),
|
||||||
|
reply_to: req.reply_to,
|
||||||
|
};
|
||||||
|
remote::send(RemoteName::new("relay", RELAY), fwd).unwrap();
|
||||||
|
println!("FORWARDED");
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
println!("REQ {}", req.text);
|
||||||
|
send_to_remote(req.reply_to, Reply(format!("echo:{}", req.text))).unwrap();
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Relay: exposes RELAY; bounces every Req straight back to the server's
|
||||||
|
/// ECHO, untouched. The client's pid inside it now crosses relay→server.
|
||||||
|
fn role_relay() {
|
||||||
|
let server_addr = std::env::var("SMARM_SERVER_ADDR").expect("SMARM_SERVER_ADDR");
|
||||||
|
smarm::run(move || {
|
||||||
|
let cluster = start(cfg("relay", vec![("server".into(), server_addr)])).expect("binds");
|
||||||
|
println!("LISTENING {}", cluster.local_addr());
|
||||||
|
let (tx, rx) = channel::<Req>();
|
||||||
|
register(RELAY, tx).unwrap();
|
||||||
|
expose(RELAY);
|
||||||
|
let ev = subscribe().unwrap();
|
||||||
|
wait_up(&ev, "server");
|
||||||
|
println!("READY");
|
||||||
|
loop {
|
||||||
|
let req = rx.recv().unwrap();
|
||||||
|
println!("RELAYING {}", req.text);
|
||||||
|
remote::send(RemoteName::new("server", ECHO), req).unwrap();
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Client: connects to server, installs a Reply inbox on its own pid,
|
||||||
|
/// declares it accepts `Reply` (`expose_type` — the RFC's one kept piece of
|
||||||
|
/// ceremony: nothing is remotely deliverable by default), sends a Req with
|
||||||
|
/// `reply_to = my pid` (auto-serialized), awaits the reply.
|
||||||
|
fn role_client() {
|
||||||
|
let server_addr = std::env::var("SMARM_SERVER_ADDR").expect("SMARM_SERVER_ADDR");
|
||||||
|
let via_relay = std::env::var("SMARM_VIA_RELAY").is_ok();
|
||||||
|
smarm::run(move || {
|
||||||
|
let _cluster = start(cfg("client", vec![("server".into(), server_addr)])).expect("binds");
|
||||||
|
let ev = subscribe().unwrap();
|
||||||
|
wait_up(&ev, "server");
|
||||||
|
println!("MEMBER-UP server");
|
||||||
|
|
||||||
|
let (tx, rx) = channel::<Reply>();
|
||||||
|
let me: Pid<Replier> = install::<Replier>(tx);
|
||||||
|
// The one deliberate line: a pid-targeted inbound is deliverable only
|
||||||
|
// for types this node has said it accepts (RFC §4, the safety).
|
||||||
|
smarm::cluster::expose::expose_type::<Reply>();
|
||||||
|
let text = if via_relay { "via-relay:ping" } else { "ping" };
|
||||||
|
remote::send(
|
||||||
|
RemoteName::new("server", ECHO),
|
||||||
|
Req {
|
||||||
|
text: text.into(),
|
||||||
|
reply_to: RemotePid::from_local(me).expect("identity set"),
|
||||||
|
},
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
println!("SENT");
|
||||||
|
let Reply(s) = rx.recv().unwrap();
|
||||||
|
println!("REPLY {s}");
|
||||||
|
loop {
|
||||||
|
smarm::sleep(Duration::from_secs(3600));
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The gate: cross-node call/reply with no ceremony.
|
||||||
|
#[test]
|
||||||
|
fn cross_node_call_reply_no_ceremony() {
|
||||||
|
maybe_child(ROLES);
|
||||||
|
let mut server = spawn_node("server", &[]);
|
||||||
|
let saddr = server.wait_listening();
|
||||||
|
server.wait_line("READY", |l| l == "READY");
|
||||||
|
let mut client = spawn_node("client", &[("SMARM_SERVER_ADDR", &saddr)]);
|
||||||
|
client.wait_line("SENT", |l| l == "SENT");
|
||||||
|
server.wait_line("REQ ping", |l| l == "REQ ping");
|
||||||
|
client.wait_line("REPLY echo:ping", |l| l == "REPLY echo:ping");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The client's pid, round-tripped through a third node, still delivers.
|
||||||
|
#[test]
|
||||||
|
fn pid_roundtrips_through_third_node() {
|
||||||
|
maybe_child(ROLES);
|
||||||
|
// Relay needs the server address; server needs the relay address —
|
||||||
|
// pre-reserve the relay port (same accepted micro-window as cluster_mesh).
|
||||||
|
let relay_addr = {
|
||||||
|
let l = std::net::TcpListener::bind("127.0.0.1:0").unwrap();
|
||||||
|
l.local_addr().unwrap().to_string()
|
||||||
|
};
|
||||||
|
let mut server = spawn_node("server", &[("SMARM_RELAY_ADDR", &relay_addr)]);
|
||||||
|
let saddr = server.wait_listening();
|
||||||
|
server.wait_line("READY", |l| l == "READY");
|
||||||
|
let mut relay = spawn_node(
|
||||||
|
"relay",
|
||||||
|
&[
|
||||||
|
("SMARM_SERVER_ADDR", &saddr),
|
||||||
|
("SMARM_LISTEN_ADDR", &relay_addr),
|
||||||
|
],
|
||||||
|
);
|
||||||
|
let _ = relay.wait_listening();
|
||||||
|
relay.wait_line("READY", |l| l == "READY");
|
||||||
|
let mut client = spawn_node(
|
||||||
|
"client",
|
||||||
|
&[("SMARM_SERVER_ADDR", &saddr), ("SMARM_VIA_RELAY", "1")],
|
||||||
|
);
|
||||||
|
client.wait_line("SENT", |l| l == "SENT");
|
||||||
|
server.wait_line("FORWARDED", |l| l == "FORWARDED");
|
||||||
|
relay.wait_line("RELAYING", |l| l.starts_with("RELAYING"));
|
||||||
|
server.wait_line("REQ relayed:ping", |l| l == "REQ relayed:ping");
|
||||||
|
client.wait_line("REPLY echo:relayed:ping", |l| {
|
||||||
|
l == "REPLY echo:relayed:ping"
|
||||||
|
});
|
||||||
|
}
|
||||||
@@ -0,0 +1,221 @@
|
|||||||
|
//! RFC 010 c9 — remote `Name` sends: the outbound seam and the single
|
||||||
|
//! inbound name-resolution seam, cross-process.
|
||||||
|
//!
|
||||||
|
//! Two node processes each run the integrated `cluster::start`. The
|
||||||
|
//! *receiver* registers a `String` inbox under a name and exposes it (and
|
||||||
|
//! registers a second name it does NOT expose); the *sender* waits for
|
||||||
|
//! `node_up`, then sends. Facts cross as stdout lines: `LISTENING <addr>`,
|
||||||
|
//! `MEMBER-UP <name>`, `GOT <payload>`, `SEND-RESULT <case> <verdict>`.
|
||||||
|
//! Roles park forever afterwards (retractable-state trap); the parent
|
||||||
|
//! SIGKILLs via `Drop`.
|
||||||
|
//!
|
||||||
|
//! What is asserted at each end (roadmap-binding):
|
||||||
|
//! - cross-node name-send delivers the payload;
|
||||||
|
//! - an unexposed name is unreachable — the receiver's inbox stays empty
|
||||||
|
//! even though the name IS registered locally;
|
||||||
|
//! - a wrong type hash is a decode failure at the receiver, never a
|
||||||
|
//! misroute — the `String` inbox does not see a `u64` delivered under a
|
||||||
|
//! made-up hash, nor a `u64` under `u64`'s hash;
|
||||||
|
//! - a send to a disconnected (never-connected) node fails locally with
|
||||||
|
//! `NotConnected`, and `Ok(())` means only "handed to the transport".
|
||||||
|
//!
|
||||||
|
//! Timing note for the "stays empty" assertions: they are proven by
|
||||||
|
//! ORDERING, not by waiting — the sender emits the negative-case frames
|
||||||
|
//! BEFORE the positive one on the same connection (in-order stream), so when
|
||||||
|
//! the receiver has seen the positive payload, the negatives have already
|
||||||
|
//! been processed and refused. No sleep-and-hope.
|
||||||
|
#![cfg(feature = "cluster")]
|
||||||
|
|
||||||
|
mod common;
|
||||||
|
|
||||||
|
use common::{maybe_child, spawn_node, Node};
|
||||||
|
use smarm::cluster::envelope::NodeMeta;
|
||||||
|
use smarm::cluster::expose::expose;
|
||||||
|
use smarm::cluster::membership::{subscribe, NodeEvent};
|
||||||
|
use smarm::cluster::remote::{send_remote_raw, RemoteName, RemoteSendError};
|
||||||
|
use smarm::cluster::{start, Config, StaticSeeds, Timing};
|
||||||
|
use smarm::{channel, register, Name};
|
||||||
|
use std::time::Duration;
|
||||||
|
|
||||||
|
const ROLES: &[(&str, fn())] = &[("receiver", role_receiver), ("sender", role_sender)];
|
||||||
|
|
||||||
|
const INBOX: Name<String> = Name::new("c9.inbox");
|
||||||
|
const HIDDEN: Name<String> = Name::new("c9.hidden");
|
||||||
|
|
||||||
|
fn base_config(name: &str, seeds: Vec<(String, String)>) -> Config {
|
||||||
|
Config {
|
||||||
|
node_name: name.to_string(),
|
||||||
|
meta: NodeMeta {
|
||||||
|
role: "c9".to_string(),
|
||||||
|
region: "local".to_string(),
|
||||||
|
},
|
||||||
|
listen_addr: "127.0.0.1:0".to_string(),
|
||||||
|
strategy: Box::new(StaticSeeds::new(seeds)),
|
||||||
|
timing: Timing::default(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Receiver: register + expose INBOX; register HIDDEN unexposed **in a
|
||||||
|
/// separate actor** (one actor holds one channel per message type — a
|
||||||
|
/// second `register` of the same `M` on one actor silently replaces the
|
||||||
|
/// first, closing it); print every payload that lands in either.
|
||||||
|
fn role_receiver() {
|
||||||
|
smarm::run(|| {
|
||||||
|
let cluster = start(base_config("recv", vec![])).expect("listener binds");
|
||||||
|
println!("LISTENING {}", cluster.local_addr());
|
||||||
|
|
||||||
|
// HIDDEN's holder: its own actor, so its String channel does not
|
||||||
|
// displace INBOX's on the root actor.
|
||||||
|
let (hidden_ready_tx, hidden_ready_rx) = channel::<()>();
|
||||||
|
smarm::spawn(move || {
|
||||||
|
let (hid_tx, hid_rx) = channel::<String>();
|
||||||
|
register(HIDDEN, hid_tx).unwrap();
|
||||||
|
hidden_ready_tx.send(()).unwrap();
|
||||||
|
loop {
|
||||||
|
match hid_rx.recv() {
|
||||||
|
Ok(s) => println!("GOT-HIDDEN {s}"),
|
||||||
|
Err(_) => break,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
});
|
||||||
|
hidden_ready_rx.recv().unwrap();
|
||||||
|
|
||||||
|
let (in_tx, in_rx) = channel::<String>();
|
||||||
|
register(INBOX, in_tx).unwrap();
|
||||||
|
expose(INBOX);
|
||||||
|
println!("READY");
|
||||||
|
loop {
|
||||||
|
match in_rx.recv() {
|
||||||
|
Ok(s) => println!("GOT {s}"),
|
||||||
|
Err(_) => break,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
loop {
|
||||||
|
smarm::sleep(Duration::from_secs(3600));
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Sender: connect to recv, wait for node_up, then in this ORDER on the one
|
||||||
|
/// connection: hidden-name send, wrong-hash sends (two flavours), then the
|
||||||
|
/// positive send. Plus a send to a node that is not connected at all.
|
||||||
|
fn role_sender() {
|
||||||
|
let recv_addr = std::env::var("SMARM_RECV_ADDR").expect("SMARM_RECV_ADDR");
|
||||||
|
smarm::run(move || {
|
||||||
|
let _cluster = start(base_config("send", vec![("recv".to_string(), recv_addr)]))
|
||||||
|
.expect("listener binds");
|
||||||
|
let events = subscribe().expect("manager is up");
|
||||||
|
loop {
|
||||||
|
match events.rx.recv() {
|
||||||
|
Ok(NodeEvent::NodeUp(info)) if info.name == "recv" => break,
|
||||||
|
Ok(_) => continue,
|
||||||
|
Err(_) => panic!("manager gone"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
println!("MEMBER-UP recv");
|
||||||
|
|
||||||
|
// Not connected: purely local knowledge, no frame leaves.
|
||||||
|
let ghost: RemoteName<String> = RemoteName::new("nowhere", INBOX);
|
||||||
|
let r = smarm::cluster::remote::send(ghost, "lost".to_string());
|
||||||
|
println!(
|
||||||
|
"SEND-RESULT not-connected {}",
|
||||||
|
match r {
|
||||||
|
Err(RemoteSendError::NotConnected(_)) => "NotConnected",
|
||||||
|
Ok(()) => "Ok",
|
||||||
|
Err(_) => "OtherErr",
|
||||||
|
}
|
||||||
|
);
|
||||||
|
|
||||||
|
// Unexposed name at the peer: the frame goes (local knowledge can't
|
||||||
|
// know the peer's exposed set) and the peer refuses it.
|
||||||
|
let hidden: RemoteName<String> = RemoteName::new("recv", HIDDEN);
|
||||||
|
let r = smarm::cluster::remote::send(hidden, "should not land".to_string());
|
||||||
|
println!(
|
||||||
|
"SEND-RESULT hidden {}",
|
||||||
|
if r.is_ok() { "Ok" } else { "Err" }
|
||||||
|
);
|
||||||
|
|
||||||
|
// Wrong hash, two flavours: (a) a u64 payload under a made-up hash
|
||||||
|
// (unknown type at the peer); (b) a u64 payload under u64's real
|
||||||
|
// hash against a String-typed name (decoder known, wrong channel).
|
||||||
|
// Both are raw sends — the typed API cannot express them, by design.
|
||||||
|
let bogus = 0xdead_beef_u64;
|
||||||
|
let r = send_remote_raw(
|
||||||
|
"recv",
|
||||||
|
"c9.inbox",
|
||||||
|
bogus,
|
||||||
|
&smarm::cluster::envelope::encode_payload(&7u64).unwrap(),
|
||||||
|
);
|
||||||
|
println!(
|
||||||
|
"SEND-RESULT wrong-hash-unknown {}",
|
||||||
|
if r.is_ok() { "Ok" } else { "Err" }
|
||||||
|
);
|
||||||
|
let r = send_remote_raw(
|
||||||
|
"recv",
|
||||||
|
"c9.inbox",
|
||||||
|
smarm::cluster::expose::type_hash::<u64>(),
|
||||||
|
&smarm::cluster::envelope::encode_payload(&7u64).unwrap(),
|
||||||
|
);
|
||||||
|
println!(
|
||||||
|
"SEND-RESULT wrong-hash-known {}",
|
||||||
|
if r.is_ok() { "Ok" } else { "Err" }
|
||||||
|
);
|
||||||
|
|
||||||
|
// Positive: last on the stream, so its arrival proves the negatives
|
||||||
|
// were already processed.
|
||||||
|
let inbox: RemoteName<String> = RemoteName::new("recv", INBOX);
|
||||||
|
let r = smarm::cluster::remote::send(inbox, "hello from send".to_string());
|
||||||
|
println!(
|
||||||
|
"SEND-RESULT positive {}",
|
||||||
|
if r.is_ok() { "Ok" } else { "Err" }
|
||||||
|
);
|
||||||
|
|
||||||
|
loop {
|
||||||
|
smarm::sleep(Duration::from_secs(3600));
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
fn wait_send_result(node: &mut Node, case: &str) -> String {
|
||||||
|
let prefix = format!("SEND-RESULT {case} ");
|
||||||
|
let line = node.wait_line(&prefix, |l| l.starts_with(&prefix));
|
||||||
|
line[prefix.len()..].to_string()
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn remote_name_send_delivers_and_refusals_never_misroute() {
|
||||||
|
maybe_child(ROLES);
|
||||||
|
|
||||||
|
let mut recv = spawn_node("receiver", &[]);
|
||||||
|
let addr = recv.wait_listening();
|
||||||
|
recv.wait_line("READY", |l| l == "READY");
|
||||||
|
|
||||||
|
let mut send = spawn_node("sender", &[("SMARM_RECV_ADDR", &addr)]);
|
||||||
|
send.wait_line("MEMBER-UP recv", |l| l == "MEMBER-UP recv");
|
||||||
|
|
||||||
|
// Local-knowledge-only failure for an unknown node.
|
||||||
|
assert_eq!(wait_send_result(&mut send, "not-connected"), "NotConnected");
|
||||||
|
// Every frame-bearing send is Ok — Ok means "handed to the transport",
|
||||||
|
// nothing about what the peer does with it (RFC §3, documented here).
|
||||||
|
assert_eq!(wait_send_result(&mut send, "hidden"), "Ok");
|
||||||
|
assert_eq!(wait_send_result(&mut send, "wrong-hash-unknown"), "Ok");
|
||||||
|
assert_eq!(wait_send_result(&mut send, "wrong-hash-known"), "Ok");
|
||||||
|
assert_eq!(wait_send_result(&mut send, "positive"), "Ok");
|
||||||
|
|
||||||
|
// The positive payload lands...
|
||||||
|
recv.wait_line("GOT hello from send", |l| l == "GOT hello from send");
|
||||||
|
// ...and, by stream ordering, every negative before it was refused: no
|
||||||
|
// GOT for the wrong-hash frames, no GOT-HIDDEN at all. The transcript
|
||||||
|
// up to this point is the proof.
|
||||||
|
let transcript = recv.transcript();
|
||||||
|
let gots: Vec<&str> = transcript
|
||||||
|
.iter()
|
||||||
|
.map(|s| s.as_str())
|
||||||
|
.filter(|l| l.starts_with("GOT"))
|
||||||
|
.collect();
|
||||||
|
assert_eq!(
|
||||||
|
gots,
|
||||||
|
["GOT hello from send"],
|
||||||
|
"exactly one delivery, the exposed one"
|
||||||
|
);
|
||||||
|
}
|
||||||
@@ -0,0 +1,272 @@
|
|||||||
|
//! RFC 010 c3 — transport conformance suite, run against both shipped impls
|
||||||
|
//! (TCP and in-memory loopback), plus impl-specific cases.
|
||||||
|
//!
|
||||||
|
//! Shared suite (roadmap): frame roundtrips through the framed codec, framing
|
||||||
|
//! across a split write, coalesced frames in one write, peer-close mid-frame
|
||||||
|
//! (must error, not EOF), clean close at a frame boundary (EOF as `Ok(None)`).
|
||||||
|
//!
|
||||||
|
//! The TCP impl parks the calling actor, so its runs live inside `smarm::run`;
|
||||||
|
//! loopback blocks the OS thread and runs as plain tests.
|
||||||
|
#![cfg(feature = "cluster")]
|
||||||
|
|
||||||
|
use smarm::cluster::envelope::Frame;
|
||||||
|
use smarm::cluster::transport::loopback::LoopbackTransport;
|
||||||
|
use smarm::cluster::transport::tcp::TcpTransport;
|
||||||
|
use smarm::cluster::transport::{Conn, FramedConn, RecvError, Transport};
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Helpers
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
/// Listener + dial + accept against one transport, both conns returned.
|
||||||
|
/// Relies on dial not requiring a concurrent accept (TCP backlog / loopback
|
||||||
|
/// queue), so a single thread or actor can hold both ends.
|
||||||
|
fn pair(t: &dyn Transport, addr: &str) -> (Box<dyn Conn>, Box<dyn Conn>) {
|
||||||
|
let mut l = t.listen(addr).unwrap();
|
||||||
|
let a = t.dial(&l.local_addr()).unwrap();
|
||||||
|
let b = l.accept().unwrap();
|
||||||
|
(a, b)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn frames() -> Vec<Frame> {
|
||||||
|
vec![
|
||||||
|
Frame::Heartbeat,
|
||||||
|
Frame::Send {
|
||||||
|
index: 42,
|
||||||
|
generation: 3,
|
||||||
|
type_hash: 0x1234_5678_9ABC_DEF0,
|
||||||
|
payload: vec![1, 2, 3, 4, 5],
|
||||||
|
},
|
||||||
|
Frame::SendNamed {
|
||||||
|
name: "the_counter".into(),
|
||||||
|
type_hash: 0xFFFF_0000_FFFF_0000,
|
||||||
|
payload: vec![],
|
||||||
|
},
|
||||||
|
Frame::Demonitor { monitor_id: 77 },
|
||||||
|
]
|
||||||
|
}
|
||||||
|
|
||||||
|
fn encode(f: &Frame) -> Vec<u8> {
|
||||||
|
let mut out = Vec::new();
|
||||||
|
f.encode(&mut out).unwrap();
|
||||||
|
out
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Shared conformance suite — generic over an established pair
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
fn suite_roundtrip(a: Box<dyn Conn>, b: Box<dyn Conn>) {
|
||||||
|
let mut fa = FramedConn::new(a);
|
||||||
|
let mut fb = FramedConn::new(b);
|
||||||
|
// a -> b, then b -> a: both directions carry every frame shape.
|
||||||
|
for f in frames() {
|
||||||
|
fa.send(&f).unwrap();
|
||||||
|
assert_eq!(fb.recv().unwrap().unwrap(), f);
|
||||||
|
}
|
||||||
|
for f in frames() {
|
||||||
|
fb.send(&f).unwrap();
|
||||||
|
assert_eq!(fa.recv().unwrap().unwrap(), f);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn suite_split_write(mut a: Box<dyn Conn>, b: Box<dyn Conn>) {
|
||||||
|
let f = Frame::Send {
|
||||||
|
index: 7,
|
||||||
|
generation: 1,
|
||||||
|
type_hash: 0xAB,
|
||||||
|
payload: vec![9; 64],
|
||||||
|
};
|
||||||
|
let bytes = encode(&f);
|
||||||
|
// Split inside the length prefix, then inside the body: the reader must
|
||||||
|
// reassemble regardless of where the boundary falls.
|
||||||
|
a.write_all(&bytes[..2]).unwrap();
|
||||||
|
a.write_all(&bytes[2..10]).unwrap();
|
||||||
|
a.write_all(&bytes[10..]).unwrap();
|
||||||
|
let mut fb = FramedConn::new(b);
|
||||||
|
assert_eq!(fb.recv().unwrap().unwrap(), f);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn suite_coalesced(mut a: Box<dyn Conn>, b: Box<dyn Conn>) {
|
||||||
|
let f1 = Frame::Heartbeat;
|
||||||
|
let f2 = Frame::Demonitor { monitor_id: 5 };
|
||||||
|
let mut bytes = encode(&f1);
|
||||||
|
bytes.extend_from_slice(&encode(&f2));
|
||||||
|
a.write_all(&bytes).unwrap();
|
||||||
|
let mut fb = FramedConn::new(b);
|
||||||
|
assert_eq!(fb.recv().unwrap().unwrap(), f1);
|
||||||
|
assert_eq!(fb.recv().unwrap().unwrap(), f2);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn suite_close_mid_frame(mut a: Box<dyn Conn>, b: Box<dyn Conn>) {
|
||||||
|
let bytes = encode(&Frame::Send {
|
||||||
|
index: 1,
|
||||||
|
generation: 1,
|
||||||
|
type_hash: 1,
|
||||||
|
payload: vec![0; 128],
|
||||||
|
});
|
||||||
|
a.write_all(&bytes[..bytes.len() / 2]).unwrap();
|
||||||
|
a.close();
|
||||||
|
let mut fb = FramedConn::new(b);
|
||||||
|
match fb.recv() {
|
||||||
|
Err(RecvError::TruncatedByPeer) => {}
|
||||||
|
other => panic!("expected TruncatedByPeer, got {other:?}"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn suite_clean_close(mut a: Box<dyn Conn>, b: Box<dyn Conn>) {
|
||||||
|
let f = Frame::Heartbeat;
|
||||||
|
a.write_all(&encode(&f)).unwrap();
|
||||||
|
a.close();
|
||||||
|
let mut fb = FramedConn::new(b);
|
||||||
|
// The buffered frame is still delivered, then EOF at the boundary.
|
||||||
|
assert_eq!(fb.recv().unwrap().unwrap(), f);
|
||||||
|
assert!(fb.recv().unwrap().is_none());
|
||||||
|
}
|
||||||
|
|
||||||
|
fn run_suite(t: &dyn Transport, addr: &str) {
|
||||||
|
let (a, b) = pair(t, addr);
|
||||||
|
suite_roundtrip(a, b);
|
||||||
|
let (a, b) = pair(t, addr);
|
||||||
|
suite_split_write(a, b);
|
||||||
|
let (a, b) = pair(t, addr);
|
||||||
|
suite_coalesced(a, b);
|
||||||
|
let (a, b) = pair(t, addr);
|
||||||
|
suite_close_mid_frame(a, b);
|
||||||
|
let (a, b) = pair(t, addr);
|
||||||
|
suite_clean_close(a, b);
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Loopback — plain tests, no runtime
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn loopback_conformance() {
|
||||||
|
// Fresh transport per pair() call is fine, but one instance must also
|
||||||
|
// support sequential re-listen on distinct addresses.
|
||||||
|
let t = LoopbackTransport::default();
|
||||||
|
run_suite(&t, "alpha");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn loopback_dial_unknown_addr_refused() {
|
||||||
|
let t = LoopbackTransport::default();
|
||||||
|
let err = t.dial("nobody-home").unwrap_err();
|
||||||
|
assert_eq!(err.kind(), std::io::ErrorKind::ConnectionRefused);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn loopback_addr_in_use() {
|
||||||
|
let t = LoopbackTransport::default();
|
||||||
|
let _l = t.listen("alpha").unwrap();
|
||||||
|
let err = t.listen("alpha").unwrap_err();
|
||||||
|
assert_eq!(err.kind(), std::io::ErrorKind::AddrInUse);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn loopback_listener_drop_frees_addr_and_refuses_dial() {
|
||||||
|
let t = LoopbackTransport::default();
|
||||||
|
let l = t.listen("alpha").unwrap();
|
||||||
|
drop(l);
|
||||||
|
let err = t.dial("alpha").unwrap_err();
|
||||||
|
assert_eq!(err.kind(), std::io::ErrorKind::ConnectionRefused);
|
||||||
|
// Address is reusable after the listener is gone.
|
||||||
|
let _l2 = t.listen("alpha").unwrap();
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn loopback_write_after_peer_close_broken_pipe() {
|
||||||
|
let t = LoopbackTransport::default();
|
||||||
|
let (mut a, mut b) = pair(&t, "alpha");
|
||||||
|
b.close();
|
||||||
|
let err = a.write_all(&[1, 2, 3]).unwrap_err();
|
||||||
|
assert_eq!(err.kind(), std::io::ErrorKind::BrokenPipe);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn loopback_cross_thread_blocking_read() {
|
||||||
|
// Reader blocks on an empty pipe until the writer thread delivers.
|
||||||
|
let t = LoopbackTransport::default();
|
||||||
|
let (a, b) = pair(&t, "alpha");
|
||||||
|
let mut fb = FramedConn::new(b);
|
||||||
|
let writer = std::thread::spawn(move || {
|
||||||
|
let mut a = a;
|
||||||
|
std::thread::sleep(std::time::Duration::from_millis(30));
|
||||||
|
a.write_all(&encode(&Frame::Heartbeat)).unwrap();
|
||||||
|
});
|
||||||
|
assert_eq!(fb.recv().unwrap().unwrap(), Frame::Heartbeat);
|
||||||
|
writer.join().unwrap();
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// TCP — inside the runtime (read/write park the calling actor)
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn tcp_conformance() {
|
||||||
|
smarm::run(|| {
|
||||||
|
run_suite(&TcpTransport, "127.0.0.1:0");
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn tcp_dial_refused() {
|
||||||
|
smarm::run(|| {
|
||||||
|
// Bind to an OS-assigned port, learn it, close the listener, dial it.
|
||||||
|
let addr = {
|
||||||
|
let l = TcpTransport.listen("127.0.0.1:0").unwrap();
|
||||||
|
l.local_addr()
|
||||||
|
};
|
||||||
|
let err = TcpTransport.dial(&addr).unwrap_err();
|
||||||
|
assert_eq!(err.kind(), std::io::ErrorKind::ConnectionRefused);
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn tcp_bad_addr_rejected_without_resolution() {
|
||||||
|
// Addresses are opaque pre-resolved strings; the c9 seam resolves names.
|
||||||
|
// A hostname is therefore invalid input here, not something to resolve.
|
||||||
|
let err = TcpTransport.dial("localhost:1234").unwrap_err();
|
||||||
|
assert_eq!(err.kind(), std::io::ErrorKind::InvalidInput);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn tcp_local_addr_reports_real_port() {
|
||||||
|
let l = TcpTransport.listen("127.0.0.1:0").unwrap();
|
||||||
|
let addr = l.local_addr();
|
||||||
|
let port: u16 = addr.rsplit(':').next().unwrap().parse().unwrap();
|
||||||
|
assert_ne!(port, 0);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn tcp_big_frame_across_socket_buffers() {
|
||||||
|
// A payload far beyond socket buffer sizes forces genuine fragmentation
|
||||||
|
// and write backpressure: writer and reader must run concurrently.
|
||||||
|
smarm::run(|| {
|
||||||
|
let (tx, rx) = smarm::channel::<Frame>();
|
||||||
|
let payload = vec![0xA5u8; 4 * 1024 * 1024];
|
||||||
|
let f = Frame::Send {
|
||||||
|
index: 9,
|
||||||
|
generation: 2,
|
||||||
|
type_hash: 0xC0FFEE,
|
||||||
|
payload,
|
||||||
|
};
|
||||||
|
let mut l = TcpTransport.listen("127.0.0.1:0").unwrap();
|
||||||
|
let addr = l.local_addr();
|
||||||
|
let fw = f.clone();
|
||||||
|
let writer = smarm::spawn(move || {
|
||||||
|
let mut fa = FramedConn::new(TcpTransport.dial(&addr).unwrap());
|
||||||
|
fa.send(&fw).unwrap();
|
||||||
|
});
|
||||||
|
let reader = smarm::spawn(move || {
|
||||||
|
let mut fb = FramedConn::new(l.accept().unwrap());
|
||||||
|
let got = fb.recv().unwrap().unwrap();
|
||||||
|
tx.send(got).unwrap();
|
||||||
|
});
|
||||||
|
let got = rx.recv().unwrap();
|
||||||
|
assert_eq!(got, f);
|
||||||
|
writer.join().unwrap();
|
||||||
|
reader.join().unwrap();
|
||||||
|
});
|
||||||
|
}
|
||||||
@@ -0,0 +1,108 @@
|
|||||||
|
//! RFC 010 c4 — two-node harness smoke tests.
|
||||||
|
//!
|
||||||
|
//! Roadmap: "spawn two, handshake-less connect, both exit clean." The
|
||||||
|
//! listener node binds port 0 and announces its concrete address; the
|
||||||
|
//! dialer connects raw (no Hello — c5 doesn't exist yet), pushes one
|
||||||
|
//! Heartbeat through the real framed codec, and closes. Assertions are on
|
||||||
|
//! protocol-visible lines only. Flake budget: see tests/common/mod.rs.
|
||||||
|
#![cfg(feature = "cluster")]
|
||||||
|
|
||||||
|
mod common;
|
||||||
|
|
||||||
|
use common::{maybe_child, spawn_node};
|
||||||
|
use smarm::cluster::envelope::Frame;
|
||||||
|
use smarm::cluster::transport::tcp::TcpTransport;
|
||||||
|
use smarm::cluster::transport::{FramedConn, Transport};
|
||||||
|
|
||||||
|
const ROLES: &[(&str, fn())] = &[
|
||||||
|
("listener", role_listener),
|
||||||
|
("dialer", role_dialer),
|
||||||
|
("hang", role_hang),
|
||||||
|
("fail", role_fail),
|
||||||
|
];
|
||||||
|
|
||||||
|
fn role_listener() {
|
||||||
|
smarm::run(|| {
|
||||||
|
let mut l = TcpTransport.listen("127.0.0.1:0").unwrap();
|
||||||
|
println!("LISTENING {}", l.local_addr());
|
||||||
|
let mut fc = FramedConn::new(l.accept().unwrap());
|
||||||
|
match fc.recv() {
|
||||||
|
Ok(Some(Frame::Heartbeat)) => println!("RECV heartbeat"),
|
||||||
|
other => {
|
||||||
|
println!("RECV unexpected: {other:?}");
|
||||||
|
std::process::exit(3);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
match fc.recv() {
|
||||||
|
Ok(None) => println!("PEER-CLOSED clean"),
|
||||||
|
other => {
|
||||||
|
println!("PEER-CLOSED unexpected: {other:?}");
|
||||||
|
std::process::exit(3);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
});
|
||||||
|
println!("EXIT ok");
|
||||||
|
}
|
||||||
|
|
||||||
|
fn role_dialer() {
|
||||||
|
let addr = std::env::var("SMARM_PEER_ADDR").expect("SMARM_PEER_ADDR not set");
|
||||||
|
smarm::run(move || {
|
||||||
|
let mut fc = FramedConn::new(TcpTransport.dial(&addr).unwrap());
|
||||||
|
fc.send(&Frame::Heartbeat).unwrap();
|
||||||
|
fc.close();
|
||||||
|
println!("SENT heartbeat");
|
||||||
|
});
|
||||||
|
println!("EXIT ok");
|
||||||
|
}
|
||||||
|
|
||||||
|
fn role_hang() {
|
||||||
|
println!("HANGING");
|
||||||
|
loop {
|
||||||
|
std::thread::sleep(std::time::Duration::from_secs(3600));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn role_fail() {
|
||||||
|
std::process::exit(7);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The roadmap smoke test: two real processes, raw transport connect, one
|
||||||
|
/// frame across, clean close observed on both sides, both exit 0.
|
||||||
|
#[test]
|
||||||
|
fn two_nodes_connect_and_exit_clean() {
|
||||||
|
maybe_child(ROLES);
|
||||||
|
let mut listener = spawn_node("listener", &[]);
|
||||||
|
let addr = listener.wait_listening();
|
||||||
|
let mut dialer = spawn_node("dialer", &[("SMARM_PEER_ADDR", &addr)]);
|
||||||
|
dialer.wait_line("SENT heartbeat", |l| l == "SENT heartbeat");
|
||||||
|
listener.wait_line("RECV heartbeat", |l| l == "RECV heartbeat");
|
||||||
|
listener.wait_line("clean peer close", |l| l == "PEER-CLOSED clean");
|
||||||
|
dialer.wait_exit_ok();
|
||||||
|
listener.wait_exit_ok();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Reap guarantee: dropping a Node kills a hung child — no orphan survives
|
||||||
|
/// a panicking test.
|
||||||
|
#[test]
|
||||||
|
fn drop_reaps_hung_node() {
|
||||||
|
maybe_child(ROLES);
|
||||||
|
let mut node = spawn_node("hang", &[]);
|
||||||
|
node.wait_line("HANGING", |l| l == "HANGING");
|
||||||
|
let pid = node.pid().expect("live child has a pid") as libc::pid_t;
|
||||||
|
drop(node);
|
||||||
|
// After Drop's kill+wait the pid is fully reaped: signalling it fails
|
||||||
|
// with ESRCH (pid-reuse in this instant is not a realistic race).
|
||||||
|
let rc = unsafe { libc::kill(pid, 0) };
|
||||||
|
assert_eq!(rc, -1, "process still signallable after Drop");
|
||||||
|
let errno = std::io::Error::last_os_error().raw_os_error();
|
||||||
|
assert_eq!(errno, Some(libc::ESRCH), "expected ESRCH, got {errno:?}");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Nonzero child exits surface as statuses, not hangs or panics.
|
||||||
|
#[test]
|
||||||
|
fn nonzero_exit_is_reported() {
|
||||||
|
maybe_child(ROLES);
|
||||||
|
let mut node = spawn_node("fail", &[]);
|
||||||
|
let status = node.wait_exit();
|
||||||
|
assert_eq!(status.code(), Some(7));
|
||||||
|
}
|
||||||
@@ -0,0 +1,250 @@
|
|||||||
|
//! RFC 010 c4 — subprocess multi-node test harness.
|
||||||
|
//!
|
||||||
|
//! The runtime is a process singleton, so two real nodes means two
|
||||||
|
//! processes. This harness re-execs the *current test binary* as node
|
||||||
|
//! processes (precedent: tests/stack_diag.rs), tails their output live,
|
||||||
|
//! waits on protocol-visible lines, and reaps reliably no matter how the
|
||||||
|
//! test dies.
|
||||||
|
//!
|
||||||
|
//! Usage, per test file:
|
||||||
|
//!
|
||||||
|
//! - Declare roles as plain `fn()`s. A role prints protocol-visible facts
|
||||||
|
//! as single lines (Rust's piped stdout is line-buffered, so `println!`
|
||||||
|
//! is enough) and exits.
|
||||||
|
//! - **Every** `#[test]` in the file starts with
|
||||||
|
//! [`maybe_child`]`(ROLES)` — in the child re-exec, whichever test
|
||||||
|
//! libtest runs first performs the role and exits before the rest of the
|
||||||
|
//! suite runs (children are spawned with `--test-threads=1 --quiet`).
|
||||||
|
//! - The parent side spawns nodes with [`spawn_node`], waits on lines with
|
||||||
|
//! [`Node::wait_line`], and on exits with [`Node::wait_exit`].
|
||||||
|
//!
|
||||||
|
//! Port assignment: children bind port 0 and *report* the concrete address
|
||||||
|
//! (e.g. `LISTENING 127.0.0.1:41733`) rather than the parent pre-picking a
|
||||||
|
//! port — no bind/steal race by construction.
|
||||||
|
//!
|
||||||
|
//! Reaping: [`Node`]'s `Drop` SIGKILLs and `wait(2)`s the child, so a
|
||||||
|
//! panicking test (including a `wait_line` timeout) leaves no orphan and
|
||||||
|
//! no zombie. Tail threads exit on pipe EOF.
|
||||||
|
//!
|
||||||
|
//! Flake budget (explicit, per roadmap): every wait is bounded by
|
||||||
|
//! [`WAIT`] (10 s) against a typical cost of well under 1 s; the smoke
|
||||||
|
//! suite ran 10/10 clean at authoring time. Treat >1 failure in 100 runs
|
||||||
|
//! as a harness or runtime regression, not weather. On timeout the panic
|
||||||
|
//! message carries the node's full transcript so far.
|
||||||
|
|
||||||
|
#![allow(dead_code)] // Reusable surface: later phases use more of it than any one file.
|
||||||
|
|
||||||
|
use std::env;
|
||||||
|
use std::io::{BufRead, BufReader};
|
||||||
|
use std::process::{Child, Command, ExitStatus, Stdio};
|
||||||
|
use std::sync::mpsc::{Receiver, RecvTimeoutError};
|
||||||
|
use std::time::{Duration, Instant};
|
||||||
|
|
||||||
|
/// Env var selecting the child role in a re-exec.
|
||||||
|
const ROLE_ENV: &str = "SMARM_TWO_NODE_ROLE";
|
||||||
|
|
||||||
|
/// Upper bound for every wait in the harness. See the flake budget above.
|
||||||
|
pub const WAIT: Duration = Duration::from_secs(10);
|
||||||
|
|
||||||
|
/// In the child re-exec: run the matching role and exit. In the parent (no
|
||||||
|
/// role env set): return immediately. Call this first in every `#[test]` of
|
||||||
|
/// any file using the harness, passing the file's full role table.
|
||||||
|
pub fn maybe_child(roles: &[(&str, fn())]) {
|
||||||
|
let role = match env::var(ROLE_ENV) {
|
||||||
|
Ok(r) => r,
|
||||||
|
Err(_) => return,
|
||||||
|
};
|
||||||
|
for (name, f) in roles {
|
||||||
|
if *name == role {
|
||||||
|
f();
|
||||||
|
std::process::exit(0);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
eprintln!("two_node harness: unknown role {role:?}");
|
||||||
|
std::process::exit(2);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One spawned node process with live-tailed output.
|
||||||
|
pub struct Node {
|
||||||
|
/// Role name, for panic messages.
|
||||||
|
pub role: String,
|
||||||
|
child: Option<Child>,
|
||||||
|
stdout_rx: Receiver<String>,
|
||||||
|
stderr_rx: Receiver<String>,
|
||||||
|
/// Every line consumed from stdout/stderr so far, for failure dumps.
|
||||||
|
transcript: Vec<String>,
|
||||||
|
}
|
||||||
|
|
||||||
|
fn tail(stream: impl std::io::Read + Send + 'static, prefix: &'static str) -> Receiver<String> {
|
||||||
|
let (tx, rx) = std::sync::mpsc::channel();
|
||||||
|
std::thread::spawn(move || {
|
||||||
|
for line in BufReader::new(stream).lines() {
|
||||||
|
let line = match line {
|
||||||
|
Ok(l) => l,
|
||||||
|
Err(_) => break,
|
||||||
|
};
|
||||||
|
// Receiver gone (Node dropped): stop tailing.
|
||||||
|
if tx.send(format!("{prefix}{line}")).is_err() {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
});
|
||||||
|
rx
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Re-exec the current test binary as `role`, with any extra env vars.
|
||||||
|
pub fn spawn_node(role: &str, extra_env: &[(&str, &str)]) -> Node {
|
||||||
|
let exe = env::current_exe().expect("current_exe");
|
||||||
|
let mut cmd = Command::new(exe);
|
||||||
|
cmd.env(ROLE_ENV, role)
|
||||||
|
// --test-threads=1: exactly one test fn starts, hits maybe_child,
|
||||||
|
// and becomes the role. --nocapture: libtest must not swallow the
|
||||||
|
// role's println! lines — the parent tails them live.
|
||||||
|
.args(["--test-threads=1", "--quiet", "--nocapture"])
|
||||||
|
.stdout(Stdio::piped())
|
||||||
|
.stderr(Stdio::piped());
|
||||||
|
for (k, v) in extra_env {
|
||||||
|
cmd.env(k, v);
|
||||||
|
}
|
||||||
|
let mut child = cmd.spawn().expect("failed to spawn node process");
|
||||||
|
let stdout_rx = tail(child.stdout.take().expect("piped stdout"), "");
|
||||||
|
let stderr_rx = tail(child.stderr.take().expect("piped stderr"), "[stderr] ");
|
||||||
|
Node {
|
||||||
|
role: role.to_string(),
|
||||||
|
child: Some(child),
|
||||||
|
stdout_rx,
|
||||||
|
stderr_rx,
|
||||||
|
transcript: Vec::new(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Node {
|
||||||
|
fn drain_stderr(&mut self) {
|
||||||
|
while let Ok(l) = self.stderr_rx.try_recv() {
|
||||||
|
self.transcript.push(l);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn dump(&self) -> String {
|
||||||
|
if self.transcript.is_empty() {
|
||||||
|
"<no output>".to_string()
|
||||||
|
} else {
|
||||||
|
self.transcript.join("\n")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Every stdout/stderr line seen so far, in arrival order. For
|
||||||
|
/// ordering-proof assertions ("by the time X arrived, Y had not").
|
||||||
|
#[allow(dead_code)]
|
||||||
|
pub fn transcript(&self) -> &[String] {
|
||||||
|
&self.transcript
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Wait until a stdout line satisfies `pred`; return it. Panics with the
|
||||||
|
/// full transcript after [`WAIT`]. `what` names the expectation in the
|
||||||
|
/// panic message.
|
||||||
|
pub fn wait_line(&mut self, what: &str, pred: impl Fn(&str) -> bool) -> String {
|
||||||
|
let deadline = Instant::now() + WAIT;
|
||||||
|
loop {
|
||||||
|
self.drain_stderr();
|
||||||
|
let left = deadline.saturating_duration_since(Instant::now());
|
||||||
|
match self.stdout_rx.recv_timeout(left) {
|
||||||
|
Ok(line) => {
|
||||||
|
self.transcript.push(line.clone());
|
||||||
|
if pred(&line) {
|
||||||
|
return line;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Err(RecvTimeoutError::Timeout) => {
|
||||||
|
// Pull in whatever stderr arrived since the last drain,
|
||||||
|
// so a role's eprintln! diagnostics survive into the dump.
|
||||||
|
self.drain_stderr();
|
||||||
|
panic!(
|
||||||
|
"node {:?}: timed out waiting for {what} after {WAIT:?}; transcript:\n{}",
|
||||||
|
self.role,
|
||||||
|
self.dump()
|
||||||
|
);
|
||||||
|
}
|
||||||
|
Err(RecvTimeoutError::Disconnected) => {
|
||||||
|
self.drain_stderr();
|
||||||
|
panic!(
|
||||||
|
"node {:?}: output closed while waiting for {what}; transcript:\n{}",
|
||||||
|
self.role,
|
||||||
|
self.dump()
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Shorthand: wait for a `LISTENING <addr>` announcement, return the addr.
|
||||||
|
pub fn wait_listening(&mut self) -> String {
|
||||||
|
let line = self.wait_line("LISTENING announcement", |l| l.starts_with("LISTENING "));
|
||||||
|
line["LISTENING ".len()..].to_string()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Wait for the process to exit; panics with the transcript on timeout.
|
||||||
|
pub fn wait_exit(&mut self) -> ExitStatus {
|
||||||
|
let deadline = Instant::now() + WAIT;
|
||||||
|
loop {
|
||||||
|
let polled = match self.child.as_mut() {
|
||||||
|
Some(c) => c.try_wait(),
|
||||||
|
None => panic!("node {:?}: already reaped", self.role),
|
||||||
|
};
|
||||||
|
match polled {
|
||||||
|
Ok(Some(status)) => {
|
||||||
|
// Drain remaining output into the transcript for dumps.
|
||||||
|
self.drain_stderr();
|
||||||
|
while let Ok(l) = self.stdout_rx.try_recv() {
|
||||||
|
self.transcript.push(l);
|
||||||
|
}
|
||||||
|
self.child = None;
|
||||||
|
return status;
|
||||||
|
}
|
||||||
|
Ok(None) => {
|
||||||
|
if Instant::now() >= deadline {
|
||||||
|
self.drain_stderr();
|
||||||
|
self.kill();
|
||||||
|
panic!(
|
||||||
|
"node {:?}: did not exit within {WAIT:?}; transcript:\n{}",
|
||||||
|
self.role,
|
||||||
|
self.dump()
|
||||||
|
);
|
||||||
|
}
|
||||||
|
std::thread::sleep(Duration::from_millis(10));
|
||||||
|
}
|
||||||
|
Err(e) => panic!("node {:?}: try_wait failed: {e}", self.role),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Wait for exit and require success, dumping the transcript otherwise.
|
||||||
|
pub fn wait_exit_ok(&mut self) {
|
||||||
|
let status = self.wait_exit();
|
||||||
|
assert!(
|
||||||
|
status.success(),
|
||||||
|
"node {:?}: exited with {status}; transcript:\n{}",
|
||||||
|
self.role,
|
||||||
|
self.dump()
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The child's OS pid, if not yet reaped.
|
||||||
|
pub fn pid(&self) -> Option<u32> {
|
||||||
|
self.child.as_ref().map(Child::id)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// SIGKILL + reap now (idempotent).
|
||||||
|
pub fn kill(&mut self) {
|
||||||
|
if let Some(mut child) = self.child.take() {
|
||||||
|
let _ = child.kill();
|
||||||
|
let _ = child.wait();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Drop for Node {
|
||||||
|
fn drop(&mut self) {
|
||||||
|
self.kill();
|
||||||
|
}
|
||||||
|
}
|
||||||
+10
-22
@@ -1,9 +1,7 @@
|
|||||||
//! Low-level context-switch tests. These poke `init_actor_stack` and the
|
//! Low-level context-switch tests. These poke `init_actor_stack` and the
|
||||||
//! naked asm shims directly — no scheduler involved.
|
//! naked asm shims directly — no scheduler involved.
|
||||||
|
|
||||||
use smarm::context::{
|
use smarm::context::{init_actor_stack, switch_to_actor, switch_to_scheduler};
|
||||||
get_actor_sp, init_actor_stack, set_actor_sp, switch_to_actor, switch_to_scheduler,
|
|
||||||
};
|
|
||||||
use smarm::stack::Stack;
|
use smarm::stack::Stack;
|
||||||
use std::cell::Cell;
|
use std::cell::Cell;
|
||||||
|
|
||||||
@@ -31,8 +29,7 @@ fn actor_runs_and_returns_to_scheduler() {
|
|||||||
reset_log();
|
reset_log();
|
||||||
let stack = Stack::new(64 * 1024, 4096).unwrap();
|
let stack = Stack::new(64 * 1024, 4096).unwrap();
|
||||||
let sp = init_actor_stack(stack.top(), actor_simple);
|
let sp = init_actor_stack(stack.top(), actor_simple);
|
||||||
set_actor_sp(sp);
|
let _ = unsafe { switch_to_actor(sp) };
|
||||||
unsafe { switch_to_actor() };
|
|
||||||
assert_eq!(get_log(), 0x1);
|
assert_eq!(get_log(), 0x1);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -48,12 +45,11 @@ fn actor_yields_and_resumes() {
|
|||||||
reset_log();
|
reset_log();
|
||||||
let stack = Stack::new(64 * 1024, 4096).unwrap();
|
let stack = Stack::new(64 * 1024, 4096).unwrap();
|
||||||
let sp = init_actor_stack(stack.top(), actor_two_steps);
|
let sp = init_actor_stack(stack.top(), actor_two_steps);
|
||||||
set_actor_sp(sp);
|
|
||||||
|
|
||||||
unsafe { switch_to_actor() };
|
let sp = unsafe { switch_to_actor(sp) };
|
||||||
assert_eq!(get_log(), 0x1, "after first resume");
|
assert_eq!(get_log(), 0x1, "after first resume");
|
||||||
|
|
||||||
unsafe { switch_to_actor() };
|
let _ = unsafe { switch_to_actor(sp) };
|
||||||
assert_eq!(get_log(), 0x1 | 0x2, "after second resume");
|
assert_eq!(get_log(), 0x1 | 0x2, "after second resume");
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -96,10 +92,9 @@ extern "C-unwind" fn actor_reg_check() {
|
|||||||
fn callee_saved_registers_survive_yield() {
|
fn callee_saved_registers_survive_yield() {
|
||||||
let stack = Stack::new(64 * 1024, 4096).unwrap();
|
let stack = Stack::new(64 * 1024, 4096).unwrap();
|
||||||
let sp = init_actor_stack(stack.top(), actor_reg_check);
|
let sp = init_actor_stack(stack.top(), actor_reg_check);
|
||||||
set_actor_sp(sp);
|
|
||||||
unsafe {
|
unsafe {
|
||||||
switch_to_actor();
|
let sp = switch_to_actor(sp);
|
||||||
switch_to_actor();
|
let _ = switch_to_actor(sp);
|
||||||
}
|
}
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
REG_BEFORE.get().copied().unwrap(),
|
REG_BEFORE.get().copied().unwrap(),
|
||||||
@@ -138,18 +133,11 @@ fn two_actors_dont_corrupt_each_other() {
|
|||||||
let sp_a = init_actor_stack(stack_a.top(), actor_a);
|
let sp_a = init_actor_stack(stack_a.top(), actor_a);
|
||||||
let sp_b = init_actor_stack(stack_b.top(), actor_b);
|
let sp_b = init_actor_stack(stack_b.top(), actor_b);
|
||||||
|
|
||||||
set_actor_sp(sp_a);
|
let sp_a = unsafe { switch_to_actor(sp_a) };
|
||||||
unsafe { switch_to_actor() };
|
let sp_b = unsafe { switch_to_actor(sp_b) };
|
||||||
let sp_a = get_actor_sp();
|
|
||||||
|
|
||||||
set_actor_sp(sp_b);
|
let _ = unsafe { switch_to_actor(sp_a) };
|
||||||
unsafe { switch_to_actor() };
|
let _ = unsafe { switch_to_actor(sp_b) };
|
||||||
let sp_b = get_actor_sp();
|
|
||||||
|
|
||||||
set_actor_sp(sp_a);
|
|
||||||
unsafe { switch_to_actor() };
|
|
||||||
set_actor_sp(sp_b);
|
|
||||||
unsafe { switch_to_actor() };
|
|
||||||
|
|
||||||
assert_eq!(A_VAL.with(|c| c.get()), 0xA00D);
|
assert_eq!(A_VAL.with(|c| c.get()), 0xA00D);
|
||||||
assert_eq!(B_VAL.with(|c| c.get()), 0xB00D);
|
assert_eq!(B_VAL.with(|c| c.get()), 0xB00D);
|
||||||
|
|||||||
+14
-3
@@ -396,8 +396,11 @@ impl GenServer for Pool {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// A worker spawned and watched from inside a handler delivers its Down to
|
// A worker spawned and watched from inside a handler delivers its Down to
|
||||||
// handle_down. Down arms outrank the inbox, so the death is in the log by
|
// handle_down. Nothing orders the worker's death before the follow-up
|
||||||
// the time the follow-up call is answered.
|
// call (the worker sits in the shared queue while call/reply wakes ride the
|
||||||
|
// wake slot), so poll: the log must become exactly [Panic] within a bounded
|
||||||
|
// number of yields. Down-outranks-inbox ordering is covered by
|
||||||
|
// `watch_dead_pid_is_noproc_down`.
|
||||||
#[test]
|
#[test]
|
||||||
fn worker_pool_down_reaches_handle_down() {
|
fn worker_pool_down_reaches_handle_down() {
|
||||||
let got = Arc::new(Mutex::new(Vec::new()));
|
let got = Arc::new(Mutex::new(Vec::new()));
|
||||||
@@ -409,7 +412,15 @@ fn worker_pool_down_reaches_handle_down() {
|
|||||||
});
|
});
|
||||||
server.cast(PoolCast::SpawnDoomedWorker).unwrap();
|
server.cast(PoolCast::SpawnDoomedWorker).unwrap();
|
||||||
let _ = server.call(()).unwrap(); // sync point: cast handled, worker live
|
let _ = server.call(()).unwrap(); // sync point: cast handled, worker live
|
||||||
*got2.lock().unwrap() = server.call(()).unwrap();
|
let mut log = Vec::new();
|
||||||
|
for _ in 0..10_000 {
|
||||||
|
log = server.call(()).unwrap();
|
||||||
|
if !log.is_empty() {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
smarm::yield_now();
|
||||||
|
}
|
||||||
|
*got2.lock().unwrap() = log;
|
||||||
});
|
});
|
||||||
assert_eq!(*got.lock().unwrap(), vec![DownReason::Panic]);
|
assert_eq!(*got.lock().unwrap(), vec![DownReason::Panic]);
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -161,3 +161,63 @@ fn demonitor_after_fire_is_none() {
|
|||||||
let _ = h.join();
|
let _ = h.join();
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// spawn_monitor: registration precedes publish, so a child that dies before
|
||||||
|
// the parent gets another instruction in still reports its real reason.
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn spawn_monitor_never_reports_noproc_multi_thread() {
|
||||||
|
// 4 schedulers, 200 instantly-dying children. With spawn+monitor this
|
||||||
|
// observes NoProc a few percent of the time (the child finishes on
|
||||||
|
// another scheduler before monitor() registers); with spawn_monitor it
|
||||||
|
// must be Exit every time.
|
||||||
|
let noproc = Arc::new(AtomicUsize::new(0));
|
||||||
|
let exit = Arc::new(AtomicUsize::new(0));
|
||||||
|
let (n, e) = (noproc.clone(), exit.clone());
|
||||||
|
smarm::init(smarm::Config::exact(4)).run(move || {
|
||||||
|
let mut hs = Vec::new();
|
||||||
|
for _ in 0..200 {
|
||||||
|
let (h, m) = smarm::spawn_monitor(|| {});
|
||||||
|
let d = m.rx.recv().expect("Down");
|
||||||
|
assert_eq!(d.pid, h.pid());
|
||||||
|
match d.reason {
|
||||||
|
DownReason::Exit => e.fetch_add(1, Ordering::Relaxed),
|
||||||
|
DownReason::NoProc => n.fetch_add(1, Ordering::Relaxed),
|
||||||
|
other => panic!("unexpected {other:?}"),
|
||||||
|
};
|
||||||
|
hs.push(h);
|
||||||
|
}
|
||||||
|
for h in hs {
|
||||||
|
h.join().unwrap();
|
||||||
|
}
|
||||||
|
});
|
||||||
|
assert_eq!(noproc.load(Ordering::Relaxed), 0);
|
||||||
|
assert_eq!(exit.load(Ordering::Relaxed), 200);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn spawn_monitor_sees_panic_and_demonitor_works() {
|
||||||
|
let ok = Arc::new(AtomicBool::new(false));
|
||||||
|
let o = ok.clone();
|
||||||
|
run(move || {
|
||||||
|
let (h, m) = smarm::spawn_monitor(|| panic!("boom"));
|
||||||
|
let d = m.rx.recv().expect("Down");
|
||||||
|
assert_eq!(d.pid, h.pid());
|
||||||
|
assert!(matches!(d.reason, DownReason::Panic));
|
||||||
|
let _ = h.join();
|
||||||
|
|
||||||
|
// demonitor before the child runs: no Down ever arrives.
|
||||||
|
let (h2, m2) = smarm::spawn_monitor(|| {});
|
||||||
|
demonitor(&m2);
|
||||||
|
h2.join().unwrap();
|
||||||
|
// Last sender gone with nothing sent: the channel is closed and empty.
|
||||||
|
assert!(
|
||||||
|
!matches!(m2.rx.try_recv(), Ok(Some(_))),
|
||||||
|
"demonitored spawn_monitor still delivered"
|
||||||
|
);
|
||||||
|
o.store(true, Ordering::SeqCst);
|
||||||
|
});
|
||||||
|
assert!(ok.load(Ordering::SeqCst));
|
||||||
|
}
|
||||||
|
|||||||
+13
-20
@@ -1,5 +1,6 @@
|
|||||||
//! Process-group tests that run under the scheduler: `join` installs a real
|
//! Process-group tests that run under the scheduler: `join` installs a real
|
||||||
//! monitor on a live actor, and a real death drives eviction on next contact.
|
//! monitor on a live actor, and a real death drives eviction (the reaper
|
||||||
|
//! actor sweeps it; the read path hides it in the meantime).
|
||||||
//! (Pure structural invariants live in the `pg` unit tests.)
|
//! (Pure structural invariants live in the `pg` unit tests.)
|
||||||
|
|
||||||
use smarm::{channel, members, pick, run, spawn};
|
use smarm::{channel, members, pick, run, spawn};
|
||||||
@@ -35,19 +36,15 @@ fn a_dead_actor_vanishes_from_every_group_it_joined() {
|
|||||||
assert_eq!(members("g1"), vec![pid]);
|
assert_eq!(members("g1"), vec![pid]);
|
||||||
assert_eq!(members("g2"), vec![pid]);
|
assert_eq!(members("g2"), vec![pid]);
|
||||||
|
|
||||||
// Release and reap the actor. finalize_actor queues the Down to our
|
// Release the actor. finalize_actor queues the Down to the reaper and
|
||||||
// monitors before unparking joiners, so by the time join() returns the
|
// marks the slot dead before unparking joiners, so by the time join()
|
||||||
// Down is already waiting in the membership channel.
|
// returns every read hides the pid whether or not the reaper has run.
|
||||||
tx.send(()).unwrap();
|
tx.send(()).unwrap();
|
||||||
h.join().unwrap();
|
h.join().unwrap();
|
||||||
|
|
||||||
// Drain-on-contact: touching g1 detects the death and sweeps the pid
|
// Gone from every group it joined, not just one.
|
||||||
// out of every group (g2 included), not just g1.
|
assert!(members("g1").is_empty(), "gone from g1");
|
||||||
assert!(members("g1").is_empty(), "evicted from the touched group");
|
assert!(members("g2").is_empty(), "and from g2");
|
||||||
assert!(
|
|
||||||
members("g2").is_empty(),
|
|
||||||
"and swept from the untouched group"
|
|
||||||
);
|
|
||||||
assert_eq!(pick("g1"), None);
|
assert_eq!(pick("g1"), None);
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
@@ -86,11 +83,7 @@ fn live_members_survive_a_peers_death() {
|
|||||||
tx_a.send(()).unwrap();
|
tx_a.send(()).unwrap();
|
||||||
a.join().unwrap();
|
a.join().unwrap();
|
||||||
|
|
||||||
assert_eq!(
|
assert_eq!(members("svc"), vec![b.pid()], "only the dead peer is gone");
|
||||||
members("svc"),
|
|
||||||
vec![b.pid()],
|
|
||||||
"only the dead peer is reaped"
|
|
||||||
);
|
|
||||||
assert_eq!(pick("svc"), Some(b.pid()));
|
assert_eq!(pick("svc"), Some(b.pid()));
|
||||||
|
|
||||||
tx_b.send(()).unwrap();
|
tx_b.send(()).unwrap();
|
||||||
@@ -123,18 +116,18 @@ fn leave_drops_a_membership_without_affecting_others() {
|
|||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn joining_an_already_dead_pid_is_evicted_on_next_contact() {
|
fn joining_an_already_dead_pid_never_shows_in_a_read() {
|
||||||
run(|| {
|
run(|| {
|
||||||
let h = spawn(|| {});
|
let h = spawn(|| {});
|
||||||
let pid = h.pid();
|
let pid = h.pid();
|
||||||
h.join().unwrap(); // actor is finalized before we join it to anything
|
h.join().unwrap(); // actor is finalized before we join it to anything
|
||||||
|
|
||||||
// monitor() on a gone pid queues a NoProc Down immediately, so the
|
// join() on a gone pid queues a NoProc Down to the reaper immediately;
|
||||||
// membership is reaped the next time the group is touched.
|
// reads never show it either way (slot-liveness backstop).
|
||||||
join("late", pid);
|
join("late", pid);
|
||||||
assert!(
|
assert!(
|
||||||
members("late").is_empty(),
|
members("late").is_empty(),
|
||||||
"dead-at-join member is reaped on read"
|
"dead-at-join member never reads as live"
|
||||||
);
|
);
|
||||||
assert_eq!(pick("late"), None);
|
assert_eq!(pick("late"), None);
|
||||||
});
|
});
|
||||||
|
|||||||
@@ -258,3 +258,46 @@ fn send_dyn_to_dead_pid_is_dead() {
|
|||||||
assert!(matches!(send_dyn::<u64>(p, 1u64), Err(SendError::Dead(_))));
|
assert!(matches!(send_dyn::<u64>(p, 1u64), Err(SendError::Dead(_))));
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// --- one channel per message type per actor -----------------------------------
|
||||||
|
|
||||||
|
/// Registering a second name of the same message type on one actor, with a
|
||||||
|
/// *fresh* channel, would silently replace and close the first — so it
|
||||||
|
/// panics (found in RFC 010 c9). The sanctioned shapes stay quiet: bind both
|
||||||
|
/// names to a clone of one sender, or use two actors.
|
||||||
|
#[test]
|
||||||
|
#[should_panic(expected = "already publishes a live channel")]
|
||||||
|
fn second_live_channel_of_same_type_on_one_actor_panics() {
|
||||||
|
run(|| {
|
||||||
|
let (tx1, _rx1) = channel::<u64>();
|
||||||
|
let (tx2, _rx2) = channel::<u64>();
|
||||||
|
register(Name::<u64>::new("dup-a"), tx1).unwrap();
|
||||||
|
register(Name::<u64>::new("dup-b"), tx2).unwrap(); // panics
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn two_names_on_one_cloned_sender_is_fine() {
|
||||||
|
run(|| {
|
||||||
|
let (tx, rx) = channel::<u64>();
|
||||||
|
register(Name::<u64>::new("twin-a"), tx.clone()).unwrap();
|
||||||
|
register(Name::<u64>::new("twin-b"), tx).unwrap();
|
||||||
|
send(Name::<u64>::new("twin-a"), 1).unwrap();
|
||||||
|
send(Name::<u64>::new("twin-b"), 2).unwrap();
|
||||||
|
assert_eq!(rx.recv().unwrap(), 1);
|
||||||
|
assert_eq!(rx.recv().unwrap(), 2);
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn replacing_a_channel_whose_receiver_is_gone_is_fine() {
|
||||||
|
run(|| {
|
||||||
|
let (tx1, rx1) = channel::<u64>();
|
||||||
|
register(Name::<u64>::new("reborn"), tx1).unwrap();
|
||||||
|
drop(rx1); // old inbox gone: replacement is the honest thing to do
|
||||||
|
let (tx2, rx2) = channel::<u64>();
|
||||||
|
register(Name::<u64>::new("reborn-2"), tx2).unwrap();
|
||||||
|
send(Name::<u64>::new("reborn-2"), 9).unwrap();
|
||||||
|
assert_eq!(rx2.recv().unwrap(), 9);
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|||||||
+3
-2
@@ -80,10 +80,11 @@ fn slot_off_means_zero_slot_traffic() {
|
|||||||
assert_eq!(rt.stats().slot_hits(), 0);
|
assert_eq!(rt.stats().slot_hits(), 0);
|
||||||
assert_eq!(rt.stats().slot_displacements(), 0);
|
assert_eq!(rt.stats().slot_displacements(), 0);
|
||||||
|
|
||||||
// …and off by default (RFC 005: default off until the shootout accepts).
|
// …and ON by default (RFC 005 accepted after the shootout — history.md
|
||||||
|
// finding 17/18: ping-pong 12.5× at 20T, ~9% at 1T, 100% slot hits).
|
||||||
let rt = init(Config::exact(1));
|
let rt = init(Config::exact(1));
|
||||||
rt.run(ping_pong(2, 100));
|
rt.run(ping_pong(2, 100));
|
||||||
assert_eq!(rt.stats().slot_hits(), 0, "wake_slot must default to OFF");
|
assert!(rt.stats().slot_hits() > 0, "wake_slot must default to ON");
|
||||||
}
|
}
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
|
|||||||
Reference in New Issue
Block a user