Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
89fb13e29a |
@@ -0,0 +1,2 @@
|
||||
[target.aarch64-unknown-linux-gnu]
|
||||
linker = "aarch64-linux-gnu-gcc"
|
||||
@@ -1,24 +0,0 @@
|
||||
#!/bin/sh
|
||||
# smarm pre-commit gate: clippy the library (src/) with warnings as errors.
|
||||
# unwrap_used / expect_used are denied (Cargo.toml [lints.clippy]): library
|
||||
# code must not hide a panic behind unwrap/expect. Tests/examples are not gated.
|
||||
#
|
||||
# Toolchain resolution: prefer an installed cargo-clippy; on machines whose
|
||||
# rust comes without the clippy component (e.g. NixOS home-manager), fall
|
||||
# back to an ephemeral nix-shell toolchain. The fallback uses its own target
|
||||
# dir (target/clippy) because the shell's rustc version may differ from the
|
||||
# default toolchain's — mixed-compiler artifacts in one target dir are an
|
||||
# E0514 hard error. MSRV (Cargo.toml rust-version) keeps the older shell
|
||||
# toolchain a legitimate gate.
|
||||
set -eu
|
||||
[ -f "$HOME/.cargo/env" ] && . "$HOME/.cargo/env"
|
||||
cd "$(git rev-parse --show-toplevel)"
|
||||
if cargo clippy --version >/dev/null 2>&1; then
|
||||
cargo clippy --lib -- -D warnings
|
||||
elif command -v nix-shell >/dev/null 2>&1; then
|
||||
nix-shell -p clippy -p cargo -p rustc \
|
||||
--run 'CARGO_TARGET_DIR=target/clippy cargo clippy --lib -- -D warnings'
|
||||
else
|
||||
echo "pre-commit: cargo clippy unavailable and no nix-shell fallback" >&2
|
||||
exit 1
|
||||
fi
|
||||
@@ -1,7 +1,3 @@
|
||||
target
|
||||
Cargo.lock
|
||||
smarm_trace.json
|
||||
/bench_results/
|
||||
__pycache__/
|
||||
*.pyc
|
||||
profile.coz
|
||||
|
||||
+1
-87
@@ -1,66 +1,17 @@
|
||||
[package]
|
||||
name = "smarm"
|
||||
version = "0.8.0"
|
||||
version = "0.4.0"
|
||||
edition = "2021"
|
||||
rust-version = "1.95"
|
||||
|
||||
[lints.rust]
|
||||
unexpected_cfgs = { level = "warn", check-cfg = ["cfg(loom)"] }
|
||||
|
||||
[lints.clippy]
|
||||
# Library code must never hide a panic behind unwrap/expect. Both are denied; an
|
||||
# intentional panic is written explicitly as `match { Err(e) => panic!(..) }`.
|
||||
# panic!/unreachable! are deliberately left un-linted as the blessed explicit
|
||||
# form. Enforced on the library target only (`cargo clippy --lib`); tests and
|
||||
# examples unwrap freely and are not gated.
|
||||
unwrap_used = "deny"
|
||||
expect_used = "deny"
|
||||
|
||||
[features]
|
||||
default = ["rq-mpmc"]
|
||||
smarm-trace = []
|
||||
# RFC 007: native causal profiling. Zero cost when off (cf. smarm-trace): the
|
||||
# hook in `maybe_preempt` and the resume-path fast-forward compile away; the
|
||||
# two Slot ledger fields exist regardless and stay 0 (budget_cycles precedent).
|
||||
smarm-causal = []
|
||||
# RFC 016 Chunk 2: cycle-accurate per-actor time-budget accounting. Off by
|
||||
# default — it costs two extra RDTSC reads per actor resume on the hot path
|
||||
# (D6). The `ActorInfo.budget_cycles` field exists regardless; it just stays 0
|
||||
# unless this is enabled.
|
||||
budget-accounting = []
|
||||
# RFC 016 Chunk 4: the live observer gen_server (src/observer.rs). Off by
|
||||
# default (DECISION D10) — the read primitive (Chunks 1–3) is always present
|
||||
# and unflagged; only the optional gen_server transport sits behind this, so a
|
||||
# release build pays nothing for an observer it never starts.
|
||||
observer = []
|
||||
# RFC 010 c1: clustering. Off by default — the default build stays libc-only,
|
||||
# byte-for-byte (gate checked per phase). serde is the payload contract,
|
||||
# postcard the payload codec; both minimal (no default features). Everything
|
||||
# cluster-shaped lives behind this flag.
|
||||
cluster = ["dep:serde", "dep:postcard"]
|
||||
# Run-queue selection: exactly one, compile-time (see src/run_queue.rs).
|
||||
# Non-default variants need --no-default-features (features are additive).
|
||||
rq-mutex = []
|
||||
rq-mpmc = []
|
||||
rq-striped = []
|
||||
|
||||
[build-dependencies]
|
||||
cc = "1"
|
||||
|
||||
[dependencies]
|
||||
libc = "0.2"
|
||||
# RFC 010 §2 — only compiled under `--features cluster`.
|
||||
serde = { version = "1", default-features = false, optional = true }
|
||||
# `alloc` (not `std`): the seam serializes to Vec; postcard stays no_std-aligned.
|
||||
postcard = { version = "1", default-features = false, features = ["alloc"], optional = true }
|
||||
|
||||
[target.'cfg(loom)'.dependencies]
|
||||
loom = "0.7"
|
||||
|
||||
[dev-dependencies]
|
||||
libc = "0.2"
|
||||
# derive + std for cluster envelope tests only; the lib itself never needs them
|
||||
serde = { version = "1", features = ["derive"] }
|
||||
tokio = { version = "1", features = ["rt", "rt-multi-thread", "macros", "sync", "time"] }
|
||||
|
||||
[profile.dev]
|
||||
@@ -71,14 +22,6 @@ panic = "unwind"
|
||||
lto = "thin"
|
||||
codegen-units = 1
|
||||
|
||||
# `cargo test --profile reltest`: release codegen for the crate (same opt-level,
|
||||
# same panic strategy) but no LTO at the final link. Thin LTO is what makes each
|
||||
# of the ~40 test binaries cost ~12 s to link instead of ~2 s; the tests don't
|
||||
# need cross-crate LTO, the benches do (they keep using `release`).
|
||||
[profile.reltest]
|
||||
inherits = "release"
|
||||
lto = false
|
||||
|
||||
[[bench]]
|
||||
name = "primes"
|
||||
harness = false
|
||||
@@ -98,32 +41,3 @@ harness = false
|
||||
[[bench]]
|
||||
name = "tokio_favored"
|
||||
harness = false
|
||||
|
||||
[[bench]]
|
||||
name = "rq_micro"
|
||||
harness = false
|
||||
|
||||
[[bench]]
|
||||
name = "rq_runtime"
|
||||
harness = false
|
||||
|
||||
[[bench]]
|
||||
name = "switch_cost"
|
||||
harness = false
|
||||
|
||||
# RFC 016 Chunk 4 — the live observer dump. Needs the optional gen_server.
|
||||
[[example]]
|
||||
name = "observer"
|
||||
required-features = ["observer"]
|
||||
|
||||
[[example]]
|
||||
name = "causal_pipeline"
|
||||
required-features = ["smarm-causal"]
|
||||
|
||||
[[example]]
|
||||
name = "causal_attrib_probe"
|
||||
required-features = ["smarm-causal"]
|
||||
|
||||
[[example]]
|
||||
name = "causal_probe"
|
||||
required-features = ["smarm-causal"]
|
||||
|
||||
@@ -1,38 +1,34 @@
|
||||
# smarm
|
||||
|
||||
> SMARM: Smarm, Marks Actor Runtime Machinery. A proof-of-concept green-thread actor runtime for Rust.
|
||||
> SMARM — Smarm, Marks Actor Runtime Machinery. A proof-of-concept green-thread actor runtime for Rust.
|
||||
|
||||
SMARM is my attempt to implement the erlang/OTP philosophy in the Rust programming language. This has yielded a fault-tolerant, fast, and scalable runtime. This runtime allows the creation of asynchronous applications in Rust without the function coloring associated with the async/await system. It encourages the writing of simple, synchronous code, and largely elides the need for lifetime annotations.
|
||||
Implements the core ideas in [`Achitecture.md`](.docs/Architecture.md): green-thread actors on a
|
||||
shared heap, scheduled cooperatively, communicating only by `Send` messages.
|
||||
Erlang's isolation model without Erlang's copying GC, Rust's zero-copy
|
||||
ownership transfers without async's function colouring.
|
||||
|
||||
The scheduler is multi-threaded — one OS thread per available CPU, all drawing
|
||||
from a shared run queue. The single-threaded `run()` entry point is kept as a
|
||||
convenience wrapper around `runtime::init(Config::exact(1)).run(f)`.
|
||||
|
||||
## Overview
|
||||
|
||||
SMARM implements green-thread actors on a shared heap, communicating only by `Send` messages. By sharing the heap, SMARM avoids the copying overhead of Erlang, which is safe to do due to Rust's borrow checker.
|
||||
|
||||
On top of the core runtime mechanics, SMARM also provides a library of primitives for making applications closely inspired by erlang/OTP. This includes generic servers (gen_servers), generic state machines (gen_statem), and supervision trees.
|
||||
|
||||
Supervision trees are the core primitive to allow your application to survive an unexpected panic. Supervisors are processes dedicated to monitoring other processes, which can restart these should they fail. This means that when set up properly an application may 'self-heal' when encountering unforeseen circumstances.
|
||||
|
||||
SMARM is not cooperatively scheduled; it uses preemption. This means a heavy task will not starve out other lighter tasks. Everything will make steady progress, which translates to very beneficial behaviour under (over)load: average latency goes up, but tail latency does not blow up.
|
||||
|
||||
To help diagnose these unforeseen circumstances, smarm may be compiled with its `tracing` feature, which emits a full trace using [Perfetto](https://perfetto.dev/).
|
||||
|
||||
Should you want to optimize your application, SMARM is unusually well poised to help. As the runtime functionally controls time, SMARM comes with a built in causal profiler, under the `causal` feature.
|
||||
|
||||
I also built a Phoenix-Framework inspired HTTP 1.1 library on top of SMARM called [URUS](https://git.kalsbeek.dev/Markk116/urus), which implements Pub/Sub, Channels, and basic amenities like Websockets and Server Sent Events.
|
||||
|
||||
## Limitations
|
||||
|
||||
This runtime requires naked assembly to function, and has thus far only been implemented for x86-64 assembly. It expects an operating system that supports virtual address space, and is therefore not (yet) suited for embedded targets. The IO implementation is currently based around the Linux kernel's `epoll` mechanism, meaning it requires a (GNU+)Linux distribution to run.
|
||||
|
||||
The preemption mechanism works by wrapping the memory allocator and checking how many CPU cycles you have used compared to your timeslice budget. This allows preemption to fire in most normal code, but tight zero-allocation loops do not get caught and require manual insertion of `check!()` if you want preemption to function.
|
||||
|
||||
This library is still in its early stages, and while I try my best with loom and tests, stable operation cannot be guaranteed. Therefore it is not (yet) recommended for production use.
|
||||
|
||||
At this moment, stack memory for each green thread is capped. Uncapping this may lead to performance benefits for deeply recursive algorithms that in a traditional async runtime might require pointer-chases through the heap. This is as yet unrealised.
|
||||
|
||||
At this stage, the codebase is largely LLM-generated, which is obvious if you start to read through the internals. While I did the design, and I keep the LLM under tight rein, the codebase is not in a state that I am very happy with. This also goes for the documentation.
|
||||
## What's here
|
||||
|
||||
| Module | What it does |
|
||||
|--------------|------------------------------------------------------------------------|
|
||||
| `stack` | `mmap`'d growable stack with guard page; SIGSEGV on overflow |
|
||||
| `context` | `#[naked]` x86-64 context-switch shims, callee-saved regs only |
|
||||
| `preempt` | Allocator-driven preemption; `check!()` macro for no-alloc loops |
|
||||
| `pid` | `(index, generation)` PIDs; stale handles are detectable, not silent |
|
||||
| `actor` | Trampoline + `catch_unwind` boundary at the actor entry point |
|
||||
| `scheduler` | Run queue, slot table, spawn/join, parking, idle path |
|
||||
| `channel` | Unbounded MPSC channel; `recv` parks the actor |
|
||||
| `mutex` | `Mutex<T>` with mandatory timeout; FIFO waiters; parks the green thread |
|
||||
| `timer` | Min-heap of `(deadline, reason)`; `Sleep` and `WaitTimeout` reasons |
|
||||
| `io` | `block_on_io` for blocking work; `wait_readable`/`wait_writable` + `read`/`write` via epoll |
|
||||
| `supervisor` | `Signal::Exit`/`Panic`/`Stopped` funnelled to a parent; `OneForOne`/`OneForAll`/`RestForOne` strategies + restart-intensity cap |
|
||||
| `monitor` | `monitor(pid)` → `Monitor { id, target, rx }`; one-shot `Down` via `rx`; `demonitor(&m)` tears one registration down; unidirectional death notice |
|
||||
| `link` | bidirectional `link`/`unlink`; abnormal death propagates (cooperative stop, or an `ExitSignal` message under `trap_exit`) |
|
||||
| `gen_server` | `call` (sync request-reply) / `cast` (async) over one inbox; `ServerRef` + `init`/`terminate` hooks; server-down via channel closure |
|
||||
|
||||
## Quick taste
|
||||
|
||||
@@ -54,29 +50,6 @@ run(|| {
|
||||
});
|
||||
```
|
||||
|
||||
## Stopping actors
|
||||
|
||||
Two strengths, as in OTP. `request_stop(pid)` is `exit(Pid, kill)`: a cooperative
|
||||
hard stop, unwinding at the actor's next observation point. `request_shutdown(pid)`
|
||||
is `exit(Pid, shutdown)`: an actor that traps exits (`trap_exit()`, or
|
||||
`ctx.trap_exit()` in a gen_server / `cx.trap_exit()` in a gen_statem) receives it
|
||||
as a signal — `handle_shutdown` / a `shutdown` row — and may drain before stopping
|
||||
itself; one that does not trap is stopped outright. Supervisors trap:
|
||||
`request_shutdown(sup)` tears the tree down top-down, each child per its
|
||||
`ChildSpec` `Shutdown` policy (`Timeout(d)`, `Infinity`, `BrutalKill`). The run's
|
||||
root actor returning means "the program is done": every top-level actor gets a
|
||||
`request_shutdown`, and `run()` returns when they are gone. From outside the
|
||||
runtime (a signal thread), `Runtime::handle().request_shutdown(pid)` does the
|
||||
same. `examples/graceful_shutdown.rs` shows all of it.
|
||||
|
||||
A gen_server or gen_statem lives until it stops, is shut down, or is killed; its
|
||||
refs are addresses — dropping them never ends it (a forgotten one is swept at
|
||||
root exit; with `--features smarm-trace` each such sweep is a `root_sweep`
|
||||
trace line). The supervised shape is `GenServerBuilder::named(N).run()` /
|
||||
`gen_statem::run_named(N, m)`: the server runs inline as the `ChildSpec` child
|
||||
itself, so the supervisor's shutdown reaches it directly, a restart re-binds the
|
||||
name, and the program addresses it by name.
|
||||
|
||||
## Layout
|
||||
|
||||
```
|
||||
@@ -93,9 +66,12 @@ benches/
|
||||
|
||||
## Building and running
|
||||
|
||||
Standard Cargo. Requires Rust 1.95 or newer (the `#[naked]` attribute went stable in 1.88; we use a few unrelated post-1.88 features). I have worked hard to keep this library as dependency-free as possible. `master` is x86-64 Linux only. An experimental, **untested** aarch64 context-switch backend lives on the `arm-port` branch (extracted into a `target_arch`-gated `src/arch/`); it has not been validated on hardware yet. macOS remains on the deferred list because of the epoll dependency.
|
||||
|
||||
|
||||
Standard Cargo. Requires Rust 1.95 or newer (the `#[naked]` attribute went stable
|
||||
in 1.88; we use a few unrelated post-1.88 features). `master` is x86-64 Linux
|
||||
only. An experimental, **untested** aarch64 context-switch backend lives on the
|
||||
`arm-port` branch (extracted into a `target_arch`-gated `src/arch/`); it has not
|
||||
been validated on hardware yet. macOS remains on the deferred list because of the
|
||||
epoll dependency.
|
||||
|
||||
```sh
|
||||
cargo test # all tests
|
||||
@@ -103,29 +79,26 @@ cargo test --test mutex # one module
|
||||
cargo bench # primes benchmark vs tokio
|
||||
```
|
||||
|
||||
## What's not here
|
||||
|
||||
See the **Defer** section of `Architecture.md`.
|
||||
`join!` for handle groups, stack growth via remap,
|
||||
hierarchical timer wheel, fd-wait timeouts, `Signal::Timeout`. Each is
|
||||
mechanism we know how to add; none belongs in this iteration.
|
||||
|
||||
## Docs
|
||||
|
||||
| Document | What it covers |
|
||||
|---|---|
|
||||
| [`Architecture.md`](./docs/Architecture.md) | Design intent, runtime model, and deferred work |
|
||||
| [`smarm - Deep Dive.html`](./docs/smarm%20-%20Deep%20Dive.html) | Generated walkthrough of the system; good starting point if you want to learn about the internals |
|
||||
| [`smarm - Deep Dive.html`](./docs/smarm%20-%20Deep%20Dive.html) | Generated walkthrough of the system; good starting point |
|
||||
| [`BENCHMARKS_AND_TUNING.md`](./docs/BENCHMARKS_AND_TUNING.md) | Where smarm wins and loses vs tokio, preemption knob recommendations |
|
||||
| [`benchmarks.md`](./docs/benchmarks.md) | Raw benchmark results, methodology, and tuning experiment log |
|
||||
|
||||
## Coming up
|
||||
|
||||
Clustering: clustering multiple SMARM nodes together is in the pipeline.
|
||||
SMARM-BEAM Interop: Running SMARM as a supervised node under the BEAM via a Rustler NIF works, including message passing and supervision trees that span the runtimes. However, this library is still too unstable to release.
|
||||
SMARM is an interesting platform for implementing a 'dataflow' library, but work on this has not yet started.
|
||||
|
||||
|
||||
## Contributing
|
||||
|
||||
This started as a personal proof-of-concept, but it is starting to outgrow that name. If you want to contribute, please get in contact to discuss what you want to work on. Code without prior communication is not welcome.
|
||||
|
||||
|
||||
## A note on open source
|
||||
|
||||
An open source project is a gift, and by giving it, it is no longer mine. I highly enourage you to fork it, to make it your own. This repository, however, is still mine.
|
||||
This is a personal proof-of-concept. There's no PR workflow. If you fork it and do something interesting, just send me an email. If it's nice, I'll upstream the changes.
|
||||
|
||||
---
|
||||
|
||||
|
||||
|
||||
-363
@@ -1,363 +0,0 @@
|
||||
# smarm — Roadmap
|
||||
|
||||
## Shipped (compacted — full cycle plans and deviation records live in git history)
|
||||
|
||||
Cycles before v0.8 (v0.4 actor primitives, v0.5 runtime decomposition &
|
||||
pluggable run queue, v0.6 actor ergonomics, v0.7 select on epoch-stamped
|
||||
consuming wakes): see `git log ROADMAP.md`.
|
||||
|
||||
### v0.8 — gen_server: handle_info / handle_down + io fd hygiene ✅
|
||||
Spent `select` on the server loop: static info arms (`type Info`,
|
||||
`ServerBuilder::with_info`) and dynamic monitor forwarding
|
||||
(`ServerCtx`/`Watcher` + a control arm), priority downs → control → infos →
|
||||
inbox. Closed the v0.2 fd hole: a drop guard in `wait_fd` DELs the kernel
|
||||
registration on unwind (the leak was worse than documented — a stale waiters
|
||||
entry permanently poisoned the fd). Deviation: the plain-inbox fast path
|
||||
narrowed; servers holding a `Watcher` select forever.
|
||||
Commits `e5d1b3b`, `24b95c9`, `f6969e5`.
|
||||
|
||||
### v0.9 — Wake-path latency ✅
|
||||
Attacked per-wake latency with the RFC 005 **wake slot**: a per-scheduler,
|
||||
thread-local, capacity-one wake cache checked before the shared queue, pushed
|
||||
only from actor context, slot-then-shared pop with the waker's residual slice
|
||||
as the starvation bound (`slot_hits`/`slot_displacements`). Benched via the
|
||||
slot on/off dimension of `rq_runtime` — ping-pong-pairs (win), yield-storm
|
||||
(regression guard), spawn-storm (neutrality); results annotated in RFC 005.
|
||||
The RFC 004 spinning-workers experiment, originally scoped here, was evaluated
|
||||
and **excised** (not worth the code cost; preserved on branch
|
||||
`rfc-004-spinning`). Also a false-sharing fix (`align(64)` on `SchedulerStats`)
|
||||
and a termination wake for idle siblings.
|
||||
Commits `2708042`, `37d9319`, `eddf3fe`.
|
||||
**Default flipped ON 2026-08-18** (history.md findings 17/18): the slot had
|
||||
shipped default-off "until the shootout accepts it" and the flip was never
|
||||
made, so every general.rs number since was slot-off. Acceptance sweep
|
||||
(rq_runtime, 1/2/4/8/20 schedulers, rq-mpmc, 7 runs): ping-pong-pairs
|
||||
−7% at 1T, 3.6×/5.3×/15.6×/10× faster at 2/4/8/20T, 100% slot hits, 0
|
||||
displaced; yield-storm and spawn-storm within ±10% noise both ways.
|
||||
general.rs on the flip: ping_pong_steady 20T 18485→1549 µs, 1T −15%,
|
||||
mpsc_contention 1T −51%, ping_pong_oneshot 20T −18%; no smarm regressions.
|
||||
`benches/baseline.json` regenerated slot-on.
|
||||
|
||||
---
|
||||
|
||||
## Decision record — queue topology 🔒 CLOSED (2026-06-10)
|
||||
|
||||
The run-queue shootout (harness `6d9f369`, 24-core sweep 1–24 schedulers,
|
||||
report `bench_report_rq_shootout.html`) landed in **RFC 005's World 3**: the
|
||||
three queue variants are within 10–15% of each other in `rq_runtime` at every
|
||||
scheduler count ≥ 4, on all three workloads.
|
||||
Consequences:
|
||||
- **`rq-mutex` stays the default** — simplest correct, no capacity
|
||||
constraints, locking model already integrated.
|
||||
- **Feature plumbing stays as is.** All three variants keep compiling in
|
||||
every build; `rq-mpmc`/`rq-striped` remain selectable for benching.
|
||||
- **Reopening is benchmark-driven only.** The report documents the
|
||||
conditional upgrade paths if a future workload qualifies: mpmc for
|
||||
message-passing-dominant loads at N ≤ 8; striped for high-contention balanced push/pop at N ≥ 16. Neither is a scheduler workload as measured.
|
||||
- **Effort redirects to the wake path**: RFC 005 (billed as a latency patch,
|
||||
per its own World 3 framing), RFC 004, and eventually per-switch cost.
|
||||
|
||||
---
|
||||
|
||||
## Typed addressable mailboxes ✅ SHIPPED (RFC 013 + RFC 014)
|
||||
|
||||
The unblocker several later items quietly assumed: a `Pid` is now messageable.
|
||||
RFC 013 reworked `registry.rs` from a name↔pid *bimap* into a name→**live
|
||||
mailbox** directory off the cold leaf; RFC 014 layered the typed addressing and
|
||||
producers on top. Two addressing modes — `Pid<A>` (direct, identity-bound) and
|
||||
`Name<M>` (durable, re-resolving, location-transparent) — compile-time typing
|
||||
preserved via phantom tokens over *contained* `Box<dyn Any>` erasure (the
|
||||
global-enum alternative was rejected: it breaks library-extensibility for
|
||||
out-of-crate actors). Channel store keyed by message `TypeId` in every path.
|
||||
|
||||
Delivered surface:
|
||||
- **Sends:** `send_to` (`Pid<A>`), `send` (`Name<M>`), `send_dyn` (bare-pid
|
||||
escape hatch, names the message type) — earlier 014 phases.
|
||||
- **Producers & discovery** (final phase, `a866e34`): `spawn_addr` (typed-path
|
||||
producer; parent-side inbox publish so an immediate `send_to` resolves, no
|
||||
race on the body); `lookup_as` / `pick_as` / `members_as` (unchecked-but-sound
|
||||
re-type of an erased pid — a wrong `A` degrades to `NoChannel`, never
|
||||
misdelivery); `dispatch` (pick-a-live-member-and-send, `SendError::NoMember`
|
||||
on empty pool).
|
||||
- **By-name gen_servers** (final phase): `ServerName<G>` over the existing
|
||||
typed-channel store (keyed by `TypeId::of::<Envelope<G>>()`, so `Envelope`
|
||||
stays private and no separate directory is needed); type-state
|
||||
`NamedServerBuilder<G>` (fallible `start`, `NameTaken`) leaving the infallible
|
||||
`ServerBuilder::start` untouched; free `call` / `cast` / `whereis_server`;
|
||||
`ServerRef::shutdown` + free `shutdown` as the sys-style synchronous stop.
|
||||
- **Root-exit teardown** (final phase): the run's initial actor is the root;
|
||||
when it exits the run winds down. *(Reworked with the graceful-shutdown work:
|
||||
root exit now delivers `request_shutdown` to every forest root — see
|
||||
"Root exit" below and `tests/root_exit.rs`.)* Closes the "app actor blocks
|
||||
AllDone" stall — see Look into, below.
|
||||
|
||||
Extends — does not retire — the "select exists; a unified per-process mailbox
|
||||
still does not" invariant: 014 adds addressable *delivery*, not a unified inbox;
|
||||
multi-port stays `select` composition over named channels. Examples:
|
||||
`examples/{typed_actor,named_genserver,worker_pool}.rs`. Earlier-phase commits
|
||||
and the full 013/014 history in git.
|
||||
|
||||
---
|
||||
|
||||
## gen_server time-related patterns ✅ SHIPPED (RFC 015)
|
||||
*(layer 2; seven commits `47d75d1`→`f454a91`, each a reviewable chunk. Builds on
|
||||
the `send_after` substrate below.)*
|
||||
|
||||
The OTP time vocabulary against the v0.8 server loop, with **no handler-signature
|
||||
change** — the capability is a handle stashed on `self` (the `Watcher` pattern),
|
||||
not a return-directive or a `&ctx` threaded through handlers. New trait surface:
|
||||
`type Timer` (server's own scheduled payload, `()` if unused, kept distinct from
|
||||
the external `type Info`), `handle_timer`, `handle_idle` (both no-op defaults).
|
||||
|
||||
- **One-shot / debounce / retry-backoff** — `ctx.timer()` hands out a clonable
|
||||
`TimerHandle`; `arm_after(d, msg)` arms, `cancel(id)` carries the substrate's
|
||||
race bool. Debounce/backoff are just arm-and-`cancel` against the latest event
|
||||
(no special mechanism).
|
||||
- **Periodic tick / heartbeat** — `tick_every(d, msg)`: loop-managed sugar over
|
||||
the one-shot substrate (re-arm at `now + d`), one stable id, `cancel` stops the
|
||||
re-arm *and* the pending instance. Requires `Timer: Clone` (method-level bound
|
||||
only); the loop re-delivers via a stored factory, so the bound never leaks onto
|
||||
`type Timer` or the loop.
|
||||
- **Idle / receive timeout** — *not* a channel: it is the timeout on the loop's
|
||||
`select_timeout` / `recv_timeout`. `ctx.idle_after(d)` (set once in `init`)
|
||||
fixes the window; reset on any dispatched message; `None` from the wait ⇒
|
||||
`handle_idle`; re-arms steady. No generation-tag race — there is no token.
|
||||
|
||||
Mechanics: control (monitor intake) + armed timers fold into one loop-internal
|
||||
`Sys` channel selected above the inbox, so **armed timers outrank infos** (a
|
||||
heartbeat can't be starved). One substrate addition — `send_after_to` (a
|
||||
channel-targeting sibling of `send_after`, lands the fire on the loop's own arm).
|
||||
Exit is leak-free: the drop guard (same one that runs `terminate`) drains and
|
||||
cancels every live timer id, then `debug_assert!`s none survive. `call_timeout`
|
||||
is unchanged — it's a *client-side* call deadline, disambiguated in docs from the
|
||||
server-side idle timeout (same word, two axes). `gen_statem` state timeouts
|
||||
(deferred, Low) will reuse the loop-owned idle deadline.
|
||||
|
||||
---
|
||||
|
||||
## Process groups — the primitive pubsub & channels should have sat on ✅ SHIPPED (RFC 012)
|
||||
*(`src/pg.rs`, four commits `b78311b`→`56f2fc5`; see HANDOFF + git history. Context retained below.)*
|
||||
|
||||
Context: urus is a webserver written on top of smarm to provide a testing target.
|
||||
A named pid→multiset map with monitor-backed removal: `registry.rs` generalised
|
||||
from name↔pid *bimap* to name→*multiset*, the death hook reused verbatim. Local
|
||||
first. The point is that urus pubsub collapses into a pg consumer (`subscribe` =
|
||||
join, `broadcast` = send-to-members) instead of being a bespoke mechanism, and the
|
||||
same group set reads two ways — fan-out (all members) vs discovery/pool (one
|
||||
member), with different netsplit consequences. Urus shipped pubsub/channels predate this
|
||||
and want reframing on top of it. Foundational, so early in the post-v0.9 stack.
|
||||
Needs an RFC.
|
||||
|
||||
---
|
||||
|
||||
## Later
|
||||
### Highest priority
|
||||
#### Per-switch cost (context shims, epoch protocol)
|
||||
The shootout's residual: per-wake latency is 0.16–0.18 µs at N=1 and
|
||||
0.8–1.2 µs at N=8+, dominated by the context-switch shims and the epoch
|
||||
protocol, not the queue. **Premise corrected 2026-08-18 (finding 17): the
|
||||
N=8+ figure was the slot-off futex path (one futex_wake per landed wake,
|
||||
woken worker steals the pair); slot-on it is ~1.3× N=1. The shims are ~5
|
||||
cycles (finding 3) and the epoch CASes ~117 cycles/roundtrip (finding 15).
|
||||
Re-measure slot-on before spending anything here.** On current evidence this is the larger constant —
|
||||
"the whole game" alongside the v0.9 work — but there is no spec yet. Needs a
|
||||
profiling spike (where do the cycles actually go per park/unpark round-trip)
|
||||
and then an RFC before it can be scheduled.
|
||||
|
||||
#### send_after / cancel_timer
|
||||
✅ SHIPPED (`61520bf`) — message-delivery timer on the `timer.rs` min-heap:
|
||||
deliver a value to an address (`Pid<A>` via `send_to`, `Name<M>` via `send`),
|
||||
resolved *on fire* so a dead target / restarted name is observed at fire time;
|
||||
failed resolve dropped (Erlang `erlang:send_after`). `Reason::Send { fire }`
|
||||
carries delivery type-erased; cancellation is an `armed` set keyed on entry
|
||||
`seq` (only `Send` uses it — `Sleep`/`WaitTimeout` stay inert-stale), exposed as
|
||||
an opaque `TimerId`; `cancel` is unscoped and returns the race signal.
|
||||
`peek_deadline` relaxed to "≤ true next deadline" for a future timing wheel. The
|
||||
gen_server time layer (RFC 015, shipped above) lands on this, adding only the
|
||||
channel-targeting `send_after_to` sibling.
|
||||
|
||||
#### Introspection — process_info / get_state / tree dump
|
||||
✅ SHIPPED (RFC 016) — runtime introspection & observability, superseding the
|
||||
RFC 000 / 006 / 009 sketches. The mechanism is an internal synchronous read,
|
||||
not a C ABI: `snapshot()` / `actor_info(pid)` return owned data (pid, names,
|
||||
fine scheduling state, parent edge, trap, monitor/link/joiner counts, mailbox
|
||||
depth) carrying `SNAPSHOT_FORMAT_VERSION` (Chunk 1, ps-semantics tearing);
|
||||
`tree()` folds that into a parentage forest with orphan re-rooting (Chunk 3).
|
||||
Per-actor counters — timeslice overruns, messages-received, and a feature-gated
|
||||
approximate time budget — ride hot `AtomicU64` slot fields (Chunk 2). The live
|
||||
`observer` gen_server is the read transport over that primitive, behind the
|
||||
off-by-default `observer` feature (Chunk 4); it is the read half of the future
|
||||
RFC 003 control plane. Wedged-runtime dumps stay gdb's job, and park-reason
|
||||
detail / C ABI are explicit non-goals.
|
||||
|
||||
#### Worker pool behaviour
|
||||
Supervised, interchangeable workers with restart semantics over a shared inbox
|
||||
(poolboy / NimblePool shape) — distinct from connection pools (bb8/deadpool), which
|
||||
pool *resources*, not *supervised processes*. Sits on `supervisor.rs` +
|
||||
`gen_server.rs`. Needs an RFC.
|
||||
|
||||
### Medium Priority
|
||||
|
||||
#### Demand-driven pipelines — GenStage / Broadway shape
|
||||
Supervised producer/consumer stages where consumers signal demand upstream, with
|
||||
batching, ack, partitioning. The clearest thing hex has and crates.io lacks (stream
|
||||
combinators and bounded channels are not a supervised demand-contract stage graph),
|
||||
and the natural fit for ingestion-shaped workloads. Builds on channels + gen_server
|
||||
+ supervisor. Needs an RFC.
|
||||
#### Unwakeable idle sleep when io is absent (terminal-wake residual)
|
||||
The `(Some(deadline), None)` idle branch — timers pending, io subsystem never
|
||||
initialized — blocks in `thread::sleep` with no wake mechanism at all. The
|
||||
terminal wake (writes the wake pipe at AllDone) cannot reach it: no io, no
|
||||
pipe. Same stall as the fixed bug, in any no-io runtime: a sibling that
|
||||
blocked on an orphaned deadline sleeps it out in full after everything else
|
||||
finished. Candidates, mutually exclusive: (a) clamp the sleep (cheap, but
|
||||
turns idle into periodic wakeups), or (b) park the branch on a condvar/futex
|
||||
the AllDone path signals — and at that point consider making the condvar the
|
||||
idle primitive for the no-io runtime generally (a cross-thread unpark could
|
||||
signal it too, see below). Decide before any no-io deployment.
|
||||
#### Cross-thread unpark
|
||||
`RuntimeInner::enqueue` does not wake idle sibling schedulers — only io
|
||||
completions write the wake pipe. Mid-flight this is masked (the enqueuing
|
||||
thread is awake and eats the work itself), but it costs parallelism: work
|
||||
enqueued by a busy thread waits until the sibling's idle poll times out. Needs bench evidence (does the shared-queue handoff latency actually show up?) before a mechanism is picked.
|
||||
#### Unbounded / configurable-bounded actor count
|
||||
Fixed slab with a loud assert (`Config::max_actors(n)`, default 16 384).
|
||||
Revisit with a segmented slab (array of `AtomicPtr<Segment>`, doubling segment
|
||||
sizes, append-only) once the cap is actually hit. Do not let it calcify.
|
||||
#### arm-port validation & merge
|
||||
`arm-port` branch carries an AAPCS64 context-switch backend, never run on
|
||||
hardware. Build + run full test suite on an aarch64 device; check
|
||||
`chained_spawn` / `yield_many` bench medians; merge and update README.
|
||||
|
||||
### Low priority
|
||||
#### gen_statem — postponement + state timeouts only
|
||||
A thin layer over gen_server, not a new behaviour. The (state, event) dispatch
|
||||
matrix is free from the type system and not worth porting. The two mechanisms that
|
||||
are: event **postponement** (defer events in the wrong state, replay on transition
|
||||
— selective receive, codified) and **state timeouts** (auto-cancel on state
|
||||
change). Device-connection FSMs are the canonical use. Wants send_after underneath.
|
||||
Needs an RFC.
|
||||
|
||||
#### Clustering — distribution epic
|
||||
Sequenced deliberately after v0.9 and the per-switch-cost spike. A fat stack of
|
||||
RFCs, not one. Spine settled in discussion; decisions still open:
|
||||
- **Explicit remote boundary, never transparency.** Serialization colours *edges*
|
||||
(channel types), not functions — local edges stay zero-copy `Send`, only remote
|
||||
edges take a `RemoteRef<T: Serialize + DeserializeOwned>`. No hidden latency when
|
||||
a peer migrates; the refactor is visible by construction.
|
||||
- **One binary, role as runtime config** (`ROLE=… REGION=… SEEDS=…`); a build-hash
|
||||
handshake enforces same-binary type identity and sidesteps cross-version type
|
||||
agreement. Roles select which supervision subtree mounts.
|
||||
- **Distributed pg falls out of local pg + a membership/gossip layer**, and
|
||||
distributed pubsub falls out of that for free; per-member metadata (region, load)
|
||||
enables fly-style nearest-member routing.
|
||||
- **Migratable gen_servers** as a sub-layer: only behaviours migrate (a raw actor's
|
||||
stack is opaque; a gen_server *between callbacks* is just its `State`), gated by
|
||||
`Serialize` bounds + an `on_arrive` reacquire hook, addressed by name not pid. The
|
||||
BEAM can't do this — leaning on the behaviour layer is what buys it. Requires `State` to be serializable, so should probaby spec a `trait MigratableGenServer: GenServer where Self::State: Migratable` , or something to that extent, so we can lean on the type system to make sure we don't accidentally make state that cannot be serialised. (The "addressed by name not pid" half is RFC 013's `Name<M>` durable address — its local form is the foundation this remote layer extends.)
|
||||
- **CRDT presence** is the high-value, genuinely-hard layer above distributed pg,
|
||||
kept *out* of the pg primitive (the pg2 strong-consistency lesson). Furthest out. Can maybe defer to rust ecosystem
|
||||
|
||||
|
||||
---
|
||||
|
||||
## Look into
|
||||
|
||||
### app actors block AllDone; no external stop path — ADDRESSED (RFC 014 root-exit teardown)
|
||||
Agent working on urus (see same git server as smarm) reported a lazily spawned actor never returning, blocking program shutdown. Maybe we should do something about it. Agent worked around it by giving the actor an atomic bool to spin on. See urus example crud for exact impl.
|
||||
|
||||
**Update (RFC 014):** root-exit teardown stops the parked-forever remainder when
|
||||
the root actor exits, so a lazily spawned daemon no longer wedges shutdown on
|
||||
`live_actors > 0`; `ServerRef::shutdown` (+ free `shutdown`) is the explicit stop
|
||||
path the atomic-bool workaround stood in for. Re-check the urus crud repro to
|
||||
confirm the workaround can be retired (the teardown is cooperative — an actor in
|
||||
a tight loop with no observation point still can't be stopped).
|
||||
|
||||
**Update (graceful shutdown):** the RFC 014 sweep was a hard `request_stop` of
|
||||
every live slot, deferred until nothing was runnable — which killed a sleeping
|
||||
actor (timer pending) but drained a queued one, for no principled reason. It is
|
||||
now the OTP semantics: root exit = "the program is done" = `request_shutdown`
|
||||
to every **forest root** (live actor whose parent is the run or is dead), run
|
||||
synchronously on the root's finalize path. Supervisors cascade with their child
|
||||
`Shutdown` policies; trapping actors may `Continue`/drain (timers keep working)
|
||||
and end the run when they stop themselves; non-trapping actors are stopped
|
||||
outright — `join` what you need finished. No forcing sweep follows.
|
||||
|
||||
---
|
||||
|
||||
### Open items from the graceful-shutdown work (not scheduled)
|
||||
- ~~gen_server / gen_statem as a direct supervised child.~~ Done: lifetime is
|
||||
the actor's (refs are addresses, the loop holds an inbox sender);
|
||||
`NamedGenServerBuilder::run` / `gen_statem::run_named` run the loop inline as
|
||||
the `ChildSpec` child; the root-exit sweep traces each leftover as
|
||||
`root_sweep` under `smarm-trace`.
|
||||
- Supervisor `Live` drop-guard sweep is `request_stop` (kill propagates as
|
||||
kill); OTP would deliver a trappable `killed`. Chosen for boundedness.
|
||||
- A root-exit shutdown reaches only actors live *at that instant*; a
|
||||
non-trapping forest root that spawns before it unwinds leaves that spawn
|
||||
to itself (Erlang: an unlinked spawn is nobody's child).
|
||||
- **Supervisor start *order* is not start *readiness*.** `start_child`
|
||||
spawns and moves straight on, so an earlier child is merely *scheduled*,
|
||||
not initialised, when a later sibling starts. A later child that resolves
|
||||
an earlier one by name (`whereis_server`) can therefore miss it — the
|
||||
classic "named registry sibling, then its consumers" tree. Ordered
|
||||
`OneForOne`/`RestForOne` shutdown is unaffected (reverse order is honoured
|
||||
and each stop *is* awaited); this is a start-side gap only.
|
||||
Making `spawn` itself block does NOT fix it — it would only shrink the
|
||||
window to "child has begun executing", while the property callers need is
|
||||
"child has bound its name / opened its socket", which only the child can
|
||||
declare. It would also tax the hot path (one round-trip per accepted
|
||||
connection) and turn every spawn into a context-switch point. OTP has the
|
||||
same async `spawn` and puts the synchronisation one level up:
|
||||
`gen_server:start_link` blocks the caller until `init/1` returns.
|
||||
Fix shape when scheduled: a readiness ack in the supervisor's child-start
|
||||
path (`ChildSpec` variant whose factory receives a ready-signal;
|
||||
`NamedGenServerBuilder::run` acks after its name bind, gen_server default
|
||||
acks after `init`; plain closures ack at spawn as today, i.e. opt-in with
|
||||
no cost to existing children). Until then the workaround is structural:
|
||||
have the registrar spawn its own consumers so the ordering is program
|
||||
order inside one actor, not a cross-actor guarantee (urus v0.3 endpoint
|
||||
does exactly this).
|
||||
|
||||
## Invariants & gotchas (respect these across all cycles)
|
||||
|
||||
- **Shared mutex is non-reentrant.** `Sender::send` can call `unpark` →
|
||||
`with_shared`. Never send on a channel while holding the shared lock. Pattern:
|
||||
`mem::take` data under the lock, send after releasing. See `finalize_actor`.
|
||||
- **`finalize_actor` order:** take stack/waiters/monitors under lock + set
|
||||
Done/outcome → recycle stack → deliver supervisor Signal + monitor Downs →
|
||||
unpark joiners → reclaim slot if `outstanding_handles==0`. Death notifications
|
||||
always precede reclamation.
|
||||
- **Slot lifecycle reset in THREE places:** `Slot::vacant()`, `reclaim_slot()`
|
||||
(runtime.rs), slot-init block in `spawn_under` (scheduler.rs). Any new `Slot`
|
||||
field must be reset in all three.
|
||||
- **Pid = (index, generation).** Stale handles caught by generation mismatch in
|
||||
`slot()/slot_mut()`. The monitor `NoProc` path relies on this.
|
||||
- **The only wildcard wake is `request_stop`, and it is terminal.** Every
|
||||
registration-based waker (channel sends, mutex grants, wait-timers, io
|
||||
completions, joiner wakes, `select` arms) carries the wait's park-epoch
|
||||
and wakes through `unpark_at`; every successful wake consumes the epoch.
|
||||
Wakes are therefore *meaningful*: one-shot park sites interpret them
|
||||
without loops, and `select` needs no cancellation pass. When adding a new
|
||||
waker, decide which form it is — if its registration handle can outlive
|
||||
the wait it was created for, it MUST be epoch-stamped; a wait that can
|
||||
exit without parking MUST `retire_wait` first (see slot_state.rs).
|
||||
- **`select` exists; a unified per-process mailbox still does not.** The
|
||||
supervisor keeps its single `supervisor_channel` funnel; `recv_match`
|
||||
stays per-channel. `select` composes channels at the wait, not into one
|
||||
queue — gen_server's `handle_info`/`handle_down` (v0.8) are built on
|
||||
exactly that composition, with documented arm priority (downs → control
|
||||
→ infos → inbox) instead of mailbox FIFO. A hot higher-priority arm
|
||||
starves lower ones by design; that's the contract.
|
||||
- **Cooperative-only.** Preemption and cancellation both depend on the actor
|
||||
reaching `check!()`/yield/alloc/blocking points.
|
||||
- **Lock order is Leaf → Channel, one of each at most** (debug-asserted in
|
||||
`raw_mutex.rs`). Leaf = cold locks / free list / stack pool / registry,
|
||||
mutual leaves. A channel lock may be taken under a Leaf (finalize/monitor
|
||||
clone senders living in slots); nothing may be locked under a channel lock.
|
||||
- **Queue ops require preemption disabled.** A producer suspended mid-publish
|
||||
stalls every consumer — livelock. `with_runtime`, `with_shared`, and
|
||||
`RawMutex` guards all disable preemption for their span.
|
||||
- **`run()` is single-thread** (`Config::exact(1)`); tests rely on deterministic
|
||||
single-thread ordering. Multi-thread via `runtime::init(Config…)`.
|
||||
|
||||
-229
@@ -1,229 +0,0 @@
|
||||
# urus / smarm handoff — updated 2026-08-19 (session 3)
|
||||
|
||||
## TL;DR for the next session
|
||||
**smarm is done for now** (5 unpushed commits on local `master`, see below).
|
||||
**Next = urus v0.3 endpoint refactor.** You should NOT need to read smarm
|
||||
scheduler internals; the contract you build on is fully described here and in
|
||||
`smarm_full/examples/graceful_shutdown.rs` (read that file first — it is the
|
||||
exact shape urus's tree will take) plus `smarm_full/tests/root_exit.rs`.
|
||||
|
||||
### The smarm contract urus builds on (all on local master, verified by tests)
|
||||
- `request_stop(pid)` = kill (cooperative hard stop). `request_shutdown(pid)` =
|
||||
polite: trapping target gets `ExitSignal{reason: Shutdown}`, non-trapping is
|
||||
stopped outright. `RuntimeHandle::{request_stop,request_shutdown}` do the same
|
||||
from any OS thread (signal handler); grab `rt.handle()` before `rt.run`.
|
||||
- Supervisor traps; `request_shutdown(sup)` = ordered reverse-start shutdown,
|
||||
per-child `ChildSpec::shutdown(Shutdown::{Timeout(d)|Infinity|BrutalKill})`
|
||||
(default Timeout(5s)); sup then returns normally. `request_stop(sup)`
|
||||
hard-stops children too (no orphans).
|
||||
- gen_server: `ctx.trap_exit()` in init; `handle_shutdown() -> Exit|Continue`;
|
||||
`handle_exit(sig)`; `ctx.stop_handle().stop()` = normal self-exit;
|
||||
`terminate()` may block only on the graceful path (Exit / stop / inbox close).
|
||||
`GenServerRef::shutdown()` is graceful and waits.
|
||||
- gen_statem: same in event clothes — `cx.trap_exit()` in initial enter,
|
||||
`shutdown` rows (default `stop`), `exit sig` rows, `cx.stop()` / `stop` tail,
|
||||
optional `terminate { }` block. `GenStatemRef::shutdown()`.
|
||||
- **Root exit = program done**: when the root actor returns, the runtime
|
||||
`request_shutdown`s every *forest root* (live actor whose parent is the run
|
||||
or dead). Supervisors cascade; trapping actors may drain (timers keep
|
||||
working) and end the run when they stop; non-trapping are stopped; **no
|
||||
forcing sweep** (`join` what must finish). The old "wait until nothing
|
||||
runnable then kill all" deferral is gone.
|
||||
- **Gotcha for urus:** a gen_server's lifetime is governed by its refs — drop
|
||||
the last `GenServerRef` and the inbox closes → clean exit *even mid-drain*.
|
||||
The endpoint must be pinned (named, or its ref held by the supervisor
|
||||
wrapper) or it will terminate the moment the root drops its ref.
|
||||
- **Known gap (ROADMAP open item):** a gen_server can't be a direct `ChildSpec`
|
||||
child; use the trapping wrapper pattern in `examples/graceful_shutdown.rs::
|
||||
drainer_child` (starts `under(self_pid())`, forwards shutdown, waits). Doing
|
||||
an inline `GenServerBuilder::run()` first may be worth a short smarm detour —
|
||||
decide with Markk.
|
||||
|
||||
### smarm commits this session (local master, NOT pushed, NOT tagged)
|
||||
`250f312` root-exit = graceful shutdown of forest roots (tests/root_exit.rs)
|
||||
`6ceb138` gen_statem shutdown parity (tests/gen_statem_shutdown.rs)
|
||||
`849a424` docs + examples/graceful_shutdown.rs + README "Stopping actors"
|
||||
On top of `1002777` (cross-thread wake) and `9c8f59c` (graceful shutdown).
|
||||
Cargo.toml still `0.6.1`. Release cut (push, tag — v0.7 is justified by the
|
||||
API surface — version bump) is Markk's. Full suite, doc tests, examples,
|
||||
`cargo fmt`, `cargo clippy --lib` all clean. (`clippy --tests` has pre-existing
|
||||
unwrap lints in tests/fd_select.rs, untouched.)
|
||||
|
||||
### Decisions taken this session (Markk)
|
||||
- Root exit means "program done" (Go/tokio/OTP), not "wait for pending work";
|
||||
the previously agreed "sleep(50ms) must finish" test was dropped as encoding
|
||||
the wrong contract (a timer-wheel gate would re-wedge periodic-timer daemons).
|
||||
- No behaviour-preserving deferral, no forcing second sweep.
|
||||
- Examples/docs done in the same session; urus next session.
|
||||
|
||||
---
|
||||
# Previous handoff (still accurate where not superseded above)
|
||||
|
||||
|
||||
## Next-session goal
|
||||
Phase 1 is **done and committed**; Phase 2 is next:
|
||||
1. **smarm v0.6.2** — cross-thread wake root fix. **DONE**, committed on `master`
|
||||
as `1002777`. Not yet tagged, not yet version-bumped (Cargo.toml still reads
|
||||
`0.6.1`), and **not yet pushed to origin** — it exists only in the delivered
|
||||
snapshot zip and the local sandbox clone. Cutting the release (push + tag
|
||||
`v0.6.2` + bump `0.6.1`→`0.6.2`) is Markk's step.
|
||||
2. **urus v0.3** — endpoint refactor. Working against `smarm = { path = "../smarm_full" }`
|
||||
with `git update-index --skip-worktree Cargo.toml` (Markk approved); release commit
|
||||
swaps back to the tag once Markk cuts it. Phase 2 plan below is STALE where it says
|
||||
drain-in-terminate; the endpoint is a trapping GenServer: `handle_shutdown` →
|
||||
`Continue`, enter Draining, `StopHandle::stop()` when the conn set empties.
|
||||
|
||||
Decisions below are locked unless marked *(confirm)*.
|
||||
|
||||
## Reconstruction (the sandbox resets between sessions)
|
||||
A fresh sandbox has an empty home and **no Rust toolchain**. To restore:
|
||||
- Install rustup/cargo. smarm reformats under **rustc 1.97.1**; urus `rust-version`
|
||||
is 1.95. Use 1.97.1.
|
||||
- urus: `git clone https://git.kalsbeek.dev/Markk116/urus` — `origin` is registered
|
||||
and public-read. master `8bdec97` = the v0.2.x line. (Zips in outputs are stale;
|
||||
prefer the remote now.)
|
||||
- smarm: `git clone https://git.kalsbeek.dev/Markk116/smarm`. Latest tag **v0.6.1**
|
||||
(`ca1c983`). The cross-thread wake fix is committed as `1002777` on top of the
|
||||
post-v0.6.1 README commit `8f2d513` (= origin/master). **It is NOT on origin
|
||||
yet** — a fresh clone won't have it until Markk pushes. Restore it from the
|
||||
snapshot zip if working before the push. **v0.6.2 is not yet tagged.**
|
||||
- urus pins smarm by git **tag** in `Cargo.toml` (currently `v0.6.0`). A trivial
|
||||
first commit bumps it to `v0.6.1` (also picks up `try_spawn` + monitor
|
||||
terminal-outcome fixes).
|
||||
|
||||
## Why (context — the finding that drives the plan)
|
||||
urus's shutdown machinery (the `AtomicBool` listener flag + the `SHUTDOWN_POLL`
|
||||
loop in `serve.rs`) is scaffolding around two smarm properties. Their statuses
|
||||
differ, which is the whole point:
|
||||
|
||||
- **Issue A — lossy stop vs a QUEUED actor: ALREADY FIXED in smarm.** Commit
|
||||
`7bab4d2` added an entry-side `check_cancelled()` in `park_current`. A
|
||||
`request_stop` against a listener parked in `wait_readable_timeout` now unwinds
|
||||
cleanly (it parks via `try_select_timeout → park_current`). urus's flag + its
|
||||
stale "smarm's lossy stop-while-QUEUED window" comment can be deleted.
|
||||
- **Issue B — foreign-thread wake is a no-op: FIXED in `1002777` (was present
|
||||
through v0.6.1).** The gap: `unpark`/`unpark_at`/`request_stop` all route through
|
||||
`try_with_runtime`, which reads a thread-local that is `None` on any non-scheduler
|
||||
thread, so no cross-thread wake worked — a signal handler / OS thread could not
|
||||
wake *or* stop a parked actor, which is why `serve.rs` polls the shutdown signal
|
||||
instead of parking on it. Now closed (see Phase 1 below): urus's `SHUTDOWN_POLL`
|
||||
loop can be deleted and its `Handle::shutdown` can park on a handle-driven stop.
|
||||
|
||||
## Phase 1 — smarm cross-thread wake (root fix) — DONE (`1002777`)
|
||||
Shipped as one commit generalizing RFC 018 (a producer reaches the runtime through
|
||||
a `Weak` it holds) from the IO backend to channel senders and a new handle:
|
||||
- **`Runtime::handle() -> RuntimeHandle`** (`Send + Sync`), holding a
|
||||
`Weak<RuntimeInner>`. Grab it before `rt.run` and hand it to the signal thread.
|
||||
- **`RuntimeHandle::request_stop<A>(Pid<A>)`** — upgrades the Weak and calls
|
||||
`request_stop_inner` on the inner; no-op if the runtime is gone. This is the
|
||||
signal-handler-drives-shutdown path; it cascades the ordered stop down the tree
|
||||
exactly like an in-runtime `request_stop`.
|
||||
- **Send-wake:** the receiver captures `scheduler::runtime_weak()` into its
|
||||
`parked_receiver` tuple **at park time** (not at channel creation — the resolved
|
||||
sub-decision; a parked receiver is a live actor so the Weak is provably upgradable,
|
||||
and it scopes the capture to when a wake is possible). `send()` and last-sender
|
||||
`drop` wake via `scheduler::unpark_at_via(pid, epoch, &weak)`: thread-local path
|
||||
when on a scheduler thread (preempt-gated, slot-eligible), captured Weak otherwise.
|
||||
In-runtime timer wakes (recv/select) were left on `scheduler::unpark_at`.
|
||||
|
||||
**API scope decision (signed off):** `RuntimeHandle` exposes **`request_stop` only**.
|
||||
No public `unpark`/`unpark_at` on the handle — send-wake needs no user-facing handle,
|
||||
and "unpark off-runtime" is covered because `request_stop` drives `unpark` on the
|
||||
upgraded inner. No `is_alive()`. Both are one-line additions if a consumer appears.
|
||||
|
||||
**No RFC written** — pattern was already established (RFC 018), agreed not needed.
|
||||
|
||||
Tests: `tests/cross_thread_wake.rs` (foreign-thread send wakes a parked receiver;
|
||||
foreign-thread `request_stop` wakes+stops a parked actor; a lingering handle never
|
||||
blocks all-done and degrades to a no-op once the runtime drops). Full suite green;
|
||||
`cargo fmt` + `cargo clippy --lib` clean.
|
||||
|
||||
**Remaining release step (Markk):** push `master`, tag `v0.6.2`, bump Cargo.toml
|
||||
`0.6.1`→`0.6.2`. Left paired with the tag as the release cut, not done in `1002777`.
|
||||
|
||||
## Phase 1b — smarm graceful shutdown (OTP lift) — DONE (`9c8f59c`, on top of `1002777`)
|
||||
Decided this session (Markk): B — fix at the smarm level rather than a two-stop
|
||||
split in urus. No RFC (Markk: "just implement it"). Shipped, tested, committed on
|
||||
the local `master`, **not pushed, not tagged**. It should ship as the same
|
||||
release as 1002777 (v0.6.2, or v0.7 given the API surface — Markk's call).
|
||||
- `request_shutdown(pid)` / `RuntimeHandle::request_shutdown` = `exit(Pid, shutdown)`;
|
||||
`request_stop` = `exit(Pid, kill)`. Trapping target gets `ExitSignal{reason:
|
||||
DownReason::Shutdown}`; non-trapping is stopped outright.
|
||||
- `ChildSpec::shutdown(Shutdown::{BrutalKill, Timeout(d), Infinity})`, default 5s.
|
||||
Supervisor traps exits; `request_shutdown(sup)` = ordered top-down shutdown,
|
||||
returns normally. **Also fixed**: `request_stop(sup)` used to ORPHAN children
|
||||
(probe-verified; the handoff's "cascade" claim was wrong) — `Live` drop guard now
|
||||
hard-stops them.
|
||||
- gen_server: `ctx.trap_exit()`, `handle_shutdown() -> ShutdownAction::{Exit,
|
||||
Continue}`, `handle_exit(ExitSignal)`, `ctx.stop_handle().stop()` = normal
|
||||
self-exit (`{stop, normal}`; previously impossible — only abnormal `Stopped`).
|
||||
`GenServerRef::shutdown()` is graceful now.
|
||||
- Root finding that forced this: gen_server `terminate()` runs from a Drop guard,
|
||||
mid-unwind on the stop path; any park in it = double panic = abort. So
|
||||
"drain-in-terminate()" (the old Phase 2 plan) was never viable.
|
||||
|
||||
### ~~Next-session smarm work~~ DONE this session (see TL;DR)
|
||||
1. **Root-exit sweep**: make `Pop::RootDrain` also require an empty timer wheel
|
||||
(and it already requires nothing runnable; io_out is only checked for AllDone —
|
||||
check whether it should gate RootDrain too). TDD: an actor in `sleep(50ms)` when
|
||||
the root returns must finish, not be swept. Then a `Reservoir`-style test that a
|
||||
*truly* parked-forever daemon still gets swept.
|
||||
2. **gen_statem parity**: `ctx.trap_exit()`, `handle_shutdown -> ShutdownAction`,
|
||||
`handle_exit`, stop handle. Mirror gen_server; mechanical.
|
||||
3. **Examples review**: `examples/*.rs` predate all of this. Rework where they show
|
||||
shutdown/teardown to use `request_shutdown`, `Shutdown` policies, and
|
||||
`StopHandle`; `named_genserver.rs` first (uses `shutdown`). Also
|
||||
`docs/smarm - Deep Dive.html` says terminate() must be non-blocking — now only
|
||||
true on the unwind paths; and README could use a "Stopping actors" paragraph
|
||||
(request_stop = kill, request_shutdown = shutdown, Shutdown policy).
|
||||
|
||||
### Open smarm items found on the way (noted, not scheduled)
|
||||
- (root-exit sweep and gen_statem parity moved up to the scheduled list.)
|
||||
- Sweep in the supervisor `Live` drop guard is `request_stop` (kill propagates as
|
||||
kill); OTP would deliver a trappable `killed`. Chosen for boundedness.
|
||||
|
||||
## Phase 2 — urus v0.3: endpoint refactor (after v0.6.2 is tagged)
|
||||
Target = the spec's original shape (`urus-spec.md` §2.1/§6: `listener_sup` under the
|
||||
**user's** root supervisor). Deviation to unwind: `serve` owning `rt.run`.
|
||||
- App owns the runtime: `smarm::init(cfg).run(|| root_sup.run())`, root e.g.
|
||||
`RestForOne[ app actors…, urus::endpoint(config, pipeline) ]`. This is what kills
|
||||
the `Arc<OnceLock>` idiom for the right reason (app state born in-runtime as a
|
||||
supervised, ordered child).
|
||||
- `urus::endpoint` = one GenServer child owning the registry + an **internal**
|
||||
listener sub-supervisor + drain-in-`terminate()`. Listeners stay internal, not
|
||||
app-visible peers.
|
||||
- Shutdown = `request_stop` the root supervisor (or via the runtime handle from a
|
||||
signal thread) → cascades down → `endpoint.terminate()` runs the drain
|
||||
(`drain_timeout`, force-stop sweep).
|
||||
- **DELETE:** the `AtomicBool` listener flag (A fixed) and the `SHUTDOWN_POLL` loop
|
||||
+ its apologetic comment (B fixed → park, don't poll).
|
||||
- Nuance: `request_stop` → `Signal::Stopped` is *abnormal* → `Transient` restarts.
|
||||
Stop-without-restart = stop the **supervisor**, not the children.
|
||||
- *(confirm)* Keep `serve`/`serve_with`/`serve_with_shutdown` as thin wrappers that
|
||||
build the one-child tree internally, so the simple case stays one line.
|
||||
- *(confirm)* Keep `Handle`/`ShutdownSignal`? Now that cross-thread wake works,
|
||||
`Handle::shutdown` can map to handle-driven `request_stop` on the endpoint.
|
||||
- Breaking → cut **urus v0.3**; bump the smarm pin to `v0.6.2` here.
|
||||
|
||||
## Working norms
|
||||
- Every bash call: `export PATH=$HOME/.cargo/bin:$PATH` (once the toolchain's in).
|
||||
- **TDD**: failing test first, then implement; keep suites green.
|
||||
- **Hammer ritual** for ANY connection-lifecycle change (the urus #2 shutdown work
|
||||
qualifies): 35× subset (`shutdown timeout reaped slowloris streaming chunked sse
|
||||
stalled ws_ channels session`) + 3× full + 1× trace. `scripts/hammer.sh` does NOT
|
||||
pass feature flags — loop manually with `--features phoenix`. Subset filter must
|
||||
NOT use `--test integration` (session tests live in lib).
|
||||
- Example smoke tests: hold the server's stdin open (`mkfifo` + `sleep > fifo`) or
|
||||
the Enter-to-shutdown thread fires on EOF instantly.
|
||||
- Background procs are reaped BETWEEN bash calls; `pkill -f` matches your own shell.
|
||||
- Artefact store (specs): `curl -H "Authorization: Bearer sk-llmingest-2e45d80c63db24c6781e761eb2a9a58e83d9f48ef77a42185bad311d07c80e68" https://artefacts.kalsbeek.dev/artifacts/<name>`
|
||||
— `urus-spec.md`, `urus-bench-spec.md`, `rfc_008-implementation-notes.md`, …
|
||||
- smarm feature flags: `smarm-trace`, `smarm-causal` (urus re-exports both).
|
||||
|
||||
## Local cross-repo testing — KEEP OUT OF COMMITS
|
||||
To test urus #2 against un-tagged smarm 0.6.2, point urus's `Cargo.toml` smarm dep
|
||||
at a local path (`smarm = { path = "../smarm" }`) instead of the git tag.
|
||||
- Must NOT land in commits. Guard: `git update-index --skip-worktree Cargo.toml`
|
||||
after editing (undo with `--no-skip-worktree`), or stash before committing.
|
||||
- The committed `Cargo.toml` stays pinned to the git tag; restore the tag (bumped to
|
||||
`v0.6.2`) for the release commit.
|
||||
@@ -1,89 +0,0 @@
|
||||
# Benches
|
||||
|
||||
Two families live here: **comparison benches** (smarm vs tokio, predating
|
||||
v0.5) and the **run-queue shootout** (v0.5 phase 4). All are plain binaries
|
||||
(`harness = false` in `Cargo.toml`), so `cargo bench` just builds in release
|
||||
and runs `main()` — no criterion, no magic.
|
||||
|
||||
```
|
||||
cargo bench --bench <name> # one bench
|
||||
cargo bench # all of them (slow; rarely what you want)
|
||||
```
|
||||
|
||||
## Catalog
|
||||
|
||||
| file | what it measures |
|
||||
|---|---|
|
||||
| `primes.rs` | Compute fan-out/fan-in: counts primes across W workers. Pure compute throughput + spawn/join/channel cost. |
|
||||
| `multi_scheduler.rs` | The original cross-runtime matrix: smarm (1 thread / N threads) vs tokio (current_thread / multi_thread) on compute, ping-pong, and spawn throughput. |
|
||||
| `general.rs` | Workloads where neither runtime has a structural edge. Large gaps here mean real per-task/per-yield overhead differences — watch these for regressions. |
|
||||
| `smarm_favored.rs` | Workloads the stackful green-thread model is built for. Single-thread numbers isolate per-switch cost from contention. |
|
||||
| `tokio_favored.rs` | Workloads tokio's model is built for. Expect to lose; the value is knowing *by how much* and catching the gap widening. |
|
||||
| `rq_micro.rs` | Run-queue **structures** in isolation (no runtime, no actors): push/pop throughput sweeping thread count × producer:consumer ratio. Covers all three queue types in one binary — the types compile in every build; only the runtime's alias is feature-selected. |
|
||||
| `rq_runtime.rs` | The **whole scheduler** with the compile-time-selected queue: yield-storm (pure queue churn), ping-pong-pairs (park/unpark latency), spawn-storm (slab + free list + queue churn), sweeping scheduler count. Comparing variants requires rebuilding per `rq-*` feature. |
|
||||
|
||||
## The run-queue shootout
|
||||
|
||||
One command; it rebuilds `rq_runtime` once per queue variant, runs `rq_micro`
|
||||
once, and aggregates:
|
||||
|
||||
```
|
||||
./scripts/bench_rq.sh
|
||||
# on a big box:
|
||||
SMARM_BENCH_THREADS="1 2 4 8 16 20" ./scripts/bench_rq.sh
|
||||
```
|
||||
|
||||
Outputs land in `bench_results/` (gitignored): one full log per run, plus
|
||||
`summary.csv` assembled from the machine-readable `RQCSV,...` lines every
|
||||
config prints alongside the human table.
|
||||
|
||||
Manual single-variant runs need the feature dance (features are additive, so
|
||||
the default `rq-mutex` must be switched off):
|
||||
|
||||
```
|
||||
cargo bench --bench rq_runtime --no-default-features --features rq-striped
|
||||
```
|
||||
|
||||
### Knobs (env vars, all optional)
|
||||
|
||||
| var | default | used by |
|
||||
|---|---|---|
|
||||
| `SMARM_BENCH_THREADS` | `"1 2 4"` | both — space-separated sweep |
|
||||
| `SMARM_BENCH_RUNS` | `5` | both — repetitions; the **median** is reported |
|
||||
| `SMARM_BENCH_ITEMS` | `200000` | `rq_micro` — items per measurement |
|
||||
| `SMARM_BENCH_YIELD_ACTORS` / `_YIELDS` | `200` / `500` | `rq_runtime` yield-storm |
|
||||
| `SMARM_BENCH_PAIRS` / `_ROUNDTRIPS` | `32` / `1000` | `rq_runtime` ping-pong |
|
||||
| `SMARM_BENCH_SPAWNS` | `5000` | `rq_runtime` spawn-storm |
|
||||
|
||||
## Reading the numbers honestly
|
||||
|
||||
- **Core count is the experiment.** On a 1-core machine (CI, sandboxes) the
|
||||
sweep only validates the harness and catches gross pathologies —
|
||||
oversubscribed schedulers measure context-switch noise, not contention.
|
||||
Variant decisions come from a many-core box.
|
||||
- The striped queue *should lose* at low thread counts (ticket overhead with
|
||||
no contention to amortize) — that's expected, not a bug.
|
||||
- Medians over `SMARM_BENCH_RUNS` absorb scheduling noise but not thermal /
|
||||
turbo drift; for publishable numbers, pin the CPU governor and run a warmup
|
||||
pass first.
|
||||
- `spawn-storm` batches joins (1024 at a time) to stay well under the slab
|
||||
cap; if you raise `SMARM_BENCH_SPAWNS` massively, that batching is why it
|
||||
still works.
|
||||
|
||||
## Adding a bench
|
||||
|
||||
1. `benches/<name>.rs` with a plain `main()`; print the house table (see any
|
||||
existing bench) and, if it belongs to a sweep, a greppable CSV line with a
|
||||
distinctive prefix (`RQCSV,` for the shootout family).
|
||||
2. Register it in `Cargo.toml`:
|
||||
```toml
|
||||
[[bench]]
|
||||
name = "<name>"
|
||||
harness = false
|
||||
```
|
||||
3. Take parameters from `SMARM_BENCH_*` env vars with modest defaults — the
|
||||
defaults must finish in seconds on one core, the env scales them up on
|
||||
real hardware.
|
||||
4. Report **medians**, and keep one measurement = one fresh runtime
|
||||
(`init(Config::exact(t))` inside the measured closure constructor, the
|
||||
`run()` inside the timed region) so runs don't contaminate each other.
|
||||
+286
-338
@@ -1,366 +1,314 @@
|
||||
{
|
||||
"catch_unwind_panics": {
|
||||
"smarm 1-thread": {
|
||||
"result": 10000,
|
||||
"median": 122512,
|
||||
"min": 113735,
|
||||
"max": 124564
|
||||
},
|
||||
"smarm 20-thread": {
|
||||
"result": 10000,
|
||||
"median": 20569,
|
||||
"min": 19671,
|
||||
"max": 21174
|
||||
},
|
||||
"tokio current_thread": {
|
||||
"result": 10000,
|
||||
"median": 10134,
|
||||
"min": 9081,
|
||||
"max": 10524
|
||||
},
|
||||
"tokio multi-thread": {
|
||||
"result": 10000,
|
||||
"median": 2461,
|
||||
"min": 2351,
|
||||
"max": 2715
|
||||
}
|
||||
},
|
||||
"chained_spawn": {
|
||||
"smarm 1-thread": {
|
||||
"result": 1000,
|
||||
"median": 498,
|
||||
"min": 491,
|
||||
"max": 503
|
||||
"median": 266,
|
||||
"min": 242,
|
||||
"max": 351
|
||||
},
|
||||
"smarm 20-thread": {
|
||||
"smarm 24-thread": {
|
||||
"result": 1000,
|
||||
"median": 2135,
|
||||
"min": 1881,
|
||||
"max": 2284
|
||||
"median": 742,
|
||||
"min": 696,
|
||||
"max": 860
|
||||
},
|
||||
"tokio current_thread": {
|
||||
"result": 1000,
|
||||
"median": 114,
|
||||
"min": 106,
|
||||
"max": 119
|
||||
"median": 62,
|
||||
"min": 61,
|
||||
"max": 68
|
||||
},
|
||||
"tokio multi-thread": {
|
||||
"result": 1000,
|
||||
"median": 164,
|
||||
"min": 160,
|
||||
"max": 186
|
||||
}
|
||||
},
|
||||
"deep_recursion": {
|
||||
"smarm 1-thread": {
|
||||
"result": 1,
|
||||
"median": 316,
|
||||
"min": 314,
|
||||
"max": 322
|
||||
},
|
||||
"smarm 20-thread": {
|
||||
"result": 1,
|
||||
"median": 758,
|
||||
"min": 725,
|
||||
"max": 1222
|
||||
},
|
||||
"tokio current_thread": {
|
||||
"result": 1,
|
||||
"median": 11,
|
||||
"min": 11,
|
||||
"max": 14
|
||||
},
|
||||
"tokio multi-thread": {
|
||||
"result": 1,
|
||||
"median": 52,
|
||||
"min": 51,
|
||||
"max": 55
|
||||
}
|
||||
},
|
||||
"fan_out_compute": {
|
||||
"smarm 1-thread": {
|
||||
"result": 33860,
|
||||
"median": 15074,
|
||||
"min": 15064,
|
||||
"max": 15081
|
||||
},
|
||||
"smarm 20-thread": {
|
||||
"result": 33860,
|
||||
"median": 2786,
|
||||
"min": 2539,
|
||||
"max": 2913
|
||||
},
|
||||
"tokio current_thread": {
|
||||
"result": 33860,
|
||||
"median": 14040,
|
||||
"min": 13749,
|
||||
"max": 14056
|
||||
},
|
||||
"tokio multi-thread": {
|
||||
"result": 33860,
|
||||
"median": 2404,
|
||||
"min": 2136,
|
||||
"max": 2436
|
||||
}
|
||||
},
|
||||
"many_timers": {
|
||||
"smarm 1-thread": {
|
||||
"result": 10000,
|
||||
"median": 117779,
|
||||
"min": 107641,
|
||||
"max": 120785
|
||||
},
|
||||
"smarm 20-thread": {
|
||||
"result": 10000,
|
||||
"median": 54031,
|
||||
"min": 53646,
|
||||
"max": 54841
|
||||
},
|
||||
"tokio current_thread": {
|
||||
"result": 10000,
|
||||
"median": 12417,
|
||||
"min": 12383,
|
||||
"max": 12479
|
||||
},
|
||||
"tokio multi-thread": {
|
||||
"result": 10000,
|
||||
"median": 13880,
|
||||
"min": 13405,
|
||||
"max": 13953
|
||||
}
|
||||
},
|
||||
"mpsc_contention": {
|
||||
"smarm 1-thread": {
|
||||
"result": 320000,
|
||||
"median": 3260,
|
||||
"min": 2896,
|
||||
"max": 3427
|
||||
},
|
||||
"smarm 20-thread": {
|
||||
"result": 320000,
|
||||
"median": 41206,
|
||||
"min": 40562,
|
||||
"max": 41754
|
||||
},
|
||||
"tokio current_thread": {
|
||||
"result": 320000,
|
||||
"median": 5895,
|
||||
"min": 5272,
|
||||
"max": 5902
|
||||
},
|
||||
"tokio multi-thread": {
|
||||
"result": 320000,
|
||||
"median": 71069,
|
||||
"min": 60076,
|
||||
"max": 80026
|
||||
}
|
||||
},
|
||||
"multi_thread_scaling": {
|
||||
"smarm 1-thread": {
|
||||
"result": 33860,
|
||||
"median": 15163,
|
||||
"min": 15125,
|
||||
"max": 15172
|
||||
},
|
||||
"smarm 2-thread": {
|
||||
"result": 33860,
|
||||
"median": 7980,
|
||||
"min": 7947,
|
||||
"max": 8103
|
||||
},
|
||||
"smarm 4-thread": {
|
||||
"result": 33860,
|
||||
"median": 4324,
|
||||
"min": 4300,
|
||||
"max": 4364
|
||||
},
|
||||
"smarm 20-thread": {
|
||||
"result": 33860,
|
||||
"median": 2704,
|
||||
"min": 2391,
|
||||
"max": 2841
|
||||
},
|
||||
"tokio multi 1-thread": {
|
||||
"result": 33860,
|
||||
"median": 14271,
|
||||
"min": 14251,
|
||||
"max": 14459
|
||||
},
|
||||
"tokio multi 2-thread": {
|
||||
"result": 33860,
|
||||
"median": 7395,
|
||||
"min": 7195,
|
||||
"max": 7443
|
||||
},
|
||||
"tokio multi 4-thread": {
|
||||
"result": 33860,
|
||||
"median": 3740,
|
||||
"min": 3709,
|
||||
"max": 3752
|
||||
},
|
||||
"tokio multi 20-thread": {
|
||||
"result": 33860,
|
||||
"median": 2373,
|
||||
"min": 2176,
|
||||
"max": 2380
|
||||
}
|
||||
},
|
||||
"ping_pong_oneshot": {
|
||||
"smarm 1-thread": {
|
||||
"result": 1000,
|
||||
"median": 935,
|
||||
"min": 910,
|
||||
"max": 941
|
||||
},
|
||||
"smarm 20-thread": {
|
||||
"result": 1000,
|
||||
"median": 6168,
|
||||
"min": 6029,
|
||||
"max": 6550
|
||||
},
|
||||
"tokio current_thread": {
|
||||
"result": 1000,
|
||||
"median": 407,
|
||||
"min": 401,
|
||||
"max": 430
|
||||
},
|
||||
"tokio multi-thread": {
|
||||
"result": 1000,
|
||||
"median": 9395,
|
||||
"min": 8795,
|
||||
"max": 9812
|
||||
}
|
||||
},
|
||||
"ping_pong_steady": {
|
||||
"smarm 1-thread": {
|
||||
"result": 10000,
|
||||
"median": 1517,
|
||||
"min": 1501,
|
||||
"max": 1527
|
||||
},
|
||||
"smarm 20-thread": {
|
||||
"result": 10000,
|
||||
"median": 1910,
|
||||
"min": 1848,
|
||||
"max": 1947
|
||||
},
|
||||
"tokio current_thread": {
|
||||
"result": 10000,
|
||||
"median": 1284,
|
||||
"min": 1278,
|
||||
"max": 1296
|
||||
},
|
||||
"tokio multi-thread": {
|
||||
"result": 10000,
|
||||
"median": 82015,
|
||||
"min": 81833,
|
||||
"max": 82993
|
||||
}
|
||||
},
|
||||
"spawn_pair_control": {
|
||||
"smarm 1-thread": {
|
||||
"result": 1000,
|
||||
"median": 637,
|
||||
"min": 632,
|
||||
"max": 661
|
||||
},
|
||||
"smarm 20-thread": {
|
||||
"result": 1000,
|
||||
"median": 5118,
|
||||
"min": 5089,
|
||||
"max": 5157
|
||||
},
|
||||
"tokio current_thread": {
|
||||
"result": 1000,
|
||||
"median": 288,
|
||||
"min": 249,
|
||||
"max": 300
|
||||
},
|
||||
"tokio multi-thread": {
|
||||
"result": 1000,
|
||||
"median": 8617,
|
||||
"min": 8534,
|
||||
"max": 8808
|
||||
}
|
||||
},
|
||||
"spawn_storm_busy": {
|
||||
"smarm 1-thread": {
|
||||
"result": 10000,
|
||||
"median": 105937,
|
||||
"min": 101927,
|
||||
"max": 107104
|
||||
},
|
||||
"smarm 20-thread": {
|
||||
"result": 10000,
|
||||
"median": 15928,
|
||||
"min": 15918,
|
||||
"max": 16581
|
||||
},
|
||||
"tokio current_thread": {
|
||||
"result": 10000,
|
||||
"median": 1248,
|
||||
"min": 1121,
|
||||
"max": 1253
|
||||
},
|
||||
"tokio multi-thread": {
|
||||
"result": 10000,
|
||||
"median": 12313,
|
||||
"min": 11641,
|
||||
"max": 16848
|
||||
}
|
||||
},
|
||||
"uncontended_channel": {
|
||||
"smarm 1-thread": {
|
||||
"result": 1000000,
|
||||
"median": 12452,
|
||||
"min": 12315,
|
||||
"max": 13700
|
||||
},
|
||||
"tokio current_thread": {
|
||||
"result": 1000000,
|
||||
"median": 15091,
|
||||
"min": 15086,
|
||||
"max": 16944
|
||||
}
|
||||
},
|
||||
"yield_in_hot_loop": {
|
||||
"smarm 1-thread": {
|
||||
"result": 1000000,
|
||||
"median": 33187,
|
||||
"min": 33000,
|
||||
"max": 33325
|
||||
},
|
||||
"tokio current_thread": {
|
||||
"result": 1000000,
|
||||
"median": 74444,
|
||||
"min": 67009,
|
||||
"max": 75753
|
||||
"median": 190,
|
||||
"min": 169,
|
||||
"max": 207
|
||||
}
|
||||
},
|
||||
"yield_many": {
|
||||
"smarm 1-thread": {
|
||||
"result": 200000,
|
||||
"median": 10921,
|
||||
"min": 10828,
|
||||
"max": 11041
|
||||
"median": 19071,
|
||||
"min": 18776,
|
||||
"max": 19396
|
||||
},
|
||||
"smarm 20-thread": {
|
||||
"smarm 24-thread": {
|
||||
"result": 200000,
|
||||
"median": 44422,
|
||||
"min": 43715,
|
||||
"max": 44595
|
||||
"median": 172454,
|
||||
"min": 166246,
|
||||
"max": 174230
|
||||
},
|
||||
"tokio current_thread": {
|
||||
"result": 200000,
|
||||
"median": 5355,
|
||||
"min": 5327,
|
||||
"max": 5426
|
||||
"median": 4737,
|
||||
"min": 4644,
|
||||
"max": 5065
|
||||
},
|
||||
"tokio multi-thread": {
|
||||
"result": 200000,
|
||||
"median": 6170,
|
||||
"min": 5982,
|
||||
"max": 6898
|
||||
"median": 8738,
|
||||
"min": 7852,
|
||||
"max": 9770
|
||||
}
|
||||
},
|
||||
"fan_out_compute": {
|
||||
"smarm 1-thread": {
|
||||
"result": 33860,
|
||||
"median": 13234,
|
||||
"min": 13196,
|
||||
"max": 13390
|
||||
},
|
||||
"smarm 24-thread": {
|
||||
"result": 33860,
|
||||
"median": 2244,
|
||||
"min": 2162,
|
||||
"max": 2380
|
||||
},
|
||||
"tokio current_thread": {
|
||||
"result": 33860,
|
||||
"median": 14049,
|
||||
"min": 14035,
|
||||
"max": 14300
|
||||
},
|
||||
"tokio multi-thread": {
|
||||
"result": 33860,
|
||||
"median": 1474,
|
||||
"min": 1285,
|
||||
"max": 1823
|
||||
}
|
||||
},
|
||||
"ping_pong_oneshot": {
|
||||
"smarm 1-thread": {
|
||||
"result": 1000,
|
||||
"median": 751,
|
||||
"min": 727,
|
||||
"max": 913
|
||||
},
|
||||
"smarm 24-thread": {
|
||||
"result": 1000,
|
||||
"median": 1308,
|
||||
"min": 1227,
|
||||
"max": 1396
|
||||
},
|
||||
"tokio current_thread": {
|
||||
"result": 1000,
|
||||
"median": 407,
|
||||
"min": 400,
|
||||
"max": 444
|
||||
},
|
||||
"tokio multi-thread": {
|
||||
"result": 1000,
|
||||
"median": 10869,
|
||||
"min": 8683,
|
||||
"max": 11688
|
||||
}
|
||||
},
|
||||
"spawn_storm_busy": {
|
||||
"smarm 1-thread": {
|
||||
"result": 10000,
|
||||
"median": 112045,
|
||||
"min": 99936,
|
||||
"max": 117329
|
||||
},
|
||||
"smarm 24-thread": {
|
||||
"result": 10000,
|
||||
"median": 137105,
|
||||
"min": 130852,
|
||||
"max": 147707
|
||||
},
|
||||
"tokio current_thread": {
|
||||
"result": 10000,
|
||||
"median": 1128,
|
||||
"min": 1123,
|
||||
"max": 1435
|
||||
},
|
||||
"tokio multi-thread": {
|
||||
"result": 10000,
|
||||
"median": 19674,
|
||||
"min": 16013,
|
||||
"max": 27234
|
||||
}
|
||||
},
|
||||
"mpsc_contention": {
|
||||
"smarm 1-thread": {
|
||||
"result": 320000,
|
||||
"median": 3667,
|
||||
"min": 3608,
|
||||
"max": 4126
|
||||
},
|
||||
"smarm 24-thread": {
|
||||
"result": 320000,
|
||||
"median": 45681,
|
||||
"min": 31908,
|
||||
"max": 51287
|
||||
},
|
||||
"tokio current_thread": {
|
||||
"result": 320000,
|
||||
"median": 6228,
|
||||
"min": 6210,
|
||||
"max": 6514
|
||||
},
|
||||
"tokio multi-thread": {
|
||||
"result": 320000,
|
||||
"median": 66173,
|
||||
"min": 42208,
|
||||
"max": 83255
|
||||
}
|
||||
},
|
||||
"many_timers": {
|
||||
"smarm 1-thread": {
|
||||
"result": 10000,
|
||||
"median": 119988,
|
||||
"min": 107308,
|
||||
"max": 123557
|
||||
},
|
||||
"smarm 24-thread": {
|
||||
"result": 10000,
|
||||
"median": 218842,
|
||||
"min": 182009,
|
||||
"max": 256988
|
||||
},
|
||||
"tokio current_thread": {
|
||||
"result": 10000,
|
||||
"median": 12432,
|
||||
"min": 12308,
|
||||
"max": 13468
|
||||
},
|
||||
"tokio multi-thread": {
|
||||
"result": 10000,
|
||||
"median": 16311,
|
||||
"min": 15026,
|
||||
"max": 16897
|
||||
}
|
||||
},
|
||||
"multi_thread_scaling": {
|
||||
"smarm 1-thread": {
|
||||
"result": 33860,
|
||||
"median": 14908,
|
||||
"min": 14857,
|
||||
"max": 15218
|
||||
},
|
||||
"smarm 2-thread": {
|
||||
"result": 33860,
|
||||
"median": 7834,
|
||||
"min": 7717,
|
||||
"max": 8033
|
||||
},
|
||||
"smarm 4-thread": {
|
||||
"result": 33860,
|
||||
"median": 4393,
|
||||
"min": 4326,
|
||||
"max": 4435
|
||||
},
|
||||
"smarm 24-thread": {
|
||||
"result": 33860,
|
||||
"median": 2173,
|
||||
"min": 2068,
|
||||
"max": 2405
|
||||
},
|
||||
"tokio multi 1-thread": {
|
||||
"result": 33860,
|
||||
"median": 14432,
|
||||
"min": 14219,
|
||||
"max": 14763
|
||||
},
|
||||
"tokio multi 2-thread": {
|
||||
"result": 33860,
|
||||
"median": 7333,
|
||||
"min": 7222,
|
||||
"max": 7477
|
||||
},
|
||||
"tokio multi 4-thread": {
|
||||
"result": 33860,
|
||||
"median": 3741,
|
||||
"min": 3681,
|
||||
"max": 3876
|
||||
},
|
||||
"tokio multi 24-thread": {
|
||||
"result": 33860,
|
||||
"median": 1513,
|
||||
"min": 1375,
|
||||
"max": 1979
|
||||
}
|
||||
},
|
||||
"deep_recursion": {
|
||||
"smarm 1-thread": {
|
||||
"result": 1,
|
||||
"median": 102,
|
||||
"min": 96,
|
||||
"max": 123
|
||||
},
|
||||
"smarm 24-thread": {
|
||||
"result": 1,
|
||||
"median": 597,
|
||||
"min": 576,
|
||||
"max": 682
|
||||
},
|
||||
"tokio current_thread": {
|
||||
"result": 1,
|
||||
"median": 13,
|
||||
"min": 11,
|
||||
"max": 35
|
||||
},
|
||||
"tokio multi-thread": {
|
||||
"result": 1,
|
||||
"median": 56,
|
||||
"min": 46,
|
||||
"max": 65
|
||||
}
|
||||
},
|
||||
"yield_in_hot_loop": {
|
||||
"smarm 1-thread": {
|
||||
"result": 1000000,
|
||||
"median": 80680,
|
||||
"min": 80308,
|
||||
"max": 81845
|
||||
},
|
||||
"tokio current_thread": {
|
||||
"result": 1000000,
|
||||
"median": 72606,
|
||||
"min": 72154,
|
||||
"max": 77206
|
||||
}
|
||||
},
|
||||
"uncontended_channel": {
|
||||
"smarm 1-thread": {
|
||||
"result": 1000000,
|
||||
"median": 9257,
|
||||
"min": 9223,
|
||||
"max": 12049
|
||||
},
|
||||
"tokio current_thread": {
|
||||
"result": 1000000,
|
||||
"median": 16925,
|
||||
"min": 16848,
|
||||
"max": 17019
|
||||
}
|
||||
},
|
||||
"catch_unwind_panics": {
|
||||
"smarm 1-thread": {
|
||||
"result": 10000,
|
||||
"median": 116821,
|
||||
"min": 111345,
|
||||
"max": 128261
|
||||
},
|
||||
"smarm 24-thread": {
|
||||
"result": 10000,
|
||||
"median": 117487,
|
||||
"min": 107011,
|
||||
"max": 129307
|
||||
},
|
||||
"tokio current_thread": {
|
||||
"result": 10000,
|
||||
"median": 10425,
|
||||
"min": 10141,
|
||||
"max": 10604
|
||||
},
|
||||
"tokio multi-thread": {
|
||||
"result": 10000,
|
||||
"median": 6418,
|
||||
"min": 3715,
|
||||
"max": 7144
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,46 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Run-queue shootout driver (ROADMAP_v0.5 phase 4; RFC 005 slot dimension
|
||||
# added for the v0.9 slot shootout).
|
||||
#
|
||||
# Rebuilds the runtime bench once per rq-* feature and runs the raw-structure
|
||||
# microbench once (it covers all structures in a single binary). The RFC 005
|
||||
# wake slot is a runtime Config knob, NOT a feature — each rq_runtime binary
|
||||
# sweeps slot off/on internally (SMARM_BENCH_SLOT, default "0 1"). Results
|
||||
# land in bench_results/ as full logs; the RQCSV lines are aggregated into
|
||||
# bench_results/summary.csv and the RQSLOT counter lines (slot hits /
|
||||
# displacements, slot-on configs only) into bench_results/slot_counters.csv.
|
||||
#
|
||||
# Tune the sweep for the box, e.g. on the 20-core machine:
|
||||
# SMARM_BENCH_THREADS="1 2 4 8 16 20" ./scripts/bench_rq.sh
|
||||
# Slot-only re-run against the frozen rq-mutex substrate:
|
||||
# SMARM_BENCH_SLOT="0 1" cargo bench --bench rq_runtime
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")/.."
|
||||
|
||||
OUT=bench_results
|
||||
mkdir -p "$OUT"
|
||||
: "${SMARM_BENCH_THREADS:=1 2 4}"
|
||||
: "${SMARM_BENCH_SLOT:=0 1}"
|
||||
export SMARM_BENCH_THREADS SMARM_BENCH_SLOT
|
||||
|
||||
echo "== raw structures (one binary, all variants) =="
|
||||
cargo bench --bench rq_micro 2>&1 | tee "$OUT/micro.txt"
|
||||
|
||||
for v in rq-mutex rq-mpmc rq-striped; do
|
||||
echo "== runtime benches: $v (slot sweep: $SMARM_BENCH_SLOT) =="
|
||||
cargo bench --bench rq_runtime --no-default-features --features "$v" \
|
||||
2>&1 | tee "$OUT/runtime-$v.txt"
|
||||
done
|
||||
|
||||
# runtime rows: kind,variant,slot,bench,threads,work,median_us,ops_per_s
|
||||
# micro rows: kind,structure,threads,p:c,items,median_us,items_per_s (one
|
||||
# column narrower, as before — split on kind when plotting)
|
||||
echo "kind,a,b,c,d,e,median_us,ops_per_s" > "$OUT/summary.csv"
|
||||
grep -h '^RQCSV,' "$OUT"/*.txt | sed 's/^RQCSV,//' >> "$OUT/summary.csv"
|
||||
|
||||
echo "variant,bench,threads,slot_hits,slot_displacements" > "$OUT/slot_counters.csv"
|
||||
grep -h '^RQSLOT,' "$OUT"/*.txt | sed 's/^RQSLOT,//' >> "$OUT/slot_counters.csv" || true
|
||||
|
||||
echo
|
||||
echo "Summary: $OUT/summary.csv ($(($(wc -l < "$OUT/summary.csv") - 1)) rows)"
|
||||
echo "Slot counters: $OUT/slot_counters.csv ($(($(wc -l < "$OUT/slot_counters.csv") - 1)) rows)"
|
||||
+34
-247
@@ -13,18 +13,7 @@
|
||||
//! completeness.
|
||||
//! 4. ping_pong_oneshot — N rounds of (spawn pair, send oneshot, await).
|
||||
//! Closer to a request/response workload than channel
|
||||
//! ping-pong. NOTE: 2 spawns + 2 channel allocs + 2
|
||||
//! joins per round; the message path is a minority
|
||||
//! of it. Sections 5 and 6 split it apart.
|
||||
//! 5. spawn_pair_control — section 4 with the messages removed: same
|
||||
//! spawn/join shape, actors return immediately.
|
||||
//! (4 − 5) ≈ per-round message-path cost.
|
||||
//! 6. ping_pong_steady — ONE persistent pair, N roundtrips over unbounded
|
||||
//! MPSC channels (smarm::channel vs
|
||||
//! tokio::sync::mpsc::unbounded_channel — both
|
||||
//! unbounded, non-blocking send). Steady-state
|
||||
//! park/unpark cost per roundtrip, no spawn in the
|
||||
//! loop.
|
||||
//! ping-pong.
|
||||
|
||||
use std::sync::atomic::{AtomicU64, Ordering};
|
||||
use std::sync::Arc;
|
||||
@@ -37,16 +26,7 @@ use std::time::Instant;
|
||||
const ITERS: u32 = 15;
|
||||
|
||||
fn available_threads() -> usize {
|
||||
std::thread::available_parallelism()
|
||||
.map(|n| n.get())
|
||||
.unwrap_or(1)
|
||||
}
|
||||
|
||||
fn env_sets() -> u32 {
|
||||
std::env::var("SMARM_BENCH_SETS")
|
||||
.ok()
|
||||
.and_then(|v| v.parse().ok())
|
||||
.unwrap_or(5)
|
||||
std::thread::available_parallelism().map(|n| n.get()).unwrap_or(1)
|
||||
}
|
||||
|
||||
fn print_header(title: &str) {
|
||||
@@ -61,17 +41,14 @@ fn print_header(title: &str) {
|
||||
}
|
||||
|
||||
fn run_n<F: FnMut() -> (u64, u128)>(name: &str, n: u32, mut f: F) {
|
||||
let sets = env_sets();
|
||||
let mut times = Vec::with_capacity((n * sets) as usize);
|
||||
let mut times = Vec::new();
|
||||
let mut last = 0u64;
|
||||
// One warmup before all sets, discarded.
|
||||
// One warmup iteration, discarded.
|
||||
let _ = f();
|
||||
for _ in 0..sets {
|
||||
for _ in 0..n {
|
||||
let (v, t) = f();
|
||||
times.push(t);
|
||||
last = v;
|
||||
}
|
||||
for _ in 0..n {
|
||||
let (v, t) = f();
|
||||
times.push(t);
|
||||
last = v;
|
||||
}
|
||||
times.sort_unstable();
|
||||
let median = times[times.len() / 2];
|
||||
@@ -121,15 +98,17 @@ fn bench_chained_smarm(threads: usize) -> (u64, u128) {
|
||||
fn bench_chained_tokio_current() -> (u64, u128) {
|
||||
let counter = Arc::new(AtomicU64::new(0));
|
||||
let c2 = counter.clone();
|
||||
let rt = tokio::runtime::Builder::new_current_thread()
|
||||
.build()
|
||||
.unwrap();
|
||||
let rt = tokio::runtime::Builder::new_current_thread().build().unwrap();
|
||||
let start = Instant::now();
|
||||
let local = tokio::task::LocalSet::new();
|
||||
local.block_on(&rt, async move {
|
||||
// Use a oneshot done channel like tokio's own chained_spawn bench.
|
||||
let (done_tx, done_rx) = tokio::sync::oneshot::channel();
|
||||
fn iter(c: Arc<AtomicU64>, done: tokio::sync::oneshot::Sender<()>, n: u64) {
|
||||
fn iter(
|
||||
c: Arc<AtomicU64>,
|
||||
done: tokio::sync::oneshot::Sender<()>,
|
||||
n: u64,
|
||||
) {
|
||||
if n == 0 {
|
||||
let _ = done.send(());
|
||||
} else {
|
||||
@@ -197,9 +176,7 @@ fn bench_yield_smarm(threads: usize) -> (u64, u128) {
|
||||
}
|
||||
|
||||
fn bench_yield_tokio_current() -> (u64, u128) {
|
||||
let rt = tokio::runtime::Builder::new_current_thread()
|
||||
.build()
|
||||
.unwrap();
|
||||
let rt = tokio::runtime::Builder::new_current_thread().build().unwrap();
|
||||
let start = Instant::now();
|
||||
let local = tokio::task::LocalSet::new();
|
||||
local.block_on(&rt, async move {
|
||||
@@ -248,22 +225,11 @@ const PRIME_N: u64 = 400_000;
|
||||
const PRIME_WORKERS: u64 = 64;
|
||||
|
||||
fn is_prime(n: u64) -> bool {
|
||||
if n < 2 {
|
||||
return false;
|
||||
}
|
||||
if n < 4 {
|
||||
return true;
|
||||
}
|
||||
if n % 2 == 0 {
|
||||
return false;
|
||||
}
|
||||
if n < 2 { return false; }
|
||||
if n < 4 { return true; }
|
||||
if n % 2 == 0 { return false; }
|
||||
let mut i = 3u64;
|
||||
while i * i <= n {
|
||||
if n % i == 0 {
|
||||
return false;
|
||||
}
|
||||
i += 2;
|
||||
}
|
||||
while i * i <= n { if n % i == 0 { return false; } i += 2; }
|
||||
true
|
||||
}
|
||||
|
||||
@@ -274,11 +240,7 @@ fn count_primes(lo: u64, hi: u64) -> u64 {
|
||||
fn primes_slice(w: u64) -> (u64, u64) {
|
||||
let per = PRIME_N / PRIME_WORKERS;
|
||||
let lo = w * per;
|
||||
let hi = if w + 1 == PRIME_WORKERS {
|
||||
PRIME_N
|
||||
} else {
|
||||
lo + per
|
||||
};
|
||||
let hi = if w + 1 == PRIME_WORKERS { PRIME_N } else { lo + per };
|
||||
(lo, hi)
|
||||
}
|
||||
|
||||
@@ -295,9 +257,7 @@ fn bench_primes_smarm(threads: usize) -> (u64, u128) {
|
||||
tc.fetch_add(count_primes(lo, hi), Ordering::Relaxed);
|
||||
}));
|
||||
}
|
||||
for h in handles {
|
||||
h.join().unwrap();
|
||||
}
|
||||
for h in handles { h.join().unwrap(); }
|
||||
});
|
||||
(total.load(Ordering::Relaxed), start.elapsed().as_micros())
|
||||
}
|
||||
@@ -305,9 +265,7 @@ fn bench_primes_smarm(threads: usize) -> (u64, u128) {
|
||||
fn bench_primes_tokio_current() -> (u64, u128) {
|
||||
let total = Arc::new(AtomicU64::new(0));
|
||||
let t2 = total.clone();
|
||||
let rt = tokio::runtime::Builder::new_current_thread()
|
||||
.build()
|
||||
.unwrap();
|
||||
let rt = tokio::runtime::Builder::new_current_thread().build().unwrap();
|
||||
let start = Instant::now();
|
||||
let local = tokio::task::LocalSet::new();
|
||||
local.block_on(&rt, async move {
|
||||
@@ -319,9 +277,7 @@ fn bench_primes_tokio_current() -> (u64, u128) {
|
||||
tc.fetch_add(count_primes(lo, hi), Ordering::Relaxed);
|
||||
}));
|
||||
}
|
||||
for h in handles {
|
||||
let _ = h.await;
|
||||
}
|
||||
for h in handles { let _ = h.await; }
|
||||
});
|
||||
(total.load(Ordering::Relaxed), start.elapsed().as_micros())
|
||||
}
|
||||
@@ -343,9 +299,7 @@ fn bench_primes_tokio_multi() -> (u64, u128) {
|
||||
tc.fetch_add(count_primes(lo, hi), Ordering::Relaxed);
|
||||
}));
|
||||
}
|
||||
for h in handles {
|
||||
let _ = h.await;
|
||||
}
|
||||
for h in handles { let _ = h.await; }
|
||||
});
|
||||
(total.load(Ordering::Relaxed), start.elapsed().as_micros())
|
||||
}
|
||||
@@ -380,9 +334,7 @@ fn bench_pp_smarm(threads: usize) -> (u64, u128) {
|
||||
}
|
||||
|
||||
fn bench_pp_tokio_current() -> (u64, u128) {
|
||||
let rt = tokio::runtime::Builder::new_current_thread()
|
||||
.build()
|
||||
.unwrap();
|
||||
let rt = tokio::runtime::Builder::new_current_thread().build().unwrap();
|
||||
let start = Instant::now();
|
||||
let local = tokio::task::LocalSet::new();
|
||||
local.block_on(&rt, async move {
|
||||
@@ -429,139 +381,11 @@ fn bench_pp_tokio_multi() -> (u64, u128) {
|
||||
(PP_ROUNDS, start.elapsed().as_micros())
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 5. spawn_pair_control — section 4 minus the messages
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Target-5 instrumentation: emit the runtime's wake-path counters for one
|
||||
/// run when SMARM_WAKE_DIAG is set. Off by default so sweep.py output is
|
||||
/// unchanged.
|
||||
fn wake_diag(section: &str, threads: usize, rt: &smarm::runtime::Runtime, us: u128) {
|
||||
if std::env::var_os("SMARM_WAKE_DIAG").is_some() {
|
||||
println!("DIAG,{section},{threads},{us},{}", rt.stats().wake_diag());
|
||||
}
|
||||
}
|
||||
|
||||
fn bench_ctl_smarm(threads: usize) -> (u64, u128) {
|
||||
let start = Instant::now();
|
||||
let rt = smarm::runtime::init(bench_cfg(threads));
|
||||
rt.run(|| {
|
||||
for _ in 0..PP_ROUNDS {
|
||||
let hb = smarm::spawn(|| {});
|
||||
let ha = smarm::spawn(|| {});
|
||||
ha.join().unwrap();
|
||||
hb.join().unwrap();
|
||||
}
|
||||
});
|
||||
let us = start.elapsed().as_micros();
|
||||
wake_diag("spawn_pair_control", threads, &rt, us);
|
||||
(PP_ROUNDS, us)
|
||||
}
|
||||
|
||||
fn bench_ctl_tokio_current() -> (u64, u128) {
|
||||
let rt = tokio::runtime::Builder::new_current_thread()
|
||||
.build()
|
||||
.unwrap();
|
||||
let start = Instant::now();
|
||||
let local = tokio::task::LocalSet::new();
|
||||
local.block_on(&rt, async move {
|
||||
for _ in 0..PP_ROUNDS {
|
||||
let hb = tokio::task::spawn_local(async {});
|
||||
let ha = tokio::task::spawn_local(async {});
|
||||
let _ = ha.await;
|
||||
let _ = hb.await;
|
||||
}
|
||||
});
|
||||
(PP_ROUNDS, start.elapsed().as_micros())
|
||||
}
|
||||
|
||||
fn bench_ctl_tokio_multi() -> (u64, u128) {
|
||||
let rt = tokio::runtime::Builder::new_multi_thread()
|
||||
.worker_threads(available_threads())
|
||||
.build()
|
||||
.unwrap();
|
||||
let start = Instant::now();
|
||||
rt.block_on(async move {
|
||||
for _ in 0..PP_ROUNDS {
|
||||
let hb = tokio::spawn(async {});
|
||||
let ha = tokio::spawn(async {});
|
||||
let _ = ha.await;
|
||||
let _ = hb.await;
|
||||
}
|
||||
});
|
||||
(PP_ROUNDS, start.elapsed().as_micros())
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 6. ping_pong_steady — one persistent pair, PP_STEADY roundtrips
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const PP_STEADY: u64 = 10_000;
|
||||
|
||||
fn bench_steady_smarm(threads: usize) -> (u64, u128) {
|
||||
let start = Instant::now();
|
||||
let rt = smarm::runtime::init(bench_cfg(threads));
|
||||
rt.run(|| {
|
||||
let (tx_ab, rx_ab) = smarm::channel::<u64>();
|
||||
let (tx_ba, rx_ba) = smarm::channel::<u64>();
|
||||
let echo = smarm::spawn(move || {
|
||||
for _ in 0..PP_STEADY {
|
||||
let v = rx_ab.recv().unwrap();
|
||||
tx_ba.send(v + 1).unwrap();
|
||||
}
|
||||
});
|
||||
for i in 0..PP_STEADY {
|
||||
tx_ab.send(i).unwrap();
|
||||
let v = rx_ba.recv().unwrap();
|
||||
assert_eq!(v, i + 1);
|
||||
}
|
||||
echo.join().unwrap();
|
||||
});
|
||||
let us = start.elapsed().as_micros();
|
||||
wake_diag("ping_pong_steady", threads, &rt, us);
|
||||
(PP_STEADY, us)
|
||||
}
|
||||
|
||||
async fn steady_tokio_body() {
|
||||
let (tx_ab, mut rx_ab) = tokio::sync::mpsc::unbounded_channel::<u64>();
|
||||
let (tx_ba, mut rx_ba) = tokio::sync::mpsc::unbounded_channel::<u64>();
|
||||
let echo = tokio::spawn(async move {
|
||||
for _ in 0..PP_STEADY {
|
||||
let v = rx_ab.recv().await.unwrap();
|
||||
tx_ba.send(v + 1).unwrap();
|
||||
}
|
||||
});
|
||||
for i in 0..PP_STEADY {
|
||||
tx_ab.send(i).unwrap();
|
||||
let v = rx_ba.recv().await.unwrap();
|
||||
assert_eq!(v, i + 1);
|
||||
}
|
||||
echo.await.unwrap();
|
||||
}
|
||||
|
||||
fn bench_steady_tokio_current() -> (u64, u128) {
|
||||
let rt = tokio::runtime::Builder::new_current_thread()
|
||||
.build()
|
||||
.unwrap();
|
||||
let start = Instant::now();
|
||||
rt.block_on(steady_tokio_body());
|
||||
(PP_STEADY, start.elapsed().as_micros())
|
||||
}
|
||||
|
||||
fn bench_steady_tokio_multi() -> (u64, u128) {
|
||||
let rt = tokio::runtime::Builder::new_multi_thread()
|
||||
.worker_threads(available_threads())
|
||||
.build()
|
||||
.unwrap();
|
||||
let start = Instant::now();
|
||||
rt.block_on(steady_tokio_body());
|
||||
(PP_STEADY, start.elapsed().as_micros())
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// main
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Knob helper — reads SMARM_ALLOC_INTERVAL / SMARM_TIMESLICE_CYCLES env vars
|
||||
// so the sweep script can override the preemption knobs without recompiling.
|
||||
@@ -570,14 +394,10 @@ fn bench_steady_tokio_multi() -> (u64, u128) {
|
||||
fn bench_cfg(threads: usize) -> smarm::runtime::Config {
|
||||
let mut cfg = smarm::runtime::Config::exact(threads);
|
||||
if let Ok(v) = std::env::var("SMARM_ALLOC_INTERVAL") {
|
||||
if let Ok(n) = v.parse::<u32>() {
|
||||
cfg = cfg.alloc_interval(n);
|
||||
}
|
||||
if let Ok(n) = v.parse::<u32>() { cfg = cfg.alloc_interval(n); }
|
||||
}
|
||||
if let Ok(v) = std::env::var("SMARM_TIMESLICE_CYCLES") {
|
||||
if let Ok(n) = v.parse::<u64>() {
|
||||
cfg = cfg.timeslice_cycles(n);
|
||||
}
|
||||
if let Ok(n) = v.parse::<u64>() { cfg = cfg.timeslice_cycles(n); }
|
||||
}
|
||||
cfg
|
||||
}
|
||||
@@ -586,43 +406,30 @@ fn main() {
|
||||
let n = available_threads();
|
||||
println!("smarm general benchmarks");
|
||||
println!("available parallelism: {n} threads");
|
||||
let sets = env_sets();
|
||||
println!(
|
||||
"ITERS={ITERS}×{sets} sets = {} samples (+1 warmup, discarded)",
|
||||
ITERS * sets
|
||||
);
|
||||
println!("ITERS={ITERS} (+1 warmup, discarded)");
|
||||
println!(
|
||||
"CHAIN_DEPTH={CHAIN_DEPTH}, YIELD_TASKS={YIELD_TASKS}×{YIELD_ROUNDS}, \
|
||||
PRIME_N={PRIME_N}/{PRIME_WORKERS} workers, PP_ROUNDS={PP_ROUNDS}, \
|
||||
PP_STEADY={PP_STEADY}"
|
||||
PRIME_N={PRIME_N}/{PRIME_WORKERS} workers, PP_ROUNDS={PP_ROUNDS}"
|
||||
);
|
||||
|
||||
// ---- 1. chained_spawn ----
|
||||
print_header(&format!("chained_spawn: depth {CHAIN_DEPTH}"));
|
||||
run_n("smarm 1-thread", ITERS, || bench_chained_smarm(1));
|
||||
run_n(&format!("smarm {n}-thread"), ITERS, || {
|
||||
bench_chained_smarm(n)
|
||||
});
|
||||
run_n(&format!("smarm {n}-thread"), ITERS, || bench_chained_smarm(n));
|
||||
run_n("tokio current_thread", ITERS, bench_chained_tokio_current);
|
||||
run_n("tokio multi-thread", ITERS, bench_chained_tokio_multi);
|
||||
|
||||
// ---- 2. yield_many ----
|
||||
print_header(&format!(
|
||||
"yield_many: {YIELD_TASKS} tasks × {YIELD_ROUNDS} yields"
|
||||
));
|
||||
print_header(&format!("yield_many: {YIELD_TASKS} tasks × {YIELD_ROUNDS} yields"));
|
||||
run_n("smarm 1-thread", ITERS, || bench_yield_smarm(1));
|
||||
run_n(&format!("smarm {n}-thread"), ITERS, || bench_yield_smarm(n));
|
||||
run_n("tokio current_thread", ITERS, bench_yield_tokio_current);
|
||||
run_n("tokio multi-thread", ITERS, bench_yield_tokio_multi);
|
||||
|
||||
// ---- 3. fan_out_compute ----
|
||||
print_header(&format!(
|
||||
"fan_out_compute: primes in [2, {PRIME_N}) across {PRIME_WORKERS}"
|
||||
));
|
||||
print_header(&format!("fan_out_compute: primes in [2, {PRIME_N}) across {PRIME_WORKERS}"));
|
||||
run_n("smarm 1-thread", ITERS, || bench_primes_smarm(1));
|
||||
run_n(&format!("smarm {n}-thread"), ITERS, || {
|
||||
bench_primes_smarm(n)
|
||||
});
|
||||
run_n(&format!("smarm {n}-thread"), ITERS, || bench_primes_smarm(n));
|
||||
run_n("tokio current_thread", ITERS, bench_primes_tokio_current);
|
||||
run_n("tokio multi-thread", ITERS, bench_primes_tokio_multi);
|
||||
|
||||
@@ -632,24 +439,4 @@ fn main() {
|
||||
run_n(&format!("smarm {n}-thread"), ITERS, || bench_pp_smarm(n));
|
||||
run_n("tokio current_thread", ITERS, bench_pp_tokio_current);
|
||||
run_n("tokio multi-thread", ITERS, bench_pp_tokio_multi);
|
||||
|
||||
// ---- 5. spawn_pair_control ----
|
||||
print_header(&format!(
|
||||
"spawn_pair_control: {PP_ROUNDS} rounds, no messages"
|
||||
));
|
||||
run_n("smarm 1-thread", ITERS, || bench_ctl_smarm(1));
|
||||
run_n(&format!("smarm {n}-thread"), ITERS, || bench_ctl_smarm(n));
|
||||
run_n("tokio current_thread", ITERS, bench_ctl_tokio_current);
|
||||
run_n("tokio multi-thread", ITERS, bench_ctl_tokio_multi);
|
||||
|
||||
// ---- 6. ping_pong_steady ----
|
||||
print_header(&format!(
|
||||
"ping_pong_steady: 1 pair × {PP_STEADY} roundtrips"
|
||||
));
|
||||
run_n("smarm 1-thread", ITERS, || bench_steady_smarm(1));
|
||||
run_n(&format!("smarm {n}-thread"), ITERS, || {
|
||||
bench_steady_smarm(n)
|
||||
});
|
||||
run_n("tokio current_thread", ITERS, bench_steady_tokio_current);
|
||||
run_n("tokio multi-thread", ITERS, bench_steady_tokio_multi);
|
||||
}
|
||||
|
||||
+41
-90
@@ -64,22 +64,11 @@ const PRIME_N: u64 = 400_000;
|
||||
const WORKERS: u64 = 64;
|
||||
|
||||
fn is_prime(n: u64) -> bool {
|
||||
if n < 2 {
|
||||
return false;
|
||||
}
|
||||
if n < 4 {
|
||||
return true;
|
||||
}
|
||||
if n % 2 == 0 {
|
||||
return false;
|
||||
}
|
||||
if n < 2 { return false; }
|
||||
if n < 4 { return true; }
|
||||
if n % 2 == 0 { return false; }
|
||||
let mut i = 3u64;
|
||||
while i * i <= n {
|
||||
if n % i == 0 {
|
||||
return false;
|
||||
}
|
||||
i += 2;
|
||||
}
|
||||
while i * i <= n { if n % i == 0 { return false; } i += 2; }
|
||||
true
|
||||
}
|
||||
|
||||
@@ -107,9 +96,7 @@ fn bench_primes_smarm(threads: usize) -> (u64, u128) {
|
||||
tc.fetch_add(count_primes(lo, hi), Ordering::Relaxed);
|
||||
}));
|
||||
}
|
||||
for h in handles {
|
||||
h.join().unwrap();
|
||||
}
|
||||
for h in handles { h.join().unwrap(); }
|
||||
});
|
||||
(total.load(Ordering::Relaxed), start.elapsed().as_micros())
|
||||
}
|
||||
@@ -117,9 +104,7 @@ fn bench_primes_smarm(threads: usize) -> (u64, u128) {
|
||||
fn bench_primes_tokio_current() -> (u64, u128) {
|
||||
let total = Arc::new(AtomicU64::new(0));
|
||||
let t2 = total.clone();
|
||||
let rt = tokio::runtime::Builder::new_current_thread()
|
||||
.build()
|
||||
.unwrap();
|
||||
let rt = tokio::runtime::Builder::new_current_thread().build().unwrap();
|
||||
let start = Instant::now();
|
||||
let local = tokio::task::LocalSet::new();
|
||||
local.block_on(&rt, async move {
|
||||
@@ -131,9 +116,7 @@ fn bench_primes_tokio_current() -> (u64, u128) {
|
||||
tc.fetch_add(count_primes(lo, hi), Ordering::Relaxed);
|
||||
}));
|
||||
}
|
||||
for h in handles {
|
||||
let _ = h.await;
|
||||
}
|
||||
for h in handles { let _ = h.await; }
|
||||
});
|
||||
(total.load(Ordering::Relaxed), start.elapsed().as_micros())
|
||||
}
|
||||
@@ -155,21 +138,17 @@ fn bench_primes_tokio_multi() -> (u64, u128) {
|
||||
tc.fetch_add(count_primes(lo, hi), Ordering::Relaxed);
|
||||
}));
|
||||
}
|
||||
for h in handles {
|
||||
let _ = h.await;
|
||||
}
|
||||
for h in handles { let _ = h.await; }
|
||||
});
|
||||
(total.load(Ordering::Relaxed), start.elapsed().as_micros())
|
||||
}
|
||||
|
||||
fn bench_primes_baseline() -> (u64, u128) {
|
||||
let start = Instant::now();
|
||||
let total: u64 = (0..WORKERS)
|
||||
.map(|w| {
|
||||
let (lo, hi) = primes_slice(w);
|
||||
count_primes(lo, hi)
|
||||
})
|
||||
.sum();
|
||||
let total: u64 = (0..WORKERS).map(|w| {
|
||||
let (lo, hi) = primes_slice(w);
|
||||
count_primes(lo, hi)
|
||||
}).sum();
|
||||
(total, start.elapsed().as_micros())
|
||||
}
|
||||
|
||||
@@ -188,17 +167,15 @@ fn bench_pingpong_smarm(threads: usize) -> (u64, u128) {
|
||||
tx_a.send(0).unwrap();
|
||||
loop {
|
||||
let v = rx_b.recv().unwrap();
|
||||
if v >= PING_ROUNDS {
|
||||
break;
|
||||
}
|
||||
if v >= PING_ROUNDS { break; }
|
||||
tx_a.send(v + 1).unwrap();
|
||||
}
|
||||
});
|
||||
let hb = smarm::spawn(move || loop {
|
||||
let v = rx_a.recv().unwrap();
|
||||
tx_b.send(v + 1).unwrap();
|
||||
if v + 1 >= PING_ROUNDS {
|
||||
break;
|
||||
let hb = smarm::spawn(move || {
|
||||
loop {
|
||||
let v = rx_a.recv().unwrap();
|
||||
tx_b.send(v + 1).unwrap();
|
||||
if v + 1 >= PING_ROUNDS { break; }
|
||||
}
|
||||
});
|
||||
ha.join().unwrap();
|
||||
@@ -221,9 +198,7 @@ fn bench_pingpong_tokio_current() -> (u64, u128) {
|
||||
tx_a.send(0).unwrap();
|
||||
loop {
|
||||
let v = rx_b.recv().await.unwrap();
|
||||
if v >= PING_ROUNDS {
|
||||
break;
|
||||
}
|
||||
if v >= PING_ROUNDS { break; }
|
||||
tx_a.send(v + 1).unwrap();
|
||||
}
|
||||
});
|
||||
@@ -231,9 +206,7 @@ fn bench_pingpong_tokio_current() -> (u64, u128) {
|
||||
loop {
|
||||
let v = rx_a.recv().await.unwrap();
|
||||
tx_b.send(v + 1).unwrap();
|
||||
if v + 1 >= PING_ROUNDS {
|
||||
break;
|
||||
}
|
||||
if v + 1 >= PING_ROUNDS { break; }
|
||||
}
|
||||
});
|
||||
let _ = ha.await;
|
||||
@@ -256,9 +229,7 @@ fn bench_pingpong_tokio_multi() -> (u64, u128) {
|
||||
tx_a.send(0).unwrap();
|
||||
loop {
|
||||
let v = rx_b.recv().await.unwrap();
|
||||
if v >= PING_ROUNDS {
|
||||
break;
|
||||
}
|
||||
if v >= PING_ROUNDS { break; }
|
||||
tx_a.send(v + 1).unwrap();
|
||||
}
|
||||
});
|
||||
@@ -266,9 +237,7 @@ fn bench_pingpong_tokio_multi() -> (u64, u128) {
|
||||
loop {
|
||||
let v = rx_a.recv().await.unwrap();
|
||||
tx_b.send(v + 1).unwrap();
|
||||
if v + 1 >= PING_ROUNDS {
|
||||
break;
|
||||
}
|
||||
if v + 1 >= PING_ROUNDS { break; }
|
||||
}
|
||||
});
|
||||
let _ = ha.await;
|
||||
@@ -295,9 +264,7 @@ fn bench_spawn_smarm(threads: usize) -> (u64, u128) {
|
||||
cc.fetch_add(1, Ordering::Relaxed);
|
||||
}));
|
||||
}
|
||||
for h in handles {
|
||||
h.join().unwrap();
|
||||
}
|
||||
for h in handles { h.join().unwrap(); }
|
||||
});
|
||||
(counter.load(Ordering::Relaxed), start.elapsed().as_micros())
|
||||
}
|
||||
@@ -305,9 +272,7 @@ fn bench_spawn_smarm(threads: usize) -> (u64, u128) {
|
||||
fn bench_spawn_tokio_current() -> (u64, u128) {
|
||||
let counter = Arc::new(AtomicU64::new(0));
|
||||
let c = counter.clone();
|
||||
let rt = tokio::runtime::Builder::new_current_thread()
|
||||
.build()
|
||||
.unwrap();
|
||||
let rt = tokio::runtime::Builder::new_current_thread().build().unwrap();
|
||||
let start = Instant::now();
|
||||
let local = tokio::task::LocalSet::new();
|
||||
local.block_on(&rt, async move {
|
||||
@@ -318,9 +283,7 @@ fn bench_spawn_tokio_current() -> (u64, u128) {
|
||||
cc.fetch_add(1, Ordering::Relaxed);
|
||||
}));
|
||||
}
|
||||
for h in handles {
|
||||
let _ = h.await;
|
||||
}
|
||||
for h in handles { let _ = h.await; }
|
||||
});
|
||||
(counter.load(Ordering::Relaxed), start.elapsed().as_micros())
|
||||
}
|
||||
@@ -341,9 +304,7 @@ fn bench_spawn_tokio_multi() -> (u64, u128) {
|
||||
cc.fetch_add(1, Ordering::Relaxed);
|
||||
}));
|
||||
}
|
||||
for h in handles {
|
||||
let _ = h.await;
|
||||
}
|
||||
for h in handles { let _ = h.await; }
|
||||
});
|
||||
(counter.load(Ordering::Relaxed), start.elapsed().as_micros())
|
||||
}
|
||||
@@ -359,34 +320,24 @@ fn main() {
|
||||
println!("PRIME_N={PRIME_N}, WORKERS={WORKERS}, PING_ROUNDS={PING_ROUNDS}, SPAWN_COUNT={SPAWN_COUNT}");
|
||||
|
||||
// ---- Primes ----
|
||||
print_header(&format!(
|
||||
"Fan-out/fan-in: count primes in [2, {PRIME_N}) across {WORKERS} workers"
|
||||
));
|
||||
run_n("baseline (serial)", ITERS, bench_primes_baseline);
|
||||
run_n("smarm single-thread", ITERS, || bench_primes_smarm(1));
|
||||
run_n(&format!("smarm {n}-thread"), ITERS, || {
|
||||
bench_primes_smarm(n)
|
||||
});
|
||||
run_n("tokio current_thread", ITERS, bench_primes_tokio_current);
|
||||
run_n("tokio multi-thread", ITERS, bench_primes_tokio_multi);
|
||||
print_header(&format!("Fan-out/fan-in: count primes in [2, {PRIME_N}) across {WORKERS} workers"));
|
||||
run_n("baseline (serial)", ITERS, bench_primes_baseline);
|
||||
run_n("smarm single-thread", ITERS, || bench_primes_smarm(1));
|
||||
run_n(&format!("smarm {n}-thread"), ITERS, || bench_primes_smarm(n));
|
||||
run_n("tokio current_thread", ITERS, bench_primes_tokio_current);
|
||||
run_n("tokio multi-thread", ITERS, bench_primes_tokio_multi);
|
||||
|
||||
// ---- Ping-pong ----
|
||||
print_header(&format!(
|
||||
"Ping-pong: {PING_ROUNDS} round-trips between two actors"
|
||||
));
|
||||
run_n("smarm single-thread", ITERS, || bench_pingpong_smarm(1));
|
||||
run_n(&format!("smarm {n}-thread"), ITERS, || {
|
||||
bench_pingpong_smarm(n)
|
||||
});
|
||||
run_n("tokio current_thread", ITERS, bench_pingpong_tokio_current);
|
||||
run_n("tokio multi-thread", ITERS, bench_pingpong_tokio_multi);
|
||||
print_header(&format!("Ping-pong: {PING_ROUNDS} round-trips between two actors"));
|
||||
run_n("smarm single-thread", ITERS, || bench_pingpong_smarm(1));
|
||||
run_n(&format!("smarm {n}-thread"), ITERS, || bench_pingpong_smarm(n));
|
||||
run_n("tokio current_thread", ITERS, bench_pingpong_tokio_current);
|
||||
run_n("tokio multi-thread", ITERS, bench_pingpong_tokio_multi);
|
||||
|
||||
// ---- Spawn throughput ----
|
||||
print_header(&format!(
|
||||
"Spawn throughput: {SPAWN_COUNT} actors spawned and joined"
|
||||
));
|
||||
run_n("smarm single-thread", ITERS, || bench_spawn_smarm(1));
|
||||
print_header(&format!("Spawn throughput: {SPAWN_COUNT} actors spawned and joined"));
|
||||
run_n("smarm single-thread", ITERS, || bench_spawn_smarm(1));
|
||||
run_n(&format!("smarm {n}-thread"), ITERS, || bench_spawn_smarm(n));
|
||||
run_n("tokio current_thread", ITERS, bench_spawn_tokio_current);
|
||||
run_n("tokio multi-thread", ITERS, bench_spawn_tokio_multi);
|
||||
run_n("tokio current_thread", ITERS, bench_spawn_tokio_current);
|
||||
run_n("tokio multi-thread", ITERS, bench_spawn_tokio_multi);
|
||||
}
|
||||
|
||||
+7
-24
@@ -16,20 +16,12 @@ const WORKERS: u64 = 16;
|
||||
const ITERATIONS: u32 = 5;
|
||||
|
||||
fn is_prime(n: u64) -> bool {
|
||||
if n < 2 {
|
||||
return false;
|
||||
}
|
||||
if n < 4 {
|
||||
return true;
|
||||
}
|
||||
if n % 2 == 0 {
|
||||
return false;
|
||||
}
|
||||
if n < 2 { return false; }
|
||||
if n < 4 { return true; }
|
||||
if n % 2 == 0 { return false; }
|
||||
let mut i = 3u64;
|
||||
while i * i <= n {
|
||||
if n % i == 0 {
|
||||
return false;
|
||||
}
|
||||
if n % i == 0 { return false; }
|
||||
i += 2;
|
||||
}
|
||||
true
|
||||
@@ -38,9 +30,7 @@ fn is_prime(n: u64) -> bool {
|
||||
fn count_primes_in(lo: u64, hi: u64) -> u64 {
|
||||
let mut count = 0u64;
|
||||
for n in lo..hi {
|
||||
if is_prime(n) {
|
||||
count += 1;
|
||||
}
|
||||
if is_prime(n) { count += 1; }
|
||||
}
|
||||
count
|
||||
}
|
||||
@@ -48,11 +38,7 @@ fn count_primes_in(lo: u64, hi: u64) -> u64 {
|
||||
fn slice(worker: u64) -> (u64, u64) {
|
||||
let per = N / WORKERS;
|
||||
let lo = worker * per;
|
||||
let hi = if worker + 1 == WORKERS {
|
||||
N
|
||||
} else {
|
||||
(worker + 1) * per
|
||||
};
|
||||
let hi = if worker + 1 == WORKERS { N } else { (worker + 1) * per };
|
||||
(lo, hi)
|
||||
}
|
||||
|
||||
@@ -139,10 +125,7 @@ fn main() {
|
||||
"Counting primes in [2, {}) across {} workers, {} iterations each\n",
|
||||
N, WORKERS, ITERATIONS
|
||||
);
|
||||
println!(
|
||||
"{:>12} | {:>15} | {:>16} | {:>15} | {:>15}",
|
||||
"runtime", "primes found", "median", "min", "max"
|
||||
);
|
||||
println!("{:>12} | {:>15} | {:>16} | {:>15} | {:>15}", "runtime", "primes found", "median", "min", "max");
|
||||
println!("{}", "-".repeat(80));
|
||||
|
||||
run_n("baseline", ITERATIONS, bench_baseline);
|
||||
|
||||
@@ -1,223 +0,0 @@
|
||||
//! Raw run-queue microbench (ROADMAP_v0.5 phase 4).
|
||||
//!
|
||||
//! Benches the three queue STRUCTURES directly — no runtime, no actors — to
|
||||
//! isolate the data structure under contention. All three types compile in
|
||||
//! every build, so this binary covers the whole matrix in one run; it does
|
||||
//! NOT need the rq-* feature rebuild dance (that's `rq_runtime`).
|
||||
//!
|
||||
//! Sweeps thread count × producer:consumer ratio. Queues are sized to the
|
||||
//! item count, so the occupancy contract holds trivially and producers never
|
||||
//! block on capacity.
|
||||
//!
|
||||
//! Knobs (env):
|
||||
//! SMARM_BENCH_THREADS space-separated sweep, default "1 2 4"
|
||||
//! SMARM_BENCH_ITEMS items per measurement, default 200_000
|
||||
//! SMARM_BENCH_RUNS repetitions per config (median reported), default 5
|
||||
//!
|
||||
//! Output: the house table, plus one machine-readable line per config:
|
||||
//! RQCSV,micro,<structure>,<threads>,<p:c>,<items>,<median_us>,<items_per_s>
|
||||
//!
|
||||
//! NOTE: numbers from a 1-core sandbox only validate the harness; real
|
||||
//! contention curves come from the many-core box (scripts/bench_rq.sh).
|
||||
|
||||
use smarm::pid::Pid;
|
||||
use smarm::run_queue::{MpmcRing, MutexQueue, StripedRing};
|
||||
use std::sync::atomic::{AtomicUsize, Ordering};
|
||||
use std::sync::Arc;
|
||||
use std::time::Instant;
|
||||
|
||||
fn env_usize(key: &str, default: usize) -> usize {
|
||||
std::env::var(key)
|
||||
.ok()
|
||||
.and_then(|v| v.parse().ok())
|
||||
.unwrap_or(default)
|
||||
}
|
||||
|
||||
fn env_threads() -> Vec<usize> {
|
||||
std::env::var("SMARM_BENCH_THREADS")
|
||||
.map(|v| {
|
||||
v.split_whitespace()
|
||||
.filter_map(|t| t.parse().ok())
|
||||
.collect()
|
||||
})
|
||||
.unwrap_or_else(|_| vec![1, 2, 4])
|
||||
}
|
||||
|
||||
/// Generic driver: `producers` threads push `items` total, `consumers`
|
||||
/// threads pop until everything is accounted for. Returns elapsed µs.
|
||||
fn drive<Q: Send + Sync + 'static>(
|
||||
q: Arc<Q>,
|
||||
push: fn(&Q, Pid),
|
||||
pop: fn(&Q) -> Option<Pid>,
|
||||
producers: usize,
|
||||
consumers: usize,
|
||||
items: usize,
|
||||
) -> u128 {
|
||||
let remaining = Arc::new(AtomicUsize::new(items));
|
||||
let start = Instant::now();
|
||||
let mut hs = Vec::new();
|
||||
let per = items / producers;
|
||||
for p in 0..producers {
|
||||
let q = q.clone();
|
||||
// Give the last producer the remainder.
|
||||
let n = if p == producers - 1 {
|
||||
items - per * (producers - 1)
|
||||
} else {
|
||||
per
|
||||
};
|
||||
hs.push(std::thread::spawn(move || {
|
||||
let pid = Pid::new(p as u32, 0);
|
||||
for _ in 0..n {
|
||||
push(&q, pid);
|
||||
}
|
||||
}));
|
||||
}
|
||||
for _ in 0..consumers {
|
||||
let q = q.clone();
|
||||
let remaining = remaining.clone();
|
||||
hs.push(std::thread::spawn(move || loop {
|
||||
// Claim-then-pop so consumers exit promptly when the budget hits
|
||||
// zero; the claim is backed out on a miss.
|
||||
let r = remaining.load(Ordering::Relaxed);
|
||||
if r == 0 {
|
||||
return;
|
||||
}
|
||||
if pop(&q).is_some() {
|
||||
remaining.fetch_sub(1, Ordering::Relaxed);
|
||||
} else {
|
||||
std::hint::spin_loop();
|
||||
}
|
||||
}));
|
||||
}
|
||||
for h in hs {
|
||||
h.join().unwrap();
|
||||
}
|
||||
start.elapsed().as_micros()
|
||||
}
|
||||
|
||||
/// Single-thread alternating push/pop (the T = 1 case).
|
||||
fn drive_single<Q>(q: &Q, push: fn(&Q, Pid), pop: fn(&Q) -> Option<Pid>, items: usize) -> u128 {
|
||||
let pid = Pid::new(0, 0);
|
||||
let start = Instant::now();
|
||||
for _ in 0..items {
|
||||
push(q, pid);
|
||||
assert!(pop(q).is_some());
|
||||
}
|
||||
start.elapsed().as_micros()
|
||||
}
|
||||
|
||||
struct Case {
|
||||
structure: &'static str,
|
||||
threads: usize,
|
||||
producers: usize,
|
||||
consumers: usize,
|
||||
}
|
||||
|
||||
fn ratios_for(threads: usize) -> Vec<(usize, usize)> {
|
||||
if threads < 2 {
|
||||
return vec![(1, 1)]; // label only; T=1 runs the alternating driver
|
||||
}
|
||||
let mut v = vec![(threads / 2, threads - threads / 2)]; // balanced
|
||||
if threads >= 4 {
|
||||
v.push((3 * threads / 4, threads - 3 * threads / 4)); // producer-heavy
|
||||
v.push((threads / 4, threads - threads / 4)); // consumer-heavy
|
||||
}
|
||||
v
|
||||
}
|
||||
|
||||
fn main() {
|
||||
let threads_sweep = env_threads();
|
||||
let items = env_usize("SMARM_BENCH_ITEMS", 200_000);
|
||||
let runs = env_usize("SMARM_BENCH_RUNS", 5);
|
||||
|
||||
println!("\n{}", "=".repeat(86));
|
||||
println!(" run-queue raw structures — items={items}, runs={runs} (median)");
|
||||
println!("{}", "=".repeat(86));
|
||||
println!(
|
||||
"{:>10} | {:>7} | {:>7} | {:>10} | {:>14}",
|
||||
"structure", "threads", "p:c", "median µs", "items/s"
|
||||
);
|
||||
println!("{}", "-".repeat(86));
|
||||
|
||||
let mut cases = Vec::new();
|
||||
for &t in &threads_sweep {
|
||||
for (p, c) in ratios_for(t) {
|
||||
for s in ["mutex", "mpmc", "striped"] {
|
||||
cases.push(Case {
|
||||
structure: s,
|
||||
threads: t,
|
||||
producers: p,
|
||||
consumers: c,
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for case in cases {
|
||||
let mut times: Vec<u128> = (0..runs)
|
||||
.map(|_| {
|
||||
// Fresh queue per run; capacity = items so pushes never stall.
|
||||
match case.structure {
|
||||
"mutex" => {
|
||||
let q = Arc::new(MutexQueue::new(case.threads, items));
|
||||
if case.threads < 2 {
|
||||
drive_single(&*q, MutexQueue::push, MutexQueue::pop, items)
|
||||
} else {
|
||||
drive(
|
||||
q,
|
||||
MutexQueue::push,
|
||||
MutexQueue::pop,
|
||||
case.producers,
|
||||
case.consumers,
|
||||
items,
|
||||
)
|
||||
}
|
||||
}
|
||||
"mpmc" => {
|
||||
let q = Arc::new(MpmcRing::with_capacity(items));
|
||||
if case.threads < 2 {
|
||||
drive_single(&*q, MpmcRing::push, MpmcRing::pop, items)
|
||||
} else {
|
||||
drive(
|
||||
q,
|
||||
MpmcRing::push,
|
||||
MpmcRing::pop,
|
||||
case.producers,
|
||||
case.consumers,
|
||||
items,
|
||||
)
|
||||
}
|
||||
}
|
||||
"striped" => {
|
||||
let q = Arc::new(StripedRing::new(case.threads.max(1), items));
|
||||
if case.threads < 2 {
|
||||
drive_single(&*q, StripedRing::push, StripedRing::pop, items)
|
||||
} else {
|
||||
drive(
|
||||
q,
|
||||
StripedRing::push,
|
||||
StripedRing::pop,
|
||||
case.producers,
|
||||
case.consumers,
|
||||
items,
|
||||
)
|
||||
}
|
||||
}
|
||||
_ => unreachable!(),
|
||||
}
|
||||
})
|
||||
.collect();
|
||||
times.sort_unstable();
|
||||
let median = times[times.len() / 2];
|
||||
let per_s = (items as f64 / (median as f64 / 1e6)) as u64;
|
||||
let ratio = format!("{}:{}", case.producers, case.consumers);
|
||||
println!(
|
||||
"{:>10} | {:>7} | {:>7} | {:>10} | {:>14}",
|
||||
case.structure, case.threads, ratio, median, per_s
|
||||
);
|
||||
println!(
|
||||
"RQCSV,micro,{},{},{},{},{},{}",
|
||||
case.structure, case.threads, ratio, items, median, per_s
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -1,281 +0,0 @@
|
||||
//! Runtime-level run-queue benches (ROADMAP_v0.5 phase 4; slot dimension
|
||||
//! added for the v0.9 slot shootout, RFC 005).
|
||||
//!
|
||||
//! These exercise the WHOLE scheduler with the compile-time-selected queue,
|
||||
//! so comparing variants means rebuilding per rq-* feature — that's what
|
||||
//! scripts/bench_rq.sh does. The RFC 005 wake slot is a *runtime* Config
|
||||
//! knob, so one binary benches both arms; the slot on/off sweep happens
|
||||
//! inside this binary. Workloads:
|
||||
//!
|
||||
//! yield-storm — N actors yield K times each. Pure queue churn:
|
||||
//! every yield is a push + pop with nothing in between.
|
||||
//! Slot role: REGRESSION GUARD — yields never touch the
|
||||
//! slot, any slot-on delta is pop-path overhead.
|
||||
//! ping-pong-pairs — P channel pairs, M roundtrips each. Park/unpark
|
||||
//! latency through the queue.
|
||||
//! Slot role: TARGET METRIC — every send-wake is an
|
||||
//! actor-context unpark, the slot's home pattern.
|
||||
//! spawn-storm — S spawn+join of trivial actors. Slab + queue + free
|
||||
//! list under churn.
|
||||
//! Slot role: NEUTRALITY CHECK — spawns bypass the slot
|
||||
//! by policy; join wakes fire from finalize (scheduler
|
||||
//! context), also shared.
|
||||
//!
|
||||
//! Knobs (env):
|
||||
//! SMARM_BENCH_THREADS scheduler-count sweep, default "1 2 4"
|
||||
//! SMARM_BENCH_SLOT wake-slot sweep, default "0 1" (off then on)
|
||||
//! SMARM_BENCH_RUNS repetitions per config (median), default 5
|
||||
//! SMARM_BENCH_YIELD_ACTORS / _YIELDS default 200 / 500
|
||||
//! SMARM_BENCH_PAIRS / _ROUNDTRIPS default 32 / 1000
|
||||
//! SMARM_BENCH_SPAWNS default 5000
|
||||
//!
|
||||
//! Output: house table + one line per config:
|
||||
//! RQCSV,runtime,<variant>,<slot>,<bench>,<threads>,<work>,<median_us>,<ops_per_s>
|
||||
//! plus, for slot-on configs, the RFC 005 observability counters:
|
||||
//! RQSLOT,<variant>,<bench>,<threads>,<slot_hits>,<slot_displacements>
|
||||
//! (hits/displacements are taken from the same run as the median time).
|
||||
//!
|
||||
//! NOTE: a 1-core sandbox validates the harness, not the scaling story;
|
||||
//! real curves come from the many-core box.
|
||||
|
||||
use smarm::runtime::{init, Config};
|
||||
use std::sync::atomic::{AtomicU64, Ordering};
|
||||
use std::sync::Arc;
|
||||
use std::time::Instant;
|
||||
|
||||
fn variant() -> &'static str {
|
||||
if cfg!(feature = "rq-mpmc") {
|
||||
"rq-mpmc"
|
||||
} else if cfg!(feature = "rq-striped") {
|
||||
"rq-striped"
|
||||
} else {
|
||||
"rq-mutex"
|
||||
}
|
||||
}
|
||||
|
||||
fn env_usize(key: &str, default: usize) -> usize {
|
||||
std::env::var(key)
|
||||
.ok()
|
||||
.and_then(|v| v.parse().ok())
|
||||
.unwrap_or(default)
|
||||
}
|
||||
|
||||
fn env_threads() -> Vec<usize> {
|
||||
std::env::var("SMARM_BENCH_THREADS")
|
||||
.map(|v| {
|
||||
v.split_whitespace()
|
||||
.filter_map(|t| t.parse().ok())
|
||||
.collect()
|
||||
})
|
||||
.unwrap_or_else(|_| vec![1, 2, 4])
|
||||
}
|
||||
|
||||
fn env_slots() -> Vec<bool> {
|
||||
std::env::var("SMARM_BENCH_SLOT")
|
||||
.map(|v| {
|
||||
v.split_whitespace()
|
||||
.filter_map(|t| match t {
|
||||
"0" | "off" | "false" => Some(false),
|
||||
"1" | "on" | "true" => Some(true),
|
||||
_ => None,
|
||||
})
|
||||
.collect()
|
||||
})
|
||||
.unwrap_or_else(|_| vec![false, true])
|
||||
}
|
||||
|
||||
/// One measured run: (total_ops, elapsed_µs, slot_hits, slot_displacements).
|
||||
struct Sample {
|
||||
ops: u64,
|
||||
us: u128,
|
||||
hits: u64,
|
||||
displacements: u64,
|
||||
diag: String,
|
||||
}
|
||||
|
||||
fn yield_storm(threads: usize, slot: bool, actors: usize, yields: usize) -> Sample {
|
||||
let rt = init(Config::exact(threads).wake_slot(slot));
|
||||
let start = Instant::now();
|
||||
rt.run(move || {
|
||||
let handles: Vec<_> = (0..actors)
|
||||
.map(|_| {
|
||||
smarm::spawn(move || {
|
||||
for _ in 0..yields {
|
||||
smarm::yield_now();
|
||||
}
|
||||
})
|
||||
})
|
||||
.collect();
|
||||
for h in handles {
|
||||
let _ = h.join();
|
||||
}
|
||||
});
|
||||
let us = start.elapsed().as_micros();
|
||||
let stats = rt.stats();
|
||||
Sample {
|
||||
ops: (actors * yields) as u64,
|
||||
us,
|
||||
hits: stats.slot_hits(),
|
||||
displacements: stats.slot_displacements(),
|
||||
diag: stats.wake_diag(),
|
||||
}
|
||||
}
|
||||
|
||||
fn ping_pong_pairs(threads: usize, slot: bool, pairs: usize, roundtrips: usize) -> Sample {
|
||||
let rt = init(Config::exact(threads).wake_slot(slot));
|
||||
let total = Arc::new(AtomicU64::new(0));
|
||||
let t2 = total.clone();
|
||||
let start = Instant::now();
|
||||
rt.run(move || {
|
||||
let handles: Vec<_> = (0..pairs)
|
||||
.map(|_| {
|
||||
let total = t2.clone();
|
||||
smarm::spawn(move || {
|
||||
let (tx_ab, rx_ab) = smarm::channel::channel::<u64>();
|
||||
let (tx_ba, rx_ba) = smarm::channel::channel::<u64>();
|
||||
let n = roundtrips as u64;
|
||||
let echo = smarm::spawn(move || {
|
||||
for _ in 0..n {
|
||||
let v = rx_ab.recv().expect("echo recv");
|
||||
tx_ba.send(v + 1).expect("echo send");
|
||||
}
|
||||
});
|
||||
for i in 0..n {
|
||||
tx_ab.send(i).expect("ping send");
|
||||
let v = rx_ba.recv().expect("ping recv");
|
||||
assert_eq!(v, i + 1);
|
||||
}
|
||||
let _ = echo.join();
|
||||
total.fetch_add(n, Ordering::Relaxed);
|
||||
})
|
||||
})
|
||||
.collect();
|
||||
for h in handles {
|
||||
let _ = h.join();
|
||||
}
|
||||
});
|
||||
let us = start.elapsed().as_micros();
|
||||
let stats = rt.stats();
|
||||
Sample {
|
||||
ops: total.load(Ordering::Relaxed),
|
||||
us,
|
||||
hits: stats.slot_hits(),
|
||||
displacements: stats.slot_displacements(),
|
||||
diag: stats.wake_diag(),
|
||||
}
|
||||
}
|
||||
|
||||
fn spawn_storm(threads: usize, slot: bool, spawns: usize) -> Sample {
|
||||
let rt = init(Config::exact(threads).wake_slot(slot));
|
||||
let start = Instant::now();
|
||||
rt.run(move || {
|
||||
// Batches bound simultaneous liveness well below the slab cap.
|
||||
const BATCH: usize = 1024;
|
||||
let mut left = spawns;
|
||||
while left > 0 {
|
||||
let n = left.min(BATCH);
|
||||
let handles: Vec<_> = (0..n).map(|_| smarm::spawn(|| {})).collect();
|
||||
for h in handles {
|
||||
let _ = h.join();
|
||||
}
|
||||
left -= n;
|
||||
}
|
||||
});
|
||||
let us = start.elapsed().as_micros();
|
||||
let stats = rt.stats();
|
||||
Sample {
|
||||
ops: spawns as u64,
|
||||
us,
|
||||
hits: stats.slot_hits(),
|
||||
displacements: stats.slot_displacements(),
|
||||
diag: stats.wake_diag(),
|
||||
}
|
||||
}
|
||||
|
||||
fn main() {
|
||||
let threads_sweep = env_threads();
|
||||
let slot_sweep = env_slots();
|
||||
let runs = env_usize("SMARM_BENCH_RUNS", 5);
|
||||
let ya = env_usize("SMARM_BENCH_YIELD_ACTORS", 200);
|
||||
let yy = env_usize("SMARM_BENCH_YIELDS", 500);
|
||||
let pp = env_usize("SMARM_BENCH_PAIRS", 32);
|
||||
let pr = env_usize("SMARM_BENCH_ROUNDTRIPS", 1000);
|
||||
let ss = env_usize("SMARM_BENCH_SPAWNS", 5000);
|
||||
|
||||
println!("\n{}", "=".repeat(106));
|
||||
println!(
|
||||
" runtime benches — variant={}, runs={runs} (median)",
|
||||
variant()
|
||||
);
|
||||
println!("{}", "=".repeat(106));
|
||||
println!(
|
||||
"{:>16} | {:>7} | {:>4} | {:>16} | {:>10} | {:>14} | {:>10} | {:>9}",
|
||||
"bench", "threads", "slot", "work", "median µs", "ops/s", "slot hits", "displaced"
|
||||
);
|
||||
println!("{}", "-".repeat(106));
|
||||
|
||||
type Bench = (&'static str, String, Box<dyn Fn(usize, bool) -> Sample>);
|
||||
let benches: Vec<Bench> = vec![
|
||||
(
|
||||
"yield-storm",
|
||||
format!("{ya}x{yy}"),
|
||||
Box::new(move |t, s| yield_storm(t, s, ya, yy)),
|
||||
),
|
||||
(
|
||||
"ping-pong-pairs",
|
||||
format!("{pp}x{pr}"),
|
||||
Box::new(move |t, s| ping_pong_pairs(t, s, pp, pr)),
|
||||
),
|
||||
(
|
||||
"spawn-storm",
|
||||
format!("{ss}"),
|
||||
Box::new(move |t, s| spawn_storm(t, s, ss)),
|
||||
),
|
||||
];
|
||||
|
||||
for (name, work, f) in &benches {
|
||||
for &t in &threads_sweep {
|
||||
for &slot in &slot_sweep {
|
||||
let mut samples: Vec<Sample> = (0..runs).map(|_| f(t, slot)).collect();
|
||||
// Median by elapsed time; report the counters from that
|
||||
// same run so hits/time stay paired.
|
||||
samples.sort_unstable_by_key(|s| s.us);
|
||||
let mid = &samples[samples.len() / 2];
|
||||
let per_s = (mid.ops as f64 / (mid.us as f64 / 1e6)) as u64;
|
||||
let slot_str = if slot { "on" } else { "off" };
|
||||
println!(
|
||||
"{:>16} | {:>7} | {:>4} | {:>16} | {:>10} | {:>14} | {:>10} | {:>9}",
|
||||
name, t, slot_str, work, mid.us, per_s, mid.hits, mid.displacements
|
||||
);
|
||||
println!(
|
||||
"RQCSV,runtime,{},{},{},{},{},{},{}",
|
||||
variant(),
|
||||
slot_str,
|
||||
name,
|
||||
t,
|
||||
work,
|
||||
mid.us,
|
||||
per_s
|
||||
);
|
||||
println!(
|
||||
"RQDIAG,{},{},{},{},{}",
|
||||
variant(),
|
||||
slot_str,
|
||||
name,
|
||||
t,
|
||||
mid.diag
|
||||
);
|
||||
if slot {
|
||||
println!(
|
||||
"RQSLOT,{},{},{},{},{}",
|
||||
variant(),
|
||||
name,
|
||||
t,
|
||||
mid.hits,
|
||||
mid.displacements
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
+25
-73
@@ -37,16 +37,7 @@ use std::time::Instant;
|
||||
const ITERS: u32 = 15;
|
||||
|
||||
fn available_threads() -> usize {
|
||||
std::thread::available_parallelism()
|
||||
.map(|n| n.get())
|
||||
.unwrap_or(1)
|
||||
}
|
||||
|
||||
fn env_sets() -> u32 {
|
||||
std::env::var("SMARM_BENCH_SETS")
|
||||
.ok()
|
||||
.and_then(|v| v.parse().ok())
|
||||
.unwrap_or(5)
|
||||
std::thread::available_parallelism().map(|n| n.get()).unwrap_or(1)
|
||||
}
|
||||
|
||||
fn print_header(title: &str) {
|
||||
@@ -61,17 +52,13 @@ fn print_header(title: &str) {
|
||||
}
|
||||
|
||||
fn run_n<F: FnMut() -> (u64, u128)>(name: &str, n: u32, mut f: F) {
|
||||
let sets = env_sets();
|
||||
let mut times = Vec::with_capacity((n * sets) as usize);
|
||||
let mut times = Vec::new();
|
||||
let mut last = 0u64;
|
||||
// One warmup before all sets, discarded.
|
||||
let _ = f();
|
||||
for _ in 0..sets {
|
||||
for _ in 0..n {
|
||||
let (v, t) = f();
|
||||
times.push(t);
|
||||
last = v;
|
||||
}
|
||||
let _ = f(); // warmup
|
||||
for _ in 0..n {
|
||||
let (v, t) = f();
|
||||
times.push(t);
|
||||
last = v;
|
||||
}
|
||||
times.sort_unstable();
|
||||
let median = times[times.len() / 2];
|
||||
@@ -118,9 +105,7 @@ fn bench_recurse_smarm(threads: usize) -> (u64, u128) {
|
||||
fn bench_recurse_tokio_current() -> (u64, u128) {
|
||||
let counter = Arc::new(AtomicU64::new(0));
|
||||
let c2 = counter.clone();
|
||||
let rt = tokio::runtime::Builder::new_current_thread()
|
||||
.build()
|
||||
.unwrap();
|
||||
let rt = tokio::runtime::Builder::new_current_thread().build().unwrap();
|
||||
let start = Instant::now();
|
||||
let local = tokio::task::LocalSet::new();
|
||||
local.block_on(&rt, async move {
|
||||
@@ -203,9 +188,7 @@ fn bench_hot_smarm() -> (u64, u128) {
|
||||
}
|
||||
|
||||
fn bench_hot_tokio_current() -> (u64, u128) {
|
||||
let rt = tokio::runtime::Builder::new_current_thread()
|
||||
.build()
|
||||
.unwrap();
|
||||
let rt = tokio::runtime::Builder::new_current_thread().build().unwrap();
|
||||
let start = Instant::now();
|
||||
let local = tokio::task::LocalSet::new();
|
||||
local.block_on(&rt, async move {
|
||||
@@ -255,9 +238,7 @@ fn bench_unc_smarm() -> (u64, u128) {
|
||||
}
|
||||
|
||||
fn bench_unc_tokio_current() -> (u64, u128) {
|
||||
let rt = tokio::runtime::Builder::new_current_thread()
|
||||
.build()
|
||||
.unwrap();
|
||||
let rt = tokio::runtime::Builder::new_current_thread().build().unwrap();
|
||||
let start = Instant::now();
|
||||
let local = tokio::task::LocalSet::new();
|
||||
local.block_on(&rt, async move {
|
||||
@@ -305,12 +286,8 @@ fn bench_panic_smarm(threads: usize) -> (u64, u128) {
|
||||
}
|
||||
for h in handles {
|
||||
match h.join() {
|
||||
Ok(()) => {
|
||||
ok2.fetch_add(1, Ordering::Relaxed);
|
||||
}
|
||||
Err(_) => {
|
||||
err2.fetch_add(1, Ordering::Relaxed);
|
||||
}
|
||||
Ok(()) => { ok2.fetch_add(1, Ordering::Relaxed); }
|
||||
Err(_) => { err2.fetch_add(1, Ordering::Relaxed); }
|
||||
}
|
||||
}
|
||||
});
|
||||
@@ -324,9 +301,7 @@ fn bench_panic_tokio_current() -> (u64, u128) {
|
||||
let err = Arc::new(AtomicU64::new(0));
|
||||
let ok2 = ok.clone();
|
||||
let err2 = err.clone();
|
||||
let rt = tokio::runtime::Builder::new_current_thread()
|
||||
.build()
|
||||
.unwrap();
|
||||
let rt = tokio::runtime::Builder::new_current_thread().build().unwrap();
|
||||
let prev_hook = std::panic::take_hook();
|
||||
std::panic::set_hook(Box::new(|_| {}));
|
||||
let start = Instant::now();
|
||||
@@ -342,12 +317,8 @@ fn bench_panic_tokio_current() -> (u64, u128) {
|
||||
}
|
||||
for h in handles {
|
||||
match h.await {
|
||||
Ok(()) => {
|
||||
ok2.fetch_add(1, Ordering::Relaxed);
|
||||
}
|
||||
Err(_) => {
|
||||
err2.fetch_add(1, Ordering::Relaxed);
|
||||
}
|
||||
Ok(()) => { ok2.fetch_add(1, Ordering::Relaxed); }
|
||||
Err(_) => { err2.fetch_add(1, Ordering::Relaxed); }
|
||||
}
|
||||
}
|
||||
});
|
||||
@@ -379,12 +350,8 @@ fn bench_panic_tokio_multi() -> (u64, u128) {
|
||||
}
|
||||
for h in handles {
|
||||
match h.await {
|
||||
Ok(()) => {
|
||||
ok2.fetch_add(1, Ordering::Relaxed);
|
||||
}
|
||||
Err(_) => {
|
||||
err2.fetch_add(1, Ordering::Relaxed);
|
||||
}
|
||||
Ok(()) => { ok2.fetch_add(1, Ordering::Relaxed); }
|
||||
Err(_) => { err2.fetch_add(1, Ordering::Relaxed); }
|
||||
}
|
||||
}
|
||||
});
|
||||
@@ -397,6 +364,7 @@ fn bench_panic_tokio_multi() -> (u64, u128) {
|
||||
// main
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Knob helper — reads SMARM_ALLOC_INTERVAL / SMARM_TIMESLICE_CYCLES env vars
|
||||
// so the sweep script can override the preemption knobs without recompiling.
|
||||
@@ -405,14 +373,10 @@ fn bench_panic_tokio_multi() -> (u64, u128) {
|
||||
fn bench_cfg(threads: usize) -> smarm::runtime::Config {
|
||||
let mut cfg = smarm::runtime::Config::exact(threads);
|
||||
if let Ok(v) = std::env::var("SMARM_ALLOC_INTERVAL") {
|
||||
if let Ok(n) = v.parse::<u32>() {
|
||||
cfg = cfg.alloc_interval(n);
|
||||
}
|
||||
if let Ok(n) = v.parse::<u32>() { cfg = cfg.alloc_interval(n); }
|
||||
}
|
||||
if let Ok(v) = std::env::var("SMARM_TIMESLICE_CYCLES") {
|
||||
if let Ok(n) = v.parse::<u64>() {
|
||||
cfg = cfg.timeslice_cycles(n);
|
||||
}
|
||||
if let Ok(n) = v.parse::<u64>() { cfg = cfg.timeslice_cycles(n); }
|
||||
}
|
||||
cfg
|
||||
}
|
||||
@@ -421,11 +385,7 @@ fn main() {
|
||||
let n = available_threads();
|
||||
println!("smarm smarm-favored benchmarks");
|
||||
println!("available parallelism: {n} threads");
|
||||
let sets = env_sets();
|
||||
println!(
|
||||
"ITERS={ITERS}×{sets} sets = {} samples (+1 warmup, discarded)",
|
||||
ITERS * sets
|
||||
);
|
||||
println!("ITERS={ITERS} (+1 warmup, discarded)");
|
||||
println!(
|
||||
"RECURSE_DEPTH={RECURSE_DEPTH}, HOT_YIELDS={HOT_YIELDS}×2, \
|
||||
UNCONT_MSGS={UNCONT_MSGS}, PANIC_TASKS={PANIC_TASKS}"
|
||||
@@ -434,30 +394,22 @@ fn main() {
|
||||
// ---- 9. deep_recursion ----
|
||||
print_header(&format!("deep_recursion: depth {RECURSE_DEPTH}"));
|
||||
run_n("smarm 1-thread", ITERS, || bench_recurse_smarm(1));
|
||||
run_n(&format!("smarm {n}-thread"), ITERS, || {
|
||||
bench_recurse_smarm(n)
|
||||
});
|
||||
run_n(&format!("smarm {n}-thread"), ITERS, || bench_recurse_smarm(n));
|
||||
run_n("tokio current_thread", ITERS, bench_recurse_tokio_current);
|
||||
run_n("tokio multi-thread", ITERS, bench_recurse_tokio_multi);
|
||||
|
||||
// ---- 10. yield_in_hot_loop ----
|
||||
print_header(&format!(
|
||||
"yield_in_hot_loop: 2 actors × {HOT_YIELDS} yields (single thread)"
|
||||
));
|
||||
print_header(&format!("yield_in_hot_loop: 2 actors × {HOT_YIELDS} yields (single thread)"));
|
||||
run_n("smarm 1-thread", ITERS, bench_hot_smarm);
|
||||
run_n("tokio current_thread", ITERS, bench_hot_tokio_current);
|
||||
|
||||
// ---- 11. uncontended_channel ----
|
||||
print_header(&format!(
|
||||
"uncontended_channel: 1→1, {UNCONT_MSGS} msgs (single thread)"
|
||||
));
|
||||
print_header(&format!("uncontended_channel: 1→1, {UNCONT_MSGS} msgs (single thread)"));
|
||||
run_n("smarm 1-thread", ITERS, bench_unc_smarm);
|
||||
run_n("tokio current_thread", ITERS, bench_unc_tokio_current);
|
||||
|
||||
// ---- 12. catch_unwind_panics ----
|
||||
print_header(&format!(
|
||||
"catch_unwind_panics: {PANIC_TASKS} tasks, 50% panic"
|
||||
));
|
||||
print_header(&format!("catch_unwind_panics: {PANIC_TASKS} tasks, 50% panic"));
|
||||
run_n("smarm 1-thread", ITERS, || bench_panic_smarm(1));
|
||||
run_n(&format!("smarm {n}-thread"), ITERS, || bench_panic_smarm(n));
|
||||
run_n("tokio current_thread", ITERS, bench_panic_tokio_current);
|
||||
|
||||
+5
-137
@@ -50,11 +50,6 @@ SWEEP_GRID = [
|
||||
(128, 1_200_000),
|
||||
]
|
||||
|
||||
# Number of independent cargo bench processes per measurement point.
|
||||
# Each process is a fully isolated run (fresh warmup, cold caches, new PID),
|
||||
# so the final median is a median of independent samples — robust to OS noise.
|
||||
BENCH_SETS = 5
|
||||
|
||||
# Regression threshold: warn if median is more than this % worse than baseline.
|
||||
REGRESSION_THRESHOLD_PCT = 10
|
||||
|
||||
@@ -108,12 +103,9 @@ def parse_output(text: str) -> dict[str, dict[str, dict]]:
|
||||
# Running
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def run_benches_once(env_extra: dict[str, str] | None = None) -> dict[str, dict[str, dict]]:
|
||||
"""Run all BENCHES once and return merged parsed results."""
|
||||
def run_benches(env_extra: dict[str, str] | None = None) -> dict[str, dict[str, dict]]:
|
||||
"""Run all BENCHES and return merged parsed results."""
|
||||
env = os.environ.copy()
|
||||
# Each process does exactly one set of ITERS samples — no within-process
|
||||
# accumulation; the caller handles multi-set aggregation.
|
||||
env["SMARM_BENCH_SETS"] = "1"
|
||||
if env_extra:
|
||||
env.update(env_extra)
|
||||
|
||||
@@ -137,45 +129,6 @@ def run_benches_once(env_extra: dict[str, str] | None = None) -> dict[str, dict[
|
||||
return all_results
|
||||
|
||||
|
||||
def run_benches(env_extra: dict[str, str] | None = None, sets: int = BENCH_SETS) -> dict[str, dict[str, dict]]:
|
||||
"""Run BENCH_SETS independent processes and return median-of-medians per label.
|
||||
|
||||
Each set is a separate cargo bench invocation with its own warmup and OS
|
||||
context, so samples are statistically independent. The final median and
|
||||
min/max are computed over the per-set medians.
|
||||
"""
|
||||
# Accumulate per-set medians: {bench: {label: [median_set1, median_set2, ...]}}
|
||||
accumulated: dict[str, dict[str, list[int]]] = {}
|
||||
last_result: dict[str, dict[str, int]] = {}
|
||||
|
||||
for i in range(sets):
|
||||
print(f" set {i + 1}/{sets}…", flush=True)
|
||||
set_results = run_benches_once(env_extra)
|
||||
for bench, labels in set_results.items():
|
||||
accumulated.setdefault(bench, {})
|
||||
last_result.setdefault(bench, {})
|
||||
for label, data in labels.items():
|
||||
accumulated[bench].setdefault(label, [])
|
||||
accumulated[bench][label].append(data["median"])
|
||||
last_result[bench][label] = data["result"]
|
||||
|
||||
# Collapse to final stats.
|
||||
final: dict[str, dict[str, dict]] = {}
|
||||
for bench, labels in accumulated.items():
|
||||
final[bench] = {}
|
||||
for label, medians in labels.items():
|
||||
medians.sort()
|
||||
mid = medians[len(medians) // 2]
|
||||
final[bench][label] = {
|
||||
"result": last_result[bench][label],
|
||||
"median": mid,
|
||||
"min": medians[0],
|
||||
"max": medians[-1],
|
||||
}
|
||||
|
||||
return final
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Baseline JSON
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -243,90 +196,6 @@ def check_regressions(current: dict, baseline: dict) -> bool:
|
||||
# Pretty print
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _threads(label: str) -> int | None:
|
||||
"""Worker-thread count implied by a runtime label.
|
||||
|
||||
tokio's `current_thread` is a single-threaded executor (1); an explicit
|
||||
`multi N-thread` is N; a bare `multi-thread` has no count (None) — tokio's
|
||||
default work-stealing pool, paired against smarm's widest config.
|
||||
"""
|
||||
if "current_thread" in label:
|
||||
return 1
|
||||
m = re.search(r"(\d+)-thread", label)
|
||||
return int(m.group(1)) if m else None
|
||||
|
||||
|
||||
def vs_tokio(results: dict) -> list[tuple]:
|
||||
"""Per bench, like-for-like smarm-vs-tokio rows matched by thread count.
|
||||
|
||||
Exact thread-count matches are paired directly (smarm 1-thread vs tokio
|
||||
current_thread, smarm 4-thread vs tokio multi 4-thread, …). tokio's bare
|
||||
`multi-thread` (no explicit count) is paired against the widest unmatched
|
||||
smarm multi-thread config. Lower median µs = faster; ratio = tokio_med /
|
||||
smarm_med, so ratio > 1 means smarm is that many times faster.
|
||||
|
||||
Returns rows of (bench, smarm_label, smarm_med, tokio_label, tokio_med,
|
||||
ratio, winner). Benches without a comparable pair are skipped.
|
||||
"""
|
||||
rows: list[tuple] = []
|
||||
for bench, runtimes in sorted(results.items()):
|
||||
smarm: dict[int, tuple[str, int]] = {}
|
||||
tokio: dict[int, tuple[str, int]] = {}
|
||||
tokio_default: tuple[str, int] | None = None # bare 'multi-thread'
|
||||
for label, data in runtimes.items():
|
||||
n = _threads(label)
|
||||
if label.startswith("smarm"):
|
||||
if n is not None:
|
||||
smarm[n] = (label, data["median"])
|
||||
elif label.startswith("tokio"):
|
||||
if n is None:
|
||||
tokio_default = (label, data["median"])
|
||||
else:
|
||||
tokio[n] = (label, data["median"])
|
||||
|
||||
def row(s: tuple[str, int], t: tuple[str, int]):
|
||||
s_label, s_med = s
|
||||
t_label, t_med = t
|
||||
if s_med == 0:
|
||||
return None
|
||||
ratio = t_med / s_med
|
||||
return (bench, s_label, s_med, t_label, t_med, ratio,
|
||||
"smarm" if ratio >= 1.0 else "tokio")
|
||||
|
||||
matched: set[int] = set()
|
||||
for n in sorted(set(smarm) & set(tokio)):
|
||||
r = row(smarm[n], tokio[n])
|
||||
if r:
|
||||
rows.append(r)
|
||||
matched.add(n)
|
||||
# tokio's default multi pool vs the widest smarm config not already paired.
|
||||
if tokio_default is not None:
|
||||
rem = [n for n in smarm if n not in matched and n > 1]
|
||||
if rem:
|
||||
r = row(smarm[max(rem)], tokio_default)
|
||||
if r:
|
||||
rows.append(r)
|
||||
return rows
|
||||
|
||||
|
||||
def print_vs_tokio(results: dict) -> None:
|
||||
"""Human summary + greppable VSTOKIO lines (best smarm vs best tokio)."""
|
||||
rows = vs_tokio(results)
|
||||
if not rows:
|
||||
return
|
||||
print("\n vs tokio (like-for-like by thread count; ratio>1 = smarm faster, lower µs better)")
|
||||
print(f" {'-'*78}")
|
||||
for bench, s_label, s_med, t_label, t_med, ratio, winner in rows:
|
||||
print(
|
||||
f" {bench:<22} {s_label} {s_med}µs vs {t_label} {t_med}µs"
|
||||
f" → {ratio:.2f}x ({winner})"
|
||||
)
|
||||
# Machine-readable, one line per bench:
|
||||
# VSTOKIO,<bench>,<smarm_label>,<smarm_us>,<tokio_label>,<tokio_us>,<ratio>,<winner>
|
||||
for bench, s_label, s_med, t_label, t_med, ratio, winner in rows:
|
||||
print(f"VSTOKIO,{bench},{s_label},{s_med},{t_label},{t_med},{ratio:.3f},{winner}")
|
||||
|
||||
|
||||
def print_results(results: dict, label: str = "") -> None:
|
||||
if label:
|
||||
print(f"\n{'='*70}")
|
||||
@@ -341,7 +210,6 @@ def print_results(results: dict, label: str = "") -> None:
|
||||
f" {rt_label:>28} | {data['result']:>10} | "
|
||||
f"{data['median']:>10} | {data['min']:>8} | {data['max']:>8}"
|
||||
)
|
||||
print_vs_tokio(results)
|
||||
|
||||
|
||||
def print_sweep_table(sweep_results: list[tuple[int, int, dict]]) -> None:
|
||||
@@ -384,7 +252,7 @@ def cmd_run(args) -> None:
|
||||
["cargo", "build", "--release", "--benches"],
|
||||
cwd=REPO, check=True, capture_output=True,
|
||||
)
|
||||
print(f"Running benches ({BENCH_SETS} independent sets)…")
|
||||
print("Running benches…")
|
||||
results = run_benches()
|
||||
print_results(results, "Results (default knobs)")
|
||||
if args.save_baseline:
|
||||
@@ -398,7 +266,7 @@ def cmd_regress(args) -> None:
|
||||
["cargo", "build", "--release", "--benches"],
|
||||
cwd=REPO, check=True, capture_output=True,
|
||||
)
|
||||
print(f"Running benches ({BENCH_SETS} independent sets)…")
|
||||
print("Running benches…")
|
||||
current = run_benches()
|
||||
print_results(current, "Current results")
|
||||
print(f"\nRegression check (threshold: >{REGRESSION_THRESHOLD_PCT}% slower than baseline)")
|
||||
@@ -420,7 +288,7 @@ def cmd_sweep(args) -> None:
|
||||
|
||||
for interval, cycles in SWEEP_GRID:
|
||||
tag = f"alloc_interval={interval}, timeslice_cycles={cycles}"
|
||||
print(f" Running: {tag} ({BENCH_SETS} sets)…", flush=True)
|
||||
print(f" Running: {tag} …", flush=True)
|
||||
env_extra = {
|
||||
"SMARM_ALLOC_INTERVAL": str(interval),
|
||||
"SMARM_TIMESLICE_CYCLES": str(cycles),
|
||||
|
||||
@@ -1,281 +0,0 @@
|
||||
//! Per-switch (context-switch) cost microbench — the profiling-spike harness
|
||||
//! for the ROADMAP "Per-switch cost (context shims, epoch protocol)" item.
|
||||
//!
|
||||
//! The spin work (RFC 004) is closed; the next perf target is the per-switch
|
||||
//! cost itself. Shootout evidence: per-wake latency is ~0.16–0.18µs at N=1 but
|
||||
//! ~0.8–1.2µs at N=8+, and the residual is attributed to the context-switch
|
||||
//! shims (`src/context.rs`) and the epoch protocol — NOT the queue. This binary
|
||||
//! isolates that round-trip so the cycles can be attributed under `perf` and an
|
||||
//! rdtsc bracket, feeding the RFC.
|
||||
//!
|
||||
//! WHAT THE ROUND-TRIP IS
|
||||
//!
|
||||
//! `yield_now()` from inside an actor does exactly one park/unpark round-trip
|
||||
//! with nothing else attached:
|
||||
//!
|
||||
//! actor: switch_to_scheduler ──► scheduler re-queues the actor (slot-word
|
||||
//! (context.rs shim) epoch/state transition, run_queue push),
|
||||
//! pops it straight back, switch_to_actor
|
||||
//! actor resumes ◄──────────────────────────────────────────────────────
|
||||
//!
|
||||
//! No IO thread traffic, no channel, no timer, no cross-thread wake. On a
|
||||
//! single-scheduler runtime the re-queue+repop never leaves this core, so the
|
||||
//! sample is the *pure* shim + epoch + queue-op cost with zero coherency
|
||||
//! traffic. That is the `local` baseline; a `remote` mode (wake straddling two
|
||||
//! schedulers, to expose the N=1→N=8 coherency/TLS-mode jump) is a deliberate
|
||||
//! follow-up and is NOT in this file yet — local first, per the spike plan.
|
||||
//!
|
||||
//! TWO LENSES ON THE SAME LOOP
|
||||
//!
|
||||
//! wall — `Instant` bracket per round-trip. Source of truth for µs, directly
|
||||
//! comparable to the shootout's per-wake latency numbers.
|
||||
//! cycles — `rdtsc` bracket per round-trip. Source of truth for the cycle
|
||||
//! budget the RFC will reason in (the spin budget is in cycles too).
|
||||
//!
|
||||
//! Reporting both lets us *derive* the effective TSC frequency (cycles/ns) from
|
||||
//! the same samples instead of hardcoding a nominal 3.7GHz — the spin_sweep
|
||||
//! lesson was that nominal-vs-actual TSC drift is exactly what produces
|
||||
//! red-herring numbers. If the derived freq matches the box's known base clock,
|
||||
//! the two lenses corroborate; if not, that mismatch is itself a finding.
|
||||
//!
|
||||
//! Knobs (env):
|
||||
//! SMARM_SWITCH_ROUNDS round-trips timed per run default 200000
|
||||
//! SMARM_SWITCH_WARMUP untimed warmup round-trips default 10000
|
||||
//! SMARM_SWITCH_RUNS runs (pooled latency, median) default 5
|
||||
//!
|
||||
//! Output: house table + one greppable line per run-set:
|
||||
//! SWITCHCSV,<variant>,<mode>,<rounds>,<runs>,<n>,<p50_ns>,<p90_ns>,<p99_ns>,
|
||||
//! <min_ns>,<max_ns>,<mean_ns>,<mean_cyc>,<derived_ghz>
|
||||
//!
|
||||
//! NOTE: a single yielding actor is the cooperative-scheduling tightest loop —
|
||||
//! it never parks on a futex (the work is always immediately re-queued), so
|
||||
//! this measures the switch+epoch+queue path, NOT the futex park. That is
|
||||
//! intentional: the futex park is the spin work's territory (RFC 004), already
|
||||
//! characterised. The unattributed constant the shootout flagged lives in the
|
||||
//! switch itself, which is what this loop hammers.
|
||||
|
||||
use smarm::runtime::{init, Config};
|
||||
use smarm::{run, spawn, yield_now};
|
||||
use std::sync::{Arc, Mutex};
|
||||
|
||||
// --------------------------------------------------------------------------
|
||||
// env helpers (house style, matching spin_sweep.rs / rq_runtime.rs)
|
||||
// --------------------------------------------------------------------------
|
||||
|
||||
fn variant() -> &'static str {
|
||||
if cfg!(feature = "rq-mpmc") {
|
||||
"rq-mpmc"
|
||||
} else if cfg!(feature = "rq-striped") {
|
||||
"rq-striped"
|
||||
} else {
|
||||
"rq-mutex"
|
||||
}
|
||||
}
|
||||
|
||||
fn env_usize(key: &str, default: usize) -> usize {
|
||||
std::env::var(key)
|
||||
.ok()
|
||||
.and_then(|v| v.parse().ok())
|
||||
.unwrap_or(default)
|
||||
}
|
||||
|
||||
// --------------------------------------------------------------------------
|
||||
// rdtsc — serialised so the bracket actually fences the round-trip.
|
||||
//
|
||||
// Plain `rdtsc` can be reordered around the work by an out-of-order core, which
|
||||
// would smear the bracket. `rdtscp` retires prior instructions before reading
|
||||
// the counter, and the trailing `lfence` blocks later instructions from
|
||||
// climbing above the second read. Pair = (rdtscp; lfence) … work … (rdtscp;
|
||||
// lfence): a standard cycle-accurate bracket. We read TSC_AUX too but ignore
|
||||
// it; the point is the ordering guarantee, not the core id.
|
||||
// --------------------------------------------------------------------------
|
||||
|
||||
#[inline(always)]
|
||||
fn rdtsc_serialised() -> u64 {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
unsafe {
|
||||
let mut aux = 0u32;
|
||||
let t = core::arch::x86_64::__rdtscp(&mut aux);
|
||||
core::arch::x86_64::_mm_lfence();
|
||||
t
|
||||
}
|
||||
#[cfg(not(target_arch = "x86_64"))]
|
||||
{
|
||||
// Non-x86 fallback: nanosecond clock standing in for cycles. The derived
|
||||
// "GHz" column then reads ~1.0 and is meaningless, but the wall lens and
|
||||
// the harness still work. The spike target box is x86-64.
|
||||
use std::time::Instant;
|
||||
thread_local! { static T0: Instant = Instant::now(); }
|
||||
T0.with(|t0| t0.elapsed().as_nanos() as u64)
|
||||
}
|
||||
}
|
||||
|
||||
// --------------------------------------------------------------------------
|
||||
// percentile / median helpers (verbatim house idiom from spin_sweep.rs)
|
||||
// --------------------------------------------------------------------------
|
||||
|
||||
/// Nearest-rank percentile over an already-sorted slice. `p` in [0, 100].
|
||||
fn pct(sorted: &[u64], p: f64) -> u64 {
|
||||
if sorted.is_empty() {
|
||||
return 0;
|
||||
}
|
||||
let idx = ((p / 100.0) * (sorted.len() - 1) as f64).round() as usize;
|
||||
sorted[idx.min(sorted.len() - 1)]
|
||||
}
|
||||
|
||||
// --------------------------------------------------------------------------
|
||||
// one run: a single actor yields ROUNDS times; we bracket each yield from
|
||||
// inside the actor (the only vantage point — the actor is suspended during the
|
||||
// scheduler half, so an external timer can't see a single round-trip).
|
||||
//
|
||||
// Per iteration we capture BOTH a wall-ns delta and a TSC-cycle delta around
|
||||
// the same `yield_now()`. The loop overhead (two clock reads + a Vec push +
|
||||
// the branch) rides along in every sample equally; we subtract an empty-loop
|
||||
// self-calibration below so the reported number is the round-trip, not the
|
||||
// instrumentation.
|
||||
// --------------------------------------------------------------------------
|
||||
|
||||
struct RunSample {
|
||||
lat_ns: Vec<u64>,
|
||||
cyc: Vec<u64>,
|
||||
}
|
||||
|
||||
fn one_run(threads: usize, rounds: usize, warmup: usize) -> RunSample {
|
||||
let out: Arc<Mutex<Option<RunSample>>> = Arc::new(Mutex::new(None));
|
||||
let out2 = out.clone();
|
||||
|
||||
let cfg = Config::exact(threads);
|
||||
init(cfg);
|
||||
|
||||
run(move || {
|
||||
let h = spawn(move || {
|
||||
// Warmup: let the actor's stack/queue slot go hot, JIT-free but
|
||||
// cache-warm, before any sample is kept.
|
||||
for _ in 0..warmup {
|
||||
yield_now();
|
||||
}
|
||||
|
||||
let mut lat_ns = Vec::with_capacity(rounds);
|
||||
let mut cyc = Vec::with_capacity(rounds);
|
||||
|
||||
for _ in 0..rounds {
|
||||
let w0 = std::time::Instant::now();
|
||||
let c0 = rdtsc_serialised();
|
||||
yield_now();
|
||||
let c1 = rdtsc_serialised();
|
||||
let w1 = w0.elapsed();
|
||||
cyc.push(c1.saturating_sub(c0));
|
||||
lat_ns.push(w1.as_nanos() as u64);
|
||||
}
|
||||
|
||||
*out2.lock().unwrap() = Some(RunSample { lat_ns, cyc });
|
||||
});
|
||||
let _ = h.join();
|
||||
});
|
||||
|
||||
let sample = out.lock().unwrap().take().expect("actor stored a sample");
|
||||
sample
|
||||
}
|
||||
|
||||
/// Empty-loop self-calibration: the same bracket with the `yield_now()` removed,
|
||||
/// run inline (no runtime). Gives the floor cost of two serialised clock reads +
|
||||
/// the push, in both lenses, to subtract from the round-trip samples.
|
||||
fn calibrate(rounds: usize) -> (u64, u64) {
|
||||
let mut lat_ns = Vec::with_capacity(rounds);
|
||||
let mut cyc = Vec::with_capacity(rounds);
|
||||
let mut sink = 0u64;
|
||||
for _ in 0..rounds {
|
||||
let w0 = std::time::Instant::now();
|
||||
let c0 = rdtsc_serialised();
|
||||
// no yield — measure the bracket itself
|
||||
let c1 = rdtsc_serialised();
|
||||
let w1 = w0.elapsed();
|
||||
sink ^= c1;
|
||||
cyc.push(c1.saturating_sub(c0));
|
||||
lat_ns.push(w1.as_nanos() as u64);
|
||||
}
|
||||
std::hint::black_box(sink);
|
||||
cyc.sort_unstable();
|
||||
lat_ns.sort_unstable();
|
||||
// Use the medians as the floor — robust to the occasional interrupt.
|
||||
(pct(&lat_ns, 50.0), pct(&cyc, 50.0))
|
||||
}
|
||||
|
||||
fn main() {
|
||||
let rounds = env_usize("SMARM_SWITCH_ROUNDS", 200_000);
|
||||
let warmup = env_usize("SMARM_SWITCH_WARMUP", 10_000);
|
||||
let runs = env_usize("SMARM_SWITCH_RUNS", 5);
|
||||
let mode = "local";
|
||||
|
||||
// Calibrate the instrumentation floor once, with a healthy sample.
|
||||
let (floor_ns, floor_cyc) = calibrate(rounds.min(50_000).max(10_000));
|
||||
|
||||
let mut pooled_ns: Vec<u64> = Vec::new();
|
||||
let mut pooled_cyc: Vec<u64> = Vec::new();
|
||||
|
||||
for _ in 0..runs {
|
||||
let s = one_run(1, rounds, warmup);
|
||||
// Subtract the instrumentation floor; saturating so a sub-floor outlier
|
||||
// (clock granularity) clamps to 0 rather than wrapping.
|
||||
pooled_ns.extend(s.lat_ns.iter().map(|&v| v.saturating_sub(floor_ns)));
|
||||
pooled_cyc.extend(s.cyc.iter().map(|&v| v.saturating_sub(floor_cyc)));
|
||||
}
|
||||
|
||||
pooled_ns.sort_unstable();
|
||||
pooled_cyc.sort_unstable();
|
||||
|
||||
let n = pooled_ns.len();
|
||||
let mean_ns = pooled_ns.iter().map(|&v| v as f64).sum::<f64>() / n.max(1) as f64;
|
||||
let mean_cyc = pooled_cyc.iter().map(|&v| v as f64).sum::<f64>() / n.max(1) as f64;
|
||||
// Derived effective frequency: cycles per ns = GHz. Cross-checks the two
|
||||
// lenses against the box's known base clock.
|
||||
let derived_ghz = if mean_ns > 0.0 {
|
||||
mean_cyc / mean_ns
|
||||
} else {
|
||||
0.0
|
||||
};
|
||||
|
||||
let p50 = pct(&pooled_ns, 50.0);
|
||||
let p90 = pct(&pooled_ns, 90.0);
|
||||
let p99 = pct(&pooled_ns, 99.0);
|
||||
let lo = *pooled_ns.first().unwrap_or(&0);
|
||||
let hi = *pooled_ns.last().unwrap_or(&0);
|
||||
|
||||
// House table.
|
||||
println!();
|
||||
println!("per-switch cost — {} mode, variant={}", mode, variant());
|
||||
println!(
|
||||
" rounds={} warmup={} runs={} (instrumentation floor: {} ns / {} cyc, subtracted)",
|
||||
rounds, warmup, runs, floor_ns, floor_cyc
|
||||
);
|
||||
println!(
|
||||
" {:<10} {:<10} {:<10} {:<10} {:<10}",
|
||||
"p50 ns", "p90 ns", "p99 ns", "min ns", "max ns"
|
||||
);
|
||||
println!(
|
||||
" {:<10} {:<10} {:<10} {:<10} {:<10}",
|
||||
p50, p90, p99, lo, hi
|
||||
);
|
||||
println!(
|
||||
" mean {:.1} ns | mean {:.0} cyc | derived {:.3} GHz",
|
||||
mean_ns, mean_cyc, derived_ghz
|
||||
);
|
||||
|
||||
// Greppable line — same spirit as SPINCSV.
|
||||
println!(
|
||||
"SWITCHCSV,{},{},{},{},{},{},{},{},{},{},{:.1},{:.0},{:.3}",
|
||||
variant(),
|
||||
mode,
|
||||
rounds,
|
||||
runs,
|
||||
n,
|
||||
p50,
|
||||
p90,
|
||||
p99,
|
||||
lo,
|
||||
hi,
|
||||
mean_ns,
|
||||
mean_cyc,
|
||||
derived_ghz
|
||||
);
|
||||
}
|
||||
+41
-125
@@ -36,16 +36,7 @@ use std::time::{Duration, Instant};
|
||||
const ITERS: u32 = 15;
|
||||
|
||||
fn available_threads() -> usize {
|
||||
std::thread::available_parallelism()
|
||||
.map(|n| n.get())
|
||||
.unwrap_or(1)
|
||||
}
|
||||
|
||||
fn env_sets() -> u32 {
|
||||
std::env::var("SMARM_BENCH_SETS")
|
||||
.ok()
|
||||
.and_then(|v| v.parse().ok())
|
||||
.unwrap_or(5)
|
||||
std::thread::available_parallelism().map(|n| n.get()).unwrap_or(1)
|
||||
}
|
||||
|
||||
fn print_header(title: &str) {
|
||||
@@ -60,17 +51,13 @@ fn print_header(title: &str) {
|
||||
}
|
||||
|
||||
fn run_n<F: FnMut() -> (u64, u128)>(name: &str, n: u32, mut f: F) {
|
||||
let sets = env_sets();
|
||||
let mut times = Vec::with_capacity((n * sets) as usize);
|
||||
let mut times = Vec::new();
|
||||
let mut last = 0u64;
|
||||
// One warmup before all sets, discarded.
|
||||
let _ = f();
|
||||
for _ in 0..sets {
|
||||
for _ in 0..n {
|
||||
let (v, t) = f();
|
||||
times.push(t);
|
||||
last = v;
|
||||
}
|
||||
let _ = f(); // warmup
|
||||
for _ in 0..n {
|
||||
let (v, t) = f();
|
||||
times.push(t);
|
||||
last = v;
|
||||
}
|
||||
times.sort_unstable();
|
||||
let median = times[times.len() / 2];
|
||||
@@ -86,8 +73,8 @@ fn run_n<F: FnMut() -> (u64, u128)>(name: &str, n: u32, mut f: F) {
|
||||
// 5. spawn_storm_busy — workers loaded, then storm of zero-work spawns
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const STORM_BACKGROUND: u64 = 8; // number of background "busy" actors
|
||||
const STORM_SPAWN: u64 = 10_000; // zero-work spawns to time
|
||||
const STORM_BACKGROUND: u64 = 8; // number of background "busy" actors
|
||||
const STORM_SPAWN: u64 = 10_000; // zero-work spawns to time
|
||||
|
||||
fn bench_storm_smarm(threads: usize) -> (u64, u128) {
|
||||
let counter = Arc::new(AtomicU64::new(0));
|
||||
@@ -116,15 +103,11 @@ fn bench_storm_smarm(threads: usize) -> (u64, u128) {
|
||||
cc.fetch_add(1, Ordering::Relaxed);
|
||||
}));
|
||||
}
|
||||
for h in handles {
|
||||
h.join().unwrap();
|
||||
}
|
||||
for h in handles { h.join().unwrap(); }
|
||||
|
||||
// Tear down background.
|
||||
s2.store(true, Ordering::Relaxed);
|
||||
for h in bg_handles {
|
||||
h.join().unwrap();
|
||||
}
|
||||
for h in bg_handles { h.join().unwrap(); }
|
||||
});
|
||||
(counter.load(Ordering::Relaxed), start.elapsed().as_micros())
|
||||
}
|
||||
@@ -135,9 +118,7 @@ fn bench_storm_tokio_current() -> (u64, u128) {
|
||||
let c2 = counter.clone();
|
||||
let s2 = stop.clone();
|
||||
|
||||
let rt = tokio::runtime::Builder::new_current_thread()
|
||||
.build()
|
||||
.unwrap();
|
||||
let rt = tokio::runtime::Builder::new_current_thread().build().unwrap();
|
||||
let start = Instant::now();
|
||||
let local = tokio::task::LocalSet::new();
|
||||
local.block_on(&rt, async move {
|
||||
@@ -157,13 +138,9 @@ fn bench_storm_tokio_current() -> (u64, u128) {
|
||||
cc.fetch_add(1, Ordering::Relaxed);
|
||||
}));
|
||||
}
|
||||
for h in handles {
|
||||
let _ = h.await;
|
||||
}
|
||||
for h in handles { let _ = h.await; }
|
||||
s2.store(true, Ordering::Relaxed);
|
||||
for h in bg_handles {
|
||||
let _ = h.await;
|
||||
}
|
||||
for h in bg_handles { let _ = h.await; }
|
||||
});
|
||||
(counter.load(Ordering::Relaxed), start.elapsed().as_micros())
|
||||
}
|
||||
@@ -196,13 +173,9 @@ fn bench_storm_tokio_multi() -> (u64, u128) {
|
||||
cc.fetch_add(1, Ordering::Relaxed);
|
||||
}));
|
||||
}
|
||||
for h in handles {
|
||||
let _ = h.await;
|
||||
}
|
||||
for h in handles { let _ = h.await; }
|
||||
s2.store(true, Ordering::Relaxed);
|
||||
for h in bg_handles {
|
||||
let _ = h.await;
|
||||
}
|
||||
for h in bg_handles { let _ = h.await; }
|
||||
});
|
||||
(counter.load(Ordering::Relaxed), start.elapsed().as_micros())
|
||||
}
|
||||
@@ -235,21 +208,14 @@ fn bench_mpsc_smarm(threads: usize) -> (u64, u128) {
|
||||
}
|
||||
let _ = count; // discard; run() closure must return ()
|
||||
});
|
||||
for h in prod_handles {
|
||||
h.join().unwrap();
|
||||
}
|
||||
for h in prod_handles { h.join().unwrap(); }
|
||||
let _ = consumer.join().unwrap();
|
||||
});
|
||||
(
|
||||
MPSC_PRODUCERS * MPSC_PER_PRODUCER,
|
||||
start.elapsed().as_micros(),
|
||||
)
|
||||
(MPSC_PRODUCERS * MPSC_PER_PRODUCER, start.elapsed().as_micros())
|
||||
}
|
||||
|
||||
fn bench_mpsc_tokio_current() -> (u64, u128) {
|
||||
let rt = tokio::runtime::Builder::new_current_thread()
|
||||
.build()
|
||||
.unwrap();
|
||||
let rt = tokio::runtime::Builder::new_current_thread().build().unwrap();
|
||||
let start = Instant::now();
|
||||
let local = tokio::task::LocalSet::new();
|
||||
local.block_on(&rt, async move {
|
||||
@@ -271,15 +237,10 @@ fn bench_mpsc_tokio_current() -> (u64, u128) {
|
||||
}
|
||||
count
|
||||
});
|
||||
for h in prod_handles {
|
||||
let _ = h.await;
|
||||
}
|
||||
for h in prod_handles { let _ = h.await; }
|
||||
let _ = consumer.await;
|
||||
});
|
||||
(
|
||||
MPSC_PRODUCERS * MPSC_PER_PRODUCER,
|
||||
start.elapsed().as_micros(),
|
||||
)
|
||||
(MPSC_PRODUCERS * MPSC_PER_PRODUCER, start.elapsed().as_micros())
|
||||
}
|
||||
|
||||
fn bench_mpsc_tokio_multi() -> (u64, u128) {
|
||||
@@ -307,15 +268,10 @@ fn bench_mpsc_tokio_multi() -> (u64, u128) {
|
||||
}
|
||||
count
|
||||
});
|
||||
for h in prod_handles {
|
||||
let _ = h.await;
|
||||
}
|
||||
for h in prod_handles { let _ = h.await; }
|
||||
let _ = consumer.await;
|
||||
});
|
||||
(
|
||||
MPSC_PRODUCERS * MPSC_PER_PRODUCER,
|
||||
start.elapsed().as_micros(),
|
||||
)
|
||||
(MPSC_PRODUCERS * MPSC_PER_PRODUCER, start.elapsed().as_micros())
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -341,9 +297,7 @@ fn bench_timers_smarm(threads: usize) -> (u64, u128) {
|
||||
smarm::sleep(Duration::from_millis(ms));
|
||||
}));
|
||||
}
|
||||
for h in handles {
|
||||
h.join().unwrap();
|
||||
}
|
||||
for h in handles { h.join().unwrap(); }
|
||||
});
|
||||
(TIMER_ACTORS, start.elapsed().as_micros())
|
||||
}
|
||||
@@ -363,9 +317,7 @@ fn bench_timers_tokio_current() -> (u64, u128) {
|
||||
tokio::time::sleep(Duration::from_millis(ms)).await;
|
||||
}));
|
||||
}
|
||||
for h in handles {
|
||||
let _ = h.await;
|
||||
}
|
||||
for h in handles { let _ = h.await; }
|
||||
});
|
||||
(TIMER_ACTORS, start.elapsed().as_micros())
|
||||
}
|
||||
@@ -385,9 +337,7 @@ fn bench_timers_tokio_multi() -> (u64, u128) {
|
||||
tokio::time::sleep(Duration::from_millis(ms)).await;
|
||||
}));
|
||||
}
|
||||
for h in handles {
|
||||
let _ = h.await;
|
||||
}
|
||||
for h in handles { let _ = h.await; }
|
||||
});
|
||||
(TIMER_ACTORS, start.elapsed().as_micros())
|
||||
}
|
||||
@@ -400,22 +350,11 @@ const SCALING_N: u64 = 400_000;
|
||||
const SCALING_WORKERS: u64 = 64;
|
||||
|
||||
fn is_prime(n: u64) -> bool {
|
||||
if n < 2 {
|
||||
return false;
|
||||
}
|
||||
if n < 4 {
|
||||
return true;
|
||||
}
|
||||
if n % 2 == 0 {
|
||||
return false;
|
||||
}
|
||||
if n < 2 { return false; }
|
||||
if n < 4 { return true; }
|
||||
if n % 2 == 0 { return false; }
|
||||
let mut i = 3u64;
|
||||
while i * i <= n {
|
||||
if n % i == 0 {
|
||||
return false;
|
||||
}
|
||||
i += 2;
|
||||
}
|
||||
while i * i <= n { if n % i == 0 { return false; } i += 2; }
|
||||
true
|
||||
}
|
||||
|
||||
@@ -426,11 +365,7 @@ fn count_primes(lo: u64, hi: u64) -> u64 {
|
||||
fn scaling_slice(w: u64) -> (u64, u64) {
|
||||
let per = SCALING_N / SCALING_WORKERS;
|
||||
let lo = w * per;
|
||||
let hi = if w + 1 == SCALING_WORKERS {
|
||||
SCALING_N
|
||||
} else {
|
||||
lo + per
|
||||
};
|
||||
let hi = if w + 1 == SCALING_WORKERS { SCALING_N } else { lo + per };
|
||||
(lo, hi)
|
||||
}
|
||||
|
||||
@@ -447,9 +382,7 @@ fn bench_scaling_smarm(threads: usize) -> (u64, u128) {
|
||||
tc.fetch_add(count_primes(lo, hi), Ordering::Relaxed);
|
||||
}));
|
||||
}
|
||||
for h in handles {
|
||||
h.join().unwrap();
|
||||
}
|
||||
for h in handles { h.join().unwrap(); }
|
||||
});
|
||||
(total.load(Ordering::Relaxed), start.elapsed().as_micros())
|
||||
}
|
||||
@@ -471,9 +404,7 @@ fn bench_scaling_tokio_multi(threads: usize) -> (u64, u128) {
|
||||
tc.fetch_add(count_primes(lo, hi), Ordering::Relaxed);
|
||||
}));
|
||||
}
|
||||
for h in handles {
|
||||
let _ = h.await;
|
||||
}
|
||||
for h in handles { let _ = h.await; }
|
||||
});
|
||||
(total.load(Ordering::Relaxed), start.elapsed().as_micros())
|
||||
}
|
||||
@@ -482,6 +413,7 @@ fn bench_scaling_tokio_multi(threads: usize) -> (u64, u128) {
|
||||
// main
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Knob helper — reads SMARM_ALLOC_INTERVAL / SMARM_TIMESLICE_CYCLES env vars
|
||||
// so the sweep script can override the preemption knobs without recompiling.
|
||||
@@ -490,14 +422,10 @@ fn bench_scaling_tokio_multi(threads: usize) -> (u64, u128) {
|
||||
fn bench_cfg(threads: usize) -> smarm::runtime::Config {
|
||||
let mut cfg = smarm::runtime::Config::exact(threads);
|
||||
if let Ok(v) = std::env::var("SMARM_ALLOC_INTERVAL") {
|
||||
if let Ok(n) = v.parse::<u32>() {
|
||||
cfg = cfg.alloc_interval(n);
|
||||
}
|
||||
if let Ok(n) = v.parse::<u32>() { cfg = cfg.alloc_interval(n); }
|
||||
}
|
||||
if let Ok(v) = std::env::var("SMARM_TIMESLICE_CYCLES") {
|
||||
if let Ok(n) = v.parse::<u64>() {
|
||||
cfg = cfg.timeslice_cycles(n);
|
||||
}
|
||||
if let Ok(n) = v.parse::<u64>() { cfg = cfg.timeslice_cycles(n); }
|
||||
}
|
||||
cfg
|
||||
}
|
||||
@@ -506,11 +434,7 @@ fn main() {
|
||||
let n = available_threads();
|
||||
println!("smarm tokio-favored benchmarks");
|
||||
println!("available parallelism: {n} threads");
|
||||
let sets = env_sets();
|
||||
println!(
|
||||
"ITERS={ITERS}×{sets} sets = {} samples (+1 warmup, discarded)",
|
||||
ITERS * sets
|
||||
);
|
||||
println!("ITERS={ITERS} (+1 warmup, discarded)");
|
||||
println!(
|
||||
"STORM_BACKGROUND={STORM_BACKGROUND}, STORM_SPAWN={STORM_SPAWN}, \
|
||||
MPSC={MPSC_PRODUCERS}×{MPSC_PER_PRODUCER}, \
|
||||
@@ -541,9 +465,7 @@ fn main() {
|
||||
"many_timers: {TIMER_ACTORS} actors sleeping {TIMER_MIN_MS}–{TIMER_MAX_MS} ms"
|
||||
));
|
||||
run_n("smarm 1-thread", ITERS, || bench_timers_smarm(1));
|
||||
run_n(&format!("smarm {n}-thread"), ITERS, || {
|
||||
bench_timers_smarm(n)
|
||||
});
|
||||
run_n(&format!("smarm {n}-thread"), ITERS, || bench_timers_smarm(n));
|
||||
run_n("tokio current_thread", ITERS, bench_timers_tokio_current);
|
||||
run_n("tokio multi-thread", ITERS, bench_timers_tokio_multi);
|
||||
|
||||
@@ -553,19 +475,13 @@ fn main() {
|
||||
));
|
||||
let sweep: Vec<usize> = {
|
||||
let mut v = vec![1usize, 2, 4];
|
||||
if n > 4 && !v.contains(&n) {
|
||||
v.push(n);
|
||||
}
|
||||
if n > 4 && !v.contains(&n) { v.push(n); }
|
||||
v.into_iter().filter(|t| *t <= n).collect()
|
||||
};
|
||||
for t in &sweep {
|
||||
run_n(&format!("smarm {t}-thread"), ITERS, || {
|
||||
bench_scaling_smarm(*t)
|
||||
});
|
||||
run_n(&format!("smarm {t}-thread"), ITERS, || bench_scaling_smarm(*t));
|
||||
}
|
||||
for t in &sweep {
|
||||
run_n(&format!("tokio multi {t}-thread"), ITERS, || {
|
||||
bench_scaling_tokio_multi(*t)
|
||||
});
|
||||
run_n(&format!("tokio multi {t}-thread"), ITERS, || bench_scaling_tokio_multi(*t));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,36 +0,0 @@
|
||||
fn main() {
|
||||
// RFC 019 §7 test canary (agreed Q3): compiled without stack-clash
|
||||
// protection so its 96 KiB local is a genuine one-displacement guard
|
||||
// jumper; distro-hardened compilers would otherwise probe it page-wise
|
||||
// and defeat the test's purpose.
|
||||
cc::Build::new()
|
||||
.file("canary/canary.c")
|
||||
.flag_if_supported("-fno-stack-clash-protection")
|
||||
.compile("smarm_canary");
|
||||
println!("cargo:rerun-if-changed=canary/canary.c");
|
||||
|
||||
// RFC 010 c6d — build_hash inputs. The compile-time facts a peer must
|
||||
// share for a mesh link: the exact toolchain and the declared (enabled)
|
||||
// feature set. Emitted as a plain string; the hashing (FNV-1a folded
|
||||
// with PROTO_VERSION) happens in src/cluster.rs where the protocol
|
||||
// version actually lives — parsing it out of a source file here would
|
||||
// be a second, fragile copy. Always emitted, even for non-cluster
|
||||
// builds: one env var costs the default build nothing.
|
||||
let rustc = std::env::var("RUSTC").unwrap_or_else(|_| "rustc".to_string());
|
||||
let version = std::process::Command::new(&rustc)
|
||||
.arg("-V")
|
||||
.output()
|
||||
.ok()
|
||||
.map(|o| String::from_utf8_lossy(&o.stdout).trim().to_string())
|
||||
.filter(|v| !v.is_empty())
|
||||
.unwrap_or_else(|| "rustc-unknown".to_string());
|
||||
let mut feats: Vec<String> = std::env::vars()
|
||||
.filter_map(|(k, _)| k.strip_prefix("CARGO_FEATURE_").map(str::to_string))
|
||||
.collect();
|
||||
feats.sort();
|
||||
println!(
|
||||
"cargo:rustc-env=SMARM_BUILD_HASH_INPUTS={version};features={}",
|
||||
feats.join(",")
|
||||
);
|
||||
println!("cargo:rerun-if-env-changed=RUSTC");
|
||||
}
|
||||
@@ -1,14 +0,0 @@
|
||||
/* RFC 019 §7 FFI canary: an honest unprobed C frame with a 96 KiB local,
|
||||
* touched from its LOW end first — the exact "one sub rsp steps over a small
|
||||
* guard" pattern the RFC's motivating incident hit (a cargo-vendored gz
|
||||
* build; cc-invoked builds do not enable -fstack-clash-protection, and this
|
||||
* file pins that off explicitly so the canary stays a canary even on
|
||||
* hardened-default toolchains). */
|
||||
void smarm_canary_burn(void) {
|
||||
volatile char buf[96 * 1024];
|
||||
buf[0] = 1; /* deepest address first */
|
||||
for (unsigned i = 0; i < sizeof buf; i += 4096) {
|
||||
buf[i] = (char)i;
|
||||
}
|
||||
buf[sizeof buf - 1] = 1;
|
||||
}
|
||||
@@ -75,7 +75,7 @@ genuine advantage over tokio's task abort model.
|
||||
|
||||
### Spawn-heavy workloads (19–70×)
|
||||
|
||||
Every smarm actor `mmap`s a 64 KiB stack reserve with a 64 KiB PROT_NONE guard below (both per-actor configurable since RFC 019; the reserve is demand-paged). This is
|
||||
Every smarm actor `mmap`s a 64 KiB stack with a guard page. This is
|
||||
a syscall. Tokio tasks are heap-allocated state machines — no stack,
|
||||
no syscall, ~100 bytes each. For workloads that spawn thousands of
|
||||
short-lived actors per second, this is a structural disadvantage.
|
||||
|
||||
@@ -1,87 +0,0 @@
|
||||
# Per-switch cost — N=1 local profile (spike findings)
|
||||
|
||||
Measured with `benches/switch_cost.rs` (local mode: one actor, one scheduler,
|
||||
tight `yield_now()` loop = one park/unpark round-trip with no IO/channel/timer
|
||||
and no cross-core traffic). Sandbox: 1 core, kernel 6.18, **no PMU** (hardware
|
||||
counters unavailable), so attribution is from `perf record -e task-clock`
|
||||
(software timer sampling) plus the bench's own rdtsc/wall brackets.
|
||||
|
||||
> **Provenance (post-excision).** This profile was captured against the
|
||||
> spin-enabled build (the pre-excision HEAD, with the RFC 004 spinning workers
|
||||
> live in `src/runtime.rs`). The RFC 004 spinning experiment has since been
|
||||
> excised from `master`; idle schedulers are back to the historical
|
||||
> `thread::sleep` wait. The `futex_wake` attribution below therefore reflects
|
||||
> spinning machinery that is **no longer present on current master** — see the
|
||||
> per-row and per-finding notes. The shim, `schedule_loop`, and run-queue
|
||||
> findings are spin-independent and remain valid.
|
||||
|
||||
## Numbers (stable across runs)
|
||||
|
||||
- Round-trip p50 ≈ **303 ns ≈ 828 cyc** (instrumentation floor subtracted).
|
||||
- Derived effective clock ≈ **2.73 GHz** (rdtsc cyc / wall ns — the two lenses
|
||||
corroborate, so the cycle counts are trustworthy).
|
||||
- p90 316 ns, p99 472 ns; max is a multi-ms OS-deschedule outlier (1 shared
|
||||
core) — ignore the max, trust the percentiles (the harness pools them).
|
||||
|
||||
## Attribution (perf task-clock, self-time, 28.6k samples / 12M round-trips)
|
||||
|
||||
| share | symbol | bucket |
|
||||
|------:|--------|--------|
|
||||
| 24.8% | `runtime::schedule_loop` | scheduler logic (slot-word/epoch + dispatch) |
|
||||
| 8.6% | `MutexQueue::push`/`pop`/`len` | run-queue ops |
|
||||
| ~12% | `do_syscall_64`+`syscall`+`futex_*` | **futex_wake on the hot path** — spinning submit-rule wake; removed by the RFC 004 excision (not on current master) |
|
||||
| 3.5% | `IoThread::drain_completions` | the always-on IO thread (`run()` starts one) |
|
||||
| ~30% | `main` + `clock_gettime`/Timespec + `quicksort` | **instrumentation** (timing + percentile sort) |
|
||||
| ~1% | `switch_to_scheduler`+`switch_to_actor_asm`+ sp accessors | **the context shims + TLS** |
|
||||
|
||||
## Headline finding — revises the handoff hypothesis
|
||||
|
||||
The handoff named the **context shims** (`context.rs`: two `call`s into the
|
||||
TLS sp accessors per switch) as the prime suspect for the per-switch cost.
|
||||
**At N=1 that is not where the time goes — the shims + TLS are ~1% of
|
||||
self-time.** The N=1 cost is dominated by:
|
||||
|
||||
1. **`schedule_loop` + run-queue ops (~33%)** — the epoch/slot-word transition
|
||||
and the mutex run-queue push/pop on every re-queue.
|
||||
2. **A `futex_wake` syscall (~12%)** fired on the hot path even though nothing
|
||||
was parked. This was the spinning **submit-rule wake** introduced by the RFC
|
||||
004 experiment — a parallelism/latency optimisation, not a liveness guard. In
|
||||
a single-scheduler always-runnable loop it was pure cost (no one was ever
|
||||
parked to wake). The RFC 004 excision removed this wake with the rest of the
|
||||
spinning machinery: on current master idle schedulers use `thread::sleep`
|
||||
again, so the N=1 hot path no longer makes this syscall.
|
||||
|
||||
## What this does and does NOT show
|
||||
|
||||
- The handoff's shim hypothesis was a **many-core** hypothesis: its evidence was
|
||||
the N=1→N=8 jump (0.18→1.2µs), attributed to TLS access mode (`__tls_get_addr`
|
||||
vs `#[thread_local]`) and cross-core coherency on the sp/epoch words. **None of
|
||||
that is observable at N=1 on one core.** This profile does NOT refute it; it
|
||||
establishes that the shim is cheap *until cores contend*.
|
||||
- So the spike question sharpens into two separable costs:
|
||||
- **N=1 floor:** scheduler logic (`schedule_loop` + run-queue ops). The
|
||||
futex_wake component was spinning machinery and is gone post-excision, so the
|
||||
remaining N=1 floor is the scheduler core itself.
|
||||
- **N→8 slope:** the shim/TLS/coherency cost. Needs the many-core box + a
|
||||
`remote` bench mode (wake straddling two schedulers) + hardware PMU counters
|
||||
(cache-misses, `MEM_LOAD…HITM` for coherency) — none available in this sandbox.
|
||||
|
||||
## Reproduce
|
||||
|
||||
```sh
|
||||
. "$HOME/.cargo/env"
|
||||
cargo build --release --bench switch_cost
|
||||
BIN=$(ls -t target/release/deps/switch_cost-* | grep -v '\.d$' | head -1)
|
||||
PERF=/usr/lib/linux-tools-6.8.0-124/perf # 6.8 perf on 6.18 kernel; sw events only here
|
||||
|
||||
# bench alone (numbers):
|
||||
SMARM_SWITCH_ROUNDS=3000000 SMARM_SWITCH_WARMUP=50000 SMARM_SWITCH_RUNS=4 "$BIN"
|
||||
|
||||
# attribution (sw task-clock; HW counters need a real PMU / the 5900X):
|
||||
SMARM_SWITCH_ROUNDS=3000000 "$PERF" record -F 4000 -g --call-graph fp -o /tmp/switch.data -- "$BIN"
|
||||
"$PERF" report -i /tmp/switch.data --stdio --no-children
|
||||
```
|
||||
|
||||
On the 5900X with a real PMU, drop `-e task-clock` for `-e cycles,instructions,
|
||||
cache-misses,mem_load_retired.l3_miss` to get the coherency picture the N=8 case
|
||||
needs.
|
||||
+438
-877
File diff suppressed because it is too large
Load Diff
@@ -1,200 +0,0 @@
|
||||
//! Attribution-efficiency probe (RFC 007 follow-up).
|
||||
//!
|
||||
//! Original hypothesis: the ~4pt impact shortfall on the 24-core
|
||||
//! validation (+29.3/+83.5 vs theoretical +33/+100) is a constant
|
||||
//! attribution efficiency eff ≈ 0.91 from site-exit tail truncation.
|
||||
//! The guard-drop flush closed that leak, yet eff held at ~0.93 —
|
||||
//! RESOLVED (2026-07-13 sweep): the residual is runnable off-CPU time
|
||||
//! inside the site (~4.9 slice-expiry yields/entry x ~5.6µs runqueue
|
||||
//! wait), wall time the ground truth below counts but on-CPU
|
||||
//! attribution correctly skips. The offcpu audit bucket now counts it;
|
||||
//! `eff+offcpu` printed per window should sit at ~1.00 — the
|
||||
//! closed-books check.
|
||||
//!
|
||||
//! Measurement: same pipeline as `causal_pipeline`, but the `reserve`
|
||||
//! actor also measures its raw in-site time directly (rdtsc at guard
|
||||
//! enter/exit) and counts site entries. For each experiment window at
|
||||
//! pct%:
|
||||
//!
|
||||
//! eff = (Δglobal_delay / (pct/100)) / Δin_site_cycles
|
||||
//!
|
||||
//! and the missing time per site entry localizes the leak:
|
||||
//!
|
||||
//! tail_us/entry = (Δin_site − Δglobal_delay/(pct/100)) / Δentries
|
||||
//!
|
||||
//! A constant eff across 25/50% with tail/entry in the tens of µs
|
||||
//! localizes a per-entry mechanism; `eff+offcpu` ≈ 1.00 confirms the
|
||||
//! runnable-gap account and rules out any remaining silent loss.
|
||||
//!
|
||||
//! Run: cargo run --release --example causal_attrib_probe --features smarm-causal
|
||||
|
||||
use std::sync::atomic::{AtomicBool, AtomicU64, Ordering};
|
||||
use std::sync::Arc;
|
||||
use std::time::{Duration, Instant};
|
||||
|
||||
fn rdtsc() -> u64 {
|
||||
// x86_64 only — same clock the ledger uses.
|
||||
unsafe { core::arch::x86_64::_rdtsc() }
|
||||
}
|
||||
|
||||
/// Same fixed-work loop as causal_pipeline (dependent LCG, preemptible).
|
||||
fn work_iters(iters: u64) {
|
||||
let mut acc = 0x2545_f491_4f6c_dd1du64;
|
||||
let mut i = 0u64;
|
||||
while i < iters {
|
||||
let chunk_end = (i + 256).min(iters);
|
||||
while i < chunk_end {
|
||||
acc = acc.wrapping_mul(6364136223846793005).wrapping_add(i);
|
||||
i += 1;
|
||||
}
|
||||
std::hint::black_box(acc);
|
||||
smarm::check!();
|
||||
}
|
||||
}
|
||||
|
||||
fn calibrate_iters_per_us() -> u64 {
|
||||
let n = 8_000_000u64;
|
||||
let t = Instant::now();
|
||||
work_iters(n);
|
||||
(n / (t.elapsed().as_micros().max(1) as u64)).max(1)
|
||||
}
|
||||
|
||||
static IN_SITE_CYCLES: AtomicU64 = AtomicU64::new(0);
|
||||
static SITE_ENTRIES: AtomicU64 = AtomicU64::new(0);
|
||||
|
||||
fn main() {
|
||||
let per_us = calibrate_iters_per_us();
|
||||
println!("calibration: {per_us} work iters/µs");
|
||||
let work_us = move |us: u64| work_iters(us * per_us);
|
||||
|
||||
let cores = std::thread::available_parallelism()
|
||||
.map(|n| n.get())
|
||||
.unwrap_or(1);
|
||||
println!("cores: {cores}");
|
||||
if cores < 4 {
|
||||
println!("probe: SKIPPED (needs the stages in parallel)");
|
||||
return;
|
||||
}
|
||||
|
||||
smarm::init(smarm::Config::default()).run(move || {
|
||||
let stop = Arc::new(AtomicBool::new(false));
|
||||
let (tx_ab, rx_ab) = smarm::channel::<u64>();
|
||||
let (tx_bc, rx_bc) = smarm::channel::<u64>();
|
||||
|
||||
let stop_p = stop.clone();
|
||||
let producer = smarm::spawn(move || {
|
||||
let mut i = 0u64;
|
||||
while !stop_p.load(Ordering::Relaxed) {
|
||||
{
|
||||
let _g = smarm::causal_site!("serialize");
|
||||
work_us(200);
|
||||
}
|
||||
if tx_ab.send(i).is_err() {
|
||||
break;
|
||||
}
|
||||
i += 1;
|
||||
}
|
||||
});
|
||||
|
||||
// Reserve: the target — instrumented with ground-truth in-site time.
|
||||
let reserve = smarm::spawn(move || {
|
||||
while let Ok(item) = rx_ab.recv() {
|
||||
{
|
||||
let t0 = rdtsc();
|
||||
let _g = smarm::causal_site!("reserve");
|
||||
work_us(400);
|
||||
// Measured before guard drop: exactly the span the
|
||||
// ledger should be attributing.
|
||||
IN_SITE_CYCLES.fetch_add(rdtsc().saturating_sub(t0), Ordering::Relaxed);
|
||||
SITE_ENTRIES.fetch_add(1, Ordering::Relaxed);
|
||||
}
|
||||
if tx_bc.send(item).is_err() {
|
||||
break;
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
let notify = smarm::spawn(move || {
|
||||
while rx_bc.recv().is_ok() {
|
||||
{
|
||||
let _g = smarm::causal_site!("notify");
|
||||
work_us(50);
|
||||
}
|
||||
smarm::progress!("orders-processed");
|
||||
}
|
||||
});
|
||||
|
||||
let stop_bg = stop.clone();
|
||||
let background = smarm::spawn(move || {
|
||||
while !stop_bg.load(Ordering::Relaxed) {
|
||||
let _g = smarm::causal_site!("background-compaction");
|
||||
work_us(500);
|
||||
}
|
||||
});
|
||||
|
||||
smarm::sleep(Duration::from_millis(300));
|
||||
|
||||
let hz = smarm::causal::tsc_hz();
|
||||
println!("tsc_hz: {:.3} GHz", hz / 1e9);
|
||||
|
||||
// Manual windows so ledger/ground-truth snapshots align exactly.
|
||||
for &pct in &[25u32, 50, 50, 25] {
|
||||
let g0 = smarm::causal::global_delay_cycles();
|
||||
let s0 = IN_SITE_CYCLES.load(Ordering::Relaxed);
|
||||
let e0 = SITE_ENTRIES.load(Ordering::Relaxed);
|
||||
let a0 = smarm::causal::ledger_counters();
|
||||
smarm::causal::begin_experiment_for_test("reserve", pct);
|
||||
smarm::sleep(Duration::from_millis(1000));
|
||||
smarm::causal::end_experiment_for_test();
|
||||
let audit = smarm::causal::ledger_counters().delta_since(&a0);
|
||||
let injected = smarm::causal::global_delay_cycles() - g0;
|
||||
let in_site = IN_SITE_CYCLES.load(Ordering::Relaxed) - s0;
|
||||
let entries = SITE_ENTRIES.load(Ordering::Relaxed) - e0;
|
||||
|
||||
let attributed = injected as f64 / (pct as f64 / 100.0);
|
||||
let eff = attributed / in_site as f64;
|
||||
// Books-closure check: add back the runnable off-CPU gaps the
|
||||
// audit counted (delta terms -> raw via /pct) — should be ~1.00.
|
||||
let eff_closed = (injected as f64 + audit.offcpu_in_site_cycles as f64)
|
||||
/ (pct as f64 / 100.0)
|
||||
/ in_site as f64;
|
||||
let missing = in_site as f64 - attributed;
|
||||
let tail_us = if entries > 0 {
|
||||
missing / entries as f64 / hz * 1e6
|
||||
} else {
|
||||
f64::NAN
|
||||
};
|
||||
println!(
|
||||
"pct {pct:>2}% in_site {:>8.1}ms attributed {:>8.1}ms eff {eff:.3} eff+offcpu {eff_closed:.3} entries {entries} missing/entry {tail_us:.1}µs",
|
||||
in_site as f64 / hz * 1e3,
|
||||
attributed / hz * 1e3,
|
||||
);
|
||||
// RFC 007 deficit hunt: name the losses. Drop/discard columns are
|
||||
// in would-be delta terms — divide by pct/100 to compare with the
|
||||
// missing attribution above.
|
||||
let ms = |c: u64| c as f64 / hz * 1e3;
|
||||
println!(
|
||||
" audit: absorbed {:>7.1}ms forgiven {:>6.1}ms drop park {:>5.2}ms/{:<5} yield {:>5.2}ms/{:<5} offcpu {:>6.2}ms/{:<5} discard >max {:>5.2}ms/{:<3} unarmed {}",
|
||||
ms(audit.spin_absorbed_cycles),
|
||||
ms(audit.park_forgiven_cycles),
|
||||
ms(audit.drop_park_cycles),
|
||||
audit.drop_park_n,
|
||||
ms(audit.drop_yield_cycles),
|
||||
audit.drop_yield_n,
|
||||
ms(audit.offcpu_in_site_cycles),
|
||||
audit.offcpu_in_site_n,
|
||||
ms(audit.discard_overmax_cycles),
|
||||
audit.discard_overmax_n,
|
||||
audit.discard_unarmed_n
|
||||
);
|
||||
smarm::sleep(Duration::from_millis(150));
|
||||
}
|
||||
|
||||
stop.store(true, Ordering::Relaxed);
|
||||
producer.join().unwrap();
|
||||
reserve.join().unwrap();
|
||||
notify.join().unwrap();
|
||||
background.join().unwrap();
|
||||
println!("probe: DONE");
|
||||
});
|
||||
}
|
||||
@@ -1,296 +0,0 @@
|
||||
//! Causal-profiling demo (RFC 007): a pipeline where conventional profiling
|
||||
//! lies and causal profiling doesn't.
|
||||
//!
|
||||
//! producer --(serialize ~200µs/item)--> reserve --(~400µs/item)--> notify
|
||||
//! background: an actor burning CPU constantly, fully off the critical path
|
||||
//!
|
||||
//! `reserve` is the true bottleneck. `serialize` is hot but overlapped with
|
||||
//! `reserve`'s backlog, and `background` is the hottest code in the process
|
||||
//! while contributing nothing to throughput. A cycle profiler ranks them
|
||||
//! background > reserve ≈ 2×serialize; the causal report instead shows
|
||||
//! throughput responding to virtual speedups of `reserve` and (near-)ignoring
|
||||
//! `serialize` and `background`.
|
||||
//!
|
||||
//! Stage cost is fixed *work* (a calibrated arithmetic loop), not fixed wall
|
||||
//! time. This matters: a timed busy-wait absorbs injected causal delay into
|
||||
//! its own budget and finishes on schedule regardless, making every
|
||||
//! experiment read as a no-op (found live on a 24-core run: dead-flat
|
||||
//! deltas). Real workloads are work-shaped, so the demo must be too.
|
||||
//!
|
||||
//! Run:
|
||||
//! cargo run --release --example causal_pipeline --features smarm-causal
|
||||
//!
|
||||
//! Modes (`SMARM_CAUSAL_MODE`), for probing what the guard placement leaves
|
||||
//! out of the measurement (a site speeds up only what it wraps; `recv`/`send`
|
||||
//! on the serialized stage sit outside the canonical guard):
|
||||
//! work (default) — guard wraps only the 400µs of work.
|
||||
//! wide — guard widened over recv + work + send, the whole
|
||||
//! serialized per-item path.
|
||||
//! occupancy — no experiments; times each segment of reserve's loop
|
||||
//! at baseline and reports the unguarded per-item
|
||||
//! overhead δ plus the impact ceiling it implies.
|
||||
//!
|
||||
//! Result (24-core run, 2026-07-13, job f9305cbb): δ measured 0.3µs/item —
|
||||
//! 0.1% of the serialized path — and `wide` does not move the @50% cell
|
||||
//! (+83.5/+86.3 vs work's +81.5/+86.7). This demo's +84-vs-+100 @50%
|
||||
//! shortfall is therefore NOT unguarded stage time; it is controller-side:
|
||||
//! injected delay reaches ~327ms of the ideal 350ms over the 700ms window,
|
||||
//! plus a ~3% real-throughput dip while experiments run. Contrast urus's
|
||||
//! causal_bench, where the same arithmetic identified a real ~70µs/request
|
||||
//! unguarded remainder (recv/reply outside the store guard). Sites measure
|
||||
//! what they wrap — and the occupancy probe tells you which case you're in.
|
||||
//!
|
||||
//! Prints a summary, writes `profile.coz` (Coz plot-compatible), and — given
|
||||
//! enough cores for the pipeline to actually run in parallel — checks the
|
||||
//! expected separation and exits nonzero if it doesn't hold, so a CI box can
|
||||
//! run this as a smoke test.
|
||||
|
||||
use std::sync::atomic::{AtomicBool, Ordering};
|
||||
use std::sync::Arc;
|
||||
use std::time::{Duration, Instant};
|
||||
|
||||
/// LCG-mix `iters` times in dependent sequence (unvectorizable, un-elidable),
|
||||
/// staying preemptible — and causal-sampleable/delayable — via `check!()`.
|
||||
fn work_iters(iters: u64) {
|
||||
let mut acc = 0x2545_f491_4f6c_dd1du64;
|
||||
let mut i = 0u64;
|
||||
while i < iters {
|
||||
let chunk_end = (i + 256).min(iters);
|
||||
while i < chunk_end {
|
||||
acc = acc.wrapping_mul(6364136223846793005).wrapping_add(i);
|
||||
i += 1;
|
||||
}
|
||||
std::hint::black_box(acc);
|
||||
smarm::check!();
|
||||
}
|
||||
}
|
||||
|
||||
/// Measure how many `work_iters` iterations fit in a microsecond on this
|
||||
/// machine, so stage costs below are meaningful in time while staying
|
||||
/// work-shaped.
|
||||
fn calibrate_iters_per_us() -> u64 {
|
||||
let n = 8_000_000u64;
|
||||
let t = Instant::now();
|
||||
work_iters(n);
|
||||
(n / (t.elapsed().as_micros().max(1) as u64)).max(1)
|
||||
}
|
||||
|
||||
/// Guard placement for the `reserve` stage — see module doc.
|
||||
#[derive(Clone, Copy, PartialEq)]
|
||||
enum Mode {
|
||||
Work,
|
||||
Wide,
|
||||
Occupancy,
|
||||
}
|
||||
|
||||
fn main() {
|
||||
let mode = match std::env::var("SMARM_CAUSAL_MODE").as_deref() {
|
||||
Err(_) | Ok("") | Ok("work") => Mode::Work,
|
||||
Ok("wide") => Mode::Wide,
|
||||
Ok("occupancy") => Mode::Occupancy,
|
||||
Ok(other) => {
|
||||
eprintln!("unknown SMARM_CAUSAL_MODE {other:?} (work|wide|occupancy)");
|
||||
std::process::exit(2);
|
||||
}
|
||||
};
|
||||
println!(
|
||||
"mode: {}",
|
||||
match mode {
|
||||
Mode::Work => "work",
|
||||
Mode::Wide => "wide",
|
||||
Mode::Occupancy => "occupancy",
|
||||
}
|
||||
);
|
||||
let per_us = calibrate_iters_per_us();
|
||||
println!("calibration: {per_us} work iters/µs");
|
||||
let work_us = move |us: u64| work_iters(us * per_us);
|
||||
|
||||
let mut failures: Vec<String> = Vec::new();
|
||||
|
||||
smarm::init(smarm::Config::default()).run(move || {
|
||||
let stop = Arc::new(AtomicBool::new(false));
|
||||
|
||||
let (tx_ab, rx_ab) = smarm::channel::<u64>();
|
||||
let (tx_bc, rx_bc) = smarm::channel::<u64>();
|
||||
|
||||
// Producer: hot serialization, but upstream of the bottleneck.
|
||||
let stop_p = stop.clone();
|
||||
let producer = smarm::spawn(move || {
|
||||
let mut i = 0u64;
|
||||
while !stop_p.load(Ordering::Relaxed) {
|
||||
{
|
||||
let _g = smarm::causal_site!("serialize");
|
||||
work_us(200);
|
||||
}
|
||||
if tx_ab.send(i).is_err() {
|
||||
break;
|
||||
}
|
||||
i += 1;
|
||||
}
|
||||
// tx_ab drops here; downstream drains and exits.
|
||||
});
|
||||
|
||||
// Reserve: the true bottleneck (~400µs of work per item).
|
||||
let reserve = smarm::spawn(move || match mode {
|
||||
Mode::Work => {
|
||||
while let Ok(item) = rx_ab.recv() {
|
||||
{
|
||||
let _g = smarm::causal_site!("reserve");
|
||||
work_us(400);
|
||||
}
|
||||
if tx_bc.send(item).is_err() {
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
// Whole serialized per-item path under the guard: a virtual
|
||||
// speedup now also compresses recv/send, so the @50% cell should
|
||||
// recover the theoretical 2× that `work` mode's placement caps.
|
||||
Mode::Wide => loop {
|
||||
let _g = smarm::causal_site!("reserve");
|
||||
let Ok(item) = rx_ab.recv() else { break };
|
||||
work_us(400);
|
||||
if tx_bc.send(item).is_err() {
|
||||
break;
|
||||
}
|
||||
},
|
||||
// Time each segment at baseline; the recv+send remainder δ is
|
||||
// the serialized time a `work`-placed guard cannot speed up.
|
||||
Mode::Occupancy => {
|
||||
let (mut recv_ns, mut work_ns, mut send_ns, mut n) = (0u64, 0u64, 0u64, 0u64);
|
||||
loop {
|
||||
let t0 = Instant::now();
|
||||
let Ok(item) = rx_ab.recv() else { break };
|
||||
let t1 = Instant::now();
|
||||
{
|
||||
let _g = smarm::causal_site!("reserve");
|
||||
work_us(400);
|
||||
}
|
||||
let t2 = Instant::now();
|
||||
if tx_bc.send(item).is_err() {
|
||||
break;
|
||||
}
|
||||
recv_ns += (t1 - t0).as_nanos() as u64;
|
||||
work_ns += (t2 - t1).as_nanos() as u64;
|
||||
send_ns += t2.elapsed().as_nanos() as u64;
|
||||
n += 1;
|
||||
}
|
||||
let items = n.max(1) as f64;
|
||||
let (r, w, s) = (
|
||||
recv_ns as f64 / items / 1e3,
|
||||
work_ns as f64 / items / 1e3,
|
||||
send_ns as f64 / items / 1e3,
|
||||
);
|
||||
let delta = r + s;
|
||||
let total = w + delta;
|
||||
println!("occupancy: {n} items; per item recv {r:.1}µs + work(guarded) {w:.1}µs + send {s:.1}µs");
|
||||
println!(
|
||||
"occupancy: unguarded δ = {delta:.1}µs/item = {:.1}% of the serialized path",
|
||||
100.0 * delta / total
|
||||
);
|
||||
for pct in [25u32, 50] {
|
||||
let f = 1.0 - f64::from(pct) / 100.0;
|
||||
println!(
|
||||
"occupancy: predicted reserve impact @{pct}% -> {:+.1}% (ceiling if δ were guarded: {:+.1}%)",
|
||||
100.0 * (total / (f * w + delta) - 1.0),
|
||||
100.0 * (1.0 / f - 1.0)
|
||||
);
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
// Notify: light tail stage; marks the unit of useful work.
|
||||
let notify = smarm::spawn(move || {
|
||||
while rx_bc.recv().is_ok() {
|
||||
{
|
||||
let _g = smarm::causal_site!("notify");
|
||||
work_us(50);
|
||||
}
|
||||
smarm::progress!("orders-processed");
|
||||
}
|
||||
});
|
||||
|
||||
// Background: hottest code in the process, zero throughput relevance.
|
||||
let stop_bg = stop.clone();
|
||||
let background = smarm::spawn(move || {
|
||||
while !stop_bg.load(Ordering::Relaxed) {
|
||||
let _g = smarm::causal_site!("background-compaction");
|
||||
work_us(500);
|
||||
}
|
||||
});
|
||||
|
||||
// Warm up so queues reach steady state before measuring.
|
||||
smarm::sleep(Duration::from_millis(300));
|
||||
|
||||
if mode == Mode::Occupancy {
|
||||
// No experiments: hold steady state for a window, then drain and
|
||||
// let the reserve actor print its segment report.
|
||||
smarm::sleep(Duration::from_millis(1500));
|
||||
stop.store(true, Ordering::Relaxed);
|
||||
producer.join().unwrap();
|
||||
reserve.join().unwrap();
|
||||
notify.join().unwrap();
|
||||
background.join().unwrap();
|
||||
return;
|
||||
}
|
||||
|
||||
let results = smarm::causal::run_experiments(&smarm::causal::ExperimentPlan {
|
||||
speedups_pct: vec![0, 25, 50],
|
||||
experiment: Duration::from_millis(700),
|
||||
cooldown: Duration::from_millis(150),
|
||||
});
|
||||
|
||||
stop.store(true, Ordering::Relaxed);
|
||||
producer.join().unwrap();
|
||||
reserve.join().unwrap();
|
||||
notify.join().unwrap();
|
||||
background.join().unwrap();
|
||||
|
||||
print!("{}", smarm::causal::render_summary(&results));
|
||||
// RFC 007 deficit hunt: SMARM_CAUSAL_AUDIT=1 appends the per-cell
|
||||
// ledger audit (injected/absorbed/forgiven + drop and discard
|
||||
// buckets) without touching the pinned summary format.
|
||||
if std::env::var_os("SMARM_CAUSAL_AUDIT").is_some() {
|
||||
print!("{}", smarm::causal::render_ledger_audit(&results));
|
||||
}
|
||||
let coz = smarm::causal::render_coz(&results);
|
||||
match std::fs::write("profile.coz", coz) {
|
||||
Ok(()) => println!("\nwrote profile.coz"),
|
||||
Err(e) => eprintln!("\nfailed to write profile.coz: {e}"),
|
||||
}
|
||||
|
||||
// Verdict. The separation only exists when the four pipeline actors
|
||||
// actually run in parallel; on a small box, report and skip.
|
||||
let cores = std::thread::available_parallelism().map(|n| n.get()).unwrap_or(1);
|
||||
if cores < 4 {
|
||||
println!("verdict: SKIPPED ({cores} cores; separation needs the stages in parallel)");
|
||||
return;
|
||||
}
|
||||
let impact = |site: &str| {
|
||||
smarm::causal::impact_pct(&results, site, 25, "orders-processed")
|
||||
};
|
||||
let mut expect = |site: &str, ok: &dyn Fn(f64) -> bool, want: &str| match impact(site) {
|
||||
Some(p) => {
|
||||
let verdict = if ok(p) { "ok" } else { "FAIL" };
|
||||
println!("verdict: {site} @25% -> {p:+.1}% (want {want}) {verdict}");
|
||||
if !ok(p) {
|
||||
failures.push(format!("{site}: {p:+.1}% (want {want})"));
|
||||
}
|
||||
}
|
||||
None => {
|
||||
println!("verdict: {site} @25% -> missing cell FAIL");
|
||||
failures.push(format!("{site}: missing cell"));
|
||||
}
|
||||
};
|
||||
expect("reserve", &|p| p > 15.0, "> +15%");
|
||||
expect("serialize", &|p| p < 10.0, "< +10%");
|
||||
expect("background-compaction", &|p| p < 10.0, "< +10%");
|
||||
|
||||
if failures.is_empty() {
|
||||
println!("verdict: PASS — causal separation holds");
|
||||
} else {
|
||||
println!("verdict: FAIL — {}", failures.join("; "));
|
||||
std::process::exit(1);
|
||||
}
|
||||
});
|
||||
}
|
||||
@@ -1,145 +0,0 @@
|
||||
//! Diagnostic probe for RFC 007 on a target box. Measures, in order:
|
||||
//! 1. TSC frequency against `Instant` (the crate assumes 3 GHz).
|
||||
//! 2. TSC sanity under actor migration: distribution of wall time actually
|
||||
//! spent in `burn_us(400)` across many runs — a bimodal/short tail means
|
||||
//! cross-core TSC offsets are cutting burns short.
|
||||
//! 3. Pipeline stage rates with no experiment running (who is the real
|
||||
//! bottleneck?).
|
||||
//! 4. The same rates during a 50% experiment on `background-compaction`
|
||||
//! (a correct implementation must slow every stage; an off-critical-path
|
||||
//! target must reduce end-to-end throughput proportionally).
|
||||
//!
|
||||
//! Run: cargo run --release --example causal_probe --features smarm-causal
|
||||
|
||||
use std::sync::atomic::{AtomicBool, AtomicU64, Ordering};
|
||||
use std::sync::Arc;
|
||||
use std::time::{Duration, Instant};
|
||||
|
||||
fn burn_us(us: u64) {
|
||||
let cycles = us * 3_000;
|
||||
let start = smarm::preempt::rdtsc();
|
||||
while smarm::preempt::rdtsc().saturating_sub(start) < cycles {
|
||||
smarm::check!();
|
||||
}
|
||||
}
|
||||
|
||||
fn main() {
|
||||
// 1. TSC calibration (plain OS thread, before the runtime starts).
|
||||
let c0 = smarm::preempt::rdtsc();
|
||||
let t0 = Instant::now();
|
||||
std::thread::sleep(Duration::from_millis(200));
|
||||
let hz = (smarm::preempt::rdtsc() - c0) as f64 / t0.elapsed().as_secs_f64();
|
||||
println!("tsc_hz: {:.3e} (crate assumes 3.0e9)", hz);
|
||||
|
||||
smarm::init(smarm::Config::default()).run(move || {
|
||||
// 2. burn_us(400) wall-time distribution inside a migrating actor.
|
||||
let h = smarm::spawn(|| {
|
||||
let mut samples: Vec<u64> = (0..500)
|
||||
.map(|_| {
|
||||
let t = Instant::now();
|
||||
burn_us(400);
|
||||
t.elapsed().as_micros() as u64
|
||||
})
|
||||
.collect();
|
||||
samples.sort_unstable();
|
||||
println!(
|
||||
"burn_us(400) wall us: min {} p10 {} p50 {} p90 {} max {}",
|
||||
samples[0], samples[50], samples[250], samples[450], samples[499]
|
||||
);
|
||||
});
|
||||
h.join().unwrap();
|
||||
|
||||
// 3+4. Pipeline with per-stage counters.
|
||||
let stop = Arc::new(AtomicBool::new(false));
|
||||
let produced = Arc::new(AtomicU64::new(0));
|
||||
let reserved = Arc::new(AtomicU64::new(0));
|
||||
let notified = Arc::new(AtomicU64::new(0));
|
||||
|
||||
let (tx_ab, rx_ab) = smarm::channel::<u64>();
|
||||
let (tx_bc, rx_bc) = smarm::channel::<u64>();
|
||||
|
||||
let stop_p = stop.clone();
|
||||
let produced2 = produced.clone();
|
||||
let producer = smarm::spawn(move || {
|
||||
let mut i = 0u64;
|
||||
while !stop_p.load(Ordering::Relaxed) {
|
||||
{
|
||||
let _g = smarm::causal_site!("serialize");
|
||||
burn_us(200);
|
||||
}
|
||||
if tx_ab.send(i).is_err() {
|
||||
break;
|
||||
}
|
||||
produced2.fetch_add(1, Ordering::Relaxed);
|
||||
i += 1;
|
||||
}
|
||||
});
|
||||
|
||||
let reserved2 = reserved.clone();
|
||||
let reserve = smarm::spawn(move || {
|
||||
while let Ok(item) = rx_ab.recv() {
|
||||
{
|
||||
let _g = smarm::causal_site!("reserve");
|
||||
burn_us(400);
|
||||
}
|
||||
reserved2.fetch_add(1, Ordering::Relaxed);
|
||||
if tx_bc.send(item).is_err() {
|
||||
break;
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
let notified2 = notified.clone();
|
||||
let notify = smarm::spawn(move || {
|
||||
while rx_bc.recv().is_ok() {
|
||||
{
|
||||
let _g = smarm::causal_site!("notify");
|
||||
burn_us(50);
|
||||
}
|
||||
notified2.fetch_add(1, Ordering::Relaxed);
|
||||
smarm::progress!("orders-processed");
|
||||
}
|
||||
});
|
||||
|
||||
let stop_bg = stop.clone();
|
||||
let background = smarm::spawn(move || {
|
||||
while !stop_bg.load(Ordering::Relaxed) {
|
||||
let _g = smarm::causal_site!("background-compaction");
|
||||
burn_us(500);
|
||||
}
|
||||
});
|
||||
|
||||
smarm::sleep(Duration::from_millis(300));
|
||||
|
||||
let window = |label: &str| {
|
||||
let (p0, r0, n0) = (
|
||||
produced.load(Ordering::Relaxed),
|
||||
reserved.load(Ordering::Relaxed),
|
||||
notified.load(Ordering::Relaxed),
|
||||
);
|
||||
let d0 = smarm::causal::global_delay_cycles();
|
||||
let t = Instant::now();
|
||||
smarm::sleep(Duration::from_millis(700));
|
||||
let secs = t.elapsed().as_secs_f64();
|
||||
println!(
|
||||
"{label}: produced {:.0}/s reserved {:.0}/s notified {:.0}/s injected {:.0}ms(assumed-3GHz)",
|
||||
(produced.load(Ordering::Relaxed) - p0) as f64 / secs,
|
||||
(reserved.load(Ordering::Relaxed) - r0) as f64 / secs,
|
||||
(notified.load(Ordering::Relaxed) - n0) as f64 / secs,
|
||||
(smarm::causal::global_delay_cycles() - d0) as f64 / 3.0e9 * 1e3,
|
||||
);
|
||||
};
|
||||
|
||||
window("no-experiment ");
|
||||
smarm::causal::begin_experiment_for_test("background-compaction", 50);
|
||||
window("bg-comp @ 50% ");
|
||||
smarm::causal::end_experiment_for_test();
|
||||
window("post-experiment");
|
||||
|
||||
stop.store(true, Ordering::Relaxed);
|
||||
producer.join().unwrap();
|
||||
reserve.join().unwrap();
|
||||
notify.join().unwrap();
|
||||
background.join().unwrap();
|
||||
});
|
||||
}
|
||||
@@ -1,308 +0,0 @@
|
||||
//! The hand-written **expansion target** of the `gen_statem!` macro: the same
|
||||
//! machine as `examples/gen_statem_macro.rs`, written out in full so the
|
||||
//! primitives can be judged standing on their own. The macro generates exactly
|
||||
//! this shape; nothing here needs the macro to be correct or safe.
|
||||
//!
|
||||
//! The design:
|
||||
//!
|
||||
//! * States and events are real enums. An invalid state is unrepresentable;
|
||||
//! there are no bitflags, no `u32` superpositions, no unsafe unions.
|
||||
//!
|
||||
//! * The dispatch `match (state, event)` IS the transition table. It is
|
||||
//! *total* — no catch-all `_` arm — so:
|
||||
//! - a forgotten (state, event) pair is a non-exhaustive `match` (E0004),
|
||||
//! - a duplicated/conflicting row is `unreachable_patterns` (denied below).
|
||||
//!
|
||||
//! * Single-target rows name their target in the table; the handler (if any)
|
||||
//! is side-effect-only. The table owns the target, so it cannot be wrong.
|
||||
//!
|
||||
//! * Branching rows have the handler return a per-row *successor enum*. A
|
||||
//! target outside that enum is E0599; a missing one is E0004. (See
|
||||
//! `UnlockOutcome` and `on_unlock`.)
|
||||
//!
|
||||
//! * Handlers are module-private. The only way to reach one is through the
|
||||
//! actor's message interface via this table, so a handler that is never
|
||||
//! wired in is dead code — and dead_code is denied below, making an orphan
|
||||
//! handler a compile error.
|
||||
//!
|
||||
//! * Drops onto the `Machine` / `Resolution` / `Cx` primitives with no change
|
||||
//! to `src/gen_statem.rs`.
|
||||
//!
|
||||
//! Default build is clean and runs. A BREAK-CASE MENU at the bottom documents
|
||||
//! how to make each of the four guarantees fire.
|
||||
//!
|
||||
//! Run: `cargo run --example gen_statem_expanded`
|
||||
|
||||
#![deny(dead_code, unreachable_patterns)]
|
||||
|
||||
use smarm::gen_statem::{spawn, Cx, GenStatemRef, Machine, Reply, Resolution, Step};
|
||||
use smarm::run;
|
||||
|
||||
// === user types ============================================================
|
||||
|
||||
#[derive(Clone, Copy, PartialEq, Eq, Debug)]
|
||||
enum Door {
|
||||
Open,
|
||||
Closed,
|
||||
Locked,
|
||||
}
|
||||
|
||||
struct Data {
|
||||
enters: u32, // total state entries (incl. initial)
|
||||
pushes: u32, // times a push closed the door
|
||||
knocks: u32, // knocks answered (a Locked knock is postponed, then counted)
|
||||
}
|
||||
|
||||
enum Cast {
|
||||
Push,
|
||||
Pull,
|
||||
Lock,
|
||||
Unlock(u32), // carries a key
|
||||
Knock, // counted when the door is reachable; postponed while Locked
|
||||
}
|
||||
|
||||
enum Call {
|
||||
GetState(Reply<Door>),
|
||||
GetEnters(Reply<u32>),
|
||||
GetPushes(Reply<u32>),
|
||||
GetKnocks(Reply<u32>),
|
||||
}
|
||||
|
||||
enum Ev {
|
||||
Cast(Cast),
|
||||
Call(Call),
|
||||
// The runtime's internal events. `Info` is out-of-band (here unused, so
|
||||
// `()`); `StateTimeout` / `Timeout` are timer fires the loop feeds back in.
|
||||
Info(()),
|
||||
StateTimeout,
|
||||
Timeout(&'static str),
|
||||
}
|
||||
|
||||
const CODE: u32 = 1234;
|
||||
|
||||
// === per-row successor enum for the one branching row ======================
|
||||
// `Locked + Unlock` may end in Closed (right key) or Locked (wrong key) — and
|
||||
// nothing else. This type IS that declared set; `on_unlock` cannot name Open.
|
||||
enum UnlockOutcome {
|
||||
Closed,
|
||||
Locked,
|
||||
}
|
||||
|
||||
impl From<UnlockOutcome> for Door {
|
||||
fn from(o: UnlockOutcome) -> Door {
|
||||
match o {
|
||||
UnlockOutcome::Closed => Door::Closed,
|
||||
UnlockOutcome::Locked => Door::Locked,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// === module-private handlers (side effects / branch choice only) ===========
|
||||
// Reachable solely through the table below. Orphan one and it is dead_code.
|
||||
|
||||
fn on_push(data: &mut Data) {
|
||||
data.pushes += 1;
|
||||
}
|
||||
|
||||
fn on_unlock(key: u32) -> UnlockOutcome {
|
||||
if key == CODE {
|
||||
UnlockOutcome::Closed
|
||||
} else {
|
||||
UnlockOutcome::Locked // wrong key: caller will see this == current -> stay
|
||||
}
|
||||
}
|
||||
|
||||
// === the machine ===========================================================
|
||||
|
||||
struct DoorSm {
|
||||
state: Door,
|
||||
data: Data,
|
||||
}
|
||||
|
||||
impl DoorSm {
|
||||
fn start(init: Door) -> GenStatemRef<DoorSm> {
|
||||
spawn(DoorSm {
|
||||
state: init,
|
||||
data: Data {
|
||||
enters: 0,
|
||||
pushes: 0,
|
||||
knocks: 0,
|
||||
},
|
||||
})
|
||||
}
|
||||
|
||||
fn enter(&mut self, cx: &mut Cx<Ev>) {
|
||||
self.data.enters += 1;
|
||||
// A state-timeout: an Open door auto-closes after a quiet window. The
|
||||
// loop auto-resets it on any transition, so it fires only if the door is
|
||||
// still Open when it elapses.
|
||||
if self.state == Door::Open {
|
||||
cx.state_timeout(std::time::Duration::from_millis(5));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl Machine for DoorSm {
|
||||
type Ev = Ev;
|
||||
|
||||
fn state_timeout_ev() -> Ev {
|
||||
Ev::StateTimeout
|
||||
}
|
||||
|
||||
fn timeout_ev(name: &'static str) -> Ev {
|
||||
Ev::Timeout(name)
|
||||
}
|
||||
|
||||
fn on_start(&mut self, cx: &mut Cx<Ev>) {
|
||||
self.enter(cx);
|
||||
}
|
||||
|
||||
fn handle(&mut self, ev: Ev, cx: &mut Cx<Ev>) -> Step<Ev> {
|
||||
let prev = self.state;
|
||||
|
||||
// ---- phase 1: postpone routing (borrow-only) ----------------------
|
||||
// A deferred event is handed back untouched for the loop's postpone
|
||||
// queue; no handler code runs on it. Here: a knock at a locked door.
|
||||
// (The macro emits this as a `match (state, &ev)` yielding a bool; the
|
||||
// hand-written form can just test directly.)
|
||||
if let (Door::Locked, Ev::Cast(Cast::Knock)) = (prev, &ev) {
|
||||
return Step::Postponed(ev);
|
||||
}
|
||||
|
||||
// ---- phase 2: the consuming transition table ----------------------
|
||||
// Total over (Door, Ev). Read it as the declared graph: each `=> To(x)`
|
||||
// is an edge, each `=> Unhandled` an explicit refusal. The postponed
|
||||
// pair above reappears as `unreachable!` so the match stays total.
|
||||
let res: Resolution<Door> = match (self.state, ev) {
|
||||
// --- Open -------------------------------------------------------
|
||||
(Door::Open, Ev::Cast(Cast::Push)) => {
|
||||
on_push(&mut self.data);
|
||||
Resolution::To(Door::Closed)
|
||||
}
|
||||
(Door::Open, Ev::Cast(Cast::Knock)) => {
|
||||
self.data.knocks += 1;
|
||||
Resolution::To(prev)
|
||||
}
|
||||
(Door::Open, Ev::Cast(Cast::Pull | Cast::Lock | Cast::Unlock(_))) => {
|
||||
Resolution::Unhandled
|
||||
}
|
||||
|
||||
// --- Closed -----------------------------------------------------
|
||||
(Door::Closed, Ev::Cast(Cast::Pull)) => Resolution::To(Door::Open),
|
||||
(Door::Closed, Ev::Cast(Cast::Lock)) => Resolution::To(Door::Locked),
|
||||
(Door::Closed, Ev::Cast(Cast::Knock)) => {
|
||||
self.data.knocks += 1;
|
||||
Resolution::To(prev)
|
||||
}
|
||||
(Door::Closed, Ev::Cast(Cast::Push | Cast::Unlock(_))) => Resolution::Unhandled,
|
||||
|
||||
// --- Locked (branching row: handler picks within UnlockOutcome) -
|
||||
(Door::Locked, Ev::Cast(Cast::Unlock(key))) => Resolution::To(on_unlock(key).into()),
|
||||
// Routed out in phase 1; listed only to keep this match total.
|
||||
(Door::Locked, Ev::Cast(Cast::Knock)) => {
|
||||
unreachable!("postponed event is replayed, not dispatched here")
|
||||
}
|
||||
(Door::Locked, Ev::Cast(Cast::Push | Cast::Pull | Cast::Lock)) => Resolution::Unhandled,
|
||||
|
||||
// --- state-independent queries (reply, then stay) ---------------
|
||||
(_, Ev::Call(Call::GetState(r))) => {
|
||||
r.reply(prev);
|
||||
Resolution::To(prev)
|
||||
}
|
||||
(_, Ev::Call(Call::GetEnters(r))) => {
|
||||
r.reply(self.data.enters);
|
||||
Resolution::To(prev)
|
||||
}
|
||||
(_, Ev::Call(Call::GetPushes(r))) => {
|
||||
r.reply(self.data.pushes);
|
||||
Resolution::To(prev)
|
||||
}
|
||||
(_, Ev::Call(Call::GetKnocks(r))) => {
|
||||
r.reply(self.data.knocks);
|
||||
Resolution::To(prev)
|
||||
}
|
||||
|
||||
// --- timeouts: an Open door auto-closes; others have none armed --
|
||||
(Door::Open, Ev::StateTimeout) => Resolution::To(Door::Closed),
|
||||
(_, Ev::StateTimeout) => Resolution::Unhandled,
|
||||
(_, Ev::Timeout(name)) => {
|
||||
// No named timeout is armed in this run; a real handler would
|
||||
// dispatch on `name`. Acknowledge it to exercise the field.
|
||||
let _ = name;
|
||||
Resolution::Unhandled
|
||||
}
|
||||
|
||||
// --- out-of-band info: silent drop (the gen_server default) ------
|
||||
(_, Ev::Info(_)) => Resolution::Unhandled,
|
||||
};
|
||||
|
||||
// ---- apply the resolution -----------------------------------------
|
||||
match res {
|
||||
Resolution::To(s) if s == prev => Step::Stayed, // stay: no enter
|
||||
Resolution::To(s) => {
|
||||
self.state = s; // sole writer of the state cell
|
||||
cx.__reset_state_timeout(); // auto-reset across transitions
|
||||
self.enter(cx);
|
||||
Step::Transitioned
|
||||
}
|
||||
Resolution::Unhandled => {
|
||||
cx.on_unhandled();
|
||||
Step::Stayed
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn main() {
|
||||
run(|| {
|
||||
let door = DoorSm::start(Door::Closed);
|
||||
|
||||
door.send(Ev::Cast(Cast::Lock)).unwrap(); // Closed -> Locked
|
||||
door.send(Ev::Cast(Cast::Knock)).unwrap(); // Locked: postponed (not yet counted)
|
||||
door.send(Ev::Cast(Cast::Push)).unwrap(); // Locked: Push invalid -> Unhandled
|
||||
door.send(Ev::Cast(Cast::Unlock(0))).unwrap(); // Locked: wrong key -> stay (knock still deferred)
|
||||
door.send(Ev::Cast(Cast::Unlock(CODE))).unwrap(); // Locked -> Closed; deferred Knock replays here
|
||||
door.send(Ev::Cast(Cast::Push)).unwrap(); // Closed: Push invalid -> Unhandled
|
||||
door.send(Ev::Cast(Cast::Pull)).unwrap(); // Closed -> Open (arms 5ms auto-close)
|
||||
door.send(Ev::Info(())).unwrap(); // out-of-band: silently dropped
|
||||
|
||||
// Wait past the auto-close window: the state-timeout fires and the door
|
||||
// closes itself, with no further input. (`smarm::sleep` parks the actor
|
||||
// without blocking a worker thread, so the timer wheel keeps turning.)
|
||||
smarm::sleep(std::time::Duration::from_millis(40));
|
||||
|
||||
let st = door.call(|r| Ev::Call(Call::GetState(r))).unwrap();
|
||||
let enters = door.call(|r| Ev::Call(Call::GetEnters(r))).unwrap();
|
||||
let pushes = door.call(|r| Ev::Call(Call::GetPushes(r))).unwrap();
|
||||
let knocks = door.call(|r| Ev::Call(Call::GetKnocks(r))).unwrap();
|
||||
|
||||
println!("state={st:?} enters={enters} pushes={pushes} knocks={knocks}");
|
||||
assert_eq!(st, Door::Closed); // auto-closed by the state-timeout
|
||||
assert_eq!(enters, 5); // Closed(start) + Locked + Closed + Open + Closed
|
||||
assert_eq!(pushes, 0); // no push ever closed it this run
|
||||
assert_eq!(knocks, 1); // the locked-door knock, replayed once unlocked
|
||||
println!("ok");
|
||||
});
|
||||
}
|
||||
|
||||
// ===========================================================================
|
||||
// BREAK-CASE MENU — each makes one compile-time guarantee fire.
|
||||
//
|
||||
// 1. ORPHAN HANDLER (dead_code -> error):
|
||||
// add `fn on_slam(_d: &mut Data) {}` and don't reference it.
|
||||
// => error: function `on_slam` is never used
|
||||
//
|
||||
// 2. CONFLICTING ROW (unreachable_patterns -> error):
|
||||
// duplicate an arm, e.g. add a second
|
||||
// `(Door::Open, Ev::Cast(Cast::Push)) => Resolution::Unhandled,`
|
||||
// => error: unreachable pattern
|
||||
//
|
||||
// 3. MISSING PAIR (non-exhaustive match, E0004):
|
||||
// delete the `(Door::Locked, Ev::Cast(Cast::Push | Cast::Pull | Cast::Lock))`
|
||||
// arm.
|
||||
// => error[E0004]: non-exhaustive patterns: ... not covered
|
||||
//
|
||||
// 4. OUT-OF-SET TARGET (E0599):
|
||||
// in `on_unlock`, return `UnlockOutcome::Open`.
|
||||
// => error[E0599]: no variant ... named `Open` found for enum `UnlockOutcome`
|
||||
// ===========================================================================
|
||||
@@ -1,194 +0,0 @@
|
||||
//! The **same** machine as `examples/gen_statem_expanded.rs`, written through the
|
||||
//! `gen_statem!` macro. Diff this file against that one to see exactly what the
|
||||
//! macro buys: every `// ===` section there that was boilerplate (the `Ev`
|
||||
//! enum, the `DoorSm` struct, `start`, the whole `Machine` impl, the `enter`
|
||||
//! dispatch, the stay/transition apply-tail) collapses into the invocation
|
||||
//! below. What stays hand-written is what carries meaning: the four types, the
|
||||
//! per-state successor enum, and the handler fns.
|
||||
//!
|
||||
//! The point of the exercise is that the macro is *pure sugar*: the four
|
||||
//! compile-time guarantees the hand-written form demonstrates are properties of
|
||||
//! the emitted code, not of the macro, so they survive expansion unchanged. The
|
||||
//! BREAK-CASE MENU at the bottom is the same four cases, re-expressed against
|
||||
//! the macro surface — flip any one on and the compiler fires identically.
|
||||
//!
|
||||
//! Run: `cargo run --example gen_statem_macro`
|
||||
|
||||
#![deny(dead_code)] // guarantee #3 (orphan handlers); the macro denies the
|
||||
// dispatch's own unreachable_patterns internally.
|
||||
|
||||
use smarm::gen_statem;
|
||||
use smarm::gen_statem::Reply;
|
||||
use smarm::run;
|
||||
|
||||
// === user types (identical to gen_statem_expanded.rs) =========================
|
||||
|
||||
#[derive(Clone, Copy, PartialEq, Eq, Debug)]
|
||||
enum Door {
|
||||
Open,
|
||||
Closed,
|
||||
Locked,
|
||||
}
|
||||
|
||||
struct Data {
|
||||
enters: u32, // total state entries (incl. initial)
|
||||
pushes: u32, // times a push closed the door
|
||||
knocks: u32, // knocks answered (a Locked knock is postponed, then counted)
|
||||
}
|
||||
|
||||
enum Cast {
|
||||
Push,
|
||||
Pull,
|
||||
Lock,
|
||||
Unlock(u32), // carries a key
|
||||
Knock, // counted when the door is reachable; postponed while Locked
|
||||
}
|
||||
|
||||
enum Call {
|
||||
GetState(Reply<Door>),
|
||||
GetEnters(Reply<u32>),
|
||||
GetPushes(Reply<u32>),
|
||||
GetKnocks(Reply<u32>),
|
||||
}
|
||||
|
||||
const CODE: u32 = 1234;
|
||||
|
||||
// === per-row successor enum for the one branching row ======================
|
||||
// `Locked + Unlock` may end in Closed (right key) or Locked (wrong key) — and
|
||||
// nothing else. This type IS that declared set; `on_unlock` cannot name Open.
|
||||
enum UnlockOutcome {
|
||||
Closed,
|
||||
Locked,
|
||||
}
|
||||
|
||||
impl From<UnlockOutcome> for Door {
|
||||
fn from(o: UnlockOutcome) -> Door {
|
||||
match o {
|
||||
UnlockOutcome::Closed => Door::Closed,
|
||||
UnlockOutcome::Locked => Door::Locked,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// === module-private handlers (side effects / branch choice only) ===========
|
||||
// Reachable solely through the table below. Orphan one and it is dead_code.
|
||||
|
||||
fn on_push(data: &mut Data) {
|
||||
data.pushes += 1;
|
||||
}
|
||||
|
||||
fn on_unlock(key: u32) -> UnlockOutcome {
|
||||
if key == CODE {
|
||||
UnlockOutcome::Closed
|
||||
} else {
|
||||
UnlockOutcome::Locked // wrong key: caller will see this == current -> stay
|
||||
}
|
||||
}
|
||||
|
||||
// === the machine ===========================================================
|
||||
// Everything below — Ev, DoorSm, start, Machine, enter — is generated.
|
||||
|
||||
gen_statem! {
|
||||
machine: DoorSm { state: Door, data: Data };
|
||||
event: Ev { cast: Cast, call: Call, info: () };
|
||||
|
||||
// You name the bindings the bodies use; the macro can't lend you its own
|
||||
// `self`/`cx` across macro hygiene. `data` = &mut Data, `prev` = current
|
||||
// state tag, `cx` = context handle (unused here).
|
||||
context(data, prev, cx);
|
||||
|
||||
enter {
|
||||
_ => data.enters += 1,
|
||||
}
|
||||
|
||||
on Door::Open => {
|
||||
cast Cast::Push => { on_push(data); Door::Closed },
|
||||
cast Cast::Knock => { data.knocks += 1; prev },
|
||||
cast Cast::Pull | Cast::Lock | Cast::Unlock(_) => unhandled,
|
||||
}
|
||||
on Door::Closed => {
|
||||
cast Cast::Pull => Door::Open,
|
||||
cast Cast::Lock => Door::Locked,
|
||||
cast Cast::Knock => { data.knocks += 1; prev },
|
||||
cast Cast::Push | Cast::Unlock(_) => unhandled,
|
||||
}
|
||||
on Door::Locked => {
|
||||
cast Cast::Unlock(key) => on_unlock(key), // branch -> UnlockOutcome
|
||||
// A knock at a locked door waits: defer it until the door is reachable,
|
||||
// where the replay counts it.
|
||||
cast Cast::Knock => postpone,
|
||||
cast Cast::Push | Cast::Pull | Cast::Lock => unhandled,
|
||||
}
|
||||
|
||||
// state-independent queries (reply, then stay via `prev`)
|
||||
on _ => {
|
||||
call Call::GetState(r) => { r.reply(prev); prev },
|
||||
call Call::GetEnters(r) => { r.reply(data.enters); prev },
|
||||
call Call::GetPushes(r) => { r.reply(data.pushes); prev },
|
||||
call Call::GetKnocks(r) => { r.reply(data.knocks); prev },
|
||||
// This machine arms no timeouts, so refuse them everywhere. (Info has a
|
||||
// built-in silent-drop default, so it needs no row.)
|
||||
state_timeout => unhandled,
|
||||
timeout _ => unhandled,
|
||||
}
|
||||
}
|
||||
|
||||
fn main() {
|
||||
run(|| {
|
||||
let door = DoorSm::start(
|
||||
Door::Closed,
|
||||
Data {
|
||||
enters: 0,
|
||||
pushes: 0,
|
||||
knocks: 0,
|
||||
},
|
||||
);
|
||||
|
||||
door.send(Ev::Cast(Cast::Lock)).unwrap(); // Closed -> Locked
|
||||
door.send(Ev::Cast(Cast::Knock)).unwrap(); // Locked: postponed (not yet counted)
|
||||
door.send(Ev::Cast(Cast::Push)).unwrap(); // Locked: Push invalid -> Unhandled
|
||||
door.send(Ev::Cast(Cast::Unlock(0))).unwrap(); // Locked: wrong key -> stay (knock still deferred)
|
||||
door.send(Ev::Cast(Cast::Unlock(CODE))).unwrap(); // Locked -> Closed; deferred Knock replays here
|
||||
door.send(Ev::Cast(Cast::Pull)).unwrap(); // Closed -> Open
|
||||
door.send(Ev::Cast(Cast::Push)).unwrap(); // Open -> Closed (pushes=1)
|
||||
|
||||
let st = door.call(|r| Ev::Call(Call::GetState(r))).unwrap();
|
||||
let enters = door.call(|r| Ev::Call(Call::GetEnters(r))).unwrap();
|
||||
let pushes = door.call(|r| Ev::Call(Call::GetPushes(r))).unwrap();
|
||||
let knocks = door.call(|r| Ev::Call(Call::GetKnocks(r))).unwrap();
|
||||
|
||||
println!("state={st:?} enters={enters} pushes={pushes} knocks={knocks}");
|
||||
assert_eq!(st, Door::Closed);
|
||||
assert_eq!(enters, 5); // Closed(start) + Locked + Closed + Open + Closed
|
||||
assert_eq!(pushes, 1);
|
||||
assert_eq!(knocks, 1); // the locked-door knock, replayed once unlocked
|
||||
println!("ok");
|
||||
});
|
||||
}
|
||||
|
||||
// ===========================================================================
|
||||
// BREAK-CASE MENU — the four guarantees, through the macro. Each fires exactly
|
||||
// as it does in the hand-written gen_statem_expanded.rs.
|
||||
//
|
||||
// 1. ORPHAN HANDLER (dead_code -> error):
|
||||
// add `fn on_slam(_d: &mut Data) {}` and don't reference it.
|
||||
// => error: function `on_slam` is never used
|
||||
//
|
||||
// 2. CONFLICTING ROW (unreachable_patterns):
|
||||
// duplicate a row, e.g. add a second
|
||||
// `cast Cast::Push => unhandled,` under `on Door::Open`.
|
||||
// NOTE: this example is a separate crate from `smarm`, so rustc's
|
||||
// in_external_macro rule SILENCES this lint here even though the macro
|
||||
// denies it — the duplicate compiles. The guarantee is real only for
|
||||
// machines defined inside the `smarm` crate itself. This is the documented
|
||||
// macro_rules! limitation.
|
||||
//
|
||||
// 3. MISSING PAIR (non-exhaustive match, E0004):
|
||||
// delete the `cast Cast::Push | Cast::Pull | Cast::Lock => unhandled,`
|
||||
// row under `on Door::Locked`.
|
||||
// => error[E0004]: non-exhaustive patterns: ... not covered
|
||||
//
|
||||
// 4. OUT-OF-SET TARGET (E0599):
|
||||
// in `on_unlock`, return `UnlockOutcome::Open`.
|
||||
// => error[E0599]: no variant ... named `Open` found for enum `UnlockOutcome`
|
||||
// ===========================================================================
|
||||
@@ -1,139 +0,0 @@
|
||||
//! Graceful shutdown, end to end: a supervised app tree, a server that
|
||||
//! drains before it exits, and the two ways the whole thing winds down.
|
||||
//!
|
||||
//! Stopping an actor comes in two strengths, as in OTP:
|
||||
//! - `request_stop(pid)` = `exit(Pid, kill)`: cooperative hard stop,
|
||||
//! unwinds at the next observation point.
|
||||
//! - `request_shutdown(pid)` = `exit(Pid, shutdown)`: a trapping target gets
|
||||
//! an `ExitSignal { reason: Shutdown }` and winds
|
||||
//! down on its own terms; a non-trapping one is
|
||||
//! stopped outright.
|
||||
//!
|
||||
//! A supervisor traps exits. `request_shutdown(sup)` runs its ordered
|
||||
//! shutdown — children in reverse start order, each per its `ChildSpec`
|
||||
//! `Shutdown` policy (`Timeout(d)` default 5s, `Infinity`, `BrutalKill`) —
|
||||
//! and the supervisor then returns normally.
|
||||
//!
|
||||
//! Two triggers are shown:
|
||||
//! 1. **Root exit.** The run's root actor returning means "the program is
|
||||
//! done": the runtime delivers `request_shutdown` to every top-level actor
|
||||
//! (here: the supervisor). Trapping actors may keep running to drain and
|
||||
//! end the run when they stop themselves; non-trapping ones are stopped.
|
||||
//! 2. **An outside thread** (e.g. a signal handler) driving it via
|
||||
//! `RuntimeHandle::request_shutdown` on the supervisor — the root then
|
||||
//! just waits for the tree to come down.
|
||||
|
||||
use smarm::gen_server::{
|
||||
GenServer, GenServerBuilder, GenServerCtx, GenServerName, ShutdownAction, StopHandle,
|
||||
TimerHandle,
|
||||
};
|
||||
use smarm::supervisor::{ChildSpec, OneForOne, Restart, Shutdown};
|
||||
use smarm::{sleep, spawn};
|
||||
use std::thread;
|
||||
use std::time::Duration;
|
||||
|
||||
/// A server with in-flight work: on shutdown it stops accepting, finishes what
|
||||
/// it has (simulated with a ticking timer), then ends itself.
|
||||
struct Drainer {
|
||||
pending: u32,
|
||||
stop: Option<StopHandle<Drainer>>,
|
||||
timer: Option<TimerHandle<Drainer>>,
|
||||
}
|
||||
|
||||
impl GenServer for Drainer {
|
||||
type Call = ();
|
||||
type Reply = ();
|
||||
type Cast = ();
|
||||
type Info = ();
|
||||
type Timer = ();
|
||||
|
||||
fn init(&mut self, ctx: &GenServerCtx<Self>) {
|
||||
ctx.trap_exit(); // opt in: shutdown arrives as handle_shutdown
|
||||
self.stop = Some(ctx.stop_handle());
|
||||
self.timer = Some(ctx.timer());
|
||||
}
|
||||
fn handle_call(&mut self, _: ()) {}
|
||||
fn handle_cast(&mut self, _: ()) {}
|
||||
fn handle_shutdown(&mut self) -> ShutdownAction {
|
||||
println!(
|
||||
"drainer: shutdown requested, {} items pending",
|
||||
self.pending
|
||||
);
|
||||
self.timer
|
||||
.as_ref()
|
||||
.unwrap()
|
||||
.tick_every(Duration::from_millis(20), ());
|
||||
ShutdownAction::Continue // keep serving until drained
|
||||
}
|
||||
fn handle_timer(&mut self, _: ()) {
|
||||
self.pending -= 1;
|
||||
if self.pending == 0 {
|
||||
println!("drainer: drained, stopping");
|
||||
self.stop.as_ref().unwrap().stop(); // normal exit
|
||||
}
|
||||
}
|
||||
fn terminate(&mut self) {
|
||||
// Graceful path: this runs on the normal path and may block.
|
||||
println!("drainer: terminate");
|
||||
}
|
||||
}
|
||||
|
||||
/// The server's name: how the rest of the app reaches it (and the only handle
|
||||
/// that survives a restart).
|
||||
const DRAINER: GenServerName<Drainer> = GenServerName::new("drainer");
|
||||
|
||||
fn app_tree() -> OneForOne {
|
||||
OneForOne::new()
|
||||
.child(
|
||||
ChildSpec::new(Restart::Permanent, || {
|
||||
// A plain worker that does not trap: stopped outright on shutdown.
|
||||
loop {
|
||||
sleep(Duration::from_millis(10));
|
||||
}
|
||||
})
|
||||
.shutdown(Shutdown::Timeout(Duration::from_millis(100))),
|
||||
)
|
||||
// A gen_server is a direct child: `named(N).run()` runs the loop as
|
||||
// the child actor itself, so the supervisor's shutdown arrives as
|
||||
// `handle_shutdown` and a restart re-binds the name.
|
||||
.child(
|
||||
ChildSpec::new(Restart::Permanent, || {
|
||||
GenServerBuilder::new(Drainer {
|
||||
pending: 3,
|
||||
stop: None,
|
||||
timer: None,
|
||||
})
|
||||
.named(DRAINER)
|
||||
.run()
|
||||
.expect("drainer name is free");
|
||||
})
|
||||
.shutdown(Shutdown::Infinity),
|
||||
)
|
||||
}
|
||||
|
||||
fn main() {
|
||||
println!("--- 1. root exit drives the shutdown ---");
|
||||
smarm::run(|| {
|
||||
spawn(|| app_tree().run());
|
||||
sleep(Duration::from_millis(50)); // the app "runs" for a while
|
||||
// Returning here asks the supervisor to shut down; the run ends when
|
||||
// the tree — drainer included — is gone.
|
||||
});
|
||||
|
||||
println!("--- 2. an outside thread drives the shutdown ---");
|
||||
let rt = smarm::init(smarm::Config::default());
|
||||
let handle = rt.handle(); // Send + Sync; grab it before run
|
||||
rt.run(move || {
|
||||
let sup = spawn(|| app_tree().run());
|
||||
let sup_pid = sup.pid();
|
||||
// Stand-in for a SIGTERM handler thread.
|
||||
thread::spawn(move || {
|
||||
thread::sleep(Duration::from_millis(50));
|
||||
println!("signal thread: requesting shutdown");
|
||||
handle.request_shutdown(sup_pid);
|
||||
});
|
||||
sup.join()
|
||||
.expect("supervisor returns normally after ordered shutdown");
|
||||
println!("supervisor down; root returns");
|
||||
});
|
||||
}
|
||||
@@ -1,77 +0,0 @@
|
||||
//! Addressing a gen_server by a durable name.
|
||||
//!
|
||||
//! A gen_server is multi-message (call / cast over one inbox), so it is named
|
||||
//! by the *server* type rather than by a single message type. Registering it
|
||||
//! under a [`GenServerName`] lets clients `call` and `cast` by name, resolving on
|
||||
//! every use — so the address keeps working across a supervised restart, with
|
||||
//! no stale [`GenServerRef`] to refresh.
|
||||
|
||||
use smarm::{
|
||||
call, cast, run, whereis_server, GenServer, GenServerBuilder, GenServerName, GenServerRef,
|
||||
};
|
||||
|
||||
/// A counter server: synchronous `Get`, asynchronous `Inc` / `Add`.
|
||||
struct Counter {
|
||||
n: u64,
|
||||
}
|
||||
enum Query {
|
||||
Get,
|
||||
}
|
||||
enum Update {
|
||||
Inc,
|
||||
Add(u64),
|
||||
}
|
||||
impl GenServer for Counter {
|
||||
type Call = Query;
|
||||
type Reply = u64;
|
||||
type Cast = Update;
|
||||
type Info = ();
|
||||
type Timer = ();
|
||||
|
||||
fn handle_call(&mut self, q: Query) -> u64 {
|
||||
match q {
|
||||
Query::Get => self.n,
|
||||
}
|
||||
}
|
||||
fn handle_cast(&mut self, u: Update) {
|
||||
match u {
|
||||
Update::Inc => self.n += 1,
|
||||
Update::Add(k) => self.n += k,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// A durable name typed by the server, so by-name `call` / `cast` check against
|
||||
/// `Counter`'s `Call` / `Cast` / `Reply`.
|
||||
const COUNTER: GenServerName<Counter> = GenServerName::new("counter");
|
||||
|
||||
fn main() {
|
||||
run(|| {
|
||||
// Start the server and bind its name in one step. A named start is
|
||||
// fallible: the name may already be held by another live server.
|
||||
GenServerBuilder::new(Counter { n: 0 })
|
||||
.named(COUNTER)
|
||||
.start()
|
||||
.unwrap();
|
||||
|
||||
// Address it purely by name. Each call/cast resolves through the
|
||||
// registry, so a server restarted under the same name is reached
|
||||
// transparently.
|
||||
cast(COUNTER, Update::Inc).unwrap();
|
||||
cast(COUNTER, Update::Add(41)).unwrap();
|
||||
assert_eq!(call(COUNTER, Query::Get).unwrap(), 42);
|
||||
|
||||
// When you want a handle to hold or pass on rather than resolve per
|
||||
// call, recover a typed `GenServerRef` from the name.
|
||||
let svc: Option<GenServerRef<Counter>> = whereis_server(COUNTER);
|
||||
if let Some(svc) = svc {
|
||||
let _ = svc.call(Query::Get);
|
||||
// A named server is pinned alive by the registry, so dropping refs
|
||||
// does not end it. Stop it explicitly: `shutdown()` asks politely
|
||||
// (a trapping server drains first; this one is stopped outright)
|
||||
// and waits until it is gone. Left running, the root's return
|
||||
// would shut it down the same way — see examples/graceful_shutdown.rs.
|
||||
svc.shutdown();
|
||||
}
|
||||
});
|
||||
}
|
||||
@@ -1,160 +0,0 @@
|
||||
//! The live observer (RFC 016 Chunk 4) producing an OTP `observer`-flavoured
|
||||
//! dump of a running system.
|
||||
//!
|
||||
//! Run it with the feature on:
|
||||
//!
|
||||
//! ```text
|
||||
//! cargo run --example observer --features observer
|
||||
//! ```
|
||||
//!
|
||||
//! It stands up a tiny tree — a named service plus two workers parked on a gate
|
||||
//! — starts the [`observer`](smarm::observer) gen_server, then asks it for a
|
||||
//! snapshot and a tree over the call channel and renders both. The observer is
|
||||
//! pure transport: every line below is the Chunk-1 read
|
||||
//! ([`snapshot`](smarm::snapshot) / [`tree`](smarm::tree)) marshalled across a
|
||||
//! `call`, nothing more.
|
||||
|
||||
use smarm::observer::{self, ObserverReply, ObserverRequest};
|
||||
use smarm::{
|
||||
channel, register, run, spawn, ActorState, Name, RuntimeSnapshot, RuntimeTree, TreeNode,
|
||||
};
|
||||
|
||||
const ECHO: Name<u64> = Name::new("echo");
|
||||
|
||||
fn state_glyph(s: ActorState) -> &'static str {
|
||||
match s {
|
||||
ActorState::Queued => "queued",
|
||||
ActorState::Running => "running",
|
||||
ActorState::Notified => "notified",
|
||||
ActorState::Parked => "parked",
|
||||
ActorState::Done => "done",
|
||||
}
|
||||
}
|
||||
|
||||
/// A `ps`-style table over the flat snapshot.
|
||||
fn print_snapshot(snap: &RuntimeSnapshot) {
|
||||
println!(
|
||||
"snapshot (format v{}, {} actors)",
|
||||
snap.format_version,
|
||||
snap.actors.len()
|
||||
);
|
||||
println!(
|
||||
" {:<10} {:<9} {:<10} {:>4} {:>4} {:>4} {:>4} {:>5} {}",
|
||||
"pid", "state", "parent", "mon", "lnk", "joi", "mbox", "msgs", "names"
|
||||
);
|
||||
for a in &snap.actors {
|
||||
let parent = if a.supervisor.index() == u32::MAX {
|
||||
"<root>".to_string()
|
||||
} else {
|
||||
format!("{}.{}", a.supervisor.index(), a.supervisor.generation())
|
||||
};
|
||||
println!(
|
||||
" {:<10} {:<9} {:<10} {:>4} {:>4} {:>4} {:>4} {:>5} {}",
|
||||
format!("{}.{}", a.pid.index(), a.pid.generation()),
|
||||
state_glyph(a.state),
|
||||
parent,
|
||||
a.monitors,
|
||||
a.links,
|
||||
a.joiners,
|
||||
a.mailbox_depth,
|
||||
a.messages_received,
|
||||
if a.names.is_empty() {
|
||||
"-".to_string()
|
||||
} else {
|
||||
a.names.join(",")
|
||||
},
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/// The parentage forest, indented.
|
||||
fn print_tree(t: &RuntimeTree) {
|
||||
println!("tree (format v{})", t.format_version);
|
||||
fn walk(node: &TreeNode, depth: usize) {
|
||||
let indent = " ".repeat(depth + 1);
|
||||
let flag = if node.orphaned { " [orphaned]" } else { "" };
|
||||
let names = if node.info.names.is_empty() {
|
||||
String::new()
|
||||
} else {
|
||||
format!(" ({})", node.info.names.join(","))
|
||||
};
|
||||
println!(
|
||||
"{indent}{}.{} {}{names}{flag}",
|
||||
node.info.pid.index(),
|
||||
node.info.pid.generation(),
|
||||
state_glyph(node.info.state),
|
||||
);
|
||||
for child in &node.children {
|
||||
walk(child, depth + 1);
|
||||
}
|
||||
}
|
||||
for root in &t.roots {
|
||||
walk(root, 0);
|
||||
}
|
||||
}
|
||||
|
||||
fn main() {
|
||||
run(|| {
|
||||
// A named echo service and two anonymous workers, all parked on a gate
|
||||
// so the system holds still while we observe it. Each gets its own gate
|
||||
// receiver (a Receiver is single-consumer); we keep the senders to
|
||||
// release them at the end.
|
||||
let (ready_tx, ready_rx) = channel::<()>();
|
||||
let mut gates = Vec::new();
|
||||
|
||||
let svc = {
|
||||
let (gate_tx, gate_rx) = channel::<()>();
|
||||
gates.push(gate_tx);
|
||||
let ready_tx = ready_tx.clone();
|
||||
spawn(move || {
|
||||
let (cmd_tx, cmd_rx) = channel::<u64>();
|
||||
register(ECHO, cmd_tx).unwrap();
|
||||
ready_tx.send(()).unwrap();
|
||||
gate_rx.recv().unwrap();
|
||||
drop(cmd_rx);
|
||||
})
|
||||
};
|
||||
let workers: Vec<_> = (0..2)
|
||||
.map(|_| {
|
||||
let (gate_tx, gate_rx) = channel::<()>();
|
||||
gates.push(gate_tx);
|
||||
let ready_tx = ready_tx.clone();
|
||||
spawn(move || {
|
||||
ready_tx.send(()).unwrap();
|
||||
gate_rx.recv().unwrap();
|
||||
})
|
||||
})
|
||||
.collect();
|
||||
|
||||
// Wait until all three have announced and parked.
|
||||
for _ in 0..3 {
|
||||
ready_rx.recv().unwrap();
|
||||
}
|
||||
// Queue two commands at the echo service so its mailbox depth is visible.
|
||||
smarm::send(ECHO, 1).unwrap();
|
||||
smarm::send(ECHO, 2).unwrap();
|
||||
|
||||
// Start the observer and dump the system through it.
|
||||
let obs = observer::start();
|
||||
|
||||
let ObserverReply::Snapshot(snap) = obs.call(ObserverRequest::Snapshot).unwrap() else {
|
||||
unreachable!()
|
||||
};
|
||||
let ObserverReply::Tree(t) = obs.call(ObserverRequest::Tree).unwrap() else {
|
||||
unreachable!()
|
||||
};
|
||||
|
||||
print_snapshot(&snap);
|
||||
println!();
|
||||
print_tree(&t);
|
||||
|
||||
// Release everyone and drain.
|
||||
for gate_tx in gates {
|
||||
gate_tx.send(()).unwrap();
|
||||
}
|
||||
svc.join().unwrap();
|
||||
for w in workers {
|
||||
w.join().unwrap();
|
||||
}
|
||||
});
|
||||
}
|
||||
@@ -1,78 +0,0 @@
|
||||
//! Addressing a single-message actor two ways: by identity and by name.
|
||||
//!
|
||||
//! A single-message actor (one implementing [`Addressable`]) is reachable
|
||||
//! through either of smarm's two address kinds:
|
||||
//!
|
||||
//! - [`Pid<A>`] — identity-bound. Names the exact incarnation, never
|
||||
//! redirects, and stops resolving once that actor dies.
|
||||
//! - [`Name<M>`] — durable. Re-resolves through the registry on every send,
|
||||
//! so it always reaches whoever currently holds the name.
|
||||
//!
|
||||
//! A name is typed by the *message* it carries; a pid by the *actor* type
|
||||
//! (whose [`Addressable::Msg`] is that message).
|
||||
|
||||
use smarm::{
|
||||
channel, lookup_as, register, run, send, send_to, spawn, spawn_addr, Addressable, Name, Pid,
|
||||
Receiver,
|
||||
};
|
||||
|
||||
/// A single-message actor. Its one message type is what a `Pid<Echo>` delivers.
|
||||
struct Echo;
|
||||
impl Addressable for Echo {
|
||||
type Msg = EchoMsg;
|
||||
}
|
||||
enum EchoMsg {
|
||||
Say(String),
|
||||
Stop,
|
||||
}
|
||||
|
||||
/// A durable, message-typed name. Declared as a constant and shared freely.
|
||||
const ECHO: Name<EchoMsg> = Name::new("echo");
|
||||
|
||||
fn main() {
|
||||
run(|| {
|
||||
// --- Identity-bound: Pid<A> ---------------------------------------
|
||||
//
|
||||
// `spawn_addr` makes the actor's inbox, hands the body its receiver,
|
||||
// installs the sender, and returns a typed `Pid<Echo>`. The inbox is
|
||||
// published before the pid is returned, so the address is live the
|
||||
// instant we hold it — an immediate `send_to` always resolves.
|
||||
let echo: Pid<Echo> = spawn_addr::<Echo>(|rx: Receiver<EchoMsg>| {
|
||||
while let Ok(msg) = rx.recv() {
|
||||
match msg {
|
||||
EchoMsg::Say(s) => println!("echo: {s}"),
|
||||
EchoMsg::Stop => break,
|
||||
}
|
||||
}
|
||||
});
|
||||
send_to(echo, EchoMsg::Say("hello".into())).unwrap();
|
||||
send_to(echo, EchoMsg::Stop).unwrap();
|
||||
|
||||
// --- Durable: Name<M> ---------------------------------------------
|
||||
//
|
||||
// An actor claims a name for its own inbox; senders resolve it on every
|
||||
// send, so the binding outlives any single holder.
|
||||
let (ready_tx, ready_rx) = channel::<()>();
|
||||
spawn(move || {
|
||||
let (tx, rx) = channel::<EchoMsg>();
|
||||
register(ECHO, tx).unwrap();
|
||||
ready_tx.send(()).unwrap();
|
||||
while let Ok(msg) = rx.recv() {
|
||||
match msg {
|
||||
EchoMsg::Say(s) => println!("named echo: {s}"),
|
||||
EchoMsg::Stop => break,
|
||||
}
|
||||
}
|
||||
});
|
||||
ready_rx.recv().unwrap(); // the actor has claimed the name
|
||||
|
||||
send(ECHO, EchoMsg::Say("by name".into())).unwrap();
|
||||
|
||||
// Recover a typed pid from the name when you want identity-bound sends:
|
||||
// `lookup_as` re-types the registry's erased pid as `Pid<Echo>`, so you
|
||||
// drop back onto the compile-checked `send_to`.
|
||||
if let Some(p) = lookup_as::<Echo>("echo") {
|
||||
send_to(p, EchoMsg::Stop).unwrap();
|
||||
}
|
||||
});
|
||||
}
|
||||
@@ -1,95 +0,0 @@
|
||||
//! A typed worker pool over process groups (`pg`).
|
||||
//!
|
||||
//! Workers enroll in a named group; the dispatcher reaches them through it.
|
||||
//! Because the pool is homogeneous — every member is a `Worker` — the group's
|
||||
//! typed reads (`pick_as` / `members_as`) and the `dispatch` combinator hand
|
||||
//! members back as `Pid<Worker>`, so every send is an ordinary compile-checked
|
||||
//! `send_to` rather than the untyped escape hatch. The untyped reads
|
||||
//! (`members` / `pick` + `send_dyn`) stay available for identity-only use —
|
||||
//! counting, logging, monitoring — where the message type isn't known.
|
||||
//!
|
||||
//! The pool drains itself: each worker retires on a sentinel and reports its
|
||||
//! tally back, and the dispatcher waits the pool out before returning. (A pool
|
||||
//! left running would be stopped anyway when the root actor exits, but draining
|
||||
//! explicitly keeps the example deterministic.)
|
||||
|
||||
use smarm::{
|
||||
channel, dispatch, join, members, members_as, pick, pick_as, run, send_dyn, send_to,
|
||||
spawn_addr, Addressable, Pid,
|
||||
};
|
||||
|
||||
/// The pool's worker actor. One message type, carried by `Pid<Worker>`.
|
||||
struct Worker;
|
||||
impl Addressable for Worker {
|
||||
type Msg = Job;
|
||||
}
|
||||
|
||||
/// A unit of work, or the sentinel that retires a worker.
|
||||
enum Job {
|
||||
Task { id: u64 },
|
||||
Retire,
|
||||
}
|
||||
|
||||
const POOL: &str = "pool";
|
||||
const WORKERS: u64 = 4;
|
||||
|
||||
fn main() {
|
||||
run(|| {
|
||||
// Mint four typed workers and enroll them. `spawn_addr` yields a
|
||||
// `Pid<Worker>`; `join` takes a typed pid directly, erasing internally.
|
||||
// Each worker reports how many tasks it handled over a shared channel,
|
||||
// so the dispatcher can wait the pool out at the end.
|
||||
let (done_tx, done_rx) = channel::<(u64, u64)>();
|
||||
for id in 0..WORKERS {
|
||||
let done_tx = done_tx.clone();
|
||||
let w: Pid<Worker> = spawn_addr::<Worker>(move |rx| {
|
||||
let mut handled = 0;
|
||||
while let Ok(job) = rx.recv() {
|
||||
match job {
|
||||
Job::Task { id: job } => {
|
||||
println!("worker {id} handling job {job}");
|
||||
handled += 1;
|
||||
}
|
||||
Job::Retire => break,
|
||||
}
|
||||
}
|
||||
done_tx.send((id, handled)).unwrap();
|
||||
});
|
||||
join(POOL, w);
|
||||
}
|
||||
drop(done_tx); // from here only the workers hold senders
|
||||
|
||||
// Pick one live member and hand it a job — typed end to end.
|
||||
if let Some(w) = pick_as::<Worker>(POOL) {
|
||||
send_to(w, Job::Task { id: 1 }).unwrap();
|
||||
}
|
||||
|
||||
// `dispatch` rolls pick-a-live-member-and-send into one call, returning
|
||||
// the member it reached (or handing the job back if the pool is empty).
|
||||
let _ = dispatch::<Worker>(POOL, Job::Task { id: 2 });
|
||||
|
||||
// Fan a job out to the whole pool — typed, so each send is `send_to`.
|
||||
for w in members_as::<Worker>(POOL) {
|
||||
let _ = send_to(w, Job::Task { id: 3 });
|
||||
}
|
||||
|
||||
// The untyped reads stay available for identity-only use. To message a
|
||||
// member reached this way, the explicit `send_dyn` escape hatch names
|
||||
// the message type.
|
||||
println!("pool size: {}", members(POOL).len());
|
||||
if let Some(any) = pick(POOL) {
|
||||
let _ = send_dyn::<Job>(any, Job::Task { id: 4 });
|
||||
}
|
||||
|
||||
// Retire every worker, then drain their tallies so the program winds
|
||||
// down on its own.
|
||||
for w in members_as::<Worker>(POOL) {
|
||||
let _ = send_to(w, Job::Retire);
|
||||
}
|
||||
for _ in 0..WORKERS {
|
||||
if let Ok((id, handled)) = done_rx.recv() {
|
||||
println!("worker {id} retired after {handled} jobs");
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
+7
-22
@@ -72,10 +72,7 @@ pub fn clear_current_pid() {
|
||||
CURRENT_PID.with(|c| c.set(None));
|
||||
}
|
||||
|
||||
/// Actor-side TLS accessor: `#[inline(never)]` + fence, see `context` docs.
|
||||
#[inline(never)]
|
||||
pub fn current_pid() -> Option<Pid> {
|
||||
crate::context::tls_fence();
|
||||
CURRENT_PID.with(|c| c.get())
|
||||
}
|
||||
|
||||
@@ -96,13 +93,11 @@ pub fn take_last_outcome() -> Option<Outcome> {
|
||||
/// unwinding to cross the boundary, but `catch_unwind` here means unwinding
|
||||
/// never actually does.
|
||||
pub extern "C-unwind" fn trampoline() {
|
||||
let b = match CURRENT_ACTOR_BOX.with(|c| c.borrow_mut().take()) {
|
||||
Some(b) => b,
|
||||
None => panic!("smarm: trampoline entered without a closure set (core corrupt)"),
|
||||
};
|
||||
let b = CURRENT_ACTOR_BOX.with(|c| c.borrow_mut().take())
|
||||
.expect("trampoline entered without a closure set");
|
||||
|
||||
let outcome = match panic::catch_unwind(panic::AssertUnwindSafe(b)) {
|
||||
Ok(()) => Outcome::Exit,
|
||||
Ok(()) => Outcome::Exit,
|
||||
Err(payload) => {
|
||||
if payload.is::<StopSentinel>() {
|
||||
Outcome::Stopped
|
||||
@@ -112,7 +107,8 @@ pub extern "C-unwind" fn trampoline() {
|
||||
}
|
||||
};
|
||||
|
||||
publish_outcome(outcome);
|
||||
LAST_OUTCOME.with(|r| *r.borrow_mut() = Some(outcome));
|
||||
ACTOR_DONE.with(|c| c.set(true));
|
||||
|
||||
// Hand control back. The scheduler will tear down our slot and never
|
||||
// resume us again.
|
||||
@@ -121,25 +117,14 @@ pub extern "C-unwind" fn trampoline() {
|
||||
unreachable!("scheduler resumed a done actor");
|
||||
}
|
||||
|
||||
/// Record the outcome for the scheduler that is about to be switched to. Kept
|
||||
/// out of line: the actor may have migrated threads while its closure ran, so
|
||||
/// the TLS base `trampoline` computed at entry must not be reused here (see
|
||||
/// `context` module docs on thread-locals and migration).
|
||||
#[inline(never)]
|
||||
fn publish_outcome(outcome: Outcome) {
|
||||
crate::context::tls_fence();
|
||||
LAST_OUTCOME.with(|r| *r.borrow_mut() = Some(outcome));
|
||||
ACTOR_DONE.with(|c| c.set(true));
|
||||
}
|
||||
|
||||
/// One actor's worth of state. Owned by the scheduler's slot table.
|
||||
pub struct Actor {
|
||||
/// The PID this actor was assigned at spawn time.
|
||||
pub pid: Pid,
|
||||
/// The stack the actor runs on. Dropped (munmap'd) when the actor dies.
|
||||
/// (The saved stack pointer lives on the `Slot` as an atomic, not here:
|
||||
/// it is hot scheduling state, read/written without the cold lock.)
|
||||
pub stack: Stack,
|
||||
/// The saved stack pointer. Updated on every yield.
|
||||
pub sp: usize,
|
||||
/// The PID of this actor's supervisor. Used to deliver `Signal` on death.
|
||||
pub supervisor: Pid,
|
||||
/// Cooperative-cancellation flag. `request_stop` sets it (and unparks a
|
||||
|
||||
@@ -0,0 +1,165 @@
|
||||
//! Cooperative context switching and cycle counter, aarch64 (AAPCS64).
|
||||
//!
|
||||
//! The aarch64 mirror of the x86-64 backend. The protocol is identical — save
|
||||
//! callee-saved state, swap stack pointers via the shared `*_SP` thread-locals,
|
||||
//! restore, return — but the mechanics differ from x86 in two ways worth
|
||||
//! stating up front, because they drive the stack layout:
|
||||
//!
|
||||
//! 1. **Return is via the link register, not the stack.** x86 `ret` pops the
|
||||
//! return address off the stack; aarch64 `ret` jumps to whatever is in
|
||||
//! `x30` (lr). So the entry point is parked in the saved-`x30` slot, and
|
||||
//! the shim restores `x30` and `ret`s to it. There is no "ret target word"
|
||||
//! sitting on the stack the way there is on x86.
|
||||
//!
|
||||
//! 2. **sp must be 16-byte aligned at all times**, not just at call entry.
|
||||
//! We save 12 registers (x19–x28, x29, x30) = 96 bytes, already a multiple
|
||||
//! of 16, so the frame is aligned by construction.
|
||||
//!
|
||||
//! FP/SIMD registers (v8–v15, whose low 64 bits are callee-saved under AAPCS64)
|
||||
//! are NOT saved, for the same reason the x86 backend skips XMM: every yield
|
||||
//! goes through a Rust call boundary, so the compiler has already spilled any
|
||||
//! live vector state. If we ever yield from a non-call-boundary, this breaks.
|
||||
|
||||
use super::{get_actor_sp, get_scheduler_sp, set_actor_sp, set_scheduler_sp};
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Initial stack layout
|
||||
//
|
||||
// Twelve 8-byte slots, all zero except the x30 slot which holds `entry`. The
|
||||
// first `switch_to_actor` loads x19–x28/x29/x30 back and `ret`s to x30,
|
||||
// landing at the top of `entry` with sp 16-aligned (the AAPCS64 requirement).
|
||||
//
|
||||
// Layout (high → low), relative to aligned_top = top & ~15:
|
||||
//
|
||||
// aligned_top - 8 : x30 = entry ← `ret` jumps here.
|
||||
// aligned_top - 16 : x29 = 0 (fp)
|
||||
// aligned_top - 24 : x28 = 0
|
||||
// aligned_top - 32 : x27 = 0
|
||||
// aligned_top - 40 : x26 = 0
|
||||
// aligned_top - 48 : x25 = 0
|
||||
// aligned_top - 56 : x24 = 0
|
||||
// aligned_top - 64 : x23 = 0
|
||||
// aligned_top - 72 : x22 = 0
|
||||
// aligned_top - 80 : x21 = 0
|
||||
// aligned_top - 88 : x20 = 0
|
||||
// aligned_top - 96 : x19 = 0 ← initial sp (16-aligned)
|
||||
//
|
||||
// The restore order in the shim (ldp pairs, low→high) must match this exactly.
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
pub fn init_actor_stack(top: *mut u8, entry: extern "C-unwind" fn()) -> usize {
|
||||
unsafe {
|
||||
let aligned_top = top as usize & !15;
|
||||
let mut sp = aligned_top;
|
||||
sp -= 8; (sp as *mut usize).write(entry as usize); // x30 (lr) → entry
|
||||
sp -= 8; (sp as *mut usize).write(0); // x29 (fp)
|
||||
sp -= 8; (sp as *mut usize).write(0); // x28
|
||||
sp -= 8; (sp as *mut usize).write(0); // x27
|
||||
sp -= 8; (sp as *mut usize).write(0); // x26
|
||||
sp -= 8; (sp as *mut usize).write(0); // x25
|
||||
sp -= 8; (sp as *mut usize).write(0); // x24
|
||||
sp -= 8; (sp as *mut usize).write(0); // x23
|
||||
sp -= 8; (sp as *mut usize).write(0); // x22
|
||||
sp -= 8; (sp as *mut usize).write(0); // x21
|
||||
sp -= 8; (sp as *mut usize).write(0); // x20
|
||||
sp -= 8; (sp as *mut usize).write(0); // x19 ← initial sp
|
||||
sp
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Context switch shims
|
||||
//
|
||||
// Each shim:
|
||||
// 1. Pushes x19–x28, x29, x30 as six 16-byte stp pairs (sp pre-decrement).
|
||||
// 2. Moves sp into x0 and calls the Rust helper that stores it.
|
||||
// 3. Calls the Rust helper that returns the *other* side's saved sp.
|
||||
// 4. Moves that into sp.
|
||||
// 5. Restores the twelve registers and rets (to the restored x30).
|
||||
//
|
||||
// The helper calls clobber x30 themselves, but that's fine: the outgoing x30
|
||||
// was already saved to the stack in step 1, and step 5 restores the incoming
|
||||
// side's x30 before `ret`. The push/pop register order is paired so that the
|
||||
// `ldp` sequence reverses the `stp` sequence exactly.
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
#[unsafe(naked)]
|
||||
unsafe extern "C" fn switch_to_actor_asm() {
|
||||
core::arch::naked_asm!(
|
||||
"stp x19, x20, [sp, #-96]!",
|
||||
"stp x21, x22, [sp, #16]",
|
||||
"stp x23, x24, [sp, #32]",
|
||||
"stp x25, x26, [sp, #48]",
|
||||
"stp x27, x28, [sp, #64]",
|
||||
"stp x29, x30, [sp, #80]",
|
||||
"mov x0, sp",
|
||||
"bl {set_sched_sp}",
|
||||
"bl {get_actor_sp}",
|
||||
"mov sp, x0",
|
||||
"ldp x21, x22, [sp, #16]",
|
||||
"ldp x23, x24, [sp, #32]",
|
||||
"ldp x25, x26, [sp, #48]",
|
||||
"ldp x27, x28, [sp, #64]",
|
||||
"ldp x29, x30, [sp, #80]",
|
||||
"ldp x19, x20, [sp], #96",
|
||||
"ret",
|
||||
set_sched_sp = sym set_scheduler_sp,
|
||||
get_actor_sp = sym get_actor_sp,
|
||||
);
|
||||
}
|
||||
|
||||
/// Resume the actor whose sp is in `ACTOR_SP`. Returns when the actor yields.
|
||||
pub unsafe fn switch_to_actor() {
|
||||
unsafe { switch_to_actor_asm() };
|
||||
}
|
||||
|
||||
#[unsafe(naked)]
|
||||
pub unsafe extern "C" fn switch_to_scheduler() {
|
||||
core::arch::naked_asm!(
|
||||
"stp x19, x20, [sp, #-96]!",
|
||||
"stp x21, x22, [sp, #16]",
|
||||
"stp x23, x24, [sp, #32]",
|
||||
"stp x25, x26, [sp, #48]",
|
||||
"stp x27, x28, [sp, #64]",
|
||||
"stp x29, x30, [sp, #80]",
|
||||
"mov x0, sp",
|
||||
"bl {set_actor_sp}",
|
||||
"bl {get_sched_sp}",
|
||||
"mov sp, x0",
|
||||
"ldp x21, x22, [sp, #16]",
|
||||
"ldp x23, x24, [sp, #32]",
|
||||
"ldp x25, x26, [sp, #48]",
|
||||
"ldp x27, x28, [sp, #64]",
|
||||
"ldp x29, x30, [sp, #80]",
|
||||
"ldp x19, x20, [sp], #96",
|
||||
"ret",
|
||||
set_actor_sp = sym set_actor_sp,
|
||||
get_sched_sp = sym get_scheduler_sp,
|
||||
);
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Cycle counter
|
||||
//
|
||||
// CNTVCT_EL0 is the virtual count register, readable from EL0 on Linux. `isb`
|
||||
// serialises so we don't sample the counter before prior instructions retire
|
||||
// (the aarch64 analogue of x86's `lfence` before rdtsc).
|
||||
//
|
||||
// Unit: CNTVCT ticks, whose frequency is CNTFRQ_EL0 — typically 1–50 MHz,
|
||||
// NOT the CPU clock. This is a different and much coarser unit than the x86
|
||||
// TSC, which is why `TIMESLICE_CYCLES` must be tuned separately for aarch64.
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
#[inline(always)]
|
||||
pub fn read_cycle_counter() -> u64 {
|
||||
let cnt: u64;
|
||||
unsafe {
|
||||
core::arch::asm!(
|
||||
"isb",
|
||||
"mrs {cnt}, cntvct_el0",
|
||||
cnt = out(reg) cnt,
|
||||
options(nostack, nomem),
|
||||
);
|
||||
}
|
||||
cnt
|
||||
}
|
||||
@@ -0,0 +1,54 @@
|
||||
//! Architecture abstraction layer.
|
||||
//!
|
||||
//! Everything that depends on the target ISA lives behind this module: the
|
||||
//! cooperative context-switch shims, the initial-stack layout, and the
|
||||
//! cycle counter used by the preemption timeslice. The rest of the runtime
|
||||
//! talks only to the API re-exported here and never names a register.
|
||||
//!
|
||||
//! Each backend (`x86_64`, `aarch64`) exposes the same four items:
|
||||
//!
|
||||
//! - `init_actor_stack(top, entry) -> usize`
|
||||
//! Build a fresh actor stack so the first `switch_to_actor` lands
|
||||
//! inside `entry` with the ABI-correct stack alignment.
|
||||
//! - `switch_to_actor()`
|
||||
//! Resume the actor whose sp is in `ACTOR_SP`; returns when it yields.
|
||||
//! - `switch_to_scheduler()`
|
||||
//! Yield from the current actor back to its scheduler thread.
|
||||
//! - `read_cycle_counter() -> u64`
|
||||
//! Monotonic per-core cycle counter for the timeslice clock. The unit
|
||||
//! is ISA-defined (x86 TSC ticks vs. aarch64 virtual-counter ticks),
|
||||
//! so `TIMESLICE_CYCLES` must be tuned per-arch — see `preempt`.
|
||||
//!
|
||||
//! The `SCHEDULER_SP` / `ACTOR_SP` thread-locals are shared across backends
|
||||
//! and live here, since the saved-stack-pointer protocol is identical
|
||||
//! regardless of which registers the shim happens to push.
|
||||
|
||||
use std::cell::Cell;
|
||||
|
||||
thread_local! {
|
||||
static SCHEDULER_SP: Cell<usize> = const { Cell::new(0) };
|
||||
static ACTOR_SP: Cell<usize> = const { Cell::new(0) };
|
||||
}
|
||||
|
||||
// Used by the naked shims (via `sym`) and by the scheduler/tests through the
|
||||
// re-exports below. `pub` because the asm references them as symbols.
|
||||
pub(crate) fn get_scheduler_sp() -> usize { SCHEDULER_SP.with(|c| c.get()) }
|
||||
pub(crate) fn set_scheduler_sp(v: usize) { SCHEDULER_SP.with(|c| c.set(v)) }
|
||||
pub fn get_actor_sp() -> usize { ACTOR_SP.with(|c| c.get()) }
|
||||
pub fn set_actor_sp(v: usize) { ACTOR_SP.with(|c| c.set(v)) }
|
||||
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
mod x86_64;
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
pub use x86_64::{init_actor_stack, read_cycle_counter, switch_to_actor, switch_to_scheduler};
|
||||
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
mod aarch64;
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
pub use aarch64::{init_actor_stack, read_cycle_counter, switch_to_actor, switch_to_scheduler};
|
||||
|
||||
#[cfg(not(any(target_arch = "x86_64", target_arch = "aarch64")))]
|
||||
compile_error!(
|
||||
"smarm's context-switch and cycle-counter layer supports only x86_64 and \
|
||||
aarch64. Add a backend in src/arch/ for this target."
|
||||
);
|
||||
@@ -0,0 +1,118 @@
|
||||
//! Cooperative context switching and cycle counter, x86-64 (SysV AMD64).
|
||||
//!
|
||||
//! Two naked-asm functions move execution between a scheduler thread and an
|
||||
//! actor running on its own mmap'd stack. The compiler cannot do this; the
|
||||
//! whole point of `#[unsafe(naked)]` is that we control every instruction.
|
||||
//!
|
||||
//! `init_actor_stack` builds the initial stack so that the first
|
||||
//! `switch_to_actor` lands inside the entry function with `rsp % 16 == 8`
|
||||
//! (the x86-64 ABI requirement at function entry).
|
||||
|
||||
use super::{get_actor_sp, get_scheduler_sp, set_actor_sp, set_scheduler_sp};
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Initial stack layout
|
||||
//
|
||||
// We start from aligned_top = top & ~15, then bias by 8 and push seven 8-byte
|
||||
// slots (downward). The first `switch_to_actor` pops r15..rbx and `ret`s —
|
||||
// landing in `entry` with rsp % 16 == 8 (the x86-64 ABI state at the point
|
||||
// just after a `call`, which is what `entry` is compiled to expect).
|
||||
//
|
||||
// Layout (high → low), relative to aligned_top = top & ~15. Note the entry
|
||||
// slot is at aligned_top - 16, NOT aligned_top - 8: the function does
|
||||
// `(top & ~15) - 8` and *then* a `-= 8` before the first write, so the first
|
||||
// stored word lands at aligned_top - 16. Verified by single-stepping the
|
||||
// `ret` under llmdbg: entry sits at an address with %16 == 0, so the post-ret
|
||||
// rsp is %16 == 8.
|
||||
//
|
||||
// aligned_top - 8 : (unused padding; keeps entry's slot %16 == 0)
|
||||
// aligned_top - 16 : entry ptr ← `ret` target. Post-ret: rsp % 16 == 8.
|
||||
// aligned_top - 24 : rbx = 0
|
||||
// aligned_top - 32 : rbp = 0
|
||||
// aligned_top - 40 : r12 = 0
|
||||
// aligned_top - 48 : r13 = 0
|
||||
// aligned_top - 56 : r14 = 0
|
||||
// aligned_top - 64 : r15 = 0 ← initial rsp
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
pub fn init_actor_stack(top: *mut u8, entry: extern "C-unwind" fn()) -> usize {
|
||||
unsafe {
|
||||
let mut sp = (top as usize & !15) - 8;
|
||||
sp -= 8; (sp as *mut usize).write(entry as usize); // ret target
|
||||
sp -= 8; (sp as *mut usize).write(0); // rbx
|
||||
sp -= 8; (sp as *mut usize).write(0); // rbp
|
||||
sp -= 8; (sp as *mut usize).write(0); // r12
|
||||
sp -= 8; (sp as *mut usize).write(0); // r13
|
||||
sp -= 8; (sp as *mut usize).write(0); // r14
|
||||
sp -= 8; (sp as *mut usize).write(0); // r15
|
||||
sp
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Context switch shims
|
||||
//
|
||||
// Each shim:
|
||||
// 1. Pushes the six callee-saved integer registers.
|
||||
// 2. Snaps rsp into rdi and calls the Rust helper that stores it.
|
||||
// 3. Calls the Rust helper that returns the *other* side's saved rsp.
|
||||
// 4. Moves that into rsp.
|
||||
// 5. Pops the six registers and rets.
|
||||
//
|
||||
// XMM registers are NOT saved here. We rely on every yield happening through
|
||||
// a Rust call site, which means the compiler has spilled any live XMM state
|
||||
// to the stack before we get here. (This is the same argument the compiler
|
||||
// uses internally — callee-saved regs are what survive a `call`, and the
|
||||
// SysV AMD64 ABI says XMM0–15 are all caller-saved.) If we ever yield from
|
||||
// a place that isn't a Rust call boundary, this assumption breaks.
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
#[unsafe(naked)]
|
||||
unsafe extern "C" fn switch_to_actor_asm() {
|
||||
core::arch::naked_asm!(
|
||||
"push rbx", "push rbp", "push r12", "push r13", "push r14", "push r15",
|
||||
"mov rdi, rsp",
|
||||
"call {set_sched_sp}",
|
||||
"call {get_actor_sp}",
|
||||
"mov rsp, rax",
|
||||
"pop r15", "pop r14", "pop r13", "pop r12", "pop rbp", "pop rbx",
|
||||
"ret",
|
||||
set_sched_sp = sym set_scheduler_sp,
|
||||
get_actor_sp = sym get_actor_sp,
|
||||
);
|
||||
}
|
||||
|
||||
/// Resume the actor whose sp is in `ACTOR_SP`. Returns when the actor yields.
|
||||
pub unsafe fn switch_to_actor() {
|
||||
unsafe { switch_to_actor_asm() };
|
||||
}
|
||||
|
||||
#[unsafe(naked)]
|
||||
pub unsafe extern "C" fn switch_to_scheduler() {
|
||||
core::arch::naked_asm!(
|
||||
"push rbx", "push rbp", "push r12", "push r13", "push r14", "push r15",
|
||||
"mov rdi, rsp",
|
||||
"call {set_actor_sp}",
|
||||
"call {get_sched_sp}",
|
||||
"mov rsp, rax",
|
||||
"pop r15", "pop r14", "pop r13", "pop r12", "pop rbp", "pop rbx",
|
||||
"ret",
|
||||
set_actor_sp = sym set_actor_sp,
|
||||
get_sched_sp = sym get_scheduler_sp,
|
||||
);
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Cycle counter
|
||||
//
|
||||
// `lfence` serialises the instruction stream so we don't measure time before
|
||||
// prior instructions retire. Unit: TSC ticks (≈ CPU base clock).
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
#[inline(always)]
|
||||
pub fn read_cycle_counter() -> u64 {
|
||||
unsafe {
|
||||
core::arch::asm!("lfence", options(nostack, nomem, preserves_flags));
|
||||
core::arch::x86_64::_rdtsc()
|
||||
}
|
||||
}
|
||||
-977
@@ -1,977 +0,0 @@
|
||||
//! Native causal profiling (RFC 007). Enabled by `--features smarm-causal`;
|
||||
//! zero cost without it (same discipline as `smarm-trace`).
|
||||
//!
|
||||
//! The Coz algorithm, transposed onto actors: to estimate what speeding up
|
||||
//! code site S by p% would do to throughput, we instead *slow everything
|
||||
//! else down* by p% of the time spent in S, and watch the progress-point
|
||||
//! rates respond. Where Coz must inject real `usleep`s into OS threads from
|
||||
//! the outside, smarm owns every clock that matters:
|
||||
//!
|
||||
//! - Sampling and delay injection happen at `maybe_preempt`'s amortised
|
||||
//! cadence — an existing, safe hook (never inside a prep-to-park region).
|
||||
//! - Injected delay is subtracted from the actor's timeslice
|
||||
//! (`preempt::extend_timeslice`), so experiments don't perturb scheduling.
|
||||
//! - Delay bookkeeping is *actor*-granular: each `Slot` carries an absorbed-
|
||||
//! delay ledger, compared against a global ledger. Parked actors absorb
|
||||
//! accrued delay for free on resume (Coz's blocked-thread rule) — waiting
|
||||
//! is never penalised.
|
||||
//!
|
||||
//! v1 scope (per RFC discussion): explicit scoped sites (`causal_site!`)
|
||||
//! rather than PC sampling (jar Q1 stays open); throughput progress points
|
||||
//! only; timer-heap deadlines are *not* shifted (documented gap — long
|
||||
//! experiments can make real-time timeouts fire early in virtual terms);
|
||||
//! multi-scheduler coherence is best-effort via global atomics.
|
||||
//!
|
||||
//! Usage:
|
||||
//! ```ignore
|
||||
//! let _g = smarm::causal_site!("inventory-reserve"); // in suspect code
|
||||
//! smarm::progress!("orders-processed"); // per unit of work
|
||||
//! let results = smarm::causal::run_experiments(&Default::default());
|
||||
//! print!("{}", smarm::causal::render_summary(&results));
|
||||
//! std::fs::write("profile.coz", smarm::causal::render_coz(&results))?;
|
||||
//! ```
|
||||
|
||||
#[cfg(feature = "smarm-causal")]
|
||||
mod inner {
|
||||
use crate::preempt;
|
||||
use std::cell::Cell;
|
||||
use std::sync::atomic::{AtomicU64, Ordering};
|
||||
use std::sync::{Mutex, OnceLock};
|
||||
use std::time::{Duration, Instant};
|
||||
|
||||
// -----------------------------------------------------------------------
|
||||
// Global state
|
||||
// -----------------------------------------------------------------------
|
||||
|
||||
/// Active experiment, packed `(site_id << 32) | speedup_pct`. 0 = idle.
|
||||
/// A single word so the hot path reads one atomic; experiments are global
|
||||
/// across scheduler threads (jar Q7, v1: plain Relaxed atomics).
|
||||
static EXPERIMENT: AtomicU64 = AtomicU64::new(0);
|
||||
|
||||
/// Monotone experiment-window counter, bumped by every `begin()`. Lets
|
||||
/// the offcpu-gap stash (RFC 007) tell apart two windows with an
|
||||
/// identical site+pct word — live in the attrib probe's 50,50 schedule
|
||||
/// — so a gap straddling `end()`/`begin()` never counts a cooldown.
|
||||
static EXPERIMENT_EPOCH: AtomicU64 = AtomicU64::new(0);
|
||||
|
||||
/// Global virtual-delay ledger, in TSC cycles: the total delay every
|
||||
/// actor *should* have experienced since startup. Grows while a sample
|
||||
/// lands in the experiment's target site; each actor's `Slot` ledger
|
||||
/// chases it by spin-absorbing at preemption checks.
|
||||
static GLOBAL_DELAY: AtomicU64 = AtomicU64::new(0);
|
||||
|
||||
/// Registered site names; site id = index + 1 (0 = "no site").
|
||||
static SITES: OnceLock<Mutex<Vec<&'static str>>> = OnceLock::new();
|
||||
|
||||
/// Registered progress points (leaked for `'static`, like trace's drain
|
||||
/// state — the set is small and lives for the process).
|
||||
static PROGRESS: OnceLock<Mutex<Vec<&'static ProgressPoint>>> = OnceLock::new();
|
||||
|
||||
thread_local! {
|
||||
/// TSC at this thread's previous causal check, the sample "period"
|
||||
/// denominator. Re-armed on every actor resume so scheduler time and
|
||||
/// a previous actor's tail never count toward a sample. 0 = unarmed.
|
||||
static LAST_SAMPLE_TSC: Cell<u64> = const { Cell::new(0) };
|
||||
}
|
||||
|
||||
/// Guard against TSC weirdness (migration between unsynced sockets,
|
||||
/// virtualisation steps): a single sample interval larger than this is
|
||||
/// discarded rather than believed. ~33ms at 3 GHz — far beyond any real
|
||||
/// gap between preemption checks inside a slice.
|
||||
const MAX_SAMPLE_CYCLES: u64 = 100_000_000;
|
||||
|
||||
/// Cap on delay spun in one visit, so one check can never wedge an actor
|
||||
/// for a human-visible pause; the remainder is absorbed on later visits.
|
||||
/// ~3ms at 3 GHz.
|
||||
const MAX_SPIN_PER_VISIT: u64 = 10_000_000;
|
||||
|
||||
// -----------------------------------------------------------------------
|
||||
// Ledger audit (RFC 007 deficit hunt): where injected delay is born,
|
||||
// paid, and forgiven — and where would-be attribution is silently lost
|
||||
// (deschedule tails, clamp discards). Monotone Relaxed totals, read via
|
||||
// `ledger_counters()`; `run_experiments` windows them into
|
||||
// `ExperimentResult`. Measure-only: nothing here changes injection or
|
||||
// absorption behaviour.
|
||||
// -----------------------------------------------------------------------
|
||||
|
||||
/// Cycles bystanders actually spun to pay down the global ledger.
|
||||
static SPIN_ABSORBED_CYCLES: AtomicU64 = AtomicU64::new(0);
|
||||
/// Cycles waived at wake after a real park (the blocked-thread rule).
|
||||
static PARK_FORGIVEN_CYCLES: AtomicU64 = AtomicU64::new(0);
|
||||
/// Would-be attribution lost when the target-site actor parks mid-site.
|
||||
static DROP_PARK_CYCLES: AtomicU64 = AtomicU64::new(0);
|
||||
static DROP_PARK_N: AtomicU64 = AtomicU64::new(0);
|
||||
/// Same loss at yields (explicit, slice-expiry, or a park that requeued).
|
||||
static DROP_YIELD_CYCLES: AtomicU64 = AtomicU64::new(0);
|
||||
static DROP_YIELD_N: AtomicU64 = AtomicU64::new(0);
|
||||
/// Samples discarded by the TSC-weirdness clamp, in would-be delta terms.
|
||||
static DISCARD_OVERMAX_CYCLES: AtomicU64 = AtomicU64::new(0);
|
||||
static DISCARD_OVERMAX_N: AtomicU64 = AtomicU64::new(0);
|
||||
/// In-site samples dropped because the thread's clock was unarmed.
|
||||
static DISCARD_UNARMED_N: AtomicU64 = AtomicU64::new(0);
|
||||
/// Would-be attribution over runnable off-CPU gaps inside the target
|
||||
/// site (yield-descheduled -> resumed within the same window). Not a
|
||||
/// loss: on-CPU-only attribution is the Coz model — queue-wait is not
|
||||
/// shrunk by speeding the site's code — but counted so the audit books
|
||||
/// close against wall in-site time (the located @50 "deficit").
|
||||
static OFFCPU_IN_SITE_CYCLES: AtomicU64 = AtomicU64::new(0);
|
||||
static OFFCPU_IN_SITE_N: AtomicU64 = AtomicU64::new(0);
|
||||
|
||||
fn sites() -> &'static Mutex<Vec<&'static str>> {
|
||||
SITES.get_or_init(|| Mutex::new(Vec::new()))
|
||||
}
|
||||
|
||||
fn progress_points() -> &'static Mutex<Vec<&'static ProgressPoint>> {
|
||||
PROGRESS.get_or_init(|| Mutex::new(Vec::new()))
|
||||
}
|
||||
|
||||
/// Recover from lock poisoning: all these registries hold plain data that
|
||||
/// is valid at every instruction boundary, so a panicked registrant can't
|
||||
/// leave them torn.
|
||||
fn lock_unpoisoned<T>(m: &Mutex<T>) -> std::sync::MutexGuard<'_, T> {
|
||||
match m.lock() {
|
||||
Ok(g) => g,
|
||||
Err(poisoned) => poisoned.into_inner(),
|
||||
}
|
||||
}
|
||||
|
||||
// -----------------------------------------------------------------------
|
||||
// Sites
|
||||
// -----------------------------------------------------------------------
|
||||
|
||||
/// Register (or look up) a causal site by name; returns its nonzero id.
|
||||
/// Called once per `causal_site!` expansion via a `OnceLock`, so the
|
||||
/// mutex is off every hot path.
|
||||
pub fn site_id(name: &'static str) -> u32 {
|
||||
let mut v = lock_unpoisoned(sites());
|
||||
if let Some(pos) = v.iter().position(|n| *n == name) {
|
||||
return (pos + 1) as u32;
|
||||
}
|
||||
v.push(name);
|
||||
v.len() as u32
|
||||
}
|
||||
|
||||
fn site_name(id: u32) -> Option<String> {
|
||||
if id == 0 {
|
||||
return None;
|
||||
}
|
||||
let v = lock_unpoisoned(sites());
|
||||
v.get((id - 1) as usize).map(|s| (*s).to_string())
|
||||
}
|
||||
|
||||
/// RAII marker: while alive, the current *actor* (not thread — the id
|
||||
/// lives in its `Slot` and survives preemption/migration) is "inside"
|
||||
/// the site. Nesting restores the outer site on drop. Inert outside an
|
||||
/// actor (scheduler/OS-thread stacks).
|
||||
pub struct SiteGuard {
|
||||
/// Slot of the actor that entered, null if entered outside an actor.
|
||||
/// Valid for the guard's whole life: the guard lives on the actor's
|
||||
/// stack, and a slot is never reclaimed while its actor is alive —
|
||||
/// the same argument as `preempt::check_cancelled`.
|
||||
slot: *const crate::runtime::Slot,
|
||||
prev: u32,
|
||||
}
|
||||
|
||||
impl SiteGuard {
|
||||
/// Enter `site` for the on-CPU actor.
|
||||
pub fn enter(site: u32) -> Self {
|
||||
let slot = preempt::current_slot_ptr();
|
||||
if slot.is_null() {
|
||||
return SiteGuard { slot, prev: 0 };
|
||||
}
|
||||
// SAFETY: non-null ⇒ points at the on-CPU actor's slot; see the
|
||||
// field docs for the lifetime argument.
|
||||
let prev = unsafe { (*slot).causal_site() };
|
||||
unsafe { (*slot).set_causal_site(site) };
|
||||
site_transition(slot, prev, site);
|
||||
SiteGuard { slot, prev }
|
||||
}
|
||||
}
|
||||
|
||||
impl Drop for SiteGuard {
|
||||
fn drop(&mut self) {
|
||||
if !self.slot.is_null() {
|
||||
// SAFETY (both): as in `enter` — the actor (and thus its
|
||||
// slot) is alive for as long as this guard is on its stack.
|
||||
let site = unsafe { (*self.slot).causal_site() };
|
||||
unsafe { (*self.slot).set_causal_site(self.prev) };
|
||||
site_transition(self.slot, site, self.prev);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Name of the site the on-CPU actor is currently inside, if any.
|
||||
/// (Introspection/testing; not a hot path.)
|
||||
pub fn current_site_name() -> Option<String> {
|
||||
let slot = preempt::current_slot_ptr();
|
||||
if slot.is_null() {
|
||||
return None;
|
||||
}
|
||||
// SAFETY: on-CPU actor's slot, valid for the whole resume.
|
||||
site_name(unsafe { (*slot).causal_site() })
|
||||
}
|
||||
|
||||
// -----------------------------------------------------------------------
|
||||
// Progress points
|
||||
// -----------------------------------------------------------------------
|
||||
|
||||
/// A named throughput counter. One per distinct name; `progress!` call
|
||||
/// sites sharing a name share the counter.
|
||||
pub struct ProgressPoint {
|
||||
name: &'static str,
|
||||
count: AtomicU64,
|
||||
}
|
||||
|
||||
impl ProgressPoint {
|
||||
/// The hot path: one Relaxed RMW. (Contended across actors by design
|
||||
/// — a progress point is a global rate meter.)
|
||||
#[inline]
|
||||
pub fn bump(&self) {
|
||||
self.count.fetch_add(1, Ordering::Relaxed);
|
||||
}
|
||||
}
|
||||
|
||||
/// Register (or look up) a progress point. Called once per `progress!`
|
||||
/// expansion via a `OnceLock`; the mutex is off the hot path.
|
||||
pub fn register_progress(name: &'static str) -> &'static ProgressPoint {
|
||||
let mut v = lock_unpoisoned(progress_points());
|
||||
if let Some(p) = v.iter().find(|p| p.name == name) {
|
||||
return p;
|
||||
}
|
||||
let p: &'static ProgressPoint = Box::leak(Box::new(ProgressPoint {
|
||||
name,
|
||||
count: AtomicU64::new(0),
|
||||
}));
|
||||
v.push(p);
|
||||
p
|
||||
}
|
||||
|
||||
/// Snapshot of all progress points as `(name, count)`.
|
||||
pub fn progress_snapshot() -> Vec<(String, u64)> {
|
||||
lock_unpoisoned(progress_points())
|
||||
.iter()
|
||||
.map(|p| (p.name.to_string(), p.count.load(Ordering::Relaxed)))
|
||||
.collect()
|
||||
}
|
||||
|
||||
// -----------------------------------------------------------------------
|
||||
// The hot hook: sample + absorb
|
||||
// -----------------------------------------------------------------------
|
||||
|
||||
/// Called from `maybe_preempt` at the amortised timeslice-check cadence,
|
||||
/// under the `PREEMPTION_ENABLED` gate (so never in a prep-to-park or
|
||||
/// no-preempt region — spinning here is as safe as yielding is).
|
||||
///
|
||||
/// One Relaxed load and out when no experiment is running.
|
||||
#[inline]
|
||||
pub(crate) fn check() {
|
||||
let exp = EXPERIMENT.load(Ordering::Relaxed);
|
||||
if exp == 0 {
|
||||
return;
|
||||
}
|
||||
cold_check(exp);
|
||||
}
|
||||
|
||||
/// The experiment-active path, kept out of the inlined fast path.
|
||||
#[cold]
|
||||
#[inline(never)]
|
||||
fn cold_check(exp: u64) {
|
||||
crate::context::tls_fence();
|
||||
let slot = preempt::current_slot_ptr();
|
||||
if slot.is_null() {
|
||||
return;
|
||||
}
|
||||
// Serialised: `now` closes an interval attributed to a code site; a
|
||||
// speculative early read would drop that site's tail (RFC 007).
|
||||
let now = preempt::rdtsc_serialising();
|
||||
let last = LAST_SAMPLE_TSC.with(|c| c.replace(now));
|
||||
let target_site = (exp >> 32) as u32;
|
||||
let pct = exp & 0xffff_ffff;
|
||||
|
||||
// SAFETY (both derefs below): non-null ⇒ the on-CPU actor's slot,
|
||||
// never reclaimed while the actor runs — see `check_cancelled`.
|
||||
let my_site = unsafe { (*slot).causal_site() };
|
||||
|
||||
if my_site == target_site && pct > 0 {
|
||||
// A sample landed in the target site: everyone else must fall
|
||||
// behind by pct% of the sampled interval. Grow the global ledger
|
||||
// and credit ourselves the same amount — the credited gap *is*
|
||||
// the virtual speedup.
|
||||
if last == 0 {
|
||||
// Unarmed clock: no interval to attribute — count the loss.
|
||||
DISCARD_UNARMED_N.fetch_add(1, Ordering::Relaxed);
|
||||
return;
|
||||
}
|
||||
// SAFETY: `slot` is the on-CPU actor's slot (checked non-null
|
||||
// above); see `check_cancelled` for the lifetime argument.
|
||||
unsafe { attribute(slot, now.saturating_sub(last), pct) };
|
||||
} else {
|
||||
// Not the winner: chase the global ledger by spinning off the
|
||||
// difference, then push the slice start forward so injected
|
||||
// delay never counts as compute (the clock correction that Coz
|
||||
// cannot do from outside).
|
||||
let global = GLOBAL_DELAY.load(Ordering::Relaxed);
|
||||
let mine = unsafe { (*slot).causal_delay() };
|
||||
if mine >= global {
|
||||
return;
|
||||
}
|
||||
let spin = (global - mine).min(MAX_SPIN_PER_VISIT);
|
||||
let start = preempt::rdtsc();
|
||||
while preempt::rdtsc().saturating_sub(start) < spin {
|
||||
core::hint::spin_loop();
|
||||
}
|
||||
SPIN_ABSORBED_CYCLES.fetch_add(spin, Ordering::Relaxed);
|
||||
unsafe { (*slot).set_causal_delay(mine.wrapping_add(spin)) };
|
||||
preempt::extend_timeslice(spin);
|
||||
// The spin is not part of the next sample interval either.
|
||||
LAST_SAMPLE_TSC.with(|c| c.set(preempt::rdtsc()));
|
||||
}
|
||||
}
|
||||
|
||||
/// Attribute one target-site sample of `interval` cycles at `pct`%:
|
||||
/// grow the global ledger and credit the sampling actor's own ledger by
|
||||
/// the same amount — the credited gap *is* the virtual speedup. Shared
|
||||
/// by the cold check and the guard-boundary flush. Applies the same
|
||||
/// clamps as sampling always has: zero intervals and clock hiccups are
|
||||
/// discarded, not the run.
|
||||
///
|
||||
/// SAFETY: `slot` must point at the on-CPU actor's slot (the
|
||||
/// `check_cancelled` lifetime argument).
|
||||
unsafe fn attribute(slot: *const crate::runtime::Slot, interval: u64, pct: u64) {
|
||||
if interval == 0 {
|
||||
return; // now == last: nothing to attribute, nothing lost
|
||||
}
|
||||
if interval > MAX_SAMPLE_CYCLES {
|
||||
// TSC-weirdness clamp: the sample is discarded, not the run.
|
||||
// Count the loss in would-be delta terms so the audit's columns
|
||||
// compare directly against `injected_cycles`.
|
||||
DISCARD_OVERMAX_N.fetch_add(1, Ordering::Relaxed);
|
||||
DISCARD_OVERMAX_CYCLES.fetch_add(interval.saturating_mul(pct) / 100, Ordering::Relaxed);
|
||||
return;
|
||||
}
|
||||
let delta = interval.saturating_mul(pct) / 100;
|
||||
GLOBAL_DELAY.fetch_add(delta, Ordering::Relaxed);
|
||||
let mine = (*slot).causal_delay();
|
||||
(*slot).set_causal_delay(mine.wrapping_add(delta));
|
||||
}
|
||||
|
||||
/// Site-boundary hook, called by `SiteGuard` enter/drop when the
|
||||
/// actor's current site changes from `old` to `new`. Sample-only —
|
||||
/// never spins — so it is safe anywhere, including no-preempt regions
|
||||
/// where `check()` cannot run.
|
||||
///
|
||||
/// - Leaving the experiment's target site: flush the pending interval.
|
||||
/// Cold checks only sample when they happen to fire in-site, so the
|
||||
/// tail between the last check and the guard drop was otherwise
|
||||
/// discarded on every site entry — measured live at ~22-29µs/entry,
|
||||
/// ~6-7% of all target time (eff 0.93), which under-reported every
|
||||
/// impact (+83.5% where theory says +100%).
|
||||
/// - Entering the target site: re-arm the sample clock, so time spent
|
||||
/// *before* the site can never be attributed to it by the first
|
||||
/// in-site check (the symmetric over-attribution).
|
||||
#[inline(never)]
|
||||
fn site_transition(slot: *const crate::runtime::Slot, old: u32, new: u32) {
|
||||
crate::context::tls_fence();
|
||||
let exp = EXPERIMENT.load(Ordering::Relaxed);
|
||||
if exp == 0 || old == new {
|
||||
return;
|
||||
}
|
||||
let target = (exp >> 32) as u32;
|
||||
let pct = exp & 0xffff_ffff;
|
||||
if old == target && new != target {
|
||||
// Serialised: guard exit bounds the site's interval exactly.
|
||||
let now = preempt::rdtsc_serialising();
|
||||
let last = LAST_SAMPLE_TSC.with(|c| c.replace(now));
|
||||
if pct > 0 {
|
||||
if last != 0 {
|
||||
// SAFETY: forwarded from the guard, which holds the on-CPU
|
||||
// actor's slot for its whole life (see `SiteGuard::slot`).
|
||||
unsafe { attribute(slot, now.saturating_sub(last), pct) };
|
||||
} else {
|
||||
DISCARD_UNARMED_N.fetch_add(1, Ordering::Relaxed);
|
||||
}
|
||||
}
|
||||
} else if new == target && old != target {
|
||||
// Serialised: an early arm would let pre-site work leak into the
|
||||
// first in-site interval — the over-attribution this guards.
|
||||
LAST_SAMPLE_TSC.with(|c| c.set(preempt::rdtsc_serialising()));
|
||||
}
|
||||
}
|
||||
|
||||
/// Resume-path hook (scheduler thread, actor off-CPU). Two duties:
|
||||
///
|
||||
/// - If the last deschedule was a *real park*, time blocked absorbs any
|
||||
/// delay accrued meanwhile for free — Coz's blocked-thread rule, which
|
||||
/// keeps experiments from punishing actors for waiting. An actor that
|
||||
/// merely yielded (slice expiry) was runnable the whole time and keeps
|
||||
/// its debt: it must pay by spinning at its next check. Forgiving on
|
||||
/// every resume would make any yield-cadence actor delay-immune and
|
||||
/// experiments inert (found live on a 24-core run: nothing slowed).
|
||||
/// - If the deschedule was a *yield* in the live experiment's target
|
||||
/// site, count the off-CPU gap it opened into the offcpu audit bucket
|
||||
/// (RFC 007: the located @50 deficit — runnable queue-wait is wall
|
||||
/// time in-site that on-CPU attribution correctly skips). Same-window
|
||||
/// only, enforced by the experiment epoch; measure-only.
|
||||
/// - Arm this thread's sample clock so the first interval of the resume
|
||||
/// excludes scheduler time.
|
||||
#[inline]
|
||||
pub(crate) fn on_resume(slot: &crate::runtime::Slot) {
|
||||
let (desched_tsc, desched_epoch) = slot.take_causal_desched();
|
||||
if desched_tsc != 0 && desched_epoch == EXPERIMENT_EPOCH.load(Ordering::Relaxed) {
|
||||
// Same epoch ⇒ no `begin()` since the stash; a nonzero word ⇒
|
||||
// no `end()` either — the gap closed inside its own window.
|
||||
let exp = EXPERIMENT.load(Ordering::Relaxed);
|
||||
if exp != 0 {
|
||||
let pct = exp & 0xffff_ffff;
|
||||
let gap = preempt::rdtsc()
|
||||
.saturating_sub(desched_tsc)
|
||||
.min(MAX_SAMPLE_CYCLES);
|
||||
OFFCPU_IN_SITE_CYCLES.fetch_add(gap.saturating_mul(pct) / 100, Ordering::Relaxed);
|
||||
OFFCPU_IN_SITE_N.fetch_add(1, Ordering::Relaxed);
|
||||
}
|
||||
}
|
||||
if slot.take_causal_parked() {
|
||||
let global = GLOBAL_DELAY.load(Ordering::Relaxed);
|
||||
let mine = slot.causal_delay();
|
||||
if mine < global {
|
||||
PARK_FORGIVEN_CYCLES.fetch_add(global - mine, Ordering::Relaxed);
|
||||
slot.set_causal_delay(global);
|
||||
}
|
||||
}
|
||||
LAST_SAMPLE_TSC.with(|c| c.set(preempt::rdtsc()));
|
||||
}
|
||||
|
||||
/// Deschedule-path hook (scheduler side, same OS thread the actor just
|
||||
/// ran on). If an experiment is live and the departing actor sits in the
|
||||
/// target site, the sample tail `[last sample -> now]` is about to be
|
||||
/// lost: nothing flushes it here, and `on_resume` re-arms the clock
|
||||
/// before the actor runs again. Measure-only (RFC 007 deficit hunt) —
|
||||
/// tally the would-be attribution into the park/yield drop buckets and
|
||||
/// leave behaviour untouched. The interval is capped at
|
||||
/// MAX_SAMPLE_CYCLES: past that the flush would have discarded it anyway
|
||||
/// (counted separately). `now` includes the few hundred ns of scheduler
|
||||
/// bookkeeping since the actor actually stopped — an acceptable
|
||||
/// overcount for a diagnostic.
|
||||
///
|
||||
/// Slice-expiry yields sample at the same checkpoint that deschedules
|
||||
/// them, so their tails are ~zero by construction; a fat yield bucket
|
||||
/// therefore points at explicit `yield_now` calls or requeued parks.
|
||||
///
|
||||
/// Yields additionally stash the deschedule instant on the slot so
|
||||
/// `on_resume` can count the runnable off-CPU gap (offcpu bucket).
|
||||
pub(crate) fn on_deschedule(slot: &crate::runtime::Slot, real_park: bool) {
|
||||
let exp = EXPERIMENT.load(Ordering::Relaxed);
|
||||
if exp == 0 {
|
||||
return;
|
||||
}
|
||||
let target = (exp >> 32) as u32;
|
||||
let pct = exp & 0xffff_ffff;
|
||||
if pct == 0 || slot.causal_site() != target {
|
||||
return;
|
||||
}
|
||||
let now = preempt::rdtsc();
|
||||
if !real_park {
|
||||
// Runnable gap opens here; `on_resume` closes and counts it
|
||||
// (offcpu bucket). Parks are excluded: blocked time is already
|
||||
// represented by forgiveness, and blocked wall time is not
|
||||
// queue-wait.
|
||||
slot.set_causal_desched(now, EXPERIMENT_EPOCH.load(Ordering::Relaxed));
|
||||
}
|
||||
let last = LAST_SAMPLE_TSC.with(|c| c.get());
|
||||
if last == 0 {
|
||||
return;
|
||||
}
|
||||
let interval = now.saturating_sub(last).min(MAX_SAMPLE_CYCLES);
|
||||
let would_be = interval.saturating_mul(pct) / 100;
|
||||
if real_park {
|
||||
DROP_PARK_N.fetch_add(1, Ordering::Relaxed);
|
||||
DROP_PARK_CYCLES.fetch_add(would_be, Ordering::Relaxed);
|
||||
} else {
|
||||
DROP_YIELD_N.fetch_add(1, Ordering::Relaxed);
|
||||
DROP_YIELD_CYCLES.fetch_add(would_be, Ordering::Relaxed);
|
||||
}
|
||||
}
|
||||
|
||||
// -----------------------------------------------------------------------
|
||||
// Experiments
|
||||
// -----------------------------------------------------------------------
|
||||
|
||||
fn begin(site: u32, pct: u32) {
|
||||
EXPERIMENT_EPOCH.fetch_add(1, Ordering::Relaxed);
|
||||
EXPERIMENT.store(((site as u64) << 32) | pct as u64, Ordering::Relaxed);
|
||||
}
|
||||
|
||||
fn end() {
|
||||
EXPERIMENT.store(0, Ordering::Relaxed);
|
||||
}
|
||||
|
||||
/// Total virtual delay injected so far, in TSC cycles.
|
||||
pub fn global_delay_cycles() -> u64 {
|
||||
GLOBAL_DELAY.load(Ordering::Relaxed)
|
||||
}
|
||||
|
||||
/// Cumulative ledger-audit totals since startup (RFC 007 deficit hunt).
|
||||
/// All monotone; window a span by snapshotting before/after and taking
|
||||
/// `delta_since`. Cycle fields are in would-be-injected delta terms so
|
||||
/// they compare directly against `injected_cycles`.
|
||||
#[derive(Clone, Copy, Debug, Default)]
|
||||
pub struct LedgerCounters {
|
||||
pub spin_absorbed_cycles: u64,
|
||||
pub park_forgiven_cycles: u64,
|
||||
pub drop_park_cycles: u64,
|
||||
pub drop_park_n: u64,
|
||||
pub drop_yield_cycles: u64,
|
||||
pub drop_yield_n: u64,
|
||||
pub discard_overmax_cycles: u64,
|
||||
pub discard_overmax_n: u64,
|
||||
pub discard_unarmed_n: u64,
|
||||
pub offcpu_in_site_cycles: u64,
|
||||
pub offcpu_in_site_n: u64,
|
||||
}
|
||||
|
||||
impl LedgerCounters {
|
||||
/// Field-wise difference against an earlier snapshot.
|
||||
pub fn delta_since(&self, before: &LedgerCounters) -> LedgerCounters {
|
||||
LedgerCounters {
|
||||
spin_absorbed_cycles: self
|
||||
.spin_absorbed_cycles
|
||||
.saturating_sub(before.spin_absorbed_cycles),
|
||||
park_forgiven_cycles: self
|
||||
.park_forgiven_cycles
|
||||
.saturating_sub(before.park_forgiven_cycles),
|
||||
drop_park_cycles: self
|
||||
.drop_park_cycles
|
||||
.saturating_sub(before.drop_park_cycles),
|
||||
drop_park_n: self.drop_park_n.saturating_sub(before.drop_park_n),
|
||||
drop_yield_cycles: self
|
||||
.drop_yield_cycles
|
||||
.saturating_sub(before.drop_yield_cycles),
|
||||
drop_yield_n: self.drop_yield_n.saturating_sub(before.drop_yield_n),
|
||||
discard_overmax_cycles: self
|
||||
.discard_overmax_cycles
|
||||
.saturating_sub(before.discard_overmax_cycles),
|
||||
discard_overmax_n: self
|
||||
.discard_overmax_n
|
||||
.saturating_sub(before.discard_overmax_n),
|
||||
discard_unarmed_n: self
|
||||
.discard_unarmed_n
|
||||
.saturating_sub(before.discard_unarmed_n),
|
||||
offcpu_in_site_cycles: self
|
||||
.offcpu_in_site_cycles
|
||||
.saturating_sub(before.offcpu_in_site_cycles),
|
||||
offcpu_in_site_n: self
|
||||
.offcpu_in_site_n
|
||||
.saturating_sub(before.offcpu_in_site_n),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Snapshot the cumulative audit counters.
|
||||
pub fn ledger_counters() -> LedgerCounters {
|
||||
LedgerCounters {
|
||||
spin_absorbed_cycles: SPIN_ABSORBED_CYCLES.load(Ordering::Relaxed),
|
||||
park_forgiven_cycles: PARK_FORGIVEN_CYCLES.load(Ordering::Relaxed),
|
||||
drop_park_cycles: DROP_PARK_CYCLES.load(Ordering::Relaxed),
|
||||
drop_park_n: DROP_PARK_N.load(Ordering::Relaxed),
|
||||
drop_yield_cycles: DROP_YIELD_CYCLES.load(Ordering::Relaxed),
|
||||
drop_yield_n: DROP_YIELD_N.load(Ordering::Relaxed),
|
||||
discard_overmax_cycles: DISCARD_OVERMAX_CYCLES.load(Ordering::Relaxed),
|
||||
discard_overmax_n: DISCARD_OVERMAX_N.load(Ordering::Relaxed),
|
||||
discard_unarmed_n: DISCARD_UNARMED_N.load(Ordering::Relaxed),
|
||||
offcpu_in_site_cycles: OFFCPU_IN_SITE_CYCLES.load(Ordering::Relaxed),
|
||||
offcpu_in_site_n: OFFCPU_IN_SITE_N.load(Ordering::Relaxed),
|
||||
}
|
||||
}
|
||||
|
||||
/// Absorbed-delay ledger of the on-CPU actor (testing/introspection).
|
||||
pub fn my_absorbed_delay_cycles() -> u64 {
|
||||
let slot = preempt::current_slot_ptr();
|
||||
if slot.is_null() {
|
||||
return 0;
|
||||
}
|
||||
// SAFETY: on-CPU actor's slot; see `check_cancelled`.
|
||||
unsafe { (*slot).causal_delay() }
|
||||
}
|
||||
|
||||
/// Test support: start an experiment targeting `site_name` at `pct`%
|
||||
/// virtual speedup. Registers the site if needed.
|
||||
pub fn begin_experiment_for_test(name: &'static str, pct: u32) {
|
||||
begin(site_id(name), pct);
|
||||
}
|
||||
|
||||
/// Test support: stop the running experiment.
|
||||
pub fn end_experiment_for_test() {
|
||||
end();
|
||||
}
|
||||
|
||||
/// Test support: grow the global delay ledger directly, as if target-site
|
||||
/// samples had attributed `cycles` — deterministic driver for the timer
|
||||
/// virtual-time tests. Calibrates the TSC eagerly so conversion later
|
||||
/// never stalls a scheduler loop.
|
||||
pub fn inject_delay_cycles_for_test(cycles: u64) {
|
||||
let _ = tsc_hz();
|
||||
GLOBAL_DELAY.fetch_add(cycles, Ordering::Relaxed);
|
||||
}
|
||||
|
||||
/// Convert ledger cycles to wall time at the measured TSC rate.
|
||||
pub fn cycles_to_duration(cycles: u64) -> Duration {
|
||||
Duration::from_secs_f64(cycles as f64 / tsc_hz())
|
||||
}
|
||||
|
||||
/// Controller parameters: which speedups to try per site, and the
|
||||
/// experiment/cooldown windows.
|
||||
pub struct ExperimentPlan {
|
||||
pub speedups_pct: Vec<u32>,
|
||||
pub experiment: Duration,
|
||||
pub cooldown: Duration,
|
||||
}
|
||||
|
||||
impl Default for ExperimentPlan {
|
||||
fn default() -> Self {
|
||||
ExperimentPlan {
|
||||
speedups_pct: vec![0, 25, 50],
|
||||
experiment: Duration::from_millis(500),
|
||||
cooldown: Duration::from_millis(100),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// One completed experiment cell.
|
||||
#[derive(Default)]
|
||||
pub struct ExperimentResult {
|
||||
pub site: String,
|
||||
pub speedup_pct: u32,
|
||||
pub duration: Duration,
|
||||
/// Progress-point deltas over the window, `(name, count)`.
|
||||
pub deltas: Vec<(String, u64)>,
|
||||
/// Virtual delay injected during the window (cycles).
|
||||
pub injected_cycles: u64,
|
||||
// Ledger-audit deltas over the window (RFC 007 deficit hunt); see
|
||||
// `LedgerCounters` for field semantics. `spin_absorbed_cycles > 0`
|
||||
// in a 0% cell means the window paid debt left over from an earlier
|
||||
// one — the baseline-contamination signature.
|
||||
pub spin_absorbed_cycles: u64,
|
||||
pub park_forgiven_cycles: u64,
|
||||
pub drop_park_cycles: u64,
|
||||
pub drop_park_n: u64,
|
||||
pub drop_yield_cycles: u64,
|
||||
pub drop_yield_n: u64,
|
||||
pub discard_overmax_cycles: u64,
|
||||
pub discard_overmax_n: u64,
|
||||
pub discard_unarmed_n: u64,
|
||||
pub offcpu_in_site_cycles: u64,
|
||||
pub offcpu_in_site_n: u64,
|
||||
}
|
||||
|
||||
/// Run the plan synchronously on the calling (OS) thread: for every
|
||||
/// registered site × speedup, run one experiment window and record
|
||||
/// progress-point deltas, with a cooldown between cells. Sites and
|
||||
/// progress points must already be registered (the workload has to be
|
||||
/// running); the caller owns workload start/stop.
|
||||
///
|
||||
/// v1 controller: exhaustive sweep, fixed windows, no adaptive site
|
||||
/// selection or confidence stopping (jar Q5).
|
||||
///
|
||||
/// Callable from a plain OS thread *or* from inside an actor: sleeping
|
||||
/// parks the green thread when we're on one (so no scheduler thread is
|
||||
/// blocked), and falls back to `thread::sleep` otherwise.
|
||||
pub fn run_experiments(plan: &ExperimentPlan) -> Vec<ExperimentResult> {
|
||||
fn controller_sleep(d: Duration) {
|
||||
if preempt::current_slot_ptr().is_null() {
|
||||
std::thread::sleep(d);
|
||||
} else {
|
||||
// Wall-anchored: the controller's window/cooldown sleeps
|
||||
// *define* the experiment's wall length; letting them chase
|
||||
// the delay it is itself injecting would stretch every window
|
||||
// (observed ~2x at 50% speedup). Deltas are rate-normalized
|
||||
// either way — this fixes cost, not bias.
|
||||
crate::scheduler::sleep_wall(d);
|
||||
}
|
||||
}
|
||||
// Calibrate before any window so report rendering never has to sleep.
|
||||
let _ = tsc_hz();
|
||||
let site_list: Vec<(u32, String)> = {
|
||||
let v = lock_unpoisoned(sites());
|
||||
v.iter()
|
||||
.enumerate()
|
||||
.map(|(i, n)| ((i + 1) as u32, (*n).to_string()))
|
||||
.collect()
|
||||
};
|
||||
let mut out = Vec::new();
|
||||
for (sid, sname) in &site_list {
|
||||
for &pct in &plan.speedups_pct {
|
||||
let before = progress_snapshot();
|
||||
let injected_before = global_delay_cycles();
|
||||
let audit_before = ledger_counters();
|
||||
let t0 = Instant::now();
|
||||
begin(*sid, pct);
|
||||
controller_sleep(plan.experiment);
|
||||
end();
|
||||
// Snapshot immediately: injection and spin freeze at `end()`
|
||||
// (checks gate on the experiment word), but forgiveness does
|
||||
// not — a later snapshot would leak cooldown wakes into the
|
||||
// window.
|
||||
let audit = ledger_counters().delta_since(&audit_before);
|
||||
let elapsed = t0.elapsed();
|
||||
let after = progress_snapshot();
|
||||
let deltas = after
|
||||
.iter()
|
||||
.map(|(n, c)| {
|
||||
let b = before
|
||||
.iter()
|
||||
.find(|(bn, _)| bn == n)
|
||||
.map(|(_, bc)| *bc)
|
||||
.unwrap_or(0);
|
||||
(n.clone(), c.saturating_sub(b))
|
||||
})
|
||||
.collect();
|
||||
out.push(ExperimentResult {
|
||||
site: sname.clone(),
|
||||
speedup_pct: pct,
|
||||
duration: elapsed,
|
||||
deltas,
|
||||
injected_cycles: global_delay_cycles() - injected_before,
|
||||
spin_absorbed_cycles: audit.spin_absorbed_cycles,
|
||||
park_forgiven_cycles: audit.park_forgiven_cycles,
|
||||
drop_park_cycles: audit.drop_park_cycles,
|
||||
drop_park_n: audit.drop_park_n,
|
||||
drop_yield_cycles: audit.drop_yield_cycles,
|
||||
drop_yield_n: audit.drop_yield_n,
|
||||
discard_overmax_cycles: audit.discard_overmax_cycles,
|
||||
discard_overmax_n: audit.discard_overmax_n,
|
||||
discard_unarmed_n: audit.discard_unarmed_n,
|
||||
offcpu_in_site_cycles: audit.offcpu_in_site_cycles,
|
||||
offcpu_in_site_n: audit.offcpu_in_site_n,
|
||||
});
|
||||
controller_sleep(plan.cooldown);
|
||||
}
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
// -----------------------------------------------------------------------
|
||||
// Reports
|
||||
// -----------------------------------------------------------------------
|
||||
|
||||
/// Measured TSC frequency (Hz), calibrated once. The crate-wide 3 GHz
|
||||
/// constant is fine for the *relative* timeslice check, but report
|
||||
/// normalisation divides wall time by injected time, so a 20% Hz error
|
||||
/// skews every impact number — measured live: a 3.7 GHz box inflated all
|
||||
/// baselines uniformly. Calibrated against `Instant` over ~50ms on first
|
||||
/// use; `run_experiments` triggers it before its first window (using the
|
||||
/// park-aware sleep, so no scheduler thread is blocked when called from
|
||||
/// an actor).
|
||||
static TSC_HZ_MEASURED: OnceLock<f64> = OnceLock::new();
|
||||
|
||||
/// Measured TSC frequency in Hz. Calibrates on first call (~50ms).
|
||||
pub fn tsc_hz() -> f64 {
|
||||
*TSC_HZ_MEASURED.get_or_init(|| {
|
||||
let c0 = preempt::rdtsc();
|
||||
let t0 = Instant::now();
|
||||
let d = Duration::from_millis(50);
|
||||
if preempt::current_slot_ptr().is_null() {
|
||||
std::thread::sleep(d);
|
||||
} else {
|
||||
// Wall-anchored: calibration divides TSC delta by *wall*
|
||||
// elapsed; a virtual sleep dilated by concurrent injection
|
||||
// would still measure correctly (elapsed() is wall) but
|
||||
// waste window time — and must never depend on the ledger
|
||||
// it exists to convert.
|
||||
crate::scheduler::sleep_wall(d);
|
||||
}
|
||||
(preempt::rdtsc().wrapping_sub(c0)) as f64 / t0.elapsed().as_secs_f64()
|
||||
})
|
||||
}
|
||||
|
||||
/// Normalized rate for one cell: count over the *virtual* window
|
||||
/// (wall − injected) — Coz's normalization: injected delay does not
|
||||
/// exist in the virtual timeline. A bottleneck site keeps its raw count
|
||||
/// while shrinking the divisor → positive impact; a fully overlapped
|
||||
/// site loses count proportionally → ~zero.
|
||||
fn normalized_rate(r: &ExperimentResult, point: &str) -> Option<f64> {
|
||||
let count = r.deltas.iter().find(|(n, _)| n == point).map(|(_, c)| *c)?;
|
||||
let injected_secs = r.injected_cycles as f64 / tsc_hz();
|
||||
let virtual_secs = (r.duration.as_secs_f64() - injected_secs).max(1e-9);
|
||||
Some(count as f64 / virtual_secs)
|
||||
}
|
||||
|
||||
/// Impact of virtually speeding up `site` by `speedup_pct` on progress
|
||||
/// point `point`, in percent relative to that site's own 0% baseline
|
||||
/// cell. `None` if either cell or the point is missing, or the baseline
|
||||
/// rate is zero. This is the machine-readable form of the summary's
|
||||
/// "vs baseline" column, for programmatic checks (CI, examples).
|
||||
pub fn impact_pct(
|
||||
results: &[ExperimentResult],
|
||||
site: &str,
|
||||
speedup_pct: u32,
|
||||
point: &str,
|
||||
) -> Option<f64> {
|
||||
let cell = results
|
||||
.iter()
|
||||
.find(|r| r.site == site && r.speedup_pct == speedup_pct)?;
|
||||
let base = results
|
||||
.iter()
|
||||
.find(|r| r.site == site && r.speedup_pct == 0)?;
|
||||
let rate = normalized_rate(cell, point)?;
|
||||
let b = normalized_rate(base, point)?;
|
||||
if b <= 0.0 {
|
||||
return None;
|
||||
}
|
||||
Some((rate / b - 1.0) * 100.0)
|
||||
}
|
||||
|
||||
/// Human-readable summary: per (site, progress point), the throughput at
|
||||
/// each virtual speedup and the change relative to that site's own 0%
|
||||
/// baseline. A near-zero column across speedups means: optimising this
|
||||
/// site buys you nothing — the RFC's headline answer.
|
||||
///
|
||||
/// Ends with a one-line fidelity note (RFC 007 Validation): reported
|
||||
/// impacts are conservative — attribution counts on-CPU site time only,
|
||||
/// so runnable queue-wait inside the site (the located @50 "deficit",
|
||||
/// eff ≈ 0.93 live) is never injected and gains are lower bounds; site
|
||||
/// *rankings* are unaffected.
|
||||
pub fn render_summary(results: &[ExperimentResult]) -> String {
|
||||
use std::fmt::Write;
|
||||
let mut s = String::new();
|
||||
let _ = writeln!(s, "== smarm causal profile ==");
|
||||
let mut sites_seen: Vec<&str> = Vec::new();
|
||||
for r in results {
|
||||
if !sites_seen.contains(&r.site.as_str()) {
|
||||
sites_seen.push(&r.site);
|
||||
}
|
||||
}
|
||||
for site in sites_seen {
|
||||
let _ = writeln!(s, "site {site}");
|
||||
for r in results.iter().filter(|r| r.site == site) {
|
||||
for (name, _) in &r.deltas {
|
||||
let rate = match normalized_rate(r, name) {
|
||||
Some(x) => x,
|
||||
None => continue,
|
||||
};
|
||||
let rel = impact_pct(results, site, r.speedup_pct, name)
|
||||
.map(|p| format!("{p:+.1}%"))
|
||||
.unwrap_or_else(|| "n/a".to_string());
|
||||
let _ = writeln!(
|
||||
s,
|
||||
" speedup {:>3}% {name:<24} {rate:>12.1}/s vs baseline {rel} (injected {:.1}ms)",
|
||||
r.speedup_pct,
|
||||
r.injected_cycles as f64 / tsc_hz() * 1e3
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
if !results.is_empty() {
|
||||
let _ = writeln!(
|
||||
s,
|
||||
"note: impacts are lower bounds — site time counts on-CPU only (runnable queue-wait is not attributed); rankings unaffected"
|
||||
);
|
||||
}
|
||||
s
|
||||
}
|
||||
|
||||
/// Ledger-audit companion to `render_summary` (RFC 007 deficit hunt):
|
||||
/// per cell, where the window's virtual delay went — born (injected),
|
||||
/// paid (absorbed), waived (forgiven at wake) — and the attribution the
|
||||
/// sampler lost: tails dropped at parks/yields inside the target site,
|
||||
/// plus clamp discards. Cycle columns in ms at the calibrated TSC rate.
|
||||
/// `absorbed` above `injected` in a cell (0% especially) means it paid
|
||||
/// debt left over from earlier windows.
|
||||
pub fn render_ledger_audit(results: &[ExperimentResult]) -> String {
|
||||
use std::fmt::Write;
|
||||
let hz = tsc_hz();
|
||||
let ms = |c: u64| c as f64 / hz * 1e3;
|
||||
let mut s = String::new();
|
||||
let _ = writeln!(s, "== smarm causal ledger audit ==");
|
||||
for r in results {
|
||||
let _ = writeln!(
|
||||
s,
|
||||
"site {:<22} @{:>2}% injected {:>7.1}ms absorbed {:>7.1}ms forgiven {:>7.1}ms \
|
||||
drop park {:>6.2}ms/{:<4} yield {:>6.2}ms/{:<4} offcpu {:>6.2}ms/{:<5} \
|
||||
discard >max {:>6.2}ms/{:<3} unarmed {}",
|
||||
r.site,
|
||||
r.speedup_pct,
|
||||
ms(r.injected_cycles),
|
||||
ms(r.spin_absorbed_cycles),
|
||||
ms(r.park_forgiven_cycles),
|
||||
ms(r.drop_park_cycles),
|
||||
r.drop_park_n,
|
||||
ms(r.drop_yield_cycles),
|
||||
r.drop_yield_n,
|
||||
ms(r.offcpu_in_site_cycles),
|
||||
r.offcpu_in_site_n,
|
||||
ms(r.discard_overmax_cycles),
|
||||
r.discard_overmax_n,
|
||||
r.discard_unarmed_n
|
||||
);
|
||||
}
|
||||
s
|
||||
}
|
||||
|
||||
/// Coz-compatible profile text (`profile.coz`), so Coz's existing plot
|
||||
/// tooling renders our experiments — the RFC's "don't build a UI" call.
|
||||
pub fn render_coz(results: &[ExperimentResult]) -> String {
|
||||
use std::fmt::Write;
|
||||
let mut s = String::new();
|
||||
let _ = writeln!(s, "startup\ttime=0");
|
||||
for r in results {
|
||||
let _ = writeln!(
|
||||
s,
|
||||
"experiment\tselected={}\tspeedup={:.2}\tduration={}\tselected-samples=1",
|
||||
r.site,
|
||||
r.speedup_pct as f64 / 100.0,
|
||||
r.duration.as_nanos()
|
||||
);
|
||||
for (name, count) in &r.deltas {
|
||||
let _ = writeln!(s, "throughput-point\tname={name}\tdelta={count}");
|
||||
}
|
||||
}
|
||||
s
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(feature = "smarm-causal")]
|
||||
pub use inner::*;
|
||||
|
||||
/// Mark one unit of useful work complete at a named throughput progress
|
||||
/// point (RFC 007). One Relaxed increment when `smarm-causal` is on; nothing
|
||||
/// at all when it's off.
|
||||
#[cfg(feature = "smarm-causal")]
|
||||
#[macro_export]
|
||||
macro_rules! progress {
|
||||
($name:literal) => {{
|
||||
static __SMARM_PP: ::std::sync::OnceLock<&'static $crate::causal::ProgressPoint> =
|
||||
::std::sync::OnceLock::new();
|
||||
__SMARM_PP
|
||||
.get_or_init(|| $crate::causal::register_progress($name))
|
||||
.bump();
|
||||
}};
|
||||
}
|
||||
|
||||
#[cfg(not(feature = "smarm-causal"))]
|
||||
#[macro_export]
|
||||
macro_rules! progress {
|
||||
($name:literal) => {{}};
|
||||
}
|
||||
|
||||
/// Enter a named causal-profiling site for the current actor; the returned
|
||||
/// guard exits it (restoring any enclosing site) on drop. Site identity is
|
||||
/// stored in the actor's slot, so it survives preemption and migration.
|
||||
/// Expands to a unit no-op without `smarm-causal`.
|
||||
#[cfg(feature = "smarm-causal")]
|
||||
#[macro_export]
|
||||
macro_rules! causal_site {
|
||||
($name:literal) => {{
|
||||
static __SMARM_SITE: ::std::sync::OnceLock<u32> = ::std::sync::OnceLock::new();
|
||||
$crate::causal::SiteGuard::enter(
|
||||
*__SMARM_SITE.get_or_init(|| $crate::causal::site_id($name)),
|
||||
)
|
||||
}};
|
||||
}
|
||||
|
||||
#[cfg(not(feature = "smarm-causal"))]
|
||||
#[macro_export]
|
||||
macro_rules! causal_site {
|
||||
($name:literal) => {
|
||||
()
|
||||
};
|
||||
}
|
||||
+68
-706
@@ -1,175 +1,51 @@
|
||||
//! Unbounded multi-producer, single-consumer channels: how actors talk to
|
||||
//! each other.
|
||||
//! Unbounded MPSC channels.
|
||||
//!
|
||||
//! A channel is a queue with a typed [`Sender`] on one end and a typed
|
||||
//! [`Receiver`] on the other. Any number of actors can hold a clone of the
|
||||
//! `Sender` and push messages onto the same queue; exactly one [`Receiver`]
|
||||
//! reads them back out, in the order they arrived. This is the basic wiring
|
||||
//! smarm's other actor primitives (`gen_server`, `pg`, the registry) are all
|
||||
//! built out of, and it is directly usable on its own for a worker that just
|
||||
//! needs an inbox.
|
||||
//! Inner state is `Arc<Mutex<Inner<T>>>` so channels can be sent across OS
|
||||
//! threads (required for the multi-scheduler runtime where a sender and
|
||||
//! receiver may run on different scheduler threads simultaneously).
|
||||
//!
|
||||
//! ## A first channel
|
||||
//!
|
||||
//! ```
|
||||
//! use smarm::{channel, run, spawn};
|
||||
//!
|
||||
//! run(|| {
|
||||
//! let (tx, rx) = channel::<u64>();
|
||||
//!
|
||||
//! let worker = spawn(move || {
|
||||
//! // Blocks until a message arrives.
|
||||
//! let n = rx.recv().unwrap();
|
||||
//! assert_eq!(n, 42);
|
||||
//!
|
||||
//! // Once every Sender is dropped, recv() reports the channel closed
|
||||
//! // instead of blocking forever.
|
||||
//! assert!(rx.recv().is_err());
|
||||
//! });
|
||||
//!
|
||||
//! tx.send(42).unwrap();
|
||||
//! drop(tx); // last sender gone: the channel is now closed
|
||||
//! worker.join().unwrap();
|
||||
//! });
|
||||
//! ```
|
||||
//!
|
||||
//! ## Sending
|
||||
//!
|
||||
//! [`Sender`] is cheaply clonable: hand a clone to every actor that needs to
|
||||
//! push messages into this queue. The channel stays open as long as at least
|
||||
//! one clone exists; [`Sender::send`] never blocks and always succeeds while
|
||||
//! the channel is open, since the queue is unbounded. Once the [`Receiver`]
|
||||
//! has been dropped, `send` returns the message back to you in
|
||||
//! [`SendError`] instead of delivering it.
|
||||
//!
|
||||
//! ## Receiving
|
||||
//!
|
||||
//! There is exactly one [`Receiver`] per channel (it is not clonable).
|
||||
//! [`Receiver::recv`] returns the next message in arrival order, parking the
|
||||
//! calling actor if the queue is currently empty. Once every `Sender` has
|
||||
//! been dropped and the queue has been drained, `recv` stops parking and
|
||||
//! returns [`RecvError`] instead, so a receiver never blocks forever waiting
|
||||
//! on senders that are never coming back.
|
||||
//!
|
||||
//! Beyond plain `recv`, three variants cover the common needs:
|
||||
//!
|
||||
//! - [`Receiver::try_recv`]: never parks: reports an empty-but-open channel
|
||||
//! as `Ok(None)` instead of waiting.
|
||||
//! - [`Receiver::recv_timeout`]: parks, but gives up and returns
|
||||
//! [`RecvTimeoutError::Timeout`] if no message arrives before a deadline.
|
||||
//! - [`Receiver::recv_match`] / [`Receiver::try_recv_match`]: selective
|
||||
//! receive. Instead of taking whatever is at the front of the queue, pick
|
||||
//! out the first message matching a predicate, leaving the rest queued in
|
||||
//! order. Handy for an actor that wants to prioritise one kind of message
|
||||
//! over others already waiting.
|
||||
//!
|
||||
//! ## Waiting on several channels: `select`
|
||||
//!
|
||||
//! [`select`] parks an actor across several receivers at once and reports
|
||||
//! the index of the first one that is ready (has a message queued, or has
|
||||
//! been closed). [`select_timeout`] adds a deadline, the way `recv_timeout`
|
||||
//! does for a single channel. See their docs for the full contract,
|
||||
//! including the priority-order and no-fairness guarantee.
|
||||
//!
|
||||
//! ## Implementation notes
|
||||
//!
|
||||
//! The queue and its bookkeeping live behind `Arc<RawMutex<Inner<T>>>`
|
||||
//! rather than a `std::sync::Mutex`, so that a channel can be freely shared
|
||||
//! and sent across the OS threads backing the multi-scheduler runtime.
|
||||
//! `RawMutex` matters here for a subtler reason too: an ordinary pthread
|
||||
//! mutex can be released from a different OS thread than the one that took
|
||||
//! it (smarm's preemption can migrate a timesliced actor between scheduler
|
||||
//! threads mid-critical-section), and doing that to a `std::sync::Mutex` is
|
||||
//! undefined behavior. `RawMutex` disables preemption for the guard's short
|
||||
//! lifetime instead, so the release always happens on the thread that
|
||||
//! acquired it, and it has no poisoning to worry about besides. Channel
|
||||
//! locks are cheap and are never held across another lock acquisition or a
|
||||
//! blocking call; the predicate passed to `recv_match` runs under this lock,
|
||||
//! which is why it needs to stay cheap, pure, and must not call back into
|
||||
//! the same channel.
|
||||
//! Semantics:
|
||||
//! - Senders are clonable; the last sender drop closes the channel.
|
||||
//! - `Receiver::recv` on an empty open channel parks the receiver.
|
||||
//! - `Receiver::recv` on an empty closed channel returns `Err(RecvError)`.
|
||||
//! - `Sender::send` on an open channel always succeeds.
|
||||
//! - `Sender::send` on a closed channel (receiver dropped) returns
|
||||
//! `Err(SendError(value))`.
|
||||
//! - When a send pushes to a previously empty queue and a receiver is
|
||||
//! parked, the receiver is unparked.
|
||||
|
||||
use crate::pid::Pid;
|
||||
use crate::raw_mutex::RawMutex;
|
||||
use crate::runtime::RuntimeInner;
|
||||
use std::collections::VecDeque;
|
||||
use std::sync::{Arc, Weak};
|
||||
use std::sync::{Arc, Mutex};
|
||||
|
||||
/// Create a new channel and return its `(Sender, Receiver)` halves.
|
||||
///
|
||||
/// The channel is unbounded (no capacity limit) and open until every
|
||||
/// `Sender` has been dropped.
|
||||
pub fn channel<T>() -> (Sender<T>, Receiver<T>) {
|
||||
let inner = Arc::new(RawMutex::new_channel(Inner {
|
||||
let inner = Arc::new(Mutex::new(Inner {
|
||||
queue: VecDeque::new(),
|
||||
parked_receiver: None,
|
||||
rt: None,
|
||||
senders: 1,
|
||||
receiver_alive: true,
|
||||
}));
|
||||
(
|
||||
Sender {
|
||||
inner: inner.clone(),
|
||||
},
|
||||
Receiver { inner },
|
||||
)
|
||||
(Sender { inner: inner.clone() }, Receiver { inner })
|
||||
}
|
||||
|
||||
struct Inner<T> {
|
||||
queue: VecDeque<T>,
|
||||
/// The parked receiver's `(pid, park-epoch)`, if one is currently
|
||||
/// waiting. The epoch identifies exactly which wait this is, so a waker
|
||||
/// left over from a wait that already ended (a losing `select` arm, a
|
||||
/// `recv_timeout` whose timer fired after it was already satisfied) is
|
||||
/// inert and does nothing when it fires.
|
||||
parked_receiver: Option<(Pid, u32)>,
|
||||
/// The receiver's runtime, captured the first time it parks (so provably
|
||||
/// alive then) and kept for the life of the channel: it lets a sender on a
|
||||
/// foreign OS thread wake the receiver without the `RUNTIME` thread-local,
|
||||
/// which is unset off a scheduler thread. Captured once rather than per
|
||||
/// park because `Arc::downgrade` + drop is a locked RMW pair on a shared
|
||||
/// counter, and parking is the channel hot path. A `Receiver` never
|
||||
/// migrates between runtimes — it is pinned to its actor — so one capture
|
||||
/// stays correct for every later park.
|
||||
rt: Option<Weak<RuntimeInner>>,
|
||||
parked_receiver: Option<Pid>,
|
||||
senders: usize,
|
||||
receiver_alive: bool,
|
||||
}
|
||||
|
||||
impl<T> Inner<T> {
|
||||
/// Capture the receiver's runtime if we have not already. Called under the
|
||||
/// channel lock at every park site; after the first park it is one branch
|
||||
/// on an `Option`, no atomics.
|
||||
fn note_runtime(&mut self) {
|
||||
if self.rt.is_none() {
|
||||
self.rt = crate::scheduler::runtime_weak();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// The sending half of a channel, created by [`channel`]. Clonable: every
|
||||
/// clone pushes onto the same queue, and the channel stays open as long as
|
||||
/// any clone is alive. Dropping the last `Sender` closes the channel, which
|
||||
/// wakes a parked [`Receiver`] so it can observe the closure.
|
||||
pub struct Sender<T> {
|
||||
inner: Arc<RawMutex<Inner<T>>>,
|
||||
inner: Arc<Mutex<Inner<T>>>,
|
||||
}
|
||||
|
||||
/// The receiving half of a channel, created by [`channel`]. Not clonable:
|
||||
/// a channel has exactly one receiver. Reads messages in the order they
|
||||
/// were sent, via [`recv`](Receiver::recv) and its variants.
|
||||
pub struct Receiver<T> {
|
||||
inner: Arc<RawMutex<Inner<T>>>,
|
||||
inner: Arc<Mutex<Inner<T>>>,
|
||||
}
|
||||
|
||||
/// Returned by [`Sender::send`] when the channel's [`Receiver`] has already
|
||||
/// been dropped. Carries the message back so it is never silently lost;
|
||||
/// recover it with `.0` or by matching.
|
||||
#[derive(Debug, PartialEq, Eq)]
|
||||
pub struct SendError<T>(pub T);
|
||||
|
||||
/// Returned by [`Receiver::recv`] (and the other receive methods, in their
|
||||
/// own error types) when the channel is closed: every `Sender` has been
|
||||
/// dropped and no message is left queued.
|
||||
#[derive(Debug, PartialEq, Eq, Clone, Copy)]
|
||||
pub struct RecvError;
|
||||
|
||||
@@ -181,46 +57,23 @@ impl std::fmt::Display for RecvError {
|
||||
|
||||
impl std::error::Error for RecvError {}
|
||||
|
||||
/// Returned by [`Receiver::recv_timeout`].
|
||||
#[derive(Debug, PartialEq, Eq, Clone, Copy)]
|
||||
pub enum RecvTimeoutError {
|
||||
/// The deadline passed with no message available.
|
||||
Timeout,
|
||||
/// Every sender was dropped with no message available. The
|
||||
/// timeout-aware counterpart of plain [`RecvError`].
|
||||
Disconnected,
|
||||
}
|
||||
|
||||
impl std::fmt::Display for RecvTimeoutError {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
match self {
|
||||
RecvTimeoutError::Timeout => write!(f, "recv timed out"),
|
||||
RecvTimeoutError::Disconnected => write!(f, "channel closed"),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl std::error::Error for RecvTimeoutError {}
|
||||
|
||||
impl<T> Clone for Sender<T> {
|
||||
fn clone(&self) -> Self {
|
||||
self.inner.lock().senders += 1;
|
||||
Sender {
|
||||
inner: self.inner.clone(),
|
||||
}
|
||||
self.inner.lock().unwrap().senders += 1;
|
||||
Sender { inner: self.inner.clone() }
|
||||
}
|
||||
}
|
||||
|
||||
impl<T> Drop for Sender<T> {
|
||||
fn drop(&mut self) {
|
||||
let unpark = {
|
||||
let mut g = self.inner.lock();
|
||||
let mut g = self.inner.lock().unwrap();
|
||||
g.senders -= 1;
|
||||
// Wake the parked receiver on the last sender drop regardless of
|
||||
// whether the queue is empty. A plain `recv` only ever parks on an
|
||||
// empty queue (so this is unchanged for it), but a selective
|
||||
// `recv_match` may be parked on a non-empty queue holding only
|
||||
// non-matching messages. It must wake to observe closure and
|
||||
// `recv_match` may be parked on a *non-empty* queue holding only
|
||||
// non-matching messages — it must wake to observe closure and
|
||||
// return Err rather than sleep forever.
|
||||
if g.senders == 0 {
|
||||
g.parked_receiver.take()
|
||||
@@ -228,271 +81,116 @@ impl<T> Drop for Sender<T> {
|
||||
None
|
||||
}
|
||||
};
|
||||
if let Some((pid, epoch)) = unpark {
|
||||
crate::scheduler::unpark_at_via(pid, epoch, || self.inner.lock().rt.clone());
|
||||
if let Some(pid) = unpark {
|
||||
crate::scheduler::unpark(pid);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl<T> Drop for Receiver<T> {
|
||||
fn drop(&mut self) {
|
||||
// The only consumer is gone: queued messages can never be delivered.
|
||||
// Drop them now instead of leaving them queued until the last Sender
|
||||
// happens to go away, which can be long after this receiver's owner
|
||||
// has exited if some other part of the runtime is still holding a
|
||||
// clone of the Sender. Draining runs each queued message's own drop
|
||||
// glue, which matters for a gen_server call: dropping a queued call
|
||||
// envelope drops its reply channel too, which wakes the caller with
|
||||
// an error instead of leaving it parked forever. Drain under the
|
||||
// lock, then run the drops after releasing it, since a message's
|
||||
// drop glue may itself touch a different channel or the scheduler.
|
||||
let drained = {
|
||||
let mut g = self.inner.lock();
|
||||
g.receiver_alive = false;
|
||||
std::mem::take(&mut g.queue)
|
||||
};
|
||||
drop(drained);
|
||||
self.inner.lock().unwrap().receiver_alive = false;
|
||||
}
|
||||
}
|
||||
|
||||
impl<T> Sender<T> {
|
||||
/// Number of messages currently queued and not yet received. For
|
||||
/// introspection and monitoring; takes the channel's internal lock, so
|
||||
/// avoid calling it from a hot path.
|
||||
pub(crate) fn queued_len(&self) -> usize {
|
||||
self.inner.lock().queue.len()
|
||||
}
|
||||
|
||||
/// Whether the [`Receiver`] is still alive (a send would be accepted).
|
||||
pub(crate) fn receiver_alive(&self) -> bool {
|
||||
self.inner.lock().receiver_alive
|
||||
}
|
||||
|
||||
/// Whether `other` is a sender of this very channel (a clone).
|
||||
pub(crate) fn same_channel(&self, other: &Sender<T>) -> bool {
|
||||
Arc::ptr_eq(&self.inner, &other.inner)
|
||||
}
|
||||
|
||||
/// Push `value` onto the channel. Succeeds unconditionally as long as
|
||||
/// the [`Receiver`] is still alive: the queue has no capacity limit, so
|
||||
/// this never blocks and never fails except when the channel is closed,
|
||||
/// in which case `value` comes back in [`SendError`].
|
||||
pub fn send(&self, value: T) -> Result<(), SendError<T>> {
|
||||
let unpark = {
|
||||
let mut g = self.inner.lock();
|
||||
let mut g = self.inner.lock().unwrap();
|
||||
if !g.receiver_alive {
|
||||
return Err(SendError(value));
|
||||
}
|
||||
g.queue.push_back(value);
|
||||
g.parked_receiver.take()
|
||||
};
|
||||
if let Some((pid, epoch)) = unpark {
|
||||
crate::te!(crate::trace::Event::Send {
|
||||
sender: crate::actor::current_pid()
|
||||
.unwrap_or(crate::pid::Pid::new(u32::MAX, u32::MAX)),
|
||||
receiver: Some(pid)
|
||||
});
|
||||
crate::scheduler::unpark_at_via(pid, epoch, || self.inner.lock().rt.clone());
|
||||
if let Some(pid) = unpark {
|
||||
crate::te!(crate::trace::Event::Send { sender: crate::actor::current_pid().unwrap_or(crate::pid::Pid::new(u32::MAX, u32::MAX)), receiver: Some(pid) });
|
||||
crate::scheduler::unpark(pid);
|
||||
} else {
|
||||
crate::te!(crate::trace::Event::Send {
|
||||
sender: crate::actor::current_pid()
|
||||
.unwrap_or(crate::pid::Pid::new(u32::MAX, u32::MAX)),
|
||||
receiver: None
|
||||
});
|
||||
crate::te!(crate::trace::Event::Send { sender: crate::actor::current_pid().unwrap_or(crate::pid::Pid::new(u32::MAX, u32::MAX)), receiver: None });
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
impl<T> Receiver<T> {
|
||||
/// Block until a message is available and return it. Messages come back
|
||||
/// in the order they were sent. If the queue is empty and every
|
||||
/// [`Sender`] has already been dropped, returns [`RecvError`] instead of
|
||||
/// blocking forever.
|
||||
pub fn recv(&self) -> Result<T, RecvError> {
|
||||
loop {
|
||||
{
|
||||
let mut g = self.inner.lock();
|
||||
let mut g = self.inner.lock().unwrap();
|
||||
if let Some(v) = g.queue.pop_front() {
|
||||
crate::preempt::note_message_received();
|
||||
return Ok(v);
|
||||
}
|
||||
if g.senders == 0 {
|
||||
return Err(RecvError);
|
||||
}
|
||||
let me = match crate::actor::current_pid() {
|
||||
Some(me) => me,
|
||||
None => panic!("smarm: recv() called outside an actor"),
|
||||
};
|
||||
let me = crate::actor::current_pid()
|
||||
.expect("recv() called outside an actor");
|
||||
debug_assert!(
|
||||
g.parked_receiver.is_none_or(|(p, _)| p == me),
|
||||
g.parked_receiver.is_none(),
|
||||
"channel has more than one receiver"
|
||||
);
|
||||
// begin_wait is lock-free, so it's legal under the Channel lock;
|
||||
// registering in the same critical section makes the epoch
|
||||
// atomic with the senders' view of the registration.
|
||||
g.note_runtime();
|
||||
g.parked_receiver = Some((me, crate::scheduler::begin_wait()));
|
||||
g.parked_receiver = Some(me);
|
||||
crate::te!(crate::trace::Event::RecvPark(me));
|
||||
}
|
||||
// Release the lock before parking: the unparker will need it.
|
||||
// Release the lock before parking — the unparker will need it.
|
||||
crate::scheduler::park_current();
|
||||
// Woken up. Record it before looping to check the queue.
|
||||
crate::te!(crate::trace::Event::RecvWake(
|
||||
match crate::actor::current_pid() {
|
||||
Some(p) => p,
|
||||
None => panic!("smarm: RecvWake outside an actor (core corrupt)"),
|
||||
}
|
||||
));
|
||||
// Woken up — record it before looping to check the queue.
|
||||
crate::te!(crate::trace::Event::RecvWake(crate::actor::current_pid().unwrap()));
|
||||
}
|
||||
}
|
||||
|
||||
/// Like [`recv`](Self::recv), but gives up and returns
|
||||
/// [`RecvTimeoutError::Timeout`] if no message has arrived by the time
|
||||
/// `timeout` elapses.
|
||||
/// Selective receive: remove and return the first queued message for which
|
||||
/// `pred` holds, leaving the rest in arrival order. If no queued message
|
||||
/// matches, parks and re-scans on every send (a selective receiver may park
|
||||
/// on a *non-empty* queue). Returns `Err(RecvError)` only once the channel
|
||||
/// is closed and no queued message matches.
|
||||
///
|
||||
/// If a message arrives at essentially the same moment the deadline
|
||||
/// passes, the message wins: you get `Ok` rather than `Timeout`. If
|
||||
/// every sender is dropped before a message arrives or the deadline
|
||||
/// passes, you get [`RecvTimeoutError::Disconnected`].
|
||||
///
|
||||
/// `Duration::ZERO` is a valid timeout: it still gives any
|
||||
/// already-queued message a chance to be returned, and only then
|
||||
/// reports `Timeout`.
|
||||
pub fn recv_timeout(&self, timeout: std::time::Duration) -> Result<T, RecvTimeoutError>
|
||||
where
|
||||
T: Send + 'static,
|
||||
{
|
||||
let me = match crate::actor::current_pid() {
|
||||
Some(me) => me,
|
||||
None => panic!("smarm: recv_timeout() called outside an actor"),
|
||||
};
|
||||
|
||||
// Fast path + wait registration, one critical section.
|
||||
let epoch;
|
||||
{
|
||||
let mut g = self.inner.lock();
|
||||
if let Some(v) = g.queue.pop_front() {
|
||||
crate::preempt::note_message_received();
|
||||
return Ok(v);
|
||||
}
|
||||
if g.senders == 0 {
|
||||
return Err(RecvTimeoutError::Disconnected);
|
||||
}
|
||||
debug_assert!(
|
||||
g.parked_receiver.is_none_or(|(p, _)| p == me),
|
||||
"channel has more than one receiver"
|
||||
);
|
||||
epoch = crate::scheduler::begin_wait();
|
||||
g.note_runtime();
|
||||
g.parked_receiver = Some((me, epoch));
|
||||
crate::te!(crate::trace::Event::RecvPark(me));
|
||||
}
|
||||
|
||||
// Arm the timer after releasing the channel lock (insert takes the
|
||||
// timers lock; never nest under a Channel lock). A send or even the
|
||||
// timer itself may unpark us before we park; the runtime's wake
|
||||
// protocol makes the park below return immediately in that case.
|
||||
let deadline = crate::timer::deadline_from_now(timeout);
|
||||
let target: std::sync::Arc<dyn crate::timer::TimerTarget> = self.inner.clone();
|
||||
crate::scheduler::insert_wait_timer(deadline, me, target, epoch);
|
||||
|
||||
crate::scheduler::park_current();
|
||||
crate::te!(crate::trace::Event::RecvWake(
|
||||
match crate::actor::current_pid() {
|
||||
Some(p) => p,
|
||||
None => panic!("smarm: RecvWake outside an actor (core corrupt)"),
|
||||
}
|
||||
));
|
||||
let mut g = self.inner.lock();
|
||||
if let Some(v) = g.queue.pop_front() {
|
||||
crate::preempt::note_message_received();
|
||||
return Ok(v);
|
||||
}
|
||||
if g.senders == 0 {
|
||||
return Err(RecvTimeoutError::Disconnected);
|
||||
}
|
||||
Err(RecvTimeoutError::Timeout)
|
||||
}
|
||||
|
||||
/// Selective receive: find and return the first queued message for
|
||||
/// which `pred` returns `true`, leaving every other message in the
|
||||
/// queue untouched and in order. Useful when an actor's inbox mixes
|
||||
/// message kinds and it wants to handle one kind out of turn, without
|
||||
/// discarding the rest.
|
||||
///
|
||||
/// If nothing queued matches, this blocks and re-checks every time a new
|
||||
/// message arrives, the same way [`recv`](Self::recv) blocks on an empty
|
||||
/// queue: a selective receiver can be waiting even while the queue holds
|
||||
/// messages, just none that match yet. Returns [`RecvError`] only once
|
||||
/// the channel is closed and still nothing matches.
|
||||
///
|
||||
/// `pred` runs while the channel is locked, so keep it cheap, side
|
||||
/// effect free, and make sure it never calls back into this same
|
||||
/// channel. It takes `&T` and is called fresh on every scan (not `FnMut`
|
||||
/// with running state), so it should judge each message purely on its
|
||||
/// own content.
|
||||
/// `pred` is run while the channel lock is held: keep it cheap and pure,
|
||||
/// and do not call back into this channel from inside it. It is modelled as
|
||||
/// `Fn` (not `FnMut`) deliberately — it is re-run from scratch on every
|
||||
/// scan, so a stateful predicate would observe surprising re-counting.
|
||||
pub fn recv_match<F>(&self, pred: F) -> Result<T, RecvError>
|
||||
where
|
||||
F: Fn(&T) -> bool,
|
||||
{
|
||||
loop {
|
||||
{
|
||||
let mut g = self.inner.lock();
|
||||
if let Some(i) = g.queue.iter().position(&pred) {
|
||||
let mut g = self.inner.lock().unwrap();
|
||||
if let Some(i) = g.queue.iter().position(|v| pred(v)) {
|
||||
// position() found it, so remove() returns Some.
|
||||
crate::preempt::note_message_received();
|
||||
let v = match g.queue.remove(i) {
|
||||
Some(v) => v,
|
||||
None => panic!("smarm: channel queue.remove after position (logic bug)"),
|
||||
};
|
||||
return Ok(v);
|
||||
return Ok(g.queue.remove(i).unwrap());
|
||||
}
|
||||
if g.senders == 0 {
|
||||
// Closed and nothing queued can ever match.
|
||||
return Err(RecvError);
|
||||
}
|
||||
let me = match crate::actor::current_pid() {
|
||||
Some(me) => me,
|
||||
None => panic!("smarm: recv_match() called outside an actor"),
|
||||
};
|
||||
let me = crate::actor::current_pid()
|
||||
.expect("recv_match() called outside an actor");
|
||||
debug_assert!(
|
||||
g.parked_receiver.is_none_or(|(p, _)| p == me),
|
||||
g.parked_receiver.is_none(),
|
||||
"channel has more than one receiver"
|
||||
);
|
||||
g.note_runtime();
|
||||
g.parked_receiver = Some((me, crate::scheduler::begin_wait()));
|
||||
g.parked_receiver = Some(me);
|
||||
crate::te!(crate::trace::Event::RecvPark(me));
|
||||
}
|
||||
// Release the lock before parking: the unparker will need it.
|
||||
// Release the lock before parking — the unparker will need it.
|
||||
crate::scheduler::park_current();
|
||||
crate::te!(crate::trace::Event::RecvWake(
|
||||
match crate::actor::current_pid() {
|
||||
Some(p) => p,
|
||||
None => panic!("smarm: RecvWake outside an actor (core corrupt)"),
|
||||
}
|
||||
));
|
||||
crate::te!(crate::trace::Event::RecvWake(crate::actor::current_pid().unwrap()));
|
||||
}
|
||||
}
|
||||
|
||||
/// The non-blocking counterpart of [`recv_match`](Self::recv_match):
|
||||
/// returns immediately either way. `Ok(Some(v))` if a queued message
|
||||
/// matched `pred` (removed; the rest stay queued in order), `Ok(None)`
|
||||
/// if the channel is open but nothing currently matches, `Err(RecvError)`
|
||||
/// if the channel is closed and nothing matches. Same predicate contract
|
||||
/// as `recv_match`.
|
||||
/// Non-blocking selective receive. `Ok(Some(v))` if a queued message
|
||||
/// matched `pred` (removed, rest left in order), `Ok(None)` if the channel
|
||||
/// is open but nothing matched, `Err(RecvError)` if closed and nothing
|
||||
/// matched. Same predicate contract as [`recv_match`](Self::recv_match).
|
||||
pub fn try_recv_match<F>(&self, pred: F) -> Result<Option<T>, RecvError>
|
||||
where
|
||||
F: Fn(&T) -> bool,
|
||||
{
|
||||
let mut g = self.inner.lock();
|
||||
if let Some(i) = g.queue.iter().position(&pred) {
|
||||
crate::preempt::note_message_received();
|
||||
let v = match g.queue.remove(i) {
|
||||
Some(v) => v,
|
||||
None => panic!("smarm: channel queue.remove after position (logic bug)"),
|
||||
};
|
||||
return Ok(Some(v));
|
||||
let mut g = self.inner.lock().unwrap();
|
||||
if let Some(i) = g.queue.iter().position(|v| pred(v)) {
|
||||
return Ok(Some(g.queue.remove(i).unwrap()));
|
||||
}
|
||||
if g.senders == 0 {
|
||||
return Err(RecvError);
|
||||
@@ -500,14 +198,11 @@ impl<T> Receiver<T> {
|
||||
Ok(None)
|
||||
}
|
||||
|
||||
/// The non-blocking counterpart of [`recv`](Self::recv): returns
|
||||
/// immediately either way. `Ok(Some(v))` if a message was queued,
|
||||
/// `Ok(None)` if the channel is open but currently empty, `Err(RecvError)`
|
||||
/// if the channel is closed and the queue is drained.
|
||||
/// Non-blocking. `Ok(Some(v))` if a message was available, `Ok(None)` if
|
||||
/// the channel is empty but open, `Err(RecvError)` if closed and drained.
|
||||
pub fn try_recv(&self) -> Result<Option<T>, RecvError> {
|
||||
let mut g = self.inner.lock();
|
||||
let mut g = self.inner.lock().unwrap();
|
||||
if let Some(v) = g.queue.pop_front() {
|
||||
crate::preempt::note_message_received();
|
||||
return Ok(Some(v));
|
||||
}
|
||||
if g.senders == 0 {
|
||||
@@ -516,336 +211,3 @@ impl<T> Receiver<T> {
|
||||
Ok(None)
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// TimerTarget: the expiry half of recv_timeout
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
impl<T: Send + 'static> crate::timer::TimerTarget for RawMutex<Inner<T>> {
|
||||
fn on_timeout(&self, pid: Pid, epoch: u32) {
|
||||
// Cancel the wait only if THIS wait (epoch match) is still
|
||||
// registered. If a sender already took `parked_receiver`, the
|
||||
// receiver is waking with a message: message wins, the timer
|
||||
// no-ops. If a later wait by the same receiver is registered, the
|
||||
// epoch mismatches: stale entry, no-op. (unpark_at would fail its
|
||||
// internal check in either case anyway; checking under the lock
|
||||
// keeps the registration bookkeeping exact.)
|
||||
let unpark = {
|
||||
let mut g = self.lock();
|
||||
if g.parked_receiver == Some((pid, epoch)) {
|
||||
g.parked_receiver = None;
|
||||
true
|
||||
} else {
|
||||
false
|
||||
}
|
||||
};
|
||||
// Unpark outside the channel lock: it may take the run-queue lock;
|
||||
// legal under a Channel lock, but pointless to nest.
|
||||
if unpark {
|
||||
crate::scheduler::unpark_at(pid, epoch);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// select: ready-index wait over multiple receivers
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
pub(crate) mod sealed {
|
||||
pub trait Sealed {}
|
||||
}
|
||||
impl<T> sealed::Sealed for Receiver<T> {}
|
||||
|
||||
/// An arm of a [`select`]: something you can wait on alongside other arms
|
||||
/// and be told when it becomes ready. Implemented by [`Receiver`]; sealed
|
||||
/// (cannot be implemented outside this crate), since the registration
|
||||
/// contract below is part of the runtime's internal wake protocol.
|
||||
///
|
||||
/// Contract (all under the arm's own lock): `sel_register` checks-or-
|
||||
/// registers atomically. If the arm is ready it does not register and
|
||||
/// returns `Ok(false)`; otherwise it publishes `(pid, epoch)` where its
|
||||
/// wakers will find it and returns `Ok(true)`. "Ready" means a receive
|
||||
/// would not block: a message is queued, or the arm is closed. `Err` means
|
||||
/// the arm could not register at all (only fd arms can fail; channel
|
||||
/// registration always succeeds), and the wait must be retired and earlier
|
||||
/// eager-cleanup arms unregistered.
|
||||
pub trait Selectable: sealed::Sealed {
|
||||
#[doc(hidden)]
|
||||
fn sel_register(&self, pid: Pid, epoch: u32) -> std::io::Result<bool>;
|
||||
#[doc(hidden)]
|
||||
fn sel_ready(&self) -> bool;
|
||||
/// Remove this arm's `(pid, epoch)` registration if, and only if, it is
|
||||
/// still in place. Default no-op: a losing channel arm's stale
|
||||
/// registration is harmless and self-cleans. Fd arms override this:
|
||||
/// their staleness would otherwise leave the fd unusable for future
|
||||
/// selects, so they need an eager cleanup pass.
|
||||
#[doc(hidden)]
|
||||
fn sel_unregister(&self, _pid: Pid, _epoch: u32) {}
|
||||
/// Whether this arm requires the eager cleanup pass at all. Gates the
|
||||
/// post-wake `sel_unregister` sweep so channel-only selects keep their
|
||||
/// cheap, cleanup-free path.
|
||||
#[doc(hidden)]
|
||||
fn sel_eager_cleanup(&self) -> bool {
|
||||
false
|
||||
}
|
||||
}
|
||||
|
||||
impl<T> Selectable for Receiver<T> {
|
||||
fn sel_register(&self, pid: Pid, epoch: u32) -> std::io::Result<bool> {
|
||||
let mut g = self.inner.lock();
|
||||
if !g.queue.is_empty() || g.senders == 0 {
|
||||
return Ok(false);
|
||||
}
|
||||
debug_assert!(
|
||||
g.parked_receiver.is_none_or(|(p, _)| p == pid),
|
||||
"channel has more than one receiver"
|
||||
);
|
||||
g.note_runtime();
|
||||
g.parked_receiver = Some((pid, epoch));
|
||||
Ok(true)
|
||||
}
|
||||
|
||||
fn sel_ready(&self) -> bool {
|
||||
let g = self.inner.lock();
|
||||
!g.queue.is_empty() || g.senders == 0
|
||||
}
|
||||
}
|
||||
|
||||
/// Wait on several channels at once and return the index of the first one
|
||||
/// that is ready, instead of blocking on just one with [`Receiver::recv`].
|
||||
///
|
||||
/// "Ready" means a receive on that arm would not block: a message is
|
||||
/// queued, or the arm is closed (so the caller's own `try_recv` observes
|
||||
/// the disconnect: a dead arm is something to react to, not something to
|
||||
/// hang on). `select` only tells you which arm is ready; read the actual
|
||||
/// message yourself, typically with [`Receiver::try_recv`] on that arm.
|
||||
///
|
||||
/// A closed arm stays ready forever. Once you have observed its disconnect,
|
||||
/// drop it from the arm set you pass in next time: otherwise, under the
|
||||
/// priority order below, it would win every subsequent call and starve
|
||||
/// every arm listed after it.
|
||||
///
|
||||
/// Arms are checked **in order**: index 0 is the highest priority, both
|
||||
/// when checking immediately and after being woken. This is a deliberate,
|
||||
/// documented guarantee, not an accident of implementation: put a control
|
||||
/// or shutdown channel first so it is always noticed promptly. The
|
||||
/// flip side is that there is **no fairness guarantee**: a busy arm 0 can
|
||||
/// starve arm 1 indefinitely by design.
|
||||
///
|
||||
/// One actor can `select` on a channel and later plain `recv` on it (or
|
||||
/// `select` again on an overlapping set of arms) with no restriction. What
|
||||
/// stays illegal is what was always illegal for a channel: two *different*
|
||||
/// actors receiving on the same one.
|
||||
///
|
||||
/// Panics if `arms` is empty, if called outside an actor, or if an fd arm
|
||||
/// fails to register (see [`try_select`] for the fallible form; a
|
||||
/// channel-only `select` can never fail).
|
||||
pub fn select(arms: &[&dyn Selectable]) -> usize {
|
||||
match try_select(arms) {
|
||||
Ok(i) => i,
|
||||
Err(e) => panic!("smarm: select() fd arm failed to register (use try_select): {e}"),
|
||||
}
|
||||
}
|
||||
|
||||
/// The fallible form of [`select`]: `Err` when an arm fails to register.
|
||||
/// Only fd arms can fail this way (for example, the file descriptor is
|
||||
/// invalid, or something else is already waiting on it); a channel-only
|
||||
/// select can never fail. On `Err` the wait is fully retired and no
|
||||
/// registration is left behind: every arm registered before the failing
|
||||
/// one has been unregistered.
|
||||
pub fn try_select(arms: &[&dyn Selectable]) -> std::io::Result<usize> {
|
||||
assert!(!arms.is_empty(), "select() on an empty arm list");
|
||||
let me = match crate::actor::current_pid() {
|
||||
Some(me) => me,
|
||||
None => panic!("smarm: select() called outside an actor"),
|
||||
};
|
||||
loop {
|
||||
let epoch = crate::scheduler::begin_wait();
|
||||
if let Some(i) = register_arms(me, epoch, arms)? {
|
||||
return Ok(i);
|
||||
}
|
||||
|
||||
// Stale fd registrations are not harmless (a losing fd arm's
|
||||
// leftover registration can make the fd unusable for the next
|
||||
// select until a kernel event happens to clear it), so selects
|
||||
// containing fd arms run an eager cleanup pass after the park,
|
||||
// including when a terminal stop unwinds out of it, via the guard.
|
||||
// Channel-only selects skip all of it: `eager` is false, the guard
|
||||
// is disarmed, and the loser-arm self-cleaning story is unchanged.
|
||||
let eager = arms.iter().any(|a| a.sel_eager_cleanup());
|
||||
let mut guard = UnregisterGuard {
|
||||
arms,
|
||||
me,
|
||||
epoch,
|
||||
armed: eager,
|
||||
};
|
||||
|
||||
crate::scheduler::park_current();
|
||||
|
||||
if eager {
|
||||
unregister_arms(arms, me, epoch);
|
||||
}
|
||||
guard.armed = false;
|
||||
drop(guard);
|
||||
|
||||
// Woken precisely: an arm's send (message) or last-sender drop
|
||||
// (closure) is what woke us, and both leave their arm ready.
|
||||
// Return the first ready one, in priority order (which may be a
|
||||
// different, higher-priority arm than the one that woke us; its
|
||||
// message stays queued and re-reports ready on the next call).
|
||||
// Fd arms classify by a fresh zero-timeout poll, so they too are
|
||||
// a pure function of current state, independent of the
|
||||
// registration the cleanup pass just removed.
|
||||
for (i, arm) in arms.iter().enumerate() {
|
||||
if arm.sel_ready() {
|
||||
return Ok(i);
|
||||
}
|
||||
}
|
||||
// Unreachable in practice (a stop wake unwinds out of
|
||||
// park_current before we get here). Defensive: re-open the wait
|
||||
// and re-register; stale own-registrations are overwritten
|
||||
// (channels) or were removed by the cleanup pass above (fds).
|
||||
}
|
||||
}
|
||||
|
||||
/// Eager-cleanup sweep: remove every fd arm's registration that is still
|
||||
/// ours. No-op per channel arm (one virtual call); one io-lock visit per
|
||||
/// fd arm.
|
||||
fn unregister_arms(arms: &[&dyn Selectable], me: Pid, epoch: u32) {
|
||||
for arm in arms {
|
||||
if arm.sel_eager_cleanup() {
|
||||
arm.sel_unregister(me, epoch);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Stop-unwind twin of the explicit cleanup pass: a terminal stop unwinds
|
||||
// out of `park_current`, and a registered fd arm must not outlive its
|
||||
// actor. Disarmed on the normal path after the explicit pass runs; never
|
||||
// armed when no fd arm is registered, keeping the channel-only path
|
||||
// guard-free in effect.
|
||||
struct UnregisterGuard<'a> {
|
||||
arms: &'a [&'a dyn Selectable],
|
||||
me: Pid,
|
||||
epoch: u32,
|
||||
armed: bool,
|
||||
}
|
||||
|
||||
impl Drop for UnregisterGuard<'_> {
|
||||
fn drop(&mut self) {
|
||||
if self.armed {
|
||||
unregister_arms(self.arms, self.me, self.epoch);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// The registration pass shared by `select` and `select_timeout`: check-or-
|
||||
// register each arm, in priority order, each atomically under its own lock.
|
||||
// Cross-arm atomicity is unnecessary: an arm becoming ready right after its
|
||||
// registration still wakes the caller through the normal wake path.
|
||||
//
|
||||
// `Ok(Some(i))` = arm `i` was already ready, the pass stopped, and the wait
|
||||
// has been fully retired (no park may follow): earlier fd arms are
|
||||
// unregistered eagerly so none are left dangling. `Err` = an arm failed to
|
||||
// register; same unwind (earlier fd arms unregistered, wait retired).
|
||||
// `Ok(None)` = every arm registered successfully; the caller parks.
|
||||
fn register_arms(me: Pid, epoch: u32, arms: &[&dyn Selectable]) -> std::io::Result<Option<usize>> {
|
||||
for (i, arm) in arms.iter().enumerate() {
|
||||
let registered = match arm.sel_register(me, epoch) {
|
||||
Ok(r) => r,
|
||||
Err(e) => {
|
||||
unregister_arms(&arms[..i], me, epoch);
|
||||
crate::scheduler::retire_wait();
|
||||
return Err(e);
|
||||
}
|
||||
};
|
||||
if !registered {
|
||||
unregister_arms(&arms[..i], me, epoch);
|
||||
crate::scheduler::retire_wait();
|
||||
return Ok(Some(i));
|
||||
}
|
||||
}
|
||||
Ok(None)
|
||||
}
|
||||
|
||||
// The `select_timeout` timer target: stateless, because a wake's cause can
|
||||
// always be read back off plain channel state (an arm ready, or not). If
|
||||
// an arm already won before the deadline, this timer's fire is simply
|
||||
// ignored, the way any other stale wakeup is.
|
||||
struct SelectTimeout;
|
||||
impl crate::timer::TimerTarget for SelectTimeout {
|
||||
fn on_timeout(&self, pid: Pid, epoch: u32) {
|
||||
crate::scheduler::unpark_at(pid, epoch);
|
||||
}
|
||||
}
|
||||
|
||||
/// Like [`select`], but gives up and returns `None` if no arm becomes
|
||||
/// ready before `timeout` elapses.
|
||||
///
|
||||
/// All of `select`'s semantics carry over: arms are still checked in
|
||||
/// priority order, a closed arm is still permanently ready, and there is
|
||||
/// still no fairness guarantee across arms. A message that arrives at
|
||||
/// essentially the same moment the deadline passes still wins, the same
|
||||
/// way [`Receiver::recv_timeout`] resolves that race.
|
||||
///
|
||||
/// `Duration::ZERO` is a valid timeout: it still gives an already-ready arm
|
||||
/// a chance to be reported before falling through to `None`.
|
||||
///
|
||||
/// Panics if `arms` is empty, if called outside an actor, or if an fd arm
|
||||
/// fails to register (see [`try_select_timeout`] for the fallible form; a
|
||||
/// channel-only select can never fail).
|
||||
pub fn select_timeout(arms: &[&dyn Selectable], timeout: std::time::Duration) -> Option<usize> {
|
||||
match try_select_timeout(arms, timeout) {
|
||||
Ok(r) => r,
|
||||
Err(e) => panic!(
|
||||
"smarm: select_timeout() fd arm failed to register (use try_select_timeout): {e}"
|
||||
),
|
||||
}
|
||||
}
|
||||
|
||||
/// The fallible form of [`select_timeout`]: `Err` when an arm fails to
|
||||
/// register (only fd arms can). On `Err` the wait is fully retired and no
|
||||
/// registration is left behind on any arm.
|
||||
pub fn try_select_timeout(
|
||||
arms: &[&dyn Selectable],
|
||||
timeout: std::time::Duration,
|
||||
) -> std::io::Result<Option<usize>> {
|
||||
assert!(!arms.is_empty(), "select_timeout() on an empty arm list");
|
||||
let me = match crate::actor::current_pid() {
|
||||
Some(me) => me,
|
||||
None => panic!("smarm: select_timeout() called outside an actor"),
|
||||
};
|
||||
let epoch = crate::scheduler::begin_wait();
|
||||
if let Some(i) = register_arms(me, epoch, arms)? {
|
||||
return Ok(Some(i)); // ready now: the timer was never armed
|
||||
}
|
||||
|
||||
// Arm the timer after the registration pass, outside every channel
|
||||
// lock (inserting a timer takes the timers lock).
|
||||
let deadline = crate::timer::deadline_from_now(timeout);
|
||||
let target: std::sync::Arc<dyn crate::timer::TimerTarget> = std::sync::Arc::new(SelectTimeout);
|
||||
crate::scheduler::insert_wait_timer(deadline, me, target, epoch);
|
||||
|
||||
// Same eager-cleanup story as `try_select`: a timer win in particular
|
||||
// leaves every fd arm's registration behind, which without this pass
|
||||
// would leave those fds unusable until a kernel event happened to
|
||||
// clear them.
|
||||
let eager = arms.iter().any(|a| a.sel_eager_cleanup());
|
||||
let mut guard = UnregisterGuard {
|
||||
arms,
|
||||
me,
|
||||
epoch,
|
||||
armed: eager,
|
||||
};
|
||||
|
||||
crate::scheduler::park_current();
|
||||
|
||||
if eager {
|
||||
unregister_arms(arms, me, epoch);
|
||||
}
|
||||
guard.armed = false;
|
||||
drop(guard);
|
||||
|
||||
// Woken precisely: an arm (ready below) or the timer (nothing ready).
|
||||
Ok(arms.iter().position(|arm| arm.sel_ready()))
|
||||
}
|
||||
|
||||
-282
@@ -1,282 +0,0 @@
|
||||
//! RFC 010 — clustering (smarm⇄smarm, explicit remote boundary).
|
||||
//!
|
||||
//! c1: feature flag + optional deps. c2: the owned envelope. c3: the
|
||||
//! transport trait (control connection), framed codec, and the TCP +
|
||||
//! loopback impls. c5: the handshake state machine. c6: the connection
|
||||
//! [`manager`] (registry) and per-peer connection actors ([`conn`]), started
|
||||
//! as an explicit supervision subtree, plus the handshake on the
|
||||
//! accept/connect path ([`connect`]). Everything above them lands in later
|
||||
//! chunks.
|
||||
|
||||
pub mod conn;
|
||||
pub mod connect;
|
||||
pub mod connector;
|
||||
pub mod discovery;
|
||||
pub mod envelope;
|
||||
pub mod expose;
|
||||
pub mod handshake;
|
||||
pub mod manager;
|
||||
pub mod membership;
|
||||
pub mod pg;
|
||||
pub mod remote;
|
||||
pub mod transport;
|
||||
|
||||
use std::io;
|
||||
use std::time::{Duration, SystemTime, UNIX_EPOCH};
|
||||
|
||||
use crate::gen_server::{self, GenServerBuilder};
|
||||
use crate::monitor::monitor;
|
||||
use crate::pg::Incarnation;
|
||||
use crate::scheduler::{sleep, spawn, JoinHandle};
|
||||
use crate::supervisor::{ChildSpec, OneForOne, Restart};
|
||||
|
||||
use envelope::NodeMeta;
|
||||
use handshake::Local;
|
||||
use transport::tcp::TcpTransport;
|
||||
use transport::Transport;
|
||||
|
||||
pub use conn::{spawn_established, ConnHandle};
|
||||
pub use connect::{dial, spawn_acceptor, AcceptorHandle};
|
||||
pub use connector::{spawn_connector, ConnectorHandle};
|
||||
pub use discovery::{Discovery, StaticSeeds, Strategy};
|
||||
pub use envelope::RemoteDownReason;
|
||||
pub use expose::{expose, expose_type, type_hash, DeliverError};
|
||||
pub use manager::{Manager, MANAGER};
|
||||
pub use membership::{subscribe, view, MembershipEvents, NodeEvent, NodeInfo};
|
||||
pub use pg::{dispatch_any, members_all, pick_any, DispatchAnyError, GroupMember, PgMsg, PG_NAME};
|
||||
pub use remote::{
|
||||
demonitor_remote, monitor_remote, send_to_remote, NotConnected, RemoteDown, RemoteMonitor,
|
||||
RemoteName, RemotePid, RemoteSendError, ToRemoteError,
|
||||
};
|
||||
|
||||
/// c6d — the derived build hash for [`handshake::LocalNode::build_hash`]:
|
||||
/// two builds may mesh only when this matches, and it is a pure function of
|
||||
/// the compile-time inputs that define wire compatibility today — the exact
|
||||
/// toolchain (`rustc -V`), the declared feature set, and
|
||||
/// [`envelope::PROTO_VERSION`]. FNV-1a 64 over the build-script string, then
|
||||
/// the proto version folded byte-wise, so a proto bump moves the hash even
|
||||
/// on an identical toolchain. The domain is deliberately lean and
|
||||
/// tightenable later without a wire change — it is just a `u64`.
|
||||
pub const BUILD_HASH: u64 = fold_u32(
|
||||
fnv1a64(env!("SMARM_BUILD_HASH_INPUTS").as_bytes()),
|
||||
envelope::PROTO_VERSION,
|
||||
);
|
||||
|
||||
/// FNV-1a 64 (const so [`BUILD_HASH`] is a compile-time fact).
|
||||
const fn fnv1a64(bytes: &[u8]) -> u64 {
|
||||
let mut h: u64 = 0xcbf2_9ce4_8422_2325;
|
||||
let mut i = 0;
|
||||
while i < bytes.len() {
|
||||
h ^= bytes[i] as u64;
|
||||
h = h.wrapping_mul(0x0000_0100_0000_01b3);
|
||||
i += 1;
|
||||
}
|
||||
h
|
||||
}
|
||||
|
||||
/// Continue an FNV-1a state over a `u32`'s little-endian bytes.
|
||||
const fn fold_u32(mut h: u64, v: u32) -> u64 {
|
||||
let b = v.to_le_bytes();
|
||||
let mut i = 0;
|
||||
while i < b.len() {
|
||||
h ^= b[i] as u64;
|
||||
h = h.wrapping_mul(0x0000_0100_0000_01b3);
|
||||
i += 1;
|
||||
}
|
||||
h
|
||||
}
|
||||
|
||||
/// The control-plane timing knobs, all with today's fixed values as
|
||||
/// defaults ([`Timing::default`]). One struct threaded explicitly to the
|
||||
/// acceptor, the dial path, every connection actor and the connector — no
|
||||
/// ambient state, so a test can run a fast mesh without touching globals.
|
||||
/// Every node in a mesh should agree on `heartbeat_interval` <
|
||||
/// `liveness_timeout`; nothing enforces it.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub struct Timing {
|
||||
/// Idle-connection heartbeat pace. Default [`conn::HEARTBEAT_INTERVAL`].
|
||||
pub heartbeat_interval: Duration,
|
||||
/// Inbound silence that tears a connection down. Default
|
||||
/// [`conn::LIVENESS_TIMEOUT`].
|
||||
pub liveness_timeout: Duration,
|
||||
/// Per-frame handshake deadline on the accept/dial path. Default
|
||||
/// [`connect::HANDSHAKE_TIMEOUT`].
|
||||
pub handshake_timeout: Duration,
|
||||
/// Connector redial delay after the first failure. Default
|
||||
/// [`connector::INITIAL_BACKOFF`].
|
||||
pub initial_backoff: Duration,
|
||||
/// Connector redial delay cap. Default [`connector::MAX_BACKOFF`].
|
||||
pub max_backoff: Duration,
|
||||
}
|
||||
|
||||
impl Default for Timing {
|
||||
fn default() -> Self {
|
||||
Timing {
|
||||
heartbeat_interval: conn::HEARTBEAT_INTERVAL,
|
||||
liveness_timeout: conn::LIVENESS_TIMEOUT,
|
||||
handshake_timeout: connect::HANDSHAKE_TIMEOUT,
|
||||
initial_backoff: connector::INITIAL_BACKOFF,
|
||||
max_backoff: connector::MAX_BACKOFF,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// How to run this node: its identity and how it finds peers.
|
||||
pub struct Config {
|
||||
/// This node's claimed name — the mesh-wide identity peers dial by and
|
||||
/// the tie-break input. Must be unique across the mesh.
|
||||
pub node_name: String,
|
||||
/// Metadata offered in this node's `Hello`.
|
||||
pub meta: NodeMeta,
|
||||
/// The control-connection listen address (e.g. `"127.0.0.1:0"`; the
|
||||
/// concrete bound address is [`Cluster::local_addr`]).
|
||||
pub listen_addr: String,
|
||||
/// The peer-discovery strategy — [`StaticSeeds`] until richer ones land.
|
||||
pub strategy: Box<dyn Strategy>,
|
||||
/// Heartbeat / liveness / handshake / backoff knobs; [`Timing::default`]
|
||||
/// is the shipping configuration.
|
||||
pub timing: Timing,
|
||||
}
|
||||
|
||||
/// A running cluster node: the supervised [`Manager`], the acceptor over the
|
||||
/// bound listener, and the connector driving its [`Strategy`]. Roles will
|
||||
/// eventually mount this; until the role mechanism lands it is started by
|
||||
/// hand (RFC 010 §7).
|
||||
///
|
||||
/// Dropping the handle stops the acceptor and connector loops (no new
|
||||
/// connections in either direction) but detaches the manager subtree, which
|
||||
/// — with every established connection — keeps running for the life of the
|
||||
/// runtime, the same split as [`AcceptorHandle`] alone.
|
||||
pub struct Cluster {
|
||||
_sup: JoinHandle,
|
||||
acceptor: AcceptorHandle,
|
||||
connector: ConnectorHandle,
|
||||
local: Local,
|
||||
}
|
||||
|
||||
impl Cluster {
|
||||
/// The concrete bound listen address, dialable as-is.
|
||||
pub fn local_addr(&self) -> &str {
|
||||
self.acceptor.local_addr()
|
||||
}
|
||||
|
||||
/// This node's handshake identity (name, incarnation, build hash, meta).
|
||||
pub fn local(&self) -> &Local {
|
||||
&self.local
|
||||
}
|
||||
|
||||
/// Stop accepting and dialing. Established connections stay up (they
|
||||
/// belong to the manager); tear those down via the manager.
|
||||
pub fn shutdown(&self) {
|
||||
self.acceptor.shutdown();
|
||||
self.connector.shutdown();
|
||||
}
|
||||
}
|
||||
|
||||
/// Start a cluster node: the supervised manager (blocking until it is
|
||||
/// registered and ready to answer), the acceptor bound per
|
||||
/// [`Config::listen_addr`], and the connector running [`Config::strategy`].
|
||||
/// The node's identity is completed here: `incarnation` is
|
||||
/// [`self_incarnation`] and `build_hash` is [`BUILD_HASH`] — c7 is its first
|
||||
/// consumer. Errs only if the listener cannot bind.
|
||||
///
|
||||
/// The manager is a supervised child (restarted on crash); per-peer
|
||||
/// connection actors are dynamic and monitored by the manager rather than
|
||||
/// statically supervised — a lost connection is re-established by the
|
||||
/// connector's dial loop, never resurrected onto a stale socket.
|
||||
pub fn start(config: Config) -> io::Result<Cluster> {
|
||||
let sup = spawn(|| {
|
||||
OneForOne::new()
|
||||
.child(ChildSpec::new(Restart::Permanent, manager_child))
|
||||
.run()
|
||||
});
|
||||
while gen_server::whereis_server(MANAGER).is_none() {
|
||||
sleep(Duration::from_millis(1));
|
||||
}
|
||||
let local = Local {
|
||||
node_name: config.node_name,
|
||||
incarnation: self_incarnation(),
|
||||
build_hash: BUILD_HASH,
|
||||
meta: config.meta,
|
||||
};
|
||||
// The wire identity serialized pids are stamped with (c10).
|
||||
remote::set_local_identity(&local.node_name, local.incarnation);
|
||||
// The pg actor (Phase 5): subscribes membership, owns the "pg" name.
|
||||
pg::attach_cluster();
|
||||
let listener = TcpTransport.listen(&config.listen_addr)?;
|
||||
let acceptor = spawn_acceptor(listener, local.clone(), config.timing);
|
||||
let connector = spawn_connector(
|
||||
Box::new(TcpTransport),
|
||||
local.clone(),
|
||||
config.strategy,
|
||||
config.timing,
|
||||
);
|
||||
Ok(Cluster {
|
||||
_sup: sup,
|
||||
acceptor,
|
||||
connector,
|
||||
local,
|
||||
})
|
||||
}
|
||||
|
||||
/// This process's incarnation epoch: milliseconds since the Unix epoch,
|
||||
/// truncated to `u32`. Not a clock — its one job is separating a node from
|
||||
/// its own restart (two starts of the same name land on the same value only
|
||||
/// if they happen within the same millisecond modulo ~49.7 days). Seconds
|
||||
/// would be too coarse: a crash-and-restart inside one second is routine
|
||||
/// under supervision.
|
||||
pub fn self_incarnation() -> Incarnation {
|
||||
let ms = SystemTime::now()
|
||||
.duration_since(UNIX_EPOCH)
|
||||
.map(|d| d.as_millis())
|
||||
.unwrap_or(0);
|
||||
Incarnation::new(ms as u32)
|
||||
}
|
||||
|
||||
/// The supervised manager child body. It *is* the child actor: it starts the
|
||||
/// named manager, then parks on the manager's own termination so this actor's
|
||||
/// lifetime tracks the manager's — the supervisor's restart accounting keys off
|
||||
/// this actor exiting.
|
||||
fn manager_child() {
|
||||
let m = match GenServerBuilder::new(Manager::new()).named(MANAGER).start() {
|
||||
Ok(m) => m,
|
||||
// Name still held by a not-yet-reaped prior instance: return and let
|
||||
// the supervisor retry under its restart policy.
|
||||
Err(_) => return,
|
||||
};
|
||||
let _ = monitor(m.pid()).rx.recv();
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
/// The hash core against the published FNV-1a 64 test vectors — the
|
||||
/// contract is "this is FNV-1a", not "whatever the fn does".
|
||||
#[test]
|
||||
fn fnv1a64_known_vectors() {
|
||||
assert_eq!(fnv1a64(b""), 0xcbf2_9ce4_8422_2325);
|
||||
assert_eq!(fnv1a64(b"a"), 0xaf63_dc4c_8601_ec8c);
|
||||
assert_eq!(fnv1a64(b"foobar"), 0x85944171f73967e8);
|
||||
}
|
||||
|
||||
/// Folding the proto version continues the same FNV state: identical
|
||||
/// inputs with a different version must land on a different hash.
|
||||
#[test]
|
||||
fn proto_version_moves_the_hash() {
|
||||
let base = fnv1a64(b"same-toolchain;features=CLUSTER");
|
||||
assert_ne!(fold_u32(base, 1), fold_u32(base, 2));
|
||||
// And it equals hashing the bytes in one pass — the fold is a
|
||||
// continuation, not a second construction.
|
||||
let mut all = b"same-toolchain;features=CLUSTER".to_vec();
|
||||
all.extend_from_slice(&1u32.to_le_bytes());
|
||||
assert_eq!(fold_u32(base, 1), fnv1a64(&all));
|
||||
}
|
||||
|
||||
/// The derived constant exists, is compile-time, and is not degenerate.
|
||||
#[test]
|
||||
fn build_hash_is_nonzero() {
|
||||
const H: u64 = BUILD_HASH;
|
||||
assert_ne!(H, 0);
|
||||
}
|
||||
}
|
||||
@@ -1,630 +0,0 @@
|
||||
//! RFC 010 c6 — the per-peer connection actor.
|
||||
//!
|
||||
//! One actor per established control connection. It owns the whole
|
||||
//! [`FramedConn`] and, in a single [`select`](crate::select), waits on two
|
||||
//! things at once: its command inbox and the connection becoming readable (the
|
||||
//! [`FdArm`](crate::scheduler::FdArm) the transport hands back). That is why it
|
||||
//! is a plain select-loop actor rather than a `gen_server` or `gen_statem` —
|
||||
//! neither of those can fold fd-readiness into its wait, and folding it in is
|
||||
//! the whole job. The single owner sends and receives on the one `FramedConn`,
|
||||
//! so no read/write split is needed.
|
||||
//!
|
||||
//! The handshake completes *before* this actor exists (on the accept/connect
|
||||
//! path — c6b) and produces the [`Peer`]; the *path* then registers the
|
||||
//! connection with the [`manager`](crate::cluster::manager), which takes
|
||||
//! ownership of its [`ConnHandle`] and monitors the actor, so any exit
|
||||
//! deregisters the connection. The actor itself holds no authority over its
|
||||
//! own lifetime: it runs until the manager drops its handle (deregistration,
|
||||
//! `Disconnect`, or manager shutdown), the connection ends, or liveness
|
||||
//! expires. Heartbeat send and fixed-timeout liveness are the timeout arm of
|
||||
//! the same `select` (c6c): [`HEARTBEAT_INTERVAL`] paces outbound
|
||||
//! [`Frame::Heartbeat`](crate::cluster::envelope::Frame::Heartbeat)s, and a
|
||||
//! [`LIVENESS_TIMEOUT`] window — reset by any inbound frame — tears the
|
||||
//! connection down when it empties.
|
||||
//!
|
||||
//! c9 adds the third arm — the connection's dedicated **outbound inbox**
|
||||
//! (`Sender<Frame>` bound in the manager-maintained outbound table, D13),
|
||||
//! drained onto the wire in the same loop — and inbound *interpretation*:
|
||||
//! `SendNamed` goes to the one resolution seam,
|
||||
//! [`remote::deliver_named`](crate::cluster::remote::deliver_named).
|
||||
//! `Send` goes to the pid seam (c10). The outbound
|
||||
//! sender is a separate channel from `cmd_tx` on purpose: closing it is not
|
||||
//! a stop signal — lifetime authority stays with the [`ConnHandle`] (D9).
|
||||
//!
|
||||
//! c12 adds the monitor plane, and it lives *here* on purpose. Two tables,
|
||||
//! both owned by this actor and dying with the connection:
|
||||
//!
|
||||
//! - **outstanding** — monitors *this* node holds on actors at the peer:
|
||||
//! `monitor_id → (target, Sender<RemoteDown>)`. Fed by
|
||||
//! [`MonCmd`](crate::cluster::remote::MonCmd) from `monitor_remote`; the
|
||||
//! actor records the id and *then* emits the `Monitor` frame, so a `Down`
|
||||
//! frame can never race an entry that isn't there yet. An inbound `Down`
|
||||
//! removes the entry and delivers.
|
||||
//! - **watched** — monitors the *peer* holds on actors here: `monitor_id →
|
||||
//! local Monitor`. An inbound `Monitor` is admitted only for a pid that
|
||||
//! was exposed or crossed the wire (`is_watchable`, D12): a corpse answers
|
||||
//! with its recorded terminal reason (RFC §6), an unwatchable or unknown
|
||||
//! pid with `NoProc` — indistinguishable from dead, so nothing leaks. A
|
||||
//! live watchable pid gets a local monitor whose `rx` is one more arm of
|
||||
//! the select; its `Down` goes back as a frame.
|
||||
//!
|
||||
//! Because both tables are actor state, connection loss (c13) needs no
|
||||
//! second bookkeeping owner: this actor's exit is the one place that knows
|
||||
//! every monitor the link was carrying. `Monitors::teardown` runs on every
|
||||
//! exit path and answers each outstanding monitor with `Disconnected` —
|
||||
//! the roadmap's "partition vs. death" contrast: an actor that dies sends
|
||||
//! its true reason over the link, a link that dies says only that.
|
||||
|
||||
use std::collections::HashMap;
|
||||
|
||||
use std::time::{Duration, Instant};
|
||||
|
||||
use crate::channel::{channel, try_select_timeout, Receiver, Selectable, Sender};
|
||||
use crate::cluster::envelope::{Frame, RemoteDownReason};
|
||||
use crate::cluster::handshake::Peer;
|
||||
use crate::cluster::manager::{Call, Registered, Reply, MANAGER};
|
||||
use crate::cluster::remote::{
|
||||
deliver_named, deliver_to_pid, InboundVerdict, MonCmd, RemoteDown, RemotePid,
|
||||
};
|
||||
use crate::cluster::transport::FramedConn;
|
||||
use crate::cluster::Timing;
|
||||
use crate::gen_server;
|
||||
use crate::monitor::{
|
||||
demonitor, is_watchable, monitor, terminal_reason, DownReason, Monitor, MonitorId,
|
||||
};
|
||||
use crate::pid::{Erased, Pid};
|
||||
use crate::scheduler::spawn;
|
||||
|
||||
/// Commands to a running connection actor.
|
||||
enum Cmd {
|
||||
Shutdown,
|
||||
}
|
||||
|
||||
/// The manager's authority over one connection actor: while this handle
|
||||
/// lives the connection lives, and dropping it stops the actor and closes
|
||||
/// the socket. Only the [`manager`](crate::cluster::manager) holds one —
|
||||
/// callers of [`spawn_established`] get a [`Pid`] and no lifetime authority,
|
||||
/// so a connection can never outlive, or die with, whichever actor happened
|
||||
/// to establish it.
|
||||
pub struct ConnHandle {
|
||||
cmd_tx: Sender<Cmd>,
|
||||
/// The connection's dedicated outbound inboxes — frames and monitor
|
||||
/// commands. The manager moves them into the outbound table on
|
||||
/// `Register` (see [`take_outbound`](ConnHandle::take_outbound)); a
|
||||
/// `Duplicate` verdict drops them with the handle.
|
||||
out_tx: Option<(Sender<Frame>, Sender<MonCmd>)>,
|
||||
}
|
||||
|
||||
impl std::fmt::Debug for ConnHandle {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
f.write_str("ConnHandle")
|
||||
}
|
||||
}
|
||||
|
||||
impl ConnHandle {
|
||||
/// Ask the connection to close and exit. Idempotent, and a no-op if the
|
||||
/// actor has already gone. Dropping the handle does the same thing; this
|
||||
/// exists for the manager's explicit `Disconnect` path.
|
||||
pub fn shutdown(&self) {
|
||||
let _ = self.cmd_tx.send(Cmd::Shutdown);
|
||||
}
|
||||
|
||||
/// Manager-only: take the outbound senders to bind into the outbound
|
||||
/// table. Once, at registration.
|
||||
pub(crate) fn take_outbound(&mut self) -> Option<(Sender<Frame>, Sender<MonCmd>)> {
|
||||
self.out_tx.take()
|
||||
}
|
||||
}
|
||||
|
||||
/// The name was already claimed by a live connection, so this one was
|
||||
/// refused; its actor has been stopped and its socket closed.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub struct RegisterRefused;
|
||||
|
||||
/// Spawn a connection actor for an **already-established** connection (the
|
||||
/// handshake completed on the path and produced `peer`) and register it with
|
||||
/// the manager, synchronously, before returning. The manager takes the
|
||||
/// actor's [`ConnHandle`]; the caller gets only the [`Pid`], because
|
||||
/// connection lifetime belongs to the table and not to the establishing
|
||||
/// actor. A refusal has already stopped the actor and closed the socket.
|
||||
pub fn spawn_established(
|
||||
framed: FramedConn,
|
||||
peer: Peer,
|
||||
timing: Timing,
|
||||
) -> Result<Pid, RegisterRefused> {
|
||||
let (cmd_tx, cmd_rx) = channel();
|
||||
let (out_tx, out_rx) = channel();
|
||||
let (mon_tx, mon_rx) = channel();
|
||||
let reg_peer = peer.clone();
|
||||
let pid = spawn(move || run(framed, peer, timing, cmd_rx, out_rx, mon_rx)).pid();
|
||||
match gen_server::call(
|
||||
MANAGER,
|
||||
Call::Register {
|
||||
peer: reg_peer,
|
||||
pid,
|
||||
handle: ConnHandle {
|
||||
cmd_tx,
|
||||
out_tx: Some((out_tx, mon_tx)),
|
||||
},
|
||||
},
|
||||
) {
|
||||
Ok(Reply::Registered(Registered::Ok)) => Ok(pid),
|
||||
// Duplicate name, or the manager is unreachable. Either way the
|
||||
// handle went with the call and is dropped there (or never arrived
|
||||
// and dropped with it), which stops the actor and closes the socket.
|
||||
_ => Err(RegisterRefused),
|
||||
}
|
||||
}
|
||||
|
||||
/// Default for [`Timing::heartbeat_interval`]: how often this end emits
|
||||
/// [`Frame::Heartbeat`] on an idle connection. The first one goes out
|
||||
/// immediately at spawn, so the peer's liveness window starts fed.
|
||||
pub const HEARTBEAT_INTERVAL: Duration = Duration::from_secs(1);
|
||||
|
||||
/// Default for [`Timing::liveness_timeout`]: how long the connection may go
|
||||
/// without a single inbound frame before it is declared dead and torn down. Any inbound frame resets the window —
|
||||
/// heartbeats keep an idle connection alive, and real traffic (c8+) counts
|
||||
/// for free. Fixed by design (RFC v2 §5): this is the control connection, a
|
||||
/// heartbeat can never queue behind bulk traffic, so a fixed timeout is an
|
||||
/// honest detector.
|
||||
pub const LIVENESS_TIMEOUT: Duration = Duration::from_secs(4);
|
||||
|
||||
fn run(
|
||||
mut framed: FramedConn,
|
||||
_peer: Peer,
|
||||
timing: Timing,
|
||||
cmd_rx: Receiver<Cmd>,
|
||||
out_rx: Receiver<Frame>,
|
||||
mon_rx: Receiver<MonCmd>,
|
||||
) {
|
||||
let mut mons = Monitors::default();
|
||||
match framed.readable_arm() {
|
||||
Some(arm) => run_live(
|
||||
&mut framed,
|
||||
arm,
|
||||
timing,
|
||||
&cmd_rx,
|
||||
&out_rx,
|
||||
&mon_rx,
|
||||
&mut mons,
|
||||
),
|
||||
None => run_inert(&cmd_rx),
|
||||
}
|
||||
framed.close();
|
||||
mons.teardown(&mon_rx);
|
||||
}
|
||||
|
||||
/// The monitor plane's two tables (module docs). Owned by the actor.
|
||||
#[derive(Default)]
|
||||
struct Monitors {
|
||||
/// Monitors this node holds on peer actors: id → (target, delivery).
|
||||
outstanding: HashMap<MonitorId, (RemotePid<Erased>, Sender<RemoteDown>)>,
|
||||
/// Monitors the peer holds on local actors: id → the local monitor.
|
||||
watched: HashMap<MonitorId, Monitor>,
|
||||
}
|
||||
|
||||
impl Monitors {
|
||||
/// The connection is gone, whatever the exit path (liveness expiry,
|
||||
/// EOF, wire failure, commanded stop): release the peer's local
|
||||
/// monitors, and answer every one of ours with `Disconnected` — nothing
|
||||
/// more can be known about those actors. Commands still sitting in the
|
||||
/// inbox are folded in first (a `Monitor` handed to us but never
|
||||
/// processed gets its notice too; a `Demonitor` still cancels), so the
|
||||
/// only registration that can miss this is one that lands after the
|
||||
/// drain and before the inbox drops — the reader side backstops that
|
||||
/// (`RemoteMonitor`). Entries leave the table as they are answered, and
|
||||
/// this runs once per actor, so no monitor sees two notices.
|
||||
fn teardown(&mut self, mon_rx: &Receiver<MonCmd>) {
|
||||
for (_, m) in self.watched.drain() {
|
||||
let _ = demonitor(&m);
|
||||
}
|
||||
while let Ok(Some(cmd)) = mon_rx.try_recv() {
|
||||
match cmd {
|
||||
MonCmd::Monitor { id, target, tx } => {
|
||||
self.outstanding.insert(id, (target, tx));
|
||||
}
|
||||
MonCmd::Demonitor { id } => {
|
||||
self.outstanding.remove(&id);
|
||||
}
|
||||
}
|
||||
}
|
||||
for (_, (pid, tx)) in self.outstanding.drain() {
|
||||
let _ = tx.send(RemoteDown {
|
||||
pid,
|
||||
reason: RemoteDownReason::Disconnected,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
/// Admit a peer's `Monitor` for local `(index, generation)`. Returns
|
||||
/// the reason to answer with at once, or `None` if a live monitor was
|
||||
/// installed. Corpse → recorded terminal reason (RFC §6, and only
|
||||
/// watchable deaths are recorded); live watchable → monitor; anything
|
||||
/// else → `NoProc`. The check-then-monitor race (dies in between) is
|
||||
/// closed on the read side: a `NoProc` from a monitor we installed on a
|
||||
/// live pid is upgraded through `terminal_reason` in `sweep_watched`.
|
||||
fn admit(&mut self, id: MonitorId, index: u32, generation: u32) -> Option<DownReason> {
|
||||
let pid = Pid::new(index, generation);
|
||||
if let Some(reason) = terminal_reason(pid) {
|
||||
return Some(reason);
|
||||
}
|
||||
if !is_watchable(pid) {
|
||||
return Some(DownReason::NoProc);
|
||||
}
|
||||
let m = monitor(pid);
|
||||
self.watched.insert(id, m);
|
||||
None
|
||||
}
|
||||
|
||||
fn cancel(&mut self, id: MonitorId) {
|
||||
if let Some(m) = self.watched.remove(&id) {
|
||||
let _ = demonitor(&m);
|
||||
}
|
||||
}
|
||||
|
||||
/// Collect every local `Down` that has arrived for a peer-held monitor.
|
||||
fn sweep_watched(&mut self) -> Vec<(MonitorId, DownReason)> {
|
||||
let mut fired = Vec::new();
|
||||
for (id, m) in self.watched.iter() {
|
||||
if let Ok(Some(down)) = m.rx.try_recv() {
|
||||
let reason = match down.reason {
|
||||
DownReason::NoProc => terminal_reason(m.target).unwrap_or(DownReason::NoProc),
|
||||
r => r,
|
||||
};
|
||||
fired.push((*id, reason));
|
||||
}
|
||||
}
|
||||
for (id, _) in &fired {
|
||||
self.watched.remove(id);
|
||||
}
|
||||
fired
|
||||
}
|
||||
|
||||
/// The peer reports a monitored actor down: deliver locally.
|
||||
fn down(&mut self, id: MonitorId, reason: RemoteDownReason) {
|
||||
if let Some((pid, tx)) = self.outstanding.remove(&id) {
|
||||
let _ = tx.send(RemoteDown { pid, reason });
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// The steady-state loop over an fd-backed connection: one
|
||||
/// `select_timeout` folds the command inbox, the outbound inbox, socket
|
||||
/// readability, and the nearer of the two deadlines (`hb_send`,
|
||||
/// `liveness`) into a single wait.
|
||||
fn run_live(
|
||||
framed: &mut FramedConn,
|
||||
arm: crate::scheduler::FdArm,
|
||||
timing: Timing,
|
||||
cmd_rx: &Receiver<Cmd>,
|
||||
out_rx: &Receiver<Frame>,
|
||||
mon_rx: &Receiver<MonCmd>,
|
||||
mons: &mut Monitors,
|
||||
) {
|
||||
let mut next_hb = Instant::now();
|
||||
let mut live_until = Instant::now() + timing.liveness_timeout;
|
||||
// The outbound senders live in the manager's table and are dropped on
|
||||
// unbind; after that these arms would wake forever, so they drop out of
|
||||
// the select (not a stop signal — see the module docs).
|
||||
let mut out_open = true;
|
||||
let mut mon_open = true;
|
||||
// Which wait each select arm stands for. Built in lockstep with the
|
||||
// `Selectable` vector each iteration, so a wake is decoded by name and
|
||||
// never by position.
|
||||
enum Arm {
|
||||
Cmd,
|
||||
Fd,
|
||||
Out,
|
||||
Mon,
|
||||
/// A peer-held local monitor (any of them: firing sweeps them all).
|
||||
Watched,
|
||||
}
|
||||
fn push<'s>(
|
||||
arms: &mut Vec<&'s dyn Selectable>,
|
||||
what: &mut Vec<Arm>,
|
||||
s: &'s dyn Selectable,
|
||||
a: Arm,
|
||||
) {
|
||||
arms.push(s);
|
||||
what.push(a);
|
||||
}
|
||||
loop {
|
||||
let now = Instant::now();
|
||||
if now >= live_until {
|
||||
break; // liveness expired: the peer is dead to us
|
||||
}
|
||||
if now >= next_hb {
|
||||
if framed.send(&Frame::Heartbeat).is_err() {
|
||||
break;
|
||||
}
|
||||
next_hb = now + timing.heartbeat_interval;
|
||||
}
|
||||
let wait = next_hb.min(live_until).saturating_duration_since(now);
|
||||
let mut arms: Vec<&dyn Selectable> = Vec::new();
|
||||
let mut what: Vec<Arm> = Vec::new();
|
||||
push(&mut arms, &mut what, cmd_rx, Arm::Cmd);
|
||||
push(&mut arms, &mut what, &arm, Arm::Fd);
|
||||
if out_open {
|
||||
push(&mut arms, &mut what, out_rx, Arm::Out);
|
||||
}
|
||||
if mon_open {
|
||||
push(&mut arms, &mut what, mon_rx, Arm::Mon);
|
||||
}
|
||||
for m in mons.watched.values() {
|
||||
push(&mut arms, &mut what, &m.rx, Arm::Watched);
|
||||
}
|
||||
match try_select_timeout(&arms, wait).map(|i| i.map(|i| &what[i])) {
|
||||
Ok(Some(Arm::Cmd)) => {
|
||||
if should_stop(cmd_rx) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
Ok(Some(Arm::Fd)) => match pump_readable(framed, mons) {
|
||||
Pump::Ended => break,
|
||||
Pump::Frames(n) => {
|
||||
if n > 0 {
|
||||
live_until = Instant::now() + timing.liveness_timeout;
|
||||
}
|
||||
}
|
||||
},
|
||||
Ok(Some(Arm::Out)) => match pump_outbound(framed, out_rx) {
|
||||
Outbound::Drained => {}
|
||||
Outbound::Closed => out_open = false,
|
||||
Outbound::WireFailed => break,
|
||||
},
|
||||
Ok(Some(Arm::Mon)) => match pump_moncmds(framed, mon_rx, mons) {
|
||||
Outbound::Drained => {}
|
||||
Outbound::Closed => mon_open = false,
|
||||
Outbound::WireFailed => break,
|
||||
},
|
||||
Ok(Some(Arm::Watched)) => {
|
||||
for (id, reason) in mons.sweep_watched() {
|
||||
let frame = Frame::Down {
|
||||
monitor_id: id.0,
|
||||
reason: reason.into(),
|
||||
};
|
||||
if framed.send(&frame).is_err() {
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
// A deadline passed; the top of the loop acts on whichever.
|
||||
Ok(None) => {}
|
||||
// The fd arm failed to register — the connection is gone.
|
||||
Err(_) => break,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Drain the monitor-command inbox: record, then emit (module docs).
|
||||
fn pump_moncmds(
|
||||
framed: &mut FramedConn,
|
||||
mon_rx: &Receiver<MonCmd>,
|
||||
mons: &mut Monitors,
|
||||
) -> Outbound {
|
||||
loop {
|
||||
match mon_rx.try_recv() {
|
||||
Ok(Some(MonCmd::Monitor { id, target, tx })) => {
|
||||
let frame = Frame::Monitor {
|
||||
monitor_id: id.0,
|
||||
index: target.index(),
|
||||
generation: target.generation(),
|
||||
};
|
||||
mons.outstanding.insert(id, (target, tx));
|
||||
if framed.send(&frame).is_err() {
|
||||
return Outbound::WireFailed;
|
||||
}
|
||||
}
|
||||
Ok(Some(MonCmd::Demonitor { id })) => {
|
||||
if mons.outstanding.remove(&id).is_some()
|
||||
&& framed.send(&Frame::Demonitor { monitor_id: id.0 }).is_err()
|
||||
{
|
||||
return Outbound::WireFailed;
|
||||
}
|
||||
}
|
||||
Ok(None) => return Outbound::Drained,
|
||||
Err(_) => return Outbound::Closed,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// What one outbound-side wake (frames or monitor commands) yielded.
|
||||
enum Outbound {
|
||||
/// Everything queued went onto the wire; the inbox is open and empty.
|
||||
Drained,
|
||||
/// The manager unbound this connection's sender; nothing more will come.
|
||||
Closed,
|
||||
/// The socket refused a write: the connection is gone.
|
||||
WireFailed,
|
||||
}
|
||||
|
||||
/// Drain every queued outbound frame onto the wire.
|
||||
fn pump_outbound(framed: &mut FramedConn, out_rx: &Receiver<Frame>) -> Outbound {
|
||||
loop {
|
||||
match out_rx.try_recv() {
|
||||
Ok(Some(frame)) => {
|
||||
if framed.send(&frame).is_err() {
|
||||
return Outbound::WireFailed;
|
||||
}
|
||||
}
|
||||
Ok(None) => return Outbound::Drained,
|
||||
Err(_) => return Outbound::Closed,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// No fd to select on (loopback): only a command can end the wait, and
|
||||
/// neither heartbeats nor liveness run — a transport that can't report
|
||||
/// readiness can't be timed either (same caveat as
|
||||
/// [`FramedConn::recv_deadline`]). Loopback is a test transport; every real
|
||||
/// connection is fd-backed.
|
||||
fn run_inert(cmd_rx: &Receiver<Cmd>) {
|
||||
loop {
|
||||
let arms: [&dyn Selectable; 1] = [cmd_rx];
|
||||
let _ = crate::channel::select(&arms);
|
||||
if should_stop(cmd_rx) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Drain the command arm. Returns `true` when the actor should exit — a
|
||||
/// shutdown was requested, or the last handle was dropped.
|
||||
fn should_stop(cmd_rx: &Receiver<Cmd>) -> bool {
|
||||
match cmd_rx.try_recv() {
|
||||
Ok(Some(Cmd::Shutdown)) => true,
|
||||
Ok(None) => false, // spurious wake
|
||||
Err(_) => true, // all senders dropped
|
||||
}
|
||||
}
|
||||
|
||||
/// What one readable wake yielded.
|
||||
enum Pump {
|
||||
/// The connection has ended: EOF (clean or mid-frame) or an
|
||||
/// unrecoverable stream error.
|
||||
Ended,
|
||||
/// Still up; this many complete frames were consumed (possibly zero, if
|
||||
/// the wake delivered only part of a frame). Any nonzero count resets
|
||||
/// the liveness window.
|
||||
Frames(usize),
|
||||
}
|
||||
|
||||
/// Surface an inbound verdict: one `smarm-trace` event, nothing else — it
|
||||
/// is local knowledge (RFC §3). A no-op without the feature.
|
||||
fn note_verdict(verdict: InboundVerdict) {
|
||||
#[cfg(feature = "smarm-trace")]
|
||||
crate::te!(crate::trace::Event::ClusterInbound(verdict.label()));
|
||||
#[cfg(not(feature = "smarm-trace"))]
|
||||
drop(verdict);
|
||||
}
|
||||
|
||||
/// Consume one readable wake: exactly one socket read (which cannot block
|
||||
/// after a level-triggered readable indication), then drain every complete
|
||||
/// frame the buffer now holds. A blocking `recv` here would park the actor
|
||||
/// past its heartbeat and liveness deadlines whenever a frame arrives split.
|
||||
/// Every consumed frame counts for liveness; `SendNamed` goes to the one
|
||||
/// name-resolution seam and `Send` to the pid seam. Verdicts are local
|
||||
/// knowledge only — nothing goes back on the wire (RFC §3) — and surface
|
||||
/// as one `smarm-trace` `ClusterInbound` event each (zero cost off).
|
||||
/// `Monitor`/`Demonitor`/`Down` go to the [`Monitors`] tables; a `Monitor`
|
||||
/// that can be answered at once is answered inline.
|
||||
fn pump_readable(framed: &mut FramedConn, mons: &mut Monitors) -> Pump {
|
||||
let eof = match framed.read_once() {
|
||||
Ok(n) => n == 0,
|
||||
Err(_) => return Pump::Ended,
|
||||
};
|
||||
let mut got = 0;
|
||||
loop {
|
||||
match framed.next_buffered() {
|
||||
Ok(Some(frame)) => {
|
||||
got += 1;
|
||||
match frame {
|
||||
Frame::SendNamed {
|
||||
name,
|
||||
type_hash,
|
||||
payload,
|
||||
} => {
|
||||
note_verdict(deliver_named(&name, type_hash, &payload));
|
||||
}
|
||||
Frame::Send {
|
||||
index,
|
||||
generation,
|
||||
type_hash,
|
||||
payload,
|
||||
} => {
|
||||
note_verdict(deliver_to_pid(index, generation, type_hash, &payload));
|
||||
}
|
||||
Frame::Monitor {
|
||||
monitor_id,
|
||||
index,
|
||||
generation,
|
||||
} => {
|
||||
let id = MonitorId(monitor_id);
|
||||
if let Some(reason) = mons.admit(id, index, generation) {
|
||||
let frame = Frame::Down {
|
||||
monitor_id,
|
||||
reason: reason.into(),
|
||||
};
|
||||
if framed.send(&frame).is_err() {
|
||||
return Pump::Ended;
|
||||
}
|
||||
}
|
||||
}
|
||||
Frame::Demonitor { monitor_id } => mons.cancel(MonitorId(monitor_id)),
|
||||
Frame::Down { monitor_id, reason } => mons.down(MonitorId(monitor_id), reason),
|
||||
// Heartbeat: liveness only. Handshake frames after
|
||||
// establishment: ignored.
|
||||
_ => {}
|
||||
}
|
||||
}
|
||||
Ok(None) => break,
|
||||
Err(_) => return Pump::Ended, // corrupt stream
|
||||
}
|
||||
}
|
||||
if eof {
|
||||
Pump::Ended
|
||||
} else {
|
||||
Pump::Frames(got)
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
//! `Monitors::teardown` in isolation: the actor-side half of c13, pinned
|
||||
//! separately because from the outside it is indistinguishable from the
|
||||
//! read-side backstop in `RemoteMonitor` (both yield `Disconnected`).
|
||||
use super::*;
|
||||
use crate::pg::Incarnation;
|
||||
|
||||
fn pid(index: u32) -> RemotePid<Erased> {
|
||||
RemotePid::from_parts("peer", Incarnation::new(1), index, 1)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn teardown_answers_every_outstanding_and_unread_monitor_once() {
|
||||
crate::run(|| {
|
||||
let mut mons = Monitors::default();
|
||||
let (mon_tx, mon_rx) = channel::<MonCmd>();
|
||||
|
||||
// Already registered.
|
||||
let (tx1, rx1) = channel::<RemoteDown>();
|
||||
mons.outstanding.insert(MonitorId(1), (pid(1), tx1));
|
||||
// In the inbox, never processed.
|
||||
let (tx2, rx2) = channel::<RemoteDown>();
|
||||
mon_tx
|
||||
.send(MonCmd::Monitor {
|
||||
id: MonitorId(2),
|
||||
target: pid(2),
|
||||
tx: tx2,
|
||||
})
|
||||
.ok()
|
||||
.unwrap();
|
||||
// Registered, then cancelled in the inbox: silence.
|
||||
let (tx3, rx3) = channel::<RemoteDown>();
|
||||
mons.outstanding.insert(MonitorId(3), (pid(3), tx3));
|
||||
mon_tx
|
||||
.send(MonCmd::Demonitor { id: MonitorId(3) })
|
||||
.ok()
|
||||
.unwrap();
|
||||
|
||||
mons.teardown(&mon_rx);
|
||||
|
||||
let d1 = rx1.recv().unwrap();
|
||||
assert_eq!(
|
||||
(d1.pid, d1.reason),
|
||||
(pid(1), RemoteDownReason::Disconnected)
|
||||
);
|
||||
let d2 = rx2.recv().unwrap();
|
||||
assert_eq!(
|
||||
(d2.pid, d2.reason),
|
||||
(pid(2), RemoteDownReason::Disconnected)
|
||||
);
|
||||
// Cancelled: no notice was sent (its sender is dropped, channel
|
||||
// closed-empty), and nobody got a second one.
|
||||
assert!(rx3.try_recv().is_err());
|
||||
assert!(rx1.try_recv().is_err());
|
||||
assert!(rx2.try_recv().is_err());
|
||||
assert!(mons.outstanding.is_empty());
|
||||
});
|
||||
}
|
||||
}
|
||||
@@ -1,365 +0,0 @@
|
||||
//! RFC 010 c6b — the handshake on the accept/connect path.
|
||||
//!
|
||||
//! Per D8 (re-amended): the c5 machines are driven by **straight-line code
|
||||
//! on the path**, not by an actor. The dial side runs [`Initiator`]; the
|
||||
//! acceptor loop runs [`Responder`]. A connection actor is spawned only
|
||||
//! *after* a successful handshake ([`spawn_established`]); every reject,
|
||||
//! protocol failure, timeout, and tie-break loss is resolved right here,
|
||||
//! on the path, by closing — no actor ever exists for a connection that
|
||||
//! didn't establish.
|
||||
//!
|
||||
//! Buffer trap (binding): the path reader and the steady-state actor share
|
||||
//! ONE [`FramedConn`]. Its decode buffer may hold read-ahead past the
|
||||
//! handshake frames, so the *whole* `FramedConn` travels into
|
||||
//! [`spawn_established`] — never a fresh codec over the same socket.
|
||||
//!
|
||||
//! Layering: [`dial_handshake`] and [`accept_handshake`] are the bare path
|
||||
//! steps — IO on a `FramedConn`, no manager, no actors — testable over the
|
||||
//! loopback transport on plain threads. [`dial`] and [`spawn_acceptor`] are
|
||||
//! the manager-integrated layer (actor context required): they keep the
|
||||
//! [`manager`](crate::cluster::manager)'s dial-intent set honest and spawn
|
||||
//! the connection actor on success.
|
||||
|
||||
use std::io;
|
||||
use std::time::{Duration, Instant};
|
||||
|
||||
use crate::channel::{channel, Receiver, Selectable, Sender};
|
||||
use crate::cluster::conn::spawn_established;
|
||||
use crate::cluster::envelope::{Frame, RejectReason};
|
||||
use crate::cluster::handshake::{
|
||||
Initiator, InitiatorOutcome, Local, Peer, PeerStanding, Responder, ResponderOutcome,
|
||||
};
|
||||
use crate::cluster::manager::{Call, Reply, MANAGER};
|
||||
use crate::cluster::transport::{FramedConn, Listener, RecvError, SendError, Transport};
|
||||
use crate::cluster::Timing;
|
||||
use crate::gen_server;
|
||||
use crate::pid::Pid;
|
||||
use crate::scheduler::{self, spawn};
|
||||
|
||||
/// Default for [`Timing::handshake_timeout`]: how long either side waits for
|
||||
/// the peer's handshake frame before giving up and closing. Enforced on the path via [`FramedConn::recv_deadline`],
|
||||
/// so a peer that connects and goes silent cannot wedge the acceptor.
|
||||
pub const HANDSHAKE_TIMEOUT: Duration = Duration::from_secs(5);
|
||||
|
||||
/// Why a handshake did not establish. In every case the connection has
|
||||
/// already been closed on the path by the time this is returned.
|
||||
#[derive(Debug)]
|
||||
pub enum HandshakeError {
|
||||
/// A `HelloReject` travelled — sent by us (accept side) or received by
|
||||
/// us (dial side).
|
||||
Rejected(RejectReason),
|
||||
/// Accept side only: the inbound dial lost the simultaneous-connect
|
||||
/// tie-break (D7) and was closed silently, no frame sent.
|
||||
TieBreakLoss,
|
||||
/// The peer spoke a valid frame that is wrong here (non-`Hello` first
|
||||
/// frame; non-response to our `Hello`), or an undecodable byte stream.
|
||||
Protocol,
|
||||
/// EOF before the handshake resolved. On the dial side this is also
|
||||
/// what losing the tie-break looks like: the peer closes silently.
|
||||
Closed,
|
||||
/// [`HANDSHAKE_TIMEOUT`] (or the caller's deadline) passed first.
|
||||
TimedOut,
|
||||
/// The transport failed mid-handshake.
|
||||
Transport(io::Error),
|
||||
}
|
||||
|
||||
fn from_send(e: SendError) -> HandshakeError {
|
||||
match e {
|
||||
// Handshake frames are small and self-made; an encode failure is a
|
||||
// protocol-level impossibility, not a transport fault.
|
||||
SendError::Encode(_) => HandshakeError::Protocol,
|
||||
SendError::Io(e) => HandshakeError::Transport(e),
|
||||
}
|
||||
}
|
||||
|
||||
fn from_recv(e: RecvError) -> HandshakeError {
|
||||
match e {
|
||||
RecvError::Corrupt(_) => HandshakeError::Protocol,
|
||||
RecvError::TruncatedByPeer => HandshakeError::Closed,
|
||||
RecvError::Io(e) => HandshakeError::Transport(e),
|
||||
RecvError::TimedOut => HandshakeError::TimedOut,
|
||||
}
|
||||
}
|
||||
|
||||
/// Dial-side path step: send our `Hello`, interpret the one response. On
|
||||
/// `Ok` the connection is established and `framed` is live (with any
|
||||
/// read-ahead intact in its buffer); on `Err` the connection is closed.
|
||||
pub fn dial_handshake(
|
||||
framed: &mut FramedConn,
|
||||
local: &Local,
|
||||
deadline: Instant,
|
||||
) -> Result<Peer, HandshakeError> {
|
||||
let (initiator, hello) = Initiator::new(local);
|
||||
if let Err(e) = framed.send(&hello) {
|
||||
framed.close();
|
||||
return Err(from_send(e));
|
||||
}
|
||||
let outcome = match framed.recv_deadline(deadline) {
|
||||
Ok(Some(frame)) => initiator.on_frame(frame),
|
||||
Ok(None) => {
|
||||
framed.close();
|
||||
return Err(HandshakeError::Closed);
|
||||
}
|
||||
Err(e) => {
|
||||
framed.close();
|
||||
return Err(from_recv(e));
|
||||
}
|
||||
};
|
||||
match outcome {
|
||||
InitiatorOutcome::Established(peer) => Ok(peer),
|
||||
InitiatorOutcome::Rejected(reason) => {
|
||||
framed.close();
|
||||
Err(HandshakeError::Rejected(reason))
|
||||
}
|
||||
InitiatorOutcome::Failed(_) => {
|
||||
framed.close();
|
||||
Err(HandshakeError::Protocol)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Accept-side path step: read the first frame, judge it, answer or close.
|
||||
///
|
||||
/// `standing_of` supplies the [`PeerStanding`] of the *offered* name — knowledge
|
||||
/// only the frame reveals, which is why it is a callback and not a value
|
||||
/// (the integrated acceptor asks the manager; loopback tests fabricate).
|
||||
/// It is not called when the first frame is not a `Hello`.
|
||||
///
|
||||
/// On `Ok` the ack has been sent and `framed` is live (read-ahead intact);
|
||||
/// on `Err` any owed reject has been sent and the connection is closed.
|
||||
pub fn accept_handshake(
|
||||
framed: &mut FramedConn,
|
||||
local: Local,
|
||||
standing_of: impl FnOnce(&str) -> PeerStanding,
|
||||
deadline: Instant,
|
||||
) -> Result<Peer, HandshakeError> {
|
||||
let frame = match framed.recv_deadline(deadline) {
|
||||
Ok(Some(frame)) => frame,
|
||||
Ok(None) => {
|
||||
framed.close();
|
||||
return Err(HandshakeError::Closed);
|
||||
}
|
||||
Err(e) => {
|
||||
framed.close();
|
||||
return Err(from_recv(e));
|
||||
}
|
||||
};
|
||||
let standing = match &frame {
|
||||
Frame::Hello { node_name, .. } => standing_of(node_name),
|
||||
_ => PeerStanding::Free,
|
||||
};
|
||||
match Responder::new(local).on_frame(frame, standing) {
|
||||
ResponderOutcome::Accepted { reply, peer } => {
|
||||
if let Err(e) = framed.send(&reply) {
|
||||
framed.close();
|
||||
return Err(from_send(e));
|
||||
}
|
||||
Ok(peer)
|
||||
}
|
||||
ResponderOutcome::Rejected { reply, reason } => {
|
||||
// Best effort: the reject is the cross-version compatibility
|
||||
// anchor, but if the write fails the peer sees a bare close,
|
||||
// which it must survive anyway.
|
||||
let _ = framed.send(&reply);
|
||||
framed.close();
|
||||
Err(HandshakeError::Rejected(reason))
|
||||
}
|
||||
ResponderOutcome::TieBreakLoss => {
|
||||
// D7: close silently — the peer computes the same verdict.
|
||||
framed.close();
|
||||
Err(HandshakeError::TieBreakLoss)
|
||||
}
|
||||
ResponderOutcome::Failed(_) => {
|
||||
framed.close();
|
||||
Err(HandshakeError::Protocol)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Manager-integrated layer
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Why an integrated [`dial`] did not produce a connection.
|
||||
#[derive(Debug)]
|
||||
pub enum DialError {
|
||||
/// Another dial to this peer name is already in flight.
|
||||
AlreadyDialing,
|
||||
/// The manager is not running (or answered nonsense).
|
||||
ManagerUnavailable,
|
||||
/// The transport could not connect.
|
||||
Connect(io::Error),
|
||||
/// Connected, but the handshake did not establish.
|
||||
Handshake(HandshakeError),
|
||||
/// The peer at `addr` established, but answered as a different name
|
||||
/// than the one we dialed — the tie-break bookkeeping (keyed by the
|
||||
/// dialed name) would be unsound, so the connection is closed.
|
||||
PeerNameMismatch { expected: String, got: String },
|
||||
/// The handshake established, but the manager refused the registration:
|
||||
/// a connection to this peer already exists. The loser has been closed.
|
||||
Duplicate,
|
||||
}
|
||||
|
||||
impl DialError {
|
||||
/// A short static label per kind, for the `smarm-trace` `ClusterDial`
|
||||
/// event; the payload (io error, names) is not carried.
|
||||
pub fn label(&self) -> &'static str {
|
||||
match self {
|
||||
DialError::AlreadyDialing => "already_dialing",
|
||||
DialError::ManagerUnavailable => "manager_unavailable",
|
||||
DialError::Connect(_) => "connect",
|
||||
DialError::Handshake(_) => "handshake",
|
||||
DialError::PeerNameMismatch { .. } => "peer_name_mismatch",
|
||||
DialError::Duplicate => "duplicate",
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Dial `peer_name` at `addr` and run the handshake, keeping the manager's
|
||||
/// dial-intent set honest around it: the intent is registered *before*
|
||||
/// connecting (so a crossing inbound `Hello` sees it) and cleared the
|
||||
/// moment the handshake resolves, before the connection actor is spawned.
|
||||
/// Must run inside an actor. Retrying is the caller's business (c7's dial
|
||||
/// loop); a lost tie-break surfaces as `Handshake(Closed)` — the peer's
|
||||
/// accepted connection is already on its way.
|
||||
pub fn dial(
|
||||
transport: &dyn Transport,
|
||||
addr: &str,
|
||||
peer_name: &str,
|
||||
local: &Local,
|
||||
timing: Timing,
|
||||
) -> Result<Pid, DialError> {
|
||||
let me = scheduler::self_pid();
|
||||
match gen_server::call(
|
||||
MANAGER,
|
||||
Call::DialBegin {
|
||||
name: peer_name.to_string(),
|
||||
pid: me,
|
||||
},
|
||||
) {
|
||||
Ok(Reply::DialBegan(true)) => {}
|
||||
Ok(Reply::DialBegan(false)) => return Err(DialError::AlreadyDialing),
|
||||
_ => return Err(DialError::ManagerUnavailable),
|
||||
}
|
||||
let result = connect_and_shake(transport, addr, local, timing);
|
||||
// Cleared immediately on outcome — a stale intent during the established
|
||||
// window would corrupt later tie-breaks. Synchronous (a call): the
|
||||
// intent is provably gone before anything else happens.
|
||||
let _ = gen_server::call(
|
||||
MANAGER,
|
||||
Call::DialEnd {
|
||||
name: peer_name.to_string(),
|
||||
},
|
||||
);
|
||||
let (mut framed, peer) = result?;
|
||||
if peer.node_name != peer_name {
|
||||
framed.close();
|
||||
return Err(DialError::PeerNameMismatch {
|
||||
expected: peer_name.to_string(),
|
||||
got: peer.node_name,
|
||||
});
|
||||
}
|
||||
spawn_established(framed, peer, timing).map_err(|_| DialError::Duplicate)
|
||||
}
|
||||
|
||||
fn connect_and_shake(
|
||||
transport: &dyn Transport,
|
||||
addr: &str,
|
||||
local: &Local,
|
||||
timing: Timing,
|
||||
) -> Result<(FramedConn, Peer), DialError> {
|
||||
let conn = transport.dial(addr).map_err(DialError::Connect)?;
|
||||
let mut framed = FramedConn::new(conn);
|
||||
let deadline = Instant::now() + timing.handshake_timeout;
|
||||
let peer = dial_handshake(&mut framed, local, deadline).map_err(DialError::Handshake)?;
|
||||
Ok((framed, peer))
|
||||
}
|
||||
|
||||
/// A running acceptor. [`shutdown`](AcceptorHandle::shutdown) (or dropping
|
||||
/// the last handle) stops the accept loop only: connections it established
|
||||
/// belong to the [`manager`](crate::cluster::manager) and keep running, to
|
||||
/// be torn down through the table (`Disconnect`, a peer close, or manager
|
||||
/// shutdown).
|
||||
pub struct AcceptorHandle {
|
||||
cmd_tx: Sender<()>,
|
||||
addr: String,
|
||||
}
|
||||
|
||||
impl AcceptorHandle {
|
||||
/// Ask the acceptor to stop. Idempotent; a no-op if it already has.
|
||||
pub fn shutdown(&self) {
|
||||
let _ = self.cmd_tx.send(());
|
||||
}
|
||||
|
||||
/// The concrete bound address, dialable as-is.
|
||||
pub fn local_addr(&self) -> &str {
|
||||
&self.addr
|
||||
}
|
||||
}
|
||||
|
||||
/// Spawn the acceptor actor over a bound listener. Each inbound connection
|
||||
/// is handshaken **inline in the loop** (a deliberate serialization: the
|
||||
/// per-frame deadline bounds how long any one peer can hold the line, and
|
||||
/// nothing concurrent exists to be starved before c7). The listener must be
|
||||
/// fd-backed ([`Listener::readable_arm`]); the loopback listener is not,
|
||||
/// and its acceptor exits immediately — loopback handshakes are driven
|
||||
/// synchronously through the path fns instead, per D8.
|
||||
pub fn spawn_acceptor(listener: Box<dyn Listener>, local: Local, timing: Timing) -> AcceptorHandle {
|
||||
let addr = listener.local_addr();
|
||||
let (cmd_tx, cmd_rx) = channel();
|
||||
spawn(move || accept_loop(listener, local, timing, cmd_rx));
|
||||
AcceptorHandle { cmd_tx, addr }
|
||||
}
|
||||
|
||||
fn accept_loop(
|
||||
mut listener: Box<dyn Listener>,
|
||||
local: Local,
|
||||
timing: Timing,
|
||||
cmd_rx: Receiver<()>,
|
||||
) {
|
||||
loop {
|
||||
let Some(arm) = listener.readable_arm() else {
|
||||
return;
|
||||
};
|
||||
let arms: [&dyn Selectable; 2] = [&cmd_rx, &arm];
|
||||
match crate::channel::try_select(&arms) {
|
||||
Ok(0) => match cmd_rx.try_recv() {
|
||||
Ok(Some(())) => return,
|
||||
Ok(None) => continue, // spurious wake
|
||||
Err(_) => return, // all handles dropped
|
||||
},
|
||||
Ok(_) => {
|
||||
// The listener is readable: accept completes without parking.
|
||||
let conn = match listener.accept() {
|
||||
Ok(conn) => conn,
|
||||
Err(_) => return, // listener itself is broken
|
||||
};
|
||||
handle_inbound(FramedConn::new(conn), &local, timing);
|
||||
}
|
||||
Err(_) => return, // fd arm failed to register: listener is gone
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Run the accept-side handshake for one inbound connection, asking the
|
||||
/// manager for the [`PeerStanding`], and hand the established connection to the
|
||||
/// manager. Every failure was already resolved on the path (reject sent /
|
||||
/// closed, or the registration refused and the actor stopped), so there is
|
||||
/// nothing for the acceptor to carry forward.
|
||||
fn handle_inbound(mut framed: FramedConn, local: &Local, timing: Timing) {
|
||||
let deadline = Instant::now() + timing.handshake_timeout;
|
||||
let standing_of = |name: &str| match gen_server::call(
|
||||
MANAGER,
|
||||
Call::Standing {
|
||||
peer_name: name.to_string(),
|
||||
},
|
||||
) {
|
||||
Ok(Reply::Standing(s)) => s,
|
||||
// Manager unreachable: nobody could register this connection anyway,
|
||||
// so claim the name taken and reject rather than accept an orphan.
|
||||
_ => PeerStanding::Claimed,
|
||||
};
|
||||
if let Ok(peer) = accept_handshake(&mut framed, local.clone(), standing_of, deadline) {
|
||||
let _ = spawn_established(framed, peer, timing);
|
||||
}
|
||||
}
|
||||
@@ -1,316 +0,0 @@
|
||||
//! RFC 010 c7b — the connector: the dial loop that turns discovered
|
||||
//! candidates into a full mesh.
|
||||
//!
|
||||
//! A plain select-loop actor (the c6 shape). It spawns its [`Strategy`] as a
|
||||
//! child actor and receives [`Discovery`] events from it; it tracks which
|
||||
//! peers are up by **subscribing to membership like any other consumer** —
|
||||
//! no privileged channel into the manager, the same snapshot-then-stream
|
||||
//! surface c8 will use. One `select` folds the command inbox, the discovery
|
||||
//! stream, the membership stream, and the earliest retry deadline into a
|
||||
//! single wait.
|
||||
//!
|
||||
//! Per-candidate state: dial on arrival; on failure retry with capped
|
||||
//! exponential backoff ([`INITIAL_BACKOFF`] doubling to [`MAX_BACKOFF`]);
|
||||
//! on the peer's `node_up` stop dialing and reset the backoff; on its
|
||||
//! `node_down` resume immediately (a fresh sequence — the reconnect case is
|
||||
//! the one backoff exists to pace, but the *first* retry after a death
|
||||
//! should be prompt). A candidate bearing our own name is parked permanently
|
||||
//! — that seed is us; so is one whose address answers as a different name
|
||||
//! (`PeerNameMismatch`: a misconfigured or stale seed — each retry would
|
||||
//! only blip the peer's membership). Every other failure retries: in
|
||||
//! particular a `NameTaken` reject can be our own ghost at the peer, not
|
||||
//! yet reaped by its liveness timer, so it must not park. Each attempt's
|
||||
//! outcome is one `smarm-trace` `ClusterDial` event. A [`Discovery::Withdrawn`]
|
||||
//! drops its `(name, addr)` from the dial set — only that: a live
|
||||
//! connection is membership's, and a re-announce re-adds it fresh.
|
||||
//!
|
||||
//! Dials run **inline in the loop** — the same deliberate serialization as
|
||||
//! the acceptor (c6b): each attempt is bounded by the connect + handshake
|
||||
//! deadlines, and nothing concurrent exists to be starved. A wall of slow
|
||||
//! unreachable seeds would stretch the loop's latency; revisit if a real
|
||||
//! deployment ever hits that shape.
|
||||
|
||||
use std::collections::HashSet;
|
||||
use std::time::{Duration, Instant};
|
||||
|
||||
use crate::channel::{channel, select, select_timeout, Receiver, Selectable, Sender};
|
||||
use crate::cluster::connect::{dial, DialError};
|
||||
use crate::cluster::discovery::{Discovery, Strategy};
|
||||
use crate::cluster::handshake::Local;
|
||||
use crate::cluster::membership::{subscribe, NodeEvent};
|
||||
use crate::cluster::transport::Transport;
|
||||
use crate::cluster::Timing;
|
||||
use crate::scheduler::spawn;
|
||||
|
||||
/// Default for [`Timing::initial_backoff`]: first retry delay after a failed
|
||||
/// dial attempt.
|
||||
pub const INITIAL_BACKOFF: Duration = Duration::from_millis(250);
|
||||
/// Default for [`Timing::max_backoff`]: an unreachable seed is retried this
|
||||
/// often, forever.
|
||||
pub const MAX_BACKOFF: Duration = Duration::from_secs(5);
|
||||
|
||||
enum Cmd {
|
||||
Shutdown,
|
||||
}
|
||||
|
||||
/// A running connector. `shutdown` (or dropping the last handle) stops the
|
||||
/// dial loop and its strategy only — established connections belong to the
|
||||
/// manager, exactly as with the acceptor.
|
||||
pub struct ConnectorHandle {
|
||||
cmd_tx: Sender<Cmd>,
|
||||
}
|
||||
|
||||
impl ConnectorHandle {
|
||||
/// Ask the connector to stop. Idempotent; a no-op if it already has.
|
||||
pub fn shutdown(&self) {
|
||||
let _ = self.cmd_tx.send(Cmd::Shutdown);
|
||||
}
|
||||
}
|
||||
|
||||
/// One discovered `(name, addr)` and our dial intent towards it.
|
||||
struct Candidate {
|
||||
name: String,
|
||||
addr: String,
|
||||
state: State,
|
||||
}
|
||||
|
||||
/// The connector's *intent* for a candidate. Whether the peer is currently
|
||||
/// up is a separate, name-keyed membership fact (`up` in [`run`]): a
|
||||
/// candidate can arrive after its peer's `node_up` (the snapshot lands
|
||||
/// before the strategy has said anything), so "up" cannot live on the
|
||||
/// candidate alone — it is a filter over dialing, not a candidate state.
|
||||
enum State {
|
||||
/// Never dialed: this seed is the local node itself, or the address
|
||||
/// answered as a *different* name than the one seeded
|
||||
/// (`DialError::PeerNameMismatch` — a misconfigured or stale seed;
|
||||
/// redialing would only blip the peer's membership forever). The way
|
||||
/// back is the strategy's: `Withdrawn` then a fresh `Candidate`.
|
||||
Parked,
|
||||
/// Dial when due; on failure, back off.
|
||||
Dialing {
|
||||
/// Delay to apply after the *next* failure.
|
||||
backoff: Duration,
|
||||
next_attempt: Instant,
|
||||
},
|
||||
}
|
||||
|
||||
impl State {
|
||||
fn fresh(timing: &Timing) -> Self {
|
||||
State::Dialing {
|
||||
backoff: timing.initial_backoff,
|
||||
next_attempt: Instant::now(),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl Candidate {
|
||||
/// The retry deadline, if this candidate is dialing at all.
|
||||
fn due(&self) -> Option<Instant> {
|
||||
match self.state {
|
||||
State::Parked => None,
|
||||
State::Dialing { next_attempt, .. } => Some(next_attempt),
|
||||
}
|
||||
}
|
||||
/// A dial attempt was made: schedule the retry, grow the backoff.
|
||||
fn attempted(&mut self, timing: &Timing) {
|
||||
if let State::Dialing {
|
||||
backoff,
|
||||
next_attempt,
|
||||
} = &mut self.state
|
||||
{
|
||||
*next_attempt = Instant::now() + *backoff;
|
||||
*backoff = (*backoff * 2).min(timing.max_backoff);
|
||||
}
|
||||
}
|
||||
/// The peer came up: the next sequence (after a later `node_down`)
|
||||
/// starts from the initial delay again.
|
||||
fn peer_up(&mut self, timing: &Timing) {
|
||||
if let State::Dialing { backoff, .. } = &mut self.state {
|
||||
*backoff = timing.initial_backoff;
|
||||
}
|
||||
}
|
||||
/// The peer went down: redial promptly, fresh sequence.
|
||||
fn peer_down(&mut self, timing: &Timing) {
|
||||
if matches!(self.state, State::Dialing { .. }) {
|
||||
self.state = State::fresh(timing);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Spawn the connector actor. The strategy is spawned as its child; the
|
||||
/// membership subscription is taken inside the actor. Must be called from
|
||||
/// inside an actor (the same requirement as `dial`).
|
||||
pub fn spawn_connector(
|
||||
transport: Box<dyn Transport>,
|
||||
local: Local,
|
||||
strategy: Box<dyn Strategy>,
|
||||
timing: Timing,
|
||||
) -> ConnectorHandle {
|
||||
let (cmd_tx, cmd_rx) = channel();
|
||||
spawn(move || run(transport, local, strategy, timing, cmd_rx));
|
||||
ConnectorHandle { cmd_tx }
|
||||
}
|
||||
|
||||
fn run(
|
||||
transport: Box<dyn Transport>,
|
||||
local: Local,
|
||||
strategy: Box<dyn Strategy>,
|
||||
timing: Timing,
|
||||
cmd_rx: Receiver<Cmd>,
|
||||
) {
|
||||
// Membership is the connector's source of truth for "who is up" — the
|
||||
// snapshot seeds `up` before any candidate arrives.
|
||||
let Some(events) = subscribe() else {
|
||||
return; // no manager, no cluster to connect
|
||||
};
|
||||
let (disc_tx, disc_rx) = channel();
|
||||
spawn(move || strategy.run(disc_tx));
|
||||
|
||||
let mut cands: Vec<Candidate> = Vec::new();
|
||||
let mut up: HashSet<String> = HashSet::new();
|
||||
let mut strategy_done = false;
|
||||
|
||||
loop {
|
||||
// Drain every input, then act. Order does not matter: acting is
|
||||
// idempotent against the resulting state.
|
||||
match drain_cmd(&cmd_rx) {
|
||||
Drained::Stop => return,
|
||||
Drained::Open => {}
|
||||
}
|
||||
if !strategy_done {
|
||||
strategy_done = drain_discoveries(&disc_rx, &local, &timing, &mut cands);
|
||||
}
|
||||
match drain_events(&events.rx, &timing, &mut up, &mut cands) {
|
||||
Drained::Stop => return, // manager gone: the cluster is tearing down
|
||||
Drained::Open => {}
|
||||
}
|
||||
|
||||
// Dial everything due, inline (see the module docs on serialization).
|
||||
let now = Instant::now();
|
||||
for c in cands
|
||||
.iter_mut()
|
||||
.filter(|c| !up.contains(&c.name) && c.due().is_some_and(|d| d <= now))
|
||||
{
|
||||
// On success the manager's node_up is on its way and lands in
|
||||
// `up` (backing off meanwhile keeps a racing re-attempt from
|
||||
// spinning); every failure retries — see the module docs —
|
||||
// except a peer-name mismatch, which parks the candidate.
|
||||
let outcome = dial(&*transport, &c.addr, &c.name, &local, timing);
|
||||
note_dial(&outcome);
|
||||
match outcome {
|
||||
Err(DialError::PeerNameMismatch { .. }) => c.state = State::Parked,
|
||||
_ => c.attempted(&timing),
|
||||
}
|
||||
}
|
||||
|
||||
// Wait: until the earliest retry deadline among actionable
|
||||
// candidates, or indefinitely if none is pending.
|
||||
let deadline = cands
|
||||
.iter()
|
||||
.filter(|c| !up.contains(&c.name))
|
||||
.filter_map(Candidate::due)
|
||||
.min();
|
||||
let mut arms: Vec<&dyn Selectable> = vec![&cmd_rx, &events.rx];
|
||||
if !strategy_done {
|
||||
arms.push(&disc_rx);
|
||||
}
|
||||
match deadline {
|
||||
Some(d) => {
|
||||
let wait = d.saturating_duration_since(Instant::now());
|
||||
let _ = select_timeout(&arms, wait);
|
||||
}
|
||||
None => {
|
||||
let _ = select(&arms);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
enum Drained {
|
||||
Open,
|
||||
Stop,
|
||||
}
|
||||
|
||||
fn drain_cmd(rx: &Receiver<Cmd>) -> Drained {
|
||||
match rx.try_recv() {
|
||||
Ok(Some(Cmd::Shutdown)) => Drained::Stop,
|
||||
Ok(None) => Drained::Open,
|
||||
Err(_) => Drained::Stop, // all handles dropped
|
||||
}
|
||||
}
|
||||
|
||||
/// Pull every pending discovery into the candidate set (deduplicated by
|
||||
/// `(name, addr)`; a candidate bearing the local name is parked; a
|
||||
/// `Withdrawn` removes its pair from the dial set and nothing else — see
|
||||
/// [`Discovery::Withdrawn`]). Returns `true` once the strategy's channel
|
||||
/// closes — it has said all it will.
|
||||
fn drain_discoveries(
|
||||
rx: &Receiver<Discovery>,
|
||||
local: &Local,
|
||||
timing: &Timing,
|
||||
cands: &mut Vec<Candidate>,
|
||||
) -> bool {
|
||||
loop {
|
||||
match rx.try_recv() {
|
||||
Ok(Some(Discovery::Withdrawn { name, addr })) => {
|
||||
cands.retain(|c| !(c.name == name && c.addr == addr));
|
||||
}
|
||||
Ok(Some(Discovery::Candidate { name, addr })) => {
|
||||
if cands.iter().any(|c| c.name == name && c.addr == addr) {
|
||||
continue;
|
||||
}
|
||||
let state = if name == local.node_name {
|
||||
State::Parked
|
||||
} else {
|
||||
State::fresh(timing)
|
||||
};
|
||||
cands.push(Candidate { name, addr, state });
|
||||
}
|
||||
Ok(None) => return false,
|
||||
Err(_) => return true, // strategy done; its candidates live on here
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Fold pending membership events into `up`; each is also a transition on
|
||||
/// that peer's candidates (see [`Candidate::peer_up`] / [`peer_down`]).
|
||||
///
|
||||
/// [`peer_down`]: Candidate::peer_down
|
||||
fn drain_events(
|
||||
rx: &Receiver<NodeEvent>,
|
||||
timing: &Timing,
|
||||
up: &mut HashSet<String>,
|
||||
cands: &mut [Candidate],
|
||||
) -> Drained {
|
||||
loop {
|
||||
match rx.try_recv() {
|
||||
Ok(Some(NodeEvent::NodeUp(info))) => {
|
||||
cands
|
||||
.iter_mut()
|
||||
.filter(|c| c.name == info.name)
|
||||
.for_each(|c| c.peer_up(timing));
|
||||
up.insert(info.name);
|
||||
}
|
||||
Ok(Some(NodeEvent::NodeDown(info))) => {
|
||||
up.remove(&info.name);
|
||||
cands
|
||||
.iter_mut()
|
||||
.filter(|c| c.name == info.name)
|
||||
.for_each(|c| c.peer_down(timing));
|
||||
}
|
||||
Ok(None) => return Drained::Open,
|
||||
Err(_) => return Drained::Stop,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Surface a dial outcome: one `smarm-trace` event, nothing else. The
|
||||
/// connector's bookkeeping is decided by the caller.
|
||||
fn note_dial(outcome: &Result<crate::pid::Pid, DialError>) {
|
||||
#[cfg(feature = "smarm-trace")]
|
||||
crate::te!(crate::trace::Event::ClusterDial(
|
||||
outcome.as_ref().map_or_else(DialError::label, |_| "ok")
|
||||
));
|
||||
#[cfg(not(feature = "smarm-trace"))]
|
||||
let _ = outcome;
|
||||
}
|
||||
@@ -1,76 +0,0 @@
|
||||
//! RFC 010 c7b — peer discovery: the [`Strategy`] seam and the static-seeds
|
||||
//! implementation.
|
||||
//!
|
||||
//! A strategy is **push-based and runs as its own actor**: the
|
||||
//! [`connector`](crate::cluster::connector) spawns it with the sending end of
|
||||
//! a channel, and the strategy emits [`Discovery`] events whenever it learns
|
||||
//! something — once at startup for a static list, continuously for a future
|
||||
//! mDNS/DNS strategy — for as long as it cares to run. Returning ends the
|
||||
//! strategy actor; the candidates it pushed live on in the connector (the
|
||||
//! connector owns all retry/backoff state, so a strategy never re-announces).
|
||||
//!
|
||||
//! A candidate is a **`(node_name, addr)` pair**, not a bare address: the
|
||||
//! dial path and the D7 tie-break are keyed by peer *name* (the dial intent
|
||||
//! must be registered before connecting so a crossing inbound `Hello` sees
|
||||
//! it), so an anonymous dial would reintroduce exactly the
|
||||
//! simultaneous-connect flap D7 exists to prevent. Discovery mechanisms know
|
||||
//! names — that is what they discover.
|
||||
|
||||
use crate::channel::Sender;
|
||||
|
||||
/// A discovery event, as pushed by a [`Strategy`].
|
||||
///
|
||||
/// `Candidate` announces, `Withdrawn` retracts — the primitive pair. A
|
||||
/// strategy that wants TTL semantics builds them on top (track its own
|
||||
/// last-seen times, emit `Withdrawn` on expiry); the connector deliberately
|
||||
/// has no clock of its own for candidates (D11: strategies never
|
||||
/// re-announce, the connector owns retry). `#[non_exhaustive]` so more can
|
||||
/// land without breaking strategies.
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
#[non_exhaustive]
|
||||
pub enum Discovery {
|
||||
/// A peer worth dialing: its claimed node name and a dialable address.
|
||||
Candidate { name: String, addr: String },
|
||||
/// Stop dialing this `(name, addr)`. Dial-set only: a connection that
|
||||
/// is already up is membership's business and is left alone; an
|
||||
/// attempt in flight completes on its own; a later `Candidate` for the
|
||||
/// same pair re-adds it with fresh backoff. Unknown pairs are ignored.
|
||||
Withdrawn { name: String, addr: String },
|
||||
}
|
||||
|
||||
/// A source of peers to dial. Implementations are spawned as actors by the
|
||||
/// connector — see the module docs for the contract.
|
||||
pub trait Strategy: Send + 'static {
|
||||
/// Run the strategy: push [`Discovery`] events into `out` as they are
|
||||
/// learned; return when done discovering (or when `out` reports closed —
|
||||
/// the connector is gone). Runs inside an actor, so blocking
|
||||
/// cooperatively is fine.
|
||||
fn run(self: Box<Self>, out: Sender<Discovery>);
|
||||
}
|
||||
|
||||
/// The static-seeds strategy: a fixed `(name, addr)` list, announced once.
|
||||
#[derive(Debug, Clone, Default)]
|
||||
pub struct StaticSeeds {
|
||||
seeds: Vec<(String, String)>,
|
||||
}
|
||||
|
||||
impl StaticSeeds {
|
||||
pub fn new(seeds: impl IntoIterator<Item = (impl Into<String>, impl Into<String>)>) -> Self {
|
||||
StaticSeeds {
|
||||
seeds: seeds
|
||||
.into_iter()
|
||||
.map(|(n, a)| (n.into(), a.into()))
|
||||
.collect(),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl Strategy for StaticSeeds {
|
||||
fn run(self: Box<Self>, out: Sender<Discovery>) {
|
||||
for (name, addr) in self.seeds {
|
||||
if out.send(Discovery::Candidate { name, addr }).is_err() {
|
||||
return; // connector gone; nobody to discover for
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,549 +0,0 @@
|
||||
//! RFC 010 c2 — the owned wire envelope.
|
||||
//!
|
||||
//! Every control-plane frame is `u32` little-endian length prefix (of tag +
|
||||
//! body), `u8` tag, hand-encoded body. postcard appears in exactly one place:
|
||||
//! the payload blob inside `Send`/`SendNamed`, via [`encode_payload`] /
|
||||
//! [`decode_payload`] — the seam where a codec swap would land (RFC 010 §2).
|
||||
//! Everything else is hand-rolled and wholly owned.
|
||||
//!
|
||||
//! Integers are little-endian. Strings are `u16` length + UTF-8 bytes.
|
||||
//! Payload blobs are `u32` length + bytes. Enum-shaped fields
|
||||
//! ([`RejectReason`], [`DownReason`]) are a single tag byte.
|
||||
|
||||
use crate::monitor::DownReason;
|
||||
use crate::pg::Incarnation;
|
||||
|
||||
/// Wire protocol version, checked in the handshake (c5).
|
||||
pub const PROTO_VERSION: u32 = 1;
|
||||
|
||||
/// Hard cap on the length prefix. The control plane never carries bulk data
|
||||
/// (RFC 010 §5 — that is the jarred rkyv plane), so anything larger is
|
||||
/// corruption or an attack, not a legitimate frame.
|
||||
pub const MAX_FRAME_LEN: usize = 16 * 1024 * 1024;
|
||||
|
||||
/// Per-node metadata exchanged in the handshake (RFC 010 §1: not identity).
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct NodeMeta {
|
||||
pub role: String,
|
||||
pub region: String,
|
||||
}
|
||||
|
||||
/// Why a `Hello` was rejected.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum RejectReason {
|
||||
/// Build hashes differ — not the same binary.
|
||||
HashMismatch,
|
||||
/// The offered node name is already claimed by a live peer.
|
||||
NameTaken,
|
||||
/// Wire protocol version mismatch.
|
||||
ProtoVersion,
|
||||
}
|
||||
|
||||
/// The control-plane frame inventory (RFC 010, *Implementation details*).
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub enum Frame {
|
||||
Hello {
|
||||
proto_version: u32,
|
||||
build_hash: u64,
|
||||
node_name: String,
|
||||
incarnation: Incarnation,
|
||||
meta: NodeMeta,
|
||||
},
|
||||
HelloAck {
|
||||
node_name: String,
|
||||
incarnation: Incarnation,
|
||||
meta: NodeMeta,
|
||||
},
|
||||
HelloReject {
|
||||
reason: RejectReason,
|
||||
},
|
||||
Heartbeat,
|
||||
Send {
|
||||
/// Target slot index (node is implicit in the connection, incarnation
|
||||
/// is bound at handshake — RFC 010 §3).
|
||||
index: u32,
|
||||
generation: u32,
|
||||
type_hash: u64,
|
||||
payload: Vec<u8>,
|
||||
},
|
||||
SendNamed {
|
||||
name: String,
|
||||
type_hash: u64,
|
||||
payload: Vec<u8>,
|
||||
},
|
||||
Monitor {
|
||||
monitor_id: u64,
|
||||
index: u32,
|
||||
generation: u32,
|
||||
},
|
||||
Demonitor {
|
||||
monitor_id: u64,
|
||||
},
|
||||
Down {
|
||||
monitor_id: u64,
|
||||
reason: RemoteDownReason,
|
||||
},
|
||||
}
|
||||
|
||||
/// Why a remotely-monitored actor is reported down: either the target's own
|
||||
/// terminal [`DownReason`] as its node recorded it, or the *link* to that
|
||||
/// node was lost (or absent) — which says nothing about the actor itself.
|
||||
///
|
||||
/// This is the cluster-side widening of `DownReason` (p5): `Disconnected`
|
||||
/// is a fact about a connection, never about a local actor, so it lives
|
||||
/// here rather than in the core enum — a local `Down` can never carry it,
|
||||
/// and matches on `DownReason` stay exhaustive over actor outcomes only.
|
||||
/// On the wire `Local(r)` uses `r`'s tag and `Disconnected` is tag 5,
|
||||
/// bound since c11; no peer emits it today (a lost link is synthesized
|
||||
/// locally), but the codec honours it both ways.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum RemoteDownReason {
|
||||
/// The target itself terminated; the peer reported this reason.
|
||||
Local(DownReason),
|
||||
/// The link to the target's node was lost or was never up.
|
||||
Disconnected,
|
||||
}
|
||||
|
||||
impl RemoteDownReason {
|
||||
/// The actor's own reason, if this was not a link loss.
|
||||
pub fn local(self) -> Option<DownReason> {
|
||||
match self {
|
||||
RemoteDownReason::Local(r) => Some(r),
|
||||
RemoteDownReason::Disconnected => None,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl From<DownReason> for RemoteDownReason {
|
||||
fn from(r: DownReason) -> Self {
|
||||
RemoteDownReason::Local(r)
|
||||
}
|
||||
}
|
||||
|
||||
// Frame tags. 0 is deliberately unassigned so an all-zero buffer never parses.
|
||||
const TAG_HELLO: u8 = 1;
|
||||
const TAG_HELLO_ACK: u8 = 2;
|
||||
const TAG_HELLO_REJECT: u8 = 3;
|
||||
const TAG_HEARTBEAT: u8 = 4;
|
||||
const TAG_SEND: u8 = 5;
|
||||
const TAG_SEND_NAMED: u8 = 6;
|
||||
const TAG_MONITOR: u8 = 7;
|
||||
const TAG_DEMONITOR: u8 = 8;
|
||||
const TAG_DOWN: u8 = 9;
|
||||
|
||||
// RejectReason tags.
|
||||
const REJ_HASH_MISMATCH: u8 = 1;
|
||||
const REJ_NAME_TAKEN: u8 = 2;
|
||||
const REJ_PROTO_VERSION: u8 = 3;
|
||||
|
||||
// DownReason tags. Do not reuse tags.
|
||||
const DR_EXIT: u8 = 1;
|
||||
const DR_PANIC: u8 = 2;
|
||||
const DR_STOPPED: u8 = 3;
|
||||
const DR_NOPROC: u8 = 4;
|
||||
const DR_DISCONNECTED: u8 = 5;
|
||||
// `Shutdown` never rides in a `Down` by contract (a target that honours the
|
||||
// request exits normally) — the tag exists so the codec stays total.
|
||||
const DR_SHUTDOWN: u8 = 6;
|
||||
|
||||
/// Frame could not be encoded. The output buffer is left exactly as it was.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum EncodeError {
|
||||
/// tag + body exceed [`MAX_FRAME_LEN`].
|
||||
FrameTooLarge { len: usize },
|
||||
/// A string field exceeds `u16::MAX` bytes.
|
||||
StringTooLong { len: usize },
|
||||
}
|
||||
|
||||
impl core::fmt::Display for EncodeError {
|
||||
fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result {
|
||||
match self {
|
||||
Self::FrameTooLarge { len } => {
|
||||
write!(f, "frame body of {len} bytes exceeds MAX_FRAME_LEN")
|
||||
}
|
||||
Self::StringTooLong { len } => {
|
||||
write!(f, "string field of {len} bytes exceeds u16::MAX")
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl std::error::Error for EncodeError {}
|
||||
|
||||
/// Frame could not be decoded. Everything here is *corruption* — "not enough
|
||||
/// bytes yet" is the `Ok(None)` streaming case, never an error.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum DecodeError {
|
||||
/// The length prefix exceeds [`MAX_FRAME_LEN`].
|
||||
FrameTooLarge { declared: usize },
|
||||
/// The length prefix is zero — there is no tag byte.
|
||||
EmptyFrame,
|
||||
/// Unknown frame tag.
|
||||
UnknownTag(u8),
|
||||
/// Unknown tag for an enum-shaped field.
|
||||
UnknownEnumTag { what: &'static str, tag: u8 },
|
||||
/// A field ran past the declared frame end (the length prefix lied long,
|
||||
/// or a length-carrying field inside the body lied).
|
||||
Truncated,
|
||||
/// Bytes were left over after the body (the length prefix lied short).
|
||||
Trailing { extra: usize },
|
||||
/// A string field was not valid UTF-8.
|
||||
Utf8,
|
||||
}
|
||||
|
||||
impl core::fmt::Display for DecodeError {
|
||||
fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result {
|
||||
match self {
|
||||
Self::FrameTooLarge { declared } => {
|
||||
write!(f, "declared frame length {declared} exceeds MAX_FRAME_LEN")
|
||||
}
|
||||
Self::EmptyFrame => write!(f, "zero-length frame (no tag byte)"),
|
||||
Self::UnknownTag(t) => write!(f, "unknown frame tag {t}"),
|
||||
Self::UnknownEnumTag { what, tag } => write!(f, "unknown {what} tag {tag}"),
|
||||
Self::Truncated => write!(f, "frame body truncated mid-field"),
|
||||
Self::Trailing { extra } => write!(f, "{extra} trailing bytes after frame body"),
|
||||
Self::Utf8 => write!(f, "string field is not valid UTF-8"),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl std::error::Error for DecodeError {}
|
||||
|
||||
impl Frame {
|
||||
/// Append this frame, length-prefixed, to `out`.
|
||||
///
|
||||
/// On error `out` is left untouched.
|
||||
pub fn encode(&self, out: &mut Vec<u8>) -> Result<(), EncodeError> {
|
||||
let start = out.len();
|
||||
out.extend_from_slice(&[0u8; 4]); // length placeholder, patched below
|
||||
let result = self.encode_body(out);
|
||||
match result {
|
||||
Ok(()) => {
|
||||
let frame_len = out.len() - start - 4;
|
||||
if frame_len > MAX_FRAME_LEN {
|
||||
out.truncate(start);
|
||||
return Err(EncodeError::FrameTooLarge { len: frame_len });
|
||||
}
|
||||
// Cast is lossless: MAX_FRAME_LEN < u32::MAX, checked above.
|
||||
let len32 = frame_len as u32;
|
||||
out[start..start + 4].copy_from_slice(&len32.to_le_bytes());
|
||||
Ok(())
|
||||
}
|
||||
Err(e) => {
|
||||
out.truncate(start);
|
||||
Err(e)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn encode_body(&self, out: &mut Vec<u8>) -> Result<(), EncodeError> {
|
||||
match self {
|
||||
Frame::Hello {
|
||||
proto_version,
|
||||
build_hash,
|
||||
node_name,
|
||||
incarnation,
|
||||
meta,
|
||||
} => {
|
||||
out.push(TAG_HELLO);
|
||||
put_u32(out, *proto_version);
|
||||
put_u64(out, *build_hash);
|
||||
put_str(out, node_name)?;
|
||||
put_u32(out, incarnation.get());
|
||||
put_meta(out, meta)?;
|
||||
}
|
||||
Frame::HelloAck {
|
||||
node_name,
|
||||
incarnation,
|
||||
meta,
|
||||
} => {
|
||||
out.push(TAG_HELLO_ACK);
|
||||
put_str(out, node_name)?;
|
||||
put_u32(out, incarnation.get());
|
||||
put_meta(out, meta)?;
|
||||
}
|
||||
Frame::HelloReject { reason } => {
|
||||
out.push(TAG_HELLO_REJECT);
|
||||
out.push(match reason {
|
||||
RejectReason::HashMismatch => REJ_HASH_MISMATCH,
|
||||
RejectReason::NameTaken => REJ_NAME_TAKEN,
|
||||
RejectReason::ProtoVersion => REJ_PROTO_VERSION,
|
||||
});
|
||||
}
|
||||
Frame::Heartbeat => out.push(TAG_HEARTBEAT),
|
||||
Frame::Send {
|
||||
index,
|
||||
generation,
|
||||
type_hash,
|
||||
payload,
|
||||
} => {
|
||||
out.push(TAG_SEND);
|
||||
put_u32(out, *index);
|
||||
put_u32(out, *generation);
|
||||
put_u64(out, *type_hash);
|
||||
put_blob(out, payload)?;
|
||||
}
|
||||
Frame::SendNamed {
|
||||
name,
|
||||
type_hash,
|
||||
payload,
|
||||
} => {
|
||||
out.push(TAG_SEND_NAMED);
|
||||
put_str(out, name)?;
|
||||
put_u64(out, *type_hash);
|
||||
put_blob(out, payload)?;
|
||||
}
|
||||
Frame::Monitor {
|
||||
monitor_id,
|
||||
index,
|
||||
generation,
|
||||
} => {
|
||||
out.push(TAG_MONITOR);
|
||||
put_u64(out, *monitor_id);
|
||||
put_u32(out, *index);
|
||||
put_u32(out, *generation);
|
||||
}
|
||||
Frame::Demonitor { monitor_id } => {
|
||||
out.push(TAG_DEMONITOR);
|
||||
put_u64(out, *monitor_id);
|
||||
}
|
||||
Frame::Down { monitor_id, reason } => {
|
||||
out.push(TAG_DOWN);
|
||||
put_u64(out, *monitor_id);
|
||||
out.push(match reason {
|
||||
RemoteDownReason::Local(DownReason::Exit) => DR_EXIT,
|
||||
RemoteDownReason::Local(DownReason::Panic) => DR_PANIC,
|
||||
RemoteDownReason::Local(DownReason::Stopped) => DR_STOPPED,
|
||||
RemoteDownReason::Local(DownReason::NoProc) => DR_NOPROC,
|
||||
RemoteDownReason::Local(DownReason::Shutdown) => DR_SHUTDOWN,
|
||||
RemoteDownReason::Disconnected => DR_DISCONNECTED,
|
||||
});
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Try to decode one frame from the start of `buf`.
|
||||
///
|
||||
/// `Ok(Some((frame, consumed)))` — a full frame; the caller advances by
|
||||
/// `consumed`. `Ok(None)` — not enough bytes yet (streaming); read more
|
||||
/// and retry. `Err(_)` — the bytes are corrupt; the connection is dead.
|
||||
pub fn decode(buf: &[u8]) -> Result<Option<(Frame, usize)>, DecodeError> {
|
||||
let Some(prefix) = buf.get(0..4) else {
|
||||
return Ok(None);
|
||||
};
|
||||
let mut len4 = [0u8; 4];
|
||||
len4.copy_from_slice(prefix);
|
||||
let declared = u32::from_le_bytes(len4) as usize;
|
||||
if declared > MAX_FRAME_LEN {
|
||||
return Err(DecodeError::FrameTooLarge { declared });
|
||||
}
|
||||
if declared == 0 {
|
||||
return Err(DecodeError::EmptyFrame);
|
||||
}
|
||||
let Some(body) = buf.get(4..4 + declared) else {
|
||||
return Ok(None);
|
||||
};
|
||||
let mut r = Reader { buf: body, pos: 0 };
|
||||
let frame = Self::decode_body(&mut r)?;
|
||||
if r.pos != body.len() {
|
||||
return Err(DecodeError::Trailing {
|
||||
extra: body.len() - r.pos,
|
||||
});
|
||||
}
|
||||
Ok(Some((frame, 4 + declared)))
|
||||
}
|
||||
|
||||
fn decode_body(r: &mut Reader<'_>) -> Result<Frame, DecodeError> {
|
||||
let tag = r.u8()?;
|
||||
let frame = match tag {
|
||||
TAG_HELLO => Frame::Hello {
|
||||
proto_version: r.u32()?,
|
||||
build_hash: r.u64()?,
|
||||
node_name: r.string()?,
|
||||
incarnation: Incarnation::new(r.u32()?),
|
||||
meta: r.meta()?,
|
||||
},
|
||||
TAG_HELLO_ACK => Frame::HelloAck {
|
||||
node_name: r.string()?,
|
||||
incarnation: Incarnation::new(r.u32()?),
|
||||
meta: r.meta()?,
|
||||
},
|
||||
TAG_HELLO_REJECT => Frame::HelloReject {
|
||||
reason: match r.u8()? {
|
||||
REJ_HASH_MISMATCH => RejectReason::HashMismatch,
|
||||
REJ_NAME_TAKEN => RejectReason::NameTaken,
|
||||
REJ_PROTO_VERSION => RejectReason::ProtoVersion,
|
||||
t => {
|
||||
return Err(DecodeError::UnknownEnumTag {
|
||||
what: "RejectReason",
|
||||
tag: t,
|
||||
})
|
||||
}
|
||||
},
|
||||
},
|
||||
TAG_HEARTBEAT => Frame::Heartbeat,
|
||||
TAG_SEND => Frame::Send {
|
||||
index: r.u32()?,
|
||||
generation: r.u32()?,
|
||||
type_hash: r.u64()?,
|
||||
payload: r.blob()?,
|
||||
},
|
||||
TAG_SEND_NAMED => Frame::SendNamed {
|
||||
name: r.string()?,
|
||||
type_hash: r.u64()?,
|
||||
payload: r.blob()?,
|
||||
},
|
||||
TAG_MONITOR => Frame::Monitor {
|
||||
monitor_id: r.u64()?,
|
||||
index: r.u32()?,
|
||||
generation: r.u32()?,
|
||||
},
|
||||
TAG_DEMONITOR => Frame::Demonitor {
|
||||
monitor_id: r.u64()?,
|
||||
},
|
||||
TAG_DOWN => Frame::Down {
|
||||
monitor_id: r.u64()?,
|
||||
reason: match r.u8()? {
|
||||
DR_EXIT => RemoteDownReason::Local(DownReason::Exit),
|
||||
DR_PANIC => RemoteDownReason::Local(DownReason::Panic),
|
||||
DR_STOPPED => RemoteDownReason::Local(DownReason::Stopped),
|
||||
DR_NOPROC => RemoteDownReason::Local(DownReason::NoProc),
|
||||
DR_SHUTDOWN => RemoteDownReason::Local(DownReason::Shutdown),
|
||||
DR_DISCONNECTED => RemoteDownReason::Disconnected,
|
||||
t => {
|
||||
return Err(DecodeError::UnknownEnumTag {
|
||||
what: "RemoteDownReason",
|
||||
tag: t,
|
||||
})
|
||||
}
|
||||
},
|
||||
},
|
||||
t => return Err(DecodeError::UnknownTag(t)),
|
||||
};
|
||||
Ok(frame)
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Body writers
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
fn put_u32(out: &mut Vec<u8>, v: u32) {
|
||||
out.extend_from_slice(&v.to_le_bytes());
|
||||
}
|
||||
|
||||
fn put_u64(out: &mut Vec<u8>, v: u64) {
|
||||
out.extend_from_slice(&v.to_le_bytes());
|
||||
}
|
||||
|
||||
fn put_str(out: &mut Vec<u8>, s: &str) -> Result<(), EncodeError> {
|
||||
let Ok(len) = u16::try_from(s.len()) else {
|
||||
return Err(EncodeError::StringTooLong { len: s.len() });
|
||||
};
|
||||
out.extend_from_slice(&len.to_le_bytes());
|
||||
out.extend_from_slice(s.as_bytes());
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn put_blob(out: &mut Vec<u8>, b: &[u8]) -> Result<(), EncodeError> {
|
||||
let Ok(len) = u32::try_from(b.len()) else {
|
||||
return Err(EncodeError::FrameTooLarge { len: b.len() });
|
||||
};
|
||||
out.extend_from_slice(&len.to_le_bytes());
|
||||
out.extend_from_slice(b);
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn put_meta(out: &mut Vec<u8>, m: &NodeMeta) -> Result<(), EncodeError> {
|
||||
put_str(out, &m.role)?;
|
||||
put_str(out, &m.region)
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Body reader
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
struct Reader<'a> {
|
||||
buf: &'a [u8],
|
||||
pos: usize,
|
||||
}
|
||||
|
||||
impl Reader<'_> {
|
||||
fn take(&mut self, n: usize) -> Result<&[u8], DecodeError> {
|
||||
let end = self.pos.checked_add(n).ok_or(DecodeError::Truncated)?;
|
||||
let s = self.buf.get(self.pos..end).ok_or(DecodeError::Truncated)?;
|
||||
self.pos = end;
|
||||
Ok(s)
|
||||
}
|
||||
|
||||
fn u8(&mut self) -> Result<u8, DecodeError> {
|
||||
Ok(self.take(1)?[0])
|
||||
}
|
||||
|
||||
fn u16(&mut self) -> Result<u16, DecodeError> {
|
||||
let mut b = [0u8; 2];
|
||||
b.copy_from_slice(self.take(2)?);
|
||||
Ok(u16::from_le_bytes(b))
|
||||
}
|
||||
|
||||
fn u32(&mut self) -> Result<u32, DecodeError> {
|
||||
let mut b = [0u8; 4];
|
||||
b.copy_from_slice(self.take(4)?);
|
||||
Ok(u32::from_le_bytes(b))
|
||||
}
|
||||
|
||||
fn u64(&mut self) -> Result<u64, DecodeError> {
|
||||
let mut b = [0u8; 8];
|
||||
b.copy_from_slice(self.take(8)?);
|
||||
Ok(u64::from_le_bytes(b))
|
||||
}
|
||||
|
||||
fn string(&mut self) -> Result<String, DecodeError> {
|
||||
let len = self.u16()? as usize;
|
||||
let bytes = self.take(len)?;
|
||||
match core::str::from_utf8(bytes) {
|
||||
Ok(s) => Ok(s.to_owned()),
|
||||
Err(_) => Err(DecodeError::Utf8),
|
||||
}
|
||||
}
|
||||
|
||||
fn blob(&mut self) -> Result<Vec<u8>, DecodeError> {
|
||||
let len = self.u32()? as usize;
|
||||
Ok(self.take(len)?.to_vec())
|
||||
}
|
||||
|
||||
fn meta(&mut self) -> Result<NodeMeta, DecodeError> {
|
||||
Ok(NodeMeta {
|
||||
role: self.string()?,
|
||||
region: self.string()?,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// The postcard seam (RFC 010 §2) — the ONLY place payload bytes are produced
|
||||
// or consumed. A codec swap lands here and nowhere else.
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Payload (de)serialization failed at the codec seam.
|
||||
#[derive(Debug)]
|
||||
pub struct PayloadError(String);
|
||||
|
||||
impl core::fmt::Display for PayloadError {
|
||||
fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result {
|
||||
write!(f, "payload codec: {}", self.0)
|
||||
}
|
||||
}
|
||||
|
||||
impl std::error::Error for PayloadError {}
|
||||
|
||||
/// Serialize a payload value to the wire blob.
|
||||
pub fn encode_payload<T: serde::Serialize + ?Sized>(value: &T) -> Result<Vec<u8>, PayloadError> {
|
||||
postcard::to_allocvec(value).map_err(|e| PayloadError(e.to_string()))
|
||||
}
|
||||
|
||||
/// Deserialize a payload value from the wire blob.
|
||||
pub fn decode_payload<T: serde::de::DeserializeOwned>(bytes: &[u8]) -> Result<T, PayloadError> {
|
||||
postcard::from_bytes(bytes).map_err(|e| PayloadError(e.to_string()))
|
||||
}
|
||||
@@ -1,238 +0,0 @@
|
||||
//! RFC 010 c8 — explicit exposure: the node's remote surface, and the
|
||||
//! fixed-seed type hash.
|
||||
//!
|
||||
//! Nothing local is remotely reachable by default (RFC §4 — "a gun needs a
|
||||
//! safety"). [`expose`] marks a registered name remotely addressable and
|
||||
//! registers `M`'s decoder under [`type_hash::<M>()`](type_hash);
|
||||
//! [`expose_type`] registers only the decoder (the reply-to path: a
|
||||
//! `RemotePid<A>` received in a message is sendable only if `A::Msg`'s
|
||||
//! decoder was explicitly registered). The exposed set is the node's
|
||||
//! visible, auditable remote surface ([`exposed_names`]).
|
||||
//!
|
||||
//! ## Where the state lives
|
||||
//!
|
||||
//! On `RuntimeInner`, the [`pg`](crate::pg) pattern: a leaf-locked table,
|
||||
//! cfg-gated behind the `cluster` feature (zero-cost-when-off, per c1).
|
||||
//! Chosen over manager-held state because c9's inbound decode consults it
|
||||
//! per frame — a hot path that must not serialize every remote delivery
|
||||
//! through one gen_server. The state resets with the runtime, like every
|
||||
//! registry.
|
||||
//!
|
||||
//! ## The watchable fold (D3), against the code as it stands
|
||||
//!
|
||||
//! RFC §4: the exposed set is not a new registry — it folds into the
|
||||
//! existing `watchable` machinery, one set, two set-sites (a pid crossing
|
||||
//! the membrane, and expose). Reading the code: `register` **already
|
||||
//! stamps every named holder watchable** ("no successfully-registered actor
|
||||
//! can die unflagged", registry.rs), so an exposed *name*'s holder needs no
|
||||
//! extra mark here — the guarantee holds by registration, and re-registration
|
||||
//! after a holder's death re-stamps the new holder for free (a per-tenancy
|
||||
//! mark taken at expose time could not do that). The cluster's own
|
||||
//! `mark_watchable` set-site is therefore the **pid crossing the wire** —
|
||||
//! serialization of a pid into a frame, c10 — the exact analog of the
|
||||
//! membrane crossing. What lives here is only the name/type-level state
|
||||
//! neither the registry nor the slot bits can carry: which names are
|
||||
//! exposed, and how to decode each type hash.
|
||||
//!
|
||||
//! ## The hash
|
||||
//!
|
||||
//! [`type_hash`] is FNV-1a 64 (fixed seed: the FNV offset basis) over
|
||||
//! `TypeId`, so it is a constant of the binary: stable across runs of the
|
||||
//! same build — exactly the scope the build-hash handshake reduces the mesh
|
||||
//! to — and deliberately *not* stable across builds (scope guard: no
|
||||
//! cross-version wire compatibility). A collision between two exposed types
|
||||
//! degrades to a decode error or a refused channel, never a misroute — the
|
||||
//! local `SendError::NoChannel` guarantee survives the network (RFC §3).
|
||||
//!
|
||||
//! ## The decoder contract
|
||||
//!
|
||||
//! A decoder is **decode-and-deliver-to-pid**: it captures `M` (the one
|
||||
//! typed site), decodes the payload, and hands the value to the target's
|
||||
//! published channel via the registry's own dynamic send. Wire-name →
|
||||
//! local-pid resolution deliberately stays *outside* — that is c9's single
|
||||
//! resolution seam, and it calls [`decode_deliver`].
|
||||
|
||||
use std::any::TypeId;
|
||||
use std::collections::HashMap;
|
||||
use std::hash::{Hash, Hasher};
|
||||
|
||||
use crate::cluster::envelope::{decode_payload, PayloadError};
|
||||
use crate::pid::{Name, Pid};
|
||||
use crate::registry::{send_dyn, SendError};
|
||||
use crate::scheduler::with_runtime;
|
||||
|
||||
/// The fixed-seed `TypeId` → `u64` hash: FNV-1a 64 over the `TypeId`'s hash
|
||||
/// bytes, seeded with the FNV offset basis. A constant of the binary — see
|
||||
/// the module docs for scope.
|
||||
pub fn type_hash<M: 'static>() -> u64 {
|
||||
let mut h = Fnv1a64::new();
|
||||
TypeId::of::<M>().hash(&mut h);
|
||||
h.finish()
|
||||
}
|
||||
|
||||
/// FNV-1a 64 as a `Hasher`, so `TypeId` (opaque, `Hash`-only) can feed it.
|
||||
/// Same constants as the const fns in [`crate::cluster`] (BUILD_HASH).
|
||||
struct Fnv1a64(u64);
|
||||
|
||||
impl Fnv1a64 {
|
||||
fn new() -> Self {
|
||||
Fnv1a64(0xcbf2_9ce4_8422_2325)
|
||||
}
|
||||
}
|
||||
|
||||
impl Hasher for Fnv1a64 {
|
||||
fn write(&mut self, bytes: &[u8]) {
|
||||
for &b in bytes {
|
||||
self.0 ^= b as u64;
|
||||
self.0 = self.0.wrapping_mul(0x0000_0100_0000_01b3);
|
||||
}
|
||||
}
|
||||
fn finish(&self) -> u64 {
|
||||
self.0
|
||||
}
|
||||
}
|
||||
|
||||
/// Why a [`decode_deliver`] did not deliver. Payload-free mirror of the
|
||||
/// registry's `SendError` where relevant — the caller (c9's inbound path)
|
||||
/// has only bytes to give back, not a typed message.
|
||||
#[derive(Debug)]
|
||||
pub enum DeliverError {
|
||||
/// No decoder is registered under this hash — the type was never
|
||||
/// exposed here.
|
||||
UnknownType,
|
||||
/// The bytes did not decode as the registered type.
|
||||
Decode(PayloadError),
|
||||
/// The target actor is dead (or was never alive).
|
||||
Dead,
|
||||
/// The target is live but has no channel for this message type, or that
|
||||
/// channel is closed — the `NoChannel` guarantee: a decoded value is
|
||||
/// refused, never misrouted.
|
||||
WrongChannel,
|
||||
}
|
||||
|
||||
/// A registered decoder: decode `bytes` as the captured type and deliver to
|
||||
/// `pid`'s published channel. `Arc`, so [`decode_deliver`] can clone it out
|
||||
/// from under the exposure lock and call it lock-free — the decoder's
|
||||
/// `send_dyn` takes the registry lock, and the two are mutual Leaves that
|
||||
/// must never nest.
|
||||
type Decoder = std::sync::Arc<dyn Fn(Pid, &[u8]) -> Result<(), DeliverError> + Send + Sync>;
|
||||
|
||||
/// The exposure state, one per runtime (a `RuntimeInner` field, pg-style).
|
||||
pub(crate) struct ExposureState {
|
||||
/// The exposed names: registry key → the type hash it expects.
|
||||
exposed: HashMap<&'static str, u64>,
|
||||
/// The decoders: type hash → decode-and-deliver.
|
||||
decoders: HashMap<u64, Decoder>,
|
||||
}
|
||||
|
||||
impl ExposureState {
|
||||
pub(crate) fn new() -> Self {
|
||||
ExposureState {
|
||||
exposed: HashMap::new(),
|
||||
decoders: HashMap::new(),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Mark `name` remotely addressable and register `M`'s decoder under its
|
||||
/// type hash (so both name-sends and pid-sends of `M` work — RFC §4).
|
||||
/// Returns the hash.
|
||||
///
|
||||
/// Exposure is a **name-level fact**, independent of who currently holds the
|
||||
/// name (names late-bind: the registry re-resolves on every send, and c9's
|
||||
/// seam resolves per delivery). Exposing an unregistered name is therefore
|
||||
/// valid — deliveries fail with "unresolved" until someone registers it.
|
||||
/// Idempotent. Must run inside [`run`](crate::run).
|
||||
pub fn expose<M>(name: Name<M>) -> u64
|
||||
where
|
||||
M: serde::de::DeserializeOwned + Send + 'static,
|
||||
{
|
||||
let h = ensure_decoder::<M>();
|
||||
with_runtime(|inner| {
|
||||
inner.exposure.lock().exposed.insert(name.as_str(), h);
|
||||
});
|
||||
h
|
||||
}
|
||||
|
||||
/// Register only `M`'s decoder (no name): the reply-to path. Returns the
|
||||
/// hash. Idempotent. Must run inside [`run`](crate::run).
|
||||
pub fn expose_type<M>() -> u64
|
||||
where
|
||||
M: serde::de::DeserializeOwned + Send + 'static,
|
||||
{
|
||||
ensure_decoder::<M>()
|
||||
}
|
||||
|
||||
fn ensure_decoder<M>() -> u64
|
||||
where
|
||||
M: serde::de::DeserializeOwned + Send + 'static,
|
||||
{
|
||||
let h = type_hash::<M>();
|
||||
with_runtime(|inner| {
|
||||
inner
|
||||
.exposure
|
||||
.lock()
|
||||
.decoders
|
||||
.entry(h)
|
||||
.or_insert_with(decoder::<M>);
|
||||
});
|
||||
h
|
||||
}
|
||||
|
||||
/// The one typed site: decode as `M`, deliver via the registry's dynamic
|
||||
/// send. See the module docs for the error mapping.
|
||||
fn decoder<M>() -> Decoder
|
||||
where
|
||||
M: serde::de::DeserializeOwned + Send + 'static,
|
||||
{
|
||||
std::sync::Arc::new(|pid, bytes| {
|
||||
let m: M = decode_payload(bytes).map_err(DeliverError::Decode)?;
|
||||
send_dyn(pid, m).map_err(|e| match e {
|
||||
SendError::Dead(_) | SendError::Unresolved(_) | SendError::NoMember(_) => {
|
||||
DeliverError::Dead
|
||||
}
|
||||
SendError::NoChannel(_) | SendError::Closed(_) => DeliverError::WrongChannel,
|
||||
})
|
||||
})
|
||||
}
|
||||
|
||||
/// The type hash `name` was exposed with, or `None` if it is not exposed.
|
||||
/// Must run inside [`run`](crate::run).
|
||||
pub fn exposed_hash(name: &str) -> Option<u64> {
|
||||
with_runtime(|inner| inner.exposure.lock().exposed.get(name).copied())
|
||||
}
|
||||
|
||||
/// Whether a decoder is registered under `hash`. Must run inside
|
||||
/// [`run`](crate::run).
|
||||
pub fn decoder_registered(hash: u64) -> bool {
|
||||
with_runtime(|inner| inner.exposure.lock().decoders.contains_key(&hash))
|
||||
}
|
||||
|
||||
/// Decode `bytes` under `hash`'s registered decoder and deliver to `pid`.
|
||||
/// This is the delivery half c9's single resolution seam calls after it has
|
||||
/// resolved a wire name to a local pid. Must run inside [`run`](crate::run).
|
||||
pub fn decode_deliver(hash: u64, to: Pid, bytes: &[u8]) -> Result<(), DeliverError> {
|
||||
// Clone the Arc under the lock, call outside it: the decoder's
|
||||
// `send_dyn` takes the registry lock — a mutual Leaf with the exposure
|
||||
// lock (the runtime asserts if Leaves nest). This also keeps unrelated
|
||||
// deliveries uncoupled from a slow decode.
|
||||
let d = with_runtime(|inner| inner.exposure.lock().decoders.get(&hash).cloned());
|
||||
match d {
|
||||
Some(d) => d(to, bytes),
|
||||
None => Err(DeliverError::UnknownType),
|
||||
}
|
||||
}
|
||||
|
||||
/// The auditable remote surface: every exposed name and its type hash,
|
||||
/// unordered. Must run inside [`run`](crate::run).
|
||||
pub fn exposed_names() -> Vec<(&'static str, u64)> {
|
||||
with_runtime(|inner| {
|
||||
inner
|
||||
.exposure
|
||||
.lock()
|
||||
.exposed
|
||||
.iter()
|
||||
.map(|(&n, &h)| (n, h))
|
||||
.collect()
|
||||
})
|
||||
}
|
||||
@@ -1,178 +0,0 @@
|
||||
//! RFC 010 c5 — the handshake as a pure state machine.
|
||||
//!
|
||||
//! Frames in, actions out — no IO, no clocks, no actors. The c6 connection
|
||||
//! actor drives these machines and executes their actions; everything
|
||||
//! time-shaped (handshake deadline, heartbeats) lives there.
|
||||
|
||||
use crate::cluster::envelope::{Frame, NodeMeta, RejectReason, PROTO_VERSION};
|
||||
use crate::pg::Incarnation;
|
||||
|
||||
/// This node's identity and metadata, as offered in (or checked against) a
|
||||
/// `Hello`.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct Local {
|
||||
pub node_name: String,
|
||||
pub incarnation: Incarnation,
|
||||
pub build_hash: u64,
|
||||
pub meta: NodeMeta,
|
||||
}
|
||||
|
||||
/// The peer identity a successful handshake yields (what c7 feeds `node_up`).
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct Peer {
|
||||
pub node_name: String,
|
||||
pub incarnation: Incarnation,
|
||||
pub meta: NodeMeta,
|
||||
}
|
||||
|
||||
/// Driver-supplied standing of the *offered* name at this node — knowledge
|
||||
/// the pure machine cannot have (c6 owns the connection table and dial
|
||||
/// set). One answer, in the responder's own precedence: an established
|
||||
/// peer under that name outranks an in-flight dial to it.
|
||||
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
|
||||
pub enum PeerStanding {
|
||||
/// Neither connected to nor dialing that name.
|
||||
#[default]
|
||||
Free,
|
||||
/// An established peer already holds that name.
|
||||
Claimed,
|
||||
/// We have our own dial in flight to that name.
|
||||
Dialing,
|
||||
}
|
||||
|
||||
/// Simultaneous-connect tie-break: does the connection dialed by
|
||||
/// `dialer_name` survive against the reverse dial?
|
||||
/// The rule (ratified 2026-08-14, a wire-protocol fact): the connection
|
||||
/// dialed by the lexicographically **smaller** name survives. Both ends know
|
||||
/// both names, so both compute the same verdict — which is why the losing
|
||||
/// side may close silently instead of sending a reject.
|
||||
pub fn dial_wins(dialer_name: &str, acceptor_name: &str) -> bool {
|
||||
dialer_name < acceptor_name
|
||||
}
|
||||
|
||||
/// Dial side: emits `Hello` at construction, interprets the single response.
|
||||
#[must_use]
|
||||
#[derive(Debug)]
|
||||
pub struct Initiator(());
|
||||
|
||||
/// What the dial side's response frame meant.
|
||||
#[must_use]
|
||||
#[derive(Debug, PartialEq, Eq)]
|
||||
pub enum InitiatorOutcome {
|
||||
Established(Peer),
|
||||
Rejected(RejectReason),
|
||||
/// Protocol violation before the ack — close. Carries the offending frame.
|
||||
Failed(Frame),
|
||||
}
|
||||
|
||||
impl Initiator {
|
||||
/// Start a dial-side handshake: the returned frame is the `Hello` to
|
||||
/// send; the returned machine is the right to interpret the response.
|
||||
pub fn new(local: &Local) -> (Self, Frame) {
|
||||
let hello = Frame::Hello {
|
||||
proto_version: PROTO_VERSION,
|
||||
build_hash: local.build_hash,
|
||||
node_name: local.node_name.clone(),
|
||||
incarnation: local.incarnation,
|
||||
meta: local.meta.clone(),
|
||||
};
|
||||
(Initiator(()), hello)
|
||||
}
|
||||
|
||||
/// Interpret the response. The `HelloAck` carries no hash or version —
|
||||
/// the responder already checked ours against its own, and equality is
|
||||
/// symmetric, so a one-sided check is sound.
|
||||
pub fn on_frame(self, frame: Frame) -> InitiatorOutcome {
|
||||
match frame {
|
||||
Frame::HelloAck {
|
||||
node_name,
|
||||
incarnation,
|
||||
meta,
|
||||
} => InitiatorOutcome::Established(Peer {
|
||||
node_name,
|
||||
incarnation,
|
||||
meta,
|
||||
}),
|
||||
Frame::HelloReject { reason } => InitiatorOutcome::Rejected(reason),
|
||||
other => InitiatorOutcome::Failed(other),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Accept side: awaits exactly one `Hello`, answers or closes.
|
||||
#[must_use]
|
||||
#[derive(Debug)]
|
||||
pub struct Responder {
|
||||
local: Local,
|
||||
}
|
||||
|
||||
/// What to do with an inbound connection's first frame.
|
||||
#[must_use]
|
||||
#[derive(Debug, PartialEq, Eq)]
|
||||
pub enum ResponderOutcome {
|
||||
/// Send the ack; the connection is established.
|
||||
Accepted { reply: Frame, peer: Peer },
|
||||
/// Send the reject, then close.
|
||||
Rejected { reply: Frame, reason: RejectReason },
|
||||
/// Lost the simultaneous-connect tie-break: close silently, no frame.
|
||||
TieBreakLoss,
|
||||
/// Protocol violation before Hello — close, no reply. Carries the frame.
|
||||
Failed(Frame),
|
||||
}
|
||||
|
||||
impl Responder {
|
||||
pub fn new(local: Local) -> Self {
|
||||
Responder { local }
|
||||
}
|
||||
|
||||
/// Judge the connection's first frame. Check order is proto → hash →
|
||||
/// name → tie-break: validity before identity. `HelloReject` is the
|
||||
/// cross-version compatibility anchor, so a version-mismatched peer
|
||||
/// still gets one.
|
||||
pub fn on_frame(self, frame: Frame, standing: PeerStanding) -> ResponderOutcome {
|
||||
let Frame::Hello {
|
||||
proto_version,
|
||||
build_hash,
|
||||
node_name,
|
||||
incarnation,
|
||||
meta,
|
||||
} = frame
|
||||
else {
|
||||
return ResponderOutcome::Failed(frame);
|
||||
};
|
||||
|
||||
let reject = |reason| ResponderOutcome::Rejected {
|
||||
reply: Frame::HelloReject { reason },
|
||||
reason,
|
||||
};
|
||||
|
||||
if proto_version != PROTO_VERSION {
|
||||
return reject(RejectReason::ProtoVersion);
|
||||
}
|
||||
if build_hash != self.local.build_hash {
|
||||
return reject(RejectReason::HashMismatch);
|
||||
}
|
||||
if node_name == self.local.node_name || standing == PeerStanding::Claimed {
|
||||
return reject(RejectReason::NameTaken);
|
||||
}
|
||||
// Simultaneous connect: the inbound frame is the peer's dial. If our
|
||||
// own in-flight dial wins instead, drop this one silently — the peer
|
||||
// computes the same verdict (see `dial_wins`).
|
||||
if standing == PeerStanding::Dialing && !dial_wins(&node_name, &self.local.node_name) {
|
||||
return ResponderOutcome::TieBreakLoss;
|
||||
}
|
||||
|
||||
ResponderOutcome::Accepted {
|
||||
reply: Frame::HelloAck {
|
||||
node_name: self.local.node_name,
|
||||
incarnation: self.local.incarnation,
|
||||
meta: self.local.meta,
|
||||
},
|
||||
peer: Peer {
|
||||
node_name,
|
||||
incarnation,
|
||||
meta,
|
||||
},
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,295 +0,0 @@
|
||||
//! RFC 010 c6 — the cluster connection manager.
|
||||
//!
|
||||
//! One manager per runtime: the single registry of live peer connections and
|
||||
//! the source of truth for whether a peer name is already claimed. The
|
||||
//! accept/connect path registers each established connection here, handing
|
||||
//! over its [`ConnHandle`] — **the manager owns connection lifetime**. A
|
||||
//! connection lives as long as its table entry, so it neither outlives nor
|
||||
//! dies with whichever actor happened to establish it. Registered actors are
|
||||
//! also *monitored*, so the table self-heals on any exit path — a connection
|
||||
//! that panics, is cancelled, or closes cleanly is removed without
|
||||
//! cooperation from the dying actor.
|
||||
//!
|
||||
//! The manager also holds the **membership state** (c7a): `node_up` fires on
|
||||
//! a successful registration and `node_down` on removal — they are derived
|
||||
//! facts of the exact events this table already owns, so holding the view
|
||||
//! here means no cross-actor race between "connection exists" and "node is
|
||||
//! up". The consumer surface (event types, [`subscribe`], [`view`],
|
||||
//! semantics) is [`membership`](crate::cluster::membership); no consumer
|
||||
//! ever touches the table itself.
|
||||
//!
|
||||
//! The connector dial loop is c7b, built on top of both.
|
||||
|
||||
use std::collections::HashMap;
|
||||
|
||||
use crate::channel::Sender;
|
||||
use crate::cluster::conn::ConnHandle;
|
||||
use crate::cluster::handshake::{Peer, PeerStanding};
|
||||
use crate::cluster::membership::{NodeEvent, NodeInfo};
|
||||
use crate::cluster::remote::{bind_outbound, unbind_outbound};
|
||||
use crate::gen_server::{GenServer, GenServerCtx, GenServerName, Watcher};
|
||||
use crate::monitor::{monitor, Down};
|
||||
use crate::pg::NodeId;
|
||||
use crate::pid::Pid;
|
||||
|
||||
/// Well-known name of the singleton manager within a runtime. Connection
|
||||
/// actors reach it by name rather than by a passed-around ref, so a restarted
|
||||
/// manager is always found at the same key.
|
||||
pub const MANAGER: GenServerName<Manager> = GenServerName::new("smarm.cluster.manager");
|
||||
|
||||
/// One live connection's entry: the actor running it, the handle whose
|
||||
/// lifetime *is* the connection's (see the module docs), and the peer's
|
||||
/// membership identity (what `node_up` announced and `node_down` will name).
|
||||
struct ConnEntry {
|
||||
pid: Pid,
|
||||
info: NodeInfo,
|
||||
_handle: ConnHandle,
|
||||
}
|
||||
|
||||
/// The connection registry: peer name → the connection actor that owns that
|
||||
/// peer's control connection. Plus the membership state layered on it (c7a):
|
||||
/// subscribers, and the `(name, incarnation)` → [`NodeId`] memo.
|
||||
pub struct Manager {
|
||||
conns: HashMap<String, ConnEntry>,
|
||||
/// In-flight dial intents: peer name -> the actor performing the dial.
|
||||
/// Registered *before* connecting so a crossing inbound `Hello` sees it
|
||||
/// ([`PeerStanding::Dialing`]); cleared the moment
|
||||
/// the dial resolves, and — because the dialer is monitored — on the
|
||||
/// dialer's death, so a panicking dial can never wedge the tie-break.
|
||||
dials: HashMap<String, Pid>,
|
||||
/// Membership subscribers; a closed channel is pruned on the next emit.
|
||||
subscribers: Vec<Sender<NodeEvent>>,
|
||||
/// The [`NodeId`] memo: a reconnect at the same incarnation keeps its id,
|
||||
/// a restart (new incarnation) allocates a fresh one. Grows one entry per
|
||||
/// distinct `(name, incarnation)` ever seen — unbounded in principle,
|
||||
/// bounded in practice by restarts actually happening.
|
||||
ids: HashMap<(String, u32), NodeId>,
|
||||
/// Next id to allocate. Starts at 1: id 0 is
|
||||
/// [`DEFAULT_NODE_ID`](crate::pg::DEFAULT_NODE_ID), the local node.
|
||||
next_id: u32,
|
||||
watcher: Option<Watcher<Manager>>,
|
||||
}
|
||||
|
||||
impl Manager {
|
||||
pub fn new() -> Self {
|
||||
Manager {
|
||||
conns: HashMap::new(),
|
||||
dials: HashMap::new(),
|
||||
subscribers: Vec::new(),
|
||||
ids: HashMap::new(),
|
||||
next_id: 1,
|
||||
watcher: None,
|
||||
}
|
||||
}
|
||||
|
||||
/// The memoized id for `(name, incarnation)` — see the field docs.
|
||||
fn node_id(&mut self, name: &str, incarnation: u32) -> NodeId {
|
||||
*self
|
||||
.ids
|
||||
.entry((name.to_string(), incarnation))
|
||||
.or_insert_with(|| {
|
||||
let id = NodeId::new(self.next_id);
|
||||
self.next_id += 1;
|
||||
id
|
||||
})
|
||||
}
|
||||
|
||||
/// Deliver `event` to every live subscriber, pruning the dead: a closed
|
||||
/// channel means the subscriber dropped its [`MembershipEvents`]
|
||||
/// (crate::cluster::membership::MembershipEvents).
|
||||
fn emit(&mut self, event: &NodeEvent) {
|
||||
self.subscribers.retain(|tx| tx.send(event.clone()).is_ok());
|
||||
}
|
||||
}
|
||||
|
||||
impl Default for Manager {
|
||||
fn default() -> Self {
|
||||
Manager::new()
|
||||
}
|
||||
}
|
||||
|
||||
/// Outcome of a [`Call::Register`].
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum Registered {
|
||||
/// The name was free; this connection is now the peer of record.
|
||||
Ok,
|
||||
/// Another live connection already holds this name — the caller lost the
|
||||
/// race (or is a duplicate) and must not run.
|
||||
Duplicate,
|
||||
}
|
||||
|
||||
/// Requests to the manager.
|
||||
pub enum Call {
|
||||
/// The path claims its peer's name for a freshly-established connection,
|
||||
/// handing the manager the actor's [`ConnHandle`] and the handshake's
|
||||
/// [`Peer`] (the membership identity `node_up` announces). The manager
|
||||
/// monitors `pid` and holds the handle for as long as the entry lives; a
|
||||
/// [`Registered::Duplicate`] verdict drops the handle here, which stops
|
||||
/// the refused actor.
|
||||
Register {
|
||||
peer: Peer,
|
||||
pid: Pid,
|
||||
handle: ConnHandle,
|
||||
},
|
||||
/// Tear down the connection to `name`: the manager drops its handle, the
|
||||
/// actor stops, and the monitor removes the entry. A no-op if no such
|
||||
/// connection is live.
|
||||
Disconnect { name: String },
|
||||
/// The current peer names, sorted. For observation and tests.
|
||||
Peers,
|
||||
/// A dialer declares an in-flight dial to `name` before connecting. The
|
||||
/// pid is the dialing actor, monitored so the intent dies with it.
|
||||
DialBegin { name: String, pid: Pid },
|
||||
/// The dial to `name` resolved (either way): drop the intent. A call,
|
||||
/// not a cast, so the intent is provably gone before the dialer moves on.
|
||||
DialEnd { name: String },
|
||||
/// The [`PeerStanding`] of an inbound `Hello` offering `peer_name` — the
|
||||
/// accept path asks this between reading the frame and judging it.
|
||||
Standing { peer_name: String },
|
||||
/// Subscribe `tx` to membership events, snapshot-then-stream: one
|
||||
/// [`NodeEvent::NodeUp`] per live peer is queued into `tx` before this
|
||||
/// call answers, so the stream is exact from its first event (handlers
|
||||
/// are serialized — nothing interleaves with the snapshot). Use
|
||||
/// [`subscribe`](crate::cluster::membership::subscribe).
|
||||
Subscribe { tx: Sender<NodeEvent> },
|
||||
/// The current view: every live peer's [`NodeInfo`], unordered. Use
|
||||
/// [`view`](crate::cluster::membership::view).
|
||||
View,
|
||||
}
|
||||
|
||||
/// Replies from the manager.
|
||||
#[derive(Debug)]
|
||||
pub enum Reply {
|
||||
Registered(Registered),
|
||||
Disconnected,
|
||||
Peers(Vec<String>),
|
||||
/// `false`: another dial to this name is already in flight — do not dial.
|
||||
DialBegan(bool),
|
||||
DialEnded,
|
||||
Standing(PeerStanding),
|
||||
Subscribed,
|
||||
View(Vec<NodeInfo>),
|
||||
}
|
||||
|
||||
impl GenServer for Manager {
|
||||
type Call = Call;
|
||||
type Reply = Reply;
|
||||
type Cast = ();
|
||||
type Info = ();
|
||||
type Timer = ();
|
||||
|
||||
fn init(&mut self, ctx: &GenServerCtx<Self>) {
|
||||
self.watcher = Some(ctx.watcher());
|
||||
}
|
||||
|
||||
/// Manager shutdown drops every entry (and with it every ConnHandle);
|
||||
/// the outbound table must not outlive the connections it names.
|
||||
fn terminate(&mut self) {
|
||||
for name in self.conns.keys() {
|
||||
unbind_outbound(name);
|
||||
}
|
||||
}
|
||||
|
||||
fn handle_call(&mut self, request: Call) -> Reply {
|
||||
match request {
|
||||
Call::Register {
|
||||
peer,
|
||||
pid,
|
||||
mut handle,
|
||||
} => {
|
||||
if self.conns.contains_key(&peer.node_name) {
|
||||
// `handle` drops here: the refused actor stops itself.
|
||||
return Reply::Registered(Registered::Duplicate);
|
||||
}
|
||||
if let Some(w) = &self.watcher {
|
||||
w.watch(monitor(pid));
|
||||
}
|
||||
// The outbound table (c9) is maintained here, inside the same
|
||||
// serialized handlers that own the connection's lifetime.
|
||||
if let Some((frames, monitors)) = handle.take_outbound() {
|
||||
bind_outbound(&peer.node_name, peer.incarnation, frames, monitors);
|
||||
}
|
||||
let info = NodeInfo {
|
||||
node: self.node_id(&peer.node_name, peer.incarnation.get()),
|
||||
name: peer.node_name.clone(),
|
||||
incarnation: peer.incarnation,
|
||||
meta: peer.meta,
|
||||
};
|
||||
self.conns.insert(
|
||||
peer.node_name,
|
||||
ConnEntry {
|
||||
pid,
|
||||
info: info.clone(),
|
||||
_handle: handle,
|
||||
},
|
||||
);
|
||||
self.emit(&NodeEvent::NodeUp(info));
|
||||
Reply::Registered(Registered::Ok)
|
||||
}
|
||||
Call::Disconnect { name } => {
|
||||
// Dropping the entry drops the handle, which stops the actor.
|
||||
if let Some(entry) = self.conns.remove(&name) {
|
||||
unbind_outbound(&name);
|
||||
self.emit(&NodeEvent::NodeDown(entry.info));
|
||||
}
|
||||
Reply::Disconnected
|
||||
}
|
||||
Call::Peers => {
|
||||
let mut names: Vec<String> = self.conns.keys().cloned().collect();
|
||||
names.sort();
|
||||
Reply::Peers(names)
|
||||
}
|
||||
Call::DialBegin { name, pid } => {
|
||||
if self.dials.contains_key(&name) {
|
||||
return Reply::DialBegan(false);
|
||||
}
|
||||
if let Some(w) = &self.watcher {
|
||||
w.watch(monitor(pid));
|
||||
}
|
||||
self.dials.insert(name, pid);
|
||||
Reply::DialBegan(true)
|
||||
}
|
||||
Call::DialEnd { name } => {
|
||||
self.dials.remove(&name);
|
||||
Reply::DialEnded
|
||||
}
|
||||
Call::Standing { peer_name } => {
|
||||
Reply::Standing(if self.conns.contains_key(&peer_name) {
|
||||
PeerStanding::Claimed
|
||||
} else if self.dials.contains_key(&peer_name) {
|
||||
PeerStanding::Dialing
|
||||
} else {
|
||||
PeerStanding::Free
|
||||
})
|
||||
}
|
||||
Call::Subscribe { tx } => {
|
||||
// The snapshot: queued before `tx` joins the list, and — the
|
||||
// handlers being serialized — before any later event.
|
||||
for entry in self.conns.values() {
|
||||
let _ = tx.send(NodeEvent::NodeUp(entry.info.clone()));
|
||||
}
|
||||
self.subscribers.push(tx);
|
||||
Reply::Subscribed
|
||||
}
|
||||
Call::View => Reply::View(self.conns.values().map(|e| e.info.clone()).collect()),
|
||||
}
|
||||
}
|
||||
|
||||
fn handle_cast(&mut self, _request: ()) {}
|
||||
|
||||
fn handle_down(&mut self, down: Down) {
|
||||
let mut downs = Vec::new();
|
||||
self.conns.retain(|name, entry| {
|
||||
let dead = entry.pid == down.pid;
|
||||
if dead {
|
||||
downs.push((name.clone(), entry.info.clone()));
|
||||
}
|
||||
!dead
|
||||
});
|
||||
for (name, info) in downs {
|
||||
unbind_outbound(&name);
|
||||
self.emit(&NodeEvent::NodeDown(info));
|
||||
}
|
||||
self.dials.retain(|_, pid| *pid != down.pid);
|
||||
}
|
||||
}
|
||||
@@ -1,97 +0,0 @@
|
||||
//! RFC 010 c7a — membership: `node_up`/`node_down` events and the view.
|
||||
//!
|
||||
//! The membership *state* lives inside the [`manager`](crate::cluster::manager)
|
||||
//! — `node_up` and `node_down` are derived facts of the exact events the
|
||||
//! manager already owns (a successful registration; a reap or `Disconnect`),
|
||||
//! so holding the view anywhere else would only add a cross-actor ordering
|
||||
//! seam. This module is the consumer surface: the event and view types, and
|
||||
//! the [`subscribe`]/[`view`] entry points. No consumer ever touches the
|
||||
//! connection table (roadmap-binding, enforced by module privacy: the table
|
||||
//! is a private field, and nothing here exposes names→pids).
|
||||
//!
|
||||
//! ## Subscription semantics (ratified 2026-08-15)
|
||||
//!
|
||||
//! [`subscribe`] is **snapshot-then-stream**: the returned receiver first
|
||||
//! yields one [`NodeEvent::NodeUp`] per currently-live peer, then live events
|
||||
//! as they happen. Because the manager is a `gen_server` (handlers are
|
||||
//! serialized), the snapshot is exact — no event can interleave with it, and
|
||||
//! per-subscriber ordering matches the manager's processing order. There is
|
||||
//! no join-race for late subscribers and no separate "get, then diff" dance;
|
||||
//! [`view`] exists for observation, not for synchronization.
|
||||
//!
|
||||
//! A dropped subscriber is pruned on the next emission (its channel reports
|
||||
//! closed) — no monitor needed, the sender itself tells us.
|
||||
//!
|
||||
//! ## NodeId identity
|
||||
//!
|
||||
//! A [`NodeId`] is a compact **local alias for the wire identity**
|
||||
//! `(node_name, incarnation)`, memoized by the manager: a reconnect blip at
|
||||
//! the same incarnation keeps its id (down, then up, same id), while a
|
||||
//! restart — a new incarnation — gets a fresh one, so a node's ghost and its
|
||||
//! successor are always distinguishable. Ids are allocated from 1;
|
||||
//! [`DEFAULT_NODE_ID`](crate::pg::DEFAULT_NODE_ID) (0) remains the local
|
||||
//! node, per [`pg`](crate::pg)'s framing.
|
||||
|
||||
use crate::channel::{channel, Receiver};
|
||||
use crate::cluster::envelope::NodeMeta;
|
||||
use crate::cluster::manager::{Call, Reply, MANAGER};
|
||||
use crate::gen_server;
|
||||
use crate::pg::{Incarnation, NodeId};
|
||||
|
||||
/// One live remote node, as the view and [`NodeEvent::NodeUp`] describe it.
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct NodeInfo {
|
||||
/// The local alias for `(name, incarnation)` — see the module docs.
|
||||
pub node: NodeId,
|
||||
/// The peer's claimed node name (handshake-verified).
|
||||
pub name: String,
|
||||
/// The peer's incarnation epoch, as offered in its `Hello`.
|
||||
pub incarnation: Incarnation,
|
||||
/// The peer's `Hello` metadata.
|
||||
pub meta: NodeMeta,
|
||||
}
|
||||
|
||||
/// A membership change, as delivered to subscribers.
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub enum NodeEvent {
|
||||
/// A peer's control connection established and registered.
|
||||
NodeUp(NodeInfo),
|
||||
/// That peer's connection ended — reaped, commanded down, or the manager
|
||||
/// itself shut down. Carries the same [`NodeInfo`] the corresponding
|
||||
/// `NodeUp` delivered, so consumers need no id→name reverse map.
|
||||
NodeDown(NodeInfo),
|
||||
}
|
||||
|
||||
/// A live membership subscription: the receiving end of the event stream
|
||||
/// (the [`Monitor`](crate::monitor::Monitor) shape — read from [`rx`], drop
|
||||
/// to unsubscribe).
|
||||
///
|
||||
/// [rx]: MembershipEvents::rx
|
||||
pub struct MembershipEvents {
|
||||
/// The event stream: the snapshot's `NodeUp`s first, then live events.
|
||||
/// Fold it into a `select` from a plain actor, or pipe it into a
|
||||
/// `gen_server` via `with_info`.
|
||||
pub rx: Receiver<NodeEvent>,
|
||||
}
|
||||
|
||||
/// Subscribe to membership events (snapshot-then-stream — see the module
|
||||
/// docs). `None`: the manager is not running. Must be called from inside an
|
||||
/// actor.
|
||||
pub fn subscribe() -> Option<MembershipEvents> {
|
||||
let (tx, rx) = channel();
|
||||
match gen_server::call(MANAGER, Call::Subscribe { tx }) {
|
||||
Ok(Reply::Subscribed) => Some(MembershipEvents { rx }),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
/// The current view: every live peer's [`NodeInfo`], unordered. For
|
||||
/// observation and tests; consumers that need to *track* the view should
|
||||
/// [`subscribe`] instead (the snapshot makes the stream self-sufficient).
|
||||
/// `None`: the manager is not running. Must be called from inside an actor.
|
||||
pub fn view() -> Option<Vec<NodeInfo>> {
|
||||
match gen_server::call(MANAGER, Call::View) {
|
||||
Ok(Reply::View(v)) => Some(v),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
@@ -1,546 +0,0 @@
|
||||
//! RFC 010 c15 — distributed process groups (Phase 5).
|
||||
//!
|
||||
//! The Erlang `pg` shape (D18): every node's group store is the union of
|
||||
//! its own local members and each peer's *announced* local members. There
|
||||
//! is one **pg actor** per node — the c14 reaper grown up — and it is the
|
||||
//! only writer of remote entries and the only sender of announcements:
|
||||
//!
|
||||
//! - **Origin owns its members.** Joins are local (`pg::join`), the eager
|
||||
//! reaper is the liveness authority, and the origin announces every
|
||||
//! change: `Join`/`Leave` incrementally to every up node, and a full
|
||||
//! `Sync` of its local groups to a peer the moment that peer comes up
|
||||
//! (`NodeUp`). Nobody monitors a remote member; a peer's `NodeDown` sweeps
|
||||
//! every member it announced.
|
||||
//! - **Transport is a pure consumer** of Phase 3/4: the exposed name
|
||||
//! [`PG_NAME`] (`"pg"`) carrying [`PgMsg`] over postcard, sent with
|
||||
//! [`remote::send`]. No new frame, no manager change.
|
||||
//! - **No anti-entropy.** Per-origin ordering rides the single TCP link:
|
||||
//! the actor sends `Sync` to a peer *before* it can send that peer any
|
||||
//! `Join`/`Leave` (both from the same loop, over the same connection), and
|
||||
//! a reconnect is a fresh `NodeUp` ⇒ fresh `Sync` replacing that peer's
|
||||
//! set wholesale.
|
||||
//! - **Local API unchanged.** `members`/`pick`/`dispatch` stay local-only
|
||||
//! (`get_local_members`); a remote entry in the store carries the peer's
|
||||
//! `NodeId` and never surfaces there. Cluster-wide reads are the new,
|
||||
//! additive [`members_all`] over [`GroupMember`] (c16 adds `pick_any` /
|
||||
//! `dispatch_any`).
|
||||
//!
|
||||
//! ## Ordering inside the node
|
||||
//!
|
||||
//! `pg::join`/`pg::leave` mutate the store on the caller's thread and then
|
||||
//! *announce* to the actor's control inbox. Because the store op precedes the
|
||||
//! announcement and the actor re-reads the store before broadcasting a
|
||||
//! `Joined`, an announcement that has been overtaken (the member left or died
|
||||
//! before the actor got to it) is dropped rather than advertised: the wire
|
||||
//! never sees a `Join` for a member the origin no longer holds. `Leave`
|
||||
//! broadcasts unconditionally — a spurious `Leave` is a no-op at the peer.
|
||||
//!
|
||||
//! Inbound: `NodeUp` is emitted by the manager on the accept/connect path,
|
||||
//! *before* the peer's connection actor exists, so it is queued on the
|
||||
//! membership stream before any frame from that peer can reach this inbox.
|
||||
//! The actor still drains membership before it interprets a `PgMsg` whose
|
||||
//! sender it does not know, and drops the message if the sender is still not
|
||||
//! up (a ghost — its next `NodeUp` brings a `Sync`).
|
||||
|
||||
use std::collections::HashMap;
|
||||
|
||||
use crate::channel::{channel, select, Receiver, Selectable};
|
||||
use crate::cluster::expose::expose;
|
||||
use crate::cluster::membership::{subscribe, MembershipEvents, NodeEvent, NodeInfo};
|
||||
use crate::cluster::remote::{
|
||||
self, local_identity, send_to_remote, RemoteName, RemotePid, ToRemoteError,
|
||||
};
|
||||
use crate::monitor::Down;
|
||||
use crate::pg::{
|
||||
live, member_for, reaper_inboxes, sweep_local_death, Incarnation, Member, Membership, PgEvent,
|
||||
};
|
||||
use crate::pid::{assert_type, Addressable, Erased, Pid};
|
||||
use crate::registry::{register, send_to, SendError};
|
||||
use crate::scheduler::with_runtime;
|
||||
use crate::Name;
|
||||
|
||||
/// The exposed name every node's pg actor answers under.
|
||||
pub const PG_NAME: Name<PgMsg> = Name::new("pg");
|
||||
|
||||
/// The pg wire protocol. Every variant is origin-authored: `from` / the
|
||||
/// pid's node is the node whose local members are being described.
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub enum PgMsg {
|
||||
/// The origin's complete local membership, sent to a peer on `NodeUp`.
|
||||
/// Replaces whatever the receiver held for that origin.
|
||||
Sync {
|
||||
from: String,
|
||||
groups: Vec<(String, Vec<RemotePid<Erased>>)>,
|
||||
},
|
||||
/// The origin added `pid` (its own) to `group`.
|
||||
Join {
|
||||
group: String,
|
||||
pid: RemotePid<Erased>,
|
||||
},
|
||||
/// The origin removed `pid` from `group` — voluntary leave or death.
|
||||
Leave {
|
||||
group: String,
|
||||
pid: RemotePid<Erased>,
|
||||
},
|
||||
}
|
||||
|
||||
// Hand-rolled serde (the crate carries no serde-derive), as a 3-tuple with a
|
||||
// leading tag: (0, from, groups) | (1, group, pid) | (2, group, pid).
|
||||
impl serde::Serialize for PgMsg {
|
||||
fn serialize<S: serde::Serializer>(&self, s: S) -> Result<S::Ok, S::Error> {
|
||||
use serde::ser::SerializeTuple;
|
||||
let mut t = s.serialize_tuple(3)?;
|
||||
match self {
|
||||
PgMsg::Sync { from, groups } => {
|
||||
t.serialize_element(&0u8)?;
|
||||
t.serialize_element(from)?;
|
||||
t.serialize_element(groups)?;
|
||||
}
|
||||
PgMsg::Join { group, pid } => {
|
||||
t.serialize_element(&1u8)?;
|
||||
t.serialize_element(group)?;
|
||||
t.serialize_element(pid)?;
|
||||
}
|
||||
PgMsg::Leave { group, pid } => {
|
||||
t.serialize_element(&2u8)?;
|
||||
t.serialize_element(group)?;
|
||||
t.serialize_element(pid)?;
|
||||
}
|
||||
}
|
||||
t.end()
|
||||
}
|
||||
}
|
||||
|
||||
impl<'de> serde::Deserialize<'de> for PgMsg {
|
||||
fn deserialize<D: serde::Deserializer<'de>>(d: D) -> Result<Self, D::Error> {
|
||||
struct V;
|
||||
impl<'de> serde::de::Visitor<'de> for V {
|
||||
type Value = PgMsg;
|
||||
fn expecting(&self, f: &mut std::fmt::Formatter) -> std::fmt::Result {
|
||||
f.write_str("a pg message tuple")
|
||||
}
|
||||
fn visit_seq<A: serde::de::SeqAccess<'de>>(
|
||||
self,
|
||||
mut seq: A,
|
||||
) -> Result<PgMsg, A::Error> {
|
||||
use serde::de::Error;
|
||||
let tag: u8 = seq
|
||||
.next_element()?
|
||||
.ok_or_else(|| A::Error::custom("pg: missing tag"))?;
|
||||
let text: String = seq
|
||||
.next_element()?
|
||||
.ok_or_else(|| A::Error::custom("pg: missing name"))?;
|
||||
match tag {
|
||||
0 => {
|
||||
let groups = seq
|
||||
.next_element()?
|
||||
.ok_or_else(|| A::Error::custom("pg: missing groups"))?;
|
||||
Ok(PgMsg::Sync { from: text, groups })
|
||||
}
|
||||
1 | 2 => {
|
||||
let pid = seq
|
||||
.next_element()?
|
||||
.ok_or_else(|| A::Error::custom("pg: missing pid"))?;
|
||||
Ok(if tag == 1 {
|
||||
PgMsg::Join { group: text, pid }
|
||||
} else {
|
||||
PgMsg::Leave { group: text, pid }
|
||||
})
|
||||
}
|
||||
t => Err(A::Error::custom(format!("pg: unknown tag {t}"))),
|
||||
}
|
||||
}
|
||||
}
|
||||
d.deserialize_tuple(3, V)
|
||||
}
|
||||
}
|
||||
|
||||
/// A member of a group as the cluster sees it: on this node (a plain
|
||||
/// [`Pid`], sendable locally) or on a peer (a [`RemotePid`], sendable via
|
||||
/// [`send_to_remote`](remote::send_to_remote)). `Pid` cannot hold a remote
|
||||
/// (D14), hence the two-variant shape.
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub enum GroupMember {
|
||||
Local(Pid),
|
||||
Remote(RemotePid<Erased>),
|
||||
}
|
||||
|
||||
/// Every member of `group` cluster-wide, in the store's order: local members
|
||||
/// filtered by the same liveness backstop as [`members`](crate::pg::members),
|
||||
/// remote members exactly as their origins last announced them. Must run
|
||||
/// inside [`run`](crate::run).
|
||||
pub fn members_all(group: &str) -> Vec<GroupMember> {
|
||||
with_runtime(|inner| {
|
||||
let me = inner.node_id;
|
||||
let pg = inner.process_groups.lock();
|
||||
pg.all_of(group)
|
||||
.into_iter()
|
||||
.filter_map(|m| {
|
||||
if m.node == me {
|
||||
live(inner, m.pid).then_some(GroupMember::Local(m.pid))
|
||||
} else {
|
||||
// A remote entry always has its node's name recorded
|
||||
// (they land under the same lock); a missing one is a
|
||||
// node already swept, so it hides rather than misnames.
|
||||
pg.node_name(m.node).map(|name| {
|
||||
GroupMember::Remote(RemotePid::from_parts(
|
||||
name,
|
||||
m.incarnation,
|
||||
m.pid.index(),
|
||||
m.pid.generation(),
|
||||
))
|
||||
})
|
||||
}
|
||||
})
|
||||
.collect()
|
||||
})
|
||||
}
|
||||
|
||||
/// One member of `group` cluster-wide, or `None` if it has none: the first
|
||||
/// entry in the store's order (this node's members in join order first when
|
||||
/// they joined first — the same stateless first-live scan as
|
||||
/// [`pick`](crate::pg::pick), extended over the peers' announced members).
|
||||
/// Must run inside [`run`](crate::run).
|
||||
pub fn pick_any(group: &str) -> Option<GroupMember> {
|
||||
members_all(group).into_iter().next()
|
||||
}
|
||||
|
||||
/// Why [`dispatch_any`] handed `msg` back.
|
||||
#[derive(Debug)]
|
||||
pub enum DispatchAnyError<M> {
|
||||
/// The group has no member anywhere.
|
||||
NoMember(M),
|
||||
/// The pick was local and the local typed send failed.
|
||||
Local(SendError<M>),
|
||||
/// The pick was remote and the remote send failed at this node.
|
||||
Remote(ToRemoteError<M>),
|
||||
}
|
||||
|
||||
impl<M> DispatchAnyError<M> {
|
||||
/// The undelivered message.
|
||||
pub fn into_inner(self) -> M {
|
||||
match self {
|
||||
DispatchAnyError::NoMember(m) => m,
|
||||
DispatchAnyError::Local(e) => e.into_inner(),
|
||||
DispatchAnyError::Remote(e) => e.into_inner(),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// [`pick_any`] and send in one step, returning the member reached: a local
|
||||
/// pick goes through [`send_to`], a remote one through [`send_to_remote`]
|
||||
/// (so `Ok` for a remote member means "handed to the connection", RFC 010
|
||||
/// §3). Homogeneous pool assumed, as for [`dispatch`](crate::pg::dispatch);
|
||||
/// a wrong `A` degrades to a clean error at the target, never a misroute.
|
||||
/// Must run inside [`run`](crate::run).
|
||||
pub fn dispatch_any<A>(group: &str, msg: A::Msg) -> Result<GroupMember, DispatchAnyError<A::Msg>>
|
||||
where
|
||||
A: Addressable,
|
||||
A::Msg: serde::Serialize,
|
||||
{
|
||||
match pick_any(group) {
|
||||
None => Err(DispatchAnyError::NoMember(msg)),
|
||||
Some(GroupMember::Local(pid)) => send_to(assert_type::<A>(pid), msg)
|
||||
.map(|()| GroupMember::Local(pid))
|
||||
.map_err(DispatchAnyError::Local),
|
||||
Some(GroupMember::Remote(rp)) => send_to_remote(rp.clone().assert_type::<A>(), msg)
|
||||
.map(|()| GroupMember::Remote(rp))
|
||||
.map_err(DispatchAnyError::Remote),
|
||||
}
|
||||
}
|
||||
|
||||
/// Attach the pg actor to the running cluster. Called once by
|
||||
/// `cluster::start` after the manager is up and the local identity is set;
|
||||
/// spawns the actor if this run has not joined anything yet.
|
||||
pub(crate) fn attach_cluster() {
|
||||
let _ = reaper_inboxes().ctl.send(PgEvent::Attach);
|
||||
}
|
||||
|
||||
/// The attached half of the actor's state: who is up (by name) and the
|
||||
/// membership stream.
|
||||
struct Attached {
|
||||
events: MembershipEvents,
|
||||
peers: HashMap<String, NodeInfo>,
|
||||
/// This node's wire identity: `Sync`'s `from`, and the stamp on every
|
||||
/// pid we ship (attach requires it, so no `None` path exists here).
|
||||
me: String,
|
||||
incarnation: Incarnation,
|
||||
}
|
||||
|
||||
/// The pg actor: the c14 reaper (`deaths`), the local API's announcements
|
||||
/// (`ctl`), and — once attached — the membership stream and the exposed
|
||||
/// `"pg"` inbox, all in one drain-then-select loop. `deaths`/`ctl` closing
|
||||
/// is the run tearing down; the membership stream closing is the manager
|
||||
/// gone (detach, keep reaping).
|
||||
pub(crate) fn actor(deaths: Receiver<Down>, ctl: Receiver<PgEvent>) {
|
||||
let (pg_tx, pg_rx) = channel::<PgMsg>();
|
||||
let mut cl: Option<Attached> = None;
|
||||
loop {
|
||||
loop {
|
||||
match deaths.try_recv() {
|
||||
Ok(Some(down)) => on_death(cl.as_ref(), down.pid),
|
||||
Ok(None) => break,
|
||||
Err(_) => return,
|
||||
}
|
||||
}
|
||||
loop {
|
||||
match ctl.try_recv() {
|
||||
Ok(Some(PgEvent::Attach)) => {
|
||||
// Own the name BEFORE subscribing (which yields to the
|
||||
// manager): a peer's first frame must find "pg" exposed
|
||||
// and resolvable, or it is dropped. Idempotent for a
|
||||
// re-attach: same actor, same channel (the registry
|
||||
// refuses a *second* live one).
|
||||
let _ = register(PG_NAME, pg_tx.clone());
|
||||
expose(PG_NAME);
|
||||
if let Some(a) = attach() {
|
||||
cl = Some(a);
|
||||
}
|
||||
}
|
||||
Ok(Some(PgEvent::Joined { group, pid })) => on_joined(cl.as_ref(), &group, pid),
|
||||
Ok(Some(PgEvent::Left { group, pid })) => on_left(cl.as_ref(), &group, pid),
|
||||
Ok(None) => break,
|
||||
Err(_) => return,
|
||||
}
|
||||
}
|
||||
if let Some(a) = cl.as_mut() {
|
||||
if !drain_events(a) {
|
||||
cl = None;
|
||||
continue;
|
||||
}
|
||||
loop {
|
||||
match pg_rx.try_recv() {
|
||||
Ok(Some(msg)) => on_msg(a, msg),
|
||||
Ok(None) => break,
|
||||
Err(_) => return, // our own inbox: only on teardown
|
||||
}
|
||||
}
|
||||
}
|
||||
// Wait. Control first (attach/teardown must be prompt), then deaths,
|
||||
// then the cluster arms.
|
||||
let mut arms: Vec<&dyn Selectable> = vec![&ctl, &deaths];
|
||||
if let Some(a) = cl.as_ref() {
|
||||
arms.push(&a.events.rx);
|
||||
arms.push(&pg_rx);
|
||||
}
|
||||
let _ = select(&arms);
|
||||
}
|
||||
}
|
||||
|
||||
fn attach() -> Option<Attached> {
|
||||
let events = subscribe()?;
|
||||
let (me, incarnation) = local_identity()?;
|
||||
Some(Attached {
|
||||
events,
|
||||
peers: HashMap::new(),
|
||||
me,
|
||||
incarnation,
|
||||
})
|
||||
}
|
||||
|
||||
/// Fold pending membership events: `NodeUp` ⇒ record + `Sync` that peer;
|
||||
/// `NodeDown` ⇒ sweep every member it announced. `false` when the stream
|
||||
/// has closed.
|
||||
fn drain_events(a: &mut Attached) -> bool {
|
||||
loop {
|
||||
match a.events.rx.try_recv() {
|
||||
Ok(Some(NodeEvent::NodeUp(info))) => {
|
||||
let name = info.name.clone();
|
||||
a.peers.insert(name.clone(), info);
|
||||
// Snapshot under the store lock, then stamp wire pids
|
||||
// outside it (`from_local` marks watchable under the slot's
|
||||
// cold lock — Leaf-on-Leaf nesting is asserted).
|
||||
let local: Vec<(String, Vec<Pid>)> =
|
||||
with_runtime(|inner| inner.process_groups.lock().groups_on(inner.node_id));
|
||||
let groups = local
|
||||
.into_iter()
|
||||
.map(|(g, pids)| (g, pids.into_iter().map(|p| wire(a, p)).collect()))
|
||||
.collect();
|
||||
let msg = PgMsg::Sync {
|
||||
from: a.me.clone(),
|
||||
groups,
|
||||
};
|
||||
let _ = remote::send(RemoteName::new(name, PG_NAME), msg);
|
||||
}
|
||||
Ok(Some(NodeEvent::NodeDown(info))) => {
|
||||
a.peers.remove(&info.name);
|
||||
with_runtime(|inner| {
|
||||
let mut pg = inner.process_groups.lock();
|
||||
pg.remove_where(|m| m.node == info.node);
|
||||
pg.forget_node_name(info.node);
|
||||
});
|
||||
}
|
||||
Ok(None) => return true,
|
||||
Err(_) => return false,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Send `msg` to every up peer. `NotConnected` is ignored: that peer's
|
||||
/// `NodeDown` is on its way and its next `NodeUp` gets a `Sync`.
|
||||
fn broadcast(a: &Attached, msg: PgMsg) {
|
||||
for name in a.peers.keys() {
|
||||
let _ = remote::send(RemoteName::new(name.clone(), PG_NAME), msg.clone());
|
||||
}
|
||||
}
|
||||
|
||||
fn on_death(a: Option<&Attached>, pid: Pid) {
|
||||
let evicted = sweep_local_death(pid);
|
||||
if let Some(a) = a {
|
||||
for (group, ms) in evicted {
|
||||
broadcast(
|
||||
a,
|
||||
PgMsg::Leave {
|
||||
group,
|
||||
pid: wire(a, ms.member.pid),
|
||||
},
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn on_joined(a: Option<&Attached>, group: &str, pid: Pid) {
|
||||
let Some(a) = a else { return };
|
||||
// Re-check: a leave/death may have overtaken the announcement.
|
||||
let still = with_runtime(|inner| {
|
||||
let m = member_for(inner, pid);
|
||||
inner.process_groups.lock().contains(group, &m)
|
||||
});
|
||||
if still {
|
||||
broadcast(
|
||||
a,
|
||||
PgMsg::Join {
|
||||
group: group.to_owned(),
|
||||
pid: wire(a, pid),
|
||||
},
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
fn on_left(a: Option<&Attached>, group: &str, pid: Pid) {
|
||||
let Some(a) = a else { return };
|
||||
broadcast(
|
||||
a,
|
||||
PgMsg::Leave {
|
||||
group: group.to_owned(),
|
||||
pid: wire(a, pid),
|
||||
},
|
||||
);
|
||||
}
|
||||
|
||||
/// The wire form of a local member pid, stamped with the identity the
|
||||
/// actor was attached with (marks watchable, like `from_local`).
|
||||
fn wire(a: &Attached, pid: Pid) -> RemotePid<Erased> {
|
||||
RemotePid::from_local_at(pid, a.me.clone(), a.incarnation)
|
||||
}
|
||||
|
||||
/// The named origin's `NodeInfo`, if it is up. A second look at the
|
||||
/// membership stream covers a `NodeUp` that landed after this loop
|
||||
/// iteration's drain; anything still unknown is a ghost and is dropped.
|
||||
fn origin(a: &mut Attached, name: &str) -> Option<NodeInfo> {
|
||||
if let Some(i) = a.peers.get(name) {
|
||||
return Some(i.clone());
|
||||
}
|
||||
drain_events(a);
|
||||
a.peers.get(name).cloned()
|
||||
}
|
||||
|
||||
/// `origin`, additionally requiring `pid` to be stamped with the origin's
|
||||
/// current incarnation — a pid from a previous life of that node is a ghost.
|
||||
fn origin_of(a: &mut Attached, pid: &RemotePid<Erased>) -> Option<NodeInfo> {
|
||||
origin(a, pid.node()).filter(|i| i.incarnation == pid.incarnation())
|
||||
}
|
||||
|
||||
fn remote_membership(origin: &NodeInfo, pid: &RemotePid<Erased>) -> Membership {
|
||||
Membership {
|
||||
member: Member {
|
||||
node: origin.node,
|
||||
incarnation: origin.incarnation,
|
||||
pid: Pid::new(pid.index(), pid.generation()),
|
||||
},
|
||||
monitor: None,
|
||||
}
|
||||
}
|
||||
|
||||
fn on_msg(a: &mut Attached, msg: PgMsg) {
|
||||
match msg {
|
||||
PgMsg::Sync { from, groups } => {
|
||||
let Some(info) = origin(a, &from) else { return };
|
||||
with_runtime(|inner| {
|
||||
let mut pg = inner.process_groups.lock();
|
||||
pg.remove_where(|m| m.node == info.node);
|
||||
pg.set_node_name(info.node, info.name.clone());
|
||||
for (group, pids) in &groups {
|
||||
// Origin-authored: only its own current-incarnation pids.
|
||||
for p in pids
|
||||
.iter()
|
||||
.filter(|p| p.node() == from && p.incarnation() == info.incarnation)
|
||||
{
|
||||
pg.join(group, remote_membership(&info, p));
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
PgMsg::Join { group, pid } => {
|
||||
let Some(info) = origin_of(a, &pid) else {
|
||||
return;
|
||||
};
|
||||
with_runtime(|inner| {
|
||||
let mut pg = inner.process_groups.lock();
|
||||
pg.set_node_name(info.node, info.name.clone());
|
||||
pg.join(&group, remote_membership(&info, &pid));
|
||||
});
|
||||
}
|
||||
PgMsg::Leave { group, pid } => {
|
||||
let Some(info) = origin_of(a, &pid) else {
|
||||
return;
|
||||
};
|
||||
with_runtime(|inner| {
|
||||
let ms = remote_membership(&info, &pid);
|
||||
inner.process_groups.lock().leave(&group, ms.member);
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::cluster::envelope::{decode_payload, encode_payload};
|
||||
|
||||
#[test]
|
||||
fn pg_msg_roundtrips_every_variant() {
|
||||
let p = RemotePid::<Erased>::from_parts("a", Incarnation::new(9), 3, 1);
|
||||
for m in [
|
||||
PgMsg::Sync {
|
||||
from: "a".into(),
|
||||
groups: vec![
|
||||
("g".into(), vec![p.clone(), p.clone()]),
|
||||
("h".into(), vec![]),
|
||||
],
|
||||
},
|
||||
PgMsg::Sync {
|
||||
from: "a".into(),
|
||||
groups: vec![],
|
||||
},
|
||||
PgMsg::Join {
|
||||
group: "g".into(),
|
||||
pid: p.clone(),
|
||||
},
|
||||
PgMsg::Leave {
|
||||
group: "g".into(),
|
||||
pid: p.clone(),
|
||||
},
|
||||
] {
|
||||
let bytes = encode_payload(&m).unwrap();
|
||||
let back: PgMsg = decode_payload(&bytes).unwrap();
|
||||
assert_eq!(back, m);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn pg_msg_rejects_unknown_tag() {
|
||||
let bytes = encode_payload(&(7u8, "x", 0u32)).unwrap();
|
||||
assert!(decode_payload::<PgMsg>(&bytes).is_err());
|
||||
}
|
||||
}
|
||||
@@ -1,835 +0,0 @@
|
||||
//! RFC 010 c9 — remote `Name` sends: the outbound path and the single
|
||||
//! inbound name-resolution seam.
|
||||
//!
|
||||
//! ## Outbound (D13, ratified 2026-08-15)
|
||||
//!
|
||||
//! A module-private table `node name → Sender<Frame>` — one dedicated
|
||||
//! outbound channel per live connection, populated and torn down by the
|
||||
//! manager inside the same serialized handlers that own the connection's
|
||||
//! lifetime (Register / Disconnect / reap), living on `RuntimeInner` beside
|
||||
//! the exposure state. [`send`] is one leaf-lock lookup + one channel send:
|
||||
//! no gen_server on the data plane, no published channel anyone holding a
|
||||
//! pid could inject raw frames into. `Ok(())` means **handed to the
|
||||
//! connection's inbox** — local knowledge only, exactly the BEAM contract
|
||||
//! (RFC §3): a missing entry or a closed channel is
|
||||
//! [`RemoteSendError::NotConnected`]; delivery confirmation is the monitor's
|
||||
//! job (c11). The entry-present/actor-dying-mid-send window is *honest*
|
||||
//! under that contract, not a bug.
|
||||
//!
|
||||
//! The outbound sender is deliberately **separate from the conn actor's
|
||||
//! command channel**: if it were a clone of `cmd_tx`, the manager dropping
|
||||
//! its `ConnHandle` would no longer close that channel and connection
|
||||
//! lifetime would leak to whoever holds a sender — a D9 violation.
|
||||
//!
|
||||
//! Buffering is unbounded toward a slow peer (the BEAM `busy_dist_port`
|
||||
//! shape); backpressure is out of c9's scope and noted here rather than
|
||||
//! silently absent.
|
||||
//!
|
||||
//! ## Inbound — the ONE resolution seam (RFC v2)
|
||||
//!
|
||||
//! Every wire-name → local-pid resolution goes through [`deliver_named`],
|
||||
//! and nothing else: the conn actor hands it the three fields of a
|
||||
//! `SendNamed` and gets back a verdict. It checks the exposed set first (an
|
||||
//! unexposed name is unreachable — the gun's safety), then the type hash
|
||||
//! against what the name was exposed with, then resolves the name through
|
||||
//! the registry and delivers via c8's [`decode_deliver`]. When an owned-name
|
||||
//! table lands beside the `&'static str` registry, it slots in here without
|
||||
//! touching call sites. Module privacy enforces the funnel: the exposed and
|
||||
//! outbound tables are `pub(crate)`, and no other module resolves names for
|
||||
//! the wire.
|
||||
//!
|
||||
//! Refusals are silent to the sender by design (§3: send failure reflects
|
||||
//! local knowledge only); they are observable locally as the returned
|
||||
//! [`InboundVerdict`], which the conn actor may log or count.
|
||||
//!
|
||||
//! ## Pids (c10, D14)
|
||||
//!
|
||||
//! [`RemotePid<A>`] = `(node_name, incarnation, index, generation)` +
|
||||
//! phantom — identity-bound, dead when that incarnation dies, never
|
||||
//! redirects. The node travels as its **name** (a global identifier, so a pid
|
||||
//! forwarded through a third node needs no re-mapping); NodeId is a local
|
||||
//! alias and never crosses. A local `Pid<A>` serializes *as* a `RemotePid`
|
||||
//! stamped from the ambient [local identity](set_local_identity); a
|
||||
//! `RemotePid` deserializes into `Pid<A>` only when it names this node (the
|
||||
//! collapse), else it is a decode error — fields that may hold a pid from
|
||||
//! anywhere are typed `RemotePid<A>`.
|
||||
//!
|
||||
//! [`send_to_remote`] is the pid-targeted send. A self-node pid short-
|
||||
//! circuits to the local typed send with the message object itself — no
|
||||
//! encode, no frame (zero-copy-equivalent). Otherwise the outbound table
|
||||
//! (widened to carry each node's **current incarnation**) does the RFC v2 §3
|
||||
//! check at the send site: a pid of a dead incarnation is
|
||||
//! [`ToRemoteError::DeadIncarnation`] and no frame is emitted. Inbound
|
||||
//! `Send` frames are delivered by index/generation through c8's
|
||||
//! [`decode_deliver`]: the target actor's published channel for the exposed
|
||||
//! type is the only route (the reply-to path requires
|
||||
//! [`expose_type`](crate::cluster::expose::expose_type) at the receiver).
|
||||
|
||||
use std::cell::Cell;
|
||||
use std::collections::HashMap;
|
||||
use std::marker::PhantomData;
|
||||
|
||||
use crate::channel::{channel, Receiver, RecvError, Selectable, Sender};
|
||||
use crate::cluster::envelope::{encode_payload, Frame, PayloadError, RemoteDownReason};
|
||||
use crate::cluster::expose::{decode_deliver, exposed_hash, type_hash, DeliverError};
|
||||
use crate::monitor::{demonitor, monitor, Monitor, MonitorId};
|
||||
use crate::pg::Incarnation;
|
||||
use crate::pid::{Addressable, Erased, Name, Pid};
|
||||
use crate::registry::{send_to, whereis, SendError};
|
||||
use crate::scheduler::with_runtime;
|
||||
|
||||
/// A name on a specific remote node: `(node_name, Name<M>)`. Sendable via
|
||||
/// [`send`]; typed, so the payload is `M` and the wire hash is
|
||||
/// [`type_hash::<M>()`](type_hash).
|
||||
pub struct RemoteName<M> {
|
||||
node: String,
|
||||
name: Name<M>,
|
||||
_marker: PhantomData<fn() -> M>,
|
||||
}
|
||||
|
||||
impl<M> RemoteName<M> {
|
||||
pub fn new(node: impl Into<String>, name: Name<M>) -> Self {
|
||||
RemoteName {
|
||||
node: node.into(),
|
||||
name,
|
||||
_marker: PhantomData,
|
||||
}
|
||||
}
|
||||
pub fn node(&self) -> &str {
|
||||
&self.node
|
||||
}
|
||||
pub fn name(&self) -> Name<M> {
|
||||
self.name
|
||||
}
|
||||
}
|
||||
|
||||
impl<M> Clone for RemoteName<M> {
|
||||
fn clone(&self) -> Self {
|
||||
RemoteName {
|
||||
node: self.node.clone(),
|
||||
name: self.name,
|
||||
_marker: PhantomData,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl<M> std::fmt::Debug for RemoteName<M> {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
write!(f, "{}@{}", self.name.as_str(), self.node)
|
||||
}
|
||||
}
|
||||
|
||||
/// Why a remote send did not leave this node. Local knowledge only.
|
||||
#[derive(Debug)]
|
||||
pub enum RemoteSendError<M> {
|
||||
/// No live connection to that node right now (never connected, or gone
|
||||
/// and not yet re-dialed). The message is handed back.
|
||||
NotConnected(M),
|
||||
/// The payload did not serialize.
|
||||
Encode(M, PayloadError),
|
||||
}
|
||||
|
||||
impl<M> RemoteSendError<M> {
|
||||
pub fn into_inner(self) -> M {
|
||||
match self {
|
||||
RemoteSendError::NotConnected(m) | RemoteSendError::Encode(m, _) => m,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// The outbound table, one per runtime (a `RuntimeInner` field): per live
|
||||
/// node, its current incarnation (the RFC v2 §3 send-site check) and the
|
||||
/// connection's dedicated outbound sender. Plus this node's own wire
|
||||
/// identity, which pid serialization stamps.
|
||||
pub(crate) struct Outbound {
|
||||
by_node: HashMap<String, Route>,
|
||||
local: Option<(String, Incarnation)>,
|
||||
}
|
||||
|
||||
/// One live connection as the outbound path sees it: the peer's current
|
||||
/// incarnation and the two inboxes of its connection actor — frames (c9)
|
||||
/// and monitor bookkeeping (c12, [`MonCmd`]).
|
||||
pub(crate) struct Route {
|
||||
incarnation: Incarnation,
|
||||
frames: Sender<Frame>,
|
||||
monitors: Sender<MonCmd>,
|
||||
}
|
||||
|
||||
impl Outbound {
|
||||
pub(crate) fn new() -> Self {
|
||||
Outbound {
|
||||
by_node: HashMap::new(),
|
||||
local: None,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Set this node's wire identity — what serialized pids are stamped with
|
||||
/// and what a `RemotePid` must name to collapse. `cluster::start` sets it;
|
||||
/// exposed for local tests. Must run inside [`run`](crate::run).
|
||||
pub fn set_local_identity(node: &str, incarnation: Incarnation) {
|
||||
with_runtime(|inner| {
|
||||
inner.outbound.lock().local = Some((node.to_string(), incarnation));
|
||||
});
|
||||
}
|
||||
|
||||
/// This node's wire identity, if set. Must run inside [`run`](crate::run).
|
||||
pub fn local_identity() -> Option<(String, Incarnation)> {
|
||||
with_runtime(|inner| inner.outbound.lock().local.clone())
|
||||
}
|
||||
|
||||
/// Manager-only: bind `node`'s outbound channels at `incarnation`. Called
|
||||
/// inside `Register`.
|
||||
pub(crate) fn bind_outbound(
|
||||
node: &str,
|
||||
incarnation: Incarnation,
|
||||
frames: Sender<Frame>,
|
||||
monitors: Sender<MonCmd>,
|
||||
) {
|
||||
with_runtime(|inner| {
|
||||
inner.outbound.lock().by_node.insert(
|
||||
node.to_string(),
|
||||
Route {
|
||||
incarnation,
|
||||
frames,
|
||||
monitors,
|
||||
},
|
||||
);
|
||||
});
|
||||
}
|
||||
|
||||
/// Test probe: bind an arbitrary sender as `node`'s outbound so a test can
|
||||
/// assert what frames leave — or don't. Same table, same lookup as the real
|
||||
/// path (this is how "no frame emitted" is asserted at the frame level).
|
||||
/// Frames only: there is no connection actor behind a probe, so a
|
||||
/// [`monitor_remote`] against a probed node reports `Disconnected`.
|
||||
pub fn bind_outbound_probe(node: &str, incarnation: Incarnation, tx: Sender<Frame>) {
|
||||
drop(bind_outbound_probe_with_monitors(node, incarnation, tx));
|
||||
}
|
||||
|
||||
/// The monitor half of a probed node's inbox: opaque, held only to be
|
||||
/// dropped. See [`bind_outbound_probe_with_monitors`].
|
||||
pub struct MonitorInbox {
|
||||
_rx: Receiver<MonCmd>,
|
||||
}
|
||||
|
||||
/// Test probe: like [`bind_outbound_probe`], but the monitor-command
|
||||
/// receiver is handed back instead of dropped, so a test can stage the
|
||||
/// c13 drain gap — a `Monitor` command that reached the connection's inbox
|
||||
/// and dies unread when the inbox is dropped. While the inbox lives,
|
||||
/// [`monitor_remote`] against the probed node is simply in flight.
|
||||
pub fn bind_outbound_probe_with_monitors(
|
||||
node: &str,
|
||||
incarnation: Incarnation,
|
||||
tx: Sender<Frame>,
|
||||
) -> MonitorInbox {
|
||||
let (mon_tx, mon_rx) = channel();
|
||||
bind_outbound(node, incarnation, tx, mon_tx);
|
||||
MonitorInbox { _rx: mon_rx }
|
||||
}
|
||||
|
||||
/// Manager-only: unbind `node`'s outbound channel. Called on `Disconnect`,
|
||||
/// reap, and manager shutdown. Dropping the sender is what closes the conn
|
||||
/// actor's outbound arm — but that arm's closure is NOT a stop signal (the
|
||||
/// cmd channel is, per D9); the actor simply stops selecting on it.
|
||||
pub(crate) fn unbind_outbound(node: &str) {
|
||||
with_runtime(|inner| {
|
||||
inner.outbound.lock().by_node.remove(node);
|
||||
});
|
||||
}
|
||||
|
||||
/// Send `msg` to `target`. `Ok(())` = handed to the connection's inbox, and
|
||||
/// nothing more — see the module docs. Must run inside
|
||||
/// [`run`](crate::run).
|
||||
pub fn send<M>(target: RemoteName<M>, msg: M) -> Result<(), RemoteSendError<M>>
|
||||
where
|
||||
M: serde::Serialize + Send + 'static,
|
||||
{
|
||||
let payload = match encode_payload(&msg) {
|
||||
Ok(p) => p,
|
||||
Err(e) => return Err(RemoteSendError::Encode(msg, e)),
|
||||
};
|
||||
let frame = Frame::SendNamed {
|
||||
name: target.name.as_str().to_string(),
|
||||
type_hash: type_hash::<M>(),
|
||||
payload,
|
||||
};
|
||||
match hand_to_connection(&target.node, frame) {
|
||||
Ok(()) => Ok(()),
|
||||
Err(NotConnected) => Err(RemoteSendError::NotConnected(msg)),
|
||||
}
|
||||
}
|
||||
|
||||
/// No live connection to the named node — the payload-free form of
|
||||
/// [`RemoteSendError::NotConnected`], for the raw path.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub struct NotConnected;
|
||||
|
||||
/// The untyped escape hatch: send pre-encoded `payload` under an explicit
|
||||
/// `type_hash`. Exists so tests (and future codecs) can put deliberately
|
||||
/// wrong frames on the wire; the typed [`send`] cannot express a hash/type
|
||||
/// mismatch, by design. Same `Ok` semantics as [`send`].
|
||||
pub fn send_remote_raw(
|
||||
node: &str,
|
||||
name: &str,
|
||||
type_hash: u64,
|
||||
payload: &[u8],
|
||||
) -> Result<(), NotConnected> {
|
||||
hand_to_connection(
|
||||
node,
|
||||
Frame::SendNamed {
|
||||
name: name.to_string(),
|
||||
type_hash,
|
||||
payload: payload.to_vec(),
|
||||
},
|
||||
)
|
||||
}
|
||||
|
||||
/// One lookup, one send. Clone the sender out under the lock and send
|
||||
/// outside it (a channel send can unpark the conn actor).
|
||||
fn hand_to_connection(node: &str, frame: Frame) -> Result<(), NotConnected> {
|
||||
let tx = with_runtime(|inner| {
|
||||
inner
|
||||
.outbound
|
||||
.lock()
|
||||
.by_node
|
||||
.get(node)
|
||||
.map(|r| r.frames.clone())
|
||||
});
|
||||
match tx {
|
||||
Some(tx) => tx.send(frame).map_err(|_| NotConnected),
|
||||
None => Err(NotConnected),
|
||||
}
|
||||
}
|
||||
|
||||
// ---- pids ---------------------------------------------------------------
|
||||
|
||||
/// A pid on some node: `(node_name, incarnation, index, generation)` plus
|
||||
/// the actor type. See the module docs. Serializes as a 4-tuple.
|
||||
pub struct RemotePid<A> {
|
||||
node: String,
|
||||
incarnation: Incarnation,
|
||||
index: u32,
|
||||
generation: u32,
|
||||
_marker: PhantomData<fn() -> A>,
|
||||
}
|
||||
|
||||
impl<A> RemotePid<A> {
|
||||
/// Build from raw parts (tests, and codecs re-hydrating a pid).
|
||||
pub fn from_parts(
|
||||
node: impl Into<String>,
|
||||
incarnation: Incarnation,
|
||||
index: u32,
|
||||
generation: u32,
|
||||
) -> Self {
|
||||
RemotePid {
|
||||
node: node.into(),
|
||||
incarnation,
|
||||
index,
|
||||
generation,
|
||||
_marker: PhantomData,
|
||||
}
|
||||
}
|
||||
|
||||
/// The wire form of a local pid, stamped with this node's identity, and
|
||||
/// **marked watchable** — asking for the wire form *is* the intent to
|
||||
/// ship the pid, so this is the same D12 set-site as `Pid::serialize`
|
||||
/// (c12 made it explicit: a peer may monitor exactly the pids that
|
||||
/// crossed, and a pid handed out via `from_local` in a hand-built reply
|
||||
/// has crossed). Must run inside [`run`](crate::run).
|
||||
///
|
||||
/// `None` when this runtime has no wire identity (no `cluster::start`,
|
||||
/// no [`set_local_identity`]): such a pid cannot name a node, and a
|
||||
/// stamped `("", 0)` would be dropped by every peer with no signal.
|
||||
/// The pid is not marked watchable in that case either.
|
||||
pub fn from_local(pid: Pid<A>) -> Option<Self> {
|
||||
let (node, incarnation) = local_identity()?;
|
||||
Some(Self::from_local_at(pid, node, incarnation))
|
||||
}
|
||||
|
||||
/// `from_local` with the identity supplied by the caller — for a holder
|
||||
/// that already carries the node's identity (the pg actor) and must not
|
||||
/// have a `None` path. Marks watchable like `from_local`.
|
||||
pub(crate) fn from_local_at(pid: Pid<A>, node: String, incarnation: Incarnation) -> Self {
|
||||
crate::monitor::mark_watchable(pid);
|
||||
RemotePid::from_parts(node, incarnation, pid.index(), pid.generation())
|
||||
}
|
||||
|
||||
/// The collapse: `Some(local pid)` iff this pid names this very node
|
||||
/// (name and incarnation). Must run inside [`run`](crate::run).
|
||||
pub fn local(&self) -> Option<Pid<A>> {
|
||||
let (n, i) = local_identity()?;
|
||||
(n == self.node && i == self.incarnation)
|
||||
.then(|| crate::pid::assert_type::<A>(Pid::new(self.index, self.generation)))
|
||||
}
|
||||
|
||||
/// Drop the actor type: the untyped `RemotePid<Erased>`, the form
|
||||
/// [`RemoteDown`] and [`RemoteMonitor`] carry (mirrors [`Pid::erase`]).
|
||||
pub fn erase(self) -> RemotePid<Erased> {
|
||||
RemotePid::from_parts(self.node, self.incarnation, self.index, self.generation)
|
||||
}
|
||||
|
||||
/// Re-type an erased pid as `RemotePid<B>` — the unchecked mirror of
|
||||
/// `pid::assert_type`, with the same degradation: a wrong `B` means the
|
||||
/// target refuses the payload's hash (never a misroute).
|
||||
pub(crate) fn assert_type<B>(self) -> RemotePid<B> {
|
||||
RemotePid::from_parts(self.node, self.incarnation, self.index, self.generation)
|
||||
}
|
||||
|
||||
pub fn node(&self) -> &str {
|
||||
&self.node
|
||||
}
|
||||
pub fn incarnation(&self) -> Incarnation {
|
||||
self.incarnation
|
||||
}
|
||||
pub fn index(&self) -> u32 {
|
||||
self.index
|
||||
}
|
||||
pub fn generation(&self) -> u32 {
|
||||
self.generation
|
||||
}
|
||||
}
|
||||
|
||||
impl<A> Clone for RemotePid<A> {
|
||||
fn clone(&self) -> Self {
|
||||
RemotePid::from_parts(
|
||||
self.node.clone(),
|
||||
self.incarnation,
|
||||
self.index,
|
||||
self.generation,
|
||||
)
|
||||
}
|
||||
}
|
||||
impl<A> PartialEq for RemotePid<A> {
|
||||
fn eq(&self, o: &Self) -> bool {
|
||||
self.node == o.node
|
||||
&& self.incarnation == o.incarnation
|
||||
&& self.index == o.index
|
||||
&& self.generation == o.generation
|
||||
}
|
||||
}
|
||||
impl<A> Eq for RemotePid<A> {}
|
||||
impl<A> std::fmt::Debug for RemotePid<A> {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
write!(
|
||||
f,
|
||||
"<{}.{}@{}#{}>",
|
||||
self.index,
|
||||
self.generation,
|
||||
self.node,
|
||||
self.incarnation.get()
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
impl<A> serde::Serialize for RemotePid<A> {
|
||||
fn serialize<S: serde::Serializer>(&self, s: S) -> Result<S::Ok, S::Error> {
|
||||
(
|
||||
self.node.as_str(),
|
||||
self.incarnation.get(),
|
||||
self.index,
|
||||
self.generation,
|
||||
)
|
||||
.serialize(s)
|
||||
}
|
||||
}
|
||||
impl<'de, A> serde::Deserialize<'de> for RemotePid<A> {
|
||||
fn deserialize<D: serde::Deserializer<'de>>(d: D) -> Result<Self, D::Error> {
|
||||
let (node, inc, index, generation) = <(String, u32, u32, u32)>::deserialize(d)?;
|
||||
Ok(RemotePid::from_parts(
|
||||
node,
|
||||
Incarnation::new(inc),
|
||||
index,
|
||||
generation,
|
||||
))
|
||||
}
|
||||
}
|
||||
|
||||
/// Why a pid-targeted send did not leave this node. Local knowledge only.
|
||||
#[derive(Debug)]
|
||||
pub enum ToRemoteError<M> {
|
||||
/// No live connection to the pid's node.
|
||||
NotConnected(M),
|
||||
/// The pid's incarnation is not that node's current one (RFC v2 §3): the
|
||||
/// actor died with its incarnation. Detected at the send site; no frame.
|
||||
DeadIncarnation(M),
|
||||
/// The payload did not serialize.
|
||||
Encode(M, PayloadError),
|
||||
/// The pid collapsed to a local one and the local typed send failed.
|
||||
Local(SendError<M>),
|
||||
}
|
||||
|
||||
impl<M> ToRemoteError<M> {
|
||||
/// The undelivered message.
|
||||
pub fn into_inner(self) -> M {
|
||||
match self {
|
||||
ToRemoteError::NotConnected(m)
|
||||
| ToRemoteError::DeadIncarnation(m)
|
||||
| ToRemoteError::Encode(m, _) => m,
|
||||
ToRemoteError::Local(e) => e.into_inner(),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Send `msg` to a pid, wherever it lives. Self-node pids short-circuit to
|
||||
/// the local typed send with `msg` itself (no encode, no frame); others go
|
||||
/// out as a `Send` frame after the incarnation check. `Ok(())` for a remote
|
||||
/// target = handed to the connection's inbox. Must run inside
|
||||
/// [`run`](crate::run).
|
||||
pub fn send_to_remote<A>(target: RemotePid<A>, msg: A::Msg) -> Result<(), ToRemoteError<A::Msg>>
|
||||
where
|
||||
A: Addressable,
|
||||
A::Msg: serde::Serialize,
|
||||
{
|
||||
if let Some(local) = target.local() {
|
||||
return send_to(local, msg).map_err(ToRemoteError::Local);
|
||||
}
|
||||
let route = with_runtime(|inner| {
|
||||
inner
|
||||
.outbound
|
||||
.lock()
|
||||
.by_node
|
||||
.get(&target.node)
|
||||
.map(|r| (r.incarnation, r.frames.clone()))
|
||||
});
|
||||
let (current, tx) = match route {
|
||||
Some(r) => r,
|
||||
None => return Err(ToRemoteError::NotConnected(msg)),
|
||||
};
|
||||
if current != target.incarnation {
|
||||
return Err(ToRemoteError::DeadIncarnation(msg));
|
||||
}
|
||||
let payload = match encode_payload(&msg) {
|
||||
Ok(p) => p,
|
||||
Err(e) => return Err(ToRemoteError::Encode(msg, e)),
|
||||
};
|
||||
let frame = Frame::Send {
|
||||
index: target.index,
|
||||
generation: target.generation,
|
||||
type_hash: type_hash::<A::Msg>(),
|
||||
payload,
|
||||
};
|
||||
tx.send(frame).map_err(|_| ToRemoteError::NotConnected(msg))
|
||||
}
|
||||
|
||||
/// The inbound `Send` seam: deliver `payload` under `type_hash` to the local
|
||||
/// actor `(index, generation)`. Node and incarnation are implicit in the
|
||||
/// connection (bound at handshake) — the frame carries only the slot
|
||||
/// identity. Delivery goes through c8's decoder table, so only types the
|
||||
/// receiver has [`expose_type`](crate::cluster::expose::expose_type)d (or
|
||||
/// exposed by name) can land; anything else is refused, never misrouted.
|
||||
pub fn deliver_to_pid(
|
||||
index: u32,
|
||||
generation: u32,
|
||||
type_hash: u64,
|
||||
payload: &[u8],
|
||||
) -> InboundVerdict {
|
||||
let pid = Pid::new(index, generation);
|
||||
match decode_deliver(type_hash, pid, payload) {
|
||||
Ok(()) => InboundVerdict::Delivered,
|
||||
Err(e) => InboundVerdict::Refused(e),
|
||||
}
|
||||
}
|
||||
|
||||
/// What the inbound seam did with a `SendNamed`. Local observability only;
|
||||
/// nothing goes back on the wire (RFC §3).
|
||||
#[derive(Debug)]
|
||||
pub enum InboundVerdict {
|
||||
/// Decoded and handed to the name's holder.
|
||||
Delivered,
|
||||
/// The name is not in this node's exposed set.
|
||||
NotExposed,
|
||||
/// The frame's hash is not the hash the name was exposed with.
|
||||
HashMismatch { expected: u64, got: u64 },
|
||||
/// Exposed, but no live holder right now (unbound, or its holder died
|
||||
/// and the binding is being pruned).
|
||||
Unresolved,
|
||||
/// Resolved, but the delivery half refused it (decode failure, or the
|
||||
/// holder's channel does not accept the exposed type — a local
|
||||
/// re-registration under a different type; never a misroute).
|
||||
Refused(DeliverError),
|
||||
}
|
||||
|
||||
impl InboundVerdict {
|
||||
/// A short static label for tracing/counting (`smarm-trace` records one
|
||||
/// `ClusterInbound` event per frame with it).
|
||||
pub fn label(&self) -> &'static str {
|
||||
match self {
|
||||
InboundVerdict::Delivered => "delivered",
|
||||
InboundVerdict::NotExposed => "not_exposed",
|
||||
InboundVerdict::HashMismatch { .. } => "hash_mismatch",
|
||||
InboundVerdict::Unresolved => "unresolved",
|
||||
InboundVerdict::Refused(_) => "refused",
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// THE inbound resolution seam: exposed-set check → hash check → registry
|
||||
/// resolution → c8 delivery. See the module docs. Must run inside
|
||||
/// [`run`](crate::run) — the conn actor's context.
|
||||
pub fn deliver_named(name: &str, type_hash: u64, payload: &[u8]) -> InboundVerdict {
|
||||
let Some(expected) = exposed_hash(name) else {
|
||||
return InboundVerdict::NotExposed;
|
||||
};
|
||||
if expected != type_hash {
|
||||
return InboundVerdict::HashMismatch {
|
||||
expected,
|
||||
got: type_hash,
|
||||
};
|
||||
}
|
||||
let Some(pid) = whereis(name) else {
|
||||
return InboundVerdict::Unresolved;
|
||||
};
|
||||
match decode_deliver(type_hash, pid, payload) {
|
||||
Ok(()) => InboundVerdict::Delivered,
|
||||
Err(e) => InboundVerdict::Refused(e),
|
||||
}
|
||||
}
|
||||
|
||||
// ---- monitors (c12) -----------------------------------------------------
|
||||
|
||||
/// Bookkeeping commands from [`monitor_remote`]/[`demonitor_remote`] to the
|
||||
/// connection actor that owns the link to the target's node. The actor
|
||||
/// records the registration and *then* emits the `Monitor` frame itself, so
|
||||
/// a `Down` can never arrive at a table that does not yet know the id. It
|
||||
/// lives in the actor (not on `RuntimeInner`) so the bookkeeping dies with
|
||||
/// the connection — exactly what c13 needs to synthesize `Disconnected`.
|
||||
pub(crate) enum MonCmd {
|
||||
Monitor {
|
||||
id: MonitorId,
|
||||
target: RemotePid<Erased>,
|
||||
tx: Sender<RemoteDown>,
|
||||
},
|
||||
Demonitor {
|
||||
id: MonitorId,
|
||||
},
|
||||
}
|
||||
|
||||
/// A remotely-monitored actor's termination notice — the cluster analog of
|
||||
/// [`Down`](crate::monitor::Down), with the pid in its wire form because it
|
||||
/// may name any node.
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct RemoteDown {
|
||||
/// The pid that was being monitored.
|
||||
pub pid: RemotePid<Erased>,
|
||||
/// How it went down. `Disconnected` means the *link* to its node was
|
||||
/// lost (or absent) — nothing is known about the actor itself.
|
||||
pub reason: RemoteDownReason,
|
||||
}
|
||||
|
||||
enum Watch {
|
||||
/// The target collapsed to this node: an ordinary local monitor,
|
||||
/// translated on read.
|
||||
Local(Monitor),
|
||||
/// The target is elsewhere: the connection actor for its node holds the
|
||||
/// registration and forwards the peer's `Down` frame here. The
|
||||
/// `RemoteState` is the read-side backstop (c13): a channel that closes
|
||||
/// while `Live` — the connection died with our command unread — reads
|
||||
/// as `Disconnected` once; afterwards, and after a cancel, closed is
|
||||
/// just closed.
|
||||
Remote(Receiver<RemoteDown>, Cell<RemoteState>),
|
||||
}
|
||||
|
||||
/// Where a remote-watch stands from the reader's side.
|
||||
#[derive(Clone, Copy, PartialEq, Eq)]
|
||||
enum RemoteState {
|
||||
/// No notice yet, not cancelled: a closed channel means `Disconnected`.
|
||||
Live,
|
||||
/// The one notice has been read (or synthesized): nothing more is due.
|
||||
Done,
|
||||
/// `demonitor_remote` ran: never synthesize.
|
||||
Cancelled,
|
||||
}
|
||||
|
||||
/// A live remote monitor: read its one [`RemoteDown`] with
|
||||
/// [`recv`](RemoteMonitor::recv)/[`try_recv`](RemoteMonitor::try_recv), or
|
||||
/// fold it into a `select` via [`arm`](RemoteMonitor::arm). Distinct from
|
||||
/// [`Monitor`] on purpose: its target is a [`RemotePid`], its notice a
|
||||
/// [`RemoteDown`], and it can report `Disconnected` — none of which a local
|
||||
/// monitor can express. Dropping it discards an unread notice, like the
|
||||
/// local one; after [`demonitor_remote`] the channel is closed and empty, so
|
||||
/// `recv` errs rather than parking — also like the local one.
|
||||
///
|
||||
/// Exactly one notice is guaranteed even if the connection actor dies with
|
||||
/// the registration unread (the c13 drain gap): a channel that closes
|
||||
/// before any notice — and before any cancel — reads as `Disconnected`,
|
||||
/// once. The next read is the ordinary closed-channel `Err`.
|
||||
pub struct RemoteMonitor {
|
||||
/// This registration's process-unique id — minted here, echoed by the
|
||||
/// peer in its `Down` frame.
|
||||
pub id: MonitorId,
|
||||
/// The pid being monitored.
|
||||
pub target: RemotePid<Erased>,
|
||||
watch: Watch,
|
||||
}
|
||||
|
||||
impl RemoteMonitor {
|
||||
/// Block (cooperatively) for the notice.
|
||||
pub fn recv(&self) -> Result<RemoteDown, RecvError> {
|
||||
match &self.watch {
|
||||
Watch::Local(m) => m.rx.recv().map(|d| RemoteDown {
|
||||
pid: self.target.clone(),
|
||||
reason: d.reason.into(),
|
||||
}),
|
||||
Watch::Remote(rx, st) => match rx.recv() {
|
||||
Ok(d) => {
|
||||
st.set(RemoteState::Done);
|
||||
Ok(d)
|
||||
}
|
||||
Err(e) => self.closed(st).ok_or(e),
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
/// The notice if it has arrived; `Ok(None)` if not yet.
|
||||
pub fn try_recv(&self) -> Result<Option<RemoteDown>, RecvError> {
|
||||
match &self.watch {
|
||||
Watch::Local(m) => m.rx.try_recv().map(|o| {
|
||||
o.map(|d| RemoteDown {
|
||||
pid: self.target.clone(),
|
||||
reason: d.reason.into(),
|
||||
})
|
||||
}),
|
||||
Watch::Remote(rx, st) => match rx.try_recv() {
|
||||
Ok(Some(d)) => {
|
||||
st.set(RemoteState::Done);
|
||||
Ok(Some(d))
|
||||
}
|
||||
Ok(None) => Ok(None),
|
||||
Err(e) => self.closed(st).map(Some).ok_or(e),
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
/// The channel closed. While `Live` — no notice yet, no cancel — that
|
||||
/// is the connection having died with our registration unread, so
|
||||
/// synthesize the one `Disconnected` and mark `Done`; otherwise closed
|
||||
/// is just closed.
|
||||
fn closed(&self, st: &Cell<RemoteState>) -> Option<RemoteDown> {
|
||||
if st.get() != RemoteState::Live {
|
||||
return None;
|
||||
}
|
||||
st.set(RemoteState::Done);
|
||||
Some(RemoteDown {
|
||||
pid: self.target.clone(),
|
||||
reason: RemoteDownReason::Disconnected,
|
||||
})
|
||||
}
|
||||
|
||||
/// The selectable arm: readiness means [`try_recv`](Self::try_recv)
|
||||
/// will yield the notice.
|
||||
pub fn arm(&self) -> &dyn Selectable {
|
||||
match &self.watch {
|
||||
Watch::Local(m) => &m.rx,
|
||||
Watch::Remote(rx, _) => rx,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl std::fmt::Debug for RemoteMonitor {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
f.debug_struct("RemoteMonitor")
|
||||
.field("id", &self.id)
|
||||
.field("target", &self.target)
|
||||
.finish_non_exhaustive()
|
||||
}
|
||||
}
|
||||
|
||||
/// Monitor `target`, wherever it lives. Exactly one [`RemoteDown`] arrives:
|
||||
///
|
||||
/// - self-node pid ⇒ an ordinary local monitor underneath (same NoProc rule);
|
||||
/// - no live connection to the pid's node ⇒ `Disconnected`, queued at once
|
||||
/// (the remote analog of NoProc: nothing can be known);
|
||||
/// - the pid's incarnation is not the node's current one ⇒ `NoProc`, queued
|
||||
/// at once — the node is *known* to have restarted, so its actor is a
|
||||
/// corpse, not a partition (RFC v2 §3);
|
||||
/// - otherwise the connection actor registers the id and sends `Monitor`;
|
||||
/// the peer answers with the true terminal reason on exit, or immediately
|
||||
/// with the recorded reason for a corpse (`terminal_reason`, RFC §6) or
|
||||
/// `NoProc` for a pid it never exposed and never shipped.
|
||||
///
|
||||
/// The connection dropping while the monitor is outstanding delivers
|
||||
/// `Disconnected` (c13): the connection actor synthesizes it on teardown,
|
||||
/// and the monitor's own read path backstops the case where the actor died
|
||||
/// with the registration still unread. Must run inside [`run`](crate::run).
|
||||
pub fn monitor_remote<A>(target: RemotePid<A>) -> RemoteMonitor {
|
||||
if let Some(local) = target.local() {
|
||||
let m = monitor(local);
|
||||
return RemoteMonitor {
|
||||
id: m.id,
|
||||
target: target.erase(),
|
||||
watch: Watch::Local(m),
|
||||
};
|
||||
}
|
||||
let target = target.erase();
|
||||
let (id, route) = with_runtime(|inner| {
|
||||
let id = inner.alloc_monitor_id();
|
||||
let route = inner
|
||||
.outbound
|
||||
.lock()
|
||||
.by_node
|
||||
.get(&target.node)
|
||||
.map(|r| (r.incarnation, r.monitors.clone()));
|
||||
(id, route)
|
||||
});
|
||||
let (tx, rx) = channel::<RemoteDown>();
|
||||
let immediate = match route {
|
||||
None => Some(RemoteDownReason::Disconnected),
|
||||
Some((current, _)) if current != target.incarnation => {
|
||||
Some(RemoteDownReason::Local(crate::monitor::DownReason::NoProc))
|
||||
}
|
||||
Some((_, mon_tx)) => {
|
||||
let cmd = MonCmd::Monitor {
|
||||
id,
|
||||
target: target.clone(),
|
||||
tx: tx.clone(),
|
||||
};
|
||||
match mon_tx.send(cmd) {
|
||||
Ok(()) => None,
|
||||
Err(_) => Some(RemoteDownReason::Disconnected), // actor already gone
|
||||
}
|
||||
}
|
||||
};
|
||||
if let Some(reason) = immediate {
|
||||
let _ = tx.send(RemoteDown {
|
||||
pid: target.clone(),
|
||||
reason,
|
||||
});
|
||||
}
|
||||
RemoteMonitor {
|
||||
id,
|
||||
target,
|
||||
watch: Watch::Remote(rx, Cell::new(RemoteState::Live)),
|
||||
}
|
||||
}
|
||||
|
||||
/// Cancel `m`. No future notice will be *sent* for it; a notice already in
|
||||
/// flight from the peer is dropped on arrival, and one already sitting in
|
||||
/// `m` is discarded when `m` is dropped (same contract as
|
||||
/// [`demonitor`]). Unlike the local form this returns nothing: the
|
||||
/// registration is owned by the connection actor, so whether the `Down`
|
||||
/// beat the cancel is not local knowledge. Must run inside
|
||||
/// [`run`](crate::run).
|
||||
pub fn demonitor_remote(m: &RemoteMonitor) {
|
||||
match &m.watch {
|
||||
Watch::Local(local) => {
|
||||
let _ = demonitor(local);
|
||||
}
|
||||
Watch::Remote(_, st) => {
|
||||
// Cancel first: a channel closing after this is closed, not a
|
||||
// Disconnected notice — the caller asked for silence.
|
||||
st.set(RemoteState::Cancelled);
|
||||
let mon_tx = with_runtime(|inner| {
|
||||
inner
|
||||
.outbound
|
||||
.lock()
|
||||
.by_node
|
||||
.get(&m.target.node)
|
||||
.map(|r| r.monitors.clone())
|
||||
});
|
||||
if let Some(mon_tx) = mon_tx {
|
||||
let _ = mon_tx.send(MonCmd::Demonitor { id: m.id });
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,298 +0,0 @@
|
||||
//! RFC 010 c3 — transport abstraction for the **control** connection.
|
||||
//!
|
||||
//! Scope, per RFC 010 v2 §5 and D2:
|
||||
//!
|
||||
//! - A "connection" here is the *control* connection: the one carrying this
|
||||
//! RFC's frame inventory ([`crate::cluster::envelope::Frame`]), whose
|
||||
//! heartbeats feed failure detection. The trait deliberately says nothing
|
||||
//! about how many connections a peer pair may hold — the jarred rkyv bulk
|
||||
//! plane opens **additional per-peer connections** outside this trait, and
|
||||
//! nothing here may foreclose that.
|
||||
//! - Homogeneous smarm⇄smarm only. The BEAM membrane is *not* a transport
|
||||
//! impl and the trait does not accommodate it (D2).
|
||||
//! - Addresses are opaque, **pre-resolved** strings. Name resolution is a
|
||||
//! single separate seam (roadmap c9); impls reject unresolved names rather
|
||||
//! than resolving them.
|
||||
//!
|
||||
//! Blocking model: [`Conn`] calls block the caller. The TCP impl parks the
|
||||
//! calling *actor* (fd readiness via the scheduler); the loopback impl blocks
|
||||
//! the calling *OS thread* and is a test transport — do not drive it from a
|
||||
//! scheduler thread.
|
||||
//!
|
||||
//! Framing is not part of the trait: [`FramedConn`] is the single shared
|
||||
//! codec that turns any byte-stream [`Conn`] into a frame pipe, feeding
|
||||
//! [`Frame::decode`]'s incremental contract. Impls never re-implement
|
||||
//! framing, and the conformance suite exercises the same codec over every
|
||||
//! impl.
|
||||
|
||||
use std::io;
|
||||
|
||||
use crate::cluster::envelope::{DecodeError, EncodeError, Frame};
|
||||
|
||||
pub mod loopback;
|
||||
pub mod tcp;
|
||||
|
||||
/// An established control connection: a bidirectional byte stream.
|
||||
pub trait Conn: Send {
|
||||
/// Read at least one byte, blocking the caller until data is available,
|
||||
/// EOF, or error. `Ok(0)` means EOF: the peer closed and all bytes it
|
||||
/// wrote before closing have been consumed.
|
||||
fn read(&mut self, buf: &mut [u8]) -> io::Result<usize>;
|
||||
|
||||
/// Write the whole buffer, blocking the caller as needed.
|
||||
fn write_all(&mut self, buf: &[u8]) -> io::Result<()>;
|
||||
|
||||
/// Close both directions. Idempotent. Bytes already written remain
|
||||
/// readable at the peer, which then observes EOF; peer writes after this
|
||||
/// fail.
|
||||
fn close(&mut self);
|
||||
|
||||
/// Diagnostic label for logs only. Mesh identity comes from the
|
||||
/// handshake (`Hello`/`HelloAck`), never from the transport.
|
||||
fn peer_addr(&self) -> String;
|
||||
|
||||
/// Readiness as a [`select`](crate::select) arm, for transports backed by
|
||||
/// a file descriptor. `Some` lets a driver wait on "this connection is
|
||||
/// readable" alongside an ordinary command inbox in a single `select`, so
|
||||
/// one actor can interleave reading with control messages without a
|
||||
/// second thread. The default is `None`: a transport with no fd (the
|
||||
/// in-memory loopback) cannot be selected on and must be driven another
|
||||
/// way.
|
||||
fn readable_arm(&self) -> Option<crate::scheduler::FdArm> {
|
||||
None
|
||||
}
|
||||
}
|
||||
|
||||
/// A bound listen point producing inbound [`Conn`]s.
|
||||
pub trait Listener: Send {
|
||||
/// Accept the next inbound connection, blocking the caller.
|
||||
fn accept(&mut self) -> io::Result<Box<dyn Conn>>;
|
||||
|
||||
/// The concrete bound address, dialable as-is (e.g. the real port when
|
||||
/// bound with port 0).
|
||||
fn local_addr(&self) -> String;
|
||||
|
||||
/// Readiness as a [`select`](crate::select) arm, mirroring
|
||||
/// [`Conn::readable_arm`]: `Some` lets an acceptor wait on "an inbound
|
||||
/// connection is pending" alongside a command inbox in one `select`, so
|
||||
/// it can be told to stop without a poll loop. Default `None` (the
|
||||
/// loopback listener has no fd and must be driven synchronously).
|
||||
fn readable_arm(&self) -> Option<crate::scheduler::FdArm> {
|
||||
None
|
||||
}
|
||||
}
|
||||
|
||||
/// A way of establishing control connections. Object-safe on purpose: the
|
||||
/// connector and membership layers hold `&dyn Transport` / boxed conns
|
||||
/// rather than growing a generic parameter.
|
||||
pub trait Transport: Send + Sync {
|
||||
/// Connect to a peer's listen address. Blocks the caller until
|
||||
/// established or failed.
|
||||
fn dial(&self, addr: &str) -> io::Result<Box<dyn Conn>>;
|
||||
|
||||
/// Bind a listen point.
|
||||
fn listen(&self, addr: &str) -> io::Result<Box<dyn Listener>>;
|
||||
}
|
||||
|
||||
impl std::fmt::Debug for dyn Conn {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
write!(f, "Conn({})", self.peer_addr())
|
||||
}
|
||||
}
|
||||
|
||||
impl std::fmt::Debug for dyn Listener {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
write!(f, "Listener({})", self.local_addr())
|
||||
}
|
||||
}
|
||||
|
||||
/// Error surface of [`FramedConn::send`].
|
||||
#[derive(Debug)]
|
||||
pub enum SendError {
|
||||
/// The frame could not be encoded (e.g. a field over its wire limit).
|
||||
Encode(EncodeError),
|
||||
/// The transport failed mid-write.
|
||||
Io(io::Error),
|
||||
}
|
||||
|
||||
impl std::fmt::Display for SendError {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
match self {
|
||||
SendError::Encode(e) => write!(f, "frame encode failed: {e:?}"),
|
||||
SendError::Io(e) => write!(f, "transport write failed: {e}"),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl std::error::Error for SendError {}
|
||||
|
||||
/// Error surface of [`FramedConn::recv`].
|
||||
#[derive(Debug)]
|
||||
pub enum RecvError {
|
||||
/// The byte stream is not a valid frame stream (bad tag, lying length,
|
||||
/// oversized frame, …). The connection is unusable.
|
||||
Corrupt(DecodeError),
|
||||
/// The peer closed mid-frame: EOF arrived with a partial frame buffered.
|
||||
/// Distinct from a clean close, which is `Ok(None)`.
|
||||
TruncatedByPeer,
|
||||
/// The transport failed mid-read.
|
||||
Io(io::Error),
|
||||
/// The deadline passed before a full frame arrived
|
||||
/// ([`FramedConn::recv_deadline`] only; plain [`recv`](FramedConn::recv)
|
||||
/// never returns this).
|
||||
TimedOut,
|
||||
}
|
||||
|
||||
impl std::fmt::Display for RecvError {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
match self {
|
||||
RecvError::Corrupt(e) => write!(f, "frame stream corrupt: {e:?}"),
|
||||
RecvError::TruncatedByPeer => write!(f, "peer closed mid-frame"),
|
||||
RecvError::Io(e) => write!(f, "transport read failed: {e}"),
|
||||
RecvError::TimedOut => write!(f, "deadline passed mid-receive"),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl std::error::Error for RecvError {}
|
||||
|
||||
/// How many bytes each blocking read asks the transport for.
|
||||
const READ_CHUNK: usize = 8 * 1024;
|
||||
|
||||
/// The shared framed codec: one of these per control connection, owning the
|
||||
/// [`Conn`] and the reassembly buffer. Frames may arrive split or coalesced
|
||||
/// arbitrarily; [`recv`](FramedConn::recv) reassembles either way.
|
||||
pub struct FramedConn {
|
||||
conn: Box<dyn Conn>,
|
||||
rbuf: Vec<u8>,
|
||||
}
|
||||
|
||||
impl FramedConn {
|
||||
pub fn new(conn: Box<dyn Conn>) -> Self {
|
||||
FramedConn {
|
||||
conn,
|
||||
rbuf: Vec::new(),
|
||||
}
|
||||
}
|
||||
|
||||
/// Encode and write one frame.
|
||||
pub fn send(&mut self, frame: &Frame) -> Result<(), SendError> {
|
||||
let mut out = Vec::new();
|
||||
frame.encode(&mut out).map_err(SendError::Encode)?;
|
||||
self.conn.write_all(&out).map_err(SendError::Io)
|
||||
}
|
||||
|
||||
/// Receive the next frame. `Ok(None)` is a clean close: EOF at a frame
|
||||
/// boundary. EOF mid-frame is [`RecvError::TruncatedByPeer`].
|
||||
pub fn recv(&mut self) -> Result<Option<Frame>, RecvError> {
|
||||
loop {
|
||||
match Frame::decode(&self.rbuf) {
|
||||
Ok(Some((frame, consumed))) => {
|
||||
self.rbuf.drain(..consumed);
|
||||
return Ok(Some(frame));
|
||||
}
|
||||
Ok(None) => {}
|
||||
Err(e) => return Err(RecvError::Corrupt(e)),
|
||||
}
|
||||
let mut chunk = [0u8; READ_CHUNK];
|
||||
let n = self.conn.read(&mut chunk).map_err(RecvError::Io)?;
|
||||
if n == 0 {
|
||||
return if self.rbuf.is_empty() {
|
||||
Ok(None)
|
||||
} else {
|
||||
Err(RecvError::TruncatedByPeer)
|
||||
};
|
||||
}
|
||||
self.rbuf.extend_from_slice(&chunk[..n]);
|
||||
}
|
||||
}
|
||||
|
||||
/// Like [`recv`](FramedConn::recv), but gives up with
|
||||
/// [`RecvError::TimedOut`] once `deadline` passes without a full frame.
|
||||
/// The deadline is enforced between reads via the connection's fd arm
|
||||
/// (so the caller must be an actor); a transport with no fd (loopback)
|
||||
/// cannot be timed out and this degrades to a plain blocking `recv` —
|
||||
/// the same caveat as liveness.
|
||||
pub fn recv_deadline(
|
||||
&mut self,
|
||||
deadline: std::time::Instant,
|
||||
) -> Result<Option<Frame>, RecvError> {
|
||||
loop {
|
||||
match Frame::decode(&self.rbuf) {
|
||||
Ok(Some((frame, consumed))) => {
|
||||
self.rbuf.drain(..consumed);
|
||||
return Ok(Some(frame));
|
||||
}
|
||||
Ok(None) => {}
|
||||
Err(e) => return Err(RecvError::Corrupt(e)),
|
||||
}
|
||||
if let Some(arm) = self.conn.readable_arm() {
|
||||
let left = deadline.saturating_duration_since(std::time::Instant::now());
|
||||
if left.is_zero() {
|
||||
return Err(RecvError::TimedOut);
|
||||
}
|
||||
match crate::channel::try_select_timeout(&[&arm], left) {
|
||||
Ok(Some(_)) => {}
|
||||
Ok(None) => return Err(RecvError::TimedOut),
|
||||
Err(e) => return Err(RecvError::Io(e)),
|
||||
}
|
||||
}
|
||||
let mut chunk = [0u8; READ_CHUNK];
|
||||
let n = self.conn.read(&mut chunk).map_err(RecvError::Io)?;
|
||||
if n == 0 {
|
||||
return if self.rbuf.is_empty() {
|
||||
Ok(None)
|
||||
} else {
|
||||
Err(RecvError::TruncatedByPeer)
|
||||
};
|
||||
}
|
||||
self.rbuf.extend_from_slice(&chunk[..n]);
|
||||
}
|
||||
}
|
||||
|
||||
/// One socket read, appended to the reassembly buffer. Returns the byte
|
||||
/// count (`0` = EOF). For select-loop callers that were just told the fd
|
||||
/// is readable: under the level-triggered IO thread exactly one read per
|
||||
/// readable wake never blocks and never loses data — leftover socket
|
||||
/// bytes re-signal on the next select, and complete frames already
|
||||
/// reassembled are drained with [`next_buffered`](FramedConn::next_buffered).
|
||||
/// (A plain [`recv`](FramedConn::recv) can block into the socket while
|
||||
/// the buffer holds a partial frame, which a loop with deadlines to keep
|
||||
/// cannot afford.)
|
||||
pub fn read_once(&mut self) -> std::io::Result<usize> {
|
||||
let mut chunk = [0u8; READ_CHUNK];
|
||||
let n = self.conn.read(&mut chunk)?;
|
||||
self.rbuf.extend_from_slice(&chunk[..n]);
|
||||
Ok(n)
|
||||
}
|
||||
|
||||
/// Decode the next complete frame already sitting in the reassembly
|
||||
/// buffer, without touching the socket. `Ok(None)` means the buffer
|
||||
/// holds no complete frame (empty, or a partial awaiting more bytes).
|
||||
pub fn next_buffered(&mut self) -> Result<Option<Frame>, DecodeError> {
|
||||
match Frame::decode(&self.rbuf) {
|
||||
Ok(Some((frame, consumed))) => {
|
||||
self.rbuf.drain(..consumed);
|
||||
Ok(Some(frame))
|
||||
}
|
||||
Ok(None) => Ok(None),
|
||||
Err(e) => Err(e),
|
||||
}
|
||||
}
|
||||
|
||||
/// Close the underlying connection (idempotent, see [`Conn::close`]).
|
||||
pub fn close(&mut self) {
|
||||
self.conn.close();
|
||||
}
|
||||
|
||||
/// Diagnostic label of the underlying connection.
|
||||
pub fn peer_addr(&self) -> String {
|
||||
self.conn.peer_addr()
|
||||
}
|
||||
|
||||
/// The underlying connection's readiness arm, if it is fd-backed (see
|
||||
/// [`Conn::readable_arm`]).
|
||||
pub fn readable_arm(&self) -> Option<crate::scheduler::FdArm> {
|
||||
self.conn.readable_arm()
|
||||
}
|
||||
}
|
||||
@@ -1,274 +0,0 @@
|
||||
//! In-memory loopback transport — a shipped **test** transport.
|
||||
//!
|
||||
//! Lets Phases 2–4 exercise protocol logic (connector, membership,
|
||||
//! monitors) through the real transport trait and the real framed codec
|
||||
//! without sockets or timing flake.
|
||||
//!
|
||||
//! Blocking model: calls block the **OS thread** on a condvar. That is the
|
||||
//! right shape for plain `#[test]`s driving protocol state machines; it is
|
||||
//! the wrong shape for scheduler threads. Do not drive a loopback conn from
|
||||
//! inside an actor — use the TCP impl there.
|
||||
//!
|
||||
//! Semantics mirror TCP shutdown where it matters for the codec: bytes
|
||||
//! written before `close` remain readable at the peer, which then sees EOF;
|
||||
//! writes toward a closed peer fail with `BrokenPipe`. Write buffers are
|
||||
//! unbounded, so writes never block — backpressure is not simulated.
|
||||
|
||||
use std::collections::{HashMap, VecDeque};
|
||||
use std::io;
|
||||
use std::sync::{Arc, Condvar, Mutex, MutexGuard};
|
||||
|
||||
use super::{Conn, Listener, Transport};
|
||||
|
||||
/// Poison-tolerant lock: a panicked holder in a *test* transport must not
|
||||
/// cascade; the byte-queue state stays consistent under every early return.
|
||||
fn lock<T>(m: &Mutex<T>) -> MutexGuard<'_, T> {
|
||||
match m.lock() {
|
||||
Ok(g) => g,
|
||||
Err(poisoned) => poisoned.into_inner(),
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// One direction of a duplex: a byte queue with close flags for both ends
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
#[derive(Default)]
|
||||
struct PipeState {
|
||||
bytes: VecDeque<u8>,
|
||||
/// The writing end closed: readers drain remaining bytes, then EOF.
|
||||
write_closed: bool,
|
||||
/// The reading end closed: writers fail with `BrokenPipe`.
|
||||
read_closed: bool,
|
||||
}
|
||||
|
||||
#[derive(Default)]
|
||||
struct Pipe {
|
||||
state: Mutex<PipeState>,
|
||||
cv: Condvar,
|
||||
}
|
||||
|
||||
impl Pipe {
|
||||
fn write_all(&self, buf: &[u8]) -> io::Result<()> {
|
||||
let mut st = lock(&self.state);
|
||||
if st.write_closed {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::NotConnected,
|
||||
"loopback conn closed locally",
|
||||
));
|
||||
}
|
||||
if st.read_closed {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::BrokenPipe,
|
||||
"loopback peer closed",
|
||||
));
|
||||
}
|
||||
st.bytes.extend(buf);
|
||||
self.cv.notify_all();
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn read(&self, buf: &mut [u8]) -> io::Result<usize> {
|
||||
if buf.is_empty() {
|
||||
return Ok(0);
|
||||
}
|
||||
let mut st = lock(&self.state);
|
||||
loop {
|
||||
if !st.bytes.is_empty() {
|
||||
let n = st.bytes.len().min(buf.len());
|
||||
for (slot, byte) in buf.iter_mut().zip(st.bytes.drain(..n)) {
|
||||
*slot = byte;
|
||||
}
|
||||
return Ok(n);
|
||||
}
|
||||
if st.write_closed || st.read_closed {
|
||||
return Ok(0); // EOF: peer closed, or our own end closed.
|
||||
}
|
||||
st = match self.cv.wait(st) {
|
||||
Ok(g) => g,
|
||||
Err(poisoned) => poisoned.into_inner(),
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
/// Close from the writer side: remaining bytes stay readable, then EOF.
|
||||
fn close_write(&self) {
|
||||
lock(&self.state).write_closed = true;
|
||||
self.cv.notify_all();
|
||||
}
|
||||
|
||||
/// Close from the reader side: peer writes fail from now on.
|
||||
fn close_read(&self) {
|
||||
lock(&self.state).read_closed = true;
|
||||
self.cv.notify_all();
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Conn: two pipes, one per direction
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// One end of an established loopback connection.
|
||||
pub struct LoopbackConn {
|
||||
tx: Arc<Pipe>,
|
||||
rx: Arc<Pipe>,
|
||||
peer: String,
|
||||
}
|
||||
|
||||
impl Conn for LoopbackConn {
|
||||
fn read(&mut self, buf: &mut [u8]) -> io::Result<usize> {
|
||||
self.rx.read(buf)
|
||||
}
|
||||
|
||||
fn write_all(&mut self, buf: &[u8]) -> io::Result<()> {
|
||||
self.tx.write_all(buf)
|
||||
}
|
||||
|
||||
fn close(&mut self) {
|
||||
self.tx.close_write();
|
||||
self.rx.close_read();
|
||||
}
|
||||
|
||||
fn peer_addr(&self) -> String {
|
||||
self.peer.clone()
|
||||
}
|
||||
}
|
||||
|
||||
impl Drop for LoopbackConn {
|
||||
fn drop(&mut self) {
|
||||
self.close();
|
||||
}
|
||||
}
|
||||
|
||||
fn conn_pair(listen_addr: &str, conn_no: u64) -> (LoopbackConn, LoopbackConn) {
|
||||
let a_to_b = Arc::new(Pipe::default());
|
||||
let b_to_a = Arc::new(Pipe::default());
|
||||
let dialer = LoopbackConn {
|
||||
tx: a_to_b.clone(),
|
||||
rx: b_to_a.clone(),
|
||||
peer: listen_addr.to_string(),
|
||||
};
|
||||
let accepted = LoopbackConn {
|
||||
tx: b_to_a,
|
||||
rx: a_to_b,
|
||||
peer: format!("{listen_addr}#dialer-{conn_no}"),
|
||||
};
|
||||
(dialer, accepted)
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Listener + registry
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
#[derive(Default)]
|
||||
struct AcceptState {
|
||||
pending: VecDeque<LoopbackConn>,
|
||||
closed: bool,
|
||||
}
|
||||
|
||||
#[derive(Default)]
|
||||
struct AcceptQueue {
|
||||
state: Mutex<AcceptState>,
|
||||
cv: Condvar,
|
||||
}
|
||||
|
||||
/// A bound loopback listen point.
|
||||
pub struct LoopbackListener {
|
||||
addr: String,
|
||||
queue: Arc<AcceptQueue>,
|
||||
registry: Arc<Mutex<Registry>>,
|
||||
}
|
||||
|
||||
impl Listener for LoopbackListener {
|
||||
fn accept(&mut self) -> io::Result<Box<dyn Conn>> {
|
||||
let mut st = lock(&self.queue.state);
|
||||
loop {
|
||||
if let Some(conn) = st.pending.pop_front() {
|
||||
return Ok(Box::new(conn));
|
||||
}
|
||||
if st.closed {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::NotConnected,
|
||||
"loopback listener closed",
|
||||
));
|
||||
}
|
||||
st = match self.queue.cv.wait(st) {
|
||||
Ok(g) => g,
|
||||
Err(poisoned) => poisoned.into_inner(),
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
fn local_addr(&self) -> String {
|
||||
self.addr.clone()
|
||||
}
|
||||
}
|
||||
|
||||
impl Drop for LoopbackListener {
|
||||
fn drop(&mut self) {
|
||||
lock(&self.registry).listeners.remove(&self.addr);
|
||||
let mut st = lock(&self.queue.state);
|
||||
st.closed = true;
|
||||
self.queue.cv.notify_all();
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Default)]
|
||||
struct Registry {
|
||||
listeners: HashMap<String, Arc<AcceptQueue>>,
|
||||
dial_count: u64,
|
||||
}
|
||||
|
||||
/// The loopback transport. Addresses are arbitrary strings scoped to one
|
||||
/// transport instance; distinct instances never see each other's listeners.
|
||||
#[derive(Default)]
|
||||
pub struct LoopbackTransport {
|
||||
registry: Arc<Mutex<Registry>>,
|
||||
}
|
||||
|
||||
impl Transport for LoopbackTransport {
|
||||
fn dial(&self, addr: &str) -> io::Result<Box<dyn Conn>> {
|
||||
let (queue, conn_no) = {
|
||||
let mut reg = lock(&self.registry);
|
||||
reg.dial_count += 1;
|
||||
let no = reg.dial_count;
|
||||
match reg.listeners.get(addr) {
|
||||
Some(q) => (q.clone(), no),
|
||||
None => {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::ConnectionRefused,
|
||||
format!("no loopback listener at {addr:?}"),
|
||||
));
|
||||
}
|
||||
}
|
||||
};
|
||||
let (dialer, accepted) = conn_pair(addr, conn_no);
|
||||
let mut st = lock(&queue.state);
|
||||
if st.closed {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::ConnectionRefused,
|
||||
format!("loopback listener at {addr:?} closed"),
|
||||
));
|
||||
}
|
||||
st.pending.push_back(accepted);
|
||||
queue.cv.notify_all();
|
||||
Ok(Box::new(dialer))
|
||||
}
|
||||
|
||||
fn listen(&self, addr: &str) -> io::Result<Box<dyn Listener>> {
|
||||
let queue = Arc::new(AcceptQueue::default());
|
||||
let mut reg = lock(&self.registry);
|
||||
if reg.listeners.contains_key(addr) {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::AddrInUse,
|
||||
format!("loopback listener already bound at {addr:?}"),
|
||||
));
|
||||
}
|
||||
reg.listeners.insert(addr.to_string(), queue.clone());
|
||||
Ok(Box::new(LoopbackListener {
|
||||
addr: addr.to_string(),
|
||||
queue,
|
||||
registry: self.registry.clone(),
|
||||
}))
|
||||
}
|
||||
}
|
||||
@@ -1,285 +0,0 @@
|
||||
//! TCP transport — the production control-plane transport.
|
||||
//!
|
||||
//! Blocking model: every blocking point parks the **calling actor** on fd
|
||||
//! readiness ([`crate::scheduler::wait_readable`] / `wait_writable`); the
|
||||
//! scheduler thread is never blocked. All conn/listener methods must
|
||||
//! therefore run inside an actor. `listen` itself only binds (no waiting)
|
||||
//! and is callable anywhere.
|
||||
//!
|
||||
//! Addresses are pre-resolved `ip:port` strings (`SocketAddr` syntax, IPv4
|
||||
//! or IPv6). Hostnames are rejected with `InvalidInput`: name resolution is
|
||||
//! the single c9 seam, not something each transport does on the side.
|
||||
//!
|
||||
//! Writes use `send(2)` with `MSG_NOSIGNAL` — a peer reset must surface as
|
||||
//! `BrokenPipe`/`ConnectionReset`, not `SIGPIPE`.
|
||||
|
||||
use std::io;
|
||||
use std::net::{SocketAddr, TcpListener as StdListener, TcpStream};
|
||||
use std::os::fd::{AsRawFd, RawFd};
|
||||
|
||||
use crate::scheduler::{wait_readable, wait_writable};
|
||||
|
||||
use super::{Conn, Listener, Transport};
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// sockaddr plumbing
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// A `sockaddr_in`/`sockaddr_in6` built from a parsed `SocketAddr`, plus its
|
||||
/// length, ready for `connect(2)`.
|
||||
union SockAddrUnion {
|
||||
v4: libc::sockaddr_in,
|
||||
v6: libc::sockaddr_in6,
|
||||
}
|
||||
|
||||
fn to_sockaddr(sa: &SocketAddr) -> (SockAddrUnion, libc::socklen_t) {
|
||||
match sa {
|
||||
SocketAddr::V4(v4) => {
|
||||
let raw = libc::sockaddr_in {
|
||||
sin_family: libc::AF_INET as libc::sa_family_t,
|
||||
sin_port: v4.port().to_be(),
|
||||
sin_addr: libc::in_addr {
|
||||
s_addr: u32::from_be_bytes(v4.ip().octets()).to_be(),
|
||||
},
|
||||
sin_zero: [0; 8],
|
||||
};
|
||||
(
|
||||
SockAddrUnion { v4: raw },
|
||||
std::mem::size_of::<libc::sockaddr_in>() as libc::socklen_t,
|
||||
)
|
||||
}
|
||||
SocketAddr::V6(v6) => {
|
||||
let raw = libc::sockaddr_in6 {
|
||||
sin6_family: libc::AF_INET6 as libc::sa_family_t,
|
||||
sin6_port: v6.port().to_be(),
|
||||
sin6_flowinfo: v6.flowinfo(),
|
||||
sin6_addr: libc::in6_addr {
|
||||
s6_addr: v6.ip().octets(),
|
||||
},
|
||||
sin6_scope_id: v6.scope_id(),
|
||||
};
|
||||
(
|
||||
SockAddrUnion { v6: raw },
|
||||
std::mem::size_of::<libc::sockaddr_in6>() as libc::socklen_t,
|
||||
)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn parse_addr(addr: &str) -> io::Result<SocketAddr> {
|
||||
addr.parse().map_err(|_| {
|
||||
io::Error::new(
|
||||
io::ErrorKind::InvalidInput,
|
||||
format!("{addr:?} is not a resolved ip:port — resolution is the c9 seam"),
|
||||
)
|
||||
})
|
||||
}
|
||||
|
||||
fn so_error(fd: RawFd) -> io::Result<()> {
|
||||
let mut err: libc::c_int = 0;
|
||||
let mut len = std::mem::size_of::<libc::c_int>() as libc::socklen_t;
|
||||
let rc = unsafe {
|
||||
libc::getsockopt(
|
||||
fd,
|
||||
libc::SOL_SOCKET,
|
||||
libc::SO_ERROR,
|
||||
(&mut err) as *mut _ as *mut libc::c_void,
|
||||
&mut len,
|
||||
)
|
||||
};
|
||||
if rc != 0 {
|
||||
return Err(io::Error::last_os_error());
|
||||
}
|
||||
if err != 0 {
|
||||
return Err(io::Error::from_raw_os_error(err));
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Conn
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// One established TCP control connection. Owns the socket; drop closes it.
|
||||
pub struct TcpConn {
|
||||
stream: TcpStream,
|
||||
closed: bool,
|
||||
}
|
||||
|
||||
impl TcpConn {
|
||||
fn fd(&self) -> RawFd {
|
||||
self.stream.as_raw_fd()
|
||||
}
|
||||
}
|
||||
|
||||
impl Conn for TcpConn {
|
||||
fn read(&mut self, buf: &mut [u8]) -> io::Result<usize> {
|
||||
if self.closed {
|
||||
return Ok(0);
|
||||
}
|
||||
if buf.is_empty() {
|
||||
return Ok(0);
|
||||
}
|
||||
loop {
|
||||
wait_readable(self.fd())?;
|
||||
let n = unsafe { libc::read(self.fd(), buf.as_mut_ptr() as *mut _, buf.len()) };
|
||||
if n >= 0 {
|
||||
return Ok(n as usize);
|
||||
}
|
||||
let e = io::Error::last_os_error();
|
||||
match e.kind() {
|
||||
// Spurious readiness or signal: park again.
|
||||
io::ErrorKind::WouldBlock | io::ErrorKind::Interrupted => continue,
|
||||
_ => return Err(e),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn write_all(&mut self, mut buf: &[u8]) -> io::Result<()> {
|
||||
if self.closed {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::NotConnected,
|
||||
"tcp conn closed locally",
|
||||
));
|
||||
}
|
||||
while !buf.is_empty() {
|
||||
wait_writable(self.fd())?;
|
||||
let n = unsafe {
|
||||
libc::send(
|
||||
self.fd(),
|
||||
buf.as_ptr() as *const _,
|
||||
buf.len(),
|
||||
libc::MSG_NOSIGNAL,
|
||||
)
|
||||
};
|
||||
if n >= 0 {
|
||||
buf = &buf[n as usize..];
|
||||
continue;
|
||||
}
|
||||
let e = io::Error::last_os_error();
|
||||
match e.kind() {
|
||||
io::ErrorKind::WouldBlock | io::ErrorKind::Interrupted => continue,
|
||||
_ => return Err(e),
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn close(&mut self) {
|
||||
if !self.closed {
|
||||
self.closed = true;
|
||||
// Best-effort: the peer sees EOF after draining. The fd itself
|
||||
// is released when the owning stream drops.
|
||||
let _ = self.stream.shutdown(std::net::Shutdown::Both);
|
||||
}
|
||||
}
|
||||
|
||||
fn peer_addr(&self) -> String {
|
||||
match self.stream.peer_addr() {
|
||||
Ok(sa) => sa.to_string(),
|
||||
Err(_) => "<disconnected>".to_string(),
|
||||
}
|
||||
}
|
||||
|
||||
fn readable_arm(&self) -> Option<crate::scheduler::FdArm> {
|
||||
Some(crate::scheduler::FdArm::readable(self.fd()))
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Listener
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// A bound TCP listen point (non-blocking socket; accept parks the actor).
|
||||
pub struct TcpListener {
|
||||
inner: StdListener,
|
||||
local: SocketAddr,
|
||||
}
|
||||
|
||||
impl Listener for TcpListener {
|
||||
fn readable_arm(&self) -> Option<crate::scheduler::FdArm> {
|
||||
Some(crate::scheduler::FdArm::readable(self.inner.as_raw_fd()))
|
||||
}
|
||||
|
||||
fn accept(&mut self) -> io::Result<Box<dyn Conn>> {
|
||||
loop {
|
||||
wait_readable(self.inner.as_raw_fd())?;
|
||||
match self.inner.accept() {
|
||||
Ok((stream, _peer)) => {
|
||||
stream.set_nonblocking(true)?;
|
||||
return Ok(Box::new(TcpConn {
|
||||
stream,
|
||||
closed: false,
|
||||
}));
|
||||
}
|
||||
Err(e)
|
||||
if e.kind() == io::ErrorKind::WouldBlock
|
||||
|| e.kind() == io::ErrorKind::Interrupted =>
|
||||
{
|
||||
continue;
|
||||
}
|
||||
Err(e) => return Err(e),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn local_addr(&self) -> String {
|
||||
self.local.to_string()
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Transport
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// The TCP transport. Stateless; every call stands alone.
|
||||
pub struct TcpTransport;
|
||||
|
||||
impl Transport for TcpTransport {
|
||||
fn dial(&self, addr: &str) -> io::Result<Box<dyn Conn>> {
|
||||
let sa = parse_addr(addr)?;
|
||||
let family = match sa {
|
||||
SocketAddr::V4(_) => libc::AF_INET,
|
||||
SocketAddr::V6(_) => libc::AF_INET6,
|
||||
};
|
||||
let fd = unsafe {
|
||||
libc::socket(
|
||||
family,
|
||||
libc::SOCK_STREAM | libc::SOCK_NONBLOCK | libc::SOCK_CLOEXEC,
|
||||
0,
|
||||
)
|
||||
};
|
||||
if fd < 0 {
|
||||
return Err(io::Error::last_os_error());
|
||||
}
|
||||
// From here the fd is owned by `stream`; any early return drops it.
|
||||
let stream = unsafe {
|
||||
use std::os::fd::FromRawFd;
|
||||
TcpStream::from_raw_fd(fd)
|
||||
};
|
||||
let (raw, len) = to_sockaddr(&sa);
|
||||
let rc = unsafe { libc::connect(fd, (&raw) as *const _ as *const libc::sockaddr, len) };
|
||||
if rc != 0 {
|
||||
let e = io::Error::last_os_error();
|
||||
if e.raw_os_error() != Some(libc::EINPROGRESS) {
|
||||
return Err(e);
|
||||
}
|
||||
// Connect in flight: park until the socket is writable, then the
|
||||
// verdict is in SO_ERROR.
|
||||
wait_writable(fd)?;
|
||||
so_error(fd)?;
|
||||
}
|
||||
Ok(Box::new(TcpConn {
|
||||
stream,
|
||||
closed: false,
|
||||
}))
|
||||
}
|
||||
|
||||
fn listen(&self, addr: &str) -> io::Result<Box<dyn Listener>> {
|
||||
let sa = parse_addr(addr)?;
|
||||
let inner = StdListener::bind(sa)?;
|
||||
inner.set_nonblocking(true)?;
|
||||
let local = inner.local_addr()?;
|
||||
Ok(Box::new(TcpListener { inner, local }))
|
||||
}
|
||||
}
|
||||
+11
-192
@@ -1,195 +1,14 @@
|
||||
//! Cooperative context switching, x86-64.
|
||||
//! Cooperative context switching — public façade over the arch backend.
|
||||
//!
|
||||
//! Two naked-asm functions move execution between a scheduler thread and an
|
||||
//! actor running on its own mmap'd stack. The compiler cannot do this; the
|
||||
//! whole point of `#[unsafe(naked)]` is that we control every instruction.
|
||||
//! The real implementation lives in `crate::arch`, selected per `target_arch`.
|
||||
//! This module re-exports the stable surface (`init_actor_stack`, the two
|
||||
//! `switch_to_*` shims, and the `*_actor_sp` accessors) so the scheduler,
|
||||
//! `preempt`, and the `tests/context.rs` integration tests keep importing
|
||||
//! `smarm::context::{...}` unchanged regardless of which ISA is built.
|
||||
//!
|
||||
//! The actor's stack pointer travels in registers: `switch_to_actor` takes
|
||||
//! the target sp as its argument and returns the sp the actor saved when it
|
||||
//! next yielded (handed back in `rax` by `switch_to_scheduler`'s shim). Only
|
||||
//! the *scheduler* sp lives in a thread-local — the yielding actor sits at
|
||||
//! arbitrary call depth with no argument channel back to the scheduler, so
|
||||
//! TLS is the one place it can find the way home. `init_actor_stack` builds
|
||||
//! the initial stack so that the first `switch_to_actor` lands inside the
|
||||
//! entry function with `rsp % 16 == 8` (the x86-64 ABI requirement at
|
||||
//! function entry).
|
||||
//!
|
||||
//! # Thread-locals and migration (read before touching any `thread_local!`)
|
||||
//!
|
||||
//! An actor may park on scheduler thread A and be resumed on thread B. LLVM
|
||||
//! treats the address of a thread-local as a loop-invariant, side-effect-free
|
||||
//! value: it computes `%fs:0 + offset` once per function and happily keeps it
|
||||
//! in a callee-saved register across calls — including across
|
||||
//! `switch_to_scheduler`. Any function that touches a scheduler thread-local
|
||||
//! both before and after a switch (or that gets *inlined* into one that does)
|
||||
//! therefore reads and writes the *old thread's* TLS after migration. Nothing
|
||||
//! at the switch can prevent this: it is not a memory clobber problem, the
|
||||
//! address is not memory-derived in LLVM's model. Under thin LTO the code
|
||||
//! happened to use the local-exec model (`%fs:imm` operands, nothing to
|
||||
//! cache) so it worked by luck; a plain `cargo build --release` of a
|
||||
//! downstream crate broke multi-thread runs (`ACTOR_DONE` written to the wrong
|
||||
//! thread → "scheduler resumed a done actor").
|
||||
//!
|
||||
//! Rule: every function that touches a thread-local and can execute on an
|
||||
//! actor stack must be `#[inline(never)]` and call [`tls_fence`] first, so the
|
||||
//! TLS base is recomputed inside a callee that cannot be inlined into a frame
|
||||
//! spanning a switch, and so LLVM cannot infer the accessor is pure and merge
|
||||
//! two calls to it. Scheduler-side code (`schedule_loop` and what it calls
|
||||
//! before/after `switch_to_actor`) never migrates and is exempt. Do not return
|
||||
//! `&Cell`/pointers into TLS from these accessors; return values.
|
||||
//! `cargo test --profile reltest` (no LTO) is the regression oracle.
|
||||
//! See `crate::arch` for the contract every backend implements, and
|
||||
//! `crate::arch::{x86_64, aarch64}` for the per-ISA register choreography.
|
||||
|
||||
use std::cell::Cell;
|
||||
|
||||
/// Compiler barrier for TLS accessors, see the module docs. Emits no code; a
|
||||
/// side-effecting empty asm keeps LLVM from marking the enclosing
|
||||
/// `#[inline(never)]` accessor `memory(none)` and merging calls to it.
|
||||
#[inline(always)]
|
||||
pub(crate) fn tls_fence() {
|
||||
// SAFETY: empty asm, no operands, no stack, no flags.
|
||||
unsafe { core::arch::asm!("", options(nostack, preserves_flags)) }
|
||||
}
|
||||
|
||||
thread_local! {
|
||||
static SCHEDULER_SP: Cell<usize> = const { Cell::new(0) };
|
||||
}
|
||||
|
||||
#[inline(never)]
|
||||
fn get_scheduler_sp() -> usize {
|
||||
tls_fence();
|
||||
SCHEDULER_SP.with(|c| c.get())
|
||||
}
|
||||
#[inline(never)]
|
||||
fn set_scheduler_sp(v: usize) {
|
||||
tls_fence();
|
||||
SCHEDULER_SP.with(|c| c.set(v))
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Initial stack layout
|
||||
//
|
||||
// We start from aligned_top = top & ~15, then bias by 8 and push seven 8-byte
|
||||
// slots (downward). The first `switch_to_actor` pops r15..rbx and `ret`s —
|
||||
// landing in `entry` with rsp % 16 == 8 (the x86-64 ABI state at the point
|
||||
// just after a `call`, which is what `entry` is compiled to expect).
|
||||
//
|
||||
// Layout (high → low), relative to aligned_top = top & ~15. Note the entry
|
||||
// slot is at aligned_top - 16, NOT aligned_top - 8: the function does
|
||||
// `(top & ~15) - 8` and *then* a `-= 8` before the first write, so the first
|
||||
// stored word lands at aligned_top - 16. Verified by single-stepping the
|
||||
// `ret` under llmdbg: entry sits at an address with %16 == 0, so the post-ret
|
||||
// rsp is %16 == 8.
|
||||
//
|
||||
// aligned_top - 8 : (unused padding; keeps entry's slot %16 == 0)
|
||||
// aligned_top - 16 : entry ptr ← `ret` target. Post-ret: rsp % 16 == 8.
|
||||
// aligned_top - 24 : rbx = 0
|
||||
// aligned_top - 32 : rbp = 0
|
||||
// aligned_top - 40 : r12 = 0
|
||||
// aligned_top - 48 : r13 = 0
|
||||
// aligned_top - 56 : r14 = 0
|
||||
// aligned_top - 64 : r15 = 0 ← initial rsp
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
pub fn init_actor_stack(top: *mut u8, entry: extern "C-unwind" fn()) -> usize {
|
||||
unsafe {
|
||||
let mut sp = (top as usize & !15) - 8;
|
||||
sp -= 8;
|
||||
(sp as *mut usize).write(entry as usize); // ret target
|
||||
sp -= 8;
|
||||
(sp as *mut usize).write(0); // rbx
|
||||
sp -= 8;
|
||||
(sp as *mut usize).write(0); // rbp
|
||||
sp -= 8;
|
||||
(sp as *mut usize).write(0); // r12
|
||||
sp -= 8;
|
||||
(sp as *mut usize).write(0); // r13
|
||||
sp -= 8;
|
||||
(sp as *mut usize).write(0); // r14
|
||||
sp -= 8;
|
||||
(sp as *mut usize).write(0); // r15
|
||||
sp
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Context switch shims
|
||||
//
|
||||
// switch_to_actor_asm (rdi = target actor sp, returns rax = the sp the actor
|
||||
// saved when it next yielded):
|
||||
// 1. Pushes the six callee-saved integer registers (scheduler side).
|
||||
// 2. Stashes the target sp in rbx — free scratch: the register's live
|
||||
// value is on the stack we just pushed to, and the pops below load the
|
||||
// *other* side's values anyway — then snaps rsp into rdi and calls the
|
||||
// Rust helper that stores it in SCHEDULER_SP.
|
||||
// 3. Installs the target sp and pops the actor's registers; `ret` lands
|
||||
// where the actor yielded (or in `entry` on first resume).
|
||||
//
|
||||
// switch_to_scheduler_asm (no args; its "return value" materialises on the
|
||||
// OTHER stack, as switch_to_actor's rax):
|
||||
// 1. Pushes the six callee-saved integer registers (actor side).
|
||||
// 2. Stashes its own rsp in rbx (same free-scratch argument), asks the
|
||||
// Rust helper for SCHEDULER_SP.
|
||||
// 3. Installs the scheduler sp, moves the saved actor sp into rax, pops
|
||||
// the scheduler's registers and rets — completing the scheduler's
|
||||
// `switch_to_actor(sp)` call with the actor's new sp as its result.
|
||||
//
|
||||
// XMM registers are NOT saved here. We rely on every yield happening through
|
||||
// a Rust call site, which means the compiler has spilled any live XMM state
|
||||
// to the stack before we get here. (This is the same argument the compiler
|
||||
// uses internally — callee-saved regs are what survive a `call`, and the
|
||||
// SysV AMD64 ABI says XMM0–15 are all caller-saved.) If we ever yield from
|
||||
// a place that isn't a Rust call boundary, this assumption breaks.
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
#[unsafe(naked)]
|
||||
unsafe extern "C" fn switch_to_actor_asm(actor_sp: usize) -> usize {
|
||||
core::arch::naked_asm!(
|
||||
"push rbx", "push rbp", "push r12", "push r13", "push r14", "push r15",
|
||||
"mov rbx, rdi",
|
||||
"mov rdi, rsp",
|
||||
"call {set_sched_sp}",
|
||||
"mov rsp, rbx",
|
||||
"pop r15", "pop r14", "pop r13", "pop r12", "pop rbp", "pop rbx",
|
||||
"ret",
|
||||
set_sched_sp = sym set_scheduler_sp,
|
||||
);
|
||||
}
|
||||
|
||||
/// Resume the actor whose saved stack pointer is `actor_sp`. Returns when the
|
||||
/// actor yields, with the stack pointer the actor saved as it did — store it
|
||||
/// back into the slot for the next resume.
|
||||
///
|
||||
/// # Safety
|
||||
///
|
||||
/// `actor_sp` must be a valid saved actor stack pointer — either from
|
||||
/// `init_actor_stack` (first resume) or the value a prior `switch_to_actor`
|
||||
/// returned for this actor (subsequent resumes). Resuming with a stale or
|
||||
/// forged sp transfers control to an arbitrary address. Must not be called
|
||||
/// from within an actor (only the scheduler side may resume).
|
||||
pub unsafe fn switch_to_actor(actor_sp: usize) -> usize {
|
||||
unsafe { switch_to_actor_asm(actor_sp) }
|
||||
}
|
||||
|
||||
/// Yield from the running actor back to its scheduler thread. Returns when the
|
||||
/// actor is next resumed via [`switch_to_actor`].
|
||||
///
|
||||
/// # Safety
|
||||
///
|
||||
/// The caller must be running on an actor stack that was entered through
|
||||
/// [`switch_to_actor`], so that `SCHEDULER_SP` holds the live saved stack
|
||||
/// pointer of the scheduler side. Calling this from the scheduler thread, or
|
||||
/// before any actor has been resumed, transfers control to an arbitrary
|
||||
/// address.
|
||||
#[unsafe(naked)]
|
||||
pub unsafe extern "C" fn switch_to_scheduler() {
|
||||
core::arch::naked_asm!(
|
||||
"push rbx", "push rbp", "push r12", "push r13", "push r14", "push r15",
|
||||
"mov rbx, rsp",
|
||||
"call {get_sched_sp}",
|
||||
"mov rsp, rax",
|
||||
"mov rax, rbx",
|
||||
"pop r15", "pop r14", "pop r13", "pop r12", "pop rbp", "pop rbx",
|
||||
"ret",
|
||||
get_sched_sp = sym get_scheduler_sp,
|
||||
);
|
||||
}
|
||||
pub use crate::arch::{
|
||||
get_actor_sp, init_actor_stack, set_actor_sp, switch_to_actor, switch_to_scheduler,
|
||||
};
|
||||
|
||||
+82
-1236
File diff suppressed because it is too large
Load Diff
-1455
File diff suppressed because it is too large
Load Diff
@@ -1,453 +0,0 @@
|
||||
//! Inspect what is running right now: which actors exist, what state each one
|
||||
//! is in, and how they are related.
|
||||
//!
|
||||
//! This is the tool for questions like "is my server still alive", "how many
|
||||
//! actors are currently parked waiting on something", or "what does the spawn
|
||||
//! tree look like". It is meant for debugging, test assertions, a health check
|
||||
//! endpoint, or a monitoring dashboard: anywhere you want to look at the
|
||||
//! runtime from the outside without stopping it or coupling your code to its
|
||||
//! internals.
|
||||
//!
|
||||
//! Three entry points, in order of scope:
|
||||
//!
|
||||
//! - [`snapshot`] returns every actor that currently exists, as a plain
|
||||
//! owned `Vec`, so you can filter, count, or search it however you like.
|
||||
//! - [`actor_info`] returns a coherent view of exactly one actor, by pid.
|
||||
//! Cheaper than filtering a whole snapshot down to one entry, and more
|
||||
//! precise (see "Consistency" below).
|
||||
//! - [`tree`] returns the same actors as [`snapshot`], folded into a
|
||||
//! parent/child forest that mirrors who spawned whom.
|
||||
//!
|
||||
//! ```
|
||||
//! use smarm::{actor_info, channel, run, snapshot, spawn, ActorState};
|
||||
//!
|
||||
//! run(|| {
|
||||
//! let (ready_tx, ready_rx) = channel::<()>();
|
||||
//! let (gate_tx, gate_rx) = channel::<()>();
|
||||
//!
|
||||
//! let worker = spawn(move || {
|
||||
//! ready_tx.send(()).unwrap();
|
||||
//! gate_rx.recv().unwrap(); // blocks here until released
|
||||
//! });
|
||||
//! ready_rx.recv().unwrap();
|
||||
//!
|
||||
//! // `snapshot` sees every actor, including this one and the worker.
|
||||
//! let snap = snapshot();
|
||||
//! assert!(snap.actors.len() >= 2);
|
||||
//!
|
||||
//! // `actor_info` gives a coherent view of just the worker. It is
|
||||
//! // blocked on the gate channel, so it must be Parked.
|
||||
//! let pid = worker.pid();
|
||||
//! let info = actor_info(pid).expect("worker is still alive");
|
||||
//! assert_eq!(info.state, ActorState::Parked);
|
||||
//!
|
||||
//! gate_tx.send(()).unwrap();
|
||||
//! worker.join().unwrap();
|
||||
//!
|
||||
//! // Once joined, the pid no longer names a live actor.
|
||||
//! assert!(actor_info(pid).is_none());
|
||||
//! });
|
||||
//! ```
|
||||
//!
|
||||
//! ## Consistency
|
||||
//!
|
||||
//! [`snapshot`] is not a single atomic pause-the-world freeze: it walks every
|
||||
//! actor's state one after another, so it is a series of independent,
|
||||
//! cheap, lock-free reads rather than one coherent moment in time. Between
|
||||
//! reading actor A and actor B, either one can change state, and an actor can
|
||||
//! even finish and disappear mid-scan. In practice this is exactly what you
|
||||
//! want: a coherent stop-the-world snapshot would mean pausing every actor in
|
||||
//! the runtime just to look at it, which is expensive and rarely necessary
|
||||
//! for a dashboard, a test assertion, or a debugging session.
|
||||
//!
|
||||
//! [`actor_info`], in contrast, is coherent for the one actor it names: all of
|
||||
//! its fields describe the same instant for that actor, because a single
|
||||
//! actor's data cannot tear the way a scan across many actors can.
|
||||
//!
|
||||
//! ## Implementation notes
|
||||
//!
|
||||
//! These details matter if you are working on smarm itself; they are not part
|
||||
//! of the public contract.
|
||||
//!
|
||||
//! The read never stops the scheduler and never holds a lock across the whole
|
||||
//! scan. Each actor's scheduling state is a single lock-free word load
|
||||
//! (hence the possible tearing described above). Reading the rest of an
|
||||
//! actor's cold data (its supervisor, monitors, links, and so on) takes a
|
||||
//! brief per-actor lock, just long enough to copy those fields out; nothing
|
||||
//! is held across actors. Locking follows the crate-wide rule that at most
|
||||
//! one "leaf" lock (a per-actor lock, the registry lock, or the free list
|
||||
//! lock) is held at a time, with no leaf lock held while acquiring another.
|
||||
//! The read is phased accordingly: first one pass over the registry to
|
||||
//! collect every actor's registered names and mailbox depth, released before
|
||||
//! the per-actor scan begins.
|
||||
|
||||
use crate::pid::Pid;
|
||||
use crate::registry::MailboxInfo;
|
||||
use crate::runtime::{Slot, ROOT_PID};
|
||||
use crate::scheduler::with_runtime;
|
||||
use crate::slot_state::{
|
||||
word_gen, word_state, ST_DONE, ST_PARKED, ST_QUEUED, ST_RUNNING, ST_RUNNING_NOTIFIED,
|
||||
};
|
||||
use std::collections::HashMap;
|
||||
|
||||
/// The format version carried by every [`RuntimeSnapshot`] and
|
||||
/// [`RuntimeTree`], as [`RuntimeSnapshot::format_version`] /
|
||||
/// [`RuntimeTree::format_version`]. If you serialize a snapshot (for example
|
||||
/// to send it somewhere else, or to compare snapshots taken with different
|
||||
/// versions of smarm) check this field: a change in its value means the shape
|
||||
/// of [`ActorInfo`] or its neighbors has changed and old and new snapshots
|
||||
/// should not be assumed compatible. If you only ever read a snapshot
|
||||
/// in-process in the same version of smarm that produced it, you can ignore
|
||||
/// this field.
|
||||
pub const SNAPSHOT_FORMAT_VERSION: u16 = 1;
|
||||
|
||||
/// What an actor is doing right now, from the scheduler's point of view.
|
||||
///
|
||||
/// - `Queued`: runnable, waiting for a scheduler thread to pick it up.
|
||||
/// - `Running`: currently executing on a scheduler thread.
|
||||
/// - `Notified`: was running and got woken up (for example, a message
|
||||
/// arrived) before it had a chance to yield or park; it will be re-queued
|
||||
/// as soon as it does.
|
||||
/// - `Parked`: blocked, waiting on something such as a channel receive, a
|
||||
/// mutex, a timer, or an IO event.
|
||||
/// - `Done`: has finished (returned or panicked) but its slot has not been
|
||||
/// reclaimed for reuse yet, so it is still visible to introspection.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum ActorState {
|
||||
Queued,
|
||||
Running,
|
||||
Notified,
|
||||
Parked,
|
||||
Done,
|
||||
}
|
||||
|
||||
/// Classify a packed state word. `None` for a Vacant slot (skipped by the
|
||||
/// scan): a vacant slot holds no actor at all, live or done.
|
||||
fn classify(w: u64) -> Option<ActorState> {
|
||||
Some(match word_state(w) {
|
||||
ST_QUEUED => ActorState::Queued,
|
||||
ST_RUNNING => ActorState::Running,
|
||||
ST_RUNNING_NOTIFIED => ActorState::Notified,
|
||||
ST_PARKED => ActorState::Parked,
|
||||
ST_DONE => ActorState::Done,
|
||||
_ => return None, // ST_VACANT
|
||||
})
|
||||
}
|
||||
|
||||
/// An owned, self-contained view of one actor at (approximately) one moment.
|
||||
/// It borrows nothing from the runtime, so you can keep it, send it
|
||||
/// elsewhere, or print it long after the actor it describes has changed
|
||||
/// state or even exited.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct ActorInfo {
|
||||
pub pid: Pid,
|
||||
/// Names this actor is currently registered under (see the
|
||||
/// [`registry`](crate::registry) module). Usually empty or one name;
|
||||
/// an actor can have more if it registered several.
|
||||
pub names: Vec<&'static str>,
|
||||
pub state: ActorState,
|
||||
/// The actor that spawned this one: whoever called `spawn` or
|
||||
/// `spawn_under` to create it. This is a parentage record, not
|
||||
/// necessarily a supervision relationship: `spawn_under` records the
|
||||
/// supervisor you asked for, while plain `spawn` records the spawning
|
||||
/// actor itself, whether or not it supervises anything. It is the
|
||||
/// runtime's root pid for the run's own root actor, and for a `Done`
|
||||
/// actor whose bookkeeping has already been cleared.
|
||||
pub supervisor: Pid,
|
||||
pub trap_exit: bool,
|
||||
pub monitors: u32,
|
||||
pub links: u32,
|
||||
pub joiners: u32,
|
||||
/// Messages currently queued and not yet delivered, summed across every
|
||||
/// channel this actor has published (via `register`, `install`,
|
||||
/// `spawn_addr`, or starting a gen_server). This is 0 for an actor that
|
||||
/// only holds a private, unpublished `channel()` receiver, since nothing
|
||||
/// outside the actor can see that channel exists.
|
||||
pub mailbox_depth: u32,
|
||||
/// How many times this actor has been preempted for running past its
|
||||
/// scheduling timeslice. Counts only since the actor's current start (a
|
||||
/// supervisor restart begins a fresh count).
|
||||
pub overruns: u64,
|
||||
/// How many messages this actor has received (taken off its inbox), since
|
||||
/// its current start. Useful for spotting an actor whose mailbox is
|
||||
/// filling up faster than it can drain it: compare this against
|
||||
/// `mailbox_depth` over time.
|
||||
pub messages_received: u64,
|
||||
/// Approximate CPU cycles this actor has spent running, since its current
|
||||
/// start. A relative measure for comparing actors against each other, not
|
||||
/// an absolute or wall-clock figure. Always 0 unless the crate's
|
||||
/// `budget-accounting` feature is enabled, since measuring it costs a
|
||||
/// timestamp read on every resume.
|
||||
pub budget_cycles: u64,
|
||||
/// RFC 019 §8 — this actor's stack, as the runtime sees it. All fields
|
||||
/// are lock-free atomic reads, coherent for this incarnation via the
|
||||
/// same generation check as the counters above. Exact RSS is
|
||||
/// deliberately absent: `mincore` is debug tooling, never a runtime
|
||||
/// path.
|
||||
pub stack: StackInfo,
|
||||
}
|
||||
|
||||
/// RFC 019 §8 — per-actor stack introspection. Sizes are page-rounded, as
|
||||
/// [`Stack::new`](crate::stack::Stack::new) rounds them.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub struct StackInfo {
|
||||
/// Usable stack size ([`SpawnOpts::stack_reserve`](crate::SpawnOpts::stack_reserve)
|
||||
/// or the Config/default).
|
||||
pub reserve: usize,
|
||||
/// PROT_NONE guard below the usable region.
|
||||
pub guard: usize,
|
||||
/// Sampled high-water depth in bytes: `top − lowest saved sp`. Sampled,
|
||||
/// not exact — the context save at yields/parks/preemptions is the
|
||||
/// sampler (RFC 019 §2), so a spike the actor never yielded inside is
|
||||
/// invisible. 0 depth means "never descheduled at any depth", not
|
||||
/// "never ran".
|
||||
pub depth_high_water: usize,
|
||||
/// Parks on this incarnation since its last shrink (or since install if
|
||||
/// it has never shrunk) — the §3 cooldown counter, live.
|
||||
pub parks_since_shrink: u32,
|
||||
/// §3 shrinks performed on this incarnation.
|
||||
pub shrinks: u32,
|
||||
}
|
||||
|
||||
/// A snapshot of every actor in the runtime at (approximately) one moment.
|
||||
/// See the module docs' "Consistency" section for what "approximately" means
|
||||
/// here.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct RuntimeSnapshot {
|
||||
pub format_version: u16,
|
||||
pub actors: Vec<ActorInfo>,
|
||||
}
|
||||
|
||||
/// Every actor that currently exists: running, queued, parked, or finished
|
||||
/// but not yet cleaned up. Cheap and lock-free per actor; see the module
|
||||
/// docs for what "approximately one moment" means for the result as a whole.
|
||||
/// Panics if called outside [`run`](crate::run).
|
||||
pub fn snapshot() -> RuntimeSnapshot {
|
||||
with_runtime(|inner| {
|
||||
// First pass: one registry lock to collect names + mailbox depth for
|
||||
// every actor, released before touching any per-actor lock below.
|
||||
let mail = inner.registry.lock().introspect_map();
|
||||
|
||||
// Second pass: walk the actor table. Each actor's scheduling state is
|
||||
// a lock-free word load; only copying its other fields takes a brief
|
||||
// per-actor lock. Tearing across actors is expected here (see the
|
||||
// module docs' "Consistency" section).
|
||||
let mut actors = Vec::new();
|
||||
for (idx, slot) in inner.slots.iter().enumerate() {
|
||||
let idx = idx as u32;
|
||||
if let Some(info) = read_slot(slot, idx, mail.get(&idx)) {
|
||||
actors.push(info);
|
||||
}
|
||||
}
|
||||
RuntimeSnapshot {
|
||||
format_version: SNAPSHOT_FORMAT_VERSION,
|
||||
actors,
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
/// A coherent view of exactly one actor, or `None` if `pid` does not name a
|
||||
/// currently-live entry: it is stale (that actor has already exited and its
|
||||
/// slot was reused by another), out of range, or was never a real pid at
|
||||
/// all. Unlike [`snapshot`], every field of the result describes the same
|
||||
/// instant, since there is only one actor to read.
|
||||
/// The stack shape `(reserve, guard)` of a live actor, page-rounded — the
|
||||
/// RFC 019 introspection surface's first field (depth sampling and shrink
|
||||
/// counters land with the shrink machinery). `None` if `pid` no longer names
|
||||
/// a live actor. Takes the actor's cold lock briefly; debugging/assertion
|
||||
/// use, not a hot-path call.
|
||||
pub fn stack_shape(pid: Pid) -> Option<(usize, usize)> {
|
||||
with_runtime(|inner| {
|
||||
let slot = inner.slot_at(pid)?;
|
||||
let cold = slot.cold.lock();
|
||||
if slot.generation() != pid.generation() {
|
||||
return None;
|
||||
}
|
||||
cold.actor.as_ref().map(|a| a.stack.shape())
|
||||
})
|
||||
}
|
||||
|
||||
pub fn actor_info(pid: Pid) -> Option<ActorInfo> {
|
||||
with_runtime(|inner| {
|
||||
let slot = inner.slot_at(pid)?;
|
||||
let mail = inner.registry.lock().introspect_one(pid.index());
|
||||
let info = read_slot(slot, pid.index(), mail.as_ref())?;
|
||||
// read_slot keys on the slab's *current* generation; reject if that
|
||||
// isn't the incarnation the caller asked about.
|
||||
(info.pid.generation() == pid.generation()).then_some(info)
|
||||
})
|
||||
}
|
||||
|
||||
/// Build one `ActorInfo` for slot `idx`, or `None` if the slot is empty or
|
||||
/// was reclaimed while this read was in progress. The scheduling state comes
|
||||
/// from a lock-free word load (the source of the tearing described in the
|
||||
/// module docs); the per-actor lock then confirms the actor has not since
|
||||
/// exited and been replaced, so the rest of the fields are coherent for this
|
||||
/// exact actor. `mail` is this slot's registry entry, if any.
|
||||
fn read_slot(slot: &Slot, idx: u32, mail: Option<&MailboxInfo>) -> Option<ActorInfo> {
|
||||
let w = slot.state_word();
|
||||
let state = classify(w)?;
|
||||
let gen = word_gen(w);
|
||||
let pid = Pid::new(idx, gen);
|
||||
|
||||
let cold = slot.cold.lock();
|
||||
// If the generation moved between the lock-free load and acquiring the
|
||||
// per-actor lock, this actor exited (and the slot may already hold a new
|
||||
// one). Drop it rather than mix one actor's state with another's data; a
|
||||
// racing actor may simply be missed by this scan, which is expected (see
|
||||
// the module docs' "Consistency" section).
|
||||
if word_gen(slot.state_word()) != gen {
|
||||
return None;
|
||||
}
|
||||
let (supervisor, trap_exit) = match cold.actor.as_ref() {
|
||||
// Live incarnation: parent + trap live on the Actor.
|
||||
Some(actor) => (actor.supervisor, actor.trap.is_some()),
|
||||
// Done tombstone: the Actor was taken at finalize and the collections
|
||||
// cleared, so report it root-less with empty counts.
|
||||
None => (ROOT_PID, false),
|
||||
};
|
||||
let monitors = cold.monitors.len() as u32;
|
||||
let links = cold.links.len() as u32;
|
||||
let joiners = cold.waiters.len() as u32;
|
||||
drop(cold);
|
||||
|
||||
// Counters are plain atomics, read lock-free.
|
||||
let (reserve, guard, top, hwm, parks_since_shrink, shrinks) = slot.stack_introspect();
|
||||
let stack = StackInfo {
|
||||
reserve,
|
||||
guard,
|
||||
depth_high_water: top.saturating_sub(hwm),
|
||||
parks_since_shrink,
|
||||
shrinks,
|
||||
};
|
||||
let overruns = slot.overruns();
|
||||
let messages_received = slot.messages_received();
|
||||
let budget_cycles = slot.budget_cycles();
|
||||
|
||||
// Names + depth belong to this incarnation only if the registry mailbox's
|
||||
// pid matches the slab generation; a stale registry entry (dead prior
|
||||
// occupant, not yet pruned) contributes nothing.
|
||||
let (names, mailbox_depth) = match mail {
|
||||
Some(mi) if mi.pid.generation() == gen => (mi.names.clone(), mi.depth),
|
||||
_ => (Vec::new(), 0),
|
||||
};
|
||||
|
||||
Some(ActorInfo {
|
||||
pid,
|
||||
names,
|
||||
state,
|
||||
supervisor,
|
||||
trap_exit,
|
||||
monitors,
|
||||
links,
|
||||
joiners,
|
||||
mailbox_depth,
|
||||
overruns,
|
||||
messages_received,
|
||||
budget_cycles,
|
||||
stack,
|
||||
})
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Tree view: a pure derivation over a snapshot
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// One node in the parentage forest returned by [`tree`]. `children` are the
|
||||
/// actors whose recorded parent (see [`ActorInfo::supervisor`]) points at
|
||||
/// this node's actor.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct TreeNode {
|
||||
pub info: ActorInfo,
|
||||
/// True if this actor's recorded parent was not found in the snapshot
|
||||
/// (it had already exited, or was itself missing), so this node was
|
||||
/// placed at the top of the forest instead of being dropped. This keeps
|
||||
/// every actor in the snapshot visible somewhere in the tree, even one
|
||||
/// whose parent is gone.
|
||||
pub orphaned: bool,
|
||||
pub children: Vec<TreeNode>,
|
||||
}
|
||||
|
||||
/// The parentage forest: every actor from a snapshot, arranged by who spawned
|
||||
/// whom. Roots are actors with no parent in the snapshot (including the
|
||||
/// run's own root actor) plus any orphaned actors (see [`TreeNode::orphaned`]).
|
||||
/// This mirrors spawn parentage, not necessarily a supervision tree; see
|
||||
/// [`ActorInfo::supervisor`].
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct RuntimeTree {
|
||||
pub format_version: u16,
|
||||
pub roots: Vec<TreeNode>,
|
||||
}
|
||||
|
||||
/// Take a fresh [`snapshot`] and fold it into the parentage forest.
|
||||
pub fn tree() -> RuntimeTree {
|
||||
tree_from(snapshot())
|
||||
}
|
||||
|
||||
/// Fold an existing snapshot into a parentage forest by grouping each actor
|
||||
/// under its parent, without taking a new snapshot. Useful if you already
|
||||
/// have one (for example, one built in a test, or one you took earlier and
|
||||
/// want to inspect again) and want the tree view of it without re-reading
|
||||
/// the runtime.
|
||||
pub fn tree_from(snap: RuntimeSnapshot) -> RuntimeTree {
|
||||
let RuntimeSnapshot {
|
||||
format_version,
|
||||
actors,
|
||||
} = snap;
|
||||
|
||||
let mut index_of: HashMap<Pid, usize> = HashMap::with_capacity(actors.len());
|
||||
for (i, a) in actors.iter().enumerate() {
|
||||
index_of.insert(a.pid, i);
|
||||
}
|
||||
|
||||
// Group children under their present parent; everything else is a root.
|
||||
// Scan order is preserved within each parent's child list.
|
||||
let mut children_of: HashMap<Pid, Vec<usize>> = HashMap::new();
|
||||
let mut roots: Vec<usize> = Vec::new();
|
||||
let mut orphaned = vec![false; actors.len()];
|
||||
for (i, a) in actors.iter().enumerate() {
|
||||
let parent = a.supervisor;
|
||||
if parent != ROOT_PID && index_of.contains_key(&parent) {
|
||||
children_of.entry(parent).or_default().push(i);
|
||||
} else {
|
||||
// Parent is the forest sentinel (genuine root) or absent from the
|
||||
// snapshot (orphan): either way, a root of the forest.
|
||||
orphaned[i] = parent != ROOT_PID;
|
||||
roots.push(i);
|
||||
}
|
||||
}
|
||||
|
||||
// `take()` each actor as it is placed, which also guards against a
|
||||
// (constructionally impossible) parentage cycle re-entering a node.
|
||||
let mut slots: Vec<Option<ActorInfo>> = actors.into_iter().map(Some).collect();
|
||||
let root_nodes = roots
|
||||
.into_iter()
|
||||
.filter_map(|i| build_node(i, &children_of, &orphaned, &mut slots))
|
||||
.collect();
|
||||
RuntimeTree {
|
||||
format_version,
|
||||
roots: root_nodes,
|
||||
}
|
||||
}
|
||||
|
||||
fn build_node(
|
||||
i: usize,
|
||||
children_of: &HashMap<Pid, Vec<usize>>,
|
||||
orphaned: &[bool],
|
||||
slots: &mut [Option<ActorInfo>],
|
||||
) -> Option<TreeNode> {
|
||||
let info = slots[i].take()?; // already placed → cycle guard / no double-attach
|
||||
let children = children_of
|
||||
.get(&info.pid)
|
||||
.map(|kids| {
|
||||
kids.iter()
|
||||
.filter_map(|&c| build_node(c, children_of, orphaned, slots))
|
||||
.collect()
|
||||
})
|
||||
.unwrap_or_default();
|
||||
Some(TreeNode {
|
||||
info,
|
||||
orphaned: orphaned[i],
|
||||
children,
|
||||
})
|
||||
}
|
||||
@@ -13,68 +13,42 @@
|
||||
//! leaves the actor, no copying through an intermediary thread. Built on
|
||||
//! these are the conveniences `read(fd, &mut buf)` and `write(fd, &buf)`.
|
||||
//!
|
||||
//! Architecture (RFC 018: driver-enqueues)
|
||||
//! =======================================
|
||||
//! Per `run()`, two OS threads, each a *producer* behind the runtime's
|
||||
//! two-call contract — make the actor runnable (`unpark_at`, whose enqueue
|
||||
//! tail wakes a parked scheduler), nothing else:
|
||||
//! Architecture
|
||||
//! ============
|
||||
//! Per `run()`, two OS threads:
|
||||
//! - **epoll thread**: owns the epollfd. Loops in `epoll_wait`. On a
|
||||
//! ready fd, pushes `Completion::FdReady { pid, fd, events }` to the
|
||||
//! shared completion queue and writes the scheduler-wake pipe. On the
|
||||
//! shutdown pipe (also registered in epollfd), exits.
|
||||
//! - **pool thread**: blocks on the request mpsc. Runs the closure
|
||||
//! inside `catch_unwind`, pushes `Completion::Blocking { pid, result }`,
|
||||
//! writes the scheduler-wake pipe.
|
||||
//!
|
||||
//! - **epoll thread**: owns `epoll_wait` on the epollfd. On a ready fd it
|
||||
//! removes the parked waiter from the shared `waiters` map and DELs the
|
||||
//! fd (both under the waiters lock — see below), then unparks the
|
||||
//! actor directly. On the shutdown pipe (also registered in the
|
||||
//! epollfd), exits.
|
||||
//! - **pool thread**: blocks on the request mpsc. Runs the closure inside
|
||||
//! `catch_unwind`, stashes the result in the actor's slot
|
||||
//! (`pending_io_result`, under the cold lock, generation-checked),
|
||||
//! decrements the runtime's `io_outstanding`, and unparks the actor.
|
||||
//! Both threads share a single `completions: Arc<Mutex<VecDeque<Completion>>>`
|
||||
//! and the same scheduler-wake pipe.
|
||||
//!
|
||||
//! There is no shared completion queue and no wake pipe: each producer
|
||||
//! routes its own completion, so the whole byte-vs-completion visibility
|
||||
//! discipline of the drain era — and the stranded-completion hazards it
|
||||
//! defended against — is unrepresentable. Producers reach the runtime
|
||||
//! through a `Weak<RuntimeInner>`: upgraded per completion (the path is
|
||||
//! syscall-bound; the refcount op is noise) and avoiding an Arc cycle
|
||||
//! through `RuntimeInner::io`.
|
||||
//!
|
||||
//! `epoll_ctl` (register fd interest) is called by the scheduler thread
|
||||
//! directly on the epollfd. That's well-defined per `epoll_ctl(2)`: a
|
||||
//! thread may be calling `epoll_wait` on the epollfd while another thread
|
||||
//! calls `epoll_ctl`.
|
||||
//! `epoll_ctl` (register/unregister fd interest) is called by the
|
||||
//! scheduler thread *directly* on the epollfd. That's well-defined per
|
||||
//! `epoll_ctl(2)`: a thread may be calling `epoll_wait` on the epollfd
|
||||
//! while another thread calls `epoll_ctl`. Avoids needing a second mpsc
|
||||
//! and a second wake mechanism.
|
||||
//!
|
||||
//! Epoll mode
|
||||
//! ==========
|
||||
//! Level-triggered with EPOLLONESHOT. After a wakeup the kernel
|
||||
//! auto-disarms the fd, so we never get two wakeups for one
|
||||
//! `wait_readable` call. The epoll thread explicitly `EPOLL_CTL_DEL`s the
|
||||
//! fd on readiness to free the slot for re-registration. Net effect: each
|
||||
//! `wait_readable` call. The scheduler explicitly `EPOLL_CTL_DEL`s the fd
|
||||
//! on completion to free the slot for re-registration. Net effect: each
|
||||
//! `wait_readable(fd)` is one ADD, one wakeup, one DEL — symmetric and
|
||||
//! stateless between calls.
|
||||
//!
|
||||
//! ## The waiters lock is the ADD/DEL serialization
|
||||
//!
|
||||
//! Registration (scheduler thread: check-vacant, defensive DEL, ADD,
|
||||
//! insert) and readiness consumption (epoll thread: remove, DEL) each run
|
||||
//! entirely under the `waiters` mutex. This is what makes the
|
||||
//! oneshot-rearm race unrepresentable: a woken actor re-registering the
|
||||
//! same fd cannot interleave with the epoll thread's DEL for the *previous*
|
||||
//! registration — whichever takes the lock second sees a consistent
|
||||
//! kernel-side state. Lock order: `io` (the runtime's outer mutex, held by
|
||||
//! scheduler-side callers) → `waiters` → slot/queue leaves via `unpark_at`.
|
||||
//! The epoll thread takes `waiters` without `io` — it must never take
|
||||
//! `io`, both for lock-order hygiene and because teardown holds `io` while
|
||||
//! joining it.
|
||||
//!
|
||||
//! Fd hygiene
|
||||
//! ==========
|
||||
//! An actor stopped while waiting on an fd unwinds out of `wait_fd`'s park;
|
||||
//! a drop guard there (armed after a successful register, forgotten on a
|
||||
//! normal wake) calls [`IoThread::cancel_waiter`], which removes the
|
||||
//! `waiters` entry iff it is still that wait's `(pid, epoch)` and only then
|
||||
//! `EPOLL_CTL_DEL`s the fd — an entry already consumed by the epoll thread
|
||||
//! means the fd may carry someone else's fresh registration, which must be
|
||||
//! left alone. `epoll_register` keeps a defensive bare DEL before ADD as
|
||||
//! belt-and-braces.
|
||||
//! If an actor dies while waiting on an fd, the registration is leaked
|
||||
//! (the fd stays in the epollfd, armed). EPOLLONESHOT bounds the damage:
|
||||
//! at most one stale wakeup, after which the kernel disarms. The stale
|
||||
//! wakeup hits a dead pid in `waiters` and is dropped. Acceptable for v0.2;
|
||||
//! a future pass should DEL on actor death.
|
||||
//!
|
||||
//! Buffers used with `read`/`write` should be on fds opened with
|
||||
//! `O_NONBLOCK`. If they aren't, the syscall may block the scheduler
|
||||
@@ -92,14 +66,13 @@
|
||||
//! they have no equivalent panic-propagation path.
|
||||
|
||||
use crate::pid::Pid;
|
||||
use crate::runtime::RuntimeInner;
|
||||
use std::any::Any;
|
||||
use std::collections::HashMap;
|
||||
use std::collections::{HashMap, VecDeque};
|
||||
use std::io;
|
||||
use std::os::fd::RawFd;
|
||||
use std::panic;
|
||||
use std::sync::atomic::Ordering;
|
||||
use std::sync::{mpsc, Arc, Mutex, Weak};
|
||||
use std::sync::mpsc;
|
||||
use std::sync::{Arc, Mutex};
|
||||
use std::thread::JoinHandle as OsJoinHandle;
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -111,31 +84,42 @@ use std::thread::JoinHandle as OsJoinHandle;
|
||||
pub type IoResult = Result<Box<dyn Any + Send>, Box<dyn Any + Send>>;
|
||||
|
||||
struct Request {
|
||||
/// The submitter's park-epoch — the eventual wake is epoch-matched.
|
||||
epoch: u32,
|
||||
pid: Pid,
|
||||
/// The work to perform. Returns the wire-form result directly.
|
||||
work: Box<dyn FnOnce() -> IoResult + Send>,
|
||||
}
|
||||
|
||||
/// The parked-waiter map, shared between scheduler-side registration and
|
||||
/// the epoll thread's readiness consumption. See the module docs on why
|
||||
/// this single lock is the ADD/DEL serialization.
|
||||
type Waiters = Arc<Mutex<HashMap<RawFd, (Pid, u32)>>>;
|
||||
/// Completion message from either IO thread back to the scheduler.
|
||||
pub enum Completion {
|
||||
/// A `block_on_io` closure has finished (Ok = return value, Err = panic
|
||||
/// payload).
|
||||
Blocking { pid: Pid, result: IoResult },
|
||||
/// An fd registered via `wait_readable`/`wait_writable` is ready. The
|
||||
/// scheduler looks up the parked pid in `waiters`, unparks it, and
|
||||
/// removes the entry. `pid` isn't in this variant because the epoll
|
||||
/// thread doesn't have access to the `waiters` map; the scheduler
|
||||
/// thread owns that.
|
||||
FdReady { fd: RawFd, events: u32 },
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// IoThread — created per `run()`, owned by `RuntimeInner::io`.
|
||||
// IoThread — created per `run()`, owned by `SchedulerState`.
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
pub struct IoThread {
|
||||
// ----- Channels & queues -----
|
||||
|
||||
/// Submission queue into the blocking-work pool.
|
||||
tx: mpsc::Sender<Request>,
|
||||
/// One parked actor per registered fd. Populated by `epoll_register`,
|
||||
/// consumed by the epoll thread on readiness or `cancel_waiter` on an
|
||||
/// unwound wait.
|
||||
waiters: Waiters,
|
||||
/// Shared completion queue, fed by both the pool and the epoll thread.
|
||||
completions: Arc<Mutex<VecDeque<Completion>>>,
|
||||
/// Pipe the scheduler polls in its idle path. Both IO threads write to
|
||||
/// `wake_write` after pushing a completion.
|
||||
wake_read: RawFd,
|
||||
wake_write: RawFd,
|
||||
|
||||
// ----- Epoll machinery -----
|
||||
|
||||
/// The epollfd, owned by `IoThread`. Callable cross-thread via
|
||||
/// `epoll_ctl` per the man page.
|
||||
epollfd: RawFd,
|
||||
@@ -144,24 +128,39 @@ pub struct IoThread {
|
||||
/// shutdown.
|
||||
shutdown_read: RawFd,
|
||||
shutdown_write: RawFd,
|
||||
/// One parked actor per registered fd. Populated by `wait_readable` /
|
||||
/// `wait_writable` and drained by the scheduler when a `FdReady`
|
||||
/// completion is processed.
|
||||
pub waiters: HashMap<RawFd, Pid>,
|
||||
|
||||
// ----- Threads -----
|
||||
|
||||
pool_thread: Option<OsJoinHandle<()>>,
|
||||
epoll_thread: Option<OsJoinHandle<()>>,
|
||||
|
||||
/// Number of `block_on_io` requests in-flight. Used by the scheduler's
|
||||
/// idle path to decide whether to wait on the pipe or exit. Fd waits
|
||||
/// are not counted here; they're counted by `waiters.len()`.
|
||||
pub outstanding: u32,
|
||||
}
|
||||
|
||||
impl IoThread {
|
||||
/// Start the pool and epoll threads. `rt` is the producers' route back
|
||||
/// into the runtime (slot table + unpark protocol); a `Weak` so the
|
||||
/// `RuntimeInner → IoThread → RuntimeInner` cycle never forms.
|
||||
pub(crate) fn start(rt: Weak<RuntimeInner>) -> io::Result<Self> {
|
||||
// Pool submission channel.
|
||||
pub fn start() -> io::Result<Self> {
|
||||
// Scheduler-facing wake pipe.
|
||||
let (wake_read, wake_write) = make_pipe()?;
|
||||
// Pool submission channel + shared completion queue.
|
||||
let (tx, rx) = mpsc::channel::<Request>();
|
||||
let waiters: Waiters = Arc::new(Mutex::new(HashMap::new()));
|
||||
let completions: Arc<Mutex<VecDeque<Completion>>> =
|
||||
Arc::new(Mutex::new(VecDeque::new()));
|
||||
|
||||
// Epoll machinery.
|
||||
let epollfd = unsafe { libc::epoll_create1(libc::EPOLL_CLOEXEC) };
|
||||
if epollfd < 0 {
|
||||
// Best-effort fd cleanup before bailing.
|
||||
unsafe {
|
||||
libc::close(wake_read);
|
||||
libc::close(wake_write);
|
||||
}
|
||||
return Err(io::Error::last_os_error());
|
||||
}
|
||||
|
||||
@@ -170,6 +169,8 @@ impl IoThread {
|
||||
Err(e) => {
|
||||
unsafe {
|
||||
libc::close(epollfd);
|
||||
libc::close(wake_read);
|
||||
libc::close(wake_write);
|
||||
}
|
||||
return Err(e);
|
||||
}
|
||||
@@ -196,77 +197,91 @@ impl IoThread {
|
||||
libc::close(epollfd);
|
||||
libc::close(shutdown_read);
|
||||
libc::close(shutdown_write);
|
||||
libc::close(wake_read);
|
||||
libc::close(wake_write);
|
||||
}
|
||||
return Err(e);
|
||||
}
|
||||
|
||||
// Spawn pool thread.
|
||||
let pool_rt = rt.clone();
|
||||
let pool_comps = completions.clone();
|
||||
let pool_thread = std::thread::Builder::new()
|
||||
.name("smarm-io-pool".into())
|
||||
.spawn(move || pool_loop(rx, pool_rt))?;
|
||||
.spawn(move || pool_loop(rx, pool_comps, wake_write))?;
|
||||
|
||||
// Spawn epoll thread.
|
||||
let epoll_waiters = waiters.clone();
|
||||
let epoll_comps = completions.clone();
|
||||
let epoll_thread = std::thread::Builder::new()
|
||||
.name("smarm-io-epoll".into())
|
||||
.spawn(move || epoll_loop(epollfd, epoll_waiters, rt))?;
|
||||
.spawn(move || epoll_loop(epollfd, epoll_comps, wake_write))?;
|
||||
|
||||
Ok(Self {
|
||||
tx,
|
||||
waiters,
|
||||
completions,
|
||||
wake_read,
|
||||
wake_write,
|
||||
epollfd,
|
||||
shutdown_read,
|
||||
shutdown_write,
|
||||
waiters: HashMap::new(),
|
||||
pool_thread: Some(pool_thread),
|
||||
epoll_thread: Some(epoll_thread),
|
||||
outstanding: 0,
|
||||
})
|
||||
}
|
||||
|
||||
/// Hand a request to the pool. The caller (scheduler.rs) increments
|
||||
/// `io_outstanding` BEFORE calling — the pool decrements on completion,
|
||||
/// and an increment that trailed the completion would underflow.
|
||||
pub fn submit(&mut self, pid: Pid, epoch: u32, work: Box<dyn FnOnce() -> IoResult + Send>) {
|
||||
/// Hand a request to the pool. Increments `outstanding`.
|
||||
pub fn submit(&mut self, pid: Pid, work: Box<dyn FnOnce() -> IoResult + Send>) {
|
||||
self.outstanding += 1;
|
||||
// Send can only fail if the pool has hung up, which only happens
|
||||
// on shutdown. submit during shutdown is a bug.
|
||||
if self.tx.send(Request { pid, epoch, work }).is_err() {
|
||||
panic!("smarm: io pool hung up unexpectedly (submit during shutdown)");
|
||||
self.tx
|
||||
.send(Request { pid, work })
|
||||
.expect("io pool hung up unexpectedly");
|
||||
}
|
||||
|
||||
/// Drain every available completion. Caller (the scheduler) routes the
|
||||
/// results and updates `outstanding` / `waiters` accordingly.
|
||||
pub fn drain_completions(&mut self) -> Vec<Completion> {
|
||||
let mut q = self.completions.lock().unwrap();
|
||||
let mut out = Vec::with_capacity(q.len());
|
||||
while let Some(c) = q.pop_front() {
|
||||
out.push(c);
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
pub fn wake_fd(&self) -> RawFd {
|
||||
self.wake_read
|
||||
}
|
||||
|
||||
/// Register interest in `fd` becoming readable/writable; record `pid`
|
||||
/// as the parked waiter. The epoll thread unparks it on readiness.
|
||||
/// The caller increments `io_fd_waiters` BEFORE calling (mirror of
|
||||
/// `submit`'s contract) and decrements it again if this errors.
|
||||
/// as the parked waiter. The epoll thread will push a `FdReady`
|
||||
/// completion when the kernel signals.
|
||||
///
|
||||
/// EPOLLONESHOT: one wakeup per registration; the epoll thread DELs on
|
||||
/// readiness, `cancel_waiter` DELs on an unwound wait.
|
||||
/// EPOLLONESHOT: one wakeup per registration. The scheduler must
|
||||
/// `epoll_del` on completion to free the slot for re-registration.
|
||||
pub fn epoll_register(
|
||||
&mut self,
|
||||
fd: RawFd,
|
||||
pid: Pid,
|
||||
epoch: u32,
|
||||
readable: bool,
|
||||
writable: bool,
|
||||
) -> io::Result<()> {
|
||||
let mut waiters = match self.waiters.lock() {
|
||||
Ok(g) => g,
|
||||
Err(e) => panic!("smarm: io waiters lock poisoned (core corrupt): {e}"),
|
||||
};
|
||||
// Two actors waiting on the same fd would be a misuse: the kernel
|
||||
// delivers exactly one EPOLLONESHOT wakeup, so the second waiter
|
||||
// would hang. Reject up front.
|
||||
if waiters.contains_key(&fd) {
|
||||
if self.waiters.contains_key(&fd) {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::AlreadyExists,
|
||||
"fd already has a parked waiter",
|
||||
));
|
||||
}
|
||||
|
||||
// Belt-and-braces: `cancel_waiter` is responsible for cleaning up a
|
||||
// stopped waiter's registration, but a bare DEL is harmless if the
|
||||
// fd isn't registered (ENOENT) and removes any leak a path we
|
||||
// haven't thought of might leave behind.
|
||||
// Defensive cleanup: if a previous actor died while waiting on this
|
||||
// fd, the kernel-side registration was leaked (we don't walk all
|
||||
// waiters on actor death). A bare DEL is harmless if the fd isn't
|
||||
// registered (ENOENT), and removes any leak.
|
||||
unsafe {
|
||||
libc::epoll_ctl(self.epollfd, libc::EPOLL_CTL_DEL, fd, std::ptr::null_mut());
|
||||
}
|
||||
@@ -282,34 +297,25 @@ impl IoThread {
|
||||
events,
|
||||
u64: fd as u64,
|
||||
};
|
||||
let r =
|
||||
unsafe { libc::epoll_ctl(self.epollfd, libc::EPOLL_CTL_ADD, fd, &mut ev as *mut _) };
|
||||
let r = unsafe {
|
||||
libc::epoll_ctl(self.epollfd, libc::EPOLL_CTL_ADD, fd, &mut ev as *mut _)
|
||||
};
|
||||
if r < 0 {
|
||||
return Err(io::Error::last_os_error());
|
||||
}
|
||||
waiters.insert(fd, (pid, epoch));
|
||||
self.waiters.insert(fd, pid);
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Remove `fd`'s waiter iff it is still `(pid, epoch)`, DELing the fd
|
||||
/// from the epollfd in the same critical section. Returns whether the
|
||||
/// entry was removed (the caller then decrements `io_fd_waiters`).
|
||||
/// `false` means the epoll thread consumed the registration first —
|
||||
/// the fd may already carry someone else's fresh ADD; hands off.
|
||||
pub fn cancel_waiter(&mut self, fd: RawFd, pid: Pid, epoch: u32) -> bool {
|
||||
let mut waiters = match self.waiters.lock() {
|
||||
Ok(g) => g,
|
||||
Err(e) => panic!("smarm: io waiters lock poisoned (core corrupt): {e}"),
|
||||
};
|
||||
if waiters.get(&fd) == Some(&(pid, epoch)) {
|
||||
waiters.remove(&fd);
|
||||
// EPOLL_CTL_DEL of an already-removed fd returns ENOENT; ignore.
|
||||
unsafe {
|
||||
libc::epoll_ctl(self.epollfd, libc::EPOLL_CTL_DEL, fd, std::ptr::null_mut());
|
||||
}
|
||||
true
|
||||
} else {
|
||||
false
|
||||
/// Remove `fd` from the epollfd. Called by the scheduler after a
|
||||
/// `FdReady` completion, so the next `wait_readable(fd)` can ADD again.
|
||||
///
|
||||
/// Does NOT touch `waiters` — that's the scheduler's bookkeeping; this
|
||||
/// is purely the kernel-side cleanup.
|
||||
pub fn epoll_deregister(&mut self, fd: RawFd) {
|
||||
// EPOLL_CTL_DEL of an already-removed fd returns ENOENT; ignore.
|
||||
unsafe {
|
||||
libc::epoll_ctl(self.epollfd, libc::EPOLL_CTL_DEL, fd, std::ptr::null_mut());
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -330,10 +336,7 @@ impl Drop for IoThread {
|
||||
let real_tx = std::mem::replace(&mut self.tx, dead_tx);
|
||||
drop(real_tx);
|
||||
|
||||
// 3. Join both threads. Safe even while the caller holds the
|
||||
// runtime's `io` mutex: neither thread ever takes it (they reach
|
||||
// the runtime through a Weak they upgrade per completion, and
|
||||
// the epoll thread's only lock is `waiters`).
|
||||
// 3. Join both threads.
|
||||
if let Some(h) = self.epoll_thread.take() {
|
||||
let _ = h.join();
|
||||
}
|
||||
@@ -346,6 +349,8 @@ impl Drop for IoThread {
|
||||
libc::close(self.epollfd);
|
||||
libc::close(self.shutdown_read);
|
||||
libc::close(self.shutdown_write);
|
||||
libc::close(self.wake_read);
|
||||
libc::close(self.wake_write);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -356,38 +361,36 @@ impl Drop for IoThread {
|
||||
const SHUTDOWN_EPOLL_TOKEN: u64 = u64::MAX;
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Pool loop (producer: Blocking completions)
|
||||
// Pool loop
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
fn pool_loop(rx: mpsc::Receiver<Request>, rt: Weak<RuntimeInner>) {
|
||||
while let Ok(Request { pid, epoch, work }) = rx.recv() {
|
||||
fn pool_loop(
|
||||
rx: mpsc::Receiver<Request>,
|
||||
completions: Arc<Mutex<VecDeque<Completion>>>,
|
||||
wake_write: RawFd,
|
||||
) {
|
||||
while let Ok(Request { pid, work }) = rx.recv() {
|
||||
let result: IoResult = match panic::catch_unwind(panic::AssertUnwindSafe(work)) {
|
||||
Ok(r) => r,
|
||||
Err(payload) => Err(payload),
|
||||
};
|
||||
let Some(inner) = rt.upgrade() else { return };
|
||||
// Stash the result under the cold lock (generation-checked: an
|
||||
// actor stopped with the op in flight discards it), decrement the
|
||||
// in-flight count, then wake through the epoch-matched unpark. The
|
||||
// unpark's enqueue tail wakes a parked scheduler; the actor stays
|
||||
// `live` until it resumes and finalizes, so the decrement's
|
||||
// ordering against the termination verdict is not load-bearing.
|
||||
if let Some(slot) = inner.slot_at(pid) {
|
||||
let mut cold = slot.cold.lock();
|
||||
if slot.generation() == pid.generation() {
|
||||
cold.pending_io_result = Some(result);
|
||||
}
|
||||
}
|
||||
inner.io_outstanding.fetch_sub(1, Ordering::AcqRel);
|
||||
inner.unpark_at(pid, epoch);
|
||||
completions
|
||||
.lock()
|
||||
.unwrap()
|
||||
.push_back(Completion::Blocking { pid, result });
|
||||
wake_scheduler(wake_write);
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Epoll loop (producer: FdReady completions)
|
||||
// Epoll loop
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
fn epoll_loop(epollfd: RawFd, waiters: Waiters, rt: Weak<RuntimeInner>) {
|
||||
fn epoll_loop(
|
||||
epollfd: RawFd,
|
||||
completions: Arc<Mutex<VecDeque<Completion>>>,
|
||||
wake_write: RawFd,
|
||||
) {
|
||||
// Buffer for epoll_wait. 64 is plenty for our scale; if a real load
|
||||
// appears that needs more, this is a one-line change.
|
||||
const MAX_EVENTS: usize = 64;
|
||||
@@ -395,7 +398,12 @@ fn epoll_loop(epollfd: RawFd, waiters: Waiters, rt: Weak<RuntimeInner>) {
|
||||
|
||||
loop {
|
||||
let n = unsafe {
|
||||
libc::epoll_wait(epollfd, events.as_mut_ptr(), MAX_EVENTS as libc::c_int, -1)
|
||||
libc::epoll_wait(
|
||||
epollfd,
|
||||
events.as_mut_ptr(),
|
||||
MAX_EVENTS as libc::c_int,
|
||||
-1,
|
||||
)
|
||||
};
|
||||
|
||||
if n < 0 {
|
||||
@@ -410,45 +418,54 @@ fn epoll_loop(epollfd: RawFd, waiters: Waiters, rt: Weak<RuntimeInner>) {
|
||||
}
|
||||
|
||||
let mut shutdown_requested = false;
|
||||
for ev in events.iter().take(n as usize) {
|
||||
if ev.u64 == SHUTDOWN_EPOLL_TOKEN {
|
||||
shutdown_requested = true;
|
||||
continue;
|
||||
}
|
||||
let fd = ev.u64 as RawFd;
|
||||
// Consume the registration: remove + DEL under the waiters
|
||||
// lock (the ADD/DEL serialization — see module docs). A
|
||||
// vanished entry means `cancel_waiter` beat us: the wake is
|
||||
// already moot.
|
||||
let entry = {
|
||||
let mut w = match waiters.lock() {
|
||||
Ok(g) => g,
|
||||
Err(e) => {
|
||||
panic!("smarm: io waiters lock poisoned (core corrupt): {e}")
|
||||
}
|
||||
};
|
||||
let entry = w.remove(&fd);
|
||||
if entry.is_some() {
|
||||
unsafe {
|
||||
libc::epoll_ctl(epollfd, libc::EPOLL_CTL_DEL, fd, std::ptr::null_mut());
|
||||
}
|
||||
let mut pushed_any = false;
|
||||
{
|
||||
let mut q = completions.lock().unwrap();
|
||||
for ev in events.iter().take(n as usize) {
|
||||
if ev.u64 == SHUTDOWN_EPOLL_TOKEN {
|
||||
shutdown_requested = true;
|
||||
continue;
|
||||
}
|
||||
entry
|
||||
};
|
||||
if let Some((pid, epoch)) = entry {
|
||||
let Some(inner) = rt.upgrade() else { return };
|
||||
inner.io_fd_waiters.fetch_sub(1, Ordering::AcqRel);
|
||||
inner.unpark_at(pid, epoch);
|
||||
let fd = ev.u64 as RawFd;
|
||||
let evs = ev.events;
|
||||
q.push_back(Completion::FdReady {
|
||||
fd,
|
||||
events: evs,
|
||||
});
|
||||
pushed_any = true;
|
||||
}
|
||||
}
|
||||
|
||||
if pushed_any {
|
||||
wake_scheduler(wake_write);
|
||||
}
|
||||
if shutdown_requested {
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Write one byte to the scheduler's wake pipe. Retries on EINTR; ignores
|
||||
/// EAGAIN (pipe full means there's already an outstanding wake we haven't
|
||||
/// consumed yet, which is sufficient).
|
||||
fn wake_scheduler(wake_write: RawFd) {
|
||||
let buf: [u8; 1] = [0];
|
||||
unsafe {
|
||||
loop {
|
||||
let n = libc::write(wake_write, buf.as_ptr() as *const _, 1);
|
||||
if n < 0 {
|
||||
let e = *libc::__errno_location();
|
||||
if e == libc::EINTR {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Pipe helper
|
||||
// Pipe helpers (unchanged from v0.2)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
fn make_pipe() -> io::Result<(RawFd, RawFd)> {
|
||||
@@ -459,3 +476,46 @@ fn make_pipe() -> io::Result<(RawFd, RawFd)> {
|
||||
}
|
||||
Ok((fds[0], fds[1]))
|
||||
}
|
||||
|
||||
/// Drain pending bytes from the wake pipe. The scheduler calls this after
|
||||
/// a `poll` wakeup so the next idle call sees an empty pipe.
|
||||
pub fn drain_wake_pipe(fd: RawFd) {
|
||||
let mut buf = [0u8; 64];
|
||||
loop {
|
||||
let n = unsafe { libc::read(fd, buf.as_mut_ptr() as *mut _, buf.len()) };
|
||||
if n <= 0 {
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Block on `fd` for up to `timeout`, returning when either there's data
|
||||
/// to read or the timeout elapses. `None` for `timeout` means wait forever.
|
||||
pub fn poll_wake(fd: RawFd, timeout: Option<std::time::Duration>) {
|
||||
let timeout_ms: libc::c_int = match timeout {
|
||||
None => -1,
|
||||
Some(d) => {
|
||||
let ms = d.as_millis();
|
||||
if ms > i32::MAX as u128 {
|
||||
i32::MAX
|
||||
} else {
|
||||
ms as i32
|
||||
}
|
||||
}
|
||||
};
|
||||
let mut pfd = libc::pollfd {
|
||||
fd,
|
||||
events: libc::POLLIN,
|
||||
revents: 0,
|
||||
};
|
||||
loop {
|
||||
let r = unsafe { libc::poll(&mut pfd as *mut _, 1, timeout_ms) };
|
||||
if r < 0 {
|
||||
let e = unsafe { *libc::__errno_location() };
|
||||
if e == libc::EINTR {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
+21
-75
@@ -11,46 +11,22 @@
|
||||
//!
|
||||
//! See `LOOM.md` for the design intent and the deferred-for-later list.
|
||||
|
||||
// Docs are part of the contract: broken/private intra-doc links fail `cargo doc`,
|
||||
// and every doctest is compiled with `deny(warnings)` under `cargo test --doc`.
|
||||
#![deny(rustdoc::broken_intra_doc_links)]
|
||||
#![deny(rustdoc::private_intra_doc_links)]
|
||||
#![deny(rustdoc::redundant_explicit_links)]
|
||||
#![deny(rustdoc::invalid_codeblock_attributes)]
|
||||
#![deny(rustdoc::invalid_rust_codeblocks)]
|
||||
#![doc(test(attr(deny(warnings))))]
|
||||
|
||||
pub mod actor;
|
||||
pub mod causal;
|
||||
pub mod channel;
|
||||
#[cfg(feature = "cluster")]
|
||||
pub mod cluster;
|
||||
pub mod context;
|
||||
pub mod gen_server;
|
||||
pub mod gen_statem;
|
||||
pub mod introspect;
|
||||
pub mod io;
|
||||
pub mod link;
|
||||
pub mod monitor;
|
||||
pub mod mutex;
|
||||
#[cfg(feature = "observer")]
|
||||
pub mod observer;
|
||||
pub(crate) mod park;
|
||||
pub mod pg;
|
||||
pub mod pid;
|
||||
pub mod preempt;
|
||||
pub(crate) mod raw_mutex;
|
||||
pub mod registry;
|
||||
#[doc(hidden)] // pub only so benches/rq_micro.rs can drive the raw structures
|
||||
pub mod run_queue;
|
||||
pub mod runtime;
|
||||
pub mod scheduler;
|
||||
pub(crate) mod signal;
|
||||
pub(crate) mod slot_state;
|
||||
pub mod arch;
|
||||
pub mod stack;
|
||||
pub mod context;
|
||||
pub mod preempt;
|
||||
pub mod pid;
|
||||
pub mod actor;
|
||||
pub mod channel;
|
||||
pub mod scheduler;
|
||||
pub mod supervisor;
|
||||
pub(crate) mod sync_shim;
|
||||
pub mod timer;
|
||||
pub mod io;
|
||||
pub mod mutex;
|
||||
pub mod monitor;
|
||||
pub mod link;
|
||||
pub mod gen_server;
|
||||
pub mod runtime;
|
||||
pub mod trace;
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -64,48 +40,18 @@ static ALLOCATOR: preempt::PreemptingAllocator = preempt::PreemptingAllocator;
|
||||
// Public API re-exports
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
pub use channel::{
|
||||
channel, select, select_timeout, try_select, try_select_timeout, Receiver, RecvError,
|
||||
RecvTimeoutError, Selectable, Sender,
|
||||
};
|
||||
pub use gen_server::{
|
||||
call, cast, shutdown, whereis_server, CallError, CallTimeoutError, CastError, GenServer,
|
||||
GenServerBuilder, GenServerCtx, GenServerName, GenServerRef, NamedGenServerBuilder,
|
||||
ShutdownAction, StopHandle, TimerHandle, Watcher,
|
||||
};
|
||||
pub use gen_statem::{
|
||||
CallError as GenStatemCallError, Cx, GenStatemName, GenStatemRef, Machine, Reply, Resolution,
|
||||
SendError as GenStatemSendError,
|
||||
};
|
||||
pub use introspect::{
|
||||
actor_info, snapshot, tree, tree_from, ActorInfo, ActorState, RuntimeSnapshot, RuntimeTree,
|
||||
StackInfo, TreeNode, SNAPSHOT_FORMAT_VERSION,
|
||||
};
|
||||
pub use channel::{channel, Receiver, RecvError, Sender};
|
||||
pub use gen_server::{CallError, CastError, GenServer, ServerRef};
|
||||
pub use link::{link, trap_exit, unlink, ExitSignal};
|
||||
pub use monitor::{
|
||||
demonitor, mark_watchable, monitor, terminal_reason, Down, DownReason, Monitor, MonitorId,
|
||||
};
|
||||
pub use monitor::{demonitor, monitor, Down, DownReason, Monitor, MonitorId};
|
||||
pub use mutex::{LockTimeout, Mutex, MutexGuard};
|
||||
#[cfg(feature = "observer")]
|
||||
pub use observer::{ObserverReply, ObserverRequest};
|
||||
pub use pg::{
|
||||
dispatch, join, leave, members, members_as, pick, pick_as, Incarnation, Member, NodeId,
|
||||
};
|
||||
pub use pid::{Addressable, Erased, Name, Pid, RawPid};
|
||||
pub use registry::{
|
||||
install, lookup_as, register, resolve_name, send, send_dyn, send_to, unregister, whereis,
|
||||
NameResolution, RegisterError, SendError,
|
||||
};
|
||||
pub use runtime::{init, Config, Runtime, RuntimeHandle};
|
||||
pub use pid::Pid;
|
||||
pub use runtime::{init, Config, Runtime};
|
||||
pub use scheduler::{
|
||||
block_on_io, cancel_timer, request_shutdown, request_stop, run, self_pid, send_after,
|
||||
send_after_named, send_after_named_wall, send_after_wall, sleep, sleep_wall, spawn, spawn_addr,
|
||||
spawn_addr_with, spawn_monitor, spawn_monitor_with, spawn_under, spawn_under_with, spawn_with,
|
||||
try_spawn, try_spawn_under_with, wait_readable, wait_readable_timeout, wait_writable,
|
||||
wait_writable_timeout, yield_now, FdArm, JoinError, JoinHandle, SpawnError, SpawnOpts,
|
||||
block_on_io, request_stop, run, self_pid, sleep, spawn, spawn_under, wait_readable,
|
||||
wait_writable, yield_now, JoinError, JoinHandle,
|
||||
};
|
||||
pub use supervisor::{ChildSpec, OneForOne, Restart, Shutdown, Signal, Strategy};
|
||||
pub use timer::TimerId;
|
||||
pub use supervisor::{ChildSpec, OneForOne, Restart, Signal, Strategy};
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// check!()
|
||||
|
||||
+47
-77
@@ -49,9 +49,10 @@
|
||||
//! [`Down`]: crate::monitor::Down
|
||||
//! [`request_stop`]: crate::scheduler::request_stop
|
||||
|
||||
use crate::channel::{channel, Receiver};
|
||||
use crate::channel::{channel, Receiver, Sender};
|
||||
use crate::monitor::DownReason;
|
||||
use crate::pid::Pid;
|
||||
use crate::runtime::State;
|
||||
use crate::scheduler::{request_stop, self_pid, with_runtime};
|
||||
|
||||
/// A linked peer's death, delivered to a trapping actor's inbox.
|
||||
@@ -77,13 +78,11 @@ pub fn trap_exit() -> Receiver<ExitSignal> {
|
||||
let (tx, rx) = channel::<ExitSignal>();
|
||||
let me = self_pid();
|
||||
with_runtime(|inner| {
|
||||
if let Some(slot) = inner.slot_at(me) {
|
||||
let mut cold = slot.cold.lock();
|
||||
// Own slot: generation is necessarily current (we're running).
|
||||
if let Some(actor) = cold.actor.as_mut() {
|
||||
inner.with_shared(|s| {
|
||||
if let Some(actor) = s.slot_mut(me).and_then(|slot| slot.actor.as_mut()) {
|
||||
actor.trap = Some(tx);
|
||||
}
|
||||
}
|
||||
})
|
||||
});
|
||||
rx
|
||||
}
|
||||
@@ -95,74 +94,51 @@ pub fn trap_exit() -> Receiver<ExitSignal> {
|
||||
/// this delivers an immediate [`DownReason::NoProc`] exit signal to the caller
|
||||
/// (a message if trapping, otherwise a cooperative stop). Linking yourself, or
|
||||
/// re-linking an existing peer, is a no-op.
|
||||
pub fn link<A>(target: Pid<A>) {
|
||||
let target = target.erase();
|
||||
pub fn link(target: Pid) {
|
||||
let me = self_pid();
|
||||
if target == me {
|
||||
return;
|
||||
}
|
||||
|
||||
// Cold locks are leaves: never hold two at once. The link is recorded one
|
||||
// side at a time, TARGET FIRST — that ordering is what makes the race
|
||||
// window sound:
|
||||
//
|
||||
// - Once `me` is in `target.links`, the target's death always reaches us
|
||||
// (its finalize cascade walks that list). So after step 1 succeeds, the
|
||||
// link semantics are already live.
|
||||
// - If the target dies between step 1 and step 2, its cascade removes
|
||||
// `target` from OUR links (a no-op, we haven't added it yet) and
|
||||
// delivers the exit signal — correct, the link was established. Our
|
||||
// subsequent step-2 insert leaves a stale `target` entry in `me.links`;
|
||||
// stale entries are benign by construction (every cascade walk
|
||||
// re-verifies the peer's word; `unlink` removes them like any other).
|
||||
//
|
||||
// The reverse order would be unsound: target dying in the window would
|
||||
// walk its links WITHOUT us — a silently dead link that we believe is live.
|
||||
let registered_on_target = with_runtime(|inner| match inner.slot_at(target) {
|
||||
Some(slot) => {
|
||||
let mut cold = slot.cold.lock();
|
||||
if slot.is_live_for(target) && cold.actor.is_some() {
|
||||
if !cold.links.contains(&me) {
|
||||
cold.links.push(me);
|
||||
// Under the lock: if the target is live, record the link both ways and
|
||||
// return `None`. If it is gone, return the caller's trap sender (if any)
|
||||
// so we can deliver the NoProc signal after releasing the lock.
|
||||
let dead_action: Option<Option<Sender<ExitSignal>>> = with_runtime(|inner| {
|
||||
inner.with_shared(|s| {
|
||||
let target_live = matches!(
|
||||
s.slot(target),
|
||||
Some(slot) if slot.actor.is_some() && !matches!(slot.state, State::Done)
|
||||
);
|
||||
if target_live {
|
||||
if let Some(slot) = s.slot_mut(me) {
|
||||
if !slot.links.contains(&target) {
|
||||
slot.links.push(target);
|
||||
}
|
||||
}
|
||||
true
|
||||
if let Some(slot) = s.slot_mut(target) {
|
||||
if !slot.links.contains(&me) {
|
||||
slot.links.push(me);
|
||||
}
|
||||
}
|
||||
None
|
||||
} else {
|
||||
false
|
||||
// Grab our own trap sender so the NoProc delivery (below)
|
||||
// doesn't need a second lock acquisition.
|
||||
Some(
|
||||
s.slot(me)
|
||||
.and_then(|slot| slot.actor.as_ref())
|
||||
.and_then(|a| a.trap.clone()),
|
||||
)
|
||||
}
|
||||
}
|
||||
None => false,
|
||||
});
|
||||
|
||||
if registered_on_target {
|
||||
with_runtime(|inner| {
|
||||
let slot = match inner.slot_at(me) {
|
||||
Some(s) => s,
|
||||
None => panic!("smarm: link own slot vanished (core corrupt)"),
|
||||
};
|
||||
let mut cold = slot.cold.lock();
|
||||
if !cold.links.contains(&target) {
|
||||
cold.links.push(target);
|
||||
}
|
||||
});
|
||||
return;
|
||||
}
|
||||
|
||||
// Target already gone: deliver NoProc to ourselves — as a message if
|
||||
// trapping, otherwise as a cooperative stop.
|
||||
let my_trap = with_runtime(|inner| {
|
||||
inner.slot_at(me).and_then(|slot| {
|
||||
let cold = slot.cold.lock();
|
||||
cold.actor.as_ref().and_then(|a| a.trap.clone())
|
||||
})
|
||||
});
|
||||
match my_trap {
|
||||
Some(tx) => {
|
||||
let _ = tx.send(ExitSignal {
|
||||
from: target,
|
||||
reason: DownReason::NoProc,
|
||||
});
|
||||
|
||||
match dead_action {
|
||||
None => {} // linked successfully
|
||||
Some(Some(tx)) => {
|
||||
let _ = tx.send(ExitSignal { from: target, reason: DownReason::NoProc });
|
||||
}
|
||||
None => request_stop(me),
|
||||
Some(None) => request_stop(me),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -170,25 +146,19 @@ pub fn link<A>(target: Pid<A>) {
|
||||
///
|
||||
/// After this, neither actor's death propagates to the other. A no-op if the
|
||||
/// two were not linked.
|
||||
pub fn unlink<A>(target: Pid<A>) {
|
||||
let target = target.erase();
|
||||
pub fn unlink(target: Pid) {
|
||||
let me = self_pid();
|
||||
if target == me {
|
||||
return;
|
||||
}
|
||||
with_runtime(|inner| {
|
||||
// One cold lock at a time (leaf rule). Order is immaterial here:
|
||||
// a half-removed link is just a stale entry on one side, and stale
|
||||
// entries are benign (re-verified on every cascade walk).
|
||||
if let Some(slot) = inner.slot_at(me) {
|
||||
let mut cold = slot.cold.lock();
|
||||
cold.links.retain(|p| *p != target);
|
||||
}
|
||||
if let Some(slot) = inner.slot_at(target) {
|
||||
let mut cold = slot.cold.lock();
|
||||
if slot.generation() == target.generation() {
|
||||
cold.links.retain(|p| *p != me);
|
||||
inner.with_shared(|s| {
|
||||
if let Some(slot) = s.slot_mut(me) {
|
||||
slot.links.retain(|p| *p != target);
|
||||
}
|
||||
}
|
||||
if let Some(slot) = s.slot_mut(target) {
|
||||
slot.links.retain(|p| *p != me);
|
||||
}
|
||||
})
|
||||
});
|
||||
}
|
||||
|
||||
+91
-236
@@ -1,94 +1,59 @@
|
||||
//! Find out when another actor dies, without it knowing or caring that you're
|
||||
//! watching.
|
||||
//! Process monitors.
|
||||
//!
|
||||
//! Say one actor manages a pool of workers and needs to know when a worker
|
||||
//! exits, so it can replace it. The worker does not need to know it is being
|
||||
//! watched, and nothing about the worker's own behavior should change because
|
||||
//! someone is watching it. That is what [`monitor`] is for: call
|
||||
//! `monitor(target)` to get a [`Monitor`], and read exactly one [`Down`]
|
||||
//! message off `monitor.rx` whenever `target` terminates, however it
|
||||
//! terminates.
|
||||
//! `monitor(target)` asks the runtime to deliver a single [`Down`] when
|
||||
//! `target` terminates, and hands back a [`Monitor`] — the [`Receiver`] to read
|
||||
//! it from, plus the identity (`id`, `target`) needed to take the registration
|
||||
//! back down with [`demonitor`]. A monitor is:
|
||||
//!
|
||||
//! ```
|
||||
//! use smarm::{monitor, run, spawn, DownReason};
|
||||
//! - **unidirectional** — the watcher learns of the target's death, but the
|
||||
//! target learns nothing of the watcher, and the watcher is unaffected by
|
||||
//! the death beyond the notification (contrast a *link*, which propagates
|
||||
//! failure);
|
||||
//! - **one-shot** — exactly one `Down` is ever sent for a given monitor.
|
||||
//! The returned channel closes afterwards, so a second `recv()` yields
|
||||
//! `Err(RecvError)`.
|
||||
//!
|
||||
//! run(|| {
|
||||
//! let worker = spawn(|| {
|
||||
//! // does some work, then returns
|
||||
//! });
|
||||
//! let pid = worker.pid();
|
||||
//! This generalizes the older single-`supervisor_channel` mechanism: a
|
||||
//! supervisor is just a hard-wired monitor that the parent installs at spawn
|
||||
//! time. Here any actor may monitor any pid, any number of times.
|
||||
//!
|
||||
//! let m = monitor(pid);
|
||||
//! let _ = worker.join();
|
||||
//! ## Reasons
|
||||
//!
|
||||
//! let down = m.rx.recv().expect("monitor channel closed before Down");
|
||||
//! assert_eq!(down.pid, pid);
|
||||
//! assert_eq!(down.reason, DownReason::Exit);
|
||||
//! });
|
||||
//! ```
|
||||
//! [`DownReason`] is deliberately payload-free. A panicking actor's payload
|
||||
//! has a single owner and is delivered to whoever `join()`s the actor (as
|
||||
//! `JoinError`); a monitor only learns *that* it panicked, not the value.
|
||||
//! Monitoring a pid that is already gone (reclaimed, or never alive) yields
|
||||
//! [`DownReason::NoProc`] immediately, mirroring Erlang's `noproc`.
|
||||
//!
|
||||
//! A monitor is one-directional and one-shot:
|
||||
//! ## Demonitoring
|
||||
//!
|
||||
//! - **One-directional**: the watcher learns that the target died, but the
|
||||
//! target is completely unaffected. It never learns it was being watched,
|
||||
//! and its own behavior and lifetime do not change because of the monitor.
|
||||
//! This is the opposite of a [`link`](mod@crate::link), which is bidirectional:
|
||||
//! linking two actors means an abnormal death on either side can bring the
|
||||
//! other down too. Reach for a monitor when you just want to *know*; reach
|
||||
//! for a link when a peer's crash should actually stop you.
|
||||
//! - **One-shot**: you get exactly one [`Down`] per `monitor()` call, then the
|
||||
//! channel closes. Calling `monitor` again on the same target (or a
|
||||
//! different one) gives you an independent registration with its own
|
||||
//! [`Monitor`] and its own one-shot channel; nothing stops you from
|
||||
//! monitoring the same actor many times over; each call is watched and
|
||||
//! fires on its own.
|
||||
//! Each `monitor()` registration is tagged with a process-unique [`MonitorId`].
|
||||
//! [`demonitor`] removes the registration named by a [`Monitor`] from its
|
||||
//! target's slot, returning `Some(id)` if a live registration was found or
|
||||
//! `None` if it had already fired (or the target is gone). Dropping the
|
||||
//! [`Monitor`] afterwards discards any `Down` that the target had *already*
|
||||
//! queued — the equivalent of Erlang's `demonitor(Ref, [flush])`.
|
||||
//!
|
||||
//! ## Why a monitor never hands you the panic value
|
||||
//! ## Races
|
||||
//!
|
||||
//! If the target panicked, [`Down`] tells you *that* it panicked
|
||||
//! ([`DownReason::Panic`]), but not the panic's payload. The payload has a
|
||||
//! single owner: it is handed to whichever caller `join()`s the actor's
|
||||
//! [`JoinHandle`](crate::JoinHandle), as a `JoinError`. A monitor only needs
|
||||
//! to know that something went wrong, not reproduce the exact value that
|
||||
//! caused it, so it gets the reason and nothing else.
|
||||
//!
|
||||
//! Monitoring a target that is already gone (it finished and was cleaned up,
|
||||
//! or the pid never pointed at a real actor) is not an error: you get a
|
||||
//! [`Down`] with [`DownReason::NoProc`] right away, instead of waiting
|
||||
//! forever for something that already happened.
|
||||
//!
|
||||
//! ## Stopping a monitor early
|
||||
//!
|
||||
//! [`demonitor`] cancels a monitor before it fires. If the registration was
|
||||
//! still live, it removes it and returns `Some` of the monitor's id: no
|
||||
//! `Down` will arrive on that channel from here on. If the target had already
|
||||
//! died and its `Down` already sent, there is nothing left to cancel and
|
||||
//! `demonitor` returns `None`; the `Down` you already have (or that is
|
||||
//! already sitting in the channel) is unaffected.
|
||||
//!
|
||||
//! If you want to cancel *and* make sure a `Down` that already arrived is
|
||||
//! discarded without reading it, just drop the [`Monitor`]: dropping it closes
|
||||
//! its receiver, and any queued `Down` is dropped along with it.
|
||||
//!
|
||||
//! ## Correctness notes for implementers
|
||||
//!
|
||||
//! A target that is still alive at the moment `monitor()` registers is
|
||||
//! guaranteed to eventually produce a real `Down`: registration and the
|
||||
//! target's own termination bookkeeping run under the same lock, so there is
|
||||
//! no window in which the target could die without the just-added
|
||||
//! registration seeing it. `demonitor` is similarly race-free against a target
|
||||
//! that has since died and had its slot reused by a new, unrelated actor: it
|
||||
//! is checked against the exact monitored incarnation, so it can never remove
|
||||
//! a different actor's registration by accident, it simply reports `None`.
|
||||
//! Registration (below) and `finalize_actor` (in `runtime`) both run under the
|
||||
//! shared-state mutex, so a target that is still alive when its monitor is
|
||||
//! registered is guaranteed to deliver a real `Down`; there is no window in
|
||||
//! which the death slips between the liveness check and the registration.
|
||||
//! `demonitor` is protected by the generation half of the pid: if the target
|
||||
//! has died and its slot index been recycled, `slot_mut(target)` fails the
|
||||
//! generation check and `demonitor` is a clean no-op — it can never strip a
|
||||
//! *different* actor's monitor that happens to share the slot index.
|
||||
|
||||
use crate::channel::{channel, Receiver, Sender};
|
||||
use crate::pid::Pid;
|
||||
use crate::runtime::State;
|
||||
use crate::scheduler::with_runtime;
|
||||
|
||||
/// Why a monitored actor went down.
|
||||
///
|
||||
/// Carries no payload: see the module docs for why a monitor never receives
|
||||
/// the panic value itself.
|
||||
/// `Copy` because it carries no payload — see the module docs for why the
|
||||
/// panic payload is *not* included here.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum DownReason {
|
||||
/// The target returned normally.
|
||||
@@ -98,11 +63,6 @@ pub enum DownReason {
|
||||
Panic,
|
||||
/// The target was cooperatively cancelled via `request_stop`.
|
||||
Stopped,
|
||||
/// A graceful shutdown was requested via `request_shutdown`. Only ever
|
||||
/// appears in an [`ExitSignal`](crate::link::ExitSignal) delivered to a
|
||||
/// trapping actor — never in a [`Down`]: a target that honours the request
|
||||
/// exits *normally*, one that does not trap is `Stopped`.
|
||||
Shutdown,
|
||||
/// The target was already gone (finished and reclaimed, or never alive)
|
||||
/// at the moment `monitor()` was called.
|
||||
NoProc,
|
||||
@@ -117,22 +77,21 @@ pub struct Down {
|
||||
pub reason: DownReason,
|
||||
}
|
||||
|
||||
/// A unique identifier for one [`monitor`] registration.
|
||||
/// A process-unique identifier for one `monitor()` registration.
|
||||
///
|
||||
/// Opaque and `Copy`. Never reused for the life of the runtime, so if you
|
||||
/// monitor the same target more than once, each call's id is distinct. This
|
||||
/// is what lets [`demonitor`] tear down exactly one of several monitors on
|
||||
/// the same target without disturbing the others.
|
||||
/// Opaque and `Copy`. Allocated from a monotonic counter in shared state, so
|
||||
/// it is never reused for the lifetime of the runtime — distinct `monitor()`
|
||||
/// calls on the same target get distinct ids, which is what lets [`demonitor`]
|
||||
/// tear down exactly one of several monitors on a target.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
|
||||
pub struct MonitorId(pub(crate) u64);
|
||||
|
||||
/// A live monitor: the receiving end of the one-shot [`Down`] channel, plus the
|
||||
/// identity needed to [`demonitor`] it.
|
||||
///
|
||||
/// Read the notification from [`Monitor::rx`]. Not `Clone`, since only one
|
||||
/// side is meant to consume it. Dropping a `Monitor` closes the receiving
|
||||
/// end; if a `Down` had already arrived but was never read, it is discarded
|
||||
/// along with it.
|
||||
/// Read the notification from [`Monitor::rx`]. Not `Clone` (the receiver is a
|
||||
/// single consumer). Dropping it closes the receiving end; if a `Down` was
|
||||
/// already queued it is discarded with the channel.
|
||||
pub struct Monitor {
|
||||
/// This registration's process-unique id.
|
||||
pub id: MonitorId,
|
||||
@@ -145,164 +104,60 @@ pub struct Monitor {
|
||||
/// Monitor `target`. Returns a [`Monitor`] whose `rx` receives exactly one
|
||||
/// [`Down`].
|
||||
///
|
||||
/// To monitor a child you are spawning yourself, prefer
|
||||
/// [`spawn_monitor`](crate::spawn_monitor): `spawn` followed by `monitor` on
|
||||
/// the returned pid can race the child's death and observe `NoProc` instead
|
||||
/// of its real reason.
|
||||
///
|
||||
/// If `target` is still live, the `Down` arrives when it terminates. If
|
||||
/// `target` is already gone, a [`DownReason::NoProc`] `Down` is queued
|
||||
/// immediately so the caller's `rx.recv()` returns without parking.
|
||||
pub fn monitor<A>(target: Pid<A>) -> Monitor {
|
||||
let target = target.erase();
|
||||
pub fn monitor(target: Pid) -> Monitor {
|
||||
let (tx, rx) = channel::<Down>();
|
||||
let id = with_runtime(|inner| inner.alloc_monitor_id());
|
||||
if !register_monitor(target, id, &tx) {
|
||||
let _ = tx.send(Down {
|
||||
pid: target,
|
||||
reason: DownReason::NoProc,
|
||||
});
|
||||
|
||||
// Register under the shared lock. We allocate the id and (if the target is
|
||||
// live) clone the sender into its monitor list, keeping the original `tx`
|
||||
// for the NoProc fallback. `tx.clone()` only touches the channel's own
|
||||
// mutex, never the shared runtime mutex, so it is safe under the lock — but
|
||||
// we must not *send* here, as `Sender::send` can call back in to unpark a
|
||||
// parked receiver and the shared mutex is not reentrant.
|
||||
let (id, registered) = with_runtime(|inner| {
|
||||
inner.with_shared(|s| {
|
||||
let id = s.alloc_monitor_id();
|
||||
let registered = match s.slot_mut(target) {
|
||||
Some(slot) if !matches!(slot.state, State::Done) => {
|
||||
slot.monitors.push((id, tx.clone()));
|
||||
true
|
||||
}
|
||||
_ => false,
|
||||
};
|
||||
(id, registered)
|
||||
})
|
||||
});
|
||||
|
||||
if !registered {
|
||||
let _ = tx.send(Down { pid: target, reason: DownReason::NoProc });
|
||||
}
|
||||
|
||||
Monitor { id, target, rx }
|
||||
}
|
||||
|
||||
/// Register a monitor `id` on `target` that delivers its `Down` to `tx` — the
|
||||
/// primitive under [`monitor`], split out so a caller can fan many monitors
|
||||
/// into ONE channel (process groups: every membership's death lands on the
|
||||
/// reaper's single inbox). Returns `false` if `target` is already gone, in
|
||||
/// which case nothing is registered and the caller decides what to queue
|
||||
/// (`monitor` sends `NoProc`). The caller allocates `id` up front so it can
|
||||
/// record the registration *before* arming it.
|
||||
/// Cancel the monitor `m`. Returns `Some(id)` if a live registration was found
|
||||
/// on the target's slot and removed, or `None` if there was nothing to remove
|
||||
/// — the target already fired its `Down` (the registration is drained on
|
||||
/// finalize), was never alive (`NoProc`), or has been reclaimed.
|
||||
///
|
||||
/// Implementation note: registration happens under the target's cold lock.
|
||||
/// `tx.clone()` takes the channel's own lock, a Channel-class RawMutex, which
|
||||
/// is explicitly permitted under a Leaf (cold) lock by the lock order
|
||||
/// documented in raw_mutex.rs. We must still not *send* under the lock, since
|
||||
/// `Sender::send` can unpark a parked receiver, and there's no reason to nest
|
||||
/// that.
|
||||
pub(crate) fn register_monitor(target: Pid, id: MonitorId, tx: &Sender<Down>) -> bool {
|
||||
with_runtime(|inner| match inner.slot_at(target) {
|
||||
Some(slot) => {
|
||||
let mut cold = slot.cold.lock();
|
||||
if slot.is_live_for(target) {
|
||||
cold.monitors.push((id, tx.clone()));
|
||||
true
|
||||
} else {
|
||||
false
|
||||
}
|
||||
}
|
||||
None => false,
|
||||
})
|
||||
}
|
||||
|
||||
/// Remove registration `id` from `target` — the primitive under
|
||||
/// [`demonitor`], for callers that hold only the id (see
|
||||
/// [`register_monitor`]). `None` if the registration is not there: already
|
||||
/// fired, already removed, or the slot has moved on to a new tenant.
|
||||
///
|
||||
/// The registration is removed under the target's cold lock, but the
|
||||
/// `Sender` is moved *out* and dropped only after the lock is released.
|
||||
/// Dropping the last sender runs `Sender::drop`, which may unpark a parked
|
||||
/// receiver; legal under a cold lock, but pointless to nest.
|
||||
pub(crate) fn unregister_monitor(target: Pid, id: MonitorId) -> Option<MonitorId> {
|
||||
/// This stops any *future* `Down`. To also discard a `Down` the target may have
|
||||
/// *already* queued (the finalize-races-demonitor case), drop `m` afterwards;
|
||||
/// dropping the [`Monitor`] closes its receiver and the queued notice goes with
|
||||
/// it — the analogue of Erlang's `demonitor(Ref, [flush])`.
|
||||
pub fn demonitor(m: &Monitor) -> Option<MonitorId> {
|
||||
// Remove the registration under the lock, but move the `Sender` *out* and
|
||||
// let it drop only after the lock is released: dropping the last sender
|
||||
// runs `Sender::drop`, which may unpark a parked receiver → `with_shared`,
|
||||
// and the shared mutex is not reentrant.
|
||||
let removed: Option<(MonitorId, Sender<Down>)> = with_runtime(|inner| {
|
||||
let slot = inner.slot_at(target)?;
|
||||
let mut cold = slot.cold.lock();
|
||||
if slot.generation() != target.generation() {
|
||||
return None; // slot reused; the Down already fired
|
||||
}
|
||||
let pos = cold.monitors.iter().position(|(mid, _)| *mid == id)?;
|
||||
Some(cold.monitors.remove(pos))
|
||||
inner.with_shared(|s| {
|
||||
let slot = s.slot_mut(m.target)?;
|
||||
let pos = slot.monitors.iter().position(|(mid, _)| *mid == m.id)?;
|
||||
Some(slot.monitors.remove(pos))
|
||||
})
|
||||
});
|
||||
// `removed`'s sender drops here, outside the lock.
|
||||
removed.map(|(id, _sender)| id)
|
||||
}
|
||||
|
||||
/// Flag `target`'s tenancy as watchable: its death will stamp the slot's
|
||||
/// terminal record (see [`terminal_reason`]), exactly as registering a name
|
||||
/// does. The bridge calls this wherever a smarm pid is *encoded across the
|
||||
/// boundary* — a contract reply, an introspection listing — because BEAM can
|
||||
/// only watch pids it holds, and can only hold pids that crossed. Keeping the
|
||||
/// bit rare is what keeps the record alive: anonymous never-exported churn
|
||||
/// (holder threads, egress tasks) stays ineligible and cannot evict a
|
||||
/// watchable tenancy's record from a LIFO-recycled slot.
|
||||
///
|
||||
/// Generation-checked and live-screened: marking a pid whose tenancy already
|
||||
/// ended is a no-op — its record either exists (it was flagged before dying)
|
||||
/// or is honestly unknowable. Same `Runtime::run()` context contract as
|
||||
/// [`monitor`].
|
||||
pub fn mark_watchable<A>(target: Pid<A>) {
|
||||
let target = target.erase();
|
||||
with_runtime(|inner| {
|
||||
if let Some(slot) = inner.slot_at(target) {
|
||||
// Cold lock FIRST: finalize publishes Done and checks the
|
||||
// watchable bit under this same lock, so the mark either lands
|
||||
// before finalize reads it (the death stamps) or observes the
|
||||
// tenancy already dead (no-op). No lost-stamp window between an
|
||||
// unlocked liveness read and the flag set.
|
||||
let mut cold = slot.cold.lock();
|
||||
if slot.is_live_for(target) {
|
||||
cold.watchable = true;
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/// Whether `target` is live *and* its tenancy is watchable. The cluster's
|
||||
/// remote-monitor admission check (RFC 010 c12): a peer may monitor a pid only
|
||||
/// if that pid was exposed or crossed the wire (the D12 set-sites), and a
|
||||
/// live-but-unwatchable pid answers exactly like a dead one — no liveness leak
|
||||
/// beyond what `watchable` already grants. Same context contract as
|
||||
/// [`monitor`].
|
||||
#[cfg(feature = "cluster")]
|
||||
pub(crate) fn is_watchable<A>(target: Pid<A>) -> bool {
|
||||
let target = target.erase();
|
||||
with_runtime(|inner| {
|
||||
inner.slot_at(target).is_some_and(|slot| {
|
||||
let cold = slot.cold.lock();
|
||||
slot.is_live_for(target) && cold.watchable
|
||||
})
|
||||
})
|
||||
}
|
||||
|
||||
/// The terminal [`DownReason`] of the tenancy `target` names, if that tenancy
|
||||
/// ever registered a name and is the *most recent named* death of its slot:
|
||||
/// finalize stamps the slot with `(generation, reason)` for once-registered
|
||||
/// tenancies (anonymous green-thread churn does not stamp — nor evict), and
|
||||
/// the record survives reclaim and the next tenant's install, until the next
|
||||
/// *named* tenant of the slot itself dies. `None` means the pid never lived,
|
||||
/// is still alive, never held a name, or its record was overwritten by a
|
||||
/// later named tenancy's death — callers fall back to `NoProc` semantics.
|
||||
///
|
||||
/// This exists for watch-installers that raced their target's death (bridge
|
||||
/// soak signature 4): a `NoProc` observed at install time can be upgraded to
|
||||
/// the real reason while the record still matches, which is exactly what an
|
||||
/// install that had won the race would have delivered. It does NOT change
|
||||
/// [`monitor`]'s own semantics — monitoring a stale pid still queues `NoProc`,
|
||||
/// the same shape Erlang gives — the upgrade is the caller's deliberate act.
|
||||
/// Same context contract as [`monitor`]: must run inside `Runtime::run()`.
|
||||
pub fn terminal_reason<A>(target: Pid<A>) -> Option<DownReason> {
|
||||
let target = target.erase();
|
||||
with_runtime(|inner| {
|
||||
let slot = inner.slot_at(target)?;
|
||||
let cold = slot.cold.lock();
|
||||
match cold.terminal {
|
||||
Some((generation, reason)) if generation == target.generation() => Some(reason),
|
||||
_ => None,
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
/// Cancel the monitor `m`. Returns `Some(id)` if a live registration was found
|
||||
/// and removed, so no `Down` will arrive on `m.rx` from here on. Returns
|
||||
/// `None` if there was nothing left to remove: the target had already gone
|
||||
/// down and its `Down` was already sent (or is already sitting in the
|
||||
/// channel, unread).
|
||||
///
|
||||
/// This only stops a *future* `Down`. If you also want to discard a `Down`
|
||||
/// that already arrived (or is about to, in a race with this call), drop `m`
|
||||
/// instead of, or in addition to, calling this: dropping the [`Monitor`]
|
||||
/// closes its receiver and any queued notice is discarded with it.
|
||||
pub fn demonitor(m: &Monitor) -> Option<MonitorId> {
|
||||
unregister_monitor(m.target, m.id)
|
||||
}
|
||||
|
||||
+58
-257
@@ -1,89 +1,12 @@
|
||||
//! Shared mutable state across actors, when a channel is overkill.
|
||||
//! Actor-aware mutex with mandatory timeout.
|
||||
//!
|
||||
//! smarm actors normally coordinate by sending messages, and for a piece of
|
||||
//! owned state the right tool is usually a `gen_server`: one actor holds the
|
||||
//! data and everyone else talks to it. Sometimes that is more machinery than
|
||||
//! you need, and plain shared, lockable state is simpler: [`Mutex<T>`] is
|
||||
//! that escape hatch. It behaves like `std::sync::Mutex<T>`, guarding a value
|
||||
//! of type `T` behind a guard that gives you `&mut T` while held, but it is
|
||||
//! built for smarm's actors rather than OS threads.
|
||||
//! `Mutex<T>` parks the calling *green* thread on contention rather than
|
||||
//! blocking the OS thread. Every lock attempt is bounded by a timeout.
|
||||
//!
|
||||
//! The key difference from `std::sync::Mutex` is what happens on contention.
|
||||
//! [`Mutex::lock`] parks the calling actor (a cooperatively scheduled green
|
||||
//! thread) rather than blocking the underlying OS thread, so other actors on
|
||||
//! the same OS thread keep running while it waits. And every lock attempt is
|
||||
//! bounded by a timeout: an actor that hangs on to the lock forever (stuck in
|
||||
//! a bug, or just slow) would otherwise wedge every other actor waiting on
|
||||
//! it, so smarm makes the wait bounded by default instead of leaving it up
|
||||
//! to you to remember.
|
||||
//! Internals use `Arc<std::sync::Mutex<...>>` so the type is genuinely
|
||||
//! `Send + Sync` and can be shared across scheduler threads.
|
||||
//!
|
||||
//! ## A first lock
|
||||
//!
|
||||
//! ```
|
||||
//! use smarm::{run, spawn, Mutex};
|
||||
//!
|
||||
//! run(|| {
|
||||
//! let counter = Mutex::new(0u32);
|
||||
//!
|
||||
//! // Mutex::clone() is cheap and hands out another handle to the SAME
|
||||
//! // underlying value, much like Arc::clone: every clone shares one lock
|
||||
//! // and one value, so mutations through one are visible through all.
|
||||
//! let a = counter.clone();
|
||||
//! let b = counter.clone();
|
||||
//!
|
||||
//! let h1 = spawn(move || {
|
||||
//! let mut guard = a.lock().unwrap();
|
||||
//! *guard += 1;
|
||||
//! });
|
||||
//! let h2 = spawn(move || {
|
||||
//! let mut guard = b.lock().unwrap();
|
||||
//! *guard += 1;
|
||||
//! });
|
||||
//! h1.join().unwrap();
|
||||
//! h2.join().unwrap();
|
||||
//!
|
||||
//! assert_eq!(*counter.lock().unwrap(), 2);
|
||||
//! });
|
||||
//! ```
|
||||
//!
|
||||
//! ## Choosing a timeout
|
||||
//!
|
||||
//! [`Mutex::lock`] waits up to [`DEFAULT_TIMEOUT`] (30 seconds) before giving
|
||||
//! up with [`LockTimeout`]. To use a different bound for one call, use
|
||||
//! [`Mutex::lock_timeout`] instead; to change the default for every future
|
||||
//! `lock()` call on this mutex (including through its clones), use
|
||||
//! [`Mutex::set_default_timeout`]. If you never want to wait at all, use
|
||||
//! [`Mutex::try_lock`], which returns immediately whether or not the lock was
|
||||
//! free.
|
||||
//!
|
||||
//! ## Fairness and panics
|
||||
//!
|
||||
//! Waiters are granted the lock in the order they started waiting (FIFO), so
|
||||
//! no actor can be starved by later arrivals repeatedly cutting in line.
|
||||
//!
|
||||
//! This mutex never poisons. `std::sync::Mutex` marks itself poisoned if a
|
||||
//! thread panics while holding the lock, because a partly mutated value might
|
||||
//! be left behind for the next lock holder to see. smarm's actors already
|
||||
//! rely on `Drop` running during unwinding to release the lock, so if a
|
||||
//! holder panics, [`MutexGuard::drop`] still runs and the next waiter is
|
||||
//! granted the lock normally. It is the same tradeoff `std::sync::Mutex`
|
||||
//! offers you if you choose to ignore poisoning: you may see a value left
|
||||
//! mid-update by the panicking actor, so a panic inside a critical section is
|
||||
//! still a bug worth fixing, just not one that also wedges every future lock
|
||||
//! attempt.
|
||||
//!
|
||||
//! Locking a mutex you already hold (on the same actor) does not queue
|
||||
//! behind yourself: it deadlocks, the same way relocking a non-reentrant
|
||||
//! `std::sync::Mutex` does. Don't call `lock` while already holding a guard
|
||||
//! from the same `Mutex`.
|
||||
//!
|
||||
//! ## Outside the runtime
|
||||
//!
|
||||
//! `Mutex<T>` also works when called from plain code that is not running as
|
||||
//! a smarm actor (for example, in a test's setup code before calling
|
||||
//! [`run`](crate::run)). There, an actor's cooperative park has no meaning,
|
||||
//! so a lock attempt instead blocks the calling OS thread directly until the
|
||||
//! mutex is free; there is no timeout on this path.
|
||||
//! Fairness: FIFO. Poisoning: none. Reentrance: deadlock (caller bug).
|
||||
|
||||
use crate::pid::Pid;
|
||||
use crate::scheduler;
|
||||
@@ -92,14 +15,8 @@ use std::collections::VecDeque;
|
||||
use std::sync::{Arc, Mutex as StdMutex};
|
||||
use std::time::Duration;
|
||||
|
||||
/// How long [`Mutex::lock`] waits for the lock before giving up, unless
|
||||
/// overridden per-mutex with [`Mutex::set_default_timeout`] or per-call with
|
||||
/// [`Mutex::lock_timeout`].
|
||||
pub const DEFAULT_TIMEOUT: Duration = Duration::from_secs(30);
|
||||
|
||||
/// Returned by [`Mutex::lock`] / [`Mutex::lock_timeout`] when the timeout
|
||||
/// elapses before the lock became available. The lock attempt is abandoned;
|
||||
/// nothing was acquired, and the mutex's value is unaffected.
|
||||
#[derive(Debug, PartialEq, Eq, Clone, Copy)]
|
||||
pub struct LockTimeout;
|
||||
|
||||
@@ -116,15 +33,13 @@ impl std::error::Error for LockTimeout {}
|
||||
|
||||
struct Wait {
|
||||
pid: Pid,
|
||||
/// The wait's park-epoch (slot-word wait identity, see slot_state.rs).
|
||||
/// Grants and timeouts wake via `unpark_at(pid, epoch)`; a stale entry
|
||||
/// can neither be granted by mistake nor wake the wrong wait.
|
||||
epoch: u32,
|
||||
seq: u64,
|
||||
}
|
||||
|
||||
struct MutexState {
|
||||
holder: Option<Pid>,
|
||||
waiters: VecDeque<Wait>,
|
||||
next_seq: u64,
|
||||
default_timeout: Duration,
|
||||
}
|
||||
|
||||
@@ -138,6 +53,7 @@ impl MutexCore {
|
||||
state: StdMutex::new(MutexState {
|
||||
holder: None,
|
||||
waiters: VecDeque::new(),
|
||||
next_seq: 0,
|
||||
default_timeout,
|
||||
}),
|
||||
}
|
||||
@@ -145,33 +61,26 @@ impl MutexCore {
|
||||
}
|
||||
|
||||
impl TimerTarget for MutexCore {
|
||||
fn on_timeout(&self, pid: Pid, epoch: u32) {
|
||||
fn on_timeout(&self, pid: Pid, wait_seq: u64) {
|
||||
let unpark = {
|
||||
let mut st = match self.state.lock() {
|
||||
Ok(g) => g,
|
||||
Err(e) => panic!("smarm: mutex state lock poisoned (core corrupt): {e}"),
|
||||
};
|
||||
// Remove from waiters only if still there with matching epoch.
|
||||
let mut st = self.state.lock().unwrap();
|
||||
// Remove from waiters only if still there with matching seq.
|
||||
// If the lock was already granted (holder == Some(pid)), the
|
||||
// timer fired after the grant: treat as no-op; the actor
|
||||
// timer fired after the grant — treat as no-op; the actor
|
||||
// will see `is_holder == true` and return Ok.
|
||||
if st.holder == Some(pid) {
|
||||
return;
|
||||
}
|
||||
match st
|
||||
.waiters
|
||||
.iter()
|
||||
.position(|w| w.pid == pid && w.epoch == epoch)
|
||||
{
|
||||
Some(pos) => {
|
||||
st.waiters.remove(pos);
|
||||
true
|
||||
}
|
||||
None => false,
|
||||
let pos = st.waiters.iter().position(|w| w.pid == pid && w.seq == wait_seq);
|
||||
if pos.is_some() {
|
||||
st.waiters.remove(pos.unwrap());
|
||||
true
|
||||
} else {
|
||||
false
|
||||
}
|
||||
};
|
||||
if unpark {
|
||||
scheduler::unpark_at(pid, epoch);
|
||||
scheduler::unpark(pid);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -187,8 +96,6 @@ pub struct Mutex<T> {
|
||||
}
|
||||
|
||||
impl<T> Mutex<T> {
|
||||
/// Wrap `value` in a new mutex, initially unlocked, with the default
|
||||
/// lock timeout ([`DEFAULT_TIMEOUT`]).
|
||||
pub fn new(value: T) -> Self {
|
||||
Self {
|
||||
core: Arc::new(MutexCore::new(DEFAULT_TIMEOUT)),
|
||||
@@ -196,36 +103,15 @@ impl<T> Mutex<T> {
|
||||
}
|
||||
}
|
||||
|
||||
/// Change how long future [`lock`](Self::lock) calls on this mutex wait
|
||||
/// before giving up. Applies to every clone of this `Mutex` (they share
|
||||
/// one underlying lock), and to `lock` calls already in progress that
|
||||
/// have not yet started waiting. Does not affect [`lock_timeout`](Self::lock_timeout)
|
||||
/// calls, which always use the timeout passed in.
|
||||
pub fn set_default_timeout(&self, timeout: Duration) {
|
||||
match self.core.state.lock() {
|
||||
Ok(mut st) => st.default_timeout = timeout,
|
||||
Err(e) => panic!("smarm: mutex state lock poisoned (core corrupt): {e}"),
|
||||
}
|
||||
self.core.state.lock().unwrap().default_timeout = timeout;
|
||||
}
|
||||
|
||||
/// Acquire the lock, waiting up to this mutex's default timeout
|
||||
/// ([`DEFAULT_TIMEOUT`], or whatever [`set_default_timeout`](Self::set_default_timeout)
|
||||
/// last set) if it is currently held elsewhere. Returns a [`MutexGuard`]
|
||||
/// that releases the lock when dropped, or [`LockTimeout`] if the
|
||||
/// deadline passes first. To use a one-off timeout instead of the
|
||||
/// mutex's default, call [`lock_timeout`](Self::lock_timeout) directly.
|
||||
pub fn lock(&self) -> Result<MutexGuard<'_, T>, LockTimeout> {
|
||||
let timeout = match self.core.state.lock() {
|
||||
Ok(st) => st.default_timeout,
|
||||
Err(e) => panic!("smarm: mutex state lock poisoned (core corrupt): {e}"),
|
||||
};
|
||||
let timeout = self.core.state.lock().unwrap().default_timeout;
|
||||
self.lock_timeout(timeout)
|
||||
}
|
||||
|
||||
/// Acquire the lock, waiting up to `timeout` (ignoring this mutex's
|
||||
/// default) if it is currently held elsewhere. Returns a [`MutexGuard`]
|
||||
/// that releases the lock when dropped, or [`LockTimeout`] if `timeout`
|
||||
/// elapses first with the lock still unavailable.
|
||||
pub fn lock_timeout(&self, timeout: Duration) -> Result<MutexGuard<'_, T>, LockTimeout> {
|
||||
// Outside the runtime (e.g. in tests, after run() returns) there is no
|
||||
// current actor PID. Fall back to a blocking std::sync::Mutex acquire.
|
||||
@@ -235,100 +121,53 @@ impl<T> Mutex<T> {
|
||||
|
||||
// Fast path: nobody holds it.
|
||||
{
|
||||
let mut st = match self.core.state.lock() {
|
||||
Ok(g) => g,
|
||||
Err(e) => panic!("smarm: mutex state lock poisoned (core corrupt): {e}"),
|
||||
};
|
||||
let mut st = self.core.state.lock().unwrap();
|
||||
if st.holder.is_none() {
|
||||
st.holder = Some(me);
|
||||
drop(st);
|
||||
let taken = match self.value.lock() {
|
||||
Ok(mut g) => g.take(),
|
||||
Err(e) => panic!("smarm: mutex value lock poisoned (core corrupt): {e}"),
|
||||
};
|
||||
let value = match taken {
|
||||
Some(v) => v,
|
||||
None => panic!("smarm: Mutex value missing on free fast path (core corrupt)"),
|
||||
};
|
||||
return Ok(MutexGuard {
|
||||
mutex: self,
|
||||
value: Some(value),
|
||||
});
|
||||
let value = self.value.lock().unwrap().take()
|
||||
.expect("Mutex: value missing on free fast path");
|
||||
return Ok(MutexGuard { mutex: self, value: Some(value) });
|
||||
}
|
||||
}
|
||||
|
||||
// Slow path: register as a waiter, set timeout, park.
|
||||
let _np = scheduler::NoPreempt::enter();
|
||||
let epoch = {
|
||||
let mut st = match self.core.state.lock() {
|
||||
Ok(g) => g,
|
||||
Err(e) => panic!("smarm: mutex state lock poisoned (core corrupt): {e}"),
|
||||
};
|
||||
// begin_wait is lock-free (legal under the state lock); this
|
||||
// makes the epoch atomic with the registration's visibility to
|
||||
// grants and timeouts.
|
||||
let epoch = scheduler::begin_wait();
|
||||
st.waiters.push_back(Wait { pid: me, epoch });
|
||||
epoch
|
||||
let seq = {
|
||||
let mut st = self.core.state.lock().unwrap();
|
||||
let seq = st.next_seq;
|
||||
st.next_seq = st.next_seq.wrapping_add(1);
|
||||
st.waiters.push_back(Wait { pid: me, seq });
|
||||
seq
|
||||
};
|
||||
|
||||
let target: Arc<dyn TimerTarget> = self.core.clone();
|
||||
let deadline = timer::deadline_from_now(timeout);
|
||||
scheduler::insert_wait_timer(deadline, me, target, epoch);
|
||||
scheduler::insert_wait_timer(deadline, me, target, seq);
|
||||
scheduler::park_current();
|
||||
|
||||
// Resumed, precisely: only our grant or our timer can wake this
|
||||
// wait (both epoch-stamped; a stop wake unwinds out of
|
||||
// park_current). The one-shot interpretation below is therefore
|
||||
// exhaustive. Are we the holder?
|
||||
let is_holder = match self.core.state.lock() {
|
||||
Ok(st) => st.holder == Some(me),
|
||||
Err(e) => panic!("smarm: mutex state lock poisoned (core corrupt): {e}"),
|
||||
};
|
||||
// Resumed. Are we the holder?
|
||||
let is_holder = self.core.state.lock().unwrap().holder == Some(me);
|
||||
if is_holder {
|
||||
let taken = match self.value.lock() {
|
||||
Ok(mut g) => g.take(),
|
||||
Err(e) => panic!("smarm: mutex value lock poisoned (core corrupt): {e}"),
|
||||
};
|
||||
let value = match taken {
|
||||
Some(v) => v,
|
||||
None => panic!("smarm: Mutex value missing after grant (core corrupt)"),
|
||||
};
|
||||
Ok(MutexGuard {
|
||||
mutex: self,
|
||||
value: Some(value),
|
||||
})
|
||||
let value = self.value.lock().unwrap().take()
|
||||
.expect("Mutex: value missing after grant");
|
||||
Ok(MutexGuard { mutex: self, value: Some(value) })
|
||||
} else {
|
||||
Err(LockTimeout)
|
||||
}
|
||||
}
|
||||
|
||||
/// Acquire the lock only if it is immediately available: never parks and
|
||||
/// never waits. Returns `Some` with a [`MutexGuard`] if the lock was
|
||||
/// free, `None` if it is currently held elsewhere.
|
||||
pub fn try_lock(&self) -> Option<MutexGuard<'_, T>> {
|
||||
let me = crate::actor::current_pid()?;
|
||||
let mut st = match self.core.state.lock() {
|
||||
Ok(g) => g,
|
||||
Err(e) => panic!("smarm: mutex state lock poisoned (core corrupt): {e}"),
|
||||
};
|
||||
let mut st = self.core.state.lock().unwrap();
|
||||
if st.holder.is_some() {
|
||||
return None;
|
||||
}
|
||||
st.holder = Some(me);
|
||||
drop(st);
|
||||
let taken = match self.value.lock() {
|
||||
Ok(mut g) => g.take(),
|
||||
Err(e) => panic!("smarm: mutex value lock poisoned (core corrupt): {e}"),
|
||||
};
|
||||
let value = match taken {
|
||||
Some(v) => v,
|
||||
None => panic!("smarm: Mutex value missing on try_lock free path (core corrupt)"),
|
||||
};
|
||||
Some(MutexGuard {
|
||||
mutex: self,
|
||||
value: Some(value),
|
||||
})
|
||||
let value = self.value.lock().unwrap().take()
|
||||
.expect("Mutex: value missing on try_lock free path");
|
||||
Some(MutexGuard { mutex: self, value: Some(value) })
|
||||
}
|
||||
|
||||
/// Blocking fallback used when called outside the smarm runtime.
|
||||
@@ -338,32 +177,17 @@ impl<T> Mutex<T> {
|
||||
// tracking and just grab the value mutex directly. This is safe because
|
||||
// outside the runtime there are no green threads competing.
|
||||
let value = loop {
|
||||
let v = match self.value.lock() {
|
||||
Ok(mut g) => g.take(),
|
||||
Err(e) => panic!("smarm: mutex value lock poisoned (core corrupt): {e}"),
|
||||
};
|
||||
if let Some(v) = v {
|
||||
break v;
|
||||
}
|
||||
let v = self.value.lock().unwrap().take();
|
||||
if let Some(v) = v { break v; }
|
||||
std::thread::yield_now();
|
||||
};
|
||||
Ok(MutexGuard {
|
||||
mutex: self,
|
||||
value: Some(value),
|
||||
})
|
||||
Ok(MutexGuard { mutex: self, value: Some(value) })
|
||||
}
|
||||
}
|
||||
|
||||
impl<T> Clone for Mutex<T> {
|
||||
/// Cheap: hands back another handle to the same underlying lock and
|
||||
/// value, the way `Arc::clone` does. All clones of a `Mutex` share one
|
||||
/// lock and one protected value; locking through any clone excludes
|
||||
/// every other clone.
|
||||
fn clone(&self) -> Self {
|
||||
Self {
|
||||
core: self.core.clone(),
|
||||
value: self.value.clone(),
|
||||
}
|
||||
Self { core: self.core.clone(), value: self.value.clone() }
|
||||
}
|
||||
}
|
||||
|
||||
@@ -375,10 +199,6 @@ unsafe impl<T: Send> Sync for Mutex<T> {}
|
||||
// Guard
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Grants access to the value inside a [`Mutex`] while the lock is held.
|
||||
/// Dereferences to `&T` and `&mut T`. Dropping the guard releases the lock
|
||||
/// and, if another actor is waiting, wakes the next one in arrival order.
|
||||
/// Returned by [`Mutex::lock`], [`Mutex::lock_timeout`], and [`Mutex::try_lock`].
|
||||
pub struct MutexGuard<'a, T> {
|
||||
mutex: &'a Mutex<T>,
|
||||
value: Option<T>,
|
||||
@@ -386,53 +206,34 @@ pub struct MutexGuard<'a, T> {
|
||||
|
||||
impl<T> std::ops::Deref for MutexGuard<'_, T> {
|
||||
type Target = T;
|
||||
fn deref(&self) -> &T {
|
||||
match self.value.as_ref() {
|
||||
Some(v) => v,
|
||||
None => panic!("smarm: MutexGuard value missing (core corrupt)"),
|
||||
}
|
||||
}
|
||||
fn deref(&self) -> &T { self.value.as_ref().expect("MutexGuard: value missing") }
|
||||
}
|
||||
|
||||
impl<T> std::ops::DerefMut for MutexGuard<'_, T> {
|
||||
fn deref_mut(&mut self) -> &mut T {
|
||||
match self.value.as_mut() {
|
||||
Some(v) => v,
|
||||
None => panic!("smarm: MutexGuard value missing (core corrupt)"),
|
||||
}
|
||||
self.value.as_mut().expect("MutexGuard: value missing")
|
||||
}
|
||||
}
|
||||
|
||||
impl<T: std::fmt::Debug> std::fmt::Debug for MutexGuard<'_, T> {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
let value = match self.value.as_ref() {
|
||||
Some(v) => v,
|
||||
None => panic!("smarm: MutexGuard value missing (core corrupt)"),
|
||||
};
|
||||
f.debug_tuple("MutexGuard").field(value).finish()
|
||||
f.debug_tuple("MutexGuard")
|
||||
.field(self.value.as_ref().expect("MutexGuard: value missing"))
|
||||
.finish()
|
||||
}
|
||||
}
|
||||
|
||||
impl<T> Drop for MutexGuard<'_, T> {
|
||||
fn drop(&mut self) {
|
||||
let v = match self.value.take() {
|
||||
Some(v) => v,
|
||||
None => panic!("smarm: MutexGuard double drop (core corrupt)"),
|
||||
};
|
||||
match self.mutex.value.lock() {
|
||||
Ok(mut g) => *g = Some(v),
|
||||
Err(e) => panic!("smarm: mutex value lock poisoned (core corrupt): {e}"),
|
||||
}
|
||||
let v = self.value.take().expect("MutexGuard: double drop");
|
||||
*self.mutex.value.lock().unwrap() = Some(v);
|
||||
|
||||
let next = {
|
||||
let mut st = match self.mutex.core.state.lock() {
|
||||
Ok(g) => g,
|
||||
Err(e) => panic!("smarm: mutex state lock poisoned (core corrupt): {e}"),
|
||||
};
|
||||
let next_pid = {
|
||||
let mut st = self.mutex.core.state.lock().unwrap();
|
||||
match st.waiters.pop_front() {
|
||||
Some(w) => {
|
||||
st.holder = Some(w.pid);
|
||||
Some((w.pid, w.epoch))
|
||||
Some(w.pid)
|
||||
}
|
||||
None => {
|
||||
st.holder = None;
|
||||
@@ -440,8 +241,8 @@ impl<T> Drop for MutexGuard<'_, T> {
|
||||
}
|
||||
}
|
||||
};
|
||||
if let Some((pid, epoch)) = next {
|
||||
scheduler::unpark_at(pid, epoch);
|
||||
if let Some(pid) = next_pid {
|
||||
scheduler::unpark(pid);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
-114
@@ -1,114 +0,0 @@
|
||||
//! RFC 016 — runtime observability (Chunk 4: the observer gen_server).
|
||||
//!
|
||||
//! A thin [`GenServer`] that consumes the Chunk-1 read primitive
|
||||
//! ([`snapshot`](crate::snapshot) / [`tree`](crate::tree) /
|
||||
//! [`actor_info`](crate::actor_info)) over a message interface — the live
|
||||
//! `observer` process, in the OTP sense. It is a *transport*, not the
|
||||
//! mechanism: the synchronous internal read stays the primitive, and the
|
||||
//! observer is just one more consumer of it alongside the test suite. This is
|
||||
//! also the read half of the future RFC 003 control plane — the same actor
|
||||
//! gains write verbs there rather than a second consumer being spun up
|
||||
//! (DECISION D12).
|
||||
//!
|
||||
//! ## Why it is feature-gated (DECISION D10)
|
||||
//!
|
||||
//! The read primitive (Chunks 1–3) is always present and unflagged: it is pure
|
||||
//! reads and the test suite leans on it. The *gen_server* sits behind the
|
||||
//! `observer` Cargo feature, off by default, matching RFC 003's dev-only
|
||||
//! feature-flag stance — a release build pays nothing for a live observer it
|
||||
//! never starts.
|
||||
//!
|
||||
//! ## The protocol is the contract (DECISION D11)
|
||||
//!
|
||||
//! [`ObserverRequest`] / [`ObserverReply`] *are* the wire contract. They carry
|
||||
//! no version field of their own because the payloads already do:
|
||||
//! [`RuntimeSnapshot`](crate::RuntimeSnapshot) and
|
||||
//! [`RuntimeTree`](crate::RuntimeTree) each carry
|
||||
//! [`SNAPSHOT_FORMAT_VERSION`](crate::SNAPSHOT_FORMAT_VERSION) (D1). The owned
|
||||
//! snapshot — a potentially large `Vec<ActorInfo>` — travels over the call
|
||||
//! channel by value; that is intended, it is exactly what a remote observer
|
||||
//! (RFC 011) will serialize across a node boundary.
|
||||
|
||||
use crate::gen_server::{GenServer, GenServerBuilder, GenServerRef};
|
||||
use crate::introspect::{actor_info, snapshot, tree};
|
||||
use crate::introspect::{ActorInfo, RuntimeSnapshot, RuntimeTree};
|
||||
use crate::pid::Pid;
|
||||
|
||||
/// A read-only request to the observer. Each verb maps one-to-one onto a
|
||||
/// Chunk-1 read; there are deliberately no mutating verbs here (those are RFC
|
||||
/// 003, D12).
|
||||
#[derive(Debug, Clone)]
|
||||
pub enum ObserverRequest {
|
||||
/// Whole-runtime [`snapshot`].
|
||||
Snapshot,
|
||||
/// Parentage forest, folded from a snapshot ([`tree`]).
|
||||
Tree,
|
||||
/// Coherent view of one actor ([`actor_info`]); `None` reply if the pid is
|
||||
/// stale, forged, or names a vacant slot.
|
||||
ActorInfo(Pid),
|
||||
}
|
||||
|
||||
/// The observer's reply, tagged to match the [`ObserverRequest`] verb. Each
|
||||
/// variant wraps the owned Chunk-1 read result unchanged — the observer adds no
|
||||
/// interpretation, it is pure transport.
|
||||
#[derive(Debug, Clone)]
|
||||
pub enum ObserverReply {
|
||||
Snapshot(RuntimeSnapshot),
|
||||
Tree(RuntimeTree),
|
||||
ActorInfo(Option<ActorInfo>),
|
||||
}
|
||||
|
||||
/// The observer server. Stateless by construction (a ZST): every reply is
|
||||
/// derived freshly from the live runtime on each call, so there is nothing to
|
||||
/// keep between requests.
|
||||
pub struct Observer;
|
||||
|
||||
impl GenServer for Observer {
|
||||
type Call = ObserverRequest;
|
||||
type Reply = ObserverReply;
|
||||
/// No async verbs: the observer is request/reply only. `Infallible` is
|
||||
/// uninhabited, so a `cast` can never be constructed and
|
||||
/// [`handle_cast`](GenServer::handle_cast) is statically unreachable.
|
||||
type Cast = core::convert::Infallible;
|
||||
type Info = ();
|
||||
type Timer = ();
|
||||
|
||||
fn handle_call(&mut self, request: ObserverRequest) -> ObserverReply {
|
||||
match request {
|
||||
ObserverRequest::Snapshot => ObserverReply::Snapshot(snapshot()),
|
||||
ObserverRequest::Tree => ObserverReply::Tree(tree()),
|
||||
ObserverRequest::ActorInfo(pid) => ObserverReply::ActorInfo(actor_info(pid)),
|
||||
}
|
||||
}
|
||||
|
||||
fn handle_cast(&mut self, request: core::convert::Infallible) {
|
||||
// Uninhabited: this match has no arms because `Cast` cannot be
|
||||
// constructed. It documents at the type level that the observer takes
|
||||
// no fire-and-forget traffic.
|
||||
match request {}
|
||||
}
|
||||
}
|
||||
|
||||
/// Spawn the observer under the current actor and hand back its [`GenServerRef`].
|
||||
/// Shorthand for `GenServerBuilder::new(Observer).start()`; use the builder
|
||||
/// directly (e.g. `.under(sup)`) to slot it into a supervision tree.
|
||||
///
|
||||
/// ```
|
||||
/// use smarm::run;
|
||||
/// use smarm::observer::{self, ObserverRequest, ObserverReply};
|
||||
///
|
||||
/// run(|| {
|
||||
/// let obs = observer::start();
|
||||
///
|
||||
/// // Ask for a whole-runtime snapshot over the call channel.
|
||||
/// let ObserverReply::Snapshot(snap) = obs.call(ObserverRequest::Snapshot).unwrap()
|
||||
/// else { panic!("snapshot verb must reply with a snapshot") };
|
||||
///
|
||||
/// // The observer is itself a scheduled actor, so it appears in the very
|
||||
/// // snapshot it produced — transport over the same read every consumer sees.
|
||||
/// assert!(snap.actors.iter().any(|a| a.pid == obs.pid()));
|
||||
/// });
|
||||
/// ```
|
||||
pub fn start() -> GenServerRef<Observer> {
|
||||
GenServerBuilder::new(Observer).start()
|
||||
}
|
||||
-1017
File diff suppressed because it is too large
Load Diff
@@ -1,934 +0,0 @@
|
||||
//! Process groups: one name, many actors.
|
||||
//!
|
||||
//! A process group is a named set of actors that you can look up, fan out to,
|
||||
//! or pick a worker from. It is the natural home for a *worker pool* (several
|
||||
//! interchangeable actors doing the same job), for *service discovery* (find
|
||||
//! everyone currently offering some capability), and for *broadcast* (reach
|
||||
//! every member of a group at once).
|
||||
//!
|
||||
//! The set is *live*: members [`join`] it, and a member that dies is removed
|
||||
//! automatically. You never deregister a dead actor — there is no bookkeeping
|
||||
//! to get wrong, and [`members`] / [`pick`] never hand you an actor that has
|
||||
//! already gone.
|
||||
//!
|
||||
//! ## Joining and reading a group
|
||||
//!
|
||||
//! ```
|
||||
//! use smarm::{channel, join, leave, members, pick, run, spawn};
|
||||
//!
|
||||
//! run(|| {
|
||||
//! let (tx1, rx1) = channel::<()>();
|
||||
//! let (tx2, rx2) = channel::<()>();
|
||||
//! let w1 = spawn(move || { rx1.recv().unwrap(); });
|
||||
//! let w2 = spawn(move || { rx2.recv().unwrap(); });
|
||||
//!
|
||||
//! // Workers join the group; `members` is the live view of it.
|
||||
//! join("pool", w1.pid());
|
||||
//! join("pool", w2.pid());
|
||||
//! assert_eq!(members("pool").len(), 2);
|
||||
//!
|
||||
//! // One worker dies. Nothing tells the group — smarm evicts it
|
||||
//! // automatically, so it is gone from `members` and never picked.
|
||||
//! tx1.send(()).unwrap();
|
||||
//! w1.join().unwrap();
|
||||
//! assert_eq!(members("pool"), vec![w2.pid()]);
|
||||
//! assert_eq!(pick("pool"), Some(w2.pid()));
|
||||
//!
|
||||
//! // Voluntary departure works too.
|
||||
//! leave("pool", w2.pid());
|
||||
//! assert!(pick("pool").is_none());
|
||||
//!
|
||||
//! tx2.send(()).unwrap();
|
||||
//! w2.join().unwrap();
|
||||
//! });
|
||||
//! ```
|
||||
//!
|
||||
//! [`members`] returns every live member in the order they joined; [`pick`]
|
||||
//! returns one of them, or `None` when the group is empty. The same actor can
|
||||
//! belong to any number of groups at once, and joining a group it is already in
|
||||
//! is a harmless no-op.
|
||||
//!
|
||||
//! ## Membership ends on its own
|
||||
//!
|
||||
//! You do not have to clean up after a member that dies. When an actor exits —
|
||||
//! for any reason — it is removed from every group it had joined, before any
|
||||
//! later read or send can observe it. [`leave`] is only for *voluntary*
|
||||
//! departure, when a still-living actor wants out of a group.
|
||||
//!
|
||||
//! This is the main difference from keeping your own `Vec<Pid>`: a plain list
|
||||
//! goes stale the instant a member dies, and you would have to notice and prune
|
||||
//! it yourself. A group prunes itself.
|
||||
//!
|
||||
//! ## Sending to a group
|
||||
//!
|
||||
//! For a worker pool you usually want to hand a job to *one* available member.
|
||||
//! [`dispatch`] picks a live member and sends it a message in a single step,
|
||||
//! returning the member it reached:
|
||||
//!
|
||||
//! ```ignore
|
||||
//! use smarm::{dispatch, join, Addressable};
|
||||
//!
|
||||
//! struct Job(String);
|
||||
//! struct Worker;
|
||||
//! impl Addressable for Worker { type Msg = Job; }
|
||||
//!
|
||||
//! // Each worker has published a `Pid<Worker>` inbox and joined the pool.
|
||||
//! join("workers", worker_a);
|
||||
//! join("workers", worker_b);
|
||||
//!
|
||||
//! // Route one job to whichever live worker `pick` lands on.
|
||||
//! match dispatch::<Worker>("workers", Job("resize image".into())) {
|
||||
//! Ok(who) => println!("sent to {who:?}"),
|
||||
//! Err(returned) => println!("no worker took it: {returned:?}"),
|
||||
//! }
|
||||
//! ```
|
||||
//!
|
||||
//! When you want the pids themselves rather than to send right away, [`pick_as`]
|
||||
//! and [`members_as`] return typed [`Pid<A>`](Pid)s for a homogeneous group, so
|
||||
//! the follow-up send stays compile-checked. The untyped [`pick`] and
|
||||
//! [`members`] are for mixed groups, where all you can rely on is identity.
|
||||
//!
|
||||
//! ## Groups vs. the registry
|
||||
//!
|
||||
//! A group is the many-actors counterpart to the [`registry`](crate::registry).
|
||||
//! The registry binds a name to *at most one* actor and re-resolves it on every
|
||||
//! send — what you want for a single well-known service. A group binds a name to
|
||||
//! *many* actors, and one actor may sit in many groups. Reach for the registry
|
||||
//! when there is exactly one of something; reach for a group when there is a set.
|
||||
//!
|
||||
//! ## Identity and clustering
|
||||
//!
|
||||
//! A group member is described by a [`Member`] — a [`Pid`] plus a [`NodeId`] and
|
||||
//! an [`Incarnation`]. Everything on this page is **local**: you pass and
|
||||
//! receive plain [`Pid`]s, and [`members`] / [`pick`] / [`dispatch`] only ever
|
||||
//! name actors on this node (Erlang's `get_local_members`). With the `cluster`
|
||||
//! feature a group also holds the members other nodes have announced, carried
|
||||
//! under their [`NodeId`]; those never surface here — the cluster-wide reads
|
||||
//! live in [`cluster::pg`](crate::cluster::pg) (`members_all` and friends) and
|
||||
//! return a `Local | Remote` member type, since a [`Pid`] cannot hold a remote.
|
||||
//!
|
||||
//! ## Running context
|
||||
//!
|
||||
//! Every function here addresses the current runtime, so each must be called
|
||||
//! from inside [`run`](crate::run) (that is, on an actor thread). Calling one
|
||||
//! from outside a running runtime panics.
|
||||
|
||||
use crate::channel::{channel, Sender};
|
||||
use crate::monitor::{register_monitor, unregister_monitor, Down, DownReason, MonitorId};
|
||||
use crate::pid::{assert_type, Addressable, Pid};
|
||||
use crate::registry::{send_to, SendError};
|
||||
use crate::scheduler::{spawn_under, with_runtime};
|
||||
use std::collections::HashMap;
|
||||
|
||||
/// A cluster node handle. A `u32` integer handle, *not* an interned atom — the
|
||||
/// single deliberate divergence from the BEAM wire shape.
|
||||
#[derive(Copy, Clone, PartialEq, Eq, Hash, Debug)]
|
||||
pub struct NodeId(u32);
|
||||
|
||||
impl NodeId {
|
||||
#[inline]
|
||||
pub const fn new(v: u32) -> Self {
|
||||
Self(v)
|
||||
}
|
||||
#[inline]
|
||||
pub const fn get(self) -> u32 {
|
||||
self.0
|
||||
}
|
||||
}
|
||||
|
||||
impl From<u32> for NodeId {
|
||||
#[inline]
|
||||
fn from(v: u32) -> Self {
|
||||
Self(v)
|
||||
}
|
||||
}
|
||||
|
||||
/// A node's incarnation epoch — the BEAM `Creation` field adopted verbatim: it
|
||||
/// separates a crashed node from its restart. Fixed for the life of a run
|
||||
/// until clustering supplies a real one.
|
||||
#[derive(Copy, Clone, PartialEq, Eq, Hash, Debug)]
|
||||
pub struct Incarnation(u32);
|
||||
|
||||
impl Incarnation {
|
||||
#[inline]
|
||||
pub const fn new(v: u32) -> Self {
|
||||
Self(v)
|
||||
}
|
||||
#[inline]
|
||||
pub const fn get(self) -> u32 {
|
||||
self.0
|
||||
}
|
||||
}
|
||||
|
||||
impl From<u32> for Incarnation {
|
||||
#[inline]
|
||||
fn from(v: u32) -> Self {
|
||||
Self(v)
|
||||
}
|
||||
}
|
||||
|
||||
/// The fixed single-node identity used until clustering supplies real values.
|
||||
/// Carried like `wake_slot` so the public API never has to change to acquire it.
|
||||
pub const DEFAULT_NODE_ID: NodeId = NodeId(0);
|
||||
/// The fixed incarnation for the single-node default. Non-zero so it never
|
||||
/// collides with a BEAM "any creation" wildcard at interop time.
|
||||
pub const DEFAULT_INCARNATION: Incarnation = Incarnation(1);
|
||||
|
||||
/// A group member's full identity: `(node, incarnation, pid)`.
|
||||
///
|
||||
/// Deliberately a field-for-field image of a modern BEAM pid (`NEW_PID_EXT`).
|
||||
/// In memory it is a plain struct — no wire packing; the packed representation
|
||||
/// belongs to the remote-reference boundary, not here.
|
||||
#[derive(Copy, Clone, PartialEq, Eq, Hash, Debug)]
|
||||
pub struct Member {
|
||||
/// Which node the pid lives on. `DEFAULT_NODE_ID` while single-node.
|
||||
pub node: NodeId,
|
||||
/// The node's incarnation epoch at the time of joining.
|
||||
pub incarnation: Incarnation,
|
||||
/// Pure local slot identity — unchanged; cluster identity is layered
|
||||
/// *around* it here rather than overloading `Pid::generation`.
|
||||
pub pid: Pid,
|
||||
}
|
||||
|
||||
/// One membership: a [`Member`] and the id of the monitor that watches its
|
||||
/// liveness. The monitor's `Down` is delivered to the group reaper's single
|
||||
/// inbox (see [`ProcessGroups::deaths`]), so the membership carries only what
|
||||
/// [`leave`] needs to tear the registration down: the id.
|
||||
pub(crate) struct Membership {
|
||||
pub(crate) member: Member,
|
||||
/// `None` for a remote member (cluster): the origin node is its liveness
|
||||
/// authority; nothing here watches it.
|
||||
pub(crate) monitor: Option<MonitorId>,
|
||||
}
|
||||
|
||||
/// The store: `name → multiset<Member>`. Within a single group a `Member`
|
||||
/// appears at most once (`join` is idempotent); the *multiset* framing is for
|
||||
/// cluster-readiness — the same pid is freely a member of many groups, and the
|
||||
/// width admits multiples in general.
|
||||
///
|
||||
/// Locking discipline. Held under one Leaf-class `RawMutex` on `RuntimeInner`,
|
||||
/// mirroring the registry, and never held together with another Leaf lock (it
|
||||
/// never touches the registry or a slot's cold lock). Monitor registration and
|
||||
/// removal take the target's cold lock (also Leaf), so they run *before* /
|
||||
/// *after* the group lock, never under it — see [`join`] for the ordering that
|
||||
/// makes that safe. Nothing under this lock ever touches a channel.
|
||||
///
|
||||
/// Eviction is *eager*: every membership's monitor delivers to the one
|
||||
/// `deaths` channel, drained by a per-run reaper actor that sweeps the dead
|
||||
/// pid out of every group the moment its `Down` is scheduled. The read path
|
||||
/// keeps a slot-liveness backstop for the window between a death and the
|
||||
/// reaper's turn.
|
||||
pub(crate) struct ProcessGroups {
|
||||
groups: HashMap<String, Vec<Membership>>,
|
||||
/// The reaper's inboxes: every membership monitor is registered against
|
||||
/// a clone of `deaths`. `None` until the first `join` of a run spawns
|
||||
/// the reaper; a stale one (receiver gone with the previous run's
|
||||
/// teardown) is detected via `receiver_alive` and replaced.
|
||||
reaper: Option<ReaperInboxes>,
|
||||
/// `NodeId → node name` for every peer with members in the store, kept
|
||||
/// by the pg actor under this lock, so a stored remote member can be
|
||||
/// rendered back to its wire identity without asking anyone.
|
||||
#[cfg(feature = "cluster")]
|
||||
node_names: HashMap<NodeId, String>,
|
||||
}
|
||||
|
||||
impl ProcessGroups {
|
||||
pub(crate) fn new() -> Self {
|
||||
Self {
|
||||
groups: HashMap::new(),
|
||||
reaper: None,
|
||||
#[cfg(feature = "cluster")]
|
||||
node_names: HashMap::new(),
|
||||
}
|
||||
}
|
||||
|
||||
/// Forget the reaper. Called at the start of every `run()` so a stopped
|
||||
/// reaper from a previous run is never sent to; `join` respawns.
|
||||
pub(crate) fn reset_reaper(&mut self) {
|
||||
self.reaper = None;
|
||||
}
|
||||
|
||||
/// Insert `ms` into `group`. Idempotent on the *member*: `false` means the
|
||||
/// member was already present and nothing changed; `true` means inserted.
|
||||
pub(crate) fn join(&mut self, group: &str, ms: Membership) -> bool {
|
||||
let v = self.groups.entry(group.to_owned()).or_default();
|
||||
if v.iter().any(|e| e.member == ms.member) {
|
||||
return false;
|
||||
}
|
||||
v.push(ms);
|
||||
true
|
||||
}
|
||||
|
||||
/// Remove `member`'s membership from `group`, returning it (so the caller
|
||||
/// can unregister its monitor outside the lock). An emptied group is pruned.
|
||||
pub(crate) fn leave(&mut self, group: &str, member: Member) -> Option<Membership> {
|
||||
let v = self.groups.get_mut(group)?;
|
||||
let pos = v.iter().position(|e| e.member == member)?;
|
||||
let removed = v.remove(pos);
|
||||
if v.is_empty() {
|
||||
self.groups.remove(group);
|
||||
}
|
||||
Some(removed)
|
||||
}
|
||||
|
||||
/// The one dumb eviction primitive: drop every member matching `pred` from
|
||||
/// every group, pruning emptied groups, and return the evicted
|
||||
/// memberships with the group each was in. The primitive does not know
|
||||
/// *why* a member leaves; that is the caller's concern. Its callers are
|
||||
/// the reaper (a local death) and the cluster's node-down / re-sync
|
||||
/// sweeps — all over this same predicate path, which is the whole reason
|
||||
/// to shape eviction as a predicate. Insertion order within a group is
|
||||
/// preserved (`members` / `pick` are order-stable).
|
||||
pub(crate) fn remove_where(
|
||||
&mut self,
|
||||
mut pred: impl FnMut(&Member) -> bool,
|
||||
) -> Vec<(String, Membership)> {
|
||||
let mut evicted = Vec::new();
|
||||
self.groups.retain(|g, v| {
|
||||
let mut i = 0;
|
||||
while i < v.len() {
|
||||
if pred(&v[i].member) {
|
||||
evicted.push((g.clone(), v.remove(i)));
|
||||
} else {
|
||||
i += 1;
|
||||
}
|
||||
}
|
||||
!v.is_empty()
|
||||
});
|
||||
evicted
|
||||
}
|
||||
|
||||
/// Raw enumeration of a group's members — no liveness filtering. Used by
|
||||
/// tests to assert storage state independently of the read-path backstop.
|
||||
#[cfg(test)]
|
||||
fn members_of(&self, group: &str) -> Vec<Member> {
|
||||
self.groups
|
||||
.get(group)
|
||||
.map(|v| v.iter().map(|e| e.member).collect())
|
||||
.unwrap_or_default()
|
||||
}
|
||||
|
||||
/// Live members of `group` **on `node`**, in insertion order. The
|
||||
/// `is_live` oracle is the read-path backstop: a member whose slot is
|
||||
/// already dead is dropped from the *result* even if the reaper has not
|
||||
/// swept it yet. Backstop only — the entry stays in storage; eviction is
|
||||
/// the reaper's job. The node filter is what keeps the local API local:
|
||||
/// a remote member's `pid` is another node's slot bits, meaningless to
|
||||
/// `is_live` and to any local send.
|
||||
fn members_where(
|
||||
&self,
|
||||
group: &str,
|
||||
node: NodeId,
|
||||
mut is_live: impl FnMut(Pid) -> bool,
|
||||
) -> Vec<Pid> {
|
||||
self.groups
|
||||
.get(group)
|
||||
.map(|v| {
|
||||
v.iter()
|
||||
.filter(|e| e.member.node == node)
|
||||
.map(|e| e.member.pid)
|
||||
.filter(|&p| is_live(p))
|
||||
.collect()
|
||||
})
|
||||
.unwrap_or_default()
|
||||
}
|
||||
|
||||
/// The first live member of `group` on `node` in insertion order —
|
||||
/// stateless first-live `pick`, with the same read-path backstop and node
|
||||
/// filter as `members_where`.
|
||||
fn first_member_where(
|
||||
&self,
|
||||
group: &str,
|
||||
node: NodeId,
|
||||
mut is_live: impl FnMut(Pid) -> bool,
|
||||
) -> Option<Pid> {
|
||||
self.groups
|
||||
.get(group)?
|
||||
.iter()
|
||||
.filter(|e| e.member.node == node)
|
||||
.map(|e| e.member.pid)
|
||||
.find(|&p| is_live(p))
|
||||
}
|
||||
}
|
||||
|
||||
/// The store's cluster-side surface: raw reads the pg actor needs to speak
|
||||
/// for this node (`Sync`, membership checks) and the peer-name memo. One
|
||||
/// `cfg` block: everything here exists only when there is a mesh.
|
||||
#[cfg(feature = "cluster")]
|
||||
impl ProcessGroups {
|
||||
/// Does `group` hold `member` right now? (Raw storage, no liveness.)
|
||||
pub(crate) fn contains(&self, group: &str, member: &Member) -> bool {
|
||||
self.groups
|
||||
.get(group)
|
||||
.is_some_and(|v| v.iter().any(|e| e.member == *member))
|
||||
}
|
||||
|
||||
/// Every stored member of `group`, any node, insertion order. Raw storage.
|
||||
pub(crate) fn all_of(&self, group: &str) -> Vec<Member> {
|
||||
self.groups
|
||||
.get(group)
|
||||
.map(|v| v.iter().map(|e| e.member).collect())
|
||||
.unwrap_or_default()
|
||||
}
|
||||
|
||||
/// `(group, [pid])` for every group with a member on `node` — the
|
||||
/// `Sync` payload. Raw storage; groups with no such member are omitted.
|
||||
pub(crate) fn groups_on(&self, node: NodeId) -> Vec<(String, Vec<Pid>)> {
|
||||
let mut out: Vec<(String, Vec<Pid>)> = self
|
||||
.groups
|
||||
.iter()
|
||||
.filter_map(|(g, v)| {
|
||||
let pids: Vec<Pid> = v
|
||||
.iter()
|
||||
.filter(|e| e.member.node == node)
|
||||
.map(|e| e.member.pid)
|
||||
.collect();
|
||||
(!pids.is_empty()).then(|| (g.clone(), pids))
|
||||
})
|
||||
.collect();
|
||||
out.sort_by(|a, b| a.0.cmp(&b.0));
|
||||
out
|
||||
}
|
||||
|
||||
/// Record / forget the name behind a peer's `NodeId`.
|
||||
pub(crate) fn set_node_name(&mut self, node: NodeId, name: String) {
|
||||
self.node_names.insert(node, name);
|
||||
}
|
||||
pub(crate) fn forget_node_name(&mut self, node: NodeId) {
|
||||
self.node_names.remove(&node);
|
||||
}
|
||||
pub(crate) fn node_name(&self, node: NodeId) -> Option<&str> {
|
||||
self.node_names.get(&node).map(String::as_str)
|
||||
}
|
||||
}
|
||||
|
||||
/// The group reaper: one detached actor per run, spawned by the first `join`,
|
||||
/// parked on the shared `deaths` inbox. Every local membership's monitor
|
||||
/// delivers here, so a death is swept out of *every* group it joined as soon
|
||||
/// as the reaper is scheduled — no group operation has to happen first.
|
||||
/// Sweeps by `(node, pid)`: only local members, since a remote member's pid
|
||||
/// bits are meaningless here. Exits when the last sender is gone, i.e. never
|
||||
/// during a run (the store holds one); the run's teardown stops it like any
|
||||
/// other parked actor. Spawned under `ROOT_PID` so its exit signal is absorbed
|
||||
/// rather than delivered to whichever supervisor's child happened to join
|
||||
/// first.
|
||||
///
|
||||
/// Under `cluster` the same actor is the node's **pg actor** (RFC 010 Phase
|
||||
/// 5, c15): it also drains a control inbox of local join/leave announcements,
|
||||
/// the membership stream and the exposed `"pg"` inbox — see
|
||||
/// [`crate::cluster::pg`]. Its store-side sweep is unchanged.
|
||||
#[cfg(not(feature = "cluster"))]
|
||||
fn reaper(rx: crate::channel::Receiver<Down>, ctl: crate::channel::Receiver<PgEvent>) {
|
||||
// No mesh: nothing to tell about joins/leaves. Drop the control inbox
|
||||
// so announcements are refused at the sender rather than queued.
|
||||
drop(ctl);
|
||||
while let Ok(down) = rx.recv() {
|
||||
sweep_local_death(down.pid);
|
||||
// Evicted memberships hold only ids; their monitors have fired.
|
||||
}
|
||||
}
|
||||
|
||||
/// What the local API tells the reaper besides deaths (which arrive as
|
||||
/// [`Down`] on their own inbox — that channel's type is fixed by the monitor
|
||||
/// primitive, so the two cannot be one enum). The default reaper has no use
|
||||
/// for these; the cluster's pg actor broadcasts them (RFC 010 Phase 5).
|
||||
// The default reaper never looks inside — that is the point, not a bug.
|
||||
#[cfg_attr(not(feature = "cluster"), allow(dead_code))]
|
||||
pub(crate) enum PgEvent {
|
||||
/// `join` inserted `pid` into `group`. The consumer re-checks the store
|
||||
/// before acting on it.
|
||||
Joined { group: String, pid: Pid },
|
||||
/// `leave` removed `pid` from `group`.
|
||||
Left { group: String, pid: Pid },
|
||||
/// `cluster::start` has the manager up and the local identity set: take
|
||||
/// a membership subscription, register + expose the `"pg"` name, and
|
||||
/// start speaking to peers.
|
||||
#[cfg(feature = "cluster")]
|
||||
Attach,
|
||||
}
|
||||
|
||||
/// Evict the local member `pid` from every group. The reaper's one store
|
||||
/// operation; returns what was evicted with its group (the cluster's
|
||||
/// `Leave` broadcast wants both).
|
||||
pub(crate) fn sweep_local_death(pid: Pid) -> Vec<(String, Membership)> {
|
||||
with_runtime(|inner| {
|
||||
let node = inner.node_id;
|
||||
inner
|
||||
.process_groups
|
||||
.lock()
|
||||
.remove_where(|m| m.node == node && m.pid == pid)
|
||||
})
|
||||
}
|
||||
|
||||
/// The reaper's inboxes. `deaths` is the liveness authority for the set
|
||||
/// (`ctl` is created and dropped with it, on the same actor).
|
||||
#[derive(Clone)]
|
||||
pub(crate) struct ReaperInboxes {
|
||||
pub(crate) deaths: Sender<Down>,
|
||||
/// The control inbox: local `join`/`leave` announce here (see
|
||||
/// [`PgEvent`]). The default reaper closes it on entry.
|
||||
pub(crate) ctl: Sender<PgEvent>,
|
||||
}
|
||||
|
||||
impl ReaperInboxes {
|
||||
fn alive(&self) -> bool {
|
||||
self.deaths.receiver_alive()
|
||||
}
|
||||
}
|
||||
|
||||
/// Live senders for the reaper's inboxes, spawning the reaper if this run has
|
||||
/// none yet. Two racing first-spawns may both spawn; the loser's senders drop
|
||||
/// on return, its spare reaper sees a closed inbox and exits.
|
||||
pub(crate) fn reaper_inboxes() -> ReaperInboxes {
|
||||
let existing = with_runtime(|inner| {
|
||||
let pg = inner.process_groups.lock();
|
||||
pg.reaper.clone().filter(ReaperInboxes::alive)
|
||||
});
|
||||
if let Some(r) = existing {
|
||||
return r;
|
||||
}
|
||||
let (tx, rx) = channel::<Down>();
|
||||
let (ctl_tx, ctl_rx) = channel::<PgEvent>();
|
||||
// Detached: the handle drops here. The reaper's lifetime is the run's.
|
||||
// The ONE seam between the local store and the cluster: same inboxes,
|
||||
// different body.
|
||||
#[cfg(not(feature = "cluster"))]
|
||||
let _ = spawn_under(crate::runtime::ROOT_PID, move || reaper(rx, ctl_rx));
|
||||
#[cfg(feature = "cluster")]
|
||||
let _ = spawn_under(crate::runtime::ROOT_PID, move || {
|
||||
crate::cluster::pg::actor(rx, ctl_rx)
|
||||
});
|
||||
let fresh = ReaperInboxes {
|
||||
deaths: tx,
|
||||
ctl: ctl_tx,
|
||||
};
|
||||
with_runtime(|inner| {
|
||||
let mut pg = inner.process_groups.lock();
|
||||
match &pg.reaper {
|
||||
Some(r) if r.alive() => r.clone(),
|
||||
_ => {
|
||||
pg.reaper = Some(fresh.clone());
|
||||
fresh
|
||||
}
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
/// A live sender for the reaper's `deaths` inbox (spawning it if needed).
|
||||
fn deaths_sender() -> Sender<Down> {
|
||||
reaper_inboxes().deaths
|
||||
}
|
||||
|
||||
/// Announce a local group change to the reaper, if this run has one. A
|
||||
/// closed inbox is the default reaper (uninterested) or a run tearing down.
|
||||
fn announce(msg: PgEvent) {
|
||||
let ctl = with_runtime(|inner| {
|
||||
inner
|
||||
.process_groups
|
||||
.lock()
|
||||
.reaper
|
||||
.as_ref()
|
||||
.map(|r| r.ctl.clone())
|
||||
});
|
||||
if let Some(ctl) = ctl {
|
||||
let _ = ctl.send(msg);
|
||||
}
|
||||
}
|
||||
|
||||
/// Build the full member identity for `pid` from runtime identity.
|
||||
pub(crate) fn member_for(inner: &crate::runtime::RuntimeInner, pid: Pid) -> Member {
|
||||
Member {
|
||||
node: inner.node_id,
|
||||
incarnation: inner.incarnation,
|
||||
pid,
|
||||
}
|
||||
}
|
||||
|
||||
/// Is `pid` a live actor right now? Generation-checked atomic slot-word read,
|
||||
/// no lock — identical to the registry's guard. The read-path backstop: a
|
||||
/// generation is never reused, so a dead member is detectable independently of
|
||||
/// whether its monitor `Down` has been drained yet.
|
||||
pub(crate) fn live(inner: &crate::runtime::RuntimeInner, pid: Pid) -> bool {
|
||||
inner.slot_at(pid).is_some_and(|s| s.is_live_for(pid))
|
||||
}
|
||||
|
||||
/// Add `pid` to `group`. The same pid may join many groups; within one group a
|
||||
/// pid is a member at most once (idempotent). Returns `true` if this call newly
|
||||
/// added the membership, `false` if it was already a member.
|
||||
///
|
||||
/// Installs a monitor on `pid` so the actor's death evicts it from the group
|
||||
/// automatically — you never have to remove a dead member yourself. Joining a
|
||||
/// pid that is already dead is accepted and evicted the same way (via a
|
||||
/// `NoProc` notice), so it never shows up in a read.
|
||||
///
|
||||
/// Panics if called outside `Runtime::run()`.
|
||||
pub fn join<A>(group: impl Into<String>, pid: Pid<A>) -> bool {
|
||||
let group = group.into();
|
||||
let pid = pid.erase();
|
||||
let deaths = deaths_sender();
|
||||
// Record the membership BEFORE arming its monitor: the reaper sweeps by
|
||||
// pid on the first `Down`, so a `Down` that could precede the entry would
|
||||
// leave a corpse in storage forever (visible to no read — the backstop
|
||||
// hides it — but a leak, and once groups are clustered a member that
|
||||
// would be announced). Arming after insertion means every `Down` finds
|
||||
// its entry. The monitor id is allocated up front so `leave` can tear the
|
||||
// registration down even if it lands in the tiny window before arming (an
|
||||
// orphaned registration is harmless: its `Down` names a pid whose
|
||||
// membership is gone, and the sweep finds nothing).
|
||||
let id = with_runtime(|inner| inner.alloc_monitor_id());
|
||||
let inserted = with_runtime(|inner| {
|
||||
let ms = Membership {
|
||||
member: member_for(inner, pid),
|
||||
monitor: Some(id),
|
||||
};
|
||||
inner.process_groups.lock().join(&group, ms)
|
||||
});
|
||||
if !inserted {
|
||||
return false;
|
||||
}
|
||||
// Tell the reaper (the cluster's pg actor re-checks the store before it
|
||||
// broadcasts, so a `leave`/death that overtakes this announcement is
|
||||
// never advertised as a join).
|
||||
announce(PgEvent::Joined {
|
||||
group: group.clone(),
|
||||
pid,
|
||||
});
|
||||
|
||||
// Outside the group lock: registration takes the target's cold lock (Leaf).
|
||||
// The registration races `finalize_actor` under that cold lock exactly as
|
||||
// every other monitor does, so no death can slip between the join and the
|
||||
// monitor being in place.
|
||||
if !register_monitor(pid, id, &deaths) {
|
||||
// Already gone: queue the notice ourselves, exactly as `monitor` does.
|
||||
let _ = deaths.send(Down {
|
||||
pid,
|
||||
reason: DownReason::NoProc,
|
||||
});
|
||||
}
|
||||
true
|
||||
}
|
||||
|
||||
/// Drop `pid`'s membership of `group`. Returns whether a membership was
|
||||
/// removed. The membership's monitor registration is torn down.
|
||||
///
|
||||
/// Panics if called outside `Runtime::run()`.
|
||||
pub fn leave<A>(group: &str, pid: Pid<A>) -> bool {
|
||||
let pid = pid.erase();
|
||||
let removed = with_runtime(|inner| {
|
||||
let member = member_for(inner, pid);
|
||||
inner.process_groups.lock().leave(group, member)
|
||||
});
|
||||
match removed {
|
||||
Some(ms) => {
|
||||
if let Some(id) = ms.monitor {
|
||||
unregister_monitor(pid, id);
|
||||
}
|
||||
announce(PgEvent::Left {
|
||||
group: group.to_owned(),
|
||||
pid,
|
||||
});
|
||||
true
|
||||
}
|
||||
None => false,
|
||||
}
|
||||
}
|
||||
|
||||
/// Every live member of `group`, in the order they joined. Returns an empty
|
||||
/// vector if the group does not exist or has no live members.
|
||||
///
|
||||
/// Dead members are never returned: the reaper evicts a member as soon as its
|
||||
/// death is processed, and as a backstop a member whose slot is already dead
|
||||
/// is dropped from the result even in the brief window before the reaper's
|
||||
/// turn.
|
||||
///
|
||||
/// Panics if called outside `Runtime::run()`.
|
||||
pub fn members(group: &str) -> Vec<Pid> {
|
||||
with_runtime(|inner| {
|
||||
inner
|
||||
.process_groups
|
||||
.lock()
|
||||
.members_where(group, inner.node_id, |pid| live(inner, pid))
|
||||
})
|
||||
}
|
||||
|
||||
/// One live member of `group`, or `None` if the group is empty (or every
|
||||
/// member has died). Selection is a stateless first-live scan in join order,
|
||||
/// with the same dead-member backstop as [`members`]; smarter, load-aware
|
||||
/// routing is a later, clustered concern.
|
||||
///
|
||||
/// Panics if called outside `Runtime::run()`.
|
||||
pub fn pick(group: &str) -> Option<Pid> {
|
||||
with_runtime(|inner| {
|
||||
inner
|
||||
.process_groups
|
||||
.lock()
|
||||
.first_member_where(group, inner.node_id, |pid| live(inner, pid))
|
||||
})
|
||||
}
|
||||
|
||||
/// Typed [`pick`]: one live member of `group` as a [`Pid<A>`](Pid).
|
||||
/// For a homogeneous pool every member is an `A`, so the picked member comes
|
||||
/// back typed and dispatch is an ordinary compile-checked [`send_to`] rather
|
||||
/// than the [`send_dyn`](crate::send_dyn) escape hatch. Re-types via the
|
||||
/// unchecked `assert_type` primitive — a wrong `A` degrades to
|
||||
/// [`SendError::NoChannel`] on the next send, never a misdelivery.
|
||||
///
|
||||
/// Panics if called outside `Runtime::run()`.
|
||||
pub fn pick_as<A: Addressable>(group: &str) -> Option<Pid<A>> {
|
||||
pick(group).map(assert_type::<A>)
|
||||
}
|
||||
|
||||
/// Typed `members`: every live member of `group` as a [`Pid<A>`], same
|
||||
/// unchecked re-type as [`pick_as`]. Fan-out stays compile-checked end to end.
|
||||
///
|
||||
/// Panics if called outside `Runtime::run()`.
|
||||
pub fn members_as<A: Addressable>(group: &str) -> Vec<Pid<A>> {
|
||||
members(group).into_iter().map(assert_type::<A>).collect()
|
||||
}
|
||||
|
||||
/// Pick a live member of `group` and send it `msg` in one step, returning the
|
||||
/// member it reached on success. The pick-a-live-member-and-send combinator
|
||||
/// over [`pick_as`] + [`send_to`].
|
||||
///
|
||||
/// Errors hand `msg` back undelivered: [`SendError::NoMember`] if the pool is
|
||||
/// empty (or all-dead), otherwise whatever the underlying [`send_to`] returns
|
||||
/// (e.g. the picked member died in the window between pick and send →
|
||||
/// [`SendError::Dead`]).
|
||||
///
|
||||
/// Panics if called outside `Runtime::run()`.
|
||||
pub fn dispatch<A: Addressable>(group: &str, msg: A::Msg) -> Result<Pid<A>, SendError<A::Msg>> {
|
||||
match pick_as::<A>(group) {
|
||||
Some(pid) => send_to::<A>(pid, msg).map(|()| pid),
|
||||
None => Err(SendError::NoMember(msg)),
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::scheduler::spawn;
|
||||
use std::time::{Duration, Instant};
|
||||
|
||||
fn member(index: u32, generation: u32) -> Member {
|
||||
Member {
|
||||
node: DEFAULT_NODE_ID,
|
||||
incarnation: DEFAULT_INCARNATION,
|
||||
pid: Pid::new(index, generation),
|
||||
}
|
||||
}
|
||||
|
||||
/// A synthetic membership: the store never looks at the id.
|
||||
fn synth(index: u32, generation: u32) -> Membership {
|
||||
Membership {
|
||||
member: member(index, generation),
|
||||
monitor: Some(MonitorId(0)),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn join_is_idempotent_within_a_group() {
|
||||
let mut pg = ProcessGroups::new();
|
||||
assert!(pg.join("workers", synth(1, 0)), "first join inserts");
|
||||
assert!(
|
||||
!pg.join("workers", synth(1, 0)),
|
||||
"second identical join is refused"
|
||||
);
|
||||
assert_eq!(pg.members_of("workers"), vec![member(1, 0)]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn same_pid_in_many_groups_is_independent() {
|
||||
let mut pg = ProcessGroups::new();
|
||||
pg.join("a", synth(1, 0));
|
||||
pg.join("b", synth(1, 0));
|
||||
pg.join("b", synth(2, 0));
|
||||
assert_eq!(pg.members_of("a"), vec![member(1, 0)]);
|
||||
assert_eq!(pg.members_of("b"), vec![member(1, 0), member(2, 0)]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn distinct_generations_are_distinct_members() {
|
||||
// ABA guard: same slot index, different generation = different actor.
|
||||
let mut pg = ProcessGroups::new();
|
||||
assert!(pg.join("g", synth(1, 0)));
|
||||
assert!(
|
||||
pg.join("g", synth(1, 1)),
|
||||
"different generation is a distinct member"
|
||||
);
|
||||
assert_eq!(pg.members_of("g"), vec![member(1, 0), member(1, 1)]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn leave_removes_one_membership_and_prunes_empty_groups() {
|
||||
let mut pg = ProcessGroups::new();
|
||||
pg.join("g", synth(1, 0));
|
||||
pg.join("g", synth(2, 0));
|
||||
assert!(pg.leave("g", member(1, 0)).is_some());
|
||||
assert_eq!(pg.members_of("g"), vec![member(2, 0)]);
|
||||
assert!(
|
||||
pg.leave("g", member(1, 0)).is_none(),
|
||||
"second leave finds nothing"
|
||||
);
|
||||
assert!(pg.leave("g", member(2, 0)).is_some());
|
||||
assert!(pg.members_of("g").is_empty(), "group is now empty");
|
||||
assert!(
|
||||
pg.leave("never", member(9, 0)).is_none(),
|
||||
"leaving an unknown group is a no-op"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn remove_where_sweeps_every_group() {
|
||||
let mut pg = ProcessGroups::new();
|
||||
for (g, m) in [
|
||||
("a", synth(1, 0)),
|
||||
("a", synth(2, 0)),
|
||||
("b", synth(1, 0)),
|
||||
("c", synth(3, 0)),
|
||||
] {
|
||||
pg.join(g, m);
|
||||
}
|
||||
// Death of pid index 1 (any generation) evicts it everywhere.
|
||||
let evicted = pg.remove_where(|mem| mem.pid.index() == 1);
|
||||
assert_eq!(evicted.len(), 2, "pid 1 was in a and b");
|
||||
assert_eq!(pg.members_of("a"), vec![member(2, 0)]);
|
||||
assert!(pg.members_of("b").is_empty(), "b held only pid 1; pruned");
|
||||
assert_eq!(pg.members_of("c"), vec![member(3, 0)]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn remove_where_can_match_an_incarnation_sweep() {
|
||||
// Shape check for the node-down / incarnation sweep caller.
|
||||
let mut pg = ProcessGroups::new();
|
||||
let stale = Membership {
|
||||
member: Member {
|
||||
node: DEFAULT_NODE_ID,
|
||||
incarnation: Incarnation::new(7),
|
||||
pid: Pid::new(1, 0),
|
||||
},
|
||||
monitor: Some(MonitorId(0)),
|
||||
};
|
||||
pg.join("g", stale);
|
||||
pg.join("g", synth(2, 0));
|
||||
let evicted = pg.remove_where(|mem| mem.incarnation == Incarnation::new(7));
|
||||
assert_eq!(evicted.len(), 1);
|
||||
assert_eq!(pg.members_of("g"), vec![member(2, 0)]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn read_backstop_hides_a_member_the_reaper_has_not_yet_swept() {
|
||||
let mut pg = ProcessGroups::new();
|
||||
pg.join("g", synth(1, 0));
|
||||
pg.join("g", synth(2, 0));
|
||||
|
||||
// The slot-word oracle already reports pid 1 dead (finalize window),
|
||||
// ahead of the reaper's turn.
|
||||
let dead = Pid::new(1, 0);
|
||||
let oracle = |pid: Pid| pid != dead;
|
||||
|
||||
assert_eq!(
|
||||
pg.members_where("g", DEFAULT_NODE_ID, oracle),
|
||||
vec![Pid::new(2, 0)],
|
||||
"dead pid filtered from read"
|
||||
);
|
||||
assert_eq!(
|
||||
pg.first_member_where("g", DEFAULT_NODE_ID, oracle),
|
||||
Some(Pid::new(2, 0)),
|
||||
"pick skips the dead first member"
|
||||
);
|
||||
|
||||
// Backstop does not evict — that stays the reaper's job; raw storage
|
||||
// still holds both until it runs.
|
||||
assert_eq!(pg.members_of("g"), vec![member(1, 0), member(2, 0)]);
|
||||
}
|
||||
|
||||
// ---- reaper: eager eviction against a live runtime ----
|
||||
|
||||
/// Raw storage view for a group, bypassing the read-path backstop.
|
||||
fn stored(group: &str) -> Vec<Member> {
|
||||
with_runtime(|inner| inner.process_groups.lock().members_of(group))
|
||||
}
|
||||
|
||||
/// Cooperative wait (`smarm::sleep`, never an OS block) until `pred`.
|
||||
fn wait_until(what: &str, mut pred: impl FnMut() -> bool) {
|
||||
let deadline = Instant::now() + Duration::from_secs(2);
|
||||
while !pred() {
|
||||
assert!(Instant::now() < deadline, "timed out waiting for: {what}");
|
||||
crate::sleep(Duration::from_millis(1));
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_death_is_swept_from_storage_without_any_group_operation() {
|
||||
crate::run(|| {
|
||||
let (tx, rx) = channel::<()>();
|
||||
let w = spawn(move || {
|
||||
rx.recv().unwrap();
|
||||
});
|
||||
let pid = w.pid();
|
||||
join("a", pid);
|
||||
join("b", pid);
|
||||
assert_eq!(stored("a"), vec![member_for_test(pid)]);
|
||||
|
||||
tx.send(()).unwrap();
|
||||
w.join().unwrap();
|
||||
// No members()/pick()/join() on a or b from here on: the reaper
|
||||
// alone must clear both.
|
||||
wait_until("reaper sweeps a and b", || {
|
||||
stored("a").is_empty() && stored("b").is_empty()
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_dead_at_join_pid_is_swept_from_storage() {
|
||||
crate::run(|| {
|
||||
let h = spawn(|| {});
|
||||
let pid = h.pid();
|
||||
h.join().unwrap();
|
||||
assert!(join("late", pid), "join is accepted; eviction is uniform");
|
||||
wait_until("reaper sweeps the NoProc member", || {
|
||||
stored("late").is_empty()
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn leave_then_death_does_not_disturb_a_rejoined_group() {
|
||||
// A monitor unregistered by `leave` must not fire later; the pid's
|
||||
// fresh membership after re-join is swept exactly once, by its own
|
||||
// monitor, on death.
|
||||
crate::run(|| {
|
||||
let (tx, rx) = channel::<()>();
|
||||
let w = spawn(move || {
|
||||
rx.recv().unwrap();
|
||||
});
|
||||
let pid = w.pid();
|
||||
join("g", pid);
|
||||
assert!(leave("g", pid));
|
||||
assert!(join("g", pid));
|
||||
assert_eq!(members("g"), vec![pid]);
|
||||
tx.send(()).unwrap();
|
||||
w.join().unwrap();
|
||||
wait_until("reaper sweeps g", || stored("g").is_empty());
|
||||
});
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn reaper_is_respawned_for_a_second_run_of_the_same_runtime() {
|
||||
let rt = crate::runtime::init(crate::runtime::Config::exact(1));
|
||||
let body = || {
|
||||
let h = spawn(|| {});
|
||||
let pid = h.pid();
|
||||
h.join().unwrap();
|
||||
join("g", pid);
|
||||
wait_until("reaper sweeps g", || stored("g").is_empty());
|
||||
};
|
||||
rt.run(body);
|
||||
rt.run(body);
|
||||
}
|
||||
|
||||
fn member_for_test(pid: Pid) -> Member {
|
||||
with_runtime(|inner| member_for(inner, pid))
|
||||
}
|
||||
}
|
||||
+13
-310
@@ -1,335 +1,38 @@
|
||||
//! Process identifiers.
|
||||
//!
|
||||
//! Identity is `(index, generation)`: the index is a slot in the scheduler's
|
||||
//! actor table, the generation increments every time that slot is reused, so a
|
||||
//! stale id (right index, wrong generation) is a *detectable* error rather than
|
||||
//! a silent misdirection — the ABA problem solved without exhausting the id
|
||||
//! space. Those raw numbers live in [`RawPid`].
|
||||
//!
|
||||
//! The public identity is the *typed* [`Pid<A>`] (RFC 014): `RawPid` plus a
|
||||
//! phantom actor type, so a pid is simultaneously an identity and a direct,
|
||||
//! identity-bound address. `Pid<Erased>` — the default — is the untyped pid
|
||||
//! used for identity-only plumbing and for actors with no single message type
|
||||
//! (raw `spawn`, gen_servers). Resolving a name yields the durable, re-resolving
|
||||
//! [`Name`] instead.
|
||||
//! A `Pid` is `(index, generation)`. The index is a slot in the scheduler's
|
||||
//! actor table; the generation increments every time that slot is reused.
|
||||
//! A stale `Pid` (correct index, wrong generation) is a detectable error,
|
||||
//! not a silent misdirection — solves the ABA problem without exhausting
|
||||
//! the PID space.
|
||||
|
||||
use std::marker::PhantomData;
|
||||
|
||||
/// The raw identity numbers, with no actor type. The key for everything that
|
||||
/// only cares about *which* actor: slab indexing, generation checks, and the
|
||||
/// heterogeneous monitor / link / pg tables (which hold actors of every type at
|
||||
/// once, so they cannot be parameterised by one).
|
||||
#[derive(Copy, Clone, PartialEq, Eq, Hash)]
|
||||
pub struct RawPid {
|
||||
pub struct Pid {
|
||||
index: u32,
|
||||
generation: u32,
|
||||
}
|
||||
|
||||
impl RawPid {
|
||||
impl Pid {
|
||||
#[inline]
|
||||
pub const fn new(index: u32, generation: u32) -> Self {
|
||||
Self { index, generation }
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub const fn index(self) -> u32 {
|
||||
self.index
|
||||
}
|
||||
pub const fn index(self) -> u32 { self.index }
|
||||
|
||||
#[inline]
|
||||
pub const fn generation(self) -> u32 {
|
||||
self.generation
|
||||
}
|
||||
pub const fn generation(self) -> u32 { self.generation }
|
||||
}
|
||||
|
||||
impl std::fmt::Debug for RawPid {
|
||||
impl std::fmt::Debug for Pid {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
write!(f, "Pid({}.{})", self.index, self.generation)
|
||||
}
|
||||
}
|
||||
|
||||
impl std::fmt::Display for RawPid {
|
||||
impl std::fmt::Display for Pid {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
write!(f, "<{}.{}>", self.index, self.generation)
|
||||
}
|
||||
}
|
||||
|
||||
/// Phantom actor type for a pid that has no single message type: raw `spawn`
|
||||
/// actors, gen_servers (intrinsically multi-message, addressed via `GenServerRef`),
|
||||
/// and every identity-only context. Deliberately **not** [`Addressable`], so a
|
||||
/// typed `send` to a `Pid<Erased>` does not compile; the runtime-checked
|
||||
/// `send_dyn` escape hatch (RFC 014 §4.6) is the sanctioned bare-pid path.
|
||||
pub enum Erased {}
|
||||
|
||||
/// A process identifier parameterised by the actor's type `A` (default
|
||||
/// [`Erased`]). Wraps the raw `(index, generation)` plus a zero-sized phantom,
|
||||
/// so a `Pid<A>` is both an identity and a direct, identity-bound address: when
|
||||
/// `A: Addressable`, a `send` delivers `A::Msg` to exactly the incarnation this
|
||||
/// pid names — no redirect (contrast the re-resolving [`Name`]).
|
||||
///
|
||||
/// Equality, hashing, and formatting are the raw identity's; the phantom is
|
||||
/// `fn() -> A`, so `Pid<A>` is unconditionally `Copy + Send + Sync` and borrows
|
||||
/// nothing from `A`. The trait impls are hand-written so no `A: Trait` bound
|
||||
/// leaks in from a `#[derive]`.
|
||||
pub struct Pid<A = Erased> {
|
||||
raw: RawPid,
|
||||
_marker: PhantomData<fn() -> A>,
|
||||
}
|
||||
|
||||
impl Pid<Erased> {
|
||||
/// Build an untyped pid from raw numbers. The runtime mints identities
|
||||
/// here; typing happens at typed-actor boundaries via `Pid::from_raw`.
|
||||
#[inline]
|
||||
pub const fn new(index: u32, generation: u32) -> Self {
|
||||
Self {
|
||||
raw: RawPid::new(index, generation),
|
||||
_marker: PhantomData,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl<A> Pid<A> {
|
||||
/// Wrap a raw identity as a typed pid. Crate-internal: minting a typed pid
|
||||
/// from raw numbers asserts an actor's type *unchecked*, which is exactly
|
||||
/// what the typed API exists to avoid outside the runtime's own spawn /
|
||||
/// resolution paths.
|
||||
#[inline]
|
||||
pub(crate) const fn from_raw(raw: RawPid) -> Self {
|
||||
Self {
|
||||
raw,
|
||||
_marker: PhantomData,
|
||||
}
|
||||
}
|
||||
|
||||
/// The raw identity, dropping the actor type — the key for identity-only
|
||||
/// tables and internal plumbing.
|
||||
#[inline]
|
||||
pub const fn raw(self) -> RawPid {
|
||||
self.raw
|
||||
}
|
||||
|
||||
/// Forget the actor type.
|
||||
#[inline]
|
||||
pub const fn erase(self) -> Pid<Erased> {
|
||||
Pid::from_raw(self.raw)
|
||||
}
|
||||
|
||||
/// Slot index in the actor table.
|
||||
#[inline]
|
||||
pub const fn index(self) -> u32 {
|
||||
self.raw.index()
|
||||
}
|
||||
|
||||
/// Reuse generation of the slot (ABA guard).
|
||||
#[inline]
|
||||
pub const fn generation(self) -> u32 {
|
||||
self.raw.generation()
|
||||
}
|
||||
}
|
||||
|
||||
/// Re-type an erased pid as `Pid<A>` *unchecked* — the one shared primitive
|
||||
/// behind `lookup_as` / `pick_as` / `members_as` (RFC 014 §4.4). The registry
|
||||
/// and pg stores are heterogeneous in `A` (they hold actors of every type at
|
||||
/// once), so resolving them yields a bare [`Pid`]; recovering the typed address
|
||||
/// is necessarily an assertion the store cannot make for us.
|
||||
///
|
||||
/// **Not unsound.** Delivery routes on the message's [`TypeId`](std::any::TypeId)
|
||||
/// (every send path keys the channel store by it), so a wrong `A` here does not
|
||||
/// mis-deliver: the next [`send_to`](crate::send_to) finds no channel for
|
||||
/// `A::Msg` on that actor and returns [`SendError::NoChannel`](crate::SendError::NoChannel).
|
||||
/// A mistyped pid degrades to a clean send error, never a silent misroute.
|
||||
#[inline]
|
||||
pub(crate) fn assert_type<A>(pid: Pid) -> Pid<A> {
|
||||
Pid::from_raw(pid.raw())
|
||||
}
|
||||
|
||||
impl<A> Copy for Pid<A> {}
|
||||
impl<A> Clone for Pid<A> {
|
||||
fn clone(&self) -> Self {
|
||||
*self
|
||||
}
|
||||
}
|
||||
impl<A> PartialEq for Pid<A> {
|
||||
fn eq(&self, other: &Self) -> bool {
|
||||
self.raw == other.raw
|
||||
}
|
||||
}
|
||||
impl<A> Eq for Pid<A> {}
|
||||
impl<A> std::hash::Hash for Pid<A> {
|
||||
fn hash<H: std::hash::Hasher>(&self, state: &mut H) {
|
||||
self.raw.hash(state);
|
||||
}
|
||||
}
|
||||
impl<A> std::fmt::Debug for Pid<A> {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
std::fmt::Debug::fmt(&self.raw, f)
|
||||
}
|
||||
}
|
||||
impl<A> std::fmt::Display for Pid<A> {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
std::fmt::Display::fmt(&self.raw, f)
|
||||
}
|
||||
}
|
||||
|
||||
/// An actor type with a single associated message type, so a [`Pid<Self>`] is a
|
||||
/// typed address. The raw channel layer has no such trait (actors are closures
|
||||
/// over channels) and `GenServer` is intrinsically multi-message (addressed via
|
||||
/// its own `GenServerRef`); this is the minimal hook that lets the single-message
|
||||
/// actors carry their message type in their pid. (RFC 014 §4.2.)
|
||||
pub trait Addressable: 'static {
|
||||
/// The message this actor receives. A `Pid<Self>` delivers `Self::Msg`.
|
||||
type Msg: Send + 'static;
|
||||
}
|
||||
|
||||
/// A durable, re-resolving address: a static name plus a phantom message type
|
||||
/// `M` (RFC 014's `Name<M>`). Declared as a constant and shared freely:
|
||||
///
|
||||
/// ```ignore
|
||||
/// const COUNTER: Name<CounterMsg> = Name::new("counter");
|
||||
/// ```
|
||||
///
|
||||
/// Unlike a [`Pid`], a `Name` is resolved through the registry on *every* send,
|
||||
/// so it always reaches whoever currently holds the name.
|
||||
pub struct Name<M> {
|
||||
name: &'static str,
|
||||
_marker: PhantomData<fn() -> M>,
|
||||
}
|
||||
|
||||
impl<M> Name<M> {
|
||||
/// Bind a static string as a typed name. `const`, so names live as
|
||||
/// associated constants at call sites.
|
||||
#[inline]
|
||||
pub const fn new(name: &'static str) -> Self {
|
||||
Self {
|
||||
name,
|
||||
_marker: PhantomData,
|
||||
}
|
||||
}
|
||||
|
||||
/// The underlying registry key.
|
||||
#[inline]
|
||||
pub const fn as_str(self) -> &'static str {
|
||||
self.name
|
||||
}
|
||||
}
|
||||
|
||||
impl<M> Copy for Name<M> {}
|
||||
impl<M> Clone for Name<M> {
|
||||
fn clone(&self) -> Self {
|
||||
*self
|
||||
}
|
||||
}
|
||||
impl<M> PartialEq for Name<M> {
|
||||
fn eq(&self, other: &Self) -> bool {
|
||||
self.name == other.name
|
||||
}
|
||||
}
|
||||
impl<M> Eq for Name<M> {}
|
||||
impl<M> std::hash::Hash for Name<M> {
|
||||
fn hash<H: std::hash::Hasher>(&self, state: &mut H) {
|
||||
self.name.hash(state);
|
||||
}
|
||||
}
|
||||
impl<M> std::fmt::Debug for Name<M> {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
write!(f, "Name<{}>({:?})", std::any::type_name::<M>(), self.name)
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod typed_pid_tests {
|
||||
use super::*;
|
||||
|
||||
// A stand-in actor type with one message type, exercising `Addressable`.
|
||||
struct Counter;
|
||||
struct CounterMsg; // used only as a phantom key; no variants needed
|
||||
impl Addressable for Counter {
|
||||
type Msg = CounterMsg;
|
||||
}
|
||||
|
||||
fn msg_type_name<A: Addressable>() -> &'static str {
|
||||
std::any::type_name::<A::Msg>()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn typed_pid_is_a_copyable_identity() {
|
||||
let p = Pid::<Counter>::from_raw(RawPid::new(3, 1));
|
||||
let q = p; // Copy, not move
|
||||
assert_eq!(p.index(), 3);
|
||||
assert_eq!(p.generation(), 1);
|
||||
assert_eq!(p, q);
|
||||
// Same index, different generation = different incarnation.
|
||||
assert_ne!(p, Pid::<Counter>::from_raw(RawPid::new(3, 2)));
|
||||
assert!(format!("{p:?}").starts_with("Pid("));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn erase_drops_the_type_but_keeps_identity() {
|
||||
let p = Pid::<Counter>::from_raw(RawPid::new(7, 4));
|
||||
assert_eq!(p.erase(), Pid::new(7, 4)); // Pid<Erased>
|
||||
assert_eq!(p.raw(), RawPid::new(7, 4));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn name_is_a_copyable_string_token() {
|
||||
const COUNTER: Name<CounterMsg> = Name::new("counter");
|
||||
let n = COUNTER; // Copy
|
||||
assert_eq!(n.as_str(), "counter");
|
||||
assert_eq!(n, COUNTER);
|
||||
assert_ne!(n, Name::<CounterMsg>::new("other"));
|
||||
assert!(format!("{n:?}").contains("\"counter\""));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn addressable_exposes_the_message_type() {
|
||||
assert!(msg_type_name::<Counter>().ends_with("CounterMsg"));
|
||||
}
|
||||
|
||||
// Identity tokens must be usable across threads.
|
||||
#[test]
|
||||
fn tokens_are_send_sync_without_key_bounds() {
|
||||
fn assert_send_sync<T: Send + Sync>() {}
|
||||
assert_send_sync::<Pid<Counter>>();
|
||||
assert_send_sync::<Pid<Erased>>();
|
||||
assert_send_sync::<Name<CounterMsg>>();
|
||||
}
|
||||
}
|
||||
|
||||
// ---- RFC 010 c10: pids auto-serialize (cluster feature) ---------------------
|
||||
|
||||
/// A local `Pid<A>` serializes as a
|
||||
/// [`RemotePid<A>`](crate::cluster::remote::RemotePid): the wire form stamps
|
||||
/// this node's name and incarnation from the ambient runtime, so a pid can
|
||||
/// sit inside any message field and reply-to needs no ceremony (RFC 010 §3,
|
||||
/// "sugar not a bear trap"). Serializing a pid also marks it **watchable**
|
||||
/// — the wire crossing is the cluster's `mark_watchable` set-site (D12), the
|
||||
/// exact analog of the membrane crossing.
|
||||
///
|
||||
/// Must run inside `run()` (the ambient identity lives on the runtime); a
|
||||
/// runtime without a cluster identity cannot serialize a pid at all — it is
|
||||
/// a serialize error, surfacing as the send's `Encode` failure — rather than
|
||||
/// a `("", 0)` stamp that every peer would silently drop.
|
||||
#[cfg(feature = "cluster")]
|
||||
impl<A: 'static> serde::Serialize for Pid<A> {
|
||||
fn serialize<S: serde::Serializer>(&self, s: S) -> Result<S::Ok, S::Error> {
|
||||
crate::cluster::remote::RemotePid::<A>::from_local(*self)
|
||||
.ok_or_else(|| serde::ser::Error::custom("pid serialized with no local node identity"))?
|
||||
.serialize(s)
|
||||
}
|
||||
}
|
||||
|
||||
/// Deserializing into a `Pid<A>` is the **collapse**: it succeeds only when
|
||||
/// the wire pid names this very node (name and incarnation both), and is a
|
||||
/// decode error otherwise — a foreign pid cannot become a local `Pid`.
|
||||
/// Fields that may hold a pid from anywhere are `RemotePid<A>`.
|
||||
#[cfg(feature = "cluster")]
|
||||
impl<'de, A: 'static> serde::Deserialize<'de> for Pid<A> {
|
||||
fn deserialize<D: serde::Deserializer<'de>>(d: D) -> Result<Self, D::Error> {
|
||||
let rp = crate::cluster::remote::RemotePid::<A>::deserialize(d)?;
|
||||
rp.local().ok_or_else(|| {
|
||||
serde::de::Error::custom(format!(
|
||||
"pid {}@{} is not local to this node",
|
||||
rp.index(),
|
||||
rp.node()
|
||||
))
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
+20
-161
@@ -30,7 +30,17 @@ use std::cell::Cell;
|
||||
use std::sync::atomic::{AtomicBool, Ordering};
|
||||
|
||||
pub const DEFAULT_ALLOC_INTERVAL: u32 = 128;
|
||||
|
||||
// The timeslice is measured in `rdtsc()` ticks, and the tick *unit* differs by
|
||||
// ISA (see `crate::arch::read_cycle_counter`). Both constants target ~100µs.
|
||||
// x86-64 : TSC ≈ CPU base clock; 300_000 ticks ≈ 100µs on a 3 GHz core.
|
||||
// aarch64: CNTVCT runs at CNTFRQ_EL0, commonly ~24 MHz; 2_400 ticks ≈ 100µs.
|
||||
// A real implementation would read the frequency at startup and compute this;
|
||||
// for now the constant is calibrated per-arch. Tune on-device if needed.
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
pub const DEFAULT_TIMESLICE_CYCLES: u64 = 300_000; // ≈ 100µs on a 3 GHz CPU
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
pub const DEFAULT_TIMESLICE_CYCLES: u64 = 2_400; // ≈ 100µs at a 24 MHz CNTVCT
|
||||
|
||||
thread_local! {
|
||||
/// While `false`, the allocator hook is a no-op.
|
||||
@@ -58,16 +68,6 @@ thread_local! {
|
||||
/// resume path free of atomic ref-count traffic; see `check_cancelled` for
|
||||
/// the safety argument.
|
||||
static CURRENT_STOP: Cell<*const AtomicBool> = const { Cell::new(std::ptr::null()) };
|
||||
|
||||
/// Raw pointer to the on-CPU actor's slot, set/cleared by the scheduler on
|
||||
/// the same resume/return boundary as `CURRENT_STOP` (RFC 016 Chunk 2).
|
||||
/// Lets the rare slice-expiry site bump that actor's overrun counter with
|
||||
/// one TLS load and no runtime lookup. Null while no actor is on-CPU. The
|
||||
/// slot lives in the fixed slab and is never reclaimed while the actor is
|
||||
/// running, so the pointer is valid for the whole resume (same lifetime
|
||||
/// argument as `CURRENT_STOP`).
|
||||
static CURRENT_SLOT: Cell<*const crate::runtime::Slot> =
|
||||
const { Cell::new(std::ptr::null()) };
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -87,88 +87,6 @@ pub(crate) fn clear_current_stop() {
|
||||
CURRENT_STOP.with(|c| c.set(std::ptr::null()));
|
||||
}
|
||||
|
||||
/// Bind the on-CPU actor's slot. Called by the scheduler immediately before
|
||||
/// `switch_to_actor`, beside `set_current_stop`.
|
||||
pub(crate) fn set_current_slot(slot: *const crate::runtime::Slot) {
|
||||
CURRENT_SLOT.with(|c| c.set(slot));
|
||||
}
|
||||
|
||||
/// Unbind the slot pointer on the return path, beside `clear_current_stop`.
|
||||
pub(crate) fn clear_current_slot() {
|
||||
CURRENT_SLOT.with(|c| c.set(std::ptr::null()));
|
||||
}
|
||||
|
||||
/// Raw pointer to the on-CPU actor's slot, null on the scheduler's own
|
||||
/// stack. Same lifetime argument as `note_overrun`: the slot is never
|
||||
/// reclaimed while its actor is on-CPU. Consumers: the `smarm-causal`
|
||||
/// profiler (RFC 007) and — unconditionally — the SIGSEGV classifier
|
||||
/// (RFC 019 §7), which additionally relies on this being a plain load of a
|
||||
/// const-initialized TLS Cell (no lazy init, no allocation, no dtor): safe
|
||||
/// from a signal handler.
|
||||
#[inline(never)]
|
||||
pub(crate) fn current_slot_ptr() -> *const crate::runtime::Slot {
|
||||
crate::context::tls_fence();
|
||||
CURRENT_SLOT.with(|c| c.get())
|
||||
}
|
||||
|
||||
/// Swap the preemption gate, returning the previous value. The one accessor
|
||||
/// for `PREEMPTION_ENABLED` from actor context (`NoPreempt`, `RawMutex`,
|
||||
/// `with_runtime`, trace): `#[inline(never)]` + fence, see `context` docs.
|
||||
#[inline(never)]
|
||||
pub(crate) fn preemption_swap(enabled: bool) -> bool {
|
||||
crate::context::tls_fence();
|
||||
PREEMPTION_ENABLED.with(|c| c.replace(enabled))
|
||||
}
|
||||
|
||||
/// Read the preemption gate (debug assertions on the queue paths).
|
||||
#[inline(never)]
|
||||
pub(crate) fn preemption_enabled() -> bool {
|
||||
crate::context::tls_fence();
|
||||
PREEMPTION_ENABLED.with(|c| c.get())
|
||||
}
|
||||
|
||||
/// RFC 007 (`smarm-causal`) — push the slice start forward by `cycles`, so
|
||||
/// virtually-injected delay spun inside `maybe_preempt` does not count against
|
||||
/// the actor's timeslice (the clock-correction half of the RFC: the runtime
|
||||
/// owns this clock, so it can subtract its own perturbation).
|
||||
#[cfg(feature = "smarm-causal")]
|
||||
#[inline(never)]
|
||||
pub(crate) fn extend_timeslice(cycles: u64) {
|
||||
crate::context::tls_fence();
|
||||
TIMESLICE_START.with(|c| c.set(c.get().wrapping_add(cycles)));
|
||||
}
|
||||
|
||||
/// Tally a timeslice overrun against the on-CPU actor (RFC 016 Chunk 2). A
|
||||
/// no-op if no actor is bound (the scheduler's own stack). Reached only from
|
||||
/// the slice-expiry branch, which is already the yield path, so its cost is
|
||||
/// irrelevant.
|
||||
#[inline(never)]
|
||||
fn note_overrun() {
|
||||
crate::context::tls_fence();
|
||||
let p = CURRENT_SLOT.with(|c| c.get());
|
||||
// SAFETY: `p` is null (no actor on-CPU) or a pointer to the on-CPU actor's
|
||||
// slot in the fixed slab. The slot is not reclaimed while the actor runs
|
||||
// (finalize/reclaim happen only after it yields back), so the deref is
|
||||
// valid for the whole resume — the same argument as `check_cancelled`.
|
||||
if !p.is_null() {
|
||||
unsafe { (*p).record_overrun() };
|
||||
}
|
||||
}
|
||||
|
||||
/// Tally one received message against the on-CPU actor (RFC 016 Chunk 2),
|
||||
/// called from the channel receive path on each successful dequeue. A no-op
|
||||
/// outside an actor (null slot). One TLS load + one Relaxed load/store on a
|
||||
/// cache line the receiving thread already owns — no atomic RMW, no lock. Same
|
||||
/// slot-lifetime safety argument as `note_overrun`.
|
||||
#[inline(never)]
|
||||
pub(crate) fn note_message_received() {
|
||||
crate::context::tls_fence();
|
||||
let p = CURRENT_SLOT.with(|c| c.get());
|
||||
if !p.is_null() {
|
||||
unsafe { (*p).record_message() };
|
||||
}
|
||||
}
|
||||
|
||||
/// Observation point for cooperative cancellation. If the on-CPU actor has
|
||||
/// been flagged for stop, raise the sentinel panic so the trampoline's
|
||||
/// `catch_unwind` tears the stack down (running Drop) and reports
|
||||
@@ -178,9 +96,8 @@ pub(crate) fn note_message_received() {
|
||||
/// Called from `maybe_preempt` (amortised, the `check!()`/alloc path) and from
|
||||
/// the wakeup side of every blocking park (`park_current`/`yield_now`), which
|
||||
/// is past the prep-to-park window — so it can never lose a wakeup.
|
||||
#[inline(never)]
|
||||
#[inline]
|
||||
pub fn check_cancelled() {
|
||||
crate::context::tls_fence();
|
||||
let p = CURRENT_STOP.with(|c| c.get());
|
||||
// SAFETY: `p` is either null (no actor on-CPU — the scheduler clears it on
|
||||
// every return) or a pointer into the on-CPU actor's `Arc<AtomicBool>`
|
||||
@@ -209,43 +126,12 @@ pub fn reset_timeslice() {
|
||||
TIMESLICE_START.with(|c| c.set(rdtsc()));
|
||||
}
|
||||
|
||||
/// Cycles elapsed since the current slice started (RFC 016 Chunk 2,
|
||||
/// `budget-accounting`). Read by the scheduler right after an actor yields back,
|
||||
/// on the same thread that armed `TIMESLICE_START`. Approximate for wake-slot
|
||||
/// resumes, which inherit the slice (see `Slot::add_budget`).
|
||||
#[cfg(feature = "budget-accounting")]
|
||||
#[inline]
|
||||
pub(crate) fn elapsed_slice_cycles() -> u64 {
|
||||
rdtsc().saturating_sub(TIMESLICE_START.with(|c| c.get()))
|
||||
}
|
||||
|
||||
/// Read the TSC, unserialised. The core may execute this before earlier
|
||||
/// instructions retire (or after later ones start), so a single stamp can
|
||||
/// land tens of cycles — worst case a stalled load's worth — early or late.
|
||||
/// That is negligible against every consumer in this module: the timeslice
|
||||
/// arm/expiry compare against a ~10^5-cycle slice, and an early stamp only
|
||||
/// makes a slice look *more* used (expires marginally sooner, never later).
|
||||
/// The `lfence` this used to carry was a pipeline drain paid on every
|
||||
/// resume; the one place that needs it is causal-site attribution, which
|
||||
/// opts in via [`rdtsc_serialising`].
|
||||
/// Per-core cycle counter for the timeslice clock. Delegates to the active
|
||||
/// arch backend (`rdtsc` on x86-64, `CNTVCT_EL0` on aarch64). The tick unit
|
||||
/// is ISA-defined, so `DEFAULT_TIMESLICE_CYCLES` is calibrated per-arch below.
|
||||
#[inline(always)]
|
||||
pub fn rdtsc() -> u64 {
|
||||
// SAFETY: x86-64 only (this crate is x86-64 Linux only).
|
||||
unsafe { core::arch::x86_64::_rdtsc() }
|
||||
}
|
||||
|
||||
/// Read the TSC after all prior instructions have completed locally.
|
||||
/// Use where a stamp bounds an interval attributed to *code* — a speculative
|
||||
/// early read would credit the tail of that code to whatever comes next.
|
||||
/// Costs a pipeline drain; keep it off the per-resume path.
|
||||
#[inline(always)]
|
||||
pub fn rdtsc_serialising() -> u64 {
|
||||
unsafe {
|
||||
// SAFETY: x86-64 only. `lfence` serialises the instruction stream so
|
||||
// we don't measure time before prior instructions retire.
|
||||
core::arch::asm!("lfence", options(nostack, nomem, preserves_flags));
|
||||
core::arch::x86_64::_rdtsc()
|
||||
}
|
||||
crate::arch::read_cycle_counter()
|
||||
}
|
||||
|
||||
pub struct PreemptingAllocator;
|
||||
@@ -287,44 +173,19 @@ unsafe impl GlobalAlloc for PreemptingAllocator {
|
||||
/// the actor would then park, and the wakeup would be lost. Library
|
||||
/// code that touches the parking primitives must keep its prep-to-park
|
||||
/// regions allocation-free and check!()-free.
|
||||
///
|
||||
/// `#[inline(never)]`: this touches thread-locals and can switch threads in
|
||||
/// the middle; inlined into a caller's loop the TLS base would be hoisted
|
||||
/// across the switch (see `context` module docs). The call is the price.
|
||||
#[inline(never)]
|
||||
#[inline(always)]
|
||||
pub fn maybe_preempt() {
|
||||
crate::context::tls_fence();
|
||||
ALLOC_COUNT.with(|c| {
|
||||
let n = c.get();
|
||||
if n == 0 {
|
||||
c.set(CONFIGURED_ALLOC_INTERVAL.with(|i| i.get()));
|
||||
// Cooperative cancellation shares the amortised cadence with the
|
||||
// timeslice check. Observe a pending stop first: if we are being
|
||||
// cancelled there is no point yielding, we unwind instead.
|
||||
check_cancelled();
|
||||
if PREEMPTION_ENABLED.with(|e| e.get()) {
|
||||
// Cooperative cancellation shares the amortised cadence with
|
||||
// the timeslice check, and shares its gate: while preemption
|
||||
// is disabled (`NoPreempt`, `with_shared`, channel critical
|
||||
// sections) the stop sentinel must NOT be raised, because an
|
||||
// allocation-triggered unwind inside a region holding a
|
||||
// `std::sync::Mutex` would poison it — one `request_stop` at
|
||||
// the wrong moment would then cascade `lock().unwrap()`
|
||||
// panics through every later user of that lock. Observation
|
||||
// is merely deferred to the next enabled allocation or the
|
||||
// wakeup side of the next park/yield, both of which are
|
||||
// lock-free points by construction.
|
||||
//
|
||||
// Observe a pending stop first: if we are being cancelled
|
||||
// there is no point yielding, we unwind instead.
|
||||
check_cancelled();
|
||||
// RFC 007: causal-profiling sample/absorb point. Shares the
|
||||
// amortised cadence, and the PREEMPTION_ENABLED gate — so it
|
||||
// can never spin inside a prep-to-park or no-preempt region.
|
||||
#[cfg(feature = "smarm-causal")]
|
||||
crate::causal::check();
|
||||
let start = TIMESLICE_START.with(|s| s.get());
|
||||
if rdtsc().saturating_sub(start) > CONFIGURED_TIMESLICE_CYCLES.with(|t| t.get()) {
|
||||
// Tally the overrun (RFC 016 Chunk 2) before handing back —
|
||||
// this is the slice-expiry site RFC 006 wanted, and it's
|
||||
// already the yield path, so the counter is near-free.
|
||||
note_overrun();
|
||||
// SAFETY: reachable only inside an actor (the scheduler
|
||||
// sets PREEMPTION_ENABLED on resume and clears it on
|
||||
// return). The scheduler stack is therefore valid.
|
||||
@@ -342,9 +203,7 @@ pub fn maybe_preempt() {
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Force-expire the timeslice so the next RDTSC check preempts.
|
||||
#[inline(never)]
|
||||
pub fn expire_timeslice_for_test() {
|
||||
crate::context::tls_fence();
|
||||
TIMESLICE_START.with(|c| c.set(0));
|
||||
ALLOC_COUNT.with(|c| c.set(0));
|
||||
}
|
||||
|
||||
@@ -1,371 +0,0 @@
|
||||
//! A minimal futex-based mutex that cannot poison.
|
||||
//!
|
||||
//! `std::sync::Mutex` poisons on unwind, turning one panic into a cascade of
|
||||
//! `lock().unwrap()` panics in every later user. The runtime's internal
|
||||
//! critical sections must never unwind anyway (the stop sentinel is gated
|
||||
//! behind `PREEMPTION_ENABLED`, and the guard below disables preemption), so
|
||||
//! poisoning buys nothing and costs a failure mode. This mutex has no poison
|
||||
//! state by construction.
|
||||
//!
|
||||
//! Two further properties the runtime wants:
|
||||
//!
|
||||
//! - **The guard enters `NoPreempt`.** A timeslice switch while holding an OS
|
||||
//! mutex would suspend the actor with the lock held, stalling every other
|
||||
//! OS thread that touches it until the actor is resumed. Disabling
|
||||
//! preemption for the (short) critical section keeps lock hold times
|
||||
//! bounded. It also closes the unwind hole structurally: with
|
||||
//! `PREEMPTION_ENABLED` false, `maybe_preempt` neither yields nor raises
|
||||
//! the stop sentinel, so no allocation inside the critical section can
|
||||
//! unwind it.
|
||||
//! - **No std machinery.** One `AtomicU32` and two futex syscalls; friendlier
|
||||
//! to an eventual embedded port than `std::sync::Mutex` (swap the futex for
|
||||
//! a spin or WFE backend).
|
||||
//!
|
||||
//! Algorithm: the classic three-state futex mutex (Drepper, "Futexes Are
|
||||
//! Tricky", mutex3). 0 = unlocked, 1 = locked, 2 = locked with (possible)
|
||||
//! waiters. Uncontended lock/unlock is one CAS / one swap, no syscall.
|
||||
//!
|
||||
//! Lock-order position — two classes (see [`LockClass`]):
|
||||
//!
|
||||
//! - **Leaf**: slot cold locks, the free list, the stack pool, the name
|
||||
//! registry. Mutual leaves — never hold two at once.
|
||||
//! - **Channel**: a channel's internal lock. May be acquired *under* a Leaf
|
||||
//! (finalize clones the supervisor/trap senders, and `monitor()` clones the
|
||||
//! Down sender, all under a cold lock — structural, the sender lives in the
|
||||
//! slot), but nothing may be acquired under a Channel lock: channel
|
||||
//! critical sections call only the lock-free unpark protocol.
|
||||
//!
|
||||
//! So the total order is Leaf → Channel, one of each at most. Holding either
|
||||
//! while pushing to the run queue is permitted (unpark from inside a cold or
|
||||
//! channel section); the reverse — taking any `RawMutex` from inside a
|
||||
//! run-queue op — cannot arise (queue ops call nothing).
|
||||
|
||||
use std::cell::UnsafeCell;
|
||||
use std::ops::{Deref, DerefMut};
|
||||
use std::sync::atomic::{AtomicU32, Ordering};
|
||||
|
||||
const UNLOCKED: u32 = 0;
|
||||
const LOCKED: u32 = 1;
|
||||
const CONTENDED: u32 = 2;
|
||||
|
||||
/// How many `pause` spins to burn before falling back to the futex. Critical
|
||||
/// sections under this lock are tens of nanoseconds (push to a Vec, clone a
|
||||
/// sender), so a short spin almost always avoids the syscall.
|
||||
const SPIN_LIMIT: u32 = 64;
|
||||
|
||||
/// Which rung of the two-rung lock order a `RawMutex` occupies. Debug builds
|
||||
/// enforce the order mechanically (see the module docs); release builds carry
|
||||
/// no state and no checks.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub(crate) enum LockClass {
|
||||
/// Runtime cold data: slot cold locks, free list, stack pool, registry.
|
||||
/// Mutual leaves among themselves; a Channel lock may be taken under one.
|
||||
Leaf,
|
||||
/// A channel's internal lock. One at a time, nothing acquired under it;
|
||||
/// may itself be acquired under a Leaf.
|
||||
Channel,
|
||||
}
|
||||
|
||||
// The ordering rules, mechanically enforced (debug builds): a deadlock from a
|
||||
// violated order is a hang waiting for the right interleaving, so fail at the
|
||||
// acquisition that violates it, not in the eventual hang.
|
||||
#[cfg(debug_assertions)]
|
||||
thread_local! {
|
||||
static LEAVES_HELD: std::cell::Cell<u32> = const { std::cell::Cell::new(0) };
|
||||
static CHANNELS_HELD: std::cell::Cell<u32> = const { std::cell::Cell::new(0) };
|
||||
}
|
||||
|
||||
// Debug-only TLS bookkeeping: out of line only when it has a body (release
|
||||
// builds must not pay a call for an empty function).
|
||||
#[cfg_attr(debug_assertions, inline(never))]
|
||||
#[cfg_attr(not(debug_assertions), inline(always))]
|
||||
fn order_check_acquire(class: LockClass) {
|
||||
#[cfg(debug_assertions)]
|
||||
crate::context::tls_fence();
|
||||
#[cfg(debug_assertions)]
|
||||
match class {
|
||||
LockClass::Leaf => LEAVES_HELD.with(|l| {
|
||||
debug_assert_eq!(
|
||||
l.get(),
|
||||
0,
|
||||
"lock order violated: acquiring a Leaf RawMutex while already \
|
||||
holding one (cold locks / free list / stack pool / registry \
|
||||
are mutual leaves)"
|
||||
);
|
||||
CHANNELS_HELD.with(|c| {
|
||||
debug_assert_eq!(
|
||||
c.get(),
|
||||
0,
|
||||
"lock order violated: acquiring a Leaf RawMutex under a \
|
||||
channel lock (order is Leaf -> Channel, never the reverse)"
|
||||
);
|
||||
});
|
||||
l.set(l.get() + 1);
|
||||
}),
|
||||
LockClass::Channel => CHANNELS_HELD.with(|c| {
|
||||
debug_assert_eq!(
|
||||
c.get(),
|
||||
0,
|
||||
"lock order violated: acquiring a channel lock while already \
|
||||
holding one (channel locks are mutual leaves)"
|
||||
);
|
||||
c.set(c.get() + 1);
|
||||
}),
|
||||
}
|
||||
#[cfg(not(debug_assertions))]
|
||||
let _ = class;
|
||||
}
|
||||
|
||||
#[cfg_attr(debug_assertions, inline(never))]
|
||||
#[cfg_attr(not(debug_assertions), inline(always))]
|
||||
fn order_check_release(class: LockClass) {
|
||||
#[cfg(debug_assertions)]
|
||||
crate::context::tls_fence();
|
||||
#[cfg(debug_assertions)]
|
||||
match class {
|
||||
LockClass::Leaf => LEAVES_HELD.with(|c| c.set(c.get() - 1)),
|
||||
LockClass::Channel => CHANNELS_HELD.with(|c| c.set(c.get() - 1)),
|
||||
}
|
||||
#[cfg(not(debug_assertions))]
|
||||
let _ = class;
|
||||
}
|
||||
|
||||
pub(crate) struct RawMutex<T> {
|
||||
state: AtomicU32,
|
||||
class: LockClass,
|
||||
data: UnsafeCell<T>,
|
||||
}
|
||||
|
||||
// SAFETY: standard mutex argument — exclusive access to `data` is mediated by
|
||||
// `state`; `T: Send` suffices for both because `&RawMutex` only ever hands out
|
||||
// access to one thread at a time.
|
||||
unsafe impl<T: Send> Send for RawMutex<T> {}
|
||||
unsafe impl<T: Send> Sync for RawMutex<T> {}
|
||||
|
||||
impl<T> RawMutex<T> {
|
||||
/// A Leaf-class mutex — the default for runtime cold data.
|
||||
pub(crate) const fn new(data: T) -> Self {
|
||||
Self::with_class(data, LockClass::Leaf)
|
||||
}
|
||||
|
||||
/// A Channel-class mutex — for channel internals only.
|
||||
pub(crate) const fn new_channel(data: T) -> Self {
|
||||
Self::with_class(data, LockClass::Channel)
|
||||
}
|
||||
|
||||
pub(crate) const fn with_class(data: T, class: LockClass) -> Self {
|
||||
Self {
|
||||
state: AtomicU32::new(UNLOCKED),
|
||||
class,
|
||||
data: UnsafeCell::new(data),
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn lock(&self) -> RawMutexGuard<'_, T> {
|
||||
// Enter NoPreempt *before* acquiring, so a preemption can't fire
|
||||
// between acquisition and guard construction.
|
||||
let prev_preempt = crate::preempt::preemption_swap(false);
|
||||
order_check_acquire(self.class);
|
||||
if self
|
||||
.state
|
||||
.compare_exchange(UNLOCKED, LOCKED, Ordering::Acquire, Ordering::Relaxed)
|
||||
.is_err()
|
||||
{
|
||||
self.lock_slow();
|
||||
}
|
||||
RawMutexGuard {
|
||||
m: self,
|
||||
prev_preempt,
|
||||
}
|
||||
}
|
||||
|
||||
#[cold]
|
||||
fn lock_slow(&self) {
|
||||
// Bounded spin first: the expected hold time is far below the cost of
|
||||
// a futex round trip.
|
||||
let mut spins = 0;
|
||||
loop {
|
||||
let s = self.state.load(Ordering::Relaxed);
|
||||
if s == UNLOCKED
|
||||
&& self
|
||||
.state
|
||||
.compare_exchange_weak(UNLOCKED, LOCKED, Ordering::Acquire, Ordering::Relaxed)
|
||||
.is_ok()
|
||||
{
|
||||
return;
|
||||
}
|
||||
spins += 1;
|
||||
if spins >= SPIN_LIMIT {
|
||||
break;
|
||||
}
|
||||
std::hint::spin_loop();
|
||||
}
|
||||
// Futex path. Mark contended and sleep until woken; on wake, retake
|
||||
// by swapping to CONTENDED (we cannot know whether other waiters
|
||||
// remain, so we must conservatively keep the contended marker).
|
||||
while self.state.swap(CONTENDED, Ordering::Acquire) != UNLOCKED {
|
||||
futex_wait(&self.state, CONTENDED);
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
fn unlock(&self) {
|
||||
if self.state.swap(UNLOCKED, Ordering::Release) == CONTENDED {
|
||||
futex_wake(&self.state, 1);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) struct RawMutexGuard<'a, T> {
|
||||
m: &'a RawMutex<T>,
|
||||
prev_preempt: bool,
|
||||
}
|
||||
|
||||
impl<T> Deref for RawMutexGuard<'_, T> {
|
||||
type Target = T;
|
||||
#[inline]
|
||||
fn deref(&self) -> &T {
|
||||
// SAFETY: guard existence implies exclusive ownership of the lock.
|
||||
unsafe { &*self.m.data.get() }
|
||||
}
|
||||
}
|
||||
|
||||
impl<T> DerefMut for RawMutexGuard<'_, T> {
|
||||
#[inline]
|
||||
fn deref_mut(&mut self) -> &mut T {
|
||||
// SAFETY: as above, plus &mut self.
|
||||
unsafe { &mut *self.m.data.get() }
|
||||
}
|
||||
}
|
||||
|
||||
impl<T> Drop for RawMutexGuard<'_, T> {
|
||||
#[inline]
|
||||
fn drop(&mut self) {
|
||||
self.m.unlock();
|
||||
order_check_release(self.m.class);
|
||||
// Restore preemption only after the lock is released.
|
||||
crate::preempt::preemption_swap(self.prev_preempt);
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// futex (x86-64 Linux; master is x86-only, see arm-port branch)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
fn futex_wait(state: &AtomicU32, expected: u32) {
|
||||
// SAFETY: `state` is a valid, aligned u32 for the duration of the call.
|
||||
// Spurious wakeups and EAGAIN (value already changed) are both handled by
|
||||
// the caller's retry loop.
|
||||
unsafe {
|
||||
libc::syscall(
|
||||
libc::SYS_futex,
|
||||
state.as_ptr(),
|
||||
libc::FUTEX_WAIT | libc::FUTEX_PRIVATE_FLAG,
|
||||
expected,
|
||||
std::ptr::null::<libc::timespec>(),
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
fn futex_wake(state: &AtomicU32, n: i32) {
|
||||
// SAFETY: as above.
|
||||
unsafe {
|
||||
libc::syscall(
|
||||
libc::SYS_futex,
|
||||
state.as_ptr(),
|
||||
libc::FUTEX_WAKE | libc::FUTEX_PRIVATE_FLAG,
|
||||
n,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use std::sync::Arc;
|
||||
|
||||
#[test]
|
||||
fn uncontended_lock_unlock() {
|
||||
let m = RawMutex::new(0u64);
|
||||
for _ in 0..1000 {
|
||||
*m.lock() += 1;
|
||||
}
|
||||
assert_eq!(*m.lock(), 1000);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn contended_counter_is_exact() {
|
||||
const THREADS: usize = 8;
|
||||
const PER: u64 = 50_000;
|
||||
let m = Arc::new(RawMutex::new(0u64));
|
||||
let hs: Vec<_> = (0..THREADS)
|
||||
.map(|_| {
|
||||
let m = m.clone();
|
||||
std::thread::spawn(move || {
|
||||
for _ in 0..PER {
|
||||
*m.lock() += 1;
|
||||
}
|
||||
})
|
||||
})
|
||||
.collect();
|
||||
for h in hs {
|
||||
h.join().unwrap();
|
||||
}
|
||||
assert_eq!(*m.lock(), THREADS as u64 * PER);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn no_poison_on_unwind() {
|
||||
let m = Arc::new(RawMutex::new(0u64));
|
||||
let m2 = m.clone();
|
||||
let _ = std::thread::spawn(move || {
|
||||
let _g = m2.lock();
|
||||
panic!("unwind while holding");
|
||||
})
|
||||
.join();
|
||||
// A std Mutex would now be poisoned; this one just works.
|
||||
*m.lock() += 1;
|
||||
assert_eq!(*m.lock(), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn channel_lock_nests_under_leaf() {
|
||||
// The permitted ordering: Leaf -> Channel (finalize/monitor clone a
|
||||
// sender under a cold lock). Must not trip the order check.
|
||||
let leaf = RawMutex::new(0u64);
|
||||
let chan = RawMutex::new_channel(0u64);
|
||||
let _l = leaf.lock();
|
||||
let _c = chan.lock();
|
||||
}
|
||||
|
||||
#[cfg(debug_assertions)]
|
||||
#[test]
|
||||
#[should_panic(expected = "lock order violated")]
|
||||
fn leaf_under_channel_is_rejected() {
|
||||
let leaf = RawMutex::new(0u64);
|
||||
let chan = RawMutex::new_channel(0u64);
|
||||
let _c = chan.lock();
|
||||
let _l = leaf.lock(); // Channel -> Leaf: forbidden
|
||||
}
|
||||
|
||||
#[cfg(debug_assertions)]
|
||||
#[test]
|
||||
#[should_panic(expected = "lock order violated")]
|
||||
fn two_leaves_are_rejected() {
|
||||
let a = RawMutex::new(0u64);
|
||||
let b = RawMutex::new(0u64);
|
||||
let _ga = a.lock();
|
||||
let _gb = b.lock(); // leaves are mutual: forbidden
|
||||
}
|
||||
|
||||
#[cfg(debug_assertions)]
|
||||
#[test]
|
||||
#[should_panic(expected = "lock order violated")]
|
||||
fn two_channel_locks_are_rejected() {
|
||||
let a = RawMutex::new_channel(0u64);
|
||||
let b = RawMutex::new_channel(0u64);
|
||||
let _ga = a.lock();
|
||||
let _gb = b.lock(); // channel locks are mutual leaves: forbidden
|
||||
}
|
||||
}
|
||||
-780
@@ -1,780 +0,0 @@
|
||||
//! Give an actor a name so other actors can find it and message it.
|
||||
//!
|
||||
//! Without the registry, the only way to reach an actor is to already be
|
||||
//! holding its [`Pid`], usually because you spawned it yourself or someone
|
||||
//! passed it to you. That is fine for a worker you just created, but it does
|
||||
//! not work for a well-known service that arbitrary parts of your program
|
||||
//! need to find independently, like a logger, a config store, or a
|
||||
//! connection pool. The registry solves this: an actor claims a name once,
|
||||
//! and from then on any other actor can look that name up, or send to it
|
||||
//! directly, without ever having been handed a `Pid`.
|
||||
//!
|
||||
//! ```
|
||||
//! use smarm::{channel, register, run, send, spawn, whereis, Name};
|
||||
//!
|
||||
//! const COUNTER: Name<u64> = Name::new("counter");
|
||||
//!
|
||||
//! run(|| {
|
||||
//! let (ready_tx, ready_rx) = channel::<()>();
|
||||
//! let (tx, rx) = channel::<u64>();
|
||||
//!
|
||||
//! let worker = spawn(move || {
|
||||
//! // Claim the name for this actor's inbox. Any actor holding
|
||||
//! // `COUNTER` can now reach this one by name.
|
||||
//! register(COUNTER, tx).unwrap();
|
||||
//! ready_tx.send(()).unwrap();
|
||||
//! assert_eq!(rx.recv().unwrap(), 42);
|
||||
//! });
|
||||
//!
|
||||
//! ready_rx.recv().unwrap(); // wait for the worker to register
|
||||
//!
|
||||
//! // Look the name up, or just send to it directly.
|
||||
//! assert_eq!(whereis("counter"), Some(worker.pid()));
|
||||
//! send(COUNTER, 42).unwrap();
|
||||
//!
|
||||
//! worker.join().unwrap();
|
||||
//!
|
||||
//! // The name dies with the actor: nobody holds it anymore.
|
||||
//! assert_eq!(whereis("counter"), None);
|
||||
//! });
|
||||
//! ```
|
||||
//!
|
||||
//! ## Names carry a message type
|
||||
//!
|
||||
//! A [`Name<M>`] is a plain string plus a type parameter `M`: the message
|
||||
//! type that name expects to receive. [`Name::new`] is `const`, so the usual
|
||||
//! pattern is a module-level constant like `COUNTER` above, shared by every
|
||||
//! caller. The type parameter means a name is only ever sent the kind of
|
||||
//! message it was declared for. If two different constants share the same
|
||||
//! string but have different message types, they still address two
|
||||
//! independent channels on the same actor: registering both just gives that
|
||||
//! actor two ways to be reached, one per message type. This is how you give
|
||||
//! one actor a "public" channel and a separate, differently-typed "admin"
|
||||
//! channel under related names, without inventing an enum to merge them.
|
||||
//!
|
||||
//! ## One actor per name, looked up fresh every time
|
||||
//!
|
||||
//! A name always points at exactly one actor at a time (contrast a *process
|
||||
//! group*, from the [`pg`](crate::pg) module, which is one name mapping to
|
||||
//! many actors). Unlike a plain [`Pid`], which names one specific actor
|
||||
//! forever and stops working the moment that actor dies, a name is
|
||||
//! re-resolved on every [`send`]: if the actor holding it dies and a new one
|
||||
//! registers under the same name, the next `send` reaches the new holder
|
||||
//! automatically. Use a name for a long-lived service whose exact identity
|
||||
//! you do not want to track by hand; use a `Pid` when you already have one
|
||||
//! and want to talk to that exact actor.
|
||||
//!
|
||||
//! ## Registration ends when the actor does
|
||||
//!
|
||||
//! There is no separate step to clean up a name when its actor exits: dying
|
||||
//! is enough. The next operation that touches a dead binding (a [`whereis`],
|
||||
//! a [`send`], or another actor's [`register`] of the same name) notices the
|
||||
//! actor is gone and clears the stale entry as a side effect, so the name
|
||||
//! becomes free again. [`unregister`] is only for a live actor voluntarily
|
||||
//! giving up a name it no longer wants; nothing has to call it on the way
|
||||
//! out.
|
||||
//!
|
||||
//! ## Implementation notes
|
||||
//!
|
||||
//! These details matter if you are working on smarm itself; they are not
|
||||
//! part of the public contract.
|
||||
//!
|
||||
//! Internally, each live actor that has published at least one channel owns
|
||||
//! a `Mailbox`: its pid plus a set of typed channels, keyed by the message
|
||||
//! type's `TypeId`. A stored channel is a `Box<dyn Any + Send>` that
|
||||
//! is concretely a `Sender<M>`; resolving for `M` looks up that exact
|
||||
//! `TypeId` and downcasts, so the downcast cannot fail on correct data (a
|
||||
//! failure would be a bug in the registry itself, checked in debug builds).
|
||||
//! Registering a name therefore means: find or create the actor's mailbox,
|
||||
//! insert the channel under its type, and point the name at the actor's pid.
|
||||
//!
|
||||
//! There is no callback when an actor exits. Every operation that touches a
|
||||
//! binding checks the target pid's liveness directly against the scheduler's
|
||||
//! slot table (which also tracks a generation counter, so a dead actor's
|
||||
//! reused slot index is never mistaken for the same actor). A binding to a
|
||||
//! dead actor is treated as absent and dropped right there. This keeps the
|
||||
//! registry decoupled from actor teardown, at the cost of a dead binding
|
||||
//! lingering until something happens to look at it.
|
||||
//!
|
||||
//! The whole registry (both the name index and the per-actor mailboxes) sits
|
||||
//! behind one lock, which is what lets a name-addressed [`send`] resolve and
|
||||
//! clone the target's sender in a single critical section. The sender is
|
||||
//! cloned while that lock is held, then the lock is released before the
|
||||
//! actual send, since delivering a message can wake a parked receiver and
|
||||
//! that wakeup work should not run while the registry is locked.
|
||||
|
||||
use crate::channel::Sender;
|
||||
use crate::pid::{Addressable, Name, Pid};
|
||||
use crate::scheduler::{self_pid, with_runtime};
|
||||
use std::any::{type_name, Any, TypeId};
|
||||
use std::collections::HashMap;
|
||||
|
||||
/// Why a [`register`] call was rejected.
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub enum RegisterError {
|
||||
/// The name is bound to a different, still-live actor.
|
||||
NameTaken { holder: Pid },
|
||||
/// The caller is not a live actor (cannot happen for `self`, kept for
|
||||
/// symmetry / future explicit-pid registration).
|
||||
NoProc,
|
||||
}
|
||||
|
||||
impl std::fmt::Display for RegisterError {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
match self {
|
||||
RegisterError::NameTaken { holder } => {
|
||||
write!(f, "name is already registered to live actor {holder}")
|
||||
}
|
||||
RegisterError::NoProc => write!(f, "caller is not a live actor"),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl std::error::Error for RegisterError {}
|
||||
|
||||
/// Why a send did not deliver. Every variant carries the undelivered message
|
||||
/// back, mirroring [`crate::channel::SendError`], so a failed send never
|
||||
/// silently drops what you tried to send.
|
||||
///
|
||||
/// `Debug` and `Display` are hand-written so neither requires `M: Debug`,
|
||||
/// since the payload is handed back to you, not printed.
|
||||
pub enum SendError<M> {
|
||||
/// No live actor is currently registered under this name. Returned only
|
||||
/// by name-addressed [`send`]; the pid-addressed counterpart of "nothing
|
||||
/// there" is [`SendError::Dead`].
|
||||
Unresolved(M),
|
||||
/// The actor this pid identifies has died, even if its slot has since
|
||||
/// been taken over by a different, live actor. A direct `Pid<A>` send
|
||||
/// never redirects to that new occupant; contrast name-addressed
|
||||
/// [`send`], which would reach it. Returned by the pid-addressed sends,
|
||||
/// [`send_to`] and [`send_dyn`].
|
||||
Dead(M),
|
||||
/// The actor is live but has not published a channel for this message
|
||||
/// type.
|
||||
NoChannel(M),
|
||||
/// The actor's channel for this message type is closed (its receiver has
|
||||
/// been dropped).
|
||||
Closed(M),
|
||||
/// No live member was available to deliver to: returned by
|
||||
/// [`dispatch`](crate::dispatch) when the target process group is empty
|
||||
/// or every member in it has died. The name-addressed counterpart of
|
||||
/// this case is [`SendError::Unresolved`].
|
||||
NoMember(M),
|
||||
}
|
||||
|
||||
impl<M> SendError<M> {
|
||||
/// Recover the undelivered message.
|
||||
pub fn into_inner(self) -> M {
|
||||
match self {
|
||||
SendError::Unresolved(m)
|
||||
| SendError::Dead(m)
|
||||
| SendError::NoChannel(m)
|
||||
| SendError::Closed(m)
|
||||
| SendError::NoMember(m) => m,
|
||||
}
|
||||
}
|
||||
|
||||
fn variant(&self) -> &'static str {
|
||||
match self {
|
||||
SendError::Unresolved(_) => "Unresolved",
|
||||
SendError::Dead(_) => "Dead",
|
||||
SendError::NoChannel(_) => "NoChannel",
|
||||
SendError::Closed(_) => "Closed",
|
||||
SendError::NoMember(_) => "NoMember",
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl<M> std::fmt::Debug for SendError<M> {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
write!(f, "SendError::{}", self.variant())
|
||||
}
|
||||
}
|
||||
|
||||
impl<M> std::fmt::Display for SendError<M> {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
match self {
|
||||
SendError::Unresolved(_) => write!(f, "no live actor registered under that name"),
|
||||
SendError::Dead(_) => {
|
||||
write!(f, "the addressed actor is no longer the live incarnation")
|
||||
}
|
||||
SendError::NoChannel(_) => write!(f, "actor has no channel for this message type"),
|
||||
SendError::Closed(_) => write!(f, "the actor's channel for this type is closed"),
|
||||
SendError::NoMember(_) => write!(f, "no live member in the process group"),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl<M> std::error::Error for SendError<M> {}
|
||||
|
||||
/// A registry-stored channel, type-erased over its message type. The stored
|
||||
/// object must serve two readers: `clone_sender` (downcast back to the concrete
|
||||
/// `Sender<M>`) and the runtime introspection snapshot (queued length without
|
||||
/// knowing `M`). A bare `Box<dyn Any>` gives the first but not the second, so
|
||||
/// we erase behind this small trait instead.
|
||||
trait ErasedSender: Send {
|
||||
fn as_any(&self) -> &dyn Any;
|
||||
fn queued_len(&self) -> usize;
|
||||
fn receiver_alive(&self) -> bool;
|
||||
}
|
||||
|
||||
impl<M: Send + 'static> ErasedSender for Sender<M> {
|
||||
fn as_any(&self) -> &dyn Any {
|
||||
self
|
||||
}
|
||||
fn queued_len(&self) -> usize {
|
||||
Sender::queued_len(self)
|
||||
}
|
||||
fn receiver_alive(&self) -> bool {
|
||||
Sender::receiver_alive(self)
|
||||
}
|
||||
}
|
||||
|
||||
/// One typed channel of an actor, type-erased. Concretely a `Sender<M>` filed
|
||||
/// under `TypeId::of::<M>()`; `msg_type` is `type_name::<M>()`, kept for
|
||||
/// observability tooling and as the debug cross-check on the downcast.
|
||||
struct Channel {
|
||||
sender: Box<dyn ErasedSender>,
|
||||
msg_type: &'static str,
|
||||
}
|
||||
|
||||
/// An actor's messageable surface: its identity plus every typed channel it has
|
||||
/// published, keyed by message [`TypeId`]. Stored once per live actor; reached
|
||||
/// by pid (directly) or by any name pointing at that pid.
|
||||
struct Mailbox {
|
||||
pid: Pid,
|
||||
channels: HashMap<TypeId, Channel>,
|
||||
}
|
||||
|
||||
impl Mailbox {
|
||||
fn new(pid: Pid) -> Self {
|
||||
Self {
|
||||
pid,
|
||||
channels: HashMap::new(),
|
||||
}
|
||||
}
|
||||
|
||||
/// Clone the `Sender<M>` for this actor, if it has one. Called **under the
|
||||
/// registry Leaf lock**: `Sender::clone` takes a Channel lock, which is
|
||||
/// legal under a Leaf (Leaf -> Channel).
|
||||
fn clone_sender<M: Send + 'static>(&self) -> Option<Sender<M>> {
|
||||
let ch = self.channels.get(&TypeId::of::<M>())?;
|
||||
let tx = match ch.sender.as_any().downcast_ref::<Sender<M>>() {
|
||||
Some(tx) => tx,
|
||||
None => panic!(
|
||||
"smarm: channel keyed by TypeId but downcast to its own type failed (core corrupt)"
|
||||
),
|
||||
};
|
||||
debug_assert_eq!(ch.msg_type, type_name::<M>(), "msg_type / TypeId disagree");
|
||||
Some(tx.clone())
|
||||
}
|
||||
}
|
||||
|
||||
/// Per-actor registry view handed to runtime introspection: registered names
|
||||
/// and summed mailbox depth, tagged with the mailbox's `pid` so a stale
|
||||
/// incarnation can be filtered against the slab. Covers only *published*
|
||||
/// channels (`register` / `install` / `spawn_addr` / gen_server start); an
|
||||
/// actor that holds only a private `channel()` receiver is invisible here and
|
||||
/// reports depth 0.
|
||||
pub(crate) struct MailboxInfo {
|
||||
pub(crate) pid: Pid,
|
||||
pub(crate) names: Vec<&'static str>,
|
||||
pub(crate) depth: u32,
|
||||
}
|
||||
|
||||
/// The directory. Invariant (held under the registry lock): every value in
|
||||
/// `by_name` is the full [`Pid`] (index *and* generation) of an actor that
|
||||
/// published a [`Mailbox`] into `by_index` at registration time. Stale entries
|
||||
/// (dead holders, including holders whose slot has since been re-tenanted by
|
||||
/// a different actor) violate nothing: they are pruned on contact, and the
|
||||
/// generation makes "dead" decidable even after slot reuse.
|
||||
pub(crate) struct Registry {
|
||||
/// `pid.index() -> the actor's mailbox`. The handle store.
|
||||
by_index: HashMap<u32, Mailbox>,
|
||||
/// `name -> holder pid`. Several names may map to one actor. The full pid
|
||||
/// (not just the index) is load-bearing: an index alone cannot tell a dead
|
||||
/// holder from the live actor now tenanting its recycled slot. Comparing
|
||||
/// only the index would make such a name read as live-held (unresolvable
|
||||
/// and unregisterable at once) and could misdeliver to whatever new,
|
||||
/// same-typed actor now sits in that slot.
|
||||
by_name: HashMap<&'static str, Pid>,
|
||||
}
|
||||
|
||||
impl Registry {
|
||||
pub(crate) fn new() -> Self {
|
||||
Self {
|
||||
by_index: HashMap::new(),
|
||||
by_name: HashMap::new(),
|
||||
}
|
||||
}
|
||||
|
||||
/// Drop a dead holder's artifacts: every name bound to it, and its
|
||||
/// mailbox, but only while the mailbox is still *its own*. A recycled
|
||||
/// slot's mailbox belongs to the live tenant (publish replaces it
|
||||
/// wholesale on pid mismatch) and is left untouched.
|
||||
fn prune_holder(&mut self, holder: Pid) {
|
||||
self.by_name.retain(|_, p| *p != holder);
|
||||
if self
|
||||
.by_index
|
||||
.get(&holder.index())
|
||||
.is_some_and(|mb| mb.pid == holder)
|
||||
{
|
||||
self.by_index.remove(&holder.index());
|
||||
}
|
||||
}
|
||||
|
||||
/// Runtime introspection input: per-slot-index registry view, giving the
|
||||
/// actor's registered names (inverted from `by_name`) and its mailbox
|
||||
/// depth (queued messages summed across every published typed channel).
|
||||
/// Carries each mailbox's full `pid` so the caller can discard a stale
|
||||
/// incarnation's entry against the slab's live generation. Names are
|
||||
/// matched to mailboxes by *full pid*, so a stale name (dead holder)
|
||||
/// still annotates the corpse's own mailbox if that survives, but never a
|
||||
/// recycled slot's new tenant; names that attach to no mailbox are
|
||||
/// dropped, since that violates no invariant and they get pruned on next
|
||||
/// contact.
|
||||
pub(crate) fn introspect_map(&self) -> HashMap<u32, MailboxInfo> {
|
||||
let mut names: HashMap<Pid, Vec<&'static str>> = HashMap::new();
|
||||
for (&name, &pid) in &self.by_name {
|
||||
names.entry(pid).or_default().push(name);
|
||||
}
|
||||
let mut out: HashMap<u32, MailboxInfo> = HashMap::with_capacity(self.by_index.len());
|
||||
for (&idx, mb) in &self.by_index {
|
||||
let depth: usize = mb.channels.values().map(|c| c.sender.queued_len()).sum();
|
||||
out.insert(
|
||||
idx,
|
||||
MailboxInfo {
|
||||
pid: mb.pid,
|
||||
names: names.remove(&mb.pid).unwrap_or_default(),
|
||||
depth: depth.min(u32::MAX as usize) as u32,
|
||||
},
|
||||
);
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
/// Single-actor form of [`introspect_map`](Self::introspect_map): the
|
||||
/// registry view for one slot index, or `None` if no mailbox is published
|
||||
/// there. Used by the runtime's per-actor introspection so its cost stays
|
||||
/// proportional to the one actor rather than locking every channel in the
|
||||
/// runtime.
|
||||
pub(crate) fn introspect_one(&self, idx: u32) -> Option<MailboxInfo> {
|
||||
let mb = self.by_index.get(&idx)?;
|
||||
let depth: usize = mb.channels.values().map(|c| c.sender.queued_len()).sum();
|
||||
let names = self
|
||||
.by_name
|
||||
.iter()
|
||||
.filter_map(|(&n, &p)| (p == mb.pid).then_some(n))
|
||||
.collect();
|
||||
Some(MailboxInfo {
|
||||
pid: mb.pid,
|
||||
names,
|
||||
depth: depth.min(u32::MAX as usize) as u32,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
/// Is `pid` a live actor right now? Atomic slot-word read; no lock.
|
||||
fn live(inner: &crate::runtime::RuntimeInner, pid: Pid) -> bool {
|
||||
inner.slot_at(pid).is_some_and(|s| s.is_live_for(pid))
|
||||
}
|
||||
|
||||
/// Give the current actor's channel a name, so other actors can find and
|
||||
/// message it by that name instead of needing its [`Pid`].
|
||||
///
|
||||
/// Calling this again with the same `(name, type)` from the same actor is
|
||||
/// harmless. Registering a *second* message type under the same (or a
|
||||
/// different) name from the same actor just adds another typed channel to
|
||||
/// that actor's mailbox; it does not replace the first.
|
||||
///
|
||||
/// Fails with [`RegisterError::NameTaken`] if the name is currently held by a
|
||||
/// *different* live actor. A name held by an actor that has since died is not
|
||||
/// considered taken: it is quietly reclaimed and handed to you. Panics if
|
||||
/// called outside [`run`](crate::run).
|
||||
pub fn register<M: Send + 'static>(name: Name<M>, tx: Sender<M>) -> Result<(), RegisterError> {
|
||||
register_with(self_pid(), name.as_str(), tx)
|
||||
}
|
||||
|
||||
/// Bind `name` to `pid`'s mailbox and publish `tx` under `M`'s [`TypeId`], for
|
||||
/// an explicit (already-live) actor rather than `self`. The shared core of
|
||||
/// [`register`] (which passes `self_pid()`) and the parent-side server-name
|
||||
/// bind in `gen_server`, which names a freshly spawned server before its body
|
||||
/// has run, so the name resolves the instant `start()` returns. Same collision
|
||||
/// rules and lock discipline as `register`.
|
||||
pub(crate) fn register_with<M: Send + 'static>(
|
||||
me: Pid,
|
||||
key: &'static str,
|
||||
tx: Sender<M>,
|
||||
) -> Result<(), RegisterError> {
|
||||
with_runtime(|inner| {
|
||||
// Stamp-eligibility for the terminal record (soak sig 4): flag the
|
||||
// tenancy BEFORE the binding lands and outside the registry lock (no
|
||||
// nesting), so no successfully-registered actor can die unflagged.
|
||||
// A register that then fails leaves a harmless overshoot; a stale
|
||||
// `me` is screened by the same live() the binding requires below.
|
||||
if live(inner, me) {
|
||||
if let Some(slot) = inner.slot_at(me) {
|
||||
slot.cold.lock().watchable = true;
|
||||
}
|
||||
}
|
||||
let mut reg = inner.registry.lock();
|
||||
if !live(inner, me) {
|
||||
return Err(RegisterError::NoProc);
|
||||
}
|
||||
if let Some(&holder) = reg.by_name.get(key) {
|
||||
if holder == me {
|
||||
// Same actor: just add the channel below.
|
||||
} else if live(inner, holder) {
|
||||
return Err(RegisterError::NameTaken { holder });
|
||||
} else {
|
||||
// Dead holder: free the name (and its other stale artifacts).
|
||||
// Liveness is judged against the *stored* pid, generation
|
||||
// included, so a recycled slot's live tenant no longer makes a
|
||||
// dead name read as taken.
|
||||
reg.prune_holder(holder);
|
||||
}
|
||||
}
|
||||
// Publish (or extend) the mailbox with this channel, then bind the name.
|
||||
publish_channel::<M>(&mut reg, me, tx);
|
||||
reg.by_name.insert(key, me);
|
||||
Ok(())
|
||||
})
|
||||
}
|
||||
|
||||
/// Insert or extend the current actor's mailbox with one typed channel, filed
|
||||
/// under its message [`TypeId`]. Shared by [`register`] (which then binds a
|
||||
/// name) and [`install`] (which does not). A leftover mailbox at this slot
|
||||
/// index from a dead prior incarnation (pid mismatch) is replaced wholesale.
|
||||
/// Caller holds the registry lock and has established that `me` is live.
|
||||
///
|
||||
/// **One channel per message type per actor.** Publishing a second `M`
|
||||
/// channel on the same live actor replaces the first — and if the first's
|
||||
/// receiver is still alive, that replacement drops its last sender, closing
|
||||
/// it, and any `recv`/`select` on it then returns "closed" immediately and
|
||||
/// forever: a silent hot loop that starves the scheduler. That is never
|
||||
/// intended, so it panics here (found the hard way in RFC 010 c9, where two
|
||||
/// `Name<String>`s registered on one actor did exactly this). Replacing a
|
||||
/// channel whose receiver is already gone is fine (an actor re-registering
|
||||
/// after dropping its old inbox) and stays silent. To hold two names of the
|
||||
/// same type, register them from two actors, or bind both names to one
|
||||
/// cloned sender.
|
||||
fn publish_channel<M: Send + 'static>(reg: &mut Registry, me: Pid, tx: Sender<M>) {
|
||||
let mb = reg
|
||||
.by_index
|
||||
.entry(me.index())
|
||||
.or_insert_with(|| Mailbox::new(me));
|
||||
if mb.pid != me {
|
||||
*mb = Mailbox::new(me);
|
||||
}
|
||||
if let Some(existing) = mb.channels.get(&TypeId::of::<M>()) {
|
||||
assert!(
|
||||
!existing.sender.receiver_alive() || same_channel::<M>(existing, &tx),
|
||||
"smarm: actor {me:?} already publishes a live channel for message type `{}`; \
|
||||
a second one would replace and CLOSE the first (its receiver would then \
|
||||
read as closed forever). Register the second name from another actor, or \
|
||||
bind both names to a clone of the same sender.",
|
||||
type_name::<M>()
|
||||
);
|
||||
}
|
||||
mb.channels.insert(
|
||||
TypeId::of::<M>(),
|
||||
Channel {
|
||||
sender: Box::new(tx),
|
||||
msg_type: type_name::<M>(),
|
||||
},
|
||||
);
|
||||
}
|
||||
|
||||
/// True if `existing` and `tx` are senders of the very same channel (a
|
||||
/// cloned sender bound under a second name is the sanctioned way to hold two
|
||||
/// names of one type on one actor).
|
||||
fn same_channel<M: Send + 'static>(existing: &Channel, tx: &Sender<M>) -> bool {
|
||||
existing
|
||||
.sender
|
||||
.as_any()
|
||||
.downcast_ref::<Sender<M>>()
|
||||
.is_some_and(|old| old.same_channel(tx))
|
||||
}
|
||||
|
||||
/// Publish the current actor's `Sender<A::Msg>` into its mailbox **without**
|
||||
/// binding a name, and hand back the typed [`Pid<A>`] that addresses this
|
||||
/// actor directly.
|
||||
///
|
||||
/// This is for an actor that wants to be reachable directly by its pid,
|
||||
/// rather than only through a re-resolving [`Name`]: call this once with your
|
||||
/// inbox sender, then hand the returned `Pid<A>` to whoever should be able to
|
||||
/// message you. Unlike [`register`] there is no name to collide on, and the
|
||||
/// current actor is always live while inside `run()`, so this cannot fail.
|
||||
/// Panics if called outside [`run`](crate::run).
|
||||
pub fn install<A: Addressable>(tx: Sender<A::Msg>) -> Pid<A> {
|
||||
let me = self_pid();
|
||||
with_runtime(|inner| {
|
||||
let mut reg = inner.registry.lock();
|
||||
debug_assert!(live(inner, me), "self_pid() is a live actor inside run()");
|
||||
publish_channel::<A::Msg>(&mut reg, me, tx);
|
||||
});
|
||||
// `me` is this actor; re-type the identity as `Pid<A>` (the channel for
|
||||
// `A::Msg` was just published, so the typed address is now messageable).
|
||||
Pid::from_raw(me.raw())
|
||||
}
|
||||
|
||||
/// Publish `tx` into `pid`'s mailbox under `M`'s [`TypeId`], for an explicit
|
||||
/// (freshly minted, already-live) actor rather than `self`. The parent-side
|
||||
/// half of [`spawn_addr`](crate::spawn_addr): the spawner makes the inbox and
|
||||
/// publishes the sender here *before* handing back the `Pid<A>`, so an
|
||||
/// immediate `send_to` on the returned pid always resolves. The address is
|
||||
/// live the instant the caller holds it, with no dependence on the spawned
|
||||
/// actor's body having run yet.
|
||||
///
|
||||
/// Caller guarantees `pid` is the just-installed actor (queued, this exact
|
||||
/// incarnation); `publish_channel` replaces any stale leftover at the slot.
|
||||
pub(crate) fn install_for<M: Send + 'static>(pid: Pid, tx: Sender<M>) {
|
||||
with_runtime(|inner| {
|
||||
let mut reg = inner.registry.lock();
|
||||
debug_assert!(
|
||||
live(inner, pid),
|
||||
"install_for: pid must be a freshly spawned, live actor"
|
||||
);
|
||||
publish_channel::<M>(&mut reg, pid, tx);
|
||||
});
|
||||
}
|
||||
|
||||
/// Look up which actor currently holds `name`, if any. Returns `None` if the
|
||||
/// name is unbound, or if it was bound to an actor that has since died (the
|
||||
/// stale binding is cleared as a side effect of this call).
|
||||
pub fn whereis(name: &str) -> Option<Pid> {
|
||||
with_runtime(|inner| {
|
||||
let mut reg = inner.registry.lock();
|
||||
let pid = *reg.by_name.get(name)?;
|
||||
if live(inner, pid) {
|
||||
Some(pid)
|
||||
} else {
|
||||
// Generation-checked against the stored holder: a recycled slot's
|
||||
// live tenant reads dead here, and the stale name heals.
|
||||
reg.prune_holder(pid);
|
||||
None
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
/// What a name is bound to, three-valued (bridge soak signature 4).
|
||||
///
|
||||
/// [`Live`](NameResolution::Live) is [`whereis`]'s `Some`.
|
||||
/// [`Corpse`](NameResolution::Corpse) carries the *stored* holder pid of a
|
||||
/// dead-but-unpruned binding — a state Erlang cannot represent (its name
|
||||
/// death unregisters atomically; smarm's prune is lazy), captured here before
|
||||
/// the prune that `whereis` performs discards it, so the caller can consult
|
||||
/// [`terminal_reason`](crate::monitor::terminal_reason) for the tenancy's
|
||||
/// real down reason. [`Unbound`](NameResolution::Unbound) matches Erlang's
|
||||
/// unregistered name.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum NameResolution {
|
||||
/// The stored holder is live (generation-checked); the binding stands.
|
||||
Live(Pid),
|
||||
/// The stored holder is dead. The binding was pruned on the way out —
|
||||
/// the name heals exactly as `whereis` heals it; only the evidence is
|
||||
/// returned instead of discarded. A second resolve is `Unbound`.
|
||||
Corpse(Pid),
|
||||
/// No binding stored (never registered, or already pruned by any reader).
|
||||
Unbound,
|
||||
}
|
||||
|
||||
/// Resolve `name` like [`whereis`], but keep the corpse: the dead-holder arm
|
||||
/// returns the stored pid it pruned instead of a bare `None`. Same lock
|
||||
/// discipline and pruning behavior as `whereis`; same `Runtime::run()`
|
||||
/// context contract.
|
||||
pub fn resolve_name(name: &str) -> NameResolution {
|
||||
with_runtime(|inner| {
|
||||
let mut reg = inner.registry.lock();
|
||||
let Some(&pid) = reg.by_name.get(name) else {
|
||||
return NameResolution::Unbound;
|
||||
};
|
||||
if live(inner, pid) {
|
||||
NameResolution::Live(pid)
|
||||
} else {
|
||||
reg.prune_holder(pid);
|
||||
NameResolution::Corpse(pid)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
/// Like [`whereis`], but returns a *typed* [`Pid<A>`] instead of a bare
|
||||
/// [`Pid`], so a follow-up [`send_to`] is compile-checked instead of needing
|
||||
/// the untyped [`send_dyn`] escape hatch. `None` if the name is unbound or its
|
||||
/// holder has died.
|
||||
///
|
||||
/// The type `A` is not checked against what the name's holder actually
|
||||
/// published: if you pick the wrong `A`, this still succeeds, but the next
|
||||
/// send against the returned pid degrades to [`SendError::NoChannel`] rather
|
||||
/// than reaching the wrong actor or the wrong channel.
|
||||
///
|
||||
/// Panics if called outside [`run`](crate::run).
|
||||
pub fn lookup_as<A: Addressable>(name: &str) -> Option<Pid<A>> {
|
||||
whereis(name).map(crate::pid::assert_type::<A>)
|
||||
}
|
||||
|
||||
/// Resolve `name` to its actor's pid and a cloned `Sender<M>`, all under one
|
||||
/// lock acquisition. The crate-internal building block for `gen_server`'s
|
||||
/// by-name addressing: a named server publishes its inbox as a
|
||||
/// `Sender<Envelope<G>>` (via [`register_with`]), and the server's `call` /
|
||||
/// `cast` / `whereis_server` recover that exact typed sender here to rebuild a
|
||||
/// `GenServerRef<G>`. `None` if unbound, dead (pruned on the way out), or
|
||||
/// holding no `M` channel.
|
||||
pub(crate) fn resolve_named_sender<M: Send + 'static>(name: &str) -> Option<(Pid, Sender<M>)> {
|
||||
with_runtime(|inner| {
|
||||
let mut reg = inner.registry.lock();
|
||||
let pid = *reg.by_name.get(name)?;
|
||||
if !live(inner, pid) {
|
||||
// Stored-pid liveness, generation included: a name whose holder
|
||||
// died is pruned (heals) even if the slot has a new tenant.
|
||||
// Otherwise the tenant's mailbox would make the name unresolvable
|
||||
// without pruning, wedging it for the tenant's lifetime.
|
||||
reg.prune_holder(pid);
|
||||
return None;
|
||||
}
|
||||
// A live holder's mailbox is its own (publish replaces wholesale on
|
||||
// pid mismatch, and one live actor per slot), so index lookup is safe.
|
||||
let tx = reg
|
||||
.by_index
|
||||
.get(&pid.index())
|
||||
.and_then(Mailbox::clone_sender::<M>)?;
|
||||
Some((pid, tx))
|
||||
})
|
||||
}
|
||||
|
||||
/// Give up a name. Returns the actor it pointed at, if that actor was still
|
||||
/// live. Only the *name* is freed; the actor's mailbox (and any other names
|
||||
/// bound to it) are unaffected. A binding to an already-dead actor reports
|
||||
/// `None`, since there was nothing live to release.
|
||||
pub fn unregister(name: &str) -> Option<Pid> {
|
||||
with_runtime(|inner| {
|
||||
let mut reg = inner.registry.lock();
|
||||
let pid = reg.by_name.remove(name)?;
|
||||
if live(inner, pid) {
|
||||
Some(pid)
|
||||
} else {
|
||||
None
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
/// Look `name` up and deliver `msg` to whichever actor currently holds it.
|
||||
/// This is the point of naming an actor: a name you can send a message to
|
||||
/// directly, without a separate lookup step.
|
||||
///
|
||||
/// On failure the message comes back to you, wrapped in the [`SendError`]
|
||||
/// variant that explains why: [`SendError::Unresolved`] if no live actor
|
||||
/// currently holds the name, [`SendError::NoChannel`] if the actor that holds
|
||||
/// it never published a channel for `M`, or [`SendError::Closed`] if it
|
||||
/// published one but has since dropped the receiving end. Panics if called
|
||||
/// outside [`run`](crate::run).
|
||||
pub fn send<M: Send + 'static>(name: Name<M>, msg: M) -> Result<(), SendError<M>> {
|
||||
let key = name.as_str();
|
||||
with_runtime(|inner| {
|
||||
// Resolve + clone the sender under the registry lock, then drop the
|
||||
// lock before sending (a send can unpark a receiver).
|
||||
let tx = {
|
||||
let mut reg = inner.registry.lock();
|
||||
let pid = match reg.by_name.get(key) {
|
||||
Some(&p) => p,
|
||||
None => return Err(SendError::Unresolved(msg)),
|
||||
};
|
||||
if !live(inner, pid) {
|
||||
// Stored-pid liveness (generation included), so a recycled
|
||||
// slot's new live tenant is never mistaken for the name's
|
||||
// original (now-dead) holder.
|
||||
reg.prune_holder(pid);
|
||||
return Err(SendError::Unresolved(msg));
|
||||
}
|
||||
match reg
|
||||
.by_index
|
||||
.get(&pid.index())
|
||||
.and_then(Mailbox::clone_sender::<M>)
|
||||
{
|
||||
Some(tx) => tx,
|
||||
None => return Err(SendError::NoChannel(msg)),
|
||||
}
|
||||
};
|
||||
tx.send(msg)
|
||||
.map_err(|crate::channel::SendError(m)| SendError::Closed(m))
|
||||
})
|
||||
}
|
||||
|
||||
/// Resolve a *raw* pid to its mailbox and deliver `msg` on the channel for `M`,
|
||||
/// with **no redirect**. The stored mailbox must be this exact incarnation
|
||||
/// (generation included) and still live; otherwise the actor this pid named
|
||||
/// is gone and the result is [`SendError::Dead`], even when the slot now
|
||||
/// holds a different, live actor (which is left untouched). Shared by
|
||||
/// [`send_to`] (typed, `M = A::Msg`, channel guaranteed on an installed
|
||||
/// actor) and [`send_dyn`] (explicit `M`, where `NoChannel` is a real
|
||||
/// outcome).
|
||||
fn send_to_pid<M: Send + 'static>(
|
||||
inner: &crate::runtime::RuntimeInner,
|
||||
pid: Pid,
|
||||
msg: M,
|
||||
) -> Result<(), SendError<M>> {
|
||||
// Resolve + clone the sender under the registry lock, then drop the lock
|
||||
// before sending (a send can unpark a receiver), same order as `send`.
|
||||
let tx = {
|
||||
let mut reg = inner.registry.lock();
|
||||
match reg.by_index.get(&pid.index()).map(|m| m.pid) {
|
||||
// Exact incarnation, still alive: its `M` channel, or NoChannel.
|
||||
Some(stored) if stored == pid && live(inner, pid) => {
|
||||
match reg
|
||||
.by_index
|
||||
.get(&pid.index())
|
||||
.and_then(Mailbox::clone_sender::<M>)
|
||||
{
|
||||
Some(tx) => tx,
|
||||
None => return Err(SendError::NoChannel(msg)),
|
||||
}
|
||||
}
|
||||
// Our incarnation's mailbox, but the actor has died: prune + Dead.
|
||||
Some(stored) if stored == pid => {
|
||||
reg.prune_holder(pid);
|
||||
return Err(SendError::Dead(msg));
|
||||
}
|
||||
// A different incarnation (or nothing) occupies the slot: the actor
|
||||
// this pid named is gone. Do not disturb any newer occupant.
|
||||
_ => return Err(SendError::Dead(msg)),
|
||||
}
|
||||
};
|
||||
tx.send(msg)
|
||||
.map_err(|crate::channel::SendError(m)| SendError::Closed(m))
|
||||
}
|
||||
|
||||
/// Deliver `msg` directly to the exact actor identified by `pid`. Unlike
|
||||
/// name-addressed [`send`], there is **no redirect**: if that specific actor
|
||||
/// has died, the message comes back as [`SendError::Dead`], even if its slot
|
||||
/// has since been taken over by a different, live actor. Use this when you
|
||||
/// already hold a `Pid<A>` and want to talk to that one actor specifically;
|
||||
/// use [`send`] with a [`Name`] when you want whichever actor currently holds
|
||||
/// a name.
|
||||
///
|
||||
/// The message type is the actor's `A::Msg`, so on a live actor that has
|
||||
/// installed its inbox (via [`install`] or [`register`]) the channel is
|
||||
/// always present; [`SendError::NoChannel`] therefore means the actor is live
|
||||
/// but never published a `Pid<A>`-reachable inbox. Panics if called outside
|
||||
/// [`run`](crate::run).
|
||||
pub fn send_to<A: Addressable>(pid: Pid<A>, msg: A::Msg) -> Result<(), SendError<A::Msg>> {
|
||||
with_runtime(|inner| send_to_pid::<A::Msg>(inner, pid.erase(), msg))
|
||||
}
|
||||
|
||||
/// The escape hatch for sending to a bare, untyped [`Pid`] when the typed
|
||||
/// [`send_to`] is unavailable, for example a pid recovered from a [`Down`]
|
||||
/// notification or a group's `members()` list, where you no longer know the
|
||||
/// actor's message type at compile time.
|
||||
///
|
||||
/// Because the message type is not checked at compile time here, this is the
|
||||
/// one send that can genuinely be live-but-wrong: the actor may be alive yet
|
||||
/// expose no channel for `M`, in which case you get [`SendError::NoChannel`]
|
||||
/// back instead of a misdelivery. Liveness and redirect behavior are
|
||||
/// otherwise identical to [`send_to`]: identity-bound, no redirect,
|
||||
/// [`SendError::Dead`] once the addressed incarnation is gone. Prefer
|
||||
/// `send_to` with a typed `Pid<A>` whenever you have one; reach for this only
|
||||
/// when you don't. Panics if called outside [`run`](crate::run).
|
||||
///
|
||||
/// [`Down`]: crate::Down
|
||||
pub fn send_dyn<M: Send + 'static>(pid: Pid, msg: M) -> Result<(), SendError<M>> {
|
||||
with_runtime(|inner| send_to_pid::<M>(inner, pid, msg))
|
||||
}
|
||||
@@ -1,748 +0,0 @@
|
||||
//! The run queue, selected at COMPILE TIME by mutually-exclusive cargo
|
||||
//! features (no runtime dispatch — the scheduler's pop loop is the hottest
|
||||
//! code in the runtime):
|
||||
//!
|
||||
//! - `rq-mutex` — `Mutex<VecDeque>`. The control/baseline:
|
||||
//! strictly FIFO, trivially correct, one global lock.
|
||||
//! - `rq-mpmc` (default) — a single hand-rolled Vyukov bounded MPMC ring (per-cell
|
||||
//! sequence numbers). Strict FIFO, lock-free, one hot
|
||||
//! enqueue/dequeue cache-line pair.
|
||||
//! - `rq-striped` — M Vyukov rings with fetch-add ticket distribution.
|
||||
//! *Relaxed* FIFO: ordering across stripes is bounded-skewed
|
||||
//! (≈ one ring's worth of reordering per stripe), in exchange
|
||||
//! for spreading the hot line M ways. Predicted winner at
|
||||
//! high core counts; phase 4's shootout decides.
|
||||
//!
|
||||
//! Select non-default variants with `--no-default-features --features rq-…`
|
||||
//! (cargo features are additive, so the default must be switched off).
|
||||
//!
|
||||
//! All variants are compiled unconditionally (so every build runs every
|
||||
//! variant's unit tests); the feature only picks which one the runtime uses
|
||||
//! via the `RunQueue` alias.
|
||||
//!
|
||||
//! # Contract (shared by all variants)
|
||||
//!
|
||||
//! - **Occupancy is bounded by `max_actors`.** A pid is in the queue at most
|
||||
//! once (pushes pair 1:1 with transitions into `Queued`; only the
|
||||
//! scheduler transitions `Queued → Running` — see the state-machine docs
|
||||
//! in `runtime.rs`), and at most `max_actors` actors exist. With the RFC
|
||||
//! 005 wake slot enabled the invariant reads "in (slot ⊕ shared queue) at
|
||||
//! most once" — a slot push *replaces* the queue push at the same
|
||||
//! protocol point, and a displacement moves the occupant, never copies
|
||||
//! it — so the bound holds verbatim. The bounded rings are sized ≥
|
||||
//! `max_actors`, so **`push` is infallible**; a full
|
||||
//! ring is an invariant violation and panics loudly rather than spinning.
|
||||
//! - **Preemption must be disabled around every push/pop** (debug-asserted).
|
||||
//! For the mutex variant this is the usual no-switch/no-unwind-under-lock
|
||||
//! rule. For the rings it is *load-bearing in a sharper way*: a producer
|
||||
//! suspended between claiming a cell and publishing its sequence number
|
||||
//! stalls every consumer behind that cell — on a busy runtime that is a
|
||||
//! livelock, since the suspended actor's own resume entry sits behind the
|
||||
//! hole. Callers get this for free: every queue op happens inside
|
||||
//! `with_runtime`/`try_with_runtime` (NoPreempt for their span since
|
||||
//! phase 2) or on a scheduler thread between resumes (preemption off).
|
||||
//! - **`pop() == None` is a snapshot, not a fence.** A push that is mid-
|
||||
//! publish (or in a stripe the probe already passed) may be missed; the
|
||||
//! caller's idle path sleeps ≤ 100µs and retries, so the cost is a bounded
|
||||
//! latency blip, never a lost entry. Termination does not lean on this:
|
||||
//! the all-clear is `live_actors == 0` (+ io quiescent), and `live == 0`
|
||||
//! already implies the queue holds nothing actionable — see the argument
|
||||
//! in `schedule_loop`.
|
||||
//! - `len()` is approximate (stats only).
|
||||
|
||||
use crate::pid::Pid;
|
||||
use crate::sync_shim::{fence, AtomicUsize, Ordering, UnsafeCell};
|
||||
use std::mem::MaybeUninit;
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Feature selection
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
#[cfg(not(any(feature = "rq-mutex", feature = "rq-mpmc", feature = "rq-striped")))]
|
||||
compile_error!(
|
||||
"smarm: no run queue selected. Enable exactly one of the features \
|
||||
`rq-mpmc` (default), `rq-mutex`, `rq-striped`."
|
||||
);
|
||||
#[cfg(all(feature = "rq-mutex", feature = "rq-mpmc"))]
|
||||
compile_error!(
|
||||
"smarm: features `rq-mutex` and `rq-mpmc` are mutually exclusive \
|
||||
(use --no-default-features to drop the default `rq-mpmc`)."
|
||||
);
|
||||
#[cfg(all(feature = "rq-mutex", feature = "rq-striped"))]
|
||||
compile_error!("smarm: features `rq-mutex` and `rq-striped` are mutually exclusive.");
|
||||
#[cfg(all(feature = "rq-mpmc", feature = "rq-striped"))]
|
||||
compile_error!(
|
||||
"smarm: features `rq-mpmc` and `rq-striped` are mutually exclusive \
|
||||
(use --no-default-features to drop the default `rq-mpmc`)."
|
||||
);
|
||||
|
||||
#[cfg(feature = "rq-mutex")]
|
||||
pub(crate) type RunQueue = MutexQueue;
|
||||
#[cfg(feature = "rq-mpmc")]
|
||||
pub(crate) type RunQueue = MpmcRing;
|
||||
#[cfg(feature = "rq-striped")]
|
||||
pub(crate) type RunQueue = StripedRing;
|
||||
|
||||
#[inline]
|
||||
fn assert_no_preempt() {
|
||||
debug_assert!(
|
||||
!crate::preempt::preemption_enabled(),
|
||||
"run-queue op with preemption enabled — a switch mid-op stalls or \
|
||||
corrupts the queue; route through with_runtime or scheduler context"
|
||||
);
|
||||
}
|
||||
|
||||
/// Escalating wait for transient ring stalls: a peer preempted by the OS
|
||||
/// inside its ~100ns claim→publish window (finding 13). Spin first (the
|
||||
/// common stall is a peer that is merely slow, gone within a few hundred
|
||||
/// cycles), then donate the timeslice — under oversubscription pure
|
||||
/// spinning STARVES the descheduled peer of the CPU it needs to publish
|
||||
/// (measured in the finding-13 soak: 10⁶ pure spins can outlast the very
|
||||
/// stall they prolong). OS-level yielding is orthogonal to
|
||||
/// `assert_no_preempt`, which guards smarm signal preemption only.
|
||||
struct Backoff(u32);
|
||||
|
||||
impl Backoff {
|
||||
const SPIN_LIMIT: u32 = 6;
|
||||
|
||||
fn new() -> Self {
|
||||
Self(0)
|
||||
}
|
||||
|
||||
/// Total waits so far — lets bounded callers cap the yield phase.
|
||||
fn steps(&self) -> u32 {
|
||||
self.0
|
||||
}
|
||||
|
||||
#[inline]
|
||||
fn wait(&mut self) {
|
||||
#[cfg(loom)]
|
||||
// Loom has no notion of spinning time; every wait is a scheduling
|
||||
// point so the model explores the stalled peer's progress.
|
||||
loom::thread::yield_now();
|
||||
#[cfg(not(loom))]
|
||||
if self.0 <= Self::SPIN_LIMIT {
|
||||
for _ in 0..1u32 << self.0 {
|
||||
std::hint::spin_loop();
|
||||
}
|
||||
} else {
|
||||
std::thread::yield_now();
|
||||
}
|
||||
self.0 += 1;
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// rq-mutex — the baseline
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
#[allow(dead_code)]
|
||||
pub struct MutexQueue {
|
||||
q: std::sync::Mutex<std::collections::VecDeque<Pid>>,
|
||||
}
|
||||
|
||||
#[allow(dead_code)]
|
||||
impl MutexQueue {
|
||||
pub fn new(_threads: usize, max_actors: usize) -> Self {
|
||||
Self {
|
||||
// Pre-size: the queue can never outgrow the slab, and one
|
||||
// allocation at init beats reallocating under the lock later.
|
||||
q: std::sync::Mutex::new(std::collections::VecDeque::with_capacity(max_actors)),
|
||||
}
|
||||
}
|
||||
|
||||
pub fn push(&self, pid: Pid) {
|
||||
assert_no_preempt();
|
||||
match self.q.lock() {
|
||||
Ok(mut g) => g.push_back(pid),
|
||||
Err(e) => panic!("smarm: run-queue q lock poisoned (core corrupt): {e}"),
|
||||
}
|
||||
}
|
||||
|
||||
pub fn pop(&self) -> Option<Pid> {
|
||||
assert_no_preempt();
|
||||
match self.q.lock() {
|
||||
Ok(mut g) => g.pop_front(),
|
||||
Err(e) => panic!("smarm: run-queue q lock poisoned (core corrupt): {e}"),
|
||||
}
|
||||
}
|
||||
|
||||
pub fn len(&self) -> u64 {
|
||||
match self.q.lock() {
|
||||
Ok(g) => g.len() as u64,
|
||||
Err(e) => panic!("smarm: run-queue q lock poisoned (core corrupt): {e}"),
|
||||
}
|
||||
}
|
||||
|
||||
pub fn is_empty(&self) -> bool {
|
||||
match self.q.lock() {
|
||||
Ok(g) => g.is_empty(),
|
||||
Err(e) => panic!("smarm: run-queue q lock poisoned (core corrupt): {e}"),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// rq-mpmc — Vyukov bounded MPMC ring
|
||||
// ---------------------------------------------------------------------------
|
||||
//
|
||||
// Dmitry Vyukov's bounded MPMC queue: each cell carries a sequence number.
|
||||
// A producer may write cell `i` when `seq == pos` (its turn); it publishes
|
||||
// with `seq = pos + 1`. A consumer may read when `seq == pos + 1`; it
|
||||
// releases the cell to the next lap with `seq = pos + capacity`. Producers
|
||||
// and consumers each contend on one counter; cell handoff is a per-cell
|
||||
// Acquire/Release pair, so unrelated push/pop pairs don't serialize.
|
||||
|
||||
/// Pad to a cache-line pair so the producer and consumer counters (and the
|
||||
/// cells) don't false-share.
|
||||
#[repr(align(128))]
|
||||
struct CachePadded<T>(T);
|
||||
|
||||
struct Cell {
|
||||
seq: AtomicUsize,
|
||||
pid: UnsafeCell<MaybeUninit<Pid>>,
|
||||
}
|
||||
|
||||
#[allow(dead_code)]
|
||||
pub struct MpmcRing {
|
||||
buf: Box<[Cell]>,
|
||||
mask: usize,
|
||||
enqueue_pos: CachePadded<AtomicUsize>,
|
||||
dequeue_pos: CachePadded<AtomicUsize>,
|
||||
}
|
||||
|
||||
// SAFETY: cells are handed off between threads via the per-cell seq
|
||||
// (Release on publish, Acquire on claim); Pid is Copy + Send.
|
||||
unsafe impl Send for MpmcRing {}
|
||||
unsafe impl Sync for MpmcRing {}
|
||||
|
||||
#[allow(dead_code)]
|
||||
impl MpmcRing {
|
||||
pub fn new(_threads: usize, max_actors: usize) -> Self {
|
||||
Self::with_capacity(max_actors)
|
||||
}
|
||||
|
||||
/// Pub for the raw-structure microbench (sized to the op count so the
|
||||
/// occupancy contract is trivially met there). Runtime code uses `new`.
|
||||
pub fn with_capacity(min_cap: usize) -> Self {
|
||||
// Occupancy ≤ max_actors (queue contract), so capacity = the next
|
||||
// power of two ≥ max_actors can never overflow. (≥ 2 so mask works.)
|
||||
let cap = min_cap.next_power_of_two().max(2);
|
||||
let buf: Box<[Cell]> = (0..cap)
|
||||
.map(|i| Cell {
|
||||
seq: AtomicUsize::new(i),
|
||||
pid: UnsafeCell::new(MaybeUninit::uninit()),
|
||||
})
|
||||
.collect();
|
||||
Self {
|
||||
buf,
|
||||
mask: cap - 1,
|
||||
enqueue_pos: CachePadded(AtomicUsize::new(0)),
|
||||
dequeue_pos: CachePadded(AtomicUsize::new(0)),
|
||||
}
|
||||
}
|
||||
|
||||
pub fn push(&self, pid: Pid) {
|
||||
assert_no_preempt();
|
||||
if self.try_push(pid) {
|
||||
return;
|
||||
}
|
||||
self.push_slow(pid);
|
||||
}
|
||||
|
||||
/// Cold path: the cell at `enqueue_pos` is still a lap behind. Two
|
||||
/// worlds are indistinguishable at the cell (finding 13): a consumer
|
||||
/// preempted between its `dequeue_pos` claim and its seq release while
|
||||
/// the ring lapped onto that cell (transient), or a genuine occupancy
|
||||
/// overflow (a runtime bug). The COUNTERS discriminate — the same shape
|
||||
/// as crossbeam `ArrayQueue::push`'s fence + opposite-counter check:
|
||||
/// occupancy < capacity ⇒ transient ⇒ wait for the stalled peer;
|
||||
/// occupancy ≥ capacity ⇒ the at-most-once-enqueued invariant really is
|
||||
/// broken ⇒ panic (a legal push starts from occupancy ≤ max_actors − 1).
|
||||
#[cold]
|
||||
fn push_slow(&self, pid: Pid) {
|
||||
let mut backoff = Backoff::new();
|
||||
loop {
|
||||
// Order the counter reads after the failed cell read.
|
||||
// `enqueue_pos` is loaded BEFORE `dequeue_pos`, so a pop racing
|
||||
// us can only make the computed occupancy an UNDERestimate —
|
||||
// conservative in the safe direction (never a spurious panic;
|
||||
// a genuine violation is persistent and caught next lap).
|
||||
fence(Ordering::SeqCst);
|
||||
let enq = self.enqueue_pos.0.load(Ordering::SeqCst);
|
||||
let deq = self.dequeue_pos.0.load(Ordering::SeqCst);
|
||||
let occ = enq.wrapping_sub(deq);
|
||||
assert!(
|
||||
occ <= self.mask,
|
||||
"smarm: run queue occupancy {} reached capacity {} — a pid \
|
||||
was enqueued more than once (or more pids exist than \
|
||||
max_actors); the at-most-once-enqueued invariant is broken \
|
||||
(enq={} deq={})",
|
||||
occ,
|
||||
self.mask + 1,
|
||||
enq,
|
||||
deq
|
||||
);
|
||||
backoff.wait();
|
||||
if self.try_push(pid) {
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// One full claim attempt; `false` when the cell at `enqueue_pos` is
|
||||
/// still a lap behind — which means EITHER genuinely full OR a
|
||||
/// lap-stalled consumer (finding 13). The cell cannot tell the two
|
||||
/// apart; callers must disambiguate via the counters (`push_slow`) or
|
||||
/// tolerate refusal (`StripedRing`'s probe).
|
||||
fn try_push(&self, pid: Pid) -> bool {
|
||||
let mut pos = self.enqueue_pos.0.load(Ordering::Relaxed);
|
||||
loop {
|
||||
let cell = &self.buf[pos & self.mask];
|
||||
let seq = cell.seq.load(Ordering::Acquire);
|
||||
let diff = seq as isize - pos as isize;
|
||||
if diff == 0 {
|
||||
// Our turn: claim the position.
|
||||
match self.enqueue_pos.0.compare_exchange_weak(
|
||||
pos,
|
||||
pos + 1,
|
||||
Ordering::Relaxed,
|
||||
Ordering::Relaxed,
|
||||
) {
|
||||
Ok(_) => {
|
||||
// SAFETY: the claim gives us exclusive write access
|
||||
// to this cell until we publish below.
|
||||
cell.pid.with_mut(|p| unsafe { (*p).write(pid) });
|
||||
cell.seq.store(pos + 1, Ordering::Release);
|
||||
return true;
|
||||
}
|
||||
Err(actual) => pos = actual,
|
||||
}
|
||||
} else if diff < 0 {
|
||||
return false; // full — a whole lap behind
|
||||
} else {
|
||||
pos = self.enqueue_pos.0.load(Ordering::Relaxed);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub fn pop(&self) -> Option<Pid> {
|
||||
assert_no_preempt();
|
||||
let mut backoff = Backoff::new();
|
||||
let mut pos = self.dequeue_pos.0.load(Ordering::Relaxed);
|
||||
loop {
|
||||
let cell = &self.buf[pos & self.mask];
|
||||
let seq = cell.seq.load(Ordering::Acquire);
|
||||
let diff = seq as isize - (pos + 1) as isize;
|
||||
if diff == 0 {
|
||||
match self.dequeue_pos.0.compare_exchange_weak(
|
||||
pos,
|
||||
pos + 1,
|
||||
Ordering::Relaxed,
|
||||
Ordering::Relaxed,
|
||||
) {
|
||||
Ok(_) => {
|
||||
// SAFETY: the claim gives us exclusive read access;
|
||||
// the producer's Release publish made `pid` visible
|
||||
// to our Acquire load of `seq`.
|
||||
let pid = cell.pid.with(|p| unsafe { (*p).assume_init_read() });
|
||||
// Release the cell for the next lap.
|
||||
cell.seq.store(pos + self.mask + 1, Ordering::Release);
|
||||
return Some(pid);
|
||||
}
|
||||
Err(actual) => pos = actual,
|
||||
}
|
||||
} else if diff < 0 {
|
||||
// The cell at `pos` is unpublished: either the queue is
|
||||
// empty, or the producer that claimed it was preempted
|
||||
// inside its claim→publish window (finding 13). The cell
|
||||
// cannot tell the two apart; the counters can.
|
||||
fence(Ordering::SeqCst);
|
||||
let deq = self.dequeue_pos.0.load(Ordering::SeqCst);
|
||||
if deq != pos {
|
||||
// Stale head — no verdict; re-probe at the real head.
|
||||
pos = deq;
|
||||
continue;
|
||||
}
|
||||
if self.enqueue_pos.0.load(Ordering::SeqCst) == pos {
|
||||
return None; // counters agree: genuinely empty
|
||||
}
|
||||
// Producer mid-publish. Wait briefly, then report None
|
||||
// anyway — a DELIBERATE bounded deviation from crossbeam's
|
||||
// unbounded retry: a spurious None is correctness-benign
|
||||
// here (the stalled push completes and its RFC 018
|
||||
// enqueue-wake re-wakes a parked scheduler), a parked
|
||||
// scheduler beats a yielding one, and StripedRing's pop
|
||||
// probe must not hang on one stripe.
|
||||
if backoff.steps() > Backoff::SPIN_LIMIT + 8 {
|
||||
return None;
|
||||
}
|
||||
backoff.wait();
|
||||
} else {
|
||||
pos = self.dequeue_pos.0.load(Ordering::Relaxed);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub fn len(&self) -> u64 {
|
||||
let e = self.enqueue_pos.0.load(Ordering::Relaxed);
|
||||
let d = self.dequeue_pos.0.load(Ordering::Relaxed);
|
||||
e.saturating_sub(d) as u64
|
||||
}
|
||||
|
||||
pub fn is_empty(&self) -> bool {
|
||||
self.len() == 0
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// rq-striped — M Vyukov rings, ticket-distributed
|
||||
// ---------------------------------------------------------------------------
|
||||
//
|
||||
// Producers fetch-add a ticket and start probing at stripe `ticket % M`;
|
||||
// consumers do the same with their own ticket. Under symmetric load the
|
||||
// tickets spread producers and consumers uniformly, so each stripe sees
|
||||
// ~1/M of the traffic and the single hot cache-line pair becomes M cooler
|
||||
// ones. FIFO is relaxed: two pushes that land in different stripes can be
|
||||
// popped in either order, with skew bounded by stripe occupancy imbalance.
|
||||
//
|
||||
// Push probes forward from its home stripe until a `try_push` succeeds.
|
||||
// Σ stripe capacity ≥ 2 × max_actors while occupancy ≤ max_actors, so at
|
||||
// every instant at least half the total capacity is free and the probe
|
||||
// terminates (in practice on the first stripe).
|
||||
|
||||
#[allow(dead_code)]
|
||||
pub struct StripedRing {
|
||||
stripes: Box<[MpmcRing]>,
|
||||
/// Stripe count minus one (count is a power of two).
|
||||
stripe_mask: usize,
|
||||
push_ticket: CachePadded<AtomicUsize>,
|
||||
pop_ticket: CachePadded<AtomicUsize>,
|
||||
}
|
||||
|
||||
#[allow(dead_code)]
|
||||
impl StripedRing {
|
||||
pub fn new(threads: usize, max_actors: usize) -> Self {
|
||||
// One stripe per scheduler thread, rounded up to a power of two —
|
||||
// more stripes than threads buys nothing (at most `threads` ops are
|
||||
// in flight) and costs pop-probe latency when mostly empty.
|
||||
let n = threads.max(1).next_power_of_two();
|
||||
// Per-stripe capacity: 2 × max_actors / n in total, and never below
|
||||
// a floor that keeps degenerate configs (tiny slab, many threads)
|
||||
// trivially correct.
|
||||
let per = ((2 * max_actors) / n).next_power_of_two().max(8);
|
||||
let stripes: Box<[MpmcRing]> = (0..n).map(|_| MpmcRing::with_capacity(per)).collect();
|
||||
Self {
|
||||
stripes,
|
||||
stripe_mask: n - 1,
|
||||
push_ticket: CachePadded(AtomicUsize::new(0)),
|
||||
pop_ticket: CachePadded(AtomicUsize::new(0)),
|
||||
}
|
||||
}
|
||||
|
||||
pub fn push(&self, pid: Pid) {
|
||||
assert_no_preempt();
|
||||
let home = self.push_ticket.0.fetch_add(1, Ordering::Relaxed);
|
||||
// Probe from the home stripe; capacity headroom (Σ ≥ 2×occupancy)
|
||||
// guarantees a free stripe exists, so the outer loop terminates.
|
||||
// The retry-from-home lap handles the racy case where every stripe
|
||||
// momentarily refused us.
|
||||
let mut backoff = Backoff::new();
|
||||
loop {
|
||||
for i in 0..=self.stripe_mask {
|
||||
let s = &self.stripes[(home + i) & self.stripe_mask];
|
||||
if s.try_push(pid) {
|
||||
return;
|
||||
}
|
||||
}
|
||||
// Every stripe refused this lap: either transiently full (the
|
||||
// headroom argument above guarantees a genuinely free stripe
|
||||
// exists) or the probes landed on lap-stalled cells (finding
|
||||
// 13 — try_push cannot tell the two apart). Waiting is correct
|
||||
// either way; the backoff escalates to an OS yield so stalled
|
||||
// consumers get the CPU they need to release their cells.
|
||||
backoff.wait();
|
||||
}
|
||||
}
|
||||
|
||||
pub fn pop(&self) -> Option<Pid> {
|
||||
assert_no_preempt();
|
||||
let home = self.pop_ticket.0.fetch_add(1, Ordering::Relaxed);
|
||||
for i in 0..=self.stripe_mask {
|
||||
if let Some(pid) = self.stripes[(home + i) & self.stripe_mask].pop() {
|
||||
return Some(pid);
|
||||
}
|
||||
}
|
||||
None // snapshot miss possible across stripes; idle-retry absorbs it
|
||||
}
|
||||
|
||||
pub fn len(&self) -> u64 {
|
||||
self.stripes.iter().map(|s| s.len()).sum()
|
||||
}
|
||||
|
||||
pub fn is_empty(&self) -> bool {
|
||||
self.stripes.iter().all(|s| s.is_empty())
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Tests — all variants, in every build (the feature only picks the alias)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
#[cfg(all(test, not(loom)))]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use std::collections::HashSet;
|
||||
use std::sync::Arc;
|
||||
|
||||
fn pid(i: u32) -> Pid {
|
||||
Pid::new(i, 0)
|
||||
}
|
||||
|
||||
fn fifo_smoke<Q>(q: &Q, push: impl Fn(&Q, Pid), pop: impl Fn(&Q) -> Option<Pid>) {
|
||||
for i in 0..100 {
|
||||
push(q, pid(i));
|
||||
}
|
||||
for i in 0..100 {
|
||||
assert_eq!(pop(q), Some(pid(i)));
|
||||
}
|
||||
assert_eq!(pop(q), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn mutex_fifo() {
|
||||
let q = MutexQueue::new(1, 1024);
|
||||
fifo_smoke(&q, |q, p| q.push(p), |q| q.pop());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn mpmc_fifo_single_thread() {
|
||||
let q = MpmcRing::new(1, 1024);
|
||||
fifo_smoke(&q, |q, p| q.push(p), |q| q.pop());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn mpmc_wraps_many_laps() {
|
||||
let q = MpmcRing::with_capacity(8);
|
||||
for lap in 0..1000u32 {
|
||||
for i in 0..8 {
|
||||
q.push(pid(lap * 8 + i));
|
||||
}
|
||||
for i in 0..8 {
|
||||
assert_eq!(q.pop(), Some(pid(lap * 8 + i)));
|
||||
}
|
||||
}
|
||||
assert_eq!(q.pop(), None);
|
||||
}
|
||||
|
||||
/// N producers, M consumers, every element exactly once. Run on plain OS
|
||||
/// threads (PREEMPTION_ENABLED defaults false, satisfying the contract).
|
||||
fn exactly_once<Q: Send + Sync + 'static>(
|
||||
q: Q,
|
||||
push: fn(&Q, Pid),
|
||||
pop: fn(&Q) -> Option<Pid>,
|
||||
producers: u32,
|
||||
consumers: u32,
|
||||
per_producer: u32,
|
||||
) {
|
||||
let q = Arc::new(q);
|
||||
let total = (producers * per_producer) as usize;
|
||||
let popped = Arc::new(std::sync::Mutex::new(Vec::with_capacity(total)));
|
||||
let remaining = Arc::new(AtomicUsize::new(total));
|
||||
|
||||
let mut hs = Vec::new();
|
||||
for p in 0..producers {
|
||||
let q = q.clone();
|
||||
hs.push(std::thread::spawn(move || {
|
||||
for i in 0..per_producer {
|
||||
push(&q, pid(p * per_producer + i));
|
||||
}
|
||||
}));
|
||||
}
|
||||
for _ in 0..consumers {
|
||||
let q = q.clone();
|
||||
let popped = popped.clone();
|
||||
let remaining = remaining.clone();
|
||||
hs.push(std::thread::spawn(move || {
|
||||
let mut local = Vec::new();
|
||||
while remaining.load(Ordering::Relaxed) > 0 {
|
||||
if let Some(pid) = pop(&q) {
|
||||
remaining.fetch_sub(1, Ordering::Relaxed);
|
||||
local.push(pid);
|
||||
} else {
|
||||
std::hint::spin_loop();
|
||||
}
|
||||
}
|
||||
popped.lock().unwrap().extend(local);
|
||||
}));
|
||||
}
|
||||
for h in hs {
|
||||
h.join().unwrap();
|
||||
}
|
||||
|
||||
let popped = popped.lock().unwrap();
|
||||
assert_eq!(popped.len(), total, "count mismatch");
|
||||
let set: HashSet<u64> = popped
|
||||
.iter()
|
||||
.map(|p| ((p.index() as u64) << 32) | p.generation() as u64)
|
||||
.collect();
|
||||
assert_eq!(set.len(), total, "duplicate or lost element");
|
||||
assert_eq!(pop(&q), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn mpmc_exactly_once_contended() {
|
||||
exactly_once(
|
||||
MpmcRing::new(8, 4096),
|
||||
|q, p| q.push(p),
|
||||
|q| q.pop(),
|
||||
4,
|
||||
4,
|
||||
1000,
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn striped_exactly_once_contended() {
|
||||
exactly_once(
|
||||
StripedRing::new(8, 4096),
|
||||
|q, p| q.push(p),
|
||||
|q| q.pop(),
|
||||
4,
|
||||
4,
|
||||
1000,
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn striped_drains_after_skewed_load() {
|
||||
// Hammer pushes from one thread (all tickets walk the stripes in
|
||||
// order) and verify a single consumer sees every element.
|
||||
let q = StripedRing::new(4, 64);
|
||||
let mut seen = HashSet::new();
|
||||
for i in 0..64 {
|
||||
q.push(pid(i));
|
||||
}
|
||||
while let Some(p) = q.pop() {
|
||||
assert!(seen.insert(p.index()));
|
||||
}
|
||||
assert_eq!(seen.len(), 64);
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// loom model tests — RUSTFLAGS="--cfg loom" cargo test --lib --release
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
#[cfg(all(test, loom))]
|
||||
mod loom_tests {
|
||||
use super::*;
|
||||
use loom::sync::Arc;
|
||||
use loom::thread;
|
||||
|
||||
fn pid(i: u32) -> Pid {
|
||||
Pid::new(i, 0)
|
||||
}
|
||||
|
||||
/// Two producers, main-thread consumer: both elements arrive exactly
|
||||
/// once, across every interleaving — including through a lap wraparound
|
||||
/// (capacity 2 forces cell reuse).
|
||||
#[test]
|
||||
fn mpmc_two_producers_exactly_once() {
|
||||
loom::model(|| {
|
||||
let q = Arc::new(MpmcRing::with_capacity(2));
|
||||
let mut hs = Vec::new();
|
||||
for i in 0..2u32 {
|
||||
let q = q.clone();
|
||||
hs.push(thread::spawn(move || q.push(pid(i))));
|
||||
}
|
||||
let mut got = Vec::new();
|
||||
while got.len() < 2 {
|
||||
match q.pop() {
|
||||
Some(p) => got.push(p.index()),
|
||||
None => thread::yield_now(),
|
||||
}
|
||||
}
|
||||
for h in hs {
|
||||
h.join().unwrap();
|
||||
}
|
||||
got.sort_unstable();
|
||||
assert_eq!(got, vec![0, 1]);
|
||||
assert!(q.pop().is_none());
|
||||
});
|
||||
}
|
||||
|
||||
/// Finding 13 regression: a consumer stalled between its dequeue_pos
|
||||
/// claim and its seq release must NOT make a lapping producer conclude
|
||||
/// "full" (the old code asserted here — occupancy never exceeded the
|
||||
/// bound; the cell just hadn't been recycled). Capacity 2: fill, pop
|
||||
/// once on each thread, then push a third element — in the
|
||||
/// interleavings where a pop's release is still pending, that push
|
||||
/// laps onto the stalled cell and must wait, not die.
|
||||
#[test]
|
||||
fn mpmc_lap_onto_stalled_consumer_completes() {
|
||||
loom::model(|| {
|
||||
let q = Arc::new(MpmcRing::with_capacity(2));
|
||||
q.push(pid(0));
|
||||
q.push(pid(1));
|
||||
let q2 = q.clone();
|
||||
let h = thread::spawn(move || q2.pop().expect("ring has two elements"));
|
||||
let a = q.pop().expect("ring has two elements");
|
||||
// enqueue_pos = 2 → cell 0: laps onto the other thread's cell
|
||||
// whenever its release is delayed.
|
||||
q.push(pid(2));
|
||||
let b = h.join().unwrap();
|
||||
let c = q.pop().expect("the lapping push must have landed");
|
||||
let mut got = vec![a.index(), b.index(), c.index()];
|
||||
got.sort_unstable();
|
||||
assert_eq!(got, vec![0, 1, 2]);
|
||||
});
|
||||
}
|
||||
|
||||
/// Producer races a consumer on a single element: the consumer either
|
||||
/// gets it or sees a clean None — never a torn/duplicated element.
|
||||
#[test]
|
||||
fn mpmc_push_pop_race() {
|
||||
loom::model(|| {
|
||||
let q = Arc::new(MpmcRing::with_capacity(2));
|
||||
let q2 = q.clone();
|
||||
let prod = thread::spawn(move || q2.push(pid(7)));
|
||||
let seen = q.pop();
|
||||
prod.join().unwrap();
|
||||
match seen {
|
||||
Some(p) => {
|
||||
assert_eq!(p.index(), 7);
|
||||
assert!(q.pop().is_none());
|
||||
}
|
||||
None => assert_eq!(q.pop().map(|p| p.index()), Some(7)),
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/// Striped: two producers landing in (potentially) different stripes,
|
||||
/// main-thread consumer drains both exactly once.
|
||||
#[test]
|
||||
fn striped_two_producers_exactly_once() {
|
||||
loom::model(|| {
|
||||
let q = Arc::new(StripedRing::new(2, 4));
|
||||
let mut hs = Vec::new();
|
||||
for i in 0..2u32 {
|
||||
let q = q.clone();
|
||||
hs.push(thread::spawn(move || q.push(pid(i))));
|
||||
}
|
||||
let mut got = Vec::new();
|
||||
while got.len() < 2 {
|
||||
match q.pop() {
|
||||
Some(p) => got.push(p.index()),
|
||||
None => thread::yield_now(),
|
||||
}
|
||||
}
|
||||
for h in hs {
|
||||
h.join().unwrap();
|
||||
}
|
||||
got.sort_unstable();
|
||||
assert_eq!(got, vec![0, 1]);
|
||||
assert!(q.pop().is_none());
|
||||
});
|
||||
}
|
||||
}
|
||||
+497
-2023
File diff suppressed because it is too large
Load Diff
+184
-1205
File diff suppressed because it is too large
Load Diff
-330
@@ -1,330 +0,0 @@
|
||||
//! RFC 019 §7 — overflow diagnostics.
|
||||
//!
|
||||
//! One process-global SIGSEGV handler, installed once at [`crate::runtime::init`]
|
||||
//! (before any scheduler thread exists, so the PRIOR save is unracing), plus a
|
||||
//! per-scheduler-thread `sigaltstack` registered at `schedule_loop` entry — a
|
||||
//! guard hit means the faulting stack has no room to run anything, so the
|
||||
//! altstack is not optional.
|
||||
//!
|
||||
//! The handler classifies `si_addr` against the *current* actor only, reached
|
||||
//! through `preempt::CURRENT_SLOT` — a const-initialized `Cell<*const Slot>`
|
||||
//! whose access is a plain TLS load (no lazy init, no allocation, no dtor
|
||||
//! registration), and which every scheduler thread has materialized before an
|
||||
//! actor can run on it. The slot's diag atomics (`diag_stack_top` & co) are
|
||||
//! written in `install_actor` before the Release publish and are only consulted
|
||||
//! here while the actor is on-CPU, so they cannot be stale.
|
||||
//!
|
||||
//! Two classification tiers:
|
||||
//! - **In-guard**: definitive. Rust frames probe pages in order
|
||||
//! (`__rust_probestack`), so Rust overflow always lands here; so does any C
|
||||
//! built with `-fstack-clash-protection` (distro-packaged libraries), and —
|
||||
//! with the 1 MiB default guard — nearly every unprobed frame too.
|
||||
//! - **Overshoot**: within [`OVERSHOOT_SLOP`] *below* the guard. An unprobed
|
||||
//! frame (cargo-built C via `cc` almost never enables clash protection)
|
||||
//! large enough to step over the guard in one `sub rsp`. Attribution is
|
||||
//! "probable": the address is in unmapped VA that nothing else owns, an
|
||||
//! actor was on-CPU, and the distance fits a frame — the diagnostic says so.
|
||||
//!
|
||||
//! Classified faults print one line (async-signal-safe: stack buffer +
|
||||
//! `write(2)`, no fmt, no alloc, no locks) and re-raise with default
|
||||
//! disposition — no unwind, no resume, no fail-soft (jarred; UB-adjacent from
|
||||
//! a handler). Unclassified faults reinstate the PRIOR handler and refault, so
|
||||
//! std's own "thread ... has overflowed its stack" diagnostics for OS-thread
|
||||
//! stacks survive our presence. Reinstating deregisters us for good, which is
|
||||
//! fine: the process is dying either way.
|
||||
|
||||
use std::cell::Cell;
|
||||
use std::mem::MaybeUninit;
|
||||
use std::sync::atomic::Ordering;
|
||||
use std::sync::Once;
|
||||
|
||||
/// Tier-2 window below the guard. Matches the guard default (and the kernel's
|
||||
/// `stack_guard_gap`): a frame that out-jumps both the guard and this window
|
||||
/// in one displacement is past what a diagnostic can honestly attribute.
|
||||
pub(crate) const OVERSHOOT_SLOP: usize = 1024 * 1024;
|
||||
|
||||
/// Per-scheduler-thread signal stack. MINSIGSTKSZ is ~11 KiB on AVX-512
|
||||
/// hardware; 64 KiB leaves the formatter room without mattering to anyone.
|
||||
/// One per OS thread, never freed: scheduler threads live for the process in
|
||||
/// practice, and repeated `run()`s on reused threads re-use the registration
|
||||
/// (the TLS flag), so the leak is bounded by the OS thread count.
|
||||
const ALTSTACK_SIZE: usize = 64 * 1024;
|
||||
|
||||
static INSTALL: Once = Once::new();
|
||||
/// The handler that was installed before ours (std's, typically). Written
|
||||
/// exactly once inside INSTALL — which completes in `runtime::init` before
|
||||
/// any scheduler thread (and thus any classifiable fault) can exist — and
|
||||
/// only read from the handler afterwards.
|
||||
static mut PRIOR: MaybeUninit<libc::sigaction> = MaybeUninit::uninit();
|
||||
|
||||
thread_local! {
|
||||
/// Whether this OS thread has registered its altstack.
|
||||
static ALTSTACK_SET: Cell<bool> = const { Cell::new(false) };
|
||||
}
|
||||
|
||||
/// Where a fault landed relative to the current actor's stack.
|
||||
#[derive(Debug, PartialEq, Eq)]
|
||||
pub(crate) enum FaultClass {
|
||||
/// Inside `[top − reserve − guard, top − reserve)`: the guard region.
|
||||
Guard,
|
||||
/// Within `OVERSHOOT_SLOP` below the guard: stepped over it. Payload is
|
||||
/// the distance below `guard_lo`.
|
||||
Overshoot(usize),
|
||||
/// Not ours to explain.
|
||||
Foreign,
|
||||
}
|
||||
|
||||
/// Pure classifier — all edges unit-tested below. `top` is the stack's usable
|
||||
/// top, `reserve`/`guard` its shape; both page-rounded by `Stack::new`.
|
||||
pub(crate) fn classify(addr: usize, top: usize, reserve: usize, guard: usize) -> FaultClass {
|
||||
let guard_hi = top.wrapping_sub(reserve);
|
||||
let guard_lo = guard_hi.wrapping_sub(guard);
|
||||
if addr >= guard_lo && addr < guard_hi {
|
||||
FaultClass::Guard
|
||||
} else if addr < guard_lo && addr >= guard_lo.saturating_sub(OVERSHOOT_SLOP) {
|
||||
FaultClass::Overshoot(guard_lo - addr)
|
||||
} else {
|
||||
FaultClass::Foreign
|
||||
}
|
||||
}
|
||||
|
||||
/// Install the process-global handler. Idempotent; called from
|
||||
/// `runtime::init`.
|
||||
pub(crate) fn install_once() {
|
||||
INSTALL.call_once(|| unsafe {
|
||||
let mut sa: libc::sigaction = std::mem::zeroed();
|
||||
sa.sa_sigaction = handler as *const () as usize;
|
||||
sa.sa_flags = libc::SA_SIGINFO | libc::SA_ONSTACK;
|
||||
libc::sigemptyset(&mut sa.sa_mask);
|
||||
let prior = &mut *std::ptr::addr_of_mut!(PRIOR);
|
||||
libc::sigaction(libc::SIGSEGV, &sa, prior.as_mut_ptr());
|
||||
});
|
||||
}
|
||||
|
||||
/// Register this OS thread's altstack (idempotent per thread). Called at
|
||||
/// `schedule_loop` entry, so every thread that can run an actor has one.
|
||||
pub(crate) fn register_altstack() {
|
||||
ALTSTACK_SET.with(|set| {
|
||||
if set.get() {
|
||||
return;
|
||||
}
|
||||
unsafe {
|
||||
let sp = libc::mmap(
|
||||
std::ptr::null_mut(),
|
||||
ALTSTACK_SIZE,
|
||||
libc::PROT_READ | libc::PROT_WRITE,
|
||||
libc::MAP_PRIVATE | libc::MAP_ANONYMOUS,
|
||||
-1,
|
||||
0,
|
||||
);
|
||||
if sp == libc::MAP_FAILED {
|
||||
// Degrade: no altstack means a guard hit dies without the
|
||||
// message (handler can't run) — the pre-RFC behavior, never
|
||||
// incorrectness.
|
||||
return;
|
||||
}
|
||||
let ss = libc::stack_t {
|
||||
ss_sp: sp,
|
||||
ss_flags: 0,
|
||||
ss_size: ALTSTACK_SIZE,
|
||||
};
|
||||
libc::sigaltstack(&ss, std::ptr::null_mut());
|
||||
}
|
||||
set.set(true);
|
||||
});
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// The handler
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
unsafe extern "C" fn handler(
|
||||
_sig: libc::c_int,
|
||||
info: *mut libc::siginfo_t,
|
||||
_ctx: *mut libc::c_void,
|
||||
) {
|
||||
let slot_ptr = crate::preempt::current_slot_ptr();
|
||||
if !slot_ptr.is_null() {
|
||||
let slot = &*slot_ptr;
|
||||
let top = slot.diag_stack_top.load(Ordering::Relaxed);
|
||||
if top != 0 {
|
||||
let reserve = slot.diag_stack_reserve.load(Ordering::Relaxed);
|
||||
let guard = slot.diag_stack_guard.load(Ordering::Relaxed);
|
||||
let pid = slot.diag_pid.load(Ordering::Relaxed);
|
||||
let addr = (*info).si_addr() as usize;
|
||||
match classify(addr, top, reserve, guard) {
|
||||
FaultClass::Guard => {
|
||||
let mut b = Buf::new();
|
||||
b.s("smarm: actor ");
|
||||
b.pid(pid);
|
||||
b.s(" overflowed its stack: fault in the guard region, depth-at-fault=");
|
||||
b.u(top - addr);
|
||||
b.s(" bytes (reserve=");
|
||||
b.u(reserve);
|
||||
b.s(", guard=");
|
||||
b.u(guard);
|
||||
b.s("). Raise stack_reserve (SpawnOpts or Config).\n");
|
||||
b.emit();
|
||||
die_by_default();
|
||||
return;
|
||||
}
|
||||
FaultClass::Overshoot(below) => {
|
||||
let mut b = Buf::new();
|
||||
b.s("smarm: actor ");
|
||||
b.pid(pid);
|
||||
b.s(" probably overflowed its stack: fault ");
|
||||
b.u(below);
|
||||
b.s(" bytes below the guard - an unprobed (FFI?) frame stepped over it (reserve=");
|
||||
b.u(reserve);
|
||||
b.s(", guard=");
|
||||
b.u(guard);
|
||||
b.s("). Raise stack_guard or stack_reserve.\n");
|
||||
b.emit();
|
||||
die_by_default();
|
||||
return;
|
||||
}
|
||||
FaultClass::Foreign => {}
|
||||
}
|
||||
}
|
||||
}
|
||||
// Not ours: put back whoever was there before us and refault into them.
|
||||
let prior = &*std::ptr::addr_of!(PRIOR);
|
||||
libc::sigaction(libc::SIGSEGV, prior.as_ptr(), std::ptr::null_mut());
|
||||
}
|
||||
|
||||
/// Reset SIGSEGV to default disposition; returning from the handler then
|
||||
/// refaults at the same instruction and the process dies the normal death
|
||||
/// (core-dumpable, correct wait status), exactly as if we were never here —
|
||||
/// but with the message already on stderr.
|
||||
unsafe fn die_by_default() {
|
||||
let mut dfl: libc::sigaction = std::mem::zeroed();
|
||||
dfl.sa_sigaction = libc::SIG_DFL;
|
||||
libc::sigemptyset(&mut dfl.sa_mask);
|
||||
libc::sigaction(libc::SIGSEGV, &dfl, std::ptr::null_mut());
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Async-signal-safe formatting: fixed buffer, decimal itoa, one write(2).
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
struct Buf {
|
||||
b: [u8; 320],
|
||||
len: usize,
|
||||
}
|
||||
|
||||
impl Buf {
|
||||
fn new() -> Self {
|
||||
Buf {
|
||||
b: [0; 320],
|
||||
len: 0,
|
||||
}
|
||||
}
|
||||
fn s(&mut self, s: &str) {
|
||||
for &c in s.as_bytes() {
|
||||
if self.len < self.b.len() {
|
||||
self.b[self.len] = c;
|
||||
self.len += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
fn u(&mut self, mut n: usize) {
|
||||
let mut tmp = [0u8; 20];
|
||||
let mut i = tmp.len();
|
||||
loop {
|
||||
i -= 1;
|
||||
tmp[i] = b'0' + (n % 10) as u8;
|
||||
n /= 10;
|
||||
if n == 0 {
|
||||
break;
|
||||
}
|
||||
}
|
||||
for &c in &tmp[i..] {
|
||||
if self.len < self.b.len() {
|
||||
self.b[self.len] = c;
|
||||
self.len += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
/// `idx.gen`, unpacked from the install-time packing.
|
||||
fn pid(&mut self, packed: u64) {
|
||||
self.u((packed >> 32) as usize);
|
||||
self.s(".");
|
||||
self.u((packed & 0xffff_ffff) as usize);
|
||||
}
|
||||
fn emit(&self) {
|
||||
unsafe {
|
||||
libc::write(2, self.b.as_ptr() as *const libc::c_void, self.len);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Classifier units — the arithmetic edges, before anything integrates.
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::{classify, FaultClass, OVERSHOOT_SLOP};
|
||||
|
||||
const PG: usize = 4096;
|
||||
// A synthetic stack far from address-space edges: top at 1 GiB.
|
||||
const TOP: usize = 1 << 30;
|
||||
const RESERVE: usize = 16 * PG;
|
||||
const GUARD: usize = 4 * PG;
|
||||
const GUARD_HI: usize = TOP - RESERVE;
|
||||
const GUARD_LO: usize = GUARD_HI - GUARD;
|
||||
|
||||
#[test]
|
||||
fn inside_guard_both_edges() {
|
||||
assert_eq!(classify(GUARD_LO, TOP, RESERVE, GUARD), FaultClass::Guard);
|
||||
assert_eq!(
|
||||
classify(GUARD_HI - 1, TOP, RESERVE, GUARD),
|
||||
FaultClass::Guard
|
||||
);
|
||||
assert_eq!(
|
||||
classify(GUARD_LO + GUARD / 2, TOP, RESERVE, GUARD),
|
||||
FaultClass::Guard
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn usable_region_is_foreign() {
|
||||
// A fault inside the RW stack itself isn't a guard hit and must not
|
||||
// be explained as one.
|
||||
assert_eq!(classify(GUARD_HI, TOP, RESERVE, GUARD), FaultClass::Foreign);
|
||||
assert_eq!(classify(TOP - 1, TOP, RESERVE, GUARD), FaultClass::Foreign);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn above_top_is_foreign() {
|
||||
assert_eq!(classify(TOP, TOP, RESERVE, GUARD), FaultClass::Foreign);
|
||||
assert_eq!(classify(TOP + PG, TOP, RESERVE, GUARD), FaultClass::Foreign);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn overshoot_window_edges() {
|
||||
assert_eq!(
|
||||
classify(GUARD_LO - 1, TOP, RESERVE, GUARD),
|
||||
FaultClass::Overshoot(1)
|
||||
);
|
||||
assert_eq!(
|
||||
classify(GUARD_LO - OVERSHOOT_SLOP, TOP, RESERVE, GUARD),
|
||||
FaultClass::Overshoot(OVERSHOOT_SLOP)
|
||||
);
|
||||
assert_eq!(
|
||||
classify(GUARD_LO - OVERSHOOT_SLOP - 1, TOP, RESERVE, GUARD),
|
||||
FaultClass::Foreign
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn low_address_stack_saturates_not_wraps() {
|
||||
// A stack mapped so low that the slop window would underflow: the
|
||||
// window clips to 0 instead of wrapping around the address space.
|
||||
let top = RESERVE + GUARD + PG; // guard_lo == PG
|
||||
assert_eq!(classify(0, top, RESERVE, GUARD), FaultClass::Overshoot(PG));
|
||||
// Null-page fault still classified only because it IS within slop
|
||||
// here; with a normal-height stack it is Foreign (covered above by
|
||||
// the window-edge test at realistic addresses).
|
||||
}
|
||||
}
|
||||
@@ -1,624 +0,0 @@
|
||||
//! The per-slot scheduling state machine, as a standalone unit.
|
||||
//!
|
||||
//! One atomic word packs `(generation << 32) | (epoch << 8) | state`; every
|
||||
//! transition is a CAS on the packed word, so the generation check is atomic
|
||||
//! with the transition — no ABA, no acting on a recycled slot. The diagram
|
||||
//! and the full protocol rationale live in `runtime.rs`; this module is the
|
||||
//! mechanism, factored out so that:
|
||||
//!
|
||||
//! - loom can model-check the production transitions directly (see the
|
||||
//! `loom_tests` module; built with `RUSTFLAGS="--cfg loom"`), and
|
||||
//! - every method asserts the precondition it relies on (`debug_assert!` —
|
||||
//! these are hot paths), per the assert-the-invariants house rule.
|
||||
//!
|
||||
//! ## The park-epoch (wait identity)
|
||||
//!
|
||||
//! The middle 24 bits carry the slot's *park-epoch*: the identity of the
|
||||
//! actor's current (or most recent) wait. The rules:
|
||||
//!
|
||||
//! - [`begin_wait`](StateWord::begin_wait) bumps the epoch and returns it;
|
||||
//! the actor calls it once per wait, *before* registering itself with any
|
||||
//! waker. Registrations carry `(pid, epoch)`.
|
||||
//! - A wake may be **epoch-matched** (`unpark(gen, Some(epoch))`): it lands
|
||||
//! only if the word still carries that epoch. Wakers whose registration
|
||||
//! handle can outlive the wait it was created for (channel senders, mutex
|
||||
//! grants, wait-timers) MUST use this form.
|
||||
//! - Every successful wake **consumes** the epoch — `Parked(e) → Queued(e+1)`,
|
||||
//! `Running(e) → RunningNotified(e+1)` — so at most one wake can ever land
|
||||
//! per wait, by construction. A loser in a multi-waker race (e.g. the
|
||||
//! non-winning arms of a `select`) fails the epoch check and no-ops; it can
|
||||
//! neither steal a future wait's wake nor leave a pending notification that
|
||||
//! would fault a later one-shot park (`Mutex::lock_timeout`, `sleep`,
|
||||
//! `block_on_io`, `wait_fd` all rely on wakes being *meaningful*).
|
||||
//! - The wildcard form (`unpark(gen, None)`) also consumes, and is reserved
|
||||
//! for terminal wakes — `request_stop` — which never return control to the
|
||||
//! code that parked.
|
||||
//!
|
||||
//! Epoch wrap (24 bits = 16.7M waits) is harmless: a collision would require
|
||||
//! a taken registration to stay in flight across a full wrap of the *same
|
||||
//! actor's* waits, and registrations are consumed at take-time under their
|
||||
//! primitive's lock — the exposure is the taker's instruction window.
|
||||
//!
|
||||
//! Atomics come from `sync_shim` (std normally, `loom::sync` under
|
||||
//! `cfg(loom)`).
|
||||
|
||||
use crate::sync_shim::{AtomicU64, Ordering};
|
||||
|
||||
pub(crate) const ST_VACANT: u64 = 0;
|
||||
pub(crate) const ST_QUEUED: u64 = 1;
|
||||
pub(crate) const ST_RUNNING: u64 = 2;
|
||||
pub(crate) const ST_RUNNING_NOTIFIED: u64 = 3;
|
||||
pub(crate) const ST_PARKED: u64 = 4;
|
||||
pub(crate) const ST_DONE: u64 = 5;
|
||||
|
||||
/// Park-epoch width: 24 bits, packed at word bits 8..32.
|
||||
pub(crate) const EPOCH_MASK: u32 = 0x00FF_FFFF;
|
||||
|
||||
#[inline]
|
||||
pub(crate) const fn pack(gen: u32, epoch: u32, st: u64) -> u64 {
|
||||
debug_assert!(epoch & !EPOCH_MASK == 0);
|
||||
((gen as u64) << 32) | ((epoch as u64) << 8) | st
|
||||
}
|
||||
#[inline]
|
||||
pub(crate) const fn word_gen(w: u64) -> u32 {
|
||||
(w >> 32) as u32
|
||||
}
|
||||
#[inline]
|
||||
pub(crate) const fn word_epoch(w: u64) -> u32 {
|
||||
((w >> 8) as u32) & EPOCH_MASK
|
||||
}
|
||||
#[inline]
|
||||
pub(crate) const fn word_state(w: u64) -> u64 {
|
||||
w & 0xFF
|
||||
}
|
||||
|
||||
/// What an unpark amounted to. The caller owns the side effects (enqueue,
|
||||
/// trace events) — this module is pure state.
|
||||
#[derive(Debug, PartialEq, Eq, Clone, Copy)]
|
||||
pub(crate) enum Unpark {
|
||||
/// Parked → Queued: the caller must enqueue the pid.
|
||||
Enqueue,
|
||||
/// Running → RunningNotified: the scheduler's park-return will re-queue.
|
||||
Notified,
|
||||
/// Stale generation, stale epoch, already queued/notified, done, or
|
||||
/// vacant.
|
||||
Noop,
|
||||
}
|
||||
|
||||
/// A pid's-eye view of the slot, for cold paths that hold the slot lock.
|
||||
#[derive(Debug, PartialEq, Eq, Clone, Copy)]
|
||||
pub(crate) enum Status {
|
||||
/// The generation no longer matches: the slot was reclaimed (and possibly
|
||||
/// reused) — the pid is stale.
|
||||
Stale,
|
||||
/// The actor terminated; its outcome is (or was) in the slot.
|
||||
Done,
|
||||
/// Alive in some scheduling state (Queued / Running / Notified / Parked).
|
||||
Live,
|
||||
}
|
||||
|
||||
pub(crate) struct StateWord(AtomicU64);
|
||||
|
||||
impl StateWord {
|
||||
pub(crate) fn new() -> Self {
|
||||
Self(AtomicU64::new(pack(0, 0, ST_VACANT)))
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn load(&self) -> u64 {
|
||||
self.0.load(Ordering::Acquire)
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn generation(&self) -> u32 {
|
||||
word_gen(self.load())
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn status_for(&self, gen: u32) -> Status {
|
||||
let w = self.load();
|
||||
if word_gen(w) != gen {
|
||||
return Status::Stale;
|
||||
}
|
||||
match word_state(w) {
|
||||
ST_DONE => Status::Done,
|
||||
// A matching generation on a Vacant slot is unreachable for any
|
||||
// ISSUED pid — reclaim bumps the generation in the very store
|
||||
// that vacates, and install publishes Queued before the pid
|
||||
// escapes. But `Pid::new` is public, so a forged / never-issued
|
||||
// pid (e.g. `Pid::new(5, 0)` against a fresh slab) can land
|
||||
// here; for those, "no such actor" is the correct total answer.
|
||||
ST_VACANT => Status::Stale,
|
||||
_ => Status::Live,
|
||||
}
|
||||
}
|
||||
|
||||
/// Spawn-side publish: Vacant → Queued. The caller owns the vacant slot
|
||||
/// exclusively (it popped the index from the free list), so this is a
|
||||
/// plain Release store; it is the moment the actor becomes visible to
|
||||
/// pops, unparks, and stops. The epoch starts at 0 for each occupancy
|
||||
/// (`set_done` zeroes it; wait identity never crosses a lifetime).
|
||||
pub(crate) fn publish_queued(&self, gen: u32) {
|
||||
debug_assert_eq!(
|
||||
self.load(),
|
||||
pack(gen, 0, ST_VACANT),
|
||||
"publish over a non-vacant slot"
|
||||
);
|
||||
self.0.store(pack(gen, 0, ST_QUEUED), Ordering::Release);
|
||||
}
|
||||
|
||||
/// Scheduler pop-side claim: Queued → Running, epoch preserved. `false`
|
||||
/// means the popped pid is stale — by the at-most-once-enqueued
|
||||
/// invariant, a generation mismatch is the only possible failure
|
||||
/// (asserted). Nothing can move a matching-gen word off Queued (wakes
|
||||
/// no-op on Queued), so the CAS loop is single-shot in practice.
|
||||
#[must_use]
|
||||
pub(crate) fn try_claim(&self, gen: u32) -> bool {
|
||||
loop {
|
||||
let w = self.load();
|
||||
if word_gen(w) != gen {
|
||||
return false;
|
||||
}
|
||||
debug_assert_eq!(
|
||||
word_state(w),
|
||||
ST_QUEUED,
|
||||
"queued pid found in unexpected state {} — double enqueue?",
|
||||
word_state(w)
|
||||
);
|
||||
if self
|
||||
.0
|
||||
.compare_exchange(
|
||||
w,
|
||||
pack(gen, word_epoch(w), ST_RUNNING),
|
||||
Ordering::AcqRel,
|
||||
Ordering::Acquire,
|
||||
)
|
||||
.is_ok()
|
||||
{
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Yield return path: Running | RunningNotified → Queued, epoch
|
||||
/// preserved. A notification that arrived mid-run coalesces into the
|
||||
/// re-queue. Caller must enqueue. CAS loop because a notify can bump the
|
||||
/// epoch between the read and the exchange.
|
||||
pub(crate) fn yield_return(&self, gen: u32) {
|
||||
loop {
|
||||
let w = self.load();
|
||||
debug_assert!(
|
||||
matches!(word_state(w), ST_RUNNING | ST_RUNNING_NOTIFIED) && word_gen(w) == gen,
|
||||
"yield return from invalid word {w:#x}"
|
||||
);
|
||||
if self
|
||||
.0
|
||||
.compare_exchange(
|
||||
w,
|
||||
pack(gen, word_epoch(w), ST_QUEUED),
|
||||
Ordering::AcqRel,
|
||||
Ordering::Acquire,
|
||||
)
|
||||
.is_ok()
|
||||
{
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Park return path. `true` = actually parked. `false` = an unpark landed
|
||||
/// in the prep-to-park window (RunningNotified); the word is already back
|
||||
/// to Queued and the caller must enqueue — the lost-wakeup window,
|
||||
/// closed. Epoch preserved on both paths (the notify already consumed
|
||||
/// it).
|
||||
#[must_use]
|
||||
pub(crate) fn park_return(&self, gen: u32) -> bool {
|
||||
loop {
|
||||
let w = self.load();
|
||||
debug_assert_eq!(word_gen(w), gen, "park return with stale gen");
|
||||
let target = match word_state(w) {
|
||||
ST_RUNNING => ST_PARKED,
|
||||
ST_RUNNING_NOTIFIED => ST_QUEUED,
|
||||
st => unreachable!("park return from invalid state {st}"),
|
||||
};
|
||||
if self
|
||||
.0
|
||||
.compare_exchange(
|
||||
w,
|
||||
pack(gen, word_epoch(w), target),
|
||||
Ordering::AcqRel,
|
||||
Ordering::Acquire,
|
||||
)
|
||||
.is_ok()
|
||||
{
|
||||
return target == ST_PARKED;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Open a new wait: bump the park-epoch and return it. Called by the
|
||||
/// waiting actor itself (so the state is Running, or RunningNotified if
|
||||
/// a terminal wake is already pending — the bump preserves the pending
|
||||
/// notification), once per wait, BEFORE registering `(pid, epoch)` with
|
||||
/// any waker.
|
||||
#[must_use]
|
||||
pub(crate) fn begin_wait(&self, gen: u32) -> u32 {
|
||||
loop {
|
||||
let w = self.load();
|
||||
debug_assert!(
|
||||
matches!(word_state(w), ST_RUNNING | ST_RUNNING_NOTIFIED) && word_gen(w) == gen,
|
||||
"begin_wait from invalid word {w:#x}"
|
||||
);
|
||||
let next = word_epoch(w).wrapping_add(1) & EPOCH_MASK;
|
||||
if self
|
||||
.0
|
||||
.compare_exchange(
|
||||
w,
|
||||
pack(gen, next, word_state(w)),
|
||||
Ordering::AcqRel,
|
||||
Ordering::Acquire,
|
||||
)
|
||||
.is_ok()
|
||||
{
|
||||
return next;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// The unpark protocol — the one way anything outside the scheduler makes
|
||||
/// an actor runnable. See [`Unpark`] for the caller's obligations.
|
||||
///
|
||||
/// `want = Some(epoch)` is the epoch-matched form: lands only if the word
|
||||
/// still carries that epoch (i.e. the wait it was registered for is still
|
||||
/// the current, un-woken wait). `want = None` is the wildcard, reserved
|
||||
/// for terminal wakes. Both forms CONSUME the epoch on success.
|
||||
#[must_use]
|
||||
pub(crate) fn unpark(&self, gen: u32, want: Option<u32>) -> Unpark {
|
||||
loop {
|
||||
let w = self.load();
|
||||
if word_gen(w) != gen {
|
||||
return Unpark::Noop;
|
||||
}
|
||||
if let Some(e) = want {
|
||||
if word_epoch(w) != e {
|
||||
return Unpark::Noop;
|
||||
}
|
||||
}
|
||||
let bumped = word_epoch(w).wrapping_add(1) & EPOCH_MASK;
|
||||
match word_state(w) {
|
||||
ST_PARKED => {
|
||||
if self
|
||||
.0
|
||||
.compare_exchange(
|
||||
w,
|
||||
pack(gen, bumped, ST_QUEUED),
|
||||
Ordering::AcqRel,
|
||||
Ordering::Acquire,
|
||||
)
|
||||
.is_ok()
|
||||
{
|
||||
return Unpark::Enqueue;
|
||||
}
|
||||
}
|
||||
ST_RUNNING => {
|
||||
if self
|
||||
.0
|
||||
.compare_exchange(
|
||||
w,
|
||||
pack(gen, bumped, ST_RUNNING_NOTIFIED),
|
||||
Ordering::AcqRel,
|
||||
Ordering::Acquire,
|
||||
)
|
||||
.is_ok()
|
||||
{
|
||||
return Unpark::Notified;
|
||||
}
|
||||
}
|
||||
_ => return Unpark::Noop, // Queued | Notified | Done | Vacant
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Eat a pending notification: RunningNotified → Running, epoch
|
||||
/// preserved; no-op on Running. Called by the RUNNING actor itself, on
|
||||
/// the no-park exit of a wait it registered for but never parked on
|
||||
/// (`select` returning a ready arm at registration time), AFTER bumping
|
||||
/// the epoch and BEFORE re-checking its stop flag:
|
||||
///
|
||||
/// - post-bump, the only wakers that can have set RunningNotified are
|
||||
/// ones stamped with the just-retired epoch (a select arm) or a
|
||||
/// terminal wildcard (`request_stop`);
|
||||
/// - the caller's stop-flag check AFTER the clear catches the terminal
|
||||
/// case (the flag is set before the wake fires), so eating its
|
||||
/// notification loses nothing — and a stop arriving later re-notifies
|
||||
/// a Running word as usual;
|
||||
/// - what remains eaten is exactly the stale arm wake that would
|
||||
/// otherwise fault the actor's next one-shot park.
|
||||
///
|
||||
/// Returns whether a notification was eaten.
|
||||
pub(crate) fn clear_notify(&self, gen: u32) -> bool {
|
||||
loop {
|
||||
let w = self.load();
|
||||
debug_assert!(
|
||||
matches!(word_state(w), ST_RUNNING | ST_RUNNING_NOTIFIED) && word_gen(w) == gen,
|
||||
"clear_notify from invalid word {w:#x}"
|
||||
);
|
||||
if word_state(w) != ST_RUNNING_NOTIFIED {
|
||||
return false;
|
||||
}
|
||||
if self
|
||||
.0
|
||||
.compare_exchange(
|
||||
w,
|
||||
pack(gen, word_epoch(w), ST_RUNNING),
|
||||
Ordering::AcqRel,
|
||||
Ordering::Acquire,
|
||||
)
|
||||
.is_ok()
|
||||
{
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Finalize: Running | RunningNotified → Done, epoch zeroed (wait
|
||||
/// identity never crosses an occupancy). Called by the scheduler that
|
||||
/// just ran the actor to completion (so those are the only legal prior
|
||||
/// states), under the slot's cold lock so join's check-or-register is
|
||||
/// linearized against it.
|
||||
pub(crate) fn set_done(&self, gen: u32) {
|
||||
let prev = self.0.swap(pack(gen, 0, ST_DONE), Ordering::AcqRel);
|
||||
debug_assert!(
|
||||
matches!(word_state(prev), ST_RUNNING | ST_RUNNING_NOTIFIED) && word_gen(prev) == gen,
|
||||
"finalize from invalid word {prev:#x}"
|
||||
);
|
||||
}
|
||||
|
||||
/// Reclaim: Done → Vacant(gen + 1). The generation bump IS the reclaim:
|
||||
/// every stale pid is dead from this store onwards. Caller holds the cold
|
||||
/// lock and has verified eligibility (asserted).
|
||||
pub(crate) fn reclaim(&self, gen: u32) {
|
||||
debug_assert_eq!(
|
||||
self.load(),
|
||||
pack(gen, 0, ST_DONE),
|
||||
"reclaim of a non-Done slot"
|
||||
);
|
||||
self.0
|
||||
.store(pack(gen.wrapping_add(1), 0, ST_VACANT), Ordering::Release);
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// loom model tests — RUSTFLAGS="--cfg loom" cargo test --lib --release
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
#[cfg(all(test, loom))]
|
||||
mod loom_tests {
|
||||
use super::*;
|
||||
use loom::sync::atomic::{AtomicBool, AtomicUsize};
|
||||
use loom::sync::Arc;
|
||||
use loom::thread;
|
||||
use std::sync::atomic::Ordering as O;
|
||||
|
||||
/// THE lost-wakeup theorem. A waiter registers a condition check then
|
||||
/// parks (as every parking site does); a waker sets the condition then
|
||||
/// unparks. In every interleaving the waiter must end up runnable —
|
||||
/// parked-forever-with-condition-set must be unreachable.
|
||||
#[test]
|
||||
fn no_lost_wakeup_park_vs_unpark() {
|
||||
loom::model(|| {
|
||||
let word = Arc::new(StateWord::new());
|
||||
word.publish_queued(0);
|
||||
assert!(word.try_claim(0)); // scheduler claimed: actor Running
|
||||
let epoch = word.begin_wait(0); // actor opens the wait
|
||||
|
||||
let ready = Arc::new(AtomicBool::new(false));
|
||||
let enqueues = Arc::new(AtomicUsize::new(0));
|
||||
|
||||
// Waker: make the condition true, then wake the registered wait.
|
||||
let w = word.clone();
|
||||
let r = ready.clone();
|
||||
let e = enqueues.clone();
|
||||
let waker = thread::spawn(move || {
|
||||
r.store(true, O::SeqCst);
|
||||
if w.unpark(0, Some(epoch)) == Unpark::Enqueue {
|
||||
e.fetch_add(1, O::SeqCst);
|
||||
}
|
||||
});
|
||||
|
||||
// Waiter (as the scheduler executes it): re-check the condition,
|
||||
// park only if still false; a Notified park-return re-queues.
|
||||
let parked = if ready.load(O::SeqCst) {
|
||||
false // condition already visible: doesn't park at all
|
||||
} else if word.park_return(0) {
|
||||
true
|
||||
} else {
|
||||
enqueues.fetch_add(1, O::SeqCst); // notified → re-queued
|
||||
false
|
||||
};
|
||||
|
||||
waker.join().unwrap();
|
||||
|
||||
let w = word.load();
|
||||
if parked {
|
||||
// Parked is only a FINAL state if the waker's unpark moved it
|
||||
// back to Queued (+ one enqueue). Parked-and-stays-parked
|
||||
// would be the lost wakeup.
|
||||
assert_eq!(word_state(w), ST_QUEUED, "lost wakeup: parked forever");
|
||||
assert_eq!(enqueues.load(O::SeqCst), 1);
|
||||
} else {
|
||||
// Never more than one enqueue (at-most-once-enqueued).
|
||||
assert!(enqueues.load(O::SeqCst) <= 1);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/// Two concurrent unparkers, one parked actor: exactly one wins the
|
||||
/// enqueue (at-most-once), regardless of interleaving. Both stamped with
|
||||
/// the live epoch — the consuming bump is what serializes them.
|
||||
#[test]
|
||||
fn two_unparkers_one_enqueue() {
|
||||
loom::model(|| {
|
||||
let word = Arc::new(StateWord::new());
|
||||
word.publish_queued(0);
|
||||
assert!(word.try_claim(0));
|
||||
let epoch = word.begin_wait(0);
|
||||
assert!(word.park_return(0)); // actor parked
|
||||
|
||||
let enqueues = Arc::new(AtomicUsize::new(0));
|
||||
let mut hs = Vec::new();
|
||||
for _ in 0..2 {
|
||||
let w = word.clone();
|
||||
let e = enqueues.clone();
|
||||
hs.push(thread::spawn(move || {
|
||||
if w.unpark(0, Some(epoch)) == Unpark::Enqueue {
|
||||
e.fetch_add(1, O::SeqCst);
|
||||
}
|
||||
}));
|
||||
}
|
||||
for h in hs {
|
||||
h.join().unwrap();
|
||||
}
|
||||
assert_eq!(enqueues.load(O::SeqCst), 1);
|
||||
assert_eq!(word_state(word.load()), ST_QUEUED);
|
||||
});
|
||||
}
|
||||
|
||||
/// The stale-epoch theorem — what `select`'s loser arms lean on. An
|
||||
/// actor opens a wait, two registered wakers race it (against the park
|
||||
/// itself, covering the prep-to-park window); afterwards the actor is
|
||||
/// runnable exactly once, and a LATE waker still stamped with the
|
||||
/// consumed epoch can neither enqueue nor notify — in every
|
||||
/// interleaving. (Under wildcard semantics the late waker would corrupt
|
||||
/// the actor's NEXT one-shot park; this is the theorem that buys
|
||||
/// `Mutex::lock_timeout`/`sleep`/`block_on_io` their unchanged code.)
|
||||
#[test]
|
||||
fn consumed_epoch_unpark_never_lands() {
|
||||
loom::model(|| {
|
||||
let word = Arc::new(StateWord::new());
|
||||
word.publish_queued(0);
|
||||
assert!(word.try_claim(0));
|
||||
let epoch = word.begin_wait(0);
|
||||
|
||||
// Two arms race the wake, concurrent with the park itself.
|
||||
let enqueues = Arc::new(AtomicUsize::new(0));
|
||||
let mut hs = Vec::new();
|
||||
for _ in 0..2 {
|
||||
let w = word.clone();
|
||||
let e = enqueues.clone();
|
||||
hs.push(thread::spawn(move || {
|
||||
if w.unpark(0, Some(epoch)) == Unpark::Enqueue {
|
||||
e.fetch_add(1, O::SeqCst);
|
||||
}
|
||||
}));
|
||||
}
|
||||
let mut runnable_via_notify = false;
|
||||
if !word.park_return(0) {
|
||||
runnable_via_notify = true; // notified in prep-to-park
|
||||
}
|
||||
for h in hs {
|
||||
h.join().unwrap();
|
||||
}
|
||||
|
||||
// Exactly one path made the actor runnable.
|
||||
let direct = enqueues.load(O::SeqCst);
|
||||
if runnable_via_notify {
|
||||
assert_eq!(direct, 0, "woken twice: notify AND enqueue");
|
||||
} else {
|
||||
assert_eq!(direct, 1, "parked forever, or woken twice");
|
||||
}
|
||||
assert_eq!(word_state(word.load()), ST_QUEUED);
|
||||
|
||||
// The actor runs again. A waker still holding the OLD epoch —
|
||||
// a select loser arm firing later — must be a strict no-op,
|
||||
// not a pending notification.
|
||||
assert!(word.try_claim(0));
|
||||
assert_eq!(word.unpark(0, Some(epoch)), Unpark::Noop);
|
||||
assert_eq!(
|
||||
word_state(word.load()),
|
||||
ST_RUNNING,
|
||||
"stale epoch notified a live run"
|
||||
);
|
||||
});
|
||||
}
|
||||
|
||||
/// The retire theorem — `select`'s no-park exit. An actor opens a wait
|
||||
/// and registers, then finds an arm ready and returns WITHOUT parking;
|
||||
/// a loser arm's waker fires concurrently, stamped with the live epoch.
|
||||
/// The exit retires the wait (bump, then eat): in every interleaving
|
||||
/// the run ends on a clean Running word — no pending notification
|
||||
/// survives to fault the actor's next one-shot park — and the waker
|
||||
/// never enqueues.
|
||||
#[test]
|
||||
fn retire_eats_late_arm_notification() {
|
||||
loom::model(|| {
|
||||
let word = Arc::new(StateWord::new());
|
||||
word.publish_queued(0);
|
||||
assert!(word.try_claim(0));
|
||||
let epoch = word.begin_wait(0); // select opens + registers
|
||||
|
||||
let w = word.clone();
|
||||
let waker = thread::spawn(move || w.unpark(0, Some(epoch)));
|
||||
|
||||
// No-park exit: bump (invalidates in-flight wakes), then eat
|
||||
// (consumes one that already landed).
|
||||
let _ = word.begin_wait(0);
|
||||
word.clear_notify(0);
|
||||
|
||||
assert_ne!(waker.join().unwrap(), Unpark::Enqueue);
|
||||
assert_eq!(
|
||||
word_state(word.load()),
|
||||
ST_RUNNING,
|
||||
"stale arm wake survived the retire"
|
||||
);
|
||||
});
|
||||
}
|
||||
|
||||
/// The ABA theorem: a stale-generation unpark racing reclaim + reuse can
|
||||
/// never touch the slot's new occupant.
|
||||
#[test]
|
||||
fn stale_unpark_never_hits_reused_slot() {
|
||||
loom::model(|| {
|
||||
let word = Arc::new(StateWord::new());
|
||||
// Gen-0 actor runs to completion.
|
||||
word.publish_queued(0);
|
||||
assert!(word.try_claim(0));
|
||||
|
||||
let w = word.clone();
|
||||
let stale = thread::spawn(move || w.unpark(0, None));
|
||||
|
||||
// Scheduler: finalize, reclaim, and a new spawn reuses the slot.
|
||||
word.set_done(0);
|
||||
word.reclaim(0);
|
||||
word.publish_queued(1);
|
||||
|
||||
// The stale unpark may have squeezed in only while gen 0 was
|
||||
// still Running (→ Notified) — in which case set_done's swap
|
||||
// absorbed it — or it observed Done/Vacant/gen-1 and no-op'd.
|
||||
// Either way it must never claim an enqueue.
|
||||
assert_ne!(stale.join().unwrap(), Unpark::Enqueue);
|
||||
// And the new occupant is exactly where its spawn put it.
|
||||
assert_eq!(word.load(), pack(1, 0, ST_QUEUED));
|
||||
});
|
||||
}
|
||||
|
||||
/// Unpark racing the claim itself: whatever the interleaving, the actor
|
||||
/// is Running or RunningNotified afterwards and nobody enqueued (it was
|
||||
/// never parked).
|
||||
#[test]
|
||||
fn unpark_vs_claim_coalesces() {
|
||||
loom::model(|| {
|
||||
let word = Arc::new(StateWord::new());
|
||||
word.publish_queued(0);
|
||||
|
||||
let w = word.clone();
|
||||
let unparker = thread::spawn(move || w.unpark(0, None));
|
||||
|
||||
assert!(word.try_claim(0)); // the entry is ours; claim must win
|
||||
let r = unparker.join().unwrap();
|
||||
assert_ne!(r, Unpark::Enqueue);
|
||||
let st = word_state(word.load());
|
||||
assert!(matches!(st, ST_RUNNING | ST_RUNNING_NOTIFIED));
|
||||
});
|
||||
}
|
||||
}
|
||||
+17
-237
@@ -1,45 +1,32 @@
|
||||
//! mmap-based actor stack with a PROT_NONE guard region below (RFC 019).
|
||||
//! mmap-based growable stack with a guard page below.
|
||||
//!
|
||||
//! Layout (low → high address):
|
||||
//! [ guard region (PROT_NONE) | stack region ]
|
||||
//! ^ top() — initial stack pointer
|
||||
//! [ guard page (PROT_NONE) | stack region ]
|
||||
//! ^ top() — initial stack pointer
|
||||
//!
|
||||
//! Stacks grow downward. Overflow lands in the guard region → SIGSEGV.
|
||||
//!
|
||||
//! Both the usable reserve and the guard are caller-chosen (page-rounded).
|
||||
//! The reserve is a *virtual* reservation: anonymous mmap is demand-paged,
|
||||
//! so RSS is touched-pages, not reserve × actors. The guard costs address
|
||||
//! space only. A wide guard (the runtime defaults to 64 KiB) exists for
|
||||
//! unprobed FFI frames: Rust frames touch pages in order (probestack), so
|
||||
//! one page catches Rust overflow, but a C frame with a large local can
|
||||
//! step over a single page in one `sub rsp`.
|
||||
//! Stacks grow downward. Overflow lands in the guard page → SIGSEGV.
|
||||
|
||||
use std::io;
|
||||
|
||||
pub struct Stack {
|
||||
/// Bottom of the entire mmap'd region (start of the guard).
|
||||
/// Bottom of the entire mmap'd region (start of guard page).
|
||||
base: *mut u8,
|
||||
/// Total mmap'd size: guard_size + stack_size.
|
||||
total_size: usize,
|
||||
/// Usable stack size (excluding the guard).
|
||||
/// Usable stack size (excluding guard page).
|
||||
stack_size: usize,
|
||||
/// PROT_NONE region below the usable stack.
|
||||
guard_size: usize,
|
||||
}
|
||||
|
||||
// Stack owns its memory; safe to send across threads.
|
||||
unsafe impl Send for Stack {}
|
||||
|
||||
impl Stack {
|
||||
/// Allocate a new stack. `stack_size` is the usable region; `guard_size`
|
||||
/// is mapped PROT_NONE below it. Both are rounded up to the page size
|
||||
/// and must be non-zero.
|
||||
pub fn new(stack_size: usize, guard_size: usize) -> io::Result<Self> {
|
||||
assert!(stack_size > 0, "stack_size must be non-zero");
|
||||
assert!(guard_size > 0, "guard_size must be non-zero");
|
||||
/// Allocate a new stack. `stack_size` is the usable region; one page is
|
||||
/// added below as a guard page. Both are rounded up to the page size.
|
||||
pub fn new(stack_size: usize) -> io::Result<Self> {
|
||||
let page = page_size();
|
||||
let stack_size = round_up(stack_size, page);
|
||||
let guard_size = round_up(guard_size, page);
|
||||
let guard_size = page;
|
||||
let total_size = guard_size + stack_size;
|
||||
|
||||
let base = unsafe {
|
||||
@@ -57,19 +44,16 @@ impl Stack {
|
||||
}
|
||||
let base = base as *mut u8;
|
||||
|
||||
let ret = unsafe { libc::mprotect(base as *mut libc::c_void, guard_size, libc::PROT_NONE) };
|
||||
let ret = unsafe {
|
||||
libc::mprotect(base as *mut libc::c_void, guard_size, libc::PROT_NONE)
|
||||
};
|
||||
if ret != 0 {
|
||||
let err = io::Error::last_os_error();
|
||||
unsafe { libc::munmap(base as *mut libc::c_void, total_size) };
|
||||
return Err(err);
|
||||
}
|
||||
|
||||
Ok(Self {
|
||||
base,
|
||||
total_size,
|
||||
stack_size,
|
||||
guard_size,
|
||||
})
|
||||
Ok(Self { base, total_size, stack_size })
|
||||
}
|
||||
|
||||
/// 16-byte-aligned top of the usable region.
|
||||
@@ -78,54 +62,14 @@ impl Stack {
|
||||
(raw_top & !15) as *mut u8
|
||||
}
|
||||
|
||||
/// Pointer to the bottom of the usable region (just above the guard).
|
||||
/// Pointer to the bottom of the usable region (just above the guard page).
|
||||
pub fn usable_base(&self) -> *mut u8 {
|
||||
unsafe { self.base.add(self.guard_size) }
|
||||
unsafe { self.base.add(page_size()) }
|
||||
}
|
||||
|
||||
pub fn stack_size(&self) -> usize {
|
||||
self.stack_size
|
||||
}
|
||||
|
||||
pub fn guard_size(&self) -> usize {
|
||||
self.guard_size
|
||||
}
|
||||
|
||||
/// `(stack_size, guard_size)` after page rounding. The pool rule
|
||||
/// (RFC 019 §1) compares this against the runtime defaults: only
|
||||
/// default-shaped stacks are pooled.
|
||||
pub fn shape(&self) -> (usize, usize) {
|
||||
(self.stack_size, self.guard_size)
|
||||
}
|
||||
|
||||
/// Pool-recycle zap (RFC 019 §6): `MADV_DONTNEED` everything below the
|
||||
/// retained entry end `[top − retain, top)` — the span the next actor's
|
||||
/// shallow frames land in stays resident, the dead spike below it is
|
||||
/// released. The stack is unowned at the call site (its actor is dead),
|
||||
/// so a synchronous eager zap races nothing and the RSS drop is
|
||||
/// immediate — a museum of worst-case spikes is exactly what a pool must
|
||||
/// not be; DONTNEED's ~8× per-page cost vs FREE is irrelevant off the
|
||||
/// hot path. Advisory like the park-path shrink: a failure degrades to
|
||||
/// "the pool keeps RSS", never to incorrectness. No-op (no syscall) when
|
||||
/// `retain` covers the whole usable region — i.e. always, at the 64 KiB
|
||||
/// default reserve.
|
||||
pub(crate) fn recycle_zap(&self, retain: usize) {
|
||||
if let Some((off, len)) = retain_range(self.stack_size, retain, page_size()) {
|
||||
unsafe {
|
||||
libc::madvise(
|
||||
self.usable_base().add(off) as *mut libc::c_void,
|
||||
len,
|
||||
libc::MADV_DONTNEED,
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Round `n` up to whole pages — the same rounding `Stack::new` applies, so
|
||||
/// runtime defaults stored pre-rounded compare exactly against [`Stack::shape`].
|
||||
pub(crate) fn round_to_pages(n: usize) -> usize {
|
||||
round_up(n, page_size())
|
||||
}
|
||||
|
||||
impl Drop for Stack {
|
||||
@@ -136,174 +80,10 @@ impl Drop for Stack {
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn page_size() -> usize {
|
||||
fn page_size() -> usize {
|
||||
unsafe { libc::sysconf(libc::_SC_PAGESIZE) as usize }
|
||||
}
|
||||
|
||||
fn round_up(n: usize, align: usize) -> usize {
|
||||
(n + align - 1) & !(align - 1)
|
||||
}
|
||||
|
||||
/// The whole-page span the park-path shrink may `MADV_FREE` (RFC 019 §3):
|
||||
/// `[page_up(hwm), page_down(sp − redzone))`, or `None` if no full page fits.
|
||||
///
|
||||
/// `hwm` is the sampled high-water (deepest observed `sp`); everything in
|
||||
/// `[hwm, sp)` is below the live frame and dead by definition. One page of
|
||||
/// redzone stays resident under live `sp` — it covers the SysV 128-byte red
|
||||
/// zone plus spill margin with room to spare. Rounding is inward on both
|
||||
/// ends so the result can never touch the redzone, cross `sp`, or dip below
|
||||
/// `hwm`; all arithmetic is checked so adversarial inputs (`sp < redzone`,
|
||||
/// `hwm ≥ sp`, values near the address-space edges) collapse to `None`
|
||||
/// rather than a wild or negative-length range.
|
||||
pub(crate) fn shrink_range(hwm: usize, sp: usize, page: usize) -> Option<(usize, usize)> {
|
||||
debug_assert!(page.is_power_of_two());
|
||||
if hwm >= sp {
|
||||
return None;
|
||||
}
|
||||
let redzone = page;
|
||||
let end = sp.checked_sub(redzone)? & !(page - 1); // page_down(sp − redzone)
|
||||
let start = hwm.checked_add(page - 1)? & !(page - 1); // page_up(hwm)
|
||||
if end > start {
|
||||
Some((start, end - start))
|
||||
} else {
|
||||
None
|
||||
}
|
||||
}
|
||||
|
||||
/// The `(offset_from_usable_base, len)` span the pool recycle DONTNEEDs
|
||||
/// (RFC 019 §6): everything below the retained entry end. "Bottom RETAIN of
|
||||
/// the stack" is read stack-wise (entry frames = highest addresses of a
|
||||
/// downward stack): the retained span is `[top − page_up(retain), top)`, the
|
||||
/// zapped span is the rest — retaining the low-address deep end instead
|
||||
/// would keep the coldest pages and release the ones the next actor faults
|
||||
/// first. `retain` rounds *up* to whole pages (retain more, zap less), so
|
||||
/// with `stack_size` page-rounded by `Stack::new` the result is always
|
||||
/// page-aligned. Checked math: `retain ≥ stack_size` (notably the default
|
||||
/// 64 KiB reserve with the 64 KiB RETAIN) and overflow collapse to `None`.
|
||||
pub(crate) fn retain_range(
|
||||
stack_size: usize,
|
||||
retain: usize,
|
||||
page: usize,
|
||||
) -> Option<(usize, usize)> {
|
||||
debug_assert!(page.is_power_of_two());
|
||||
let retain = retain.checked_add(page - 1)? & !(page - 1); // page_up(retain)
|
||||
let len = stack_size.checked_sub(retain)?;
|
||||
if len == 0 {
|
||||
return None;
|
||||
}
|
||||
Some((0, len))
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::{retain_range, shrink_range};
|
||||
|
||||
const PG: usize = 4096;
|
||||
|
||||
#[test]
|
||||
fn retain_covers_whole_stack_is_a_noop() {
|
||||
// The default config: reserve == RETAIN == 64 KiB. No zap, no syscall.
|
||||
assert_eq!(retain_range(16 * PG, 16 * PG, PG), None);
|
||||
assert_eq!(retain_range(PG, PG, PG), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn retain_larger_than_stack_is_a_noop() {
|
||||
assert_eq!(retain_range(16 * PG, 17 * PG, PG), None);
|
||||
assert_eq!(retain_range(PG, usize::MAX, PG), None); // page_up overflows
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn retain_zero_zaps_everything() {
|
||||
assert_eq!(retain_range(16 * PG, 0, PG), Some((0, 16 * PG)));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn retain_rounds_up_zapping_less() {
|
||||
// 1 byte of retain keeps a whole page.
|
||||
assert_eq!(retain_range(16 * PG, 1, PG), Some((0, 15 * PG)));
|
||||
assert_eq!(retain_range(16 * PG, PG + 1, PG), Some((0, 14 * PG)));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn retain_one_page_short_of_stack() {
|
||||
assert_eq!(retain_range(2 * PG, PG, PG), Some((0, PG)));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn retain_range_is_page_aligned() {
|
||||
for size_pg in [1usize, 2, 3, 16, 1024] {
|
||||
for retain in [0usize, 1, PG - 1, PG, PG + 1, 4 * PG, size_pg * PG] {
|
||||
if let Some((off, len)) = retain_range(size_pg * PG, retain, PG) {
|
||||
assert_eq!(off, 0);
|
||||
assert_eq!(len % PG, 0);
|
||||
assert!(len <= size_pg * PG);
|
||||
assert!(len > 0);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn empty_and_inverted_spans_are_none() {
|
||||
assert_eq!(shrink_range(0x8000_0000, 0x8000_0000, PG), None); // hwm == sp
|
||||
assert_eq!(shrink_range(0x8000_1000, 0x8000_0000, PG), None); // hwm > sp
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn span_smaller_than_redzone_plus_page_is_none() {
|
||||
let sp = 0x8000_0000;
|
||||
// Everything within redzone+1 page of sp: no full page clears both
|
||||
// the redzone and the page_up(hwm) rounding.
|
||||
assert_eq!(shrink_range(sp - PG, sp, PG), None);
|
||||
assert_eq!(shrink_range(sp - 2 * PG + 1, sp, PG), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn exact_two_pages_frees_one() {
|
||||
let sp = 0x8000_0000;
|
||||
let hwm = sp - 2 * PG;
|
||||
// [hwm, hwm+PG) frees; [sp−PG, sp) is redzone.
|
||||
assert_eq!(shrink_range(hwm, sp, PG), Some((hwm, PG)));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn unaligned_ends_round_inward() {
|
||||
let sp = 0x8000_0123; // live sp mid-page
|
||||
let hwm = 0x7f00_0abc; // high-water mid-page
|
||||
let (start, len) = shrink_range(hwm, sp, PG).unwrap();
|
||||
assert_eq!(start % PG, 0);
|
||||
assert_eq!(len % PG, 0);
|
||||
assert!(start >= hwm); // never below the sampled high-water
|
||||
assert!(start + len <= (sp - PG) & !(PG - 1)); // never into the redzone
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn result_never_crosses_sp() {
|
||||
// Sweep hwm across every offset of the page straddling the boundary.
|
||||
let sp = 0x8000_0000 + 137;
|
||||
for hwm in (sp - 4 * PG)..(sp) {
|
||||
if let Some((start, len)) = shrink_range(hwm, sp, PG) {
|
||||
assert!(start >= hwm);
|
||||
assert!(start + len + PG <= sp + PG); // end ≤ page_down(sp − PG) < sp
|
||||
assert!(len > 0);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn underflow_near_zero_is_none() {
|
||||
assert_eq!(shrink_range(0, PG - 1, PG), None); // sp < redzone
|
||||
assert_eq!(shrink_range(0, 0, PG), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn big_span_frees_interior() {
|
||||
let sp = 0x8000_0000;
|
||||
let spike = 4 * 1024 * 1024;
|
||||
let hwm = sp - spike;
|
||||
let (start, len) = shrink_range(hwm, sp, PG).unwrap();
|
||||
assert_eq!(start, hwm); // aligned input: starts exactly at hwm
|
||||
assert_eq!(len, spike - PG); // everything but the redzone page
|
||||
}
|
||||
}
|
||||
|
||||
+107
-306
@@ -1,114 +1,16 @@
|
||||
//! Supervision: keep a set of actors alive.
|
||||
//! Supervision signals.
|
||||
//!
|
||||
//! A *supervisor* is an actor whose only job is to start a fixed set of child
|
||||
//! actors and react when one of them terminates — restarting it (and, depending
|
||||
//! on the strategy, some of its siblings) according to a policy, or giving up
|
||||
//! when failures arrive too fast. It is how you turn "an actor that might crash"
|
||||
//! into "a service that stays up": a crash becomes a restart instead of a hole
|
||||
//! in the process tree.
|
||||
//! Every actor has a supervisor, which is itself just an actor with a
|
||||
//! `Receiver<Signal>`. When a child actor terminates, the scheduler sends
|
||||
//! a `Signal` on the supervisor's channel. The supervisor decides what to
|
||||
//! do — restart, escalate, ignore.
|
||||
//!
|
||||
//! The supervisor type is [`OneForOne`]. The name is historical — the restart
|
||||
//! *strategy* is selectable via [`OneForOne::strategy`], and
|
||||
//! [`Strategy::OneForOne`] is merely the default. You declare the children up
|
||||
//! front as [`ChildSpec`]s, each carrying a [`Restart`] policy, then hand the
|
||||
//! supervision loop an actor of its own with [`OneForOne::run`].
|
||||
//!
|
||||
//! ## A child that crashes and recovers
|
||||
//!
|
||||
//! ```
|
||||
//! use smarm::{run, spawn, ChildSpec, OneForOne, Restart};
|
||||
//! use std::sync::Arc;
|
||||
//! use std::sync::atomic::{AtomicUsize, Ordering};
|
||||
//! use std::time::Duration;
|
||||
//!
|
||||
//! run(|| {
|
||||
//! // A flaky child: it panics on its first two starts, then settles.
|
||||
//! let starts = Arc::new(AtomicUsize::new(0));
|
||||
//! let s = starts.clone();
|
||||
//! let child = move || {
|
||||
//! let n = s.fetch_add(1, Ordering::SeqCst) + 1;
|
||||
//! if n < 3 {
|
||||
//! panic!("boom {n}");
|
||||
//! }
|
||||
//! // The third start returns normally.
|
||||
//! };
|
||||
//!
|
||||
//! // The supervisor runs on its own actor. `Transient` restarts a child
|
||||
//! // that panics but treats a clean return as "done", so once the child
|
||||
//! // finally succeeds the supervisor has nothing left to do and `run()`
|
||||
//! // returns. smarm catches the child's panic and turns it into a restart;
|
||||
//! // it never reaches the process as a real crash.
|
||||
//! let sup = spawn(move || {
|
||||
//! OneForOne::new()
|
||||
//! .intensity(5, Duration::from_secs(60))
|
||||
//! .child(ChildSpec::new(Restart::Transient, child))
|
||||
//! .run();
|
||||
//! });
|
||||
//! sup.join().unwrap();
|
||||
//!
|
||||
//! assert_eq!(starts.load(Ordering::SeqCst), 3); // one start, two restarts
|
||||
//! });
|
||||
//! ```
|
||||
//!
|
||||
//! ## Restart policies
|
||||
//!
|
||||
//! Each child carries a [`Restart`] policy that decides whether *that child*
|
||||
//! comes back when it terminates:
|
||||
//!
|
||||
//! - [`Restart::Permanent`] restarts on any termination, normal or panic —
|
||||
//! for a service that should never be down.
|
||||
//! - [`Restart::Transient`] restarts only on an abnormal exit (a panic or a
|
||||
//! cooperative stop); a clean return means "done" — for work that runs to
|
||||
//! completion but should be retried if it crashes.
|
||||
//! - [`Restart::Temporary`] never restarts; the death is simply noted.
|
||||
//!
|
||||
//! ## Strategies: which siblings get cycled
|
||||
//!
|
||||
//! When a restart is due, the [`Strategy`] decides which *other* children are
|
||||
//! cycled along with the one that died. The triggering child's own policy still
|
||||
//! decides whether anything restarts at all.
|
||||
//!
|
||||
//! - [`Strategy::OneForOne`] restarts only the child that died — the default.
|
||||
//! - [`Strategy::OneForAll`] restarts every child: the survivors are stopped,
|
||||
//! then the whole set is restarted.
|
||||
//! - [`Strategy::RestForOne`] restarts the dead child and every child started
|
||||
//! after it, leaving earlier children untouched.
|
||||
//!
|
||||
//! ## Stopping a sibling is cooperative
|
||||
//!
|
||||
//! Cycling a sibling means stopping it first, and a supervisor never tears a
|
||||
//! running actor down from outside: smarm actors share a heap and rely on
|
||||
//! Drop/RAII, so unwinding a peer's stack from elsewhere would be unsound.
|
||||
//! Instead the supervisor *requests* the stop and the child unwinds at its next
|
||||
//! observation point — a `check!()`, an allocation, or a blocking call. A child
|
||||
//! wedged in a tight loop with no observation point cannot be stopped, for the
|
||||
//! same reason it cannot be preempted.
|
||||
//!
|
||||
//! ## Giving up: the restart-intensity cap
|
||||
//!
|
||||
//! A child that crashes the instant it starts would otherwise restart forever.
|
||||
//! [`OneForOne::intensity`] bounds that: at most `max` restarts within any
|
||||
//! `period`-long sliding window. One terminating child counts as a single
|
||||
//! restart event even when the strategy cycles several siblings. When the cap
|
||||
//! trips, the supervisor stops restarting, cooperatively stops any survivors in
|
||||
//! reverse start order, and `run()` returns.
|
||||
//!
|
||||
//! ## Running context
|
||||
//!
|
||||
//! [`OneForOne::run`] takes over the calling actor as the supervision loop, so
|
||||
//! a supervisor gets an actor of its own — typically
|
||||
//! `spawn(|| OneForOne::new()/* … */.run())`, all from inside
|
||||
//! [`run`](crate::run). Each child is spawned beneath the supervisor's pid, so
|
||||
//! every child termination funnels back to it as a [`Signal`].
|
||||
//! For v0.1 there is no built-in restart-intensity cap. That's policy and
|
||||
//! lives in user code; library is mechanism only.
|
||||
|
||||
use crate::pid::Pid;
|
||||
use std::any::Any;
|
||||
|
||||
/// A child-termination notice delivered to its supervisor.
|
||||
///
|
||||
/// Every child a supervisor starts is spawned beneath the supervisor's pid, so
|
||||
/// each child's termination funnels back to it as one of these. The variant
|
||||
/// records *how* the child went — which is what its [`Restart`] policy keys off.
|
||||
pub enum Signal {
|
||||
/// The child exited normally.
|
||||
Exit(Pid),
|
||||
@@ -123,9 +25,9 @@ pub enum Signal {
|
||||
impl std::fmt::Debug for Signal {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
match self {
|
||||
Signal::Exit(pid) => write!(f, "Signal::Exit({:?})", pid),
|
||||
Signal::Exit(pid) => write!(f, "Signal::Exit({:?})", pid),
|
||||
Signal::Panic(pid, _) => write!(f, "Signal::Panic({:?}, ..)", pid),
|
||||
Signal::Stopped(pid) => write!(f, "Signal::Stopped({:?})", pid),
|
||||
Signal::Stopped(pid) => write!(f, "Signal::Stopped({:?})", pid),
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -133,15 +35,35 @@ impl std::fmt::Debug for Signal {
|
||||
impl Signal {
|
||||
pub fn pid(&self) -> Pid {
|
||||
match self {
|
||||
Signal::Exit(p) => *p,
|
||||
Signal::Exit(p) => *p,
|
||||
Signal::Panic(p, _) => *p,
|
||||
Signal::Stopped(p) => *p,
|
||||
Signal::Stopped(p) => *p,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
use crate::channel::{channel, RecvTimeoutError};
|
||||
use crate::monitor::DownReason;
|
||||
// ---------------------------------------------------------------------------
|
||||
// One-for-one supervisor
|
||||
//
|
||||
// A supervisor is itself an actor. It spawns each child under its own pid so
|
||||
// that every child death funnels into one mailbox (the `supervisor_channel`),
|
||||
// then loops: receive a Signal, decide per the child's `Restart` policy
|
||||
// whether to restart, and enforce a restart-intensity cap so a child that
|
||||
// crashes in a tight loop eventually gives up instead of spinning forever.
|
||||
//
|
||||
// What this does NOT do: *forcibly* terminate a running child. smarm actors
|
||||
// share a heap and rely on Drop/RAII, so tearing down a peer's stack from
|
||||
// outside is unsound. Stopping a sibling — whether for `one_for_all` /
|
||||
// `rest_for_one`, for a propagated link death, or for the ordered shutdown
|
||||
// below — is therefore *cooperative*: `request_stop` flags the child and it
|
||||
// unwinds at its next observation point. A child with no observation points
|
||||
// (a tight loop with no `check!()`, allocation, or blocking op) cannot be
|
||||
// stopped, exactly as it cannot be preempted. When the intensity cap trips,
|
||||
// this supervisor stops restarting and tears the remaining children down in
|
||||
// reverse start order before returning.
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
use crate::channel::channel;
|
||||
use std::collections::{HashMap, VecDeque};
|
||||
use std::sync::Arc;
|
||||
use std::time::{Duration, Instant};
|
||||
@@ -166,52 +88,11 @@ pub enum Restart {
|
||||
pub struct ChildSpec {
|
||||
start: Arc<dyn Fn() + Send + Sync + 'static>,
|
||||
restart: Restart,
|
||||
shutdown: Shutdown,
|
||||
}
|
||||
|
||||
impl ChildSpec {
|
||||
/// A child with the given restart policy and the default
|
||||
/// [`Shutdown::Timeout`] of 5 seconds.
|
||||
pub fn new(restart: Restart, start: impl Fn() + Send + Sync + 'static) -> Self {
|
||||
Self {
|
||||
start: Arc::new(start),
|
||||
restart,
|
||||
shutdown: Shutdown::default(),
|
||||
}
|
||||
}
|
||||
|
||||
/// Set how the supervisor stops this child (see [`Shutdown`]). A child
|
||||
/// that is itself a supervisor should use [`Shutdown::Infinity`] so its
|
||||
/// own subtree gets its full grace periods.
|
||||
pub fn shutdown(mut self, shutdown: Shutdown) -> Self {
|
||||
self.shutdown = shutdown;
|
||||
self
|
||||
}
|
||||
}
|
||||
|
||||
/// How a supervisor stops a child it is taking down — the OTP child-spec
|
||||
/// `shutdown` value. Applies to every supervisor-initiated stop: the ordered
|
||||
/// shutdown of the whole set and the sibling cycling of
|
||||
/// [`Strategy::OneForAll`] / [`Strategy::RestForOne`].
|
||||
///
|
||||
/// A graceful stop is a [`request_shutdown`](crate::request_shutdown): a child
|
||||
/// that traps exits receives the request as a message and winds down in its
|
||||
/// own time; one that does not is stopped outright.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum Shutdown {
|
||||
/// `request_stop` immediately; no request, no grace period.
|
||||
BrutalKill,
|
||||
/// `request_shutdown`, wait up to the duration for the child to exit, then
|
||||
/// `request_stop` it. The default, at 5 seconds.
|
||||
Timeout(Duration),
|
||||
/// `request_shutdown` and wait however long the child takes. Use for a
|
||||
/// child supervisor, whose subtree has its own timeouts.
|
||||
Infinity,
|
||||
}
|
||||
|
||||
impl Default for Shutdown {
|
||||
fn default() -> Self {
|
||||
Shutdown::Timeout(Duration::from_secs(5))
|
||||
Self { start: Arc::new(start), restart }
|
||||
}
|
||||
}
|
||||
|
||||
@@ -234,10 +115,11 @@ pub enum Strategy {
|
||||
|
||||
/// A supervisor over a fixed set of children.
|
||||
///
|
||||
/// Build it with [`new`](Self::new), add children with [`child`](Self::child),
|
||||
/// pick a [`strategy`](Self::strategy) and an [`intensity`](Self::intensity)
|
||||
/// cap, then drive the loop with [`run`](Self::run) on an actor of its own. See
|
||||
/// the [module docs](self) for the full picture.
|
||||
/// Despite the name (kept for backwards compatibility), the restart strategy
|
||||
/// is selectable via [`OneForOne::strategy`]; the default is
|
||||
/// [`Strategy::OneForOne`]. Build with `new()`, add children with `child()`,
|
||||
/// tune the cap with `intensity()`, then drive it with `run()` from inside an
|
||||
/// actor (typically `spawn(|| OneForOne::new()....run())`).
|
||||
pub struct OneForOne {
|
||||
children: Vec<ChildSpec>,
|
||||
strategy: Strategy,
|
||||
@@ -283,35 +165,23 @@ impl OneForOne {
|
||||
}
|
||||
|
||||
/// Run the supervision loop on the current actor. Returns when every child
|
||||
/// has reached a terminal, non-restartable state, when the restart
|
||||
/// intensity cap is tripped, or when the supervisor is asked to shut down
|
||||
/// (a [`request_shutdown`](crate::request_shutdown) — from its own
|
||||
/// supervisor, or from the app). On every one of those exits the survivors
|
||||
/// are stopped in reverse start order, each per its
|
||||
/// [`Shutdown`] policy, before this returns.
|
||||
///
|
||||
/// The supervisor traps exits for the length of the loop (that is how the
|
||||
/// shutdown request reaches it as a message). Should the supervisor itself
|
||||
/// be hard-stopped with [`request_stop`](crate::request_stop), it unwinds
|
||||
/// without waiting for anything — but a drop guard hard-stops its live
|
||||
/// children on the way out, so the subtree is not orphaned (a child
|
||||
/// supervisor unwinds the same way, recursively).
|
||||
/// has reached a terminal, non-restartable state, or when the restart
|
||||
/// intensity cap is tripped.
|
||||
pub fn run(self) {
|
||||
let me = crate::scheduler::self_pid();
|
||||
let (tx, rx) = channel::<Signal>();
|
||||
crate::scheduler::register_supervisor_channel(me, tx);
|
||||
let exits = crate::link::trap_exit();
|
||||
|
||||
// pid -> index into `self.children`, for the children currently alive.
|
||||
let mut live = Live::default();
|
||||
let mut by_pid: HashMap<Pid, usize> = HashMap::new();
|
||||
let mut active: usize = 0;
|
||||
// Sliding window of recent restart instants, for the intensity cap.
|
||||
let mut restarts: Vec<Instant> = Vec::new();
|
||||
|
||||
let start_child = |idx: usize, live: &mut Live| {
|
||||
let start_child = |idx: usize, by_pid: &mut HashMap<Pid, usize>| {
|
||||
let start = self.children[idx].start.clone();
|
||||
let h = crate::scheduler::spawn_under(me, move || (start)());
|
||||
live.insert(h.pid(), idx);
|
||||
by_pid.insert(h.pid(), idx);
|
||||
// We supervise via the signal funnel, not by joining; drop the
|
||||
// handle so the child's slot is reclaimed promptly on death (the
|
||||
// termination Signal is delivered before reclamation regardless).
|
||||
@@ -319,105 +189,28 @@ impl OneForOne {
|
||||
};
|
||||
|
||||
for idx in 0..self.children.len() {
|
||||
start_child(idx, &mut live);
|
||||
start_child(idx, &mut by_pid);
|
||||
active += 1;
|
||||
}
|
||||
|
||||
// A signal that arrives while we are awaiting stop-confirmations (for a
|
||||
// child we are *not* currently stopping) is stashed here and processed
|
||||
// by the main loop before it blocks again.
|
||||
// by the main loop before it blocks on `recv` again.
|
||||
let mut pending: VecDeque<Signal> = VecDeque::new();
|
||||
|
||||
// Stop one child per its policy and wait for its termination signal.
|
||||
// Signals for other pids that arrive meanwhile are stashed. Bounded by
|
||||
// construction: `request_stop` (used directly, or as the fallback once
|
||||
// the grace period lapses) always produces a signal.
|
||||
let stop_child = |pid: Pid, idx: usize, pending: &mut VecDeque<Signal>| {
|
||||
let await_one = |deadline: Option<Instant>, pending: &mut VecDeque<Signal>| -> bool {
|
||||
loop {
|
||||
let sig = match pending.iter().position(|s| s.pid() == pid) {
|
||||
Some(i) => pending.remove(i),
|
||||
None => match deadline {
|
||||
None => rx.recv().ok(),
|
||||
Some(dl) => {
|
||||
match rx.recv_timeout(dl.saturating_duration_since(Instant::now()))
|
||||
{
|
||||
Ok(s) => Some(s),
|
||||
Err(RecvTimeoutError::Timeout) => return false,
|
||||
Err(RecvTimeoutError::Disconnected) => None,
|
||||
}
|
||||
}
|
||||
},
|
||||
};
|
||||
match sig {
|
||||
Some(s) if s.pid() == pid => return true,
|
||||
Some(s) => pending.push_back(s),
|
||||
None => return true, // funnel closed: nothing more can arrive
|
||||
}
|
||||
}
|
||||
};
|
||||
match self.children[idx].shutdown {
|
||||
Shutdown::BrutalKill => {
|
||||
crate::scheduler::request_stop(pid);
|
||||
await_one(None, pending);
|
||||
}
|
||||
Shutdown::Timeout(grace) => {
|
||||
crate::scheduler::request_shutdown(pid);
|
||||
if !await_one(Some(Instant::now() + grace), pending) {
|
||||
crate::scheduler::request_stop(pid);
|
||||
await_one(None, pending);
|
||||
}
|
||||
}
|
||||
Shutdown::Infinity => {
|
||||
crate::scheduler::request_shutdown(pid);
|
||||
await_one(None, pending);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
// Stop a set of children in reverse start order, one at a time.
|
||||
let stop_set =
|
||||
|set: &mut Vec<(Pid, usize)>, live: &mut Live, pending: &mut VecDeque<Signal>| {
|
||||
set.sort_unstable_by_key(|x| std::cmp::Reverse(x.1));
|
||||
for (pid, idx) in set.iter() {
|
||||
live.remove(pid);
|
||||
stop_child(*pid, *idx, pending);
|
||||
}
|
||||
};
|
||||
|
||||
// Wait for the next event: a stashed signal, a child signal, or a
|
||||
// shutdown request. `Ok(sig)`, or `Err(())` when we must wind down.
|
||||
let next_event = |pending: &mut VecDeque<Signal>| -> Result<Signal, ()> {
|
||||
loop {
|
||||
if let Some(s) = pending.pop_front() {
|
||||
return Ok(s);
|
||||
}
|
||||
// The trap inbox is arm 0: a shutdown request is noticed even
|
||||
// under a flood of child signals.
|
||||
match crate::channel::select(&[&exits, &rx]) {
|
||||
0 => match exits.try_recv() {
|
||||
Ok(Some(sig)) if sig.reason == DownReason::Shutdown => return Err(()),
|
||||
// Any other exit signal (a linked peer's death — a
|
||||
// supervisor links nothing itself, but may be linked
|
||||
// to) is not ours to act on; a closed trap inbox is
|
||||
// impossible while `exits` is held here.
|
||||
_ => {}
|
||||
},
|
||||
_ => match rx.try_recv() {
|
||||
Ok(Some(s)) => return Ok(s),
|
||||
Ok(None) => {}
|
||||
Err(_) => return Err(()), // funnel closed: nothing left to supervise
|
||||
},
|
||||
}
|
||||
let next_signal = |pending: &mut VecDeque<Signal>| -> Option<Signal> {
|
||||
if let Some(s) = pending.pop_front() {
|
||||
Some(s)
|
||||
} else {
|
||||
rx.recv().ok()
|
||||
}
|
||||
};
|
||||
|
||||
while active > 0 {
|
||||
let sig = match next_event(&mut pending) {
|
||||
Ok(s) => s,
|
||||
Err(()) => break,
|
||||
let sig = match next_signal(&mut pending) {
|
||||
Some(s) => s,
|
||||
None => break, // mailbox closed: nothing left to supervise
|
||||
};
|
||||
let idx = match live.remove(&sig.pid()) {
|
||||
let idx = match by_pid.remove(&sig.pid()) {
|
||||
Some(i) => i,
|
||||
None => continue, // stray/duplicate signal
|
||||
};
|
||||
@@ -449,70 +242,78 @@ impl OneForOne {
|
||||
restarts.push(now);
|
||||
|
||||
// Which *live* siblings get cycled along with the failed child.
|
||||
// (The failed child is already gone — removed from `live` above.)
|
||||
// (The failed child is already gone — removed from `by_pid` above.)
|
||||
let mut to_stop: Vec<(Pid, usize)> = match self.strategy {
|
||||
Strategy::OneForOne => Vec::new(),
|
||||
Strategy::OneForAll => live.iter().map(|(p, i)| (*p, *i)).collect(),
|
||||
Strategy::RestForOne => live
|
||||
Strategy::OneForAll => by_pid.iter().map(|(p, i)| (*p, *i)).collect(),
|
||||
Strategy::RestForOne => by_pid
|
||||
.iter()
|
||||
.filter(|(_, i)| **i > idx)
|
||||
.map(|(p, i)| (*p, *i))
|
||||
.collect(),
|
||||
};
|
||||
// Stop survivors in reverse start order (highest child index first).
|
||||
to_stop.sort_unstable_by(|a, b| b.1.cmp(&a.1));
|
||||
|
||||
// The set we will restart: the failed child plus every sibling we
|
||||
// are about to stop, restarted in start (ascending index) order.
|
||||
let mut restart_set: Vec<usize> = Vec::with_capacity(to_stop.len() + 1);
|
||||
restart_set.push(idx);
|
||||
restart_set.extend(to_stop.iter().map(|(_, i)| *i));
|
||||
|
||||
// Stop the survivors (each per its policy, reverse start order),
|
||||
// then restart the whole set in start order. Net effect on
|
||||
// `active`: one child died (idx), `to_stop.len()` were stopped,
|
||||
// and `restart_set.len() == 1 + to_stop.len()` are started — so
|
||||
// Request stops, then await each survivor's termination signal
|
||||
// before restarting. `request_stop` on an already-dead pid is a
|
||||
// no-op; in that case its (already-sent) Exit signal serves as the
|
||||
// confirmation. Any signal for a pid we are *not* awaiting is
|
||||
// stashed for the main loop.
|
||||
let mut awaiting: Vec<Pid> = Vec::with_capacity(to_stop.len());
|
||||
for (pid, cidx) in &to_stop {
|
||||
by_pid.remove(pid);
|
||||
restart_set.push(*cidx);
|
||||
crate::scheduler::request_stop(*pid);
|
||||
awaiting.push(*pid);
|
||||
}
|
||||
while !awaiting.is_empty() {
|
||||
let s = match next_signal(&mut pending) {
|
||||
Some(s) => s,
|
||||
None => break, // mailbox closed mid-await; stop waiting
|
||||
};
|
||||
if let Some(pos) = awaiting.iter().position(|p| *p == s.pid()) {
|
||||
awaiting.swap_remove(pos);
|
||||
} else {
|
||||
pending.push_back(s);
|
||||
}
|
||||
}
|
||||
|
||||
// Restart the whole set in start order. Net effect on `active`:
|
||||
// one child died (idx), `to_stop.len()` were stopped, and
|
||||
// `restart_set.len() == 1 + to_stop.len()` are started — so
|
||||
// `active` is unchanged and needs no adjustment here.
|
||||
stop_set(&mut to_stop, &mut live, &mut pending);
|
||||
restart_set.sort_unstable();
|
||||
for cidx in restart_set {
|
||||
start_child(cidx, &mut live);
|
||||
start_child(cidx, &mut by_pid);
|
||||
}
|
||||
}
|
||||
|
||||
// Ordered shutdown: stop any survivors in reverse start order, each per
|
||||
// its policy. On the normal `active == 0` exit `live` is empty and this
|
||||
// is a no-op; on a shutdown request, a cap-trip, or a closed funnel it
|
||||
// tears the remaining children down deterministically.
|
||||
let mut survivors: Vec<(Pid, usize)> = live.iter().map(|(p, i)| (*p, *i)).collect();
|
||||
stop_set(&mut survivors, &mut live, &mut pending);
|
||||
}
|
||||
}
|
||||
|
||||
/// The live children of a supervisor, with a drop guard: if the supervisor is
|
||||
/// unwound (a hard `request_stop`, or a panic in the loop) its children are
|
||||
/// hard-stopped rather than orphaned. Fire-and-forget by necessity — a guard
|
||||
/// running mid-unwind cannot park to await anything.
|
||||
#[derive(Default)]
|
||||
struct Live(HashMap<Pid, usize>);
|
||||
|
||||
impl std::ops::Deref for Live {
|
||||
type Target = HashMap<Pid, usize>;
|
||||
fn deref(&self) -> &Self::Target {
|
||||
&self.0
|
||||
}
|
||||
}
|
||||
|
||||
impl std::ops::DerefMut for Live {
|
||||
fn deref_mut(&mut self) -> &mut Self::Target {
|
||||
&mut self.0
|
||||
}
|
||||
}
|
||||
|
||||
impl Drop for Live {
|
||||
fn drop(&mut self) {
|
||||
if std::thread::panicking() {
|
||||
for pid in self.0.keys() {
|
||||
crate::scheduler::request_stop(*pid);
|
||||
// Ordered shutdown: stop any survivors in reverse start order and await
|
||||
// their termination. On the normal `active == 0` exit `by_pid` is empty
|
||||
// and this is a no-op; on a cap-trip or mailbox-closed break it tears
|
||||
// the remaining children down deterministically instead of leaking them.
|
||||
let mut survivors: Vec<(Pid, usize)> = by_pid.iter().map(|(p, i)| (*p, *i)).collect();
|
||||
survivors.sort_unstable_by(|a, b| b.1.cmp(&a.1));
|
||||
let mut awaiting: Vec<Pid> = Vec::with_capacity(survivors.len());
|
||||
for (pid, _) in &survivors {
|
||||
crate::scheduler::request_stop(*pid);
|
||||
awaiting.push(*pid);
|
||||
}
|
||||
while !awaiting.is_empty() {
|
||||
let s = match next_signal(&mut pending) {
|
||||
Some(s) => s,
|
||||
None => break,
|
||||
};
|
||||
if let Some(pos) = awaiting.iter().position(|p| *p == s.pid()) {
|
||||
awaiting.swap_remove(pos);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -1,46 +0,0 @@
|
||||
//! std vs loom indirection for the modules that loom model-checks
|
||||
//! (`slot_state`, `run_queue`). Everything else uses std paths directly —
|
||||
//! the full runtime (context switches, futexes, real TLS) is not loom-able
|
||||
//! and is never executed under `cfg(loom)`.
|
||||
//!
|
||||
//! Build the loom models with: `RUSTFLAGS="--cfg loom" cargo test --lib --release`
|
||||
|
||||
#[cfg(loom)]
|
||||
pub(crate) use loom::sync::atomic::{fence, AtomicU64, AtomicUsize, Ordering};
|
||||
|
||||
#[cfg(not(loom))]
|
||||
pub(crate) use std::sync::atomic::{fence, AtomicU64, AtomicUsize, Ordering};
|
||||
|
||||
// park.rs condvar-parker (loom + non-Linux builds only; the Linux non-loom
|
||||
// build parks on a futex and never touches these — gating them identically
|
||||
// keeps the default build free of unused imports).
|
||||
#[cfg(loom)]
|
||||
pub(crate) use loom::sync::{Condvar, Mutex};
|
||||
|
||||
#[cfg(all(not(loom), not(target_os = "linux")))]
|
||||
pub(crate) use std::sync::{Condvar, Mutex};
|
||||
|
||||
/// `UnsafeCell` with loom's `with`/`with_mut` access API; pass-through cost
|
||||
/// is zero in normal builds (`#[inline]`, newtype over std's cell).
|
||||
#[cfg(loom)]
|
||||
pub(crate) use loom::cell::UnsafeCell;
|
||||
|
||||
#[cfg(not(loom))]
|
||||
pub(crate) struct UnsafeCell<T>(std::cell::UnsafeCell<T>);
|
||||
|
||||
#[cfg(not(loom))]
|
||||
impl<T> UnsafeCell<T> {
|
||||
pub(crate) fn new(v: T) -> Self {
|
||||
Self(std::cell::UnsafeCell::new(v))
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn with<R>(&self, f: impl FnOnce(*const T) -> R) -> R {
|
||||
f(self.0.get())
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn with_mut<R>(&self, f: impl FnOnce(*mut T) -> R) -> R {
|
||||
f(self.0.get())
|
||||
}
|
||||
}
|
||||
+27
-247
@@ -15,14 +15,12 @@
|
||||
//! `BinaryHeap` is a max-heap; entries are wrapped in `Reverse` to get
|
||||
//! min-heap behaviour.
|
||||
//!
|
||||
//! Cancellation is selective. A `Sleep` / `WaitTimeout` entry is left in the
|
||||
//! heap on a non-timer wakeup (lock granted before timeout): it is popped
|
||||
//! eventually and no-ops because a stale unpark fails its epoch CAS — cheap
|
||||
//! (~32 bytes per stale entry plus a few cycles on pop), bounded by one entry
|
||||
//! per parked actor. A `Send` entry is different: running its thunk delivers a
|
||||
//! real message, so a stale one is *not* inert. `send_after` therefore carries
|
||||
//! true cancellation via the `armed` set keyed on the entry's `seq`; `pop_due`
|
||||
//! fires a `Send` only while it is still armed, and `cancel` removes the arm.
|
||||
//! No cancellation. When a non-timer wakeup happens (e.g. lock granted
|
||||
//! before timeout), the timer entry is left in the heap. It will be popped
|
||||
//! eventually and the dispatch will observe "actor is no longer parked /
|
||||
//! wait_seq is stale" and no-op. Cost is ~32 bytes per stale entry plus a
|
||||
//! few cycles on pop; acceptable given the upper bound is "one entry per
|
||||
//! parked actor".
|
||||
//!
|
||||
//! Stale pids (slot reused since the timer was inserted) are filtered on
|
||||
//! pop by the scheduler — same convention as the run queue.
|
||||
@@ -37,53 +35,19 @@ use std::time::{Duration, Instant};
|
||||
///
|
||||
/// Held inside `Entry`, dispatched by the scheduler in `pop_due`.
|
||||
pub enum Reason {
|
||||
/// `sleep(d)`. Wake `pid` via the epoch-matched unpark: if anything
|
||||
/// else (necessarily a terminal wake) already consumed the wait, the
|
||||
/// entry is stale and no-ops at the CAS.
|
||||
Sleep { epoch: u32 },
|
||||
/// A bounded wait (`Mutex::lock_timeout`, `Receiver::recv_timeout`,
|
||||
/// `select_timeout`). On expiry the scheduler calls
|
||||
/// `target.on_timeout(pid, epoch)`. The target then decides whether
|
||||
/// `pid` was actually still waiting (registration still present under
|
||||
/// its lock), and if so takes the registration and unparks via
|
||||
/// `unpark_at`. The epoch is the slot-word park-epoch — the runtime-wide
|
||||
/// wait identity — so a stale entry is doubly inert: the registration
|
||||
/// check misses, and even a racing unpark fails the word's epoch CAS.
|
||||
/// `loom::sleep(d)`. Unpark `pid` unconditionally (modulo the usual
|
||||
/// "still parked?" check the scheduler applies).
|
||||
Sleep,
|
||||
/// A bounded wait — currently only `Mutex::lock_timeout`. On expiry the
|
||||
/// scheduler calls `target.on_timeout(pid, wait_seq)`. The target then
|
||||
/// decides whether `pid` was actually still waiting, and if so unparks
|
||||
/// it with whatever error the wait was bounded for. `wait_seq` lets the
|
||||
/// target tell apart "this wait" from "a later wait by the same actor
|
||||
/// on the same target".
|
||||
WaitTimeout {
|
||||
target: Arc<dyn TimerTarget>,
|
||||
epoch: u32,
|
||||
wait_seq: u64,
|
||||
},
|
||||
/// `send_after`: deliver a message to an address at the deadline,
|
||||
/// cancellable. The destination (a `Pid<A>` / `Name<M>`) and the message
|
||||
/// are captured inside `fire`, which resolves the address through the
|
||||
/// registry and sends *when run* — so a target that died or, for a name,
|
||||
/// was restarted is observed at fire time, not arm time. A failed resolve
|
||||
/// or send is dropped (Erlang `erlang:send_after` semantics).
|
||||
///
|
||||
/// Unlike `Sleep` / `WaitTimeout`, a stale `Send` is **not** inert — running
|
||||
/// the thunk delivers a real message — so these are the only timers that
|
||||
/// carry true cancellation (the `armed` set on [`Timers`], keyed by the
|
||||
/// entry's `seq`). `pop_due` fires the thunk only for an entry still armed.
|
||||
Send { fire: Box<dyn FnOnce() + Send> },
|
||||
}
|
||||
|
||||
/// Opaque handle to an armed `send_after` timer, returned by
|
||||
/// [`Timers::insert_send`] and consumed by [`Timers::cancel`]. The inner value
|
||||
/// is the entry's insertion `seq`; callers must treat it as opaque so the
|
||||
/// backing structure can change (e.g. a future hierarchical timing wheel) with
|
||||
/// no API churn.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
|
||||
pub struct TimerId(u64);
|
||||
|
||||
impl TimerId {
|
||||
/// Wrap a raw value. Crate-internal: the gen_server timer layer mints its
|
||||
/// own loop-local `TimerId`s (the public ids it hands out, decoupled from
|
||||
/// the per-re-arm substrate `seq`) and maps them to live substrate ids.
|
||||
/// These local ids are only ever resolved through that layer's registry —
|
||||
/// never passed back to [`Timers::cancel`] — so the two id roles do not mix.
|
||||
pub(crate) fn from_raw(v: u64) -> Self {
|
||||
TimerId(v)
|
||||
}
|
||||
}
|
||||
|
||||
/// Callback the scheduler invokes when a `WaitTimeout` entry pops.
|
||||
@@ -91,7 +55,7 @@ impl TimerId {
|
||||
/// Implementors: do not touch `SchedulerState` other than via the public
|
||||
/// `unpark` / channel APIs. The scheduler is mid-iteration when this fires.
|
||||
pub trait TimerTarget: Send + Sync {
|
||||
fn on_timeout(&self, pid: Pid, epoch: u32);
|
||||
fn on_timeout(&self, pid: Pid, wait_seq: u64);
|
||||
}
|
||||
|
||||
pub struct Entry {
|
||||
@@ -102,19 +66,6 @@ pub struct Entry {
|
||||
seq: u64,
|
||||
pub pid: Pid,
|
||||
pub reason: Reason,
|
||||
/// RFC 007 virtual time: the global delay ledger reading when this entry
|
||||
/// was (re-)queued. `pop_due` shifts the effective deadline by any delay
|
||||
/// injected since, so timers dilate together with the causally-delayed
|
||||
/// workload instead of firing early in virtual terms.
|
||||
#[cfg(feature = "smarm-causal")]
|
||||
delay_stamp: u64,
|
||||
/// RFC 007: a wall-anchored entry opts out of the virtual-time shift —
|
||||
/// its deadline is honoured in wall time regardless of injected delay.
|
||||
/// Used by the causal controller's own measurement/cooldown sleeps so
|
||||
/// experiment windows keep a fixed wall length; ordinary workload timers
|
||||
/// stay virtual (`false`).
|
||||
#[cfg(feature = "smarm-causal")]
|
||||
wall: bool,
|
||||
}
|
||||
|
||||
impl PartialEq for Entry {
|
||||
@@ -129,9 +80,7 @@ impl Ord for Entry {
|
||||
// Earlier deadline first; ties broken by insertion order so the
|
||||
// ordering is total. `Reason` and `Pid` deliberately don't
|
||||
// participate.
|
||||
self.deadline
|
||||
.cmp(&other.deadline)
|
||||
.then_with(|| self.seq.cmp(&other.seq))
|
||||
self.deadline.cmp(&other.deadline).then_with(|| self.seq.cmp(&other.seq))
|
||||
}
|
||||
}
|
||||
|
||||
@@ -143,134 +92,27 @@ impl PartialOrd for Entry {
|
||||
|
||||
#[derive(Default)]
|
||||
pub struct Timers {
|
||||
/// RFC 018: the scheduler coordination layer. Attached once at
|
||||
/// `RuntimeInner::new`; every insert notes its deadline (min-maintained
|
||||
/// snapshot for the busy-path due-check + the timekeeper re-arm wake)
|
||||
/// and every pop/clear re-anchors the snapshot to the heap minimum.
|
||||
/// All calls happen under the timers mutex — the serialization the
|
||||
/// coordinator's timer protocol mandates. `None` only in unit tests
|
||||
/// that construct a bare `Timers`.
|
||||
coord: Option<std::sync::Arc<crate::park::Coordinator>>,
|
||||
/// Reverse-wrapped so the smallest deadline is at the top.
|
||||
heap: BinaryHeap<Reverse<Entry>>,
|
||||
/// Monotonic counter for the tiebreaker `seq` field (and the `TimerId` of a
|
||||
/// `Send` timer — the two are the same value).
|
||||
/// Monotonic counter for the tiebreaker `seq` field.
|
||||
next_seq: u64,
|
||||
/// Presence set of *live* `Send` timers, keyed by `seq`. Populated on
|
||||
/// `insert_send`, removed on fire (in `pop_due`) and on `cancel`. A `Send`
|
||||
/// entry fires only while present, so a `cancel` that lands before the
|
||||
/// entry pops prevents delivery; a `cancel` after it has fired finds
|
||||
/// nothing (the race signal). Bounded by armed-but-not-yet-resolved timers
|
||||
/// and self-collecting — no sweep. `Sleep` / `WaitTimeout` never touch it.
|
||||
armed: std::collections::HashSet<u64>,
|
||||
}
|
||||
|
||||
impl Timers {
|
||||
pub fn new() -> Self {
|
||||
Self {
|
||||
coord: None,
|
||||
heap: BinaryHeap::new(),
|
||||
next_seq: 0,
|
||||
armed: std::collections::HashSet::new(),
|
||||
}
|
||||
}
|
||||
|
||||
/// Attach the scheduler coordination layer (RFC 018). Called once, at
|
||||
/// runtime construction, before any scheduler thread exists.
|
||||
pub(crate) fn attach_coordinator(&mut self, c: std::sync::Arc<crate::park::Coordinator>) {
|
||||
self.coord = Some(c);
|
||||
Self { heap: BinaryHeap::new(), next_seq: 0 }
|
||||
}
|
||||
|
||||
/// Insert a `Sleep` timer. Convenience for the common case.
|
||||
pub fn insert_sleep(&mut self, deadline: Instant, pid: Pid, epoch: u32) {
|
||||
self.insert(deadline, pid, Reason::Sleep { epoch });
|
||||
pub fn insert_sleep(&mut self, deadline: Instant, pid: Pid) {
|
||||
self.insert(deadline, pid, Reason::Sleep);
|
||||
}
|
||||
|
||||
/// Insert a *wall-anchored* `Sleep` timer: fires at `deadline` in wall
|
||||
/// time even while causal profiling (feature `smarm-causal`) is injecting
|
||||
/// virtual delay — it never chases the delay ledger. Without the feature
|
||||
/// this is identical to [`insert_sleep`](Self::insert_sleep).
|
||||
///
|
||||
/// Intended for measurement machinery (the causal controller's window and
|
||||
/// cooldown sleeps, TSC calibration) whose durations *define* wall time
|
||||
/// rather than participate in the workload. Workload code should use the
|
||||
/// ordinary virtual-anchored timers.
|
||||
pub fn insert_sleep_wall(&mut self, deadline: Instant, pid: Pid, epoch: u32) {
|
||||
self.push(deadline, pid, Reason::Sleep { epoch }, true);
|
||||
}
|
||||
|
||||
/// Arm a cancellable `send_after` timer: run `fire` at `deadline` unless
|
||||
/// [`cancel`](Self::cancel)led first. `pid` is informational only (the
|
||||
/// destination, or who armed it — useful for introspection); it is *not*
|
||||
/// used to wake anyone, the delivery lives entirely inside `fire`. Returns
|
||||
/// a [`TimerId`] for cancellation.
|
||||
pub fn insert_send(
|
||||
&mut self,
|
||||
deadline: Instant,
|
||||
pid: Pid,
|
||||
fire: Box<dyn FnOnce() + Send>,
|
||||
) -> TimerId {
|
||||
self.armed.insert(self.next_seq);
|
||||
TimerId(self.push(deadline, pid, Reason::Send { fire }, false))
|
||||
}
|
||||
|
||||
/// Arm a *wall-anchored* cancellable `send_after` timer (RFC 007): the
|
||||
/// same contract as [`insert_send`](Self::insert_send), but the entry
|
||||
/// opts out of the virtual-time shift and fires at its raw deadline
|
||||
/// regardless of injected delay — the `Send`-reason sibling of
|
||||
/// [`insert_sleep_wall`](Self::insert_sleep_wall). Without the
|
||||
/// `smarm-causal` feature this is identical to `insert_send`.
|
||||
pub fn insert_send_wall(
|
||||
&mut self,
|
||||
deadline: Instant,
|
||||
pid: Pid,
|
||||
fire: Box<dyn FnOnce() + Send>,
|
||||
) -> TimerId {
|
||||
self.armed.insert(self.next_seq);
|
||||
TimerId(self.push(deadline, pid, Reason::Send { fire }, true))
|
||||
}
|
||||
|
||||
/// Cancel an armed `send_after` timer. Returns `true` if the timer was
|
||||
/// still armed (delivery is now prevented), `false` if it had already
|
||||
/// fired or been cancelled. The heap entry, if still pending, is left to be
|
||||
/// discarded when its deadline passes — `pop_due` drops any `Send` entry
|
||||
/// whose `seq` is no longer armed.
|
||||
pub fn cancel(&mut self, id: TimerId) -> bool {
|
||||
self.armed.remove(&id.0)
|
||||
}
|
||||
|
||||
/// Insert an arbitrary (virtual-anchored) timer entry.
|
||||
/// Insert an arbitrary timer entry.
|
||||
pub fn insert(&mut self, deadline: Instant, pid: Pid, reason: Reason) {
|
||||
self.push(deadline, pid, reason, false);
|
||||
}
|
||||
|
||||
/// Common insertion path. `wall` selects the RFC 007 anchor (see
|
||||
/// [`insert_sleep_wall`](Self::insert_sleep_wall)); it is accepted — and
|
||||
/// ignored — without the `smarm-causal` feature so callers don't fork.
|
||||
/// Returns the entry's `seq`.
|
||||
fn push(&mut self, deadline: Instant, pid: Pid, reason: Reason, wall: bool) -> u64 {
|
||||
#[cfg(not(feature = "smarm-causal"))]
|
||||
let _ = wall;
|
||||
let seq = self.next_seq;
|
||||
self.next_seq = self.next_seq.wrapping_add(1);
|
||||
self.heap.push(Reverse(Entry {
|
||||
deadline,
|
||||
seq,
|
||||
pid,
|
||||
reason,
|
||||
#[cfg(feature = "smarm-causal")]
|
||||
delay_stamp: crate::causal::global_delay_cycles(),
|
||||
#[cfg(feature = "smarm-causal")]
|
||||
wall,
|
||||
}));
|
||||
// RFC 018: publish the (possibly new-minimum) deadline to the
|
||||
// busy-path snapshot and wake the timekeeper if it is parked
|
||||
// toward a later one. We hold the timers mutex — the mandated
|
||||
// serialization for both.
|
||||
if let Some(c) = &self.coord {
|
||||
c.note_deadline(deadline);
|
||||
}
|
||||
seq
|
||||
self.heap.push(Reverse(Entry { deadline, seq, pid, reason }));
|
||||
}
|
||||
|
||||
pub fn is_empty(&self) -> bool {
|
||||
@@ -282,10 +124,6 @@ impl Timers {
|
||||
/// discarded so it can't keep the runtime alive.
|
||||
pub fn clear(&mut self) {
|
||||
self.heap.clear();
|
||||
self.armed.clear();
|
||||
if let Some(c) = &self.coord {
|
||||
c.refresh_deadline(None);
|
||||
}
|
||||
}
|
||||
|
||||
/// Soonest pending deadline, or `None` if the heap is empty.
|
||||
@@ -295,72 +133,14 @@ impl Timers {
|
||||
|
||||
/// Pop every entry whose deadline is ≤ `now`, in deadline order.
|
||||
/// The scheduler dispatches each entry by inspecting `entry.reason`.
|
||||
///
|
||||
/// A due `Send` entry is returned only if it is still armed; a cancelled
|
||||
/// one is silently dropped here (its `seq` was already removed from
|
||||
/// `armed` by [`cancel`](Self::cancel)). Returning it removes it from
|
||||
/// `armed`, so a later `cancel` of a fired timer reports `false`.
|
||||
///
|
||||
/// RFC 007 virtual time (feature `smarm-causal`): before an entry fires,
|
||||
/// any global delay injected since it was (re-)queued is added to its
|
||||
/// deadline; an entry whose *effective* deadline hasn't passed is pushed
|
||||
/// back with the shifted deadline and a fresh stamp, so it keeps chasing
|
||||
/// delay injected while it waits. Consequences, both benign:
|
||||
/// [`peek_deadline`](Self::peek_deadline) may under-report (raw deadline
|
||||
/// earlier than effective), costing at most one spurious scheduler wake
|
||||
/// per injected chunk; and a shift never converts wall time — with zero
|
||||
/// debt the path is byte-identical to the featureless one. Wall-anchored
|
||||
/// entries ([`insert_sleep_wall`](Self::insert_sleep_wall)) are exempt
|
||||
/// from the shift and always fire at their raw deadline.
|
||||
pub fn pop_due(&mut self, now: Instant) -> Vec<Entry> {
|
||||
let mut out = Vec::new();
|
||||
#[cfg(feature = "smarm-causal")]
|
||||
let global = crate::causal::global_delay_cycles();
|
||||
while let Some(r) = self.heap.peek() {
|
||||
if r.0.deadline > now {
|
||||
if r.0.deadline <= now {
|
||||
out.push(self.heap.pop().unwrap().0);
|
||||
} else {
|
||||
break;
|
||||
}
|
||||
#[allow(unused_mut)]
|
||||
let mut entry = match self.heap.pop() {
|
||||
Some(e) => e.0,
|
||||
None => panic!("smarm: timer heap pop after peek returned None (core corrupt)"),
|
||||
};
|
||||
if matches!(entry.reason, Reason::Send { .. }) && !self.armed.contains(&entry.seq) {
|
||||
// Cancelled before it came due: discard, do not deliver.
|
||||
// (Checked before any shift so a cancelled entry is never
|
||||
// re-queued just to be discarded later.)
|
||||
continue;
|
||||
}
|
||||
#[cfg(feature = "smarm-causal")]
|
||||
if !entry.wall {
|
||||
let debt = global.saturating_sub(entry.delay_stamp);
|
||||
if debt > 0 {
|
||||
let shifted = entry
|
||||
.deadline
|
||||
.checked_add(crate::causal::cycles_to_duration(debt))
|
||||
.unwrap_or(entry.deadline);
|
||||
if shifted > now {
|
||||
// Not due in virtual time: re-queue at the shifted
|
||||
// deadline, stamped, keeping `seq` (and thus `Send`
|
||||
// cancellation identity) intact.
|
||||
entry.deadline = shifted;
|
||||
entry.delay_stamp = global;
|
||||
self.heap.push(Reverse(entry));
|
||||
continue;
|
||||
}
|
||||
}
|
||||
}
|
||||
if matches!(entry.reason, Reason::Send { .. }) {
|
||||
self.armed.remove(&entry.seq);
|
||||
}
|
||||
out.push(entry);
|
||||
}
|
||||
// RFC 018: re-anchor the busy-path snapshot to the new heap minimum
|
||||
// (still under the timers mutex). A causal-shift re-queue above went
|
||||
// through `heap.push` directly, so this peek is the one place the
|
||||
// snapshot is guaranteed to catch up.
|
||||
if let Some(c) = &self.coord {
|
||||
c.refresh_deadline(self.peek_deadline());
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
+37
-100
@@ -11,22 +11,18 @@
|
||||
//! cargo test --test runtime <test_name> --features smarm-trace
|
||||
//!
|
||||
//! Output: smarm_trace.json in cwd, or $SMARM_TRACE_FILE.
|
||||
//! View: <https://ui.perfetto.dev> or chrome://tracing
|
||||
//! View: https://ui.perfetto.dev or chrome://tracing
|
||||
|
||||
#[cfg(feature = "smarm-trace")]
|
||||
#[macro_export]
|
||||
macro_rules! te {
|
||||
($kind:expr) => {
|
||||
$crate::trace::record($kind)
|
||||
};
|
||||
($kind:expr) => { $crate::trace::record($kind) };
|
||||
}
|
||||
|
||||
#[cfg(not(feature = "smarm-trace"))]
|
||||
#[macro_export]
|
||||
macro_rules! te {
|
||||
($kind:expr) => {
|
||||
()
|
||||
};
|
||||
($kind:expr) => { () };
|
||||
}
|
||||
|
||||
#[cfg(feature = "smarm-trace")]
|
||||
@@ -46,47 +42,22 @@ mod inner {
|
||||
#[derive(Clone, Debug)]
|
||||
pub enum Event {
|
||||
// Actor lifecycle
|
||||
Spawn {
|
||||
parent: Pid,
|
||||
child: Pid,
|
||||
},
|
||||
Spawn { parent: Pid, child: Pid },
|
||||
Resume(Pid),
|
||||
Yield(Pid),
|
||||
Park(Pid),
|
||||
Done(Pid),
|
||||
/// Root exit found a live forest root (an actor nobody supervises)
|
||||
/// and delivered `request_shutdown` to it. `trapping` says whether it
|
||||
/// got the chance to drain (true) or was stopped outright (false).
|
||||
/// Every such line is an actor whose lifetime was nobody's business
|
||||
/// but the runtime's — the way to *see* unsupervised leftovers.
|
||||
RootSweep {
|
||||
target: Pid,
|
||||
trapping: bool,
|
||||
},
|
||||
// Wakeup paths
|
||||
UnparkDirect(Pid), // unpark() saw Parked -> re-queued immediately
|
||||
UnparkDeferred(Pid), // unpark() saw Runnable -> set pending_unpark flag
|
||||
UnparkFlagConsumed(Pid), // scheduler saw flag on Park -> re-queued instead
|
||||
// Channel
|
||||
Send {
|
||||
sender: Pid,
|
||||
receiver: Option<Pid>,
|
||||
},
|
||||
Send { sender: Pid, receiver: Option<Pid> },
|
||||
RecvPark(Pid),
|
||||
RecvWake(Pid),
|
||||
// Queue
|
||||
Enqueue(Pid),
|
||||
Dequeue(Pid),
|
||||
// RFC 005 wake slot
|
||||
SlotPush(Pid), // actor-context wake parked in the waking thread's slot
|
||||
SlotPop(Pid), // scheduler resumed a pid from its own slot
|
||||
// Cluster (RFC 010): the conn actor's verdict on one inbound frame —
|
||||
// local knowledge only, never on the wire; the label is
|
||||
// `InboundVerdict::label()`. No pid: a refused frame has none.
|
||||
ClusterInbound(&'static str),
|
||||
// Cluster (RFC 010): the connector's verdict on one dial attempt —
|
||||
// `"ok"` or `DialError::label()`. No pid.
|
||||
ClusterDial(&'static str),
|
||||
}
|
||||
|
||||
// -----------------------------------------------------------------------
|
||||
@@ -94,8 +65,8 @@ mod inner {
|
||||
// -----------------------------------------------------------------------
|
||||
|
||||
struct Record {
|
||||
nanos: u64, // ns since open()
|
||||
tid: u64, // OS thread id
|
||||
nanos: u64, // ns since open()
|
||||
tid: u64, // OS thread id
|
||||
event: Event,
|
||||
}
|
||||
|
||||
@@ -110,8 +81,8 @@ mod inner {
|
||||
// -----------------------------------------------------------------------
|
||||
|
||||
struct Global {
|
||||
sender: mpsc::Sender<Msg>,
|
||||
start: Instant,
|
||||
sender: mpsc::Sender<Msg>,
|
||||
start: Instant,
|
||||
}
|
||||
|
||||
static GLOBAL: Mutex<Option<Global>> = Mutex::new(None);
|
||||
@@ -121,13 +92,13 @@ mod inner {
|
||||
// The start Instant is copied alongside it — also one mutex hit per thread.
|
||||
// record() never touches GLOBAL after that.
|
||||
struct LocalState {
|
||||
tx: mpsc::Sender<Msg>,
|
||||
tx: mpsc::Sender<Msg>,
|
||||
start: Instant,
|
||||
}
|
||||
|
||||
thread_local! {
|
||||
static LOCAL_STATE: std::cell::RefCell<Option<LocalState>> =
|
||||
const { std::cell::RefCell::new(None) };
|
||||
std::cell::RefCell::new(None);
|
||||
}
|
||||
|
||||
// -----------------------------------------------------------------------
|
||||
@@ -135,26 +106,20 @@ mod inner {
|
||||
// -----------------------------------------------------------------------
|
||||
|
||||
pub fn open() {
|
||||
let path =
|
||||
std::env::var("SMARM_TRACE_FILE").unwrap_or_else(|_| "smarm_trace.json".to_owned());
|
||||
let path = std::env::var("SMARM_TRACE_FILE")
|
||||
.unwrap_or_else(|_| "smarm_trace.json".to_owned());
|
||||
|
||||
let (tx, rx) = mpsc::channel::<Msg>();
|
||||
let start = Instant::now();
|
||||
|
||||
match GLOBAL.lock() {
|
||||
Ok(mut g) => *g = Some(Global { sender: tx, start }),
|
||||
Err(e) => panic!("smarm: trace lock poisoned (core corrupt): {e}"),
|
||||
}
|
||||
*GLOBAL.lock().unwrap() = Some(Global { sender: tx, start });
|
||||
|
||||
// Drain thread: owns the Receiver, writes to disk.
|
||||
let path_for_thread = path.clone();
|
||||
match std::thread::Builder::new()
|
||||
std::thread::Builder::new()
|
||||
.name("smarm-trace-drain".into())
|
||||
.spawn(move || drain_thread(rx, &path_for_thread))
|
||||
{
|
||||
Ok(_) => {}
|
||||
Err(e) => panic!("smarm: failed to spawn trace drain thread: {e}"),
|
||||
}
|
||||
.expect("failed to spawn trace drain thread");
|
||||
|
||||
eprintln!("[smarm-trace] writing to {}", path);
|
||||
}
|
||||
@@ -165,10 +130,7 @@ mod inner {
|
||||
// Drop the global sender so the drain thread's recv() returns Err
|
||||
// after the Flush sentinel, signalling clean shutdown.
|
||||
let sender = {
|
||||
let mut g = match GLOBAL.lock() {
|
||||
Ok(g) => g,
|
||||
Err(e) => panic!("smarm: trace lock poisoned (core corrupt): {e}"),
|
||||
};
|
||||
let mut g = GLOBAL.lock().unwrap();
|
||||
g.take().map(|g| g.sender)
|
||||
};
|
||||
if let Some(tx) = sender {
|
||||
@@ -183,38 +145,33 @@ mod inner {
|
||||
// Hot path
|
||||
// -----------------------------------------------------------------------
|
||||
|
||||
#[inline(never)]
|
||||
pub fn record(event: Event) {
|
||||
crate::context::tls_fence();
|
||||
// Disable preemption for the entire duration of record(). Any
|
||||
// allocation here (mutex internals, channel send, lazy init) would
|
||||
// trigger PreemptingAllocator -> maybe_preempt -> switch_to_scheduler,
|
||||
// which would try to re-acquire inner.shared (already held at many
|
||||
// te!() call sites) -> deadlock. Guard at the very top, before any
|
||||
// allocation-capable call.
|
||||
let was_enabled = crate::preempt::preemption_swap(false);
|
||||
let was_enabled = crate::preempt::PREEMPTION_ENABLED
|
||||
.with(|e| { let v = e.get(); e.set(false); v });
|
||||
|
||||
LOCAL_STATE.with(|cell| {
|
||||
let mut opt = cell.borrow_mut();
|
||||
// Lazily initialise: one mutex hit per thread, ever.
|
||||
if opt.is_none() {
|
||||
let guard = match GLOBAL.lock() {
|
||||
Ok(g) => g,
|
||||
Err(e) => panic!("smarm: trace lock poisoned (core corrupt): {e}"),
|
||||
};
|
||||
if let Some(g) = guard.as_ref() {
|
||||
if let Some(g) = GLOBAL.lock().unwrap().as_ref() {
|
||||
let tx = g.sender.clone();
|
||||
*opt = Some(LocalState { tx, start: g.start });
|
||||
}
|
||||
}
|
||||
if let Some(ls) = opt.as_ref() {
|
||||
let nanos = ls.start.elapsed().as_nanos() as u64;
|
||||
let tid = os_tid();
|
||||
let tid = os_tid();
|
||||
let _ = ls.tx.send(Msg::Event(Record { nanos, tid, event }));
|
||||
}
|
||||
});
|
||||
|
||||
crate::preempt::preemption_swap(was_enabled);
|
||||
crate::preempt::PREEMPTION_ENABLED.with(|e| e.set(was_enabled));
|
||||
}
|
||||
|
||||
// -----------------------------------------------------------------------
|
||||
@@ -224,10 +181,7 @@ mod inner {
|
||||
fn drain_thread(rx: mpsc::Receiver<Msg>, path: &str) {
|
||||
let f = match std::fs::File::create(path) {
|
||||
Ok(f) => f,
|
||||
Err(e) => {
|
||||
eprintln!("[smarm-trace] create failed: {}", e);
|
||||
return;
|
||||
}
|
||||
Err(e) => { eprintln!("[smarm-trace] create failed: {}", e); return; }
|
||||
};
|
||||
let mut w = std::io::BufWriter::new(f);
|
||||
let _ = writeln!(w, "{{\"traceEvents\":[");
|
||||
@@ -240,9 +194,7 @@ mod inner {
|
||||
Ok(Msg::Event(r)) => {
|
||||
let (name, actor_idx) = chrome_fields(&r.event);
|
||||
let ts_us = r.nanos as f64 / 1000.0;
|
||||
if !first {
|
||||
let _ = w.write_all(b",\n");
|
||||
}
|
||||
if !first { let _ = w.write_all(b",\n"); }
|
||||
first = false;
|
||||
let _ = write!(w,
|
||||
"{{\"ph\":\"i\",\"ts\":{:.3},\"pid\":{},\"tid\":{},\"name\":{:?},\"s\":\"g\"}}",
|
||||
@@ -266,40 +218,25 @@ mod inner {
|
||||
|
||||
fn chrome_fields(ev: &Event) -> (String, u32) {
|
||||
match ev {
|
||||
Event::Spawn { parent, child } => {
|
||||
(format!("spawn c={}", child.index()), parent.index())
|
||||
}
|
||||
Event::Resume(p) => ("resume".into(), p.index()),
|
||||
Event::Yield(p) => ("yield".into(), p.index()),
|
||||
Event::Park(p) => ("park".into(), p.index()),
|
||||
Event::Done(p) => ("done".into(), p.index()),
|
||||
Event::RootSweep { target, trapping } => (
|
||||
format!(
|
||||
"root_sweep {}",
|
||||
if *trapping { "shutdown" } else { "stopped" }
|
||||
),
|
||||
target.index(),
|
||||
),
|
||||
Event::UnparkDirect(p) => ("unpark_direct".into(), p.index()),
|
||||
Event::UnparkDeferred(p) => ("unpark_deferred".into(), p.index()),
|
||||
Event::Spawn { parent, child } =>
|
||||
(format!("spawn c={}", child.index()), parent.index()),
|
||||
Event::Resume(p) => ("resume".into(), p.index()),
|
||||
Event::Yield(p) => ("yield".into(), p.index()),
|
||||
Event::Park(p) => ("park".into(), p.index()),
|
||||
Event::Done(p) => ("done".into(), p.index()),
|
||||
Event::UnparkDirect(p) => ("unpark_direct".into(), p.index()),
|
||||
Event::UnparkDeferred(p) => ("unpark_deferred".into(), p.index()),
|
||||
Event::UnparkFlagConsumed(p) => ("unpark_flag_consumed".into(), p.index()),
|
||||
Event::Send { sender, receiver } => (
|
||||
format!(
|
||||
"send rx={}",
|
||||
receiver
|
||||
.map(|p| p.index().to_string())
|
||||
.unwrap_or_else(|| "none".into())
|
||||
),
|
||||
format!("send rx={}", receiver
|
||||
.map(|p| p.index().to_string())
|
||||
.unwrap_or_else(|| "none".into())),
|
||||
sender.index(),
|
||||
),
|
||||
Event::RecvPark(p) => ("recv_park".into(), p.index()),
|
||||
Event::RecvWake(p) => ("recv_wake".into(), p.index()),
|
||||
Event::Enqueue(p) => ("enqueue".into(), p.index()),
|
||||
Event::Dequeue(p) => ("dequeue".into(), p.index()),
|
||||
Event::SlotPush(p) => ("slot_push".into(), p.index()),
|
||||
Event::SlotPop(p) => ("slot_pop".into(), p.index()),
|
||||
Event::ClusterInbound(v) => (format!("cluster_inbound {v}"), 0),
|
||||
Event::ClusterDial(v) => (format!("cluster_dial {v}"), 0),
|
||||
Event::Enqueue(p) => ("enqueue".into(), p.index()),
|
||||
Event::Dequeue(p) => ("dequeue".into(), p.index()),
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,263 @@
|
||||
# smarm — task.md (next steps)
|
||||
|
||||
Handoff for a future session reusing this sandbox. Read top to bottom once
|
||||
before starting; the gotchas section is hard-won and will save you a faceplant.
|
||||
|
||||
## Resume the environment
|
||||
|
||||
- Repo: `smarm`. Two branches (the old single `arm-port` stack was split):
|
||||
- `master` — the mainline, and HEAD. Carries roadmap #1–#5: cooperative
|
||||
cancellation, supervisor strategies (one_for_one/all, rest_for_one) + the
|
||||
orphaned-timer shutdown fix, links/trap_exit, selective receive, gen_server,
|
||||
and demonitor/`MonitorId`. Tagged `v0.4.0`. x86-64 Linux only.
|
||||
- `arm-port` — `master` plus a single commit: `feat(arch): aarch64 context
|
||||
switch + cycle counter`. Extracts the x86-64 context-switch / stack-init /
|
||||
cycle-counter out of `context.rs` into a `target_arch`-gated `src/arch/`
|
||||
(x86_64 + aarch64 backends) and adds an AAPCS64 backend. ⚠️ UNTESTED: never
|
||||
built or run on real ARM hardware. The x86-64 path is unchanged
|
||||
(`arch/x86_64.rs` is the old `context.rs` body verbatim), so the x86 suite
|
||||
passing says nothing about the aarch64 backend. Build + test on-device
|
||||
before trusting it.
|
||||
- Toolchain is installed but NOT on PATH in a fresh shell. First line of every
|
||||
session: `. "$HOME/.cargo/env"` (rustc/cargo 1.96).
|
||||
- Build `cargo build` · all tests `cargo test` · one suite `cargo test --test monitor`.
|
||||
- Bench probe `cargo bench --bench general` (custom print-only harness; compiles
|
||||
tokio in release the first time — slow — but `target/` persists across git
|
||||
checkouts so it's paid once).
|
||||
- Perf regression check: `git checkout <pre-change-sha>` →
|
||||
`cargo bench --bench general | tee before.txt` → `git checkout arm-port` →
|
||||
run again → diff the `smarm 1-thread` medians for `chained_spawn` and
|
||||
`yield_many` (those exercise spawn/finalize/scheduler). Numbers are noisy on
|
||||
this shared CPU; treat as "regression beyond noise?" not a precise delta.
|
||||
|
||||
## Roadmap (dependency order)
|
||||
|
||||
### 1. Cooperative cancellation — the keystone ✅ DONE (`a8ddb4a`)
|
||||
Everything below (one_for_all/rest_for_one, links) needs a *safe* way to stop a
|
||||
running peer. Forcible teardown of another green thread's stack is unsound here
|
||||
(shared heap + Drop). So: cooperative stop the actor observes and unwinds itself.
|
||||
|
||||
Shipped as designed (sentinel unwind, not Result-threading). Notes for what
|
||||
came next / future readers:
|
||||
- Stop flag lives on `Actor` behind `Arc<AtomicBool>` (fresh per spawn), NOT a
|
||||
`Slot` field — sidesteps the three-place reset, at the cost of one small
|
||||
alloc per spawn. The scheduler hands the resume path a raw `*const AtomicBool`
|
||||
(no per-resume refcount traffic); `yield_many` bench stayed at baseline,
|
||||
`chained_spawn` ~+6% from that alloc (left as-is; move to a `Slot` field if it
|
||||
ever matters).
|
||||
- Observation points: amortised `maybe_preempt`/`check!()` path + the wakeup
|
||||
side of `park_current`/`yield_now`. Sentinel = `StopSentinel` (zero-size),
|
||||
recognised in the trampoline → `Outcome::Stopped`. `join()` on a stopped actor
|
||||
returns `Ok(())` (no payload to propagate; reason is on the monitor channel).
|
||||
- Documented gaps confirmed by tests: no-observation-point loop can't be stopped
|
||||
(same as preemption); a user `catch_unwind` can swallow the sentinel but the
|
||||
flag stays set so the next yield re-raises.
|
||||
|
||||
Original plan, for reference:
|
||||
- Add a per-actor stop flag (Slot field + atomic, or check via shared state).
|
||||
- `request_stop(pid)`: set the flag, unpark if parked.
|
||||
- Realize the stop as a **controlled unwind**: when the scheduler resumes a
|
||||
stop-requested actor, inject a sentinel panic (dedicated payload type) so the
|
||||
existing `trampoline` `catch_unwind` tears the stack down and runs Drop. The
|
||||
trampoline recognizes the sentinel and reports a new `Outcome::Stopped`
|
||||
(distinct from a user `Panic`). This avoids changing every blocking-op
|
||||
signature.
|
||||
- Alternative considered: thread `Result<_, Cancelled>` through recv/sleep/
|
||||
lock/io. Rejected — large API churn. Go with the sentinel unwind.
|
||||
- Caveat to document: user code with its own `catch_unwind` can swallow the
|
||||
sentinel (cf. Erlang `catch`); re-check the flag at the next yield and/or
|
||||
re-raise. And a tight no-alloc loop without `check!()` can't be stopped —
|
||||
same inherent limitation as preemption.
|
||||
- Observation points: `maybe_preempt()`/`check!()` (cheap flag check) and the
|
||||
blocking parks (recv/sleep/mutex/io) on the stop-driven unpark.
|
||||
- Tests: looping actor on `check!()` gets stopped → `Outcome::Stopped`, Drop
|
||||
guards ran; parked-on-recv actor gets stopped; no-check loop documents the gap.
|
||||
|
||||
### 2. one_for_all / rest_for_one + ordered shutdown ✅ DONE (`351dc9c`)
|
||||
Shipped. What landed vs the plan:
|
||||
- `Strategy::{OneForOne,OneForAll,RestForOne}` selected via `.strategy()`,
|
||||
default `OneForOne`. The struct keeps the `OneForOne` name (compat; existing
|
||||
tests untouched) — a rename to `Supervisor` is a deferred refactor.
|
||||
- The *triggering* child's `Restart` policy decides whether anything restarts;
|
||||
the strategy decides which live siblings are cycled (all / index-> after the
|
||||
failed one). Survivors are `request_stop`'d in reverse start order, awaited on
|
||||
the existing `supervisor_channel` funnel (no new channel, no `select`),
|
||||
restarted in start order. One failure = one intensity tick regardless of group
|
||||
size. Out-of-band signals during an await are stashed and replayed.
|
||||
- `Signal::Stopped(pid)` + `DownReason::Stopped` added (kept distinct from Exit,
|
||||
as planned). A `Stopped` signal counts as abnormal for the restart decision.
|
||||
- Ordered shutdown: on cap-trip / mailbox-close-with-survivors, stop remaining
|
||||
children in reverse start order and await them (no-op on the normal exit).
|
||||
- ⚠️ Surfaced + fixed a latent keystone bug (`e80334b`): a cancelled
|
||||
sleeping/timeout actor orphans its timer entry, and the scheduler's shutdown
|
||||
check counted pending timers → `run()` hung until the dead actor's deadline
|
||||
fired (a `sleep(30s)` sleeper hung shutdown 30s). Fix: timers no longer gate
|
||||
shutdown (`live == 0` already implies nothing a timer could wake); heap is
|
||||
cleared on exit. Independent of the supervisor work.
|
||||
- Tests: all-restart (sibling cycled despite clean exit), suffix-restart
|
||||
(prefix child left alone), reverse-order teardown.
|
||||
|
||||
Original plan, for reference:
|
||||
- one_for_all: on any child failure, `request_stop` all siblings, await their
|
||||
termination signals, restart all per spec.
|
||||
- rest_for_one: stop+restart the failed child and those started after it.
|
||||
- Supervisor shutdown: stop children in reverse start order.
|
||||
- Decide signal surface: add `Signal::Stopped(pid)` + `DownReason::Stopped`
|
||||
rather than folding into Exit (clearer for the supervisor's await logic).
|
||||
- Tests: all-restart, suffix-restart, reverse-order shutdown.
|
||||
|
||||
### 3. Links + trap_exit ✅ DONE (`6581484`)
|
||||
- `Slot.links: Vec<Pid>` (bidirectional); `link`/`unlink`; `trap_exit()` flag
|
||||
lives on `Actor` (fresh per spawn → a restarted child starts un-trapped, and
|
||||
no fourth slot-reset site).
|
||||
- On finalize, reverse links are cleared under the lock (always — keeps the
|
||||
cascade acyclic), then for each linked peer: abnormal death (`Panic`/
|
||||
`Stopped`) → `request_stop(peer)` unless peer traps, in which case deliver an
|
||||
`ExitSignal` *message* instead. Normal exit does NOT propagate. Linking an
|
||||
already-dead pid delivers an immediate `NoProc` signal (message if trapping,
|
||||
else `request_stop(self)` — not a silent no-op).
|
||||
- Resolved: the trap inbox is a **dedicated** channel (`trap_exit() ->
|
||||
Receiver<ExitSignal>`), distinct from the monitor `Down` channel; `ExitSignal`
|
||||
reuses `DownReason` and carries no panic payload (joiner-only, as with
|
||||
monitors). `spawn_link` deferred to #5.
|
||||
- Tests (`tests/link.rs`): linked pair one panics → other stopped (+Drop ran);
|
||||
trap_exit → other gets a message and survives; normal exit doesn't propagate;
|
||||
dead-pid link stops a non-trapper / messages a trapper; `unlink` prevents
|
||||
propagation.
|
||||
|
||||
### 4. Selective receive (independent track) ✅ DONE (`03f3875`)
|
||||
Shipped. What landed vs the plan:
|
||||
- `Receiver::recv_match(pred) -> Result<T, RecvError>` scans the queued
|
||||
`VecDeque` front-to-back, removes+returns the first match, leaves the rest in
|
||||
arrival order; parks and re-scans when nothing matches. `try_recv_match` (the
|
||||
non-blocking variant, mirroring `try_recv`) rolled in same commit.
|
||||
- Wakeup turned out cheaper than feared: `Sender::send` *already* took
|
||||
`parked_receiver` on every push, so "wake on ANY send" needed no send-side
|
||||
change. The one real edit was relaxing `Sender::drop` to unpark the parked
|
||||
receiver on the last-sender drop regardless of queue emptiness — a selective
|
||||
receiver can park on a non-empty no-match queue and must wake to observe
|
||||
closure. No-op for plain `recv` (only ever parks on an empty queue); stress
|
||||
suite stays green.
|
||||
- `pred` is `Fn(&T) -> bool` (not `FnMut`) on purpose: it's re-run from scratch
|
||||
on every scan, so a stateful predicate would re-count surprisingly. It runs
|
||||
under the channel lock — keep it cheap/pure, don't re-enter the channel.
|
||||
- Close semantics: `recv_match` returns `Err(RecvError)` only when closed AND no
|
||||
queued message matches; a match is still returned on a closed channel.
|
||||
Non-matches are left for a later `recv`.
|
||||
- Tests (`tests/selective_recv.rs`): out-of-order match pulled first; non-matches
|
||||
remain in order; park-on-non-empty then wake on a match; closed-with-only-
|
||||
non-matches → Err; closed-but-match-present → match; `try_recv_match` states.
|
||||
|
||||
Original plan, for reference:
|
||||
- Add `Receiver::recv_match(pred) -> T`: scan the queued `VecDeque`, remove+return
|
||||
first match, leave the rest in order; park and re-scan on new arrivals.
|
||||
- This changes channel wakeup: a parked selective receiver must wake on ANY send
|
||||
(not just empty→nonempty) and re-scan. Touches `channel.rs` carefully — the
|
||||
stress tests guard lost-wakeup invariants; keep them green.
|
||||
- Tests: messages arrive out of interest-order; match pulled first; non-matches
|
||||
remain for a later `recv`.
|
||||
|
||||
### 5. Grab-bag (each its own small commit)
|
||||
- `spawn_link`: spawn-and-link atomically (deferred from #3); thin wrapper over
|
||||
`spawn_under` + `link`, but do it under one lock so there's no window where
|
||||
the child dies before the link is recorded.
|
||||
- `demonitor`: needs a per-monitor id to remove a specific sender. Decide the
|
||||
monitor API NOW before more code depends on it — likely return a
|
||||
`Monitor { id, rx }` instead of a bare `Receiver<Down>`.
|
||||
✅ DONE (this commit). What landed vs the plan:
|
||||
- `monitor()` now returns `Monitor { id, target, rx }` (added `target` over
|
||||
the sketched `{id, rx}` so `demonitor` jumps straight to the slot instead of
|
||||
scanning every slot for the id). `MonitorId(u64)` is opaque, from a
|
||||
monotonic `next_monitor_id` counter on `SharedState`, bumped under the
|
||||
shared lock in `monitor()` — no atomics, deterministic, never reused.
|
||||
- `Slot.monitors: Vec<(MonitorId, Sender<Down>)>`. The three slot-reset sites
|
||||
were untouched — they `.clear()`/`Vec::new()`, which is element-type-
|
||||
agnostic, so no new reset obligation. `finalize_actor` just destructures
|
||||
`(_, m)` and sends as before.
|
||||
- `demonitor(&Monitor) -> Option<MonitorId>`: `Some(id)` when a live
|
||||
registration was found+removed, `None` when it had already fired (drained on
|
||||
finalize), was `NoProc`, or the slot was reclaimed. Chose `Option<MonitorId>`
|
||||
over a bare bool — names which registration went. Generation half of the pid
|
||||
makes a stale demonitor a clean no-op: a recycled slot index fails
|
||||
`slot_mut`'s generation check, so it can never strip a different actor's
|
||||
monitor.
|
||||
- ⚠️ Reentrancy: the removed `Sender` is `remove`d out of the Vec under the
|
||||
lock but **dropped after the lock is released** — `Sender::drop` can unpark a
|
||||
parked receiver → `with_shared`, and the shared mutex is non-reentrant. Same
|
||||
discipline as `finalize_actor`.
|
||||
- "Flush" (discard a `Down` the target already queued) falls out of dropping
|
||||
the `Monitor`: `demonitor(&m); drop(m)`. That's the cleanup the still-to-come
|
||||
gen_server **call timeout** wants — monitor the server, wait reply-or-Down-
|
||||
or-deadline, then demonitor+drop so a timed-out call leaks no registration
|
||||
and no stale `Down`.
|
||||
- Perf: touches `Slot` + `finalize_actor`, but `chained_spawn`/`yield_many`
|
||||
register no monitors, so the Vec stays empty (take-empty is identical cost,
|
||||
finalize loop runs zero times). before/after `general` probe medians within
|
||||
noise. Tests (`tests/monitor.rs`): demonitor-stops-delivery, one-of-many
|
||||
(siblings untouched), after-fire-is-None.
|
||||
- Named registry: `register(name,pid)`/`whereis`/`send_by_name`; a
|
||||
`HashMap<String,Pid>` in `SharedState`.
|
||||
- gen_server-style call/cast: request-reply correlation as a thin layer over
|
||||
channels (`call` sends `{req, reply_tx}`, awaits `reply_rx`); no runtime change.
|
||||
✅ DONE (`a4fcf6c`). What landed vs the plan:
|
||||
- `GenServer` trait on the state value: assoc `Call`/`Reply`/`Cast` types,
|
||||
required `handle_call`/`handle_cast`, optional `init`/`terminate` hooks.
|
||||
`ServerRef<G>` is a clonable inbox sender + `pid()`; `start` / `start_under`.
|
||||
- One inbox, not two: a single `Envelope { Call(req, reply_tx) | Cast }`
|
||||
channel, dispatched by variant. Forced by no-`select`/no-unified-mailbox —
|
||||
a server can't wait on a call channel and a cast channel at once.
|
||||
- Server-down falls out of channel closure (no monitor needed): `send` fails
|
||||
if the inbox is gone; the reply sender drops on the server's unwind so a
|
||||
parked caller wakes to `Err`. Both → `Call/CastError::ServerDown`.
|
||||
- `terminate` runs via a drop guard → fires on *every* exit path (clean inbox
|
||||
close, handler panic, `request_stop`), not just the clean one. Caveat: it
|
||||
may run mid-unwind, so keep it non-blocking (a panic inside it during an
|
||||
unwind double-panics → abort).
|
||||
- No `handle_info`, no call timeout — both deferred to land with timeouts
|
||||
(`handle_info` needs the still-unmade cross-channel mailbox merge; a call
|
||||
timeout needs a per-`recv` deadline / `Signal::Timeout`).
|
||||
- Pure additive layer (no Slot/scheduler/spawn/finalize change) → no perf
|
||||
check run. Tests (`tests/gen_server.rs`): cast→call roundtrip, init/terminate
|
||||
ordering, both server-down paths.
|
||||
- Docs: README now points at the experimental, untested aarch64 port on the
|
||||
`arm-port` branch. The module table still calls `context` x86-64-only — true
|
||||
for `master`, since the `src/arch/` split rides on `arm-port`. Fold the arch/
|
||||
split and ARM64-supported wording into the README module table + build section
|
||||
once `arm-port` is validated on hardware and merged.
|
||||
|
||||
## Gotchas / invariants (respect these)
|
||||
|
||||
- **Shared mutex is non-reentrant.** `Sender::send` can call `unpark` →
|
||||
`with_shared`. NEVER send on a channel while holding the shared lock. Pattern:
|
||||
`mem::take` the senders/data under the lock, send after releasing. See
|
||||
`finalize_actor` (supervisor signal + monitor Downs both sent post-lock).
|
||||
- **`finalize_actor` order:** take stack/waiters/monitors under lock + set
|
||||
Done/outcome → recycle stack (post-lock) → deliver supervisor Signal + monitor
|
||||
Downs (post-lock) → unpark joiners → reclaim slot iff `outstanding_handles==0`.
|
||||
Death notifications always precede reclamation, so a pid carried in a
|
||||
Signal/Down is still matchable even as its slot is about to be reused.
|
||||
- **Slot lifecycle is reset in THREE places** — `Slot::vacant()`,
|
||||
`reclaim_slot()` (runtime.rs), and the slot-init block in `spawn_under`
|
||||
(scheduler.rs). Any new Slot field must be reset in all three (monitors was).
|
||||
- **Pid = (index, generation);** stale handles caught by generation mismatch in
|
||||
`slot()/slot_mut()`. The monitor `NoProc` path relies on this.
|
||||
- **No `select`, no unified per-process mailbox.** Why the supervisor uses the
|
||||
single `supervisor_channel` funnel rather than N monitor channels. trap_exit
|
||||
resolved this by giving each trapping actor a dedicated `Receiver<ExitSignal>`
|
||||
inbox (see #3); selective receive (#4) stayed *per-channel* (`recv_match`
|
||||
scans one channel's queue) rather than introducing a cross-channel mailbox —
|
||||
if selective receive ever needs to span the monitor/trap inboxes too, that
|
||||
cross-channel merge is the still-unmade decision.
|
||||
- **Cooperative-only**: preemption and (future) cancellation both depend on the
|
||||
actor reaching `check!()`/yield/alloc/blocking points.
|
||||
- `run()` is single-thread (`Config::exact(1)`); tests rely on deterministic
|
||||
single-thread ordering (parent runs until it parks). Multi-thread via
|
||||
`runtime::init(Config…)`.
|
||||
|
||||
## Workflow expectations (from the human)
|
||||
|
||||
- TDD: write the failing test first, then implement.
|
||||
- Commit incrementally with conventional-commit messages; keep each commit a
|
||||
reviewable unit (they diff in their IDE and are the filter to the codebase).
|
||||
- Run the full suite before each commit; check perf when a change touches
|
||||
Slot/scheduler/spawn/finalize hot paths.
|
||||
-120
@@ -1,120 +0,0 @@
|
||||
# Tests
|
||||
|
||||
Integration tests for the runtime. Each file owns one feature area or one
|
||||
class of bug. Everything here runs under plain `cargo test`; the loom model
|
||||
tests are the exception — they live **in the library** (`src/slot_state.rs`,
|
||||
`src/run_queue.rs`), not in this directory, because loom must compile the
|
||||
production code with shimmed atomics (see "Loom" below).
|
||||
|
||||
## Running
|
||||
|
||||
```
|
||||
cargo test # debug build — RUN THIS ONE: all invariant asserts live
|
||||
cargo test --release # what users actually execute (LTO, no debug_asserts)
|
||||
```
|
||||
|
||||
Debug builds are not just "slower tests": the runtime self-checks its
|
||||
invariants only there — every `StateWord` transition asserts its
|
||||
precondition, `enqueue` asserts the exact `(gen, Queued)` word, `RawMutex`
|
||||
enforces the never-two-cold-locks leaf rule with a per-thread held-count,
|
||||
`live_actors` checks for double-finalize underflow. A green release run with
|
||||
a red debug run means an invariant broke without (yet) corrupting behavior —
|
||||
treat it as a real failure.
|
||||
|
||||
### The queue-variant matrix
|
||||
|
||||
The run queue is compile-time selected; the suite must pass under all three
|
||||
(features are additive, so drop the default first):
|
||||
|
||||
```
|
||||
cargo test # rq-mutex (default)
|
||||
cargo test --no-default-features --features rq-mpmc
|
||||
cargo test --no-default-features --features rq-striped
|
||||
```
|
||||
|
||||
### Loom (model checking)
|
||||
|
||||
```
|
||||
RUSTFLAGS="--cfg loom" cargo test --lib --release
|
||||
```
|
||||
|
||||
Exhaustively explores interleavings of the slot state machine
|
||||
(`src/slot_state.rs`: lost-wakeup, at-most-once-enqueue, the stale-pid ABA
|
||||
theorem, unpark-vs-claim) and the ring queues (`src/run_queue.rs`:
|
||||
exactly-once through lap wraparound, push/pop races). Models run the
|
||||
production transitions through `src/sync_shim.rs` — std atomics normally,
|
||||
`loom::sync` under `--cfg loom`. `RawMutex` is deliberately not modeled:
|
||||
futexes can't be, and it's the textbook Drepper mutex3 with stress and
|
||||
unwind-safety tests of its own.
|
||||
|
||||
### Trace feature
|
||||
|
||||
`cargo test --features smarm-trace` exists mainly to catch bit-rot in the
|
||||
`te!()` call sites; run it after touching scheduler paths.
|
||||
|
||||
### Before a runtime-core PR
|
||||
|
||||
The full matrix, in rough order of bug-finding power per minute:
|
||||
|
||||
1. `cargo test` (debug, default variant)
|
||||
2. debug under `rq-mpmc` and `rq-striped`
|
||||
3. `cargo test --release`
|
||||
4. loom
|
||||
5. `cargo build --features smarm-trace`
|
||||
|
||||
## Catalog
|
||||
|
||||
**Low-level units (no scheduler)**
|
||||
| file | covers |
|
||||
|---|---|
|
||||
| `context.rs` | `init_actor_stack` + the naked-asm context-switch shims, poked directly |
|
||||
| `stack.rs` | the mmap'd stack allocator |
|
||||
| `pid.rs` | pid packing/equality |
|
||||
|
||||
**Feature areas (run under a real runtime)**
|
||||
| file | covers |
|
||||
|---|---|
|
||||
| `runtime.rs` | `Config`, `Runtime::run`, re-running a runtime, correctness under genuine parallelism |
|
||||
| `scheduler.rs` | spawn / join / panic delivery / `yield_now` / `self_pid` |
|
||||
| `channel.rs` | send/recv (recv parks, so these need the runtime) |
|
||||
| `selective_recv.rs` | `recv_match` / `try_recv_match` |
|
||||
| `mutex.rs` | the actor-blocking `Mutex<T>` (lock parks) |
|
||||
| `timer.rs` | `sleep` ordering — time-sensitive, generous tolerances by design |
|
||||
| `io.rs` | `block_on_io`: blocking closures on the pool while the actor parks |
|
||||
| `io_epoll.rs` | `wait_readable` / `wait_writable` + the `read`/`write` sugar |
|
||||
| `preempt.rs` | explicit preemption via `smarm::check!()` |
|
||||
| `cancel.rs` | cooperative cancellation (`request_stop`) — the keystone semantics |
|
||||
| `monitor.rs` | `monitor` delivers exactly one `Down`; `demonitor` |
|
||||
| `link.rs` | bidirectional links + `trap_exit` |
|
||||
| `supervisor.rs` | one-for-one supervision |
|
||||
| `gen_server.rs` | call/cast round-trips, lifecycle callbacks, server-down detection |
|
||||
|
||||
**Regression & stress**
|
||||
| file | covers |
|
||||
|---|---|
|
||||
| `stress.rs` | lost wakeups, pid-table pressure, thundering herds, panic isolation under concurrency. Where the phase-2 RefCell-migration bug was caught. |
|
||||
| `poison_stop.rs` | `request_stop` racing an alloc-under-lock must not poison/abort. See its header for the full story. |
|
||||
| `many_timers_multi_thread.rs` | multi-thread sleep-timer lost-wakeup regression |
|
||||
|
||||
## Conventions
|
||||
|
||||
- **Each test owns its runtime.** `init(Config::exact(N))` + `rt.run(...)`;
|
||||
never share a `Runtime` between tests. Oversubscription (`exact(4)` on one
|
||||
core) is deliberate — forced interleaving at yield points is how
|
||||
single-core CI finds races at all.
|
||||
- **Regression tests must be validated against the bug.** A regression test
|
||||
that passes with the bug reintroduced is documentation, not a test.
|
||||
Reintroduce the fix's inverse locally and watch it fail before trusting it
|
||||
(`poison_stop.rs` went through exactly this: its first version never fired
|
||||
the sentinel under a lock, and was rewritten until it SIGABRT'd pre-fix).
|
||||
- **Stochastic tests get the odds stacked.** Use `Config::alloc_interval(1)`
|
||||
to make every allocation an observation point, many actors, and both
|
||||
phases of any every-other-allocation cadence (see
|
||||
`poison_stop::self_stop_during_spawn...`).
|
||||
- **Time-based assertions use ordering, not durations.** Assert
|
||||
"didn't return instantly" / "A woke before B", with generous tolerances;
|
||||
CI machines are slow and noisy.
|
||||
- New invariants added to the runtime should come with the assert at the
|
||||
point of reliance (debug_assert on hot paths) *and*, where the invariant is
|
||||
a protocol, a loom model in the owning module — that combination is what
|
||||
made phases 2–5 land without a single post-merge race so far.
|
||||
+4
-80
@@ -49,14 +49,8 @@ fn looping_actor_on_check_is_stopped() {
|
||||
}
|
||||
let _ = h.join();
|
||||
});
|
||||
assert!(
|
||||
saw_stopped.load(Ordering::SeqCst),
|
||||
"expected DownReason::Stopped"
|
||||
);
|
||||
assert!(
|
||||
dropped.load(Ordering::SeqCst),
|
||||
"Drop guard must run during the cancellation unwind"
|
||||
);
|
||||
assert!(saw_stopped.load(Ordering::SeqCst), "expected DownReason::Stopped");
|
||||
assert!(dropped.load(Ordering::SeqCst), "Drop guard must run during the cancellation unwind");
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -85,14 +79,8 @@ fn parked_on_recv_actor_is_stopped() {
|
||||
}
|
||||
let _ = h.join();
|
||||
});
|
||||
assert!(
|
||||
saw_stopped.load(Ordering::SeqCst),
|
||||
"expected DownReason::Stopped"
|
||||
);
|
||||
assert!(
|
||||
dropped.load(Ordering::SeqCst),
|
||||
"Drop guard must run on cancellation of a parked actor"
|
||||
);
|
||||
assert!(saw_stopped.load(Ordering::SeqCst), "expected DownReason::Stopped");
|
||||
assert!(dropped.load(Ordering::SeqCst), "Drop guard must run on cancellation of a parked actor");
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -142,67 +130,3 @@ fn join_on_stopped_actor_returns_ok() {
|
||||
assert!(h.join().is_ok(), "join on a stopped actor returns Ok(())");
|
||||
});
|
||||
}
|
||||
|
||||
/// Regression: `request_stop` against a QUEUED actor must not be lossy.
|
||||
///
|
||||
/// The stop flag is set, but the wildcard unpark no-ops on a Queued actor
|
||||
/// (the pending run "is" the wake). If the actor's first action on resume
|
||||
/// is a blocking park — no allocation, no `check!()` on the way — a
|
||||
/// wake-side-only check in `park_current` never runs: the actor parks with
|
||||
/// the stop flag already set, and nothing will ever wake it. The runtime
|
||||
/// then idles forever (the root is parked on the monitor channel).
|
||||
///
|
||||
/// Fix: an entry-side `check_cancelled` in `park_current`. The remaining
|
||||
/// window (flag set after the entry check, before the park lands) is closed
|
||||
/// by the existing protocol: the stop's unpark then finds Running /
|
||||
/// the prep-to-park window, sets Notified, and the park-return re-queues
|
||||
/// into the wake-side check.
|
||||
///
|
||||
/// Watchdog harness: without the fix this deadlocks, so the runtime runs on
|
||||
/// a side thread and the test fails on a timeout instead of hanging cargo.
|
||||
#[test]
|
||||
fn stop_flagged_while_queued_lands_at_first_park() {
|
||||
use std::sync::mpsc;
|
||||
use std::time::Duration;
|
||||
|
||||
let dropped = Arc::new(AtomicBool::new(false));
|
||||
let saw_stopped = Arc::new(AtomicBool::new(false));
|
||||
let (d, s) = (dropped.clone(), saw_stopped.clone());
|
||||
|
||||
let (done_tx, done_rx) = mpsc::channel::<()>();
|
||||
std::thread::spawn(move || {
|
||||
let rt = smarm::init(smarm::Config::exact(1));
|
||||
rt.run(move || {
|
||||
let h = spawn(move || {
|
||||
let _g = DropFlag(d);
|
||||
let (tx, rx) = channel::<u8>();
|
||||
let _keep = tx; // keep the channel open: recv() parks
|
||||
let _ = rx.recv(); // first observation point is this park
|
||||
});
|
||||
let pid = h.pid();
|
||||
let down = monitor(pid);
|
||||
// Stop while the child is still QUEUED — before it ever runs.
|
||||
// The unpark no-ops; only the flag is left behind.
|
||||
request_stop(pid);
|
||||
let dn = down.rx.recv().expect("monitor channel closed before Down");
|
||||
assert_eq!(dn.pid, pid);
|
||||
if matches!(dn.reason, DownReason::Stopped) {
|
||||
s.store(true, Ordering::SeqCst);
|
||||
}
|
||||
let _ = h.join();
|
||||
});
|
||||
let _ = done_tx.send(());
|
||||
});
|
||||
done_rx
|
||||
.recv_timeout(Duration::from_secs(10))
|
||||
.expect("runtime deadlocked: stop against a QUEUED actor was lost at its first park");
|
||||
|
||||
assert!(
|
||||
saw_stopped.load(Ordering::SeqCst),
|
||||
"expected DownReason::Stopped"
|
||||
);
|
||||
assert!(
|
||||
dropped.load(Ordering::SeqCst),
|
||||
"Drop guard must run during the cancellation unwind"
|
||||
);
|
||||
}
|
||||
|
||||
-1039
File diff suppressed because it is too large
Load Diff
@@ -108,199 +108,3 @@ fn recv_returns_err_when_all_senders_dropped() {
|
||||
|
||||
assert!(saw_err.load(std::sync::atomic::Ordering::SeqCst));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn channel_ops_interleaved_with_monitor_churn_multi_thread() {
|
||||
// Regression for the RawMutex migration: monitor registration clones the
|
||||
// Down sender under the target's cold (Leaf) lock, which now nests a
|
||||
// Channel-class lock under it. Debug builds enforce the Leaf -> Channel
|
||||
// ordering on every acquisition, so driving channels, monitors, and actor
|
||||
// death concurrently across schedulers makes any ordering regression
|
||||
// panic here rather than deadlock in the field.
|
||||
use std::sync::atomic::{AtomicI64, Ordering};
|
||||
use std::sync::Arc;
|
||||
|
||||
let total = Arc::new(AtomicI64::new(0));
|
||||
let total2 = total.clone();
|
||||
smarm::init(smarm::Config::exact(4)).run(move || {
|
||||
let (tx, rx) = channel::<i64>();
|
||||
let consumer = spawn(move || {
|
||||
let mut sum = 0;
|
||||
while let Ok(v) = rx.recv() {
|
||||
sum += v;
|
||||
}
|
||||
OUT.with(|c| c.set(sum)); // not asserted cross-thread; see total
|
||||
total2.fetch_add(sum, Ordering::Relaxed);
|
||||
});
|
||||
|
||||
let mut handles = Vec::new();
|
||||
for i in 0..32i64 {
|
||||
let tx = tx.clone();
|
||||
handles.push(spawn(move || {
|
||||
// Short-lived target whose death fires the monitor below.
|
||||
// spawn_monitor: registered before publish, so the Down is
|
||||
// the finalize-sent Exit this test is about, never NoProc
|
||||
// (spawn-then-monitor raced ~8% at 4 threads).
|
||||
let (t, m) = smarm::spawn_monitor(move || {
|
||||
tx.send(i).unwrap();
|
||||
});
|
||||
t.join().unwrap();
|
||||
// Down delivery exercises send-from-finalize.
|
||||
let d = m.rx.recv().unwrap();
|
||||
assert_eq!(d.reason, smarm::DownReason::Exit);
|
||||
}));
|
||||
}
|
||||
drop(tx);
|
||||
for h in handles {
|
||||
h.join().unwrap();
|
||||
}
|
||||
consumer.join().unwrap();
|
||||
});
|
||||
assert_eq!(
|
||||
total.load(std::sync::atomic::Ordering::Relaxed),
|
||||
(0..32).sum::<i64>()
|
||||
);
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// recv_timeout
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
use smarm::RecvTimeoutError;
|
||||
use std::time::{Duration, Instant};
|
||||
|
||||
#[test]
|
||||
fn recv_timeout_returns_queued_message_immediately() {
|
||||
run(|| {
|
||||
let (tx, rx) = channel::<i64>();
|
||||
tx.send(5).unwrap();
|
||||
assert_eq!(rx.recv_timeout(Duration::from_secs(10)), Ok(5));
|
||||
});
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn recv_timeout_times_out_on_silent_channel() {
|
||||
run(|| {
|
||||
let (_tx, rx) = channel::<i64>();
|
||||
let start = Instant::now();
|
||||
let r = rx.recv_timeout(Duration::from_millis(50));
|
||||
assert_eq!(r, Err(RecvTimeoutError::Timeout));
|
||||
assert!(start.elapsed() >= Duration::from_millis(50));
|
||||
});
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn recv_timeout_wakes_promptly_on_send() {
|
||||
run(|| {
|
||||
let (tx, rx) = channel::<i64>();
|
||||
let h = spawn(move || {
|
||||
let start = Instant::now();
|
||||
assert_eq!(rx.recv_timeout(Duration::from_secs(10)), Ok(9));
|
||||
// Far below the timeout: the send woke us, not the deadline.
|
||||
assert!(start.elapsed() < Duration::from_secs(1));
|
||||
});
|
||||
smarm::yield_now();
|
||||
tx.send(9).unwrap();
|
||||
h.join().unwrap();
|
||||
});
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn recv_timeout_reports_disconnected_on_close() {
|
||||
run(|| {
|
||||
let (tx, rx) = channel::<i64>();
|
||||
let h = spawn(move || {
|
||||
assert_eq!(
|
||||
rx.recv_timeout(Duration::from_secs(10)),
|
||||
Err(RecvTimeoutError::Disconnected)
|
||||
);
|
||||
});
|
||||
smarm::yield_now();
|
||||
drop(tx);
|
||||
h.join().unwrap();
|
||||
});
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn recv_timeout_zero_duration_is_a_bounded_poll() {
|
||||
run(|| {
|
||||
let (_tx, rx) = channel::<i64>();
|
||||
assert_eq!(
|
||||
rx.recv_timeout(Duration::ZERO),
|
||||
Err(RecvTimeoutError::Timeout)
|
||||
);
|
||||
});
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn channel_remains_usable_after_a_timeout() {
|
||||
// The stale timer entry from the first (timed-out) wait must not cancel
|
||||
// or corrupt later waits — seq isolation.
|
||||
run(|| {
|
||||
let (tx, rx) = channel::<i64>();
|
||||
assert_eq!(
|
||||
rx.recv_timeout(Duration::from_millis(10)),
|
||||
Err(RecvTimeoutError::Timeout)
|
||||
);
|
||||
// Plain recv still works...
|
||||
tx.send(1).unwrap();
|
||||
assert_eq!(rx.recv(), Ok(1));
|
||||
// ...and so does a second bounded wait, woken by a send.
|
||||
let h = spawn(move || {
|
||||
tx.send(2).unwrap();
|
||||
});
|
||||
assert_eq!(rx.recv_timeout(Duration::from_secs(10)), Ok(2));
|
||||
h.join().unwrap();
|
||||
});
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn recv_timeout_many_waiters_multi_thread() {
|
||||
// Mixed outcomes under real parallelism: half the channels get fed,
|
||||
// half time out; every actor must resolve correctly.
|
||||
use std::sync::atomic::{AtomicU32, Ordering};
|
||||
use std::sync::Arc;
|
||||
|
||||
let got = Arc::new(AtomicU32::new(0));
|
||||
let timed_out = Arc::new(AtomicU32::new(0));
|
||||
let (got2, timed_out2) = (got.clone(), timed_out.clone());
|
||||
smarm::init(smarm::Config::exact(4)).run(move || {
|
||||
let mut handles = Vec::new();
|
||||
for i in 0..24i64 {
|
||||
let (tx, rx) = channel::<i64>();
|
||||
let got = got2.clone();
|
||||
let timed_out = timed_out2.clone();
|
||||
handles.push(spawn(move || {
|
||||
match rx.recv_timeout(Duration::from_millis(100)) {
|
||||
Ok(v) => {
|
||||
assert_eq!(v, i);
|
||||
got.fetch_add(1, Ordering::Relaxed);
|
||||
}
|
||||
Err(RecvTimeoutError::Timeout) => {
|
||||
timed_out.fetch_add(1, Ordering::Relaxed);
|
||||
}
|
||||
Err(e) => panic!("unexpected: {e}"),
|
||||
}
|
||||
}));
|
||||
if i % 2 == 0 {
|
||||
handles.push(spawn(move || {
|
||||
tx.send(i).unwrap();
|
||||
}));
|
||||
}
|
||||
// odd i: tx drops here -> Disconnected, not Timeout! Keep it alive
|
||||
// instead by leaking the sender into a holder actor that outlives
|
||||
// the deadline.
|
||||
else {
|
||||
handles.push(spawn(move || {
|
||||
smarm::sleep(Duration::from_millis(200));
|
||||
drop(tx);
|
||||
}));
|
||||
}
|
||||
}
|
||||
for h in handles {
|
||||
h.join().unwrap();
|
||||
}
|
||||
});
|
||||
assert_eq!(got.load(std::sync::atomic::Ordering::Relaxed), 12);
|
||||
assert_eq!(timed_out.load(std::sync::atomic::Ordering::Relaxed), 12);
|
||||
}
|
||||
|
||||
@@ -1,117 +0,0 @@
|
||||
//! RFC 010 c6a — connection-actor lifecycle against the manager table.
|
||||
//!
|
||||
//! The handshake is bypassed here (c6b wires it): each connection is
|
||||
//! constructed already-established over a real localhost TCP pair, handed a
|
||||
//! fabricated `Peer`, and spawned. `spawn_established` registers it with the
|
||||
//! manager, which takes its handle and monitors it, so the table reflects the
|
||||
//! connection while it lives and reaps it on any exit path. This proves three
|
||||
//! things at once: a live connection shows up, a commanded `Disconnect`
|
||||
//! removes exactly that one, and a peer close (EOF, no command) removes the
|
||||
//! other.
|
||||
//!
|
||||
//! TCP parks the calling actor, so everything runs inside `smarm::run`; the
|
||||
//! single-threaded runtime is fine because every wait is a cooperative fd park.
|
||||
#![cfg(feature = "cluster")]
|
||||
|
||||
use std::time::Duration;
|
||||
|
||||
use smarm::cluster::envelope::NodeMeta;
|
||||
use smarm::cluster::handshake::Peer;
|
||||
use smarm::cluster::manager::{Call, Manager, Reply, MANAGER};
|
||||
use smarm::cluster::spawn_established;
|
||||
use smarm::cluster::transport::tcp::TcpTransport;
|
||||
use smarm::cluster::transport::{Conn, FramedConn, Transport};
|
||||
use smarm::cluster::Timing;
|
||||
use smarm::gen_server::{self, GenServerBuilder};
|
||||
use smarm::pg::Incarnation;
|
||||
use smarm::{run, sleep};
|
||||
|
||||
/// A fabricated post-handshake peer identity. Only `node_name` matters to the
|
||||
/// manager table; the rest is filler until c7 consumes it.
|
||||
fn peer(name: &str) -> Peer {
|
||||
Peer {
|
||||
node_name: name.to_string(),
|
||||
incarnation: Incarnation::new(1),
|
||||
meta: NodeMeta {
|
||||
role: "test".to_string(),
|
||||
region: "test".to_string(),
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
/// One established transport pair over localhost. Relies on TCP backlog so the
|
||||
/// sequential dial-then-accept needs no concurrent acceptor (same assumption as
|
||||
/// the c3 conformance suite).
|
||||
fn pair(t: &dyn Transport) -> (Box<dyn Conn>, Box<dyn Conn>) {
|
||||
let mut l = t.listen("127.0.0.1:0").unwrap();
|
||||
let a = t.dial(&l.local_addr()).unwrap();
|
||||
let b = l.accept().unwrap();
|
||||
(a, b)
|
||||
}
|
||||
|
||||
/// Poll the manager until its peer set matches `expected` (sorted), or fail.
|
||||
/// The bound is generous against a sub-millisecond real cost.
|
||||
fn wait_peers(expected: &[&str]) {
|
||||
let want: Vec<String> = expected.iter().map(|s| s.to_string()).collect();
|
||||
for _ in 0..2000 {
|
||||
if let Ok(Reply::Peers(got)) = gen_server::call(MANAGER, Call::Peers) {
|
||||
if got == want {
|
||||
return;
|
||||
}
|
||||
}
|
||||
sleep(Duration::from_millis(1));
|
||||
}
|
||||
let got = gen_server::call(MANAGER, Call::Peers);
|
||||
panic!("timed out waiting for peers == {want:?}; last = {got:?}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn connection_up_commanded_shutdown_and_eof_all_reflected_in_table() {
|
||||
run(|| {
|
||||
// The manager, started plainly and reachable at its well-known name.
|
||||
// (The supervised subtree in `cluster::start` is permanent by design;
|
||||
// a plainly-started manager lets this test terminate cleanly.)
|
||||
let mgr = GenServerBuilder::new(Manager::new())
|
||||
.named(MANAGER)
|
||||
.start()
|
||||
.expect("manager name is free");
|
||||
|
||||
let t = TcpTransport;
|
||||
let (a1, b1) = pair(&t);
|
||||
let (a2, b2) = pair(&t);
|
||||
|
||||
// Manage the `a` ends as peers node-b and node-c; keep the `b` far ends
|
||||
// open so neither socket is closed from the far side yet.
|
||||
spawn_established(FramedConn::new(a1), peer("node-b"), Timing::default())
|
||||
.expect("node-b registers");
|
||||
spawn_established(FramedConn::new(a2), peer("node-c"), Timing::default())
|
||||
.expect("node-c registers");
|
||||
|
||||
// Up: both connections register and the table shows them.
|
||||
wait_peers(&["node-b", "node-c"]);
|
||||
|
||||
// A commanded disconnect reaps exactly its own connection: the
|
||||
// manager drops that entry's handle and the actor stops.
|
||||
assert!(matches!(
|
||||
gen_server::call(
|
||||
MANAGER,
|
||||
Call::Disconnect {
|
||||
name: "node-b".to_string()
|
||||
}
|
||||
),
|
||||
Ok(Reply::Disconnected)
|
||||
));
|
||||
wait_peers(&["node-c"]);
|
||||
|
||||
// A peer close (EOF) reaps the other with no command at all.
|
||||
drop(b2);
|
||||
wait_peers(&[]);
|
||||
|
||||
// node-b's far end stayed open until here, so its removal above was the
|
||||
// disconnect command and not an EOF.
|
||||
drop(b1);
|
||||
|
||||
// All connection actors have exited; stop the manager so `run` returns.
|
||||
mgr.shutdown();
|
||||
});
|
||||
}
|
||||
@@ -1,180 +0,0 @@
|
||||
//! RFC 010 c6c — heartbeat send + fixed-timeout liveness + teardown.
|
||||
//!
|
||||
//! Each case runs one real connection actor over an in-process localhost TCP
|
||||
//! pair, with the far end held as a raw `FramedConn` (no actor) so the test
|
||||
//! controls exactly what — if anything — the peer says. That gives the three
|
||||
//! protocol-visible facts direct handles: heartbeats appear on the wire
|
||||
//! unprompted; a mute peer is torn down (and reaped from the manager table)
|
||||
//! once `LIVENESS_TIMEOUT` empties; and a peer that does nothing but send
|
||||
//! heartbeats keeps the connection alive past that same window.
|
||||
//!
|
||||
//! Loopback has no fd and cannot drive liveness (documented on the actor),
|
||||
//! so everything here is TCP. TCP parks the calling actor, so everything
|
||||
//! runs inside `smarm::run`.
|
||||
#![cfg(feature = "cluster")]
|
||||
|
||||
use std::time::{Duration, Instant};
|
||||
|
||||
use smarm::cluster::conn::{HEARTBEAT_INTERVAL, LIVENESS_TIMEOUT};
|
||||
use smarm::cluster::envelope::{Frame, NodeMeta};
|
||||
use smarm::cluster::handshake::Peer;
|
||||
use smarm::cluster::manager::{Call, Manager, Reply, MANAGER};
|
||||
use smarm::cluster::spawn_established;
|
||||
use smarm::cluster::transport::tcp::TcpTransport;
|
||||
use smarm::cluster::transport::{Conn, FramedConn, Transport};
|
||||
use smarm::cluster::Timing;
|
||||
use smarm::gen_server::{self, GenServerBuilder};
|
||||
use smarm::pg::Incarnation;
|
||||
use smarm::{run, sleep, spawn};
|
||||
|
||||
/// A fabricated post-handshake peer identity (same shape as the c6a suite).
|
||||
fn peer(name: &str) -> Peer {
|
||||
Peer {
|
||||
node_name: name.to_string(),
|
||||
incarnation: Incarnation::new(1),
|
||||
meta: NodeMeta {
|
||||
role: "test".to_string(),
|
||||
region: "test".to_string(),
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
/// One established transport pair over localhost (TCP backlog covers the
|
||||
/// sequential dial-then-accept, as in the c3 conformance suite).
|
||||
fn pair(t: &dyn Transport) -> (Box<dyn Conn>, Box<dyn Conn>) {
|
||||
let mut l = t.listen("127.0.0.1:0").unwrap();
|
||||
let a = t.dial(&l.local_addr()).unwrap();
|
||||
let b = l.accept().unwrap();
|
||||
(a, b)
|
||||
}
|
||||
|
||||
fn peers() -> Vec<String> {
|
||||
match gen_server::call(MANAGER, Call::Peers) {
|
||||
Ok(Reply::Peers(p)) => p,
|
||||
other => panic!("manager unreachable: {other:?}"),
|
||||
}
|
||||
}
|
||||
|
||||
/// Poll until the manager's peer set matches `expected` (sorted) or `budget`
|
||||
/// runs out.
|
||||
fn wait_peers(expected: &[&str], budget: Duration) {
|
||||
let want: Vec<String> = expected.iter().map(|s| s.to_string()).collect();
|
||||
let deadline = Instant::now() + budget;
|
||||
while Instant::now() < deadline {
|
||||
if peers() == want {
|
||||
return;
|
||||
}
|
||||
sleep(Duration::from_millis(10));
|
||||
}
|
||||
panic!(
|
||||
"timed out waiting for peers == {want:?}; last = {:?}",
|
||||
peers()
|
||||
);
|
||||
}
|
||||
|
||||
/// The actor emits heartbeats unprompted: the raw far end, saying nothing,
|
||||
/// sees a `Frame::Heartbeat` well within one interval (the first goes out at
|
||||
/// spawn).
|
||||
#[test]
|
||||
fn heartbeats_are_sent_unprompted() {
|
||||
run(|| {
|
||||
let mgr = GenServerBuilder::new(Manager::new())
|
||||
.named(MANAGER)
|
||||
.start()
|
||||
.expect("manager name is free");
|
||||
|
||||
let (a, b) = pair(&TcpTransport);
|
||||
spawn_established(FramedConn::new(a), peer("hb-send"), Timing::default())
|
||||
.expect("register");
|
||||
let mut far = FramedConn::new(b);
|
||||
|
||||
let frame = far
|
||||
.recv_deadline(Instant::now() + HEARTBEAT_INTERVAL)
|
||||
.expect("a heartbeat before one interval elapses");
|
||||
assert_eq!(frame, Some(Frame::Heartbeat));
|
||||
|
||||
// Teardown: closing the far end is an EOF at the actor.
|
||||
far.close();
|
||||
wait_peers(&[], Duration::from_secs(2));
|
||||
mgr.shutdown();
|
||||
});
|
||||
}
|
||||
|
||||
/// A mute peer is dead: no inbound frame for `LIVENESS_TIMEOUT` tears the
|
||||
/// connection down and the manager's monitor reaps the table entry. The
|
||||
/// entry is still present well inside the window — the teardown is the
|
||||
/// timer, not an accident of setup.
|
||||
#[test]
|
||||
fn mute_peer_is_torn_down_after_liveness_timeout() {
|
||||
run(|| {
|
||||
let mgr = GenServerBuilder::new(Manager::new())
|
||||
.named(MANAGER)
|
||||
.start()
|
||||
.expect("manager name is free");
|
||||
|
||||
let (a, b) = pair(&TcpTransport);
|
||||
spawn_established(FramedConn::new(a), peer("mute"), Timing::default()).expect("register");
|
||||
// Held open and silent: no frames, no EOF. (Unread inbound
|
||||
// heartbeats sit in kernel buffers; they are 5 bytes each.)
|
||||
let _far = FramedConn::new(b);
|
||||
|
||||
// Well inside the window the connection is still up.
|
||||
sleep(LIVENESS_TIMEOUT / 2);
|
||||
assert_eq!(peers(), vec!["mute".to_string()], "torn down too early");
|
||||
|
||||
// ...and once the window empties it is gone. Generous budget over
|
||||
// the remaining half-window.
|
||||
wait_peers(&[], LIVENESS_TIMEOUT);
|
||||
mgr.shutdown();
|
||||
});
|
||||
}
|
||||
|
||||
/// Heartbeats alone keep a connection alive past `LIVENESS_TIMEOUT`: a far
|
||||
/// end that sends `Frame::Heartbeat` at the interval (and nothing else)
|
||||
/// holds the entry; when it goes quiet, liveness finally fires.
|
||||
#[test]
|
||||
fn heartbeats_keep_the_connection_alive() {
|
||||
run(|| {
|
||||
let mgr = GenServerBuilder::new(Manager::new())
|
||||
.named(MANAGER)
|
||||
.start()
|
||||
.expect("manager name is free");
|
||||
|
||||
let (a, b) = pair(&TcpTransport);
|
||||
spawn_established(FramedConn::new(a), peer("kept"), Timing::default()).expect("register");
|
||||
|
||||
// The far heartbeat pump: interval-paced sends until told to stop,
|
||||
// then holds the socket open, silent, so the eventual teardown is
|
||||
// liveness — not EOF.
|
||||
let (ctl_tx, ctl_rx) = smarm::channel::channel::<()>();
|
||||
spawn(move || {
|
||||
let mut far = FramedConn::new(b);
|
||||
// Phase 1: heartbeat at the interval until the first signal.
|
||||
while matches!(ctl_rx.try_recv(), Ok(None)) {
|
||||
far.send(&Frame::Heartbeat).expect("far send");
|
||||
sleep(HEARTBEAT_INTERVAL);
|
||||
}
|
||||
// Phase 2: silent but with the socket held open — dropping
|
||||
// `far` here would EOF the actor and mask the liveness path.
|
||||
// Exits when the test's closure ends and drops `ctl_tx` (an
|
||||
// eternal park would stop `run` from ever returning).
|
||||
while matches!(ctl_rx.try_recv(), Ok(None)) {
|
||||
sleep(Duration::from_millis(20));
|
||||
}
|
||||
});
|
||||
|
||||
// Past the liveness window with margin: still up.
|
||||
sleep(LIVENESS_TIMEOUT + LIVENESS_TIMEOUT / 2);
|
||||
assert_eq!(
|
||||
peers(),
|
||||
vec!["kept".to_string()],
|
||||
"liveness fired despite heartbeats"
|
||||
);
|
||||
|
||||
// Silence the pump; liveness now empties and the entry goes.
|
||||
ctl_tx.send(()).expect("pump alive");
|
||||
wait_peers(&[], LIVENESS_TIMEOUT * 2);
|
||||
mgr.shutdown();
|
||||
// `ctl_tx` drops here, releasing the pump's phase-2 wait.
|
||||
});
|
||||
}
|
||||
@@ -1,482 +0,0 @@
|
||||
//! RFC 010 c6b — the handshake on the accept/connect path.
|
||||
//!
|
||||
//! Path-level tests drive [`dial_handshake`]/[`accept_handshake`] over the
|
||||
//! loopback transport on plain threads (its intended use — synchronous, no
|
||||
//! runtime). Integration tests run the manager-backed [`dial`] and
|
||||
//! [`spawn_acceptor`] over real localhost TCP inside `smarm::run`, and the
|
||||
//! two-node case as subprocesses via the c4 harness. Flake budget: see
|
||||
//! tests/common/mod.rs.
|
||||
#![cfg(feature = "cluster")]
|
||||
|
||||
mod common;
|
||||
|
||||
use std::sync::mpsc;
|
||||
use std::time::{Duration, Instant};
|
||||
|
||||
use common::{maybe_child, spawn_node, WAIT};
|
||||
use smarm::cluster::connect::{
|
||||
accept_handshake, dial, dial_handshake, spawn_acceptor, DialError, HandshakeError,
|
||||
HANDSHAKE_TIMEOUT,
|
||||
};
|
||||
use smarm::cluster::envelope::{Frame, NodeMeta, RejectReason};
|
||||
use smarm::cluster::handshake::{Local, PeerStanding};
|
||||
use smarm::cluster::manager::{Call, Manager, Reply, MANAGER};
|
||||
use smarm::cluster::transport::loopback::LoopbackTransport;
|
||||
use smarm::cluster::transport::tcp::TcpTransport;
|
||||
use smarm::cluster::transport::{FramedConn, Transport};
|
||||
use smarm::cluster::Timing;
|
||||
use smarm::gen_server::{self, GenServerBuilder};
|
||||
use smarm::pg::Incarnation;
|
||||
use smarm::{run, sleep};
|
||||
|
||||
const ROLES: &[(&str, fn())] = &[
|
||||
("hs_listener", role_hs_listener),
|
||||
("hs_dialer", role_hs_dialer),
|
||||
];
|
||||
|
||||
const HASH: u64 = 0xC6B0_C6B0_C6B0_C6B0;
|
||||
|
||||
fn local(name: &str) -> Local {
|
||||
Local {
|
||||
node_name: name.into(),
|
||||
incarnation: Incarnation::new(3),
|
||||
build_hash: HASH,
|
||||
meta: NodeMeta {
|
||||
role: "test".into(),
|
||||
region: "test".into(),
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
/// A loopback conn pair as `FramedConn`s, ready for a threaded handshake.
|
||||
fn loopback_pair() -> (FramedConn, FramedConn) {
|
||||
let t = LoopbackTransport::default();
|
||||
let mut l = t.listen("hs").unwrap();
|
||||
let dialer = FramedConn::new(t.dial("hs").unwrap());
|
||||
let accepted = FramedConn::new(l.accept().unwrap());
|
||||
(dialer, accepted)
|
||||
}
|
||||
|
||||
/// Far-future deadline for loopback paths, where it cannot fire anyway.
|
||||
fn no_deadline() -> Instant {
|
||||
Instant::now() + Duration::from_secs(3600)
|
||||
}
|
||||
|
||||
/// Park the node forever: it has announced everything the parent asserts on,
|
||||
/// and must now hold its connection open until SIGKILLed.
|
||||
fn park() -> ! {
|
||||
loop {
|
||||
sleep(Duration::from_secs(1));
|
||||
}
|
||||
}
|
||||
|
||||
/// Cooperative bounded receive across the closure/actor boundary. A blocking
|
||||
/// `std::mpsc` wait would park the OS thread and starve the single-threaded
|
||||
/// scheduler, so every wait inside `run` polls with [`sleep`] instead.
|
||||
fn poll_recv<T>(rx: &mpsc::Receiver<T>, what: &str) -> T {
|
||||
let deadline = Instant::now() + WAIT;
|
||||
loop {
|
||||
match rx.try_recv() {
|
||||
Ok(v) => return v,
|
||||
Err(mpsc::TryRecvError::Empty) => {
|
||||
assert!(Instant::now() < deadline, "timed out waiting for {what}");
|
||||
sleep(Duration::from_millis(1));
|
||||
}
|
||||
Err(mpsc::TryRecvError::Disconnected) => panic!("channel closed waiting for {what}"),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Path level, over loopback on plain threads
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
#[test]
|
||||
fn loopback_happy_path_establishes_both_ends() {
|
||||
maybe_child(ROLES);
|
||||
let (mut dialer, mut accepted) = loopback_pair();
|
||||
let responder = std::thread::spawn(move || {
|
||||
accept_handshake(
|
||||
&mut accepted,
|
||||
local("node-b"),
|
||||
|name| {
|
||||
assert_eq!(name, "node-a");
|
||||
PeerStanding::Free
|
||||
},
|
||||
no_deadline(),
|
||||
)
|
||||
});
|
||||
let peer_of_dialer = dial_handshake(&mut dialer, &local("node-a"), no_deadline()).unwrap();
|
||||
let peer_of_acceptor = responder.join().unwrap().unwrap();
|
||||
assert_eq!(peer_of_dialer.node_name, "node-b");
|
||||
assert_eq!(peer_of_acceptor.node_name, "node-a");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn loopback_hash_mismatch_rejected_with_frame_then_eof() {
|
||||
maybe_child(ROLES);
|
||||
let (mut dialer, mut accepted) = loopback_pair();
|
||||
let mut wrong = local("node-b");
|
||||
wrong.build_hash ^= 1;
|
||||
let responder = std::thread::spawn(move || {
|
||||
accept_handshake(&mut accepted, wrong, |_| PeerStanding::Free, no_deadline())
|
||||
});
|
||||
// The dial side receives the reject frame — the compatibility anchor.
|
||||
match dial_handshake(&mut dialer, &local("node-a"), no_deadline()) {
|
||||
Err(HandshakeError::Rejected(RejectReason::HashMismatch)) => {}
|
||||
other => panic!("expected HashMismatch reject, got {other:?}"),
|
||||
}
|
||||
match responder.join().unwrap() {
|
||||
Err(HandshakeError::Rejected(RejectReason::HashMismatch)) => {}
|
||||
other => panic!("expected accept side to report the reject, got {other:?}"),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn loopback_tie_break_loser_closed_silently() {
|
||||
maybe_child(ROLES);
|
||||
// The inbound dial is from "node-z"; we are "node-a" with our own dial to
|
||||
// node-z in flight. dial_wins("node-z", "node-a") is false, so the
|
||||
// inbound loses: closed with no frame at all.
|
||||
let (mut dialer, mut accepted) = loopback_pair();
|
||||
let responder = std::thread::spawn(move || {
|
||||
accept_handshake(
|
||||
&mut accepted,
|
||||
local("node-a"),
|
||||
|_| PeerStanding::Dialing,
|
||||
no_deadline(),
|
||||
)
|
||||
});
|
||||
// Silent close: the dial side sees EOF, never a frame.
|
||||
match dial_handshake(&mut dialer, &local("node-z"), no_deadline()) {
|
||||
Err(HandshakeError::Closed) => {}
|
||||
other => panic!("expected silent close (Closed), got {other:?}"),
|
||||
}
|
||||
match responder.join().unwrap() {
|
||||
Err(HandshakeError::TieBreakLoss) => {}
|
||||
other => panic!("expected TieBreakLoss on the accept side, got {other:?}"),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn loopback_read_ahead_past_hello_survives_into_established_conn() {
|
||||
maybe_child(ROLES);
|
||||
// The buffer trap, proven: the dialer coalesces Hello + Heartbeat before
|
||||
// the responder's first read, so the Heartbeat lands in the shared
|
||||
// FramedConn's decode buffer during the handshake. The dialer sends
|
||||
// nothing afterwards — the post-handshake recv can only succeed if the
|
||||
// read-ahead travelled with the FramedConn.
|
||||
let (mut dialer, mut accepted) = loopback_pair();
|
||||
let (_init, hello) = smarm::cluster::handshake::Initiator::new(&local("node-a"));
|
||||
dialer.send(&hello).unwrap();
|
||||
dialer.send(&Frame::Heartbeat).unwrap();
|
||||
// Both frames are buffered before the responder reads at all.
|
||||
let (tx, rx) = mpsc::channel();
|
||||
std::thread::spawn(move || {
|
||||
let peer = accept_handshake(
|
||||
&mut accepted,
|
||||
local("node-b"),
|
||||
|_| PeerStanding::Free,
|
||||
no_deadline(),
|
||||
)
|
||||
.unwrap();
|
||||
let next = accepted.recv();
|
||||
let _ = tx.send((peer, next));
|
||||
});
|
||||
// A bounded wait: if the Heartbeat were NOT carried in the buffer, the
|
||||
// recv above would block forever (the dialer stays open and silent).
|
||||
let (peer, next) = rx
|
||||
.recv_timeout(Duration::from_secs(5))
|
||||
.expect("read-ahead lost: post-handshake recv blocked");
|
||||
assert_eq!(peer.node_name, "node-a");
|
||||
match next {
|
||||
Ok(Some(Frame::Heartbeat)) => {}
|
||||
other => panic!("expected the read-ahead Heartbeat, got {other:?}"),
|
||||
}
|
||||
drop(dialer);
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Deadline + manager integration, over TCP inside the runtime
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
#[test]
|
||||
fn tcp_silent_peer_times_out_on_the_accept_path() {
|
||||
maybe_child(ROLES);
|
||||
run(|| {
|
||||
let t = TcpTransport;
|
||||
let mut l = t.listen("127.0.0.1:0").unwrap();
|
||||
// Connect and then say nothing at all.
|
||||
let silent = t.dial(&l.local_addr()).unwrap();
|
||||
let mut accepted = FramedConn::new(l.accept().unwrap());
|
||||
let (tx, rx) = mpsc::channel();
|
||||
smarm::spawn(move || {
|
||||
let r = accept_handshake(
|
||||
&mut accepted,
|
||||
local("node-b"),
|
||||
|_| PeerStanding::Free,
|
||||
Instant::now() + Duration::from_millis(200),
|
||||
);
|
||||
let _ = tx.send(r);
|
||||
});
|
||||
match poll_recv(&rx, "accept-path outcome") {
|
||||
Err(HandshakeError::TimedOut) => {}
|
||||
other => panic!("expected TimedOut, got {other:?}"),
|
||||
}
|
||||
drop(silent);
|
||||
});
|
||||
}
|
||||
|
||||
/// Poll the manager until its peer set matches `expected` (sorted), or fail.
|
||||
fn wait_peers(expected: &[&str]) {
|
||||
let want: Vec<String> = expected.iter().map(|s| s.to_string()).collect();
|
||||
for _ in 0..5000 {
|
||||
if let Ok(Reply::Peers(got)) = gen_server::call(MANAGER, Call::Peers) {
|
||||
if got == want {
|
||||
return;
|
||||
}
|
||||
}
|
||||
sleep(Duration::from_millis(1));
|
||||
}
|
||||
let got = gen_server::call(MANAGER, Call::Peers);
|
||||
panic!("timed out waiting for peers == {want:?}; last = {got:?}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tcp_duplicate_name_rejected_by_acceptor() {
|
||||
maybe_child(ROLES);
|
||||
run(|| {
|
||||
let mgr = GenServerBuilder::new(Manager::new())
|
||||
.named(MANAGER)
|
||||
.start()
|
||||
.expect("manager name is free");
|
||||
let listener = TcpTransport.listen("127.0.0.1:0").unwrap();
|
||||
let acceptor = spawn_acceptor(listener, local("node-b"), Timing::default());
|
||||
let addr = acceptor.local_addr().to_string();
|
||||
|
||||
// First dial offering "dup-node": establishes and registers.
|
||||
let mut first = FramedConn::new(TcpTransport.dial(&addr).unwrap());
|
||||
let peer = dial_handshake(
|
||||
&mut first,
|
||||
&local("dup-node"),
|
||||
Instant::now() + HANDSHAKE_TIMEOUT,
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(peer.node_name, "node-b");
|
||||
wait_peers(&["dup-node"]);
|
||||
|
||||
// Second dial offering the same name: deterministic NameTaken.
|
||||
let mut second = FramedConn::new(TcpTransport.dial(&addr).unwrap());
|
||||
match dial_handshake(
|
||||
&mut second,
|
||||
&local("dup-node"),
|
||||
Instant::now() + HANDSHAKE_TIMEOUT,
|
||||
) {
|
||||
Err(HandshakeError::Rejected(RejectReason::NameTaken)) => {}
|
||||
other => panic!("expected NameTaken, got {other:?}"),
|
||||
}
|
||||
// The established connection was untouched by the rejected one.
|
||||
wait_peers(&["dup-node"]);
|
||||
|
||||
// Teardown: the acceptor owns no connections, so the established one
|
||||
// is torn down through the table.
|
||||
acceptor.shutdown();
|
||||
assert!(matches!(
|
||||
gen_server::call(
|
||||
MANAGER,
|
||||
Call::Disconnect {
|
||||
name: "dup-node".to_string()
|
||||
}
|
||||
),
|
||||
Ok(Reply::Disconnected)
|
||||
));
|
||||
wait_peers(&[]);
|
||||
first.close();
|
||||
mgr.shutdown();
|
||||
});
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn dial_intent_cleared_when_dialer_dies() {
|
||||
maybe_child(ROLES);
|
||||
run(|| {
|
||||
let mgr = GenServerBuilder::new(Manager::new())
|
||||
.named(MANAGER)
|
||||
.start()
|
||||
.expect("manager name is free");
|
||||
let (begun_tx, begun_rx) = mpsc::channel();
|
||||
let (go_tx, go_rx) = mpsc::channel::<()>();
|
||||
smarm::spawn(move || {
|
||||
let me = smarm::self_pid();
|
||||
match gen_server::call(
|
||||
MANAGER,
|
||||
Call::DialBegin {
|
||||
name: "ghost".into(),
|
||||
pid: me,
|
||||
},
|
||||
) {
|
||||
Ok(Reply::DialBegan(true)) => {}
|
||||
other => panic!("DialBegin failed: {other:?}"),
|
||||
}
|
||||
let _ = begun_tx.send(());
|
||||
let () = poll_recv(&go_rx, "go signal");
|
||||
panic!("dialer dies mid-dial");
|
||||
});
|
||||
poll_recv(&begun_rx, "DialBegin done");
|
||||
// While the dialer lives, the intent is visible.
|
||||
match gen_server::call(
|
||||
MANAGER,
|
||||
Call::Standing {
|
||||
peer_name: "ghost".into(),
|
||||
},
|
||||
) {
|
||||
Ok(Reply::Standing(s)) => assert_eq!(s, PeerStanding::Dialing),
|
||||
other => panic!("PeerStanding failed: {other:?}"),
|
||||
}
|
||||
// Kill it; the monitor must clear the intent without cooperation.
|
||||
go_tx.send(()).unwrap();
|
||||
let deadline = Instant::now() + WAIT;
|
||||
loop {
|
||||
match gen_server::call(
|
||||
MANAGER,
|
||||
Call::Standing {
|
||||
peer_name: "ghost".into(),
|
||||
},
|
||||
) {
|
||||
Ok(Reply::Standing(s)) if s != PeerStanding::Dialing => break,
|
||||
_ if Instant::now() > deadline => {
|
||||
panic!("dial intent not cleared after dialer death")
|
||||
}
|
||||
_ => sleep(Duration::from_millis(1)),
|
||||
}
|
||||
}
|
||||
mgr.shutdown();
|
||||
});
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Two nodes, two processes: the integrated dial against a real acceptor
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Announce, then park forever. Neither role ever tears its connection
|
||||
/// down: a table entry only exists while the *peer* holds its side open, so
|
||||
/// any teardown here would retract the other node's observation before it
|
||||
/// had made it. The parent reaps both with SIGKILL once it has both
|
||||
/// announcements (see [`common::Node`]'s `Drop`).
|
||||
fn role_hs_listener() {
|
||||
run(|| {
|
||||
let _mgr = GenServerBuilder::new(Manager::new())
|
||||
.named(MANAGER)
|
||||
.start()
|
||||
.expect("manager name is free");
|
||||
let listener = TcpTransport.listen("127.0.0.1:0").unwrap();
|
||||
let acceptor = spawn_acceptor(listener, local("node-b"), Timing::default());
|
||||
println!("LISTENING {}", acceptor.local_addr());
|
||||
wait_peers(&["node-a"]);
|
||||
println!("PEERS node-a");
|
||||
park();
|
||||
});
|
||||
}
|
||||
|
||||
fn role_hs_dialer() {
|
||||
let addr = std::env::var("SMARM_PEER_ADDR").expect("SMARM_PEER_ADDR not set");
|
||||
run(move || {
|
||||
let _mgr = GenServerBuilder::new(Manager::new())
|
||||
.named(MANAGER)
|
||||
.start()
|
||||
.expect("manager name is free");
|
||||
let (tx, rx) = mpsc::channel();
|
||||
smarm::spawn(move || {
|
||||
let r = dial(
|
||||
&TcpTransport,
|
||||
&addr,
|
||||
"node-b",
|
||||
&local("node-a"),
|
||||
Timing::default(),
|
||||
);
|
||||
let _ = tx.send(r);
|
||||
});
|
||||
if let Err(e) = poll_recv(&rx, "dial outcome") {
|
||||
println!("DIAL failed: {e:?}");
|
||||
std::process::exit(3);
|
||||
}
|
||||
wait_peers(&["node-b"]);
|
||||
println!("PEERS node-b");
|
||||
park();
|
||||
});
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn two_node_integrated_handshake_over_tcp() {
|
||||
maybe_child(ROLES);
|
||||
let mut listener = spawn_node("hs_listener", &[]);
|
||||
let addr = listener.wait_listening();
|
||||
let mut dialer = spawn_node("hs_dialer", &[("SMARM_PEER_ADDR", &addr)]);
|
||||
// Each node reports its own table naming the other: a real dial against a
|
||||
// real acceptor established in both directions. Both nodes then park —
|
||||
// clean-exit behaviour is the c4 harness's own smoke test, and demanding
|
||||
// it here would mean a teardown, which is exactly what cannot be ordered
|
||||
// safely across two processes. Dropping the nodes SIGKILLs them.
|
||||
dialer.wait_line("PEERS node-b", |l| l == "PEERS node-b");
|
||||
listener.wait_line("PEERS node-a", |l| l == "PEERS node-a");
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Integrated-dial guardrails (no acceptor involved)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
#[test]
|
||||
fn concurrent_dial_to_same_name_refused() {
|
||||
maybe_child(ROLES);
|
||||
run(|| {
|
||||
let mgr = GenServerBuilder::new(Manager::new())
|
||||
.named(MANAGER)
|
||||
.start()
|
||||
.expect("manager name is free");
|
||||
let (begun_tx, begun_rx) = mpsc::channel();
|
||||
let (go_tx, go_rx) = mpsc::channel::<()>();
|
||||
// First dialer parks with the intent held (it never connects —
|
||||
// 'holding the intent' is all this test needs from it).
|
||||
smarm::spawn(move || {
|
||||
let me = smarm::self_pid();
|
||||
assert!(matches!(
|
||||
gen_server::call(
|
||||
MANAGER,
|
||||
Call::DialBegin {
|
||||
name: "node-x".into(),
|
||||
pid: me,
|
||||
}
|
||||
),
|
||||
Ok(Reply::DialBegan(true))
|
||||
));
|
||||
let _ = begun_tx.send(());
|
||||
let () = poll_recv(&go_rx, "go signal");
|
||||
let _ = gen_server::call(
|
||||
MANAGER,
|
||||
Call::DialEnd {
|
||||
name: "node-x".into(),
|
||||
},
|
||||
);
|
||||
});
|
||||
poll_recv(&begun_rx, "DialBegin done");
|
||||
// Second integrated dial to the same name: refused before connecting
|
||||
// (the addr is unroutable on purpose — it must never be dialed).
|
||||
let (tx, rx) = mpsc::channel();
|
||||
smarm::spawn(move || {
|
||||
let r = dial(
|
||||
&TcpTransport,
|
||||
"127.0.0.1:1",
|
||||
"node-x",
|
||||
&local("node-a"),
|
||||
Timing::default(),
|
||||
);
|
||||
let _ = tx.send(r);
|
||||
});
|
||||
match poll_recv(&rx, "second dial outcome") {
|
||||
Err(DialError::AlreadyDialing) => {}
|
||||
other => panic!("expected AlreadyDialing, got {other:?}"),
|
||||
}
|
||||
go_tx.send(()).unwrap();
|
||||
mgr.shutdown();
|
||||
});
|
||||
}
|
||||
@@ -1,115 +0,0 @@
|
||||
//! RFC 010 — a seed whose address answers as a *different* name
|
||||
//! (`DialError::PeerNameMismatch`) is dialed once and then parked: the
|
||||
//! connector must not redial it on backoff forever.
|
||||
//!
|
||||
//! Observed from the misdialed peer: each such dial establishes at the
|
||||
//! responder (it registers, `node_up`), then the dialer closes on the name
|
||||
//! check (`node_down`) — one membership blip per attempt. Cross-process: a
|
||||
//! *server* named `server` subscribes and reports; a *client* on fast
|
||||
//! timing (50–500ms backoff) seeds `("wrongname", server_addr)`. After the
|
||||
//! first blip the server counts further `NodeUp`s across 2s — several
|
||||
//! backoff periods. Parked ⇒ zero. Negative-control-verified: with the park
|
||||
//! stubbed out the count is ≥ 1 in the same window.
|
||||
#![cfg(feature = "cluster")]
|
||||
|
||||
mod common;
|
||||
|
||||
use common::{maybe_child, spawn_node};
|
||||
use smarm::cluster::envelope::NodeMeta;
|
||||
use smarm::cluster::membership::{subscribe, NodeEvent};
|
||||
use smarm::cluster::{start, Config, StaticSeeds, Timing};
|
||||
use std::time::{Duration, Instant};
|
||||
|
||||
const ROLES: &[(&str, fn())] = &[("server", role_server), ("client", role_client)];
|
||||
|
||||
fn meta() -> NodeMeta {
|
||||
NodeMeta {
|
||||
role: "mismatch".into(),
|
||||
region: "local".into(),
|
||||
}
|
||||
}
|
||||
|
||||
fn timing() -> Timing {
|
||||
Timing {
|
||||
initial_backoff: Duration::from_millis(50),
|
||||
max_backoff: Duration::from_millis(500),
|
||||
..Timing::default()
|
||||
}
|
||||
}
|
||||
|
||||
fn role_server() {
|
||||
smarm::run(|| {
|
||||
let cluster = start(Config {
|
||||
node_name: "server".into(),
|
||||
meta: meta(),
|
||||
listen_addr: std::env::var("SMARM_LISTEN_ADDR")
|
||||
.unwrap_or_else(|_| "127.0.0.1:0".into()),
|
||||
strategy: Box::new(StaticSeeds::new(Vec::<(String, String)>::new())),
|
||||
timing: timing(),
|
||||
})
|
||||
.expect("binds");
|
||||
let ev = subscribe().unwrap();
|
||||
println!("LISTENING {}", cluster.local_addr());
|
||||
// First blip: the misdialed client establishes, then closes on us.
|
||||
loop {
|
||||
match ev.rx.recv() {
|
||||
Ok(NodeEvent::NodeDown(i)) if i.name == "client" => break,
|
||||
Ok(_) => continue,
|
||||
Err(_) => panic!("manager gone"),
|
||||
}
|
||||
}
|
||||
println!("BLIP");
|
||||
// Now count further NodeUps across several backoff periods.
|
||||
let mut more = 0usize;
|
||||
let t0 = Instant::now();
|
||||
while t0.elapsed() < Duration::from_millis(2000) {
|
||||
match ev.rx.try_recv() {
|
||||
Ok(Some(NodeEvent::NodeUp(i))) if i.name == "client" => more += 1,
|
||||
Ok(_) => {}
|
||||
Err(_) => panic!("manager gone"),
|
||||
}
|
||||
smarm::sleep(Duration::from_millis(50));
|
||||
}
|
||||
println!("MORE {more}");
|
||||
loop {
|
||||
smarm::sleep(Duration::from_secs(3600));
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
fn role_client() {
|
||||
let server_addr = std::env::var("SMARM_SERVER_ADDR").expect("SMARM_SERVER_ADDR");
|
||||
smarm::run(move || {
|
||||
let _cluster = start(Config {
|
||||
node_name: "client".into(),
|
||||
meta: meta(),
|
||||
listen_addr: "127.0.0.1:0".into(),
|
||||
strategy: Box::new(StaticSeeds::new(vec![(
|
||||
"wrongname".to_string(),
|
||||
server_addr,
|
||||
)])),
|
||||
timing: timing(),
|
||||
})
|
||||
.expect("binds");
|
||||
println!("CLIENT UP");
|
||||
loop {
|
||||
smarm::sleep(Duration::from_secs(3600));
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn mismatched_seed_is_dialed_once_then_parked() {
|
||||
maybe_child(ROLES);
|
||||
let mut server = spawn_node("server", &[]);
|
||||
let saddr = server.wait_listening();
|
||||
let mut client = spawn_node("client", &[("SMARM_SERVER_ADDR", &saddr)]);
|
||||
client.wait_line("CLIENT UP", |l| l == "CLIENT UP");
|
||||
server.wait_line("BLIP", |l| l == "BLIP");
|
||||
let line = server.wait_line("MORE", |l| l.starts_with("MORE "));
|
||||
let more: usize = line.split_whitespace().nth(1).unwrap().parse().unwrap();
|
||||
assert_eq!(
|
||||
more, 0,
|
||||
"mismatched seed was redialed {more}× after being parked"
|
||||
);
|
||||
}
|
||||
@@ -1,379 +0,0 @@
|
||||
//! RFC 010 c13 — connection-loss synthesis.
|
||||
//!
|
||||
//! Local suite (`run()`, no network): the read-side backstop. A
|
||||
//! `RemoteMonitor` whose channel closes without a notice reads as
|
||||
//! `Disconnected` exactly once (a `Monitor` command that reached the conn
|
||||
//! actor's inbox but was never processed — the drain gap); after
|
||||
//! `demonitor_remote` a closed channel stays a plain `Err`, never a notice.
|
||||
//!
|
||||
//! Cross-process: the headline contrast — an actor's own death gives its
|
||||
//! TRUE reason, loss of the LINK gives `Disconnected` (both a commanded
|
||||
//! `Disconnect` and a SIGKILLed peer process are `Disconnected` from the
|
||||
//! monitor's view: nobody is left to say otherwise). Reconnect does not
|
||||
//! resurrect: the old monitor yields nothing more, proven by stream ORDER
|
||||
//! (a fresh monitor over the new link delivers first). The ignored test
|
||||
//! trips liveness by SIGSTOP and then drops the link too, asserting exactly
|
||||
//! one notice for one monitor.
|
||||
#![cfg(feature = "cluster")]
|
||||
|
||||
mod common;
|
||||
|
||||
use common::{maybe_child, spawn_node};
|
||||
use smarm::cluster::envelope::NodeMeta;
|
||||
use smarm::cluster::expose::{expose, expose_type};
|
||||
use smarm::cluster::manager::{Call, Reply, MANAGER};
|
||||
use smarm::cluster::membership::{subscribe, NodeEvent};
|
||||
use smarm::cluster::remote::{
|
||||
self, demonitor_remote, monitor_remote, send_to_remote, RemoteName, RemotePid,
|
||||
};
|
||||
use smarm::cluster::{start, Config, RemoteDownReason, StaticSeeds, Timing};
|
||||
use smarm::pg::Incarnation;
|
||||
use smarm::{
|
||||
channel, gen_server, install, register, run, spawn, Addressable, DownReason, Erased, Name, Pid,
|
||||
};
|
||||
use std::collections::HashMap;
|
||||
use std::time::Duration;
|
||||
|
||||
// ---- message types (hand-rolled serde; the crate is derive-less) ---------
|
||||
|
||||
#[derive(Debug)]
|
||||
struct Ctl {
|
||||
cmd: String,
|
||||
reply_to: RemotePid<Client>,
|
||||
}
|
||||
#[derive(Debug)]
|
||||
struct Answer {
|
||||
text: String,
|
||||
pid: Option<RemotePid<Erased>>,
|
||||
}
|
||||
struct Client;
|
||||
impl Addressable for Client {
|
||||
type Msg = Answer;
|
||||
}
|
||||
|
||||
impl serde::Serialize for Ctl {
|
||||
fn serialize<S: serde::Serializer>(&self, s: S) -> Result<S::Ok, S::Error> {
|
||||
use serde::ser::SerializeTuple;
|
||||
let mut t = s.serialize_tuple(2)?;
|
||||
t.serialize_element(&self.cmd)?;
|
||||
t.serialize_element(&self.reply_to)?;
|
||||
t.end()
|
||||
}
|
||||
}
|
||||
impl<'de> serde::Deserialize<'de> for Ctl {
|
||||
fn deserialize<D: serde::Deserializer<'de>>(d: D) -> Result<Self, D::Error> {
|
||||
let (cmd, reply_to) = <(String, RemotePid<Client>)>::deserialize(d)?;
|
||||
Ok(Ctl { cmd, reply_to })
|
||||
}
|
||||
}
|
||||
impl serde::Serialize for Answer {
|
||||
fn serialize<S: serde::Serializer>(&self, s: S) -> Result<S::Ok, S::Error> {
|
||||
use serde::ser::SerializeTuple;
|
||||
let mut t = s.serialize_tuple(2)?;
|
||||
t.serialize_element(&self.text)?;
|
||||
t.serialize_element(&self.pid)?;
|
||||
t.end()
|
||||
}
|
||||
}
|
||||
impl<'de> serde::Deserialize<'de> for Answer {
|
||||
fn deserialize<D: serde::Deserializer<'de>>(d: D) -> Result<Self, D::Error> {
|
||||
let (text, pid) = <(String, Option<RemotePid<Erased>>)>::deserialize(d)?;
|
||||
Ok(Answer { text, pid })
|
||||
}
|
||||
}
|
||||
|
||||
// ================= local suite =========================================
|
||||
|
||||
/// A `Monitor` command handed to the connection but never processed (its
|
||||
/// receiver dropped unread) reads as `Disconnected` — once. A second read
|
||||
/// is the ordinary closed-channel `Err`, so "exactly one notice" holds.
|
||||
#[test]
|
||||
fn unread_command_reads_as_disconnected_once() {
|
||||
maybe_child(ROLES);
|
||||
run(|| {
|
||||
remote::set_local_identity("me", Incarnation::new(7));
|
||||
let (probe_tx, _probe_rx) = channel();
|
||||
let inbox =
|
||||
remote::bind_outbound_probe_with_monitors("peer", Incarnation::new(5), probe_tx);
|
||||
let target = RemotePid::<Erased>::from_parts("peer", Incarnation::new(5), 9, 1);
|
||||
let m = monitor_remote(target.clone());
|
||||
assert!(
|
||||
matches!(m.try_recv(), Ok(None)),
|
||||
"command is in flight, no notice yet"
|
||||
);
|
||||
drop(inbox); // the conn actor died with the command unread
|
||||
let d = m.recv().unwrap();
|
||||
assert_eq!(d.pid, target);
|
||||
assert_eq!(d.reason, RemoteDownReason::Disconnected);
|
||||
assert!(
|
||||
m.recv().is_err(),
|
||||
"second read is closed, not a second notice"
|
||||
);
|
||||
assert!(m.try_recv().is_err());
|
||||
});
|
||||
}
|
||||
|
||||
/// After `demonitor_remote`, a closed channel is a closed channel: no
|
||||
/// notice is synthesized for a monitor the caller cancelled.
|
||||
#[test]
|
||||
fn cancelled_monitor_never_synthesizes() {
|
||||
maybe_child(ROLES);
|
||||
run(|| {
|
||||
remote::set_local_identity("me", Incarnation::new(7));
|
||||
let (probe_tx, _probe_rx) = channel();
|
||||
let inbox =
|
||||
remote::bind_outbound_probe_with_monitors("peer", Incarnation::new(5), probe_tx);
|
||||
let target = RemotePid::<Erased>::from_parts("peer", Incarnation::new(5), 9, 1);
|
||||
let m = monitor_remote(target);
|
||||
demonitor_remote(&m);
|
||||
drop(inbox);
|
||||
assert!(m.recv().is_err());
|
||||
assert!(m.try_recv().is_err());
|
||||
});
|
||||
}
|
||||
|
||||
// ================= cross-process ======================================
|
||||
|
||||
const ROLES: &[(&str, fn())] = &[
|
||||
("server", role_server),
|
||||
("client", role_client),
|
||||
("client_stop", role_client_stop),
|
||||
];
|
||||
|
||||
const CTL: Name<Ctl> = Name::new("c13.ctl");
|
||||
|
||||
fn cfg(name: &str, seeds: Vec<(String, String)>) -> Config {
|
||||
Config {
|
||||
node_name: name.to_string(),
|
||||
meta: NodeMeta {
|
||||
role: "c13".into(),
|
||||
region: "local".into(),
|
||||
},
|
||||
listen_addr: std::env::var("SMARM_LISTEN_ADDR").unwrap_or_else(|_| "127.0.0.1:0".into()),
|
||||
strategy: Box::new(StaticSeeds::new(seeds)),
|
||||
timing: timing(),
|
||||
}
|
||||
}
|
||||
|
||||
/// The p11 knobs make the liveness test fast: both roles of that test are
|
||||
/// spawned with `SMARM_FAST_TIMING=1` and agree on a 100ms heartbeat /
|
||||
/// 500ms liveness window. Everything else runs the shipping defaults.
|
||||
fn timing() -> Timing {
|
||||
if std::env::var_os("SMARM_FAST_TIMING").is_some() {
|
||||
Timing {
|
||||
heartbeat_interval: Duration::from_millis(100),
|
||||
liveness_timeout: Duration::from_millis(500),
|
||||
initial_backoff: Duration::from_millis(50),
|
||||
max_backoff: Duration::from_millis(500),
|
||||
..Timing::default()
|
||||
}
|
||||
} else {
|
||||
Timing::default()
|
||||
}
|
||||
}
|
||||
|
||||
fn wait_up(events: &smarm::cluster::membership::MembershipEvents, who: &str) {
|
||||
loop {
|
||||
match events.rx.recv() {
|
||||
Ok(NodeEvent::NodeUp(i)) if i.name == who => return,
|
||||
Ok(_) => continue,
|
||||
Err(_) => panic!("manager gone"),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn disconnect(name: &str) {
|
||||
assert!(matches!(
|
||||
gen_server::call(
|
||||
MANAGER,
|
||||
Call::Disconnect {
|
||||
name: name.to_string()
|
||||
}
|
||||
),
|
||||
Ok(Reply::Disconnected)
|
||||
));
|
||||
}
|
||||
|
||||
/// Server: `spawn` ⇒ a parked worker (answer carries its pid);
|
||||
/// `kill:<index>` releases it, whereupon it returns (Exit).
|
||||
fn role_server() {
|
||||
smarm::run(move || {
|
||||
let cluster = start(cfg("server", vec![])).expect("binds");
|
||||
println!("LISTENING {}", cluster.local_addr());
|
||||
let (tx, rx) = channel::<Ctl>();
|
||||
register(CTL, tx).unwrap();
|
||||
expose(CTL);
|
||||
println!("READY");
|
||||
let mut workers: HashMap<u32, smarm::channel::Sender<()>> = HashMap::new();
|
||||
loop {
|
||||
let ctl = rx.recv().unwrap();
|
||||
println!("CTL {}", ctl.cmd);
|
||||
let (text, pid): (String, Option<RemotePid<Erased>>) = match ctl.cmd.as_str() {
|
||||
"spawn" => {
|
||||
let (go_tx, go_rx) = channel::<()>();
|
||||
let p: Pid = spawn(move || {
|
||||
let _ = go_rx.recv();
|
||||
})
|
||||
.pid();
|
||||
workers.insert(p.index(), go_tx);
|
||||
(
|
||||
"ok".into(),
|
||||
Some(RemotePid::from_local(p).expect("identity set")),
|
||||
)
|
||||
}
|
||||
other => {
|
||||
let idx: u32 = other.strip_prefix("kill:").unwrap().parse().unwrap();
|
||||
if let Some(go) = workers.remove(&idx) {
|
||||
let _ = go.send(());
|
||||
}
|
||||
("killed".into(), None)
|
||||
}
|
||||
};
|
||||
send_to_remote(ctl.reply_to, Answer { text, pid }).unwrap();
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/// Client-side setup shared by both client roles: join, expose the reply
|
||||
/// path, hand back an `ask` closure and the membership stream.
|
||||
fn client_setup() -> (
|
||||
smarm::cluster::Cluster,
|
||||
smarm::cluster::membership::MembershipEvents,
|
||||
impl Fn(&str) -> Answer,
|
||||
) {
|
||||
let server_addr = std::env::var("SMARM_SERVER_ADDR").expect("SMARM_SERVER_ADDR");
|
||||
let cluster = start(cfg("client", vec![("server".into(), server_addr)])).expect("binds");
|
||||
let ev = subscribe().unwrap();
|
||||
wait_up(&ev, "server");
|
||||
let (tx, rx) = channel::<Answer>();
|
||||
let me: Pid<Client> = install::<Client>(tx);
|
||||
expose_type::<Answer>();
|
||||
let ask = move |cmd: &str| -> Answer {
|
||||
remote::send(
|
||||
RemoteName::new("server", CTL),
|
||||
Ctl {
|
||||
cmd: cmd.into(),
|
||||
reply_to: RemotePid::from_local(me).expect("identity set"),
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
rx.recv().unwrap()
|
||||
};
|
||||
(cluster, ev, ask)
|
||||
}
|
||||
|
||||
fn role_client() {
|
||||
smarm::run(move || {
|
||||
let (_cluster, ev, ask) = client_setup();
|
||||
|
||||
// 1. Headline: actor death ⇒ TRUE reason; link cut ⇒ Disconnected.
|
||||
let a = ask("spawn").pid.unwrap();
|
||||
let b = ask("spawn").pid.unwrap();
|
||||
let ma = monitor_remote(a.clone());
|
||||
let mb = monitor_remote(b.clone());
|
||||
ask(&format!("kill:{}", a.index()));
|
||||
let d = ma.recv().unwrap();
|
||||
assert_eq!(d.pid, a);
|
||||
println!("DOWN actor {:?}", d.reason);
|
||||
disconnect("server");
|
||||
let d = mb.recv().unwrap();
|
||||
assert_eq!(d.pid, b);
|
||||
println!("DOWN link {:?}", d.reason);
|
||||
|
||||
// 2. Reconnect does not resurrect. The connector redials on
|
||||
// node_down; over the NEW link a fresh monitor delivers, while
|
||||
// the old one (already answered) yields nothing further — order
|
||||
// proves it, and `b` is even still alive on the server.
|
||||
wait_up(&ev, "server");
|
||||
println!("RECONNECTED");
|
||||
let c = ask("spawn").pid.unwrap();
|
||||
let mc = monitor_remote(c.clone());
|
||||
ask(&format!("kill:{}", b.index()));
|
||||
ask(&format!("kill:{}", c.index()));
|
||||
assert_eq!(mc.recv().unwrap().reason, DownReason::Exit.into());
|
||||
let stray = matches!(mb.try_recv(), Ok(Some(_)));
|
||||
println!("RESURRECT stray={stray}");
|
||||
|
||||
// 3. Peer PROCESS killed ⇒ Disconnected too (nobody is left to send
|
||||
// Down): the parent SIGKILLs the server once it sees the marker.
|
||||
let e = ask("spawn").pid.unwrap();
|
||||
let me_ = monitor_remote(e.clone());
|
||||
println!("KILL SERVER NOW");
|
||||
let d = me_.recv().unwrap();
|
||||
assert_eq!(d.pid, e);
|
||||
println!("DOWN procdeath {:?}", d.reason);
|
||||
|
||||
println!("CLIENT DONE");
|
||||
loop {
|
||||
smarm::sleep(Duration::from_secs(3600));
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/// The slow role: liveness expiry (peer SIGSTOPped) followed by the link
|
||||
/// dropping for real (peer SIGKILLed) — one monitor, exactly one notice.
|
||||
fn role_client_stop() {
|
||||
smarm::run(move || {
|
||||
let (_cluster, _ev, ask) = client_setup();
|
||||
let a = ask("spawn").pid.unwrap();
|
||||
let ma = monitor_remote(a.clone());
|
||||
println!("STOP SERVER NOW");
|
||||
let d = ma.recv().unwrap(); // liveness expiry, ~liveness_timeout
|
||||
assert_eq!(d.pid, a);
|
||||
println!("DOWN stopped {:?}", d.reason);
|
||||
println!("KILL SERVER NOW");
|
||||
// Give the drop every chance to produce a second notice, then look.
|
||||
smarm::sleep(Duration::from_secs(1));
|
||||
let dup = matches!(ma.try_recv(), Ok(Some(_)));
|
||||
println!("DUPLICATE dup={dup}");
|
||||
println!("CLIENT DONE");
|
||||
loop {
|
||||
smarm::sleep(Duration::from_secs(3600));
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/// The Phase 4 c13 gate: partition vs. death distinguishable; nothing
|
||||
/// survives reconnect; a dead peer process is a Disconnected too.
|
||||
#[test]
|
||||
fn link_loss_is_disconnected_and_does_not_survive_reconnect() {
|
||||
maybe_child(ROLES);
|
||||
let mut server = spawn_node("server", &[]);
|
||||
let saddr = server.wait_listening();
|
||||
server.wait_line("READY", |l| l == "READY");
|
||||
let mut client = spawn_node("client", &[("SMARM_SERVER_ADDR", &saddr)]);
|
||||
client.wait_line("DOWN actor Local(Exit)", |l| l == "DOWN actor Local(Exit)");
|
||||
client.wait_line("DOWN link Disconnected", |l| l == "DOWN link Disconnected");
|
||||
client.wait_line("RECONNECTED", |l| l == "RECONNECTED");
|
||||
client.wait_line("RESURRECT stray=false", |l| l == "RESURRECT stray=false");
|
||||
client.wait_line("KILL SERVER NOW", |l| l == "KILL SERVER NOW");
|
||||
server.kill();
|
||||
client.wait_line("DOWN procdeath Disconnected", |l| {
|
||||
l == "DOWN procdeath Disconnected"
|
||||
});
|
||||
client.wait_line("CLIENT DONE", |l| l == "CLIENT DONE");
|
||||
}
|
||||
|
||||
/// Covers the invariant the headline test cannot: liveness expiry and the
|
||||
/// transport drop both firing for the same connection yield ONE notice.
|
||||
/// Runs on the fast [`timing`] (both roles) — was `#[ignore]`d at the 4s
|
||||
/// default until the p11 knobs landed.
|
||||
#[test]
|
||||
fn timeout_then_drop_yields_one_notice() {
|
||||
maybe_child(ROLES);
|
||||
let fast = ("SMARM_FAST_TIMING", "1");
|
||||
let mut server = spawn_node("server", &[fast]);
|
||||
let saddr = server.wait_listening();
|
||||
server.wait_line("READY", |l| l == "READY");
|
||||
let mut client = spawn_node("client_stop", &[("SMARM_SERVER_ADDR", &saddr), fast]);
|
||||
client.wait_line("STOP SERVER NOW", |l| l == "STOP SERVER NOW");
|
||||
let spid = server.pid().expect("server alive") as libc::pid_t;
|
||||
assert_eq!(unsafe { libc::kill(spid, libc::SIGSTOP) }, 0);
|
||||
client.wait_line("DOWN stopped Disconnected", |l| {
|
||||
l == "DOWN stopped Disconnected"
|
||||
});
|
||||
client.wait_line("KILL SERVER NOW", |l| l == "KILL SERVER NOW");
|
||||
server.kill(); // SIGKILL works on a stopped process; Drop would too
|
||||
client.wait_line("DUPLICATE dup=false", |l| l == "DUPLICATE dup=false");
|
||||
client.wait_line("CLIENT DONE", |l| l == "CLIENT DONE");
|
||||
}
|
||||
@@ -1,161 +0,0 @@
|
||||
//! RFC 010 — `Discovery::Withdrawn`: a strategy retracts a candidate and the
|
||||
//! connector stops dialing it.
|
||||
//!
|
||||
//! Cross-process: a plain *server* node, and a *client* whose strategy is a
|
||||
//! script: announce a decoy `(ghost, addr)` where `addr` is a raw
|
||||
//! `TcpListener` the client itself holds (an OS thread accepts and
|
||||
//! immediately closes, so every dial fails at handshake and the connector
|
||||
//! keeps retrying on backoff — the accept count is the dial count); after a
|
||||
//! beat, withdraw the decoy and announce the real server. The client waits
|
||||
//! for the server's `node_up` — which is *after* the withdrawal in the
|
||||
//! strategy's own stream — then watches the decoy's accept count stay flat
|
||||
//! across a window longer than the pending backoff. Before withdrawal it
|
||||
//! must have been climbing (≥ 1), or the negative proves nothing.
|
||||
#![cfg(feature = "cluster")]
|
||||
|
||||
mod common;
|
||||
|
||||
use common::{maybe_child, spawn_node};
|
||||
use smarm::channel::Sender;
|
||||
use smarm::cluster::discovery::{Discovery, Strategy};
|
||||
use smarm::cluster::envelope::NodeMeta;
|
||||
use smarm::cluster::membership::{subscribe, NodeEvent};
|
||||
use smarm::cluster::{start, Config, StaticSeeds, Timing};
|
||||
use std::net::TcpListener;
|
||||
use std::sync::atomic::{AtomicUsize, Ordering};
|
||||
use std::sync::Arc;
|
||||
use std::time::{Duration, Instant};
|
||||
|
||||
const ROLES: &[(&str, fn())] = &[("server", role_server), ("client", role_client)];
|
||||
|
||||
fn meta() -> NodeMeta {
|
||||
NodeMeta {
|
||||
role: "withdraw".into(),
|
||||
region: "local".into(),
|
||||
}
|
||||
}
|
||||
|
||||
fn role_server() {
|
||||
smarm::run(|| {
|
||||
let cluster = start(Config {
|
||||
node_name: "server".into(),
|
||||
meta: meta(),
|
||||
listen_addr: std::env::var("SMARM_LISTEN_ADDR")
|
||||
.unwrap_or_else(|_| "127.0.0.1:0".into()),
|
||||
strategy: Box::new(StaticSeeds::new(Vec::<(String, String)>::new())),
|
||||
timing: Timing::default(),
|
||||
})
|
||||
.expect("binds");
|
||||
println!("LISTENING {}", cluster.local_addr());
|
||||
loop {
|
||||
smarm::sleep(Duration::from_secs(3600));
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/// Scripted strategy: decoy, pause, withdraw decoy, real server, done.
|
||||
struct Script {
|
||||
decoy: String,
|
||||
server: String,
|
||||
}
|
||||
|
||||
impl Strategy for Script {
|
||||
fn run(self: Box<Self>, out: Sender<Discovery>) {
|
||||
let _ = out.send(Discovery::Candidate {
|
||||
name: "ghost".into(),
|
||||
addr: self.decoy.clone(),
|
||||
});
|
||||
// Long enough for the 250ms/500ms retries to land: ≥ 3 dials.
|
||||
smarm::sleep(Duration::from_millis(1100));
|
||||
let _ = out.send(Discovery::Withdrawn {
|
||||
name: "ghost".into(),
|
||||
addr: self.decoy,
|
||||
});
|
||||
let _ = out.send(Discovery::Candidate {
|
||||
name: "server".into(),
|
||||
addr: self.server,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
fn role_client() {
|
||||
let server_addr = std::env::var("SMARM_SERVER_ADDR").expect("SMARM_SERVER_ADDR");
|
||||
// The decoy: accept-and-close on an OS thread; count every accept.
|
||||
let decoy = TcpListener::bind("127.0.0.1:0").unwrap();
|
||||
let decoy_addr = decoy.local_addr().unwrap().to_string();
|
||||
let dials = Arc::new(AtomicUsize::new(0));
|
||||
let counter = dials.clone();
|
||||
std::thread::spawn(move || {
|
||||
for conn in decoy.incoming() {
|
||||
counter.fetch_add(1, Ordering::SeqCst);
|
||||
drop(conn);
|
||||
}
|
||||
});
|
||||
|
||||
smarm::run(move || {
|
||||
let _cluster = start(Config {
|
||||
node_name: "client".into(),
|
||||
meta: meta(),
|
||||
listen_addr: "127.0.0.1:0".into(),
|
||||
strategy: Box::new(Script {
|
||||
decoy: decoy_addr,
|
||||
server: server_addr,
|
||||
}),
|
||||
timing: Timing::default(),
|
||||
})
|
||||
.expect("binds");
|
||||
let ev = subscribe().unwrap();
|
||||
loop {
|
||||
match ev.rx.recv() {
|
||||
Ok(NodeEvent::NodeUp(i)) if i.name == "server" => break,
|
||||
Ok(_) => continue,
|
||||
Err(_) => panic!("manager gone"),
|
||||
}
|
||||
}
|
||||
// The withdrawal preceded the server candidate in the strategy's
|
||||
// stream, so it has been applied. Any dial that started before it
|
||||
// is bounded by the connect+handshake deadlines; let it drain, then
|
||||
// hold the count flat across a window longer than the pending
|
||||
// backoff would be (1s at this point, 2s next).
|
||||
let before = dials.load(Ordering::SeqCst);
|
||||
smarm::sleep(Duration::from_millis(500));
|
||||
let settled = dials.load(Ordering::SeqCst);
|
||||
let t0 = Instant::now();
|
||||
while t0.elapsed() < Duration::from_millis(3000) {
|
||||
smarm::sleep(Duration::from_millis(100));
|
||||
}
|
||||
let after = dials.load(Ordering::SeqCst);
|
||||
println!("WITHDRAWN before={before} settled={settled} after={after}");
|
||||
println!("CLIENT DONE");
|
||||
loop {
|
||||
smarm::sleep(Duration::from_secs(3600));
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn withdrawn_candidate_is_no_longer_dialed() {
|
||||
maybe_child(ROLES);
|
||||
let mut server = spawn_node("server", &[]);
|
||||
let saddr = server.wait_listening();
|
||||
let mut client = spawn_node("client", &[("SMARM_SERVER_ADDR", &saddr)]);
|
||||
let line = client.wait_line("WITHDRAWN", |l| l.starts_with("WITHDRAWN "));
|
||||
let mut nums = line
|
||||
.split_whitespace()
|
||||
.skip(1)
|
||||
.map(|kv| kv.split_once('=').unwrap().1.parse::<usize>().unwrap());
|
||||
let (before, settled, after) = (
|
||||
nums.next().unwrap(),
|
||||
nums.next().unwrap(),
|
||||
nums.next().unwrap(),
|
||||
);
|
||||
assert!(
|
||||
before >= 1,
|
||||
"decoy was never dialed; the negative proves nothing: {line}"
|
||||
);
|
||||
assert_eq!(
|
||||
settled, after,
|
||||
"connector kept dialing a withdrawn candidate: {line}"
|
||||
);
|
||||
client.wait_line("CLIENT DONE", |l| l == "CLIENT DONE");
|
||||
}
|
||||
@@ -1,276 +0,0 @@
|
||||
//! RFC 010 c2 — owned envelope tests (roadmap: per-frame roundtrip,
|
||||
//! truncation mid-field, unknown tag, length prefix lying long and short,
|
||||
//! zero-length payload, adversarial lengths).
|
||||
#![cfg(feature = "cluster")]
|
||||
|
||||
use serde::{Deserialize, Serialize};
|
||||
use smarm::cluster::envelope::{
|
||||
decode_payload, encode_payload, DecodeError, Frame, NodeMeta, RejectReason, MAX_FRAME_LEN,
|
||||
PROTO_VERSION,
|
||||
};
|
||||
use smarm::cluster::RemoteDownReason;
|
||||
use smarm::monitor::DownReason;
|
||||
use smarm::pg::Incarnation;
|
||||
|
||||
fn meta() -> NodeMeta {
|
||||
NodeMeta {
|
||||
role: "worker".into(),
|
||||
region: "eu-west".into(),
|
||||
}
|
||||
}
|
||||
|
||||
fn all_frames() -> Vec<Frame> {
|
||||
vec![
|
||||
Frame::Hello {
|
||||
proto_version: PROTO_VERSION,
|
||||
build_hash: 0xDEAD_BEEF_CAFE_F00D,
|
||||
node_name: "alpha".into(),
|
||||
incarnation: Incarnation::new(7),
|
||||
meta: meta(),
|
||||
},
|
||||
Frame::HelloAck {
|
||||
node_name: "beta".into(),
|
||||
incarnation: Incarnation::new(9),
|
||||
meta: meta(),
|
||||
},
|
||||
Frame::HelloReject {
|
||||
reason: RejectReason::NameTaken,
|
||||
},
|
||||
Frame::Heartbeat,
|
||||
Frame::Send {
|
||||
index: 42,
|
||||
generation: 3,
|
||||
type_hash: 0x1234_5678_9ABC_DEF0,
|
||||
payload: vec![1, 2, 3, 4, 5],
|
||||
},
|
||||
Frame::SendNamed {
|
||||
name: "the_counter".into(),
|
||||
type_hash: 0xFFFF_0000_FFFF_0000,
|
||||
payload: vec![],
|
||||
},
|
||||
Frame::Monitor {
|
||||
monitor_id: 77,
|
||||
index: 42,
|
||||
generation: 3,
|
||||
},
|
||||
Frame::Demonitor { monitor_id: 77 },
|
||||
Frame::Down {
|
||||
monitor_id: 77,
|
||||
reason: RemoteDownReason::Local(DownReason::Panic),
|
||||
},
|
||||
Frame::Down {
|
||||
monitor_id: 78,
|
||||
reason: RemoteDownReason::Disconnected,
|
||||
},
|
||||
]
|
||||
}
|
||||
|
||||
fn encode_one(f: &Frame) -> Vec<u8> {
|
||||
let mut buf = Vec::new();
|
||||
f.encode(&mut buf).unwrap();
|
||||
buf
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn per_frame_roundtrip() {
|
||||
for f in all_frames() {
|
||||
let buf = encode_one(&f);
|
||||
let (decoded, consumed) = Frame::decode(&buf).unwrap().unwrap();
|
||||
assert_eq!(decoded, f, "roundtrip mismatch");
|
||||
assert_eq!(consumed, buf.len(), "consumed != buffer length for {f:?}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn back_to_back_frames_decode_sequentially() {
|
||||
let mut buf = Vec::new();
|
||||
for f in all_frames() {
|
||||
f.encode(&mut buf).unwrap();
|
||||
}
|
||||
let mut off = 0;
|
||||
let mut decoded = Vec::new();
|
||||
while off < buf.len() {
|
||||
let (f, n) = Frame::decode(&buf[off..]).unwrap().unwrap();
|
||||
decoded.push(f);
|
||||
off += n;
|
||||
}
|
||||
assert_eq!(decoded, all_frames());
|
||||
assert_eq!(off, buf.len());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn heartbeat_golden_bytes() {
|
||||
// Locks the layout: u32 LE length prefix, then the tag byte.
|
||||
let buf = encode_one(&Frame::Heartbeat);
|
||||
assert_eq!(buf, vec![1, 0, 0, 0, 4]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn zero_length_payload_roundtrips() {
|
||||
let f = Frame::Send {
|
||||
index: 0,
|
||||
generation: 0,
|
||||
type_hash: 0,
|
||||
payload: vec![],
|
||||
};
|
||||
let buf = encode_one(&f);
|
||||
let (decoded, consumed) = Frame::decode(&buf).unwrap().unwrap();
|
||||
assert_eq!(decoded, f);
|
||||
assert_eq!(consumed, buf.len());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn incomplete_is_none_not_error() {
|
||||
let buf = encode_one(&all_frames()[0]);
|
||||
// Every strict prefix short of the full frame must report "need more".
|
||||
for cut in 0..buf.len() {
|
||||
assert_eq!(
|
||||
Frame::decode(&buf[..cut]).unwrap(),
|
||||
None,
|
||||
"cut at {cut} should be incomplete"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn unknown_frame_tag() {
|
||||
let buf = vec![1, 0, 0, 0, 250];
|
||||
assert_eq!(Frame::decode(&buf), Err(DecodeError::UnknownTag(250)));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn unknown_enum_tags() {
|
||||
// HelloReject with a bogus reason tag.
|
||||
let buf = vec![2, 0, 0, 0, 3, 99];
|
||||
assert_eq!(
|
||||
Frame::decode(&buf),
|
||||
Err(DecodeError::UnknownEnumTag {
|
||||
what: "RejectReason",
|
||||
tag: 99
|
||||
})
|
||||
);
|
||||
// Down with a bogus reason tag (id = 0u64).
|
||||
let mut buf = vec![10, 0, 0, 0, 9];
|
||||
buf.extend_from_slice(&0u64.to_le_bytes());
|
||||
buf.push(200);
|
||||
assert_eq!(
|
||||
Frame::decode(&buf),
|
||||
Err(DecodeError::UnknownEnumTag {
|
||||
what: "RemoteDownReason",
|
||||
tag: 200
|
||||
})
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn length_prefix_lying_long_with_bytes_present_is_trailing() {
|
||||
let mut buf = encode_one(&Frame::Heartbeat);
|
||||
// Declare 3 extra body bytes and actually supply them.
|
||||
let declared = u32::from_le_bytes([buf[0], buf[1], buf[2], buf[3]]) + 3;
|
||||
buf[0..4].copy_from_slice(&declared.to_le_bytes());
|
||||
buf.extend_from_slice(&[0xAA, 0xBB, 0xCC]);
|
||||
assert_eq!(Frame::decode(&buf), Err(DecodeError::Trailing { extra: 3 }));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn length_prefix_lying_long_without_bytes_is_incomplete() {
|
||||
// Indistinguishable from a partial read — must be None, not an error.
|
||||
let mut buf = encode_one(&Frame::Heartbeat);
|
||||
let declared = u32::from_le_bytes([buf[0], buf[1], buf[2], buf[3]]) + 3;
|
||||
buf[0..4].copy_from_slice(&declared.to_le_bytes());
|
||||
assert_eq!(Frame::decode(&buf).unwrap(), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn length_prefix_lying_short_truncates_a_field() {
|
||||
let f = &all_frames()[0]; // Hello: plenty of fields to cut into
|
||||
let mut buf = encode_one(f);
|
||||
let declared = u32::from_le_bytes([buf[0], buf[1], buf[2], buf[3]]);
|
||||
let lie = declared - 4; // cut mid-field
|
||||
buf[0..4].copy_from_slice(&lie.to_le_bytes());
|
||||
assert_eq!(Frame::decode(&buf), Err(DecodeError::Truncated));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn truncation_mid_string_field() {
|
||||
// A frame whose declared length is intact but whose inner string length
|
||||
// runs past the body: SendNamed claiming a 1000-byte name in a tiny body.
|
||||
let mut body = vec![6u8]; // TAG_SEND_NAMED
|
||||
body.extend_from_slice(&1000u16.to_le_bytes());
|
||||
body.extend_from_slice(b"short");
|
||||
let mut buf = (body.len() as u32).to_le_bytes().to_vec();
|
||||
buf.extend_from_slice(&body);
|
||||
assert_eq!(Frame::decode(&buf), Err(DecodeError::Truncated));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn adversarial_lengths() {
|
||||
// Length prefix of u32::MAX: reject as oversized, do not wait for 4 GiB.
|
||||
let buf = [0xFF, 0xFF, 0xFF, 0xFF, 0];
|
||||
assert_eq!(
|
||||
Frame::decode(&buf),
|
||||
Err(DecodeError::FrameTooLarge {
|
||||
declared: u32::MAX as usize
|
||||
})
|
||||
);
|
||||
// Just over the cap: also rejected.
|
||||
let over = (MAX_FRAME_LEN as u32 + 1).to_le_bytes();
|
||||
assert!(matches!(
|
||||
Frame::decode(&over),
|
||||
Err(DecodeError::FrameTooLarge { .. })
|
||||
));
|
||||
// Zero-length frame: there is no tag byte; corrupt, not incomplete.
|
||||
let buf = [0, 0, 0, 0];
|
||||
assert_eq!(Frame::decode(&buf), Err(DecodeError::EmptyFrame));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn invalid_utf8_in_string_field() {
|
||||
let mut buf = encode_one(&Frame::SendNamed {
|
||||
name: "abcd".into(),
|
||||
type_hash: 0,
|
||||
payload: vec![],
|
||||
});
|
||||
// name bytes start after: 4 (len) + 1 (tag) + 2 (str len) = offset 7
|
||||
buf[7] = 0xFF;
|
||||
assert_eq!(Frame::decode(&buf), Err(DecodeError::Utf8));
|
||||
}
|
||||
|
||||
#[derive(Debug, PartialEq, Serialize, Deserialize)]
|
||||
struct Ping {
|
||||
seq: u64,
|
||||
label: String,
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn payload_seam_roundtrip() {
|
||||
let ping = Ping {
|
||||
seq: 31337,
|
||||
label: "hello".into(),
|
||||
};
|
||||
let blob = encode_payload(&ping).unwrap();
|
||||
// Carry it through a real frame, as it will travel in c9.
|
||||
let f = Frame::Send {
|
||||
index: 1,
|
||||
generation: 1,
|
||||
type_hash: 0xABCD,
|
||||
payload: blob,
|
||||
};
|
||||
let buf = encode_one(&f);
|
||||
let (decoded, _) = Frame::decode(&buf).unwrap().unwrap();
|
||||
let Frame::Send { payload, .. } = decoded else {
|
||||
panic!("wrong frame");
|
||||
};
|
||||
let back: Ping = decode_payload(&payload).unwrap();
|
||||
assert_eq!(back, ping);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn payload_seam_rejects_truncated_blob() {
|
||||
let blob = encode_payload(&Ping {
|
||||
seq: 1,
|
||||
label: "x".into(),
|
||||
})
|
||||
.unwrap();
|
||||
assert!(decode_payload::<Ping>(&blob[..blob.len() - 1]).is_err());
|
||||
}
|
||||
@@ -1,161 +0,0 @@
|
||||
//! RFC 010 c8 — exposure registry + type hashing. Purely local, no network.
|
||||
//!
|
||||
//! Payload types are std types (`String`, `u64`) because the crate's serde is
|
||||
//! deliberately derive-less (`default-features = false`) — user crates bring
|
||||
//! their own derive; the contract here is `DeserializeOwned`.
|
||||
//!
|
||||
//! The hash-stability test re-execs the current binary (the c4 harness): the
|
||||
//! guarantee under test is "stable across runs in the SAME binary" — exactly
|
||||
//! what the build-hash handshake reduces the mesh to — not stability across
|
||||
//! builds, which the scope guard explicitly rejects.
|
||||
#![cfg(feature = "cluster")]
|
||||
|
||||
mod common;
|
||||
|
||||
use common::{maybe_child, spawn_node};
|
||||
use smarm::cluster::envelope::encode_payload;
|
||||
use smarm::cluster::expose::{
|
||||
decode_deliver, decoder_registered, expose, expose_type, exposed_hash, exposed_names,
|
||||
type_hash, DeliverError,
|
||||
};
|
||||
use smarm::monitor::{monitor, terminal_reason, DownReason};
|
||||
use smarm::{channel, register, run, spawn, Name};
|
||||
|
||||
const ROLES: &[(&str, fn())] = &[("hasher", role_hasher)];
|
||||
|
||||
/// Print the hashes this process computes; the parent (a different run of
|
||||
/// the same binary) compares against its own.
|
||||
fn role_hasher() {
|
||||
println!("HASH-STRING {}", type_hash::<String>());
|
||||
println!("HASH-U64 {}", type_hash::<u64>());
|
||||
}
|
||||
|
||||
const GREETER: Name<String> = Name::new("expose-test.greeter");
|
||||
|
||||
/// Exposed and unexposed lookup, the returned hash, and the audit listing.
|
||||
#[test]
|
||||
fn exposed_and_unexposed_lookup() {
|
||||
maybe_child(ROLES);
|
||||
run(|| {
|
||||
let h = expose(GREETER);
|
||||
assert_eq!(h, type_hash::<String>());
|
||||
assert_eq!(exposed_hash("expose-test.greeter"), Some(h));
|
||||
assert_eq!(exposed_hash("never-exposed"), None);
|
||||
assert!(exposed_names().contains(&("expose-test.greeter", h)));
|
||||
});
|
||||
}
|
||||
|
||||
/// Distinct types land on distinct hashes (FNV over distinct TypeIds — a
|
||||
/// smoke assertion; a collision would degrade to a decode error, never a
|
||||
/// misroute, per RFC §3).
|
||||
#[test]
|
||||
fn distinct_types_distinct_hashes() {
|
||||
maybe_child(ROLES);
|
||||
run(|| {
|
||||
assert_ne!(type_hash::<String>(), type_hash::<u64>());
|
||||
assert_ne!(type_hash::<String>(), type_hash::<Vec<u8>>());
|
||||
});
|
||||
}
|
||||
|
||||
/// The decode-and-deliver contract: a registered hash decodes into the
|
||||
/// target's typed channel; an unknown hash, corrupt bytes, and a missing
|
||||
/// channel each fail without delivering — `WrongChannel`, never a misroute.
|
||||
#[test]
|
||||
fn decoder_registration_and_delivery() {
|
||||
maybe_child(ROLES);
|
||||
run(|| {
|
||||
let h_string = expose_type::<String>();
|
||||
let h_u64 = expose_type::<u64>();
|
||||
assert!(decoder_registered(h_string));
|
||||
assert!(!decoder_registered(h_string.wrapping_add(1)));
|
||||
|
||||
// A live actor with a String channel (registered from its own body,
|
||||
// announced via a ready signal — the tests/registry.rs idiom).
|
||||
let (ready_tx, ready_rx) = channel::<()>();
|
||||
let (stop_tx, stop_rx) = channel::<()>();
|
||||
let (msg_tx, msg_rx) = channel::<String>();
|
||||
let pid = spawn(move || {
|
||||
register(Name::<String>::new("expose-test.sink"), msg_tx).unwrap();
|
||||
ready_tx.send(()).unwrap();
|
||||
let _ = stop_rx.recv();
|
||||
})
|
||||
.pid();
|
||||
ready_rx.recv().unwrap();
|
||||
|
||||
// Happy path: decode + deliver through the published channel.
|
||||
let bytes = encode_payload("hello across the seam").unwrap();
|
||||
decode_deliver(h_string, pid, &bytes).unwrap();
|
||||
assert_eq!(msg_rx.recv().unwrap(), "hello across the seam");
|
||||
|
||||
// Unknown hash: nothing was registered under it.
|
||||
assert!(matches!(
|
||||
decode_deliver(h_string.wrapping_add(1), pid, &bytes),
|
||||
Err(DeliverError::UnknownType)
|
||||
));
|
||||
|
||||
// Corrupt bytes: the decoder fails before any send.
|
||||
assert!(matches!(
|
||||
decode_deliver(h_string, pid, &[0xff; 3]),
|
||||
Err(DeliverError::Decode(_))
|
||||
));
|
||||
|
||||
// Right decoder, wrong channel: the actor has no u64 channel, so the
|
||||
// decoded value is refused — the NoChannel guarantee.
|
||||
let u64_bytes = encode_payload(&7u64).unwrap();
|
||||
assert!(matches!(
|
||||
decode_deliver(h_u64, pid, &u64_bytes),
|
||||
Err(DeliverError::WrongChannel)
|
||||
));
|
||||
|
||||
stop_tx.send(()).unwrap();
|
||||
});
|
||||
}
|
||||
|
||||
/// `expose` and the bridge crossing agree on the resulting set: both funnel
|
||||
/// the pid-boundary mark through the watchable machinery, so an exposed
|
||||
/// name's holder dies with a terminal record — the exact observable
|
||||
/// `mark_watchable` guarantees the membrane. (For named holders the mark is
|
||||
/// already stamped by `register` itself; this pins the shared contract.)
|
||||
#[test]
|
||||
fn expose_and_bridge_crossing_agree_on_the_set() {
|
||||
maybe_child(ROLES);
|
||||
run(|| {
|
||||
let (ready_tx, ready_rx) = channel::<()>();
|
||||
let (stop_tx, stop_rx) = channel::<()>();
|
||||
let (msg_tx, _msg_rx) = channel::<String>();
|
||||
let pid = spawn(move || {
|
||||
register(GREETER, msg_tx).unwrap();
|
||||
ready_tx.send(()).unwrap();
|
||||
let _ = stop_rx.recv();
|
||||
})
|
||||
.pid();
|
||||
ready_rx.recv().unwrap();
|
||||
|
||||
expose(GREETER);
|
||||
let m = monitor(pid);
|
||||
stop_tx.send(()).unwrap();
|
||||
assert_eq!(m.rx.recv().unwrap().reason, DownReason::Exit);
|
||||
assert_eq!(terminal_reason(pid), Some(DownReason::Exit));
|
||||
});
|
||||
}
|
||||
|
||||
/// Hash stability across runs in the same binary: a re-exec of this binary
|
||||
/// computes the same hashes this process does.
|
||||
#[test]
|
||||
fn hash_stable_across_runs_in_same_binary() {
|
||||
maybe_child(ROLES);
|
||||
let (mine_string, mine_u64) = {
|
||||
// Computing a TypeId hash needs no runtime, but keep the contract
|
||||
// uniform with real call sites.
|
||||
(type_hash::<String>(), type_hash::<u64>())
|
||||
};
|
||||
let mut child = spawn_node("hasher", &[]);
|
||||
let line = child.wait_line("HASH-STRING", |l| l.starts_with("HASH-STRING "));
|
||||
assert_eq!(
|
||||
line["HASH-STRING ".len()..].parse::<u64>().unwrap(),
|
||||
mine_string
|
||||
);
|
||||
let line = child.wait_line("HASH-U64", |l| l.starts_with("HASH-U64 "));
|
||||
assert_eq!(line["HASH-U64 ".len()..].parse::<u64>().unwrap(), mine_u64);
|
||||
child.wait_exit();
|
||||
}
|
||||
@@ -1,240 +0,0 @@
|
||||
//! RFC 010 c5 — handshake state-machine tests (roadmap: happy path; hash
|
||||
//! mismatch; proto-version mismatch; name already claimed; simultaneous-connect
|
||||
//! tie-break; garbage before Hello). Pure — no IO, no actors, no runtime.
|
||||
#![cfg(feature = "cluster")]
|
||||
|
||||
use smarm::cluster::envelope::{Frame, NodeMeta, RejectReason, PROTO_VERSION};
|
||||
use smarm::cluster::handshake::{
|
||||
dial_wins, Initiator, InitiatorOutcome, Local, PeerStanding, Responder, ResponderOutcome,
|
||||
};
|
||||
use smarm::pg::Incarnation;
|
||||
|
||||
const HASH: u64 = 0xDEAD_BEEF_CAFE_F00D;
|
||||
|
||||
fn local(name: &str) -> Local {
|
||||
Local {
|
||||
node_name: name.into(),
|
||||
incarnation: Incarnation::new(7),
|
||||
build_hash: HASH,
|
||||
meta: NodeMeta {
|
||||
role: "worker".into(),
|
||||
region: "eu-west".into(),
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
/// The Hello that `Initiator::new(&local(name))` emits, built by hand.
|
||||
fn hello_from(name: &str) -> Frame {
|
||||
let l = local(name);
|
||||
Frame::Hello {
|
||||
proto_version: PROTO_VERSION,
|
||||
build_hash: l.build_hash,
|
||||
node_name: l.node_name,
|
||||
incarnation: l.incarnation,
|
||||
meta: l.meta,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn happy_path_establishes_both_ends() {
|
||||
// alpha dials beta.
|
||||
let (initiator, hello) = Initiator::new(&local("alpha"));
|
||||
assert_eq!(hello, hello_from("alpha"), "initiator emits its identity");
|
||||
|
||||
let responder = Responder::new(local("beta"));
|
||||
let (reply, peer) = match responder.on_frame(hello, PeerStanding::Free) {
|
||||
ResponderOutcome::Accepted { reply, peer } => (reply, peer),
|
||||
other => panic!("expected Accepted, got {other:?}"),
|
||||
};
|
||||
assert_eq!(peer.node_name, "alpha");
|
||||
assert_eq!(peer.incarnation, Incarnation::new(7));
|
||||
assert_eq!(peer.meta.role, "worker");
|
||||
|
||||
// The ack carries the responder's identity, no hash/version (one-sided
|
||||
// check — sound because equality is symmetric).
|
||||
let l = local("beta");
|
||||
assert_eq!(
|
||||
reply,
|
||||
Frame::HelloAck {
|
||||
node_name: l.node_name,
|
||||
incarnation: l.incarnation,
|
||||
meta: l.meta,
|
||||
}
|
||||
);
|
||||
|
||||
match initiator.on_frame(reply) {
|
||||
InitiatorOutcome::Established(peer) => {
|
||||
assert_eq!(peer.node_name, "beta");
|
||||
assert_eq!(peer.incarnation, Incarnation::new(7));
|
||||
assert_eq!(peer.meta.region, "eu-west");
|
||||
}
|
||||
other => panic!("expected Established, got {other:?}"),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn hash_mismatch_rejected() {
|
||||
let responder = Responder::new(local("beta"));
|
||||
let hello = Frame::Hello {
|
||||
proto_version: PROTO_VERSION,
|
||||
build_hash: HASH ^ 1,
|
||||
node_name: "alpha".into(),
|
||||
incarnation: Incarnation::new(7),
|
||||
meta: local("alpha").meta,
|
||||
};
|
||||
match responder.on_frame(hello, PeerStanding::Free) {
|
||||
ResponderOutcome::Rejected { reply, reason } => {
|
||||
assert_eq!(reason, RejectReason::HashMismatch);
|
||||
assert_eq!(reply, Frame::HelloReject { reason });
|
||||
}
|
||||
other => panic!("expected Rejected, got {other:?}"),
|
||||
}
|
||||
|
||||
// The dialer side of the same story: a reject frame comes back.
|
||||
let (initiator, _hello) = Initiator::new(&local("alpha"));
|
||||
match initiator.on_frame(Frame::HelloReject {
|
||||
reason: RejectReason::HashMismatch,
|
||||
}) {
|
||||
InitiatorOutcome::Rejected(RejectReason::HashMismatch) => {}
|
||||
other => panic!("expected Rejected(HashMismatch), got {other:?}"),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn proto_version_mismatch_rejected_and_checked_first() {
|
||||
// Both proto and hash wrong: proto wins — nothing after the version can
|
||||
// be trusted, and HelloReject is the cross-version compatibility anchor.
|
||||
let responder = Responder::new(local("beta"));
|
||||
let hello = Frame::Hello {
|
||||
proto_version: PROTO_VERSION + 1,
|
||||
build_hash: HASH ^ 1,
|
||||
node_name: "alpha".into(),
|
||||
incarnation: Incarnation::new(7),
|
||||
meta: local("alpha").meta,
|
||||
};
|
||||
match responder.on_frame(hello, PeerStanding::Free) {
|
||||
ResponderOutcome::Rejected { reason, .. } => {
|
||||
assert_eq!(reason, RejectReason::ProtoVersion);
|
||||
}
|
||||
other => panic!("expected Rejected, got {other:?}"),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn claimed_name_rejected() {
|
||||
let responder = Responder::new(local("beta"));
|
||||
let ctx = PeerStanding::Claimed;
|
||||
match responder.on_frame(hello_from("alpha"), ctx) {
|
||||
ResponderOutcome::Rejected { reason, .. } => {
|
||||
assert_eq!(reason, RejectReason::NameTaken);
|
||||
}
|
||||
other => panic!("expected Rejected, got {other:?}"),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn own_name_offered_rejected_as_name_taken() {
|
||||
// Self-connect or genuine collision: the responder's own name arrives.
|
||||
let responder = Responder::new(local("beta"));
|
||||
match responder.on_frame(hello_from("beta"), PeerStanding::Free) {
|
||||
ResponderOutcome::Rejected { reason, .. } => {
|
||||
assert_eq!(reason, RejectReason::NameTaken);
|
||||
}
|
||||
other => panic!("expected Rejected, got {other:?}"),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn hash_checked_before_name() {
|
||||
// Wrong hash AND claimed name: hash wins (validity before identity).
|
||||
let responder = Responder::new(local("beta"));
|
||||
let hello = Frame::Hello {
|
||||
proto_version: PROTO_VERSION,
|
||||
build_hash: HASH ^ 1,
|
||||
node_name: "alpha".into(),
|
||||
incarnation: Incarnation::new(7),
|
||||
meta: local("alpha").meta,
|
||||
};
|
||||
let ctx = PeerStanding::Claimed;
|
||||
match responder.on_frame(hello, ctx) {
|
||||
ResponderOutcome::Rejected { reason, .. } => {
|
||||
assert_eq!(reason, RejectReason::HashMismatch);
|
||||
}
|
||||
other => panic!("expected Rejected, got {other:?}"),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn dial_wins_is_deterministic_and_antisymmetric() {
|
||||
// The smaller name's dial survives; both ends compute the same verdict.
|
||||
assert!(dial_wins("alpha", "beta"));
|
||||
assert!(!dial_wins("beta", "alpha"));
|
||||
for (a, b) in [("a", "b"), ("node-1", "node-2"), ("x", "xx")] {
|
||||
assert_ne!(dial_wins(a, b), dial_wins(b, a), "({a}, {b})");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn simultaneous_connect_exactly_one_side_accepts() {
|
||||
// alpha and beta dial each other at once. Each responder sees the peer's
|
||||
// Hello while its own dial is in flight.
|
||||
let ctx = PeerStanding::Dialing;
|
||||
|
||||
// On beta: inbound is alpha's dial; alpha < beta, so the inbound wins.
|
||||
let on_beta = Responder::new(local("beta")).on_frame(hello_from("alpha"), ctx);
|
||||
assert!(
|
||||
matches!(on_beta, ResponderOutcome::Accepted { .. }),
|
||||
"beta must accept alpha's dial, got {on_beta:?}"
|
||||
);
|
||||
|
||||
// On alpha: inbound is beta's dial; it loses — close silently, no frame
|
||||
// (ratified: both ends can compute the outcome, a reject adds nothing).
|
||||
let on_alpha = Responder::new(local("alpha")).on_frame(hello_from("beta"), ctx);
|
||||
assert!(
|
||||
matches!(on_alpha, ResponderOutcome::TieBreakLoss),
|
||||
"alpha must silently drop beta's dial, got {on_alpha:?}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tiebreak_loss_only_applies_when_dialing() {
|
||||
// Same inbound Hello, no dial in flight: plain accept.
|
||||
let on_alpha = Responder::new(local("alpha")).on_frame(hello_from("beta"), PeerStanding::Free);
|
||||
assert!(matches!(on_alpha, ResponderOutcome::Accepted { .. }));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn garbage_before_hello_fails_without_reply() {
|
||||
// Any valid-but-wrong frame before Hello is a protocol violation: close,
|
||||
// no reject frame. (Undecodable bytes are the codec's Err, not ours.)
|
||||
for frame in [
|
||||
Frame::Heartbeat,
|
||||
Frame::HelloAck {
|
||||
node_name: "alpha".into(),
|
||||
incarnation: Incarnation::new(7),
|
||||
meta: local("alpha").meta,
|
||||
},
|
||||
Frame::Demonitor { monitor_id: 3 },
|
||||
] {
|
||||
let out = Responder::new(local("beta")).on_frame(frame.clone(), PeerStanding::Free);
|
||||
match out {
|
||||
ResponderOutcome::Failed(f) => assert_eq!(f, frame),
|
||||
other => panic!("expected Failed({frame:?}), got {other:?}"),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn garbage_before_ack_fails_the_initiator() {
|
||||
for frame in [
|
||||
Frame::Heartbeat,
|
||||
hello_from("beta"),
|
||||
Frame::Demonitor { monitor_id: 3 },
|
||||
] {
|
||||
let (initiator, _hello) = Initiator::new(&local("alpha"));
|
||||
match initiator.on_frame(frame.clone()) {
|
||||
InitiatorOutcome::Failed(f) => assert_eq!(f, frame),
|
||||
other => panic!("expected Failed({frame:?}), got {other:?}"),
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,270 +0,0 @@
|
||||
//! RFC 010 c7a — membership events and the view, at the manager.
|
||||
//!
|
||||
//! Same construction as the c6a lifecycle suite: the handshake is bypassed,
|
||||
//! connections are built already-established over localhost TCP pairs with
|
||||
//! fabricated `Peer`s, and the manager is started plainly so the test can
|
||||
//! terminate. What is under test is the membership layer that c7 adds to the
|
||||
//! manager: `node_up`/`node_down` events to subscribers (snapshot-then-stream),
|
||||
//! the view, and NodeId identity — memoized per `(name, incarnation)`, so a
|
||||
//! reconnect blip keeps its id and a restart (new incarnation) gets a fresh one.
|
||||
#![cfg(feature = "cluster")]
|
||||
|
||||
use std::time::Duration;
|
||||
|
||||
use smarm::cluster::envelope::NodeMeta;
|
||||
use smarm::cluster::handshake::Peer;
|
||||
use smarm::cluster::manager::{Call, Manager, Reply, MANAGER};
|
||||
use smarm::cluster::membership::{subscribe, view, MembershipEvents, NodeEvent};
|
||||
use smarm::cluster::spawn_established;
|
||||
use smarm::cluster::transport::tcp::TcpTransport;
|
||||
use smarm::cluster::transport::{Conn, FramedConn, Transport};
|
||||
use smarm::cluster::Timing;
|
||||
use smarm::gen_server::{self, GenServerBuilder};
|
||||
use smarm::pg::{Incarnation, NodeId};
|
||||
use smarm::run;
|
||||
|
||||
/// A fabricated post-handshake peer identity, with the incarnation under the
|
||||
/// test's control (it is identity-bearing here, unlike in the c6a suite).
|
||||
fn peer(name: &str, inc: u32) -> Peer {
|
||||
Peer {
|
||||
node_name: name.to_string(),
|
||||
incarnation: Incarnation::new(inc),
|
||||
meta: NodeMeta {
|
||||
role: "test".to_string(),
|
||||
region: "test".to_string(),
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
/// One established transport pair over localhost (TCP backlog covers the
|
||||
/// sequential dial-then-accept, as in the c3 conformance suite).
|
||||
fn pair(t: &dyn Transport) -> (Box<dyn Conn>, Box<dyn Conn>) {
|
||||
let mut l = t.listen("127.0.0.1:0").unwrap();
|
||||
let a = t.dial(&l.local_addr()).unwrap();
|
||||
let b = l.accept().unwrap();
|
||||
(a, b)
|
||||
}
|
||||
|
||||
/// The next event, or a panic naming the wait. The bound is generous against
|
||||
/// a sub-millisecond real cost.
|
||||
fn next_event(ev: &MembershipEvents, waiting_for: &str) -> NodeEvent {
|
||||
ev.rx
|
||||
.recv_timeout(Duration::from_secs(5))
|
||||
.unwrap_or_else(|e| panic!("timed out waiting for {waiting_for}: {e:?}"))
|
||||
}
|
||||
|
||||
/// Assert the subscription is drained: no event is pending.
|
||||
fn assert_quiet(ev: &MembershipEvents) {
|
||||
assert!(matches!(ev.rx.try_recv(), Ok(None)));
|
||||
}
|
||||
|
||||
fn disconnect(name: &str) {
|
||||
assert!(matches!(
|
||||
gen_server::call(
|
||||
MANAGER,
|
||||
Call::Disconnect {
|
||||
name: name.to_string()
|
||||
}
|
||||
),
|
||||
Ok(Reply::Disconnected)
|
||||
));
|
||||
}
|
||||
|
||||
/// Live subscription: an empty snapshot, then `NodeUp` on registration and
|
||||
/// `NodeDown` (same id) on commanded disconnect and on peer EOF alike.
|
||||
#[test]
|
||||
fn subscriber_sees_up_and_down() {
|
||||
run(|| {
|
||||
let mgr = GenServerBuilder::new(Manager::new())
|
||||
.named(MANAGER)
|
||||
.start()
|
||||
.expect("manager name is free");
|
||||
|
||||
let ev = subscribe().expect("manager is up");
|
||||
assert_quiet(&ev); // nothing live: the snapshot is empty
|
||||
|
||||
let t = TcpTransport;
|
||||
let (a1, b1) = pair(&t);
|
||||
let (a2, b2) = pair(&t);
|
||||
spawn_established(FramedConn::new(a1), peer("node-b", 1), Timing::default())
|
||||
.expect("node-b registers");
|
||||
spawn_established(FramedConn::new(a2), peer("node-c", 1), Timing::default())
|
||||
.expect("node-c registers");
|
||||
|
||||
let up_b = match next_event(&ev, "node_up(node-b)") {
|
||||
NodeEvent::NodeUp(info) => {
|
||||
assert_eq!(info.name, "node-b");
|
||||
assert_eq!(info.incarnation, Incarnation::new(1));
|
||||
assert_eq!(info.meta.role, "test");
|
||||
info
|
||||
}
|
||||
other => panic!("expected node_up(node-b), got {other:?}"),
|
||||
};
|
||||
let up_c = match next_event(&ev, "node_up(node-c)") {
|
||||
NodeEvent::NodeUp(info) => {
|
||||
assert_eq!(info.name, "node-c");
|
||||
info
|
||||
}
|
||||
other => panic!("expected node_up(node-c), got {other:?}"),
|
||||
};
|
||||
assert_ne!(up_b.node, up_c.node, "distinct peers get distinct ids");
|
||||
|
||||
// Commanded disconnect: down with node-b's id.
|
||||
disconnect("node-b");
|
||||
assert_eq!(
|
||||
next_event(&ev, "node_down(node-b)"),
|
||||
NodeEvent::NodeDown(up_b.clone())
|
||||
);
|
||||
|
||||
// Peer EOF, no command: down with node-c's id.
|
||||
drop(b2);
|
||||
assert_eq!(
|
||||
next_event(&ev, "node_down(node-c)"),
|
||||
NodeEvent::NodeDown(up_c.clone())
|
||||
);
|
||||
assert_quiet(&ev);
|
||||
|
||||
drop(b1);
|
||||
mgr.shutdown();
|
||||
});
|
||||
}
|
||||
|
||||
/// Snapshot-then-stream: a subscriber arriving after connections established
|
||||
/// receives one `NodeUp` per live peer before anything else, and the view
|
||||
/// call agrees with it.
|
||||
#[test]
|
||||
fn late_subscriber_gets_snapshot_and_view_agrees() {
|
||||
run(|| {
|
||||
let mgr = GenServerBuilder::new(Manager::new())
|
||||
.named(MANAGER)
|
||||
.start()
|
||||
.expect("manager name is free");
|
||||
|
||||
let t = TcpTransport;
|
||||
let (a1, b1) = pair(&t);
|
||||
let (a2, b2) = pair(&t);
|
||||
spawn_established(FramedConn::new(a1), peer("node-b", 1), Timing::default())
|
||||
.expect("node-b registers");
|
||||
spawn_established(FramedConn::new(a2), peer("node-c", 1), Timing::default())
|
||||
.expect("node-c registers");
|
||||
|
||||
let ev = subscribe().expect("manager is up");
|
||||
let mut names = Vec::new();
|
||||
for _ in 0..2 {
|
||||
match next_event(&ev, "a snapshot node_up") {
|
||||
NodeEvent::NodeUp(info) => names.push(info.name),
|
||||
other => panic!("expected a snapshot node_up, got {other:?}"),
|
||||
}
|
||||
}
|
||||
names.sort();
|
||||
assert_eq!(names, ["node-b", "node-c"]);
|
||||
assert_quiet(&ev); // the snapshot is exactly the live set
|
||||
|
||||
let mut v = view().expect("manager is up");
|
||||
v.sort_by(|a, b| a.name.cmp(&b.name));
|
||||
assert_eq!(v.len(), 2);
|
||||
assert_eq!(v[0].name, "node-b");
|
||||
assert_eq!(v[1].name, "node-c");
|
||||
|
||||
disconnect("node-b");
|
||||
disconnect("node-c");
|
||||
drop((b1, b2));
|
||||
// Drain the two downs so the subscription ends quiet.
|
||||
let _ = next_event(&ev, "node_down");
|
||||
let _ = next_event(&ev, "node_down");
|
||||
mgr.shutdown();
|
||||
});
|
||||
}
|
||||
|
||||
/// NodeId identity: a restart (same name, new incarnation) is a NEW id — the
|
||||
/// ghost and its successor are distinguishable — while a reconnect blip (same
|
||||
/// name, same incarnation) keeps its id.
|
||||
#[test]
|
||||
fn restart_gets_new_id_blip_keeps_id() {
|
||||
run(|| {
|
||||
let mgr = GenServerBuilder::new(Manager::new())
|
||||
.named(MANAGER)
|
||||
.start()
|
||||
.expect("manager name is free");
|
||||
let ev = subscribe().expect("manager is up");
|
||||
let t = TcpTransport;
|
||||
|
||||
let id = |e: NodeEvent, what: &str| -> NodeId {
|
||||
match e {
|
||||
NodeEvent::NodeUp(info) => info.node,
|
||||
other => panic!("expected node_up ({what}), got {other:?}"),
|
||||
}
|
||||
};
|
||||
let down_id = |e: NodeEvent, what: &str| -> NodeId {
|
||||
match e {
|
||||
NodeEvent::NodeDown(info) => info.node,
|
||||
other => panic!("expected node_down ({what}), got {other:?}"),
|
||||
}
|
||||
};
|
||||
|
||||
// Up at incarnation 1, then the peer dies (EOF).
|
||||
let (a1, b1) = pair(&t);
|
||||
spawn_established(FramedConn::new(a1), peer("node-b", 1), Timing::default())
|
||||
.expect("registers");
|
||||
let id1 = id(next_event(&ev, "node_up inc 1"), "inc 1");
|
||||
drop(b1);
|
||||
assert_eq!(down_id(next_event(&ev, "node_down inc 1"), "inc 1"), id1);
|
||||
|
||||
// Restart: new incarnation, new id — the ghost's id is not reused.
|
||||
let (a2, b2) = pair(&t);
|
||||
spawn_established(FramedConn::new(a2), peer("node-b", 2), Timing::default())
|
||||
.expect("registers");
|
||||
let id2 = id(next_event(&ev, "node_up inc 2"), "inc 2");
|
||||
assert_ne!(
|
||||
id1, id2,
|
||||
"a restarted node must be distinguishable from its ghost"
|
||||
);
|
||||
|
||||
// Blip: the same incarnation reconnects and keeps its id.
|
||||
disconnect("node-b");
|
||||
assert_eq!(down_id(next_event(&ev, "node_down inc 2"), "inc 2"), id2);
|
||||
let (a3, b3) = pair(&t);
|
||||
spawn_established(FramedConn::new(a3), peer("node-b", 2), Timing::default())
|
||||
.expect("registers");
|
||||
let id3 = id(next_event(&ev, "node_up after blip"), "blip");
|
||||
assert_eq!(
|
||||
id2, id3,
|
||||
"a reconnect at the same incarnation is the same node"
|
||||
);
|
||||
|
||||
disconnect("node-b");
|
||||
let _ = next_event(&ev, "final node_down");
|
||||
drop((b2, b3));
|
||||
mgr.shutdown();
|
||||
});
|
||||
}
|
||||
|
||||
/// A dropped subscriber is pruned on the next emit and never disturbs the
|
||||
/// manager or a live subscriber.
|
||||
#[test]
|
||||
fn dead_subscriber_is_pruned() {
|
||||
run(|| {
|
||||
let mgr = GenServerBuilder::new(Manager::new())
|
||||
.named(MANAGER)
|
||||
.start()
|
||||
.expect("manager name is free");
|
||||
|
||||
let dead = subscribe().expect("manager is up");
|
||||
drop(dead);
|
||||
let live = subscribe().expect("manager is up");
|
||||
|
||||
let t = TcpTransport;
|
||||
let (a1, b1) = pair(&t);
|
||||
spawn_established(FramedConn::new(a1), peer("node-b", 1), Timing::default())
|
||||
.expect("registers");
|
||||
match next_event(&live, "node_up despite a dead co-subscriber") {
|
||||
NodeEvent::NodeUp(info) => assert_eq!(info.name, "node-b"),
|
||||
other => panic!("expected node_up, got {other:?}"),
|
||||
}
|
||||
|
||||
disconnect("node-b");
|
||||
let _ = next_event(&live, "node_down");
|
||||
drop(b1);
|
||||
mgr.shutdown();
|
||||
});
|
||||
}
|
||||
@@ -1,175 +0,0 @@
|
||||
//! RFC 010 c7 — the Phase 2 gate: a 3-node mesh under the subprocess
|
||||
//! harness, repeatable.
|
||||
//!
|
||||
//! Each node process runs the integrated `cluster::start` (manager +
|
||||
//! acceptor + connector + static seeds), subscribes to membership like any
|
||||
//! consumer, and announces protocol-visible facts as lines:
|
||||
//! `LISTENING <addr>`, `MEMBER-UP <name> inc=<n>`, `MEMBER-DOWN <name>`.
|
||||
//! Then it **parks forever** — cross-process teardown is retractable state
|
||||
//! (binding trap), so the parent SIGKILLs via `Node`'s `Drop` and clean exit
|
||||
//! stays the c4 harness's own smoke test.
|
||||
//!
|
||||
//! Ports: nodes bind `:0` and report, so the mesh is built by seeding each
|
||||
//! node with the previously-reported addresses (n1: no seeds; n2: n1;
|
||||
//! n3: n1+n2 — inbound covers the reverse edges). The late-seed test is the
|
||||
//! one exception: the parent pre-reserves a port by binding-and-closing it,
|
||||
//! seeds one node with it, then starts the second node on that exact
|
||||
//! address. In principle another process could steal the port in the gap;
|
||||
//! in practice the window is microseconds on a local runner — accepted, and
|
||||
//! confined to that one test.
|
||||
#![cfg(feature = "cluster")]
|
||||
|
||||
mod common;
|
||||
|
||||
use common::{maybe_child, spawn_node, Node};
|
||||
use smarm::cluster::envelope::NodeMeta;
|
||||
use smarm::cluster::membership::{subscribe, NodeEvent};
|
||||
use smarm::cluster::{start, Config, StaticSeeds, Timing};
|
||||
use std::time::Duration;
|
||||
|
||||
const ROLES: &[(&str, fn())] = &[("node", role_node)];
|
||||
|
||||
/// A mesh node: identity and seeds from env, membership events to stdout,
|
||||
/// park forever (the parent reaps).
|
||||
fn role_node() {
|
||||
let name = std::env::var("SMARM_NODE_NAME").expect("SMARM_NODE_NAME not set");
|
||||
let listen = std::env::var("SMARM_LISTEN_ADDR").unwrap_or_else(|_| "127.0.0.1:0".to_string());
|
||||
// Seeds: comma-separated `name=addr` pairs; empty or unset means none.
|
||||
let seeds: Vec<(String, String)> = std::env::var("SMARM_SEEDS")
|
||||
.unwrap_or_default()
|
||||
.split(',')
|
||||
.filter(|s| !s.is_empty())
|
||||
.map(|s| {
|
||||
let (n, a) = s.split_once('=').expect("seed must be name=addr");
|
||||
(n.to_string(), a.to_string())
|
||||
})
|
||||
.collect();
|
||||
|
||||
smarm::run(move || {
|
||||
let cluster = start(Config {
|
||||
node_name: name,
|
||||
meta: NodeMeta {
|
||||
role: "mesh-test".to_string(),
|
||||
region: "local".to_string(),
|
||||
},
|
||||
listen_addr: listen,
|
||||
strategy: Box::new(StaticSeeds::new(seeds)),
|
||||
timing: Timing::default(),
|
||||
})
|
||||
.expect("listener binds");
|
||||
println!("LISTENING {}", cluster.local_addr());
|
||||
|
||||
let events = subscribe().expect("manager is up");
|
||||
loop {
|
||||
match events.rx.recv() {
|
||||
Ok(NodeEvent::NodeUp(info)) => {
|
||||
println!("MEMBER-UP {} inc={}", info.name, info.incarnation.get());
|
||||
}
|
||||
Ok(NodeEvent::NodeDown(info)) => {
|
||||
println!("MEMBER-DOWN {}", info.name);
|
||||
}
|
||||
Err(_) => break, // manager gone; park below regardless
|
||||
}
|
||||
}
|
||||
loop {
|
||||
smarm::sleep(Duration::from_secs(3600));
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
fn spawn_mesh_node(name: &str, seeds: &str, listen: Option<&str>) -> Node {
|
||||
let mut env: Vec<(&str, &str)> = vec![("SMARM_NODE_NAME", name), ("SMARM_SEEDS", seeds)];
|
||||
if let Some(addr) = listen {
|
||||
env.push(("SMARM_LISTEN_ADDR", addr));
|
||||
}
|
||||
spawn_node("node", &env)
|
||||
}
|
||||
|
||||
/// Wait for `MEMBER-UP <peer> inc=<n>` and return the incarnation.
|
||||
fn wait_member_up(node: &mut Node, peer: &str) -> u32 {
|
||||
let prefix = format!("MEMBER-UP {peer} inc=");
|
||||
let line = node.wait_line(&format!("MEMBER-UP {peer}"), |l| l.starts_with(&prefix));
|
||||
line[prefix.len()..].parse().expect("incarnation parses")
|
||||
}
|
||||
|
||||
fn wait_member_down(node: &mut Node, peer: &str) {
|
||||
let want = format!("MEMBER-DOWN {peer}");
|
||||
node.wait_line(&want, |l| l == want);
|
||||
}
|
||||
|
||||
/// The gate, plus the kill and restart facts, as one mesh's life: three
|
||||
/// nodes form a full mesh (every node sees both others up); killing one
|
||||
/// yields `node_down` at both survivors; its restart under the same name
|
||||
/// arrives as a NEW incarnation — the ghost and its successor are
|
||||
/// distinguishable at every observer.
|
||||
#[test]
|
||||
fn three_node_mesh_forms_then_kill_then_restart_distinguishable() {
|
||||
maybe_child(ROLES);
|
||||
|
||||
let mut n1 = spawn_mesh_node("node-1", "", None);
|
||||
let a1 = n1.wait_listening();
|
||||
let mut n2 = spawn_mesh_node("node-2", &format!("node-1={a1}"), None);
|
||||
let a2 = n2.wait_listening();
|
||||
let mut n3 = spawn_mesh_node("node-3", &format!("node-1={a1},node-2={a2}"), None);
|
||||
let _a3 = n3.wait_listening();
|
||||
|
||||
// Full mesh: each node reports both peers up (dialed or inbound alike).
|
||||
wait_member_up(&mut n1, "node-2");
|
||||
let inc3_at_n1 = wait_member_up(&mut n1, "node-3");
|
||||
wait_member_up(&mut n2, "node-1");
|
||||
let inc3_at_n2 = wait_member_up(&mut n2, "node-3");
|
||||
wait_member_up(&mut n3, "node-1");
|
||||
wait_member_up(&mut n3, "node-2");
|
||||
assert_eq!(
|
||||
inc3_at_n1, inc3_at_n2,
|
||||
"one node, one incarnation, all observers"
|
||||
);
|
||||
|
||||
// Kill node-3 (SIGKILL via Drop): node_down at both survivors.
|
||||
drop(n3);
|
||||
wait_member_down(&mut n1, "node-3");
|
||||
wait_member_down(&mut n2, "node-3");
|
||||
|
||||
// Restart node-3 under the same name: it re-dials its seeds and comes
|
||||
// up everywhere as a new incarnation — never the ghost's.
|
||||
let mut n3b = spawn_mesh_node("node-3", &format!("node-1={a1},node-2={a2}"), None);
|
||||
let _ = n3b.wait_listening();
|
||||
let inc3b_at_n1 = wait_member_up(&mut n1, "node-3");
|
||||
let inc3b_at_n2 = wait_member_up(&mut n2, "node-3");
|
||||
assert_eq!(inc3b_at_n1, inc3b_at_n2);
|
||||
assert_ne!(
|
||||
inc3_at_n1, inc3b_at_n1,
|
||||
"a restarted node must be distinguishable from its ghost"
|
||||
);
|
||||
wait_member_up(&mut n3b, "node-1");
|
||||
wait_member_up(&mut n3b, "node-2");
|
||||
}
|
||||
|
||||
/// A seed that is unreachable at start is not fatal: the connector retries
|
||||
/// on backoff, and when a node finally appears at that address, the mesh
|
||||
/// edge forms.
|
||||
#[test]
|
||||
fn seed_unreachable_at_start_then_arriving_later() {
|
||||
maybe_child(ROLES);
|
||||
|
||||
// Pre-reserve an address by binding and immediately closing it (see the
|
||||
// module docs for the accepted steal window). Dials to it are refused
|
||||
// until node-b starts there.
|
||||
let reserved = {
|
||||
let l = std::net::TcpListener::bind("127.0.0.1:0").expect("bind");
|
||||
l.local_addr().expect("addr").to_string()
|
||||
};
|
||||
|
||||
let mut a = spawn_mesh_node("node-a", &format!("node-b={reserved}"), None);
|
||||
let _ = a.wait_listening();
|
||||
|
||||
// Let a few refused attempts happen before the seed comes up, so the
|
||||
// retry path is what forms the edge (backoff cap 5s < harness WAIT 10s).
|
||||
std::thread::sleep(Duration::from_millis(600));
|
||||
|
||||
let mut b = spawn_mesh_node("node-b", "", Some(&reserved));
|
||||
let _ = b.wait_listening();
|
||||
|
||||
wait_member_up(&mut a, "node-b");
|
||||
wait_member_up(&mut b, "node-a");
|
||||
}
|
||||
@@ -1,359 +0,0 @@
|
||||
//! RFC 010 c12 — remote monitors.
|
||||
//!
|
||||
//! Local suite (`run()`, no network): the immediate answers — no connection
|
||||
//! ⇒ `Disconnected`, dead incarnation ⇒ `NoProc` — and the self-node
|
||||
//! collapse (a plain local monitor underneath, incl. `demonitor_remote`).
|
||||
//!
|
||||
//! Cross-process: a *server* exposes a control name and spawns workers on
|
||||
//! request, replying with each worker's pid (via `RemotePid::from_local`,
|
||||
//! the D12 set-site) or, for the deliberately unshipped one, only its raw
|
||||
//! slot numbers. The *client* monitors them and asserts: kill ⇒ the true
|
||||
//! reason (Exit / Panic); a corpse ⇒ its recorded terminal reason, not
|
||||
//! NoProc; a live pid that never crossed the wire ⇒ NoProc (no liveness
|
||||
//! leak); a demonitor racing the kill ⇒ no notice, proven by stream ORDER
|
||||
//! (a later notice on the same connection arrives while the earlier slot
|
||||
//! is still empty), not by sleeping.
|
||||
#![cfg(feature = "cluster")]
|
||||
|
||||
mod common;
|
||||
|
||||
use common::{maybe_child, spawn_node};
|
||||
use smarm::cluster::envelope::NodeMeta;
|
||||
use smarm::cluster::expose::{expose, expose_type};
|
||||
use smarm::cluster::membership::{subscribe, NodeEvent};
|
||||
use smarm::cluster::remote::{
|
||||
self, demonitor_remote, monitor_remote, send_to_remote, RemoteName, RemotePid,
|
||||
};
|
||||
use smarm::cluster::{start, Config, RemoteDownReason, StaticSeeds, Timing};
|
||||
use smarm::pg::Incarnation;
|
||||
use smarm::{channel, install, register, run, spawn, Addressable, DownReason, Erased, Name, Pid};
|
||||
use std::collections::HashMap;
|
||||
use std::time::Duration;
|
||||
|
||||
// ---- message types (hand-rolled serde; the crate is derive-less) ---------
|
||||
|
||||
#[derive(Debug)]
|
||||
struct Ctl {
|
||||
cmd: String,
|
||||
reply_to: RemotePid<Client>,
|
||||
}
|
||||
#[derive(Debug)]
|
||||
struct Answer {
|
||||
text: String,
|
||||
pid: Option<RemotePid<Erased>>,
|
||||
}
|
||||
struct Client;
|
||||
impl Addressable for Client {
|
||||
type Msg = Answer;
|
||||
}
|
||||
|
||||
impl serde::Serialize for Ctl {
|
||||
fn serialize<S: serde::Serializer>(&self, s: S) -> Result<S::Ok, S::Error> {
|
||||
use serde::ser::SerializeTuple;
|
||||
let mut t = s.serialize_tuple(2)?;
|
||||
t.serialize_element(&self.cmd)?;
|
||||
t.serialize_element(&self.reply_to)?;
|
||||
t.end()
|
||||
}
|
||||
}
|
||||
impl<'de> serde::Deserialize<'de> for Ctl {
|
||||
fn deserialize<D: serde::Deserializer<'de>>(d: D) -> Result<Self, D::Error> {
|
||||
let (cmd, reply_to) = <(String, RemotePid<Client>)>::deserialize(d)?;
|
||||
Ok(Ctl { cmd, reply_to })
|
||||
}
|
||||
}
|
||||
impl serde::Serialize for Answer {
|
||||
fn serialize<S: serde::Serializer>(&self, s: S) -> Result<S::Ok, S::Error> {
|
||||
use serde::ser::SerializeTuple;
|
||||
let mut t = s.serialize_tuple(2)?;
|
||||
t.serialize_element(&self.text)?;
|
||||
t.serialize_element(&self.pid)?;
|
||||
t.end()
|
||||
}
|
||||
}
|
||||
impl<'de> serde::Deserialize<'de> for Answer {
|
||||
fn deserialize<D: serde::Deserializer<'de>>(d: D) -> Result<Self, D::Error> {
|
||||
let (text, pid) = <(String, Option<RemotePid<Erased>>)>::deserialize(d)?;
|
||||
Ok(Answer { text, pid })
|
||||
}
|
||||
}
|
||||
|
||||
// ================= local suite =========================================
|
||||
|
||||
/// No connection to the pid's node: `Disconnected` at once — the remote
|
||||
/// analog of NoProc, and the first thing c11's variant is for.
|
||||
#[test]
|
||||
fn unconnected_node_is_disconnected_immediately() {
|
||||
maybe_child(ROLES);
|
||||
run(|| {
|
||||
remote::set_local_identity("me", Incarnation::new(7));
|
||||
let ghost = RemotePid::<Erased>::from_parts("nowhere", Incarnation::new(1), 3, 1);
|
||||
let m = monitor_remote(ghost.clone());
|
||||
let d = m.recv().unwrap();
|
||||
assert_eq!(d.pid, ghost);
|
||||
assert_eq!(d.reason, RemoteDownReason::Disconnected);
|
||||
});
|
||||
}
|
||||
|
||||
/// The node is connected but the pid names an earlier incarnation: the
|
||||
/// actor is a known corpse (RFC v2 §3), so `NoProc` at once — never
|
||||
/// `Disconnected`, nothing on the wire.
|
||||
#[test]
|
||||
fn dead_incarnation_is_noproc_immediately() {
|
||||
maybe_child(ROLES);
|
||||
run(|| {
|
||||
remote::set_local_identity("me", Incarnation::new(7));
|
||||
let (probe_tx, probe_rx) = channel();
|
||||
remote::bind_outbound_probe("peer", Incarnation::new(5), probe_tx);
|
||||
let stale = RemotePid::<Erased>::from_parts("peer", Incarnation::new(4), 9, 1);
|
||||
let m = monitor_remote(stale);
|
||||
assert_eq!(m.recv().unwrap().reason, DownReason::NoProc.into());
|
||||
assert!(probe_rx.try_recv().unwrap().is_none(), "no frame emitted");
|
||||
});
|
||||
}
|
||||
|
||||
/// A self-node pid collapses to an ordinary local monitor: the true reason
|
||||
/// on exit, and `demonitor_remote` cancels it.
|
||||
#[test]
|
||||
fn self_node_pid_collapses_to_local_monitor() {
|
||||
maybe_child(ROLES);
|
||||
run(|| {
|
||||
remote::set_local_identity("me", Incarnation::new(7));
|
||||
let (go_tx, go_rx) = channel::<()>();
|
||||
let (go2_tx, go2_rx) = channel::<()>();
|
||||
let a = spawn(move || {
|
||||
let _ = go_rx.recv();
|
||||
})
|
||||
.pid();
|
||||
let b = spawn(move || {
|
||||
let _ = go2_rx.recv();
|
||||
})
|
||||
.pid();
|
||||
let ma = monitor_remote(RemotePid::from_local(a).expect("identity set"));
|
||||
let mb = monitor_remote(RemotePid::from_local(b).expect("identity set"));
|
||||
assert_ne!(ma.id, mb.id);
|
||||
assert!(ma.target.local() == Some(a));
|
||||
|
||||
demonitor_remote(&mb);
|
||||
go2_tx.send(()).unwrap();
|
||||
go_tx.send(()).unwrap();
|
||||
let d = ma.recv().unwrap();
|
||||
assert_eq!(d.reason, DownReason::Exit.into());
|
||||
assert_eq!(d.pid.local(), Some(a));
|
||||
// `a` is down (its notice arrived), and `b` was killed first on the
|
||||
// same scheduler — a notice for `b` would be here by now. After a
|
||||
// demonitor the channel is closed-empty (`Err`), like the local one.
|
||||
assert!(matches!(mb.try_recv(), Ok(None) | Err(_)));
|
||||
});
|
||||
}
|
||||
|
||||
// ================= cross-process ======================================
|
||||
|
||||
const ROLES: &[(&str, fn())] = &[("server", role_server), ("client", role_client)];
|
||||
|
||||
const CTL: Name<Ctl> = Name::new("c12.ctl");
|
||||
|
||||
fn cfg(name: &str, seeds: Vec<(String, String)>) -> Config {
|
||||
Config {
|
||||
node_name: name.to_string(),
|
||||
meta: NodeMeta {
|
||||
role: "c12".into(),
|
||||
region: "local".into(),
|
||||
},
|
||||
listen_addr: std::env::var("SMARM_LISTEN_ADDR").unwrap_or_else(|_| "127.0.0.1:0".into()),
|
||||
strategy: Box::new(StaticSeeds::new(seeds)),
|
||||
timing: Timing::default(),
|
||||
}
|
||||
}
|
||||
|
||||
fn wait_up(events: &smarm::cluster::membership::MembershipEvents, who: &str) {
|
||||
loop {
|
||||
match events.rx.recv() {
|
||||
Ok(NodeEvent::NodeUp(i)) if i.name == who => return,
|
||||
Ok(_) => continue,
|
||||
Err(_) => panic!("manager gone"),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Server commands (all answered to `reply_to`):
|
||||
/// - `spawn:exit` / `spawn:panic` — a parked worker; `kill:<index>` releases
|
||||
/// it, whereupon it returns / panics. Answer carries its pid.
|
||||
/// - `spawn:corpse` — a worker that has already exited when the answer is
|
||||
/// sent; the pid was shipped (watchable) before it died.
|
||||
/// - `spawn:unwatched` — a parked worker whose pid is NEVER shipped; the
|
||||
/// answer carries only `text = "slot:<index>:<generation>"`.
|
||||
fn role_server() {
|
||||
smarm::run(move || {
|
||||
let cluster = start(cfg("server", vec![])).expect("binds");
|
||||
println!("LISTENING {}", cluster.local_addr());
|
||||
let (tx, rx) = channel::<Ctl>();
|
||||
register(CTL, tx).unwrap();
|
||||
expose(CTL);
|
||||
println!("READY");
|
||||
let mut workers: HashMap<u32, smarm::channel::Sender<()>> = HashMap::new();
|
||||
loop {
|
||||
let ctl = rx.recv().unwrap();
|
||||
println!("CTL {}", ctl.cmd);
|
||||
let (text, pid): (String, Option<RemotePid<Erased>>) = match ctl.cmd.as_str() {
|
||||
"spawn:exit" | "spawn:panic" => {
|
||||
let panic = ctl.cmd == "spawn:panic";
|
||||
let (go_tx, go_rx) = channel::<()>();
|
||||
let p: Pid = spawn(move || {
|
||||
let _ = go_rx.recv();
|
||||
if panic {
|
||||
panic!("worker asked to panic");
|
||||
}
|
||||
})
|
||||
.pid();
|
||||
workers.insert(p.index(), go_tx);
|
||||
(
|
||||
"ok".into(),
|
||||
Some(RemotePid::from_local(p).expect("identity set")),
|
||||
)
|
||||
}
|
||||
"spawn:corpse" => {
|
||||
let p: Pid = spawn(|| {}).pid();
|
||||
let rp = RemotePid::from_local(p).expect("identity set"); // shipped ⇒ watchable
|
||||
let m = smarm::monitor(p);
|
||||
let _ = m.rx.recv(); // dead before the answer goes out
|
||||
("ok".into(), Some(rp))
|
||||
}
|
||||
"spawn:unwatched" => {
|
||||
let (go_tx, go_rx) = channel::<()>();
|
||||
let p: Pid = spawn(move || {
|
||||
let _ = go_rx.recv();
|
||||
})
|
||||
.pid();
|
||||
workers.insert(p.index(), go_tx);
|
||||
(format!("slot:{}:{}", p.index(), p.generation()), None)
|
||||
}
|
||||
other => {
|
||||
let idx: u32 = other.strip_prefix("kill:").unwrap().parse().unwrap();
|
||||
if let Some(go) = workers.remove(&idx) {
|
||||
let _ = go.send(());
|
||||
}
|
||||
("killed".into(), None)
|
||||
}
|
||||
};
|
||||
send_to_remote(ctl.reply_to, Answer { text, pid }).unwrap();
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
fn role_client() {
|
||||
let server_addr = std::env::var("SMARM_SERVER_ADDR").expect("SMARM_SERVER_ADDR");
|
||||
smarm::run(move || {
|
||||
let _cluster = start(cfg("client", vec![("server".into(), server_addr)])).expect("binds");
|
||||
let ev = subscribe().unwrap();
|
||||
wait_up(&ev, "server");
|
||||
let (tx, rx) = channel::<Answer>();
|
||||
let me: Pid<Client> = install::<Client>(tx);
|
||||
expose_type::<Answer>();
|
||||
let ask = |cmd: &str| -> Answer {
|
||||
remote::send(
|
||||
RemoteName::new("server", CTL),
|
||||
Ctl {
|
||||
cmd: cmd.into(),
|
||||
reply_to: RemotePid::from_local(me).expect("identity set"),
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
rx.recv().unwrap()
|
||||
};
|
||||
let server_inc = ev_incarnation();
|
||||
|
||||
// 1. kill ⇒ true reason (Exit).
|
||||
let a = ask("spawn:exit").pid.unwrap();
|
||||
let ma = monitor_remote(a.clone());
|
||||
ask(&format!("kill:{}", a.index()));
|
||||
let d = ma.recv().unwrap();
|
||||
assert_eq!(d.pid, a);
|
||||
println!("DOWN exit {:?}", d.reason);
|
||||
|
||||
// 2. kill ⇒ true reason (Panic).
|
||||
let b = ask("spawn:panic").pid.unwrap();
|
||||
let mb = monitor_remote(b.clone());
|
||||
ask(&format!("kill:{}", b.index()));
|
||||
println!("DOWN panic {:?}", mb.recv().unwrap().reason);
|
||||
|
||||
// 3. corpse ⇒ recorded terminal reason, not NoProc.
|
||||
let c = ask("spawn:corpse").pid.unwrap();
|
||||
println!("DOWN corpse {:?}", monitor_remote(c).recv().unwrap().reason);
|
||||
|
||||
// 4. live but never shipped/exposed ⇒ NoProc (no leak); a made-up
|
||||
// slot on the same node ⇒ NoProc too, indistinguishably.
|
||||
let ans = ask("spawn:unwatched");
|
||||
let mut it = ans.text.strip_prefix("slot:").unwrap().split(':');
|
||||
let (idx, gen): (u32, u32) = (
|
||||
it.next().unwrap().parse().unwrap(),
|
||||
it.next().unwrap().parse().unwrap(),
|
||||
);
|
||||
let hidden = RemotePid::<Erased>::from_parts("server", server_inc, idx, gen);
|
||||
println!(
|
||||
"DOWN hidden {:?}",
|
||||
monitor_remote(hidden).recv().unwrap().reason
|
||||
);
|
||||
let bogus = RemotePid::<Erased>::from_parts("server", server_inc, 100_000, 1);
|
||||
println!(
|
||||
"DOWN bogus {:?}",
|
||||
monitor_remote(bogus).recv().unwrap().reason
|
||||
);
|
||||
|
||||
// 5. demonitor races the kill: no notice for `d1`, proven by order —
|
||||
// `d2`'s notice (same connection, later) arrives while `d1`'s
|
||||
// slot is still empty.
|
||||
let d1 = ask("spawn:exit").pid.unwrap();
|
||||
let m1 = monitor_remote(d1.clone());
|
||||
demonitor_remote(&m1);
|
||||
ask(&format!("kill:{}", d1.index()));
|
||||
let d2 = ask("spawn:exit").pid.unwrap();
|
||||
let m2 = monitor_remote(d2.clone());
|
||||
ask(&format!("kill:{}", d2.index()));
|
||||
assert_eq!(m2.recv().unwrap().reason, DownReason::Exit.into());
|
||||
// Closed-empty (`Err`) or open-empty (`Ok(None)`) both mean no notice.
|
||||
let stray = matches!(m1.try_recv(), Ok(Some(_)));
|
||||
println!("DEMONITOR stray={stray}");
|
||||
|
||||
println!("CLIENT DONE");
|
||||
loop {
|
||||
smarm::sleep(Duration::from_secs(3600));
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/// The server's incarnation as this node sees it — for building pids by hand.
|
||||
fn ev_incarnation() -> Incarnation {
|
||||
smarm::cluster::membership::view()
|
||||
.expect("manager up")
|
||||
.into_iter()
|
||||
.find(|i| i.name == "server")
|
||||
.map(|i| i.incarnation)
|
||||
.expect("server in view")
|
||||
}
|
||||
|
||||
/// The Phase 4 c12 gate: remote monitors report the true reason, honour
|
||||
/// corpses, leak nothing for unshipped pids, and cancel cleanly.
|
||||
#[test]
|
||||
fn remote_monitors_report_true_reasons() {
|
||||
maybe_child(ROLES);
|
||||
let mut server = spawn_node("server", &[]);
|
||||
let saddr = server.wait_listening();
|
||||
server.wait_line("READY", |l| l == "READY");
|
||||
let mut client = spawn_node("client", &[("SMARM_SERVER_ADDR", &saddr)]);
|
||||
client.wait_line("DOWN exit Local(Exit)", |l| l == "DOWN exit Local(Exit)");
|
||||
client.wait_line("DOWN panic Local(Panic)", |l| {
|
||||
l == "DOWN panic Local(Panic)"
|
||||
});
|
||||
client.wait_line("DOWN corpse Local(Exit)", |l| {
|
||||
l == "DOWN corpse Local(Exit)"
|
||||
});
|
||||
client.wait_line("DOWN hidden Local(NoProc)", |l| {
|
||||
l == "DOWN hidden Local(NoProc)"
|
||||
});
|
||||
client.wait_line("DOWN bogus Local(NoProc)", |l| {
|
||||
l == "DOWN bogus Local(NoProc)"
|
||||
});
|
||||
client.wait_line("DEMONITOR stray=false", |l| l == "DEMONITOR stray=false");
|
||||
client.wait_line("CLIENT DONE", |l| l == "CLIENT DONE");
|
||||
}
|
||||
@@ -1,254 +0,0 @@
|
||||
//! RFC 010 c15 — distributed pg: sync on `NodeUp`, incremental
|
||||
//! `Join`/`Leave`, eager eviction announced, `NodeDown` sweep.
|
||||
//!
|
||||
//! Two nodes. The *origin* joins two local workers to `"pool"` before the
|
||||
//! *observer* connects (so the observer's view comes from `Sync`), exposes a
|
||||
//! `"go"` command inbox and then does exactly what the observer tells it:
|
||||
//! kill one worker, join a third, leave with the second. The observer drives
|
||||
//! that script through the cluster itself and asserts every step from
|
||||
//! `members_all` — never touching the group on its own side, except once to
|
||||
//! prove a mixed local+remote group reads correctly and that `members` stays
|
||||
//! local. `dispatch_any` is exercised both ways: into the origin's worker
|
||||
//! (remote pick, `send_to_remote`) and, once the origin is gone, into the
|
||||
//! observer's own (local pick, `send_to`). Finally the parent SIGKILLs the
|
||||
//! origin: the observer must sweep every remote member on `NodeDown`.
|
||||
#![cfg(feature = "cluster")]
|
||||
|
||||
mod common;
|
||||
|
||||
use common::{maybe_child, spawn_node};
|
||||
use smarm::cluster::envelope::NodeMeta;
|
||||
use smarm::cluster::expose::{expose, expose_type};
|
||||
use smarm::cluster::membership::{subscribe, NodeEvent};
|
||||
use smarm::cluster::remote::{self, RemoteName};
|
||||
use smarm::cluster::{
|
||||
dispatch_any, members_all, pick_any, start, Config, DispatchAnyError, GroupMember, StaticSeeds,
|
||||
Timing,
|
||||
};
|
||||
use smarm::{channel, join, leave, members, register, send_to, spawn_addr, Addressable, Name, Pid};
|
||||
use std::time::{Duration, Instant};
|
||||
|
||||
const GO: Name<u8> = Name::new("go");
|
||||
const POOL: &str = "pool";
|
||||
|
||||
/// A pool worker's message: `"die"` stops it, anything else is printed.
|
||||
#[derive(Debug, PartialEq)]
|
||||
struct Job(String);
|
||||
struct Worker;
|
||||
impl Addressable for Worker {
|
||||
type Msg = Job;
|
||||
}
|
||||
impl serde::Serialize for Job {
|
||||
fn serialize<S: serde::Serializer>(&self, s: S) -> Result<S::Ok, S::Error> {
|
||||
self.0.serialize(s)
|
||||
}
|
||||
}
|
||||
impl<'de> serde::Deserialize<'de> for Job {
|
||||
fn deserialize<D: serde::Deserializer<'de>>(d: D) -> Result<Self, D::Error> {
|
||||
String::deserialize(d).map(Job)
|
||||
}
|
||||
}
|
||||
|
||||
const ROLES: &[(&str, fn())] = &[("origin", role_origin), ("observer", role_observer)];
|
||||
|
||||
fn cfg(name: &str, seeds: Vec<(String, String)>) -> Config {
|
||||
Config {
|
||||
node_name: name.into(),
|
||||
meta: NodeMeta {
|
||||
role: "c15".into(),
|
||||
region: "local".into(),
|
||||
},
|
||||
listen_addr: "127.0.0.1:0".into(),
|
||||
strategy: Box::new(StaticSeeds::new(seeds)),
|
||||
timing: Timing::default(),
|
||||
}
|
||||
}
|
||||
|
||||
/// A pool worker: prints every job it is handed, exits on `"die"`.
|
||||
fn worker() -> Pid<Worker> {
|
||||
spawn_addr::<Worker>(|rx| {
|
||||
while let Ok(Job(s)) = rx.recv() {
|
||||
if s == "die" {
|
||||
return;
|
||||
}
|
||||
println!("JOB {s}");
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
fn role_origin() {
|
||||
smarm::run(|| {
|
||||
let cluster = start(cfg("origin", vec![])).expect("binds");
|
||||
// Remote dispatch lands here only for a type this node accepts.
|
||||
expose_type::<Job>();
|
||||
let w1 = worker();
|
||||
let w2 = worker();
|
||||
assert!(join(POOL, w1));
|
||||
assert!(join(POOL, w2));
|
||||
let (go_tx, go_rx) = channel::<u8>();
|
||||
register(GO, go_tx).unwrap();
|
||||
expose(GO);
|
||||
println!("LISTENING {}", cluster.local_addr());
|
||||
println!("JOINED 2");
|
||||
loop {
|
||||
match go_rx.recv().unwrap() {
|
||||
1 => {
|
||||
send_to(w1, Job("die".into())).unwrap();
|
||||
println!("KILLED w1");
|
||||
}
|
||||
2 => {
|
||||
assert!(leave(POOL, w2));
|
||||
println!("LEFT w2");
|
||||
}
|
||||
3 => {
|
||||
let w3 = worker();
|
||||
assert!(join(POOL, w3));
|
||||
println!("JOINED w3");
|
||||
}
|
||||
n => panic!("unknown command {n}"),
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
fn remote_count(group: &str) -> usize {
|
||||
members_all(group)
|
||||
.iter()
|
||||
.filter(|m| matches!(m, GroupMember::Remote(_)))
|
||||
.count()
|
||||
}
|
||||
|
||||
/// Cooperative poll until `pred`; panics (with the last view) on timeout.
|
||||
fn wait_view(what: &str, group: &str, pred: impl Fn(&[GroupMember]) -> bool) {
|
||||
let deadline = Instant::now() + Duration::from_secs(5);
|
||||
loop {
|
||||
let v = members_all(group);
|
||||
if pred(&v) {
|
||||
return;
|
||||
}
|
||||
assert!(
|
||||
Instant::now() < deadline,
|
||||
"timed out waiting for {what}; view = {v:?}"
|
||||
);
|
||||
smarm::sleep(Duration::from_millis(5));
|
||||
}
|
||||
}
|
||||
|
||||
fn role_observer() {
|
||||
let origin_addr = std::env::var("SMARM_ORIGIN_ADDR").expect("SMARM_ORIGIN_ADDR");
|
||||
smarm::run(move || {
|
||||
let _cluster = start(cfg("observer", vec![("origin".into(), origin_addr)])).expect("binds");
|
||||
let ev = subscribe().unwrap();
|
||||
loop {
|
||||
match ev.rx.recv().unwrap() {
|
||||
NodeEvent::NodeUp(i) if i.name == "origin" => break,
|
||||
_ => {}
|
||||
}
|
||||
}
|
||||
let go = |n: u8| remote::send(RemoteName::new("origin", GO), n).unwrap();
|
||||
|
||||
// Sync: both pre-existing members arrive with no join on this side.
|
||||
wait_view("sync of 2 remote members", POOL, |v| {
|
||||
v.len() == 2 && v.iter().all(|m| matches!(m, GroupMember::Remote(_)))
|
||||
});
|
||||
let synced = members_all(POOL);
|
||||
assert!(synced.iter().all(|m| match m {
|
||||
GroupMember::Remote(p) => p.node() == "origin",
|
||||
GroupMember::Local(_) => false,
|
||||
}));
|
||||
println!("SEES 2");
|
||||
|
||||
// Origin-side death: the origin's reaper announces the leave.
|
||||
go(1);
|
||||
wait_view("death evicted on observer", POOL, |v| v.len() == 1);
|
||||
println!("SEES 1 after death");
|
||||
|
||||
// Incremental Join.
|
||||
go(3);
|
||||
wait_view("incremental join", POOL, |v| v.len() == 2);
|
||||
println!("SEES 2 after join");
|
||||
|
||||
// Voluntary Leave.
|
||||
go(2);
|
||||
wait_view("incremental leave", POOL, |v| v.len() == 1);
|
||||
println!("SEES 1 after leave");
|
||||
|
||||
// Mixed group: our own member sits beside the remote one in
|
||||
// `members_all`; `members` stays local-only.
|
||||
let me = worker();
|
||||
assert!(join(POOL, me));
|
||||
wait_view("mixed local+remote", POOL, |v| {
|
||||
v.len() == 2 && v.contains(&GroupMember::Local(me.erase()))
|
||||
});
|
||||
assert_eq!(
|
||||
members(POOL),
|
||||
vec![me.erase()],
|
||||
"local API never shows remotes"
|
||||
);
|
||||
assert_eq!(remote_count(POOL), 1);
|
||||
println!("MIXED ok");
|
||||
|
||||
// dispatch_any: the store's first entry is the origin's w3 (it was
|
||||
// announced before we joined), so the pick is remote and the job
|
||||
// crosses the wire — the origin's worker prints it.
|
||||
let picked = pick_any(POOL).expect("pool has members");
|
||||
assert!(
|
||||
matches!(picked, GroupMember::Remote(_)),
|
||||
"first entry is remote: {picked:?}"
|
||||
);
|
||||
let reached = dispatch_any::<Worker>(POOL, Job("from-observer".into())).unwrap();
|
||||
assert_eq!(reached, picked);
|
||||
println!("DISPATCHED remote");
|
||||
|
||||
println!("PARK");
|
||||
// Parent SIGKILLs the origin now: NodeDown must sweep its member,
|
||||
// ours must survive.
|
||||
wait_view("node_down sweep", POOL, |v| {
|
||||
v == [GroupMember::Local(me.erase())]
|
||||
});
|
||||
assert_eq!(members(POOL), vec![me.erase()]);
|
||||
println!("SWEPT");
|
||||
|
||||
// Now the only member is ours: a local pick, a local send.
|
||||
let reached = dispatch_any::<Worker>(POOL, Job("local".into())).unwrap();
|
||||
assert_eq!(reached, GroupMember::Local(me.erase()));
|
||||
// And an empty group hands the message back.
|
||||
match dispatch_any::<Worker>("nobody", Job("lost".into())) {
|
||||
Err(DispatchAnyError::NoMember(Job(s))) => assert_eq!(s, "lost"),
|
||||
other => panic!("expected NoMember, got {other:?}"),
|
||||
}
|
||||
println!("DISPATCHED local");
|
||||
loop {
|
||||
smarm::sleep(Duration::from_secs(3600));
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/// The Phase 5 gate: sync, join, leave, death, node_down — all observed from
|
||||
/// the peer, none of them a group operation on the peer — plus dispatch_any
|
||||
/// reaching a remote member and a local one.
|
||||
#[test]
|
||||
fn groups_span_two_nodes() {
|
||||
maybe_child(ROLES);
|
||||
let mut origin = spawn_node("origin", &[]);
|
||||
let addr = origin.wait_listening();
|
||||
origin.wait_line("JOINED 2", |l| l == "JOINED 2");
|
||||
let mut observer = spawn_node("observer", &[("SMARM_ORIGIN_ADDR", &addr)]);
|
||||
observer.wait_line("SEES 2", |l| l == "SEES 2");
|
||||
origin.wait_line("KILLED w1", |l| l == "KILLED w1");
|
||||
observer.wait_line("SEES 1 after death", |l| l == "SEES 1 after death");
|
||||
origin.wait_line("JOINED w3", |l| l == "JOINED w3");
|
||||
observer.wait_line("SEES 2 after join", |l| l == "SEES 2 after join");
|
||||
origin.wait_line("LEFT w2", |l| l == "LEFT w2");
|
||||
observer.wait_line("SEES 1 after leave", |l| l == "SEES 1 after leave");
|
||||
observer.wait_line("MIXED ok", |l| l == "MIXED ok");
|
||||
observer.wait_line("DISPATCHED remote", |l| l == "DISPATCHED remote");
|
||||
origin.wait_line("JOB from-observer", |l| l == "JOB from-observer");
|
||||
observer.wait_line("PARK", |l| l == "PARK");
|
||||
origin.kill();
|
||||
observer.wait_line("SWEPT", |l| l == "SWEPT");
|
||||
// Order between the root's line and the worker's is scheduling; wait
|
||||
// for the later one to be certain both happened.
|
||||
observer.wait_line("DISPATCHED local", |l| l == "DISPATCHED local");
|
||||
observer.wait_line("JOB local", |l| l == "JOB local");
|
||||
}
|
||||
@@ -1,356 +0,0 @@
|
||||
//! RFC 010 c10 — pid targeting + auto-serialization. The Phase 3 gate:
|
||||
//! cross-node call/reply with no ceremony, under the subprocess harness.
|
||||
//!
|
||||
//! Local suite (`run()`, no network): serialize/deserialize shapes,
|
||||
//! self-collapse, the outside-runtime contract, the local send-site
|
||||
//! incarnation check with a probe proving **no frame is emitted**.
|
||||
//!
|
||||
//! Cross-process: two nodes. The *server* exposes a `Name<Req>`; the
|
||||
//! *client* sends a `Req` carrying its own `Pid<Reply>` (auto-serialized to
|
||||
//! a `RemotePid` on the wire); the server replies via `send_to_remote`
|
||||
//! straight back to that pid — no name at the client end, no ceremony. A
|
||||
//! third-node roundtrip: the client's pid travels client→server→relay→
|
||||
//! server→client, and still delivers.
|
||||
#![cfg(feature = "cluster")]
|
||||
|
||||
mod common;
|
||||
|
||||
use common::{maybe_child, spawn_node};
|
||||
use smarm::cluster::envelope::{encode_payload, Frame, NodeMeta};
|
||||
use smarm::cluster::expose::{expose, type_hash};
|
||||
use smarm::cluster::membership::{subscribe, NodeEvent};
|
||||
use smarm::cluster::remote::{self, send_to_remote, RemoteName, RemotePid, ToRemoteError};
|
||||
use smarm::cluster::{start, Config, StaticSeeds, Timing};
|
||||
use smarm::pg::Incarnation;
|
||||
use smarm::{channel, install, register, run, Addressable, Name, Pid};
|
||||
use std::time::Duration;
|
||||
|
||||
// ---- message types (std-only payloads; the crate's serde is derive-less,
|
||||
// so wire types are hand-rolled with serde's tuple/seq API via `serde::ser`
|
||||
// impls below — the same thing a user's derive would generate) ------------
|
||||
|
||||
/// A request carrying a reply-to. Serialize/Deserialize are written by hand
|
||||
/// here for exactly one reason: this crate deliberately does not pull in
|
||||
/// serde-derive. Field 1 is the auto-serializing pid.
|
||||
#[derive(Debug, PartialEq)]
|
||||
struct Req {
|
||||
text: String,
|
||||
reply_to: RemotePid<Replier>,
|
||||
}
|
||||
|
||||
#[derive(Debug, PartialEq)]
|
||||
struct Reply(String);
|
||||
|
||||
struct Replier;
|
||||
impl Addressable for Replier {
|
||||
type Msg = Reply;
|
||||
}
|
||||
|
||||
impl serde::Serialize for Req {
|
||||
fn serialize<S: serde::Serializer>(&self, s: S) -> Result<S::Ok, S::Error> {
|
||||
use serde::ser::SerializeTuple;
|
||||
let mut t = s.serialize_tuple(2)?;
|
||||
t.serialize_element(&self.text)?;
|
||||
t.serialize_element(&self.reply_to)?;
|
||||
t.end()
|
||||
}
|
||||
}
|
||||
impl<'de> serde::Deserialize<'de> for Req {
|
||||
fn deserialize<D: serde::Deserializer<'de>>(d: D) -> Result<Self, D::Error> {
|
||||
let (text, reply_to) = <(String, RemotePid<Replier>)>::deserialize(d)?;
|
||||
Ok(Req { text, reply_to })
|
||||
}
|
||||
}
|
||||
impl serde::Serialize for Reply {
|
||||
fn serialize<S: serde::Serializer>(&self, s: S) -> Result<S::Ok, S::Error> {
|
||||
self.0.serialize(s)
|
||||
}
|
||||
}
|
||||
impl<'de> serde::Deserialize<'de> for Reply {
|
||||
fn deserialize<D: serde::Deserializer<'de>>(d: D) -> Result<Self, D::Error> {
|
||||
String::deserialize(d).map(Reply)
|
||||
}
|
||||
}
|
||||
|
||||
// ================= local suite =========================================
|
||||
|
||||
/// A local `Pid<A>` serializes as a `RemotePid<A>` stamped with this node's
|
||||
/// identity; deserializing it back on the same node collapses to the same
|
||||
/// local pid (`local()` is `Some`, `Pid` round-trips).
|
||||
#[test]
|
||||
fn local_pid_serializes_and_collapses_on_self() {
|
||||
maybe_child(ROLES);
|
||||
run(|| {
|
||||
// The local identity is set by cluster::start; the local suite sets
|
||||
// it directly.
|
||||
remote::set_local_identity("me", Incarnation::new(7));
|
||||
let (tx, _rx) = channel::<Reply>();
|
||||
let me: Pid<Replier> = install::<Replier>(tx);
|
||||
|
||||
let bytes = encode_payload(&me).unwrap();
|
||||
let rp: RemotePid<Replier> = smarm::cluster::envelope::decode_payload(&bytes).unwrap();
|
||||
assert_eq!(rp.node(), "me");
|
||||
assert_eq!(rp.incarnation(), Incarnation::new(7));
|
||||
assert_eq!(
|
||||
rp.local(),
|
||||
Some(me),
|
||||
"self-node pid collapses to the local pid"
|
||||
);
|
||||
|
||||
// Deserializing straight into Pid<A> works for a self-node pid...
|
||||
let back: Pid<Replier> = smarm::cluster::envelope::decode_payload(&bytes).unwrap();
|
||||
assert_eq!(back, me);
|
||||
|
||||
// ...and FAILS for a foreign one (collapse is literal: node == self).
|
||||
let foreign = RemotePid::<Replier>::from_parts("elsewhere", Incarnation::new(1), 3, 1);
|
||||
let fbytes = encode_payload(&foreign).unwrap();
|
||||
assert!(smarm::cluster::envelope::decode_payload::<Pid<Replier>>(&fbytes).is_err());
|
||||
assert_eq!(foreign.local(), None);
|
||||
});
|
||||
}
|
||||
|
||||
/// `send_to_remote` short-circuits locally for a self-node pid — the
|
||||
/// zero-copy-equivalent collapse: the message object itself lands in the
|
||||
/// local channel, no encode, no frame.
|
||||
#[test]
|
||||
fn send_to_remote_collapses_locally_for_self() {
|
||||
maybe_child(ROLES);
|
||||
run(|| {
|
||||
remote::set_local_identity("me", Incarnation::new(7));
|
||||
let (tx, rx) = channel::<Reply>();
|
||||
let me: Pid<Replier> = install::<Replier>(tx);
|
||||
let rp = RemotePid::from_local(me).expect("identity set");
|
||||
// Probe the outbound path: nothing must be handed to any connection.
|
||||
let (probe_tx, probe_rx) = channel::<Frame>();
|
||||
remote::bind_outbound_probe("me", Incarnation::new(7), probe_tx);
|
||||
|
||||
send_to_remote(rp, Reply("hi".into())).unwrap();
|
||||
assert_eq!(rx.recv().unwrap(), Reply("hi".into()));
|
||||
assert!(
|
||||
matches!(probe_rx.try_recv(), Ok(None)),
|
||||
"no frame for a local collapse"
|
||||
);
|
||||
});
|
||||
}
|
||||
|
||||
/// RFC v2 §3: a `RemotePid` whose incarnation is not the current one for its
|
||||
/// node fails at the local send site with `DeadIncarnation`, and NO frame
|
||||
/// is emitted — asserted on a probe sender bound as that node's outbound.
|
||||
#[test]
|
||||
fn stale_incarnation_rejected_locally_no_frame() {
|
||||
maybe_child(ROLES);
|
||||
run(|| {
|
||||
remote::set_local_identity("me", Incarnation::new(7));
|
||||
let (probe_tx, probe_rx) = channel::<Frame>();
|
||||
remote::bind_outbound_probe("peer", Incarnation::new(5), probe_tx);
|
||||
|
||||
let stale = RemotePid::<Replier>::from_parts("peer", Incarnation::new(4), 9, 1);
|
||||
match send_to_remote(stale, Reply("late".into())) {
|
||||
Err(ToRemoteError::DeadIncarnation(Reply(s))) => assert_eq!(s, "late"),
|
||||
other => panic!("expected DeadIncarnation, got {other:?}"),
|
||||
}
|
||||
assert!(
|
||||
matches!(probe_rx.try_recv(), Ok(None)),
|
||||
"stale pid must emit no frame"
|
||||
);
|
||||
|
||||
// The current incarnation goes through: a Send frame with the pid's
|
||||
// (index, generation) and Reply's hash lands on the probe.
|
||||
let live = RemotePid::<Replier>::from_parts("peer", Incarnation::new(5), 9, 1);
|
||||
send_to_remote(live, Reply("now".into())).unwrap();
|
||||
match probe_rx.recv().unwrap() {
|
||||
Frame::Send {
|
||||
index,
|
||||
generation,
|
||||
type_hash: h,
|
||||
payload,
|
||||
} => {
|
||||
assert_eq!((index, generation), (9, 1));
|
||||
assert_eq!(h, type_hash::<Reply>());
|
||||
let r: Reply = smarm::cluster::envelope::decode_payload(&payload).unwrap();
|
||||
assert_eq!(r, Reply("now".into()));
|
||||
}
|
||||
f => panic!("expected Send, got {f:?}"),
|
||||
}
|
||||
|
||||
// Unknown node: NotConnected, no frame anywhere.
|
||||
let nowhere = RemotePid::<Replier>::from_parts("nowhere", Incarnation::new(1), 1, 1);
|
||||
assert!(matches!(
|
||||
send_to_remote(nowhere, Reply("x".into())),
|
||||
Err(ToRemoteError::NotConnected(_))
|
||||
));
|
||||
});
|
||||
}
|
||||
|
||||
// ================= cross-process gate ==================================
|
||||
|
||||
const ROLES: &[(&str, fn())] = &[
|
||||
("server", role_server),
|
||||
("client", role_client),
|
||||
("relay", role_relay),
|
||||
];
|
||||
|
||||
const ECHO: Name<Req> = Name::new("c10.echo");
|
||||
const RELAY: Name<Req> = Name::new("c10.relay");
|
||||
|
||||
fn cfg(name: &str, seeds: Vec<(String, String)>) -> Config {
|
||||
Config {
|
||||
node_name: name.to_string(),
|
||||
meta: NodeMeta {
|
||||
role: "c10".into(),
|
||||
region: "local".into(),
|
||||
},
|
||||
listen_addr: std::env::var("SMARM_LISTEN_ADDR").unwrap_or_else(|_| "127.0.0.1:0".into()),
|
||||
strategy: Box::new(StaticSeeds::new(seeds)),
|
||||
timing: Timing::default(),
|
||||
}
|
||||
}
|
||||
|
||||
fn wait_up(events: &smarm::cluster::membership::MembershipEvents, who: &str) {
|
||||
loop {
|
||||
match events.rx.recv() {
|
||||
Ok(NodeEvent::NodeUp(i)) if i.name == who => return,
|
||||
Ok(_) => continue,
|
||||
Err(_) => panic!("manager gone"),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Server: exposes ECHO; each Req is answered by `send_to_remote` to its
|
||||
/// reply_to — the server never learns a name for the client. If the Req text
|
||||
/// starts with "via-relay:", it forwards the whole Req (reply_to and all) to
|
||||
/// the relay node instead, which sends it back here; the second arrival is
|
||||
/// answered normally. That is the pid's third-node roundtrip.
|
||||
fn role_server() {
|
||||
let relay_addr = std::env::var("SMARM_RELAY_ADDR").ok();
|
||||
smarm::run(move || {
|
||||
let seeds = relay_addr
|
||||
.map(|a| vec![("relay".to_string(), a)])
|
||||
.unwrap_or_default();
|
||||
let cluster = start(cfg("server", seeds)).expect("binds");
|
||||
println!("LISTENING {}", cluster.local_addr());
|
||||
let (tx, rx) = channel::<Req>();
|
||||
register(ECHO, tx).unwrap();
|
||||
expose(ECHO);
|
||||
println!("READY");
|
||||
loop {
|
||||
let req = rx.recv().unwrap();
|
||||
if let Some(rest) = req.text.strip_prefix("via-relay:") {
|
||||
let fwd = Req {
|
||||
text: format!("relayed:{rest}"),
|
||||
reply_to: req.reply_to,
|
||||
};
|
||||
remote::send(RemoteName::new("relay", RELAY), fwd).unwrap();
|
||||
println!("FORWARDED");
|
||||
continue;
|
||||
}
|
||||
println!("REQ {}", req.text);
|
||||
send_to_remote(req.reply_to, Reply(format!("echo:{}", req.text))).unwrap();
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/// Relay: exposes RELAY; bounces every Req straight back to the server's
|
||||
/// ECHO, untouched. The client's pid inside it now crosses relay→server.
|
||||
fn role_relay() {
|
||||
let server_addr = std::env::var("SMARM_SERVER_ADDR").expect("SMARM_SERVER_ADDR");
|
||||
smarm::run(move || {
|
||||
let cluster = start(cfg("relay", vec![("server".into(), server_addr)])).expect("binds");
|
||||
println!("LISTENING {}", cluster.local_addr());
|
||||
let (tx, rx) = channel::<Req>();
|
||||
register(RELAY, tx).unwrap();
|
||||
expose(RELAY);
|
||||
let ev = subscribe().unwrap();
|
||||
wait_up(&ev, "server");
|
||||
println!("READY");
|
||||
loop {
|
||||
let req = rx.recv().unwrap();
|
||||
println!("RELAYING {}", req.text);
|
||||
remote::send(RemoteName::new("server", ECHO), req).unwrap();
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/// Client: connects to server, installs a Reply inbox on its own pid,
|
||||
/// declares it accepts `Reply` (`expose_type` — the RFC's one kept piece of
|
||||
/// ceremony: nothing is remotely deliverable by default), sends a Req with
|
||||
/// `reply_to = my pid` (auto-serialized), awaits the reply.
|
||||
fn role_client() {
|
||||
let server_addr = std::env::var("SMARM_SERVER_ADDR").expect("SMARM_SERVER_ADDR");
|
||||
let via_relay = std::env::var("SMARM_VIA_RELAY").is_ok();
|
||||
smarm::run(move || {
|
||||
let _cluster = start(cfg("client", vec![("server".into(), server_addr)])).expect("binds");
|
||||
let ev = subscribe().unwrap();
|
||||
wait_up(&ev, "server");
|
||||
println!("MEMBER-UP server");
|
||||
|
||||
let (tx, rx) = channel::<Reply>();
|
||||
let me: Pid<Replier> = install::<Replier>(tx);
|
||||
// The one deliberate line: a pid-targeted inbound is deliverable only
|
||||
// for types this node has said it accepts (RFC §4, the safety).
|
||||
smarm::cluster::expose::expose_type::<Reply>();
|
||||
let text = if via_relay { "via-relay:ping" } else { "ping" };
|
||||
remote::send(
|
||||
RemoteName::new("server", ECHO),
|
||||
Req {
|
||||
text: text.into(),
|
||||
reply_to: RemotePid::from_local(me).expect("identity set"),
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
println!("SENT");
|
||||
let Reply(s) = rx.recv().unwrap();
|
||||
println!("REPLY {s}");
|
||||
loop {
|
||||
smarm::sleep(Duration::from_secs(3600));
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/// The gate: cross-node call/reply with no ceremony.
|
||||
#[test]
|
||||
fn cross_node_call_reply_no_ceremony() {
|
||||
maybe_child(ROLES);
|
||||
let mut server = spawn_node("server", &[]);
|
||||
let saddr = server.wait_listening();
|
||||
server.wait_line("READY", |l| l == "READY");
|
||||
let mut client = spawn_node("client", &[("SMARM_SERVER_ADDR", &saddr)]);
|
||||
client.wait_line("SENT", |l| l == "SENT");
|
||||
server.wait_line("REQ ping", |l| l == "REQ ping");
|
||||
client.wait_line("REPLY echo:ping", |l| l == "REPLY echo:ping");
|
||||
}
|
||||
|
||||
/// The client's pid, round-tripped through a third node, still delivers.
|
||||
#[test]
|
||||
fn pid_roundtrips_through_third_node() {
|
||||
maybe_child(ROLES);
|
||||
// Relay needs the server address; server needs the relay address —
|
||||
// pre-reserve the relay port (same accepted micro-window as cluster_mesh).
|
||||
let relay_addr = {
|
||||
let l = std::net::TcpListener::bind("127.0.0.1:0").unwrap();
|
||||
l.local_addr().unwrap().to_string()
|
||||
};
|
||||
let mut server = spawn_node("server", &[("SMARM_RELAY_ADDR", &relay_addr)]);
|
||||
let saddr = server.wait_listening();
|
||||
server.wait_line("READY", |l| l == "READY");
|
||||
let mut relay = spawn_node(
|
||||
"relay",
|
||||
&[
|
||||
("SMARM_SERVER_ADDR", &saddr),
|
||||
("SMARM_LISTEN_ADDR", &relay_addr),
|
||||
],
|
||||
);
|
||||
let _ = relay.wait_listening();
|
||||
relay.wait_line("READY", |l| l == "READY");
|
||||
let mut client = spawn_node(
|
||||
"client",
|
||||
&[("SMARM_SERVER_ADDR", &saddr), ("SMARM_VIA_RELAY", "1")],
|
||||
);
|
||||
client.wait_line("SENT", |l| l == "SENT");
|
||||
server.wait_line("FORWARDED", |l| l == "FORWARDED");
|
||||
relay.wait_line("RELAYING", |l| l.starts_with("RELAYING"));
|
||||
server.wait_line("REQ relayed:ping", |l| l == "REQ relayed:ping");
|
||||
client.wait_line("REPLY echo:relayed:ping", |l| {
|
||||
l == "REPLY echo:relayed:ping"
|
||||
});
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user