From 7a42d49746c9e2a2ea6107499b41930c101fb9bb Mon Sep 17 00:00:00 2001 From: smarm-agent Date: Sun, 14 Jun 2026 18:56:16 +0000 Subject: [PATCH] bench: add switch_cost local-mode per-switch microbench (rdtsc+wall, RFC per-switch spike) --- Cargo.toml | 4 + benches/switch_cost.rs | 256 +++++++++++++++++++++++++++++++++++++++++ 2 files changed, 260 insertions(+) create mode 100644 benches/switch_cost.rs diff --git a/Cargo.toml b/Cargo.toml index acca772..62f1d80 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -61,3 +61,7 @@ harness = false [[bench]] name = "rq_runtime" harness = false + +[[bench]] +name = "switch_cost" +harness = false diff --git a/benches/switch_cost.rs b/benches/switch_cost.rs new file mode 100644 index 0000000..c4ff508 --- /dev/null +++ b/benches/switch_cost.rs @@ -0,0 +1,256 @@ +//! Per-switch (context-switch) cost microbench — the profiling-spike harness +//! for the ROADMAP "Per-switch cost (context shims, epoch protocol)" item. +//! +//! The spin work (RFC 004) is closed; the next perf target is the per-switch +//! cost itself. Shootout evidence: per-wake latency is ~0.16–0.18µs at N=1 but +//! ~0.8–1.2µs at N=8+, and the residual is attributed to the context-switch +//! shims (`src/context.rs`) and the epoch protocol — NOT the queue. This binary +//! isolates that round-trip so the cycles can be attributed under `perf` and an +//! rdtsc bracket, feeding the RFC. +//! +//! WHAT THE ROUND-TRIP IS +//! +//! `yield_now()` from inside an actor does exactly one park/unpark round-trip +//! with nothing else attached: +//! +//! actor: switch_to_scheduler ──► scheduler re-queues the actor (slot-word +//! (context.rs shim) epoch/state transition, run_queue push), +//! pops it straight back, switch_to_actor +//! actor resumes ◄────────────────────────────────────────────────────── +//! +//! No IO thread traffic, no channel, no timer, no cross-thread wake. On a +//! single-scheduler runtime the re-queue+repop never leaves this core, so the +//! sample is the *pure* shim + epoch + queue-op cost with zero coherency +//! traffic. That is the `local` baseline; a `remote` mode (wake straddling two +//! schedulers, to expose the N=1→N=8 coherency/TLS-mode jump) is a deliberate +//! follow-up and is NOT in this file yet — local first, per the spike plan. +//! +//! TWO LENSES ON THE SAME LOOP +//! +//! wall — `Instant` bracket per round-trip. Source of truth for µs, directly +//! comparable to the shootout's per-wake latency numbers. +//! cycles — `rdtsc` bracket per round-trip. Source of truth for the cycle +//! budget the RFC will reason in (the spin budget is in cycles too). +//! +//! Reporting both lets us *derive* the effective TSC frequency (cycles/ns) from +//! the same samples instead of hardcoding a nominal 3.7GHz — the spin_sweep +//! lesson was that nominal-vs-actual TSC drift is exactly what produces +//! red-herring numbers. If the derived freq matches the box's known base clock, +//! the two lenses corroborate; if not, that mismatch is itself a finding. +//! +//! Knobs (env): +//! SMARM_SWITCH_ROUNDS round-trips timed per run default 200000 +//! SMARM_SWITCH_WARMUP untimed warmup round-trips default 10000 +//! SMARM_SWITCH_RUNS runs (pooled latency, median) default 5 +//! +//! Output: house table + one greppable line per run-set: +//! SWITCHCSV,,,,,,,,, +//! ,,,, +//! +//! NOTE: a single yielding actor is the cooperative-scheduling tightest loop — +//! it never parks on a futex (the work is always immediately re-queued), so +//! this measures the switch+epoch+queue path, NOT the futex park. That is +//! intentional: the futex park is the spin work's territory (RFC 004), already +//! characterised. The unattributed constant the shootout flagged lives in the +//! switch itself, which is what this loop hammers. + +use smarm::runtime::{init, Config}; +use smarm::{run, spawn, yield_now}; +use std::sync::{Arc, Mutex}; + +// -------------------------------------------------------------------------- +// env helpers (house style, matching spin_sweep.rs / rq_runtime.rs) +// -------------------------------------------------------------------------- + +fn variant() -> &'static str { + if cfg!(feature = "rq-mpmc") { + "rq-mpmc" + } else if cfg!(feature = "rq-striped") { + "rq-striped" + } else { + "rq-mutex" + } +} + +fn env_usize(key: &str, default: usize) -> usize { + std::env::var(key).ok().and_then(|v| v.parse().ok()).unwrap_or(default) +} + +// -------------------------------------------------------------------------- +// rdtsc — serialised so the bracket actually fences the round-trip. +// +// Plain `rdtsc` can be reordered around the work by an out-of-order core, which +// would smear the bracket. `rdtscp` retires prior instructions before reading +// the counter, and the trailing `lfence` blocks later instructions from +// climbing above the second read. Pair = (rdtscp; lfence) … work … (rdtscp; +// lfence): a standard cycle-accurate bracket. We read TSC_AUX too but ignore +// it; the point is the ordering guarantee, not the core id. +// -------------------------------------------------------------------------- + +#[inline(always)] +fn rdtsc_serialised() -> u64 { + #[cfg(target_arch = "x86_64")] + unsafe { + let mut aux = 0u32; + let t = core::arch::x86_64::__rdtscp(&mut aux); + core::arch::x86_64::_mm_lfence(); + t + } + #[cfg(not(target_arch = "x86_64"))] + { + // Non-x86 fallback: nanosecond clock standing in for cycles. The derived + // "GHz" column then reads ~1.0 and is meaningless, but the wall lens and + // the harness still work. The spike target box is x86-64. + use std::time::Instant; + thread_local! { static T0: Instant = Instant::now(); } + T0.with(|t0| t0.elapsed().as_nanos() as u64) + } +} + +// -------------------------------------------------------------------------- +// percentile / median helpers (verbatim house idiom from spin_sweep.rs) +// -------------------------------------------------------------------------- + +/// Nearest-rank percentile over an already-sorted slice. `p` in [0, 100]. +fn pct(sorted: &[u64], p: f64) -> u64 { + if sorted.is_empty() { + return 0; + } + let idx = ((p / 100.0) * (sorted.len() - 1) as f64).round() as usize; + sorted[idx.min(sorted.len() - 1)] +} + +// -------------------------------------------------------------------------- +// one run: a single actor yields ROUNDS times; we bracket each yield from +// inside the actor (the only vantage point — the actor is suspended during the +// scheduler half, so an external timer can't see a single round-trip). +// +// Per iteration we capture BOTH a wall-ns delta and a TSC-cycle delta around +// the same `yield_now()`. The loop overhead (two clock reads + a Vec push + +// the branch) rides along in every sample equally; we subtract an empty-loop +// self-calibration below so the reported number is the round-trip, not the +// instrumentation. +// -------------------------------------------------------------------------- + +struct RunSample { + lat_ns: Vec, + cyc: Vec, +} + +fn one_run(threads: usize, rounds: usize, warmup: usize) -> RunSample { + let out: Arc>> = Arc::new(Mutex::new(None)); + let out2 = out.clone(); + + let cfg = Config::exact(threads); + init(cfg); + + run(move || { + let h = spawn(move || { + // Warmup: let the actor's stack/queue slot go hot, JIT-free but + // cache-warm, before any sample is kept. + for _ in 0..warmup { + yield_now(); + } + + let mut lat_ns = Vec::with_capacity(rounds); + let mut cyc = Vec::with_capacity(rounds); + + for _ in 0..rounds { + let w0 = std::time::Instant::now(); + let c0 = rdtsc_serialised(); + yield_now(); + let c1 = rdtsc_serialised(); + let w1 = w0.elapsed(); + cyc.push(c1.saturating_sub(c0)); + lat_ns.push(w1.as_nanos() as u64); + } + + *out2.lock().unwrap() = Some(RunSample { lat_ns, cyc }); + }); + let _ = h.join(); + }); + + let sample = out.lock().unwrap().take().expect("actor stored a sample"); + sample +} + +/// Empty-loop self-calibration: the same bracket with the `yield_now()` removed, +/// run inline (no runtime). Gives the floor cost of two serialised clock reads + +/// the push, in both lenses, to subtract from the round-trip samples. +fn calibrate(rounds: usize) -> (u64, u64) { + let mut lat_ns = Vec::with_capacity(rounds); + let mut cyc = Vec::with_capacity(rounds); + let mut sink = 0u64; + for _ in 0..rounds { + let w0 = std::time::Instant::now(); + let c0 = rdtsc_serialised(); + // no yield — measure the bracket itself + let c1 = rdtsc_serialised(); + let w1 = w0.elapsed(); + sink ^= c1; + cyc.push(c1.saturating_sub(c0)); + lat_ns.push(w1.as_nanos() as u64); + } + std::hint::black_box(sink); + cyc.sort_unstable(); + lat_ns.sort_unstable(); + // Use the medians as the floor — robust to the occasional interrupt. + (pct(&lat_ns, 50.0), pct(&cyc, 50.0)) +} + +fn main() { + let rounds = env_usize("SMARM_SWITCH_ROUNDS", 200_000); + let warmup = env_usize("SMARM_SWITCH_WARMUP", 10_000); + let runs = env_usize("SMARM_SWITCH_RUNS", 5); + let mode = "local"; + + // Calibrate the instrumentation floor once, with a healthy sample. + let (floor_ns, floor_cyc) = calibrate(rounds.min(50_000).max(10_000)); + + let mut pooled_ns: Vec = Vec::new(); + let mut pooled_cyc: Vec = Vec::new(); + + for _ in 0..runs { + let s = one_run(1, rounds, warmup); + // Subtract the instrumentation floor; saturating so a sub-floor outlier + // (clock granularity) clamps to 0 rather than wrapping. + pooled_ns.extend(s.lat_ns.iter().map(|&v| v.saturating_sub(floor_ns))); + pooled_cyc.extend(s.cyc.iter().map(|&v| v.saturating_sub(floor_cyc))); + } + + pooled_ns.sort_unstable(); + pooled_cyc.sort_unstable(); + + let n = pooled_ns.len(); + let mean_ns = pooled_ns.iter().map(|&v| v as f64).sum::() / n.max(1) as f64; + let mean_cyc = pooled_cyc.iter().map(|&v| v as f64).sum::() / n.max(1) as f64; + // Derived effective frequency: cycles per ns = GHz. Cross-checks the two + // lenses against the box's known base clock. + let derived_ghz = if mean_ns > 0.0 { mean_cyc / mean_ns } else { 0.0 }; + + let p50 = pct(&pooled_ns, 50.0); + let p90 = pct(&pooled_ns, 90.0); + let p99 = pct(&pooled_ns, 99.0); + let lo = *pooled_ns.first().unwrap_or(&0); + let hi = *pooled_ns.last().unwrap_or(&0); + + // House table. + println!(); + println!("per-switch cost — {} mode, variant={}", mode, variant()); + println!( + " rounds={} warmup={} runs={} (instrumentation floor: {} ns / {} cyc, subtracted)", + rounds, warmup, runs, floor_ns, floor_cyc + ); + println!(" {:<10} {:<10} {:<10} {:<10} {:<10}", "p50 ns", "p90 ns", "p99 ns", "min ns", "max ns"); + println!(" {:<10} {:<10} {:<10} {:<10} {:<10}", p50, p90, p99, lo, hi); + println!( + " mean {:.1} ns | mean {:.0} cyc | derived {:.3} GHz", + mean_ns, mean_cyc, derived_ghz + ); + + // Greppable line — same spirit as SPINCSV. + println!( + "SWITCHCSV,{},{},{},{},{},{},{},{},{},{},{:.1},{:.0},{:.3}", + variant(), mode, rounds, runs, n, p50, p90, p99, lo, hi, mean_ns, mean_cyc, derived_ghz + ); +}