//! Per-switch (context-switch) cost microbench — the profiling-spike harness //! for the ROADMAP "Per-switch cost (context shims, epoch protocol)" item. //! //! The spin work (RFC 004) is closed; the next perf target is the per-switch //! cost itself. Shootout evidence: per-wake latency is ~0.16–0.18µs at N=1 but //! ~0.8–1.2µs at N=8+, and the residual is attributed to the context-switch //! shims (`src/context.rs`) and the epoch protocol — NOT the queue. This binary //! isolates that round-trip so the cycles can be attributed under `perf` and an //! rdtsc bracket, feeding the RFC. //! //! WHAT THE ROUND-TRIP IS //! //! `yield_now()` from inside an actor does exactly one park/unpark round-trip //! with nothing else attached: //! //! actor: switch_to_scheduler ──► scheduler re-queues the actor (slot-word //! (context.rs shim) epoch/state transition, run_queue push), //! pops it straight back, switch_to_actor //! actor resumes ◄────────────────────────────────────────────────────── //! //! No IO thread traffic, no channel, no timer, no cross-thread wake. On a //! single-scheduler runtime the re-queue+repop never leaves this core, so the //! sample is the *pure* shim + epoch + queue-op cost with zero coherency //! traffic. That is the `local` baseline; a `remote` mode (wake straddling two //! schedulers, to expose the N=1→N=8 coherency/TLS-mode jump) is a deliberate //! follow-up and is NOT in this file yet — local first, per the spike plan. //! //! TWO LENSES ON THE SAME LOOP //! //! wall — `Instant` bracket per round-trip. Source of truth for µs, directly //! comparable to the shootout's per-wake latency numbers. //! cycles — `rdtsc` bracket per round-trip. Source of truth for the cycle //! budget the RFC will reason in (the spin budget is in cycles too). //! //! Reporting both lets us *derive* the effective TSC frequency (cycles/ns) from //! the same samples instead of hardcoding a nominal 3.7GHz — the spin_sweep //! lesson was that nominal-vs-actual TSC drift is exactly what produces //! red-herring numbers. If the derived freq matches the box's known base clock, //! the two lenses corroborate; if not, that mismatch is itself a finding. //! //! Knobs (env): //! SMARM_SWITCH_ROUNDS round-trips timed per run default 200000 //! SMARM_SWITCH_WARMUP untimed warmup round-trips default 10000 //! SMARM_SWITCH_RUNS runs (pooled latency, median) default 5 //! //! Output: house table + one greppable line per run-set: //! SWITCHCSV,,,,,,,,, //! ,,,, //! //! NOTE: a single yielding actor is the cooperative-scheduling tightest loop — //! it never parks on a futex (the work is always immediately re-queued), so //! this measures the switch+epoch+queue path, NOT the futex park. That is //! intentional: the futex park is the spin work's territory (RFC 004), already //! characterised. The unattributed constant the shootout flagged lives in the //! switch itself, which is what this loop hammers. use smarm::runtime::{init, Config}; use smarm::{run, spawn, yield_now}; use std::sync::{Arc, Mutex}; // -------------------------------------------------------------------------- // env helpers (house style, matching spin_sweep.rs / rq_runtime.rs) // -------------------------------------------------------------------------- fn variant() -> &'static str { if cfg!(feature = "rq-mpmc") { "rq-mpmc" } else if cfg!(feature = "rq-striped") { "rq-striped" } else { "rq-mutex" } } fn env_usize(key: &str, default: usize) -> usize { std::env::var(key).ok().and_then(|v| v.parse().ok()).unwrap_or(default) } // -------------------------------------------------------------------------- // rdtsc — serialised so the bracket actually fences the round-trip. // // Plain `rdtsc` can be reordered around the work by an out-of-order core, which // would smear the bracket. `rdtscp` retires prior instructions before reading // the counter, and the trailing `lfence` blocks later instructions from // climbing above the second read. Pair = (rdtscp; lfence) … work … (rdtscp; // lfence): a standard cycle-accurate bracket. We read TSC_AUX too but ignore // it; the point is the ordering guarantee, not the core id. // -------------------------------------------------------------------------- #[inline(always)] fn rdtsc_serialised() -> u64 { #[cfg(target_arch = "x86_64")] unsafe { let mut aux = 0u32; let t = core::arch::x86_64::__rdtscp(&mut aux); core::arch::x86_64::_mm_lfence(); t } #[cfg(not(target_arch = "x86_64"))] { // Non-x86 fallback: nanosecond clock standing in for cycles. The derived // "GHz" column then reads ~1.0 and is meaningless, but the wall lens and // the harness still work. The spike target box is x86-64. use std::time::Instant; thread_local! { static T0: Instant = Instant::now(); } T0.with(|t0| t0.elapsed().as_nanos() as u64) } } // -------------------------------------------------------------------------- // percentile / median helpers (verbatim house idiom from spin_sweep.rs) // -------------------------------------------------------------------------- /// Nearest-rank percentile over an already-sorted slice. `p` in [0, 100]. fn pct(sorted: &[u64], p: f64) -> u64 { if sorted.is_empty() { return 0; } let idx = ((p / 100.0) * (sorted.len() - 1) as f64).round() as usize; sorted[idx.min(sorted.len() - 1)] } // -------------------------------------------------------------------------- // one run: a single actor yields ROUNDS times; we bracket each yield from // inside the actor (the only vantage point — the actor is suspended during the // scheduler half, so an external timer can't see a single round-trip). // // Per iteration we capture BOTH a wall-ns delta and a TSC-cycle delta around // the same `yield_now()`. The loop overhead (two clock reads + a Vec push + // the branch) rides along in every sample equally; we subtract an empty-loop // self-calibration below so the reported number is the round-trip, not the // instrumentation. // -------------------------------------------------------------------------- struct RunSample { lat_ns: Vec, cyc: Vec, } fn one_run(threads: usize, rounds: usize, warmup: usize) -> RunSample { let out: Arc>> = Arc::new(Mutex::new(None)); let out2 = out.clone(); let cfg = Config::exact(threads); init(cfg); run(move || { let h = spawn(move || { // Warmup: let the actor's stack/queue slot go hot, JIT-free but // cache-warm, before any sample is kept. for _ in 0..warmup { yield_now(); } let mut lat_ns = Vec::with_capacity(rounds); let mut cyc = Vec::with_capacity(rounds); for _ in 0..rounds { let w0 = std::time::Instant::now(); let c0 = rdtsc_serialised(); yield_now(); let c1 = rdtsc_serialised(); let w1 = w0.elapsed(); cyc.push(c1.saturating_sub(c0)); lat_ns.push(w1.as_nanos() as u64); } *out2.lock().unwrap() = Some(RunSample { lat_ns, cyc }); }); let _ = h.join(); }); let sample = out.lock().unwrap().take().expect("actor stored a sample"); sample } /// Empty-loop self-calibration: the same bracket with the `yield_now()` removed, /// run inline (no runtime). Gives the floor cost of two serialised clock reads + /// the push, in both lenses, to subtract from the round-trip samples. fn calibrate(rounds: usize) -> (u64, u64) { let mut lat_ns = Vec::with_capacity(rounds); let mut cyc = Vec::with_capacity(rounds); let mut sink = 0u64; for _ in 0..rounds { let w0 = std::time::Instant::now(); let c0 = rdtsc_serialised(); // no yield — measure the bracket itself let c1 = rdtsc_serialised(); let w1 = w0.elapsed(); sink ^= c1; cyc.push(c1.saturating_sub(c0)); lat_ns.push(w1.as_nanos() as u64); } std::hint::black_box(sink); cyc.sort_unstable(); lat_ns.sort_unstable(); // Use the medians as the floor — robust to the occasional interrupt. (pct(&lat_ns, 50.0), pct(&cyc, 50.0)) } fn main() { let rounds = env_usize("SMARM_SWITCH_ROUNDS", 200_000); let warmup = env_usize("SMARM_SWITCH_WARMUP", 10_000); let runs = env_usize("SMARM_SWITCH_RUNS", 5); let mode = "local"; // Calibrate the instrumentation floor once, with a healthy sample. let (floor_ns, floor_cyc) = calibrate(rounds.min(50_000).max(10_000)); let mut pooled_ns: Vec = Vec::new(); let mut pooled_cyc: Vec = Vec::new(); for _ in 0..runs { let s = one_run(1, rounds, warmup); // Subtract the instrumentation floor; saturating so a sub-floor outlier // (clock granularity) clamps to 0 rather than wrapping. pooled_ns.extend(s.lat_ns.iter().map(|&v| v.saturating_sub(floor_ns))); pooled_cyc.extend(s.cyc.iter().map(|&v| v.saturating_sub(floor_cyc))); } pooled_ns.sort_unstable(); pooled_cyc.sort_unstable(); let n = pooled_ns.len(); let mean_ns = pooled_ns.iter().map(|&v| v as f64).sum::() / n.max(1) as f64; let mean_cyc = pooled_cyc.iter().map(|&v| v as f64).sum::() / n.max(1) as f64; // Derived effective frequency: cycles per ns = GHz. Cross-checks the two // lenses against the box's known base clock. let derived_ghz = if mean_ns > 0.0 { mean_cyc / mean_ns } else { 0.0 }; let p50 = pct(&pooled_ns, 50.0); let p90 = pct(&pooled_ns, 90.0); let p99 = pct(&pooled_ns, 99.0); let lo = *pooled_ns.first().unwrap_or(&0); let hi = *pooled_ns.last().unwrap_or(&0); // House table. println!(); println!("per-switch cost — {} mode, variant={}", mode, variant()); println!( " rounds={} warmup={} runs={} (instrumentation floor: {} ns / {} cyc, subtracted)", rounds, warmup, runs, floor_ns, floor_cyc ); println!(" {:<10} {:<10} {:<10} {:<10} {:<10}", "p50 ns", "p90 ns", "p99 ns", "min ns", "max ns"); println!(" {:<10} {:<10} {:<10} {:<10} {:<10}", p50, p90, p99, lo, hi); println!( " mean {:.1} ns | mean {:.0} cyc | derived {:.3} GHz", mean_ns, mean_cyc, derived_ghz ); // Greppable line — same spirit as SPINCSV. println!( "SWITCHCSV,{},{},{},{},{},{},{},{},{},{},{:.1},{:.0},{:.3}", variant(), mode, rounds, runs, n, p50, p90, p99, lo, hi, mean_ns, mean_cyc, derived_ghz ); }