Files
smarm/benches/switch_cost.rs

257 lines
11 KiB
Rust
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
//! Per-switch (context-switch) cost microbench — the profiling-spike harness
//! for the ROADMAP "Per-switch cost (context shims, epoch protocol)" item.
//!
//! The spin work (RFC 004) is closed; the next perf target is the per-switch
//! cost itself. Shootout evidence: per-wake latency is ~0.160.18µs at N=1 but
//! ~0.81.2µs at N=8+, and the residual is attributed to the context-switch
//! shims (`src/context.rs`) and the epoch protocol — NOT the queue. This binary
//! isolates that round-trip so the cycles can be attributed under `perf` and an
//! rdtsc bracket, feeding the RFC.
//!
//! WHAT THE ROUND-TRIP IS
//!
//! `yield_now()` from inside an actor does exactly one park/unpark round-trip
//! with nothing else attached:
//!
//! actor: switch_to_scheduler ──► scheduler re-queues the actor (slot-word
//! (context.rs shim) epoch/state transition, run_queue push),
//! pops it straight back, switch_to_actor
//! actor resumes ◄──────────────────────────────────────────────────────
//!
//! No IO thread traffic, no channel, no timer, no cross-thread wake. On a
//! single-scheduler runtime the re-queue+repop never leaves this core, so the
//! sample is the *pure* shim + epoch + queue-op cost with zero coherency
//! traffic. That is the `local` baseline; a `remote` mode (wake straddling two
//! schedulers, to expose the N=1→N=8 coherency/TLS-mode jump) is a deliberate
//! follow-up and is NOT in this file yet — local first, per the spike plan.
//!
//! TWO LENSES ON THE SAME LOOP
//!
//! wall — `Instant` bracket per round-trip. Source of truth for µs, directly
//! comparable to the shootout's per-wake latency numbers.
//! cycles — `rdtsc` bracket per round-trip. Source of truth for the cycle
//! budget the RFC will reason in (the spin budget is in cycles too).
//!
//! Reporting both lets us *derive* the effective TSC frequency (cycles/ns) from
//! the same samples instead of hardcoding a nominal 3.7GHz — the spin_sweep
//! lesson was that nominal-vs-actual TSC drift is exactly what produces
//! red-herring numbers. If the derived freq matches the box's known base clock,
//! the two lenses corroborate; if not, that mismatch is itself a finding.
//!
//! Knobs (env):
//! SMARM_SWITCH_ROUNDS round-trips timed per run default 200000
//! SMARM_SWITCH_WARMUP untimed warmup round-trips default 10000
//! SMARM_SWITCH_RUNS runs (pooled latency, median) default 5
//!
//! Output: house table + one greppable line per run-set:
//! SWITCHCSV,<variant>,<mode>,<rounds>,<runs>,<n>,<p50_ns>,<p90_ns>,<p99_ns>,
//! <min_ns>,<max_ns>,<mean_ns>,<mean_cyc>,<derived_ghz>
//!
//! NOTE: a single yielding actor is the cooperative-scheduling tightest loop —
//! it never parks on a futex (the work is always immediately re-queued), so
//! this measures the switch+epoch+queue path, NOT the futex park. That is
//! intentional: the futex park is the spin work's territory (RFC 004), already
//! characterised. The unattributed constant the shootout flagged lives in the
//! switch itself, which is what this loop hammers.
use smarm::runtime::{init, Config};
use smarm::{run, spawn, yield_now};
use std::sync::{Arc, Mutex};
// --------------------------------------------------------------------------
// env helpers (house style, matching spin_sweep.rs / rq_runtime.rs)
// --------------------------------------------------------------------------
fn variant() -> &'static str {
if cfg!(feature = "rq-mpmc") {
"rq-mpmc"
} else if cfg!(feature = "rq-striped") {
"rq-striped"
} else {
"rq-mutex"
}
}
fn env_usize(key: &str, default: usize) -> usize {
std::env::var(key).ok().and_then(|v| v.parse().ok()).unwrap_or(default)
}
// --------------------------------------------------------------------------
// rdtsc — serialised so the bracket actually fences the round-trip.
//
// Plain `rdtsc` can be reordered around the work by an out-of-order core, which
// would smear the bracket. `rdtscp` retires prior instructions before reading
// the counter, and the trailing `lfence` blocks later instructions from
// climbing above the second read. Pair = (rdtscp; lfence) … work … (rdtscp;
// lfence): a standard cycle-accurate bracket. We read TSC_AUX too but ignore
// it; the point is the ordering guarantee, not the core id.
// --------------------------------------------------------------------------
#[inline(always)]
fn rdtsc_serialised() -> u64 {
#[cfg(target_arch = "x86_64")]
unsafe {
let mut aux = 0u32;
let t = core::arch::x86_64::__rdtscp(&mut aux);
core::arch::x86_64::_mm_lfence();
t
}
#[cfg(not(target_arch = "x86_64"))]
{
// Non-x86 fallback: nanosecond clock standing in for cycles. The derived
// "GHz" column then reads ~1.0 and is meaningless, but the wall lens and
// the harness still work. The spike target box is x86-64.
use std::time::Instant;
thread_local! { static T0: Instant = Instant::now(); }
T0.with(|t0| t0.elapsed().as_nanos() as u64)
}
}
// --------------------------------------------------------------------------
// percentile / median helpers (verbatim house idiom from spin_sweep.rs)
// --------------------------------------------------------------------------
/// Nearest-rank percentile over an already-sorted slice. `p` in [0, 100].
fn pct(sorted: &[u64], p: f64) -> u64 {
if sorted.is_empty() {
return 0;
}
let idx = ((p / 100.0) * (sorted.len() - 1) as f64).round() as usize;
sorted[idx.min(sorted.len() - 1)]
}
// --------------------------------------------------------------------------
// one run: a single actor yields ROUNDS times; we bracket each yield from
// inside the actor (the only vantage point — the actor is suspended during the
// scheduler half, so an external timer can't see a single round-trip).
//
// Per iteration we capture BOTH a wall-ns delta and a TSC-cycle delta around
// the same `yield_now()`. The loop overhead (two clock reads + a Vec push +
// the branch) rides along in every sample equally; we subtract an empty-loop
// self-calibration below so the reported number is the round-trip, not the
// instrumentation.
// --------------------------------------------------------------------------
struct RunSample {
lat_ns: Vec<u64>,
cyc: Vec<u64>,
}
fn one_run(threads: usize, rounds: usize, warmup: usize) -> RunSample {
let out: Arc<Mutex<Option<RunSample>>> = Arc::new(Mutex::new(None));
let out2 = out.clone();
let cfg = Config::exact(threads);
init(cfg);
run(move || {
let h = spawn(move || {
// Warmup: let the actor's stack/queue slot go hot, JIT-free but
// cache-warm, before any sample is kept.
for _ in 0..warmup {
yield_now();
}
let mut lat_ns = Vec::with_capacity(rounds);
let mut cyc = Vec::with_capacity(rounds);
for _ in 0..rounds {
let w0 = std::time::Instant::now();
let c0 = rdtsc_serialised();
yield_now();
let c1 = rdtsc_serialised();
let w1 = w0.elapsed();
cyc.push(c1.saturating_sub(c0));
lat_ns.push(w1.as_nanos() as u64);
}
*out2.lock().unwrap() = Some(RunSample { lat_ns, cyc });
});
let _ = h.join();
});
let sample = out.lock().unwrap().take().expect("actor stored a sample");
sample
}
/// Empty-loop self-calibration: the same bracket with the `yield_now()` removed,
/// run inline (no runtime). Gives the floor cost of two serialised clock reads +
/// the push, in both lenses, to subtract from the round-trip samples.
fn calibrate(rounds: usize) -> (u64, u64) {
let mut lat_ns = Vec::with_capacity(rounds);
let mut cyc = Vec::with_capacity(rounds);
let mut sink = 0u64;
for _ in 0..rounds {
let w0 = std::time::Instant::now();
let c0 = rdtsc_serialised();
// no yield — measure the bracket itself
let c1 = rdtsc_serialised();
let w1 = w0.elapsed();
sink ^= c1;
cyc.push(c1.saturating_sub(c0));
lat_ns.push(w1.as_nanos() as u64);
}
std::hint::black_box(sink);
cyc.sort_unstable();
lat_ns.sort_unstable();
// Use the medians as the floor — robust to the occasional interrupt.
(pct(&lat_ns, 50.0), pct(&cyc, 50.0))
}
fn main() {
let rounds = env_usize("SMARM_SWITCH_ROUNDS", 200_000);
let warmup = env_usize("SMARM_SWITCH_WARMUP", 10_000);
let runs = env_usize("SMARM_SWITCH_RUNS", 5);
let mode = "local";
// Calibrate the instrumentation floor once, with a healthy sample.
let (floor_ns, floor_cyc) = calibrate(rounds.min(50_000).max(10_000));
let mut pooled_ns: Vec<u64> = Vec::new();
let mut pooled_cyc: Vec<u64> = Vec::new();
for _ in 0..runs {
let s = one_run(1, rounds, warmup);
// Subtract the instrumentation floor; saturating so a sub-floor outlier
// (clock granularity) clamps to 0 rather than wrapping.
pooled_ns.extend(s.lat_ns.iter().map(|&v| v.saturating_sub(floor_ns)));
pooled_cyc.extend(s.cyc.iter().map(|&v| v.saturating_sub(floor_cyc)));
}
pooled_ns.sort_unstable();
pooled_cyc.sort_unstable();
let n = pooled_ns.len();
let mean_ns = pooled_ns.iter().map(|&v| v as f64).sum::<f64>() / n.max(1) as f64;
let mean_cyc = pooled_cyc.iter().map(|&v| v as f64).sum::<f64>() / n.max(1) as f64;
// Derived effective frequency: cycles per ns = GHz. Cross-checks the two
// lenses against the box's known base clock.
let derived_ghz = if mean_ns > 0.0 { mean_cyc / mean_ns } else { 0.0 };
let p50 = pct(&pooled_ns, 50.0);
let p90 = pct(&pooled_ns, 90.0);
let p99 = pct(&pooled_ns, 99.0);
let lo = *pooled_ns.first().unwrap_or(&0);
let hi = *pooled_ns.last().unwrap_or(&0);
// House table.
println!();
println!("per-switch cost — {} mode, variant={}", mode, variant());
println!(
" rounds={} warmup={} runs={} (instrumentation floor: {} ns / {} cyc, subtracted)",
rounds, warmup, runs, floor_ns, floor_cyc
);
println!(" {:<10} {:<10} {:<10} {:<10} {:<10}", "p50 ns", "p90 ns", "p99 ns", "min ns", "max ns");
println!(" {:<10} {:<10} {:<10} {:<10} {:<10}", p50, p90, p99, lo, hi);
println!(
" mean {:.1} ns | mean {:.0} cyc | derived {:.3} GHz",
mean_ns, mean_cyc, derived_ghz
);
// Greppable line — same spirit as SPINCSV.
println!(
"SWITCHCSV,{},{},{},{},{},{},{},{},{},{},{:.1},{:.0},{:.3}",
variant(), mode, rounds, runs, n, p50, p90, p99, lo, hi, mean_ns, mean_cyc, derived_ghz
);
}