bench: add switch_cost local-mode per-switch microbench (rdtsc+wall, RFC per-switch spike)
This commit is contained in:
@@ -61,3 +61,7 @@ harness = false
|
||||
[[bench]]
|
||||
name = "rq_runtime"
|
||||
harness = false
|
||||
|
||||
[[bench]]
|
||||
name = "switch_cost"
|
||||
harness = false
|
||||
|
||||
@@ -0,0 +1,256 @@
|
||||
//! Per-switch (context-switch) cost microbench — the profiling-spike harness
|
||||
//! for the ROADMAP "Per-switch cost (context shims, epoch protocol)" item.
|
||||
//!
|
||||
//! The spin work (RFC 004) is closed; the next perf target is the per-switch
|
||||
//! cost itself. Shootout evidence: per-wake latency is ~0.16–0.18µs at N=1 but
|
||||
//! ~0.8–1.2µs at N=8+, and the residual is attributed to the context-switch
|
||||
//! shims (`src/context.rs`) and the epoch protocol — NOT the queue. This binary
|
||||
//! isolates that round-trip so the cycles can be attributed under `perf` and an
|
||||
//! rdtsc bracket, feeding the RFC.
|
||||
//!
|
||||
//! WHAT THE ROUND-TRIP IS
|
||||
//!
|
||||
//! `yield_now()` from inside an actor does exactly one park/unpark round-trip
|
||||
//! with nothing else attached:
|
||||
//!
|
||||
//! actor: switch_to_scheduler ──► scheduler re-queues the actor (slot-word
|
||||
//! (context.rs shim) epoch/state transition, run_queue push),
|
||||
//! pops it straight back, switch_to_actor
|
||||
//! actor resumes ◄──────────────────────────────────────────────────────
|
||||
//!
|
||||
//! No IO thread traffic, no channel, no timer, no cross-thread wake. On a
|
||||
//! single-scheduler runtime the re-queue+repop never leaves this core, so the
|
||||
//! sample is the *pure* shim + epoch + queue-op cost with zero coherency
|
||||
//! traffic. That is the `local` baseline; a `remote` mode (wake straddling two
|
||||
//! schedulers, to expose the N=1→N=8 coherency/TLS-mode jump) is a deliberate
|
||||
//! follow-up and is NOT in this file yet — local first, per the spike plan.
|
||||
//!
|
||||
//! TWO LENSES ON THE SAME LOOP
|
||||
//!
|
||||
//! wall — `Instant` bracket per round-trip. Source of truth for µs, directly
|
||||
//! comparable to the shootout's per-wake latency numbers.
|
||||
//! cycles — `rdtsc` bracket per round-trip. Source of truth for the cycle
|
||||
//! budget the RFC will reason in (the spin budget is in cycles too).
|
||||
//!
|
||||
//! Reporting both lets us *derive* the effective TSC frequency (cycles/ns) from
|
||||
//! the same samples instead of hardcoding a nominal 3.7GHz — the spin_sweep
|
||||
//! lesson was that nominal-vs-actual TSC drift is exactly what produces
|
||||
//! red-herring numbers. If the derived freq matches the box's known base clock,
|
||||
//! the two lenses corroborate; if not, that mismatch is itself a finding.
|
||||
//!
|
||||
//! Knobs (env):
|
||||
//! SMARM_SWITCH_ROUNDS round-trips timed per run default 200000
|
||||
//! SMARM_SWITCH_WARMUP untimed warmup round-trips default 10000
|
||||
//! SMARM_SWITCH_RUNS runs (pooled latency, median) default 5
|
||||
//!
|
||||
//! Output: house table + one greppable line per run-set:
|
||||
//! SWITCHCSV,<variant>,<mode>,<rounds>,<runs>,<n>,<p50_ns>,<p90_ns>,<p99_ns>,
|
||||
//! <min_ns>,<max_ns>,<mean_ns>,<mean_cyc>,<derived_ghz>
|
||||
//!
|
||||
//! NOTE: a single yielding actor is the cooperative-scheduling tightest loop —
|
||||
//! it never parks on a futex (the work is always immediately re-queued), so
|
||||
//! this measures the switch+epoch+queue path, NOT the futex park. That is
|
||||
//! intentional: the futex park is the spin work's territory (RFC 004), already
|
||||
//! characterised. The unattributed constant the shootout flagged lives in the
|
||||
//! switch itself, which is what this loop hammers.
|
||||
|
||||
use smarm::runtime::{init, Config};
|
||||
use smarm::{run, spawn, yield_now};
|
||||
use std::sync::{Arc, Mutex};
|
||||
|
||||
// --------------------------------------------------------------------------
|
||||
// env helpers (house style, matching spin_sweep.rs / rq_runtime.rs)
|
||||
// --------------------------------------------------------------------------
|
||||
|
||||
fn variant() -> &'static str {
|
||||
if cfg!(feature = "rq-mpmc") {
|
||||
"rq-mpmc"
|
||||
} else if cfg!(feature = "rq-striped") {
|
||||
"rq-striped"
|
||||
} else {
|
||||
"rq-mutex"
|
||||
}
|
||||
}
|
||||
|
||||
fn env_usize(key: &str, default: usize) -> usize {
|
||||
std::env::var(key).ok().and_then(|v| v.parse().ok()).unwrap_or(default)
|
||||
}
|
||||
|
||||
// --------------------------------------------------------------------------
|
||||
// rdtsc — serialised so the bracket actually fences the round-trip.
|
||||
//
|
||||
// Plain `rdtsc` can be reordered around the work by an out-of-order core, which
|
||||
// would smear the bracket. `rdtscp` retires prior instructions before reading
|
||||
// the counter, and the trailing `lfence` blocks later instructions from
|
||||
// climbing above the second read. Pair = (rdtscp; lfence) … work … (rdtscp;
|
||||
// lfence): a standard cycle-accurate bracket. We read TSC_AUX too but ignore
|
||||
// it; the point is the ordering guarantee, not the core id.
|
||||
// --------------------------------------------------------------------------
|
||||
|
||||
#[inline(always)]
|
||||
fn rdtsc_serialised() -> u64 {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
unsafe {
|
||||
let mut aux = 0u32;
|
||||
let t = core::arch::x86_64::__rdtscp(&mut aux);
|
||||
core::arch::x86_64::_mm_lfence();
|
||||
t
|
||||
}
|
||||
#[cfg(not(target_arch = "x86_64"))]
|
||||
{
|
||||
// Non-x86 fallback: nanosecond clock standing in for cycles. The derived
|
||||
// "GHz" column then reads ~1.0 and is meaningless, but the wall lens and
|
||||
// the harness still work. The spike target box is x86-64.
|
||||
use std::time::Instant;
|
||||
thread_local! { static T0: Instant = Instant::now(); }
|
||||
T0.with(|t0| t0.elapsed().as_nanos() as u64)
|
||||
}
|
||||
}
|
||||
|
||||
// --------------------------------------------------------------------------
|
||||
// percentile / median helpers (verbatim house idiom from spin_sweep.rs)
|
||||
// --------------------------------------------------------------------------
|
||||
|
||||
/// Nearest-rank percentile over an already-sorted slice. `p` in [0, 100].
|
||||
fn pct(sorted: &[u64], p: f64) -> u64 {
|
||||
if sorted.is_empty() {
|
||||
return 0;
|
||||
}
|
||||
let idx = ((p / 100.0) * (sorted.len() - 1) as f64).round() as usize;
|
||||
sorted[idx.min(sorted.len() - 1)]
|
||||
}
|
||||
|
||||
// --------------------------------------------------------------------------
|
||||
// one run: a single actor yields ROUNDS times; we bracket each yield from
|
||||
// inside the actor (the only vantage point — the actor is suspended during the
|
||||
// scheduler half, so an external timer can't see a single round-trip).
|
||||
//
|
||||
// Per iteration we capture BOTH a wall-ns delta and a TSC-cycle delta around
|
||||
// the same `yield_now()`. The loop overhead (two clock reads + a Vec push +
|
||||
// the branch) rides along in every sample equally; we subtract an empty-loop
|
||||
// self-calibration below so the reported number is the round-trip, not the
|
||||
// instrumentation.
|
||||
// --------------------------------------------------------------------------
|
||||
|
||||
struct RunSample {
|
||||
lat_ns: Vec<u64>,
|
||||
cyc: Vec<u64>,
|
||||
}
|
||||
|
||||
fn one_run(threads: usize, rounds: usize, warmup: usize) -> RunSample {
|
||||
let out: Arc<Mutex<Option<RunSample>>> = Arc::new(Mutex::new(None));
|
||||
let out2 = out.clone();
|
||||
|
||||
let cfg = Config::exact(threads);
|
||||
init(cfg);
|
||||
|
||||
run(move || {
|
||||
let h = spawn(move || {
|
||||
// Warmup: let the actor's stack/queue slot go hot, JIT-free but
|
||||
// cache-warm, before any sample is kept.
|
||||
for _ in 0..warmup {
|
||||
yield_now();
|
||||
}
|
||||
|
||||
let mut lat_ns = Vec::with_capacity(rounds);
|
||||
let mut cyc = Vec::with_capacity(rounds);
|
||||
|
||||
for _ in 0..rounds {
|
||||
let w0 = std::time::Instant::now();
|
||||
let c0 = rdtsc_serialised();
|
||||
yield_now();
|
||||
let c1 = rdtsc_serialised();
|
||||
let w1 = w0.elapsed();
|
||||
cyc.push(c1.saturating_sub(c0));
|
||||
lat_ns.push(w1.as_nanos() as u64);
|
||||
}
|
||||
|
||||
*out2.lock().unwrap() = Some(RunSample { lat_ns, cyc });
|
||||
});
|
||||
let _ = h.join();
|
||||
});
|
||||
|
||||
let sample = out.lock().unwrap().take().expect("actor stored a sample");
|
||||
sample
|
||||
}
|
||||
|
||||
/// Empty-loop self-calibration: the same bracket with the `yield_now()` removed,
|
||||
/// run inline (no runtime). Gives the floor cost of two serialised clock reads +
|
||||
/// the push, in both lenses, to subtract from the round-trip samples.
|
||||
fn calibrate(rounds: usize) -> (u64, u64) {
|
||||
let mut lat_ns = Vec::with_capacity(rounds);
|
||||
let mut cyc = Vec::with_capacity(rounds);
|
||||
let mut sink = 0u64;
|
||||
for _ in 0..rounds {
|
||||
let w0 = std::time::Instant::now();
|
||||
let c0 = rdtsc_serialised();
|
||||
// no yield — measure the bracket itself
|
||||
let c1 = rdtsc_serialised();
|
||||
let w1 = w0.elapsed();
|
||||
sink ^= c1;
|
||||
cyc.push(c1.saturating_sub(c0));
|
||||
lat_ns.push(w1.as_nanos() as u64);
|
||||
}
|
||||
std::hint::black_box(sink);
|
||||
cyc.sort_unstable();
|
||||
lat_ns.sort_unstable();
|
||||
// Use the medians as the floor — robust to the occasional interrupt.
|
||||
(pct(&lat_ns, 50.0), pct(&cyc, 50.0))
|
||||
}
|
||||
|
||||
fn main() {
|
||||
let rounds = env_usize("SMARM_SWITCH_ROUNDS", 200_000);
|
||||
let warmup = env_usize("SMARM_SWITCH_WARMUP", 10_000);
|
||||
let runs = env_usize("SMARM_SWITCH_RUNS", 5);
|
||||
let mode = "local";
|
||||
|
||||
// Calibrate the instrumentation floor once, with a healthy sample.
|
||||
let (floor_ns, floor_cyc) = calibrate(rounds.min(50_000).max(10_000));
|
||||
|
||||
let mut pooled_ns: Vec<u64> = Vec::new();
|
||||
let mut pooled_cyc: Vec<u64> = Vec::new();
|
||||
|
||||
for _ in 0..runs {
|
||||
let s = one_run(1, rounds, warmup);
|
||||
// Subtract the instrumentation floor; saturating so a sub-floor outlier
|
||||
// (clock granularity) clamps to 0 rather than wrapping.
|
||||
pooled_ns.extend(s.lat_ns.iter().map(|&v| v.saturating_sub(floor_ns)));
|
||||
pooled_cyc.extend(s.cyc.iter().map(|&v| v.saturating_sub(floor_cyc)));
|
||||
}
|
||||
|
||||
pooled_ns.sort_unstable();
|
||||
pooled_cyc.sort_unstable();
|
||||
|
||||
let n = pooled_ns.len();
|
||||
let mean_ns = pooled_ns.iter().map(|&v| v as f64).sum::<f64>() / n.max(1) as f64;
|
||||
let mean_cyc = pooled_cyc.iter().map(|&v| v as f64).sum::<f64>() / n.max(1) as f64;
|
||||
// Derived effective frequency: cycles per ns = GHz. Cross-checks the two
|
||||
// lenses against the box's known base clock.
|
||||
let derived_ghz = if mean_ns > 0.0 { mean_cyc / mean_ns } else { 0.0 };
|
||||
|
||||
let p50 = pct(&pooled_ns, 50.0);
|
||||
let p90 = pct(&pooled_ns, 90.0);
|
||||
let p99 = pct(&pooled_ns, 99.0);
|
||||
let lo = *pooled_ns.first().unwrap_or(&0);
|
||||
let hi = *pooled_ns.last().unwrap_or(&0);
|
||||
|
||||
// House table.
|
||||
println!();
|
||||
println!("per-switch cost — {} mode, variant={}", mode, variant());
|
||||
println!(
|
||||
" rounds={} warmup={} runs={} (instrumentation floor: {} ns / {} cyc, subtracted)",
|
||||
rounds, warmup, runs, floor_ns, floor_cyc
|
||||
);
|
||||
println!(" {:<10} {:<10} {:<10} {:<10} {:<10}", "p50 ns", "p90 ns", "p99 ns", "min ns", "max ns");
|
||||
println!(" {:<10} {:<10} {:<10} {:<10} {:<10}", p50, p90, p99, lo, hi);
|
||||
println!(
|
||||
" mean {:.1} ns | mean {:.0} cyc | derived {:.3} GHz",
|
||||
mean_ns, mean_cyc, derived_ghz
|
||||
);
|
||||
|
||||
// Greppable line — same spirit as SPINCSV.
|
||||
println!(
|
||||
"SWITCHCSV,{},{},{},{},{},{},{},{},{},{},{:.1},{:.0},{:.3}",
|
||||
variant(), mode, rounds, runs, n, p50, p90, p99, lo, hi, mean_ns, mean_cyc, derived_ghz
|
||||
);
|
||||
}
|
||||
Reference in New Issue
Block a user