From 6172f4231dd08b70a2fd28df2eef230fa0b4b5e0 Mon Sep 17 00:00:00 2001 From: claude-asm-audit Date: Sat, 15 Aug 2026 12:25:15 +0000 Subject: [PATCH] perf(runtime): skip take_closure's locked swap after first resume Every resume paid an unconditional AtomicPtr::swap (lock xchg, full barrier) to check for a first-resume closure that is null on all resumes after the first. A Relaxed null-load fast path is sound: store_closure runs only before publish_queued, whose Release pairing with try_claim's Acquire orders it before this call, so no writer can race the load within an occupancy. Measured on switch_cost (1-core sandbox, rq-mutex, cycles): mean roundtrip 350-355 -> 324-328, ~7.5%. All lib + scheduler/channel/ supervisor tests pass. --- src/runtime.rs | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/src/runtime.rs b/src/runtime.rs index 5f6cf35..fe57066 100644 --- a/src/runtime.rs +++ b/src/runtime.rs @@ -873,6 +873,15 @@ impl Slot { } fn take_closure(&self) -> Option { + // Fast path: every resume after the first (the overwhelming case) + // finds null. A plain load suffices to prove it — `store_closure` + // runs only before `publish_queued`, whose Release/Acquire pairing + // with the claimer's `try_claim` orders it before this call, so no + // writer can race the load within an occupancy. This keeps the + // locked RMW (full barrier, ~20+ cycles) off the per-resume path. + if self.closure.load(Ordering::Relaxed).is_null() { + return None; + } let raw = self.closure.swap(std::ptr::null_mut(), Ordering::Acquire); if raw.is_null() { None