feat(runtime,io): driver-enqueues + park/wake idle path — retire the wake pipe
The swap (RFC 018). Schedulers no longer sleep on a shared level-triggered wake pipe — the herd source that made the default 8-thread config 7x slower than 2 threads (E1). They park on per-thread futex parkers via the coordination layer; IO backends become producers behind a two-call contract (make runnable, then the enqueue tail wakes exactly one parked scheduler). Deleted: the drain lock and the one-winner phase-1 drain; the shared completions VecDeque; the wake pipe fds, poll_wake, drain_wake_pipe, wake_scheduler, the FdReady/Blocking Completion enum; the 100us idle nap; the per-pop io.lock liveness read; io.rs's as_millis timeout truncation. Added: - enqueue wake tail (fixes the silent enqueue): wake_one_if_idle, a fence + one Relaxed mask load when everyone is busy — the pure-compute hot path pays almost nothing. - driver-enqueues: the pool thread stashes its result in the slot, decrements io_outstanding, unparks; the epoll thread removes+DELs the waiter under the waiters lock and unparks. Both reach the runtime via a Weak (no Arc cycle). The waiters map moves behind its own Arc<Mutex> so the epoll thread never takes the runtime io lock (teardown holds it while joining that thread). - io_outstanding / io_fd_waiters atomics: the termination verdict reads two atomics instead of taking io.lock on every pop. - timekeeper idle path: at most one parked scheduler holds the timer deadline (an expiry wakes one, not a herd); everyone else parks indefinitely and is woken by the enqueue tail. - busy-path timer due-check (ratified design point (a)): under saturation nobody parks and no timekeeper exists, yet due timers must still fire — one Relaxed load of the earliest-deadline snapshot per loop, clock read only when a timer is armed. Maintained under the timers mutex. - chain rule: a scheduler that pops with more work queued and a sibling parked wakes one, so surplus runs in parallel rather than behind it. tests/park_wake.rs pins the two new observable properties: timers fire under full scheduler saturation, and sub-ms sleeps are prompt (the as_millis truncation regression). Full suite + all loom models green; clippy --lib clean.
This commit is contained in:
+36
-9
@@ -72,6 +72,7 @@ use crate::runtime::{
|
||||
self, RuntimeInner, YieldIntent, RUNTIME,
|
||||
};
|
||||
use crate::supervisor::Signal;
|
||||
use std::sync::atomic::Ordering;
|
||||
use std::sync::Arc;
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -753,7 +754,14 @@ where
|
||||
Err(e) => panic!("smarm: io lock poisoned (core corrupt): {e}"),
|
||||
};
|
||||
match io.as_mut() {
|
||||
Some(io) => io.submit(me, epoch, work),
|
||||
Some(io) => {
|
||||
// RFC 018: count the op in flight BEFORE submit — the
|
||||
// pool decrements on completion, and an increment that
|
||||
// trailed the completion would underflow. Under the io
|
||||
// lock, so ordered against the same-lock submit.
|
||||
inner.io_outstanding.fetch_add(1, Ordering::AcqRel);
|
||||
io.submit(me, epoch, work);
|
||||
}
|
||||
None => panic!("io thread not started"),
|
||||
}
|
||||
});
|
||||
@@ -813,7 +821,17 @@ fn wait_fd(fd: std::os::fd::RawFd, readable: bool, writable: bool) -> std::io::R
|
||||
Err(e) => panic!("smarm: io lock poisoned (core corrupt): {e}"),
|
||||
};
|
||||
match io.as_mut() {
|
||||
Some(io) => io.epoll_register(fd, me, epoch, readable, writable),
|
||||
Some(io) => {
|
||||
// RFC 018: count the waiter BEFORE the ADD (mirror of
|
||||
// submit); roll back if the registration fails so a
|
||||
// rejected wait leaves the verdict counters clean.
|
||||
inner.io_fd_waiters.fetch_add(1, Ordering::AcqRel);
|
||||
let r = io.epoll_register(fd, me, epoch, readable, writable);
|
||||
if r.is_err() {
|
||||
inner.io_fd_waiters.fetch_sub(1, Ordering::AcqRel);
|
||||
}
|
||||
r
|
||||
}
|
||||
None => panic!("io thread not started"),
|
||||
}
|
||||
})?;
|
||||
@@ -838,9 +856,12 @@ fn wait_fd(fd: std::os::fd::RawFd, readable: bool, writable: bool) -> std::io::R
|
||||
Err(e) => panic!("smarm: io lock poisoned (core corrupt): {e}"),
|
||||
};
|
||||
if let Some(io) = io.as_mut() {
|
||||
if io.waiters.get(&self.fd) == Some(&(self.me, self.epoch)) {
|
||||
io.waiters.remove(&self.fd);
|
||||
io.epoll_deregister(self.fd);
|
||||
// `cancel_waiter` removes + DELs iff still ours, all
|
||||
// under the waiters lock (the ADD/DEL serialization);
|
||||
// decrement only when we actually removed it — a
|
||||
// FdReady that consumed it already did the decrement.
|
||||
if io.cancel_waiter(self.fd, self.me, self.epoch) {
|
||||
inner.io_fd_waiters.fetch_sub(1, Ordering::AcqRel);
|
||||
}
|
||||
}
|
||||
});
|
||||
@@ -908,7 +929,14 @@ impl crate::channel::Selectable for FdArm {
|
||||
};
|
||||
match io.as_mut() {
|
||||
Some(io) => {
|
||||
io.epoll_register(self.fd, pid, epoch, self.readable, self.writable)
|
||||
inner.io_fd_waiters.fetch_add(1, Ordering::AcqRel);
|
||||
let r = io.epoll_register(
|
||||
self.fd, pid, epoch, self.readable, self.writable,
|
||||
);
|
||||
if r.is_err() {
|
||||
inner.io_fd_waiters.fetch_sub(1, Ordering::AcqRel);
|
||||
}
|
||||
r
|
||||
}
|
||||
None => panic!("io thread not started"),
|
||||
}
|
||||
@@ -936,9 +964,8 @@ impl crate::channel::Selectable for FdArm {
|
||||
Err(e) => panic!("smarm: io lock poisoned (core corrupt): {e}"),
|
||||
};
|
||||
if let Some(io) = io.as_mut() {
|
||||
if io.waiters.get(&self.fd) == Some(&(pid, epoch)) {
|
||||
io.waiters.remove(&self.fd);
|
||||
io.epoll_deregister(self.fd);
|
||||
if io.cancel_waiter(self.fd, pid, epoch) {
|
||||
inner.io_fd_waiters.fetch_sub(1, Ordering::AcqRel);
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
Reference in New Issue
Block a user