feat(cluster): RFC 010 c3 — transport trait, framed codec, TCP + loopback impls

The control-connection abstraction (RFC v2 §5): object-safe Transport/
Listener/Conn over opaque pre-resolved addresses (resolution stays the c9
seam), with FramedConn as the single shared byte->Frame codec feeding
Frame::decode's incremental contract. Nothing forecloses additional
per-peer connections for the jarred bulk plane; the membrane is not a
transport (D2).

TCP parks the calling actor via scheduler fd readiness (MSG_NOSIGNAL
writes, EINPROGRESS dial resolved through SO_ERROR). Loopback is the
shipped in-memory test transport: OS-thread-blocking condvar pipes with
TCP-shaped close semantics, per-instance address registry.

Conformance suite runs the same codec over both impls: roundtrips both
directions, framing across split writes, coalesced frames, peer-close
mid-frame as TruncatedByPeer (not EOF), clean close as Ok(None). Plus
impl-specific establishment/error cases and a 4 MiB cross-buffer TCP
frame under real backpressure.
This commit is contained in:
Claude
2026-08-14 14:31:39 +00:00
parent 3850f6099b
commit 39ab92871e
5 changed files with 1021 additions and 4 deletions
+277
View File
@@ -0,0 +1,277 @@
//! TCP transport — the production control-plane transport.
//!
//! Blocking model: every blocking point parks the **calling actor** on fd
//! readiness ([`crate::scheduler::wait_readable`] / `wait_writable`); the
//! scheduler thread is never blocked. All conn/listener methods must
//! therefore run inside an actor. `listen` itself only binds (no waiting)
//! and is callable anywhere.
//!
//! Addresses are pre-resolved `ip:port` strings (`SocketAddr` syntax, IPv4
//! or IPv6). Hostnames are rejected with `InvalidInput`: name resolution is
//! the single c9 seam, not something each transport does on the side.
//!
//! Writes use `send(2)` with `MSG_NOSIGNAL` — a peer reset must surface as
//! `BrokenPipe`/`ConnectionReset`, not `SIGPIPE`.
use std::io;
use std::net::{SocketAddr, TcpListener as StdListener, TcpStream};
use std::os::fd::{AsRawFd, RawFd};
use crate::scheduler::{wait_readable, wait_writable};
use super::{Conn, Listener, Transport};
// ---------------------------------------------------------------------------
// sockaddr plumbing
// ---------------------------------------------------------------------------
/// A `sockaddr_in`/`sockaddr_in6` built from a parsed `SocketAddr`, plus its
/// length, ready for `connect(2)`.
union SockAddrUnion {
v4: libc::sockaddr_in,
v6: libc::sockaddr_in6,
}
fn to_sockaddr(sa: &SocketAddr) -> (SockAddrUnion, libc::socklen_t) {
match sa {
SocketAddr::V4(v4) => {
let raw = libc::sockaddr_in {
sin_family: libc::AF_INET as libc::sa_family_t,
sin_port: v4.port().to_be(),
sin_addr: libc::in_addr {
s_addr: u32::from_be_bytes(v4.ip().octets()).to_be(),
},
sin_zero: [0; 8],
};
(
SockAddrUnion { v4: raw },
std::mem::size_of::<libc::sockaddr_in>() as libc::socklen_t,
)
}
SocketAddr::V6(v6) => {
let raw = libc::sockaddr_in6 {
sin6_family: libc::AF_INET6 as libc::sa_family_t,
sin6_port: v6.port().to_be(),
sin6_flowinfo: v6.flowinfo(),
sin6_addr: libc::in6_addr {
s6_addr: v6.ip().octets(),
},
sin6_scope_id: v6.scope_id(),
};
(
SockAddrUnion { v6: raw },
std::mem::size_of::<libc::sockaddr_in6>() as libc::socklen_t,
)
}
}
}
fn parse_addr(addr: &str) -> io::Result<SocketAddr> {
addr.parse().map_err(|_| {
io::Error::new(
io::ErrorKind::InvalidInput,
format!("{addr:?} is not a resolved ip:port — resolution is the c9 seam"),
)
})
}
fn so_error(fd: RawFd) -> io::Result<()> {
let mut err: libc::c_int = 0;
let mut len = std::mem::size_of::<libc::c_int>() as libc::socklen_t;
let rc = unsafe {
libc::getsockopt(
fd,
libc::SOL_SOCKET,
libc::SO_ERROR,
(&mut err) as *mut _ as *mut libc::c_void,
&mut len,
)
};
if rc != 0 {
return Err(io::Error::last_os_error());
}
if err != 0 {
return Err(io::Error::from_raw_os_error(err));
}
Ok(())
}
// ---------------------------------------------------------------------------
// Conn
// ---------------------------------------------------------------------------
/// One established TCP control connection. Owns the socket; drop closes it.
pub struct TcpConn {
stream: TcpStream,
closed: bool,
}
impl TcpConn {
fn fd(&self) -> RawFd {
self.stream.as_raw_fd()
}
}
impl Conn for TcpConn {
fn read(&mut self, buf: &mut [u8]) -> io::Result<usize> {
if self.closed {
return Ok(0);
}
if buf.is_empty() {
return Ok(0);
}
loop {
wait_readable(self.fd())?;
let n = unsafe { libc::read(self.fd(), buf.as_mut_ptr() as *mut _, buf.len()) };
if n >= 0 {
return Ok(n as usize);
}
let e = io::Error::last_os_error();
match e.kind() {
// Spurious readiness or signal: park again.
io::ErrorKind::WouldBlock | io::ErrorKind::Interrupted => continue,
_ => return Err(e),
}
}
}
fn write_all(&mut self, mut buf: &[u8]) -> io::Result<()> {
if self.closed {
return Err(io::Error::new(
io::ErrorKind::NotConnected,
"tcp conn closed locally",
));
}
while !buf.is_empty() {
wait_writable(self.fd())?;
let n = unsafe {
libc::send(
self.fd(),
buf.as_ptr() as *const _,
buf.len(),
libc::MSG_NOSIGNAL,
)
};
if n >= 0 {
buf = &buf[n as usize..];
continue;
}
let e = io::Error::last_os_error();
match e.kind() {
io::ErrorKind::WouldBlock | io::ErrorKind::Interrupted => continue,
_ => return Err(e),
}
}
Ok(())
}
fn close(&mut self) {
if !self.closed {
self.closed = true;
// Best-effort: the peer sees EOF after draining. The fd itself
// is released when the owning stream drops.
let _ = self.stream.shutdown(std::net::Shutdown::Both);
}
}
fn peer_addr(&self) -> String {
match self.stream.peer_addr() {
Ok(sa) => sa.to_string(),
Err(_) => "<disconnected>".to_string(),
}
}
}
// ---------------------------------------------------------------------------
// Listener
// ---------------------------------------------------------------------------
/// A bound TCP listen point (non-blocking socket; accept parks the actor).
pub struct TcpListener {
inner: StdListener,
local: SocketAddr,
}
impl Listener for TcpListener {
fn accept(&mut self) -> io::Result<Box<dyn Conn>> {
loop {
wait_readable(self.inner.as_raw_fd())?;
match self.inner.accept() {
Ok((stream, _peer)) => {
stream.set_nonblocking(true)?;
return Ok(Box::new(TcpConn {
stream,
closed: false,
}));
}
Err(e)
if e.kind() == io::ErrorKind::WouldBlock
|| e.kind() == io::ErrorKind::Interrupted =>
{
continue;
}
Err(e) => return Err(e),
}
}
}
fn local_addr(&self) -> String {
self.local.to_string()
}
}
// ---------------------------------------------------------------------------
// Transport
// ---------------------------------------------------------------------------
/// The TCP transport. Stateless; every call stands alone.
pub struct TcpTransport;
impl Transport for TcpTransport {
fn dial(&self, addr: &str) -> io::Result<Box<dyn Conn>> {
let sa = parse_addr(addr)?;
let family = match sa {
SocketAddr::V4(_) => libc::AF_INET,
SocketAddr::V6(_) => libc::AF_INET6,
};
let fd = unsafe {
libc::socket(
family,
libc::SOCK_STREAM | libc::SOCK_NONBLOCK | libc::SOCK_CLOEXEC,
0,
)
};
if fd < 0 {
return Err(io::Error::last_os_error());
}
// From here the fd is owned by `stream`; any early return drops it.
let stream = unsafe {
use std::os::fd::FromRawFd;
TcpStream::from_raw_fd(fd)
};
let (raw, len) = to_sockaddr(&sa);
let rc = unsafe { libc::connect(fd, (&raw) as *const _ as *const libc::sockaddr, len) };
if rc != 0 {
let e = io::Error::last_os_error();
if e.raw_os_error() != Some(libc::EINPROGRESS) {
return Err(e);
}
// Connect in flight: park until the socket is writable, then the
// verdict is in SO_ERROR.
wait_writable(fd)?;
so_error(fd)?;
}
Ok(Box::new(TcpConn {
stream,
closed: false,
}))
}
fn listen(&self, addr: &str) -> io::Result<Box<dyn Listener>> {
let sa = parse_addr(addr)?;
let inner = StdListener::bind(sa)?;
inner.set_nonblocking(true)?;
let local = inner.local_addr()?;
Ok(Box::new(TcpListener { inner, local }))
}
}