diff --git a/issues/design-debt/toyos-has-its-own-network-stack.md b/issues/design-debt/toyos-has-its-own-network-stack.md index a43b140d96a..a512ac65046 100644 --- a/issues/design-debt/toyos-has-its-own-network-stack.md +++ b/issues/design-debt/toyos-has-its-own-network-stack.md @@ -16,6 +16,8 @@ netd runs smoltcp. Its replacement is ToyOS's own stack, built clean-room: reade - Stage 5: netd on one shard, pipe ABI unchanged; smoltcp leaves netd, `Cargo.toml` and `userland/Cargo.lock` in the same PR. - Then multi-core, netring (blocked on the owner's ABI ruling), TCP and IP hardening, IPv6, offloads, soak. +The listener defects are this track's: `issues/hardware/a-handshake-nobody-finishes-holds-a-listeners-port-shut.md` and `issues/hardware/a-connect-between-two-accepts-is-reset.md`, on smoltcp until stage 5, and `issues/hardware/an-accept-that-never-reaches-netd-strands-its-listener.md`, in std's accept. + Owed from the wire specification by stages 3–5: ETH-32, whose subnet broadcast needs the subnet `toyos-net-ip` holds; and those whose layer tags are `[ip]`, `[shell]` or `[udp]`: ETH-11, 12, 22–25, 33; ARP-16–18; IP-25, 26, 35; IPP-01–13; ICMP-32–46; IGMP-23–25, 29; UDP-17, 18; and the policy halves of ETH-10, 14, 19, IP-02, 20–23, 29, IPO-15, 16, ICMP-23, 24, 26 and UDP-22. What `toyos-net-wire` does not yet meet: diff --git a/issues/hardware/a-handshake-nobody-finishes-holds-a-listeners-port-shut.md b/issues/hardware/a-handshake-nobody-finishes-holds-a-listeners-port-shut.md new file mode 100644 index 00000000000..0bd71ff9299 --- /dev/null +++ b/issues/hardware/a-handshake-nobody-finishes-holds-a-listeners-port-shut.md @@ -0,0 +1,31 @@ +--- +status: open +kind: defect +opened: 2026-09-27 +--- + +# A handshake nobody finishes holds a listener's port shut + +netd's listener is one smoltcp socket, and the port listens only while that +socket is in `Listen` (`userland/netd/src/listen.rs`). A SYN whose sender never +answers the SYN-ACK (a peer gone, a spoofed source) leaves it in `SynReceived`, +and smoltcp 0.12 retransmits the SYN-ACK without end. + +Measured on smoltcp's interface over a wire played by hand, as +`userland/netd/src/listen/tests.rs` plays it: after one SYN and 600 s of +silence the socket was still `SynReceived`, having sent 72 SYN-ACKs, and +another peer's SYN in that state was answered with a reset. With +`set_timeout(10 s)` the socket went `Closed` at 10 s, which `listen::settle` +turns back into `Listen`. + +So one packet shuts sshd's port for the rest of the boot. Nothing has chosen a +bound on a listener's half-open handshake, and the timeout smoltcp offers also +bounds the connection the socket becomes, so it would have to come off at the +hand-over. + +**Owner**: whoever holds `issues/design-debt/toyos-has-its-own-network-stack.md`, +which carries this and `issues/hardware/a-connect-between-two-accepts-is-reset.md`. + +**Exit**: a listener's half-open handshake let go within a bound, and a test +that sends one SYN and nothing more, then finds the port answering the next +peer with a SYN-ACK. diff --git a/issues/hardware/an-accept-that-never-reaches-netd-strands-its-listener.md b/issues/hardware/an-accept-that-never-reaches-netd-strands-its-listener.md new file mode 100644 index 00000000000..0d9d9fb5fca --- /dev/null +++ b/issues/hardware/an-accept-that-never-reaches-netd-strands-its-listener.md @@ -0,0 +1,26 @@ +--- +status: open +kind: defect +opened: 2026-09-27 +--- + +# An accept that never reaches netd strands its listener's owner + +std's `TcpListener::accept` (`rust/library/std/src/sys/net/connection/toyos.rs`) +reads netd's wake byte first, and only then calls `toyos::net::tcp_accept`, +which reaches netd (`NetdConn::connect`) and makes the data path +(`DataPath::create`) before it sends the request. If either fails, or the send +does, the accept returns an error with the wake spent and no request made. netd +spends a wake only on an accept it answers (`userland/netd/src/listen.rs`), so +it still counts the owner as woken, writes no second wake, and the owner's next +`accept` blocks for the rest of the boot while the connection holds the port. + +Read from the code and not reproduced. `NetdConn::connect` fails with +`ResourceExhausted` when the kernel's port queue refuses it, and +`DataPath::create` with `Io` when a pipe cannot be made. + +**Owner**: whoever holds `issues/design-debt/toyos-has-its-own-network-stack.md`. + +**Exit**: no failure between the wake and netd's answer leaves the owner +holding a spent wake netd still counts, and a test in which that step fails and +the next `accept` still returns the waiting connection. diff --git a/tests/toyos-rust-tests/src/bin/netd_refused_accept.rs b/tests/toyos-rust-tests/src/bin/netd_refused_accept.rs new file mode 100644 index 00000000000..7485fcb3464 --- /dev/null +++ b/tests/toyos-rust-tests/src/bin/netd_refused_accept.rs @@ -0,0 +1,129 @@ +//! An accept netd refuses for room still spends its owner's wake, so the +//! connection it left is announced again once room returns, and not before. +//! +//! The host dials this program's listener through the forward. Woken, this +//! program fills netd's connections to the host until one is refused, and +//! asks for the connection; netd refuses it for room. Once a request netd +//! answered after that refusal says room is still gone, no wake may be +//! waiting. One connection closed, the next wake is the verdict. +//! +//! argv[1] is the port of the harness's host server on `HOST`, and the harness +//! forwards a host port to this guest's `FORWARDED_PORT`. +//! `netd_refused_accept: ok` is the only success line. + +#[path = "../netd_stream.rs"] +mod netd_stream; + +use netd_stream::{ask, Ask, FORWARDED_PORT, HOST}; +use toyos::net::{ + MsgType, NetError, NetdConn, TcpAcceptPipedRequest, TcpAcceptPipedResponse, TcpBound, TcpConnectPipedRequest, + TcpConnectResponse, TcpConnection, TcpSocketId, DATA_FROM_CLIENT, DATA_HANDLES, DATA_TO_CLIENT, +}; +use toyos_abi::syscall::SyscallError; + +fn main() { + let port: u16 = std::env::args() + .nth(1) + .and_then(|p| p.parse().ok()) + .expect("usage: netd_refused_accept "); + let listener = toyos::net::tcp_bind([0; 4], FORWARDED_PORT).expect("bind the forwarded port"); + + let dial = toyos::net::tcp_connect(HOST, port, 0).expect("connect to the host server"); + ask(&dial.tx, Ask::Dial); + wake(&listener, "the host's dial"); + let mut held = Vec::new(); + let refused = loop { + match connect(port) { + Ok(conn) => { + ask(&conn.tx, Ask::Held(0)); + held.push(conn); + } + Err(e) => break e, + } + }; + assert_eq!(refused, NetError::ResourceExhausted, "a connect after {} held", held.len()); + assert_eq!( + accept(listener.socket_id), + Err(NetError::ResourceExhausted), + "an accept with every connection taken" + ); + println!("netd_refused_accept: an accept refused for room, {} connections held", held.len()); + assert_eq!( + connect(port).err(), + Some(NetError::ResourceExhausted), + "a connect after an accept refused for room" + ); + let mut byte = [0u8; 1]; + assert_eq!( + listener.notify.read_nonblock(&mut byte), + Err(SyscallError::WouldBlock), + "netd woke its owner for a connection there is no room to take" + ); + end(held.pop().expect("the cap holds at least the connection before the refusal")); + wake(&listener, "the connection an accept refused for room left, once room returned"); + accept(listener.socket_id).unwrap_or_else(|e| panic!("the connection an accept refused for room left: {e:?}")); + end(dial); + held.into_iter().for_each(end); + toyos::net::tcp_close(listener.socket_id).expect("close the listener"); + println!("netd_refused_accept: ok"); +} + +/// Wait for netd's wake on `listener`, and take it. +fn wake(listener: &TcpBound, what: &str) { + println!("netd_refused_accept: waiting for a wake for {what}"); + let mut byte = [0u8; 1]; + match listener.notify.read(&mut byte) { + Ok(1) => {} + Ok(_) => panic!("a wake for {what}: netd closed the listener"), + Err(e) => panic!("a wake for {what}: {e:?}"), + } +} + +/// netd's answer to an accept on `listener`. An accepted connection is closed +/// at once. +fn accept(listener: TcpSocketId) -> Result<(), NetError> { + let (_rx, _tx, handles) = data_path(); + let resp: TcpAcceptPipedResponse = reach_netd() + .request_with_handles(&handles, MsgType::TcpAcceptPiped, &TcpAcceptPipedRequest { socket_id: listener.0 }) + .expect("netd takes the request") + .response()?; + toyos::net::tcp_close(TcpSocketId(resp.socket_id)).expect("close the accepted connection"); + Ok(()) +} + +/// netd's answer to a connect to the host server. +fn connect(port: u16) -> Result { + let (rx, tx, handles) = data_path(); + let resp: TcpConnectResponse = reach_netd() + .request_with_handles( + &handles, + MsgType::TcpConnectPiped, + &TcpConnectPipedRequest { addr: HOST, port, _pad: 0, timeout_ms: 0 }, + ) + .expect("netd takes the request") + .response()?; + Ok(TcpConnection { rx, tx, socket_id: TcpSocketId(resp.socket_id), local_port: resp.local_port }) +} + +/// A connection to netd. **Refused only by the kernel's port queue**, which +/// `toyos::net` spells as netd's own `ResourceExhausted`, so it is no answer +/// of netd's here. +fn reach_netd() -> NetdConn { + NetdConn::connect().unwrap_or_else(|e| panic!("reach netd: {e:?}")) +} + +/// The ends of a duplex data path this side keeps, and the two it hands netd. +fn data_path() -> (toyos::Pipe, toyos::Pipe, [toyos_abi::RawHandle; DATA_HANDLES]) { + let (rx, to_client) = toyos::pipe_pair().expect("the pipe netd writes into"); + let (from_client, tx) = toyos::pipe_pair().expect("the pipe netd reads from"); + let mut handles = [toyos_abi::HANDLE_INVALID; DATA_HANDLES]; + handles[DATA_TO_CLIENT] = to_client.into_raw(); + handles[DATA_FROM_CLIENT] = from_client.into_raw(); + (rx, tx, handles) +} + +fn end(conn: TcpConnection) { + let TcpConnection { rx, tx, socket_id, .. } = conn; + drop((rx, tx)); + toyos::net::tcp_close(socket_id).expect("close a connection"); +} diff --git a/tests/toyos.rs b/tests/toyos.rs index f2ed0e90973..0bc082ba6b5 100644 --- a/tests/toyos.rs +++ b/tests/toyos.rs @@ -307,15 +307,16 @@ const RUST_SKIP: &[&str] = &[ "netd_listener_forgery", // Needs a NIC in front of netd and a host server behind it. // `netd_slow_reader`, `netd_held_open`, `netd_stalled_peer`, - // `netd_udp_refused`, `netd_udp_any_address` and `netd_refused_pipes` run - // them on `tests/netcase`, and `netd_lookup_let_go` on it with its frames - // held. + // `netd_udp_refused`, `netd_udp_any_address`, `netd_refused_pipes` and + // `netd_refused_accept` run them on `tests/netcase`, and + // `netd_lookup_let_go` on it with its frames held. "netd_slow_reader", "netd_held_open", "netd_stalled_peer", "netd_udp_refused", "netd_udp_any_address", "netd_refused_pipes", + "netd_refused_accept", "netd_lookup_let_go", // It asserts nothing at all: it holds a `tests/lancase` boot open for // twenty seconds so the host can reach this machine over the cable. On a @@ -916,6 +917,10 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // round trip after each case, a named line per refusal and a clean // console; its clocks are liveness guards. ("netd_refused_pipes", Sched::Parallel, Tier::Fast), + // The netcase boot again: an accept netd refuses for room leaves its owner + // a wake for the connection it left, once room returns. The verdict + // is the guest's wake or its absence. + ("netd_refused_accept", Sched::Parallel, Tier::Fast), // The netcase boot again: bytes held back past a full pipe move on the // pipe's room alone, the peer holding the connection open and silent. The // verdict is the guest's byte-for-byte comparison; its clocks are @@ -1683,6 +1688,7 @@ const CARRIES: &[(&str, &[&str])] = &[ ("netd_listener_forgery", &["test_rs_netd_listener_forgery"]), ("netd_slow_reader", &["test_rs_netd_slow_reader"]), ("netd_refused_pipes", &["test_rs_netd_refused_pipes"]), + ("netd_refused_accept", &["test_rs_netd_refused_accept"]), ("netd_held_open", &["test_rs_netd_held_open"]), ("netd_stalled_peer", &["test_rs_netd_stalled_peer"]), ("netd_udp_refused", &["test_rs_netd_udp_refused"]), @@ -10535,6 +10541,22 @@ fn netd_refused_pipes(rust_bins: &[(String, Vec)]) -> Result<(), String> { Ok(()) } +/// An accept netd refuses for room leaves its owner a wake for the connection +/// it left: the guest's wakes are the verdict. This side carries that netd named the +/// refusal for room and that no program panicked. +fn netd_refused_accept(rust_bins: &[(String, Vec)]) -> Result<(), String> { + let HostRun { result, console, .. } = netcase_against_host(rust_bins, "netd_refused_accept", true, "")?; + if !result.stdout.lines().any(|l| l.trim_end().ends_with("netd_refused_accept: ok")) { + return Err(format!("the guest never said it was done:\n{}", result.stdout)); + } + if !console.contains("netd: refusing accept, ") { + return Err(format!("netd refused an accept for room without saying so:\n{console}")); + } + serial::Serial::named("boot console", console.as_str()).must_be_clean()?; + eprintln!(" [netcase] an accept refused for room left a wake once room returned"); + Ok(()) +} + /// Ctrl+Alt+D at a live desktop: every CPU answers, and the two halves of the /// report agree. /// @@ -16065,6 +16087,7 @@ fn run_machine_test( } "netd_slow_reader" => netd_slow_reader(rust_bins), "netd_refused_pipes" => netd_refused_pipes(rust_bins), + "netd_refused_accept" => netd_refused_accept(rust_bins), "netd_held_open" => netd_held_open(rust_bins), "netd_stalled_peer" => netd_stalled_peer(rust_bins), "netd_udp_refused" => netd_udp_refused(rust_bins), diff --git a/userland/netd/src/listen.rs b/userland/netd/src/listen.rs new file mode 100644 index 00000000000..1b7af579cf1 --- /dev/null +++ b/userland/netd/src/listen.rs @@ -0,0 +1,88 @@ +//! Whether a listener's owner is owed a wake, and what its accept finds. +//! +//! **A listener is one smoltcp socket that becomes the connection it +//! accepts**, so the port listens only while that socket is in `Listen`: one +//! that left it is handed to its owner or listens again, or the port answers +//! every other peer with a reset for the rest of the boot. +//! +//! **An accept spends the owner's wake whatever it answers, a refusal +//! included, and a wake is owed only for a connection there is room to +//! take.** An owner refused holds no wake, so the connection it left is +//! announced again, and an owner refused for room is not woken until room +//! returns. + +use smoltcp::socket::tcp; + +/// A listener's port, and whether its owner holds a wake it has not spent on +/// an accept. +pub struct Listening { + port: u16, + woken: bool, +} + +/// What an accept finds, handed the pipes `P` its request carried. +#[derive(Debug, PartialEq, Eq)] +pub enum Accept

{ + /// A connection, to hand over on the pipes. + Take(P), + /// A request that carried no pipes, whatever waits. + NoPipes, + /// A connection, and no room to take it. + NoRoom, + /// No connection. + Nothing, +} + +impl Listening { + pub fn new(port: u16) -> Self { + Self { port, woken: false } + } + + pub fn port(&self) -> u16 { + self.port + } + + /// The bytes to write the owner for `socket`: one wake if a connection + /// waits, there is `room` to take it, and the owner holds no wake, and + /// none otherwise. The wake is held from here on, so the caller ends the + /// listener if the owner is not handed it. + pub fn wake(&mut self, socket: &mut tcp::Socket, room: bool) -> &'static [u8] { + let owed = settle(socket, self.port) && room && !self.woken; + self.woken |= owed; + if owed { &[1] } else { &[] } + } + + /// An accept, with `room` for another connection or not, and the pipes + /// its request carried. + pub fn accept

(&mut self, socket: &mut tcp::Socket, room: bool, pipes: Option

) -> Accept

{ + self.woken = false; + match (settle(socket, self.port), room, pipes) { + (_, _, None) => Accept::NoPipes, + (false, _, Some(_)) => Accept::Nothing, + (true, false, Some(_)) => Accept::NoRoom, + (true, true, Some(pipes)) => Accept::Take(pipes), + } + } +} + +/// Puts `socket` back to listening on `port` if its peer reset it before its +/// owner took it, and says whether it holds a connection: a handshake +/// finished, whatever the peer did since. Its FIN included, which can land in +/// the same pass as the handshake's last ACK, so no pass ever sees the socket +/// `Established`. +fn settle(socket: &mut tcp::Socket, port: u16) -> bool { + match socket.state() { + tcp::State::Listen | tcp::State::SynReceived => false, + tcp::State::Established | tcp::State::CloseWait => true, + tcp::State::Closed => { + socket + .listen(port) + .unwrap_or_else(|e| panic!("netd: a closed socket refused to listen on {port}: {e:?}")); + false + } + other => panic!("netd: a listener's socket is {other:?}, which only netd closing or connecting it reaches"), + } +} + +#[cfg(test)] +mod tests; diff --git a/userland/netd/src/listen/tests.rs b/userland/netd/src/listen/tests.rs new file mode 100644 index 00000000000..d8fcb66548f --- /dev/null +++ b/userland/netd/src/listen/tests.rs @@ -0,0 +1,314 @@ +//! A listener's socket on a wire: smoltcp's own `Interface` on an Ethernet +//! device whose far end is played here segment by segment, so what is judged +//! is what the port answers the next peer. + +use super::*; +use std::collections::VecDeque; + +use smoltcp::iface::{Config, Interface, PollResult, SocketHandle, SocketSet}; +use smoltcp::phy::{self, ChecksumCapabilities, Device, DeviceCapabilities, Medium}; +use smoltcp::time::Instant; +use smoltcp::wire::{ + ArpOperation, ArpPacket, ArpRepr, EthernetAddress, EthernetFrame, EthernetProtocol, EthernetRepr, + HardwareAddress, IpAddress, IpCidr, IpProtocol, Ipv4Address, Ipv4Packet, Ipv4Repr, TcpControl, TcpPacket, + TcpRepr, TcpSeqNumber, +}; + +const OUR_MAC: EthernetAddress = EthernetAddress([0x02, 0, 0, 0, 0, 0x01]); +const OURS: Ipv4Address = Ipv4Address::new(10, 0, 2, 15); +const PEER_MAC: EthernetAddress = EthernetAddress([0x02, 0, 0, 0, 0, 0x02]); +const PEER: Ipv4Address = Ipv4Address::new(10, 0, 2, 2); +const PORT: u16 = 22; +/// The peer's initial sequence number, on every connection it opens. +const PEER_ISN: u32 = 1000; + +#[derive(Default)] +struct Wire { + inbound: VecDeque>, + outbound: Vec>, +} + +struct Rx(Vec); +struct Tx<'a>(&'a mut Vec>); + +impl phy::RxToken for Rx { + fn consume R>(self, f: F) -> R { + f(&self.0) + } +} + +impl phy::TxToken for Tx<'_> { + fn consume R>(self, len: usize, f: F) -> R { + let mut frame = vec![0u8; len]; + let result = f(&mut frame); + self.0.push(frame); + result + } +} + +impl Device for Wire { + type RxToken<'a> = Rx; + type TxToken<'a> = Tx<'a>; + + fn receive(&mut self, _: Instant) -> Option<(Rx, Tx<'_>)> { + let frame = self.inbound.pop_front()?; + Some((Rx(frame), Tx(&mut self.outbound))) + } + + fn transmit(&mut self, _: Instant) -> Option> { + Some(Tx(&mut self.outbound)) + } + + fn capabilities(&self) -> DeviceCapabilities { + let mut caps = DeviceCapabilities::default(); + caps.max_transmission_unit = 1514; + caps.medium = Medium::Ethernet; + caps + } +} + +/// One TCP segment our side sent: its flags, sequence number and the peer +/// port it went to. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +struct Sent { + control: TcpControl, + ack: bool, + seq: u32, + to: u16, +} + +struct Net { + iface: Interface, + wire: Wire, + sockets: SocketSet<'static>, + listener: SocketHandle, + listening: Listening, + sent: Vec, +} + +impl Net { + fn new() -> Self { + let mut wire = Wire::default(); + let mut iface = Interface::new(Config::new(HardwareAddress::Ethernet(OUR_MAC)), &mut wire, Instant::from_millis(0)); + iface.update_ip_addrs(|addrs| addrs.push(IpCidr::new(IpAddress::Ipv4(OURS), 24)).unwrap()); + let mut socket = tcp::Socket::new(tcp::SocketBuffer::new(vec![0; 4096]), tcp::SocketBuffer::new(vec![0; 4096])); + socket.listen(PORT).expect("a fresh socket listens"); + let mut sockets = SocketSet::new(Vec::new()); + let listener = sockets.add(socket); + Self { iface, wire, sockets, listener, listening: Listening::new(PORT), sent: Vec::new() } + } + + fn socket(&mut self) -> &mut tcp::Socket<'static> { + self.sockets.get_mut::(self.listener) + } + + /// One pass of netd's loop: everything the wire holds, in one batch, and + /// the far end's ARP answered. + fn pass(&mut self) { + loop { + while self.iface.poll(Instant::from_millis(0), &mut self.wire, &mut self.sockets) != PollResult::None {} + if self.wire.outbound.is_empty() { + return; + } + for frame in std::mem::take(&mut self.wire.outbound) { + self.far_end(&frame); + } + } + } + + fn far_end(&mut self, frame: &[u8]) { + let eth = EthernetFrame::new_checked(frame).expect("the interface sent an Ethernet frame"); + match eth.ethertype() { + EthernetProtocol::Arp => { + let arp = ArpRepr::parse(&ArpPacket::new_checked(eth.payload()).unwrap()).unwrap(); + let ArpRepr::EthernetIpv4 { operation: ArpOperation::Request, .. } = arp else { return }; + let reply = ArpRepr::EthernetIpv4 { + operation: ArpOperation::Reply, + source_hardware_addr: PEER_MAC, + source_protocol_addr: PEER, + target_hardware_addr: OUR_MAC, + target_protocol_addr: OURS, + }; + let mut out = vec![0u8; 14 + reply.buffer_len()]; + let mut e = EthernetFrame::new_unchecked(&mut out); + EthernetRepr { src_addr: PEER_MAC, dst_addr: OUR_MAC, ethertype: EthernetProtocol::Arp }.emit(&mut e); + reply.emit(&mut ArpPacket::new_unchecked(e.payload_mut())); + self.wire.inbound.push_back(out); + } + EthernetProtocol::Ipv4 => { + let ip = Ipv4Packet::new_checked(eth.payload()).unwrap(); + assert_eq!(ip.next_header(), IpProtocol::Tcp, "a listener sends only TCP"); + let tcp = TcpPacket::new_checked(ip.payload()).unwrap(); + let control = match (tcp.syn(), tcp.fin(), tcp.rst()) { + (true, _, _) => TcpControl::Syn, + (_, true, _) => TcpControl::Fin, + (_, _, true) => TcpControl::Rst, + _ => TcpControl::None, + }; + self.sent.push(Sent { control, ack: tcp.ack(), seq: tcp.seq_number().0 as u32, to: tcp.dst_port() }); + } + other => panic!("the interface sent a frame of type {other}"), + } + } + + /// A segment from the peer's `from` port onto the wire, delivered at the + /// next [`Net::pass`]. + fn send(&mut self, from: u16, control: TcpControl, seq: u32, ack: Option) { + let caps = ChecksumCapabilities::default(); + let tcp = TcpRepr { + src_port: from, + dst_port: PORT, + control, + seq_number: TcpSeqNumber(seq as i32), + ack_number: ack.map(|a| TcpSeqNumber(a as i32)), + window_len: 64000, + window_scale: None, + max_seg_size: None, + sack_permitted: false, + sack_ranges: [None, None, None], + timestamp: None, + payload: &[], + }; + let ip = Ipv4Repr { + src_addr: PEER, + dst_addr: OURS, + next_header: IpProtocol::Tcp, + payload_len: tcp.buffer_len(), + hop_limit: 64, + }; + let mut out = vec![0u8; 14 + ip.buffer_len() + ip.payload_len]; + let mut e = EthernetFrame::new_unchecked(&mut out); + EthernetRepr { src_addr: PEER_MAC, dst_addr: OUR_MAC, ethertype: EthernetProtocol::Ipv4 }.emit(&mut e); + let mut packet = Ipv4Packet::new_unchecked(e.payload_mut()); + ip.emit(&mut packet, &caps); + tcp.emit(&mut TcpPacket::new_unchecked(packet.payload_mut()), &IpAddress::Ipv4(PEER), &IpAddress::Ipv4(OURS), &caps); + self.wire.inbound.push_back(out); + } + + /// The peer's SYN from `from`, and the SYN-ACK's sequence number. + fn syn(&mut self, from: u16) -> u32 { + self.send(from, TcpControl::Syn, PEER_ISN, None); + self.pass(); + self.sent + .iter() + .rev() + .find(|s| s.control == TcpControl::Syn && s.ack && s.to == from) + .unwrap_or_else(|| panic!("a SYN from {from} was answered {:?}", self.sent.last())) + .seq + } + + /// The peer's answer to the SYN-ACK and `then`, in one batch. + fn ack_and(&mut self, from: u16, our_isn: u32, then: TcpControl) { + self.send(from, TcpControl::None, PEER_ISN + 1, Some(our_isn + 1)); + self.send(from, then, PEER_ISN + 1, Some(our_isn + 1)); + self.pass(); + } + + /// Whether the port answers a new peer's SYN with a SYN-ACK: it listens. + fn listens(&mut self, from: u16) -> bool { + self.send(from, TcpControl::Syn, PEER_ISN, None); + self.pass(); + let answer = *self.sent.iter().rev().find(|s| s.to == from).expect("a SYN is answered"); + answer.control == TcpControl::Syn + } + + /// Whether the owner is woken on a pass with `room` or without. + fn wakes(&mut self, room: bool) -> bool { + match self.listening.wake(self.sockets.get_mut::(self.listener), room) { + [] => false, + [1] => true, + other => panic!("a wake of {other:?}"), + } + } + + fn accept(&mut self, room: bool, pipes: bool) -> Accept<()> { + self.listening.accept(self.sockets.get_mut::(self.listener), room, pipes.then_some(())) + } +} + +#[test] +fn a_finished_handshake_is_owed_one_wake() { + let mut net = Net::new(); + let isn = net.syn(5001); + assert!(!net.wakes(true), "a SYN alone was taken for a connection"); + net.send(5001, TcpControl::None, PEER_ISN + 1, Some(isn + 1)); + net.pass(); + assert!(net.wakes(true)); + assert!(!net.wakes(true), "a connection its owner holds a wake for was announced twice"); + assert_eq!(net.accept(true, true), Accept::Take(())); +} + +/// **A peer that sends its FIN with the handshake's last ACK is still a +/// connection.** Both land in one pass, so the socket goes from `SynReceived` +/// to `CloseWait` with no pass ever seeing it `Established`. +#[test] +fn a_peer_that_closes_with_its_last_ack_is_a_connection() { + let mut net = Net::new(); + let isn = net.syn(5001); + net.ack_and(5001, isn, TcpControl::Fin); + assert_eq!(net.socket().state(), tcp::State::CloseWait, "the premise: both in one pass"); + assert!(net.wakes(true), "a connection the peer half-closed was never announced"); + assert_eq!(net.accept(true, true), Accept::Take(()), "a connection the peer half-closed was not handed over"); +} + +/// A peer that resets before its owner took it frees the port: the next peer +/// is answered a SYN-ACK and not a reset. +#[test] +fn a_peer_that_resets_before_it_is_taken_frees_the_port() { + let mut net = Net::new(); + let isn = net.syn(5001); + net.ack_and(5001, isn, TcpControl::Rst); + assert_eq!(net.socket().state(), tcp::State::Closed, "the premise: both in one pass"); + assert!(!net.wakes(false)); + assert!(net.listens(5002), "the port answered the next peer {:?}", net.sent.last()); +} + +/// A wake written for a connection its peer then reset is spent by the +/// accept that finds nothing, and the next connection is announced. +#[test] +fn a_wake_spent_on_a_reset_connection_announces_the_next() { + let mut net = Net::new(); + let isn = net.syn(5001); + net.send(5001, TcpControl::None, PEER_ISN + 1, Some(isn + 1)); + net.pass(); + assert!(net.wakes(true)); + net.send(5001, TcpControl::Rst, PEER_ISN + 1, Some(isn + 1)); + net.pass(); + assert!(!net.wakes(true), "a reset connection was announced"); + let isn = net.syn(5002); + assert_eq!(net.accept(true, true), Accept::Nothing, "an accept took a connection its peer had reset"); + net.send(5002, TcpControl::None, PEER_ISN + 1, Some(isn + 1)); + net.pass(); + assert!(net.wakes(true), "the connection after a reset one was never announced"); +} + +/// An accept that handed netd no pipes spends the owner's wake, and the +/// connection it left is announced again at once. +#[test] +fn an_accept_refused_for_its_pipes_is_woken_again() { + let mut net = Net::new(); + let isn = net.syn(5001); + net.send(5001, TcpControl::None, PEER_ISN + 1, Some(isn + 1)); + net.pass(); + assert!(net.wakes(true)); + assert_eq!(net.accept(true, false), Accept::NoPipes); + assert!(net.wakes(true), "an owner refused for its pipes was never woken again"); + assert_eq!(net.accept(true, true), Accept::Take(())); +} + +/// An accept refused for room spends the owner's wake, and the connection it +/// left is announced again once room returns, and not before. +#[test] +fn an_accept_refused_for_room_is_woken_again_when_room_returns() { + let mut net = Net::new(); + assert_eq!(net.accept(false, true), Accept::Nothing, "an accept with nothing waiting was refused for room"); + let isn = net.syn(5001); + net.send(5001, TcpControl::None, PEER_ISN + 1, Some(isn + 1)); + net.pass(); + assert!(!net.wakes(false), "the owner was woken for a connection there is no room to take"); + assert!(net.wakes(true)); + assert_eq!(net.accept(false, true), Accept::NoRoom); + assert!(!net.wakes(false), "an owner refused for room was woken again with room still gone"); + assert!(net.wakes(true), "an owner refused for room was never woken again"); + assert_eq!(net.accept(true, true), Accept::Take(())); +} diff --git a/userland/netd/src/main.rs b/userland/netd/src/main.rs index b5a5682bd7a..25b7689578b 100644 --- a/userland/netd/src/main.rs +++ b/userland/netd/src/main.rs @@ -9,6 +9,7 @@ use toyos::say; mod device; mod dhcp; mod i219; +mod listen; mod mdns; mod report; mod resolve; @@ -399,7 +400,7 @@ impl PipedConnection { struct PipedListener { handle: SocketHandle, notify_write: Pipe, - notified: bool, + listening: listen::Listening, } struct PendingPipedConnect { @@ -1223,7 +1224,7 @@ impl NetDaemon { self.piped_listeners.insert(socket_id, PipedListener { handle, notify_write, - notified: false, + listening: listen::Listening::new(port), }); msg.client.result(&TcpBindResponse { @@ -1238,39 +1239,38 @@ impl NetDaemon { msg.client.error(ERR_INVALID_INPUT); return; }; - if !self.piped_room() { - say!( - "netd: refusing accept, {} piped connections already (max {})", - self.piped_live(), - self.max_piped_connections, - ); - msg.client.error(ERR_RESOURCE_EXHAUSTED); - return; - } - let Some(pipes) = DataPipes::take(&msg.client) else { - msg.client.error(ERR_INVALID_INPUT); - return; - }; - let Some(listener) = self.piped_listeners.get(&req.socket_id) else { + let (room, pipes) = (self.piped_room(), DataPipes::take(&msg.client)); + let Some(listener) = self.piped_listeners.get_mut(&req.socket_id) else { msg.client.error(ERR_NOT_CONNECTED); return; }; + let (old_handle, local_port) = (listener.handle, listener.listening.port()); + let pipes = match listener.listening.accept(socket_set.get_mut::(old_handle), room, pipes) { + listen::Accept::Take(pipes) => pipes, + listen::Accept::NoPipes => { + msg.client.error(ERR_INVALID_INPUT); + return; + } + listen::Accept::NoRoom => { + say!( + "netd: refusing accept, {} piped connections already (max {})", + self.piped_live(), + self.max_piped_connections, + ); + msg.client.error(ERR_RESOURCE_EXHAUSTED); + return; + } + listen::Accept::Nothing => { + msg.client.error(ERR_NOT_CONNECTED); + return; + } + }; - let socket = socket_set.get_mut::(listener.handle); - // Not `is_active`: that is already true in SynReceived, where - // `remote_endpoint()` is still None. - if socket.state() != tcp::State::Established { - msg.client.error(ERR_NOT_CONNECTED); - return; - } - - let remote = socket.remote_endpoint().unwrap(); - let local_port = socket.local_endpoint().unwrap().port; + let remote = socket_set.get_mut::(old_handle).remote_endpoint().unwrap(); let remote_addr = match remote.addr { IpAddress::Ipv4(a) => a.octets(), }; - let old_handle = listener.handle; let stream_id = self.alloc_id(); self.sockets.insert(stream_id, SocketKind::TcpStream(old_handle)); @@ -1288,10 +1288,10 @@ impl NetDaemon { let new_handle = socket_set.add(new_listener); self.sockets.insert(req.socket_id, SocketKind::TcpListener(new_handle)); - if let Some(pl) = self.piped_listeners.get_mut(&req.socket_id) { - pl.handle = new_handle; - pl.notified = false; - } + self.piped_listeners + .get_mut(&req.socket_id) + .expect("looked up above; nothing between there and here removes a piped_listeners entry") + .handle = new_handle; msg.client.result(&TcpAcceptPipedResponse { socket_id: stream_id, @@ -1429,18 +1429,14 @@ impl NetDaemon { /// is what tells it instead — its notify pipe reads EOF. A full pipe is /// that refusal too: it is an owner that has left a whole pipe of wakes /// unread. - fn serve_piped_listeners(&mut self, socket_set: &mut SocketSet<'_>) { + fn serve_piped_listeners(&mut self, socket_set: &mut SocketSet<'_>, room: bool) { use toyos_abi::syscall::SyscallError; let mut dead = Vec::new(); for (&socket_id, listener) in &mut self.piped_listeners { - let socket = socket_set.get_mut::(listener.handle); - // Not `is_active`: that is already true in SynReceived, before the - // three-way handshake completes. - let owed = socket.state() == tcp::State::Established && !listener.notified; - let wake: &[u8] = if owed { &[1] } else { &[] }; + let wake = listener.listening.wake(socket_set.get_mut::(listener.handle), room); match toyos_abi::syscall::write_nonblock(listener.notify_write.as_handle(), wake) { - Ok(_) => listener.notified |= owed, - Err(SyscallError::WouldBlock) if !owed => {} + Ok(_) => {} + Err(SyscallError::WouldBlock) if wake.is_empty() => {} // Its owner has gone, which is the ordinary end of a listener. Err(SyscallError::Gone) => dead.push(socket_id), Err(e) => { @@ -1461,8 +1457,9 @@ impl NetDaemon { } } - /// Process pending async operations (UDP recvs, lookups, piped connects). - fn process_pending(&mut self, socket_set: &mut SocketSet<'_>) { + /// Process pending async operations (UDP recvs, lookups, piped connects), + /// and say whether there is room for another piped connection after them. + fn process_pending(&mut self, socket_set: &mut SocketSet<'_>) -> bool { let now = Instant::now(); for pr in std::mem::take(&mut self.pending_udp_recvs) { @@ -1527,6 +1524,7 @@ impl NetDaemon { } i += 1; } + self.piped_room() } } @@ -1739,9 +1737,8 @@ fn main() { daemon.bridge_piped(&mut socket_set); - daemon.serve_piped_listeners(&mut socket_set); - - daemon.process_pending(&mut socket_set); + let room = daemon.process_pending(&mut socket_set); + daemon.serve_piped_listeners(&mut socket_set, room); // smoltcp's own next deadline — a retransmit, a persist probe, a // delayed ACK — and zero when it has a frame to send now. A piped