diff --git a/Cargo.lock b/Cargo.lock index 7551623381..7910f62e32 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -3211,12 +3211,6 @@ version = "0.6.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "dae608c151f68243f2b000364e1f7b186d9c29845f7d2d85bd31b9ad77ad552b" -[[package]] -name = "managed" -version = "0.8.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0ca88d725a0a943b096803bd34e73a4437208b6077654cc4ecb2947a5f91618d" - [[package]] name = "matchers" version = "0.2.0" @@ -3362,14 +3356,20 @@ dependencies = [ name = "netstack" version = "0.0.0" dependencies = [ - "smoltcp", "toyos", "toyos-abi", "toyos-device-memory", + "toyos-dhcp", "toyos-dns", "toyos-i219", "toyos-inspect", "toyos-mdns", + "toyos-net-ip", + "toyos-net-node", + "toyos-net-shard", + "toyos-net-tcp", + "toyos-net-udp", + "toyos-net-wire", "toyos-tco", "toyos-virtio", ] @@ -5215,20 +5215,6 @@ dependencies = [ "serde", ] -[[package]] -name = "smoltcp" -version = "0.12.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dad095989c1533c1c266d9b1e8d70a1329dd3723c3edac6d03bbd67e7bf6f4bb" -dependencies = [ - "bitflags 1.3.2", - "byteorder", - "cfg-if", - "defmt 0.3.100", - "heapless", - "managed", -] - [[package]] name = "snake" version = "0.1.0" diff --git a/issues/a-connect-between-two-accepts-is-reset.md b/issues/a-connect-between-two-accepts-is-reset.md index e71d0d5c06..dc55970315 100644 --- a/issues/a-connect-between-two-accepts-is-reset.md +++ b/issues/a-connect-between-two-accepts-is-reset.md @@ -32,5 +32,22 @@ metal harness's swapping boots, which that row was the one user of, and `issues/the-host-cannot-reach-the-t14-while-it-runs-toyos.md` records the commit that restores those. +netstack's listener is the node's now, whose queue is [tcp]'s: a connect +that arrives while another waits to be accepted is queued +(`a_connect_between_two_accepts_is_queued_not_reset`, +`userland/netstack/node/tests/listeners.rs`), and in a guest two host peers +that dial before any accept are both accepted (`netstack_streams`, which on +smoltcp ended `the listener was woken for 1 of two peers`). What is left of the +exit is `lan_swap`. + +QEMU's own forward is a third place such a reset can come from, and it was +not ruled out of the reading above: its listener queues one connection, and a +host dial that arrives while one is queued is reset by QEMU with no SYN sent to +the guest. Measured while `netstack_streams` was written, with the host at a +load of 30 and twelve guests beside it: 3 of 8 runs red, the host's second +dial ending `Connection reset by peer (os error 54)` and the frames recorded +on the guest's card holding one SYN; none of 5 once the harness dialled its +second peer only after QEMU's table showed the first carried. + **Exit**: a listener that queues a connect arriving between two accepts, and `lan_swap` restored and green. diff --git a/issues/a-handshake-nobody-finishes-holds-a-listeners-port-shut.md b/issues/a-handshake-nobody-finishes-holds-a-listeners-port-shut.md deleted file mode 100644 index 92dd74dbdf..0000000000 --- a/issues/a-handshake-nobody-finishes-holds-a-listeners-port-shut.md +++ /dev/null @@ -1,31 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-09-27 ---- - -# A handshake nobody finishes holds a listener's port shut - -netd's listener is one smoltcp socket, and the port listens only while that -socket is in `Listen` (`userland/netd/src/listen.rs`). A SYN whose sender never -answers the SYN-ACK (a peer gone, a spoofed source) leaves it in `SynReceived`, -and smoltcp 0.12 retransmits the SYN-ACK without end. - -Measured on smoltcp's interface over a wire played by hand, as -`userland/netd/src/listen/tests.rs` plays it: after one SYN and 600 s of -silence the socket was still `SynReceived`, having sent 72 SYN-ACKs, and -another peer's SYN in that state was answered with a reset. With -`set_timeout(10 s)` the socket went `Closed` at 10 s, which `listen::settle` -turns back into `Listen`. - -So one packet shuts sshd's port for the rest of the boot. Nothing has chosen a -bound on a listener's half-open handshake, and the timeout smoltcp offers also -bounds the connection the socket becomes, so it would have to come off at the -hand-over. - -**Owner**: whoever holds `issues/toyos-has-its-own-network-stack.md`, -which carries this and `issues/a-connect-between-two-accepts-is-reset.md`. - -**Exit**: a listener's half-open handshake let go within a bound, and a test -that sends one SYN and nothing more, then finds the port answering the next -peer with a SYN-ACK. diff --git a/issues/a-handshake-reset-before-it-ends-hands-its-option-to-the-next-connection.md b/issues/a-handshake-reset-before-it-ends-hands-its-option-to-the-next-connection.md deleted file mode 100644 index 17e83abab5..0000000000 --- a/issues/a-handshake-reset-before-it-ends-hands-its-option-to-the-next-connection.md +++ /dev/null @@ -1,48 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-10-09 ---- - -# A handshake reset before it ends hands its option to the next connection - -netstack on smoltcp gives a connection the `TCP_NODELAY` its listener held -when its SYN arrived (`userland/netstack/src/listen.rs`), which is a host's -rule (`userland/netstack/node/tests/host.rs`), except for the connection that -follows a handshake its peer reset before it ended. - -Read from the two sources, not run. A listener there is one smoltcp socket. A -set of the listener's option is written to that socket only while it is in -`Listen` (`Listening::set_nodelay`); once a SYN has arrived the set is kept in -`Listening` for the socket that listens next, which `Listening::open` gives it -after an accept, or after a reset that left the socket `Closed` -(`Listening::settle`). smoltcp 0.12.0 puts a socket that takes an RST in -`SynReceived` straight back to `Listen` (`src/socket/tcp.rs`, the -`(State::SynReceived, TcpControl::Rst)` arm: `self.tuple = None; -self.set_state(State::Listen)`), with the Nagle switch it had. No pass of -netstack sees that socket `Closed`, so nothing writes the listener's option to -it again. - -The sequence: the listener holds the option, a SYN arrives, the owner clears -the option, the peer resets, as a SYN scan does. The socket listens again with -Nagle's algorithm off, and the next connection begins with `TCP_NODELAY` at a -listener that held none when its SYN arrived; its accept answers the option -on. The mirror gives a connection without the option at a listener whose -`getsockopt` reads 1. A set that reaches the socket while it listens again, or -the accept that replaces it, ends the difference: one connection a reset -handshake is affected, if the owner changed the option during that handshake. - -No test holds it, and none is to be built: the owner's ruling is that no new -test is built on smoltcp. The node, which replaces this stack, has the rule -for this case: [tcp] keeps the options with the listener and copies them to a -connection when its SYN arrives (LS-10), a reset handshake leaves nothing -behind, and -`a_handshake_reset_before_it_ends_leaves_the_next_connection_its_listeners_option` -(`userland/netstack/node/tests/listeners.rs`) plays the sequence both ways -round and reads the next connection's option from the accept's answer. - -**Exit condition**: netd runs on the node and `userland/netstack/src/listen.rs` -and smoltcp are deleted, with that test of the node green. - -**Owner**: whoever holds `issues/toyos-has-its-own-network-stack.md`, at the -move. diff --git a/issues/a-lease-kept-across-a-link-flap-is-not-verified-until-its-renewal.md b/issues/a-lease-kept-across-a-link-flap-is-not-verified-until-its-renewal.md deleted file mode 100644 index 8ec4bfb9b6..0000000000 --- a/issues/a-lease-kept-across-a-link-flap-is-not-verified-until-its-renewal.md +++ /dev/null @@ -1,33 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-09-23 ---- - -# A lease kept across a link flap is not verified until its renewal - -netd keeps a bound DHCP lease when the link goes down and comes back -(`userland/netd/src/main.rs`'s main loop, `dhcp::restart`'s header): the -client restarts discovery only on a link that comes up with no lease held. The -client offers no way to renew early, so the lease is next checked against a -server at its own renewal time, T1 (RFC 2131 §4.4.5), which is half the lease -by default. - -Until then a cable moved to another network keeps the old address, route and -resolvers. Every frame the machine sends in that window goes out under an -address the new network never leased it. The compromise was chosen over the -alternative the client offers, a restart, which gives the address up before it -asks again and so takes a machine whose cable only flapped off its network for -a whole exchange. - -## Owner - -The successor of the I219 PHY branch (PR #453) on the LAN track, "The LAN -reaches a router, and is not yet production grade" (stage 4, the stack). - -## What would close it - -The DHCP client verifies a kept lease when the link comes back: a renew-now -request, or RFC 2131 §3.2's INIT-REBOOT (a DHCPREQUEST for the address it -holds), with the lease kept while the answer is outstanding and given up on a -DHCPNAK. diff --git a/issues/a-listeners-tcp-nodelay-has-no-setter-in-std-or-the-forks-and-a-c-program-cannot-order-the-waiting-case.md b/issues/a-listeners-tcp-nodelay-has-no-setter-in-std-or-the-forks-and-a-c-program-cannot-order-the-waiting-case.md index b99e8c9cc8..4cf0c83ce8 100644 --- a/issues/a-listeners-tcp-nodelay-has-no-setter-in-std-or-the-forks-and-a-c-program-cannot-order-the-waiting-case.md +++ b/issues/a-listeners-tcp-nodelay-has-no-setter-in-std-or-the-forks-and-a-c-program-cannot-order-the-waiting-case.md @@ -24,10 +24,8 @@ loopback delivers in. A red there prints the host's name and both answers. The pipe ABI carries both halves (`toyos/src/net.rs`): a bind's request carries its listener's options and `MsgType::TcpListenerSetOption` sets one afterwards, and an accept's answer says what its connection holds -(`TcpOptions`). netstack answers the rule for both -(`userland/netstack/src/listen.rs`), but for the connection after a reset -handshake -(`issues/a-handshake-reset-before-it-ends-hands-its-option-to-the-next-connection.md`), +(`TcpOptions`). netstack answers the rule for both by the node's calls +(`userland/netstack/src/serve.rs`, `userland/netstack/node/src/listeners.rs`), and the `libc_sockets` guest test reads it: `nodelay_accepted` by `toyos::net`, for a connection that waited when the option was set, its wake read first, and for the next one dialled; `tests/netcase/nodelay_kept.c` by libc's `setsockopt` and diff --git a/issues/a-netstack-client-cannot-tell-a-reset-from-the-peers-fin.md b/issues/a-netstack-client-cannot-tell-a-reset-from-the-peers-fin.md deleted file mode 100644 index b1513cae8e..0000000000 --- a/issues/a-netstack-client-cannot-tell-a-reset-from-the-peers-fin.md +++ /dev/null @@ -1,58 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-09-26 ---- - -# A netstack client cannot tell a reset connection from the peer's FIN - -A stream reaches its client as two pipes, and the receive pipe's end says only -that it ended. std (`ended`, `sdk/std/sys/net/connection.rs`) reads the kind -off the send pipe: one with no reader behind a receive pipe at its end is a -failure. libc reads no kind: its `recv` answers every end 0 and every refused -read `EIO` (`userland/libc/src/socket.rs`), so a C client reads a reset as the -peer's FIN. - -On the node that order holds: a failure lets the send pipe go before the -receive pipe, and an orderly end keeps a send pipe whose last byte and FIN are -queued until its writer leaves (`userland/netstack/node/src/streams.rs`'s -header; its host tests in `userland/netstack/node/tests/streams.rs`, each red -with the order broken). - -What `main` ships is netstack on smoltcp, whose `bridge_piped` -(`userland/netstack/src/main.rs`) closes the receive pipe and then, in the same -pass, the send pipe of a connection that is no longer open, whether it failed -or ended in order. A client that reads the end between the two reads a -failure as a FIN, and one that reads it after both reads an orderly end as a -failure. A client that shut its sending half down before the peer's FIN is -the second: its connection is done in both directions at that FIN. Measured -in a guest on `main`, std, against the host kernel's TCP behind QEMU's user -network, the same program on the harness host's TCP (macOS) as the oracle: - -- a client that shut its sending half down with nothing pending, then read the - peer's four bytes and its FIN, read the four and then `ConnectionReset`, in - each of two runs, where the host read `Ok(0)`; a C client doing the same - read `recv` 0 with `main`'s libc, which reads every end so, and - `ECONNRESET` with a libc that reads the kind as std does; -- a client that read the peer's FIN and then shut its sending half down read - its next read as `Ok(0)` in two runs and as `ConnectionReset` in two others, - where the host read `Ok(0)`: the race between the two closes; -- a reset mid-stream, a reset after the client's half-close and a FIN before - the client's write read as the host read them, in each of four runs. - -**Exit condition**: netstack runs on the node, and on `tests/netcase`: - -- a guest std test whose client shuts its sending half down reads its peer's - bytes and then the end as an end, one whose peer resets a stream reads the - reset as a reset, and each line it prints is the host kernel's for the same - program against the same peer; -- libc reads an end's kind as std does, and a guest C case reads `recv` 0 at - the peer's FIN after `shutdown(SHUT_WR)`, `ECONNRESET` on a reset mid-stream - and on one after `SHUT_WR`, `send` `EPIPE` after `SHUT_WR` and `recv` 0 - after `SHUT_RD`; and it is red with `recv`'s probe of the send pipe replaced - by a plain 0, with `shutdown` not marking the sending half shut, and with - `recv` not answering 0 after `SHUT_RD`. The libc change and its host test - were written for the node and are posted on pull request #803 for the move - to carry with that case. - -**Owner**: whoever holds `issues/toyos-has-its-own-network-stack.md`. diff --git a/issues/a-received-handle-has-no-knowable-type.md b/issues/a-received-handle-has-no-knowable-type.md index 6859121aa5..ad91cb0255 100644 --- a/issues/a-received-handle-has-no-knowable-type.md +++ b/issues/a-received-handle-has-no-knowable-type.md @@ -33,9 +33,12 @@ typed can be ended by whoever sent it.** The sender needs nothing but - A window client receives its buffer the same way (`userland/toyos-window/src/lib.rs`). - blockd maps the region a client sends with its open (`Region::adopt` in `userland/blockd/src/region.rs`), so a *client* holding its connector ends it. -- netd adopts as pipes the two handles a client sends with a piped socket - (`DataPipes::take`) and the one with a piped bind (`handle_tcp_bind_piped`), - both in `userland/netd/src/main.rs`: a *client* ends it the same way. +- netstack adopts as pipes the two handles a client sends with a piped socket + (`data_pipes`) and the one with a piped bind (`Sockets::listen`), both in + `userland/netstack/src/serve.rs`: a handle that is no pipe end answers a + refusal the node ends that client's stream or listener for + (`userland/netstack/src/pipes.rs`), and a datagram socket's are netstack's + own to read and write. Nothing in the tree is hostile today, so nothing fails. The property the architecture claims — that a process cannot be harmed by what it was not given — diff --git a/issues/a-shutdown-of-the-sending-half-drops-what-the-send-pipe-still-holds.md b/issues/a-shutdown-of-the-sending-half-drops-what-the-send-pipe-still-holds.md deleted file mode 100644 index e592ad2b78..0000000000 --- a/issues/a-shutdown-of-the-sending-half-drops-what-the-send-pipe-still-holds.md +++ /dev/null @@ -1,37 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-10-08 ---- - -# A shutdown of the sending half drops what the send pipe still holds - -`handle_tcp_shutdown` (`userland/netstack/src/main.rs`) calls `socket.close()` -on the pass that reads the request. A client's bytes reach netstack through -its send pipe and its request through its connection, and nothing orders the -two: bytes the client wrote before it asked may still be in the pipe. -`bridge_piped` reads the pipe only while `send_room` holds, which is false -from `FIN-WAIT-1` on, so those bytes are never sent and the peer reads a -stream that ends short with a clean FIN. - -Dropping the write end instead is the path that works: the bridge reads the -pipe to its end and closes the socket after the last byte. - -Measured in a guest on `main`, against the host kernel's TCP behind QEMU's -user network, with a peer that echoes what it read once the client's FIN -arrives: a client that wrote 1,048,576 bytes and shut its sending half down -at once read 65,536 bytes back as sent in two runs and 65,535 in two others, -where the same program on a host's TCP reads all 1,048,576 and then the end. -What it read after them, `ConnectionReset`, is not this defect but -`issues/a-netstack-client-cannot-tell-a-reset-from-the-peers-fin.md`'s: a -client that shut its sending half down with nothing pending reads it too. -`netstack_socket_churn` -(`tests/toyos-rust-tests/src/bin/netstack_socket_churn.rs`) writes into the -pipe after the shutdown, which is the client's own error and not this. - -**Exit condition**: a shutdown of the sending half closes the socket after -the bytes the pipe held when the request was read, and a guest test on -`tests/netcase` whose client writes, shuts down at once and reads the peer's -echo sees every byte it wrote. - -**Owner**: whoever holds `issues/toyos-has-its-own-network-stack.md`. diff --git a/issues/a-stream-its-peer-reset-refuses-the-option-requests-a-host-answers.md b/issues/a-stream-its-peer-reset-refuses-the-option-requests-a-host-answers.md index 34b1520236..91c4fa4772 100644 --- a/issues/a-stream-its-peer-reset-refuses-the-option-requests-a-host-answers.md +++ b/issues/a-stream-its-peer-reset-refuses-the-option-requests-a-host-answers.md @@ -6,12 +6,12 @@ opened: 2026-10-08 # A stream its peer reset refuses the option requests a host answers -netstack lets a piped connection's socket and id go once the wire is finished -and it has closed both of its pipe ends (`bridge_piped`, -`userland/netstack/src/main.rs`). A peer's reset does both at once, so a -client that still holds the stream names an id netstack no longer has, and -`handle_tcp_set_option` answers it `ERR_NOT_CONNECTED`. Before the socket left -with its bridge it answered from the kept socket. +The node lets a stream go once it holds neither of its pipe ends +(`userland/netstack/node/src/streams.rs`), and netstack's id for it goes with +it (`userland/netstack/src/serve.rs`, `Sockets::settle`). A peer's reset ends +both pipes at once, so a client that still holds the stream names an id +netstack no longer has, and `Sockets::set_option` answers it +`ERR_NOT_CONNECTED`. What a host answers `std::net::TcpStream` on a stream whose peer reset it, after the read that reported the reset: diff --git a/issues/a-streams-failure-reaches-its-client-as-a-reset-whatever-it-was.md b/issues/a-streams-failure-reaches-its-client-as-a-reset-whatever-it-was.md index 0e177c786a..f3e91e7aa9 100644 --- a/issues/a-streams-failure-reaches-its-client-as-a-reset-whatever-it-was.md +++ b/issues/a-streams-failure-reaches-its-client-as-a-reset-whatever-it-was.md @@ -12,9 +12,8 @@ one bit: the peer's FIN, or a failure. The node knows which failure (`toyos_net_tcp::Failure`: a reset, R2's timeout, an ICMP error), and std answers every one `ConnectionReset`, where Linux answers a connection [tcp] gave up on `ETIMEDOUT` and one an ICMP error ended `EHOSTUNREACH` or -`ENETUNREACH`. libc reads no failure yet -(`issues/a-netstack-client-cannot-tell-a-reset-from-the-peers-fin.md`), and -the change the move carries for it answers every one `ECONNRESET`. +`ENETUNREACH`. libc answers every one `ECONNRESET` +(`userland/libc/src/streamend.rs`). A pipe's end carries no word, and every other carrier costs what this one does not: a third pipe or a kept request connection is a 2 MiB page each diff --git a/issues/libc-answers-epipe-and-raises-no-sigpipe.md b/issues/libc-answers-epipe-and-raises-no-sigpipe.md index 139b02096d..fa97758331 100644 --- a/issues/libc-answers-epipe-and-raises-no-sigpipe.md +++ b/issues/libc-answers-epipe-and-raises-no-sigpipe.md @@ -16,10 +16,8 @@ and none of those `issues/a-childs-end-is-an-event-and-a-parent-takes-its-children-down.md` has it imitate is `SIGPIPE`. So every write behaves as if `MSG_NOSIGNAL` were given, and a C program that relies on the default action to stop writing into a -closed pipe runs on, and stops only if it reads the error. `send` on a stream answers `EIO` for every refusal today; the libc -change the move carries for -`issues/a-netstack-client-cannot-tell-a-reset-from-the-peers-fin.md` answers -`EPIPE` after `shutdown(SHUT_WR)`, and raises nothing either. +closed pipe runs on, and stops only if it reads the error. `send` on a stream answers `EPIPE` after +`shutdown(SHUT_WR)` (`userland/libc/src/socket.rs`), and raises nothing either. Whether libc raises `SIGPIPE` on these, or states the departure as it states others, is a decision of libc's design that nobody has made. diff --git a/issues/most-t14-leases-land-one-dhcp-retry-late.md b/issues/most-t14-leases-land-one-dhcp-retry-late.md index d550b67ae4..f28c21d26d 100644 --- a/issues/most-t14-leases-land-one-dhcp-retry-late.md +++ b/issues/most-t14-leases-land-one-dhcp-retry-late.md @@ -26,6 +26,16 @@ The network track owns it: netd leaves smoltcp in stage 5 of `issues/the-loader-does-only-what-must-precede-the-handover.md` waits on it. +netstack runs `toyos-dhcp` now (`userland/netstack/node`): an unanswered +DISCOVER is sent again on RFC 2131 §4.1's schedule, and a link that comes up +with no lease starts the exchange over +(`a_link_that_returns_with_no_lease_starts_discovery_over_at_once`, +`userland/netstack/node/tests/lease.rs`). Every reading above is of the +smoltcp client. On this one, four T14 boots of pull request #801's +measurement branches took their leases 3,419, 7,206, 8,172 and 8,397 ms after +netstack came up; netstack logs no DISCOVER and no OFFER, so what became of +the first DISCOVER of each is unread. + **Exit**: netd logs each DISCOVER it sends and each OFFER it receives, and a T14 boot's log shows what became of the DISCOVER sent as the link came up; on every boot of a T14 run an unanswered DISCOVER is sent again within RFC 2131 diff --git a/issues/netstack-cuts-a-departed-clients-unsent-tail-at-the-ceiling.md b/issues/netstack-cuts-a-departed-clients-unsent-tail-at-the-ceiling.md deleted file mode 100644 index 5e125bf6c5..0000000000 --- a/issues/netstack-cuts-a-departed-clients-unsent-tail-at-the-ceiling.md +++ /dev/null @@ -1,46 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-10-08 ---- - -# netstack cuts a departed client's unsent tail at the ceiling - -A piped connection whose client holds no end of a direction still open -(`userland/netstack/src/main.rs`, `clientless`) has `OWNERLESS_LIFE`, 100 -seconds from the pass that found it so, to finish on the wire (`ownerless`). -That is every client that is gone, and one that is not: a client that let go -of its send end and still holds the read end of a receive pipe netstack closed -at the peer's FIN is timed and cut the same way, alive. The bound is absolute: at the -ceiling the socket is reset whatever it still owes its peer. A program that -writes and exits is the ordinary case of a client that is gone, so when its -peer takes longer than 100 seconds to acknowledge what was written, the peer's -stream ends in a reset with the tail undelivered: whatever is left of the -64 KiB send buffer, and whatever the client's send pipe still held, up to the -pipe's 2 MiB. - -The cut is loud at both ends: netstack logs it and the peer reads a reset, not -a short stream. Nobody is left to tell on the client's side. - -Hosts do otherwise for an orphaned connection with data still to send: it is -cut on no progress or under orphan pressure and not on a clock from the close. -From knowledge of Linux's `tcp_orphan_retries` and `tcp_max_orphans`, not -measured here. - -The bound is absolute because the slot table has no per-client share -(`issues/netstack-lookup-slots-have-no-per-client-share.md` records the same of -the lookups): a peer that acknowledges a byte at a time would hold a slot of a -client that is gone for as long as it liked, and -`max_piped_connections` such peers hold every slot. - -Read from the code, not measured: the bench rows that reach the ceiling -(`userland/netstack/src/listen/tests.rs`) cut a connection whose peer -acknowledges nothing. - -**Exit condition**: a connection with no client whose peer keeps -acknowledging new bytes is kept while it makes progress, under a bound on how -many such connections one peer or one departed client may hold; and a bench -row whose peer acknowledges one segment a second past the ceiling sees the -whole tail and a FIN. - -**Owner**: whoever holds `issues/toyos-has-its-own-network-stack.md`. diff --git a/issues/netstack-datagram-sockets-and-listeners-have-no-bound.md b/issues/netstack-datagram-sockets-and-listeners-have-no-bound.md deleted file mode 100644 index a1cc23c784..0000000000 --- a/issues/netstack-datagram-sockets-and-listeners-have-no-bound.md +++ /dev/null @@ -1,41 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-10-08 ---- - -# netstack's datagram sockets and listeners have no bound - -`max_piped_connections` (`userland/netstack/src/main.rs`) bounds piped TCP -connections and pending connects, and `resolve::MAX_LOOKUPS` the lookups in -flight. `handle_udp_bind` and `handle_tcp_bind_piped` check neither and -nothing else: every bind adds a socket with two 64 KiB buffers and holds the -client's 2 MiB pipes with it, two for a UDP socket and one for a listener. One holder of the `netstack` -connector that binds in a loop takes netstack's memory, and every dynamic UDP -port, from every other program. No bound is per client either: -`issues/netstack-lookup-slots-have-no-per-client-share.md` records that for -the two that exist. - -Read from the code, not measured. - -On the node (`userland/netstack/node/src/places.rs`) a datagram socket and a -listener each hold a place, and a bind or a listen with none left is refused -with nothing made: `a_datagram_socket_holds_a_place_and_a_bind_without_one_makes_nothing` -and `a_listener_holds_a_place_and_a_listen_without_one_makes_nothing` -(`userland/netstack/node/tests/listeners.rs`). The sockets of the node's own, -the responder's for the machine's name and each query's of a lookup, hold no -place: a lookup is bounded by `toyos_dns::MAX_LOOKUPS` and its rounds, so -clients at the bound refuse no lookup -(`a_query_with_no_port_to_leave_from_ends_its_lookup_by_name`, -`userland/netstack/node/tests/resolve.rs`). What is left: netstack as it -ships is the code above until it runs on the node; the node's places are one -number for every client, so its tests bind past the bound and see the refusal -but have no second client whose bind is answered, which the track records -with its own exit; and the word the pipe ABI answers a refused listen in is -the move's to map. - -**Exit condition**: a bind past a stated bound is refused -`ERR_RESOURCE_EXHAUSTED` with nothing made, and a test binds past it and sees -the refusal and another client's bind answered. - -**Owner**: whoever holds `issues/toyos-has-its-own-network-stack.md`. diff --git a/issues/netstack-holds-a-pending-request-for-a-client-that-left.md b/issues/netstack-holds-a-pending-request-for-a-client-that-left.md index 5b672a07b0..2e7ad50794 100644 --- a/issues/netstack-holds-a-pending-request-for-a-client-that-left.md +++ b/issues/netstack-holds-a-pending-request-for-a-client-that-left.md @@ -6,17 +6,16 @@ opened: 2026-09-26 # netd holds a pending request for a client that left -A UDP receive with no datagram yet and a piped connect waiting for its SYN-ACK -keep their client's connection in `pending_udp_recvs` and -`pending_piped_connects` (`userland/netd/src/main.rs`) until what they wait -for happens. Neither watches the connection, so a client that hangs up is -noticed only when netd writes its answer: a receive on a socket nothing sends -to holds its entry for the life of the socket, and a connect with no timeout -holds its socket and its piped-connection slot until smoltcp gives the -handshake up. +A UDP receive with no datagram yet and a connect waiting for its handshake keep +their client's connection in `Sockets::receiving` and `Sockets::connecting` +(`userland/netstack/src/serve.rs`) until what they wait for happens. Neither +watches the connection, so a client that hangs up is noticed only when netstack +writes its answer: a receive on a socket nothing sends to holds its entry for +the life of the socket, and a connect with no timeout holds its stream and its +place until [tcp] gives the handshake up. -A lookup's client is watched and let go at once (`resolve::Resolver::let_go`, -reached from the `TOKEN_LOOKUP_BASE` watches in `main`). +A lookup's client is watched and let go at once (`Node::let_go`, reached from +the `TOKEN_LOOKUP` watches in `Sockets::watch`). Exit condition: both pending lists watch their clients' connections the same way, and a guest test that hangs up a pending receive and a pending connect diff --git a/issues/netstack-keeps-receiving-for-a-client-that-closed-its-receive-end.md b/issues/netstack-keeps-receiving-for-a-client-that-closed-its-receive-end.md index 98680b28f7..d4a9a30f73 100644 --- a/issues/netstack-keeps-receiving-for-a-client-that-closed-its-receive-end.md +++ b/issues/netstack-keeps-receiving-for-a-client-that-closed-its-receive-end.md @@ -6,17 +6,17 @@ opened: 2026-09-26 # netd keeps receiving for a client that closed its receive end -When a piped TCP connection's client closes its receive pipe while the peer is -still sending, `bridge_piped` (`userland/netd/src/main.rs`) drops its end of -that pipe and nothing else: the socket stays open, keeps its unread bytes and -advertises a zero window, so the peer's further bytes sit in flight and its -sender blocks until the client also closes its send pipe — and even then the -socket only sends a FIN into a peer that is still trying to send. Nothing -tells the peer the bytes will never be read. +When a stream's client closes its receive pipe while the peer is still +sending, the node lets its end of that pipe go and nothing else +(`userland/netstack/node/src/streams.rs`, `Stream::pass`): the connection +stays open, [tcp] keeps its unread bytes and advertises a closing window, so +the peer's further bytes sit in flight and its sender blocks until the client +is done writing too. Then the node closes the connection, and [tcp] resets one +closed with text unread (`a_close_with_text_unread_resets`, +`userland/netstack/node/tests/streams.rs`). Until then nothing tells the peer +the bytes will never be read. -Pre-existing, and unchanged by the fix that stopped netd dropping bytes the -pipe had refused. `netd_refused_pipes`' fourth case builds exactly this state -and only asserts that netd survives it. +Read from the code, not measured. Exit condition: a connection whose receive end is gone with bytes still owed is reset (or its receive side shut down) so the peer learns at once, with a diff --git a/issues/netstack-keeps-the-datagram-socket-of-a-client-that-sent-no-close.md b/issues/netstack-keeps-the-datagram-socket-of-a-client-that-sent-no-close.md deleted file mode 100644 index ace4fdf606..0000000000 --- a/issues/netstack-keeps-the-datagram-socket-of-a-client-that-sent-no-close.md +++ /dev/null @@ -1,36 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-10-08 ---- - -# netstack keeps the datagram socket of a client that sent no close - -A UDP socket is three things in netstack (`userland/netstack/src/main.rs`): -the smoltcp socket with its two 64 KiB buffers, the id's entry in `sockets`, -and the two pipe ends in `udp_pipes`, each keeping a 2 MiB kernel pipe alive. -A close request removes all three. Nothing else looks at the pipes: a client -that dies, or drops its ends without the request, leaves the socket and its -bound port for the life of the process, and a restarted program that binds -the same port is answered `ERR_ADDR_IN_USE`. The one other way out is -`deliver_datagram`, which ends the socket when a receive that was already -pending writes into a pipe with no reader. - -A listener's owner is heard, and only on a pass something else causes: -`serve_piped_listeners` writes zero bytes to every notify pipe each pass and -ends the listener the kernel answers `Gone` for, and no watch wakes a pass for -it, so an idle netstack keeps a dead owner's port until other traffic arrives. - -A piped TCP connection is the shape to copy: `bridge_piped` ends it on what -the kernel says of the client's pipe ends, and the watch on each of them is -what wakes that pass. - -Read from the code, not measured. - -**Exit condition**: a datagram socket and a listener whose owner's pipe ends -are closed leave the socket set and the table with no other traffic, and a -guest test on `tests/netcase` that binds each, drops it without a close -request, and reads `net.sockets.udp` and `net.sockets.listeners` back at what -they were. - -**Owner**: whoever holds `issues/toyos-has-its-own-network-stack.md`. diff --git a/issues/netstack-lookup-slots-have-no-per-client-share.md b/issues/netstack-lookup-slots-have-no-per-client-share.md index 7ab352bbeb..36d29b4712 100644 --- a/issues/netstack-lookup-slots-have-no-per-client-share.md +++ b/issues/netstack-lookup-slots-have-no-per-client-share.md @@ -6,7 +6,7 @@ opened: 2026-09-26 # netd's lookup slots have no per-client share -`resolve::MAX_LOOKUPS` (`userland/netd/src/resolve.rs`) bounds the lookups in +`toyos_dns::MAX_LOOKUPS` (`userland/netstack/node/src/resolve.rs`) bounds the lookups in flight across every client at once. One holder of `netd` that asks for names its servers answer slowly, or not at all, holds all of them for up to a lookup's whole schedule each, and every other program's lookup is refused with @@ -14,8 +14,9 @@ lookup's whole schedule each, and every other program's lookup is refused with netd cannot share the slots out itself: a client is a connection, a program opens as many as it likes, and netd holds no word for the program behind one -(there is no pid-as-authority, by design). The same is true of piped -connections, bounded by `max_piped_connections` across all clients. +(there is no pid-as-authority, by design). The same is true of the +node's places (`userland/netstack/node/src/places.rs`), one number across all +clients. Exit condition: a lookup is charged to something a program cannot multiply by opening connections (a per-program grant from init, or a kernel-side diff --git a/issues/netstack-passes-every-millisecond-while-a-request-is-pending.md b/issues/netstack-passes-every-millisecond-while-a-request-is-pending.md deleted file mode 100644 index 6d65bcb04a..0000000000 --- a/issues/netstack-passes-every-millisecond-while-a-request-is-pending.md +++ /dev/null @@ -1,25 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-09-26 ---- - -# netd passes every millisecond while a request is pending - -netd's loop (`userland/netd/src/main.rs`, `main`) waits at most 1 ms whenever -a UDP receive or a piped connect is pending, whether or not anything is due. -It is a flat wait standing in for events those requests do not have: - -- a pending UDP receive's datagram wakes the NIC, but a receive whose socket - another request closes is answered only by some later pass; -- a pending connect's `timeout_ms` deadline is not folded into the loop's - timeout the way the handshake sweep's is. - -A piped connection no longer holds the loop to 1 ms: its peer's bytes wake the -NIC, its client's bytes wake its send pipe's watch while the socket has room, -held bytes wake its receive pipe's watch, and smoltcp's timers are in -`poll_delay`, whose zero is a pass at once. Neither does a lookup: its reply -wakes the NIC and its wait is `resolve::Resolver::wake_in`. - -Exit condition: the loop's timeout is the earliest real deadline, each pending -request is answered on its own event, and no 1 ms constant remains. diff --git a/issues/netstack-removes-a-closed-stream-before-its-fin-leaves.md b/issues/netstack-removes-a-closed-stream-before-its-fin-leaves.md deleted file mode 100644 index b04a57b19a..0000000000 --- a/issues/netstack-removes-a-closed-stream-before-its-fin-leaves.md +++ /dev/null @@ -1,23 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-09-25 ---- - -# netd removes a closed stream before its FIN leaves - -`handle_tcp_close` (`userland/netd/src/main.rs`) answers a client's close of a -TCP stream with `socket.close()` and then `socket_set.remove(handle)` in the -same pass. smoltcp's `close` only queues the FIN for the next `poll`, and the -socket is gone before that poll runs, so the FIN is never sent: the peer keeps -an established connection that nothing on this machine will ever write to -again, and learns otherwise only if it sends a segment and gets the reset back. - -Read from the code, not measured. Every server that closes a connection it -accepted is affected — `logd` turning a reader away past its `MAX_READERS`, -sshd ending a session — and a peer that only reads, as the host's log reader -does, waits on it for ever. - -Exit condition: a closed stream stays in the socket set until its FIN has been -sent (or the peer's reset has ended it), with a test whose host peer only reads -and sees the close. diff --git a/issues/netstack-resolver-asks-a-dead-primary-first-on-every-lookup.md b/issues/netstack-resolver-asks-a-dead-primary-first-on-every-lookup.md index 9fbac8051b..f44b733f7e 100644 --- a/issues/netstack-resolver-asks-a-dead-primary-first-on-every-lookup.md +++ b/issues/netstack-resolver-asks-a-dead-primary-first-on-every-lookup.md @@ -9,35 +9,20 @@ opened: 2026-09-26 Every lookup asks the lease's first server first (`toyos_dns::Lookup::start`), with no memory of that server's silence. On a network whose primary resolver is down, every lookup spends one whole wait (`toyos_dns::WAIT_MS`, 2 s) on it -before the next server is asked. With `resolve::MAX_LOOKUPS` = 16 slots, that -caps netd at 8 lookups a second, and every lookup past that is refused +before the next server is asked. With `toyos_dns::MAX_LOOKUPS` = 16 slots, that +caps netstack at 8 lookups a second, and every lookup past that is refused `ERR_RESOURCE_EXHAUSTED`. -A second cost comes on top of the first: smoltcp asks for one neighbour's link -address per second for the whole interface (`neighbor::Cache::SILENT_TIME`), -and a query to a primary that answers no ARP takes that second. So the working -server's address is learned a second late, and it is learned again whenever -its cache entry expires (`ENTRY_LIFETIME`, 60 s). A frame from the neighbour -restarts the entry's lifetime, so the entry expires only after 60 s in which -the working server sent nothing. - -Measured on smoltcp 0.12's own `Interface` with the host harness of -`userland/netd/src/resolve/tests.rs` (`Net`). Time moved to netd's own wakes. -The servers were `[SILENT, ANSWERS]`: the first on the link and answering no -ARP, the second answering. The scratch test that drove it is not committed. - -- **One lookup started every 50 ms for 60 s:** - - 1201 lookups started, and 737 of them refused `ERR_RESOURCE_EXHAUSTED`; - - 448 answered, none failed, and 16 still in flight at the end; - - the first answer came 3000 ms after the first lookup started, and every - later one 2000 ms after its lookup started; - - smoltcp asked for the silent server's address 58 times and the answering - server's once. -- **The same stream for 180 s:** 3601 started, 2195 refused, and the answering - server's address asked once. -- **Lookups for 10 s, none for 65 s, then lookups again:** the answering - server's address was asked a second time. The first answer after the pause - took 3000 ms again, where the ones after it took 2000 ms. +The second cost this issue recorded, one neighbour request a second for the +whole interface, left with smoltcp: [ip] asks for each next hop on its own +(`rfc_4861_7_2_2_a_silent_next_hop_holds_up_no_other`, +`toyos-net-ip/tests/nud.rs`), and a query whose resolver's link address [ip] +gives up is let go at once +(`a_query_for_a_resolver_ip_has_given_up_ends_its_lookup_in_the_opportunity_that_would_have_carried_it`, +`userland/netstack/node/tests/resolve.rs`). The first cost is read from the +constants and not measured on the node: [ip] gives a silent next hop up after +3 s, past the lookup's own wait, so a silent primary still costs each lookup +one `WAIT_MS`. Exit condition: either the resolver keeps a history of each server's answers and orders the servers by it (RFC 1035 §7.2), so a silent primary is asked last, diff --git a/issues/netstack-spends-two-poller-slots-per-piped-connection.md b/issues/netstack-spends-two-poller-slots-per-piped-connection.md index 7c0a60c78d..cde07b542f 100644 --- a/issues/netstack-spends-two-poller-slots-per-piped-connection.md +++ b/issues/netstack-spends-two-poller-slots-per-piped-connection.md @@ -6,21 +6,23 @@ opened: 2026-09-26 # netstack spends two poller slots per piped connection -A piped connection registers both its pipes on every pass, for as long as -netstack holds them: its send pipe `READABLE` while the socket has send room -and `OTHER_END_GONE` until the kernel has answered that, and its receive pipe -`OTHER_END_GONE` and, while it holds bytes the pipe refused, `WRITABLE` -(`userland/netstack/src/main.rs`, the loop in `main`). `MAX_PIPED_SLOTS` -divides the batch one poller can carry (`Poller::MAX_HANDLES`, less the fixed +A stream registers both its pipes on every pass, for as long as the node holds +them: its send pipe `READABLE` while the stack has room for its bytes and +`OTHER_END_GONE` until the kernel has answered that, and its receive pipe +`OTHER_END_GONE` and, while the stack holds bytes the pipe refused, `WRITABLE` +(`userland/netstack/src/serve.rs`, `Sockets::watch`, from the node's +`Node::watches`). `MAX_PLACES` (`userland/netstack/src/main.rs`) divides the +batch one poller can carry (`Poller::MAX_HANDLES`, less the fixed registrations, the pending connections and the lookups' clients) by -`POLL_HANDLES_PER_PIPED = 2`, so the ceiling on live piped connections is half +`serve::WATCHES_PER_PLACE = 2`, so the ceiling on the node's places is half what one registration each would give. -The memory budget (an eighth of memory at 4 MiB a connection) binds first only -on a machine whose eighth holds fewer connections than that ceiling. +The memory budget (an eighth of memory at what a listener can be made to hold, +`PLACE_BYTES`) binds first only on a machine whose eighth holds fewer places +than that ceiling. **Exit condition**: one registration per connection, and -`POLL_HANDLES_PER_PIPED` deleted. One `OP_WATCH` on a joined `Connection` +`WATCHES_PER_PLACE` deleted. One `OP_WATCH` on a joined `Connection` carries `READABLE | WRITABLE` for both its pipes, and is refused `OTHER_END_GONE`: `ops::pipe_end_watch` (`kernel/src/object/ops.rs`) answers a pipe end alone, and a connection has two other ends for the one bit. So the diff --git a/issues/netstack-watches-both-pipes-of-every-connection-on-every-pass.md b/issues/netstack-watches-both-pipes-of-every-connection-on-every-pass.md index 5b0421f65c..6a50b070dc 100644 --- a/issues/netstack-watches-both-pipes-of-every-connection-on-every-pass.md +++ b/issues/netstack-watches-both-pipes-of-every-connection-on-every-pass.md @@ -7,8 +7,8 @@ opened: 2026-10-08 # netstack watches both pipes of every connection on every pass Every pass of netstack's loop submits a watch for each pipe of each live piped -connection, ready or not, changed or not (`userland/netstack/src/main.rs`, the -loop in `main`): an idle connection submits two. A watch replaces its handle's +connection, ready or not, changed or not (`userland/netstack/src/serve.rs`, +`Sockets::watch`, called by the loop in `main`): an idle connection submits two. A watch replaces its handle's earlier one, so each is a poll allocated and registered again, with the takes of the global pipe lock that `arm` makes to read the pipe and find its watches (`kernel/src/inbox/mod.rs`). diff --git a/issues/netstacks-places-are-one-number-for-every-client.md b/issues/netstacks-places-are-one-number-for-every-client.md new file mode 100644 index 0000000000..2792bdc742 --- /dev/null +++ b/issues/netstacks-places-are-one-number-for-every-client.md @@ -0,0 +1,33 @@ +--- +status: open +kind: defect +opened: 2026-10-08 +--- + +# netstack's places are one number for every client + +Every stream, listener and datagram socket a client makes netstack hold takes +one of the node's places (`userland/netstack/node/src/places.rs`), which +netstack sets from an eighth of physical memory at what a listener can be made +to hold (`userland/netstack/src/main.rs`, `places_for`), and a connect, a +listen, an accept and a bind with none left are refused with nothing made: +`a_connect_past_the_places_is_refused_and_sends_nothing`, +`a_listener_holds_a_place_and_a_listen_without_one_makes_nothing` and +`a_datagram_socket_holds_a_place_and_a_bind_without_one_makes_nothing` +(`userland/netstack/node/tests/listeners.rs`), and in a guest the connect past +them, answered `ERR_RESOURCE_EXHAUSTED` (`netstack_socket_churn`). The sockets +of the node's own, the responder's for the machine's name and each query's of +a lookup, hold no place: a lookup is bounded by `toyos_dns::MAX_LOOKUPS` and +its rounds. + +The places are one number for every client. One holder of the `netstack` +connector that connects, listens or binds in a loop takes them all, and every +dynamic UDP port, from every other program: +`issues/netstack-lookup-slots-have-no-per-client-share.md` records the same of +the lookups, and why netstack cannot share them out itself. + +**Exit condition**: a test takes one client to its share of the places, sees +its next bind refused `ERR_RESOURCE_EXHAUSTED` with nothing made, and another +client's bind answered. + +**Owner**: whoever holds `issues/toyos-has-its-own-network-stack.md`. diff --git a/issues/no-physical-memory-fairness.md b/issues/no-physical-memory-fairness.md index 9c6e73e2c2..0454f4d516 100644 --- a/issues/no-physical-memory-fairness.md +++ b/issues/no-physical-memory-fairness.md @@ -21,8 +21,8 @@ request is an allocation request, and every one of them needs an owner who can say no.* Three instances were filed under it — the compositor's windows, netd's piped connections, and `SYS_CONNECT` pinning 4 MiB into an unbounded pending queue — and all three now have a bound *and* a caller that hears the refusal, -which is the pair the class asks for: `toyos_desktop::max_windows` and netd's -`max_piped_connections` each divide an eighth of physical memory by what one unit +which is the pair the class asks for: `toyos_desktop::max_windows` and netstack's +`places_for` each divide an eighth of physical memory by what one unit costs and refuse past it, and a pipe allocates its ring page on first use rather than at `create`. **What a bound alone still does not answer is whose window to refuse.** The memory is charged to nobody, so a cap is the only thing between one diff --git a/issues/the-guest-suite-runs-only-what-no-cheaper-tier-reaches.md b/issues/the-guest-suite-runs-only-what-no-cheaper-tier-reaches.md index 5e941e8bfc..96881f0e44 100644 --- a/issues/the-guest-suite-runs-only-what-no-cheaper-tier-reaches.md +++ b/issues/the-guest-suite-runs-only-what-no-cheaper-tier-reaches.md @@ -226,9 +226,8 @@ Host: `toyos-net-tcp`, `toyos-dns`, `toyos-mdns`, `toyos-swap`, `toyos-inspect`, - host `sshd_key_auth`: sshd refuses a key not authorized. - host `lan_lease_report`: a link that goes down and comes back after a lease neither gives the lease up nor starts the client over. The T14 cannot flap its cable, and no T14 row reads a lease - (`issues/the-host-cannot-reach-the-t14-while-it-runs-toyos.md`). Exit: netd's link-up decision (its main loop and - `dhcp::restart`) lifted into a function a netd `#[test]` drives with a lease held, red when the - `!dhcp.leased()` guard goes. + (`issues/the-host-cannot-reach-the-t14-while-it-runs-toyos.md`). Held on a host by the node's `a_link_that_returns_keeps_a_held_lease_and_announces_it` and + `a_link_that_returns_with_no_lease_starts_discovery_over_at_once` (`userland/netstack/node/tests/lease.rs`). - metal `sshd_exec`, the arms `lan_talk`'s one command did not reach, each a step of that row's exchange red when its arm in `userland/sshd/src/main.rs` is reverted; the row and its exchange are deleted, and this waits on `issues/the-host-cannot-reach-the-t14-while-it-runs-toyos.md`: diff --git a/issues/the-pipe-abi-has-no-word-for-an-unreachable-host-or-a-lookup-to-try-again.md b/issues/the-pipe-abi-has-no-word-for-an-unreachable-host-or-a-lookup-to-try-again.md new file mode 100644 index 0000000000..02bb2e4be1 --- /dev/null +++ b/issues/the-pipe-abi-has-no-word-for-an-unreachable-host-or-a-lookup-to-try-again.md @@ -0,0 +1,33 @@ +--- +status: open +kind: defect +opened: 2026-10-09 +--- + +# The pipe ABI has no word for an unreachable host or a lookup to try again + +netstack answers three of the node's endings in a word of the pipe ABI +(`toyos/src/net.rs`) that is not theirs (`userland/netstack/src/serve.rs`, +`connect_failed` and `Sockets::settle`): + +- **A connect an ICMP error ended**, `Failure::Unreachable(_)` and + `Failure::Prohibited`, is answered `ERR_OTHER`: std `Other`, libc `EIO`, + with the ICMP code only in the log. Owed: network unreachable and host + unreachable as std `NetworkUnreachable` / `HostUnreachable`, libc + `ENETUNREACH` / `EHOSTUNREACH`; port unreachable as `ECONNREFUSED`; and + `Prohibited` as `EHOSTUNREACH`, which is what Linux makes of an + administratively filtered ICMP error. +- **A lookup ending `Unreachable`** is answered `ERR_NOT_CONNECTED`: std + `NotConnected`, libc `ENOTCONN`, a stream's word and the wrong family for a + lookup. Owed: network unreachable. +- **A lookup ending `LeaseChanged`** is answered the same. Owed: a word to try + again, getaddrinfo's `EAI_AGAIN`. + +No caller in the tree branches on any of the three today, so a program reads a +wrong reason and nothing acts on it. + +**Exit condition**: an ABI change after the move gives the pipe ABI the words +above, std and libc map them as named, and netstack answers each ending in its +own word. + +**Owner**: the network track, `issues/toyos-has-its-own-network-stack.md`. diff --git a/issues/toyos-has-its-own-network-stack.md b/issues/toyos-has-its-own-network-stack.md index 44c6c5e362..3332765731 100644 --- a/issues/toyos-has-its-own-network-stack.md +++ b/issues/toyos-has-its-own-network-stack.md @@ -6,7 +6,7 @@ opened: 2026-09-27 # ToyOS has its own network stack -netd runs smoltcp. Its replacement is ToyOS's own stack, built clean-room: readers specify each stage from the RFCs, writers build from those specifications and the RFCs alone and never open another stack's source. The specifications live outside the tree; every test names its scenario id, and the review checks each scenario has one. Where a departure's exit names a test or an RFC rather than the specifications, that is the orchestrator's reading of the owner's words of 2026-10-04: "why do we need to persist prose. The specs exist we can reference them cant we?" +netstack ran smoltcp until stage 5. Its replacement is ToyOS's own stack, built clean-room: readers specify each stage from the RFCs, writers build from those specifications and the RFCs alone and never open another stack's source. The specifications live outside the tree; every test names its scenario id, and the review checks each scenario has one. Where a departure's exit names a test or an RFC rather than the specifications, that is the orchestrator's reading of the owner's words of 2026-10-04: "why do we need to persist prose. The specs exist we can reference them cant we?" - Stage 0: the acceptance bar, on smoltcp. - Stage 1: `toyos-net-wire`, in the tree. @@ -17,37 +17,36 @@ netd runs smoltcp. Its replacement is ToyOS's own stack, built clean-room: reade - Then multi-core, netring (blocked on the owner's ABI ruling), TCP and IP hardening, IPv6, offloads, soak. - This track owns the T14's outbound rows, the machine reaching its router and the internet on the I219, judged from the stick: they arrive on this stack and not on smoltcp (owner: "no smoltcp."), and until they do no T14 row reads the wired card (`issues/the-host-cannot-reach-the-t14-while-it-runs-toyos.md`). -Stage 5 lands in slices, the orchestrator's cut under the owner's words "i want smoltcp out as fast as possible": netd's decisions go into `toyos-net-node` (`userland/netstack/node`), pure and host-tested and shipped in nothing, and then one change moves netd onto it and deletes smoltcp. In the tree: the node, its DHCP lease, its clients' datagram sockets and their broadcast permission, its mDNS name, its resolver, streams, listeners and the one bound on its clients' streams, listeners and datagram sockets (`userland/netstack/node/src/places.rs`); a lookup's sockets stand outside that bound, under `toyos_dns::MAX_LOOKUPS` and its rounds; and datagram senders that wait for their next hop in their own queue, with [ip] taking none of them. Still to build on it: the move, which hands `Node::transmit` its draws, since a lookup carries on inside a transmit opportunity. +Stage 5 lands in slices, the orchestrator's cut under the owner's words "i want smoltcp out as fast as possible": netd's decisions go into `toyos-net-node` (`userland/netstack/node`), pure and host-tested and shipped in nothing, and then one change moves netd onto it and deletes smoltcp. In the tree: the node, its DHCP lease, its clients' datagram sockets and their broadcast permission, its mDNS name, its resolver, streams, listeners and the one bound on its clients' streams, listeners and datagram sockets (`userland/netstack/node/src/places.rs`); a lookup's sockets stand outside that bound, under `toyos_dns::MAX_LOOKUPS` and its rounds; and datagram senders that wait for their next hop in their own queue, with [ip] taking none of them. The move is in the tree too: `userland/netstack` is the card, the clock, the kernel's random source and pipes and the clients' requests around the node (`userland/netstack/src/main.rs`), and smoltcp is in no manifest and no lockfile. What the node does not yet meet: -- Of what a third party wrote, only slirp's recorded OFFER and ACK have reached it (`userland/netstack/node/tests/slirp.rs`); every other frame it has answered is the tests' own, from the RFCs' layouts. Exit: netd runs on it against slirp in a guest and a router on the T14. +- Of what a third party wrote, it has met QEMU's user network in a guest, on virtio and on QEMU's e1000e: its DHCP server, its ARP and the host kernel's TCP behind it (`netstack_streams`, `netstack_socket_churn`, `libc_sockets`). It has met no router and no switch. Exit: the T14's outbound rows are green on it. - It counts a DHCP message [udp] refused as `node.dhcp-unsent`, whatever the rule; the `dhcp.renew-unroutable` scenario owed above is not written. Exit: that scenario names the counter, or the node counts the renewal apart. - With the link down the DHCP client keeps its timers: every 4 to 64 s the node has a deadline, builds a DISCOVER that [udp] refuses, and counts it in `dhcp.tx.discover` and `node.dhcp-unsent`, from `Node::new` on. Exit: `toyos-dhcp`'s client is told the link went down and waits for it, and the node's first DISCOVER is the one that leaves. -- No second TCP has been on the far end of a stream: the peer of `userland/netstack/node/tests/streams.rs` is the tests' own script, which acknowledges what arrives in order and loses, reorders and repeats nothing; what is not ours there is `etherparse`, which reads every segment the node emits and builds every one it receives. Exit: netd runs on the node against the host kernel's TCP through slirp in a guest. +- The node's link-down handling (`Node::link`, `userland/netstack/node/src/lib.rs`) and its INIT-REBOOT, a held lease asked for again by REQUEST when the link returns (`Client::link_up`, `toyos-dhcp/src/lib.rs`), have never run on a real card or against a real DHCP server: host tests alone hold them, and no T14 boot has had its link go. Exit: one attended T14 boot with the cable pulled and returned, whose log shows `netstack: I219: link down`, then on the link's return the held lease verified by a REQUEST and its ACK, or taken again, with neither DHCP line that says this machine has no address, then the name's claim line, `netstack: mDNS: no host answered for toyos-t14.local; this machine answers as it`, again. - The node's places are one number for every client: a program that connects, listens or binds in a loop takes them all, and so does a peer that keeps the connections [tcp] is finishing alive, each of which holds its stream's place (the line below on a client gone with nothing left in its pipe). The node has no word for the program behind a request, as `issues/netstack-lookup-slots-have-no-per-client-share.md` records of netd. Exit: a test of the node in which one program at its share of the places leaves another's connect, listen, accept and bind answered. - A port is one listener's, whatever address either listen named: a listen on a port a listener holds is refused `ListenRefused::InUse` (`userland/netstack/node/src/listeners.rs`), so [tcp]'s rule that a SYN goes to the listener that named its address before one at every address (LS-09) is not reachable from the node, and one program's two listeners on one port at two addresses, which a host allows one user, are refused too. The node has no word for the program behind a listen: with LS-09 reachable, any program that reaches netstack would take every later connection of another program's listener at every address by naming the machine's address on its port. Exit: a listen carries its program, a second listener on a held port is allowed only to the program that holds every listener on it, and a test of the node finds one program's pair standing, a SYN handed to the one that named its address, and another program's listen on that port refused. -- A connection that waits to be accepted holds no place: a listener's one place stands for [tcp]'s `limits::LISTEN_READY` of them, each with at most one receive buffer of text taken when its first byte arrives, 128 x 65,535 bytes or 8.4 MB a listener as netd configures [tcp]. Counting each as a place would let any peer of any listener take every place by handshakes alone and refuse every client its connect, listen and bind; as built a peer fills the queue of the listener it reaches. So the places bound the node's memory only if the shell prices a listener at that product. Exit: the move derives netd's places from a memory figure in which a listener costs its 128 receive buffers, and its body says the figure. - A listener's handshakes in progress are [tcp]'s `limits::LISTEN_PENDING`, each held until `limits::SYNACK_GIVE_UP`, and a SYN past them is dropped: 256 SYNs a minute from addresses that never answer shut a port to every other peer for as long as they keep coming. [tcp] has no SYN cookie and lets no older handshake go for a newer one. Exit: a test in `toyos-net-shard/tcp/tests/` in which a peer's handshake completes while LS-12's flood runs. - The node counts a wake unspent until an accept arrives, and cannot tell an accept on its way from one that will never be sent: an owner that reads a wake and sends no accept is owed one wake fewer from then on, and the last connection waiting is not announced. That is the node's half of `issues/an-accept-that-never-reaches-netstack-strands-its-listener.md`. Exit: that issue's, by a client that cannot spend a wake without its accept arriving, or by an accept that waits in netd and needs no wake. -- A stream its client can see no more, its reader gone or the peer's FIN read through and its client writing no more, is reset once its pipe has given up no byte for 100 s, and a peer address restarts that clock for at most 16 such streams at once (`OWNERLESS_PER_PEER`, `userland/netstack/node/src/streams.rs`); one past them has 100 s from the pass that found it so, until one of the 16 is done and it takes its place. What holds in this stage, from the code: per address, 16, an estimate no measurement set. There is no floor on the rate: each of the 16 is kept for as long as its pipe gives up one byte per 100 s, which for a full 2 MiB pipe is 2,097,152 x 100 s, over six years, and then for its tail in [tcp] (the line below on a client gone with nothing left in its pipe). The 16 is no bound across addresses: addresses are not counted, and where a connect's address is its client's choice, an accepted stream's is the peer's, which on the link answers for as many addresses as it likes. The bound across them is the node's places (`userland/netstack/node/src/places.rs`), of which every stream holds one until the node lets it go, and there is no second one: in `peers_at_more_addresses_than_there_are_places_hold_no_stream_past_them` (`userland/netstack/node/tests/listeners.rs`) peers at four addresses bring sixteen connections each to a node with three places, which holds two such streams at a time, each kept alive past its 100 s, and refuses every accept and connect past them. `issues/netstack-cuts-a-departed-clients-unsent-tail-at-the-ceiling.md` stays open until netd runs on the node: on a host its exit's second clause is met (`a_departed_clients_tail_arrives_whole_while_its_peer_takes_it`) and its first for 16 streams an address. Exit: the T14 or a guest shows a peer holding one at a byte per 100 s, and the cut takes a floor on the rate, or the owner rules it needs none; and the move closes that issue with its bench row. -- A stream its peer reset is gone from the node at once, so for a client that still holds it `Node::set_nodelay` answers `false` and `Node::nodelay` `None`: `issues/a-stream-its-peer-reset-refuses-the-option-requests-a-host-answers.md`, carried by the node as built. std's `nodelay()` asks netd nothing: it answers from the value its `TcpStream` holds (`sdk/std/sys/net/connection.rs`), and libc's `getsockopt` of `TCP_NODELAY` from its socket's entry (`userland/libc/src/socket.rs`). What the node carries is `set_nodelay` on a reset stream, refused where macOS refuses it under another kind and Linux answers. The node cannot hold the stream for the client: its two pipe ends are all it has of the client, letting both go is how the client learns of the reset, and with both gone nothing tells it when the client leaves, so a kept stream would be kept for ever. Exit, outside the node: that issue's guest test is green, and a `getsockopt` of `TCP_NODELAY` on a stream its peer reset is answered, by libc from a value its socket holds as std's does or by a pipe ABI that gives netd a handle whose other end's leaving the kernel reports. -- How a stream ended reaches its client as the order its two pipes end in, which is one bit: the peer's FIN, or a failure. The node keeps a send pipe whose FIN is queued until its writer leaves and lets a failed stream's send pipe go first (`userland/netstack/node/src/streams.rs`), so std reads a FIN as a FIN and a failure as `ConnectionReset`, whichever failure it was: `issues/a-streams-failure-reaches-its-client-as-a-reset-whatever-it-was.md`. libc reads no end's kind until the move carries its change for `issues/a-netstack-client-cannot-tell-a-reset-from-the-peers-fin.md` (posted on pull request #803) with that issue's guest C case on `tests/netcase`. Exit: those two issues'. -- Every frame and every deadline ends in a pass over every stream, every transmit opportunity walks every stream twice (once for the count each peer address keeps alive, once to pass those whose connect is not answered), and `Node::next_deadline` reads every stream: linear in streams per frame, as netd's `bridge_piped` is today. For an idle stream a pass is one `recv_with`, two `status` and one read of its client's pipe, which in netd is a system call: 1,000 system calls a frame with 1,000 idle streams. A pass's `recv_with` also re-files its connection's deadline and offers it to the transmit round, so one frame makes every idle stream eligible and the next opportunity serves each to learn it has nothing. [tcp] cannot say which connections a segment, a timer, an ICMP error or a failed next hop moved: it reports `drain_eligible` and `drain_gone`, which are the round's. The change: [tcp] marks a held connection in `for_conn`, `tick`, `icmp` and `end` and hands the marks out once (`drain_moved`); the node maps a connection to its stream, keeps its deadlines ordered, and passes only the streams [tcp] moved, the one a pipe's wake names, and those due. Exit: at the move, netd's CPU time per frame on the T14 with 1, 100 and 1,000 idle streams beside one bulk stream is measured, and the change lands if 1,000 idle streams cost more than 12.3 µs a frame over what one costs. The 12.3 µs is the time a full frame takes on a wire of 1 Gbit/s, 1,538 bytes of 8 ns with its preamble and gap: past it the passes alone take longer than the bulk stream's frames take to arrive. It is a threshold by arithmetic and no measurement; by estimate, not measured, 1,000 system calls exceed it. +- A stream its client can see no more, its reader gone or the peer's FIN read through and its client writing no more, is reset once its pipe has given up no byte for 100 s, and a peer address restarts that clock for at most 16 such streams at once (`OWNERLESS_PER_PEER`, `userland/netstack/node/src/streams.rs`); one past them has 100 s from the pass that found it so, until one of the 16 is done and it takes its place. What holds in this stage, from the code: per address, 16, an estimate no measurement set. There is no floor on the rate: each of the 16 is kept for as long as its pipe gives up one byte per 100 s, which for a full 2 MiB pipe is 2,097,152 x 100 s, over six years, and then for its tail in [tcp] (the line below on a client gone with nothing left in its pipe). The 16 is no bound across addresses: addresses are not counted, and where a connect's address is its client's choice, an accepted stream's is the peer's, which on the link answers for as many addresses as it likes. The bound across them is the node's places (`userland/netstack/node/src/places.rs`), of which every stream holds one until the node lets it go, and there is no second one: in `peers_at_more_addresses_than_there_are_places_hold_no_stream_past_them` (`userland/netstack/node/tests/listeners.rs`) peers at four addresses bring sixteen connections each to a node with three places, which holds two such streams at a time, each kept alive past its 100 s, and refuses every accept and connect past them. Exit: the T14 or a guest shows a peer holding one at a byte per 100 s, and the cut takes a floor on the rate, or the owner rules it needs none. +- A stream its peer reset is gone from the node at once, so for a client that still holds it `Node::set_nodelay` answers `false`: `issues/a-stream-its-peer-reset-refuses-the-option-requests-a-host-answers.md`, carried by the node as built. std's `nodelay()` asks netd nothing: it answers from the value its `TcpStream` holds (`sdk/std/sys/net/connection.rs`), and libc's `getsockopt` of `TCP_NODELAY` from its socket's entry (`userland/libc/src/socket.rs`). What the node carries is `set_nodelay` on a reset stream, refused where macOS refuses it under another kind and Linux answers. The node cannot hold the stream for the client: its two pipe ends are all it has of the client, letting both go is how the client learns of the reset, and with both gone nothing tells it when the client leaves, so a kept stream would be kept for ever. Exit, outside the node: that issue's guest test is green, and a `getsockopt` of `TCP_NODELAY` on a stream its peer reset is answered, by libc from a value its socket holds as std's does or by a pipe ABI that gives netd a handle whose other end's leaving the kernel reports. +- How a stream ended reaches its client as the order its two pipes end in, which is one bit: the peer's FIN, or a failure. The node keeps a send pipe whose FIN is queued until its writer leaves and lets a failed stream's send pipe go first (`userland/netstack/node/src/streams.rs`), so std and libc read a FIN as a FIN and a failure as `ConnectionReset` or `ECONNRESET`, whichever failure it was: `issues/a-streams-failure-reaches-its-client-as-a-reset-whatever-it-was.md`. Exit: that issue's. +- Every frame and every deadline ends in a pass over every stream, every transmit opportunity walks every stream twice (once for the count each peer address keeps alive, once to pass those whose connect is not answered), and `Node::next_deadline` reads every stream: linear in streams per frame, as netstack's bridge on smoltcp was. For an idle stream a pass is one `recv_with`, two `status` and one read of its client's pipe, which in netstack is a system call: 1,000 system calls a frame with 1,000 idle streams. A pass's `recv_with` also re-files its connection's deadline and offers it to the transmit round, so one frame makes every idle stream eligible and the next opportunity serves each to learn it has nothing. [tcp] cannot say which connections a segment, a timer, an ICMP error or a failed next hop moved: it reports `drain_eligible` and `drain_gone`, which are the round's. The change: [tcp] marks a held connection in `for_conn`, `tick`, `icmp` and `end` and hands the marks out once (`drain_moved`); the node maps a connection to its stream, keeps its deadlines ordered, and passes only the streams [tcp] moved, the one a pipe's wake names, and those due. A pipe's end-gone answer (`Node::pipe_gone`) runs a pass of its own as well, beside the one `Sockets::bridge` runs for a wake's readiness answers, so k clients leaving in one wake cost k + 1 passes. Not measured at the move. The count is what a guest can read and a time is not: netstack's places end at 103, what one poller watches (`MAX_PLACES`, `userland/netstack/src/main.rs`), so 1,000 idle streams cannot exist, the T14's bench opens no peer that holds streams, and QEMU gives no verdict on time. Exit: a guest test counts netstack's reads of its clients' pipes per frame received beside one bulk stream, with 1 idle stream and with 100, and the line closes when the two counts are equal. - A client that is gone with nothing left in its pipe is finished by [tcp]'s rules for a user who let go: a reset once it has been idle 60 s, where netd on smoltcp reset it 100 s after the client left whatever the peer said. A peer that acknowledges a byte a minute holds such a connection for as long as it has bytes to acknowledge. Exit: the T14 or a guest shows such a peer holding one, and [tcp]'s rule takes a bound from the client's leaving; or the owner rules the idle bound is the one. - `Tcp::ready`, `Tcp::orphans` and `Tcp::options` are additions to [tcp] the specifications do not hold, made for the node's wakes, its places and the copy a stream keeps of its connection's options; their tests (`toyos-net-shard/tcp/tests/held.rs`) name no scenario id. `Shard::listen` refuses an address [ip] does not hold usable, by the rule [udp]'s bind refuses by (`udp.bind-address-not-local`), and counts nothing for it; the node's test of it (`a_listener_is_at_the_address_it_named_and_only_one_the_machine_holds`) names no scenario either. A listener whose address the machine loses stands as a datagram socket bound to it does, with its port and its place and its owner told nothing, and answers again once the address is back (`a_listener_stands_while_its_address_is_lost_and_answers_when_it_is_back`, `userland/netstack/node/tests/lease.rs`); no host was read for it. Exit: the specifications hold the three calls and the refusal, and the tests name their scenarios. - The word each of [udp]'s refusals is answered in on the pipe ABI is `Refused` in `userland/netstack/node/src/datagram.rs`, tested by `each_refusal_is_answered_in_the_pipes_word_for_it` against no reader's scenario; `udp.no-ephemeral-port` has a word and no test that reaches it. Exit: the scenario owed below exists, and the test names its id. -- A client's datagram to a broadcast address needs its socket's permission, where smoltcp sends it for every socket. The owner's ruling: "Of course carry the permission why would you even ask. The premise is and always has been to do things properly". The pipe ABI carries the permission as far as netd: `MsgType::UdpSetOption` with `OPT_BROADCAST` (`toyos/src/net.rs`), sent by std's `set_broadcast` (`sdk/std/sys/net/connection.rs`) and by libc's `setsockopt` of `SO_BROADCAST` (`userland/libc/src/socket.rs`), which keeps it for a socket not yet bound and sends it at `bind`; the word for a send refused for want of it is `ERR_PERMISSION_DENIED`, `ErrorKind::PermissionDenied` in std and `EACCES` in libc. The node enforces it and nothing that ships does: `Node::udp_set_broadcast` hands the permission to `Udp::set_broadcast`, and a send without it is refused `udp.broadcast-not-permitted` and answered `Refused::PermissionDenied` (`userland/netstack/node/src/datagram.rs`), which `a_broadcast_needs_its_permission` (US-21, `userland/netstack/node/tests/datagram.rs`) holds, with both broadcast addresses read off the wire in link-broadcast frames once the socket holds it and refused again once it is taken back, and `each_refusal_is_answered_in_the_pipes_word_for_it` holds the word. The permission of the call goes with the datagram to the route it leaves by: [udp] queues it with what its socket held, a closed socket's included, and `Ip::send_udp` refuses a link broadcast to one without it, `ip.broadcast-not-permitted`, counted and logged, so a prefix a DHCP server renews under a held address between the call and the frame makes no broadcast of a datagram accepted for a host (`s_udp_us_021_shard_a_datagram_accepted_for_a_host_is_no_broadcast_after_a_renewal_narrows_the_prefix` and its two neighbours in `toyos-net-shard/tests/udp.rs`, and from a server's ACK `a_datagram_accepted_for_a_host_is_no_broadcast_after_the_server_renews_a_narrower_prefix` in the node's tests); netd on smoltcp answers the request done and sends to a broadcast address for every socket. The search is run. `grep -rnI -E 'SO_BROADCAST|INADDR_BROADCAST|255\.255\.255\.255|sendto|setsockopt|sys/socket\.h|socket\(' userland/doom/doomgeneric`, the one third-party C program an image builds, at the commit `userland/doom/build.rs` pins, exits 1 with no line; the same without `sys/socket\.h` over `tests/testcases` exits 1 with no line; `SO_BROADCAST|INADDR_BROADCAST|255\.255\.255\.255` over the trees of `src/libcxx.rs`'s `SOURCES` finds LLVM libc's own macro and its documentation and no caller; and in ToyOS's tree outside the stack's crates the option is named only by the two setters. The socket2 fork's setter reaches nobody: `issues/the-socket2-forks-so-broadcast-setter-does-nothing.md`. Exit, at the move: netd answers `MsgType::UdpSetOption`'s `OPT_BROADCAST` by `Node::udp_set_broadcast` and writes `ERR_PERMISSION_DENIED` for `Refused::PermissionDenied`. Exit, after the move, which keeps this line open until both are green on netd on the node, since no test can read the refusal on smoltcp: a guest test, `udp_broadcast_needs_its_permission`, reads `broadcast()` false, then true after `set_broadcast(true)`, on the socket and on a `try_clone` of it, which `broadcast()` answering a constant or a duplicate holding a value of its own turns red, and sends to the limited broadcast address before and after the set and is refused `PermissionDenied` only before, which std's setter reverted to `Ok(())` turns red; and a guest C case reads libc's setter, `socket`, `setsockopt(SO_BROADCAST)`, `bind`, `sendto` to the limited broadcast address, refused `EACCES` without the option and sent with it, and with the option set before `bind`, which `bind` no longer handing the kept option over turns red. The case runs the sequence without `bind` too, where the first `sendto` binds the socket and hands the kept option over; `tests/netcase/sendto_unbound.c` holds that send for a socket permitted to make it. -- The machine's name is told to the node twice, as `toyos_dhcp::HostName` in `Node::new` and as `toyos_mdns::Host` in `Node::answer_as`: the first owns its bytes and the second borrows them. Exit: one type carries the label to both; due at the move if handing `answer_as` its `&'static str` takes a leak. -- The machine's name is probed for and defended (RFC 6762 §8, §9) by `toyos-mdns` (`userland/netstack/mdns`), on the node and in the netstack that ships. What that does not yet meet. Every machine is named `toyos-t14` (`dhcp::HOSTNAME`, `userland/netstack/src/dhcp.rs`) and a responder that loses its name holds none, so of two on one network the second answers to no name: asked "When two machines on one network have the same ToyOS host name, what should the second one do?", the owner answered "Hold no name, say so (Recommended)". Asked on 2026-10-09 "A ToyOS machine that loses its .local name to a conflict stays nameless until its network link next returns, even after the other machine has left. Should it try for its name again by itself?", he answered "Retry on a slow interval (Recommended)". Built: a lost name is probed for again a minute after each loss (`RETRY_MS`, `userland/netstack/mdns/src/lib.rs`) and at once on a link after none, and is claimed by the first probing no host answers; RFC 6762 permits probing again at §8.1's rate ("wait five seconds after any failed probe attempt before trying again") and does not require it. Two packets from any host on the link still take a held name, a response to take it back to probing and a response under the probe, and one more a minute keeps it: no path back exists that a peer cannot hold shut, since a host that answers every probe is what a holder of the name is. The caller does not say how a message was addressed, so two rules that turn on it are not applied: a response is read however it came, where §6 reads a unicast one only within two seconds of a question that asked for one; and a message from a source outside the subnet is ignored even when it was sent to the group, where §11 deems that on the link "regardless of source IP address", so a host of the same name on another subnet of the link is never a conflict. Exit: each machine's name is its own, and a test with two guests on one network finds each by its name; and the caller says how a message was addressed, with a test in which a unicast response nobody asked for takes no name and one in which a response sent to the group from another subnet does. Not measured, each with its exit: the shipped responder's claim on the T14's wired card, which no `system.toml` in the tree gives netstack (when the outbound rows land, their boot shows the lease line, then the claim line, and no loss line); and the link's return on metal (the orchestrator's attended session with the owner, never automated). A link that goes and returns inside one pass of the Intel driver reaches the name as no change: `issues/a-link-that-goes-and-returns-inside-one-pass-of-the-intel-driver-is-reported-as-no-change.md`. -- A socket's datagrams that wait for a next hop wait in the one queue it has, `toyos_net_udp::limits::TX_DATAGRAMS` of them: with all 16 waiting for one next hop, the socket's sends to every other destination are refused `udp.tx-queue-full`, `Refused::ResourceExhausted` to a client, until that hop answers or [ip] gives it up, 3 s after its first request left (`BROADCAST_SOLICIT` requests a `RETRANS` apart). It is no regression by this track's earlier record of what ships, that smoltcp kept each datagram in its own socket until the neighbour answered. What a closed socket had accepted waits in the closed sender's `CLOSED_DATAGRAMS` the same way, so 16 datagrams of closed sockets that wait for one silent host have the next closes' datagrams discarded, `udp.tx-discarded-on-close`, for those 3 s: the line on US-53 among stage 3's departures. `s_udp_us_025_what_waits_for_a_next_hop_is_bounded_by_its_sockets_queue` holds the first as it stands. Owner: the worker of the move, with which a program first reaches this queue. Exit, by the move's first run on the T14 or in a guest: it shows a program refused for it, and a socket's bound is counted per next hop; or the owner rules the one queue is the bound. -- A link that goes down drops no waiting datagram by itself: every waiting one is handed to [ip] again at the next transmit opportunity, which refuses it for want of a route (`route.no-source-address`), and one for which the link is back before that opportunity waits for its next hop anew, where the readers' link-down rule (NUD-23) drops what waits at once. `s_udp_us_024_shard_a_waiting_datagram_whose_route_goes_is_dropped` holds the first as it stands, and `a_waiting_datagram_whose_link_returns_before_an_opportunity_waits_anew_and_leaves`, beside it in `toyos-net-shard/tests/udp.rs` and under no scenario's id, the second. Owner: the worker of the move. Exit, before the move: that second test's last assertion is inverted and finds the datagram that waited gone, or the specifications have it wait. -- `toyos_dns::Lookup::on_datagram` still takes a reply's source address and port, which the node's connected sockets have already matched and the node hands it from its own record of the query: the two parameters are read only by netd's resolver on smoltcp. Exit, at the move: they go with that resolver. +- A client's datagram to a broadcast address needs its socket's permission, where smoltcp sends it for every socket. The owner's ruling: "Of course carry the permission why would you even ask. The premise is and always has been to do things properly". The pipe ABI carries the permission as far as netd: `MsgType::UdpSetOption` with `OPT_BROADCAST` (`toyos/src/net.rs`), sent by std's `set_broadcast` (`sdk/std/sys/net/connection.rs`) and by libc's `setsockopt` of `SO_BROADCAST` (`userland/libc/src/socket.rs`), which keeps it for a socket not yet bound and sends it at `bind`; the word for a send refused for want of it is `ERR_PERMISSION_DENIED`, `ErrorKind::PermissionDenied` in std and `EACCES` in libc. The node enforces it, and netstack hands it the request (`Sockets::udp_set_option`, `userland/netstack/src/serve.rs`): `Node::udp_set_broadcast` hands the permission to `Udp::set_broadcast`, and a send without it is refused `udp.broadcast-not-permitted` and answered `Refused::PermissionDenied` (`userland/netstack/node/src/datagram.rs`), which `a_broadcast_needs_its_permission` (US-21, `userland/netstack/node/tests/datagram.rs`) holds, with both broadcast addresses read off the wire in link-broadcast frames once the socket holds it and refused again once it is taken back, and `each_refusal_is_answered_in_the_pipes_word_for_it` holds the word. The permission of the call goes with the datagram to the route it leaves by: [udp] queues it with what its socket held, a closed socket's included, and `Ip::send_udp` refuses a link broadcast to one without it, `ip.broadcast-not-permitted`, counted and logged, so a prefix a DHCP server renews under a held address between the call and the frame makes no broadcast of a datagram accepted for a host (`s_udp_us_021_shard_a_datagram_accepted_for_a_host_is_no_broadcast_after_a_renewal_narrows_the_prefix` and its two neighbours in `toyos-net-shard/tests/udp.rs`, and from a server's ACK `a_datagram_accepted_for_a_host_is_no_broadcast_after_the_server_renews_a_narrower_prefix` in the node's tests). The search is run. `grep -rnI -E 'SO_BROADCAST|INADDR_BROADCAST|255\.255\.255\.255|sendto|setsockopt|sys/socket\.h|socket\(' userland/doom/doomgeneric`, the one third-party C program an image builds, at the commit `userland/doom/build.rs` pins, exits 1 with no line; the same without `sys/socket\.h` over `tests/testcases` exits 1 with no line; `SO_BROADCAST|INADDR_BROADCAST|255\.255\.255\.255` over the trees of `src/libcxx.rs`'s `SOURCES` finds LLVM libc's own macro and its documentation and no caller; and in ToyOS's tree outside the stack's crates the option is named only by the two setters. The socket2 fork's setter reaches nobody: `issues/the-socket2-forks-so-broadcast-setter-does-nothing.md`. Exit: a guest test, `udp_broadcast_needs_its_permission`, reads `broadcast()` false, then true after `set_broadcast(true)`, on the socket and on a `try_clone` of it, which `broadcast()` answering a constant or a duplicate holding a value of its own turns red, and sends to the limited broadcast address before and after the set and is refused `PermissionDenied` only before, which std's setter reverted to `Ok(())` turns red; and a guest C case reads libc's setter, `socket`, `setsockopt(SO_BROADCAST)`, `bind`, `sendto` to the limited broadcast address, refused `EACCES` without the option and sent with it, and with the option set before `bind`, which `bind` no longer handing the kept option over turns red. The case runs the sequence without `bind` too, where the first `sendto` binds the socket and hands the kept option over; `tests/netcase/sendto_unbound.c` holds that send for a socket permitted to make it. +- The machine's name is told to the node twice, as `toyos_dhcp::HostName` in `Node::new` and as `toyos_mdns::Host` in `Node::answer_as`: the first owns its bytes and the second borrows them. netstack's is a constant, so neither takes a leak. Exit: one type carries the label to both. +- The machine's name is probed for and defended (RFC 6762 §8, §9) by `toyos-mdns` (`userland/netstack/mdns`), on the node, which netstack runs. What that does not yet meet. Every machine is named `toyos-t14` (`HOSTNAME`, `userland/netstack/src/main.rs`) and a responder that loses its name holds none, so of two on one network the second answers to no name: asked "When two machines on one network have the same ToyOS host name, what should the second one do?", the owner answered "Hold no name, say so (Recommended)". Asked on 2026-10-09 "A ToyOS machine that loses its .local name to a conflict stays nameless until its network link next returns, even after the other machine has left. Should it try for its name again by itself?", he answered "Retry on a slow interval (Recommended)". Built: a lost name is probed for again a minute after each loss (`RETRY_MS`, `userland/netstack/mdns/src/lib.rs`) and at once on a link after none, and is claimed by the first probing no host answers; RFC 6762 permits probing again at §8.1's rate ("wait five seconds after any failed probe attempt before trying again") and does not require it. Two packets from any host on the link still take a held name, a response to take it back to probing and a response under the probe, and one more a minute keeps it: no path back exists that a peer cannot hold shut, since a host that answers every probe is what a holder of the name is. The caller does not say how a message was addressed, so two rules that turn on it are not applied: a response is read however it came, where §6 reads a unicast one only within two seconds of a question that asked for one; and a message from a source outside the subnet is ignored even when it was sent to the group, where §11 deems that on the link "regardless of source IP address", so a host of the same name on another subnet of the link is never a conflict. Exit: each machine's name is its own, and a test with two guests on one network finds each by its name; and the caller says how a message was addressed, with a test in which a unicast response nobody asked for takes no name and one in which a response sent to the group from another subnet does. Not measured, each with its exit: the shipped responder's claim on the T14's wired card, which no `system.toml` in the tree gives netstack (when the outbound rows land, their boot shows the lease line, then the claim line, and no loss line); and the link's return on metal (the orchestrator's attended session with the owner, never automated). A link that goes and returns inside one pass of the Intel driver reaches the name as no change: `issues/a-link-that-goes-and-returns-inside-one-pass-of-the-intel-driver-is-reported-as-no-change.md`. +- A socket's datagrams that wait for a next hop wait in the one queue it has, `toyos_net_udp::limits::TX_DATAGRAMS` of them: with all 16 waiting for one next hop, the socket's sends to every other destination are refused `udp.tx-queue-full`, `Refused::ResourceExhausted` to a client, until that hop answers or [ip] gives it up, 3 s after its first request left (`BROADCAST_SOLICIT` requests a `RETRANS` apart). It is no regression by this track's earlier record of what ships, that smoltcp kept each datagram in its own socket until the neighbour answered. What a closed socket had accepted waits in the closed sender's `CLOSED_DATAGRAMS` the same way, so 16 datagrams of closed sockets that wait for one silent host have the next closes' datagrams discarded, `udp.tx-discarded-on-close`, for those 3 s: the line on US-53 among stage 3's departures. `s_udp_us_025_what_waits_for_a_next_hop_is_bounded_by_its_sockets_queue` holds the first as it stands. The move's guest runs showed no program refused for it: none of their jobs sends sixteen datagrams past a hop that does not answer. Owner: whoever reads the T14's first outbound run on this stack. Exit: a run on the T14 or in a guest shows a program refused for it, and a socket's bound is counted per next hop; or the owner rules the one queue is the bound. +- A link that goes down drops no waiting datagram by itself: every waiting one is handed to [ip] again at the next transmit opportunity, which refuses it for want of a route (`route.no-source-address`), and one for which the link is back before that opportunity waits for its next hop anew, where the readers' link-down rule (NUD-23) drops what waits at once. `s_udp_us_024_shard_a_waiting_datagram_whose_route_goes_is_dropped` holds the first as it stands, and `a_waiting_datagram_whose_link_returns_before_an_opportunity_waits_anew_and_leaves`, beside it in `toyos-net-shard/tests/udp.rs` and under no scenario's id, the second. Not done before the move, whose worker changed nothing below netstack. Owner: whoever next changes `toyos-net-udp`. Exit: that second test's last assertion is inverted and finds the datagram that waited gone, or the specifications have it wait. +- `toyos_dns::Lookup::on_datagram` still takes a reply's source address and port, which the node's connected sockets have already matched and the node hands it from its own record of the query: the resolver that read them left with smoltcp, and the node hands both from its own record. Not removed at the move, which left `toyos-dns` as it was. Exit: the two parameters and the comparison inside `on_datagram` go. - `node.address-refused` has no test: `toyos-dhcp` accepts no address or prefix [ip] refuses, by the same `toyos-net-wire` checks in both, so the refusal cannot be reached from the wire. Exit: the client hands [ip] a type that carries the check, and the counter goes. -The listener defects are this track's: `issues/a-handshake-nobody-finishes-holds-a-listeners-port-shut.md` and `issues/a-connect-between-two-accepts-is-reset.md`, on smoltcp until stage 5, and `issues/an-accept-that-never-reaches-netstack-strands-its-listener.md`, in std's accept; and `issues/a-listeners-tcp-nodelay-has-no-setter-in-std-or-the-forks-and-a-c-program-cannot-order-the-waiting-case.md`: the pipe ABI gives a listener its option in its bind's request and by `MsgType::TcpListenerSetOption` and an accept's answer its connection's (`TcpOptions`, `toyos/src/net.rs`), netd on smoltcp answers them but for `issues/a-handshake-reset-before-it-ends-hands-its-option-to-the-next-connection.md`, which the move ends with smoltcp, and the move maps the bind to `Node::listen`, whose `nodelay` the request's is, the set to `Node::set_listener_nodelay` and the answer's options to `Accepted`'s; what is left of its exit is libc's `poll` and the socket2 and mio forks. +The listener defects are this track's: `issues/a-connect-between-two-accepts-is-reset.md`, whose listener queues now and whose `lan_swap` is not restored, and `issues/an-accept-that-never-reaches-netstack-strands-its-listener.md`, in std's accept; and `issues/a-listeners-tcp-nodelay-has-no-setter-in-std-or-the-forks-and-a-c-program-cannot-order-the-waiting-case.md`: the pipe ABI gives a listener its option in its bind's request and by `MsgType::TcpListenerSetOption` and an accept's answer its connection's (`TcpOptions`, `toyos/src/net.rs`), netstack maps the bind to `Node::listen`, whose `nodelay` the request's is, the set to `Node::set_listener_nodelay` and the answer's options to `Accepted`'s (`userland/netstack/src/serve.rs`); what is left of its exit is libc's `poll` and the socket2 and mio forks. -Owed from the stage 3 specifications: by stage 5, the scenario for netd's mapping of UDP's refusals onto the pipe ABI, and `dhcp.renew-unroutable`, which netd counts where `toyos-net-udp` refuses the renewal `udp.no-route`; by IP hardening, the scenarios for fragment reassembly and path MTU discovery. +Owed from the stage 3 specifications: since stage 5, the scenario for netd's mapping of UDP's refusals onto the pipe ABI, and `dhcp.renew-unroutable`, which netd counts where `toyos-net-udp` refuses the renewal `udp.no-route`; by IP hardening, the scenarios for fragment reassembly and path MTU discovery. What stage 3 departs from its specifications: @@ -87,4 +86,4 @@ What `toyos-net-wire` does not yet meet: - Its corpus is the specification's vectors alone, so no header layout has an oracle independent of the reader. Exit: frames captured from slirp and the T14 join the corpus. - It departs from the wire specification: RT-06 sweeps frames and datagrams only, because a message-only vector has no length of its own to be cut against. Exit: the specification covers it. -Exit: `rg smoltcp` is empty outside `issues/`. +smoltcp is gone: `rg -i smoltcp` is empty outside `issues/`. Exit: every stage above is built, and every line above is closed or is an issue of its own. diff --git a/issues/toyos-records-carry-the-sticks-serial-the-cards-mac-and-the-resolvers-in-full.md b/issues/toyos-records-carry-the-sticks-serial-the-cards-mac-and-the-resolvers-in-full.md index 1042d69106..b94a426672 100644 --- a/issues/toyos-records-carry-the-sticks-serial-the-cards-mac-and-the-resolvers-in-full.md +++ b/issues/toyos-records-carry-the-sticks-serial-the-cards-mac-and-the-resolvers-in-full.md @@ -13,9 +13,9 @@ public carries them unless whoever pastes it masks them by hand: - the USB stick's serial number: `kernel/src/drivers/xhci/wait/msc.rs:1407`, on every bind, and `:1415`, when a disk comes back; -- the network card's MAC: `userland/netstack/src/main.rs:1605`; +- the network card's MAC: `userland/netstack/src/main.rs`, `main`; - the resolvers the lease named, which on the bench are the provider's public - ones: `userland/netstack/src/dhcp.rs:149`. + ones: `userland/netstack/src/main.rs`, `Leases::pass`. `src/sourcegate.rs` reads tracked files only. A pull request's body, a comment and a commit message are read by nothing before they are public. diff --git a/tests/common/qemu.rs b/tests/common/qemu.rs index 758ee8281a..6df69de4ce 100644 --- a/tests/common/qemu.rs +++ b/tests/common/qemu.rs @@ -685,6 +685,11 @@ pub enum Profile { /// `iommu_platform=on`, and the harness sets that only where a unit exists, /// so the guest's own negotiation comes out the other way here. HeadlessNoIommu, + /// [`Profile::Headless`] with QEMU's `e1000e` for its NIC, the 82574L + /// `toyos-i219` drives beside the T14's I219: the one guest in which + /// netstack runs its Intel driver, a ring of fifteen frames and a wake + /// for transmit room. + HeadlessE1000e, /// [`Profile::Headless`] with no USB controller: the machine whose NVMe /// disk is the only storage it has, and the only device its firmware can /// boot. @@ -731,6 +736,7 @@ impl Profile { Self::Virt | Self::VirtNoRng | Self::VirtEl2 | Self::VirtEl2NoVhe | Self::VirtTcg => Arch::Aarch64, Self::Headless | Self::HeadlessNoIommu + | Self::HeadlessE1000e | Self::HeadlessNoUsb | Self::Metal => Arch::X86_64, } @@ -815,6 +821,8 @@ impl Virtio { enum Nic { Absent, Virtio, + /// QEMU's model of the 82574L. + E1000e, } /// Everything a profile decides about the machine, in one table. A new @@ -937,6 +945,7 @@ impl Profile { rng: false, }, Self::HeadlessNoIommu => Shape { iommu: None, ..Self::Headless.shape() }, + Self::HeadlessE1000e => Shape { nic: Nic::E1000e, ..Self::Headless.shape() }, Self::HeadlessNoUsb => Shape { xhci: &[], usb: &[], ..Self::Headless.shape() }, } } @@ -2186,6 +2195,9 @@ fn qemu_command( "virtio-net-pci-non-transitional,netdev=net0{platform}" )); } + Nic::E1000e => { + qemu.arg("-netdev").arg("user,id=net0").arg("-device").arg("e1000e,netdev=net0"); + } } if shape.virtio.present() { if shape.virtio.sound() { diff --git a/tests/libc-arch/src/lib.rs b/tests/libc-arch/src/lib.rs index a6fa47e5df..fa2c1def04 100644 --- a/tests/libc-arch/src/lib.rs +++ b/tests/libc-arch/src/lib.rs @@ -9,7 +9,7 @@ //! `long double` widening against compiler-builtins', the errno codes against //! `include/errno.h`, IPv4 addresses and their texts against the host C //! library's, and the socket option rule against the host's `setsockopt` and -//! `getsockopt`. +//! `getsockopt`, and the rule that reads a stream's end. #[cfg(test)] extern crate alloc; @@ -48,6 +48,9 @@ mod sigmask; #[path = "../../../userland/libc/src/sockopt.rs"] mod sockopt; #[cfg(test)] +#[path = "../../../userland/libc/src/streamend.rs"] +mod streamend; +#[cfg(test)] #[path = "../../../userland/libc/src/strtonum.rs"] mod strtonum; #[cfg(test)] @@ -88,6 +91,8 @@ mod signal_masks; #[cfg(test)] mod socket_options; #[cfg(test)] +mod stream_ends; +#[cfg(test)] mod strtonum_differential; #[cfg(test)] mod text_differential; diff --git a/tests/libc-arch/src/stream_ends.rs b/tests/libc-arch/src/stream_ends.rs new file mode 100644 index 0000000000..d25480b320 --- /dev/null +++ b/tests/libc-arch/src/stream_ends.rs @@ -0,0 +1,30 @@ +//! libc's rule for a stream's end (`streamend.rs`): a read of 0 is the peer's FIN only while the +//! send pipe still has its reader, a write the send pipe refuses for its reader's leaving is a +//! reset, and nothing else either pipe says is read as one. + +use toyos_abi::syscall::SyscallError; + +use crate::streamend::{read_end, write_refused, Refusal}; + +/// Every refusal but the two a pipe answers on purpose. +const OTHERS: [SyscallError; 4] = + [SyscallError::InvalidArgument, SyscallError::PermissionDenied, SyscallError::BadAddress, SyscallError::Io]; + +#[test] +fn a_read_of_zero_is_the_peers_fin_only_while_the_send_pipe_is_read() { + assert_eq!(read_end(Ok(0)), Ok(())); + assert_eq!(read_end(Err(SyscallError::WouldBlock)), Ok(()), "a full send pipe still has its reader"); + assert_eq!(read_end(Err(SyscallError::Gone)), Err(Refusal::Reset)); + for other in OTHERS { + assert_eq!(read_end(Err(other)), Err(Refusal::Other), "{other:?}"); + } +} + +#[test] +fn a_write_the_send_pipe_refuses_for_its_reader_is_a_reset() { + assert_eq!(write_refused(SyscallError::Gone), Refusal::Reset); + assert_eq!(write_refused(SyscallError::WouldBlock), Refusal::Again); + for other in OTHERS { + assert_eq!(write_refused(other), Refusal::Other, "{other:?}"); + } +} diff --git a/tests/netcase/stream_ends.c b/tests/netcase/stream_ends.c new file mode 100644 index 0000000000..057811f9f1 --- /dev/null +++ b/tests/netcase/stream_ends.c @@ -0,0 +1,91 @@ +/* How libc tells a C client a stream ended: recv 0 at the peer's FIN after + its own SHUT_WR, ECONNRESET on a reset mid-stream and after SHUT_WR, EPIPE + for a send after SHUT_WR, and recv 0 after SHUT_RD. argv: the address and + the port of the harness's peer, whose first byte read names how it ends the + stream (`tests/toyos-rust-tests/src/stream_ends.rs`). No stream is closed: + libc's close of a socket ends the program + (`issues/libc-close-of-a-socket-ends-the-process.md`), and the job's exit + lets each go. */ +#include +#include +#include +#include +#include +#include +#include + +/* What the peer sends before its reset of 'R'. */ +#define AHEAD 65536 + +static int wrong; +static struct sockaddr_in peer; + +static void said(const char *what, long got, long want) { + printf("%s: %ld%s\n", what, got, got == want ? "" : " <-- WRONG"); + if (got != want) wrong++; +} + +/* A call that answered -1: the errno it set, named. */ +static void refused(const char *what, long got, int err, int want) { + printf("%s: %ld, errno %d%s\n", what, got, got < 0 ? err : 0, got < 0 && err == want ? "" : " <-- WRONG"); + if (got >= 0 || err != want) wrong++; +} + +/* A stream to the peer that has been told how to end it. */ +static int dial(char how) { + int fd = socket(AF_INET, SOCK_STREAM, 0); + if (fd < 0 || connect(fd, (struct sockaddr *)&peer, sizeof peer) != 0 || send(fd, &how, 1, 0) != 1) { + printf("stream_ends: the peer could not be dialled for '%c'\n", how); + exit(1); + } + return fd; +} + +/* Reads until `want` bytes or what ends them first: the bytes read. */ +static long take(int fd, long want) { + char buf[4096]; + long got = 0, r; + while (got < want && (r = recv(fd, buf, sizeof buf < (size_t)(want - got) ? sizeof buf : (size_t)(want - got), 0)) > 0) + got += r; + return got; +} + +int main(int argc, char **argv) { + if (argc != 3) return 2; + char c; + long r; + memset(&peer, 0, sizeof peer); + peer.sin_family = AF_INET; + peer.sin_port = htons((uint16_t)atoi(argv[2])); + said("the peer's address is one", inet_pton(AF_INET, argv[1], &peer.sin_addr), 1); + + int fd = dial('S'); + said("shut_first: the peer's answer", recv(fd, &c, 1, 0), 1); + said("shut_first: shutdown(SHUT_WR)", shutdown(fd, SHUT_WR), 0); + r = send(fd, "x", 1, 0); + refused("shut_first: a send after it", r, errno, EPIPE); + said("shut_first: the peer's four bytes after it", take(fd, 4), 4); + said("shut_first: then the peer's FIN", recv(fd, &c, 1, 0), 0); + + fd = dial('R'); + said("reset_mid_stream: the bytes ahead of it", take(fd, AHEAD), AHEAD); + said("reset_mid_stream: the answer", send(fd, "k", 1, 0), 1); + r = recv(fd, &c, 1, 0); + refused("reset_mid_stream: then", r, errno, ECONNRESET); + + fd = dial('D'); + said("reset_after_half_close: shutdown(SHUT_WR)", shutdown(fd, SHUT_WR), 0); + r = recv(fd, &c, 1, 0); + refused("reset_after_half_close: then", r, errno, ECONNRESET); + + fd = dial('B'); + said("shut_rd: shutdown(SHUT_RD) with the peer's three bytes on their way", shutdown(fd, SHUT_RD), 0); + said("shut_rd: then recv", recv(fd, &c, 1, 0), 0); + + if (wrong) { + printf("stream_ends: %d wrong\n", wrong); + return 1; + } + printf("stream_ends: ok\n"); + return 0; +} diff --git a/tests/netcase/system.toml b/tests/netcase/system.toml index 4aaa4e8995..e3bbd82f85 100644 --- a/tests/netcase/system.toml +++ b/tests/netcase/system.toml @@ -24,7 +24,7 @@ syscap = ["logread"] [programs.netstack] service = true serves = ["netstack"] -devices = ["pci:1af4:1041"] +devices = ["pci:1af4:1041", "pci:8086:10d3"] # `netstack` because `netstack_socket_churn` connects through it and reads its # counts. diff --git a/tests/toyos-rust-tests/src/bin/netstack_socket_churn.rs b/tests/toyos-rust-tests/src/bin/netstack_socket_churn.rs index f6d379b283..0534d66d57 100644 --- a/tests/toyos-rust-tests/src/bin/netstack_socket_churn.rs +++ b/tests/toyos-rust-tests/src/bin/netstack_socket_churn.rs @@ -1,7 +1,7 @@ //! Piped TCP connections that end with no close request from their client, as //! a client that died leaves them: once netstack has let each connection go, //! its stream count is what it was before the first, and so is its count of -//! sockets no table entry names. +//! places held, once the stack has finished each connection alone. //! //! argv[1] is the port of the harness's host server, which ends every //! connection it accepts at once. Each round reads the peer's end of stream, @@ -9,19 +9,36 @@ //! connections to return. //! //! argv[2] is the port of a second, which holds every connection it accepts -//! and reads nothing. A client that shut its sending half down and left is one -//! whose send pipe netstack never reads again, so nothing but the kernel's -//! word on that pipe's other end can tell netstack it is gone: each such -//! connection is counted ownerless, with the pipe empty and with bytes in it. -//! -//! On the same server, a client that moves netstack a handle that is no pipe -//! end where its receive end belongs, and keeps its send end: the kernel +//! and reads nothing. On it, a client that moves netstack a handle that is no +//! pipe end where its receive end belongs, and keeps its send end: the kernel //! refuses netstack's watch of it, and that refusal is all that ends the //! connection. +//! +//! On the same server, a client that shut its sending half down and left: +//! netstack let its send pipe go at the shutdown and its peer sends nothing, +//! so nothing but the kernel's word on its receive pipe's other end can tell +//! netstack it is gone. +//! +//! A listener and a datagram socket whose owner drops its pipe ends with no +//! close request, and no peer near either: the kernel's word on the pipe's +//! other end is all that ends them, and netstack's count of each returns. +//! +//! argv[3] is the port of a third, which answers each datagram with itself. A +//! receive is asked of netstack before any datagram has been sent, so it waits +//! there, and is answered by the datagram the server sends back; a second +//! receive on that socket while the first waits is refused. +//! +//! Last, connects to the holding server until netstack refuses one: it holds +//! as many places as it said it has, and the connect past them is answered +//! `ResourceExhausted`. Last because the stack keeps each of those +//! connections' places until it has finished them with a peer that never +//! ends its half. use std::time::{Duration, Instant}; -use toyos::net::{MsgType, NetstackConn, TcpConnectPipedRequest, TcpConnectResponse}; +use toyos::net::{ + MsgType, NetError, NetstackConn, TcpConnectPipedRequest, TcpConnectResponse, UdpRecvFromRequest, UdpRecvResponse, +}; use toyos::OwnedHandle; use toyos_inspect::{Value, NET}; @@ -30,9 +47,9 @@ const HOST: [u8; 4] = [10, 0, 2, 2]; const ROUNDS: usize = 4; -/// A hang ceiling on netstack letting one ended connection go, or counting -/// one ownerless: it does the first on the pass after the peer acknowledges -/// its FIN, and the second on the pass the client's leaving wakes. +/// A hang ceiling on netstack letting one ended connection go: it does so on +/// the pass the client's leaving wakes, and the stack finishes the connection +/// when the peer acknowledges its FIN. const LET_GO: Duration = Duration::from_secs(20); fn count(key: &str) -> u64 { @@ -43,25 +60,85 @@ fn count(key: &str) -> u64 { } } -/// A connection the peer holds open, whose client shuts its sending half down, -/// leaves `unread` in a send pipe netstack no longer reads, and is gone. -fn left_after_a_shutdown(port: u16, unread: &[u8]) { - let ownerless = count("net.piped.ownerless"); +/// Asks netstack for `key` until it is `wanted`: each question is a round +/// trip netstack answers on a pass of its own. +fn returns(key: &str, wanted: u64, what: &str) { + let asked = Instant::now(); + loop { + let found = count(key); + if found == wanted { + return; + } + assert!(asked.elapsed() < LET_GO, "netstack's {key} is {found} and not {wanted} {LET_GO:?} after {what}"); + } +} + +/// A connection the peer holds open, whose client shuts its sending half down +/// and is gone. +fn left_after_a_shutdown(port: u16) { + let live = count("net.piped.live"); let conn = toyos::net::tcp_connect(HOST, port, 0).unwrap_or_else(|e| panic!("the holding server: {e:?}")); + assert_eq!(count("net.piped.live"), live + 1, "netstack counts the connection it just answered"); toyos::net::tcp_shutdown(conn.socket_id, 1).unwrap_or_else(|e| panic!("shutting the sending half: {e:?}")); - if !unread.is_empty() { - assert_eq!(conn.tx.write_nonblock(unread), Ok(unread.len()), "bytes into a send pipe nobody reads"); - } drop(conn); - let asked = Instant::now(); - while count("net.piped.ownerless") != ownerless + 1 { - assert!( - asked.elapsed() < LET_GO, - "netstack was not told a client that shut its sending half down, left {} byte(s) in its send pipe \ - and dropped both ends is gone", - unread.len() - ); - } + returns("net.piped.live", live, "a client that shut its sending half down dropped both its ends"); +} + +fn owners_that_left_are_let_go() { + let (listeners, udp) = (count("net.sockets.listeners"), count("net.sockets.udp")); + let listener = toyos::net::tcp_bind([0, 0, 0, 0], 0).unwrap_or_else(|e| panic!("a listener: {e:?}")); + let socket = toyos::net::udp_bind([0, 0, 0, 0], 0).unwrap_or_else(|e| panic!("a datagram socket: {e:?}")); + assert_eq!(count("net.sockets.listeners"), listeners + 1, "netstack counts the listener it just answered"); + assert_eq!(count("net.sockets.udp"), udp + 1, "netstack counts the datagram socket it just answered"); + // Their pipe ends, and no close request. + drop(listener); + drop(socket); + returns("net.sockets.listeners", listeners, "a listener's owner dropped its wake pipe"); + returns("net.sockets.udp", udp, "a datagram socket's owner dropped its pipes"); +} + +fn a_receive_that_waits_is_answered(port: u16) { + const SAID: &[u8] = b"a datagram for a receive that waits"; + let socket = toyos::net::udp_bind([0, 0, 0, 0], 0).unwrap_or_else(|e| panic!("a datagram socket: {e:?}")); + let waiting = NetstackConn::connect() + .and_then(|netstack| { + netstack.request(MsgType::UdpRecvFrom, &UdpRecvFromRequest { socket_id: socket.socket_id.0, max_len: 64 }) + }) + .unwrap_or_else(|e| panic!("asking for a datagram: {e:?}")); + // netstack reads its clients' requests in the order they connected, so + // this answer says the receive above is waiting there. + count("net.sockets.udp"); + let second = toyos::net::udp_recv_from(socket.socket_id, 64); + assert_eq!(second.err(), Some(NetError::ResourceExhausted), "a second receive on a socket whose first waits"); + assert_eq!(socket.tx.write_nonblock(SAID), Ok(SAID.len()), "a datagram into the socket's send pipe"); + toyos::net::udp_send_to(socket.socket_id, HOST, port, SAID.len() as u16) + .unwrap_or_else(|e| panic!("sending to the answering server: {e:?}")); + let answer: UdpRecvResponse = waiting.response().unwrap_or_else(|e| panic!("the receive that waited: {e:?}")); + assert_eq!((answer.addr, answer.port, usize::from(answer.len)), (HOST, port, SAID.len()), "the answer's source and length"); + let mut back = [0u8; 64]; + assert_eq!(socket.rx.read(&mut back), Ok(SAID.len()), "the answer's bytes in the socket's receive pipe"); + assert_eq!(&back[..SAID.len()], SAID); + toyos::net::udp_close(socket.socket_id).unwrap_or_else(|e| panic!("closing the datagram socket: {e:?}")); +} + +fn a_connect_past_the_places_is_refused(port: u16) { + let (max, held) = (count("net.places.max"), count("net.places.held")); + let mut kept = Vec::new(); + let refused = loop { + match toyos::net::tcp_connect(HOST, port, 0) { + Ok(conn) => kept.push(conn), + Err(refusal) => break refusal, + } + assert!(kept.len() as u64 <= max, "netstack holds more connections than the {max} places it said it has"); + }; + assert_eq!(refused, NetError::ResourceExhausted, "the connect past netstack's places"); + assert_eq!( + kept.len() as u64, + max - held, + "netstack said it has {max} places with {held} held, and refused the connect after {} more", + kept.len() + ); + assert_eq!(count("net.places.held"), max, "every place is held where the connect was refused"); } /// A connection the peer holds open, whose client keeps its send end and moved @@ -84,15 +161,11 @@ fn a_receive_end_that_is_no_pipe_end_is_refused(port: u16) { }) .and_then(|pending| pending.response()) .unwrap_or_else(|e| panic!("the holding server, with an acceptor for a receive end: {e:?}")); - let id = connected.socket_id; - let asked = Instant::now(); - while count("net.piped.live") != live { - assert!( - asked.elapsed() < LET_GO, - "netstack still counts connection {id} live {LET_GO:?} after the kernel refused the watch of its \ - receive end, which is no pipe end" - ); - } + returns( + "net.piped.live", + live, + &format!("the kernel refused the watch of connection {}'s receive end, which is no pipe end", connected.socket_id), + ); drop(kept); } @@ -101,11 +174,11 @@ fn main() { std::env::args() .nth(at) .and_then(|p| p.parse().ok()) - .expect("usage: netstack_socket_churn ") + .expect("usage: netstack_socket_churn ") }; - let (port, holding) = (port(1), port(2)); + let (port, holding, answering) = (port(1), port(2), port(3)); let (streams, live) = (count("net.sockets.tcp"), count("net.piped.live")); - let untabled = count("net.sockets.untabled"); + let held = count("net.places.held"); for round in 1..=ROUNDS { let conn = toyos::net::tcp_connect(HOST, port, 0) .unwrap_or_else(|e| panic!("connection {round} to the host server: {e:?}")); @@ -113,14 +186,7 @@ fn main() { assert_eq!(conn.rx.read(&mut byte), Ok(0), "connection {round}: the host ends it without a byte"); // Both pipe ends, and no close request. drop(conn); - let asked = Instant::now(); - // Each question is a round trip netstack answers on a pass of its own. - while count("net.piped.live") != live { - assert!( - asked.elapsed() < LET_GO, - "netstack still counts connection {round} live {LET_GO:?} after both its ends closed" - ); - } + returns("net.piped.live", live, &format!("both ends of connection {round} closed")); } let left = count("net.sockets.tcp"); println!( @@ -128,15 +194,16 @@ fn main() { them and holds {left} after" ); assert_eq!(left, streams, "netstack keeps a stream for a connection it let go"); - assert_eq!( - count("net.sockets.untabled"), - untabled, - "netstack keeps the socket of a connection whose table entry it let go" - ); + returns("net.places.held", held, "the last of the connections it let go ended on the wire"); a_receive_end_that_is_no_pipe_end_is_refused(holding); println!("netstack_socket_churn: a connection whose receive end is no pipe end was reset"); - left_after_a_shutdown(holding, &[]); - left_after_a_shutdown(holding, b"never read"); - println!("netstack_socket_churn: two clients that shut down and left are ownerless"); + left_after_a_shutdown(holding); + println!("netstack_socket_churn: a client that shut down and left was let go"); + owners_that_left_are_let_go(); + println!("netstack_socket_churn: a listener and a datagram socket whose owner left were let go"); + a_receive_that_waits_is_answered(answering); + println!("netstack_socket_churn: a receive that waited was answered by its datagram"); + a_connect_past_the_places_is_refused(holding); + println!("netstack_socket_churn: the connect past netstack's places was refused"); println!("netstack_socket_churn: ok"); } diff --git a/tests/toyos-rust-tests/src/bin/netstack_streams.rs b/tests/toyos-rust-tests/src/bin/netstack_streams.rs new file mode 100644 index 0000000000..cba943fbab --- /dev/null +++ b/tests/toyos-rust-tests/src/bin/netstack_streams.rs @@ -0,0 +1,126 @@ +//! Streams through netstack against a TCP nobody here wrote: the host +//! kernel's, behind QEMU's user network. +//! +//! argv[1] is the port of the harness's host server, which sends back every +//! byte it reads. The job writes [`BULK`] bytes of a sequence to it and shuts +//! its sending half down at once, while it reads them back, and compares each: +//! more than a pipe and both of the stack's buffers hold, so a byte lost, +//! repeated or moved where either side made the other wait shows as the first +//! that differs. After the last byte it reads the stream's end: the server's +//! FIN, which std reads as an end and not a reset behind the shutdown. +//! +//! argv[2] is a port the job listens on. It says so and takes no connection +//! until its listener's pipe has given up two wakes: the harness dials twice, +//! so the second connection arrives while the first waits to be accepted, and +//! both are then accepted, answered with the byte each sent and closed, which +//! each peer reads as its stream's end. + +use std::io::{Read, Write}; +use std::net::{Ipv4Addr, Shutdown, TcpStream}; +use std::time::{Duration, Instant}; + +use toyos::poller::{Poller, READABLE}; +use toyos_abi::syscall::SyscallError; + +/// The host, as QEMU's user network names it to a guest. +const HOST: Ipv4Addr = Ipv4Addr::new(10, 0, 2, 2); + +/// Twice a kernel pipe, so the client's pipe fills behind the stack's window. +const BULK: usize = 4 * 1024 * 1024; + +/// A hang ceiling on the two peers' wakes: the harness dials both as it reads +/// the line that says the listener waits. +const WOKEN: Duration = Duration::from_secs(60); + +/// The sequence's byte at `at`: no period a buffer's or a segment's size +/// divides. +fn byte(at: usize) -> u8 { + (at % 251) as u8 ^ (at / 251) as u8 +} + +fn bulk(port: u16) { + let mut reading = TcpStream::connect((HOST, port)).unwrap_or_else(|e| panic!("the echoing server: {e}")); + let mut writing = reading.try_clone().expect("a second handle on the stream"); + let writer = std::thread::spawn(move || { + let mut sent = 0; + let mut chunk = [0u8; 8192]; + while sent < BULK { + let len = chunk.len().min(BULK - sent); + for (i, b) in chunk[..len].iter_mut().enumerate() { + *b = byte(sent + i); + } + writing.write_all(&chunk[..len]).unwrap_or_else(|e| panic!("writing byte {sent} of the bulk: {e}")); + sent += len; + } + // At once: what the pipe still holds is the stack's to send first. + writing.shutdown(Shutdown::Write).expect("shutting the sending half"); + }); + let mut read = 0; + let mut chunk = [0u8; 8192]; + while read < BULK { + let len = reading.read(&mut chunk).unwrap_or_else(|e| panic!("reading byte {read} of the bulk: {e}")); + assert_ne!(len, 0, "the stream ended at byte {read} of the bulk"); + for (i, b) in chunk[..len].iter().enumerate() { + assert_eq!(*b, byte(read + i), "byte {} came back as another", read + i); + } + read += len; + } + writer.join().expect("the writer"); + match reading.read(&mut chunk) { + Ok(0) => {} + ended => panic!("the read after the bulk's last byte answered {ended:?} and not the stream's end"), + } + println!("netstack_streams: {BULK} bytes went out and came back as sent, then the stream's end"); +} + +fn two_peers(port: u16) { + let bound = toyos::net::tcp_bind([0, 0, 0, 0], port).unwrap_or_else(|e| panic!("listening on {port}: {e:?}")); + println!("netstack_streams: the listener waits for two peers"); + let poller = Poller::new(1); + let asked = Instant::now(); + let mut wakes = [0u8; 2]; + let mut woken = 0; + while woken < wakes.len() { + let Some(left) = WOKEN.checked_sub(asked.elapsed()) else { + // Which half is missing: a connection that never finished its + // handshake, or a wake for one that did. + let waiting: Vec<_> = (0..wakes.len()) + .map(|_| toyos::net::tcp_accept(bound.socket_id).map(|accepted| accepted.remote_port)) + .collect(); + panic!("the listener was woken for {woken} of two peers in {WOKEN:?}, and two accepts then answered {waiting:?}"); + }; + poller.watch(&bound.notify, READABLE, 0); + poller.wait(1, left.as_nanos() as u64, |_| {}); + match bound.notify.read_nonblock(&mut wakes[woken..]) { + Ok(0) => panic!("netstack ended the listener after {woken} wake(s)"), + Ok(n) => woken += n, + Err(SyscallError::WouldBlock) => {} + Err(e) => panic!("reading the listener's wakes: {e:?}"), + } + } + let mut answered = Vec::new(); + for _ in 0..wakes.len() { + let accepted = toyos::net::tcp_accept(bound.socket_id).unwrap_or_else(|e| panic!("an accept after its wake: {e:?}")); + let mut said = [0u8; 1]; + assert_eq!(accepted.rx.read(&mut said), Ok(1), "a peer's one byte"); + assert_eq!(accepted.tx.write(&said), Ok(1), "its answer"); + answered.push(said[0]); + toyos::net::tcp_close(accepted.socket_id).expect("closing an accepted stream"); + } + toyos::net::tcp_close(bound.socket_id).expect("closing the listener"); + answered.sort_unstable(); + assert_eq!(answered, *b"12", "each peer was accepted once"); + println!("netstack_streams: two peers that arrived before any accept were both accepted"); +} + +fn main() { + let port = |at: usize| -> u16 { + std::env::args() + .nth(at) + .and_then(|p| p.parse().ok()) + .expect("usage: netstack_streams ") + }; + bulk(port(1)); + two_peers(port(2)); + println!("netstack_streams: ok"); +} diff --git a/tests/toyos-rust-tests/src/bin/stream_ends_std.rs b/tests/toyos-rust-tests/src/bin/stream_ends_std.rs new file mode 100644 index 0000000000..3de8b6df58 --- /dev/null +++ b/tests/toyos-rust-tests/src/bin/stream_ends_std.rs @@ -0,0 +1,23 @@ +//! `stream_ends.rs`'s report, in a guest: one line each, then `ok` where it +//! is the one a host's TCP gives, and an exit of 1 where it is not. +//! +//! argv: the address and port of the harness's peer. + +#[path = "../stream_ends.rs"] +mod stream_ends; + +fn main() { + let args: Vec = std::env::args().collect(); + let [_, host, port] = &args[..] else { panic!("usage: stream_ends_std ") }; + let port = port.parse().unwrap_or_else(|e| panic!("the port {port}: {e}")); + let mut report = Vec::new(); + stream_ends::run(host, port, |line| { + println!("stream_ends_std: {line}"); + report.push(line); + }); + if report != stream_ends::EXPECTED { + println!("stream_ends_std: a host's TCP reports\n{}", stream_ends::EXPECTED.join("\n")); + std::process::exit(1); + } + println!("stream_ends_std: ok"); +} diff --git a/tests/toyos-rust-tests/src/stream_ends.rs b/tests/toyos-rust-tests/src/stream_ends.rs new file mode 100644 index 0000000000..3dccb26e64 --- /dev/null +++ b/tests/toyos-rust-tests/src/stream_ends.rs @@ -0,0 +1,126 @@ +//! Each way a stream ends, as std reports it to a client: the same code runs +//! in a guest against netstack and on the harness's host against the host +//! kernel's TCP, and the two reports are compared line for line. +//! +//! The peer is the harness's: a connection's first byte names what it does, +//! and `tests/netcase/stream_ends.c` dials it with the same bytes. Every end it +//! makes waits on something the client did, so no line depends on how long +//! anything took. +//! +//! Std alone: the harness compiles this file too. + +use std::io::{ErrorKind, Read, Write}; +use std::net::{Shutdown, TcpStream}; + +/// What the peer sends before the reset of `reset_mid_stream`. +pub const AHEAD: usize = 1 << 16; + +/// The peer sends its FIN before the client writes. +pub const PEER_FIN_FIRST: u8 = b'F'; +/// The peer sends [`AHEAD`] bytes, reads the client's one, then resets. +pub const RESET_MID_STREAM: u8 = b'R'; +/// The peer reads to the client's end, then resets. +pub const RESET_AFTER_HALF_CLOSE: u8 = b'D'; +/// The peer sends three bytes and its FIN. +pub const BOTH_CLOSE: u8 = b'B'; +/// The peer answers one byte; the client then shuts its sending half with +/// nothing pending, and the peer, at that FIN, sends four bytes and its own. +pub const SHUT_FIRST: u8 = b'S'; + +/// The report a host's TCP gives, and so the one netstack owes. +pub const EXPECTED: [&str; 6] = [ + "peer_fin_first: the read Ok(0), then a write Ok(5)", + "reset_mid_stream: 65536 bytes came as sent, the answer Ok(1), then Err(ConnectionReset)", + "reset_after_half_close: shut the sending half: Ok, then Err(ConnectionReset)", + "both_close: 3 bytes came as sent, then Ok(0)", + "both_close: shut the sending half: Ok, then Ok(0)", + "shut_first: the answer Ok(1), shut the sending half: Ok, 4 bytes came as sent, then Ok(0)", +]; + +/// Byte `i` of what the peer sends ahead of a reset: a pattern no shift or +/// cut reproduces. +pub fn pattern(i: usize) -> u8 { + (i % 251) as u8 +} + +fn kind(result: std::io::Result) -> String { + match result { + Ok(_) => "Ok".to_string(), + Err(e) => format!("Err({:?})", e.kind()), + } +} + +fn count(result: std::io::Result) -> String { + match result { + Ok(n) => format!("Ok({n})"), + Err(e) => format!("Err({:?})", e.kind()), + } +} + +/// What one more read answers: the end, or a byte that is not one. +fn the_end(stream: &mut TcpStream) -> String { + count(stream.read(&mut [0u8; 1])) +} + +/// Reads until `want` bytes, the end or a refusal: how many matched +/// `expect`, and what stopped it, if anything did before `want`. +fn take(stream: &mut TcpStream, want: usize, expect: impl Fn(usize) -> u8) -> (String, Option) { + let mut buf = vec![0u8; 65536]; + let mut got = 0; + while got < want { + let room = buf.len().min(want - got); + match stream.read(&mut buf[..room]) { + Ok(0) => return (format!("{got} bytes came"), Some("Ok(0)".to_string())), + Ok(n) => { + if let Some(at) = (0..n).find(|&i| buf[i] != expect(got + i)) { + return (format!("byte {} differs after {got} bytes", got + at), None); + } + got += n; + } + Err(e) if e.kind() == ErrorKind::Interrupted => {} + Err(e) => return (format!("{got} bytes came"), Some(format!("Err({:?})", e.kind()))), + } + } + (format!("{got} bytes came"), None) +} + +fn dial(host: &str, port: u16, what: u8) -> TcpStream { + let mut stream = TcpStream::connect((host, port)).unwrap_or_else(|e| panic!("the peer at {host}:{port}: {e}")); + stream.write_all(&[what]).unwrap_or_else(|e| panic!("naming the end to the peer: {e}")); + stream +} + +/// Every end, in order: one line each as [`EXPECTED`] spells them. +pub fn run(host: &str, port: u16, mut say: impl FnMut(String)) { + let mut stream = dial(host, port, PEER_FIN_FIRST); + let read = the_end(&mut stream); + say(format!("peer_fin_first: the read {read}, then a write {}", count(stream.write(b"after")))); + + let mut stream = dial(host, port, RESET_MID_STREAM); + let line = match take(&mut stream, AHEAD, pattern) { + (came, Some(end)) => format!("{came}, then {end}"), + (came, None) => { + let answer = count(stream.write(b"k")); + format!("{came} as sent, the answer {answer}, then {}", the_end(&mut stream)) + } + }; + say(format!("reset_mid_stream: {line}")); + + let mut stream = dial(host, port, RESET_AFTER_HALF_CLOSE); + let shut = kind(stream.shutdown(Shutdown::Write)); + say(format!("reset_after_half_close: shut the sending half: {shut}, then {}", the_end(&mut stream))); + + let mut stream = dial(host, port, BOTH_CLOSE); + let (came, end) = take(&mut stream, 3, |i| b"bye"[i]); + let end = end.unwrap_or_else(|| the_end(&mut stream)); + say(format!("both_close: {came} as sent, then {end}")); + let shut = kind(stream.shutdown(Shutdown::Write)); + say(format!("both_close: shut the sending half: {shut}, then {}", the_end(&mut stream))); + + let mut stream = dial(host, port, SHUT_FIRST); + let answer = the_end(&mut stream); + let shut = kind(stream.shutdown(Shutdown::Write)); + let (came, end) = take(&mut stream, 4, |i| b"late"[i]); + let end = end.unwrap_or_else(|| the_end(&mut stream)); + say(format!("shut_first: the answer {answer}, shut the sending half: {shut}, {came} as sent, then {end}")); +} diff --git a/tests/toyos.rs b/tests/toyos.rs index fc6d31ef4f..54fdff5ff1 100644 --- a/tests/toyos.rs +++ b/tests/toyos.rs @@ -2,6 +2,8 @@ extern crate toyos_build; mod common; +#[path = "toyos-rust-tests/src/stream_ends.rs"] +mod stream_ends; use std::collections::{BTreeMap, BTreeSet}; use std::fs; @@ -134,9 +136,15 @@ const RUST_SKIP: &[&str] = &[ // It needs a NIC in front of netstack and a host server behind it: // `netstack_socket_churn` runs it on `tests/netcase`. "netstack_socket_churn", + // It needs a NIC in front of netstack, a host server behind it and a host + // that dials its listener: `netstack_streams` runs it on `tests/netcase`. + "netstack_streams", // It needs a host that dials its listeners when it says they wait: // `libc_sockets` runs it on `tests/netcase`. "nodelay_accepted", + // It needs a host peer that ends each stream as the stream asks: + // `libc_sockets` runs it on `tests/netcase`. + "stream_ends_std", // It asserts nothing at all: it holds `dump_nmi_probe`'s boot open for // twenty seconds. On a shared boot it would be twenty seconds of nothing. "lan_hold", @@ -282,9 +290,18 @@ const MACHINE_TESTS: &[&str] = &[ // connections: netstack is one binary that owns its NIC, with no host // build, and the T14's peer is the bench's network. "netstack_socket_churn", + // The stack against a TCP its writers did not write, the host kernel's + // behind QEMU's user network, through the kernel's own pipes and each of + // netstack's two drivers: netstack has no host build, the stack's host + // tests answer it from the tests' own script, and the T14's bench may + // open no peer that echoes megabytes or dials in. + "netstack_streams", + "netstack_streams_e1000e", // What libc's and std's socket calls ask of netstack, read back from a - // peer that answers: the calls are requests on netstack's port, netstack - // has no host build, and the T14's peer is the bench's network. + // peer that answers, and each way a stream ends as libc and std read it: + // the calls are requests on netstack's port, an end is what netstack, the + // kernel's pipes and the library read together, netstack has no host + // build, and the T14's bench has no peer that answers or resets on cue. "libc_sockets", // The nested-NMI report is a raw write to the 16550, which the T14 does not // have. @@ -3002,21 +3019,23 @@ fn scanout_wc(console: &str) -> Result<(), String> { /// netstack's stream count returns once connections that ended without their /// client's close request are let go, a client that left a connection its /// peer holds is counted gone, and a connection whose receive end the kernel -/// refuses netstack's watch of is reset. One host server here ends each -/// connection it accepts at once and one holds each; the guest's comparisons -/// are the verdict. +/// refuses netstack's watch of is reset; a listener and a datagram socket +/// whose owner left are let go, a receive that waits is answered by its +/// datagram, and the connect past netstack's places is refused. One host +/// server here ends each connection it accepts at once, one holds each and +/// one answers each datagram; the guest's comparisons are the verdict. fn netstack_socket_churn() -> Result<(), String> { const JOB: &str = "netstack_socket_churn"; let server = std::net::TcpListener::bind(("127.0.0.1", 0)).map_err(|e| format!("the host server: {e}"))?; let port = server.local_addr().map_err(|e| format!("the host server's port: {e}"))?.port(); // Ends with the process: a guest that never dials leaves it in `accept`. thread::spawn(move || server.incoming().for_each(drop)); - let holding = holding_server()?; + let (holding, answering) = (holding_server()?, answering_server()?); let bin = qemu::build_toyos_bin(qemu::SUITE_ARCH, &compile::repo_root().join("tests/toyos-rust-tests"), JOB); let mut qemu = boot_netcase(&[], &[(JOB.to_string(), bin)], BootOptions::default())?; - let result = - qemu.run_test(&format!("test_rs_netstack_socket_churn {port} {holding}"), Duration::from_secs(120)); + let result = qemu + .run_test(&format!("test_rs_netstack_socket_churn {port} {holding} {answering}"), Duration::from_secs(120)); if let Some(why) = &result.error { return Err(format!("{why}\nthe job said:\n{}", result.stdout)); } @@ -3029,6 +3048,118 @@ fn netstack_socket_churn() -> Result<(), String> { Ok(()) } +/// What one job of netstack's said, as its verdict: it ended by itself, with +/// 0, and with its last line. +fn job_ok(job: &str, result: &qemu::TestResult) -> Result<(), String> { + if let Some(why) = &result.error { + return Err(format!("{why}\nthe job said:\n{}", result.stdout)); + } + if result.exit_code != Some(0) { + return Err(format!("the job ended {:?}:\n{}", result.exit_code, result.stdout)); + } + if !result.stdout.lines().any(|l| l.trim_end().ends_with(&format!("{job}: ok"))) { + return Err(format!("the guest never said it was done:\n{}", result.stdout)); + } + Ok(()) +} + +/// Megabytes out to the host kernel's TCP and back unchanged, and two host +/// peers that dial the guest's listener before it accepts either, both +/// accepted, answered and read their stream's end: through netstack on +/// `profile`'s card. The host server here sends back what it reads; the +/// guest's comparisons and each peer's answer are the verdict. +fn netstack_streams(profile: qemu::Profile) -> Result<(), String> { + use std::io::Read; + const JOB: &str = "netstack_streams"; + /// The port the job listens on in the guest, which nothing else on its + /// boot binds. + const LISTENER: u16 = 7010; + const WAITS: &str = "netstack_streams: the listener waits for two peers"; + /// A hang ceiling on a peer's answer, which the job wrote before it + /// ended. + const ANSWERED: Duration = Duration::from_secs(30); + let echo = std::net::TcpListener::bind(("127.0.0.1", 0)).map_err(|e| format!("the echoing server: {e}"))?; + let port = echo.local_addr().map_err(|e| format!("the echoing server's port: {e}"))?.port(); + // Ends with the process: a guest that never dials leaves it in `accept`. + thread::spawn(move || { + for stream in echo.incoming().flatten() { + thread::spawn(move || { + let mut back = stream.try_clone().expect("a second handle on an accepted stream"); + // A guest that resets the stream ends its echo, and reads the + // loss itself. + let _ = std::io::copy(&mut &stream, &mut back); + }); + } + }); + + let bin = qemu::build_toyos_bin(qemu::SUITE_ARCH, &compile::repo_root().join("tests/toyos-rust-tests"), JOB); + let options = BootOptions { qmp: true, profile, ..Default::default() }; + let mut qemu = boot_netcase(&[], &[(JOB.to_string(), bin)], options)?; + let [to_listener] = forwards_into(&qemu, [LISTENER])?; + let mut monitor = qemu::QmpMonitor::open(qemu.qmp_socket()); + // The connections QEMU's user network has taken off the forwarded port + // and is carrying to the guest's listener: every row of its table that + // names the guest's port, but the forward's own. + let mut carried = move || { + let table = monitor.human("info usernet"); + table + .lines() + .map(|row| row.split_whitespace().collect::>()) + .filter(|row| row.first().is_some_and(|kind| kind.starts_with("TCP[") && *kind != "TCP[HOST_FORWARD]")) + .filter(|row| [3, 5].iter().any(|&at| row.get(at) == Some(&LISTENER.to_string().as_str()))) + .count() + }; + // Each peer says its own byte as soon as it has dialled, and the next + // dials once QEMU carries this one: its forward queues one connection, + // and resets a dial that arrives while one is queued. + let mut dial = |said: u8| -> Result { + let before = carried(); + let mut peer = std::net::TcpStream::connect(("127.0.0.1", to_listener)).map_err(|e| e.to_string())?; + peer.write_all(&[said]).map_err(|e| e.to_string())?; + peer.set_read_timeout(Some(ANSWERED)).map_err(|e| e.to_string())?; + let dialled = Instant::now(); + // QEMU says nothing when it carries a connection, so its table is + // asked again after a wait that doubles to 64 ms. + let mut pause = Duration::from_millis(1); + while carried() == before { + if dialled.elapsed() > ANSWERED { + return Err(format!("QEMU's user network took no connection off its forward in {ANSWERED:?}")); + } + thread::sleep(pause); + pause = (pause * 2).min(Duration::from_millis(64)); + } + Ok(peer) + }; + let mut peers = Vec::new(); + let result = + qemu.run_test_paced(&format!("test_rs_{JOB} {port} {LISTENER}"), Duration::from_secs(240), |_, line| { + if line.trim_end().ends_with(WAITS) { + peers.extend(b"12".iter().map(|&said| dial(said).map(|peer| (said, peer)))); + } + }); + // A dial that failed first: it is why the job's wakes did not come. + let peers: Vec<_> = peers.into_iter().collect::>().map_err(|e| { + format!("the host could not dial the guest's listener: {e}\nthe job said:\n{}", result.stdout) + })?; + job_ok(JOB, &result)?; + if peers.len() != 2 { + return Err(format!("the job never said its listener waits:\n{}", result.stdout)); + } + for (said, mut peer) in peers { + let mut answer = [0u8; 1]; + peer.read_exact(&mut answer).map_err(|e| format!("the peer that said {:?} was not answered: {e}", said as char))?; + if answer[0] != said { + return Err(format!("the peer that said {:?} was answered {:?}", said as char, answer[0] as char)); + } + // The guest closed the stream behind its answer. + match peer.read(&mut answer) { + Ok(0) => {} + ended => return Err(format!("the peer that said {:?} read {ended:?} where its stream ends", said as char)), + } + } + Ok(()) +} + /// A host server that holds each connection it accepts and reads none of it, /// for as long as the process lives: its port. A guest that never dials /// leaves it in `accept`. @@ -3039,6 +3170,81 @@ fn holding_server() -> Result { Ok(port) } +/// The peer `stream_ends` dials, for as long as the process lives: its port. +/// Each connection's first byte names how it ends, and each end waits on what +/// the client did before it. +fn ending_server() -> Result { + use std::io::Read; + use std::os::fd::AsRawFd; + /// A close that sends a reset (RFC 9293 §3.10.4's ABORT): a zero linger. + fn reset(stream: std::net::TcpStream) { + let linger = libc::linger { l_onoff: 1, l_linger: 0 }; + // SAFETY: a live socket, and an option of the size its type says. + let set = unsafe { + libc::setsockopt( + stream.as_raw_fd(), + libc::SOL_SOCKET, + libc::SO_LINGER, + (&raw const linger).cast(), + size_of::() as libc::socklen_t, + ) + }; + assert_eq!(set, 0, "a zero linger: {}", std::io::Error::last_os_error()); + } + fn serve(mut stream: std::net::TcpStream) { + let mut what = [0u8; 1]; + if stream.read_exact(&mut what).is_err() { + return; + } + match what[0] { + stream_ends::PEER_FIN_FIRST => {} + stream_ends::RESET_MID_STREAM => { + let ahead: Vec = (0..stream_ends::AHEAD).map(stream_ends::pattern).collect(); + if stream.write_all(&ahead).is_ok() && stream.read_exact(&mut what).is_ok() { + reset(stream); + } + } + stream_ends::RESET_AFTER_HALF_CLOSE => { + if stream.read_to_end(&mut Vec::new()).is_ok() { + reset(stream); + } + } + stream_ends::BOTH_CLOSE => { + let _ = stream.write_all(b"bye"); + } + stream_ends::SHUT_FIRST => { + let _ = stream + .write_all(b".") + .and_then(|()| stream.read_to_end(&mut Vec::new())) + .and_then(|_| stream.write_all(b"late")); + } + _ => {} + } + } + let peer = std::net::TcpListener::bind(("127.0.0.1", 0)).map_err(|e| format!("the ending peer: {e}"))?; + let port = peer.local_addr().map_err(|e| format!("the ending peer's port: {e}"))?.port(); + thread::spawn(move || { + for stream in peer.incoming().flatten() { + thread::spawn(move || serve(stream)); + } + }); + Ok(port) +} + +/// A host server that answers each datagram with itself, for as long as the +/// process lives: its port. +fn answering_server() -> Result { + let echo = std::net::UdpSocket::bind(("127.0.0.1", 0)).map_err(|e| format!("the answering server: {e}"))?; + let port = echo.local_addr().map_err(|e| format!("the answering server's port: {e}"))?.port(); + thread::spawn(move || { + let mut datagram = [0u8; 64]; + while let Ok((len, from)) = echo.recv_from(&mut datagram) { + echo.send_to(&datagram[..len], from).expect("answer a datagram"); + } + }); + Ok(port) +} + /// Boot `tests/netcase` with these binaries staged, to netstack's lease, since /// its jobs name their peer by an address, and then to the claim of its name: /// a responder whose probing never ends leaves the machine nameless, and @@ -3049,7 +3255,7 @@ fn boot_netcase( options: BootOptions, ) -> Result { const LEASED: &str = "netstack: DHCP: lease "; - /// `userland/netstack/src/mdns.rs`'s line for a probing no host answered. + /// `userland/netstack/src/serve.rs`'s line for a probing no host answered. const CLAIMED: &str = "netstack: mDNS: no host answered for "; let case = compile::repo_root().join("tests/netcase"); let mut qemu = QemuInstance::boot_with_options(&case, c_bins, rust_bins, options); @@ -3090,40 +3296,39 @@ fn forwards_into(qemu: &QemuInstance, guest_ports: [u16; N]) -> /// one boot of `tests/netcase`: each job dials a host server that holds what /// it accepts, or sends to one that answers each datagram with itself, at the /// address the guest's network gives the host; and each time a job says its -/// listeners wait, the host dials them through the ports QEMU forwards. A -/// job's own comparisons are its verdict. +/// listeners wait, the host dials them through the ports QEMU forwards; and +/// `stream_ends` in C and in std dial a host peer that ends each stream as +/// its first byte names, std's report first read on this host's TCP. A job's +/// own comparisons are its verdict. fn libc_sockets() -> Result<(), String> { const HOST: &str = "10.0.2.2"; /// The ports `nodelay_kept` and then `nodelay_accepted` listen on in the /// guest, which nothing else on its boot binds. const LISTENERS: [u16; 4] = [7001, 7002, 7003, 7004]; - const RUST_JOB: &str = "nodelay_accepted"; - let holding = holding_server()?; - let echo = std::net::UdpSocket::bind(("127.0.0.1", 0)).map_err(|e| format!("the answering server: {e}"))?; - let answering = echo.local_addr().map_err(|e| format!("the answering server's port: {e}"))?.port(); - // Ends with the process, as the holding server does. - thread::spawn(move || { - let mut datagram = [0u8; 64]; - while let Ok((len, from)) = echo.recv_from(&mut datagram) { - echo.send_to(&datagram[..len], from).expect("answer a datagram"); - } - }); + const RUST_JOBS: [&str; 2] = ["nodelay_accepted", "stream_ends_std"]; + let (holding, answering, ending) = (holding_server()?, answering_server()?, ending_server()?); + // The oracle: the report std's job owes is this host's TCP's. + let mut host = Vec::new(); + stream_ends::run("127.0.0.1", ending, |line| host.push(line)); + if host != stream_ends::EXPECTED { + return Err(format!("this host's TCP reports\n{}\nand not\n{}", host.join("\n"), stream_ends::EXPECTED.join("\n"))); + } let case = compile::repo_root().join("tests/netcase"); - let c_bins: Vec<(String, Vec)> = ["addr_order", "nodelay_kept", "sendto_unbound"] + let c_bins: Vec<(String, Vec)> = ["addr_order", "nodelay_kept", "sendto_unbound", "stream_ends"] .map(|name| (name.to_string(), compile::link_toyos(&compile::compile_own_c(&case, name), name))) .into(); - let rust_bin = - qemu::build_toyos_bin(qemu::SUITE_ARCH, &compile::repo_root().join("tests/toyos-rust-tests"), RUST_JOB); - let mut qemu = - boot_netcase(&c_bins, &[(RUST_JOB.to_string(), rust_bin)], BootOptions { qmp: true, ..Default::default() })?; + let rust_bins: Vec<(String, Vec)> = RUST_JOBS + .map(|job| (job.to_string(), qemu::build_toyos_bin(qemu::SUITE_ARCH, &compile::repo_root().join("tests/toyos-rust-tests"), job))) + .into(); + let mut qemu = boot_netcase(&c_bins, &rust_bins, BootOptions { qmp: true, ..Default::default() })?; let [held, clear, std_port, pipe_port] = LISTENERS; let [to_held, to_clear, to_std, to_pipe] = forwards_into(&qemu, LISTENERS)?; /// Each line a job says its listeners wait in, the constants of its /// source, and the forwarded ports the host dials when it does. type Waits<'a> = &'a [(&'a str, &'a [u16])]; // The address first: every job names its peer by one. - let jobs: [(&str, String, Waits); 4] = [ + let jobs: [(&str, String, Waits); 6] = [ ("test_c_addr_order", holding.to_string(), &[]), ( "test_c_nodelay_kept", @@ -3142,6 +3347,8 @@ fn libc_sockets() -> Result<(), String> { ("nodelay_accepted: the listener waits for a second peer", &[to_pipe]), ], ), + ("test_c_stream_ends", ending.to_string(), &[]), + ("test_rs_stream_ends_std", ending.to_string(), &[]), ]; // Held to the test's end: a peer gone before a job's `accept` is a // connection its listener no longer holds. @@ -3178,6 +3385,8 @@ fn run_machine_test(name: &str, test_config: &Path) -> Result<(), String> { match name { "iommu_virtio_platform" => common::iommu::iommu_virtio_platform(test_config), "netstack_socket_churn" => netstack_socket_churn(), + "netstack_streams" => netstack_streams(qemu::Profile::Headless), + "netstack_streams_e1000e" => netstack_streams(qemu::Profile::HeadlessE1000e), "libc_sockets" => libc_sockets(), "nested_nmi_is_loud" => faults::nested_nmi_is_loud(test_config), "machine_shutdown" => power::machine_shutdown(test_config), diff --git a/toyos-i219/src/lib.rs b/toyos-i219/src/lib.rs index a8fce532e2..6fa33319e3 100644 --- a/toyos-i219/src/lib.rs +++ b/toyos-i219/src/lib.rs @@ -69,7 +69,7 @@ //! Every number in a written-back descriptor is the device's, and this driver //! is on the far side of an IOMMU domain from the rest of the machine but on //! the *same* side as its own memory. A length longer than the buffer it was -//! given becomes the length of a slice netstack hands to smoltcp, so `parse_rx` +//! given becomes the length of a slice netstack hands to its stack, so `parse_rx` //! bounds it, and `RxRefusal` is every way a written-back descriptor is refused //! rather than believed. //! diff --git a/toyos-i219/src/tests.rs b/toyos-i219/src/tests.rs index 175b87d24d..7d2f159f8a 100644 --- a/toyos-i219/src/tests.rs +++ b/toyos-i219/src/tests.rs @@ -315,7 +315,7 @@ mod refusals { const DONE: u8 = rx_desc::status::DD | rx_desc::status::EOP; /// The one that matters most: this number becomes the length of a slice - /// handed to smoltcp, so a device claiming more than the buffer holds + /// handed to the stack, so a device claiming more than the buffer holds /// would be a read past it. #[test] fn more_bytes_than_the_buffer_holds_is_refused() { diff --git a/toyos-net-shard/tcp/Cargo.toml b/toyos-net-shard/tcp/Cargo.toml index 5eec4d1102..33dad5ac2b 100644 --- a/toyos-net-shard/tcp/Cargo.toml +++ b/toyos-net-shard/tcp/Cargo.toml @@ -1,5 +1,3 @@ -# Host-tested; nothing ships it until the stack replaces smoltcp (issues/toyos-has-its-own-network-stack.md). - [package] name = "toyos-net-tcp" version = "0.1.0" diff --git a/userland/libc/src/lib.rs b/userland/libc/src/lib.rs index 94e87be16c..ef7401847d 100644 --- a/userland/libc/src/lib.rs +++ b/userland/libc/src/lib.rs @@ -28,6 +28,7 @@ mod sigmask; mod socket; mod sockopt; mod stdio; +mod streamend; mod string; mod strtonum; mod text; diff --git a/userland/libc/src/socket.rs b/userland/libc/src/socket.rs index 1a5986a94d..0992bf2186 100644 --- a/userland/libc/src/socket.rs +++ b/userland/libc/src/socket.rs @@ -8,9 +8,10 @@ use toyos_abi::syscall; use toyos::net::{NetError, TcpOptions, TcpSocketId, UdpSocketId, OPT_BROADCAST, OPT_NODELAY}; use crate::errno::{ - EACCES, EADDRINUSE, EAFNOSUPPORT, EBADF, ECONNREFUSED, ECONNRESET, EFAULT, EINVAL, EIO, ENOMEM, ENOPROTOOPT, ENOSPC, - ENOTCONN, EOPNOTSUPP, ETIMEDOUT, + EACCES, EADDRINUSE, EAFNOSUPPORT, EAGAIN, EBADF, ECONNREFUSED, ECONNRESET, EFAULT, EINVAL, EIO, ENOMEM, ENOPROTOOPT, + ENOSPC, ENOTCONN, EOPNOTSUPP, EPIPE, ETIMEDOUT, }; +use crate::streamend::{self, Refusal}; use crate::inaddr::{self, SockaddrIn, AF_INET}; use crate::sockopt::{self, Kept}; @@ -69,6 +70,9 @@ struct SocketEntry { // after its `connect`. nodelay: bool, broadcast: bool, + /// `shutdown` was asked for this half of a stream. + read_shut: bool, + write_shut: bool, } const MAX_SOCKETS: usize = 128; @@ -178,6 +182,8 @@ pub unsafe extern "C" fn socket(domain: i32, sock_type: i32, _protocol: i32) -> notify_fd: 0, nodelay: false, broadcast: false, + read_shut: false, + write_shut: false, }; let fd = alloc_socket(entry); if fd < 0 { @@ -319,6 +325,8 @@ pub unsafe extern "C" fn accept( notify_fd: 0, nodelay: accepted.options.nodelay(), broadcast: false, + read_shut: false, + write_shut: false, }; let new_fd = alloc_socket(new_entry); if new_fd < 0 { @@ -346,10 +354,14 @@ pub unsafe extern "C" fn send(fd: i32, buf: *const u8, len: usize, _flags: i32) match entry.kind { SocketKind::Tcp => { + if entry.write_shut { + set_errno(EPIPE); + return -1; + } let data = core::slice::from_raw_parts(buf, len); match syscall::write(RawHandle(entry.tx_fd as u32), data) { Ok(n) => n as isize, - Err(_) => { set_errno(EIO); -1 } + Err(e) => { set_errno(stream_errno(streamend::write_refused(e))); -1 } } } SocketKind::Udp => { @@ -380,10 +392,17 @@ pub unsafe extern "C" fn recv(fd: i32, buf: *mut u8, len: usize, _flags: i32) -> match entry.kind { SocketKind::Tcp => { + if entry.read_shut { + return 0; + } let data = core::slice::from_raw_parts_mut(buf, len); - match syscall::read(RawHandle(entry.rx_fd as u32), data) { + let read = syscall::read(RawHandle(entry.rx_fd as u32), data).map_err(|_| Refusal::Other).and_then(|n| match n { + 0 => streamend::read_end(syscall::write_nonblock(RawHandle(entry.tx_fd as u32), &[])).map(|()| 0), + n => Ok(n), + }); + match read { Ok(n) => n as isize, - Err(_) => { set_errno(EIO); -1 } + Err(refusal) => { set_errno(stream_errno(refusal)); -1 } } } SocketKind::Udp => { @@ -504,7 +523,7 @@ pub unsafe extern "C" fn shutdown(fd: i32, how: i32) -> i32 { Some(s) => s, None => { set_errno(EBADF); return -1; } }; - let entry = match slot.as_ref() { + let entry = match slot.as_mut() { Some(e) => e, None => { set_errno(EBADF); return -1; } }; @@ -514,10 +533,24 @@ pub unsafe extern "C" fn shutdown(fd: i32, how: i32) -> i32 { set_errno(net_err_to_errno(e)); return -1; } + entry.read_shut |= matches!(how, SHUT_RD | SHUT_RDWR); + entry.write_shut |= matches!(how, SHUT_WR | SHUT_RDWR); } 0 } +const SHUT_RD: i32 = 0; +const SHUT_WR: i32 = 1; +const SHUT_RDWR: i32 = 2; + +fn stream_errno(refusal: Refusal) -> i32 { + match refusal { + Refusal::Reset => ECONNRESET, + Refusal::Again => EAGAIN, + Refusal::Other => EIO, + } +} + // close (for socket fds) diff --git a/userland/libc/src/streamend.rs b/userland/libc/src/streamend.rs new file mode 100644 index 0000000000..0831218f80 --- /dev/null +++ b/userland/libc/src/streamend.rs @@ -0,0 +1,38 @@ +//! How a stream socket's end reaches its caller. A stream is two of netstack's pipes, and the +//! receive pipe says only that it ended: netstack lets the send pipe go first when the connection +//! failed, and keeps it after an orderly end for as long as its writer holds it, so what a +//! zero-byte write into it answers after the end is the end's kind. It reads nothing but what it +//! is handed, so the host tests it (`toyos-libc-copies`). + +use toyos_abi::syscall::SyscallError; + +/// What a call on a stream answers instead of bytes. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub(crate) enum Refusal { + /// `ECONNRESET`: the connection failed. + Reset, + /// `EAGAIN`. + Again, + /// `EIO`: the handle is not the pipe end the socket was made with. + Other, +} + +/// A read of 0 from the receive pipe, given what a zero-byte write into the send pipe then +/// answered: `Ok` is the peer's FIN, after its last byte. +pub(crate) fn read_end(probe: Result) -> Result<(), Refusal> { + match probe { + Ok(_) | Err(SyscallError::WouldBlock) => Ok(()), + Err(SyscallError::Gone) => Err(Refusal::Reset), + Err(_) => Err(Refusal::Other), + } +} + +/// A write the send pipe refused. One after a shutdown of the sending half is refused before it +/// reaches the pipe, which netstack reads no more but keeps. +pub(crate) fn write_refused(refused: SyscallError) -> Refusal { + match refused { + SyscallError::Gone => Refusal::Reset, + SyscallError::WouldBlock => Refusal::Again, + _ => Refusal::Other, + } +} diff --git a/userland/netstack/Cargo.toml b/userland/netstack/Cargo.toml index ea4f0beb03..498ad8ba17 100644 --- a/userland/netstack/Cargo.toml +++ b/userland/netstack/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "netstack" -description = "The network server: one Ethernet device claimed from the kernel, smoltcp over it, DHCP, the DNS resolver and mDNS, served to programs over IPC." +description = "The network server: one Ethernet device claimed from the kernel, ToyOS's own stack over it (toyos-net-node), its DHCP lease, resolver and mDNS name, served to programs over IPC." edition = "2024" license = "MIT OR Apache-2.0" @@ -13,22 +13,14 @@ toyos-virtio = { path = "../../toyos-virtio" } toyos-tco = { path = "../../toyos-tco" } toyos-mdns = { path = "mdns" } toyos-dns = { path = "../../toyos-dns" } +toyos-dhcp = { path = "../../toyos-dhcp" } toyos-inspect = { path = "../../toyos-inspect" } - -[dependencies.smoltcp] -version = "0.12" -default-features = false -features = [ - "medium-ethernet", - "proto-ipv4", - "socket-tcp", - "socket-udp", - "socket-dhcpv4", - # The multicast DNS group, which `mdns.rs` joins to answer for this - # machine's name. - "multicast", - "alloc", -] +toyos-net-node = { path = "node" } +toyos-net-ip = { path = "../../toyos-net-ip" } +toyos-net-shard = { path = "../../toyos-net-shard" } +toyos-net-tcp = { path = "../../toyos-net-shard/tcp" } +toyos-net-udp = { path = "../../toyos-net-udp" } +toyos-net-wire = { path = "../../toyos-net-wire" } [package.metadata.toyos.host] exempt.owns = "the NIC ToyOS claims for it" diff --git a/userland/netstack/README.md b/userland/netstack/README.md index 025cf719a9..4316392bf0 100644 --- a/userland/netstack/README.md +++ b/userland/netstack/README.md @@ -1,3 +1,3 @@ # netstack -Network daemon that provides networking to applications via message-passing IPC, built on smoltcp. +Network daemon that provides networking to applications via message-passing IPC. diff --git a/userland/netstack/node/src/datagram.rs b/userland/netstack/node/src/datagram.rs index 99a6e04b21..09ae843897 100644 --- a/userland/netstack/node/src/datagram.rs +++ b/userland/netstack/node/src/datagram.rs @@ -138,7 +138,7 @@ impl Node { /// Queues `payload` for `destination:port`; it leaves in a later [`Node::transmit`]. pub fn udp_send_to(&mut self, now: Instant, id: DatagramId, destination: Ipv4Addr, port: u16, payload: &[u8]) -> Result<(), Refused> { let sent = self.stack.send_to(now, id.0, destination, port, payload).map_err(refused); - self.log(); + self.log(now); sent } diff --git a/userland/netstack/node/src/lib.rs b/userland/netstack/node/src/lib.rs index 077eae0a87..f0506792f6 100644 --- a/userland/netstack/node/src/lib.rs +++ b/userland/netstack/node/src/lib.rs @@ -17,7 +17,8 @@ //! **Untrusted input.** A received frame is never read here: every byte goes through //! `toyos-net-wire`'s parsers inside the shard, and a DHCP payload through the client's. What //! either refuses is counted where it was refused and, where it is a log line, comes out of -//! [`Node::drain_events`]. +//! [`Node::drain_events`], at most one a rule in any 10 s with a count of the rest: a peer's +//! frames never write the log faster than that. //! //! **Draws.** Each `draw` is handed to the client, whose order is its own: a call that starts an //! exchange draws its transaction id first. The name draws after it, the delay of each probing it @@ -73,7 +74,8 @@ toyos_net_wire::counters! { pub enum Event { /// A refusal of the stack's, and how many of its rule it stands for beyond itself. Stack { refusal: toyos_net_shard::Refusal, suppressed: u64 }, - Dhcp(toyos_dhcp::Refusal), + /// A refusal of the client's, and how many of its rule it stands for beyond itself. + Dhcp { refusal: toyos_dhcp::Refusal, suppressed: u64 }, /// What became of the machine's name. Name(toyos_mdns::Event), } @@ -88,6 +90,9 @@ pub struct Node { resolver: resolve::Resolver, counters: Counters, events: Vec, + /// The client's refusals pass through it as the stack's pass through the shard's: one line a + /// rule in any 10 s, carrying how many were not. + dhcp_log: toyos_dhcp::RefusalLog, /// Room for the largest message the client accepts. datagram: Vec, streams: streams::Streams, @@ -115,6 +120,7 @@ impl Node { resolver: resolve::Resolver::new(), counters: Counters::default(), events: Vec::new(), + dhcp_log: toyos_dhcp::RefusalLog::default(), datagram: vec![0; usize::from(toyos_dhcp::limits::MAX_MESSAGE)], streams: streams::Streams::default(), listeners: listeners::Listeners::default(), @@ -127,10 +133,6 @@ impl Node { self.stack.shard() } - pub fn dhcp(&self) -> &Client { - &self.client - } - pub fn counters(&self) -> &Counters { &self.counters } @@ -143,7 +145,7 @@ impl Node { /// Log lines since the last call. pub fn drain_events(&mut self) -> impl Iterator + '_ { - self.events.drain(..).chain(self.client.drain_refusals().map(Event::Dhcp)) + self.events.drain(..) } // ---- the device ---- @@ -163,7 +165,7 @@ impl Node { loop { sent = sent.saturating_add(self.stack.transmit(now, credit.saturating_sub(sent), &mut sink)); // A datagram [ip] refused as it left is a line of this opportunity. - self.log(); + self.log(now); let asked = self.resolver.pass(now, &mut self.stack, &mut self.counters, &mut draw); if !asked || sent >= credit { break; @@ -208,7 +210,7 @@ impl Node { /// carried on, against the lease as that left it. fn settle(&mut self, now: Instant, draw: &mut impl FnMut() -> u32) { loop { - self.log(); + self.log(now); let (out, verified) = if let Some(report) = self.stack.report() { match report { Report::Verified(verified) => (self.client.verified(now, &mut *draw), Some(verified)), @@ -226,11 +228,16 @@ impl Node { self.resolver.pass(now, &mut self.stack, &mut self.counters, draw); } - /// The stack's log lines so far, into [`Self::drain_events`]: the one way a line leaves the - /// stack. - fn log(&mut self) { + /// The stack's and the client's log lines so far, into [`Self::drain_events`]: the one way a + /// line leaves either. + fn log(&mut self, now: Instant) { let events = &mut self.events; self.stack.refusals(|refusal, suppressed| events.push(Event::Stack { refusal, suppressed })); + for refusal in self.client.drain_refusals() { + if let Some(suppressed) = self.dhcp_log.admit(now, refusal.rule) { + events.push(Event::Dhcp { refusal, suppressed }); + } + } } /// What one call of the client's asked for: the lease first, so a message leaves from the diff --git a/userland/netstack/node/src/listeners.rs b/userland/netstack/node/src/listeners.rs index 008db818e9..6ee247b8ec 100644 --- a/userland/netstack/node/src/listeners.rs +++ b/userland/netstack/node/src/listeners.rs @@ -149,10 +149,6 @@ impl Node { true } - pub fn listener_nodelay(&self, id: ListenerId) -> Option { - self.listeners.live.get(&id).map(|listener| listener.options.nodelay) - } - /// The owner's accept: the oldest connection waiting at `id` becomes a stream on `pipes`, /// and what it already received moves at once. pub fn accept(&mut self, now: Instant, id: ListenerId, pipes: Option) -> Result { diff --git a/userland/netstack/node/src/streams.rs b/userland/netstack/node/src/streams.rs index 6e5bc08f44..2266477e27 100644 --- a/userland/netstack/node/src/streams.rs +++ b/userland/netstack/node/src/streams.rs @@ -429,10 +429,6 @@ impl Node { true } - pub fn nodelay(&self, id: StreamId) -> Option { - self.streams.live.get(&id).map(|stream| stream.options.nodelay) - } - /// The kernel said nobody holds the other end of one of an established stream's pipes: the /// to-client pipe has no reader, or the from-client pipe no writer, whatever it still holds. pub fn pipe_gone(&mut self, now: Instant, id: StreamId, end: PipeEnd) { diff --git a/userland/netstack/node/tests/lease.rs b/userland/netstack/node/tests/lease.rs index 1d4fce39cb..a6ee8cd6d1 100644 --- a/userland/netstack/node/tests/lease.rs +++ b/userland/netstack/node/tests/lease.rs @@ -153,7 +153,7 @@ fn a_nak_takes_address_route_and_resolvers_in_one_step() { assert_eq!((wire.node.lease(), wire.address(A), wire.gateways()), (None, None, vec![])); assert!(matches!(wire.sent.last(), Some(Seen::Dhcp { source, .. }) if source.is_unspecified()), "{:?}", wire.sent.last()); assert_ne!(xid(wire.last(DISCOVER)), id, "a new exchange"); - let refused = Event::Dhcp(toyos_dhcp::Refusal { rule: toyos_dhcp::Counter::Nak, peer: toyos_dhcp::Peer::From(R) }); + let refused = Event::Dhcp { refusal: toyos_dhcp::Refusal { rule: toyos_dhcp::Counter::Nak, peer: toyos_dhcp::Peer::From(R) }, suppressed: 0 }; assert!(wire.node.drain_events().any(|event| event == refused)); assert!(!wire.answers_arp_for(A)); } @@ -370,7 +370,7 @@ fn a_conflict_under_the_probe_declines_and_holds_nothing() { assert_eq!((wire.node.lease(), wire.address(A)), (None, None)); let events: Vec = wire.node.drain_events().collect(); let stack = |event: &Event| matches!(event, Event::Stack { refusal: Refusal::Ip(refusal), .. } if refusal.rule == toyos_net_ip::Counter::AcdConflict); - let dhcp = |event: &Event| matches!(event, Event::Dhcp(refusal) if refusal.rule == toyos_dhcp::Counter::Declined); + let dhcp = |event: &Event| matches!(event, Event::Dhcp { refusal, .. } if refusal.rule == toyos_dhcp::Counter::Declined); assert!(events.iter().any(stack) && events.iter().any(dhcp), "{events:?}"); assert!(wire.run_until(Duration::from_secs(11), |wire| wire.dhcp().last().is_some_and(|(kind, _)| *kind == DISCOVER)), "discovery starts over"); @@ -436,20 +436,18 @@ fn next_deadline_is_the_earliest_of_the_clients_and_the_stacks() { // Selecting: only the client waits. let mut wire = Wire::new(); wire.link(true); - let (client, stack) = (wire.node.dhcp().next_deadline(), wire.node.shard().next_deadline()); - assert!(client.is_some() && stack.is_none(), "{client:?} {stack:?}"); - assert_eq!(wire.node.next_deadline(), client); + let stack = wire.node.shard().next_deadline(); + assert!(stack.is_none() && wire.node.next_deadline().is_some(), "{stack:?}"); // Probing: only the stack does. let wire = Wire::acknowledged(&terms(600, Some(R))); - let (client, stack) = (wire.node.dhcp().next_deadline(), wire.node.shard().next_deadline()); - assert!(client.is_none() && stack.is_some(), "{client:?} {stack:?}"); + let stack = wire.node.shard().next_deadline(); + assert!(stack.is_some(), "{stack:?}"); assert_eq!(wire.node.next_deadline(), stack); - // Held: both, the stack's second announcement before the client's renewal. + // Held: both, the stack's second announcement before the client's renewal at T1, 300 s. let mut wire = Wire::leased(&terms(600, Some(R))); - let (client, stack) = (wire.node.dhcp().next_deadline(), wire.node.shard().next_deadline()); - assert_eq!(client, Some(after(300))); + let stack = wire.node.shard().next_deadline(); assert!(stack.is_some_and(|stack| stack < after(300)), "{stack:?}"); assert_eq!(wire.node.next_deadline(), stack); @@ -471,11 +469,31 @@ fn a_reply_without_the_cookie_is_refused_by_name() { bootp[236..240].fill(0); wire.deliver(&from_server(MAC, A, &bootp)); wire.last(DISCOVER); - let refused = Event::Dhcp(toyos_dhcp::Refusal { rule: toyos_dhcp::Counter::BootpReply, peer: toyos_dhcp::Peer::From(R) }); + let refused = Event::Dhcp { refusal: toyos_dhcp::Refusal { rule: toyos_dhcp::Counter::BootpReply, peer: toyos_dhcp::Peer::From(R) }, suppressed: 0 }; assert!(wire.node.drain_events().any(|event| event == refused)); assert_eq!(wire.node.drain_events().count(), 0, "drained"); } +/// Any host on the link can send replies the client refuses, one a frame: the log takes one line +/// a rule in any 10 s, and the next line carries how many it did not. +#[test] +fn a_hundred_refused_replies_are_one_line_and_a_count() { + let mut wire = Wire::new(); + wire.link(true); + let id = xid(wire.last(DISCOVER)); + let mut bootp = message_of(OFFER, id, &terms(HOUR, Some(R))); + bootp[236..240].fill(0); + let refusal = toyos_dhcp::Refusal { rule: toyos_dhcp::Counter::BootpReply, peer: toyos_dhcp::Peer::From(R) }; + for _ in 0..100 { + wire.deliver(&from_server(MAC, A, &bootp)); + } + assert_eq!(wire.node.drain_events().collect::>(), [Event::Dhcp { refusal, suppressed: 0 }]); + + wire.now = wire.now.after(toyos_net_wire::REFUSAL_LOG_INTERVAL); + wire.deliver(&from_server(MAC, A, &bootp)); + assert_eq!(wire.node.drain_events().collect::>(), [Event::Dhcp { refusal, suppressed: 99 }]); +} + /// The node waiting for the ACK of its REQUEST, and that ACK's frame. fn requesting() -> (Wire, Vec) { let terms = terms(HOUR, Some(R)); diff --git a/userland/netstack/node/tests/listeners.rs b/userland/netstack/node/tests/listeners.rs index 23c238e7fd..d901574823 100644 --- a/userland/netstack/node/tests/listeners.rs +++ b/userland/netstack/node/tests/listeners.rs @@ -383,12 +383,10 @@ impl Net { answer.map(|accepted| (accepted, client)) } - /// An accept that takes `peer`'s connection, which the answer and the node say the same - /// option of. + /// An accept that takes `peer`'s connection. fn accepts(&mut self, id: ListenerId, peer: Peer) -> (StreamId, Client) { let (accepted, client) = self.accept(id).expect("a connection and a place"); assert_eq!((accepted.remote, accepted.local), (peer.endpoint(), Port::new(SSH).unwrap())); - assert_eq!(Some(accepted.nodelay), self.node.nodelay(accepted.id), "the accept's answer and the stream it made"); (accepted.id, client) } @@ -463,10 +461,7 @@ fn a_listener_answers_a_syn_and_wakes_its_owner_when_the_handshake_ends() { assert_eq!((wakes(&owner), client.borrow().dropped), (1, 0)); } -// The recorded failure of `issues/a-handshake-nobody-finishes-holds-a-listeners-port-shut.md`: -// one SYN and nothing more, and the stack this one replaces answered the next peer's SYN with a -// reset for as long as the first handshake hung, which was for good. Here the next peer is -// answered at once, and the first handshake is given up within [tcp]'s bound, after which its +// One SYN and nothing more: the next peer is answered at once, and the first handshake is given up within [tcp]'s bound, after which its // late ACK meets LISTEN: RFC 9293 §3.10.7.2, second check, . #[test] fn a_handshake_nobody_finishes_leaves_the_port_open_and_is_given_up() { @@ -852,15 +847,14 @@ fn a_datagram_socket_holds_a_place_and_a_bind_without_one_makes_nothing() { fn a_stream_starts_with_the_options_its_connection_took_from_its_listener() { let mut net = Net::new(); let (id, _owner) = net.listen(SSH); - assert_eq!(net.node.listener_nodelay(id), Some(false)); // One connection begins before the option is set and one after; both are accepted after. net.handshake(P1); assert!(net.node.set_listener_nodelay(id, true)); net.handshake(P2); - assert_eq!(net.node.listener_nodelay(id), Some(true)); - let (first, a) = net.accepts(id, P1); - let (second, b) = net.accepts(id, P2); - assert_eq!((net.node.nodelay(first), net.node.nodelay(second)), (Some(false), Some(true))); + let (first, a) = net.accept(id).unwrap(); + let (second, b) = net.accept(id).unwrap(); + assert_eq!((first.remote, first.nodelay, second.remote, second.nodelay), (P1.endpoint(), false, P2.endpoint(), true)); + let (first, second) = (first.id, second.id); for client in [&a, &b] { net.says(client, b"a"); net.says(client, b"b"); @@ -875,18 +869,17 @@ fn a_stream_starts_with_the_options_its_connection_took_from_its_listener() { assert!(net.node.set_nodelay(net.now, first, true)); net.pump(); assert_eq!(net.texts(P1), [b"a".to_vec(), b"b".to_vec()], "and the first stream's second waits no longer"); - assert_eq!((net.node.nodelay(first), net.node.nodelay(second), net.node.listener_nodelay(id)), (Some(true), Some(false), Some(true))); assert!(net.node.set_listener_nodelay(id, false)); net.handshake(P3); - let (third, c) = net.accepts(id, P3); - assert_eq!((net.node.nodelay(third), net.node.nodelay(first)), (Some(false), Some(true))); + let (third, c) = net.accept(id).unwrap(); + assert_eq!((third.remote, third.nodelay), (P3.endpoint(), false)); net.says(&c, b"a"); net.says(&c, b"b"); assert_eq!(net.texts(P3), [b"a".to_vec()]); assert!(net.node.close_listener(net.now, id)); - assert_eq!((net.node.set_listener_nodelay(id, true), net.node.listener_nodelay(id)), (false, None)); + assert!(!net.node.set_listener_nodelay(id, true), "a closed listener's id names nothing"); } // A listen names its listener's option, so no SYN reaches the port between the passive open and @@ -896,10 +889,9 @@ fn a_stream_starts_with_the_options_its_connection_took_from_its_listener() { fn a_listener_holds_the_option_its_listen_named_from_its_first_connection() { let mut net = Net::new(); let (id, _owner) = net.listen_with(SSH, true); - assert_eq!(net.node.listener_nodelay(id), Some(true)); net.handshake(P1); let (accepted, client) = net.accept(id).unwrap(); - assert_eq!((accepted.nodelay, net.node.nodelay(accepted.id)), (true, Some(true))); + assert!(accepted.nodelay); net.says(&client, b"a"); net.says(&client, b"b"); assert_eq!(net.texts(P1), [b"a".to_vec(), b"b".to_vec()], "the second write waits for no acknowledgment"); @@ -907,14 +899,13 @@ fn a_listener_holds_the_option_its_listen_named_from_its_first_connection() { assert!(net.node.set_listener_nodelay(id, false)); net.handshake(P2); let (accepted, _client) = net.accept(id).unwrap(); - assert_eq!((accepted.nodelay, net.node.listener_nodelay(id)), (false, Some(false)), "a listen's option is the listener's to change"); + assert!(!accepted.nodelay, "a listen's option is the listener's to change"); } // A handshake its peer resets before it ends (RFC 9293 §3.10.7.4, first check, in SYN-RECEIVED) // leaves nothing of its options behind: the listener's option changed while it was in progress, // and the next connection begins with what the listener holds when its own SYN arrives. Both -// ways round, since the stack this one replaces keeps the reset handshake's option for the next -// connection (`issues/a-handshake-reset-before-it-ends-hands-its-option-to-the-next-connection.md`). +// ways round. #[test] fn a_handshake_reset_before_it_ends_leaves_the_next_connection_its_listeners_option() { for held in [true, false] { @@ -927,7 +918,6 @@ fn a_handshake_reset_before_it_ends_leaves_the_next_connection_its_listeners_opt net.handshake(P2); let (accepted, _client) = net.accept(id).unwrap(); assert_eq!((accepted.remote, accepted.nodelay), (P2.endpoint(), !held), "the listener held {held} at the reset handshake's SYN"); - assert_eq!((net.node.nodelay(accepted.id), net.node.listener_nodelay(id)), (Some(!held), Some(!held))); // And a set while the reset handshake was in progress, taken back before the next SYN. net.syn(P3); @@ -952,8 +942,7 @@ impl ToClient for BrokenEnd { // The accept's own pass moves what the connection already received, and a pipe that refuses it // for good ends the stream there (RFC 9293 §3.10.7.4's reset is the peer's notice). The answer -// still says what the stream was handed over with, where `Node::nodelay` of its id has no stream -// left to answer for. +// still says what the stream was handed over with, though no stream is left to answer for it. #[test] fn an_accept_answers_the_option_of_a_stream_its_own_pass_let_go() { let mut net = Net::new(); @@ -965,7 +954,7 @@ fn an_accept_answers_the_option_of_a_stream_its_own_pass_let_go() { net.pump(); assert!(net.last(P1).rst, "{:?}", net.heard); assert_eq!((accepted.remote, accepted.nodelay), (P1.endpoint(), true)); - assert_eq!((net.node.nodelay(accepted.id), net.node.streams(), client.borrow().dropped), (None, 0, 2)); + assert_eq!((net.node.streams(), client.borrow().dropped), (0, 2)); } // `streams` counts a stream its client can see no more by its peer's address, and lets an @@ -1024,12 +1013,11 @@ fn a_request_for_a_stream_that_was_cut_names_nothing() { net.node.close(net.now, cut); assert!(!net.node.shutdown_write(net.now, cut)); assert!(!net.node.set_nodelay(net.now, cut, true)); - assert_eq!(net.node.nodelay(cut), None); net.node.pipe_gone(net.now, cut, PipeEnd::FromClient); net.node.pipe_broken(net.now, cut, PipeEnd::FromClient); net.pump(); assert_eq!((net.node.streams(), net.node.held(), other.borrow().dropped, wakes(&owner)), (1, 2, 0, 2), "the stream in its place stands"); - assert_eq!((net.node.nodelay(next), net.heard.len(), net.events()), (Some(false), said, vec![])); + assert_eq!((net.heard.len(), net.events()), (said, vec![])); } // The track's exit for the bound across peer addresses. `streams` lets each address keep diff --git a/userland/netstack/node/tests/streams.rs b/userland/netstack/node/tests/streams.rs index 93ea1799a0..699296b573 100644 --- a/userland/netstack/node/tests/streams.rs +++ b/userland/netstack/node/tests/streams.rs @@ -538,7 +538,7 @@ fn a_connect_is_a_handshake_and_its_answer_names_the_port() { let [syn, ack] = &net.far.segments[..] else { panic!("a SYN and an ACK, not {:?}", net.far.segments) }; assert!(syn.syn && syn.ack.is_none() && syn.text.is_empty() && !syn.fin && !syn.rst, "{syn:?}"); assert_eq!((ack.syn, ack.seq, ack.ack, ack.text.len()), (false, syn.seq.wrapping_add(1), Some(ISS + 1), 0)); - assert_eq!((net.node.streams(), net.watch(id), net.node.nodelay(id)), (1, Some(IDLE), Some(false))); + assert_eq!((net.node.streams(), net.watch(id)), (1, Some(IDLE))); assert_eq!(dropped(&client), (false, false)); } @@ -980,7 +980,7 @@ fn a_close_sends_what_the_pipe_held_and_then_the_fin() { assert!(net.far.resets.is_empty()); // The id names nothing. net.node.close(net.now, id); - assert_eq!((net.node.nodelay(id), net.node.set_nodelay(net.now, id, true), net.node.shutdown_write(net.now, id)), (None, false, false)); + assert_eq!((net.node.set_nodelay(net.now, id, true), net.node.shutdown_write(net.now, id)), (false, false)); } // RFC 9293 §3.6.1 (SHLD-3): a close with text unread is a reset, which shows the peer it was @@ -1301,7 +1301,7 @@ fn nodelay_reaches_the_stack() { assert_eq!(net.far.texts(), [100], "the second write waits for the first's acknowledgment"); assert!(net.node.set_nodelay(net.now, id, true)); net.pump(); - assert_eq!((net.far.texts(), net.node.nodelay(id)), (vec![100, 50], Some(true))); + assert_eq!(net.far.texts(), [100, 50]); } // ---- the wire is not trusted ---- diff --git a/userland/netstack/src/card.rs b/userland/netstack/src/card.rs index 3869f5d0b7..38b7e31dce 100644 --- a/userland/netstack/src/card.rs +++ b/userland/netstack/src/card.rs @@ -53,7 +53,7 @@ impl Card { /// **The record has to be taken, not merely noticed.** A claim reads ready /// while it holds an undrained interrupt, so a pass that saw the token and /// left it would find the same one on the next `wait` and every one after - /// it. What the message meant is in the rings, which `iface.poll` reads. + /// it. /// This is also where a driver with a per-pass budget gets it back. /// /// A claim that refuses the read for anything but `WouldBlock` is the @@ -131,6 +131,24 @@ impl Card { snap.put("errors.crc", wire.crc_errors); } + /// Hands the next received frame to `take` and gives its buffer back to + /// the card once `take` returns; `false` when none waits. + pub fn rx(&self, take: impl FnOnce(&[u8])) -> bool { + match self { + Self::Virtio(nic) => { + let Some((index, len)) = nic.poll_rx() else { return false }; + take(nic.rx_frame(index, len)); + nic.rx_done(index); + } + Self::Intel(nic) => { + let Some(frame) = nic.poll_rx() else { return false }; + take(nic.rx_frame(&frame)); + nic.rx_done(frame); + } + } + true + } + /// How many frames the card takes now. [`Self::tx`] is for a caller this /// answered: a card with no room is not offered a frame. pub fn tx_room(&self) -> usize { diff --git a/userland/netstack/src/dhcp.rs b/userland/netstack/src/dhcp.rs deleted file mode 100644 index a5025042e2..0000000000 --- a/userland/netstack/src/dhcp.rs +++ /dev/null @@ -1,217 +0,0 @@ -//! This machine's address, taken from the network rather than written down. -//! -//! What the lease decides is the whole of the interface: the address and its -//! prefix, the default route, and the servers `crate::resolve` asks. All -//! three are written together on every lease and cleared together when one is -//! lost, because a route left standing over an address that is gone sends -//! frames out with a source nothing will answer. -//! -//! **A machine that gets no lease says so and goes on serving.** Its clients -//! then get their connects refused, one refusal at a time, which is what they -//! are already written to survive. - -use std::time::{Duration, Instant}; - -use smoltcp::iface::Interface; -use smoltcp::socket::dhcpv4; -use smoltcp::wire::{DhcpOption, IpCidr, Ipv4Address, Ipv4Cidr}; - -use crate::resolve::Resolver; - -/// RFC 2132 §3.14. -const OPT_HOST_NAME: u8 = 12; - -/// The name this machine asks its network to record for it, and answers to on -/// it as `.local` (`crate::mdns`). One name, because there is one machine. -pub const HOSTNAME: &str = "toyos-t14"; - -/// The options every DISCOVER and REQUEST carries. -static OUTGOING: [DhcpOption<'static>; 1] = - [DhcpOption { kind: OPT_HOST_NAME, data: HOSTNAME.as_bytes() }]; - -/// How long this machine waits for its first lease before saying it has none. -/// -/// It bounds the *report*, never the client: the socket retries for the life of -/// the boot and a lease that lands later is applied like any other. What it buys -/// is a line in the log on a machine whose network never answers. -const LEASE_BOUND: Duration = Duration::from_millis(toyos_tco::LEASE_BOUND_MS); - -/// The DHCP client socket this machine runs, asking for a lease under -/// [`HOSTNAME`]. -/// -/// **Its discovery restarts when the link comes up with no lease held** -/// (`restart`): the client's own retry is ten seconds apart, so a DISCOVER sent -/// into a link still negotiating would otherwise cost the lease that long. -pub fn socket() -> dhcpv4::Socket<'static> { - let mut socket = dhcpv4::Socket::new(); - socket.set_outgoing_options(&OUTGOING); - socket -} - -/// Ask again from the start, now: the link has just come up and no lease is -/// held. -/// -/// **Never on a bound lease.** The client's restart gives the address up -/// before it asks again, so a link that went down and came back would take -/// this machine off its network for a whole exchange. The client offers no -/// way to renew early, so a bound lease is kept across a flap and renews on -/// its own timer. -pub fn restart(client: &mut dhcpv4::Socket) { - client.reset(); -} - -/// What the client decided, owned, so that applying it borrows nothing of the -/// socket set the client lives in. -pub enum Change { - Leased { address: Ipv4Cidr, router: Option, server: Ipv4Address, dns: Vec }, - Lost, -} - -impl Change { - /// Whatever the client has to say this pass. - pub fn of(client: &mut dhcpv4::Socket) -> Option { - match client.poll()? { - dhcpv4::Event::Configured(config) => Some(Self::Leased { - address: config.address, - router: config.router, - server: config.server.address, - dns: config.dns_servers.to_vec(), - }), - dhcpv4::Event::Deconfigured => Some(Self::Lost), - } - } -} - -/// The lease the interface holds: everything the server decided. -struct Held { - address: Ipv4Cidr, - router: Option, - server: Ipv4Address, - dns: Vec, -} - -/// The lease's own state, and what the boot's log still owes about it. -pub struct Dhcp { - began: Instant, - /// The lease the interface currently holds, if it holds one. - lease: Option, - /// Whether this boot has settled the question once — a lease landed, or the - /// bound passed with none. netstack announces itself on the edge of this. - settled: bool, -} - -impl Dhcp { - pub fn new() -> Self { - Self { began: Instant::now(), lease: None, settled: false } - } - - /// Whether the interface holds a lease now. - pub fn leased(&self) -> bool { - self.lease.is_some() - } - - /// The lease as `inspect` reads it: what the interface holds now. - pub fn inspect(&self, snap: &mut toyos_inspect::Snapshot) { - let Some(held) = &self.lease else { - snap.put("lease.held", false); - return; - }; - snap.put("lease.held", true); - snap.put("lease.address", held.address.to_string()); - snap.put("lease.server", held.server.to_string()); - if let Some(router) = held.router { - snap.put("lease.router", router.to_string()); - } - let dns: Vec = held.dns.iter().map(ToString::to_string).collect(); - snap.put("lease.dns", dns.join(" ")); - } - - /// Apply what the client decided, and answer whether this machine's address - /// question has just been settled — which is the moment netstack has something - /// to serve with. - pub fn pass u16>(&mut self, change: Option, iface: &mut Interface, resolver: &mut Resolver) -> bool { - if let Some(change) = change { - let held = match change { - Change::Leased { address, router, server, dns } => { - // **One record carrying every field the lease decided.** A - // boot read off a stick or a stream has this line and - // nothing else to say what this machine's network was. - crate::say!( - "netstack: DHCP: lease {}/{} from {server}, gateway {}, dns [{}], {} ms after \ - netstack came up", - address.address(), - address.prefix_len(), - match router { - Some(router) => router.to_string(), - None => "none".to_string(), - }, - dns.iter().map(ToString::to_string).collect::>().join(" "), - self.began.elapsed().as_millis(), - ); - Some(Held { address, router, server, dns }) - } - // Only worth a line where there was something to lose: the - // client reports this on its way to a first lease too. - Change::Lost => { - if self.lease.is_some() { - crate::say!("netstack: DHCP: the lease is gone; this machine has no address"); - } - None - } - }; - Self::write( - held.as_ref().map(|h| (h.address, h.router)), - held.as_ref().map_or(&[][..], |h| &h.dns), - iface, - resolver, - ); - self.lease = held; - } - if self.settled { - return false; - } - if self.lease.is_some() { - self.settled = true; - return true; - } - if self.began.elapsed() >= LEASE_BOUND { - crate::say!( - "netstack: DHCP: no lease as {} in {} s; this machine has no address and every \ - connect through it is refused", - HOSTNAME, - self.began.elapsed().as_secs(), - ); - self.settled = true; - return true; - } - false - } - - /// The address, the default route and the resolvers, written together; - /// `None` writes the absence of all three. **One writer, reached by every - /// change**, so a route left standing over an address that is gone cannot - /// be arranged without breaking the path every boot takes to its lease. - fn write u16>( - lease: Option<(Ipv4Cidr, Option)>, - dns: &[Ipv4Address], - iface: &mut Interface, - resolver: &mut Resolver, - ) { - iface.update_ip_addrs(|addrs| { - // Cleared before the push, so a list already holding an address - // cannot leave the old one standing beside the new. - addrs.clear(); - if let Some((address, _)) = lease { - addrs.push(IpCidr::Ipv4(address)).expect("an emptied address list takes one"); - } - }); - iface.routes_mut().remove_default_ipv4_route(); - if let Some(router) = lease.and_then(|(_, router)| router) { - iface - .routes_mut() - .add_default_ipv4_route(router) - .expect("an emptied route table takes one default route"); - } - resolver.set_servers(dns); - } -} diff --git a/userland/netstack/src/listen.rs b/userland/netstack/src/listen.rs deleted file mode 100644 index edf5460520..0000000000 --- a/userland/netstack/src/listen.rs +++ /dev/null @@ -1,116 +0,0 @@ -//! Whether a listener's owner is owed a wake, what its accept finds, and the -//! option its connections begin with. -//! -//! **A listener is one smoltcp socket that becomes the connection it -//! accepts**, so the port listens only while that socket is in `Listen`: one -//! that left it is handed to its owner or listens again, or the port answers -//! every other peer with a reset for the rest of the boot. -//! -//! **An accept spends the owner's wake whatever it answers, a refusal -//! included, and a wake is owed only for a connection there is room to -//! take.** An owner refused holds no wake, so the connection it left is -//! announced again, and an owner refused for room is not woken until room -//! returns. -//! -//! **A connection begins with the option its listener held when its SYN -//! arrived**, which is what a host gives it, except the connection after a -//! handshake its peer reset: the socket holds the listener's option from the -//! moment it listens, its bind's from the first, a set reaches it only while -//! it still does, and what it holds once it is a connection is that -//! connection's. smoltcp puts a socket reset in `SynReceived` straight back to -//! `Listen` with what it held, which no pass here sees, so a set that arrived -//! during that handshake reaches no socket until the next accept or the next -//! set: `issues/a-handshake-reset-before-it-ends-hands-its-option-to-the-next-connection.md`. - -use smoltcp::socket::tcp; - -/// A listener's port, whether its owner holds a wake it has not spent on an -/// accept, and whether Nagle's algorithm is off for the connections that begin -/// from here on. -pub struct Listening { - port: u16, - woken: bool, - nodelay: bool, -} - -/// What an accept finds, handed the pipes `P` its request carried. -#[derive(Debug, PartialEq, Eq)] -pub enum Accept

{ - /// A connection, to hand over on the pipes. - Take(P), - /// A request that carried no pipes, whatever waits. - NoPipes, - /// A connection, and no room to take it. - NoRoom, - /// No connection. - Nothing, -} - -impl Listening { - /// A listener on `port` that holds `nodelay` from its first socket on. - pub fn new(port: u16, nodelay: bool) -> Self { - Self { port, woken: false, nodelay } - } - - pub fn port(&self) -> u16 { - self.port - } - - /// Has a closed `socket` listen on the port, with the listener's option. - pub fn open(&self, socket: &mut tcp::Socket) { - let port = self.port; - socket.listen(port).unwrap_or_else(|e| panic!("netstack: a socket refused to listen on {port}: {e:?}")); - socket.set_nagle_enabled(!self.nodelay); - } - - /// The listener's option from here on. `socket` takes it only while no - /// SYN has reached it. - pub fn set_nodelay(&mut self, socket: &mut tcp::Socket, nodelay: bool) { - self.nodelay = nodelay; - if socket.state() == tcp::State::Listen { - socket.set_nagle_enabled(!nodelay); - } - } - - /// The bytes to write the owner for `socket`: one wake if a connection - /// waits, there is `room` to take it, and the owner holds no wake, and - /// none otherwise. The wake is held from here on, so the caller ends the - /// listener if the owner is not handed it. - pub fn wake(&mut self, socket: &mut tcp::Socket, room: bool) -> &'static [u8] { - let owed = self.settle(socket) && room && !self.woken; - self.woken |= owed; - if owed { &[1] } else { &[] } - } - - /// An accept, with `room` for another connection or not, and the pipes - /// its request carried. - pub fn accept

(&mut self, socket: &mut tcp::Socket, room: bool, pipes: Option

) -> Accept

{ - self.woken = false; - match (self.settle(socket), room, pipes) { - (_, _, None) => Accept::NoPipes, - (false, _, Some(_)) => Accept::Nothing, - (true, false, Some(_)) => Accept::NoRoom, - (true, true, Some(pipes)) => Accept::Take(pipes), - } - } - - /// Puts `socket` back to listening if its peer reset it before its owner - /// took it, and says whether it holds a connection: a handshake finished, - /// whatever the peer did since. Its FIN included, which can land in the - /// same pass as the handshake's last ACK, so no pass ever sees the socket - /// `Established`. - fn settle(&self, socket: &mut tcp::Socket) -> bool { - match socket.state() { - tcp::State::Listen | tcp::State::SynReceived => false, - tcp::State::Established | tcp::State::CloseWait => true, - tcp::State::Closed => { - self.open(socket); - false - } - other => panic!("netstack: a listener's socket is {other:?}, which only netstack closing or connecting it reaches"), - } - } -} - -#[cfg(test)] -mod tests; diff --git a/userland/netstack/src/listen/tests.rs b/userland/netstack/src/listen/tests.rs deleted file mode 100644 index ac210aaf03..0000000000 --- a/userland/netstack/src/listen/tests.rs +++ /dev/null @@ -1,523 +0,0 @@ -//! A listener's socket on a wire: smoltcp's own `Interface` on an Ethernet -//! device whose far end is played here segment by segment, so what is judged -//! is what the port answers the next peer. - -use super::*; -use crate::{Ownerless, OWNERLESS_LIFE, RESET_LIFE}; -use std::collections::VecDeque; -use std::time::Duration; - -use smoltcp::iface::{Config, Interface, PollResult, SocketHandle, SocketSet}; -use smoltcp::phy::{self, ChecksumCapabilities, Device, DeviceCapabilities, Medium}; -use smoltcp::time::Instant; -use smoltcp::wire::{ - ArpOperation, ArpPacket, ArpRepr, EthernetAddress, EthernetFrame, EthernetProtocol, EthernetRepr, - HardwareAddress, IpAddress, IpCidr, IpProtocol, Ipv4Address, Ipv4Packet, Ipv4Repr, TcpControl, TcpPacket, - TcpRepr, TcpSeqNumber, -}; - -const OUR_MAC: EthernetAddress = EthernetAddress([0x02, 0, 0, 0, 0, 0x01]); -const OURS: Ipv4Address = Ipv4Address::new(10, 0, 2, 15); -const PEER_MAC: EthernetAddress = EthernetAddress([0x02, 0, 0, 0, 0, 0x02]); -const PEER: Ipv4Address = Ipv4Address::new(10, 0, 2, 2); -const PORT: u16 = 22; -/// The peer's initial sequence number, on every connection it opens. -const PEER_ISN: u32 = 1000; - -#[derive(Default)] -struct Wire { - inbound: VecDeque>, - outbound: Vec>, -} - -struct Rx(Vec); -struct Tx<'a>(&'a mut Vec>); - -impl phy::RxToken for Rx { - fn consume R>(self, f: F) -> R { - f(&self.0) - } -} - -impl phy::TxToken for Tx<'_> { - fn consume R>(self, len: usize, f: F) -> R { - let mut frame = vec![0u8; len]; - let result = f(&mut frame); - self.0.push(frame); - result - } -} - -impl Device for Wire { - type RxToken<'a> = Rx; - type TxToken<'a> = Tx<'a>; - - fn receive(&mut self, _: Instant) -> Option<(Rx, Tx<'_>)> { - let frame = self.inbound.pop_front()?; - Some((Rx(frame), Tx(&mut self.outbound))) - } - - fn transmit(&mut self, _: Instant) -> Option> { - Some(Tx(&mut self.outbound)) - } - - fn capabilities(&self) -> DeviceCapabilities { - let mut caps = DeviceCapabilities::default(); - caps.max_transmission_unit = 1514; - caps.medium = Medium::Ethernet; - caps - } -} - -/// One TCP segment our side sent: its flags, sequence number and the peer -/// port it went to. -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -struct Sent { - control: TcpControl, - ack: bool, - seq: u32, - to: u16, -} - -struct Net { - iface: Interface, - wire: Wire, - sockets: SocketSet<'static>, - listener: SocketHandle, - listening: Listening, - sent: Vec, - /// The clock every poll reads. - now: Instant, - /// Whether the far end answers ARP. - arp: bool, -} - -impl Net { - fn new() -> Self { - let mut wire = Wire::default(); - let mut iface = Interface::new(Config::new(HardwareAddress::Ethernet(OUR_MAC)), &mut wire, Instant::from_millis(0)); - iface.update_ip_addrs(|addrs| addrs.push(IpCidr::new(IpAddress::Ipv4(OURS), 24)).unwrap()); - let mut socket = tcp::Socket::new(tcp::SocketBuffer::new(vec![0; 4096]), tcp::SocketBuffer::new(vec![0; 4096])); - socket.listen(PORT).expect("a fresh socket listens"); - let mut sockets = SocketSet::new(Vec::new()); - let listener = sockets.add(socket); - Self { - iface, - wire, - sockets, - listener, - listening: Listening::new(PORT, false), - sent: Vec::new(), - now: Instant::from_millis(0), - arp: true, - } - } - - fn socket(&mut self) -> &mut tcp::Socket<'static> { - self.sockets.get_mut::(self.listener) - } - - /// One pass of netstack's loop: everything the wire holds, in one batch, and - /// the far end's ARP answered while it answers any. - fn pass(&mut self) { - loop { - while self.iface.poll(self.now, &mut self.wire, &mut self.sockets) != PollResult::None {} - if self.wire.outbound.is_empty() { - return; - } - for frame in std::mem::take(&mut self.wire.outbound) { - self.far_end(&frame); - } - } - } - - fn far_end(&mut self, frame: &[u8]) { - let eth = EthernetFrame::new_checked(frame).expect("the interface sent an Ethernet frame"); - match eth.ethertype() { - EthernetProtocol::Arp => { - let arp = ArpRepr::parse(&ArpPacket::new_checked(eth.payload()).unwrap()).unwrap(); - let ArpRepr::EthernetIpv4 { operation: ArpOperation::Request, .. } = arp else { return }; - if !self.arp { - return; - } - let reply = ArpRepr::EthernetIpv4 { - operation: ArpOperation::Reply, - source_hardware_addr: PEER_MAC, - source_protocol_addr: PEER, - target_hardware_addr: OUR_MAC, - target_protocol_addr: OURS, - }; - let mut out = vec![0u8; 14 + reply.buffer_len()]; - let mut e = EthernetFrame::new_unchecked(&mut out); - EthernetRepr { src_addr: PEER_MAC, dst_addr: OUR_MAC, ethertype: EthernetProtocol::Arp }.emit(&mut e); - reply.emit(&mut ArpPacket::new_unchecked(e.payload_mut())); - self.wire.inbound.push_back(out); - } - EthernetProtocol::Ipv4 => { - let ip = Ipv4Packet::new_checked(eth.payload()).unwrap(); - assert_eq!(ip.next_header(), IpProtocol::Tcp, "a listener sends only TCP"); - let tcp = TcpPacket::new_checked(ip.payload()).unwrap(); - let control = match (tcp.syn(), tcp.fin(), tcp.rst()) { - (true, _, _) => TcpControl::Syn, - (_, true, _) => TcpControl::Fin, - (_, _, true) => TcpControl::Rst, - _ => TcpControl::None, - }; - self.sent.push(Sent { control, ack: tcp.ack(), seq: tcp.seq_number().0 as u32, to: tcp.dst_port() }); - } - other => panic!("the interface sent a frame of type {other}"), - } - } - - /// A segment from the peer's `from` port onto the wire, delivered at the - /// next [`Net::pass`]. - fn send(&mut self, from: u16, control: TcpControl, seq: u32, ack: Option) { - let caps = ChecksumCapabilities::default(); - let tcp = TcpRepr { - src_port: from, - dst_port: PORT, - control, - seq_number: TcpSeqNumber(seq as i32), - ack_number: ack.map(|a| TcpSeqNumber(a as i32)), - window_len: 64000, - window_scale: None, - max_seg_size: None, - sack_permitted: false, - sack_ranges: [None, None, None], - timestamp: None, - payload: &[], - }; - let ip = Ipv4Repr { - src_addr: PEER, - dst_addr: OURS, - next_header: IpProtocol::Tcp, - payload_len: tcp.buffer_len(), - hop_limit: 64, - }; - let mut out = vec![0u8; 14 + ip.buffer_len() + ip.payload_len]; - let mut e = EthernetFrame::new_unchecked(&mut out); - EthernetRepr { src_addr: PEER_MAC, dst_addr: OUR_MAC, ethertype: EthernetProtocol::Ipv4 }.emit(&mut e); - let mut packet = Ipv4Packet::new_unchecked(e.payload_mut()); - ip.emit(&mut packet, &caps); - tcp.emit(&mut TcpPacket::new_unchecked(packet.payload_mut()), &IpAddress::Ipv4(PEER), &IpAddress::Ipv4(OURS), &caps); - self.wire.inbound.push_back(out); - } - - /// The peer's SYN from `from`, and the SYN-ACK's sequence number. - fn syn(&mut self, from: u16) -> u32 { - self.send(from, TcpControl::Syn, PEER_ISN, None); - self.pass(); - self.sent - .iter() - .rev() - .find(|s| s.control == TcpControl::Syn && s.ack && s.to == from) - .unwrap_or_else(|| panic!("a SYN from {from} was answered {:?}", self.sent.last())) - .seq - } - - /// The peer's answer to the SYN-ACK and `then`, in one batch. - fn ack_and(&mut self, from: u16, our_isn: u32, then: TcpControl) { - self.send(from, TcpControl::None, PEER_ISN + 1, Some(our_isn + 1)); - self.send(from, then, PEER_ISN + 1, Some(our_isn + 1)); - self.pass(); - } - - /// Whether the port answers a new peer's SYN with a SYN-ACK: it listens. - fn listens(&mut self, from: u16) -> bool { - self.send(from, TcpControl::Syn, PEER_ISN, None); - self.pass(); - let answer = *self.sent.iter().rev().find(|s| s.to == from).expect("a SYN is answered"); - answer.control == TcpControl::Syn - } - - /// Whether the owner is woken on a pass with `room` or without. - fn wakes(&mut self, room: bool) -> bool { - match self.listening.wake(self.sockets.get_mut::(self.listener), room) { - [] => false, - [1] => true, - other => panic!("a wake of {other:?}"), - } - } - - fn accept(&mut self, room: bool, pipes: bool) -> Accept<()> { - self.listening.accept(self.sockets.get_mut::(self.listener), room, pipes.then_some(())) - } -} - -#[test] -fn a_finished_handshake_is_owed_one_wake() { - let mut net = Net::new(); - let isn = net.syn(5001); - assert!(!net.wakes(true), "a SYN alone was taken for a connection"); - net.send(5001, TcpControl::None, PEER_ISN + 1, Some(isn + 1)); - net.pass(); - assert!(net.wakes(true)); - assert!(!net.wakes(true), "a connection its owner holds a wake for was announced twice"); - assert_eq!(net.accept(true, true), Accept::Take(())); -} - -/// **A peer that sends its FIN with the handshake's last ACK is still a -/// connection.** Both land in one pass, so the socket goes from `SynReceived` -/// to `CloseWait` with no pass ever seeing it `Established`. -#[test] -fn a_peer_that_closes_with_its_last_ack_is_a_connection() { - let mut net = Net::new(); - let isn = net.syn(5001); - net.ack_and(5001, isn, TcpControl::Fin); - assert_eq!(net.socket().state(), tcp::State::CloseWait, "the premise: both in one pass"); - assert!(net.wakes(true), "a connection the peer half-closed was never announced"); - assert_eq!(net.accept(true, true), Accept::Take(()), "a connection the peer half-closed was not handed over"); -} - -/// A peer that resets before its owner took it frees the port: the next peer -/// is answered a SYN-ACK and not a reset. -#[test] -fn a_peer_that_resets_before_it_is_taken_frees_the_port() { - let mut net = Net::new(); - let isn = net.syn(5001); - net.ack_and(5001, isn, TcpControl::Rst); - assert_eq!(net.socket().state(), tcp::State::Closed, "the premise: both in one pass"); - assert!(!net.wakes(false)); - assert!(net.listens(5002), "the port answered the next peer {:?}", net.sent.last()); -} - -/// A wake written for a connection its peer then reset is spent by the -/// accept that finds nothing, and the next connection is announced. -#[test] -fn a_wake_spent_on_a_reset_connection_announces_the_next() { - let mut net = Net::new(); - let isn = net.syn(5001); - net.send(5001, TcpControl::None, PEER_ISN + 1, Some(isn + 1)); - net.pass(); - assert!(net.wakes(true)); - net.send(5001, TcpControl::Rst, PEER_ISN + 1, Some(isn + 1)); - net.pass(); - assert!(!net.wakes(true), "a reset connection was announced"); - let isn = net.syn(5002); - assert_eq!(net.accept(true, true), Accept::Nothing, "an accept took a connection its peer had reset"); - net.send(5002, TcpControl::None, PEER_ISN + 1, Some(isn + 1)); - net.pass(); - assert!(net.wakes(true), "the connection after a reset one was never announced"); -} - -/// An accept that handed netstack no pipes spends the owner's wake, and the -/// connection it left is announced again at once. -#[test] -fn an_accept_refused_for_its_pipes_is_woken_again() { - let mut net = Net::new(); - let isn = net.syn(5001); - net.send(5001, TcpControl::None, PEER_ISN + 1, Some(isn + 1)); - net.pass(); - assert!(net.wakes(true)); - assert_eq!(net.accept(true, false), Accept::NoPipes); - assert!(net.wakes(true), "an owner refused for its pipes was never woken again"); - assert_eq!(net.accept(true, true), Accept::Take(())); -} - -/// An accept refused for room spends the owner's wake, and the connection it -/// left is announced again once room returns, and not before. -#[test] -fn an_accept_refused_for_room_is_woken_again_when_room_returns() { - let mut net = Net::new(); - assert_eq!(net.accept(false, true), Accept::Nothing, "an accept with nothing waiting was refused for room"); - let isn = net.syn(5001); - net.send(5001, TcpControl::None, PEER_ISN + 1, Some(isn + 1)); - net.pass(); - assert!(!net.wakes(false), "the owner was woken for a connection there is no room to take"); - assert!(net.wakes(true)); - assert_eq!(net.accept(false, true), Accept::NoRoom); - assert!(!net.wakes(false), "an owner refused for room was woken again with room still gone"); - assert!(net.wakes(true), "an owner refused for room was never woken again"); - assert_eq!(net.accept(true, true), Accept::Take(())); -} - -/// **A connection netstack aborts keeps its socket until the reset has left.** -/// A socket taken out of the set on the pass that aborted it sends nothing, -/// and its peer holds a connection nobody will ever answer. -#[test] -fn an_aborted_connection_is_spent_once_its_reset_has_left() { - let mut net = Net::new(); - let isn = net.syn(5001); - net.send(5001, TcpControl::None, PEER_ISN + 1, Some(isn + 1)); - net.pass(); - assert!(!crate::spent(net.socket()), "an established connection was spent"); - net.socket().abort(); - assert!(!crate::spent(net.socket()), "an aborted connection was spent with its reset still owed"); - net.pass(); - let last = net.sent.last().expect("an abort is said on the wire"); - assert_eq!((last.control, last.to), (TcpControl::Rst, 5001)); - assert!(crate::spent(net.socket()), "a connection whose reset has left was kept"); -} - -/// A connection both ends have closed is spent while it only waits out -/// `TimeWait`, and not while its own FIN is unanswered. -#[test] -fn a_connection_closed_by_both_ends_is_spent() { - let mut net = Net::new(); - let isn = net.syn(5001); - net.send(5001, TcpControl::None, PEER_ISN + 1, Some(isn + 1)); - net.pass(); - net.socket().close(); - net.pass(); - assert!(!crate::spent(net.socket()), "a connection was spent with its FIN unanswered"); - net.send(5001, TcpControl::Fin, PEER_ISN + 1, Some(isn + 2)); - net.pass(); - assert_eq!(net.socket().state(), tcp::State::TimeWait, "the premise: the peer's FIN answered ours"); - assert!(crate::spent(net.socket()), "a connection both ends closed was kept"); - assert_eq!(crate::ownerless(net.socket(), Duration::ZERO, false), Ownerless::Over); -} - -/// The handshake from `from` finished: a connection a client would hold. -fn established(net: &mut Net, from: u16) -> u32 { - let isn = net.syn(from); - net.send(from, TcpControl::None, PEER_ISN + 1, Some(isn + 1)); - net.pass(); - assert_eq!(net.socket().state(), tcp::State::Established); - isn -} - -/// A second at a time from the client's leaving to [`OWNERLESS_LIFE`], the peer -/// doing `each_second`: the connection waits for all of it, and what it is at -/// the end is the answer. -fn at_the_ceiling(net: &mut Net, mut each_second: impl FnMut(&mut Net)) -> Ownerless { - for second in 0..OWNERLESS_LIFE.as_secs() { - each_second(net); - net.pass(); - let gone = Duration::from_secs(second); - assert_eq!( - crate::ownerless(net.socket(), gone, false), - Ownerless::Waits, - "after {second}s in {}", - net.socket().state() - ); - net.now += smoltcp::time::Duration::from_secs(1); - } - net.pass(); - crate::ownerless(net.socket(), OWNERLESS_LIFE, false) -} - -/// The reset [`Ownerless::Cut`] owes leaves at the next poll, and the -/// connection is over. -fn the_cut_is_said(net: &mut Net) { - net.pass(); - let last = net.sent.last().expect("a cut connection's reset"); - assert_eq!((last.control, last.to), (TcpControl::Rst, 5001)); - assert_eq!(crate::ownerless(net.socket(), Duration::ZERO, true), Ownerless::Over); -} - -/// The half of netstack's pass that comes before its bridge: the interface -/// polled until it has nothing to do, and nothing read off the wire since. -/// What it sent stays on the wire for [`Net::pass`] to answer. -fn poll_only(net: &mut Net) { - while net.iface.poll(net.now, &mut net.wire, &mut net.sockets) != PollResult::None {} -} - -/// A connection in `FIN-WAIT-2` whose peer acknowledged the FIN once and then -/// said nothing, cut at the ceiling: its next hop's neighbour entry ran out -/// forty seconds before. -fn cut_after_a_silence(net: &mut Net) { - let isn = established(net, 5001); - net.socket().close(); - net.pass(); - net.send(5001, TcpControl::None, PEER_ISN + 1, Some(isn + 2)); - assert_eq!(at_the_ceiling(net, |_| {}), Ownerless::Cut); -} - -/// **A cut connection is kept until its reset has left.** The pass after the -/// cut asks the next hop for its address and sends nothing else; the reset -/// leaves on the pass that reads the answer. -#[test] -fn a_cut_connection_whose_neighbour_entry_ran_out_is_kept_until_its_reset_leaves() { - let mut net = Net::new(); - cut_after_a_silence(&mut net); - let said = net.sent.len(); - poll_only(&mut net); - let asked: Vec<_> = - net.wire.outbound.iter().map(|f| EthernetFrame::new_checked(&f[..]).unwrap().ethertype()).collect(); - assert_eq!(asked, [EthernetProtocol::Arp], "the premise: the reset waits on the next hop's address"); - assert_eq!( - crate::ownerless(net.socket(), Duration::ZERO, true), - Ownerless::Waits, - "a cut connection was let go before its next hop could answer" - ); - net.pass(); - assert_eq!(net.sent.len(), said + 1, "the next hop answered and the reset stayed"); - the_cut_is_said(&mut net); -} - -/// **And no longer than [`RESET_LIFE`] when the next hop answers nothing.** -#[test] -fn a_cut_connection_whose_next_hop_answers_no_arp_is_let_go_after_the_reset_bound() { - let mut net = Net::new(); - cut_after_a_silence(&mut net); - net.arp = false; - let said = net.sent.len(); - for second in 0..RESET_LIFE.as_secs() { - net.pass(); - assert_eq!( - crate::ownerless(net.socket(), Duration::from_secs(second), true), - Ownerless::Waits, - "{second}s after the cut" - ); - net.now += smoltcp::time::Duration::from_secs(1); - } - net.pass(); - assert_eq!(crate::ownerless(net.socket(), RESET_LIFE, true), Ownerless::Unsaid, "the slot never comes back"); - assert_eq!(net.sent.len(), said, "the premise: no segment leaves without a next hop"); -} - -/// **A connection netstack aborted before the ceiling is cut there too**, its -/// reset still owed: the neighbour entry the handshake made has run out and -/// the next hop answers no ARP. -#[test] -fn an_aborted_connection_whose_next_hop_answers_no_arp_is_cut_at_the_ceiling() { - let mut net = Net::new(); - established(&mut net, 5001); - net.now += smoltcp::time::Duration::from_secs(61); - net.arp = false; - net.socket().abort(); - let said = net.sent.len(); - assert_eq!(at_the_ceiling(&mut net, |_| {}), Ownerless::Cut); - assert_eq!(crate::ownerless(net.socket(), RESET_LIFE, true), Ownerless::Unsaid); - assert_eq!(net.sent.len(), said, "the premise: no segment leaves without a next hop"); -} - -/// **A peer that acknowledges nothing of a full send buffer holds a connection -/// with no client no longer than the ceiling.** The socket takes no more of -/// the client's bytes for as long as it lives, so what its send pipe still -/// holds is never read. -#[test] -fn a_peer_that_acknowledges_nothing_of_a_full_send_buffer_is_reset_at_the_ceiling() { - let mut net = Net::new(); - established(&mut net, 5001); - assert_eq!(net.socket().send_slice(&[0u8; 4096]), Ok(4096), "the premise: the bench's send buffer is full"); - let cut = at_the_ceiling(&mut net, |net| assert!(!crate::send_room(net.socket()), "the premise: no send room")); - assert_eq!(cut, Ownerless::Cut, "the slot of a connection with no client never comes back"); - the_cut_is_said(&mut net); -} - -/// **A peer that goes silent after the handshake holds a connection with no -/// client no longer than the ceiling.** Its FIN is never acknowledged, and -/// smoltcp retransmits it for as long as the socket lives. -#[test] -fn a_peer_gone_silent_is_reset_at_the_ceiling() { - let mut net = Net::new(); - established(&mut net, 5001); - net.socket().close(); - assert_eq!(at_the_ceiling(&mut net, |_| {}), Ownerless::Cut, "the slot of a connection with no client never comes back"); - assert_eq!(net.socket().state(), tcp::State::Closed); - the_cut_is_said(&mut net); -} - -/// **Nor does a peer that keeps answering.** It acknowledges the FIN, never -/// sends its own, and says so again every second: `FinWait2` has no timer, and -/// a socket timeout counted from the peer's last word would never run out. -#[test] -fn a_peer_that_answers_and_never_closes_is_reset_at_the_ceiling() { - let mut net = Net::new(); - let isn = established(&mut net, 5001); - net.socket().close(); - let answer = |net: &mut Net| net.send(5001, TcpControl::None, PEER_ISN + 1, Some(isn + 2)); - assert_eq!(at_the_ceiling(&mut net, answer), Ownerless::Cut, "the slot of a connection with no client never comes back"); - the_cut_is_said(&mut net); -} diff --git a/userland/netstack/src/main.rs b/userland/netstack/src/main.rs index b0c27ccc70..619aa93f2a 100644 --- a/userland/netstack/src/main.rs +++ b/userland/netstack/src/main.rs @@ -1,21 +1,58 @@ -use std::collections::HashMap; -use std::time::{Duration, Instant}; -use toyos::poller::{OTHER_END_GONE, READABLE, WRITABLE, Poller}; -use toyos::ipc; -use toyos::AsHandle; -use toyos::ipc::RxStep; +//! The network server: one Ethernet card claimed from the kernel, ToyOS's own +//! stack over it (`toyos-net-node`), served to programs over the pipe ABI +//! (`toyos::net`). +//! +//! **This program holds no protocol and reads no frame.** Every byte off the +//! wire goes to the node as the card handed it over, and every decision about +//! an address, a socket or a segment is the node's. What is here is what the +//! node cannot do and stay pure: the card (`card`), the clock, the kernel's +//! random source, the kernel's pipes (`pipes`), the clients' connections +//! (`client`) and their requests (`serve`). +//! +//! **One pass**: the card's link and its received frames go to the node, then +//! every deadline that is due, then the node is offered the card's transmit +//! room until it has no frame left or the card no room; what the node has to +//! say is written to the log and to the clients that waited for it; and the +//! loop sleeps on the kernel until the card, a client, a watched pipe or the +//! node's next deadline wakes it. A request is carried out in the pass that +//! read it and its frames leave in the next, which follows at once. +//! +//! **A frame leaves only into room the card said it has** (`Card::tx_room`): +//! the node builds none without it, and a card with none wakes the pass that +//! has (`Card::wake_on_room`). +//! +//! **Every draw is the kernel's**, the stack's secrets and each id, port and +//! transaction id after them, and a kernel that refuses one ends netstack by +//! name: a value anyone can predict is a forged answer's way in. + +use std::time::{Duration, Instant as Wall}; + +use toyos::endow; +use toyos::ipc::{self, RxStep}; +use toyos::poller::{Poller, READABLE}; use toyos::say; +use toyos::AsHandle; +use toyos_abi::syscall::PciId; +use toyos_dhcp::{HostName, Lease}; +use toyos_i219::Part; +use toyos_inspect::Snapshot; +use toyos_net_ip::Nud; +use toyos_net_node::Node; +use toyos_net_shard::{Config, Secrets}; +use toyos_net_wire::ethernet::{IndividualMac, MacAddr}; +use toyos_net_wire::Instant; mod card; mod client; mod device; -mod dhcp; mod i219; -mod listen; -mod mdns; -mod resolve; +mod pipes; +mod serve; mod virtio_net; +use card::Card; +use client::{Client, ClientRx, HANDSHAKE_TIMEOUT, MAX_KEPT_REQUEST, MAX_PENDING_CONNS, PendingConn, Request}; + /// The cards this program can drive, named by what identifies one rather than /// by the slot firmware put it in, and each with the driver that opens it. The /// manifest row spells the same pair and the claim arrives under a label @@ -34,398 +71,61 @@ const CARDS: [(PciId, fn(toyos::PciDev) -> Card); 3] = [ (PciId { vendor: 0x1af4, device: 0x1041 }, Card::virtio), ]; -use toyos::endow; -use toyos::Pipe; -use toyos_abi::syscall::PciId; -use toyos_i219::Part; -use toyos_inspect::Snapshot; -use virtio_net::VirtioNet; - -use card::Card; -use client::{Client, ClientRx, HANDSHAKE_TIMEOUT, MAX_KEPT_REQUEST, MAX_PENDING_CONNS, PendingConn, Request}; - -use toyos::net::*; - -use smoltcp::iface::{Config, Interface, PollResult, SocketHandle, SocketSet}; -use smoltcp::phy::{self, Device, DeviceCapabilities, Medium}; -use smoltcp::socket::{dhcpv4, tcp, udp}; -use smoltcp::time::Instant as SmoltcpInstant; -use smoltcp::wire::{EthernetAddress, HardwareAddress, IpAddress, IpEndpoint, IpListenEndpoint}; - -use std::net::Ipv4Addr; - -// --- smoltcp Device wrapper --- - -/// The driver, as smoltcp's `Device`. -/// -/// A thin adapter: every token below borrows the driver rather than a claim -/// handle, because the ring the token gives back to is this process's own. -struct DmaNic { - nic: Card, -} - -impl DmaNic { - /// Whether the card takes a frame now. Where it does not, its claim reads - /// ready when it will. - fn room(&self) -> bool { - self.nic.tx_room() > 0 || self.nic.wake_on_room() > 0 - } -} - -impl Device for DmaNic { - type RxToken<'a> = DmaRxToken<'a>; - type TxToken<'a> = DmaTxToken<'a>; - - fn receive(&mut self, _timestamp: SmoltcpInstant) -> Option<(Self::RxToken<'_>, Self::TxToken<'_>)> { - // A frame is taken off the receive ring only with room to answer it: - // smoltcp takes a transmit token with every frame it receives. - if !self.room() { - return None; - } - let token = match &self.nic { - Card::Virtio(nic) => { - nic.poll_rx().map(|(index, len)| DmaRxToken::Virtio { nic, index, len }) - } - Card::Intel(nic) => nic.poll_rx().map(|frame| DmaRxToken::Intel { nic, frame }), - }?; - Some((token, DmaTxToken { nic: &self.nic })) - } - - fn transmit(&mut self, _timestamp: SmoltcpInstant) -> Option> { - self.room().then_some(DmaTxToken { nic: &self.nic }) - } - - fn capabilities(&self) -> DeviceCapabilities { - let mut caps = DeviceCapabilities::default(); - caps.max_transmission_unit = 1514; - caps.medium = Medium::Ethernet; - caps - } -} - -/// One received frame, holding the driver it came from rather than a tag that -/// says which — so a frame and a driver that do not go together is not a state -/// this program can be in. -enum DmaRxToken<'a> { - Virtio { nic: &'a VirtioNet, index: usize, len: usize }, - Intel { nic: &'a i219::Nic, frame: toyos_i219::Frame }, -} - -impl phy::RxToken for DmaRxToken<'_> { - fn consume(self, f: F) -> R - where - F: FnOnce(&[u8]) -> R, - { - // The borrow ends with `f`, and the buffer goes back to the device only - // afterwards: smoltcp's `consume` takes `FnOnce(&[u8])`, so the - // callback cannot keep the reference past its own return. - match self { - Self::Virtio { nic, index, len } => { - let result = f(nic.rx_frame(index, len)); - nic.rx_done(index); - result - } - Self::Intel { nic, frame } => { - let result = f(nic.rx_frame(&frame)); - nic.rx_done(frame); - result - } - } - } -} - -struct DmaTxToken<'a> { - nic: &'a Card, -} - -impl phy::TxToken for DmaTxToken<'_> { - fn consume(self, len: usize, f: F) -> R - where - F: FnOnce(&mut [u8]) -> R, - { - // Filling and sending are one call, because the buffer belongs to the - // descriptor the driver picks: a frame written before one was taken - // would be written into a buffer the device may still be reading. - self.nic.tx(len, f) - } -} - -// --- Socket tracking --- - -enum SocketKind { - TcpStream(SocketHandle), - TcpListener(SocketHandle), - Udp(SocketHandle), -} - -struct UdpPipes { - tx_read: Pipe, - rx_write: Pipe, -} - -struct PendingUdpRecv { - client: Client, - socket_id: u32, - max_len: u32, -} - -/// A piped TCP connection: data flows through kernel pipes instead of IPC messages. -/// -/// **The socket and its id live exactly as long as this does.** What ends a -/// connection is what the kernel says of the client's pipe ends and what the -/// peer says on the wire; a close request only asks for that end early, and a -/// client that dies sends none. -/// -/// **A client is gone when the kernel says so of both its ends** -/// ([`OTHER_END_GONE`], watched on each pipe for as long as netstack holds -/// it), whatever its send pipe still holds and whether or not the socket -/// takes bytes. A direction netstack itself has closed counts as gone. Such a -/// connection is [`ownerless`]: the wire has [`OWNERLESS_LIFE`] to finish it, -/// and a reset that then cannot leave has [`RESET_LIFE`]. A client holding an -/// end of a direction still open is never timed. -struct PipedConnection { - socket_id: u32, - handle: SocketHandle, - rx_write: Option, - tx_read: Option, - /// The kernel said no holder of the send pipe's write end is left. What - /// the pipe holds is still the peer's. - writer_gone: bool, - /// When a pass first found the client gone, or cut the connection. - ownerless: Option, - /// [`Ownerless::Cut`] was this connection's answer. - cut: bool, - /// The client's receive pipe refused bytes the socket still holds, so the - /// pipe is watched for room. - held: bool, -} - -impl PipedConnection { - /// **A client's handle that refuses netstack for any reason but a full pipe or - /// a vanished reader ends that client's connection, never netstack.** The - /// ends are whatever the client moved, and nothing checks their kind at - /// intake: a read end or a handle with no `WRITE` right answers a refusal - /// here, and one that is no pipe end has its watch refused. So does a pipe - /// whose ring page could not be allocated, which no wait cures. - fn refuse(&mut self, socket: &mut tcp::Socket, end: &str, e: toyos_abi::syscall::SyscallError) { - say!("netstack: resetting a connection — its {end} pipe refused netstack: {e:?}"); - socket.abort(); - self.close_all(); - } - - fn close_rx(&mut self) { - self.rx_write.take(); - } - - fn close_tx(&mut self) { - self.tx_read.take(); - } - - fn close_all(&mut self) { - self.close_rx(); - self.close_tx(); - } - - /// The client holds no end of a direction that is still open. - fn clientless(&self) -> bool { - self.rx_write.is_none() && (self.tx_read.is_none() || self.writer_gone) - } - - /// What the kernel answered a watch on the client's send pipe, or on its - /// receive pipe. **Of a pipe netstack still holds**: closing one ends its - /// watch, and that end is an answer too. - fn pipe_answered( - &mut self, - socket: &mut tcp::Socket, - send: bool, - answer: Result, - ) { - let (held, end) = if send { (&self.tx_read, "send") } else { (&self.rx_write, "receive") }; - match answer { - _ if held.is_none() => {} - Err(e) => self.refuse(socket, end, e), - Ok(met) if met & OTHER_END_GONE == 0 => {} - Ok(_) if send => self.writer_gone = true, - // Nobody is left to read it. - Ok(_) => self.close_rx(), - } - } -} - -/// A piped TCP listener: netstack writes 1 byte to notify pipe on new connection. -struct PipedListener { - handle: SocketHandle, - notify_write: Pipe, - listening: listen::Listening, -} - -struct PendingPipedConnect { - client: Client, - socket_id: u32, - handle: SocketHandle, - /// Held from the moment the request arrived. The ends came *with* it, so - /// there is nothing left to open when the handshake completes and nothing - /// to fail there — where a pipe id could still be refused after netstack had - /// already told smoltcp to connect. - pipes: DataPipes, - deadline: Option, -} - -/// The two ends of a client's data path, as the client's request handed them -/// over. -/// -/// A pipe end travels as itself now: the client makes both pipes, keeps the -/// ends facing itself, and moves these two. They used to be ids in the request -/// payload, which netstack reopened by number — and any peer of the pipe's creator -/// could have named the same one. -struct DataPipes { - to_client: Pipe, - from_client: Pipe, -} - -impl DataPipes { - /// Take the pair the frame just read off `client` promised. - fn take(client: &Client) -> Option { - let [to_client, from_client] = client.conn.recv_handles_exact::<{ DATA_HANDLES }>()?; - Some(Self { - to_client: unsafe { Pipe::from_raw(to_client) }, - from_client: unsafe { Pipe::from_raw(from_client) }, - }) - } -} - -/// Whether `socket` takes a client's bytes now: it is sending, and its send -/// buffer has room. -fn send_room(socket: &tcp::Socket) -> bool { - socket.can_send() && socket.send_capacity() > socket.send_queue() -} - -/// Whether `socket` has said its last to its peer: it is closed, and the reset -/// an abort owes has left. `TimeWait` only waits. -fn spent(socket: &tcp::Socket) -> bool { - !socket.is_open() && !(socket.state() == tcp::State::Closed && socket.remote_endpoint().is_some()) -} - -/// How long the wire has to finish a connection whose client's pipe ends are -/// both gone, before netstack resets it: R2, the time RFC 9293 §3.8.3 gives a -/// segment's retransmission before the connection is closed, at the 100 -/// seconds it asks for at least. -/// -/// **From the client's leaving and not from the peer's last word**, so a peer -/// that keeps answering holds a slot no longer than one that says nothing: -/// RFC 9293 §3.8.6.1 lets a system reclaim a connection its peer holds open. -const OWNERLESS_LIFE: Duration = Duration::from_secs(100); - -/// How long the reset of a connection cut at [`OWNERLESS_LIFE`] has to leave. -/// A connection that sent and heard nothing for that long has outlived its -/// next hop's neighbour entry, so the reset waits on an ARP answer. -/// -/// Address resolution's own budget: RFC 4861 §7.2.2 fails it after -/// `MAX_MULTICAST_SOLICIT` solicitations `RETRANS_TIMER` apart, 3 and 1,000 -/// milliseconds in §10. That is IPv6's; RFC 1122 §2.3.2.1 gives ARP a rate of -/// one request a second per destination and no count, and smoltcp asks at that -/// rate for as long as the socket lives. -const RESET_LIFE: Duration = Duration::from_secs(3); - -/// What a pass makes of a connection whose client is gone. -#[derive(Debug, PartialEq, Eq)] -enum Ownerless { - /// The wire still owes something, and has time left. - Waits, - /// Reset at [`OWNERLESS_LIFE`] with the wire unfinished, and kept until - /// the reset has left or [`RESET_LIFE`] is over. - Cut, - /// The wire is finished. - Over, - /// Let go with a reset that never left: no next hop took it. - Unsaid, -} - -/// The pass's answer for `socket`, whose client has been gone for `waited`, or -/// which was cut `waited` ago. -fn ownerless(socket: &mut tcp::Socket, waited: Duration, cut: bool) -> Ownerless { - if spent(socket) { - Ownerless::Over - } else if waited < if cut { RESET_LIFE } else { OWNERLESS_LIFE } { - Ownerless::Waits - } else if cut { - Ownerless::Unsaid - } else { - socket.abort(); - Ownerless::Cut - } -} - -fn piped_connection(socket_id: u32, handle: SocketHandle, pipes: DataPipes) -> PipedConnection { - PipedConnection { - socket_id, - handle, - rx_write: Some(pipes.to_client), - tx_read: Some(pipes.from_client), - writer_gone: false, - ownerless: None, - cut: false, - held: false, - } -} - -/// Poll registrations that are not piped connections: the service listener and -/// the NIC claim. -const FIXED_POLL_HANDLES: u32 = 2; - -/// Registrations one piped connection can make in a batch: its send pipe and -/// its receive pipe. -const POLL_HANDLES_PER_PIPED: u32 = 2; - -/// Registrations the lookups make in a batch: each waiting client's -/// connection, which is how netstack hears it hang up. -const LOOKUP_POLL_HANDLES: u32 = resolve::MAX_LOOKUPS as u32; - -/// Hard ceiling on live piped connections, from the poller rather than from -/// memory: netstack registers every connection's pipes in the same batch as the two -/// fixed registrations, the pending connections and the lookups' clients, and -/// `Poller::MAX_HANDLES` is the widest set one poller can carry. The memory -/// budget below binds first on a machine whose eighth holds fewer connections. -const MAX_PIPED_SLOTS: u64 = ((Poller::MAX_HANDLES - FIXED_POLL_HANDLES - MAX_PENDING_CONNS - LOOKUP_POLL_HANDLES) - / POLL_HANDLES_PER_PIPED) as u64; - -/// Payload bytes a UDP socket's receive buffer holds, and therefore the longest -/// datagram netstack can ever hand back — which is what bounds the buffer -/// [`Netstack::deliver_datagram`] sizes from a client's `max_len`. -const UDP_SOCKET_BUFFER: usize = 65536; - -/// Payload bytes each direction of a TCP socket buffers inside netstack, before the -/// window closes and the peer is asked to wait. -const TCP_SOCKET_BUFFER: usize = 65536; - -/// Physical memory one piped connection costs. A kernel pipe is exactly one -/// 2 MiB page (`kernel/src/pipe.rs`: `PIPE_SIZE = PAGE_2M`) and a piped socket -/// is two of them, one per direction. The client allocates them, but netstack -/// holding the far ends is what keeps them alive, so this is netstack's to bound. -const PIPED_CONNECTION_BYTES: u64 = 2 * 2 * 1024 * 1024; - -/// Share of physical memory netstack will keep tied up in client pipes. +/// The name this machine asks its network to record for it, and answers to on +/// it as `.local` once no other host does. One name, because there is +/// one machine. +pub const HOSTNAME: &str = "toyos-t14"; + +/// How long this machine waits for its first lease before saying it has none. +/// It bounds the report, never the client: the node asks for the life of the +/// boot and a lease that lands later is applied like any other. +const LEASE_BOUND: Duration = Duration::from_millis(toyos_tco::LEASE_BOUND_MS); + +/// Payload bytes each direction of a connection buffers in the stack, before +/// the window closes on the peer or the client's pipe stops being read. +const TCP_BUFFER: u32 = 65_535; + +/// A kernel pipe is one 2 MiB page (`kernel/src/pipe.rs`). The client +/// allocates it, and netstack holding its far end is what keeps it alive. +const PIPE_BYTES: u64 = 2 * 1024 * 1024; + +/// What one place can make this machine hold, at the largest of the three +/// things a place is (`toyos-net-node`'s `places`): a listener, whose peers +/// fill its queue of `LISTEN_READY` finished connections with a receive +/// buffer of text each, and its wake pipe. A stream is its two pipes and two +/// buffers, a datagram socket its two pipes and two queues, and both are +/// less. +const PLACE_BYTES: u64 = { + let listener = toyos_net_tcp::limits::LISTEN_READY as u64 * TCP_BUFFER as u64 + PIPE_BYTES; + let stream = 2 * PIPE_BYTES + 2 * TCP_BUFFER as u64; + let datagram = 2 * PIPE_BYTES + (toyos_net_udp::limits::RX_BYTES + toyos_net_udp::limits::TX_BYTES) as u64; + assert!(listener >= stream && listener >= datagram); + listener +}; + +/// Share of physical memory netstack lets its clients' places tie up. /// /// Policy, not derivation, and the same eighth the compositor takes for the -/// same reason: nothing in the kernel says what a process may use — no -/// per-process limit, no pressure signal, no OOM killer — so the quantity that -/// would make this derivable does not exist yet. -const PIPE_BUDGET_SHARE: u64 = 8; - -/// How many piped connections netstack will hold, given total physical memory. -/// -/// An eighth of memory divided by the two pipes a connection costs, floored at -/// one and capped at what one poller can watch. -/// -/// **A mitigation, not a policy anyone chose.** A piped connection's 4 MiB is -/// charged to nobody — no per-process limit, no pressure signal, no OOM killer -/// (`issues/no-physical-memory-fairness.md`) — so without a cap a client that opens sockets -/// in a loop walks the machine into exhaustion, and netstack has no way to tell -/// that from ordinary use. Delete this in favour of a kernel memory limit, not -/// in favour of a bigger number. -fn max_piped_connections(total_mem: u64) -> usize { - let budget = total_mem / PIPE_BUDGET_SHARE; - (budget / PIPED_CONNECTION_BYTES).clamp(1, MAX_PIPED_SLOTS) as usize +/// same reason: nothing in the kernel says what a process may use +/// (`issues/no-physical-memory-fairness.md`), so without a bound a client +/// that opens sockets in a loop walks the machine into exhaustion. Delete +/// this in favour of a kernel memory limit, not in favour of a bigger number. +const PLACE_BUDGET_SHARE: u64 = 8; + +/// Watches that are no place's: the service's acceptor and the card's claim. +const FIXED_WATCHES: u32 = 2; + +/// The most places one poller can watch: every place's pipes are asked about +/// in the same batch as the fixed watches, the connections that have not said +/// what they want and the lookups' clients. +const MAX_PLACES: u32 = + (Poller::MAX_HANDLES - FIXED_WATCHES - MAX_PENDING_CONNS - serve::LOOKUP_WATCHES) / serve::WATCHES_PER_PLACE; + +/// How many places the node is given, of `total_mem` bytes of physical +/// memory: its share by what a place costs, at least one, at most what the +/// poller watches. +fn places_for(total_mem: u64) -> usize { + (total_mem / PLACE_BUDGET_SHARE / PLACE_BYTES).clamp(1, u64::from(MAX_PLACES)) as usize } /// Total physical memory, as the kernel reports it. @@ -436,964 +136,159 @@ fn total_memory() -> u64 { toyos_abi::syscall::SysinfoHeader::decode(&buf).memory_total } -/// Where this netstack's socket ids start: at random, and never 0. -/// -/// **A client holds a socket id across netstack being replaced** (`toyos-swap`): it -/// learns the old netstack is gone when a request fails, and closing what it held -/// is its first reaction. Every netstack counting from 1 made that stale number -/// another client's live socket in the new one. A random start makes two -/// instances' ranges overlap only by a chance the size of their lengths over -/// 2^32, which bounds the harm and does not remove it: an id is a number any -/// client can name (`issues/netstack-socket-ids-are-ambient.md`). -fn first_socket_id() -> u32 { +/// One draw of the kernel's random source. +fn draw() -> u32 { let mut bytes = [0u8; 4]; toyos_abi::syscall::random(&mut bytes) - .unwrap_or_else(|e| panic!("netstack: the kernel's random source refused the first socket id: {e:?}")); - u32::from_le_bytes(bytes).max(1) -} - -/// smoltcp's one random source, which every TCP connection's initial sequence -/// number (RFC 6528 asks for one an off-path sender cannot predict) and every -/// DHCP transaction ID are drawn from. `Config::new` seeds it with 0, which -/// is the same sequence on every boot of every machine. -fn smoltcp_seed() -> u64 { - let mut bytes = [0u8; 8]; - toyos_abi::syscall::random(&mut bytes) - .unwrap_or_else(|e| panic!("netstack: the kernel's random source refused smoltcp's seed: {e:?}")); - u64::from_le_bytes(bytes) -} - -struct Netstack { - sockets: HashMap, - next_id: u32, - next_local_port: u16, - pending_udp_recvs: Vec, - resolver: resolve::Resolver u16>, - piped_connections: Vec, - piped_listeners: HashMap, - pending_piped_connects: Vec, - udp_pipes: HashMap, - max_piped_connections: usize, + .unwrap_or_else(|e| panic!("netstack: the kernel's random source refused a draw: {e:?}")); + u32::from_le_bytes(bytes) } -impl Netstack { - fn new(max_piped_connections: usize) -> Self { - Self { - sockets: HashMap::new(), - next_id: first_socket_id(), - next_local_port: 49152, - pending_udp_recvs: Vec::new(), - resolver: resolve::Resolver::new(Instant::now(), resolve::random_u16), - piped_connections: Vec::new(), - piped_listeners: HashMap::new(), - pending_piped_connects: Vec::new(), - udp_pipes: HashMap::new(), - max_piped_connections, - } - } - - /// Is there room for one more piped connection? - /// - /// Counts the connects still waiting for their SYN-ACK: they each already - /// name a pair of pipes, so leaving them out would let a burst of - /// `TCP_CONNECT_PIPED` overshoot the cap by the whole burst. - fn piped_room(&self) -> bool { - self.piped_live() < self.max_piped_connections - } - - /// Connections the cap is counting. Reported by both refusals, because - /// `piped_connections.len()` alone reads as "0 already, max 126" when a - /// burst of connects fills the pending list — a refusal that looks like a - /// bug in the check rather than the check working. - fn piped_live(&self) -> usize { - self.piped_connections.len() + self.pending_piped_connects.len() - } - - /// The socket table's size, as `inspect` reads it: counts, and no - /// endpoint, because every client holding `netstack` can ask. - /// - /// `sockets.untabled` is every socket the stack holds that no table entry - /// names, the resolver's left out: netstack's own, and any that outlived its - /// entry, which moves it and no other count. The resolver's are left out - /// because their number is every program's lookups in flight. - fn inspect(&self, snap: &mut Snapshot, socket_set: &SocketSet<'_>) { - let untabled = socket_set - .iter() - .count() - .checked_sub(self.sockets.len() + self.resolver.sockets()) - .expect("netstack: a table entry or a lookup names a socket the stack does not hold"); - snap.put("sockets.untabled", untabled); - let (mut streams, mut listeners, mut udp) = (0u32, 0u32, 0u32); - for kind in self.sockets.values() { - match kind { - SocketKind::TcpStream(_) => streams += 1, - SocketKind::TcpListener(_) => listeners += 1, - SocketKind::Udp(_) => udp += 1, - } - } - snap.put("sockets.tcp", streams); - snap.put("sockets.listeners", listeners); - snap.put("sockets.udp", udp); - snap.put("piped.live", self.piped_live()); - snap.put("piped.ownerless", self.piped_connections.iter().filter(|c| c.ownerless.is_some()).count()); - snap.put("piped.max", self.max_piped_connections); - } - - fn alloc_id(&mut self) -> u32 { - let id = self.next_id; - self.next_id = match self.next_id.wrapping_add(1) { - 0 => 1, - next => next, - }; - id - } - - fn alloc_port(&mut self) -> u16 { - let port = self.next_local_port; - self.next_local_port = if self.next_local_port >= 65535 { 49152 } else { self.next_local_port + 1 }; - port - } - - /// The first port from [`alloc_port`](Self::alloc_port)'s cursor that no - /// UDP socket holds, and the cursor moves past it; `None` once every one - /// is held. - fn alloc_free_udp_port(&mut self, socket_set: &SocketSet<'_>) -> Option { - self.next_local_port = resolve::free_port(socket_set, self.next_local_port)?; - Some(self.alloc_port()) - } - - /// Dispatch one whole request. - /// - /// A synchronous handler answers and lets the connection close where it - /// stands; an asynchronous one moves the [`Client`] into its pending list - /// and answers when what it started finishes. - fn handle_message( - &mut self, - req: Request, - socket_set: &mut SocketSet<'_>, - iface: &mut Interface, - ) { - match MsgType::from_u32(req.msg_type) { - Some(MsgType::TcpClose) => self.handle_tcp_close(&req, socket_set), - Some(MsgType::TcpShutdown) => self.handle_tcp_shutdown(&req, socket_set), - Some(MsgType::UdpBind) => self.handle_udp_bind(&req, socket_set), - Some(MsgType::UdpSendTo) => self.handle_udp_send_to(&req, socket_set), - Some(MsgType::UdpRecvFrom) => self.handle_udp_recv_from(req, socket_set), - Some(MsgType::UdpClose) => self.handle_udp_close(&req, socket_set), - Some(MsgType::DnsLookup) => self.handle_dns_lookup(req, socket_set), - Some(MsgType::TcpSetOption) => self.handle_tcp_set_option(&req, socket_set), - Some(MsgType::TcpListenerSetOption) => self.handle_tcp_listener_set_option(&req, socket_set), - Some(MsgType::UdpSetOption) => self.handle_udp_set_option(&req), - Some(MsgType::TcpConnectPiped) => self.handle_tcp_connect_piped(req, socket_set, iface), - Some(MsgType::TcpBindPiped) => self.handle_tcp_bind_piped(&req, socket_set), - Some(MsgType::TcpAcceptPiped) => self.handle_tcp_accept_piped(&req, socket_set), - None => { - say!("netstack: unknown message type {}", req.msg_type); - req.client.error(ERR_INVALID_INPUT); - } - } - } - - fn handle_tcp_close(&mut self, msg: &Request, socket_set: &mut SocketSet<'_>) { - let Ok(req) = ipc::decode_payload::(msg.payload()) else { - msg.client.error(ERR_INVALID_INPUT); - return; - }; - if let Some(kind) = self.sockets.remove(&req.socket_id) { - match kind { - SocketKind::TcpStream(handle) => { - socket_set.get_mut::(handle).close(); - socket_set.remove(handle); - if let Some(pos) = self.piped_connections.iter().position(|c| c.handle == handle) { - self.piped_connections.swap_remove(pos).close_all(); - } - // A connect still waiting for its SYN-ACK names the - // handle just removed, and the pass that would read it - // next is a panic; its client is answered instead. - if let Some(pos) = - self.pending_piped_connects.iter().position(|c| c.handle == handle) - { - self.pending_piped_connects.swap_remove(pos).client.error(ERR_CONNECTION_REFUSED); - } - } - SocketKind::TcpListener(handle) => { - socket_set.get_mut::(handle).abort(); - socket_set.remove(handle); - self.piped_listeners.remove(&req.socket_id); - } - SocketKind::Udp(handle) => { - socket_set.get_mut::(handle).close(); - socket_set.remove(handle); - self.udp_pipes.remove(&req.socket_id); - } - } - } - msg.client.done(); - } - - fn handle_tcp_shutdown(&mut self, msg: &Request, socket_set: &mut SocketSet<'_>) { - let Ok(req) = ipc::decode_payload::(msg.payload()) else { - msg.client.error(ERR_INVALID_INPUT); - return; - }; - let Some(SocketKind::TcpStream(handle)) = self.sockets.get(&req.socket_id) else { - msg.client.error(ERR_NOT_CONNECTED); - return; - }; - let socket = socket_set.get_mut::(*handle); - if req.how == 1 || req.how == 2 { - socket.close(); - } - msg.client.done(); - } - - fn handle_udp_bind(&mut self, msg: &Request, socket_set: &mut SocketSet<'_>) { - let Ok(req) = ipc::decode_payload::(msg.payload()) else { - msg.client.error(ERR_INVALID_INPUT); - return; - }; - let Some(pipes) = DataPipes::take(&msg.client) else { - msg.client.error(ERR_INVALID_INPUT); - return; - }; - let (rx_write, tx_read) = (pipes.to_client, pipes.from_client); - let port = match req.port { - 0 => match self.alloc_free_udp_port(socket_set) { - Some(port) => port, - None => { - msg.client.error(ERR_ADDR_IN_USE); - return; - } - }, - port if resolve::udp_port_taken(socket_set, port) => { - msg.client.error(ERR_ADDR_IN_USE); - return; - } - port => port, - }; - - let rx_buf = udp::PacketBuffer::new( - vec![udp::PacketMetadata::EMPTY; 16], - vec![0u8; UDP_SOCKET_BUFFER], - ); - let tx_buf = udp::PacketBuffer::new( - vec![udp::PacketMetadata::EMPTY; 16], - vec![0u8; UDP_SOCKET_BUFFER], - ); - let mut socket = udp::Socket::new(rx_buf, tx_buf); - // **The unspecified address binds as no address at all.** smoltcp - // reads `Some(0.0.0.0)` as a socket for datagrams addressed to - // 0.0.0.0, and none ever is, so a socket bound the ordinary way to - // receive on every address would receive nothing but broadcast. - let addr = Ipv4Addr::from(req.addr); - let endpoint = IpListenEndpoint { addr: (!addr.is_unspecified()).then_some(IpAddress::Ipv4(addr)), port }; - socket - .bind(endpoint) - .unwrap_or_else(|e| panic!("netstack: a fresh socket refused to bind the free port {port}: {e:?}")); - - let handle = socket_set.add(socket); - let socket_id = self.alloc_id(); - self.sockets.insert(socket_id, SocketKind::Udp(handle)); - self.udp_pipes.insert(socket_id, UdpPipes { tx_read, rx_write }); - - msg.client.result(&UdpBindResponse { - socket_id, - bound_port: port, - _pad: 0, - }); - } - - fn handle_udp_send_to(&mut self, msg: &Request, socket_set: &mut SocketSet<'_>) { - let Ok(req) = ipc::decode_payload::(msg.payload()) else { - msg.client.error(ERR_INVALID_INPUT); - return; - }; - - let Some(SocketKind::Udp(handle)) = self.sockets.get(&req.socket_id) else { - msg.client.error(ERR_NOT_CONNECTED); - return; - }; - let handle = *handle; - - let Some(pipes) = self.udp_pipes.get(&req.socket_id) else { - msg.client.error(ERR_NOT_CONNECTED); - return; - }; - - let mut buf = vec![0u8; req.len as usize]; - let n = match toyos_abi::syscall::read_nonblock(pipes.tx_read.as_handle(), &mut buf) { - Ok(n) => n, - // The client writes the datagram into the pipe and *then* sends this - // request, so an empty pipe is a client naming bytes it never put - // there. A blocking read here waits for a second write that a - // conforming client never makes. - Err(toyos_abi::syscall::SyscallError::WouldBlock) => { - msg.client.error(ERR_INVALID_INPUT); - return; - } - Err(_) => { - msg.client.error(ERR_OTHER); - return; - } - }; - - let addr = Ipv4Addr::from(req.addr); - let endpoint = IpEndpoint::new(IpAddress::Ipv4(addr), req.port); - let socket = socket_set.get_mut::(handle); - match socket.send_slice(&buf[..n], endpoint) { - Ok(()) => msg.client.result(&(n as u32)), - Err(_) => msg.client.error(ERR_OTHER), - } +/// One of the stack's keys: no draw is part of two. +fn key() -> [u8; 16] { + let mut key = [0u8; 16]; + for word in key.chunks_exact_mut(4) { + word.copy_from_slice(&draw().to_le_bytes()); } + key +} - /// Take one waiting datagram off `socket_id` for `client`, or hand the - /// client back when none has arrived. - /// - /// **A datagram goes into the client's pipe whole, or its socket ends.** - /// The answer names a length, and a write takes as much as the pipe has - /// room for and cannot be taken back: a client reading that length out of - /// a pipe holding part of this datagram would splice the next one onto it. - /// So a pipe that will not take one whole — full, gone, or not a pipe netstack - /// can write — ends the socket by name and answers its client a reset, and - /// nothing can follow the part it did take. - fn deliver_datagram( - &mut self, - client: Client, - socket_id: u32, - max_len: u32, - socket_set: &mut SocketSet<'_>, - ) -> Option { - let (Some(&SocketKind::Udp(handle)), Some(pipes)) = - (self.sockets.get(&socket_id), self.udp_pipes.get(&socket_id)) - else { - client.error(ERR_NOT_CONNECTED); - return None; - }; - let socket = socket_set.get_mut::(handle); - if !socket.can_recv() { - return Some(client); - } - // `max_len` is the client's number. Clamped rather than trusted: the - // socket's own receive buffer is 65536 bytes, so no datagram it can hand - // back is longer, and an unclamped `vec!` here is a 4 GiB allocation any - // client can ask netstack to make. - let mut bytes = vec![0u8; (max_len as usize).min(UDP_SOCKET_BUFFER)]; - let (n, meta) = match socket.recv_slice(&mut bytes) { - Ok(got) => got, - Err(_) => { - client.error(ERR_OTHER); - return None; - } - }; - let wrote = toyos_abi::syscall::write_nonblock(pipes.rx_write.as_handle(), &bytes[..n]); - if wrote == Ok(n) { - let IpAddress::Ipv4(addr) = meta.endpoint.addr; - client.result(&UdpRecvResponse { addr: addr.octets(), port: meta.endpoint.port, len: n as u16 }); - return None; - } - match wrote { - Ok(took) => say!("netstack: ending UDP socket {socket_id} — its receive pipe took {took} of a {n}-byte datagram"), - Err(e) => say!("netstack: ending UDP socket {socket_id} — its receive pipe refused a {n}-byte datagram: {e:?}"), - } - socket.close(); - socket_set.remove(handle); - self.sockets.remove(&socket_id); - self.udp_pipes.remove(&socket_id); - client.error(ERR_CONNECTION_RESET); - None +fn secrets() -> Secrets { + Secrets { + ip: key(), + resets: key(), + tcp: toyos_net_tcp::Secrets { + isn: key(), + timestamp: key(), + port_offset: key(), + port_index: key(), + port_table: std::array::from_fn(|_| draw() as u16), + }, } +} - fn handle_udp_recv_from(&mut self, msg: Request, socket_set: &mut SocketSet<'_>) { - let Ok(req) = ipc::decode_payload::(msg.payload()) else { - msg.client.error(ERR_INVALID_INPUT); - return; - }; - if let Some(client) = self.deliver_datagram(msg.client, req.socket_id, req.max_len, socket_set) { - // Nothing has arrived yet: keep the connection open until one does. - self.pending_udp_recvs.push(PendingUdpRecv { - client, - socket_id: req.socket_id, - max_len: req.max_len, - }); - } - } +/// What of a lease the log and `inspect` say: everything the server decided. +#[derive(Clone, PartialEq, Eq)] +struct Said { + address: std::net::Ipv4Addr, + prefix_len: u8, + router: Option, + server: std::net::Ipv4Addr, + dns: Vec, +} - fn handle_udp_close(&mut self, msg: &Request, socket_set: &mut SocketSet<'_>) { - let Ok(req) = ipc::decode_payload::(msg.payload()) else { - msg.client.error(ERR_INVALID_INPUT); - return; - }; - if let Some(SocketKind::Udp(handle)) = self.sockets.remove(&req.socket_id) { - socket_set.get_mut::(handle).close(); - socket_set.remove(handle); - self.udp_pipes.remove(&req.socket_id); - } - msg.client.done(); +impl Said { + fn of(lease: &Lease) -> Self { + Self { address: lease.address, prefix_len: lease.prefix_len, router: lease.router, server: lease.server, dns: lease.dns.clone() } } - /// Start resolving the name `msg` carries, or answer at once where there - /// is nothing to ask: an address written as one, or a name no server can - /// be asked for. - fn handle_dns_lookup(&mut self, msg: Request, socket_set: &mut SocketSet<'_>) { - let Ok(hostname) = std::str::from_utf8(msg.payload()) else { - msg.client.error(ERR_INVALID_INPUT); - return; - }; - if let Ok(ip) = hostname.parse::() { - answer_lookup(&msg.client, &[ip.octets()]); - return; - } - let Ok(name) = toyos_dns::Name::parse(hostname) else { - msg.client.error(ERR_INVALID_INPUT); - return; - }; - if let Err((client, why)) = self.resolver.start(msg.client, name, socket_set, Instant::now()) { - client.error(why.code()); - } + fn dns(&self) -> String { + self.dns.iter().map(ToString::to_string).collect::>().join(" ") } +} - fn handle_tcp_set_option(&mut self, msg: &Request, socket_set: &mut SocketSet<'_>) { - let Ok(req) = ipc::decode_payload::(msg.payload()) else { - msg.client.error(ERR_INVALID_INPUT); - return; - }; - let Some(SocketKind::TcpStream(handle)) = self.sockets.get(&req.socket_id) else { - msg.client.error(ERR_NOT_CONNECTED); - return; - }; - let socket = socket_set.get_mut::(*handle); - match req.option { - OPT_NODELAY => { - socket.set_nagle_enabled(req.value == 0); - msg.client.done(); - } - _ => msg.client.error(ERR_INVALID_INPUT), - } - } +/// What the boot's log has said of the lease, and still owes. +struct Leases { + began: Wall, + said: Option, + /// Whether this boot has settled the question once: a lease landed, or + /// the bound passed with none. + settled: bool, +} - fn handle_tcp_listener_set_option(&mut self, msg: &Request, socket_set: &mut SocketSet<'_>) { - let Ok(req) = ipc::decode_payload::(msg.payload()) else { - msg.client.error(ERR_INVALID_INPUT); - return; - }; - let Some(listener) = self.piped_listeners.get_mut(&req.socket_id) else { - msg.client.error(ERR_NOT_CONNECTED); - return; - }; - match req.option { - OPT_NODELAY => { - listener.listening.set_nodelay(socket_set.get_mut::(listener.handle), req.value != 0); - msg.client.done(); +impl Leases { + /// Says what changed of the lease the node holds, and answers whether + /// this machine's address question has just been settled. + fn pass(&mut self, held: Option<&Lease>) -> bool { + let held = held.map(Said::of); + if held != self.said { + match &held { + // One record carrying every field the lease decided: a boot + // read off a stick or a stream has this line and nothing else + // to say what this machine's network was. + Some(lease) => say!( + "netstack: DHCP: lease {}/{} from {}, gateway {}, dns [{}], {} ms after \ + netstack came up", + lease.address, + lease.prefix_len, + lease.server, + match lease.router { + Some(router) => router.to_string(), + None => "none".to_string(), + }, + lease.dns(), + self.began.elapsed().as_millis(), + ), + None => say!("netstack: DHCP: the lease is gone; this machine has no address"), + } + self.said = held; + } + if self.settled { + return false; + } + if self.said.is_none() { + if self.began.elapsed() < LEASE_BOUND { + return false; } - _ => msg.client.error(ERR_INVALID_INPUT), - } - } - - /// smoltcp sends to a broadcast address for every socket, so the - /// permission has nothing here to switch. - fn handle_udp_set_option(&self, msg: &Request) { - let Ok(req) = ipc::decode_payload::(msg.payload()) else { - msg.client.error(ERR_INVALID_INPUT); - return; - }; - let Some(SocketKind::Udp(_)) = self.sockets.get(&req.socket_id) else { - msg.client.error(ERR_NOT_CONNECTED); - return; - }; - match req.option { - OPT_BROADCAST => msg.client.done(), - _ => msg.client.error(ERR_INVALID_INPUT), - } - } - - // --- Piped socket handlers --- - - fn handle_tcp_connect_piped( - &mut self, - msg: Request, - socket_set: &mut SocketSet<'_>, - iface: &mut Interface, - ) { - let Ok(req) = ipc::decode_payload::(msg.payload()) else { - msg.client.error(ERR_INVALID_INPUT); - return; - }; - // Refused before the socket exists, so a refusal leaves nothing to - // unwind and no SYN on the wire. An error return, never a panic: the - // request is a client's and asking for one connection too many is not - // a bug in netstack. - // - // Not `ERR_CONNECTION_REFUSED`, which this file already uses below for - // a pending connect whose socket reached `Closed` — the peer's answer. - // On one code a client cannot tell "this machine is full, back off" - // from "that peer says no, give up". - if !self.piped_room() { say!( - "netstack: refusing connect, {} piped connections already (max {})", - self.piped_live(), - self.max_piped_connections, + "netstack: DHCP: no lease as {} in {} s; this machine has no address and every \ + connect through it is refused", + HOSTNAME, + self.began.elapsed().as_secs(), ); - msg.client.error(ERR_RESOURCE_EXHAUSTED); - return; - } - // Taken before the socket exists, for the same reason the capacity - // check is: a missing pair leaves nothing to unwind and no SYN on the - // wire. - let Some(pipes) = DataPipes::take(&msg.client) else { - msg.client.error(ERR_INVALID_INPUT); - return; - }; - let remote = IpEndpoint::new( - IpAddress::Ipv4(Ipv4Addr::from(req.addr)), - req.port, - ); - if req.port == 0 || remote.addr.is_unspecified() { - msg.client.error(ERR_INVALID_INPUT); - return; - } - // **This machine holding no address is not a peer's refusal.** Before - // the lease there is no source for a SYN, and the socket's own - // `Unaddressable` would reach the client as `ERR_CONNECTION_REFUSED`, - // which says "that peer says no, give up" about a condition of this - // machine that clears when the lease lands. - if iface.ipv4_addr().is_none() { - msg.client.error(ERR_NOT_CONNECTED); - return; - } - let local_port = self.alloc_port(); - - let rx_buf = tcp::SocketBuffer::new(vec![0u8; TCP_SOCKET_BUFFER]); - let tx_buf = tcp::SocketBuffer::new(vec![0u8; TCP_SOCKET_BUFFER]); - let mut socket = tcp::Socket::new(rx_buf, tx_buf); - if socket.connect(iface.context(), remote, local_port).is_err() { - msg.client.error(ERR_CONNECTION_REFUSED); - return; } - - let handle = socket_set.add(socket); - let socket_id = self.alloc_id(); - self.sockets.insert(socket_id, SocketKind::TcpStream(handle)); - - let deadline = if req.timeout_ms > 0 { - Some(Instant::now() + Duration::from_millis(req.timeout_ms as u64)) - } else { - None - }; - - // Async — hold the connection until the handshake completes. - self.pending_piped_connects.push(PendingPipedConnect { - client: msg.client, - socket_id, - handle, - pipes, - deadline, - }); + self.settled = true; + true } - fn handle_tcp_bind_piped(&mut self, msg: &Request, socket_set: &mut SocketSet<'_>) { - let Ok(req) = ipc::decode_payload::(msg.payload()) else { - msg.client.error(ERR_INVALID_INPUT); - return; - }; - let port = if req.port == 0 { self.alloc_port() } else { req.port }; - - // Take the pipe before the socket goes into socket_set: a missing one - // then has no half-built socket to unwind. - let Some([notify]) = msg.client.conn.recv_handles_exact::<{ NOTIFY_HANDLES }>() else { - msg.client.error(ERR_INVALID_INPUT); - return; - }; - let notify_write = unsafe { Pipe::from_raw(notify) }; - - let rx_buf = tcp::SocketBuffer::new(vec![0u8; TCP_SOCKET_BUFFER]); - let tx_buf = tcp::SocketBuffer::new(vec![0u8; TCP_SOCKET_BUFFER]); - let mut socket = tcp::Socket::new(rx_buf, tx_buf); - let listening = listen::Listening::new(port, req.options.nodelay()); - // Before the socket is in the set, so before a SYN can reach it. A - // fresh socket and a port that is not zero: neither refusal `listen` - // has can be this one. - listening.open(&mut socket); - - let handle = socket_set.add(socket); - let socket_id = self.alloc_id(); - self.sockets.insert(socket_id, SocketKind::TcpListener(handle)); - - self.piped_listeners.insert(socket_id, PipedListener { handle, notify_write, listening }); - - msg.client.result(&TcpBindResponse { - socket_id, - bound_port: port, - _pad: 0, - }); + /// How long until the bound's report is due, while it is owed. + fn due_in(&self) -> Option { + (!self.settled).then(|| LEASE_BOUND.saturating_sub(self.began.elapsed())) } - fn handle_tcp_accept_piped(&mut self, msg: &Request, socket_set: &mut SocketSet<'_>) { - let Ok(req) = ipc::decode_payload::(msg.payload()) else { - msg.client.error(ERR_INVALID_INPUT); + /// The lease as `inspect` reads it, and the router's entry in the + /// stack's neighbour table where the lease names one. + fn inspect(&self, node: &Node, snap: &mut Snapshot) { + let Some(held) = &self.said else { + snap.put("lease.held", false); return; }; - let (room, pipes) = (self.piped_room(), DataPipes::take(&msg.client)); - let Some(listener) = self.piped_listeners.get_mut(&req.socket_id) else { - msg.client.error(ERR_NOT_CONNECTED); - return; - }; - let (old_handle, local_port) = (listener.handle, listener.listening.port()); - let pipes = match listener.listening.accept(socket_set.get_mut::(old_handle), room, pipes) { - listen::Accept::Take(pipes) => pipes, - listen::Accept::NoPipes => { - msg.client.error(ERR_INVALID_INPUT); - return; - } - listen::Accept::NoRoom => { - say!( - "netstack: refusing accept, {} piped connections already (max {})", - self.piped_live(), - self.max_piped_connections, - ); - msg.client.error(ERR_RESOURCE_EXHAUSTED); - return; - } - listen::Accept::Nothing => { - msg.client.error(ERR_NOT_CONNECTED); - return; - } - }; - - let accepted = socket_set.get_mut::(old_handle); - let remote = accepted.remote_endpoint().unwrap(); - let options = TcpOptions::new(!accepted.nagle_enabled()); - let remote_addr = match remote.addr { - IpAddress::Ipv4(a) => a.octets(), - }; - - let stream_id = self.alloc_id(); - self.sockets.insert(stream_id, SocketKind::TcpStream(old_handle)); - - self.piped_connections.push(piped_connection(stream_id, old_handle, pipes)); - - // Create replacement listener - let rx_buf = tcp::SocketBuffer::new(vec![0u8; TCP_SOCKET_BUFFER]); - let tx_buf = tcp::SocketBuffer::new(vec![0u8; TCP_SOCKET_BUFFER]); - let mut new_listener = tcp::Socket::new(rx_buf, tx_buf); - let listener = self - .piped_listeners - .get_mut(&req.socket_id) - .expect("looked up above; nothing between there and here removes a piped_listeners entry"); - listener.listening.open(&mut new_listener); - listener.handle = socket_set.add(new_listener); - self.sockets.insert(req.socket_id, SocketKind::TcpListener(listener.handle)); - - msg.client.result(&TcpAcceptPipedResponse { - socket_id: stream_id, - remote_addr, - remote_port: remote.port, - local_port, - options, - }); - } - - /// Bridge data between smoltcp sockets and kernel pipes for piped connections. - /// Drains both directions as far as the other side takes — when a pipe is - /// full, data stays in smoltcp's buffer and the TCP window shrinks. - fn bridge_piped(&mut self, socket_set: &mut SocketSet<'_>) { - use toyos_abi::syscall::SyscallError; - let mut closed = Vec::new(); - for i in 0..self.piped_connections.len() { - let conn = &mut self.piped_connections[i]; - let socket = socket_set.get_mut::(conn.handle); - - // smoltcp rx → the client's pipe. **Nothing leaves the socket that - // the pipe did not take**: a byte dequeued here has already been - // acknowledged to the peer, so one the pipe refused is cut out of - // the middle of the client's stream with nothing saying so. The - // rest waits in the socket, and the pipe's room is what wakes the - // pass that moves it. - conn.held = false; - if let Some(ref pipe) = conn.rx_write { - let mut refused = None; - while socket.can_recv() { - let moved = socket.recv(|queued| { - match toyos_abi::syscall::write_nonblock(pipe.as_handle(), queued) { - Ok(n) => (n, n), - Err(e) => { - refused = Some(e); - (0, 0) - } - } - }); - if !matches!(moved, Ok(n) if n > 0) { - break; - } - } - match refused { - None => {} - // Full: the client has not read yet. - Some(SyscallError::WouldBlock) => conn.held = true, - Some(SyscallError::Gone) => conn.close_rx(), - Some(e) => conn.refuse(socket, "receive", e), - } - } - - // pipe read → smoltcp tx. Ok(0) is the kernel's EOF — ring drained, - // no writer — which says the client stopped writing; not the - // forgeable closed flags. [`send_room`] and never `can_send` alone: - // a zero-length read answers `Ok(0)`, which the arm below reads as - // the client hanging up. - while send_room(socket) { - if let Some(ref pipe) = conn.tx_read { - // **No more is taken out of the pipe than the socket will - // take from us.** `send_slice` answers how many bytes it - // enqueued and takes fewer when the send buffer is short of - // room; bytes read past that are gone, and the peer's stream - // is short in the middle with nothing saying so. The pipe is - // where the rest belongs until there is room. - let mut buf = [0u8; 4096]; - let want = (socket.send_capacity() - socket.send_queue()).min(buf.len()); - match toyos_abi::syscall::read_nonblock(pipe.as_handle(), &mut buf[..want]) { - Ok(0) => { - socket.close(); - conn.close_tx(); - break; - } - Ok(n) => { - // Both refusals are bytes the pipe has already given - // up, so neither may be swallowed here of all places. - let sent = socket.send_slice(&buf[..n]).unwrap_or_else(|e| { - panic!("netstack: a socket that could send refused {n} byte(s): {e:?}") - }); - assert_eq!(sent, n, "netstack: the send buffer took {sent} of {n} byte(s) it had room for"); - } - Err(SyscallError::WouldBlock) => break, - Err(e) => { - conn.refuse(socket, "send", e); - break; - } - } - } else { - break; - } - } - - // Signal EOF to client when remote has closed and all data is drained - if !socket.may_recv() && !socket.can_recv() && conn.rx_write.is_some() { - conn.close_rx(); - } - - // **A connection that is over takes no more of the client's bytes.** - // A peer's reset leaves the socket `Closed` and `can_send` false for - // good, so the loop above never reads the pipe again; left open, the - // client's writes fill a pipe nobody drains and then block, and a - // writer that is never told its peer is gone cannot say so. - if !socket.is_open() && conn.tx_read.is_some() { - conn.close_tx(); - } - - if conn.clientless() { - let waited = conn.ownerless.get_or_insert_with(Instant::now).elapsed(); - match ownerless(socket, waited, conn.cut) { - Ownerless::Waits => {} - Ownerless::Cut => { - say!( - "netstack: resetting a connection — its client left {}s ago and its peer has not finished it", - waited.as_secs() - ); - (conn.ownerless, conn.cut) = (Some(Instant::now()), true); - } - Ownerless::Over => { - if conn.cut { - say!("netstack: the reset has left"); - } - closed.push(i); - } - Ownerless::Unsaid => { - say!( - "netstack: letting a connection go with its reset unsent — no next hop took it in {}s", - RESET_LIFE.as_secs() - ); - closed.push(i); - } - } - } - } - - for &i in closed.iter().rev() { - let conn = self.piped_connections.swap_remove(i); - socket_set.remove(conn.handle); - self.sockets.remove(&conn.socket_id); - } - } - - /// How long until the first connection with no client left reaches - /// [`OWNERLESS_LIFE`], or a cut one [`RESET_LIFE`], which nothing on the - /// wire wakes a pass for. - fn ownerless_wake_in(&self) -> Option { - self.piped_connections - .iter() - .filter_map(|c| Some((c.ownerless?, if c.cut { RESET_LIFE } else { OWNERLESS_LIFE }))) - .map(|(since, life)| life.saturating_sub(since.elapsed())) - .min() - } - - /// What the kernel answered a watch on a pipe of connection `socket_id`, - /// which may be gone since: the watch of a pipe closed with it ends too. - fn pipe_answered( - &mut self, - socket_set: &mut SocketSet<'_>, - socket_id: u32, - send: bool, - answer: Result, - ) { - if let Some(conn) = self.piped_connections.iter_mut().find(|c| c.socket_id == socket_id) { - conn.pipe_answered(socket_set.get_mut::(conn.handle), send, answer); - } - } - - /// Tell each piped listener's owner about a connection it can accept, and - /// close every listener whose notify pipe refuses netstack. - /// - /// **One write a pass, and any refusal ends the listener.** A listener - /// owed a wake writes its byte; the rest write zero bytes, which move - /// nothing and are still refused by name once the owner has gone. A wake - /// the pipe will not take is one the owner never gets, so its `accept` - /// would wait forever on a listener netstack still held; closing the listener - /// is what tells it instead — its notify pipe reads EOF. A full pipe is - /// that refusal too: it is an owner that has left a whole pipe of wakes - /// unread. - fn serve_piped_listeners(&mut self, socket_set: &mut SocketSet<'_>, room: bool) { - use toyos_abi::syscall::SyscallError; - let mut dead = Vec::new(); - for (&socket_id, listener) in &mut self.piped_listeners { - let wake = listener.listening.wake(socket_set.get_mut::(listener.handle), room); - match toyos_abi::syscall::write_nonblock(listener.notify_write.as_handle(), wake) { - Ok(_) => {} - Err(SyscallError::WouldBlock) if wake.is_empty() => {} - // Its owner has gone, which is the ordinary end of a listener. - Err(SyscallError::Gone) => dead.push(socket_id), - Err(e) => { - say!("netstack: closing listener {socket_id} — its notify pipe refused netstack: {e:?}"); - dead.push(socket_id); - } - } - } - for socket_id in dead { - if let Some(_listener) = self.piped_listeners.remove(&socket_id) { - if let Some(kind) = self.sockets.remove(&socket_id) { - if let SocketKind::TcpListener(handle) = kind { - socket_set.get_mut::(handle).abort(); - socket_set.remove(handle); - } - } - } - } - } - - /// Process pending async operations (UDP recvs, lookups, piped connects), - /// and say whether there is room for another piped connection after them. - fn process_pending(&mut self, socket_set: &mut SocketSet<'_>) -> bool { - let now = Instant::now(); - - for pr in std::mem::take(&mut self.pending_udp_recvs) { - if let Some(client) = self.deliver_datagram(pr.client, pr.socket_id, pr.max_len, socket_set) { - self.pending_udp_recvs.push(PendingUdpRecv { client, ..pr }); - } - } - - for (client, name, ended) in self.resolver.pass(socket_set, now) { - use resolve::Ended; - use toyos_dns::Failure; - match ended { - Ok(addrs) => answer_lookup(&client, &addrs), - // The protocol's one answer for a name with no address, - // whether the name or only its address is missing. - Err(Ended::Failed(Failure::NoSuchName | Failure::NoAddress)) => answer_lookup(&client, &[]), - Err(Ended::Failed(Failure::TimedOut)) => client.error(ERR_TIMED_OUT), - Err(Ended::Failed(Failure::Unreachable)) => { - unreachable!("netstack: no lookup here is told a query did not reach its server, and {name}'s ended so") - } - Err(Ended::Failed(why @ (Failure::Truncated | Failure::ServerFailed(_) | Failure::TooManyAliases))) => { - say!("netstack: a lookup of {name} ended without an answer: {why:?}"); - client.error(ERR_OTHER); - } - Err(Ended::NoPort) => { - say!("netstack: a lookup of {name} ended with every dynamic port bound, none left for its next query"); - client.error(ERR_RESOURCE_EXHAUSTED); - } - } - } - - // Pending piped connects - let mut i = 0; - while i < self.pending_piped_connects.len() { - let pc = &self.pending_piped_connects[i]; - let socket = socket_set.get_mut::(pc.handle); - if socket.may_send() { - let local_port = socket.local_endpoint().map(|e| e.port).unwrap_or(0); - let resp = TcpConnectResponse { - socket_id: pc.socket_id, - local_port, - _pad: 0, - }; - pc.client.result(&resp); - let pc = self.pending_piped_connects.swap_remove(i); - self.piped_connections.push(piped_connection(pc.socket_id, pc.handle, pc.pipes)); - continue; - } - if socket.state() == tcp::State::Closed { - pc.client.error(ERR_CONNECTION_REFUSED); - let (socket_id, handle) = (pc.socket_id, pc.handle); - self.sockets.remove(&socket_id); - socket_set.remove(handle); - self.pending_piped_connects.swap_remove(i); - continue; - } - if pc.deadline.is_some_and(|d| now >= d) { - pc.client.error(ERR_TIMED_OUT); - socket.abort(); - let (socket_id, handle) = (pc.socket_id, pc.handle); - self.sockets.remove(&socket_id); - socket_set.remove(handle); - self.pending_piped_connects.swap_remove(i); - continue; - } - i += 1; - } - self.piped_room() - } -} - -/// The most addresses one lookup's answer carries: what -/// `toyos::net::dns_lookup`'s 256-byte buffer holds, a count byte and five -/// bytes an address. A resolver may answer with a subset of a name's -/// addresses, and these are the ones the server put first. -const MAX_ANSWERED: usize = (256 - 1) / 5; - -/// A lookup's answer: a count, then each address behind the family tag 4. -fn answer_lookup(client: &Client, addrs: &[[u8; 4]]) { - let addrs = &addrs[..addrs.len().min(MAX_ANSWERED)]; - let mut answer = vec![addrs.len() as u8]; - for addr in addrs { - answer.push(4); - answer.extend_from_slice(addr); + snap.put("lease.held", true); + snap.put("lease.address", format!("{}/{}", held.address, held.prefix_len)); + snap.put("lease.server", held.server.to_string()); + snap.put("lease.dns", held.dns()); + let Some(router) = held.router else { return }; + snap.put("lease.router", router.to_string()); + let shard = node.shard(); + snap.put( + "neighbour.router", + match shard.ip().neighbour(shard.iface(), router) { + None => "none", + Some(Nud::Incomplete(_)) => "incomplete", + Some(Nud::Reachable(_)) => "reachable", + Some(Nud::Stale(_)) => "stale", + Some(Nud::Delay(_)) => "delay", + Some(Nud::Probe(_)) => "probe", + Some(Nud::Unreachable(_)) => "unreachable", + Some(Nud::Failed) => "failed", + }, + ); } - client.result_bytes(&answer); } /// Answer `inspect` with what this pass knows, in one non-blocking write, and /// let the connection close as every other answer does. -/// -/// Here and not in [`Netstack::handle_message`] because the card and the -/// lease are the loop's and not the socket table's. -fn answer_inspect(request: &Request, daemon: &Netstack, card: &Card, dhcp: &dhcp::Dhcp, socket_set: &SocketSet<'_>) { +fn answer_inspect(request: &Request, card: &Card, leases: &Leases, sockets: &serve::Sockets, node: &Node) { // The request is a bare header, and anything riding on one is not this // protocol. if request.payload_len != 0 { - request.client.error(ERR_INVALID_INPUT); + request.client.error(toyos::net::ERR_INVALID_INPUT); return; } let mut snap = Snapshot::new(toyos_inspect::NET); card.inspect(&mut snap); - dhcp.inspect(&mut snap); - daemon.inspect(&mut snap, socket_set); + leases.inspect(node, &mut snap); + sockets.inspect(node, &mut snap); let encoded = snap.encode().unwrap_or_else(|why| panic!("netstack: its snapshot: {why}")); request.client.snapshot(&encoded); } @@ -1404,15 +299,10 @@ const _: () = assert!( ); fn main() { - // **The order this used to have was load-bearing and is now moot.** The - // device was claimed before the name was published, because a client that - // connected while netstack was still in `DmaNic::open` reached a listener owned - // by a process about to return and got its request answered by nobody — - // sshserver found it, took its `panic!` arm and put a tokio backtrace across the - // boot. There is no window left to order around: the `netstack` port exists - // before either process does, a client's connection is queued on it whether - // or not this program ever reaches `accept`, and if netstack exits the queued - // client sees `Gone` rather than silence. + // The `netstack` port exists before this process does: a client's + // connection is queued on it whether or not this program ever reaches + // `accept`, and if netstack exits the queued client sees `Gone` rather + // than silence. let Some((open, claim)) = CARDS .iter() .find_map(|(id, open)| endow::pci_function::(*id).map(|c| (*open, c))) @@ -1422,197 +312,135 @@ fn main() { }; let acceptor = endow::acceptor("netstack") .expect("the manifest declares this program serves `netstack`"); - let nic = open(claim); - // The link as the card came up with it, which the first change a pass - // reports is measured against. Virtio reports no link changes at all. - let mut link_up = match &nic { - Card::Intel(intel) => intel.link().is_up(), - Card::Virtio(_) => true, - }; - let mac = nic.mac(); - let mut device = DmaNic { nic }; - + let card = open(claim); + let mac = card.mac(); say!( "netstack: MAC {:02x}:{:02x}:{:02x}:{:02x}:{:02x}:{:02x}", mac[0], mac[1], mac[2], mac[3], mac[4], mac[5] ); - let mut config = Config::new(HardwareAddress::Ethernet(EthernetAddress(mac))); - config.random_seed = smoltcp_seed(); - let epoch = Instant::now(); - let now = SmoltcpInstant::from_millis(0); - let mut iface = Interface::new(config, &mut device, now); - let mut socket_set = SocketSet::new(vec![]); + let began = Wall::now(); + let clock = || Instant::from_nanos(u64::try_from(began.elapsed().as_nanos()).unwrap_or(u64::MAX)); + let config = Config { + mac: IndividualMac::new(MacAddr(mac)).unwrap_or_else(|| panic!("netstack: the card's address is a group's")), + receive_buffer: TCP_BUFFER, + send_buffer: TCP_BUFFER, + secrets: secrets(), + }; + let host = HostName::new(HOSTNAME).unwrap_or_else(|| panic!("netstack: {HOSTNAME:?} is no host name")); + let mut node = Node::new(clock(), config, Some(host), draw) + .unwrap_or_else(|why| panic!("netstack: the stack refused its configuration: {why:?}")); - let dhcp_handle = socket_set.add(dhcp::socket()); - let mut dhcp = dhcp::Dhcp::new(); - if let Card::Intel(nic) = &device.nic { + if let Card::Intel(nic) = &card { nic.accept_multicast(toyos_mdns::GROUP_MAC); } - let mut mdns = mdns::Responder::new(dhcp::HOSTNAME, &mut iface, &mut socket_set); + let name = toyos_mdns::Host::new(HOSTNAME).unwrap_or_else(|_| panic!("netstack: {HOSTNAME:?} is no host name")); + node.answer_as(clock(), name, draw) + .unwrap_or_else(|why| panic!("netstack: a new node refused the multicast DNS port: {why:?}")); let total_mem = total_memory(); - let max_piped = max_piped_connections(total_mem); - let mut daemon = Netstack::new(max_piped); + let places = places_for(total_mem); + node.set_places(clock(), places); + let mut sockets = serve::Sockets::new(draw(), places); + let mut leases = Leases { began, said: None, settled: false }; + + // The link as the card came up with it, which the first change a pass + // reports is measured against. Virtio reports no link changes at all. The + // node begins with its link down. + let mut link_up = match &card { + Card::Intel(intel) => intel.link().is_up(), + Card::Virtio(_) => true, + }; + if link_up { + node.link(clock(), true, draw); + } - // Sized for the slot ceiling rather than for `max_piped`: the batch - // between two `wait` calls is the two fixed registrations, at most two per live - // piped connection, one per pending connection and one per lookup, and the - // ceiling is what that can never exceed. let poller = Poller::new( - FIXED_POLL_HANDLES - + POLL_HANDLES_PER_PIPED * MAX_PIPED_SLOTS as u32 - + MAX_PENDING_CONNS - + LOOKUP_POLL_HANDLES, + FIXED_WATCHES + serve::WATCHES_PER_PLACE * MAX_PLACES + MAX_PENDING_CONNS + serve::LOOKUP_WATCHES, ); - const TOKEN_LISTENER: u64 = 0; + const TOKEN_ACCEPTOR: u64 = 0; const TOKEN_NIC: u64 = 1; - // A piped connection's two pipes, by its socket id in the low word: an - // answer names the connection it was asked of and no place in a list, - // which a connection let go since would hand to another. - const TOKEN_SEND_PIPE: u64 = 1 << 32; - const TOKEN_RECEIVE_PIPE: u64 = 2 << 32; - // Clear of a connection's own handle by more than `MAX_HANDLES` (4096, - // `kernel/src/object/handle.rs`). + // Clear of the two above and of a connection's own handle by more than + // `MAX_HANDLES` (4096, `kernel/src/object/handle.rs`); `serve`'s tokens + // are all above the low word. const TOKEN_PENDING_BASE: u64 = 0x1_0000; - // Clear of the pending range by the same margin. - const TOKEN_LOOKUP_BASE: u64 = 0x2_0000; let mut pending: Vec = Vec::new(); + // Accepts the kernel refused since the last it did not. + let mut accept_refused: u64 = 0; loop { - // Before `iface.poll`, because it is what makes the interrupt taken and - // what gives a driver with a per-pass receive budget that budget back. - if let Some(link) = device.nic.begin_pass() { - // Down to up only, and only with no lease held: a speed change is - // no new network, and a bound lease is kept across a flap rather - // than given up — `dhcp::restart`'s own header. Before the poll - // below, so the DISCOVER goes out on this pass. - if link.is_up() && !link_up && !dhcp.leased() { - dhcp::restart(socket_set.get_mut::(dhcp_handle)); + // First, because it is what makes the interrupt taken and what gives + // a driver with a per-pass receive budget that budget back. + if let Some(link) = card.begin_pass() { + // A change of state only: a speed change is no new network. + if link.is_up() != link_up { + link_up = link.is_up(); + node.link(clock(), link_up, draw); + } + } + while card.rx(|frame| node.receive(clock(), frame, draw)) {} + let now = clock(); + if node.next_deadline().is_some_and(|at| at <= now) { + node.fire(now, draw); + } + loop { + // A card with no room is asked to say when it has some, and a + // frame the node still holds leaves in the pass that wakes. + let room = match card.tx_room() { + 0 => card.wake_on_room(), + room => room, + }; + if room == 0 || node.transmit(now, room, |frame| card.tx(frame.len(), |slot| slot.copy_from_slice(frame)), draw) < room { + break; } - link_up = link.is_up(); } - let now = SmoltcpInstant::from_millis(epoch.elapsed().as_millis() as i64); - while iface.poll(now, &mut device, &mut socket_set) != PollResult::None {} - device.nic.report(); + card.report(); - // **After the poll and before anything is served.** The lease is what - // gives this machine an address, a route and its resolvers, so a client - // answered before it was applied would be answered on a machine that is - // on no network. - let change = dhcp::Change::of(socket_set.get_mut::(dhcp_handle)); - if dhcp.pass(change, &mut iface, &mut daemon.resolver) { + sockets.settle(&mut node, now); + // Before anything is served: the lease is what gives this machine an + // address, a route and its resolvers. + if leases.pass(node.lease()) { say!( - "netstack: ready, at most {max_piped} piped connections \ - ({} MiB each of {} MiB total)", - PIPED_CONNECTION_BYTES / (1024 * 1024), + "netstack: ready, at most {places} places ({} MiB each of {} MiB total)", + PLACE_BYTES / (1024 * 1024), total_mem / (1024 * 1024), ); } - mdns.pass(&iface, &mut socket_set, link_up, Instant::now()); - - daemon.bridge_piped(&mut socket_set); - - let room = daemon.process_pending(&mut socket_set); - daemon.serve_piped_listeners(&mut socket_set, room); - - // smoltcp's own next deadline — a retransmit, a persist probe, a - // delayed ACK — and zero when it has a frame to send now. A piped - // connection needs nothing else: its peer's bytes wake the NIC, and its - // client's bytes and room wake the watches below. - // - // **None of them while the card has no room.** Each is a frame to send - // or to take, the card refuses both, and smoltcp's "now" would be a - // pass every time round until it stops; the card's claim begins the - // pass that can. - let smoltcp_due = if device.room() { - iface.poll_delay(now, &socket_set).map_or(u64::MAX, |d| d.total_micros().saturating_mul(1000)) - } else { - u64::MAX - }; - - // A pending UDP receive or connect has no wake of its own. - let has_pending_async = !daemon.pending_udp_recvs.is_empty() - || !daemon.pending_piped_connects.is_empty(); - let timeout = if has_pending_async { - smoltcp_due.min(Duration::from_millis(1).as_nanos() as u64) - } else { - smoltcp_due - }; - - poller.watch(&acceptor, READABLE, TOKEN_LISTENER); - poller.watch(device.nic.claim(), READABLE, TOKEN_NIC); - - // The client's bytes to send, room in a receive pipe that is holding - // the peer's back, and the client letting go of either end: each is a - // pass's worth of work. - for conn in daemon.piped_connections.iter() { - // Readable only while the socket can take the bytes: a pipe - // holding some is readable until read, so its watch would complete - // on every pass while the peer's window is shut. The ACK that makes - // room wakes the NIC. Its writer's leaving is asked until answered, - // for the same reason: it stays so. - let room = send_room(socket_set.get::(conn.handle)); - let send = if room { READABLE } else { 0 } | if conn.writer_gone { 0 } else { OTHER_END_GONE }; - if let (true, Some(pipe)) = (send != 0, &conn.tx_read) { - poller.watch(pipe, send, TOKEN_SEND_PIPE | u64::from(conn.socket_id)); - } - if let Some(pipe) = &conn.rx_write { - let room = if conn.held { WRITABLE } else { 0 }; - poller.watch(pipe, room | OTHER_END_GONE, TOKEN_RECEIVE_PIPE | u64::from(conn.socket_id)); - } - } - + poller.watch(&acceptor, READABLE, TOKEN_ACCEPTOR); + poller.watch(card.claim(), READABLE, TOKEN_NIC); + sockets.watch(&node, &poller); for p in pending.iter() { poller.watch(&p.conn, READABLE, TOKEN_PENDING_BASE + p.conn.as_handle().0 as u64); } - // A client waiting on a lookup hangs up by closing its connection, - // which makes it readable. - for client in daemon.resolver.clients() { - poller.watch(&client.conn, READABLE, TOKEN_LOOKUP_BASE + client.conn.as_handle().0 as u64); + // The node's next deadline: a retransmission, a lease's timer, a + // lookup's wait, a connect's. Zero when one is due. + let nanos = |left: Duration| u64::try_from(left.as_nanos()).unwrap_or(u64::MAX); + let mut timeout = node.next_deadline().map_or(u64::MAX, |at| nanos(at.since(clock()))); + if let Some(left) = leases.due_in() { + timeout = timeout.min(nanos(left)); } - - let timeout = match mdns.wake_in(Instant::now()) { - Some(left) => timeout.min(left.as_nanos() as u64), - None => timeout, - }; - // A lookup waiting on its answer is woken when its wait is over, to - // ask the next server. - let timeout = match daemon.resolver.wake_in(Instant::now()) { - Some(left) => timeout.min(left.as_nanos() as u64), - None => timeout, - }; - let timeout = match daemon.ownerless_wake_in() { - Some(left) => timeout.min(left.as_nanos() as u64), - None => timeout, - }; // A card that never does what it owes sends no interrupt to say so. - let timeout = timeout.min(device.nic.pass_due_in().unwrap_or(u64::MAX)); + timeout = timeout.min(card.pass_due_in().unwrap_or(u64::MAX)); // A client that connects and then says nothing wakes nothing, so the - // deadline that removes it has to be a wake in its own right: without - // this netstack can sit in `wait` forever with `pending` full of clients - // whose handshake is already over its time. - let timeout = if pending.is_empty() { - timeout - } else { - timeout.min(HANDSHAKE_TIMEOUT.as_nanos() as u64) - }; + // deadline that removes it is a wake in its own right. + if !pending.is_empty() { + timeout = timeout.min(nanos(HANDSHAKE_TIMEOUT)); + } let mut ready: Vec = Vec::new(); - poller.wait_answers(1, timeout, |token, answer| match token & !u64::from(u32::MAX) { - TOKEN_SEND_PIPE => daemon.pipe_answered(&mut socket_set, token as u32, true, answer), - TOKEN_RECEIVE_PIPE => daemon.pipe_answered(&mut socket_set, token as u32, false, answer), - _ => ready.push(token), + poller.wait_answers(1, timeout, |token, answer| { + if !sockets.answered(&mut node, clock(), token, answer) { + ready.push(token); + } }); + sockets.bridge(&mut node, clock()); - // A handshake that never completes is why this deadline exists, and the - // sweep has to happen on a pass that found nothing ready too — - // otherwise a silent client is only ever timed out by some *other* - // client's traffic. - let now_wall = Instant::now(); + // On a pass that found nothing ready too: otherwise a silent client + // is only ever timed out by some other client's traffic. + let now_wall = Wall::now(); for p in pending.iter().filter(|p| now_wall.duration_since(p.since) >= HANDSHAKE_TIMEOUT) { say!( "netstack: dropping client {} — it never finished its request", @@ -1621,16 +449,30 @@ fn main() { } pending.retain(|p| now_wall.duration_since(p.since) < HANDSHAKE_TIMEOUT); - // A lookup whose client has left ends now, not when its servers are - // done with it: its sockets and its slot are another client's. - let spoke = |c: &Client| ready.contains(&(TOKEN_LOOKUP_BASE + c.conn.as_handle().0 as u64)); - daemon.resolver.let_go(&mut socket_set, |c| spoke(c) && c.gone()); - // Accept and the request are two events. Nothing is read here: a client // that connects and then says nothing costs a slot and a deadline, not // the network stack. - if ready.contains(&TOKEN_LISTENER) { - let conn = acceptor.accept().expect("accept failed"); + // A connection the kernel would not hand over is gone from its + // queue, and its client told: the first of a run is named, and the + // next accept says how many followed it. + let accepted = match ready.contains(&TOKEN_ACCEPTOR).then(|| acceptor.accept()) { + Some(Err(why)) => { + if accept_refused == 0 { + say!("netstack: the kernel refused a client's connection: {why:?}"); + } + accept_refused = accept_refused.saturating_add(1); + None + } + Some(Ok(conn)) => { + if accept_refused > 1 { + say!("netstack: accepting again, after {accept_refused} connections the kernel refused"); + } + accept_refused = 0; + Some(conn) + } + None => None, + }; + if let Some(conn) = accepted { if pending.len() >= MAX_PENDING_CONNS as usize { say!( "netstack: refusing client {} — {MAX_PENDING_CONNS} connections are already \ @@ -1638,7 +480,7 @@ fn main() { conn.as_handle().0 ); } else { - pending.push(PendingConn { conn, rx: ClientRx::new(), since: Instant::now() }); + pending.push(PendingConn { conn, rx: ClientRx::new(), since: Wall::now() }); } } @@ -1690,10 +532,10 @@ fn main() { for request in requests { if request.msg_type == toyos_inspect::MSG_INSPECT { - answer_inspect(&request, &daemon, &device.nic, &dhcp, &socket_set); + answer_inspect(&request, &card, &leases, &sockets, &node); continue; } - daemon.handle_message(request, &mut socket_set, &mut iface); + sockets.request(&mut node, clock(), request, draw); } } } diff --git a/userland/netstack/src/mdns.rs b/userland/netstack/src/mdns.rs deleted file mode 100644 index 9c0d66a77a..0000000000 --- a/userland/netstack/src/mdns.rs +++ /dev/null @@ -1,157 +0,0 @@ -//! This machine's name on its network: `.local` answers with the -//! address the lease gave it (`toyos_mdns`, RFC 6762), so another machine on -//! the network reaches it by name with nothing configured on either — the -//! development host finds the T14's served log this way. -//! -//! **Answered only while an address is held on a link that is up, and the -//! name is claimed**: every address after none and every return of the link -//! is probed on first (§8, §8.1), which takes the name's first answer three -//! quarters of a second and a drawn delay past it, and a name another host -//! answers for is not this machine's. Every decision is -//! `toyos_mdns::Responder`'s; this is the socket, the clock, the draw of the -//! delay the responder asks for, and the log line for what became of the -//! name. A pass tells the responder its link, hands it every message that -//! has arrived, and only then asks what the name is owed: a conflicting -//! response received as a probing ends takes the name before it is claimed. -//! What the name is owed later — a probe, an announcement (§8.3), or an -//! answer §6 held back — is a wake of netstack's own loop -//! ([`Responder::wake_in`]) rather than a sleep, because the protocol names -//! the interval and nothing on the wire says when it has passed. - -use std::net::Ipv4Addr; -use std::time::{Duration, Instant}; - -use smoltcp::iface::{Interface, SocketHandle, SocketSet}; -use smoltcp::socket::udp; -use smoltcp::wire::{IpAddress, IpCidr, IpEndpoint}; -use toyos_mdns::{Event, Host, Link, Source, To, GROUP, PORT, RETRY_MS}; - -/// A message is a few hundred bytes; this holds a handful of them between two -/// passes, and one past it is dropped by the socket, which is what its -/// sender's own retry is for. -const BUFFER: usize = 4096; - -pub struct Responder { - handle: SocketHandle, - host: &'static str, - record: toyos_mdns::Responder<'static>, - /// The origin of the responder's clock. - born: Instant, -} - -impl Responder { - /// Join the group and bind its port. `host` is the name this machine asks - /// its network to record for it (`dhcp::HOSTNAME`). - pub fn new(host: &'static str, iface: &mut Interface, socket_set: &mut SocketSet<'static>) -> Self { - let label = Host::new(host).unwrap_or_else(|_| panic!("netstack: {host:?} is no host name")); - iface - .join_multicast_group(IpAddress::Ipv4(Ipv4Addr::from(GROUP))) - .expect("netstack: the multicast DNS group is the one group this interface joins"); - let buffer = || { - udp::PacketBuffer::new(vec![udp::PacketMetadata::EMPTY; 8], vec![0u8; BUFFER]) - }; - let mut socket = udp::Socket::new(buffer(), buffer()); - socket.bind(PORT).expect("netstack: nothing else binds the multicast DNS port"); - Self { handle: socket_set.add(socket), host, record: toyos_mdns::Responder::new(label), born: Instant::now() } - } - - /// After each poll: tell the responder its link, which is the interface's - /// address while the card's link is up; hand it every message that - /// arrived and send what each is answered; then send what the name is - /// owed now. What became of the name is said as it happens. - pub fn pass(&mut self, iface: &Interface, socket_set: &mut SocketSet<'_>, link_up: bool, now: Instant) { - let socket = socket_set.get_mut::(self.handle); - // IPv4 is the one protocol this netstack is built with, so every address is one. - let link = iface.ip_addrs().first().filter(|_| link_up).map(|&IpCidr::Ipv4(cidr)| Link { - addr: cidr.address().octets(), - prefix: cidr.prefix_len(), - }); - let now_ms = self.ms(now); - let group = IpEndpoint::new(IpAddress::Ipv4(Ipv4Addr::from(GROUP)), PORT); - self.record.on(link, now_ms); - while let Ok((message, meta)) = socket.recv() { - let IpAddress::Ipv4(from) = meta.endpoint.addr; - let from = Source { addr: from.octets(), port: meta.endpoint.port }; - let (answer, event) = self.record.heard(message, from, now_ms); - said(self.host, event); - let Some(answer) = answer else { - continue; - }; - let to = match answer.to { - To::Group => group, - To::Asker => meta.endpoint, - }; - send(socket, &answer.bytes, to); - } - let (owed, event) = self.record.owed(now_ms, || u32::from(crate::resolve::random_u16())); - said(self.host, event); - if let Some(owed) = owed { - send(socket, &owed, group); - } - } - - /// When the loop must wake for what the name is owed, if anything is. - pub fn wake_in(&self, now: Instant) -> Option { - self.record.owed_at().map(|at| Duration::from_millis(at.saturating_sub(self.ms(now)))) - } - - fn ms(&self, now: Instant) -> u64 { - now.saturating_duration_since(self.born).as_millis() as u64 - } -} - -/// The log line for what became of `host`'s name. -fn said(host: &str, event: Option) { - match event { - Some(Event::Claimed) => crate::say!("netstack: mDNS: no host answered for {host}.local; this machine answers as it"), - Some(Event::Lost) => crate::say!( - "netstack: mDNS: another host answered for {host}.local; this machine answers to no name and asks for {host}.local again every {} s", - RETRY_MS / 1000 - ), - None => {} - } -} - -fn send(socket: &mut udp::Socket, bytes: &[u8], to: IpEndpoint) { - match socket.send_slice(bytes, to) { - Ok(()) => {} - // A burst of queries this pass cannot answer; the asker retries, and - // nothing here waits. - Err(udp::SendError::BufferFull) => {} - // Only an asker's own source can be this, an address or a port of - // zero, and nothing on the wire reaches it. - Err(udp::SendError::Unaddressable) => {} - } -} - -#[cfg(test)] -mod tests { - use super::*; - use smoltcp::iface::SocketSet; - - const LINK: Link = Link { addr: [10, 0, 2, 15], prefix: 24 }; - - /// A `Responder` over a socket taken from a set of its own — `wake_in` - /// touches neither, only `record` and `born`. - fn responder() -> Responder { - let buffer = || udp::PacketBuffer::new(vec![udp::PacketMetadata::EMPTY], vec![0u8; 64]); - let mut set = SocketSet::new(Vec::new()); - let handle = set.add(udp::Socket::new(buffer(), buffer())); - Responder { handle, host: "t14", record: toyos_mdns::Responder::new(Host::new("t14").unwrap()), born: Instant::now() } - } - - /// **A wake is asked for exactly what the name owes.** Nothing before an - /// address is held; the first probe's own instant, the drawn delay after - /// the address, once the record schedules it. A responder that never asks - /// for this wake sends that probe, and every one after it, only on some - /// other, unrelated wake. - #[test] - fn wake_in_asks_for_what_the_name_owes_and_nothing_else() { - let mut r = responder(); - assert_eq!(r.wake_in(r.born), None, "nothing is owed before an address is held"); - r.record.on(Some(LINK), 0); - assert_eq!(r.record.owed(0, || 100), (None, None), "§8.1: the delay before the first probe"); - assert_eq!(r.wake_in(r.born), Some(Duration::from_millis(100))); - assert_eq!(r.wake_in(r.born + Duration::from_millis(40)), Some(Duration::from_millis(60))); - } -} diff --git a/userland/netstack/src/pipes.rs b/userland/netstack/src/pipes.rs new file mode 100644 index 0000000000..5aa1b4cc44 --- /dev/null +++ b/userland/netstack/src/pipes.rs @@ -0,0 +1,72 @@ +//! The kernel's pipe ends as the node's [`ToClient`], [`FromClient`] and +//! [`Wake`]: one non-blocking call each, and the kernel's refusal in the +//! node's word for it. +//! +//! **The node owns an end and netstack only watches it.** [`hold`] splits a +//! pipe end into the [`Held`] the node is handed and the [`Watched`] netstack +//! keeps: the handle closes when the node drops its half, which is how a +//! client reads the end of its stream or finds its writes refused, and from +//! then the watched half names nothing. +//! +//! **Untrusted input.** An end is whatever handle its client moved, and +//! nothing checks its kind at intake: one that is no pipe end, or the wrong +//! end, answers a refusal here that is neither a full pipe nor a vanished +//! reader, and the node ends that client's stream or listener for it. + +use std::rc::{Rc, Weak}; + +use toyos::Pipe; +use toyos_abi::syscall::SyscallError; +use toyos_net_node::{FromClient, ReadRefusal, ToClient, Wake, WriteRefusal}; + +/// The half the node holds: the handle lives as long as this does. +pub struct Held(Rc); + +/// The half netstack watches by. +pub struct Watched(Weak); + +pub fn hold(pipe: Pipe) -> (Held, Watched) { + let pipe = Rc::new(pipe); + let watched = Watched(Rc::downgrade(&pipe)); + (Held(pipe), watched) +} + +impl Watched { + /// The end, while the node holds it. + pub fn held(&self) -> Option> { + self.0.upgrade() + } +} + +fn write_refused(why: SyscallError) -> WriteRefusal { + match why { + SyscallError::WouldBlock => WriteRefusal::Full, + SyscallError::Gone => WriteRefusal::Gone, + _ => WriteRefusal::Broken, + } +} + +impl ToClient for Held { + fn write(&mut self, bytes: &[u8]) -> Result { + self.0.write_nonblock(bytes).map_err(write_refused) + } +} + +impl FromClient for Held { + fn read(&mut self, out: &mut [u8]) -> Result { + self.0.read_nonblock(out).map_err(|why| match why { + SyscallError::WouldBlock => ReadRefusal::Empty, + _ => ReadRefusal::Broken, + }) + } +} + +impl Wake for Held { + fn wake(&mut self) -> Result<(), WriteRefusal> { + match self.0.write_nonblock(&[1]) { + Ok(1) => Ok(()), + Ok(_) => Err(WriteRefusal::Full), + Err(why) => Err(write_refused(why)), + } + } +} diff --git a/userland/netstack/src/resolve.rs b/userland/netstack/src/resolve.rs deleted file mode 100644 index 526316c878..0000000000 --- a/userland/netstack/src/resolve.rs +++ /dev/null @@ -1,323 +0,0 @@ -//! The machine's one resolver: a name's IPv4 addresses, asked of the servers -//! the lease named. std's `lookup_host` and libc's `getaddrinfo` both reach it -//! through `toyos::net::dns_lookup`. Every decision is `toyos_dns`'s. This -//! module owns the sockets, the clock, the query IDs and the clients waiting -//! for answers. -//! -//! **Each query leaves from a socket of its own**, bound to a port drawn from -//! the kernel's random source, and carries an ID drawn from the same source -//! (RFC 5452 §9.2). An off-path sender has to guess both to be read at all. -//! A socket of its own is also what keeps one query from waiting on another: -//! smoltcp keeps a datagram at the head of its socket's queue while its -//! server's link address is unresolved or no route leads to it, and a query -//! queued behind it on the same socket would never leave. -//! -//! **A lookup's wait is a wake of netstack's loop** ([`Resolver::wake_in`]), and a -//! reply is a frame, which the NIC's interrupt already wakes it for. - -use std::net::Ipv4Addr; -use std::time::{Duration, Instant}; - -use smoltcp::iface::{SocketHandle, SocketSet}; -use smoltcp::socket::udp; -use smoltcp::wire::{IpAddress, IpEndpoint, IpListenEndpoint, Ipv4Address, DHCP_MAX_DNS_SERVER_COUNT}; -use toyos_dns::{Asked, Failure, Lookup, Name, Step}; -pub use toyos_dns::MAX_LOOKUPS; - -/// Datagrams, and bytes, a query's socket holds between two passes: two of the -/// largest an unfragmented Ethernet frame carries. A reply to a query sent -/// without EDNS is at most 512 bytes (RFC 1035 §4.2.1), and a larger one still -/// reaches the reader, which judges it. -const RECEIVE_PACKETS: usize = 2; -const RECEIVE_BUFFER: usize = RECEIVE_PACKETS * (1500 - 20 - 8); - -/// A query is at most a header, a 255-byte name and four bytes, and its socket -/// sends nothing else. -const QUERY_BYTES: usize = 12 + 255 + 4; - -/// The ports a query's socket is bound in: IANA's dynamic range (RFC 6335 -/// §6). -const EPHEMERAL: std::ops::RangeInclusive = 49152..=65535; - -/// Why a lookup was not started. -#[derive(Debug)] -pub enum Refused { - /// The lease named no server, or there is no lease. - NoServer, - /// [`MAX_LOOKUPS`] are in flight, or every port in [`EPHEMERAL`] is bound. - Full, -} - -impl Refused { - /// What the client is answered. - pub fn code(&self) -> u32 { - match self { - // This machine is on no network that answers names, which clears - // when a lease lands. - Self::NoServer => toyos::net::ERR_NOT_CONNECTED, - Self::Full => toyos::net::ERR_RESOURCE_EXHAUSTED, - } - } -} - -/// How a lookup ended without an address. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum Ended { - /// What the servers' answers, or their silence, came to. - Failed(Failure), - /// Every port in [`EPHEMERAL`] is bound, so the next query had none to - /// leave from. - NoPort, -} - -struct Pending { - client: C, - name: Name, - lookup: Lookup, - /// The socket each query waiting for its answer left from, and the newest - /// query's, which may not have left yet. - queries: Vec<(Asked, SocketHandle)>, -} - -/// The lookups in flight, and the servers a new one asks. -pub struct Resolver { - servers: Vec<[u8; 4]>, - lookups: Vec>, - /// The origin of every lookup's clock. - born: Instant, - /// Every query's ID and port. - draw: D, -} - -/// A value the kernel's random source drew. netstack ends by name if the source -/// refuses: an ID or a port anyone can predict is a forged answer's way in. -pub fn random_u16() -> u16 { - let mut bytes = [0u8; 2]; - toyos_abi::syscall::random(&mut bytes) - .unwrap_or_else(|e| panic!("netstack: the kernel's random source refused a query's ID or port: {e:?}")); - u16::from_le_bytes(bytes) -} - -/// Whether a UDP socket in `socket_set` is bound to `port`, whatever its -/// address. smoltcp hands a datagram to the first socket that accepts it, so a -/// second socket on a port would receive nothing its first did not refuse. -pub fn udp_port_taken(socket_set: &SocketSet<'_>, port: u16) -> bool { - socket_set.iter().any(|(_, socket)| match socket { - smoltcp::socket::Socket::Udp(s) => s.endpoint().port == port, - _ => false, - }) -} - -impl u16> Resolver { - /// A resolver whose clock starts at `born`, drawing every query's ID and - /// port from `draw`. - pub fn new(born: Instant, draw: D) -> Self { - Self { servers: Vec::new(), lookups: Vec::new(), born, draw } - } - - /// The servers the lease named, or none. A lookup already in flight keeps - /// the servers it started with. - /// - /// **An address no query can be sent to is not kept.** The lease is the - /// network's word: an unspecified, broadcast or multicast server would have - /// every query to it refused by the socket, so it is named and left out. - pub fn set_servers(&mut self, servers: &[Ipv4Address]) { - assert!( - servers.len() <= DHCP_MAX_DNS_SERVER_COUNT, - "netstack: a lease carries at most {DHCP_MAX_DNS_SERVER_COUNT} resolvers, and this one {}", - servers.len() - ); - self.servers.clear(); - for server in servers { - if server.is_unspecified() || server.is_broadcast() || server.is_multicast() { - crate::say!("netstack: the lease names {server} as a resolver, which no query can reach; not asking it"); - continue; - } - self.servers.push(server.octets()); - } - } - - /// Start looking up `name` for `client`. - pub fn start(&mut self, client: C, name: Name, socket_set: &mut SocketSet<'_>, now: Instant) -> Result<(), (C, Refused)> { - if self.servers.is_empty() { - return Err((client, Refused::NoServer)); - } - if self.lookups.len() == MAX_LOOKUPS { - return Err((client, Refused::Full)); - } - let (lookup, step) = Lookup::start(name.clone(), &self.servers, self.ms(now), &mut self.draw) - .expect("the server list was checked not empty"); - let mut pending = Pending { client, name, lookup, queries: Vec::new() }; - match act(&mut pending, step, socket_set, &mut self.draw) { - None => { - self.lookups.push(pending); - Ok(()) - } - Some(Err(Ended::NoPort)) => Err((pending.client, Refused::Full)), - Some(other) => panic!("netstack: a lookup's first step is a query, not {other:?}"), - } - } - - /// Read every reply that arrived and end every wait that is over; answer - /// each lookup that ended with its client, the name it asked, and how it - /// ended. - pub fn pass(&mut self, socket_set: &mut SocketSet<'_>, now: Instant) -> Vec<(C, Name, Result, Ended>)> { - let now_ms = self.ms(now); - let mut ended = Vec::new(); - let mut i = 0; - while i < self.lookups.len() { - let pending = &mut self.lookups[i]; - let mut done = None; - let mut q = 0; - while done.is_none() && q < pending.queries.len() { - let (asked, handle) = pending.queries[q]; - let socket = socket_set.get_mut::(handle); - let (reply, from, port) = match socket.recv() { - Ok((reply, meta)) => { - let IpAddress::Ipv4(from) = meta.endpoint.addr; - (reply.to_vec(), from.octets(), meta.endpoint.port) - } - Err(udp::RecvError::Exhausted) => { - q += 1; - continue; - } - Err(udp::RecvError::Truncated) => { - unreachable!("netstack: recv hands back the whole datagram and truncates nothing") - } - }; - let step = pending.lookup.on_datagram(asked, from, port, &reply, now_ms, &mut self.draw); - done = act(pending, step, socket_set, &mut self.draw); - // The step may have let this query go: its socket is read - // again from wherever it now is, or every socket from the - // first. - q = pending.queries.iter().position(|&(a, _)| a == asked).unwrap_or(0); - } - if done.is_none() { - let step = pending.lookup.on_time(now_ms, &mut self.draw); - done = act(pending, step, socket_set, &mut self.draw); - } - match done { - None => i += 1, - Some(result) => { - let pending = self.lookups.swap_remove(i); - close(&pending, socket_set); - ended.push((pending.client, pending.name, result)); - } - } - } - ended - } - - /// End every lookup whose client `gone` says has left, at once: nobody is - /// waiting for its answer, and its sockets and its slot are another's. - pub fn let_go(&mut self, socket_set: &mut SocketSet<'_>, mut gone: impl FnMut(&C) -> bool) { - let mut i = 0; - while i < self.lookups.len() { - if gone(&self.lookups[i].client) { - close(&self.lookups.swap_remove(i), socket_set); - } else { - i += 1; - } - } - } - - /// The clients waiting for an answer. - pub fn clients(&self) -> impl Iterator { - self.lookups.iter().map(|p| &p.client) - } - - /// How many of the stack's sockets are the resolver's. - pub fn sockets(&self) -> usize { - self.lookups.iter().map(|p| p.queries.len()).sum() - } - - /// How long until the soonest lookup's wait is over, if one is in flight. - pub fn wake_in(&self, now: Instant) -> Option { - let now_ms = self.ms(now); - self.lookups.iter().map(|p| Duration::from_millis(p.lookup.due().saturating_sub(now_ms))).min() - } - - fn ms(&self, now: Instant) -> u64 { - now.saturating_duration_since(self.born).as_millis() as u64 - } -} - -/// Carry out `step` for `pending`, let go of the socket of every query no -/// longer answered, and answer the lookup's end where it ended. -/// -/// **A query whose wait ended before it left is let go with its socket.** -/// Nothing can answer it, and smoltcp would go on asking for its server's link -/// address once a second, which is its one discovery a second for every -/// neighbour: the next server's would wait behind it for as long as it lived. -fn act( - pending: &mut Pending, - step: Step, - socket_set: &mut SocketSet<'_>, - draw: &mut impl FnMut() -> u16, -) -> Option, Ended>> { - let done = match step { - Step::Ask { asked, to, query } => match free_port(socket_set, draw()) { - Some(port) => { - let handle = socket_set.add(query_socket(port)); - send(socket_set.get_mut::(handle), to, &query); - pending.queries.push((asked, handle)); - None - } - None => Some(Err(Ended::NoPort)), - }, - Step::Wait => None, - Step::Done(result) => Some(result.map_err(Ended::Failed)), - }; - let lookup = &pending.lookup; - let newest = pending.queries.last().map(|&(asked, _)| asked); - pending.queries.retain(|&(asked, handle)| { - let unsent = Some(asked) != newest && socket_set.get::(handle).send_queue() > 0; - let waiting = !unsent && lookup.waiting().any(|w| w == asked); - if !waiting { - socket_set.remove(handle); - } - waiting - }); - done -} - -/// Let go of every socket `pending` holds. -fn close(pending: &Pending, socket_set: &mut SocketSet<'_>) { - for &(_, handle) in &pending.queries { - socket_set.remove(handle); - } -} - -/// A socket for one query, bound to `port` on every address. -fn query_socket(port: u16) -> udp::Socket<'static> { - let buffer = |packets, bytes| udp::PacketBuffer::new(vec![udp::PacketMetadata::EMPTY; packets], vec![0u8; bytes]); - let mut socket = udp::Socket::new(buffer(RECEIVE_PACKETS, RECEIVE_BUFFER), buffer(1, QUERY_BYTES)); - socket - .bind(IpListenEndpoint { addr: None, port }) - .unwrap_or_else(|e| panic!("netstack: a fresh socket refused the free port {port}: {e:?}")); - socket -} - -/// Queue `query` for port 53 of `to` on its own fresh socket. -fn send(socket: &mut udp::Socket, to: [u8; 4], query: &[u8]) { - let at = IpEndpoint::new(IpAddress::Ipv4(Ipv4Addr::from(to)), toyos_dns::PORT); - socket - .send_slice(query, at) - .unwrap_or_else(|e| panic!("netstack: a fresh socket's empty one-query queue refused its query to {at}: {e:?}")); -} - -/// The first port in [`EPHEMERAL`] no UDP socket holds, searched upward from -/// `from` and wrapping; `None` once every one is held. `from` names -/// `EPHEMERAL`'s start plus `from` modulo its size, which for a port already -/// in it is that port: the range starts at a multiple of its size. -pub fn free_port(socket_set: &SocketSet<'_>, from: u16) -> Option { - const SPAN: u32 = *EPHEMERAL.end() as u32 - *EPHEMERAL.start() as u32 + 1; - const _: () = assert!(*EPHEMERAL.start() as u32 % SPAN == 0); - let start = u32::from(from) % SPAN; - (0..SPAN) - .map(|k| EPHEMERAL.start() + ((start + k) % SPAN) as u16) - .find(|&port| !udp_port_taken(socket_set, port)) -} - -#[cfg(test)] -mod tests; diff --git a/userland/netstack/src/resolve/tests.rs b/userland/netstack/src/resolve/tests.rs deleted file mode 100644 index 3b046247a4..0000000000 --- a/userland/netstack/src/resolve/tests.rs +++ /dev/null @@ -1,490 +0,0 @@ -//! The resolver on a wire: smoltcp's own `Interface` on an Ethernet device -//! whose far end is played here frame by frame, so what is judged is what -//! leaves the interface, not what the resolver queued. - -use super::*; -use std::collections::VecDeque; - -use smoltcp::iface::{Config, Interface, PollResult}; -use smoltcp::phy::{self, ChecksumCapabilities, Device, DeviceCapabilities, Medium}; -use smoltcp::time::Instant as SmolInstant; -use smoltcp::wire::{ - ArpOperation, ArpPacket, ArpRepr, EthernetAddress, EthernetFrame, EthernetProtocol, EthernetRepr, - HardwareAddress, IpCidr, IpProtocol, Ipv4Packet, Ipv4Repr, UdpPacket, UdpRepr, -}; - -const OUR_MAC: EthernetAddress = EthernetAddress([0x02, 0, 0, 0, 0, 0x01]); -const OURS: Ipv4Address = Ipv4Address::new(10, 0, 0, 2); -/// On the link, and nothing there answers ARP: a LAN's resolver that is down. -const SILENT: [u8; 4] = [10, 0, 0, 53]; -/// On the link, answering ARP and every query it is given an answer for. -const ANSWERS: [u8; 4] = [10, 0, 0, 54]; -const ANSWERS_MAC: EthernetAddress = EthernetAddress([0x02, 0, 0, 0, 0, 0x54]); -/// Off the link, and the interface holds no route. -const UNROUTED: [u8; 4] = [192, 0, 2, 53]; -const ADDRESS: [u8; 4] = [192, 0, 2, 7]; - -/// Frames for the interface, and the frames it sent. -#[derive(Default)] -struct Wire { - inbound: VecDeque>, - outbound: Vec>, -} - -struct Rx(Vec); -struct Tx<'a>(&'a mut Vec>); - -impl phy::RxToken for Rx { - fn consume(self, f: F) -> R - where - F: FnOnce(&[u8]) -> R, - { - f(&self.0) - } -} - -impl phy::TxToken for Tx<'_> { - fn consume(self, len: usize, f: F) -> R - where - F: FnOnce(&mut [u8]) -> R, - { - let mut frame = vec![0u8; len]; - let result = f(&mut frame); - self.0.push(frame); - result - } -} - -impl Device for Wire { - type RxToken<'a> = Rx; - type TxToken<'a> = Tx<'a>; - - fn receive(&mut self, _: SmolInstant) -> Option<(Rx, Tx<'_>)> { - let frame = self.inbound.pop_front()?; - Some((Rx(frame), Tx(&mut self.outbound))) - } - - fn transmit(&mut self, _: SmolInstant) -> Option> { - Some(Tx(&mut self.outbound)) - } - - fn capabilities(&self) -> DeviceCapabilities { - let mut caps = DeviceCapabilities::default(); - caps.max_transmission_unit = 1514; - caps.medium = Medium::Ethernet; - caps - } -} - -/// What [`ANSWERS`] says for a name. -enum Says { - Address([u8; 4]), - /// A CNAME to this name and nothing else, which the resolver asks again - /// at its end. - Alias(&'static str), -} - -type Draw = Box u16>; - -/// A draw an off-path sender could predict, which is all a test needs. -fn counter() -> Draw { - let mut x: u32 = 0x9e37_79b9; - Box::new(move || { - x ^= x << 13; - x ^= x >> 17; - x ^= x << 5; - x as u16 - }) -} - -struct Net { - iface: Interface, - wire: Wire, - sockets: SocketSet<'static>, - resolver: Resolver, - born: Instant, - now_ms: u64, - /// [`ANSWERS`]'s zone: a name, the time before which its answer is held - /// back, and what it says. - zone: Vec<(&'static str, u64, Says)>, - /// Frames [`ANSWERS`] has sent and the wire has not yet delivered, with - /// the time they arrive. - held: Vec<(u64, Vec)>, - /// Every address the interface asked ARP for. - arp_asked: Vec<[u8; 4]>, - /// Every server a query reached. - queried: Vec<[u8; 4]>, - /// The port every query left from. - sources: Vec, - ended: Vec<(u32, Name, Result, Ended>)>, -} - -impl Net { - fn new(servers: &[[u8; 4]]) -> Self { - let mut wire = Wire::default(); - let mut iface = - Interface::new(Config::new(HardwareAddress::Ethernet(OUR_MAC)), &mut wire, SmolInstant::from_millis(0)); - iface.update_ip_addrs(|addrs| addrs.push(IpCidr::new(IpAddress::Ipv4(OURS), 24)).unwrap()); - let born = Instant::now(); - let mut resolver = Resolver::new(born, counter()); - let servers: Vec = servers.iter().map(|s| Ipv4Address::from(*s)).collect(); - resolver.set_servers(&servers); - Self { - iface, - wire, - sockets: SocketSet::new(Vec::new()), - resolver, - born, - now_ms: 0, - zone: Vec::new(), - held: Vec::new(), - arp_asked: Vec::new(), - queried: Vec::new(), - sources: Vec::new(), - ended: Vec::new(), - } - } - - fn now(&self) -> Instant { - self.born + Duration::from_millis(self.now_ms) - } - - fn start(&mut self, client: u32, name: &str) -> Result<(), Refused> { - let now = self.now(); - self.resolver.start(client, Name::parse(name).unwrap(), &mut self.sockets, now).map_err(|(_, why)| why) - } - - fn poll(&mut self) { - let at = SmolInstant::from_millis(self.now_ms as i64); - while self.iface.poll(at, &mut self.wire, &mut self.sockets) != PollResult::None {} - } - - /// Passes of netstack's loop, each followed by the far end's turn, for as long - /// as a frame is due now: a frame is the NIC's interrupt, which wakes the - /// loop at once. - fn pass(&mut self) { - loop { - let now_ms = self.now_ms; - let (due, later): (Vec<_>, Vec<_>) = self.held.drain(..).partition(|(at, _)| *at <= now_ms); - self.held = later; - self.wire.inbound.extend(due.into_iter().map(|(_, frame)| frame)); - self.poll(); - let now = self.now(); - let ended = self.resolver.pass(&mut self.sockets, now); - self.ended.extend(ended); - self.poll(); - for frame in std::mem::take(&mut self.wire.outbound) { - self.far_end(&frame); - } - if self.wire.inbound.is_empty() && self.held.iter().all(|(at, _)| *at > now_ms) { - return; - } - } - } - - /// When netstack's loop next wakes, as its `main` computes it: the soonest of - /// the resolver's wake, smoltcp's own and the next frame the far end has - /// on the wire. `None` is a loop asleep until something else wakes it. - fn next_wake(&mut self) -> Option { - let now = self.now(); - let resolver = self.resolver.wake_in(now).map(|d| d.as_millis() as u64); - let smoltcp = self - .iface - .poll_delay(SmolInstant::from_millis(self.now_ms as i64), &self.sockets) - .map(|d| d.total_micros().div_ceil(1000)); - let frame = self.held.iter().map(|(at, _)| at - self.now_ms).min(); - let wake = [resolver, smoltcp, frame].into_iter().flatten().min()?; - assert!(wake > 0, "netstack's loop would wake at once again at {} ms, having just passed", self.now_ms); - Some(self.now_ms + wake) - } - - /// Pass at each of netstack's wakes up to `ms`, and at `ms`. - fn until(&mut self, ms: u64) { - loop { - self.pass(); - if self.now_ms == ms { - return; - } - self.now_ms = self.next_wake().map_or(ms, |at| at.min(ms)); - } - } - - /// Pass at each of netstack's wakes until `client`'s lookup has ended, or - /// `until_ms` has come. A lookup in flight with no wake to carry it is a - /// loop that would sleep through it, and ends the test. - fn run(&mut self, client: u32, until_ms: u64) -> Option, Ended>> { - loop { - self.pass(); - if let Some(at) = self.ended.iter().position(|(c, ..)| *c == client) { - return Some(self.ended.remove(at).2); - } - let at = self.next_wake().unwrap_or_else(|| { - panic!("lookup {client} is in flight at {} ms and nothing would wake netstack's loop", self.now_ms) - }); - if at > until_ms { - return None; - } - self.now_ms = at; - } - } - - fn far_end(&mut self, frame: &[u8]) { - let frame = EthernetFrame::new_checked(frame).expect("the interface sent an Ethernet frame"); - match frame.ethertype() { - EthernetProtocol::Arp => { - let arp = ArpRepr::parse(&ArpPacket::new_checked(frame.payload()).unwrap()).unwrap(); - let ArpRepr::EthernetIpv4 { operation: ArpOperation::Request, target_protocol_addr, .. } = arp else { - return; - }; - self.arp_asked.push(target_protocol_addr.octets()); - if target_protocol_addr.octets() == ANSWERS { - let reply = ArpRepr::EthernetIpv4 { - operation: ArpOperation::Reply, - source_hardware_addr: ANSWERS_MAC, - source_protocol_addr: Ipv4Address::from(ANSWERS), - target_hardware_addr: OUR_MAC, - target_protocol_addr: OURS, - }; - let mut out = vec![0u8; 14 + reply.buffer_len()]; - let mut eth = EthernetFrame::new_unchecked(&mut out); - EthernetRepr { src_addr: ANSWERS_MAC, dst_addr: OUR_MAC, ethertype: EthernetProtocol::Arp } - .emit(&mut eth); - reply.emit(&mut ArpPacket::new_unchecked(eth.payload_mut())); - self.wire.inbound.push_back(out); - } - } - EthernetProtocol::Ipv4 => { - let ip = Ipv4Packet::new_checked(frame.payload()).unwrap(); - assert_eq!(ip.next_header(), IpProtocol::Udp, "the resolver sends only UDP"); - let udp = UdpPacket::new_checked(ip.payload()).unwrap(); - assert_eq!(udp.dst_port(), toyos_dns::PORT); - let to = ip.dst_addr().octets(); - self.queried.push(to); - self.sources.push(udp.src_port()); - assert_eq!(to, ANSWERS, "a query left for a server the wire cannot reach"); - self.answer(udp.src_port(), udp.payload()); - } - other => panic!("the interface sent a frame of type {other}"), - } - } - - /// [`ANSWERS`]'s reply to `query`, from its port 53 to `port`, spelled by - /// hand: the question echoed and one answer record behind a pointer to it. - fn answer(&mut self, port: u16, query: &[u8]) { - let asked = question(query); - let Some((_, not_before, says)) = self.zone.iter().find(|(name, ..)| *name == asked) else { - return; - }; - let mut reply = query.to_vec(); - reply[2..4].copy_from_slice(&0x8180u16.to_be_bytes()); - reply[6..8].copy_from_slice(&1u16.to_be_bytes()); - reply.extend_from_slice(&[0xc0, 12]); - let (rtype, data) = match says { - Says::Address(addr) => (1u16, addr.to_vec()), - Says::Alias(target) => (5, Name::parse(target).unwrap().wire().to_vec()), - }; - reply.extend_from_slice(&rtype.to_be_bytes()); - reply.extend_from_slice(&1u16.to_be_bytes()); - reply.extend_from_slice(&60u32.to_be_bytes()); - reply.extend_from_slice(&(data.len() as u16).to_be_bytes()); - reply.extend_from_slice(&data); - - let caps = ChecksumCapabilities::default(); - let udp = UdpRepr { src_port: toyos_dns::PORT, dst_port: port }; - let ip = Ipv4Repr { - src_addr: Ipv4Address::from(ANSWERS), - dst_addr: OURS, - next_header: IpProtocol::Udp, - payload_len: udp.header_len() + reply.len(), - hop_limit: 64, - }; - let mut out = vec![0u8; 14 + ip.buffer_len() + ip.payload_len]; - let mut eth = EthernetFrame::new_unchecked(&mut out); - EthernetRepr { src_addr: ANSWERS_MAC, dst_addr: OUR_MAC, ethertype: EthernetProtocol::Ipv4 }.emit(&mut eth); - let mut packet = Ipv4Packet::new_unchecked(eth.payload_mut()); - ip.emit(&mut packet, &caps); - udp.emit( - &mut UdpPacket::new_unchecked(packet.payload_mut()), - &IpAddress::Ipv4(ip.src_addr), - &IpAddress::Ipv4(ip.dst_addr), - reply.len(), - |payload| payload.copy_from_slice(&reply), - &caps, - ); - self.held.push(((*not_before).max(self.now_ms), out)); - } -} - -/// The dotted name a query asks, read by hand. -fn question(query: &[u8]) -> String { - let mut labels = Vec::new(); - let mut at = 12; - while query[at] != 0 { - let len = usize::from(query[at]); - labels.push(std::str::from_utf8(&query[at + 1..at + 1 + len]).unwrap().to_string()); - at += 1 + len; - } - labels.join(".") -} - -/// **A query to a server whose link address never resolves holds up no other -/// query.** smoltcp keeps a datagram at the head of its socket's queue while -/// its neighbour is missing, so a query queued behind it on the same socket -/// never leaves. The first server is on the link and answers no ARP; the -/// second is asked when the first query's wait is over, and answers. -#[test] -fn a_server_that_answers_no_arp_holds_up_no_other_server() { - let mut net = Net::new(&[SILENT, ANSWERS]); - net.zone.push(("www.example", 0, Says::Address(ADDRESS))); - net.start(1, "www.example").unwrap(); - let ended = net.run(1, 25_000); - assert!(net.arp_asked.contains(&SILENT), "the premise: the first server was asked for its link address"); - assert_eq!(ended, Some(Ok(vec![ADDRESS])), "the second server was never asked: it heard {:?}", net.queried); - assert!(net.now_ms < 2 * toyos_dns::WAIT_MS, "answered at {} ms", net.now_ms); -} - -/// The same, for a server no route leads to: smoltcp keeps that datagram at -/// the head of its queue too. -#[test] -fn a_server_with_no_route_holds_up_no_other_server() { - let mut net = Net::new(&[UNROUTED, ANSWERS]); - net.zone.push(("www.example", 0, Says::Address(ADDRESS))); - net.start(1, "www.example").unwrap(); - let ended = net.run(1, 25_000); - assert_eq!(ended, Some(Ok(vec![ADDRESS])), "the second server was never asked: it heard {:?}", net.queried); - assert!(net.now_ms < 2 * toyos_dns::WAIT_MS, "answered at {} ms", net.now_ms); -} - -/// **An alias answered late, while queries sit behind a missing neighbour, -/// restarts the lookup without overflowing anything.** The first server -/// answers its first query only after five more have been sent, the second -/// server's among them, and answers with an alias alone, so the lookup asks -/// again at the alias with every query afresh. The reply reaches netstack from -/// the network, so nothing it arranges may end netstack. -#[test] -fn an_alias_answered_while_queries_are_stuck_restarts_the_lookup() { - let mut net = Net::new(&[ANSWERS, SILENT]); - net.zone.push(("www.example", 10_500, Says::Alias("cdn.example"))); - net.zone.push(("cdn.example", 0, Says::Address(ADDRESS))); - net.start(1, "www.example").unwrap(); - let ended = net.run(1, 40_000); - assert_eq!(ended, Some(Ok(vec![ADDRESS])), "the servers heard {:?}", net.queried); -} - -/// **At most [`MAX_LOOKUPS`] are in flight**, and one past it is refused with -/// the code a client reads as "this machine is full". -#[test] -fn the_lookup_past_the_cap_is_refused_as_exhausted() { - let mut net = Net::new(&[SILENT]); - for client in 0..MAX_LOOKUPS as u32 { - net.start(client, "www.example").unwrap_or_else(|_| panic!("lookup {client} was refused")); - } - let refused = net.start(MAX_LOOKUPS as u32, "www.example").expect_err("one lookup past the cap"); - assert!(matches!(refused, Refused::Full)); - assert_eq!(refused.code(), toyos::net::ERR_RESOURCE_EXHAUSTED); -} - -/// **A lookup whose client has left is let go at once**: every socket its -/// queries left from leaves the stack, and its slot takes another lookup. -#[test] -fn a_lookup_whose_client_left_is_let_go_at_once() { - let mut net = Net::new(&[ANSWERS]); - for client in 0..MAX_LOOKUPS as u32 { - net.start(client, "www.example").unwrap(); - } - net.until(toyos_dns::WAIT_MS); - assert!(net.ended.is_empty(), "nothing answers, so nothing has ended"); - assert_eq!(net.queried.len(), 2 * MAX_LOOKUPS, "two queries each left, neither answered"); - assert_eq!(net.resolver.sockets(), 2 * MAX_LOOKUPS); - assert_eq!(net.sockets.iter().count(), 2 * MAX_LOOKUPS); - net.resolver.let_go(&mut net.sockets, |&client| client == 3); - assert_eq!(net.sockets.iter().count(), 2 * (MAX_LOOKUPS - 1), "both of its sockets left the stack"); - assert_eq!(net.resolver.sockets(), 2 * (MAX_LOOKUPS - 1)); - assert!(net.resolver.clients().all(|&client| client != 3)); - net.start(MAX_LOOKUPS as u32, "www.example").expect("the slot it left takes another lookup"); -} - -/// **Each query waiting for its answer holds one socket; one whose wait ended -/// before it left holds none, and neither does a lookup that ended.** -#[test] -fn a_lookup_holds_a_socket_per_query_that_left_and_none_once_ended() { - let mut net = Net::new(&[ANSWERS, SILENT]); - net.start(1, "www.example").unwrap(); - net.until(toyos_dns::WAIT_MS - 1); - assert_eq!(net.queried, [ANSWERS]); - assert_eq!(net.sockets.iter().count(), 1); - net.until(toyos_dns::WAIT_MS); - assert_eq!(net.sockets.iter().count(), 2, "the first query is still answered, the second is new"); - net.until(2 * toyos_dns::WAIT_MS); - assert_eq!(net.queried, [ANSWERS, ANSWERS], "the second query never left"); - assert_eq!(net.sockets.iter().count(), 2, "the second query's socket went when its wait ended"); - net.zone.push(("www.example", 0, Says::Address(ADDRESS))); - assert_eq!(net.run(1, 5 * toyos_dns::WAIT_MS), Some(Ok(vec![ADDRESS])), "the fifth query, to the first server"); - assert_eq!(net.sockets.iter().count(), 0, "an ended lookup holds no socket"); -} - -/// **A query never leaves from a port a client holds.** smoltcp hands a -/// datagram to the first socket that takes it, so a query on a client's port -/// would have its answer read by the client, or the client's datagrams read -/// by the lookup. The client here sits on the port the draw lands on. -#[test] -fn a_query_leaves_from_no_port_a_client_holds() { - let mut draw = counter(); - let _id = draw(); - let drawn = free_port(&SocketSet::new(Vec::new()), draw()).expect("an empty stack holds no port"); - - let mut open = Net::new(&[ANSWERS]); - open.zone.push(("www.example", 0, Says::Address(ADDRESS))); - open.start(1, "www.example").unwrap(); - assert_eq!(open.run(1, toyos_dns::WAIT_MS), Some(Ok(vec![ADDRESS]))); - assert_eq!(open.sources, [drawn], "the premise: with nothing bound, the query leaves from the drawn port"); - - let mut net = Net::new(&[ANSWERS]); - net.zone.push(("www.example", 0, Says::Address(ADDRESS))); - let buffer = || udp::PacketBuffer::new(vec![udp::PacketMetadata::EMPTY; 1], vec![0u8; 512]); - let mut client = udp::Socket::new(buffer(), buffer()); - client.bind(IpListenEndpoint { addr: None, port: drawn }).unwrap(); - let client = net.sockets.add(client); - net.start(1, "www.example").unwrap(); - let ended = net.run(1, toyos_dns::WAIT_MS); - assert!(!net.sources.contains(&drawn), "a query left from port {drawn}, which a client holds"); - assert_eq!(net.sources.len(), 1, "one query left"); - assert_eq!(ended, Some(Ok(vec![ADDRESS])), "the lookup did not read its own answer"); - assert!(!net.sockets.get_mut::(client).can_recv(), "the client was handed the lookup's answer"); -} - -/// **A lookup's waits are netstack's wakes**: a server that is reached and never -/// answers is asked again the moment each wait ends, and the lookup ends -/// timed out the moment its last one does, with nothing but the resolver's -/// own wake to carry it there. -#[test] -fn a_server_that_never_answers_is_asked_at_each_waits_end() { - let mut net = Net::new(&[ANSWERS]); - net.start(1, "www.example").unwrap(); - let ended = net.run(1, 10 * toyos_dns::WAIT_MS); - assert_eq!(ended, Some(Err(Ended::Failed(Failure::TimedOut)))); - assert_eq!(net.now_ms, toyos_dns::ROUNDS as u64 * toyos_dns::WAIT_MS, "the lookup ended late"); - assert_eq!(net.queried, [ANSWERS; toyos_dns::ROUNDS]); -} - -/// **A lookup's wait is its own, not the latest of every lookup in flight.** -/// A second lookup started half a wait behind the first must not push the -/// first's wake back to the second's: `wake_in` names the soonest due lookup, -/// and the first here ends on its own schedule regardless of the second. -#[test] -fn a_lookup_is_not_carried_by_a_later_ones_schedule() { - let mut net = Net::new(&[ANSWERS]); - net.start(1, "www.example").unwrap(); - net.until(toyos_dns::WAIT_MS / 2); - net.start(2, "other.example").unwrap(); - let ended = net.run(1, 10 * toyos_dns::WAIT_MS); - assert_eq!(ended, Some(Err(Ended::Failed(Failure::TimedOut)))); - assert_eq!(net.now_ms, toyos_dns::ROUNDS as u64 * toyos_dns::WAIT_MS, "lookup 1 waited on lookup 2's schedule"); - let ended = net.run(2, 10 * toyos_dns::WAIT_MS); - assert_eq!(ended, Some(Err(Ended::Failed(Failure::TimedOut)))); - assert_eq!( - net.now_ms, - toyos_dns::WAIT_MS / 2 + toyos_dns::ROUNDS as u64 * toyos_dns::WAIT_MS, - "lookup 2 waited on lookup 1's schedule" - ); -} diff --git a/userland/netstack/src/serve.rs b/userland/netstack/src/serve.rs new file mode 100644 index 0000000000..478cd9b2ff --- /dev/null +++ b/userland/netstack/src/serve.rs @@ -0,0 +1,854 @@ +//! A client's requests onto the node's calls, and the node's answers back +//! onto the pipe ABI (`toyos::net`). +//! +//! **One request is one call of the node's**, and every decision about a +//! socket is the node's: what is here is the number a client names a socket +//! by, the clients that wait for an answer, the pipe ends netstack watches, +//! and the word each of the node's refusals is written in. Two requests reach +//! no call: a datagram's bytes travel in the socket's own pipes, which are +//! netstack's and not the node's, and a shutdown of the receiving half alone +//! asks nothing of the stack, std keeping that half's state itself. +//! +//! **Untrusted input.** A request's every field is its client's number: an id +//! that names nothing, or a socket of another kind, is refused +//! `ERR_NOT_CONNECTED`, a payload that is not the request's struct, a port of +//! zero and a name that is none `ERR_INVALID_INPUT`, and a length is a bound +//! on a read and never an allocation's size past what one frame carries. A +//! request that waits holds its client's connection, so what waits is +//! bounded: one receive a socket, a second refused `ERR_RESOURCE_EXHAUSTED`. +//! Nothing a client sends panics netstack. +//! +//! **An id is a number any client can name** +//! (`issues/netstack-socket-ids-are-ambient.md`): the table starts at a +//! random one so that an id a client kept across a replaced netstack names +//! nothing in the next. + +use std::collections::{BTreeMap, HashMap}; +use std::net::Ipv4Addr; +use std::time::Duration; + +use toyos::net::*; +use toyos::poller::{OTHER_END_GONE, READABLE, WRITABLE, Poller}; +use toyos::{ipc, say, AsHandle, Pipe}; +use toyos_abi::syscall::SyscallError; +use toyos_inspect::Snapshot; +use toyos_net_node::{ + AcceptRefused, ConnectRefused, DatagramId, Ended, Event, ListenRefused, ListenerId, LookupId, Node, NotStarted, + PipeEnd, Pipes, Refused, StreamEvent, StreamId, WriteRefusal, +}; +use toyos_net_shard::ConnectError; +use toyos_net_tcp::{Endpoint, Failure}; +use toyos_net_wire::{Instant, Port}; + +use crate::client::{Client, Request}; +use crate::pipes::{self, Watched}; +use crate::HOSTNAME; + +/// A watch's token names the socket it was asked of by its id, in the low +/// word, and never a place in a list: an answer can arrive after its socket +/// is gone, and must then name nothing. +const TOKEN_FROM_CLIENT: u64 = 1 << 32; +const TOKEN_TO_CLIENT: u64 = 2 << 32; +const TOKEN_LISTENER: u64 = 3 << 32; +const TOKEN_DATAGRAM: u64 = 4 << 32; +/// A client that waits for its lookup's answer, by its connection's handle. +/// Clear of the loop's own tokens and of a pending connection's. +const TOKEN_LOOKUP: u64 = 5 << 32; +const TOKEN_KIND: u64 = !(u32::MAX as u64); + +/// Watches one place makes at most: a stream's two pipes. +pub const WATCHES_PER_PLACE: u32 = 2; + +/// Watches the lookups make: each waiting client's connection. +pub const LOOKUP_WATCHES: u32 = toyos_dns::MAX_LOOKUPS as u32; + +/// The most addresses one lookup's answer carries: what +/// `toyos::net::dns_lookup`'s 256-byte buffer holds, a count byte and five +/// bytes an address. +const MAX_ANSWERED: usize = (256 - 1) / 5; + +/// What a client's id names. +enum Socket { + Stream(StreamId), + Listener { id: ListenerId, wakes: Watched }, + Datagram(Datagram), +} + +/// A datagram socket's two pipes, which carry its payloads between its client +/// and netstack: the node holds a datagram only between two calls. +struct Datagram { + id: DatagramId, + to_client: Pipe, + from_client: Pipe, +} + +/// The pipe ends of a stream the node holds, for as long as it holds either: +/// longer than its client's id names it, since a closed stream's pipe is +/// still read until it is empty. +struct Ends { + stream: StreamId, + to_client: Watched, + from_client: Watched, +} + +/// A client waiting for a datagram to reach its socket. +struct Receiving { + client: Client, + max_len: u32, +} + +pub struct Sockets { + ids: HashMap, + next_id: u32, + ends: HashMap, + /// The id each stream's ends are kept under. + by_stream: BTreeMap, + /// The client each unanswered connect is answered to, and the id it is + /// told. + connecting: BTreeMap, + lookups: Vec<(LookupId, Client, toyos_dns::Name)>, + /// At most one receive waits on a socket, so a client's requests alone + /// hold no more of netstack's handles than its sockets do. + receiving: BTreeMap, + /// The places the node was given. + places: usize, + /// A stream's pipe was answered ready in this wake: one pass over the + /// streams follows the wake, however many answers it carried. + bridge: bool, +} + +/// The two ends a connect or an accept moves, each split into the node's half +/// and the watched one. +fn data_pipes(client: &Client) -> Option<(Pipes, Watched, Watched)> { + let [to_client, from_client] = client.conn.recv_handles_exact::<{ DATA_HANDLES }>()?; + // SAFETY: the kernel moved both handles into this process with the frame + // just read, and nothing else answers for either. + let (to_client, from_client) = unsafe { (Pipe::from_raw(to_client), Pipe::from_raw(from_client)) }; + let (to_held, to_watched) = pipes::hold(to_client); + let (from_held, from_watched) = pipes::hold(from_client); + Some((Pipes { to_client: Box::new(to_held), from_client: Box::new(from_held) }, to_watched, from_watched)) +} + +/// The pipe ABI's word for each of the node's. +fn refused(refusal: Refused) -> u32 { + match refusal { + Refused::AddrInUse => ERR_ADDR_IN_USE, + Refused::NotConnected => ERR_NOT_CONNECTED, + Refused::InvalidInput => ERR_INVALID_INPUT, + Refused::PermissionDenied => ERR_PERMISSION_DENIED, + Refused::ResourceExhausted => ERR_RESOURCE_EXHAUSTED, + } +} + +/// A connect that made no stream. A machine with no address or no route yet +/// is `ERR_NOT_CONNECTED`, which clears when a lease lands, and never a +/// peer's refusal; no free port is `ERR_ADDR_IN_USE`. +fn connect_refused(refusal: ConnectRefused) -> u32 { + use toyos_net_tcp::Error; + match refusal { + ConnectRefused::Full => ERR_RESOURCE_EXHAUSTED, + ConnectRefused::Stack(ConnectError::Route(_)) => ERR_NOT_CONNECTED, + ConnectRefused::Stack(ConnectError::NotUnicast | ConnectError::Tcp(Error::InvalidRemote)) => ERR_INVALID_INPUT, + ConnectRefused::Stack(ConnectError::Tcp(Error::AddrInUse)) => ERR_ADDR_IN_USE, + ConnectRefused::Stack(ConnectError::Tcp( + why @ (Error::NoSuchSocket | Error::NotConnected | Error::Exists | Error::Closing | Error::WouldBlock | Error::Failed(_)), + )) => { + say!("netstack: the stack refused an active open as {why:?}, which is no refusal of one"); + ERR_OTHER + } + } +} + +/// A handshake that ended without a connection. The pipe ABI has a word for a +/// peer's refusal, its reset and its silence, and none for an ICMP error's: +/// those are named in the log and answered `ERR_OTHER` +/// (`issues/the-pipe-abi-has-no-word-for-an-unreachable-host-or-a-lookup-to-try-again.md`). +fn connect_failed(failure: Failure) -> u32 { + match failure { + Failure::Refused => ERR_CONNECTION_REFUSED, + Failure::Reset => ERR_CONNECTION_RESET, + Failure::TimedOut => ERR_TIMED_OUT, + Failure::Unreachable(_) | Failure::Prohibited => { + say!("netstack: a connect ended {failure:?}"); + ERR_OTHER + } + } +} + +fn listen_refused(refusal: ListenRefused) -> u32 { + match refusal { + ListenRefused::Full => ERR_RESOURCE_EXHAUSTED, + // The word a datagram socket's bind to such an address is answered in. + ListenRefused::NotLocal => ERR_INVALID_INPUT, + ListenRefused::InUse => ERR_ADDR_IN_USE, + } +} + +fn accept_refused(refusal: AcceptRefused) -> u32 { + match refusal { + AcceptRefused::NoListener | AcceptRefused::Nothing => ERR_NOT_CONNECTED, + AcceptRefused::NoPipes => ERR_INVALID_INPUT, + AcceptRefused::Full => ERR_RESOURCE_EXHAUSTED, + } +} + +/// A lookup's answer: a count, then each address behind the family tag 4. A +/// resolver may answer with a subset of a name's addresses, and these are the +/// ones the server put first. +fn answer_lookup(client: &Client, addrs: &[[u8; 4]]) { + let mut answer = vec![addrs.len().min(MAX_ANSWERED) as u8]; + for addr in addrs.iter().take(MAX_ANSWERED) { + answer.push(4); + answer.extend_from_slice(addr); + } + client.result_bytes(&answer); +} + +impl Sockets { + pub fn new(first_id: u32, places: usize) -> Self { + Self { + ids: HashMap::new(), + next_id: first_id.max(1), + ends: HashMap::new(), + by_stream: BTreeMap::new(), + connecting: BTreeMap::new(), + lookups: Vec::new(), + receiving: BTreeMap::new(), + places, + bridge: false, + } + } + + /// An id no socket and no stream's ends are kept under, and never 0. + fn alloc_id(&mut self) -> u32 { + loop { + let id = self.next_id; + self.next_id = match self.next_id.wrapping_add(1) { + 0 => 1, + next => next, + }; + if !self.ids.contains_key(&id) && !self.ends.contains_key(&id) { + return id; + } + } + } + + fn hold_stream(&mut self, stream: StreamId, to_client: Watched, from_client: Watched) -> u32 { + let id = self.alloc_id(); + self.ids.insert(id, Socket::Stream(stream)); + self.ends.insert(id, Ends { stream, to_client, from_client }); + self.by_stream.insert(stream, id); + id + } + + fn stream(&self, socket_id: u32) -> Option { + match self.ids.get(&socket_id) { + Some(Socket::Stream(id)) => Some(*id), + Some(Socket::Listener { .. } | Socket::Datagram(_)) | None => None, + } + } + + fn listener(&self, socket_id: u32) -> Option { + match self.ids.get(&socket_id) { + Some(Socket::Listener { id, .. }) => Some(*id), + Some(Socket::Stream(_) | Socket::Datagram(_)) | None => None, + } + } + + fn datagram(&self, socket_id: u32) -> Option<&Datagram> { + match self.ids.get(&socket_id) { + Some(Socket::Datagram(socket)) => Some(socket), + Some(Socket::Stream(_) | Socket::Listener { .. }) | None => None, + } + } + + /// One whole request. A call that answers at once drops its client where + /// it answers; a connect, a lookup and a receive with nothing to hand + /// over keep theirs until the node has the answer ([`Self::settle`]). + pub fn request(&mut self, node: &mut Node, now: Instant, req: Request, draw: fn() -> u32) { + match MsgType::from_u32(req.msg_type) { + Some(MsgType::TcpClose) => self.close(node, now, &req), + Some(MsgType::TcpShutdown) => self.shutdown(node, now, &req), + Some(MsgType::UdpBind) => self.udp_bind(node, &req, draw), + Some(MsgType::UdpSendTo) => self.udp_send_to(node, now, &req), + Some(MsgType::UdpRecvFrom) => self.udp_recv_from(node, now, req), + Some(MsgType::UdpClose) => self.udp_close(node, now, &req), + Some(MsgType::DnsLookup) => self.lookup(node, now, req, draw), + Some(MsgType::TcpSetOption) => self.set_option(node, now, &req), + Some(MsgType::TcpListenerSetOption) => self.listener_set_option(node, &req), + Some(MsgType::UdpSetOption) => self.udp_set_option(node, &req), + Some(MsgType::TcpConnectPiped) => self.connect(node, now, req), + Some(MsgType::TcpBindPiped) => self.listen(node, &req, draw), + Some(MsgType::TcpAcceptPiped) => self.accept(node, now, &req), + None => { + say!("netstack: unknown message type {}", req.msg_type); + req.client.error(ERR_INVALID_INPUT); + } + } + } + + /// The client lets go of whatever its id names; an id that names nothing + /// is let go already. + fn close(&mut self, node: &mut Node, now: Instant, msg: &Request) { + let Ok(req) = ipc::decode_payload::(msg.payload()) else { + msg.client.error(ERR_INVALID_INPUT); + return; + }; + match self.ids.remove(&req.socket_id) { + Some(Socket::Stream(id)) => node.close(now, id), + Some(Socket::Listener { id, .. }) => { + node.close_listener(now, id); + } + Some(Socket::Datagram(socket)) => self.end_datagram(node, now, socket), + None => {} + } + msg.client.done(); + } + + fn shutdown(&mut self, node: &mut Node, now: Instant, msg: &Request) { + let Ok(req) = ipc::decode_payload::(msg.payload()) else { + msg.client.error(ERR_INVALID_INPUT); + return; + }; + let Some(id) = self.stream(req.socket_id) else { + msg.client.error(ERR_NOT_CONNECTED); + return; + }; + let known = match req.how { + // The receiving half alone: std keeps its state, and the id names + // a stream. + 0 => true, + 1 | 2 => node.shutdown_write(now, id), + _ => { + msg.client.error(ERR_INVALID_INPUT); + return; + } + }; + if known { + msg.client.done(); + } else { + msg.client.error(ERR_NOT_CONNECTED); + } + } + + fn set_option(&mut self, node: &mut Node, now: Instant, msg: &Request) { + let Ok(req) = ipc::decode_payload::(msg.payload()) else { + msg.client.error(ERR_INVALID_INPUT); + return; + }; + let Some(id) = self.stream(req.socket_id) else { + msg.client.error(ERR_NOT_CONNECTED); + return; + }; + match req.option { + OPT_NODELAY if node.set_nodelay(now, id, req.value != 0) => msg.client.done(), + OPT_NODELAY => msg.client.error(ERR_NOT_CONNECTED), + _ => msg.client.error(ERR_INVALID_INPUT), + } + } + + fn listener_set_option(&mut self, node: &mut Node, msg: &Request) { + let Ok(req) = ipc::decode_payload::(msg.payload()) else { + msg.client.error(ERR_INVALID_INPUT); + return; + }; + let Some(id) = self.listener(req.socket_id) else { + msg.client.error(ERR_NOT_CONNECTED); + return; + }; + match req.option { + OPT_NODELAY if node.set_listener_nodelay(id, req.value != 0) => msg.client.done(), + OPT_NODELAY => msg.client.error(ERR_NOT_CONNECTED), + _ => msg.client.error(ERR_INVALID_INPUT), + } + } + + fn udp_set_option(&mut self, node: &mut Node, msg: &Request) { + let Ok(req) = ipc::decode_payload::(msg.payload()) else { + msg.client.error(ERR_INVALID_INPUT); + return; + }; + let Some(socket) = self.datagram(req.socket_id) else { + msg.client.error(ERR_NOT_CONNECTED); + return; + }; + match req.option { + OPT_BROADCAST => match node.udp_set_broadcast(socket.id, req.value != 0) { + Ok(()) => msg.client.done(), + Err(refusal) => msg.client.error(refused(refusal)), + }, + _ => msg.client.error(ERR_INVALID_INPUT), + } + } + + fn connect(&mut self, node: &mut Node, now: Instant, msg: Request) { + let Ok(req) = ipc::decode_payload::(msg.payload()) else { + msg.client.error(ERR_INVALID_INPUT); + return; + }; + let (Some((pipes, to_client, from_client)), Some(port)) = (data_pipes(&msg.client), Port::new(req.port)) else { + msg.client.error(ERR_INVALID_INPUT); + return; + }; + let remote = Endpoint { addr: Ipv4Addr::from(req.addr), port }; + let timeout = (req.timeout_ms > 0).then(|| Duration::from_millis(u64::from(req.timeout_ms))); + match node.connect(now, remote, timeout, pipes) { + Ok(stream) => { + let id = self.hold_stream(stream, to_client, from_client); + self.connecting.insert(stream, (msg.client, id)); + } + Err(refusal) => { + if refusal == ConnectRefused::Full { + say!("netstack: refusing connect, {} of {} places held", node.held(), self.places); + } + msg.client.error(connect_refused(refusal)); + } + } + } + + fn listen(&mut self, node: &mut Node, msg: &Request, draw: fn() -> u32) { + let Ok(req) = ipc::decode_payload::(msg.payload()) else { + msg.client.error(ERR_INVALID_INPUT); + return; + }; + let Some([wakes]) = msg.client.conn.recv_handles_exact::<{ NOTIFY_HANDLES }>() else { + msg.client.error(ERR_INVALID_INPUT); + return; + }; + // SAFETY: the kernel moved the handle into this process with the frame + // just read, and nothing else answers for it. + let (held, wakes) = pipes::hold(unsafe { Pipe::from_raw(wakes) }); + match node.listen(Ipv4Addr::from(req.addr), Port::new(req.port), req.options.nodelay(), Box::new(held), draw) { + Ok((listener, port)) => { + let socket_id = self.alloc_id(); + self.ids.insert(socket_id, Socket::Listener { id: listener, wakes }); + msg.client.result(&TcpBindResponse { socket_id, bound_port: port.get(), _pad: 0 }); + } + Err(refusal) => msg.client.error(listen_refused(refusal)), + } + } + + fn accept(&mut self, node: &mut Node, now: Instant, msg: &Request) { + let Ok(req) = ipc::decode_payload::(msg.payload()) else { + msg.client.error(ERR_INVALID_INPUT); + return; + }; + let moved = data_pipes(&msg.client); + let Some(listener) = self.listener(req.socket_id) else { + msg.client.error(accept_refused(AcceptRefused::NoListener)); + return; + }; + let (pipes, watched) = match moved { + Some((pipes, to_client, from_client)) => (Some(pipes), Some((to_client, from_client))), + None => (None, None), + }; + match (node.accept(now, listener, pipes), watched) { + (Ok(accepted), Some((to_client, from_client))) => { + let socket_id = self.hold_stream(accepted.id, to_client, from_client); + msg.client.result(&TcpAcceptPipedResponse { + socket_id, + remote_addr: accepted.remote.addr.octets(), + remote_port: accepted.remote.port.get(), + local_port: accepted.local.get(), + options: TcpOptions::new(accepted.nodelay), + }); + } + (Ok(_), None) => unreachable!("the node made a stream of an accept that moved no pipes"), + (Err(refusal), _) => { + if refusal == AcceptRefused::Full { + say!("netstack: refusing accept, {} of {} places held", node.held(), self.places); + } + msg.client.error(accept_refused(refusal)); + } + } + } + + fn udp_bind(&mut self, node: &mut Node, msg: &Request, draw: fn() -> u32) { + let Ok(req) = ipc::decode_payload::(msg.payload()) else { + msg.client.error(ERR_INVALID_INPUT); + return; + }; + let Some([to_client, from_client]) = msg.client.conn.recv_handles_exact::<{ DATA_HANDLES }>() else { + msg.client.error(ERR_INVALID_INPUT); + return; + }; + // SAFETY: the kernel moved both handles into this process with the + // frame just read, and nothing else answers for either. + let (to_client, from_client) = unsafe { (Pipe::from_raw(to_client), Pipe::from_raw(from_client)) }; + match node.udp_bind(Ipv4Addr::from(req.addr), Port::new(req.port), draw) { + Ok((id, port)) => { + let socket_id = self.alloc_id(); + self.ids.insert(socket_id, Socket::Datagram(Datagram { id, to_client, from_client })); + msg.client.result(&UdpBindResponse { socket_id, bound_port: port.get(), _pad: 0 }); + } + Err(refusal) => msg.client.error(refused(refusal)), + } + } + + /// The client wrote the datagram into its socket's pipe and then sent this + /// request, so the bytes the request names are there: a pipe that holds + /// fewer is a client naming bytes it never wrote, and a read that would + /// wait for them would wait on that client. + fn udp_send_to(&mut self, node: &mut Node, now: Instant, msg: &Request) { + let Ok(req) = ipc::decode_payload::(msg.payload()) else { + msg.client.error(ERR_INVALID_INPUT); + return; + }; + let Some(socket) = self.datagram(req.socket_id) else { + msg.client.error(ERR_NOT_CONNECTED); + return; + }; + let mut payload = vec![0u8; usize::from(req.len)]; + // A datagram of no bytes put none in the pipe. + let read = if payload.is_empty() { Ok(0) } else { socket.from_client.read_nonblock(&mut payload) }; + match read { + Ok(read) if read == payload.len() => {} + Ok(_) | Err(SyscallError::WouldBlock) => { + msg.client.error(ERR_INVALID_INPUT); + return; + } + Err(_) => { + msg.client.error(ERR_OTHER); + return; + } + } + match node.udp_send_to(now, socket.id, Ipv4Addr::from(req.addr), req.port, &payload) { + Ok(()) => msg.client.result(&u32::from(req.len)), + Err(refusal) => msg.client.error(refused(refusal)), + } + } + + fn udp_recv_from(&mut self, node: &mut Node, now: Instant, msg: Request) { + let Ok(req) = ipc::decode_payload::(msg.payload()) else { + msg.client.error(ERR_INVALID_INPUT); + return; + }; + // A waiting client that hung up holds the socket's one wait for nobody. + if let Some(waiting) = self.receiving.remove(&req.socket_id) { + if !waiting.client.gone() { + self.receiving.insert(req.socket_id, waiting); + msg.client.error(ERR_RESOURCE_EXHAUSTED); + return; + } + } + let waiting = Receiving { client: msg.client, max_len: req.max_len }; + if let Some(waiting) = self.deliver(node, now, req.socket_id, waiting) { + self.receiving.insert(req.socket_id, waiting); + } + } + + /// Hands `waiting` its socket's oldest datagram, or hands `waiting` back + /// when none has arrived. + /// + /// **A datagram goes into the client's pipe whole, or its socket ends.** + /// The answer names a length, and a write takes what the pipe has room + /// for and cannot be taken back: a client reading that length out of a + /// pipe holding part of this datagram would splice the next one onto it. + fn deliver(&mut self, node: &mut Node, now: Instant, socket_id: u32, waiting: Receiving) -> Option { + let Some(socket) = self.datagram(socket_id) else { + waiting.client.error(ERR_NOT_CONNECTED); + return None; + }; + // The client's number bounds what it is handed and never what is + // allocated: no datagram is longer than [udp] delivers. + let room = usize::try_from(waiting.max_len).unwrap_or(usize::MAX).min(toyos_net_udp::limits::MAX_PAYLOAD); + let mut payload = vec![0u8; room]; + let datagram = match node.udp_recv_from(socket.id, &mut payload) { + Ok(Some(datagram)) => datagram, + Ok(None) => return Some(waiting), + Err(refusal) => { + waiting.client.error(refused(refusal)); + return None; + } + }; + let wrote = socket.to_client.write_nonblock(&payload[..datagram.len]); + if wrote == Ok(datagram.len) { + waiting.client.result(&UdpRecvResponse { + addr: datagram.source.octets(), + port: datagram.source_port.map_or(0, Port::get), + len: datagram.len as u16, + }); + return None; + } + say!( + "netstack: ending UDP socket {socket_id} — its receive pipe answered {wrote:?} to a {}-byte datagram", + datagram.len + ); + if let Some(Socket::Datagram(socket)) = self.ids.remove(&socket_id) { + self.end_datagram(node, now, socket); + } + waiting.client.error(ERR_CONNECTION_RESET); + None + } + + fn end_datagram(&mut self, node: &mut Node, now: Instant, socket: Datagram) { + if let Err(refusal) = node.udp_close(now, socket.id) { + unreachable!("the node refused the close of a datagram socket its table held: {refusal:?}"); + } + } + + fn udp_close(&mut self, node: &mut Node, now: Instant, msg: &Request) { + let Ok(req) = ipc::decode_payload::(msg.payload()) else { + msg.client.error(ERR_INVALID_INPUT); + return; + }; + if self.datagram(req.socket_id).is_some() { + if let Some(Socket::Datagram(socket)) = self.ids.remove(&req.socket_id) { + self.end_datagram(node, now, socket); + } + } + msg.client.done(); + } + + /// Starts resolving the name `msg` carries, or answers at once where + /// there is nothing to ask: an address written as one. + fn lookup(&mut self, node: &mut Node, now: Instant, msg: Request, draw: fn() -> u32) { + let Ok(hostname) = std::str::from_utf8(msg.payload()) else { + msg.client.error(ERR_INVALID_INPUT); + return; + }; + if let Ok(ip) = hostname.parse::() { + answer_lookup(&msg.client, &[ip.octets()]); + return; + } + let Ok(name) = toyos_dns::Name::parse(hostname) else { + msg.client.error(ERR_INVALID_INPUT); + return; + }; + match node.resolve(now, name.clone(), draw) { + Ok(id) => self.lookups.push((id, msg.client, name)), + Err(NotStarted::NotConnected) => msg.client.error(ERR_NOT_CONNECTED), + Err(NotStarted::ResourceExhausted) => msg.client.error(ERR_RESOURCE_EXHAUSTED), + } + } + + /// Everything the node has to say since the last call, to the log and to + /// the clients that waited for it, and the table entries of what the node + /// let go. + pub fn settle(&mut self, node: &mut Node, now: Instant) { + for (socket_id, waiting) in std::mem::take(&mut self.receiving) { + if let Some(waiting) = self.deliver(node, now, socket_id, waiting) { + self.receiving.insert(socket_id, waiting); + } + } + + let events: Vec = node.drain_stream_events().collect(); + for event in events { + let (id, answer) = match event { + StreamEvent::Connected { id, local } => (id, Ok(local)), + StreamEvent::Failed { id, failure } => (id, Err(connect_failed(failure))), + StreamEvent::TimedOut { id } => (id, Err(ERR_TIMED_OUT)), + // Its client closed it from another connection. + StreamEvent::Closed { id } => (id, Err(ERR_CONNECTION_REFUSED)), + StreamEvent::Cut { .. } => { + say!("netstack: resetting a connection — its client is gone and its peer took none of its bytes in 100 s"); + continue; + } + }; + let Some((client, socket_id)) = self.connecting.remove(&id) else { + unreachable!("the node answered a connect nobody waits for") + }; + match answer { + Ok(local) => client.result(&TcpConnectResponse { socket_id, local_port: local.get(), _pad: 0 }), + Err(code) => client.error(code), + } + } + + for resolved in node.take_resolved() { + let Some(at) = self.lookups.iter().position(|(id, ..)| *id == resolved.id) else { + unreachable!("the node ended a lookup nobody waits for") + }; + let (_, client, name) = self.lookups.swap_remove(at); + use toyos_dns::Failure as Dns; + match resolved.result { + Ok(addrs) => answer_lookup(&client, &addrs), + // The protocol's one answer for a name with no address, + // whether the name or only its address is missing. + Err(Ended::Failed(Dns::NoSuchName | Dns::NoAddress)) => answer_lookup(&client, &[]), + Err(Ended::Failed(Dns::TimedOut)) => client.error(ERR_TIMED_OUT), + // No query found a way out, or the lease the lookup asked + // under went: this machine is on no network that answers the + // name, which a lease clears. Neither is that word's (see + // `connect_failed`). + Err(Ended::Failed(Dns::Unreachable) | Ended::LeaseChanged) => client.error(ERR_NOT_CONNECTED), + Err(Ended::Failed(why @ (Dns::Truncated | Dns::ServerFailed(_) | Dns::TooManyAliases))) => { + say!("netstack: a lookup of {name} ended without an answer: {why:?}"); + client.error(ERR_OTHER); + } + Err(Ended::NoPort) => { + say!("netstack: a lookup of {name} ended with every dynamic port bound, none left for its next query"); + client.error(ERR_RESOURCE_EXHAUSTED); + } + } + } + + let ended: Vec<(ListenerId, WriteRefusal)> = node.drain_ended_listeners().collect(); + for (ended, refusal) in ended { + let found = self.ids.iter().find_map(|(socket_id, socket)| match socket { + Socket::Listener { id, .. } if *id == ended => Some(*socket_id), + Socket::Listener { .. } | Socket::Stream(_) | Socket::Datagram(_) => None, + }); + // None where its owner's close arrived in the pass that ended it. + let Some(socket_id) = found else { continue }; + self.ids.remove(&socket_id); + // Its owner gone is the ordinary end of a listener. + if refusal != WriteRefusal::Gone { + say!("netstack: closing listener {socket_id} — its notify pipe refused a wake: {refusal:?}"); + } + } + + let events: Vec = node.drain_events().collect(); + for event in events { + match event { + Event::Stack { refusal, suppressed: 0 } => say!("netstack: refused {refusal:?}"), + Event::Stack { refusal, suppressed } => say!("netstack: refused {refusal:?}, and {suppressed} more by its rule"), + Event::Dhcp { refusal, suppressed: 0 } => say!("netstack: DHCP: refused {refusal:?}"), + Event::Dhcp { refusal, suppressed } => say!("netstack: DHCP: refused {refusal:?}, and {suppressed} more by its rule"), + Event::Name(toyos_mdns::Event::Claimed) => { + say!("netstack: mDNS: no host answered for {HOSTNAME}.local; this machine answers as it") + } + Event::Name(toyos_mdns::Event::Lost) => say!( + "netstack: mDNS: another host answered for {HOSTNAME}.local; this machine answers to no name and asks for {HOSTNAME}.local again every {} s", + toyos_mdns::RETRY_MS / 1000 + ), + } + } + + // A stream the node let go holds neither end, and its id names + // nothing from then. + let (ids, by_stream) = (&mut self.ids, &mut self.by_stream); + self.ends.retain(|socket_id, ends| { + let held = ends.to_client.held().is_some() || ends.from_client.held().is_some(); + if !held { + by_stream.remove(&ends.stream); + if matches!(ids.get(socket_id), Some(Socket::Stream(_))) { + ids.remove(socket_id); + } + } + held + }); + } + + /// Asks the kernel about every pipe a pass could be owed for: a stream's + /// as the node's last pass left them, the wake pipe of each listener and + /// one pipe of each datagram socket for its owner's leaving, and the + /// connection of each client that waits for a lookup, which hangs up by + /// closing it. + pub fn watch(&self, node: &Node, poller: &Poller) { + for (stream, watch) in node.watches() { + let Some((socket_id, ends)) = self.by_stream.get(&stream).and_then(|id| Some((*id, self.ends.get(id)?))) else { + unreachable!("the node holds a stream whose ends netstack does not") + }; + let from = if watch.readable { READABLE } else { 0 } | if watch.writer { OTHER_END_GONE } else { 0 }; + if let (true, Some(pipe)) = (from != 0, ends.from_client.held()) { + poller.watch(&*pipe, from, TOKEN_FROM_CLIENT | u64::from(socket_id)); + } + let to = if watch.writable { WRITABLE } else { 0 } | if watch.reader { OTHER_END_GONE } else { 0 }; + if let (true, Some(pipe)) = (to != 0, ends.to_client.held()) { + poller.watch(&*pipe, to, TOKEN_TO_CLIENT | u64::from(socket_id)); + } + } + for (socket_id, socket) in &self.ids { + match socket { + Socket::Listener { wakes, .. } => { + if let Some(pipe) = wakes.held() { + poller.watch(&*pipe, OTHER_END_GONE, TOKEN_LISTENER | u64::from(*socket_id)); + } + } + Socket::Datagram(socket) => poller.watch(&socket.to_client, OTHER_END_GONE, TOKEN_DATAGRAM | u64::from(*socket_id)), + Socket::Stream(_) => {} + } + } + for (_, client, _) in &self.lookups { + poller.watch(&client.conn, READABLE, TOKEN_LOOKUP | u64::from(client.conn.as_handle().0)); + } + } + + /// What the kernel answered one of [`Self::watch`]'s watches, or `false` + /// for a token that is none of them. An answer about a socket that is + /// gone since, or a pipe end the node let go, says nothing: closing an + /// end ends its watch, and that end is an answer too. + pub fn answered(&mut self, node: &mut Node, now: Instant, token: u64, answer: Result) -> bool { + let socket_id = token as u32; + match token & TOKEN_KIND { + kind @ (TOKEN_FROM_CLIENT | TOKEN_TO_CLIENT) => { + let Some(ends) = self.ends.get(&socket_id) else { return true }; + let (end, held) = if kind == TOKEN_FROM_CLIENT { + (PipeEnd::FromClient, ends.from_client.held().is_some()) + } else { + (PipeEnd::ToClient, ends.to_client.held().is_some()) + }; + match answer { + _ if !held => {} + Err(why) => { + say!("netstack: resetting a connection — the kernel refused the watch of its {end:?} pipe: {why:?}"); + node.pipe_broken(now, ends.stream, end); + } + Ok(met) if met & OTHER_END_GONE != 0 => node.pipe_gone(now, ends.stream, end), + Ok(_) => self.bridge = true, + } + } + TOKEN_LISTENER => { + let gone = answer.map_or(true, |met| met & OTHER_END_GONE != 0); + if let (true, Some(id)) = (gone, self.listener(socket_id)) { + if let Err(why) = answer { + say!("netstack: closing listener {socket_id} — the kernel refused the watch of its notify pipe: {why:?}"); + } + self.ids.remove(&socket_id); + node.close_listener(now, id); + } + } + TOKEN_DATAGRAM => { + let gone = answer.map_or(true, |met| met & OTHER_END_GONE != 0); + if gone && self.datagram(socket_id).is_some() { + if let Err(why) = answer { + say!("netstack: ending UDP socket {socket_id} — the kernel refused the watch of its receive pipe: {why:?}"); + } + if let Some(Socket::Datagram(socket)) = self.ids.remove(&socket_id) { + self.end_datagram(node, now, socket); + } + } + } + TOKEN_LOOKUP => { + // A lookup whose client has left ends now: its sockets and + // its place are another client's. + let spoke = |client: &Client| u64::from(client.conn.as_handle().0) == u64::from(socket_id); + self.lookups.retain(|(id, client, _)| { + let left = spoke(client) && client.gone(); + if left { + node.let_go(now, *id); + } + !left + }); + } + _ => return false, + } + true + } + + /// The pass over the streams the last wake's answers asked for, once. + pub fn bridge(&mut self, node: &mut Node, now: Instant) { + if std::mem::take(&mut self.bridge) { + node.bridge(now); + } + } + + /// The socket table as `inspect` reads it: counts, and no endpoint, + /// because every client holding `netstack` can ask. + pub fn inspect(&self, node: &Node, snap: &mut Snapshot) { + let (mut streams, mut listeners, mut udp) = (0u32, 0u32, 0u32); + for socket in self.ids.values() { + match socket { + Socket::Stream(_) => streams += 1, + Socket::Listener { .. } => listeners += 1, + Socket::Datagram(_) => udp += 1, + } + } + snap.put("sockets.tcp", streams); + snap.put("sockets.listeners", listeners); + snap.put("sockets.udp", udp); + snap.put("piped.live", node.streams()); + snap.put("places.held", node.held()); + snap.put("places.max", self.places); + } +} diff --git a/userland/netstack/src/virtio_net.rs b/userland/netstack/src/virtio_net.rs index de692807e1..752063ca83 100644 --- a/userland/netstack/src/virtio_net.rs +++ b/userland/netstack/src/virtio_net.rs @@ -63,7 +63,7 @@ pub const RX_BUF_SIZE: usize = 4096; /// One descriptor per transmit buffer, and the two counts are one number for /// the same reason the receive side's are: buffer `i` is published at head `i` /// and nowhere else, so a head this driver holds names a buffer nothing else -/// is writing. Sixteen heads over one buffer would be sixteen aliases — smoltcp +/// is writing. Sixteen heads over one buffer would be sixteen aliases — the stack /// emits several frames per poll, and the device reads a descriptor whenever it /// likes. const TX_QUEUE_SIZE: u16 = 16;