diff --git a/Cargo.lock b/Cargo.lock index c8cce35baf7..13dd732cf84 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -3363,11 +3363,13 @@ dependencies = [ "smoltcp", "toyos", "toyos-abi", + "toyos-device-memory", "toyos-dns", "toyos-i219", "toyos-inspect", "toyos-mdns", "toyos-tco", + "toyos-virtio", ] [[package]] @@ -5798,6 +5800,10 @@ dependencies = [ "toyos-abi", ] +[[package]] +name = "toyos-device-memory" +version = "0.1.0" + [[package]] name = "toyos-dhcp" version = "0.1.0" @@ -5862,6 +5868,9 @@ version = "0.1.0" [[package]] name = "toyos-i219" version = "0.1.0" +dependencies = [ + "toyos-device-memory", +] [[package]] name = "toyos-inspect" @@ -6065,6 +6074,14 @@ dependencies = [ "toyos-bootmap", ] +[[package]] +name = "toyos-virtio" +version = "0.1.0" +dependencies = [ + "toyos-device-memory", + "toyos-untrusted", +] + [[package]] name = "toyos-wallclock" version = "0.1.0" diff --git a/Cargo.toml b/Cargo.toml index 788315a9e7a..8bd6a1e6d22 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -34,6 +34,7 @@ members = [ "toyos-blockring/transport", "toyos-bootmap", "toyos-cpuvuln", + "toyos-device-memory", "toyos-dhcp", "toyos-dns", "toyos-elf", @@ -66,6 +67,7 @@ members = [ "toyos-untrusted", "toyos-update", "toyos-userbound", + "toyos-virtio", "toyos-wallclock", "toyos-xhci", "toyos-xhci/sim", diff --git a/issues/a-userland-device-driver-has-no-shared-boundary-to-be-written-against.md b/issues/a-userland-device-driver-has-no-shared-boundary-to-be-written-against.md deleted file mode 100644 index b6b1eeadc5d..00000000000 --- a/issues/a-userland-device-driver-has-no-shared-boundary-to-be-written-against.md +++ /dev/null @@ -1,24 +0,0 @@ ---- -status: open -kind: finding -opened: 2026-09-08 ---- - -# A userland device driver has no shared boundary to be written against - -`toyos-i219` declares `Registers`, `Clock`, `DmaBuffers` and `Interrupts` in its -own `lib.rs`, and netd implements them in `userland/netd/src/i219.rs`. Each has -exactly one real implementation, so what they name is this driver's private -boundary rather than the one every userland driver is written against. - -Two of the four are already chip-shaped. `DmaBuffers::read`/`write` deal in -64-bit words because a legacy e1000 descriptor is two `u64` halves, which is not -the shape an xHCI TRB or an NVMe submission entry fits; `Registers` is 32-bit -dwords for the same reason. A second userland driver would either widen them or -declare its own four, and at that point there are two answers to one question. - -What is not yet decided is whether there is one boundary at all, or whether each -driver crate keeps its own and the shared part is only netd's `device.rs`. The -call is the owner's, and nothing is owed until a second userland driver exists -to be written against it — the virtio NIC beside `toyos-i219` in netd does not -use these traits at all. diff --git a/issues/every-driver-is-still-in-the-kernel.md b/issues/every-driver-is-still-in-the-kernel.md index ef745e3f4ef..dbcd2ca1226 100644 --- a/issues/every-driver-is-still-in-the-kernel.md +++ b/issues/every-driver-is-still-in-the-kernel.md @@ -37,6 +37,15 @@ What is left of the staged work: `virtio-sound` classes, their arms of `SYS_DEVICE_REG_READ`/`WRITE`, and `SYS_GPU_*`, which is an ABI change. GOP stays: it is memory the loader hands over, and the panic console paints it. + A virtio holder stands on `toyos-virtio`, whose first client is netstack's + NIC, and the second one owes that crate two things. The walk of the + capability list moves into it from `userland/netstack/src/virtio_net.rs`, + over a configuration read the caller passes in, which closes + `issues/the-virtio-capability-walk-reads-a-refused-configuration-read-as-zeros.md`. + And before a client ends its device on `UsedRefusal::Written` for a chain + the device only reads, whose bound is 0, what QEMU's device reports as + `len` on such a queue is measured: the NIC's transmit queue is the only + one read so far. 2. Done: **BAR sizing and re-assignment onto 2 MiB boundaries** is `pcidev::place_bar`, with the overlap refusal kept as the assertion that it worked rather than as the mechanism. diff --git a/issues/netstack-ends-for-the-boot-on-a-nic-it-cannot-drive-on.md b/issues/netstack-ends-for-the-boot-on-a-nic-it-cannot-drive-on.md new file mode 100644 index 00000000000..e35c645de6b --- /dev/null +++ b/issues/netstack-ends-for-the-boot-on-a-nic-it-cannot-drive-on.md @@ -0,0 +1,35 @@ +--- +status: open +kind: defect +opened: 2026-10-09 +--- + +# netstack ends for the boot on a NIC it cannot drive on + +`system.toml` gives `restart = true` to diskserver and fileserver and not to +netstack. netstack ends by name, "this NIC cannot be driven on", wherever a +card stops being one it can drive: a claim that refuses its interrupt read, an +Intel part that kept stranded descriptors past the driver's deadline +(`Card::begin_pass`, `userland/netstack/src/card.rs`), and a virtio device +whose used ring is not believed (`finished`, +`userland/netstack/src/virtio_net.rs`), which is any of a head past the table, +a head with no chain in flight, a length past a chain's writable bytes and a +used index past the available one. After any of them the machine has no +network until it boots again, and every program holding the `netstack` port +holds a port nobody serves. + +Ending is the right answer to the device: none of the four is something a +conforming device writes, and a ring read on past one loses frames silently. +What is missing is what comes after the end. + +A restart is not one line in `system.toml`. A second netstack on the same +address meets `issues/a-replacement-netstack-reuses-its-predecessors-ports.md`, +every claim it mints spends device addresses +(`issues/a-claim-spends-device-addresses-its-slot-never-gets-back.md`), and +section 3 of `issues/the-lan-is-not-yet-production-grade.md` splits the driver +from the stack so that a driver's end does not take the stack's connections. + +Owned by whoever takes that section. Exit: a guest test ends netstack's driver +and reads the network served again without a boot, or the owner rules that a +NIC whose device broke its ring stays down for the boot and this file becomes +that ruling. diff --git a/issues/the-virtio-capability-walk-reads-a-refused-configuration-read-as-zeros.md b/issues/the-virtio-capability-walk-reads-a-refused-configuration-read-as-zeros.md new file mode 100644 index 00000000000..a8cf43c6eaa --- /dev/null +++ b/issues/the-virtio-capability-walk-reads-a-refused-configuration-read-as-zeros.md @@ -0,0 +1,35 @@ +--- +status: open +kind: defect +opened: 2026-10-09 +--- + +# The virtio capability walk reads a refused configuration read as zeros + +`vendor_caps` (`userland/netstack/src/virtio_net.rs`) walks a function's +capability list through its claim, and two things in it believe more than they +were told: + +- A field of a vendor capability is read as + `dev.config_read(next + at, width).unwrap_or(0)`, so a read the kernel + refused becomes a `cfg_type`, `bar`, `offset` or `length` of 0. A capability + made of zeros is then refused or ignored by `toyos_virtio::pci::Layout::of` + under some other name (`MissingCap`, `TooShort`), and the kernel's word for + what happened is gone. +- The capabilities pointer and every `next` link are used as read. The *PCI + Local Bus Specification*, revision 3.0, section 6.7, reserves the bottom two + bits of each and has software mask them before using the value as an offset + (cited from memory: the document was not at hand when this was filed). + An unmasked pointer with either bit set is an offset no capability is at, + and the 32-bit reads at `next + 8` and on are then misaligned, which the + claim refuses and the first item turns into zeros. + +Neither reaches memory: the claim bounds every read to the function's own +configuration space (`config_space_is_bounded`). What they cost is a device +refused for the wrong reason. + +Owned by the second client of `toyos-virtio`, which moves the walk into the +crate (stage 1 of `issues/every-driver-is-still-in-the-kernel.md`). Exit: the +walk is the crate's, over a configuration read its caller passes in; a refused +read is a refusal by name and a link is masked before it is an offset; and a +host test is red against each of the two as they stand today. diff --git a/toyos-device-memory/Cargo.toml b/toyos-device-memory/Cargo.toml new file mode 100644 index 00000000000..60251c81328 --- /dev/null +++ b/toyos-device-memory/Cargo.toml @@ -0,0 +1,18 @@ +# A member of the host workspace (root `Cargo.toml`), like toyos-i219 and +# toyos-virtio, the two driver crates written against it: it is pure and has +# no dependencies. +# +# What lives here is two traits and their contract: the memory a userland +# driver and its device both reach, which is the whole of what a driver's +# decisions are separated from the instructions that carry them out by. A +# driver crate is generic over them and host-tested against a model of its +# device; the program that holds the claim implements them once, for every +# driver it runs. + +[package] +name = "toyos-device-memory" +description = "The memory a userland driver and its device share, as the boundary a driver crate is written against: a register window, a DMA grant and the two barriers that order a store against the device, pure." +version = "0.1.0" +edition = "2021" +license = "MIT OR Apache-2.0" +publish = false diff --git a/toyos-device-memory/src/lib.rs b/toyos-device-memory/src/lib.rs new file mode 100644 index 00000000000..3f0cec5c02d --- /dev/null +++ b/toyos-device-memory/src/lib.rs @@ -0,0 +1,96 @@ +//! The memory a userland driver and its device both reach, as the boundary a +//! driver crate is written against. +//! +//! Two traits, named by what they do and not by what device is behind them: +//! [`Registers`] is one access to a mapped BAR, and [`DmaBuffers`] is the +//! grant the device reaches through its IOMMU domain, with the two barriers +//! that order the driver's accesses to it against the device's. A driver crate +//! takes each as a generic parameter, resolved at compile time, and makes +//! every decision above them, where its host tests drive it against a model of +//! its device. +//! +//! **An implementation decides nothing**: each method is one volatile load or +//! store of the width its name says, or one barrier. The program that holds +//! the claim implements both once, for every driver it runs, so what an +//! architecture needs for a store to reach a device is one body. +//! +//! **Widths are in the names**, because the width of an access to a device is +//! a decision — a register file says which each field takes — and one a +//! driver states at the site rather than has inferred. Each is there because a +//! driver in the tree makes that access. +//! +//! Every word is little-endian, as a PCI device's registers and rings are and +//! as every machine ToyOS runs on is. +//! +//! What only one driver needs of its substrate — a clock, the claim's +//! interrupt record — is that driver crate's own trait until a second driver +//! needs the same. + +#![no_std] +#![forbid(unsafe_code)] + +/// One access to the BAR that holds a device's registers. +/// +/// **The driver bounds an offset before it reaches an implementation**: +/// against [`Self::bytes`], and aligned for the access's width. Volatile, +/// because the device writes the same bytes: a plain load of a status +/// register can be hoisted out of the loop that waits on it. +pub trait Registers { + /// Bytes the window covers. + fn bytes(&self) -> usize; + fn read8(&self, at: usize) -> u8; + fn read16(&self, at: usize) -> u16; + fn read32(&self, at: usize) -> u32; + fn write8(&self, at: usize, value: u8); + fn write16(&self, at: usize, value: u16); + fn write32(&self, at: usize, value: u32); +} + +/// The memory this process and its device both reach: one DMA grant. +/// +/// **Two addresses, because neither derives from the other**: an offset is +/// where the driver's own loads and stores land, and [`Self::device_addr`] is +/// what the device is told for the same byte. Never a physical address once a +/// unit translates for the function. +/// +/// Every access is volatile and aligned for its width: the device reads and +/// writes the same bytes. +/// +/// # The order of a driver's accesses against its device's +/// +/// A device acts on the grant when a register tells it to, or whenever it +/// likes, and on a weakly ordered machine neither a store nor a load of this +/// memory is ordered against another without a barrier. The two are here +/// rather than beside a doorbell because the order they impose is over *this* +/// memory: +/// +/// - **[`Self::publish`] before the device is pointed at what was stored.** +/// A driver stores what it publishes, calls `publish`, and only then makes +/// the store that tells the device of it — a register write, or the index +/// of a ring the device polls. +/// - **[`Self::observe`] after the load that said the device was done.** A +/// driver loads the word by which the device hands memory back, calls +/// `observe`, and only then loads what that word covers. +/// +/// **Neither orders a store against a later load of what the device wrote in +/// answer** — an index stored, then the device's suppression word loaded to +/// decide whether to notify it. No driver here makes that pair; the first +/// that does adds a method for it and does not reach for `publish`. +pub trait DmaBuffers { + /// Bytes in the grant. + fn bytes(&self) -> usize; + /// Where the device reaches byte `at`. + fn device_addr(&self, at: usize) -> u64; + fn read16(&self, at: usize) -> u16; + fn read32(&self, at: usize) -> u32; + fn read64(&self, at: usize) -> u64; + fn write16(&self, at: usize, value: u16); + fn write32(&self, at: usize, value: u32); + fn write64(&self, at: usize, value: u64); + /// One release barrier: every store made above this call is visible to + /// the device before any store made below it. + fn publish(&self); + /// One acquire barrier: every load made below this call sees memory at + /// least as new as the load made above it. + fn observe(&self); +} diff --git a/toyos-i219/Cargo.toml b/toyos-i219/Cargo.toml index c75edc8c8d0..6af17a51985 100644 --- a/toyos-i219/Cargo.toml +++ b/toyos-i219/Cargo.toml @@ -1,14 +1,15 @@ # A member of the host workspace (root `Cargo.toml`), like toyos-hda, toyos-pci -# and toyos-dma: it is pure, it has no dependencies, and its tests run on the -# host with no guest and no QEMU. +# and toyos-dma: it is pure, and its tests run on the host with no guest and no +# QEMU. # # What lives here is every decision the I219 driver makes — the descriptor # rings, the interrupt causes, the link — against four traits netstack implements -# one instruction deep. The device is on the far side of a trust boundary, so -# the numbers it writes into a descriptor have to be driven with values no -# working card would send far more often than a boot allows; and the machine -# that matters cannot be booted by a test at all, so the specification is the -# oracle and `stub.rs` is what it is written into. +# one instruction deep, two of them the boundary every userland driver shares. +# The device is on the far side of a trust boundary, so the numbers it writes +# into a descriptor have to be driven with values no working card would send +# far more often than a boot allows; and the machine that matters cannot be +# booted by a test at all, so the specification is the oracle and `stub.rs` is +# what it is written into. [package] name = "toyos-i219" @@ -17,3 +18,8 @@ version = "0.1.0" edition = "2021" license = "MIT OR Apache-2.0" publish = false + +[dependencies] +# `Registers` and `DmaBuffers`: the boundary every userland driver is written +# against, and the order of its two barriers. +toyos-device-memory = { path = "../toyos-device-memory" } diff --git a/toyos-i219/src/lib.rs b/toyos-i219/src/lib.rs index 9ed9f1df007..a8fce532e27 100644 --- a/toyos-i219/src/lib.rs +++ b/toyos-i219/src/lib.rs @@ -18,14 +18,25 @@ //! //! # The boundary //! -//! Four traits, named by what they do and not by what chip is behind them: -//! [`Registers`] is one memory-mapped access, [`Clock`] is one counter read, -//! [`DmaBuffers`] is the memory the device and this process share, and -//! [`Interrupts`] is what the claim answered. Each is a generic parameter -//! resolved at compile time — no trait object, no dispatch — and no +//! Four traits, named by what they do and not by what chip is behind them. +//! `Registers`, one memory-mapped access, and `DmaBuffers`, the memory the +//! device and this process share, are `toyos-device-memory`'s: the boundary +//! every userland driver is written against, where the order of the two +//! barriers is stated. [`Clock`] is one counter read and [`Interrupts`] is +//! what the claim answered, and both are this crate's. Each is a generic +//! parameter resolved at compile time — no trait object, no dispatch — and no //! implementation of one decides anything: every branch is above the boundary, //! in this crate, where the host tests it. //! +//! **Every register is a 32-bit access and every descriptor a 64-bit one**: +//! the register offsets are this crate's own constants, every one of them +//! under [`regs::REGISTER_BYTES`], which [`I219::open`] refuses a smaller +//! window than, and a legacy descriptor is two 64-bit halves. `publish` is +//! rung before a tail register is written, so a part that fetches on the +//! doorbell cannot fetch a descriptor that is half-written; `observe` is +//! taken after the load that found `DD` set, so a written-back length cannot +//! be read from before the write-back. +//! //! In netstack those four are the substrate's own: the mapped BAR, the monotonic //! clock, a `DmaRegion` in this function's IOMMU domain, and the claim's //! interrupt record. On the host they are `stub.rs`, which is the datasheet @@ -104,23 +115,9 @@ mod stub; #[cfg(test)] mod tests; -use regs::{cause, ctrl, ivar, rah, rctl, rx_desc, status, tctl, tx_desc, txdctl}; +use toyos_device_memory::{DmaBuffers, Registers}; -/// One memory-mapped register access. -/// -/// **The offsets are this crate's own constants**, every one of them under -/// [`regs::REGISTER_BYTES`], which [`I219::open`] refuses a smaller window -/// than. -pub trait Registers { - /// Bytes the window covers. - fn bytes(&self) -> usize; - /// One volatile 32-bit load. Volatile because the device writes the same - /// bytes, and a plain load of a status register can be hoisted out of the - /// loop that waits on it. - fn read(&self, reg: usize) -> u32; - /// One volatile 32-bit store. - fn write(&self, reg: usize, value: u32); -} +use regs::{cause, ctrl, ivar, rah, rctl, rx_desc, status, tctl, tx_desc, txdctl}; /// The clock, and the one way this driver gives the processor away. pub trait Clock { @@ -135,39 +132,6 @@ pub trait Clock { fn pause(&self, nanos: u64); } -/// The memory this process and the device both reach. -/// -/// **Two addresses, because neither derives from the other**: the offsets this -/// driver computes are where its own loads and stores land, and -/// [`Self::device_addr`] is what a descriptor must carry for the device to -/// reach the same bytes through the unit. -/// -/// The two fences are here rather than beside the doorbell because the -/// ordering they impose is over *this* memory: the architecture's rules about -/// when a store to DMA memory becomes visible to a device live in the -/// implementation and nowhere above it. -pub trait DmaBuffers { - /// Bytes in the grant. - fn bytes(&self) -> usize; - /// Where the device reaches byte `at`. Never a physical address once a - /// unit translates for this function. - fn device_addr(&self, at: usize) -> u64; - /// One volatile aligned 64-bit load of the grant. - fn read(&self, at: usize) -> u64; - /// One volatile aligned 64-bit store into the grant. - fn write(&self, at: usize, word: u64); - /// One release barrier: every store made above this call is visible to the - /// device before any store made below it. Rung before a tail register is, - /// so a device that fetches on the doorbell cannot fetch a descriptor that - /// is half-written. - fn publish(&self); - /// One acquire barrier: every load made below this call sees memory at - /// least as new as the load above it that found `DD` set. Taken before a - /// descriptor's other fields are read, so a written-back length cannot be - /// read from before the write-back. - fn observe(&self); -} - /// What the claim answered when its interrupt record was read. /// /// **The implementation hands the answer over and decides nothing about it**: @@ -585,11 +549,11 @@ pub(crate) enum Whole { /// function where it advertises a reset; this is the driver not trusting that /// it did. Register writes only, so it is safe with mastering off. pub fn quiesce(regs: &R) { - regs.write(regs::IMC, u32::MAX); - let rctl = regs.read(regs::RCTL); - regs.write(regs::RCTL, rctl & !rctl::EN); - let tctl = regs.read(regs::TCTL); - regs.write(regs::TCTL, tctl & !tctl::EN); + regs.write32(regs::IMC, u32::MAX); + let rctl = regs.read32(regs::RCTL); + regs.write32(regs::RCTL, rctl & !rctl::EN); + let tctl = regs.read32(regs::TCTL); + regs.write32(regs::TCTL, tctl & !tctl::EN); } /// §4.6.1's reset, with every interrupt masked on both sides of it. @@ -601,7 +565,7 @@ fn reset( // A window nothing decodes answers ones on every access, and a driver // that went on would read a MAC address of `ff:ff:ff:ff:ff:ff` out of // it and never say why nothing arrived. - if regs.read(regs::STATUS) == u32::MAX { + if regs.read32(regs::STATUS) == u32::MAX { return Err(Refusal::Dead); } @@ -609,8 +573,8 @@ fn reset( // while the register file is being rebuilt. `ICR` is read afterwards // because §10.2.4.1's case 1 — mask all — is the one arm that clears // it unconditionally. - regs.write(regs::IMC, u32::MAX); - let _ = regs.read(regs::ICR); + regs.write32(regs::IMC, u32::MAX); + let _ = regs.read32(regs::ICR); // §10.2.2.1: "Before issuing this reset, software has to insure that Tx // and Rx processes are stopped by following the procedure described in @@ -620,11 +584,11 @@ fn reset( // section sanctions it: "the software device driver might time out if // the PCIe Master Enable Status bit is not cleared within a given // time", and the reset clears the bit either way. - let held = regs.read(regs::CTRL); - regs.write(regs::CTRL, held | ctrl::GIO_MASTER_DISABLE); + let held = regs.read32(regs::CTRL); + regs.write32(regs::CTRL, held | ctrl::GIO_MASTER_DISABLE); let started = clock.nanos(); let master_quiet = loop { - if regs.read(regs::STATUS) & status::GIO_MASTER_ENABLE == 0 { + if regs.read32(regs::STATUS) & status::GIO_MASTER_ENABLE == 0 { break true; } if clock.nanos().saturating_sub(started) >= MASTER_QUIESCE_DEADLINE_NANOS { @@ -636,14 +600,14 @@ fn reset( // reserved bits are documented as "Set to 1b" and one as "must be set // to 1b" — a driver that wrote a value it composed itself would clear // them. - let held = regs.read(regs::CTRL); + let held = regs.read32(regs::CTRL); // Held until the reset has finished and the PHY's configuration with it: // the drop is paced from the reset's own write, and the waits below are // measured from that write too. let mut flag = None; let (reset_at, full) = match whole { Whole::MacAlone => { - regs.write(regs::CTRL, held | ctrl::RST); + regs.write32(regs::CTRL, held | ctrl::RST); // The write is the event §10.2.2.1's two bounds and §9.2's delay // before the first MDIO access are measured from. let reset_at = clock.nanos(); @@ -656,7 +620,7 @@ fn reset( (reset_at, None) } Whole::WithThePhy { after } => { - let firmware = regs.read(regs::FWSM); + let firmware = regs.read32(regs::FWSM); let word = wake::reset_word(held, firmware); // Under §8.2.4's flag, which arbitrates the CSRs this MAC shares // with its firmware — and a flag that would not come is no reason @@ -666,7 +630,7 @@ fn reset( let claimed = phy::Owned::claim(regs, clock, after); match &claimed { Ok(mdi) => mdi.write_paced(regs::CTRL, word), - Err(_) => regs.write(regs::CTRL, word), + Err(_) => regs.write32(regs::CTRL, word), } let reset_at = clock.nanos(); // Nothing is touched while the part resets both ends of its @@ -682,7 +646,7 @@ fn reset( } }; loop { - if regs.read(regs::CTRL) & ctrl::RST == 0 { + if regs.read32(regs::CTRL) & ctrl::RST == 0 { break; } let waited = clock.nanos().saturating_sub(reset_at); @@ -703,8 +667,8 @@ fn reset( }); // Again after the reset: §4.6.1 keeps them masked until the rings // exist, and the reset itself is an event the part may have recorded. - regs.write(regs::IMC, u32::MAX); - let _ = regs.read(regs::ICR); + regs.write32(regs::IMC, u32::MAX); + let _ = regs.read32(regs::ICR); Ok(Reset { master_quiet, at: reset_at, full, released }) } @@ -712,7 +676,7 @@ fn reset( /// where [`wake::PHY_CONFIGURED_DEADLINE_NANOS`] ran out first. fn phy_configured(regs: &R, clock: &C, since: u64) -> Option { loop { - let done = regs.read(regs::STATUS) & status::PHY_CONFIGURED != 0; + let done = regs.read32(regs::STATUS) & status::PHY_CONFIGURED != 0; let waited = clock.nanos().saturating_sub(since); if done { return Some(waited); @@ -869,8 +833,8 @@ impl I219 { // §10.2.5.23: the reset reloads entry 0 from the NVM, so this is read // after it and not before. - let low = regs.read(regs::RAL0); - let high = regs.read(regs::RAH0); + let low = regs.read32(regs::RAL0); + let high = regs.read32(regs::RAH0); if high & rah::AV == 0 { return Err(Refusal::NoStationAddress); } @@ -898,7 +862,7 @@ impl I219 { Part::I219 => regs::MTA_DWORDS_PCH, }; for entry in 0..table { - regs.write(regs::MTA + entry * 4, 0); + regs.write32(regs::MTA + entry * 4, 0); } // §4.6.3.1: "Refer to the PHY documentation for the initialization and @@ -919,7 +883,7 @@ impl I219 { // goes with them and is not preserved: a part that came out of the // reset still blocking master requests would fetch no descriptor and // write back no frame, and nothing else in this bring-up would say so. - let held = regs.read(regs::CTRL); + let held = regs.read32(regs::CTRL); let wanted = (held & !(ctrl::GIO_MASTER_DISABLE | ctrl::PHY_RST @@ -931,12 +895,12 @@ impl I219 { | ctrl::TFCE | ctrl::VME)) | ctrl::SLU; - regs.write(regs::CTRL, wanted); + regs.write32(regs::CTRL, wanted); // §10.2.4.7: "If any bits are set in EIAC, the ICR register should not // be read" — and this driver reads it, so auto-clear stays off and // every cause is acknowledged by the write-back in `begin_pass`. - regs.write(regs::EIAC, 0); + regs.write32(regs::EIAC, 0); let mut nic = Self { part, @@ -966,37 +930,37 @@ impl I219 { // "in MSI-X mode" and says nothing about what a part outside that mode // answers — so a part that does not take the write is refused here // rather than driven on a guess about which interrupt it would raise. - nic.regs.write(regs::IVAR, ivar::ALL_ON_VECTOR_ZERO); + nic.regs.write32(regs::IVAR, ivar::ALL_ON_VECTOR_ZERO); nic.accepted(regs::IVAR, ivar::ALL_ON_VECTOR_ZERO)?; // No moderation on either side: §10.2.4.2's throttle and the two // receive timers all hold an interrupt back, and what this driver // waits on is the frame that has already arrived. - nic.regs.write(regs::ITR, 0); - nic.regs.write(regs::RDTR, 0); - nic.regs.write(regs::RADV, 0); - nic.regs.write(regs::TIDV, 0); - nic.regs.write(regs::TADV, 0); + nic.regs.write32(regs::ITR, 0); + nic.regs.write32(regs::RDTR, 0); + nic.regs.write32(regs::RADV, 0); + nic.regs.write32(regs::TIDV, 0); + nic.regs.write32(regs::TADV, 0); nic.arm_rx_ring(); nic.arm_tx_ring(); // §4.6.6, in its order: the write-back policy, the gap, then the // transmitter. - nic.regs.write(regs::TXDCTL, txdctl::SUGGESTED); - nic.regs.write(regs::TIPG, regs::TIPG_DEFAULT); - nic.regs.write(regs::TCTL, TX_CONTROL); + nic.regs.write32(regs::TXDCTL, txdctl::SUGGESTED); + nic.regs.write32(regs::TIPG, regs::TIPG_DEFAULT); + nic.regs.write32(regs::TCTL, TX_CONTROL); nic.accepted(regs::TCTL, TX_CONTROL)?; // §4.6.5.1: the receiver last, "only after all other setup is // accomplished". let rx = rctl::EN | rctl::BAM | rctl::SECRC | rctl::BSIZE_2048; - nic.regs.write(regs::RCTL, rx); + nic.regs.write32(regs::RCTL, rx); nic.accepted(regs::RCTL, rx)?; // §4.6.5: and only now the mask, so no cause can arrive before there // is a ring to answer it with. - nic.regs.write(regs::IMS, cause::ENABLED | cause::ENABLED_MSIX); + nic.regs.write32(regs::IMS, cause::ENABLED | cause::ENABLED_MSIX); nic.opened_at = nic.clock.nanos(); nic.refresh_link(); @@ -1013,13 +977,13 @@ impl I219 { Part::I219 => regs::mta_bit_pch(group), }; let at = regs::MTA + dword * 4; - self.regs.write(at, self.regs.read(at) | 1 << bit); + self.regs.write32(at, self.regs.read32(at) | 1 << bit); } /// A register wrote what it was told, so a window that is not this register /// file is refused here instead of looking like a dead network. fn accepted(&self, reg: usize, wrote: u32) -> Result<(), Refusal> { - let read = self.regs.read(reg); + let read = self.regs.read32(reg); if read & wrote == wrote { Ok(()) } else { @@ -1039,26 +1003,26 @@ impl I219 { } let base = self.dma.device_addr(OFF_RX_RING); self.dma.publish(); - self.regs.write(regs::RDBAL, base as u32); - self.regs.write(regs::RDBAH, (base >> 32) as u32); - self.regs.write(regs::RDLEN, (RX_RING * rx_desc::BYTES) as u32); - self.regs.write(regs::RDH, 0); - self.regs.write(regs::RDT, self.rx_tail as u32); + self.regs.write32(regs::RDBAL, base as u32); + self.regs.write32(regs::RDBAH, (base >> 32) as u32); + self.regs.write32(regs::RDLEN, (RX_RING * rx_desc::BYTES) as u32); + self.regs.write32(regs::RDH, 0); + self.regs.write32(regs::RDT, self.rx_tail as u32); } fn arm_tx_ring(&mut self) { for index in 0..TX_RING { let at = OFF_TX_RING + index * tx_desc::BYTES; - self.dma.write(at, 0); - self.dma.write(at + 8, 0); + self.dma.write64(at, 0); + self.dma.write64(at + 8, 0); } let base = self.dma.device_addr(OFF_TX_RING); self.dma.publish(); - self.regs.write(regs::TDBAL, base as u32); - self.regs.write(regs::TDBAH, (base >> 32) as u32); - self.regs.write(regs::TDLEN, (TX_RING * tx_desc::BYTES) as u32); - self.regs.write(regs::TDH, 0); - self.regs.write(regs::TDT, 0); + self.regs.write32(regs::TDBAL, base as u32); + self.regs.write32(regs::TDBAH, (base >> 32) as u32); + self.regs.write32(regs::TDLEN, (TX_RING * tx_desc::BYTES) as u32); + self.regs.write32(regs::TDH, 0); + self.regs.write32(regs::TDT, 0); } /// Write descriptor `index` back as an empty buffer the device may fill. @@ -1070,8 +1034,8 @@ impl I219 { fn publish_rx(&mut self, index: usize) { let at = OFF_RX_RING + index * rx_desc::BYTES; let buffer = OFF_RX_BUFS + index * RX_BUF_BYTES; - self.dma.write(at + 8, 0); - self.dma.write(at, self.dma.device_addr(buffer)); + self.dma.write64(at + 8, 0); + self.dma.write64(at, self.dma.device_addr(buffer)); } pub fn mac(&self) -> [u8; 6] { @@ -1103,7 +1067,7 @@ impl I219 { /// what separates "nothing arrived" from "something arrived and went /// nowhere" on a boot with no other record. pub fn wire(&mut self) -> Wire { - let take = |reg| u64::from(self.regs.read(reg)); + let take = |reg| u64::from(self.regs.read32(reg)); let read = Wire { sent: take(regs::GPTC), received: take(regs::GPRC), @@ -1152,7 +1116,7 @@ impl I219 { // again in [`Self::wake_on_room`]. let waited = core::mem::take(&mut self.tx_wake); if waited { - self.regs.write(regs::IMC, self.part.tx_done()); + self.regs.write32(regs::IMC, self.part.tx_done()); } // Once, and written back. §10.2.4.1's case 3 says a read with no @@ -1161,10 +1125,10 @@ impl I219 { // for ever; and §7.4.5's "Write to Clear" is the arm that always // works. `INT_ASSERTED` is left out because the same section says // writing it has no effect and it clears when its causes do. - let causes = self.regs.read(regs::ICR); + let causes = self.regs.read32(regs::ICR); let acknowledged = causes & !cause::INT_ASSERTED; if acknowledged != 0 { - self.regs.write(regs::ICR, acknowledged); + self.regs.write32(regs::ICR, acknowledged); } if causes & cause::RXO != 0 { @@ -1234,7 +1198,7 @@ impl I219 { /// Re-read `STATUS` and take what it says about the link. fn refresh_link(&mut self) { - let status = self.regs.read(regs::STATUS); + let status = self.regs.read32(regs::STATUS); self.link = if status & status::LU != 0 { Link::Up { speed: Speed::in_status(status), @@ -1261,7 +1225,7 @@ impl I219 { } let index = self.rx_next; let at = OFF_RX_RING + index * rx_desc::BYTES; - let word = self.dma.read(at + 8); + let word = self.dma.read64(at + 8); let status = ((word >> rx_desc::STATUS_SHIFT) & rx_desc::BYTE_MASK) as u8; if status & rx_desc::status::DD == 0 { return None; @@ -1269,7 +1233,7 @@ impl I219 { // §7.1.7.1: the descriptor was written back in a batch, so its // other fields are read only after the load that found `DD`. self.dma.observe(); - let word = self.dma.read(at + 8); + let word = self.dma.read64(at + 8); self.rx_next = (index + 1) % RX_RING; self.rx_budget -= 1; match parse_rx(word) { @@ -1328,7 +1292,7 @@ impl I219 { if next == self.rx_next { break; } - if self.dma.read(OFF_RX_RING + next * rx_desc::BYTES + 8) != 0 { + if self.dma.read64(OFF_RX_RING + next * rx_desc::BYTES + 8) != 0 { break; } tail = next; @@ -1340,7 +1304,7 @@ impl I219 { // §7.1.8: the device fetches on the tail write, so the descriptors have // to be there before the write is. self.dma.publish(); - self.regs.write(regs::RDT, tail as u32); + self.regs.write32(regs::RDT, tail as u32); } /// How many frames the transmit ring takes now, every descriptor the part @@ -1385,7 +1349,7 @@ impl I219 { } self.counters.tx_full = self.counters.tx_full.saturating_add(1); if !self.tx_wake { - self.regs.write(regs::IMS, self.part.tx_done()); + self.regs.write32(regs::IMS, self.part.tx_done()); self.tx_wake = true; self.counters.tx_wake_armed = self.counters.tx_wake_armed.saturating_add(1); } @@ -1414,14 +1378,14 @@ impl I219 { let command = (tx_desc::cmd::ONE_FRAME as u64) << tx_desc::CMD_SHIFT; // The status nibble is zeroed with the rest of the word, so the `DD` // this driver waits for is one the device wrote. - self.dma.write(at + 8, (slot.len as u64 & tx_desc::LENGTH_MASK) | command); - self.dma.write(at, self.dma.device_addr(slot.at)); + self.dma.write64(at + 8, (slot.len as u64 & tx_desc::LENGTH_MASK) | command); + self.dma.write64(at, self.dma.device_addr(slot.at)); self.tx_next = (slot.index + 1) % TX_RING; // §7.2.4.1: "The 82574 NEVER fetches descriptors beyond the descriptor // tail pointer", so the descriptor and the frame both have to be // visible before the tail moves over them. self.dma.publish(); - self.regs.write(regs::TDT, self.tx_next as u32); + self.regs.write32(regs::TDT, self.tx_next as u32); } /// Take back every transmit descriptor the device has written back. @@ -1432,7 +1396,7 @@ impl I219 { fn reclaim_tx(&mut self) { while self.tx_clean != self.tx_next { let at = OFF_TX_RING + self.tx_clean * tx_desc::BYTES; - let word = self.dma.read(at + 8); + let word = self.dma.read64(at + 8); let status = ((word >> tx_desc::STATUS_SHIFT) & tx_desc::STATUS_MASK) as u8; if status & tx_desc::STATUS_DD == 0 { return; diff --git a/toyos-i219/src/pch.rs b/toyos-i219/src/pch.rs index f31d8b72906..17c77fa5131 100644 --- a/toyos-i219/src/pch.rs +++ b/toyos-i219/src/pch.rs @@ -31,8 +31,8 @@ use crate::Registers; /// One read-modify-write: `set` raised and `clear` lowered, every other bit as /// the part holds it. fn modify(regs: &R, reg: usize, set: u32, clear: u32) { - let held = regs.read(reg); - regs.write(reg, (held | set) & !clear); + let held = regs.read32(reg); + regs.write32(reg, (held | set) & !clear); } /// Everything this module owes the part, after its reset and before its rings. @@ -54,8 +54,8 @@ pub(crate) fn prepare(regs: &R) { /// the read, and the word composed from it would set every field of the /// register. pub(crate) fn release(regs: &R) { - let held = regs.read(regs::CTRL_EXT); + let held = regs.read32(regs::CTRL_EXT); if held != u32::MAX { - regs.write(regs::CTRL_EXT, held & !ctrl_ext::DRIVER_HOLDS_THE_FUNCTION); + regs.write32(regs::CTRL_EXT, held & !ctrl_ext::DRIVER_HOLDS_THE_FUNCTION); } } diff --git a/toyos-i219/src/phy.rs b/toyos-i219/src/phy.rs index c8b7b9b5450..1410254855e 100644 --- a/toyos-i219/src/phy.rs +++ b/toyos-i219/src/phy.rs @@ -482,14 +482,14 @@ impl<'a, R: Registers, C: Clock> Arbitration<'a, R, C> { /// poll is the one exception (it is the transaction, not the arbitration). fn read_at(&self, reg: usize) -> u32 { self.pace(); - let reading = self.regs.read(reg); + let reading = self.regs.read32(reg); self.last.set(Some(self.clock.nanos())); reading } fn write_at(&self, reg: usize, value: u32) { self.pace(); - self.regs.write(reg, value); + self.regs.write32(reg, value); self.last.set(Some(self.clock.nanos())); } @@ -659,7 +659,7 @@ impl<'a, R: Registers, C: Clock> Owned<'a, R, C> { // and a part that never clears it is refused with nothing written. let waiting_since = self.mdio.clock.nanos(); loop { - if self.mdio.regs.read(regs::MDIC) & mdic::WAIT == 0 { + if self.mdio.regs.read32(regs::MDIC) & mdic::WAIT == 0 { break; } let waited = self.mdio.clock.nanos().saturating_sub(waiting_since); @@ -668,10 +668,10 @@ impl<'a, R: Registers, C: Clock> Owned<'a, R, C> { } } let command = command(phy, reg, op, data); - self.mdio.regs.write(regs::MDIC, command); + self.mdio.regs.write32(regs::MDIC, command); let started = self.mdio.clock.nanos(); loop { - let answer = self.mdio.regs.read(regs::MDIC); + let answer = self.mdio.regs.read32(regs::MDIC); if answer & mdic::ERROR != 0 { return Err(PhyRefusal::MdiError { phy, reg }); } @@ -776,7 +776,7 @@ impl<'a, R: Registers, C: Clock> Owned<'a, R, C> { /// cycle that never ends stops the ask**, because the interconnect that did /// not carry one transaction carries none, whatever address it names. pub(crate) fn ask(&self) -> Answer { - let mdic = || self.mdio.regs.read(regs::MDIC); + let mdic = || self.mdio.regs.read32(regs::MDIC); let mut last = 0; for addr in [SPECIFIC, GENERAL] { let high = self.read(addr, reg::IDENTIFIER_HIGH); diff --git a/toyos-i219/src/stub.rs b/toyos-i219/src/stub.rs index 4d14562c8a9..85b93426c99 100644 --- a/toyos-i219/src/stub.rs +++ b/toyos-i219/src/stub.rs @@ -12,6 +12,12 @@ //! Every run takes a seed; every assertion this file makes prints it; and the //! seed reproduces the run exactly. //! +//! **The order of the driver's accesses is recorded, and judged by a test.** +//! This model has one memory and reads it whole, so a barrier left out changes +//! nothing it does; [`Access`] is every store, load and barrier the driver +//! made on the grant and every register it wrote, in one trace. A frame's +//! bytes are its caller's accesses and are not in it. +//! //! **Two parts, one model.** [`Nic::new`] is the 82574 the QEMU arm runs //! against; §10.2.2.7 addresses its PHY as "1 = Gigabit PHY. 2 = PCIe PHY" and //! none of that register set is modelled here, so an `MDIC` access to it is an @@ -57,6 +63,21 @@ use crate::{phy, wake, Clock, DmaBuffers, Interrupts, Part, Registers}; /// mistake a refusal instead of a pass. pub const DEVICE_BASE: u64 = 0x0000_0001_0000_0000; +/// What the driver did to the part, in the order it did it. +#[derive(Clone, Copy, PartialEq, Eq, Debug)] +pub enum Access { + /// A store into the grant. + Store { at: usize }, + /// A load from the grant, and the word it found. + Load { at: usize, word: u64 }, + /// [`DmaBuffers::publish`]. + Publish, + /// [`DmaBuffers::observe`]. + Observe, + /// A write of a register. + Write { reg: usize }, +} + /// The station address the modelled NVM holds (§10.2.5.23: entry 0 is loaded /// from the `IA` field after a reset). pub const NVM_MAC: [u8; 6] = [0x54, 0xbf, 0x64, 0x11, 0x22, 0x33]; @@ -613,6 +634,7 @@ struct Model { /// Every register offset the driver has written, so a test about what a /// path does *not* write can say so. written: BTreeSet, + trace: Vec, /// Every restart of auto-negotiation the driver wrote, by the register it /// wrote it to, in order. restarts: Vec, @@ -716,6 +738,7 @@ impl Model { messages: 0, not_decoding: None, written: BTreeSet::new(), + trace: Vec::new(), restarts: Vec::new(), held: false, arbitration_at: Vec::new(), @@ -2232,6 +2255,11 @@ impl Nic { self.0.borrow().written.clone() } + /// Everything the driver did to the grant and every register it wrote. + pub fn trace(&self) -> Vec { + self.0.borrow().trace.clone() + } + /// Nothing in the window decodes `reg`, so reads of it answer ones. pub fn window_does_not_decode(&self, reg: usize) { self.0.borrow_mut().not_decoding = Some(reg); @@ -2451,12 +2479,28 @@ impl Registers for Bar { regs::REGISTER_BYTES } - fn read(&self, reg: usize) -> u32 { + fn read32(&self, reg: usize) -> u32 { self.0.borrow_mut().read(reg) } - fn write(&self, reg: usize, value: u32) { - self.0.borrow_mut().write(reg, value); + fn write32(&self, reg: usize, value: u32) { + let mut model = self.0.borrow_mut(); + model.trace.push(Access::Write { reg }); + model.write(reg, value); + } + + // Every register of this part is thirty-two bits, and is reached as one. + fn read8(&self, at: usize) -> u8 { + panic!("a 32-bit register was read at {at:#x} as 8 bits") + } + fn read16(&self, at: usize) -> u16 { + panic!("a 32-bit register was read at {at:#x} as 16 bits") + } + fn write8(&self, at: usize, _: u8) { + panic!("a 32-bit register was written at {at:#x} as 8 bits") + } + fn write16(&self, at: usize, _: u16) { + panic!("a 32-bit register was written at {at:#x} as 16 bits") } } @@ -2494,18 +2538,40 @@ impl DmaBuffers for Grant { DEVICE_BASE + at as u64 } - fn read(&self, at: usize) -> u64 { - self.0.borrow().desc_read(at) + fn read64(&self, at: usize) -> u64 { + let mut model = self.0.borrow_mut(); + let word = model.desc_read(at); + model.trace.push(Access::Load { at, word }); + word } - fn write(&self, at: usize, word: u64) { - self.0.borrow_mut().desc_write(at, word); + fn write64(&self, at: usize, word: u64) { + let mut model = self.0.borrow_mut(); + model.trace.push(Access::Store { at }); + model.desc_write(at, word); } - // The host has one memory and one observer of it, so the two barriers are - // where they are on a machine — at the boundary — and cost nothing here. - fn publish(&self) {} - fn observe(&self) {} + // A legacy descriptor is two sixty-four-bit halves, and is reached as them. + fn read16(&self, at: usize) -> u16 { + panic!("a descriptor was read at {at:#x} as 16 bits") + } + fn read32(&self, at: usize) -> u32 { + panic!("a descriptor was read at {at:#x} as 32 bits") + } + fn write16(&self, at: usize, _: u16) { + panic!("a descriptor was written at {at:#x} as 16 bits") + } + fn write32(&self, at: usize, _: u32) { + panic!("a descriptor was written at {at:#x} as 32 bits") + } + + fn publish(&self) { + self.0.borrow_mut().trace.push(Access::Publish); + } + + fn observe(&self) { + self.0.borrow_mut().trace.push(Access::Observe); + } } pub struct Line(Rc>); diff --git a/toyos-i219/src/tests.rs b/toyos-i219/src/tests.rs index dd98a9a7ba1..175b87d24d7 100644 --- a/toyos-i219/src/tests.rs +++ b/toyos-i219/src/tests.rs @@ -12,7 +12,7 @@ use std::{format, vec}; use crate::phy as toyos_phy; use crate::phy::{Others, PhyRefusal}; use crate::regs::{self, cause, ctrl, extcnf, ivar, rctl, rx_desc, tctl, tx_desc}; -use crate::stub::{Nic, Permits, Unanswered, NVM_MAC}; +use crate::stub::{Access, Nic, Permits, Unanswered, NVM_MAC}; use crate::*; type Driver = I219; @@ -154,10 +154,22 @@ fn a_window_or_grant_too_small_is_refused() { fn bytes(&self) -> usize { self.0 } - fn read(&self, _: usize) -> u32 { + fn read8(&self, _: usize) -> u8 { panic!("a refused window was read") } - fn write(&self, _: usize, _: u32) { + fn read16(&self, _: usize) -> u16 { + panic!("a refused window was read") + } + fn read32(&self, _: usize) -> u32 { + panic!("a refused window was read") + } + fn write8(&self, _: usize, _: u8) { + panic!("a refused window was written") + } + fn write16(&self, _: usize, _: u16) { + panic!("a refused window was written") + } + fn write32(&self, _: usize, _: u32) { panic!("a refused window was written") } } @@ -178,10 +190,22 @@ fn a_window_or_grant_too_small_is_refused() { fn device_addr(&self, _: usize) -> u64 { panic!("a refused grant was addressed") } - fn read(&self, _: usize) -> u64 { + fn read16(&self, at: usize) -> u16 { + panic!("a refused grant was read at {at:#x} as 16 bits") + } + fn read32(&self, at: usize) -> u32 { + panic!("a refused grant was read at {at:#x} as 32 bits") + } + fn write16(&self, at: usize, _: u16) { + panic!("a refused grant was written at {at:#x} as 16 bits") + } + fn write32(&self, at: usize, _: u32) { + panic!("a refused grant was written at {at:#x} as 32 bits") + } + fn read64(&self, _: usize) -> u64 { panic!("a refused grant was read") } - fn write(&self, _: usize, _: u64) { + fn write64(&self, _: usize, _: u64) { panic!("a refused grant was written") } fn publish(&self) {} @@ -232,10 +256,22 @@ fn a_window_that_reads_ones_is_refused() { fn bytes(&self) -> usize { regs::REGISTER_BYTES } - fn read(&self, _: usize) -> u32 { + fn read8(&self, at: usize) -> u8 { + panic!("a 32-bit register was read at {at:#x} as 8 bits") + } + fn read16(&self, at: usize) -> u16 { + panic!("a 32-bit register was read at {at:#x} as 16 bits") + } + fn write8(&self, at: usize, _: u8) { + panic!("a 32-bit register was written at {at:#x} as 8 bits") + } + fn write16(&self, at: usize, _: u16) { + panic!("a 32-bit register was written at {at:#x} as 16 bits") + } + fn read32(&self, _: usize) -> u32 { u32::MAX } - fn write(&self, _: usize, _: u32) {} + fn write32(&self, _: usize, _: u32) {} } struct NoClock; impl Clock for NoClock { @@ -1294,6 +1330,123 @@ fn a_seeded_workload_loses_nothing_and_invents_nothing() { assert_eq!(counters.split, 0, "{}", nic.because("a frame was split")); } } + +// --- the order of the driver's accesses against the part's --- + +/// Frames both ways on `part` with every latitude on, and everything the +/// driver did to the grant and the registers while they moved. +fn a_worked_trace(seed: u64, part: Part) -> (Vec, usize, usize) { + let nic = Nic::with(seed, part, Permits::default()); + let mut driver = open(&nic); + link_up(&nic, &mut driver); + let (mut sent, mut received) = (0, 0); + for round in 0..(TX_RING as u8 * 3) { + nic.deliver(&frame(round + 1, 64 + round as usize)); + if let Some(slot) = driver.tx_reserve(64) { + nic.put_bytes(slot.at, &frame(0x80 | round, 64)); + driver.tx_commit(slot); + sent += 1; + } + nic.run(); + one_pass(&mut driver); + received += drain(&nic, &mut driver).len(); + } + driver.reclaim(); + (nic.trace(), sent, received) +} + +/// Whether `word`, loaded from `at`, is a descriptor's status with `DD` set. +fn found_done(at: usize, word: u64) -> bool { + let status_of = |ring: usize, count: usize, bytes: usize| { + (ring..ring + count * bytes).contains(&at) && (at - ring) % bytes == 8 + }; + if status_of(OFF_RX_RING, RX_RING, rx_desc::BYTES) { + (word >> rx_desc::STATUS_SHIFT) as u8 & rx_desc::status::DD != 0 + } else if status_of(OFF_TX_RING, TX_RING, tx_desc::BYTES) { + (word >> tx_desc::STATUS_SHIFT) as u8 & tx_desc::STATUS_DD != 0 + } else { + false + } +} + +/// `toyos_device_memory::DmaBuffers::publish`'s order, at both tails: §7.2.4.1 +/// and §7.1.8 have the part fetch on the tail write, so no store into the +/// grant stands between the last `publish` and a write of `TDT` or `RDT`. +/// The model reads one memory whole and cannot fail on it; the trace can. +#[test] +fn a_tail_register_is_written_only_after_a_publish_that_follows_the_last_descriptor_store() { + for (part, _) in PARTS { + for seed in 1..=8u64 { + let (trace, sent, received) = a_worked_trace(seed, part); + let mut unpublished = None; + let (mut stored, mut moved) = ([false; 2], [0usize; 2]); + for access in trace { + match access { + Access::Store { at } => { + unpublished = Some(at); + stored[(at >= OFF_TX_RING) as usize] = true; + } + Access::Publish => unpublished = None, + Access::Write { reg } if reg == regs::RDT || reg == regs::TDT => { + assert_eq!( + unpublished, None, + "seed {seed}, {part:?}: tail register {reg:#x} was written over a \ + store into the grant that no publish followed" + ); + let ring = (reg == regs::TDT) as usize; + moved[ring] += core::mem::take(&mut stored[ring]) as usize; + } + Access::Load { .. } | Access::Observe | Access::Write { .. } => {} + } + } + // The arming of each ring is one more than the frames. + assert!(received > 0 && sent > 0, "seed {seed}, {part:?}: no frame moved"); + assert!(moved[0] > 1, "seed {seed}, {part:?}: RDT never moved over a stored descriptor"); + assert_eq!(moved[1], sent + 1, "seed {seed}, {part:?}: TDT did not move once a frame"); + } + } +} + +/// `toyos_device_memory::DmaBuffers::observe`'s order, on both rings: §7.1.7.1 +/// and §7.2.4.2 write descriptors back in batches, so after the load that +/// first finds `DD` in a descriptor the driver loads nothing from the grant +/// before an `observe` — the descriptor's own length and errors least of all. +#[test] +fn a_descriptors_fields_are_read_only_after_an_observe_that_follows_the_load_that_found_dd() { + for (part, _) in PARTS { + for seed in 1..=8u64 { + let (trace, sent, received) = a_worked_trace(seed, part); + // Status words whose `DD` the driver has found since it last + // stored them, and the one no `observe` has followed yet. + let mut taken = std::collections::BTreeSet::new(); + let mut unobserved = None; + let mut found = [0usize; 2]; + for access in trace { + match access { + Access::Store { at } => { + taken.remove(&at); + } + Access::Load { at, word } => { + assert_eq!( + unobserved, None, + "seed {seed}, {part:?}: {at:#x} was loaded after the load that found \ + DD, with no observe between them" + ); + if found_done(at, word) && taken.insert(at) { + unobserved = Some(at); + found[(at >= OFF_TX_RING) as usize] += 1; + } + } + Access::Observe => unobserved = None, + Access::Publish | Access::Write { .. } => {} + } + } + assert_eq!(unobserved, None, "seed {seed}, {part:?}: a DD was found and never observed"); + assert_eq!(found[0], received, "seed {seed}, {part:?}: not every frame was found by its DD"); + assert_eq!(found[1], sent, "seed {seed}, {part:?}: not every sent descriptor was found by its DD"); + } + } +} // --- the PHY behind MDIC --- /// With a partner on the wire and the PHY where a part whose Management Engine @@ -2933,20 +3086,32 @@ fn quiesce_stops_what_a_previous_holder_left_running() { fn bytes(&self) -> usize { crate::regs::REGISTER_BYTES } - fn read(&self, reg: usize) -> u32 { + fn read8(&self, at: usize) -> u8 { + panic!("a 32-bit register was read at {at:#x} as 8 bits") + } + fn read16(&self, at: usize) -> u16 { + panic!("a 32-bit register was read at {at:#x} as 16 bits") + } + fn write8(&self, at: usize, _: u8) { + panic!("a 32-bit register was written at {at:#x} as 8 bits") + } + fn write16(&self, at: usize, _: u16) { + panic!("a 32-bit register was written at {at:#x} as 16 bits") + } + fn read32(&self, reg: usize) -> u32 { *self.0.borrow().get(®).unwrap_or(&0) } - fn write(&self, reg: usize, value: u32) { + fn write32(&self, reg: usize, value: u32) { self.0.borrow_mut().insert(reg, value); } } let regs = Regs(core::cell::RefCell::default()); let rctl = crate::regs::rctl::EN | crate::regs::rctl::BAM; let tctl = crate::regs::tctl::EN | crate::regs::tctl::PSP; - crate::Registers::write(®s, crate::regs::RCTL, rctl); - crate::Registers::write(®s, crate::regs::TCTL, tctl); + crate::Registers::write32(®s, crate::regs::RCTL, rctl); + crate::Registers::write32(®s, crate::regs::TCTL, tctl); crate::quiesce(®s); - let read = |reg| crate::Registers::read(®s, reg); + let read = |reg| crate::Registers::read32(®s, reg); assert_eq!(read(crate::regs::RCTL), crate::regs::rctl::BAM); assert_eq!(read(crate::regs::TCTL), crate::regs::tctl::PSP); assert_eq!(read(crate::regs::IMC), u32::MAX); diff --git a/toyos-virtio/Cargo.toml b/toyos-virtio/Cargo.toml new file mode 100644 index 00000000000..40f277aa30f --- /dev/null +++ b/toyos-virtio/Cargo.toml @@ -0,0 +1,28 @@ +# A member of the host workspace (root `Cargo.toml`), like toyos-i219 and +# toyos-hda: it is pure, and its tests run on the host with no guest and no +# QEMU. +# +# What lives here is what every virtio driver in userland decides before its +# own device type begins — where the device's structures are, what was +# negotiated, what a used-ring element is allowed to say — against two traits +# the driver's program implements one instruction deep. The device is on the +# far side of a trust boundary, so its registers and its used ring have to be +# driven with values no working device would write far more often than a boot +# allows; the specification is the oracle and `stub.rs` is what it is written +# into. + +[package] +name = "toyos-virtio" +description = "The virtio 1.2 PCI transport and split virtqueue: every decision a driver makes about them, and none of the instructions that carry them out, pure." +version = "0.1.0" +edition = "2021" +license = "MIT OR Apache-2.0" +publish = false + +[dependencies] +# The type a used-ring element's two fields stay in until each has met its +# bound. +toyos-untrusted = { path = "../toyos-untrusted" } +# `Registers` and `DmaBuffers`: the boundary every userland driver is written +# against, and the order of its two barriers. +toyos-device-memory = { path = "../toyos-device-memory" } diff --git a/toyos-virtio/src/lib.rs b/toyos-virtio/src/lib.rs new file mode 100644 index 00000000000..ff89774760a --- /dev/null +++ b/toyos-virtio/src/lib.rs @@ -0,0 +1,68 @@ +//! The virtio PCI transport and the split virtqueue — every decision a driver +//! makes about them, and none of the instructions that carry them out. +//! +//! Every `§` in this crate is a section of *Virtual I/O Device (VIRTIO) +//! Version 1.2*, OASIS Committee Specification 01. [`pci`] is §4.1 with the +//! initialisation §3.1.1 orders and the negotiation §2.2 bounds; [`queue`] is +//! §2.7. A device type — its feature bits, its configuration fields, what its +//! buffers carry — is its driver's and is not here. +//! +//! # The boundary +//! +//! `toyos-device-memory`'s two traits, the ones every userland driver is +//! written against: `Registers` is one access to the BAR the device's +//! structures are in, and `DmaBuffers` is the grant a queue's three parts are +//! laid out in. Each is a generic parameter resolved at compile time, and no +//! implementation of one decides anything: every branch is above the +//! boundary, in this crate, where the host tests it against `stub.rs`. +//! +//! **The offsets are bounded here before any reaches an implementation**, and +//! **the width of each register access is §4.1.3.1's**, chosen here. The two +//! barriers are where §2.7.13 and §2.7.8.2 put them ([`queue`]). +//! +//! # The device is not trusted +//! +//! Everything a device publishes is input from outside the driver: the +//! capabilities that say where its structures are, its feature bits, its +//! status, a queue's size and notify offset, and every word of a used ring. +//! Each is bounded before it is an offset, an index or a length, and one that +//! fails its bound is a refusal by name — [`pci::Refusal`] for the transport, +//! [`queue::UsedRefusal`] for a used ring — never a panic and never an access +//! the bound did not cover. +//! +//! **A refusal is the end of the device's use.** The transport's ends the +//! bring-up, and takes the [`pci::Setup`] it was made on with it. A used +//! ring's is something no conforming device writes (§2.7.8, §2.7.8.2), and +//! the queue it was read from is not one to go on reading: the element was +//! spent, and the used index may never pass the available one, so every chain +//! still in flight is one element short of coming back. +//! +//! What panics here is the *driver's* own mistake — a chain published over +//! descriptors still in flight, a ring laid out past its grant — which no +//! device can cause. +//! +//! # What is not here +//! +//! - **No notification suppression, either way.** No `VIRTIO_F_EVENT_IDX` is +//! accepted and `avail.flags` stays 0, so §2.7.7.2 has the device notify for +//! every buffer it uses; and every published chain is notified, which +//! §2.7.10.1 permits whatever `used.flags` says — reading it would buy one +//! skipped register write for one more device-written word believed. +//! - **No packed ring, no indirect descriptors, no legacy interface.** +//! - **No walk of configuration space**: [`pci::Layout::of`] takes the vendor +//! capabilities a driver read through its claim. + +#![no_std] +#![forbid(unsafe_code)] + +extern crate alloc; +#[cfg(test)] +extern crate std; + +pub mod pci; +pub mod queue; + +#[cfg(test)] +mod stub; +#[cfg(test)] +mod tests; diff --git a/toyos-virtio/src/pci.rs b/toyos-virtio/src/pci.rs new file mode 100644 index 00000000000..6c08b11757e --- /dev/null +++ b/toyos-virtio/src/pci.rs @@ -0,0 +1,457 @@ +//! The PCI transport (§4.1): where a device's structures are, the +//! initialisation §3.1.1 orders, and a queue's configuration. +//! +//! **The order is in the types.** [`Offer`] is a device reset and +//! acknowledged, its feature bits read; [`Offer::accept`] answers a [`Setup`], +//! on which queues are configured; [`Setup::driver_ok`] answers a [`Live`], +//! and only a `Live` sends a notification — §3.1.1 has none sent before +//! `DRIVER_OK`. A queue is enabled by the one call that gave it its addresses +//! and its vector (§4.1.4.3.2), so none is enabled without them. **Every step +//! takes the device by value and a refusal does not give it back**, so +//! nothing is written to a device after `FAILED` was. +//! +//! **Access widths are §4.1.3.1's**: a byte for `device_status`, sixteen bits +//! for a sixteen-bit field, and a sixty-four-bit field as its two halves. +//! +//! A refusal after the device was acknowledged sets `FAILED` (§3.1.1), over +//! the bits already set: §2.1.1 has the driver clear none. + +use alloc::vec::Vec; + +use toyos_device_memory::{DmaBuffers, Registers}; + +use crate::queue::{Published, Virtqueue}; + +/// §4.1.4: `cfg_type`. +const CFG_COMMON: u8 = 1; +const CFG_NOTIFY: u8 = 2; +const CFG_ISR: u8 = 3; +const CFG_DEVICE: u8 = 4; + +/// §4.1.4: `bar` "values 0x0 to 0x5 specify a Base Address register", and +/// "any other value is reserved". +const LAST_BAR: u8 = 5; + +/// §4.1.4.3: `virtio_pci_common_cfg`. +mod common { + pub const DEVICE_FEATURE_SELECT: usize = 0x00; + pub const DEVICE_FEATURE: usize = 0x04; + pub const DRIVER_FEATURE_SELECT: usize = 0x08; + pub const DRIVER_FEATURE: usize = 0x0C; + pub const CONFIG_MSIX_VECTOR: usize = 0x10; + pub const DEVICE_STATUS: usize = 0x14; + pub const QUEUE_SELECT: usize = 0x16; + pub const QUEUE_SIZE: usize = 0x18; + pub const QUEUE_MSIX_VECTOR: usize = 0x1A; + pub const QUEUE_ENABLE: usize = 0x1C; + pub const QUEUE_NOTIFY_OFF: usize = 0x1E; + pub const QUEUE_DESC: usize = 0x20; + pub const QUEUE_DRIVER: usize = 0x28; + pub const QUEUE_DEVICE: usize = 0x30; + /// Through `queue_device`: every field this transport reaches. + pub const BYTES: usize = 0x38; +} + +/// §2.1: the device status bits. +pub mod status { + pub const ACKNOWLEDGE: u8 = 1; + pub const DRIVER: u8 = 2; + pub const DRIVER_OK: u8 = 4; + pub const FEATURES_OK: u8 = 8; + pub const FAILED: u8 = 128; +} + +/// §6: "This indicates compliance with this specification". +pub const VIRTIO_F_VERSION_1: u64 = 1 << 32; +/// §6: the device reaches memory through addresses the platform translates, +/// which is what [`DmaBuffers::device_addr`] answers. +pub const VIRTIO_F_ACCESS_PLATFORM: u64 = 1 << 33; + +/// One vendor-specific capability (§4.1.4's `virtio_pci_cap`), as a driver +/// read it out of configuration space. Every field is the device's. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct VendorCap { + pub cfg_type: u8, + pub bar: u8, + pub offset: u32, + pub length: u32, + /// The dword after the structure, which is `notify_off_multiplier` where + /// `cfg_type` is the notification structure's (§4.1.4.4) and nothing + /// anywhere else. + pub notify_off_multiplier: u32, +} + +/// Which of a device's interrupt sources. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum Source { + Config, + Queue(u16), +} + +/// Why the device is not driven. Each is something the device published or +/// answered. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum Refusal { + /// No capability of a structure every device presents (§4.1.4.3.1, + /// §4.1.4.4.1, §4.1.4.5.1, §4.1.4.6) in a BAR that is one. + MissingCap(&'static str), + /// The four structures are not in one BAR, and one BAR is what a + /// [`Registers`] is. + SplitAcrossBars, + /// A structure's offset and length run past its BAR. + OutsideBar(&'static str), + /// A structure is shorter than the fields this transport reaches in it. + TooShort(&'static str), + /// A structure is off the alignment its section requires. + Misaligned(&'static str), + /// `device_status` did not read 0 after 0 was written (§4.1.4.3.2). + ResetUnanswered, + /// The device does not offer `VIRTIO_F_VERSION_1` (§6.1). + NotVersion1 { offered: u64 }, + /// `FEATURES_OK` did not stay set (§3.1.1 step 6). + FeaturesRefused { accepted: u64, status: u8 }, + /// A vector field did not read back the entry written to it + /// (§4.1.5.1.2.2). + NoVector(Source), + /// The queue is absent, or shallower than the driver's rings + /// (§4.1.4.3: `queue_size`). + QueueTooShallow { queue: u16, offered: u16, wanted: u16 }, + /// The queue's `queue_notify_off` puts its notify address outside the + /// notification structure, or off its alignment (§4.1.4.4.1). + Doorbell { queue: u16, notify_off: u16 }, + /// A read past the device-specific structure. + PastDeviceConfig { at: usize, bytes: usize }, +} + +impl core::fmt::Display for Refusal { + fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { + match self { + Self::MissingCap(what) => write!(f, "it published no usable {what}"), + Self::SplitAcrossBars => { + write!(f, "it published its four configuration structures in more than one BAR") + } + Self::OutsideBar(what) => write!(f, "its {what} runs past the BAR it is in"), + Self::TooShort(what) => { + write!(f, "its {what} is shorter than the fields a driver reaches in it") + } + Self::Misaligned(what) => write!(f, "its {what} is not aligned for its fields"), + Self::ResetUnanswered => write!(f, "it never zeroed DEVICE_STATUS for its reset"), + Self::NotVersion1 { offered } => { + write!(f, "it offers the feature set {offered:#x}, without VERSION_1") + } + Self::FeaturesRefused { accepted, status } => write!( + f, + "it refused the feature set {accepted:#x} this driver accepted, leaving \ + DEVICE_STATUS={status:#x} without FEATURES_OK" + ), + Self::NoVector(Source::Config) => { + write!(f, "it refused a vector for its configuration-change interrupt") + } + Self::NoVector(Source::Queue(queue)) => { + write!(f, "it refused a vector for queue {queue}") + } + Self::QueueTooShallow { queue, offered, wanted } => write!( + f, + "its queue {queue} holds {offered} descriptor(s), where this driver's rings \ + are {wanted}" + ), + Self::Doorbell { queue, notify_off } => write!( + f, + "its queue {queue} names notify offset {notify_off}, which is no place in its \ + notification structure" + ), + Self::PastDeviceConfig { at, bytes } => write!( + f, + "its device configuration is {bytes} byte(s), and the driver's field at {at:#x} \ + is not in it" + ), + } + } +} + +/// Where a device's structures are: the capabilities a driver uses, and the +/// one BAR they are in. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct Layout { + common: VendorCap, + notify: VendorCap, + device: VendorCap, +} + +impl Layout { + /// Choose among `caps`, in the order the device's capability list has + /// them. + /// + /// §4.1.4.1: the first of each type is the one used, and a capability with + /// a reserved `bar` or `cfg_type` is ignored. The ISR structure is not + /// reached — a vector is bound, and §4.1.4.5.2 has it left alone then — + /// but a device without one is not a device §4.1.4.5.1 describes. + pub fn of(caps: &[VendorCap]) -> Result { + let first = |cfg_type: u8, what: &'static str| { + caps.iter() + .find(|cap| cap.cfg_type == cfg_type && cap.bar <= LAST_BAR) + .copied() + .ok_or(Refusal::MissingCap(what)) + }; + let common = first(CFG_COMMON, "COMMON_CFG")?; + let notify = first(CFG_NOTIFY, "NOTIFY_CFG")?; + let isr = first(CFG_ISR, "ISR_CFG")?; + let device = first(CFG_DEVICE, "DEVICE_CFG")?; + if [notify, isr, device].iter().any(|cap| cap.bar != common.bar) { + return Err(Refusal::SplitAcrossBars); + } + Ok(Self { common, notify, device }) + } + + /// The BAR a [`Registers`] for this device is a window over. + pub fn bar(&self) -> u8 { + self.common.bar + } +} + +#[derive(Clone, Copy)] +struct Region { + at: usize, + bytes: usize, +} + +impl Region { + /// The structure `cap` names, held to the window it is in: no shorter + /// than `need`, and on a multiple of `align`. + fn of( + cap: &VendorCap, + what: &'static str, + window: usize, + need: usize, + align: u32, + ) -> Result { + // Sixty-four bits, so two device-chosen dwords cannot wrap. + if cap.offset as u64 + cap.length as u64 > window as u64 { + return Err(Refusal::OutsideBar(what)); + } + if (cap.length as usize) < need { + return Err(Refusal::TooShort(what)); + } + if !cap.offset.is_multiple_of(align) { + return Err(Refusal::Misaligned(what)); + } + Ok(Self { at: cap.offset as usize, bytes: cap.length as usize }) + } +} + +/// The registers, and where the three structures are in them. +struct Wires { + regs: R, + common: usize, + notify: Region, + notify_off_multiplier: u32, + device: Region, + /// What `device_status` was last written with. + status: u8, +} + +impl Wires { + fn set_status(&mut self, bit: u8) { + self.status |= bit; + self.regs.write8(self.common + common::DEVICE_STATUS, self.status); + } + + /// Give the device up (§3.1.1), and hand the reason back. + fn refuse(&mut self, why: Refusal) -> Refusal { + self.set_status(status::FAILED); + why + } +} + +/// A device reset and acknowledged, and the features it offers. +pub struct Offer { + wires: Wires, + offered: u64, +} + +impl Offer { + /// §3.1.1 steps 1 to 3 and the first half of 4, over the BAR `layout` + /// names. + /// + /// The reset is one write and one read: a device that has not zeroed its + /// status by then is refused, where §4.1.4.3.2 would have it waited on, + /// and nothing more is written to it. + pub fn acknowledge(regs: R, layout: &Layout) -> Result { + let window = regs.bytes(); + // §4.1.4.3.1 and §4.1.4.4.1: the alignment of each offset. + let common = Region::of(&layout.common, "COMMON_CFG", window, common::BYTES, 4)?; + let notify = Region::of(&layout.notify, "NOTIFY_CFG", window, 0, 2)?; + let device = Region::of(&layout.device, "DEVICE_CFG", window, 0, 1)?; + let mut wires = Wires { + regs, + common: common.at, + notify, + notify_off_multiplier: layout.notify.notify_off_multiplier, + device, + status: 0, + }; + + let at = wires.common; + wires.regs.write8(at + common::DEVICE_STATUS, 0); + if wires.regs.read8(at + common::DEVICE_STATUS) != 0 { + return Err(Refusal::ResetUnanswered); + } + wires.set_status(status::ACKNOWLEDGE); + wires.set_status(status::DRIVER); + + wires.regs.write32(at + common::DEVICE_FEATURE_SELECT, 0); + let low = wires.regs.read32(at + common::DEVICE_FEATURE); + wires.regs.write32(at + common::DEVICE_FEATURE_SELECT, 1); + let high = wires.regs.read32(at + common::DEVICE_FEATURE); + Ok(Self { wires, offered: (high as u64) << 32 | low as u64 }) + } + + /// Every feature bit the device offers. + pub fn features(&self) -> u64 { + self.offered + } + + /// The rest of §3.1.1 step 4, and steps 5 and 6: accept `wanted` of what + /// the device offers. + /// + /// `wanted` is the device type's bits. The transport's own are not the + /// caller's to choose (§6.1): `VIRTIO_F_VERSION_1` is accepted, and a + /// device without it is refused; `VIRTIO_F_ACCESS_PLATFORM` is accepted + /// wherever it is offered. Nothing the device did not offer is written + /// (§2.2.1). + pub fn accept(mut self, wanted: u64) -> Result, Refusal> { + if self.offered & VIRTIO_F_VERSION_1 == 0 { + return Err(self.wires.refuse(Refusal::NotVersion1 { offered: self.offered })); + } + let features = self.offered & (wanted | VIRTIO_F_VERSION_1 | VIRTIO_F_ACCESS_PLATFORM); + let wires = &mut self.wires; + let at = wires.common; + wires.regs.write32(at + common::DRIVER_FEATURE_SELECT, 0); + wires.regs.write32(at + common::DRIVER_FEATURE, features as u32); + wires.regs.write32(at + common::DRIVER_FEATURE_SELECT, 1); + wires.regs.write32(at + common::DRIVER_FEATURE, (features >> 32) as u32); + wires.set_status(status::FEATURES_OK); + let answered = wires.regs.read8(at + common::DEVICE_STATUS); + if answered & status::FEATURES_OK == 0 { + return Err(wires.refuse(Refusal::FeaturesRefused { accepted: features, status: answered })); + } + Ok(Setup { wires: self.wires, features, doorbells: Vec::new() }) + } +} + +/// A device whose features are settled and which is not yet live: §3.1.1 +/// step 7. +/// +/// Each step answers the `Setup` it was made on, or the refusal that ended +/// it: §3.1.1 has a driver that set `FAILED` go no further. +pub struct Setup { + wires: Wires, + features: u64, + /// Each enabled queue's index, and where in the BAR its notify address is. + doorbells: Vec<(u16, usize)>, +} + +impl Setup { + /// The feature bits negotiated. + pub fn features(&self) -> u64 { + self.features + } + + /// Map configuration changes to MSI-X table entry `entry` (§4.1.5.1.2). + pub fn config_vector(self, entry: u16) -> Result { + self.bind(common::CONFIG_MSIX_VECTOR, entry, Source::Config) + } + + fn bind(mut self, field: usize, entry: u16, source: Source) -> Result { + let at = self.wires.common + field; + self.wires.regs.write16(at, entry); + // §4.1.5.1.2.2: "on success, the previously written value is returned". + if self.wires.regs.read16(at) != entry { + return Err(self.wires.refuse(Refusal::NoVector(source))); + } + Ok(self) + } + + /// Give the device `queue`: its size, its three addresses and MSI-X table + /// entry `entry` for its used-buffer notifications, and then — and only + /// then (§4.1.4.3.2) — its enable. + pub fn enable( + mut self, + queue: &Virtqueue, + entry: u16, + ) -> Result { + let (index, wanted) = (queue.index(), queue.size()); + let at = self.wires.common; + self.wires.regs.write16(at + common::QUEUE_SELECT, index); + // The most the device takes, and 0 for a queue it does not have. + let offered = self.wires.regs.read16(at + common::QUEUE_SIZE); + if offered < wanted { + return Err(self.wires.refuse(Refusal::QueueTooShallow { queue: index, offered, wanted })); + } + self.wires.regs.write16(at + common::QUEUE_SIZE, wanted); + for (field, addr) in [ + (common::QUEUE_DESC, queue.desc_addr()), + (common::QUEUE_DRIVER, queue.avail_addr()), + (common::QUEUE_DEVICE, queue.used_addr()), + ] { + self.wires.regs.write32(at + field, addr as u32); + self.wires.regs.write32(at + field + 4, (addr >> 32) as u32); + } + + // §4.1.4.4: the notify address is `cap.offset + queue_notify_off * + // notify_off_multiplier`, and §4.1.4.4.1 has the structure hold the + // two bytes written there. Both factors are the device's; their + // product fits sixty-four bits. + let notify_off = self.wires.regs.read16(at + common::QUEUE_NOTIFY_OFF); + let into = notify_off as u64 * self.wires.notify_off_multiplier as u64; + if into + 2 > self.wires.notify.bytes as u64 || !into.is_multiple_of(2) { + return Err(self.wires.refuse(Refusal::Doorbell { queue: index, notify_off })); + } + let doorbell = self.wires.notify.at + into as usize; + + let mut bound = self.bind(common::QUEUE_MSIX_VECTOR, entry, Source::Queue(index))?; + bound.wires.regs.write16(at + common::QUEUE_ENABLE, 1); + bound.doorbells.push((index, doorbell)); + Ok(bound) + } + + /// One byte of the device-specific structure, `at` bytes into it. + pub fn device_read8(mut self, at: usize) -> Result<(Self, u8), Refusal> { + let Region { at: base, bytes } = self.wires.device; + if at >= bytes { + return Err(self.wires.refuse(Refusal::PastDeviceConfig { at, bytes })); + } + let byte = self.wires.regs.read8(base + at); + Ok((self, byte)) + } + + /// §3.1.1 step 8: the device is live. + pub fn driver_ok(mut self) -> Live { + self.wires.set_status(status::DRIVER_OK); + Live { regs: self.wires.regs, doorbells: self.doorbells } + } +} + +/// A live device: what is left of the transport is its notify addresses. +pub struct Live { + regs: R, + doorbells: Vec<(u16, usize)>, +} + +impl Live { + /// Tell the device of the chain `published` made available: the queue's + /// index, sixteen bits, at its notify address (§4.1.5.2). + /// + /// # Panics + /// If the chain's queue is not one [`Setup::enable`] enabled on this + /// device, which is the driver's own mistake. + pub fn notify(&self, published: Published) { + let queue = published.queue(); + let (_, at) = self + .doorbells + .iter() + .find(|(index, _)| *index == queue) + .unwrap_or_else(|| panic!("virtio: queue {queue} was published to and never enabled")); + self.regs.write16(*at, queue); + } +} diff --git a/toyos-virtio/src/queue.rs b/toyos-virtio/src/queue.rs new file mode 100644 index 00000000000..87aae8a855d --- /dev/null +++ b/toyos-virtio/src/queue.rs @@ -0,0 +1,368 @@ +//! One split virtqueue (§2.7): its three parts in the grant, the chains in +//! flight, and what a used-ring element has to satisfy. +//! +//! **A chain is the caller's own run of descriptors**, `head` and the ones +//! after it, and the caller picks the head: a driver that gives buffer `i` the +//! head `i` gets back, in a completion's head, the buffer it filled. +//! +//! **Nothing the device can write is read back but the used ring.** The +//! descriptor table and the available ring are the driver's (§2.7.5.1 has the +//! device write neither) and the device reaches both all the same, so what is +//! in flight, how long each chain is and where the available index stands are +//! kept here, in memory the device does not reach. A descriptor table the +//! device has rewritten into a loop is therefore a loop nothing follows. +//! +//! **The order of a publication is §2.7.13's** (steps 4 and 6): the +//! descriptors and the ring entry, `publish`, the index, `publish`, and only +//! then the notification — which [`Published`] is owed for and only a live +//! transport sends. **An element is read after the index that counts it** +//! (§2.7.8.2): the used index, `observe`, the element. + +use alloc::vec; +use alloc::vec::Vec; + +use toyos_untrusted::{Refused, Untrusted}; + +use toyos_device_memory::DmaBuffers; + +/// §2.7: "The maximum Queue Size value is 32768." +pub const MAX_QUEUE_SIZE: u16 = 32768; + +/// §2.7.5: one `virtq_desc`. +pub const DESC_BYTES: usize = 16; +const DESC_LEN: usize = 8; +const DESC_FLAGS: usize = 12; +const DESC_NEXT: usize = 14; +const VIRTQ_DESC_F_NEXT: u16 = 1; +const VIRTQ_DESC_F_WRITE: u16 = 2; + +/// §2.7.6 and §2.7.8: both rings open with `flags` and `idx`, sixteen bits +/// each — `flags` zeroed with the ring and never written again — and the +/// entries follow. +const RING_IDX: usize = 2; +pub const RING_ENTRIES: usize = 4; +pub const AVAIL_ENTRY_BYTES: usize = 2; +/// §2.7.8: one `virtq_used_elem`, `id` then `len`. +pub const USED_ELEM_BYTES: usize = 8; +const USED_ELEM_LEN: usize = 4; + +/// §2.7's table of sizes. The rings' include the event word that follows the +/// entries, which only `VIRTIO_F_EVENT_IDX` gives a meaning. +pub const fn desc_bytes(size: u16) -> usize { + DESC_BYTES * size as usize +} + +pub const fn avail_bytes(size: u16) -> usize { + 6 + AVAIL_ENTRY_BYTES * size as usize +} + +pub const fn used_bytes(size: u16) -> usize { + 6 + USED_ELEM_BYTES * size as usize +} + +/// Where a queue's three parts start in the grant. +#[derive(Clone, Copy)] +pub struct Parts { + pub desc: usize, + pub avail: usize, + pub used: usize, +} + +impl Parts { + /// The three parts one after another from `at`, each on the alignment + /// §2.7's table gives it where `at` is on the descriptor table's. + pub const fn contiguous(at: usize, size: u16) -> Self { + let avail = at + desc_bytes(size); + let used = (avail + avail_bytes(size) + 3) & !3; + Self { desc: at, avail, used } + } + + /// One past the used ring's last byte. + pub const fn end(&self, size: u16) -> usize { + self.used + used_bytes(size) + } +} + +/// One element of a chain: `len` bytes at the device's `addr`, for the device +/// to read or to write. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct Buffer { + pub addr: u64, + pub len: u32, + pub writable: bool, +} + +impl Buffer { + pub const fn readable(addr: u64, len: u32) -> Self { + Self { addr, len, writable: false } + } + + pub const fn writable(addr: u64, len: u32) -> Self { + Self { addr, len, writable: true } + } +} + +/// A chain the device has finished with. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct Used { + pub head: u16, + /// Bytes the device says it wrote, never more than the chain's writable + /// elements hold. §2.7.8.3: nothing past them is to be believed. + pub written: u32, +} + +/// Why the used ring was not believed. The device wrote every word of it, and +/// none of these is one a conforming device writes: the queue is not one to +/// go on reading after any of them. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum UsedRefusal { + /// An element's `id` is past the descriptor table. The element is taken + /// and names no chain. + Head(Refused), + /// An element's `id` is a head no chain is in flight at: one the device + /// already gave back, or one never published. The element is taken. + NoChain { head: u16 }, + /// An element's `len` is more than its chain's device-writable elements + /// hold (§2.7.8). The element is taken and its chain stays in flight: its + /// buffers are the device's still, and are not handed up. + Written { head: u16, len: Refused }, + /// The used index stands further on than the chains made available: the + /// device says it used buffers it was never given, so no element from the + /// last one taken on is one this driver can tell from a stale one. + /// **Nothing is taken, and the ring answers this until the device takes + /// the index back**: the queue is not one to go on reading. + Jumped { used: u16, taken: u16, available: u16 }, +} + +impl core::fmt::Display for UsedRefusal { + fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { + match self { + Self::Head(why) => write!(f, "a used element's head {why}"), + Self::NoChain { head } => { + write!(f, "a used element names head {head}, where no chain is in flight") + } + Self::Written { head, len } => write!( + f, + "the used element for head {head} claims more bytes than its chain may be \ + written: {len}" + ), + Self::Jumped { used, taken, available } => write!( + f, + "the used index reads {used}, past the available index {available}, with the \ + elements before {taken} taken" + ), + } + } +} + +/// A chain made visible to the device, and the notification owed for it: +/// [`crate::pci::Live::notify`] takes it, and nothing else does. +#[must_use = "a published chain the device is never told of is never taken"] +pub struct Published { + queue: u16, +} + +impl Published { + pub fn queue(&self) -> u16 { + self.queue + } +} + +#[derive(Clone, Copy, Default)] +struct Slot { + /// Part of a chain in flight. + held: bool, + /// At a head: how many descriptors its chain is. 0 anywhere else. + descs: u16, + /// At a head: bytes its device-writable elements hold. + writable: u32, +} + +/// One split virtqueue over the grant `M`. +pub struct Virtqueue { + mem: M, + index: u16, + size: u16, + parts: Parts, + /// The available index, as published. Never read back from the ring. + next_avail: u16, + /// How many used elements have been taken, as the ring counts them. + last_used: u16, + /// One per descriptor: the table's own length is the bound a used + /// element's head is held to. + slots: Vec, +} + +impl Virtqueue { + /// Queue `index` of `size` descriptors, its parts where `parts` says, all + /// three zeroed (§4.1.5.1.3 step 4, and §2.7.10.1 of `used.flags`). + /// + /// # Panics + /// The layout is the caller's own, so one the specification refuses is its + /// mistake: a size that is no power of two or past [`MAX_QUEUE_SIZE`] + /// (§2.7, §4.1.4.3.2), a part past the grant, or one off §2.7.1's + /// alignment, here or at the device. + pub fn new(mem: M, index: u16, size: u16, parts: Parts) -> Self { + assert!( + size.is_power_of_two() && size <= MAX_QUEUE_SIZE, + "virtqueue {index}: {size} descriptors is no queue size" + ); + for (what, at, bytes, align) in [ + ("descriptor table", parts.desc, desc_bytes(size), 16), + ("available ring", parts.avail, avail_bytes(size), 2), + ("used ring", parts.used, used_bytes(size), 4), + ] { + assert!( + at.checked_add(bytes).is_some_and(|end| end <= mem.bytes()), + "virtqueue {index}: its {what} at {at:#x} runs past a grant of {:#x} bytes", + mem.bytes() + ); + assert!( + at % align == 0 && mem.device_addr(at) % align as u64 == 0, + "virtqueue {index}: its {what} at {at:#x} is not on a multiple of {align}" + ); + } + for at in (0..desc_bytes(size)).step_by(8) { + mem.write64(parts.desc + at, 0); + } + for at in (0..avail_bytes(size)).step_by(2) { + mem.write16(parts.avail + at, 0); + } + for at in (0..used_bytes(size)).step_by(2) { + mem.write16(parts.used + at, 0); + } + Self { + mem, + index, + size, + parts, + next_avail: 0, + last_used: 0, + slots: vec![Slot::default(); size as usize], + } + } + + pub fn index(&self) -> u16 { + self.index + } + + pub fn size(&self) -> u16 { + self.size + } + + /// What the device is told for each part (§4.1.4.3's `queue_desc`, + /// `queue_driver` and `queue_device`). + pub fn desc_addr(&self) -> u64 { + self.mem.device_addr(self.parts.desc) + } + + pub fn avail_addr(&self) -> u64 { + self.mem.device_addr(self.parts.avail) + } + + pub fn used_addr(&self) -> u64 { + self.mem.device_addr(self.parts.used) + } + + /// Make `chain` available at `head` and the descriptors after it. + /// + /// # Panics + /// The chain is the caller's own, so one the specification refuses is its + /// mistake: an empty one, one past the table, one over a descriptor still + /// in flight, a device-readable element after a device-writable one + /// (§2.7.4.2), or 2^32 bytes and more (§2.7.5.2). + pub fn publish(&mut self, head: u16, chain: &[Buffer]) -> Published { + let (queue, size) = (self.index, self.size); + let first = head as usize; + assert!(!chain.is_empty(), "virtqueue {queue}: a chain of no buffers at head {head}"); + assert!( + first + chain.len() <= size as usize, + "virtqueue {queue}: a chain of {} at head {head} runs past a table of {size}", + chain.len() + ); + let mut total = 0u32; + let mut writable = 0u32; + for (nth, buffer) in chain.iter().enumerate() { + assert!( + !self.slots[first + nth].held, + "virtqueue {queue}: a chain at head {head} takes descriptor {}, which is in flight", + first + nth + ); + assert!( + buffer.writable || writable == 0, + "virtqueue {queue}: the chain at head {head} has a device-readable element after \ + a device-writable one" + ); + total = total.checked_add(buffer.len).unwrap_or_else(|| { + panic!("virtqueue {queue}: the chain at head {head} is 2^32 bytes or more") + }); + if buffer.writable { + writable += buffer.len; + } + } + + for (nth, buffer) in chain.iter().enumerate() { + let last = nth + 1 == chain.len(); + let at = self.parts.desc + (first + nth) * DESC_BYTES; + let direction = if buffer.writable { VIRTQ_DESC_F_WRITE } else { 0 }; + self.mem.write64(at, buffer.addr); + self.mem.write32(at + DESC_LEN, buffer.len); + self.mem.write16(at + DESC_FLAGS, direction | if last { 0 } else { VIRTQ_DESC_F_NEXT }); + // In range: `first + nth + 1 < first + chain.len() <= size`. + self.mem.write16(at + DESC_NEXT, if last { 0 } else { (first + nth + 1) as u16 }); + self.slots[first + nth].held = true; + } + self.slots[first].descs = chain.len() as u16; + self.slots[first].writable = writable; + + let entry = (self.next_avail % size) as usize; + self.mem.write16(self.parts.avail + RING_ENTRIES + entry * AVAIL_ENTRY_BYTES, head); + self.mem.publish(); + self.next_avail = self.next_avail.wrapping_add(1); + self.mem.write16(self.parts.avail + RING_IDX, self.next_avail); + self.mem.publish(); + Published { queue } + } + + /// The next chain the device has finished with, `None` when the ring holds + /// no more, or why what the ring holds was not believed. + /// + /// A chain is retired as it is answered, so a head the device reports + /// twice is refused the second time. + pub fn poll_used(&mut self) -> Result, UsedRefusal> { + let used: u16 = self.mem.read16(self.parts.used + RING_IDX); + let pending = used.wrapping_sub(self.last_used); + if pending == 0 { + return Ok(None); + } + // §2.7.8: an element "matches an entry placed in the available ring by + // the guest earlier", so the device's count never passes the driver's. + let available = self.next_avail.wrapping_sub(self.last_used); + if pending > available { + return Err(UsedRefusal::Jumped { used, taken: self.last_used, available: self.next_avail }); + } + self.mem.observe(); + let at = self.parts.used + + RING_ENTRIES + + (self.last_used % self.size) as usize * USED_ELEM_BYTES; + let id = Untrusted::new(self.mem.read32(at)); + let len = Untrusted::new(self.mem.read32(at + USED_ELEM_LEN)); + self.last_used = self.last_used.wrapping_add(1); + + let first = id.index(self.slots.len()).map_err(UsedRefusal::Head)?; + // Exact: `index` proved it below the table's length, a `u16`. + let head = first as u16; + let chain = self.slots[first]; + if chain.descs == 0 { + return Err(UsedRefusal::NoChain { head }); + } + let written = len + .at_most(chain.writable as u64) + .map_err(|len| UsedRefusal::Written { head, len })?; + for slot in &mut self.slots[first..first + chain.descs as usize] { + *slot = Slot::default(); + } + // Exact: `at_most` proved it no more than `chain.writable`, a `u32`. + Ok(Some(Used { head, written: written as u32 })) + } +} diff --git a/toyos-virtio/src/stub.rs b/toyos-virtio/src/stub.rs new file mode 100644 index 00000000000..a7c0e2f36c7 --- /dev/null +++ b/toyos-virtio/src/stub.rs @@ -0,0 +1,548 @@ +//! A virtio PCI device, as the specification describes it — not as the driver +//! beside it expects it. +//! +//! Written from *VIRTIO Version 1.2* and cited to it clause by clause. **Where +//! the specification binds the driver, this file is what holds it to that**: an +//! access of the wrong width, a write to a read-only field, a feature accepted +//! that was not offered, a status bit cleared, a queue enabled before it was +//! configured, a notification before `DRIVER_OK` — each is an assertion here, +//! by section, and a test that reaches one is red. Where the specification +//! binds the *device*, [`Permits`] is how a test has it break the rule, and +//! the raw ring writes below are how a test has it say anything at all. +//! +//! The grant is the same object: the registers and the memory are one +//! machine's, and one [`Event`] trace orders what the driver did to both. + +use std::cell::RefCell; +use std::rc::Rc; +use std::vec; +use std::vec::Vec; + +use toyos_device_memory::{DmaBuffers, Registers}; + +use crate::pci::{status, Source, VendorCap, VIRTIO_F_ACCESS_PLATFORM, VIRTIO_F_VERSION_1}; +use crate::queue::{ + Buffer, Parts, AVAIL_ENTRY_BYTES, DESC_BYTES, RING_ENTRIES, USED_ELEM_BYTES, +}; + +/// Where the device reaches the grant. Not zero: a driver that told the +/// device a grant offset instead of a device address is then not inside it. +pub const DEVICE_BASE: u64 = 0x0000_0001_0000_0000; +pub const GRANT_BYTES: usize = 0x8000; + +/// The BAR, laid out the way a common emulated device lays its own out: a +/// page for each structure. +pub const BAR_BYTES: usize = 0x4000; +pub const COMMON_AT: u32 = 0x1000; +pub const ISR_AT: u32 = 0x0000; +pub const DEVICE_AT: u32 = 0x2000; +pub const NOTIFY_AT: u32 = 0x3000; +pub const NOTIFY_BYTES: u32 = 0x1000; +pub const NOTIFY_OFF_MULTIPLIER: u32 = 4; +pub const QUEUES: usize = 3; +pub const QUEUE_MAX: u16 = 256; + +/// §4.1.5.1.2: what a vector field reads when the device mapped none. +const NO_VECTOR: u16 = 0xFFFF; + +/// A device-type feature bit, and one the device does not offer. +pub const FEATURE_OFFERED: u64 = 1 << 5; +pub const FEATURE_WITHHELD: u64 = 1 << 7; + +/// What the driver did, in the order it did it. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum Event { + /// A write of `device_status`. + Status(u8), + /// A write of any other common-configuration field, by its offset. + Common { field: usize, value: u32 }, + /// A read of the device-specific structure. + DeviceConfig { at: usize }, + /// A store into the grant. + Store { at: usize, bytes: usize, value: u64 }, + /// A load from the grant. + Load { at: usize, bytes: usize }, + /// [`DmaBuffers::publish`]. + Publish, + /// [`DmaBuffers::observe`]. + Observe, + /// An available-buffer notification (§4.1.5.2). + Notify { at: usize, value: u16 }, +} + +/// What a device does that the specification forbids it, or leaves open. +#[derive(Clone, Copy, Default)] +pub struct Permits { + /// §4.1.4.3.1 has `device_status` read 0 once the reset is done. This one + /// is never done. + pub reset_never_completes: bool, + /// §2.2.2: the device "MUST fail to set the FEATURES_OK device status bit" + /// for a subset it does not take. + pub refuses_features: bool, + /// §4.1.5.1.2: mapping a vector "could fail", and then reads `NO_VECTOR`. + pub no_vector_for: Option, +} + +#[derive(Clone, Copy, Default)] +pub struct QueueRegs { + /// What `queue_size` reads: the maximum on reset, then what was written. + pub size: u16, + pub vector: u16, + pub enabled: bool, + pub notify_off: u16, + pub desc: u64, + pub driver: u64, + pub device: u64, +} + +pub struct State { + pub permits: Permits, + pub offered: u64, + pub accepted: u64, + pub status: u8, + pub config_vector: u16, + pub queues: [QueueRegs; QUEUES], + pub device_config: Vec, + /// §4.1.4.4: what the notification capability carries. + pub notify_off_multiplier: u32, + pub trace: Vec, + pub grant: Vec, + device_feature_select: u32, + driver_feature_select: u32, + queue_select: u16, +} + +/// The machine. A clone is the same machine: the driver holds one as its +/// [`Registers`], one in each queue as its [`DmaBuffers`], and the test keeps +/// one to be the device with. +#[derive(Clone)] +pub struct Machine(Rc>); + +impl Machine { + /// A device after power-on, offering [`FEATURE_OFFERED`] and the two + /// transport bits, with a six-byte device-specific structure. + pub fn new() -> Self { + let mut state = State { + permits: Permits::default(), + offered: VIRTIO_F_VERSION_1 | VIRTIO_F_ACCESS_PLATFORM | FEATURE_OFFERED, + accepted: 0, + status: 0, + config_vector: NO_VECTOR, + queues: [QueueRegs::default(); QUEUES], + device_config: vec![0x52, 0x54, 0x00, 0x12, 0x34, 0x56], + notify_off_multiplier: NOTIFY_OFF_MULTIPLIER, + trace: Vec::new(), + grant: vec![0xA5; GRANT_BYTES], + device_feature_select: 0, + driver_feature_select: 0, + queue_select: 0, + }; + state.reset(); + Self(Rc::new(RefCell::new(state))) + } + + /// The capabilities a walk of this device's configuration space yields, + /// in the list's order: its four structures, and the configuration-access + /// one (§4.1.4.9) no driver here uses. + pub fn caps(&self) -> Vec { + let cap = |cfg_type, offset, length, notify_off_multiplier| VendorCap { + cfg_type, + bar: 4, + offset, + length, + notify_off_multiplier, + }; + vec![ + cap(1, COMMON_AT, 0x1000, 0), + cap(3, ISR_AT, 0x1000, 0), + cap(4, DEVICE_AT, self.state(|s| s.device_config.len() as u32), 0), + cap(2, NOTIFY_AT, NOTIFY_BYTES, self.state(|s| s.notify_off_multiplier)), + cap(5, 0, 0, 0), + ] + } + + pub fn state(&self, read: impl FnOnce(&mut State) -> T) -> T { + read(&mut self.0.borrow_mut()) + } + + pub fn trace(&self) -> Vec { + self.state(|s| s.trace.clone()) + } + + pub fn forget_trace(&self) { + self.state(|s| s.trace.clear()); + } + + /// Every `device_status` the driver wrote, in order. + pub fn statuses(&self) -> Vec { + let trace = self.trace(); + trace.iter().filter_map(|event| if let Event::Status(s) = event { Some(*s) } else { None }).collect() + } + + /// The device's view of a queue's rings. + pub fn ring(&self, parts: Parts, size: u16) -> Ring { + Ring { machine: self.clone(), parts, size } + } +} + +impl State { + /// §2.4.1 and §4.1.5.1.2.1: status 0, every event unmapped; §4.1.4.3.1: + /// `queue_enable` 0 and `queue_size` the maximum. + fn reset(&mut self) { + self.status = 0; + self.accepted = 0; + self.config_vector = NO_VECTOR; + for (index, queue) in self.queues.iter_mut().enumerate() { + *queue = QueueRegs { + size: QUEUE_MAX, + vector: NO_VECTOR, + notify_off: index as u16, + ..QueueRegs::default() + }; + } + } + + fn selected(&mut self) -> Option<&mut QueueRegs> { + self.queues.get_mut(self.queue_select as usize) + } + + /// §4.1.3.1: each field at its own width, a sixty-four-bit one as two + /// halves. + fn common_read(&mut self, field: usize, bytes: usize) -> u32 { + let queue = self.selected().map(|q| *q); + let half = |value: u64, high: bool| if high { (value >> 32) as u32 } else { value as u32 }; + let (width, value) = match field { + 0x00 => (4, self.device_feature_select), + // §4.1.4.3.1: 0 for any select but 0 and 1. + 0x04 => (4, match self.device_feature_select { + 0 => self.offered as u32, + 1 => (self.offered >> 32) as u32, + _ => 0, + }), + 0x08 => (4, self.driver_feature_select), + 0x0C => (4, half(self.accepted, self.driver_feature_select == 1)), + 0x10 => (2, self.config_vector as u32), + 0x12 => (2, QUEUES as u32), + 0x14 => (1, self.status as u32), + 0x15 => (1, 0), + 0x16 => (2, self.queue_select as u32), + // §4.1.4.3.1: 0 for a queue the device does not have. + 0x18 => (2, queue.map_or(0, |q| q.size as u32)), + 0x1A => (2, queue.map_or(NO_VECTOR as u32, |q| q.vector as u32)), + 0x1C => (2, queue.map_or(0, |q| q.enabled as u32)), + 0x1E => (2, queue.map_or(0, |q| q.notify_off as u32)), + 0x20 | 0x24 => (4, half(queue.map_or(0, |q| q.desc), field == 0x24)), + 0x28 | 0x2C => (4, half(queue.map_or(0, |q| q.driver), field == 0x2C)), + 0x30 | 0x34 => (4, half(queue.map_or(0, |q| q.device), field == 0x34)), + _ => panic!("stub: a read at {field:#x} of the common structure is of no field"), + }; + assert_eq!(bytes, width, "stub: §4.1.3.1: a {bytes}-byte read of the {width}-byte field at {field:#x}"); + value + } + + fn common_write(&mut self, field: usize, bytes: usize, value: u32) { + let width = match field { + 0x00 | 0x08 | 0x0C | 0x20..=0x37 if field.is_multiple_of(4) => 4, + 0x10 | 0x16 | 0x18 | 0x1A | 0x1C => 2, + 0x14 => 1, + 0x04 | 0x12 | 0x15 | 0x1E => { + panic!("stub: §4.1.4.3.2: a write to the read-only field at {field:#x}") + } + _ => panic!("stub: a write at {field:#x} of the common structure is of no field"), + }; + assert_eq!(bytes, width, "stub: §4.1.3.1: a {bytes}-byte write of the {width}-byte field at {field:#x}"); + if field == 0x14 { + return self.write_status(value as u8); + } + self.trace.push(Event::Common { field, value }); + let (live, settled) = (self.status & status::DRIVER_OK != 0, self.status & status::FEATURES_OK != 0); + let set_half = |whole: &mut u64, high: bool| { + *whole = if high { + (*whole & 0xFFFF_FFFF) | (value as u64) << 32 + } else { + (*whole & !0xFFFF_FFFF) | value as u64 + } + }; + match field { + 0x00 => self.device_feature_select = value, + 0x08 => self.driver_feature_select = value, + 0x0C => { + assert!( + self.status & status::DRIVER != 0 && !settled, + "stub: §3.1.1: features written with DEVICE_STATUS={:#x}", + self.status + ); + let high = self.driver_feature_select == 1; + let offered = if high { (self.offered >> 32) as u32 } else { self.offered as u32 }; + assert_eq!(value & !offered, 0, "stub: §2.2.1: accepted a feature never offered"); + set_half(&mut self.accepted, high); + } + 0x10 => { + self.config_vector = + if self.permits.no_vector_for == Some(Source::Config) { NO_VECTOR } else { value as u16 }; + } + 0x16 => self.queue_select = value as u16, + _ => { + let refuses = self.permits.no_vector_for == Some(Source::Queue(self.queue_select)); + let select = self.queue_select; + let queue = self + .selected() + .unwrap_or_else(|| panic!("stub: a write to queue {select}, which this device has not")); + assert!(!queue.enabled, "stub: §4.1.4.3.2: queue {select} configured after its enable"); + assert!(settled && !live, "stub: §3.1.1: queue {select} configured outside step 7"); + match field { + 0x18 => { + assert!( + (value as u16).is_power_of_two() && value as u16 <= QUEUE_MAX, + "stub: §4.1.4.3.2: queue size {value}" + ); + queue.size = value as u16; + } + 0x1A => queue.vector = if refuses { NO_VECTOR } else { value as u16 }, + 0x1C => { + assert_eq!(value, 1, "stub: §4.1.4.3.2: {value} written to queue_enable"); + assert!( + queue.desc != 0 && queue.driver != 0 && queue.device != 0, + "stub: §4.1.4.3.2: queue {select} enabled before its addresses" + ); + queue.enabled = true; + } + 0x20 | 0x24 => set_half(&mut queue.desc, field == 0x24), + 0x28 | 0x2C => set_half(&mut queue.driver, field == 0x2C), + 0x30 | 0x34 => set_half(&mut queue.device, field == 0x34), + _ => unreachable!("the width match above named every field"), + } + } + } + } + + fn write_status(&mut self, value: u8) { + self.trace.push(Event::Status(value)); + if value == 0 { + if !self.permits.reset_never_completes { + self.reset(); + } + return; + } + assert_eq!(self.status & !value, 0, "stub: §2.1.1: status {value:#x} clears a bit of {:#x}", self.status); + self.status = value; + if value & status::FEATURES_OK != 0 && self.permits.refuses_features { + self.status &= !status::FEATURES_OK; + } + } + + /// §4.1.5.2 and §4.1.4.4: the queue's index, sixteen bits, at its own + /// notify address — and §3.1.1: none before `DRIVER_OK`. + fn notify(&mut self, at: usize, bytes: usize, value: u32) { + assert_eq!(bytes, 2, "stub: §4.1.5.2: a {bytes}-byte notification"); + assert!(self.status & status::DRIVER_OK != 0, "stub: §3.1.1: a notification before DRIVER_OK"); + let queue = self + .queues + .get(value as usize) + .unwrap_or_else(|| panic!("stub: a notification for queue {value}, which this device has not")); + assert!(queue.enabled, "stub: a notification for queue {value}, which is not enabled"); + assert_eq!( + at, + queue.notify_off as usize * self.notify_off_multiplier as usize, + "stub: §4.1.4.4: queue {value} notified at another queue's address" + ); + self.trace.push(Event::Notify { at, value: value as u16 }); + } + + fn register(&mut self, at: usize, bytes: usize, write: Option) -> u32 { + assert!(at + bytes <= BAR_BYTES, "stub: a {bytes}-byte access at {at:#x} is past the BAR"); + assert_eq!(at % bytes, 0, "stub: a {bytes}-byte access at {at:#x} is not aligned"); + let device = DEVICE_AT as usize..DEVICE_AT as usize + self.device_config.len(); + let notify = NOTIFY_AT as usize..(NOTIFY_AT + NOTIFY_BYTES) as usize; + let common = COMMON_AT as usize..COMMON_AT as usize + 0x1000; + if common.contains(&at) { + match write { + Some(value) => self.common_write(at - COMMON_AT as usize, bytes, value), + None => return self.common_read(at - COMMON_AT as usize, bytes), + } + } else if notify.contains(&at) { + let value = write.expect("stub: a read of the notification structure"); + self.notify(at - NOTIFY_AT as usize, bytes, value); + } else if device.contains(&at) { + assert!(write.is_none() && bytes == 1, "stub: the device structure is bytes, read"); + self.trace.push(Event::DeviceConfig { at: at - DEVICE_AT as usize }); + return self.device_config[at - DEVICE_AT as usize] as u32; + } else { + panic!("stub: an access at {at:#x}, which is in none of the device's structures"); + } + 0 + } + + fn load(&mut self, at: usize, bytes: usize) -> u64 { + assert!(at + bytes <= self.grant.len(), "stub: a {bytes}-byte load at {at:#x} is past the grant"); + assert_eq!(at % bytes, 0, "stub: a {bytes}-byte load at {at:#x} is not aligned"); + self.trace.push(Event::Load { at, bytes }); + self.peek(at, bytes) + } + + fn store(&mut self, at: usize, bytes: usize, value: u64) { + assert!(at + bytes <= self.grant.len(), "stub: a {bytes}-byte store at {at:#x} is past the grant"); + assert_eq!(at % bytes, 0, "stub: a {bytes}-byte store at {at:#x} is not aligned"); + self.trace.push(Event::Store { at, bytes, value }); + self.poke(at, bytes, value); + } + + /// The device's own access to the grant, which is no [`Event`]. + pub fn peek(&self, at: usize, bytes: usize) -> u64 { + let mut word = [0u8; 8]; + word[..bytes].copy_from_slice(&self.grant[at..at + bytes]); + u64::from_le_bytes(word) + } + + pub fn poke(&mut self, at: usize, bytes: usize, value: u64) { + self.grant[at..at + bytes].copy_from_slice(&value.to_le_bytes()[..bytes]); + } +} + +impl Registers for Machine { + fn bytes(&self) -> usize { + BAR_BYTES + } + + fn read8(&self, at: usize) -> u8 { + self.state(|s| s.register(at, 1, None)) as u8 + } + + fn read16(&self, at: usize) -> u16 { + self.state(|s| s.register(at, 2, None)) as u16 + } + + fn read32(&self, at: usize) -> u32 { + self.state(|s| s.register(at, 4, None)) + } + + fn write8(&self, at: usize, value: u8) { + self.state(|s| s.register(at, 1, Some(value as u32))); + } + + fn write16(&self, at: usize, value: u16) { + self.state(|s| s.register(at, 2, Some(value as u32))); + } + + fn write32(&self, at: usize, value: u32) { + self.state(|s| s.register(at, 4, Some(value))); + } +} + +impl DmaBuffers for Machine { + fn bytes(&self) -> usize { + GRANT_BYTES + } + + fn device_addr(&self, at: usize) -> u64 { + DEVICE_BASE + at as u64 + } + + fn read16(&self, at: usize) -> u16 { + self.state(|s| s.load(at, 2)) as u16 + } + + fn read32(&self, at: usize) -> u32 { + self.state(|s| s.load(at, 4)) as u32 + } + + fn read64(&self, at: usize) -> u64 { + self.state(|s| s.load(at, 8)) + } + + fn write16(&self, at: usize, value: u16) { + self.state(|s| s.store(at, 2, value as u64)); + } + + fn write32(&self, at: usize, value: u32) { + self.state(|s| s.store(at, 4, value as u64)); + } + + fn write64(&self, at: usize, value: u64) { + self.state(|s| s.store(at, 8, value)); + } + + fn publish(&self) { + self.state(|s| s.trace.push(Event::Publish)); + } + + fn observe(&self) { + self.state(|s| s.trace.push(Event::Observe)); + } +} + +/// One queue's rings, as the device reaches them. +pub struct Ring { + machine: Machine, + pub parts: Parts, + pub size: u16, +} + +impl Ring { + pub fn avail_idx(&self) -> u16 { + self.machine.state(|s| s.peek(self.parts.avail + 2, 2)) as u16 + } + + pub fn used_idx(&self) -> u16 { + self.machine.state(|s| s.peek(self.parts.used + 2, 2)) as u16 + } + + /// The head in available-ring entry `nth`, counted as the index counts. + pub fn avail_entry(&self, nth: u16) -> u16 { + let at = self.parts.avail + RING_ENTRIES + (nth % self.size) as usize * AVAIL_ENTRY_BYTES; + self.machine.state(|s| s.peek(at, 2)) as u16 + } + + /// The chain at `head`, read the way §2.7.5 has a device read it: each + /// descriptor, and the next while `VIRTQ_DESC_F_NEXT` says there is one. + pub fn chain(&self, head: u16) -> Vec { + let mut chain = Vec::new(); + let mut index = head; + loop { + assert!(index < self.size, "stub: descriptor {index} is past a table of {}", self.size); + assert!(chain.len() < self.size as usize, "stub: §2.7.5.2: the chain at {head} loops"); + let at = self.parts.desc + index as usize * DESC_BYTES; + let (addr, len, flags, next) = self.machine.state(|s| { + (s.peek(at, 8), s.peek(at + 8, 4) as u32, s.peek(at + 12, 2) as u16, s.peek(at + 14, 2) as u16) + }); + assert_eq!(flags & !3, 0, "stub: descriptor {index} carries flags {flags:#x}"); + chain.push(Buffer { addr, len, writable: flags & 2 != 0 }); + if flags & 1 == 0 { + return chain; + } + index = next; + } + } + + /// Write used element `nth`, counted as the index counts, and nothing + /// else: the index stays where it was. + pub fn write_used(&self, nth: u16, id: u32, len: u32) { + let at = self.parts.used + RING_ENTRIES + (nth % self.size) as usize * USED_ELEM_BYTES; + self.machine.state(|s| { + s.poke(at, 4, id as u64); + s.poke(at + 4, 4, len as u64); + }); + } + + pub fn set_used_idx(&self, idx: u16) { + self.machine.state(|s| s.poke(self.parts.used + 2, 2, idx as u64)); + } + + /// §2.7.8: return the chain at `id`, `len` bytes written — the element, + /// and then the index that counts it (§2.7.8.2). + pub fn use_chain(&self, id: u32, len: u32) { + let idx = self.used_idx(); + self.write_used(idx, id, len); + self.set_used_idx(idx.wrapping_add(1)); + } + + /// Overwrite one descriptor, which §2.7.5.1 forbids a device and nothing + /// stops it doing. + pub fn scribble_desc(&self, index: u16, flags: u16, next: u16) { + let at = self.parts.desc + index as usize * DESC_BYTES; + self.machine.state(|s| { + s.poke(at + 12, 2, flags as u64); + s.poke(at + 14, 2, next as u64); + }); + } +} diff --git a/toyos-virtio/src/tests.rs b/toyos-virtio/src/tests.rs new file mode 100644 index 00000000000..4a426bee9c6 --- /dev/null +++ b/toyos-virtio/src/tests.rs @@ -0,0 +1,908 @@ +//! The transport and the queue against [`crate::stub`]: the sequences the +//! specification orders, and every word a device can lie in. + +use std::collections::BTreeMap; +use std::vec::Vec; + +use toyos_untrusted::Refused; + +use crate::pci::{ + status, Layout, Live, Offer, Refusal, Setup, Source, VendorCap, VIRTIO_F_ACCESS_PLATFORM, + VIRTIO_F_VERSION_1, +}; +use crate::queue::{desc_bytes, Buffer, Parts, Used, UsedRefusal, Virtqueue}; +use crate::stub::{ + Event, Machine, Ring, BAR_BYTES, COMMON_AT, DEVICE_BASE, FEATURE_OFFERED, FEATURE_WITHHELD, + GRANT_BYTES, NOTIFY_AT, NOTIFY_OFF_MULTIPLIER, QUEUES, +}; + +const SIZE: u16 = 8; +/// The MSI-X table entry every source is mapped to. +const ENTRY: u16 = 0; +/// Where the buffers a chain names are: past every queue's rings. +const BUFFERS: u64 = DEVICE_BASE + 0x2000; + +fn parts(index: u16) -> Parts { + Parts::contiguous(index as usize * 0x400, SIZE) +} + +fn queue(machine: &Machine, index: u16) -> Virtqueue { + Virtqueue::new(machine.clone(), index, SIZE, parts(index)) +} + +fn ring(machine: &Machine, index: u16) -> Ring { + machine.ring(parts(index), SIZE) +} + +fn layout(machine: &Machine) -> Layout { + Layout::of(&machine.caps()).expect("the stub's own capabilities") +} + +fn setup(machine: &Machine) -> Setup { + Offer::acknowledge(machine.clone(), &layout(machine)) + .and_then(|offer| offer.accept(FEATURE_OFFERED)) + .expect("the stub negotiates") +} + +/// A live device with queue 0 enabled. +fn live(machine: &Machine) -> (Live, Virtqueue) { + let queue = queue(machine, 0); + let setup = setup(machine) + .config_vector(ENTRY) + .and_then(|setup| setup.enable(&queue, ENTRY)) + .unwrap_or_else(|why| panic!("the stub takes a vector and queue 0: {why}")); + (setup.driver_ok(), queue) +} + +fn one(len: u32, writable: bool) -> [Buffer; 1] { + [Buffer { addr: BUFFERS, len, writable }] +} + +fn cap_of(caps: &mut [VendorCap], cfg_type: u8) -> &mut VendorCap { + caps.iter_mut().find(|cap| cap.cfg_type == cfg_type).expect("the stub publishes it") +} + +// --- §4.1.4: where the structures are --- + +/// §4.1.4.1: "The driver SHOULD use the first instance of each virtio +/// structure type", and "MUST ignore any vendor-specific capability structure +/// which has a reserved bar value". +#[test] +fn the_first_capability_of_each_type_in_a_real_bar_is_the_one_used() { + let machine = Machine::new(); + let mut caps = machine.caps(); + let bar = caps[0].bar; + // A common structure in a BAR that is none, ahead of the real one; and a + // second one behind it, over the ISR structure nothing may touch. + caps.insert(0, VendorCap { cfg_type: 1, bar: 6, offset: 0, length: 0x1000, notify_off_multiplier: 0 }); + caps.push(VendorCap { cfg_type: 1, bar, offset: 0, length: 0x1000, notify_off_multiplier: 0 }); + let layout = Layout::of(&caps).expect("the real capabilities are all there"); + assert_eq!(layout.bar(), bar); + // The stub reds on an access outside the structure it published first. + Offer::acknowledge(machine, &layout).expect("the first common structure answers"); +} + +#[test] +fn a_device_missing_a_structure_is_refused_by_its_name() { + let machine = Machine::new(); + for (cfg_type, name) in [(1, "COMMON_CFG"), (2, "NOTIFY_CFG"), (3, "ISR_CFG"), (4, "DEVICE_CFG")] { + let caps: Vec = + machine.caps().into_iter().filter(|cap| cap.cfg_type != cfg_type).collect(); + assert_eq!(Layout::of(&caps), Err(Refusal::MissingCap(name))); + // And one that is only in a reserved BAR is as missing. + let mut caps = machine.caps(); + cap_of(&mut caps, cfg_type).bar = 6; + assert_eq!(Layout::of(&caps), Err(Refusal::MissingCap(name))); + } + assert_eq!(Layout::of(&[]), Err(Refusal::MissingCap("COMMON_CFG"))); +} + +#[test] +fn structures_in_two_bars_are_refused() { + let machine = Machine::new(); + for cfg_type in [2, 3, 4] { + let mut caps = machine.caps(); + cap_of(&mut caps, cfg_type).bar = 0; + assert_eq!(Layout::of(&caps), Err(Refusal::SplitAcrossBars)); + } +} + +/// A capability's offset and length are the device's. One that names bytes +/// past the BAR, too few bytes, or an offset its section forbids is refused +/// before anything is read or written through it. +#[test] +fn a_structure_its_bar_does_not_hold_is_refused_and_never_reached() { + let bar = BAR_BYTES as u32; + let refused = |cfg_type: u8, offset: u32, length: u32| { + let machine = Machine::new(); + let mut caps = machine.caps(); + let cap = cap_of(&mut caps, cfg_type); + (cap.offset, cap.length) = (offset, length); + let layout = Layout::of(&caps).expect("the capabilities are all there"); + let refusal = Offer::acknowledge(machine.clone(), &layout).err(); + if refusal.is_some() { + assert_eq!(machine.trace(), [], "the device was reached"); + } + refusal + }; + for (cfg_type, name) in [(1, "COMMON_CFG"), (2, "NOTIFY_CFG"), (4, "DEVICE_CFG")] { + assert_eq!(refused(cfg_type, bar, 0x1000), Some(Refusal::OutsideBar(name))); + assert_eq!(refused(cfg_type, bar - 0x1000, 0x1001), Some(Refusal::OutsideBar(name))); + // The two dwords sum past thirty-two bits, and to a small number. + assert_eq!(refused(cfg_type, 0xFFFF_F000, 0x2000), Some(Refusal::OutsideBar(name))); + assert_eq!(refused(cfg_type, 0x1000, u32::MAX), Some(Refusal::OutsideBar(name))); + } + // §4.1.4.3: through `queue_device` is 0x38 bytes. + assert_eq!(refused(1, COMMON_AT, 0x37), Some(Refusal::TooShort("COMMON_CFG"))); + assert_eq!(refused(1, COMMON_AT, 0x38), None); + // §4.1.4.3.1: "offset MUST be 4-byte aligned"; §4.1.4.4.1: two for the + // notification structure's. + assert_eq!(refused(1, COMMON_AT + 2, 0x38), Some(Refusal::Misaligned("COMMON_CFG"))); + assert_eq!(refused(2, NOTIFY_AT + 1, 0x100), Some(Refusal::Misaligned("NOTIFY_CFG"))); + // A structure that ends on the BAR's last byte is inside it. + assert_eq!(refused(2, bar - 0x1000, 0x1000), None); +} + +// --- §3.1.1 and §2.2: the initialisation --- + +/// The whole of §3.1.1 as the device sees it: every write, in order, each at +/// the width §4.1.3.1 gives its field (the stub reds on any other). +#[test] +fn the_initialisation_is_the_sequence_the_specification_orders() { + let machine = Machine::new(); + let (_live, queue) = live(&machine); + let features = VIRTIO_F_VERSION_1 | VIRTIO_F_ACCESS_PLATFORM | FEATURE_OFFERED; + let (desc, avail, used) = (queue.desc_addr(), queue.avail_addr(), queue.used_addr()); + assert_eq!((desc, avail, used), (DEVICE_BASE, DEVICE_BASE + 128, DEVICE_BASE + 152)); + let common = |field, value| Event::Common { field, value }; + let registers: Vec = machine + .trace() + .into_iter() + .filter(|event| matches!(event, Event::Status(_) | Event::Common { .. })) + .collect(); + assert_eq!( + registers, + [ + // 1: reset. 2: ACKNOWLEDGE. 3: DRIVER. + Event::Status(0), + Event::Status(status::ACKNOWLEDGE), + Event::Status(status::ACKNOWLEDGE | status::DRIVER), + // 4: the offer read, both halves, and the subset written. + common(0x00, 0), + common(0x00, 1), + common(0x08, 0), + common(0x0C, features as u32), + common(0x08, 1), + common(0x0C, (features >> 32) as u32), + // 5: FEATURES_OK (and 6, its read back, which is no write). + Event::Status(status::ACKNOWLEDGE | status::DRIVER | status::FEATURES_OK), + // 7: the configuration vector; then the queue — selected, sized, + // its three addresses a half at a time, its vector, and last its + // enable (§4.1.4.3.2). + common(0x10, ENTRY as u32), + common(0x16, 0), + common(0x18, SIZE as u32), + common(0x20, desc as u32), + common(0x24, (desc >> 32) as u32), + common(0x28, avail as u32), + common(0x2C, (avail >> 32) as u32), + common(0x30, used as u32), + common(0x34, (used >> 32) as u32), + common(0x1A, ENTRY as u32), + common(0x1C, 1), + // 8: DRIVER_OK. + Event::Status( + status::ACKNOWLEDGE | status::DRIVER | status::FEATURES_OK | status::DRIVER_OK + ), + ] + ); + machine.state(|s| { + assert_eq!(s.accepted, features); + assert!(s.queues[0].enabled); + assert_eq!((s.queues[0].desc, s.queues[0].driver, s.queues[0].device), (desc, avail, used)); + }); +} + +/// §4.1.4.3.2: the driver waits for `device_status` to read 0 "before +/// reinitializing the device". One that does not read 0 is not reinitialised. +#[test] +fn a_device_that_never_finishes_its_reset_is_refused_and_written_nothing_more() { + let machine = Machine::new(); + machine.state(|s| { + s.permits.reset_never_completes = true; + s.status = status::ACKNOWLEDGE | status::DRIVER | status::FEATURES_OK | status::DRIVER_OK; + }); + let refused = Offer::acknowledge(machine.clone(), &layout(&machine)).err(); + assert_eq!(refused, Some(Refusal::ResetUnanswered)); + assert_eq!(machine.trace(), [Event::Status(0)]); +} + +/// §2.2.1: "The driver MUST NOT accept a feature which the device did not +/// offer" — and the stub reds on one written. §6.1: `VERSION_1` and +/// `ACCESS_PLATFORM` are accepted where offered, whatever the caller wanted. +#[test] +fn only_what_the_device_offers_is_accepted() { + for (offered, wanted, accepted) in [ + ( + VIRTIO_F_VERSION_1 | VIRTIO_F_ACCESS_PLATFORM | FEATURE_OFFERED, + FEATURE_OFFERED | FEATURE_WITHHELD, + VIRTIO_F_VERSION_1 | VIRTIO_F_ACCESS_PLATFORM | FEATURE_OFFERED, + ), + ( + VIRTIO_F_VERSION_1 | VIRTIO_F_ACCESS_PLATFORM | FEATURE_OFFERED, + 0, + VIRTIO_F_VERSION_1 | VIRTIO_F_ACCESS_PLATFORM, + ), + (VIRTIO_F_VERSION_1 | FEATURE_OFFERED, u64::MAX, VIRTIO_F_VERSION_1 | FEATURE_OFFERED), + ] { + let machine = Machine::new(); + machine.state(|s| s.offered = offered); + let offer = Offer::acknowledge(machine.clone(), &layout(&machine)).expect("acknowledged"); + assert_eq!(offer.features(), offered); + let setup = offer.accept(wanted).expect("a subset of the offer is taken"); + assert_eq!(setup.features(), accepted); + assert_eq!(machine.state(|s| s.accepted), accepted); + } +} + +/// §6.1: "A driver MAY fail to operate further if VIRTIO_F_VERSION_1 is not +/// offered" — this one is no legacy driver. §3.1.1: it says so with `FAILED`, +/// over the bits it had set (§2.1.1). +#[test] +fn a_device_without_version_1_is_refused_and_told_so() { + let machine = Machine::new(); + machine.state(|s| s.offered = VIRTIO_F_ACCESS_PLATFORM | FEATURE_OFFERED); + let refused = Offer::acknowledge(machine.clone(), &layout(&machine)) + .and_then(|offer| offer.accept(FEATURE_OFFERED)) + .err(); + assert_eq!( + refused, + Some(Refusal::NotVersion1 { offered: VIRTIO_F_ACCESS_PLATFORM | FEATURE_OFFERED }) + ); + assert_eq!(machine.statuses(), [0, 1, 3, 3 | status::FAILED]); + assert_eq!(machine.state(|s| s.accepted), 0); +} + +/// §3.1.1 step 6: "Re-read device status to ensure the FEATURES_OK bit is +/// still set: otherwise, the device does not support our subset of features +/// and the device is unusable." +#[test] +fn a_device_that_does_not_keep_features_ok_is_refused_and_told_so() { + let machine = Machine::new(); + machine.state(|s| s.permits.refuses_features = true); + let refused = Offer::acknowledge(machine.clone(), &layout(&machine)) + .and_then(|offer| offer.accept(FEATURE_OFFERED)) + .err(); + assert_eq!( + refused, + Some(Refusal::FeaturesRefused { + accepted: VIRTIO_F_VERSION_1 | VIRTIO_F_ACCESS_PLATFORM | FEATURE_OFFERED, + status: status::ACKNOWLEDGE | status::DRIVER, + }) + ); + assert_eq!(machine.statuses(), [0, 1, 3, 11, 11 | status::FAILED]); +} + +// --- §4.1.4.3 and §4.1.5.1: a queue's configuration --- + +/// `queue_size` is the most the device takes, and 0 for a queue it does not +/// have (§4.1.4.3.1). Either way the queue is not enabled. +#[test] +fn a_queue_shallower_than_the_drivers_rings_is_refused() { + let machine = Machine::new(); + let shallow = setup(&machine); + // After the reset, which is what gives a queue its maximum back. + machine.state(|s| s.queues[0].size = SIZE / 2); + assert_eq!( + shallow.enable(&queue(&machine, 0), ENTRY).err(), + Some(Refusal::QueueTooShallow { queue: 0, offered: SIZE / 2, wanted: SIZE }) + ); + assert!(!machine.state(|s| s.queues[0].enabled)); + assert_eq!(machine.statuses().last(), Some(&(11 | status::FAILED))); + + let machine = Machine::new(); + let absent = QUEUES as u16; + let rings = Virtqueue::new(machine.clone(), absent, SIZE, parts(0)); + assert_eq!( + setup(&machine).enable(&rings, ENTRY).err(), + Some(Refusal::QueueTooShallow { queue: absent, offered: 0, wanted: SIZE }) + ); +} + +/// §4.1.5.1.2.2: "the driver MUST verify success by reading the Vector field +/// value". A queue whose vector the device did not map is never enabled. +#[test] +fn a_vector_the_device_does_not_map_is_refused_and_its_queue_stays_disabled() { + let machine = Machine::new(); + machine.state(|s| s.permits.no_vector_for = Some(Source::Config)); + assert_eq!( + setup(&machine).config_vector(ENTRY).err(), + Some(Refusal::NoVector(Source::Config)) + ); + assert_eq!(machine.statuses().last(), Some(&(11 | status::FAILED))); + + let machine = Machine::new(); + machine.state(|s| s.permits.no_vector_for = Some(Source::Queue(1))); + let setup = setup(&machine) + .config_vector(ENTRY) + .and_then(|setup| setup.enable(&queue(&machine, 0), ENTRY)) + .unwrap_or_else(|why| panic!("the configuration vector and queue 0's map: {why}")); + assert_eq!( + setup.enable(&queue(&machine, 1), ENTRY).err(), + Some(Refusal::NoVector(Source::Queue(1))) + ); + machine.state(|s| assert!(s.queues[0].enabled && !s.queues[1].enabled)); + assert_eq!(machine.statuses().last(), Some(&(11 | status::FAILED))); +} + +/// §4.1.4.4: the notify address is `cap.offset + queue_notify_off * +/// notify_off_multiplier`, and §4.1.4.4.1 has the structure hold the two bytes +/// written there. Both factors are the device's, so a product outside the +/// structure is refused — the stub reds on a write anywhere else. +#[test] +fn a_notify_offset_outside_the_notification_structure_is_refused() { + let enable = |notify_off: u16, multiplier: u32| { + let machine = Machine::new(); + machine.state(|s| s.notify_off_multiplier = multiplier); + let setup = setup(&machine); + // After the reset, which is what gives a queue its offset back. + machine.state(|s| s.queues[0].notify_off = notify_off); + let enabled = setup.enable(&queue(&machine, 0), ENTRY).map(|_| ()); + (machine, enabled) + }; + let refused = |notify_off| Err(Refusal::Doorbell { queue: 0, notify_off }); + + // 0x1000 bytes of structure: the last two-byte address in it is 0xFFE. + for (notify_off, multiplier) in [(0x400, 4), (0xFFF, 2), (0xFFFF, u32::MAX), (1, 0x1000)] { + let (machine, enabled) = enable(notify_off, multiplier); + assert_eq!(enabled, refused(notify_off), "{notify_off:#x} * {multiplier:#x}"); + assert!(!machine.state(|s| s.queues[0].enabled)); + } + // An odd address is not sixteen-bit aligned. + assert_eq!(enable(1, 1).1, refused(1)); + + for (notify_off, multiplier) in [(0x3FF, 4), (0x7FF, 2), (0xFFFF, 0)] { + let (_machine, enabled) = enable(notify_off, multiplier); + assert_eq!(enabled, Ok(()), "{notify_off:#x} * {multiplier:#x}"); + } +} + +/// §4.1.5.2: "the driver sends an available buffer notification to the device +/// by writing the 16-bit virtqueue index of this virtqueue to the Queue Notify +/// address" — each queue's own, and one address for all where the multiplier +/// is 0 (§4.1.4.4). +#[test] +fn a_notification_is_the_queues_index_at_the_queues_address() { + for multiplier in [NOTIFY_OFF_MULTIPLIER, 0] { + let machine = Machine::new(); + machine.state(|s| s.notify_off_multiplier = multiplier); + let mut queues = [queue(&machine, 0), queue(&machine, 1), queue(&machine, 2)]; + let live = queues + .iter() + .try_fold(setup(&machine), |setup, queue| setup.enable(queue, ENTRY)) + .unwrap_or_else(|why| panic!("the stub takes three queues: {why}")) + .driver_ok(); + machine.forget_trace(); + for index in [2usize, 0, 1] { + live.notify(queues[index].publish(0, &one(64, true))); + } + let notified: Vec = + machine.trace().into_iter().filter(|e| matches!(e, Event::Notify { .. })).collect(); + let at = |index: usize| index * multiplier as usize; + assert_eq!( + notified, + [ + Event::Notify { at: at(2), value: 2 }, + Event::Notify { at: at(0), value: 0 }, + Event::Notify { at: at(1), value: 1 }, + ] + ); + } +} + +#[test] +#[should_panic(expected = "queue 1 was published to and never enabled")] +fn a_chain_on_a_queue_the_device_was_never_given_is_the_drivers_mistake() { + let machine = Machine::new(); + let (live, _queue) = live(&machine); + live.notify(queue(&machine, 1).publish(0, &one(64, true))); +} + +/// The device-specific structure is as long as its capability says, which is +/// the device's number: a field past it is refused and not read. +#[test] +fn a_field_past_the_device_structure_is_refused_and_not_read() { + let machine = Machine::new(); + let mut device = setup(&machine); + machine.forget_trace(); + let mut mac = [0u8; 6]; + for (at, byte) in mac.iter_mut().enumerate() { + (device, *byte) = + device.device_read8(at).unwrap_or_else(|why| panic!("byte {at} is inside: {why}")); + } + assert_eq!(mac, [0x52, 0x54, 0x00, 0x12, 0x34, 0x56]); + assert_eq!(machine.trace().len(), 6); + for at in [6, 7, usize::MAX] { + let machine = Machine::new(); + let device = setup(&machine); + machine.forget_trace(); + assert_eq!( + device.device_read8(at).err(), + Some(Refusal::PastDeviceConfig { at, bytes: 6 }) + ); + assert_eq!(machine.trace(), [Event::Status(11 | status::FAILED)]); + } +} + +// --- §2.7: the queue --- + +/// §4.1.5.1.3 step 4 has the three parts zeroed, and §2.7.10.1 has the driver +/// "initialize flags in the used ring to 0". The stub's grant starts as +/// anything but. +#[test] +fn a_new_queue_zeroes_its_parts_and_nothing_else() { + let machine = Machine::new(); + let _queue = queue(&machine, 0); + let parts = parts(0); + machine.state(|s| { + assert!(s.grant[..parts.end(SIZE)].iter().enumerate().all(|(at, byte)| { + // The padding between the available ring and the used one is not + // a part. + *byte == 0 || (parts.avail + 22..parts.used).contains(&at) + })); + assert!(s.grant[parts.end(SIZE)..].iter().all(|byte| *byte == 0xA5)); + }); +} + +/// §2.7.13.1: each element's address, length and direction, chained by `next` +/// under `VIRTQ_DESC_F_NEXT` — read back the way a device reads a chain — and +/// §2.7.13.2: its head in the next available-ring entry. +#[test] +fn a_published_chain_is_what_the_device_reads() { + let machine = Machine::new(); + let mut queue = queue(&machine, 0); + let ring = ring(&machine, 0); + let request = [ + Buffer::readable(BUFFERS, 16), + Buffer::readable(BUFFERS + 0x100, 1024), + Buffer::writable(BUFFERS + 0x800, 4), + ]; + let _ = queue.publish(2, &request); + let _ = queue.publish(7, &one(2048, true)); + assert_eq!(ring.avail_idx(), 2); + assert_eq!((ring.avail_entry(0), ring.avail_entry(1)), (2, 7)); + assert_eq!(ring.chain(2), request); + assert_eq!(ring.chain(7), one(2048, true)); +} + +/// §2.7.13: the descriptors and the ring entry (steps 1 and 2), a barrier +/// (4), the index (5), a barrier (6), the notification (7). A device may read +/// the chain the moment the index moves (§2.7.13.3). +#[test] +fn a_chain_is_whole_before_its_index_and_its_index_before_its_notification() { + let machine = Machine::new(); + let (live, mut queue) = live(&machine); + machine.forget_trace(); + live.notify(queue.publish(3, &[Buffer::readable(BUFFERS, 16), Buffer::writable(BUFFERS + 16, 64)])); + + let parts = parts(0); + let trace = machine.trace(); + let position = |what: &dyn Fn(&Event) -> bool| { + let found: Vec = + trace.iter().enumerate().filter(|(_, e)| what(e)).map(|(at, _)| at).collect(); + assert!(!found.is_empty(), "missing from {trace:?}"); + found + }; + let table = parts.desc..parts.desc + desc_bytes(SIZE); + let descriptors = position(&|e| matches!(e, Event::Store { at, .. } if table.contains(at))); + let entry = position(&|e| *e == Event::Store { at: parts.avail + 4, bytes: 2, value: 3 }); + let index = position(&|e| *e == Event::Store { at: parts.avail + 2, bytes: 2, value: 1 }); + let barriers = position(&|e| *e == Event::Publish); + let notified = position(&|e| matches!(e, Event::Notify { .. })); + // Two descriptors, four fields each. + assert_eq!(descriptors.len(), 8); + assert_eq!((entry.len(), index.len(), barriers.len(), notified.len()), (1, 1, 2, 1)); + assert!(descriptors.iter().all(|at| *at < barriers[0])); + assert!(entry[0] < barriers[0]); + assert!(barriers[0] < index[0]); + assert!(index[0] < barriers[1]); + assert!(barriers[1] < notified[0]); +} + +/// §2.7.8.2: "The device MUST set len prior to updating the used idx" — so +/// the element is loaded after the index that counts it, across a barrier. +#[test] +fn a_used_element_is_read_after_the_index_that_counts_it() { + let machine = Machine::new(); + let mut queue = queue(&machine, 0); + let _ = queue.publish(0, &one(64, true)); + ring(&machine, 0).use_chain(0, 64); + machine.forget_trace(); + assert_eq!(queue.poll_used(), Ok(Some(Used { head: 0, written: 64 }))); + let used = parts(0).used; + assert_eq!( + machine.trace(), + [ + Event::Load { at: used + 2, bytes: 2 }, + Event::Observe, + Event::Load { at: used + 4, bytes: 4 }, + Event::Load { at: used + 8, bytes: 4 }, + ] + ); + assert_eq!(queue.poll_used(), Ok(None)); +} + +/// A used element's `id` is thirty-two bits and a head is sixteen: one past +/// the table is refused whole, never narrowed into it. +#[test] +fn a_head_past_the_table_is_refused() { + let machine = Machine::new(); + let mut queue = queue(&machine, 0); + let ring = ring(&machine, 0); + for head in 0..4 { + let _ = queue.publish(head, &one(64, true)); + } + // 0x1_0002 narrows to head 2, which is in flight. + for id in [SIZE as u32, 0x1_0002, u32::MAX] { + ring.use_chain(id, 0); + assert_eq!( + queue.poll_used(), + Err(UsedRefusal::Head(Refused::PastTable { value: id as u64, len: SIZE as u64 })) + ); + } + assert_eq!(queue.poll_used(), Ok(None)); + // Head 2 was not retired by the element that narrowed to it. + ring.use_chain(2, 64); + assert_eq!(queue.poll_used(), Ok(Some(Used { head: 2, written: 64 }))); +} + +/// A head nothing is in flight at: never published, or reported a second +/// time — which is how a device would hand one buffer up twice. +#[test] +fn a_head_with_no_chain_in_flight_is_refused() { + let machine = Machine::new(); + let mut queue = queue(&machine, 0); + let ring = ring(&machine, 0); + let _ = queue.publish(0, &one(64, true)); + let _ = queue.publish(4, &[Buffer::readable(BUFFERS, 8), Buffer::writable(BUFFERS + 8, 8)]); + let _ = queue.publish(6, &one(64, true)); + + ring.use_chain(3, 0); + assert_eq!(queue.poll_used(), Err(UsedRefusal::NoChain { head: 3 })); + // Descriptor 5 is in flight and is no head. + ring.use_chain(5, 0); + assert_eq!(queue.poll_used(), Err(UsedRefusal::NoChain { head: 5 })); + ring.use_chain(4, 8); + assert_eq!(queue.poll_used(), Ok(Some(Used { head: 4, written: 8 }))); + let _ = queue.publish(1, &one(64, true)); + ring.use_chain(4, 8); + assert_eq!(queue.poll_used(), Err(UsedRefusal::NoChain { head: 4 })); +} + +/// The one that matters most: `written` becomes the length of what a driver +/// reads out of its buffer. §2.7.8: `len` is "the number of bytes written +/// into the device writable portion of the buffer", so the bound is that +/// portion and not the chain. The chain refused stays in flight, for as long +/// as the device has an element left to answer it with. +#[test] +fn more_bytes_than_the_chain_may_be_written_is_refused() { + for len in [65, 100, 164, u32::MAX] { + let machine = Machine::new(); + let mut queue = queue(&machine, 0); + let ring = ring(&machine, 0); + let _ = queue.publish(0, &[Buffer::readable(BUFFERS, 100), Buffer::writable(BUFFERS + 100, 64)]); + let _ = queue.publish(2, &one(1514, false)); + // Two more, so the device has four elements to spend on two chains. + let _ = queue.publish(3, &one(1, true)); + let _ = queue.publish(4, &one(1, true)); + + ring.use_chain(0, len); + assert_eq!( + queue.poll_used(), + Err(UsedRefusal::Written { + head: 0, + len: Refused::PastBound { value: len as u64, bound: 64 } + }) + ); + ring.use_chain(0, 64); + assert_eq!(queue.poll_used(), Ok(Some(Used { head: 0, written: 64 }))); + + // A chain the device may only read has nothing to be written. + ring.use_chain(2, 1); + assert_eq!( + queue.poll_used(), + Err(UsedRefusal::Written { head: 2, len: Refused::PastBound { value: 1, bound: 0 } }) + ); + ring.use_chain(2, 0); + assert_eq!(queue.poll_used(), Ok(Some(Used { head: 2, written: 0 }))); + } +} + +/// §2.7.8: every used element answers a chain made available. A used index +/// past the available one counts elements nothing was published for, so none +/// is read — not the stale ones, and not the ring's memory under them. +#[test] +fn a_used_index_past_what_was_made_available_is_refused_and_nothing_is_read() { + let machine = Machine::new(); + let mut queue = queue(&machine, 0); + let ring = ring(&machine, 0); + let _ = queue.publish(0, &one(64, true)); + let _ = queue.publish(1, &one(64, true)); + ring.write_used(0, 0, 64); + ring.write_used(1, 1, 64); + ring.write_used(2, 0, 64); + + for jumped in [3, 9, 0x8000, u16::MAX] { + ring.set_used_idx(jumped); + machine.forget_trace(); + for _ in 0..2 { + assert_eq!( + queue.poll_used(), + Err(UsedRefusal::Jumped { used: jumped, taken: 0, available: 2 }) + ); + } + let used = parts(0).used; + assert_eq!(machine.trace(), [Event::Load { at: used + 2, bytes: 2 }; 2]); + } + + // The index taken back to what was made available, the ring reads on. + ring.set_used_idx(2); + assert_eq!(queue.poll_used(), Ok(Some(Used { head: 0, written: 64 }))); + assert_eq!(queue.poll_used(), Ok(Some(Used { head: 1, written: 64 }))); + assert_eq!(queue.poll_used(), Ok(None)); + // And an index that goes backwards is one that went all the way round. + ring.set_used_idx(1); + assert_eq!(queue.poll_used(), Err(UsedRefusal::Jumped { used: 1, taken: 2, available: 2 })); +} + +/// One element refused is one element: the ones behind it are still read. +#[test] +fn a_refused_element_does_not_hide_the_ones_behind_it() { + let machine = Machine::new(); + let mut queue = queue(&machine, 0); + let ring = ring(&machine, 0); + for head in 0..3 { + let _ = queue.publish(head, &one(64, true)); + } + ring.use_chain(0x55, 0); + ring.use_chain(1, 64); + ring.use_chain(2, 3); + assert!(matches!(queue.poll_used(), Err(UsedRefusal::Head(_)))); + assert_eq!(queue.poll_used(), Ok(Some(Used { head: 1, written: 64 }))); + assert_eq!(queue.poll_used(), Ok(Some(Used { head: 2, written: 3 }))); + assert_eq!(queue.poll_used(), Ok(None)); +} + +/// §2.7.5.1: "A device MUST NOT write to any descriptor table entry", and it +/// reaches the table all the same. One rewritten into a loop is a loop the +/// driver never follows: a chain is retired by what the driver recorded of +/// it, and the table is never loaded. +#[test] +fn a_descriptor_table_the_device_rewrote_into_a_loop_is_never_followed() { + let machine = Machine::new(); + let mut queue = queue(&machine, 0); + let ring = ring(&machine, 0); + let chain = [Buffer::readable(BUFFERS, 16), Buffer::writable(BUFFERS + 16, 64)]; + let _ = queue.publish(0, &chain); + let _ = queue.publish(2, &one(64, true)); + // Descriptor 0 chains to itself; descriptor 1 chains past the table. + ring.scribble_desc(0, 1, 0); + ring.scribble_desc(1, 1, 0xFFFF); + ring.use_chain(0, 64); + + machine.forget_trace(); + assert_eq!(queue.poll_used(), Ok(Some(Used { head: 0, written: 64 }))); + let table = 0..desc_bytes(SIZE); + assert!(machine + .trace() + .iter() + .all(|event| !matches!(event, Event::Load { at, .. } if table.contains(at)))); + // Exactly the two descriptors of the chain came back: both take a chain + // again, and descriptor 2 is in flight still. + let _ = queue.publish(0, &chain); + assert_eq!(ring.chain(0), chain); + ring.use_chain(2, 0); + assert_eq!(queue.poll_used(), Ok(Some(Used { head: 2, written: 0 }))); +} + +/// §2.7.13.3: "idx always increments, and wraps naturally at 65536" — both +/// indices, past the wrap, on a ring of eight. +#[test] +fn both_indices_wrap_at_65536() { + let machine = Machine::new(); + let mut queue = queue(&machine, 0); + let ring = ring(&machine, 0); + for nth in 0..70_000u32 { + let head = (nth % 3) as u16; + let _ = queue.publish(head, &one(64, true)); + assert_eq!(ring.avail_entry(nth as u16), head); + ring.use_chain(head as u32, nth % 65); + assert_eq!(queue.poll_used(), Ok(Some(Used { head, written: nth % 65 })), "at {nth}"); + machine.forget_trace(); + } + assert_eq!(ring.avail_idx(), (70_000u32 % 65_536) as u16); + assert_eq!(queue.poll_used(), Ok(None)); +} + +/// xorshift64*: every draw comes from the seed. +struct Draws(u64); + +impl Draws { + fn next(&mut self) -> u64 { + self.0 ^= self.0 >> 12; + self.0 ^= self.0 << 25; + self.0 ^= self.0 >> 27; + self.0.wrapping_mul(0x2545_F491_4F6C_DD1D) + } + + fn below(&mut self, bound: u64) -> u64 { + self.next() % bound + } +} + +/// A device that uses buffers in any order (§2.6 permits it) and, between +/// them, writes elements of its own invention. The oracle is a model of what +/// is in flight kept apart from the queue's: each element is believed exactly +/// when it names a chain in flight and no more bytes than that chain's +/// writable elements hold, and an index past the available one is refused. +/// +/// Many short lives rather than one long one: every element the device invents +/// spends one a chain was owed, so a queue this device has been at for long +/// has only chains nothing can answer. +#[test] +fn a_hostile_device_is_believed_only_where_the_model_agrees() { + let mut seeds = Draws(0x9E37_79B9_7F4A_7C15); + let mut believed = 0u32; + let mut refused = [0u32; 4]; + for _ in 0..2_000 { + let seed = seeds.next() | 1; + let mut draws = Draws(seed); + let machine = Machine::new(); + let mut queue = queue(&machine, 0); + let ring = ring(&machine, 0); + // What is in flight, by head: how many descriptors, and writable bytes. + let mut flying: BTreeMap = BTreeMap::new(); + let (mut made, mut taken) = (0u16, 0u16); + + for step in 0..64 { + let held = |flying: &BTreeMap, at: u16| { + flying.iter().any(|(head, (descs, _))| (*head..*head + *descs).contains(&at)) + }; + // The driver: a chain of one to three, where the table is free. + let descs = 1 + draws.below(3) as u16; + let head = draws.below((SIZE - descs + 1) as u64) as u16; + if (head..head + descs).all(|at| !held(&flying, at)) { + let readable = draws.below(descs as u64 + 1) as u16; + let chain: Vec = (0..descs) + .map(|nth| Buffer { + addr: BUFFERS + nth as u64 * 0x100, + len: draws.below(200) as u32, + writable: nth >= readable, + }) + .collect(); + let writable = chain.iter().filter(|b| b.writable).map(|b| b.len).sum(); + let _ = queue.publish(head, &chain); + flying.insert(head, (descs, writable)); + made = made.wrapping_add(1); + } + + // The device: up to three elements, each honest or invented. + let mut written: Vec<(u32, u32)> = Vec::new(); + for _ in 0..draws.below(4) { + let honest = flying.iter().nth(draws.below(SIZE as u64) as usize); + let element = match (draws.below(8), honest) { + (0, _) => (draws.next() as u32, draws.next() as u32), + (1, _) => (draws.below(SIZE as u64 * 2) as u32, draws.below(400) as u32), + (_, Some((head, (_, writable)))) => { + (*head as u32, draws.below(*writable as u64 + 2) as u32) + } + (_, None) => continue, + }; + ring.write_used(ring.used_idx().wrapping_add(written.len() as u16), element.0, element.1); + written.push(element); + } + let room = made.wrapping_sub(taken); + ring.set_used_idx(ring.used_idx().wrapping_add(written.len() as u16)); + if written.len() as u16 > room { + let used = ring.used_idx(); + assert_eq!( + queue.poll_used(), + Err(UsedRefusal::Jumped { used, taken, available: made }), + "seed {seed:#x} step {step}" + ); + refused[3] += 1; + // The device takes back what it was never given. + ring.set_used_idx(taken.wrapping_add(room)); + written.truncate(room as usize); + } + + for (id, len) in written { + let got = queue.poll_used(); + taken = taken.wrapping_add(1); + let chain = u16::try_from(id).ok().and_then(|head| flying.get(&head).map(|c| (head, *c))); + let want = match chain { + _ if id >= SIZE as u32 => Err(UsedRefusal::Head(Refused::PastTable { + value: id as u64, + len: SIZE as u64, + })), + None => Err(UsedRefusal::NoChain { head: id as u16 }), + Some((head, (_, writable))) if len > writable => Err(UsedRefusal::Written { + head, + len: Refused::PastBound { value: len as u64, bound: writable as u64 }, + }), + Some((head, _)) => { + flying.remove(&head); + believed += 1; + Ok(Some(Used { head, written: len })) + } + }; + match want { + Err(UsedRefusal::Head(_)) => refused[0] += 1, + Err(UsedRefusal::NoChain { .. }) => refused[1] += 1, + Err(UsedRefusal::Written { .. }) => refused[2] += 1, + Err(UsedRefusal::Jumped { .. }) | Ok(_) => {} + } + assert_eq!(got, want, "seed {seed:#x} step {step}: element ({id:#x}, {len:#x})"); + } + assert_eq!(queue.poll_used(), Ok(None), "seed {seed:#x} step {step}"); + machine.forget_trace(); + } + } + // The runs reached every verdict, so none of them is asserted of nothing. + assert!(believed > 5_000, "only {believed} chain(s) ever came back"); + assert!(refused.iter().all(|count| *count > 100), "refusals by kind: {refused:?}"); +} + +// --- the driver's own mistakes, which no device causes --- + +#[test] +#[should_panic(expected = "takes descriptor 3, which is in flight")] +fn a_chain_over_a_descriptor_in_flight_is_the_drivers_mistake() { + let machine = Machine::new(); + let mut queue = queue(&machine, 0); + let _ = queue.publish(2, &[Buffer::readable(BUFFERS, 8), Buffer::writable(BUFFERS + 8, 8)]); + let _ = queue.publish(3, &one(64, true)); +} + +/// §2.7.4.2: "The driver MUST place any device-writable descriptor elements +/// after any device-readable descriptor elements." +#[test] +#[should_panic(expected = "a device-readable element after a device-writable one")] +fn a_readable_element_after_a_writable_one_is_the_drivers_mistake() { + let machine = Machine::new(); + let mut queue = queue(&machine, 0); + let _ = queue.publish(0, &[Buffer::writable(BUFFERS, 8), Buffer::readable(BUFFERS + 8, 8)]); +} + +#[test] +#[should_panic(expected = "runs past a table of 8")] +fn a_chain_past_the_table_is_the_drivers_mistake() { + let machine = Machine::new(); + let mut queue = queue(&machine, 0); + let _ = queue.publish(7, &[Buffer::readable(BUFFERS, 8), Buffer::writable(BUFFERS + 8, 8)]); +} + +/// §4.1.4.3.2: "the driver MUST NOT write a value which is not a power of 2 +/// to queue_size". +#[test] +#[should_panic(expected = "6 descriptors is no queue size")] +fn a_queue_size_that_is_no_power_of_two_is_the_drivers_mistake() { + let machine = Machine::new(); + let _ = Virtqueue::new(machine, 0, 6, Parts::contiguous(0, 6)); +} + +#[test] +#[should_panic(expected = "runs past a grant of 0x8000 bytes")] +fn a_ring_past_the_grant_is_the_drivers_mistake() { + let machine = Machine::new(); + let _ = Virtqueue::new(machine, 0, SIZE, Parts::contiguous(GRANT_BYTES - 0x80, SIZE)); +} + +/// §2.7.1: each part on its alignment. A used ring two bytes off a multiple +/// of four is one the device may not take. +#[test] +#[should_panic(expected = "its used ring at 0x9a is not on a multiple of 4")] +fn a_part_off_its_alignment_is_the_drivers_mistake() { + let machine = Machine::new(); + let _ = Virtqueue::new(machine, 0, SIZE, Parts { desc: 0, avail: 128, used: 154 }); +} diff --git a/userland/netstack/Cargo.toml b/userland/netstack/Cargo.toml index 0c28c27b7ef..ea4f0beb032 100644 --- a/userland/netstack/Cargo.toml +++ b/userland/netstack/Cargo.toml @@ -7,7 +7,9 @@ license = "MIT OR Apache-2.0" [dependencies] toyos-abi = { path = "../../toyos-abi" } toyos = { path = "../../toyos" } +toyos-device-memory = { path = "../../toyos-device-memory" } toyos-i219 = { path = "../../toyos-i219" } +toyos-virtio = { path = "../../toyos-virtio" } toyos-tco = { path = "../../toyos-tco" } toyos-mdns = { path = "mdns" } toyos-dns = { path = "../../toyos-dns" } diff --git a/userland/netstack/src/card.rs b/userland/netstack/src/card.rs index b83a0aae0e1..3869f5d0b7d 100644 --- a/userland/netstack/src/card.rs +++ b/userland/netstack/src/card.rs @@ -172,9 +172,10 @@ impl Card { /// Say what the driver counted, once a pass and after every frame the pass /// sent: a line per refused descriptor is itself more frames to send. + /// virtio counts nothing: what its device is not believed on ends it. pub fn report(&self) { match self { - Self::Virtio(nic) => nic.report(), + Self::Virtio(_) => {} Self::Intel(nic) => nic.report(), } } diff --git a/userland/netstack/src/device.rs b/userland/netstack/src/device.rs index 7d3a4f19086..74fe0aa6092 100644 --- a/userland/netstack/src/device.rs +++ b/userland/netstack/src/device.rs @@ -1,10 +1,124 @@ -//! What both NIC drivers need from the substrate beside the SDK's -//! `toyos::volatile::Window`: the kernel's word for a call a bring-up cannot go -//! on without, and the latch a diagnostic is printed on. +//! What both NIC drivers need from the substrate: `toyos-device-memory`'s +//! boundary over the SDK's `toyos::volatile::Window`, the kernel's word for a +//! call a bring-up cannot go on without, and the latch a diagnostic is printed +//! on. +//! +//! **[`Bar`] and [`Grant`] are the one implementation of the boundary both +//! driver crates are written against**, an instruction deep: every offset +//! that reaches them a driver crate has bounded, and the two barriers are here +//! and nowhere else in this program. Neither owns its mapping: a driver's +//! holder keeps the `SharedMemory` and the `DmaRegion` for as long as it keeps +//! the driver. use std::cell::Cell; +use std::sync::atomic::{fence, Ordering}; +use toyos::volatile::Window; use toyos_abi::syscall::SyscallError; +use toyos_device_memory::{DmaBuffers, Registers}; + +/// A mapped BAR, as a driver reaches its device's registers. +#[derive(Clone, Copy)] +pub struct Bar(Window); + +impl Bar { + pub fn over(window: Window) -> Self { + Self(window) + } +} + +impl Registers for Bar { + fn bytes(&self) -> usize { + self.0.bytes() + } + + fn read8(&self, at: usize) -> u8 { + self.0.read(at) + } + + fn read16(&self, at: usize) -> u16 { + self.0.read(at) + } + + fn read32(&self, at: usize) -> u32 { + self.0.read(at) + } + + fn write8(&self, at: usize, value: u8) { + self.0.write(at, value); + } + + fn write16(&self, at: usize, value: u16) { + self.0.write(at, value); + } + + fn write32(&self, at: usize, value: u32) { + self.0.write(at, value); + } +} + +/// A DMA grant, as a driver reaches its rings and descriptors. +#[derive(Clone, Copy)] +pub struct Grant { + window: Window, + /// Where the device reaches the grant's first byte. Not a physical address: + /// what the unit translates for this function and for nothing else. + device_base: u64, +} + +impl Grant { + pub fn over(window: Window, device_base: u64) -> Self { + Self { window, device_base } + } + + /// The same bytes, for the holder's own view of them: the frames, which no + /// driver crate reaches. + pub fn window(&self) -> Window { + self.window + } +} + +impl DmaBuffers for Grant { + fn bytes(&self) -> usize { + self.window.bytes() + } + + fn device_addr(&self, at: usize) -> u64 { + self.device_base + at as u64 + } + + fn read16(&self, at: usize) -> u16 { + self.window.read(at) + } + + fn read32(&self, at: usize) -> u32 { + self.window.read(at) + } + + fn read64(&self, at: usize) -> u64 { + self.window.read(at) + } + + fn write16(&self, at: usize, value: u16) { + self.window.write(at, value); + } + + fn write32(&self, at: usize, value: u32) { + self.window.write(at, value); + } + + fn write64(&self, at: usize, value: u64) { + self.window.write(at, value); + } + + fn publish(&self) { + fence(Ordering::Release); + } + + fn observe(&self) { + fence(Ordering::Acquire); + } +} /// The kernel refused a call a bring-up cannot go on without, and the word is /// the kernel's own. diff --git a/userland/netstack/src/i219.rs b/userland/netstack/src/i219.rs index 14f25d26f16..f1ebc450c62 100644 --- a/userland/netstack/src/i219.rs +++ b/userland/netstack/src/i219.rs @@ -1,5 +1,6 @@ -//! The I219, as netstack drives it: the four implementations the driver's logic is -//! written against, and nothing else. +//! The I219, as netstack drives it: the clock and the claim the driver's logic +//! is written against beside `device.rs`'s register window and grant, and +//! nothing else. //! //! What the kernel keeps is the *claim*: config space, the interrupt vector it //! programmed into this function, and the address space the function @@ -7,7 +8,7 @@ //! [`PciDev::dma_alloc`] answered; one this driver invented instead is refused //! at the unit and recorded against this process. //! -//! **Two views of one grant, and the split is the boundary itself.** [`Grant`] +//! **Two views of one grant, and the split is the boundary itself.** `Grant` //! is the driver's, and it reaches descriptors — sixteen bytes each, read and //! written a word at a time. [`Nic::frames`] is netstack's, and it reaches //! payloads. The driver never touches a payload and netstack never touches a @@ -21,32 +22,10 @@ use toyos::shm::SharedMemory; use toyos::volatile::Window; use toyos::{DmaRegion, PciDev}; use toyos_abi::syscall::SyscallError; -use toyos_i219::{Clock, DmaBuffers, Interrupts, Registers}; +use toyos_device_memory::Registers; +use toyos_i219::{Clock, Interrupts}; -use crate::device::{KernelRefused, Latch}; - -/// The register window. The offsets are `toyos-i219`'s own constants, all under -/// `toyos_i219::regs::REGISTER_BYTES`, which its `open` refuses a shorter window -/// than. -pub struct Bar { - window: Window, - /// The mapping the window points into, kept for as long as the window is. - _mapped: SharedMemory, -} - -impl Registers for Bar { - fn bytes(&self) -> usize { - self.window.bytes() - } - - fn read(&self, reg: usize) -> u32 { - self.window.read::(reg) - } - - fn write(&self, reg: usize, value: u32) { - self.window.write::(reg, value); - } -} +use crate::device::{Bar, Grant, KernelRefused, Latch}; /// The machine's monotonic clock: one syscall, which the kernel serves from an /// anchor plus the timestamp counter — and the thread's own sleep, which is how @@ -64,40 +43,6 @@ impl Clock for Monotonic { } } -/// The DMA grant, as the driver reaches its descriptors. -pub struct Grant { - window: Window, - device_base: u64, - /// The region the window points into, kept for as long as the window is. - _region: DmaRegion, -} - -impl DmaBuffers for Grant { - fn bytes(&self) -> usize { - self.window.bytes() - } - - fn device_addr(&self, at: usize) -> u64 { - self.device_base + at as u64 - } - - fn read(&self, at: usize) -> u64 { - self.window.read::(at) - } - - fn write(&self, at: usize, word: u64) { - self.window.write::(at, word); - } - - fn publish(&self) { - std::sync::atomic::fence(std::sync::atomic::Ordering::Release); - } - - fn observe(&self) { - std::sync::atomic::fence(std::sync::atomic::Ordering::Acquire); - } -} - /// The claim's interrupt record, handed over as the kernel answered it. /// /// Shared with [`Nic`] rather than moved into the driver: the poller watches the @@ -137,7 +82,13 @@ impl std::fmt::Display for Opening { /// The card, brought up and driving. pub struct Nic { + /// Above the mappings it reaches: dropping the driver writes the part's + /// registers, and fields drop in the order they are declared. driver: RefCell, + /// Held for their mappings' lives: the register window points into the + /// first and the grant's into the second. + _mapped: SharedMemory, + _region: DmaRegion, claim: Rc, /// netstack's view of the same grant: the frame payloads, and nothing else. frames: Window, @@ -146,8 +97,8 @@ pub struct Nic { reported: Latch<(toyos_i219::Counters, toyos_i219::Link)>, } -/// The claim's register window, mapped. -fn registers(dev: &PciDev) -> Result { +/// The claim's register window, and the mapping it points into. +fn registers(dev: &PciDev) -> Result<(Bar, SharedMemory), Opening> { let info = dev .describe() .map_err(KernelRefused::on("the claim's description")) @@ -172,25 +123,26 @@ fn registers(dev: &PciDev) -> Result { info.dev, info.func, ); - // SAFETY: `map_bar` answered `bytes` bytes of live mapping, and - // `mapped` is moved into the `Bar` that carries the window. + // SAFETY: `map_bar` answered `bytes` bytes of live mapping, and `mapped` + // goes to the caller with the window, which keeps both for as long as it + // keeps either. let window = unsafe { Window::new(mapped.as_ptr(), bytes as usize) }; - Ok(Bar { window, _mapped: mapped }) + Ok((Bar::over(window), mapped)) } /// Everything the claim is asked for before the part is reached, in the order /// it is asked: the register window, the part stopped, then one grant — as the -/// driver reaches its descriptors, and as netstack reaches its frames. -fn granted(dev: &PciDev) -> Result<(Bar, Grant, Window), Opening> { - let bar = registers(dev)?; +/// driver reaches its descriptors — with the two mappings they point into. +fn granted(dev: &PciDev) -> Result<(Bar, Grant, SharedMemory, DmaRegion), Opening> { + let (bar, mapped) = registers(dev)?; // Before the grant: the grant is what starts the function mastering, and a // part a previous holder left receiving would write into its descriptors. // Said first: whether the kernel's release reset the part is read here, in // the receive and transmit enables it handed over. crate::say!( "netstack: I219: inherited RCTL {:#010x} TCTL {:#010x}", - toyos_i219::Registers::read(&bar, toyos_i219::regs::RCTL), - toyos_i219::Registers::read(&bar, toyos_i219::regs::TCTL) + bar.read32(toyos_i219::regs::RCTL), + bar.read32(toyos_i219::regs::TCTL) ); toyos_i219::quiesce(&bar); let region = dev @@ -199,11 +151,11 @@ fn granted(dev: &PciDev) -> Result<(Bar, Grant, Window), Opening> { .map_err(Opening::Kernel)?; let device_base = region.device_addr; // SAFETY: `dma_alloc` answered `GRANT_BYTES` bytes of live mapping, and - // `region` is moved into the `Grant`, which its caller keeps for as long - // as it keeps either window. - let frames = + // `region` goes to the caller with the grant, which keeps both for as + // long as it keeps either. + let window = unsafe { Window::new(region.memory.as_ptr(), toyos_i219::GRANT_BYTES as usize) }; - Ok((bar, Grant { window: frames, device_base, _region: region }, frames)) + Ok((bar, Grant::over(window, device_base), mapped, region)) } /// What a bring-up says about itself: a sentence each for what it asked the @@ -243,7 +195,10 @@ impl Nic { /// Take the claim's register window and one grant, and bring the part up. pub fn open(dev: PciDev, part: toyos_i219::Part) -> Result { let dev = Rc::new(dev); - let (bar, grant, frames) = granted(&dev)?; + // Bound before the driver, so one that is refused lets its part go + // while both are still mapped. + let (bar, grant, mapped, region) = granted(&dev)?; + let frames = grant.window(); let driver = toyos_i219::I219::open(part, bar, Monotonic, grant, Claim(Rc::clone(&dev))) .map_err(Opening::Driver)?; @@ -251,6 +206,8 @@ impl Nic { say_brought_up(driver.brought_up()); Ok(Self { driver: RefCell::new(driver), + _mapped: mapped, + _region: region, claim: dev, frames, mac, diff --git a/userland/netstack/src/virtio_net.rs b/userland/netstack/src/virtio_net.rs index 759dc15266f..de692807e1f 100644 --- a/userland/netstack/src/virtio_net.rs +++ b/userland/netstack/src/virtio_net.rs @@ -2,20 +2,24 @@ //! //! What the kernel keeps is the *claim*: config space, the interrupt vector it //! programmed into this function's MSI-X table, and the address space the -//! function translates through. Everything below that is this file, and none of -//! it is authority: an address written into a descriptor is one -//! `PciDev::dma_alloc` answered, and one this driver invents instead is refused -//! at the unit and recorded against this process. +//! function translates through. Everything below that is this file and +//! `toyos-virtio`, and none of it is authority: an address written into a +//! descriptor is one `PciDev::dma_alloc` answered, and one this driver invents +//! instead is refused at the unit and recorded against this process. //! -//! **The device is not trusted, and neither is a used-ring element it wrote.** -//! [`Rings::parse_used`] bounds the head against the descriptor table's own -//! length and the written length against what that chain was given; an element -//! failing either is counted and dropped, because losing a token costs -//! throughput and believing a bad one costs memory. The device is on the far -//! side of the same boundary wherever the driver sits. +//! **The transport and the rings are `toyos-virtio`'s**, where every word the +//! device writes is bounded and every refusal is host-tested, over +//! `device.rs`'s register window and grant. What is this file's is the +//! network device (§5.1): which buffer goes with which head, the header in +//! front of every frame, and what becomes of a refusal. //! -//! Virtio 1.2 throughout: §4.1.4 for the PCI capability layout, §2.7 for the -//! split virtqueue, §5.1 for the network device. +//! **A used ring the device is not believed on ends this program**, by the +//! refusal's own name ([`finished`]): no conforming device writes one, the +//! refused element spent one a chain in flight was owed, and every frame +//! after it is one that silently never arrives. +//! +//! Virtio 1.2 throughout: §4.1.4 for the PCI capability layout, §5.1 for the +//! network device. use std::cell::RefCell; @@ -23,16 +27,15 @@ use toyos::shm::SharedMemory; use toyos::volatile::Window; use toyos::{DmaRegion, PciDev}; use toyos_abi::syscall::{RegWidth, SyscallError}; +use toyos_device_memory::DmaBuffers; +use toyos_virtio::pci::{Layout, Live, Offer, VendorCap}; +use toyos_virtio::queue::{avail_bytes, desc_bytes, Buffer, Parts, Published, Used, Virtqueue}; -use crate::device::{KernelRefused, Latch}; +use crate::device::{Bar, Grant, KernelRefused}; -/// PCI's own vendor-specific capability id; virtio's four config windows are -/// all published under it (§4.1.4). +/// PCI's own vendor-specific capability id; virtio's config structures are all +/// published under it (§4.1.4). const CAP_ID_VENDOR: u8 = 0x09; -const CAP_COMMON_CFG: u8 = 1; -const CAP_NOTIFY_CFG: u8 = 2; -const CAP_ISR_CFG: u8 = 3; -const CAP_DEVICE_CFG: u8 = 4; /// Where the capability list starts, and how far a walk may follow it. The /// pointer is the *device's*, so a malformed or cyclic chain ends the walk @@ -40,47 +43,14 @@ const CAP_DEVICE_CFG: u8 = 4; const CAPABILITIES_PTR: u32 = 0x34; const MAX_CAPABILITIES: usize = 48; -// Common configuration structure, §4.1.4.3. -const COMMON_DEVICE_FEATURE_SELECT: usize = 0x00; -const COMMON_DEVICE_FEATURE: usize = 0x04; -const COMMON_DRIVER_FEATURE_SELECT: usize = 0x08; -const COMMON_DRIVER_FEATURE: usize = 0x0C; -const COMMON_MSIX_CONFIG: usize = 0x10; -const COMMON_DEVICE_STATUS: usize = 0x14; -const COMMON_QUEUE_SELECT: usize = 0x16; -const COMMON_QUEUE_SIZE: usize = 0x18; -const COMMON_QUEUE_MSIX: usize = 0x1A; -const COMMON_QUEUE_ENABLE: usize = 0x1C; -const COMMON_QUEUE_NOTIFY_OFF: usize = 0x1E; -const COMMON_QUEUE_DESC: usize = 0x20; -const COMMON_QUEUE_DRIVER: usize = 0x28; -const COMMON_QUEUE_DEVICE: usize = 0x30; - -const STATUS_ACKNOWLEDGE: u32 = 1; -const STATUS_DRIVER: u32 = 2; -const STATUS_DRIVER_OK: u32 = 4; -const STATUS_FEATURES_OK: u32 = 8; - -const VIRTIO_F_VERSION_1: u64 = 1 << 32; -/// §6: the device's buffer addresses are the platform's, so what translates any -/// other function's translates this one's. A device never offered it is one no -/// unit sees; the driver accepts it wherever it is offered, and the negotiated -/// set is logged for a gate to read back. -const VIRTIO_F_ACCESS_PLATFORM: u64 = 1 << 33; +/// §5.1.3: the device has a MAC address of its own to read. const VIRTIO_NET_F_MAC: u64 = 1 << 5; +/// §6, for the feature line: the bit `toyos-virtio` accepts wherever it is +/// offered, and a gate reads back. +const VIRTIO_F_ACCESS_PLATFORM: u64 = toyos_virtio::pci::VIRTIO_F_ACCESS_PLATFORM; -/// The one MSI-X table entry the kernel programs. A device answering -/// [`NO_VECTOR`] for it has refused the binding (§4.1.5.1.2). +/// The one MSI-X table entry the kernel programs. const MSIX_ENTRY: u16 = 0; -const NO_VECTOR: u16 = 0xFFFF; - -const VIRTQ_DESC_F_WRITE: u16 = 2; -const DESC_BYTES: usize = 16; -const AVAIL_IDX_OFF: usize = 2; -const AVAIL_RING_OFF: usize = 4; -const USED_IDX_OFF: usize = 2; -const USED_RING_OFF: usize = 4; -const USED_ELEM_BYTES: usize = 8; const RX_QUEUE: u16 = 0; const TX_QUEUE: u16 = 1; @@ -103,198 +73,46 @@ pub const TX_BUF_SIZE: usize = 4096; /// twelve bytes with `VERSION_1`, `num_buffers` included (§5.1.6). pub const NET_HDR_SIZE: usize = 12; -/// The grant's layout: the receive queue's three rings on a page each, the +/// The grant's layout: the receive queue's three parts on a page each, the /// transmit queue's three in one page, then the frames. One grant, because /// every byte of it is this process's own — what a wrong offset here corrupts /// is this driver's memory and nobody else's. -const OFF_RX_DESC: usize = 0x0000; -const OFF_RX_AVAIL: usize = 0x1000; -const OFF_RX_USED: usize = 0x2000; -const OFF_TX_RINGS: usize = 0x3000; +const RX_PARTS: Parts = Parts { desc: 0x0000, avail: 0x1000, used: 0x2000 }; +const TX_PARTS: Parts = Parts::contiguous(0x3000, TX_QUEUE_SIZE); const OFF_RX_BUFS: usize = 0x4000; const OFF_TX_BUFS: usize = OFF_RX_BUFS + RX_BUF_COUNT * RX_BUF_SIZE; const GRANT_BYTES: u64 = (OFF_TX_BUFS + TX_BUF_COUNT * TX_BUF_SIZE) as u64; -const fn tx_avail_off() -> usize { - (TX_QUEUE_SIZE as usize * DESC_BYTES + 1) & !1 -} - -const fn tx_used_off() -> usize { - (tx_avail_off() + AVAIL_RING_OFF + TX_QUEUE_SIZE as usize * 2 + 3) & !3 -} - const _: () = { - assert!(RX_QUEUE_SIZE as usize * DESC_BYTES <= OFF_RX_AVAIL - OFF_RX_DESC); - assert!(AVAIL_RING_OFF + RX_QUEUE_SIZE as usize * 2 <= OFF_RX_USED - OFF_RX_AVAIL); - assert!( - USED_RING_OFF + RX_QUEUE_SIZE as usize * USED_ELEM_BYTES <= OFF_TX_RINGS - OFF_RX_USED - ); - assert!(tx_used_off() + USED_RING_OFF + TX_QUEUE_SIZE as usize * USED_ELEM_BYTES <= 0x1000); - assert!(OFF_TX_RINGS + 0x1000 <= OFF_RX_BUFS); + assert!(desc_bytes(RX_QUEUE_SIZE) <= RX_PARTS.avail - RX_PARTS.desc); + assert!(avail_bytes(RX_QUEUE_SIZE) <= RX_PARTS.used - RX_PARTS.avail); + assert!(RX_PARTS.end(RX_QUEUE_SIZE) <= TX_PARTS.desc); + assert!(TX_PARTS.end(TX_QUEUE_SIZE) <= OFF_RX_BUFS); assert!(NET_HDR_SIZE < RX_BUF_SIZE && NET_HDR_SIZE < TX_BUF_SIZE); }; -/// One split-virtqueue descriptor (§2.7.5). -#[repr(C)] -#[derive(Clone, Copy)] -struct Desc { - addr: u64, - len: u32, - flags: u16, - next: u16, -} - -/// Why a used-ring element was refused. The device wrote it, so it is a claim -/// about this driver's own memory and the only safe answer is to drop it. -#[derive(Clone, Copy, PartialEq, Eq, Debug)] -enum UsedRefusal { - /// A head index past the descriptor table. - Head(u32), - /// A head this driver has no chain published at — a completion for a - /// buffer it has already taken back, or one it never posted. - NoChain(u16), - /// More bytes written than the chain was given. - Written { id: u16, len: u32, chain: u32 }, -} - -/// One queue's rings and the bookkeeping that says what is in flight. -struct Rings { - desc: Window, - avail: Window, - used: Window, - size: u16, - last_used: u16, - /// Bytes the chain at each head was given, indexed by head: the one bound a - /// device-reported length is checked against, and 0 means no chain is - /// published there. - chain_bytes: Vec, - refused: u32, - /// Where this queue's doorbell sits, from `queue_notify_off`. - notify_off: u16, -} - -impl Rings { - fn new(desc: Window, avail: Window, used: Window, size: u16) -> Self { - desc.zero(); - avail.zero(); - used.zero(); - Self { - desc, - avail, - used, - size, - last_used: 0, - chain_bytes: vec![0; size as usize], - refused: 0, - notify_off: 0, - } - } - - /// Publish a one-descriptor chain at `head` and ring the doorbell. - fn submit(&mut self, head: u16, addr: u64, len: u32, writable: bool, doorbell: Doorbell) { - self.chain_bytes[head as usize] = len; - let flags = if writable { VIRTQ_DESC_F_WRITE } else { 0 }; - // One descriptor per chain, so nothing here sets `VIRTQ_DESC_F_NEXT` - // and `next` is never followed. - self.desc.write(head as usize * DESC_BYTES, Desc { addr, len, flags, next: 0 }); - - let avail_idx: u16 = self.avail.read(AVAIL_IDX_OFF); - self.avail.write(AVAIL_RING_OFF + (avail_idx % self.size) as usize * 2, head); - // The device must see the ring entry before the index that publishes - // it, and the index before the doorbell (§2.7.13). - std::sync::atomic::fence(std::sync::atomic::Ordering::Release); - self.avail.write(AVAIL_IDX_OFF, avail_idx.wrapping_add(1)); - std::sync::atomic::fence(std::sync::atomic::Ordering::Release); - doorbell.ring(); - } - - /// The next completion, or `None`. A refused element is counted and - /// skipped, never returned, so one bad element cannot hide the ones behind - /// it. - /// - /// The chain is retired here, so a device that reports the same head twice - /// is refused by the second report rather than handing a caller a buffer it - /// has already been given. - fn poll_used(&mut self) -> Option<(u16, u32)> { - loop { - let used_idx: u16 = self.used.read(USED_IDX_OFF); - if used_idx == self.last_used { - return None; - } - std::sync::atomic::fence(std::sync::atomic::Ordering::Acquire); - let at = USED_RING_OFF + (self.last_used % self.size) as usize * USED_ELEM_BYTES; - let id: u32 = self.used.read(at); - let len: u32 = self.used.read(at + 4); - self.last_used = self.last_used.wrapping_add(1); - match self.parse_used(id, len) { - Ok((head, written)) => { - self.chain_bytes[head as usize] = 0; - return Some((head, written)); - } - Err(_) => { - self.refused = self.refused.saturating_add(1); - continue; - } - } - } - } - - /// What a used element must satisfy. Separate from the volatile reads, so - /// the tests below can drive it with elements no device would send. - fn parse_used(&self, id: u32, len: u32) -> Result<(u16, u32), UsedRefusal> { - if id as usize >= self.chain_bytes.len() { - return Err(UsedRefusal::Head(id)); - } - let id = id as u16; - let chain = self.chain_bytes[id as usize]; - if chain == 0 { - return Err(UsedRefusal::NoChain(id)); - } - if len > chain { - return Err(UsedRefusal::Written { id, len, chain }); - } - Ok((id, len)) - } -} - -/// Where one queue's doorbell is, taken out of the notify window once. -#[derive(Clone, Copy)] -struct Doorbell { - window: Window, - queue: u16, -} - -impl Doorbell { - fn ring(self) { - self.window.write(0, self.queue); - } -} - /// Why the device was not brought up. Each keeps its own word: a machine with /// no such device and one that refused a feature set ask different things of a /// caller. #[derive(Clone, Copy, PartialEq, Eq, Debug)] pub enum Refusal { - MissingCap(&'static str), - ResetUnanswered, - FeaturesRefused { offered: u64, status: u32 }, - NoVector(&'static str), + /// What the device published or answered, as the transport refused it. + Device(toyos_virtio::pci::Refusal), Kernel(KernelRefused), /// The claim answered a configuration read it had to refuse. Unbounded(&'static str, u32), } +impl From for Refusal { + fn from(why: toyos_virtio::pci::Refusal) -> Self { + Self::Device(why) + } +} + impl std::fmt::Display for Refusal { fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { match self { - Self::MissingCap(what) => write!(f, "it published no usable {what}"), - Self::ResetUnanswered => write!(f, "it never zeroed DEVICE_STATUS for its reset"), - Self::FeaturesRefused { offered, status } => write!( - f, - "it refused the feature set {offered:#x} this driver accepted, leaving \ - DEVICE_STATUS={status:#x} without FEATURES_OK" - ), - Self::NoVector(what) => write!(f, "it refused a vector for {what}"), + Self::Device(why) => write!(f, "{why}"), Self::Kernel(why) => write!(f, "{why}"), Self::Unbounded(what, at) => write!( f, @@ -305,6 +123,18 @@ impl std::fmt::Display for Refusal { } } +/// The next chain the device has finished with on `rings`, or `None`. +/// +/// **Every way the used ring is not believed ends the device's use**: a head +/// past the table, a head with no chain in flight, more bytes than a chain's +/// writable ones, and an index past what was made available. Nothing drives +/// this NIC from there, and this dies where it can be read. +fn finished(rings: &mut Virtqueue) -> Option { + rings + .poll_used() + .unwrap_or_else(|why| panic!("netstack: this NIC cannot be driven on — {why}")) +} + /// The bound the capability walk rests on, asked once before the walk. /// /// **The walk below indexes configuration space by numbers the *device* wrote** @@ -338,50 +168,34 @@ fn config_space_is_bounded(dev: &PciDev) -> Result<(), Refusal> { /// The virtio-net function, brought up and driving. pub struct VirtioNet { dev: PciDev, - rx_doorbell: Doorbell, - /// Held for their mappings' lives: every window above and below points into - /// one of these two. + device: Live, + /// Held for their mappings' lives: the register window points into the + /// first and the grant's into the second. _bar: SharedMemory, - _grant: DmaRegion, - dma: Window, - /// Where the device reaches the grant's first byte. Not a physical address: - /// what the unit translates for this function and for nothing else. - dma_device_addr: u64, - rx: RefCell, + _region: DmaRegion, + grant: Grant, + rx: RefCell>, tx: RefCell, - reported: Latch, mac: [u8; 6], } impl VirtioNet { - /// Bring the function up: the capability chain, the reset handshake, the - /// feature negotiation, both queues and the vector binding — in the order - /// virtio 1.2 §3.1.1 fixes. + /// Bring the function up: the capability chain, then the reset, the + /// feature negotiation, both queues and their vectors in the order + /// `toyos-virtio`'s types fix, which is virtio 1.2 §3.1.1's. pub fn open(dev: PciDev) -> Result { let info = dev.describe().map_err(KernelRefused::on("the claim's description")).map_err(Refusal::Kernel)?; config_space_is_bounded(&dev)?; - let caps = Capabilities::walk(&dev); - let common_cap = caps.find(CAP_COMMON_CFG).ok_or(Refusal::MissingCap("COMMON_CFG"))?; - let notify_cap = caps.find(CAP_NOTIFY_CFG).ok_or(Refusal::MissingCap("NOTIFY_CFG"))?; - // Found and not used: the vector is bound, so nothing here polls the - // ISR status byte. Its absence still refuses the device — a chain - // missing it is a chain this driver did not understand. - let isr_cap = caps.find(CAP_ISR_CFG).ok_or(Refusal::MissingCap("ISR_CFG"))?; - let device_cap = caps.find(CAP_DEVICE_CFG).ok_or(Refusal::MissingCap("DEVICE_CFG"))?; - - // One BAR carries all four on every virtio device in reach, and the - // kernel hands out a BAR at a time; a device that spread them would be - // refused here rather than driven half-mapped. - let bar = common_cap.bar; - if [notify_cap, isr_cap, device_cap].iter().any(|cap| cap.bar != bar) { - return Err(Refusal::MissingCap("its four config windows in one BAR")); - } + let layout = Layout::of(&vendor_caps(&dev))?; + // The kernel hands out a BAR at a time, and reports 0 bytes for one it + // keeps back. + let bar = layout.bar(); let bar_bytes = *info .bar_bytes .get(bar as usize) .filter(|bytes| **bytes > 0) - .ok_or(Refusal::MissingCap("a BAR this claim may map"))?; + .ok_or(toyos_virtio::pci::Refusal::MissingCap("a BAR this claim may map"))?; let mapped = dev .map_bar(bar as u32, bar_bytes) .map_err(KernelRefused::on("the register window")) @@ -390,25 +204,10 @@ impl VirtioNet { // `mapped`, which this struct holds for its own life. let window = unsafe { Window::new(mapped.as_ptr(), bar_bytes as usize) }; - let common = common_cap.window(window)?; - let notify = notify_cap.window(window)?; - let device_cfg = device_cap.window(window)?; - - common.write::(COMMON_DEVICE_STATUS, 0); - // The reset is acknowledged by the register reading back zero (§4.1.4.3.2). - if common.read::(COMMON_DEVICE_STATUS) != 0 { - return Err(Refusal::ResetUnanswered); - } - common.write::(COMMON_DEVICE_STATUS, STATUS_ACKNOWLEDGE); - common.write::(COMMON_DEVICE_STATUS, STATUS_ACKNOWLEDGE | STATUS_DRIVER); - - common.write::(COMMON_DEVICE_FEATURE_SELECT, 0); - let lo: u32 = common.read(COMMON_DEVICE_FEATURE); - common.write::(COMMON_DEVICE_FEATURE_SELECT, 1); - let hi: u32 = common.read(COMMON_DEVICE_FEATURE); - let offered = ((hi as u64) << 32) | lo as u64; - let features = - offered & (VIRTIO_F_VERSION_1 | VIRTIO_F_ACCESS_PLATFORM | VIRTIO_NET_F_MAC); + let offer = Offer::acknowledge(Bar::over(window), &layout)?; + let offered = offer.features(); + let setup = offer.accept(VIRTIO_NET_F_MAC)?; + let features = setup.features(); // The line the kernel's virtio drivers print, in the same shape: // `iommu_virtio_platform` reads it back for every virtio function the // machine creates. @@ -421,91 +220,39 @@ impl VirtioNet { if features & VIRTIO_F_ACCESS_PLATFORM != 0 { 'y' } else { 'n' }, ); - common.write::(COMMON_DRIVER_FEATURE_SELECT, 0); - common.write::(COMMON_DRIVER_FEATURE, features as u32); - common.write::(COMMON_DRIVER_FEATURE_SELECT, 1); - common.write::(COMMON_DRIVER_FEATURE, (features >> 32) as u32); - common.write::( - COMMON_DEVICE_STATUS, - STATUS_ACKNOWLEDGE | STATUS_DRIVER | STATUS_FEATURES_OK, - ); - let answered: u32 = common.read(COMMON_DEVICE_STATUS); - if answered & STATUS_FEATURES_OK == 0 { - return Err(Refusal::FeaturesRefused { offered: features, status: answered }); - } - - let grant = dev + let region = dev .dma_alloc(GRANT_BYTES) .map_err(KernelRefused::on("a DMA grant")) .map_err(Refusal::Kernel)?; // SAFETY: the grant covers at least `GRANT_BYTES` — the kernel rounds // the request up to whole pages, never down — and lives as long as - // `grant`, which this struct holds. - let dma = unsafe { Window::new(grant.memory.as_ptr(), GRANT_BYTES as usize) }; - dma.zero(); - let dma_device_addr = grant.device_addr; - - let mut rx = Rings::new( - dma.sub(OFF_RX_DESC, RX_QUEUE_SIZE as usize * DESC_BYTES), - dma.sub(OFF_RX_AVAIL, AVAIL_RING_OFF + RX_QUEUE_SIZE as usize * 2), - dma.sub(OFF_RX_USED, USED_RING_OFF + RX_QUEUE_SIZE as usize * USED_ELEM_BYTES), - RX_QUEUE_SIZE, - ); - let mut tx = tx_rings(dma); - - setup_queue(common, RX_QUEUE, &mut rx, dma_device_addr, OFF_RX_DESC, OFF_RX_AVAIL, OFF_RX_USED)?; - setup_queue( - common, - TX_QUEUE, - &mut tx, - dma_device_addr, - OFF_TX_RINGS, - OFF_TX_RINGS + tx_avail_off(), - OFF_TX_RINGS + tx_used_off(), - )?; - - // The vector the kernel already put in the table, named to the device. - // After the queues are configured and before they are enabled, so no - // queue is ever live with no vector bound to it. - common.write::(COMMON_MSIX_CONFIG, MSIX_ENTRY); - if common.read::(COMMON_MSIX_CONFIG) == NO_VECTOR { - return Err(Refusal::NoVector("its configuration-change interrupt")); - } - for (queue, what) in [(RX_QUEUE, "the receive queue"), (TX_QUEUE, "the transmit queue")] { - common.write::(COMMON_QUEUE_SELECT, queue); - common.write::(COMMON_QUEUE_MSIX, MSIX_ENTRY); - if common.read::(COMMON_QUEUE_MSIX) == NO_VECTOR { - return Err(Refusal::NoVector(what)); - } - } - for queue in [RX_QUEUE, TX_QUEUE] { - common.write::(COMMON_QUEUE_SELECT, queue); - common.write::(COMMON_QUEUE_ENABLE, 1); - } + // `region`, which this struct holds. + let window = unsafe { Window::new(region.memory.as_ptr(), GRANT_BYTES as usize) }; + window.zero(); + let grant = Grant::over(window, region.device_addr); + + let rx = Virtqueue::new(grant, RX_QUEUE, RX_QUEUE_SIZE, RX_PARTS); + let tx = TxQueue::new(grant); + // The vector the kernel already put in the table, named to the device + // for each of its sources. + let mut setup = setup + .config_vector(MSIX_ENTRY)? + .enable(&rx, MSIX_ENTRY)? + .enable(&tx.rings, MSIX_ENTRY)?; let mut mac = [0u8; 6]; - for (i, byte) in mac.iter_mut().enumerate() { - *byte = device_cfg.read::(i); + for (at, byte) in mac.iter_mut().enumerate() { + (setup, *byte) = setup.device_read8(at)?; } - let rx_doorbell = doorbell(notify, notify_cap.notify_multiplier, rx.notify_off, RX_QUEUE); - let tx_doorbell = doorbell(notify, notify_cap.notify_multiplier, tx.notify_off, TX_QUEUE); - - common.write::( - COMMON_DEVICE_STATUS, - STATUS_ACKNOWLEDGE | STATUS_DRIVER | STATUS_FEATURES_OK | STATUS_DRIVER_OK, - ); - let nic = Self { dev, - rx_doorbell, + device: setup.driver_ok(), _bar: mapped, - _grant: grant, - dma, - dma_device_addr, + _region: region, + grant, rx: RefCell::new(rx), - tx: RefCell::new(TxQueue::new(tx, dma, dma_device_addr, tx_doorbell)), - reported: Latch::default(), + tx: RefCell::new(tx), mac, }; @@ -520,19 +267,6 @@ impl VirtioNet { self.mac } - /// Say what this driver has refused, when the count has moved. Once a - /// pass, never per element: a burst of refusals is one line. - pub fn report(&self) { - let refused = self.rx.borrow().refused + self.tx.borrow().rings.refused; - if self.reported.moved(refused).is_none() { - return; - } - crate::say!( - "netstack: this NIC has refused {refused} used-ring element(s) — the device named a \ - descriptor this driver never published or claimed more bytes than it was given" - ); - } - /// Drain the interrupt record, so the claim stops reading ready. /// /// The count is not acted on — what a message meant is in the rings — but @@ -550,30 +284,23 @@ impl VirtioNet { /// Post receive buffer `index` on its own head. fn post_rx(&self, index: usize) { - let head = index as u16; let at = OFF_RX_BUFS + index * RX_BUF_SIZE; // The header is zeroed before the buffer is published: what the device // writes there is its own, and what it leaves behind is this driver's. - self.dma.sub(at, NET_HDR_SIZE).zero(); - self.rx.borrow_mut().submit( - head, - self.dma_device_addr + at as u64, - RX_BUF_SIZE as u32, - true, - self.rx_doorbell, - ); + self.grant.window().sub(at, NET_HDR_SIZE).zero(); + let buffer = Buffer::writable(self.grant.device_addr(at), RX_BUF_SIZE as u32); + let published = self.rx.borrow_mut().publish(index as u16, &[buffer]); + self.device.notify(published); } /// The next received frame as `(buffer index, frame bytes)`, the virtio /// header excluded, or `None`. pub fn poll_rx(&self) -> Option<(usize, usize)> { loop { - let polled = self.rx.borrow_mut().poll_used(); - let (head, written) = polled?; - // `parse_used` bounded the head by the descriptor table's length, - // which is this queue's size, which is the buffer count. - let index = head as usize; - let total = written as usize; + // The head is below this queue's size, which is the buffer count, + // and `written` no more than the one buffer its chain is. + let Used { head, written } = finished(&mut self.rx.borrow_mut())?; + let (index, total) = (head as usize, written as usize); if total <= NET_HDR_SIZE { // Shorter than its own header is nothing to hand up, and the // buffer goes straight back. @@ -595,7 +322,7 @@ impl VirtioNet { /// Where a received frame's bytes are, past the virtio header. pub fn rx_frame(&self, index: usize, len: usize) -> &[u8] { - let window = self.dma.sub(OFF_RX_BUFS + index * RX_BUF_SIZE + NET_HDR_SIZE, len); + let window = self.grant.window().sub(OFF_RX_BUFS + index * RX_BUF_SIZE + NET_HDR_SIZE, len); // SAFETY: the window is inside the grant, which lives as long as // `self`; the device has finished with this buffer — its used-ring // element is what said so — and it is not posted again until @@ -612,7 +339,9 @@ impl VirtioNet { /// Fill a transmit buffer with a `len`-byte frame and hand it to the /// device ([`TxQueue::send`]). pub fn tx(&self, len: usize, fill: impl FnOnce(&mut [u8]) -> R) -> R { - self.tx.borrow_mut().send(len, fill) + let (result, published) = self.tx.borrow_mut().send(len, fill); + self.device.notify(published); + result } /// The claim, for the poller: readable means an interrupt has landed. @@ -621,49 +350,42 @@ impl VirtioNet { } } -/// The transmit queue's three rings, where the grant's layout puts them. -fn tx_rings(dma: Window) -> Rings { - Rings::new( - dma.sub(OFF_TX_RINGS, TX_QUEUE_SIZE as usize * DESC_BYTES), - dma.sub(OFF_TX_RINGS + tx_avail_off(), AVAIL_RING_OFF + TX_QUEUE_SIZE as usize * 2), - dma.sub(OFF_TX_RINGS + tx_used_off(), USED_RING_OFF + TX_QUEUE_SIZE as usize * USED_ELEM_BYTES), - TX_QUEUE_SIZE, - ) -} - /// The transmit queue: its rings, the heads nothing is in flight on, and the /// buffer each head owns. Everything in it is memory, so the host drives it /// with a plain allocation for the grant. struct TxQueue { - rings: Rings, + rings: Virtqueue, /// Transmit heads nothing is in flight on. free: Vec, - /// The grant, and where the device reaches its first byte. - dma: Window, - dma_device_addr: u64, - doorbell: Doorbell, + grant: Grant, } impl TxQueue { - fn new(rings: Rings, dma: Window, dma_device_addr: u64, doorbell: Doorbell) -> Self { - Self { rings, free: (0..TX_QUEUE_SIZE).rev().collect(), dma, dma_device_addr, doorbell } + fn new(grant: Grant) -> Self { + Self { + rings: Virtqueue::new(grant, TX_QUEUE, TX_QUEUE_SIZE, TX_PARTS), + free: (0..TX_QUEUE_SIZE).rev().collect(), + grant, + } } /// How many frames the queue takes now, every head the device has /// finished with taken back first. /// - /// **Room returning is an interrupt already**: this driver negotiates no + /// **Room returning is an interrupt already**: `toyos-virtio` accepts no /// `VIRTIO_F_EVENT_IDX` and leaves the transmit queue's `avail.flags` 0, /// and §2.7.7 has the device notify for every buffer it uses on such a /// queue. fn room(&mut self) -> usize { - while let Some((head, _)) = self.rings.poll_used() { - self.free.push(head); + while let Some(done) = finished(&mut self.rings) { + self.free.push(done.head); } self.free.len() } - /// Fill a transmit buffer with a `len`-byte frame and hand it to the device. + /// Fill a transmit buffer with a `len`-byte frame and make it available: + /// the device is told of it by whoever holds the transport, with what this + /// answers. /// /// **The buffer is the head's own and the two are taken together**, so /// nothing is written into a buffer the device is reading: a head leaves @@ -672,7 +394,7 @@ impl TxQueue { /// /// Non-blocking, and for a caller [`Self::room`] answered: a frame /// offered with no head free is a caller that did not ask. - fn send(&mut self, len: usize, fill: impl FnOnce(&mut [u8]) -> R) -> R { + fn send(&mut self, len: usize, fill: impl FnOnce(&mut [u8]) -> R) -> (R, Published) { assert!( NET_HDR_SIZE + len <= TX_BUF_SIZE, "netstack: a {len}-byte frame does not fit this NIC's transmit buffer" @@ -683,170 +405,88 @@ impl TxQueue { .expect("netstack: a frame was offered to a transmit queue that had said it has no room"); let at = OFF_TX_BUFS + head as usize * TX_BUF_SIZE; // The header is this driver's and zeroed before the frame goes in. - self.dma.sub(at, NET_HDR_SIZE).zero(); - let window = self.dma.sub(at + NET_HDR_SIZE, len); + self.grant.window().sub(at, NET_HDR_SIZE).zero(); + let window = self.grant.window().sub(at + NET_HDR_SIZE, len); // SAFETY: the window is inside the grant, which lives as long as the // driver that holds this queue; the device is not reading it, because // this head is out of `free` and its descriptor is published only // after `fill` returns. let result = fill(unsafe { std::slice::from_raw_parts_mut(window.as_ptr(), len) }); - self.rings.submit( - head, - self.dma_device_addr + at as u64, - (NET_HDR_SIZE + len) as u32, - false, - self.doorbell, - ); - result + let frame = Buffer::readable(self.grant.device_addr(at), (NET_HDR_SIZE + len) as u32); + (result, self.rings.publish(head, &[frame])) } } -/// Where a queue's doorbell is, from `queue_notify_off` and the multiplier the -/// notify capability published (§4.1.4.4). -fn doorbell(notify: Window, multiplier: u32, notify_off: u16, queue: u16) -> Doorbell { - let at = notify_off as usize * multiplier as usize; - Doorbell { window: notify.sub(at, 2), queue } -} - -/// Program one queue's size and ring addresses, and read back where its -/// doorbell sits. Does not enable it: a queue must have its vector first. -fn setup_queue( - common: Window, - index: u16, - rings: &mut Rings, - dma_device_addr: u64, - desc_off: usize, - avail_off: usize, - used_off: usize, -) -> Result<(), Refusal> { - common.write::(COMMON_QUEUE_SELECT, index); - let max: u16 = common.read(COMMON_QUEUE_SIZE); - if max < rings.size { - // The device's own bound against rings sized at compile time: a device - // offering fewer is one this layout does not fit, and shrinking to it - // would silently change how many buffers are posted. - return Err(Refusal::MissingCap("a queue as deep as this driver's rings")); - } - common.write::(COMMON_QUEUE_SIZE, rings.size); - common.write::(COMMON_QUEUE_DESC, dma_device_addr + desc_off as u64); - common.write::(COMMON_QUEUE_DRIVER, dma_device_addr + avail_off as u64); - common.write::(COMMON_QUEUE_DEVICE, dma_device_addr + used_off as u64); - rings.notify_off = common.read(COMMON_QUEUE_NOTIFY_OFF); - Ok(()) -} - -/// One virtio PCI capability, as read out of config space. -#[derive(Clone, Copy)] -struct Cap { - cfg_type: u8, - bar: u8, - offset: u32, - length: u32, - notify_multiplier: u32, -} - -impl Cap { - /// The sub-window this capability names inside its BAR's mapping. - /// - /// Offset and length are the *device's*, so both are checked against the - /// window before a sub-window is taken: a capability naming bytes past the - /// BAR refuses the device rather than being clamped to one that fits. - fn window(&self, bar: Window) -> Result { - let length = (self.length as usize).max(4); - match (self.offset as usize).checked_add(length) { - Some(end) if end <= bar.bytes() => Ok(bar.sub(self.offset as usize, length)), - _ => Err(Refusal::MissingCap("a capability inside its own BAR")), +/// The vendor capabilities a function published, in its list's order, walked +/// once. +fn vendor_caps(dev: &PciDev) -> Vec { + let mut found = Vec::new(); + let mut seen = 0usize; + let Ok(first) = dev.config_read(CAPABILITIES_PTR, RegWidth::U8) else { + return found; + }; + let mut next = first; + // The pointer is the device's: a chain that does not terminate, or one + // pointing outside the header, ends the walk rather than running off + // the window or for ever. + while next >= 0x40 && next < 0x100 && seen < MAX_CAPABILITIES { + seen += 1; + let Ok(id) = dev.config_read(next, RegWidth::U8) else { break }; + if id as u8 == CAP_ID_VENDOR { + let read = |at: u32, width| dev.config_read(next + at, width).unwrap_or(0); + found.push(VendorCap { + cfg_type: read(3, RegWidth::U8) as u8, + bar: read(4, RegWidth::U8) as u8, + offset: read(8, RegWidth::U32), + length: read(12, RegWidth::U32), + notify_off_multiplier: read(16, RegWidth::U32), + }); } + let Ok(link) = dev.config_read(next + 1, RegWidth::U8) else { break }; + next = link; } + found } -/// The vendor capabilities a function published, walked once. -struct Capabilities(Vec); - -impl Capabilities { - fn walk(dev: &PciDev) -> Self { - let mut found = Vec::new(); - let mut seen = 0usize; - let Ok(first) = dev.config_read(CAPABILITIES_PTR, RegWidth::U8) else { - return Self(found); - }; - let mut next = first; - // The pointer is the device's: a chain that does not terminate, or one - // pointing outside the header, ends the walk rather than running off - // the window or for ever. - while next >= 0x40 && next < 0x100 && seen < MAX_CAPABILITIES { - seen += 1; - let Ok(id) = dev.config_read(next, RegWidth::U8) else { break }; - if id as u8 == CAP_ID_VENDOR { - let read = |at: u32, width| dev.config_read(next + at, width).unwrap_or(0); - found.push(Cap { - cfg_type: read(3, RegWidth::U8) as u8, - bar: read(4, RegWidth::U8) as u8, - offset: read(8, RegWidth::U32), - length: read(12, RegWidth::U32), - notify_multiplier: read(16, RegWidth::U32), - }); - } - let Ok(link) = dev.config_read(next + 1, RegWidth::U8) else { break }; - next = link; - } - Self(found) - } - - /// The first capability of `cfg_type` — the one §4.1.4.1 says to use where - /// a device publishes several. - fn find(&self, cfg_type: u8) -> Option { - self.0.iter().find(|cap| cap.cfg_type == cfg_type).copied() - } -} - -/// What a used-ring element must satisfy, driven with elements no device would -/// send. -/// -/// **The device is on the far side of a trust boundary and moving the driver -/// out of the kernel did not change that.** Each of these was a way to make the -/// old kernel-side driver act on a number the device chose; the arms are the -/// same and the code they guard is this file's now. +/// The transmit queue's room, over a plain allocation for the grant. What a +/// used-ring element must satisfy is `toyos-virtio`'s, and tested there. #[cfg(test)] mod tests { + use toyos_virtio::queue::{AVAIL_ENTRY_BYTES, DESC_BYTES, RING_ENTRIES, USED_ELEM_BYTES}; + use super::*; - /// Rings over a plain allocation. `parse_used` reads none of them — only - /// `chain_bytes` and `size` — so what they point at does not matter. - fn rings(size: u16) -> Rings { - let backing = vec![0u8; 0x1000].leak(); - // SAFETY: `leak` gives the allocation the `'static` lifetime the window - // needs, and nothing here reads or writes through it. - let window = unsafe { Window::new(backing.as_mut_ptr(), backing.len()) }; - Rings::new(window.sub(0, 0x400), window.sub(0x400, 0x400), window.sub(0x800, 0x400), size) - } + const DEVICE_BASE: u64 = 0x4000_0000; - /// A transmit queue over a plain allocation the size of the grant, and - /// the doorbell a write to which nothing hears. + /// A transmit queue over a plain allocation the size of the grant. fn queue() -> TxQueue { - let grant = vec![0u8; GRANT_BYTES as usize].leak(); - let bell = vec![0u8; 2].leak(); - // SAFETY: `leak` gives both allocations the `'static` lifetime the - // windows need, and nothing but this queue and the test reaches them. - let (dma, bell) = unsafe { - (Window::new(grant.as_mut_ptr(), grant.len()), Window::new(bell.as_mut_ptr(), bell.len())) - }; - TxQueue::new(tx_rings(dma), dma, 0x4000_0000, Doorbell { window: bell, queue: TX_QUEUE }) + let backing = vec![0u64; GRANT_BYTES as usize / 8].leak(); + // SAFETY: `leak` gives the allocation the `'static` lifetime the + // window needs, and nothing but this queue and the test reaches it. + let window = unsafe { Window::new(backing.as_mut_ptr().cast(), GRANT_BYTES as usize) }; + TxQueue::new(Grant::over(window, DEVICE_BASE)) + } + + /// The head in available-ring entry `nth` (§2.7.6). + fn made_available(queue: &TxQueue, nth: u16) -> u16 { + let entry = (nth % TX_QUEUE_SIZE) as usize; + queue.grant.window().read(TX_PARTS.avail + RING_ENTRIES + entry * AVAIL_ENTRY_BYTES) } /// The device, finishing with the `count` oldest chains it was given and /// answering their heads (§2.7.8). fn device_uses(queue: &TxQueue, count: u16) -> Vec { - let rings = &queue.rings; - let used_idx: u16 = rings.used.read(USED_IDX_OFF); + let ring = queue.grant.window(); + let used_idx: u16 = ring.read(TX_PARTS.used + 2); (0..count) .map(|nth| { let at = used_idx.wrapping_add(nth); - let head: u16 = rings.avail.read(AVAIL_RING_OFF + (at % rings.size) as usize * 2); - let element = USED_RING_OFF + (at % rings.size) as usize * USED_ELEM_BYTES; - rings.used.write(element, head as u32); - rings.used.write(element + 4, 0u32); - rings.used.write(USED_IDX_OFF, at.wrapping_add(1)); + let head = made_available(queue, at); + let element = + TX_PARTS.used + RING_ENTRIES + (at % TX_QUEUE_SIZE) as usize * USED_ELEM_BYTES; + ring.write(element, head as u32); + ring.write(element + 4, 0u32); + ring.write(TX_PARTS.used + 2, at.wrapping_add(1)); head }) .collect() @@ -862,18 +502,19 @@ mod tests { let mut published = 0u8; while queue.room() > 0 { assert!(published < TX_QUEUE_SIZE as u8 * 2, "the transmit queue never filled"); - queue.send(60, |frame| frame.fill(published + 1)); + let ((), told) = queue.send(60, |frame| frame.fill(published + 1)); + assert_eq!(told.queue(), TX_QUEUE); published += 1; } assert_eq!(published as u16, TX_QUEUE_SIZE); - let avail_idx: u16 = queue.rings.avail.read(AVAIL_IDX_OFF); - assert_eq!(avail_idx, TX_QUEUE_SIZE); + let ring = queue.grant.window(); + assert_eq!(ring.read::(TX_PARTS.avail + 2), TX_QUEUE_SIZE); for nth in 0..TX_QUEUE_SIZE { - let head: u16 = queue.rings.avail.read(AVAIL_RING_OFF + nth as usize * 2); - let chain: Desc = queue.rings.desc.read(head as usize * DESC_BYTES); - assert_eq!(chain.len as usize, NET_HDR_SIZE + 60); - let buffer = queue.dma.sub((chain.addr - queue.dma_device_addr) as usize, chain.len as usize); - let bytes: Vec = (0..chain.len as usize).map(|at| buffer.read::(at)).collect(); + let chain = TX_PARTS.desc + made_available(&queue, nth) as usize * DESC_BYTES; + let (addr, len): (u64, u32) = (ring.read(chain), ring.read(chain + 8)); + assert_eq!(len as usize, NET_HDR_SIZE + 60); + let buffer = ring.sub((addr - DEVICE_BASE) as usize, len as usize); + let bytes: Vec = (0..len as usize).map(|at| buffer.read::(at)).collect(); assert_eq!(bytes[..NET_HDR_SIZE], [0; NET_HDR_SIZE]); assert_eq!(bytes[NET_HDR_SIZE..], [nth as u8 + 1; 60]); } @@ -882,9 +523,8 @@ mod tests { assert_eq!(queue.room(), 3); // The next frame goes out on a head the device gave back, and on no // head still in flight. - queue.send(60, |frame| frame.fill(0xEE)); - let reused: u16 = queue.rings.avail.read(AVAIL_RING_OFF + (TX_QUEUE_SIZE % queue.rings.size) as usize * 2); - assert!(back.contains(&reused)); + let _ = queue.send(60, |frame| frame.fill(0xEE)); + assert!(back.contains(&made_available(&queue, TX_QUEUE_SIZE))); assert_eq!(queue.room(), 2); } @@ -895,42 +535,55 @@ mod tests { fn a_frame_offered_to_a_full_transmit_queue_is_not_taken() { let mut queue = queue(); while queue.room() > 0 { - queue.send(60, |frame| frame.fill(1)); + let _ = queue.send(60, |frame| frame.fill(1)); } - queue.send(60, |frame| frame.fill(2)); + let _ = queue.send(60, |frame| frame.fill(2)); } + /// A queue with two frames in flight, on heads 0 and 1, whose device then + /// writes one used element, `id` and `len`, and counts `used_idx` of them. + fn after_the_device_wrote(id: u32, len: u32, used_idx: u16) -> TxQueue { + let mut queue = queue(); + let _ = queue.send(60, |frame| frame.fill(1)); + let _ = queue.send(60, |frame| frame.fill(2)); + assert_eq!([made_available(&queue, 0), made_available(&queue, 1)], [0, 1]); + let ring = queue.grant.window(); + ring.write(TX_PARTS.used + RING_ENTRIES, id); + ring.write(TX_PARTS.used + RING_ENTRIES + 4, len); + ring.write(TX_PARTS.used + 2, used_idx); + queue + } + + /// Each way the used ring is not believed ends the program by that + /// refusal's own name, where it is read: none is counted and passed over. #[test] - fn a_head_past_the_table_is_refused() { - let queue = rings(16); - assert_eq!(queue.parse_used(16, 4), Err(UsedRefusal::Head(16))); - assert_eq!(queue.parse_used(u32::MAX, 4), Err(UsedRefusal::Head(u32::MAX))); + #[should_panic(expected = "this NIC cannot be driven on — a used element's head")] + fn a_used_head_past_the_table_ends_the_driver() { + after_the_device_wrote(0xFFFF, 0, 1).room(); } - /// A completion for a head with nothing published at it — a second report - /// of one buffer, which is how a device would hand the same frame up twice. #[test] - fn a_head_with_no_chain_is_refused() { - let queue = rings(16); - assert_eq!(queue.parse_used(3, 4), Err(UsedRefusal::NoChain(3))); + #[should_panic( + expected = "this NIC cannot be driven on — a used element names head 5, where no chain \ + is in flight" + )] + fn a_used_head_with_no_chain_in_flight_ends_the_driver() { + after_the_device_wrote(5, 0, 1).room(); } - /// The one that matters most: `written` becomes the length of a slice this - /// process hands to smoltcp, so a device claiming more than it was given - /// would be a read past the buffer. + /// A transmit chain is the device's to read and has no byte it may write. #[test] - fn more_bytes_than_the_chain_was_given_is_refused() { - let mut queue = rings(16); - queue.chain_bytes[3] = 100; - assert_eq!(queue.parse_used(3, 100), Ok((3, 100))); - assert_eq!(queue.parse_used(3, 99), Ok((3, 99))); - assert_eq!( - queue.parse_used(3, 101), - Err(UsedRefusal::Written { id: 3, len: 101, chain: 100 }) - ); - assert_eq!( - queue.parse_used(3, u32::MAX), - Err(UsedRefusal::Written { id: 3, len: u32::MAX, chain: 100 }) - ); + #[should_panic( + expected = "this NIC cannot be driven on — the used element for head 0 claims more bytes \ + than its chain may be written" + )] + fn a_used_length_past_the_chains_writable_bytes_ends_the_driver() { + after_the_device_wrote(0, 1, 1).room(); + } + + #[test] + #[should_panic(expected = "this NIC cannot be driven on — the used index reads 3")] + fn a_used_index_past_every_frame_offered_ends_the_driver() { + after_the_device_wrote(0, 0, 3).room(); } }