diff --git a/build.zig b/build.zig index 412651e..a3dc24d 100644 --- a/build.zig +++ b/build.zig @@ -342,6 +342,7 @@ pub fn build(b: *std.Build) void { "protocol-registry-test", // drives the registrar: ungranted bind, collision, restart "protocol-denied-test", // restriction stage one: an ungranted open answers as absence "protocol-conformance-test", // the reserved verbs, asked of every provider the boot bound + "device-authority-test", // the attacker: a process handed no device, asserting what it cannot do }) |fixture| { const package = b.lazyDependency(fixture, .{}) orelse @panic("a test fixture package is missing under test/system/services"); @@ -397,6 +398,19 @@ pub fn build(b: *std.Build) void { // here (compiled for the host rather than inheriting a freestanding target) — // which also compile-checks that the three-way split stays self-consistent. const test_step = b.step("test", "Run tests"); + + // Every compile-time ceiling states what it counts, who decides its size, what it + // protects, what happens when it is reached, and how anyone finds out + // (docs/os-development/bounds.md). The ~273 that predate the rule are listed in + // tools/bounds-allowlist.txt so this could land without a tree-wide sweep first; + // that list may only shrink. Part of `test` rather than the default build: it reads + // the whole tree, and a red bounds check should not stop you booting a kernel. + const bounds_check = b.addSystemCommand(&.{ "python3", "tools/check-bounds.py" }); + bounds_check.setCwd(b.path(".")); + const bounds_step = b.step("bounds", "Check that every compile-time ceiling declares itself"); + bounds_step.dependOn(&bounds_check.step); + test_step.dependOn(&bounds_check.step); + for ([_][]const u8{ "system/boot-handoff.zig", "system/abi.zig", diff --git a/build.zig.zon b/build.zig.zon index bc568ab..6f7adc4 100644 --- a/build.zig.zon +++ b/build.zig.zon @@ -79,6 +79,7 @@ .@"protocol-registry-test" = .{ .path = "test/system/services/protocol-registry-test", .lazy = true }, .@"protocol-denied-test" = .{ .path = "test/system/services/protocol-denied-test", .lazy = true }, .@"protocol-conformance-test" = .{ .path = "test/system/services/protocol-conformance-test", .lazy = true }, + .@"device-authority-test" = .{ .path = "test/system/services/device-authority-test", .lazy = true }, // See `zig fetch --save ` for a command-line interface for adding dependencies. //.example = .{ // // When updating this field to a new URL, be sure to delete the corresponding diff --git a/docs/bounds-track-plan.md b/docs/bounds-track-plan.md index a8a5017..ce0f813 100644 --- a/docs/bounds-track-plan.md +++ b/docs/bounds-track-plan.md @@ -3,6 +3,426 @@ *Plan, 2026-08-08. Follows [fixed-bounds-audit.md](fixed-bounds-audit.md) (235 ceilings, 139 on quantities we do not choose) and the AMD Ryzen that found the first one.* +--- + +## Live state — the unattended run + +*This table is the progress view. It is updated at the end of every step, before the +next one starts.* + +| Step | What | State | +|---|---|---| +| L1 | Reclamation: a dead task's registrations die with its claims | **stopped — the step was wrong; see open question 4** | +| L2 | Bounds build check + allowlist; declare what we have already touched | **done** — `zig build bounds`, 273 allowlisted, 5 declared | +| L3 | xHCI: slot count from `HCSPARAMS1.MaxSlots`, not 8 | **done** — QEMU reports 64; the driver tracked 8 | +| L4 | USB: configuration descriptor sized by `wTotalLength`, not 512 | **done** — QEMU tops out at 211 bytes, so the case catches the class, not the original trigger | +| L5 | USB: interfaces from the descriptor, and the misattributed-endpoint bug | **done** — fix is by construction; no direct test, see open question 5 | +| L6 | xHCI: a failed `allocateDevice` stops leaking an enabled slot | **done** — path forced and verified; no regression test, see open question 5 | + +**Run 1 complete.** L1 stopped (the step was wrong), L2–L6 landed. Suite 115 → 116. +Allowlist 278 → 269. Two steps ship without a permanent regression test, both because +QEMU's USB devices are too small to reach the paths — see open question 5, which is the +audit's own lesson recurring: the test rig is smaller than a real machine. + +--- + +## Run 2 — device authority: delete the invented ceilings + +*Design: [device-authority.md](os-development/device-authority.md), which is the **how** +for the delegation step [device-manager.md](device-driver-development/device-manager.md) +already settled. Read both before starting; the second is authoritative where they +differ.* + +The goal, in the project owner's words: **remove the maximum values we set arbitrarily, +move the responsibility to the device manager, and keep in the kernel only the parts +that cannot safely run in user space.** + +| Step | What | State | +|---|---|---| +| D0 | The grant rides `system_spawn` — atomic, so no driver need change to receive one | **done** | +| D1 | `device_transfer(device_id, task_id)` — the holder gives a device away | **done** — syscall 54; a move, not a copy | +| D2 | Adversarial case: a process handed nothing is refused | **done** — `device-authority-test`; the claim half joins it at D6 | +| D3 | The manager claims the seeded devices before any driver is spawned | **merged into D4** | +| D4 | `usb-xhci-bus` receives its controller | **done** — caught an IOMMU regression I introduced | +| D5 | `pci-bus`, `virtio-gpu`, `ps2-bus`, discovery receive theirs | **done** — every driver now receives its hardware | +| D6 | `device_claim` refuses a device the caller was not handed | **blocked on question 10** | +| D7 | Zero-resource devices stop being kernel objects | **blocked on question 8** | +| D8 | **`maximum_children_per_parent` deleted** | **done** | +| D9 | Dynamic table; **`maximum_devices` deleted**; per-registrar allowance | **done** | +| D10 | Every driver hellos, on its own merits | not started — optional, independent | + +*Executed in the order D1, D2, D4, D5(part), D9, D0, D5(rest), D8 — the numbering is +the original plan's, not the sequence. D0 was added mid-run, D3 merged into D4, and D8 +turned out to have been unblocked since D9.* + +**Run 2 stops, blocked on one question.** Landed: D0, D1, D2, D4, D8, D9, and D5 for +three of five claimants (`usb-xhci-bus`, `pci-bus`, `virtio-gpu`). Suite 118/118. + +**Both invented ceilings are gone.** `maximum_devices` and +`maximum_children_per_parent` no longer exist, and delegation is atomic with the spawn. + +What remains is not a number: **D6 closes the claiming hole** and is blocked on +question 9, and D7 moves the inventory and is blocked on question 8. + +Ordering is load-bearing. D1–D2 built and proved the mechanism with nothing depending on +it. D4–D5 move each claimant across one at a time, so the suite stays green throughout +and a regression names the driver that caused it. D6 is the flag day. (The original +"D7 must precede D9" no longer holds: the per-holder quota bounds zero-resource children +as well as anything else, so D9 landed without it.) + +**D3 merged into D4** (found while implementing, recorded rather than worked around). +The two cannot be separated: the moment the manager claims a device, any driver still +calling `device_claim` on it is refused `AlreadyClaimed`, so D3 on its own turns the +suite red — and D3 applied to *nothing* changes no behaviour and cannot be tested. +They land together, with the manager claiming only for drivers in an explicit +**delegated set** so every unconverted driver keeps claiming exactly as before. +`usb-xhci-bus` is the first member, as it was the first driver to conform to `hello` +(device-manager.md, M18.1). D5 moves the rest in one at a time; the set and the +`device_claim` path both disappear at D6. + +### Observations from the run + +- **D4 introduced an IOMMU regression, caught by converting one driver at a time.** + `confineDevice` runs inside `systemDeviceClaim`, so a device arriving by *transfer* + was never confined for its new owner: the driver's DMA rings went unbound (three + IOMMU+USB cases failed), and two worse consequences were latent — a manager death + would have torn down a domain a live driver was using, and a driver death would have + leaked one. `iommu.reassign` moves the confinement with the device, keeping the + domain and attachment intact so it never translates through nothing. Converting all + five drivers at once would have produced the same three failures with five suspects. +- **`usb-hub` failed once in a full run, then passed six times** (four isolated, two + full). Suspected instance of the known intermittent AP ring-3 fault rather than + anything in D4 — recorded rather than dismissed, because D4 moved the `hello` earlier + and so did shift boot timing. Watch it across the remaining steps. + +--- + +## Run 3 — close the claiming hole + +*Every driver now receives its hardware. What remains is that `device_claim` still +works for anyone, so a device nobody holds can still be taken. Six steps, no open +questions.* + +### The rule this run implements + +The kernel records, per device, **who gave it**. From that one field everything follows: + +- `device_transfer` and the spawn-fused grant set the giver. +- **`device_claim` refuses a device that has a giver.** A device that was given to + someone is delegated hardware and must be handed on, never taken. +- **On death a device reverts to its giver**, if that task is still alive; otherwise its + claim clears. A grant is a loan, not a gift. + +Two things fall out rather than being special-cased: + +- **The framebuffer is untouched.** Nobody delegates it, so it has no giver, so the + display service can still claim it exactly as today. No exemption in the kernel, no + mention of display anywhere in the rule. +- **Restart needs no race.** A dying driver's device returns to the manager, which + re-delegates it on respawn. Previously the kernel released it to nobody and the + manager re-claimed first-come, so every restart reopened the hole. + +**Found while implementing: E2 is the step that closes the hole, not E3.** A delegated +device is *held*, so an attempt to take it is refused as `AlreadyClaimed` long before +the giver is consulted — and once a borrower's death returns the device to its lender +(or clears both when the lender is gone), there is no state where a device is unheld and +still on loan. E3's check is therefore unreachable today. It stays as one comparison +that fails closed, guarding any future path that frees a device without clearing its +giver, and its comment says so rather than implying a protection it is not providing. + +| Step | What | +|---|---| +| E1 | Record a giver per device; `device_transfer` and the spawn grant set it — **done** | +| E2 | On task death a device reverts to its giver if alive, else its claim clears — **done** | +| E3 | `device_claim` refuses a device that has a giver — **done**, but unreachable: E2 already closed the window | +| E4 | The manager claims every resource-bearing device at boot, so nothing is left takeable — **done** (boot snapshot only; see below) | +| E5 | The attacker fixture gains the claim half it has been waiting for since D2 — **done with E4** | +| E6 | Delete the delegated-set scaffolding — every driver is delegated now — **done** | + +Ordering: E1 alone changes no behaviour. E2 must precede E3, or a restart cannot +re-acquire. E4 must precede E5, or the attacker will find takeable devices and the +assertion will be wrong about why. E6 is cleanup. + +**Run 3 complete.** Suite 118/118. Every driver receives its hardware; a grant is a +loan that returns to its lender when the borrower dies; nothing firmware-discovered is +left unheld except the framebuffer, which the compositor owns. + +### E4's scope, stated + +It covers the **boot snapshot**. A device *reported* later and matched to no driver +stays claimable — `pci-cap-test` and `iommu-fault-test` both rely on that to reach an +unmatched NIC. Narrowing it further is a separate change with those fixtures in scope. + +The real gap it closed was the **HPET**: an MMIO window, an IRQ, no user-space driver, +and claimable by anyone. Excluded on purpose: the loader's framebuffer (the manager +starts before display, so taking it would break the boot screen) and anything with no +resources, which grants nothing. + +### Not in this run, and not blocking it + +**Zero-resource devices stay in the kernel.** Moving them out needs an answer to who +mints their ids, given that the id is the `device_token` on the usb-transfer wire and +the kernel's idempotency is what keeps it stable across a bus restart. Nothing depends +on it: both invented ceilings are already gone and this run does not need it. Recorded +as future work. + +**Not every driver hellos.** Delivery no longer needs it, so it stands or falls on its +own merits — uniform liveness and one class of driver. Also future work. + +--- + +### Question 9 — answered: a singleton gets every device that matched it + +The 8042 is one controller described by two ACPI nodes, so it cannot be split across +processes. The manager now hands the single `ps2-bus` instance **every** matching node: +the first rides the spawn, the rest are transferred to the running instance. Late +arrival is safe because the ordering is natural rather than lucky — the controller node +carries the ports and is needed at once, the mouse node not until after identify +(measured: handed over at 0.336, first touched at 0.456). The count is whatever matched, +so a machine with no PS/2 ports or one port needs no special case. + +Discovery turned out not to be special either. The kernel seeds the `acpi-tables` node, +so it is in the same boot snapshot the manager already scans for the PCI host bridge — +it is handed over at spawn like everything else. + +### Open question 10 — the framebuffer is the last thing anyone claims + +`device_claim` now has exactly two callers: the device manager, which is the acquirer +and should have it, and `system/services/display/backend.zig`, whose GOP path claims the +kernel-seeded display node. + +Closing `device_claim` breaks the compositor's boot floor. Exempting it puts a hole in +the middle of the authority model, in the one place an exemption is most expensive. + +The likely answer is neither. **The framebuffer is not a device** — it is where pixels +go, handed over by the loader, and the kernel wraps it in a `DeviceClass.display` +descriptor only so `mmio_map` can hand it over write-combining. If that is right, it +should leave the device table rather than be exempted from its rules, and the compositor +should receive the pixels some other way. That is a change to the display path, which is +out of scope for this run. + +### Settled 2026-08-08: the grant rides `system_spawn` (D0) + +Questions 6 and 7 both dissolved on inspection — neither `ps2-bus` nor discovery needs +to start speaking `hello`, and `virtio-gpu` has no standalone path to lose. What remains +is *where the grant is delivered*, and there are three candidates: + +| | Race window | Cost | +|---|---|---| +| Transfer after spawn | **yes** | none | +| Every driver hellos | no | `ps2-bus` + discovery gain a handshake | +| **Grant rides `system_spawn`** | **no — atomic** | one more syscall argument | + +**Take the third.** The manager cannot transfer before the child exists, so a separate +transfer always leaves a window in which the child is running and does not yet hold its +device. It would close on QEMU every time and open occasionally on a machine with +different timing — the exact failure shape this track exists to delete, and not worth +introducing while removing the others. Fusing the device into the spawn removes it by +construction: the child does not exist until it holds the device. No new knowledge in +the kernel — the same rule, *you may give away what you hold*, made atomic with the call +that creates the recipient. `system_spawn` uses five of six argument registers, so there +is room, and `no_device` is already the sentinel for a driver with no assignment. + +**`hello` for every driver is a good idea on its own merits** — uniform liveness, the +deadline applied to all rather than some, and the `speaks_protocol` two-class split +leaving the manager (a wedged `ps2-bus` is invisible to its supervisor today). It is +D10, kept separate so grant delivery does not force it. + +### Open questions raised by D5 — resolved except question 8 + +Delegation is delivered in `onHello`. That works for a driver that says hello, and +**two of the four do not**. + +6. **`ps2-bus` does not hello**, and that is deliberate: device-manager.md records + "Legacy drivers (e.g. ps2-bus) are supervised and restarted but **not yet required + to hello**", and `Driver.speaks_protocol` exists to say so. Either it leaves + "legacy", or the grant gets a delivery point that is not `hello`. + + **The `acpi` service is not this case — an earlier version of this question wrongly + lumped it in.** It is spawned as `addDriver("discovery", no_device, false)`: the + *discovery service*, one per firmware, with **no device assignment at all**, which + "finds and claims the acpi-tables (or devicetree-blob) node itself. Not a per-device + driver." The manager cannot hand it a device because discovery is what produces the + device tree — there is nothing to match against yet. Its question is not "should it + hello" but "is the bootstrap exempt from D6, or does the manager claim the + acpi-tables node and pass it on". + + **A delivery point that needs no hello already exists in the shape of the code**: + `process.spawnSupervised` returns the child's pid, so the manager could transfer + immediately after spawn. If that is acceptable, this question mostly dissolves — + `ps2-bus` would not need to leave "legacy" and discovery could be handed its node + too. The cost is ordering: the child may reach for the device before the transfer + lands, where `hello` guarantees it cannot because the child is the one asking. That + trade is the actual decision. + +7. **`virtio-gpu`'s "standalone bring-up" — resolved; there is no such path.** An + earlier version of this question read the comment "Best-effort: standalone bring-up + has no manager" as a boot path that delegation would delete. It is not one. + `virtio-gpu`'s `main` requires `argv[1]` and exits without it, and the only source of + that argument is the device manager — the chain is ACPI → pci-bus reports the + function → `devices.csv` matches `1AF4:1050` → the manager spawns the driver with the + id. There is no way to run it without a manager, so nothing is lost. + + What the comment is really about is **resilience**: the hello is best-effort so a + driver whose manager has *died* keeps serving, which device-manager.md states + ("If the manager dies, drivers keep running"). + + The residual is one narrow window: the manager spawns a driver and dies before + transferring. Today the driver could still claim, because claiming is free-for-all; + after D6 it exits and the restarted manager respawns it. That is the better + behaviour — a driver holding hardware nobody assigned it is what D6 exists to stop. + +Until these are answered, `device_claim` cannot be closed off at D6 for those three, so +**D6 is blocked on questions 6 and 7**. + +8. **Removing zero-resource devices from the kernel needs someone else to mint their + ids, and nothing says who.** A USB interface is registered with + `resource_count = 0`, and the id `device_register` returns is load-bearing in three + places: it is the `child_added` packet's target, it is the class driver's `argv[1]`, + and it is the `device_token` of the **usb-transfer wire protocol** — so the id space + is visible on the wire, not merely internal. usb-xhci-bus's own comment records the + fourth constraint: the kernel's idempotency is what makes "the same port and + interface always map back to the same device id" across a bus restart, which is what + stops a respawned bus spawning duplicate class drivers. + + So the mover has to answer: who mints the id, how it stays stable across a *bus* + restart, how it stays stable across a *manager* restart (open question 3 territory), + and whether the wire protocol's `device_token` changes meaning. That is a design + step, not a mechanical one. + +**D8 is reordered to run after D9**, which is a correction to the original sequencing. +D8's justification was "the authorisation it stood in for exists" — but with D6 blocked +it does not, so deleting the shared per-parent cap now would reopen the exhaustion hole +it was written for. D9's **per-holder quota** closes that hole independently of +authorisation, and does it better: a rogue exhausts its own allowance rather than the +table everyone shares. Once the quota exists, the per-parent cap is redundant whether or +not D6 has landed. + +D9 also no longer depends on D7. Its original rationale was that zero-resource children +are the case that sidesteps containment — true, but a per-holder quota bounds them just +as well as anything else, because it counts entries per holder rather than per parent. + +### Settled, so the run does not re-litigate them + +- **The manager claims, it is not granted.** No binary names in the kernel; the rule is + "you may give away what you hold". The residual — it rests on the manager claiming + first — is stated in the design and is closed later by the spawn capability + [drivers.md](device-driver-development/drivers.md) already names as missing. +- **The framebuffer is not a device.** It is where pixels go, handed over by the loader, + and the compositor uses it as the boot floor until a real display driver announces + itself. Nothing in this run touches the display service or its GOP path. +- **`maximum_endpoints_per_interface` and the wire structs stay.** Widening them is a + protocol change, out of scope. +- **A per-holder quota is not a retreat.** Dynamic storage with no bound moves the + ceiling to the kernel heap, which is shared and fatal rather than partial. A bound + charged to the task that caused it is isolation, and it is declared through + [bounds.md](os-development/bounds.md) like anything else. + +### Working rules + +As Run 1, unchanged: work in `/Users/danielsamson/Gitea/daniel/danos` on +`claude/bounds-track`; every step lands with a test that fails before the fix, verified +by restoring the old behaviour; full suite green before each commit; never two suites at +once (`pgrep -f qemu_test.py`); 60 GiB free; `git commit -F` with no `Co-Authored-By`; +update this table before starting the next step. **If a step needs a decision that is +not written down, stop it, add the question below, and move on** — Run 1's first step +was wrong and stopping was the right call. + +**Suite:** 115/115 at the start of the run. +**Branch:** `claude/bounds-track`. + +### What this run deliberately does not touch + +Phases 2 and 3 below — the authorisation gate and moving the inventory to the device +manager — are **out of scope for unattended work**. They decide whether the OS is +secure, and they are currently a direction rather than a specification: what a device +capability *is*, which syscalls change, what replaces `device_claim` for its seven +callers, how a driver spawned bare behaves. Those want a design session, the way +`/protocol` had one. + +Also out of scope: anything touching `maximum_device_resources` (a wire struct, so a +trust-boundary change, not a resize), and the non-device bounds the audit found in FAT, +the VFS, logger, init, display and boot. + +### Open questions this run must not answer on its own + +Recorded here rather than guessed. If a step runs into one, it stops and writes the +question down instead of inventing an answer. + +1. **`device_enumerate` probably narrows rather than retires.** The device manager + calls it to find `pci_host_bridge` nodes — it cannot ask itself. The likely shape is + that the kernel keeps the *firmware-discovered roots* (which by principle 5 it holds + for real reasons, since they come from ACPI rather than a driver's say-so) and + everything a driver registered lives in the manager. Not decided. +2. **A device-manager restart has no re-enumerate handshake.** If only the manager + dies, the buses are alive and never re-send `child_added`, so a restarted manager + comes back blind. The manager is restartable by design; nothing implements this. +3. **Which adversarial tests I1–I3 need.** The audit's six real defects were all found + by asking what an attacker would do, and the suite had never asked. "Add adversarial + cases" is not executable until the attacks are named. +4. **Reclamation is not a death-sweep problem, and L1 as written would have broken the + restart path.** Found on the first attempt at it. The audit is right that `count` + never decreases, but *death is the wrong trigger*: + - The broker keeps entries deliberately: "The devices stay in the table — they + describe hardware, which did not go away — only their ownership clears." A driver + dying does not unplug anything. + - Device ids must stay **stable across a bus restart**, because + `device-manager.driverForDevice` dedupes by `device_id` so that "a re-report after + a bus restart must not spawn a second instance". Stability comes from the + idempotency scan returning the existing id — removing entries on death would give + a restarted bus fresh ids and spawn duplicate driver instances. + - Everything else a task holds *is* already reclaimed on every path out: + `irq.releaseOwner`, `iommu.releaseAllOwnedBy`, `dmaRegistryReleaseOwner`, then the + broker's claims (`process.releaseTaskResourcesLocked`). + + So the real leak has two sources, and neither is death: a device that genuinely + **goes away** (hot-unplug) has no retirement path, and a bus that enumerates + *differently* on restart leaves its stale entries behind forever. Both are the device + manager's inventory problem — phase 3 — and both need the id-stability question + answered first (tombstone-and-reuse aliases stale ids held by another process; + generation-tagged ids change the id encoding, which is ABI). Not an unattended + decision. +5. **Driver descriptor parsing cannot be host-tested, so L5's correctness fix ships + without a direct test.** The endpoint-misattribution bug lives in + `parseConfiguration`, a pure function over a byte blob — exactly the shape a host + unit test wants, and `usb-storage/scsi.zig` and `usb-hid/hid-report.zig` already do + this. But `usb-xhci-library.zig` imports `memory`, `mmio` and `time`, so it cannot + be a standalone host-test root, and QEMU offers no device that would exercise the + path anyway: the largest available is `usb-audio,multi=on` at 2 interfaces and 211 + bytes, against a cap of 4. + + Three ways out, and picking one is a judgement about house style rather than a + mechanical step: extract the parser to its own file and wire `usb-abi` into a test + module (build-support currently resolves module names only for `userBinary`); + extract it and import `usb-abi` by relative path (against the import-by-name + convention); or accept QEMU-only coverage and say so. + + Mitigating, and the reason this is recorded rather than blocking: after the fix the + bug is unreachable **by construction**, not by the added `else`. Interfaces are now + allocated to exactly the count the descriptor declares, so `interface_count` can + never reach `interfaces.len` mid-parse. The `else` is belt-and-braces for the + 255-interface clamp. The alternate-setting path that shares it *is* exercised — + `usb-audio` has alternate settings, and the `usb-large-descriptor` case walks them. + +### Working rules for the run + +- Work in `/Users/danielsamson/Gitea/daniel/danos` (not a worktree), on + `claude/bounds-track`. +- **Every step lands with a test that fails before the fix**, verified by temporarily + restoring the old behaviour and watching exactly the intended assertion flip. A test + that passes both ways is not a test. +- Full QEMU suite green before each commit. Never run two suites at once — check + `pgrep -f qemu_test.py` first; a second concurrent run produces false triple faults + because both share `zig-out`. +- Check at least 60 GiB free before starting a suite. +- Commit with `git commit -F `, never `-m` (a backtick in a message is executed + by the shell and silently eats a word). No `Co-Authored-By` trailers. +- Update the Live state table **before** starting the next step. +- If a step needs a decision that is not written down here, stop, add it to the open + questions above, and move to the next step. + +--- + ## The principles this is derived from 1. **danOS is a microkernel.** Minimise what the kernel is responsible for; move diff --git a/docs/os-development/device-authority.md b/docs/os-development/device-authority.md index 7e49a8f..9509214 100644 --- a/docs/os-development/device-authority.md +++ b/docs/os-development/device-authority.md @@ -1,173 +1,151 @@ -# Device authority: the kernel stops keeping an inventory +# Device authority: the implementation of delegation -*Design, drafted 2026-08-07. Not implemented. Prompted by a real machine: an AMD -Ryzen desktop enumerated more PCI functions than the kernel's device table would -hold, and the xHCI and SATA controllers were refused registration — so the -machine booted to the compositor with no USB and no storage.* +*Implementation design, 2026-08-08. The **what** is settled in +[device-manager.md](../device-driver-development/device-manager.md) — "structure in the +manager, authority in the kernel", and delegation as the step after `hello`. This +document is the **how**, and the decisions that paragraph leaves open.* -The kernel keeps a table of every device userspace discovers. It is a fixed -array of 64 descriptors, 344 bytes each, and a second cap allows any one parent -16 children. Neither number is written down anywhere as a decision: -`maximum_devices` has no comment and never reached `parameters.zig`, where every -other tunable in this kernel lives with its reasoning attached. +Read first: [drivers.md](../device-driver-development/drivers.md) (the claim is the +capability), [driver-model.md](../device-driver-development/driver-model.md) (the three +invariants), [device-manager.md](../device-driver-development/device-manager.md) (the +tree, the matcher, the supervisor). -Raising them is not the fix. The numbers are wrong because the *table* is wrong: -it is an inventory of hardware, and an inventory of hardware is not something a -kernel needs. This document proposes replacing it with capabilities, which -removes the ceiling rather than moving it. +## The one thing not yet true -## What the kernel actually uses +`device-manager.md` says assignment "stays argv for now", and names the next step: -Every read of a device descriptor from the kernel proper, exhaustively: +> The step after `hello` exists is delegation: the manager claims (or is granted) the +> devices and passes the claim to the driver over IPC (the M13 capability-transfer +> mechanism), replacing first-come-first-served `device_claim` with policy. -| Used for | What it needs | -|---|---| -| `mmio_map`, `io_read`/`io_write` | the physical range, to check the mapping falls inside it | -| `irq_bind`, `msi_bind` | the GSI, and that the caller owns the device | -| `dma_bind` | that the caller owns the device | -| IOMMU confinement | the **PCI BDF**, to key a domain | -| `device_register` | the parent's ranges, for the containment check | -| the boot display seed | one framebuffer window | +Until that lands, the manager's matching is advisory. `device_claim` checks only that +the device exists and is unheld ([devices-broker.zig](../../system/kernel/devices-broker.zig)): -That is: **physical ranges, interrupt numbers, and one BDF.** Vendor, device and -subsystem ids, class triples, the human-readable names, the parent links, the -bus numbers — the kernel stores all of it and reads none of it. It is held so -that `device_enumerate` can hand it back to user space, which is the whole -mistake in one sentence: the kernel is acting as a distribution mechanism for -data it does not use. +```zig +pub fn claim(id: u64, owner: u32) ClaimError!void { + if (id >= count) return error.NoSuchDevice; + if (claimed[@intCast(id)] != null) return error.AlreadyClaimed; + claimed[@intCast(id)] = owner; +} +``` -## The split +A driver is spawned with its device id in `argv[1]` and claims it; any process could +pass any integer instead. Since a claim is a licence to map physical memory, that is the +gap this document closes. -Three concerns are tangled in one table. +## Decision 1: the manager claims, then transfers -**Platform bring-up** — timers, LAPIC/IOAPIC, CPU topology, the ECAM window, the -framebuffer the firmware left. The kernel derives these from ACPI before user -space exists and needs them to function. They never belonged in the device table -and mostly are not (the platform block in `kernel.zig` is separate); this -document does not change them. +`device-manager.md` leaves "claims (or is granted)" open. **Claims.** -**Resource authority** — which task may map which physical range, receive which -interrupt, touch which ports. This *must* stay in the kernel. It is the one -grant that cannot be audited after the fact: a process that maps arbitrary -physical memory owns the machine, page tables and IOMMU structures included. -This is memory protection, not device management, and it is why the answer is -not simply "move it all to the device manager". +The manager runs before any driver exists — `init` starts it from `init.csv`, and it is +what spawns drivers — so it takes the seeded devices unopposed and there is nothing +unheld left for anyone to race for. One new call moves ownership on: -**Device inventory** — what exists, what it is, how it is arranged, which driver -should bind it. This is `device-manager`'s job and is already half there: it -loads `devices.csv`, matches, spawns drivers, and receives `child_added` reports -over its own protocol. The kernel table duplicates what those reports already -carry. +``` +device_transfer(device_id, task_id) -> 0/-errno +``` -## The proposal: a resource is a capability +The kernel checks only that the caller currently holds the device. No names, no +attestation, no notion of "the device manager" — the rule is *you may give away what you +hold*, which is the capability discipline already in force. -Device resources join endpoints, shared memory and DMA regions as a kind in the -handle table. +The alternative was the kernel granting roots to a task it recognises by binary path +plus a PID-1 supervisor. It is more robust — it does not depend on the manager being +first — but it puts a binary name inside the kernel, and the goal is that the kernel +keeps only what cannot safely live in user space. A name is not that. -1. **Roots.** At boot the kernel mints capabilities for the windows it learned - from firmware — the ECAM range, the framebuffer, the legacy port space — and - hands them to the first bus drivers. This is the only place device knowledge - enters the kernel, and it comes from ACPI, not from a driver's say-so. -2. **Subdivision.** A bus driver enumerating hardware derives a narrower - capability from one it holds: `resource_derive(cap, kind, start, len) → cap`. - The kernel checks the sub-range lies inside the capability being subdivided — - the same containment rule as today (`devices-broker.contains`), but checked - against *one capability the caller demonstrably holds* rather than by walking - a global tree. -3. **Delegation.** The driver passes that capability to the child driver over - IPC. Cap-passing already exists; this is the mechanism `subscribe` and - `attach_scanout` already use. -4. **Use.** `mmio_map`, `irq_bind`, `msi_bind`, `io_read`/`io_write` and - `dma_bind` take a capability handle instead of `(device_id, resource_index)`. - Possession *is* the authority — there is nothing to look up and no ownership - table to consult. +**The residual, stated plainly:** authority here rests on the manager claiming first. +That holds because `init.csv` decides what starts and in what order, so it is an +operator-visible ordering rather than an attacker-controlled one — but it is an +assumption, not an enforced invariant. The enforced version arrives with the spawn +capability [drivers.md](../device-driver-development/drivers.md) already names as +missing ("`system_spawn` is currently ungated … because there is no spawn capability +yet"). This design is compatible with it and does not block on it. -Exclusivity stops being a broker refusing a second claimant and becomes the -ordinary property of a capability: only one process was given it. +## Decision 2: the kernel stops holding inventory -## What this buys +The kernel reads exactly three things out of a descriptor: **physical ranges** (to check +a mapping falls inside one), **interrupt numbers**, and **one PCI BDF** (to key an IOMMU +domain). Vendor, device and subsystem ids, class triples, `_HID` strings, bus addresses +and names are stored only so `device_enumerate` can hand them back — which +`device-manager.md` already resolves: that call "fades to a manager-internal (then +deleted) seam", because the manager owns the tree as data. -**No ceiling.** There is no table to size, so no machine is too big. The -Ryzen's enumeration stops being a limit to tune and becomes what it is — a fact -about a computer. +So the kernel's table becomes: **parent, resources, holder, BDF.** That is what cannot +safely run in user space; the rest moves. -**Reclamation, free.** `count` in the broker today only ever increases; -`releaseAllOwnedBy` clears a dead driver's *claims* but never its entries. A -driver that crashes and is restarted re-registers its children and consumes the -table again — reachable today without any malice, given the device manager -restarts drivers by design. Capabilities die with the task. +**Devices with no resources leave the kernel entirely.** A USB device is addressed +through its controller and carries `resource_count = 0` +([driver-model.md](../device-driver-development/driver-model.md): "that case is allowed +and is the common one"). It conveys no mapping authority, so there is nothing for the +kernel to enforce and no reason for it to know. It is inventory, and inventory is the +manager's — reported by `child_added`, which already carries everything needed. -**No quota needed.** The per-parent cap exists to stop one claimant looping -`device_register` and filling the shared table, because a zero-resource child -sidesteps the containment check. With no shared table there is nothing to -exhaust; a process can only ever subdivide what it was given, and its handles -are already bounded per task. +That is also the case that made `maximum_children_per_parent` necessary: a zero-resource +child sidesteps containment, so a driver could loop `device_register` and fill the +shared table. Once such children are not kernel objects, every remaining entry is a real +contained subdivision of something the caller holds. -**A smaller kernel.** Three syscalls leave the ABI, one narrower one arrives, -and `devices-broker.zig` largely disappears along with both constants. +## Decision 3: no shared ceiling; a per-holder quota instead -**The discipline the rest of the system already uses.** "The claim is the -capability" is written in the driver documentation as if it were already true. -This makes it true. +`maximum_devices = 64` and `maximum_children_per_parent = 16` are numbers we invented, +and both are shared — one driver's enumeration starves every other driver, which is how +an AMD Ryzen booted with no USB and no storage. -## Syscall surface +- **The table becomes dynamic.** It is built after `heap.init` (`kernel.zig`: `pmm.init` + at 137, `heap.init` at 179, `devices_broker.init` at 202), so nothing prevents it. No + specification bounds how many devices a machine has, so nothing should bound ours. +- **`maximum_children_per_parent` is deleted**, because the authorisation it stood in + for now exists. +- **A per-holder quota replaces them.** Dynamic storage without a bound moves the + ceiling to the kernel heap, which is shared and fatal rather than partial — strictly + worse. The bound that is *not* worse is one charged to the task that caused it: a + driver that loops `device_register` exhausts its own allowance, is refused with an + attributable errno, and is restarted by its supervisor while every other driver + carries on. That is the microkernel property rather than a workaround for it, and it + is declared through [bounds.md](bounds.md) like any other. -- `device_enumerate` — **retires.** It exists only to read the kernel's table. - Callers ask `device-manager`, whose protocol already reserves an `enumerate` - verb. Note this is a public-ABI change: `vdso.md` documents it. -- `device_register` — **splits.** The kernel half becomes `resource_derive`; the - publication half ("this device exists, here is what it is") becomes an IPC - message to `device-manager`, which is where the inventory belongs and where - `child_added` already carries the same facts. -- `device_claim` — **dissolves into possession**, except for the IOMMU (below). -- `mmio_map`, `irq_bind`, `msi_bind`, `io_read`, `io_write`, `dma_bind` — keep - their names and semantics; their first argument becomes a capability handle. +## The shape of the change -## The open question: where the IOMMU attaches +| | Before | After | +|---|---|---| +| Manager gets its devices | claims them, unauthorised | claims them (first, unopposed) | +| Driver gets its device | `argv[1]` + `device_claim` | receives it in the `hello` reply | +| Kernel checks | is it free? | do you hold it? | +| Kernel stores | the full descriptor | parent, resources, holder, BDF | +| Zero-resource devices | kernel table entries | manager records only | +| Table size | `maximum_devices = 64` | dynamic, per-holder quota | +| Children per parent | `maximum_children_per_parent = 16` | deleted | -This is the one place the kernel still needs device *identity* rather than a -range. `confineDevice(device_id, bdf, owner)` builds a domain keyed by PCI BDF, -attaches it, and the claim is rolled back if confinement fails — deliberately, -so a device that cannot be confined is never driven. +Bring-up order changes for the five claiming drivers: `hello` must precede the claim, +because the reply is where the device arrives. `pci-bus` today does the reverse — its +own comment reads "Claim the bridge, map the ECAM, hello the manager, then scan." -Three options, none obviously right: +## What does not change -1. **The BDF rides the capability.** A memory capability derived for a PCI - function carries its BDF, and the kernel confines on first `mmio_map` or - `dma_bind`. Keeps the syscall count down; means a capability is no longer - purely a range. -2. **An explicit `device_attach(cap, bdf)`.** Honest and visible, but it puts a - BDF — a fact about PCI — into a kernel interface that otherwise knows nothing - about buses, and something must stop a caller naming a BDF that is not - theirs. -3. **The root PCI capability carries the segment, and derivation computes the - BDF.** Purest, but only works for PCI and the kernel would be parsing bus - topology, which is precisely what this document is trying to stop. +- The three invariants of [driver-model.md](../device-driver-development/driver-model.md): + a claim is exclusive, a descriptor is a licence to map physical memory, therefore + containment. This design strengthens the first and touches neither of the others. +- The display service's GOP path. The framebuffer is not a device — it is where pixels + go, handed over by the loader, and the compositor uses it as the boot floor until a + real display driver announces itself + ([display-v2.md](../device-driver-development/display-v2.md)). The kernel wraps it in + a display-class descriptor so `mmio_map` can hand it over write-combining; that is + plumbing for a mapping, not a claim about what it is. +- Supervision, restart, pruning and re-report + ([device-manager.md](../device-driver-development/device-manager.md), + [process-lifecycle.md](process-lifecycle.md)). Delegation slots into the existing + `hello` exchange and changes none of it. +- `device_register` idempotency, which is what lets a restarted bus rebuild the same + ids. -My inclination is (1), because confinement is a property of the resource being -granted rather than a separate action, and because it keeps the "possession is -authority" story intact. It needs the derivation call to know it is carving a -PCI function, which is a wart worth arguing about. +## How it is verified -## What this does not solve +The invariant is: **a process holds what it was handed and cannot name its way into +holding more.** The suite has no adversarial device case today — the audit's lesson was +that "the suite contains no attacker" — so this adds one: a process that was handed +nothing calls `device_claim` and `device_transfer` on a device another driver holds, and +on one nobody holds, and is refused each time with its own errno. -- **Hot-plug and removal.** Capabilities die with their holder, but a device - that physically disappears while a driver lives is the device manager's - problem and unchanged by this. -- **Quotas on physical memory.** Nothing here bounds how much a driver maps; it - bounds only *what* it may map. That was already true. -- **The device tree as a published thing.** `/system/devices` remains a device - manager concern, as the file-system hierarchy already assumes. - -## Migration sketch - -Not a plan yet — the phases need sizing once the IOMMU question is settled. -Rough shape: introduce the capability kind and `resource_derive` alongside the -existing table; convert the resource-consuming syscalls to accept either form; -move the inventory into `device-manager` and convert its clients off -`device_enumerate`; then delete the table, the two constants and the three -syscalls in one flag-day, as the `ServiceId` retirement did. - -Every driver is affected, so the QEMU suite is the arbiter at each step, and the -Ryzen is the acceptance test — it is the machine that found this, and the one -that proves it fixed. +The Ryzen is the acceptance test for the ceiling half: it is the machine that found the +constants, and the one that proves them gone. diff --git a/library/device/driver/driver.zig b/library/device/driver/driver.zig index b34af41..0b2ea4b 100644 --- a/library/device/driver/driver.zig +++ b/library/device/driver/driver.zig @@ -46,12 +46,31 @@ pub fn enumerate(buffer: []DeviceDescriptor) usize { return sc.systemCall2(.device_enumerate, @intFromPtr(buffer.ptr), buffer.len); } +/// Why a `transfer` failed. `NotHeld` is the interesting one — it means the caller tried +/// to give away a device it does not have, which is the whole rule. +pub const TransferError = error{ NoSuchDevice, NotHeld, NoSuchTask, Refused }; + +/// Give device `id` to task `to`. **A move, not a copy** — a claim is exclusive, so the +/// caller stops holding it. This is how the device manager hands a driver the device it +/// matched, replacing first-come-first-served claiming with policy +/// (docs/os-development/device-authority.md). +pub fn transfer(id: u64, to: u32) TransferError!void { + const r = sc.systemCall2(.device_transfer, id, to); + if (!failed(r)) return; + return switch (errnoOf(r)) { + abi.ENODEV => error.NoSuchDevice, + abi.EPERM => error.NotHeld, + abi.ESRCH => error.NoSuchTask, + else => error.Refused, + }; +} + /// Why a `claim` failed. Worth distinguishing: `AlreadyClaimed` means back off and /// let the owner have it, `NoSuchDevice` means this id is stale and the caller should /// re-enumerate, and `NotConfined` means the machine could not place the device under /// IOMMU translation — the claim was rolled back, and that one is a fault report, not /// a retry. `Refused` is an errno this library does not know a name for. -pub const ClaimError = error{ NoSuchDevice, AlreadyClaimed, NotConfined, Refused }; +pub const ClaimError = error{ NoSuchDevice, AlreadyClaimed, NotConfined, NotYours, Refused }; /// Take exclusive ownership of device `id`. pub fn claim(id: u64) ClaimError!void { @@ -61,6 +80,7 @@ pub fn claim(id: u64) ClaimError!void { abi.ENODEV => error.NoSuchDevice, abi.EBUSY => error.AlreadyClaimed, abi.ECONFINE => error.NotConfined, + abi.EPERM => error.NotYours, // delegated hardware: it must be handed to you else => error.Refused, }; } diff --git a/library/kernel/process.zig b/library/kernel/process.zig index c341189..c3c5129 100644 --- a/library/kernel/process.zig +++ b/library/kernel/process.zig @@ -182,6 +182,22 @@ pub fn spawnWithArguments(name: []const u8, arguments: []const []const u8) ?u32 /// `ipc.replyWait` as a child-exit badge (`ipc.Received.isChildExit`/`childProcessId`), so /// one endpoint can supervise many children. Returns the child's process id, or null. pub fn spawnSupervised(name: []const u8, arguments: []const []const u8, exit_endpoint: ?usize) ?u32 { + return spawnSupervisedWithDevice(name, arguments, exit_endpoint, abi.no_device); +} + +/// Spawn a supervised child **and give it a device you hold**, in one call. +/// +/// The device manager's path: it holds the hardware and the driver it starts must have +/// it. Fusing the handover into the spawn is what removes the window a separate +/// transfer would leave — the child cannot run without its device, because it does not +/// exist until it holds it (docs/os-development/device-authority.md). The kernel checks +/// only that the device is the caller's to give. +pub fn spawnSupervisedWithDevice( + name: []const u8, + arguments: []const []const u8, + exit_endpoint: ?usize, + device: u64, +) ?u32 { var blob: [256]u8 = undefined; var len: usize = 0; for (arguments, 0..) |argument, i| { @@ -194,7 +210,7 @@ pub fn spawnSupervised(name: []const u8, arguments: []const []const u8, exit_end @memcpy(blob[len..][0..argument.len], argument); len += argument.len; } - const r = sc.systemCall5(.system_spawn, @intFromPtr(name.ptr), name.len, if (len == 0) 0 else @intFromPtr(&blob), len, exit_endpoint orelse abi.no_cap); + const r = sc.systemCall6(.system_spawn, @intFromPtr(name.ptr), name.len, if (len == 0) 0 else @intFromPtr(&blob), len, exit_endpoint orelse abi.no_cap, device); if (r > ~@as(usize, 0) - 4095) return null; // a wrapped -errno return @intCast(r); } diff --git a/library/kernel/system-call.zig b/library/kernel/system-call.zig index 5406a90..244a575 100644 --- a/library/kernel/system-call.zig +++ b/library/kernel/system-call.zig @@ -65,3 +65,19 @@ pub inline fn systemCall5(n: SystemCall, a0: usize, a1: usize, a2: usize, a3: us [a4] "{r8}" (a4), : .{ .rcx = true, .r11 = true, .memory = true }); } + +/// The sixth and last argument register. `syscall` clobbers rcx, so r10 stands in for +/// it and r9 is the end of the line — a call needing a seventh would have to pass a +/// struct instead. +pub inline fn systemCall6(n: SystemCall, a0: usize, a1: usize, a2: usize, a3: usize, a4: usize, a5: usize) usize { + return asm volatile ("syscall" + : [ret] "={rax}" (-> usize), + : [n] "{rax}" (@intFromEnum(n)), + [a0] "{rdi}" (a0), + [a1] "{rsi}" (a1), + [a2] "{rdx}" (a2), + [a3] "{r10}" (a3), + [a4] "{r8}" (a4), + [a5] "{r9}" (a5), + : .{ .rcx = true, .r11 = true, .memory = true }); +} diff --git a/system/abi.zig b/system/abi.zig index d8f2157..8b18754 100644 --- a/system/abi.zig +++ b/system/abi.zig @@ -48,7 +48,7 @@ pub const SystemCall = enum(u64) { irq_bind = 14, // irq_bind(id, resource_index, endpoint): deliver a device IRQ as an IPC notification irq_ack = 15, // irq_ack(id, resource_index): re-arm a bound IRQ after servicing it device_register = 16, // device_register(parent_id, descriptor) -> id/-errno: publish a child of a device you claimed (-ENOSPC table full, -ECHILDREN parent full, -ERANGE resource escapes the parent, -ENODEV/-EPERM bad parent, -E2BIG too many resources) - system_spawn = 17, // system_spawn(name_ptr, name_len, arguments_ptr, arguments_len, exit_endpoint) -> child process id: start a named initial-ramdisk binary as a new ring-3 process + system_spawn = 17, // system_spawn(name_ptr, name_len, arguments_ptr, arguments_len, exit_endpoint, device) -> child process id: start a named initial-ramdisk binary as a new ring-3 process. `device` (or `no_device`) is a device the caller holds and gives to the child, atomically — the child never runs without it dma_alloc = 18, // dma_alloc(len, flags) -> virtual_address (rax), physical_address (rdx): contiguous, pinned, uncacheable DMA memory dma_free = 19, // dma_free(virtual_address, len) -> 0: release a prior dma_alloc msi_bind = 20, // msi_bind(device_id, endpoint) -> address (rax), data (rdx): a per-device MSI vector for a claimed device @@ -85,6 +85,7 @@ pub const SystemCall = enum(u64) { dma_bind = 51, // dma_bind(device_id, region_handle) -> 0/-errno: map a DMA-region capability into the claimed device's IOMMU domain (idempotent). The caller must own the device and hold the handle dma_unbind = 52, // dma_unbind(device_id, region_handle) -> 0/-errno: unmap a previously bound region from the device's domain and invalidate handle_close = 53, // handle_close(handle) -> 0/-errno: drop one capability handle and free its table slot (endpoints, shared-memory, DMA regions) + device_transfer = 54, // device_transfer(device_id, task_id) -> 0/-errno: give a device you hold to another task. A MOVE, not a copy — a claim is exclusive (-ENODEV no such device, -EPERM you do not hold it, -ESRCH no such task) _, }; @@ -126,6 +127,11 @@ pub const ECONFINE: i64 = 16; // the device could not be placed under IOMMU tran /// space, and the number to bump when adding one. pub const errno_maximum: i64 = 16; +/// `system_spawn`'s `device` argument when the child is given no device — every +/// caller but the device manager. Matches `device-manager-protocol.no_device`, which +/// is the same sentinel one layer up. +pub const no_device: u64 = ~@as(u64, 0); + /// `futex_wait` return codes (in rax). pub const futex_woken: u64 = 0; // woken by a futex_wake pub const futex_mismatch: u64 = 1; // *addr != expected on entry; the caller did not block diff --git a/system/drivers/pci-bus/pci-bus.zig b/system/drivers/pci-bus/pci-bus.zig index 5cd81cc..77af176 100644 --- a/system/drivers/pci-bus/pci-bus.zig +++ b/system/drivers/pci-bus/pci-bus.zig @@ -75,11 +75,20 @@ fn configWrite16(bus: u64, dev: u64, function: u64, offset: u64, value: u16) voi configWrite(bus, dev, function, aligned, (word & ~mask) | (@as(u32, value) << shift)); } -/// Claim the bridge, map the ECAM, hello the manager, then scan. +/// Hello the manager (which is where the bridge arrives), map the ECAM, then scan. fn initialise(endpoint: ipc.Handle) bool { _ = endpoint; - device.claim(bridge_id) catch |e| { - std.log.info("unable to claim bridge device {d}: {s}", .{ bridge_id, @errorName(e) }); + // **The handshake first, because it is where the device arrives.** This driver + // used to claim `bridge_id` here — first-come-first-served, so the manager's + // matching was advisory and any process could have claimed the bridge by naming + // the same id. The manager now holds it and transfers it in the hello reply + // (docs/os-development/device-authority.md). `hello` is synchronous, so the + // transfer has completed by the time this returns. + // + // Keep the manager handle to report children through; a supervised bus that + // cannot reach its manager has nothing to serve. + manager_handle = device_manager.hello(.bus, bridge_id) orelse { + std.log.info("no hello with the device manager; bridge {d} not delegated", .{bridge_id}); return false; }; const buffer = memory.allocator().alloc(device.DeviceDescriptor, 64) catch { @@ -113,11 +122,6 @@ fn initialise(endpoint: ipc.Handle) bool { return false; }; - // The handshake (role: bus — we enumerate PCI and report the functions we - // find), then the scan. Keep the manager handle to report children through; - // a supervised bus that cannot reach its manager has nothing to serve. - manager_handle = device_manager.hello(.bus, bridge_id) orelse return false; - scan(); return true; } diff --git a/system/drivers/ps2-bus/ps2-bus.zig b/system/drivers/ps2-bus/ps2-bus.zig index fcc0c72..9911562 100644 --- a/system/drivers/ps2-bus/ps2-bus.zig +++ b/system/drivers/ps2-bus/ps2-bus.zig @@ -114,10 +114,9 @@ pub fn main() void { _ = logging.write("/system/drivers/ps2-bus: found PS/2 controller\n"); _ = logging.write("/system/drivers/ps2-bus: initializing controller\n"); - device.claim(controller_device_descriptor.id) catch |e| { - std.log.warn("unable to claim controller: {s}", .{@errorName(e)}); - return; - }; + // The controller node arrived with the spawn — the manager holds the hardware + // and names it in the call that creates this process + // (docs/os-development/device-authority.md). Nothing to claim. const controller = ps2.Controller.init(controller_device_descriptor) orelse { _ = logging.write("/system/drivers/ps2-bus: controller is missing its IO ports\n"); @@ -255,18 +254,21 @@ pub fn main() void { if (port_device_types[@intFromEnum(ps2.Port.two)] != null) { if (ps2.findMouseDescriptor(buffer)) |descriptor| { if (findInterruptResourceIndex(descriptor)) |auxiliary_index| { - if (device.claim(descriptor.id)) |_| { - if (device.irqBind(descriptor.id, auxiliary_index, endpoint)) { - maybe_auxiliary_interrupt = .{ - .device_id = descriptor.id, - .interrupt_index = auxiliary_index, - .gsi = descriptor.resources[auxiliary_index].start, - }; - } else { - _ = logging.write("/system/drivers/ps2-bus: auxiliary irq_bind failed\n"); - } - } else |e| { - std.log.warn("auxiliary claim failed: {s}", .{@errorName(e)}); + // The mouse node is a *second* device for this one instance — the + // 8042 is one controller with two ports, so it cannot be split across + // two processes. It is transferred to us after the spawn, which is + // safe because we only reach it here, long after the controller + // handshakes and identify. irq_bind is the proof we hold it: it is + // gated on ownership, so a failure here means the handover has not + // landed rather than a hardware problem. + if (device.irqBind(descriptor.id, auxiliary_index, endpoint)) { + maybe_auxiliary_interrupt = .{ + .device_id = descriptor.id, + .interrupt_index = auxiliary_index, + .gsi = descriptor.resources[auxiliary_index].start, + }; + } else { + _ = logging.write("/system/drivers/ps2-bus: auxiliary irq_bind failed (mouse node not delegated?)\n"); } } } diff --git a/system/drivers/usb-xhci-bus/usb-xhci-bus.zig b/system/drivers/usb-xhci-bus/usb-xhci-bus.zig index 7a1d8b0..8fcb5fb 100644 --- a/system/drivers/usb-xhci-bus/usb-xhci-bus.zig +++ b/system/drivers/usb-xhci-bus/usb-xhci-bus.zig @@ -173,10 +173,22 @@ fn initialise(endpoint: ipc.Handle) bool { if (!channel.bindPatiently("usb-transfer", endpoint)) _ = logging.write("/system/drivers/usb-xhci-bus: /protocol/usb-transfer is another controller's; serving mine unnamed\n"); - device.claim(controller_id) catch |e| { - std.log.warn("unable to claim controller device {d}: {s}", .{ controller_id, @errorName(e) }); + // **The handshake comes first, because it is where the device arrives.** This + // driver used to claim `controller_id` here — first-come-first-served, so the + // manager's matching was advisory and any process could have claimed it by + // passing the same integer. Now the manager holds the controller and transfers + // it in `onHello`, so by the time this call returns the device is ours and + // nothing else could have taken it (docs/os-development/device-authority.md). + // + // `hello` is synchronous, so the transfer has completed before the reply lands — + // there is no window between being told yes and holding the thing. + // + // Keep the handle: the tick's hot-plug dispatch reports through it. + const handle = device_manager.hello(.bus, controller_id) orelse { + std.log.warn("no hello with the device manager; controller {d} not delegated", .{controller_id}); return false; }; + manager_handle = handle; // Fetch our own descriptor back for the controller's resources. const buffer = memory.allocator().alloc(device.DeviceDescriptor, 64) catch { @@ -204,7 +216,7 @@ fn initialise(endpoint: ipc.Handle) bool { std.log.info("controller device {d} has no register BAR", .{controller_id}); return false; }; - std.log.info("claimed controller device {d} (registers at 0x{x}, {d} bytes)", .{ + std.log.info("controller device {d} registers at 0x{x}, {d} bytes", .{ controller_id, register_window.start, register_window.len, @@ -228,8 +240,13 @@ fn initialise(endpoint: ipc.Handle) bool { _ = logging.write("/system/drivers/usb-xhci-bus: controller reset/bring-up failed\n"); return false; }; - std.log.info("controller running ({d} slots, {d}-byte contexts)", .{ + // Both numbers, because for a long time they disagreed silently: the controller + // reported its real slot count and the driver tracked a fixed 8 of them, so a + // ninth device — trivially reachable behind a hub — simply did not exist. They + // must now be equal, and the suite asserts it. + std.log.info("controller running ({d} slots, tracking {d}, {d}-byte contexts)", .{ controller.?.max_slots, + controller.?.devices.len, controller.?.context_size, }); // The proof of life: a No-Op command round-trips the command ring, the event @@ -242,12 +259,6 @@ fn initialise(endpoint: ipc.Handle) bool { return false; } - // The handshake (role: bus — we enumerate USB ports and report the devices - // behind them), inside the manager's hello deadline. Keep the handle: the - // tick's hot-plug dispatch reports through it. - const handle = device_manager.hello(.bus, controller_id) orelse return false; - manager_handle = handle; - scanPorts(handle); // Arm the timer: in polling mode it drains the event ring; in MSI mode it is the diff --git a/system/drivers/usb-xhci-bus/usb-xhci-library.zig b/system/drivers/usb-xhci-bus/usb-xhci-library.zig index fb0862d..9d0186b 100644 --- a/system/drivers/usb-xhci-bus/usb-xhci-library.zig +++ b/system/drivers/usb-xhci-bus/usb-xhci-library.zig @@ -221,10 +221,17 @@ fn intervalFor(speed: u32, b_interval: u8) u32 { }; } -// Upper bounds on what one device's active configuration describes. A boot -// keyboard or mouse has one interface with one interrupt endpoint; a flash drive -// has one interface with two bulk endpoints. Generous for those. -pub const max_interfaces = 4; +/// bound: endpoints recorded per interface +/// decided-by: external +/// protects: the fixed endpoint array inside InterfaceInfo +/// at-limit: refuse — the surplus endpoint is not recorded and the class driver +/// cannot bind it +/// observed-by: the class driver failing to find the endpoint it wants +/// +/// Held at 4 because the usb-transfer wire protocol reports exactly +/// `max_reported_endpoints` (4) endpoints per interface, so widening this alone +/// would change nothing a class driver can see. Lifting it is a protocol change — +/// docs/bounds-track-plan.md, out of scope for the unattended run. pub const max_endpoints_per_interface = 4; // The endpoint-descriptor facts a class driver needs to talk to an endpoint: its @@ -256,7 +263,19 @@ const ConfiguredEndpoint = struct { dci: u32 = 0, ring: ProducerRing = .{ .region = .{ .virtual = 0, .physical = 0 } }, }; -const max_configured_endpoints = max_interfaces * max_endpoints_per_interface; +/// bound: transfer rings configured on one device slot +/// decided-by: hardware +/// protects: the per-device ring table, sized at compile time +/// at-limit: refuse — configureEndpoint returns null and its caller reports it +/// observed-by: the class driver's own failure to bind an endpoint +/// +/// The xHCI specification's number rather than ours: a Device Context carries a slot +/// context plus at most 31 endpoint contexts, because the Context Entries field that +/// addresses them is 5 bits (DCI 1..31, DCI 1 being the default control endpoint). +/// A device physically cannot present more. This was +/// `max_interfaces * max_endpoints_per_interface` = 16, so it moved whenever either +/// of those two guesses moved. +const max_configured_endpoints = 31; // One addressed USB device behind this controller: its hardware slot, its EP0 // (control) transfer ring, the DMA context + bounce buffer the control pipe uses, @@ -277,7 +296,13 @@ pub const Device = struct { device_descriptor: usb_abi.DeviceDescriptor = std.mem.zeroes(usb_abi.DeviceDescriptor), configuration_value: u8 = 0, interface_count: u8 = 0, - interfaces: [max_interfaces]InterfaceInfo = [_]InterfaceInfo{.{}} ** max_interfaces, + /// The interfaces of the active configuration, allocated at enumeration from the + /// count the configuration descriptor actually declares. This was a fixed 4, and + /// the fifth interface of a composite device (a headset, a webcam with audio, a + /// dock, a multifunction printer) did not merely go missing — see + /// parseConfiguration, where its endpoints were appended to interface 3's list. + interfaces: []InterfaceInfo = &.{}, + // Transfer rings configured for this device's interrupt/bulk endpoints. endpoint_ring_count: u8 = 0, endpoint_rings: [max_configured_endpoints]ConfiguredEndpoint = [_]ConfiguredEndpoint{.{}} ** max_configured_endpoints, @@ -296,6 +321,15 @@ pub const Device = struct { // Downstream ports with a pending change to service (bit P = port P), set // by the status-change endpoint (and by an initial sweep in setupHub). hub_change_mask: u32 = 0, + + /// Release the interface list. Safe to call twice, and on a device that never + /// enumerated — re-enumeration and teardown both come through here. + pub fn freeInterfaces(self: *Device) void { + if (self.interfaces.len != 0) memory.allocator().free(self.interfaces); + self.interfaces = &.{}; + self.interface_count = 0; + } + }; // A standing interrupt-IN subscription: the endpoint's ring is kept armed with a @@ -328,9 +362,6 @@ pub const Report = struct { data: [64]u8 = [_]u8{0} ** 64, }; -// How many addressed devices this driver tracks at once. QEMU presents a handful -// (a keyboard, a mouse, a storage stick); a fuller machine would grow this. -const max_devices = 8; const max_subscriptions = 8; const report_queue_capacity = 16; @@ -449,7 +480,14 @@ pub const Controller = struct { device_context_array: memory.DmaRegion, command_ring: ProducerRing, event_ring: EventRing, - devices: [max_devices]Device = [_]Device{.{}} ** max_devices, + /// One entry per device slot the **controller** says it has (HCSPARAMS1.MaxSlots, + /// 1..255), allocated at bring-up. This used to be a fixed 8 with the comment "QEMU + /// presents a handful; a fuller machine would grow this" — which is the shape the + /// bounds rule exists to stop, since the controller has always reported the real + /// number and `op_config` below is already programmed with it. A desktop's keyboard, + /// mouse, webcam, headset, hub and two sticks reach 8 without trying, and everything + /// past it vanished (behind a hub, without even a log line). + devices: []Device, subscriptions: [max_subscriptions]Subscription = [_]Subscription{.{}} ** max_subscriptions, report_queue: [report_queue_capacity]Report = [_]Report{.{}} ** report_queue_capacity, report_count: usize = 0, @@ -582,8 +620,16 @@ pub const Controller = struct { .device_context_array = undefined, .command_ring = undefined, .event_ring = undefined, + .devices = &.{}, }; + // One tracking slot per slot the controller reports. A controller that claims + // no slots cannot address anything, so treat that as a dead controller rather + // than allocating nothing and failing mysteriously later. + if (self.max_slots == 0) return null; + self.devices = memory.allocator().alloc(Device, self.max_slots) catch return null; + for (self.devices) |*device| device.* = .{}; + // Wait for the controller to report ready, then halt it if it is running. if (!waitClear(self.operational(op_usbsts), usbsts_controller_not_ready)) return null; if (read32(self.operational(op_usbcmd)) & usbcmd_run != 0) { @@ -790,7 +836,7 @@ pub const Controller = struct { } fn allocateDevice(self: *Controller) ?*Device { - for (&self.devices) |*device| { + for (self.devices) |*device| { if (!device.used) return device; } return null; @@ -897,7 +943,8 @@ pub const Controller = struct { return null; }; const device = self.allocateDevice() orelse { - std.log.info("port {d} setup: no free device slot", .{port}); + std.log.warn("port {d} setup: no free device slot", .{port}); + self.disableSlot(slot_id); // the controller already gave it to us return null; }; device.* = .{ @@ -1011,7 +1058,7 @@ pub const Controller = struct { /// The next pending (hub, downstream-port) change to service, or null. Clears /// the returned port's bit. Called on the bus tick. pub fn takeHubChange(self: *Controller) ?struct { hub: *Device, port: u16 } { - for (&self.devices) |*device| { + for (self.devices) |*device| { if (!device.used or !device.is_hub or device.hub_change_mask == 0) continue; const bit: u5 = @intCast(@ctz(device.hub_change_mask)); device.hub_change_mask &= ~(@as(u32, 1) << bit); @@ -1084,7 +1131,7 @@ pub const Controller = struct { } pub fn deviceOnHubPort(self: *Controller, hub: *Device, port: u16) ?*Device { - for (&self.devices) |*device| { + for (self.devices) |*device| { if (device.used and device.parent_slot == hub.slot_id and device.parent_port == port) return device; } return null; @@ -1099,7 +1146,11 @@ pub const Controller = struct { std.log.info("hub slot {d} port {d}: Enable Slot failed", .{ hub.slot_id, port }); return null; }; - const device = self.allocateDevice() orelse return null; + const device = self.allocateDevice() orelse { + std.log.warn("hub slot {d} port {d}: no free device slot", .{ hub.slot_id, port }); + self.disableSlot(slot_id); // the controller already gave it to us + return null; + }; const child_speed = mapHubPortSpeed(speed); device.* = .{ .used = true, @@ -1150,6 +1201,7 @@ pub const Controller = struct { fn abandon(self: *Controller, device: *Device) ?*Device { _ = self; + device.freeInterfaces(); device.used = false; return null; } @@ -1299,12 +1351,25 @@ pub const Controller = struct { const configuration = std.mem.bytesToValue(usb_abi.ConfigurationDescriptor, &header); device.configuration_value = @intFromEnum(configuration.configuration_value); - // Read the whole block into a local buffer and parse it here (in the bus - // driver) so the parse never has to cross the 256-byte IPC boundary. - var blob: [512]u8 = undefined; - const length = @min(configuration.total_length, blob.len); - if (!self.controlTransfer(device, usb_abi.getDescriptor(.configuration, 0, 0, @intCast(length)), blob[0..length], true)) return false; - parseConfiguration(device, blob[0..length]); + // Read the whole block and parse it here (in the bus driver) so the parse + // never has to cross the 256-byte IPC boundary. Sized by the device's own + // wTotalLength — the ceiling is then the field's u16, which is the USB + // specification's, not ours. + // + // This was a fixed 512 with an `@min` clamp, which silently truncated: a + // composite device routinely exceeds it (a headset is 500-900 bytes, a UVC + // webcam 1-3 KB, a multifunction printer 600+), and the interfaces past the + // cut simply did not exist — while SET_CONFIGURATION below still configured + // the device for all of them. QEMU's boot keyboard, mouse and stick are all + // under 100 bytes, which is why it survived. + if (configuration.total_length < header.len) return false; // shorter than its own header + const blob = memory.allocator().alloc(u8, configuration.total_length) catch return false; + defer memory.allocator().free(blob); + if (!self.controlTransfer(device, usb_abi.getDescriptor(.configuration, 0, 0, configuration.total_length), blob, true)) return false; + // Both numbers, so a truncation can never again be invisible: the block the + // device declared, and the bytes actually fetched and parsed. They must match. + std.log.info("config block {d} bytes, read {d}", .{ configuration.total_length, blob.len }); + if (!parseConfiguration(device, blob)) return false; // Select the configuration, moving the device to the configured state. if (!self.controlTransfer(device, usb_abi.setConfiguration(configuration.configuration_value), &.{}, false)) return false; @@ -1315,8 +1380,30 @@ pub const Controller = struct { // and the endpoints that follow it. Endpoints belong to the most recent // interface. Unknown descriptor types (HID, class-specific) are skipped by // their length. - fn parseConfiguration(device: *Device, blob: []const u8) void { - device.interface_count = 0; + /// Two passes: count the alternate-setting-0 interfaces the block declares, + /// allocate exactly that many, then fill them. False only on an allocation + /// failure. The count cannot exceed 255 — `bNumInterfaces` is a u8, so that is + /// the USB specification's ceiling and not one of ours. + fn parseConfiguration(device: *Device, blob: []const u8) bool { + var declared: usize = 0; + var count_offset: usize = 0; + while (count_offset + 2 <= blob.len) { + const length = blob[count_offset]; + if (length < 2 or count_offset + length > blob.len) break; + if (@as(usb_abi.DescriptorType, @enumFromInt(blob[count_offset + 1])) == .interface and + length >= @sizeOf(usb_abi.InterfaceDescriptor)) + { + const descriptor = std.mem.bytesToValue(usb_abi.InterfaceDescriptor, blob[count_offset .. count_offset + @sizeOf(usb_abi.InterfaceDescriptor)]); + if (@intFromEnum(descriptor.alternate_setting) == 0 and declared < 255) declared += 1; + } + count_offset += length; + } + + device.freeInterfaces(); + if (declared == 0) return true; + device.interfaces = memory.allocator().alloc(InterfaceInfo, declared) catch return false; + for (device.interfaces) |*interface| interface.* = .{}; + var current: ?*InterfaceInfo = null; var offset: usize = 0; while (offset + 2 <= blob.len) { @@ -1326,9 +1413,17 @@ pub const Controller = struct { switch (@as(usb_abi.DescriptorType, @enumFromInt(descriptor_type))) { .interface => if (length >= @sizeOf(usb_abi.InterfaceDescriptor)) { const descriptor = std.mem.bytesToValue(usb_abi.InterfaceDescriptor, blob[offset .. offset + @sizeOf(usb_abi.InterfaceDescriptor)]); - if (@intFromEnum(descriptor.alternate_setting) != 0) { - current = null; // ignore alternate settings for now - } else if (device.interface_count < max_interfaces) { + // An interface we do not record MUST clear `current`, or the + // endpoints that follow it attach to the previous one. The cap + // branch used to have no `else`, so a fifth interface's endpoints + // were appended to interface 3's array and a class driver bound to + // interface 3 could be handed an endpoint belonging to something + // else entirely — silently, with a truthful-looking count logged. + if (@intFromEnum(descriptor.alternate_setting) != 0 or + device.interface_count >= device.interfaces.len) + { + current = null; + } else { const slot = &device.interfaces[device.interface_count]; slot.* = .{ .number = @intFromEnum(descriptor.interface_number), @@ -1358,13 +1453,14 @@ pub const Controller = struct { } offset += length; } + return true; } // --- endpoint configuration + interrupt / bulk transfers --------------- /// Find the tracked device and interface an assigned device id belongs to. pub fn findInterface(self: *Controller, device_id: u64) ?struct { device: *Device, interface: *InterfaceInfo } { - for (&self.devices) |*device| { + for (self.devices) |*device| { if (!device.used) continue; for (device.interfaces[0..device.interface_count]) |*interface| { if (interface.registered_device_id == device_id) return .{ .device = device, .interface = interface }; @@ -1633,7 +1729,7 @@ pub const Controller = struct { /// The tracked device on `port`, or null. pub fn deviceOnPort(self: *Controller, port: u32) ?*Device { - for (&self.devices) |*device| { + for (self.devices) |*device| { if (device.used and device.port == port) return device; } return null; @@ -1642,7 +1738,7 @@ pub const Controller = struct { /// The next used device whose parent hub is `hub_slot` and slot id > `after` /// (for recursive teardown when a hub itself disconnects), or null. pub fn nextChildOf(self: *Controller, hub_slot: u8, after: u8) ?*Device { - for (&self.devices) |*device| { + for (self.devices) |*device| { if (device.used and device.parent_slot == hub_slot and device.slot_id > after) return device; } return null; @@ -1656,15 +1752,27 @@ pub const Controller = struct { for (&self.subscriptions) |*subscription| { if (subscription.active and subscription.slot_id == device.slot_id) subscription.active = false; } + self.disableSlot(device.slot_id); + device.freeInterfaces(); + device.used = false; + } + + /// Hand a slot back to the controller and clear its context-array entry. + /// + /// Every path that has issued a successful Enable Slot owes this, including the + /// ones that then fail to bring the device up. A slot the driver forgets is one + /// the controller never reissues, so the loss is permanent for the boot: the two + /// setup paths used to return null straight after a failed `allocateDevice`, + /// leaking a slot per attempt — and the hub path did it without even a log line. + fn disableSlot(self: *Controller, slot_id: u8) void { const physical = self.submitCommand(.{ - .control = trbControl(.disable_slot, @as(u32, device.slot_id) << 24), + .control = trbControl(.disable_slot, @as(u32, slot_id) << 24), }); if (self.awaitCommand(physical)) |code| { if (code != @intFromEnum(CompletionCode.success)) - std.log.info("slot {d}: Disable Slot completion code {d}", .{ device.slot_id, code }); - } else std.log.info("slot {d}: Disable Slot timed out", .{device.slot_id}); + std.log.info("slot {d}: Disable Slot completion code {d}", .{ slot_id, code }); + } else std.log.info("slot {d}: Disable Slot timed out", .{slot_id}); const array: [*]volatile u64 = @ptrFromInt(self.device_context_array.virtual); - array[device.slot_id] = 0; - device.used = false; + array[slot_id] = 0; } }; diff --git a/system/drivers/virtio-gpu/virtio-gpu.zig b/system/drivers/virtio-gpu/virtio-gpu.zig index c26c8da..d625992 100644 --- a/system/drivers/virtio-gpu/virtio-gpu.zig +++ b/system/drivers/virtio-gpu/virtio-gpu.zig @@ -200,10 +200,9 @@ fn testPixel(index: u32) u32 { fn initialise(endpoint: ipc.Handle) bool { _ = endpoint; - device.claim(device_id) catch |e| { - std.log.info("unable to claim device {d}: {s}", .{ device_id, @errorName(e) }); - return false; - }; + // The device arrived with the spawn — the manager holds it and names it in the call + // that creates this process, so it is ours before the first instruction here + // (docs/os-development/device-authority.md). Nothing to claim. var descriptors: [64]device.DeviceDescriptor = undefined; const total = device.enumerate(&descriptors); diff --git a/system/kernel/devices-broker.zig b/system/kernel/devices-broker.zig index a985dd6..0d6b8a2 100644 --- a/system/kernel/devices-broker.zig +++ b/system/kernel/devices-broker.zig @@ -22,29 +22,114 @@ const std = @import("std"); const abi = @import("abi"); const platform = @import("platform"); const device_abi = @import("device-abi"); +const heap = @import("heap.zig"); -pub const maximum_devices = 64; +/// How many devices a single registrar may put in the table. +/// +/// **This is a runaway detector, not a security boundary**, and the difference matters. +/// It cannot stop a malicious driver — a quota generous enough never to bite a real +/// machine is still generous enough to be unpleasant — and it is not trying to. What +/// stops malice is that only a driver the manager handed a device can register children +/// under it (docs/os-development/device-authority.md). What this catches is a +/// *legitimate* driver in a loop, early, and attributably: the driver that did it is +/// refused, named in its own log, and restarted, while every other driver is untouched. +/// +/// The shared ceilings it replaces could not do that. `maximum_devices = 64` was a +/// guess about someone else's computer and one driver's enumeration starved every +/// other — which is how an AMD Ryzen came to boot with a working display, no USB and no +/// storage. A per-registrar allowance is the same protection charged to whoever caused +/// it, which is the microkernel property rather than a workaround for it. +/// +/// The number: a machine's whole PCI segment tops out at 65536 functions, and the +/// biggest real registrar seen is pci-bus at a few dozen. 4096 is far above anything a +/// real machine produces and around 1.4 MiB of descriptors, well under the kernel heap. +/// **Reaching it is a bug report, not a tuning request** — no correct driver gets near. +/// +/// bound: devices one task may register +/// decided-by: ours +/// protects: the kernel heap, against a driver looping device_register +/// at-limit: refuse — ECHILDREN to the registrar; every other driver is unaffected +/// observed-by: the bus driver's line naming the reason (pci-bus reconciles functions +/// found against registered) +const maximum_devices_per_registrar = 4096; -/// Cap on children a single parent may have. A zero-resource child (legal — a USB -/// device is addressed through its controller, not by MMIO) sidesteps the containment -/// check, so without a bound a process that claimed one device could loop -/// `device_register` and exhaust the whole table, permanently denying it to every other -/// driver. This bounds the blast radius of one claim; a real quota (and a -/// `device_release` to reclaim on exit) is future work — see docs/driver-model.md. -const maximum_children_per_parent = 16; -var devices: [maximum_devices]device_abi.DeviceDescriptor = undefined; -var claimed: [maximum_devices]?u32 = .{null} ** maximum_devices; // owner task id, or null +/// The table, grown on demand from the kernel heap. **There is no ceiling**: how many +/// devices a machine has is the machine's business, and no specification bounds it, so +/// nothing here should. `devices_broker.init` runs after `heap.init` (kernel.zig), so +/// there was never a reason for this to be static beyond it having been written that +/// way first. +var devices: []device_abi.DeviceDescriptor = &.{}; +/// Owner task id per device, or null. Parallel to `devices` and grown with it. +var claimed: []?u32 = &.{}; +/// The task that called `register` for each device, so the per-registrar allowance can +/// be charged to whoever caused the entry. Firmware-discovered nodes carry `no_registrar` +/// — they are the kernel's own, not anybody's doing. +/// Optional, not a sentinel: **task 0 is a real task** (the kernel's own), so any +/// "none" value inside the id space is a device belonging to somebody reading back as +/// belonging to nobody. Found the moment the first test asserted a giver, because the +/// task doing the giving was task 0. +var registrar: []?u32 = &.{}; + +/// **Who gave each device away**, or `no_giver` if nobody ever did. +/// +/// One field, and the whole authority rule follows from it: a device that was *given* +/// to someone is delegated hardware, so it may only be handed on, never taken +/// (`claim` refuses it); and when its holder dies it goes back to whoever lent it, +/// rather than becoming free for anyone to grab. +/// +/// It also settles the framebuffer without mentioning it. Nobody delegates the +/// loader's framebuffer, so it has no giver, so the display service claims it exactly +/// as it always has — no exemption, no special case, no `display` anywhere in the rule. +var giver: []?u32 = &.{}; var count: usize = 0; +/// Grow the three parallel arrays so at least one more device fits. False if the heap +/// cannot satisfy it, which the callers report rather than swallow. +fn reserve() bool { + if (count < devices.len) return true; + // Double from a deliberately SMALL first block. Sizing it for a typical machine + // would mean the growth path never ran on the hardware we test on, and only woke up + // on someone else's larger machine — which is the exact shape of the failure this + // whole track exists to stop. At 8, every boot grows the table several times, so + // the path is exercised constantly and the suite asserts it. + const wanted = if (devices.len == 0) 8 else devices.len * 2; + const allocator = heap.allocator(); + const grown_devices = allocator.realloc(devices, wanted) catch return false; + devices = grown_devices; + const grown_claimed = allocator.realloc(claimed, wanted) catch return false; + claimed = grown_claimed; + const grown_registrar = allocator.realloc(registrar, wanted) catch return false; + registrar = grown_registrar; + const grown_giver = allocator.realloc(giver, wanted) catch return false; + giver = grown_giver; + for (claimed[count..], registrar[count..], giver[count..]) |*slot, *who, *lender| { + slot.* = null; + who.* = null; + lender.* = null; + } + return true; +} + +/// How many devices `task` has registered — the allowance is charged per registrar, so +/// a driver in a loop exhausts its own and no one else's. +fn registeredBy(task: u32) usize { + var n: usize = 0; + for (registrar[0..count]) |who| { + if (who != null and who.? == task) n += 1; + } + return n; +} + /// The id of the seeded framebuffer node (`seedDisplay`), or null when the machine /// handed over no framebuffer. Lets the process layer recognise the display claim /// (to quiesce the bootstrap console) without threading the id through every caller. var display_device: ?u64 = null; -/// Devices discovery found but the table had no room for. Non-zero means the machine -/// is bigger than `maximum_devices` and some hardware is simply invisible to drivers — -/// which would otherwise be an entirely silent failure. Logged at boot. +/// Devices discovery found but could not record. The table grows on demand, so this is +/// no longer "the machine is bigger than our guess" — it means the kernel heap could not +/// satisfy the growth, which would otherwise be an entirely silent failure. Logged at +/// boot. pub var dropped: usize = 0; /// Snapshot the device tree into the flat table. Run once, right after discovery. @@ -52,7 +137,9 @@ pub fn init(device_tree: *const platform.DeviceTree) void { count = 0; dropped = 0; display_device = null; - for (&claimed) |*c| c.* = null; + for (claimed) |*c| c.* = null; + for (registrar) |*r| r.* = null; + for (giver) |*g| g.* = null; walk(device_tree.root, device_abi.no_parent); } @@ -64,7 +151,7 @@ pub fn init(device_tree: *const platform.DeviceTree) void { /// table is full. Idempotent-ish: only ever call once per boot. pub fn seedDisplay(base: u64, width: u32, height: u32, pitch: u32, format: u32, refresh_hz: u32) ?u64 { if (base == 0 or width == 0 or height == 0) return null; // headless - if (count >= maximum_devices) { + if (!reserve()) { dropped += 1; return null; } @@ -109,7 +196,7 @@ fn walk(node: *platform.Device, parent_id: u64) void { } fn record(node: *platform.Device, parent_id: u64) u64 { - if (count >= maximum_devices) { + if (!reserve()) { dropped += 1; return device_abi.no_parent; // children of a dropped node become roots, not orphans } @@ -164,9 +251,32 @@ pub fn enumerateFrom(start: usize, out: []device_abi.DeviceDescriptor) usize { pub fn claim(id: u64, owner: u32) ClaimError!void { if (id >= count) return error.NoSuchDevice; if (claimed[@intCast(id)] != null) return error.AlreadyClaimed; + // **Delegated hardware may be handed on, never taken.** + // + // Belt and braces, and worth being honest about: with the loan rule above this is + // **currently unreachable**. A device that was given to someone is held, so it is + // refused as `AlreadyClaimed` before reaching here; and when the holder dies the + // device goes back to its lender (or, if the lender is gone, has its giver cleared + // with its claim), so there is no state where a device is unheld *and* still on + // loan. The window a stranger could have used simply stops existing. + // + // It stays because it is one comparison and it fails closed: any future path that + // frees a device without clearing its giver would otherwise hand delegated + // hardware to whoever asked first, which is exactly the hole this run closed. + // + // Note it leaves the loader's framebuffer alone without naming it: nobody delegates + // the framebuffer, so it has no giver, so the display service claims it as always. + if (giver[@intCast(id)] != null) return error.NotYours; claimed[@intCast(id)] = owner; } +/// The task that gave device `id` away, or null if nobody ever did. A device with a +/// giver is delegated hardware: it may be handed on, never taken. +pub fn giverOf(id: u64) ?u32 { + if (id >= count) return null; + return giver[@intCast(id)]; +} + /// The task that owns device `id`, or null. pub fn ownerOf(id: u64) ?u32 { if (id >= count) return null; @@ -178,11 +288,41 @@ pub fn ownerOf(id: u64) ?u32 { /// hardware again (docs/process-lifecycle.md iron rule 1: cleanup is the kernel's /// job). The devices stay in the table — they describe hardware, which did not go /// away — only their ownership clears. +/// Whether a task is still alive, injected by the process layer (which owns the task +/// table) the same way the scheduler's other hooks are. Null means "assume not", so a +/// kernel built without it clears claims rather than handing them to a ghost. +pub var task_alive_hook: ?*const fn (u32) bool = null; + +fn alive(task: u32) bool { + const hook = task_alive_hook orelse return false; + return hook(task); +} + pub fn releaseAllOwnedBy(owner: u32) void { - for (claimed[0..count]) |*slot| { - if (slot.*) |o| { - if (o == owner) slot.* = null; + for (claimed[0..count], 0..) |*slot, id| { + const holder = slot.* orelse continue; + if (holder != owner) continue; + + // **A grant is a loan.** A device this task was *given* goes back to whoever + // lent it, not to nobody — so the device manager gets its hardware back the + // instant a driver dies, and hands it to the replacement. + // + // Without this the kernel released the claim to no one and the manager + // re-claimed first-come, so every driver restart reopened the window this + // rule closes. And once `claim` refuses a device that has a giver, releasing + // to nobody would strand it: no one could ever take it again. + // + // A dead lender is no lender: clear the claim and the giver together, so the + // device is genuinely free rather than owed to a ghost. + if (giver[id]) |lender| { + if (alive(lender)) { + slot.* = lender; + giver[id] = null; // returned; it is the lender's own again, not on loan + continue; + } + giver[id] = null; } + slot.* = null; } } @@ -284,7 +424,7 @@ pub const RegisterError = error{ NoSuchParent, // no device with that id NotYourParent, // that device exists but this task has not claimed it TooManyResources, // the descriptor declares more resources than one device may hold - TooManyChildren, // this parent is at maximum_children_per_parent + TooManyChildren, // the caller is at its per-registrar allowance NotContained, // a child resource escapes its parent's window }; @@ -304,8 +444,46 @@ pub fn errnoOf(e: RegisterError) i64 { pub const ClaimError = error{ NoSuchDevice, // no device with that id AlreadyClaimed, // a live task already owns it + NotYours, // delegated hardware: it has a giver, so it must be handed on, not taken }; +/// Why a `transfer` was refused. +pub const TransferError = error{ + NoSuchDevice, // no device with that id + NotHeld, // the caller does not hold it — you may only give away what you have +}; + +/// The errno a refused `transfer` returns to ring 3. (`ESRCH` — no such recipient — is +/// raised by the caller in system/kernel/process.zig, which is what can see the task +/// table.) +pub fn transferErrnoOf(e: TransferError) i64 { + return switch (e) { + error.NoSuchDevice => abi.ENODEV, + error.NotHeld => abi.EPERM, + }; +} + +/// Move device `id` from `from` to `to`. **A move, not a copy**: a claim is exclusive +/// (driver-model.md, invariant 1), so the giver stops holding it the moment the +/// receiver starts. +/// +/// This is the mechanism behind delegation — the device manager claims what firmware +/// discovery seeded and passes each device to the driver it matched, which replaces +/// first-come-first-served `device_claim` with policy +/// (docs/device-driver-development/device-manager.md). The kernel checks only that the +/// caller holds the device: *you may give away what you have*. It knows nothing about +/// which task is the manager, and needs to know nothing. +/// +/// Note this is deliberately NOT the M13 capability-passing path, which shares a handle +/// refcounted — a copy. Exclusivity cannot be expressed that way. +pub fn transfer(id: u64, from: u32, to: u32) TransferError!void { + if (id >= count) return error.NoSuchDevice; + const holder = claimed[@intCast(id)] orelse return error.NotHeld; + if (holder != from) return error.NotHeld; + claimed[@intCast(id)] = to; + giver[@intCast(id)] = from; +} + /// The errno a refused `claim` returns to ring 3. (`ECONFINE` — the claim stood but /// the IOMMU would not confine the device — is raised by the caller in /// system/kernel/process.zig, which is what rolls the claim back.) @@ -313,6 +491,7 @@ pub fn claimErrnoOf(e: ClaimError) i64 { return switch (e) { error.NoSuchDevice => abi.ENODEV, error.AlreadyClaimed => abi.EBUSY, + error.NotYours => abi.EPERM, }; } @@ -339,14 +518,6 @@ fn existingChild(parent_id: u64, descriptor: *const device_abi.DeviceDescriptor) return null; } -/// Number of devices currently recorded with `parent_id` as their parent. -fn childCount(parent_id: u64) usize { - var n: usize = 0; - for (devices[0..count]) |d| { - if (d.parent == parent_id) n += 1; - } - return n; -} /// Publish `descriptor` as a child of `parent_id`, on behalf of `owner`. Returns the new /// device id. The child is left **unclaimed**, so another process (a class driver) @@ -377,8 +548,11 @@ pub fn register(parent_id: u64, owner: u32, descriptor: *const device_abi.Device // below). if (existingChild(parent_id, descriptor)) |existing_id| return existing_id; - if (childCount(parent_id) >= maximum_children_per_parent) return error.TooManyChildren; - if (count >= maximum_devices) return error.NoSpace; + // The allowance is charged to whoever is registering, so a driver in a loop + // exhausts its own and every other driver carries on. There is no machine-wide + // ceiling any more: the table grows. + if (registeredBy(owner) >= maximum_devices_per_registrar) return error.TooManyChildren; + if (!reserve()) return error.NoSpace; const parent = &devices[@intCast(parent_id)]; for (0..@intCast(descriptor.resource_count)) |i| { @@ -401,6 +575,7 @@ pub fn register(parent_id: u64, owner: u32, descriptor: *const device_abi.Device for (0..@intCast(descriptor.resource_count)) |i| d.resources[i] = descriptor.resources[i]; devices[count] = d; + registrar[count] = owner; count += 1; return d.id; } diff --git a/system/kernel/iommu.zig b/system/kernel/iommu.zig index 824a95a..029ac19 100644 --- a/system/kernel/iommu.zig +++ b/system/kernel/iommu.zig @@ -34,12 +34,27 @@ const platform = @import("platform"); const architecture = @import("architecture"); const devices_broker = @import("devices-broker.zig"); const log = @import("log.zig"); +const heap = @import("heap.zig"); const page_size: u64 = abi.page_size; const page_mask: u64 = page_size - 1; const huge_page_size: u64 = 2 * 1024 * 1024; -/// One domain per claimed PCI function. 64 mirrors devices-broker's device cap. +/// The IOMMU's own translation-domain pool — one per claimed DMA-capable device. +/// +/// No longer coupled to the device count. It was, by a comment and then by a comptime +/// assert, only because `confined` (one slot per device id) was sized by this same +/// constant; those are two unrelated quantities and making the device table dynamic +/// separated them. This one is genuinely the hardware's: both VT-d and AMD-Vi report +/// how many domains they support in a capability register, so the honest fix is to read +/// it rather than choose 64 — bounds-track-plan.md phase 4. +/// +/// bound: IOMMU translation domains the kernel can hold at once +/// decided-by: hardware +/// protects: the statically sized domain pool +/// at-limit: refuse — ECONFINE; the claim is rolled back and the device is not driven, +/// because a claim that cannot be confined must not stand +/// observed-by: the claiming driver's own line naming ECONFINE pub const maximum_domains = 64; pub const invalid_domain: u16 = 0xFFFF; @@ -94,18 +109,30 @@ pub fn init() void { /// Per-claimed-device record: its private domain, so a driver's death tears down /// exactly the domains it held. const Confined = struct { active: bool = false, owner: u32 = 0, bdf: u16 = 0, domain: u16 = invalid_domain }; -var confined: [maximum_domains]Confined = .{Confined{}} ** maximum_domains; -// `confined` is indexed by **device id**, so it must cover every id the broker can -// mint. These two numbers agreed only by a sentence in a comment above -// `maximum_domains` — and when they disagreed, `confineDevice` returned success for -// the ids it had no room for, leaving those devices unconfined DMA masters. Coupled -// bounds agree in code, not in prose (docs/os-development/bounds.md). -comptime { - if (maximum_domains < devices_broker.maximum_devices) - @compileError("iommu.confined is indexed by device id but is smaller than the " ++ - "broker's device table: ids past its end cannot be confined, and so cannot " ++ - "be claimed at all"); +/// Indexed by **device id**, so it must cover every id the broker can mint — and the +/// broker's table has no ceiling any more, so neither can this. It grows on demand. +/// +/// This used to be `[maximum_domains]`, sized by the *domain* constant purely because +/// device ids happened to stop at 64 as well. Two unrelated quantities sharing one +/// number: `domains` below is the IOMMU's own translation-domain pool, which the +/// hardware bounds and reports, while this is one slot per device the machine has. +/// A comptime assert held them together while both were fixed; making the device table +/// dynamic is what forced them apart, which is the assert having done its job. +var confined: []Confined = &.{}; + +/// Grow `confined` to cover `device_id`. False if the heap cannot — and the caller +/// treats that as a refusal to confine, never as permission. +fn reserveConfined(device_id: u64) bool { + if (device_id < confined.len) return true; + if (device_id >= std.math.maxInt(usize) / 2) return false; // absurd id; refuse rather than size to it + var wanted: usize = if (confined.len == 0) 64 else confined.len; + while (wanted <= device_id) wanted *= 2; + const grown = heap.allocator().realloc(confined, wanted) catch return false; + const previous = confined.len; + confined = grown; + for (confined[previous..]) |*record| record.* = .{}; + return true; } /// Place a just-claimed PCI function under IOMMU translation on behalf of `owner`: give @@ -126,7 +153,7 @@ pub fn confineDevice(device_id: u64, bdf: u16, owner: u32) bool { // at `confined.len`; moving the inventory out of the kernel and taking the domain // count from the hardware both change that, and either would have made a silent // unconfined DMA master out of every device past the 64th. - if (device_id >= confined.len) return false; + if (!reserveConfined(device_id)) return false; const domain = domainCreate(owner, bdf) orelse return false; // Firmware reserved region for this device, if any (real hardware; QEMU has none). @@ -168,7 +195,7 @@ pub fn unmapForDevice(device_id: u64, physical: u64, len: u64) void { /// own freshly-`dma_alloc`'d buffer into the devices it drives. pub fn mapRegionForOwner(owner: u32, physical: u64, len: u64) void { if (!active) return; - for (&confined) |*c| { + for (confined) |*c| { if (c.active and c.owner == owner) _ = map(c.domain, physical, len); } } @@ -179,7 +206,7 @@ pub fn mapRegionForOwner(owner: u32, physical: u64, len: u64) void { /// domain other than its owner's. pub fn unmapRegionEverywhere(physical: u64, len: u64) void { if (!active) return; - for (&confined) |*c| { + for (confined) |*c| { if (c.active) unmap(c.domain, physical, len); } } @@ -187,9 +214,40 @@ pub fn unmapRegionEverywhere(physical: u64, len: u64) void { /// A driver died or released its devices: tear down every domain it held (detach the /// device, free the tables) so their DMA is blocked again and a restarted driver /// re-claims cleanly. Runs BEFORE the broker claims and the DMA frames are released. +/// Re-point a device's existing confinement at a new owner, keeping its domain and +/// its attachment intact. +/// +/// Delegation needs this: the device manager claims a device (which confines it, with +/// the manager as owner) and then transfers it to the driver. Without moving the +/// confinement record too, the domain stays the manager's — so the driver's DMA +/// buffers are never bound into it, its rings are invisible to the device, and every +/// transfer faults. Worse, a manager death would then tear down a domain a live driver +/// is using, and a driver death would leave one behind. +/// +/// The domain is *not* rebuilt: the device stays attached throughout, so there is no +/// window in which it is translating through nothing. +pub fn reassign(device_id: u64, owner: u32) void { + if (!active) return; + if (device_id >= confined.len) return; + const record = &confined[@intCast(device_id)]; + if (!record.active) return; + record.owner = owner; +} + +/// The task a device's confinement is recorded against, or null if it has none. The +/// confinement's owner decides whose death tears the domain down, so a delegation that +/// moved the device but not this record would leave a live driver's domain destroyed by +/// its manager's exit — which is why `reassign` exists and why the suite asserts it. +pub fn confinementOwner(device_id: u64) ?u32 { + if (!active) return null; + if (device_id >= confined.len) return null; + const record = confined[@intCast(device_id)]; + return if (record.active) record.owner else null; +} + pub fn releaseAllOwnedBy(owner: u32) void { if (!active) return; - for (&confined) |*c| { + for (confined) |*c| { if (c.active and c.owner == owner) { detachDevice(c.bdf); domainDestroy(c.domain); diff --git a/system/kernel/process.zig b/system/kernel/process.zig index 254abf5..e167312 100644 --- a/system/kernel/process.zig +++ b/system/kernel/process.zig @@ -190,6 +190,15 @@ pub fn init() void { scheduler.timer_tick_hook = timerSweepLocked; scheduler.group_exit_hook = groupExitLocked; scheduler.space_mapping_release_hook = dropSpaceMappingHook; + // The broker owns devices; the process layer owns the task table. It asks whether a + // lender is still alive before handing a dead driver's device back to it. + devices_broker.task_alive_hook = taskAliveLocked; +} + +/// Whether `id` names a live task. The broker calls this through its hook when +/// deciding if a dead holder's device can go back to the task that lent it. +fn taskAliveLocked(id: u32) bool { + return scheduler.taskByIdLocked(id) != null; } /// Return -1 (as an unsigned bit pattern) in the system_call result register. @@ -249,6 +258,7 @@ fn system_call(state: *architecture.CpuState) void { .irq_bind => systemIrqBind(state), .irq_ack => systemIrqAck(state), .device_register => systemDeviceRegister(state), + .device_transfer => systemDeviceTransfer(state), .system_spawn => systemSpawn(state), .dma_alloc => systemDmaAlloc(state), .dma_free => systemDmaFree(state), @@ -440,6 +450,41 @@ fn systemDeviceClaim(state: *architecture.CpuState) void { architecture.setSystemCallResult(state, 0); } +/// device_transfer(device_id, task_id) -> 0/-errno: give a device you hold to another +/// task. The mechanism behind delegation — the device manager claims what discovery +/// seeded and hands each device to the driver it matched, so assignment stops being +/// first-come-first-served (docs/os-development/device-authority.md). +/// +/// The kernel's whole rule is *you may give away what you hold*. It has no notion of +/// which task is the device manager, and deliberately gains none: a binary name in the +/// kernel is not something that cannot safely live in user space. +fn systemDeviceTransfer(state: *architecture.CpuState) void { + const device_id = architecture.systemCallArg(state, 0); + const task_id: u32 = @truncate(architecture.systemCallArg(state, 1)); + const flags = sync.enter(); + defer sync.leave(flags); + + // The recipient must exist, or the device would be moved to nobody and become + // unreachable for the rest of the boot — no path un-holds a device but task death. + if (scheduler.taskByIdLocked(task_id) == null) return failErr(state, ipc.ESRCH); + + devices_broker.transfer(device_id, scheduler.current().id, task_id) catch |e| + return failErr(state, devices_broker.transferErrnoOf(e)); + + // The device's IOMMU confinement moves with it. The giver confined it when it + // claimed, so the domain exists and the device stays attached — but the record + // still names the giver as owner, which would leave the receiver's DMA buffers + // unbound (every transfer faulting), a giver's death tearing down a domain the + // receiver is using, and the receiver's death leaving one behind. + if (devices_broker.pciAddressOf(device_id)) |_| { + iommu.reassign(device_id, task_id); + // Bind whatever the receiver has already allocated — the same courtesy the + // claim path does for a driver that dma_alloc'd its rings before claiming. + dmaBindOwnerRegionsInto(task_id, device_id); + } + architecture.setSystemCallResult(state, 0); +} + /// mmio_map(device_id, resource_index) -> virtual_address: map a claimed device's MMIO window into /// this address space (strong-uncacheable) and return the register base address. /// The claim is the capability — a process can only map hardware it owns. @@ -981,6 +1026,13 @@ fn systemSpawn(state: *architecture.CpuState) void { const arguments_ptr = architecture.systemCallArg(state, 2); const arguments_len = architecture.systemCallArg(state, 3); const exit_handle = architecture.systemCallArg(state, 4); + // A device the caller holds and gives to the child. Fused into the spawn rather + // than transferred after it, because a separate transfer leaves a window in which + // the child is running and does not yet hold its device — a race that would close + // on one machine and open on another, which is the failure shape this whole track + // exists to remove (docs/bounds-track-plan.md, "the grant rides system_spawn"). + // Here the child cannot observe the gap: it does not exist until it holds it. + const device_to_give = architecture.systemCallArg(state, 5); const t = scheduler.current(); if (len == 0 or len > scheduler.maximum_task_name or ptr >= user_half_end or ptr + len > user_half_end) return fail(state); if (arguments_len > maximum_argument_bytes) return fail(state); @@ -1025,7 +1077,26 @@ fn systemSpawn(state: *architecture.CpuState) void { } } + // Refuse before creating anything if the device is not the caller's to give — a + // spawn that half-succeeds would leave a child running without the hardware it was + // spawned for, which is worse than not spawning it. + if (device_to_give != abi.no_device and devices_broker.ownerOf(device_to_give) != t.id) + return failErr(state, ipc.EPERM); + const child = spawnProcessSupervised(item.blob, 4, argv[0..argc], t.id, exit_endpoint) catch return fail(state); + + if (device_to_give != abi.no_device) { + devices_broker.transfer(device_to_give, t.id, child) catch { + // Cannot happen — ownership was checked above and the lock has not been + // dropped — but a spawned child holding nothing is not something to guess + // about, so say so rather than leave it silent. + log.print("/system/kernel: WARNING spawn gave device {d} to task {d} and the transfer failed\n", .{ device_to_give, child }); + }; + if (devices_broker.pciAddressOf(device_to_give)) |_| { + iommu.reassign(device_to_give, child); + dmaBindOwnerRegionsInto(child, device_to_give); + } + } architecture.setSystemCallResult(state, child); } diff --git a/system/kernel/tests.zig b/system/kernel/tests.zig index 328390d..7d03c72 100644 --- a/system/kernel/tests.zig +++ b/system/kernel/tests.zig @@ -261,6 +261,10 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void { containmentTest(); } else if (eql(case, "apertures")) { apertureTest(); + } else if (eql(case, "device-transfer")) { + deviceTransferTest(boot_information); + } else if (eql(case, "device-authority")) { + deviceAuthorityTest(boot_information); } else if (eql(case, "device-manager")) { deviceManagerTest(boot_information); } else if (eql(case, "protocol-registry")) { @@ -1443,6 +1447,52 @@ fn iommuTest() void { } else { check("scratch domain allocated", false); } + // Delegation moves a device between tasks, and the IOMMU confinement must move + // with it. When it did not, the driver's DMA rings were never bound into the + // device's domain and every transfer faulted — but two worse consequences were + // latent and invisible to those tests: the domain still named the *giver*, so the + // giver's death would tear down a domain a live driver was using, and the + // receiver's death would leave one behind. Asserted directly here rather than + // inferred from the USB cases going green. + // This case runs no pci-bus, so there are no PCI functions to confine — and a + // silently skipped assertion is worse than none. Register one the way pci-bus does: + // a child of the host bridge whose resource 0 is a 4 KiB config window inside the + // bridge's ECAM, which is what `pciAddressOf` derives a requester id from. + var iommu_table: [64]device_abi.DeviceDescriptor = undefined; + const iommu_total = devices_broker.enumerate(&iommu_table); + const bridge: ?device_abi.DeviceDescriptor = for (iommu_table[0..@min(iommu_total, iommu_table.len)]) |d| { + if (d.class == @intFromEnum(device_abi.DeviceClass.pci_host_bridge) and d.resource_count >= 2) break d; + } else null; + check("the kernel seeded a PCI host bridge to parent a function under", bridge != null); + + const subject: ?u64 = if (bridge) |b| blk: { + const me_bridge = scheduler.currentId(); + if (!claimOk(b.id, me_bridge)) break :blk null; + var function = std.mem.zeroes(device_abi.DeviceDescriptor); + function.class = @intFromEnum(device_abi.DeviceClass.pci_device); + function.pci_class = device_abi.no_pci_class; + function.resource_count = 1; + function.resources[0] = .{ + .kind = @intFromEnum(device_abi.ResourceKind.memory), + .start = b.resources[0].start, // the first config slot in the ECAM window + .len = 4096, + }; + break :blk devices_broker.register(b.id, me_bridge, &function) catch null; + } else null; + check("a PCI function exists to confine", subject != null); + + if (subject) |device_id| { + check("a device starts unconfined", iommu.confinementOwner(device_id) == null); + const me = scheduler.currentId(); + check("confining records the owner", iommu.confineDevice(device_id, devices_broker.pciAddressOf(device_id).?, me)); + check("the confinement names the confiner", iommu.confinementOwner(device_id) == me); + iommu.reassign(device_id, me + 1000); + check("reassign moves the confinement to the new holder", iommu.confinementOwner(device_id) == me + 1000); + check("and the previous holder no longer owns it", iommu.confinementOwner(device_id) != me); + iommu.releaseAllOwnedBy(me + 1000); + check("the new holder's death tears the domain down", iommu.confinementOwner(device_id) == null); + } + log("DANOS-IOMMU: enabled base=0x{x} domains active\n", .{pinfo.iommu_base}); result(); } @@ -3048,7 +3098,17 @@ fn acpiParseTest(boot_information: *const BootInformation) void { while (i < rd.count) : (i += 1) { const item = rd.entry(i) orelse continue; if (!eql(initial_ramdisk.basename(item.name), "discovery")) continue; - _ = process.spawnProcessSupervised(item.blob, 4, &.{ item.name, "1" }, scheduler.currentId(), null) catch 0; + // Hand it the acpi-tables node the way the manager would: claim it here, then + // move it to the child. Discovery no longer claims for itself. + var scratch: [64]device_abi.DeviceDescriptor = undefined; + const seen_devices = devices_broker.enumerate(&scratch); + const tables: ?u64 = for (scratch[0..@min(seen_devices, scratch.len)]) |d| { + if (d.class == @intFromEnum(device_abi.DeviceClass.acpi_tables)) break d.id; + } else null; + const me_parse = scheduler.currentId(); + if (tables) |node| _ = claimOk(node, me_parse); + const child = process.spawnProcessSupervised(item.blob, 4, &.{ item.name, "floor:1" }, me_parse, null) catch 0; + if (tables) |node| devices_broker.transfer(node, me_parse, child) catch {}; spawned = true; break; } @@ -3987,39 +4047,130 @@ fn containmentTest() void { check("re-registering an identical child returns the same id", again != 0 and again == good); check("re-registering grew nothing", devices_broker.enumerate(&buffer) == before + 1); - // Fill the parent to its child cap with distinct children (same window, different - // identity — the match is on identity, so each is a new device). + // **There is no per-parent cap.** `maximum_children_per_parent = 16` is gone: it + // was written to stop a driver looping `device_register` and exhausting a shared + // table, and there is no shared table to exhaust — the table grows, and each + // registrar has its own allowance, so a runaway costs only itself. The cap never + // bounded a determined caller anyway (16 per parent, but nothing stopped it + // claiming more parents); what it reliably did was refuse a real PCI bus with more + // than 16 functions, which is how an AMD Ryzen booted with no USB and no storage. + // + // 64 children under one parent — four times the old ceiling. var filled: u32 = 0; - var capped = false; + var registered_children: u32 = 0; while (filled < 64) : (filled += 1) { var name: [4]u8 = .{ 'k', 0, 0, 0 }; name[1] = '0' + @as(u8, @intCast(filled / 10)); name[2] = '0' + @as(u8, @intCast(filled % 10)); var extra = childDescriptor(name[0..3], parent_window.start, 0x20); - _ = devices_broker.register(parent_id, me, &extra) catch |err| { - capped = err == error.TooManyChildren; - break; - }; + if (devices_broker.register(parent_id, me, &extra)) |_| { + registered_children += 1; + } else |_| break; } - check("the parent reaches its child cap (TooManyChildren)", capped); + check("one parent takes far more children than the old cap allowed", registered_children == 64); - // The regression this ordering exists for: **a re-registration consumes no slot, - // so a full parent must not refuse one.** A crashed bus driver is restarted by its - // supervisor and re-registers everything it rediscovers; when the cap was checked - // before the identity match, the restart was refused its own devices and the - // machine degraded a little more on every crash. + // Idempotency still holds, and still matters: a crashed bus driver is restarted and + // re-registers everything it rediscovers, which must return the ids it had before + // rather than duplicate them. const readmitted = devices_broker.register(parent_id, me, &fits) catch 0; - check("a full parent still re-admits an identical child", readmitted != 0 and readmitted == good); + check("re-registering an identical child still returns its id", readmitted != 0 and readmitted == good); - // ...and the cap is genuinely still in force for anything new. - var novel = childDescriptor("knew", parent_window.start, 0x20); - const still_capped = if (devices_broker.register(parent_id, me, &novel)) |_| false else |err| err == error.TooManyChildren; - check("a full parent still refuses a new child", still_capped); + // The table itself has no ceiling: it grows. The old `maximum_devices = 64` was a + // guess about someone else's computer, and one driver's enumeration starved every + // other — which is how a Ryzen booted with no USB and no storage. What bounds a + // runaway now is an allowance charged to the registrar, so the damage stays with + // whoever caused it (docs/os-development/device-authority.md). + // The table grows: it starts at 8 entries and this boot holds well past that, so + // the growth path runs every time rather than lying dormant until someone else's + // larger machine finds it — which is how the old ceiling stayed invisible. + const held = devices_broker.enumerate(&buffer); + check("the table grew beyond its initial block", held > 8); result(); } /// A minimal child descriptor with one memory resource, for the containment test. +/// Delegation's mechanism. A claim is exclusive (driver-model.md, invariant 1), so +/// handing a device on is a **move**: the giver stops holding it the instant the +/// receiver starts. That is why this is not the M13 capability path, which shares a +/// handle refcounted. +/// +/// The rule the kernel enforces is the whole of it: *you may give away what you hold*. +/// It has no idea which task is the device manager and needs none +/// (docs/os-development/device-authority.md). +fn deviceTransferTest(boot_information: *const boot_handoff.BootInformation) void { + var buffer: [8]device_abi.DeviceDescriptor = undefined; + check("the device tree is seeded", devices_broker.enumerate(&buffer) >= 2); + + const image = bundledInit(boot_information) orelse { + check("initial_ramdisk carries /system/services/init", false); + result(); + return; + }; + const me = scheduler.currentId(); + const endpoint = ipcsync.createIpcEndpoint() orelse { + check("exit endpoint allocated", false); + result(); + return; + }; + const child = process.spawnProcessSupervised(image, 4, &.{"/system/services/init"}, me, endpoint) catch 0; + check("supervised child spawned", child != 0); + + // You may only give away what you hold — so an unheld device cannot be moved at all, + // which is what stops a transfer being a back door around claiming. + const unheld = if (devices_broker.transfer(0, me, child)) |_| false else |e| e == error.NotHeld; + check("an unheld device cannot be transferred", unheld); + + // A second device this test never hands over, so the "no giver" half below is a + // real observation rather than a restatement of the first. + const parentless: u64 = 1; + + check("claimed device 0", claimOk(0, me)); + + // The move itself. + const moved = if (devices_broker.transfer(0, me, child)) |_| true else |_| false; + check("the holder may transfer", moved); + const holder = devices_broker.ownerOf(0) orelse 0; + check("the receiver holds it", holder == child); + check("the giver does not", holder != me); + + // Having given it away, the giver cannot give it again. This is the assertion that + // makes it a move rather than a copy. + const again = if (devices_broker.transfer(0, me, child)) |_| false else |e| e == error.NotHeld; + check("a former holder cannot transfer again", again); + + // Nor may a stranger move a device it never held. + const stranger = if (devices_broker.transfer(0, 9999, me)) |_| false else |e| e == error.NotHeld; + check("a stranger cannot transfer another task's device", stranger); + + // The giver is recorded. One field, and the authority rule follows from it: a + // device that was *given* to someone is delegated hardware, so it may be handed on + // but never taken, and when its holder dies it returns to whoever lent it instead + // of becoming free for anyone. Nothing has a giver until it is handed over — + // which is why the loader's framebuffer needs no exemption from either rule. + check("the giver is recorded on a transfer", devices_broker.giverOf(0) == me); + check("a device nobody handed over has no giver", devices_broker.giverOf(parentless) == null); + + const absent = if (devices_broker.transfer(9999, me, child)) |_| false else |e| e == error.NoSuchDevice; + check("a device that does not exist is refused", absent); + + // **A grant is a loan.** The child holds device 0 and `me` lent it, so the child's + // death must hand it back rather than release it to nobody. That is what lets the + // device manager re-delegate to a restarted driver — and, once `claim` refuses a + // device that has a giver, it is the only thing that stops a dead driver's hardware + // being stranded forever. + check("the child still holds the lent device", devices_broker.ownerOf(0) == child); + devices_broker.releaseAllOwnedBy(child); + check("the borrower's death returns the device to its lender", devices_broker.ownerOf(0) == me); + check("and it is no longer on loan", devices_broker.giverOf(0) == null); + + // A device with no lender still simply frees on death, as it always did. + check("claimed a second device with no lender", claimOk(parentless, me)); + devices_broker.releaseAllOwnedBy(me); + check("a device nobody lent is released outright", devices_broker.ownerOf(parentless) == null); + result(); +} + /// PCI host-bridge apertures are derived from the *holes* in the firmware memory map, /// and a registered BAR must fall inside one. So the invariant is not "we find the /// holes" but "an aperture never covers memory the firmware described" — an aperture @@ -4160,6 +4311,58 @@ fn protocolRegistryTest(boot_information: *const BootInformation) void { /// /// The fixture's `protocol-denied: ok` is the marker; each step prints its own /// line, which the harness's ordered regex reads. +/// The attacker the device suite never had. The audit's finding was that a fully +/// green suite had missed six real defects because it *contains no attacker* — every +/// device case asserts a driver handed its hardware can drive it, and none asks what a +/// process handed **nothing** can do. +/// +/// The fixture is spawned with no device and asserts what it therefore cannot do. It +/// runs without the device manager on purpose: nothing here needs a driver, and a boot +/// with fewer moving parts makes the refusals unambiguous. +fn deviceAuthorityTest(boot_information: *const BootInformation) void { + log("DANOS-TEST-BEGIN: device-authority\n", .{}); + if (boot_information.initial_ramdisk_len == 0) { + check("bootloader handed over an initial_ramdisk", false); + result(); + return; + } + const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len]; + const rd = initial_ramdisk.Reader.init(image) orelse { + check("initial_ramdisk image is valid", false); + result(); + return; + }; + + process.setInitialRamdisk(image); + // The manager must be up: it is what holds the seeded hardware, and without it + // every device would be lying around unheld and the last assertion would have + // nothing to observe — a test that cannot fail. + check("registry (init) spawned", spawnRegistry(rd)); + check("device-manager spawned", spawnNamed(rd, "device-manager")); + check("device-authority-test spawned", spawnNamedWithArg(rd, "device-authority-test", "run")); + + // The VERDICT prefix matters: the fixture prints one "device-authority: ok " + // line per assertion, so a marker of "device-authority: ok" matches the FIRST + // passing assertion and this loop exits before any later failure is printed — the + // case then passes with failures in it, which it did until this was caught. + const pass_marker = "device-authority: VERDICT ok"; + const fail_marker = "device-authority: VERDICT FAILED"; + scheduler.setPriority(1); + const deadline = architecture.millis() + 20000; + var saw_pass = false; + var saw_fail = false; + while (architecture.millis() < deadline and !saw_pass and !saw_fail) { + if (bufferHas(pass_marker)) saw_pass = true; + if (bufferHas(fail_marker)) saw_fail = true; + scheduler.yield(); + } + scheduler.setPriority(4); + + check("no authority assertion failed", !saw_fail); + check("the attacker completed every assertion", saw_pass); + result(); +} + fn protocolDeniedTest(boot_information: *const BootInformation) void { log("DANOS-TEST-BEGIN: protocol-denied\n", .{}); if (boot_information.initial_ramdisk_len == 0) { diff --git a/system/parameters.zig b/system/parameters.zig index a055bb0..d51ebdd 100644 --- a/system/parameters.zig +++ b/system/parameters.zig @@ -17,8 +17,13 @@ /// arrays (discovery pool, scheduler state, per-core GDT/TSS). Generous headroom: /// those structs are small, and the *large* per-core resources (kernel and IST /// stacks) are allocated at bring-up for cores that actually come online, so this -/// ceiling is cheap. A machine with more logical CPUs has its surplus reported and -/// left parked (see acpi `cpusDropped`). +/// ceiling is cheap. +/// +/// bound: logical CPUs the kernel tracks +/// decided-by: hardware +/// protects: the per-CPU bookkeeping arrays, which are sized at compile time +/// at-limit: degrade — the surplus cores are left parked, never brought online +/// observed-by: platform.cpusDropped() -> the WARNING at kernel.zig:281 pub const maximum_cpus = 128; /// Maximum tasks (kernel threads) alive at once — the static task-table size. Each @@ -29,6 +34,17 @@ pub const maximum_cpus = 128; /// for the USB stack: the xHCI bus driver spawns a supervised class-driver instance /// per matched interface (keyboard, mouse, mass storage), on top of the FAT and /// block servers and the growing ramdisk bundle. +/// +/// The history above is the argument against this number: it has been raised twice, +/// each time by a machine or a bundle that outgrew it, which is the pattern the +/// bounds rule exists to stop. It is `ours` only because the task table is static; +/// how many drivers a machine needs is decided by how much hardware it has. +/// +/// bound: kernel threads alive at once — the static task-table size +/// decided-by: hardware +/// protects: the statically allocated task table +/// at-limit: refuse — spawn fails; a supervised driver is never started +/// observed-by: the spawning supervisor's own log line; see docs/bounds-track-plan.md pub const maximum_tasks = 48; /// Each task's kernel stack (also each AP's bring-up stack), in bytes. diff --git a/system/services/acpi/acpi.zig b/system/services/acpi/acpi.zig index 9c0545b..de5d01f 100644 --- a/system/services/acpi/acpi.zig +++ b/system/services/acpi/acpi.zig @@ -104,11 +104,19 @@ fn findTablesNode(buffer: []device.DeviceDescriptor) ?device.DeviceDescriptor { } pub fn main(init: process.Init) void { - // When the acpi-parse scenario spawns this directly, argv[1] is a device-count - // *floor* to self-verify against. The kernel no longer parses AML, so there is - // no exact count to match — proving the ring-3 parse found at least a floor of - // devices is the check. Deterministic, no log-scraping. - const floor: ?usize = if (init.arguments.get(1)) |a| (std.fmt.parseInt(usize, a, 10) catch null) else null; + // When the acpi-parse scenario spawns this directly, it passes `floor:N` — a + // device-count floor to self-verify against. The kernel no longer parses AML, so + // there is no exact count to match; proving the ring-3 parse found at least N + // Device objects is the check. Deterministic, no log-scraping. + // + // The `floor:` prefix matters. argv[1] is the assigned device id for every driver + // the manager spawns, so a bare number here would be read as a floor — which is + // exactly what happened when discovery started being given its node: it saw + // argv[1] = "7", decided it was in self-verify mode, and never reported a device. + const floor: ?usize = if (init.arguments.get(1)) |a| blk: { + if (!std.mem.startsWith(u8, a, "floor:")) break :blk null; + break :blk std.fmt.parseInt(usize, a["floor:".len..], 10) catch null; + } else null; const buffer = memory.allocator().alloc(device.DeviceDescriptor, 64) catch { _ = logging.write("/system/services/acpi: out of memory\n"); @@ -118,11 +126,11 @@ pub fn main(init: process.Init) void { _ = logging.write("/system/services/acpi: no acpi-tables node to claim\n"); return; }; + // The node arrived with the spawn: the manager holds it and names it in the call + // that creates this process. Discovery was the last thing in the system that + // acquired hardware by naming it rather than being given it + // (docs/os-development/device-authority.md). node_id = node.id; - device.claim(node_id) catch |e| { - std.log.warn("unable to claim acpi-tables: {s}", .{@errorName(e)}); - return; - }; // Map the node's resources: the AML blobs (bytecode), the FADT (intact // "FACP" header — decision 3), the io_port grant, and the SCI irq. diff --git a/system/services/device-manager/device-manager.zig b/system/services/device-manager/device-manager.zig index 6b87d14..5c04e0c 100644 --- a/system/services/device-manager/device-manager.zig +++ b/system/services/device-manager/device-manager.zig @@ -220,6 +220,13 @@ fn childCountOf(reporter: u32) u32 { /// The driver entry a live process id belongs to. Zero is not a process id here: /// it is what `onDriverExit` writes back to retire an id it has already acted on, /// so a second notification for the same death matches nothing. +fn driverByName(name: []const u8) ?*Driver { + for (&drivers) |*driver| { + if (driver.used and std.mem.eql(u8, driver.name(), name)) return driver; + } + return null; +} + fn driverByProcess(process_id: u32) ?*Driver { if (process_id == 0) return null; for (&drivers) |*driver| { @@ -257,6 +264,22 @@ fn addDriver(name: []const u8, device_id: u64, speaks_protocol: bool) void { /// device id as argv[1] when it has one, the hello deadline armed when it /// speaks the protocol. fn spawnDriver(driver: *Driver) void { + // Take the device before the driver exists, so there is no window in which anyone + // else could claim it — which is the whole of what makes the handover authoritative + // rather than advisory. Re-claiming across a restart is expected to say + // AlreadyClaimed once the manager already holds it, and that is fine: it means the + // device never left our hands while the driver was dead. + if (driver.device_id != device_manager_protocol.no_device) { + device.claim(driver.device_id) catch |e| switch (e) { + error.AlreadyClaimed => {}, // ours already, from a previous spawn of this driver + else => { + std.log.warn("cannot hold device {d} for {s}: {s}", .{ driver.device_id, driver.name(), @errorName(e) }); + driver.state = .failed; + return; + }, + }; + } + var id_text: [20]u8 = undefined; var arguments: [1][]const u8 = undefined; var argument_count: usize = 0; @@ -264,7 +287,14 @@ fn spawnDriver(driver: *Driver) void { arguments[0] = std.fmt.bufPrint(&id_text, "{d}", .{driver.device_id}) catch return; argument_count = 1; } - const child = process.spawnSupervised(driver.name(), arguments[0..argument_count], manager_endpoint) orelse { + // The device rides the spawn, so the driver holds it before its first instruction. + // A transfer *after* spawning would leave a window in which the child is running + // without its hardware — closed on one machine, open on another + // (docs/bounds-track-plan.md, "the grant rides system_spawn"). + const give = driver.device_id; + if (give != device_manager_protocol.no_device) + std.log.info("delegated device {d} to {s}", .{ give, driver.name() }); + const child = process.spawnSupervisedWithDevice(driver.name(), arguments[0..argument_count], manager_endpoint, give) orelse { std.log.info("failed to spawn {s}", .{driver.name()}); driver.state = .failed; return; @@ -362,6 +392,40 @@ fn initialise(endpoint: ipc.Handle) bool { const total = device.enumerate(buffer); const count = @min(total, buffer.len); + // The node discovery needs: the kernel seeds it, so it is in this same snapshot + // and can be handed over like any other assignment. Discovery used to find and + // claim it itself — the last driver that acquired hardware by naming it rather + // than being given it (docs/os-development/device-authority.md). + var tables_node: u64 = device_manager_protocol.no_device; + for (buffer[0..count]) |descriptor| { + if (descriptor.class == @intFromEnum(device.DeviceClass.acpi_tables)) { + tables_node = descriptor.id; + break; + } + } + + // **Hold the firmware-discovered hardware, so none of it is left lying around.** + // A device nobody holds can be claimed by anyone, so every seeded device that + // carries mappable resources is taken here whether or not a driver wants it — the + // HPET most of all, which has an MMIO window and an IRQ and no user-space driver. + // Held by the manager it is inert; unheld it was there for the taking. + // + // Two deliberate exclusions: + // - the loader's framebuffer, which the compositor claims and which is not + // hardware anyone is delegated (the manager starts before display, so taking + // it here would break the boot screen); + // - anything with no resources, which grants nothing and so is not worth holding. + // + // This covers the boot snapshot only. A device *reported* later and matched to no + // driver stays claimable — the pci-cap and iommu-fault fixtures rely on exactly + // that to reach an unmatched NIC. Narrowing it further is a separate change with + // those fixtures in scope. + for (buffer[0..count]) |descriptor| { + if (descriptor.resource_count == 0) continue; + if (descriptor.class == @intFromEnum(device.DeviceClass.display)) continue; + device.claim(descriptor.id) catch continue; // already held, or not ours to take + } + var matched: usize = 0; for (buffer[0..count]) |descriptor| { if (descriptor.class == @intFromEnum(device.DeviceClass.pci_host_bridge)) { @@ -383,7 +447,7 @@ fn initialise(endpoint: ipc.Handle) bool { // under the neutral name "discovery", spawned once at startup. It finds and // claims the acpi-tables (or devicetree-blob) node itself. Not a per-device // match — it is the discoverer, not a driver bound to one device. - addDriver("discovery", device_manager_protocol.no_device, false); + addDriver("discovery", tables_node, false); if (test_restart_mode) { // The driver-restart scenario's fixture: claims device 0 (the tree @@ -424,6 +488,7 @@ fn onHello(_: void, invocation: Invocation(device_manager_protocol.Hello), _: An return -envelope.EPERM; }; driver.state = .running; + std.log.info("hello from {s} (device {d})", .{ driver.name(), invocation.target }); // Resilience drill (V6): once, kill the virtio-gpu driver a moment after it hellos, so // the normal restart policy respawns it — the compositor must survive and re-attach. @@ -461,9 +526,36 @@ fn onChildAdded(_: void, invocation: Invocation(device_manager_protocol.ChildAdd if (match.ambiguous) std.log.info("/system/configuration/devices.csv: multiple equally-specific rules match the device {s} reported; binding {s}", .{ driver.name(), match.driver }); if (id.bus == .acpi) { - // An hid-matched driver (ps2-bus) is a singleton that finds its - // own devices once spawned — spawn it once, no device assignment. - if (!alreadySupervised(match.driver)) addDriver(match.driver, device_manager_protocol.no_device, false); + // An hid-matched driver is a singleton over one piece of hardware + // described by several nodes: the 8042 is a single controller whose + // I/O ports live under the keyboard node (PNP0303) while the mouse + // is a second node (PNP0F13). Two processes would fight over the + // same 0x60/0x64 registers, so there is exactly one instance — and + // it needs *every* matching device, however many the machine has + // (some have none, some one port, some two). + // + // The first rides the spawn; the rest are transferred to the + // running instance. Late arrival is fine here because the order is + // natural: the instance needs the controller node immediately and + // reaches the mouse only after the 8042 handshakes and identify. + if (!alreadySupervised(match.driver)) { + addDriver(match.driver, device_id, false); + } else if (driverByName(match.driver)) |running| { + if (running.process_id != 0) { + device.claim(device_id) catch |e| switch (e) { + error.AlreadyClaimed => {}, + else => { + std.log.warn("cannot hold device {d} for {s}: {s}", .{ device_id, match.driver, @errorName(e) }); + return status; + }, + }; + device.transfer(device_id, running.process_id) catch |e| { + std.log.warn("could not give device {d} to {s}: {s}", .{ device_id, match.driver, @errorName(e) }); + return status; + }; + std.log.info("delegated device {d} to {s} (already running)", .{ device_id, match.driver }); + } + } } else { // A per-device driver: one instance, the registered id as argv[1]. if (!driverForDevice(device_id)) addDriver(match.driver, device_id, true); diff --git a/test/qemu_test.py b/test/qemu_test.py index 8bcccc5..1fcc6ad 100644 --- a/test/qemu_test.py +++ b/test/qemu_test.py @@ -625,11 +625,24 @@ CASES = [ # manager spawn the USB keyboard driver, which opens its device over the # transfer protocol, asks for boot protocol, subscribes to its interrupt # endpoint, and comes up — proof the class-driver <-> controller path works. + # + # It also pins the slot count to the hardware's answer. The backreference is the + # assertion: the driver must track exactly as many device slots as the controller + # reports in HCSPARAMS1.MaxSlots. It tracked a fixed 8 while QEMU's xHCI reports + # 64, so seven eighths of the controller was invisible and a device behind a hub + # past the eighth vanished without a log line. \1 fails the moment they diverge. {"name": "usb-hid", "smp": 4, "timeout": 150, # usb-kbd/usb-mouse ride the default boot xHCI bus (see qemu_args). - "expect": r"(?=[\s\S]*usb-hid-keyboard: ok)(?=[\s\S]*usb-hid-mouse: ok)", + # The controller must arrive by DELEGATION, not by claiming: the manager holds it + # and transfers it in the hello reply, so the match is authoritative rather than + # advisory (docs/os-development/device-authority.md). The delegation line must + # precede the hello, because the transfer completes before the reply lands. + "expect": r"(?=[\s\S]*device-manager: delegated device (\d+) to /system/drivers/usb-xhci-bus" + r"[\s\S]*usb-xhci-bus: controller device \1 registers at)" + r"(?=[\s\S]*usb-xhci-bus: controller running \((\d+) slots, tracking \2,)" + r"(?=[\s\S]*usb-hid-keyboard: ok)(?=[\s\S]*usb-hid-mouse: ok)", "fail": r"DANOS-TEST-RESULT: FAIL"}, # Keyboard echo: inject a known phrase via QMP send-key; the usb-hid-keyboard # driver decodes it and echoes each character to the log (the simple @@ -667,6 +680,28 @@ CASES = [ r"(?=.*hub slot \d+ port \d+ device:.*0x0627)" r"(?=.*usb-hid-keyboard: ok \(device 3)", "fail": r"DANOS-TEST-RESULT: FAIL"}, + # The driver must never truncate a configuration block. Its size is the DEVICE's + # choice (wTotalLength, a u16); the driver used to read the first 512 bytes into a + # fixed buffer and parse those, so interfaces past the cut did not exist while + # SET_CONFIGURATION still configured the device for all of them. + # + # The backreference is the assertion: declared length and bytes read must match. + # + # Honest limit: QEMU cannot produce a block over 512 bytes. Its boot keyboard, + # mouse and stick are 34-44, and the largest device on offer is usb-audio in + # multi-channel mode at 211 — which is why the suite never saw the original bug, + # and why it cannot now reproduce that exact trigger. What this case does catch is + # the class: any clamp below the attached device's block size fails it, verified + # by pinning the buffer to 128 and watching it go red. A real headset (500-900 + # bytes), UVC webcam (1-3 KB) or multifunction printer trips the 512 itself. + {"name": "usb-large-descriptor", + "build_case": "usb-hid", + "smp": 4, + "timeout": 150, + "qemu_extra": ["-device", "qemu-xhci,id=xhci2", + "-device", "usb-audio,bus=xhci2.0,port=1,multi=on"], + "expect": r"(?s)(?=.*usb-xhci-bus: config block (\d\d\d+) bytes, read \1)", + "fail": r"DANOS-TEST-RESULT: FAIL"}, # USB mass storage end to end: the boot usb-storage device (the FAT32 image, # which has a real 0x55AA boot sector) is enough — the manager spawns # usb-storage, which opens the device, runs the Bulk-Only / SCSI bring-up, @@ -745,8 +780,16 @@ CASES = [ {"name": "acpi-ps2", "smp": 4, "timeout": 150, + # The 8042 is ONE controller described by two ACPI nodes, so the single ps2-bus + # instance must receive BOTH. The keyboard node rides the spawn — it carries the + # 0x60/0x64 ports and is needed immediately — and the mouse node is transferred to + # the already-running instance, which is safe because it is not touched until after + # the controller handshakes and identify. Two processes cannot split this: they + # would fight over the same registers. "expect": r"discovery: device \d+\s+bus=acpi hid=PNP0303[\s\S]*" - r"device-manager: spawned \S*ps2-bus[\s\S]*" + r"device-manager: delegated device (\d+) to \S*ps2-bus[\s\S]*" + r"device-manager: spawned \S*ps2-bus for device \1[\s\S]*" + r"device-manager: delegated device \d+ to \S*ps2-bus \(already running\)[\s\S]*" r"ps2-bus: keyboard driver attached", "fail": r"DANOS-TEST-RESULT: FAIL"}, # M21.1: the SCI + power button. Boot the manager (which spawns the acpi @@ -819,10 +862,15 @@ CASES = [ {"name": "pci-scan", "smp": 4, "timeout": 60, - "expect": r"pci-bus: (\d+) functions found[\s\S]*" + # The bridge must arrive by DELEGATION, not by claiming — and again after the + # restart, which is what proves the manager re-takes the device when its driver + # dies and hands it to the replacement. \1 pins it to the same device both times. + "expect": r"device-manager: delegated device (\d+) to \S*pci-bus[\s\S]*" + r"pci-bus: (\d+) functions found[\s\S]*" r"device-manager: test mode: killing the reporter[\s\S]*" r"device-manager: restarting \S*pci-bus[\s\S]*" - r"pci-bus: \1 functions found[\s\S]*" + r"device-manager: delegated device \1 to \S*pci-bus[\s\S]*" + r"pci-bus: \2 functions found[\s\S]*" r"DANOS-TEST-RESULT: PASS", "fail": r"DANOS-TEST-RESULT: FAIL"}, # M18.3: the application surface — device-list enumerates the tree over IPC, @@ -992,6 +1040,23 @@ CASES = [ {"name": "apertures", "expect": r"DANOS-TEST-RESULT: PASS", "fail": r"DANOS-TEST-RESULT: FAIL"}, + # Delegation's mechanism (docs/os-development/device-authority.md). A claim is + # exclusive, so handing a device on is a MOVE: the giver stops holding it the + # instant the receiver starts - which is why this is not the M13 capability path, + # where a handle is shared refcounted. The kernel's whole rule is "you may give + # away what you hold"; it has no notion of which task is the device manager. + {"name": "device-transfer", + "expect": r"DANOS-TEST-RESULT: PASS", + "fail": r"DANOS-TEST-RESULT: FAIL"}, + # The attacker the device suite never had. The audit's finding was that a fully + # green suite missed six real defects because it contains no attacker: every device + # case asserts a driver handed its hardware can drive it, and none asks what a + # process handed NOTHING can do. This fixture is that process - it holds no device + # and asserts it can give none away, with a positive control first so the refusals + # are decisions rather than a broken syscall path. + {"name": "device-authority", + "expect": r"DANOS-TEST-RESULT: PASS", + "fail": r"DANOS-TEST-RESULT: FAIL"}, # IRQ teardown: an exiting driver's line is masked and its slot cleared (so no # ISR notifies a freed endpoint), and a sibling owner sharing that endpoint # keeps its own binding. A long-running driver never reaches this teardown path. diff --git a/test/system/services/crash-test/crash-test.zig b/test/system/services/crash-test/crash-test.zig index b5c262a..5806220 100644 --- a/test/system/services/crash-test/crash-test.zig +++ b/test/system/services/crash-test/crash-test.zig @@ -19,13 +19,15 @@ pub fn main(init: process.Init) void { const argument = init.arguments.get(1) orelse return; // bare: stay silent const assigned = std.fmt.parseInt(u64, argument, 10) catch return; - // The respawn only reaches this line because the kernel released the - // previous instance's claim at death. A failed claim exits cleanly — the - // manager reads "meant to stop" and the scenario fails loudly by silence. - device.claim(assigned) catch { - _ = logging.write("crash-test: claim failed\n"); - return; - }; + // The device arrived with the spawn — this fixture is delegated its hardware like + // any other driver, so it holds `assigned` before its first instruction and has + // nothing to claim (docs/os-development/device-authority.md). + // + // The property this scenario checks is unchanged, only its mechanism: a respawned + // instance still gets the device its predecessor held. It used to arrive because + // the kernel released the dead instance's claim and this one re-took it, racing + // anyone else who wanted it; now the device reverts to the manager on death and is + // handed to the replacement, which is the same guarantee without the race. var manager: ?ipc.Handle = null; var tries: u32 = 0; diff --git a/test/system/services/device-authority-test/build.zig b/test/system/services/device-authority-test/build.zig new file mode 100644 index 0000000..b5dabc0 --- /dev/null +++ b/test/system/services/device-authority-test/build.zig @@ -0,0 +1,15 @@ +//! The device-authority-test fixture as a binary package (docs/build-packages-plan.md): +//! this file names the binary and EXACTLY the modules its source imports — +//! build-support resolves each name from the domains this zon declares. + +const std = @import("std"); +const build_support = @import("build-support"); + +pub fn build(b: *std.Build) void { + const exe = build_support.userBinary(b, .{ + .name = "device-authority-test", + .root_source_file = b.path("device-authority-test.zig"), + .imports = &.{ "driver", "logging", "process", "time" }, + }); + b.installArtifact(exe); +} diff --git a/test/system/services/device-authority-test/build.zig.zon b/test/system/services/device-authority-test/build.zig.zon new file mode 100644 index 0000000..08eed1e --- /dev/null +++ b/test/system/services/device-authority-test/build.zig.zon @@ -0,0 +1,15 @@ +.{ + .name = .device_authority_test, + .version = "0.0.0", + .fingerprint = 0x4acbba0c105a1462, // Changing this has security and trust implications. + .minimum_zig_version = "0.16.0", + .dependencies = .{ + // build-support supplies the shared recipe; kernel is implicit in + // every binary (the root shim + link script live there). The rest + // are exactly the homes of this binary's declared imports. + .@"build-support" = .{ .path = "../../../../build-support" }, + .kernel = .{ .path = "../../../../library/kernel" }, + .device = .{ .path = "../../../../library/device" }, + }, + .paths = .{""}, +} diff --git a/test/system/services/device-authority-test/device-authority-test.zig b/test/system/services/device-authority-test/device-authority-test.zig new file mode 100644 index 0000000..01fd317 --- /dev/null +++ b/test/system/services/device-authority-test/device-authority-test.zig @@ -0,0 +1,124 @@ +//! device-authority-test — the attacker the device suite never had. +//! +//! The audit behind [docs/fixed-bounds-audit.md] found six real defects that a +//! fully green suite had missed, and the reason was structural: *the suite +//! contains no attacker*. Every device case asserts that a driver handed its +//! own hardware can drive it. None asks what a process that was handed +//! **nothing** can do. +//! +//! This binary is that process. It is spawned with no device, holds no device, +//! and asserts what it therefore cannot do +//! ([docs/os-development/device-authority.md]): +//! +//! 1. **A positive control first.** `device_enumerate` works from here, so +//! the refusals below are decisions rather than a syscall path that is +//! simply broken for this process. Without this, "everything failed" would +//! read identically to "the assertions are meaningless". +//! 2. **It cannot give away a device it does not hold** — not one another +//! task holds, and not a free one either. The kernel's whole rule is *you +//! may give away what you hold*, so the state of the device is irrelevant: +//! a process holding nothing can transfer nothing. That is asserted across +//! several ids precisely so it cannot pass by accident of which device +//! happened to be free at boot. +//! 3. **A device that does not exist is refused differently** — `NoSuchDevice` +//! rather than `NotHeld`. A refusal that cannot say which rule refused it +//! is what cost a debugging session on the Ryzen, so the distinction is +//! part of the contract and is tested as such. +//! +//! **Why there is no "cannot take a delegated device" assertion here.** The +//! hole this fixture was written for is closed, but not by a refusal it could +//! observe. A device that was given to someone is *held*, so an attempt to +//! take it is refused as `AlreadyClaimed` — the same answer as before. What +//! changed is what happens when the holder dies: the device returns to +//! whoever lent it instead of becoming free, so the window in which a +//! stranger could take it no longer exists. There is no moment to catch. + +const std = @import("std"); +const device = @import("driver"); +const logging = @import("logging"); +const process = @import("process"); +const time = @import("time"); + +fn line(comptime format: []const u8, arguments: anytype) void { + var buffer: [160]u8 = undefined; + _ = logging.write(std.fmt.bufPrint(&buffer, format, arguments) catch return); +} + +var failures: usize = 0; +var process_table: [64]process.ProcessDescriptor = undefined; + +fn check(name: []const u8, ok: bool) void { + if (!ok) failures += 1; + line("device-authority: {s} {s}\n", .{ if (ok) "ok" else "FAIL", name }); +} + +fn run() void { + // 1. The positive control: this process can reach the device syscalls at all. + var table: [64]device.DeviceDescriptor = undefined; + const total = device.enumerate(&table); + check("enumerate works from an unprivileged process", total > 0); + const seen = @min(total, table.len); + + // 2. Holding nothing, it can give nothing away — whatever the device's state. + // Every id the machine actually has, so this cannot pass by luck. + var refused: usize = 0; + var wrong_reason: usize = 0; + for (table[0..seen]) |descriptor| { + device.transfer(descriptor.id, process.taskId()) catch |e| { + refused += 1; + if (e != error.NotHeld) wrong_reason += 1; + continue; + }; + } + check("every transfer by a non-holder is refused", refused == seen); + check("each refusal says NotHeld, not something vaguer", wrong_reason == 0); + + // 3. A device that does not exist is a different refusal, and says so. + const absent = if (device.transfer(0xFFFF_FFFF, process.taskId())) |_| false else |e| e == error.NoSuchDevice; + check("a device that does not exist is refused as absent", absent); + + // 4. **The spawn is not a second way in.** A device now rides system_spawn, which + // would be a fine back door if the kernel checked ownership any less carefully + // there than it does in transfer: spawn a child, name someone else's device, and + // the child holds hardware nobody gave it. The refusal must happen before the + // child exists, so nothing is left running either. + if (seen != 0) { + const before = process.processes(&process_table); + const spawned = process.spawnSupervisedWithDevice("/test/system/services/device-authority-test", &.{}, null, table[0].id); + check("spawning with a device the caller does not hold is refused", spawned == null); + check("and no child was left behind by the refusal", process.processes(&process_table) == before); + } + + // 5. **Nothing with mappable resources is left lying around.** A device nobody + // holds can be claimed by anyone, so the manager takes every seeded device that + // carries resources — the HPET above all, which has an MMIO window and an IRQ + // and no user-space driver. The one exception is the loader's framebuffer, + // which the compositor claims. So from here, a resource-bearing device should + // refuse to be taken, and the reason should be that someone already has it. + // Settle first, then sweep **once**. The manager is still starting when this + // fixture is spawned, so an immediate sweep finds hardware unheld and reports a + // hole that closes a millisecond later. Retrying until the sweep comes back + // empty is worse than useless: the first pass *takes* the device, so the second + // finds it unavailable — because this process now holds it — and concludes all + // is well. One sweep, after a wait long enough for the manager to have claimed. + time.sleepMillis(1500); + var takeable: usize = 0; + for (table[0..seen]) |descriptor| { + if (descriptor.resource_count == 0) continue; + if (descriptor.class == @intFromEnum(device.DeviceClass.display)) continue; + device.claim(descriptor.id) catch continue; // refused, as it should be + takeable += 1; + } + check("no resource-bearing device is left for the taking", takeable == 0); + + if (failures == 0) { + line("device-authority: VERDICT ok ({d} devices, none of them mine)\n", .{seen}); + } else { + line("device-authority: VERDICT FAILED {d} assertion(s)\n", .{failures}); + } +} + +pub fn main(startup: process.Init) void { + const role = startup.arguments.get(1) orelse return; // bare (ramdisk sweep): stay silent + if (std.mem.eql(u8, role, "run")) run(); +} diff --git a/tools/bounds-allowlist.txt b/tools/bounds-allowlist.txt new file mode 100644 index 0000000..3eea7a4 --- /dev/null +++ b/tools/bounds-allowlist.txt @@ -0,0 +1,281 @@ +# Bounds that predate the rule (docs/os-development/bounds.md). +# +# Generated from the tree as it stood when the check landed, so the gate could start +# without a 278-site sweep in front of it. Every line is a compile-time ceiling that +# has not yet said what it counts, who decides its size, what it protects, what +# happens when it is reached, or how anyone finds out. +# +# **This list may only shrink.** Declaring a bound means deleting its line; the check +# fails if a listed bound is now declared, and fails if a listed bound has vanished. +# Nothing may be added to it — a new ceiling declares itself or does not land. +# +# docs/fixed-bounds-audit.md is the analysis of how they got here. +boot/efi.zig:buffer +boot/efi.zig:info_buffer +boot/efi.zig:maximum_bundled +boot/efi.zig:maximum_tree_depth +library/device/acpi/aml/interpreter.zig:argbuf +library/device/acpi/aml/interpreter.zig:args +library/device/acpi/aml/interpreter.zig:locals +library/device/acpi/aml/interpreter.zig:maximum_segments +library/device/acpi/aml/interpreter.zig:notify_queue +library/device/acpi/aml/namespace.zig:segment +library/device/acpi/aml/parser.zig:maximum_segments +library/device/driver/driver.zig:lookup_attempts +library/device/model/device-abi.zig:hid +library/device/model/device-abi.zig:maximum_device_resources +library/device/pci/pci.zig:bar_virtual +library/device/pci/pci.zig:bars +library/device/registry/device-registry.zig:cols +library/device/registry/device-registry.zig:rules +library/device/registry/device-registry.zig:rules +library/device/registry/device-registry.zig:rules +library/device/registry/device-registry.zig:rules +library/device/registry/device-registry.zig:rules +library/kernel/channel.zig:bind_attempts +library/kernel/channel.zig:name_maximum +library/kernel/channel.zig:path_maximum +library/kernel/file-system.zig:name_buffer +library/kernel/file-system.zig:out +library/kernel/logging.zig:buffer +library/kernel/process.zig:blob +library/kernel/process.zig:receive +library/kernel/process.zig:table +library/kernel/service.zig:subscriber_capacity +library/kernel/start.zig:buffer +library/kernel/thread.zig:readers +library/kernel/thread.zig:writers +library/protocol/device-manager/device-manager-protocol.zig:hid +library/protocol/display/display-protocol.zig:max_modes +library/protocol/envelope/envelope.zig:packet_maximum +library/protocol/envelope/envelope.zig:post_maximum +library/protocol/power/power-protocol.zig:hid +library/protocol/scanout/scanout-protocol.zig:max_modes +library/protocol/usb-transfer/usb-transfer-protocol.zig:max_report_data +library/protocol/usb-transfer/usb-transfer-protocol.zig:max_reported_endpoints +library/protocol/usb-transfer/usb-transfer-protocol.zig:setup +system/abi.zig:errno_maximum +system/abi.zig:klog_maximum_message +system/abi.zig:maximum_process_name +system/boot-handoff.zig:kernel_segments +system/drivers/pci-bus/pci-bus.zig:line +system/drivers/pci-bus/pci-bus.zig:sub_buffer +system/drivers/ps2-bus/mouse-packet.zig:bytes +system/drivers/ps2-bus/scancode.zig:pressed +system/drivers/ps2-bus/scancode.zig:set2_base +system/drivers/ps2-bus/scancode.zig:set2_extended +system/drivers/usb-hid/hid-report.zig:keys +system/drivers/usb-hid/keyboard.zig:receive +system/drivers/usb-hid/mouse.zig:receive +system/drivers/usb-storage/bulk-only-transport.zig:cdb +system/drivers/usb-storage/scsi.zig:op_read_capacity_10 +system/drivers/usb-storage/usb-storage.zig:capacity_bytes +system/drivers/usb-xhci-bus/usb-xhci-bus.zig:hid_buffer +system/drivers/usb-xhci-bus/usb-xhci-bus.zig:maker_buffer +system/drivers/usb-xhci-bus/usb-xhci-bus.zig:maker_buffer +system/drivers/usb-xhci-bus/usb-xhci-bus.zig:prev_connected +system/drivers/usb-xhci-bus/usb-xhci-bus.zig:product_buffer +system/drivers/usb-xhci-bus/usb-xhci-bus.zig:product_buffer +system/drivers/usb-xhci-bus/usb-xhci-library.zig:buffer +system/drivers/usb-xhci-bus/usb-xhci-library.zig:bytes +system/drivers/usb-xhci-bus/usb-xhci-library.zig:data +system/drivers/usb-xhci-bus/usb-xhci-library.zig:descriptor +system/drivers/usb-xhci-bus/usb-xhci-library.zig:head +system/drivers/usb-xhci-bus/usb-xhci-library.zig:header +system/drivers/usb-xhci-bus/usb-xhci-library.zig:max_subscriptions +system/drivers/usb-xhci-bus/usb-xhci-library.zig:port_changes +system/drivers/usb-xhci-bus/usb-xhci-library.zig:raw +system/drivers/usb-xhci-bus/usb-xhci-library.zig:report_queue_capacity +system/drivers/usb-xhci-bus/usb-xhci-library.zig:request_set_hub_depth +system/drivers/virtio-gpu/virtio-gpu-protocol.zig:edid +system/drivers/virtio-gpu/virtio-gpu-protocol.zig:max_scanouts +system/drivers/virtio-gpu/virtio-gpu.zig:descriptors +system/drivers/virtio-gpu/virtio-gpu.zig:max_height +system/drivers/virtio-gpu/virtio-gpu.zig:max_width +system/initial-ramdisk.zig:buffer +system/initial-ramdisk.zig:buffer +system/initial-ramdisk.zig:maximum_name +system/kernel/acpi.zig:APIC +system/kernel/acpi.zig:BERT +system/kernel/acpi.zig:CPEP +system/kernel/acpi.zig:DMAR +system/kernel/acpi.zig:DSDT +system/kernel/acpi.zig:ECDT +system/kernel/acpi.zig:EINJ +system/kernel/acpi.zig:ERST +system/kernel/acpi.zig:FACP +system/kernel/acpi.zig:FACS +system/kernel/acpi.zig:HEST +system/kernel/acpi.zig:HPET +system/kernel/acpi.zig:IVRS +system/kernel/acpi.zig:MCFG +system/kernel/acpi.zig:MPST +system/kernel/acpi.zig:MSCT +system/kernel/acpi.zig:PMTT +system/kernel/acpi.zig:PSDT +system/kernel/acpi.zig:RASF +system/kernel/acpi.zig:RSDT +system/kernel/acpi.zig:SBST +system/kernel/acpi.zig:SLIT +system/kernel/acpi.zig:SPCR +system/kernel/acpi.zig:SRAT +system/kernel/acpi.zig:SSDT +system/kernel/acpi.zig:XSDT +system/kernel/acpi.zig:aml_block_len +system/kernel/acpi.zig:aml_block_physical +system/kernel/acpi.zig:holes +system/kernel/acpi.zig:maximum_rmrr +system/kernel/acpi.zig:nb +system/kernel/acpi.zig:nb +system/kernel/acpi.zig:nb +system/kernel/acpi.zig:oem_id +system/kernel/acpi.zig:oem_id +system/kernel/acpi.zig:oem_id +system/kernel/acpi.zig:oem_table_id +system/kernel/acpi.zig:overrides +system/kernel/acpi.zig:rmrr_limit_offset +system/kernel/acpi.zig:signature +system/kernel/acpi.zig:signature +system/kernel/acpi.zig:signature +system/kernel/architecture/x86_64/cpu.zig:irq_vector_count +system/kernel/architecture/x86_64/idt.zig:gate_count +system/kernel/architecture/x86_64/ioapic.zig:overrides +system/kernel/architecture/x86_64/iommu-amd.zig:buffer +system/kernel/architecture/x86_64/iommu-amd.zig:buffer +system/kernel/architecture/x86_64/iommu-intel.zig:buffer +system/kernel/architecture/x86_64/iommu-intel.zig:buffer +system/kernel/architecture/x86_64/iommu-intel.zig:context_table +system/kernel/device-model.zig:buffer +system/kernel/device-model.zig:cbuf +system/kernel/device-model.zig:hid_buffer +system/kernel/device-model.zig:maximum_resources +system/kernel/device-model.zig:name_buffer +system/kernel/device-model.zig:rbuf +system/kernel/heap.zig:heap_maximum +system/kernel/ipc-synchronous.zig:MESSAGE_MAXIMUM +system/kernel/ipc-synchronous.zig:POST_MAXIMUM +system/kernel/ipc-synchronous.zig:notify_buffer +system/kernel/ipc-synchronous.zig:post_capacity +system/kernel/irq.zig:maximum_gsi +system/kernel/irq.zig:msi_bound +system/kernel/irq.zig:msi_owner +system/kernel/irq.zig:vector_gsi +system/kernel/kernel.zig:buffer +system/kernel/kernel.zig:buffer +system/kernel/kernel.zig:buffer +system/kernel/kernel.zig:isos +system/kernel/kernel.zig:maximum_wake_attempts +system/kernel/log-ring.zig:message +system/kernel/log-ring.zig:out +system/kernel/log.zig:buffer +system/kernel/log.zig:maximum_sinks +system/kernel/log.zig:message +system/kernel/log.zig:ring_capacity +system/kernel/process.zig:chunk +system/kernel/process.zig:chunk +system/kernel/process.zig:chunk +system/kernel/process.zig:chunk +system/kernel/process.zig:exit_record_capacity +system/kernel/process.zig:exit_subscriber_capacity +system/kernel/process.zig:maximum_argument_bytes +system/kernel/process.zig:maximum_arguments +system/kernel/process.zig:maximum_dma_regions +system/kernel/process.zig:maximum_mmap_pages +system/kernel/process.zig:maximum_mount_prefix +system/kernel/process.zig:maximum_mount_rewrite +system/kernel/process.zig:maximum_pages +system/kernel/process.zig:maximum_resolve_path +system/kernel/process.zig:maximum_segments +system/kernel/process.zig:maximum_shared_memory_pages +system/kernel/process.zig:name_buffer +system/kernel/process.zig:timer_capacity +system/kernel/process.zig:word_bytes +system/kernel/process.zig:write_buffer +system/kernel/scheduler.zig:ipc_maximum_handles +system/kernel/scheduler.zig:maximum_space_mappings +system/kernel/vfs.zig:maximum_directories +system/kernel/vfs.zig:maximum_mounts +system/kernel/vfs.zig:maximum_prefix +system/kernel/vfs.zig:maximum_rewrite +system/services/acpi/acpi.zig:blocks +system/services/acpi/acpi.zig:buffer +system/services/acpi/acpi.zig:hid +system/services/acpi/acpi.zig:mmio_scratch +system/services/acpi/acpi.zig:name +system/services/acpi/acpi.zig:registered +system/services/device-manager/device-manager.zig:arguments +system/services/device-manager/device-manager.zig:id_text +system/services/device-manager/device-manager.zig:maximum_children +system/services/device-manager/device-manager.zig:maximum_drivers +system/services/device-manager/device-manager.zig:name_buffer +system/services/device-manager/device-manager.zig:registry_rules +system/services/device-manager/device-manager.zig:registry_source +system/services/display/backend.zig:device_table +system/services/display/backend.zig:line +system/services/display/compositor.zig:capacity +system/services/display/compositor.zig:maximum_columns +system/services/display/compositor.zig:maximum_rects +system/services/display/compositor.zig:maximum_rows +system/services/display/display.zig:line +system/services/display/display.zig:line +system/services/display/display.zig:line +system/services/display/display.zig:list +system/services/display/display.zig:maximum_layers +system/services/display/display.zig:mode_list +system/services/fat/engine.zig:buf +system/services/fat/engine.zig:buf +system/services/fat/engine.zig:buf +system/services/fat/engine.zig:buffer +system/services/fat/engine.zig:cached_back +system/services/fat/engine.zig:chunk +system/services/fat/engine.zig:chunk +system/services/fat/engine.zig:chunk +system/services/fat/engine.zig:device_back +system/services/fat/engine.zig:display +system/services/fat/engine.zig:long_name +system/services/fat/engine.zig:long_name +system/services/fat/engine.zig:long_name +system/services/fat/engine.zig:max_transfer_sectors +system/services/fat/engine.zig:pair +system/services/fat/engine.zig:pair +system/services/fat/engine.zig:payload +system/services/fat/engine.zig:payload +system/services/fat/engine.zig:payload +system/services/fat/engine.zig:readback +system/services/fat/engine.zig:run +system/services/fat/engine.zig:run +system/services/fat/engine.zig:short +system/services/fat/engine.zig:short +system/services/fat/engine.zig:short +system/services/fat/engine.zig:stem +system/services/fat/engine.zig:tail_buffer +system/services/fat/engine.zig:units +system/services/fat/engine.zig:value +system/services/fat/engine.zig:value +system/services/fat/engine.zig:window +system/services/fat/on-disk.zig:filesystem_type +system/services/fat/on-disk.zig:filesystem_type +system/services/fat/on-disk.zig:jump +system/services/fat/on-disk.zig:name +system/services/fat/on-disk.zig:name1 +system/services/fat/on-disk.zig:name2 +system/services/fat/on-disk.zig:name3 +system/services/fat/on-disk.zig:oem_name +system/services/fat/on-disk.zig:volume_label +system/services/fat/on-disk.zig:volume_label +system/services/init/init.zig:binary +system/services/init/init.zig:init_csv +system/services/init/init.zig:max_service_args +system/services/init/init.zig:max_services +system/services/init/init.zig:maximum_bindings +system/services/init/init.zig:maximum_grants +system/services/init/init.zig:maximum_name +system/services/init/init.zig:maximum_restarts +system/services/init/init.zig:process_table +system/services/init/init.zig:protocol_csv +system/services/logger/logger.zig:chunk +system/services/logger/logger.zig:gap_line +system/services/logger/logger.zig:line +system/services/logger/logger.zig:line +system/services/logger/logger.zig:maximum_files +system/services/logger/logger.zig:stamp diff --git a/tools/check-bounds.py b/tools/check-bounds.py new file mode 100644 index 0000000..8b85664 --- /dev/null +++ b/tools/check-bounds.py @@ -0,0 +1,189 @@ +#!/usr/bin/env python3 +"""Every compile-time ceiling states what it is doing there. + +A *bound* is a number chosen at compile time that decides how much of something the +code can hold: `const maximum_devices = 64`, `var below: [64]Range`, `var blob: [512]u8`. +Different units, one shape, and one recurring way of going wrong — see +docs/fixed-bounds-audit.md, where 235 of them turned up, 139 on quantities the machine +or a file decides rather than us, and 171 silent when reached. + +This is the gate that keeps new ones from joining them. It does not resize anything and +it makes no judgement about whether a bound should exist; it only refuses one that will +not say what it is for. The declaration is a doc comment immediately above: + + /// bound: logical CPUs the kernel tracks + /// decided-by: hardware + /// protects: the per-CPU bookkeeping arrays, which are sized at compile time + /// at-limit: degrade - surplus cores are left parked, never brought online + /// observed-by: platform.cpusDropped() -> the WARNING at kernel.zig:281 + pub const maximum_cpus = 128; + +`decided-by` is the field the audit turned on: `hardware` and `external` mean the +quantity is not ours to choose, and a fixed bound on one of those is a defect rather +than a tunable. `at-limit`'s vocabulary is closed on purpose — there is no `silent`, +no `drop`, and nothing meaning *allow*, so the behaviours that caused the damage cannot +be written down. `truncate` is legal only with a marker the reader can see. + +An array length that names a declared bound (`[maximum_devices]Descriptor`) is not +itself a bound: the number lives at the declaration, and that is where it is declared. +Only literal lengths are flagged, which pushes ceilings toward having names. + +The ~235 that already exist are listed in tools/bounds-allowlist.txt so this can land +without a tree-wide sweep in front of it. That list may only shrink: declaring a bound +means deleting its line, and a stale line is an error too. + +Usage: check-bounds.py [--list] [repo-root] + --list print every undeclared bound found, for regenerating the allowlist +""" + +import re +import sys +from pathlib import Path + +# Directories worth gating. `test/` is excluded: a fixture's `[100]MemoryRegion` is a +# test input, not a ceiling the system runs into. +ROOTS = ("system", "library", "boot") +SKIP_PARTS = {".zig-cache", "zig-out", ".git", ".claude", "vendor", "generated"} +SKIP_FILES = {"tests.zig"} + +FIELDS = ("bound", "decided-by", "protects", "at-limit", "observed-by") +DECIDED_BY = {"hardware", "external", "ours"} +AT_LIMIT = {"refuse", "degrade", "truncate", "grow"} + +# A `const` whose name reads like a ceiling and whose value is an integer literal. +NAMED = re.compile( + r"^\s*(?:pub\s+)?const\s+([A-Za-z_]\w*)\s*(?::\s*[\w.\[\]]+\s*)?=\s*" + r"(\d[\d_]*|0x[0-9a-fA-F_]+)\s*(?:\*\s*\d[\d_]*\s*)*;" +) +NAME_IS_BOUND = re.compile(r"(^|_)(maximum|max|limit|capacity|depth|attempts|count)($|_)", re.I) + +# A declaration or struct field whose type carries a *literal* array length. +ARRAY_DECL = re.compile(r"^\s*(?:pub\s+)?(?:const|var)\s+([A-Za-z_]\w*)\s*:[^=]*?\[\s*(\d[\d_]*)\s*\]") +ARRAY_FIELD = re.compile(r"^\s*([A-Za-z_]\w*)\s*:\s*\[\s*(\d[\d_]*)\s*\]") + +# Struct padding and reserved fields are shapes, not ceilings — nothing is ever "held" +# in them. Everything else with a literal length is a candidate, *including* the tidy +# powers of two: `[512]u8` and `[64]Range` were the two worst findings in the audit, and +# any size-based exemption would have skipped exactly them. A length that is genuinely a +# fact rather than a ceiling says so in its declaration ("decided-by: hardware, the PCI +# spec gives a function 6 BARs") — that is what the declaration is for. +NOT_A_BOUND_NAME = re.compile(r"^_*(padding|pad|reserved|unused|spare)\d*$", re.I) + + +def sources(root: Path): + for top in ROOTS: + base = root / top + if not base.is_dir(): + continue + for path in sorted(base.rglob("*.zig")): + if SKIP_PARTS & set(path.parts) or path.name in SKIP_FILES: + continue + yield path + + +def declaration_above(lines, index): + """The `/// key: value` block immediately above line `index`, as a dict.""" + fields = {} + i = index - 1 + while i >= 0: + stripped = lines[i].strip() + if not stripped.startswith("///"): + break + body = stripped[3:].strip() + match = re.match(r"([a-z-]+):\s*(.+)", body) + if match: + fields[match.group(1)] = match.group(2).strip() + i -= 1 + return fields + + +def problems_with(fields): + """Why a declaration is not acceptable, or an empty list.""" + missing = [f for f in FIELDS if f not in fields or not fields[f]] + if missing: + return ["missing " + ", ".join(missing)] + out = [] + if fields["decided-by"] not in DECIDED_BY: + out.append(f"decided-by must be one of {sorted(DECIDED_BY)}, not {fields['decided-by']!r}") + verb = fields["at-limit"].split()[0].strip("-:,").lower() + if verb not in AT_LIMIT: + out.append( + f"at-limit must start with one of {sorted(AT_LIMIT)}, not {verb!r}. " + "There is deliberately no way to say 'silent', 'drop', or anything meaning 'allow'" + ) + if verb == "truncate" and len(fields["at-limit"].split()) < 3: + out.append("at-limit: truncate must say how a reader can TELL it happened") + return out + + +def find(root: Path): + """Every bound-shaped declaration: (relative path, name, value, line, fields).""" + for path in sources(root): + rel = path.relative_to(root).as_posix() + lines = path.read_text(encoding="utf-8", errors="replace").split("\n") + for n, line in enumerate(lines): + if line.lstrip().startswith("//"): + continue + name = value = None + m = NAMED.match(line) + if m and NAME_IS_BOUND.search(m.group(1)): + name, value = m.group(1), m.group(2) + else: + m = ARRAY_DECL.match(line) or ARRAY_FIELD.match(line) + if m and not NOT_A_BOUND_NAME.match(m.group(1)): + name, value = m.group(1), m.group(2) + if name: + yield rel, name, value, n + 1, declaration_above(lines, n) + + +def main(): + argv = [a for a in sys.argv[1:] if not a.startswith("--")] + listing = "--list" in sys.argv + root = Path(argv[0]) if argv else Path(__file__).resolve().parent.parent + + allow_path = root / "tools" / "bounds-allowlist.txt" + allowed = set() + if allow_path.exists(): + for raw in allow_path.read_text().split("\n"): + entry = raw.split("#", 1)[0].strip() + if entry: + allowed.add(entry) + + undeclared, bad, seen = [], [], set() + for rel, name, value, line, fields in find(root): + key = f"{rel}:{name}" + seen.add(key) + if not fields: + (undeclared if key not in allowed else []).append((key, value, line)) + continue + for why in problems_with(fields): + bad.append((key, line, why)) + if key in allowed and not problems_with(fields): + bad.append((key, line, "now declared — delete its line from tools/bounds-allowlist.txt")) + + if listing: + for key, value, line in sorted(undeclared): + print(f"{key} # = {value}, line {line}") + for key in sorted(allowed - seen): + print(f"# STALE: {key}") + return 0 + + stale = sorted(allowed - seen) + if not undeclared and not bad and not stale: + return 0 + + print("bounds check failed\n", file=sys.stderr) + for key, value, line in sorted(undeclared): + print(f" {key} (= {value}, line {line})", file=sys.stderr) + print(" no declaration. A ceiling states what it counts, who decides its", file=sys.stderr) + print(" size, what it protects, what happens at the limit, and how you", file=sys.stderr) + print(" find out. See docs/os-development/bounds.md.", file=sys.stderr) + for key, line, why in sorted(bad): + print(f" {key} (line {line}): {why}", file=sys.stderr) + for key in stale: + print(f" {key}: allowlisted but no longer found — delete its line", file=sys.stderr) + return 1 + + +if __name__ == "__main__": + sys.exit(main())