Merge the bounds track: the numbers we invented are gone
An AMD Ryzen booted to a working compositor with no USB and no storage. The cause was maximum_children_per_parent = 16 — a number nobody had justified, in a kernel where 235 compile-time ceilings turned out to exist, 139 of them on quantities the machine or a file decides rather than us, and 171 silent when reached. Both invented ceilings are deleted rather than raised. The device table grows, because no specification bounds how many devices a machine has, and a runaway is bounded per-registrar so it costs only the driver that caused it. Device authority became real rather than advisory. Matching was always policy, but claiming was first-come, so any process could take any device nobody happened to be holding. Now every driver receives its hardware in the spawn that creates it — it never runs without it — a grant is a loan that returns to its lender when the borrower dies, and nothing firmware-discovered is left unheld except the framebuffer, which the compositor owns. Along the way: the AMD boot fix itself (STAR's SYSRET base carries the RPL, which Intel forgives and AMD does not); one errno space with every refusal naming the rule that refused it; the IOMMU confinement moving with its device, and refusing rather than silently succeeding at the edge of its table; and PCI apertures derived without copying the firmware memory map into a fixed array, which had let a large map turn occupied RAM into a window a driver could map. A gate now stands behind the audit: zig build bounds refuses a new ceiling that will not say what it counts, who decides its size, what happens when it is reached, and how anyone finds out. The 269 that predate the rule are allowlisted and that list can only shrink. Suite 114 -> 118, green at every step. Docs: fixed-bounds-audit.md, os-development/bounds.md, bounds-track-plan.md, os-development/device-authority.md.
This commit is contained in:
@@ -342,6 +342,7 @@ pub fn build(b: *std.Build) void {
|
|||||||
"protocol-registry-test", // drives the registrar: ungranted bind, collision, restart
|
"protocol-registry-test", // drives the registrar: ungranted bind, collision, restart
|
||||||
"protocol-denied-test", // restriction stage one: an ungranted open answers as absence
|
"protocol-denied-test", // restriction stage one: an ungranted open answers as absence
|
||||||
"protocol-conformance-test", // the reserved verbs, asked of every provider the boot bound
|
"protocol-conformance-test", // the reserved verbs, asked of every provider the boot bound
|
||||||
|
"device-authority-test", // the attacker: a process handed no device, asserting what it cannot do
|
||||||
}) |fixture| {
|
}) |fixture| {
|
||||||
const package = b.lazyDependency(fixture, .{}) orelse
|
const package = b.lazyDependency(fixture, .{}) orelse
|
||||||
@panic("a test fixture package is missing under test/system/services");
|
@panic("a test fixture package is missing under test/system/services");
|
||||||
@@ -397,6 +398,19 @@ pub fn build(b: *std.Build) void {
|
|||||||
// here (compiled for the host rather than inheriting a freestanding target) —
|
// here (compiled for the host rather than inheriting a freestanding target) —
|
||||||
// which also compile-checks that the three-way split stays self-consistent.
|
// which also compile-checks that the three-way split stays self-consistent.
|
||||||
const test_step = b.step("test", "Run tests");
|
const test_step = b.step("test", "Run tests");
|
||||||
|
|
||||||
|
// Every compile-time ceiling states what it counts, who decides its size, what it
|
||||||
|
// protects, what happens when it is reached, and how anyone finds out
|
||||||
|
// (docs/os-development/bounds.md). The ~273 that predate the rule are listed in
|
||||||
|
// tools/bounds-allowlist.txt so this could land without a tree-wide sweep first;
|
||||||
|
// that list may only shrink. Part of `test` rather than the default build: it reads
|
||||||
|
// the whole tree, and a red bounds check should not stop you booting a kernel.
|
||||||
|
const bounds_check = b.addSystemCommand(&.{ "python3", "tools/check-bounds.py" });
|
||||||
|
bounds_check.setCwd(b.path("."));
|
||||||
|
const bounds_step = b.step("bounds", "Check that every compile-time ceiling declares itself");
|
||||||
|
bounds_step.dependOn(&bounds_check.step);
|
||||||
|
test_step.dependOn(&bounds_check.step);
|
||||||
|
|
||||||
for ([_][]const u8{
|
for ([_][]const u8{
|
||||||
"system/boot-handoff.zig",
|
"system/boot-handoff.zig",
|
||||||
"system/abi.zig",
|
"system/abi.zig",
|
||||||
|
|||||||
@@ -79,6 +79,7 @@
|
|||||||
.@"protocol-registry-test" = .{ .path = "test/system/services/protocol-registry-test", .lazy = true },
|
.@"protocol-registry-test" = .{ .path = "test/system/services/protocol-registry-test", .lazy = true },
|
||||||
.@"protocol-denied-test" = .{ .path = "test/system/services/protocol-denied-test", .lazy = true },
|
.@"protocol-denied-test" = .{ .path = "test/system/services/protocol-denied-test", .lazy = true },
|
||||||
.@"protocol-conformance-test" = .{ .path = "test/system/services/protocol-conformance-test", .lazy = true },
|
.@"protocol-conformance-test" = .{ .path = "test/system/services/protocol-conformance-test", .lazy = true },
|
||||||
|
.@"device-authority-test" = .{ .path = "test/system/services/device-authority-test", .lazy = true },
|
||||||
// See `zig fetch --save <url>` for a command-line interface for adding dependencies.
|
// See `zig fetch --save <url>` for a command-line interface for adding dependencies.
|
||||||
//.example = .{
|
//.example = .{
|
||||||
// // When updating this field to a new URL, be sure to delete the corresponding
|
// // When updating this field to a new URL, be sure to delete the corresponding
|
||||||
|
|||||||
@@ -3,6 +3,426 @@
|
|||||||
*Plan, 2026-08-08. Follows [fixed-bounds-audit.md](fixed-bounds-audit.md) (235 ceilings,
|
*Plan, 2026-08-08. Follows [fixed-bounds-audit.md](fixed-bounds-audit.md) (235 ceilings,
|
||||||
139 on quantities we do not choose) and the AMD Ryzen that found the first one.*
|
139 on quantities we do not choose) and the AMD Ryzen that found the first one.*
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Live state — the unattended run
|
||||||
|
|
||||||
|
*This table is the progress view. It is updated at the end of every step, before the
|
||||||
|
next one starts.*
|
||||||
|
|
||||||
|
| Step | What | State |
|
||||||
|
|---|---|---|
|
||||||
|
| L1 | Reclamation: a dead task's registrations die with its claims | **stopped — the step was wrong; see open question 4** |
|
||||||
|
| L2 | Bounds build check + allowlist; declare what we have already touched | **done** — `zig build bounds`, 273 allowlisted, 5 declared |
|
||||||
|
| L3 | xHCI: slot count from `HCSPARAMS1.MaxSlots`, not 8 | **done** — QEMU reports 64; the driver tracked 8 |
|
||||||
|
| L4 | USB: configuration descriptor sized by `wTotalLength`, not 512 | **done** — QEMU tops out at 211 bytes, so the case catches the class, not the original trigger |
|
||||||
|
| L5 | USB: interfaces from the descriptor, and the misattributed-endpoint bug | **done** — fix is by construction; no direct test, see open question 5 |
|
||||||
|
| L6 | xHCI: a failed `allocateDevice` stops leaking an enabled slot | **done** — path forced and verified; no regression test, see open question 5 |
|
||||||
|
|
||||||
|
**Run 1 complete.** L1 stopped (the step was wrong), L2–L6 landed. Suite 115 → 116.
|
||||||
|
Allowlist 278 → 269. Two steps ship without a permanent regression test, both because
|
||||||
|
QEMU's USB devices are too small to reach the paths — see open question 5, which is the
|
||||||
|
audit's own lesson recurring: the test rig is smaller than a real machine.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Run 2 — device authority: delete the invented ceilings
|
||||||
|
|
||||||
|
*Design: [device-authority.md](os-development/device-authority.md), which is the **how**
|
||||||
|
for the delegation step [device-manager.md](device-driver-development/device-manager.md)
|
||||||
|
already settled. Read both before starting; the second is authoritative where they
|
||||||
|
differ.*
|
||||||
|
|
||||||
|
The goal, in the project owner's words: **remove the maximum values we set arbitrarily,
|
||||||
|
move the responsibility to the device manager, and keep in the kernel only the parts
|
||||||
|
that cannot safely run in user space.**
|
||||||
|
|
||||||
|
| Step | What | State |
|
||||||
|
|---|---|---|
|
||||||
|
| D0 | The grant rides `system_spawn` — atomic, so no driver need change to receive one | **done** |
|
||||||
|
| D1 | `device_transfer(device_id, task_id)` — the holder gives a device away | **done** — syscall 54; a move, not a copy |
|
||||||
|
| D2 | Adversarial case: a process handed nothing is refused | **done** — `device-authority-test`; the claim half joins it at D6 |
|
||||||
|
| D3 | The manager claims the seeded devices before any driver is spawned | **merged into D4** |
|
||||||
|
| D4 | `usb-xhci-bus` receives its controller | **done** — caught an IOMMU regression I introduced |
|
||||||
|
| D5 | `pci-bus`, `virtio-gpu`, `ps2-bus`, discovery receive theirs | **done** — every driver now receives its hardware |
|
||||||
|
| D6 | `device_claim` refuses a device the caller was not handed | **blocked on question 10** |
|
||||||
|
| D7 | Zero-resource devices stop being kernel objects | **blocked on question 8** |
|
||||||
|
| D8 | **`maximum_children_per_parent` deleted** | **done** |
|
||||||
|
| D9 | Dynamic table; **`maximum_devices` deleted**; per-registrar allowance | **done** |
|
||||||
|
| D10 | Every driver hellos, on its own merits | not started — optional, independent |
|
||||||
|
|
||||||
|
*Executed in the order D1, D2, D4, D5(part), D9, D0, D5(rest), D8 — the numbering is
|
||||||
|
the original plan's, not the sequence. D0 was added mid-run, D3 merged into D4, and D8
|
||||||
|
turned out to have been unblocked since D9.*
|
||||||
|
|
||||||
|
**Run 2 stops, blocked on one question.** Landed: D0, D1, D2, D4, D8, D9, and D5 for
|
||||||
|
three of five claimants (`usb-xhci-bus`, `pci-bus`, `virtio-gpu`). Suite 118/118.
|
||||||
|
|
||||||
|
**Both invented ceilings are gone.** `maximum_devices` and
|
||||||
|
`maximum_children_per_parent` no longer exist, and delegation is atomic with the spawn.
|
||||||
|
|
||||||
|
What remains is not a number: **D6 closes the claiming hole** and is blocked on
|
||||||
|
question 9, and D7 moves the inventory and is blocked on question 8.
|
||||||
|
|
||||||
|
Ordering is load-bearing. D1–D2 built and proved the mechanism with nothing depending on
|
||||||
|
it. D4–D5 move each claimant across one at a time, so the suite stays green throughout
|
||||||
|
and a regression names the driver that caused it. D6 is the flag day. (The original
|
||||||
|
"D7 must precede D9" no longer holds: the per-holder quota bounds zero-resource children
|
||||||
|
as well as anything else, so D9 landed without it.)
|
||||||
|
|
||||||
|
**D3 merged into D4** (found while implementing, recorded rather than worked around).
|
||||||
|
The two cannot be separated: the moment the manager claims a device, any driver still
|
||||||
|
calling `device_claim` on it is refused `AlreadyClaimed`, so D3 on its own turns the
|
||||||
|
suite red — and D3 applied to *nothing* changes no behaviour and cannot be tested.
|
||||||
|
They land together, with the manager claiming only for drivers in an explicit
|
||||||
|
**delegated set** so every unconverted driver keeps claiming exactly as before.
|
||||||
|
`usb-xhci-bus` is the first member, as it was the first driver to conform to `hello`
|
||||||
|
(device-manager.md, M18.1). D5 moves the rest in one at a time; the set and the
|
||||||
|
`device_claim` path both disappear at D6.
|
||||||
|
|
||||||
|
### Observations from the run
|
||||||
|
|
||||||
|
- **D4 introduced an IOMMU regression, caught by converting one driver at a time.**
|
||||||
|
`confineDevice` runs inside `systemDeviceClaim`, so a device arriving by *transfer*
|
||||||
|
was never confined for its new owner: the driver's DMA rings went unbound (three
|
||||||
|
IOMMU+USB cases failed), and two worse consequences were latent — a manager death
|
||||||
|
would have torn down a domain a live driver was using, and a driver death would have
|
||||||
|
leaked one. `iommu.reassign` moves the confinement with the device, keeping the
|
||||||
|
domain and attachment intact so it never translates through nothing. Converting all
|
||||||
|
five drivers at once would have produced the same three failures with five suspects.
|
||||||
|
- **`usb-hub` failed once in a full run, then passed six times** (four isolated, two
|
||||||
|
full). Suspected instance of the known intermittent AP ring-3 fault rather than
|
||||||
|
anything in D4 — recorded rather than dismissed, because D4 moved the `hello` earlier
|
||||||
|
and so did shift boot timing. Watch it across the remaining steps.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Run 3 — close the claiming hole
|
||||||
|
|
||||||
|
*Every driver now receives its hardware. What remains is that `device_claim` still
|
||||||
|
works for anyone, so a device nobody holds can still be taken. Six steps, no open
|
||||||
|
questions.*
|
||||||
|
|
||||||
|
### The rule this run implements
|
||||||
|
|
||||||
|
The kernel records, per device, **who gave it**. From that one field everything follows:
|
||||||
|
|
||||||
|
- `device_transfer` and the spawn-fused grant set the giver.
|
||||||
|
- **`device_claim` refuses a device that has a giver.** A device that was given to
|
||||||
|
someone is delegated hardware and must be handed on, never taken.
|
||||||
|
- **On death a device reverts to its giver**, if that task is still alive; otherwise its
|
||||||
|
claim clears. A grant is a loan, not a gift.
|
||||||
|
|
||||||
|
Two things fall out rather than being special-cased:
|
||||||
|
|
||||||
|
- **The framebuffer is untouched.** Nobody delegates it, so it has no giver, so the
|
||||||
|
display service can still claim it exactly as today. No exemption in the kernel, no
|
||||||
|
mention of display anywhere in the rule.
|
||||||
|
- **Restart needs no race.** A dying driver's device returns to the manager, which
|
||||||
|
re-delegates it on respawn. Previously the kernel released it to nobody and the
|
||||||
|
manager re-claimed first-come, so every restart reopened the hole.
|
||||||
|
|
||||||
|
**Found while implementing: E2 is the step that closes the hole, not E3.** A delegated
|
||||||
|
device is *held*, so an attempt to take it is refused as `AlreadyClaimed` long before
|
||||||
|
the giver is consulted — and once a borrower's death returns the device to its lender
|
||||||
|
(or clears both when the lender is gone), there is no state where a device is unheld and
|
||||||
|
still on loan. E3's check is therefore unreachable today. It stays as one comparison
|
||||||
|
that fails closed, guarding any future path that frees a device without clearing its
|
||||||
|
giver, and its comment says so rather than implying a protection it is not providing.
|
||||||
|
|
||||||
|
| Step | What |
|
||||||
|
|---|---|
|
||||||
|
| E1 | Record a giver per device; `device_transfer` and the spawn grant set it — **done** |
|
||||||
|
| E2 | On task death a device reverts to its giver if alive, else its claim clears — **done** |
|
||||||
|
| E3 | `device_claim` refuses a device that has a giver — **done**, but unreachable: E2 already closed the window |
|
||||||
|
| E4 | The manager claims every resource-bearing device at boot, so nothing is left takeable — **done** (boot snapshot only; see below) |
|
||||||
|
| E5 | The attacker fixture gains the claim half it has been waiting for since D2 — **done with E4** |
|
||||||
|
| E6 | Delete the delegated-set scaffolding — every driver is delegated now — **done** |
|
||||||
|
|
||||||
|
Ordering: E1 alone changes no behaviour. E2 must precede E3, or a restart cannot
|
||||||
|
re-acquire. E4 must precede E5, or the attacker will find takeable devices and the
|
||||||
|
assertion will be wrong about why. E6 is cleanup.
|
||||||
|
|
||||||
|
**Run 3 complete.** Suite 118/118. Every driver receives its hardware; a grant is a
|
||||||
|
loan that returns to its lender when the borrower dies; nothing firmware-discovered is
|
||||||
|
left unheld except the framebuffer, which the compositor owns.
|
||||||
|
|
||||||
|
### E4's scope, stated
|
||||||
|
|
||||||
|
It covers the **boot snapshot**. A device *reported* later and matched to no driver
|
||||||
|
stays claimable — `pci-cap-test` and `iommu-fault-test` both rely on that to reach an
|
||||||
|
unmatched NIC. Narrowing it further is a separate change with those fixtures in scope.
|
||||||
|
|
||||||
|
The real gap it closed was the **HPET**: an MMIO window, an IRQ, no user-space driver,
|
||||||
|
and claimable by anyone. Excluded on purpose: the loader's framebuffer (the manager
|
||||||
|
starts before display, so taking it would break the boot screen) and anything with no
|
||||||
|
resources, which grants nothing.
|
||||||
|
|
||||||
|
### Not in this run, and not blocking it
|
||||||
|
|
||||||
|
**Zero-resource devices stay in the kernel.** Moving them out needs an answer to who
|
||||||
|
mints their ids, given that the id is the `device_token` on the usb-transfer wire and
|
||||||
|
the kernel's idempotency is what keeps it stable across a bus restart. Nothing depends
|
||||||
|
on it: both invented ceilings are already gone and this run does not need it. Recorded
|
||||||
|
as future work.
|
||||||
|
|
||||||
|
**Not every driver hellos.** Delivery no longer needs it, so it stands or falls on its
|
||||||
|
own merits — uniform liveness and one class of driver. Also future work.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### Question 9 — answered: a singleton gets every device that matched it
|
||||||
|
|
||||||
|
The 8042 is one controller described by two ACPI nodes, so it cannot be split across
|
||||||
|
processes. The manager now hands the single `ps2-bus` instance **every** matching node:
|
||||||
|
the first rides the spawn, the rest are transferred to the running instance. Late
|
||||||
|
arrival is safe because the ordering is natural rather than lucky — the controller node
|
||||||
|
carries the ports and is needed at once, the mouse node not until after identify
|
||||||
|
(measured: handed over at 0.336, first touched at 0.456). The count is whatever matched,
|
||||||
|
so a machine with no PS/2 ports or one port needs no special case.
|
||||||
|
|
||||||
|
Discovery turned out not to be special either. The kernel seeds the `acpi-tables` node,
|
||||||
|
so it is in the same boot snapshot the manager already scans for the PCI host bridge —
|
||||||
|
it is handed over at spawn like everything else.
|
||||||
|
|
||||||
|
### Open question 10 — the framebuffer is the last thing anyone claims
|
||||||
|
|
||||||
|
`device_claim` now has exactly two callers: the device manager, which is the acquirer
|
||||||
|
and should have it, and `system/services/display/backend.zig`, whose GOP path claims the
|
||||||
|
kernel-seeded display node.
|
||||||
|
|
||||||
|
Closing `device_claim` breaks the compositor's boot floor. Exempting it puts a hole in
|
||||||
|
the middle of the authority model, in the one place an exemption is most expensive.
|
||||||
|
|
||||||
|
The likely answer is neither. **The framebuffer is not a device** — it is where pixels
|
||||||
|
go, handed over by the loader, and the kernel wraps it in a `DeviceClass.display`
|
||||||
|
descriptor only so `mmio_map` can hand it over write-combining. If that is right, it
|
||||||
|
should leave the device table rather than be exempted from its rules, and the compositor
|
||||||
|
should receive the pixels some other way. That is a change to the display path, which is
|
||||||
|
out of scope for this run.
|
||||||
|
|
||||||
|
### Settled 2026-08-08: the grant rides `system_spawn` (D0)
|
||||||
|
|
||||||
|
Questions 6 and 7 both dissolved on inspection — neither `ps2-bus` nor discovery needs
|
||||||
|
to start speaking `hello`, and `virtio-gpu` has no standalone path to lose. What remains
|
||||||
|
is *where the grant is delivered*, and there are three candidates:
|
||||||
|
|
||||||
|
| | Race window | Cost |
|
||||||
|
|---|---|---|
|
||||||
|
| Transfer after spawn | **yes** | none |
|
||||||
|
| Every driver hellos | no | `ps2-bus` + discovery gain a handshake |
|
||||||
|
| **Grant rides `system_spawn`** | **no — atomic** | one more syscall argument |
|
||||||
|
|
||||||
|
**Take the third.** The manager cannot transfer before the child exists, so a separate
|
||||||
|
transfer always leaves a window in which the child is running and does not yet hold its
|
||||||
|
device. It would close on QEMU every time and open occasionally on a machine with
|
||||||
|
different timing — the exact failure shape this track exists to delete, and not worth
|
||||||
|
introducing while removing the others. Fusing the device into the spawn removes it by
|
||||||
|
construction: the child does not exist until it holds the device. No new knowledge in
|
||||||
|
the kernel — the same rule, *you may give away what you hold*, made atomic with the call
|
||||||
|
that creates the recipient. `system_spawn` uses five of six argument registers, so there
|
||||||
|
is room, and `no_device` is already the sentinel for a driver with no assignment.
|
||||||
|
|
||||||
|
**`hello` for every driver is a good idea on its own merits** — uniform liveness, the
|
||||||
|
deadline applied to all rather than some, and the `speaks_protocol` two-class split
|
||||||
|
leaving the manager (a wedged `ps2-bus` is invisible to its supervisor today). It is
|
||||||
|
D10, kept separate so grant delivery does not force it.
|
||||||
|
|
||||||
|
### Open questions raised by D5 — resolved except question 8
|
||||||
|
|
||||||
|
Delegation is delivered in `onHello`. That works for a driver that says hello, and
|
||||||
|
**two of the four do not**.
|
||||||
|
|
||||||
|
6. **`ps2-bus` does not hello**, and that is deliberate: device-manager.md records
|
||||||
|
"Legacy drivers (e.g. ps2-bus) are supervised and restarted but **not yet required
|
||||||
|
to hello**", and `Driver.speaks_protocol` exists to say so. Either it leaves
|
||||||
|
"legacy", or the grant gets a delivery point that is not `hello`.
|
||||||
|
|
||||||
|
**The `acpi` service is not this case — an earlier version of this question wrongly
|
||||||
|
lumped it in.** It is spawned as `addDriver("discovery", no_device, false)`: the
|
||||||
|
*discovery service*, one per firmware, with **no device assignment at all**, which
|
||||||
|
"finds and claims the acpi-tables (or devicetree-blob) node itself. Not a per-device
|
||||||
|
driver." The manager cannot hand it a device because discovery is what produces the
|
||||||
|
device tree — there is nothing to match against yet. Its question is not "should it
|
||||||
|
hello" but "is the bootstrap exempt from D6, or does the manager claim the
|
||||||
|
acpi-tables node and pass it on".
|
||||||
|
|
||||||
|
**A delivery point that needs no hello already exists in the shape of the code**:
|
||||||
|
`process.spawnSupervised` returns the child's pid, so the manager could transfer
|
||||||
|
immediately after spawn. If that is acceptable, this question mostly dissolves —
|
||||||
|
`ps2-bus` would not need to leave "legacy" and discovery could be handed its node
|
||||||
|
too. The cost is ordering: the child may reach for the device before the transfer
|
||||||
|
lands, where `hello` guarantees it cannot because the child is the one asking. That
|
||||||
|
trade is the actual decision.
|
||||||
|
|
||||||
|
7. **`virtio-gpu`'s "standalone bring-up" — resolved; there is no such path.** An
|
||||||
|
earlier version of this question read the comment "Best-effort: standalone bring-up
|
||||||
|
has no manager" as a boot path that delegation would delete. It is not one.
|
||||||
|
`virtio-gpu`'s `main` requires `argv[1]` and exits without it, and the only source of
|
||||||
|
that argument is the device manager — the chain is ACPI → pci-bus reports the
|
||||||
|
function → `devices.csv` matches `1AF4:1050` → the manager spawns the driver with the
|
||||||
|
id. There is no way to run it without a manager, so nothing is lost.
|
||||||
|
|
||||||
|
What the comment is really about is **resilience**: the hello is best-effort so a
|
||||||
|
driver whose manager has *died* keeps serving, which device-manager.md states
|
||||||
|
("If the manager dies, drivers keep running").
|
||||||
|
|
||||||
|
The residual is one narrow window: the manager spawns a driver and dies before
|
||||||
|
transferring. Today the driver could still claim, because claiming is free-for-all;
|
||||||
|
after D6 it exits and the restarted manager respawns it. That is the better
|
||||||
|
behaviour — a driver holding hardware nobody assigned it is what D6 exists to stop.
|
||||||
|
|
||||||
|
Until these are answered, `device_claim` cannot be closed off at D6 for those three, so
|
||||||
|
**D6 is blocked on questions 6 and 7**.
|
||||||
|
|
||||||
|
8. **Removing zero-resource devices from the kernel needs someone else to mint their
|
||||||
|
ids, and nothing says who.** A USB interface is registered with
|
||||||
|
`resource_count = 0`, and the id `device_register` returns is load-bearing in three
|
||||||
|
places: it is the `child_added` packet's target, it is the class driver's `argv[1]`,
|
||||||
|
and it is the `device_token` of the **usb-transfer wire protocol** — so the id space
|
||||||
|
is visible on the wire, not merely internal. usb-xhci-bus's own comment records the
|
||||||
|
fourth constraint: the kernel's idempotency is what makes "the same port and
|
||||||
|
interface always map back to the same device id" across a bus restart, which is what
|
||||||
|
stops a respawned bus spawning duplicate class drivers.
|
||||||
|
|
||||||
|
So the mover has to answer: who mints the id, how it stays stable across a *bus*
|
||||||
|
restart, how it stays stable across a *manager* restart (open question 3 territory),
|
||||||
|
and whether the wire protocol's `device_token` changes meaning. That is a design
|
||||||
|
step, not a mechanical one.
|
||||||
|
|
||||||
|
**D8 is reordered to run after D9**, which is a correction to the original sequencing.
|
||||||
|
D8's justification was "the authorisation it stood in for exists" — but with D6 blocked
|
||||||
|
it does not, so deleting the shared per-parent cap now would reopen the exhaustion hole
|
||||||
|
it was written for. D9's **per-holder quota** closes that hole independently of
|
||||||
|
authorisation, and does it better: a rogue exhausts its own allowance rather than the
|
||||||
|
table everyone shares. Once the quota exists, the per-parent cap is redundant whether or
|
||||||
|
not D6 has landed.
|
||||||
|
|
||||||
|
D9 also no longer depends on D7. Its original rationale was that zero-resource children
|
||||||
|
are the case that sidesteps containment — true, but a per-holder quota bounds them just
|
||||||
|
as well as anything else, because it counts entries per holder rather than per parent.
|
||||||
|
|
||||||
|
### Settled, so the run does not re-litigate them
|
||||||
|
|
||||||
|
- **The manager claims, it is not granted.** No binary names in the kernel; the rule is
|
||||||
|
"you may give away what you hold". The residual — it rests on the manager claiming
|
||||||
|
first — is stated in the design and is closed later by the spawn capability
|
||||||
|
[drivers.md](device-driver-development/drivers.md) already names as missing.
|
||||||
|
- **The framebuffer is not a device.** It is where pixels go, handed over by the loader,
|
||||||
|
and the compositor uses it as the boot floor until a real display driver announces
|
||||||
|
itself. Nothing in this run touches the display service or its GOP path.
|
||||||
|
- **`maximum_endpoints_per_interface` and the wire structs stay.** Widening them is a
|
||||||
|
protocol change, out of scope.
|
||||||
|
- **A per-holder quota is not a retreat.** Dynamic storage with no bound moves the
|
||||||
|
ceiling to the kernel heap, which is shared and fatal rather than partial. A bound
|
||||||
|
charged to the task that caused it is isolation, and it is declared through
|
||||||
|
[bounds.md](os-development/bounds.md) like anything else.
|
||||||
|
|
||||||
|
### Working rules
|
||||||
|
|
||||||
|
As Run 1, unchanged: work in `/Users/danielsamson/Gitea/daniel/danos` on
|
||||||
|
`claude/bounds-track`; every step lands with a test that fails before the fix, verified
|
||||||
|
by restoring the old behaviour; full suite green before each commit; never two suites at
|
||||||
|
once (`pgrep -f qemu_test.py`); 60 GiB free; `git commit -F` with no `Co-Authored-By`;
|
||||||
|
update this table before starting the next step. **If a step needs a decision that is
|
||||||
|
not written down, stop it, add the question below, and move on** — Run 1's first step
|
||||||
|
was wrong and stopping was the right call.
|
||||||
|
|
||||||
|
**Suite:** 115/115 at the start of the run.
|
||||||
|
**Branch:** `claude/bounds-track`.
|
||||||
|
|
||||||
|
### What this run deliberately does not touch
|
||||||
|
|
||||||
|
Phases 2 and 3 below — the authorisation gate and moving the inventory to the device
|
||||||
|
manager — are **out of scope for unattended work**. They decide whether the OS is
|
||||||
|
secure, and they are currently a direction rather than a specification: what a device
|
||||||
|
capability *is*, which syscalls change, what replaces `device_claim` for its seven
|
||||||
|
callers, how a driver spawned bare behaves. Those want a design session, the way
|
||||||
|
`/protocol` had one.
|
||||||
|
|
||||||
|
Also out of scope: anything touching `maximum_device_resources` (a wire struct, so a
|
||||||
|
trust-boundary change, not a resize), and the non-device bounds the audit found in FAT,
|
||||||
|
the VFS, logger, init, display and boot.
|
||||||
|
|
||||||
|
### Open questions this run must not answer on its own
|
||||||
|
|
||||||
|
Recorded here rather than guessed. If a step runs into one, it stops and writes the
|
||||||
|
question down instead of inventing an answer.
|
||||||
|
|
||||||
|
1. **`device_enumerate` probably narrows rather than retires.** The device manager
|
||||||
|
calls it to find `pci_host_bridge` nodes — it cannot ask itself. The likely shape is
|
||||||
|
that the kernel keeps the *firmware-discovered roots* (which by principle 5 it holds
|
||||||
|
for real reasons, since they come from ACPI rather than a driver's say-so) and
|
||||||
|
everything a driver registered lives in the manager. Not decided.
|
||||||
|
2. **A device-manager restart has no re-enumerate handshake.** If only the manager
|
||||||
|
dies, the buses are alive and never re-send `child_added`, so a restarted manager
|
||||||
|
comes back blind. The manager is restartable by design; nothing implements this.
|
||||||
|
3. **Which adversarial tests I1–I3 need.** The audit's six real defects were all found
|
||||||
|
by asking what an attacker would do, and the suite had never asked. "Add adversarial
|
||||||
|
cases" is not executable until the attacks are named.
|
||||||
|
4. **Reclamation is not a death-sweep problem, and L1 as written would have broken the
|
||||||
|
restart path.** Found on the first attempt at it. The audit is right that `count`
|
||||||
|
never decreases, but *death is the wrong trigger*:
|
||||||
|
- The broker keeps entries deliberately: "The devices stay in the table — they
|
||||||
|
describe hardware, which did not go away — only their ownership clears." A driver
|
||||||
|
dying does not unplug anything.
|
||||||
|
- Device ids must stay **stable across a bus restart**, because
|
||||||
|
`device-manager.driverForDevice` dedupes by `device_id` so that "a re-report after
|
||||||
|
a bus restart must not spawn a second instance". Stability comes from the
|
||||||
|
idempotency scan returning the existing id — removing entries on death would give
|
||||||
|
a restarted bus fresh ids and spawn duplicate driver instances.
|
||||||
|
- Everything else a task holds *is* already reclaimed on every path out:
|
||||||
|
`irq.releaseOwner`, `iommu.releaseAllOwnedBy`, `dmaRegistryReleaseOwner`, then the
|
||||||
|
broker's claims (`process.releaseTaskResourcesLocked`).
|
||||||
|
|
||||||
|
So the real leak has two sources, and neither is death: a device that genuinely
|
||||||
|
**goes away** (hot-unplug) has no retirement path, and a bus that enumerates
|
||||||
|
*differently* on restart leaves its stale entries behind forever. Both are the device
|
||||||
|
manager's inventory problem — phase 3 — and both need the id-stability question
|
||||||
|
answered first (tombstone-and-reuse aliases stale ids held by another process;
|
||||||
|
generation-tagged ids change the id encoding, which is ABI). Not an unattended
|
||||||
|
decision.
|
||||||
|
5. **Driver descriptor parsing cannot be host-tested, so L5's correctness fix ships
|
||||||
|
without a direct test.** The endpoint-misattribution bug lives in
|
||||||
|
`parseConfiguration`, a pure function over a byte blob — exactly the shape a host
|
||||||
|
unit test wants, and `usb-storage/scsi.zig` and `usb-hid/hid-report.zig` already do
|
||||||
|
this. But `usb-xhci-library.zig` imports `memory`, `mmio` and `time`, so it cannot
|
||||||
|
be a standalone host-test root, and QEMU offers no device that would exercise the
|
||||||
|
path anyway: the largest available is `usb-audio,multi=on` at 2 interfaces and 211
|
||||||
|
bytes, against a cap of 4.
|
||||||
|
|
||||||
|
Three ways out, and picking one is a judgement about house style rather than a
|
||||||
|
mechanical step: extract the parser to its own file and wire `usb-abi` into a test
|
||||||
|
module (build-support currently resolves module names only for `userBinary`);
|
||||||
|
extract it and import `usb-abi` by relative path (against the import-by-name
|
||||||
|
convention); or accept QEMU-only coverage and say so.
|
||||||
|
|
||||||
|
Mitigating, and the reason this is recorded rather than blocking: after the fix the
|
||||||
|
bug is unreachable **by construction**, not by the added `else`. Interfaces are now
|
||||||
|
allocated to exactly the count the descriptor declares, so `interface_count` can
|
||||||
|
never reach `interfaces.len` mid-parse. The `else` is belt-and-braces for the
|
||||||
|
255-interface clamp. The alternate-setting path that shares it *is* exercised —
|
||||||
|
`usb-audio` has alternate settings, and the `usb-large-descriptor` case walks them.
|
||||||
|
|
||||||
|
### Working rules for the run
|
||||||
|
|
||||||
|
- Work in `/Users/danielsamson/Gitea/daniel/danos` (not a worktree), on
|
||||||
|
`claude/bounds-track`.
|
||||||
|
- **Every step lands with a test that fails before the fix**, verified by temporarily
|
||||||
|
restoring the old behaviour and watching exactly the intended assertion flip. A test
|
||||||
|
that passes both ways is not a test.
|
||||||
|
- Full QEMU suite green before each commit. Never run two suites at once — check
|
||||||
|
`pgrep -f qemu_test.py` first; a second concurrent run produces false triple faults
|
||||||
|
because both share `zig-out`.
|
||||||
|
- Check at least 60 GiB free before starting a suite.
|
||||||
|
- Commit with `git commit -F <file>`, never `-m` (a backtick in a message is executed
|
||||||
|
by the shell and silently eats a word). No `Co-Authored-By` trailers.
|
||||||
|
- Update the Live state table **before** starting the next step.
|
||||||
|
- If a step needs a decision that is not written down here, stop, add it to the open
|
||||||
|
questions above, and move to the next step.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
## The principles this is derived from
|
## The principles this is derived from
|
||||||
|
|
||||||
1. **danOS is a microkernel.** Minimise what the kernel is responsible for; move
|
1. **danOS is a microkernel.** Minimise what the kernel is responsible for; move
|
||||||
|
|||||||
@@ -1,173 +1,151 @@
|
|||||||
# Device authority: the kernel stops keeping an inventory
|
# Device authority: the implementation of delegation
|
||||||
|
|
||||||
*Design, drafted 2026-08-07. Not implemented. Prompted by a real machine: an AMD
|
*Implementation design, 2026-08-08. The **what** is settled in
|
||||||
Ryzen desktop enumerated more PCI functions than the kernel's device table would
|
[device-manager.md](../device-driver-development/device-manager.md) — "structure in the
|
||||||
hold, and the xHCI and SATA controllers were refused registration — so the
|
manager, authority in the kernel", and delegation as the step after `hello`. This
|
||||||
machine booted to the compositor with no USB and no storage.*
|
document is the **how**, and the decisions that paragraph leaves open.*
|
||||||
|
|
||||||
The kernel keeps a table of every device userspace discovers. It is a fixed
|
Read first: [drivers.md](../device-driver-development/drivers.md) (the claim is the
|
||||||
array of 64 descriptors, 344 bytes each, and a second cap allows any one parent
|
capability), [driver-model.md](../device-driver-development/driver-model.md) (the three
|
||||||
16 children. Neither number is written down anywhere as a decision:
|
invariants), [device-manager.md](../device-driver-development/device-manager.md) (the
|
||||||
`maximum_devices` has no comment and never reached `parameters.zig`, where every
|
tree, the matcher, the supervisor).
|
||||||
other tunable in this kernel lives with its reasoning attached.
|
|
||||||
|
|
||||||
Raising them is not the fix. The numbers are wrong because the *table* is wrong:
|
## The one thing not yet true
|
||||||
it is an inventory of hardware, and an inventory of hardware is not something a
|
|
||||||
kernel needs. This document proposes replacing it with capabilities, which
|
|
||||||
removes the ceiling rather than moving it.
|
|
||||||
|
|
||||||
## What the kernel actually uses
|
`device-manager.md` says assignment "stays argv for now", and names the next step:
|
||||||
|
|
||||||
Every read of a device descriptor from the kernel proper, exhaustively:
|
> The step after `hello` exists is delegation: the manager claims (or is granted) the
|
||||||
|
> devices and passes the claim to the driver over IPC (the M13 capability-transfer
|
||||||
|
> mechanism), replacing first-come-first-served `device_claim` with policy.
|
||||||
|
|
||||||
| Used for | What it needs |
|
Until that lands, the manager's matching is advisory. `device_claim` checks only that
|
||||||
|---|---|
|
the device exists and is unheld ([devices-broker.zig](../../system/kernel/devices-broker.zig)):
|
||||||
| `mmio_map`, `io_read`/`io_write` | the physical range, to check the mapping falls inside it |
|
|
||||||
| `irq_bind`, `msi_bind` | the GSI, and that the caller owns the device |
|
|
||||||
| `dma_bind` | that the caller owns the device |
|
|
||||||
| IOMMU confinement | the **PCI BDF**, to key a domain |
|
|
||||||
| `device_register` | the parent's ranges, for the containment check |
|
|
||||||
| the boot display seed | one framebuffer window |
|
|
||||||
|
|
||||||
That is: **physical ranges, interrupt numbers, and one BDF.** Vendor, device and
|
```zig
|
||||||
subsystem ids, class triples, the human-readable names, the parent links, the
|
pub fn claim(id: u64, owner: u32) ClaimError!void {
|
||||||
bus numbers — the kernel stores all of it and reads none of it. It is held so
|
if (id >= count) return error.NoSuchDevice;
|
||||||
that `device_enumerate` can hand it back to user space, which is the whole
|
if (claimed[@intCast(id)] != null) return error.AlreadyClaimed;
|
||||||
mistake in one sentence: the kernel is acting as a distribution mechanism for
|
claimed[@intCast(id)] = owner;
|
||||||
data it does not use.
|
}
|
||||||
|
```
|
||||||
|
|
||||||
## The split
|
A driver is spawned with its device id in `argv[1]` and claims it; any process could
|
||||||
|
pass any integer instead. Since a claim is a licence to map physical memory, that is the
|
||||||
|
gap this document closes.
|
||||||
|
|
||||||
Three concerns are tangled in one table.
|
## Decision 1: the manager claims, then transfers
|
||||||
|
|
||||||
**Platform bring-up** — timers, LAPIC/IOAPIC, CPU topology, the ECAM window, the
|
`device-manager.md` leaves "claims (or is granted)" open. **Claims.**
|
||||||
framebuffer the firmware left. The kernel derives these from ACPI before user
|
|
||||||
space exists and needs them to function. They never belonged in the device table
|
|
||||||
and mostly are not (the platform block in `kernel.zig` is separate); this
|
|
||||||
document does not change them.
|
|
||||||
|
|
||||||
**Resource authority** — which task may map which physical range, receive which
|
The manager runs before any driver exists — `init` starts it from `init.csv`, and it is
|
||||||
interrupt, touch which ports. This *must* stay in the kernel. It is the one
|
what spawns drivers — so it takes the seeded devices unopposed and there is nothing
|
||||||
grant that cannot be audited after the fact: a process that maps arbitrary
|
unheld left for anyone to race for. One new call moves ownership on:
|
||||||
physical memory owns the machine, page tables and IOMMU structures included.
|
|
||||||
This is memory protection, not device management, and it is why the answer is
|
|
||||||
not simply "move it all to the device manager".
|
|
||||||
|
|
||||||
**Device inventory** — what exists, what it is, how it is arranged, which driver
|
```
|
||||||
should bind it. This is `device-manager`'s job and is already half there: it
|
device_transfer(device_id, task_id) -> 0/-errno
|
||||||
loads `devices.csv`, matches, spawns drivers, and receives `child_added` reports
|
```
|
||||||
over its own protocol. The kernel table duplicates what those reports already
|
|
||||||
carry.
|
|
||||||
|
|
||||||
## The proposal: a resource is a capability
|
The kernel checks only that the caller currently holds the device. No names, no
|
||||||
|
attestation, no notion of "the device manager" — the rule is *you may give away what you
|
||||||
|
hold*, which is the capability discipline already in force.
|
||||||
|
|
||||||
Device resources join endpoints, shared memory and DMA regions as a kind in the
|
The alternative was the kernel granting roots to a task it recognises by binary path
|
||||||
handle table.
|
plus a PID-1 supervisor. It is more robust — it does not depend on the manager being
|
||||||
|
first — but it puts a binary name inside the kernel, and the goal is that the kernel
|
||||||
|
keeps only what cannot safely live in user space. A name is not that.
|
||||||
|
|
||||||
1. **Roots.** At boot the kernel mints capabilities for the windows it learned
|
**The residual, stated plainly:** authority here rests on the manager claiming first.
|
||||||
from firmware — the ECAM range, the framebuffer, the legacy port space — and
|
That holds because `init.csv` decides what starts and in what order, so it is an
|
||||||
hands them to the first bus drivers. This is the only place device knowledge
|
operator-visible ordering rather than an attacker-controlled one — but it is an
|
||||||
enters the kernel, and it comes from ACPI, not from a driver's say-so.
|
assumption, not an enforced invariant. The enforced version arrives with the spawn
|
||||||
2. **Subdivision.** A bus driver enumerating hardware derives a narrower
|
capability [drivers.md](../device-driver-development/drivers.md) already names as
|
||||||
capability from one it holds: `resource_derive(cap, kind, start, len) → cap`.
|
missing ("`system_spawn` is currently ungated … because there is no spawn capability
|
||||||
The kernel checks the sub-range lies inside the capability being subdivided —
|
yet"). This design is compatible with it and does not block on it.
|
||||||
the same containment rule as today (`devices-broker.contains`), but checked
|
|
||||||
against *one capability the caller demonstrably holds* rather than by walking
|
|
||||||
a global tree.
|
|
||||||
3. **Delegation.** The driver passes that capability to the child driver over
|
|
||||||
IPC. Cap-passing already exists; this is the mechanism `subscribe` and
|
|
||||||
`attach_scanout` already use.
|
|
||||||
4. **Use.** `mmio_map`, `irq_bind`, `msi_bind`, `io_read`/`io_write` and
|
|
||||||
`dma_bind` take a capability handle instead of `(device_id, resource_index)`.
|
|
||||||
Possession *is* the authority — there is nothing to look up and no ownership
|
|
||||||
table to consult.
|
|
||||||
|
|
||||||
Exclusivity stops being a broker refusing a second claimant and becomes the
|
## Decision 2: the kernel stops holding inventory
|
||||||
ordinary property of a capability: only one process was given it.
|
|
||||||
|
|
||||||
## What this buys
|
The kernel reads exactly three things out of a descriptor: **physical ranges** (to check
|
||||||
|
a mapping falls inside one), **interrupt numbers**, and **one PCI BDF** (to key an IOMMU
|
||||||
|
domain). Vendor, device and subsystem ids, class triples, `_HID` strings, bus addresses
|
||||||
|
and names are stored only so `device_enumerate` can hand them back — which
|
||||||
|
`device-manager.md` already resolves: that call "fades to a manager-internal (then
|
||||||
|
deleted) seam", because the manager owns the tree as data.
|
||||||
|
|
||||||
**No ceiling.** There is no table to size, so no machine is too big. The
|
So the kernel's table becomes: **parent, resources, holder, BDF.** That is what cannot
|
||||||
Ryzen's enumeration stops being a limit to tune and becomes what it is — a fact
|
safely run in user space; the rest moves.
|
||||||
about a computer.
|
|
||||||
|
|
||||||
**Reclamation, free.** `count` in the broker today only ever increases;
|
**Devices with no resources leave the kernel entirely.** A USB device is addressed
|
||||||
`releaseAllOwnedBy` clears a dead driver's *claims* but never its entries. A
|
through its controller and carries `resource_count = 0`
|
||||||
driver that crashes and is restarted re-registers its children and consumes the
|
([driver-model.md](../device-driver-development/driver-model.md): "that case is allowed
|
||||||
table again — reachable today without any malice, given the device manager
|
and is the common one"). It conveys no mapping authority, so there is nothing for the
|
||||||
restarts drivers by design. Capabilities die with the task.
|
kernel to enforce and no reason for it to know. It is inventory, and inventory is the
|
||||||
|
manager's — reported by `child_added`, which already carries everything needed.
|
||||||
|
|
||||||
**No quota needed.** The per-parent cap exists to stop one claimant looping
|
That is also the case that made `maximum_children_per_parent` necessary: a zero-resource
|
||||||
`device_register` and filling the shared table, because a zero-resource child
|
child sidesteps containment, so a driver could loop `device_register` and fill the
|
||||||
sidesteps the containment check. With no shared table there is nothing to
|
shared table. Once such children are not kernel objects, every remaining entry is a real
|
||||||
exhaust; a process can only ever subdivide what it was given, and its handles
|
contained subdivision of something the caller holds.
|
||||||
are already bounded per task.
|
|
||||||
|
|
||||||
**A smaller kernel.** Three syscalls leave the ABI, one narrower one arrives,
|
## Decision 3: no shared ceiling; a per-holder quota instead
|
||||||
and `devices-broker.zig` largely disappears along with both constants.
|
|
||||||
|
|
||||||
**The discipline the rest of the system already uses.** "The claim is the
|
`maximum_devices = 64` and `maximum_children_per_parent = 16` are numbers we invented,
|
||||||
capability" is written in the driver documentation as if it were already true.
|
and both are shared — one driver's enumeration starves every other driver, which is how
|
||||||
This makes it true.
|
an AMD Ryzen booted with no USB and no storage.
|
||||||
|
|
||||||
## Syscall surface
|
- **The table becomes dynamic.** It is built after `heap.init` (`kernel.zig`: `pmm.init`
|
||||||
|
at 137, `heap.init` at 179, `devices_broker.init` at 202), so nothing prevents it. No
|
||||||
|
specification bounds how many devices a machine has, so nothing should bound ours.
|
||||||
|
- **`maximum_children_per_parent` is deleted**, because the authorisation it stood in
|
||||||
|
for now exists.
|
||||||
|
- **A per-holder quota replaces them.** Dynamic storage without a bound moves the
|
||||||
|
ceiling to the kernel heap, which is shared and fatal rather than partial — strictly
|
||||||
|
worse. The bound that is *not* worse is one charged to the task that caused it: a
|
||||||
|
driver that loops `device_register` exhausts its own allowance, is refused with an
|
||||||
|
attributable errno, and is restarted by its supervisor while every other driver
|
||||||
|
carries on. That is the microkernel property rather than a workaround for it, and it
|
||||||
|
is declared through [bounds.md](bounds.md) like any other.
|
||||||
|
|
||||||
- `device_enumerate` — **retires.** It exists only to read the kernel's table.
|
## The shape of the change
|
||||||
Callers ask `device-manager`, whose protocol already reserves an `enumerate`
|
|
||||||
verb. Note this is a public-ABI change: `vdso.md` documents it.
|
|
||||||
- `device_register` — **splits.** The kernel half becomes `resource_derive`; the
|
|
||||||
publication half ("this device exists, here is what it is") becomes an IPC
|
|
||||||
message to `device-manager`, which is where the inventory belongs and where
|
|
||||||
`child_added` already carries the same facts.
|
|
||||||
- `device_claim` — **dissolves into possession**, except for the IOMMU (below).
|
|
||||||
- `mmio_map`, `irq_bind`, `msi_bind`, `io_read`, `io_write`, `dma_bind` — keep
|
|
||||||
their names and semantics; their first argument becomes a capability handle.
|
|
||||||
|
|
||||||
## The open question: where the IOMMU attaches
|
| | Before | After |
|
||||||
|
|---|---|---|
|
||||||
|
| Manager gets its devices | claims them, unauthorised | claims them (first, unopposed) |
|
||||||
|
| Driver gets its device | `argv[1]` + `device_claim` | receives it in the `hello` reply |
|
||||||
|
| Kernel checks | is it free? | do you hold it? |
|
||||||
|
| Kernel stores | the full descriptor | parent, resources, holder, BDF |
|
||||||
|
| Zero-resource devices | kernel table entries | manager records only |
|
||||||
|
| Table size | `maximum_devices = 64` | dynamic, per-holder quota |
|
||||||
|
| Children per parent | `maximum_children_per_parent = 16` | deleted |
|
||||||
|
|
||||||
This is the one place the kernel still needs device *identity* rather than a
|
Bring-up order changes for the five claiming drivers: `hello` must precede the claim,
|
||||||
range. `confineDevice(device_id, bdf, owner)` builds a domain keyed by PCI BDF,
|
because the reply is where the device arrives. `pci-bus` today does the reverse — its
|
||||||
attaches it, and the claim is rolled back if confinement fails — deliberately,
|
own comment reads "Claim the bridge, map the ECAM, hello the manager, then scan."
|
||||||
so a device that cannot be confined is never driven.
|
|
||||||
|
|
||||||
Three options, none obviously right:
|
## What does not change
|
||||||
|
|
||||||
1. **The BDF rides the capability.** A memory capability derived for a PCI
|
- The three invariants of [driver-model.md](../device-driver-development/driver-model.md):
|
||||||
function carries its BDF, and the kernel confines on first `mmio_map` or
|
a claim is exclusive, a descriptor is a licence to map physical memory, therefore
|
||||||
`dma_bind`. Keeps the syscall count down; means a capability is no longer
|
containment. This design strengthens the first and touches neither of the others.
|
||||||
purely a range.
|
- The display service's GOP path. The framebuffer is not a device — it is where pixels
|
||||||
2. **An explicit `device_attach(cap, bdf)`.** Honest and visible, but it puts a
|
go, handed over by the loader, and the compositor uses it as the boot floor until a
|
||||||
BDF — a fact about PCI — into a kernel interface that otherwise knows nothing
|
real display driver announces itself
|
||||||
about buses, and something must stop a caller naming a BDF that is not
|
([display-v2.md](../device-driver-development/display-v2.md)). The kernel wraps it in
|
||||||
theirs.
|
a display-class descriptor so `mmio_map` can hand it over write-combining; that is
|
||||||
3. **The root PCI capability carries the segment, and derivation computes the
|
plumbing for a mapping, not a claim about what it is.
|
||||||
BDF.** Purest, but only works for PCI and the kernel would be parsing bus
|
- Supervision, restart, pruning and re-report
|
||||||
topology, which is precisely what this document is trying to stop.
|
([device-manager.md](../device-driver-development/device-manager.md),
|
||||||
|
[process-lifecycle.md](process-lifecycle.md)). Delegation slots into the existing
|
||||||
|
`hello` exchange and changes none of it.
|
||||||
|
- `device_register` idempotency, which is what lets a restarted bus rebuild the same
|
||||||
|
ids.
|
||||||
|
|
||||||
My inclination is (1), because confinement is a property of the resource being
|
## How it is verified
|
||||||
granted rather than a separate action, and because it keeps the "possession is
|
|
||||||
authority" story intact. It needs the derivation call to know it is carving a
|
|
||||||
PCI function, which is a wart worth arguing about.
|
|
||||||
|
|
||||||
## What this does not solve
|
The invariant is: **a process holds what it was handed and cannot name its way into
|
||||||
|
holding more.** The suite has no adversarial device case today — the audit's lesson was
|
||||||
|
that "the suite contains no attacker" — so this adds one: a process that was handed
|
||||||
|
nothing calls `device_claim` and `device_transfer` on a device another driver holds, and
|
||||||
|
on one nobody holds, and is refused each time with its own errno.
|
||||||
|
|
||||||
- **Hot-plug and removal.** Capabilities die with their holder, but a device
|
The Ryzen is the acceptance test for the ceiling half: it is the machine that found the
|
||||||
that physically disappears while a driver lives is the device manager's
|
constants, and the one that proves them gone.
|
||||||
problem and unchanged by this.
|
|
||||||
- **Quotas on physical memory.** Nothing here bounds how much a driver maps; it
|
|
||||||
bounds only *what* it may map. That was already true.
|
|
||||||
- **The device tree as a published thing.** `/system/devices` remains a device
|
|
||||||
manager concern, as the file-system hierarchy already assumes.
|
|
||||||
|
|
||||||
## Migration sketch
|
|
||||||
|
|
||||||
Not a plan yet — the phases need sizing once the IOMMU question is settled.
|
|
||||||
Rough shape: introduce the capability kind and `resource_derive` alongside the
|
|
||||||
existing table; convert the resource-consuming syscalls to accept either form;
|
|
||||||
move the inventory into `device-manager` and convert its clients off
|
|
||||||
`device_enumerate`; then delete the table, the two constants and the three
|
|
||||||
syscalls in one flag-day, as the `ServiceId` retirement did.
|
|
||||||
|
|
||||||
Every driver is affected, so the QEMU suite is the arbiter at each step, and the
|
|
||||||
Ryzen is the acceptance test — it is the machine that found this, and the one
|
|
||||||
that proves it fixed.
|
|
||||||
|
|||||||
@@ -46,12 +46,31 @@ pub fn enumerate(buffer: []DeviceDescriptor) usize {
|
|||||||
return sc.systemCall2(.device_enumerate, @intFromPtr(buffer.ptr), buffer.len);
|
return sc.systemCall2(.device_enumerate, @intFromPtr(buffer.ptr), buffer.len);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Why a `transfer` failed. `NotHeld` is the interesting one — it means the caller tried
|
||||||
|
/// to give away a device it does not have, which is the whole rule.
|
||||||
|
pub const TransferError = error{ NoSuchDevice, NotHeld, NoSuchTask, Refused };
|
||||||
|
|
||||||
|
/// Give device `id` to task `to`. **A move, not a copy** — a claim is exclusive, so the
|
||||||
|
/// caller stops holding it. This is how the device manager hands a driver the device it
|
||||||
|
/// matched, replacing first-come-first-served claiming with policy
|
||||||
|
/// (docs/os-development/device-authority.md).
|
||||||
|
pub fn transfer(id: u64, to: u32) TransferError!void {
|
||||||
|
const r = sc.systemCall2(.device_transfer, id, to);
|
||||||
|
if (!failed(r)) return;
|
||||||
|
return switch (errnoOf(r)) {
|
||||||
|
abi.ENODEV => error.NoSuchDevice,
|
||||||
|
abi.EPERM => error.NotHeld,
|
||||||
|
abi.ESRCH => error.NoSuchTask,
|
||||||
|
else => error.Refused,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
/// Why a `claim` failed. Worth distinguishing: `AlreadyClaimed` means back off and
|
/// Why a `claim` failed. Worth distinguishing: `AlreadyClaimed` means back off and
|
||||||
/// let the owner have it, `NoSuchDevice` means this id is stale and the caller should
|
/// let the owner have it, `NoSuchDevice` means this id is stale and the caller should
|
||||||
/// re-enumerate, and `NotConfined` means the machine could not place the device under
|
/// re-enumerate, and `NotConfined` means the machine could not place the device under
|
||||||
/// IOMMU translation — the claim was rolled back, and that one is a fault report, not
|
/// IOMMU translation — the claim was rolled back, and that one is a fault report, not
|
||||||
/// a retry. `Refused` is an errno this library does not know a name for.
|
/// a retry. `Refused` is an errno this library does not know a name for.
|
||||||
pub const ClaimError = error{ NoSuchDevice, AlreadyClaimed, NotConfined, Refused };
|
pub const ClaimError = error{ NoSuchDevice, AlreadyClaimed, NotConfined, NotYours, Refused };
|
||||||
|
|
||||||
/// Take exclusive ownership of device `id`.
|
/// Take exclusive ownership of device `id`.
|
||||||
pub fn claim(id: u64) ClaimError!void {
|
pub fn claim(id: u64) ClaimError!void {
|
||||||
@@ -61,6 +80,7 @@ pub fn claim(id: u64) ClaimError!void {
|
|||||||
abi.ENODEV => error.NoSuchDevice,
|
abi.ENODEV => error.NoSuchDevice,
|
||||||
abi.EBUSY => error.AlreadyClaimed,
|
abi.EBUSY => error.AlreadyClaimed,
|
||||||
abi.ECONFINE => error.NotConfined,
|
abi.ECONFINE => error.NotConfined,
|
||||||
|
abi.EPERM => error.NotYours, // delegated hardware: it must be handed to you
|
||||||
else => error.Refused,
|
else => error.Refused,
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -182,6 +182,22 @@ pub fn spawnWithArguments(name: []const u8, arguments: []const []const u8) ?u32
|
|||||||
/// `ipc.replyWait` as a child-exit badge (`ipc.Received.isChildExit`/`childProcessId`), so
|
/// `ipc.replyWait` as a child-exit badge (`ipc.Received.isChildExit`/`childProcessId`), so
|
||||||
/// one endpoint can supervise many children. Returns the child's process id, or null.
|
/// one endpoint can supervise many children. Returns the child's process id, or null.
|
||||||
pub fn spawnSupervised(name: []const u8, arguments: []const []const u8, exit_endpoint: ?usize) ?u32 {
|
pub fn spawnSupervised(name: []const u8, arguments: []const []const u8, exit_endpoint: ?usize) ?u32 {
|
||||||
|
return spawnSupervisedWithDevice(name, arguments, exit_endpoint, abi.no_device);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Spawn a supervised child **and give it a device you hold**, in one call.
|
||||||
|
///
|
||||||
|
/// The device manager's path: it holds the hardware and the driver it starts must have
|
||||||
|
/// it. Fusing the handover into the spawn is what removes the window a separate
|
||||||
|
/// transfer would leave — the child cannot run without its device, because it does not
|
||||||
|
/// exist until it holds it (docs/os-development/device-authority.md). The kernel checks
|
||||||
|
/// only that the device is the caller's to give.
|
||||||
|
pub fn spawnSupervisedWithDevice(
|
||||||
|
name: []const u8,
|
||||||
|
arguments: []const []const u8,
|
||||||
|
exit_endpoint: ?usize,
|
||||||
|
device: u64,
|
||||||
|
) ?u32 {
|
||||||
var blob: [256]u8 = undefined;
|
var blob: [256]u8 = undefined;
|
||||||
var len: usize = 0;
|
var len: usize = 0;
|
||||||
for (arguments, 0..) |argument, i| {
|
for (arguments, 0..) |argument, i| {
|
||||||
@@ -194,7 +210,7 @@ pub fn spawnSupervised(name: []const u8, arguments: []const []const u8, exit_end
|
|||||||
@memcpy(blob[len..][0..argument.len], argument);
|
@memcpy(blob[len..][0..argument.len], argument);
|
||||||
len += argument.len;
|
len += argument.len;
|
||||||
}
|
}
|
||||||
const r = sc.systemCall5(.system_spawn, @intFromPtr(name.ptr), name.len, if (len == 0) 0 else @intFromPtr(&blob), len, exit_endpoint orelse abi.no_cap);
|
const r = sc.systemCall6(.system_spawn, @intFromPtr(name.ptr), name.len, if (len == 0) 0 else @intFromPtr(&blob), len, exit_endpoint orelse abi.no_cap, device);
|
||||||
if (r > ~@as(usize, 0) - 4095) return null; // a wrapped -errno
|
if (r > ~@as(usize, 0) - 4095) return null; // a wrapped -errno
|
||||||
return @intCast(r);
|
return @intCast(r);
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -65,3 +65,19 @@ pub inline fn systemCall5(n: SystemCall, a0: usize, a1: usize, a2: usize, a3: us
|
|||||||
[a4] "{r8}" (a4),
|
[a4] "{r8}" (a4),
|
||||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The sixth and last argument register. `syscall` clobbers rcx, so r10 stands in for
|
||||||
|
/// it and r9 is the end of the line — a call needing a seventh would have to pass a
|
||||||
|
/// struct instead.
|
||||||
|
pub inline fn systemCall6(n: SystemCall, a0: usize, a1: usize, a2: usize, a3: usize, a4: usize, a5: usize) usize {
|
||||||
|
return asm volatile ("syscall"
|
||||||
|
: [ret] "={rax}" (-> usize),
|
||||||
|
: [n] "{rax}" (@intFromEnum(n)),
|
||||||
|
[a0] "{rdi}" (a0),
|
||||||
|
[a1] "{rsi}" (a1),
|
||||||
|
[a2] "{rdx}" (a2),
|
||||||
|
[a3] "{r10}" (a3),
|
||||||
|
[a4] "{r8}" (a4),
|
||||||
|
[a5] "{r9}" (a5),
|
||||||
|
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||||
|
}
|
||||||
|
|||||||
+7
-1
@@ -48,7 +48,7 @@ pub const SystemCall = enum(u64) {
|
|||||||
irq_bind = 14, // irq_bind(id, resource_index, endpoint): deliver a device IRQ as an IPC notification
|
irq_bind = 14, // irq_bind(id, resource_index, endpoint): deliver a device IRQ as an IPC notification
|
||||||
irq_ack = 15, // irq_ack(id, resource_index): re-arm a bound IRQ after servicing it
|
irq_ack = 15, // irq_ack(id, resource_index): re-arm a bound IRQ after servicing it
|
||||||
device_register = 16, // device_register(parent_id, descriptor) -> id/-errno: publish a child of a device you claimed (-ENOSPC table full, -ECHILDREN parent full, -ERANGE resource escapes the parent, -ENODEV/-EPERM bad parent, -E2BIG too many resources)
|
device_register = 16, // device_register(parent_id, descriptor) -> id/-errno: publish a child of a device you claimed (-ENOSPC table full, -ECHILDREN parent full, -ERANGE resource escapes the parent, -ENODEV/-EPERM bad parent, -E2BIG too many resources)
|
||||||
system_spawn = 17, // system_spawn(name_ptr, name_len, arguments_ptr, arguments_len, exit_endpoint) -> child process id: start a named initial-ramdisk binary as a new ring-3 process
|
system_spawn = 17, // system_spawn(name_ptr, name_len, arguments_ptr, arguments_len, exit_endpoint, device) -> child process id: start a named initial-ramdisk binary as a new ring-3 process. `device` (or `no_device`) is a device the caller holds and gives to the child, atomically — the child never runs without it
|
||||||
dma_alloc = 18, // dma_alloc(len, flags) -> virtual_address (rax), physical_address (rdx): contiguous, pinned, uncacheable DMA memory
|
dma_alloc = 18, // dma_alloc(len, flags) -> virtual_address (rax), physical_address (rdx): contiguous, pinned, uncacheable DMA memory
|
||||||
dma_free = 19, // dma_free(virtual_address, len) -> 0: release a prior dma_alloc
|
dma_free = 19, // dma_free(virtual_address, len) -> 0: release a prior dma_alloc
|
||||||
msi_bind = 20, // msi_bind(device_id, endpoint) -> address (rax), data (rdx): a per-device MSI vector for a claimed device
|
msi_bind = 20, // msi_bind(device_id, endpoint) -> address (rax), data (rdx): a per-device MSI vector for a claimed device
|
||||||
@@ -85,6 +85,7 @@ pub const SystemCall = enum(u64) {
|
|||||||
dma_bind = 51, // dma_bind(device_id, region_handle) -> 0/-errno: map a DMA-region capability into the claimed device's IOMMU domain (idempotent). The caller must own the device and hold the handle
|
dma_bind = 51, // dma_bind(device_id, region_handle) -> 0/-errno: map a DMA-region capability into the claimed device's IOMMU domain (idempotent). The caller must own the device and hold the handle
|
||||||
dma_unbind = 52, // dma_unbind(device_id, region_handle) -> 0/-errno: unmap a previously bound region from the device's domain and invalidate
|
dma_unbind = 52, // dma_unbind(device_id, region_handle) -> 0/-errno: unmap a previously bound region from the device's domain and invalidate
|
||||||
handle_close = 53, // handle_close(handle) -> 0/-errno: drop one capability handle and free its table slot (endpoints, shared-memory, DMA regions)
|
handle_close = 53, // handle_close(handle) -> 0/-errno: drop one capability handle and free its table slot (endpoints, shared-memory, DMA regions)
|
||||||
|
device_transfer = 54, // device_transfer(device_id, task_id) -> 0/-errno: give a device you hold to another task. A MOVE, not a copy — a claim is exclusive (-ENODEV no such device, -EPERM you do not hold it, -ESRCH no such task)
|
||||||
_,
|
_,
|
||||||
};
|
};
|
||||||
|
|
||||||
@@ -126,6 +127,11 @@ pub const ECONFINE: i64 = 16; // the device could not be placed under IOMMU tran
|
|||||||
/// space, and the number to bump when adding one.
|
/// space, and the number to bump when adding one.
|
||||||
pub const errno_maximum: i64 = 16;
|
pub const errno_maximum: i64 = 16;
|
||||||
|
|
||||||
|
/// `system_spawn`'s `device` argument when the child is given no device — every
|
||||||
|
/// caller but the device manager. Matches `device-manager-protocol.no_device`, which
|
||||||
|
/// is the same sentinel one layer up.
|
||||||
|
pub const no_device: u64 = ~@as(u64, 0);
|
||||||
|
|
||||||
/// `futex_wait` return codes (in rax).
|
/// `futex_wait` return codes (in rax).
|
||||||
pub const futex_woken: u64 = 0; // woken by a futex_wake
|
pub const futex_woken: u64 = 0; // woken by a futex_wake
|
||||||
pub const futex_mismatch: u64 = 1; // *addr != expected on entry; the caller did not block
|
pub const futex_mismatch: u64 = 1; // *addr != expected on entry; the caller did not block
|
||||||
|
|||||||
@@ -75,11 +75,20 @@ fn configWrite16(bus: u64, dev: u64, function: u64, offset: u64, value: u16) voi
|
|||||||
configWrite(bus, dev, function, aligned, (word & ~mask) | (@as(u32, value) << shift));
|
configWrite(bus, dev, function, aligned, (word & ~mask) | (@as(u32, value) << shift));
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Claim the bridge, map the ECAM, hello the manager, then scan.
|
/// Hello the manager (which is where the bridge arrives), map the ECAM, then scan.
|
||||||
fn initialise(endpoint: ipc.Handle) bool {
|
fn initialise(endpoint: ipc.Handle) bool {
|
||||||
_ = endpoint;
|
_ = endpoint;
|
||||||
device.claim(bridge_id) catch |e| {
|
// **The handshake first, because it is where the device arrives.** This driver
|
||||||
std.log.info("unable to claim bridge device {d}: {s}", .{ bridge_id, @errorName(e) });
|
// used to claim `bridge_id` here — first-come-first-served, so the manager's
|
||||||
|
// matching was advisory and any process could have claimed the bridge by naming
|
||||||
|
// the same id. The manager now holds it and transfers it in the hello reply
|
||||||
|
// (docs/os-development/device-authority.md). `hello` is synchronous, so the
|
||||||
|
// transfer has completed by the time this returns.
|
||||||
|
//
|
||||||
|
// Keep the manager handle to report children through; a supervised bus that
|
||||||
|
// cannot reach its manager has nothing to serve.
|
||||||
|
manager_handle = device_manager.hello(.bus, bridge_id) orelse {
|
||||||
|
std.log.info("no hello with the device manager; bridge {d} not delegated", .{bridge_id});
|
||||||
return false;
|
return false;
|
||||||
};
|
};
|
||||||
const buffer = memory.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
const buffer = memory.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||||
@@ -113,11 +122,6 @@ fn initialise(endpoint: ipc.Handle) bool {
|
|||||||
return false;
|
return false;
|
||||||
};
|
};
|
||||||
|
|
||||||
// The handshake (role: bus — we enumerate PCI and report the functions we
|
|
||||||
// find), then the scan. Keep the manager handle to report children through;
|
|
||||||
// a supervised bus that cannot reach its manager has nothing to serve.
|
|
||||||
manager_handle = device_manager.hello(.bus, bridge_id) orelse return false;
|
|
||||||
|
|
||||||
scan();
|
scan();
|
||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -114,10 +114,9 @@ pub fn main() void {
|
|||||||
_ = logging.write("/system/drivers/ps2-bus: found PS/2 controller\n");
|
_ = logging.write("/system/drivers/ps2-bus: found PS/2 controller\n");
|
||||||
_ = logging.write("/system/drivers/ps2-bus: initializing controller\n");
|
_ = logging.write("/system/drivers/ps2-bus: initializing controller\n");
|
||||||
|
|
||||||
device.claim(controller_device_descriptor.id) catch |e| {
|
// The controller node arrived with the spawn — the manager holds the hardware
|
||||||
std.log.warn("unable to claim controller: {s}", .{@errorName(e)});
|
// and names it in the call that creates this process
|
||||||
return;
|
// (docs/os-development/device-authority.md). Nothing to claim.
|
||||||
};
|
|
||||||
|
|
||||||
const controller = ps2.Controller.init(controller_device_descriptor) orelse {
|
const controller = ps2.Controller.init(controller_device_descriptor) orelse {
|
||||||
_ = logging.write("/system/drivers/ps2-bus: controller is missing its IO ports\n");
|
_ = logging.write("/system/drivers/ps2-bus: controller is missing its IO ports\n");
|
||||||
@@ -255,18 +254,21 @@ pub fn main() void {
|
|||||||
if (port_device_types[@intFromEnum(ps2.Port.two)] != null) {
|
if (port_device_types[@intFromEnum(ps2.Port.two)] != null) {
|
||||||
if (ps2.findMouseDescriptor(buffer)) |descriptor| {
|
if (ps2.findMouseDescriptor(buffer)) |descriptor| {
|
||||||
if (findInterruptResourceIndex(descriptor)) |auxiliary_index| {
|
if (findInterruptResourceIndex(descriptor)) |auxiliary_index| {
|
||||||
if (device.claim(descriptor.id)) |_| {
|
// The mouse node is a *second* device for this one instance — the
|
||||||
if (device.irqBind(descriptor.id, auxiliary_index, endpoint)) {
|
// 8042 is one controller with two ports, so it cannot be split across
|
||||||
maybe_auxiliary_interrupt = .{
|
// two processes. It is transferred to us after the spawn, which is
|
||||||
.device_id = descriptor.id,
|
// safe because we only reach it here, long after the controller
|
||||||
.interrupt_index = auxiliary_index,
|
// handshakes and identify. irq_bind is the proof we hold it: it is
|
||||||
.gsi = descriptor.resources[auxiliary_index].start,
|
// gated on ownership, so a failure here means the handover has not
|
||||||
};
|
// landed rather than a hardware problem.
|
||||||
} else {
|
if (device.irqBind(descriptor.id, auxiliary_index, endpoint)) {
|
||||||
_ = logging.write("/system/drivers/ps2-bus: auxiliary irq_bind failed\n");
|
maybe_auxiliary_interrupt = .{
|
||||||
}
|
.device_id = descriptor.id,
|
||||||
} else |e| {
|
.interrupt_index = auxiliary_index,
|
||||||
std.log.warn("auxiliary claim failed: {s}", .{@errorName(e)});
|
.gsi = descriptor.resources[auxiliary_index].start,
|
||||||
|
};
|
||||||
|
} else {
|
||||||
|
_ = logging.write("/system/drivers/ps2-bus: auxiliary irq_bind failed (mouse node not delegated?)\n");
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -173,10 +173,22 @@ fn initialise(endpoint: ipc.Handle) bool {
|
|||||||
if (!channel.bindPatiently("usb-transfer", endpoint))
|
if (!channel.bindPatiently("usb-transfer", endpoint))
|
||||||
_ = logging.write("/system/drivers/usb-xhci-bus: /protocol/usb-transfer is another controller's; serving mine unnamed\n");
|
_ = logging.write("/system/drivers/usb-xhci-bus: /protocol/usb-transfer is another controller's; serving mine unnamed\n");
|
||||||
|
|
||||||
device.claim(controller_id) catch |e| {
|
// **The handshake comes first, because it is where the device arrives.** This
|
||||||
std.log.warn("unable to claim controller device {d}: {s}", .{ controller_id, @errorName(e) });
|
// driver used to claim `controller_id` here — first-come-first-served, so the
|
||||||
|
// manager's matching was advisory and any process could have claimed it by
|
||||||
|
// passing the same integer. Now the manager holds the controller and transfers
|
||||||
|
// it in `onHello`, so by the time this call returns the device is ours and
|
||||||
|
// nothing else could have taken it (docs/os-development/device-authority.md).
|
||||||
|
//
|
||||||
|
// `hello` is synchronous, so the transfer has completed before the reply lands —
|
||||||
|
// there is no window between being told yes and holding the thing.
|
||||||
|
//
|
||||||
|
// Keep the handle: the tick's hot-plug dispatch reports through it.
|
||||||
|
const handle = device_manager.hello(.bus, controller_id) orelse {
|
||||||
|
std.log.warn("no hello with the device manager; controller {d} not delegated", .{controller_id});
|
||||||
return false;
|
return false;
|
||||||
};
|
};
|
||||||
|
manager_handle = handle;
|
||||||
|
|
||||||
// Fetch our own descriptor back for the controller's resources.
|
// Fetch our own descriptor back for the controller's resources.
|
||||||
const buffer = memory.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
const buffer = memory.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||||
@@ -204,7 +216,7 @@ fn initialise(endpoint: ipc.Handle) bool {
|
|||||||
std.log.info("controller device {d} has no register BAR", .{controller_id});
|
std.log.info("controller device {d} has no register BAR", .{controller_id});
|
||||||
return false;
|
return false;
|
||||||
};
|
};
|
||||||
std.log.info("claimed controller device {d} (registers at 0x{x}, {d} bytes)", .{
|
std.log.info("controller device {d} registers at 0x{x}, {d} bytes", .{
|
||||||
controller_id,
|
controller_id,
|
||||||
register_window.start,
|
register_window.start,
|
||||||
register_window.len,
|
register_window.len,
|
||||||
@@ -228,8 +240,13 @@ fn initialise(endpoint: ipc.Handle) bool {
|
|||||||
_ = logging.write("/system/drivers/usb-xhci-bus: controller reset/bring-up failed\n");
|
_ = logging.write("/system/drivers/usb-xhci-bus: controller reset/bring-up failed\n");
|
||||||
return false;
|
return false;
|
||||||
};
|
};
|
||||||
std.log.info("controller running ({d} slots, {d}-byte contexts)", .{
|
// Both numbers, because for a long time they disagreed silently: the controller
|
||||||
|
// reported its real slot count and the driver tracked a fixed 8 of them, so a
|
||||||
|
// ninth device — trivially reachable behind a hub — simply did not exist. They
|
||||||
|
// must now be equal, and the suite asserts it.
|
||||||
|
std.log.info("controller running ({d} slots, tracking {d}, {d}-byte contexts)", .{
|
||||||
controller.?.max_slots,
|
controller.?.max_slots,
|
||||||
|
controller.?.devices.len,
|
||||||
controller.?.context_size,
|
controller.?.context_size,
|
||||||
});
|
});
|
||||||
// The proof of life: a No-Op command round-trips the command ring, the event
|
// The proof of life: a No-Op command round-trips the command ring, the event
|
||||||
@@ -242,12 +259,6 @@ fn initialise(endpoint: ipc.Handle) bool {
|
|||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
|
|
||||||
// The handshake (role: bus — we enumerate USB ports and report the devices
|
|
||||||
// behind them), inside the manager's hello deadline. Keep the handle: the
|
|
||||||
// tick's hot-plug dispatch reports through it.
|
|
||||||
const handle = device_manager.hello(.bus, controller_id) orelse return false;
|
|
||||||
manager_handle = handle;
|
|
||||||
|
|
||||||
scanPorts(handle);
|
scanPorts(handle);
|
||||||
|
|
||||||
// Arm the timer: in polling mode it drains the event ring; in MSI mode it is the
|
// Arm the timer: in polling mode it drains the event ring; in MSI mode it is the
|
||||||
|
|||||||
@@ -221,10 +221,17 @@ fn intervalFor(speed: u32, b_interval: u8) u32 {
|
|||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
// Upper bounds on what one device's active configuration describes. A boot
|
/// bound: endpoints recorded per interface
|
||||||
// keyboard or mouse has one interface with one interrupt endpoint; a flash drive
|
/// decided-by: external
|
||||||
// has one interface with two bulk endpoints. Generous for those.
|
/// protects: the fixed endpoint array inside InterfaceInfo
|
||||||
pub const max_interfaces = 4;
|
/// at-limit: refuse — the surplus endpoint is not recorded and the class driver
|
||||||
|
/// cannot bind it
|
||||||
|
/// observed-by: the class driver failing to find the endpoint it wants
|
||||||
|
///
|
||||||
|
/// Held at 4 because the usb-transfer wire protocol reports exactly
|
||||||
|
/// `max_reported_endpoints` (4) endpoints per interface, so widening this alone
|
||||||
|
/// would change nothing a class driver can see. Lifting it is a protocol change —
|
||||||
|
/// docs/bounds-track-plan.md, out of scope for the unattended run.
|
||||||
pub const max_endpoints_per_interface = 4;
|
pub const max_endpoints_per_interface = 4;
|
||||||
|
|
||||||
// The endpoint-descriptor facts a class driver needs to talk to an endpoint: its
|
// The endpoint-descriptor facts a class driver needs to talk to an endpoint: its
|
||||||
@@ -256,7 +263,19 @@ const ConfiguredEndpoint = struct {
|
|||||||
dci: u32 = 0,
|
dci: u32 = 0,
|
||||||
ring: ProducerRing = .{ .region = .{ .virtual = 0, .physical = 0 } },
|
ring: ProducerRing = .{ .region = .{ .virtual = 0, .physical = 0 } },
|
||||||
};
|
};
|
||||||
const max_configured_endpoints = max_interfaces * max_endpoints_per_interface;
|
/// bound: transfer rings configured on one device slot
|
||||||
|
/// decided-by: hardware
|
||||||
|
/// protects: the per-device ring table, sized at compile time
|
||||||
|
/// at-limit: refuse — configureEndpoint returns null and its caller reports it
|
||||||
|
/// observed-by: the class driver's own failure to bind an endpoint
|
||||||
|
///
|
||||||
|
/// The xHCI specification's number rather than ours: a Device Context carries a slot
|
||||||
|
/// context plus at most 31 endpoint contexts, because the Context Entries field that
|
||||||
|
/// addresses them is 5 bits (DCI 1..31, DCI 1 being the default control endpoint).
|
||||||
|
/// A device physically cannot present more. This was
|
||||||
|
/// `max_interfaces * max_endpoints_per_interface` = 16, so it moved whenever either
|
||||||
|
/// of those two guesses moved.
|
||||||
|
const max_configured_endpoints = 31;
|
||||||
|
|
||||||
// One addressed USB device behind this controller: its hardware slot, its EP0
|
// One addressed USB device behind this controller: its hardware slot, its EP0
|
||||||
// (control) transfer ring, the DMA context + bounce buffer the control pipe uses,
|
// (control) transfer ring, the DMA context + bounce buffer the control pipe uses,
|
||||||
@@ -277,7 +296,13 @@ pub const Device = struct {
|
|||||||
device_descriptor: usb_abi.DeviceDescriptor = std.mem.zeroes(usb_abi.DeviceDescriptor),
|
device_descriptor: usb_abi.DeviceDescriptor = std.mem.zeroes(usb_abi.DeviceDescriptor),
|
||||||
configuration_value: u8 = 0,
|
configuration_value: u8 = 0,
|
||||||
interface_count: u8 = 0,
|
interface_count: u8 = 0,
|
||||||
interfaces: [max_interfaces]InterfaceInfo = [_]InterfaceInfo{.{}} ** max_interfaces,
|
/// The interfaces of the active configuration, allocated at enumeration from the
|
||||||
|
/// count the configuration descriptor actually declares. This was a fixed 4, and
|
||||||
|
/// the fifth interface of a composite device (a headset, a webcam with audio, a
|
||||||
|
/// dock, a multifunction printer) did not merely go missing — see
|
||||||
|
/// parseConfiguration, where its endpoints were appended to interface 3's list.
|
||||||
|
interfaces: []InterfaceInfo = &.{},
|
||||||
|
|
||||||
// Transfer rings configured for this device's interrupt/bulk endpoints.
|
// Transfer rings configured for this device's interrupt/bulk endpoints.
|
||||||
endpoint_ring_count: u8 = 0,
|
endpoint_ring_count: u8 = 0,
|
||||||
endpoint_rings: [max_configured_endpoints]ConfiguredEndpoint = [_]ConfiguredEndpoint{.{}} ** max_configured_endpoints,
|
endpoint_rings: [max_configured_endpoints]ConfiguredEndpoint = [_]ConfiguredEndpoint{.{}} ** max_configured_endpoints,
|
||||||
@@ -296,6 +321,15 @@ pub const Device = struct {
|
|||||||
// Downstream ports with a pending change to service (bit P = port P), set
|
// Downstream ports with a pending change to service (bit P = port P), set
|
||||||
// by the status-change endpoint (and by an initial sweep in setupHub).
|
// by the status-change endpoint (and by an initial sweep in setupHub).
|
||||||
hub_change_mask: u32 = 0,
|
hub_change_mask: u32 = 0,
|
||||||
|
|
||||||
|
/// Release the interface list. Safe to call twice, and on a device that never
|
||||||
|
/// enumerated — re-enumeration and teardown both come through here.
|
||||||
|
pub fn freeInterfaces(self: *Device) void {
|
||||||
|
if (self.interfaces.len != 0) memory.allocator().free(self.interfaces);
|
||||||
|
self.interfaces = &.{};
|
||||||
|
self.interface_count = 0;
|
||||||
|
}
|
||||||
|
|
||||||
};
|
};
|
||||||
|
|
||||||
// A standing interrupt-IN subscription: the endpoint's ring is kept armed with a
|
// A standing interrupt-IN subscription: the endpoint's ring is kept armed with a
|
||||||
@@ -328,9 +362,6 @@ pub const Report = struct {
|
|||||||
data: [64]u8 = [_]u8{0} ** 64,
|
data: [64]u8 = [_]u8{0} ** 64,
|
||||||
};
|
};
|
||||||
|
|
||||||
// How many addressed devices this driver tracks at once. QEMU presents a handful
|
|
||||||
// (a keyboard, a mouse, a storage stick); a fuller machine would grow this.
|
|
||||||
const max_devices = 8;
|
|
||||||
const max_subscriptions = 8;
|
const max_subscriptions = 8;
|
||||||
const report_queue_capacity = 16;
|
const report_queue_capacity = 16;
|
||||||
|
|
||||||
@@ -449,7 +480,14 @@ pub const Controller = struct {
|
|||||||
device_context_array: memory.DmaRegion,
|
device_context_array: memory.DmaRegion,
|
||||||
command_ring: ProducerRing,
|
command_ring: ProducerRing,
|
||||||
event_ring: EventRing,
|
event_ring: EventRing,
|
||||||
devices: [max_devices]Device = [_]Device{.{}} ** max_devices,
|
/// One entry per device slot the **controller** says it has (HCSPARAMS1.MaxSlots,
|
||||||
|
/// 1..255), allocated at bring-up. This used to be a fixed 8 with the comment "QEMU
|
||||||
|
/// presents a handful; a fuller machine would grow this" — which is the shape the
|
||||||
|
/// bounds rule exists to stop, since the controller has always reported the real
|
||||||
|
/// number and `op_config` below is already programmed with it. A desktop's keyboard,
|
||||||
|
/// mouse, webcam, headset, hub and two sticks reach 8 without trying, and everything
|
||||||
|
/// past it vanished (behind a hub, without even a log line).
|
||||||
|
devices: []Device,
|
||||||
subscriptions: [max_subscriptions]Subscription = [_]Subscription{.{}} ** max_subscriptions,
|
subscriptions: [max_subscriptions]Subscription = [_]Subscription{.{}} ** max_subscriptions,
|
||||||
report_queue: [report_queue_capacity]Report = [_]Report{.{}} ** report_queue_capacity,
|
report_queue: [report_queue_capacity]Report = [_]Report{.{}} ** report_queue_capacity,
|
||||||
report_count: usize = 0,
|
report_count: usize = 0,
|
||||||
@@ -582,8 +620,16 @@ pub const Controller = struct {
|
|||||||
.device_context_array = undefined,
|
.device_context_array = undefined,
|
||||||
.command_ring = undefined,
|
.command_ring = undefined,
|
||||||
.event_ring = undefined,
|
.event_ring = undefined,
|
||||||
|
.devices = &.{},
|
||||||
};
|
};
|
||||||
|
|
||||||
|
// One tracking slot per slot the controller reports. A controller that claims
|
||||||
|
// no slots cannot address anything, so treat that as a dead controller rather
|
||||||
|
// than allocating nothing and failing mysteriously later.
|
||||||
|
if (self.max_slots == 0) return null;
|
||||||
|
self.devices = memory.allocator().alloc(Device, self.max_slots) catch return null;
|
||||||
|
for (self.devices) |*device| device.* = .{};
|
||||||
|
|
||||||
// Wait for the controller to report ready, then halt it if it is running.
|
// Wait for the controller to report ready, then halt it if it is running.
|
||||||
if (!waitClear(self.operational(op_usbsts), usbsts_controller_not_ready)) return null;
|
if (!waitClear(self.operational(op_usbsts), usbsts_controller_not_ready)) return null;
|
||||||
if (read32(self.operational(op_usbcmd)) & usbcmd_run != 0) {
|
if (read32(self.operational(op_usbcmd)) & usbcmd_run != 0) {
|
||||||
@@ -790,7 +836,7 @@ pub const Controller = struct {
|
|||||||
}
|
}
|
||||||
|
|
||||||
fn allocateDevice(self: *Controller) ?*Device {
|
fn allocateDevice(self: *Controller) ?*Device {
|
||||||
for (&self.devices) |*device| {
|
for (self.devices) |*device| {
|
||||||
if (!device.used) return device;
|
if (!device.used) return device;
|
||||||
}
|
}
|
||||||
return null;
|
return null;
|
||||||
@@ -897,7 +943,8 @@ pub const Controller = struct {
|
|||||||
return null;
|
return null;
|
||||||
};
|
};
|
||||||
const device = self.allocateDevice() orelse {
|
const device = self.allocateDevice() orelse {
|
||||||
std.log.info("port {d} setup: no free device slot", .{port});
|
std.log.warn("port {d} setup: no free device slot", .{port});
|
||||||
|
self.disableSlot(slot_id); // the controller already gave it to us
|
||||||
return null;
|
return null;
|
||||||
};
|
};
|
||||||
device.* = .{
|
device.* = .{
|
||||||
@@ -1011,7 +1058,7 @@ pub const Controller = struct {
|
|||||||
/// The next pending (hub, downstream-port) change to service, or null. Clears
|
/// The next pending (hub, downstream-port) change to service, or null. Clears
|
||||||
/// the returned port's bit. Called on the bus tick.
|
/// the returned port's bit. Called on the bus tick.
|
||||||
pub fn takeHubChange(self: *Controller) ?struct { hub: *Device, port: u16 } {
|
pub fn takeHubChange(self: *Controller) ?struct { hub: *Device, port: u16 } {
|
||||||
for (&self.devices) |*device| {
|
for (self.devices) |*device| {
|
||||||
if (!device.used or !device.is_hub or device.hub_change_mask == 0) continue;
|
if (!device.used or !device.is_hub or device.hub_change_mask == 0) continue;
|
||||||
const bit: u5 = @intCast(@ctz(device.hub_change_mask));
|
const bit: u5 = @intCast(@ctz(device.hub_change_mask));
|
||||||
device.hub_change_mask &= ~(@as(u32, 1) << bit);
|
device.hub_change_mask &= ~(@as(u32, 1) << bit);
|
||||||
@@ -1084,7 +1131,7 @@ pub const Controller = struct {
|
|||||||
}
|
}
|
||||||
|
|
||||||
pub fn deviceOnHubPort(self: *Controller, hub: *Device, port: u16) ?*Device {
|
pub fn deviceOnHubPort(self: *Controller, hub: *Device, port: u16) ?*Device {
|
||||||
for (&self.devices) |*device| {
|
for (self.devices) |*device| {
|
||||||
if (device.used and device.parent_slot == hub.slot_id and device.parent_port == port) return device;
|
if (device.used and device.parent_slot == hub.slot_id and device.parent_port == port) return device;
|
||||||
}
|
}
|
||||||
return null;
|
return null;
|
||||||
@@ -1099,7 +1146,11 @@ pub const Controller = struct {
|
|||||||
std.log.info("hub slot {d} port {d}: Enable Slot failed", .{ hub.slot_id, port });
|
std.log.info("hub slot {d} port {d}: Enable Slot failed", .{ hub.slot_id, port });
|
||||||
return null;
|
return null;
|
||||||
};
|
};
|
||||||
const device = self.allocateDevice() orelse return null;
|
const device = self.allocateDevice() orelse {
|
||||||
|
std.log.warn("hub slot {d} port {d}: no free device slot", .{ hub.slot_id, port });
|
||||||
|
self.disableSlot(slot_id); // the controller already gave it to us
|
||||||
|
return null;
|
||||||
|
};
|
||||||
const child_speed = mapHubPortSpeed(speed);
|
const child_speed = mapHubPortSpeed(speed);
|
||||||
device.* = .{
|
device.* = .{
|
||||||
.used = true,
|
.used = true,
|
||||||
@@ -1150,6 +1201,7 @@ pub const Controller = struct {
|
|||||||
|
|
||||||
fn abandon(self: *Controller, device: *Device) ?*Device {
|
fn abandon(self: *Controller, device: *Device) ?*Device {
|
||||||
_ = self;
|
_ = self;
|
||||||
|
device.freeInterfaces();
|
||||||
device.used = false;
|
device.used = false;
|
||||||
return null;
|
return null;
|
||||||
}
|
}
|
||||||
@@ -1299,12 +1351,25 @@ pub const Controller = struct {
|
|||||||
const configuration = std.mem.bytesToValue(usb_abi.ConfigurationDescriptor, &header);
|
const configuration = std.mem.bytesToValue(usb_abi.ConfigurationDescriptor, &header);
|
||||||
device.configuration_value = @intFromEnum(configuration.configuration_value);
|
device.configuration_value = @intFromEnum(configuration.configuration_value);
|
||||||
|
|
||||||
// Read the whole block into a local buffer and parse it here (in the bus
|
// Read the whole block and parse it here (in the bus driver) so the parse
|
||||||
// driver) so the parse never has to cross the 256-byte IPC boundary.
|
// never has to cross the 256-byte IPC boundary. Sized by the device's own
|
||||||
var blob: [512]u8 = undefined;
|
// wTotalLength — the ceiling is then the field's u16, which is the USB
|
||||||
const length = @min(configuration.total_length, blob.len);
|
// specification's, not ours.
|
||||||
if (!self.controlTransfer(device, usb_abi.getDescriptor(.configuration, 0, 0, @intCast(length)), blob[0..length], true)) return false;
|
//
|
||||||
parseConfiguration(device, blob[0..length]);
|
// This was a fixed 512 with an `@min` clamp, which silently truncated: a
|
||||||
|
// composite device routinely exceeds it (a headset is 500-900 bytes, a UVC
|
||||||
|
// webcam 1-3 KB, a multifunction printer 600+), and the interfaces past the
|
||||||
|
// cut simply did not exist — while SET_CONFIGURATION below still configured
|
||||||
|
// the device for all of them. QEMU's boot keyboard, mouse and stick are all
|
||||||
|
// under 100 bytes, which is why it survived.
|
||||||
|
if (configuration.total_length < header.len) return false; // shorter than its own header
|
||||||
|
const blob = memory.allocator().alloc(u8, configuration.total_length) catch return false;
|
||||||
|
defer memory.allocator().free(blob);
|
||||||
|
if (!self.controlTransfer(device, usb_abi.getDescriptor(.configuration, 0, 0, configuration.total_length), blob, true)) return false;
|
||||||
|
// Both numbers, so a truncation can never again be invisible: the block the
|
||||||
|
// device declared, and the bytes actually fetched and parsed. They must match.
|
||||||
|
std.log.info("config block {d} bytes, read {d}", .{ configuration.total_length, blob.len });
|
||||||
|
if (!parseConfiguration(device, blob)) return false;
|
||||||
|
|
||||||
// Select the configuration, moving the device to the configured state.
|
// Select the configuration, moving the device to the configured state.
|
||||||
if (!self.controlTransfer(device, usb_abi.setConfiguration(configuration.configuration_value), &.{}, false)) return false;
|
if (!self.controlTransfer(device, usb_abi.setConfiguration(configuration.configuration_value), &.{}, false)) return false;
|
||||||
@@ -1315,8 +1380,30 @@ pub const Controller = struct {
|
|||||||
// and the endpoints that follow it. Endpoints belong to the most recent
|
// and the endpoints that follow it. Endpoints belong to the most recent
|
||||||
// interface. Unknown descriptor types (HID, class-specific) are skipped by
|
// interface. Unknown descriptor types (HID, class-specific) are skipped by
|
||||||
// their length.
|
// their length.
|
||||||
fn parseConfiguration(device: *Device, blob: []const u8) void {
|
/// Two passes: count the alternate-setting-0 interfaces the block declares,
|
||||||
device.interface_count = 0;
|
/// allocate exactly that many, then fill them. False only on an allocation
|
||||||
|
/// failure. The count cannot exceed 255 — `bNumInterfaces` is a u8, so that is
|
||||||
|
/// the USB specification's ceiling and not one of ours.
|
||||||
|
fn parseConfiguration(device: *Device, blob: []const u8) bool {
|
||||||
|
var declared: usize = 0;
|
||||||
|
var count_offset: usize = 0;
|
||||||
|
while (count_offset + 2 <= blob.len) {
|
||||||
|
const length = blob[count_offset];
|
||||||
|
if (length < 2 or count_offset + length > blob.len) break;
|
||||||
|
if (@as(usb_abi.DescriptorType, @enumFromInt(blob[count_offset + 1])) == .interface and
|
||||||
|
length >= @sizeOf(usb_abi.InterfaceDescriptor))
|
||||||
|
{
|
||||||
|
const descriptor = std.mem.bytesToValue(usb_abi.InterfaceDescriptor, blob[count_offset .. count_offset + @sizeOf(usb_abi.InterfaceDescriptor)]);
|
||||||
|
if (@intFromEnum(descriptor.alternate_setting) == 0 and declared < 255) declared += 1;
|
||||||
|
}
|
||||||
|
count_offset += length;
|
||||||
|
}
|
||||||
|
|
||||||
|
device.freeInterfaces();
|
||||||
|
if (declared == 0) return true;
|
||||||
|
device.interfaces = memory.allocator().alloc(InterfaceInfo, declared) catch return false;
|
||||||
|
for (device.interfaces) |*interface| interface.* = .{};
|
||||||
|
|
||||||
var current: ?*InterfaceInfo = null;
|
var current: ?*InterfaceInfo = null;
|
||||||
var offset: usize = 0;
|
var offset: usize = 0;
|
||||||
while (offset + 2 <= blob.len) {
|
while (offset + 2 <= blob.len) {
|
||||||
@@ -1326,9 +1413,17 @@ pub const Controller = struct {
|
|||||||
switch (@as(usb_abi.DescriptorType, @enumFromInt(descriptor_type))) {
|
switch (@as(usb_abi.DescriptorType, @enumFromInt(descriptor_type))) {
|
||||||
.interface => if (length >= @sizeOf(usb_abi.InterfaceDescriptor)) {
|
.interface => if (length >= @sizeOf(usb_abi.InterfaceDescriptor)) {
|
||||||
const descriptor = std.mem.bytesToValue(usb_abi.InterfaceDescriptor, blob[offset .. offset + @sizeOf(usb_abi.InterfaceDescriptor)]);
|
const descriptor = std.mem.bytesToValue(usb_abi.InterfaceDescriptor, blob[offset .. offset + @sizeOf(usb_abi.InterfaceDescriptor)]);
|
||||||
if (@intFromEnum(descriptor.alternate_setting) != 0) {
|
// An interface we do not record MUST clear `current`, or the
|
||||||
current = null; // ignore alternate settings for now
|
// endpoints that follow it attach to the previous one. The cap
|
||||||
} else if (device.interface_count < max_interfaces) {
|
// branch used to have no `else`, so a fifth interface's endpoints
|
||||||
|
// were appended to interface 3's array and a class driver bound to
|
||||||
|
// interface 3 could be handed an endpoint belonging to something
|
||||||
|
// else entirely — silently, with a truthful-looking count logged.
|
||||||
|
if (@intFromEnum(descriptor.alternate_setting) != 0 or
|
||||||
|
device.interface_count >= device.interfaces.len)
|
||||||
|
{
|
||||||
|
current = null;
|
||||||
|
} else {
|
||||||
const slot = &device.interfaces[device.interface_count];
|
const slot = &device.interfaces[device.interface_count];
|
||||||
slot.* = .{
|
slot.* = .{
|
||||||
.number = @intFromEnum(descriptor.interface_number),
|
.number = @intFromEnum(descriptor.interface_number),
|
||||||
@@ -1358,13 +1453,14 @@ pub const Controller = struct {
|
|||||||
}
|
}
|
||||||
offset += length;
|
offset += length;
|
||||||
}
|
}
|
||||||
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
// --- endpoint configuration + interrupt / bulk transfers ---------------
|
// --- endpoint configuration + interrupt / bulk transfers ---------------
|
||||||
|
|
||||||
/// Find the tracked device and interface an assigned device id belongs to.
|
/// Find the tracked device and interface an assigned device id belongs to.
|
||||||
pub fn findInterface(self: *Controller, device_id: u64) ?struct { device: *Device, interface: *InterfaceInfo } {
|
pub fn findInterface(self: *Controller, device_id: u64) ?struct { device: *Device, interface: *InterfaceInfo } {
|
||||||
for (&self.devices) |*device| {
|
for (self.devices) |*device| {
|
||||||
if (!device.used) continue;
|
if (!device.used) continue;
|
||||||
for (device.interfaces[0..device.interface_count]) |*interface| {
|
for (device.interfaces[0..device.interface_count]) |*interface| {
|
||||||
if (interface.registered_device_id == device_id) return .{ .device = device, .interface = interface };
|
if (interface.registered_device_id == device_id) return .{ .device = device, .interface = interface };
|
||||||
@@ -1633,7 +1729,7 @@ pub const Controller = struct {
|
|||||||
|
|
||||||
/// The tracked device on `port`, or null.
|
/// The tracked device on `port`, or null.
|
||||||
pub fn deviceOnPort(self: *Controller, port: u32) ?*Device {
|
pub fn deviceOnPort(self: *Controller, port: u32) ?*Device {
|
||||||
for (&self.devices) |*device| {
|
for (self.devices) |*device| {
|
||||||
if (device.used and device.port == port) return device;
|
if (device.used and device.port == port) return device;
|
||||||
}
|
}
|
||||||
return null;
|
return null;
|
||||||
@@ -1642,7 +1738,7 @@ pub const Controller = struct {
|
|||||||
/// The next used device whose parent hub is `hub_slot` and slot id > `after`
|
/// The next used device whose parent hub is `hub_slot` and slot id > `after`
|
||||||
/// (for recursive teardown when a hub itself disconnects), or null.
|
/// (for recursive teardown when a hub itself disconnects), or null.
|
||||||
pub fn nextChildOf(self: *Controller, hub_slot: u8, after: u8) ?*Device {
|
pub fn nextChildOf(self: *Controller, hub_slot: u8, after: u8) ?*Device {
|
||||||
for (&self.devices) |*device| {
|
for (self.devices) |*device| {
|
||||||
if (device.used and device.parent_slot == hub_slot and device.slot_id > after) return device;
|
if (device.used and device.parent_slot == hub_slot and device.slot_id > after) return device;
|
||||||
}
|
}
|
||||||
return null;
|
return null;
|
||||||
@@ -1656,15 +1752,27 @@ pub const Controller = struct {
|
|||||||
for (&self.subscriptions) |*subscription| {
|
for (&self.subscriptions) |*subscription| {
|
||||||
if (subscription.active and subscription.slot_id == device.slot_id) subscription.active = false;
|
if (subscription.active and subscription.slot_id == device.slot_id) subscription.active = false;
|
||||||
}
|
}
|
||||||
|
self.disableSlot(device.slot_id);
|
||||||
|
device.freeInterfaces();
|
||||||
|
device.used = false;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Hand a slot back to the controller and clear its context-array entry.
|
||||||
|
///
|
||||||
|
/// Every path that has issued a successful Enable Slot owes this, including the
|
||||||
|
/// ones that then fail to bring the device up. A slot the driver forgets is one
|
||||||
|
/// the controller never reissues, so the loss is permanent for the boot: the two
|
||||||
|
/// setup paths used to return null straight after a failed `allocateDevice`,
|
||||||
|
/// leaking a slot per attempt — and the hub path did it without even a log line.
|
||||||
|
fn disableSlot(self: *Controller, slot_id: u8) void {
|
||||||
const physical = self.submitCommand(.{
|
const physical = self.submitCommand(.{
|
||||||
.control = trbControl(.disable_slot, @as(u32, device.slot_id) << 24),
|
.control = trbControl(.disable_slot, @as(u32, slot_id) << 24),
|
||||||
});
|
});
|
||||||
if (self.awaitCommand(physical)) |code| {
|
if (self.awaitCommand(physical)) |code| {
|
||||||
if (code != @intFromEnum(CompletionCode.success))
|
if (code != @intFromEnum(CompletionCode.success))
|
||||||
std.log.info("slot {d}: Disable Slot completion code {d}", .{ device.slot_id, code });
|
std.log.info("slot {d}: Disable Slot completion code {d}", .{ slot_id, code });
|
||||||
} else std.log.info("slot {d}: Disable Slot timed out", .{device.slot_id});
|
} else std.log.info("slot {d}: Disable Slot timed out", .{slot_id});
|
||||||
const array: [*]volatile u64 = @ptrFromInt(self.device_context_array.virtual);
|
const array: [*]volatile u64 = @ptrFromInt(self.device_context_array.virtual);
|
||||||
array[device.slot_id] = 0;
|
array[slot_id] = 0;
|
||||||
device.used = false;
|
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|||||||
@@ -200,10 +200,9 @@ fn testPixel(index: u32) u32 {
|
|||||||
|
|
||||||
fn initialise(endpoint: ipc.Handle) bool {
|
fn initialise(endpoint: ipc.Handle) bool {
|
||||||
_ = endpoint;
|
_ = endpoint;
|
||||||
device.claim(device_id) catch |e| {
|
// The device arrived with the spawn — the manager holds it and names it in the call
|
||||||
std.log.info("unable to claim device {d}: {s}", .{ device_id, @errorName(e) });
|
// that creates this process, so it is ours before the first instruction here
|
||||||
return false;
|
// (docs/os-development/device-authority.md). Nothing to claim.
|
||||||
};
|
|
||||||
|
|
||||||
var descriptors: [64]device.DeviceDescriptor = undefined;
|
var descriptors: [64]device.DeviceDescriptor = undefined;
|
||||||
const total = device.enumerate(&descriptors);
|
const total = device.enumerate(&descriptors);
|
||||||
|
|||||||
@@ -22,29 +22,114 @@ const std = @import("std");
|
|||||||
const abi = @import("abi");
|
const abi = @import("abi");
|
||||||
const platform = @import("platform");
|
const platform = @import("platform");
|
||||||
const device_abi = @import("device-abi");
|
const device_abi = @import("device-abi");
|
||||||
|
const heap = @import("heap.zig");
|
||||||
|
|
||||||
pub const maximum_devices = 64;
|
/// How many devices a single registrar may put in the table.
|
||||||
|
///
|
||||||
|
/// **This is a runaway detector, not a security boundary**, and the difference matters.
|
||||||
|
/// It cannot stop a malicious driver — a quota generous enough never to bite a real
|
||||||
|
/// machine is still generous enough to be unpleasant — and it is not trying to. What
|
||||||
|
/// stops malice is that only a driver the manager handed a device can register children
|
||||||
|
/// under it (docs/os-development/device-authority.md). What this catches is a
|
||||||
|
/// *legitimate* driver in a loop, early, and attributably: the driver that did it is
|
||||||
|
/// refused, named in its own log, and restarted, while every other driver is untouched.
|
||||||
|
///
|
||||||
|
/// The shared ceilings it replaces could not do that. `maximum_devices = 64` was a
|
||||||
|
/// guess about someone else's computer and one driver's enumeration starved every
|
||||||
|
/// other — which is how an AMD Ryzen came to boot with a working display, no USB and no
|
||||||
|
/// storage. A per-registrar allowance is the same protection charged to whoever caused
|
||||||
|
/// it, which is the microkernel property rather than a workaround for it.
|
||||||
|
///
|
||||||
|
/// The number: a machine's whole PCI segment tops out at 65536 functions, and the
|
||||||
|
/// biggest real registrar seen is pci-bus at a few dozen. 4096 is far above anything a
|
||||||
|
/// real machine produces and around 1.4 MiB of descriptors, well under the kernel heap.
|
||||||
|
/// **Reaching it is a bug report, not a tuning request** — no correct driver gets near.
|
||||||
|
///
|
||||||
|
/// bound: devices one task may register
|
||||||
|
/// decided-by: ours
|
||||||
|
/// protects: the kernel heap, against a driver looping device_register
|
||||||
|
/// at-limit: refuse — ECHILDREN to the registrar; every other driver is unaffected
|
||||||
|
/// observed-by: the bus driver's line naming the reason (pci-bus reconciles functions
|
||||||
|
/// found against registered)
|
||||||
|
const maximum_devices_per_registrar = 4096;
|
||||||
|
|
||||||
/// Cap on children a single parent may have. A zero-resource child (legal — a USB
|
|
||||||
/// device is addressed through its controller, not by MMIO) sidesteps the containment
|
|
||||||
/// check, so without a bound a process that claimed one device could loop
|
|
||||||
/// `device_register` and exhaust the whole table, permanently denying it to every other
|
|
||||||
/// driver. This bounds the blast radius of one claim; a real quota (and a
|
|
||||||
/// `device_release` to reclaim on exit) is future work — see docs/driver-model.md.
|
|
||||||
const maximum_children_per_parent = 16;
|
|
||||||
|
|
||||||
var devices: [maximum_devices]device_abi.DeviceDescriptor = undefined;
|
/// The table, grown on demand from the kernel heap. **There is no ceiling**: how many
|
||||||
var claimed: [maximum_devices]?u32 = .{null} ** maximum_devices; // owner task id, or null
|
/// devices a machine has is the machine's business, and no specification bounds it, so
|
||||||
|
/// nothing here should. `devices_broker.init` runs after `heap.init` (kernel.zig), so
|
||||||
|
/// there was never a reason for this to be static beyond it having been written that
|
||||||
|
/// way first.
|
||||||
|
var devices: []device_abi.DeviceDescriptor = &.{};
|
||||||
|
/// Owner task id per device, or null. Parallel to `devices` and grown with it.
|
||||||
|
var claimed: []?u32 = &.{};
|
||||||
|
/// The task that called `register` for each device, so the per-registrar allowance can
|
||||||
|
/// be charged to whoever caused the entry. Firmware-discovered nodes carry `no_registrar`
|
||||||
|
/// — they are the kernel's own, not anybody's doing.
|
||||||
|
/// Optional, not a sentinel: **task 0 is a real task** (the kernel's own), so any
|
||||||
|
/// "none" value inside the id space is a device belonging to somebody reading back as
|
||||||
|
/// belonging to nobody. Found the moment the first test asserted a giver, because the
|
||||||
|
/// task doing the giving was task 0.
|
||||||
|
var registrar: []?u32 = &.{};
|
||||||
|
|
||||||
|
/// **Who gave each device away**, or `no_giver` if nobody ever did.
|
||||||
|
///
|
||||||
|
/// One field, and the whole authority rule follows from it: a device that was *given*
|
||||||
|
/// to someone is delegated hardware, so it may only be handed on, never taken
|
||||||
|
/// (`claim` refuses it); and when its holder dies it goes back to whoever lent it,
|
||||||
|
/// rather than becoming free for anyone to grab.
|
||||||
|
///
|
||||||
|
/// It also settles the framebuffer without mentioning it. Nobody delegates the
|
||||||
|
/// loader's framebuffer, so it has no giver, so the display service claims it exactly
|
||||||
|
/// as it always has — no exemption, no special case, no `display` anywhere in the rule.
|
||||||
|
var giver: []?u32 = &.{};
|
||||||
var count: usize = 0;
|
var count: usize = 0;
|
||||||
|
|
||||||
|
/// Grow the three parallel arrays so at least one more device fits. False if the heap
|
||||||
|
/// cannot satisfy it, which the callers report rather than swallow.
|
||||||
|
fn reserve() bool {
|
||||||
|
if (count < devices.len) return true;
|
||||||
|
// Double from a deliberately SMALL first block. Sizing it for a typical machine
|
||||||
|
// would mean the growth path never ran on the hardware we test on, and only woke up
|
||||||
|
// on someone else's larger machine — which is the exact shape of the failure this
|
||||||
|
// whole track exists to stop. At 8, every boot grows the table several times, so
|
||||||
|
// the path is exercised constantly and the suite asserts it.
|
||||||
|
const wanted = if (devices.len == 0) 8 else devices.len * 2;
|
||||||
|
const allocator = heap.allocator();
|
||||||
|
const grown_devices = allocator.realloc(devices, wanted) catch return false;
|
||||||
|
devices = grown_devices;
|
||||||
|
const grown_claimed = allocator.realloc(claimed, wanted) catch return false;
|
||||||
|
claimed = grown_claimed;
|
||||||
|
const grown_registrar = allocator.realloc(registrar, wanted) catch return false;
|
||||||
|
registrar = grown_registrar;
|
||||||
|
const grown_giver = allocator.realloc(giver, wanted) catch return false;
|
||||||
|
giver = grown_giver;
|
||||||
|
for (claimed[count..], registrar[count..], giver[count..]) |*slot, *who, *lender| {
|
||||||
|
slot.* = null;
|
||||||
|
who.* = null;
|
||||||
|
lender.* = null;
|
||||||
|
}
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// How many devices `task` has registered — the allowance is charged per registrar, so
|
||||||
|
/// a driver in a loop exhausts its own and no one else's.
|
||||||
|
fn registeredBy(task: u32) usize {
|
||||||
|
var n: usize = 0;
|
||||||
|
for (registrar[0..count]) |who| {
|
||||||
|
if (who != null and who.? == task) n += 1;
|
||||||
|
}
|
||||||
|
return n;
|
||||||
|
}
|
||||||
|
|
||||||
/// The id of the seeded framebuffer node (`seedDisplay`), or null when the machine
|
/// The id of the seeded framebuffer node (`seedDisplay`), or null when the machine
|
||||||
/// handed over no framebuffer. Lets the process layer recognise the display claim
|
/// handed over no framebuffer. Lets the process layer recognise the display claim
|
||||||
/// (to quiesce the bootstrap console) without threading the id through every caller.
|
/// (to quiesce the bootstrap console) without threading the id through every caller.
|
||||||
var display_device: ?u64 = null;
|
var display_device: ?u64 = null;
|
||||||
|
|
||||||
/// Devices discovery found but the table had no room for. Non-zero means the machine
|
/// Devices discovery found but could not record. The table grows on demand, so this is
|
||||||
/// is bigger than `maximum_devices` and some hardware is simply invisible to drivers —
|
/// no longer "the machine is bigger than our guess" — it means the kernel heap could not
|
||||||
/// which would otherwise be an entirely silent failure. Logged at boot.
|
/// satisfy the growth, which would otherwise be an entirely silent failure. Logged at
|
||||||
|
/// boot.
|
||||||
pub var dropped: usize = 0;
|
pub var dropped: usize = 0;
|
||||||
|
|
||||||
/// Snapshot the device tree into the flat table. Run once, right after discovery.
|
/// Snapshot the device tree into the flat table. Run once, right after discovery.
|
||||||
@@ -52,7 +137,9 @@ pub fn init(device_tree: *const platform.DeviceTree) void {
|
|||||||
count = 0;
|
count = 0;
|
||||||
dropped = 0;
|
dropped = 0;
|
||||||
display_device = null;
|
display_device = null;
|
||||||
for (&claimed) |*c| c.* = null;
|
for (claimed) |*c| c.* = null;
|
||||||
|
for (registrar) |*r| r.* = null;
|
||||||
|
for (giver) |*g| g.* = null;
|
||||||
walk(device_tree.root, device_abi.no_parent);
|
walk(device_tree.root, device_abi.no_parent);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -64,7 +151,7 @@ pub fn init(device_tree: *const platform.DeviceTree) void {
|
|||||||
/// table is full. Idempotent-ish: only ever call once per boot.
|
/// table is full. Idempotent-ish: only ever call once per boot.
|
||||||
pub fn seedDisplay(base: u64, width: u32, height: u32, pitch: u32, format: u32, refresh_hz: u32) ?u64 {
|
pub fn seedDisplay(base: u64, width: u32, height: u32, pitch: u32, format: u32, refresh_hz: u32) ?u64 {
|
||||||
if (base == 0 or width == 0 or height == 0) return null; // headless
|
if (base == 0 or width == 0 or height == 0) return null; // headless
|
||||||
if (count >= maximum_devices) {
|
if (!reserve()) {
|
||||||
dropped += 1;
|
dropped += 1;
|
||||||
return null;
|
return null;
|
||||||
}
|
}
|
||||||
@@ -109,7 +196,7 @@ fn walk(node: *platform.Device, parent_id: u64) void {
|
|||||||
}
|
}
|
||||||
|
|
||||||
fn record(node: *platform.Device, parent_id: u64) u64 {
|
fn record(node: *platform.Device, parent_id: u64) u64 {
|
||||||
if (count >= maximum_devices) {
|
if (!reserve()) {
|
||||||
dropped += 1;
|
dropped += 1;
|
||||||
return device_abi.no_parent; // children of a dropped node become roots, not orphans
|
return device_abi.no_parent; // children of a dropped node become roots, not orphans
|
||||||
}
|
}
|
||||||
@@ -164,9 +251,32 @@ pub fn enumerateFrom(start: usize, out: []device_abi.DeviceDescriptor) usize {
|
|||||||
pub fn claim(id: u64, owner: u32) ClaimError!void {
|
pub fn claim(id: u64, owner: u32) ClaimError!void {
|
||||||
if (id >= count) return error.NoSuchDevice;
|
if (id >= count) return error.NoSuchDevice;
|
||||||
if (claimed[@intCast(id)] != null) return error.AlreadyClaimed;
|
if (claimed[@intCast(id)] != null) return error.AlreadyClaimed;
|
||||||
|
// **Delegated hardware may be handed on, never taken.**
|
||||||
|
//
|
||||||
|
// Belt and braces, and worth being honest about: with the loan rule above this is
|
||||||
|
// **currently unreachable**. A device that was given to someone is held, so it is
|
||||||
|
// refused as `AlreadyClaimed` before reaching here; and when the holder dies the
|
||||||
|
// device goes back to its lender (or, if the lender is gone, has its giver cleared
|
||||||
|
// with its claim), so there is no state where a device is unheld *and* still on
|
||||||
|
// loan. The window a stranger could have used simply stops existing.
|
||||||
|
//
|
||||||
|
// It stays because it is one comparison and it fails closed: any future path that
|
||||||
|
// frees a device without clearing its giver would otherwise hand delegated
|
||||||
|
// hardware to whoever asked first, which is exactly the hole this run closed.
|
||||||
|
//
|
||||||
|
// Note it leaves the loader's framebuffer alone without naming it: nobody delegates
|
||||||
|
// the framebuffer, so it has no giver, so the display service claims it as always.
|
||||||
|
if (giver[@intCast(id)] != null) return error.NotYours;
|
||||||
claimed[@intCast(id)] = owner;
|
claimed[@intCast(id)] = owner;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The task that gave device `id` away, or null if nobody ever did. A device with a
|
||||||
|
/// giver is delegated hardware: it may be handed on, never taken.
|
||||||
|
pub fn giverOf(id: u64) ?u32 {
|
||||||
|
if (id >= count) return null;
|
||||||
|
return giver[@intCast(id)];
|
||||||
|
}
|
||||||
|
|
||||||
/// The task that owns device `id`, or null.
|
/// The task that owns device `id`, or null.
|
||||||
pub fn ownerOf(id: u64) ?u32 {
|
pub fn ownerOf(id: u64) ?u32 {
|
||||||
if (id >= count) return null;
|
if (id >= count) return null;
|
||||||
@@ -178,11 +288,41 @@ pub fn ownerOf(id: u64) ?u32 {
|
|||||||
/// hardware again (docs/process-lifecycle.md iron rule 1: cleanup is the kernel's
|
/// hardware again (docs/process-lifecycle.md iron rule 1: cleanup is the kernel's
|
||||||
/// job). The devices stay in the table — they describe hardware, which did not go
|
/// job). The devices stay in the table — they describe hardware, which did not go
|
||||||
/// away — only their ownership clears.
|
/// away — only their ownership clears.
|
||||||
|
/// Whether a task is still alive, injected by the process layer (which owns the task
|
||||||
|
/// table) the same way the scheduler's other hooks are. Null means "assume not", so a
|
||||||
|
/// kernel built without it clears claims rather than handing them to a ghost.
|
||||||
|
pub var task_alive_hook: ?*const fn (u32) bool = null;
|
||||||
|
|
||||||
|
fn alive(task: u32) bool {
|
||||||
|
const hook = task_alive_hook orelse return false;
|
||||||
|
return hook(task);
|
||||||
|
}
|
||||||
|
|
||||||
pub fn releaseAllOwnedBy(owner: u32) void {
|
pub fn releaseAllOwnedBy(owner: u32) void {
|
||||||
for (claimed[0..count]) |*slot| {
|
for (claimed[0..count], 0..) |*slot, id| {
|
||||||
if (slot.*) |o| {
|
const holder = slot.* orelse continue;
|
||||||
if (o == owner) slot.* = null;
|
if (holder != owner) continue;
|
||||||
|
|
||||||
|
// **A grant is a loan.** A device this task was *given* goes back to whoever
|
||||||
|
// lent it, not to nobody — so the device manager gets its hardware back the
|
||||||
|
// instant a driver dies, and hands it to the replacement.
|
||||||
|
//
|
||||||
|
// Without this the kernel released the claim to no one and the manager
|
||||||
|
// re-claimed first-come, so every driver restart reopened the window this
|
||||||
|
// rule closes. And once `claim` refuses a device that has a giver, releasing
|
||||||
|
// to nobody would strand it: no one could ever take it again.
|
||||||
|
//
|
||||||
|
// A dead lender is no lender: clear the claim and the giver together, so the
|
||||||
|
// device is genuinely free rather than owed to a ghost.
|
||||||
|
if (giver[id]) |lender| {
|
||||||
|
if (alive(lender)) {
|
||||||
|
slot.* = lender;
|
||||||
|
giver[id] = null; // returned; it is the lender's own again, not on loan
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
giver[id] = null;
|
||||||
}
|
}
|
||||||
|
slot.* = null;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -284,7 +424,7 @@ pub const RegisterError = error{
|
|||||||
NoSuchParent, // no device with that id
|
NoSuchParent, // no device with that id
|
||||||
NotYourParent, // that device exists but this task has not claimed it
|
NotYourParent, // that device exists but this task has not claimed it
|
||||||
TooManyResources, // the descriptor declares more resources than one device may hold
|
TooManyResources, // the descriptor declares more resources than one device may hold
|
||||||
TooManyChildren, // this parent is at maximum_children_per_parent
|
TooManyChildren, // the caller is at its per-registrar allowance
|
||||||
NotContained, // a child resource escapes its parent's window
|
NotContained, // a child resource escapes its parent's window
|
||||||
};
|
};
|
||||||
|
|
||||||
@@ -304,8 +444,46 @@ pub fn errnoOf(e: RegisterError) i64 {
|
|||||||
pub const ClaimError = error{
|
pub const ClaimError = error{
|
||||||
NoSuchDevice, // no device with that id
|
NoSuchDevice, // no device with that id
|
||||||
AlreadyClaimed, // a live task already owns it
|
AlreadyClaimed, // a live task already owns it
|
||||||
|
NotYours, // delegated hardware: it has a giver, so it must be handed on, not taken
|
||||||
};
|
};
|
||||||
|
|
||||||
|
/// Why a `transfer` was refused.
|
||||||
|
pub const TransferError = error{
|
||||||
|
NoSuchDevice, // no device with that id
|
||||||
|
NotHeld, // the caller does not hold it — you may only give away what you have
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The errno a refused `transfer` returns to ring 3. (`ESRCH` — no such recipient — is
|
||||||
|
/// raised by the caller in system/kernel/process.zig, which is what can see the task
|
||||||
|
/// table.)
|
||||||
|
pub fn transferErrnoOf(e: TransferError) i64 {
|
||||||
|
return switch (e) {
|
||||||
|
error.NoSuchDevice => abi.ENODEV,
|
||||||
|
error.NotHeld => abi.EPERM,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Move device `id` from `from` to `to`. **A move, not a copy**: a claim is exclusive
|
||||||
|
/// (driver-model.md, invariant 1), so the giver stops holding it the moment the
|
||||||
|
/// receiver starts.
|
||||||
|
///
|
||||||
|
/// This is the mechanism behind delegation — the device manager claims what firmware
|
||||||
|
/// discovery seeded and passes each device to the driver it matched, which replaces
|
||||||
|
/// first-come-first-served `device_claim` with policy
|
||||||
|
/// (docs/device-driver-development/device-manager.md). The kernel checks only that the
|
||||||
|
/// caller holds the device: *you may give away what you have*. It knows nothing about
|
||||||
|
/// which task is the manager, and needs to know nothing.
|
||||||
|
///
|
||||||
|
/// Note this is deliberately NOT the M13 capability-passing path, which shares a handle
|
||||||
|
/// refcounted — a copy. Exclusivity cannot be expressed that way.
|
||||||
|
pub fn transfer(id: u64, from: u32, to: u32) TransferError!void {
|
||||||
|
if (id >= count) return error.NoSuchDevice;
|
||||||
|
const holder = claimed[@intCast(id)] orelse return error.NotHeld;
|
||||||
|
if (holder != from) return error.NotHeld;
|
||||||
|
claimed[@intCast(id)] = to;
|
||||||
|
giver[@intCast(id)] = from;
|
||||||
|
}
|
||||||
|
|
||||||
/// The errno a refused `claim` returns to ring 3. (`ECONFINE` — the claim stood but
|
/// The errno a refused `claim` returns to ring 3. (`ECONFINE` — the claim stood but
|
||||||
/// the IOMMU would not confine the device — is raised by the caller in
|
/// the IOMMU would not confine the device — is raised by the caller in
|
||||||
/// system/kernel/process.zig, which is what rolls the claim back.)
|
/// system/kernel/process.zig, which is what rolls the claim back.)
|
||||||
@@ -313,6 +491,7 @@ pub fn claimErrnoOf(e: ClaimError) i64 {
|
|||||||
return switch (e) {
|
return switch (e) {
|
||||||
error.NoSuchDevice => abi.ENODEV,
|
error.NoSuchDevice => abi.ENODEV,
|
||||||
error.AlreadyClaimed => abi.EBUSY,
|
error.AlreadyClaimed => abi.EBUSY,
|
||||||
|
error.NotYours => abi.EPERM,
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -339,14 +518,6 @@ fn existingChild(parent_id: u64, descriptor: *const device_abi.DeviceDescriptor)
|
|||||||
return null;
|
return null;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Number of devices currently recorded with `parent_id` as their parent.
|
|
||||||
fn childCount(parent_id: u64) usize {
|
|
||||||
var n: usize = 0;
|
|
||||||
for (devices[0..count]) |d| {
|
|
||||||
if (d.parent == parent_id) n += 1;
|
|
||||||
}
|
|
||||||
return n;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Publish `descriptor` as a child of `parent_id`, on behalf of `owner`. Returns the new
|
/// Publish `descriptor` as a child of `parent_id`, on behalf of `owner`. Returns the new
|
||||||
/// device id. The child is left **unclaimed**, so another process (a class driver)
|
/// device id. The child is left **unclaimed**, so another process (a class driver)
|
||||||
@@ -377,8 +548,11 @@ pub fn register(parent_id: u64, owner: u32, descriptor: *const device_abi.Device
|
|||||||
// below).
|
// below).
|
||||||
if (existingChild(parent_id, descriptor)) |existing_id| return existing_id;
|
if (existingChild(parent_id, descriptor)) |existing_id| return existing_id;
|
||||||
|
|
||||||
if (childCount(parent_id) >= maximum_children_per_parent) return error.TooManyChildren;
|
// The allowance is charged to whoever is registering, so a driver in a loop
|
||||||
if (count >= maximum_devices) return error.NoSpace;
|
// exhausts its own and every other driver carries on. There is no machine-wide
|
||||||
|
// ceiling any more: the table grows.
|
||||||
|
if (registeredBy(owner) >= maximum_devices_per_registrar) return error.TooManyChildren;
|
||||||
|
if (!reserve()) return error.NoSpace;
|
||||||
|
|
||||||
const parent = &devices[@intCast(parent_id)];
|
const parent = &devices[@intCast(parent_id)];
|
||||||
for (0..@intCast(descriptor.resource_count)) |i| {
|
for (0..@intCast(descriptor.resource_count)) |i| {
|
||||||
@@ -401,6 +575,7 @@ pub fn register(parent_id: u64, owner: u32, descriptor: *const device_abi.Device
|
|||||||
for (0..@intCast(descriptor.resource_count)) |i| d.resources[i] = descriptor.resources[i];
|
for (0..@intCast(descriptor.resource_count)) |i| d.resources[i] = descriptor.resources[i];
|
||||||
|
|
||||||
devices[count] = d;
|
devices[count] = d;
|
||||||
|
registrar[count] = owner;
|
||||||
count += 1;
|
count += 1;
|
||||||
return d.id;
|
return d.id;
|
||||||
}
|
}
|
||||||
|
|||||||
+74
-16
@@ -34,12 +34,27 @@ const platform = @import("platform");
|
|||||||
const architecture = @import("architecture");
|
const architecture = @import("architecture");
|
||||||
const devices_broker = @import("devices-broker.zig");
|
const devices_broker = @import("devices-broker.zig");
|
||||||
const log = @import("log.zig");
|
const log = @import("log.zig");
|
||||||
|
const heap = @import("heap.zig");
|
||||||
|
|
||||||
const page_size: u64 = abi.page_size;
|
const page_size: u64 = abi.page_size;
|
||||||
const page_mask: u64 = page_size - 1;
|
const page_mask: u64 = page_size - 1;
|
||||||
const huge_page_size: u64 = 2 * 1024 * 1024;
|
const huge_page_size: u64 = 2 * 1024 * 1024;
|
||||||
|
|
||||||
/// One domain per claimed PCI function. 64 mirrors devices-broker's device cap.
|
/// The IOMMU's own translation-domain pool — one per claimed DMA-capable device.
|
||||||
|
///
|
||||||
|
/// No longer coupled to the device count. It was, by a comment and then by a comptime
|
||||||
|
/// assert, only because `confined` (one slot per device id) was sized by this same
|
||||||
|
/// constant; those are two unrelated quantities and making the device table dynamic
|
||||||
|
/// separated them. This one is genuinely the hardware's: both VT-d and AMD-Vi report
|
||||||
|
/// how many domains they support in a capability register, so the honest fix is to read
|
||||||
|
/// it rather than choose 64 — bounds-track-plan.md phase 4.
|
||||||
|
///
|
||||||
|
/// bound: IOMMU translation domains the kernel can hold at once
|
||||||
|
/// decided-by: hardware
|
||||||
|
/// protects: the statically sized domain pool
|
||||||
|
/// at-limit: refuse — ECONFINE; the claim is rolled back and the device is not driven,
|
||||||
|
/// because a claim that cannot be confined must not stand
|
||||||
|
/// observed-by: the claiming driver's own line naming ECONFINE
|
||||||
pub const maximum_domains = 64;
|
pub const maximum_domains = 64;
|
||||||
pub const invalid_domain: u16 = 0xFFFF;
|
pub const invalid_domain: u16 = 0xFFFF;
|
||||||
|
|
||||||
@@ -94,18 +109,30 @@ pub fn init() void {
|
|||||||
/// Per-claimed-device record: its private domain, so a driver's death tears down
|
/// Per-claimed-device record: its private domain, so a driver's death tears down
|
||||||
/// exactly the domains it held.
|
/// exactly the domains it held.
|
||||||
const Confined = struct { active: bool = false, owner: u32 = 0, bdf: u16 = 0, domain: u16 = invalid_domain };
|
const Confined = struct { active: bool = false, owner: u32 = 0, bdf: u16 = 0, domain: u16 = invalid_domain };
|
||||||
var confined: [maximum_domains]Confined = .{Confined{}} ** maximum_domains;
|
|
||||||
|
|
||||||
// `confined` is indexed by **device id**, so it must cover every id the broker can
|
/// Indexed by **device id**, so it must cover every id the broker can mint — and the
|
||||||
// mint. These two numbers agreed only by a sentence in a comment above
|
/// broker's table has no ceiling any more, so neither can this. It grows on demand.
|
||||||
// `maximum_domains` — and when they disagreed, `confineDevice` returned success for
|
///
|
||||||
// the ids it had no room for, leaving those devices unconfined DMA masters. Coupled
|
/// This used to be `[maximum_domains]`, sized by the *domain* constant purely because
|
||||||
// bounds agree in code, not in prose (docs/os-development/bounds.md).
|
/// device ids happened to stop at 64 as well. Two unrelated quantities sharing one
|
||||||
comptime {
|
/// number: `domains` below is the IOMMU's own translation-domain pool, which the
|
||||||
if (maximum_domains < devices_broker.maximum_devices)
|
/// hardware bounds and reports, while this is one slot per device the machine has.
|
||||||
@compileError("iommu.confined is indexed by device id but is smaller than the " ++
|
/// A comptime assert held them together while both were fixed; making the device table
|
||||||
"broker's device table: ids past its end cannot be confined, and so cannot " ++
|
/// dynamic is what forced them apart, which is the assert having done its job.
|
||||||
"be claimed at all");
|
var confined: []Confined = &.{};
|
||||||
|
|
||||||
|
/// Grow `confined` to cover `device_id`. False if the heap cannot — and the caller
|
||||||
|
/// treats that as a refusal to confine, never as permission.
|
||||||
|
fn reserveConfined(device_id: u64) bool {
|
||||||
|
if (device_id < confined.len) return true;
|
||||||
|
if (device_id >= std.math.maxInt(usize) / 2) return false; // absurd id; refuse rather than size to it
|
||||||
|
var wanted: usize = if (confined.len == 0) 64 else confined.len;
|
||||||
|
while (wanted <= device_id) wanted *= 2;
|
||||||
|
const grown = heap.allocator().realloc(confined, wanted) catch return false;
|
||||||
|
const previous = confined.len;
|
||||||
|
confined = grown;
|
||||||
|
for (confined[previous..]) |*record| record.* = .{};
|
||||||
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Place a just-claimed PCI function under IOMMU translation on behalf of `owner`: give
|
/// Place a just-claimed PCI function under IOMMU translation on behalf of `owner`: give
|
||||||
@@ -126,7 +153,7 @@ pub fn confineDevice(device_id: u64, bdf: u16, owner: u32) bool {
|
|||||||
// at `confined.len`; moving the inventory out of the kernel and taking the domain
|
// at `confined.len`; moving the inventory out of the kernel and taking the domain
|
||||||
// count from the hardware both change that, and either would have made a silent
|
// count from the hardware both change that, and either would have made a silent
|
||||||
// unconfined DMA master out of every device past the 64th.
|
// unconfined DMA master out of every device past the 64th.
|
||||||
if (device_id >= confined.len) return false;
|
if (!reserveConfined(device_id)) return false;
|
||||||
const domain = domainCreate(owner, bdf) orelse return false;
|
const domain = domainCreate(owner, bdf) orelse return false;
|
||||||
|
|
||||||
// Firmware reserved region for this device, if any (real hardware; QEMU has none).
|
// Firmware reserved region for this device, if any (real hardware; QEMU has none).
|
||||||
@@ -168,7 +195,7 @@ pub fn unmapForDevice(device_id: u64, physical: u64, len: u64) void {
|
|||||||
/// own freshly-`dma_alloc`'d buffer into the devices it drives.
|
/// own freshly-`dma_alloc`'d buffer into the devices it drives.
|
||||||
pub fn mapRegionForOwner(owner: u32, physical: u64, len: u64) void {
|
pub fn mapRegionForOwner(owner: u32, physical: u64, len: u64) void {
|
||||||
if (!active) return;
|
if (!active) return;
|
||||||
for (&confined) |*c| {
|
for (confined) |*c| {
|
||||||
if (c.active and c.owner == owner) _ = map(c.domain, physical, len);
|
if (c.active and c.owner == owner) _ = map(c.domain, physical, len);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -179,7 +206,7 @@ pub fn mapRegionForOwner(owner: u32, physical: u64, len: u64) void {
|
|||||||
/// domain other than its owner's.
|
/// domain other than its owner's.
|
||||||
pub fn unmapRegionEverywhere(physical: u64, len: u64) void {
|
pub fn unmapRegionEverywhere(physical: u64, len: u64) void {
|
||||||
if (!active) return;
|
if (!active) return;
|
||||||
for (&confined) |*c| {
|
for (confined) |*c| {
|
||||||
if (c.active) unmap(c.domain, physical, len);
|
if (c.active) unmap(c.domain, physical, len);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -187,9 +214,40 @@ pub fn unmapRegionEverywhere(physical: u64, len: u64) void {
|
|||||||
/// A driver died or released its devices: tear down every domain it held (detach the
|
/// A driver died or released its devices: tear down every domain it held (detach the
|
||||||
/// device, free the tables) so their DMA is blocked again and a restarted driver
|
/// device, free the tables) so their DMA is blocked again and a restarted driver
|
||||||
/// re-claims cleanly. Runs BEFORE the broker claims and the DMA frames are released.
|
/// re-claims cleanly. Runs BEFORE the broker claims and the DMA frames are released.
|
||||||
|
/// Re-point a device's existing confinement at a new owner, keeping its domain and
|
||||||
|
/// its attachment intact.
|
||||||
|
///
|
||||||
|
/// Delegation needs this: the device manager claims a device (which confines it, with
|
||||||
|
/// the manager as owner) and then transfers it to the driver. Without moving the
|
||||||
|
/// confinement record too, the domain stays the manager's — so the driver's DMA
|
||||||
|
/// buffers are never bound into it, its rings are invisible to the device, and every
|
||||||
|
/// transfer faults. Worse, a manager death would then tear down a domain a live driver
|
||||||
|
/// is using, and a driver death would leave one behind.
|
||||||
|
///
|
||||||
|
/// The domain is *not* rebuilt: the device stays attached throughout, so there is no
|
||||||
|
/// window in which it is translating through nothing.
|
||||||
|
pub fn reassign(device_id: u64, owner: u32) void {
|
||||||
|
if (!active) return;
|
||||||
|
if (device_id >= confined.len) return;
|
||||||
|
const record = &confined[@intCast(device_id)];
|
||||||
|
if (!record.active) return;
|
||||||
|
record.owner = owner;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The task a device's confinement is recorded against, or null if it has none. The
|
||||||
|
/// confinement's owner decides whose death tears the domain down, so a delegation that
|
||||||
|
/// moved the device but not this record would leave a live driver's domain destroyed by
|
||||||
|
/// its manager's exit — which is why `reassign` exists and why the suite asserts it.
|
||||||
|
pub fn confinementOwner(device_id: u64) ?u32 {
|
||||||
|
if (!active) return null;
|
||||||
|
if (device_id >= confined.len) return null;
|
||||||
|
const record = confined[@intCast(device_id)];
|
||||||
|
return if (record.active) record.owner else null;
|
||||||
|
}
|
||||||
|
|
||||||
pub fn releaseAllOwnedBy(owner: u32) void {
|
pub fn releaseAllOwnedBy(owner: u32) void {
|
||||||
if (!active) return;
|
if (!active) return;
|
||||||
for (&confined) |*c| {
|
for (confined) |*c| {
|
||||||
if (c.active and c.owner == owner) {
|
if (c.active and c.owner == owner) {
|
||||||
detachDevice(c.bdf);
|
detachDevice(c.bdf);
|
||||||
domainDestroy(c.domain);
|
domainDestroy(c.domain);
|
||||||
|
|||||||
@@ -190,6 +190,15 @@ pub fn init() void {
|
|||||||
scheduler.timer_tick_hook = timerSweepLocked;
|
scheduler.timer_tick_hook = timerSweepLocked;
|
||||||
scheduler.group_exit_hook = groupExitLocked;
|
scheduler.group_exit_hook = groupExitLocked;
|
||||||
scheduler.space_mapping_release_hook = dropSpaceMappingHook;
|
scheduler.space_mapping_release_hook = dropSpaceMappingHook;
|
||||||
|
// The broker owns devices; the process layer owns the task table. It asks whether a
|
||||||
|
// lender is still alive before handing a dead driver's device back to it.
|
||||||
|
devices_broker.task_alive_hook = taskAliveLocked;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether `id` names a live task. The broker calls this through its hook when
|
||||||
|
/// deciding if a dead holder's device can go back to the task that lent it.
|
||||||
|
fn taskAliveLocked(id: u32) bool {
|
||||||
|
return scheduler.taskByIdLocked(id) != null;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Return -1 (as an unsigned bit pattern) in the system_call result register.
|
/// Return -1 (as an unsigned bit pattern) in the system_call result register.
|
||||||
@@ -249,6 +258,7 @@ fn system_call(state: *architecture.CpuState) void {
|
|||||||
.irq_bind => systemIrqBind(state),
|
.irq_bind => systemIrqBind(state),
|
||||||
.irq_ack => systemIrqAck(state),
|
.irq_ack => systemIrqAck(state),
|
||||||
.device_register => systemDeviceRegister(state),
|
.device_register => systemDeviceRegister(state),
|
||||||
|
.device_transfer => systemDeviceTransfer(state),
|
||||||
.system_spawn => systemSpawn(state),
|
.system_spawn => systemSpawn(state),
|
||||||
.dma_alloc => systemDmaAlloc(state),
|
.dma_alloc => systemDmaAlloc(state),
|
||||||
.dma_free => systemDmaFree(state),
|
.dma_free => systemDmaFree(state),
|
||||||
@@ -440,6 +450,41 @@ fn systemDeviceClaim(state: *architecture.CpuState) void {
|
|||||||
architecture.setSystemCallResult(state, 0);
|
architecture.setSystemCallResult(state, 0);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// device_transfer(device_id, task_id) -> 0/-errno: give a device you hold to another
|
||||||
|
/// task. The mechanism behind delegation — the device manager claims what discovery
|
||||||
|
/// seeded and hands each device to the driver it matched, so assignment stops being
|
||||||
|
/// first-come-first-served (docs/os-development/device-authority.md).
|
||||||
|
///
|
||||||
|
/// The kernel's whole rule is *you may give away what you hold*. It has no notion of
|
||||||
|
/// which task is the device manager, and deliberately gains none: a binary name in the
|
||||||
|
/// kernel is not something that cannot safely live in user space.
|
||||||
|
fn systemDeviceTransfer(state: *architecture.CpuState) void {
|
||||||
|
const device_id = architecture.systemCallArg(state, 0);
|
||||||
|
const task_id: u32 = @truncate(architecture.systemCallArg(state, 1));
|
||||||
|
const flags = sync.enter();
|
||||||
|
defer sync.leave(flags);
|
||||||
|
|
||||||
|
// The recipient must exist, or the device would be moved to nobody and become
|
||||||
|
// unreachable for the rest of the boot — no path un-holds a device but task death.
|
||||||
|
if (scheduler.taskByIdLocked(task_id) == null) return failErr(state, ipc.ESRCH);
|
||||||
|
|
||||||
|
devices_broker.transfer(device_id, scheduler.current().id, task_id) catch |e|
|
||||||
|
return failErr(state, devices_broker.transferErrnoOf(e));
|
||||||
|
|
||||||
|
// The device's IOMMU confinement moves with it. The giver confined it when it
|
||||||
|
// claimed, so the domain exists and the device stays attached — but the record
|
||||||
|
// still names the giver as owner, which would leave the receiver's DMA buffers
|
||||||
|
// unbound (every transfer faulting), a giver's death tearing down a domain the
|
||||||
|
// receiver is using, and the receiver's death leaving one behind.
|
||||||
|
if (devices_broker.pciAddressOf(device_id)) |_| {
|
||||||
|
iommu.reassign(device_id, task_id);
|
||||||
|
// Bind whatever the receiver has already allocated — the same courtesy the
|
||||||
|
// claim path does for a driver that dma_alloc'd its rings before claiming.
|
||||||
|
dmaBindOwnerRegionsInto(task_id, device_id);
|
||||||
|
}
|
||||||
|
architecture.setSystemCallResult(state, 0);
|
||||||
|
}
|
||||||
|
|
||||||
/// mmio_map(device_id, resource_index) -> virtual_address: map a claimed device's MMIO window into
|
/// mmio_map(device_id, resource_index) -> virtual_address: map a claimed device's MMIO window into
|
||||||
/// this address space (strong-uncacheable) and return the register base address.
|
/// this address space (strong-uncacheable) and return the register base address.
|
||||||
/// The claim is the capability — a process can only map hardware it owns.
|
/// The claim is the capability — a process can only map hardware it owns.
|
||||||
@@ -981,6 +1026,13 @@ fn systemSpawn(state: *architecture.CpuState) void {
|
|||||||
const arguments_ptr = architecture.systemCallArg(state, 2);
|
const arguments_ptr = architecture.systemCallArg(state, 2);
|
||||||
const arguments_len = architecture.systemCallArg(state, 3);
|
const arguments_len = architecture.systemCallArg(state, 3);
|
||||||
const exit_handle = architecture.systemCallArg(state, 4);
|
const exit_handle = architecture.systemCallArg(state, 4);
|
||||||
|
// A device the caller holds and gives to the child. Fused into the spawn rather
|
||||||
|
// than transferred after it, because a separate transfer leaves a window in which
|
||||||
|
// the child is running and does not yet hold its device — a race that would close
|
||||||
|
// on one machine and open on another, which is the failure shape this whole track
|
||||||
|
// exists to remove (docs/bounds-track-plan.md, "the grant rides system_spawn").
|
||||||
|
// Here the child cannot observe the gap: it does not exist until it holds it.
|
||||||
|
const device_to_give = architecture.systemCallArg(state, 5);
|
||||||
const t = scheduler.current();
|
const t = scheduler.current();
|
||||||
if (len == 0 or len > scheduler.maximum_task_name or ptr >= user_half_end or ptr + len > user_half_end) return fail(state);
|
if (len == 0 or len > scheduler.maximum_task_name or ptr >= user_half_end or ptr + len > user_half_end) return fail(state);
|
||||||
if (arguments_len > maximum_argument_bytes) return fail(state);
|
if (arguments_len > maximum_argument_bytes) return fail(state);
|
||||||
@@ -1025,7 +1077,26 @@ fn systemSpawn(state: *architecture.CpuState) void {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Refuse before creating anything if the device is not the caller's to give — a
|
||||||
|
// spawn that half-succeeds would leave a child running without the hardware it was
|
||||||
|
// spawned for, which is worse than not spawning it.
|
||||||
|
if (device_to_give != abi.no_device and devices_broker.ownerOf(device_to_give) != t.id)
|
||||||
|
return failErr(state, ipc.EPERM);
|
||||||
|
|
||||||
const child = spawnProcessSupervised(item.blob, 4, argv[0..argc], t.id, exit_endpoint) catch return fail(state);
|
const child = spawnProcessSupervised(item.blob, 4, argv[0..argc], t.id, exit_endpoint) catch return fail(state);
|
||||||
|
|
||||||
|
if (device_to_give != abi.no_device) {
|
||||||
|
devices_broker.transfer(device_to_give, t.id, child) catch {
|
||||||
|
// Cannot happen — ownership was checked above and the lock has not been
|
||||||
|
// dropped — but a spawned child holding nothing is not something to guess
|
||||||
|
// about, so say so rather than leave it silent.
|
||||||
|
log.print("/system/kernel: WARNING spawn gave device {d} to task {d} and the transfer failed\n", .{ device_to_give, child });
|
||||||
|
};
|
||||||
|
if (devices_broker.pciAddressOf(device_to_give)) |_| {
|
||||||
|
iommu.reassign(device_to_give, child);
|
||||||
|
dmaBindOwnerRegionsInto(child, device_to_give);
|
||||||
|
}
|
||||||
|
}
|
||||||
architecture.setSystemCallResult(state, child);
|
architecture.setSystemCallResult(state, child);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+222
-19
@@ -261,6 +261,10 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
|
|||||||
containmentTest();
|
containmentTest();
|
||||||
} else if (eql(case, "apertures")) {
|
} else if (eql(case, "apertures")) {
|
||||||
apertureTest();
|
apertureTest();
|
||||||
|
} else if (eql(case, "device-transfer")) {
|
||||||
|
deviceTransferTest(boot_information);
|
||||||
|
} else if (eql(case, "device-authority")) {
|
||||||
|
deviceAuthorityTest(boot_information);
|
||||||
} else if (eql(case, "device-manager")) {
|
} else if (eql(case, "device-manager")) {
|
||||||
deviceManagerTest(boot_information);
|
deviceManagerTest(boot_information);
|
||||||
} else if (eql(case, "protocol-registry")) {
|
} else if (eql(case, "protocol-registry")) {
|
||||||
@@ -1443,6 +1447,52 @@ fn iommuTest() void {
|
|||||||
} else {
|
} else {
|
||||||
check("scratch domain allocated", false);
|
check("scratch domain allocated", false);
|
||||||
}
|
}
|
||||||
|
// Delegation moves a device between tasks, and the IOMMU confinement must move
|
||||||
|
// with it. When it did not, the driver's DMA rings were never bound into the
|
||||||
|
// device's domain and every transfer faulted — but two worse consequences were
|
||||||
|
// latent and invisible to those tests: the domain still named the *giver*, so the
|
||||||
|
// giver's death would tear down a domain a live driver was using, and the
|
||||||
|
// receiver's death would leave one behind. Asserted directly here rather than
|
||||||
|
// inferred from the USB cases going green.
|
||||||
|
// This case runs no pci-bus, so there are no PCI functions to confine — and a
|
||||||
|
// silently skipped assertion is worse than none. Register one the way pci-bus does:
|
||||||
|
// a child of the host bridge whose resource 0 is a 4 KiB config window inside the
|
||||||
|
// bridge's ECAM, which is what `pciAddressOf` derives a requester id from.
|
||||||
|
var iommu_table: [64]device_abi.DeviceDescriptor = undefined;
|
||||||
|
const iommu_total = devices_broker.enumerate(&iommu_table);
|
||||||
|
const bridge: ?device_abi.DeviceDescriptor = for (iommu_table[0..@min(iommu_total, iommu_table.len)]) |d| {
|
||||||
|
if (d.class == @intFromEnum(device_abi.DeviceClass.pci_host_bridge) and d.resource_count >= 2) break d;
|
||||||
|
} else null;
|
||||||
|
check("the kernel seeded a PCI host bridge to parent a function under", bridge != null);
|
||||||
|
|
||||||
|
const subject: ?u64 = if (bridge) |b| blk: {
|
||||||
|
const me_bridge = scheduler.currentId();
|
||||||
|
if (!claimOk(b.id, me_bridge)) break :blk null;
|
||||||
|
var function = std.mem.zeroes(device_abi.DeviceDescriptor);
|
||||||
|
function.class = @intFromEnum(device_abi.DeviceClass.pci_device);
|
||||||
|
function.pci_class = device_abi.no_pci_class;
|
||||||
|
function.resource_count = 1;
|
||||||
|
function.resources[0] = .{
|
||||||
|
.kind = @intFromEnum(device_abi.ResourceKind.memory),
|
||||||
|
.start = b.resources[0].start, // the first config slot in the ECAM window
|
||||||
|
.len = 4096,
|
||||||
|
};
|
||||||
|
break :blk devices_broker.register(b.id, me_bridge, &function) catch null;
|
||||||
|
} else null;
|
||||||
|
check("a PCI function exists to confine", subject != null);
|
||||||
|
|
||||||
|
if (subject) |device_id| {
|
||||||
|
check("a device starts unconfined", iommu.confinementOwner(device_id) == null);
|
||||||
|
const me = scheduler.currentId();
|
||||||
|
check("confining records the owner", iommu.confineDevice(device_id, devices_broker.pciAddressOf(device_id).?, me));
|
||||||
|
check("the confinement names the confiner", iommu.confinementOwner(device_id) == me);
|
||||||
|
iommu.reassign(device_id, me + 1000);
|
||||||
|
check("reassign moves the confinement to the new holder", iommu.confinementOwner(device_id) == me + 1000);
|
||||||
|
check("and the previous holder no longer owns it", iommu.confinementOwner(device_id) != me);
|
||||||
|
iommu.releaseAllOwnedBy(me + 1000);
|
||||||
|
check("the new holder's death tears the domain down", iommu.confinementOwner(device_id) == null);
|
||||||
|
}
|
||||||
|
|
||||||
log("DANOS-IOMMU: enabled base=0x{x} domains active\n", .{pinfo.iommu_base});
|
log("DANOS-IOMMU: enabled base=0x{x} domains active\n", .{pinfo.iommu_base});
|
||||||
result();
|
result();
|
||||||
}
|
}
|
||||||
@@ -3048,7 +3098,17 @@ fn acpiParseTest(boot_information: *const BootInformation) void {
|
|||||||
while (i < rd.count) : (i += 1) {
|
while (i < rd.count) : (i += 1) {
|
||||||
const item = rd.entry(i) orelse continue;
|
const item = rd.entry(i) orelse continue;
|
||||||
if (!eql(initial_ramdisk.basename(item.name), "discovery")) continue;
|
if (!eql(initial_ramdisk.basename(item.name), "discovery")) continue;
|
||||||
_ = process.spawnProcessSupervised(item.blob, 4, &.{ item.name, "1" }, scheduler.currentId(), null) catch 0;
|
// Hand it the acpi-tables node the way the manager would: claim it here, then
|
||||||
|
// move it to the child. Discovery no longer claims for itself.
|
||||||
|
var scratch: [64]device_abi.DeviceDescriptor = undefined;
|
||||||
|
const seen_devices = devices_broker.enumerate(&scratch);
|
||||||
|
const tables: ?u64 = for (scratch[0..@min(seen_devices, scratch.len)]) |d| {
|
||||||
|
if (d.class == @intFromEnum(device_abi.DeviceClass.acpi_tables)) break d.id;
|
||||||
|
} else null;
|
||||||
|
const me_parse = scheduler.currentId();
|
||||||
|
if (tables) |node| _ = claimOk(node, me_parse);
|
||||||
|
const child = process.spawnProcessSupervised(item.blob, 4, &.{ item.name, "floor:1" }, me_parse, null) catch 0;
|
||||||
|
if (tables) |node| devices_broker.transfer(node, me_parse, child) catch {};
|
||||||
spawned = true;
|
spawned = true;
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
@@ -3987,39 +4047,130 @@ fn containmentTest() void {
|
|||||||
check("re-registering an identical child returns the same id", again != 0 and again == good);
|
check("re-registering an identical child returns the same id", again != 0 and again == good);
|
||||||
check("re-registering grew nothing", devices_broker.enumerate(&buffer) == before + 1);
|
check("re-registering grew nothing", devices_broker.enumerate(&buffer) == before + 1);
|
||||||
|
|
||||||
// Fill the parent to its child cap with distinct children (same window, different
|
// **There is no per-parent cap.** `maximum_children_per_parent = 16` is gone: it
|
||||||
// identity — the match is on identity, so each is a new device).
|
// was written to stop a driver looping `device_register` and exhausting a shared
|
||||||
|
// table, and there is no shared table to exhaust — the table grows, and each
|
||||||
|
// registrar has its own allowance, so a runaway costs only itself. The cap never
|
||||||
|
// bounded a determined caller anyway (16 per parent, but nothing stopped it
|
||||||
|
// claiming more parents); what it reliably did was refuse a real PCI bus with more
|
||||||
|
// than 16 functions, which is how an AMD Ryzen booted with no USB and no storage.
|
||||||
|
//
|
||||||
|
// 64 children under one parent — four times the old ceiling.
|
||||||
var filled: u32 = 0;
|
var filled: u32 = 0;
|
||||||
var capped = false;
|
var registered_children: u32 = 0;
|
||||||
while (filled < 64) : (filled += 1) {
|
while (filled < 64) : (filled += 1) {
|
||||||
var name: [4]u8 = .{ 'k', 0, 0, 0 };
|
var name: [4]u8 = .{ 'k', 0, 0, 0 };
|
||||||
name[1] = '0' + @as(u8, @intCast(filled / 10));
|
name[1] = '0' + @as(u8, @intCast(filled / 10));
|
||||||
name[2] = '0' + @as(u8, @intCast(filled % 10));
|
name[2] = '0' + @as(u8, @intCast(filled % 10));
|
||||||
var extra = childDescriptor(name[0..3], parent_window.start, 0x20);
|
var extra = childDescriptor(name[0..3], parent_window.start, 0x20);
|
||||||
_ = devices_broker.register(parent_id, me, &extra) catch |err| {
|
if (devices_broker.register(parent_id, me, &extra)) |_| {
|
||||||
capped = err == error.TooManyChildren;
|
registered_children += 1;
|
||||||
break;
|
} else |_| break;
|
||||||
};
|
|
||||||
}
|
}
|
||||||
check("the parent reaches its child cap (TooManyChildren)", capped);
|
check("one parent takes far more children than the old cap allowed", registered_children == 64);
|
||||||
|
|
||||||
// The regression this ordering exists for: **a re-registration consumes no slot,
|
// Idempotency still holds, and still matters: a crashed bus driver is restarted and
|
||||||
// so a full parent must not refuse one.** A crashed bus driver is restarted by its
|
// re-registers everything it rediscovers, which must return the ids it had before
|
||||||
// supervisor and re-registers everything it rediscovers; when the cap was checked
|
// rather than duplicate them.
|
||||||
// before the identity match, the restart was refused its own devices and the
|
|
||||||
// machine degraded a little more on every crash.
|
|
||||||
const readmitted = devices_broker.register(parent_id, me, &fits) catch 0;
|
const readmitted = devices_broker.register(parent_id, me, &fits) catch 0;
|
||||||
check("a full parent still re-admits an identical child", readmitted != 0 and readmitted == good);
|
check("re-registering an identical child still returns its id", readmitted != 0 and readmitted == good);
|
||||||
|
|
||||||
// ...and the cap is genuinely still in force for anything new.
|
// The table itself has no ceiling: it grows. The old `maximum_devices = 64` was a
|
||||||
var novel = childDescriptor("knew", parent_window.start, 0x20);
|
// guess about someone else's computer, and one driver's enumeration starved every
|
||||||
const still_capped = if (devices_broker.register(parent_id, me, &novel)) |_| false else |err| err == error.TooManyChildren;
|
// other — which is how a Ryzen booted with no USB and no storage. What bounds a
|
||||||
check("a full parent still refuses a new child", still_capped);
|
// runaway now is an allowance charged to the registrar, so the damage stays with
|
||||||
|
// whoever caused it (docs/os-development/device-authority.md).
|
||||||
|
// The table grows: it starts at 8 entries and this boot holds well past that, so
|
||||||
|
// the growth path runs every time rather than lying dormant until someone else's
|
||||||
|
// larger machine finds it — which is how the old ceiling stayed invisible.
|
||||||
|
const held = devices_broker.enumerate(&buffer);
|
||||||
|
check("the table grew beyond its initial block", held > 8);
|
||||||
|
|
||||||
result();
|
result();
|
||||||
}
|
}
|
||||||
|
|
||||||
/// A minimal child descriptor with one memory resource, for the containment test.
|
/// A minimal child descriptor with one memory resource, for the containment test.
|
||||||
|
/// Delegation's mechanism. A claim is exclusive (driver-model.md, invariant 1), so
|
||||||
|
/// handing a device on is a **move**: the giver stops holding it the instant the
|
||||||
|
/// receiver starts. That is why this is not the M13 capability path, which shares a
|
||||||
|
/// handle refcounted.
|
||||||
|
///
|
||||||
|
/// The rule the kernel enforces is the whole of it: *you may give away what you hold*.
|
||||||
|
/// It has no idea which task is the device manager and needs none
|
||||||
|
/// (docs/os-development/device-authority.md).
|
||||||
|
fn deviceTransferTest(boot_information: *const boot_handoff.BootInformation) void {
|
||||||
|
var buffer: [8]device_abi.DeviceDescriptor = undefined;
|
||||||
|
check("the device tree is seeded", devices_broker.enumerate(&buffer) >= 2);
|
||||||
|
|
||||||
|
const image = bundledInit(boot_information) orelse {
|
||||||
|
check("initial_ramdisk carries /system/services/init", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
const me = scheduler.currentId();
|
||||||
|
const endpoint = ipcsync.createIpcEndpoint() orelse {
|
||||||
|
check("exit endpoint allocated", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
const child = process.spawnProcessSupervised(image, 4, &.{"/system/services/init"}, me, endpoint) catch 0;
|
||||||
|
check("supervised child spawned", child != 0);
|
||||||
|
|
||||||
|
// You may only give away what you hold — so an unheld device cannot be moved at all,
|
||||||
|
// which is what stops a transfer being a back door around claiming.
|
||||||
|
const unheld = if (devices_broker.transfer(0, me, child)) |_| false else |e| e == error.NotHeld;
|
||||||
|
check("an unheld device cannot be transferred", unheld);
|
||||||
|
|
||||||
|
// A second device this test never hands over, so the "no giver" half below is a
|
||||||
|
// real observation rather than a restatement of the first.
|
||||||
|
const parentless: u64 = 1;
|
||||||
|
|
||||||
|
check("claimed device 0", claimOk(0, me));
|
||||||
|
|
||||||
|
// The move itself.
|
||||||
|
const moved = if (devices_broker.transfer(0, me, child)) |_| true else |_| false;
|
||||||
|
check("the holder may transfer", moved);
|
||||||
|
const holder = devices_broker.ownerOf(0) orelse 0;
|
||||||
|
check("the receiver holds it", holder == child);
|
||||||
|
check("the giver does not", holder != me);
|
||||||
|
|
||||||
|
// Having given it away, the giver cannot give it again. This is the assertion that
|
||||||
|
// makes it a move rather than a copy.
|
||||||
|
const again = if (devices_broker.transfer(0, me, child)) |_| false else |e| e == error.NotHeld;
|
||||||
|
check("a former holder cannot transfer again", again);
|
||||||
|
|
||||||
|
// Nor may a stranger move a device it never held.
|
||||||
|
const stranger = if (devices_broker.transfer(0, 9999, me)) |_| false else |e| e == error.NotHeld;
|
||||||
|
check("a stranger cannot transfer another task's device", stranger);
|
||||||
|
|
||||||
|
// The giver is recorded. One field, and the authority rule follows from it: a
|
||||||
|
// device that was *given* to someone is delegated hardware, so it may be handed on
|
||||||
|
// but never taken, and when its holder dies it returns to whoever lent it instead
|
||||||
|
// of becoming free for anyone. Nothing has a giver until it is handed over —
|
||||||
|
// which is why the loader's framebuffer needs no exemption from either rule.
|
||||||
|
check("the giver is recorded on a transfer", devices_broker.giverOf(0) == me);
|
||||||
|
check("a device nobody handed over has no giver", devices_broker.giverOf(parentless) == null);
|
||||||
|
|
||||||
|
const absent = if (devices_broker.transfer(9999, me, child)) |_| false else |e| e == error.NoSuchDevice;
|
||||||
|
check("a device that does not exist is refused", absent);
|
||||||
|
|
||||||
|
// **A grant is a loan.** The child holds device 0 and `me` lent it, so the child's
|
||||||
|
// death must hand it back rather than release it to nobody. That is what lets the
|
||||||
|
// device manager re-delegate to a restarted driver — and, once `claim` refuses a
|
||||||
|
// device that has a giver, it is the only thing that stops a dead driver's hardware
|
||||||
|
// being stranded forever.
|
||||||
|
check("the child still holds the lent device", devices_broker.ownerOf(0) == child);
|
||||||
|
devices_broker.releaseAllOwnedBy(child);
|
||||||
|
check("the borrower's death returns the device to its lender", devices_broker.ownerOf(0) == me);
|
||||||
|
check("and it is no longer on loan", devices_broker.giverOf(0) == null);
|
||||||
|
|
||||||
|
// A device with no lender still simply frees on death, as it always did.
|
||||||
|
check("claimed a second device with no lender", claimOk(parentless, me));
|
||||||
|
devices_broker.releaseAllOwnedBy(me);
|
||||||
|
check("a device nobody lent is released outright", devices_broker.ownerOf(parentless) == null);
|
||||||
|
result();
|
||||||
|
}
|
||||||
|
|
||||||
/// PCI host-bridge apertures are derived from the *holes* in the firmware memory map,
|
/// PCI host-bridge apertures are derived from the *holes* in the firmware memory map,
|
||||||
/// and a registered BAR must fall inside one. So the invariant is not "we find the
|
/// and a registered BAR must fall inside one. So the invariant is not "we find the
|
||||||
/// holes" but "an aperture never covers memory the firmware described" — an aperture
|
/// holes" but "an aperture never covers memory the firmware described" — an aperture
|
||||||
@@ -4160,6 +4311,58 @@ fn protocolRegistryTest(boot_information: *const BootInformation) void {
|
|||||||
///
|
///
|
||||||
/// The fixture's `protocol-denied: ok` is the marker; each step prints its own
|
/// The fixture's `protocol-denied: ok` is the marker; each step prints its own
|
||||||
/// line, which the harness's ordered regex reads.
|
/// line, which the harness's ordered regex reads.
|
||||||
|
/// The attacker the device suite never had. The audit's finding was that a fully
|
||||||
|
/// green suite had missed six real defects because it *contains no attacker* — every
|
||||||
|
/// device case asserts a driver handed its hardware can drive it, and none asks what a
|
||||||
|
/// process handed **nothing** can do.
|
||||||
|
///
|
||||||
|
/// The fixture is spawned with no device and asserts what it therefore cannot do. It
|
||||||
|
/// runs without the device manager on purpose: nothing here needs a driver, and a boot
|
||||||
|
/// with fewer moving parts makes the refusals unambiguous.
|
||||||
|
fn deviceAuthorityTest(boot_information: *const BootInformation) void {
|
||||||
|
log("DANOS-TEST-BEGIN: device-authority\n", .{});
|
||||||
|
if (boot_information.initial_ramdisk_len == 0) {
|
||||||
|
check("bootloader handed over an initial_ramdisk", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||||
|
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||||
|
check("initial_ramdisk image is valid", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
|
||||||
|
process.setInitialRamdisk(image);
|
||||||
|
// The manager must be up: it is what holds the seeded hardware, and without it
|
||||||
|
// every device would be lying around unheld and the last assertion would have
|
||||||
|
// nothing to observe — a test that cannot fail.
|
||||||
|
check("registry (init) spawned", spawnRegistry(rd));
|
||||||
|
check("device-manager spawned", spawnNamed(rd, "device-manager"));
|
||||||
|
check("device-authority-test spawned", spawnNamedWithArg(rd, "device-authority-test", "run"));
|
||||||
|
|
||||||
|
// The VERDICT prefix matters: the fixture prints one "device-authority: ok <name>"
|
||||||
|
// line per assertion, so a marker of "device-authority: ok" matches the FIRST
|
||||||
|
// passing assertion and this loop exits before any later failure is printed — the
|
||||||
|
// case then passes with failures in it, which it did until this was caught.
|
||||||
|
const pass_marker = "device-authority: VERDICT ok";
|
||||||
|
const fail_marker = "device-authority: VERDICT FAILED";
|
||||||
|
scheduler.setPriority(1);
|
||||||
|
const deadline = architecture.millis() + 20000;
|
||||||
|
var saw_pass = false;
|
||||||
|
var saw_fail = false;
|
||||||
|
while (architecture.millis() < deadline and !saw_pass and !saw_fail) {
|
||||||
|
if (bufferHas(pass_marker)) saw_pass = true;
|
||||||
|
if (bufferHas(fail_marker)) saw_fail = true;
|
||||||
|
scheduler.yield();
|
||||||
|
}
|
||||||
|
scheduler.setPriority(4);
|
||||||
|
|
||||||
|
check("no authority assertion failed", !saw_fail);
|
||||||
|
check("the attacker completed every assertion", saw_pass);
|
||||||
|
result();
|
||||||
|
}
|
||||||
|
|
||||||
fn protocolDeniedTest(boot_information: *const BootInformation) void {
|
fn protocolDeniedTest(boot_information: *const BootInformation) void {
|
||||||
log("DANOS-TEST-BEGIN: protocol-denied\n", .{});
|
log("DANOS-TEST-BEGIN: protocol-denied\n", .{});
|
||||||
if (boot_information.initial_ramdisk_len == 0) {
|
if (boot_information.initial_ramdisk_len == 0) {
|
||||||
|
|||||||
+18
-2
@@ -17,8 +17,13 @@
|
|||||||
/// arrays (discovery pool, scheduler state, per-core GDT/TSS). Generous headroom:
|
/// arrays (discovery pool, scheduler state, per-core GDT/TSS). Generous headroom:
|
||||||
/// those structs are small, and the *large* per-core resources (kernel and IST
|
/// those structs are small, and the *large* per-core resources (kernel and IST
|
||||||
/// stacks) are allocated at bring-up for cores that actually come online, so this
|
/// stacks) are allocated at bring-up for cores that actually come online, so this
|
||||||
/// ceiling is cheap. A machine with more logical CPUs has its surplus reported and
|
/// ceiling is cheap.
|
||||||
/// left parked (see acpi `cpusDropped`).
|
///
|
||||||
|
/// bound: logical CPUs the kernel tracks
|
||||||
|
/// decided-by: hardware
|
||||||
|
/// protects: the per-CPU bookkeeping arrays, which are sized at compile time
|
||||||
|
/// at-limit: degrade — the surplus cores are left parked, never brought online
|
||||||
|
/// observed-by: platform.cpusDropped() -> the WARNING at kernel.zig:281
|
||||||
pub const maximum_cpus = 128;
|
pub const maximum_cpus = 128;
|
||||||
|
|
||||||
/// Maximum tasks (kernel threads) alive at once — the static task-table size. Each
|
/// Maximum tasks (kernel threads) alive at once — the static task-table size. Each
|
||||||
@@ -29,6 +34,17 @@ pub const maximum_cpus = 128;
|
|||||||
/// for the USB stack: the xHCI bus driver spawns a supervised class-driver instance
|
/// for the USB stack: the xHCI bus driver spawns a supervised class-driver instance
|
||||||
/// per matched interface (keyboard, mouse, mass storage), on top of the FAT and
|
/// per matched interface (keyboard, mouse, mass storage), on top of the FAT and
|
||||||
/// block servers and the growing ramdisk bundle.
|
/// block servers and the growing ramdisk bundle.
|
||||||
|
///
|
||||||
|
/// The history above is the argument against this number: it has been raised twice,
|
||||||
|
/// each time by a machine or a bundle that outgrew it, which is the pattern the
|
||||||
|
/// bounds rule exists to stop. It is `ours` only because the task table is static;
|
||||||
|
/// how many drivers a machine needs is decided by how much hardware it has.
|
||||||
|
///
|
||||||
|
/// bound: kernel threads alive at once — the static task-table size
|
||||||
|
/// decided-by: hardware
|
||||||
|
/// protects: the statically allocated task table
|
||||||
|
/// at-limit: refuse — spawn fails; a supervised driver is never started
|
||||||
|
/// observed-by: the spawning supervisor's own log line; see docs/bounds-track-plan.md
|
||||||
pub const maximum_tasks = 48;
|
pub const maximum_tasks = 48;
|
||||||
|
|
||||||
/// Each task's kernel stack (also each AP's bring-up stack), in bytes.
|
/// Each task's kernel stack (also each AP's bring-up stack), in bytes.
|
||||||
|
|||||||
@@ -104,11 +104,19 @@ fn findTablesNode(buffer: []device.DeviceDescriptor) ?device.DeviceDescriptor {
|
|||||||
}
|
}
|
||||||
|
|
||||||
pub fn main(init: process.Init) void {
|
pub fn main(init: process.Init) void {
|
||||||
// When the acpi-parse scenario spawns this directly, argv[1] is a device-count
|
// When the acpi-parse scenario spawns this directly, it passes `floor:N` — a
|
||||||
// *floor* to self-verify against. The kernel no longer parses AML, so there is
|
// device-count floor to self-verify against. The kernel no longer parses AML, so
|
||||||
// no exact count to match — proving the ring-3 parse found at least a floor of
|
// there is no exact count to match; proving the ring-3 parse found at least N
|
||||||
// devices is the check. Deterministic, no log-scraping.
|
// Device objects is the check. Deterministic, no log-scraping.
|
||||||
const floor: ?usize = if (init.arguments.get(1)) |a| (std.fmt.parseInt(usize, a, 10) catch null) else null;
|
//
|
||||||
|
// The `floor:` prefix matters. argv[1] is the assigned device id for every driver
|
||||||
|
// the manager spawns, so a bare number here would be read as a floor — which is
|
||||||
|
// exactly what happened when discovery started being given its node: it saw
|
||||||
|
// argv[1] = "7", decided it was in self-verify mode, and never reported a device.
|
||||||
|
const floor: ?usize = if (init.arguments.get(1)) |a| blk: {
|
||||||
|
if (!std.mem.startsWith(u8, a, "floor:")) break :blk null;
|
||||||
|
break :blk std.fmt.parseInt(usize, a["floor:".len..], 10) catch null;
|
||||||
|
} else null;
|
||||||
|
|
||||||
const buffer = memory.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
const buffer = memory.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||||
_ = logging.write("/system/services/acpi: out of memory\n");
|
_ = logging.write("/system/services/acpi: out of memory\n");
|
||||||
@@ -118,11 +126,11 @@ pub fn main(init: process.Init) void {
|
|||||||
_ = logging.write("/system/services/acpi: no acpi-tables node to claim\n");
|
_ = logging.write("/system/services/acpi: no acpi-tables node to claim\n");
|
||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
|
// The node arrived with the spawn: the manager holds it and names it in the call
|
||||||
|
// that creates this process. Discovery was the last thing in the system that
|
||||||
|
// acquired hardware by naming it rather than being given it
|
||||||
|
// (docs/os-development/device-authority.md).
|
||||||
node_id = node.id;
|
node_id = node.id;
|
||||||
device.claim(node_id) catch |e| {
|
|
||||||
std.log.warn("unable to claim acpi-tables: {s}", .{@errorName(e)});
|
|
||||||
return;
|
|
||||||
};
|
|
||||||
|
|
||||||
// Map the node's resources: the AML blobs (bytecode), the FADT (intact
|
// Map the node's resources: the AML blobs (bytecode), the FADT (intact
|
||||||
// "FACP" header — decision 3), the io_port grant, and the SCI irq.
|
// "FACP" header — decision 3), the io_port grant, and the SCI irq.
|
||||||
|
|||||||
@@ -220,6 +220,13 @@ fn childCountOf(reporter: u32) u32 {
|
|||||||
/// The driver entry a live process id belongs to. Zero is not a process id here:
|
/// The driver entry a live process id belongs to. Zero is not a process id here:
|
||||||
/// it is what `onDriverExit` writes back to retire an id it has already acted on,
|
/// it is what `onDriverExit` writes back to retire an id it has already acted on,
|
||||||
/// so a second notification for the same death matches nothing.
|
/// so a second notification for the same death matches nothing.
|
||||||
|
fn driverByName(name: []const u8) ?*Driver {
|
||||||
|
for (&drivers) |*driver| {
|
||||||
|
if (driver.used and std.mem.eql(u8, driver.name(), name)) return driver;
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
fn driverByProcess(process_id: u32) ?*Driver {
|
fn driverByProcess(process_id: u32) ?*Driver {
|
||||||
if (process_id == 0) return null;
|
if (process_id == 0) return null;
|
||||||
for (&drivers) |*driver| {
|
for (&drivers) |*driver| {
|
||||||
@@ -257,6 +264,22 @@ fn addDriver(name: []const u8, device_id: u64, speaks_protocol: bool) void {
|
|||||||
/// device id as argv[1] when it has one, the hello deadline armed when it
|
/// device id as argv[1] when it has one, the hello deadline armed when it
|
||||||
/// speaks the protocol.
|
/// speaks the protocol.
|
||||||
fn spawnDriver(driver: *Driver) void {
|
fn spawnDriver(driver: *Driver) void {
|
||||||
|
// Take the device before the driver exists, so there is no window in which anyone
|
||||||
|
// else could claim it — which is the whole of what makes the handover authoritative
|
||||||
|
// rather than advisory. Re-claiming across a restart is expected to say
|
||||||
|
// AlreadyClaimed once the manager already holds it, and that is fine: it means the
|
||||||
|
// device never left our hands while the driver was dead.
|
||||||
|
if (driver.device_id != device_manager_protocol.no_device) {
|
||||||
|
device.claim(driver.device_id) catch |e| switch (e) {
|
||||||
|
error.AlreadyClaimed => {}, // ours already, from a previous spawn of this driver
|
||||||
|
else => {
|
||||||
|
std.log.warn("cannot hold device {d} for {s}: {s}", .{ driver.device_id, driver.name(), @errorName(e) });
|
||||||
|
driver.state = .failed;
|
||||||
|
return;
|
||||||
|
},
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
var id_text: [20]u8 = undefined;
|
var id_text: [20]u8 = undefined;
|
||||||
var arguments: [1][]const u8 = undefined;
|
var arguments: [1][]const u8 = undefined;
|
||||||
var argument_count: usize = 0;
|
var argument_count: usize = 0;
|
||||||
@@ -264,7 +287,14 @@ fn spawnDriver(driver: *Driver) void {
|
|||||||
arguments[0] = std.fmt.bufPrint(&id_text, "{d}", .{driver.device_id}) catch return;
|
arguments[0] = std.fmt.bufPrint(&id_text, "{d}", .{driver.device_id}) catch return;
|
||||||
argument_count = 1;
|
argument_count = 1;
|
||||||
}
|
}
|
||||||
const child = process.spawnSupervised(driver.name(), arguments[0..argument_count], manager_endpoint) orelse {
|
// The device rides the spawn, so the driver holds it before its first instruction.
|
||||||
|
// A transfer *after* spawning would leave a window in which the child is running
|
||||||
|
// without its hardware — closed on one machine, open on another
|
||||||
|
// (docs/bounds-track-plan.md, "the grant rides system_spawn").
|
||||||
|
const give = driver.device_id;
|
||||||
|
if (give != device_manager_protocol.no_device)
|
||||||
|
std.log.info("delegated device {d} to {s}", .{ give, driver.name() });
|
||||||
|
const child = process.spawnSupervisedWithDevice(driver.name(), arguments[0..argument_count], manager_endpoint, give) orelse {
|
||||||
std.log.info("failed to spawn {s}", .{driver.name()});
|
std.log.info("failed to spawn {s}", .{driver.name()});
|
||||||
driver.state = .failed;
|
driver.state = .failed;
|
||||||
return;
|
return;
|
||||||
@@ -362,6 +392,40 @@ fn initialise(endpoint: ipc.Handle) bool {
|
|||||||
const total = device.enumerate(buffer);
|
const total = device.enumerate(buffer);
|
||||||
const count = @min(total, buffer.len);
|
const count = @min(total, buffer.len);
|
||||||
|
|
||||||
|
// The node discovery needs: the kernel seeds it, so it is in this same snapshot
|
||||||
|
// and can be handed over like any other assignment. Discovery used to find and
|
||||||
|
// claim it itself — the last driver that acquired hardware by naming it rather
|
||||||
|
// than being given it (docs/os-development/device-authority.md).
|
||||||
|
var tables_node: u64 = device_manager_protocol.no_device;
|
||||||
|
for (buffer[0..count]) |descriptor| {
|
||||||
|
if (descriptor.class == @intFromEnum(device.DeviceClass.acpi_tables)) {
|
||||||
|
tables_node = descriptor.id;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// **Hold the firmware-discovered hardware, so none of it is left lying around.**
|
||||||
|
// A device nobody holds can be claimed by anyone, so every seeded device that
|
||||||
|
// carries mappable resources is taken here whether or not a driver wants it — the
|
||||||
|
// HPET most of all, which has an MMIO window and an IRQ and no user-space driver.
|
||||||
|
// Held by the manager it is inert; unheld it was there for the taking.
|
||||||
|
//
|
||||||
|
// Two deliberate exclusions:
|
||||||
|
// - the loader's framebuffer, which the compositor claims and which is not
|
||||||
|
// hardware anyone is delegated (the manager starts before display, so taking
|
||||||
|
// it here would break the boot screen);
|
||||||
|
// - anything with no resources, which grants nothing and so is not worth holding.
|
||||||
|
//
|
||||||
|
// This covers the boot snapshot only. A device *reported* later and matched to no
|
||||||
|
// driver stays claimable — the pci-cap and iommu-fault fixtures rely on exactly
|
||||||
|
// that to reach an unmatched NIC. Narrowing it further is a separate change with
|
||||||
|
// those fixtures in scope.
|
||||||
|
for (buffer[0..count]) |descriptor| {
|
||||||
|
if (descriptor.resource_count == 0) continue;
|
||||||
|
if (descriptor.class == @intFromEnum(device.DeviceClass.display)) continue;
|
||||||
|
device.claim(descriptor.id) catch continue; // already held, or not ours to take
|
||||||
|
}
|
||||||
|
|
||||||
var matched: usize = 0;
|
var matched: usize = 0;
|
||||||
for (buffer[0..count]) |descriptor| {
|
for (buffer[0..count]) |descriptor| {
|
||||||
if (descriptor.class == @intFromEnum(device.DeviceClass.pci_host_bridge)) {
|
if (descriptor.class == @intFromEnum(device.DeviceClass.pci_host_bridge)) {
|
||||||
@@ -383,7 +447,7 @@ fn initialise(endpoint: ipc.Handle) bool {
|
|||||||
// under the neutral name "discovery", spawned once at startup. It finds and
|
// under the neutral name "discovery", spawned once at startup. It finds and
|
||||||
// claims the acpi-tables (or devicetree-blob) node itself. Not a per-device
|
// claims the acpi-tables (or devicetree-blob) node itself. Not a per-device
|
||||||
// match — it is the discoverer, not a driver bound to one device.
|
// match — it is the discoverer, not a driver bound to one device.
|
||||||
addDriver("discovery", device_manager_protocol.no_device, false);
|
addDriver("discovery", tables_node, false);
|
||||||
|
|
||||||
if (test_restart_mode) {
|
if (test_restart_mode) {
|
||||||
// The driver-restart scenario's fixture: claims device 0 (the tree
|
// The driver-restart scenario's fixture: claims device 0 (the tree
|
||||||
@@ -424,6 +488,7 @@ fn onHello(_: void, invocation: Invocation(device_manager_protocol.Hello), _: An
|
|||||||
return -envelope.EPERM;
|
return -envelope.EPERM;
|
||||||
};
|
};
|
||||||
driver.state = .running;
|
driver.state = .running;
|
||||||
|
|
||||||
std.log.info("hello from {s} (device {d})", .{ driver.name(), invocation.target });
|
std.log.info("hello from {s} (device {d})", .{ driver.name(), invocation.target });
|
||||||
// Resilience drill (V6): once, kill the virtio-gpu driver a moment after it hellos, so
|
// Resilience drill (V6): once, kill the virtio-gpu driver a moment after it hellos, so
|
||||||
// the normal restart policy respawns it — the compositor must survive and re-attach.
|
// the normal restart policy respawns it — the compositor must survive and re-attach.
|
||||||
@@ -461,9 +526,36 @@ fn onChildAdded(_: void, invocation: Invocation(device_manager_protocol.ChildAdd
|
|||||||
if (match.ambiguous)
|
if (match.ambiguous)
|
||||||
std.log.info("/system/configuration/devices.csv: multiple equally-specific rules match the device {s} reported; binding {s}", .{ driver.name(), match.driver });
|
std.log.info("/system/configuration/devices.csv: multiple equally-specific rules match the device {s} reported; binding {s}", .{ driver.name(), match.driver });
|
||||||
if (id.bus == .acpi) {
|
if (id.bus == .acpi) {
|
||||||
// An hid-matched driver (ps2-bus) is a singleton that finds its
|
// An hid-matched driver is a singleton over one piece of hardware
|
||||||
// own devices once spawned — spawn it once, no device assignment.
|
// described by several nodes: the 8042 is a single controller whose
|
||||||
if (!alreadySupervised(match.driver)) addDriver(match.driver, device_manager_protocol.no_device, false);
|
// I/O ports live under the keyboard node (PNP0303) while the mouse
|
||||||
|
// is a second node (PNP0F13). Two processes would fight over the
|
||||||
|
// same 0x60/0x64 registers, so there is exactly one instance — and
|
||||||
|
// it needs *every* matching device, however many the machine has
|
||||||
|
// (some have none, some one port, some two).
|
||||||
|
//
|
||||||
|
// The first rides the spawn; the rest are transferred to the
|
||||||
|
// running instance. Late arrival is fine here because the order is
|
||||||
|
// natural: the instance needs the controller node immediately and
|
||||||
|
// reaches the mouse only after the 8042 handshakes and identify.
|
||||||
|
if (!alreadySupervised(match.driver)) {
|
||||||
|
addDriver(match.driver, device_id, false);
|
||||||
|
} else if (driverByName(match.driver)) |running| {
|
||||||
|
if (running.process_id != 0) {
|
||||||
|
device.claim(device_id) catch |e| switch (e) {
|
||||||
|
error.AlreadyClaimed => {},
|
||||||
|
else => {
|
||||||
|
std.log.warn("cannot hold device {d} for {s}: {s}", .{ device_id, match.driver, @errorName(e) });
|
||||||
|
return status;
|
||||||
|
},
|
||||||
|
};
|
||||||
|
device.transfer(device_id, running.process_id) catch |e| {
|
||||||
|
std.log.warn("could not give device {d} to {s}: {s}", .{ device_id, match.driver, @errorName(e) });
|
||||||
|
return status;
|
||||||
|
};
|
||||||
|
std.log.info("delegated device {d} to {s} (already running)", .{ device_id, match.driver });
|
||||||
|
}
|
||||||
|
}
|
||||||
} else {
|
} else {
|
||||||
// A per-device driver: one instance, the registered id as argv[1].
|
// A per-device driver: one instance, the registered id as argv[1].
|
||||||
if (!driverForDevice(device_id)) addDriver(match.driver, device_id, true);
|
if (!driverForDevice(device_id)) addDriver(match.driver, device_id, true);
|
||||||
|
|||||||
+69
-4
@@ -625,11 +625,24 @@ CASES = [
|
|||||||
# manager spawn the USB keyboard driver, which opens its device over the
|
# manager spawn the USB keyboard driver, which opens its device over the
|
||||||
# transfer protocol, asks for boot protocol, subscribes to its interrupt
|
# transfer protocol, asks for boot protocol, subscribes to its interrupt
|
||||||
# endpoint, and comes up — proof the class-driver <-> controller path works.
|
# endpoint, and comes up — proof the class-driver <-> controller path works.
|
||||||
|
#
|
||||||
|
# It also pins the slot count to the hardware's answer. The backreference is the
|
||||||
|
# assertion: the driver must track exactly as many device slots as the controller
|
||||||
|
# reports in HCSPARAMS1.MaxSlots. It tracked a fixed 8 while QEMU's xHCI reports
|
||||||
|
# 64, so seven eighths of the controller was invisible and a device behind a hub
|
||||||
|
# past the eighth vanished without a log line. \1 fails the moment they diverge.
|
||||||
{"name": "usb-hid",
|
{"name": "usb-hid",
|
||||||
"smp": 4,
|
"smp": 4,
|
||||||
"timeout": 150,
|
"timeout": 150,
|
||||||
# usb-kbd/usb-mouse ride the default boot xHCI bus (see qemu_args).
|
# usb-kbd/usb-mouse ride the default boot xHCI bus (see qemu_args).
|
||||||
"expect": r"(?=[\s\S]*usb-hid-keyboard: ok)(?=[\s\S]*usb-hid-mouse: ok)",
|
# The controller must arrive by DELEGATION, not by claiming: the manager holds it
|
||||||
|
# and transfers it in the hello reply, so the match is authoritative rather than
|
||||||
|
# advisory (docs/os-development/device-authority.md). The delegation line must
|
||||||
|
# precede the hello, because the transfer completes before the reply lands.
|
||||||
|
"expect": r"(?=[\s\S]*device-manager: delegated device (\d+) to /system/drivers/usb-xhci-bus"
|
||||||
|
r"[\s\S]*usb-xhci-bus: controller device \1 registers at)"
|
||||||
|
r"(?=[\s\S]*usb-xhci-bus: controller running \((\d+) slots, tracking \2,)"
|
||||||
|
r"(?=[\s\S]*usb-hid-keyboard: ok)(?=[\s\S]*usb-hid-mouse: ok)",
|
||||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
# Keyboard echo: inject a known phrase via QMP send-key; the usb-hid-keyboard
|
# Keyboard echo: inject a known phrase via QMP send-key; the usb-hid-keyboard
|
||||||
# driver decodes it and echoes each character to the log (the simple
|
# driver decodes it and echoes each character to the log (the simple
|
||||||
@@ -667,6 +680,28 @@ CASES = [
|
|||||||
r"(?=.*hub slot \d+ port \d+ device:.*0x0627)"
|
r"(?=.*hub slot \d+ port \d+ device:.*0x0627)"
|
||||||
r"(?=.*usb-hid-keyboard: ok \(device 3)",
|
r"(?=.*usb-hid-keyboard: ok \(device 3)",
|
||||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
|
# The driver must never truncate a configuration block. Its size is the DEVICE's
|
||||||
|
# choice (wTotalLength, a u16); the driver used to read the first 512 bytes into a
|
||||||
|
# fixed buffer and parse those, so interfaces past the cut did not exist while
|
||||||
|
# SET_CONFIGURATION still configured the device for all of them.
|
||||||
|
#
|
||||||
|
# The backreference is the assertion: declared length and bytes read must match.
|
||||||
|
#
|
||||||
|
# Honest limit: QEMU cannot produce a block over 512 bytes. Its boot keyboard,
|
||||||
|
# mouse and stick are 34-44, and the largest device on offer is usb-audio in
|
||||||
|
# multi-channel mode at 211 — which is why the suite never saw the original bug,
|
||||||
|
# and why it cannot now reproduce that exact trigger. What this case does catch is
|
||||||
|
# the class: any clamp below the attached device's block size fails it, verified
|
||||||
|
# by pinning the buffer to 128 and watching it go red. A real headset (500-900
|
||||||
|
# bytes), UVC webcam (1-3 KB) or multifunction printer trips the 512 itself.
|
||||||
|
{"name": "usb-large-descriptor",
|
||||||
|
"build_case": "usb-hid",
|
||||||
|
"smp": 4,
|
||||||
|
"timeout": 150,
|
||||||
|
"qemu_extra": ["-device", "qemu-xhci,id=xhci2",
|
||||||
|
"-device", "usb-audio,bus=xhci2.0,port=1,multi=on"],
|
||||||
|
"expect": r"(?s)(?=.*usb-xhci-bus: config block (\d\d\d+) bytes, read \1)",
|
||||||
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
# USB mass storage end to end: the boot usb-storage device (the FAT32 image,
|
# USB mass storage end to end: the boot usb-storage device (the FAT32 image,
|
||||||
# which has a real 0x55AA boot sector) is enough — the manager spawns
|
# which has a real 0x55AA boot sector) is enough — the manager spawns
|
||||||
# usb-storage, which opens the device, runs the Bulk-Only / SCSI bring-up,
|
# usb-storage, which opens the device, runs the Bulk-Only / SCSI bring-up,
|
||||||
@@ -745,8 +780,16 @@ CASES = [
|
|||||||
{"name": "acpi-ps2",
|
{"name": "acpi-ps2",
|
||||||
"smp": 4,
|
"smp": 4,
|
||||||
"timeout": 150,
|
"timeout": 150,
|
||||||
|
# The 8042 is ONE controller described by two ACPI nodes, so the single ps2-bus
|
||||||
|
# instance must receive BOTH. The keyboard node rides the spawn — it carries the
|
||||||
|
# 0x60/0x64 ports and is needed immediately — and the mouse node is transferred to
|
||||||
|
# the already-running instance, which is safe because it is not touched until after
|
||||||
|
# the controller handshakes and identify. Two processes cannot split this: they
|
||||||
|
# would fight over the same registers.
|
||||||
"expect": r"discovery: device \d+\s+bus=acpi hid=PNP0303[\s\S]*"
|
"expect": r"discovery: device \d+\s+bus=acpi hid=PNP0303[\s\S]*"
|
||||||
r"device-manager: spawned \S*ps2-bus[\s\S]*"
|
r"device-manager: delegated device (\d+) to \S*ps2-bus[\s\S]*"
|
||||||
|
r"device-manager: spawned \S*ps2-bus for device \1[\s\S]*"
|
||||||
|
r"device-manager: delegated device \d+ to \S*ps2-bus \(already running\)[\s\S]*"
|
||||||
r"ps2-bus: keyboard driver attached",
|
r"ps2-bus: keyboard driver attached",
|
||||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
# M21.1: the SCI + power button. Boot the manager (which spawns the acpi
|
# M21.1: the SCI + power button. Boot the manager (which spawns the acpi
|
||||||
@@ -819,10 +862,15 @@ CASES = [
|
|||||||
{"name": "pci-scan",
|
{"name": "pci-scan",
|
||||||
"smp": 4,
|
"smp": 4,
|
||||||
"timeout": 60,
|
"timeout": 60,
|
||||||
"expect": r"pci-bus: (\d+) functions found[\s\S]*"
|
# The bridge must arrive by DELEGATION, not by claiming — and again after the
|
||||||
|
# restart, which is what proves the manager re-takes the device when its driver
|
||||||
|
# dies and hands it to the replacement. \1 pins it to the same device both times.
|
||||||
|
"expect": r"device-manager: delegated device (\d+) to \S*pci-bus[\s\S]*"
|
||||||
|
r"pci-bus: (\d+) functions found[\s\S]*"
|
||||||
r"device-manager: test mode: killing the reporter[\s\S]*"
|
r"device-manager: test mode: killing the reporter[\s\S]*"
|
||||||
r"device-manager: restarting \S*pci-bus[\s\S]*"
|
r"device-manager: restarting \S*pci-bus[\s\S]*"
|
||||||
r"pci-bus: \1 functions found[\s\S]*"
|
r"device-manager: delegated device \1 to \S*pci-bus[\s\S]*"
|
||||||
|
r"pci-bus: \2 functions found[\s\S]*"
|
||||||
r"DANOS-TEST-RESULT: PASS",
|
r"DANOS-TEST-RESULT: PASS",
|
||||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
# M18.3: the application surface — device-list enumerates the tree over IPC,
|
# M18.3: the application surface — device-list enumerates the tree over IPC,
|
||||||
@@ -992,6 +1040,23 @@ CASES = [
|
|||||||
{"name": "apertures",
|
{"name": "apertures",
|
||||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
|
# Delegation's mechanism (docs/os-development/device-authority.md). A claim is
|
||||||
|
# exclusive, so handing a device on is a MOVE: the giver stops holding it the
|
||||||
|
# instant the receiver starts - which is why this is not the M13 capability path,
|
||||||
|
# where a handle is shared refcounted. The kernel's whole rule is "you may give
|
||||||
|
# away what you hold"; it has no notion of which task is the device manager.
|
||||||
|
{"name": "device-transfer",
|
||||||
|
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||||
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
|
# The attacker the device suite never had. The audit's finding was that a fully
|
||||||
|
# green suite missed six real defects because it contains no attacker: every device
|
||||||
|
# case asserts a driver handed its hardware can drive it, and none asks what a
|
||||||
|
# process handed NOTHING can do. This fixture is that process - it holds no device
|
||||||
|
# and asserts it can give none away, with a positive control first so the refusals
|
||||||
|
# are decisions rather than a broken syscall path.
|
||||||
|
{"name": "device-authority",
|
||||||
|
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||||
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
# IRQ teardown: an exiting driver's line is masked and its slot cleared (so no
|
# IRQ teardown: an exiting driver's line is masked and its slot cleared (so no
|
||||||
# ISR notifies a freed endpoint), and a sibling owner sharing that endpoint
|
# ISR notifies a freed endpoint), and a sibling owner sharing that endpoint
|
||||||
# keeps its own binding. A long-running driver never reaches this teardown path.
|
# keeps its own binding. A long-running driver never reaches this teardown path.
|
||||||
|
|||||||
@@ -19,13 +19,15 @@ pub fn main(init: process.Init) void {
|
|||||||
const argument = init.arguments.get(1) orelse return; // bare: stay silent
|
const argument = init.arguments.get(1) orelse return; // bare: stay silent
|
||||||
const assigned = std.fmt.parseInt(u64, argument, 10) catch return;
|
const assigned = std.fmt.parseInt(u64, argument, 10) catch return;
|
||||||
|
|
||||||
// The respawn only reaches this line because the kernel released the
|
// The device arrived with the spawn — this fixture is delegated its hardware like
|
||||||
// previous instance's claim at death. A failed claim exits cleanly — the
|
// any other driver, so it holds `assigned` before its first instruction and has
|
||||||
// manager reads "meant to stop" and the scenario fails loudly by silence.
|
// nothing to claim (docs/os-development/device-authority.md).
|
||||||
device.claim(assigned) catch {
|
//
|
||||||
_ = logging.write("crash-test: claim failed\n");
|
// The property this scenario checks is unchanged, only its mechanism: a respawned
|
||||||
return;
|
// instance still gets the device its predecessor held. It used to arrive because
|
||||||
};
|
// the kernel released the dead instance's claim and this one re-took it, racing
|
||||||
|
// anyone else who wanted it; now the device reverts to the manager on death and is
|
||||||
|
// handed to the replacement, which is the same guarantee without the race.
|
||||||
|
|
||||||
var manager: ?ipc.Handle = null;
|
var manager: ?ipc.Handle = null;
|
||||||
var tries: u32 = 0;
|
var tries: u32 = 0;
|
||||||
|
|||||||
@@ -0,0 +1,15 @@
|
|||||||
|
//! The device-authority-test fixture as a binary package (docs/build-packages-plan.md):
|
||||||
|
//! this file names the binary and EXACTLY the modules its source imports —
|
||||||
|
//! build-support resolves each name from the domains this zon declares.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const build_support = @import("build-support");
|
||||||
|
|
||||||
|
pub fn build(b: *std.Build) void {
|
||||||
|
const exe = build_support.userBinary(b, .{
|
||||||
|
.name = "device-authority-test",
|
||||||
|
.root_source_file = b.path("device-authority-test.zig"),
|
||||||
|
.imports = &.{ "driver", "logging", "process", "time" },
|
||||||
|
});
|
||||||
|
b.installArtifact(exe);
|
||||||
|
}
|
||||||
@@ -0,0 +1,15 @@
|
|||||||
|
.{
|
||||||
|
.name = .device_authority_test,
|
||||||
|
.version = "0.0.0",
|
||||||
|
.fingerprint = 0x4acbba0c105a1462, // Changing this has security and trust implications.
|
||||||
|
.minimum_zig_version = "0.16.0",
|
||||||
|
.dependencies = .{
|
||||||
|
// build-support supplies the shared recipe; kernel is implicit in
|
||||||
|
// every binary (the root shim + link script live there). The rest
|
||||||
|
// are exactly the homes of this binary's declared imports.
|
||||||
|
.@"build-support" = .{ .path = "../../../../build-support" },
|
||||||
|
.kernel = .{ .path = "../../../../library/kernel" },
|
||||||
|
.device = .{ .path = "../../../../library/device" },
|
||||||
|
},
|
||||||
|
.paths = .{""},
|
||||||
|
}
|
||||||
@@ -0,0 +1,124 @@
|
|||||||
|
//! device-authority-test — the attacker the device suite never had.
|
||||||
|
//!
|
||||||
|
//! The audit behind [docs/fixed-bounds-audit.md] found six real defects that a
|
||||||
|
//! fully green suite had missed, and the reason was structural: *the suite
|
||||||
|
//! contains no attacker*. Every device case asserts that a driver handed its
|
||||||
|
//! own hardware can drive it. None asks what a process that was handed
|
||||||
|
//! **nothing** can do.
|
||||||
|
//!
|
||||||
|
//! This binary is that process. It is spawned with no device, holds no device,
|
||||||
|
//! and asserts what it therefore cannot do
|
||||||
|
//! ([docs/os-development/device-authority.md]):
|
||||||
|
//!
|
||||||
|
//! 1. **A positive control first.** `device_enumerate` works from here, so
|
||||||
|
//! the refusals below are decisions rather than a syscall path that is
|
||||||
|
//! simply broken for this process. Without this, "everything failed" would
|
||||||
|
//! read identically to "the assertions are meaningless".
|
||||||
|
//! 2. **It cannot give away a device it does not hold** — not one another
|
||||||
|
//! task holds, and not a free one either. The kernel's whole rule is *you
|
||||||
|
//! may give away what you hold*, so the state of the device is irrelevant:
|
||||||
|
//! a process holding nothing can transfer nothing. That is asserted across
|
||||||
|
//! several ids precisely so it cannot pass by accident of which device
|
||||||
|
//! happened to be free at boot.
|
||||||
|
//! 3. **A device that does not exist is refused differently** — `NoSuchDevice`
|
||||||
|
//! rather than `NotHeld`. A refusal that cannot say which rule refused it
|
||||||
|
//! is what cost a debugging session on the Ryzen, so the distinction is
|
||||||
|
//! part of the contract and is tested as such.
|
||||||
|
//!
|
||||||
|
//! **Why there is no "cannot take a delegated device" assertion here.** The
|
||||||
|
//! hole this fixture was written for is closed, but not by a refusal it could
|
||||||
|
//! observe. A device that was given to someone is *held*, so an attempt to
|
||||||
|
//! take it is refused as `AlreadyClaimed` — the same answer as before. What
|
||||||
|
//! changed is what happens when the holder dies: the device returns to
|
||||||
|
//! whoever lent it instead of becoming free, so the window in which a
|
||||||
|
//! stranger could take it no longer exists. There is no moment to catch.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const device = @import("driver");
|
||||||
|
const logging = @import("logging");
|
||||||
|
const process = @import("process");
|
||||||
|
const time = @import("time");
|
||||||
|
|
||||||
|
fn line(comptime format: []const u8, arguments: anytype) void {
|
||||||
|
var buffer: [160]u8 = undefined;
|
||||||
|
_ = logging.write(std.fmt.bufPrint(&buffer, format, arguments) catch return);
|
||||||
|
}
|
||||||
|
|
||||||
|
var failures: usize = 0;
|
||||||
|
var process_table: [64]process.ProcessDescriptor = undefined;
|
||||||
|
|
||||||
|
fn check(name: []const u8, ok: bool) void {
|
||||||
|
if (!ok) failures += 1;
|
||||||
|
line("device-authority: {s} {s}\n", .{ if (ok) "ok" else "FAIL", name });
|
||||||
|
}
|
||||||
|
|
||||||
|
fn run() void {
|
||||||
|
// 1. The positive control: this process can reach the device syscalls at all.
|
||||||
|
var table: [64]device.DeviceDescriptor = undefined;
|
||||||
|
const total = device.enumerate(&table);
|
||||||
|
check("enumerate works from an unprivileged process", total > 0);
|
||||||
|
const seen = @min(total, table.len);
|
||||||
|
|
||||||
|
// 2. Holding nothing, it can give nothing away — whatever the device's state.
|
||||||
|
// Every id the machine actually has, so this cannot pass by luck.
|
||||||
|
var refused: usize = 0;
|
||||||
|
var wrong_reason: usize = 0;
|
||||||
|
for (table[0..seen]) |descriptor| {
|
||||||
|
device.transfer(descriptor.id, process.taskId()) catch |e| {
|
||||||
|
refused += 1;
|
||||||
|
if (e != error.NotHeld) wrong_reason += 1;
|
||||||
|
continue;
|
||||||
|
};
|
||||||
|
}
|
||||||
|
check("every transfer by a non-holder is refused", refused == seen);
|
||||||
|
check("each refusal says NotHeld, not something vaguer", wrong_reason == 0);
|
||||||
|
|
||||||
|
// 3. A device that does not exist is a different refusal, and says so.
|
||||||
|
const absent = if (device.transfer(0xFFFF_FFFF, process.taskId())) |_| false else |e| e == error.NoSuchDevice;
|
||||||
|
check("a device that does not exist is refused as absent", absent);
|
||||||
|
|
||||||
|
// 4. **The spawn is not a second way in.** A device now rides system_spawn, which
|
||||||
|
// would be a fine back door if the kernel checked ownership any less carefully
|
||||||
|
// there than it does in transfer: spawn a child, name someone else's device, and
|
||||||
|
// the child holds hardware nobody gave it. The refusal must happen before the
|
||||||
|
// child exists, so nothing is left running either.
|
||||||
|
if (seen != 0) {
|
||||||
|
const before = process.processes(&process_table);
|
||||||
|
const spawned = process.spawnSupervisedWithDevice("/test/system/services/device-authority-test", &.{}, null, table[0].id);
|
||||||
|
check("spawning with a device the caller does not hold is refused", spawned == null);
|
||||||
|
check("and no child was left behind by the refusal", process.processes(&process_table) == before);
|
||||||
|
}
|
||||||
|
|
||||||
|
// 5. **Nothing with mappable resources is left lying around.** A device nobody
|
||||||
|
// holds can be claimed by anyone, so the manager takes every seeded device that
|
||||||
|
// carries resources — the HPET above all, which has an MMIO window and an IRQ
|
||||||
|
// and no user-space driver. The one exception is the loader's framebuffer,
|
||||||
|
// which the compositor claims. So from here, a resource-bearing device should
|
||||||
|
// refuse to be taken, and the reason should be that someone already has it.
|
||||||
|
// Settle first, then sweep **once**. The manager is still starting when this
|
||||||
|
// fixture is spawned, so an immediate sweep finds hardware unheld and reports a
|
||||||
|
// hole that closes a millisecond later. Retrying until the sweep comes back
|
||||||
|
// empty is worse than useless: the first pass *takes* the device, so the second
|
||||||
|
// finds it unavailable — because this process now holds it — and concludes all
|
||||||
|
// is well. One sweep, after a wait long enough for the manager to have claimed.
|
||||||
|
time.sleepMillis(1500);
|
||||||
|
var takeable: usize = 0;
|
||||||
|
for (table[0..seen]) |descriptor| {
|
||||||
|
if (descriptor.resource_count == 0) continue;
|
||||||
|
if (descriptor.class == @intFromEnum(device.DeviceClass.display)) continue;
|
||||||
|
device.claim(descriptor.id) catch continue; // refused, as it should be
|
||||||
|
takeable += 1;
|
||||||
|
}
|
||||||
|
check("no resource-bearing device is left for the taking", takeable == 0);
|
||||||
|
|
||||||
|
if (failures == 0) {
|
||||||
|
line("device-authority: VERDICT ok ({d} devices, none of them mine)\n", .{seen});
|
||||||
|
} else {
|
||||||
|
line("device-authority: VERDICT FAILED {d} assertion(s)\n", .{failures});
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn main(startup: process.Init) void {
|
||||||
|
const role = startup.arguments.get(1) orelse return; // bare (ramdisk sweep): stay silent
|
||||||
|
if (std.mem.eql(u8, role, "run")) run();
|
||||||
|
}
|
||||||
@@ -0,0 +1,281 @@
|
|||||||
|
# Bounds that predate the rule (docs/os-development/bounds.md).
|
||||||
|
#
|
||||||
|
# Generated from the tree as it stood when the check landed, so the gate could start
|
||||||
|
# without a 278-site sweep in front of it. Every line is a compile-time ceiling that
|
||||||
|
# has not yet said what it counts, who decides its size, what it protects, what
|
||||||
|
# happens when it is reached, or how anyone finds out.
|
||||||
|
#
|
||||||
|
# **This list may only shrink.** Declaring a bound means deleting its line; the check
|
||||||
|
# fails if a listed bound is now declared, and fails if a listed bound has vanished.
|
||||||
|
# Nothing may be added to it — a new ceiling declares itself or does not land.
|
||||||
|
#
|
||||||
|
# docs/fixed-bounds-audit.md is the analysis of how they got here.
|
||||||
|
boot/efi.zig:buffer
|
||||||
|
boot/efi.zig:info_buffer
|
||||||
|
boot/efi.zig:maximum_bundled
|
||||||
|
boot/efi.zig:maximum_tree_depth
|
||||||
|
library/device/acpi/aml/interpreter.zig:argbuf
|
||||||
|
library/device/acpi/aml/interpreter.zig:args
|
||||||
|
library/device/acpi/aml/interpreter.zig:locals
|
||||||
|
library/device/acpi/aml/interpreter.zig:maximum_segments
|
||||||
|
library/device/acpi/aml/interpreter.zig:notify_queue
|
||||||
|
library/device/acpi/aml/namespace.zig:segment
|
||||||
|
library/device/acpi/aml/parser.zig:maximum_segments
|
||||||
|
library/device/driver/driver.zig:lookup_attempts
|
||||||
|
library/device/model/device-abi.zig:hid
|
||||||
|
library/device/model/device-abi.zig:maximum_device_resources
|
||||||
|
library/device/pci/pci.zig:bar_virtual
|
||||||
|
library/device/pci/pci.zig:bars
|
||||||
|
library/device/registry/device-registry.zig:cols
|
||||||
|
library/device/registry/device-registry.zig:rules
|
||||||
|
library/device/registry/device-registry.zig:rules
|
||||||
|
library/device/registry/device-registry.zig:rules
|
||||||
|
library/device/registry/device-registry.zig:rules
|
||||||
|
library/device/registry/device-registry.zig:rules
|
||||||
|
library/kernel/channel.zig:bind_attempts
|
||||||
|
library/kernel/channel.zig:name_maximum
|
||||||
|
library/kernel/channel.zig:path_maximum
|
||||||
|
library/kernel/file-system.zig:name_buffer
|
||||||
|
library/kernel/file-system.zig:out
|
||||||
|
library/kernel/logging.zig:buffer
|
||||||
|
library/kernel/process.zig:blob
|
||||||
|
library/kernel/process.zig:receive
|
||||||
|
library/kernel/process.zig:table
|
||||||
|
library/kernel/service.zig:subscriber_capacity
|
||||||
|
library/kernel/start.zig:buffer
|
||||||
|
library/kernel/thread.zig:readers
|
||||||
|
library/kernel/thread.zig:writers
|
||||||
|
library/protocol/device-manager/device-manager-protocol.zig:hid
|
||||||
|
library/protocol/display/display-protocol.zig:max_modes
|
||||||
|
library/protocol/envelope/envelope.zig:packet_maximum
|
||||||
|
library/protocol/envelope/envelope.zig:post_maximum
|
||||||
|
library/protocol/power/power-protocol.zig:hid
|
||||||
|
library/protocol/scanout/scanout-protocol.zig:max_modes
|
||||||
|
library/protocol/usb-transfer/usb-transfer-protocol.zig:max_report_data
|
||||||
|
library/protocol/usb-transfer/usb-transfer-protocol.zig:max_reported_endpoints
|
||||||
|
library/protocol/usb-transfer/usb-transfer-protocol.zig:setup
|
||||||
|
system/abi.zig:errno_maximum
|
||||||
|
system/abi.zig:klog_maximum_message
|
||||||
|
system/abi.zig:maximum_process_name
|
||||||
|
system/boot-handoff.zig:kernel_segments
|
||||||
|
system/drivers/pci-bus/pci-bus.zig:line
|
||||||
|
system/drivers/pci-bus/pci-bus.zig:sub_buffer
|
||||||
|
system/drivers/ps2-bus/mouse-packet.zig:bytes
|
||||||
|
system/drivers/ps2-bus/scancode.zig:pressed
|
||||||
|
system/drivers/ps2-bus/scancode.zig:set2_base
|
||||||
|
system/drivers/ps2-bus/scancode.zig:set2_extended
|
||||||
|
system/drivers/usb-hid/hid-report.zig:keys
|
||||||
|
system/drivers/usb-hid/keyboard.zig:receive
|
||||||
|
system/drivers/usb-hid/mouse.zig:receive
|
||||||
|
system/drivers/usb-storage/bulk-only-transport.zig:cdb
|
||||||
|
system/drivers/usb-storage/scsi.zig:op_read_capacity_10
|
||||||
|
system/drivers/usb-storage/usb-storage.zig:capacity_bytes
|
||||||
|
system/drivers/usb-xhci-bus/usb-xhci-bus.zig:hid_buffer
|
||||||
|
system/drivers/usb-xhci-bus/usb-xhci-bus.zig:maker_buffer
|
||||||
|
system/drivers/usb-xhci-bus/usb-xhci-bus.zig:maker_buffer
|
||||||
|
system/drivers/usb-xhci-bus/usb-xhci-bus.zig:prev_connected
|
||||||
|
system/drivers/usb-xhci-bus/usb-xhci-bus.zig:product_buffer
|
||||||
|
system/drivers/usb-xhci-bus/usb-xhci-bus.zig:product_buffer
|
||||||
|
system/drivers/usb-xhci-bus/usb-xhci-library.zig:buffer
|
||||||
|
system/drivers/usb-xhci-bus/usb-xhci-library.zig:bytes
|
||||||
|
system/drivers/usb-xhci-bus/usb-xhci-library.zig:data
|
||||||
|
system/drivers/usb-xhci-bus/usb-xhci-library.zig:descriptor
|
||||||
|
system/drivers/usb-xhci-bus/usb-xhci-library.zig:head
|
||||||
|
system/drivers/usb-xhci-bus/usb-xhci-library.zig:header
|
||||||
|
system/drivers/usb-xhci-bus/usb-xhci-library.zig:max_subscriptions
|
||||||
|
system/drivers/usb-xhci-bus/usb-xhci-library.zig:port_changes
|
||||||
|
system/drivers/usb-xhci-bus/usb-xhci-library.zig:raw
|
||||||
|
system/drivers/usb-xhci-bus/usb-xhci-library.zig:report_queue_capacity
|
||||||
|
system/drivers/usb-xhci-bus/usb-xhci-library.zig:request_set_hub_depth
|
||||||
|
system/drivers/virtio-gpu/virtio-gpu-protocol.zig:edid
|
||||||
|
system/drivers/virtio-gpu/virtio-gpu-protocol.zig:max_scanouts
|
||||||
|
system/drivers/virtio-gpu/virtio-gpu.zig:descriptors
|
||||||
|
system/drivers/virtio-gpu/virtio-gpu.zig:max_height
|
||||||
|
system/drivers/virtio-gpu/virtio-gpu.zig:max_width
|
||||||
|
system/initial-ramdisk.zig:buffer
|
||||||
|
system/initial-ramdisk.zig:buffer
|
||||||
|
system/initial-ramdisk.zig:maximum_name
|
||||||
|
system/kernel/acpi.zig:APIC
|
||||||
|
system/kernel/acpi.zig:BERT
|
||||||
|
system/kernel/acpi.zig:CPEP
|
||||||
|
system/kernel/acpi.zig:DMAR
|
||||||
|
system/kernel/acpi.zig:DSDT
|
||||||
|
system/kernel/acpi.zig:ECDT
|
||||||
|
system/kernel/acpi.zig:EINJ
|
||||||
|
system/kernel/acpi.zig:ERST
|
||||||
|
system/kernel/acpi.zig:FACP
|
||||||
|
system/kernel/acpi.zig:FACS
|
||||||
|
system/kernel/acpi.zig:HEST
|
||||||
|
system/kernel/acpi.zig:HPET
|
||||||
|
system/kernel/acpi.zig:IVRS
|
||||||
|
system/kernel/acpi.zig:MCFG
|
||||||
|
system/kernel/acpi.zig:MPST
|
||||||
|
system/kernel/acpi.zig:MSCT
|
||||||
|
system/kernel/acpi.zig:PMTT
|
||||||
|
system/kernel/acpi.zig:PSDT
|
||||||
|
system/kernel/acpi.zig:RASF
|
||||||
|
system/kernel/acpi.zig:RSDT
|
||||||
|
system/kernel/acpi.zig:SBST
|
||||||
|
system/kernel/acpi.zig:SLIT
|
||||||
|
system/kernel/acpi.zig:SPCR
|
||||||
|
system/kernel/acpi.zig:SRAT
|
||||||
|
system/kernel/acpi.zig:SSDT
|
||||||
|
system/kernel/acpi.zig:XSDT
|
||||||
|
system/kernel/acpi.zig:aml_block_len
|
||||||
|
system/kernel/acpi.zig:aml_block_physical
|
||||||
|
system/kernel/acpi.zig:holes
|
||||||
|
system/kernel/acpi.zig:maximum_rmrr
|
||||||
|
system/kernel/acpi.zig:nb
|
||||||
|
system/kernel/acpi.zig:nb
|
||||||
|
system/kernel/acpi.zig:nb
|
||||||
|
system/kernel/acpi.zig:oem_id
|
||||||
|
system/kernel/acpi.zig:oem_id
|
||||||
|
system/kernel/acpi.zig:oem_id
|
||||||
|
system/kernel/acpi.zig:oem_table_id
|
||||||
|
system/kernel/acpi.zig:overrides
|
||||||
|
system/kernel/acpi.zig:rmrr_limit_offset
|
||||||
|
system/kernel/acpi.zig:signature
|
||||||
|
system/kernel/acpi.zig:signature
|
||||||
|
system/kernel/acpi.zig:signature
|
||||||
|
system/kernel/architecture/x86_64/cpu.zig:irq_vector_count
|
||||||
|
system/kernel/architecture/x86_64/idt.zig:gate_count
|
||||||
|
system/kernel/architecture/x86_64/ioapic.zig:overrides
|
||||||
|
system/kernel/architecture/x86_64/iommu-amd.zig:buffer
|
||||||
|
system/kernel/architecture/x86_64/iommu-amd.zig:buffer
|
||||||
|
system/kernel/architecture/x86_64/iommu-intel.zig:buffer
|
||||||
|
system/kernel/architecture/x86_64/iommu-intel.zig:buffer
|
||||||
|
system/kernel/architecture/x86_64/iommu-intel.zig:context_table
|
||||||
|
system/kernel/device-model.zig:buffer
|
||||||
|
system/kernel/device-model.zig:cbuf
|
||||||
|
system/kernel/device-model.zig:hid_buffer
|
||||||
|
system/kernel/device-model.zig:maximum_resources
|
||||||
|
system/kernel/device-model.zig:name_buffer
|
||||||
|
system/kernel/device-model.zig:rbuf
|
||||||
|
system/kernel/heap.zig:heap_maximum
|
||||||
|
system/kernel/ipc-synchronous.zig:MESSAGE_MAXIMUM
|
||||||
|
system/kernel/ipc-synchronous.zig:POST_MAXIMUM
|
||||||
|
system/kernel/ipc-synchronous.zig:notify_buffer
|
||||||
|
system/kernel/ipc-synchronous.zig:post_capacity
|
||||||
|
system/kernel/irq.zig:maximum_gsi
|
||||||
|
system/kernel/irq.zig:msi_bound
|
||||||
|
system/kernel/irq.zig:msi_owner
|
||||||
|
system/kernel/irq.zig:vector_gsi
|
||||||
|
system/kernel/kernel.zig:buffer
|
||||||
|
system/kernel/kernel.zig:buffer
|
||||||
|
system/kernel/kernel.zig:buffer
|
||||||
|
system/kernel/kernel.zig:isos
|
||||||
|
system/kernel/kernel.zig:maximum_wake_attempts
|
||||||
|
system/kernel/log-ring.zig:message
|
||||||
|
system/kernel/log-ring.zig:out
|
||||||
|
system/kernel/log.zig:buffer
|
||||||
|
system/kernel/log.zig:maximum_sinks
|
||||||
|
system/kernel/log.zig:message
|
||||||
|
system/kernel/log.zig:ring_capacity
|
||||||
|
system/kernel/process.zig:chunk
|
||||||
|
system/kernel/process.zig:chunk
|
||||||
|
system/kernel/process.zig:chunk
|
||||||
|
system/kernel/process.zig:chunk
|
||||||
|
system/kernel/process.zig:exit_record_capacity
|
||||||
|
system/kernel/process.zig:exit_subscriber_capacity
|
||||||
|
system/kernel/process.zig:maximum_argument_bytes
|
||||||
|
system/kernel/process.zig:maximum_arguments
|
||||||
|
system/kernel/process.zig:maximum_dma_regions
|
||||||
|
system/kernel/process.zig:maximum_mmap_pages
|
||||||
|
system/kernel/process.zig:maximum_mount_prefix
|
||||||
|
system/kernel/process.zig:maximum_mount_rewrite
|
||||||
|
system/kernel/process.zig:maximum_pages
|
||||||
|
system/kernel/process.zig:maximum_resolve_path
|
||||||
|
system/kernel/process.zig:maximum_segments
|
||||||
|
system/kernel/process.zig:maximum_shared_memory_pages
|
||||||
|
system/kernel/process.zig:name_buffer
|
||||||
|
system/kernel/process.zig:timer_capacity
|
||||||
|
system/kernel/process.zig:word_bytes
|
||||||
|
system/kernel/process.zig:write_buffer
|
||||||
|
system/kernel/scheduler.zig:ipc_maximum_handles
|
||||||
|
system/kernel/scheduler.zig:maximum_space_mappings
|
||||||
|
system/kernel/vfs.zig:maximum_directories
|
||||||
|
system/kernel/vfs.zig:maximum_mounts
|
||||||
|
system/kernel/vfs.zig:maximum_prefix
|
||||||
|
system/kernel/vfs.zig:maximum_rewrite
|
||||||
|
system/services/acpi/acpi.zig:blocks
|
||||||
|
system/services/acpi/acpi.zig:buffer
|
||||||
|
system/services/acpi/acpi.zig:hid
|
||||||
|
system/services/acpi/acpi.zig:mmio_scratch
|
||||||
|
system/services/acpi/acpi.zig:name
|
||||||
|
system/services/acpi/acpi.zig:registered
|
||||||
|
system/services/device-manager/device-manager.zig:arguments
|
||||||
|
system/services/device-manager/device-manager.zig:id_text
|
||||||
|
system/services/device-manager/device-manager.zig:maximum_children
|
||||||
|
system/services/device-manager/device-manager.zig:maximum_drivers
|
||||||
|
system/services/device-manager/device-manager.zig:name_buffer
|
||||||
|
system/services/device-manager/device-manager.zig:registry_rules
|
||||||
|
system/services/device-manager/device-manager.zig:registry_source
|
||||||
|
system/services/display/backend.zig:device_table
|
||||||
|
system/services/display/backend.zig:line
|
||||||
|
system/services/display/compositor.zig:capacity
|
||||||
|
system/services/display/compositor.zig:maximum_columns
|
||||||
|
system/services/display/compositor.zig:maximum_rects
|
||||||
|
system/services/display/compositor.zig:maximum_rows
|
||||||
|
system/services/display/display.zig:line
|
||||||
|
system/services/display/display.zig:line
|
||||||
|
system/services/display/display.zig:line
|
||||||
|
system/services/display/display.zig:list
|
||||||
|
system/services/display/display.zig:maximum_layers
|
||||||
|
system/services/display/display.zig:mode_list
|
||||||
|
system/services/fat/engine.zig:buf
|
||||||
|
system/services/fat/engine.zig:buf
|
||||||
|
system/services/fat/engine.zig:buf
|
||||||
|
system/services/fat/engine.zig:buffer
|
||||||
|
system/services/fat/engine.zig:cached_back
|
||||||
|
system/services/fat/engine.zig:chunk
|
||||||
|
system/services/fat/engine.zig:chunk
|
||||||
|
system/services/fat/engine.zig:chunk
|
||||||
|
system/services/fat/engine.zig:device_back
|
||||||
|
system/services/fat/engine.zig:display
|
||||||
|
system/services/fat/engine.zig:long_name
|
||||||
|
system/services/fat/engine.zig:long_name
|
||||||
|
system/services/fat/engine.zig:long_name
|
||||||
|
system/services/fat/engine.zig:max_transfer_sectors
|
||||||
|
system/services/fat/engine.zig:pair
|
||||||
|
system/services/fat/engine.zig:pair
|
||||||
|
system/services/fat/engine.zig:payload
|
||||||
|
system/services/fat/engine.zig:payload
|
||||||
|
system/services/fat/engine.zig:payload
|
||||||
|
system/services/fat/engine.zig:readback
|
||||||
|
system/services/fat/engine.zig:run
|
||||||
|
system/services/fat/engine.zig:run
|
||||||
|
system/services/fat/engine.zig:short
|
||||||
|
system/services/fat/engine.zig:short
|
||||||
|
system/services/fat/engine.zig:short
|
||||||
|
system/services/fat/engine.zig:stem
|
||||||
|
system/services/fat/engine.zig:tail_buffer
|
||||||
|
system/services/fat/engine.zig:units
|
||||||
|
system/services/fat/engine.zig:value
|
||||||
|
system/services/fat/engine.zig:value
|
||||||
|
system/services/fat/engine.zig:window
|
||||||
|
system/services/fat/on-disk.zig:filesystem_type
|
||||||
|
system/services/fat/on-disk.zig:filesystem_type
|
||||||
|
system/services/fat/on-disk.zig:jump
|
||||||
|
system/services/fat/on-disk.zig:name
|
||||||
|
system/services/fat/on-disk.zig:name1
|
||||||
|
system/services/fat/on-disk.zig:name2
|
||||||
|
system/services/fat/on-disk.zig:name3
|
||||||
|
system/services/fat/on-disk.zig:oem_name
|
||||||
|
system/services/fat/on-disk.zig:volume_label
|
||||||
|
system/services/fat/on-disk.zig:volume_label
|
||||||
|
system/services/init/init.zig:binary
|
||||||
|
system/services/init/init.zig:init_csv
|
||||||
|
system/services/init/init.zig:max_service_args
|
||||||
|
system/services/init/init.zig:max_services
|
||||||
|
system/services/init/init.zig:maximum_bindings
|
||||||
|
system/services/init/init.zig:maximum_grants
|
||||||
|
system/services/init/init.zig:maximum_name
|
||||||
|
system/services/init/init.zig:maximum_restarts
|
||||||
|
system/services/init/init.zig:process_table
|
||||||
|
system/services/init/init.zig:protocol_csv
|
||||||
|
system/services/logger/logger.zig:chunk
|
||||||
|
system/services/logger/logger.zig:gap_line
|
||||||
|
system/services/logger/logger.zig:line
|
||||||
|
system/services/logger/logger.zig:line
|
||||||
|
system/services/logger/logger.zig:maximum_files
|
||||||
|
system/services/logger/logger.zig:stamp
|
||||||
@@ -0,0 +1,189 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Every compile-time ceiling states what it is doing there.
|
||||||
|
|
||||||
|
A *bound* is a number chosen at compile time that decides how much of something the
|
||||||
|
code can hold: `const maximum_devices = 64`, `var below: [64]Range`, `var blob: [512]u8`.
|
||||||
|
Different units, one shape, and one recurring way of going wrong — see
|
||||||
|
docs/fixed-bounds-audit.md, where 235 of them turned up, 139 on quantities the machine
|
||||||
|
or a file decides rather than us, and 171 silent when reached.
|
||||||
|
|
||||||
|
This is the gate that keeps new ones from joining them. It does not resize anything and
|
||||||
|
it makes no judgement about whether a bound should exist; it only refuses one that will
|
||||||
|
not say what it is for. The declaration is a doc comment immediately above:
|
||||||
|
|
||||||
|
/// bound: logical CPUs the kernel tracks
|
||||||
|
/// decided-by: hardware
|
||||||
|
/// protects: the per-CPU bookkeeping arrays, which are sized at compile time
|
||||||
|
/// at-limit: degrade - surplus cores are left parked, never brought online
|
||||||
|
/// observed-by: platform.cpusDropped() -> the WARNING at kernel.zig:281
|
||||||
|
pub const maximum_cpus = 128;
|
||||||
|
|
||||||
|
`decided-by` is the field the audit turned on: `hardware` and `external` mean the
|
||||||
|
quantity is not ours to choose, and a fixed bound on one of those is a defect rather
|
||||||
|
than a tunable. `at-limit`'s vocabulary is closed on purpose — there is no `silent`,
|
||||||
|
no `drop`, and nothing meaning *allow*, so the behaviours that caused the damage cannot
|
||||||
|
be written down. `truncate` is legal only with a marker the reader can see.
|
||||||
|
|
||||||
|
An array length that names a declared bound (`[maximum_devices]Descriptor`) is not
|
||||||
|
itself a bound: the number lives at the declaration, and that is where it is declared.
|
||||||
|
Only literal lengths are flagged, which pushes ceilings toward having names.
|
||||||
|
|
||||||
|
The ~235 that already exist are listed in tools/bounds-allowlist.txt so this can land
|
||||||
|
without a tree-wide sweep in front of it. That list may only shrink: declaring a bound
|
||||||
|
means deleting its line, and a stale line is an error too.
|
||||||
|
|
||||||
|
Usage: check-bounds.py [--list] [repo-root]
|
||||||
|
--list print every undeclared bound found, for regenerating the allowlist
|
||||||
|
"""
|
||||||
|
|
||||||
|
import re
|
||||||
|
import sys
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
# Directories worth gating. `test/` is excluded: a fixture's `[100]MemoryRegion` is a
|
||||||
|
# test input, not a ceiling the system runs into.
|
||||||
|
ROOTS = ("system", "library", "boot")
|
||||||
|
SKIP_PARTS = {".zig-cache", "zig-out", ".git", ".claude", "vendor", "generated"}
|
||||||
|
SKIP_FILES = {"tests.zig"}
|
||||||
|
|
||||||
|
FIELDS = ("bound", "decided-by", "protects", "at-limit", "observed-by")
|
||||||
|
DECIDED_BY = {"hardware", "external", "ours"}
|
||||||
|
AT_LIMIT = {"refuse", "degrade", "truncate", "grow"}
|
||||||
|
|
||||||
|
# A `const` whose name reads like a ceiling and whose value is an integer literal.
|
||||||
|
NAMED = re.compile(
|
||||||
|
r"^\s*(?:pub\s+)?const\s+([A-Za-z_]\w*)\s*(?::\s*[\w.\[\]]+\s*)?=\s*"
|
||||||
|
r"(\d[\d_]*|0x[0-9a-fA-F_]+)\s*(?:\*\s*\d[\d_]*\s*)*;"
|
||||||
|
)
|
||||||
|
NAME_IS_BOUND = re.compile(r"(^|_)(maximum|max|limit|capacity|depth|attempts|count)($|_)", re.I)
|
||||||
|
|
||||||
|
# A declaration or struct field whose type carries a *literal* array length.
|
||||||
|
ARRAY_DECL = re.compile(r"^\s*(?:pub\s+)?(?:const|var)\s+([A-Za-z_]\w*)\s*:[^=]*?\[\s*(\d[\d_]*)\s*\]")
|
||||||
|
ARRAY_FIELD = re.compile(r"^\s*([A-Za-z_]\w*)\s*:\s*\[\s*(\d[\d_]*)\s*\]")
|
||||||
|
|
||||||
|
# Struct padding and reserved fields are shapes, not ceilings — nothing is ever "held"
|
||||||
|
# in them. Everything else with a literal length is a candidate, *including* the tidy
|
||||||
|
# powers of two: `[512]u8` and `[64]Range` were the two worst findings in the audit, and
|
||||||
|
# any size-based exemption would have skipped exactly them. A length that is genuinely a
|
||||||
|
# fact rather than a ceiling says so in its declaration ("decided-by: hardware, the PCI
|
||||||
|
# spec gives a function 6 BARs") — that is what the declaration is for.
|
||||||
|
NOT_A_BOUND_NAME = re.compile(r"^_*(padding|pad|reserved|unused|spare)\d*$", re.I)
|
||||||
|
|
||||||
|
|
||||||
|
def sources(root: Path):
|
||||||
|
for top in ROOTS:
|
||||||
|
base = root / top
|
||||||
|
if not base.is_dir():
|
||||||
|
continue
|
||||||
|
for path in sorted(base.rglob("*.zig")):
|
||||||
|
if SKIP_PARTS & set(path.parts) or path.name in SKIP_FILES:
|
||||||
|
continue
|
||||||
|
yield path
|
||||||
|
|
||||||
|
|
||||||
|
def declaration_above(lines, index):
|
||||||
|
"""The `/// key: value` block immediately above line `index`, as a dict."""
|
||||||
|
fields = {}
|
||||||
|
i = index - 1
|
||||||
|
while i >= 0:
|
||||||
|
stripped = lines[i].strip()
|
||||||
|
if not stripped.startswith("///"):
|
||||||
|
break
|
||||||
|
body = stripped[3:].strip()
|
||||||
|
match = re.match(r"([a-z-]+):\s*(.+)", body)
|
||||||
|
if match:
|
||||||
|
fields[match.group(1)] = match.group(2).strip()
|
||||||
|
i -= 1
|
||||||
|
return fields
|
||||||
|
|
||||||
|
|
||||||
|
def problems_with(fields):
|
||||||
|
"""Why a declaration is not acceptable, or an empty list."""
|
||||||
|
missing = [f for f in FIELDS if f not in fields or not fields[f]]
|
||||||
|
if missing:
|
||||||
|
return ["missing " + ", ".join(missing)]
|
||||||
|
out = []
|
||||||
|
if fields["decided-by"] not in DECIDED_BY:
|
||||||
|
out.append(f"decided-by must be one of {sorted(DECIDED_BY)}, not {fields['decided-by']!r}")
|
||||||
|
verb = fields["at-limit"].split()[0].strip("-:,").lower()
|
||||||
|
if verb not in AT_LIMIT:
|
||||||
|
out.append(
|
||||||
|
f"at-limit must start with one of {sorted(AT_LIMIT)}, not {verb!r}. "
|
||||||
|
"There is deliberately no way to say 'silent', 'drop', or anything meaning 'allow'"
|
||||||
|
)
|
||||||
|
if verb == "truncate" and len(fields["at-limit"].split()) < 3:
|
||||||
|
out.append("at-limit: truncate must say how a reader can TELL it happened")
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def find(root: Path):
|
||||||
|
"""Every bound-shaped declaration: (relative path, name, value, line, fields)."""
|
||||||
|
for path in sources(root):
|
||||||
|
rel = path.relative_to(root).as_posix()
|
||||||
|
lines = path.read_text(encoding="utf-8", errors="replace").split("\n")
|
||||||
|
for n, line in enumerate(lines):
|
||||||
|
if line.lstrip().startswith("//"):
|
||||||
|
continue
|
||||||
|
name = value = None
|
||||||
|
m = NAMED.match(line)
|
||||||
|
if m and NAME_IS_BOUND.search(m.group(1)):
|
||||||
|
name, value = m.group(1), m.group(2)
|
||||||
|
else:
|
||||||
|
m = ARRAY_DECL.match(line) or ARRAY_FIELD.match(line)
|
||||||
|
if m and not NOT_A_BOUND_NAME.match(m.group(1)):
|
||||||
|
name, value = m.group(1), m.group(2)
|
||||||
|
if name:
|
||||||
|
yield rel, name, value, n + 1, declaration_above(lines, n)
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
argv = [a for a in sys.argv[1:] if not a.startswith("--")]
|
||||||
|
listing = "--list" in sys.argv
|
||||||
|
root = Path(argv[0]) if argv else Path(__file__).resolve().parent.parent
|
||||||
|
|
||||||
|
allow_path = root / "tools" / "bounds-allowlist.txt"
|
||||||
|
allowed = set()
|
||||||
|
if allow_path.exists():
|
||||||
|
for raw in allow_path.read_text().split("\n"):
|
||||||
|
entry = raw.split("#", 1)[0].strip()
|
||||||
|
if entry:
|
||||||
|
allowed.add(entry)
|
||||||
|
|
||||||
|
undeclared, bad, seen = [], [], set()
|
||||||
|
for rel, name, value, line, fields in find(root):
|
||||||
|
key = f"{rel}:{name}"
|
||||||
|
seen.add(key)
|
||||||
|
if not fields:
|
||||||
|
(undeclared if key not in allowed else []).append((key, value, line))
|
||||||
|
continue
|
||||||
|
for why in problems_with(fields):
|
||||||
|
bad.append((key, line, why))
|
||||||
|
if key in allowed and not problems_with(fields):
|
||||||
|
bad.append((key, line, "now declared — delete its line from tools/bounds-allowlist.txt"))
|
||||||
|
|
||||||
|
if listing:
|
||||||
|
for key, value, line in sorted(undeclared):
|
||||||
|
print(f"{key} # = {value}, line {line}")
|
||||||
|
for key in sorted(allowed - seen):
|
||||||
|
print(f"# STALE: {key}")
|
||||||
|
return 0
|
||||||
|
|
||||||
|
stale = sorted(allowed - seen)
|
||||||
|
if not undeclared and not bad and not stale:
|
||||||
|
return 0
|
||||||
|
|
||||||
|
print("bounds check failed\n", file=sys.stderr)
|
||||||
|
for key, value, line in sorted(undeclared):
|
||||||
|
print(f" {key} (= {value}, line {line})", file=sys.stderr)
|
||||||
|
print(" no declaration. A ceiling states what it counts, who decides its", file=sys.stderr)
|
||||||
|
print(" size, what it protects, what happens at the limit, and how you", file=sys.stderr)
|
||||||
|
print(" find out. See docs/os-development/bounds.md.", file=sys.stderr)
|
||||||
|
for key, line, why in sorted(bad):
|
||||||
|
print(f" {key} (line {line}): {why}", file=sys.stderr)
|
||||||
|
for key in stale:
|
||||||
|
print(f" {key}: allowlisted but no longer found — delete its line", file=sys.stderr)
|
||||||
|
return 1
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
sys.exit(main())
|
||||||
Reference in New Issue
Block a user