Compare commits
49
Commits
a85f29edc9
...
3c9f454398
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
3c9f454398 | ||
|
|
8710944a92 | ||
|
|
1d7850239d | ||
|
|
d603d40b5c | ||
|
|
77fe4d220e | ||
|
|
0d5a7394ef | ||
|
|
5930c9653c | ||
|
|
981ff7cd25 | ||
|
|
5cca580066 | ||
|
|
35f43057f4 | ||
|
|
72807c20e4 | ||
|
|
0eb2420690 | ||
|
|
df9c1ed827 | ||
|
|
ba195fa0a2 | ||
|
|
1a1d92cba9 | ||
|
|
4ca57fc37e | ||
|
|
ca1126537d | ||
|
|
337b981c07 | ||
|
|
b7d97ebb5d | ||
|
|
6b3a381626 | ||
|
|
7d8aa51234 | ||
|
|
3ae541214f | ||
|
|
2ebfccc8ed | ||
|
|
f23f073624 | ||
|
|
637bf2e0b1 | ||
|
|
cb8d1e2e51 | ||
|
|
8b628a4a7d | ||
|
|
a5840789dd | ||
|
|
f7151ed577 | ||
|
|
5492607a61 | ||
|
|
81dd1e9318 | ||
|
|
a2d6056772 | ||
|
|
33376f24ab | ||
|
|
3111c7c5e6 | ||
|
|
547d0ec46b | ||
|
|
5ad42ceac3 | ||
|
|
8e388e048a | ||
|
|
ef793d1320 | ||
|
|
139c4f624f | ||
|
|
3b24b541b0 | ||
|
|
5d55217212 | ||
|
|
729b40ece7 | ||
|
|
6328823ef1 | ||
|
|
69fbef40c0 | ||
|
|
f4eb88e7d2 | ||
|
|
568823a4fb | ||
|
|
4398eb7cc4 | ||
|
|
a86559648e | ||
|
|
7db5fba884 |
@@ -342,6 +342,7 @@ pub fn build(b: *std.Build) void {
|
||||
"protocol-registry-test", // drives the registrar: ungranted bind, collision, restart
|
||||
"protocol-denied-test", // restriction stage one: an ungranted open answers as absence
|
||||
"protocol-conformance-test", // the reserved verbs, asked of every provider the boot bound
|
||||
"device-authority-test", // the attacker: a process handed no device, asserting what it cannot do
|
||||
}) |fixture| {
|
||||
const package = b.lazyDependency(fixture, .{}) orelse
|
||||
@panic("a test fixture package is missing under test/system/services");
|
||||
@@ -397,6 +398,19 @@ pub fn build(b: *std.Build) void {
|
||||
// here (compiled for the host rather than inheriting a freestanding target) —
|
||||
// which also compile-checks that the three-way split stays self-consistent.
|
||||
const test_step = b.step("test", "Run tests");
|
||||
|
||||
// Every compile-time ceiling states what it counts, who decides its size, what it
|
||||
// protects, what happens when it is reached, and how anyone finds out
|
||||
// (docs/os-development/bounds.md). The ~273 that predate the rule are listed in
|
||||
// tools/bounds-allowlist.txt so this could land without a tree-wide sweep first;
|
||||
// that list may only shrink. Part of `test` rather than the default build: it reads
|
||||
// the whole tree, and a red bounds check should not stop you booting a kernel.
|
||||
const bounds_check = b.addSystemCommand(&.{ "python3", "tools/check-bounds.py" });
|
||||
bounds_check.setCwd(b.path("."));
|
||||
const bounds_step = b.step("bounds", "Check that every compile-time ceiling declares itself");
|
||||
bounds_step.dependOn(&bounds_check.step);
|
||||
test_step.dependOn(&bounds_check.step);
|
||||
|
||||
for ([_][]const u8{
|
||||
"system/boot-handoff.zig",
|
||||
"system/abi.zig",
|
||||
|
||||
@@ -79,6 +79,7 @@
|
||||
.@"protocol-registry-test" = .{ .path = "test/system/services/protocol-registry-test", .lazy = true },
|
||||
.@"protocol-denied-test" = .{ .path = "test/system/services/protocol-denied-test", .lazy = true },
|
||||
.@"protocol-conformance-test" = .{ .path = "test/system/services/protocol-conformance-test", .lazy = true },
|
||||
.@"device-authority-test" = .{ .path = "test/system/services/device-authority-test", .lazy = true },
|
||||
// See `zig fetch --save <url>` for a command-line interface for adding dependencies.
|
||||
//.example = .{
|
||||
// // When updating this field to a new URL, be sure to delete the corresponding
|
||||
|
||||
@@ -0,0 +1,632 @@
|
||||
# The bounds track: removing the numbers we invented
|
||||
|
||||
*Plan, 2026-08-08. Follows [fixed-bounds-audit.md](fixed-bounds-audit.md) (235 ceilings,
|
||||
139 on quantities we do not choose) and the AMD Ryzen that found the first one.*
|
||||
|
||||
---
|
||||
|
||||
## Live state — the unattended run
|
||||
|
||||
*This table is the progress view. It is updated at the end of every step, before the
|
||||
next one starts.*
|
||||
|
||||
| Step | What | State |
|
||||
|---|---|---|
|
||||
| L1 | Reclamation: a dead task's registrations die with its claims | **stopped — the step was wrong; see open question 4** |
|
||||
| L2 | Bounds build check + allowlist; declare what we have already touched | **done** — `zig build bounds`, 273 allowlisted, 5 declared |
|
||||
| L3 | xHCI: slot count from `HCSPARAMS1.MaxSlots`, not 8 | **done** — QEMU reports 64; the driver tracked 8 |
|
||||
| L4 | USB: configuration descriptor sized by `wTotalLength`, not 512 | **done** — QEMU tops out at 211 bytes, so the case catches the class, not the original trigger |
|
||||
| L5 | USB: interfaces from the descriptor, and the misattributed-endpoint bug | **done** — fix is by construction; no direct test, see open question 5 |
|
||||
| L6 | xHCI: a failed `allocateDevice` stops leaking an enabled slot | **done** — path forced and verified; no regression test, see open question 5 |
|
||||
|
||||
**Run 1 complete.** L1 stopped (the step was wrong), L2–L6 landed. Suite 115 → 116.
|
||||
Allowlist 278 → 269. Two steps ship without a permanent regression test, both because
|
||||
QEMU's USB devices are too small to reach the paths — see open question 5, which is the
|
||||
audit's own lesson recurring: the test rig is smaller than a real machine.
|
||||
|
||||
---
|
||||
|
||||
## Run 2 — device authority: delete the invented ceilings
|
||||
|
||||
*Design: [device-authority.md](os-development/device-authority.md), which is the **how**
|
||||
for the delegation step [device-manager.md](device-driver-development/device-manager.md)
|
||||
already settled. Read both before starting; the second is authoritative where they
|
||||
differ.*
|
||||
|
||||
The goal, in the project owner's words: **remove the maximum values we set arbitrarily,
|
||||
move the responsibility to the device manager, and keep in the kernel only the parts
|
||||
that cannot safely run in user space.**
|
||||
|
||||
| Step | What | State |
|
||||
|---|---|---|
|
||||
| D0 | The grant rides `system_spawn` — atomic, so no driver need change to receive one | **done** |
|
||||
| D1 | `device_transfer(device_id, task_id)` — the holder gives a device away | **done** — syscall 54; a move, not a copy |
|
||||
| D2 | Adversarial case: a process handed nothing is refused | **done** — `device-authority-test`; the claim half joins it at D6 |
|
||||
| D3 | The manager claims the seeded devices before any driver is spawned | **merged into D4** |
|
||||
| D4 | `usb-xhci-bus` receives its controller | **done** — caught an IOMMU regression I introduced |
|
||||
| D5 | `pci-bus`, `virtio-gpu`, `ps2-bus`, discovery receive theirs | **done** — every driver now receives its hardware |
|
||||
| D6 | `device_claim` refuses a device the caller was not handed | **blocked on question 10** |
|
||||
| D7 | Zero-resource devices stop being kernel objects | **blocked on question 8** |
|
||||
| D8 | **`maximum_children_per_parent` deleted** | **done** |
|
||||
| D9 | Dynamic table; **`maximum_devices` deleted**; per-registrar allowance | **done** |
|
||||
| D10 | Every driver hellos, on its own merits | not started — optional, independent |
|
||||
|
||||
*Executed in the order D1, D2, D4, D5(part), D9, D0, D5(rest), D8 — the numbering is
|
||||
the original plan's, not the sequence. D0 was added mid-run, D3 merged into D4, and D8
|
||||
turned out to have been unblocked since D9.*
|
||||
|
||||
**Run 2 stops, blocked on one question.** Landed: D0, D1, D2, D4, D8, D9, and D5 for
|
||||
three of five claimants (`usb-xhci-bus`, `pci-bus`, `virtio-gpu`). Suite 118/118.
|
||||
|
||||
**Both invented ceilings are gone.** `maximum_devices` and
|
||||
`maximum_children_per_parent` no longer exist, and delegation is atomic with the spawn.
|
||||
|
||||
What remains is not a number: **D6 closes the claiming hole** and is blocked on
|
||||
question 9, and D7 moves the inventory and is blocked on question 8.
|
||||
|
||||
Ordering is load-bearing. D1–D2 built and proved the mechanism with nothing depending on
|
||||
it. D4–D5 move each claimant across one at a time, so the suite stays green throughout
|
||||
and a regression names the driver that caused it. D6 is the flag day. (The original
|
||||
"D7 must precede D9" no longer holds: the per-holder quota bounds zero-resource children
|
||||
as well as anything else, so D9 landed without it.)
|
||||
|
||||
**D3 merged into D4** (found while implementing, recorded rather than worked around).
|
||||
The two cannot be separated: the moment the manager claims a device, any driver still
|
||||
calling `device_claim` on it is refused `AlreadyClaimed`, so D3 on its own turns the
|
||||
suite red — and D3 applied to *nothing* changes no behaviour and cannot be tested.
|
||||
They land together, with the manager claiming only for drivers in an explicit
|
||||
**delegated set** so every unconverted driver keeps claiming exactly as before.
|
||||
`usb-xhci-bus` is the first member, as it was the first driver to conform to `hello`
|
||||
(device-manager.md, M18.1). D5 moves the rest in one at a time; the set and the
|
||||
`device_claim` path both disappear at D6.
|
||||
|
||||
### Observations from the run
|
||||
|
||||
- **D4 introduced an IOMMU regression, caught by converting one driver at a time.**
|
||||
`confineDevice` runs inside `systemDeviceClaim`, so a device arriving by *transfer*
|
||||
was never confined for its new owner: the driver's DMA rings went unbound (three
|
||||
IOMMU+USB cases failed), and two worse consequences were latent — a manager death
|
||||
would have torn down a domain a live driver was using, and a driver death would have
|
||||
leaked one. `iommu.reassign` moves the confinement with the device, keeping the
|
||||
domain and attachment intact so it never translates through nothing. Converting all
|
||||
five drivers at once would have produced the same three failures with five suspects.
|
||||
- **`usb-hub` failed once in a full run, then passed six times** (four isolated, two
|
||||
full). Suspected instance of the known intermittent AP ring-3 fault rather than
|
||||
anything in D4 — recorded rather than dismissed, because D4 moved the `hello` earlier
|
||||
and so did shift boot timing. Watch it across the remaining steps.
|
||||
|
||||
---
|
||||
|
||||
## Run 3 — close the claiming hole
|
||||
|
||||
*Every driver now receives its hardware. What remains is that `device_claim` still
|
||||
works for anyone, so a device nobody holds can still be taken. Six steps, no open
|
||||
questions.*
|
||||
|
||||
### The rule this run implements
|
||||
|
||||
The kernel records, per device, **who gave it**. From that one field everything follows:
|
||||
|
||||
- `device_transfer` and the spawn-fused grant set the giver.
|
||||
- **`device_claim` refuses a device that has a giver.** A device that was given to
|
||||
someone is delegated hardware and must be handed on, never taken.
|
||||
- **On death a device reverts to its giver**, if that task is still alive; otherwise its
|
||||
claim clears. A grant is a loan, not a gift.
|
||||
|
||||
Two things fall out rather than being special-cased:
|
||||
|
||||
- **The framebuffer is untouched.** Nobody delegates it, so it has no giver, so the
|
||||
display service can still claim it exactly as today. No exemption in the kernel, no
|
||||
mention of display anywhere in the rule.
|
||||
- **Restart needs no race.** A dying driver's device returns to the manager, which
|
||||
re-delegates it on respawn. Previously the kernel released it to nobody and the
|
||||
manager re-claimed first-come, so every restart reopened the hole.
|
||||
|
||||
**Found while implementing: E2 is the step that closes the hole, not E3.** A delegated
|
||||
device is *held*, so an attempt to take it is refused as `AlreadyClaimed` long before
|
||||
the giver is consulted — and once a borrower's death returns the device to its lender
|
||||
(or clears both when the lender is gone), there is no state where a device is unheld and
|
||||
still on loan. E3's check is therefore unreachable today. It stays as one comparison
|
||||
that fails closed, guarding any future path that frees a device without clearing its
|
||||
giver, and its comment says so rather than implying a protection it is not providing.
|
||||
|
||||
| Step | What |
|
||||
|---|---|
|
||||
| E1 | Record a giver per device; `device_transfer` and the spawn grant set it — **done** |
|
||||
| E2 | On task death a device reverts to its giver if alive, else its claim clears — **done** |
|
||||
| E3 | `device_claim` refuses a device that has a giver — **done**, but unreachable: E2 already closed the window |
|
||||
| E4 | The manager claims every resource-bearing device at boot, so nothing is left takeable — **done** (boot snapshot only; see below) |
|
||||
| E5 | The attacker fixture gains the claim half it has been waiting for since D2 — **done with E4** |
|
||||
| E6 | Delete the delegated-set scaffolding — every driver is delegated now — **done** |
|
||||
|
||||
Ordering: E1 alone changes no behaviour. E2 must precede E3, or a restart cannot
|
||||
re-acquire. E4 must precede E5, or the attacker will find takeable devices and the
|
||||
assertion will be wrong about why. E6 is cleanup.
|
||||
|
||||
**Run 3 complete.** Suite 118/118. Every driver receives its hardware; a grant is a
|
||||
loan that returns to its lender when the borrower dies; nothing firmware-discovered is
|
||||
left unheld except the framebuffer, which the compositor owns.
|
||||
|
||||
### E4's scope, stated
|
||||
|
||||
It covers the **boot snapshot**. A device *reported* later and matched to no driver
|
||||
stays claimable — `pci-cap-test` and `iommu-fault-test` both rely on that to reach an
|
||||
unmatched NIC. Narrowing it further is a separate change with those fixtures in scope.
|
||||
|
||||
The real gap it closed was the **HPET**: an MMIO window, an IRQ, no user-space driver,
|
||||
and claimable by anyone. Excluded on purpose: the loader's framebuffer (the manager
|
||||
starts before display, so taking it would break the boot screen) and anything with no
|
||||
resources, which grants nothing.
|
||||
|
||||
### Not in this run, and not blocking it
|
||||
|
||||
**Zero-resource devices stay in the kernel.** Moving them out needs an answer to who
|
||||
mints their ids, given that the id is the `device_token` on the usb-transfer wire and
|
||||
the kernel's idempotency is what keeps it stable across a bus restart. Nothing depends
|
||||
on it: both invented ceilings are already gone and this run does not need it. Recorded
|
||||
as future work.
|
||||
|
||||
**Not every driver hellos.** Delivery no longer needs it, so it stands or falls on its
|
||||
own merits — uniform liveness and one class of driver. Also future work.
|
||||
|
||||
---
|
||||
|
||||
### Question 9 — answered: a singleton gets every device that matched it
|
||||
|
||||
The 8042 is one controller described by two ACPI nodes, so it cannot be split across
|
||||
processes. The manager now hands the single `ps2-bus` instance **every** matching node:
|
||||
the first rides the spawn, the rest are transferred to the running instance. Late
|
||||
arrival is safe because the ordering is natural rather than lucky — the controller node
|
||||
carries the ports and is needed at once, the mouse node not until after identify
|
||||
(measured: handed over at 0.336, first touched at 0.456). The count is whatever matched,
|
||||
so a machine with no PS/2 ports or one port needs no special case.
|
||||
|
||||
Discovery turned out not to be special either. The kernel seeds the `acpi-tables` node,
|
||||
so it is in the same boot snapshot the manager already scans for the PCI host bridge —
|
||||
it is handed over at spawn like everything else.
|
||||
|
||||
### Open question 10 — the framebuffer is the last thing anyone claims
|
||||
|
||||
`device_claim` now has exactly two callers: the device manager, which is the acquirer
|
||||
and should have it, and `system/services/display/backend.zig`, whose GOP path claims the
|
||||
kernel-seeded display node.
|
||||
|
||||
Closing `device_claim` breaks the compositor's boot floor. Exempting it puts a hole in
|
||||
the middle of the authority model, in the one place an exemption is most expensive.
|
||||
|
||||
The likely answer is neither. **The framebuffer is not a device** — it is where pixels
|
||||
go, handed over by the loader, and the kernel wraps it in a `DeviceClass.display`
|
||||
descriptor only so `mmio_map` can hand it over write-combining. If that is right, it
|
||||
should leave the device table rather than be exempted from its rules, and the compositor
|
||||
should receive the pixels some other way. That is a change to the display path, which is
|
||||
out of scope for this run.
|
||||
|
||||
### Settled 2026-08-08: the grant rides `system_spawn` (D0)
|
||||
|
||||
Questions 6 and 7 both dissolved on inspection — neither `ps2-bus` nor discovery needs
|
||||
to start speaking `hello`, and `virtio-gpu` has no standalone path to lose. What remains
|
||||
is *where the grant is delivered*, and there are three candidates:
|
||||
|
||||
| | Race window | Cost |
|
||||
|---|---|---|
|
||||
| Transfer after spawn | **yes** | none |
|
||||
| Every driver hellos | no | `ps2-bus` + discovery gain a handshake |
|
||||
| **Grant rides `system_spawn`** | **no — atomic** | one more syscall argument |
|
||||
|
||||
**Take the third.** The manager cannot transfer before the child exists, so a separate
|
||||
transfer always leaves a window in which the child is running and does not yet hold its
|
||||
device. It would close on QEMU every time and open occasionally on a machine with
|
||||
different timing — the exact failure shape this track exists to delete, and not worth
|
||||
introducing while removing the others. Fusing the device into the spawn removes it by
|
||||
construction: the child does not exist until it holds the device. No new knowledge in
|
||||
the kernel — the same rule, *you may give away what you hold*, made atomic with the call
|
||||
that creates the recipient. `system_spawn` uses five of six argument registers, so there
|
||||
is room, and `no_device` is already the sentinel for a driver with no assignment.
|
||||
|
||||
**`hello` for every driver is a good idea on its own merits** — uniform liveness, the
|
||||
deadline applied to all rather than some, and the `speaks_protocol` two-class split
|
||||
leaving the manager (a wedged `ps2-bus` is invisible to its supervisor today). It is
|
||||
D10, kept separate so grant delivery does not force it.
|
||||
|
||||
### Open questions raised by D5 — resolved except question 8
|
||||
|
||||
Delegation is delivered in `onHello`. That works for a driver that says hello, and
|
||||
**two of the four do not**.
|
||||
|
||||
6. **`ps2-bus` does not hello**, and that is deliberate: device-manager.md records
|
||||
"Legacy drivers (e.g. ps2-bus) are supervised and restarted but **not yet required
|
||||
to hello**", and `Driver.speaks_protocol` exists to say so. Either it leaves
|
||||
"legacy", or the grant gets a delivery point that is not `hello`.
|
||||
|
||||
**The `acpi` service is not this case — an earlier version of this question wrongly
|
||||
lumped it in.** It is spawned as `addDriver("discovery", no_device, false)`: the
|
||||
*discovery service*, one per firmware, with **no device assignment at all**, which
|
||||
"finds and claims the acpi-tables (or devicetree-blob) node itself. Not a per-device
|
||||
driver." The manager cannot hand it a device because discovery is what produces the
|
||||
device tree — there is nothing to match against yet. Its question is not "should it
|
||||
hello" but "is the bootstrap exempt from D6, or does the manager claim the
|
||||
acpi-tables node and pass it on".
|
||||
|
||||
**A delivery point that needs no hello already exists in the shape of the code**:
|
||||
`process.spawnSupervised` returns the child's pid, so the manager could transfer
|
||||
immediately after spawn. If that is acceptable, this question mostly dissolves —
|
||||
`ps2-bus` would not need to leave "legacy" and discovery could be handed its node
|
||||
too. The cost is ordering: the child may reach for the device before the transfer
|
||||
lands, where `hello` guarantees it cannot because the child is the one asking. That
|
||||
trade is the actual decision.
|
||||
|
||||
7. **`virtio-gpu`'s "standalone bring-up" — resolved; there is no such path.** An
|
||||
earlier version of this question read the comment "Best-effort: standalone bring-up
|
||||
has no manager" as a boot path that delegation would delete. It is not one.
|
||||
`virtio-gpu`'s `main` requires `argv[1]` and exits without it, and the only source of
|
||||
that argument is the device manager — the chain is ACPI → pci-bus reports the
|
||||
function → `devices.csv` matches `1AF4:1050` → the manager spawns the driver with the
|
||||
id. There is no way to run it without a manager, so nothing is lost.
|
||||
|
||||
What the comment is really about is **resilience**: the hello is best-effort so a
|
||||
driver whose manager has *died* keeps serving, which device-manager.md states
|
||||
("If the manager dies, drivers keep running").
|
||||
|
||||
The residual is one narrow window: the manager spawns a driver and dies before
|
||||
transferring. Today the driver could still claim, because claiming is free-for-all;
|
||||
after D6 it exits and the restarted manager respawns it. That is the better
|
||||
behaviour — a driver holding hardware nobody assigned it is what D6 exists to stop.
|
||||
|
||||
Until these are answered, `device_claim` cannot be closed off at D6 for those three, so
|
||||
**D6 is blocked on questions 6 and 7**.
|
||||
|
||||
8. **Removing zero-resource devices from the kernel needs someone else to mint their
|
||||
ids, and nothing says who.** A USB interface is registered with
|
||||
`resource_count = 0`, and the id `device_register` returns is load-bearing in three
|
||||
places: it is the `child_added` packet's target, it is the class driver's `argv[1]`,
|
||||
and it is the `device_token` of the **usb-transfer wire protocol** — so the id space
|
||||
is visible on the wire, not merely internal. usb-xhci-bus's own comment records the
|
||||
fourth constraint: the kernel's idempotency is what makes "the same port and
|
||||
interface always map back to the same device id" across a bus restart, which is what
|
||||
stops a respawned bus spawning duplicate class drivers.
|
||||
|
||||
So the mover has to answer: who mints the id, how it stays stable across a *bus*
|
||||
restart, how it stays stable across a *manager* restart (open question 3 territory),
|
||||
and whether the wire protocol's `device_token` changes meaning. That is a design
|
||||
step, not a mechanical one.
|
||||
|
||||
**D8 is reordered to run after D9**, which is a correction to the original sequencing.
|
||||
D8's justification was "the authorisation it stood in for exists" — but with D6 blocked
|
||||
it does not, so deleting the shared per-parent cap now would reopen the exhaustion hole
|
||||
it was written for. D9's **per-holder quota** closes that hole independently of
|
||||
authorisation, and does it better: a rogue exhausts its own allowance rather than the
|
||||
table everyone shares. Once the quota exists, the per-parent cap is redundant whether or
|
||||
not D6 has landed.
|
||||
|
||||
D9 also no longer depends on D7. Its original rationale was that zero-resource children
|
||||
are the case that sidesteps containment — true, but a per-holder quota bounds them just
|
||||
as well as anything else, because it counts entries per holder rather than per parent.
|
||||
|
||||
### Settled, so the run does not re-litigate them
|
||||
|
||||
- **The manager claims, it is not granted.** No binary names in the kernel; the rule is
|
||||
"you may give away what you hold". The residual — it rests on the manager claiming
|
||||
first — is stated in the design and is closed later by the spawn capability
|
||||
[drivers.md](device-driver-development/drivers.md) already names as missing.
|
||||
- **The framebuffer is not a device.** It is where pixels go, handed over by the loader,
|
||||
and the compositor uses it as the boot floor until a real display driver announces
|
||||
itself. Nothing in this run touches the display service or its GOP path.
|
||||
- **`maximum_endpoints_per_interface` and the wire structs stay.** Widening them is a
|
||||
protocol change, out of scope.
|
||||
- **A per-holder quota is not a retreat.** Dynamic storage with no bound moves the
|
||||
ceiling to the kernel heap, which is shared and fatal rather than partial. A bound
|
||||
charged to the task that caused it is isolation, and it is declared through
|
||||
[bounds.md](os-development/bounds.md) like anything else.
|
||||
|
||||
### Working rules
|
||||
|
||||
As Run 1, unchanged: work in `/Users/danielsamson/Gitea/daniel/danos` on
|
||||
`claude/bounds-track`; every step lands with a test that fails before the fix, verified
|
||||
by restoring the old behaviour; full suite green before each commit; never two suites at
|
||||
once (`pgrep -f qemu_test.py`); 60 GiB free; `git commit -F` with no `Co-Authored-By`;
|
||||
update this table before starting the next step. **If a step needs a decision that is
|
||||
not written down, stop it, add the question below, and move on** — Run 1's first step
|
||||
was wrong and stopping was the right call.
|
||||
|
||||
**Suite:** 115/115 at the start of the run.
|
||||
**Branch:** `claude/bounds-track`.
|
||||
|
||||
### What this run deliberately does not touch
|
||||
|
||||
Phases 2 and 3 below — the authorisation gate and moving the inventory to the device
|
||||
manager — are **out of scope for unattended work**. They decide whether the OS is
|
||||
secure, and they are currently a direction rather than a specification: what a device
|
||||
capability *is*, which syscalls change, what replaces `device_claim` for its seven
|
||||
callers, how a driver spawned bare behaves. Those want a design session, the way
|
||||
`/protocol` had one.
|
||||
|
||||
Also out of scope: anything touching `maximum_device_resources` (a wire struct, so a
|
||||
trust-boundary change, not a resize), and the non-device bounds the audit found in FAT,
|
||||
the VFS, logger, init, display and boot.
|
||||
|
||||
### Open questions this run must not answer on its own
|
||||
|
||||
Recorded here rather than guessed. If a step runs into one, it stops and writes the
|
||||
question down instead of inventing an answer.
|
||||
|
||||
1. **`device_enumerate` probably narrows rather than retires.** The device manager
|
||||
calls it to find `pci_host_bridge` nodes — it cannot ask itself. The likely shape is
|
||||
that the kernel keeps the *firmware-discovered roots* (which by principle 5 it holds
|
||||
for real reasons, since they come from ACPI rather than a driver's say-so) and
|
||||
everything a driver registered lives in the manager. Not decided.
|
||||
2. **A device-manager restart has no re-enumerate handshake.** If only the manager
|
||||
dies, the buses are alive and never re-send `child_added`, so a restarted manager
|
||||
comes back blind. The manager is restartable by design; nothing implements this.
|
||||
3. **Which adversarial tests I1–I3 need.** The audit's six real defects were all found
|
||||
by asking what an attacker would do, and the suite had never asked. "Add adversarial
|
||||
cases" is not executable until the attacks are named.
|
||||
4. **Reclamation is not a death-sweep problem, and L1 as written would have broken the
|
||||
restart path.** Found on the first attempt at it. The audit is right that `count`
|
||||
never decreases, but *death is the wrong trigger*:
|
||||
- The broker keeps entries deliberately: "The devices stay in the table — they
|
||||
describe hardware, which did not go away — only their ownership clears." A driver
|
||||
dying does not unplug anything.
|
||||
- Device ids must stay **stable across a bus restart**, because
|
||||
`device-manager.driverForDevice` dedupes by `device_id` so that "a re-report after
|
||||
a bus restart must not spawn a second instance". Stability comes from the
|
||||
idempotency scan returning the existing id — removing entries on death would give
|
||||
a restarted bus fresh ids and spawn duplicate driver instances.
|
||||
- Everything else a task holds *is* already reclaimed on every path out:
|
||||
`irq.releaseOwner`, `iommu.releaseAllOwnedBy`, `dmaRegistryReleaseOwner`, then the
|
||||
broker's claims (`process.releaseTaskResourcesLocked`).
|
||||
|
||||
So the real leak has two sources, and neither is death: a device that genuinely
|
||||
**goes away** (hot-unplug) has no retirement path, and a bus that enumerates
|
||||
*differently* on restart leaves its stale entries behind forever. Both are the device
|
||||
manager's inventory problem — phase 3 — and both need the id-stability question
|
||||
answered first (tombstone-and-reuse aliases stale ids held by another process;
|
||||
generation-tagged ids change the id encoding, which is ABI). Not an unattended
|
||||
decision.
|
||||
5. **Driver descriptor parsing cannot be host-tested, so L5's correctness fix ships
|
||||
without a direct test.** The endpoint-misattribution bug lives in
|
||||
`parseConfiguration`, a pure function over a byte blob — exactly the shape a host
|
||||
unit test wants, and `usb-storage/scsi.zig` and `usb-hid/hid-report.zig` already do
|
||||
this. But `usb-xhci-library.zig` imports `memory`, `mmio` and `time`, so it cannot
|
||||
be a standalone host-test root, and QEMU offers no device that would exercise the
|
||||
path anyway: the largest available is `usb-audio,multi=on` at 2 interfaces and 211
|
||||
bytes, against a cap of 4.
|
||||
|
||||
Three ways out, and picking one is a judgement about house style rather than a
|
||||
mechanical step: extract the parser to its own file and wire `usb-abi` into a test
|
||||
module (build-support currently resolves module names only for `userBinary`);
|
||||
extract it and import `usb-abi` by relative path (against the import-by-name
|
||||
convention); or accept QEMU-only coverage and say so.
|
||||
|
||||
Mitigating, and the reason this is recorded rather than blocking: after the fix the
|
||||
bug is unreachable **by construction**, not by the added `else`. Interfaces are now
|
||||
allocated to exactly the count the descriptor declares, so `interface_count` can
|
||||
never reach `interfaces.len` mid-parse. The `else` is belt-and-braces for the
|
||||
255-interface clamp. The alternate-setting path that shares it *is* exercised —
|
||||
`usb-audio` has alternate settings, and the `usb-large-descriptor` case walks them.
|
||||
|
||||
### Working rules for the run
|
||||
|
||||
- Work in `/Users/danielsamson/Gitea/daniel/danos` (not a worktree), on
|
||||
`claude/bounds-track`.
|
||||
- **Every step lands with a test that fails before the fix**, verified by temporarily
|
||||
restoring the old behaviour and watching exactly the intended assertion flip. A test
|
||||
that passes both ways is not a test.
|
||||
- Full QEMU suite green before each commit. Never run two suites at once — check
|
||||
`pgrep -f qemu_test.py` first; a second concurrent run produces false triple faults
|
||||
because both share `zig-out`.
|
||||
- Check at least 60 GiB free before starting a suite.
|
||||
- Commit with `git commit -F <file>`, never `-m` (a backtick in a message is executed
|
||||
by the shell and silently eats a word). No `Co-Authored-By` trailers.
|
||||
- Update the Live state table **before** starting the next step.
|
||||
- If a step needs a decision that is not written down here, stop, add it to the open
|
||||
questions above, and move to the next step.
|
||||
|
||||
---
|
||||
|
||||
## The principles this is derived from
|
||||
|
||||
1. **danOS is a microkernel.** Minimise what the kernel is responsible for; move
|
||||
responsibility to user space so it can be restarted, or fixed live during
|
||||
development, without taking the system down.
|
||||
2. **Implement the specifications correctly**, with the limits those specifications
|
||||
define — not limits we decide.
|
||||
3. **Move as much responsibility as possible to user space** (the device manager).
|
||||
4. **What remains in the kernel is minimal.**
|
||||
5. **What remains in the kernel is there for security or for a hardware limitation.**
|
||||
Nothing else earns a place.
|
||||
|
||||
Principle 5 is the test every bound is put to. For each one: *is this here because of
|
||||
security, or because of a hardware limitation?* If neither, the storage does not belong
|
||||
in the kernel and the bound is not a number to be resized — it is a thing to be moved or
|
||||
deleted.
|
||||
|
||||
Applying it to the case that started this:
|
||||
|
||||
- `maximum_devices = 64` bounds an inventory of hardware. An inventory is neither a
|
||||
security control nor a hardware limitation. **The table is in the wrong place**; the
|
||||
number is a symptom.
|
||||
- `maximum_children_per_parent = 16` exists because `device_claim` is unauthenticated —
|
||||
any process can claim any unclaimed device ([devices-broker.zig:164](../system/kernel/devices-broker.zig:164)
|
||||
checks only that the device exists and is free). The cap is a crude proxy for an
|
||||
authorisation the kernel does not perform. **Fix the authorisation and the cap has
|
||||
nothing to defend.**
|
||||
- `maximum_domains = 64` bounds IOMMU translation domains. Security — stays in the
|
||||
kernel. But VT-d and AMD-Vi both *report* how many domains they support in a
|
||||
capability register. Principle 2: read it. We chose 64 without asking.
|
||||
|
||||
## The security invariants
|
||||
|
||||
Every phase must leave all five standing. This is the "without punching a hole" half of
|
||||
the brief, and each phase below states how it is checked.
|
||||
|
||||
- **I1 Containment.** A process may map only physical memory inside a resource it was
|
||||
granted. A bus may subdivide only what it already holds.
|
||||
- **I2 Confinement.** A DMA-capable device is under IOMMU translation before its driver
|
||||
can program it, or it is not driven at all.
|
||||
- **I3 No self-granted authority.** A process holds what it was handed. It cannot name
|
||||
its way into holding more.
|
||||
- **I4 Death releases everything.** Every resource a task held is reclaimed when it
|
||||
dies, on every path out.
|
||||
- **I5 Refusal is attributable.** Every refusal names the rule that refused it.
|
||||
|
||||
## Phase 0 — Done
|
||||
|
||||
- **Errno attribution.** One errno space in `system/abi.zig`; `device_register`'s six
|
||||
refusals and `device_claim`'s three are distinct codes; call sites name the reason;
|
||||
`pci-bus` reconciles found against registered. (I5)
|
||||
- **Idempotency ordering.** A re-registration consumes no slot, so a full parent
|
||||
re-admits an identical child. A restarted bus is no longer billed for what it
|
||||
rediscovers.
|
||||
|
||||
Suite 114/114.
|
||||
|
||||
## Phase 1 — Reclamation
|
||||
|
||||
**Nothing may become dynamic before this.** Today `count` only ever increases and
|
||||
`releaseAllOwnedBy` clears a dead driver's *claims* but not its *registrations*. With a
|
||||
fixed table that is a slow march to the cap; with dynamic storage it is an unbounded
|
||||
leak, and every supervisor restart makes it worse.
|
||||
|
||||
- Extend the existing death sweep so a task's registrations go with its claims.
|
||||
- A registration whose owner is gone is removed; its children are re-parented or removed
|
||||
with it (they cannot outlive the authority that published them).
|
||||
- Test: register under a claimed parent, kill the owner, assert the entries are gone and
|
||||
the ids are not reused while any handle to them lives.
|
||||
|
||||
Invariant: **I4**.
|
||||
|
||||
## Phase 2 — Close the authorisation hole
|
||||
|
||||
The device manager already decides which driver gets which device — it matches against
|
||||
`devices.csv` and spawns the driver with the device id as `argv[1]`. Nothing binds that
|
||||
decision to the kernel's `claim`. A driver passes an integer; the kernel checks only
|
||||
that the device is free.
|
||||
|
||||
Per principles 3 and 5: **the decision stays in user space; the kernel enforces only
|
||||
possession.** The manager hands the driver the device it matched; the kernel's job is
|
||||
that a driver holds what it was handed and nothing else.
|
||||
|
||||
- The manager passes a device to the driver it spawned, over the existing cap-passing
|
||||
path. Possession is the authority.
|
||||
- `device_claim` stops being a way to *acquire* a device by naming it.
|
||||
- Exclusivity stops being a broker refusing a second claimant and becomes the ordinary
|
||||
property of a thing only one process was given.
|
||||
|
||||
**`maximum_children_per_parent` is deleted here**, because after this a bus driver's
|
||||
children are the devices it actually enumerated under a bus it was actually given, and
|
||||
the rogue-driver-fills-the-table threat the cap was written for no longer exists.
|
||||
|
||||
Invariants: **I3** (the point of the phase), **I1** (containment is unchanged and still
|
||||
checked on every subdivision), **I5**.
|
||||
|
||||
Acceptance: a driver that names a device it was not given is refused, with its own
|
||||
errno. The QEMU suite gains an adversarial case for it — the audit's lesson was that
|
||||
"the suite contains no attacker".
|
||||
|
||||
## Phase 3 — The inventory moves to user space
|
||||
|
||||
The kernel reads only three things out of a device descriptor: **physical ranges** (to
|
||||
check a mapping falls inside one), **interrupt numbers**, and **one PCI BDF** (to key an
|
||||
IOMMU domain). Vendor and device ids, class triples, subsystem ids, human-readable
|
||||
names, bus numbers and parent links are stored solely so `device_enumerate` can hand
|
||||
them back. That is the kernel acting as a distribution mechanism for data it does not
|
||||
use — principle 5 excludes it.
|
||||
|
||||
- **Zero-resource devices leave the kernel entirely.** A USB device addressed through
|
||||
its controller conveys no mapping authority; there is nothing for the kernel to
|
||||
enforce. It is pure inventory and belongs to the device manager. (This is also the
|
||||
case that sidesteps containment, which is why the cap existed.)
|
||||
- Identity and topology move to the manager, which already receives them as
|
||||
`child_added` reports and already holds the authoritative picture.
|
||||
- `device_enumerate` retires; callers ask the manager, whose protocol already reserves
|
||||
an `enumerate` verb. Public-ABI change — `docs/os-development/vdso.md` documents it.
|
||||
- What the kernel keeps: for each device that carries resources, the ranges, the GSIs,
|
||||
the BDF, and the owner.
|
||||
|
||||
After this, the kernel's table holds only resource-bearing devices, and the remaining
|
||||
count is bounded by what the machine physically has rather than by us.
|
||||
|
||||
Invariants: **I1**, **I2** unchanged — both operate on resources, which do not move.
|
||||
**I4** must be re-checked: the manager's table now needs its own reclamation, and it is
|
||||
restartable, so it must be able to rebuild from the buses.
|
||||
|
||||
## Phase 4 — Ask the hardware and the specification
|
||||
|
||||
Principle 2, applied to every remaining bound. Each of these is a number the machine or
|
||||
the standard already states, which we replaced with a guess. Independent of each other;
|
||||
can proceed in any order.
|
||||
|
||||
| Today | Ask instead |
|
||||
|---|---|
|
||||
| `maximum_domains = 64` (IOMMU) | the VT-d / AMD-Vi capability register reports the domains supported |
|
||||
| `max_devices = 8` (xHCI slots) | `HCSPARAMS1.MaxSlots` — the controller says (1–255) |
|
||||
| `max_interfaces = 4`, `max_endpoints` | the configuration descriptor says |
|
||||
| `blob: [512]u8` (USB config) | the device's `wTotalLength` |
|
||||
| `below: [64]Range` (memory map) | UEFI reports the descriptor count |
|
||||
| AML blobs capped at 6 | the XSDT's length field gives the entry count |
|
||||
| `maximum_cpus = 128` | the MADT entry count |
|
||||
| `maximum_gsi = 24` | the I/O APIC's redirection-entry count; and more than one I/O APIC exists |
|
||||
| MSI-X vectors | the capability's table-size field (up to 2048) |
|
||||
|
||||
Several of these are in user space already (xHCI, USB descriptors) and are ordinary
|
||||
allocations — principle 1 means those are also the safest to do first, since a mistake
|
||||
restarts a driver rather than the machine.
|
||||
|
||||
Two in this table are **also** correctness fixes the audit found, and should carry their
|
||||
regression tests: the xHCI `max_interfaces` path misattributes a fifth interface's
|
||||
endpoints to interface 3, and `below: [64]Range` silently turns occupied RAM into a PCI
|
||||
aperture — which is an **I1 violation reachable on real hardware**, not merely a lost
|
||||
device. That one is the highest-priority item in this phase.
|
||||
|
||||
## Phase 5 — What legitimately remains
|
||||
|
||||
After phases 1–4 the survivors should be only:
|
||||
|
||||
- **Pre-allocator storage**: the PMM's own frame bitmap, the memory map the loader hands
|
||||
over, the bootstrap page tables. You cannot allocate the allocator. (Hardware/boot
|
||||
limitation — principle 5 admits these.)
|
||||
- **Interrupt-context storage**: the IST stack and anything an exception path touches
|
||||
without allocating.
|
||||
- **Wire structures** whose layout the other side of a trust boundary parses.
|
||||
- **Facts that are not ceilings**: a page is 4096 bytes; an ACPI name segment is 4.
|
||||
|
||||
Each is declared per [bounds.md](os-development/bounds.md) — what it counts, who decides
|
||||
its size, what it protects, what happens at the limit, how you find out. And two numbers
|
||||
that must agree agree in code, not in a comment:
|
||||
|
||||
```zig
|
||||
comptime {
|
||||
if (maximum_domains != devices_broker.maximum_devices)
|
||||
@compileError("iommu.confined is indexed by device id; an id past its end is " ++
|
||||
"left unconfined while confineDevice still reports success");
|
||||
}
|
||||
```
|
||||
|
||||
## The one that must not wait
|
||||
|
||||
[`iommu.zig:107`](../system/kernel/iommu.zig:107) — `if (device_id >= confined.len) return
|
||||
true;` — returns *success* without confining. It is unreachable today only because
|
||||
device ids stop at 64. **Phases 3 and 4 both change the device count, and either makes
|
||||
it live.** Fix it before them: out of range must refuse, never allow. (I2)
|
||||
|
||||
This is also the standing rule the audit argues for: at a bound, the safe direction is
|
||||
refusal. A ceiling that fails open is not a limit, it is a switch that turns the
|
||||
protection off.
|
||||
|
||||
## How this is verified
|
||||
|
||||
- The QEMU suite is the arbiter at every step; it is 114 cases and must stay green.
|
||||
- Each fix lands with a test that **fails before it** — as the idempotency reorder did,
|
||||
where exactly one assertion flipped.
|
||||
- Adversarial cases for I1–I3 specifically: the audit's six real defects were all found
|
||||
by asking "what would an attacker do", and the suite had never asked.
|
||||
- The Ryzen is the acceptance test. It is the machine that found this, and the one that
|
||||
proves it fixed.
|
||||
|
||||
## Sequencing
|
||||
|
||||
Phase 1 gates everything. Phase 2 gates phase 3 — the inventory cannot move until
|
||||
authority is sound, or moving it is the hole. Phase 4 is independent and its user-space
|
||||
items are the safest work in the track. Phase 5 is the record of what survived.
|
||||
|
||||
The IOMMU fail-open is fixed before phase 3 or 4 touches the device count.
|
||||
@@ -74,13 +74,26 @@ one world.
|
||||
|
||||
| Direction | Packet | Purpose |
|
||||
|---|---|---|
|
||||
| driver → manager | `hello { role, version }` @ the assigned device | confirms the argv assignment, starts the deadline clock |
|
||||
| driver → manager | `hello { role, wants_channel, version }` @ the assigned device | confirms the assignment, starts the deadline clock — **and moves the channels** (below) |
|
||||
| bus → manager | `child_added { parent, bus_address, identity, bus, vendor, device, subsystem, hid }` @ the registered device id | one node the bus discovered |
|
||||
| bus → manager | `child_removed { parent, bus_address }` | unplug, or the bus lost it |
|
||||
| app → manager | `enumerate` (reserved verb 1) | snapshot of the tree: one `ChildEntry` per record in the reply's tail |
|
||||
| app → manager | `enumerate` (reserved verb 1) @ start index | one page of the tree, one `ChildEntry` per record in the reply's tail; a short page is the end |
|
||||
| app → manager | `subscribe` (reserved verb 2) | receive published add/remove events; the subscriber's endpoint rides as the call's capability |
|
||||
| manager → app | `child_added` / `child_removed` events | the same two structs, pushed rather than called |
|
||||
|
||||
**`hello` is also the establishment plane for driver-layer channels**
|
||||
([communication.md](../os-development/communication.md) "Establishment: two
|
||||
planes, one namespace"). A provider's hello carries its serving endpoint UP as
|
||||
the call's capability (the xHCI bus its transfer endpoint, usb-storage its
|
||||
block endpoint — several processes provide those contracts, so neither is ever
|
||||
a registry name). A hello with `wants_channel` set is answered with a
|
||||
capability DOWN: for a `.device`-role driver, the channel of its device's
|
||||
*reporter* (a class driver reaching its own controller); for a `.consumer`
|
||||
(not a spawned driver at all — fat looking for its volume), the channel of the
|
||||
driver *bound to* the target device. No channel in an acked reply means the
|
||||
provider is mid-restart and its re-report is on the way — retryable, never a
|
||||
verdict.
|
||||
|
||||
The watcher table behind those last two rows is the **service harness's**
|
||||
(`service.Subscribers`, shared with input and power), not the manager's: it
|
||||
answers `subscribe`/`unsubscribe`, frames each event once for the fan-out, and
|
||||
@@ -113,10 +126,11 @@ the stop sequence and the restart policy. Everything else lifecycle-shaped
|
||||
(terminate, the common `ping` liveness call, exit reasons) arrives through
|
||||
[process-lifecycle.md](../os-development/process-lifecycle.md)'s vocabulary, not this protocol.
|
||||
|
||||
Assignment stays argv (`usb-xhci-bus <device id>`) for now — simple, and it works.
|
||||
The step after `hello` exists is delegation: the manager claims (or is granted) the
|
||||
devices and passes the claim to the driver over IPC (the M13 capability-transfer
|
||||
mechanism), replacing first-come-first-served `device_claim` with policy. Identity in
|
||||
Delegation is BUILT (docs/os-development/device-authority.md "As built"): the
|
||||
device arrives WITH the spawn — `system_spawn`'s sixth argument moves the
|
||||
claim before the child's first instruction — and the id still rides argv for
|
||||
the driver to name itself by. A given device is a loan that returns to the
|
||||
manager on the holder's death. Identity in
|
||||
`child_added` is per-bus: PCI children carry the class triple (`pci_class`, as the
|
||||
xHCI match already uses); USB children carry the (class, subclass, protocol) triple
|
||||
from usb-ids.zig — each bus's native language, decoded by the shared ids modules.
|
||||
@@ -133,13 +147,19 @@ On a death notification:
|
||||
Clean exit → it meant to; don't restart. Fault or missed `hello` deadline →
|
||||
restart with **backoff**, and a crash-loop cap (three fast deaths → mark failed,
|
||||
stop respawning, log loudly; a later `reload` to the manager can retry).
|
||||
2. **Prune the subtree** the dead bus driver reported. Its children describe
|
||||
protocol state (xHCI slot ids, transfer rings) that died with the process;
|
||||
keeping the nodes would be keeping a lie. Watchers receive `child_removed` — the
|
||||
input service losing, then regaining, a keyboard is the *honest* description of
|
||||
what happened. The restarted instance rediscovers and re-reports.
|
||||
3. **The claim is already free** because the kernel released it at death — the
|
||||
restarted instance claims the same controller and comes up.
|
||||
2. **Prune the subtree** the dead bus driver reported, **reaping the class
|
||||
drivers bound to its children**. Its children describe protocol state (xHCI
|
||||
slot ids, transfer rings) that died with the process; keeping the nodes
|
||||
would be keeping a lie — and the class drivers hold channels into the dead
|
||||
process they cannot observe dying (an HID driver blocks on reports that
|
||||
simply never come). Each is killed and its entry cleared, so the matcher
|
||||
can respawn a fresh generation when the re-report arrives; their hellos
|
||||
fetch the successor's channel. Watchers receive `child_removed` — the input
|
||||
service losing, then regaining, a keyboard is the *honest* description of
|
||||
what happened.
|
||||
3. **The device is already back** because the loan returned at death — the
|
||||
manager re-delegates the same controller with the respawn, and the kernel
|
||||
confines it afresh.
|
||||
|
||||
Who supervises the supervisor: **init** (PID 1), which already supervises the
|
||||
services it starts. If the manager dies, drivers keep running (they hold their
|
||||
|
||||
@@ -0,0 +1,122 @@
|
||||
# Establishment planes: the plan
|
||||
|
||||
*2026-08-09. The design is settled in
|
||||
[communication.md](os-development/communication.md) ("Establishment: two planes, one
|
||||
namespace"): driver-layer channels stop being registry names and are routed at
|
||||
establishment by the device manager, which owns the topology. This converts both live
|
||||
seams — `usb-transfer` and `block` — in one flag-day, and fixes the restart hole the
|
||||
scoping audit found on the way. Found by a real machine: three xHCI controllers, one
|
||||
name, a mouse nobody could reach.*
|
||||
|
||||
**Non-goals, stated up front.** fat stays single-volume (multi-volume mounting, volume
|
||||
identity/content-probing, and fat's own reconnection after a storage restart are the
|
||||
M21 remount track). The input service's last-writer-wins button mask across two mice
|
||||
is a known quirk, not part of this change. Nothing here touches the app-facing names —
|
||||
`input`, `display`, `ufs` keep their exclusive registry binds.
|
||||
|
||||
## What the scoping reads established
|
||||
|
||||
- **Capabilities already travel both directions of a synchronous call, and only
|
||||
there.** Request-direction via `ipc.callCap` (r9 → `shareCapability` at server
|
||||
dequeue, a refcounted copy); reply-direction via `ipc_reply_wait`'s `send_cap`
|
||||
(surfaces as the client's `Reply.cap`) — init's registry is the working in-tree
|
||||
precedent (`pending_capability`). Pushes structurally cannot carry one. **No kernel
|
||||
change is needed.**
|
||||
- **The one mechanism gap is the service harness**: `service.run` hardcodes the reply
|
||||
capability to null and `Answer` has no capability slot; the device manager runs on
|
||||
that harness. It needs a reply-cap out-slot following init's pattern, with null
|
||||
staying the untouched default for every other service.
|
||||
- **The manager already stores the routing fact**: `Child{parent, bus_address,
|
||||
device_id, reporter}` — consumer's device id → `children[].reporter` →
|
||||
`driverByProcess(reporter)` → the (new) stored endpoint. `Hello.role` (.bus vs
|
||||
.device) already exists on the wire to distinguish provider hellos (cap up) from
|
||||
consumer hellos (cap down); the manager currently never reads it.
|
||||
- **The restart hole (new finding):** today, when a bus instance dies, its class
|
||||
drivers are zombies — HID drivers block forever on interrupt reports that will
|
||||
never come, usb-storage answers fat with errors forever; the manager respawns
|
||||
neither, and the matcher's dedupe actively prevents a respawn on re-report. The
|
||||
user's core goal is restart-a-driver-live; this plan closes the hole.
|
||||
- **Blast radius:** every QEMU case boots from USB storage, so this seam failing
|
||||
fails the whole suite, and the load-bearing serial markers (`fat: mounted
|
||||
/volumes/usb`, `usb-storage: ready`, `fat-test: ok`) must not change.
|
||||
|
||||
## P0 — mechanics (library + manager, no behavior change yet)
|
||||
|
||||
1. **Harness reply-capability out-slot.** A handler nominates a capability for its
|
||||
reply (init's `pending_capability` idiom moved into `service.run`'s loop); default
|
||||
null; the existing Arrival take/release ownership discipline is unchanged. Every
|
||||
other service compiles and behaves identically.
|
||||
2. **`driver.hello` grows both directions.** Provider form: carry the caller's
|
||||
serving endpoint up (`ipc.callCap`). Consumer form: return `Reply.cap`. One call
|
||||
can do both (a storage driver hands its block endpoint up and receives its bus
|
||||
channel down in the same hello).
|
||||
3. **Manager stores and routes.** `Driver` gains an endpoint field; the cap is
|
||||
claimed at `onMessage` level (the `capability_claimed` pattern usb-xhci-bus
|
||||
already uses — the harness's private claim flag composes, verified). `onDriverExit`
|
||||
**must** `ipc.close` the stored handle (a dead owner's endpoint survives as a
|
||||
refcounted slot; repeated restarts would exhaust the manager's 32-slot table —
|
||||
boot-fatal). Consumer hello with an unknown device id (bus died, not yet
|
||||
re-reported) gets a *retryable* refusal, never a permanent one.
|
||||
|
||||
## P1 — the usb-transfer seam
|
||||
|
||||
- Bus: delete the `bindPatiently("usb-transfer")` block and its serve-unnamed
|
||||
fallback; hello carries the endpoint up.
|
||||
- `usb.open(device_id)` takes the channel from the class driver's hello instead of
|
||||
the name lookup; retry patience stays ≥ today's window (the provider's hello
|
||||
precedes the consumer's spawn in the normal order — the retry covers respawn races).
|
||||
- usb-storage's *consumer* side uses the same lineage (its own device id).
|
||||
- `protocol.csv`: delete the bind row (67) and the three open rows (119/120/122) in
|
||||
the same commit as the code — grants and code move together or the old path dies
|
||||
silently first.
|
||||
- Fixtures: conformance table row, the usb-report kernel-side registry spawn.
|
||||
|
||||
## P2 — the block seam
|
||||
|
||||
- usb-storage: drop `.service = "block"` (the field is optional; the harness serves
|
||||
nameless — pci-bus and logger already do); its `initialise` keeps the endpoint it
|
||||
currently discards and hellos it up. Exit-code contract preserved: device absent =
|
||||
clean exit (no restart), device-present failure = exit(1) (restart with backoff) —
|
||||
establishment failure is the latter, never a silent pre-hello death like today's
|
||||
second-stick `-EBUSY`.
|
||||
- fat: `block.tryOpen`'s name lookup becomes manager-routed — enumerate, find the
|
||||
block-class provider, hello for its channel. One volume: unambiguous. Two volumes:
|
||||
fat takes the first by device-id order — deterministic within a boot, and no worse
|
||||
than today's bind race; choosing *the boot volume* by content is deferred to M21
|
||||
(risk noted: `/system/configuration` and `/system/logs` mounts ride this choice).
|
||||
- fat gets a `protocol.csv` open grant for `device-manager`; rows 68/99 die.
|
||||
- Delete `block.open` (zero callers, dead code).
|
||||
|
||||
## P3 — restart integrity (the zombie fix)
|
||||
|
||||
- **A reporter's death reaps its subtree.** When a bus/storage instance dies, the
|
||||
manager already prunes its children; now it also kills the class drivers spawned
|
||||
for those children and clears their entries, so the matcher's dedupe stops blocking
|
||||
the respawn. On re-report (ids are stable — the `P<port>I<iface>` registration
|
||||
identity), the matcher spawns fresh class drivers, which hello and receive the
|
||||
successor's channel. Restart a bus driver live: the subtree rebuilds itself.
|
||||
- usb-storage on `-EPEER` from its bus: exit(1) → supervisor restart with backoff →
|
||||
fresh hello. (fat's own reconnection: M21.)
|
||||
- Extend the existing bus-restart drill (`test-usb-restart`, the usb-report case) to
|
||||
assert the subtree *works* after the restart — a post-restart keyboard event, not
|
||||
just a re-report. **Discrimination: today that assertion fails (zombie).**
|
||||
|
||||
## P4 — the multi-instance proof
|
||||
|
||||
- New QEMU case: **two xHCI controllers**, keyboard on one, storage on the other;
|
||||
assert both class drivers come up and function. **Discrimination: without the
|
||||
conversion this case fails exactly like the Ryzen mouse** (second controller's
|
||||
child unreachable). This is the machine's bug, pinned in the suite forever.
|
||||
- Full suite green; the four existing second-controller cases (usb-hub and friends)
|
||||
re-checked against their expectations.
|
||||
- Docs updated with the code: device-manager.md, drivers.md (hello), the usb-transfer
|
||||
and block protocol docblocks, device-authority.md as-built appendix.
|
||||
|
||||
## Order and verification discipline
|
||||
|
||||
P0 → P1 → P2 → P3 → P4, one commit per coherent step, suite green at every phase
|
||||
boundary (P0 and P1 may share a verification run — P0 alone changes no behavior).
|
||||
Every new test must be shown to fail against the old behavior before it counts
|
||||
(the discrimination rule). One QEMU suite at a time. The Ryzen re-test at the end is
|
||||
the acceptance run: mouse moves, and killing one bus instance from a shell one day
|
||||
only blinks the devices on that controller.
|
||||
@@ -0,0 +1,403 @@
|
||||
# Fixed-bounds audit
|
||||
|
||||
*2026-08-07. A tree-wide audit of every compile-time ceiling on a runtime quantity,
|
||||
commissioned after an AMD Ryzen desktop booted to a working compositor with no USB
|
||||
and no storage. Seven parallel sweeps, adversarial verification of each finding, and
|
||||
a completeness critic. 235 bounds confirmed.*
|
||||
|
||||
The audit was not commissioned to fix that machine. It was commissioned to answer a
|
||||
different question: **why did a machine have to find this?** The answer is in the
|
||||
first table below, and it is not that anyone failed to predict an AMD desktop.
|
||||
|
||||
## What the audit found
|
||||
|
||||
| | |
|
||||
|---|---|
|
||||
| Bounds confirmed | **235** |
|
||||
| Decided by hardware or external data, not by us | **139** (59%) |
|
||||
| Recorded in `system/parameters.zig` | **5** (2%) |
|
||||
| With no comment explaining the number at all | **97** (41%) |
|
||||
| Silent when exceeded — no log, no counter, no error | **171** (73%) |
|
||||
| Severity critical / high | **17 / 26** |
|
||||
|
||||
The project has a stated convention: tunables live in `system/parameters.zig` with
|
||||
their reasoning attached. Two percent of them do. That is the finding — not any
|
||||
individual number.
|
||||
|
||||
And 59% of these are not tunables at all. They are guesses about someone else's
|
||||
computer: how many PCI functions a board has, how many SSDTs its firmware ships,
|
||||
how many descriptors its memory map carries, how many interfaces a USB headset
|
||||
declares. A fixed bound on a quantity the machine decides is not a knob. It is a
|
||||
defect with a plausible-looking number in it.
|
||||
|
||||
## Why "raise the number" is not available
|
||||
|
||||
This is the part that matters most, and it was found by the completeness critic
|
||||
rather than by any of the seven sweeps.
|
||||
|
||||
`system/kernel/iommu.zig:97` sizes the per-device confinement table by
|
||||
`maximum_domains = 64` — "64 mirrors devices-broker's device cap" — and indexes it
|
||||
by **device id**. Line 107:
|
||||
|
||||
```zig
|
||||
pub fn confineDevice(device_id: u64, bdf: u16, owner: u32) bool {
|
||||
if (!active) return true;
|
||||
if (device_id >= confined.len) return true; // unusual id; leave it to fail-open
|
||||
```
|
||||
|
||||
It returns `true` — success — without confining the device. The doc comment three
|
||||
lines above states the opposite invariant: *"a claim that can't be confined must not
|
||||
stand."*
|
||||
|
||||
Device ids are assigned `d.id = count` with `count < maximum_devices`, so today ids
|
||||
run 0–63 and that branch is unreachable. **It becomes reachable the moment
|
||||
`maximum_devices` is raised above 64.** Every device with id ≥ 64 would then be
|
||||
claimed by a ring-3 driver, reported as successfully confined, and left outside every
|
||||
IOMMU domain — an unconfined DMA master with a driver holding it.
|
||||
|
||||
So the one-line fix for the Ryzen — raise 64 to 512 — is a privilege escalation. Not
|
||||
inelegant: escalating. Nothing in the type system, the tests or the comments would
|
||||
have caught it, because the two constants agree only by a sentence in a comment.
|
||||
|
||||
## The Ryzen was three bugs, not one
|
||||
|
||||
Each of these independently produces "the xHCI and SATA controllers are missing."
|
||||
Fixing any one of them leaves the machine broken by the next.
|
||||
|
||||
1. **`devices-broker.zig:33`, `maximum_children_per_parent = 16`** — `pci-bus`
|
||||
registers *every* discovered function as a direct child of the one host bridge, so
|
||||
16 is the ceiling on PCI functions for the whole machine. Enumeration runs in
|
||||
bus/device/function order, so low-numbered chipset functions consume all 16 and the
|
||||
high-numbered controllers — xHCI at 0x14, SATA at 0x17 — are refused. This is the
|
||||
one that fired.
|
||||
2. **`device-manager.zig:174`, `maximum_children = 64`** — the *userspace* inventory
|
||||
has its own flat table. ACPI contributes ~34 nodes before PCI is scanned. Refusal
|
||||
returns `-ENOSPC`, which `pci-bus.zig:246` discards and `acpi.zig:257` swallows in
|
||||
an empty `catch {}`. The log line "child added" is printed *before* the status is
|
||||
consulted, so the log says the device was added when it was not.
|
||||
3. **`acpi.zig:651`, `gaps = [3]Range`** — only three sub-4 GiB MMIO holes become PCI
|
||||
bridge apertures. A BAR landing in a fourth hole fails containment and is refused,
|
||||
indistinguishably from a full table.
|
||||
|
||||
Three ceilings, three teams of one, one symptom. "Raise the limit" would have moved
|
||||
the failure to the next one and produced a second debugging session from a second
|
||||
photograph.
|
||||
|
||||
## Two latent security findings
|
||||
|
||||
**`acpi.zig:627`, `below = [64]Range`.** Firmware memory-map entries below 4 GiB, used
|
||||
to derive the PCI bridge's apertures *from the gaps between described regions*. Past the
|
||||
64th entry: `continue`. A dropped region is not merely missing — it vanishes from the
|
||||
"described space" the gap-finder subtracts, so occupied physical memory is concluded to
|
||||
be a free MMIO hole and registered as a bridge aperture. `devices-broker.contains()`
|
||||
then admits a child BAR covering RAM, and its claimant can `mmio_map` it: a ring-3
|
||||
read/write window onto kernel memory. Real UEFI maps carry 60–200 descriptors; OVMF
|
||||
carries 15–25, which is why the suite has never approached it.
|
||||
|
||||
**`devices-broker.zig:217`, the PCI requester id is a `u16`** with no segment field,
|
||||
propagated unbroken into both IOMMU backends. On a multi-segment machine two physically
|
||||
distinct functions alias onto one translation structure.
|
||||
|
||||
Both are the same shape as the finding above: a bound whose failure mode is not "we run
|
||||
out" but "the protection silently stops applying."
|
||||
|
||||
## Bugs found in passing
|
||||
|
||||
Not bounds, but found by looking at what happens at the bound:
|
||||
|
||||
- **`usb-xhci-library.zig:1331`** — `else if (device.interface_count < max_interfaces)`
|
||||
has no `else`, so at the limit `current` is left pointing at interface 3 and the 5th
|
||||
interface's endpoints are appended to interface 3's array. A class driver bound to
|
||||
interface 3 can be handed an endpoint belonging to another interface. The
|
||||
alternate-setting arm one line above does `current = null` correctly.
|
||||
- **`usb-xhci-library.zig:899/1102`** — `allocateDevice()` failing returns after
|
||||
`enableSlot()` already succeeded, with no Disable Slot. Every failed attempt
|
||||
permanently leaks a controller slot.
|
||||
- **`acpi.zig:445`** — the AML loop reserves `maximum_resources - 2` slots and the
|
||||
function then adds **four** more resources. At 5 AML blocks the FADT is lost; at 6+
|
||||
the broad IRQ window is lost too, so every legacy-IRQ device (PS/2 keyboard at IRQ 1)
|
||||
fails containment. Real firmware ships 5–15 SSDTs; QEMU ships 2.
|
||||
- **`fat.zig:252`** — a `u64` protocol offset reaches a `u32` engine parameter through a
|
||||
bare `@intCast`. In ReleaseSafe — this project's build mode — that panics.
|
||||
- **`boot/efi.zig:318`** — the bootstrap page tables map [0, 4 GiB) and nothing checks
|
||||
that the handoff buffers UEFI allocated (BootInformation, the memory-map pool, the
|
||||
initial ramdisk) landed below it. Several firmware implementations allocate top-down.
|
||||
Above 4 GiB it is a triple fault with no output at all.
|
||||
|
||||
## Headroom actually measured
|
||||
|
||||
Every other bound here is hypothetical. One is a schedule:
|
||||
`system/configuration/protocol.csv`, the manifest deciding which binary may bind which
|
||||
protocol name, is at **50 of 64 grant rows and 12,668 of 16,384 bytes** — 78% and 77%.
|
||||
|
||||
To its credit, both overflow paths log (`init.zig:147` and `:238`). It will fail
|
||||
loudly. It will still fail.
|
||||
|
||||
## The rule this suggests
|
||||
|
||||
The audit's own exemplar turned out to be broken, which is worth recording. Two sweeps
|
||||
held up `devices_broker.dropped` as the model the others should be rewritten against.
|
||||
It is not: `register()` never increments it, and `kernel.zig:203` reads it once at boot
|
||||
*before* `seedDisplay` and long before any ring-3 driver exists. It can only report
|
||||
firmware-discovery losses — precisely the opposite of the runtime case under audit.
|
||||
|
||||
`parameters.maximum_cpus` is the one bound that meets the standard: hardware-determined
|
||||
and bounded, but it states what happens to the surplus (parked) and how you find out
|
||||
(`platform.cpusDropped()` → a WARNING at `kernel.zig:281`). Verified.
|
||||
|
||||
So:
|
||||
|
||||
1. **A fixed bound on a quantity the hardware or an external file decides is a defect,
|
||||
not a tunable.** It does not need a better number; it needs to not be fixed. This
|
||||
covers 139 of the 235.
|
||||
2. **A bound that must exist states three things**: what it protects against, what the
|
||||
system does when it is reached, and how an operator finds out. One of 235 does.
|
||||
3. **Refusal must be attributable.** `process.zig:944` collapses five distinct
|
||||
`RegisterError` variants into a bare `-1`; `pci-bus` can only log "register refused"
|
||||
with no reason. Half this audit's difficulty was that the machine could not say
|
||||
which ceiling it hit.
|
||||
4. **Two constants that must agree may not agree by comment.** `maximum_domains = 64`
|
||||
and `maximum_devices = 64` are coupled by prose, and the coupling fails open.
|
||||
|
||||
## Appendix: the inventory
|
||||
|
||||
Sorted by severity, then file. `Decided by` is the audit's judgment of who chooses the
|
||||
quantity — the machine, an external file or peer, or us.
|
||||
| File:line | Bound | Value | Decided by | At the limit | Observable |
|
||||
|---|---|---|---|---|---|
|
||||
| `boot/efi.zig:318` | <inline literal> (4 * gib) | 4 GiB | hardware | No check exists. buildBootstrapTables maps [0, 4 GiB) as identity + physmap 2 MiB leaves and stops; the only carve-out is the framebuffer window at ef… | Nothing on this path. No log, no error, no panic. The last thing printed is the unconditional con_out line at … |
|
||||
| `library/device/model/device-abi.zig:81` | maximum_device_resources | 8 | hardware | Silent drop, but at a different site than claimed. The live enforcement is system/kernel/device-model.zig:114 `if (self.resource_count >= maximum_reso… | None on any drop path, and the code says otherwise: device-model.zig:111-112 claims "Silently drops beyond `ma… |
|
||||
| `system/drivers/pci-bus/pci-bus.zig:228` | device.register refusal path (kernel maximum_chi… | 16 children per parent / 64 … | hardware | system/kernel/devices-broker.zig:305 `if (childCount(parent_id) >= maximum_children_per_parent) return error.TooManyChildren;` and :306 `if (count >= … | One line per lost function: `std.log.info("register refused for {d}:{d}.{d}", .{ bus, dev, function })` (pci-b… |
|
||||
| `system/drivers/usb-xhci-bus/usb-xhci-library.zig:227` | max_interfaces | 4 | external-data | CORRECTION to this claim's description: it is not only a silent drop. `parseConfiguration` (1326-1342) has no else on `else if (device.interface_count… | None. usb-xhci-bus.zig:368/425 log `{d} interface(s)` with the truncated count; nothing says any were dropped … |
|
||||
| `system/drivers/usb-xhci-bus/usb-xhci-library.zig:333` | max_devices | 8 | hardware | Two different behaviours, exactly as claimed. Root port, `setupDevice` (899-902): `const device = self.allocateDevice() orelse { std.log.info("port {d… | Root ports: one log line. Hub-attached: nothing. I confirmed the leak is real — the only Disable Slot in the f… |
|
||||
| `system/drivers/usb-xhci-bus/usb-xhci-library.zig:1304` | <inline literal> blob: [512]u8 | 512 | external-data | Silent truncation of device-supplied data. Lines 1304-1307: `var blob: [512]u8 = undefined; const length = @min(configuration.total_length, blob.len);… | None. `configuration.total_length` is read at 1299 and never compared to `blob.len`, never logged. The `{d} in… |
|
||||
| `system/drivers/usb-xhci-bus/usb-xhci-library.zig:1305` | blob (inline [512]u8, declared line 1304) | 512 bytes | external-data | Silent truncation of externally-supplied data. The @min clamps the GET_DESCRIPTOR request to 512; parseConfiguration then walks it and stops dead at l… | Nothing. No log line, no counter, no comparison of configuration.total_length against blob.len. The symptom is… |
|
||||
| `system/drivers/usb-xhci-bus/usb-xhci-library.zig:1331` | max_interfaces (= 4), used as `else if (device.i… | 4 | external-data | CORRUPTING, not merely dropping. When interface_count == 4, neither branch of the if/else-if runs, so `current` is NOT cleared — it still points at in… | Nothing. No log, no counter, no error. The device enumerates, `port N device: ... 4 interface(s)` is logged (a… |
|
||||
| `system/kernel/acpi.zig:445` | <inline literal> device_model.maximum_resources … | 6 AML blocks, inside an 8-re… | hardware | Silent double loss. The loop stops at 6 blocks, so SSDTs 7+ are never published. Then acpi.zig:454-463 adds the io_port grant (7th), the SCI irq (8th)… | Nothing at all. No counter, no log, no error. The ring-3 acpi service simply finds an acpi-tables node without… |
|
||||
| `system/kernel/acpi.zig:445` | device_model.maximum_resources - 2 (inline expre… | 6 AML blocks, out of 8 total… | hardware | Silent drop, then a silent cascade. Resources are appended in order: up to 6 AML memory resources (445-448), io_port (454), the SCI irq (459), the bro… | Nothing at the kernel. The only downstream trace is the acpi service writing "acpi: no FADT on the node — powe… |
|
||||
| `system/kernel/acpi.zig:627` | below | [64]Range | hardware | Silent skip: `if (region.base >= (1 << 32) or below_count == below.len) continue;`. Every region past the 64th is treated as 'not described', i.e. as … | None directly, though kernel.zig:133 does print `" regions : {d} - entries in the firmware memory map"`, w… |
|
||||
| `system/kernel/acpi.zig:627` | below (inline [64]Range) | 64 entries | hardware | Silent drop, and — worse than a drop — a corrupted result. A dropped region is not merely missing from the list; it disappears from the "described spa… | Nothing at all. No counter, no log line. The kernel prints the device tree including the bogus aperture, with … |
|
||||
| `system/kernel/acpi.zig:651` | gaps | [3]Range | hardware | Silent replacement of the smallest kept gap (acpi.zig:658-664): a fourth (or fifth) MMIO hole is simply forgotten. A PCI function whose BAR lands in a… | Nothing. No log, no counter, and the resulting failure is indistinguishable at the caller from a table-full re… |
|
||||
| `system/kernel/device-model.zig:66` | maximum_resources | 8 | external-data | `addResource` (line 113-118) returns false without recording. I grepped every call site: system/kernel/acpi.zig lines 448, 454, 459, 460, 463, 552, 55… | Nothing at the drop. Downstream there are two ring-3 lines that name the symptom and misattribute the cause: s… |
|
||||
| `system/kernel/devices-broker.zig:33` | maximum_children_per_parent | 16 | hardware | Confirmed at devices-broker.zig:305: `if (childCount(parent_id) >= maximum_children_per_parent) return error.TooManyChildren;`, where `childCount` (28… | One reasonless line per lost device, and nothing from the kernel. Confirmed at pci-bus.zig:228-231: `const reg… |
|
||||
| `system/services/acpi/acpi.zig:129` | blocks | [8][]const u8 | external-data | The `blocks: [8][]const u8` array at line 129 and its `if (block_count == blocks.len) continue;` at line 152 are dead — they can never fire, because t… | Only an indirect count that cannot reveal the loss: `std.log.info("parsed {d} AML blob(s), {d} namespace devic… |
|
||||
| `system/services/device-manager/device-manager.zig:174` | maximum_children | 64 | hardware | Silent drop with a misleading log. `addChild` (180-194) returns false when no free slot exists. `onChildAdded` (442-452): `if (!addChild(...)) status … | Effectively none, and actively misleading. The -ENOSPC goes back to the bus driver, which discards it — verifi… |
|
||||
| `boot/efi.zig:108` | <inline literal> (handles[0]) | 1 | hardware | Silent selection of GOP handle 0, and the two halves genuinely disagree — nativeResolution (efi.zig:167-176) iterates `for (handles) \|h\|` over every… | Nothing. No log names the handle count, the chosen adapter, or the resolved mode; the only con_out writes in t… |
|
||||
| `library/device/driver/driver.zig:149` | <inline literal> | @min(total, buffer.len) | hardware | Silently searches a prefix: `const total = enumerate(buffer); const n = @min(total, buffer.len); for (@as([]DeviceDescriptor, buffer[0..n])) \|d\| { …… | Nothing. The discarded `total` is the only evidence that the buffer was too small, and it is thrown away on th… |
|
||||
| `library/kernel/file-system.zig:227` | Entry.name_buffer | [64]u8 | external-data | Silent truncation on both readdir paths, exactly as claimed. Backend path, Directory.next line 264-266: `const nlen = @min(@min(@as(usize, header.name… | Nothing. No log, no flag, no short-count anywhere in library/kernel/file-system.zig. Entry.name() returns a pl… |
|
||||
| `system/drivers/virtio-gpu/virtio-gpu.zig:49` | max_width / max_height (line 50) | 800 x 600 | hardware | Two distinct behaviours, and the claim conflates their visibility. (a) onSetMode, line 533-540: `if (w == 0 or h == 0 or w > max_width or h > max_heig… | Partial and misleading, as claimed. Confirmed the two adjacent lines: `std.log.info("EDID preferred mode {d}x{… |
|
||||
| `system/kernel/architecture/x86_64/cpu.zig:489` | irq_vector_count | 14 | hardware | irq.zig:138 allocVector() scans base..base+count for a vector free of both an MSI binding and a GSI binding and returns null when none is; bind (irq.z… | No kernel log line. Driver-side, both messages verified: usb-xhci-bus.zig:279 `std.log.info("msi_bind unavaila… |
|
||||
| `system/kernel/architecture/x86_64/idt.zig:17` | gate_count | 48 | our-design | Nothing is enforced at this line. `pub fn init()` at line 136 does `inline for (0..gate_count) \|vector\| { const stub = @extern(*const anyopaque, .{ … | None at runtime. The bound only ever becomes visible as the vector-exhaustion path in irq.zig, which each driv… |
|
||||
| `system/kernel/architecture/x86_64/ioapic.zig:21` | base (module-level singleton) | 1 I/O APIC | hardware | Confirmed exactly. system/kernel/acpi.zig:547-556 adds EVERY MADT I/O APIC record to the device tree as ioapic0, ioapic1, ...; kernel.zig:545-548 then… | None. No log counts the discarded units, unlike the neighbouring drop counters that ARE logged (devices_broker… |
|
||||
| `system/kernel/architecture/x86_64/iommu-amd.zig:56` | ring_entries (event log) | 256 | hardware | Confirmed. faultDrain (lines 180-198) reads EventHead/EventTail, walks the ring wrapping at `if (head >= ring_entries * 16) head = 0;`, and writes the… | None. I grepped the file: `const reg_status = 0x2020;` appears at line 32 and at NO other line — it is declare… |
|
||||
| `system/kernel/architecture/x86_64/iommu-intel.zig:102` | <inline literal> 16 * 1024 | 16384 (bytes of VT-d registe… | hardware | No check. `faultDrain` computes `frcd_base = fro*16` with fro up to 0x3FF (16368 bytes) and `nfr = ((cap >> 40) & 0xFF) + 1` up to 256 registers of 16… | A kernel-mode #PF: the fault handler reports the vector/CR2 and halts the core. Loud, but reported as a page f… |
|
||||
| `system/kernel/architecture/x86_64/iommu.zig:17` | Discovery.register_base (single unit) | 1 IOMMU unit | hardware | Confirmed. `pub const Discovery = struct { register_base: u64, amd: bool };` at iommu.zig:16-19 carries exactly one base. Intel (acpi.zig:826-852): th… | Asymmetric, exactly as claimed. Intel warns: system/kernel/iommu.zig:395-396 `if (info.iommu_extra_units > 0) … |
|
||||
| `system/kernel/devices-broker.zig:25` | maximum_devices | 64 | hardware | Three inconsistent behaviours, all confirmed. Boot discovery: devices-broker.zig:111-114 `if (count >= maximum_devices) { dropped += 1; return device_… | Partial and aimed at the wrong path, exactly as claimed. kernel.zig:203-206 prints `"/system/kernel: WARNING {… |
|
||||
| `system/kernel/iommu.zig:43` | maximum_domains | 64 | hardware | Two behaviours. Domain exhaustion is safe: `domainCreate` returns null → `confineDevice` false → process.zig:418-420 rolls the claim back and returns … | The rollback path is visible as a failed `device_claim`. The fail-open path is completely silent — no log, no … |
|
||||
| `system/kernel/irq.zig:43` | maximum_gsi | 24 | hardware | Refusal at three places, all becoming a bare -1. irq.zig:158 `if (gsi >= maximum_gsi or !architecture.irqOwnsGsi(gsi)) return error.BadGsi;` -> proces… | Nothing in the kernel, but every caller logs: ps2-bus.zig:246 and :265, acpi.zig:315 "acpi: SCI irq_bind faile… |
|
||||
| `system/kernel/irq.zig:139` | allocVector over architecture.irq_vector_count | 14 (vectors 33..46, from arc… | hardware | `fn allocVector() ?u8` (line 139) scans v in [irq_vector_base, irq_vector_base + irq_vector_count) = [33, 47) — 14 vectors — and returns null when all… | Better than claimed at the driver level, absent at the kernel level. The kernel logs nothing and has no counte… |
|
||||
| `system/kernel/process.zig:546` | maximum_dma_regions | 256 | hardware | Confirmed, and the safety-invariant break is real. process.zig:549-556 `fn dmaRegistryAdd(object: ?*ipc.DmaRegionObject, physical: u64, len: u64, owne… | None whatsoever — confirmed. `dma_alloc` returns success, no log, no counter, no failed syscall. The symptom i… |
|
||||
| `system/kernel/scheduler.zig:546` | maximum_tasks (via freeSlot in secondaryMain) | 48 (parameters.maximum_tasks… | hardware | Confirmed: `const t = freeSlot() orelse @panic("sched: task table full (AP idle task)");` at scheduler.zig:546, reached from `secondaryMain` with the … | A named panic on console/serial — loud, but the machine is dead and the message points at the task table rathe… |
|
||||
| `system/kernel/vfs.zig:58` | maximum_mounts | 8 | our-design | FALSE SUCCESS plus a reference leak, confirmed end to end. installMount, vfs.zig:182: `const m = slot orelse return;` — it gives up silently when no s… | Worse than nothing: the syscall reports success, so the mounting service logs its own success — system/service… |
|
||||
| `system/kernel/vfs.zig:86` | maximum_directories | 8 | external-data | Silent skip inside the boot walk: `if (directoryIndex(parent) == null and directory_count < maximum_directories)` (vfs.zig:141). An unregistered direc… | Nothing. No log, no counter, no `dropped` variable of the kind acpi.zig:542 and the device manager both mainta… |
|
||||
| `system/kernel/vfs.zig:141` | maximum_directories (declared line 86) | 8 | external-data | Silent drop with a mount-level knock-on, confirmed. The ninth distinct ancestor is skipped by the guard at vfs.zig:141; it then fails to resolve (reso… | Nothing — no counter, no log line, no `dropped` variable of the sort acpi.zig:542 (cpu_information.dropped, pr… |
|
||||
| `system/kernel/vfs.zig:182` | maximum_mounts (declared line 58) | 8 | our-design | False success, confirmed: installMount gives up at vfs.zig:182 (`const m = slot orelse return;`) and mountBackend returns true regardless at vfs.zig:3… | Nothing in the kernel. The mounting service logs its own success — fat.zig:173/:178/:183 all `std.log.info("mo… |
|
||||
| `system/services/acpi/acpi.zig:542` | buffer (readHid) / Registered.hid / Notice.hid | [8]u8 | external-data | Both failures confirmed. (a) readHid (lines 540-566) accepts only an integer _HID it can EISA-decode: a method result must be `.integer`, and a static… | None for (a): a skipped device produces no line at all. For (b) the log at line 408, `std.log.info("power: not… |
|
||||
| `system/services/acpi/acpi.zig:648` | descriptor.resources.len | 8 (maximum_device_resources,… | external-data | Silent drop of the surplus. `fn addResource(descriptor: *device.DeviceDescriptor, kind: device.ResourceKind, start: u64, len: u64) void { if (descript… | Nothing at the drop. The only echo is the per-device summary at line 241, `std.log.info("device {d} bus=acpi … |
|
||||
| `system/services/device-manager/device-manager.zig:149` | maximum_drivers | 16 | hardware | `addDriver` (240-254) walks for a free slot and, finding none, falls through the loop to `std.log.info("driver table full; cannot supervise {s}", .{na… | One log line only (quoted above). No counter, no status to any caller, and it lands amid the stream of `child … |
|
||||
| `system/services/device-manager/device-manager.zig:358` | <inline literal> | 64 | hardware | Silent truncation. Line 362-363: `const total = device.enumerate(buffer); const count = @min(total, buffer.len);` — the difference is computed and thr… | Nothing. No log mentions `total`. The only related line, `"/system/services/device-manager: no matchable devic… |
|
||||
| `system/services/fat/fat.zig:73` | open_nodes | 32 | our-design | An error is returned, but a maximally confusing one: `const index = allocOpen() orelse return refused;` (line 242), where `refused` is `-envelope.ENOE… | Nothing on the server side — no log at all in allocOpen or onOpen. The client sees ENOENT and will report "fil… |
|
||||
| `system/services/fat/fat.zig:252` | <inline literal> | u32 (via @intCast of a u64 p… | external-data | Panic. `vfs_protocol.Read.offset` is a `u64` (library/protocol/vfs/vfs-protocol.zig:82-86) and `engine.readFile` takes `offset: u32`; the bridge is a … | A process fault and whatever the supervisor logs about the death; nothing identifies the offending request or … |
|
||||
| `boot/efi.zig:371` | maximum_bundled | 64 | external-data | Two different behaviours for one constant, confirmed. loadByManifest (efi.zig:507): `if (count.* == maximum_bundled) return;` — silent success with th… | Manifest path: nothing. Walk path: the unconditional con_out line at efi.zig:71-75, `log("EFI: no /system bina… |
|
||||
| `boot/efi.zig:375` | maximum_tree_depth | 3 | our-design | `if (depth == maximum_tree_depth) continue;` (efi.zig:580) — the subdirectory is silently skipped and nothing beneath it is bundled. No error, no coun… | Nothing. The boot proceeds; the missing binary surfaces much later as an init spawn failure or a device-manage… |
|
||||
| `boot/efi.zig:507` | maximum_bundled (declared line 371) | 64 | external-data | As described and verified: efi.zig:507 `if (count.* == maximum_bundled) return;` in loadByManifest is a silent stop keeping the first 64; efi.zig:590 … | Manifest path: nothing. Enumeration path: efi.zig:71-75 prints 'EFI: no /system binaries (TooManyBinaries) - b… |
|
||||
| `boot/efi.zig:582` | <inline literal> (child_prefix bufPrint) | 64 bytes (initial_ramdisk.ma… | external-data | `const child = try std.fmt.bufPrint(&child_prefix, "{s}/{s}", .{ prefix, name });` into a [initial_ramdisk.maximum_name]u8 = [64]u8. error.NoSpaceLeft… | efi.zig:71-75 prints 'EFI: no /system binaries (NoSpaceLeft) - booting without user space' on con_out. Visible… |
|
||||
| `boot/efi.zig:633` | ehdr.e_phnum / e_phoff / e_phentsize (no bound a… | unbounded — read straight fr… | external-data | Unchecked out-of-bounds read. efi.zig:632-636 forms `image.ptr + ehdr.e_phoff + i * ehdr.e_phentsize` as raw pointer arithmetic and @ptrCasts it — no … | Nothing until it faults or panics, at which point the machine is still in the firmware with no kernel. |
|
||||
| `library/device/acpi/aml/interpreter.zig:158` | notify_queue | [16]NotifyEvent | external-data | Silent drop of the 17th and later Notify. `notify()` (line 571-576): `if (target) \|node\| { if (self.notify_count < self.notify_queue.len) { ...appen… | None. No log in interpreter.zig at all (grepped: zero std.log/log.print/logging.write calls in the file). `tak… |
|
||||
| `library/device/acpi/aml/interpreter.zig:523` | <inline literal> | 100_000 | our-design | Silently exits the loop and continues executing the method as if the loop had terminated normally: `while (guard < 100_000) : (guard += 1) { … }` then… | Nothing. No log, no `error.Unsupported`, no distinguishable result — the caller receives a normal-looking valu… |
|
||||
| `library/device/acpi/aml/parser.zig:24` | maximum_segments (and its disagreeing twin, libr… | 64 in the parser, 16 in the … | external-data | Both files silently drop the surplus segment after consuming its 4 bytes. Parser: appendSegment, parser.zig:137-143. Interpreter: `fn segment(self: *C… | None on either path. Neither file logs; acpi.zig discards ParseResult.consumed/total; interpreter errors surfa… |
|
||||
| `library/device/usb/usb.zig:167` | <inline literal> | 100 attempts × 20 ms = 2 s | our-design | Gives up and returns null: `const bus = while (attempts < 100) : (attempts += 1) { if (channel.openEndpoint("usb-transfer")) \|handle\| break handle; … | Nothing here. The driver's own bail-out is what an operator sees, with no indication that it was a timeout rat… |
|
||||
| `library/kernel/channel.zig:124` | <inline literal> | @min(available, into.len) | our-design | Silent truncation to the caller's buffer: `const available = @min(answer.len - envelope.prefix_size, @as(usize, status.len)); const taken = @min(avail… | Weak: `Response.status.len` holds the true length the provider sent, so a caller *could* compare it against `p… |
|
||||
| `library/kernel/file-system.zig:85` | <inline literal> | [224]u8 (twice: lines 77 and… | our-design | Errors invisibly, but the enforcing code is in the KERNEL, not where the claim points. system/kernel/process.zig:1898 rejects the input outright: `if … | None. A too-long path and a nonexistent file are both `null` from open(); no log on either side — the kernel's… |
|
||||
| `library/kernel/logging.zig:82` | <inline literal> | [256]u8 | our-design | Truncation, marked with a `~` — but the truncated slice is the *whole* buffer, not the written prefix: `const line = std.fmt.bufPrint(&buffer, prefix … | The trailing `~` marks the line as truncated, which is good; the garbage tail is not marked at all. |
|
||||
| `library/kernel/process.zig:211` | <inline literal> | [32]ProcessDescriptor | our-design | Silent false negative: `var table: [32]ProcessDescriptor = undefined; const total = processes(&table); for (table[0..@min(total, table.len)]) \|descri… | Nothing. The function returns a plain bool; the discarded `total` is the only evidence and it is thrown away b… |
|
||||
| `library/protocol/device-manager/device-manager-protocol.zig:161` | entries_per_reply | (envelope.packet_maximum - e… | hardware | Silent cut with a success status. `onEnumerate` (device-manager.zig:524-536): `for (&children) \|*child\| { if (!child.used) continue; if (written + e… | None, and structurally impossible for a client to detect: the protocol's own comment (149-152) says the count … |
|
||||
| `library/protocol/envelope/envelope.zig:133` | packet_maximum | 256 | our-design | Two behaviours, exactly as claimed. Compile time: `Define` rejects an oversized fixed part with a message naming protocol, verb, size, prefix and floo… | Compile-time half is excellent. Run-time half is silent everywhere I traced it: library/device/block/block.zig… |
|
||||
| `library/protocol/envelope/envelope.zig:134` | post_maximum | 64 | our-design | Compile error for an event payload that does not fit (`Define`, envelope.zig:365-370, message names the event and the push floor). At run time `encode… | Compile-time: named and precise. Run-time: nothing. The `orelse return` at service.zig:239 is the whole handli… |
|
||||
| `library/protocol/usb-transfer/usb-transfer-protocol.zig:58` | max_report_data | 40 | hardware | Two stacked silent truncations. First in the engine: usb-xhci-library.zig:1569 `const n = @min(length, report.data.len);` against a `data: [64]u8` (li… | None. No log at either truncation site; the surrounding comment (usb-xhci-bus.zig:766-768) states the truncati… |
|
||||
| `library/protocol/usb-transfer/usb-transfer-protocol.zig:62` | max_reported_endpoints | 4 | hardware | Silent drop, and the primary site is upstream of the one claimed. The bus driver discards surplus endpoint descriptors while parsing the configuration… | None. No log at the parse-time drop, none at the client cap; a driver's `findEndpoint` just returns null for a… |
|
||||
| `system/boot-handoff.zig:143` | kernel_segments | 8 | our-design | Silent drop with a delayed fatal consequence, as described. boot/efi.zig:659-668 records only `if (n < boot_information.kernel_segments.len)` with no … | The loader says nothing. The kernel prints the count it received — kernel.zig:159 `log.print(" kernel segs: {… |
|
||||
| `system/drivers/pci-bus/pci-bus.zig:81` | <inline literal> alloc(device.DeviceDescriptor, … | 64 | our-design | Not reachable today, so nothing happens. If the kernel table grew past the buffer: `device.enumerate(buffer)` returns the TOTAL, the search clamps wit… | One misleading line, `std.log.info("device {d} not in the device tree", .{bridge_id})` — it names a missing de… |
|
||||
| `system/drivers/pci-bus/pci-bus.zig:81` | 64 (bare inline literal — no constant, no link t… | 64 | our-design | Same site as the earlier claim on line 81 — this is a duplicate. Unreachable today (`total` is bounded by the kernel's own maximum_devices = 64). If i… | `std.log.info("device {d} not in the device tree", .{bridge_id})` — misdescribes the cause and never prints `t… |
|
||||
| `system/drivers/usb-hid/hid-report.zig:20` | KeyboardReport.keys: [6]u8, and max_transitions … | 6 concurrent keys; 14 transi… | hardware | THE CLAIM IS BACKWARDS. `max_transitions = 8 + 6` is derived from a wrong worst case, and `Transitions.add`'s guard (`if (self.count < self.items.len)… | None. `add` drops without logging and `Transitions` carries no overflow flag; `slice()` just returns the first… |
|
||||
| `system/drivers/usb-storage/usb-storage.zig:61` | <inline literal> .lun = 0 in the CommandBlockWra… | 1 LUN (LUN 0 only) | hardware | There is no limit check — LUNs 1..15 simply do not exist to this driver. Every CBW is built with `.lun = 0` (usb-storage.zig:57-63), and I confirmed t… | Nothing. Nothing reads bMaxLUN, so nothing can report that a device has more than one logical unit. |
|
||||
| `system/drivers/usb-storage/usb-storage.zig:121` | <inline literal> while (tries < 10) with time.sl… | 10 attempts = ~500 ms | hardware | Falls out of the loop and proceeds regardless (usb-storage.zig:120-127). INQUIRY's result is discarded (`_ = transact(...)`, line 129), then READ CAPA… | Present but misattributed, exactly as claimed: `_ = logging.write("/system/drivers/usb-storage: READ CAPACITY … |
|
||||
| `system/drivers/usb-storage/usb-storage.zig:172` | <inline literal> @intCast(request.lba) -> u32, @… | LBA < 2^32, count < 2^16 (RE… | external-data | Bare, unchecked @intCast on both fields — verified there is no range check anywhere between the wire struct and the CDB. scsi.read10/write10 take (lba… | In a safe build, driver death with a fault exit reason and a device-manager restart with backoff — attributed … |
|
||||
| `system/drivers/usb-storage/usb-storage.zig:172` | @intCast(request.lba) / @intCast(request.count) … | u64 -> u32 LBA, u32 -> u16 b… | external-data | Unchecked narrowing of two peer-chosen wire fields. library/protocol/block/block-protocol.zig:30-35 declares `Transfer { lba: u64, count: u32, physica… | Nothing in ReleaseFast; in a safe build a driver crash with a fault exit reason and a supervised restart, attr… |
|
||||
| `system/drivers/usb-xhci-bus/usb-xhci-bus.zig:80` | opens (= [_]Open{.{}} ** 16) | 16 | our-design | Confirmed, including the success-on-failure shape. recordOpen (line 87) walks `opens` for a matching token, then for a free slot, and falls through to… | Nothing on the bus side — no log when the table fills, and the harness closes the unclaimed report endpoint si… |
|
||||
| `system/drivers/usb-xhci-bus/usb-xhci-bus.zig:728` | prev_connected (declared line 326 as [64]bool), … | 64 | hardware | Confirmed off-by-one. Line 728: `while (port <= engine.max_ports and port <= prev_connected.len) : (port += 1)` with `var port: u32 = 1` — at port == … | The OOB half surfaces as a driver fault and a supervised restart into the same panic — loud but attributed to … |
|
||||
| `system/drivers/usb-xhci-bus/usb-xhci-bus.zig:728` | prev_connected (inline [64]bool, declared line 3… | 64 | hardware | Same site as the earlier line-728 claim — this is a duplicate. Two failures, both confirmed. (1) Off-by-one out-of-bounds: `while (port <= engine.max_… | The panic half is loud (process fault + supervised restart loop) but misattributed. The silent-drop half has n… |
|
||||
| `system/drivers/usb-xhci-bus/usb-xhci-library.zig:228` | max_endpoints_per_interface (and max_configured_… | 4 per interface, 16 configur… | hardware | Parsing: `if (interface.endpoint_count < max_endpoints_per_interface)` with no else — the 5th endpoint descriptor is silently discarded, so a class dr… | The endpoint-drop during parsing is entirely silent. The ring-exhaustion path reaches the class driver as a ge… |
|
||||
| `system/drivers/usb-xhci-bus/usb-xhci-library.zig:228` | max_endpoints_per_interface | 4 | external-data | Silent drop, no else (1345-1354). An endpoint never recorded cannot be found by `endpointForAddress` (1377-1382), so a class driver's subscribe or bul… | None at the drop. The secondary consequence the claim names is real and I verified it: usb-xhci-bus.zig:612 se… |
|
||||
| `system/drivers/usb-xhci-bus/usb-xhci-library.zig:328` | <inline literal> Report.data: [64]u8 (and usb_tr… | 64 in the driver, then 40 ov… | hardware | Truncated twice, silently: `const n = @min(length, report.data.len);` in enqueueReport, then `const n = @min(report.length, usb_transfer_protocol.max_… | Nothing. Both truncations are @min with no branch and no log. |
|
||||
| `system/drivers/usb-xhci-bus/usb-xhci-library.zig:334` | max_subscriptions | 8 | hardware | Two paths, as claimed. Class driver: `subscribeInterrupt` (1482) `const subscription = self.allocateSubscription() orelse return false;` -> usb-xhci-b… | Class drivers log it: usb-hid/keyboard.zig:94 and mouse.zig:57 both write `interrupt subscribe failed`. The hu… |
|
||||
| `system/drivers/usb-xhci-bus/usb-xhci-library.zig:335` | report_queue_capacity | 16 | hardware | `enqueueReport` (1561-1562): `if (self.report_count >= self.report_queue.len) return; // full: drop the newest` — the report is discarded and the subs… | None. The comment in the source is the only trace; no log, no counter. The symptom is dropped keystrokes or a … |
|
||||
| `system/drivers/usb-xhci-bus/usb-xhci-library.zig:968` | hub_change_mask: u32 (declared line 298) and the… | 31 usable downstream ports | hardware | Ports 32..255 are never seeded and never serviced: the seed clamps to 0xFFFF_FFFE for hub_ports >= 31, and takeHubChange picks bits via @ctz on a u32,… | Nothing. No log line mentions bNbrPorts exceeding what the mask holds; the 'N downstream ports powered' line p… |
|
||||
| `system/drivers/usb-xhci-bus/usb-xhci-library.zig:1456` | <inline literal> `.status = length` on the bulk … | 65535 (the xHCI TRB Transfer… | hardware | Unchecked write of a u32 into a 17-bit field. `bulkTransfer` (1452-1458) pushes `.status = length` with no mask and no split across TRBs; the status d… | Nothing on the way in — no check, no log. It surfaces later as a short/failed transfer or a lost completion at… |
|
||||
| `system/drivers/virtio-gpu/virtio-gpu.zig:49` | max_width / max_height | 800 x 600 | hardware | set_mode errors visibly (line 536 refuses anything larger). But the EDID path is the silent one: readEdid (lines 413-442) parses the panel's preferred… | One std.log.info line naming the preferred mode, in the driver's log file, with nothing saying it was ignored. |
|
||||
| `system/drivers/virtio-gpu/virtio-gpu.zig:208` | <inline literal> descriptors: [64]device.DeviceD… | 64 | hardware | NOT reachable today, contrary to the claim's framing. The kernel's own table is the binding wall: system/kernel/devices-broker.zig:25 `const maximum_d… | `device {d} not in the device tree` (line 213), which today only means a real absence. The kernel-side drop th… |
|
||||
| `system/initial-ramdisk.zig:26` | maximum_name | 64 bytes | our-design | Four different behaviours, verified individually. (a) Capsule build: tools/pack-system-image.py:35-37 writes 'pack-system-image: path too long (>63)' … | (a) clear build error. (b)(c)(d) nothing: boot/efi.zig has exactly two output helpers (lines 773 and 795) and … |
|
||||
| `system/kernel/acpi.zig:94` | overrides | [16]IsoEntry | hardware | Silent drop: `if (platform_information.override_count < platform_information.overrides.len) { ... }` with no else branch and no counter — unlike the C… | Nothing. A lost override means a legacy line is routed at the wrong GSI or the wrong polarity, which shows up … |
|
||||
| `system/kernel/acpi.zig:94` | PlatformInformation.overrides | [16]IsoEntry | hardware | Silent drop, at three separate 16s that must agree by hand: system/kernel/acpi.zig:558 (no else branch), system/kernel/kernel.zig:248 (`var isos: [16]… | Nothing at any of the three clamps. No counter of the cpusDropped kind, no log line. |
|
||||
| `system/kernel/acpi.zig:186` | aml_block_physical / aml_block_len | [32]u64 / [32]usize | hardware | Confirmed at acpi.zig:190-197: `fn addAmlBlock(sdt_physical: u64) void { if (aml_block_count >= aml_block_physical.len or sdt_physical == 0) return; .… | None — confirmed, no log and no dropped counter, unlike `cpu_information.dropped` (acpi.zig:539) and `rmrr_ski… |
|
||||
| `system/kernel/architecture/x86_64/apic.zig:320` | <inline literal> 100_000_000 (calibratePit guard… | 100000000 | our-design | Confirmed: the loop falls out and calibration proceeds on a window that never happened. `ticks_per_ms = elapsed / calib_ms` and `tsc_hz = (tsc_end -% … | No failure line, but not quite "none": kernel.zig:322 prints `timer online ({d} Hz tick; timer clock {d} MHz, … |
|
||||
| `system/kernel/architecture/x86_64/apic.zig:629` | localId (return type u8) / apic_id << 24 | 8-bit APIC id (0..255) | hardware | The claimed truncations are not reachable; the real defect is an omission. `sendInit`/`sendStartup` (lines 174-184) take a `u32` but every caller pass… | None for the dropped type-9 records. `cpu_information.dropped` (acpi.zig:542) counts only overflow past the 12… |
|
||||
| `system/kernel/architecture/x86_64/iommu-amd.zig:56` | ring_entries (command buffer) | 256 | our-design | Confirmed. submitCommand (lines 221-228) writes the two qwords, advances `command_tail += 16`, wraps modulo the ring, and writes the tail register — w… | None — no head check, no full-ring detection, no log. |
|
||||
| `system/kernel/architecture/x86_64/iommu-amd.zig:74` | fault_log_budget | 32 | hardware | logFault (lines 200-211) opens with `if (fault_log_budget == 0) return;` — the record's bdf and address are discarded outright, with no suppressed-fau… | One "(further faults suppressed)" line (line 210) and then permanent silence for the detail. Correcting the cl… |
|
||||
| `system/kernel/architecture/x86_64/iommu-amd.zig:246` | <inline literal> 100_000 | 100000 | our-design | Confirmed verbatim at lines 243-253: the spin loop polls the completion frame for the 0xC0FFEE sentinel, and at `if (spins > 100_000)` warns once behi… | One line for the whole boot: "/system/kernel: AMD-Vi COMPLETION_WAIT store not observed — proceeding (QEMU pro… |
|
||||
| `system/kernel/architecture/x86_64/iommu-intel.zig:77` | fault_log_budget | 32 | hardware | `logFault` (lines 252-270) takes the else branch and does `faults_suppressed += 1`. The budget is only ever decremented (line 255) — nothing resets it… | One line at the transition: 'DANOS-IOMMU-FAULT: (further faults suppressed)' (line 266). After that a device D… |
|
||||
| `system/kernel/architecture/x86_64/iommu-intel.zig:294` | <inline literal> 10_000_000 (spin64) | 10000000 | our-design | `if (spins > 10_000_000) return;` — the function returns as if the invalidation completed. Callers (`invalidateDomain`, `globalInvalidate`, `detach`) … | Nothing at all — a bare `return` with no message, unlike `spinStatus` which prints "WARNING VT-d status bit ne… |
|
||||
| `system/kernel/architecture/x86_64/paging.zig:206` | <inline literal> 0xFEE00000 | 0xFEE00000 (one 4 KiB page) | hardware | Confirmed and worse than "inconsistent": the two halves disagree with no reconciliation. paging.zig:206 maps exactly one hardcoded page `mapPage(pml4,… | A boot-time kernel fault reported as a fault address, never as "LAPIC relocated". The parsed truth is thrown a… |
|
||||
| `system/kernel/architecture/x86_64/serial.zig:112` | <inline literal> 5_000 (writeByte guard) | 5000 | hardware | The loop simply falls through and `setRegister(0, c)` runs anyway (serial.zig:111-113) — the byte is written into a transmit-holding register that may… | Nothing. The corruption appears in the serial log itself, which is the channel you would use to notice it. |
|
||||
| `system/kernel/heap.zig:28` | heap_maximum | 64 * 1024 * 1024 | our-design | Confirmed mixed. `grow` returns false at heap.zig:63 `if (start + bytes > heap_base + heap_maximum) return false;` -> `rawAlloc` null -> std.mem.Alloc… | For the graceful callers, a bare -1 to user space with no reason. For scheduler.zig:704, a panic naming the st… |
|
||||
| `system/kernel/heap.zig:161` | <inline literal> alignment ceiling | 16 | our-design | `if (alignment.toByteUnits() > 16) return null;` confirmed at heap.zig:161 inside `allocImpl`. Returning null from the vtable alloc is how the allocat… | None. No log, no distinct error; the caller sees error.OutOfMemory and investigates memory pressure. |
|
||||
| `system/kernel/ipc-synchronous.zig:40` | MESSAGE_MAXIMUM | 256 | our-design | Two different behaviours, both confirmed. The cap itself errors visibly: `if (message_len > MESSAGE_MAXIMUM or reply_cap > MESSAGE_MAXIMUM) return -E2… | -E2BIG for the cap. Nothing for the truncation — no counter, no flag bit, no errno. |
|
||||
| `system/kernel/ipc-synchronous.zig:75` | post_capacity | 16 | external-data | Deliberate silent drop of the OLDEST message: sendLocked, ipc-synchronous.zig:568 `if (endpoint.post_tail -% endpoint.post_head >= post_capacity) endp… | None, and I checked the struct: PostSlot (ipc-synchronous.zig:79-83) carries only length and sender_id — no se… |
|
||||
| `system/kernel/ipc-synchronous.zig:513` | @min(caller.ipc_send_len, receive_cap) (no const… | whatever the receiver's buff… | external-data | Silent truncation. replyWait, ipc-synchronous.zig:512-513: `if (dequeueSender(endpoint)) \|caller\| { const n = @min(caller.ipc_send_len, receive_cap)… | Nothing — no counter, no errno, no flag bit anywhere on this path. |
|
||||
| `system/kernel/log.zig:202` | buffer (log.print) | [256]u8 | our-design | Confirmed: the entire line is discarded, not truncated. Lines 201-204 verbatim: `pub fn print(comptime fmt: []const u8, args: anytype) void { var buff… | Nothing. The message never appears and its absence is indistinguishable from the code path not having run. Not… |
|
||||
| `system/kernel/pmm.zig:94` | <inline literal> (implicit: bitmap must land bel… | 4 GiB, stated only in prose | hardware | Two distinct failures. If NO usable region is large enough: `@panic("pmm: no region large enough for the frame bitmap")` — loud and fatal. If the chos… | The panic is legible. The above-4-GiB case is a triple fault or an early #PF with essentially no diagnostics. |
|
||||
| `system/kernel/process.zig:98` | maximum_shared_memory_pages | 8192 (32 MiB) | hardware | Confirmed at process.zig:790: `if (pages == 0 or pages > maximum_shared_memory_pages) return fail(state);` in `systemSharedMemoryCreate`, giving a cle… | No kernel log; -1 only — confirmed. And the ambiguity the claim notes is real: `pmm.allocContiguous` failing t… |
|
||||
| `system/kernel/process.zig:104` | maximum_mmap_pages | 8192 (32 MiB) | our-design | Confirmed at process.zig:2050: `if (pages == 0 or pages > maximum_mmap_pages) return fail(state);` in `systemMmap` — a clean -1. The independent secon… | -1 only, no log — confirmed. Indistinguishable from arena exhaustion and from 'not a user process' (line 2049)… |
|
||||
| `system/kernel/process.zig:118` | maximum_resolve_path | 224 | external-data | Confirmed at process.zig:1898: `if (path_len == 0 or path_len > maximum_resolve_path or path_ptr >= user_half_end or path_ptr + path_len > user_half_e… | No log; -1 only — confirmed. There is no ENAMETOOLONG in this path (the function does have `failErr(state, ipc… |
|
||||
| `system/kernel/process.zig:1542` | exit_subscriber_capacity | 16 | hardware | Confirmed at process.zig:1573-1581: `subscribeExits` scans `for (&exit_subscribers) \|*slot\| { if (slot.* == null) { endpoint.refcount += 1; slot.* =… | Materially worse than claimed, and this is my main correction. The kernel does return a specific -ENOSPC, but … |
|
||||
| `system/kernel/process.zig:1649` | timer_capacity | 16 | our-design | Confirmed at process.zig:1684-1692: `systemTimerBind` scans `for (&one_shot_timers) \|*slot\| { if (slot.* == null) { ...; return architecture.setSyst… | Effectively none, correcting the claim. The kernel's -ENOSPC is narrowed to a bool by library/kernel/time.zig:… |
|
||||
| `system/kernel/process.zig:2172` | maximum_segments | 16 | external-data | Confirmed at process.zig:2200: `if (ehdr.e_phnum > maximum_segments) return error.BadElf;`. The claim's key observation is right — the test is on `e_p… | None — confirmed. `error.BadElf` is erased at process.zig:1018 and `system_spawn` returns -1, so a developer i… |
|
||||
| `system/kernel/process.zig:2173` | maximum_pages | 256 | external-data | Confirmed at process.zig:2238, inside the PT_LOAD loop: `total_pages += seg.pages(); if (total_pages > maximum_pages) return error.ProgramTooBig;`. Th… | No kernel log — confirmed. `error.ProgramTooBig` is erased at the syscall boundary and becomes the same -1 as … |
|
||||
| `system/kernel/scheduler.zig:206` | maximum_space_mappings | 16 | our-design | Clean refusal, confirmed. `recordSpaceMappingLocked` (scheduler.zig:343-355) walks `entry.mappings` for a null slot and `return false` when full; proc… | A bare -1 from `shared_memory_map`/`shared_memory_create` — the same value as a bad handle, an exhausted arena… |
|
||||
| `system/kernel/vfs.zig:143` | @memcpy(d.path[0..parent.len], parent) — implici… | 64 == 64, by coincidence | external-data | Safe today, and safe with slightly more margin than the claim states. parentOf (vfs.zig:119-123) returns a strict prefix ending before the last '/', s… | Nothing would warn. The coupling is invisible from either file: system/initial-ramdisk.zig:26-28 reasons about… |
|
||||
| `system/kernel/vfs.zig:315` | name_out (via process.zig:1972 name_buffer) | [64]u8 | external-data | Silent truncation with a header that agrees with the truncation rather than reporting it: vfs.zig:314-317 `const n = @min(name.len, name_out.len); @me… | None. The contrast is fair — system/kernel/log-ring.zig sets abi.klog_flag_truncated for exactly this situatio… |
|
||||
| `system/services/acpi/acpi.zig:81` | mmio_scratch | [4096]u8 | external-data | No overflow — the aliasing IS the failure. library/device/acpi/aml/interpreter.zig:708 (readRegionByte) and :720 (writeRegionByte) both do `const virt… | None. No log fires on a SystemMemory access in either the interpreter or the HAL, so an _STA or _CRS that cons… |
|
||||
| `system/services/acpi/acpi.zig:113` | <inline literal> | 64 | hardware | Falls through silently and proceeds as though ACPI mode were enabled. `while (tries < 1000 and (halPioRead(2, pm1a_cnt) & sci_en_bit) == 0) : (tries +… | None that distinguishes the two outcomes. No log fires on give-up, and the success message "acpi: power button… |
|
||||
| `system/services/acpi/acpi.zig:306` | <inline literal> | 1000 | hardware | Falls through silently and proceeds as though ACPI mode were enabled. `while (tries < 1000 and (halPioRead(2, pm1a_cnt) & sci_en_bit) == 0) : (tries +… | None that distinguishes the two outcomes. No log fires on give-up, and the success message "acpi: power button… |
|
||||
| `system/services/device-manager/device-manager.zig:50` | registry_source | [8192]u8 | external-data | Silent truncation of the file. `loadRegistry` (63-68): `while (used < registry_source.len) { const n = file.read(registry_source[used..]) orelse break… | Nothing says the file was cut. Only downstream symptoms: `{d} malformed line(s) skipped` (line 70) if the frag… |
|
||||
| `system/services/device-manager/device-manager.zig:51` | registry_source | 8192 bytes | external-data | Silent truncation of the file. `loadRegistry` (63-68): `while (used < registry_source.len) { const n = file.read(registry_source[used..]) orelse break… | Nothing says the file was cut. Only downstream symptoms: `{d} malformed line(s) skipped` (line 70) if the frag… |
|
||||
| `system/services/device-manager/device-manager.zig:531` | <inline literal> | reply tail of a 256-byte pac… | our-design | Silent truncation with a success status: `if (written + entry_size > tail.len) break;` and the handler returns the byte count. There is no cursor and … | None. The client sees a short but well-formed list. |
|
||||
| `system/services/display/backend.zig:27` | device_table | [64]device.DeviceDescriptor | hardware | Confirmed as written but NOT reachable today. findDisplay (lines 54-63) does `const total = device.enumerate(&device_table); const n = @min(total, dev… | Misleading when it does fire, as claimed: Gop.init exhausts its retries and writes `display: no framebuffer de… |
|
||||
| `system/services/display/compositor.zig:288` | <inline literal> | src.len >= w*h*4, against a … | our-design | Silent no-op reported as success. blitTile does `if (src.len < @as(usize, w) * h * 4) return;`, but blitLayer (display.zig:236-242) ignores that, stil… | None whatsoever: no log, and a success status. The symptom is a blank region. |
|
||||
| `system/services/display/display.zig:76` | maximum_layers | 16 | our-design | Confirmed. `createLayer` (display.zig:208-210) is `const slot = freeLayer() orelse return null;`, and `onCreateLayer` (695-700) is `const slot = creat… | Nothing server-side for a client-facing exhaustion — confirmed. The service's own cursor path does log (`_ = l… |
|
||||
| `system/services/display/display.zig:395` | mode_list / list | [4]backend_mod.Mode | hardware | Silent truncation on both hops: the backend clamps with `const count = @min(@min(offered.count, scanout_protocol.max_modes), out.len);` (backend.zig:1… | None — no log names the number offered versus the number kept. The mode-set self-check only says "no alternate… |
|
||||
| `system/services/fat/engine.zig:70` | sector_size | 512 | external-data | Mount is refused, confirmed. Both paths test `if (geometry.bytes_per_sector == sector_size)` — engine.zig:193 (bare FAT at LBA 0) and engine.zig:209 (… | A wrong diagnosis: `_ = logging.write("/system/services/fat: not a FAT filesystem\n")` at system/services/fat/… |
|
||||
| `system/services/fat/engine.zig:193` | sector_size (declared line 70) | 512 | external-data | Refuses to mount rather than corrupting, confirmed at engine.zig:193 and the MBR-partition path at engine.zig:209; both fall through to `return null`. | system/services/fat/fat.zig:162 `_ = logging.write("/system/services/fat: not a FAT filesystem\n")` — false fo… |
|
||||
| `system/services/fat/fat.zig:73` | open_nodes (inline [_]OpenNode{.{}} ** 32) | 32 | our-design | Error returned, but the WRONG error, and silently: `const index = allocOpen() orelse return refused;` where `const refused: isize = -envelope.ENOENT;`… | Nothing. No log line at exhaustion, and the client is told the file does not exist. An operator debugging this… |
|
||||
| `system/services/fat/fat.zig:285` | <inline literal> | reply tail of a 256-byte pac… | our-design | Silent truncation with a success status: `const name_len = @min(listing.name_len, into.len);` then `return @intCast(name_len);` — the client receives … | None. A directory listing shows a chopped filename that then fails to open. |
|
||||
| `system/services/fat/fat.zig:285` | into.len (inline; = envelope.packet_maximum - pr… | 224 bytes | external-data | Silent truncation, and the reply's name_len is set to the TRUNCATED value, so the client cannot tell there was more. The engine's own buffer holds 260… | Nothing. |
|
||||
| `system/services/init/init.zig:70` | max_service_args | 4 | external-data | Silent drop of the surplus arguments: loadServices, init.zig:122-127 `while (it.next()) \|argument\| { if (argument.len == 0) continue; if (service.ar… | None, and the contrast with its immediate sibling is confirmed: the max_services cap ten lines above logs `"/s… |
|
||||
| `system/services/init/init.zig:161` | maximum_bindings | 16 | our-design | An error is returned to the binder: onBind, init.zig:620-622 `const slot = for (&bindings) \|*binding\| { if (!binding.used) break binding; } else ret… | Weak, and the asymmetry is confirmed: within the same function the EPERM path logs (init.zig:602-608) and the … |
|
||||
| `system/services/init/init.zig:162` | maximum_grants | 64 | external-data | Parsing stops at 64 rows: loadGrants, init.zig:237-240 `if (grant_count >= grants.len) { _ = logging.write("... protocol.csv has more rows than the ta… | The overflow itself is logged with a direct `logging.write` naming the file (init.zig:238). The consequences a… |
|
||||
| `system/services/init/init.zig:220` | protocol_csv | 16384 bytes | external-data | Silent truncation mid-line, then a heuristic report. readConfiguration (init.zig:138-146) fills the buffer and stops; the tail rows vanish and the las… | One heuristic info log: init.zig:147 `if (used == into.len) std.log.info("{s} filled the read buffer — rows pa… |
|
||||
| `system/services/init/init.zig:220` | protocol_csv (with maximum_grants = 64 rows alon… | 16384 bytes / 64 rows | external-data | Truncation, announced on both limits. readConfiguration reports the byte cap at init.zig:147 (`std.log.info("{s} filled the read buffer — rows past {d… | Both limits log, which is what makes this the calibration point. The byte-limit line rides std.log.info (the r… |
|
||||
| `system/services/init/init.zig:274` | process_table | [64]process.ProcessDescripto… | our-design | Partly handled, partly not — confirmed exactly. refreshProcessTable (init.zig:278-282) sets `process_truncated = total > process_table.len`, and taskA… | A bind refusal outside identify does log (init.zig:602-608), but a failed identify in onBind returns -EPERM wi… |
|
||||
| `boot/efi.zig:308` | pool_pages | 64 frames (256 KiB) | hardware | TablePool.alloc, efi.zig:264-265: `if (self.next >= self.cap) return error.OutOfBootstrapFrames;`. Every map2M/map4K path is `try`, so it unwinds clea… | Good, and verified: main (efi.zig:34-39) writes unconditionally to con_out — `log("\r\nEFI: boot failed: "); l… |
|
||||
| `boot/efi.zig:543` | info_buffer | 1024 bytes | external-data | `const n = try directory.read(&info_buffer);` (efi.zig:545) into a [1024]u8. UEFI answers EFI_BUFFER_TOO_SMALL for an EFI_FILE_INFO that will not fit;… | efi.zig:71-75 on con_out, naming the UEFI error but not the file. |
|
||||
| `boot/efi.zig:543` | info_buffer (inline [1024]u8) | 1024 bytes | external-data | `const n = try directory.read(&info_buffer);` (efi.zig:545) — EFI_BUFFER_TOO_SMALL becomes an error and the `try` propagates out of walkDirectory and … | efi.zig:71-75 on con_out; names the UEFI error, not the file. |
|
||||
| `boot/efi.zig:660` | BootInformation.kernel_segments (declared system… | [8]KernelSegment | our-design | `if (n < boot_information.kernel_segments.len) { ... }` (efi.zig:659-668) with no else — the segment is copied to its physical address but never recor… | Nearly nothing, but not quite nothing: kernel.zig:159 prints `log.print(" kernel segs: {d} (mapped with W^X p… |
|
||||
| `boot/efi.zig:680` | <inline literal> (attempts) and cap = info.len +… | 8 attempts; +8 spare descrip… | hardware | Correct and safe, as claimed. efi.zig:680-700: on each attempt both pool buffers are sized from the firmware's own descriptor count plus 8, a failed g… | The fatal case reaches main's handler: 'EFI: boot failed: ExitBootServicesFailed' on con_out, then hlt. No bre… |
|
||||
| `boot/efi.zig:787` | buffer (logBytes) | 128 UTF-16 units | our-design | `if (i + 1 >= buffer.len) break;` (efi.zig:790) — the tail of the string is dropped, the result is NUL-terminated at the truncation point and printed.… | This IS the observability path — logBytes is the only way a runtime string (in practice an @errorName from mai… |
|
||||
| `build/images.zig:152` | <inline literal> (mk_fat.addArg("64")) | 64 (MiB) | our-design | Build-time hard failure. tools/make-fat-image.py:67-68 — `if cluster >= self.cluster_count + 2: sys.exit("error: image out of clusters")` inside the c… | "error: image out of clusters" on stderr, which aborts the build. It names neither the file being added nor a … |
|
||||
| `library/device/acpi/aml/interpreter.zig:55` | maximum_segments | 16 | external-data | Silent truncation in `Cursor.segment` (123-129): `const s = try self.take(4); if (name_path.count < maximum_segments) { ...store... }` — the cursor ad… | None. Line 299's comment `// unknown name -> treat as uninitialised` is by design, so a truncated name is indi… |
|
||||
| `library/device/acpi/aml/parser.zig:24` | maximum_segments | 64 | external-data | appendSegment (parser.zig:137-143) always consumes the 4 name bytes via readNameSegment — so the cursor stays in sync — but silently discards the segm… | None. aml.zig:36-56 builds a ParseResult{consumed,total} precisely so a desync is detectable, but the only con… |
|
||||
| `library/device/block/block.zig:95` | <inline literal> | 600 attempts × 50 ms = 30 s | our-design | Returns null; the caller (the FAT server) reports no volume. | Nothing logged at this level, but the number is justified against a measurement. |
|
||||
| `library/device/driver/driver.zig:165` | lookup_attempts | 100 (× lookup_pause_ms = 20 … | our-design | `hello()` runs `while (attempts < lookup_attempts) : (attempts += 1) { if (channel.openEndpoint("device-manager")) \|handle\| break handle; time.sleep… | Good, and I confirmed it at driver.zig:181 — the else branch IS the log. Requiring callers also announce their… |
|
||||
| `library/device/model/device-abi.zig:139` | DeviceDescriptor.hid | [8]u8 | hardware | No truncation is reachable today. The cited site (devices-broker.zig:121-122) cannot truncate: `node.hid()` returns `hid_buffer[0..hid_len]` from devi… | Not applicable — nothing truncates. The 8-byte-with-no-NUL width does cause a live defect elsewhere: system/se… |
|
||||
| `library/device/pci/pci.zig:37` | bar_virtual | [6]usize | hardware | Errors: `pub fn mapBar(self: *Function, bar: u8) ?usize { if (bar >= 6) return null;` — a caller asking for BAR 6 gets null rather than an out-of-boun… | Null from `mapBar`, which each driver reports in its own words. |
|
||||
| `library/device/pci/pci.zig:283` | CapabilityIterator.guard (and ExtendedCapability… | 48 and 480 | hardware | Iteration ends. Silent, but correctly so: the bound is derived from the size of the address space being walked, so reaching it means the device's chai… | Nothing — and it needs nothing, because the bound cannot cut a well-formed chain short. |
|
||||
| `library/device/pci/pci.zig:286` | <inline literal> | 48 | hardware | The iterator silently ends: `if (self.cursor == 0 or self.guard >= 48) return null;` — indistinguishable from the end of a well-formed list. `findCapa… | Nothing. No log distinguishes 'no MSI capability' from 'capability list is a loop'. |
|
||||
| `library/device/pci/pci.zig:371` | <inline literal> | 480 | hardware | Silently ends the iteration: `if (self.cursor == 0 or self.guard >= 480) return null;`. `findExtendedCapability` then reports absence. | Nothing. |
|
||||
| `library/device/registry/device-registry.zig:200` | cols | [9][]const u8 | external-data | Errors and counts. parseLine: `if (count >= cols.len) return .malformed; // too many columns` and after the loop `if (count != cols.len) return .malfo… | Verified end to end, and it is genuinely good: system/services/device-manager/device-manager.zig:71 `if (resul… |
|
||||
| `library/device/registry/device-registry.zig:239` | out_rules.len (caller-supplied) | caller's buffer | external-data | `if (result.count >= out_rules.len) { result.truncated = true; continue; }` inside parse — the surplus valid rules are dropped and the flag records it… | The best in the tree, and I confirmed the caller actually uses it: device-manager.zig:72 `if (result.truncated… |
|
||||
| `library/kernel/channel.zig:37` | path_maximum | 224 | our-design | Refuses: `join` (285) `if (total > buffer.len) return null;`. `reach` (182-193) returns null when `file_system.fsResolve` cannot write the rewritten p… | None — null, no log, same documented three-way ambiguity. |
|
||||
| `library/kernel/channel.zig:46` | name_maximum | 64 | our-design | Refuses, never truncates: `join` (282-290) `if (name.len == 0 or name.len > name_maximum) return null;`, which `openEndpoint` (238-242) turns into a n… | None (no log), and the null is deliberately three-way ambiguous per the doc at 231-237: no such contract / not… |
|
||||
| `library/kernel/memory/heap.zig:199` | <inline literal> | 16 | our-design | Refuses cleanly: `fn allocImpl(_: *anyopaque, len: usize, alignment: std.mem.Alignment, _: usize) ?[*]u8 { if (alignment.toByteUnits() > 16) return nu… | An allocation failure, which std reports as OutOfMemory — a misleading name for an alignment refusal, and no l… |
|
||||
| `library/kernel/process.zig:116` | <inline literal> | [8]u8 | our-design | The kernel refuses the *sender* rather than truncating (the receive capacity travels in r8 to `ipc_reply_wait`), so a client that calls into a process… | The refusal reaches the caller as an IPC failure; the stopping process logs nothing. |
|
||||
| `library/kernel/process.zig:185` | <inline literal> | [256]u8 | our-design | Errors, cleanly: `if (len >= blob.len) return null;` and `if (len + argument.len > blob.len) return null;` — the spawn never happens and null is retur… | Poor. `spawn*` returning null is the same answer as 'no such binary' or 'task table full'; the caller (e.g. th… |
|
||||
| `library/kernel/service.zig:94` | subscriber_capacity | 8 | our-design | onSubscribe (service.zig:264) walks `slots`, and if no slot is free returns `-envelope.ENOSPC`. The turn then closes the endpoint capability that arri… | No log in the provider. On the subscriber side the only real caller in the tree does log: system/services/disp… |
|
||||
| `library/kernel/thread.zig:41` | tls_block_size | 64 | our-design | No check exists anywhere. thread.zig:88-90 carves `closure_addr` down from the stack top and then `const tls_base = (closure_addr - tls_block_size) & … | None — nothing tracks a TLS extent, so nothing could report it. The symptom would be a thread running with cor… |
|
||||
| `library/protocol/display/display-protocol.zig:99` | max_modes | 4 | our-design | Silent drop of the surplus, but it can never fire. Provider side confirmed at system/services/display/display.zig:749-757: `var list: [4]backend_mod.M… | Nothing at either end — no log — but nothing is dropped either. |
|
||||
| `library/protocol/scanout/scanout-protocol.zig:29` | max_modes | 4 | hardware | Silent truncation on both sides, exactly as claimed. Driver: virtio-gpu.zig:525-527 `var offered = scanout_protocol.Modes{ .count = offered_modes.len … | None. Confirmed: neither `modes()` in backend.zig nor the get_modes handler in virtio-gpu.zig emits anything a… |
|
||||
| `library/xkeyboard-config/xkeyboard-config.zig:80` | map(layout, hid_usage: u8, …) | u8 (indexing Layout.keys: [2… | hardware | No overflow is possible — `keys: [256]Key` (library/xkeyboard-config/generated/layouts.zig:20) exactly covers the u8 domain, and an unmapped usage yie… | A key that maps to nothing produces no character; nothing distinguishes 'unmapped' from 'narrowed to the wrong… |
|
||||
| `system/drivers/pci-bus/pci-bus.zig:36` | <inline literal> var line: [320]u8 (the per-func… | 320 bytes | our-design | `const text = std.fmt.bufPrint(&line, "...", .{...}) catch return;` at line 37 — the whole breadcrumb for that function is dropped and logFunction ret… | Nothing. The function simply does not appear in the boot log, which reads identically to "the walk did not fin… |
|
||||
| `system/drivers/pci-bus/pci-bus.zig:182` | <inline literal> if (descriptor.resource_count >… | 8 | our-design | `break` out of the `while (i < 6)` BAR-sizing loop at line 181. Remaining BARs are never sized, never recorded in the descriptor and never restored-th… | Nothing. No log, no counter, and the loop's `configWrite16(bus, dev, function, 0x04, command)` decode-restore … |
|
||||
| `system/drivers/ps2-bus/ps2-library.zig:464` | <inline literal> while (guard < 16) in drainOutp… | 16 bytes | hardware | Returns with the output-buffer-full bit possibly still set. No error, no retry, no signal to the caller (the function returns void). Callers are ps2-b… | None — silent void return. |
|
||||
| `system/drivers/usb-hid/keyboard.zig:107` | <inline literal> receive: [64]u8 (identically at… | 64 | our-design | Currently sufficient with 4 bytes to spare: the protocol's own test asserts Protocol.event_maximum == 60 and event_maximum <= envelope.post_maximum (l… | Nothing here would report an over-long message; the driver's guard is `if (message.length < @sizeOf(hid.Keyboa… |
|
||||
| `system/drivers/usb-xhci-bus/usb-xhci-bus.zig:521` | <inline literal> hid_buffer: [8]u8 (the "P<port>… | 8 bytes | our-design | Confirmed at lines 521-523: `var hid_buffer: [8]u8 = undefined; const hid_text = std.fmt.bufPrint(&hid_buffer, "P{d}I{d}", .{ port, interface.number }… | Nothing. `catch ""` swallows the bufPrint failure, register returns an existing id which looks like a normal i… |
|
||||
| `system/drivers/usb-xhci-bus/usb-xhci-bus.zig:763` | <inline literal> if (serviced >= 32) break | 32 hub changes per tick | our-design | Deferral, not loss. `while (engine.takeHubChange()) \|change\| { ... serviced += 1; if (serviced >= 32) break; }` — the remaining changes stay in each… | Not needed; nothing is lost. There is no log, correctly. |
|
||||
| `system/drivers/usb-xhci-bus/usb-xhci-library.zig:131` | trbs_per_ring (= page_size / @sizeOf(Trb)) | 256 | our-design | Correct wraparound. `push` (155-178) writes slot `enqueue_index`, and when the index reaches `trbs_per_ring - 1` (the Link slot) it re-installs the Li… | Not needed. |
|
||||
| `system/drivers/usb-xhci-bus/usb-xhci-library.zig:456` | <inline literal> port_changes: [16]u32 | 16 | hardware | Silently dropped in `pump` (1610-1613): `if (self.port_change_count < self.port_changes.len) { self.port_changes[...] = port; self.port_change_count +… | Nothing at the drop site. Recovered in practice by the level reconcile in serviceController (usb-xhci-bus.zig:… |
|
||||
| `system/drivers/usb-xhci-bus/usb-xhci-library.zig:456` | port_changes (inline [16]u32) | 16 | hardware | Silent drop, no else (pump, 1610-1613). The PORTSC change bits are already acknowledged at 1608-1609, so the edge is consumed even when the queue entr… | Nothing at the drop site; recovered by the independent per-port level reconcile at usb-xhci-bus.zig:727-741, w… |
|
||||
| `system/drivers/usb-xhci-bus/usb-xhci-library.zig:496` | <inline literal> guard < 64 in the xECP capabili… | 64 extended capabilities | hardware | The `while (offset != 0 and guard < 64)` walk simply stops; a Supported Protocol capability past the 64th is not logged. Diagnostic only. | Nothing distinguishes 'chain ended' (the `if (next == 0) break;` at 509) from 'guard tripped'. |
|
||||
| `system/drivers/usb-xhci-bus/usb-xhci-library.zig:603` | <inline literal> device_context_array = memory.d… | 512 entries (4096 / 8), vs M… | hardware | Safe, but only by accident of type width: slot_id is a u8 (`return @truncate(event.control >> 24)` in enableSlot), so `array[device.slot_id]` cannot e… | Not applicable today. |
|
||||
| `system/drivers/usb-xhci-bus/usb-xhci-library.zig:1069` | <inline literal> while (tries < 200) with time.s… | 200 tries = ~1 s | hardware | Falls out of the loop, re-reads status at 1076, and if still not enabled: `std.log.info("hub slot {d} port {d}: reset did not enable", .{ hub.slot_id,… | Good — a specific log line naming hub slot and port, quoted above. |
|
||||
| `system/drivers/virtio-gpu/virtio-gpu.zig:77` | queue_size | 16 | our-design | The only failure direction is a device offering fewer than 16, and it is checked and refused loudly — confirmed at lines 282-286: `const device_qsize … | Explicit log line carrying the device's actual value, then a clean bring-up failure. |
|
||||
| `system/drivers/virtio-gpu/virtio-gpu.zig:163` | <inline literal> while (tries < 2000) in waitUse… | 2000 iterations ≈ 2 s | our-design | Returns false (line 172). submit returns false, command_nodata returns 0 (line 184), which never equals ok_nodata, so each bring-up call site logs its… | Good on the BRING-UP path only. The serving path is silent: presentFull's two failure returns (lines 461 and 4… |
|
||||
| `system/drivers/virtio-gpu/virtio-gpu.zig:405` | <inline literal> .scanout_id = 0 (and resource_i… | 1 scanout, 1 resource, 1 bac… | hardware | No runtime limit exists — confirmed. Grepped the whole tree: `get_display_info`, `RespDisplayInfo` and `max_scanouts` appear ONLY in their declaration… | Nothing about head count, because it is never asked. The file header does state the scope: "this instance clai… |
|
||||
| `system/drivers/virtio-gpu/virtio-gpu.zig:486` | <inline literal> while (tries < 50) with time.sl… | 50 tries = 1 s | our-design | Logs and returns, confirmed at lines 485-492: `const display = while (tries < 50) : (tries += 1) { if (channel.openEndpoint("display")) \|h\| break h;… | One clearly worded log line. The lasting consequence — the display stays on the boot framebuffer permanently —… |
|
||||
| `system/drivers/virtio-gpu/virtio-gpu.zig:525` | <inline literal> .count = offered_modes.len set … | 2 offered vs 4 slots | our-design | Not reachable at current constants (offered_modes.len = 2 <= scanout_protocol.max_modes = 4), so nothing happens today. The shape is unsafe as describ… | Would be silent at the point of the bug; the downstream re-clamps mean the surplus is dropped rather than read… |
|
||||
| `system/kernel/acpi.zig:130` | maximum_rmrr | 8 | hardware | Counted, not silent — confirmed at acpi.zig:873-882: `if (platform_information.rmrr_count < maximum_rmrr) { ...record... } else { platform_information… | Good, and I verified the log fires: system/kernel/iommu.zig:393-394 `if (info.rmrr_skipped > 0) log.print(" r… |
|
||||
| `system/kernel/architecture/x86_64/cpu.zig:69` | systemCallArg (switch arms 0..5) | 6 arguments | our-design | `else => 0` — an argument index of 6 or more silently reads as zero rather than failing. Same shape at cpu.zig:788 (`pioRead` returns 0 for a width ot… | None; a caller asking for a seventh argument gets a plausible-looking 0. |
|
||||
| `system/kernel/architecture/x86_64/gdt.zig:28` | entries | 7 | our-design | Compile-time only. `const entries = 7;` sizes the template (gdt.zig:36-44) and both gdts rows; setTssFor writes fixed indices 5 and 6, and loadOnThisC… | n/a — a build-time constraint. |
|
||||
| `system/kernel/architecture/x86_64/gdt.zig:50` | gdts | [maximum_cpus][7]u64 (128 ta… | hardware | `loadOnThisCpu(cpu)` (gdt.zig:78-84) takes `@intFromPtr(&gdts[cpu])` with no guard in this file; in a safe build an out-of-range cpu is a bounds-check… | Via the ACPI/SMP path only: kernel.zig:281 prints 'cpus : WARNING {d} core(s) beyond pool cap dropped' from pl… |
|
||||
| `system/kernel/architecture/x86_64/ioapic.zig:24` | overrides | [16]IsoEntry | hardware | Silent truncation at three independent layers, all verified: acpi.zig:558 `if (platform_information.override_count < platform_information.overrides.le… | None at any of the three layers — no counter, no log. |
|
||||
| `system/kernel/architecture/x86_64/iommu-intel.zig:284` | <inline literal> 10_000_000 (spinStatus) | 10000000 | our-design | Logs a warning and returns; `enable()` then sets `gcmd_shadow \|= gcmd_te` and the core reports the IOMMU as online even though translation may not be… | "/system/kernel: WARNING VT-d status bit never set — translation may be incomplete" — a real log line, which i… |
|
||||
| `system/kernel/architecture/x86_64/paging.zig:70` | bootstrap_physmap_limit | 4 << 30 (4 GiB) | hardware | `@panic("paging: table frame above the 4 GiB bootstrap physmap")` at paging.zig:84-85, guarded by `if (!on_own_tables and frame >= bootstrap_physmap_l… | A named panic message on the console/serial. The 9-line comment at lines 55-62 states the window, why it holds… |
|
||||
| `system/kernel/architecture/x86_64/per-cpu.zig:50` | blocks | [maximum_cpus]ArchitecturePe… | hardware | `setLocal(index, …)` and `setKernelRsp(index, …)` index unguarded; bounded upstream, ReleaseSafe bounds check as backstop. A ReleaseFast build would w… | Nothing in this file. |
|
||||
| `system/kernel/architecture/x86_64/serial.zig:83` | <inline literal> 10_000 (probe guard) | 10000 | hardware | The loop ends, `echo = register(0)` is read regardless, and `return echo == 0xAE` (serial.zig:83-86). A false answer sets uart_present = false, after … | Recoverable but indirect: present() is exposed (serial.zig:91-93) and the boot log reports the posture, and re… |
|
||||
| `system/kernel/architecture/x86_64/smp.zig:124` | <inline literal> (1 << 32) | 4294967296 | our-design | `if (cr3 >= (1 << 32)) @panic("smp: kernel page tables above 4 GiB");` — the first statement of startAp, before arm() and before any INIT/SIPI is sent… | Named panic message that states the actual condition. |
|
||||
| `system/kernel/architecture/x86_64/smp.zig:141` | <inline literal> @intCast(tramp_physical >> 12) | u8 SIPI vector (frame must b… | hardware | `const vector: u8 = @intCast(tramp_physical >> 12);` — in a safety-checked build (the default), a trampoline frame at or above 1 MiB triggers a "cast … | A safety-check panic with the generic cast message — it does not name the trampoline or the 1 MiB constraint. … |
|
||||
| `system/kernel/architecture/x86_64/smp.zig:149` | <inline literal> 100 (ms AP wake window) | 100 | our-design | `const deadline = apic.millis() + 100;` then a pause-spin on ap_alive; on expiry `return false`. The caller at system/kernel/kernel.zig:437-444 retrie… | Good, and the best in this set. On give-up: `log.print("/system/kernel: cpu apic_id {d}: no response after {d… |
|
||||
| `system/kernel/architecture/x86_64/tss.zig:48` | tss_table / ap_ist_top | [maximum_cpus]Tss and [maxim… | hardware | `rsp0Ptr` (62-64), `setApIstStack` (68-70) and `setupThisCpu` index `tss_table[cpu]` / `ap_ist_top[cpu]` with no guard in this file; the bound is enfo… | Reported by the layer above, and it really is reported: kernel.zig:280-281 `if (platform.cpusDropped() > 0) lo… |
|
||||
| `system/kernel/device-model.zig:72` | name_buffer | [24]u8 | our-design | Silent truncation in `setName` (lines 94-98, `@min` then `@memcpy`), reached from `DeviceTree.init` (line 138) and `addChild` (line 159). In practice … | None, but nothing incorrect happens either; names are diagnostic only (the boot dump), never a matching key. |
|
||||
| `system/kernel/device-model.zig:72` | Device.name_buffer | [24]u8 | our-design | Silent truncation in setName; unreachable in practice. See the kernel-core duplicate: every caller passes either a literal or the result of a bufPrint… | None needed. |
|
||||
| `system/kernel/ipc-synchronous.zig:71` | POST_MAXIMUM | 64 | our-design | `if (len > POST_MAXIMUM) return -E2BIG;` — sendLocked, ipc-synchronous.zig:566, before anything is touched. A distinct errno straight back to the call… | -E2BIG, specific and actionable. |
|
||||
| `system/kernel/ipc-synchronous.zig:113` | notify_buffer | [8]u64 | our-design | Silent drop of the NEWEST badge: notifyLocked, ipc-synchronous.zig:596-602 — the store happens only inside `if (endpoint.notify_tail -% endpoint.notif… | None, and the file argues at 593-595 why that is correct for a level rather than merely tolerable. |
|
||||
| `system/kernel/ipc-synchronous.zig:113` | Endpoint.notify_buffer | [8]u64 | our-design | Silent drop of the newest badge — outside notifyLocked's guard nothing but the wake happens (ipc-synchronous.zig:596-602). | Nothing, by design. |
|
||||
| `system/kernel/kernel.zig:246` | isos | [16]architecture.IsoEntry | hardware | Silent clamp: `var isos: [16]architecture.IsoEntry = undefined; const iso_n = @min(pinfo.override_count, isos.len);` (246-247) — a second truncation s… | None. The boot log prints the I/O APIC base and route information but never the override count. |
|
||||
| `system/kernel/kernel.zig:421` | maximum_wake_attempts | 3 | our-design | The core is left parked and bring-up continues: the `while (attempt <= maximum_wake_attempts)` loop at 437 falls through to the log at 443-444; the ad… | Exemplary, and verified verbatim: `log.print("/system/kernel: cpu apic_id {d}: no response after {d} attempts… |
|
||||
| `system/kernel/log.zig:41` | maximum_sinks | 8 | our-design | Silent ignore, confirmed at lines 47-52: `pub fn addSink(sink: SinkFn) void { if (sink_count < maximum_sinks) { sinks[sink_count] = sink; sink_count +… | None in code, though self-announcing in practice (output does not appear on the missing channel). |
|
||||
| `system/kernel/log.zig:85` | ring_capacity | 512 * 1024 | our-design | Whole records are reclaimed from the tail, never a torn record — confirmed at log-ring.zig:48, `while (self.head + record_len - self.tail > capacity) … | Best in the tree, as claimed, and I verified each mechanism: per-boot monotonic `sequence` stamped into every … |
|
||||
| `system/kernel/log.zig:244` | PanicRecord.message | [512]u8 | our-design | Silent truncation that is self-describing, confirmed at lines 251-256: `const n: u32 = @intCast(@min(message.len, panic_record.message.len)); @memcpy(… | The record carries its own length and the magic-last discipline prevents a torn read. Confirmed the caller's p… |
|
||||
| `system/kernel/process.zig:109` | maximum_arguments / maximum_argument_bytes | 8 / 256 | our-design | All three checks confirmed. process.zig:976 `if (arguments_len > maximum_argument_bytes) return fail(state);`; process.zig:1012 `if (argc == maximum_a… | -1 from system_spawn with no log and the error name discarded at process.zig:1018 — confirmed. |
|
||||
| `system/kernel/process.zig:148` | write_buffer | [256]u8 | our-design | Confirmed at process.zig:1801: `if (len <= write_buffer.len and ptr < user_half_end and ptr + len <= user_half_end) { ... } else { fail(state); }` — I… | -1 to the caller at the syscall, and the ring-level truncation is flagged in the record header — the only boun… |
|
||||
| `system/kernel/process.zig:1460` | exit_record_capacity | 64 | our-design | Confirmed. `recordExitLocked` (1467-1470) overwrites the oldest via `exit_record_next = (exit_record_next + 1) % exit_record_capacity;` and an evicted… | -ESRCH with a documented interpretation, and the docstring (1455-1459) states the equivalence to 'an id that n… |
|
||||
| `system/kernel/scheduler.zig:155` | ipc_maximum_handles | 32 | our-design | Visible and correctly unwound, confirmed. `installEntry` (ipc-synchronous.zig:618-627) scans `t.handles` and `return -ENOSPC`; the cap-passing path at… | -ENOSPC reaches the caller through the IPC status — a distinct, actionable error, unlike the -1 that most othe… |
|
||||
| `system/kernel/scheduler.zig:485` | cpus (PerCpu pool) | [maximum_cpus]PerCpu = [128] | hardware | Cannot be overrun. acpi.zig:534-543 gates recording on `if (cpu_information.count < cpu_information.cpus.len)` and counts the surplus into `dropped`, … | kernel.zig:280-281 `if (platform.cpusDropped() > 0) log.print(" cpus : WARNING {d} core(s) beyond pool … |
|
||||
| `system/kernel/tests.zig:410` | buffer (device enumerate staging) | [64]device_abi.DeviceDescrip… | our-design | Two coupled problems. (1) The value 64 duplicates devices-broker's `maximum_devices` in eight places (tests.zig:410, 1449, 2834, 3931, 4230, 4246, 437… | None for either. A test that silently examines a truncated table still passes. |
|
||||
| `system/kernel/vfs.zig:59` | maximum_prefix / maximum_rewrite | 64 / 32 | our-design | Visible refusal on the mount path: mountBackend, vfs.zig:383-384 `if (!isAbsolute(prefix) or prefix.len < 2 or prefix.len > maximum_prefix) return fal… | A -1 return with no log and no distinct errno. The mounting service's own failure message is what an operator … |
|
||||
| `system/kernel/vfs.zig:143` | Directory.path | [maximum_prefix]u8 = [64]u8 | external-data | `@memcpy(d.path[0..parent.len], parent);` (vfs.zig:143) into `path: [maximum_prefix]u8` (vfs.zig:88) with no @min and no guard — the only copy in the … | A boot-time panic in safe modes (loud but fatal, before init runs); nothing at all plus corrupted neighbours i… |
|
||||
| `system/services/acpi/acpi.zig:74` | registered | [64]Registered | our-design | Unreachable — this is NOT the operative ceiling. `if (registered_count >= registered.len) return;` at line 513 is the first statement of registerDevic… | The real ceiling does log, once per lost device: `std.log.info("register refused for {s}", .{hid[0..@intCast(h… |
|
||||
| `system/services/device-manager/device-manager.zig:51` | registry_rules | [64]registry.Rule | external-data | Rules past the 64th are dropped by `registry.parse` (library/device/registry/device-registry.zig:239-241): `if (result.count >= out_rules.len) { resul… | Good: device-manager.zig:72 `if (result.truncated) _ = logging.write("/system/services/device-manager: /system… |
|
||||
| `system/services/device-manager/device-manager.zig:129` | Driver.name_buffer | [64]u8 | our-design | Silent truncation, not refusal: `addDriver` (242-244) `const n = @min(name.len, driver.name_buffer.len); @memcpy(driver.name_buffer[0..n], name[0..n])… | Indirect: the failure appears one step later as `failed to spawn {s}` (line 268) printing the truncated path, … |
|
||||
| `system/services/device-manager/device-manager.zig:260` | id_text | [20]u8 | our-design | `arguments[0] = std.fmt.bufPrint(&id_text, "{d}", .{driver.device_id}) catch return;` (264) — on overflow `spawnDriver` returns without spawning AND w… | None — the only silent return in spawnDriver; every other failure path there logs (268 `failed to spawn`, 285/… |
|
||||
| `system/services/display/backend.zig:69` | <inline literal> | 100 | our-design | Confirmed at lines 68-75: `var tries: u32 = 0; const found = while (tries < 100) : (tries += 1) { if (findDisplay()) \|f\| break f; time.sleepMillis(5… | Logged, but misleadingly — "headless?" is one of at least three possible causes (genuinely headless, discovery… |
|
||||
| `system/services/display/compositor.zig:74` | DamageList.capacity | 16 | our-design | No loss: `add` (lines 79-93) ends `self.rects[capacity - 1] = self.rects[capacity - 1].unite(r);`, so the overflow rectangle is united into the last e… | Not needed — nothing is dropped and nothing becomes incorrect. |
|
||||
| `system/services/display/compositor.zig:121` | TileGrid.maximum_columns / maximum_rows | 128 / 128 | hardware | Graceful, documented degradation. `reset` clamps with `@min((width + tile_size - 1) / tile_size, maximum_columns)` and the same for rows (lines 139-14… | No log, and none needed — the behaviour stays correct, only the repaint granularity changes. |
|
||||
| `system/services/fat/engine.zig:24` | max_transfer_sectors | 8 | our-design | Not a failure mode, confirmed. The transfer loop clamps and issues more commands: `const run: u32 = @intCast(@min(full, max_transfer_sectors));`. The … | Not applicable — correctness is unaffected, only the number of device round-trips. |
|
||||
| `system/services/fat/engine.zig:80` | block_cache_lines | 16 | our-design | Round-robin eviction, write-through — correctness preserved, only hit rate degrades. | Not needed. |
|
||||
| `system/services/fat/engine.zig:871` | <inline literal> | 4096 | external-data | `if (sector_index > 4096) return null;` ends the search, so findFreeRun/addEntry return null, createFile/createDirectory return null and the VFS handl… | None in the service; no log, and the errno collapses into the same ENOENT everything else uses. |
|
||||
| `system/services/fat/engine.zig:1047` | run | [21]EntryLoc | external-data | Silent partial cleanup, confirmed. `var run: [21]EntryLoc = undefined;` at engine.zig:1047 (removeFile, line 1046) and 1114 (rename, line 1110), recor… | None. |
|
||||
| `system/services/init/init.zig:69` | max_services | 16 | external-data | Rows past the 16th are refused and parsing stops: init.zig:114-117 `if (service_count >= services.len) { _ = logging.write(...); break; }`. | Good: a direct `logging.write` (not just a ring entry) naming the file — `"/system/services/init: /system/conf… |
|
||||
| `system/services/init/init.zig:84` | init_csv / protocol_csv | [4096]u8 / [16384]u8 | external-data | Truncation, detected and reported. readConfiguration (init.zig:138-149) fills the buffer and stops (`while (used < into.len) { const n = file.read(int… | An explicit log naming the file and the byte count. It is a heuristic (a file exactly the buffer's size gives … |
|
||||
| `system/services/init/init.zig:84` | init_csv | 4096 bytes | external-data | Silent truncation mid-line via the shared readConfiguration (init.zig:141-146), after which the severed line parses as a service path that does not ex… | The same heuristic info log as protocol.csv: init.zig:147 '{s} filled the read buffer — rows past {d} bytes ar… |
|
||||
| `system/services/init/init.zig:160` | maximum_name | 64 | our-design | Refused, not truncated — contractName, init.zig:498-502 `if (name.len == 0 or name.len > maximum_name) return null;`, and both callers answer with an … | None on the open path, deliberately (init.zig:652-677 explains why a refusal and an absence must be the same a… |
|
||||
| `system/services/init/init.zig:173` | Binding.binary | [64]u8 | our-design | Silent truncation: onBind, init.zig:630-632 `const binary_len = @min(identity.binary.len, slot.binary.len); @memcpy(slot.binary[0..binary_len], identi… | The truncated path is what appears in the provenance line (init.zig:635) and in the EBUSY refusal message (ini… |
|
||||
| `system/services/logger/logger.zig:53` | maximum_files | 24 | our-design | No loss. fileFor (line 229) looks for a matching cached name, then a free slot, then `const cached = slot orelse evictOne() orelse return null;` at li… | Nothing logged on eviction. The symptom is slow logging, not lost logging, which is the right trade. |
|
||||
| `system/services/logger/logger.zig:57` | CachedFile.name | [logging.maximum_process_nam… | external-data | Panic, in a narrower window than claimed. fileFor formats the path FIRST (line 245, `std.fmt.bufPrint(&path, "{s}/{s}.log", ...) catch return null`) i… | A process fault, visible as the logger dying; init's crash-loop policy stops restarting it after three deaths,… |
|
||||
| `system/services/logger/logger.zig:72` | carry_capacity | 64 + 256 + 64 | our-design | consume (line 187-191): `const rest = bytes.len - offset; if (rest > carry_capacity) { carry_len = 0; // cannot happen with sane frames; drop rather t… | Partly self-reporting, and the claim was right to credit it: deliver (lines 199-206) compares header.sequence … |
|
||||
| `system/services/logger/logger.zig:82` | <inline literal> | 64 | our-design | Benign today. `service.run(64, .{ .init = initialise, .on_message = onMessage, ... })` at line 82 sizes the harness receive/reply buffer. The logger s… | None, and none needed while onMessage is a no-op. |
|
||||
| `system/services/logger/logger.zig:243` | path | [base.len + 1 + 19 + 1 + log… | our-design | Silent record loss. `const full = std.fmt.bufPrint(&path, "{s}/{s}.log", .{ boot_directory[0..boot_directory_len], relative }) catch return null;` (li… | None. Records for that process never appear on disk, and the gap accounting cannot flag it: next_expected_sequ… |
|
||||
| `tools/make-fat-image.py:53` | <inline literal> (65525) | 65525 clusters (~33 MiB at 5… | our-design | Loud build failure: `sys.exit(f"error: image too small for FAT32 ({self.cluster_count} clusters < 65525); use a larger size")` — and --verify repeats … | Build stops with a message naming both the actual count and the requirement, and telling the operator what to … |
|
||||
| `tools/make-fat-image.py:212` | <inline literal> (LFN sequence numbering) | implicitly 20 entries / 255 … | our-design | Silent corruption (my reading; not exercised): `count = len(pairs) // 13` then `entry[0] = sequence \| (0x40 if sequence == count else 0)` — for a nam… | Nothing at build time; --verify only resolves EFI/BOOT/BOOTX64.EFI by short name, so a mangled LFN chain elsew… |
|
||||
| `tools/make-fat-image.py:329` | <inline literal> (guard) | 100000 cluster hops | our-design | `return None`, which the caller turns into a misleading verdict: `sys.exit("verify: EFI/BOOT/BOOTX64.efi not found")` — a chain-too-long or cyclic FAT… | A wrong-but-visible error message. The operator is told the stub is absent when the real problem is FAT struct… |
|
||||
| `tools/make-iso-image.py:151` | sector_count | 0xFFFF (32 MiB at 512 B/sect… | our-design | Deliberate silent clamp, confirmed at line 151: `sector_count = min(0xFFFF, esp_size // 512)`, packed into the El Torito default entry's 16-bit field … | Nothing at build time about the clamp itself, but the true geometry is visible: the builder prints the ESP siz… |
|
||||
| `tools/make-xkeyboard-config.py:308` | <inline literal> (range(256)) / generated keys: … | 256 HID usages | hardware | No overflow inside the generator (HID_TO_NAME only maps usages below 0x100). The bound is exported into the generated `Layout.keys: [256]Key`, so any … | Nothing here; whatever the input service does with an out-of-range usage is its own concern (outside this area… |
|
||||
| `tools/make-xkeyboard-config.py:314` | <inline literal> (range(4)) / generated levels: … | 4 shift levels per key | external-data | Silent truncation at generation time: `levels = [resolve_keysym(kd.levels[i], keysymdef) if i < len(kd.levels) else (0, 0) for i in range(4)]` — level… | Nothing. The generator prints no warning and the generated file looks complete; the missing characters only ap… |
|
||||
|
||||
@@ -0,0 +1,140 @@
|
||||
# Bounds: how a ceiling is declared
|
||||
|
||||
*Design, 2026-08-08. Follows [fixed-bounds-audit.md](../fixed-bounds-audit.md), which
|
||||
found 235 compile-time ceilings in this tree: 139 on quantities we do not choose, 5
|
||||
recorded anywhere with their reasoning, and 171 that pass in silence when reached.*
|
||||
|
||||
A bound is a number chosen at compile time that decides how much of something the code
|
||||
can hold. `const maximum_devices = 64`. `var below: [64]Range`. `var blob: [512]u8`.
|
||||
Different units — devices, firmware memory-map entries, bytes of a USB descriptor — but
|
||||
one shape, and one recurring way of going wrong.
|
||||
|
||||
## Where a bound lives
|
||||
|
||||
**Where the thing it bounds lives.** A driver's transfer-ring size belongs to that
|
||||
driver; a protocol's payload cap belongs to that protocol; the kernel's task-table size
|
||||
belongs to the kernel. There is no central list and this document does not propose one.
|
||||
|
||||
`system/parameters.zig` is not a counter-example. It is kernel-only, and it exists for a
|
||||
specific historical reason: tunables had accumulated inside the loader↔kernel handoff
|
||||
contract, and splitting them out kept that contract to what it actually is. It is a
|
||||
tidying of one file's contents, not a registry the rest of the system reports to.
|
||||
|
||||
This matters for the mechanism below. An earlier draft had every bound declared through
|
||||
a shared `bounds` module — which would have meant adding a dependency to roughly eight
|
||||
package manifests, including `library/protocol`, which deliberately depends on nothing.
|
||||
That is a coupling the problem does not require: a bound is a local fact about local
|
||||
storage, and the only thing worth sharing is the *shape of the statement*, not a module.
|
||||
|
||||
## The declaration
|
||||
|
||||
A structured doc comment, immediately above the declaration, in the file that owns it:
|
||||
|
||||
```zig
|
||||
/// bound: logical CPUs the kernel tracks
|
||||
/// decided-by: hardware
|
||||
/// protects: the per-CPU bookkeeping arrays, which are sized at compile time
|
||||
/// at-limit: degrade — surplus cores are left parked, never brought online
|
||||
/// observed-by: platform.cpusDropped() -> the WARNING at kernel.zig:281
|
||||
pub const maximum_cpus = 128;
|
||||
```
|
||||
|
||||
Five fields, all mandatory:
|
||||
|
||||
| Field | Answers |
|
||||
|---|---|
|
||||
| `bound` | what is counted, in plain words |
|
||||
| `decided-by` | `hardware`, `external`, or `ours` — who chooses how large it gets |
|
||||
| `protects` | what this ceiling defends against |
|
||||
| `at-limit` | `refuse` / `degrade` / `truncate` / `grow`, and the detail |
|
||||
| `observed-by` | how an operator finds out it was reached |
|
||||
|
||||
`decided-by` is the classification the audit turned on. `hardware` means the machine
|
||||
chooses — PCI functions, CPUs, ACPI rows, memory-map entries. `external` means a file,
|
||||
disk structure or peer chooses. `ours` means we do: a stack size, a tick rate, our own
|
||||
protocol's payload. A fixed bound on the first two is a defect rather than a tunable.
|
||||
|
||||
## What the build step enforces
|
||||
|
||||
A step in `build.zig` reads the tree and fails on:
|
||||
|
||||
1. **A bound with no declaration.** A fixed-size array or a `maximum_*`/`max_*` constant
|
||||
with no `bound:` block above it. The 235 that exist today are allowlisted by
|
||||
file+line+name, so only newly written ones are gated — the rule can land without a
|
||||
235-site sweep in front of it.
|
||||
2. **A missing field.** All five or it fails. This alone is the 97 bounds with no
|
||||
comment at all and the 171 with no observability.
|
||||
3. **An unspeakable `at-limit`.** The vocabulary is closed. There is no `silent`, no
|
||||
`drop`, and nothing meaning *allow*. `truncate` is legal only with a marker the
|
||||
reader can see — `klog_maximum_message` qualifies because the record carries
|
||||
`klog_flag_truncated`; the USB configuration descriptor cut at 512 bytes does not,
|
||||
because nothing records that anything was lost.
|
||||
4. **A stale allowlist entry.** If a listed bound is fixed or deleted, its entry goes
|
||||
too, so the list can only shrink.
|
||||
|
||||
Because this is text and not a Zig type, it also covers `boot/`, which imports almost
|
||||
nothing, and `tools/*.py`, where the audit found bounds as well. One mechanism, whole
|
||||
tree, no new dependency edges.
|
||||
|
||||
## Coupled bounds
|
||||
|
||||
Two numbers that must agree, agreeing in code rather than in a comment — no module
|
||||
needed, just a `comptime` block where one of them lives:
|
||||
|
||||
```zig
|
||||
comptime {
|
||||
if (maximum_domains != devices_broker.maximum_devices)
|
||||
@compileError("iommu.confined is indexed by device id; an id past its end is " ++
|
||||
"left unconfined while confineDevice still reports success");
|
||||
}
|
||||
```
|
||||
|
||||
`maximum_domains = 64` and `maximum_devices = 64` agree today only by a sentence in a
|
||||
comment, and the agreement fails open. This is the clause with a live hole behind it,
|
||||
and the reason raising `maximum_devices` alone would be a privilege escalation rather
|
||||
than a fix.
|
||||
|
||||
## The worked bad case
|
||||
|
||||
`devices_broker.maximum_devices`, which had no comment at all:
|
||||
|
||||
```zig
|
||||
/// bound: device nodes for the whole machine — firmware-discovered plus registered
|
||||
/// decided-by: hardware
|
||||
/// protects: nothing; this is a sizing guess about someone else's computer
|
||||
/// at-limit: refuse — ENOSPC from device_register, dropped++ during discovery
|
||||
/// observed-by: kernel.zig:203 counts discovery drops only, NOT runtime refusals
|
||||
const maximum_devices = 64;
|
||||
```
|
||||
|
||||
Writing it out is the argument. `decided-by: hardware` alongside a `protects` that
|
||||
admits there is no threat describes a bound that should not be fixed at all, and
|
||||
`observed-by` cannot be filled in honestly. The build step does not reject this — it
|
||||
makes it impossible to write down without noticing.
|
||||
|
||||
## The work-list this produces
|
||||
|
||||
`decided-by` is machine-readable, so the sweep is a query: every `hardware` or
|
||||
`external` bound whose `at-limit` is not `grow`. That is 139 of the 235, and it is the
|
||||
order the fixing takes — by class, not by our guesses about which machines get run.
|
||||
Reachability is exactly what a new computer changes; the Ryzen's cap was unreachable
|
||||
until it wasn't.
|
||||
|
||||
## What this does not do
|
||||
|
||||
**It is a statement, not a proof.** Nothing checks that the code does what `at-limit`
|
||||
claims. The enforcement is completeness and vocabulary: you cannot leave the question
|
||||
unanswered, and you cannot answer it with "silently".
|
||||
|
||||
**It carries no occupancy.** Knowing `system/configuration/protocol.csv` sits at 50 of
|
||||
64 grant rows still needs a counter and somewhere to report it. The declarations make
|
||||
that cheap to add later; it is not here.
|
||||
|
||||
**It resizes nothing.** Declaring `maximum_devices` honestly does not make the Ryzen
|
||||
work. It makes the next machine's failure legible, and it names the 139 that need real
|
||||
fixes.
|
||||
|
||||
**Enforcement is at build time, not compile time.** A Zig type could have made a missing
|
||||
field a compile error. That version needed the shared module, and the module was not
|
||||
worth the coupling — so a missing field is a failed build step instead. In practice both
|
||||
mean `zig build` stops; the difference is which stage prints the message.
|
||||
@@ -55,7 +55,40 @@ source/destination addressing internally at L0 — the way IP runs under TCP —
|
||||
and none of it surfaces into the packet header. Protocols stay ignorant of
|
||||
distance.
|
||||
|
||||
## The transport (L0): a buffer and a doorbell
|
||||
## Establishment: two planes, one namespace
|
||||
|
||||
`target` routes objects *within* a provider. It cannot route *between* provider
|
||||
processes, because "who am I talking to" was decided at establishment — so when a
|
||||
protocol's objects live in several processes, something must resolve establishment
|
||||
to the right one. A real machine forced the question: three xHCI controllers,
|
||||
three `usb-xhci-bus` processes, one `usb-transfer` name, and whichever instance
|
||||
bound first owned every class driver's discovery — a mouse on the second
|
||||
controller was unreachable (`could not open device 50`). The registry's exclusive
|
||||
bind was built for parties that are genuinely singular; per-device driver
|
||||
processes are deliberately *not* singular — one process per device found is what
|
||||
makes a single driver recompilable and restartable while the machine runs.
|
||||
|
||||
So establishment has two planes, and the namespace stays instance-free on both:
|
||||
|
||||
- **Applications establish by name.** `/protocol/input`, `/protocol/display`,
|
||||
`/protocol/ufs` — served by *services*, the layer that aggregates
|
||||
(driver-model.md: services face applications). These are genuinely singular,
|
||||
the registry's exclusive bind is correct for them, and multiplicity behind
|
||||
them is `target`'s job. Ten mice still merge into one `/protocol/input`.
|
||||
- **Drivers establish by topology.** A driver instance *creates its channel* —
|
||||
its own endpoint — and hands it up in the `hello` it already sends, as a
|
||||
capability. The device manager is the establishment router for the device
|
||||
tree because it already owns the routing fact: it knows which instance
|
||||
reported which child. A consumer's `hello` for device N returns a capability
|
||||
to N's provider. The name `usb-transfer` names the *contract*; no process
|
||||
binds it.
|
||||
|
||||
Establishment is always a synchronous call, and always consumer-initiated: the
|
||||
kernel carries one capability in each direction of a call (request and reply),
|
||||
and deliberately none on a push — a provider cannot foist a channel on anyone.
|
||||
Re-establishment after a provider restart is therefore also consumer-initiated:
|
||||
the consumer's channel dies with the provider, and a fresh `hello` fetches the
|
||||
successor's.
|
||||
|
||||
Strip any transport to its skeleton and the same two parts remain:
|
||||
|
||||
|
||||
@@ -0,0 +1,187 @@
|
||||
# Device authority: the implementation of delegation
|
||||
|
||||
*Implementation design, 2026-08-08. The **what** is settled in
|
||||
[device-manager.md](../device-driver-development/device-manager.md) — "structure in the
|
||||
manager, authority in the kernel", and delegation as the step after `hello`. This
|
||||
document is the **how**, and the decisions that paragraph leaves open.*
|
||||
|
||||
Read first: [drivers.md](../device-driver-development/drivers.md) (the claim is the
|
||||
capability), [driver-model.md](../device-driver-development/driver-model.md) (the three
|
||||
invariants), [device-manager.md](../device-driver-development/device-manager.md) (the
|
||||
tree, the matcher, the supervisor).
|
||||
|
||||
## The one thing not yet true
|
||||
|
||||
`device-manager.md` says assignment "stays argv for now", and names the next step:
|
||||
|
||||
> The step after `hello` exists is delegation: the manager claims (or is granted) the
|
||||
> devices and passes the claim to the driver over IPC (the M13 capability-transfer
|
||||
> mechanism), replacing first-come-first-served `device_claim` with policy.
|
||||
|
||||
Until that lands, the manager's matching is advisory. `device_claim` checks only that
|
||||
the device exists and is unheld ([devices-broker.zig](../../system/kernel/devices-broker.zig)):
|
||||
|
||||
```zig
|
||||
pub fn claim(id: u64, owner: u32) ClaimError!void {
|
||||
if (id >= count) return error.NoSuchDevice;
|
||||
if (claimed[@intCast(id)] != null) return error.AlreadyClaimed;
|
||||
claimed[@intCast(id)] = owner;
|
||||
}
|
||||
```
|
||||
|
||||
A driver is spawned with its device id in `argv[1]` and claims it; any process could
|
||||
pass any integer instead. Since a claim is a licence to map physical memory, that is the
|
||||
gap this document closes.
|
||||
|
||||
## Decision 1: the manager claims, then transfers
|
||||
|
||||
`device-manager.md` leaves "claims (or is granted)" open. **Claims.**
|
||||
|
||||
The manager runs before any driver exists — `init` starts it from `init.csv`, and it is
|
||||
what spawns drivers — so it takes the seeded devices unopposed and there is nothing
|
||||
unheld left for anyone to race for. One new call moves ownership on:
|
||||
|
||||
```
|
||||
device_transfer(device_id, task_id) -> 0/-errno
|
||||
```
|
||||
|
||||
The kernel checks only that the caller currently holds the device. No names, no
|
||||
attestation, no notion of "the device manager" — the rule is *you may give away what you
|
||||
hold*, which is the capability discipline already in force.
|
||||
|
||||
The alternative was the kernel granting roots to a task it recognises by binary path
|
||||
plus a PID-1 supervisor. It is more robust — it does not depend on the manager being
|
||||
first — but it puts a binary name inside the kernel, and the goal is that the kernel
|
||||
keeps only what cannot safely live in user space. A name is not that.
|
||||
|
||||
**The residual, stated plainly:** authority here rests on the manager claiming first.
|
||||
That holds because `init.csv` decides what starts and in what order, so it is an
|
||||
operator-visible ordering rather than an attacker-controlled one — but it is an
|
||||
assumption, not an enforced invariant. The enforced version arrives with the spawn
|
||||
capability [drivers.md](../device-driver-development/drivers.md) already names as
|
||||
missing ("`system_spawn` is currently ungated … because there is no spawn capability
|
||||
yet"). This design is compatible with it and does not block on it.
|
||||
|
||||
## Decision 2: the kernel stops holding inventory
|
||||
|
||||
The kernel reads exactly three things out of a descriptor: **physical ranges** (to check
|
||||
a mapping falls inside one), **interrupt numbers**, and **one PCI BDF** (to key an IOMMU
|
||||
domain). Vendor, device and subsystem ids, class triples, `_HID` strings, bus addresses
|
||||
and names are stored only so `device_enumerate` can hand them back — which
|
||||
`device-manager.md` already resolves: that call "fades to a manager-internal (then
|
||||
deleted) seam", because the manager owns the tree as data.
|
||||
|
||||
So the kernel's table becomes: **parent, resources, holder, BDF.** That is what cannot
|
||||
safely run in user space; the rest moves.
|
||||
|
||||
**Devices with no resources leave the kernel entirely.** A USB device is addressed
|
||||
through its controller and carries `resource_count = 0`
|
||||
([driver-model.md](../device-driver-development/driver-model.md): "that case is allowed
|
||||
and is the common one"). It conveys no mapping authority, so there is nothing for the
|
||||
kernel to enforce and no reason for it to know. It is inventory, and inventory is the
|
||||
manager's — reported by `child_added`, which already carries everything needed.
|
||||
|
||||
That is also the case that made `maximum_children_per_parent` necessary: a zero-resource
|
||||
child sidesteps containment, so a driver could loop `device_register` and fill the
|
||||
shared table. Once such children are not kernel objects, every remaining entry is a real
|
||||
contained subdivision of something the caller holds.
|
||||
|
||||
## Decision 3: no shared ceiling; a per-holder quota instead
|
||||
|
||||
`maximum_devices = 64` and `maximum_children_per_parent = 16` are numbers we invented,
|
||||
and both are shared — one driver's enumeration starves every other driver, which is how
|
||||
an AMD Ryzen booted with no USB and no storage.
|
||||
|
||||
- **The table becomes dynamic.** It is built after `heap.init` (`kernel.zig`: `pmm.init`
|
||||
at 137, `heap.init` at 179, `devices_broker.init` at 202), so nothing prevents it. No
|
||||
specification bounds how many devices a machine has, so nothing should bound ours.
|
||||
- **`maximum_children_per_parent` is deleted**, because the authorisation it stood in
|
||||
for now exists.
|
||||
- **A per-holder quota replaces them.** Dynamic storage without a bound moves the
|
||||
ceiling to the kernel heap, which is shared and fatal rather than partial — strictly
|
||||
worse. The bound that is *not* worse is one charged to the task that caused it: a
|
||||
driver that loops `device_register` exhausts its own allowance, is refused with an
|
||||
attributable errno, and is restarted by its supervisor while every other driver
|
||||
carries on. That is the microkernel property rather than a workaround for it, and it
|
||||
is declared through [bounds.md](bounds.md) like any other.
|
||||
|
||||
## The shape of the change
|
||||
|
||||
| | Before | After |
|
||||
|---|---|---|
|
||||
| Manager gets its devices | claims them, unauthorised | claims them (first, unopposed) |
|
||||
| Driver gets its device | `argv[1]` + `device_claim` | receives it in the `hello` reply |
|
||||
| Kernel checks | is it free? | do you hold it? |
|
||||
| Kernel stores | the full descriptor | parent, resources, holder, BDF |
|
||||
| Zero-resource devices | kernel table entries | manager records only |
|
||||
| Table size | `maximum_devices = 64` | dynamic, per-holder quota |
|
||||
| Children per parent | `maximum_children_per_parent = 16` | deleted |
|
||||
|
||||
Bring-up order changes for the five claiming drivers: `hello` must precede the claim,
|
||||
because the reply is where the device arrives. `pci-bus` today does the reverse — its
|
||||
own comment reads "Claim the bridge, map the ECAM, hello the manager, then scan."
|
||||
|
||||
## What does not change
|
||||
|
||||
- The three invariants of [driver-model.md](../device-driver-development/driver-model.md):
|
||||
a claim is exclusive, a descriptor is a licence to map physical memory, therefore
|
||||
containment. This design strengthens the first and touches neither of the others.
|
||||
- The display service's GOP path. The framebuffer is not a device — it is where pixels
|
||||
go, handed over by the loader, and the compositor uses it as the boot floor until a
|
||||
real display driver announces itself
|
||||
([display-v2.md](../device-driver-development/display-v2.md)). The kernel wraps it in
|
||||
a display-class descriptor so `mmio_map` can hand it over write-combining; that is
|
||||
plumbing for a mapping, not a claim about what it is.
|
||||
- Supervision, restart, pruning and re-report
|
||||
([device-manager.md](../device-driver-development/device-manager.md),
|
||||
[process-lifecycle.md](process-lifecycle.md)). Delegation slots into the existing
|
||||
`hello` exchange and changes none of it.
|
||||
- `device_register` idempotency, which is what lets a restarted bus rebuild the same
|
||||
ids.
|
||||
|
||||
## How it is verified
|
||||
|
||||
The invariant is: **a process holds what it was handed and cannot name its way into
|
||||
holding more.** The suite has no adversarial device case today — the audit's lesson was
|
||||
that "the suite contains no attacker" — so this adds one: a process that was handed
|
||||
nothing calls `device_claim` and `device_transfer` on a device another driver holds, and
|
||||
on one nobody holds, and is refused each time with its own errno.
|
||||
|
||||
The Ryzen is the acceptance test for the ceiling half: it is the machine that found the
|
||||
constants, and the one that proves them gone.
|
||||
|
||||
## As built (2026-08-09)
|
||||
|
||||
Three details settled differently, or beyond, what the sections above say:
|
||||
|
||||
- **The device arrives with the spawn, not in the `hello` reply.** `system_spawn` grew a
|
||||
sixth argument: the manager names the device it is giving, the kernel verifies the
|
||||
caller holds it before the child exists, and the child holds it before its first
|
||||
instruction. A give that fails after the spawn (the device stopped being the giver's,
|
||||
or its confinement was refused) kills the child — a driver running without the
|
||||
hardware it was spawned for is worse than no driver. `hello` stays what it was: the
|
||||
liveness handshake.
|
||||
- **A given device is a loan.** When the holder dies, the device returns to the giver if
|
||||
the giver is still alive — so a respawned driver is handed the same device by its
|
||||
manager instead of racing anyone for a released claim. Only if the lender is also dead
|
||||
does the device become unheld.
|
||||
- **Confinement moves with the device, and is rebuilt when the loan comes back.** A
|
||||
transfer re-points the existing IOMMU record; but a death tears the domain down
|
||||
*before* the loan returns, so the next delegation of that device finds no record and
|
||||
confines afresh — under the same fail-closed rule as a first claim (`ECONFINE`, the
|
||||
give does not stand). Without that, one driver crash left its device silently
|
||||
unconfined forever. Both give paths (`device_transfer` and spawn's give) share one
|
||||
body in [process.zig](../../system/kernel/process.zig) (`giveDeviceLocked`), and the
|
||||
`iommu` kernel test drives the death-and-respawn sequence against it directly.
|
||||
|
||||
A fourth followed on 2026-08-09, when a real three-controller machine broke the last
|
||||
name-shaped assumption ([communication.md](communication.md) "Establishment: two
|
||||
planes, one namespace"): **driver-layer channels ride the same hello.** A driver's
|
||||
serving endpoint goes up as the hello's capability; consumers get their provider's
|
||||
channel down in the reply, routed by the manager's lineage — the reporter for a
|
||||
class driver, the bound driver for a `.consumer`. `usb-transfer` and `block`
|
||||
stopped being registry names, and a reporter's death now reaps its class-driver
|
||||
subtree so the re-report can rebuild it against the successor — the manager half of
|
||||
the loan story above. One correction to the earlier bullet: `hello` is no longer
|
||||
*only* the liveness handshake; it is also where establishment happens. The device
|
||||
itself still arrives with the spawn.
|
||||
@@ -150,6 +150,57 @@ they are wire values a Rust program needs verbatim. What stays private in
|
||||
`abi.zig` is exactly the thing the vDSO exists to hide: the `SystemCall`
|
||||
numbers and the trap convention.
|
||||
|
||||
## Errors: the errno space
|
||||
|
||||
A failed call returns `-errno`. The runtime detects failure the way Linux
|
||||
does — a return value in the top 4096 — so every code stays inside 1..4095.
|
||||
These are **public**: unlike the call numbers, a caller must be able to read
|
||||
them verbatim, and they are the same vocabulary whether the number came from
|
||||
the kernel or from a user-space provider answering over IPC.
|
||||
|
||||
They are defined once in `system/abi.zig`. The kernel restates them in
|
||||
`system/kernel/ipc-synchronous.zig` and the envelope restates the
|
||||
provider-facing subset in `library/protocol/envelope/envelope.zig` (the
|
||||
`protocol` package deliberately depends on nothing, so it cannot import
|
||||
`abi`); a comptime check in `library/device/driver/driver.zig` makes drift a
|
||||
compile error.
|
||||
|
||||
| # | Name | Meaning |
|
||||
|---|------|---------|
|
||||
| 1 | `EBADF` | bad handle |
|
||||
| 2 | `E2BIG` | an argument exceeds its maximum (a message, a descriptor's resource count) |
|
||||
| 3 | `EFAULT` | buffer unmapped, or outside the user half |
|
||||
| 4 | `ENOENT` | no such name |
|
||||
| 5 | `ENOSPC` | a kernel table is full (handles, devices) |
|
||||
| 6 | `ENOMEM` | out of memory |
|
||||
| 7 | `EPEER` | the peer died before replying — its process exited or was killed |
|
||||
| 8 | `ESRCH` | no such process |
|
||||
| 9 | `EPERM` | not permitted: the caller is not the owner or supervisor |
|
||||
| 10 | `ENOSYS` | this protocol has no such operation |
|
||||
| 11 | `EPROTO` | malformed packet: shorter than the verb it names |
|
||||
| 12 | `EBUSY` | the thing asked for is held by someone still alive |
|
||||
| 13 | `ENODEV` | no such device id |
|
||||
| 14 | `ECHILDREN` | this parent already holds as many children as it can |
|
||||
| 15 | `ERANGE` | a resource escapes the window it must fall inside |
|
||||
| 16 | `ECONFINE` | the device could not be placed under IOMMU translation |
|
||||
|
||||
`EPEER` is the one with no POSIX counterpart and it is worth stating plainly:
|
||||
synchronous IPC blocks the caller until the server replies, so the caller
|
||||
needs an answer for "the server died while I was waiting." It is not a
|
||||
transport error and not a refusal — the request may well have been carried
|
||||
out — it says only that no reply is coming. A client that treats it as
|
||||
"retry" can duplicate work; the honest response is to re-resolve the protocol
|
||||
name, because the provider it held is gone.
|
||||
|
||||
**A refusal names the rule that refused it.** This is a rule and not a
|
||||
courtesy. `device_register` alone can fail six ways, and until each got its
|
||||
own code a bus driver could only report "refused" — which is how an AMD
|
||||
desktop came to boot with a working display, no USB and no storage, with
|
||||
three independent causes indistinguishable in the log. See
|
||||
[fixed-bounds-audit.md](../fixed-bounds-audit.md). A new failure mode that
|
||||
does not fit an existing code gets a new one here rather than borrowing the
|
||||
nearest.
|
||||
|
||||
## Enforcement, and an honest threat model
|
||||
|
||||
Renumbering only has teeth if the kernel **refuses syscalls that don't come
|
||||
|
||||
@@ -7,10 +7,8 @@
|
||||
//! `runtime.dma.alloc`), so whole sectors move without crossing the IPC size
|
||||
//! limit — the same handoff usb-storage uses toward the controller.
|
||||
|
||||
const channel = @import("channel");
|
||||
const envelope = @import("envelope");
|
||||
const ipc = @import("ipc");
|
||||
const time = @import("time");
|
||||
const block_protocol = @import("block-protocol");
|
||||
|
||||
const Protocol = block_protocol.Protocol;
|
||||
@@ -75,26 +73,7 @@ pub const Device = struct {
|
||||
}
|
||||
};
|
||||
|
||||
/// One open attempt, no waiting — for a server that retries on its own
|
||||
/// timer (the fat service) instead of blocking its harness in here.
|
||||
pub fn tryOpen() ?Device {
|
||||
if (channel.openEndpoint("block")) |handle| return .{ .endpoint = handle };
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Open `/protocol/block`, retrying generously while the USB storage chain
|
||||
/// (controller reset, enumeration, mass-storage bring-up) comes up.
|
||||
pub fn open() ?Device {
|
||||
// Patient: the whole USB storage chain (firmware discovery, xHCI reset and
|
||||
// enumeration, mass-storage bring-up) must complete first, which can take
|
||||
// tens of seconds under emulation.
|
||||
var attempts: usize = 0;
|
||||
// 30 s covers the slowest observed healthy chain (a flaky QEMU enumeration
|
||||
// completed at ~24 s); a machine whose stick genuinely failed setup should
|
||||
// not sit a further minute pretending otherwise.
|
||||
while (attempts < 600) : (attempts += 1) {
|
||||
if (channel.openEndpoint("block")) |handle| return .{ .endpoint = handle };
|
||||
time.sleepMillis(50);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
// There is deliberately no open-by-name here: `block` is not a registry name.
|
||||
// One storage process serves each volume, and a consumer receives its volume's
|
||||
// channel from the device manager (establishment by lineage, communication.md
|
||||
// "Establishment: two planes"), then wraps it: `block.Device{ .endpoint = c }`.
|
||||
|
||||
@@ -22,14 +22,67 @@ inline fn failed(r: usize) bool {
|
||||
return r > ~@as(usize, 0) - 4095;
|
||||
}
|
||||
|
||||
/// The errno inside a failed return. Only meaningful when `failed(r)`.
|
||||
inline fn errnoOf(r: usize) i64 {
|
||||
return -@as(i64, @bitCast(r));
|
||||
}
|
||||
|
||||
// The envelope restates the kernel's errno numbering by hand, because the
|
||||
// `protocol` package deliberately depends on nothing (so it cannot import `abi`).
|
||||
// This module is one of the few that can see both halves, so it is where they are
|
||||
// held together: drift becomes a compile error here rather than a driver reporting
|
||||
// the wrong reason for a refusal. Anything linking a driver compiles this.
|
||||
comptime {
|
||||
if (envelope.ENOENT != abi.ENOENT) @compileError("envelope.ENOENT has drifted from abi.ENOENT");
|
||||
if (envelope.ENOSPC != abi.ENOSPC) @compileError("envelope.ENOSPC has drifted from abi.ENOSPC");
|
||||
if (envelope.EPERM != abi.EPERM) @compileError("envelope.EPERM has drifted from abi.EPERM");
|
||||
if (envelope.ENOSYS != abi.ENOSYS) @compileError("envelope.ENOSYS has drifted from abi.ENOSYS");
|
||||
if (envelope.EPROTO != abi.EPROTO) @compileError("envelope.EPROTO has drifted from abi.EPROTO");
|
||||
if (envelope.EBUSY != abi.EBUSY) @compileError("envelope.EBUSY has drifted from abi.EBUSY");
|
||||
}
|
||||
|
||||
/// Copy up to `buffer.len` device descriptors into `buffer`; returns the total count.
|
||||
pub fn enumerate(buffer: []DeviceDescriptor) usize {
|
||||
return sc.systemCall2(.device_enumerate, @intFromPtr(buffer.ptr), buffer.len);
|
||||
}
|
||||
|
||||
/// Take exclusive ownership of device `id`. Returns false if taken or invalid.
|
||||
pub fn claim(id: u64) bool {
|
||||
return !failed(sc.systemCall1(.device_claim, id));
|
||||
/// Why a `transfer` failed. `NotHeld` is the interesting one — it means the caller tried
|
||||
/// to give away a device it does not have, which is the whole rule.
|
||||
pub const TransferError = error{ NoSuchDevice, NotHeld, NoSuchTask, Refused };
|
||||
|
||||
/// Give device `id` to task `to`. **A move, not a copy** — a claim is exclusive, so the
|
||||
/// caller stops holding it. This is how the device manager hands a driver the device it
|
||||
/// matched, replacing first-come-first-served claiming with policy
|
||||
/// (docs/os-development/device-authority.md).
|
||||
pub fn transfer(id: u64, to: u32) TransferError!void {
|
||||
const r = sc.systemCall2(.device_transfer, id, to);
|
||||
if (!failed(r)) return;
|
||||
return switch (errnoOf(r)) {
|
||||
abi.ENODEV => error.NoSuchDevice,
|
||||
abi.EPERM => error.NotHeld,
|
||||
abi.ESRCH => error.NoSuchTask,
|
||||
else => error.Refused,
|
||||
};
|
||||
}
|
||||
|
||||
/// Why a `claim` failed. Worth distinguishing: `AlreadyClaimed` means back off and
|
||||
/// let the owner have it, `NoSuchDevice` means this id is stale and the caller should
|
||||
/// re-enumerate, and `NotConfined` means the machine could not place the device under
|
||||
/// IOMMU translation — the claim was rolled back, and that one is a fault report, not
|
||||
/// a retry. `Refused` is an errno this library does not know a name for.
|
||||
pub const ClaimError = error{ NoSuchDevice, AlreadyClaimed, NotConfined, NotYours, Refused };
|
||||
|
||||
/// Take exclusive ownership of device `id`.
|
||||
pub fn claim(id: u64) ClaimError!void {
|
||||
const r = sc.systemCall1(.device_claim, id);
|
||||
if (!failed(r)) return;
|
||||
return switch (errnoOf(r)) {
|
||||
abi.ENODEV => error.NoSuchDevice,
|
||||
abi.EBUSY => error.AlreadyClaimed,
|
||||
abi.ECONFINE => error.NotConfined,
|
||||
abi.EPERM => error.NotYours, // delegated hardware: it must be handed to you
|
||||
else => error.Refused,
|
||||
};
|
||||
}
|
||||
|
||||
/// Map resource `resource_index` (which must be an MMIO window) of claimed device
|
||||
@@ -56,11 +109,38 @@ pub const no_pci_class = device_abi.no_pci_class;
|
||||
/// a bus driver may only subdivide what it already owns. `descriptor.id` and `descriptor.parent`
|
||||
/// are ignored. A device with no resources at all is fine — a USB device is reached
|
||||
/// through its controller, not by MMIO.
|
||||
pub fn register(parent_id: u64, descriptor: *const DeviceDescriptor) ?u64 {
|
||||
pub fn register(parent_id: u64, descriptor: *const DeviceDescriptor) RegisterError!u64 {
|
||||
const r = sc.systemCall2(.device_register, parent_id, @intFromPtr(descriptor));
|
||||
return if (failed(r)) null else r;
|
||||
if (!failed(r)) return r;
|
||||
return switch (errnoOf(r)) {
|
||||
abi.ENOSPC => error.TableFull,
|
||||
abi.ENODEV => error.NoSuchParent,
|
||||
abi.EPERM => error.NotYourParent,
|
||||
abi.E2BIG => error.TooManyResources,
|
||||
abi.ECHILDREN => error.ParentFull,
|
||||
abi.ERANGE => error.NotContained,
|
||||
abi.EFAULT => error.BadDescriptor,
|
||||
else => error.Refused,
|
||||
};
|
||||
}
|
||||
|
||||
/// Why a `register` failed. These are not interchangeable and a bus driver should
|
||||
/// say which one it hit: `ParentFull` and `TableFull` are different ceilings with
|
||||
/// different fixes, and `NotContained` is not a ceiling at all — it means the child
|
||||
/// resource escaped the window the parent actually owns. Reporting all of them as
|
||||
/// one refusal is what made an AMD desktop boot with no USB and no storage, and gave
|
||||
/// no way to tell which of three causes it was (docs/fixed-bounds-audit.md).
|
||||
pub const RegisterError = error{
|
||||
TableFull, // the kernel's device table is full, machine-wide
|
||||
ParentFull, // this parent already holds as many children as it can
|
||||
NoSuchParent, // no device with that id
|
||||
NotYourParent, // that device exists but this process has not claimed it
|
||||
TooManyResources, // the descriptor declares more resources than one device may hold
|
||||
NotContained, // a resource escapes the parent's window
|
||||
BadDescriptor, // the descriptor pointer did not read back
|
||||
Refused, // an errno this library does not know a name for
|
||||
};
|
||||
|
||||
/// Bind resource `resource_index` (which must be an IRQ) of claimed device `device_id` to
|
||||
/// `endpoint`. From then on the interrupt arrives as an asynchronous notification:
|
||||
/// `ipc.replyWait` on that endpoint returns with the high bit set in `badge` and the
|
||||
@@ -173,6 +253,51 @@ const lookup_pause_ms: u64 = 20;
|
||||
/// The device this driver was assigned is the packet's `Header.target` — the manager's
|
||||
/// object addressing, so `no_device` here is a driver that serves none.
|
||||
pub fn hello(role: Role, device_id: u64) ?ipc.Handle {
|
||||
const exchanged = helloExchange(role, device_id, null, false) orelse return null;
|
||||
return exchanged.manager;
|
||||
}
|
||||
|
||||
/// What one hello moved, besides the handshake itself: the manager's endpoint
|
||||
/// (every hello), and — when asked — the channel to the driver that provides
|
||||
/// this device, shared into our handle table by the reply.
|
||||
pub const Exchange = struct {
|
||||
manager: ipc.Handle,
|
||||
/// The provider's channel, when `want_channel` asked and the manager's
|
||||
/// lineage had one. Null with `want_channel` set means the provider is not
|
||||
/// there YET (its own hello has not landed, or it is mid-restart) — a
|
||||
/// retryable condition, never a verdict.
|
||||
channel: ?ipc.Handle,
|
||||
};
|
||||
|
||||
/// A consumer's whole establishment step: hello until the channel to this
|
||||
/// device's provider arrives. `serving` (a provider-and-consumer like
|
||||
/// usb-storage: block endpoint up, bus channel down) rides the FIRST exchange
|
||||
/// only — the manager keeps it, so retries need not resend it. The manager
|
||||
/// acks a hello whose provider is not there yet (mid-restart, re-report on
|
||||
/// the way) with no channel — retryable by design — so this re-hellos on the
|
||||
/// ONE manager handle, on the same cadence the old name lookup used, and
|
||||
/// gives up on a refusal or a vanished manager. Re-hello is benign: the
|
||||
/// manager just re-marks the entry running.
|
||||
pub fn helloForChannel(role: Role, device_id: u64, serving: ?ipc.Handle) ?ipc.Handle {
|
||||
const first = helloExchange(role, device_id, serving, true) orelse return null;
|
||||
if (first.channel) |bus| return bus;
|
||||
var attempts: u32 = 0;
|
||||
while (attempts < lookup_attempts) : (attempts += 1) {
|
||||
time.sleepMillis(lookup_pause_ms);
|
||||
const again = helloOn(first.manager, role, device_id, null, true) orelse return null;
|
||||
if (again.channel) |bus| return bus;
|
||||
}
|
||||
std.log.info("no provider channel for device {d}", .{device_id});
|
||||
return null;
|
||||
}
|
||||
|
||||
/// The full handshake (communication.md "Establishment: two planes, one
|
||||
/// namespace"): a provider hands `serving` — the endpoint its consumers will
|
||||
/// be routed to — up with the request; a consumer sets `want_channel` and
|
||||
/// receives its device's provider channel with the reply. One call can do
|
||||
/// both (usb-storage serves block and consumes usb-transfer). The kernel
|
||||
/// shares capabilities as refcounted copies, so `serving` stays ours too.
|
||||
pub fn helloExchange(role: Role, device_id: u64, serving: ?ipc.Handle, want_channel: bool) ?Exchange {
|
||||
var attempts: u32 = 0;
|
||||
const manager = while (attempts < lookup_attempts) : (attempts += 1) {
|
||||
if (channel.openEndpoint("device-manager")) |handle| break handle;
|
||||
@@ -181,29 +306,38 @@ pub fn hello(role: Role, device_id: u64) ?ipc.Handle {
|
||||
std.log.info("no device manager to hello", .{});
|
||||
return null;
|
||||
};
|
||||
return helloOn(manager, role, device_id, serving, want_channel);
|
||||
}
|
||||
|
||||
/// One hello on an already-open manager handle — the exchange without the
|
||||
/// lookup, so a retry loop never spends a handle-table slot per attempt.
|
||||
/// Public for parties that keep their own manager handle across a long retry
|
||||
/// cadence (fat polls for its volume on a timer).
|
||||
pub fn helloOn(manager: ipc.Handle, role: Role, device_id: u64, serving: ?ipc.Handle, want_channel: bool) ?Exchange {
|
||||
var packet: [device_manager_protocol.message_maximum]u8 = undefined;
|
||||
const framed = device_manager_protocol.Protocol.encodeRequest(
|
||||
.hello,
|
||||
device_id,
|
||||
.{ .role = @intFromEnum(role) },
|
||||
.{ .role = @intFromEnum(role), .wants_channel = @intFromBool(want_channel) },
|
||||
&.{},
|
||||
&packet,
|
||||
) orelse return null;
|
||||
|
||||
var reply: [device_manager_protocol.message_maximum]u8 = undefined;
|
||||
const length = ipc.call(manager, framed, &reply) catch {
|
||||
const answered = ipc.callCap(manager, framed, &reply, serving) catch {
|
||||
std.log.info("hello call failed", .{});
|
||||
return null;
|
||||
};
|
||||
const status = envelope.statusOf(reply[0..length]) orelse {
|
||||
const status = envelope.statusOf(reply[0..answered.len]) orelse {
|
||||
if (answered.cap) |stray| _ = ipc.close(stray); // never keep what we cannot read
|
||||
std.log.info("hello answered nothing readable", .{});
|
||||
return null;
|
||||
};
|
||||
if (status.status != 0) {
|
||||
if (answered.cap) |stray| _ = ipc.close(stray);
|
||||
std.log.info("hello refused", .{});
|
||||
return null;
|
||||
}
|
||||
std.log.info("hello acknowledged", .{});
|
||||
return manager;
|
||||
return .{ .manager = manager, .channel = answered.cap };
|
||||
}
|
||||
|
||||
+14
-16
@@ -5,12 +5,16 @@
|
||||
//! service and `device.zig` over the raw device calls.
|
||||
//!
|
||||
//! A class driver, spawned with its interface's assigned device id as argv[1]:
|
||||
//! if (device_manager.hello(.device, id) == null) return; // meet the spawn deadline
|
||||
//! var device = usb.open(id) orelse return; // open + get its endpoints
|
||||
//! const bus = device_manager.helloForChannel(.device, id) orelse return;
|
||||
//! var device = usb.open(bus, id) orelse return; // open + get its endpoints
|
||||
//! _ = device.controlOut(usb_abi.setProtocol(...));// class requests, descriptors
|
||||
//! _ = device.subscribeInterrupt(address, length); // reports arrive asynchronously
|
||||
//! while (true) { ... ipc.replyWait(device.endpoint, ...) ... } // its own loop
|
||||
//!
|
||||
//! The bus channel arrives from the device manager's hello — routed to THIS
|
||||
//! device's controller by lineage — never from a registry name; several bus
|
||||
//! processes provide this contract on a multi-controller machine.
|
||||
//!
|
||||
//! Reports are delivered to `device.endpoint` as asynchronous `interrupt_report`
|
||||
//! event packets, decoded with `reportOf` (the class driver runs a bare `replyWait`
|
||||
//! loop to read them, because the service harness drops buffered-message payloads
|
||||
@@ -21,10 +25,8 @@
|
||||
//! a control transfer's data stage in the tail.
|
||||
|
||||
const std = @import("std");
|
||||
const channel = @import("channel");
|
||||
const envelope = @import("envelope");
|
||||
const ipc = @import("ipc");
|
||||
const time = @import("time");
|
||||
const usb_transfer_protocol = @import("usb-transfer-protocol");
|
||||
|
||||
const Protocol = usb_transfer_protocol.Protocol;
|
||||
@@ -157,18 +159,14 @@ pub fn reportOf(packet: []const u8) ?InterruptReport {
|
||||
return Protocol.decodeEvent(.interrupt_report, packet);
|
||||
}
|
||||
|
||||
/// Open `/protocol/usb-transfer` and, on that channel, open the device with the
|
||||
/// assigned id, handing over a freshly created endpoint for asynchronous interrupt
|
||||
/// reports. Retries while the bus is still coming up (a class driver races the bus
|
||||
/// driver at boot). Two opens, deliberately: the first names the contract, the
|
||||
/// second names an object within it.
|
||||
pub fn open(device_id: u64) ?Device {
|
||||
var attempts: usize = 0;
|
||||
const bus = while (attempts < 100) : (attempts += 1) {
|
||||
if (channel.openEndpoint("usb-transfer")) |handle| break handle;
|
||||
time.sleepMillis(20);
|
||||
} else return null;
|
||||
|
||||
/// On `bus` — the channel to this device's own controller, handed to the class
|
||||
/// driver by the device manager's hello (establishment by lineage,
|
||||
/// communication.md "Establishment: two planes") — open the device with the
|
||||
/// assigned id, handing over a freshly created endpoint for asynchronous
|
||||
/// interrupt reports. The channel is never found by name: one machine carries
|
||||
/// several controllers, several processes provide this contract, and only the
|
||||
/// manager knows which one reported this device.
|
||||
pub fn open(bus: ipc.Handle, device_id: u64) ?Device {
|
||||
const endpoint = ipc.createIpcEndpoint() orelse return null;
|
||||
// The assigned device id is the target: it is what the caller has before a
|
||||
// token exists, and the token the reply hands back addresses every packet
|
||||
|
||||
@@ -182,6 +182,22 @@ pub fn spawnWithArguments(name: []const u8, arguments: []const []const u8) ?u32
|
||||
/// `ipc.replyWait` as a child-exit badge (`ipc.Received.isChildExit`/`childProcessId`), so
|
||||
/// one endpoint can supervise many children. Returns the child's process id, or null.
|
||||
pub fn spawnSupervised(name: []const u8, arguments: []const []const u8, exit_endpoint: ?usize) ?u32 {
|
||||
return spawnSupervisedWithDevice(name, arguments, exit_endpoint, abi.no_device);
|
||||
}
|
||||
|
||||
/// Spawn a supervised child **and give it a device you hold**, in one call.
|
||||
///
|
||||
/// The device manager's path: it holds the hardware and the driver it starts must have
|
||||
/// it. Fusing the handover into the spawn is what removes the window a separate
|
||||
/// transfer would leave — the child cannot run without its device, because it does not
|
||||
/// exist until it holds it (docs/os-development/device-authority.md). The kernel checks
|
||||
/// only that the device is the caller's to give.
|
||||
pub fn spawnSupervisedWithDevice(
|
||||
name: []const u8,
|
||||
arguments: []const []const u8,
|
||||
exit_endpoint: ?usize,
|
||||
device: u64,
|
||||
) ?u32 {
|
||||
var blob: [256]u8 = undefined;
|
||||
var len: usize = 0;
|
||||
for (arguments, 0..) |argument, i| {
|
||||
@@ -194,7 +210,7 @@ pub fn spawnSupervised(name: []const u8, arguments: []const []const u8, exit_end
|
||||
@memcpy(blob[len..][0..argument.len], argument);
|
||||
len += argument.len;
|
||||
}
|
||||
const r = sc.systemCall5(.system_spawn, @intFromPtr(name.ptr), name.len, if (len == 0) 0 else @intFromPtr(&blob), len, exit_endpoint orelse abi.no_cap);
|
||||
const r = sc.systemCall6(.system_spawn, @intFromPtr(name.ptr), name.len, if (len == 0) 0 else @intFromPtr(&blob), len, exit_endpoint orelse abi.no_cap, device);
|
||||
if (r > ~@as(usize, 0) - 4095) return null; // a wrapped -errno
|
||||
return @intCast(r);
|
||||
}
|
||||
|
||||
@@ -192,6 +192,14 @@ pub fn Subscribers(comptime Protocol: type, comptime Context: type) type {
|
||||
return false;
|
||||
}
|
||||
|
||||
/// A provider's own handler kept this turn's capability (stored it
|
||||
/// somewhere with a lifetime beyond the turn) — the same claim the
|
||||
/// reserved `subscribe` makes for its slot table. Composes with it:
|
||||
/// one flag, one `take()`, whoever claims first wins the turn.
|
||||
pub fn claimArrival() void {
|
||||
claimed = true;
|
||||
}
|
||||
|
||||
/// Answer one received packet, with the reserved `subscribe` and
|
||||
/// `unsubscribe` verbs already wired — a provider that leaves those two
|
||||
/// handlers null (every provider should) gets the harness's. The turn's
|
||||
@@ -296,6 +304,22 @@ pub fn Subscribers(comptime Protocol: type, comptime Context: type) type {
|
||||
/// the clean exit the supervisor reads as `ExitReason.exited`.
|
||||
/// `maximum_message` sizes the receive and reply buffers (a service passes its
|
||||
/// protocol's message maximum).
|
||||
/// The capability the current turn's handler nominates to ride out with its
|
||||
/// reply — init's registry idiom (`pending_capability`), lifted into the
|
||||
/// harness so any service can answer an establishment request with a channel
|
||||
/// (communication.md "Establishment: two planes"). Consumed by the loop at the
|
||||
/// very next `replyWait`, which is the reply this turn owes; null is the
|
||||
/// untouched common path. The kernel shares the endpoint as a refcounted copy,
|
||||
/// so the nominating service keeps its own handle.
|
||||
var pending_reply_capability: ?ipc.Handle = null;
|
||||
|
||||
/// Called from inside an `on_message` handler: send `handle` with this turn's
|
||||
/// reply. One capability per turn — the last nomination wins, matching the
|
||||
/// transport (a reply carries at most one).
|
||||
pub fn replyWithCapability(handle: ipc.Handle) void {
|
||||
pending_reply_capability = handle;
|
||||
}
|
||||
|
||||
pub fn run(comptime maximum_message: usize, callbacks: Callbacks) void {
|
||||
const endpoint = ipc.createIpcEndpoint() orelse return;
|
||||
if (callbacks.service) |name| {
|
||||
@@ -313,7 +337,12 @@ pub fn run(comptime maximum_message: usize, callbacks: Callbacks) void {
|
||||
var reply_len: usize = 0;
|
||||
var receive: [maximum_message]u8 = undefined;
|
||||
while (true) {
|
||||
const got = ipc.replyWait(endpoint, reply_buffer[0..reply_len], &receive, null);
|
||||
// The reply going out is the one the just-run handler wrote, so the
|
||||
// capability it nominated (if any) rides this exact replyWait and is
|
||||
// reset before the next turn can see a stale one.
|
||||
const reply_capability = pending_reply_capability;
|
||||
pending_reply_capability = null;
|
||||
const got = ipc.replyWait(endpoint, reply_buffer[0..reply_len], &receive, reply_capability);
|
||||
// Whatever capability came with this turn is the turn's, and the turn
|
||||
// closes it unless a callback claims it (`ipc.Arrival`). Structural
|
||||
// rather than a close per branch, because the branches are exactly what
|
||||
|
||||
@@ -65,3 +65,19 @@ pub inline fn systemCall5(n: SystemCall, a0: usize, a1: usize, a2: usize, a3: us
|
||||
[a4] "{r8}" (a4),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
}
|
||||
|
||||
/// The sixth and last argument register. `syscall` clobbers rcx, so r10 stands in for
|
||||
/// it and r9 is the end of the line — a call needing a seventh would have to pass a
|
||||
/// struct instead.
|
||||
pub inline fn systemCall6(n: SystemCall, a0: usize, a1: usize, a2: usize, a3: usize, a4: usize, a5: usize) usize {
|
||||
return asm volatile ("syscall"
|
||||
: [ret] "={rax}" (-> usize),
|
||||
: [n] "{rax}" (@intFromEnum(n)),
|
||||
[a0] "{rdi}" (a0),
|
||||
[a1] "{rsi}" (a1),
|
||||
[a2] "{rdx}" (a2),
|
||||
[a3] "{r10}" (a3),
|
||||
[a4] "{r8}" (a4),
|
||||
[a5] "{r9}" (a5),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
}
|
||||
|
||||
@@ -63,6 +63,11 @@ pub const Role = enum(u8) {
|
||||
bus = 1,
|
||||
/// Serves one device, reached through a bus's transfer protocol.
|
||||
device = 2,
|
||||
/// Not a spawned driver at all: a party asking for the channel of the
|
||||
/// driver BOUND TO the target device (a service consuming a driver-layer
|
||||
/// contract — fat asking for its volume's block provider). No deadline, no
|
||||
/// state, no driver entry; just establishment by lineage.
|
||||
consumer = 3,
|
||||
};
|
||||
|
||||
// --- the per-operation request parts ----------------------------------------
|
||||
@@ -79,7 +84,13 @@ pub const Role = enum(u8) {
|
||||
pub const Hello = extern struct {
|
||||
/// A `Role` value.
|
||||
role: u8,
|
||||
_padding: u8 = 0,
|
||||
/// 1 = the caller asks the reply to carry a capability to the channel of
|
||||
/// the driver that provides the caller's device — the target's reporter;
|
||||
/// establishment by lineage (communication.md "Establishment: two
|
||||
/// planes"). 0 = a plain handshake, and the reply carries nothing — which
|
||||
/// keeps every caller that never reads a reply capability from leaking
|
||||
/// one. Was padding, so old callers wire-compatibly say 0.
|
||||
wants_channel: u8 = 0,
|
||||
/// The protocol version this driver was built against (`version`).
|
||||
version: u16 = version,
|
||||
};
|
||||
@@ -154,10 +165,18 @@ pub const ChildEntry = extern struct {
|
||||
parent: u64,
|
||||
bus_address: u64,
|
||||
identity: u64,
|
||||
/// The kernel device id this child was registered as (`no_device` for an
|
||||
/// unregistered leaf) — what a `.consumer` hello names as its target to be
|
||||
/// routed to the driver bound to this child.
|
||||
device_id: u64,
|
||||
};
|
||||
|
||||
/// How many `ChildEntry` records one `enumerate` reply can carry. Paging joins
|
||||
/// the protocol if a tree ever outgrows one packet.
|
||||
/// How many `ChildEntry` records one `enumerate` reply can carry. The verb is
|
||||
/// PAGED: the request's `Header.target` is the start index (skip that many
|
||||
/// known children), and a short or empty page means the tree is exhausted — a
|
||||
/// real tree outgrew one packet the day ACPI reported a dozen nodes before
|
||||
/// the first USB child, and an unpaged reply silently truncated exactly the
|
||||
/// entries a consumer was looking for.
|
||||
pub const entries_per_reply: usize = (envelope.packet_maximum - envelope.prefix_size) / @sizeOf(ChildEntry);
|
||||
|
||||
pub const Protocol = envelope.Define(.{
|
||||
|
||||
+47
-3
@@ -43,12 +43,12 @@ pub const SystemCall = enum(u64) {
|
||||
ipc_call = 9, // ipc_call(h, message, len, reply, cap) -> reply_len: send + block for reply
|
||||
ipc_reply_wait = 10, // ipc_reply_wait(h, reply, len, receive, cap) -> receive_len (+badge in rdx)
|
||||
device_enumerate = 11, // device_enumerate(buffer, maximum) -> count: snapshot the device table
|
||||
device_claim = 12, // device_claim(id) -> ok: take exclusive ownership of a device
|
||||
device_claim = 12, // device_claim(id) -> 0/-errno: take exclusive ownership of a device (-ENODEV no such id, -EBUSY someone owns it, -ECONFINE the IOMMU would not confine it)
|
||||
mmio_map = 13, // mmio_map(id, resource_index) -> virtual_address: map a claimed device's MMIO into this address space
|
||||
irq_bind = 14, // irq_bind(id, resource_index, endpoint): deliver a device IRQ as an IPC notification
|
||||
irq_ack = 15, // irq_ack(id, resource_index): re-arm a bound IRQ after servicing it
|
||||
device_register = 16, // device_register(parent_id, descriptor) -> id: publish a child of a device you claimed
|
||||
system_spawn = 17, // system_spawn(name_ptr, name_len, arguments_ptr, arguments_len, exit_endpoint) -> child process id: start a named initial-ramdisk binary as a new ring-3 process
|
||||
device_register = 16, // device_register(parent_id, descriptor) -> id/-errno: publish a child of a device you claimed (-ENOSPC table full, -ECHILDREN parent full, -ERANGE resource escapes the parent, -ENODEV/-EPERM bad parent, -E2BIG too many resources)
|
||||
system_spawn = 17, // system_spawn(name_ptr, name_len, arguments_ptr, arguments_len, exit_endpoint, device) -> child process id: start a named initial-ramdisk binary as a new ring-3 process. `device` (or `no_device`) is a device the caller holds and gives to the child, atomically — the child never runs without it
|
||||
dma_alloc = 18, // dma_alloc(len, flags) -> virtual_address (rax), physical_address (rdx): contiguous, pinned, uncacheable DMA memory
|
||||
dma_free = 19, // dma_free(virtual_address, len) -> 0: release a prior dma_alloc
|
||||
msi_bind = 20, // msi_bind(device_id, endpoint) -> address (rax), data (rdx): a per-device MSI vector for a claimed device
|
||||
@@ -85,9 +85,53 @@ pub const SystemCall = enum(u64) {
|
||||
dma_bind = 51, // dma_bind(device_id, region_handle) -> 0/-errno: map a DMA-region capability into the claimed device's IOMMU domain (idempotent). The caller must own the device and hold the handle
|
||||
dma_unbind = 52, // dma_unbind(device_id, region_handle) -> 0/-errno: unmap a previously bound region from the device's domain and invalidate
|
||||
handle_close = 53, // handle_close(handle) -> 0/-errno: drop one capability handle and free its table slot (endpoints, shared-memory, DMA regions)
|
||||
device_transfer = 54, // device_transfer(device_id, task_id) -> 0/-errno: give a device you hold to another task. A MOVE, not a copy — a claim is exclusive (-ENODEV no such device, -EPERM you do not hold it, -ESRCH no such task)
|
||||
_,
|
||||
};
|
||||
|
||||
/// **The errno space** — the one vocabulary of refusal, returned as `-value` in the
|
||||
/// system_call result register and echoed by user-space providers in a reply status.
|
||||
/// Canonical here because it crosses the kernel↔user boundary in both directions:
|
||||
/// the kernel restates these in system/kernel/ipc-synchronous.zig, and the envelope
|
||||
/// restates the provider-facing subset in library/protocol/envelope/envelope.zig
|
||||
/// (which cannot import this module — the `protocol` package deliberately has no
|
||||
/// dependencies, so a comptime cross-check in library/device/driver/driver.zig and
|
||||
/// system/kernel/tests.zig holds the two halves together).
|
||||
///
|
||||
/// The rule these serve: **a refusal must say which rule refused it.** A caller that
|
||||
/// gets one number for five different reasons cannot report, retry or route around
|
||||
/// any of them — see docs/fixed-bounds-audit.md, where a bare -1 turned "this bus is
|
||||
/// at its child cap" into a machine that booted with no USB and no storage, and cost
|
||||
/// a debugging session to tell apart from four other causes.
|
||||
///
|
||||
/// Values are stable: `failed()` in the runtime treats the top 4096 return values as
|
||||
/// errors (the Linux convention), so anything here must stay well inside 1..4095.
|
||||
pub const EBADF: i64 = 1; // bad handle
|
||||
pub const E2BIG: i64 = 2; // argument exceeds its maximum (a message, a descriptor's resource count)
|
||||
pub const EFAULT: i64 = 3; // buffer unmapped / outside the user half
|
||||
pub const ENOENT: i64 = 4; // no such name
|
||||
pub const ENOSPC: i64 = 5; // a kernel table is full (handles, devices)
|
||||
pub const ENOMEM: i64 = 6; // out of memory
|
||||
pub const EPEER: i64 = 7; // peer died before replying (its process exited or was killed)
|
||||
pub const ESRCH: i64 = 8; // no such process (process_kill of an unknown/dead id)
|
||||
pub const EPERM: i64 = 9; // not permitted (the caller is not the owner/supervisor)
|
||||
pub const ENOSYS: i64 = 10; // this protocol has no such operation
|
||||
pub const EPROTO: i64 = 11; // malformed packet: shorter than the verb it names
|
||||
pub const EBUSY: i64 = 12; // the thing asked for is held by someone still alive
|
||||
pub const ENODEV: i64 = 13; // no such device id
|
||||
pub const ECHILDREN: i64 = 14; // this parent already holds as many children as it can
|
||||
pub const ERANGE: i64 = 15; // a resource escapes the window it must fall inside
|
||||
pub const ECONFINE: i64 = 16; // the device could not be placed under IOMMU translation
|
||||
|
||||
/// The highest errno defined above. A cheap guard for anyone switching over the
|
||||
/// space, and the number to bump when adding one.
|
||||
pub const errno_maximum: i64 = 16;
|
||||
|
||||
/// `system_spawn`'s `device` argument when the child is given no device — every
|
||||
/// caller but the device manager. Matches `device-manager-protocol.no_device`, which
|
||||
/// is the same sentinel one layer up.
|
||||
pub const no_device: u64 = ~@as(u64, 0);
|
||||
|
||||
/// `futex_wait` return codes (in rax).
|
||||
pub const futex_woken: u64 = 0; // woken by a futex_wake
|
||||
pub const futex_mismatch: u64 = 1; // *addr != expected on entry; the caller did not block
|
||||
|
||||
@@ -63,9 +63,11 @@
|
||||
/system/services/discovery, /system/services/device-manager, bind, power
|
||||
|
||||
# --- the drivers, which the device manager spawns ---------------------------
|
||||
# usb-transfer and block have NO bind rows: several processes provide each (one
|
||||
# per controller, one per volume), so neither is ever a registry name —
|
||||
# consumers get their provider's channel from the device manager's hello,
|
||||
# routed by lineage (communication.md "Establishment: two planes").
|
||||
/system/drivers/ps2-bus, /system/services/device-manager, bind, ps2-bus
|
||||
/system/drivers/usb-xhci-bus, /system/services/device-manager, bind, usb-transfer
|
||||
/system/drivers/usb-storage, /system/services/device-manager, bind, block
|
||||
/system/drivers/virtio-gpu, /system/services/device-manager, bind, scanout
|
||||
|
||||
# --- the same providers when the kernel test harness starts them directly ---
|
||||
@@ -92,11 +94,12 @@
|
||||
# ============================================================================
|
||||
|
||||
# --- init's own services ----------------------------------------------------
|
||||
# fat reaches the block device behind the volume it mounts; the compositor
|
||||
# reaches the scanout its driver announced, its own endpoint (the mouse-listener
|
||||
# thread opens /protocol/display like any other client — threads share no
|
||||
# handles), and the input stream that moves the cursor.
|
||||
/system/services/fat, /system/services/init, open, block
|
||||
# fat reaches the device manager to be routed to its volume's block provider
|
||||
# (block is not a name — see the bind section); the compositor reaches the
|
||||
# scanout its driver announced, its own endpoint (the mouse-listener thread
|
||||
# opens /protocol/display like any other client — threads share no handles),
|
||||
# and the input stream that moves the cursor.
|
||||
/system/services/fat, /system/services/init, open, device-manager
|
||||
/system/services/display, /system/services/init, open, scanout
|
||||
/system/services/display, /system/services/init, open, display
|
||||
/system/services/display, /system/services/init, open, input
|
||||
@@ -116,10 +119,7 @@
|
||||
# to the compositor.
|
||||
/system/drivers/*, /system/services/device-manager, open, device-manager
|
||||
/system/services/discovery, /system/services/device-manager, open, device-manager
|
||||
/system/drivers/usb-storage, /system/services/device-manager, open, usb-transfer
|
||||
/system/drivers/usb-hid-keyboard, /system/services/device-manager, open, usb-transfer
|
||||
/system/drivers/usb-hid-keyboard, /system/services/device-manager, open, input
|
||||
/system/drivers/usb-hid-mouse, /system/services/device-manager, open, usb-transfer
|
||||
/system/drivers/usb-hid-mouse, /system/services/device-manager, open, input
|
||||
/system/drivers/virtio-gpu, /system/services/device-manager, open, display
|
||||
|
||||
|
||||
|
Can't render this file because it contains an unexpected character in line 12 and column 15.
|
@@ -44,6 +44,10 @@ var ecam_physical: u64 = 0;
|
||||
var start_bus: u64 = 0;
|
||||
var bus_count: u64 = 0;
|
||||
var manager_handle: ipc.Handle = 0;
|
||||
/// Functions this scan discovered but could not publish. Counted so the end of the
|
||||
/// scan can reconcile "found" against "registered" — a scan that silently returns a
|
||||
/// subset of the machine is the failure this driver is most able to hide.
|
||||
var refused: u32 = 0;
|
||||
|
||||
/// One aligned 32-bit read from a function's configuration space.
|
||||
fn configRead(bus: u64, dev: u64, function: u64, offset: u64) u32 {
|
||||
@@ -71,13 +75,22 @@ fn configWrite16(bus: u64, dev: u64, function: u64, offset: u64, value: u16) voi
|
||||
configWrite(bus, dev, function, aligned, (word & ~mask) | (@as(u32, value) << shift));
|
||||
}
|
||||
|
||||
/// Claim the bridge, map the ECAM, hello the manager, then scan.
|
||||
/// Hello the manager (which is where the bridge arrives), map the ECAM, then scan.
|
||||
fn initialise(endpoint: ipc.Handle) bool {
|
||||
_ = endpoint;
|
||||
if (!device.claim(bridge_id)) {
|
||||
std.log.info("unable to claim bridge device {d}", .{bridge_id});
|
||||
// **The handshake first, because it is where the device arrives.** This driver
|
||||
// used to claim `bridge_id` here — first-come-first-served, so the manager's
|
||||
// matching was advisory and any process could have claimed the bridge by naming
|
||||
// the same id. The manager now holds it and transfers it in the hello reply
|
||||
// (docs/os-development/device-authority.md). `hello` is synchronous, so the
|
||||
// transfer has completed by the time this returns.
|
||||
//
|
||||
// Keep the manager handle to report children through; a supervised bus that
|
||||
// cannot reach its manager has nothing to serve.
|
||||
manager_handle = device_manager.hello(.bus, bridge_id) orelse {
|
||||
std.log.info("no hello with the device manager; bridge {d} not delegated", .{bridge_id});
|
||||
return false;
|
||||
}
|
||||
};
|
||||
const buffer = memory.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||
_ = logging.write("/system/drivers/pci-bus: out of memory\n");
|
||||
return false;
|
||||
@@ -109,11 +122,6 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
return false;
|
||||
};
|
||||
|
||||
// The handshake (role: bus — we enumerate PCI and report the functions we
|
||||
// find), then the scan. Keep the manager handle to report children through;
|
||||
// a supervised bus that cannot reach its manager has nothing to serve.
|
||||
manager_handle = device_manager.hello(.bus, bridge_id) orelse return false;
|
||||
|
||||
scan();
|
||||
return true;
|
||||
}
|
||||
@@ -141,7 +149,13 @@ fn scan() void {
|
||||
}
|
||||
}
|
||||
}
|
||||
std.log.info("{d} functions found", .{found});
|
||||
// Reconcile: "found" alone reads as success even when most of the machine was
|
||||
// refused. If the two disagree, say so at a level that survives a scrollback.
|
||||
if (refused == 0) {
|
||||
std.log.info("{d} functions found, all registered", .{found});
|
||||
} else {
|
||||
std.log.warn("{d} functions found, {d} REFUSED — {d} registered", .{ found, refused, found - refused });
|
||||
}
|
||||
}
|
||||
|
||||
/// Register one function under the bridge and report it to the manager. The
|
||||
@@ -225,8 +239,12 @@ fn registerAndReport(bus: u64, dev: u64, function: u64, class_triple: u32) void
|
||||
// subsystem are read. Every discovered function is logged, matched or not.
|
||||
logFunction(bus, dev, function, class_triple, descriptor.vendor, descriptor.device, descriptor.subsystem);
|
||||
|
||||
const registered = device.register(bridge_id, &descriptor) orelse {
|
||||
std.log.info("register refused for {d}:{d}.{d}", .{ bus, dev, function });
|
||||
// Name the rule that refused. `ParentFull` and `TableFull` are different ceilings
|
||||
// with different fixes, and `NotContained` is not a ceiling at all — it means the
|
||||
// BAR escaped the bridge's own window (docs/fixed-bounds-audit.md).
|
||||
const registered = device.register(bridge_id, &descriptor) catch |e| {
|
||||
refused += 1;
|
||||
std.log.warn("register refused for {d}:{d}.{d}: {s}", .{ bus, dev, function, @errorName(e) });
|
||||
return;
|
||||
};
|
||||
// The registered device id is the packet's target — the manager's object
|
||||
|
||||
@@ -114,10 +114,9 @@ pub fn main() void {
|
||||
_ = logging.write("/system/drivers/ps2-bus: found PS/2 controller\n");
|
||||
_ = logging.write("/system/drivers/ps2-bus: initializing controller\n");
|
||||
|
||||
if (!device.claim(controller_device_descriptor.id)) {
|
||||
_ = logging.write("/system/drivers/ps2-bus: unable to claim controller \n");
|
||||
return;
|
||||
}
|
||||
// The controller node arrived with the spawn — the manager holds the hardware
|
||||
// and names it in the call that creates this process
|
||||
// (docs/os-development/device-authority.md). Nothing to claim.
|
||||
|
||||
const controller = ps2.Controller.init(controller_device_descriptor) orelse {
|
||||
_ = logging.write("/system/drivers/ps2-bus: controller is missing its IO ports\n");
|
||||
@@ -255,14 +254,21 @@ pub fn main() void {
|
||||
if (port_device_types[@intFromEnum(ps2.Port.two)] != null) {
|
||||
if (ps2.findMouseDescriptor(buffer)) |descriptor| {
|
||||
if (findInterruptResourceIndex(descriptor)) |auxiliary_index| {
|
||||
if (device.claim(descriptor.id) and device.irqBind(descriptor.id, auxiliary_index, endpoint)) {
|
||||
// The mouse node is a *second* device for this one instance — the
|
||||
// 8042 is one controller with two ports, so it cannot be split across
|
||||
// two processes. It is transferred to us after the spawn, which is
|
||||
// safe because we only reach it here, long after the controller
|
||||
// handshakes and identify. irq_bind is the proof we hold it: it is
|
||||
// gated on ownership, so a failure here means the handover has not
|
||||
// landed rather than a hardware problem.
|
||||
if (device.irqBind(descriptor.id, auxiliary_index, endpoint)) {
|
||||
maybe_auxiliary_interrupt = .{
|
||||
.device_id = descriptor.id,
|
||||
.interrupt_index = auxiliary_index,
|
||||
.gsi = descriptor.resources[auxiliary_index].start,
|
||||
};
|
||||
} else {
|
||||
_ = logging.write("/system/drivers/ps2-bus: auxiliary irq_bind failed\n");
|
||||
_ = logging.write("/system/drivers/ps2-bus: auxiliary irq_bind failed (mouse node not delegated?)\n");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -75,9 +75,10 @@ pub fn main(init: process.Init) void {
|
||||
};
|
||||
const layout = xkb.byName(init.arguments.get(2) orelse "us") orelse xkb.us;
|
||||
|
||||
// Hello the manager first (meet the spawn deadline), then open the device.
|
||||
if (device_manager.hello(.device, device_id) == null) return;
|
||||
var device = usb.open(device_id) orelse {
|
||||
// The hello both meets the spawn deadline and brings back the channel to
|
||||
// THIS device's controller — routed by lineage, never found by name.
|
||||
const bus = device_manager.helloForChannel(.device, device_id, null) orelse return;
|
||||
var device = usb.open(bus, device_id) orelse {
|
||||
std.log.info("could not open device {d}", .{device_id});
|
||||
return;
|
||||
};
|
||||
|
||||
@@ -41,8 +41,10 @@ pub fn main(init: process.Init) void {
|
||||
return;
|
||||
};
|
||||
|
||||
if (device_manager.hello(.device, device_id) == null) return;
|
||||
var device = usb.open(device_id) orelse {
|
||||
// The hello both meets the spawn deadline and brings back the channel to
|
||||
// THIS device's controller — routed by lineage, never found by name.
|
||||
const bus = device_manager.helloForChannel(.device, device_id, null) orelse return;
|
||||
var device = usb.open(bus, device_id) orelse {
|
||||
std.log.info("could not open device {d}", .{device_id});
|
||||
return;
|
||||
};
|
||||
|
||||
@@ -82,9 +82,14 @@ fn transact(cdb: []const u8, direction_in: bool, data_physical: u64, data_length
|
||||
var bring_up_failed = false;
|
||||
|
||||
fn initialise(endpoint: ipc.Handle) bool {
|
||||
_ = endpoint;
|
||||
if (device_manager.hello(.device, device_id) == null) return false;
|
||||
device = usb.open(device_id) orelse {
|
||||
// One hello, both directions: the block-serving endpoint goes UP (the
|
||||
// manager routes fat's consumer hello here — this driver serves one
|
||||
// volume, one process per stick, so `block` is never a registry name),
|
||||
// and the channel to THIS device's controller comes DOWN, routed by
|
||||
// lineage. A second stick used to die silently on the exclusive bind;
|
||||
// now every instance is reachable through its lineage.
|
||||
const bus = device_manager.helloForChannel(.device, device_id, endpoint) orelse return false;
|
||||
device = usb.open(bus, device_id) orelse {
|
||||
std.log.info("could not open device {d}", .{device_id});
|
||||
return false;
|
||||
};
|
||||
@@ -222,8 +227,10 @@ pub fn main(init: process.Init) void {
|
||||
std.log.info("malformed device id '{s}'", .{argument});
|
||||
return;
|
||||
};
|
||||
// No `.service` name: several instances provide the block contract (one
|
||||
// per stick), so consumers are routed here by the device manager's
|
||||
// lineage, never by a registry bind — the endpoint goes up in the hello.
|
||||
service.run(block_protocol.message_maximum, .{
|
||||
.service = "block",
|
||||
.init = initialise,
|
||||
.on_message = onMessage,
|
||||
});
|
||||
|
||||
@@ -10,10 +10,10 @@ pub fn build(b: *std.Build) void {
|
||||
.name = "usb-xhci-bus",
|
||||
.root_source_file = b.path("usb-xhci-bus.zig"),
|
||||
.imports = &.{
|
||||
"channel", "device-manager-protocol", "driver", "envelope",
|
||||
"input-client", "ipc", "logging", "memory",
|
||||
"mmio", "pci", "process", "service",
|
||||
"time", "usb-abi", "usb-ids", "usb-transfer-protocol",
|
||||
"device-manager-protocol", "driver", "envelope", "ipc",
|
||||
"logging", "memory", "mmio", "pci",
|
||||
"process", "service", "time", "usb-abi",
|
||||
"usb-ids", "usb-transfer-protocol",
|
||||
},
|
||||
});
|
||||
b.installArtifact(exe);
|
||||
|
||||
@@ -15,12 +15,10 @@
|
||||
|
||||
const std = @import("std");
|
||||
const device = @import("driver");
|
||||
const channel = @import("channel");
|
||||
const ipc = @import("ipc");
|
||||
const process = @import("process");
|
||||
const service = @import("service");
|
||||
const time = @import("time");
|
||||
const input = @import("input-client");
|
||||
const device_manager = @import("driver");
|
||||
const memory = @import("memory");
|
||||
const logging = @import("logging");
|
||||
@@ -158,25 +156,27 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
// its device open, or its restarted instance could never claim it back.
|
||||
_ = process.subscribeExits(endpoint);
|
||||
|
||||
// The transfer contract, bound by hand rather than through the harness's
|
||||
// `.service`, because **losing it is not fatal here**. One machine can carry
|
||||
// several xHCI controllers and the driver model spawns one process per
|
||||
// controller, so several processes provide the same contract for different
|
||||
// hardware — and `/protocol` holds exactly one name, deliberately (addressing
|
||||
// lives inside the protocol, never in the path). Whoever binds first is the
|
||||
// one clients reach by name; a later instance still owns its controller,
|
||||
// enumerates its bus, and reports its children to the device manager, so it
|
||||
// keeps running. **Known gap:** a class driver behind a second controller
|
||||
// cannot reach it — the transfer protocol has no controller field for
|
||||
// `target`, and the fix is either one process multiplexing every controller
|
||||
// or the spawner wiring the child's channel (P5), not a second name.
|
||||
if (!channel.bindPatiently("usb-transfer", endpoint))
|
||||
_ = logging.write("/system/drivers/usb-xhci-bus: /protocol/usb-transfer is another controller's; serving mine unnamed\n");
|
||||
|
||||
if (!device.claim(controller_id)) {
|
||||
std.log.info("unable to claim controller device {d}", .{controller_id});
|
||||
// **The handshake comes first, because it is where everything moves.** The
|
||||
// manager holds the controller and transfers it in `onHello`, so by the time
|
||||
// this call returns the device is ours and nothing else could have taken it
|
||||
// (docs/os-development/device-authority.md). And the serving endpoint rides
|
||||
// up with the same call: one machine can carry several xHCI controllers, one
|
||||
// process per controller, so the transfer contract is never a registry name
|
||||
// — `/protocol` holds no instances, deliberately. Class drivers reach *this*
|
||||
// controller because the manager answers their hellos with this endpoint,
|
||||
// routed by lineage (communication.md "Establishment: two planes"). The bind
|
||||
// race this replaces left every device behind a losing controller
|
||||
// unreachable — a real machine's mouse, "could not open device 50".
|
||||
//
|
||||
// `hello` is synchronous, so the transfer has completed before the reply lands —
|
||||
// there is no window between being told yes and holding the thing.
|
||||
//
|
||||
// Keep the handle: the tick's hot-plug dispatch reports through it.
|
||||
const exchanged = device_manager.helloExchange(.bus, controller_id, endpoint, false) orelse {
|
||||
std.log.warn("no hello with the device manager; controller {d} not delegated", .{controller_id});
|
||||
return false;
|
||||
}
|
||||
};
|
||||
manager_handle = exchanged.manager;
|
||||
|
||||
// Fetch our own descriptor back for the controller's resources.
|
||||
const buffer = memory.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||
@@ -204,7 +204,7 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
std.log.info("controller device {d} has no register BAR", .{controller_id});
|
||||
return false;
|
||||
};
|
||||
std.log.info("claimed controller device {d} (registers at 0x{x}, {d} bytes)", .{
|
||||
std.log.info("controller device {d} registers at 0x{x}, {d} bytes", .{
|
||||
controller_id,
|
||||
register_window.start,
|
||||
register_window.len,
|
||||
@@ -228,8 +228,13 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
_ = logging.write("/system/drivers/usb-xhci-bus: controller reset/bring-up failed\n");
|
||||
return false;
|
||||
};
|
||||
std.log.info("controller running ({d} slots, {d}-byte contexts)", .{
|
||||
// Both numbers, because for a long time they disagreed silently: the controller
|
||||
// reported its real slot count and the driver tracked a fixed 8 of them, so a
|
||||
// ninth device — trivially reachable behind a hub — simply did not exist. They
|
||||
// must now be equal, and the suite asserts it.
|
||||
std.log.info("controller running ({d} slots, tracking {d}, {d}-byte contexts)", .{
|
||||
controller.?.max_slots,
|
||||
controller.?.devices.len,
|
||||
controller.?.context_size,
|
||||
});
|
||||
// The proof of life: a No-Op command round-trips the command ring, the event
|
||||
@@ -242,13 +247,7 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
return false;
|
||||
}
|
||||
|
||||
// The handshake (role: bus — we enumerate USB ports and report the devices
|
||||
// behind them), inside the manager's hello deadline. Keep the handle: the
|
||||
// tick's hot-plug dispatch reports through it.
|
||||
const handle = device_manager.hello(.bus, controller_id) orelse return false;
|
||||
manager_handle = handle;
|
||||
|
||||
scanPorts(handle);
|
||||
scanPorts(exchanged.manager);
|
||||
|
||||
// Arm the timer: in polling mode it drains the event ring; in MSI mode it is the
|
||||
// slower port-reconcile/safety-net tick. Re-armed on each tick in onNotification.
|
||||
@@ -522,8 +521,8 @@ fn reportInterface(manager: ipc.Handle, port: u32, interface: library.InterfaceI
|
||||
const hid_text = std.fmt.bufPrint(&hid_buffer, "P{d}I{d}", .{ port, interface.number }) catch "";
|
||||
descriptor.hid_len = hid_text.len;
|
||||
@memcpy(descriptor.hid[0..hid_text.len], hid_text);
|
||||
const registered = device.register(controller_id, &descriptor) orelse {
|
||||
std.log.info("register refused for port {d} interface {d}", .{ port, interface.number });
|
||||
const registered = device.register(controller_id, &descriptor) catch |e| {
|
||||
std.log.warn("register refused for port {d} interface {d}: {s}", .{ port, interface.number, @errorName(e) });
|
||||
return null;
|
||||
};
|
||||
|
||||
|
||||
@@ -221,10 +221,17 @@ fn intervalFor(speed: u32, b_interval: u8) u32 {
|
||||
};
|
||||
}
|
||||
|
||||
// Upper bounds on what one device's active configuration describes. A boot
|
||||
// keyboard or mouse has one interface with one interrupt endpoint; a flash drive
|
||||
// has one interface with two bulk endpoints. Generous for those.
|
||||
pub const max_interfaces = 4;
|
||||
/// bound: endpoints recorded per interface
|
||||
/// decided-by: external
|
||||
/// protects: the fixed endpoint array inside InterfaceInfo
|
||||
/// at-limit: refuse — the surplus endpoint is not recorded and the class driver
|
||||
/// cannot bind it
|
||||
/// observed-by: the class driver failing to find the endpoint it wants
|
||||
///
|
||||
/// Held at 4 because the usb-transfer wire protocol reports exactly
|
||||
/// `max_reported_endpoints` (4) endpoints per interface, so widening this alone
|
||||
/// would change nothing a class driver can see. Lifting it is a protocol change —
|
||||
/// docs/bounds-track-plan.md, out of scope for the unattended run.
|
||||
pub const max_endpoints_per_interface = 4;
|
||||
|
||||
// The endpoint-descriptor facts a class driver needs to talk to an endpoint: its
|
||||
@@ -256,7 +263,19 @@ const ConfiguredEndpoint = struct {
|
||||
dci: u32 = 0,
|
||||
ring: ProducerRing = .{ .region = .{ .virtual = 0, .physical = 0 } },
|
||||
};
|
||||
const max_configured_endpoints = max_interfaces * max_endpoints_per_interface;
|
||||
/// bound: transfer rings configured on one device slot
|
||||
/// decided-by: hardware
|
||||
/// protects: the per-device ring table, sized at compile time
|
||||
/// at-limit: refuse — configureEndpoint returns null and its caller reports it
|
||||
/// observed-by: the class driver's own failure to bind an endpoint
|
||||
///
|
||||
/// The xHCI specification's number rather than ours: a Device Context carries a slot
|
||||
/// context plus at most 31 endpoint contexts, because the Context Entries field that
|
||||
/// addresses them is 5 bits (DCI 1..31, DCI 1 being the default control endpoint).
|
||||
/// A device physically cannot present more. This was
|
||||
/// `max_interfaces * max_endpoints_per_interface` = 16, so it moved whenever either
|
||||
/// of those two guesses moved.
|
||||
const max_configured_endpoints = 31;
|
||||
|
||||
// One addressed USB device behind this controller: its hardware slot, its EP0
|
||||
// (control) transfer ring, the DMA context + bounce buffer the control pipe uses,
|
||||
@@ -277,7 +296,13 @@ pub const Device = struct {
|
||||
device_descriptor: usb_abi.DeviceDescriptor = std.mem.zeroes(usb_abi.DeviceDescriptor),
|
||||
configuration_value: u8 = 0,
|
||||
interface_count: u8 = 0,
|
||||
interfaces: [max_interfaces]InterfaceInfo = [_]InterfaceInfo{.{}} ** max_interfaces,
|
||||
/// The interfaces of the active configuration, allocated at enumeration from the
|
||||
/// count the configuration descriptor actually declares. This was a fixed 4, and
|
||||
/// the fifth interface of a composite device (a headset, a webcam with audio, a
|
||||
/// dock, a multifunction printer) did not merely go missing — see
|
||||
/// parseConfiguration, where its endpoints were appended to interface 3's list.
|
||||
interfaces: []InterfaceInfo = &.{},
|
||||
|
||||
// Transfer rings configured for this device's interrupt/bulk endpoints.
|
||||
endpoint_ring_count: u8 = 0,
|
||||
endpoint_rings: [max_configured_endpoints]ConfiguredEndpoint = [_]ConfiguredEndpoint{.{}} ** max_configured_endpoints,
|
||||
@@ -296,6 +321,15 @@ pub const Device = struct {
|
||||
// Downstream ports with a pending change to service (bit P = port P), set
|
||||
// by the status-change endpoint (and by an initial sweep in setupHub).
|
||||
hub_change_mask: u32 = 0,
|
||||
|
||||
/// Release the interface list. Safe to call twice, and on a device that never
|
||||
/// enumerated — re-enumeration and teardown both come through here.
|
||||
pub fn freeInterfaces(self: *Device) void {
|
||||
if (self.interfaces.len != 0) memory.allocator().free(self.interfaces);
|
||||
self.interfaces = &.{};
|
||||
self.interface_count = 0;
|
||||
}
|
||||
|
||||
};
|
||||
|
||||
// A standing interrupt-IN subscription: the endpoint's ring is kept armed with a
|
||||
@@ -328,9 +362,6 @@ pub const Report = struct {
|
||||
data: [64]u8 = [_]u8{0} ** 64,
|
||||
};
|
||||
|
||||
// How many addressed devices this driver tracks at once. QEMU presents a handful
|
||||
// (a keyboard, a mouse, a storage stick); a fuller machine would grow this.
|
||||
const max_devices = 8;
|
||||
const max_subscriptions = 8;
|
||||
const report_queue_capacity = 16;
|
||||
|
||||
@@ -449,7 +480,14 @@ pub const Controller = struct {
|
||||
device_context_array: memory.DmaRegion,
|
||||
command_ring: ProducerRing,
|
||||
event_ring: EventRing,
|
||||
devices: [max_devices]Device = [_]Device{.{}} ** max_devices,
|
||||
/// One entry per device slot the **controller** says it has (HCSPARAMS1.MaxSlots,
|
||||
/// 1..255), allocated at bring-up. This used to be a fixed 8 with the comment "QEMU
|
||||
/// presents a handful; a fuller machine would grow this" — which is the shape the
|
||||
/// bounds rule exists to stop, since the controller has always reported the real
|
||||
/// number and `op_config` below is already programmed with it. A desktop's keyboard,
|
||||
/// mouse, webcam, headset, hub and two sticks reach 8 without trying, and everything
|
||||
/// past it vanished (behind a hub, without even a log line).
|
||||
devices: []Device,
|
||||
subscriptions: [max_subscriptions]Subscription = [_]Subscription{.{}} ** max_subscriptions,
|
||||
report_queue: [report_queue_capacity]Report = [_]Report{.{}} ** report_queue_capacity,
|
||||
report_count: usize = 0,
|
||||
@@ -582,8 +620,16 @@ pub const Controller = struct {
|
||||
.device_context_array = undefined,
|
||||
.command_ring = undefined,
|
||||
.event_ring = undefined,
|
||||
.devices = &.{},
|
||||
};
|
||||
|
||||
// One tracking slot per slot the controller reports. A controller that claims
|
||||
// no slots cannot address anything, so treat that as a dead controller rather
|
||||
// than allocating nothing and failing mysteriously later.
|
||||
if (self.max_slots == 0) return null;
|
||||
self.devices = memory.allocator().alloc(Device, self.max_slots) catch return null;
|
||||
for (self.devices) |*device| device.* = .{};
|
||||
|
||||
// Wait for the controller to report ready, then halt it if it is running.
|
||||
if (!waitClear(self.operational(op_usbsts), usbsts_controller_not_ready)) return null;
|
||||
if (read32(self.operational(op_usbcmd)) & usbcmd_run != 0) {
|
||||
@@ -790,7 +836,7 @@ pub const Controller = struct {
|
||||
}
|
||||
|
||||
fn allocateDevice(self: *Controller) ?*Device {
|
||||
for (&self.devices) |*device| {
|
||||
for (self.devices) |*device| {
|
||||
if (!device.used) return device;
|
||||
}
|
||||
return null;
|
||||
@@ -897,7 +943,8 @@ pub const Controller = struct {
|
||||
return null;
|
||||
};
|
||||
const device = self.allocateDevice() orelse {
|
||||
std.log.info("port {d} setup: no free device slot", .{port});
|
||||
std.log.warn("port {d} setup: no free device slot", .{port});
|
||||
self.disableSlot(slot_id); // the controller already gave it to us
|
||||
return null;
|
||||
};
|
||||
device.* = .{
|
||||
@@ -1011,7 +1058,7 @@ pub const Controller = struct {
|
||||
/// The next pending (hub, downstream-port) change to service, or null. Clears
|
||||
/// the returned port's bit. Called on the bus tick.
|
||||
pub fn takeHubChange(self: *Controller) ?struct { hub: *Device, port: u16 } {
|
||||
for (&self.devices) |*device| {
|
||||
for (self.devices) |*device| {
|
||||
if (!device.used or !device.is_hub or device.hub_change_mask == 0) continue;
|
||||
const bit: u5 = @intCast(@ctz(device.hub_change_mask));
|
||||
device.hub_change_mask &= ~(@as(u32, 1) << bit);
|
||||
@@ -1084,7 +1131,7 @@ pub const Controller = struct {
|
||||
}
|
||||
|
||||
pub fn deviceOnHubPort(self: *Controller, hub: *Device, port: u16) ?*Device {
|
||||
for (&self.devices) |*device| {
|
||||
for (self.devices) |*device| {
|
||||
if (device.used and device.parent_slot == hub.slot_id and device.parent_port == port) return device;
|
||||
}
|
||||
return null;
|
||||
@@ -1099,7 +1146,11 @@ pub const Controller = struct {
|
||||
std.log.info("hub slot {d} port {d}: Enable Slot failed", .{ hub.slot_id, port });
|
||||
return null;
|
||||
};
|
||||
const device = self.allocateDevice() orelse return null;
|
||||
const device = self.allocateDevice() orelse {
|
||||
std.log.warn("hub slot {d} port {d}: no free device slot", .{ hub.slot_id, port });
|
||||
self.disableSlot(slot_id); // the controller already gave it to us
|
||||
return null;
|
||||
};
|
||||
const child_speed = mapHubPortSpeed(speed);
|
||||
device.* = .{
|
||||
.used = true,
|
||||
@@ -1150,6 +1201,7 @@ pub const Controller = struct {
|
||||
|
||||
fn abandon(self: *Controller, device: *Device) ?*Device {
|
||||
_ = self;
|
||||
device.freeInterfaces();
|
||||
device.used = false;
|
||||
return null;
|
||||
}
|
||||
@@ -1299,12 +1351,25 @@ pub const Controller = struct {
|
||||
const configuration = std.mem.bytesToValue(usb_abi.ConfigurationDescriptor, &header);
|
||||
device.configuration_value = @intFromEnum(configuration.configuration_value);
|
||||
|
||||
// Read the whole block into a local buffer and parse it here (in the bus
|
||||
// driver) so the parse never has to cross the 256-byte IPC boundary.
|
||||
var blob: [512]u8 = undefined;
|
||||
const length = @min(configuration.total_length, blob.len);
|
||||
if (!self.controlTransfer(device, usb_abi.getDescriptor(.configuration, 0, 0, @intCast(length)), blob[0..length], true)) return false;
|
||||
parseConfiguration(device, blob[0..length]);
|
||||
// Read the whole block and parse it here (in the bus driver) so the parse
|
||||
// never has to cross the 256-byte IPC boundary. Sized by the device's own
|
||||
// wTotalLength — the ceiling is then the field's u16, which is the USB
|
||||
// specification's, not ours.
|
||||
//
|
||||
// This was a fixed 512 with an `@min` clamp, which silently truncated: a
|
||||
// composite device routinely exceeds it (a headset is 500-900 bytes, a UVC
|
||||
// webcam 1-3 KB, a multifunction printer 600+), and the interfaces past the
|
||||
// cut simply did not exist — while SET_CONFIGURATION below still configured
|
||||
// the device for all of them. QEMU's boot keyboard, mouse and stick are all
|
||||
// under 100 bytes, which is why it survived.
|
||||
if (configuration.total_length < header.len) return false; // shorter than its own header
|
||||
const blob = memory.allocator().alloc(u8, configuration.total_length) catch return false;
|
||||
defer memory.allocator().free(blob);
|
||||
if (!self.controlTransfer(device, usb_abi.getDescriptor(.configuration, 0, 0, configuration.total_length), blob, true)) return false;
|
||||
// Both numbers, so a truncation can never again be invisible: the block the
|
||||
// device declared, and the bytes actually fetched and parsed. They must match.
|
||||
std.log.info("config block {d} bytes, read {d}", .{ configuration.total_length, blob.len });
|
||||
if (!parseConfiguration(device, blob)) return false;
|
||||
|
||||
// Select the configuration, moving the device to the configured state.
|
||||
if (!self.controlTransfer(device, usb_abi.setConfiguration(configuration.configuration_value), &.{}, false)) return false;
|
||||
@@ -1315,8 +1380,30 @@ pub const Controller = struct {
|
||||
// and the endpoints that follow it. Endpoints belong to the most recent
|
||||
// interface. Unknown descriptor types (HID, class-specific) are skipped by
|
||||
// their length.
|
||||
fn parseConfiguration(device: *Device, blob: []const u8) void {
|
||||
device.interface_count = 0;
|
||||
/// Two passes: count the alternate-setting-0 interfaces the block declares,
|
||||
/// allocate exactly that many, then fill them. False only on an allocation
|
||||
/// failure. The count cannot exceed 255 — `bNumInterfaces` is a u8, so that is
|
||||
/// the USB specification's ceiling and not one of ours.
|
||||
fn parseConfiguration(device: *Device, blob: []const u8) bool {
|
||||
var declared: usize = 0;
|
||||
var count_offset: usize = 0;
|
||||
while (count_offset + 2 <= blob.len) {
|
||||
const length = blob[count_offset];
|
||||
if (length < 2 or count_offset + length > blob.len) break;
|
||||
if (@as(usb_abi.DescriptorType, @enumFromInt(blob[count_offset + 1])) == .interface and
|
||||
length >= @sizeOf(usb_abi.InterfaceDescriptor))
|
||||
{
|
||||
const descriptor = std.mem.bytesToValue(usb_abi.InterfaceDescriptor, blob[count_offset .. count_offset + @sizeOf(usb_abi.InterfaceDescriptor)]);
|
||||
if (@intFromEnum(descriptor.alternate_setting) == 0 and declared < 255) declared += 1;
|
||||
}
|
||||
count_offset += length;
|
||||
}
|
||||
|
||||
device.freeInterfaces();
|
||||
if (declared == 0) return true;
|
||||
device.interfaces = memory.allocator().alloc(InterfaceInfo, declared) catch return false;
|
||||
for (device.interfaces) |*interface| interface.* = .{};
|
||||
|
||||
var current: ?*InterfaceInfo = null;
|
||||
var offset: usize = 0;
|
||||
while (offset + 2 <= blob.len) {
|
||||
@@ -1326,9 +1413,17 @@ pub const Controller = struct {
|
||||
switch (@as(usb_abi.DescriptorType, @enumFromInt(descriptor_type))) {
|
||||
.interface => if (length >= @sizeOf(usb_abi.InterfaceDescriptor)) {
|
||||
const descriptor = std.mem.bytesToValue(usb_abi.InterfaceDescriptor, blob[offset .. offset + @sizeOf(usb_abi.InterfaceDescriptor)]);
|
||||
if (@intFromEnum(descriptor.alternate_setting) != 0) {
|
||||
current = null; // ignore alternate settings for now
|
||||
} else if (device.interface_count < max_interfaces) {
|
||||
// An interface we do not record MUST clear `current`, or the
|
||||
// endpoints that follow it attach to the previous one. The cap
|
||||
// branch used to have no `else`, so a fifth interface's endpoints
|
||||
// were appended to interface 3's array and a class driver bound to
|
||||
// interface 3 could be handed an endpoint belonging to something
|
||||
// else entirely — silently, with a truthful-looking count logged.
|
||||
if (@intFromEnum(descriptor.alternate_setting) != 0 or
|
||||
device.interface_count >= device.interfaces.len)
|
||||
{
|
||||
current = null;
|
||||
} else {
|
||||
const slot = &device.interfaces[device.interface_count];
|
||||
slot.* = .{
|
||||
.number = @intFromEnum(descriptor.interface_number),
|
||||
@@ -1358,13 +1453,14 @@ pub const Controller = struct {
|
||||
}
|
||||
offset += length;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
// --- endpoint configuration + interrupt / bulk transfers ---------------
|
||||
|
||||
/// Find the tracked device and interface an assigned device id belongs to.
|
||||
pub fn findInterface(self: *Controller, device_id: u64) ?struct { device: *Device, interface: *InterfaceInfo } {
|
||||
for (&self.devices) |*device| {
|
||||
for (self.devices) |*device| {
|
||||
if (!device.used) continue;
|
||||
for (device.interfaces[0..device.interface_count]) |*interface| {
|
||||
if (interface.registered_device_id == device_id) return .{ .device = device, .interface = interface };
|
||||
@@ -1633,7 +1729,7 @@ pub const Controller = struct {
|
||||
|
||||
/// The tracked device on `port`, or null.
|
||||
pub fn deviceOnPort(self: *Controller, port: u32) ?*Device {
|
||||
for (&self.devices) |*device| {
|
||||
for (self.devices) |*device| {
|
||||
if (device.used and device.port == port) return device;
|
||||
}
|
||||
return null;
|
||||
@@ -1642,7 +1738,7 @@ pub const Controller = struct {
|
||||
/// The next used device whose parent hub is `hub_slot` and slot id > `after`
|
||||
/// (for recursive teardown when a hub itself disconnects), or null.
|
||||
pub fn nextChildOf(self: *Controller, hub_slot: u8, after: u8) ?*Device {
|
||||
for (&self.devices) |*device| {
|
||||
for (self.devices) |*device| {
|
||||
if (device.used and device.parent_slot == hub_slot and device.slot_id > after) return device;
|
||||
}
|
||||
return null;
|
||||
@@ -1656,15 +1752,27 @@ pub const Controller = struct {
|
||||
for (&self.subscriptions) |*subscription| {
|
||||
if (subscription.active and subscription.slot_id == device.slot_id) subscription.active = false;
|
||||
}
|
||||
self.disableSlot(device.slot_id);
|
||||
device.freeInterfaces();
|
||||
device.used = false;
|
||||
}
|
||||
|
||||
/// Hand a slot back to the controller and clear its context-array entry.
|
||||
///
|
||||
/// Every path that has issued a successful Enable Slot owes this, including the
|
||||
/// ones that then fail to bring the device up. A slot the driver forgets is one
|
||||
/// the controller never reissues, so the loss is permanent for the boot: the two
|
||||
/// setup paths used to return null straight after a failed `allocateDevice`,
|
||||
/// leaking a slot per attempt — and the hub path did it without even a log line.
|
||||
fn disableSlot(self: *Controller, slot_id: u8) void {
|
||||
const physical = self.submitCommand(.{
|
||||
.control = trbControl(.disable_slot, @as(u32, device.slot_id) << 24),
|
||||
.control = trbControl(.disable_slot, @as(u32, slot_id) << 24),
|
||||
});
|
||||
if (self.awaitCommand(physical)) |code| {
|
||||
if (code != @intFromEnum(CompletionCode.success))
|
||||
std.log.info("slot {d}: Disable Slot completion code {d}", .{ device.slot_id, code });
|
||||
} else std.log.info("slot {d}: Disable Slot timed out", .{device.slot_id});
|
||||
std.log.info("slot {d}: Disable Slot completion code {d}", .{ slot_id, code });
|
||||
} else std.log.info("slot {d}: Disable Slot timed out", .{slot_id});
|
||||
const array: [*]volatile u64 = @ptrFromInt(self.device_context_array.virtual);
|
||||
array[device.slot_id] = 0;
|
||||
device.used = false;
|
||||
array[slot_id] = 0;
|
||||
}
|
||||
};
|
||||
|
||||
@@ -200,10 +200,9 @@ fn testPixel(index: u32) u32 {
|
||||
|
||||
fn initialise(endpoint: ipc.Handle) bool {
|
||||
_ = endpoint;
|
||||
if (!device.claim(device_id)) {
|
||||
std.log.info("unable to claim device {d}", .{device_id});
|
||||
return false;
|
||||
}
|
||||
// The device arrived with the spawn — the manager holds it and names it in the call
|
||||
// that creates this process, so it is ours before the first instruction here
|
||||
// (docs/os-development/device-authority.md). Nothing to claim.
|
||||
|
||||
var descriptors: [64]device.DeviceDescriptor = undefined;
|
||||
const total = device.enumerate(&descriptors);
|
||||
|
||||
+82
-40
@@ -615,17 +615,78 @@ var boot_memory_regions: []const boot_handoff.MemoryRegion = &.{};
|
||||
/// I/O-APIC region, the high one from 4 GiB (or the end of RAM above it) to
|
||||
/// the 46-bit line. Coarse, mechanical, and AML-free — available at boot no
|
||||
/// matter what later moved to user space.
|
||||
pub const AddressRange = struct { base: u64, end: u64 };
|
||||
|
||||
/// The largest holes below 4 GiB in a firmware memory map: the ranges the firmware
|
||||
/// described *nothing* in, which is where a PCI BAR may legitimately live. Fills `out`
|
||||
/// (keeping the largest, replacing the smallest held so far), ignores holes shorter
|
||||
/// than `minimum`, and returns how many entries are non-empty; entries the caller
|
||||
/// reads may be empty and are skipped by `end > base`.
|
||||
///
|
||||
/// **It never copies the map, and that is the point.** The firmware chooses how many
|
||||
/// descriptors its map has — commonly 60–200 on a real machine, 15–25 under OVMF. The
|
||||
/// previous version copied the sub-4 GiB entries into a fixed `[64]` array and
|
||||
/// `continue`d past the rest, which did not merely lose them: a region absent from the
|
||||
/// walk is a region this function concludes is *free*, so a large enough map yields an
|
||||
/// "aperture" lying over live RAM. `device_register` containment would then admit a
|
||||
/// child BAR covering kernel memory, and its claimant could `mmio_map` it. A bound
|
||||
/// whose overflow hands out authority is not a limit; the fix is not a bigger array.
|
||||
///
|
||||
/// Pure, allocation-free, and linear-ish in the map (each inner pass consumes at least
|
||||
/// one region, and it runs once at boot).
|
||||
pub fn largestHolesBelow4G(
|
||||
regions: []const boot_handoff.MemoryRegion,
|
||||
minimum: u64,
|
||||
out: []AddressRange,
|
||||
) usize {
|
||||
const limit: u64 = 1 << 32;
|
||||
for (out) |*hole| hole.* = .{ .base = 0, .end = 0 };
|
||||
|
||||
var cursor: u64 = 0;
|
||||
while (cursor < limit) {
|
||||
// Step over every described region covering the cursor. Regions may overlap
|
||||
// and chain, so repeat until the cursor stops moving.
|
||||
var moved = true;
|
||||
while (moved) {
|
||||
moved = false;
|
||||
for (regions) |region| {
|
||||
if (region.base >= limit) continue;
|
||||
const end = @min(region.base + region.pages * 4096, limit);
|
||||
if (region.base <= cursor and end > cursor) {
|
||||
cursor = end;
|
||||
moved = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (cursor >= limit) break;
|
||||
|
||||
// The cursor now sits in a hole; it runs to the next described base, or to
|
||||
// 4 GiB if nothing is described above it.
|
||||
var next: u64 = limit;
|
||||
for (regions) |region| {
|
||||
if (region.base >= limit) continue;
|
||||
if (region.base > cursor and region.base < next) next = region.base;
|
||||
}
|
||||
|
||||
if (next - cursor >= minimum and out.len != 0) {
|
||||
var smallest: usize = 0;
|
||||
for (out, 0..) |hole, i| {
|
||||
if (hole.end - hole.base < out[smallest].end - out[smallest].base) smallest = i;
|
||||
}
|
||||
if (next - cursor > out[smallest].end - out[smallest].base)
|
||||
out[smallest] = .{ .base = cursor, .end = next };
|
||||
}
|
||||
cursor = next;
|
||||
}
|
||||
|
||||
var found: usize = 0;
|
||||
for (out) |hole| {
|
||||
if (hole.end > hole.base) found += 1;
|
||||
}
|
||||
return found;
|
||||
}
|
||||
|
||||
fn addBridgeApertures(bridge: *device_model.Device) void {
|
||||
// Below 4 GiB the described regions are sparse (RAM low, firmware flash
|
||||
// and tables high), so the holes are the *gaps between* them — a single
|
||||
// "after the last region" rule dies on OVMF's flash at the very top.
|
||||
// Sort-merge the described ranges, then keep the three largest gaps
|
||||
// (resource slots are bounded at 8 per device; ECAM + bus range + 3 + the
|
||||
// high aperture fits). Above 4 GiB one aperture runs from the end of the
|
||||
// described space to the 46-bit line.
|
||||
const Range = struct { base: u64, end: u64 };
|
||||
var below: [64]Range = undefined;
|
||||
var below_count: usize = 0;
|
||||
var high_end: u64 = 1 << 32;
|
||||
for (boot_memory_regions) |region| {
|
||||
const end = region.base + region.pages * 4096;
|
||||
@@ -636,37 +697,18 @@ fn addBridgeApertures(bridge: *device_model.Device) void {
|
||||
// kernel image, the tables, the ramdisk all live there). Bring-up
|
||||
// trust: only the bridge's claimant can register into the aperture.
|
||||
if (region.kind == .usable and end > high_end) high_end = end;
|
||||
if (region.base >= (1 << 32) or below_count == below.len) continue;
|
||||
below[below_count] = .{ .base = region.base, .end = @min(end, 1 << 32) };
|
||||
below_count += 1;
|
||||
}
|
||||
// Insertion sort by base (the map is small and this runs once at boot).
|
||||
for (1..below_count) |i| {
|
||||
const key = below[i];
|
||||
var j = i;
|
||||
while (j > 0 and below[j - 1].base > key.base) : (j -= 1) below[j] = below[j - 1];
|
||||
below[j] = key;
|
||||
}
|
||||
// Walk the sorted ranges, collecting inter-region gaps of at least 1 MiB.
|
||||
var gaps: [3]Range = .{Range{ .base = 0, .end = 0 }} ** 3;
|
||||
var cursor: u64 = 0;
|
||||
var index: usize = 0;
|
||||
while (index <= below_count) : (index += 1) {
|
||||
const gap_end = if (index == below_count) (1 << 32) else below[index].base;
|
||||
if (gap_end > cursor and gap_end - cursor >= (1 << 20)) {
|
||||
// Keep the three largest, replacing the smallest kept so far.
|
||||
var smallest: usize = 0;
|
||||
for (gaps, 0..) |gap, gi| {
|
||||
if (gap.end - gap.base < gaps[smallest].end - gaps[smallest].base) smallest = gi;
|
||||
}
|
||||
if (gap_end - cursor > gaps[smallest].end - gaps[smallest].base) {
|
||||
gaps[smallest] = .{ .base = cursor, .end = gap_end };
|
||||
}
|
||||
}
|
||||
if (index < below_count and below[index].end > cursor) cursor = below[index].end;
|
||||
}
|
||||
for (gaps) |gap| {
|
||||
if (gap.end > gap.base) _ = bridge.addResource(.memory, gap.base, gap.end - gap.base);
|
||||
|
||||
// Three holes below 4 GiB. This three is not a guess about hardware: a
|
||||
// `DeviceDescriptor` carries `maximum_device_resources` (8) resources, and the
|
||||
// bridge spends them on ECAM + bus range + these + the high aperture. That cap is
|
||||
// a wire struct in the kernel↔user ABI, so widening it is a separate change —
|
||||
// docs/bounds-track-plan.md, phase 4. Keeping *fewer* holes is fail-closed: it
|
||||
// refuses BARs, it never admits one.
|
||||
var holes: [3]AddressRange = undefined;
|
||||
_ = largestHolesBelow4G(boot_memory_regions, 1 << 20, &holes);
|
||||
for (holes) |hole| {
|
||||
if (hole.end > hole.base) _ = bridge.addResource(.memory, hole.base, hole.end - hole.base);
|
||||
}
|
||||
_ = bridge.addResource(.memory, high_end, (@as(u64, 1) << 46) - high_end);
|
||||
}
|
||||
|
||||
@@ -59,8 +59,15 @@ const command_opcode_shift = 60; // opcode in bits 63:60 of qword 0
|
||||
const command_completion_wait: u64 = 0x01;
|
||||
const command_invalidate_devtab: u64 = 0x02;
|
||||
const command_invalidate_pages: u64 = 0x03;
|
||||
const completion_wait_store: u64 = 1 << 1; // S: store `data` to the supplied address
|
||||
const invalidate_pages_all: u64 = 0x000F_FFFF_FFFF_F000 | 1; // address bits 51:12 all-ones + S
|
||||
// COMPLETION_WAIT qword 0: bit 0 is S (store `data` to the supplied address), bit 1 is
|
||||
// I (raise an interrupt). An earlier revision set bit 1 and then blamed QEMU for the
|
||||
// sentinel never landing — with I instead of S the IOMMU is never *asked* to store, on
|
||||
// QEMU or on silicon, and completeAndWait was no barrier at all.
|
||||
const completion_wait_store: u64 = 1 << 0;
|
||||
// INVALIDATE_IOMMU_PAGES address qword, S=1: the architected invalidate-everything
|
||||
// encoding is bits 62:12 all-ones (the spec's literal 0x7FFF_FFFF_FFFF_F000). The
|
||||
// previous 51:12 value is a non-architected range real silicon is free to misread.
|
||||
const invalidate_pages_all: u64 = 0x7FFF_FFFF_FFFF_F000 | 1;
|
||||
|
||||
const levels: u8 = 4; // 48-bit IOVA, matching the Intel 4-level path
|
||||
|
||||
@@ -228,11 +235,14 @@ fn submitCommand(qword0: u64, qword1: u64) void {
|
||||
}
|
||||
|
||||
/// Append a COMPLETION_WAIT (store form) and spin until the IOMMU writes our sentinel to
|
||||
/// the completion frame. QEMU consumes the command buffer synchronously on the tail-
|
||||
/// register write, so by the time we poll the prior invalidation is already applied; the
|
||||
/// store confirmation is belt-and-suspenders for real hardware. If it never lands
|
||||
/// (QEMU's amd-iommu does not implement the store form), warn ONCE and proceed — the
|
||||
/// invalidation itself has happened.
|
||||
/// the completion frame. This is the driver's only ordering barrier: real hardware
|
||||
/// fetches commands asynchronously, so a preceding invalidation has not happened until
|
||||
/// this store lands — the callers that free DMA frames after an unmap depend on it.
|
||||
/// (QEMU consumes the ring synchronously on the tail write and implements the store
|
||||
/// form fine; the warning below once fired there only because the command carried the
|
||||
/// I bit instead of S, so no store was ever requested.) If the store never lands, warn
|
||||
/// ONCE and proceed rather than wedge the claim path — but on real silicon that line
|
||||
/// means invalidations are unconfirmed and must be treated as a bug report.
|
||||
fn completeAndWait() void {
|
||||
const sentinel: u64 = 0xC0FFEE;
|
||||
ram(completion_frame)[0] = 0;
|
||||
@@ -246,7 +256,7 @@ fn completeAndWait() void {
|
||||
if (spins > 100_000) {
|
||||
if (!completion_warned) {
|
||||
completion_warned = true;
|
||||
iommu.environment.write("/system/kernel: AMD-Vi COMPLETION_WAIT store not observed — proceeding (QEMU processes commands synchronously)\n");
|
||||
iommu.environment.write("/system/kernel: AMD-Vi COMPLETION_WAIT store not observed — proceeding WITHOUT an invalidation barrier\n");
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -518,4 +518,13 @@ isr_smap_patch:
|
||||
testb $3, 8(%rsp)
|
||||
jz 1f
|
||||
swapgs
|
||||
1: iretq
|
||||
# Named so a fault reporter can recognise its own return instruction. A #GP
|
||||
# here is the frame's fault, not this code's — the five words below RSP are
|
||||
# what the CPU rejected, and they are the only evidence of why. Note that on
|
||||
# the ring-3 path the swapgs above has already run, so a fault at this exact
|
||||
# address re-enters the kernel with the *user's* GS base: per-CPU reads in
|
||||
# that handler are reading user-controlled state and must not be trusted.
|
||||
1:
|
||||
.global isr_return_iretq
|
||||
isr_return_iretq:
|
||||
iretq
|
||||
|
||||
@@ -57,10 +57,21 @@ pub fn setLocal(index: usize, scheduler_ptr: usize) void {
|
||||
io.wrmsr(ia32_gs_base, @intFromPtr(&blocks[index]));
|
||||
}
|
||||
|
||||
/// The scheduler pointer for the running core (via the GS base). Valid in any
|
||||
/// ring-0 context under the swapgs discipline.
|
||||
/// The scheduler pointer for the running core (via the GS base), or 0 before this
|
||||
/// core has one. Valid in any ring-0 context under the swapgs discipline.
|
||||
///
|
||||
/// **Zero is an answer here, not a fault.** A core has no per-CPU block between
|
||||
/// reset and its `setLocal`, and the fault reporters are documented to be callable
|
||||
/// unconditionally: `currentIdSafe`, `currentNameSafe`, `currentCpuIndex` and
|
||||
/// `sync.releaseIfHeldHere` all test this for 0 and mean it. Casting the base
|
||||
/// before testing it put the null one instruction out of their reach — an early
|
||||
/// fault reported itself by panicking on the cast, and the panic handler, calling
|
||||
/// those same reporters on its next line, panicked again, so the machine reset
|
||||
/// instead of halting with the message that would have said what went wrong.
|
||||
pub fn scheduler() usize {
|
||||
return @as(*const ArchitecturePerCpu, @ptrFromInt(io.rdmsr(ia32_gs_base))).scheduler;
|
||||
const base = io.rdmsr(ia32_gs_base);
|
||||
if (base == 0) return 0;
|
||||
return @as(*const ArchitecturePerCpu, @ptrFromInt(base)).scheduler;
|
||||
}
|
||||
|
||||
/// Record core `index`'s kernel stack top, used by the system_call entry stub to
|
||||
@@ -76,16 +87,35 @@ const ia32_star = 0xC000_0081;
|
||||
const ia32_lstar = 0xC000_0082;
|
||||
const ia32_sfmask = 0xC000_0084;
|
||||
|
||||
/// The SYSRET half of STAR: `sysret` loads CS from base+16 and SS from base+8.
|
||||
/// The base carries RPL 3 itself — 0x13, not the 0x10 the GDT layout suggests —
|
||||
/// because **only CS is guaranteed to come out at ring 3**. CS must: `sysret`
|
||||
/// sets CPL to 3, so the selector it loads is forced to match. SS is under no
|
||||
/// such obligation, and the two vendors differ on it: Intel ORs 3 into the SS
|
||||
/// selector as well, AMD hands it over exactly as the arithmetic produced it.
|
||||
///
|
||||
/// With base 0x10 an AMD machine therefore enters ring 3 carrying SS 0x18 — RPL
|
||||
/// 0 — and nothing complains, because a data access is checked against CPL and
|
||||
/// the descriptor's DPL, never the selector's RPL. It runs perfectly until the
|
||||
/// first interrupt: the CPU pushes that SS, and the `iretq` returning to ring 3
|
||||
/// requires SS.RPL to equal CS.RPL, refusing 0 against 3 with #GP(0x18). A
|
||||
/// process that made system calls happily then dies on its first timer tick.
|
||||
///
|
||||
/// Putting the 3 in the base makes both selectors right by construction on
|
||||
/// either vendor: 0x13 + 8 = 0x1B, 0x13 + 16 = 0x23. Intel's OR is then a no-op
|
||||
/// rather than the thing holding it together.
|
||||
const star_sysret_base: u64 = 0x13; // user data 0x18 / user code 0x20, with RPL 3
|
||||
const star_syscall_base: u64 = 0x08; // kernel code 0x08 / kernel data 0x10
|
||||
|
||||
/// Enable the `system_call`/`sysret` fast path on this core (BSP and each AP). EFER.SCE
|
||||
/// turns the instructions on; STAR sets the selectors system_call/sysret load; LSTAR
|
||||
/// is the entry stub (isr.s); SFMASK clears RFLAGS bits on entry (notably IF —
|
||||
/// the handler runs with interrupts off, like the int-gate path). The GDT is laid
|
||||
/// out (kernel code 0x08, then user data 0x18 / code 0x20) precisely so these line
|
||||
/// up: system_call loads CS 0x08 / SS 0x10; sysret loads CS = base+16 and SS = base+8
|
||||
/// with RPL forced to 3, so base 0x10 gives CS 0x23 (user code|3) and SS 0x1B.
|
||||
/// up: system_call loads CS 0x08 / SS 0x10; sysret loads CS 0x23 and SS 0x1B.
|
||||
pub fn initSystemCall() void {
|
||||
io.wrmsr(ia32_efer, io.rdmsr(ia32_efer) | 1); // SCE
|
||||
io.wrmsr(ia32_star, (@as(u64, 0x08) << 32) | (@as(u64, 0x10) << 48));
|
||||
io.wrmsr(ia32_star, (star_syscall_base << 32) | (star_sysret_base << 48));
|
||||
const entry = @extern(*const anyopaque, .{ .name = "syscall_entry" });
|
||||
io.wrmsr(ia32_lstar, @intFromPtr(entry));
|
||||
// Clear IF, TF, DF, AC and NT on entry. The first four are the usual
|
||||
|
||||
@@ -19,31 +19,117 @@
|
||||
//! ever subdivide what it was already given.
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const platform = @import("platform");
|
||||
const device_abi = @import("device-abi");
|
||||
const heap = @import("heap.zig");
|
||||
|
||||
const maximum_devices = 64;
|
||||
/// How many devices a single registrar may put in the table.
|
||||
///
|
||||
/// **This is a runaway detector, not a security boundary**, and the difference matters.
|
||||
/// It cannot stop a malicious driver — a quota generous enough never to bite a real
|
||||
/// machine is still generous enough to be unpleasant — and it is not trying to. What
|
||||
/// stops malice is that only a driver the manager handed a device can register children
|
||||
/// under it (docs/os-development/device-authority.md). What this catches is a
|
||||
/// *legitimate* driver in a loop, early, and attributably: the driver that did it is
|
||||
/// refused, named in its own log, and restarted, while every other driver is untouched.
|
||||
///
|
||||
/// The shared ceilings it replaces could not do that. `maximum_devices = 64` was a
|
||||
/// guess about someone else's computer and one driver's enumeration starved every
|
||||
/// other — which is how an AMD Ryzen came to boot with a working display, no USB and no
|
||||
/// storage. A per-registrar allowance is the same protection charged to whoever caused
|
||||
/// it, which is the microkernel property rather than a workaround for it.
|
||||
///
|
||||
/// The number: a machine's whole PCI segment tops out at 65536 functions, and the
|
||||
/// biggest real registrar seen is pci-bus at a few dozen. 4096 is far above anything a
|
||||
/// real machine produces and around 1.4 MiB of descriptors, well under the kernel heap.
|
||||
/// **Reaching it is a bug report, not a tuning request** — no correct driver gets near.
|
||||
///
|
||||
/// bound: devices one task may register
|
||||
/// decided-by: ours
|
||||
/// protects: the kernel heap, against a driver looping device_register
|
||||
/// at-limit: refuse — ECHILDREN to the registrar; every other driver is unaffected
|
||||
/// observed-by: the bus driver's line naming the reason (pci-bus reconciles functions
|
||||
/// found against registered)
|
||||
const maximum_devices_per_registrar = 4096;
|
||||
|
||||
/// Cap on children a single parent may have. A zero-resource child (legal — a USB
|
||||
/// device is addressed through its controller, not by MMIO) sidesteps the containment
|
||||
/// check, so without a bound a process that claimed one device could loop
|
||||
/// `device_register` and exhaust the whole table, permanently denying it to every other
|
||||
/// driver. This bounds the blast radius of one claim; a real quota (and a
|
||||
/// `device_release` to reclaim on exit) is future work — see docs/driver-model.md.
|
||||
const maximum_children_per_parent = 16;
|
||||
|
||||
var devices: [maximum_devices]device_abi.DeviceDescriptor = undefined;
|
||||
var claimed: [maximum_devices]?u32 = .{null} ** maximum_devices; // owner task id, or null
|
||||
/// The table, grown on demand from the kernel heap. **There is no ceiling**: how many
|
||||
/// devices a machine has is the machine's business, and no specification bounds it, so
|
||||
/// nothing here should. `devices_broker.init` runs after `heap.init` (kernel.zig), so
|
||||
/// there was never a reason for this to be static beyond it having been written that
|
||||
/// way first.
|
||||
var devices: []device_abi.DeviceDescriptor = &.{};
|
||||
/// Owner task id per device, or null. Parallel to `devices` and grown with it.
|
||||
var claimed: []?u32 = &.{};
|
||||
/// The task that called `register` for each device, so the per-registrar allowance can
|
||||
/// be charged to whoever caused the entry. Firmware-discovered nodes carry `no_registrar`
|
||||
/// — they are the kernel's own, not anybody's doing.
|
||||
/// Optional, not a sentinel: **task 0 is a real task** (the kernel's own), so any
|
||||
/// "none" value inside the id space is a device belonging to somebody reading back as
|
||||
/// belonging to nobody. Found the moment the first test asserted a giver, because the
|
||||
/// task doing the giving was task 0.
|
||||
var registrar: []?u32 = &.{};
|
||||
|
||||
/// **Who gave each device away**, or `no_giver` if nobody ever did.
|
||||
///
|
||||
/// One field, and the whole authority rule follows from it: a device that was *given*
|
||||
/// to someone is delegated hardware, so it may only be handed on, never taken
|
||||
/// (`claim` refuses it); and when its holder dies it goes back to whoever lent it,
|
||||
/// rather than becoming free for anyone to grab.
|
||||
///
|
||||
/// It also settles the framebuffer without mentioning it. Nobody delegates the
|
||||
/// loader's framebuffer, so it has no giver, so the display service claims it exactly
|
||||
/// as it always has — no exemption, no special case, no `display` anywhere in the rule.
|
||||
var giver: []?u32 = &.{};
|
||||
var count: usize = 0;
|
||||
|
||||
/// Grow the three parallel arrays so at least one more device fits. False if the heap
|
||||
/// cannot satisfy it, which the callers report rather than swallow.
|
||||
fn reserve() bool {
|
||||
if (count < devices.len) return true;
|
||||
// Double from a deliberately SMALL first block. Sizing it for a typical machine
|
||||
// would mean the growth path never ran on the hardware we test on, and only woke up
|
||||
// on someone else's larger machine — which is the exact shape of the failure this
|
||||
// whole track exists to stop. At 8, every boot grows the table several times, so
|
||||
// the path is exercised constantly and the suite asserts it.
|
||||
const wanted = if (devices.len == 0) 8 else devices.len * 2;
|
||||
const allocator = heap.allocator();
|
||||
const grown_devices = allocator.realloc(devices, wanted) catch return false;
|
||||
devices = grown_devices;
|
||||
const grown_claimed = allocator.realloc(claimed, wanted) catch return false;
|
||||
claimed = grown_claimed;
|
||||
const grown_registrar = allocator.realloc(registrar, wanted) catch return false;
|
||||
registrar = grown_registrar;
|
||||
const grown_giver = allocator.realloc(giver, wanted) catch return false;
|
||||
giver = grown_giver;
|
||||
for (claimed[count..], registrar[count..], giver[count..]) |*slot, *who, *lender| {
|
||||
slot.* = null;
|
||||
who.* = null;
|
||||
lender.* = null;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/// How many devices `task` has registered — the allowance is charged per registrar, so
|
||||
/// a driver in a loop exhausts its own and no one else's.
|
||||
fn registeredBy(task: u32) usize {
|
||||
var n: usize = 0;
|
||||
for (registrar[0..count]) |who| {
|
||||
if (who != null and who.? == task) n += 1;
|
||||
}
|
||||
return n;
|
||||
}
|
||||
|
||||
/// The id of the seeded framebuffer node (`seedDisplay`), or null when the machine
|
||||
/// handed over no framebuffer. Lets the process layer recognise the display claim
|
||||
/// (to quiesce the bootstrap console) without threading the id through every caller.
|
||||
var display_device: ?u64 = null;
|
||||
|
||||
/// Devices discovery found but the table had no room for. Non-zero means the machine
|
||||
/// is bigger than `maximum_devices` and some hardware is simply invisible to drivers —
|
||||
/// which would otherwise be an entirely silent failure. Logged at boot.
|
||||
/// Devices discovery found but could not record. The table grows on demand, so this is
|
||||
/// no longer "the machine is bigger than our guess" — it means the kernel heap could not
|
||||
/// satisfy the growth, which would otherwise be an entirely silent failure. Logged at
|
||||
/// boot.
|
||||
pub var dropped: usize = 0;
|
||||
|
||||
/// Snapshot the device tree into the flat table. Run once, right after discovery.
|
||||
@@ -51,7 +137,9 @@ pub fn init(device_tree: *const platform.DeviceTree) void {
|
||||
count = 0;
|
||||
dropped = 0;
|
||||
display_device = null;
|
||||
for (&claimed) |*c| c.* = null;
|
||||
for (claimed) |*c| c.* = null;
|
||||
for (registrar) |*r| r.* = null;
|
||||
for (giver) |*g| g.* = null;
|
||||
walk(device_tree.root, device_abi.no_parent);
|
||||
}
|
||||
|
||||
@@ -63,7 +151,7 @@ pub fn init(device_tree: *const platform.DeviceTree) void {
|
||||
/// table is full. Idempotent-ish: only ever call once per boot.
|
||||
pub fn seedDisplay(base: u64, width: u32, height: u32, pitch: u32, format: u32, refresh_hz: u32) ?u64 {
|
||||
if (base == 0 or width == 0 or height == 0) return null; // headless
|
||||
if (count >= maximum_devices) {
|
||||
if (!reserve()) {
|
||||
dropped += 1;
|
||||
return null;
|
||||
}
|
||||
@@ -108,7 +196,7 @@ fn walk(node: *platform.Device, parent_id: u64) void {
|
||||
}
|
||||
|
||||
fn record(node: *platform.Device, parent_id: u64) u64 {
|
||||
if (count >= maximum_devices) {
|
||||
if (!reserve()) {
|
||||
dropped += 1;
|
||||
return device_abi.no_parent; // children of a dropped node become roots, not orphans
|
||||
}
|
||||
@@ -157,13 +245,36 @@ pub fn enumerateFrom(start: usize, out: []device_abi.DeviceDescriptor) usize {
|
||||
return n;
|
||||
}
|
||||
|
||||
/// Take exclusive ownership of device `id` for task `owner`. Fails if the id is
|
||||
/// out of range or already claimed.
|
||||
pub fn claim(id: u64, owner: u32) bool {
|
||||
if (id >= count) return false;
|
||||
if (claimed[@intCast(id)] != null) return false;
|
||||
/// Take exclusive ownership of device `id` for task `owner`. The two ways this can
|
||||
/// fail want different responses from a driver — a stale id means re-enumerate, a
|
||||
/// live claimant means back off — so they are distinguishable (`ClaimError`).
|
||||
pub fn claim(id: u64, owner: u32) ClaimError!void {
|
||||
if (id >= count) return error.NoSuchDevice;
|
||||
if (claimed[@intCast(id)] != null) return error.AlreadyClaimed;
|
||||
// **Delegated hardware may be handed on, never taken.**
|
||||
//
|
||||
// Belt and braces, and worth being honest about: with the loan rule above this is
|
||||
// **currently unreachable**. A device that was given to someone is held, so it is
|
||||
// refused as `AlreadyClaimed` before reaching here; and when the holder dies the
|
||||
// device goes back to its lender (or, if the lender is gone, has its giver cleared
|
||||
// with its claim), so there is no state where a device is unheld *and* still on
|
||||
// loan. The window a stranger could have used simply stops existing.
|
||||
//
|
||||
// It stays because it is one comparison and it fails closed: any future path that
|
||||
// frees a device without clearing its giver would otherwise hand delegated
|
||||
// hardware to whoever asked first, which is exactly the hole this run closed.
|
||||
//
|
||||
// Note it leaves the loader's framebuffer alone without naming it: nobody delegates
|
||||
// the framebuffer, so it has no giver, so the display service claims it as always.
|
||||
if (giver[@intCast(id)] != null) return error.NotYours;
|
||||
claimed[@intCast(id)] = owner;
|
||||
return true;
|
||||
}
|
||||
|
||||
/// The task that gave device `id` away, or null if nobody ever did. A device with a
|
||||
/// giver is delegated hardware: it may be handed on, never taken.
|
||||
pub fn giverOf(id: u64) ?u32 {
|
||||
if (id >= count) return null;
|
||||
return giver[@intCast(id)];
|
||||
}
|
||||
|
||||
/// The task that owns device `id`, or null.
|
||||
@@ -177,11 +288,41 @@ pub fn ownerOf(id: u64) ?u32 {
|
||||
/// hardware again (docs/process-lifecycle.md iron rule 1: cleanup is the kernel's
|
||||
/// job). The devices stay in the table — they describe hardware, which did not go
|
||||
/// away — only their ownership clears.
|
||||
/// Whether a task is still alive, injected by the process layer (which owns the task
|
||||
/// table) the same way the scheduler's other hooks are. Null means "assume not", so a
|
||||
/// kernel built without it clears claims rather than handing them to a ghost.
|
||||
pub var task_alive_hook: ?*const fn (u32) bool = null;
|
||||
|
||||
fn alive(task: u32) bool {
|
||||
const hook = task_alive_hook orelse return false;
|
||||
return hook(task);
|
||||
}
|
||||
|
||||
pub fn releaseAllOwnedBy(owner: u32) void {
|
||||
for (claimed[0..count]) |*slot| {
|
||||
if (slot.*) |o| {
|
||||
if (o == owner) slot.* = null;
|
||||
for (claimed[0..count], 0..) |*slot, id| {
|
||||
const holder = slot.* orelse continue;
|
||||
if (holder != owner) continue;
|
||||
|
||||
// **A grant is a loan.** A device this task was *given* goes back to whoever
|
||||
// lent it, not to nobody — so the device manager gets its hardware back the
|
||||
// instant a driver dies, and hands it to the replacement.
|
||||
//
|
||||
// Without this the kernel released the claim to no one and the manager
|
||||
// re-claimed first-come, so every driver restart reopened the window this
|
||||
// rule closes. And once `claim` refuses a device that has a giver, releasing
|
||||
// to nobody would strand it: no one could ever take it again.
|
||||
//
|
||||
// A dead lender is no lender: clear the claim and the giver together, so the
|
||||
// device is genuinely free rather than owed to a ghost.
|
||||
if (giver[id]) |lender| {
|
||||
if (alive(lender)) {
|
||||
slot.* = lender;
|
||||
giver[id] = null; // returned; it is the lender's own again, not on loan
|
||||
continue;
|
||||
}
|
||||
giver[id] = null;
|
||||
}
|
||||
slot.* = null;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -274,51 +415,99 @@ fn contains(parent: device_abi.ResourceDescriptor, child: device_abi.ResourceDes
|
||||
return child.start >= parent.start and child_end <= parent_end;
|
||||
}
|
||||
|
||||
/// Why a `register` was refused. Each variant maps to its own errno (`errnoOf`), so
|
||||
/// a bus driver's log line can name the rule that stopped it — "this parent is at
|
||||
/// its child cap" and "the table is full" want different fixes, and telling them
|
||||
/// apart from a bare -1 cost a debugging session (docs/fixed-bounds-audit.md).
|
||||
pub const RegisterError = error{
|
||||
NoSpace, // the device table is full
|
||||
BadParent, // no such device, or not claimed by this task
|
||||
TooManyResources,
|
||||
TooManyChildren, // this parent is at maximum_children_per_parent
|
||||
NoSuchParent, // no device with that id
|
||||
NotYourParent, // that device exists but this task has not claimed it
|
||||
TooManyResources, // the descriptor declares more resources than one device may hold
|
||||
TooManyChildren, // the caller is at its per-registrar allowance
|
||||
NotContained, // a child resource escapes its parent's window
|
||||
};
|
||||
|
||||
/// Number of devices currently recorded with `parent_id` as their parent.
|
||||
fn childCount(parent_id: u64) usize {
|
||||
var n: usize = 0;
|
||||
for (devices[0..count]) |d| {
|
||||
if (d.parent == parent_id) n += 1;
|
||||
}
|
||||
return n;
|
||||
/// The errno a refused `register` returns to ring 3.
|
||||
pub fn errnoOf(e: RegisterError) i64 {
|
||||
return switch (e) {
|
||||
error.NoSpace => abi.ENOSPC,
|
||||
error.NoSuchParent => abi.ENODEV,
|
||||
error.NotYourParent => abi.EPERM,
|
||||
error.TooManyResources => abi.E2BIG,
|
||||
error.TooManyChildren => abi.ECHILDREN,
|
||||
error.NotContained => abi.ERANGE,
|
||||
};
|
||||
}
|
||||
|
||||
/// Publish `descriptor` as a child of `parent_id`, on behalf of `owner`. Returns the new
|
||||
/// device id. The child is left **unclaimed**, so another process (a class driver)
|
||||
/// can claim it — that is how a bus hands a device to its driver.
|
||||
/// Why a `claim` was refused.
|
||||
pub const ClaimError = error{
|
||||
NoSuchDevice, // no device with that id
|
||||
AlreadyClaimed, // a live task already owns it
|
||||
NotYours, // delegated hardware: it has a giver, so it must be handed on, not taken
|
||||
};
|
||||
|
||||
/// Why a `transfer` was refused.
|
||||
pub const TransferError = error{
|
||||
NoSuchDevice, // no device with that id
|
||||
NotHeld, // the caller does not hold it — you may only give away what you have
|
||||
};
|
||||
|
||||
/// The errno a refused `transfer` returns to ring 3. (`ESRCH` — no such recipient — is
|
||||
/// raised by the caller in system/kernel/process.zig, which is what can see the task
|
||||
/// table.)
|
||||
pub fn transferErrnoOf(e: TransferError) i64 {
|
||||
return switch (e) {
|
||||
error.NoSuchDevice => abi.ENODEV,
|
||||
error.NotHeld => abi.EPERM,
|
||||
};
|
||||
}
|
||||
|
||||
/// Move device `id` from `from` to `to`. **A move, not a copy**: a claim is exclusive
|
||||
/// (driver-model.md, invariant 1), so the giver stops holding it the moment the
|
||||
/// receiver starts.
|
||||
///
|
||||
/// `owner` must have claimed `parent_id`, and every resource in `descriptor` must be
|
||||
/// contained in a parent resource of the same kind. A device with no resources is
|
||||
/// fine and common: a USB device is addressed through its controller, not by MMIO.
|
||||
pub fn register(parent_id: u64, owner: u32, descriptor: *const device_abi.DeviceDescriptor) RegisterError!u64 {
|
||||
const parent_owner = ownerOf(parent_id) orelse return error.BadParent;
|
||||
if (parent_owner != owner) return error.BadParent;
|
||||
if (descriptor.resource_count > device_abi.maximum_device_resources) return error.TooManyResources;
|
||||
if (childCount(parent_id) >= maximum_children_per_parent) return error.TooManyChildren;
|
||||
if (count >= maximum_devices) return error.NoSpace;
|
||||
/// This is the mechanism behind delegation — the device manager claims what firmware
|
||||
/// discovery seeded and passes each device to the driver it matched, which replaces
|
||||
/// first-come-first-served `device_claim` with policy
|
||||
/// (docs/device-driver-development/device-manager.md). The kernel checks only that the
|
||||
/// caller holds the device: *you may give away what you have*. It knows nothing about
|
||||
/// which task is the manager, and needs to know nothing.
|
||||
///
|
||||
/// Note this is deliberately NOT the M13 capability-passing path, which shares a handle
|
||||
/// refcounted — a copy. Exclusivity cannot be expressed that way.
|
||||
pub fn transfer(id: u64, from: u32, to: u32) TransferError!void {
|
||||
try canTransfer(id, from);
|
||||
claimed[@intCast(id)] = to;
|
||||
giver[@intCast(id)] = from;
|
||||
}
|
||||
|
||||
const parent = &devices[@intCast(parent_id)];
|
||||
for (0..@intCast(descriptor.resource_count)) |i| {
|
||||
const r = descriptor.resources[i];
|
||||
var ok = false;
|
||||
for (0..@intCast(parent.resource_count)) |j| {
|
||||
if (contains(parent.resources[j], r)) ok = true;
|
||||
}
|
||||
if (!ok) return error.NotContained;
|
||||
}
|
||||
/// The checks `transfer` will make, without the move. The syscall layer runs them
|
||||
/// first — under the same lock hold that the transfer itself will run under — so it
|
||||
/// can refuse, or arrange the IOMMU confinement the move needs, while nothing has
|
||||
/// mutated yet and there is nothing to roll back.
|
||||
pub fn canTransfer(id: u64, from: u32) TransferError!void {
|
||||
if (id >= count) return error.NoSuchDevice;
|
||||
const holder = claimed[@intCast(id)] orelse return error.NotHeld;
|
||||
if (holder != from) return error.NotHeld;
|
||||
}
|
||||
|
||||
// Idempotent on exact match (docs/device-manager.md): a restarted
|
||||
// registering bus re-registers what it rediscovers, and the table has no
|
||||
// unregister — an identical (class, identity, resources) child under the
|
||||
// same parent returns the existing id instead of appending a duplicate.
|
||||
/// The errno a refused `claim` returns to ring 3. (`ECONFINE` — the claim stood but
|
||||
/// the IOMMU would not confine the device — is raised by the caller in
|
||||
/// system/kernel/process.zig, which is what rolls the claim back.)
|
||||
pub fn claimErrnoOf(e: ClaimError) i64 {
|
||||
return switch (e) {
|
||||
error.NoSuchDevice => abi.ENODEV,
|
||||
error.AlreadyClaimed => abi.EBUSY,
|
||||
error.NotYours => abi.EPERM,
|
||||
};
|
||||
}
|
||||
|
||||
/// The id of a child of `parent_id` already identical to `descriptor`, or null.
|
||||
/// Exact on class, identity and every resource — anything less would let a bus
|
||||
/// silently adopt an entry that is not the device it just found. The caller must
|
||||
/// have bounded `descriptor.resource_count` first.
|
||||
fn existingChild(parent_id: u64, descriptor: *const device_abi.DeviceDescriptor) ?u64 {
|
||||
for (devices[0..count]) |*existing| {
|
||||
if (existing.parent != parent_id) continue;
|
||||
if (existing.class != descriptor.class) continue;
|
||||
@@ -334,6 +523,54 @@ pub fn register(parent_id: u64, owner: u32, descriptor: *const device_abi.Device
|
||||
}
|
||||
if (same) return existing.id;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
|
||||
/// Publish `descriptor` as a child of `parent_id`, on behalf of `owner`. Returns the new
|
||||
/// device id. The child is left **unclaimed**, so another process (a class driver)
|
||||
/// can claim it — that is how a bus hands a device to its driver.
|
||||
///
|
||||
/// `owner` must have claimed `parent_id`, and every resource in `descriptor` must be
|
||||
/// contained in a parent resource of the same kind. A device with no resources is
|
||||
/// fine and common: a USB device is addressed through its controller, not by MMIO.
|
||||
pub fn register(parent_id: u64, owner: u32, descriptor: *const device_abi.DeviceDescriptor) RegisterError!u64 {
|
||||
if (parent_id >= count) return error.NoSuchParent;
|
||||
const parent_owner = ownerOf(parent_id) orelse return error.NotYourParent;
|
||||
if (parent_owner != owner) return error.NotYourParent;
|
||||
// Bounds every `descriptor.resources` read below, including the match scan's.
|
||||
if (descriptor.resource_count > device_abi.maximum_device_resources) return error.TooManyResources;
|
||||
|
||||
// Idempotent on exact match (docs/device-manager.md): a restarted registering
|
||||
// bus re-registers what it rediscovers, and the table has no unregister — an
|
||||
// identical child under the same parent returns the existing id instead of
|
||||
// appending a duplicate.
|
||||
//
|
||||
// Checked **before the caps**, because a re-registration consumes no slot.
|
||||
// Charging it against the child cap refused a restarted bus its own devices the
|
||||
// second time it started, which turned the supervision restart this system leans
|
||||
// on into a one-way ratchet toward a degraded machine. The match is exact — class,
|
||||
// identity, and every resource — so an entry returned this way was contained when
|
||||
// it was first admitted, and a stored descriptor's resources never change
|
||||
// afterwards (they are written only by `record`, `seedDisplay` and the append
|
||||
// below).
|
||||
if (existingChild(parent_id, descriptor)) |existing_id| return existing_id;
|
||||
|
||||
// The allowance is charged to whoever is registering, so a driver in a loop
|
||||
// exhausts its own and every other driver carries on. There is no machine-wide
|
||||
// ceiling any more: the table grows.
|
||||
if (registeredBy(owner) >= maximum_devices_per_registrar) return error.TooManyChildren;
|
||||
if (!reserve()) return error.NoSpace;
|
||||
|
||||
const parent = &devices[@intCast(parent_id)];
|
||||
for (0..@intCast(descriptor.resource_count)) |i| {
|
||||
const r = descriptor.resources[i];
|
||||
var ok = false;
|
||||
for (0..@intCast(parent.resource_count)) |j| {
|
||||
if (contains(parent.resources[j], r)) ok = true;
|
||||
}
|
||||
if (!ok) return error.NotContained;
|
||||
}
|
||||
|
||||
var d = std.mem.zeroes(device_abi.DeviceDescriptor);
|
||||
d.id = count;
|
||||
@@ -346,6 +583,7 @@ pub fn register(parent_id: u64, owner: u32, descriptor: *const device_abi.Device
|
||||
for (0..@intCast(descriptor.resource_count)) |i| d.resources[i] = descriptor.resources[i];
|
||||
|
||||
devices[count] = d;
|
||||
registrar[count] = owner;
|
||||
count += 1;
|
||||
return d.id;
|
||||
}
|
||||
|
||||
+89
-10
@@ -34,12 +34,27 @@ const platform = @import("platform");
|
||||
const architecture = @import("architecture");
|
||||
const devices_broker = @import("devices-broker.zig");
|
||||
const log = @import("log.zig");
|
||||
const heap = @import("heap.zig");
|
||||
|
||||
const page_size: u64 = abi.page_size;
|
||||
const page_mask: u64 = page_size - 1;
|
||||
const huge_page_size: u64 = 2 * 1024 * 1024;
|
||||
|
||||
/// One domain per claimed PCI function. 64 mirrors devices-broker's device cap.
|
||||
/// The IOMMU's own translation-domain pool — one per claimed DMA-capable device.
|
||||
///
|
||||
/// No longer coupled to the device count. It was, by a comment and then by a comptime
|
||||
/// assert, only because `confined` (one slot per device id) was sized by this same
|
||||
/// constant; those are two unrelated quantities and making the device table dynamic
|
||||
/// separated them. This one is genuinely the hardware's: both VT-d and AMD-Vi report
|
||||
/// how many domains they support in a capability register, so the honest fix is to read
|
||||
/// it rather than choose 64 — bounds-track-plan.md phase 4.
|
||||
///
|
||||
/// bound: IOMMU translation domains the kernel can hold at once
|
||||
/// decided-by: hardware
|
||||
/// protects: the statically sized domain pool
|
||||
/// at-limit: refuse — ECONFINE; the claim is rolled back and the device is not driven,
|
||||
/// because a claim that cannot be confined must not stand
|
||||
/// observed-by: the claiming driver's own line naming ECONFINE
|
||||
pub const maximum_domains = 64;
|
||||
pub const invalid_domain: u16 = 0xFFFF;
|
||||
|
||||
@@ -94,18 +109,51 @@ pub fn init() void {
|
||||
/// Per-claimed-device record: its private domain, so a driver's death tears down
|
||||
/// exactly the domains it held.
|
||||
const Confined = struct { active: bool = false, owner: u32 = 0, bdf: u16 = 0, domain: u16 = invalid_domain };
|
||||
var confined: [maximum_domains]Confined = .{Confined{}} ** maximum_domains;
|
||||
|
||||
/// Indexed by **device id**, so it must cover every id the broker can mint — and the
|
||||
/// broker's table has no ceiling any more, so neither can this. It grows on demand.
|
||||
///
|
||||
/// This used to be `[maximum_domains]`, sized by the *domain* constant purely because
|
||||
/// device ids happened to stop at 64 as well. Two unrelated quantities sharing one
|
||||
/// number: `domains` below is the IOMMU's own translation-domain pool, which the
|
||||
/// hardware bounds and reports, while this is one slot per device the machine has.
|
||||
/// A comptime assert held them together while both were fixed; making the device table
|
||||
/// dynamic is what forced them apart, which is the assert having done its job.
|
||||
var confined: []Confined = &.{};
|
||||
|
||||
/// Grow `confined` to cover `device_id`. False if the heap cannot — and the caller
|
||||
/// treats that as a refusal to confine, never as permission.
|
||||
fn reserveConfined(device_id: u64) bool {
|
||||
if (device_id < confined.len) return true;
|
||||
if (device_id >= std.math.maxInt(usize) / 2) return false; // absurd id; refuse rather than size to it
|
||||
var wanted: usize = if (confined.len == 0) 64 else confined.len;
|
||||
while (wanted <= device_id) wanted *= 2;
|
||||
const grown = heap.allocator().realloc(confined, wanted) catch return false;
|
||||
const previous = confined.len;
|
||||
confined = grown;
|
||||
for (confined[previous..]) |*record| record.* = .{};
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Place a just-claimed PCI function under IOMMU translation on behalf of `owner`: give
|
||||
/// it a private empty domain, seed it with the device's own firmware reserved region,
|
||||
/// and attach. Its DMA buffers arrive afterward as explicit grants — the owner's own
|
||||
/// `dma_alloc`'d regions are bound by the claim path (`mapForDevice`), and cross-process
|
||||
/// buffers by `dma_bind`. false only if a domain can't be allocated — the caller rolls
|
||||
/// the claim back (a claim that can't be confined must not stand). No-op success when no
|
||||
/// IOMMU exists (fail-open).
|
||||
/// buffers by `dma_bind`. false when the device cannot be confined — the caller rolls
|
||||
/// the claim back (a claim that can't be confined must not stand).
|
||||
///
|
||||
/// **Fail-closed at the table's edge.** There is exactly one deliberate fail-open here:
|
||||
/// a machine with no IOMMU, which is a fact about the hardware rather than the size of
|
||||
/// anything. Running out of *room to record* a confinement is not that, and must refuse.
|
||||
pub fn confineDevice(device_id: u64, bdf: u16, owner: u32) bool {
|
||||
if (!active) return true;
|
||||
if (device_id >= confined.len) return true; // unusual id; leave it to fail-open
|
||||
if (!active) return true; // no IOMMU on this machine — nothing to confine with
|
||||
// A device id past the end of the record table. This returned `true` — success —
|
||||
// leaving the device outside every domain while telling the caller it was
|
||||
// confined, and rolling nothing back. It is unreachable only while device ids stop
|
||||
// at `confined.len`; moving the inventory out of the kernel and taking the domain
|
||||
// count from the hardware both change that, and either would have made a silent
|
||||
// unconfined DMA master out of every device past the 64th.
|
||||
if (!reserveConfined(device_id)) return false;
|
||||
const domain = domainCreate(owner, bdf) orelse return false;
|
||||
|
||||
// Firmware reserved region for this device, if any (real hardware; QEMU has none).
|
||||
@@ -147,7 +195,7 @@ pub fn unmapForDevice(device_id: u64, physical: u64, len: u64) void {
|
||||
/// own freshly-`dma_alloc`'d buffer into the devices it drives.
|
||||
pub fn mapRegionForOwner(owner: u32, physical: u64, len: u64) void {
|
||||
if (!active) return;
|
||||
for (&confined) |*c| {
|
||||
for (confined) |*c| {
|
||||
if (c.active and c.owner == owner) _ = map(c.domain, physical, len);
|
||||
}
|
||||
}
|
||||
@@ -158,7 +206,7 @@ pub fn mapRegionForOwner(owner: u32, physical: u64, len: u64) void {
|
||||
/// domain other than its owner's.
|
||||
pub fn unmapRegionEverywhere(physical: u64, len: u64) void {
|
||||
if (!active) return;
|
||||
for (&confined) |*c| {
|
||||
for (confined) |*c| {
|
||||
if (c.active) unmap(c.domain, physical, len);
|
||||
}
|
||||
}
|
||||
@@ -166,9 +214,40 @@ pub fn unmapRegionEverywhere(physical: u64, len: u64) void {
|
||||
/// A driver died or released its devices: tear down every domain it held (detach the
|
||||
/// device, free the tables) so their DMA is blocked again and a restarted driver
|
||||
/// re-claims cleanly. Runs BEFORE the broker claims and the DMA frames are released.
|
||||
/// Re-point a device's existing confinement at a new owner, keeping its domain and
|
||||
/// its attachment intact.
|
||||
///
|
||||
/// Delegation needs this: the device manager claims a device (which confines it, with
|
||||
/// the manager as owner) and then transfers it to the driver. Without moving the
|
||||
/// confinement record too, the domain stays the manager's — so the driver's DMA
|
||||
/// buffers are never bound into it, its rings are invisible to the device, and every
|
||||
/// transfer faults. Worse, a manager death would then tear down a domain a live driver
|
||||
/// is using, and a driver death would leave one behind.
|
||||
///
|
||||
/// The domain is *not* rebuilt: the device stays attached throughout, so there is no
|
||||
/// window in which it is translating through nothing.
|
||||
pub fn reassign(device_id: u64, owner: u32) void {
|
||||
if (!active) return;
|
||||
if (device_id >= confined.len) return;
|
||||
const record = &confined[@intCast(device_id)];
|
||||
if (!record.active) return;
|
||||
record.owner = owner;
|
||||
}
|
||||
|
||||
/// The task a device's confinement is recorded against, or null if it has none. The
|
||||
/// confinement's owner decides whose death tears the domain down, so a delegation that
|
||||
/// moved the device but not this record would leave a live driver's domain destroyed by
|
||||
/// its manager's exit — which is why `reassign` exists and why the suite asserts it.
|
||||
pub fn confinementOwner(device_id: u64) ?u32 {
|
||||
if (!active) return null;
|
||||
if (device_id >= confined.len) return null;
|
||||
const record = confined[@intCast(device_id)];
|
||||
return if (record.active) record.owner else null;
|
||||
}
|
||||
|
||||
pub fn releaseAllOwnedBy(owner: u32) void {
|
||||
if (!active) return;
|
||||
for (&confined) |*c| {
|
||||
for (confined) |*c| {
|
||||
if (c.active and c.owner == owner) {
|
||||
detachDevice(c.bdf);
|
||||
domainDestroy(c.domain);
|
||||
|
||||
@@ -42,15 +42,22 @@ pub const MESSAGE_MAXIMUM: usize = 256;
|
||||
pub const maximum_handles = scheduler.ipc_maximum_handles;
|
||||
|
||||
/// Errno-style failures, returned as `-value` in the system_call result register.
|
||||
pub const EBADF: i64 = 1; // bad handle
|
||||
pub const E2BIG: i64 = 2; // message exceeds MESSAGE_MAXIMUM
|
||||
pub const EFAULT: i64 = 3; // buffer unmapped / out of the user half
|
||||
pub const ENOENT: i64 = 4; // no such name
|
||||
pub const ENOSPC: i64 = 5; // handle table full
|
||||
pub const ENOMEM: i64 = 6; // out of memory
|
||||
pub const EPEER: i64 = 7; // peer died before replying (its process exited or was killed)
|
||||
pub const ESRCH: i64 = 8; // no such process (process_kill of an unknown/dead id)
|
||||
pub const EPERM: i64 = 9; // not permitted (process_kill by anyone but the supervisor)
|
||||
/// Restated from the shared kernel↔user ABI (system/abi.zig), because ring 3 reads
|
||||
/// the same numbers — the same reason `notify_badge_bit` below is restated. New
|
||||
/// codes are added *there*, which is where the space is documented.
|
||||
pub const EBADF = abi.EBADF; // bad handle
|
||||
pub const E2BIG = abi.E2BIG; // message exceeds MESSAGE_MAXIMUM
|
||||
pub const EFAULT = abi.EFAULT; // buffer unmapped / out of the user half
|
||||
pub const ENOENT = abi.ENOENT; // no such name
|
||||
pub const ENOSPC = abi.ENOSPC; // a kernel table is full
|
||||
pub const ENOMEM = abi.ENOMEM; // out of memory
|
||||
pub const EPEER = abi.EPEER; // peer died before replying (its process exited or was killed)
|
||||
pub const ESRCH = abi.ESRCH; // no such process (process_kill of an unknown/dead id)
|
||||
pub const EPERM = abi.EPERM; // not permitted (process_kill by anyone but the supervisor)
|
||||
pub const ENODEV = abi.ENODEV; // no such device id
|
||||
pub const ECHILDREN = abi.ECHILDREN; // this parent is at its child cap
|
||||
pub const ERANGE = abi.ERANGE; // a resource escapes its parent's window
|
||||
pub const ECONFINE = abi.ECONFINE; // the device could not be placed under IOMMU translation
|
||||
|
||||
/// A badge with this bit set is an asynchronous notification (e.g. an IRQ), not a
|
||||
/// message from a client — there is no reply owed. The low bits carry the source
|
||||
|
||||
@@ -525,6 +525,40 @@ fn exitReasonForVector(vector: u64) abi.ExitReason {
|
||||
};
|
||||
}
|
||||
|
||||
/// A fault *at* the interrupt return itself has one piece of evidence worth
|
||||
/// having: the five words the CPU refused to load. `iretq` reports a selector in
|
||||
/// the error code but never says which of them carried it, and the frame is
|
||||
/// otherwise gone the moment the machine halts — so print it while it is still
|
||||
/// on the stack. Silent for every other fault, which is nearly all of them.
|
||||
///
|
||||
/// Two honesty notes ride along. On the ring-3 path the entry stub has already
|
||||
/// swapped in the user's GS base by the time this instruction runs, so the task
|
||||
/// and core named above came from user-controlled state and mean nothing; the
|
||||
/// frame's own CS says which case this is, so say so rather than let the reader
|
||||
/// trust a number. And the stack is only read after checking it is a kernel
|
||||
/// address, because a reporter that faults tells you nothing at all.
|
||||
fn reportPendingReturnFrame(state: *const architecture.CpuState) void {
|
||||
const site = @intFromPtr(@extern(*const anyopaque, .{ .name = "isr_return_iretq" }));
|
||||
if (architecture.instructionPointer(state) != site) return;
|
||||
|
||||
const sp = architecture.stackPointer(state);
|
||||
if (sp < 0xFFFF_8000_0000_0000 or sp % 8 != 0) {
|
||||
fatalPrint(" pending return frame: stack pointer unusable\n", .{});
|
||||
return;
|
||||
}
|
||||
const frame: [*]const u64 = @ptrFromInt(sp);
|
||||
fatalPrint(" the return this refused (rip, cs, rflags, rsp, ss):\n", .{});
|
||||
fatalPrint(" rip : 0x{x:0>16}\n", .{frame[0]});
|
||||
fatalPrint(" cs : 0x{x:0>4} (rpl {d})\n", .{ frame[1], frame[1] & 3 });
|
||||
fatalPrint(" rflags : 0x{x:0>16}\n", .{frame[2]});
|
||||
fatalPrint(" rsp : 0x{x:0>16}\n", .{frame[3]});
|
||||
fatalPrint(" ss : 0x{x:0>4} (rpl {d})\n", .{ frame[4], frame[4] & 3 });
|
||||
if (frame[1] & 3 == 3) {
|
||||
fatalPrint(" note: returning to ring 3, so the user GS base is installed —\n", .{});
|
||||
fatalPrint(" the task and core reported above are not to be believed.\n", .{});
|
||||
}
|
||||
}
|
||||
|
||||
fn onException(state: *const architecture.CpuState) noreturn {
|
||||
if (architecture.fromUser(state) and scheduler.currentIsUserProcess() and recoverableFault(state.vector)) {
|
||||
statusPrint("\n/system/kernel: process {d} ({s}) killed by {s} (vector {d}) on core {d}\n", .{ scheduler.currentId(), scheduler.current().name(), architecture.exceptionName(state.vector), state.vector, scheduler.currentCpuIndex() });
|
||||
@@ -548,6 +582,7 @@ fn onException(state: *const architecture.CpuState) noreturn {
|
||||
fatalPrint(" IP : 0x{x:0>16}\n", .{architecture.instructionPointer(state)});
|
||||
fatalPrint(" SP : 0x{x:0>16}\n", .{architecture.stackPointer(state)});
|
||||
if (architecture.faultAddress(state)) |address| fatalPrint(" fault addr : 0x{x:0>16}\n", .{address});
|
||||
reportPendingReturnFrame(state);
|
||||
|
||||
var buffer: [128]u8 = undefined;
|
||||
log.recordPanic(std.fmt.bufPrint(&buffer, "CPU exception {s} (vector {d}) on core {d} at IP 0x{x}", .{ architecture.exceptionName(state.vector), state.vector, core, architecture.instructionPointer(state) }) catch "cpu exception");
|
||||
@@ -561,11 +596,23 @@ fn onException(state: *const architecture.CpuState) noreturn {
|
||||
/// Freestanding has no OS to receive a panic. Emit it to every output sink, drop a
|
||||
/// POST code + a persistent breadcrumb (so a post-mortem can recover it even with
|
||||
/// no live console), then halt. Assumes no console — the sinks self-guard.
|
||||
/// Set on entry to the panic handler, and never cleared: a panic is terminal, so
|
||||
/// the only reason to see this true is that reporting the first one faulted.
|
||||
var panicking: bool = false;
|
||||
|
||||
pub const panic = std.debug.FullPanic(struct {
|
||||
fn panic(message: []const u8, first_trace_address: ?usize) noreturn {
|
||||
_ = first_trace_address;
|
||||
// A panic while reporting a panic: stop where we are. Whatever the first
|
||||
// one already printed is the diagnosis, and every line after this point
|
||||
// risks faulting again — which on real hardware means a reset, taking the
|
||||
// message off the screen of the person who needed to read it. Halting
|
||||
// with a partial report beats rebooting with none.
|
||||
if (panicking) architecture.halt();
|
||||
panicking = true;
|
||||
|
||||
log.checkpoint(cp_panic);
|
||||
log.recordPanic(message);
|
||||
log.recordPanic(message); // the breadcrumb first: it survives what follows
|
||||
fatal("\nKERNEL PANIC: "); // a panic outranks any display service holding the screen
|
||||
fatal(message);
|
||||
fatal("\n");
|
||||
|
||||
@@ -26,6 +26,14 @@ pub const RegisterAccess = acpi.RegisterAccess;
|
||||
pub const IsoEntry = acpi.IsoEntry;
|
||||
pub const Cpu = acpi.Cpu;
|
||||
|
||||
/// Where a PCI BAR may legitimately live: the holes in the firmware memory map.
|
||||
/// Firmware-agnostic — it takes a boot-handoff map, not an ACPI table — and pure, so
|
||||
/// the kernel self-test can drive it with a synthetic map. That is the only way to
|
||||
/// check the invariant that matters here: an aperture must never cover memory the
|
||||
/// firmware described, because containment would then admit a BAR over live RAM.
|
||||
pub const AddressRange = acpi.AddressRange;
|
||||
pub const largestHolesBelow4G = acpi.largestHolesBelow4G;
|
||||
|
||||
/// The FADT power register map discovery extracted (PM1 control, reset register),
|
||||
/// for kernel reboot and diagnostics. Sleep-state values are userspace's (S5 is
|
||||
/// owned by the ring-3 acpi service), so they are not here.
|
||||
|
||||
+163
-30
@@ -190,6 +190,15 @@ pub fn init() void {
|
||||
scheduler.timer_tick_hook = timerSweepLocked;
|
||||
scheduler.group_exit_hook = groupExitLocked;
|
||||
scheduler.space_mapping_release_hook = dropSpaceMappingHook;
|
||||
// The broker owns devices; the process layer owns the task table. It asks whether a
|
||||
// lender is still alive before handing a dead driver's device back to it.
|
||||
devices_broker.task_alive_hook = taskAliveLocked;
|
||||
}
|
||||
|
||||
/// Whether `id` names a live task. The broker calls this through its hook when
|
||||
/// deciding if a dead holder's device can go back to the task that lent it.
|
||||
fn taskAliveLocked(id: u32) bool {
|
||||
return scheduler.taskByIdLocked(id) != null;
|
||||
}
|
||||
|
||||
/// Return -1 (as an unsigned bit pattern) in the system_call result register.
|
||||
@@ -249,6 +258,7 @@ fn system_call(state: *architecture.CpuState) void {
|
||||
.irq_bind => systemIrqBind(state),
|
||||
.irq_ack => systemIrqAck(state),
|
||||
.device_register => systemDeviceRegister(state),
|
||||
.device_transfer => systemDeviceTransfer(state),
|
||||
.system_spawn => systemSpawn(state),
|
||||
.dma_alloc => systemDmaAlloc(state),
|
||||
.dma_free => systemDmaFree(state),
|
||||
@@ -391,7 +401,16 @@ fn systemDeviceEnumerate(state: *architecture.CpuState) void {
|
||||
var copied: u64 = 0;
|
||||
var start: usize = 0;
|
||||
while (copied < cap) {
|
||||
const filled = devices_broker.enumerateFrom(start, &chunk);
|
||||
// Snapshot each chunk under the lock: ring-3 device_register grows the table
|
||||
// with realloc on other cores, so an unlocked read walks a slice that may
|
||||
// already have been freed — and pairs a fresh `count` with a stale slice.
|
||||
// The user copy stays outside; the chunk is the kernel's own bytes, and the
|
||||
// lock windows stay as small as one chunk.
|
||||
const filled = filled: {
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
break :filled devices_broker.enumerateFrom(start, &chunk);
|
||||
};
|
||||
if (filled == 0) break;
|
||||
start += filled;
|
||||
const take = @min(@as(u64, filled), cap - copied);
|
||||
@@ -399,39 +418,109 @@ fn systemDeviceEnumerate(state: *architecture.CpuState) void {
|
||||
if (!user_memory.copyToUser(t.address_space, buffer_ptr + copied * sz, bytes)) return failErr(state, ipc.EFAULT);
|
||||
copied += take;
|
||||
}
|
||||
architecture.setSystemCallResult(state, devices_broker.deviceCount());
|
||||
const total = total: {
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
break :total devices_broker.deviceCount();
|
||||
};
|
||||
architecture.setSystemCallResult(state, total);
|
||||
}
|
||||
|
||||
/// device_claim(id) -> 0/-1: take exclusive ownership of a device for this process.
|
||||
/// device_claim(id) -> 0/-errno: take exclusive ownership of a device for this process.
|
||||
/// `-ENODEV` no such id, `-EBUSY` a live task already owns it, `-ECONFINE` the claim
|
||||
/// could not be placed under IOMMU translation and was rolled back.
|
||||
fn systemDeviceClaim(state: *architecture.CpuState) void {
|
||||
const device_id = architecture.systemCallArg(state, 0);
|
||||
const claim_flags = sync.enter();
|
||||
defer sync.leave(claim_flags);
|
||||
if (devices_broker.claim(device_id, scheduler.current().id)) {
|
||||
// Confine the device's DMA before the driver can program it: a PCI function
|
||||
// becomes reachable to the IOMMU only once claimed (until now its DMA is
|
||||
// blocked). A claim that cannot be confined must not stand — roll it back —
|
||||
// since the whole point is that claiming a DMA device is no longer equivalent
|
||||
// to ring 0. No-op when no IOMMU exists (fail-open).
|
||||
if (devices_broker.pciAddressOf(device_id)) |bdf| {
|
||||
const owner = scheduler.current().id;
|
||||
if (!iommu.confineDevice(device_id, bdf, owner)) {
|
||||
_ = devices_broker.unclaim(device_id, owner);
|
||||
return fail(state);
|
||||
}
|
||||
// Bind the buffers this task allocated before claiming the device (a driver
|
||||
// that dma_alloc'd its rings, then claimed the controller).
|
||||
dmaBindOwnerRegionsInto(owner, device_id);
|
||||
devices_broker.claim(device_id, scheduler.current().id) catch |e|
|
||||
return failErr(state, devices_broker.claimErrnoOf(e));
|
||||
|
||||
// Confine the device's DMA before the driver can program it: a PCI function
|
||||
// becomes reachable to the IOMMU only once claimed (until now its DMA is
|
||||
// blocked). A claim that cannot be confined must not stand — roll it back —
|
||||
// since the whole point is that claiming a DMA device is no longer equivalent
|
||||
// to ring 0. No-op when no IOMMU exists (fail-open).
|
||||
if (devices_broker.pciAddressOf(device_id)) |bdf| {
|
||||
const owner = scheduler.current().id;
|
||||
if (!iommu.confineDevice(device_id, bdf, owner)) {
|
||||
_ = devices_broker.unclaim(device_id, owner);
|
||||
// Its own errno: "nobody could confine this" is a different world from
|
||||
// "someone else already has it", and a driver that cannot tell them apart
|
||||
// cannot report the one that means the machine's DMA protection ran out.
|
||||
return failErr(state, ipc.ECONFINE);
|
||||
}
|
||||
// A display service just took the framebuffer — quiesce the bootstrap console
|
||||
// so the kernel and the service don't scribble over each other's pixels. The
|
||||
// claim releases (and the console resumes) automatically if the service dies;
|
||||
// see releaseTaskResourcesLocked.
|
||||
if (devices_broker.displayDevice()) |display_id| {
|
||||
if (device_id == display_id) console.setSuppressed(true);
|
||||
}
|
||||
architecture.setSystemCallResult(state, 0);
|
||||
} else fail(state);
|
||||
// Bind the buffers this task allocated before claiming the device (a driver
|
||||
// that dma_alloc'd its rings, then claimed the controller).
|
||||
dmaBindOwnerRegionsInto(owner, device_id);
|
||||
}
|
||||
// A display service just took the framebuffer — quiesce the bootstrap console
|
||||
// so the kernel and the service don't scribble over each other's pixels. The
|
||||
// claim releases (and the console resumes) automatically if the service dies;
|
||||
// see releaseTaskResourcesLocked.
|
||||
if (devices_broker.displayDevice()) |display_id| {
|
||||
if (device_id == display_id) console.setSuppressed(true);
|
||||
}
|
||||
architecture.setSystemCallResult(state, 0);
|
||||
}
|
||||
|
||||
/// device_transfer(device_id, task_id) -> 0/-errno: give a device you hold to another
|
||||
/// task. The mechanism behind delegation — the device manager claims what discovery
|
||||
/// seeded and hands each device to the driver it matched, so assignment stops being
|
||||
/// first-come-first-served (docs/os-development/device-authority.md).
|
||||
///
|
||||
/// The kernel's whole rule is *you may give away what you hold*. It has no notion of
|
||||
/// which task is the device manager, and deliberately gains none: a binary name in the
|
||||
/// kernel is not something that cannot safely live in user space.
|
||||
fn systemDeviceTransfer(state: *architecture.CpuState) void {
|
||||
const device_id = architecture.systemCallArg(state, 0);
|
||||
const task_id: u32 = @truncate(architecture.systemCallArg(state, 1));
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
|
||||
// The recipient must exist, or the device would be moved to nobody and become
|
||||
// unreachable for the rest of the boot — no path un-holds a device but task death.
|
||||
if (scheduler.taskByIdLocked(task_id) == null) return failErr(state, ipc.ESRCH);
|
||||
|
||||
const errno = giveDeviceLocked(device_id, scheduler.current().id, task_id);
|
||||
if (errno != 0) return failErr(state, errno);
|
||||
architecture.setSystemCallResult(state, 0);
|
||||
}
|
||||
|
||||
/// Move a device from `from` to `to` **with its IOMMU confinement** — the shared body
|
||||
/// of `device_transfer` and spawn's give. Returns 0 or the errno to refuse with.
|
||||
/// Caller holds the big kernel lock, and has verified the recipient exists.
|
||||
///
|
||||
/// The confinement must move with the device. Usually the giver confined it at claim,
|
||||
/// so the domain exists, the device stays attached throughout, and `reassign` re-points
|
||||
/// the record. But after a driver's death the loan came back with the domain torn down
|
||||
/// (`releaseAllOwnedBy` runs before the broker returns the device to its lender) — so a
|
||||
/// re-delegation finds no active record, and a bare `reassign` would no-op and hand the
|
||||
/// device over silently unconfined: V=0 device-table entry, no domain, every later
|
||||
/// `dma_alloc` bound into nothing. That is the exact fail-open the fail-closed claim
|
||||
/// exists to remove, so the same rule applies here: confine afresh, and a give that
|
||||
/// cannot be confined must not stand.
|
||||
///
|
||||
/// Order matters: the checks run first (nothing has mutated, nothing to roll back),
|
||||
/// the confinement second (its failure refuses cleanly), the move last (it cannot fail
|
||||
/// once `canTransfer` passed — same lock hold). Public for the in-kernel iommu test,
|
||||
/// which drives the death-and-respawn sequence against it directly.
|
||||
pub fn giveDeviceLocked(device_id: u64, from: u32, to: u32) i64 {
|
||||
devices_broker.canTransfer(device_id, from) catch |e|
|
||||
return devices_broker.transferErrnoOf(e);
|
||||
if (devices_broker.pciAddressOf(device_id)) |bdf| {
|
||||
if (iommu.confinementOwner(device_id) == null and !iommu.confineDevice(device_id, bdf, to))
|
||||
return ipc.ECONFINE;
|
||||
}
|
||||
devices_broker.transfer(device_id, from, to) catch |e|
|
||||
return devices_broker.transferErrnoOf(e); // unreachable: checked above under this lock
|
||||
if (devices_broker.pciAddressOf(device_id)) |_| {
|
||||
iommu.reassign(device_id, to);
|
||||
// Bind whatever the receiver has already allocated — the same courtesy the
|
||||
// claim path does for a driver that dma_alloc'd its rings before claiming.
|
||||
dmaBindOwnerRegionsInto(to, device_id);
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// mmio_map(device_id, resource_index) -> virtual_address: map a claimed device's MMIO window into
|
||||
@@ -931,17 +1020,21 @@ fn systemDeviceRegister(state: *architecture.CpuState) void {
|
||||
const parent_id = architecture.systemCallArg(state, 0);
|
||||
const descriptor_ptr = architecture.systemCallArg(state, 1);
|
||||
const t = scheduler.current();
|
||||
if (t.address_space == 0) return fail(state);
|
||||
if (t.address_space == 0) return failErr(state, ipc.EPERM);
|
||||
|
||||
var descriptor: device_abi.DeviceDescriptor = undefined;
|
||||
if (!ipc.copyFromUser(t.address_space, descriptor_ptr, std.mem.asBytes(&descriptor))) return fail(state);
|
||||
if (!ipc.copyFromUser(t.address_space, descriptor_ptr, std.mem.asBytes(&descriptor))) return failErr(state, ipc.EFAULT);
|
||||
|
||||
// Under the big kernel lock: the broker's table is also mutated by the
|
||||
// death sweep (releaseAllOwnedBy) and read by enumerate on other cores —
|
||||
// ring-3 registration (M19) made those genuinely concurrent.
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
const id = devices_broker.register(parent_id, t.id, &descriptor) catch return fail(state);
|
||||
// Each refusal carries its own errno (devices-broker.errnoOf) — the bus driver
|
||||
// logs which rule stopped it, so "this parent is full" is never again mistaken
|
||||
// for "the table is full" or "that resource escapes your window".
|
||||
const id = devices_broker.register(parent_id, t.id, &descriptor) catch |e|
|
||||
return failErr(state, devices_broker.errnoOf(e));
|
||||
architecture.setSystemCallResult(state, id);
|
||||
}
|
||||
|
||||
@@ -971,6 +1064,13 @@ fn systemSpawn(state: *architecture.CpuState) void {
|
||||
const arguments_ptr = architecture.systemCallArg(state, 2);
|
||||
const arguments_len = architecture.systemCallArg(state, 3);
|
||||
const exit_handle = architecture.systemCallArg(state, 4);
|
||||
// A device the caller holds and gives to the child. Fused into the spawn rather
|
||||
// than transferred after it, because a separate transfer leaves a window in which
|
||||
// the child is running and does not yet hold its device — a race that would close
|
||||
// on one machine and open on another, which is the failure shape this whole track
|
||||
// exists to remove (docs/bounds-track-plan.md, "the grant rides system_spawn").
|
||||
// Here the child cannot observe the gap: it does not exist until it holds it.
|
||||
const device_to_give = architecture.systemCallArg(state, 5);
|
||||
const t = scheduler.current();
|
||||
if (len == 0 or len > scheduler.maximum_task_name or ptr >= user_half_end or ptr + len > user_half_end) return fail(state);
|
||||
if (arguments_len > maximum_argument_bytes) return fail(state);
|
||||
@@ -1015,7 +1115,40 @@ fn systemSpawn(state: *architecture.CpuState) void {
|
||||
}
|
||||
}
|
||||
|
||||
// Refuse before creating anything if the device is not the caller's to give — a
|
||||
// spawn that half-succeeds would leave a child running without the hardware it was
|
||||
// spawned for, which is worse than not spawning it. Read under the lock; it drops
|
||||
// across the spawn, so the give below re-checks under its own hold — this one only
|
||||
// spares creating a child that was always going to be killed.
|
||||
if (device_to_give != abi.no_device) {
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
devices_broker.canTransfer(device_to_give, t.id) catch |e|
|
||||
return failErr(state, devices_broker.transferErrnoOf(e));
|
||||
}
|
||||
|
||||
const child = spawnProcessSupervised(item.blob, 4, argv[0..argc], t.id, exit_endpoint) catch return fail(state);
|
||||
|
||||
if (device_to_give != abi.no_device) {
|
||||
// The give runs in one lock hold: check, confine, move (giveDeviceLocked).
|
||||
// spawnProcessSupervised took and released the lock internally, so the
|
||||
// pre-check above holds no authority here. The child cannot outrun the
|
||||
// hand-over — its first device syscall serializes behind this same lock.
|
||||
const errno = give: {
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
break :give giveDeviceLocked(device_to_give, t.id, child);
|
||||
};
|
||||
if (errno != 0) {
|
||||
// The device stopped being the caller's between the pre-check and here, or
|
||||
// its confinement was refused. A child running without the hardware it was
|
||||
// spawned for is worse than no child — undo the spawn. The kill's ownership
|
||||
// sweep also releases anything the give half-did (a fresh confinement dies
|
||||
// with the child).
|
||||
_ = killProcess(t.id, child);
|
||||
return failErr(state, errno);
|
||||
}
|
||||
}
|
||||
architecture.setSystemCallResult(state, child);
|
||||
}
|
||||
|
||||
|
||||
+335
-10
@@ -51,6 +51,15 @@ fn check(name: []const u8, ok: bool) void {
|
||||
}
|
||||
}
|
||||
|
||||
/// `devices_broker.claim` reduced to a bool, for the `check` assertions below. The
|
||||
/// broker returns `ClaimError` so ring 3 can tell "stale id" from "someone already
|
||||
/// owns it" (the errno space in system/abi.zig); a test that only asserts the claim
|
||||
/// succeeded does not care which, and the ones that do match the error directly.
|
||||
fn claimOk(id: u64, owner: u32) bool {
|
||||
devices_broker.claim(id, owner) catch return false;
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Emit the overall result line the harness matches, then the done sentinel.
|
||||
fn result() void {
|
||||
log("DANOS-TEST-RESULT: {s} ({d} passed, {d} failed)\n", .{
|
||||
@@ -250,6 +259,12 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
|
||||
irqFreeTest();
|
||||
} else if (eql(case, "containment")) {
|
||||
containmentTest();
|
||||
} else if (eql(case, "apertures")) {
|
||||
apertureTest();
|
||||
} else if (eql(case, "device-transfer")) {
|
||||
deviceTransferTest(boot_information);
|
||||
} else if (eql(case, "device-authority")) {
|
||||
deviceAuthorityTest(boot_information);
|
||||
} else if (eql(case, "device-manager")) {
|
||||
deviceManagerTest(boot_information);
|
||||
} else if (eql(case, "protocol-registry")) {
|
||||
@@ -1432,6 +1447,79 @@ fn iommuTest() void {
|
||||
} else {
|
||||
check("scratch domain allocated", false);
|
||||
}
|
||||
// Delegation moves a device between tasks, and the IOMMU confinement must move
|
||||
// with it. When it did not, the driver's DMA rings were never bound into the
|
||||
// device's domain and every transfer faulted — but two worse consequences were
|
||||
// latent and invisible to those tests: the domain still named the *giver*, so the
|
||||
// giver's death would tear down a domain a live driver was using, and the
|
||||
// receiver's death would leave one behind. Asserted directly here rather than
|
||||
// inferred from the USB cases going green.
|
||||
// This case runs no pci-bus, so there are no PCI functions to confine — and a
|
||||
// silently skipped assertion is worse than none. Register one the way pci-bus does:
|
||||
// a child of the host bridge whose resource 0 is a 4 KiB config window inside the
|
||||
// bridge's ECAM, which is what `pciAddressOf` derives a requester id from.
|
||||
var iommu_table: [64]device_abi.DeviceDescriptor = undefined;
|
||||
const iommu_total = devices_broker.enumerate(&iommu_table);
|
||||
const bridge: ?device_abi.DeviceDescriptor = for (iommu_table[0..@min(iommu_total, iommu_table.len)]) |d| {
|
||||
if (d.class == @intFromEnum(device_abi.DeviceClass.pci_host_bridge) and d.resource_count >= 2) break d;
|
||||
} else null;
|
||||
check("the kernel seeded a PCI host bridge to parent a function under", bridge != null);
|
||||
|
||||
const subject: ?u64 = if (bridge) |b| blk: {
|
||||
const me_bridge = scheduler.currentId();
|
||||
if (!claimOk(b.id, me_bridge)) break :blk null;
|
||||
var function = std.mem.zeroes(device_abi.DeviceDescriptor);
|
||||
function.class = @intFromEnum(device_abi.DeviceClass.pci_device);
|
||||
function.pci_class = device_abi.no_pci_class;
|
||||
function.resource_count = 1;
|
||||
function.resources[0] = .{
|
||||
.kind = @intFromEnum(device_abi.ResourceKind.memory),
|
||||
.start = b.resources[0].start, // the first config slot in the ECAM window
|
||||
.len = 4096,
|
||||
};
|
||||
break :blk devices_broker.register(b.id, me_bridge, &function) catch null;
|
||||
} else null;
|
||||
check("a PCI function exists to confine", subject != null);
|
||||
|
||||
if (subject) |device_id| {
|
||||
check("a device starts unconfined", iommu.confinementOwner(device_id) == null);
|
||||
const me = scheduler.currentId();
|
||||
check("confining records the owner", iommu.confineDevice(device_id, devices_broker.pciAddressOf(device_id).?, me));
|
||||
check("the confinement names the confiner", iommu.confinementOwner(device_id) == me);
|
||||
iommu.reassign(device_id, me + 1000);
|
||||
check("reassign moves the confinement to the new holder", iommu.confinementOwner(device_id) == me + 1000);
|
||||
check("and the previous holder no longer owns it", iommu.confinementOwner(device_id) != me);
|
||||
iommu.releaseAllOwnedBy(me + 1000);
|
||||
check("the new holder's death tears the domain down", iommu.confinementOwner(device_id) == null);
|
||||
|
||||
// The restart hole. A driver's death returns its device to the lender with the
|
||||
// domain torn down (asserted just above) — so the NEXT delegation of the same
|
||||
// device finds no active confinement record, and the transfer path's bare
|
||||
// `reassign` no-ops: the respawned driver would run the device with a V=0
|
||||
// device-table entry and no domain, silently unconfined. The give path must
|
||||
// confine afresh in that case, exactly as a first claim would.
|
||||
check("the lender holds the returned device", claimOk(device_id, me));
|
||||
const driver: u32 = me + 2000;
|
||||
check("delegating it confines it to the receiver", process.giveDeviceLocked(device_id, me, driver) == 0);
|
||||
check("the give's confinement names the receiver", iommu.confinementOwner(device_id) == driver);
|
||||
// The receiver dies: confinement torn down first, then the broker loans the
|
||||
// device back to the lender — the same order releaseTaskResourcesLocked runs.
|
||||
iommu.releaseAllOwnedBy(driver);
|
||||
devices_broker.releaseAllOwnedBy(driver);
|
||||
check("death returns the loan to the lender", devices_broker.ownerOf(device_id) == me);
|
||||
check("and leaves the device unconfined", iommu.confinementOwner(device_id) == null);
|
||||
// The regression this guards: re-delegation after that death.
|
||||
const respawned: u32 = me + 3000;
|
||||
check("re-delegation after the death succeeds", process.giveDeviceLocked(device_id, me, respawned) == 0);
|
||||
check(
|
||||
"and the respawned driver's device is confined, not silently naked",
|
||||
iommu.confinementOwner(device_id) == respawned,
|
||||
);
|
||||
iommu.releaseAllOwnedBy(respawned);
|
||||
devices_broker.releaseAllOwnedBy(respawned);
|
||||
_ = devices_broker.unclaim(device_id, me);
|
||||
}
|
||||
|
||||
log("DANOS-IOMMU: enabled base=0x{x} domains active\n", .{pinfo.iommu_base});
|
||||
result();
|
||||
}
|
||||
@@ -1475,7 +1563,7 @@ fn ioPortTest() void {
|
||||
check("discovered the acpi-tables I/O window", true);
|
||||
|
||||
const me = scheduler.current();
|
||||
check("claimed the io_port device", devices_broker.claim(id, me.id));
|
||||
check("claimed the io_port device", claimOk(id, me.id));
|
||||
check("an in-range access resolves to port 0x64", process.resolveIoPort(me, id, found_res, 0x64, 1) == 0x64);
|
||||
check("a 4-byte access at the last port is refused", process.resolveIoPort(me, id, found_res, 0xFFFF, 4) == null);
|
||||
check("an out-of-range offset is refused", process.resolveIoPort(me, id, found_res, 0x10000, 1) == null);
|
||||
@@ -2470,8 +2558,8 @@ fn claimReleaseTest(boot_information: *const BootInformation) void {
|
||||
}
|
||||
|
||||
// The broker release in isolation.
|
||||
check("device 0 claimed by owner 111", devices_broker.claim(0, 111));
|
||||
check("device 1 claimed by owner 222", devices_broker.claim(1, 222));
|
||||
check("device 0 claimed by owner 111", claimOk(0, 111));
|
||||
check("device 1 claimed by owner 222", claimOk(1, 222));
|
||||
devices_broker.releaseAllOwnedBy(111);
|
||||
check("owner 111's claim is released", devices_broker.ownerOf(0) == null);
|
||||
check("owner 222's claim survives", (devices_broker.ownerOf(1) orelse 0) == 222);
|
||||
@@ -2493,7 +2581,7 @@ fn claimReleaseTest(boot_information: *const BootInformation) void {
|
||||
};
|
||||
const child = process.spawnProcessSupervised(image, 4, &.{"/system/services/init"}, me, endpoint) catch 0;
|
||||
check("supervised child spawned", child != 0);
|
||||
check("device 0 claimed on the child's behalf", devices_broker.claim(0, child));
|
||||
check("device 0 claimed on the child's behalf", claimOk(0, child));
|
||||
|
||||
check("the kill is accepted", process.killProcess(me, child) == 0);
|
||||
var badge: u64 = 0;
|
||||
@@ -2501,7 +2589,7 @@ fn claimReleaseTest(boot_information: *const BootInformation) void {
|
||||
_ = ipcsync.replyWait(endpoint, 0, 0, 0, 0, abi.no_cap, &badge, &received_cap);
|
||||
check("the exit notification arrived", badge == abi.notify_badge_bit | abi.notify_exit_bit | child);
|
||||
check("death released the child's claim", devices_broker.ownerOf(0) == null);
|
||||
check("the device is claimable again", devices_broker.claim(0, me));
|
||||
check("the device is claimable again", claimOk(0, me));
|
||||
devices_broker.releaseAllOwnedBy(me);
|
||||
result();
|
||||
}
|
||||
@@ -2690,7 +2778,7 @@ fn usbReportTest(boot_information: *const BootInformation) void {
|
||||
};
|
||||
|
||||
process.setInitialRamdisk(image);
|
||||
_ = spawnRegistry(rd); // the xhci driver binds /protocol/usb-transfer
|
||||
_ = spawnRegistry(rd); // drivers reach /protocol/device-manager through it
|
||||
var manager: u32 = 0;
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
@@ -3037,7 +3125,17 @@ fn acpiParseTest(boot_information: *const BootInformation) void {
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
if (!eql(initial_ramdisk.basename(item.name), "discovery")) continue;
|
||||
_ = process.spawnProcessSupervised(item.blob, 4, &.{ item.name, "1" }, scheduler.currentId(), null) catch 0;
|
||||
// Hand it the acpi-tables node the way the manager would: claim it here, then
|
||||
// move it to the child. Discovery no longer claims for itself.
|
||||
var scratch: [64]device_abi.DeviceDescriptor = undefined;
|
||||
const seen_devices = devices_broker.enumerate(&scratch);
|
||||
const tables: ?u64 = for (scratch[0..@min(seen_devices, scratch.len)]) |d| {
|
||||
if (d.class == @intFromEnum(device_abi.DeviceClass.acpi_tables)) break d.id;
|
||||
} else null;
|
||||
const me_parse = scheduler.currentId();
|
||||
if (tables) |node| _ = claimOk(node, me_parse);
|
||||
const child = process.spawnProcessSupervised(item.blob, 4, &.{ item.name, "floor:1" }, me_parse, null) catch 0;
|
||||
if (tables) |node| devices_broker.transfer(node, me_parse, child) catch {};
|
||||
spawned = true;
|
||||
break;
|
||||
}
|
||||
@@ -3765,8 +3863,8 @@ fn killThreadedGroupTest(boot_information: *const BootInformation) void {
|
||||
// Claims for BOTH members: group death must release every member's claims
|
||||
// before the supervisor hears anything — the worker's by the deferred
|
||||
// (condemned) path.
|
||||
check("device 0 claimed for the leader", devices_broker.claim(0, child));
|
||||
check("device 1 claimed for the worker", devices_broker.claim(1, worker));
|
||||
check("device 0 claimed for the leader", claimOk(0, child));
|
||||
check("device 1 claimed for the worker", claimOk(1, worker));
|
||||
scheduler.sleep(100); // let the worker really be running on another core
|
||||
check("the supervisor's kill is accepted", process.killProcess(me, child) == 0);
|
||||
const badge = awaitExitBadge(endpoint);
|
||||
@@ -3954,7 +4052,7 @@ fn containmentTest() void {
|
||||
result();
|
||||
return;
|
||||
}
|
||||
check("claimed the parent device", devices_broker.claim(parent_id, me));
|
||||
check("claimed the parent device", claimOk(parent_id, me));
|
||||
defer devices_broker.releaseAllOwnedBy(me);
|
||||
|
||||
// A child whose window lies inside the parent's is accepted.
|
||||
@@ -3976,10 +4074,185 @@ fn containmentTest() void {
|
||||
check("re-registering an identical child returns the same id", again != 0 and again == good);
|
||||
check("re-registering grew nothing", devices_broker.enumerate(&buffer) == before + 1);
|
||||
|
||||
// **There is no per-parent cap.** `maximum_children_per_parent = 16` is gone: it
|
||||
// was written to stop a driver looping `device_register` and exhausting a shared
|
||||
// table, and there is no shared table to exhaust — the table grows, and each
|
||||
// registrar has its own allowance, so a runaway costs only itself. The cap never
|
||||
// bounded a determined caller anyway (16 per parent, but nothing stopped it
|
||||
// claiming more parents); what it reliably did was refuse a real PCI bus with more
|
||||
// than 16 functions, which is how an AMD Ryzen booted with no USB and no storage.
|
||||
//
|
||||
// 64 children under one parent — four times the old ceiling.
|
||||
var filled: u32 = 0;
|
||||
var registered_children: u32 = 0;
|
||||
while (filled < 64) : (filled += 1) {
|
||||
var name: [4]u8 = .{ 'k', 0, 0, 0 };
|
||||
name[1] = '0' + @as(u8, @intCast(filled / 10));
|
||||
name[2] = '0' + @as(u8, @intCast(filled % 10));
|
||||
var extra = childDescriptor(name[0..3], parent_window.start, 0x20);
|
||||
if (devices_broker.register(parent_id, me, &extra)) |_| {
|
||||
registered_children += 1;
|
||||
} else |_| break;
|
||||
}
|
||||
check("one parent takes far more children than the old cap allowed", registered_children == 64);
|
||||
|
||||
// Idempotency still holds, and still matters: a crashed bus driver is restarted and
|
||||
// re-registers everything it rediscovers, which must return the ids it had before
|
||||
// rather than duplicate them.
|
||||
const readmitted = devices_broker.register(parent_id, me, &fits) catch 0;
|
||||
check("re-registering an identical child still returns its id", readmitted != 0 and readmitted == good);
|
||||
|
||||
// The table itself has no ceiling: it grows. The old `maximum_devices = 64` was a
|
||||
// guess about someone else's computer, and one driver's enumeration starved every
|
||||
// other — which is how a Ryzen booted with no USB and no storage. What bounds a
|
||||
// runaway now is an allowance charged to the registrar, so the damage stays with
|
||||
// whoever caused it (docs/os-development/device-authority.md).
|
||||
// The table grows: it starts at 8 entries and this boot holds well past that, so
|
||||
// the growth path runs every time rather than lying dormant until someone else's
|
||||
// larger machine finds it — which is how the old ceiling stayed invisible.
|
||||
const held = devices_broker.enumerate(&buffer);
|
||||
check("the table grew beyond its initial block", held > 8);
|
||||
|
||||
result();
|
||||
}
|
||||
|
||||
/// A minimal child descriptor with one memory resource, for the containment test.
|
||||
/// Delegation's mechanism. A claim is exclusive (driver-model.md, invariant 1), so
|
||||
/// handing a device on is a **move**: the giver stops holding it the instant the
|
||||
/// receiver starts. That is why this is not the M13 capability path, which shares a
|
||||
/// handle refcounted.
|
||||
///
|
||||
/// The rule the kernel enforces is the whole of it: *you may give away what you hold*.
|
||||
/// It has no idea which task is the device manager and needs none
|
||||
/// (docs/os-development/device-authority.md).
|
||||
fn deviceTransferTest(boot_information: *const boot_handoff.BootInformation) void {
|
||||
var buffer: [8]device_abi.DeviceDescriptor = undefined;
|
||||
check("the device tree is seeded", devices_broker.enumerate(&buffer) >= 2);
|
||||
|
||||
const image = bundledInit(boot_information) orelse {
|
||||
check("initial_ramdisk carries /system/services/init", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
const me = scheduler.currentId();
|
||||
const endpoint = ipcsync.createIpcEndpoint() orelse {
|
||||
check("exit endpoint allocated", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
const child = process.spawnProcessSupervised(image, 4, &.{"/system/services/init"}, me, endpoint) catch 0;
|
||||
check("supervised child spawned", child != 0);
|
||||
|
||||
// You may only give away what you hold — so an unheld device cannot be moved at all,
|
||||
// which is what stops a transfer being a back door around claiming.
|
||||
const unheld = if (devices_broker.transfer(0, me, child)) |_| false else |e| e == error.NotHeld;
|
||||
check("an unheld device cannot be transferred", unheld);
|
||||
|
||||
// A second device this test never hands over, so the "no giver" half below is a
|
||||
// real observation rather than a restatement of the first.
|
||||
const parentless: u64 = 1;
|
||||
|
||||
check("claimed device 0", claimOk(0, me));
|
||||
|
||||
// The move itself.
|
||||
const moved = if (devices_broker.transfer(0, me, child)) |_| true else |_| false;
|
||||
check("the holder may transfer", moved);
|
||||
const holder = devices_broker.ownerOf(0) orelse 0;
|
||||
check("the receiver holds it", holder == child);
|
||||
check("the giver does not", holder != me);
|
||||
|
||||
// Having given it away, the giver cannot give it again. This is the assertion that
|
||||
// makes it a move rather than a copy.
|
||||
const again = if (devices_broker.transfer(0, me, child)) |_| false else |e| e == error.NotHeld;
|
||||
check("a former holder cannot transfer again", again);
|
||||
|
||||
// Nor may a stranger move a device it never held.
|
||||
const stranger = if (devices_broker.transfer(0, 9999, me)) |_| false else |e| e == error.NotHeld;
|
||||
check("a stranger cannot transfer another task's device", stranger);
|
||||
|
||||
// The giver is recorded. One field, and the authority rule follows from it: a
|
||||
// device that was *given* to someone is delegated hardware, so it may be handed on
|
||||
// but never taken, and when its holder dies it returns to whoever lent it instead
|
||||
// of becoming free for anyone. Nothing has a giver until it is handed over —
|
||||
// which is why the loader's framebuffer needs no exemption from either rule.
|
||||
check("the giver is recorded on a transfer", devices_broker.giverOf(0) == me);
|
||||
check("a device nobody handed over has no giver", devices_broker.giverOf(parentless) == null);
|
||||
|
||||
const absent = if (devices_broker.transfer(9999, me, child)) |_| false else |e| e == error.NoSuchDevice;
|
||||
check("a device that does not exist is refused", absent);
|
||||
|
||||
// **A grant is a loan.** The child holds device 0 and `me` lent it, so the child's
|
||||
// death must hand it back rather than release it to nobody. That is what lets the
|
||||
// device manager re-delegate to a restarted driver — and, once `claim` refuses a
|
||||
// device that has a giver, it is the only thing that stops a dead driver's hardware
|
||||
// being stranded forever.
|
||||
check("the child still holds the lent device", devices_broker.ownerOf(0) == child);
|
||||
devices_broker.releaseAllOwnedBy(child);
|
||||
check("the borrower's death returns the device to its lender", devices_broker.ownerOf(0) == me);
|
||||
check("and it is no longer on loan", devices_broker.giverOf(0) == null);
|
||||
|
||||
// A device with no lender still simply frees on death, as it always did.
|
||||
check("claimed a second device with no lender", claimOk(parentless, me));
|
||||
devices_broker.releaseAllOwnedBy(me);
|
||||
check("a device nobody lent is released outright", devices_broker.ownerOf(parentless) == null);
|
||||
result();
|
||||
}
|
||||
|
||||
/// PCI host-bridge apertures are derived from the *holes* in the firmware memory map,
|
||||
/// and a registered BAR must fall inside one. So the invariant is not "we find the
|
||||
/// holes" but "an aperture never covers memory the firmware described" — an aperture
|
||||
/// over RAM means `device_register` containment admits a child BAR over kernel memory,
|
||||
/// and its claimant can `mmio_map` it.
|
||||
///
|
||||
/// The map's length is the firmware's choice: 60–200 descriptors on a real machine,
|
||||
/// 15–25 under OVMF, which is why the suite never saw this. The derivation used to
|
||||
/// copy sub-4 GiB entries into a fixed `[64]` array and skip the rest — and a skipped
|
||||
/// region is not merely lost, it is one the gap finder concludes is *free*.
|
||||
fn apertureTest() void {
|
||||
// 100 described one-page regions, 2 MiB apart: 99 small holes between them, then
|
||||
// one large hole from the last region up to 4 GiB. Under the old fixed array the
|
||||
// 36 regions past the 64th vanished, so the "largest hole" ran from ~128 MiB to
|
||||
// 4 GiB — straight across 36 regions the firmware had described.
|
||||
const spacing: u64 = 2 << 20;
|
||||
var regions: [100]boot_handoff.MemoryRegion = undefined;
|
||||
for (®ions, 0..) |*region, i| {
|
||||
region.* = .{ .base = @as(u64, i) * spacing, .pages = 1, .kind = .usable };
|
||||
}
|
||||
|
||||
var holes: [3]platform.AddressRange = undefined;
|
||||
const found = platform.largestHolesBelow4G(®ions, 1 << 20, &holes);
|
||||
check("apertures were derived from a 100-entry map", found > 0);
|
||||
|
||||
var overlaps: usize = 0;
|
||||
var largest: platform.AddressRange = .{ .base = 0, .end = 0 };
|
||||
for (holes) |hole| {
|
||||
if (hole.end <= hole.base) continue;
|
||||
if (hole.end - hole.base > largest.end - largest.base) largest = hole;
|
||||
for (regions) |region| {
|
||||
const region_end = region.base + region.pages * 4096;
|
||||
if (region.base < hole.end and region_end > hole.base) overlaps += 1;
|
||||
}
|
||||
}
|
||||
check("no aperture overlaps described memory", overlaps == 0);
|
||||
|
||||
// The big hole is above the last described region, not across it.
|
||||
const last_end = (regions.len - 1) * spacing + 4096;
|
||||
check("the largest aperture starts after the last described region", largest.base == last_end);
|
||||
check("the largest aperture runs to 4 GiB", largest.end == (1 << 32));
|
||||
|
||||
// A map the firmware describes nothing in is one whole hole; a map that describes
|
||||
// everything has none. Neither may invent an aperture over something described.
|
||||
var empty: [3]platform.AddressRange = undefined;
|
||||
check("an empty map yields one hole", platform.largestHolesBelow4G(&.{}, 1 << 20, &empty) == 1);
|
||||
check("that hole is the whole low space", empty[0].base == 0 and empty[0].end == (1 << 32));
|
||||
|
||||
const whole = [_]boot_handoff.MemoryRegion{.{ .base = 0, .pages = (1 << 32) / 4096, .kind = .usable }};
|
||||
var none: [3]platform.AddressRange = undefined;
|
||||
check("a fully described map yields no aperture", platform.largestHolesBelow4G(&whole, 1 << 20, &none) == 0);
|
||||
|
||||
result();
|
||||
}
|
||||
|
||||
fn childDescriptor(hid: []const u8, start: u64, len: u64) device_abi.DeviceDescriptor {
|
||||
var child = std.mem.zeroes(device_abi.DeviceDescriptor);
|
||||
child.class = @intFromEnum(device_abi.DeviceClass.unknown);
|
||||
@@ -4065,6 +4338,58 @@ fn protocolRegistryTest(boot_information: *const BootInformation) void {
|
||||
///
|
||||
/// The fixture's `protocol-denied: ok` is the marker; each step prints its own
|
||||
/// line, which the harness's ordered regex reads.
|
||||
/// The attacker the device suite never had. The audit's finding was that a fully
|
||||
/// green suite had missed six real defects because it *contains no attacker* — every
|
||||
/// device case asserts a driver handed its hardware can drive it, and none asks what a
|
||||
/// process handed **nothing** can do.
|
||||
///
|
||||
/// The fixture is spawned with no device and asserts what it therefore cannot do. It
|
||||
/// runs without the device manager on purpose: nothing here needs a driver, and a boot
|
||||
/// with fewer moving parts makes the refusals unambiguous.
|
||||
fn deviceAuthorityTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: device-authority\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over an initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
|
||||
process.setInitialRamdisk(image);
|
||||
// The manager must be up: it is what holds the seeded hardware, and without it
|
||||
// every device would be lying around unheld and the last assertion would have
|
||||
// nothing to observe — a test that cannot fail.
|
||||
check("registry (init) spawned", spawnRegistry(rd));
|
||||
check("device-manager spawned", spawnNamed(rd, "device-manager"));
|
||||
check("device-authority-test spawned", spawnNamedWithArg(rd, "device-authority-test", "run"));
|
||||
|
||||
// The VERDICT prefix matters: the fixture prints one "device-authority: ok <name>"
|
||||
// line per assertion, so a marker of "device-authority: ok" matches the FIRST
|
||||
// passing assertion and this loop exits before any later failure is printed — the
|
||||
// case then passes with failures in it, which it did until this was caught.
|
||||
const pass_marker = "device-authority: VERDICT ok";
|
||||
const fail_marker = "device-authority: VERDICT FAILED";
|
||||
scheduler.setPriority(1);
|
||||
const deadline = architecture.millis() + 20000;
|
||||
var saw_pass = false;
|
||||
var saw_fail = false;
|
||||
while (architecture.millis() < deadline and !saw_pass and !saw_fail) {
|
||||
if (bufferHas(pass_marker)) saw_pass = true;
|
||||
if (bufferHas(fail_marker)) saw_fail = true;
|
||||
scheduler.yield();
|
||||
}
|
||||
scheduler.setPriority(4);
|
||||
|
||||
check("no authority assertion failed", !saw_fail);
|
||||
check("the attacker completed every assertion", saw_pass);
|
||||
result();
|
||||
}
|
||||
|
||||
fn protocolDeniedTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: protocol-denied\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
|
||||
+26
-4
@@ -4,15 +4,26 @@
|
||||
//! hiding the trade-offs. Keeping them here makes them visible at a glance and gives
|
||||
//! one spot to change them. They're plain `comptime` constants (zero runtime cost);
|
||||
//! any one can later be promoted to a `-D` build option if a target needs to vary it
|
||||
//! (see build.zig's `-Dtest-case` for the pattern). This keeps ps2-library.zig to what it
|
||||
//! actually is — the bootloader↔kernel handoff *contract* — with tunables living here.
|
||||
//! (see build.zig's `-Dtest-case` for the pattern). This keeps [[boot-handoff]] to what
|
||||
//! it actually is — the loader↔kernel handoff *contract* — with tunables living here.
|
||||
//!
|
||||
//! **Kernel only, and deliberately so.** This file exists because tunables were
|
||||
//! crowding the loader↔kernel contract they were split out of; it is not a registry for
|
||||
//! the whole system. A driver's ring size belongs to that driver, a protocol's payload
|
||||
//! cap to that protocol. How a ceiling is *declared*, wherever it lives, is
|
||||
//! docs/os-development/bounds.md — a shape, not a shared list.
|
||||
|
||||
/// Ceiling on logical CPUs the kernel tracks — the size of the per-CPU bookkeeping
|
||||
/// arrays (discovery pool, scheduler state, per-core GDT/TSS). Generous headroom:
|
||||
/// those structs are small, and the *large* per-core resources (kernel and IST
|
||||
/// stacks) are allocated at bring-up for cores that actually come online, so this
|
||||
/// ceiling is cheap. A machine with more logical CPUs has its surplus reported and
|
||||
/// left parked (see acpi `cpusDropped`).
|
||||
/// ceiling is cheap.
|
||||
///
|
||||
/// bound: logical CPUs the kernel tracks
|
||||
/// decided-by: hardware
|
||||
/// protects: the per-CPU bookkeeping arrays, which are sized at compile time
|
||||
/// at-limit: degrade — the surplus cores are left parked, never brought online
|
||||
/// observed-by: platform.cpusDropped() -> the WARNING at kernel.zig:281
|
||||
pub const maximum_cpus = 128;
|
||||
|
||||
/// Maximum tasks (kernel threads) alive at once — the static task-table size. Each
|
||||
@@ -23,6 +34,17 @@ pub const maximum_cpus = 128;
|
||||
/// for the USB stack: the xHCI bus driver spawns a supervised class-driver instance
|
||||
/// per matched interface (keyboard, mouse, mass storage), on top of the FAT and
|
||||
/// block servers and the growing ramdisk bundle.
|
||||
///
|
||||
/// The history above is the argument against this number: it has been raised twice,
|
||||
/// each time by a machine or a bundle that outgrew it, which is the pattern the
|
||||
/// bounds rule exists to stop. It is `ours` only because the task table is static;
|
||||
/// how many drivers a machine needs is decided by how much hardware it has.
|
||||
///
|
||||
/// bound: kernel threads alive at once — the static task-table size
|
||||
/// decided-by: hardware
|
||||
/// protects: the statically allocated task table
|
||||
/// at-limit: refuse — spawn fails; a supervised driver is never started
|
||||
/// observed-by: the spawning supervisor's own log line; see docs/bounds-track-plan.md
|
||||
pub const maximum_tasks = 48;
|
||||
|
||||
/// Each task's kernel stack (also each AP's bring-up stack), in bytes.
|
||||
|
||||
@@ -104,11 +104,19 @@ fn findTablesNode(buffer: []device.DeviceDescriptor) ?device.DeviceDescriptor {
|
||||
}
|
||||
|
||||
pub fn main(init: process.Init) void {
|
||||
// When the acpi-parse scenario spawns this directly, argv[1] is a device-count
|
||||
// *floor* to self-verify against. The kernel no longer parses AML, so there is
|
||||
// no exact count to match — proving the ring-3 parse found at least a floor of
|
||||
// devices is the check. Deterministic, no log-scraping.
|
||||
const floor: ?usize = if (init.arguments.get(1)) |a| (std.fmt.parseInt(usize, a, 10) catch null) else null;
|
||||
// When the acpi-parse scenario spawns this directly, it passes `floor:N` — a
|
||||
// device-count floor to self-verify against. The kernel no longer parses AML, so
|
||||
// there is no exact count to match; proving the ring-3 parse found at least N
|
||||
// Device objects is the check. Deterministic, no log-scraping.
|
||||
//
|
||||
// The `floor:` prefix matters. argv[1] is the assigned device id for every driver
|
||||
// the manager spawns, so a bare number here would be read as a floor — which is
|
||||
// exactly what happened when discovery started being given its node: it saw
|
||||
// argv[1] = "7", decided it was in self-verify mode, and never reported a device.
|
||||
const floor: ?usize = if (init.arguments.get(1)) |a| blk: {
|
||||
if (!std.mem.startsWith(u8, a, "floor:")) break :blk null;
|
||||
break :blk std.fmt.parseInt(usize, a["floor:".len..], 10) catch null;
|
||||
} else null;
|
||||
|
||||
const buffer = memory.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||
_ = logging.write("/system/services/acpi: out of memory\n");
|
||||
@@ -118,11 +126,11 @@ pub fn main(init: process.Init) void {
|
||||
_ = logging.write("/system/services/acpi: no acpi-tables node to claim\n");
|
||||
return;
|
||||
};
|
||||
// The node arrived with the spawn: the manager holds it and names it in the call
|
||||
// that creates this process. Discovery was the last thing in the system that
|
||||
// acquired hardware by naming it rather than being given it
|
||||
// (docs/os-development/device-authority.md).
|
||||
node_id = node.id;
|
||||
if (!device.claim(node_id)) {
|
||||
_ = logging.write("/system/services/acpi: unable to claim acpi-tables\n");
|
||||
return;
|
||||
}
|
||||
|
||||
// Map the node's resources: the AML blobs (bytecode), the FADT (intact
|
||||
// "FACP" header — decision 3), the io_port grant, and the SCI irq.
|
||||
@@ -519,8 +527,8 @@ fn registerDevice(node: *aml.Node, hid: [8]u8, interpreter: *aml.Interpreter) vo
|
||||
@memcpy(descriptor.hid[0..@intCast(hid_len)], hid[0..@intCast(hid_len)]);
|
||||
applyCrs(&descriptor, node, interpreter);
|
||||
|
||||
const id = device.register(node_id, &descriptor) orelse {
|
||||
std.log.info("register refused for {s}", .{hid[0..@intCast(hid_len)]});
|
||||
const id = device.register(node_id, &descriptor) catch |e| {
|
||||
std.log.warn("register refused for {s}: {s}", .{ hid[0..@intCast(hid_len)], @errorName(e) });
|
||||
return;
|
||||
};
|
||||
registered[registered_count] = .{ .hid = hid, .hid_len = @intCast(hid_len), .device_id = id, .resource_count = descriptor.resource_count };
|
||||
|
||||
@@ -99,10 +99,19 @@ fn identityFromReport(report: device_manager_protocol.ChildAdded) registry.Ident
|
||||
/// Whether some driver entry already serves registered device `device_id` —
|
||||
/// a re-report after a bus restart must not spawn a second instance.
|
||||
fn driverForDevice(device_id: u64) bool {
|
||||
return driverEntryForDevice(device_id) != null;
|
||||
}
|
||||
|
||||
/// The driver entry BOUND TO a device — matched and spawned for it. Distinct
|
||||
/// from the device's reporter: a mouse's provider is the bus that reported it
|
||||
/// (lineage via `children[].reporter`), while a volume's provider is the
|
||||
/// storage driver spawned FOR it — which is what a `.consumer` hello asks for.
|
||||
fn driverEntryForDevice(device_id: u64) ?*Driver {
|
||||
if (device_id == device_manager_protocol.no_device) return null;
|
||||
for (&drivers) |*driver| {
|
||||
if (driver.used and driver.device_id == device_id) return true;
|
||||
if (driver.used and driver.device_id == device_id) return driver;
|
||||
}
|
||||
return false;
|
||||
return null;
|
||||
}
|
||||
|
||||
// --- supervision -------------------------------------------------------------
|
||||
@@ -140,6 +149,12 @@ const Driver = struct {
|
||||
spawn_ns: u64 = 0,
|
||||
hello_deadline_ns: u64 = 0,
|
||||
restart_due_ns: u64 = 0,
|
||||
// The serving endpoint this instance handed up in its hello — what consumer
|
||||
// hellos for its reported children are answered with (establishment by
|
||||
// lineage, communication.md "Establishment: two planes"). A refcounted
|
||||
// handle in OUR table: onDriverExit must close it, or every restart leaks a
|
||||
// slot of the manager's 32 until no capability can arrive at all.
|
||||
endpoint: ?ipc.Handle = null,
|
||||
|
||||
fn name(driver: *const Driver) []const u8 {
|
||||
return driver.name_buffer[0..driver.name_len];
|
||||
@@ -197,16 +212,41 @@ fn addChild(parent: u64, bus_address: u64, identity: u64, device_id: u64, report
|
||||
/// protocol state (slots, rings) that died with the process — keeping the nodes
|
||||
/// would be keeping a lie. The restarted instance rediscovers and re-reports.
|
||||
/// Watchers hear the honest story: removed now, added again on rediscovery.
|
||||
///
|
||||
/// **A reporter's death reaps its subtree.** The class drivers spawned for the
|
||||
/// pruned children hold channels into the dead process; they cannot observe
|
||||
/// the death themselves (an HID driver blocks on interrupt reports that will
|
||||
/// simply never come — a silent zombie), and their still-`used` entries would
|
||||
/// make the matcher's dedupe refuse the respawn when the re-report arrives.
|
||||
/// So each is killed and its entry cleared: the re-report spawns a fresh
|
||||
/// instance, whose hello fetches the successor's channel. That is the restart
|
||||
/// story working — kill a bus driver and only its subtree blinks
|
||||
/// (docs/establishment-planes-plan.md P3).
|
||||
fn pruneChildrenOf(reporter: u32) void {
|
||||
for (&children) |*child| {
|
||||
if (child.used and child.reporter == reporter) {
|
||||
std.log.info("child removed (device {d} port {d})", .{ child.parent, child.bus_address });
|
||||
reapDriverBoundTo(child.device_id);
|
||||
child.used = false;
|
||||
Serve.publish(.child_removed, 0, .{ .parent = child.parent, .bus_address = child.bus_address });
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Kill the driver bound to a pruned child and clear its entry — the exit
|
||||
/// notification that follows finds no entry and is ignored, so the death is
|
||||
/// never double-counted, and the freed entry is what lets the re-report
|
||||
/// respawn. The device it was given returns to us by the kernel's loan rule.
|
||||
fn reapDriverBoundTo(device_id: u64) void {
|
||||
const driver = driverEntryForDevice(device_id) orelse return;
|
||||
if (driver.process_id != 0) {
|
||||
std.log.info("reaping {s} (its provider died)", .{driver.name()});
|
||||
_ = process.kill(driver.process_id);
|
||||
}
|
||||
if (driver.endpoint) |endpoint| _ = ipc.close(endpoint);
|
||||
driver.* = .{};
|
||||
}
|
||||
|
||||
/// How many children a driver instance has reported (the test-usb-restart
|
||||
/// trigger counts these).
|
||||
fn childCountOf(reporter: u32) u32 {
|
||||
@@ -217,9 +257,29 @@ fn childCountOf(reporter: u32) u32 {
|
||||
return n;
|
||||
}
|
||||
|
||||
/// The child a kernel device id belongs to — the lineage lookup: its
|
||||
/// `.reporter` names the driver instance that provides it, which is how a
|
||||
/// consumer's hello is routed to the right provider (communication.md
|
||||
/// "Establishment: two planes"). Null when nothing reported it, or its
|
||||
/// reporter died and pruned it — a retryable gap, not a verdict.
|
||||
fn childByDevice(device_id: u64) ?*Child {
|
||||
if (device_id == device_manager_protocol.no_device) return null;
|
||||
for (&children) |*child| {
|
||||
if (child.used and child.device_id == device_id) return child;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// The driver entry a live process id belongs to. Zero is not a process id here:
|
||||
/// it is what `onDriverExit` writes back to retire an id it has already acted on,
|
||||
/// so a second notification for the same death matches nothing.
|
||||
fn driverByName(name: []const u8) ?*Driver {
|
||||
for (&drivers) |*driver| {
|
||||
if (driver.used and std.mem.eql(u8, driver.name(), name)) return driver;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
fn driverByProcess(process_id: u32) ?*Driver {
|
||||
if (process_id == 0) return null;
|
||||
for (&drivers) |*driver| {
|
||||
@@ -257,6 +317,22 @@ fn addDriver(name: []const u8, device_id: u64, speaks_protocol: bool) void {
|
||||
/// device id as argv[1] when it has one, the hello deadline armed when it
|
||||
/// speaks the protocol.
|
||||
fn spawnDriver(driver: *Driver) void {
|
||||
// Take the device before the driver exists, so there is no window in which anyone
|
||||
// else could claim it — which is the whole of what makes the handover authoritative
|
||||
// rather than advisory. Re-claiming across a restart is expected to say
|
||||
// AlreadyClaimed once the manager already holds it, and that is fine: it means the
|
||||
// device never left our hands while the driver was dead.
|
||||
if (driver.device_id != device_manager_protocol.no_device) {
|
||||
device.claim(driver.device_id) catch |e| switch (e) {
|
||||
error.AlreadyClaimed => {}, // ours already, from a previous spawn of this driver
|
||||
else => {
|
||||
std.log.warn("cannot hold device {d} for {s}: {s}", .{ driver.device_id, driver.name(), @errorName(e) });
|
||||
driver.state = .failed;
|
||||
return;
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
var id_text: [20]u8 = undefined;
|
||||
var arguments: [1][]const u8 = undefined;
|
||||
var argument_count: usize = 0;
|
||||
@@ -264,7 +340,14 @@ fn spawnDriver(driver: *Driver) void {
|
||||
arguments[0] = std.fmt.bufPrint(&id_text, "{d}", .{driver.device_id}) catch return;
|
||||
argument_count = 1;
|
||||
}
|
||||
const child = process.spawnSupervised(driver.name(), arguments[0..argument_count], manager_endpoint) orelse {
|
||||
// The device rides the spawn, so the driver holds it before its first instruction.
|
||||
// A transfer *after* spawning would leave a window in which the child is running
|
||||
// without its hardware — closed on one machine, open on another
|
||||
// (docs/bounds-track-plan.md, "the grant rides system_spawn").
|
||||
const give = driver.device_id;
|
||||
if (give != device_manager_protocol.no_device)
|
||||
std.log.info("delegated device {d} to {s}", .{ give, driver.name() });
|
||||
const child = process.spawnSupervisedWithDevice(driver.name(), arguments[0..argument_count], manager_endpoint, give) orelse {
|
||||
std.log.info("failed to spawn {s}", .{driver.name()});
|
||||
driver.state = .failed;
|
||||
return;
|
||||
@@ -298,6 +381,13 @@ fn onDriverExit(driver: *Driver) void {
|
||||
// Without this the backoff would count one death twice and the crash-loop cap
|
||||
// would fire at half the deaths it names.
|
||||
driver.process_id = 0;
|
||||
// The dead instance's serving endpoint is stale the moment it died — the
|
||||
// kernel marked the endpoint dead, but our refcounted handle would sit in
|
||||
// the 32-slot table forever. The successor's hello stores a fresh one.
|
||||
if (driver.endpoint) |stale| {
|
||||
_ = ipc.close(stale);
|
||||
driver.endpoint = null;
|
||||
}
|
||||
pruneChildrenOf(dead);
|
||||
const reason = process.exitReason(dead) orelse .fault;
|
||||
if (reason == .exited) {
|
||||
@@ -362,6 +452,40 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
const total = device.enumerate(buffer);
|
||||
const count = @min(total, buffer.len);
|
||||
|
||||
// The node discovery needs: the kernel seeds it, so it is in this same snapshot
|
||||
// and can be handed over like any other assignment. Discovery used to find and
|
||||
// claim it itself — the last driver that acquired hardware by naming it rather
|
||||
// than being given it (docs/os-development/device-authority.md).
|
||||
var tables_node: u64 = device_manager_protocol.no_device;
|
||||
for (buffer[0..count]) |descriptor| {
|
||||
if (descriptor.class == @intFromEnum(device.DeviceClass.acpi_tables)) {
|
||||
tables_node = descriptor.id;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
// **Hold the firmware-discovered hardware, so none of it is left lying around.**
|
||||
// A device nobody holds can be claimed by anyone, so every seeded device that
|
||||
// carries mappable resources is taken here whether or not a driver wants it — the
|
||||
// HPET most of all, which has an MMIO window and an IRQ and no user-space driver.
|
||||
// Held by the manager it is inert; unheld it was there for the taking.
|
||||
//
|
||||
// Two deliberate exclusions:
|
||||
// - the loader's framebuffer, which the compositor claims and which is not
|
||||
// hardware anyone is delegated (the manager starts before display, so taking
|
||||
// it here would break the boot screen);
|
||||
// - anything with no resources, which grants nothing and so is not worth holding.
|
||||
//
|
||||
// This covers the boot snapshot only. A device *reported* later and matched to no
|
||||
// driver stays claimable — the pci-cap and iommu-fault fixtures rely on exactly
|
||||
// that to reach an unmatched NIC. Narrowing it further is a separate change with
|
||||
// those fixtures in scope.
|
||||
for (buffer[0..count]) |descriptor| {
|
||||
if (descriptor.resource_count == 0) continue;
|
||||
if (descriptor.class == @intFromEnum(device.DeviceClass.display)) continue;
|
||||
device.claim(descriptor.id) catch continue; // already held, or not ours to take
|
||||
}
|
||||
|
||||
var matched: usize = 0;
|
||||
for (buffer[0..count]) |descriptor| {
|
||||
if (descriptor.class == @intFromEnum(device.DeviceClass.pci_host_bridge)) {
|
||||
@@ -383,7 +507,7 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
// under the neutral name "discovery", spawned once at startup. It finds and
|
||||
// claims the acpi-tables (or devicetree-blob) node itself. Not a per-device
|
||||
// match — it is the discoverer, not a driver bound to one device.
|
||||
addDriver("discovery", device_manager_protocol.no_device, false);
|
||||
addDriver("discovery", tables_node, false);
|
||||
|
||||
if (test_restart_mode) {
|
||||
// The driver-restart scenario's fixture: claims device 0 (the tree
|
||||
@@ -419,11 +543,54 @@ fn onHello(_: void, invocation: Invocation(device_manager_protocol.Hello), _: An
|
||||
std.log.info("refused hello (version {d}) from process {d}", .{ invocation.request.version, invocation.sender });
|
||||
return -envelope.EPROTO;
|
||||
}
|
||||
// A consumer is not a spawned driver: no entry, no deadline, no state —
|
||||
// just establishment. It asks for the channel of the driver BOUND TO its
|
||||
// target (fat asking for its volume's block provider). No channel is a
|
||||
// retryable ack, exactly as for a device-role consumer.
|
||||
//
|
||||
// The residual, stated plainly: any process granted `open device-manager`
|
||||
// can ask. The grant rows are the gate today, as they were when the block
|
||||
// name was open-granted; a finer per-channel policy belongs to the same
|
||||
// future as the spawn capability (device-authority.md).
|
||||
if (invocation.request.role == @intFromEnum(device_manager_protocol.Role.consumer)) {
|
||||
if (invocation.request.wants_channel != 0) {
|
||||
if (driverEntryForDevice(invocation.target)) |provider| {
|
||||
if (provider.endpoint) |serving| service.replyWithCapability(serving);
|
||||
}
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
const driver = driverByProcess(invocation.sender) orelse {
|
||||
std.log.info("hello from unknown process {d}", .{invocation.sender});
|
||||
return -envelope.EPERM;
|
||||
};
|
||||
driver.state = .running;
|
||||
|
||||
// A provider's hello carries its serving endpoint — the channel consumer
|
||||
// hellos for its reported children are answered with. Stored on the entry
|
||||
// (claimed from the turn, so the harness's defer leaves it alone); a
|
||||
// re-hello replaces, closing the old handle rather than leaking the slot.
|
||||
if (invocation.capability) |serving| {
|
||||
if (driver.endpoint) |previous| _ = ipc.close(previous);
|
||||
driver.endpoint = serving;
|
||||
Serve.claimArrival();
|
||||
}
|
||||
|
||||
// A consumer's hello asks for its device's provider: the child's reporter
|
||||
// is the routing fact (establishment by lineage). No channel is not a
|
||||
// refusal — the provider may be mid-restart and its re-report on the way —
|
||||
// so the hello still acks and the consumer retries. Nothing is nominated
|
||||
// unless asked: a capability sent to a caller that never reads one is a
|
||||
// leaked slot in ITS table.
|
||||
if (invocation.request.wants_channel != 0) {
|
||||
if (childByDevice(invocation.target)) |child| {
|
||||
if (driverByProcess(child.reporter)) |provider| {
|
||||
if (provider.endpoint) |serving| service.replyWithCapability(serving);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
std.log.info("hello from {s} (device {d})", .{ driver.name(), invocation.target });
|
||||
// Resilience drill (V6): once, kill the virtio-gpu driver a moment after it hellos, so
|
||||
// the normal restart policy respawns it — the compositor must survive and re-attach.
|
||||
@@ -461,9 +628,36 @@ fn onChildAdded(_: void, invocation: Invocation(device_manager_protocol.ChildAdd
|
||||
if (match.ambiguous)
|
||||
std.log.info("/system/configuration/devices.csv: multiple equally-specific rules match the device {s} reported; binding {s}", .{ driver.name(), match.driver });
|
||||
if (id.bus == .acpi) {
|
||||
// An hid-matched driver (ps2-bus) is a singleton that finds its
|
||||
// own devices once spawned — spawn it once, no device assignment.
|
||||
if (!alreadySupervised(match.driver)) addDriver(match.driver, device_manager_protocol.no_device, false);
|
||||
// An hid-matched driver is a singleton over one piece of hardware
|
||||
// described by several nodes: the 8042 is a single controller whose
|
||||
// I/O ports live under the keyboard node (PNP0303) while the mouse
|
||||
// is a second node (PNP0F13). Two processes would fight over the
|
||||
// same 0x60/0x64 registers, so there is exactly one instance — and
|
||||
// it needs *every* matching device, however many the machine has
|
||||
// (some have none, some one port, some two).
|
||||
//
|
||||
// The first rides the spawn; the rest are transferred to the
|
||||
// running instance. Late arrival is fine here because the order is
|
||||
// natural: the instance needs the controller node immediately and
|
||||
// reaches the mouse only after the 8042 handshakes and identify.
|
||||
if (!alreadySupervised(match.driver)) {
|
||||
addDriver(match.driver, device_id, false);
|
||||
} else if (driverByName(match.driver)) |running| {
|
||||
if (running.process_id != 0) {
|
||||
device.claim(device_id) catch |e| switch (e) {
|
||||
error.AlreadyClaimed => {},
|
||||
else => {
|
||||
std.log.warn("cannot hold device {d} for {s}: {s}", .{ device_id, match.driver, @errorName(e) });
|
||||
return status;
|
||||
},
|
||||
};
|
||||
device.transfer(device_id, running.process_id) catch |e| {
|
||||
std.log.warn("could not give device {d} to {s}: {s}", .{ device_id, match.driver, @errorName(e) });
|
||||
return status;
|
||||
};
|
||||
std.log.info("delegated device {d} to {s} (already running)", .{ device_id, match.driver });
|
||||
}
|
||||
}
|
||||
} else {
|
||||
// A per-device driver: one instance, the registered id as argv[1].
|
||||
if (!driverForDevice(device_id)) addDriver(match.driver, device_id, true);
|
||||
@@ -522,14 +716,24 @@ fn onChildRemoved(_: void, invocation: Invocation(device_manager_protocol.ChildR
|
||||
/// The reserved `enumerate` verb: the mirror, one `ChildEntry` per known child,
|
||||
/// packed into the reply's tail. How many arrived is the reply's own length —
|
||||
/// `Status.len` — so no count header is spent saying it twice.
|
||||
fn onEnumerate(_: void, _: Invocation(void), answer: Answer(void)) isize {
|
||||
fn onEnumerate(_: void, invocation: Invocation(void), answer: Answer(void)) isize {
|
||||
const entry_size = @sizeOf(device_manager_protocol.ChildEntry);
|
||||
const tail = answer.tail();
|
||||
// `target` is the page cursor: skip that many known children first. One
|
||||
// reply holds only a handful of entries, and a real tree (a dozen ACPI
|
||||
// nodes before the first USB child) outgrew one packet — the storage
|
||||
// child silently never fit, which is precisely the truncation shape the
|
||||
// bounds audit exists to forbid. A caller pages until a short page.
|
||||
var skip = invocation.target;
|
||||
var written: usize = 0;
|
||||
for (&children) |*child| {
|
||||
if (!child.used) continue;
|
||||
if (skip > 0) {
|
||||
skip -= 1;
|
||||
continue;
|
||||
}
|
||||
if (written + entry_size > tail.len) break;
|
||||
const entry = device_manager_protocol.ChildEntry{ .parent = child.parent, .bus_address = child.bus_address, .identity = child.identity };
|
||||
const entry = device_manager_protocol.ChildEntry{ .parent = child.parent, .bus_address = child.bus_address, .identity = child.identity, .device_id = child.device_id };
|
||||
@memcpy(tail[written..][0..entry_size], std.mem.asBytes(&entry));
|
||||
written += entry_size;
|
||||
}
|
||||
|
||||
@@ -74,10 +74,12 @@ pub const Gop = struct {
|
||||
return null;
|
||||
};
|
||||
|
||||
if (!device.claim(found.id)) {
|
||||
_ = logging.write("display: could not claim the framebuffer\n");
|
||||
device.claim(found.id) catch |e| {
|
||||
var line: [96]u8 = undefined;
|
||||
_ = logging.write(std.fmt.bufPrint(&line, "display: could not claim the framebuffer: {s}\n", .{@errorName(e)}) catch
|
||||
"display: could not claim the framebuffer\n");
|
||||
return null;
|
||||
}
|
||||
};
|
||||
// Resource 0 is the framebuffer memory window; the kernel maps it write-combining
|
||||
// because the resource carries that flag (docs/display-plan.md D1).
|
||||
const front_base = device.mmioMap(found.id, 0) orelse {
|
||||
|
||||
@@ -10,8 +10,10 @@ pub fn build(b: *std.Build) void {
|
||||
.name = "fat",
|
||||
.root_source_file = b.path("fat.zig"),
|
||||
.imports = &.{
|
||||
"block", "envelope", "file-system", "ipc", "logging", "memory", "process",
|
||||
"service", "time", "vfs-protocol",
|
||||
"block", "channel", "device-manager-protocol", "driver",
|
||||
"envelope", "file-system", "ipc", "logging",
|
||||
"memory", "process", "service", "time",
|
||||
"vfs-protocol",
|
||||
},
|
||||
});
|
||||
b.installArtifact(exe);
|
||||
|
||||
@@ -10,6 +10,9 @@
|
||||
//! it.
|
||||
|
||||
const std = @import("std");
|
||||
const channel = @import("channel");
|
||||
const device_manager_protocol = @import("device-manager-protocol");
|
||||
const driver = @import("driver");
|
||||
const ipc = @import("ipc");
|
||||
const process = @import("process");
|
||||
const service = @import("service");
|
||||
@@ -116,6 +119,56 @@ const mount_retry_ms = 500;
|
||||
|
||||
var mounted = false;
|
||||
var service_endpoint: ipc.Handle = 0;
|
||||
/// The one channel to the device manager, opened on first need and kept — the
|
||||
/// poll retries on it, never spending a handle-table slot per attempt.
|
||||
var manager_handle: ?ipc.Handle = null;
|
||||
|
||||
/// Find the volume's provider through the device manager (establishment by
|
||||
/// lineage, communication.md "Establishment: two planes" — `block` is not a
|
||||
/// registry name; one storage process serves each stick): enumerate the
|
||||
/// manager's tree, take the FIRST usb mass-storage child by enumeration order
|
||||
/// (deterministic within a boot; single-volume by construction, and choosing
|
||||
/// the BOOT volume by content when two sticks are present is the M21 remount
|
||||
/// track), and consumer-hello for the channel of the driver bound to it.
|
||||
/// Null until the chain is up — the caller's poll retries.
|
||||
fn acquireVolume() ?block.Device {
|
||||
const manager = manager_handle orelse opened: {
|
||||
const handle = channel.openEndpoint("device-manager") orelse return null;
|
||||
manager_handle = handle;
|
||||
break :opened handle;
|
||||
};
|
||||
|
||||
// The envelope's reserved `enumerate` verb, PAGED: one reply carries only
|
||||
// a handful of entries and a real tree (a dozen ACPI nodes before the
|
||||
// first USB child) is bigger, so `Header.target` is the start cursor and
|
||||
// a short page is the end. Identity is the bus's native triple, for USB
|
||||
// (base << 16) | (class << 8) | protocol — mass storage is base 0x08,
|
||||
// subclass 0x06 (SCSI transparent), the same key devices.csv matches on.
|
||||
const Entry = device_manager_protocol.ChildEntry;
|
||||
var start: u64 = 0;
|
||||
while (true) {
|
||||
const enumerate = envelope.Header{ .operation = envelope.operation_enumerate, .target = start };
|
||||
var reply: [device_manager_protocol.message_maximum]u8 = undefined;
|
||||
const length = ipc.call(manager, std.mem.asBytes(&enumerate), &reply) catch return null;
|
||||
const status = envelope.statusOf(reply[0..length]) orelse return null;
|
||||
if (status.status != 0) return null;
|
||||
|
||||
const carried = @min(@as(usize, status.len), length -| envelope.prefix_size);
|
||||
const tail = reply[envelope.prefix_size..][0..carried];
|
||||
const count = tail.len / @sizeOf(Entry);
|
||||
if (count == 0) return null; // the tree is exhausted; no volume yet
|
||||
var index: usize = 0;
|
||||
while (index < count) : (index += 1) {
|
||||
const entry = std.mem.bytesToValue(Entry, tail[index * @sizeOf(Entry) ..][0..@sizeOf(Entry)]);
|
||||
if (entry.device_id == device_manager_protocol.no_device) continue;
|
||||
if ((entry.identity >> 16) & 0xff != 0x08 or (entry.identity >> 8) & 0xff != 0x06) continue;
|
||||
const exchanged = driver.helloOn(manager, .consumer, entry.device_id, null, true) orelse return null;
|
||||
const provider = exchanged.channel orelse continue; // its driver not up yet — next tick
|
||||
return .{ .endpoint = provider };
|
||||
}
|
||||
start += count;
|
||||
}
|
||||
}
|
||||
|
||||
fn initialise(endpoint: ipc.Handle) bool {
|
||||
service_endpoint = endpoint;
|
||||
@@ -133,7 +186,7 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
/// `mounted` on success; a failure leaves everything untouched for the next tick.
|
||||
fn tryBringUp() void {
|
||||
if (mounted) return;
|
||||
const device = block.tryOpen() orelse return;
|
||||
const device = acquireVolume() orelse return;
|
||||
const geometry = device.geometry() orelse {
|
||||
_ = logging.write("/system/services/fat: block geometry unavailable\n");
|
||||
return;
|
||||
|
||||
+118
-8
@@ -606,9 +606,17 @@ CASES = [
|
||||
"smp": 4,
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# M18.2: bus tree reports — the xHCI driver scans its root-hub ports and
|
||||
# reports both QEMU devices; the manager mirrors, prunes on the reporter's
|
||||
# death, and the respawned driver re-reports (docs/device-manager.md).
|
||||
# M18.2 + establishment P3: bus tree reports, and the restart REBUILDING
|
||||
# the subtree. The manager prunes on the reporter's death — reaping the
|
||||
# class drivers bound to the pruned children, which hold channels into the
|
||||
# dead process and cannot observe the death themselves — and when the
|
||||
# respawned instance re-reports, the matcher spawns fresh class drivers
|
||||
# whose hellos fetch the successor's channel. The tail asserts the subtree
|
||||
# WORKS again: the respawned usb-storage opens its device on the NEW bus
|
||||
# instance and reads block 0. (The HID "ok" lines never appear in this
|
||||
# scenario — it boots no input service — so storage is the functional
|
||||
# proof.) Before the reap existed, the stale entries blocked the respawn
|
||||
# and the survivors were silent zombies, so this tail could not match.
|
||||
{"name": "usb-report",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
@@ -617,19 +625,36 @@ CASES = [
|
||||
"expect": r"device-manager: child added[\s\S]*"
|
||||
r"device-manager: child added[\s\S]*"
|
||||
r"device-manager: test mode: killing the reporter[\s\S]*"
|
||||
r"device-manager: reaping \S*usb-storage[\s\S]*"
|
||||
r"device-manager: child removed[\s\S]*"
|
||||
r"device-manager: restarting \S*usb-xhci-bus[\s\S]*"
|
||||
r"device-manager: child added",
|
||||
r"device-manager: child added[\s\S]*"
|
||||
r"device-manager: delegated device \d+ to /system/drivers/usb-storage[\s\S]*"
|
||||
r"usb-storage: ready[\s\S]*"
|
||||
r"usb-storage: block 0 signature 0x55aa",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# USB HID end to end: boot the full tree, enumerate the xHCI, and let the
|
||||
# manager spawn the USB keyboard driver, which opens its device over the
|
||||
# transfer protocol, asks for boot protocol, subscribes to its interrupt
|
||||
# endpoint, and comes up — proof the class-driver <-> controller path works.
|
||||
#
|
||||
# It also pins the slot count to the hardware's answer. The backreference is the
|
||||
# assertion: the driver must track exactly as many device slots as the controller
|
||||
# reports in HCSPARAMS1.MaxSlots. It tracked a fixed 8 while QEMU's xHCI reports
|
||||
# 64, so seven eighths of the controller was invisible and a device behind a hub
|
||||
# past the eighth vanished without a log line. \1 fails the moment they diverge.
|
||||
{"name": "usb-hid",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
# usb-kbd/usb-mouse ride the default boot xHCI bus (see qemu_args).
|
||||
"expect": r"(?=[\s\S]*usb-hid-keyboard: ok)(?=[\s\S]*usb-hid-mouse: ok)",
|
||||
# The controller must arrive by DELEGATION, not by claiming: the manager holds it
|
||||
# and transfers it in the hello reply, so the match is authoritative rather than
|
||||
# advisory (docs/os-development/device-authority.md). The delegation line must
|
||||
# precede the hello, because the transfer completes before the reply lands.
|
||||
"expect": r"(?=[\s\S]*device-manager: delegated device (\d+) to /system/drivers/usb-xhci-bus"
|
||||
r"[\s\S]*usb-xhci-bus: controller device \1 registers at)"
|
||||
r"(?=[\s\S]*usb-xhci-bus: controller running \((\d+) slots, tracking \2,)"
|
||||
r"(?=[\s\S]*usb-hid-keyboard: ok)(?=[\s\S]*usb-hid-mouse: ok)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# Keyboard echo: inject a known phrase via QMP send-key; the usb-hid-keyboard
|
||||
# driver decodes it and echoes each character to the log (the simple
|
||||
@@ -667,6 +692,50 @@ CASES = [
|
||||
r"(?=.*hub slot \d+ port \d+ device:.*0x0627)"
|
||||
r"(?=.*usb-hid-keyboard: ok \(device 3)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# Establishment P4 — the Ryzen mouse bug, pinned (docs/establishment-planes-plan.md).
|
||||
# A real machine carried THREE xHCI controllers; the transfer contract was one
|
||||
# exclusive registry name, so every class driver reached whichever bus instance
|
||||
# bound first, and a device behind any other controller was unreachable
|
||||
# ("could not open device 50"). This case is the minimal reproduction: a second
|
||||
# controller with its own keyboard, while the boot controller keeps the default
|
||||
# keyboard — BOTH must come up, on different device ids, which requires each
|
||||
# driver to reach ITS OWN controller. Under name-based establishment exactly one
|
||||
# can; under lineage routing (the manager answers each hello with the reporter's
|
||||
# channel) both do.
|
||||
{"name": "usb-two-controllers",
|
||||
"build_case": "usb-hid",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"qemu_extra": ["-device", "qemu-xhci,id=xhci2",
|
||||
"-device", "usb-kbd,bus=xhci2.0,port=1"],
|
||||
# Two bus delegations, then two keyboards alive with distinct ids: the
|
||||
# second `ok` must name a different device than the first, so one keyboard
|
||||
# answering twice can never satisfy it.
|
||||
"expect": r"(?s)(?=(?:[\s\S]*delegated device \d+ to /system/drivers/usb-xhci-bus){2})"
|
||||
r"(?=[\s\S]*usb-hid-keyboard: ok \(device (\d+)\b[\s\S]*usb-hid-keyboard: ok \(device (?!\1\b)\d+)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# The driver must never truncate a configuration block. Its size is the DEVICE's
|
||||
# choice (wTotalLength, a u16); the driver used to read the first 512 bytes into a
|
||||
# fixed buffer and parse those, so interfaces past the cut did not exist while
|
||||
# SET_CONFIGURATION still configured the device for all of them.
|
||||
#
|
||||
# The backreference is the assertion: declared length and bytes read must match.
|
||||
#
|
||||
# Honest limit: QEMU cannot produce a block over 512 bytes. Its boot keyboard,
|
||||
# mouse and stick are 34-44, and the largest device on offer is usb-audio in
|
||||
# multi-channel mode at 211 — which is why the suite never saw the original bug,
|
||||
# and why it cannot now reproduce that exact trigger. What this case does catch is
|
||||
# the class: any clamp below the attached device's block size fails it, verified
|
||||
# by pinning the buffer to 128 and watching it go red. A real headset (500-900
|
||||
# bytes), UVC webcam (1-3 KB) or multifunction printer trips the 512 itself.
|
||||
{"name": "usb-large-descriptor",
|
||||
"build_case": "usb-hid",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"qemu_extra": ["-device", "qemu-xhci,id=xhci2",
|
||||
"-device", "usb-audio,bus=xhci2.0,port=1,multi=on"],
|
||||
"expect": r"(?s)(?=.*usb-xhci-bus: config block (\d\d\d+) bytes, read \1)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# USB mass storage end to end: the boot usb-storage device (the FAT32 image,
|
||||
# which has a real 0x55AA boot sector) is enough — the manager spawns
|
||||
# usb-storage, which opens the device, runs the Bulk-Only / SCSI bring-up,
|
||||
@@ -745,8 +814,16 @@ CASES = [
|
||||
{"name": "acpi-ps2",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
# The 8042 is ONE controller described by two ACPI nodes, so the single ps2-bus
|
||||
# instance must receive BOTH. The keyboard node rides the spawn — it carries the
|
||||
# 0x60/0x64 ports and is needed immediately — and the mouse node is transferred to
|
||||
# the already-running instance, which is safe because it is not touched until after
|
||||
# the controller handshakes and identify. Two processes cannot split this: they
|
||||
# would fight over the same registers.
|
||||
"expect": r"discovery: device \d+\s+bus=acpi hid=PNP0303[\s\S]*"
|
||||
r"device-manager: spawned \S*ps2-bus[\s\S]*"
|
||||
r"device-manager: delegated device (\d+) to \S*ps2-bus[\s\S]*"
|
||||
r"device-manager: spawned \S*ps2-bus for device \1[\s\S]*"
|
||||
r"device-manager: delegated device \d+ to \S*ps2-bus \(already running\)[\s\S]*"
|
||||
r"ps2-bus: keyboard driver attached",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# M21.1: the SCI + power button. Boot the manager (which spawns the acpi
|
||||
@@ -819,10 +896,15 @@ CASES = [
|
||||
{"name": "pci-scan",
|
||||
"smp": 4,
|
||||
"timeout": 60,
|
||||
"expect": r"pci-bus: (\d+) functions found[\s\S]*"
|
||||
# The bridge must arrive by DELEGATION, not by claiming — and again after the
|
||||
# restart, which is what proves the manager re-takes the device when its driver
|
||||
# dies and hands it to the replacement. \1 pins it to the same device both times.
|
||||
"expect": r"device-manager: delegated device (\d+) to \S*pci-bus[\s\S]*"
|
||||
r"pci-bus: (\d+) functions found[\s\S]*"
|
||||
r"device-manager: test mode: killing the reporter[\s\S]*"
|
||||
r"device-manager: restarting \S*pci-bus[\s\S]*"
|
||||
r"pci-bus: \1 functions found[\s\S]*"
|
||||
r"device-manager: delegated device \1 to \S*pci-bus[\s\S]*"
|
||||
r"pci-bus: \2 functions found[\s\S]*"
|
||||
r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# M18.3: the application surface — device-list enumerates the tree over IPC,
|
||||
@@ -978,9 +1060,37 @@ CASES = [
|
||||
# device_register containment (in-kernel): registering a child whose MMIO window
|
||||
# escapes its parent's grant is refused (NotContained) — else dev_register would map
|
||||
# arbitrary physical memory — while an identical re-register stays idempotent.
|
||||
# Also pins the cap ordering: a parent already at maximum_children_per_parent must
|
||||
# still re-admit an identical child (a restarted bus consumes no slot re-reporting
|
||||
# what it rediscovers) while still refusing a genuinely new one.
|
||||
{"name": "containment",
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# PCI host-bridge apertures come from the holes in the firmware memory map, and a
|
||||
# registered BAR must fall inside one. The invariant is that an aperture never
|
||||
# covers memory the firmware described - otherwise device_register containment
|
||||
# admits a child BAR over live RAM. Driven with a synthetic 100-entry map, since
|
||||
# OVMF only ever produces 15-25 and real firmware 60-200.
|
||||
{"name": "apertures",
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# Delegation's mechanism (docs/os-development/device-authority.md). A claim is
|
||||
# exclusive, so handing a device on is a MOVE: the giver stops holding it the
|
||||
# instant the receiver starts - which is why this is not the M13 capability path,
|
||||
# where a handle is shared refcounted. The kernel's whole rule is "you may give
|
||||
# away what you hold"; it has no notion of which task is the device manager.
|
||||
{"name": "device-transfer",
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# The attacker the device suite never had. The audit's finding was that a fully
|
||||
# green suite missed six real defects because it contains no attacker: every device
|
||||
# case asserts a driver handed its hardware can drive it, and none asks what a
|
||||
# process handed NOTHING can do. This fixture is that process - it holds no device
|
||||
# and asserts it can give none away, with a positive control first so the refusals
|
||||
# are decisions rather than a broken syscall path.
|
||||
{"name": "device-authority",
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# IRQ teardown: an exiting driver's line is masked and its slot cleared (so no
|
||||
# ISR notifies a freed endpoint), and a sibling owner sharing that endpoint
|
||||
# keeps its own binding. A long-running driver never reaches this teardown path.
|
||||
|
||||
@@ -19,13 +19,15 @@ pub fn main(init: process.Init) void {
|
||||
const argument = init.arguments.get(1) orelse return; // bare: stay silent
|
||||
const assigned = std.fmt.parseInt(u64, argument, 10) catch return;
|
||||
|
||||
// The respawn only reaches this line because the kernel released the
|
||||
// previous instance's claim at death. A failed claim exits cleanly — the
|
||||
// manager reads "meant to stop" and the scenario fails loudly by silence.
|
||||
if (!device.claim(assigned)) {
|
||||
_ = logging.write("crash-test: claim failed\n");
|
||||
return;
|
||||
}
|
||||
// The device arrived with the spawn — this fixture is delegated its hardware like
|
||||
// any other driver, so it holds `assigned` before its first instruction and has
|
||||
// nothing to claim (docs/os-development/device-authority.md).
|
||||
//
|
||||
// The property this scenario checks is unchanged, only its mechanism: a respawned
|
||||
// instance still gets the device its predecessor held. It used to arrive because
|
||||
// the kernel released the dead instance's claim and this one re-took it, racing
|
||||
// anyone else who wanted it; now the device reverts to the manager on death and is
|
||||
// handed to the replacement, which is the same guarantee without the race.
|
||||
|
||||
var manager: ?ipc.Handle = null;
|
||||
var tries: u32 = 0;
|
||||
|
||||
@@ -0,0 +1,15 @@
|
||||
//! The device-authority-test fixture as a binary package (docs/build-packages-plan.md):
|
||||
//! this file names the binary and EXACTLY the modules its source imports —
|
||||
//! build-support resolves each name from the domains this zon declares.
|
||||
|
||||
const std = @import("std");
|
||||
const build_support = @import("build-support");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
const exe = build_support.userBinary(b, .{
|
||||
.name = "device-authority-test",
|
||||
.root_source_file = b.path("device-authority-test.zig"),
|
||||
.imports = &.{ "driver", "logging", "process", "time" },
|
||||
});
|
||||
b.installArtifact(exe);
|
||||
}
|
||||
@@ -0,0 +1,15 @@
|
||||
.{
|
||||
.name = .device_authority_test,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0x4acbba0c105a1462, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{
|
||||
// build-support supplies the shared recipe; kernel is implicit in
|
||||
// every binary (the root shim + link script live there). The rest
|
||||
// are exactly the homes of this binary's declared imports.
|
||||
.@"build-support" = .{ .path = "../../../../build-support" },
|
||||
.kernel = .{ .path = "../../../../library/kernel" },
|
||||
.device = .{ .path = "../../../../library/device" },
|
||||
},
|
||||
.paths = .{""},
|
||||
}
|
||||
@@ -0,0 +1,124 @@
|
||||
//! device-authority-test — the attacker the device suite never had.
|
||||
//!
|
||||
//! The audit behind [docs/fixed-bounds-audit.md] found six real defects that a
|
||||
//! fully green suite had missed, and the reason was structural: *the suite
|
||||
//! contains no attacker*. Every device case asserts that a driver handed its
|
||||
//! own hardware can drive it. None asks what a process that was handed
|
||||
//! **nothing** can do.
|
||||
//!
|
||||
//! This binary is that process. It is spawned with no device, holds no device,
|
||||
//! and asserts what it therefore cannot do
|
||||
//! ([docs/os-development/device-authority.md]):
|
||||
//!
|
||||
//! 1. **A positive control first.** `device_enumerate` works from here, so
|
||||
//! the refusals below are decisions rather than a syscall path that is
|
||||
//! simply broken for this process. Without this, "everything failed" would
|
||||
//! read identically to "the assertions are meaningless".
|
||||
//! 2. **It cannot give away a device it does not hold** — not one another
|
||||
//! task holds, and not a free one either. The kernel's whole rule is *you
|
||||
//! may give away what you hold*, so the state of the device is irrelevant:
|
||||
//! a process holding nothing can transfer nothing. That is asserted across
|
||||
//! several ids precisely so it cannot pass by accident of which device
|
||||
//! happened to be free at boot.
|
||||
//! 3. **A device that does not exist is refused differently** — `NoSuchDevice`
|
||||
//! rather than `NotHeld`. A refusal that cannot say which rule refused it
|
||||
//! is what cost a debugging session on the Ryzen, so the distinction is
|
||||
//! part of the contract and is tested as such.
|
||||
//!
|
||||
//! **Why there is no "cannot take a delegated device" assertion here.** The
|
||||
//! hole this fixture was written for is closed, but not by a refusal it could
|
||||
//! observe. A device that was given to someone is *held*, so an attempt to
|
||||
//! take it is refused as `AlreadyClaimed` — the same answer as before. What
|
||||
//! changed is what happens when the holder dies: the device returns to
|
||||
//! whoever lent it instead of becoming free, so the window in which a
|
||||
//! stranger could take it no longer exists. There is no moment to catch.
|
||||
|
||||
const std = @import("std");
|
||||
const device = @import("driver");
|
||||
const logging = @import("logging");
|
||||
const process = @import("process");
|
||||
const time = @import("time");
|
||||
|
||||
fn line(comptime format: []const u8, arguments: anytype) void {
|
||||
var buffer: [160]u8 = undefined;
|
||||
_ = logging.write(std.fmt.bufPrint(&buffer, format, arguments) catch return);
|
||||
}
|
||||
|
||||
var failures: usize = 0;
|
||||
var process_table: [64]process.ProcessDescriptor = undefined;
|
||||
|
||||
fn check(name: []const u8, ok: bool) void {
|
||||
if (!ok) failures += 1;
|
||||
line("device-authority: {s} {s}\n", .{ if (ok) "ok" else "FAIL", name });
|
||||
}
|
||||
|
||||
fn run() void {
|
||||
// 1. The positive control: this process can reach the device syscalls at all.
|
||||
var table: [64]device.DeviceDescriptor = undefined;
|
||||
const total = device.enumerate(&table);
|
||||
check("enumerate works from an unprivileged process", total > 0);
|
||||
const seen = @min(total, table.len);
|
||||
|
||||
// 2. Holding nothing, it can give nothing away — whatever the device's state.
|
||||
// Every id the machine actually has, so this cannot pass by luck.
|
||||
var refused: usize = 0;
|
||||
var wrong_reason: usize = 0;
|
||||
for (table[0..seen]) |descriptor| {
|
||||
device.transfer(descriptor.id, process.taskId()) catch |e| {
|
||||
refused += 1;
|
||||
if (e != error.NotHeld) wrong_reason += 1;
|
||||
continue;
|
||||
};
|
||||
}
|
||||
check("every transfer by a non-holder is refused", refused == seen);
|
||||
check("each refusal says NotHeld, not something vaguer", wrong_reason == 0);
|
||||
|
||||
// 3. A device that does not exist is a different refusal, and says so.
|
||||
const absent = if (device.transfer(0xFFFF_FFFF, process.taskId())) |_| false else |e| e == error.NoSuchDevice;
|
||||
check("a device that does not exist is refused as absent", absent);
|
||||
|
||||
// 4. **The spawn is not a second way in.** A device now rides system_spawn, which
|
||||
// would be a fine back door if the kernel checked ownership any less carefully
|
||||
// there than it does in transfer: spawn a child, name someone else's device, and
|
||||
// the child holds hardware nobody gave it. The refusal must happen before the
|
||||
// child exists, so nothing is left running either.
|
||||
if (seen != 0) {
|
||||
const before = process.processes(&process_table);
|
||||
const spawned = process.spawnSupervisedWithDevice("/test/system/services/device-authority-test", &.{}, null, table[0].id);
|
||||
check("spawning with a device the caller does not hold is refused", spawned == null);
|
||||
check("and no child was left behind by the refusal", process.processes(&process_table) == before);
|
||||
}
|
||||
|
||||
// 5. **Nothing with mappable resources is left lying around.** A device nobody
|
||||
// holds can be claimed by anyone, so the manager takes every seeded device that
|
||||
// carries resources — the HPET above all, which has an MMIO window and an IRQ
|
||||
// and no user-space driver. The one exception is the loader's framebuffer,
|
||||
// which the compositor claims. So from here, a resource-bearing device should
|
||||
// refuse to be taken, and the reason should be that someone already has it.
|
||||
// Settle first, then sweep **once**. The manager is still starting when this
|
||||
// fixture is spawned, so an immediate sweep finds hardware unheld and reports a
|
||||
// hole that closes a millisecond later. Retrying until the sweep comes back
|
||||
// empty is worse than useless: the first pass *takes* the device, so the second
|
||||
// finds it unavailable — because this process now holds it — and concludes all
|
||||
// is well. One sweep, after a wait long enough for the manager to have claimed.
|
||||
time.sleepMillis(1500);
|
||||
var takeable: usize = 0;
|
||||
for (table[0..seen]) |descriptor| {
|
||||
if (descriptor.resource_count == 0) continue;
|
||||
if (descriptor.class == @intFromEnum(device.DeviceClass.display)) continue;
|
||||
device.claim(descriptor.id) catch continue; // refused, as it should be
|
||||
takeable += 1;
|
||||
}
|
||||
check("no resource-bearing device is left for the taking", takeable == 0);
|
||||
|
||||
if (failures == 0) {
|
||||
line("device-authority: VERDICT ok ({d} devices, none of them mine)\n", .{seen});
|
||||
} else {
|
||||
line("device-authority: VERDICT FAILED {d} assertion(s)\n", .{failures});
|
||||
}
|
||||
}
|
||||
|
||||
pub fn main(startup: process.Init) void {
|
||||
const role = startup.arguments.get(1) orelse return; // bare (ramdisk sweep): stay silent
|
||||
if (std.mem.eql(u8, role, "run")) run();
|
||||
}
|
||||
@@ -71,10 +71,10 @@ pub fn main() void {
|
||||
return;
|
||||
};
|
||||
|
||||
if (!device.claim(nic_id)) {
|
||||
device.claim(nic_id) catch {
|
||||
_ = logging.write("iommu-fault-test: FAIL claim\n");
|
||||
return;
|
||||
}
|
||||
};
|
||||
var function = pci.Function.map(nic_id, &descriptor) orelse {
|
||||
_ = logging.write("iommu-fault-test: FAIL config-space map\n");
|
||||
return;
|
||||
|
||||
@@ -64,7 +64,7 @@ pub fn main() void {
|
||||
return;
|
||||
};
|
||||
writeLine("pci-cap-test: claiming ethernet function (device {d})\n", .{nic_id});
|
||||
if (!check("claim", device.claim(nic_id))) return;
|
||||
if (!check("claim", if (device.claim(nic_id)) |_| true else |_| false)) return;
|
||||
var function = pci.Function.map(nic_id, &descriptor) orelse {
|
||||
_ = logging.write("pci-cap-test: FAIL config-space map\n");
|
||||
return;
|
||||
|
||||
@@ -35,11 +35,14 @@
|
||||
//! enumerating an xHCI bus (the `usb-storage` scenario);
|
||||
//! - `scanout` — the virtio-gpu driver, which needs an emulated virtio-gpu the
|
||||
//! default harness does not attach (the `virtio-gpu` scenario);
|
||||
//! - `device-manager`, `power` and `usb-transfer` — the three P4b rebased. All
|
||||
//! three come with the device manager: it *is* the first, it spawns the
|
||||
//! discovery service that binds the second, and the xHCI driver it spawns
|
||||
//! binds the third. So booting a provider for any one of them means booting
|
||||
//! the whole driver tree here.
|
||||
//! - `device-manager` and `power` — both come with the device manager: it *is*
|
||||
//! the first and spawns the discovery service that binds the second, so
|
||||
//! booting a provider for either means booting the whole driver tree here.
|
||||
//! - `usb-transfer` — never a registry name at all: several bus processes
|
||||
//! provide it (one per controller) and consumers get their controller's
|
||||
//! channel from the device manager's hello, routed by lineage
|
||||
//! (communication.md "Establishment: two planes"). Its row below checks the
|
||||
//! contract's shape only.
|
||||
//!
|
||||
//! That is the reason this scenario stays at two providers rather than five or
|
||||
//! eight. It is not only the cost of booting half the system to send two
|
||||
@@ -120,12 +123,11 @@ const contracts = [_]Contract{
|
||||
contractOf(vfs_protocol.Protocol, false), // the FAT server — needs a volume
|
||||
contractOf(block_protocol.Protocol, false), // usb-storage — needs the xHCI chain
|
||||
contractOf(scanout_protocol.Protocol, false), // virtio-gpu — needs the device
|
||||
// The three P4b rebased. Each needs the device manager (and, for the last
|
||||
// two, what the device manager starts), which is more than this scenario
|
||||
// boots — see the header.
|
||||
// device-manager and power need the device manager (and what it starts),
|
||||
// which is more than this scenario boots — see the header.
|
||||
contractOf(device_manager_protocol.Protocol, false),
|
||||
contractOf(power_protocol.Protocol, false), // the discovery service
|
||||
contractOf(usb_transfer_protocol.Protocol, false), // the xHCI bus driver
|
||||
contractOf(usb_transfer_protocol.Protocol, false), // never bound: per-controller, routed by hello
|
||||
};
|
||||
|
||||
/// A verb number no protocol in the system defines, and none plausibly will: far
|
||||
|
||||
@@ -0,0 +1,281 @@
|
||||
# Bounds that predate the rule (docs/os-development/bounds.md).
|
||||
#
|
||||
# Generated from the tree as it stood when the check landed, so the gate could start
|
||||
# without a 278-site sweep in front of it. Every line is a compile-time ceiling that
|
||||
# has not yet said what it counts, who decides its size, what it protects, what
|
||||
# happens when it is reached, or how anyone finds out.
|
||||
#
|
||||
# **This list may only shrink.** Declaring a bound means deleting its line; the check
|
||||
# fails if a listed bound is now declared, and fails if a listed bound has vanished.
|
||||
# Nothing may be added to it — a new ceiling declares itself or does not land.
|
||||
#
|
||||
# docs/fixed-bounds-audit.md is the analysis of how they got here.
|
||||
boot/efi.zig:buffer
|
||||
boot/efi.zig:info_buffer
|
||||
boot/efi.zig:maximum_bundled
|
||||
boot/efi.zig:maximum_tree_depth
|
||||
library/device/acpi/aml/interpreter.zig:argbuf
|
||||
library/device/acpi/aml/interpreter.zig:args
|
||||
library/device/acpi/aml/interpreter.zig:locals
|
||||
library/device/acpi/aml/interpreter.zig:maximum_segments
|
||||
library/device/acpi/aml/interpreter.zig:notify_queue
|
||||
library/device/acpi/aml/namespace.zig:segment
|
||||
library/device/acpi/aml/parser.zig:maximum_segments
|
||||
library/device/driver/driver.zig:lookup_attempts
|
||||
library/device/model/device-abi.zig:hid
|
||||
library/device/model/device-abi.zig:maximum_device_resources
|
||||
library/device/pci/pci.zig:bar_virtual
|
||||
library/device/pci/pci.zig:bars
|
||||
library/device/registry/device-registry.zig:cols
|
||||
library/device/registry/device-registry.zig:rules
|
||||
library/device/registry/device-registry.zig:rules
|
||||
library/device/registry/device-registry.zig:rules
|
||||
library/device/registry/device-registry.zig:rules
|
||||
library/device/registry/device-registry.zig:rules
|
||||
library/kernel/channel.zig:bind_attempts
|
||||
library/kernel/channel.zig:name_maximum
|
||||
library/kernel/channel.zig:path_maximum
|
||||
library/kernel/file-system.zig:name_buffer
|
||||
library/kernel/file-system.zig:out
|
||||
library/kernel/logging.zig:buffer
|
||||
library/kernel/process.zig:blob
|
||||
library/kernel/process.zig:receive
|
||||
library/kernel/process.zig:table
|
||||
library/kernel/service.zig:subscriber_capacity
|
||||
library/kernel/start.zig:buffer
|
||||
library/kernel/thread.zig:readers
|
||||
library/kernel/thread.zig:writers
|
||||
library/protocol/device-manager/device-manager-protocol.zig:hid
|
||||
library/protocol/display/display-protocol.zig:max_modes
|
||||
library/protocol/envelope/envelope.zig:packet_maximum
|
||||
library/protocol/envelope/envelope.zig:post_maximum
|
||||
library/protocol/power/power-protocol.zig:hid
|
||||
library/protocol/scanout/scanout-protocol.zig:max_modes
|
||||
library/protocol/usb-transfer/usb-transfer-protocol.zig:max_report_data
|
||||
library/protocol/usb-transfer/usb-transfer-protocol.zig:max_reported_endpoints
|
||||
library/protocol/usb-transfer/usb-transfer-protocol.zig:setup
|
||||
system/abi.zig:errno_maximum
|
||||
system/abi.zig:klog_maximum_message
|
||||
system/abi.zig:maximum_process_name
|
||||
system/boot-handoff.zig:kernel_segments
|
||||
system/drivers/pci-bus/pci-bus.zig:line
|
||||
system/drivers/pci-bus/pci-bus.zig:sub_buffer
|
||||
system/drivers/ps2-bus/mouse-packet.zig:bytes
|
||||
system/drivers/ps2-bus/scancode.zig:pressed
|
||||
system/drivers/ps2-bus/scancode.zig:set2_base
|
||||
system/drivers/ps2-bus/scancode.zig:set2_extended
|
||||
system/drivers/usb-hid/hid-report.zig:keys
|
||||
system/drivers/usb-hid/keyboard.zig:receive
|
||||
system/drivers/usb-hid/mouse.zig:receive
|
||||
system/drivers/usb-storage/bulk-only-transport.zig:cdb
|
||||
system/drivers/usb-storage/scsi.zig:op_read_capacity_10
|
||||
system/drivers/usb-storage/usb-storage.zig:capacity_bytes
|
||||
system/drivers/usb-xhci-bus/usb-xhci-bus.zig:hid_buffer
|
||||
system/drivers/usb-xhci-bus/usb-xhci-bus.zig:maker_buffer
|
||||
system/drivers/usb-xhci-bus/usb-xhci-bus.zig:maker_buffer
|
||||
system/drivers/usb-xhci-bus/usb-xhci-bus.zig:prev_connected
|
||||
system/drivers/usb-xhci-bus/usb-xhci-bus.zig:product_buffer
|
||||
system/drivers/usb-xhci-bus/usb-xhci-bus.zig:product_buffer
|
||||
system/drivers/usb-xhci-bus/usb-xhci-library.zig:buffer
|
||||
system/drivers/usb-xhci-bus/usb-xhci-library.zig:bytes
|
||||
system/drivers/usb-xhci-bus/usb-xhci-library.zig:data
|
||||
system/drivers/usb-xhci-bus/usb-xhci-library.zig:descriptor
|
||||
system/drivers/usb-xhci-bus/usb-xhci-library.zig:head
|
||||
system/drivers/usb-xhci-bus/usb-xhci-library.zig:header
|
||||
system/drivers/usb-xhci-bus/usb-xhci-library.zig:max_subscriptions
|
||||
system/drivers/usb-xhci-bus/usb-xhci-library.zig:port_changes
|
||||
system/drivers/usb-xhci-bus/usb-xhci-library.zig:raw
|
||||
system/drivers/usb-xhci-bus/usb-xhci-library.zig:report_queue_capacity
|
||||
system/drivers/usb-xhci-bus/usb-xhci-library.zig:request_set_hub_depth
|
||||
system/drivers/virtio-gpu/virtio-gpu-protocol.zig:edid
|
||||
system/drivers/virtio-gpu/virtio-gpu-protocol.zig:max_scanouts
|
||||
system/drivers/virtio-gpu/virtio-gpu.zig:descriptors
|
||||
system/drivers/virtio-gpu/virtio-gpu.zig:max_height
|
||||
system/drivers/virtio-gpu/virtio-gpu.zig:max_width
|
||||
system/initial-ramdisk.zig:buffer
|
||||
system/initial-ramdisk.zig:buffer
|
||||
system/initial-ramdisk.zig:maximum_name
|
||||
system/kernel/acpi.zig:APIC
|
||||
system/kernel/acpi.zig:BERT
|
||||
system/kernel/acpi.zig:CPEP
|
||||
system/kernel/acpi.zig:DMAR
|
||||
system/kernel/acpi.zig:DSDT
|
||||
system/kernel/acpi.zig:ECDT
|
||||
system/kernel/acpi.zig:EINJ
|
||||
system/kernel/acpi.zig:ERST
|
||||
system/kernel/acpi.zig:FACP
|
||||
system/kernel/acpi.zig:FACS
|
||||
system/kernel/acpi.zig:HEST
|
||||
system/kernel/acpi.zig:HPET
|
||||
system/kernel/acpi.zig:IVRS
|
||||
system/kernel/acpi.zig:MCFG
|
||||
system/kernel/acpi.zig:MPST
|
||||
system/kernel/acpi.zig:MSCT
|
||||
system/kernel/acpi.zig:PMTT
|
||||
system/kernel/acpi.zig:PSDT
|
||||
system/kernel/acpi.zig:RASF
|
||||
system/kernel/acpi.zig:RSDT
|
||||
system/kernel/acpi.zig:SBST
|
||||
system/kernel/acpi.zig:SLIT
|
||||
system/kernel/acpi.zig:SPCR
|
||||
system/kernel/acpi.zig:SRAT
|
||||
system/kernel/acpi.zig:SSDT
|
||||
system/kernel/acpi.zig:XSDT
|
||||
system/kernel/acpi.zig:aml_block_len
|
||||
system/kernel/acpi.zig:aml_block_physical
|
||||
system/kernel/acpi.zig:holes
|
||||
system/kernel/acpi.zig:maximum_rmrr
|
||||
system/kernel/acpi.zig:nb
|
||||
system/kernel/acpi.zig:nb
|
||||
system/kernel/acpi.zig:nb
|
||||
system/kernel/acpi.zig:oem_id
|
||||
system/kernel/acpi.zig:oem_id
|
||||
system/kernel/acpi.zig:oem_id
|
||||
system/kernel/acpi.zig:oem_table_id
|
||||
system/kernel/acpi.zig:overrides
|
||||
system/kernel/acpi.zig:rmrr_limit_offset
|
||||
system/kernel/acpi.zig:signature
|
||||
system/kernel/acpi.zig:signature
|
||||
system/kernel/acpi.zig:signature
|
||||
system/kernel/architecture/x86_64/cpu.zig:irq_vector_count
|
||||
system/kernel/architecture/x86_64/idt.zig:gate_count
|
||||
system/kernel/architecture/x86_64/ioapic.zig:overrides
|
||||
system/kernel/architecture/x86_64/iommu-amd.zig:buffer
|
||||
system/kernel/architecture/x86_64/iommu-amd.zig:buffer
|
||||
system/kernel/architecture/x86_64/iommu-intel.zig:buffer
|
||||
system/kernel/architecture/x86_64/iommu-intel.zig:buffer
|
||||
system/kernel/architecture/x86_64/iommu-intel.zig:context_table
|
||||
system/kernel/device-model.zig:buffer
|
||||
system/kernel/device-model.zig:cbuf
|
||||
system/kernel/device-model.zig:hid_buffer
|
||||
system/kernel/device-model.zig:maximum_resources
|
||||
system/kernel/device-model.zig:name_buffer
|
||||
system/kernel/device-model.zig:rbuf
|
||||
system/kernel/heap.zig:heap_maximum
|
||||
system/kernel/ipc-synchronous.zig:MESSAGE_MAXIMUM
|
||||
system/kernel/ipc-synchronous.zig:POST_MAXIMUM
|
||||
system/kernel/ipc-synchronous.zig:notify_buffer
|
||||
system/kernel/ipc-synchronous.zig:post_capacity
|
||||
system/kernel/irq.zig:maximum_gsi
|
||||
system/kernel/irq.zig:msi_bound
|
||||
system/kernel/irq.zig:msi_owner
|
||||
system/kernel/irq.zig:vector_gsi
|
||||
system/kernel/kernel.zig:buffer
|
||||
system/kernel/kernel.zig:buffer
|
||||
system/kernel/kernel.zig:buffer
|
||||
system/kernel/kernel.zig:isos
|
||||
system/kernel/kernel.zig:maximum_wake_attempts
|
||||
system/kernel/log-ring.zig:message
|
||||
system/kernel/log-ring.zig:out
|
||||
system/kernel/log.zig:buffer
|
||||
system/kernel/log.zig:maximum_sinks
|
||||
system/kernel/log.zig:message
|
||||
system/kernel/log.zig:ring_capacity
|
||||
system/kernel/process.zig:chunk
|
||||
system/kernel/process.zig:chunk
|
||||
system/kernel/process.zig:chunk
|
||||
system/kernel/process.zig:chunk
|
||||
system/kernel/process.zig:exit_record_capacity
|
||||
system/kernel/process.zig:exit_subscriber_capacity
|
||||
system/kernel/process.zig:maximum_argument_bytes
|
||||
system/kernel/process.zig:maximum_arguments
|
||||
system/kernel/process.zig:maximum_dma_regions
|
||||
system/kernel/process.zig:maximum_mmap_pages
|
||||
system/kernel/process.zig:maximum_mount_prefix
|
||||
system/kernel/process.zig:maximum_mount_rewrite
|
||||
system/kernel/process.zig:maximum_pages
|
||||
system/kernel/process.zig:maximum_resolve_path
|
||||
system/kernel/process.zig:maximum_segments
|
||||
system/kernel/process.zig:maximum_shared_memory_pages
|
||||
system/kernel/process.zig:name_buffer
|
||||
system/kernel/process.zig:timer_capacity
|
||||
system/kernel/process.zig:word_bytes
|
||||
system/kernel/process.zig:write_buffer
|
||||
system/kernel/scheduler.zig:ipc_maximum_handles
|
||||
system/kernel/scheduler.zig:maximum_space_mappings
|
||||
system/kernel/vfs.zig:maximum_directories
|
||||
system/kernel/vfs.zig:maximum_mounts
|
||||
system/kernel/vfs.zig:maximum_prefix
|
||||
system/kernel/vfs.zig:maximum_rewrite
|
||||
system/services/acpi/acpi.zig:blocks
|
||||
system/services/acpi/acpi.zig:buffer
|
||||
system/services/acpi/acpi.zig:hid
|
||||
system/services/acpi/acpi.zig:mmio_scratch
|
||||
system/services/acpi/acpi.zig:name
|
||||
system/services/acpi/acpi.zig:registered
|
||||
system/services/device-manager/device-manager.zig:arguments
|
||||
system/services/device-manager/device-manager.zig:id_text
|
||||
system/services/device-manager/device-manager.zig:maximum_children
|
||||
system/services/device-manager/device-manager.zig:maximum_drivers
|
||||
system/services/device-manager/device-manager.zig:name_buffer
|
||||
system/services/device-manager/device-manager.zig:registry_rules
|
||||
system/services/device-manager/device-manager.zig:registry_source
|
||||
system/services/display/backend.zig:device_table
|
||||
system/services/display/backend.zig:line
|
||||
system/services/display/compositor.zig:capacity
|
||||
system/services/display/compositor.zig:maximum_columns
|
||||
system/services/display/compositor.zig:maximum_rects
|
||||
system/services/display/compositor.zig:maximum_rows
|
||||
system/services/display/display.zig:line
|
||||
system/services/display/display.zig:line
|
||||
system/services/display/display.zig:line
|
||||
system/services/display/display.zig:list
|
||||
system/services/display/display.zig:maximum_layers
|
||||
system/services/display/display.zig:mode_list
|
||||
system/services/fat/engine.zig:buf
|
||||
system/services/fat/engine.zig:buf
|
||||
system/services/fat/engine.zig:buf
|
||||
system/services/fat/engine.zig:buffer
|
||||
system/services/fat/engine.zig:cached_back
|
||||
system/services/fat/engine.zig:chunk
|
||||
system/services/fat/engine.zig:chunk
|
||||
system/services/fat/engine.zig:chunk
|
||||
system/services/fat/engine.zig:device_back
|
||||
system/services/fat/engine.zig:display
|
||||
system/services/fat/engine.zig:long_name
|
||||
system/services/fat/engine.zig:long_name
|
||||
system/services/fat/engine.zig:long_name
|
||||
system/services/fat/engine.zig:max_transfer_sectors
|
||||
system/services/fat/engine.zig:pair
|
||||
system/services/fat/engine.zig:pair
|
||||
system/services/fat/engine.zig:payload
|
||||
system/services/fat/engine.zig:payload
|
||||
system/services/fat/engine.zig:payload
|
||||
system/services/fat/engine.zig:readback
|
||||
system/services/fat/engine.zig:run
|
||||
system/services/fat/engine.zig:run
|
||||
system/services/fat/engine.zig:short
|
||||
system/services/fat/engine.zig:short
|
||||
system/services/fat/engine.zig:short
|
||||
system/services/fat/engine.zig:stem
|
||||
system/services/fat/engine.zig:tail_buffer
|
||||
system/services/fat/engine.zig:units
|
||||
system/services/fat/engine.zig:value
|
||||
system/services/fat/engine.zig:value
|
||||
system/services/fat/engine.zig:window
|
||||
system/services/fat/on-disk.zig:filesystem_type
|
||||
system/services/fat/on-disk.zig:filesystem_type
|
||||
system/services/fat/on-disk.zig:jump
|
||||
system/services/fat/on-disk.zig:name
|
||||
system/services/fat/on-disk.zig:name1
|
||||
system/services/fat/on-disk.zig:name2
|
||||
system/services/fat/on-disk.zig:name3
|
||||
system/services/fat/on-disk.zig:oem_name
|
||||
system/services/fat/on-disk.zig:volume_label
|
||||
system/services/fat/on-disk.zig:volume_label
|
||||
system/services/init/init.zig:binary
|
||||
system/services/init/init.zig:init_csv
|
||||
system/services/init/init.zig:max_service_args
|
||||
system/services/init/init.zig:max_services
|
||||
system/services/init/init.zig:maximum_bindings
|
||||
system/services/init/init.zig:maximum_grants
|
||||
system/services/init/init.zig:maximum_name
|
||||
system/services/init/init.zig:maximum_restarts
|
||||
system/services/init/init.zig:process_table
|
||||
system/services/init/init.zig:protocol_csv
|
||||
system/services/logger/logger.zig:chunk
|
||||
system/services/logger/logger.zig:gap_line
|
||||
system/services/logger/logger.zig:line
|
||||
system/services/logger/logger.zig:line
|
||||
system/services/logger/logger.zig:maximum_files
|
||||
system/services/logger/logger.zig:stamp
|
||||
@@ -0,0 +1,189 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Every compile-time ceiling states what it is doing there.
|
||||
|
||||
A *bound* is a number chosen at compile time that decides how much of something the
|
||||
code can hold: `const maximum_devices = 64`, `var below: [64]Range`, `var blob: [512]u8`.
|
||||
Different units, one shape, and one recurring way of going wrong — see
|
||||
docs/fixed-bounds-audit.md, where 235 of them turned up, 139 on quantities the machine
|
||||
or a file decides rather than us, and 171 silent when reached.
|
||||
|
||||
This is the gate that keeps new ones from joining them. It does not resize anything and
|
||||
it makes no judgement about whether a bound should exist; it only refuses one that will
|
||||
not say what it is for. The declaration is a doc comment immediately above:
|
||||
|
||||
/// bound: logical CPUs the kernel tracks
|
||||
/// decided-by: hardware
|
||||
/// protects: the per-CPU bookkeeping arrays, which are sized at compile time
|
||||
/// at-limit: degrade - surplus cores are left parked, never brought online
|
||||
/// observed-by: platform.cpusDropped() -> the WARNING at kernel.zig:281
|
||||
pub const maximum_cpus = 128;
|
||||
|
||||
`decided-by` is the field the audit turned on: `hardware` and `external` mean the
|
||||
quantity is not ours to choose, and a fixed bound on one of those is a defect rather
|
||||
than a tunable. `at-limit`'s vocabulary is closed on purpose — there is no `silent`,
|
||||
no `drop`, and nothing meaning *allow*, so the behaviours that caused the damage cannot
|
||||
be written down. `truncate` is legal only with a marker the reader can see.
|
||||
|
||||
An array length that names a declared bound (`[maximum_devices]Descriptor`) is not
|
||||
itself a bound: the number lives at the declaration, and that is where it is declared.
|
||||
Only literal lengths are flagged, which pushes ceilings toward having names.
|
||||
|
||||
The ~235 that already exist are listed in tools/bounds-allowlist.txt so this can land
|
||||
without a tree-wide sweep in front of it. That list may only shrink: declaring a bound
|
||||
means deleting its line, and a stale line is an error too.
|
||||
|
||||
Usage: check-bounds.py [--list] [repo-root]
|
||||
--list print every undeclared bound found, for regenerating the allowlist
|
||||
"""
|
||||
|
||||
import re
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
# Directories worth gating. `test/` is excluded: a fixture's `[100]MemoryRegion` is a
|
||||
# test input, not a ceiling the system runs into.
|
||||
ROOTS = ("system", "library", "boot")
|
||||
SKIP_PARTS = {".zig-cache", "zig-out", ".git", ".claude", "vendor", "generated"}
|
||||
SKIP_FILES = {"tests.zig"}
|
||||
|
||||
FIELDS = ("bound", "decided-by", "protects", "at-limit", "observed-by")
|
||||
DECIDED_BY = {"hardware", "external", "ours"}
|
||||
AT_LIMIT = {"refuse", "degrade", "truncate", "grow"}
|
||||
|
||||
# A `const` whose name reads like a ceiling and whose value is an integer literal.
|
||||
NAMED = re.compile(
|
||||
r"^\s*(?:pub\s+)?const\s+([A-Za-z_]\w*)\s*(?::\s*[\w.\[\]]+\s*)?=\s*"
|
||||
r"(\d[\d_]*|0x[0-9a-fA-F_]+)\s*(?:\*\s*\d[\d_]*\s*)*;"
|
||||
)
|
||||
NAME_IS_BOUND = re.compile(r"(^|_)(maximum|max|limit|capacity|depth|attempts|count)($|_)", re.I)
|
||||
|
||||
# A declaration or struct field whose type carries a *literal* array length.
|
||||
ARRAY_DECL = re.compile(r"^\s*(?:pub\s+)?(?:const|var)\s+([A-Za-z_]\w*)\s*:[^=]*?\[\s*(\d[\d_]*)\s*\]")
|
||||
ARRAY_FIELD = re.compile(r"^\s*([A-Za-z_]\w*)\s*:\s*\[\s*(\d[\d_]*)\s*\]")
|
||||
|
||||
# Struct padding and reserved fields are shapes, not ceilings — nothing is ever "held"
|
||||
# in them. Everything else with a literal length is a candidate, *including* the tidy
|
||||
# powers of two: `[512]u8` and `[64]Range` were the two worst findings in the audit, and
|
||||
# any size-based exemption would have skipped exactly them. A length that is genuinely a
|
||||
# fact rather than a ceiling says so in its declaration ("decided-by: hardware, the PCI
|
||||
# spec gives a function 6 BARs") — that is what the declaration is for.
|
||||
NOT_A_BOUND_NAME = re.compile(r"^_*(padding|pad|reserved|unused|spare)\d*$", re.I)
|
||||
|
||||
|
||||
def sources(root: Path):
|
||||
for top in ROOTS:
|
||||
base = root / top
|
||||
if not base.is_dir():
|
||||
continue
|
||||
for path in sorted(base.rglob("*.zig")):
|
||||
if SKIP_PARTS & set(path.parts) or path.name in SKIP_FILES:
|
||||
continue
|
||||
yield path
|
||||
|
||||
|
||||
def declaration_above(lines, index):
|
||||
"""The `/// key: value` block immediately above line `index`, as a dict."""
|
||||
fields = {}
|
||||
i = index - 1
|
||||
while i >= 0:
|
||||
stripped = lines[i].strip()
|
||||
if not stripped.startswith("///"):
|
||||
break
|
||||
body = stripped[3:].strip()
|
||||
match = re.match(r"([a-z-]+):\s*(.+)", body)
|
||||
if match:
|
||||
fields[match.group(1)] = match.group(2).strip()
|
||||
i -= 1
|
||||
return fields
|
||||
|
||||
|
||||
def problems_with(fields):
|
||||
"""Why a declaration is not acceptable, or an empty list."""
|
||||
missing = [f for f in FIELDS if f not in fields or not fields[f]]
|
||||
if missing:
|
||||
return ["missing " + ", ".join(missing)]
|
||||
out = []
|
||||
if fields["decided-by"] not in DECIDED_BY:
|
||||
out.append(f"decided-by must be one of {sorted(DECIDED_BY)}, not {fields['decided-by']!r}")
|
||||
verb = fields["at-limit"].split()[0].strip("-:,").lower()
|
||||
if verb not in AT_LIMIT:
|
||||
out.append(
|
||||
f"at-limit must start with one of {sorted(AT_LIMIT)}, not {verb!r}. "
|
||||
"There is deliberately no way to say 'silent', 'drop', or anything meaning 'allow'"
|
||||
)
|
||||
if verb == "truncate" and len(fields["at-limit"].split()) < 3:
|
||||
out.append("at-limit: truncate must say how a reader can TELL it happened")
|
||||
return out
|
||||
|
||||
|
||||
def find(root: Path):
|
||||
"""Every bound-shaped declaration: (relative path, name, value, line, fields)."""
|
||||
for path in sources(root):
|
||||
rel = path.relative_to(root).as_posix()
|
||||
lines = path.read_text(encoding="utf-8", errors="replace").split("\n")
|
||||
for n, line in enumerate(lines):
|
||||
if line.lstrip().startswith("//"):
|
||||
continue
|
||||
name = value = None
|
||||
m = NAMED.match(line)
|
||||
if m and NAME_IS_BOUND.search(m.group(1)):
|
||||
name, value = m.group(1), m.group(2)
|
||||
else:
|
||||
m = ARRAY_DECL.match(line) or ARRAY_FIELD.match(line)
|
||||
if m and not NOT_A_BOUND_NAME.match(m.group(1)):
|
||||
name, value = m.group(1), m.group(2)
|
||||
if name:
|
||||
yield rel, name, value, n + 1, declaration_above(lines, n)
|
||||
|
||||
|
||||
def main():
|
||||
argv = [a for a in sys.argv[1:] if not a.startswith("--")]
|
||||
listing = "--list" in sys.argv
|
||||
root = Path(argv[0]) if argv else Path(__file__).resolve().parent.parent
|
||||
|
||||
allow_path = root / "tools" / "bounds-allowlist.txt"
|
||||
allowed = set()
|
||||
if allow_path.exists():
|
||||
for raw in allow_path.read_text().split("\n"):
|
||||
entry = raw.split("#", 1)[0].strip()
|
||||
if entry:
|
||||
allowed.add(entry)
|
||||
|
||||
undeclared, bad, seen = [], [], set()
|
||||
for rel, name, value, line, fields in find(root):
|
||||
key = f"{rel}:{name}"
|
||||
seen.add(key)
|
||||
if not fields:
|
||||
(undeclared if key not in allowed else []).append((key, value, line))
|
||||
continue
|
||||
for why in problems_with(fields):
|
||||
bad.append((key, line, why))
|
||||
if key in allowed and not problems_with(fields):
|
||||
bad.append((key, line, "now declared — delete its line from tools/bounds-allowlist.txt"))
|
||||
|
||||
if listing:
|
||||
for key, value, line in sorted(undeclared):
|
||||
print(f"{key} # = {value}, line {line}")
|
||||
for key in sorted(allowed - seen):
|
||||
print(f"# STALE: {key}")
|
||||
return 0
|
||||
|
||||
stale = sorted(allowed - seen)
|
||||
if not undeclared and not bad and not stale:
|
||||
return 0
|
||||
|
||||
print("bounds check failed\n", file=sys.stderr)
|
||||
for key, value, line in sorted(undeclared):
|
||||
print(f" {key} (= {value}, line {line})", file=sys.stderr)
|
||||
print(" no declaration. A ceiling states what it counts, who decides its", file=sys.stderr)
|
||||
print(" size, what it protects, what happens at the limit, and how you", file=sys.stderr)
|
||||
print(" find out. See docs/os-development/bounds.md.", file=sys.stderr)
|
||||
for key, line, why in sorted(bad):
|
||||
print(f" {key} (line {line}): {why}", file=sys.stderr)
|
||||
for key in stale:
|
||||
print(f" {key}: allowlisted but no longer found — delete its line", file=sys.stderr)
|
||||
return 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
Reference in New Issue
Block a user