From a86559648ecab6a3c1bc5c1a868f519d2d4b1737 Mon Sep 17 00:00:00 2001 From: Daniel Samson <12231216+daniel-samson@users.noreply.github.com> Date: Sat, 8 Aug 2026 11:09:54 +0100 Subject: [PATCH] kernel: a refusal names its rule, and two bounds stop failing open MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit An AMD Ryzen booted to a working compositor with no USB and no storage, and the log said only "register refused". A tree-wide audit of every compile-time ceiling followed: 235 of them, 139 on quantities the machine or a file decides rather than us, 5 documented anywhere, 171 silent when reached. docs/fixed-bounds-audit.md has the inventory. Errno attribution. The errno space was split between the kernel and the envelope, free to drift; it is now one list in system/abi.zig, restated on both sides, with a comptime check in library/device/driver where the two halves are visible. device_register's six refusals and device_claim's three are distinct codes, so a bus driver can say which rule stopped it, and BadParent splits into NoSuchParent and NotYourParent. pci-bus reconciles found against registered instead of counting refused functions as found. Idempotency ordering. The child cap was checked before the identity match, so a restarted bus was refused its own devices — the supervision restart the system leans on ratcheted toward a degraded machine. A re-registration consumes no slot and is now admitted first. IOMMU fail-closed. confineDevice returned success for a device id past the confinement table, leaving the device outside every domain while the caller believed it confined — unreachable only while ids stop at 64, which both the inventory move and a hardware-reported domain count would change. It refuses now, and the coupling to the broker's device cap is a comptime assert rather than a sentence in a comment. PCI apertures. The bridge's MMIO apertures are derived from the holes in the firmware memory map, and the derivation copied sub-4 GiB entries into a fixed [64] array and skipped the rest. A skipped region is not merely lost: the gap finder concludes it is free, so a real machine's 60-200 entry map yields an aperture over live RAM, and containment then admits a child BAR covering kernel memory. Rewritten to walk the map in place, with the hole finder extracted as a pure function and driven by a synthetic 100-entry map in a new test case. Both new tests were verified to fail on the old code. parameters.zig gains the rationale it was missing and loses a stale sentence pointing at the wrong file; vdso.md documents the errno space, including EPEER, which had no written meaning anywhere. docs/os-development/bounds.md is how a ceiling is declared from here. docs/bounds-track-plan.md is the plan to remove the ones we invented. Suite 114 -> 115. --- docs/bounds-track-plan.md | 212 +++++++++ docs/fixed-bounds-audit.md | 403 ++++++++++++++++++ docs/os-development/bounds.md | 140 ++++++ docs/os-development/device-authority.md | 173 ++++++++ docs/os-development/vdso.md | 51 +++ library/device/driver/driver.zig | 70 ++- system/abi.zig | 42 +- system/drivers/pci-bus/pci-bus.zig | 26 +- system/drivers/ps2-bus/ps2-bus.zig | 26 +- system/drivers/usb-xhci-bus/usb-xhci-bus.zig | 10 +- system/drivers/virtio-gpu/virtio-gpu.zig | 6 +- system/kernel/acpi.zig | 122 ++++-- system/kernel/devices-broker.zig | 117 +++-- system/kernel/iommu.zig | 31 +- system/kernel/ipc-synchronous.zig | 25 +- system/kernel/platform.zig | 8 + system/kernel/process.zig | 66 +-- system/kernel/tests.zig | 111 ++++- system/parameters.zig | 10 +- system/services/acpi/acpi.zig | 10 +- system/services/display/backend.zig | 8 +- test/qemu_test.py | 11 + .../system/services/crash-test/crash-test.zig | 4 +- .../iommu-fault-test/iommu-fault-test.zig | 4 +- .../services/pci-cap-test/pci-cap-test.zig | 2 +- 25 files changed, 1520 insertions(+), 168 deletions(-) create mode 100644 docs/bounds-track-plan.md create mode 100644 docs/fixed-bounds-audit.md create mode 100644 docs/os-development/bounds.md create mode 100644 docs/os-development/device-authority.md diff --git a/docs/bounds-track-plan.md b/docs/bounds-track-plan.md new file mode 100644 index 0000000..a8a5017 --- /dev/null +++ b/docs/bounds-track-plan.md @@ -0,0 +1,212 @@ +# The bounds track: removing the numbers we invented + +*Plan, 2026-08-08. Follows [fixed-bounds-audit.md](fixed-bounds-audit.md) (235 ceilings, +139 on quantities we do not choose) and the AMD Ryzen that found the first one.* + +## The principles this is derived from + +1. **danOS is a microkernel.** Minimise what the kernel is responsible for; move + responsibility to user space so it can be restarted, or fixed live during + development, without taking the system down. +2. **Implement the specifications correctly**, with the limits those specifications + define — not limits we decide. +3. **Move as much responsibility as possible to user space** (the device manager). +4. **What remains in the kernel is minimal.** +5. **What remains in the kernel is there for security or for a hardware limitation.** + Nothing else earns a place. + +Principle 5 is the test every bound is put to. For each one: *is this here because of +security, or because of a hardware limitation?* If neither, the storage does not belong +in the kernel and the bound is not a number to be resized — it is a thing to be moved or +deleted. + +Applying it to the case that started this: + +- `maximum_devices = 64` bounds an inventory of hardware. An inventory is neither a + security control nor a hardware limitation. **The table is in the wrong place**; the + number is a symptom. +- `maximum_children_per_parent = 16` exists because `device_claim` is unauthenticated — + any process can claim any unclaimed device ([devices-broker.zig:164](../system/kernel/devices-broker.zig:164) + checks only that the device exists and is free). The cap is a crude proxy for an + authorisation the kernel does not perform. **Fix the authorisation and the cap has + nothing to defend.** +- `maximum_domains = 64` bounds IOMMU translation domains. Security — stays in the + kernel. But VT-d and AMD-Vi both *report* how many domains they support in a + capability register. Principle 2: read it. We chose 64 without asking. + +## The security invariants + +Every phase must leave all five standing. This is the "without punching a hole" half of +the brief, and each phase below states how it is checked. + +- **I1 Containment.** A process may map only physical memory inside a resource it was + granted. A bus may subdivide only what it already holds. +- **I2 Confinement.** A DMA-capable device is under IOMMU translation before its driver + can program it, or it is not driven at all. +- **I3 No self-granted authority.** A process holds what it was handed. It cannot name + its way into holding more. +- **I4 Death releases everything.** Every resource a task held is reclaimed when it + dies, on every path out. +- **I5 Refusal is attributable.** Every refusal names the rule that refused it. + +## Phase 0 — Done + +- **Errno attribution.** One errno space in `system/abi.zig`; `device_register`'s six + refusals and `device_claim`'s three are distinct codes; call sites name the reason; + `pci-bus` reconciles found against registered. (I5) +- **Idempotency ordering.** A re-registration consumes no slot, so a full parent + re-admits an identical child. A restarted bus is no longer billed for what it + rediscovers. + +Suite 114/114. + +## Phase 1 — Reclamation + +**Nothing may become dynamic before this.** Today `count` only ever increases and +`releaseAllOwnedBy` clears a dead driver's *claims* but not its *registrations*. With a +fixed table that is a slow march to the cap; with dynamic storage it is an unbounded +leak, and every supervisor restart makes it worse. + +- Extend the existing death sweep so a task's registrations go with its claims. +- A registration whose owner is gone is removed; its children are re-parented or removed + with it (they cannot outlive the authority that published them). +- Test: register under a claimed parent, kill the owner, assert the entries are gone and + the ids are not reused while any handle to them lives. + +Invariant: **I4**. + +## Phase 2 — Close the authorisation hole + +The device manager already decides which driver gets which device — it matches against +`devices.csv` and spawns the driver with the device id as `argv[1]`. Nothing binds that +decision to the kernel's `claim`. A driver passes an integer; the kernel checks only +that the device is free. + +Per principles 3 and 5: **the decision stays in user space; the kernel enforces only +possession.** The manager hands the driver the device it matched; the kernel's job is +that a driver holds what it was handed and nothing else. + +- The manager passes a device to the driver it spawned, over the existing cap-passing + path. Possession is the authority. +- `device_claim` stops being a way to *acquire* a device by naming it. +- Exclusivity stops being a broker refusing a second claimant and becomes the ordinary + property of a thing only one process was given. + +**`maximum_children_per_parent` is deleted here**, because after this a bus driver's +children are the devices it actually enumerated under a bus it was actually given, and +the rogue-driver-fills-the-table threat the cap was written for no longer exists. + +Invariants: **I3** (the point of the phase), **I1** (containment is unchanged and still +checked on every subdivision), **I5**. + +Acceptance: a driver that names a device it was not given is refused, with its own +errno. The QEMU suite gains an adversarial case for it — the audit's lesson was that +"the suite contains no attacker". + +## Phase 3 — The inventory moves to user space + +The kernel reads only three things out of a device descriptor: **physical ranges** (to +check a mapping falls inside one), **interrupt numbers**, and **one PCI BDF** (to key an +IOMMU domain). Vendor and device ids, class triples, subsystem ids, human-readable +names, bus numbers and parent links are stored solely so `device_enumerate` can hand +them back. That is the kernel acting as a distribution mechanism for data it does not +use — principle 5 excludes it. + +- **Zero-resource devices leave the kernel entirely.** A USB device addressed through + its controller conveys no mapping authority; there is nothing for the kernel to + enforce. It is pure inventory and belongs to the device manager. (This is also the + case that sidesteps containment, which is why the cap existed.) +- Identity and topology move to the manager, which already receives them as + `child_added` reports and already holds the authoritative picture. +- `device_enumerate` retires; callers ask the manager, whose protocol already reserves + an `enumerate` verb. Public-ABI change — `docs/os-development/vdso.md` documents it. +- What the kernel keeps: for each device that carries resources, the ranges, the GSIs, + the BDF, and the owner. + +After this, the kernel's table holds only resource-bearing devices, and the remaining +count is bounded by what the machine physically has rather than by us. + +Invariants: **I1**, **I2** unchanged — both operate on resources, which do not move. +**I4** must be re-checked: the manager's table now needs its own reclamation, and it is +restartable, so it must be able to rebuild from the buses. + +## Phase 4 — Ask the hardware and the specification + +Principle 2, applied to every remaining bound. Each of these is a number the machine or +the standard already states, which we replaced with a guess. Independent of each other; +can proceed in any order. + +| Today | Ask instead | +|---|---| +| `maximum_domains = 64` (IOMMU) | the VT-d / AMD-Vi capability register reports the domains supported | +| `max_devices = 8` (xHCI slots) | `HCSPARAMS1.MaxSlots` — the controller says (1–255) | +| `max_interfaces = 4`, `max_endpoints` | the configuration descriptor says | +| `blob: [512]u8` (USB config) | the device's `wTotalLength` | +| `below: [64]Range` (memory map) | UEFI reports the descriptor count | +| AML blobs capped at 6 | the XSDT's length field gives the entry count | +| `maximum_cpus = 128` | the MADT entry count | +| `maximum_gsi = 24` | the I/O APIC's redirection-entry count; and more than one I/O APIC exists | +| MSI-X vectors | the capability's table-size field (up to 2048) | + +Several of these are in user space already (xHCI, USB descriptors) and are ordinary +allocations — principle 1 means those are also the safest to do first, since a mistake +restarts a driver rather than the machine. + +Two in this table are **also** correctness fixes the audit found, and should carry their +regression tests: the xHCI `max_interfaces` path misattributes a fifth interface's +endpoints to interface 3, and `below: [64]Range` silently turns occupied RAM into a PCI +aperture — which is an **I1 violation reachable on real hardware**, not merely a lost +device. That one is the highest-priority item in this phase. + +## Phase 5 — What legitimately remains + +After phases 1–4 the survivors should be only: + +- **Pre-allocator storage**: the PMM's own frame bitmap, the memory map the loader hands + over, the bootstrap page tables. You cannot allocate the allocator. (Hardware/boot + limitation — principle 5 admits these.) +- **Interrupt-context storage**: the IST stack and anything an exception path touches + without allocating. +- **Wire structures** whose layout the other side of a trust boundary parses. +- **Facts that are not ceilings**: a page is 4096 bytes; an ACPI name segment is 4. + +Each is declared per [bounds.md](os-development/bounds.md) — what it counts, who decides +its size, what it protects, what happens at the limit, how you find out. And two numbers +that must agree agree in code, not in a comment: + +```zig +comptime { + if (maximum_domains != devices_broker.maximum_devices) + @compileError("iommu.confined is indexed by device id; an id past its end is " ++ + "left unconfined while confineDevice still reports success"); +} +``` + +## The one that must not wait + +[`iommu.zig:107`](../system/kernel/iommu.zig:107) — `if (device_id >= confined.len) return +true;` — returns *success* without confining. It is unreachable today only because +device ids stop at 64. **Phases 3 and 4 both change the device count, and either makes +it live.** Fix it before them: out of range must refuse, never allow. (I2) + +This is also the standing rule the audit argues for: at a bound, the safe direction is +refusal. A ceiling that fails open is not a limit, it is a switch that turns the +protection off. + +## How this is verified + +- The QEMU suite is the arbiter at every step; it is 114 cases and must stay green. +- Each fix lands with a test that **fails before it** — as the idempotency reorder did, + where exactly one assertion flipped. +- Adversarial cases for I1–I3 specifically: the audit's six real defects were all found + by asking "what would an attacker do", and the suite had never asked. +- The Ryzen is the acceptance test. It is the machine that found this, and the one that + proves it fixed. + +## Sequencing + +Phase 1 gates everything. Phase 2 gates phase 3 — the inventory cannot move until +authority is sound, or moving it is the hole. Phase 4 is independent and its user-space +items are the safest work in the track. Phase 5 is the record of what survived. + +The IOMMU fail-open is fixed before phase 3 or 4 touches the device count. diff --git a/docs/fixed-bounds-audit.md b/docs/fixed-bounds-audit.md new file mode 100644 index 0000000..7803f71 --- /dev/null +++ b/docs/fixed-bounds-audit.md @@ -0,0 +1,403 @@ +# Fixed-bounds audit + +*2026-08-07. A tree-wide audit of every compile-time ceiling on a runtime quantity, +commissioned after an AMD Ryzen desktop booted to a working compositor with no USB +and no storage. Seven parallel sweeps, adversarial verification of each finding, and +a completeness critic. 235 bounds confirmed.* + +The audit was not commissioned to fix that machine. It was commissioned to answer a +different question: **why did a machine have to find this?** The answer is in the +first table below, and it is not that anyone failed to predict an AMD desktop. + +## What the audit found + +| | | +|---|---| +| Bounds confirmed | **235** | +| Decided by hardware or external data, not by us | **139** (59%) | +| Recorded in `system/parameters.zig` | **5** (2%) | +| With no comment explaining the number at all | **97** (41%) | +| Silent when exceeded — no log, no counter, no error | **171** (73%) | +| Severity critical / high | **17 / 26** | + +The project has a stated convention: tunables live in `system/parameters.zig` with +their reasoning attached. Two percent of them do. That is the finding — not any +individual number. + +And 59% of these are not tunables at all. They are guesses about someone else's +computer: how many PCI functions a board has, how many SSDTs its firmware ships, +how many descriptors its memory map carries, how many interfaces a USB headset +declares. A fixed bound on a quantity the machine decides is not a knob. It is a +defect with a plausible-looking number in it. + +## Why "raise the number" is not available + +This is the part that matters most, and it was found by the completeness critic +rather than by any of the seven sweeps. + +`system/kernel/iommu.zig:97` sizes the per-device confinement table by +`maximum_domains = 64` — "64 mirrors devices-broker's device cap" — and indexes it +by **device id**. Line 107: + +```zig +pub fn confineDevice(device_id: u64, bdf: u16, owner: u32) bool { + if (!active) return true; + if (device_id >= confined.len) return true; // unusual id; leave it to fail-open +``` + +It returns `true` — success — without confining the device. The doc comment three +lines above states the opposite invariant: *"a claim that can't be confined must not +stand."* + +Device ids are assigned `d.id = count` with `count < maximum_devices`, so today ids +run 0–63 and that branch is unreachable. **It becomes reachable the moment +`maximum_devices` is raised above 64.** Every device with id ≥ 64 would then be +claimed by a ring-3 driver, reported as successfully confined, and left outside every +IOMMU domain — an unconfined DMA master with a driver holding it. + +So the one-line fix for the Ryzen — raise 64 to 512 — is a privilege escalation. Not +inelegant: escalating. Nothing in the type system, the tests or the comments would +have caught it, because the two constants agree only by a sentence in a comment. + +## The Ryzen was three bugs, not one + +Each of these independently produces "the xHCI and SATA controllers are missing." +Fixing any one of them leaves the machine broken by the next. + +1. **`devices-broker.zig:33`, `maximum_children_per_parent = 16`** — `pci-bus` + registers *every* discovered function as a direct child of the one host bridge, so + 16 is the ceiling on PCI functions for the whole machine. Enumeration runs in + bus/device/function order, so low-numbered chipset functions consume all 16 and the + high-numbered controllers — xHCI at 0x14, SATA at 0x17 — are refused. This is the + one that fired. +2. **`device-manager.zig:174`, `maximum_children = 64`** — the *userspace* inventory + has its own flat table. ACPI contributes ~34 nodes before PCI is scanned. Refusal + returns `-ENOSPC`, which `pci-bus.zig:246` discards and `acpi.zig:257` swallows in + an empty `catch {}`. The log line "child added" is printed *before* the status is + consulted, so the log says the device was added when it was not. +3. **`acpi.zig:651`, `gaps = [3]Range`** — only three sub-4 GiB MMIO holes become PCI + bridge apertures. A BAR landing in a fourth hole fails containment and is refused, + indistinguishably from a full table. + +Three ceilings, three teams of one, one symptom. "Raise the limit" would have moved +the failure to the next one and produced a second debugging session from a second +photograph. + +## Two latent security findings + +**`acpi.zig:627`, `below = [64]Range`.** Firmware memory-map entries below 4 GiB, used +to derive the PCI bridge's apertures *from the gaps between described regions*. Past the +64th entry: `continue`. A dropped region is not merely missing — it vanishes from the +"described space" the gap-finder subtracts, so occupied physical memory is concluded to +be a free MMIO hole and registered as a bridge aperture. `devices-broker.contains()` +then admits a child BAR covering RAM, and its claimant can `mmio_map` it: a ring-3 +read/write window onto kernel memory. Real UEFI maps carry 60–200 descriptors; OVMF +carries 15–25, which is why the suite has never approached it. + +**`devices-broker.zig:217`, the PCI requester id is a `u16`** with no segment field, +propagated unbroken into both IOMMU backends. On a multi-segment machine two physically +distinct functions alias onto one translation structure. + +Both are the same shape as the finding above: a bound whose failure mode is not "we run +out" but "the protection silently stops applying." + +## Bugs found in passing + +Not bounds, but found by looking at what happens at the bound: + +- **`usb-xhci-library.zig:1331`** — `else if (device.interface_count < max_interfaces)` + has no `else`, so at the limit `current` is left pointing at interface 3 and the 5th + interface's endpoints are appended to interface 3's array. A class driver bound to + interface 3 can be handed an endpoint belonging to another interface. The + alternate-setting arm one line above does `current = null` correctly. +- **`usb-xhci-library.zig:899/1102`** — `allocateDevice()` failing returns after + `enableSlot()` already succeeded, with no Disable Slot. Every failed attempt + permanently leaks a controller slot. +- **`acpi.zig:445`** — the AML loop reserves `maximum_resources - 2` slots and the + function then adds **four** more resources. At 5 AML blocks the FADT is lost; at 6+ + the broad IRQ window is lost too, so every legacy-IRQ device (PS/2 keyboard at IRQ 1) + fails containment. Real firmware ships 5–15 SSDTs; QEMU ships 2. +- **`fat.zig:252`** — a `u64` protocol offset reaches a `u32` engine parameter through a + bare `@intCast`. In ReleaseSafe — this project's build mode — that panics. +- **`boot/efi.zig:318`** — the bootstrap page tables map [0, 4 GiB) and nothing checks + that the handoff buffers UEFI allocated (BootInformation, the memory-map pool, the + initial ramdisk) landed below it. Several firmware implementations allocate top-down. + Above 4 GiB it is a triple fault with no output at all. + +## Headroom actually measured + +Every other bound here is hypothetical. One is a schedule: +`system/configuration/protocol.csv`, the manifest deciding which binary may bind which +protocol name, is at **50 of 64 grant rows and 12,668 of 16,384 bytes** — 78% and 77%. + +To its credit, both overflow paths log (`init.zig:147` and `:238`). It will fail +loudly. It will still fail. + +## The rule this suggests + +The audit's own exemplar turned out to be broken, which is worth recording. Two sweeps +held up `devices_broker.dropped` as the model the others should be rewritten against. +It is not: `register()` never increments it, and `kernel.zig:203` reads it once at boot +*before* `seedDisplay` and long before any ring-3 driver exists. It can only report +firmware-discovery losses — precisely the opposite of the runtime case under audit. + +`parameters.maximum_cpus` is the one bound that meets the standard: hardware-determined +and bounded, but it states what happens to the surplus (parked) and how you find out +(`platform.cpusDropped()` → a WARNING at `kernel.zig:281`). Verified. + +So: + +1. **A fixed bound on a quantity the hardware or an external file decides is a defect, + not a tunable.** It does not need a better number; it needs to not be fixed. This + covers 139 of the 235. +2. **A bound that must exist states three things**: what it protects against, what the + system does when it is reached, and how an operator finds out. One of 235 does. +3. **Refusal must be attributable.** `process.zig:944` collapses five distinct + `RegisterError` variants into a bare `-1`; `pci-bus` can only log "register refused" + with no reason. Half this audit's difficulty was that the machine could not say + which ceiling it hit. +4. **Two constants that must agree may not agree by comment.** `maximum_domains = 64` + and `maximum_devices = 64` are coupled by prose, and the coupling fails open. + +## Appendix: the inventory + +Sorted by severity, then file. `Decided by` is the audit's judgment of who chooses the +quantity — the machine, an external file or peer, or us. +| File:line | Bound | Value | Decided by | At the limit | Observable | +|---|---|---|---|---|---| +| `boot/efi.zig:318` | (4 * gib) | 4 GiB | hardware | No check exists. buildBootstrapTables maps [0, 4 GiB) as identity + physmap 2 MiB leaves and stops; the only carve-out is the framebuffer window at ef… | Nothing on this path. No log, no error, no panic. The last thing printed is the unconditional con_out line at … | +| `library/device/model/device-abi.zig:81` | maximum_device_resources | 8 | hardware | Silent drop, but at a different site than claimed. The live enforcement is system/kernel/device-model.zig:114 `if (self.resource_count >= maximum_reso… | None on any drop path, and the code says otherwise: device-model.zig:111-112 claims "Silently drops beyond `ma… | +| `system/drivers/pci-bus/pci-bus.zig:228` | device.register refusal path (kernel maximum_chi… | 16 children per parent / 64 … | hardware | system/kernel/devices-broker.zig:305 `if (childCount(parent_id) >= maximum_children_per_parent) return error.TooManyChildren;` and :306 `if (count >= … | One line per lost function: `std.log.info("register refused for {d}:{d}.{d}", .{ bus, dev, function })` (pci-b… | +| `system/drivers/usb-xhci-bus/usb-xhci-library.zig:227` | max_interfaces | 4 | external-data | CORRECTION to this claim's description: it is not only a silent drop. `parseConfiguration` (1326-1342) has no else on `else if (device.interface_count… | None. usb-xhci-bus.zig:368/425 log `{d} interface(s)` with the truncated count; nothing says any were dropped … | +| `system/drivers/usb-xhci-bus/usb-xhci-library.zig:333` | max_devices | 8 | hardware | Two different behaviours, exactly as claimed. Root port, `setupDevice` (899-902): `const device = self.allocateDevice() orelse { std.log.info("port {d… | Root ports: one log line. Hub-attached: nothing. I confirmed the leak is real — the only Disable Slot in the f… | +| `system/drivers/usb-xhci-bus/usb-xhci-library.zig:1304` | blob: [512]u8 | 512 | external-data | Silent truncation of device-supplied data. Lines 1304-1307: `var blob: [512]u8 = undefined; const length = @min(configuration.total_length, blob.len);… | None. `configuration.total_length` is read at 1299 and never compared to `blob.len`, never logged. The `{d} in… | +| `system/drivers/usb-xhci-bus/usb-xhci-library.zig:1305` | blob (inline [512]u8, declared line 1304) | 512 bytes | external-data | Silent truncation of externally-supplied data. The @min clamps the GET_DESCRIPTOR request to 512; parseConfiguration then walks it and stops dead at l… | Nothing. No log line, no counter, no comparison of configuration.total_length against blob.len. The symptom is… | +| `system/drivers/usb-xhci-bus/usb-xhci-library.zig:1331` | max_interfaces (= 4), used as `else if (device.i… | 4 | external-data | CORRUPTING, not merely dropping. When interface_count == 4, neither branch of the if/else-if runs, so `current` is NOT cleared — it still points at in… | Nothing. No log, no counter, no error. The device enumerates, `port N device: ... 4 interface(s)` is logged (a… | +| `system/kernel/acpi.zig:445` | device_model.maximum_resources … | 6 AML blocks, inside an 8-re… | hardware | Silent double loss. The loop stops at 6 blocks, so SSDTs 7+ are never published. Then acpi.zig:454-463 adds the io_port grant (7th), the SCI irq (8th)… | Nothing at all. No counter, no log, no error. The ring-3 acpi service simply finds an acpi-tables node without… | +| `system/kernel/acpi.zig:445` | device_model.maximum_resources - 2 (inline expre… | 6 AML blocks, out of 8 total… | hardware | Silent drop, then a silent cascade. Resources are appended in order: up to 6 AML memory resources (445-448), io_port (454), the SCI irq (459), the bro… | Nothing at the kernel. The only downstream trace is the acpi service writing "acpi: no FADT on the node — powe… | +| `system/kernel/acpi.zig:627` | below | [64]Range | hardware | Silent skip: `if (region.base >= (1 << 32) or below_count == below.len) continue;`. Every region past the 64th is treated as 'not described', i.e. as … | None directly, though kernel.zig:133 does print `" regions : {d} - entries in the firmware memory map"`, w… | +| `system/kernel/acpi.zig:627` | below (inline [64]Range) | 64 entries | hardware | Silent drop, and — worse than a drop — a corrupted result. A dropped region is not merely missing from the list; it disappears from the "described spa… | Nothing at all. No counter, no log line. The kernel prints the device tree including the bogus aperture, with … | +| `system/kernel/acpi.zig:651` | gaps | [3]Range | hardware | Silent replacement of the smallest kept gap (acpi.zig:658-664): a fourth (or fifth) MMIO hole is simply forgotten. A PCI function whose BAR lands in a… | Nothing. No log, no counter, and the resulting failure is indistinguishable at the caller from a table-full re… | +| `system/kernel/device-model.zig:66` | maximum_resources | 8 | external-data | `addResource` (line 113-118) returns false without recording. I grepped every call site: system/kernel/acpi.zig lines 448, 454, 459, 460, 463, 552, 55… | Nothing at the drop. Downstream there are two ring-3 lines that name the symptom and misattribute the cause: s… | +| `system/kernel/devices-broker.zig:33` | maximum_children_per_parent | 16 | hardware | Confirmed at devices-broker.zig:305: `if (childCount(parent_id) >= maximum_children_per_parent) return error.TooManyChildren;`, where `childCount` (28… | One reasonless line per lost device, and nothing from the kernel. Confirmed at pci-bus.zig:228-231: `const reg… | +| `system/services/acpi/acpi.zig:129` | blocks | [8][]const u8 | external-data | The `blocks: [8][]const u8` array at line 129 and its `if (block_count == blocks.len) continue;` at line 152 are dead — they can never fire, because t… | Only an indirect count that cannot reveal the loss: `std.log.info("parsed {d} AML blob(s), {d} namespace devic… | +| `system/services/device-manager/device-manager.zig:174` | maximum_children | 64 | hardware | Silent drop with a misleading log. `addChild` (180-194) returns false when no free slot exists. `onChildAdded` (442-452): `if (!addChild(...)) status … | Effectively none, and actively misleading. The -ENOSPC goes back to the bus driver, which discards it — verifi… | +| `boot/efi.zig:108` | (handles[0]) | 1 | hardware | Silent selection of GOP handle 0, and the two halves genuinely disagree — nativeResolution (efi.zig:167-176) iterates `for (handles) \|h\|` over every… | Nothing. No log names the handle count, the chosen adapter, or the resolved mode; the only con_out writes in t… | +| `library/device/driver/driver.zig:149` | | @min(total, buffer.len) | hardware | Silently searches a prefix: `const total = enumerate(buffer); const n = @min(total, buffer.len); for (@as([]DeviceDescriptor, buffer[0..n])) \|d\| { …… | Nothing. The discarded `total` is the only evidence that the buffer was too small, and it is thrown away on th… | +| `library/kernel/file-system.zig:227` | Entry.name_buffer | [64]u8 | external-data | Silent truncation on both readdir paths, exactly as claimed. Backend path, Directory.next line 264-266: `const nlen = @min(@min(@as(usize, header.name… | Nothing. No log, no flag, no short-count anywhere in library/kernel/file-system.zig. Entry.name() returns a pl… | +| `system/drivers/virtio-gpu/virtio-gpu.zig:49` | max_width / max_height (line 50) | 800 x 600 | hardware | Two distinct behaviours, and the claim conflates their visibility. (a) onSetMode, line 533-540: `if (w == 0 or h == 0 or w > max_width or h > max_heig… | Partial and misleading, as claimed. Confirmed the two adjacent lines: `std.log.info("EDID preferred mode {d}x{… | +| `system/kernel/architecture/x86_64/cpu.zig:489` | irq_vector_count | 14 | hardware | irq.zig:138 allocVector() scans base..base+count for a vector free of both an MSI binding and a GSI binding and returns null when none is; bind (irq.z… | No kernel log line. Driver-side, both messages verified: usb-xhci-bus.zig:279 `std.log.info("msi_bind unavaila… | +| `system/kernel/architecture/x86_64/idt.zig:17` | gate_count | 48 | our-design | Nothing is enforced at this line. `pub fn init()` at line 136 does `inline for (0..gate_count) \|vector\| { const stub = @extern(*const anyopaque, .{ … | None at runtime. The bound only ever becomes visible as the vector-exhaustion path in irq.zig, which each driv… | +| `system/kernel/architecture/x86_64/ioapic.zig:21` | base (module-level singleton) | 1 I/O APIC | hardware | Confirmed exactly. system/kernel/acpi.zig:547-556 adds EVERY MADT I/O APIC record to the device tree as ioapic0, ioapic1, ...; kernel.zig:545-548 then… | None. No log counts the discarded units, unlike the neighbouring drop counters that ARE logged (devices_broker… | +| `system/kernel/architecture/x86_64/iommu-amd.zig:56` | ring_entries (event log) | 256 | hardware | Confirmed. faultDrain (lines 180-198) reads EventHead/EventTail, walks the ring wrapping at `if (head >= ring_entries * 16) head = 0;`, and writes the… | None. I grepped the file: `const reg_status = 0x2020;` appears at line 32 and at NO other line — it is declare… | +| `system/kernel/architecture/x86_64/iommu-intel.zig:102` | 16 * 1024 | 16384 (bytes of VT-d registe… | hardware | No check. `faultDrain` computes `frcd_base = fro*16` with fro up to 0x3FF (16368 bytes) and `nfr = ((cap >> 40) & 0xFF) + 1` up to 256 registers of 16… | A kernel-mode #PF: the fault handler reports the vector/CR2 and halts the core. Loud, but reported as a page f… | +| `system/kernel/architecture/x86_64/iommu.zig:17` | Discovery.register_base (single unit) | 1 IOMMU unit | hardware | Confirmed. `pub const Discovery = struct { register_base: u64, amd: bool };` at iommu.zig:16-19 carries exactly one base. Intel (acpi.zig:826-852): th… | Asymmetric, exactly as claimed. Intel warns: system/kernel/iommu.zig:395-396 `if (info.iommu_extra_units > 0) … | +| `system/kernel/devices-broker.zig:25` | maximum_devices | 64 | hardware | Three inconsistent behaviours, all confirmed. Boot discovery: devices-broker.zig:111-114 `if (count >= maximum_devices) { dropped += 1; return device_… | Partial and aimed at the wrong path, exactly as claimed. kernel.zig:203-206 prints `"/system/kernel: WARNING {… | +| `system/kernel/iommu.zig:43` | maximum_domains | 64 | hardware | Two behaviours. Domain exhaustion is safe: `domainCreate` returns null → `confineDevice` false → process.zig:418-420 rolls the claim back and returns … | The rollback path is visible as a failed `device_claim`. The fail-open path is completely silent — no log, no … | +| `system/kernel/irq.zig:43` | maximum_gsi | 24 | hardware | Refusal at three places, all becoming a bare -1. irq.zig:158 `if (gsi >= maximum_gsi or !architecture.irqOwnsGsi(gsi)) return error.BadGsi;` -> proces… | Nothing in the kernel, but every caller logs: ps2-bus.zig:246 and :265, acpi.zig:315 "acpi: SCI irq_bind faile… | +| `system/kernel/irq.zig:139` | allocVector over architecture.irq_vector_count | 14 (vectors 33..46, from arc… | hardware | `fn allocVector() ?u8` (line 139) scans v in [irq_vector_base, irq_vector_base + irq_vector_count) = [33, 47) — 14 vectors — and returns null when all… | Better than claimed at the driver level, absent at the kernel level. The kernel logs nothing and has no counte… | +| `system/kernel/process.zig:546` | maximum_dma_regions | 256 | hardware | Confirmed, and the safety-invariant break is real. process.zig:549-556 `fn dmaRegistryAdd(object: ?*ipc.DmaRegionObject, physical: u64, len: u64, owne… | None whatsoever — confirmed. `dma_alloc` returns success, no log, no counter, no failed syscall. The symptom i… | +| `system/kernel/scheduler.zig:546` | maximum_tasks (via freeSlot in secondaryMain) | 48 (parameters.maximum_tasks… | hardware | Confirmed: `const t = freeSlot() orelse @panic("sched: task table full (AP idle task)");` at scheduler.zig:546, reached from `secondaryMain` with the … | A named panic on console/serial — loud, but the machine is dead and the message points at the task table rathe… | +| `system/kernel/vfs.zig:58` | maximum_mounts | 8 | our-design | FALSE SUCCESS plus a reference leak, confirmed end to end. installMount, vfs.zig:182: `const m = slot orelse return;` — it gives up silently when no s… | Worse than nothing: the syscall reports success, so the mounting service logs its own success — system/service… | +| `system/kernel/vfs.zig:86` | maximum_directories | 8 | external-data | Silent skip inside the boot walk: `if (directoryIndex(parent) == null and directory_count < maximum_directories)` (vfs.zig:141). An unregistered direc… | Nothing. No log, no counter, no `dropped` variable of the kind acpi.zig:542 and the device manager both mainta… | +| `system/kernel/vfs.zig:141` | maximum_directories (declared line 86) | 8 | external-data | Silent drop with a mount-level knock-on, confirmed. The ninth distinct ancestor is skipped by the guard at vfs.zig:141; it then fails to resolve (reso… | Nothing — no counter, no log line, no `dropped` variable of the sort acpi.zig:542 (cpu_information.dropped, pr… | +| `system/kernel/vfs.zig:182` | maximum_mounts (declared line 58) | 8 | our-design | False success, confirmed: installMount gives up at vfs.zig:182 (`const m = slot orelse return;`) and mountBackend returns true regardless at vfs.zig:3… | Nothing in the kernel. The mounting service logs its own success — fat.zig:173/:178/:183 all `std.log.info("mo… | +| `system/services/acpi/acpi.zig:542` | buffer (readHid) / Registered.hid / Notice.hid | [8]u8 | external-data | Both failures confirmed. (a) readHid (lines 540-566) accepts only an integer _HID it can EISA-decode: a method result must be `.integer`, and a static… | None for (a): a skipped device produces no line at all. For (b) the log at line 408, `std.log.info("power: not… | +| `system/services/acpi/acpi.zig:648` | descriptor.resources.len | 8 (maximum_device_resources,… | external-data | Silent drop of the surplus. `fn addResource(descriptor: *device.DeviceDescriptor, kind: device.ResourceKind, start: u64, len: u64) void { if (descript… | Nothing at the drop. The only echo is the per-device summary at line 241, `std.log.info("device {d} bus=acpi … | +| `system/services/device-manager/device-manager.zig:149` | maximum_drivers | 16 | hardware | `addDriver` (240-254) walks for a free slot and, finding none, falls through the loop to `std.log.info("driver table full; cannot supervise {s}", .{na… | One log line only (quoted above). No counter, no status to any caller, and it lands amid the stream of `child … | +| `system/services/device-manager/device-manager.zig:358` | | 64 | hardware | Silent truncation. Line 362-363: `const total = device.enumerate(buffer); const count = @min(total, buffer.len);` — the difference is computed and thr… | Nothing. No log mentions `total`. The only related line, `"/system/services/device-manager: no matchable devic… | +| `system/services/fat/fat.zig:73` | open_nodes | 32 | our-design | An error is returned, but a maximally confusing one: `const index = allocOpen() orelse return refused;` (line 242), where `refused` is `-envelope.ENOE… | Nothing on the server side — no log at all in allocOpen or onOpen. The client sees ENOENT and will report "fil… | +| `system/services/fat/fat.zig:252` | | u32 (via @intCast of a u64 p… | external-data | Panic. `vfs_protocol.Read.offset` is a `u64` (library/protocol/vfs/vfs-protocol.zig:82-86) and `engine.readFile` takes `offset: u32`; the bridge is a … | A process fault and whatever the supervisor logs about the death; nothing identifies the offending request or … | +| `boot/efi.zig:371` | maximum_bundled | 64 | external-data | Two different behaviours for one constant, confirmed. loadByManifest (efi.zig:507): `if (count.* == maximum_bundled) return;` — silent success with th… | Manifest path: nothing. Walk path: the unconditional con_out line at efi.zig:71-75, `log("EFI: no /system bina… | +| `boot/efi.zig:375` | maximum_tree_depth | 3 | our-design | `if (depth == maximum_tree_depth) continue;` (efi.zig:580) — the subdirectory is silently skipped and nothing beneath it is bundled. No error, no coun… | Nothing. The boot proceeds; the missing binary surfaces much later as an init spawn failure or a device-manage… | +| `boot/efi.zig:507` | maximum_bundled (declared line 371) | 64 | external-data | As described and verified: efi.zig:507 `if (count.* == maximum_bundled) return;` in loadByManifest is a silent stop keeping the first 64; efi.zig:590 … | Manifest path: nothing. Enumeration path: efi.zig:71-75 prints 'EFI: no /system binaries (TooManyBinaries) - b… | +| `boot/efi.zig:582` | (child_prefix bufPrint) | 64 bytes (initial_ramdisk.ma… | external-data | `const child = try std.fmt.bufPrint(&child_prefix, "{s}/{s}", .{ prefix, name });` into a [initial_ramdisk.maximum_name]u8 = [64]u8. error.NoSpaceLeft… | efi.zig:71-75 prints 'EFI: no /system binaries (NoSpaceLeft) - booting without user space' on con_out. Visible… | +| `boot/efi.zig:633` | ehdr.e_phnum / e_phoff / e_phentsize (no bound a… | unbounded — read straight fr… | external-data | Unchecked out-of-bounds read. efi.zig:632-636 forms `image.ptr + ehdr.e_phoff + i * ehdr.e_phentsize` as raw pointer arithmetic and @ptrCasts it — no … | Nothing until it faults or panics, at which point the machine is still in the firmware with no kernel. | +| `library/device/acpi/aml/interpreter.zig:158` | notify_queue | [16]NotifyEvent | external-data | Silent drop of the 17th and later Notify. `notify()` (line 571-576): `if (target) \|node\| { if (self.notify_count < self.notify_queue.len) { ...appen… | None. No log in interpreter.zig at all (grepped: zero std.log/log.print/logging.write calls in the file). `tak… | +| `library/device/acpi/aml/interpreter.zig:523` | | 100_000 | our-design | Silently exits the loop and continues executing the method as if the loop had terminated normally: `while (guard < 100_000) : (guard += 1) { … }` then… | Nothing. No log, no `error.Unsupported`, no distinguishable result — the caller receives a normal-looking valu… | +| `library/device/acpi/aml/parser.zig:24` | maximum_segments (and its disagreeing twin, libr… | 64 in the parser, 16 in the … | external-data | Both files silently drop the surplus segment after consuming its 4 bytes. Parser: appendSegment, parser.zig:137-143. Interpreter: `fn segment(self: *C… | None on either path. Neither file logs; acpi.zig discards ParseResult.consumed/total; interpreter errors surfa… | +| `library/device/usb/usb.zig:167` | | 100 attempts × 20 ms = 2 s | our-design | Gives up and returns null: `const bus = while (attempts < 100) : (attempts += 1) { if (channel.openEndpoint("usb-transfer")) \|handle\| break handle; … | Nothing here. The driver's own bail-out is what an operator sees, with no indication that it was a timeout rat… | +| `library/kernel/channel.zig:124` | | @min(available, into.len) | our-design | Silent truncation to the caller's buffer: `const available = @min(answer.len - envelope.prefix_size, @as(usize, status.len)); const taken = @min(avail… | Weak: `Response.status.len` holds the true length the provider sent, so a caller *could* compare it against `p… | +| `library/kernel/file-system.zig:85` | | [224]u8 (twice: lines 77 and… | our-design | Errors invisibly, but the enforcing code is in the KERNEL, not where the claim points. system/kernel/process.zig:1898 rejects the input outright: `if … | None. A too-long path and a nonexistent file are both `null` from open(); no log on either side — the kernel's… | +| `library/kernel/logging.zig:82` | | [256]u8 | our-design | Truncation, marked with a `~` — but the truncated slice is the *whole* buffer, not the written prefix: `const line = std.fmt.bufPrint(&buffer, prefix … | The trailing `~` marks the line as truncated, which is good; the garbage tail is not marked at all. | +| `library/kernel/process.zig:211` | | [32]ProcessDescriptor | our-design | Silent false negative: `var table: [32]ProcessDescriptor = undefined; const total = processes(&table); for (table[0..@min(total, table.len)]) \|descri… | Nothing. The function returns a plain bool; the discarded `total` is the only evidence and it is thrown away b… | +| `library/protocol/device-manager/device-manager-protocol.zig:161` | entries_per_reply | (envelope.packet_maximum - e… | hardware | Silent cut with a success status. `onEnumerate` (device-manager.zig:524-536): `for (&children) \|*child\| { if (!child.used) continue; if (written + e… | None, and structurally impossible for a client to detect: the protocol's own comment (149-152) says the count … | +| `library/protocol/envelope/envelope.zig:133` | packet_maximum | 256 | our-design | Two behaviours, exactly as claimed. Compile time: `Define` rejects an oversized fixed part with a message naming protocol, verb, size, prefix and floo… | Compile-time half is excellent. Run-time half is silent everywhere I traced it: library/device/block/block.zig… | +| `library/protocol/envelope/envelope.zig:134` | post_maximum | 64 | our-design | Compile error for an event payload that does not fit (`Define`, envelope.zig:365-370, message names the event and the push floor). At run time `encode… | Compile-time: named and precise. Run-time: nothing. The `orelse return` at service.zig:239 is the whole handli… | +| `library/protocol/usb-transfer/usb-transfer-protocol.zig:58` | max_report_data | 40 | hardware | Two stacked silent truncations. First in the engine: usb-xhci-library.zig:1569 `const n = @min(length, report.data.len);` against a `data: [64]u8` (li… | None. No log at either truncation site; the surrounding comment (usb-xhci-bus.zig:766-768) states the truncati… | +| `library/protocol/usb-transfer/usb-transfer-protocol.zig:62` | max_reported_endpoints | 4 | hardware | Silent drop, and the primary site is upstream of the one claimed. The bus driver discards surplus endpoint descriptors while parsing the configuration… | None. No log at the parse-time drop, none at the client cap; a driver's `findEndpoint` just returns null for a… | +| `system/boot-handoff.zig:143` | kernel_segments | 8 | our-design | Silent drop with a delayed fatal consequence, as described. boot/efi.zig:659-668 records only `if (n < boot_information.kernel_segments.len)` with no … | The loader says nothing. The kernel prints the count it received — kernel.zig:159 `log.print(" kernel segs: {… | +| `system/drivers/pci-bus/pci-bus.zig:81` | alloc(device.DeviceDescriptor, … | 64 | our-design | Not reachable today, so nothing happens. If the kernel table grew past the buffer: `device.enumerate(buffer)` returns the TOTAL, the search clamps wit… | One misleading line, `std.log.info("device {d} not in the device tree", .{bridge_id})` — it names a missing de… | +| `system/drivers/pci-bus/pci-bus.zig:81` | 64 (bare inline literal — no constant, no link t… | 64 | our-design | Same site as the earlier claim on line 81 — this is a duplicate. Unreachable today (`total` is bounded by the kernel's own maximum_devices = 64). If i… | `std.log.info("device {d} not in the device tree", .{bridge_id})` — misdescribes the cause and never prints `t… | +| `system/drivers/usb-hid/hid-report.zig:20` | KeyboardReport.keys: [6]u8, and max_transitions … | 6 concurrent keys; 14 transi… | hardware | THE CLAIM IS BACKWARDS. `max_transitions = 8 + 6` is derived from a wrong worst case, and `Transitions.add`'s guard (`if (self.count < self.items.len)… | None. `add` drops without logging and `Transitions` carries no overflow flag; `slice()` just returns the first… | +| `system/drivers/usb-storage/usb-storage.zig:61` | .lun = 0 in the CommandBlockWra… | 1 LUN (LUN 0 only) | hardware | There is no limit check — LUNs 1..15 simply do not exist to this driver. Every CBW is built with `.lun = 0` (usb-storage.zig:57-63), and I confirmed t… | Nothing. Nothing reads bMaxLUN, so nothing can report that a device has more than one logical unit. | +| `system/drivers/usb-storage/usb-storage.zig:121` | while (tries < 10) with time.sl… | 10 attempts = ~500 ms | hardware | Falls out of the loop and proceeds regardless (usb-storage.zig:120-127). INQUIRY's result is discarded (`_ = transact(...)`, line 129), then READ CAPA… | Present but misattributed, exactly as claimed: `_ = logging.write("/system/drivers/usb-storage: READ CAPACITY … | +| `system/drivers/usb-storage/usb-storage.zig:172` | @intCast(request.lba) -> u32, @… | LBA < 2^32, count < 2^16 (RE… | external-data | Bare, unchecked @intCast on both fields — verified there is no range check anywhere between the wire struct and the CDB. scsi.read10/write10 take (lba… | In a safe build, driver death with a fault exit reason and a device-manager restart with backoff — attributed … | +| `system/drivers/usb-storage/usb-storage.zig:172` | @intCast(request.lba) / @intCast(request.count) … | u64 -> u32 LBA, u32 -> u16 b… | external-data | Unchecked narrowing of two peer-chosen wire fields. library/protocol/block/block-protocol.zig:30-35 declares `Transfer { lba: u64, count: u32, physica… | Nothing in ReleaseFast; in a safe build a driver crash with a fault exit reason and a supervised restart, attr… | +| `system/drivers/usb-xhci-bus/usb-xhci-bus.zig:80` | opens (= [_]Open{.{}} ** 16) | 16 | our-design | Confirmed, including the success-on-failure shape. recordOpen (line 87) walks `opens` for a matching token, then for a free slot, and falls through to… | Nothing on the bus side — no log when the table fills, and the harness closes the unclaimed report endpoint si… | +| `system/drivers/usb-xhci-bus/usb-xhci-bus.zig:728` | prev_connected (declared line 326 as [64]bool), … | 64 | hardware | Confirmed off-by-one. Line 728: `while (port <= engine.max_ports and port <= prev_connected.len) : (port += 1)` with `var port: u32 = 1` — at port == … | The OOB half surfaces as a driver fault and a supervised restart into the same panic — loud but attributed to … | +| `system/drivers/usb-xhci-bus/usb-xhci-bus.zig:728` | prev_connected (inline [64]bool, declared line 3… | 64 | hardware | Same site as the earlier line-728 claim — this is a duplicate. Two failures, both confirmed. (1) Off-by-one out-of-bounds: `while (port <= engine.max_… | The panic half is loud (process fault + supervised restart loop) but misattributed. The silent-drop half has n… | +| `system/drivers/usb-xhci-bus/usb-xhci-library.zig:228` | max_endpoints_per_interface (and max_configured_… | 4 per interface, 16 configur… | hardware | Parsing: `if (interface.endpoint_count < max_endpoints_per_interface)` with no else — the 5th endpoint descriptor is silently discarded, so a class dr… | The endpoint-drop during parsing is entirely silent. The ring-exhaustion path reaches the class driver as a ge… | +| `system/drivers/usb-xhci-bus/usb-xhci-library.zig:228` | max_endpoints_per_interface | 4 | external-data | Silent drop, no else (1345-1354). An endpoint never recorded cannot be found by `endpointForAddress` (1377-1382), so a class driver's subscribe or bul… | None at the drop. The secondary consequence the claim names is real and I verified it: usb-xhci-bus.zig:612 se… | +| `system/drivers/usb-xhci-bus/usb-xhci-library.zig:328` | Report.data: [64]u8 (and usb_tr… | 64 in the driver, then 40 ov… | hardware | Truncated twice, silently: `const n = @min(length, report.data.len);` in enqueueReport, then `const n = @min(report.length, usb_transfer_protocol.max_… | Nothing. Both truncations are @min with no branch and no log. | +| `system/drivers/usb-xhci-bus/usb-xhci-library.zig:334` | max_subscriptions | 8 | hardware | Two paths, as claimed. Class driver: `subscribeInterrupt` (1482) `const subscription = self.allocateSubscription() orelse return false;` -> usb-xhci-b… | Class drivers log it: usb-hid/keyboard.zig:94 and mouse.zig:57 both write `interrupt subscribe failed`. The hu… | +| `system/drivers/usb-xhci-bus/usb-xhci-library.zig:335` | report_queue_capacity | 16 | hardware | `enqueueReport` (1561-1562): `if (self.report_count >= self.report_queue.len) return; // full: drop the newest` — the report is discarded and the subs… | None. The comment in the source is the only trace; no log, no counter. The symptom is dropped keystrokes or a … | +| `system/drivers/usb-xhci-bus/usb-xhci-library.zig:968` | hub_change_mask: u32 (declared line 298) and the… | 31 usable downstream ports | hardware | Ports 32..255 are never seeded and never serviced: the seed clamps to 0xFFFF_FFFE for hub_ports >= 31, and takeHubChange picks bits via @ctz on a u32,… | Nothing. No log line mentions bNbrPorts exceeding what the mask holds; the 'N downstream ports powered' line p… | +| `system/drivers/usb-xhci-bus/usb-xhci-library.zig:1456` | `.status = length` on the bulk … | 65535 (the xHCI TRB Transfer… | hardware | Unchecked write of a u32 into a 17-bit field. `bulkTransfer` (1452-1458) pushes `.status = length` with no mask and no split across TRBs; the status d… | Nothing on the way in — no check, no log. It surfaces later as a short/failed transfer or a lost completion at… | +| `system/drivers/virtio-gpu/virtio-gpu.zig:49` | max_width / max_height | 800 x 600 | hardware | set_mode errors visibly (line 536 refuses anything larger). But the EDID path is the silent one: readEdid (lines 413-442) parses the panel's preferred… | One std.log.info line naming the preferred mode, in the driver's log file, with nothing saying it was ignored. | +| `system/drivers/virtio-gpu/virtio-gpu.zig:208` | descriptors: [64]device.DeviceD… | 64 | hardware | NOT reachable today, contrary to the claim's framing. The kernel's own table is the binding wall: system/kernel/devices-broker.zig:25 `const maximum_d… | `device {d} not in the device tree` (line 213), which today only means a real absence. The kernel-side drop th… | +| `system/initial-ramdisk.zig:26` | maximum_name | 64 bytes | our-design | Four different behaviours, verified individually. (a) Capsule build: tools/pack-system-image.py:35-37 writes 'pack-system-image: path too long (>63)' … | (a) clear build error. (b)(c)(d) nothing: boot/efi.zig has exactly two output helpers (lines 773 and 795) and … | +| `system/kernel/acpi.zig:94` | overrides | [16]IsoEntry | hardware | Silent drop: `if (platform_information.override_count < platform_information.overrides.len) { ... }` with no else branch and no counter — unlike the C… | Nothing. A lost override means a legacy line is routed at the wrong GSI or the wrong polarity, which shows up … | +| `system/kernel/acpi.zig:94` | PlatformInformation.overrides | [16]IsoEntry | hardware | Silent drop, at three separate 16s that must agree by hand: system/kernel/acpi.zig:558 (no else branch), system/kernel/kernel.zig:248 (`var isos: [16]… | Nothing at any of the three clamps. No counter of the cpusDropped kind, no log line. | +| `system/kernel/acpi.zig:186` | aml_block_physical / aml_block_len | [32]u64 / [32]usize | hardware | Confirmed at acpi.zig:190-197: `fn addAmlBlock(sdt_physical: u64) void { if (aml_block_count >= aml_block_physical.len or sdt_physical == 0) return; .… | None — confirmed, no log and no dropped counter, unlike `cpu_information.dropped` (acpi.zig:539) and `rmrr_ski… | +| `system/kernel/architecture/x86_64/apic.zig:320` | 100_000_000 (calibratePit guard… | 100000000 | our-design | Confirmed: the loop falls out and calibration proceeds on a window that never happened. `ticks_per_ms = elapsed / calib_ms` and `tsc_hz = (tsc_end -% … | No failure line, but not quite "none": kernel.zig:322 prints `timer online ({d} Hz tick; timer clock {d} MHz, … | +| `system/kernel/architecture/x86_64/apic.zig:629` | localId (return type u8) / apic_id << 24 | 8-bit APIC id (0..255) | hardware | The claimed truncations are not reachable; the real defect is an omission. `sendInit`/`sendStartup` (lines 174-184) take a `u32` but every caller pass… | None for the dropped type-9 records. `cpu_information.dropped` (acpi.zig:542) counts only overflow past the 12… | +| `system/kernel/architecture/x86_64/iommu-amd.zig:56` | ring_entries (command buffer) | 256 | our-design | Confirmed. submitCommand (lines 221-228) writes the two qwords, advances `command_tail += 16`, wraps modulo the ring, and writes the tail register — w… | None — no head check, no full-ring detection, no log. | +| `system/kernel/architecture/x86_64/iommu-amd.zig:74` | fault_log_budget | 32 | hardware | logFault (lines 200-211) opens with `if (fault_log_budget == 0) return;` — the record's bdf and address are discarded outright, with no suppressed-fau… | One "(further faults suppressed)" line (line 210) and then permanent silence for the detail. Correcting the cl… | +| `system/kernel/architecture/x86_64/iommu-amd.zig:246` | 100_000 | 100000 | our-design | Confirmed verbatim at lines 243-253: the spin loop polls the completion frame for the 0xC0FFEE sentinel, and at `if (spins > 100_000)` warns once behi… | One line for the whole boot: "/system/kernel: AMD-Vi COMPLETION_WAIT store not observed — proceeding (QEMU pro… | +| `system/kernel/architecture/x86_64/iommu-intel.zig:77` | fault_log_budget | 32 | hardware | `logFault` (lines 252-270) takes the else branch and does `faults_suppressed += 1`. The budget is only ever decremented (line 255) — nothing resets it… | One line at the transition: 'DANOS-IOMMU-FAULT: (further faults suppressed)' (line 266). After that a device D… | +| `system/kernel/architecture/x86_64/iommu-intel.zig:294` | 10_000_000 (spin64) | 10000000 | our-design | `if (spins > 10_000_000) return;` — the function returns as if the invalidation completed. Callers (`invalidateDomain`, `globalInvalidate`, `detach`) … | Nothing at all — a bare `return` with no message, unlike `spinStatus` which prints "WARNING VT-d status bit ne… | +| `system/kernel/architecture/x86_64/paging.zig:206` | 0xFEE00000 | 0xFEE00000 (one 4 KiB page) | hardware | Confirmed and worse than "inconsistent": the two halves disagree with no reconciliation. paging.zig:206 maps exactly one hardcoded page `mapPage(pml4,… | A boot-time kernel fault reported as a fault address, never as "LAPIC relocated". The parsed truth is thrown a… | +| `system/kernel/architecture/x86_64/serial.zig:112` | 5_000 (writeByte guard) | 5000 | hardware | The loop simply falls through and `setRegister(0, c)` runs anyway (serial.zig:111-113) — the byte is written into a transmit-holding register that may… | Nothing. The corruption appears in the serial log itself, which is the channel you would use to notice it. | +| `system/kernel/heap.zig:28` | heap_maximum | 64 * 1024 * 1024 | our-design | Confirmed mixed. `grow` returns false at heap.zig:63 `if (start + bytes > heap_base + heap_maximum) return false;` -> `rawAlloc` null -> std.mem.Alloc… | For the graceful callers, a bare -1 to user space with no reason. For scheduler.zig:704, a panic naming the st… | +| `system/kernel/heap.zig:161` | alignment ceiling | 16 | our-design | `if (alignment.toByteUnits() > 16) return null;` confirmed at heap.zig:161 inside `allocImpl`. Returning null from the vtable alloc is how the allocat… | None. No log, no distinct error; the caller sees error.OutOfMemory and investigates memory pressure. | +| `system/kernel/ipc-synchronous.zig:40` | MESSAGE_MAXIMUM | 256 | our-design | Two different behaviours, both confirmed. The cap itself errors visibly: `if (message_len > MESSAGE_MAXIMUM or reply_cap > MESSAGE_MAXIMUM) return -E2… | -E2BIG for the cap. Nothing for the truncation — no counter, no flag bit, no errno. | +| `system/kernel/ipc-synchronous.zig:75` | post_capacity | 16 | external-data | Deliberate silent drop of the OLDEST message: sendLocked, ipc-synchronous.zig:568 `if (endpoint.post_tail -% endpoint.post_head >= post_capacity) endp… | None, and I checked the struct: PostSlot (ipc-synchronous.zig:79-83) carries only length and sender_id — no se… | +| `system/kernel/ipc-synchronous.zig:513` | @min(caller.ipc_send_len, receive_cap) (no const… | whatever the receiver's buff… | external-data | Silent truncation. replyWait, ipc-synchronous.zig:512-513: `if (dequeueSender(endpoint)) \|caller\| { const n = @min(caller.ipc_send_len, receive_cap)… | Nothing — no counter, no errno, no flag bit anywhere on this path. | +| `system/kernel/log.zig:202` | buffer (log.print) | [256]u8 | our-design | Confirmed: the entire line is discarded, not truncated. Lines 201-204 verbatim: `pub fn print(comptime fmt: []const u8, args: anytype) void { var buff… | Nothing. The message never appears and its absence is indistinguishable from the code path not having run. Not… | +| `system/kernel/pmm.zig:94` | (implicit: bitmap must land bel… | 4 GiB, stated only in prose | hardware | Two distinct failures. If NO usable region is large enough: `@panic("pmm: no region large enough for the frame bitmap")` — loud and fatal. If the chos… | The panic is legible. The above-4-GiB case is a triple fault or an early #PF with essentially no diagnostics. | +| `system/kernel/process.zig:98` | maximum_shared_memory_pages | 8192 (32 MiB) | hardware | Confirmed at process.zig:790: `if (pages == 0 or pages > maximum_shared_memory_pages) return fail(state);` in `systemSharedMemoryCreate`, giving a cle… | No kernel log; -1 only — confirmed. And the ambiguity the claim notes is real: `pmm.allocContiguous` failing t… | +| `system/kernel/process.zig:104` | maximum_mmap_pages | 8192 (32 MiB) | our-design | Confirmed at process.zig:2050: `if (pages == 0 or pages > maximum_mmap_pages) return fail(state);` in `systemMmap` — a clean -1. The independent secon… | -1 only, no log — confirmed. Indistinguishable from arena exhaustion and from 'not a user process' (line 2049)… | +| `system/kernel/process.zig:118` | maximum_resolve_path | 224 | external-data | Confirmed at process.zig:1898: `if (path_len == 0 or path_len > maximum_resolve_path or path_ptr >= user_half_end or path_ptr + path_len > user_half_e… | No log; -1 only — confirmed. There is no ENAMETOOLONG in this path (the function does have `failErr(state, ipc… | +| `system/kernel/process.zig:1542` | exit_subscriber_capacity | 16 | hardware | Confirmed at process.zig:1573-1581: `subscribeExits` scans `for (&exit_subscribers) \|*slot\| { if (slot.* == null) { endpoint.refcount += 1; slot.* =… | Materially worse than claimed, and this is my main correction. The kernel does return a specific -ENOSPC, but … | +| `system/kernel/process.zig:1649` | timer_capacity | 16 | our-design | Confirmed at process.zig:1684-1692: `systemTimerBind` scans `for (&one_shot_timers) \|*slot\| { if (slot.* == null) { ...; return architecture.setSyst… | Effectively none, correcting the claim. The kernel's -ENOSPC is narrowed to a bool by library/kernel/time.zig:… | +| `system/kernel/process.zig:2172` | maximum_segments | 16 | external-data | Confirmed at process.zig:2200: `if (ehdr.e_phnum > maximum_segments) return error.BadElf;`. The claim's key observation is right — the test is on `e_p… | None — confirmed. `error.BadElf` is erased at process.zig:1018 and `system_spawn` returns -1, so a developer i… | +| `system/kernel/process.zig:2173` | maximum_pages | 256 | external-data | Confirmed at process.zig:2238, inside the PT_LOAD loop: `total_pages += seg.pages(); if (total_pages > maximum_pages) return error.ProgramTooBig;`. Th… | No kernel log — confirmed. `error.ProgramTooBig` is erased at the syscall boundary and becomes the same -1 as … | +| `system/kernel/scheduler.zig:206` | maximum_space_mappings | 16 | our-design | Clean refusal, confirmed. `recordSpaceMappingLocked` (scheduler.zig:343-355) walks `entry.mappings` for a null slot and `return false` when full; proc… | A bare -1 from `shared_memory_map`/`shared_memory_create` — the same value as a bad handle, an exhausted arena… | +| `system/kernel/vfs.zig:143` | @memcpy(d.path[0..parent.len], parent) — implici… | 64 == 64, by coincidence | external-data | Safe today, and safe with slightly more margin than the claim states. parentOf (vfs.zig:119-123) returns a strict prefix ending before the last '/', s… | Nothing would warn. The coupling is invisible from either file: system/initial-ramdisk.zig:26-28 reasons about… | +| `system/kernel/vfs.zig:315` | name_out (via process.zig:1972 name_buffer) | [64]u8 | external-data | Silent truncation with a header that agrees with the truncation rather than reporting it: vfs.zig:314-317 `const n = @min(name.len, name_out.len); @me… | None. The contrast is fair — system/kernel/log-ring.zig sets abi.klog_flag_truncated for exactly this situatio… | +| `system/services/acpi/acpi.zig:81` | mmio_scratch | [4096]u8 | external-data | No overflow — the aliasing IS the failure. library/device/acpi/aml/interpreter.zig:708 (readRegionByte) and :720 (writeRegionByte) both do `const virt… | None. No log fires on a SystemMemory access in either the interpreter or the HAL, so an _STA or _CRS that cons… | +| `system/services/acpi/acpi.zig:113` | | 64 | hardware | Falls through silently and proceeds as though ACPI mode were enabled. `while (tries < 1000 and (halPioRead(2, pm1a_cnt) & sci_en_bit) == 0) : (tries +… | None that distinguishes the two outcomes. No log fires on give-up, and the success message "acpi: power button… | +| `system/services/acpi/acpi.zig:306` | | 1000 | hardware | Falls through silently and proceeds as though ACPI mode were enabled. `while (tries < 1000 and (halPioRead(2, pm1a_cnt) & sci_en_bit) == 0) : (tries +… | None that distinguishes the two outcomes. No log fires on give-up, and the success message "acpi: power button… | +| `system/services/device-manager/device-manager.zig:50` | registry_source | [8192]u8 | external-data | Silent truncation of the file. `loadRegistry` (63-68): `while (used < registry_source.len) { const n = file.read(registry_source[used..]) orelse break… | Nothing says the file was cut. Only downstream symptoms: `{d} malformed line(s) skipped` (line 70) if the frag… | +| `system/services/device-manager/device-manager.zig:51` | registry_source | 8192 bytes | external-data | Silent truncation of the file. `loadRegistry` (63-68): `while (used < registry_source.len) { const n = file.read(registry_source[used..]) orelse break… | Nothing says the file was cut. Only downstream symptoms: `{d} malformed line(s) skipped` (line 70) if the frag… | +| `system/services/device-manager/device-manager.zig:531` | | reply tail of a 256-byte pac… | our-design | Silent truncation with a success status: `if (written + entry_size > tail.len) break;` and the handler returns the byte count. There is no cursor and … | None. The client sees a short but well-formed list. | +| `system/services/display/backend.zig:27` | device_table | [64]device.DeviceDescriptor | hardware | Confirmed as written but NOT reachable today. findDisplay (lines 54-63) does `const total = device.enumerate(&device_table); const n = @min(total, dev… | Misleading when it does fire, as claimed: Gop.init exhausts its retries and writes `display: no framebuffer de… | +| `system/services/display/compositor.zig:288` | | src.len >= w*h*4, against a … | our-design | Silent no-op reported as success. blitTile does `if (src.len < @as(usize, w) * h * 4) return;`, but blitLayer (display.zig:236-242) ignores that, stil… | None whatsoever: no log, and a success status. The symptom is a blank region. | +| `system/services/display/display.zig:76` | maximum_layers | 16 | our-design | Confirmed. `createLayer` (display.zig:208-210) is `const slot = freeLayer() orelse return null;`, and `onCreateLayer` (695-700) is `const slot = creat… | Nothing server-side for a client-facing exhaustion — confirmed. The service's own cursor path does log (`_ = l… | +| `system/services/display/display.zig:395` | mode_list / list | [4]backend_mod.Mode | hardware | Silent truncation on both hops: the backend clamps with `const count = @min(@min(offered.count, scanout_protocol.max_modes), out.len);` (backend.zig:1… | None — no log names the number offered versus the number kept. The mode-set self-check only says "no alternate… | +| `system/services/fat/engine.zig:70` | sector_size | 512 | external-data | Mount is refused, confirmed. Both paths test `if (geometry.bytes_per_sector == sector_size)` — engine.zig:193 (bare FAT at LBA 0) and engine.zig:209 (… | A wrong diagnosis: `_ = logging.write("/system/services/fat: not a FAT filesystem\n")` at system/services/fat/… | +| `system/services/fat/engine.zig:193` | sector_size (declared line 70) | 512 | external-data | Refuses to mount rather than corrupting, confirmed at engine.zig:193 and the MBR-partition path at engine.zig:209; both fall through to `return null`. | system/services/fat/fat.zig:162 `_ = logging.write("/system/services/fat: not a FAT filesystem\n")` — false fo… | +| `system/services/fat/fat.zig:73` | open_nodes (inline [_]OpenNode{.{}} ** 32) | 32 | our-design | Error returned, but the WRONG error, and silently: `const index = allocOpen() orelse return refused;` where `const refused: isize = -envelope.ENOENT;`… | Nothing. No log line at exhaustion, and the client is told the file does not exist. An operator debugging this… | +| `system/services/fat/fat.zig:285` | | reply tail of a 256-byte pac… | our-design | Silent truncation with a success status: `const name_len = @min(listing.name_len, into.len);` then `return @intCast(name_len);` — the client receives … | None. A directory listing shows a chopped filename that then fails to open. | +| `system/services/fat/fat.zig:285` | into.len (inline; = envelope.packet_maximum - pr… | 224 bytes | external-data | Silent truncation, and the reply's name_len is set to the TRUNCATED value, so the client cannot tell there was more. The engine's own buffer holds 260… | Nothing. | +| `system/services/init/init.zig:70` | max_service_args | 4 | external-data | Silent drop of the surplus arguments: loadServices, init.zig:122-127 `while (it.next()) \|argument\| { if (argument.len == 0) continue; if (service.ar… | None, and the contrast with its immediate sibling is confirmed: the max_services cap ten lines above logs `"/s… | +| `system/services/init/init.zig:161` | maximum_bindings | 16 | our-design | An error is returned to the binder: onBind, init.zig:620-622 `const slot = for (&bindings) \|*binding\| { if (!binding.used) break binding; } else ret… | Weak, and the asymmetry is confirmed: within the same function the EPERM path logs (init.zig:602-608) and the … | +| `system/services/init/init.zig:162` | maximum_grants | 64 | external-data | Parsing stops at 64 rows: loadGrants, init.zig:237-240 `if (grant_count >= grants.len) { _ = logging.write("... protocol.csv has more rows than the ta… | The overflow itself is logged with a direct `logging.write` naming the file (init.zig:238). The consequences a… | +| `system/services/init/init.zig:220` | protocol_csv | 16384 bytes | external-data | Silent truncation mid-line, then a heuristic report. readConfiguration (init.zig:138-146) fills the buffer and stops; the tail rows vanish and the las… | One heuristic info log: init.zig:147 `if (used == into.len) std.log.info("{s} filled the read buffer — rows pa… | +| `system/services/init/init.zig:220` | protocol_csv (with maximum_grants = 64 rows alon… | 16384 bytes / 64 rows | external-data | Truncation, announced on both limits. readConfiguration reports the byte cap at init.zig:147 (`std.log.info("{s} filled the read buffer — rows past {d… | Both limits log, which is what makes this the calibration point. The byte-limit line rides std.log.info (the r… | +| `system/services/init/init.zig:274` | process_table | [64]process.ProcessDescripto… | our-design | Partly handled, partly not — confirmed exactly. refreshProcessTable (init.zig:278-282) sets `process_truncated = total > process_table.len`, and taskA… | A bind refusal outside identify does log (init.zig:602-608), but a failed identify in onBind returns -EPERM wi… | +| `boot/efi.zig:308` | pool_pages | 64 frames (256 KiB) | hardware | TablePool.alloc, efi.zig:264-265: `if (self.next >= self.cap) return error.OutOfBootstrapFrames;`. Every map2M/map4K path is `try`, so it unwinds clea… | Good, and verified: main (efi.zig:34-39) writes unconditionally to con_out — `log("\r\nEFI: boot failed: "); l… | +| `boot/efi.zig:543` | info_buffer | 1024 bytes | external-data | `const n = try directory.read(&info_buffer);` (efi.zig:545) into a [1024]u8. UEFI answers EFI_BUFFER_TOO_SMALL for an EFI_FILE_INFO that will not fit;… | efi.zig:71-75 on con_out, naming the UEFI error but not the file. | +| `boot/efi.zig:543` | info_buffer (inline [1024]u8) | 1024 bytes | external-data | `const n = try directory.read(&info_buffer);` (efi.zig:545) — EFI_BUFFER_TOO_SMALL becomes an error and the `try` propagates out of walkDirectory and … | efi.zig:71-75 on con_out; names the UEFI error, not the file. | +| `boot/efi.zig:660` | BootInformation.kernel_segments (declared system… | [8]KernelSegment | our-design | `if (n < boot_information.kernel_segments.len) { ... }` (efi.zig:659-668) with no else — the segment is copied to its physical address but never recor… | Nearly nothing, but not quite nothing: kernel.zig:159 prints `log.print(" kernel segs: {d} (mapped with W^X p… | +| `boot/efi.zig:680` | (attempts) and cap = info.len +… | 8 attempts; +8 spare descrip… | hardware | Correct and safe, as claimed. efi.zig:680-700: on each attempt both pool buffers are sized from the firmware's own descriptor count plus 8, a failed g… | The fatal case reaches main's handler: 'EFI: boot failed: ExitBootServicesFailed' on con_out, then hlt. No bre… | +| `boot/efi.zig:787` | buffer (logBytes) | 128 UTF-16 units | our-design | `if (i + 1 >= buffer.len) break;` (efi.zig:790) — the tail of the string is dropped, the result is NUL-terminated at the truncation point and printed.… | This IS the observability path — logBytes is the only way a runtime string (in practice an @errorName from mai… | +| `build/images.zig:152` | (mk_fat.addArg("64")) | 64 (MiB) | our-design | Build-time hard failure. tools/make-fat-image.py:67-68 — `if cluster >= self.cluster_count + 2: sys.exit("error: image out of clusters")` inside the c… | "error: image out of clusters" on stderr, which aborts the build. It names neither the file being added nor a … | +| `library/device/acpi/aml/interpreter.zig:55` | maximum_segments | 16 | external-data | Silent truncation in `Cursor.segment` (123-129): `const s = try self.take(4); if (name_path.count < maximum_segments) { ...store... }` — the cursor ad… | None. Line 299's comment `// unknown name -> treat as uninitialised` is by design, so a truncated name is indi… | +| `library/device/acpi/aml/parser.zig:24` | maximum_segments | 64 | external-data | appendSegment (parser.zig:137-143) always consumes the 4 name bytes via readNameSegment — so the cursor stays in sync — but silently discards the segm… | None. aml.zig:36-56 builds a ParseResult{consumed,total} precisely so a desync is detectable, but the only con… | +| `library/device/block/block.zig:95` | | 600 attempts × 50 ms = 30 s | our-design | Returns null; the caller (the FAT server) reports no volume. | Nothing logged at this level, but the number is justified against a measurement. | +| `library/device/driver/driver.zig:165` | lookup_attempts | 100 (× lookup_pause_ms = 20 … | our-design | `hello()` runs `while (attempts < lookup_attempts) : (attempts += 1) { if (channel.openEndpoint("device-manager")) \|handle\| break handle; time.sleep… | Good, and I confirmed it at driver.zig:181 — the else branch IS the log. Requiring callers also announce their… | +| `library/device/model/device-abi.zig:139` | DeviceDescriptor.hid | [8]u8 | hardware | No truncation is reachable today. The cited site (devices-broker.zig:121-122) cannot truncate: `node.hid()` returns `hid_buffer[0..hid_len]` from devi… | Not applicable — nothing truncates. The 8-byte-with-no-NUL width does cause a live defect elsewhere: system/se… | +| `library/device/pci/pci.zig:37` | bar_virtual | [6]usize | hardware | Errors: `pub fn mapBar(self: *Function, bar: u8) ?usize { if (bar >= 6) return null;` — a caller asking for BAR 6 gets null rather than an out-of-boun… | Null from `mapBar`, which each driver reports in its own words. | +| `library/device/pci/pci.zig:283` | CapabilityIterator.guard (and ExtendedCapability… | 48 and 480 | hardware | Iteration ends. Silent, but correctly so: the bound is derived from the size of the address space being walked, so reaching it means the device's chai… | Nothing — and it needs nothing, because the bound cannot cut a well-formed chain short. | +| `library/device/pci/pci.zig:286` | | 48 | hardware | The iterator silently ends: `if (self.cursor == 0 or self.guard >= 48) return null;` — indistinguishable from the end of a well-formed list. `findCapa… | Nothing. No log distinguishes 'no MSI capability' from 'capability list is a loop'. | +| `library/device/pci/pci.zig:371` | | 480 | hardware | Silently ends the iteration: `if (self.cursor == 0 or self.guard >= 480) return null;`. `findExtendedCapability` then reports absence. | Nothing. | +| `library/device/registry/device-registry.zig:200` | cols | [9][]const u8 | external-data | Errors and counts. parseLine: `if (count >= cols.len) return .malformed; // too many columns` and after the loop `if (count != cols.len) return .malfo… | Verified end to end, and it is genuinely good: system/services/device-manager/device-manager.zig:71 `if (resul… | +| `library/device/registry/device-registry.zig:239` | out_rules.len (caller-supplied) | caller's buffer | external-data | `if (result.count >= out_rules.len) { result.truncated = true; continue; }` inside parse — the surplus valid rules are dropped and the flag records it… | The best in the tree, and I confirmed the caller actually uses it: device-manager.zig:72 `if (result.truncated… | +| `library/kernel/channel.zig:37` | path_maximum | 224 | our-design | Refuses: `join` (285) `if (total > buffer.len) return null;`. `reach` (182-193) returns null when `file_system.fsResolve` cannot write the rewritten p… | None — null, no log, same documented three-way ambiguity. | +| `library/kernel/channel.zig:46` | name_maximum | 64 | our-design | Refuses, never truncates: `join` (282-290) `if (name.len == 0 or name.len > name_maximum) return null;`, which `openEndpoint` (238-242) turns into a n… | None (no log), and the null is deliberately three-way ambiguous per the doc at 231-237: no such contract / not… | +| `library/kernel/memory/heap.zig:199` | | 16 | our-design | Refuses cleanly: `fn allocImpl(_: *anyopaque, len: usize, alignment: std.mem.Alignment, _: usize) ?[*]u8 { if (alignment.toByteUnits() > 16) return nu… | An allocation failure, which std reports as OutOfMemory — a misleading name for an alignment refusal, and no l… | +| `library/kernel/process.zig:116` | | [8]u8 | our-design | The kernel refuses the *sender* rather than truncating (the receive capacity travels in r8 to `ipc_reply_wait`), so a client that calls into a process… | The refusal reaches the caller as an IPC failure; the stopping process logs nothing. | +| `library/kernel/process.zig:185` | | [256]u8 | our-design | Errors, cleanly: `if (len >= blob.len) return null;` and `if (len + argument.len > blob.len) return null;` — the spawn never happens and null is retur… | Poor. `spawn*` returning null is the same answer as 'no such binary' or 'task table full'; the caller (e.g. th… | +| `library/kernel/service.zig:94` | subscriber_capacity | 8 | our-design | onSubscribe (service.zig:264) walks `slots`, and if no slot is free returns `-envelope.ENOSPC`. The turn then closes the endpoint capability that arri… | No log in the provider. On the subscriber side the only real caller in the tree does log: system/services/disp… | +| `library/kernel/thread.zig:41` | tls_block_size | 64 | our-design | No check exists anywhere. thread.zig:88-90 carves `closure_addr` down from the stack top and then `const tls_base = (closure_addr - tls_block_size) & … | None — nothing tracks a TLS extent, so nothing could report it. The symptom would be a thread running with cor… | +| `library/protocol/display/display-protocol.zig:99` | max_modes | 4 | our-design | Silent drop of the surplus, but it can never fire. Provider side confirmed at system/services/display/display.zig:749-757: `var list: [4]backend_mod.M… | Nothing at either end — no log — but nothing is dropped either. | +| `library/protocol/scanout/scanout-protocol.zig:29` | max_modes | 4 | hardware | Silent truncation on both sides, exactly as claimed. Driver: virtio-gpu.zig:525-527 `var offered = scanout_protocol.Modes{ .count = offered_modes.len … | None. Confirmed: neither `modes()` in backend.zig nor the get_modes handler in virtio-gpu.zig emits anything a… | +| `library/xkeyboard-config/xkeyboard-config.zig:80` | map(layout, hid_usage: u8, …) | u8 (indexing Layout.keys: [2… | hardware | No overflow is possible — `keys: [256]Key` (library/xkeyboard-config/generated/layouts.zig:20) exactly covers the u8 domain, and an unmapped usage yie… | A key that maps to nothing produces no character; nothing distinguishes 'unmapped' from 'narrowed to the wrong… | +| `system/drivers/pci-bus/pci-bus.zig:36` | var line: [320]u8 (the per-func… | 320 bytes | our-design | `const text = std.fmt.bufPrint(&line, "...", .{...}) catch return;` at line 37 — the whole breadcrumb for that function is dropped and logFunction ret… | Nothing. The function simply does not appear in the boot log, which reads identically to "the walk did not fin… | +| `system/drivers/pci-bus/pci-bus.zig:182` | if (descriptor.resource_count >… | 8 | our-design | `break` out of the `while (i < 6)` BAR-sizing loop at line 181. Remaining BARs are never sized, never recorded in the descriptor and never restored-th… | Nothing. No log, no counter, and the loop's `configWrite16(bus, dev, function, 0x04, command)` decode-restore … | +| `system/drivers/ps2-bus/ps2-library.zig:464` | while (guard < 16) in drainOutp… | 16 bytes | hardware | Returns with the output-buffer-full bit possibly still set. No error, no retry, no signal to the caller (the function returns void). Callers are ps2-b… | None — silent void return. | +| `system/drivers/usb-hid/keyboard.zig:107` | receive: [64]u8 (identically at… | 64 | our-design | Currently sufficient with 4 bytes to spare: the protocol's own test asserts Protocol.event_maximum == 60 and event_maximum <= envelope.post_maximum (l… | Nothing here would report an over-long message; the driver's guard is `if (message.length < @sizeOf(hid.Keyboa… | +| `system/drivers/usb-xhci-bus/usb-xhci-bus.zig:521` | hid_buffer: [8]u8 (the "P… | 8 bytes | our-design | Confirmed at lines 521-523: `var hid_buffer: [8]u8 = undefined; const hid_text = std.fmt.bufPrint(&hid_buffer, "P{d}I{d}", .{ port, interface.number }… | Nothing. `catch ""` swallows the bufPrint failure, register returns an existing id which looks like a normal i… | +| `system/drivers/usb-xhci-bus/usb-xhci-bus.zig:763` | if (serviced >= 32) break | 32 hub changes per tick | our-design | Deferral, not loss. `while (engine.takeHubChange()) \|change\| { ... serviced += 1; if (serviced >= 32) break; }` — the remaining changes stay in each… | Not needed; nothing is lost. There is no log, correctly. | +| `system/drivers/usb-xhci-bus/usb-xhci-library.zig:131` | trbs_per_ring (= page_size / @sizeOf(Trb)) | 256 | our-design | Correct wraparound. `push` (155-178) writes slot `enqueue_index`, and when the index reaches `trbs_per_ring - 1` (the Link slot) it re-installs the Li… | Not needed. | +| `system/drivers/usb-xhci-bus/usb-xhci-library.zig:456` | port_changes: [16]u32 | 16 | hardware | Silently dropped in `pump` (1610-1613): `if (self.port_change_count < self.port_changes.len) { self.port_changes[...] = port; self.port_change_count +… | Nothing at the drop site. Recovered in practice by the level reconcile in serviceController (usb-xhci-bus.zig:… | +| `system/drivers/usb-xhci-bus/usb-xhci-library.zig:456` | port_changes (inline [16]u32) | 16 | hardware | Silent drop, no else (pump, 1610-1613). The PORTSC change bits are already acknowledged at 1608-1609, so the edge is consumed even when the queue entr… | Nothing at the drop site; recovered by the independent per-port level reconcile at usb-xhci-bus.zig:727-741, w… | +| `system/drivers/usb-xhci-bus/usb-xhci-library.zig:496` | guard < 64 in the xECP capabili… | 64 extended capabilities | hardware | The `while (offset != 0 and guard < 64)` walk simply stops; a Supported Protocol capability past the 64th is not logged. Diagnostic only. | Nothing distinguishes 'chain ended' (the `if (next == 0) break;` at 509) from 'guard tripped'. | +| `system/drivers/usb-xhci-bus/usb-xhci-library.zig:603` | device_context_array = memory.d… | 512 entries (4096 / 8), vs M… | hardware | Safe, but only by accident of type width: slot_id is a u8 (`return @truncate(event.control >> 24)` in enableSlot), so `array[device.slot_id]` cannot e… | Not applicable today. | +| `system/drivers/usb-xhci-bus/usb-xhci-library.zig:1069` | while (tries < 200) with time.s… | 200 tries = ~1 s | hardware | Falls out of the loop, re-reads status at 1076, and if still not enabled: `std.log.info("hub slot {d} port {d}: reset did not enable", .{ hub.slot_id,… | Good — a specific log line naming hub slot and port, quoted above. | +| `system/drivers/virtio-gpu/virtio-gpu.zig:77` | queue_size | 16 | our-design | The only failure direction is a device offering fewer than 16, and it is checked and refused loudly — confirmed at lines 282-286: `const device_qsize … | Explicit log line carrying the device's actual value, then a clean bring-up failure. | +| `system/drivers/virtio-gpu/virtio-gpu.zig:163` | while (tries < 2000) in waitUse… | 2000 iterations ≈ 2 s | our-design | Returns false (line 172). submit returns false, command_nodata returns 0 (line 184), which never equals ok_nodata, so each bring-up call site logs its… | Good on the BRING-UP path only. The serving path is silent: presentFull's two failure returns (lines 461 and 4… | +| `system/drivers/virtio-gpu/virtio-gpu.zig:405` | .scanout_id = 0 (and resource_i… | 1 scanout, 1 resource, 1 bac… | hardware | No runtime limit exists — confirmed. Grepped the whole tree: `get_display_info`, `RespDisplayInfo` and `max_scanouts` appear ONLY in their declaration… | Nothing about head count, because it is never asked. The file header does state the scope: "this instance clai… | +| `system/drivers/virtio-gpu/virtio-gpu.zig:486` | while (tries < 50) with time.sl… | 50 tries = 1 s | our-design | Logs and returns, confirmed at lines 485-492: `const display = while (tries < 50) : (tries += 1) { if (channel.openEndpoint("display")) \|h\| break h;… | One clearly worded log line. The lasting consequence — the display stays on the boot framebuffer permanently —… | +| `system/drivers/virtio-gpu/virtio-gpu.zig:525` | .count = offered_modes.len set … | 2 offered vs 4 slots | our-design | Not reachable at current constants (offered_modes.len = 2 <= scanout_protocol.max_modes = 4), so nothing happens today. The shape is unsafe as describ… | Would be silent at the point of the bug; the downstream re-clamps mean the surplus is dropped rather than read… | +| `system/kernel/acpi.zig:130` | maximum_rmrr | 8 | hardware | Counted, not silent — confirmed at acpi.zig:873-882: `if (platform_information.rmrr_count < maximum_rmrr) { ...record... } else { platform_information… | Good, and I verified the log fires: system/kernel/iommu.zig:393-394 `if (info.rmrr_skipped > 0) log.print(" r… | +| `system/kernel/architecture/x86_64/cpu.zig:69` | systemCallArg (switch arms 0..5) | 6 arguments | our-design | `else => 0` — an argument index of 6 or more silently reads as zero rather than failing. Same shape at cpu.zig:788 (`pioRead` returns 0 for a width ot… | None; a caller asking for a seventh argument gets a plausible-looking 0. | +| `system/kernel/architecture/x86_64/gdt.zig:28` | entries | 7 | our-design | Compile-time only. `const entries = 7;` sizes the template (gdt.zig:36-44) and both gdts rows; setTssFor writes fixed indices 5 and 6, and loadOnThisC… | n/a — a build-time constraint. | +| `system/kernel/architecture/x86_64/gdt.zig:50` | gdts | [maximum_cpus][7]u64 (128 ta… | hardware | `loadOnThisCpu(cpu)` (gdt.zig:78-84) takes `@intFromPtr(&gdts[cpu])` with no guard in this file; in a safe build an out-of-range cpu is a bounds-check… | Via the ACPI/SMP path only: kernel.zig:281 prints 'cpus : WARNING {d} core(s) beyond pool cap dropped' from pl… | +| `system/kernel/architecture/x86_64/ioapic.zig:24` | overrides | [16]IsoEntry | hardware | Silent truncation at three independent layers, all verified: acpi.zig:558 `if (platform_information.override_count < platform_information.overrides.le… | None at any of the three layers — no counter, no log. | +| `system/kernel/architecture/x86_64/iommu-intel.zig:284` | 10_000_000 (spinStatus) | 10000000 | our-design | Logs a warning and returns; `enable()` then sets `gcmd_shadow \|= gcmd_te` and the core reports the IOMMU as online even though translation may not be… | "/system/kernel: WARNING VT-d status bit never set — translation may be incomplete" — a real log line, which i… | +| `system/kernel/architecture/x86_64/paging.zig:70` | bootstrap_physmap_limit | 4 << 30 (4 GiB) | hardware | `@panic("paging: table frame above the 4 GiB bootstrap physmap")` at paging.zig:84-85, guarded by `if (!on_own_tables and frame >= bootstrap_physmap_l… | A named panic message on the console/serial. The 9-line comment at lines 55-62 states the window, why it holds… | +| `system/kernel/architecture/x86_64/per-cpu.zig:50` | blocks | [maximum_cpus]ArchitecturePe… | hardware | `setLocal(index, …)` and `setKernelRsp(index, …)` index unguarded; bounded upstream, ReleaseSafe bounds check as backstop. A ReleaseFast build would w… | Nothing in this file. | +| `system/kernel/architecture/x86_64/serial.zig:83` | 10_000 (probe guard) | 10000 | hardware | The loop ends, `echo = register(0)` is read regardless, and `return echo == 0xAE` (serial.zig:83-86). A false answer sets uart_present = false, after … | Recoverable but indirect: present() is exposed (serial.zig:91-93) and the boot log reports the posture, and re… | +| `system/kernel/architecture/x86_64/smp.zig:124` | (1 << 32) | 4294967296 | our-design | `if (cr3 >= (1 << 32)) @panic("smp: kernel page tables above 4 GiB");` — the first statement of startAp, before arm() and before any INIT/SIPI is sent… | Named panic message that states the actual condition. | +| `system/kernel/architecture/x86_64/smp.zig:141` | @intCast(tramp_physical >> 12) | u8 SIPI vector (frame must b… | hardware | `const vector: u8 = @intCast(tramp_physical >> 12);` — in a safety-checked build (the default), a trampoline frame at or above 1 MiB triggers a "cast … | A safety-check panic with the generic cast message — it does not name the trampoline or the 1 MiB constraint. … | +| `system/kernel/architecture/x86_64/smp.zig:149` | 100 (ms AP wake window) | 100 | our-design | `const deadline = apic.millis() + 100;` then a pause-spin on ap_alive; on expiry `return false`. The caller at system/kernel/kernel.zig:437-444 retrie… | Good, and the best in this set. On give-up: `log.print("/system/kernel: cpu apic_id {d}: no response after {d… | +| `system/kernel/architecture/x86_64/tss.zig:48` | tss_table / ap_ist_top | [maximum_cpus]Tss and [maxim… | hardware | `rsp0Ptr` (62-64), `setApIstStack` (68-70) and `setupThisCpu` index `tss_table[cpu]` / `ap_ist_top[cpu]` with no guard in this file; the bound is enfo… | Reported by the layer above, and it really is reported: kernel.zig:280-281 `if (platform.cpusDropped() > 0) lo… | +| `system/kernel/device-model.zig:72` | name_buffer | [24]u8 | our-design | Silent truncation in `setName` (lines 94-98, `@min` then `@memcpy`), reached from `DeviceTree.init` (line 138) and `addChild` (line 159). In practice … | None, but nothing incorrect happens either; names are diagnostic only (the boot dump), never a matching key. | +| `system/kernel/device-model.zig:72` | Device.name_buffer | [24]u8 | our-design | Silent truncation in setName; unreachable in practice. See the kernel-core duplicate: every caller passes either a literal or the result of a bufPrint… | None needed. | +| `system/kernel/ipc-synchronous.zig:71` | POST_MAXIMUM | 64 | our-design | `if (len > POST_MAXIMUM) return -E2BIG;` — sendLocked, ipc-synchronous.zig:566, before anything is touched. A distinct errno straight back to the call… | -E2BIG, specific and actionable. | +| `system/kernel/ipc-synchronous.zig:113` | notify_buffer | [8]u64 | our-design | Silent drop of the NEWEST badge: notifyLocked, ipc-synchronous.zig:596-602 — the store happens only inside `if (endpoint.notify_tail -% endpoint.notif… | None, and the file argues at 593-595 why that is correct for a level rather than merely tolerable. | +| `system/kernel/ipc-synchronous.zig:113` | Endpoint.notify_buffer | [8]u64 | our-design | Silent drop of the newest badge — outside notifyLocked's guard nothing but the wake happens (ipc-synchronous.zig:596-602). | Nothing, by design. | +| `system/kernel/kernel.zig:246` | isos | [16]architecture.IsoEntry | hardware | Silent clamp: `var isos: [16]architecture.IsoEntry = undefined; const iso_n = @min(pinfo.override_count, isos.len);` (246-247) — a second truncation s… | None. The boot log prints the I/O APIC base and route information but never the override count. | +| `system/kernel/kernel.zig:421` | maximum_wake_attempts | 3 | our-design | The core is left parked and bring-up continues: the `while (attempt <= maximum_wake_attempts)` loop at 437 falls through to the log at 443-444; the ad… | Exemplary, and verified verbatim: `log.print("/system/kernel: cpu apic_id {d}: no response after {d} attempts… | +| `system/kernel/log.zig:41` | maximum_sinks | 8 | our-design | Silent ignore, confirmed at lines 47-52: `pub fn addSink(sink: SinkFn) void { if (sink_count < maximum_sinks) { sinks[sink_count] = sink; sink_count +… | None in code, though self-announcing in practice (output does not appear on the missing channel). | +| `system/kernel/log.zig:85` | ring_capacity | 512 * 1024 | our-design | Whole records are reclaimed from the tail, never a torn record — confirmed at log-ring.zig:48, `while (self.head + record_len - self.tail > capacity) … | Best in the tree, as claimed, and I verified each mechanism: per-boot monotonic `sequence` stamped into every … | +| `system/kernel/log.zig:244` | PanicRecord.message | [512]u8 | our-design | Silent truncation that is self-describing, confirmed at lines 251-256: `const n: u32 = @intCast(@min(message.len, panic_record.message.len)); @memcpy(… | The record carries its own length and the magic-last discipline prevents a torn read. Confirmed the caller's p… | +| `system/kernel/process.zig:109` | maximum_arguments / maximum_argument_bytes | 8 / 256 | our-design | All three checks confirmed. process.zig:976 `if (arguments_len > maximum_argument_bytes) return fail(state);`; process.zig:1012 `if (argc == maximum_a… | -1 from system_spawn with no log and the error name discarded at process.zig:1018 — confirmed. | +| `system/kernel/process.zig:148` | write_buffer | [256]u8 | our-design | Confirmed at process.zig:1801: `if (len <= write_buffer.len and ptr < user_half_end and ptr + len <= user_half_end) { ... } else { fail(state); }` — I… | -1 to the caller at the syscall, and the ring-level truncation is flagged in the record header — the only boun… | +| `system/kernel/process.zig:1460` | exit_record_capacity | 64 | our-design | Confirmed. `recordExitLocked` (1467-1470) overwrites the oldest via `exit_record_next = (exit_record_next + 1) % exit_record_capacity;` and an evicted… | -ESRCH with a documented interpretation, and the docstring (1455-1459) states the equivalence to 'an id that n… | +| `system/kernel/scheduler.zig:155` | ipc_maximum_handles | 32 | our-design | Visible and correctly unwound, confirmed. `installEntry` (ipc-synchronous.zig:618-627) scans `t.handles` and `return -ENOSPC`; the cap-passing path at… | -ENOSPC reaches the caller through the IPC status — a distinct, actionable error, unlike the -1 that most othe… | +| `system/kernel/scheduler.zig:485` | cpus (PerCpu pool) | [maximum_cpus]PerCpu = [128] | hardware | Cannot be overrun. acpi.zig:534-543 gates recording on `if (cpu_information.count < cpu_information.cpus.len)` and counts the surplus into `dropped`, … | kernel.zig:280-281 `if (platform.cpusDropped() > 0) log.print(" cpus : WARNING {d} core(s) beyond pool … | +| `system/kernel/tests.zig:410` | buffer (device enumerate staging) | [64]device_abi.DeviceDescrip… | our-design | Two coupled problems. (1) The value 64 duplicates devices-broker's `maximum_devices` in eight places (tests.zig:410, 1449, 2834, 3931, 4230, 4246, 437… | None for either. A test that silently examines a truncated table still passes. | +| `system/kernel/vfs.zig:59` | maximum_prefix / maximum_rewrite | 64 / 32 | our-design | Visible refusal on the mount path: mountBackend, vfs.zig:383-384 `if (!isAbsolute(prefix) or prefix.len < 2 or prefix.len > maximum_prefix) return fal… | A -1 return with no log and no distinct errno. The mounting service's own failure message is what an operator … | +| `system/kernel/vfs.zig:143` | Directory.path | [maximum_prefix]u8 = [64]u8 | external-data | `@memcpy(d.path[0..parent.len], parent);` (vfs.zig:143) into `path: [maximum_prefix]u8` (vfs.zig:88) with no @min and no guard — the only copy in the … | A boot-time panic in safe modes (loud but fatal, before init runs); nothing at all plus corrupted neighbours i… | +| `system/services/acpi/acpi.zig:74` | registered | [64]Registered | our-design | Unreachable — this is NOT the operative ceiling. `if (registered_count >= registered.len) return;` at line 513 is the first statement of registerDevic… | The real ceiling does log, once per lost device: `std.log.info("register refused for {s}", .{hid[0..@intCast(h… | +| `system/services/device-manager/device-manager.zig:51` | registry_rules | [64]registry.Rule | external-data | Rules past the 64th are dropped by `registry.parse` (library/device/registry/device-registry.zig:239-241): `if (result.count >= out_rules.len) { resul… | Good: device-manager.zig:72 `if (result.truncated) _ = logging.write("/system/services/device-manager: /system… | +| `system/services/device-manager/device-manager.zig:129` | Driver.name_buffer | [64]u8 | our-design | Silent truncation, not refusal: `addDriver` (242-244) `const n = @min(name.len, driver.name_buffer.len); @memcpy(driver.name_buffer[0..n], name[0..n])… | Indirect: the failure appears one step later as `failed to spawn {s}` (line 268) printing the truncated path, … | +| `system/services/device-manager/device-manager.zig:260` | id_text | [20]u8 | our-design | `arguments[0] = std.fmt.bufPrint(&id_text, "{d}", .{driver.device_id}) catch return;` (264) — on overflow `spawnDriver` returns without spawning AND w… | None — the only silent return in spawnDriver; every other failure path there logs (268 `failed to spawn`, 285/… | +| `system/services/display/backend.zig:69` | | 100 | our-design | Confirmed at lines 68-75: `var tries: u32 = 0; const found = while (tries < 100) : (tries += 1) { if (findDisplay()) \|f\| break f; time.sleepMillis(5… | Logged, but misleadingly — "headless?" is one of at least three possible causes (genuinely headless, discovery… | +| `system/services/display/compositor.zig:74` | DamageList.capacity | 16 | our-design | No loss: `add` (lines 79-93) ends `self.rects[capacity - 1] = self.rects[capacity - 1].unite(r);`, so the overflow rectangle is united into the last e… | Not needed — nothing is dropped and nothing becomes incorrect. | +| `system/services/display/compositor.zig:121` | TileGrid.maximum_columns / maximum_rows | 128 / 128 | hardware | Graceful, documented degradation. `reset` clamps with `@min((width + tile_size - 1) / tile_size, maximum_columns)` and the same for rows (lines 139-14… | No log, and none needed — the behaviour stays correct, only the repaint granularity changes. | +| `system/services/fat/engine.zig:24` | max_transfer_sectors | 8 | our-design | Not a failure mode, confirmed. The transfer loop clamps and issues more commands: `const run: u32 = @intCast(@min(full, max_transfer_sectors));`. The … | Not applicable — correctness is unaffected, only the number of device round-trips. | +| `system/services/fat/engine.zig:80` | block_cache_lines | 16 | our-design | Round-robin eviction, write-through — correctness preserved, only hit rate degrades. | Not needed. | +| `system/services/fat/engine.zig:871` | | 4096 | external-data | `if (sector_index > 4096) return null;` ends the search, so findFreeRun/addEntry return null, createFile/createDirectory return null and the VFS handl… | None in the service; no log, and the errno collapses into the same ENOENT everything else uses. | +| `system/services/fat/engine.zig:1047` | run | [21]EntryLoc | external-data | Silent partial cleanup, confirmed. `var run: [21]EntryLoc = undefined;` at engine.zig:1047 (removeFile, line 1046) and 1114 (rename, line 1110), recor… | None. | +| `system/services/init/init.zig:69` | max_services | 16 | external-data | Rows past the 16th are refused and parsing stops: init.zig:114-117 `if (service_count >= services.len) { _ = logging.write(...); break; }`. | Good: a direct `logging.write` (not just a ring entry) naming the file — `"/system/services/init: /system/conf… | +| `system/services/init/init.zig:84` | init_csv / protocol_csv | [4096]u8 / [16384]u8 | external-data | Truncation, detected and reported. readConfiguration (init.zig:138-149) fills the buffer and stops (`while (used < into.len) { const n = file.read(int… | An explicit log naming the file and the byte count. It is a heuristic (a file exactly the buffer's size gives … | +| `system/services/init/init.zig:84` | init_csv | 4096 bytes | external-data | Silent truncation mid-line via the shared readConfiguration (init.zig:141-146), after which the severed line parses as a service path that does not ex… | The same heuristic info log as protocol.csv: init.zig:147 '{s} filled the read buffer — rows past {d} bytes ar… | +| `system/services/init/init.zig:160` | maximum_name | 64 | our-design | Refused, not truncated — contractName, init.zig:498-502 `if (name.len == 0 or name.len > maximum_name) return null;`, and both callers answer with an … | None on the open path, deliberately (init.zig:652-677 explains why a refusal and an absence must be the same a… | +| `system/services/init/init.zig:173` | Binding.binary | [64]u8 | our-design | Silent truncation: onBind, init.zig:630-632 `const binary_len = @min(identity.binary.len, slot.binary.len); @memcpy(slot.binary[0..binary_len], identi… | The truncated path is what appears in the provenance line (init.zig:635) and in the EBUSY refusal message (ini… | +| `system/services/logger/logger.zig:53` | maximum_files | 24 | our-design | No loss. fileFor (line 229) looks for a matching cached name, then a free slot, then `const cached = slot orelse evictOne() orelse return null;` at li… | Nothing logged on eviction. The symptom is slow logging, not lost logging, which is the right trade. | +| `system/services/logger/logger.zig:57` | CachedFile.name | [logging.maximum_process_nam… | external-data | Panic, in a narrower window than claimed. fileFor formats the path FIRST (line 245, `std.fmt.bufPrint(&path, "{s}/{s}.log", ...) catch return null`) i… | A process fault, visible as the logger dying; init's crash-loop policy stops restarting it after three deaths,… | +| `system/services/logger/logger.zig:72` | carry_capacity | 64 + 256 + 64 | our-design | consume (line 187-191): `const rest = bytes.len - offset; if (rest > carry_capacity) { carry_len = 0; // cannot happen with sane frames; drop rather t… | Partly self-reporting, and the claim was right to credit it: deliver (lines 199-206) compares header.sequence … | +| `system/services/logger/logger.zig:82` | | 64 | our-design | Benign today. `service.run(64, .{ .init = initialise, .on_message = onMessage, ... })` at line 82 sizes the harness receive/reply buffer. The logger s… | None, and none needed while onMessage is a no-op. | +| `system/services/logger/logger.zig:243` | path | [base.len + 1 + 19 + 1 + log… | our-design | Silent record loss. `const full = std.fmt.bufPrint(&path, "{s}/{s}.log", .{ boot_directory[0..boot_directory_len], relative }) catch return null;` (li… | None. Records for that process never appear on disk, and the gap accounting cannot flag it: next_expected_sequ… | +| `tools/make-fat-image.py:53` | (65525) | 65525 clusters (~33 MiB at 5… | our-design | Loud build failure: `sys.exit(f"error: image too small for FAT32 ({self.cluster_count} clusters < 65525); use a larger size")` — and --verify repeats … | Build stops with a message naming both the actual count and the requirement, and telling the operator what to … | +| `tools/make-fat-image.py:212` | (LFN sequence numbering) | implicitly 20 entries / 255 … | our-design | Silent corruption (my reading; not exercised): `count = len(pairs) // 13` then `entry[0] = sequence \| (0x40 if sequence == count else 0)` — for a nam… | Nothing at build time; --verify only resolves EFI/BOOT/BOOTX64.EFI by short name, so a mangled LFN chain elsew… | +| `tools/make-fat-image.py:329` | (guard) | 100000 cluster hops | our-design | `return None`, which the caller turns into a misleading verdict: `sys.exit("verify: EFI/BOOT/BOOTX64.efi not found")` — a chain-too-long or cyclic FAT… | A wrong-but-visible error message. The operator is told the stub is absent when the real problem is FAT struct… | +| `tools/make-iso-image.py:151` | sector_count | 0xFFFF (32 MiB at 512 B/sect… | our-design | Deliberate silent clamp, confirmed at line 151: `sector_count = min(0xFFFF, esp_size // 512)`, packed into the El Torito default entry's 16-bit field … | Nothing at build time about the clamp itself, but the true geometry is visible: the builder prints the ESP siz… | +| `tools/make-xkeyboard-config.py:308` | (range(256)) / generated keys: … | 256 HID usages | hardware | No overflow inside the generator (HID_TO_NAME only maps usages below 0x100). The bound is exported into the generated `Layout.keys: [256]Key`, so any … | Nothing here; whatever the input service does with an out-of-range usage is its own concern (outside this area… | +| `tools/make-xkeyboard-config.py:314` | (range(4)) / generated levels: … | 4 shift levels per key | external-data | Silent truncation at generation time: `levels = [resolve_keysym(kd.levels[i], keysymdef) if i < len(kd.levels) else (0, 0) for i in range(4)]` — level… | Nothing. The generator prints no warning and the generated file looks complete; the missing characters only ap… | + diff --git a/docs/os-development/bounds.md b/docs/os-development/bounds.md new file mode 100644 index 0000000..26e7411 --- /dev/null +++ b/docs/os-development/bounds.md @@ -0,0 +1,140 @@ +# Bounds: how a ceiling is declared + +*Design, 2026-08-08. Follows [fixed-bounds-audit.md](../fixed-bounds-audit.md), which +found 235 compile-time ceilings in this tree: 139 on quantities we do not choose, 5 +recorded anywhere with their reasoning, and 171 that pass in silence when reached.* + +A bound is a number chosen at compile time that decides how much of something the code +can hold. `const maximum_devices = 64`. `var below: [64]Range`. `var blob: [512]u8`. +Different units — devices, firmware memory-map entries, bytes of a USB descriptor — but +one shape, and one recurring way of going wrong. + +## Where a bound lives + +**Where the thing it bounds lives.** A driver's transfer-ring size belongs to that +driver; a protocol's payload cap belongs to that protocol; the kernel's task-table size +belongs to the kernel. There is no central list and this document does not propose one. + +`system/parameters.zig` is not a counter-example. It is kernel-only, and it exists for a +specific historical reason: tunables had accumulated inside the loader↔kernel handoff +contract, and splitting them out kept that contract to what it actually is. It is a +tidying of one file's contents, not a registry the rest of the system reports to. + +This matters for the mechanism below. An earlier draft had every bound declared through +a shared `bounds` module — which would have meant adding a dependency to roughly eight +package manifests, including `library/protocol`, which deliberately depends on nothing. +That is a coupling the problem does not require: a bound is a local fact about local +storage, and the only thing worth sharing is the *shape of the statement*, not a module. + +## The declaration + +A structured doc comment, immediately above the declaration, in the file that owns it: + +```zig +/// bound: logical CPUs the kernel tracks +/// decided-by: hardware +/// protects: the per-CPU bookkeeping arrays, which are sized at compile time +/// at-limit: degrade — surplus cores are left parked, never brought online +/// observed-by: platform.cpusDropped() -> the WARNING at kernel.zig:281 +pub const maximum_cpus = 128; +``` + +Five fields, all mandatory: + +| Field | Answers | +|---|---| +| `bound` | what is counted, in plain words | +| `decided-by` | `hardware`, `external`, or `ours` — who chooses how large it gets | +| `protects` | what this ceiling defends against | +| `at-limit` | `refuse` / `degrade` / `truncate` / `grow`, and the detail | +| `observed-by` | how an operator finds out it was reached | + +`decided-by` is the classification the audit turned on. `hardware` means the machine +chooses — PCI functions, CPUs, ACPI rows, memory-map entries. `external` means a file, +disk structure or peer chooses. `ours` means we do: a stack size, a tick rate, our own +protocol's payload. A fixed bound on the first two is a defect rather than a tunable. + +## What the build step enforces + +A step in `build.zig` reads the tree and fails on: + +1. **A bound with no declaration.** A fixed-size array or a `maximum_*`/`max_*` constant + with no `bound:` block above it. The 235 that exist today are allowlisted by + file+line+name, so only newly written ones are gated — the rule can land without a + 235-site sweep in front of it. +2. **A missing field.** All five or it fails. This alone is the 97 bounds with no + comment at all and the 171 with no observability. +3. **An unspeakable `at-limit`.** The vocabulary is closed. There is no `silent`, no + `drop`, and nothing meaning *allow*. `truncate` is legal only with a marker the + reader can see — `klog_maximum_message` qualifies because the record carries + `klog_flag_truncated`; the USB configuration descriptor cut at 512 bytes does not, + because nothing records that anything was lost. +4. **A stale allowlist entry.** If a listed bound is fixed or deleted, its entry goes + too, so the list can only shrink. + +Because this is text and not a Zig type, it also covers `boot/`, which imports almost +nothing, and `tools/*.py`, where the audit found bounds as well. One mechanism, whole +tree, no new dependency edges. + +## Coupled bounds + +Two numbers that must agree, agreeing in code rather than in a comment — no module +needed, just a `comptime` block where one of them lives: + +```zig +comptime { + if (maximum_domains != devices_broker.maximum_devices) + @compileError("iommu.confined is indexed by device id; an id past its end is " ++ + "left unconfined while confineDevice still reports success"); +} +``` + +`maximum_domains = 64` and `maximum_devices = 64` agree today only by a sentence in a +comment, and the agreement fails open. This is the clause with a live hole behind it, +and the reason raising `maximum_devices` alone would be a privilege escalation rather +than a fix. + +## The worked bad case + +`devices_broker.maximum_devices`, which had no comment at all: + +```zig +/// bound: device nodes for the whole machine — firmware-discovered plus registered +/// decided-by: hardware +/// protects: nothing; this is a sizing guess about someone else's computer +/// at-limit: refuse — ENOSPC from device_register, dropped++ during discovery +/// observed-by: kernel.zig:203 counts discovery drops only, NOT runtime refusals +const maximum_devices = 64; +``` + +Writing it out is the argument. `decided-by: hardware` alongside a `protects` that +admits there is no threat describes a bound that should not be fixed at all, and +`observed-by` cannot be filled in honestly. The build step does not reject this — it +makes it impossible to write down without noticing. + +## The work-list this produces + +`decided-by` is machine-readable, so the sweep is a query: every `hardware` or +`external` bound whose `at-limit` is not `grow`. That is 139 of the 235, and it is the +order the fixing takes — by class, not by our guesses about which machines get run. +Reachability is exactly what a new computer changes; the Ryzen's cap was unreachable +until it wasn't. + +## What this does not do + +**It is a statement, not a proof.** Nothing checks that the code does what `at-limit` +claims. The enforcement is completeness and vocabulary: you cannot leave the question +unanswered, and you cannot answer it with "silently". + +**It carries no occupancy.** Knowing `system/configuration/protocol.csv` sits at 50 of +64 grant rows still needs a counter and somewhere to report it. The declarations make +that cheap to add later; it is not here. + +**It resizes nothing.** Declaring `maximum_devices` honestly does not make the Ryzen +work. It makes the next machine's failure legible, and it names the 139 that need real +fixes. + +**Enforcement is at build time, not compile time.** A Zig type could have made a missing +field a compile error. That version needed the shared module, and the module was not +worth the coupling — so a missing field is a failed build step instead. In practice both +mean `zig build` stops; the difference is which stage prints the message. diff --git a/docs/os-development/device-authority.md b/docs/os-development/device-authority.md new file mode 100644 index 0000000..7e49a8f --- /dev/null +++ b/docs/os-development/device-authority.md @@ -0,0 +1,173 @@ +# Device authority: the kernel stops keeping an inventory + +*Design, drafted 2026-08-07. Not implemented. Prompted by a real machine: an AMD +Ryzen desktop enumerated more PCI functions than the kernel's device table would +hold, and the xHCI and SATA controllers were refused registration — so the +machine booted to the compositor with no USB and no storage.* + +The kernel keeps a table of every device userspace discovers. It is a fixed +array of 64 descriptors, 344 bytes each, and a second cap allows any one parent +16 children. Neither number is written down anywhere as a decision: +`maximum_devices` has no comment and never reached `parameters.zig`, where every +other tunable in this kernel lives with its reasoning attached. + +Raising them is not the fix. The numbers are wrong because the *table* is wrong: +it is an inventory of hardware, and an inventory of hardware is not something a +kernel needs. This document proposes replacing it with capabilities, which +removes the ceiling rather than moving it. + +## What the kernel actually uses + +Every read of a device descriptor from the kernel proper, exhaustively: + +| Used for | What it needs | +|---|---| +| `mmio_map`, `io_read`/`io_write` | the physical range, to check the mapping falls inside it | +| `irq_bind`, `msi_bind` | the GSI, and that the caller owns the device | +| `dma_bind` | that the caller owns the device | +| IOMMU confinement | the **PCI BDF**, to key a domain | +| `device_register` | the parent's ranges, for the containment check | +| the boot display seed | one framebuffer window | + +That is: **physical ranges, interrupt numbers, and one BDF.** Vendor, device and +subsystem ids, class triples, the human-readable names, the parent links, the +bus numbers — the kernel stores all of it and reads none of it. It is held so +that `device_enumerate` can hand it back to user space, which is the whole +mistake in one sentence: the kernel is acting as a distribution mechanism for +data it does not use. + +## The split + +Three concerns are tangled in one table. + +**Platform bring-up** — timers, LAPIC/IOAPIC, CPU topology, the ECAM window, the +framebuffer the firmware left. The kernel derives these from ACPI before user +space exists and needs them to function. They never belonged in the device table +and mostly are not (the platform block in `kernel.zig` is separate); this +document does not change them. + +**Resource authority** — which task may map which physical range, receive which +interrupt, touch which ports. This *must* stay in the kernel. It is the one +grant that cannot be audited after the fact: a process that maps arbitrary +physical memory owns the machine, page tables and IOMMU structures included. +This is memory protection, not device management, and it is why the answer is +not simply "move it all to the device manager". + +**Device inventory** — what exists, what it is, how it is arranged, which driver +should bind it. This is `device-manager`'s job and is already half there: it +loads `devices.csv`, matches, spawns drivers, and receives `child_added` reports +over its own protocol. The kernel table duplicates what those reports already +carry. + +## The proposal: a resource is a capability + +Device resources join endpoints, shared memory and DMA regions as a kind in the +handle table. + +1. **Roots.** At boot the kernel mints capabilities for the windows it learned + from firmware — the ECAM range, the framebuffer, the legacy port space — and + hands them to the first bus drivers. This is the only place device knowledge + enters the kernel, and it comes from ACPI, not from a driver's say-so. +2. **Subdivision.** A bus driver enumerating hardware derives a narrower + capability from one it holds: `resource_derive(cap, kind, start, len) → cap`. + The kernel checks the sub-range lies inside the capability being subdivided — + the same containment rule as today (`devices-broker.contains`), but checked + against *one capability the caller demonstrably holds* rather than by walking + a global tree. +3. **Delegation.** The driver passes that capability to the child driver over + IPC. Cap-passing already exists; this is the mechanism `subscribe` and + `attach_scanout` already use. +4. **Use.** `mmio_map`, `irq_bind`, `msi_bind`, `io_read`/`io_write` and + `dma_bind` take a capability handle instead of `(device_id, resource_index)`. + Possession *is* the authority — there is nothing to look up and no ownership + table to consult. + +Exclusivity stops being a broker refusing a second claimant and becomes the +ordinary property of a capability: only one process was given it. + +## What this buys + +**No ceiling.** There is no table to size, so no machine is too big. The +Ryzen's enumeration stops being a limit to tune and becomes what it is — a fact +about a computer. + +**Reclamation, free.** `count` in the broker today only ever increases; +`releaseAllOwnedBy` clears a dead driver's *claims* but never its entries. A +driver that crashes and is restarted re-registers its children and consumes the +table again — reachable today without any malice, given the device manager +restarts drivers by design. Capabilities die with the task. + +**No quota needed.** The per-parent cap exists to stop one claimant looping +`device_register` and filling the shared table, because a zero-resource child +sidesteps the containment check. With no shared table there is nothing to +exhaust; a process can only ever subdivide what it was given, and its handles +are already bounded per task. + +**A smaller kernel.** Three syscalls leave the ABI, one narrower one arrives, +and `devices-broker.zig` largely disappears along with both constants. + +**The discipline the rest of the system already uses.** "The claim is the +capability" is written in the driver documentation as if it were already true. +This makes it true. + +## Syscall surface + +- `device_enumerate` — **retires.** It exists only to read the kernel's table. + Callers ask `device-manager`, whose protocol already reserves an `enumerate` + verb. Note this is a public-ABI change: `vdso.md` documents it. +- `device_register` — **splits.** The kernel half becomes `resource_derive`; the + publication half ("this device exists, here is what it is") becomes an IPC + message to `device-manager`, which is where the inventory belongs and where + `child_added` already carries the same facts. +- `device_claim` — **dissolves into possession**, except for the IOMMU (below). +- `mmio_map`, `irq_bind`, `msi_bind`, `io_read`, `io_write`, `dma_bind` — keep + their names and semantics; their first argument becomes a capability handle. + +## The open question: where the IOMMU attaches + +This is the one place the kernel still needs device *identity* rather than a +range. `confineDevice(device_id, bdf, owner)` builds a domain keyed by PCI BDF, +attaches it, and the claim is rolled back if confinement fails — deliberately, +so a device that cannot be confined is never driven. + +Three options, none obviously right: + +1. **The BDF rides the capability.** A memory capability derived for a PCI + function carries its BDF, and the kernel confines on first `mmio_map` or + `dma_bind`. Keeps the syscall count down; means a capability is no longer + purely a range. +2. **An explicit `device_attach(cap, bdf)`.** Honest and visible, but it puts a + BDF — a fact about PCI — into a kernel interface that otherwise knows nothing + about buses, and something must stop a caller naming a BDF that is not + theirs. +3. **The root PCI capability carries the segment, and derivation computes the + BDF.** Purest, but only works for PCI and the kernel would be parsing bus + topology, which is precisely what this document is trying to stop. + +My inclination is (1), because confinement is a property of the resource being +granted rather than a separate action, and because it keeps the "possession is +authority" story intact. It needs the derivation call to know it is carving a +PCI function, which is a wart worth arguing about. + +## What this does not solve + +- **Hot-plug and removal.** Capabilities die with their holder, but a device + that physically disappears while a driver lives is the device manager's + problem and unchanged by this. +- **Quotas on physical memory.** Nothing here bounds how much a driver maps; it + bounds only *what* it may map. That was already true. +- **The device tree as a published thing.** `/system/devices` remains a device + manager concern, as the file-system hierarchy already assumes. + +## Migration sketch + +Not a plan yet — the phases need sizing once the IOMMU question is settled. +Rough shape: introduce the capability kind and `resource_derive` alongside the +existing table; convert the resource-consuming syscalls to accept either form; +move the inventory into `device-manager` and convert its clients off +`device_enumerate`; then delete the table, the two constants and the three +syscalls in one flag-day, as the `ServiceId` retirement did. + +Every driver is affected, so the QEMU suite is the arbiter at each step, and the +Ryzen is the acceptance test — it is the machine that found this, and the one +that proves it fixed. diff --git a/docs/os-development/vdso.md b/docs/os-development/vdso.md index 24b6b7a..5f56a8e 100644 --- a/docs/os-development/vdso.md +++ b/docs/os-development/vdso.md @@ -150,6 +150,57 @@ they are wire values a Rust program needs verbatim. What stays private in `abi.zig` is exactly the thing the vDSO exists to hide: the `SystemCall` numbers and the trap convention. +## Errors: the errno space + +A failed call returns `-errno`. The runtime detects failure the way Linux +does — a return value in the top 4096 — so every code stays inside 1..4095. +These are **public**: unlike the call numbers, a caller must be able to read +them verbatim, and they are the same vocabulary whether the number came from +the kernel or from a user-space provider answering over IPC. + +They are defined once in `system/abi.zig`. The kernel restates them in +`system/kernel/ipc-synchronous.zig` and the envelope restates the +provider-facing subset in `library/protocol/envelope/envelope.zig` (the +`protocol` package deliberately depends on nothing, so it cannot import +`abi`); a comptime check in `library/device/driver/driver.zig` makes drift a +compile error. + +| # | Name | Meaning | +|---|------|---------| +| 1 | `EBADF` | bad handle | +| 2 | `E2BIG` | an argument exceeds its maximum (a message, a descriptor's resource count) | +| 3 | `EFAULT` | buffer unmapped, or outside the user half | +| 4 | `ENOENT` | no such name | +| 5 | `ENOSPC` | a kernel table is full (handles, devices) | +| 6 | `ENOMEM` | out of memory | +| 7 | `EPEER` | the peer died before replying — its process exited or was killed | +| 8 | `ESRCH` | no such process | +| 9 | `EPERM` | not permitted: the caller is not the owner or supervisor | +| 10 | `ENOSYS` | this protocol has no such operation | +| 11 | `EPROTO` | malformed packet: shorter than the verb it names | +| 12 | `EBUSY` | the thing asked for is held by someone still alive | +| 13 | `ENODEV` | no such device id | +| 14 | `ECHILDREN` | this parent already holds as many children as it can | +| 15 | `ERANGE` | a resource escapes the window it must fall inside | +| 16 | `ECONFINE` | the device could not be placed under IOMMU translation | + +`EPEER` is the one with no POSIX counterpart and it is worth stating plainly: +synchronous IPC blocks the caller until the server replies, so the caller +needs an answer for "the server died while I was waiting." It is not a +transport error and not a refusal — the request may well have been carried +out — it says only that no reply is coming. A client that treats it as +"retry" can duplicate work; the honest response is to re-resolve the protocol +name, because the provider it held is gone. + +**A refusal names the rule that refused it.** This is a rule and not a +courtesy. `device_register` alone can fail six ways, and until each got its +own code a bus driver could only report "refused" — which is how an AMD +desktop came to boot with a working display, no USB and no storage, with +three independent causes indistinguishable in the log. See +[fixed-bounds-audit.md](../fixed-bounds-audit.md). A new failure mode that +does not fit an existing code gets a new one here rather than borrowing the +nearest. + ## Enforcement, and an honest threat model Renumbering only has teeth if the kernel **refuses syscalls that don't come diff --git a/library/device/driver/driver.zig b/library/device/driver/driver.zig index e2651de..b34af41 100644 --- a/library/device/driver/driver.zig +++ b/library/device/driver/driver.zig @@ -22,14 +22,47 @@ inline fn failed(r: usize) bool { return r > ~@as(usize, 0) - 4095; } +/// The errno inside a failed return. Only meaningful when `failed(r)`. +inline fn errnoOf(r: usize) i64 { + return -@as(i64, @bitCast(r)); +} + +// The envelope restates the kernel's errno numbering by hand, because the +// `protocol` package deliberately depends on nothing (so it cannot import `abi`). +// This module is one of the few that can see both halves, so it is where they are +// held together: drift becomes a compile error here rather than a driver reporting +// the wrong reason for a refusal. Anything linking a driver compiles this. +comptime { + if (envelope.ENOENT != abi.ENOENT) @compileError("envelope.ENOENT has drifted from abi.ENOENT"); + if (envelope.ENOSPC != abi.ENOSPC) @compileError("envelope.ENOSPC has drifted from abi.ENOSPC"); + if (envelope.EPERM != abi.EPERM) @compileError("envelope.EPERM has drifted from abi.EPERM"); + if (envelope.ENOSYS != abi.ENOSYS) @compileError("envelope.ENOSYS has drifted from abi.ENOSYS"); + if (envelope.EPROTO != abi.EPROTO) @compileError("envelope.EPROTO has drifted from abi.EPROTO"); + if (envelope.EBUSY != abi.EBUSY) @compileError("envelope.EBUSY has drifted from abi.EBUSY"); +} + /// Copy up to `buffer.len` device descriptors into `buffer`; returns the total count. pub fn enumerate(buffer: []DeviceDescriptor) usize { return sc.systemCall2(.device_enumerate, @intFromPtr(buffer.ptr), buffer.len); } -/// Take exclusive ownership of device `id`. Returns false if taken or invalid. -pub fn claim(id: u64) bool { - return !failed(sc.systemCall1(.device_claim, id)); +/// Why a `claim` failed. Worth distinguishing: `AlreadyClaimed` means back off and +/// let the owner have it, `NoSuchDevice` means this id is stale and the caller should +/// re-enumerate, and `NotConfined` means the machine could not place the device under +/// IOMMU translation — the claim was rolled back, and that one is a fault report, not +/// a retry. `Refused` is an errno this library does not know a name for. +pub const ClaimError = error{ NoSuchDevice, AlreadyClaimed, NotConfined, Refused }; + +/// Take exclusive ownership of device `id`. +pub fn claim(id: u64) ClaimError!void { + const r = sc.systemCall1(.device_claim, id); + if (!failed(r)) return; + return switch (errnoOf(r)) { + abi.ENODEV => error.NoSuchDevice, + abi.EBUSY => error.AlreadyClaimed, + abi.ECONFINE => error.NotConfined, + else => error.Refused, + }; } /// Map resource `resource_index` (which must be an MMIO window) of claimed device @@ -56,11 +89,38 @@ pub const no_pci_class = device_abi.no_pci_class; /// a bus driver may only subdivide what it already owns. `descriptor.id` and `descriptor.parent` /// are ignored. A device with no resources at all is fine — a USB device is reached /// through its controller, not by MMIO. -pub fn register(parent_id: u64, descriptor: *const DeviceDescriptor) ?u64 { +pub fn register(parent_id: u64, descriptor: *const DeviceDescriptor) RegisterError!u64 { const r = sc.systemCall2(.device_register, parent_id, @intFromPtr(descriptor)); - return if (failed(r)) null else r; + if (!failed(r)) return r; + return switch (errnoOf(r)) { + abi.ENOSPC => error.TableFull, + abi.ENODEV => error.NoSuchParent, + abi.EPERM => error.NotYourParent, + abi.E2BIG => error.TooManyResources, + abi.ECHILDREN => error.ParentFull, + abi.ERANGE => error.NotContained, + abi.EFAULT => error.BadDescriptor, + else => error.Refused, + }; } +/// Why a `register` failed. These are not interchangeable and a bus driver should +/// say which one it hit: `ParentFull` and `TableFull` are different ceilings with +/// different fixes, and `NotContained` is not a ceiling at all — it means the child +/// resource escaped the window the parent actually owns. Reporting all of them as +/// one refusal is what made an AMD desktop boot with no USB and no storage, and gave +/// no way to tell which of three causes it was (docs/fixed-bounds-audit.md). +pub const RegisterError = error{ + TableFull, // the kernel's device table is full, machine-wide + ParentFull, // this parent already holds as many children as it can + NoSuchParent, // no device with that id + NotYourParent, // that device exists but this process has not claimed it + TooManyResources, // the descriptor declares more resources than one device may hold + NotContained, // a resource escapes the parent's window + BadDescriptor, // the descriptor pointer did not read back + Refused, // an errno this library does not know a name for +}; + /// Bind resource `resource_index` (which must be an IRQ) of claimed device `device_id` to /// `endpoint`. From then on the interrupt arrives as an asynchronous notification: /// `ipc.replyWait` on that endpoint returns with the high bit set in `badge` and the diff --git a/system/abi.zig b/system/abi.zig index 9953f98..d8f2157 100644 --- a/system/abi.zig +++ b/system/abi.zig @@ -43,11 +43,11 @@ pub const SystemCall = enum(u64) { ipc_call = 9, // ipc_call(h, message, len, reply, cap) -> reply_len: send + block for reply ipc_reply_wait = 10, // ipc_reply_wait(h, reply, len, receive, cap) -> receive_len (+badge in rdx) device_enumerate = 11, // device_enumerate(buffer, maximum) -> count: snapshot the device table - device_claim = 12, // device_claim(id) -> ok: take exclusive ownership of a device + device_claim = 12, // device_claim(id) -> 0/-errno: take exclusive ownership of a device (-ENODEV no such id, -EBUSY someone owns it, -ECONFINE the IOMMU would not confine it) mmio_map = 13, // mmio_map(id, resource_index) -> virtual_address: map a claimed device's MMIO into this address space irq_bind = 14, // irq_bind(id, resource_index, endpoint): deliver a device IRQ as an IPC notification irq_ack = 15, // irq_ack(id, resource_index): re-arm a bound IRQ after servicing it - device_register = 16, // device_register(parent_id, descriptor) -> id: publish a child of a device you claimed + device_register = 16, // device_register(parent_id, descriptor) -> id/-errno: publish a child of a device you claimed (-ENOSPC table full, -ECHILDREN parent full, -ERANGE resource escapes the parent, -ENODEV/-EPERM bad parent, -E2BIG too many resources) system_spawn = 17, // system_spawn(name_ptr, name_len, arguments_ptr, arguments_len, exit_endpoint) -> child process id: start a named initial-ramdisk binary as a new ring-3 process dma_alloc = 18, // dma_alloc(len, flags) -> virtual_address (rax), physical_address (rdx): contiguous, pinned, uncacheable DMA memory dma_free = 19, // dma_free(virtual_address, len) -> 0: release a prior dma_alloc @@ -88,6 +88,44 @@ pub const SystemCall = enum(u64) { _, }; +/// **The errno space** — the one vocabulary of refusal, returned as `-value` in the +/// system_call result register and echoed by user-space providers in a reply status. +/// Canonical here because it crosses the kernel↔user boundary in both directions: +/// the kernel restates these in system/kernel/ipc-synchronous.zig, and the envelope +/// restates the provider-facing subset in library/protocol/envelope/envelope.zig +/// (which cannot import this module — the `protocol` package deliberately has no +/// dependencies, so a comptime cross-check in library/device/driver/driver.zig and +/// system/kernel/tests.zig holds the two halves together). +/// +/// The rule these serve: **a refusal must say which rule refused it.** A caller that +/// gets one number for five different reasons cannot report, retry or route around +/// any of them — see docs/fixed-bounds-audit.md, where a bare -1 turned "this bus is +/// at its child cap" into a machine that booted with no USB and no storage, and cost +/// a debugging session to tell apart from four other causes. +/// +/// Values are stable: `failed()` in the runtime treats the top 4096 return values as +/// errors (the Linux convention), so anything here must stay well inside 1..4095. +pub const EBADF: i64 = 1; // bad handle +pub const E2BIG: i64 = 2; // argument exceeds its maximum (a message, a descriptor's resource count) +pub const EFAULT: i64 = 3; // buffer unmapped / outside the user half +pub const ENOENT: i64 = 4; // no such name +pub const ENOSPC: i64 = 5; // a kernel table is full (handles, devices) +pub const ENOMEM: i64 = 6; // out of memory +pub const EPEER: i64 = 7; // peer died before replying (its process exited or was killed) +pub const ESRCH: i64 = 8; // no such process (process_kill of an unknown/dead id) +pub const EPERM: i64 = 9; // not permitted (the caller is not the owner/supervisor) +pub const ENOSYS: i64 = 10; // this protocol has no such operation +pub const EPROTO: i64 = 11; // malformed packet: shorter than the verb it names +pub const EBUSY: i64 = 12; // the thing asked for is held by someone still alive +pub const ENODEV: i64 = 13; // no such device id +pub const ECHILDREN: i64 = 14; // this parent already holds as many children as it can +pub const ERANGE: i64 = 15; // a resource escapes the window it must fall inside +pub const ECONFINE: i64 = 16; // the device could not be placed under IOMMU translation + +/// The highest errno defined above. A cheap guard for anyone switching over the +/// space, and the number to bump when adding one. +pub const errno_maximum: i64 = 16; + /// `futex_wait` return codes (in rax). pub const futex_woken: u64 = 0; // woken by a futex_wake pub const futex_mismatch: u64 = 1; // *addr != expected on entry; the caller did not block diff --git a/system/drivers/pci-bus/pci-bus.zig b/system/drivers/pci-bus/pci-bus.zig index 4420664..5cd81cc 100644 --- a/system/drivers/pci-bus/pci-bus.zig +++ b/system/drivers/pci-bus/pci-bus.zig @@ -44,6 +44,10 @@ var ecam_physical: u64 = 0; var start_bus: u64 = 0; var bus_count: u64 = 0; var manager_handle: ipc.Handle = 0; +/// Functions this scan discovered but could not publish. Counted so the end of the +/// scan can reconcile "found" against "registered" — a scan that silently returns a +/// subset of the machine is the failure this driver is most able to hide. +var refused: u32 = 0; /// One aligned 32-bit read from a function's configuration space. fn configRead(bus: u64, dev: u64, function: u64, offset: u64) u32 { @@ -74,10 +78,10 @@ fn configWrite16(bus: u64, dev: u64, function: u64, offset: u64, value: u16) voi /// Claim the bridge, map the ECAM, hello the manager, then scan. fn initialise(endpoint: ipc.Handle) bool { _ = endpoint; - if (!device.claim(bridge_id)) { - std.log.info("unable to claim bridge device {d}", .{bridge_id}); + device.claim(bridge_id) catch |e| { + std.log.info("unable to claim bridge device {d}: {s}", .{ bridge_id, @errorName(e) }); return false; - } + }; const buffer = memory.allocator().alloc(device.DeviceDescriptor, 64) catch { _ = logging.write("/system/drivers/pci-bus: out of memory\n"); return false; @@ -141,7 +145,13 @@ fn scan() void { } } } - std.log.info("{d} functions found", .{found}); + // Reconcile: "found" alone reads as success even when most of the machine was + // refused. If the two disagree, say so at a level that survives a scrollback. + if (refused == 0) { + std.log.info("{d} functions found, all registered", .{found}); + } else { + std.log.warn("{d} functions found, {d} REFUSED — {d} registered", .{ found, refused, found - refused }); + } } /// Register one function under the bridge and report it to the manager. The @@ -225,8 +235,12 @@ fn registerAndReport(bus: u64, dev: u64, function: u64, class_triple: u32) void // subsystem are read. Every discovered function is logged, matched or not. logFunction(bus, dev, function, class_triple, descriptor.vendor, descriptor.device, descriptor.subsystem); - const registered = device.register(bridge_id, &descriptor) orelse { - std.log.info("register refused for {d}:{d}.{d}", .{ bus, dev, function }); + // Name the rule that refused. `ParentFull` and `TableFull` are different ceilings + // with different fixes, and `NotContained` is not a ceiling at all — it means the + // BAR escaped the bridge's own window (docs/fixed-bounds-audit.md). + const registered = device.register(bridge_id, &descriptor) catch |e| { + refused += 1; + std.log.warn("register refused for {d}:{d}.{d}: {s}", .{ bus, dev, function, @errorName(e) }); return; }; // The registered device id is the packet's target — the manager's object diff --git a/system/drivers/ps2-bus/ps2-bus.zig b/system/drivers/ps2-bus/ps2-bus.zig index 6e21666..fcc0c72 100644 --- a/system/drivers/ps2-bus/ps2-bus.zig +++ b/system/drivers/ps2-bus/ps2-bus.zig @@ -114,10 +114,10 @@ pub fn main() void { _ = logging.write("/system/drivers/ps2-bus: found PS/2 controller\n"); _ = logging.write("/system/drivers/ps2-bus: initializing controller\n"); - if (!device.claim(controller_device_descriptor.id)) { - _ = logging.write("/system/drivers/ps2-bus: unable to claim controller \n"); + device.claim(controller_device_descriptor.id) catch |e| { + std.log.warn("unable to claim controller: {s}", .{@errorName(e)}); return; - } + }; const controller = ps2.Controller.init(controller_device_descriptor) orelse { _ = logging.write("/system/drivers/ps2-bus: controller is missing its IO ports\n"); @@ -255,14 +255,18 @@ pub fn main() void { if (port_device_types[@intFromEnum(ps2.Port.two)] != null) { if (ps2.findMouseDescriptor(buffer)) |descriptor| { if (findInterruptResourceIndex(descriptor)) |auxiliary_index| { - if (device.claim(descriptor.id) and device.irqBind(descriptor.id, auxiliary_index, endpoint)) { - maybe_auxiliary_interrupt = .{ - .device_id = descriptor.id, - .interrupt_index = auxiliary_index, - .gsi = descriptor.resources[auxiliary_index].start, - }; - } else { - _ = logging.write("/system/drivers/ps2-bus: auxiliary irq_bind failed\n"); + if (device.claim(descriptor.id)) |_| { + if (device.irqBind(descriptor.id, auxiliary_index, endpoint)) { + maybe_auxiliary_interrupt = .{ + .device_id = descriptor.id, + .interrupt_index = auxiliary_index, + .gsi = descriptor.resources[auxiliary_index].start, + }; + } else { + _ = logging.write("/system/drivers/ps2-bus: auxiliary irq_bind failed\n"); + } + } else |e| { + std.log.warn("auxiliary claim failed: {s}", .{@errorName(e)}); } } } diff --git a/system/drivers/usb-xhci-bus/usb-xhci-bus.zig b/system/drivers/usb-xhci-bus/usb-xhci-bus.zig index 16bfb7a..7a1d8b0 100644 --- a/system/drivers/usb-xhci-bus/usb-xhci-bus.zig +++ b/system/drivers/usb-xhci-bus/usb-xhci-bus.zig @@ -173,10 +173,10 @@ fn initialise(endpoint: ipc.Handle) bool { if (!channel.bindPatiently("usb-transfer", endpoint)) _ = logging.write("/system/drivers/usb-xhci-bus: /protocol/usb-transfer is another controller's; serving mine unnamed\n"); - if (!device.claim(controller_id)) { - std.log.info("unable to claim controller device {d}", .{controller_id}); + device.claim(controller_id) catch |e| { + std.log.warn("unable to claim controller device {d}: {s}", .{ controller_id, @errorName(e) }); return false; - } + }; // Fetch our own descriptor back for the controller's resources. const buffer = memory.allocator().alloc(device.DeviceDescriptor, 64) catch { @@ -522,8 +522,8 @@ fn reportInterface(manager: ipc.Handle, port: u32, interface: library.InterfaceI const hid_text = std.fmt.bufPrint(&hid_buffer, "P{d}I{d}", .{ port, interface.number }) catch ""; descriptor.hid_len = hid_text.len; @memcpy(descriptor.hid[0..hid_text.len], hid_text); - const registered = device.register(controller_id, &descriptor) orelse { - std.log.info("register refused for port {d} interface {d}", .{ port, interface.number }); + const registered = device.register(controller_id, &descriptor) catch |e| { + std.log.warn("register refused for port {d} interface {d}: {s}", .{ port, interface.number, @errorName(e) }); return null; }; diff --git a/system/drivers/virtio-gpu/virtio-gpu.zig b/system/drivers/virtio-gpu/virtio-gpu.zig index 33d16a5..c26c8da 100644 --- a/system/drivers/virtio-gpu/virtio-gpu.zig +++ b/system/drivers/virtio-gpu/virtio-gpu.zig @@ -200,10 +200,10 @@ fn testPixel(index: u32) u32 { fn initialise(endpoint: ipc.Handle) bool { _ = endpoint; - if (!device.claim(device_id)) { - std.log.info("unable to claim device {d}", .{device_id}); + device.claim(device_id) catch |e| { + std.log.info("unable to claim device {d}: {s}", .{ device_id, @errorName(e) }); return false; - } + }; var descriptors: [64]device.DeviceDescriptor = undefined; const total = device.enumerate(&descriptors); diff --git a/system/kernel/acpi.zig b/system/kernel/acpi.zig index 3b0fe5d..0e4e90b 100644 --- a/system/kernel/acpi.zig +++ b/system/kernel/acpi.zig @@ -615,17 +615,78 @@ var boot_memory_regions: []const boot_handoff.MemoryRegion = &.{}; /// I/O-APIC region, the high one from 4 GiB (or the end of RAM above it) to /// the 46-bit line. Coarse, mechanical, and AML-free — available at boot no /// matter what later moved to user space. +pub const AddressRange = struct { base: u64, end: u64 }; + +/// The largest holes below 4 GiB in a firmware memory map: the ranges the firmware +/// described *nothing* in, which is where a PCI BAR may legitimately live. Fills `out` +/// (keeping the largest, replacing the smallest held so far), ignores holes shorter +/// than `minimum`, and returns how many entries are non-empty; entries the caller +/// reads may be empty and are skipped by `end > base`. +/// +/// **It never copies the map, and that is the point.** The firmware chooses how many +/// descriptors its map has — commonly 60–200 on a real machine, 15–25 under OVMF. The +/// previous version copied the sub-4 GiB entries into a fixed `[64]` array and +/// `continue`d past the rest, which did not merely lose them: a region absent from the +/// walk is a region this function concludes is *free*, so a large enough map yields an +/// "aperture" lying over live RAM. `device_register` containment would then admit a +/// child BAR covering kernel memory, and its claimant could `mmio_map` it. A bound +/// whose overflow hands out authority is not a limit; the fix is not a bigger array. +/// +/// Pure, allocation-free, and linear-ish in the map (each inner pass consumes at least +/// one region, and it runs once at boot). +pub fn largestHolesBelow4G( + regions: []const boot_handoff.MemoryRegion, + minimum: u64, + out: []AddressRange, +) usize { + const limit: u64 = 1 << 32; + for (out) |*hole| hole.* = .{ .base = 0, .end = 0 }; + + var cursor: u64 = 0; + while (cursor < limit) { + // Step over every described region covering the cursor. Regions may overlap + // and chain, so repeat until the cursor stops moving. + var moved = true; + while (moved) { + moved = false; + for (regions) |region| { + if (region.base >= limit) continue; + const end = @min(region.base + region.pages * 4096, limit); + if (region.base <= cursor and end > cursor) { + cursor = end; + moved = true; + } + } + } + if (cursor >= limit) break; + + // The cursor now sits in a hole; it runs to the next described base, or to + // 4 GiB if nothing is described above it. + var next: u64 = limit; + for (regions) |region| { + if (region.base >= limit) continue; + if (region.base > cursor and region.base < next) next = region.base; + } + + if (next - cursor >= minimum and out.len != 0) { + var smallest: usize = 0; + for (out, 0..) |hole, i| { + if (hole.end - hole.base < out[smallest].end - out[smallest].base) smallest = i; + } + if (next - cursor > out[smallest].end - out[smallest].base) + out[smallest] = .{ .base = cursor, .end = next }; + } + cursor = next; + } + + var found: usize = 0; + for (out) |hole| { + if (hole.end > hole.base) found += 1; + } + return found; +} + fn addBridgeApertures(bridge: *device_model.Device) void { - // Below 4 GiB the described regions are sparse (RAM low, firmware flash - // and tables high), so the holes are the *gaps between* them — a single - // "after the last region" rule dies on OVMF's flash at the very top. - // Sort-merge the described ranges, then keep the three largest gaps - // (resource slots are bounded at 8 per device; ECAM + bus range + 3 + the - // high aperture fits). Above 4 GiB one aperture runs from the end of the - // described space to the 46-bit line. - const Range = struct { base: u64, end: u64 }; - var below: [64]Range = undefined; - var below_count: usize = 0; var high_end: u64 = 1 << 32; for (boot_memory_regions) |region| { const end = region.base + region.pages * 4096; @@ -636,37 +697,18 @@ fn addBridgeApertures(bridge: *device_model.Device) void { // kernel image, the tables, the ramdisk all live there). Bring-up // trust: only the bridge's claimant can register into the aperture. if (region.kind == .usable and end > high_end) high_end = end; - if (region.base >= (1 << 32) or below_count == below.len) continue; - below[below_count] = .{ .base = region.base, .end = @min(end, 1 << 32) }; - below_count += 1; } - // Insertion sort by base (the map is small and this runs once at boot). - for (1..below_count) |i| { - const key = below[i]; - var j = i; - while (j > 0 and below[j - 1].base > key.base) : (j -= 1) below[j] = below[j - 1]; - below[j] = key; - } - // Walk the sorted ranges, collecting inter-region gaps of at least 1 MiB. - var gaps: [3]Range = .{Range{ .base = 0, .end = 0 }} ** 3; - var cursor: u64 = 0; - var index: usize = 0; - while (index <= below_count) : (index += 1) { - const gap_end = if (index == below_count) (1 << 32) else below[index].base; - if (gap_end > cursor and gap_end - cursor >= (1 << 20)) { - // Keep the three largest, replacing the smallest kept so far. - var smallest: usize = 0; - for (gaps, 0..) |gap, gi| { - if (gap.end - gap.base < gaps[smallest].end - gaps[smallest].base) smallest = gi; - } - if (gap_end - cursor > gaps[smallest].end - gaps[smallest].base) { - gaps[smallest] = .{ .base = cursor, .end = gap_end }; - } - } - if (index < below_count and below[index].end > cursor) cursor = below[index].end; - } - for (gaps) |gap| { - if (gap.end > gap.base) _ = bridge.addResource(.memory, gap.base, gap.end - gap.base); + + // Three holes below 4 GiB. This three is not a guess about hardware: a + // `DeviceDescriptor` carries `maximum_device_resources` (8) resources, and the + // bridge spends them on ECAM + bus range + these + the high aperture. That cap is + // a wire struct in the kernel↔user ABI, so widening it is a separate change — + // docs/bounds-track-plan.md, phase 4. Keeping *fewer* holes is fail-closed: it + // refuses BARs, it never admits one. + var holes: [3]AddressRange = undefined; + _ = largestHolesBelow4G(boot_memory_regions, 1 << 20, &holes); + for (holes) |hole| { + if (hole.end > hole.base) _ = bridge.addResource(.memory, hole.base, hole.end - hole.base); } _ = bridge.addResource(.memory, high_end, (@as(u64, 1) << 46) - high_end); } diff --git a/system/kernel/devices-broker.zig b/system/kernel/devices-broker.zig index 0b1bc91..a985dd6 100644 --- a/system/kernel/devices-broker.zig +++ b/system/kernel/devices-broker.zig @@ -19,10 +19,11 @@ //! ever subdivide what it was already given. const std = @import("std"); +const abi = @import("abi"); const platform = @import("platform"); const device_abi = @import("device-abi"); -const maximum_devices = 64; +pub const maximum_devices = 64; /// Cap on children a single parent may have. A zero-resource child (legal — a USB /// device is addressed through its controller, not by MMIO) sidesteps the containment @@ -157,13 +158,13 @@ pub fn enumerateFrom(start: usize, out: []device_abi.DeviceDescriptor) usize { return n; } -/// Take exclusive ownership of device `id` for task `owner`. Fails if the id is -/// out of range or already claimed. -pub fn claim(id: u64, owner: u32) bool { - if (id >= count) return false; - if (claimed[@intCast(id)] != null) return false; +/// Take exclusive ownership of device `id` for task `owner`. The two ways this can +/// fail want different responses from a driver — a stale id means re-enumerate, a +/// live claimant means back off — so they are distinguishable (`ClaimError`). +pub fn claim(id: u64, owner: u32) ClaimError!void { + if (id >= count) return error.NoSuchDevice; + if (claimed[@intCast(id)] != null) return error.AlreadyClaimed; claimed[@intCast(id)] = owner; - return true; } /// The task that owns device `id`, or null. @@ -274,14 +275,70 @@ fn contains(parent: device_abi.ResourceDescriptor, child: device_abi.ResourceDes return child.start >= parent.start and child_end <= parent_end; } +/// Why a `register` was refused. Each variant maps to its own errno (`errnoOf`), so +/// a bus driver's log line can name the rule that stopped it — "this parent is at +/// its child cap" and "the table is full" want different fixes, and telling them +/// apart from a bare -1 cost a debugging session (docs/fixed-bounds-audit.md). pub const RegisterError = error{ NoSpace, // the device table is full - BadParent, // no such device, or not claimed by this task - TooManyResources, + NoSuchParent, // no device with that id + NotYourParent, // that device exists but this task has not claimed it + TooManyResources, // the descriptor declares more resources than one device may hold TooManyChildren, // this parent is at maximum_children_per_parent NotContained, // a child resource escapes its parent's window }; +/// The errno a refused `register` returns to ring 3. +pub fn errnoOf(e: RegisterError) i64 { + return switch (e) { + error.NoSpace => abi.ENOSPC, + error.NoSuchParent => abi.ENODEV, + error.NotYourParent => abi.EPERM, + error.TooManyResources => abi.E2BIG, + error.TooManyChildren => abi.ECHILDREN, + error.NotContained => abi.ERANGE, + }; +} + +/// Why a `claim` was refused. +pub const ClaimError = error{ + NoSuchDevice, // no device with that id + AlreadyClaimed, // a live task already owns it +}; + +/// The errno a refused `claim` returns to ring 3. (`ECONFINE` — the claim stood but +/// the IOMMU would not confine the device — is raised by the caller in +/// system/kernel/process.zig, which is what rolls the claim back.) +pub fn claimErrnoOf(e: ClaimError) i64 { + return switch (e) { + error.NoSuchDevice => abi.ENODEV, + error.AlreadyClaimed => abi.EBUSY, + }; +} + +/// The id of a child of `parent_id` already identical to `descriptor`, or null. +/// Exact on class, identity and every resource — anything less would let a bus +/// silently adopt an entry that is not the device it just found. The caller must +/// have bounded `descriptor.resource_count` first. +fn existingChild(parent_id: u64, descriptor: *const device_abi.DeviceDescriptor) ?u64 { + for (devices[0..count]) |*existing| { + if (existing.parent != parent_id) continue; + if (existing.class != descriptor.class) continue; + if (existing.pci_class != descriptor.pci_class) continue; + if (existing.hid_len != descriptor.hid_len) continue; + if (!std.mem.eql(u8, existing.hid[0..@intCast(existing.hid_len)], descriptor.hid[0..@intCast(descriptor.hid_len)])) continue; + if (existing.resource_count != descriptor.resource_count) continue; + var same = true; + for (0..@intCast(descriptor.resource_count)) |i| { + const a = existing.resources[i]; + const b = descriptor.resources[i]; + if (a.kind != b.kind or a.start != b.start or a.len != b.len) same = false; + } + if (same) return existing.id; + } + return null; +} + /// Number of devices currently recorded with `parent_id` as their parent. fn childCount(parent_id: u64) usize { var n: usize = 0; @@ -299,9 +356,27 @@ fn childCount(parent_id: u64) usize { /// contained in a parent resource of the same kind. A device with no resources is /// fine and common: a USB device is addressed through its controller, not by MMIO. pub fn register(parent_id: u64, owner: u32, descriptor: *const device_abi.DeviceDescriptor) RegisterError!u64 { - const parent_owner = ownerOf(parent_id) orelse return error.BadParent; - if (parent_owner != owner) return error.BadParent; + if (parent_id >= count) return error.NoSuchParent; + const parent_owner = ownerOf(parent_id) orelse return error.NotYourParent; + if (parent_owner != owner) return error.NotYourParent; + // Bounds every `descriptor.resources` read below, including the match scan's. if (descriptor.resource_count > device_abi.maximum_device_resources) return error.TooManyResources; + + // Idempotent on exact match (docs/device-manager.md): a restarted registering + // bus re-registers what it rediscovers, and the table has no unregister — an + // identical child under the same parent returns the existing id instead of + // appending a duplicate. + // + // Checked **before the caps**, because a re-registration consumes no slot. + // Charging it against the child cap refused a restarted bus its own devices the + // second time it started, which turned the supervision restart this system leans + // on into a one-way ratchet toward a degraded machine. The match is exact — class, + // identity, and every resource — so an entry returned this way was contained when + // it was first admitted, and a stored descriptor's resources never change + // afterwards (they are written only by `record`, `seedDisplay` and the append + // below). + if (existingChild(parent_id, descriptor)) |existing_id| return existing_id; + if (childCount(parent_id) >= maximum_children_per_parent) return error.TooManyChildren; if (count >= maximum_devices) return error.NoSpace; @@ -315,26 +390,6 @@ pub fn register(parent_id: u64, owner: u32, descriptor: *const device_abi.Device if (!ok) return error.NotContained; } - // Idempotent on exact match (docs/device-manager.md): a restarted - // registering bus re-registers what it rediscovers, and the table has no - // unregister — an identical (class, identity, resources) child under the - // same parent returns the existing id instead of appending a duplicate. - for (devices[0..count]) |*existing| { - if (existing.parent != parent_id) continue; - if (existing.class != descriptor.class) continue; - if (existing.pci_class != descriptor.pci_class) continue; - if (existing.hid_len != descriptor.hid_len) continue; - if (!std.mem.eql(u8, existing.hid[0..@intCast(existing.hid_len)], descriptor.hid[0..@intCast(descriptor.hid_len)])) continue; - if (existing.resource_count != descriptor.resource_count) continue; - var same = true; - for (0..@intCast(descriptor.resource_count)) |i| { - const a = existing.resources[i]; - const b = descriptor.resources[i]; - if (a.kind != b.kind or a.start != b.start or a.len != b.len) same = false; - } - if (same) return existing.id; - } - var d = std.mem.zeroes(device_abi.DeviceDescriptor); d.id = count; d.parent = parent_id; diff --git a/system/kernel/iommu.zig b/system/kernel/iommu.zig index 3e53939..824a95a 100644 --- a/system/kernel/iommu.zig +++ b/system/kernel/iommu.zig @@ -96,16 +96,37 @@ pub fn init() void { const Confined = struct { active: bool = false, owner: u32 = 0, bdf: u16 = 0, domain: u16 = invalid_domain }; var confined: [maximum_domains]Confined = .{Confined{}} ** maximum_domains; +// `confined` is indexed by **device id**, so it must cover every id the broker can +// mint. These two numbers agreed only by a sentence in a comment above +// `maximum_domains` — and when they disagreed, `confineDevice` returned success for +// the ids it had no room for, leaving those devices unconfined DMA masters. Coupled +// bounds agree in code, not in prose (docs/os-development/bounds.md). +comptime { + if (maximum_domains < devices_broker.maximum_devices) + @compileError("iommu.confined is indexed by device id but is smaller than the " ++ + "broker's device table: ids past its end cannot be confined, and so cannot " ++ + "be claimed at all"); +} + /// Place a just-claimed PCI function under IOMMU translation on behalf of `owner`: give /// it a private empty domain, seed it with the device's own firmware reserved region, /// and attach. Its DMA buffers arrive afterward as explicit grants — the owner's own /// `dma_alloc`'d regions are bound by the claim path (`mapForDevice`), and cross-process -/// buffers by `dma_bind`. false only if a domain can't be allocated — the caller rolls -/// the claim back (a claim that can't be confined must not stand). No-op success when no -/// IOMMU exists (fail-open). +/// buffers by `dma_bind`. false when the device cannot be confined — the caller rolls +/// the claim back (a claim that can't be confined must not stand). +/// +/// **Fail-closed at the table's edge.** There is exactly one deliberate fail-open here: +/// a machine with no IOMMU, which is a fact about the hardware rather than the size of +/// anything. Running out of *room to record* a confinement is not that, and must refuse. pub fn confineDevice(device_id: u64, bdf: u16, owner: u32) bool { - if (!active) return true; - if (device_id >= confined.len) return true; // unusual id; leave it to fail-open + if (!active) return true; // no IOMMU on this machine — nothing to confine with + // A device id past the end of the record table. This returned `true` — success — + // leaving the device outside every domain while telling the caller it was + // confined, and rolling nothing back. It is unreachable only while device ids stop + // at `confined.len`; moving the inventory out of the kernel and taking the domain + // count from the hardware both change that, and either would have made a silent + // unconfined DMA master out of every device past the 64th. + if (device_id >= confined.len) return false; const domain = domainCreate(owner, bdf) orelse return false; // Firmware reserved region for this device, if any (real hardware; QEMU has none). diff --git a/system/kernel/ipc-synchronous.zig b/system/kernel/ipc-synchronous.zig index 6fe3647..28208bb 100644 --- a/system/kernel/ipc-synchronous.zig +++ b/system/kernel/ipc-synchronous.zig @@ -42,15 +42,22 @@ pub const MESSAGE_MAXIMUM: usize = 256; pub const maximum_handles = scheduler.ipc_maximum_handles; /// Errno-style failures, returned as `-value` in the system_call result register. -pub const EBADF: i64 = 1; // bad handle -pub const E2BIG: i64 = 2; // message exceeds MESSAGE_MAXIMUM -pub const EFAULT: i64 = 3; // buffer unmapped / out of the user half -pub const ENOENT: i64 = 4; // no such name -pub const ENOSPC: i64 = 5; // handle table full -pub const ENOMEM: i64 = 6; // out of memory -pub const EPEER: i64 = 7; // peer died before replying (its process exited or was killed) -pub const ESRCH: i64 = 8; // no such process (process_kill of an unknown/dead id) -pub const EPERM: i64 = 9; // not permitted (process_kill by anyone but the supervisor) +/// Restated from the shared kernel↔user ABI (system/abi.zig), because ring 3 reads +/// the same numbers — the same reason `notify_badge_bit` below is restated. New +/// codes are added *there*, which is where the space is documented. +pub const EBADF = abi.EBADF; // bad handle +pub const E2BIG = abi.E2BIG; // message exceeds MESSAGE_MAXIMUM +pub const EFAULT = abi.EFAULT; // buffer unmapped / out of the user half +pub const ENOENT = abi.ENOENT; // no such name +pub const ENOSPC = abi.ENOSPC; // a kernel table is full +pub const ENOMEM = abi.ENOMEM; // out of memory +pub const EPEER = abi.EPEER; // peer died before replying (its process exited or was killed) +pub const ESRCH = abi.ESRCH; // no such process (process_kill of an unknown/dead id) +pub const EPERM = abi.EPERM; // not permitted (process_kill by anyone but the supervisor) +pub const ENODEV = abi.ENODEV; // no such device id +pub const ECHILDREN = abi.ECHILDREN; // this parent is at its child cap +pub const ERANGE = abi.ERANGE; // a resource escapes its parent's window +pub const ECONFINE = abi.ECONFINE; // the device could not be placed under IOMMU translation /// A badge with this bit set is an asynchronous notification (e.g. an IRQ), not a /// message from a client — there is no reply owed. The low bits carry the source diff --git a/system/kernel/platform.zig b/system/kernel/platform.zig index 2fef984..98b7614 100644 --- a/system/kernel/platform.zig +++ b/system/kernel/platform.zig @@ -26,6 +26,14 @@ pub const RegisterAccess = acpi.RegisterAccess; pub const IsoEntry = acpi.IsoEntry; pub const Cpu = acpi.Cpu; +/// Where a PCI BAR may legitimately live: the holes in the firmware memory map. +/// Firmware-agnostic — it takes a boot-handoff map, not an ACPI table — and pure, so +/// the kernel self-test can drive it with a synthetic map. That is the only way to +/// check the invariant that matters here: an aperture must never cover memory the +/// firmware described, because containment would then admit a BAR over live RAM. +pub const AddressRange = acpi.AddressRange; +pub const largestHolesBelow4G = acpi.largestHolesBelow4G; + /// The FADT power register map discovery extracted (PM1 control, reset register), /// for kernel reboot and diagnostics. Sleep-state values are userspace's (S5 is /// owned by the ring-3 acpi service), so they are not here. diff --git a/system/kernel/process.zig b/system/kernel/process.zig index 20371a9..254abf5 100644 --- a/system/kernel/process.zig +++ b/system/kernel/process.zig @@ -402,36 +402,42 @@ fn systemDeviceEnumerate(state: *architecture.CpuState) void { architecture.setSystemCallResult(state, devices_broker.deviceCount()); } -/// device_claim(id) -> 0/-1: take exclusive ownership of a device for this process. +/// device_claim(id) -> 0/-errno: take exclusive ownership of a device for this process. +/// `-ENODEV` no such id, `-EBUSY` a live task already owns it, `-ECONFINE` the claim +/// could not be placed under IOMMU translation and was rolled back. fn systemDeviceClaim(state: *architecture.CpuState) void { const device_id = architecture.systemCallArg(state, 0); const claim_flags = sync.enter(); defer sync.leave(claim_flags); - if (devices_broker.claim(device_id, scheduler.current().id)) { - // Confine the device's DMA before the driver can program it: a PCI function - // becomes reachable to the IOMMU only once claimed (until now its DMA is - // blocked). A claim that cannot be confined must not stand — roll it back — - // since the whole point is that claiming a DMA device is no longer equivalent - // to ring 0. No-op when no IOMMU exists (fail-open). - if (devices_broker.pciAddressOf(device_id)) |bdf| { - const owner = scheduler.current().id; - if (!iommu.confineDevice(device_id, bdf, owner)) { - _ = devices_broker.unclaim(device_id, owner); - return fail(state); - } - // Bind the buffers this task allocated before claiming the device (a driver - // that dma_alloc'd its rings, then claimed the controller). - dmaBindOwnerRegionsInto(owner, device_id); + devices_broker.claim(device_id, scheduler.current().id) catch |e| + return failErr(state, devices_broker.claimErrnoOf(e)); + + // Confine the device's DMA before the driver can program it: a PCI function + // becomes reachable to the IOMMU only once claimed (until now its DMA is + // blocked). A claim that cannot be confined must not stand — roll it back — + // since the whole point is that claiming a DMA device is no longer equivalent + // to ring 0. No-op when no IOMMU exists (fail-open). + if (devices_broker.pciAddressOf(device_id)) |bdf| { + const owner = scheduler.current().id; + if (!iommu.confineDevice(device_id, bdf, owner)) { + _ = devices_broker.unclaim(device_id, owner); + // Its own errno: "nobody could confine this" is a different world from + // "someone else already has it", and a driver that cannot tell them apart + // cannot report the one that means the machine's DMA protection ran out. + return failErr(state, ipc.ECONFINE); } - // A display service just took the framebuffer — quiesce the bootstrap console - // so the kernel and the service don't scribble over each other's pixels. The - // claim releases (and the console resumes) automatically if the service dies; - // see releaseTaskResourcesLocked. - if (devices_broker.displayDevice()) |display_id| { - if (device_id == display_id) console.setSuppressed(true); - } - architecture.setSystemCallResult(state, 0); - } else fail(state); + // Bind the buffers this task allocated before claiming the device (a driver + // that dma_alloc'd its rings, then claimed the controller). + dmaBindOwnerRegionsInto(owner, device_id); + } + // A display service just took the framebuffer — quiesce the bootstrap console + // so the kernel and the service don't scribble over each other's pixels. The + // claim releases (and the console resumes) automatically if the service dies; + // see releaseTaskResourcesLocked. + if (devices_broker.displayDevice()) |display_id| { + if (device_id == display_id) console.setSuppressed(true); + } + architecture.setSystemCallResult(state, 0); } /// mmio_map(device_id, resource_index) -> virtual_address: map a claimed device's MMIO window into @@ -931,17 +937,21 @@ fn systemDeviceRegister(state: *architecture.CpuState) void { const parent_id = architecture.systemCallArg(state, 0); const descriptor_ptr = architecture.systemCallArg(state, 1); const t = scheduler.current(); - if (t.address_space == 0) return fail(state); + if (t.address_space == 0) return failErr(state, ipc.EPERM); var descriptor: device_abi.DeviceDescriptor = undefined; - if (!ipc.copyFromUser(t.address_space, descriptor_ptr, std.mem.asBytes(&descriptor))) return fail(state); + if (!ipc.copyFromUser(t.address_space, descriptor_ptr, std.mem.asBytes(&descriptor))) return failErr(state, ipc.EFAULT); // Under the big kernel lock: the broker's table is also mutated by the // death sweep (releaseAllOwnedBy) and read by enumerate on other cores — // ring-3 registration (M19) made those genuinely concurrent. const flags = sync.enter(); defer sync.leave(flags); - const id = devices_broker.register(parent_id, t.id, &descriptor) catch return fail(state); + // Each refusal carries its own errno (devices-broker.errnoOf) — the bus driver + // logs which rule stopped it, so "this parent is full" is never again mistaken + // for "the table is full" or "that resource escapes your window". + const id = devices_broker.register(parent_id, t.id, &descriptor) catch |e| + return failErr(state, devices_broker.errnoOf(e)); architecture.setSystemCallResult(state, id); } diff --git a/system/kernel/tests.zig b/system/kernel/tests.zig index d815c89..328390d 100644 --- a/system/kernel/tests.zig +++ b/system/kernel/tests.zig @@ -51,6 +51,15 @@ fn check(name: []const u8, ok: bool) void { } } +/// `devices_broker.claim` reduced to a bool, for the `check` assertions below. The +/// broker returns `ClaimError` so ring 3 can tell "stale id" from "someone already +/// owns it" (the errno space in system/abi.zig); a test that only asserts the claim +/// succeeded does not care which, and the ones that do match the error directly. +fn claimOk(id: u64, owner: u32) bool { + devices_broker.claim(id, owner) catch return false; + return true; +} + /// Emit the overall result line the harness matches, then the done sentinel. fn result() void { log("DANOS-TEST-RESULT: {s} ({d} passed, {d} failed)\n", .{ @@ -250,6 +259,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void { irqFreeTest(); } else if (eql(case, "containment")) { containmentTest(); + } else if (eql(case, "apertures")) { + apertureTest(); } else if (eql(case, "device-manager")) { deviceManagerTest(boot_information); } else if (eql(case, "protocol-registry")) { @@ -1475,7 +1486,7 @@ fn ioPortTest() void { check("discovered the acpi-tables I/O window", true); const me = scheduler.current(); - check("claimed the io_port device", devices_broker.claim(id, me.id)); + check("claimed the io_port device", claimOk(id, me.id)); check("an in-range access resolves to port 0x64", process.resolveIoPort(me, id, found_res, 0x64, 1) == 0x64); check("a 4-byte access at the last port is refused", process.resolveIoPort(me, id, found_res, 0xFFFF, 4) == null); check("an out-of-range offset is refused", process.resolveIoPort(me, id, found_res, 0x10000, 1) == null); @@ -2470,8 +2481,8 @@ fn claimReleaseTest(boot_information: *const BootInformation) void { } // The broker release in isolation. - check("device 0 claimed by owner 111", devices_broker.claim(0, 111)); - check("device 1 claimed by owner 222", devices_broker.claim(1, 222)); + check("device 0 claimed by owner 111", claimOk(0, 111)); + check("device 1 claimed by owner 222", claimOk(1, 222)); devices_broker.releaseAllOwnedBy(111); check("owner 111's claim is released", devices_broker.ownerOf(0) == null); check("owner 222's claim survives", (devices_broker.ownerOf(1) orelse 0) == 222); @@ -2493,7 +2504,7 @@ fn claimReleaseTest(boot_information: *const BootInformation) void { }; const child = process.spawnProcessSupervised(image, 4, &.{"/system/services/init"}, me, endpoint) catch 0; check("supervised child spawned", child != 0); - check("device 0 claimed on the child's behalf", devices_broker.claim(0, child)); + check("device 0 claimed on the child's behalf", claimOk(0, child)); check("the kill is accepted", process.killProcess(me, child) == 0); var badge: u64 = 0; @@ -2501,7 +2512,7 @@ fn claimReleaseTest(boot_information: *const BootInformation) void { _ = ipcsync.replyWait(endpoint, 0, 0, 0, 0, abi.no_cap, &badge, &received_cap); check("the exit notification arrived", badge == abi.notify_badge_bit | abi.notify_exit_bit | child); check("death released the child's claim", devices_broker.ownerOf(0) == null); - check("the device is claimable again", devices_broker.claim(0, me)); + check("the device is claimable again", claimOk(0, me)); devices_broker.releaseAllOwnedBy(me); result(); } @@ -3765,8 +3776,8 @@ fn killThreadedGroupTest(boot_information: *const BootInformation) void { // Claims for BOTH members: group death must release every member's claims // before the supervisor hears anything — the worker's by the deferred // (condemned) path. - check("device 0 claimed for the leader", devices_broker.claim(0, child)); - check("device 1 claimed for the worker", devices_broker.claim(1, worker)); + check("device 0 claimed for the leader", claimOk(0, child)); + check("device 1 claimed for the worker", claimOk(1, worker)); scheduler.sleep(100); // let the worker really be running on another core check("the supervisor's kill is accepted", process.killProcess(me, child) == 0); const badge = awaitExitBadge(endpoint); @@ -3954,7 +3965,7 @@ fn containmentTest() void { result(); return; } - check("claimed the parent device", devices_broker.claim(parent_id, me)); + check("claimed the parent device", claimOk(parent_id, me)); defer devices_broker.releaseAllOwnedBy(me); // A child whose window lies inside the parent's is accepted. @@ -3976,10 +3987,94 @@ fn containmentTest() void { check("re-registering an identical child returns the same id", again != 0 and again == good); check("re-registering grew nothing", devices_broker.enumerate(&buffer) == before + 1); + // Fill the parent to its child cap with distinct children (same window, different + // identity — the match is on identity, so each is a new device). + var filled: u32 = 0; + var capped = false; + while (filled < 64) : (filled += 1) { + var name: [4]u8 = .{ 'k', 0, 0, 0 }; + name[1] = '0' + @as(u8, @intCast(filled / 10)); + name[2] = '0' + @as(u8, @intCast(filled % 10)); + var extra = childDescriptor(name[0..3], parent_window.start, 0x20); + _ = devices_broker.register(parent_id, me, &extra) catch |err| { + capped = err == error.TooManyChildren; + break; + }; + } + check("the parent reaches its child cap (TooManyChildren)", capped); + + // The regression this ordering exists for: **a re-registration consumes no slot, + // so a full parent must not refuse one.** A crashed bus driver is restarted by its + // supervisor and re-registers everything it rediscovers; when the cap was checked + // before the identity match, the restart was refused its own devices and the + // machine degraded a little more on every crash. + const readmitted = devices_broker.register(parent_id, me, &fits) catch 0; + check("a full parent still re-admits an identical child", readmitted != 0 and readmitted == good); + + // ...and the cap is genuinely still in force for anything new. + var novel = childDescriptor("knew", parent_window.start, 0x20); + const still_capped = if (devices_broker.register(parent_id, me, &novel)) |_| false else |err| err == error.TooManyChildren; + check("a full parent still refuses a new child", still_capped); + result(); } /// A minimal child descriptor with one memory resource, for the containment test. +/// PCI host-bridge apertures are derived from the *holes* in the firmware memory map, +/// and a registered BAR must fall inside one. So the invariant is not "we find the +/// holes" but "an aperture never covers memory the firmware described" — an aperture +/// over RAM means `device_register` containment admits a child BAR over kernel memory, +/// and its claimant can `mmio_map` it. +/// +/// The map's length is the firmware's choice: 60–200 descriptors on a real machine, +/// 15–25 under OVMF, which is why the suite never saw this. The derivation used to +/// copy sub-4 GiB entries into a fixed `[64]` array and skip the rest — and a skipped +/// region is not merely lost, it is one the gap finder concludes is *free*. +fn apertureTest() void { + // 100 described one-page regions, 2 MiB apart: 99 small holes between them, then + // one large hole from the last region up to 4 GiB. Under the old fixed array the + // 36 regions past the 64th vanished, so the "largest hole" ran from ~128 MiB to + // 4 GiB — straight across 36 regions the firmware had described. + const spacing: u64 = 2 << 20; + var regions: [100]boot_handoff.MemoryRegion = undefined; + for (®ions, 0..) |*region, i| { + region.* = .{ .base = @as(u64, i) * spacing, .pages = 1, .kind = .usable }; + } + + var holes: [3]platform.AddressRange = undefined; + const found = platform.largestHolesBelow4G(®ions, 1 << 20, &holes); + check("apertures were derived from a 100-entry map", found > 0); + + var overlaps: usize = 0; + var largest: platform.AddressRange = .{ .base = 0, .end = 0 }; + for (holes) |hole| { + if (hole.end <= hole.base) continue; + if (hole.end - hole.base > largest.end - largest.base) largest = hole; + for (regions) |region| { + const region_end = region.base + region.pages * 4096; + if (region.base < hole.end and region_end > hole.base) overlaps += 1; + } + } + check("no aperture overlaps described memory", overlaps == 0); + + // The big hole is above the last described region, not across it. + const last_end = (regions.len - 1) * spacing + 4096; + check("the largest aperture starts after the last described region", largest.base == last_end); + check("the largest aperture runs to 4 GiB", largest.end == (1 << 32)); + + // A map the firmware describes nothing in is one whole hole; a map that describes + // everything has none. Neither may invent an aperture over something described. + var empty: [3]platform.AddressRange = undefined; + check("an empty map yields one hole", platform.largestHolesBelow4G(&.{}, 1 << 20, &empty) == 1); + check("that hole is the whole low space", empty[0].base == 0 and empty[0].end == (1 << 32)); + + const whole = [_]boot_handoff.MemoryRegion{.{ .base = 0, .pages = (1 << 32) / 4096, .kind = .usable }}; + var none: [3]platform.AddressRange = undefined; + check("a fully described map yields no aperture", platform.largestHolesBelow4G(&whole, 1 << 20, &none) == 0); + + result(); +} + fn childDescriptor(hid: []const u8, start: u64, len: u64) device_abi.DeviceDescriptor { var child = std.mem.zeroes(device_abi.DeviceDescriptor); child.class = @intFromEnum(device_abi.DeviceClass.unknown); diff --git a/system/parameters.zig b/system/parameters.zig index 1b6d8c3..a055bb0 100644 --- a/system/parameters.zig +++ b/system/parameters.zig @@ -4,8 +4,14 @@ //! hiding the trade-offs. Keeping them here makes them visible at a glance and gives //! one spot to change them. They're plain `comptime` constants (zero runtime cost); //! any one can later be promoted to a `-D` build option if a target needs to vary it -//! (see build.zig's `-Dtest-case` for the pattern). This keeps ps2-library.zig to what it -//! actually is — the bootloader↔kernel handoff *contract* — with tunables living here. +//! (see build.zig's `-Dtest-case` for the pattern). This keeps [[boot-handoff]] to what +//! it actually is — the loader↔kernel handoff *contract* — with tunables living here. +//! +//! **Kernel only, and deliberately so.** This file exists because tunables were +//! crowding the loader↔kernel contract they were split out of; it is not a registry for +//! the whole system. A driver's ring size belongs to that driver, a protocol's payload +//! cap to that protocol. How a ceiling is *declared*, wherever it lives, is +//! docs/os-development/bounds.md — a shape, not a shared list. /// Ceiling on logical CPUs the kernel tracks — the size of the per-CPU bookkeeping /// arrays (discovery pool, scheduler state, per-core GDT/TSS). Generous headroom: diff --git a/system/services/acpi/acpi.zig b/system/services/acpi/acpi.zig index b29a7c2..9c0545b 100644 --- a/system/services/acpi/acpi.zig +++ b/system/services/acpi/acpi.zig @@ -119,10 +119,10 @@ pub fn main(init: process.Init) void { return; }; node_id = node.id; - if (!device.claim(node_id)) { - _ = logging.write("/system/services/acpi: unable to claim acpi-tables\n"); + device.claim(node_id) catch |e| { + std.log.warn("unable to claim acpi-tables: {s}", .{@errorName(e)}); return; - } + }; // Map the node's resources: the AML blobs (bytecode), the FADT (intact // "FACP" header — decision 3), the io_port grant, and the SCI irq. @@ -519,8 +519,8 @@ fn registerDevice(node: *aml.Node, hid: [8]u8, interpreter: *aml.Interpreter) vo @memcpy(descriptor.hid[0..@intCast(hid_len)], hid[0..@intCast(hid_len)]); applyCrs(&descriptor, node, interpreter); - const id = device.register(node_id, &descriptor) orelse { - std.log.info("register refused for {s}", .{hid[0..@intCast(hid_len)]}); + const id = device.register(node_id, &descriptor) catch |e| { + std.log.warn("register refused for {s}: {s}", .{ hid[0..@intCast(hid_len)], @errorName(e) }); return; }; registered[registered_count] = .{ .hid = hid, .hid_len = @intCast(hid_len), .device_id = id, .resource_count = descriptor.resource_count }; diff --git a/system/services/display/backend.zig b/system/services/display/backend.zig index 182fea5..2ebeeab 100644 --- a/system/services/display/backend.zig +++ b/system/services/display/backend.zig @@ -74,10 +74,12 @@ pub const Gop = struct { return null; }; - if (!device.claim(found.id)) { - _ = logging.write("display: could not claim the framebuffer\n"); + device.claim(found.id) catch |e| { + var line: [96]u8 = undefined; + _ = logging.write(std.fmt.bufPrint(&line, "display: could not claim the framebuffer: {s}\n", .{@errorName(e)}) catch + "display: could not claim the framebuffer\n"); return null; - } + }; // Resource 0 is the framebuffer memory window; the kernel maps it write-combining // because the resource carries that flag (docs/display-plan.md D1). const front_base = device.mmioMap(found.id, 0) orelse { diff --git a/test/qemu_test.py b/test/qemu_test.py index 936fe92..8bcccc5 100644 --- a/test/qemu_test.py +++ b/test/qemu_test.py @@ -978,9 +978,20 @@ CASES = [ # device_register containment (in-kernel): registering a child whose MMIO window # escapes its parent's grant is refused (NotContained) — else dev_register would map # arbitrary physical memory — while an identical re-register stays idempotent. + # Also pins the cap ordering: a parent already at maximum_children_per_parent must + # still re-admit an identical child (a restarted bus consumes no slot re-reporting + # what it rediscovers) while still refusing a genuinely new one. {"name": "containment", "expect": r"DANOS-TEST-RESULT: PASS", "fail": r"DANOS-TEST-RESULT: FAIL"}, + # PCI host-bridge apertures come from the holes in the firmware memory map, and a + # registered BAR must fall inside one. The invariant is that an aperture never + # covers memory the firmware described - otherwise device_register containment + # admits a child BAR over live RAM. Driven with a synthetic 100-entry map, since + # OVMF only ever produces 15-25 and real firmware 60-200. + {"name": "apertures", + "expect": r"DANOS-TEST-RESULT: PASS", + "fail": r"DANOS-TEST-RESULT: FAIL"}, # IRQ teardown: an exiting driver's line is masked and its slot cleared (so no # ISR notifies a freed endpoint), and a sibling owner sharing that endpoint # keeps its own binding. A long-running driver never reaches this teardown path. diff --git a/test/system/services/crash-test/crash-test.zig b/test/system/services/crash-test/crash-test.zig index 8107a2e..b5c262a 100644 --- a/test/system/services/crash-test/crash-test.zig +++ b/test/system/services/crash-test/crash-test.zig @@ -22,10 +22,10 @@ pub fn main(init: process.Init) void { // The respawn only reaches this line because the kernel released the // previous instance's claim at death. A failed claim exits cleanly — the // manager reads "meant to stop" and the scenario fails loudly by silence. - if (!device.claim(assigned)) { + device.claim(assigned) catch { _ = logging.write("crash-test: claim failed\n"); return; - } + }; var manager: ?ipc.Handle = null; var tries: u32 = 0; diff --git a/test/system/services/iommu-fault-test/iommu-fault-test.zig b/test/system/services/iommu-fault-test/iommu-fault-test.zig index 44b183e..fd0d6de 100644 --- a/test/system/services/iommu-fault-test/iommu-fault-test.zig +++ b/test/system/services/iommu-fault-test/iommu-fault-test.zig @@ -71,10 +71,10 @@ pub fn main() void { return; }; - if (!device.claim(nic_id)) { + device.claim(nic_id) catch { _ = logging.write("iommu-fault-test: FAIL claim\n"); return; - } + }; var function = pci.Function.map(nic_id, &descriptor) orelse { _ = logging.write("iommu-fault-test: FAIL config-space map\n"); return; diff --git a/test/system/services/pci-cap-test/pci-cap-test.zig b/test/system/services/pci-cap-test/pci-cap-test.zig index f4ae39e..1b9f644 100644 --- a/test/system/services/pci-cap-test/pci-cap-test.zig +++ b/test/system/services/pci-cap-test/pci-cap-test.zig @@ -64,7 +64,7 @@ pub fn main() void { return; }; writeLine("pci-cap-test: claiming ethernet function (device {d})\n", .{nic_id}); - if (!check("claim", device.claim(nic_id))) return; + if (!check("claim", if (device.claim(nic_id)) |_| true else |_| false)) return; var function = pci.Function.map(nic_id, &descriptor) orelse { _ = logging.write("pci-cap-test: FAIL config-space map\n"); return;