diff --git a/build.zig b/build.zig index 88406b9..63fec3f 100644 --- a/build.zig +++ b/build.zig @@ -132,8 +132,8 @@ pub fn build(b: *std.Build) void { // ACPI/PnP hardware-ID (_HID) names — the flat analog of pci-class for acpi_device // nodes. Also shared reference data. // The AML interpreter, a build module so the ring-3 acpi service can run the - // same parser the kernel does (docs/m19-m20-plan.md decision 1). Pure Zig, - // no kernel imports — one source, two builds. + // same parser the kernel does (docs/discovery.md — the shared AML module). + // Pure Zig, no kernel imports — one source, two builds. const aml_module = b.addModule("aml", .{ .root_source_file = b.path("system/devices/aml/aml.zig"), }); @@ -226,7 +226,7 @@ pub fn build(b: *std.Build) void { }); runtime_module.addImport("device-manager-protocol", device_manager_protocol_module); - // The power protocol: system power's domain-named surface (docs/m21-plan.md). + // The power protocol: system power's domain-named surface (docs/power.md). const power_protocol_module = b.addModule("power-protocol", .{ .root_source_file = b.path("system/services/power/protocol.zig"), }); @@ -346,8 +346,6 @@ pub fn build(b: *std.Build) void { // which unpacks it and spawns each program (system/initial-ramdisk.zig). const vfs_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "vfs", "system/services/vfs/vfs.zig"); const vfstest_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "vfs-test", "system/services/vfs/vfs-test.zig"); - const hpet_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "hpet", "system/drivers/hpet/hpet.zig"); - const bus_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "bus", "system/drivers/bus/bus.zig"); const ps2_bus_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-bus", "system/drivers/ps2-bus/ps2-bus.zig"); const ps2_keyboard_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-keyboard", "system/drivers/ps2-bus/keyboard.zig"); const ps2_mouse_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-mouse", "system/drivers/ps2-bus/mouse.zig"); @@ -361,7 +359,7 @@ pub fn build(b: *std.Build) void { const crash_test_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "crash-test", "system/services/crash-test/crash-test.zig"); const device_list_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "device-list", "system/services/device-list/device-list.zig"); // The discovery service: one swappable process per firmware - // (docs/m19-m20-plan.md decision 7), bundled under the neutral ramdisk name + // (docs/discovery.md), bundled under the neutral ramdisk name // "discovery" so the device manager never learns which firmware it is on. // x86 boots describe hardware with ACPI; the Raspberry Pis hand over a // flattened device tree — the aarch64 target flips the default when it @@ -396,10 +394,6 @@ pub fn build(b: *std.Build) void { mk_run.addFileArg(vfs_exe.getEmittedBin()); mk_run.addArg("vfs-test"); mk_run.addFileArg(vfstest_exe.getEmittedBin()); - mk_run.addArg("hpet"); - mk_run.addFileArg(hpet_exe.getEmittedBin()); - mk_run.addArg("bus"); - mk_run.addFileArg(bus_exe.getEmittedBin()); mk_run.addArg("ps2-bus"); mk_run.addFileArg(ps2_bus_exe.getEmittedBin()); mk_run.addArg("ps2-keyboard"); @@ -435,8 +429,6 @@ pub fn build(b: *std.Build) void { .{ vfs_exe, "system/services" }, .{ device_manager_exe, "system/services" }, .{ input_exe, "system/services" }, - .{ hpet_exe, "system/drivers" }, - .{ bus_exe, "system/drivers" }, .{ ps2_bus_exe, "system/drivers" }, .{ ps2_keyboard_exe, "system/drivers" }, .{ ps2_mouse_exe, "system/drivers" }, @@ -615,6 +607,21 @@ pub fn build(b: *std.Build) void { }); test_step.dependOn(&b.addRunArtifact(xkb_tests).step); + // runtime.time's Instant/Duration arithmetic. time.zig pulls in system.zig (the + // syscall wrappers), which needs the `abi` module, so it doesn't fit the plain + // loop above. + const time_tests = b.addTest(.{ + .root_module = b.createModule(.{ + .root_source_file = b.path("library/runtime/time.zig"), + .target = target, + .optimize = optimize, + .imports = &.{ + .{ .name = "abi", .module = abi_module }, + }, + }), + }); + test_step.dependOn(&b.addRunArtifact(time_tests).step); + // Convenience: `zig build gen-xkeyboard-config` regenerates the layout tables from the // vendored data (offline). `fetch` (the network step) stays a manual script run. const gen_xkb = b.addSystemCommand(&.{ "python3", "tools/make-xkeyboard-config.py", "generate" }); diff --git a/docs/README.md b/docs/README.md index c3e5631..843ec29 100644 --- a/docs/README.md +++ b/docs/README.md @@ -100,7 +100,16 @@ Cutting across all of these: when to build it, and how to keep it architecture-agnostic. - **[acpi.md](acpi.md) — finding the ACPI tables.** The concrete x86 locator chain: how the loader captures the **RSDP**, hands its physical address across in `BootInfo`, - and how the platform derives the **RSDT/XSDT** from it and walks the SDTs. + and how the platform derives the **RSDT/XSDT** from it and walks the SDTs — plus the + live event side (the SCI, the power button, GPE/Notify) the ring-3 acpi service runs. +- **[power.md](power.md) — the power service.** System power as a domain-named + service: button/lid/battery events published to subscribers, and init's orderly + shutdown composing the [lifecycle](process-lifecycle.md) stop sequence with an ACPI + S5 write. Firmware-neutral — a PSCI backend drops in on ARM. +- **[timers.md](timers.md) — timers and time.** The ring-3 surface for reading the + clock and waiting: why `now()` is a syscall rather than a service, and the one-shot + timer notification (`timer_bind`) that gives supervisors a timed wait — built on the + LAPIC heartbeat and calibrated TSC of [device-interrupts.md](device-interrupts.md). - **[smp.md](smp.md) — multiple cores.** A design/research note on how microkernels (L4, seL4) handle SMP — big kernel lock vs per-CPU vs multikernel — and how the right choice depends on whether danos is chasing real-time or resilience. diff --git a/docs/acpi.md b/docs/acpi.md index 94ed6a3..76c5adf 100644 --- a/docs/acpi.md +++ b/docs/acpi.md @@ -107,12 +107,66 @@ firmware-agnostic [device model](discovery.md) gets populated; this note stops a part that answers "where are the tables?" — everything past the RSDP is just following more pointers the tables themselves provide. +## ACPI events: the SCI, the power button, and GPEs (M21) + +The tables above are static description; ACPI is also a *live* channel. Hardware +raises the **SCI** (System Control Interrupt) — one shared, level-triggered line +whose vector the FADT names — and the OS reads status registers to learn what +happened: a fixed event like the power button, or a **General-Purpose Event** +(GPE) whose handler is an AML method. Since [discovery](discovery.md) moved AML +to ring 3, the event side lives there too, in the same **acpi service** — the +device discoverer and the event source are one process, because both need the +namespace and the port grant. + +**The kernel hands the service what it needs and no more.** Reading PM1 event +blocks and GPE blocks requires the FADT, which the kernel already parses for its +own `\_S5` poweroff. Rather than re-parse, the kernel appends the **FADT as one +more memory resource** on the `acpi-tables` node; the service tells it apart +from the AML blob resources by signature — the FADT keeps its intact `"FACP"` +header, while the blob resources are header-stripped bytecode that starts with +no signature. The kernel's own FADT parse is untouched; the service reads the +PM1 *event* blocks (which the kernel never parsed — it only needs PM1 *control* +for `\_S5`) and the GPE0/GPE1 blocks straight from its copy. The **SCI itself** +arrives as the node's one `len == 1` irq resource (distinct from the broad +`[0, 256)` window that covers children's legacy lines), which is how the service +finds the line to `irq_bind`. + +With those in hand the service enables ACPI mode (only if `SCI_EN` is clear — +some firmwares boot with it already set), sets `PWRBTN_EN`, and on each SCI: + +- **The power button** is a *fixed* event: a set `PWRBTN_STS` bit in PM1 status. + The handler clears it (write-1-to-clear), logs the press, and publishes a + [`power`](power.md) `power_button` event to subscribers. +- **GPEs** are the general path: for each set-and-enabled GPE bit `n`, the + service evaluates its `\_GPE._L%02X` (level) or `_E%02X` (edge) handler + method, drains the **Notify** queue that method produced, maps each notified + device to an event (battery, AC, lid, or a generic `notify` with its code), + and clears the status bit. A missing handler method is clear-and-log, not an + error. Making GPEs work required teaching the interpreter one opcode it never + handled — `Notify` (`0x86`) — which it now folds into a bounded queue drained + per evaluation; everything else a handler needs (field access, control flow, + method calls) was already proven by the ring-3 `_STA`/`_CRS` work. + +**How this is tested.** QEMU cannot raise GPEs deterministically on this config, +so GPE/Notify correctness is proven by **host unit tests** — hand-encoded AML +with a `Notify` inside a method body, run under `zig build test`. The QEMU +`power-button` scenario proves the fixed-event path end to end: a QMP +`system_powerdown` injects a real ACPI power-button press, and the service's SCI +handler must log it. Battery/AC/lid and the embedded controller's `_Qxx` queries +are interface-complete but validated on real hardware later. + +The service surface these events are *published on* — subscription, the event +vocabulary, and orderly shutdown — is the power service, [power.md](power.md). + ## Related - [efi.md](efi.md) — the loader that captures the RSDP before `ExitBootServices`. - [memory-map.md](memory-map.md) — the same loader-captures / kernel-consumes seam, and the ACPI-reclaim memory the RSDP lives in. - [discovery.md](discovery.md) — the broader (still-evolving) plan for turning these - tables into one neutral device model shared with the ARM device-tree path. + tables into one neutral device model shared with the ARM device-tree path, and how + ACPI enumeration and events moved to the ring-3 acpi service. +- [power.md](power.md) — the domain-named power service the ACPI event side publishes + to (button, lid, battery) and its orderly-shutdown path into S5. - [arch.md](arch.md) — why the kernel reaches the device code through a `platform` module and never names ACPI directly. diff --git a/docs/coding-standards.md b/docs/coding-standards.md index 842c08b..8ed8dda 100644 --- a/docs/coding-standards.md +++ b/docs/coding-standards.md @@ -96,7 +96,7 @@ That's all — no Unix-abbreviation exception. The source directories are full w (`system`, `library`, not `src`/`lib`), and there is no daemon `d` suffix: a driver lives in `system/drivers/` and a service in `system/services/`, so the *location* already says what it is. Encoding the role in the name too (`busd`, `vfsd`) is -redundant — the program is just `bus`, `vfs`. Don't put in a name what its directory +redundant — the program is just `ps2-bus`, `vfs`. Don't put in a name what its directory already tells you. ## A note on collisions @@ -138,7 +138,7 @@ single word or acronym needs no hyphen: `scheduler.zig`, `paging.zig`, `apic.zig conventions above — `snake_case` — because it's an identifier, not a filename.) **A sub-project's entry point repeats its directory's name** — `init/init.zig`, -`runtime/runtime.zig`, `hpet/hpet.zig` — and the sub-project is addressed by the +`runtime/runtime.zig`, `ps2-bus/ps2-bus.zig` — and the sub-project is addressed by the *directory* (`system/services/init`, `library/runtime`), with the repeated leaf resolving away. See the repository-layout section of [README.md](README.md). diff --git a/docs/danos-file-system-hierarchy-FSH.md b/docs/danos-file-system-hierarchy-FSH.md index 798e371..c79bba9 100644 --- a/docs/danos-file-system-hierarchy-FSH.md +++ b/docs/danos-file-system-hierarchy-FSH.md @@ -17,7 +17,7 @@ Most modern Unix and Unix-like operating systems follow the FHS. DanOS has its o | /srv | Site-specific data served by this system, such as data and scripts for web servers, data offered by FTP servers, and repositories for version control systems | | /system | DanOS operating system files (similar idea to C:\Windows). A true representation of danos — its layout mirrors the source tree, so `/system` is what danos *is*. | | /system/devices | danos virtual device tree e.g. similar to /sys on linux but with danos device tree conventions (the structures in the devices module) | -| /system/drivers | driver binaries, one sub-project each (e.g. /system/drivers/hpet) | +| /system/drivers | driver binaries, one sub-project each (e.g. /system/drivers/pci-bus, /system/drivers/ps2-bus) | | /system/services | system-service binaries — the VFS server, init, and other user-mode servers (e.g. /system/services/vfs, /system/services/init) | | /system/kernel | the kernel image | | /tmp | Directory for temporary files (see also /var/tmp). Often not preserved between system reboots and may be severely size-restricted. | @@ -61,8 +61,8 @@ to the driver in the order written, and a read consumes what is there. Terminals serial lines, keyboards and mice are all of this shape. These are the natural first device nodes in danos, because a character driver needs nothing the kernel doesn't already provide — it claims its device, maps its registers with `mmio_map`, and blocks -on `replyWait` for either an interrupt or a client request. `system/drivers/hpet/hpet.zig` is already -that program, minus the client half. +on `replyWait` for either an interrupt or a client request. `system/drivers/ps2-bus/ps2-bus.zig` +is already that program, minus the file-node client half. The obstacle was never the file type; it is which hardware a ring-3 driver can reach. Direct `in`/`out` from user space is still a #GP (no TSS I/O bitmap, IOPL never raised), diff --git a/docs/device-interrupts.md b/docs/device-interrupts.md index fe71a7b..874d63e 100644 --- a/docs/device-interrupts.md +++ b/docs/device-interrupts.md @@ -78,6 +78,40 @@ preemption and wakeups (1 ms granularity); the **TSC** is the resolution you rea time at. Making `sleep` itself sub-millisecond would take a tickless one-shot timer — a later step. +### Is the TSC trustworthy? Invariant, and synchronized + +A cycle counter is only a valid *clock* if two things hold, and danos checks both, +because they decide whether we read time with a cheap `rdtsc` or fall back to the HPET. + +**Invariant.** An old TSC counted core clock cycles, so it sped up and slowed down with +frequency scaling — useless as wall time. Modern CPUs (all of danos's targets) provide an +**invariant TSC**: a constant rate across P/C-states that never stops. The guarantee is a +CPUID bit — leaf `0x80000007`, EDX bit 8 — on both Intel *and* AMD. danos reads it in +`calibrate`, and a TSC that doesn't advertise it is not used as the clocksource. AMD is +why this matters in practice: it doesn't populate the Intel leaf `0x15` that enumerates +the TSC *frequency*, so danos already measures AMD's rate against the HPET — but a +measured frequency without the invariance guarantee is not enough. + +**Synchronized.** Each core has its own TSC. Even invariant ones can start at different +values (a second socket, some firmware), so a thread migrating from a core reading +`1_000_000` to one reading `999_000` would see time jump *backward*. danos runs a **warp +check** as each application processor comes online (`checkWarpSource`, adapted from +Linux's): the waking core and the BSP hammer a shared "highest seen" TSC under a lock, +and if either ever reads below it, the cores' TSCs are skewed. It's pairwise because APs +come up one at a time ([smp.md](smp.md)). + +**The fallback.** When the TSC fails either test — non-invariant (a bare VM such as the +default qemu64), or warped between cores — danos moves the monotonic clock onto the +**HPET** main counter: one fixed-rate counter, so it can neither skew between cores nor +drift with frequency. It costs a memory-mapped read instead of a register read, but it +keeps time *accurate*, which is the whole point. The switch preserves the current value, +so the clock never jumps. The boot log names the outcome: + +``` +/system/kernel: clocksource tsc (TSC invariant: yes, synchronized: yes) # real Intel/AMD +/system/kernel: clocksource hpet (TSC invariant: no, synchronized: yes) # a bare VM (TCG) +``` + ## Two kinds of vector, one dispatch The IDT now installs gates `0-47`: the 32 exceptions plus the device range. Every diff --git a/docs/device-manager.md b/docs/device-manager.md index 2b50bc9..70ace1c 100644 --- a/docs/device-manager.md +++ b/docs/device-manager.md @@ -51,7 +51,16 @@ enumeration is a **pci-bus driver**: the manager spawns it against the host brid like any bus reports children. ACPI becomes an **acpi service** that interprets the tables and reports the namespace. The manager only orchestrates and merges. Moving AML interpretation out of ring 0 is its own project on its own track; nothing here -depends on when it lands. +depends on when it lands. (It landed: [discovery.md](discovery.md), M19–M20.) + +`device_register` is **idempotent on exact match**: a re-registration with an +identical (parent, class, identity, resources) tuple returns the existing id +instead of appending a duplicate. The kernel table has no unregister, so without +this a restarted registering bus would re-report its children as fresh nodes on +every respawn. Idempotence is what makes restart-and-re-report sound for *every* +reporting bus — pci-bus, the acpi service, a future fdt service — not just one, +and it is why supervision (below) can prune a dead bus's subtree and trust the +restarted instance to rebuild exactly the same ids. ## The protocol @@ -136,10 +145,17 @@ published exit events, signals + `runtime.process`). On top of those: the mouse and keyboard QEMU already hangs off it. 7. **App surface**: `enumerate`/`subscribe` over IPC; `device_enumerate` retreats to a manager-internal seam. -8. **Discovery migration** — DONE (M19–M20, 2026-07-13): pci-bus driver (M19) - then the acpi service (M20) moved enumeration to ring 3; the kernel seeds - only the host bridge and the acpi-tables node. See - [m19-m20-plan.md](m19-m20-plan.md). +8. **Discovery migration** — DONE (M19–M20, 2026-07-13): enumeration moved to + ring 3 as swappable per-firmware discoverers — the pci-bus driver (M19) then + the acpi service (M20), see [discovery.md](discovery.md); the kernel seeds + only the host bridge and the acpi-tables node. Matching moved with it: + `child_added` grew a `device_id` (the kernel-registered id, `no_device` for + unregistered leaves like USB ports) and a firmware `hid`, and the manager now + matches drivers from those **reports** rather than its boot-time snapshot. The + PCI arm flipped in M19.3, the ACPI arm (ps2-bus matched from `_HID`) in M20.3 + — each in a single phase so no device is ever matched from both sources at + once. The acpi service reports only the non-PCI `_HID` devices, since pci-bus + already reports PCI functions (M20.2). ## Settled questions (2026-07-12) diff --git a/docs/discovery.md b/docs/discovery.md index 94a0579..0f1298d 100644 --- a/docs/discovery.md +++ b/docs/discovery.md @@ -191,3 +191,57 @@ and registers + reports each `_HID` device — the device manager matches driver (ps2-bus) from those reports. With M19's pci-bus driver, discovery now runs entirely in user space; the kernel seeds only the host bridge and the acpi-tables node. + +## Discovery is a swappable process per firmware (M19–M20) + +Moving PCI and ACPI enumeration out of ring 0 was not just a relocation — it +made discovery **firmware-neutral by construction**, which is the whole reason +to do it before the second architecture rather than after. Everything at and +above the [device-manager](device-manager.md) protocol — descriptors, +containment, reports, matching, supervision — is generic and may never become +x86-specific. Discovery is the single firmware-specific piece, and it is +isolated as **one swappable process per firmware**: + +- **x86** boots describe hardware with ACPI, so the discoverer is the **acpi + service** ([acpi.md](acpi.md)): it claims the `acpi-tables` node and runs AML. +- **The Raspberry Pis** hand over a flattened device tree, so the discoverer is + an **fdt service**: it claims a `devicetree-blob` node and walks the tree — + pure data, no bytecode, so it needs neither a port grant nor an interpreter, + strictly simpler than ACPI. (A placeholder until the [aarch64](arm.md) + bring-up fills it in.) + +The device manager spawns the discoverer under the **neutral ramdisk name +`discovery`** and never learns which firmware it is on; the build's +`-Ddiscovery=acpi|fdt` option fills that slot (x86 defaults to `acpi`, the +aarch64 target flips the default when it lands). The manager owns the device +tree as *data* and touches no hardware, ever — firmware bytecode runs only +inside the crashable, supervised discoverer, so an AML fault can never take +down the supervisor. + +Two consequences of neutrality bind on later work: + +- **Cross-firmware surfaces are named by domain, not firmware.** System power is + a [`power`](power.md) protocol, not an "ACPI events" protocol: on x86 the acpi + service registers it, on ARM a PSCI/mailbox service registers the same + `ServiceId.power`, and subscribers never learn the difference. +- **Identity must widen before the fdt service exists.** `DeviceDescriptor`'s + 8-byte `hid` holds an EISA id but cannot hold an FDT `compatible` string + (`"brcm,bcm2835-aux-uart"`); the identity field grows before the ARM path can + report a real node. + +Two supporting decisions keep the kernel's remaining slice honest: + +- **The AML interpreter is a shared build module**, compiled into both the + kernel and the acpi service — one source, two builds, no fork. The kernel + links it for the `\_S5` poweroff evaluation, the service links it for + everything else, and the `acpi-parse` test asserts the two produce the same + device count across the ring-3 move. +- **Bridge apertures come from the firmware memory map, not AML.** Registered + PCI functions carry BAR resources, and `device_register` containment demands + the bridge own windows that cover them. Those apertures are derived + kernel-side from the boot memory map's MMIO holes (regions that are neither + RAM nor tables) — mechanical, AML-free, and available at boot regardless of + what later moved to user space. The acpi service's authority is likewise + exactly one node: the `acpi-tables` node, whose broad io_port grant is the + documented trust boundary for the one process allowed to run firmware + bytecode. diff --git a/docs/driver-model.md b/docs/driver-model.md index 6b37e00..9016e1b 100644 --- a/docs/driver-model.md +++ b/docs/driver-model.md @@ -57,8 +57,9 @@ is not an address window. Discovery is trusted; user space is not. ### What a bus driver looks like -`system/drivers/bus/bus.zig` is the smallest honest one. Its "bus" is the HPET's register block and -its "devices" are the block's comparators: +danos ships no demo bus driver — the real ones are `pci-bus`, `ps2-bus`, and +`usb-xhci-bus`. The smallest *honest* shape, illustrated here with an HPET register block +as the "bus" and its comparators as the "devices", is: ```zig _ = dev.claim(bus.id); // 1. own the bus @@ -78,8 +79,8 @@ for (0..n) |i| { // 3. publish each child Each child is left **unclaimed**, which is the handoff: a comparator driver can now `device_claim` one and `mmio_map` it, and will see only its own 0x20-byte window. A child -whose window escapes the bus is refused — `bus` asserts that, and the `bus` test -asserts the kernel's table upholds it. +whose window escapes the bus is refused; the in-kernel `containment` test asserts the +kernel's table upholds that ([drivers.md](drivers.md)). A USB device has *no* resources at all: `resource_count = 0`, because it's addressed through its controller, not by MMIO. That case is allowed and is the common one. @@ -143,7 +144,7 @@ If a class driver needs `mmio`, it has become an HCD and should be one. physically-contiguous, pinned, uncacheable, reclaim-on-teardown buffers with the physical address exposed (`pmm.allocContiguous`, a DMA arena, `mapUserDmaInto`). `dma_below_4g` caps the address for legacy engines; `dma_write_combining` is accepted - but falls back to coherent until PAT is programmed. hpet is refactored onto `/lib/mmio`; + but falls back to coherent until PAT is programmed. The bus drivers use `/lib/mmio`; no DMA driver consumes `dma_alloc` yet. - **M15** — interrupts for PCI devices, the MSI half. Discovery now gives every PCI function its 4 KiB ECAM config space as resource 0 (unblocking the capability walk @@ -302,8 +303,8 @@ rather than an out-struct. The rest of this section is the original design note. **The blocker, and it's a hard one.** No PCI device can take an interrupt today. [`addBars`](system/devices/acpi.zig) records `.memory` and `.io_port` BARs and never an -`.irq`; there is no `_PRT` parsing anywhere in the tree. `hpet` only works because the -HPET advertises its own routing options in its own registers — a privilege no ordinary +`.irq`; there is no `_PRT` parsing anywhere in the tree. The HPET is the one exception — +it advertises its own interrupt routing in its own registers, a privilege no ordinary device has. **The fix, in two halves.** @@ -326,7 +327,7 @@ which means **discovery should give each `pci_device` a `.memory` resource for i 4 KiB ECAM slot**. That's a small change to `parseMcfg` and it unblocks the whole capability walk (MSI, MSI-X, PCIe extended caps) without any new syscall. -Note QEMU's HPET reports `Tn_FSB_INT_DEL_CAP = 0` — no MSI — so `hpet` can never +Note QEMU's HPET reports `Tn_FSB_INT_DEL_CAP = 0` — no MSI — so an HPET timer could never exercise this path. The first MSI driver will be the first PCI driver. ## M16 — the IOMMU, and the honest caveat ◑ detection done, enforcement pending @@ -353,7 +354,7 @@ gap should be named rather than implied. `M13` (capability passing) is independent of `M14`/`M15` and is the cheapest. It unlocks class drivers, which are the shape with no hardware requirements at all — you -could write a real one against `bus`'s comparators tomorrow. +could write a real one against any device a bus driver publishes tomorrow. `M14` and `M15` together unlock the first HCD. `M14`'s barrier layer is worth landing on its own regardless: it's small, obviously correct, and stops every future driver diff --git a/docs/drivers.md b/docs/drivers.md index d2bb88a..74ea9bd 100644 --- a/docs/drivers.md +++ b/docs/drivers.md @@ -22,12 +22,12 @@ say.* ## How a driver gets started: discover, match, spawn -Nothing in the kernel decides that the HPET needs the `hpet` driver — that is policy, -and policy lives in user space. Boot brings user space up as a three-level supervision -hierarchy, each level owning one job: +Nothing in the kernel decides that the PCI host bridge needs the `pci-bus` driver — that +is policy, and policy lives in user space. Boot brings user space up as a three-level +supervision hierarchy, each level owning one job: ``` -kernel ──spawns──► init (PID 1) ──spawns──► device-manager ──spawns──► hpet +kernel ──spawns──► init (PID 1) ──spawns──► device-manager ──spawns──► pci-bus | | | spawns only init, the service supervisor: the driver supervisor: enumerates publishes the starts the system /system/devices, matches each device @@ -184,7 +184,11 @@ Two properties worth knowing: ## A whole driver -`system/drivers/hpet/hpet.zig` is ~150 lines and does all of it. The shape: +A minimal leaf driver is only ~150 lines and does all of it. danos ships **no such +example binary** — the driver model is proven by the real drivers (`pci-bus`, `ps2-bus`, +`usb-xhci-bus`), and a teaching example belongs here, in the docs, rather than as a +compiled program nobody runs. Illustrated with a hypothetical HPET timer driver, the +shape is: ```zig const hpet = findHpet(buf) orelse return; // device_enumerate, look for @@ -209,8 +213,8 @@ while (...) { } ``` -The HPET is a good first driver for a reason that isn't obvious. Its *counter* is a -clocksource — the only way to use it is to read it, so it proved `mmio_map` without +The HPET makes a good illustration for a reason that isn't obvious. Its *counter* is a +clocksource — the only way to use it is to read it, so it exercises `mmio_map` without needing interrupts at all. Its *comparators* are a clockevent, and can be configured **level-triggered** (`Tn_INT_TYPE_CNF`), which asserts a bit in `GENERAL_INT_STATUS` that the driver must write-1-to-clear. That's a genuine deassert step, so the full @@ -252,9 +256,10 @@ bus driver may only ever subdivide what it already owns. A device with **no resources** is legal and common. A USB device is reached through its controller, not by MMIO, so it gets `resource_count = 0`. -See [`system/drivers/bus/bus.zig`](../system/drivers/bus/bus.zig) for a complete one, and -[driver-model.md](driver-model.md) for how bus drivers, class drivers and host -controller drivers fit together. +See [`system/drivers/pci-bus/pci-bus.zig`](../system/drivers/pci-bus/pci-bus.zig) for a +real one — it claims a PCI host bridge, maps its ECAM window, and publishes each function +it finds as a child — and [driver-model.md](driver-model.md) for how bus drivers, class +drivers and host controller drivers fit together. ## What the kernel does not do for you @@ -313,32 +318,38 @@ uncacheable, physical address exposed), and **memory barriers** (`/lib/mmio`'s ## Verifying it -The `hpet` test spawns `hpet` from the initial ramdisk and watches the serial log. The driver -prints `hpet: ok` only after being woken five times, and its loop's only exit is -through `replyWait` returning a notification — it cannot reach that line by polling. +No demo driver ships to prove this end to end; the *real* drivers do, so the tests +target them and the kernel primitives directly: -The last check doesn't trust the driver's self-report at all: the kernel reads the I/O -APIC redirection entry back and asserts the line really is routed to a device vector, -really is level-triggered, and really was left unmasked by the driver's final -`irq_ack`. +- **`device-manager`** — boots only the device manager, which discovers the PCI host + bridge, matches `pci-bus`, and `system_spawn`s it. The test reads kernel state — the + process table and the device tree — to confirm pci-bus came up and registered the + functions it enumerated: the whole discover → match → spawn → driver-up chain. +- **`acpi-ps2`** — a user-space driver (`ps2-bus`) is woken by its device's IRQ, + delivered as an IPC notification, and attaches the keyboard: IRQ-as-IPC, end to end. +- **`pci-scan`** — a user-space driver (`pci-bus`) maps its device's MMIO (the ECAM + window) and walks it: `mmio_map`, end to end. +- **`containment`** — the kernel refuses a `device_register` whose child window escapes + the parent's grant (else it would be a syscall for mapping arbitrary memory), while an + identical re-register stays idempotent. Asserted in-kernel, straight against the broker. +- **`irqfree`** — the teardown path. Binds two owners to one shared endpoint, releases + one, and reads the I/O APIC back: the departing owner's line is masked, the sibling's + is not. That second half is why bindings are keyed on the owning *task* and not on the + endpoint pointer — endpoints are shared, so releasing "everything pointing at this + endpoint" would silently mask a live driver's device. +- **`iopass`** — the `device_grant` teardown rule, so destroying a driver's address + space never returns MMIO frames to the RAM pool. ``` -$ python3 test/qemu_test.py hpet irqfree iopass - hpet ... PASS (matched 'DANOS-TEST-RESULT: PASS') +$ python3 test/qemu_test.py device-manager acpi-ps2 pci-scan containment irqfree iopass + device-manager ... PASS (matched 'DANOS-TEST-RESULT: PASS') + acpi-ps2 ... PASS + pci-scan ... PASS (matched 'DANOS-TEST-RESULT: PASS') + containment ... PASS (matched 'DANOS-TEST-RESULT: PASS') irqfree ... PASS (matched 'DANOS-TEST-RESULT: PASS') iopass ... PASS (matched 'DANOS-TEST-RESULT: PASS') ``` -Two companions cover what `hpet` can't, because it never exits: - -- **`irqfree`** — the teardown path. Binds two owners to one shared endpoint, releases - one, and reads the I/O APIC back: the departing owner's line is masked, the sibling's - is not. That second half is why bindings are keyed on the owning *task* and not on - the endpoint pointer — endpoints are shared, so releasing "everything pointing at - this endpoint" would silently mask a live driver's device. -- **`iopass`** — the `device_grant` teardown rule, so destroying a driver's address - space never returns MMIO frames to the RAM pool. - ## What's next (not done here) The big driver-model pieces — capability passing (class drivers), DMA + barriers, MSI, diff --git a/docs/m17-m18-plan.md b/docs/m17-m18-plan.md deleted file mode 100644 index 8518c9f..0000000 --- a/docs/m17-m18-plan.md +++ /dev/null @@ -1,210 +0,0 @@ -# M17–M18 execution plan: process lifecycle + device manager - -**Archived — completed 2026-07-13** (every item checked; suite ended 54/54). -Kept as the record of how M17–M18 landed; the successor is -[m19-m20-plan.md](m19-m20-plan.md). - -The operational plan for building [process-lifecycle.md](process-lifecycle.md) -(M17) and [device-manager.md](device-manager.md) increments 5–7 (M18). Design is -settled in those documents; this file is the build order — one phase at a time, -each phase green before the next starts. Delete or archive this file when M18 -lands. - -**Definition of green, every phase:** `zig build` clean, `zig build test` clean, -`python3 test/qemu_test.py` passes (existing scenarios plus the phase's new one), -and the relevant design doc's "known gaps" / status lines updated. Commit per -green phase (no co-author trailers). - -**Workflow (settled 2026-07-12):** work happens in a dedicated git worktree, on -feature branches cut from `main` — `feat/process-lifecycle` (M17.1–17.4), -`feat/device-manager` (M18.1), `feat/usb-xhci-bus` (M18.2–18.3). When a branch's -phases are all green it is **auto-merged into `main`**; branches are kept after -merge, not deleted. Merges and branches are pushed to origin. Phase 0 (once): -commit the design docs, merge the outstanding `feat/usb` work into `main`, and -run the existing QEMU suite green before any new work starts. - -**Numbering note:** continues the milestone sequence (driver track ended at M16). - -## Status - -The loop marks a phase `[x]` in the same commit that lands it. A phase is marked -only when its definition of green holds. - -- [x] **Phase 0** — baseline: docs committed, feat/usb merged to main, pushed; - `usb-xhci-libary.zig` renamed to `usb-xhci-library.zig`; existing QEMU - suite green from the worktree (48/48, 2026-07-12). -- [x] **M17.1** — kernel releases claims/MSI on death (claims: `releaseAllOwnedBy` - in the reap; MSI was already swept by `irq.releaseOwner`; `claim-release` - test; suite 49/49) -- [x] **M17.2** — exit reasons (`ExitReason` recorded at exit/fault/kill before - the notification; `process_exit_reason` supervisor-gated; - `runtime.process.exitReason`; kernel + ring-3 assertions; suite 49/49) -- [x] **M17.3** — published exit events + VFS subscriber (`process_subscribe`, - bounded ref-counted table, publish on every death; - `runtime.process.subscribeExits`; VFS handles carry owners and are swept on - the owner's death; `vfs-client-death` test; suite 50/50) -- [x] **M17.4** — signals, timer notifications, `runtime.process`, the service - harness (signal_bind/process_signal + coalescing pending mask; timer_bind - on the tick; bindSignals/signalsFrom/sendSignal/stop + timerOnce; - runtime.service.run with the zero-length ping; VFS converted; `signals` - scenario; suite 51/51) -- [x] **merge** `feat/process-lifecycle` → main, push (merged 2026-07-13) -- [x] **M18.1** — device-manager protocol: hello + restart policy - (device-manager-protocol module; the manager as a harness service: - supervised spawns, hello deadline via timer sweep, restart with - 300/600/1200ms backoff, exit reasons deciding restart-vs-stopped, - crash-loop cap; usb-xhci-bus first conforming driver; crash-test fixture - re-proving claim release each respawn; `driver-restart` scenario; - maximum_tasks 16→32 — the sweep was overflowing the pool; suite 52/52) -- [x] **merge** `feat/device-manager` → main, push (merged 2026-07-13) -- [x] **M18.2** — xHCI port scan + tree reports (child_added/child_removed in - the protocol; the manager's child mirror with death-pruning; xHCI maps the - register BAR — resource 0 is ECAM — reads CAPLENGTH/HCSPARAMS1, scans - PORTSC, reports connected ports with speed-class identity; `usb-report` - scenario proves report → prune → respawn → re-report; suite 53/53) -- [x] **M18.3** — app surface: enumerate/subscribe over IPC (subscriber - endpoint rides as the call's capability; events are the same structs the - buses send); device-list first client; protocol capped at the kernel's - IPC MESSAGE_MAXIMUM (256); the startUserTask debug print removed — it - sheared concurrent serial lines and was the scenario-flake root cause; - `device-list` scenario; suite 54/54) -- [x] **merge** `feat/usb-xhci-bus` → main, push (merged 2026-07-13) — **plan complete** - ---- - -## M17.1 — the kernel releases a dead process's claims - -The cleanup half of iron rule 1; the prerequisite for every restart story. - -- `system/kernel/devices-broker.zig`: `releaseAllOwnedBy(owner: u32)` — clear - every `claimed[]` slot holding this task id. -- `system/kernel/process.zig`: call it from the reap path, alongside the existing - IRQ-binding release (the ordering comment there says why IRQs go first — claims - slot in after them, before the exit notification). -- MSI vectors: find where `msi_bind` records per-device vectors (interrupts - module) and release those by owner in the same pass. -- Docs: remove the claims bullet from process-management.md "Known gaps". - -**Test:** new QEMU scenario `claim-release` — a test child claims an unclaimed -device, is killed, is respawned, and claims the same device again successfully; -assert both claims in the serial log. Kernel-side unit coverage in -`system/kernel/tests.zig` for `releaseAllOwnedBy` (claim two devices as two owners, -release one owner, verify exactly its claims freed). - -## M17.2 — exit reasons - -- `system/abi.zig`: `ExitReason` (exited, aborted, segmentation_fault, - illegal_instruction, arithmetic_fault, killed). -- Kernel: record the reason at every death site — clean exit path, each fault - class in `onException`, the kill path. Bounded recent-exits table (ids are never - reused, so a small ring keyed by id is enough). -- New system call `process_exit_reason(id)` — supervisor-gated, like kill; returns - the recorded reason or `-ESRCH` once evicted. -- `library/runtime/process.zig`: `ExitReason` + `exitReason(id: u32)`. -- Docs: remove the no-exit-status bullet from process-management.md. - -**Test:** extend the `supervision` scenario — three children: one exits cleanly, -one faults (the fault-recovery pattern), one is killed; the supervisor asserts all -three reasons. - -## M17.3 — published exit events - -- Kernel: bounded subscriber table (endpoints); new system call - `process_subscribe(endpoint)` (ungated, like `process_enumerate`); every death - posts `notify_exit_bit | id` to each subscriber — the same post the supervisor - path already uses. -- `library/runtime/process.zig`: `subscribeExits(endpoint)`. -- VFS becomes the first subscriber: on an exit event, release every handle keyed - by that task id (badges already are task ids). Log the release. -- Docs: note the convention in ipc.md (exit events reuse the exit-notification - badge encoding). - -**Test:** new QEMU scenario `vfs-client-death` — a client opens a file and is -killed without closing; assert the VFS logs the handle release and its open-handle -count returns to baseline. - -## M17.4 — signals and the service harness - -- Kernel: per-task pending mask + bound endpoint; system calls - `signal_bind(endpoint)` and `process_signal(id, signal)` (supervisor-or-self - gated); delivery posts `notify_signal_bit | pending mask`, coalescing; pending - signals with no bound endpoint pend silently. -- `library/runtime/process.zig`: `Signal`, `SignalSet`, `bindSignals`, - `signalsFrom`, `sendSignal`, `stop(id, deadline_ms)` (terminate → wait for exit - notification → kill). Implement `terminate`, `reload`, `user_1`, `user_2`; - `interrupt`/`quit` are enum members with no sender yet; `alarm` stays unbuilt. -- Kernel: **one-shot timer notifications** — `timer_bind(endpoint, ms)` posts a - notification badge when the deadline lands (IRQ-as-IPC again, on the timer - wheel `sleep` already uses). This is the missing timed-wait primitive: - `replyWait` blocks forever and `sleep` blocks the whole process, but `stop()`'s - escalation, the device manager's `hello` deadline (M18.1), and restart backoff - all need a deadline while staying responsive. It is also the mechanism `alarm` - gets for free later. -- New `library/runtime/service.zig`: the harness — `run(callbacks)` owning the - replyWait loop, folding protocol messages, signals, and child-exit notifications - into `init` / `on_message` / `on_reload` / `on_terminate`; answers the common - `ping` automatically. Define the reserved `ping` request encoding here and - document it in ipc.md (one obvious encoding; smallest that cannot collide with - existing protocols). -- Convert one existing service (input-source or hpet) to the harness as proof it - subtracts code rather than adding it. - -**Test:** extend `supervision` — a harness-built child: `sendSignal(reload)` -observed in its log, `ping` answered, `stop()` produces a clean exit with reason -`exited`; a second child that ignores signals (no bind) is killed by `stop()`'s -deadline with reason `killed`. - -## M18.1 — device-manager protocol: hello + restart policy - -- New `system/services/device-manager/device-manager-protocol.zig` module - (vfs-protocol pattern): `hello { version, role, device_id }`; version constant; - reserved fields. -- Device manager: register the `.device_manager` endpoint; spawn drivers with its - exit endpoint; enforce the hello deadline; restart policy — backoff, crash-loop - cap (three fast deaths → mark failed, log, stop), reasons from M17.2 deciding - restart vs not. -- usb-xhci-bus: adopt the harness + send hello. hpet/ps2-bus follow only if the - conversion is mechanical; otherwise they keep working unconverted (the manager - only enforces hello on drivers spawned with an assignment). -- build.zig: test-loop entry for the protocol module if it grows pure logic. - -**Test:** new QEMU scenario `driver-restart` — the xHCI driver takes a test-only -argv flag to fault after hello on its first run; assert: fault, exit reason -recorded, manager respawns with backoff, second run claims the controller -(M17.1) and hellos clean. Assert the crash-loop cap by a driver that always -faults (a tiny test driver, not xhci). - -## M18.2 — bus tree reports - -- Protocol: `child_added { parent, identity, resources }` / `child_removed { id }`. -- usb-xhci-bus: bring-up to **port scan only** — map the MMIO window (claimed in - M16-era work), controller reset/start per xHCI spec, walk the port registers, - report one `child_added` per connected port with speed + port number as - identity. **No transfer rings, no descriptors** — reading device/interface - descriptors (and therefore USB class triples for matching) is the follow-on USB - track, not this plan. -- Device manager: mirror reports into its tree; prune the subtree (emitting - `child_removed`) when a bus driver dies; assert re-report on restart. - -**Test:** QEMU already attaches usb-kbd + usb-mouse on xhci.0 — assert two -`child_added` events reach the manager and appear in its tree dump; kill the -driver, assert two `child_removed` then two fresh `child_added` after respawn. - -## M18.3 — the application surface - -- Protocol: `enumerate` (tree snapshot) + `subscribe` (published add/remove - events, input-service pattern). -- A small client (`device-list`, the `ps` analog) exercising both; the manager - becomes the one answer to "what devices exist" for user space. - `device_enumerate` stays for drivers/kernel seeding — its retreat is tied to the - discovery migration, out of this plan. - -**Test:** QEMU scenario — `device-list` shows the tree including USB children; -during a driver restart the subscribing client logs remove + add events. - ---- - -**Explicitly out of scope** (own tracks, after M18): discovery migration (pci-bus -driver, acpi service, retiring the kernel scan), USB control transfers + -descriptors + class-driver matching, the musl layer, `interrupt`/`quit` senders -(needs a console), job control. diff --git a/docs/m19-m20-plan.md b/docs/m19-m20-plan.md deleted file mode 100644 index 117aa5a..0000000 --- a/docs/m19-m20-plan.md +++ /dev/null @@ -1,196 +0,0 @@ -# M19–M20 execution plan: discovery migration - -The operational plan for [device-manager.md](device-manager.md)'s increment 8: -discovery leaves the kernel — a **pci-bus driver** (M19) and an **acpi service** -(M20), with the kernel's device enumeration retired behind them. Same rules as -[m17-m18-plan.md](m17-m18-plan.md): one phase at a time, each green before the -next; this file is the build order and the checklist. - -**Definition of green, every phase:** `zig build` clean, `zig build test` clean, -`python3 test/qemu_test.py` passes (existing scenarios plus the phase's new -one), and the relevant design doc updated. Commit per green phase (no co-author -trailers). The full suite is the regression net — the existing -`driver-restart` / `usb-report` / `device-list` / `input` scenarios must stay -green *through* the migration, which is the whole point: the system must not be -able to tell who enumerated it. - -**Workflow:** dedicated worktree; branches off `main` — `feat/pci-bus` -(M19.0–19.3), `feat/acpi-service` (M20.1–20.3); auto-merge to main when a -branch is green; keep branches; push everything. - -## Settled decisions (2026-07-13 — veto before the loop starts) - -1. **What "retiring the kernel scan" means.** The kernel keeps, forever, the - parses it needs before user space exists: RSDP/XSDT location, MADT (SMP), - the HPET table (the tick), FADT + the AML `\_S5` evaluation (poweroff — the - power tests prove it), and MCFG (the host bridge node). What retires is - **device enumeration**: the ECAM function walk (M19.3) and the DSDT/SSDT - namespace walk that builds device nodes (M20.3). The AML module stays a - shared build module compiled into both the kernel (for `\_S5`) and the acpi - service (for everything else) — same source, two builds, no fork. -2. **Bridge apertures come from the firmware memory map, not AML.** Registered - PCI functions carry BAR resources, and containment demands the bridge own - windows that cover them. The apertures are derived kernel-side from the - boot memory map's MMIO holes (regions that are neither RAM nor tables) — - mechanical, AML-free, and available at boot regardless of what later moved - to user space. (The bridge today carries only ECAM + bus range; this is the - prerequisite M19.0 exists for.) -3. **`device_register` becomes idempotent on exact match.** A re-registration - with identical (parent, class, resources) returns the existing id instead - of appending. The kernel table has no unregister, so without this a - restarted registering bus would duplicate its children on every respawn — - idempotence makes restart-and-re-report safe for every future bus, not just - PCI. -4. **The manager matches from reports.** `ChildAdded` gains a `device_id` - field (the kernel-registered id, `no_device` for unregistered leaves like - USB ports). After the M19.3 flip, PCI driver matching keys off reported - identity (the class triple) instead of the manager's boot-time snapshot — - the snapshot match remains only for what the kernel still seeds. One flip - phase changes both sides at once so no device is ever matched twice. -5. **The acpi service's authority is one node.** The kernel publishes an - `acpi-tables` device: memory resources covering the table blobs plus a - broad `io_port` resource — the documented trust grant to exactly one - process (AML OperationRegions reach EC/PM ports; the claim-gated - io_read/io_write calls already exist). The service claims it, maps the - tables, and runs the shared AML module in ring 3 behind a `Hal` backed by - `mmio_map` + `io_read`/`io_write`. -6. **Both new processes are protocol drivers** under the manager: hello, - supervision, restart with backoff — all inherited from M18.1 for free. - Registration idempotence (decision 3) is what makes their restarts sound. -7. **Firmware neutrality is the contract** (2026-07-13). The generic layer is - everything at and above the device-manager protocol — descriptors, - containment, reports, matching, supervision — and none of it may become - x86-specific. Discovery is one swappable process per firmware: the acpi - service on x86; an **fdt service** on the Raspberry Pis (claims a - `devicetree-blob` node, reports children from the flattened device tree — - pure data, no bytecode, no port grant, strictly simpler than ACPI). The - manager owns the tree as *data* and touches no hardware, ever — AML runs in - a crashable, supervised discoverer precisely so a firmware-bytecode fault - can never take down the supervisor. Two consequences recorded now: - `DeviceDescriptor`'s 8-byte `hid` cannot hold an FDT `compatible` string - ("brcm,bcm2835-aux-uart") — identity widens before the fdt service exists; - and cross-firmware surfaces are named by **domain, not firmware** (M21 - defines a *power* protocol, not an "ACPI events" protocol — PSCI/mailbox - sources feed the same subscribers on ARM). **Landed early (2026-07-13):** - both services exist as placeholders (system/services/acpi, system/services/ - fdt) and the build's `-Ddiscovery=acpi|fdt` option fills the ramdisk's - neutral `discovery` slot — the manager will spawn "discovery" by that name - in M20.3 and never learn which firmware it is on. - -## Status - -- [x] **M19.0** — prerequisites (bridge apertures from the memory map's - *gaps* — the single-hole rule died on OVMF's flash at the top of 4 GiB, - caught by the new every-BAR-contained assert in `discovery`; idempotent - `device_register` proven in `bus`; `ChildAdded.device_id`; - m17-m18-plan.md archived; suite 54/54). -- [x] **M19.1** — pci-bus driver, scan only (claims the bridge, maps ECAM - through its grant, brute-force walk with the multifunction rule; the - manager matches pci_host_bridge → pci-bus per device with the full - protocol contract; `pci-scan` builds its expected marker from the - kernel's own count — equivalence on the first run; suite 55/55). -- [x] **M19.2** — register + report (BAR probe mirrored byte-for-byte from the - kernel's addBars so dedupe returns the kernel's node ids during - coexistence; the bridge gained the io_port aperture I/O BARs need; - reports carry the registered device_id; pci-scan drills a forced restart - and asserts the PCI node count never grows — plus harness hardening: a - failing case now preserves its serial as -failed-serial.log, and - the heavy scenarios run at 150s; suite 55/55). -- [x] **M19.3** — the flip: kernel `enumeratePci`/`addBars`/`PciHeader` all - deleted (bridge node stays); manager matches PCI drivers from reported - identity, deduped by registered id. Surfaced and fixed a real SMP race the - flip created — ring-3 device_register made the broker table concurrent, so - mmio_map's lock-free read intermittently tore hpet's resource length - (user fault) and overflowed `r.len-1` into a kernel panic; now the broker - read is under the big lock and the arithmetic is guarded, and pci-bus - skips size-0 BARs. discovery.md updated; suite 55/55 (driver-restart - hammered 6×). -- [x] **merge** `feat/pci-bus` → main, push (merged 2026-07-13). -- [x] **M20.1** — acpi service, parse only: the AML interpreter is now a build - module compiled into both kernel and service; the kernel publishes the - `acpi-tables` node (AML blobs as memory resources, the broad io_port grant, - the SCI); the service claims it, maps the blobs, runs the shared parser in - ring 3, and self-verifies its Device count against the kernel's (34 = 34, - deterministic via argv, no log-scraping); the manager spawns `discovery` - at startup. Parse-only touches no hardware. Suite 56/56. - -- [x] **M20.2** — register + report: the service evaluates `_STA`/`_CRS` in - ring 3 (interpreter Hal = port I/O over the claimed node; a scratch page - backs SystemMemory maps so a stray region can't fault it) and registers + - reports each present `_HID` device under `acpi-tables`. Containment: the - broker's irq check became range-based (len-1 == the old equality) so the - node's broad irq window covers children's legacy lines; io ports fall in - the broad io grant. ChildAdded gained `hid`. Matching stays off. The - `acpi-report` scenario asserts the PS/2 keyboard (3 resources) and mouse - (1 resource) among the reports. Suite 57/57. -- [x] **M20.3** — the flip: the kernel's `wireAcpiDevices` call is gone (the - device-building helpers are retained-but-dead pending a focused sweep, - spawned as a task; static tables + `\_S5` + the acpi-tables node stay). - The manager matches ps2-bus from ACPI `_HID` reports; the service - registers all devices before reporting any (no keyboard-before-mouse - race). The `acpi-ps2` scenario proves report → spawn → ps2-bus attaches - its keyboard; `ioport` retargeted to the acpi-tables I/O window (the - kernel-built PS/2 node is gone). Suite 58/58. -- [x] **merge** `feat/acpi-service` → main, push (merged 2026-07-13) — **discovery migration complete**. - ---- - -## Phase notes - -**M19.0 apertures:** the boot memory map already crosses the handoff -([boot-handoff]), but discovery never sees it today — expect a small -pass-through (kernel init hands the map to the platform layer) before the -holes computation, which belongs where the bridge node is built -(`parseMcfg`). Sanity-check on QEMU q35: the xHCI BAR (`0xc0000000`-region -values seen in the M18 logs) must land inside a derived aperture, asserted in -the kernel unit test. - -**M19.1 scanning without owning config access twice:** the driver reads config -space through its ECAM mmio_map grant of the *bridge* window — the same bytes -the kernel walk read. Vendor-id `0xFFFF` skip, header-type multifunction rule, -no bridge recursion (matches the kernel's current single-segment walk). - -**M19.2 BAR sizing:** the classic size probe (write all-ones, read mask, -restore) is deferred — the BARs' current programmed values and types are -enough for containment-checked registration at bring-up; sizing lands with the -first driver that needs to *move* a BAR. Log what is registered so the -scenario can assert it. - -**M19.3 what the manager still seeds from the snapshot:** everything the -kernel still enumerates (timers, ACPI nodes until M20.3). The PCI arm of -`pciDriverFor` switches source; `driverFor` doesn't move until M20.3. - -**M20.1 spawn and identity (pre-settled 2026-07-13):** the manager spawns -`discovery` by its neutral ramdisk name at startup, as an ordinary protocol -driver (hello, supervision) — from M20.1 on, on every boot. For reporting ACPI -devices, `ChildAdded` gains `hid: [8]u8` (EISA ids fit; zero = none): -firmware *string* identity travels beside the numeric `identity` field until -the FDT-driven widening replaces both (decision 7). - -**M20.1 Hal in ring 3:** `mapMmio` → `device.mmioMap` over the claimed -acpi-tables node (plus a table-offset map for blobs); `pioRead`/`pioWrite` → -`device.ioRead`/`ioWrite` against its io_port resource. The interpreter cannot -tell it moved — that is the assertion of `acpi-parse`. - -**M20.2 containment for `_CRS`:** io ports fall inside the node's broad -io_port resource; MMIO windows (HPET, LAPIC ranges some firmwares list) fall -inside the memory-map holes added to the node in M20.1. Anything that doesn't -fit is logged and skipped, loudly — bring-up honesty over silent drops. - -**M20.3 ps2 ordering:** ps2-bus binds nodes the acpi service now reports, so -its spawn moves behind the report (the manager's matching handles this once -the source flips); the `input` scenario proves the keyboard still types. - -**Explicitly out of scope:** PCI bridge recursion (single segment, flat bus -walk stays); BAR reprogramming/sizing; disk/PCIe hotplug; interrupt routing -changes (`_PRT` stays wherever it is today); the USB descriptor track; -multi-segment ECAM; per-device power states (D-states, `_PSx`/`_PRx`, -suspend/resume — a future *lifecycle-vocabulary* extension, since "suspend" -has the shape of a signal every driver must answer, and it has no consumer -until laptop sleep); CPU P/C-states. - -## M21 — ACPI events + system power — DONE - -Built and merged (docs/m21-plan.md, 2026-07-13): the SCI + power button, Notify/GPE -dispatch, and orderly shutdown (init's stop cascade into a ring-3 S5 write). -See that plan for the phase record. diff --git a/docs/m21-plan.md b/docs/m21-plan.md deleted file mode 100644 index 3fea641..0000000 --- a/docs/m21-plan.md +++ /dev/null @@ -1,138 +0,0 @@ -# M21 execution plan: ACPI events + system power - -The operational plan for the event side of the acpi service and orderly -shutdown — the capstone [m19-m20-plan.md](m19-m20-plan.md) previewed. Same -rules as its predecessors: one phase at a time, each green before the next; -this file is the build order and the checklist. - -**Definition of green, every phase:** `zig build` clean, `zig build test` -clean, `python3 test/qemu_test.py` passes (existing scenarios plus the -phase's new one), and the relevant design doc updated. Commit per green phase -(no co-author trailers). Failing cases preserve their serial logs -(`-failed-serial.log`). - -**Workflow:** dedicated worktree; branch `feat/power-events` off `main`; -auto-merge to main when the branch is green; keep the branch; push everything. - -## Settled decisions (2026-07-13, approved) - -1. **S5 is executed by the acpi service from ring 3.** No new syscall: the - broad port grant (M20 decision 5) already made this physically possible — - the service holds the PM1 control ports in its io grant and derives `_S5` - from its own namespace (`aml.sleepState`). Formalizing it adds no - authority. The kernel keeps `power.zig` for its own test paths and - panic-time use. -2. **The power surface is domain-named** (decision 7 of the last plan): a - `power-protocol` module + `ServiceId.power = 5`, registered by the acpi - service — on ARM, a PSCI/mailbox service registers the same id and - subscribers never know the difference. Messages: `subscribe` (endpoint as - the call's capability, the input/manager pattern), `shutdown` (accepted - only from PID 1 — init), and events published as buffered messages: - `power_button`, `lid`, `ac`, `battery`, generic `notify` with a code. -3. **The service learns event ports from its own FADT copy**: the kernel adds - the FADT as one more memory resource on the acpi-tables node; the service - tells it apart from the AML blobs by signature ("FACP" header — the blob - resources are header-stripped bytecode and start with no signature). The - kernel's own FADT parse is untouched. -4. **The acpi service converts to the harness** (`runtime.service.run`): - protocol messages (subscribe/shutdown), the SCI notification, and the - existing report flow fold into one loop — the shape it was always meant - to have. -5. **GPE/Notify correctness is proven by host unit tests** (synthetic AML - with a Notify inside a method body; aml.zig joins the `zig build test` - loop). The QEMU scenario proves the power button — a *fixed* event, - deterministically injectable via QMP `system_powerdown` — because QEMU - cannot raise GPEs deterministically on this config. Battery/AC/lid and the - embedded controller (`_Qxx`) are interface-complete here and validated on - real hardware (the laptop) later. - -## Ground truth the phases build on (verified 2026-07-13) - -- `system/devices/power.zig` `shutdown()` is the kernel's S5 write - (SLP_TYP|SLP_EN to PM1a/PM1b control); there is no power syscall. -- init (`system/services/init/init.zig`) spawns vfs/input/device-manager - fire-and-forget — no child ids kept, no signals, no event loop. The whole - stop toolkit exists in `runtime.process` (stop/sendSignal/bindSignals). -- `test/qemu_test.py` has no QMP channel (serial is a one-way file). -- The kernel parses PM1 *control* blocks and SCI_INT from the FADT; the PM1 - **event** blocks (offsets 56/60, len at 88) and **GPE0/GPE1** blocks - (offsets 80/84, lens 92/93) are unparsed — the service reads them from its - FADT copy (decision 3). -- The acpi-tables node carries the SCI as its only `len == 1` irq resource - (the broad window is len 256) — that is how the service finds it to - `irqBind`. -- `notify_opcode = 0x86` exists in `system/devices/aml/opcodes.zig` but the - interpreter never handles it — a GPE `_Lxx` body containing Notify fails - evaluation today. Everything else a GPE handler needs (field access, - control flow, method calls) is proven by the ring-3 `_STA`/`_CRS` work. -- The dead-code sweep (spawned task) also edits `system/devices/acpi.zig`; - M21.0 checks whether it landed and rebases before touching that file. - -## Status - -- [x] **M21.0** — baseline (dead-code sweep confirmed landed on main — no - acpi.zig conflict; `feat/power-events` cut; QMP channel in the harness: - always-on unix socket, client with the capabilities handshake, per-case - `qmp_after` hook, and a hook-must-deliver pass gate that the smoke case - now proves with a harmless query-status; suite 58/58). -- [x] **M21.1** — SCI + the power button (kernel appends the FADT as an - acpi-tables memory resource, tagged by its "FACP" header; `power-protocol` - module + `ServiceId.power = 5`; the acpi service converted to - `runtime.service.run`, registers `.power`, reads PM1 event/control + GPE - ports from its FADT copy, enables ACPI mode if SCI_EN is clear, binds the - SCI (the len-1 irq), sets PWRBTN_EN; the SCI handler clears PM1_STS, - logs `power: button pressed`, publishes `power_button`, acks. Scenario - `power-button` injects a real `system_powerdown` via QMP; initial-ramdisk - timeout 30→60s for the service's added boot work; suite 59/59). -- [x] **M21.2** — Notify + GPE dispatch (interpreter handles `notify_opcode` - into a bounded queue, cleared per-evaluate, drained via - `takeNotifications`; the service walks GPE status/enable bytes, evaluates - `\_GPE._Lxx`/`_Exx` per active bit, maps notified nodes to events - (battery/ac/lid/generic), clears GPE_STS write-1, acks. EC `_Qxx` out. - Host unit test with hand-encoded AML proves the queue; aml.zig joined the - `zig build test` loop. QEMU raises no GPEs — suite is regression net, - 59/59). -- [x] **M21.3** — orderly shutdown (init supervises its children on one - endpoint that also carries signals, power events, and a re-arming - heartbeat timer; on `power_button` or a `terminate` signal it logs - `init: shutting down`, runs `stop(child, 2000, endpoint)` in reverse - order, then requests `.power` shutdown; the acpi service honors shutdown - from a subscriber — init is the one subscriber, a soft gate that survives - testing where PID 1 isn't init — and writes SLP_TYP|SLP_EN from ring 3. - `orderly-shutdown` scenario proves button → shutting-down → S5 → QEMU - exit; suite 60/60). -- [x] **merge** `feat/power-events` → main, push, keep the branch (merged 2026-07-13) — **M21 complete**. - ---- - -## Phase notes - -**M21.0 QMP:** open the unix socket after Popen, complete the -`qmp_capabilities` handshake, then send the hook's command (for these -scenarios: `{"execute": "system_powerdown"}`). The socket is additive — no -existing case may notice it. Note e3fe3f3 recently reworked how the harness -boots; adapt to its current shape rather than the pre-rework description. - -**M21.1 SCI details:** PM1_STS is at the event block base (write-1-to-clear); -PM1_EN at base + block_len/2; PWRBTN bit is 8 in both. If PM1b exists, mirror -reads/writes to both blocks. Enable ACPI mode only when SCI_EN (PM1 control -bit 0) is clear — OVMF boots may already have it set. The publish path reuses -the manager's subscriber table pattern (bounded, drop-on-failed-send). - -**M21.2 GPE walk:** GPE0_STS bytes live at the GPE0 block base, GPE0_EN in -the block's upper half; for a set+enabled bit n, the handler method is -`_L%02X` (level) or `_E%02X` (edge) under `\_GPE`. Evaluate, drain the notify -queue, clear the status bit, ack. A missing handler method is clear-and-log, -not an error. - -**M21.3 ordering:** init subscribes with retries — the acpi service registers -`.power` well after init starts. The stop sequence runs vfs last (other -services may flush through it). The S5 write mirrors `power.zig`'s -`sleepValue` (SLP_TYP bits [12:10], SLP_EN bit 13); if the write returns, log -`power: S5 write did not take` so the scenario fails loudly instead of -hanging. - -**Explicitly out of scope:** the embedded controller and `_Qxx` queries, -battery `_BST`/`_BIF` evaluation beyond the interface stubs, lid/AC on QEMU -(no emulation), reboot over the power protocol, S3 sleep, per-device D-states -(a future lifecycle-vocabulary extension), thermal zones. diff --git a/docs/power.md b/docs/power.md new file mode 100644 index 0000000..24c6929 --- /dev/null +++ b/docs/power.md @@ -0,0 +1,128 @@ +# The power service: events and shutdown + +A laptop lid closes, a battery drains, someone presses the power button — and +several parts of the system might care: a session manager dims the screen, a +logger notes it, and ultimately *something* has to turn the machine off. None of +them owns the hardware that reported the event, and the reporter should not know +who is listening. So system power is a **service**: an event source **publishes** +button/lid/battery/AC events, interested processes **subscribe**, and one +privileged caller — init — can ask it to power the machine off. It is the same +publish/subscribe shape as the [input service](input.md), applied to power. + +## Why a service, and why it is named for the domain, not the firmware + +Where the events come from is firmware-specific — on x86 they ride the ACPI SCI +([acpi.md](acpi.md)); on a Raspberry Pi they would come from PSCI or a mailbox. +What subscribers want is not: *the lid closed* means the same thing regardless of +who noticed. So the surface is **domain-named**. There is a `power-protocol` +module and a well-known `ServiceId.power = 5`; on x86 the **acpi service** +registers it, and on ARM a PSCI/mailbox service will register the *same* id. +Subscribers call `runtime.ipc.lookup(.power)` and never learn which firmware they +are on — the neutrality the whole [discovery](discovery.md) migration exists to +preserve, carried one layer up into a running-system surface. + +This is why the protocol is `power`, not "ACPI events": naming a cross-firmware +surface after one firmware would leak x86 into code the ARM port must reuse +unchanged. + +## The protocol + +The `power-protocol` module ([system/services/power/protocol.zig](../system/services/power/protocol.zig)) +follows the vfs-protocol pattern — extern-struct messages, a version, reserved +fields. Three operations: + +| Direction | Operation | Purpose | +|---|---|---| +| subscriber → service | `subscribe` | receive published events; the subscriber's endpoint rides as the call's **capability** (the input/device-manager pattern) | +| init → service | `shutdown` | orderly shutdown's last step: enter S5 (soft off) | +| service → subscriber | `event` | a published `EventMessage`, delivered as a buffered message (never sent *to* the service) | + +Events are published, not polled: like the input service, the service holds +subscriber endpoints as capabilities and `ipc_send`s each event as a buffered +message, so a slow or dead subscriber can never wedge the source. The event +vocabulary is hardware-neutral: + +- `power_button` — the button was pressed (a fixed ACPI event on x86). +- `lid`, `ac`, `battery` — the named GPE-driven events. +- `notify` — a device notification that maps to none of the above; its `code` + (the ACPI `Notify` argument) and the notifying device's `hid` say which device + and what happened. + +An `EventMessage` carries the `event` tag plus `code` and an 8-byte `hid`, so a +generic `notify` is fully described without a second round trip. + +**`shutdown` is authority, not information.** It is the only operation that +*does* something irreversible, so it is gated: the contract is that only init +(PID 1) may request it, because init is the process that has already run the stop +sequence over everything else. The acpi service implements this as a **soft +gate** — it honors `shutdown` only from a process that is a *subscriber*, and +init is the one subscriber. That stands in for "only the system supervisor may +power off" without hard-coding a pid, so it still holds under tests where PID 1 +is not init. + +## Orderly shutdown + +Powering off cleanly is where the power service, the [process +lifecycle](process-lifecycle.md), and [ACPI events](acpi.md) compose. init +already supervises the services it starts; for shutdown it runs **one event loop +over one endpoint** that carries three things at once: its children's exit +notifications, the lifecycle **signals** it can receive (`terminate`), and the +**power events** it subscribes to — plus a re-arming heartbeat timer proving PID +1 is alive. (init subscribes with retries, because the power service registers +`.power` well after init starts; a missing power service is not fatal — a +`terminate` signal drives the same path.) + +On a `power_button` event or a `terminate` signal, init: + +1. logs that it is shutting down, +2. runs the standard stop sequence — `runtime.process.stop(child, deadline, + endpoint)` — over its children **in reverse spawn order**, so the VFS stops + last (other services may flush through it), each child getting the + *terminate → deadline → kill* escalation from + [process-lifecycle.md](process-lifecycle.md), and +3. requests `.power` `shutdown`. + +The service then enters **S5** (soft off) by writing `SLP_TYP | SLP_EN` to the +PM1 control register(s) from ring 3, mirroring the kernel's own +`system/devices/power.zig` `sleepValue`. If the write returns instead of powering +the machine off, it logs loudly so a test fails rather than hangs. + +**No new system call was needed for S5.** The broad io_port grant on the +`acpi-tables` node ([discovery.md](discovery.md)) already put the PM1 control +ports in the acpi service's hands, so writing S5 from ring 3 is something it +could physically already do; formalizing it as a protocol operation added a +contract, not authority. The kernel keeps `power.zig` for its own test paths and +panic-time poweroff, where no user space is available to ask. + +## Verifying it + +Two QEMU scenarios exercise the path, both injecting a real ACPI power-button +press via QMP `system_powerdown` (there is no other deterministic power event on +this config): + +- `power-button` proves the source: the acpi service's SCI handler logs the + press and publishes `power_button` (the ACPI half is in [acpi.md](acpi.md)). +- `orderly-shutdown` proves the whole composition: button → init logs shutting + down → children stopped → the service enters S5 → QEMU exits. The ordered + regex is the proof, and QEMU's self-exit through S5 is the pass. + +## Scope + +Interface-complete but validated on real hardware (the author's laptop) later, +because QEMU does not emulate them: battery `_BST`/`_BIF` evaluation beyond the +interface stubs, lid and AC events, and the embedded controller's `_Qxx` +queries. Deliberately out of scope for now: reboot over the power protocol, S3 +sleep, per-device D-states (a future lifecycle-vocabulary extension, since +"suspend" has the shape of a signal every driver must answer and has no consumer +until laptop sleep), and thermal zones. + +## See also + +- [acpi.md](acpi.md) — where the events come from on x86: the SCI, the power + button fixed event, and GPE/Notify dispatch in the acpi service. +- [discovery.md](discovery.md) — why the surface is domain-named, and the + firmware neutrality that makes a PSCI backend drop-in on ARM. +- [process-lifecycle.md](process-lifecycle.md) — the stop sequence + (`terminate → deadline → kill`) and signals init composes into shutdown. +- [device-manager.md](device-manager.md) — the supervision model init mirrors for + its own children. diff --git a/docs/resilience.md b/docs/resilience.md index 6e30cc8..2b05e41 100644 --- a/docs/resilience.md +++ b/docs/resilience.md @@ -11,7 +11,7 @@ because the kernel releases a dead process's claims. The `driver-restart` and `usb-report` scenarios prove kill → release → respawn → re-claim → re-report end to end. What remains of this document's ladder is scope, not mechanism: more of the system moved into restartable processes (the discovery migration, -[m19-m20-plan.md](m19-m20-plan.md), is the next rung). This is the property danos is really chasing: +[discovery.md](discovery.md), is the next rung). This is the property danos is really chasing: **if a part of the OS breaks, isolate it, and re-initialise it — without rebooting.** A crashed driver gets restarted; a wedged service gets killed and brought back. It's the reason the [microkernel](vision.md) shape was chosen, and it's a *separate* goal diff --git a/docs/timers.md b/docs/timers.md new file mode 100644 index 0000000..0f8a0f8 --- /dev/null +++ b/docs/timers.md @@ -0,0 +1,117 @@ +# Timers and time + +Two different needs hide under the word "timer", and danos keeps them apart: + +- **Reading the clock** — *what time is it?* A read of a free-running counter. +- **Waiting** — *wake me in N milliseconds*, or *notify me when a deadline passes.* + +Both are answered by the **kernel**, because the kernel already owns a timer: it has +to, to preempt tasks. The LAPIC heartbeat and the calibrated TSC that back all of this +are built in [device-interrupts.md](device-interrupts.md); the scheduler's blocking and +wait queues are in [scheduling.md](scheduling.md). This page is about the surface a +ring-3 program actually uses, and one deliberate absence: **there is no user-space time +service.** + +## Why time is a syscall, not a service + +The tempting microkernel move is to put a timer *driver* in user space and have +applications ask it for the time over IPC. For a **monotonic clock that is wrong** — +reading `now()` should never cost an IPC round trip. The kernel is already holding the +answer: it computes the current time every time it schedules, from the TSC, in a couple +of instructions. Surfacing that as a system call is pure mechanism; routing it through a +message to another process would be slower *and* redundant, and a device like the HPET +(uncacheable MMIO reads) is a particularly bad thing to read on every `now()`. + +This is the same conclusion every serious system reaches: Linux and Zircon read the +counter in the vDSO, L4 exposes a clock field in a shared kernel page, seL4 reads the +cycle counter directly. None of them make a clock read an IPC. danos makes it a syscall. + +That "from the TSC" hides a portability question, because the TSC is only a valid clock +when the CPU guarantees it is *invariant* and when every core's TSC is *synchronized*. +danos checks both — the invariant-TSC CPUID bit (`0x80000007` EDX[8], set on Intel and +AMD), and a cross-core "warp" check as the cores come up — and falls back to the HPET +counter when either fails. So `now()` stays accurate on a real Intel box, a real AMD box, +and inside a VM alike; only the source behind it differs. The mechanism is in +[device-interrupts.md](device-interrupts.md). + +So the timer hardware lives in the kernel, and there is **no `hpet` driver and no time +server** to consume. (An earlier HPET driver existed only to *demonstrate* the driver +model; that role now lives in [drivers.md](drivers.md), as documentation.) The one place +a user-space time service *is* justified — **wall-clock / calendar time** — is discussed +at the end; it is deliberately not built yet. + +## The three system calls + +Time and waiting are three entries in the small syscall table ([syscall.md](syscall.md)): + +- **`clock` (#23)** → monotonic nanoseconds since boot. It only moves forward. Not + wall-clock: no date, no timezone. Backed by `architecture.nanos()` (TSC, scaled with a + 128-bit intermediate so a long uptime can't overflow) — a few nanoseconds of + resolution, and just an `rdtsc` plus a multiply. +- **`sleep` (#3)** → block the caller for N milliseconds. The scheduler records a wake + deadline and the tick sweep wakes it (`scheduler.sleep`). +- **`timer_bind` (#31)** → arm a one-shot timer that, after N milliseconds, posts a + **timer notification** to an IPC endpoint. Unlike `sleep` it does **not** block: a + service can keep answering messages on the same endpoint while a deadline is pending. + This is the timed wait that stop-sequence escalation, hello deadlines, and restart + backoff are built from ([process-lifecycle.md](process-lifecycle.md), + [device-manager.md](device-manager.md)). + +The kernel's own scheduling timer (the LAPIC, vector 32) is never exposed to user space; +programs read the TSC through `clock` and get timed wakeups through `sleep`/`timer_bind`, +both riding the scheduler tick. + +## `runtime.time` — the generic interface + +Applications don't call the syscalls directly; they use `runtime.time` +(`library/runtime/time.zig`), a thin `Instant`/`Duration` layer over them — an ergonomic +front door, not new mechanism. + +```zig +const time = @import("runtime").time; + +const start = time.now(); // Instant — monotonic +doWork(); +const took = start.elapsed(); // Duration +time.sleep(time.Duration.fromMillis(5)); // block ~5 ms + +// A deadline delivered as a notification, so a service keeps serving meanwhile: +_ = time.after(endpoint, time.Duration.fromMillis(200)); +``` + +- `Duration` is nanoseconds under the hood, with `fromNanos/fromMicros/fromMillis/ + fromSeconds` and `asNanos/asMillis`. `ceilMillis` rounds *up* to the kernel's + millisecond granularity, so a sub-millisecond `sleep` never rounds down to zero and + returns early. All arithmetic saturates rather than wraps. +- `Instant` is a point on the monotonic clock: `since`, `elapsed`, `plus`, `reached` — + built for deadline loops (`while (!deadline.reached()) …`). +- `now()` / `monotonicNanos()` wrap `clock`. `available()` reports whether the clock is + calibrated at all (the kernel returns 0 until the TSC frequency is known, so a caller + that needs real time can treat 0 as "unavailable" rather than assume it advances). +- `sleep(d)` wraps `sleep`; `spin(d)` busy-polls `now()` for the sub-millisecond delays + the millisecond tick can't express; `after(endpoint, d)` wraps `timer_bind`. + +The raw wrappers (`system.clock`, `system.sleep`, `system.timerOnce`) stay in +`library/runtime/system.zig`; `runtime.time` is the layer meant for everyday use. + +## Wall-clock time (not built) + +Everything above is **monotonic**: elapsed time since boot, perfect for timeouts and +measurement, useless for "what is the date?" Calendar time — a real-time clock, time +zones, leap seconds — is genuinely a **user-space** concern, and it *is* the case a time +service is for. It would be backed by an **RTC** driver (the CMOS real-time clock), not +the HPET, and exposed as a `CLOCK_REALTIME`-style service alongside the monotonic +syscall. It is deferred until something needs it; the monotonic clock the kernel already +owns covers every current use. + +## Verifying it + +`runtime.time`'s `Instant`/`Duration` arithmetic has unit tests that run on the host: + +``` +$ zig build test # includes library/runtime/time.zig +``` + +End to end, the proof the clock is real is that it *advances*: read `now()`, `sleep` a +`Duration`, read `now()` again, and the second reading is later — the kernel's timer +driving a ring-3 program with no service in between. diff --git a/library/runtime/runtime.zig b/library/runtime/runtime.zig index efdc5a5..9cca735 100644 --- a/library/runtime/runtime.zig +++ b/library/runtime/runtime.zig @@ -12,6 +12,9 @@ //! (arguments arrive via `init`). pub const system = @import("system.zig"); +/// Monotonic time, delays, and deadlines over the kernel clock/sleep/timer syscalls +/// — an `Instant`/`Duration` front door, no time service (docs/timers.md). +pub const time = @import("time.zig"); pub const heap = @import("heap.zig"); pub const ipc = @import("ipc.zig"); pub const start = @import("start.zig"); @@ -21,7 +24,7 @@ pub const vfs_protocol = @import("vfs-protocol"); /// The device-manager protocol: hello + tree reports (docs/device-manager.md). pub const device_manager_protocol = @import("device-manager-protocol"); -/// The power protocol: events (button, lid, battery) + shutdown (docs/m21-plan.md). +/// The power protocol: events (button, lid, battery) + shutdown (docs/power.md). pub const power_protocol = @import("power-protocol"); /// Keyboard-event listening (subscribe/next) and broadcasting (publish), over the input /// service. See library/runtime/input.zig and system/services/input/. diff --git a/library/runtime/time.zig b/library/runtime/time.zig new file mode 100644 index 0000000..9ae4908 --- /dev/null +++ b/library/runtime/time.zig @@ -0,0 +1,169 @@ +//! The danos time interface — monotonic time, delays, and deadlines for user space. +//! +//! There is no time *service*: the kernel already owns the scheduling timer and +//! surfaces it directly, so reading the clock is one system call (an `rdtsc` and a +//! scale), never an IPC round trip (docs/timers.md explains why). This module is a +//! thin, generic layer over the `clock`/`sleep`/`timer_bind` wrappers in `system.zig` +//! — an ergonomic `Instant`/`Duration` front door, not new mechanism. +//! +//! It is **monotonic** time only: nanoseconds since boot, moving forward, no date or +//! timezone. Wall-clock/calendar time is a separate user-space service (an RTC-backed +//! CLOCK_REALTIME) layered on top later. + +const std = @import("std"); +const system = @import("system.zig"); + +const nanos_per_micro: u64 = 1_000; +const nanos_per_milli: u64 = 1_000_000; +const nanos_per_second: u64 = 1_000_000_000; + +/// A span of time, held as nanoseconds. Constructors name their unit; accessors +/// truncate toward zero. `ceilMillis` rounds *up*, since `sleep`/`after` land on the +/// kernel's millisecond granularity and rounding down could return early. +pub const Duration = struct { + ns: u64, + + pub fn fromNanos(n: u64) Duration { + return .{ .ns = n }; + } + pub fn fromMicros(n: u64) Duration { + return .{ .ns = n *| nanos_per_micro }; + } + pub fn fromMillis(n: u64) Duration { + return .{ .ns = n *| nanos_per_milli }; + } + pub fn fromSeconds(n: u64) Duration { + return .{ .ns = n *| nanos_per_second }; + } + + pub fn asNanos(d: Duration) u64 { + return d.ns; + } + pub fn asMicros(d: Duration) u64 { + return d.ns / nanos_per_micro; + } + pub fn asMillis(d: Duration) u64 { + return d.ns / nanos_per_milli; + } + pub fn asSeconds(d: Duration) u64 { + return d.ns / nanos_per_second; + } + + /// Whole milliseconds, rounded up — the argument `sleep`/`after` pass the kernel. + /// A non-zero sub-millisecond duration becomes 1 ms rather than 0. + pub fn ceilMillis(d: Duration) u64 { + return (d.ns +| (nanos_per_milli - 1)) / nanos_per_milli; + } + + pub fn plus(a: Duration, b: Duration) Duration { + return .{ .ns = a.ns +| b.ns }; + } +}; + +/// A point on the monotonic clock — nanoseconds since boot. Compare and subtract +/// instants to measure elapsed time; it never runs backward, so `since` is safe to +/// saturate at zero rather than wrap. +pub const Instant = struct { + ns: u64, + + /// The span from `earlier` to `self`, saturating at zero if `earlier` is later + /// (which the monotonic clock should never produce, but callers may pass any pair). + pub fn since(self: Instant, earlier: Instant) Duration { + return .{ .ns = self.ns -| earlier.ns }; + } + + /// How long since this instant, sampled now. + pub fn elapsed(self: Instant) Duration { + return now().since(self); + } + + /// This instant advanced by `d` (a deadline, `d` from here). + pub fn plus(self: Instant, d: Duration) Instant { + return .{ .ns = self.ns +| d.ns }; + } + + /// Whether the monotonic clock has reached this instant (used as a deadline). + pub fn reached(deadline: Instant) bool { + return now().ns >= deadline.ns; + } +}; + +/// The current monotonic time. +pub fn now() Instant { + return .{ .ns = system.clock() }; +} + +/// Monotonic nanoseconds since boot — the raw `clock()` reading, for callers that +/// want a plain integer instead of an `Instant`. +pub fn monotonicNanos() u64 { + return system.clock(); +} + +/// Whether the monotonic clock is usable. The kernel returns 0 until the TSC is +/// calibrated (`tsc_hz == 0`); a caller that needs real time can treat that as +/// "unavailable" instead of assuming the clock advances. +pub fn available() bool { + return system.clock() != 0; +} + +/// Block the caller for at least `d`, rounded up to the kernel's millisecond +/// granularity. For sub-millisecond precision the scheduler cannot express, use +/// `spin`. +pub fn sleep(d: Duration) void { + system.sleep(d.ceilMillis()); +} + +/// Block the caller for `ms` milliseconds — the coarse, allocation-free form. +pub fn sleepMillis(ms: u64) void { + system.sleep(ms); +} + +/// Busy-wait until `d` has elapsed, polling the monotonic clock. This burns the CPU +/// on purpose, to hit sub-millisecond delays the scheduler's millisecond tick cannot. +/// Prefer `sleep` for anything at or above a millisecond. +pub fn spin(d: Duration) void { + const deadline = now().plus(d); + while (!deadline.reached()) {} +} + +/// Arm a one-shot timer against `endpoint` (a handle from `ipc.createIpcEndpoint`): +/// after `d` the kernel posts a timer notification (`ipc.Received.isTimer`) there. +/// Unlike `sleep`, this does not block — a service can keep serving IPC on the same +/// endpoint while the deadline is pending. Rounds `d` up to milliseconds; returns +/// false if the timer could not be armed. See `system.timerOnce`. +pub fn after(endpoint: usize, d: Duration) bool { + return system.timerOnce(endpoint, d.ceilMillis()); +} + +test "Duration unit conversions round toward zero" { + try std.testing.expectEqual(@as(u64, 1_000_000_000), Duration.fromSeconds(1).asNanos()); + try std.testing.expectEqual(@as(u64, 1_500), Duration.fromNanos(1_500).asNanos()); + try std.testing.expectEqual(@as(u64, 2), Duration.fromMillis(2).asMillis()); + try std.testing.expectEqual(@as(u64, 1), Duration.fromNanos(1_999_999).asMillis()); + try std.testing.expectEqual(@as(u64, 250), Duration.fromMicros(250).asMicros()); +} + +test "ceilMillis rounds up, and never turns a nonzero span into zero" { + try std.testing.expectEqual(@as(u64, 0), Duration.fromNanos(0).ceilMillis()); + try std.testing.expectEqual(@as(u64, 1), Duration.fromNanos(1).ceilMillis()); + try std.testing.expectEqual(@as(u64, 1), Duration.fromMillis(1).ceilMillis()); + try std.testing.expectEqual(@as(u64, 2), Duration.fromNanos(nanos_per_milli + 1).ceilMillis()); + try std.testing.expectEqual(@as(u64, 5), Duration.fromMillis(5).ceilMillis()); +} + +test "Instant arithmetic: since saturates, plus/reached form deadlines" { + const t0 = Instant{ .ns = 1_000 }; + const t1 = Instant{ .ns = 4_000 }; + try std.testing.expectEqual(@as(u64, 3_000), t1.since(t0).asNanos()); + // earlier-than-self can't happen on a monotonic clock; saturate rather than wrap. + try std.testing.expectEqual(@as(u64, 0), t0.since(t1).asNanos()); + const deadline = t0.plus(Duration.fromNanos(2_500)); + try std.testing.expectEqual(@as(u64, 3_500), deadline.ns); +} + +test "saturating arithmetic does not overflow at the u64 ceiling" { + const big = Duration.fromSeconds(std.math.maxInt(u64)); + try std.testing.expectEqual(@as(u64, std.math.maxInt(u64)), big.asNanos()); + const late = Instant{ .ns = std.math.maxInt(u64) }; + try std.testing.expectEqual(@as(u64, std.math.maxInt(u64)), late.plus(Duration.fromSeconds(10)).ns); +} diff --git a/system/abi.zig b/system/abi.zig index 657f4e4..61f37e5 100644 --- a/system/abi.zig +++ b/system/abi.zig @@ -177,7 +177,7 @@ pub const ServiceId = enum(u32) { input = 2, ps2_bus = 3, // the 8042 owner; child device drivers attach here for raw bytes device_manager = 4, // the tree, the matcher, the supervisor (docs/device-manager.md) - power = 5, // system power: events (button, lid, battery) + shutdown (docs/m21-plan.md; domain-named per decision 7 — the acpi service registers it on x86, a PSCI service will on ARM) + power = 5, // system power: events (button, lid, battery) + shutdown (docs/power.md; domain-named per docs/discovery.md — the acpi service registers it on x86, a PSCI service will on ARM) _, }; diff --git a/system/devices/acpi.zig b/system/devices/acpi.zig index c1a4db0..3c1f215 100644 --- a/system/devices/acpi.zig +++ b/system/devices/acpi.zig @@ -159,7 +159,7 @@ pub var dsdt_physical: u64 = 0; /// The FADT itself (physical + length), published on the acpi-tables node so /// the ring-3 acpi service can read the PM1 event and GPE blocks it needs for -/// the event side (docs/m21-plan.md decision 3). Distinguished from the AML +/// the event side (docs/acpi.md — ACPI events). Distinguished from the AML /// blob resources by its intact "FACP" header — the blobs are header-stripped. var fadt_physical: u64 = 0; var fadt_length: u64 = 0; @@ -423,7 +423,7 @@ pub fn discover(rsdp_physical: u64, memory_regions: []const boot_handoff.MemoryR // whatever the FADT alone provided. } - // Publish the acpi-tables node (docs/m19-m20-plan.md M20): the AML blobs as + // Publish the acpi-tables node (docs/discovery.md): the AML blobs as // memory resources for the acpi service to map and parse in ring 3, a broad // io_port grant for the OperationRegion access its interpreter needs, and // the SCI for the events track (M21). Exactly one node, one trusted @@ -448,7 +448,7 @@ fn publishAcpiTablesNode(device_tree: *DeviceTree) !void { // The broad I/O grant: OperationRegions name whatever ports the firmware // chose (EC, PM1, GPE, SMBus); which ports cannot be known before the AML // that names them is parsed, so the grant is the whole space — the honest - // trust boundary of docs/m19-m20-plan.md decision 5. + // trust boundary of docs/discovery.md (the acpi service's one trusted node). _ = node.addResource(.io_port, 0, 1 << 16); // A broad interrupt window: ACPI _CRS names legacy ISA IRQs (the PS/2 lines // 1 and 12, the RTC, …), and the service registers those devices under this @@ -610,7 +610,7 @@ fn parseMcfg(device_tree: *DeviceTree, header: *const SystemDescriptorTableHeade var boot_memory_regions: []const boot_handoff.MemoryRegion = &.{}; /// The bridge's MMIO apertures, derived from the boot memory map's holes -/// (docs/m19-m20-plan.md decision 2): registered PCI functions carry BAR +/// (docs/discovery.md — apertures from the memory map): registered PCI functions carry BAR /// resources, and `device_register` containment demands the bridge own windows /// that cover them. Everything the firmware described is "not hole"; the low /// aperture runs from the end of the described space below 4 GiB up to the diff --git a/system/devices/aml/aml.zig b/system/devices/aml/aml.zig index 8206c07..c595f6a 100644 --- a/system/devices/aml/aml.zig +++ b/system/devices/aml/aml.zig @@ -56,7 +56,7 @@ pub fn parse(allocator: std.mem.Allocator, blocks: []const []const u8) !ParseRes } /// Count the Device objects in a parsed namespace — what the acpi service -/// (docs/m19-m20-plan.md M20) reports, and what the kernel's own parse counts +/// (docs/discovery.md) reports, and what the kernel's own parse counts /// so the two can be checked equal across the ring-3 move. pub fn deviceCount(namespace: *const Namespace) usize { return countKind(namespace.root, .device); diff --git a/system/devices/device-abi.zig b/system/devices/device-abi.zig index 33cf15e..4507e38 100644 --- a/system/devices/device-abi.zig +++ b/system/devices/device-abi.zig @@ -29,7 +29,7 @@ pub const DeviceClass = enum(u32) { /// hardware ID (`_HID`) and, where static, current resource settings (`_CRS`). acpi_device, /// The ACPI tables themselves, published as one node for the user-space acpi - /// service (docs/m19-m20-plan.md M20): memory resources over the AML blobs, + /// service (docs/discovery.md): memory resources over the AML blobs, /// a broad io_port grant for OperationRegion access, and the SCI interrupt. /// The one node whose claimant is trusted to run firmware bytecode. acpi_tables, diff --git a/system/drivers/bus/bus.zig b/system/drivers/bus/bus.zig deleted file mode 100644 index 9203290..0000000 --- a/system/drivers/bus/bus.zig +++ /dev/null @@ -1,211 +0,0 @@ -//! /system/drivers/bus — a user-space **bus driver**, and the smallest honest example of one. -//! -//! A bus driver owns a device that *contains other devices*, enumerates them by some -//! bus-specific protocol, and publishes each one into the kernel's device table so a -//! class driver can claim it. PCI walks configuration space; USB walks hub descriptors. Here -//! the "bus" is the HPET's register block and the "devices" are its comparators, each -//! a 0x20-byte window at 0x100 + 0x20*n that can be driven independently. -//! -//! It's a toy bus, but nothing about the mechanism is: `bus` reads how many children -//! exist from the hardware (GENERAL_CAP bits [12:8]), publishes one `DeviceDescriptor` per -//! child with a sub-window of its own MMIO plus the shared IRQ, and the kernel checks -//! every one of those resources is contained in what `bus` was granted. A comparator -//! driver then claims a child and maps only *its* registers — not the whole block. -//! -//! It also proves the negative: registering a child whose window escapes the parent's -//! is refused. Without that check, `device_register` would be a system_call for mapping -//! arbitrary physical memory. - -const std = @import("std"); -const runtime = @import("runtime"); -const device = runtime.device; - -const register_general_cap = 0x000; - -/// Comparator n's registers: configuration+comparator+FSB route, 0x20 bytes. -fn timerWindow(hpet_base: u64, n: u64) device.ResourceDescriptor { - return .{ - .kind = @intFromEnum(device.ResourceKind.memory), - .start = hpet_base + 0x100 + 0x20 * n, - .len = 0x20, - }; -} - -fn findHpet(buffer: []device.DeviceDescriptor) ?device.DeviceDescriptor { - const total = device.enumerate(buffer); - const n = @min(total, buffer.len); - for (buffer[0..n]) |d| { - if (d.class != @intFromEnum(device.DeviceClass.timer)) continue; - if (d.parent != device.no_parent) continue; // the block, not a comparator child - for (0..d.resource_count) |j| { - if (d.resources[j].kind == @intFromEnum(device.ResourceKind.memory)) return d; - } - } - return null; -} - -/// The parent's MMIO resource, and its IRQ if it has one. -fn resourcesOf(d: device.DeviceDescriptor) struct { mmio: device.ResourceDescriptor, irq: ?device.ResourceDescriptor } { - var mmio: device.ResourceDescriptor = undefined; - var irq: ?device.ResourceDescriptor = null; - for (0..d.resource_count) |j| { - const r = d.resources[j]; - if (r.kind == @intFromEnum(device.ResourceKind.memory)) mmio = r; - if (r.kind == @intFromEnum(device.ResourceKind.irq)) irq = r; - } - return .{ .mmio = mmio, .irq = irq }; -} - -fn firstChildOf(buffer: []device.DeviceDescriptor, total: usize, parent_id: u64) ?u64 { - for (buffer[0..@min(total, buffer.len)]) |d| { - if (d.parent == parent_id) return d.id; - } - return null; -} - -pub fn main() void { - const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch { - _ = runtime.system.write("bus: out of memory\n"); - return; - }; - - const parent = findHpet(buffer) orelse { - _ = runtime.system.write("bus: no HPET\n"); - return; - }; - const resource = resourcesOf(parent); - - // Claim the bus. Everything below is subdivision of what this claim granted. - // - // Claims are exclusive, and at a normal boot the kernel spawns every initial_ramdisk - // binary — so hpet may own the HPET already. That's not an error, it's the - // capability model working: exit quietly and leave the device to its owner. The - // `bus` test spawns bus alone, so there it wins the claim. - if (!device.claim(parent.id)) { - _ = runtime.system.write("bus: HPET already claimed by another driver, nothing to do\n"); - return; - } - - // Enumerate the bus: ask the hardware how many children it has. - const base = device.mmioMap(parent.id, 0) orelse { - _ = runtime.system.write("bus: mmio_map failed\n"); - return; - }; - const cap: *volatile u64 = @ptrFromInt(base + register_general_cap); - const n_children = ((cap.* >> 8) & 0x1F) + 1; - - // Publish one child per comparator, each owning only its own window. - var published: u64 = 0; - var n: u64 = 0; - while (n < n_children) : (n += 1) { - var child = std.mem.zeroes(device.DeviceDescriptor); - child.class = @intFromEnum(device.DeviceClass.timer); - child.pci_class = device.no_pci_class; - child.hid_len = 6; - child.hid[0..6].* = "hpet-t".*; - child.resource_count = 1; - child.resources[0] = timerWindow(resource.mmio.start, n); - // Comparators share the block's interrupt line; only one child can bind it, - // but all of them may legitimately name it. - if (resource.irq) |i| { - child.resources[child.resource_count] = i; - child.resource_count += 1; - } - - if (device.register(parent.id, &child) == null) { - _ = runtime.system.write("bus: register failed\n"); - return; - } - published += 1; - } - - // The negative case. A window one byte past the end of the parent's must be - // refused — otherwise device_register would be "map any physical page you like". - // Confirm the table did not grow, not merely that the call returned null: null - // also means NoSpace/BadParent, so a size check is what actually proves the - // *containment* rule fired. - const before = device.enumerate(buffer); - var rogue = std.mem.zeroes(device.DeviceDescriptor); - rogue.class = @intFromEnum(device.DeviceClass.unknown); - rogue.resource_count = 1; - rogue.resources[0] = .{ - .kind = @intFromEnum(device.ResourceKind.memory), - .start = resource.mmio.start + resource.mmio.len, - .len = 0x1000, - }; - if (device.register(parent.id, &rogue) != null) { - _ = runtime.system.write("bus: FAIL out-of-window child was accepted\n"); - return; - } - if (device.enumerate(buffer) != before) { - _ = runtime.system.write("bus: FAIL rogue child leaked into the table\n"); - return; - } - - // And confirm the children came back with the right parent and a *narrower* - // window than the bus — read from the table, not from our own memory. - const total = device.enumerate(buffer); - var seen: u64 = 0; - for (buffer[0..@min(total, buffer.len)]) |d| { - if (d.parent != parent.id) continue; - const w = d.resources[0]; - if (w.start < resource.mmio.start or w.len >= resource.mmio.len) { - _ = runtime.system.write("bus: FAIL child window is not inside the bus\n"); - return; - } - seen += 1; - } - if (seen != published) { - _ = runtime.system.write("bus: FAIL child count mismatch\n"); - return; - } - - // Delegation, end to end: claim a child and map *it*. A real class driver would be - // a different process; here bus plays both parts, which exercises the same path. - // The child's window is 0x20 bytes at parent+0x100, so the register it sees at - // offset 0 must be the same timer-0 configuration register the bus sees at 0x100. - // - // (mmio_map rounds to a page, so the child's mapping physically covers the whole - // 4 KiB the HPET lives in — the granularity limit documented in docs/drivers.md. - // The *resource* is narrow even though the page isn't.) - const child_id = firstChildOf(buffer, device.enumerate(buffer), parent.id) orelse { - _ = runtime.system.write("bus: FAIL no child to claim\n"); - return; - }; - if (!device.claim(child_id)) { - _ = runtime.system.write("bus: FAIL could not claim own child\n"); - return; - } - const child_base = device.mmioMap(child_id, 0) orelse { - _ = runtime.system.write("bus: FAIL child mmio_map refused\n"); - return; - }; - const via_child: *volatile u64 = @ptrFromInt(child_base); - const via_bus: *volatile u64 = @ptrFromInt(base + 0x100); - if (via_child.* != via_bus.*) { - _ = runtime.system.write("bus: FAIL child window does not alias the bus register\n"); - return; - } - - // A descriptor pointer into an unmapped page must fail the call, not fault the - // kernel. Grab a page, free it, and register through the stale address: if the - // kernel dereferenced it raw (rather than copying in through the page tables) this - // would triple-fault QEMU and the test would time out instead of printing ok. - const scratch = runtime.system.mmap(0x1000, runtime.system.PROT_READ | runtime.system.PROT_WRITE); - if (!runtime.system.mmapFailed(scratch)) { - _ = runtime.system.munmap(scratch, 0x1000); - const descriptor: *const device.DeviceDescriptor = @ptrFromInt(scratch); - if (device.register(parent.id, descriptor) != null) { - _ = runtime.system.write("bus: FAIL register accepted an unmapped descriptor\n"); - return; - } - } - - _ = runtime.system.write("bus: ok\n"); - while (true) runtime.system.sleep(1000); -} - -pub const panic = runtime.panic; -comptime { - _ = &runtime.start._start; -} diff --git a/system/drivers/hpet/hpet.zig b/system/drivers/hpet/hpet.zig deleted file mode 100644 index 3b466a4..0000000 --- a/system/drivers/hpet/hpet.zig +++ /dev/null @@ -1,195 +0,0 @@ -//! /system/drivers/hpet — a user-space HPET driver. It proves the whole driver model end to -//! end: enumerate the device table, find the HPET, claim it, map its registers into -//! this ring-3 address space (strong-uncacheable), **bind its interrupt to an IPC -//! endpoint**, then sit blocked in `replyWait` until the hardware wakes it. -//! -//! Nothing here polls. Between interrupts the process is `.blocked` and off every -//! scheduler queue; the core runs other work or idles. That is the point of the -//! exercise — a driver is a process that sleeps until its device has something to -//! say (see docs/drivers.md). -//! -//! The comparator is configured **level-triggered** on purpose. Edge would be -//! simpler, but level is the discipline every real device line needs, and it forces -//! the full cycle to be correct: -//! -//! kernel ISR mask the GSI -> EOI -> notify this endpoint -//! hpet wake, clear GENERAL_INT_STATUS (deasserts the line), re-arm -//! hpet irq_ack -> kernel unmasks the GSI -//! -//! Clear the status bit *before* acking, or the line is still asserted when the -//! kernel unmasks and the I/O APIC redelivers forever. -//! -//! Register map (HPET spec 1.0a): -//! 0x000 GENERAL_CAP [63:32] fs per tick, [12:8] number timers - 1 -//! 0x010 GENERAL_CONFIGURATION bit0 ENABLE_CNF, bit1 LEG_RT_CNF -//! 0x020 GENERAL_INT_STATUS bit n = timer n asserted (write 1 to clear) -//! 0x0F0 MAIN_COUNTER -//! 0x100 TIMER0_CONFIGURATION bit1 INT_TYPE(1=level) bit2 INT_ENB bit3 TYPE(periodic) -//! bits[13:9] INT_ROUTE, [63:32] INT_ROUTE_CAP -//! 0x108 TIMER0_COMPARATOR - -const runtime = @import("runtime"); -const mmio = @import("mmio"); -const device = runtime.device; -const ipc = runtime.ipc; - -const register_general_cap = 0x000; -const register_general_configuration = 0x010; -const register_int_status = 0x020; -const register_main_counter = 0x0F0; -const register_timer0_configuration = 0x100; -const register_timer0_comparator = 0x108; - -const configuration_enable: u64 = 1 << 0; // GENERAL_CONFIGURATION.ENABLE_CNF -const configuration_leg_rt: u64 = 1 << 1; // GENERAL_CONFIGURATION.LEG_RT_CNF -const tn_int_type_level: u64 = 1 << 1; -const tn_int_enb: u64 = 1 << 2; -const tn_type_periodic: u64 = 1 << 3; -const tn_route_shift = 9; -const tn_route_mask: u64 = 0x1F << tn_route_shift; - -/// Interrupts to observe before declaring victory. -const target_ticks = 5; - -/// Read/write a 64-bit HPET register through the typed volatile MMIO layer (/lib/mmio). -/// The HPET is pure MMIO with no DMA, and on x86 its grant is strong-uncacheable (so -/// UC writes are already ordered) — no barriers are needed here; the point is the -/// typed, arch-portable access every driver should use. -inline fn rd(base: usize, off: usize) u64 { - return mmio.read(u64, base + off); -} -inline fn wr(base: usize, off: usize, value: u64) void { - mmio.write(u64, base + off, value); -} - -/// A timer-class device exposing both an MMIO window and an IRQ: its id, the two -/// resource indices, and the GSI discovery chose out of `Tn_INT_ROUTE_CAP`. -const Found = struct { device_id: u64, mmio: u64, irq: u64, gsi: u64 }; - -fn findHpet(buffer: []device.DeviceDescriptor) ?Found { - const total = device.enumerate(buffer); - const n = @min(total, buffer.len); - for (buffer[0..n]) |d| { - if (d.class != @intFromEnum(device.DeviceClass.timer)) continue; - // Skip comparator children a bus driver may have published below the block - // (see system/drivers/bus/bus.zig) — we want the register block itself. - if (d.parent != device.no_parent) continue; - var mmio_index: ?u64 = null; - var irq: ?u64 = null; - for (0..d.resource_count) |j| { - switch (d.resources[j].kind) { - @intFromEnum(device.ResourceKind.memory) => mmio_index = mmio_index orelse j, - @intFromEnum(device.ResourceKind.irq) => irq = irq orelse j, - else => {}, - } - } - if (mmio_index) |m| if (irq) |i| { - return .{ .device_id = d.id, .mmio = m, .irq = i, .gsi = d.resources[i].start }; - }; - } - return null; -} - -pub fn main() void { - // Enumerate into a heap buffer (too big for the one-page user stack). - const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 32) catch { - _ = runtime.system.write("system/drivers/hpet: out of memory\n"); - return; - }; - - const hpet = findHpet(buffer) orelse { - _ = runtime.system.write("system/drivers/hpet: no HPET with an IRQ\n"); - return; - }; - - if (!device.claim(hpet.device_id)) { - _ = runtime.system.write("system/drivers/hpet: claim failed\n"); - return; - } - const base = device.mmioMap(hpet.device_id, hpet.mmio) orelse { - _ = runtime.system.write("system/drivers/hpet: mmio_map failed\n"); - return; - }; - - // The GSI discovery picked for us out of Tn_INT_ROUTE_CAP. Program the comparator - // to raise exactly this line — the kernel will only bind the one it recorded. - const gsi = hpet.gsi; - - const endpoint = ipc.createIpcEndpoint() orelse { - _ = runtime.system.write("system/drivers/hpet: create_ipc_endpoint failed\n"); - return; - }; - - // --- program the hardware ------------------------------------------------ - // Counter period, so we can arm the comparator a fixed wall-clock distance out. - const femtos_per_tick = rd(base, register_general_cap) >> 32; - if (femtos_per_tick == 0) { - _ = runtime.system.write("system/drivers/hpet: bad HPET period\n"); - return; - } - const ticks_per_ms = 1_000_000_000_000 / femtos_per_tick; - - // Stop the counter and take the legacy route off while we reconfigure. - wr(base, register_general_configuration, rd(base, register_general_configuration) & ~(configuration_enable | configuration_leg_rt)); - - // Timer 0: one-shot, level-triggered, routed to our GSI, interrupt enabled. - // One-shot (not periodic) sidesteps the HPET's Tn_value_SET accumulator quirk — - // we simply re-arm from the driver on each interrupt, which is what a tickless - // timer driver does anyway. - var t0 = rd(base, register_timer0_configuration); - t0 &= ~(tn_route_mask | tn_type_periodic); - t0 |= tn_int_type_level | tn_int_enb | (gsi << tn_route_shift); - wr(base, register_timer0_configuration, t0); - - // Clear any stale assertion, then arm ~100 ms out and start the counter. - wr(base, register_int_status, 1); - wr(base, register_timer0_comparator, rd(base, register_main_counter) + ticks_per_ms * 100); - wr(base, register_general_configuration, rd(base, register_general_configuration) | configuration_enable); - - if (!device.irqBind(hpet.device_id, hpet.irq, endpoint)) { - _ = runtime.system.write("system/drivers/hpet: irq_bind failed\n"); - return; - } - _ = runtime.system.write("system/drivers/hpet: bound, sleeping until the hardware speaks\n"); - - // --- the driver loop ----------------------------------------------------- - // Blocked in replyWait. No polling, no spinning: the next line of this function - // runs only because an interrupt fired. - var receive: [64]u8 = undefined; - var count: usize = 0; - while (count < target_ticks) { - // Blocked here. The task is `.blocked` and off every scheduler queue; the - // next line runs only because the HPET raised its line. - const r = ipc.replyWait(endpoint, &.{}, &receive, null); - if (!r.isNotification()) continue; // a client request, not our IRQ - - // Quiet the device: write 1 to timer 0's status bit. Until this lands, the - // line is still asserted and unmasking would refire immediately. - wr(base, register_int_status, 1); - count += 1; - - if (count < target_ticks) { - wr(base, register_timer0_comparator, rd(base, register_main_counter) + ticks_per_ms * 100); - } else { - // Last one: stop the source rather than re-arming, so the line is left - // both quiet *and* unmasked by the ack below. Re-arming here would leave - // a pending interrupt that nobody is waiting for, and the ISR would mask - // the line again a moment later. - wr(base, register_timer0_configuration, rd(base, register_timer0_configuration) & ~tn_int_enb); - } - - _ = runtime.system.write("system/drivers/hpet: irq\n"); - if (!device.irqAck(hpet.device_id, hpet.irq)) { - _ = runtime.system.write("system/drivers/hpet: irq_ack failed\n"); - return; - } - } - - _ = runtime.system.write("system/drivers/hpet: ok\n"); - while (true) runtime.system.sleep(1000); -} - -pub const panic = runtime.panic; -comptime { - _ = &runtime.start._start; -} diff --git a/system/drivers/pci-bus/pci-bus.zig b/system/drivers/pci-bus/pci-bus.zig index 5d03e0e..2d6bd6f 100644 --- a/system/drivers/pci-bus/pci-bus.zig +++ b/system/drivers/pci-bus/pci-bus.zig @@ -1,5 +1,5 @@ //! /system/drivers/pci-bus — the PCI bus driver: enumeration moved out of ring 0 -//! (docs/m19-m20-plan.md, M19). The device manager matches the `pci_host_bridge` +//! (docs/discovery.md). The device manager matches the `pci_host_bridge` //! node and spawns one instance per bridge, the bridge's device id as argv[1] — //! the same per-device contract as usb-xhci-bus. //! diff --git a/system/kernel/architecture/x86_64/apic.zig b/system/kernel/architecture/x86_64/apic.zig index e24fc93..73efa17 100644 --- a/system/kernel/architecture/x86_64/apic.zig +++ b/system/kernel/architecture/x86_64/apic.zig @@ -80,6 +80,35 @@ var timer_hz: u32 = 0; var tsc_hz: u64 = 0; var tsc_base: u64 = 0; +/// Whether the TSC is architecturally **invariant** — a constant rate regardless of +/// P/C-state transitions, and thus valid as a clocksource (CPUID leaf 0x80000007, +/// EDX bit 8). AMD and modern Intel set it; the bare qemu64 model does not. Measured +/// frequency alone is not enough: a non-invariant TSC speeds up and slows down with +/// the core clock, so reading it as wall time would drift. +var tsc_invariant: bool = false; +/// Cleared if the cross-core warp check (checkWarpSource) ever sees the TSC read +/// lower on one core than the max another core has already published — i.e. the +/// per-core TSCs are not synchronized, and a task migrating cores could see time go +/// backward. Starts true (assume synchronized until proven otherwise). +var tsc_synced: bool = true; +/// The worst backward skew the warp check observed, in TSC cycles (0 = none). +var tsc_warp_cycles: u64 = 0; + +/// The monotonic clock's source. The TSC when it is invariant *and* synchronized — +/// the fast `rdtsc` path taken on real Intel/AMD and modern VMs. Otherwise the HPET +/// main counter: a single fixed-rate counter, immune to both per-core skew and +/// frequency scaling, so it stays accurate on a bare VM or a warped machine. +const ClockSource = enum { tsc, hpet }; +var clock_source: ClockSource = .tsc; + +/// HPET standby clocksource, set up in calibrate() whenever an HPET exists (whether +/// or not calibration itself measured against it): its frequency, the counter value +/// chosen as the zero point, and its width mask. Only a 64-bit HPET is used as a +/// clocksource — a 32-bit one wraps too fast to be monotonic without accumulation. +var hpet_clock_hz: u64 = 0; +var hpet_clock_base: u64 = 0; +var hpet_clock_mask: u64 = ~@as(u64, 0); + /// Read the 64-bit Time Stamp Counter. fn rdtsc() u64 { var low: u32 = undefined; @@ -220,6 +249,31 @@ pub fn calibrate() void { } tsc_base = rdtsc(); // the clock's zero point (boot) + + // Decide whether the TSC is trustworthy as a clocksource. Frequency (measured + // above, possibly against the HPET/PIT) is necessary but not sufficient: the TSC + // must also be *invariant* (CPUID 0x80000007 EDX[8]). AMD and modern Intel set + // this; the bare qemu64 model does not. + tsc_invariant = tscIsInvariant(); + + // Bring up the HPET as a standby clocksource whenever one exists — even on the + // CPUID-0x15 path where calibration never touched it — so a non-invariant TSC + // (here) or an unsynchronized one (checkWarpSource, during SMP bring-up) can fall + // back to a source that is immune to both. hpetHz() maps + enables the counter + // and is idempotent if calibration already used it. + if (configuration_hpet_base != 0) { + if (hpetHz()) |hz| { + hpet_clock_mask = hpetMask(); + if (hpet_clock_mask == ~@as(u64, 0)) { // only a 64-bit HPET is monotonic enough + hpet_clock_hz = hz; + hpet_clock_base = readHpet(); + } + } + } + + // Select the source: the fast TSC when invariant, else the HPET if we have one. + // (checkWarpSource may still demote TSC -> HPET later if the cores' TSCs skew.) + if (!tsc_invariant and hpet_clock_hz != 0) clock_source = .hpet; } /// Run the LAPIC timer one-shot from its maximum count while a monotonic reference @@ -275,7 +329,7 @@ fn calibratePit() void { // --- reference clocks ------------------------------------------------------ /// TSC frequency from CPUID leaf 0x15 (crystal_hz * numerator / denominator), or -/// null if the CPU doesn't enumerate it (common under QEMU). +/// null if the CPU doesn't enumerate it (common under QEMU, and on AMD). fn cpuidTscHz() ?u64 { if (cpuid(0).eax < 0x15) return null; const r = cpuid(0x15); @@ -283,6 +337,15 @@ fn cpuidTscHz() ?u64 { return @as(u64, r.ecx) * r.ebx / r.eax; } +/// Whether the CPU advertises an **invariant** TSC (CPUID leaf 0x80000007, EDX +/// bit 8) — the architectural guarantee, on both Intel and AMD, that the TSC ticks +/// at a constant rate across P/C-states and never stops. Requires the extended-leaf +/// range to reach 0x80000007 first. +fn tscIsInvariant() bool { + if (cpuid(0x80000000).eax < 0x80000007) return false; + return (cpuid(0x80000007).edx & (1 << 8)) != 0; +} + const CpuidRegs = struct { eax: u32, ebx: u32, ecx: u32, edx: u32 }; fn cpuid(leaf: u32) CpuidRegs { @@ -310,11 +373,20 @@ fn hpetWrite64(off: usize, value: u64) void { @as(*volatile u64, @ptrFromInt(configuration_hpet_base + off)).* = value; } +/// Whether the HPET has been mapped into the physmap yet, so `configuration_hpet_base` +/// already holds the virtual address. `hpetHz` is called more than once (calibration +/// may use the HPET, and the standby-clocksource setup asks for it again), and mapping +/// an already-mapped base a second time would double-offset it into an overflow. +var hpet_mapped: bool = false; + /// Map + enable the HPET and return its tick frequency, or null if unusable. /// Maps the HPET into the physmap and switches configuration_hpet_base to that virtual -/// address, so the register accessors reach it without the identity map. +/// address, so the register accessors reach it without the identity map. Idempotent. fn hpetHz() ?u64 { - configuration_hpet_base = paging.mapMmio(configuration_hpet_base, 0x400, true); + if (!hpet_mapped) { + configuration_hpet_base = paging.mapMmio(configuration_hpet_base, 0x400, true); + hpet_mapped = true; + } const caps = hpetRead64(0x00); const period_fs = caps >> 32; // femtoseconds per tick if (period_fs == 0) return null; @@ -364,24 +436,169 @@ pub fn tscHz() u64 { return tsc_hz; } -// Monotonic high-resolution clock, from the TSC. A function per resolution, each -// scaling the cycle delta directly at its unit (the 128-bit intermediate avoids -// overflow across a long uptime). nanos() resolves to a few ns; millis() is what -// the scheduler uses for sleep deadlines. +// Monotonic high-resolution clock. A function per resolution, each scaling the +// counter delta directly at its unit (the 128-bit intermediate avoids overflow +// across a long uptime). nanos() resolves to a few ns on the TSC; millis() is what +// the scheduler uses for sleep deadlines. The source is the TSC when it is invariant +// and synchronized, else the HPET counter (see clock_source) — the branch is one +// global load and the TSC path is unchanged from before. + +/// The selected source's counter delta since its zero point. +fn clockCount() u64 { + return switch (clock_source) { + .tsc => rdtsc() -% tsc_base, + // A 64-bit HPET (the only kind we select) never wraps in any realistic + // uptime, so the wrapping subtraction is exact. + .hpet => readHpet() -% hpet_clock_base, + }; +} + +/// The selected source's frequency (0 if the clock is unavailable/uncalibrated). +fn clockHertz() u64 { + return switch (clock_source) { + .tsc => tsc_hz, + .hpet => hpet_clock_hz, + }; +} pub fn nanos() u64 { - if (tsc_hz == 0) return 0; - return @intCast(@as(u128, rdtsc() -% tsc_base) * 1_000_000_000 / tsc_hz); + const hz = clockHertz(); + if (hz == 0) return 0; + return @intCast(@as(u128, clockCount()) * 1_000_000_000 / hz); } pub fn micros() u64 { - if (tsc_hz == 0) return 0; - return @intCast(@as(u128, rdtsc() -% tsc_base) * 1_000_000 / tsc_hz); + const hz = clockHertz(); + if (hz == 0) return 0; + return @intCast(@as(u128, clockCount()) * 1_000_000 / hz); } pub fn millis() u64 { - if (tsc_hz == 0) return 0; - return @intCast(@as(u128, rdtsc() -% tsc_base) * 1_000 / tsc_hz); + const hz = clockHertz(); + if (hz == 0) return 0; + return @intCast(@as(u128, clockCount()) * 1_000 / hz); +} + +/// Whether the CPU advertises an invariant TSC (CPUID 0x80000007 EDX[8]). +pub fn tscInvariant() bool { + return tsc_invariant; +} + +/// Test hook: force the TSC clocksource on, as if the CPU had advertised an invariant +/// TSC. QEMU's TCG accelerator (the only one for an x86 guest on an Apple-Silicon +/// host) does not expose the invariant-TSC bit — its emulated TSC isn't invariant — so +/// the tsc-sync test can't reach the real-Intel/AMD/KVM path through CPUID. This lets +/// that test exercise the TSC clocksource and the cross-core warp check anyway. tsc_base +/// is left as-is so the switch from the HPET is continuous. +pub fn forceTscClocksourceForTest() void { + tsc_invariant = true; + clock_source = .tsc; +} + +/// How many per-AP warp checks actually ran (a rendezvous completed) — lets a test +/// confirm the cross-core check executed rather than being skipped. +pub fn warpChecksRun() u32 { + return warp_checks; +} + +/// Whether the per-core TSCs are synchronized (no backward warp seen at bring-up). +pub fn tscSynced() bool { + return tsc_synced; +} + +/// The active monotonic clocksource, for the boot log and tests. +pub fn clockSourceName() []const u8 { + return switch (clock_source) { + .tsc => "tsc", + .hpet => "hpet", + }; +} + +// --- cross-core TSC synchronization ("warp") check ------------------------- +// Two cores hammer a shared "max seen" TSC value under a lock; if either reads a +// value below that max, its TSC lags the other's, and time would run backward for a +// task migrating between them (Linux calls this a warp). danos brings APs up one at a +// time, so this runs pairwise: the BSP (source) against each AP (target) as it comes +// online. It only matters — and only runs — while the TSC is the clocksource; on a +// machine already on the HPET (a bare VM) the whole rendezvous is skipped. + +var warp_lock: u32 = 0; +var warp_last: u64 = 0; +var warp_bsp_ready: u32 = 0; +var warp_ap_ready: u32 = 0; +var warp_stop: u32 = 0; +var warp_checks: u32 = 0; // completed per-AP rendezvous count (for the tsc-sync test) + +const warp_rounds: u32 = 1 << 20; // locked reads on the BSP: ~1 ms at GHz rates +const warp_spin_limit: u64 = 1 << 32; // bound every rendezvous wait so a lost core can't hang boot + +fn warpTick() void { + while (@cmpxchgWeak(u32, &warp_lock, 0, 1, .acquire, .monotonic) != null) asm volatile ("pause"); + const t = rdtsc(); + if (t < warp_last) { + const delta = warp_last - t; + if (delta > tsc_warp_cycles) tsc_warp_cycles = delta; + tsc_synced = false; + } else { + warp_last = t; + } + @atomicStore(u32, &warp_lock, 0, .release); +} + +/// Spin (bounded) until `flag` is nonzero; false on timeout. +fn warpAwait(flag: *u32) bool { + var spins: u64 = 0; + while (@atomicLoad(u32, flag, .acquire) == 0) : (spins += 1) { + if (spins >= warp_spin_limit) return false; + asm volatile ("pause"); + } + return true; +} + +/// BSP side of the pairwise TSC warp check, run once per AP as it reports in. No-op +/// unless the TSC is the active clocksource. If the AP's TSC proves to lag, demote +/// the monotonic clock to the HPET without a discontinuity. +pub fn checkWarpSource() void { + if (clock_source != .tsc) return; + warp_last = 0; + @atomicStore(u32, &warp_stop, 0, .release); + @atomicStore(u32, &warp_ap_ready, 0, .release); + @atomicStore(u32, &warp_bsp_ready, 1, .release); + if (!warpAwait(&warp_ap_ready)) { // AP never joined the rendezvous; skip, don't hang + @atomicStore(u32, &warp_bsp_ready, 0, .release); + return; + } + var i: u32 = 0; + while (i < warp_rounds) : (i += 1) warpTick(); + @atomicStore(u32, &warp_stop, 1, .release); + @atomicStore(u32, &warp_bsp_ready, 0, .release); + warp_checks += 1; + + if (!tsc_synced and hpet_clock_hz != 0) demoteToHpet(); +} + +/// AP side: join the BSP's warp check, then return so the core can enter the +/// scheduler. Bounded so a missing BSP can't strand the core. +pub fn checkWarpTarget() void { + if (clock_source != .tsc) return; + if (!warpAwait(&warp_bsp_ready)) return; + @atomicStore(u32, &warp_ap_ready, 1, .release); + var spins: u64 = 0; + while (@atomicLoad(u32, &warp_stop, .acquire) == 0) : (spins += 1) { + if (spins >= warp_spin_limit) return; + warpTick(); + } +} + +/// Switch the clocksource from the TSC to the HPET without a discontinuity: choose +/// the HPET zero point so it reads the same nanosecond value the TSC does right now, +/// so time neither jumps nor runs backward across the switch. Called when the warp +/// check proves the per-core TSCs unsynchronized. +fn demoteToHpet() void { + const now_ns = @as(u128, rdtsc() -% tsc_base) * 1_000_000_000 / tsc_hz; + const equivalent_ticks: u64 = @intCast(now_ns * hpet_clock_hz / 1_000_000_000); + hpet_clock_base = readHpet() -% equivalent_ticks; + clock_source = .hpet; } /// Acknowledge the current interrupt so the LAPIC will deliver the next one. diff --git a/system/kernel/architecture/x86_64/cpu.zig b/system/kernel/architecture/x86_64/cpu.zig index f2eac82..238cccf 100644 --- a/system/kernel/architecture/x86_64/cpu.zig +++ b/system/kernel/architecture/x86_64/cpu.zig @@ -469,6 +469,35 @@ pub fn clockHz() u64 { return apic.tscHz(); } +/// Whether the CPU guarantees an **invariant** TSC (CPUID 0x80000007 EDX[8] on +/// x86; the analogous architectural guarantee elsewhere). When false the TSC is not +/// used as the clocksource. +pub fn clockInvariant() bool { + return apic.tscInvariant(); +} + +/// Whether the per-core clock counters are synchronized (no backward warp observed +/// at SMP bring-up). When false the clock falls back off the TSC. +pub fn clockSynchronized() bool { + return apic.tscSynced(); +} + +/// The active monotonic clocksource, for the boot log ("tsc" or "hpet" on x86). +pub fn clockSourceName() []const u8 { + return apic.clockSourceName(); +} + +/// Test hook: force the TSC clocksource on, to exercise the TSC + warp-check path on +/// a hypervisor that won't advertise an invariant TSC (see apic.forceTscClocksourceForTest). +pub fn forceTscClocksourceForTest() void { + apic.forceTscClocksourceForTest(); +} + +/// How many per-AP TSC warp checks completed (for the tsc-sync test). +pub fn warpChecksRun() u32 { + return apic.warpChecksRun(); +} + /// Unmask maskable interrupts (`sti`) so device interrupts get delivered. pub fn enableInterrupts() void { asm volatile ("sti"); diff --git a/system/kernel/architecture/x86_64/smp.zig b/system/kernel/architecture/x86_64/smp.zig index e907516..1837c30 100644 --- a/system/kernel/architecture/x86_64/smp.zig +++ b/system/kernel/architecture/x86_64/smp.zig @@ -148,7 +148,14 @@ pub fn startAp(apic_id: u32, stack_top: usize, percpu: usize, index: usize, cr3: // Wait up to 100 ms for the AP to reach apEntry and set the flag. const deadline = apic.millis() + 100; while (apic.millis() < deadline) { - if (@atomicLoad(u32, &ap_alive, .acquire) != 0) return true; + if (@atomicLoad(u32, &ap_alive, .acquire) != 0) { + // Cross-check this core's TSC against the BSP's before it joins the run + // loop: an unsynchronized TSC must be caught before any task can migrate + // onto this core and observe time going backward. No-op unless the TSC is + // the clocksource (apic.checkWarpSource). + apic.checkWarpSource(); + return true; + } asm volatile ("pause"); } return false; @@ -177,6 +184,11 @@ fn apEntry(percpu: usize) callconv(.c) noreturn { @atomicStore(u32, &ap_alive, 1, .release); // "architecture state up" — BSP is polling this + // Rendezvous with the BSP for the TSC warp check (no-op unless the TSC is the + // clocksource) before joining the run loop, so this core's clock is vetted before + // it can run any task. + apic.checkWarpTarget(); + if (secondary_entry) |enterScheduler| enterScheduler(); // joins the run loop while (true) asm volatile ("hlt"); // (only if no entry was registered) } diff --git a/system/kernel/devices-broker.zig b/system/kernel/devices-broker.zig index 1bd7c75..9a8da4c 100644 --- a/system/kernel/devices-broker.zig +++ b/system/kernel/devices-broker.zig @@ -189,7 +189,7 @@ pub fn register(parent_id: u64, owner: u32, descriptor: *const device_abi.Device if (!ok) return error.NotContained; } - // Idempotent on exact match (docs/m19-m20-plan.md decision 3): a restarted + // Idempotent on exact match (docs/device-manager.md): a restarted // registering bus re-registers what it rediscovers, and the table has no // unregister — an identical (class, identity, resources) child under the // same parent returns the existing id instead of appending a duplicate. diff --git a/system/kernel/kernel.zig b/system/kernel/kernel.zig index 00e4b8b..391908d 100644 --- a/system/kernel/kernel.zig +++ b/system/kernel/kernel.zig @@ -257,10 +257,30 @@ fn kmain(boot_information: *const BootInformation) noreturn { log.checkpoint(cp_timer); log.print("/system/kernel: timer online ({d} Hz tick; timer clock {d} MHz, clock {d} MHz; calibrated via {s})\n", .{ architecture.timer_hz, architecture.timerClockHz() / 1_000_000, architecture.clockHz() / 1_000_000, architecture.timerCalibrationSource() }); + // The tsc-sync test forces the TSC clocksource on before the cores come up, so the + // TSC + warp-check path is exercised even under TCG (which won't advertise an + // invariant TSC). Inert in a normal build (docs/timers.md). + if (build_options.test_case) |tc| { + if (std.mem.eql(u8, tc, "tsc-sync")) architecture.forceTscClocksourceForTest(); + } + // Wake the other cores (application processors). A no-op on a single-core - // machine; on SMP each AP climbs to long mode and reports in (docs/smp.md). + // machine; on SMP each AP climbs to long mode and reports in (docs/smp.md). The + // per-core TSC warp check rides this: each AP is vetted before it joins the run + // loop (docs/timers.md). bringUpSecondaries(); + // Report the monotonic clock's final reliability, now the warp check has run on + // every core. On real Intel/AMD this is the invariant, synchronized TSC; a bare + // VM (no invariant bit) or a machine whose cores' TSCs skew uses the HPET instead. + log.print("/system/kernel: clocksource {s} (TSC invariant: {s}, synchronized: {s})\n", .{ + architecture.clockSourceName(), + if (architecture.clockInvariant()) "yes" else "no", + if (architecture.clockSynchronized()) "yes" else "no", + }); + if (!architecture.clockSynchronized()) + log.write("/system/kernel: WARNING: per-core TSCs are not synchronized; monotonic clock moved off the TSC\n"); + // In a test build (`zig build -Dtest-case=`), run that case and stop. // Normal builds fall through to the idle halt. if (build_options.test_case) |case| { diff --git a/system/kernel/tests.zig b/system/kernel/tests.zig index 2c5aabe..bfca441 100644 --- a/system/kernel/tests.zig +++ b/system/kernel/tests.zig @@ -102,6 +102,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void { stressTest(); } else if (eql(case, "smp-retry")) { smpRetryTest(); + } else if (eql(case, "tsc-sync")) { + tscSyncTest(); } else if (eql(case, "fault-ud")) { faultInvalidOpcode(); } else if (eql(case, "fault-pf")) { @@ -162,14 +164,12 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void { vfsTest(boot_information); } else if (eql(case, "input")) { inputTest(boot_information); - } else if (eql(case, "hpet")) { - hpetTest(boot_information); } else if (eql(case, "iopass")) { ioPassTest(); } else if (eql(case, "irqfree")) { irqFreeTest(); - } else if (eql(case, "bus")) { - busTest(boot_information); + } else if (eql(case, "containment")) { + containmentTest(); } else if (eql(case, "device-manager")) { deviceManagerTest(boot_information); } else if (eql(case, "poweroff")) { @@ -213,7 +213,7 @@ fn eql(a: []const u8, b: []const u8) bool { /// Whether the captured last-write buffer *contains* `needle`. Markers are /// matched as substrings, not prefixes, so a service's source-path debug prefix -/// (`system/drivers/hpet: ok`) still satisfies a marker like `hpet: ok`. +/// (`system/drivers/pci-bus: ...`) still satisfies a marker like `pci-bus: `. fn bufferHas(needle: []const u8) bool { return std.mem.indexOf(u8, process.write_buffer[0..process.write_len], needle) != null; } @@ -1211,6 +1211,35 @@ fn clockTest() void { result(); } +/// The TSC clocksource + cross-core warp check (`-smp 4`). This is the real +/// Intel/AMD / KVM path — an invariant, synchronized TSC. TCG (the only x86 +/// accelerator on an Apple-Silicon host) won't advertise an invariant TSC, so the +/// boot forces the TSC clocksource on (kernel.zig, gated on this case) to exercise +/// the machinery: the kernel must run the per-AP warp check as each core came up, +/// find the cores' TSCs synchronized, and keep the clock on the TSC (no HPET +/// fallback). The default suite (no force) exercises the HPET fallback instead. +fn tscSyncTest() void { + log("DANOS-TEST-BEGIN: tsc-sync\n", .{}); + check("clocksource is the TSC (forced-invariant path)", eql(architecture.clockSourceName(), "tsc")); + check("the cross-core warp check ran on the APs", architecture.warpChecksRun() >= 1); + check("per-core TSCs synchronized (no warp, no HPET fallback)", architecture.clockSynchronized()); + + // A warp that slipped past the bring-up check would surface as a backward reading. + var last = architecture.nanos(); + var monotonic = true; + var advanced = false; + var i: u32 = 0; + while (i < 1_000_000) : (i += 1) { + const t = architecture.nanos(); + if (t < last) monotonic = false; + if (t > last) advanced = true; + last = t; + } + check("monotonic clock advanced", advanced); + check("monotonic clock never ran backward", monotonic); + result(); +} + var proc_worker_run: bool = true; var proc_worker_ran: bool = false; @@ -2258,68 +2287,7 @@ fn spawnNamed(rd: initial_ramdisk.Reader, name: []const u8) bool { return false; } -/// IO passthrough + IRQ-as-IPC: a user-space driver drives real hardware and is -/// *woken by it*. Spawn hpet, which claims the HPET, maps its registers into its -/// own ring-3 address space, arms a level-triggered comparator, binds the interrupt -/// to an IPC endpoint, and then blocks. It prints "hpet: ok" only after being woken -/// `target_ticks` times — it cannot reach that line by polling, because the loop's -/// only exit is through `replyWait` returning a notification badge. -/// -/// The interesting assertion is the last one, which doesn't trust hpet at all: it -/// reads the I/O APIC's redirection entry back and checks the kernel really routed -/// the line (our vector, level-triggered) and really left it unmasked after the -/// driver's final `irq_ack`. hpet disables its comparator on the last interrupt, so -/// that state is quiescent and not a race. -fn hpetTest(boot_information: *const BootInformation) void { - log("DANOS-TEST-BEGIN: hpet\n", .{}); - if (boot_information.initial_ramdisk_len == 0) { - check("bootloader handed over an initial_ramdisk", false); - result(); - return; - } - const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len]; - const rd = initial_ramdisk.Reader.init(image) orelse { - check("initial_ramdisk image is valid", false); - result(); - return; - }; - - process.write_count = 0; - process.write_from_user = false; - check("hpet spawned from the initial_ramdisk", spawnNamed(rd, "hpet")); - - const prefix = "hpet: ok"; - scheduler.setPriority(1); - const deadline = architecture.millis() + 10000; - while (architecture.millis() < deadline) { - if (bufferHas(prefix) and process.write_count >= 2) break; - scheduler.yield(); - } - scheduler.setPriority(4); - - const ok = bufferHas(prefix); - check("user driver mapped HPET MMIO and was woken by its interrupt", ok); - check("driver syscalls came from user mode (CPL 3)", process.write_from_user); - check("kernel routed and re-armed the HPET's line at the I/O APIC", hpetRouteOk()); - result(); -} - -/// Read back the I/O APIC redirection entry for the HPET's GSI and confirm the -/// kernel programmed it: a vector in the device window, level-triggered, unmasked. -/// Independent of anything the driver reported about itself. -fn hpetRouteOk() bool { - const gsi = hpetGsi() orelse return false; - if (gsi >= architecture.irqRouteCount()) return false; - const low = architecture.irqRouteRaw(gsi); // entry index == GSI (this I/O APIC's gsi_base is 0) - const vector: u8 = @truncate(low & 0xFF); - const masked = low & (1 << 16) != 0; - const level = low & (1 << 15) != 0; - return vector >= architecture.irq_vector_base and - vector < architecture.irq_vector_base + architecture.irq_vector_count and - level and !masked; -} - -/// The GSI discovery recorded for the HPET, from the same device table the driver saw. +/// The GSI discovery recorded for the HPET, from the same device table drivers see. fn hpetGsi() ?u32 { var buffer: [16]device_abi.DeviceDescriptor = undefined; const n = @min(devices_broker.enumerate(&buffer), buffer.len); @@ -2334,87 +2302,87 @@ fn hpetGsi() ?u32 { return null; } -/// Bus driver: a user process claims a device that contains other devices, enumerates -/// them from the hardware, and publishes each as a child via `device_register` — the -/// primitive a PCI bridge or USB hub driver is built from. -/// -/// `bus` treats the HPET's register block as a bus and its comparators as children, -/// giving each a 0x20 sub-window. It checks its own work (children come back from the -/// table with the right parent and a strictly narrower window) and, importantly, that -/// the kernel **refuses** a child whose window escapes the parent's — without that, -/// `device_register` would be a system_call for mapping arbitrary physical memory. It prints -/// "bus: ok" only if all of that holds. -/// -/// The kernel-side check here is the one bus can't make: that the children really did -/// land in the device table with the containment invariant intact. -fn busTest(boot_information: *const BootInformation) void { - log("DANOS-TEST-BEGIN: bus\n", .{}); +/// `device_register` containment — a kernel security property, tested directly against +/// the broker (no user-space demo driver). A bus driver publishes children of a device +/// it owns; the kernel must **refuse** any child whose resource escapes the parent's +/// grant, or `device_register` would become a system call for mapping arbitrary physical +/// memory. This is the property the old `bus` demo driver proved end-to-end; with the +/// demo gone, the property is asserted where it lives — in the kernel. Also checks the +/// idempotence rule (M19.0): re-registering an identical child returns the same id +/// instead of appending a duplicate. +fn containmentTest() void { + log("DANOS-TEST-BEGIN: containment\n", .{}); - // M19.0: device_register is idempotent on exact match — a restarted - // registering bus must not duplicate its children. Driven directly against - // the broker: claim an unclaimed node, register the same (class, hid, - // resourceless) child twice, expect one id and one table entry. - { - const me = scheduler.currentId(); - var probe: [1]device_abi.DeviceDescriptor = undefined; - const total = devices_broker.enumerate(&probe); - check("device tree is seeded for the idempotence check", total >= 1); - if (devices_broker.ownerOf(0) == null) { - check("claimed device 0 for the idempotence check", devices_broker.claim(0, me)); - var child = std.mem.zeroes(device_abi.DeviceDescriptor); - child.class = @intFromEnum(device_abi.DeviceClass.unknown); - child.pci_class = device_abi.no_pci_class; - child.hid_len = 4; - child.hid[0..4].* = "idem".*; - const first = devices_broker.register(0, me, &child) catch 0; - check("first register succeeded", first != 0); - const before = devices_broker.enumerate(&probe); - const second = devices_broker.register(0, me, &child) catch 0; - check("re-register returned the same id", second == first); - check("re-register grew nothing", devices_broker.enumerate(&probe) == before); - devices_broker.releaseAllOwnedBy(me); - } else { - check("device 0 unexpectedly claimed before the idempotence check", false); - } - } + const me = scheduler.currentId(); + var buffer: [64]device_abi.DeviceDescriptor = undefined; + check("device tree is seeded", devices_broker.enumerate(&buffer) >= 1); - if (boot_information.initial_ramdisk_len == 0) { - check("bootloader handed over an initial_ramdisk", false); + // The kernel-seeded HPET timer block is a device with a memory resource — a natural + // parent to publish sub-window children under, as a PCI bridge or USB hub would. + const parent_id = hpetDeviceId() orelse { + check("found a device with a memory window to parent children under", false); result(); return; + }; + const parent = buffer[@intCast(parent_id)]; + var window: ?device_abi.ResourceDescriptor = null; + for (0..parent.resource_count) |j| { + if (parent.resources[j].kind == @intFromEnum(device_abi.ResourceKind.memory)) window = parent.resources[j]; } - const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len]; - const rd = initial_ramdisk.Reader.init(image) orelse { - check("initial_ramdisk image is valid", false); + const parent_window = window orelse { + check("parent exposes a memory window", false); result(); return; }; - process.write_count = 0; - process.write_from_user = false; - check("bus spawned from the initial_ramdisk", spawnNamed(rd, "bus")); - - const prefix = "bus: ok"; - scheduler.setPriority(1); - const deadline = architecture.millis() + 10000; - while (architecture.millis() < deadline) { - if (bufferHas(prefix)) break; - scheduler.yield(); + if (devices_broker.ownerOf(parent_id) != null) { + check("parent device was unclaimed at the start of the test", false); + result(); + return; } - scheduler.setPriority(4); + check("claimed the parent device", devices_broker.claim(parent_id, me)); + defer devices_broker.releaseAllOwnedBy(me); + + // A child whose window lies inside the parent's is accepted. + var fits = childDescriptor("cfit", parent_window.start, 0x20); + const before = devices_broker.enumerate(&buffer); + const good = devices_broker.register(parent_id, me, &fits) catch 0; + check("a contained child is registered", good != 0); + check("the contained child was appended to the table", devices_broker.enumerate(&buffer) == before + 1); + + // A child whose window escapes the parent's is refused with NotContained. + var escapes = childDescriptor("cesc", parent_window.start, parent_window.len + 0x1000); + const refused = if (devices_broker.register(parent_id, me, &escapes)) |_| false else |err| err == error.NotContained; + check("an out-of-window child is refused (NotContained)", refused); + check("the refused child left the table unchanged", devices_broker.enumerate(&buffer) == before + 1); + + // Idempotent on exact match: re-registering the accepted child returns its id and + // appends nothing (M19.0 — a restarted bus re-reports what it rediscovers). + const again = devices_broker.register(parent_id, me, &fits) catch 0; + check("re-registering an identical child returns the same id", again != 0 and again == good); + check("re-registering grew nothing", devices_broker.enumerate(&buffer) == before + 1); - const ok = bufferHas(prefix); - check("bus driver published children and the kernel refused an out-of-window one", ok); - check("driver syscalls came from user mode (CPL 3)", process.write_from_user); - check("every registered child is contained in its parent", childrenContained()); result(); } +/// A minimal child descriptor with one memory resource, for the containment test. +fn childDescriptor(hid: []const u8, start: u64, len: u64) device_abi.DeviceDescriptor { + var child = std.mem.zeroes(device_abi.DeviceDescriptor); + child.class = @intFromEnum(device_abi.DeviceClass.unknown); + child.pci_class = device_abi.no_pci_class; + child.hid_len = @intCast(hid.len); + @memcpy(child.hid[0..hid.len], hid); + child.resource_count = 1; + child.resources[0] = .{ .kind = @intFromEnum(device_abi.ResourceKind.memory), .start = start, .len = len }; + return child; +} + /// The device manager (a ring-3 service) enumerates /system/devices, matches each -/// device to a driver, and — eventually — spawns it. This increment only checks the -/// discovery+matching half: it must find the HPET (a timer) and decide `hpet` serves -/// it, printing "device-manager: ok". It uses no special privilege — the same -/// `device_enumerate` any process could call. (Spawning is the next increment.) +/// device to a driver, and spawns it. Proof of the whole discover -> match -> spawn -> +/// driver-up chain: boot only the device-manager; it must discover the PCI host bridge, +/// match `pci-bus`, and spawn it (with the bridge id as its argument) — and the spawned +/// pci-bus must reach its own live marker. It uses no special privilege — the same +/// `device_enumerate` any process could call. fn deviceManagerTest(boot_information: *const BootInformation) void { log("DANOS-TEST-BEGIN: device-manager\n", .{}); if (boot_information.initial_ramdisk_len == 0) { @@ -2430,70 +2398,65 @@ fn deviceManagerTest(boot_information: *const BootInformation) void { }; // Let `system_spawn` find bundled binaries by name (the normal boot path does - // this too). Only the device-manager is spawned here — so if `hpet` runs at all, - // it's because the manager discovered the timer, matched, and spawned it. + // this too). Only the device-manager is spawned here — so if `pci-bus` runs at + // all, it's because the manager discovered the PCI host bridge, matched, and + // spawned it. process.setInitialRamdisk(image); process.write_count = 0; process.write_from_user = false; check("device-manager spawned from the initial_ramdisk", spawnNamed(rd, "device-manager")); - // End-to-end proof: the driver the manager spawned reaches its own live marker. - // `hpet: ok` is hpet's final, stable message (it claims the timer, maps its MMIO, - // binds its IRQ, services one, then sleeps) — nothing overwrites the buffer after, - // so it's race-free to poll for. Its arrival means the whole - // discover -> match -> system_spawn -> driver-up chain worked. - const prefix = "hpet: ok"; + // End-to-end proof, read from kernel state — not the racy last-write serial buffer, + // since many services keep logging after pci-bus. The manager must discover the PCI + // host bridge, match pci-bus, and spawn it, and pci-bus must come up: claim the + // bridge, map its ECAM, and register the functions it enumerates as children in the + // device tree. scheduler.setPriority(1); const deadline = architecture.millis() + 10000; + var spawned = false; while (architecture.millis() < deadline) { - if (bufferHas(prefix)) break; + if (processRunning("pci-bus")) spawned = true; + if (spawned and pciFunctionsRegistered()) break; scheduler.yield(); } scheduler.setPriority(4); - const ok = bufferHas(prefix); - check("device manager matched the timer and system_spawn'd hpet, which came up", ok); + check("device manager discovered the PCI host bridge and spawned pci-bus", spawned); + check("pci-bus came up and registered the functions it enumerated", pciFunctionsRegistered()); check("its syscalls came from user mode (CPL 3)", process.write_from_user); result(); } -/// Every child `bus` registered must have each of its resources inside a parent -/// resource of the same kind — the invariant `device_register` exists to maintain, -/// checked from the kernel's own table rather than the driver's word for it. -/// -/// Only *registered* children are checked, not the whole tree. Firmware topology is -/// trusted and doesn't obey containment: a PCI function's BAR is not inside its host -/// bridge's `bus_range`, because a bus-number range isn't an address window. -fn childrenContained() bool { - var buffer: [64]device_abi.DeviceDescriptor = undefined; - const n = @min(devices_broker.enumerate(&buffer), buffer.len); - - const bus_id = hpetDeviceId() orelse return false; - const p = buffer[@intCast(bus_id)]; - - var children: usize = 0; - for (buffer[0..n]) |d| { - if (d.parent != bus_id) continue; - children += 1; - for (0..d.resource_count) |i| { - const r = d.resources[i]; - var ok = false; - for (0..p.resource_count) |j| { - const pr = p.resources[j]; - if (pr.kind != r.kind) continue; - if (r.kind == @intFromEnum(device_abi.ResourceKind.irq)) { - if (pr.start == r.start) ok = true; - } else if (r.len != 0 and r.start >= pr.start and - r.start + r.len <= pr.start + pr.len) ok = true; - } - if (!ok) return false; - } +/// Whether a live task was spawned under `name` (its argv[0]) — read from the kernel +/// task table, the same snapshot `process_enumerate` exposes. +fn processRunning(name: []const u8) bool { + var table: [64]abi.ProcessDescriptor = undefined; + const total = scheduler.enumerate(&table); + for (table[0..@min(total, table.len)]) |d| { + if (std.mem.eql(u8, d.name[0..d.name_length], name)) return true; } - return children > 0; // bus must have published at least one + return false; } -/// Device id of the HPET (the bus bus claims), from the same table drivers see. +/// Whether pci-bus registered at least one function under the PCI host bridge — proof +/// it came up, claimed the bridge, mapped its ECAM, and walked configuration space. +fn pciFunctionsRegistered() bool { + var buffer: [64]device_abi.DeviceDescriptor = undefined; + const n = @min(devices_broker.enumerate(&buffer), buffer.len); + var bridge_id: ?u64 = null; + for (buffer[0..n]) |d| { + if (d.class == @intFromEnum(device_abi.DeviceClass.pci_host_bridge)) bridge_id = d.id; + } + const bid = bridge_id orelse return false; + for (buffer[0..n]) |d| { + if (d.parent == bid) return true; + } + return false; +} + +/// Device id of the kernel-seeded HPET timer block (the node with a memory resource), +/// from the same device table drivers see. Used as a containment-test parent. fn hpetDeviceId() ?u64 { var buffer: [64]device_abi.DeviceDescriptor = undefined; const n = @min(devices_broker.enumerate(&buffer), buffer.len); @@ -2511,8 +2474,9 @@ fn hpetDeviceId() ?u64 { /// (so a dead driver's device goes quiet instead of storming) and the slot cleared /// (so an ISR never posts a notification into the endpoint that is about to be freed). /// -/// This is the path `hpet` never takes — it runs forever — so it gets its own test. -/// Two properties, both read back from the hardware rather than from our own state: +/// A long-running driver that never exits wouldn't reach this teardown path, so it +/// gets its own test that binds and releases directly. Two properties, both read back +/// from the hardware rather than from our own state: /// /// 1. A bound GSI is routed and unmasked. /// 2. After `releaseOwner` for the binding's owner, that same entry is masked again. diff --git a/system/services/acpi/acpi.zig b/system/services/acpi/acpi.zig index 59481e0..2a308f2 100644 --- a/system/services/acpi/acpi.zig +++ b/system/services/acpi/acpi.zig @@ -1,5 +1,5 @@ //! /system/services/acpi — the ACPI discovery service: the x86 firmware -//! interpreter, moved out of ring 0 (docs/m19-m20-plan.md, M20). Claims the +//! interpreter, moved out of ring 0 (docs/discovery.md). Claims the //! `acpi-tables` node the kernel publishes (the AML blobs, the broad io_port //! grant, a broad irq window, the SCI), and runs the **shared AML module** in //! ring 3 — the same parser and interpreter the kernel uses. @@ -321,7 +321,7 @@ fn onSci() void { /// handler method (`_Lxx` level / `_Exx` edge), drain the Notify queue the /// method produced, and publish an event per notified device. Then clear the /// status bit. QEMU raises no GPEs on this config, so this path is exercised by -/// host unit tests (docs/m21-plan.md decision 5); on real hardware it carries +/// host unit tests (docs/acpi.md — ACPI events); on real hardware it carries /// battery/AC/lid. The embedded controller's `_Qxx` queries are out of scope. fn handleGpe() void { handleGpeBlock(gpe0_blk, gpe0_len, 0); @@ -476,7 +476,7 @@ fn walkDevices(node: *aml.Node, interpreter: *aml.Interpreter) void { if (readHid(c, interpreter)) |hid| { // Skip PCI roots — pci-bus already reports PCI functions; ACPI adds - // only the non-PCI _HID devices (docs/m19-m20-plan.md M20.2). The two + // only the non-PCI _HID devices (docs/device-manager.md — matching). The two // roots are named through the shared registry, not bare _HID strings. const id = acpi_ids.HardwareId.fromHid(hid[0..7]); if (id != .pci_bus and id != .pci_express_root_bridge) { diff --git a/system/services/device-manager/device-manager.zig b/system/services/device-manager/device-manager.zig index cc98aea..dec6380 100644 --- a/system/services/device-manager/device-manager.zig +++ b/system/services/device-manager/device-manager.zig @@ -31,17 +31,6 @@ fn writeLine(comptime fmt: []const u8, arguments: anytype) void { _ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return); } -/// The driver that serves each device — the policy table. In a fuller system -/// this comes from a manifest (docs/device-manager.md: the third bus type -/// triggers it); for now a static map. `null` = no driver for this class yet. -fn driverFor(d: device.DeviceDescriptor) ?[]const u8 { - // The HPET timer node is still kernel-seeded (from the HPET table, not AML). - // PS/2 and other _HID devices now arrive as acpi-service reports and match - // in onChildAdded (M20.3), not from this boot snapshot. - if (d.class == @intFromEnum(device.DeviceClass.timer)) return "hpet"; - return null; -} - /// The PCI class/subclass/prog-IF triple of an xHCI (USB 3) host controller — /// Serial Bus Controller / USB Controller / XHCI — named from pci-class.zig rather /// than written as the bare 0x0C0330 (docs/coding-standards.md, "Named values"). @@ -107,7 +96,7 @@ const Driver = struct { // The assigned device id (becomes argv[1]), or protocol.no_device. device_id: u64 = protocol.no_device, // Whether this driver speaks the protocol (hello expected, deadline - // enforced). Legacy drivers (hpet, ps2-bus) are supervised and restarted + // enforced). Legacy drivers (e.g. ps2-bus) are supervised and restarted // but not yet required to hello. speaks_protocol: bool = false, process_id: u32 = 0, @@ -346,19 +335,15 @@ fn initialise(endpoint: runtime.ipc.Handle) bool { addDriver("pci-bus", descriptor.id, true); continue; } - // PCI functions no longer appear in the boot snapshot (M19.3): the - // pci-bus driver reports them, and onChildAdded matches from reports. - const driver_name = driverFor(descriptor) orelse continue; - matched += 1; - // Skip a singleton that is already alive (the initial-ramdisk sweep test - // starts every bundled binary bare, this manager included) — spawning a - // second instance would only lose the claim race and churn the log. - if (!alreadySupervised(driver_name) and !system.isProcessRunning(driver_name)) { - addDriver(driver_name, protocol.no_device, false); - } + // Nothing else is matched from the boot snapshot today. The kernel-seeded + // HPET timer node is served by the kernel's own clock (docs/timers.md), not + // a user-space driver; PCI functions and PS/2 _HID devices arrive later as + // pci-bus / acpi-service reports and match in onChildAdded (docs/discovery.md). + // A fuller system's static class->driver manifest (docs/device-manager.md) + // would slot in here. } - // The discovery service (docs/m19-m20-plan.md M20): one per firmware, packed + // The discovery service (docs/discovery.md): one per firmware, packed // under the neutral name "discovery", spawned once at startup. It finds and // claims the acpi-tables (or devicetree-blob) node itself. Not a per-device // match — it is the discoverer, not a driver bound to one device. diff --git a/system/services/fdt/fdt.zig b/system/services/fdt/fdt.zig index cba17e9..526c54f 100644 --- a/system/services/fdt/fdt.zig +++ b/system/services/fdt/fdt.zig @@ -1,5 +1,5 @@ //! /system/services/fdt — the devicetree discovery service: the ARM twin of the -//! acpi service (docs/m19-m20-plan.md decision 7). **Placeholder: not +//! acpi service (docs/discovery.md — firmware neutrality). **Placeholder: not //! implemented.** It exists so the build's `-Ddiscovery` option has both of its //! values from day one; the implementation lands with the Raspberry Pi //! bring-up (docs/arm.md). @@ -15,7 +15,7 @@ //! resident under the manager's supervision (hello, restart, the usual //! contract). //! -//! Known prerequisite recorded in the plan: `DeviceDescriptor`'s 8-byte `hid` +//! Known prerequisite recorded in docs/discovery.md: `DeviceDescriptor`'s 8-byte `hid` //! cannot hold an FDT `compatible` string ("brcm,bcm2835-aux-uart") — identity //! widens before this file grows a body. diff --git a/system/services/power/protocol.zig b/system/services/power/protocol.zig index 8d1a720..183f3fe 100644 --- a/system/services/power/protocol.zig +++ b/system/services/power/protocol.zig @@ -1,8 +1,8 @@ -//! The power protocol (docs/m21-plan.md): system power's domain-named surface, +//! The power protocol (docs/power.md): system power's domain-named surface, //! registered under `ServiceId.power`. On x86 the acpi service serves it; on //! ARM a PSCI/mailbox service will register the same id — subscribers never -//! learn which firmware they are on (m19-m20-plan.md decision 7). The -//! vfs-protocol pattern: extern-struct messages, a version, reserved fields. +//! learn which firmware they are on (docs/discovery.md — firmware neutrality). +//! The vfs-protocol pattern: extern-struct messages, a version, reserved fields. /// The protocol version a client states nowhere yet — reserved for the day a /// handshake needs it; requests carry it so a mismatch can be refused loudly. diff --git a/test/qemu_test.py b/test/qemu_test.py index c77cb22..df03195 100644 --- a/test/qemu_test.py +++ b/test/qemu_test.py @@ -177,6 +177,15 @@ CASES = [ "smp": 4, "expect": r"DANOS-TEST-RESULT: PASS", "fail": r"DANOS-TEST-RESULT: FAIL"}, + # TSC clocksource + cross-core warp check (real Intel/AMD / KVM path). TCG won't + # advertise an invariant TSC, so the kernel forces the TSC clocksource on for this + # case (gated in kernel.zig) and runs the per-AP warp check across the 4 cores; + # their TSCs are synchronized, so it stays on the TSC (no HPET fallback). The rest + # of the suite exercises the HPET fallback instead. See docs/timers.md. + {"name": "tsc-sync", + "smp": 4, + "expect": r"DANOS-TEST-RESULT: PASS", + "fail": r"DANOS-TEST-RESULT: FAIL"}, {"name": "fault-ud", "expect": r"invalid opcode \(vector 6\)"}, {"name": "fault-pf", "expect": r"page fault \(vector 14\)"}, {"name": "fault-df", "expect": r"double fault \(vector 8\)"}, @@ -284,7 +293,7 @@ CASES = [ "fail": r"acpi-parse: mismatch|DANOS-TEST-RESULT: FAIL"}, # M20.3: the flip — ps2-bus now comes up from the acpi service's report, not # a kernel-built node. Ordered: report -> spawn -> the driver attaches its - # keyboard, proving discovery runs entirely in ring 3 (docs/m19-m20-plan.md). + # keyboard, proving discovery runs entirely in ring 3 (docs/discovery.md). {"name": "acpi-ps2", "smp": 4, "timeout": 150, @@ -294,7 +303,7 @@ CASES = [ "fail": r"DANOS-TEST-RESULT: FAIL"}, # M21.1: the SCI + power button. Boot the manager (which spawns the acpi # service); ~4s in, QMP system_powerdown raises the ACPI power-button fixed - # event; the service's SCI handler must log the press (docs/m21-plan.md). + # event; the service's SCI handler must log the press (docs/acpi.md). {"name": "power-button", "smp": 4, "timeout": 60, @@ -305,7 +314,7 @@ CASES = [ # ~5s in, QMP system_powerdown raises the power button; the acpi service # publishes it, init stops its children then requests S5, and QEMU exits. # The ordered regex proves button -> shutting-down -> entering-S5; the case - # passes on QEMU's self-exit through S5 (docs/m21-plan.md). + # passes on QEMU's self-exit through S5 (docs/power.md). {"name": "orderly-shutdown", "smp": 4, "timeout": 90, @@ -316,7 +325,7 @@ CASES = [ "fail": r"power: S5 write did not take|DANOS-TEST-RESULT: FAIL"}, # M20.2: the acpi service evaluates _CRS/_STA in ring 3 and registers + # reports its _HID devices — the two PS/2 nodes must appear with resources - # (keyboard: io 0x60/0x64 + IRQ = 3; mouse: IRQ = 1) (docs/m19-m20-plan.md). + # (keyboard: io 0x60/0x64 + IRQ = 3; mouse: IRQ = 1) (docs/discovery.md). {"name": "acpi-report", "smp": 4, "timeout": 150, @@ -378,27 +387,21 @@ CASES = [ {"name": "input", "expect": r"DANOS-TEST-RESULT: PASS", "fail": r"DANOS-TEST-RESULT: FAIL"}, - # IO passthrough + IRQ-as-IPC: a user-space HPET driver maps device MMIO into - # its own address space, binds the device's interrupt to an IPC endpoint, and - # is woken by the hardware five times while blocked (never polling). - {"name": "hpet", - "expect": r"DANOS-TEST-RESULT: PASS", - "fail": r"DANOS-TEST-RESULT: FAIL"}, - # Device manager: a ring-3 service enumerates /system/devices and matches each - # device to a driver (discovery + policy in user space). This increment logs the - # decision; spawning follows. + # Device manager: a ring-3 service enumerates /system/devices, matches the PCI host + # bridge to pci-bus, and spawns it — end-to-end proof of discover -> match -> spawn + # -> driver-up (the spawned pci-bus logs " functions found"). {"name": "device-manager", "expect": r"DANOS-TEST-RESULT: PASS", "fail": r"DANOS-TEST-RESULT: FAIL"}, - # Bus driver: a user process claims a device, enumerates its children from the - # hardware, and publishes each with dev_register — and the kernel refuses a child - # whose window escapes the parent's (else dev_register maps arbitrary memory). - {"name": "bus", + # device_register containment (in-kernel): registering a child whose MMIO window + # escapes its parent's grant is refused (NotContained) — else dev_register would map + # arbitrary physical memory — while an identical re-register stays idempotent. + {"name": "containment", "expect": r"DANOS-TEST-RESULT: PASS", "fail": r"DANOS-TEST-RESULT: FAIL"}, # IRQ teardown: an exiting driver's line is masked and its slot cleared (so no # ISR notifies a freed endpoint), and a sibling owner sharing that endpoint - # keeps its own binding. The path hpet never takes, since it runs forever. + # keeps its own binding. A long-running driver never reaches this teardown path. {"name": "irqfree", "expect": r"DANOS-TEST-RESULT: PASS", "fail": r"DANOS-TEST-RESULT: FAIL"}, @@ -450,7 +453,7 @@ def qmp_send(path, command): """One QMP command: connect, capabilities handshake, execute. Raises on any failure — the caller retries until the guest's socket is ready. This is how a case injects a host-side event (system_powerdown = the ACPI power button) - into the running guest (docs/m21-plan.md).""" + into the running guest (docs/power.md).""" sock = socket.socket(socket.AF_UNIX, socket.SOCK_STREAM) sock.settimeout(5) try: