From 452080e997532705ed5464969fcae7036ace832b Mon Sep 17 00:00:00 2001 From: Daniel Samson <12231216+daniel-samson@users.noreply.github.com> Date: Mon, 13 Jul 2026 11:56:43 +0100 Subject: [PATCH] Add runtime.time, drop demo drivers, harden TSC timekeeping Time is a kernel concern in danos: the kernel owns the scheduling timer and already exposes monotonic time via the clock/sleep/timer_bind syscalls, so a userspace time service would be a redundant, slower path. This adds the generic runtime.time module over those syscalls, retires the two demonstration drivers, reorganizes the milestone docs, and makes the monotonic clock correct on Intel, AMD, and inside any VM. runtime.time (library/runtime/time.zig) - Instant/Duration interface: now, sleep, spin, after, monotonicNanos, available - a thin layer over system.clock/sleep/timerOnce; unit-tested arithmetic Remove the demo drivers hpet and bus (a teaching example belongs in the docs, not shipped in the tree) - system/drivers/ now holds only real drivers: pci-bus, ps2-bus, usb-xhci-bus - device-manager end-to-end test repointed to pci-bus (asserts on kernel state: the process table and the device tree, not a racy serial marker) - device_register containment moved to a new in-kernel `containment` test - the driver-model worked example moved inline into docs/drivers.md Reorganize milestone docs into topic docs - m17-m18 / m19-m20 / m21 plans dissolved into process-lifecycle, device-manager, discovery, and acpi docs; new docs/power.md and docs/timers.md; ~20 citations repointed; plan docs deleted TSC reliability (apic.zig, smp.zig, cpu.zig, kernel.zig) - check the invariant-TSC bit (CPUID 0x80000007 EDX[8]) on Intel and AMD - cross-core "warp" check at SMP bring-up, pairwise BSP<->AP as each core comes up - fall back to the HPET clocksource when the TSC is not invariant (a bare VM) or not synchronized (a warp), switched continuously so time never jumps - boot log reports the outcome; new tsc-sync test exercises the TSC + warp path Verified: zig build; zig build test; 60/60 QEMU cases (incl. new containment and tsc-sync). --- build.zig | 31 +- docs/README.md | 11 +- docs/acpi.md | 56 ++- docs/coding-standards.md | 4 +- docs/danos-file-system-hierarchy-FSH.md | 6 +- docs/device-interrupts.md | 34 ++ docs/device-manager.md | 26 +- docs/discovery.md | 54 +++ docs/driver-model.md | 19 +- docs/drivers.md | 69 ++-- docs/m17-m18-plan.md | 210 ------------ docs/m19-m20-plan.md | 196 ----------- docs/m21-plan.md | 138 -------- docs/power.md | 128 +++++++ docs/resilience.md | 2 +- docs/timers.md | 117 +++++++ library/runtime/runtime.zig | 5 +- library/runtime/time.zig | 169 +++++++++ system/abi.zig | 2 +- system/devices/acpi.zig | 8 +- system/devices/aml/aml.zig | 2 +- system/devices/device-abi.zig | 2 +- system/drivers/bus/bus.zig | 211 ------------ system/drivers/hpet/hpet.zig | 195 ----------- system/drivers/pci-bus/pci-bus.zig | 2 +- system/kernel/architecture/x86_64/apic.zig | 243 ++++++++++++- system/kernel/architecture/x86_64/cpu.zig | 29 ++ system/kernel/architecture/x86_64/smp.zig | 14 +- system/kernel/devices-broker.zig | 2 +- system/kernel/kernel.zig | 22 +- system/kernel/tests.zig | 322 ++++++++---------- system/services/acpi/acpi.zig | 6 +- .../device-manager/device-manager.zig | 31 +- system/services/fdt/fdt.zig | 4 +- system/services/power/protocol.zig | 6 +- test/qemu_test.py | 41 +-- 36 files changed, 1150 insertions(+), 1267 deletions(-) delete mode 100644 docs/m17-m18-plan.md delete mode 100644 docs/m19-m20-plan.md delete mode 100644 docs/m21-plan.md create mode 100644 docs/power.md create mode 100644 docs/timers.md create mode 100644 library/runtime/time.zig delete mode 100644 system/drivers/bus/bus.zig delete mode 100644 system/drivers/hpet/hpet.zig diff --git a/build.zig b/build.zig index 88406b9..63fec3f 100644 --- a/build.zig +++ b/build.zig @@ -132,8 +132,8 @@ pub fn build(b: *std.Build) void { // ACPI/PnP hardware-ID (_HID) names — the flat analog of pci-class for acpi_device // nodes. Also shared reference data. // The AML interpreter, a build module so the ring-3 acpi service can run the - // same parser the kernel does (docs/m19-m20-plan.md decision 1). Pure Zig, - // no kernel imports — one source, two builds. + // same parser the kernel does (docs/discovery.md — the shared AML module). + // Pure Zig, no kernel imports — one source, two builds. const aml_module = b.addModule("aml", .{ .root_source_file = b.path("system/devices/aml/aml.zig"), }); @@ -226,7 +226,7 @@ pub fn build(b: *std.Build) void { }); runtime_module.addImport("device-manager-protocol", device_manager_protocol_module); - // The power protocol: system power's domain-named surface (docs/m21-plan.md). + // The power protocol: system power's domain-named surface (docs/power.md). const power_protocol_module = b.addModule("power-protocol", .{ .root_source_file = b.path("system/services/power/protocol.zig"), }); @@ -346,8 +346,6 @@ pub fn build(b: *std.Build) void { // which unpacks it and spawns each program (system/initial-ramdisk.zig). const vfs_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "vfs", "system/services/vfs/vfs.zig"); const vfstest_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "vfs-test", "system/services/vfs/vfs-test.zig"); - const hpet_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "hpet", "system/drivers/hpet/hpet.zig"); - const bus_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "bus", "system/drivers/bus/bus.zig"); const ps2_bus_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-bus", "system/drivers/ps2-bus/ps2-bus.zig"); const ps2_keyboard_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-keyboard", "system/drivers/ps2-bus/keyboard.zig"); const ps2_mouse_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-mouse", "system/drivers/ps2-bus/mouse.zig"); @@ -361,7 +359,7 @@ pub fn build(b: *std.Build) void { const crash_test_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "crash-test", "system/services/crash-test/crash-test.zig"); const device_list_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "device-list", "system/services/device-list/device-list.zig"); // The discovery service: one swappable process per firmware - // (docs/m19-m20-plan.md decision 7), bundled under the neutral ramdisk name + // (docs/discovery.md), bundled under the neutral ramdisk name // "discovery" so the device manager never learns which firmware it is on. // x86 boots describe hardware with ACPI; the Raspberry Pis hand over a // flattened device tree — the aarch64 target flips the default when it @@ -396,10 +394,6 @@ pub fn build(b: *std.Build) void { mk_run.addFileArg(vfs_exe.getEmittedBin()); mk_run.addArg("vfs-test"); mk_run.addFileArg(vfstest_exe.getEmittedBin()); - mk_run.addArg("hpet"); - mk_run.addFileArg(hpet_exe.getEmittedBin()); - mk_run.addArg("bus"); - mk_run.addFileArg(bus_exe.getEmittedBin()); mk_run.addArg("ps2-bus"); mk_run.addFileArg(ps2_bus_exe.getEmittedBin()); mk_run.addArg("ps2-keyboard"); @@ -435,8 +429,6 @@ pub fn build(b: *std.Build) void { .{ vfs_exe, "system/services" }, .{ device_manager_exe, "system/services" }, .{ input_exe, "system/services" }, - .{ hpet_exe, "system/drivers" }, - .{ bus_exe, "system/drivers" }, .{ ps2_bus_exe, "system/drivers" }, .{ ps2_keyboard_exe, "system/drivers" }, .{ ps2_mouse_exe, "system/drivers" }, @@ -615,6 +607,21 @@ pub fn build(b: *std.Build) void { }); test_step.dependOn(&b.addRunArtifact(xkb_tests).step); + // runtime.time's Instant/Duration arithmetic. time.zig pulls in system.zig (the + // syscall wrappers), which needs the `abi` module, so it doesn't fit the plain + // loop above. + const time_tests = b.addTest(.{ + .root_module = b.createModule(.{ + .root_source_file = b.path("library/runtime/time.zig"), + .target = target, + .optimize = optimize, + .imports = &.{ + .{ .name = "abi", .module = abi_module }, + }, + }), + }); + test_step.dependOn(&b.addRunArtifact(time_tests).step); + // Convenience: `zig build gen-xkeyboard-config` regenerates the layout tables from the // vendored data (offline). `fetch` (the network step) stays a manual script run. const gen_xkb = b.addSystemCommand(&.{ "python3", "tools/make-xkeyboard-config.py", "generate" }); diff --git a/docs/README.md b/docs/README.md index c3e5631..843ec29 100644 --- a/docs/README.md +++ b/docs/README.md @@ -100,7 +100,16 @@ Cutting across all of these: when to build it, and how to keep it architecture-agnostic. - **[acpi.md](acpi.md) — finding the ACPI tables.** The concrete x86 locator chain: how the loader captures the **RSDP**, hands its physical address across in `BootInfo`, - and how the platform derives the **RSDT/XSDT** from it and walks the SDTs. + and how the platform derives the **RSDT/XSDT** from it and walks the SDTs — plus the + live event side (the SCI, the power button, GPE/Notify) the ring-3 acpi service runs. +- **[power.md](power.md) — the power service.** System power as a domain-named + service: button/lid/battery events published to subscribers, and init's orderly + shutdown composing the [lifecycle](process-lifecycle.md) stop sequence with an ACPI + S5 write. Firmware-neutral — a PSCI backend drops in on ARM. +- **[timers.md](timers.md) — timers and time.** The ring-3 surface for reading the + clock and waiting: why `now()` is a syscall rather than a service, and the one-shot + timer notification (`timer_bind`) that gives supervisors a timed wait — built on the + LAPIC heartbeat and calibrated TSC of [device-interrupts.md](device-interrupts.md). - **[smp.md](smp.md) — multiple cores.** A design/research note on how microkernels (L4, seL4) handle SMP — big kernel lock vs per-CPU vs multikernel — and how the right choice depends on whether danos is chasing real-time or resilience. diff --git a/docs/acpi.md b/docs/acpi.md index 94ed6a3..76c5adf 100644 --- a/docs/acpi.md +++ b/docs/acpi.md @@ -107,12 +107,66 @@ firmware-agnostic [device model](discovery.md) gets populated; this note stops a part that answers "where are the tables?" — everything past the RSDP is just following more pointers the tables themselves provide. +## ACPI events: the SCI, the power button, and GPEs (M21) + +The tables above are static description; ACPI is also a *live* channel. Hardware +raises the **SCI** (System Control Interrupt) — one shared, level-triggered line +whose vector the FADT names — and the OS reads status registers to learn what +happened: a fixed event like the power button, or a **General-Purpose Event** +(GPE) whose handler is an AML method. Since [discovery](discovery.md) moved AML +to ring 3, the event side lives there too, in the same **acpi service** — the +device discoverer and the event source are one process, because both need the +namespace and the port grant. + +**The kernel hands the service what it needs and no more.** Reading PM1 event +blocks and GPE blocks requires the FADT, which the kernel already parses for its +own `\_S5` poweroff. Rather than re-parse, the kernel appends the **FADT as one +more memory resource** on the `acpi-tables` node; the service tells it apart +from the AML blob resources by signature — the FADT keeps its intact `"FACP"` +header, while the blob resources are header-stripped bytecode that starts with +no signature. The kernel's own FADT parse is untouched; the service reads the +PM1 *event* blocks (which the kernel never parsed — it only needs PM1 *control* +for `\_S5`) and the GPE0/GPE1 blocks straight from its copy. The **SCI itself** +arrives as the node's one `len == 1` irq resource (distinct from the broad +`[0, 256)` window that covers children's legacy lines), which is how the service +finds the line to `irq_bind`. + +With those in hand the service enables ACPI mode (only if `SCI_EN` is clear — +some firmwares boot with it already set), sets `PWRBTN_EN`, and on each SCI: + +- **The power button** is a *fixed* event: a set `PWRBTN_STS` bit in PM1 status. + The handler clears it (write-1-to-clear), logs the press, and publishes a + [`power`](power.md) `power_button` event to subscribers. +- **GPEs** are the general path: for each set-and-enabled GPE bit `n`, the + service evaluates its `\_GPE._L%02X` (level) or `_E%02X` (edge) handler + method, drains the **Notify** queue that method produced, maps each notified + device to an event (battery, AC, lid, or a generic `notify` with its code), + and clears the status bit. A missing handler method is clear-and-log, not an + error. Making GPEs work required teaching the interpreter one opcode it never + handled — `Notify` (`0x86`) — which it now folds into a bounded queue drained + per evaluation; everything else a handler needs (field access, control flow, + method calls) was already proven by the ring-3 `_STA`/`_CRS` work. + +**How this is tested.** QEMU cannot raise GPEs deterministically on this config, +so GPE/Notify correctness is proven by **host unit tests** — hand-encoded AML +with a `Notify` inside a method body, run under `zig build test`. The QEMU +`power-button` scenario proves the fixed-event path end to end: a QMP +`system_powerdown` injects a real ACPI power-button press, and the service's SCI +handler must log it. Battery/AC/lid and the embedded controller's `_Qxx` queries +are interface-complete but validated on real hardware later. + +The service surface these events are *published on* — subscription, the event +vocabulary, and orderly shutdown — is the power service, [power.md](power.md). + ## Related - [efi.md](efi.md) — the loader that captures the RSDP before `ExitBootServices`. - [memory-map.md](memory-map.md) — the same loader-captures / kernel-consumes seam, and the ACPI-reclaim memory the RSDP lives in. - [discovery.md](discovery.md) — the broader (still-evolving) plan for turning these - tables into one neutral device model shared with the ARM device-tree path. + tables into one neutral device model shared with the ARM device-tree path, and how + ACPI enumeration and events moved to the ring-3 acpi service. +- [power.md](power.md) — the domain-named power service the ACPI event side publishes + to (button, lid, battery) and its orderly-shutdown path into S5. - [arch.md](arch.md) — why the kernel reaches the device code through a `platform` module and never names ACPI directly. diff --git a/docs/coding-standards.md b/docs/coding-standards.md index 842c08b..8ed8dda 100644 --- a/docs/coding-standards.md +++ b/docs/coding-standards.md @@ -96,7 +96,7 @@ That's all — no Unix-abbreviation exception. The source directories are full w (`system`, `library`, not `src`/`lib`), and there is no daemon `d` suffix: a driver lives in `system/drivers/` and a service in `system/services/`, so the *location* already says what it is. Encoding the role in the name too (`busd`, `vfsd`) is -redundant — the program is just `bus`, `vfs`. Don't put in a name what its directory +redundant — the program is just `ps2-bus`, `vfs`. Don't put in a name what its directory already tells you. ## A note on collisions @@ -138,7 +138,7 @@ single word or acronym needs no hyphen: `scheduler.zig`, `paging.zig`, `apic.zig conventions above — `snake_case` — because it's an identifier, not a filename.) **A sub-project's entry point repeats its directory's name** — `init/init.zig`, -`runtime/runtime.zig`, `hpet/hpet.zig` — and the sub-project is addressed by the +`runtime/runtime.zig`, `ps2-bus/ps2-bus.zig` — and the sub-project is addressed by the *directory* (`system/services/init`, `library/runtime`), with the repeated leaf resolving away. See the repository-layout section of [README.md](README.md). diff --git a/docs/danos-file-system-hierarchy-FSH.md b/docs/danos-file-system-hierarchy-FSH.md index 798e371..c79bba9 100644 --- a/docs/danos-file-system-hierarchy-FSH.md +++ b/docs/danos-file-system-hierarchy-FSH.md @@ -17,7 +17,7 @@ Most modern Unix and Unix-like operating systems follow the FHS. DanOS has its o | /srv | Site-specific data served by this system, such as data and scripts for web servers, data offered by FTP servers, and repositories for version control systems | | /system | DanOS operating system files (similar idea to C:\Windows). A true representation of danos — its layout mirrors the source tree, so `/system` is what danos *is*. | | /system/devices | danos virtual device tree e.g. similar to /sys on linux but with danos device tree conventions (the structures in the devices module) | -| /system/drivers | driver binaries, one sub-project each (e.g. /system/drivers/hpet) | +| /system/drivers | driver binaries, one sub-project each (e.g. /system/drivers/pci-bus, /system/drivers/ps2-bus) | | /system/services | system-service binaries — the VFS server, init, and other user-mode servers (e.g. /system/services/vfs, /system/services/init) | | /system/kernel | the kernel image | | /tmp | Directory for temporary files (see also /var/tmp). Often not preserved between system reboots and may be severely size-restricted. | @@ -61,8 +61,8 @@ to the driver in the order written, and a read consumes what is there. Terminals serial lines, keyboards and mice are all of this shape. These are the natural first device nodes in danos, because a character driver needs nothing the kernel doesn't already provide — it claims its device, maps its registers with `mmio_map`, and blocks -on `replyWait` for either an interrupt or a client request. `system/drivers/hpet/hpet.zig` is already -that program, minus the client half. +on `replyWait` for either an interrupt or a client request. `system/drivers/ps2-bus/ps2-bus.zig` +is already that program, minus the file-node client half. The obstacle was never the file type; it is which hardware a ring-3 driver can reach. Direct `in`/`out` from user space is still a #GP (no TSS I/O bitmap, IOPL never raised), diff --git a/docs/device-interrupts.md b/docs/device-interrupts.md index fe71a7b..874d63e 100644 --- a/docs/device-interrupts.md +++ b/docs/device-interrupts.md @@ -78,6 +78,40 @@ preemption and wakeups (1 ms granularity); the **TSC** is the resolution you rea time at. Making `sleep` itself sub-millisecond would take a tickless one-shot timer — a later step. +### Is the TSC trustworthy? Invariant, and synchronized + +A cycle counter is only a valid *clock* if two things hold, and danos checks both, +because they decide whether we read time with a cheap `rdtsc` or fall back to the HPET. + +**Invariant.** An old TSC counted core clock cycles, so it sped up and slowed down with +frequency scaling — useless as wall time. Modern CPUs (all of danos's targets) provide an +**invariant TSC**: a constant rate across P/C-states that never stops. The guarantee is a +CPUID bit — leaf `0x80000007`, EDX bit 8 — on both Intel *and* AMD. danos reads it in +`calibrate`, and a TSC that doesn't advertise it is not used as the clocksource. AMD is +why this matters in practice: it doesn't populate the Intel leaf `0x15` that enumerates +the TSC *frequency*, so danos already measures AMD's rate against the HPET — but a +measured frequency without the invariance guarantee is not enough. + +**Synchronized.** Each core has its own TSC. Even invariant ones can start at different +values (a second socket, some firmware), so a thread migrating from a core reading +`1_000_000` to one reading `999_000` would see time jump *backward*. danos runs a **warp +check** as each application processor comes online (`checkWarpSource`, adapted from +Linux's): the waking core and the BSP hammer a shared "highest seen" TSC under a lock, +and if either ever reads below it, the cores' TSCs are skewed. It's pairwise because APs +come up one at a time ([smp.md](smp.md)). + +**The fallback.** When the TSC fails either test — non-invariant (a bare VM such as the +default qemu64), or warped between cores — danos moves the monotonic clock onto the +**HPET** main counter: one fixed-rate counter, so it can neither skew between cores nor +drift with frequency. It costs a memory-mapped read instead of a register read, but it +keeps time *accurate*, which is the whole point. The switch preserves the current value, +so the clock never jumps. The boot log names the outcome: + +``` +/system/kernel: clocksource tsc (TSC invariant: yes, synchronized: yes) # real Intel/AMD +/system/kernel: clocksource hpet (TSC invariant: no, synchronized: yes) # a bare VM (TCG) +``` + ## Two kinds of vector, one dispatch The IDT now installs gates `0-47`: the 32 exceptions plus the device range. Every diff --git a/docs/device-manager.md b/docs/device-manager.md index 2b50bc9..70ace1c 100644 --- a/docs/device-manager.md +++ b/docs/device-manager.md @@ -51,7 +51,16 @@ enumeration is a **pci-bus driver**: the manager spawns it against the host brid like any bus reports children. ACPI becomes an **acpi service** that interprets the tables and reports the namespace. The manager only orchestrates and merges. Moving AML interpretation out of ring 0 is its own project on its own track; nothing here -depends on when it lands. +depends on when it lands. (It landed: [discovery.md](discovery.md), M19–M20.) + +`device_register` is **idempotent on exact match**: a re-registration with an +identical (parent, class, identity, resources) tuple returns the existing id +instead of appending a duplicate. The kernel table has no unregister, so without +this a restarted registering bus would re-report its children as fresh nodes on +every respawn. Idempotence is what makes restart-and-re-report sound for *every* +reporting bus — pci-bus, the acpi service, a future fdt service — not just one, +and it is why supervision (below) can prune a dead bus's subtree and trust the +restarted instance to rebuild exactly the same ids. ## The protocol @@ -136,10 +145,17 @@ published exit events, signals + `runtime.process`). On top of those: the mouse and keyboard QEMU already hangs off it. 7. **App surface**: `enumerate`/`subscribe` over IPC; `device_enumerate` retreats to a manager-internal seam. -8. **Discovery migration** — DONE (M19–M20, 2026-07-13): pci-bus driver (M19) - then the acpi service (M20) moved enumeration to ring 3; the kernel seeds - only the host bridge and the acpi-tables node. See - [m19-m20-plan.md](m19-m20-plan.md). +8. **Discovery migration** — DONE (M19–M20, 2026-07-13): enumeration moved to + ring 3 as swappable per-firmware discoverers — the pci-bus driver (M19) then + the acpi service (M20), see [discovery.md](discovery.md); the kernel seeds + only the host bridge and the acpi-tables node. Matching moved with it: + `child_added` grew a `device_id` (the kernel-registered id, `no_device` for + unregistered leaves like USB ports) and a firmware `hid`, and the manager now + matches drivers from those **reports** rather than its boot-time snapshot. The + PCI arm flipped in M19.3, the ACPI arm (ps2-bus matched from `_HID`) in M20.3 + — each in a single phase so no device is ever matched from both sources at + once. The acpi service reports only the non-PCI `_HID` devices, since pci-bus + already reports PCI functions (M20.2). ## Settled questions (2026-07-12) diff --git a/docs/discovery.md b/docs/discovery.md index 94a0579..0f1298d 100644 --- a/docs/discovery.md +++ b/docs/discovery.md @@ -191,3 +191,57 @@ and registers + reports each `_HID` device — the device manager matches driver (ps2-bus) from those reports. With M19's pci-bus driver, discovery now runs entirely in user space; the kernel seeds only the host bridge and the acpi-tables node. + +## Discovery is a swappable process per firmware (M19–M20) + +Moving PCI and ACPI enumeration out of ring 0 was not just a relocation — it +made discovery **firmware-neutral by construction**, which is the whole reason +to do it before the second architecture rather than after. Everything at and +above the [device-manager](device-manager.md) protocol — descriptors, +containment, reports, matching, supervision — is generic and may never become +x86-specific. Discovery is the single firmware-specific piece, and it is +isolated as **one swappable process per firmware**: + +- **x86** boots describe hardware with ACPI, so the discoverer is the **acpi + service** ([acpi.md](acpi.md)): it claims the `acpi-tables` node and runs AML. +- **The Raspberry Pis** hand over a flattened device tree, so the discoverer is + an **fdt service**: it claims a `devicetree-blob` node and walks the tree — + pure data, no bytecode, so it needs neither a port grant nor an interpreter, + strictly simpler than ACPI. (A placeholder until the [aarch64](arm.md) + bring-up fills it in.) + +The device manager spawns the discoverer under the **neutral ramdisk name +`discovery`** and never learns which firmware it is on; the build's +`-Ddiscovery=acpi|fdt` option fills that slot (x86 defaults to `acpi`, the +aarch64 target flips the default when it lands). The manager owns the device +tree as *data* and touches no hardware, ever — firmware bytecode runs only +inside the crashable, supervised discoverer, so an AML fault can never take +down the supervisor. + +Two consequences of neutrality bind on later work: + +- **Cross-firmware surfaces are named by domain, not firmware.** System power is + a [`power`](power.md) protocol, not an "ACPI events" protocol: on x86 the acpi + service registers it, on ARM a PSCI/mailbox service registers the same + `ServiceId.power`, and subscribers never learn the difference. +- **Identity must widen before the fdt service exists.** `DeviceDescriptor`'s + 8-byte `hid` holds an EISA id but cannot hold an FDT `compatible` string + (`"brcm,bcm2835-aux-uart"`); the identity field grows before the ARM path can + report a real node. + +Two supporting decisions keep the kernel's remaining slice honest: + +- **The AML interpreter is a shared build module**, compiled into both the + kernel and the acpi service — one source, two builds, no fork. The kernel + links it for the `\_S5` poweroff evaluation, the service links it for + everything else, and the `acpi-parse` test asserts the two produce the same + device count across the ring-3 move. +- **Bridge apertures come from the firmware memory map, not AML.** Registered + PCI functions carry BAR resources, and `device_register` containment demands + the bridge own windows that cover them. Those apertures are derived + kernel-side from the boot memory map's MMIO holes (regions that are neither + RAM nor tables) — mechanical, AML-free, and available at boot regardless of + what later moved to user space. The acpi service's authority is likewise + exactly one node: the `acpi-tables` node, whose broad io_port grant is the + documented trust boundary for the one process allowed to run firmware + bytecode. diff --git a/docs/driver-model.md b/docs/driver-model.md index 6b37e00..9016e1b 100644 --- a/docs/driver-model.md +++ b/docs/driver-model.md @@ -57,8 +57,9 @@ is not an address window. Discovery is trusted; user space is not. ### What a bus driver looks like -`system/drivers/bus/bus.zig` is the smallest honest one. Its "bus" is the HPET's register block and -its "devices" are the block's comparators: +danos ships no demo bus driver — the real ones are `pci-bus`, `ps2-bus`, and +`usb-xhci-bus`. The smallest *honest* shape, illustrated here with an HPET register block +as the "bus" and its comparators as the "devices", is: ```zig _ = dev.claim(bus.id); // 1. own the bus @@ -78,8 +79,8 @@ for (0..n) |i| { // 3. publish each child Each child is left **unclaimed**, which is the handoff: a comparator driver can now `device_claim` one and `mmio_map` it, and will see only its own 0x20-byte window. A child -whose window escapes the bus is refused — `bus` asserts that, and the `bus` test -asserts the kernel's table upholds it. +whose window escapes the bus is refused; the in-kernel `containment` test asserts the +kernel's table upholds that ([drivers.md](drivers.md)). A USB device has *no* resources at all: `resource_count = 0`, because it's addressed through its controller, not by MMIO. That case is allowed and is the common one. @@ -143,7 +144,7 @@ If a class driver needs `mmio`, it has become an HCD and should be one. physically-contiguous, pinned, uncacheable, reclaim-on-teardown buffers with the physical address exposed (`pmm.allocContiguous`, a DMA arena, `mapUserDmaInto`). `dma_below_4g` caps the address for legacy engines; `dma_write_combining` is accepted - but falls back to coherent until PAT is programmed. hpet is refactored onto `/lib/mmio`; + but falls back to coherent until PAT is programmed. The bus drivers use `/lib/mmio`; no DMA driver consumes `dma_alloc` yet. - **M15** — interrupts for PCI devices, the MSI half. Discovery now gives every PCI function its 4 KiB ECAM config space as resource 0 (unblocking the capability walk @@ -302,8 +303,8 @@ rather than an out-struct. The rest of this section is the original design note. **The blocker, and it's a hard one.** No PCI device can take an interrupt today. [`addBars`](system/devices/acpi.zig) records `.memory` and `.io_port` BARs and never an -`.irq`; there is no `_PRT` parsing anywhere in the tree. `hpet` only works because the -HPET advertises its own routing options in its own registers — a privilege no ordinary +`.irq`; there is no `_PRT` parsing anywhere in the tree. The HPET is the one exception — +it advertises its own interrupt routing in its own registers, a privilege no ordinary device has. **The fix, in two halves.** @@ -326,7 +327,7 @@ which means **discovery should give each `pci_device` a `.memory` resource for i 4 KiB ECAM slot**. That's a small change to `parseMcfg` and it unblocks the whole capability walk (MSI, MSI-X, PCIe extended caps) without any new syscall. -Note QEMU's HPET reports `Tn_FSB_INT_DEL_CAP = 0` — no MSI — so `hpet` can never +Note QEMU's HPET reports `Tn_FSB_INT_DEL_CAP = 0` — no MSI — so an HPET timer could never exercise this path. The first MSI driver will be the first PCI driver. ## M16 — the IOMMU, and the honest caveat ◑ detection done, enforcement pending @@ -353,7 +354,7 @@ gap should be named rather than implied. `M13` (capability passing) is independent of `M14`/`M15` and is the cheapest. It unlocks class drivers, which are the shape with no hardware requirements at all — you -could write a real one against `bus`'s comparators tomorrow. +could write a real one against any device a bus driver publishes tomorrow. `M14` and `M15` together unlock the first HCD. `M14`'s barrier layer is worth landing on its own regardless: it's small, obviously correct, and stops every future driver diff --git a/docs/drivers.md b/docs/drivers.md index d2bb88a..74ea9bd 100644 --- a/docs/drivers.md +++ b/docs/drivers.md @@ -22,12 +22,12 @@ say.* ## How a driver gets started: discover, match, spawn -Nothing in the kernel decides that the HPET needs the `hpet` driver — that is policy, -and policy lives in user space. Boot brings user space up as a three-level supervision -hierarchy, each level owning one job: +Nothing in the kernel decides that the PCI host bridge needs the `pci-bus` driver — that +is policy, and policy lives in user space. Boot brings user space up as a three-level +supervision hierarchy, each level owning one job: ``` -kernel ──spawns──► init (PID 1) ──spawns──► device-manager ──spawns──► hpet +kernel ──spawns──► init (PID 1) ──spawns──► device-manager ──spawns──► pci-bus | | | spawns only init, the service supervisor: the driver supervisor: enumerates publishes the starts the system /system/devices, matches each device @@ -184,7 +184,11 @@ Two properties worth knowing: ## A whole driver -`system/drivers/hpet/hpet.zig` is ~150 lines and does all of it. The shape: +A minimal leaf driver is only ~150 lines and does all of it. danos ships **no such +example binary** — the driver model is proven by the real drivers (`pci-bus`, `ps2-bus`, +`usb-xhci-bus`), and a teaching example belongs here, in the docs, rather than as a +compiled program nobody runs. Illustrated with a hypothetical HPET timer driver, the +shape is: ```zig const hpet = findHpet(buf) orelse return; // device_enumerate, look for @@ -209,8 +213,8 @@ while (...) { } ``` -The HPET is a good first driver for a reason that isn't obvious. Its *counter* is a -clocksource — the only way to use it is to read it, so it proved `mmio_map` without +The HPET makes a good illustration for a reason that isn't obvious. Its *counter* is a +clocksource — the only way to use it is to read it, so it exercises `mmio_map` without needing interrupts at all. Its *comparators* are a clockevent, and can be configured **level-triggered** (`Tn_INT_TYPE_CNF`), which asserts a bit in `GENERAL_INT_STATUS` that the driver must write-1-to-clear. That's a genuine deassert step, so the full @@ -252,9 +256,10 @@ bus driver may only ever subdivide what it already owns. A device with **no resources** is legal and common. A USB device is reached through its controller, not by MMIO, so it gets `resource_count = 0`. -See [`system/drivers/bus/bus.zig`](../system/drivers/bus/bus.zig) for a complete one, and -[driver-model.md](driver-model.md) for how bus drivers, class drivers and host -controller drivers fit together. +See [`system/drivers/pci-bus/pci-bus.zig`](../system/drivers/pci-bus/pci-bus.zig) for a +real one — it claims a PCI host bridge, maps its ECAM window, and publishes each function +it finds as a child — and [driver-model.md](driver-model.md) for how bus drivers, class +drivers and host controller drivers fit together. ## What the kernel does not do for you @@ -313,32 +318,38 @@ uncacheable, physical address exposed), and **memory barriers** (`/lib/mmio`'s ## Verifying it -The `hpet` test spawns `hpet` from the initial ramdisk and watches the serial log. The driver -prints `hpet: ok` only after being woken five times, and its loop's only exit is -through `replyWait` returning a notification — it cannot reach that line by polling. +No demo driver ships to prove this end to end; the *real* drivers do, so the tests +target them and the kernel primitives directly: -The last check doesn't trust the driver's self-report at all: the kernel reads the I/O -APIC redirection entry back and asserts the line really is routed to a device vector, -really is level-triggered, and really was left unmasked by the driver's final -`irq_ack`. +- **`device-manager`** — boots only the device manager, which discovers the PCI host + bridge, matches `pci-bus`, and `system_spawn`s it. The test reads kernel state — the + process table and the device tree — to confirm pci-bus came up and registered the + functions it enumerated: the whole discover → match → spawn → driver-up chain. +- **`acpi-ps2`** — a user-space driver (`ps2-bus`) is woken by its device's IRQ, + delivered as an IPC notification, and attaches the keyboard: IRQ-as-IPC, end to end. +- **`pci-scan`** — a user-space driver (`pci-bus`) maps its device's MMIO (the ECAM + window) and walks it: `mmio_map`, end to end. +- **`containment`** — the kernel refuses a `device_register` whose child window escapes + the parent's grant (else it would be a syscall for mapping arbitrary memory), while an + identical re-register stays idempotent. Asserted in-kernel, straight against the broker. +- **`irqfree`** — the teardown path. Binds two owners to one shared endpoint, releases + one, and reads the I/O APIC back: the departing owner's line is masked, the sibling's + is not. That second half is why bindings are keyed on the owning *task* and not on the + endpoint pointer — endpoints are shared, so releasing "everything pointing at this + endpoint" would silently mask a live driver's device. +- **`iopass`** — the `device_grant` teardown rule, so destroying a driver's address + space never returns MMIO frames to the RAM pool. ``` -$ python3 test/qemu_test.py hpet irqfree iopass - hpet ... PASS (matched 'DANOS-TEST-RESULT: PASS') +$ python3 test/qemu_test.py device-manager acpi-ps2 pci-scan containment irqfree iopass + device-manager ... PASS (matched 'DANOS-TEST-RESULT: PASS') + acpi-ps2 ... PASS + pci-scan ... PASS (matched 'DANOS-TEST-RESULT: PASS') + containment ... PASS (matched 'DANOS-TEST-RESULT: PASS') irqfree ... PASS (matched 'DANOS-TEST-RESULT: PASS') iopass ... PASS (matched 'DANOS-TEST-RESULT: PASS') ``` -Two companions cover what `hpet` can't, because it never exits: - -- **`irqfree`** — the teardown path. Binds two owners to one shared endpoint, releases - one, and reads the I/O APIC back: the departing owner's line is masked, the sibling's - is not. That second half is why bindings are keyed on the owning *task* and not on - the endpoint pointer — endpoints are shared, so releasing "everything pointing at - this endpoint" would silently mask a live driver's device. -- **`iopass`** — the `device_grant` teardown rule, so destroying a driver's address - space never returns MMIO frames to the RAM pool. - ## What's next (not done here) The big driver-model pieces — capability passing (class drivers), DMA + barriers, MSI, diff --git a/docs/m17-m18-plan.md b/docs/m17-m18-plan.md deleted file mode 100644 index 8518c9f..0000000 --- a/docs/m17-m18-plan.md +++ /dev/null @@ -1,210 +0,0 @@ -# M17–M18 execution plan: process lifecycle + device manager - -**Archived — completed 2026-07-13** (every item checked; suite ended 54/54). -Kept as the record of how M17–M18 landed; the successor is -[m19-m20-plan.md](m19-m20-plan.md). - -The operational plan for building [process-lifecycle.md](process-lifecycle.md) -(M17) and [device-manager.md](device-manager.md) increments 5–7 (M18). Design is -settled in those documents; this file is the build order — one phase at a time, -each phase green before the next starts. Delete or archive this file when M18 -lands. - -**Definition of green, every phase:** `zig build` clean, `zig build test` clean, -`python3 test/qemu_test.py` passes (existing scenarios plus the phase's new one), -and the relevant design doc's "known gaps" / status lines updated. Commit per -green phase (no co-author trailers). - -**Workflow (settled 2026-07-12):** work happens in a dedicated git worktree, on -feature branches cut from `main` — `feat/process-lifecycle` (M17.1–17.4), -`feat/device-manager` (M18.1), `feat/usb-xhci-bus` (M18.2–18.3). When a branch's -phases are all green it is **auto-merged into `main`**; branches are kept after -merge, not deleted. Merges and branches are pushed to origin. Phase 0 (once): -commit the design docs, merge the outstanding `feat/usb` work into `main`, and -run the existing QEMU suite green before any new work starts. - -**Numbering note:** continues the milestone sequence (driver track ended at M16). - -## Status - -The loop marks a phase `[x]` in the same commit that lands it. A phase is marked -only when its definition of green holds. - -- [x] **Phase 0** — baseline: docs committed, feat/usb merged to main, pushed; - `usb-xhci-libary.zig` renamed to `usb-xhci-library.zig`; existing QEMU - suite green from the worktree (48/48, 2026-07-12). -- [x] **M17.1** — kernel releases claims/MSI on death (claims: `releaseAllOwnedBy` - in the reap; MSI was already swept by `irq.releaseOwner`; `claim-release` - test; suite 49/49) -- [x] **M17.2** — exit reasons (`ExitReason` recorded at exit/fault/kill before - the notification; `process_exit_reason` supervisor-gated; - `runtime.process.exitReason`; kernel + ring-3 assertions; suite 49/49) -- [x] **M17.3** — published exit events + VFS subscriber (`process_subscribe`, - bounded ref-counted table, publish on every death; - `runtime.process.subscribeExits`; VFS handles carry owners and are swept on - the owner's death; `vfs-client-death` test; suite 50/50) -- [x] **M17.4** — signals, timer notifications, `runtime.process`, the service - harness (signal_bind/process_signal + coalescing pending mask; timer_bind - on the tick; bindSignals/signalsFrom/sendSignal/stop + timerOnce; - runtime.service.run with the zero-length ping; VFS converted; `signals` - scenario; suite 51/51) -- [x] **merge** `feat/process-lifecycle` → main, push (merged 2026-07-13) -- [x] **M18.1** — device-manager protocol: hello + restart policy - (device-manager-protocol module; the manager as a harness service: - supervised spawns, hello deadline via timer sweep, restart with - 300/600/1200ms backoff, exit reasons deciding restart-vs-stopped, - crash-loop cap; usb-xhci-bus first conforming driver; crash-test fixture - re-proving claim release each respawn; `driver-restart` scenario; - maximum_tasks 16→32 — the sweep was overflowing the pool; suite 52/52) -- [x] **merge** `feat/device-manager` → main, push (merged 2026-07-13) -- [x] **M18.2** — xHCI port scan + tree reports (child_added/child_removed in - the protocol; the manager's child mirror with death-pruning; xHCI maps the - register BAR — resource 0 is ECAM — reads CAPLENGTH/HCSPARAMS1, scans - PORTSC, reports connected ports with speed-class identity; `usb-report` - scenario proves report → prune → respawn → re-report; suite 53/53) -- [x] **M18.3** — app surface: enumerate/subscribe over IPC (subscriber - endpoint rides as the call's capability; events are the same structs the - buses send); device-list first client; protocol capped at the kernel's - IPC MESSAGE_MAXIMUM (256); the startUserTask debug print removed — it - sheared concurrent serial lines and was the scenario-flake root cause; - `device-list` scenario; suite 54/54) -- [x] **merge** `feat/usb-xhci-bus` → main, push (merged 2026-07-13) — **plan complete** - ---- - -## M17.1 — the kernel releases a dead process's claims - -The cleanup half of iron rule 1; the prerequisite for every restart story. - -- `system/kernel/devices-broker.zig`: `releaseAllOwnedBy(owner: u32)` — clear - every `claimed[]` slot holding this task id. -- `system/kernel/process.zig`: call it from the reap path, alongside the existing - IRQ-binding release (the ordering comment there says why IRQs go first — claims - slot in after them, before the exit notification). -- MSI vectors: find where `msi_bind` records per-device vectors (interrupts - module) and release those by owner in the same pass. -- Docs: remove the claims bullet from process-management.md "Known gaps". - -**Test:** new QEMU scenario `claim-release` — a test child claims an unclaimed -device, is killed, is respawned, and claims the same device again successfully; -assert both claims in the serial log. Kernel-side unit coverage in -`system/kernel/tests.zig` for `releaseAllOwnedBy` (claim two devices as two owners, -release one owner, verify exactly its claims freed). - -## M17.2 — exit reasons - -- `system/abi.zig`: `ExitReason` (exited, aborted, segmentation_fault, - illegal_instruction, arithmetic_fault, killed). -- Kernel: record the reason at every death site — clean exit path, each fault - class in `onException`, the kill path. Bounded recent-exits table (ids are never - reused, so a small ring keyed by id is enough). -- New system call `process_exit_reason(id)` — supervisor-gated, like kill; returns - the recorded reason or `-ESRCH` once evicted. -- `library/runtime/process.zig`: `ExitReason` + `exitReason(id: u32)`. -- Docs: remove the no-exit-status bullet from process-management.md. - -**Test:** extend the `supervision` scenario — three children: one exits cleanly, -one faults (the fault-recovery pattern), one is killed; the supervisor asserts all -three reasons. - -## M17.3 — published exit events - -- Kernel: bounded subscriber table (endpoints); new system call - `process_subscribe(endpoint)` (ungated, like `process_enumerate`); every death - posts `notify_exit_bit | id` to each subscriber — the same post the supervisor - path already uses. -- `library/runtime/process.zig`: `subscribeExits(endpoint)`. -- VFS becomes the first subscriber: on an exit event, release every handle keyed - by that task id (badges already are task ids). Log the release. -- Docs: note the convention in ipc.md (exit events reuse the exit-notification - badge encoding). - -**Test:** new QEMU scenario `vfs-client-death` — a client opens a file and is -killed without closing; assert the VFS logs the handle release and its open-handle -count returns to baseline. - -## M17.4 — signals and the service harness - -- Kernel: per-task pending mask + bound endpoint; system calls - `signal_bind(endpoint)` and `process_signal(id, signal)` (supervisor-or-self - gated); delivery posts `notify_signal_bit | pending mask`, coalescing; pending - signals with no bound endpoint pend silently. -- `library/runtime/process.zig`: `Signal`, `SignalSet`, `bindSignals`, - `signalsFrom`, `sendSignal`, `stop(id, deadline_ms)` (terminate → wait for exit - notification → kill). Implement `terminate`, `reload`, `user_1`, `user_2`; - `interrupt`/`quit` are enum members with no sender yet; `alarm` stays unbuilt. -- Kernel: **one-shot timer notifications** — `timer_bind(endpoint, ms)` posts a - notification badge when the deadline lands (IRQ-as-IPC again, on the timer - wheel `sleep` already uses). This is the missing timed-wait primitive: - `replyWait` blocks forever and `sleep` blocks the whole process, but `stop()`'s - escalation, the device manager's `hello` deadline (M18.1), and restart backoff - all need a deadline while staying responsive. It is also the mechanism `alarm` - gets for free later. -- New `library/runtime/service.zig`: the harness — `run(callbacks)` owning the - replyWait loop, folding protocol messages, signals, and child-exit notifications - into `init` / `on_message` / `on_reload` / `on_terminate`; answers the common - `ping` automatically. Define the reserved `ping` request encoding here and - document it in ipc.md (one obvious encoding; smallest that cannot collide with - existing protocols). -- Convert one existing service (input-source or hpet) to the harness as proof it - subtracts code rather than adding it. - -**Test:** extend `supervision` — a harness-built child: `sendSignal(reload)` -observed in its log, `ping` answered, `stop()` produces a clean exit with reason -`exited`; a second child that ignores signals (no bind) is killed by `stop()`'s -deadline with reason `killed`. - -## M18.1 — device-manager protocol: hello + restart policy - -- New `system/services/device-manager/device-manager-protocol.zig` module - (vfs-protocol pattern): `hello { version, role, device_id }`; version constant; - reserved fields. -- Device manager: register the `.device_manager` endpoint; spawn drivers with its - exit endpoint; enforce the hello deadline; restart policy — backoff, crash-loop - cap (three fast deaths → mark failed, log, stop), reasons from M17.2 deciding - restart vs not. -- usb-xhci-bus: adopt the harness + send hello. hpet/ps2-bus follow only if the - conversion is mechanical; otherwise they keep working unconverted (the manager - only enforces hello on drivers spawned with an assignment). -- build.zig: test-loop entry for the protocol module if it grows pure logic. - -**Test:** new QEMU scenario `driver-restart` — the xHCI driver takes a test-only -argv flag to fault after hello on its first run; assert: fault, exit reason -recorded, manager respawns with backoff, second run claims the controller -(M17.1) and hellos clean. Assert the crash-loop cap by a driver that always -faults (a tiny test driver, not xhci). - -## M18.2 — bus tree reports - -- Protocol: `child_added { parent, identity, resources }` / `child_removed { id }`. -- usb-xhci-bus: bring-up to **port scan only** — map the MMIO window (claimed in - M16-era work), controller reset/start per xHCI spec, walk the port registers, - report one `child_added` per connected port with speed + port number as - identity. **No transfer rings, no descriptors** — reading device/interface - descriptors (and therefore USB class triples for matching) is the follow-on USB - track, not this plan. -- Device manager: mirror reports into its tree; prune the subtree (emitting - `child_removed`) when a bus driver dies; assert re-report on restart. - -**Test:** QEMU already attaches usb-kbd + usb-mouse on xhci.0 — assert two -`child_added` events reach the manager and appear in its tree dump; kill the -driver, assert two `child_removed` then two fresh `child_added` after respawn. - -## M18.3 — the application surface - -- Protocol: `enumerate` (tree snapshot) + `subscribe` (published add/remove - events, input-service pattern). -- A small client (`device-list`, the `ps` analog) exercising both; the manager - becomes the one answer to "what devices exist" for user space. - `device_enumerate` stays for drivers/kernel seeding — its retreat is tied to the - discovery migration, out of this plan. - -**Test:** QEMU scenario — `device-list` shows the tree including USB children; -during a driver restart the subscribing client logs remove + add events. - ---- - -**Explicitly out of scope** (own tracks, after M18): discovery migration (pci-bus -driver, acpi service, retiring the kernel scan), USB control transfers + -descriptors + class-driver matching, the musl layer, `interrupt`/`quit` senders -(needs a console), job control. diff --git a/docs/m19-m20-plan.md b/docs/m19-m20-plan.md deleted file mode 100644 index 117aa5a..0000000 --- a/docs/m19-m20-plan.md +++ /dev/null @@ -1,196 +0,0 @@ -# M19–M20 execution plan: discovery migration - -The operational plan for [device-manager.md](device-manager.md)'s increment 8: -discovery leaves the kernel — a **pci-bus driver** (M19) and an **acpi service** -(M20), with the kernel's device enumeration retired behind them. Same rules as -[m17-m18-plan.md](m17-m18-plan.md): one phase at a time, each green before the -next; this file is the build order and the checklist. - -**Definition of green, every phase:** `zig build` clean, `zig build test` clean, -`python3 test/qemu_test.py` passes (existing scenarios plus the phase's new -one), and the relevant design doc updated. Commit per green phase (no co-author -trailers). The full suite is the regression net — the existing -`driver-restart` / `usb-report` / `device-list` / `input` scenarios must stay -green *through* the migration, which is the whole point: the system must not be -able to tell who enumerated it. - -**Workflow:** dedicated worktree; branches off `main` — `feat/pci-bus` -(M19.0–19.3), `feat/acpi-service` (M20.1–20.3); auto-merge to main when a -branch is green; keep branches; push everything. - -## Settled decisions (2026-07-13 — veto before the loop starts) - -1. **What "retiring the kernel scan" means.** The kernel keeps, forever, the - parses it needs before user space exists: RSDP/XSDT location, MADT (SMP), - the HPET table (the tick), FADT + the AML `\_S5` evaluation (poweroff — the - power tests prove it), and MCFG (the host bridge node). What retires is - **device enumeration**: the ECAM function walk (M19.3) and the DSDT/SSDT - namespace walk that builds device nodes (M20.3). The AML module stays a - shared build module compiled into both the kernel (for `\_S5`) and the acpi - service (for everything else) — same source, two builds, no fork. -2. **Bridge apertures come from the firmware memory map, not AML.** Registered - PCI functions carry BAR resources, and containment demands the bridge own - windows that cover them. The apertures are derived kernel-side from the - boot memory map's MMIO holes (regions that are neither RAM nor tables) — - mechanical, AML-free, and available at boot regardless of what later moved - to user space. (The bridge today carries only ECAM + bus range; this is the - prerequisite M19.0 exists for.) -3. **`device_register` becomes idempotent on exact match.** A re-registration - with identical (parent, class, resources) returns the existing id instead - of appending. The kernel table has no unregister, so without this a - restarted registering bus would duplicate its children on every respawn — - idempotence makes restart-and-re-report safe for every future bus, not just - PCI. -4. **The manager matches from reports.** `ChildAdded` gains a `device_id` - field (the kernel-registered id, `no_device` for unregistered leaves like - USB ports). After the M19.3 flip, PCI driver matching keys off reported - identity (the class triple) instead of the manager's boot-time snapshot — - the snapshot match remains only for what the kernel still seeds. One flip - phase changes both sides at once so no device is ever matched twice. -5. **The acpi service's authority is one node.** The kernel publishes an - `acpi-tables` device: memory resources covering the table blobs plus a - broad `io_port` resource — the documented trust grant to exactly one - process (AML OperationRegions reach EC/PM ports; the claim-gated - io_read/io_write calls already exist). The service claims it, maps the - tables, and runs the shared AML module in ring 3 behind a `Hal` backed by - `mmio_map` + `io_read`/`io_write`. -6. **Both new processes are protocol drivers** under the manager: hello, - supervision, restart with backoff — all inherited from M18.1 for free. - Registration idempotence (decision 3) is what makes their restarts sound. -7. **Firmware neutrality is the contract** (2026-07-13). The generic layer is - everything at and above the device-manager protocol — descriptors, - containment, reports, matching, supervision — and none of it may become - x86-specific. Discovery is one swappable process per firmware: the acpi - service on x86; an **fdt service** on the Raspberry Pis (claims a - `devicetree-blob` node, reports children from the flattened device tree — - pure data, no bytecode, no port grant, strictly simpler than ACPI). The - manager owns the tree as *data* and touches no hardware, ever — AML runs in - a crashable, supervised discoverer precisely so a firmware-bytecode fault - can never take down the supervisor. Two consequences recorded now: - `DeviceDescriptor`'s 8-byte `hid` cannot hold an FDT `compatible` string - ("brcm,bcm2835-aux-uart") — identity widens before the fdt service exists; - and cross-firmware surfaces are named by **domain, not firmware** (M21 - defines a *power* protocol, not an "ACPI events" protocol — PSCI/mailbox - sources feed the same subscribers on ARM). **Landed early (2026-07-13):** - both services exist as placeholders (system/services/acpi, system/services/ - fdt) and the build's `-Ddiscovery=acpi|fdt` option fills the ramdisk's - neutral `discovery` slot — the manager will spawn "discovery" by that name - in M20.3 and never learn which firmware it is on. - -## Status - -- [x] **M19.0** — prerequisites (bridge apertures from the memory map's - *gaps* — the single-hole rule died on OVMF's flash at the top of 4 GiB, - caught by the new every-BAR-contained assert in `discovery`; idempotent - `device_register` proven in `bus`; `ChildAdded.device_id`; - m17-m18-plan.md archived; suite 54/54). -- [x] **M19.1** — pci-bus driver, scan only (claims the bridge, maps ECAM - through its grant, brute-force walk with the multifunction rule; the - manager matches pci_host_bridge → pci-bus per device with the full - protocol contract; `pci-scan` builds its expected marker from the - kernel's own count — equivalence on the first run; suite 55/55). -- [x] **M19.2** — register + report (BAR probe mirrored byte-for-byte from the - kernel's addBars so dedupe returns the kernel's node ids during - coexistence; the bridge gained the io_port aperture I/O BARs need; - reports carry the registered device_id; pci-scan drills a forced restart - and asserts the PCI node count never grows — plus harness hardening: a - failing case now preserves its serial as -failed-serial.log, and - the heavy scenarios run at 150s; suite 55/55). -- [x] **M19.3** — the flip: kernel `enumeratePci`/`addBars`/`PciHeader` all - deleted (bridge node stays); manager matches PCI drivers from reported - identity, deduped by registered id. Surfaced and fixed a real SMP race the - flip created — ring-3 device_register made the broker table concurrent, so - mmio_map's lock-free read intermittently tore hpet's resource length - (user fault) and overflowed `r.len-1` into a kernel panic; now the broker - read is under the big lock and the arithmetic is guarded, and pci-bus - skips size-0 BARs. discovery.md updated; suite 55/55 (driver-restart - hammered 6×). -- [x] **merge** `feat/pci-bus` → main, push (merged 2026-07-13). -- [x] **M20.1** — acpi service, parse only: the AML interpreter is now a build - module compiled into both kernel and service; the kernel publishes the - `acpi-tables` node (AML blobs as memory resources, the broad io_port grant, - the SCI); the service claims it, maps the blobs, runs the shared parser in - ring 3, and self-verifies its Device count against the kernel's (34 = 34, - deterministic via argv, no log-scraping); the manager spawns `discovery` - at startup. Parse-only touches no hardware. Suite 56/56. - -- [x] **M20.2** — register + report: the service evaluates `_STA`/`_CRS` in - ring 3 (interpreter Hal = port I/O over the claimed node; a scratch page - backs SystemMemory maps so a stray region can't fault it) and registers + - reports each present `_HID` device under `acpi-tables`. Containment: the - broker's irq check became range-based (len-1 == the old equality) so the - node's broad irq window covers children's legacy lines; io ports fall in - the broad io grant. ChildAdded gained `hid`. Matching stays off. The - `acpi-report` scenario asserts the PS/2 keyboard (3 resources) and mouse - (1 resource) among the reports. Suite 57/57. -- [x] **M20.3** — the flip: the kernel's `wireAcpiDevices` call is gone (the - device-building helpers are retained-but-dead pending a focused sweep, - spawned as a task; static tables + `\_S5` + the acpi-tables node stay). - The manager matches ps2-bus from ACPI `_HID` reports; the service - registers all devices before reporting any (no keyboard-before-mouse - race). The `acpi-ps2` scenario proves report → spawn → ps2-bus attaches - its keyboard; `ioport` retargeted to the acpi-tables I/O window (the - kernel-built PS/2 node is gone). Suite 58/58. -- [x] **merge** `feat/acpi-service` → main, push (merged 2026-07-13) — **discovery migration complete**. - ---- - -## Phase notes - -**M19.0 apertures:** the boot memory map already crosses the handoff -([boot-handoff]), but discovery never sees it today — expect a small -pass-through (kernel init hands the map to the platform layer) before the -holes computation, which belongs where the bridge node is built -(`parseMcfg`). Sanity-check on QEMU q35: the xHCI BAR (`0xc0000000`-region -values seen in the M18 logs) must land inside a derived aperture, asserted in -the kernel unit test. - -**M19.1 scanning without owning config access twice:** the driver reads config -space through its ECAM mmio_map grant of the *bridge* window — the same bytes -the kernel walk read. Vendor-id `0xFFFF` skip, header-type multifunction rule, -no bridge recursion (matches the kernel's current single-segment walk). - -**M19.2 BAR sizing:** the classic size probe (write all-ones, read mask, -restore) is deferred — the BARs' current programmed values and types are -enough for containment-checked registration at bring-up; sizing lands with the -first driver that needs to *move* a BAR. Log what is registered so the -scenario can assert it. - -**M19.3 what the manager still seeds from the snapshot:** everything the -kernel still enumerates (timers, ACPI nodes until M20.3). The PCI arm of -`pciDriverFor` switches source; `driverFor` doesn't move until M20.3. - -**M20.1 spawn and identity (pre-settled 2026-07-13):** the manager spawns -`discovery` by its neutral ramdisk name at startup, as an ordinary protocol -driver (hello, supervision) — from M20.1 on, on every boot. For reporting ACPI -devices, `ChildAdded` gains `hid: [8]u8` (EISA ids fit; zero = none): -firmware *string* identity travels beside the numeric `identity` field until -the FDT-driven widening replaces both (decision 7). - -**M20.1 Hal in ring 3:** `mapMmio` → `device.mmioMap` over the claimed -acpi-tables node (plus a table-offset map for blobs); `pioRead`/`pioWrite` → -`device.ioRead`/`ioWrite` against its io_port resource. The interpreter cannot -tell it moved — that is the assertion of `acpi-parse`. - -**M20.2 containment for `_CRS`:** io ports fall inside the node's broad -io_port resource; MMIO windows (HPET, LAPIC ranges some firmwares list) fall -inside the memory-map holes added to the node in M20.1. Anything that doesn't -fit is logged and skipped, loudly — bring-up honesty over silent drops. - -**M20.3 ps2 ordering:** ps2-bus binds nodes the acpi service now reports, so -its spawn moves behind the report (the manager's matching handles this once -the source flips); the `input` scenario proves the keyboard still types. - -**Explicitly out of scope:** PCI bridge recursion (single segment, flat bus -walk stays); BAR reprogramming/sizing; disk/PCIe hotplug; interrupt routing -changes (`_PRT` stays wherever it is today); the USB descriptor track; -multi-segment ECAM; per-device power states (D-states, `_PSx`/`_PRx`, -suspend/resume — a future *lifecycle-vocabulary* extension, since "suspend" -has the shape of a signal every driver must answer, and it has no consumer -until laptop sleep); CPU P/C-states. - -## M21 — ACPI events + system power — DONE - -Built and merged (docs/m21-plan.md, 2026-07-13): the SCI + power button, Notify/GPE -dispatch, and orderly shutdown (init's stop cascade into a ring-3 S5 write). -See that plan for the phase record. diff --git a/docs/m21-plan.md b/docs/m21-plan.md deleted file mode 100644 index 3fea641..0000000 --- a/docs/m21-plan.md +++ /dev/null @@ -1,138 +0,0 @@ -# M21 execution plan: ACPI events + system power - -The operational plan for the event side of the acpi service and orderly -shutdown — the capstone [m19-m20-plan.md](m19-m20-plan.md) previewed. Same -rules as its predecessors: one phase at a time, each green before the next; -this file is the build order and the checklist. - -**Definition of green, every phase:** `zig build` clean, `zig build test` -clean, `python3 test/qemu_test.py` passes (existing scenarios plus the -phase's new one), and the relevant design doc updated. Commit per green phase -(no co-author trailers). Failing cases preserve their serial logs -(`-failed-serial.log`). - -**Workflow:** dedicated worktree; branch `feat/power-events` off `main`; -auto-merge to main when the branch is green; keep the branch; push everything. - -## Settled decisions (2026-07-13, approved) - -1. **S5 is executed by the acpi service from ring 3.** No new syscall: the - broad port grant (M20 decision 5) already made this physically possible — - the service holds the PM1 control ports in its io grant and derives `_S5` - from its own namespace (`aml.sleepState`). Formalizing it adds no - authority. The kernel keeps `power.zig` for its own test paths and - panic-time use. -2. **The power surface is domain-named** (decision 7 of the last plan): a - `power-protocol` module + `ServiceId.power = 5`, registered by the acpi - service — on ARM, a PSCI/mailbox service registers the same id and - subscribers never know the difference. Messages: `subscribe` (endpoint as - the call's capability, the input/manager pattern), `shutdown` (accepted - only from PID 1 — init), and events published as buffered messages: - `power_button`, `lid`, `ac`, `battery`, generic `notify` with a code. -3. **The service learns event ports from its own FADT copy**: the kernel adds - the FADT as one more memory resource on the acpi-tables node; the service - tells it apart from the AML blobs by signature ("FACP" header — the blob - resources are header-stripped bytecode and start with no signature). The - kernel's own FADT parse is untouched. -4. **The acpi service converts to the harness** (`runtime.service.run`): - protocol messages (subscribe/shutdown), the SCI notification, and the - existing report flow fold into one loop — the shape it was always meant - to have. -5. **GPE/Notify correctness is proven by host unit tests** (synthetic AML - with a Notify inside a method body; aml.zig joins the `zig build test` - loop). The QEMU scenario proves the power button — a *fixed* event, - deterministically injectable via QMP `system_powerdown` — because QEMU - cannot raise GPEs deterministically on this config. Battery/AC/lid and the - embedded controller (`_Qxx`) are interface-complete here and validated on - real hardware (the laptop) later. - -## Ground truth the phases build on (verified 2026-07-13) - -- `system/devices/power.zig` `shutdown()` is the kernel's S5 write - (SLP_TYP|SLP_EN to PM1a/PM1b control); there is no power syscall. -- init (`system/services/init/init.zig`) spawns vfs/input/device-manager - fire-and-forget — no child ids kept, no signals, no event loop. The whole - stop toolkit exists in `runtime.process` (stop/sendSignal/bindSignals). -- `test/qemu_test.py` has no QMP channel (serial is a one-way file). -- The kernel parses PM1 *control* blocks and SCI_INT from the FADT; the PM1 - **event** blocks (offsets 56/60, len at 88) and **GPE0/GPE1** blocks - (offsets 80/84, lens 92/93) are unparsed — the service reads them from its - FADT copy (decision 3). -- The acpi-tables node carries the SCI as its only `len == 1` irq resource - (the broad window is len 256) — that is how the service finds it to - `irqBind`. -- `notify_opcode = 0x86` exists in `system/devices/aml/opcodes.zig` but the - interpreter never handles it — a GPE `_Lxx` body containing Notify fails - evaluation today. Everything else a GPE handler needs (field access, - control flow, method calls) is proven by the ring-3 `_STA`/`_CRS` work. -- The dead-code sweep (spawned task) also edits `system/devices/acpi.zig`; - M21.0 checks whether it landed and rebases before touching that file. - -## Status - -- [x] **M21.0** — baseline (dead-code sweep confirmed landed on main — no - acpi.zig conflict; `feat/power-events` cut; QMP channel in the harness: - always-on unix socket, client with the capabilities handshake, per-case - `qmp_after` hook, and a hook-must-deliver pass gate that the smoke case - now proves with a harmless query-status; suite 58/58). -- [x] **M21.1** — SCI + the power button (kernel appends the FADT as an - acpi-tables memory resource, tagged by its "FACP" header; `power-protocol` - module + `ServiceId.power = 5`; the acpi service converted to - `runtime.service.run`, registers `.power`, reads PM1 event/control + GPE - ports from its FADT copy, enables ACPI mode if SCI_EN is clear, binds the - SCI (the len-1 irq), sets PWRBTN_EN; the SCI handler clears PM1_STS, - logs `power: button pressed`, publishes `power_button`, acks. Scenario - `power-button` injects a real `system_powerdown` via QMP; initial-ramdisk - timeout 30→60s for the service's added boot work; suite 59/59). -- [x] **M21.2** — Notify + GPE dispatch (interpreter handles `notify_opcode` - into a bounded queue, cleared per-evaluate, drained via - `takeNotifications`; the service walks GPE status/enable bytes, evaluates - `\_GPE._Lxx`/`_Exx` per active bit, maps notified nodes to events - (battery/ac/lid/generic), clears GPE_STS write-1, acks. EC `_Qxx` out. - Host unit test with hand-encoded AML proves the queue; aml.zig joined the - `zig build test` loop. QEMU raises no GPEs — suite is regression net, - 59/59). -- [x] **M21.3** — orderly shutdown (init supervises its children on one - endpoint that also carries signals, power events, and a re-arming - heartbeat timer; on `power_button` or a `terminate` signal it logs - `init: shutting down`, runs `stop(child, 2000, endpoint)` in reverse - order, then requests `.power` shutdown; the acpi service honors shutdown - from a subscriber — init is the one subscriber, a soft gate that survives - testing where PID 1 isn't init — and writes SLP_TYP|SLP_EN from ring 3. - `orderly-shutdown` scenario proves button → shutting-down → S5 → QEMU - exit; suite 60/60). -- [x] **merge** `feat/power-events` → main, push, keep the branch (merged 2026-07-13) — **M21 complete**. - ---- - -## Phase notes - -**M21.0 QMP:** open the unix socket after Popen, complete the -`qmp_capabilities` handshake, then send the hook's command (for these -scenarios: `{"execute": "system_powerdown"}`). The socket is additive — no -existing case may notice it. Note e3fe3f3 recently reworked how the harness -boots; adapt to its current shape rather than the pre-rework description. - -**M21.1 SCI details:** PM1_STS is at the event block base (write-1-to-clear); -PM1_EN at base + block_len/2; PWRBTN bit is 8 in both. If PM1b exists, mirror -reads/writes to both blocks. Enable ACPI mode only when SCI_EN (PM1 control -bit 0) is clear — OVMF boots may already have it set. The publish path reuses -the manager's subscriber table pattern (bounded, drop-on-failed-send). - -**M21.2 GPE walk:** GPE0_STS bytes live at the GPE0 block base, GPE0_EN in -the block's upper half; for a set+enabled bit n, the handler method is -`_L%02X` (level) or `_E%02X` (edge) under `\_GPE`. Evaluate, drain the notify -queue, clear the status bit, ack. A missing handler method is clear-and-log, -not an error. - -**M21.3 ordering:** init subscribes with retries — the acpi service registers -`.power` well after init starts. The stop sequence runs vfs last (other -services may flush through it). The S5 write mirrors `power.zig`'s -`sleepValue` (SLP_TYP bits [12:10], SLP_EN bit 13); if the write returns, log -`power: S5 write did not take` so the scenario fails loudly instead of -hanging. - -**Explicitly out of scope:** the embedded controller and `_Qxx` queries, -battery `_BST`/`_BIF` evaluation beyond the interface stubs, lid/AC on QEMU -(no emulation), reboot over the power protocol, S3 sleep, per-device D-states -(a future lifecycle-vocabulary extension), thermal zones. diff --git a/docs/power.md b/docs/power.md new file mode 100644 index 0000000..24c6929 --- /dev/null +++ b/docs/power.md @@ -0,0 +1,128 @@ +# The power service: events and shutdown + +A laptop lid closes, a battery drains, someone presses the power button — and +several parts of the system might care: a session manager dims the screen, a +logger notes it, and ultimately *something* has to turn the machine off. None of +them owns the hardware that reported the event, and the reporter should not know +who is listening. So system power is a **service**: an event source **publishes** +button/lid/battery/AC events, interested processes **subscribe**, and one +privileged caller — init — can ask it to power the machine off. It is the same +publish/subscribe shape as the [input service](input.md), applied to power. + +## Why a service, and why it is named for the domain, not the firmware + +Where the events come from is firmware-specific — on x86 they ride the ACPI SCI +([acpi.md](acpi.md)); on a Raspberry Pi they would come from PSCI or a mailbox. +What subscribers want is not: *the lid closed* means the same thing regardless of +who noticed. So the surface is **domain-named**. There is a `power-protocol` +module and a well-known `ServiceId.power = 5`; on x86 the **acpi service** +registers it, and on ARM a PSCI/mailbox service will register the *same* id. +Subscribers call `runtime.ipc.lookup(.power)` and never learn which firmware they +are on — the neutrality the whole [discovery](discovery.md) migration exists to +preserve, carried one layer up into a running-system surface. + +This is why the protocol is `power`, not "ACPI events": naming a cross-firmware +surface after one firmware would leak x86 into code the ARM port must reuse +unchanged. + +## The protocol + +The `power-protocol` module ([system/services/power/protocol.zig](../system/services/power/protocol.zig)) +follows the vfs-protocol pattern — extern-struct messages, a version, reserved +fields. Three operations: + +| Direction | Operation | Purpose | +|---|---|---| +| subscriber → service | `subscribe` | receive published events; the subscriber's endpoint rides as the call's **capability** (the input/device-manager pattern) | +| init → service | `shutdown` | orderly shutdown's last step: enter S5 (soft off) | +| service → subscriber | `event` | a published `EventMessage`, delivered as a buffered message (never sent *to* the service) | + +Events are published, not polled: like the input service, the service holds +subscriber endpoints as capabilities and `ipc_send`s each event as a buffered +message, so a slow or dead subscriber can never wedge the source. The event +vocabulary is hardware-neutral: + +- `power_button` — the button was pressed (a fixed ACPI event on x86). +- `lid`, `ac`, `battery` — the named GPE-driven events. +- `notify` — a device notification that maps to none of the above; its `code` + (the ACPI `Notify` argument) and the notifying device's `hid` say which device + and what happened. + +An `EventMessage` carries the `event` tag plus `code` and an 8-byte `hid`, so a +generic `notify` is fully described without a second round trip. + +**`shutdown` is authority, not information.** It is the only operation that +*does* something irreversible, so it is gated: the contract is that only init +(PID 1) may request it, because init is the process that has already run the stop +sequence over everything else. The acpi service implements this as a **soft +gate** — it honors `shutdown` only from a process that is a *subscriber*, and +init is the one subscriber. That stands in for "only the system supervisor may +power off" without hard-coding a pid, so it still holds under tests where PID 1 +is not init. + +## Orderly shutdown + +Powering off cleanly is where the power service, the [process +lifecycle](process-lifecycle.md), and [ACPI events](acpi.md) compose. init +already supervises the services it starts; for shutdown it runs **one event loop +over one endpoint** that carries three things at once: its children's exit +notifications, the lifecycle **signals** it can receive (`terminate`), and the +**power events** it subscribes to — plus a re-arming heartbeat timer proving PID +1 is alive. (init subscribes with retries, because the power service registers +`.power` well after init starts; a missing power service is not fatal — a +`terminate` signal drives the same path.) + +On a `power_button` event or a `terminate` signal, init: + +1. logs that it is shutting down, +2. runs the standard stop sequence — `runtime.process.stop(child, deadline, + endpoint)` — over its children **in reverse spawn order**, so the VFS stops + last (other services may flush through it), each child getting the + *terminate → deadline → kill* escalation from + [process-lifecycle.md](process-lifecycle.md), and +3. requests `.power` `shutdown`. + +The service then enters **S5** (soft off) by writing `SLP_TYP | SLP_EN` to the +PM1 control register(s) from ring 3, mirroring the kernel's own +`system/devices/power.zig` `sleepValue`. If the write returns instead of powering +the machine off, it logs loudly so a test fails rather than hangs. + +**No new system call was needed for S5.** The broad io_port grant on the +`acpi-tables` node ([discovery.md](discovery.md)) already put the PM1 control +ports in the acpi service's hands, so writing S5 from ring 3 is something it +could physically already do; formalizing it as a protocol operation added a +contract, not authority. The kernel keeps `power.zig` for its own test paths and +panic-time poweroff, where no user space is available to ask. + +## Verifying it + +Two QEMU scenarios exercise the path, both injecting a real ACPI power-button +press via QMP `system_powerdown` (there is no other deterministic power event on +this config): + +- `power-button` proves the source: the acpi service's SCI handler logs the + press and publishes `power_button` (the ACPI half is in [acpi.md](acpi.md)). +- `orderly-shutdown` proves the whole composition: button → init logs shutting + down → children stopped → the service enters S5 → QEMU exits. The ordered + regex is the proof, and QEMU's self-exit through S5 is the pass. + +## Scope + +Interface-complete but validated on real hardware (the author's laptop) later, +because QEMU does not emulate them: battery `_BST`/`_BIF` evaluation beyond the +interface stubs, lid and AC events, and the embedded controller's `_Qxx` +queries. Deliberately out of scope for now: reboot over the power protocol, S3 +sleep, per-device D-states (a future lifecycle-vocabulary extension, since +"suspend" has the shape of a signal every driver must answer and has no consumer +until laptop sleep), and thermal zones. + +## See also + +- [acpi.md](acpi.md) — where the events come from on x86: the SCI, the power + button fixed event, and GPE/Notify dispatch in the acpi service. +- [discovery.md](discovery.md) — why the surface is domain-named, and the + firmware neutrality that makes a PSCI backend drop-in on ARM. +- [process-lifecycle.md](process-lifecycle.md) — the stop sequence + (`terminate → deadline → kill`) and signals init composes into shutdown. +- [device-manager.md](device-manager.md) — the supervision model init mirrors for + its own children. diff --git a/docs/resilience.md b/docs/resilience.md index 6e30cc8..2b05e41 100644 --- a/docs/resilience.md +++ b/docs/resilience.md @@ -11,7 +11,7 @@ because the kernel releases a dead process's claims. The `driver-restart` and `usb-report` scenarios prove kill → release → respawn → re-claim → re-report end to end. What remains of this document's ladder is scope, not mechanism: more of the system moved into restartable processes (the discovery migration, -[m19-m20-plan.md](m19-m20-plan.md), is the next rung). This is the property danos is really chasing: +[discovery.md](discovery.md), is the next rung). This is the property danos is really chasing: **if a part of the OS breaks, isolate it, and re-initialise it — without rebooting.** A crashed driver gets restarted; a wedged service gets killed and brought back. It's the reason the [microkernel](vision.md) shape was chosen, and it's a *separate* goal diff --git a/docs/timers.md b/docs/timers.md new file mode 100644 index 0000000..0f8a0f8 --- /dev/null +++ b/docs/timers.md @@ -0,0 +1,117 @@ +# Timers and time + +Two different needs hide under the word "timer", and danos keeps them apart: + +- **Reading the clock** — *what time is it?* A read of a free-running counter. +- **Waiting** — *wake me in N milliseconds*, or *notify me when a deadline passes.* + +Both are answered by the **kernel**, because the kernel already owns a timer: it has +to, to preempt tasks. The LAPIC heartbeat and the calibrated TSC that back all of this +are built in [device-interrupts.md](device-interrupts.md); the scheduler's blocking and +wait queues are in [scheduling.md](scheduling.md). This page is about the surface a +ring-3 program actually uses, and one deliberate absence: **there is no user-space time +service.** + +## Why time is a syscall, not a service + +The tempting microkernel move is to put a timer *driver* in user space and have +applications ask it for the time over IPC. For a **monotonic clock that is wrong** — +reading `now()` should never cost an IPC round trip. The kernel is already holding the +answer: it computes the current time every time it schedules, from the TSC, in a couple +of instructions. Surfacing that as a system call is pure mechanism; routing it through a +message to another process would be slower *and* redundant, and a device like the HPET +(uncacheable MMIO reads) is a particularly bad thing to read on every `now()`. + +This is the same conclusion every serious system reaches: Linux and Zircon read the +counter in the vDSO, L4 exposes a clock field in a shared kernel page, seL4 reads the +cycle counter directly. None of them make a clock read an IPC. danos makes it a syscall. + +That "from the TSC" hides a portability question, because the TSC is only a valid clock +when the CPU guarantees it is *invariant* and when every core's TSC is *synchronized*. +danos checks both — the invariant-TSC CPUID bit (`0x80000007` EDX[8], set on Intel and +AMD), and a cross-core "warp" check as the cores come up — and falls back to the HPET +counter when either fails. So `now()` stays accurate on a real Intel box, a real AMD box, +and inside a VM alike; only the source behind it differs. The mechanism is in +[device-interrupts.md](device-interrupts.md). + +So the timer hardware lives in the kernel, and there is **no `hpet` driver and no time +server** to consume. (An earlier HPET driver existed only to *demonstrate* the driver +model; that role now lives in [drivers.md](drivers.md), as documentation.) The one place +a user-space time service *is* justified — **wall-clock / calendar time** — is discussed +at the end; it is deliberately not built yet. + +## The three system calls + +Time and waiting are three entries in the small syscall table ([syscall.md](syscall.md)): + +- **`clock` (#23)** → monotonic nanoseconds since boot. It only moves forward. Not + wall-clock: no date, no timezone. Backed by `architecture.nanos()` (TSC, scaled with a + 128-bit intermediate so a long uptime can't overflow) — a few nanoseconds of + resolution, and just an `rdtsc` plus a multiply. +- **`sleep` (#3)** → block the caller for N milliseconds. The scheduler records a wake + deadline and the tick sweep wakes it (`scheduler.sleep`). +- **`timer_bind` (#31)** → arm a one-shot timer that, after N milliseconds, posts a + **timer notification** to an IPC endpoint. Unlike `sleep` it does **not** block: a + service can keep answering messages on the same endpoint while a deadline is pending. + This is the timed wait that stop-sequence escalation, hello deadlines, and restart + backoff are built from ([process-lifecycle.md](process-lifecycle.md), + [device-manager.md](device-manager.md)). + +The kernel's own scheduling timer (the LAPIC, vector 32) is never exposed to user space; +programs read the TSC through `clock` and get timed wakeups through `sleep`/`timer_bind`, +both riding the scheduler tick. + +## `runtime.time` — the generic interface + +Applications don't call the syscalls directly; they use `runtime.time` +(`library/runtime/time.zig`), a thin `Instant`/`Duration` layer over them — an ergonomic +front door, not new mechanism. + +```zig +const time = @import("runtime").time; + +const start = time.now(); // Instant — monotonic +doWork(); +const took = start.elapsed(); // Duration +time.sleep(time.Duration.fromMillis(5)); // block ~5 ms + +// A deadline delivered as a notification, so a service keeps serving meanwhile: +_ = time.after(endpoint, time.Duration.fromMillis(200)); +``` + +- `Duration` is nanoseconds under the hood, with `fromNanos/fromMicros/fromMillis/ + fromSeconds` and `asNanos/asMillis`. `ceilMillis` rounds *up* to the kernel's + millisecond granularity, so a sub-millisecond `sleep` never rounds down to zero and + returns early. All arithmetic saturates rather than wraps. +- `Instant` is a point on the monotonic clock: `since`, `elapsed`, `plus`, `reached` — + built for deadline loops (`while (!deadline.reached()) …`). +- `now()` / `monotonicNanos()` wrap `clock`. `available()` reports whether the clock is + calibrated at all (the kernel returns 0 until the TSC frequency is known, so a caller + that needs real time can treat 0 as "unavailable" rather than assume it advances). +- `sleep(d)` wraps `sleep`; `spin(d)` busy-polls `now()` for the sub-millisecond delays + the millisecond tick can't express; `after(endpoint, d)` wraps `timer_bind`. + +The raw wrappers (`system.clock`, `system.sleep`, `system.timerOnce`) stay in +`library/runtime/system.zig`; `runtime.time` is the layer meant for everyday use. + +## Wall-clock time (not built) + +Everything above is **monotonic**: elapsed time since boot, perfect for timeouts and +measurement, useless for "what is the date?" Calendar time — a real-time clock, time +zones, leap seconds — is genuinely a **user-space** concern, and it *is* the case a time +service is for. It would be backed by an **RTC** driver (the CMOS real-time clock), not +the HPET, and exposed as a `CLOCK_REALTIME`-style service alongside the monotonic +syscall. It is deferred until something needs it; the monotonic clock the kernel already +owns covers every current use. + +## Verifying it + +`runtime.time`'s `Instant`/`Duration` arithmetic has unit tests that run on the host: + +``` +$ zig build test # includes library/runtime/time.zig +``` + +End to end, the proof the clock is real is that it *advances*: read `now()`, `sleep` a +`Duration`, read `now()` again, and the second reading is later — the kernel's timer +driving a ring-3 program with no service in between. diff --git a/library/runtime/runtime.zig b/library/runtime/runtime.zig index efdc5a5..9cca735 100644 --- a/library/runtime/runtime.zig +++ b/library/runtime/runtime.zig @@ -12,6 +12,9 @@ //! (arguments arrive via `init`). pub const system = @import("system.zig"); +/// Monotonic time, delays, and deadlines over the kernel clock/sleep/timer syscalls +/// — an `Instant`/`Duration` front door, no time service (docs/timers.md). +pub const time = @import("time.zig"); pub const heap = @import("heap.zig"); pub const ipc = @import("ipc.zig"); pub const start = @import("start.zig"); @@ -21,7 +24,7 @@ pub const vfs_protocol = @import("vfs-protocol"); /// The device-manager protocol: hello + tree reports (docs/device-manager.md). pub const device_manager_protocol = @import("device-manager-protocol"); -/// The power protocol: events (button, lid, battery) + shutdown (docs/m21-plan.md). +/// The power protocol: events (button, lid, battery) + shutdown (docs/power.md). pub const power_protocol = @import("power-protocol"); /// Keyboard-event listening (subscribe/next) and broadcasting (publish), over the input /// service. See library/runtime/input.zig and system/services/input/. diff --git a/library/runtime/time.zig b/library/runtime/time.zig new file mode 100644 index 0000000..9ae4908 --- /dev/null +++ b/library/runtime/time.zig @@ -0,0 +1,169 @@ +//! The danos time interface — monotonic time, delays, and deadlines for user space. +//! +//! There is no time *service*: the kernel already owns the scheduling timer and +//! surfaces it directly, so reading the clock is one system call (an `rdtsc` and a +//! scale), never an IPC round trip (docs/timers.md explains why). This module is a +//! thin, generic layer over the `clock`/`sleep`/`timer_bind` wrappers in `system.zig` +//! — an ergonomic `Instant`/`Duration` front door, not new mechanism. +//! +//! It is **monotonic** time only: nanoseconds since boot, moving forward, no date or +//! timezone. Wall-clock/calendar time is a separate user-space service (an RTC-backed +//! CLOCK_REALTIME) layered on top later. + +const std = @import("std"); +const system = @import("system.zig"); + +const nanos_per_micro: u64 = 1_000; +const nanos_per_milli: u64 = 1_000_000; +const nanos_per_second: u64 = 1_000_000_000; + +/// A span of time, held as nanoseconds. Constructors name their unit; accessors +/// truncate toward zero. `ceilMillis` rounds *up*, since `sleep`/`after` land on the +/// kernel's millisecond granularity and rounding down could return early. +pub const Duration = struct { + ns: u64, + + pub fn fromNanos(n: u64) Duration { + return .{ .ns = n }; + } + pub fn fromMicros(n: u64) Duration { + return .{ .ns = n *| nanos_per_micro }; + } + pub fn fromMillis(n: u64) Duration { + return .{ .ns = n *| nanos_per_milli }; + } + pub fn fromSeconds(n: u64) Duration { + return .{ .ns = n *| nanos_per_second }; + } + + pub fn asNanos(d: Duration) u64 { + return d.ns; + } + pub fn asMicros(d: Duration) u64 { + return d.ns / nanos_per_micro; + } + pub fn asMillis(d: Duration) u64 { + return d.ns / nanos_per_milli; + } + pub fn asSeconds(d: Duration) u64 { + return d.ns / nanos_per_second; + } + + /// Whole milliseconds, rounded up — the argument `sleep`/`after` pass the kernel. + /// A non-zero sub-millisecond duration becomes 1 ms rather than 0. + pub fn ceilMillis(d: Duration) u64 { + return (d.ns +| (nanos_per_milli - 1)) / nanos_per_milli; + } + + pub fn plus(a: Duration, b: Duration) Duration { + return .{ .ns = a.ns +| b.ns }; + } +}; + +/// A point on the monotonic clock — nanoseconds since boot. Compare and subtract +/// instants to measure elapsed time; it never runs backward, so `since` is safe to +/// saturate at zero rather than wrap. +pub const Instant = struct { + ns: u64, + + /// The span from `earlier` to `self`, saturating at zero if `earlier` is later + /// (which the monotonic clock should never produce, but callers may pass any pair). + pub fn since(self: Instant, earlier: Instant) Duration { + return .{ .ns = self.ns -| earlier.ns }; + } + + /// How long since this instant, sampled now. + pub fn elapsed(self: Instant) Duration { + return now().since(self); + } + + /// This instant advanced by `d` (a deadline, `d` from here). + pub fn plus(self: Instant, d: Duration) Instant { + return .{ .ns = self.ns +| d.ns }; + } + + /// Whether the monotonic clock has reached this instant (used as a deadline). + pub fn reached(deadline: Instant) bool { + return now().ns >= deadline.ns; + } +}; + +/// The current monotonic time. +pub fn now() Instant { + return .{ .ns = system.clock() }; +} + +/// Monotonic nanoseconds since boot — the raw `clock()` reading, for callers that +/// want a plain integer instead of an `Instant`. +pub fn monotonicNanos() u64 { + return system.clock(); +} + +/// Whether the monotonic clock is usable. The kernel returns 0 until the TSC is +/// calibrated (`tsc_hz == 0`); a caller that needs real time can treat that as +/// "unavailable" instead of assuming the clock advances. +pub fn available() bool { + return system.clock() != 0; +} + +/// Block the caller for at least `d`, rounded up to the kernel's millisecond +/// granularity. For sub-millisecond precision the scheduler cannot express, use +/// `spin`. +pub fn sleep(d: Duration) void { + system.sleep(d.ceilMillis()); +} + +/// Block the caller for `ms` milliseconds — the coarse, allocation-free form. +pub fn sleepMillis(ms: u64) void { + system.sleep(ms); +} + +/// Busy-wait until `d` has elapsed, polling the monotonic clock. This burns the CPU +/// on purpose, to hit sub-millisecond delays the scheduler's millisecond tick cannot. +/// Prefer `sleep` for anything at or above a millisecond. +pub fn spin(d: Duration) void { + const deadline = now().plus(d); + while (!deadline.reached()) {} +} + +/// Arm a one-shot timer against `endpoint` (a handle from `ipc.createIpcEndpoint`): +/// after `d` the kernel posts a timer notification (`ipc.Received.isTimer`) there. +/// Unlike `sleep`, this does not block — a service can keep serving IPC on the same +/// endpoint while the deadline is pending. Rounds `d` up to milliseconds; returns +/// false if the timer could not be armed. See `system.timerOnce`. +pub fn after(endpoint: usize, d: Duration) bool { + return system.timerOnce(endpoint, d.ceilMillis()); +} + +test "Duration unit conversions round toward zero" { + try std.testing.expectEqual(@as(u64, 1_000_000_000), Duration.fromSeconds(1).asNanos()); + try std.testing.expectEqual(@as(u64, 1_500), Duration.fromNanos(1_500).asNanos()); + try std.testing.expectEqual(@as(u64, 2), Duration.fromMillis(2).asMillis()); + try std.testing.expectEqual(@as(u64, 1), Duration.fromNanos(1_999_999).asMillis()); + try std.testing.expectEqual(@as(u64, 250), Duration.fromMicros(250).asMicros()); +} + +test "ceilMillis rounds up, and never turns a nonzero span into zero" { + try std.testing.expectEqual(@as(u64, 0), Duration.fromNanos(0).ceilMillis()); + try std.testing.expectEqual(@as(u64, 1), Duration.fromNanos(1).ceilMillis()); + try std.testing.expectEqual(@as(u64, 1), Duration.fromMillis(1).ceilMillis()); + try std.testing.expectEqual(@as(u64, 2), Duration.fromNanos(nanos_per_milli + 1).ceilMillis()); + try std.testing.expectEqual(@as(u64, 5), Duration.fromMillis(5).ceilMillis()); +} + +test "Instant arithmetic: since saturates, plus/reached form deadlines" { + const t0 = Instant{ .ns = 1_000 }; + const t1 = Instant{ .ns = 4_000 }; + try std.testing.expectEqual(@as(u64, 3_000), t1.since(t0).asNanos()); + // earlier-than-self can't happen on a monotonic clock; saturate rather than wrap. + try std.testing.expectEqual(@as(u64, 0), t0.since(t1).asNanos()); + const deadline = t0.plus(Duration.fromNanos(2_500)); + try std.testing.expectEqual(@as(u64, 3_500), deadline.ns); +} + +test "saturating arithmetic does not overflow at the u64 ceiling" { + const big = Duration.fromSeconds(std.math.maxInt(u64)); + try std.testing.expectEqual(@as(u64, std.math.maxInt(u64)), big.asNanos()); + const late = Instant{ .ns = std.math.maxInt(u64) }; + try std.testing.expectEqual(@as(u64, std.math.maxInt(u64)), late.plus(Duration.fromSeconds(10)).ns); +} diff --git a/system/abi.zig b/system/abi.zig index 657f4e4..61f37e5 100644 --- a/system/abi.zig +++ b/system/abi.zig @@ -177,7 +177,7 @@ pub const ServiceId = enum(u32) { input = 2, ps2_bus = 3, // the 8042 owner; child device drivers attach here for raw bytes device_manager = 4, // the tree, the matcher, the supervisor (docs/device-manager.md) - power = 5, // system power: events (button, lid, battery) + shutdown (docs/m21-plan.md; domain-named per decision 7 — the acpi service registers it on x86, a PSCI service will on ARM) + power = 5, // system power: events (button, lid, battery) + shutdown (docs/power.md; domain-named per docs/discovery.md — the acpi service registers it on x86, a PSCI service will on ARM) _, }; diff --git a/system/devices/acpi.zig b/system/devices/acpi.zig index c1a4db0..3c1f215 100644 --- a/system/devices/acpi.zig +++ b/system/devices/acpi.zig @@ -159,7 +159,7 @@ pub var dsdt_physical: u64 = 0; /// The FADT itself (physical + length), published on the acpi-tables node so /// the ring-3 acpi service can read the PM1 event and GPE blocks it needs for -/// the event side (docs/m21-plan.md decision 3). Distinguished from the AML +/// the event side (docs/acpi.md — ACPI events). Distinguished from the AML /// blob resources by its intact "FACP" header — the blobs are header-stripped. var fadt_physical: u64 = 0; var fadt_length: u64 = 0; @@ -423,7 +423,7 @@ pub fn discover(rsdp_physical: u64, memory_regions: []const boot_handoff.MemoryR // whatever the FADT alone provided. } - // Publish the acpi-tables node (docs/m19-m20-plan.md M20): the AML blobs as + // Publish the acpi-tables node (docs/discovery.md): the AML blobs as // memory resources for the acpi service to map and parse in ring 3, a broad // io_port grant for the OperationRegion access its interpreter needs, and // the SCI for the events track (M21). Exactly one node, one trusted @@ -448,7 +448,7 @@ fn publishAcpiTablesNode(device_tree: *DeviceTree) !void { // The broad I/O grant: OperationRegions name whatever ports the firmware // chose (EC, PM1, GPE, SMBus); which ports cannot be known before the AML // that names them is parsed, so the grant is the whole space — the honest - // trust boundary of docs/m19-m20-plan.md decision 5. + // trust boundary of docs/discovery.md (the acpi service's one trusted node). _ = node.addResource(.io_port, 0, 1 << 16); // A broad interrupt window: ACPI _CRS names legacy ISA IRQs (the PS/2 lines // 1 and 12, the RTC, …), and the service registers those devices under this @@ -610,7 +610,7 @@ fn parseMcfg(device_tree: *DeviceTree, header: *const SystemDescriptorTableHeade var boot_memory_regions: []const boot_handoff.MemoryRegion = &.{}; /// The bridge's MMIO apertures, derived from the boot memory map's holes -/// (docs/m19-m20-plan.md decision 2): registered PCI functions carry BAR +/// (docs/discovery.md — apertures from the memory map): registered PCI functions carry BAR /// resources, and `device_register` containment demands the bridge own windows /// that cover them. Everything the firmware described is "not hole"; the low /// aperture runs from the end of the described space below 4 GiB up to the diff --git a/system/devices/aml/aml.zig b/system/devices/aml/aml.zig index 8206c07..c595f6a 100644 --- a/system/devices/aml/aml.zig +++ b/system/devices/aml/aml.zig @@ -56,7 +56,7 @@ pub fn parse(allocator: std.mem.Allocator, blocks: []const []const u8) !ParseRes } /// Count the Device objects in a parsed namespace — what the acpi service -/// (docs/m19-m20-plan.md M20) reports, and what the kernel's own parse counts +/// (docs/discovery.md) reports, and what the kernel's own parse counts /// so the two can be checked equal across the ring-3 move. pub fn deviceCount(namespace: *const Namespace) usize { return countKind(namespace.root, .device); diff --git a/system/devices/device-abi.zig b/system/devices/device-abi.zig index 33cf15e..4507e38 100644 --- a/system/devices/device-abi.zig +++ b/system/devices/device-abi.zig @@ -29,7 +29,7 @@ pub const DeviceClass = enum(u32) { /// hardware ID (`_HID`) and, where static, current resource settings (`_CRS`). acpi_device, /// The ACPI tables themselves, published as one node for the user-space acpi - /// service (docs/m19-m20-plan.md M20): memory resources over the AML blobs, + /// service (docs/discovery.md): memory resources over the AML blobs, /// a broad io_port grant for OperationRegion access, and the SCI interrupt. /// The one node whose claimant is trusted to run firmware bytecode. acpi_tables, diff --git a/system/drivers/bus/bus.zig b/system/drivers/bus/bus.zig deleted file mode 100644 index 9203290..0000000 --- a/system/drivers/bus/bus.zig +++ /dev/null @@ -1,211 +0,0 @@ -//! /system/drivers/bus — a user-space **bus driver**, and the smallest honest example of one. -//! -//! A bus driver owns a device that *contains other devices*, enumerates them by some -//! bus-specific protocol, and publishes each one into the kernel's device table so a -//! class driver can claim it. PCI walks configuration space; USB walks hub descriptors. Here -//! the "bus" is the HPET's register block and the "devices" are its comparators, each -//! a 0x20-byte window at 0x100 + 0x20*n that can be driven independently. -//! -//! It's a toy bus, but nothing about the mechanism is: `bus` reads how many children -//! exist from the hardware (GENERAL_CAP bits [12:8]), publishes one `DeviceDescriptor` per -//! child with a sub-window of its own MMIO plus the shared IRQ, and the kernel checks -//! every one of those resources is contained in what `bus` was granted. A comparator -//! driver then claims a child and maps only *its* registers — not the whole block. -//! -//! It also proves the negative: registering a child whose window escapes the parent's -//! is refused. Without that check, `device_register` would be a system_call for mapping -//! arbitrary physical memory. - -const std = @import("std"); -const runtime = @import("runtime"); -const device = runtime.device; - -const register_general_cap = 0x000; - -/// Comparator n's registers: configuration+comparator+FSB route, 0x20 bytes. -fn timerWindow(hpet_base: u64, n: u64) device.ResourceDescriptor { - return .{ - .kind = @intFromEnum(device.ResourceKind.memory), - .start = hpet_base + 0x100 + 0x20 * n, - .len = 0x20, - }; -} - -fn findHpet(buffer: []device.DeviceDescriptor) ?device.DeviceDescriptor { - const total = device.enumerate(buffer); - const n = @min(total, buffer.len); - for (buffer[0..n]) |d| { - if (d.class != @intFromEnum(device.DeviceClass.timer)) continue; - if (d.parent != device.no_parent) continue; // the block, not a comparator child - for (0..d.resource_count) |j| { - if (d.resources[j].kind == @intFromEnum(device.ResourceKind.memory)) return d; - } - } - return null; -} - -/// The parent's MMIO resource, and its IRQ if it has one. -fn resourcesOf(d: device.DeviceDescriptor) struct { mmio: device.ResourceDescriptor, irq: ?device.ResourceDescriptor } { - var mmio: device.ResourceDescriptor = undefined; - var irq: ?device.ResourceDescriptor = null; - for (0..d.resource_count) |j| { - const r = d.resources[j]; - if (r.kind == @intFromEnum(device.ResourceKind.memory)) mmio = r; - if (r.kind == @intFromEnum(device.ResourceKind.irq)) irq = r; - } - return .{ .mmio = mmio, .irq = irq }; -} - -fn firstChildOf(buffer: []device.DeviceDescriptor, total: usize, parent_id: u64) ?u64 { - for (buffer[0..@min(total, buffer.len)]) |d| { - if (d.parent == parent_id) return d.id; - } - return null; -} - -pub fn main() void { - const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch { - _ = runtime.system.write("bus: out of memory\n"); - return; - }; - - const parent = findHpet(buffer) orelse { - _ = runtime.system.write("bus: no HPET\n"); - return; - }; - const resource = resourcesOf(parent); - - // Claim the bus. Everything below is subdivision of what this claim granted. - // - // Claims are exclusive, and at a normal boot the kernel spawns every initial_ramdisk - // binary — so hpet may own the HPET already. That's not an error, it's the - // capability model working: exit quietly and leave the device to its owner. The - // `bus` test spawns bus alone, so there it wins the claim. - if (!device.claim(parent.id)) { - _ = runtime.system.write("bus: HPET already claimed by another driver, nothing to do\n"); - return; - } - - // Enumerate the bus: ask the hardware how many children it has. - const base = device.mmioMap(parent.id, 0) orelse { - _ = runtime.system.write("bus: mmio_map failed\n"); - return; - }; - const cap: *volatile u64 = @ptrFromInt(base + register_general_cap); - const n_children = ((cap.* >> 8) & 0x1F) + 1; - - // Publish one child per comparator, each owning only its own window. - var published: u64 = 0; - var n: u64 = 0; - while (n < n_children) : (n += 1) { - var child = std.mem.zeroes(device.DeviceDescriptor); - child.class = @intFromEnum(device.DeviceClass.timer); - child.pci_class = device.no_pci_class; - child.hid_len = 6; - child.hid[0..6].* = "hpet-t".*; - child.resource_count = 1; - child.resources[0] = timerWindow(resource.mmio.start, n); - // Comparators share the block's interrupt line; only one child can bind it, - // but all of them may legitimately name it. - if (resource.irq) |i| { - child.resources[child.resource_count] = i; - child.resource_count += 1; - } - - if (device.register(parent.id, &child) == null) { - _ = runtime.system.write("bus: register failed\n"); - return; - } - published += 1; - } - - // The negative case. A window one byte past the end of the parent's must be - // refused — otherwise device_register would be "map any physical page you like". - // Confirm the table did not grow, not merely that the call returned null: null - // also means NoSpace/BadParent, so a size check is what actually proves the - // *containment* rule fired. - const before = device.enumerate(buffer); - var rogue = std.mem.zeroes(device.DeviceDescriptor); - rogue.class = @intFromEnum(device.DeviceClass.unknown); - rogue.resource_count = 1; - rogue.resources[0] = .{ - .kind = @intFromEnum(device.ResourceKind.memory), - .start = resource.mmio.start + resource.mmio.len, - .len = 0x1000, - }; - if (device.register(parent.id, &rogue) != null) { - _ = runtime.system.write("bus: FAIL out-of-window child was accepted\n"); - return; - } - if (device.enumerate(buffer) != before) { - _ = runtime.system.write("bus: FAIL rogue child leaked into the table\n"); - return; - } - - // And confirm the children came back with the right parent and a *narrower* - // window than the bus — read from the table, not from our own memory. - const total = device.enumerate(buffer); - var seen: u64 = 0; - for (buffer[0..@min(total, buffer.len)]) |d| { - if (d.parent != parent.id) continue; - const w = d.resources[0]; - if (w.start < resource.mmio.start or w.len >= resource.mmio.len) { - _ = runtime.system.write("bus: FAIL child window is not inside the bus\n"); - return; - } - seen += 1; - } - if (seen != published) { - _ = runtime.system.write("bus: FAIL child count mismatch\n"); - return; - } - - // Delegation, end to end: claim a child and map *it*. A real class driver would be - // a different process; here bus plays both parts, which exercises the same path. - // The child's window is 0x20 bytes at parent+0x100, so the register it sees at - // offset 0 must be the same timer-0 configuration register the bus sees at 0x100. - // - // (mmio_map rounds to a page, so the child's mapping physically covers the whole - // 4 KiB the HPET lives in — the granularity limit documented in docs/drivers.md. - // The *resource* is narrow even though the page isn't.) - const child_id = firstChildOf(buffer, device.enumerate(buffer), parent.id) orelse { - _ = runtime.system.write("bus: FAIL no child to claim\n"); - return; - }; - if (!device.claim(child_id)) { - _ = runtime.system.write("bus: FAIL could not claim own child\n"); - return; - } - const child_base = device.mmioMap(child_id, 0) orelse { - _ = runtime.system.write("bus: FAIL child mmio_map refused\n"); - return; - }; - const via_child: *volatile u64 = @ptrFromInt(child_base); - const via_bus: *volatile u64 = @ptrFromInt(base + 0x100); - if (via_child.* != via_bus.*) { - _ = runtime.system.write("bus: FAIL child window does not alias the bus register\n"); - return; - } - - // A descriptor pointer into an unmapped page must fail the call, not fault the - // kernel. Grab a page, free it, and register through the stale address: if the - // kernel dereferenced it raw (rather than copying in through the page tables) this - // would triple-fault QEMU and the test would time out instead of printing ok. - const scratch = runtime.system.mmap(0x1000, runtime.system.PROT_READ | runtime.system.PROT_WRITE); - if (!runtime.system.mmapFailed(scratch)) { - _ = runtime.system.munmap(scratch, 0x1000); - const descriptor: *const device.DeviceDescriptor = @ptrFromInt(scratch); - if (device.register(parent.id, descriptor) != null) { - _ = runtime.system.write("bus: FAIL register accepted an unmapped descriptor\n"); - return; - } - } - - _ = runtime.system.write("bus: ok\n"); - while (true) runtime.system.sleep(1000); -} - -pub const panic = runtime.panic; -comptime { - _ = &runtime.start._start; -} diff --git a/system/drivers/hpet/hpet.zig b/system/drivers/hpet/hpet.zig deleted file mode 100644 index 3b466a4..0000000 --- a/system/drivers/hpet/hpet.zig +++ /dev/null @@ -1,195 +0,0 @@ -//! /system/drivers/hpet — a user-space HPET driver. It proves the whole driver model end to -//! end: enumerate the device table, find the HPET, claim it, map its registers into -//! this ring-3 address space (strong-uncacheable), **bind its interrupt to an IPC -//! endpoint**, then sit blocked in `replyWait` until the hardware wakes it. -//! -//! Nothing here polls. Between interrupts the process is `.blocked` and off every -//! scheduler queue; the core runs other work or idles. That is the point of the -//! exercise — a driver is a process that sleeps until its device has something to -//! say (see docs/drivers.md). -//! -//! The comparator is configured **level-triggered** on purpose. Edge would be -//! simpler, but level is the discipline every real device line needs, and it forces -//! the full cycle to be correct: -//! -//! kernel ISR mask the GSI -> EOI -> notify this endpoint -//! hpet wake, clear GENERAL_INT_STATUS (deasserts the line), re-arm -//! hpet irq_ack -> kernel unmasks the GSI -//! -//! Clear the status bit *before* acking, or the line is still asserted when the -//! kernel unmasks and the I/O APIC redelivers forever. -//! -//! Register map (HPET spec 1.0a): -//! 0x000 GENERAL_CAP [63:32] fs per tick, [12:8] number timers - 1 -//! 0x010 GENERAL_CONFIGURATION bit0 ENABLE_CNF, bit1 LEG_RT_CNF -//! 0x020 GENERAL_INT_STATUS bit n = timer n asserted (write 1 to clear) -//! 0x0F0 MAIN_COUNTER -//! 0x100 TIMER0_CONFIGURATION bit1 INT_TYPE(1=level) bit2 INT_ENB bit3 TYPE(periodic) -//! bits[13:9] INT_ROUTE, [63:32] INT_ROUTE_CAP -//! 0x108 TIMER0_COMPARATOR - -const runtime = @import("runtime"); -const mmio = @import("mmio"); -const device = runtime.device; -const ipc = runtime.ipc; - -const register_general_cap = 0x000; -const register_general_configuration = 0x010; -const register_int_status = 0x020; -const register_main_counter = 0x0F0; -const register_timer0_configuration = 0x100; -const register_timer0_comparator = 0x108; - -const configuration_enable: u64 = 1 << 0; // GENERAL_CONFIGURATION.ENABLE_CNF -const configuration_leg_rt: u64 = 1 << 1; // GENERAL_CONFIGURATION.LEG_RT_CNF -const tn_int_type_level: u64 = 1 << 1; -const tn_int_enb: u64 = 1 << 2; -const tn_type_periodic: u64 = 1 << 3; -const tn_route_shift = 9; -const tn_route_mask: u64 = 0x1F << tn_route_shift; - -/// Interrupts to observe before declaring victory. -const target_ticks = 5; - -/// Read/write a 64-bit HPET register through the typed volatile MMIO layer (/lib/mmio). -/// The HPET is pure MMIO with no DMA, and on x86 its grant is strong-uncacheable (so -/// UC writes are already ordered) — no barriers are needed here; the point is the -/// typed, arch-portable access every driver should use. -inline fn rd(base: usize, off: usize) u64 { - return mmio.read(u64, base + off); -} -inline fn wr(base: usize, off: usize, value: u64) void { - mmio.write(u64, base + off, value); -} - -/// A timer-class device exposing both an MMIO window and an IRQ: its id, the two -/// resource indices, and the GSI discovery chose out of `Tn_INT_ROUTE_CAP`. -const Found = struct { device_id: u64, mmio: u64, irq: u64, gsi: u64 }; - -fn findHpet(buffer: []device.DeviceDescriptor) ?Found { - const total = device.enumerate(buffer); - const n = @min(total, buffer.len); - for (buffer[0..n]) |d| { - if (d.class != @intFromEnum(device.DeviceClass.timer)) continue; - // Skip comparator children a bus driver may have published below the block - // (see system/drivers/bus/bus.zig) — we want the register block itself. - if (d.parent != device.no_parent) continue; - var mmio_index: ?u64 = null; - var irq: ?u64 = null; - for (0..d.resource_count) |j| { - switch (d.resources[j].kind) { - @intFromEnum(device.ResourceKind.memory) => mmio_index = mmio_index orelse j, - @intFromEnum(device.ResourceKind.irq) => irq = irq orelse j, - else => {}, - } - } - if (mmio_index) |m| if (irq) |i| { - return .{ .device_id = d.id, .mmio = m, .irq = i, .gsi = d.resources[i].start }; - }; - } - return null; -} - -pub fn main() void { - // Enumerate into a heap buffer (too big for the one-page user stack). - const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 32) catch { - _ = runtime.system.write("system/drivers/hpet: out of memory\n"); - return; - }; - - const hpet = findHpet(buffer) orelse { - _ = runtime.system.write("system/drivers/hpet: no HPET with an IRQ\n"); - return; - }; - - if (!device.claim(hpet.device_id)) { - _ = runtime.system.write("system/drivers/hpet: claim failed\n"); - return; - } - const base = device.mmioMap(hpet.device_id, hpet.mmio) orelse { - _ = runtime.system.write("system/drivers/hpet: mmio_map failed\n"); - return; - }; - - // The GSI discovery picked for us out of Tn_INT_ROUTE_CAP. Program the comparator - // to raise exactly this line — the kernel will only bind the one it recorded. - const gsi = hpet.gsi; - - const endpoint = ipc.createIpcEndpoint() orelse { - _ = runtime.system.write("system/drivers/hpet: create_ipc_endpoint failed\n"); - return; - }; - - // --- program the hardware ------------------------------------------------ - // Counter period, so we can arm the comparator a fixed wall-clock distance out. - const femtos_per_tick = rd(base, register_general_cap) >> 32; - if (femtos_per_tick == 0) { - _ = runtime.system.write("system/drivers/hpet: bad HPET period\n"); - return; - } - const ticks_per_ms = 1_000_000_000_000 / femtos_per_tick; - - // Stop the counter and take the legacy route off while we reconfigure. - wr(base, register_general_configuration, rd(base, register_general_configuration) & ~(configuration_enable | configuration_leg_rt)); - - // Timer 0: one-shot, level-triggered, routed to our GSI, interrupt enabled. - // One-shot (not periodic) sidesteps the HPET's Tn_value_SET accumulator quirk — - // we simply re-arm from the driver on each interrupt, which is what a tickless - // timer driver does anyway. - var t0 = rd(base, register_timer0_configuration); - t0 &= ~(tn_route_mask | tn_type_periodic); - t0 |= tn_int_type_level | tn_int_enb | (gsi << tn_route_shift); - wr(base, register_timer0_configuration, t0); - - // Clear any stale assertion, then arm ~100 ms out and start the counter. - wr(base, register_int_status, 1); - wr(base, register_timer0_comparator, rd(base, register_main_counter) + ticks_per_ms * 100); - wr(base, register_general_configuration, rd(base, register_general_configuration) | configuration_enable); - - if (!device.irqBind(hpet.device_id, hpet.irq, endpoint)) { - _ = runtime.system.write("system/drivers/hpet: irq_bind failed\n"); - return; - } - _ = runtime.system.write("system/drivers/hpet: bound, sleeping until the hardware speaks\n"); - - // --- the driver loop ----------------------------------------------------- - // Blocked in replyWait. No polling, no spinning: the next line of this function - // runs only because an interrupt fired. - var receive: [64]u8 = undefined; - var count: usize = 0; - while (count < target_ticks) { - // Blocked here. The task is `.blocked` and off every scheduler queue; the - // next line runs only because the HPET raised its line. - const r = ipc.replyWait(endpoint, &.{}, &receive, null); - if (!r.isNotification()) continue; // a client request, not our IRQ - - // Quiet the device: write 1 to timer 0's status bit. Until this lands, the - // line is still asserted and unmasking would refire immediately. - wr(base, register_int_status, 1); - count += 1; - - if (count < target_ticks) { - wr(base, register_timer0_comparator, rd(base, register_main_counter) + ticks_per_ms * 100); - } else { - // Last one: stop the source rather than re-arming, so the line is left - // both quiet *and* unmasked by the ack below. Re-arming here would leave - // a pending interrupt that nobody is waiting for, and the ISR would mask - // the line again a moment later. - wr(base, register_timer0_configuration, rd(base, register_timer0_configuration) & ~tn_int_enb); - } - - _ = runtime.system.write("system/drivers/hpet: irq\n"); - if (!device.irqAck(hpet.device_id, hpet.irq)) { - _ = runtime.system.write("system/drivers/hpet: irq_ack failed\n"); - return; - } - } - - _ = runtime.system.write("system/drivers/hpet: ok\n"); - while (true) runtime.system.sleep(1000); -} - -pub const panic = runtime.panic; -comptime { - _ = &runtime.start._start; -} diff --git a/system/drivers/pci-bus/pci-bus.zig b/system/drivers/pci-bus/pci-bus.zig index 5d03e0e..2d6bd6f 100644 --- a/system/drivers/pci-bus/pci-bus.zig +++ b/system/drivers/pci-bus/pci-bus.zig @@ -1,5 +1,5 @@ //! /system/drivers/pci-bus — the PCI bus driver: enumeration moved out of ring 0 -//! (docs/m19-m20-plan.md, M19). The device manager matches the `pci_host_bridge` +//! (docs/discovery.md). The device manager matches the `pci_host_bridge` //! node and spawns one instance per bridge, the bridge's device id as argv[1] — //! the same per-device contract as usb-xhci-bus. //! diff --git a/system/kernel/architecture/x86_64/apic.zig b/system/kernel/architecture/x86_64/apic.zig index e24fc93..73efa17 100644 --- a/system/kernel/architecture/x86_64/apic.zig +++ b/system/kernel/architecture/x86_64/apic.zig @@ -80,6 +80,35 @@ var timer_hz: u32 = 0; var tsc_hz: u64 = 0; var tsc_base: u64 = 0; +/// Whether the TSC is architecturally **invariant** — a constant rate regardless of +/// P/C-state transitions, and thus valid as a clocksource (CPUID leaf 0x80000007, +/// EDX bit 8). AMD and modern Intel set it; the bare qemu64 model does not. Measured +/// frequency alone is not enough: a non-invariant TSC speeds up and slows down with +/// the core clock, so reading it as wall time would drift. +var tsc_invariant: bool = false; +/// Cleared if the cross-core warp check (checkWarpSource) ever sees the TSC read +/// lower on one core than the max another core has already published — i.e. the +/// per-core TSCs are not synchronized, and a task migrating cores could see time go +/// backward. Starts true (assume synchronized until proven otherwise). +var tsc_synced: bool = true; +/// The worst backward skew the warp check observed, in TSC cycles (0 = none). +var tsc_warp_cycles: u64 = 0; + +/// The monotonic clock's source. The TSC when it is invariant *and* synchronized — +/// the fast `rdtsc` path taken on real Intel/AMD and modern VMs. Otherwise the HPET +/// main counter: a single fixed-rate counter, immune to both per-core skew and +/// frequency scaling, so it stays accurate on a bare VM or a warped machine. +const ClockSource = enum { tsc, hpet }; +var clock_source: ClockSource = .tsc; + +/// HPET standby clocksource, set up in calibrate() whenever an HPET exists (whether +/// or not calibration itself measured against it): its frequency, the counter value +/// chosen as the zero point, and its width mask. Only a 64-bit HPET is used as a +/// clocksource — a 32-bit one wraps too fast to be monotonic without accumulation. +var hpet_clock_hz: u64 = 0; +var hpet_clock_base: u64 = 0; +var hpet_clock_mask: u64 = ~@as(u64, 0); + /// Read the 64-bit Time Stamp Counter. fn rdtsc() u64 { var low: u32 = undefined; @@ -220,6 +249,31 @@ pub fn calibrate() void { } tsc_base = rdtsc(); // the clock's zero point (boot) + + // Decide whether the TSC is trustworthy as a clocksource. Frequency (measured + // above, possibly against the HPET/PIT) is necessary but not sufficient: the TSC + // must also be *invariant* (CPUID 0x80000007 EDX[8]). AMD and modern Intel set + // this; the bare qemu64 model does not. + tsc_invariant = tscIsInvariant(); + + // Bring up the HPET as a standby clocksource whenever one exists — even on the + // CPUID-0x15 path where calibration never touched it — so a non-invariant TSC + // (here) or an unsynchronized one (checkWarpSource, during SMP bring-up) can fall + // back to a source that is immune to both. hpetHz() maps + enables the counter + // and is idempotent if calibration already used it. + if (configuration_hpet_base != 0) { + if (hpetHz()) |hz| { + hpet_clock_mask = hpetMask(); + if (hpet_clock_mask == ~@as(u64, 0)) { // only a 64-bit HPET is monotonic enough + hpet_clock_hz = hz; + hpet_clock_base = readHpet(); + } + } + } + + // Select the source: the fast TSC when invariant, else the HPET if we have one. + // (checkWarpSource may still demote TSC -> HPET later if the cores' TSCs skew.) + if (!tsc_invariant and hpet_clock_hz != 0) clock_source = .hpet; } /// Run the LAPIC timer one-shot from its maximum count while a monotonic reference @@ -275,7 +329,7 @@ fn calibratePit() void { // --- reference clocks ------------------------------------------------------ /// TSC frequency from CPUID leaf 0x15 (crystal_hz * numerator / denominator), or -/// null if the CPU doesn't enumerate it (common under QEMU). +/// null if the CPU doesn't enumerate it (common under QEMU, and on AMD). fn cpuidTscHz() ?u64 { if (cpuid(0).eax < 0x15) return null; const r = cpuid(0x15); @@ -283,6 +337,15 @@ fn cpuidTscHz() ?u64 { return @as(u64, r.ecx) * r.ebx / r.eax; } +/// Whether the CPU advertises an **invariant** TSC (CPUID leaf 0x80000007, EDX +/// bit 8) — the architectural guarantee, on both Intel and AMD, that the TSC ticks +/// at a constant rate across P/C-states and never stops. Requires the extended-leaf +/// range to reach 0x80000007 first. +fn tscIsInvariant() bool { + if (cpuid(0x80000000).eax < 0x80000007) return false; + return (cpuid(0x80000007).edx & (1 << 8)) != 0; +} + const CpuidRegs = struct { eax: u32, ebx: u32, ecx: u32, edx: u32 }; fn cpuid(leaf: u32) CpuidRegs { @@ -310,11 +373,20 @@ fn hpetWrite64(off: usize, value: u64) void { @as(*volatile u64, @ptrFromInt(configuration_hpet_base + off)).* = value; } +/// Whether the HPET has been mapped into the physmap yet, so `configuration_hpet_base` +/// already holds the virtual address. `hpetHz` is called more than once (calibration +/// may use the HPET, and the standby-clocksource setup asks for it again), and mapping +/// an already-mapped base a second time would double-offset it into an overflow. +var hpet_mapped: bool = false; + /// Map + enable the HPET and return its tick frequency, or null if unusable. /// Maps the HPET into the physmap and switches configuration_hpet_base to that virtual -/// address, so the register accessors reach it without the identity map. +/// address, so the register accessors reach it without the identity map. Idempotent. fn hpetHz() ?u64 { - configuration_hpet_base = paging.mapMmio(configuration_hpet_base, 0x400, true); + if (!hpet_mapped) { + configuration_hpet_base = paging.mapMmio(configuration_hpet_base, 0x400, true); + hpet_mapped = true; + } const caps = hpetRead64(0x00); const period_fs = caps >> 32; // femtoseconds per tick if (period_fs == 0) return null; @@ -364,24 +436,169 @@ pub fn tscHz() u64 { return tsc_hz; } -// Monotonic high-resolution clock, from the TSC. A function per resolution, each -// scaling the cycle delta directly at its unit (the 128-bit intermediate avoids -// overflow across a long uptime). nanos() resolves to a few ns; millis() is what -// the scheduler uses for sleep deadlines. +// Monotonic high-resolution clock. A function per resolution, each scaling the +// counter delta directly at its unit (the 128-bit intermediate avoids overflow +// across a long uptime). nanos() resolves to a few ns on the TSC; millis() is what +// the scheduler uses for sleep deadlines. The source is the TSC when it is invariant +// and synchronized, else the HPET counter (see clock_source) — the branch is one +// global load and the TSC path is unchanged from before. + +/// The selected source's counter delta since its zero point. +fn clockCount() u64 { + return switch (clock_source) { + .tsc => rdtsc() -% tsc_base, + // A 64-bit HPET (the only kind we select) never wraps in any realistic + // uptime, so the wrapping subtraction is exact. + .hpet => readHpet() -% hpet_clock_base, + }; +} + +/// The selected source's frequency (0 if the clock is unavailable/uncalibrated). +fn clockHertz() u64 { + return switch (clock_source) { + .tsc => tsc_hz, + .hpet => hpet_clock_hz, + }; +} pub fn nanos() u64 { - if (tsc_hz == 0) return 0; - return @intCast(@as(u128, rdtsc() -% tsc_base) * 1_000_000_000 / tsc_hz); + const hz = clockHertz(); + if (hz == 0) return 0; + return @intCast(@as(u128, clockCount()) * 1_000_000_000 / hz); } pub fn micros() u64 { - if (tsc_hz == 0) return 0; - return @intCast(@as(u128, rdtsc() -% tsc_base) * 1_000_000 / tsc_hz); + const hz = clockHertz(); + if (hz == 0) return 0; + return @intCast(@as(u128, clockCount()) * 1_000_000 / hz); } pub fn millis() u64 { - if (tsc_hz == 0) return 0; - return @intCast(@as(u128, rdtsc() -% tsc_base) * 1_000 / tsc_hz); + const hz = clockHertz(); + if (hz == 0) return 0; + return @intCast(@as(u128, clockCount()) * 1_000 / hz); +} + +/// Whether the CPU advertises an invariant TSC (CPUID 0x80000007 EDX[8]). +pub fn tscInvariant() bool { + return tsc_invariant; +} + +/// Test hook: force the TSC clocksource on, as if the CPU had advertised an invariant +/// TSC. QEMU's TCG accelerator (the only one for an x86 guest on an Apple-Silicon +/// host) does not expose the invariant-TSC bit — its emulated TSC isn't invariant — so +/// the tsc-sync test can't reach the real-Intel/AMD/KVM path through CPUID. This lets +/// that test exercise the TSC clocksource and the cross-core warp check anyway. tsc_base +/// is left as-is so the switch from the HPET is continuous. +pub fn forceTscClocksourceForTest() void { + tsc_invariant = true; + clock_source = .tsc; +} + +/// How many per-AP warp checks actually ran (a rendezvous completed) — lets a test +/// confirm the cross-core check executed rather than being skipped. +pub fn warpChecksRun() u32 { + return warp_checks; +} + +/// Whether the per-core TSCs are synchronized (no backward warp seen at bring-up). +pub fn tscSynced() bool { + return tsc_synced; +} + +/// The active monotonic clocksource, for the boot log and tests. +pub fn clockSourceName() []const u8 { + return switch (clock_source) { + .tsc => "tsc", + .hpet => "hpet", + }; +} + +// --- cross-core TSC synchronization ("warp") check ------------------------- +// Two cores hammer a shared "max seen" TSC value under a lock; if either reads a +// value below that max, its TSC lags the other's, and time would run backward for a +// task migrating between them (Linux calls this a warp). danos brings APs up one at a +// time, so this runs pairwise: the BSP (source) against each AP (target) as it comes +// online. It only matters — and only runs — while the TSC is the clocksource; on a +// machine already on the HPET (a bare VM) the whole rendezvous is skipped. + +var warp_lock: u32 = 0; +var warp_last: u64 = 0; +var warp_bsp_ready: u32 = 0; +var warp_ap_ready: u32 = 0; +var warp_stop: u32 = 0; +var warp_checks: u32 = 0; // completed per-AP rendezvous count (for the tsc-sync test) + +const warp_rounds: u32 = 1 << 20; // locked reads on the BSP: ~1 ms at GHz rates +const warp_spin_limit: u64 = 1 << 32; // bound every rendezvous wait so a lost core can't hang boot + +fn warpTick() void { + while (@cmpxchgWeak(u32, &warp_lock, 0, 1, .acquire, .monotonic) != null) asm volatile ("pause"); + const t = rdtsc(); + if (t < warp_last) { + const delta = warp_last - t; + if (delta > tsc_warp_cycles) tsc_warp_cycles = delta; + tsc_synced = false; + } else { + warp_last = t; + } + @atomicStore(u32, &warp_lock, 0, .release); +} + +/// Spin (bounded) until `flag` is nonzero; false on timeout. +fn warpAwait(flag: *u32) bool { + var spins: u64 = 0; + while (@atomicLoad(u32, flag, .acquire) == 0) : (spins += 1) { + if (spins >= warp_spin_limit) return false; + asm volatile ("pause"); + } + return true; +} + +/// BSP side of the pairwise TSC warp check, run once per AP as it reports in. No-op +/// unless the TSC is the active clocksource. If the AP's TSC proves to lag, demote +/// the monotonic clock to the HPET without a discontinuity. +pub fn checkWarpSource() void { + if (clock_source != .tsc) return; + warp_last = 0; + @atomicStore(u32, &warp_stop, 0, .release); + @atomicStore(u32, &warp_ap_ready, 0, .release); + @atomicStore(u32, &warp_bsp_ready, 1, .release); + if (!warpAwait(&warp_ap_ready)) { // AP never joined the rendezvous; skip, don't hang + @atomicStore(u32, &warp_bsp_ready, 0, .release); + return; + } + var i: u32 = 0; + while (i < warp_rounds) : (i += 1) warpTick(); + @atomicStore(u32, &warp_stop, 1, .release); + @atomicStore(u32, &warp_bsp_ready, 0, .release); + warp_checks += 1; + + if (!tsc_synced and hpet_clock_hz != 0) demoteToHpet(); +} + +/// AP side: join the BSP's warp check, then return so the core can enter the +/// scheduler. Bounded so a missing BSP can't strand the core. +pub fn checkWarpTarget() void { + if (clock_source != .tsc) return; + if (!warpAwait(&warp_bsp_ready)) return; + @atomicStore(u32, &warp_ap_ready, 1, .release); + var spins: u64 = 0; + while (@atomicLoad(u32, &warp_stop, .acquire) == 0) : (spins += 1) { + if (spins >= warp_spin_limit) return; + warpTick(); + } +} + +/// Switch the clocksource from the TSC to the HPET without a discontinuity: choose +/// the HPET zero point so it reads the same nanosecond value the TSC does right now, +/// so time neither jumps nor runs backward across the switch. Called when the warp +/// check proves the per-core TSCs unsynchronized. +fn demoteToHpet() void { + const now_ns = @as(u128, rdtsc() -% tsc_base) * 1_000_000_000 / tsc_hz; + const equivalent_ticks: u64 = @intCast(now_ns * hpet_clock_hz / 1_000_000_000); + hpet_clock_base = readHpet() -% equivalent_ticks; + clock_source = .hpet; } /// Acknowledge the current interrupt so the LAPIC will deliver the next one. diff --git a/system/kernel/architecture/x86_64/cpu.zig b/system/kernel/architecture/x86_64/cpu.zig index f2eac82..238cccf 100644 --- a/system/kernel/architecture/x86_64/cpu.zig +++ b/system/kernel/architecture/x86_64/cpu.zig @@ -469,6 +469,35 @@ pub fn clockHz() u64 { return apic.tscHz(); } +/// Whether the CPU guarantees an **invariant** TSC (CPUID 0x80000007 EDX[8] on +/// x86; the analogous architectural guarantee elsewhere). When false the TSC is not +/// used as the clocksource. +pub fn clockInvariant() bool { + return apic.tscInvariant(); +} + +/// Whether the per-core clock counters are synchronized (no backward warp observed +/// at SMP bring-up). When false the clock falls back off the TSC. +pub fn clockSynchronized() bool { + return apic.tscSynced(); +} + +/// The active monotonic clocksource, for the boot log ("tsc" or "hpet" on x86). +pub fn clockSourceName() []const u8 { + return apic.clockSourceName(); +} + +/// Test hook: force the TSC clocksource on, to exercise the TSC + warp-check path on +/// a hypervisor that won't advertise an invariant TSC (see apic.forceTscClocksourceForTest). +pub fn forceTscClocksourceForTest() void { + apic.forceTscClocksourceForTest(); +} + +/// How many per-AP TSC warp checks completed (for the tsc-sync test). +pub fn warpChecksRun() u32 { + return apic.warpChecksRun(); +} + /// Unmask maskable interrupts (`sti`) so device interrupts get delivered. pub fn enableInterrupts() void { asm volatile ("sti"); diff --git a/system/kernel/architecture/x86_64/smp.zig b/system/kernel/architecture/x86_64/smp.zig index e907516..1837c30 100644 --- a/system/kernel/architecture/x86_64/smp.zig +++ b/system/kernel/architecture/x86_64/smp.zig @@ -148,7 +148,14 @@ pub fn startAp(apic_id: u32, stack_top: usize, percpu: usize, index: usize, cr3: // Wait up to 100 ms for the AP to reach apEntry and set the flag. const deadline = apic.millis() + 100; while (apic.millis() < deadline) { - if (@atomicLoad(u32, &ap_alive, .acquire) != 0) return true; + if (@atomicLoad(u32, &ap_alive, .acquire) != 0) { + // Cross-check this core's TSC against the BSP's before it joins the run + // loop: an unsynchronized TSC must be caught before any task can migrate + // onto this core and observe time going backward. No-op unless the TSC is + // the clocksource (apic.checkWarpSource). + apic.checkWarpSource(); + return true; + } asm volatile ("pause"); } return false; @@ -177,6 +184,11 @@ fn apEntry(percpu: usize) callconv(.c) noreturn { @atomicStore(u32, &ap_alive, 1, .release); // "architecture state up" — BSP is polling this + // Rendezvous with the BSP for the TSC warp check (no-op unless the TSC is the + // clocksource) before joining the run loop, so this core's clock is vetted before + // it can run any task. + apic.checkWarpTarget(); + if (secondary_entry) |enterScheduler| enterScheduler(); // joins the run loop while (true) asm volatile ("hlt"); // (only if no entry was registered) } diff --git a/system/kernel/devices-broker.zig b/system/kernel/devices-broker.zig index 1bd7c75..9a8da4c 100644 --- a/system/kernel/devices-broker.zig +++ b/system/kernel/devices-broker.zig @@ -189,7 +189,7 @@ pub fn register(parent_id: u64, owner: u32, descriptor: *const device_abi.Device if (!ok) return error.NotContained; } - // Idempotent on exact match (docs/m19-m20-plan.md decision 3): a restarted + // Idempotent on exact match (docs/device-manager.md): a restarted // registering bus re-registers what it rediscovers, and the table has no // unregister — an identical (class, identity, resources) child under the // same parent returns the existing id instead of appending a duplicate. diff --git a/system/kernel/kernel.zig b/system/kernel/kernel.zig index 00e4b8b..391908d 100644 --- a/system/kernel/kernel.zig +++ b/system/kernel/kernel.zig @@ -257,10 +257,30 @@ fn kmain(boot_information: *const BootInformation) noreturn { log.checkpoint(cp_timer); log.print("/system/kernel: timer online ({d} Hz tick; timer clock {d} MHz, clock {d} MHz; calibrated via {s})\n", .{ architecture.timer_hz, architecture.timerClockHz() / 1_000_000, architecture.clockHz() / 1_000_000, architecture.timerCalibrationSource() }); + // The tsc-sync test forces the TSC clocksource on before the cores come up, so the + // TSC + warp-check path is exercised even under TCG (which won't advertise an + // invariant TSC). Inert in a normal build (docs/timers.md). + if (build_options.test_case) |tc| { + if (std.mem.eql(u8, tc, "tsc-sync")) architecture.forceTscClocksourceForTest(); + } + // Wake the other cores (application processors). A no-op on a single-core - // machine; on SMP each AP climbs to long mode and reports in (docs/smp.md). + // machine; on SMP each AP climbs to long mode and reports in (docs/smp.md). The + // per-core TSC warp check rides this: each AP is vetted before it joins the run + // loop (docs/timers.md). bringUpSecondaries(); + // Report the monotonic clock's final reliability, now the warp check has run on + // every core. On real Intel/AMD this is the invariant, synchronized TSC; a bare + // VM (no invariant bit) or a machine whose cores' TSCs skew uses the HPET instead. + log.print("/system/kernel: clocksource {s} (TSC invariant: {s}, synchronized: {s})\n", .{ + architecture.clockSourceName(), + if (architecture.clockInvariant()) "yes" else "no", + if (architecture.clockSynchronized()) "yes" else "no", + }); + if (!architecture.clockSynchronized()) + log.write("/system/kernel: WARNING: per-core TSCs are not synchronized; monotonic clock moved off the TSC\n"); + // In a test build (`zig build -Dtest-case=`), run that case and stop. // Normal builds fall through to the idle halt. if (build_options.test_case) |case| { diff --git a/system/kernel/tests.zig b/system/kernel/tests.zig index 2c5aabe..bfca441 100644 --- a/system/kernel/tests.zig +++ b/system/kernel/tests.zig @@ -102,6 +102,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void { stressTest(); } else if (eql(case, "smp-retry")) { smpRetryTest(); + } else if (eql(case, "tsc-sync")) { + tscSyncTest(); } else if (eql(case, "fault-ud")) { faultInvalidOpcode(); } else if (eql(case, "fault-pf")) { @@ -162,14 +164,12 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void { vfsTest(boot_information); } else if (eql(case, "input")) { inputTest(boot_information); - } else if (eql(case, "hpet")) { - hpetTest(boot_information); } else if (eql(case, "iopass")) { ioPassTest(); } else if (eql(case, "irqfree")) { irqFreeTest(); - } else if (eql(case, "bus")) { - busTest(boot_information); + } else if (eql(case, "containment")) { + containmentTest(); } else if (eql(case, "device-manager")) { deviceManagerTest(boot_information); } else if (eql(case, "poweroff")) { @@ -213,7 +213,7 @@ fn eql(a: []const u8, b: []const u8) bool { /// Whether the captured last-write buffer *contains* `needle`. Markers are /// matched as substrings, not prefixes, so a service's source-path debug prefix -/// (`system/drivers/hpet: ok`) still satisfies a marker like `hpet: ok`. +/// (`system/drivers/pci-bus: ...`) still satisfies a marker like `pci-bus: `. fn bufferHas(needle: []const u8) bool { return std.mem.indexOf(u8, process.write_buffer[0..process.write_len], needle) != null; } @@ -1211,6 +1211,35 @@ fn clockTest() void { result(); } +/// The TSC clocksource + cross-core warp check (`-smp 4`). This is the real +/// Intel/AMD / KVM path — an invariant, synchronized TSC. TCG (the only x86 +/// accelerator on an Apple-Silicon host) won't advertise an invariant TSC, so the +/// boot forces the TSC clocksource on (kernel.zig, gated on this case) to exercise +/// the machinery: the kernel must run the per-AP warp check as each core came up, +/// find the cores' TSCs synchronized, and keep the clock on the TSC (no HPET +/// fallback). The default suite (no force) exercises the HPET fallback instead. +fn tscSyncTest() void { + log("DANOS-TEST-BEGIN: tsc-sync\n", .{}); + check("clocksource is the TSC (forced-invariant path)", eql(architecture.clockSourceName(), "tsc")); + check("the cross-core warp check ran on the APs", architecture.warpChecksRun() >= 1); + check("per-core TSCs synchronized (no warp, no HPET fallback)", architecture.clockSynchronized()); + + // A warp that slipped past the bring-up check would surface as a backward reading. + var last = architecture.nanos(); + var monotonic = true; + var advanced = false; + var i: u32 = 0; + while (i < 1_000_000) : (i += 1) { + const t = architecture.nanos(); + if (t < last) monotonic = false; + if (t > last) advanced = true; + last = t; + } + check("monotonic clock advanced", advanced); + check("monotonic clock never ran backward", monotonic); + result(); +} + var proc_worker_run: bool = true; var proc_worker_ran: bool = false; @@ -2258,68 +2287,7 @@ fn spawnNamed(rd: initial_ramdisk.Reader, name: []const u8) bool { return false; } -/// IO passthrough + IRQ-as-IPC: a user-space driver drives real hardware and is -/// *woken by it*. Spawn hpet, which claims the HPET, maps its registers into its -/// own ring-3 address space, arms a level-triggered comparator, binds the interrupt -/// to an IPC endpoint, and then blocks. It prints "hpet: ok" only after being woken -/// `target_ticks` times — it cannot reach that line by polling, because the loop's -/// only exit is through `replyWait` returning a notification badge. -/// -/// The interesting assertion is the last one, which doesn't trust hpet at all: it -/// reads the I/O APIC's redirection entry back and checks the kernel really routed -/// the line (our vector, level-triggered) and really left it unmasked after the -/// driver's final `irq_ack`. hpet disables its comparator on the last interrupt, so -/// that state is quiescent and not a race. -fn hpetTest(boot_information: *const BootInformation) void { - log("DANOS-TEST-BEGIN: hpet\n", .{}); - if (boot_information.initial_ramdisk_len == 0) { - check("bootloader handed over an initial_ramdisk", false); - result(); - return; - } - const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len]; - const rd = initial_ramdisk.Reader.init(image) orelse { - check("initial_ramdisk image is valid", false); - result(); - return; - }; - - process.write_count = 0; - process.write_from_user = false; - check("hpet spawned from the initial_ramdisk", spawnNamed(rd, "hpet")); - - const prefix = "hpet: ok"; - scheduler.setPriority(1); - const deadline = architecture.millis() + 10000; - while (architecture.millis() < deadline) { - if (bufferHas(prefix) and process.write_count >= 2) break; - scheduler.yield(); - } - scheduler.setPriority(4); - - const ok = bufferHas(prefix); - check("user driver mapped HPET MMIO and was woken by its interrupt", ok); - check("driver syscalls came from user mode (CPL 3)", process.write_from_user); - check("kernel routed and re-armed the HPET's line at the I/O APIC", hpetRouteOk()); - result(); -} - -/// Read back the I/O APIC redirection entry for the HPET's GSI and confirm the -/// kernel programmed it: a vector in the device window, level-triggered, unmasked. -/// Independent of anything the driver reported about itself. -fn hpetRouteOk() bool { - const gsi = hpetGsi() orelse return false; - if (gsi >= architecture.irqRouteCount()) return false; - const low = architecture.irqRouteRaw(gsi); // entry index == GSI (this I/O APIC's gsi_base is 0) - const vector: u8 = @truncate(low & 0xFF); - const masked = low & (1 << 16) != 0; - const level = low & (1 << 15) != 0; - return vector >= architecture.irq_vector_base and - vector < architecture.irq_vector_base + architecture.irq_vector_count and - level and !masked; -} - -/// The GSI discovery recorded for the HPET, from the same device table the driver saw. +/// The GSI discovery recorded for the HPET, from the same device table drivers see. fn hpetGsi() ?u32 { var buffer: [16]device_abi.DeviceDescriptor = undefined; const n = @min(devices_broker.enumerate(&buffer), buffer.len); @@ -2334,87 +2302,87 @@ fn hpetGsi() ?u32 { return null; } -/// Bus driver: a user process claims a device that contains other devices, enumerates -/// them from the hardware, and publishes each as a child via `device_register` — the -/// primitive a PCI bridge or USB hub driver is built from. -/// -/// `bus` treats the HPET's register block as a bus and its comparators as children, -/// giving each a 0x20 sub-window. It checks its own work (children come back from the -/// table with the right parent and a strictly narrower window) and, importantly, that -/// the kernel **refuses** a child whose window escapes the parent's — without that, -/// `device_register` would be a system_call for mapping arbitrary physical memory. It prints -/// "bus: ok" only if all of that holds. -/// -/// The kernel-side check here is the one bus can't make: that the children really did -/// land in the device table with the containment invariant intact. -fn busTest(boot_information: *const BootInformation) void { - log("DANOS-TEST-BEGIN: bus\n", .{}); +/// `device_register` containment — a kernel security property, tested directly against +/// the broker (no user-space demo driver). A bus driver publishes children of a device +/// it owns; the kernel must **refuse** any child whose resource escapes the parent's +/// grant, or `device_register` would become a system call for mapping arbitrary physical +/// memory. This is the property the old `bus` demo driver proved end-to-end; with the +/// demo gone, the property is asserted where it lives — in the kernel. Also checks the +/// idempotence rule (M19.0): re-registering an identical child returns the same id +/// instead of appending a duplicate. +fn containmentTest() void { + log("DANOS-TEST-BEGIN: containment\n", .{}); - // M19.0: device_register is idempotent on exact match — a restarted - // registering bus must not duplicate its children. Driven directly against - // the broker: claim an unclaimed node, register the same (class, hid, - // resourceless) child twice, expect one id and one table entry. - { - const me = scheduler.currentId(); - var probe: [1]device_abi.DeviceDescriptor = undefined; - const total = devices_broker.enumerate(&probe); - check("device tree is seeded for the idempotence check", total >= 1); - if (devices_broker.ownerOf(0) == null) { - check("claimed device 0 for the idempotence check", devices_broker.claim(0, me)); - var child = std.mem.zeroes(device_abi.DeviceDescriptor); - child.class = @intFromEnum(device_abi.DeviceClass.unknown); - child.pci_class = device_abi.no_pci_class; - child.hid_len = 4; - child.hid[0..4].* = "idem".*; - const first = devices_broker.register(0, me, &child) catch 0; - check("first register succeeded", first != 0); - const before = devices_broker.enumerate(&probe); - const second = devices_broker.register(0, me, &child) catch 0; - check("re-register returned the same id", second == first); - check("re-register grew nothing", devices_broker.enumerate(&probe) == before); - devices_broker.releaseAllOwnedBy(me); - } else { - check("device 0 unexpectedly claimed before the idempotence check", false); - } - } + const me = scheduler.currentId(); + var buffer: [64]device_abi.DeviceDescriptor = undefined; + check("device tree is seeded", devices_broker.enumerate(&buffer) >= 1); - if (boot_information.initial_ramdisk_len == 0) { - check("bootloader handed over an initial_ramdisk", false); + // The kernel-seeded HPET timer block is a device with a memory resource — a natural + // parent to publish sub-window children under, as a PCI bridge or USB hub would. + const parent_id = hpetDeviceId() orelse { + check("found a device with a memory window to parent children under", false); result(); return; + }; + const parent = buffer[@intCast(parent_id)]; + var window: ?device_abi.ResourceDescriptor = null; + for (0..parent.resource_count) |j| { + if (parent.resources[j].kind == @intFromEnum(device_abi.ResourceKind.memory)) window = parent.resources[j]; } - const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len]; - const rd = initial_ramdisk.Reader.init(image) orelse { - check("initial_ramdisk image is valid", false); + const parent_window = window orelse { + check("parent exposes a memory window", false); result(); return; }; - process.write_count = 0; - process.write_from_user = false; - check("bus spawned from the initial_ramdisk", spawnNamed(rd, "bus")); - - const prefix = "bus: ok"; - scheduler.setPriority(1); - const deadline = architecture.millis() + 10000; - while (architecture.millis() < deadline) { - if (bufferHas(prefix)) break; - scheduler.yield(); + if (devices_broker.ownerOf(parent_id) != null) { + check("parent device was unclaimed at the start of the test", false); + result(); + return; } - scheduler.setPriority(4); + check("claimed the parent device", devices_broker.claim(parent_id, me)); + defer devices_broker.releaseAllOwnedBy(me); + + // A child whose window lies inside the parent's is accepted. + var fits = childDescriptor("cfit", parent_window.start, 0x20); + const before = devices_broker.enumerate(&buffer); + const good = devices_broker.register(parent_id, me, &fits) catch 0; + check("a contained child is registered", good != 0); + check("the contained child was appended to the table", devices_broker.enumerate(&buffer) == before + 1); + + // A child whose window escapes the parent's is refused with NotContained. + var escapes = childDescriptor("cesc", parent_window.start, parent_window.len + 0x1000); + const refused = if (devices_broker.register(parent_id, me, &escapes)) |_| false else |err| err == error.NotContained; + check("an out-of-window child is refused (NotContained)", refused); + check("the refused child left the table unchanged", devices_broker.enumerate(&buffer) == before + 1); + + // Idempotent on exact match: re-registering the accepted child returns its id and + // appends nothing (M19.0 — a restarted bus re-reports what it rediscovers). + const again = devices_broker.register(parent_id, me, &fits) catch 0; + check("re-registering an identical child returns the same id", again != 0 and again == good); + check("re-registering grew nothing", devices_broker.enumerate(&buffer) == before + 1); - const ok = bufferHas(prefix); - check("bus driver published children and the kernel refused an out-of-window one", ok); - check("driver syscalls came from user mode (CPL 3)", process.write_from_user); - check("every registered child is contained in its parent", childrenContained()); result(); } +/// A minimal child descriptor with one memory resource, for the containment test. +fn childDescriptor(hid: []const u8, start: u64, len: u64) device_abi.DeviceDescriptor { + var child = std.mem.zeroes(device_abi.DeviceDescriptor); + child.class = @intFromEnum(device_abi.DeviceClass.unknown); + child.pci_class = device_abi.no_pci_class; + child.hid_len = @intCast(hid.len); + @memcpy(child.hid[0..hid.len], hid); + child.resource_count = 1; + child.resources[0] = .{ .kind = @intFromEnum(device_abi.ResourceKind.memory), .start = start, .len = len }; + return child; +} + /// The device manager (a ring-3 service) enumerates /system/devices, matches each -/// device to a driver, and — eventually — spawns it. This increment only checks the -/// discovery+matching half: it must find the HPET (a timer) and decide `hpet` serves -/// it, printing "device-manager: ok". It uses no special privilege — the same -/// `device_enumerate` any process could call. (Spawning is the next increment.) +/// device to a driver, and spawns it. Proof of the whole discover -> match -> spawn -> +/// driver-up chain: boot only the device-manager; it must discover the PCI host bridge, +/// match `pci-bus`, and spawn it (with the bridge id as its argument) — and the spawned +/// pci-bus must reach its own live marker. It uses no special privilege — the same +/// `device_enumerate` any process could call. fn deviceManagerTest(boot_information: *const BootInformation) void { log("DANOS-TEST-BEGIN: device-manager\n", .{}); if (boot_information.initial_ramdisk_len == 0) { @@ -2430,70 +2398,65 @@ fn deviceManagerTest(boot_information: *const BootInformation) void { }; // Let `system_spawn` find bundled binaries by name (the normal boot path does - // this too). Only the device-manager is spawned here — so if `hpet` runs at all, - // it's because the manager discovered the timer, matched, and spawned it. + // this too). Only the device-manager is spawned here — so if `pci-bus` runs at + // all, it's because the manager discovered the PCI host bridge, matched, and + // spawned it. process.setInitialRamdisk(image); process.write_count = 0; process.write_from_user = false; check("device-manager spawned from the initial_ramdisk", spawnNamed(rd, "device-manager")); - // End-to-end proof: the driver the manager spawned reaches its own live marker. - // `hpet: ok` is hpet's final, stable message (it claims the timer, maps its MMIO, - // binds its IRQ, services one, then sleeps) — nothing overwrites the buffer after, - // so it's race-free to poll for. Its arrival means the whole - // discover -> match -> system_spawn -> driver-up chain worked. - const prefix = "hpet: ok"; + // End-to-end proof, read from kernel state — not the racy last-write serial buffer, + // since many services keep logging after pci-bus. The manager must discover the PCI + // host bridge, match pci-bus, and spawn it, and pci-bus must come up: claim the + // bridge, map its ECAM, and register the functions it enumerates as children in the + // device tree. scheduler.setPriority(1); const deadline = architecture.millis() + 10000; + var spawned = false; while (architecture.millis() < deadline) { - if (bufferHas(prefix)) break; + if (processRunning("pci-bus")) spawned = true; + if (spawned and pciFunctionsRegistered()) break; scheduler.yield(); } scheduler.setPriority(4); - const ok = bufferHas(prefix); - check("device manager matched the timer and system_spawn'd hpet, which came up", ok); + check("device manager discovered the PCI host bridge and spawned pci-bus", spawned); + check("pci-bus came up and registered the functions it enumerated", pciFunctionsRegistered()); check("its syscalls came from user mode (CPL 3)", process.write_from_user); result(); } -/// Every child `bus` registered must have each of its resources inside a parent -/// resource of the same kind — the invariant `device_register` exists to maintain, -/// checked from the kernel's own table rather than the driver's word for it. -/// -/// Only *registered* children are checked, not the whole tree. Firmware topology is -/// trusted and doesn't obey containment: a PCI function's BAR is not inside its host -/// bridge's `bus_range`, because a bus-number range isn't an address window. -fn childrenContained() bool { - var buffer: [64]device_abi.DeviceDescriptor = undefined; - const n = @min(devices_broker.enumerate(&buffer), buffer.len); - - const bus_id = hpetDeviceId() orelse return false; - const p = buffer[@intCast(bus_id)]; - - var children: usize = 0; - for (buffer[0..n]) |d| { - if (d.parent != bus_id) continue; - children += 1; - for (0..d.resource_count) |i| { - const r = d.resources[i]; - var ok = false; - for (0..p.resource_count) |j| { - const pr = p.resources[j]; - if (pr.kind != r.kind) continue; - if (r.kind == @intFromEnum(device_abi.ResourceKind.irq)) { - if (pr.start == r.start) ok = true; - } else if (r.len != 0 and r.start >= pr.start and - r.start + r.len <= pr.start + pr.len) ok = true; - } - if (!ok) return false; - } +/// Whether a live task was spawned under `name` (its argv[0]) — read from the kernel +/// task table, the same snapshot `process_enumerate` exposes. +fn processRunning(name: []const u8) bool { + var table: [64]abi.ProcessDescriptor = undefined; + const total = scheduler.enumerate(&table); + for (table[0..@min(total, table.len)]) |d| { + if (std.mem.eql(u8, d.name[0..d.name_length], name)) return true; } - return children > 0; // bus must have published at least one + return false; } -/// Device id of the HPET (the bus bus claims), from the same table drivers see. +/// Whether pci-bus registered at least one function under the PCI host bridge — proof +/// it came up, claimed the bridge, mapped its ECAM, and walked configuration space. +fn pciFunctionsRegistered() bool { + var buffer: [64]device_abi.DeviceDescriptor = undefined; + const n = @min(devices_broker.enumerate(&buffer), buffer.len); + var bridge_id: ?u64 = null; + for (buffer[0..n]) |d| { + if (d.class == @intFromEnum(device_abi.DeviceClass.pci_host_bridge)) bridge_id = d.id; + } + const bid = bridge_id orelse return false; + for (buffer[0..n]) |d| { + if (d.parent == bid) return true; + } + return false; +} + +/// Device id of the kernel-seeded HPET timer block (the node with a memory resource), +/// from the same device table drivers see. Used as a containment-test parent. fn hpetDeviceId() ?u64 { var buffer: [64]device_abi.DeviceDescriptor = undefined; const n = @min(devices_broker.enumerate(&buffer), buffer.len); @@ -2511,8 +2474,9 @@ fn hpetDeviceId() ?u64 { /// (so a dead driver's device goes quiet instead of storming) and the slot cleared /// (so an ISR never posts a notification into the endpoint that is about to be freed). /// -/// This is the path `hpet` never takes — it runs forever — so it gets its own test. -/// Two properties, both read back from the hardware rather than from our own state: +/// A long-running driver that never exits wouldn't reach this teardown path, so it +/// gets its own test that binds and releases directly. Two properties, both read back +/// from the hardware rather than from our own state: /// /// 1. A bound GSI is routed and unmasked. /// 2. After `releaseOwner` for the binding's owner, that same entry is masked again. diff --git a/system/services/acpi/acpi.zig b/system/services/acpi/acpi.zig index 59481e0..2a308f2 100644 --- a/system/services/acpi/acpi.zig +++ b/system/services/acpi/acpi.zig @@ -1,5 +1,5 @@ //! /system/services/acpi — the ACPI discovery service: the x86 firmware -//! interpreter, moved out of ring 0 (docs/m19-m20-plan.md, M20). Claims the +//! interpreter, moved out of ring 0 (docs/discovery.md). Claims the //! `acpi-tables` node the kernel publishes (the AML blobs, the broad io_port //! grant, a broad irq window, the SCI), and runs the **shared AML module** in //! ring 3 — the same parser and interpreter the kernel uses. @@ -321,7 +321,7 @@ fn onSci() void { /// handler method (`_Lxx` level / `_Exx` edge), drain the Notify queue the /// method produced, and publish an event per notified device. Then clear the /// status bit. QEMU raises no GPEs on this config, so this path is exercised by -/// host unit tests (docs/m21-plan.md decision 5); on real hardware it carries +/// host unit tests (docs/acpi.md — ACPI events); on real hardware it carries /// battery/AC/lid. The embedded controller's `_Qxx` queries are out of scope. fn handleGpe() void { handleGpeBlock(gpe0_blk, gpe0_len, 0); @@ -476,7 +476,7 @@ fn walkDevices(node: *aml.Node, interpreter: *aml.Interpreter) void { if (readHid(c, interpreter)) |hid| { // Skip PCI roots — pci-bus already reports PCI functions; ACPI adds - // only the non-PCI _HID devices (docs/m19-m20-plan.md M20.2). The two + // only the non-PCI _HID devices (docs/device-manager.md — matching). The two // roots are named through the shared registry, not bare _HID strings. const id = acpi_ids.HardwareId.fromHid(hid[0..7]); if (id != .pci_bus and id != .pci_express_root_bridge) { diff --git a/system/services/device-manager/device-manager.zig b/system/services/device-manager/device-manager.zig index cc98aea..dec6380 100644 --- a/system/services/device-manager/device-manager.zig +++ b/system/services/device-manager/device-manager.zig @@ -31,17 +31,6 @@ fn writeLine(comptime fmt: []const u8, arguments: anytype) void { _ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return); } -/// The driver that serves each device — the policy table. In a fuller system -/// this comes from a manifest (docs/device-manager.md: the third bus type -/// triggers it); for now a static map. `null` = no driver for this class yet. -fn driverFor(d: device.DeviceDescriptor) ?[]const u8 { - // The HPET timer node is still kernel-seeded (from the HPET table, not AML). - // PS/2 and other _HID devices now arrive as acpi-service reports and match - // in onChildAdded (M20.3), not from this boot snapshot. - if (d.class == @intFromEnum(device.DeviceClass.timer)) return "hpet"; - return null; -} - /// The PCI class/subclass/prog-IF triple of an xHCI (USB 3) host controller — /// Serial Bus Controller / USB Controller / XHCI — named from pci-class.zig rather /// than written as the bare 0x0C0330 (docs/coding-standards.md, "Named values"). @@ -107,7 +96,7 @@ const Driver = struct { // The assigned device id (becomes argv[1]), or protocol.no_device. device_id: u64 = protocol.no_device, // Whether this driver speaks the protocol (hello expected, deadline - // enforced). Legacy drivers (hpet, ps2-bus) are supervised and restarted + // enforced). Legacy drivers (e.g. ps2-bus) are supervised and restarted // but not yet required to hello. speaks_protocol: bool = false, process_id: u32 = 0, @@ -346,19 +335,15 @@ fn initialise(endpoint: runtime.ipc.Handle) bool { addDriver("pci-bus", descriptor.id, true); continue; } - // PCI functions no longer appear in the boot snapshot (M19.3): the - // pci-bus driver reports them, and onChildAdded matches from reports. - const driver_name = driverFor(descriptor) orelse continue; - matched += 1; - // Skip a singleton that is already alive (the initial-ramdisk sweep test - // starts every bundled binary bare, this manager included) — spawning a - // second instance would only lose the claim race and churn the log. - if (!alreadySupervised(driver_name) and !system.isProcessRunning(driver_name)) { - addDriver(driver_name, protocol.no_device, false); - } + // Nothing else is matched from the boot snapshot today. The kernel-seeded + // HPET timer node is served by the kernel's own clock (docs/timers.md), not + // a user-space driver; PCI functions and PS/2 _HID devices arrive later as + // pci-bus / acpi-service reports and match in onChildAdded (docs/discovery.md). + // A fuller system's static class->driver manifest (docs/device-manager.md) + // would slot in here. } - // The discovery service (docs/m19-m20-plan.md M20): one per firmware, packed + // The discovery service (docs/discovery.md): one per firmware, packed // under the neutral name "discovery", spawned once at startup. It finds and // claims the acpi-tables (or devicetree-blob) node itself. Not a per-device // match — it is the discoverer, not a driver bound to one device. diff --git a/system/services/fdt/fdt.zig b/system/services/fdt/fdt.zig index cba17e9..526c54f 100644 --- a/system/services/fdt/fdt.zig +++ b/system/services/fdt/fdt.zig @@ -1,5 +1,5 @@ //! /system/services/fdt — the devicetree discovery service: the ARM twin of the -//! acpi service (docs/m19-m20-plan.md decision 7). **Placeholder: not +//! acpi service (docs/discovery.md — firmware neutrality). **Placeholder: not //! implemented.** It exists so the build's `-Ddiscovery` option has both of its //! values from day one; the implementation lands with the Raspberry Pi //! bring-up (docs/arm.md). @@ -15,7 +15,7 @@ //! resident under the manager's supervision (hello, restart, the usual //! contract). //! -//! Known prerequisite recorded in the plan: `DeviceDescriptor`'s 8-byte `hid` +//! Known prerequisite recorded in docs/discovery.md: `DeviceDescriptor`'s 8-byte `hid` //! cannot hold an FDT `compatible` string ("brcm,bcm2835-aux-uart") — identity //! widens before this file grows a body. diff --git a/system/services/power/protocol.zig b/system/services/power/protocol.zig index 8d1a720..183f3fe 100644 --- a/system/services/power/protocol.zig +++ b/system/services/power/protocol.zig @@ -1,8 +1,8 @@ -//! The power protocol (docs/m21-plan.md): system power's domain-named surface, +//! The power protocol (docs/power.md): system power's domain-named surface, //! registered under `ServiceId.power`. On x86 the acpi service serves it; on //! ARM a PSCI/mailbox service will register the same id — subscribers never -//! learn which firmware they are on (m19-m20-plan.md decision 7). The -//! vfs-protocol pattern: extern-struct messages, a version, reserved fields. +//! learn which firmware they are on (docs/discovery.md — firmware neutrality). +//! The vfs-protocol pattern: extern-struct messages, a version, reserved fields. /// The protocol version a client states nowhere yet — reserved for the day a /// handshake needs it; requests carry it so a mismatch can be refused loudly. diff --git a/test/qemu_test.py b/test/qemu_test.py index c77cb22..df03195 100644 --- a/test/qemu_test.py +++ b/test/qemu_test.py @@ -177,6 +177,15 @@ CASES = [ "smp": 4, "expect": r"DANOS-TEST-RESULT: PASS", "fail": r"DANOS-TEST-RESULT: FAIL"}, + # TSC clocksource + cross-core warp check (real Intel/AMD / KVM path). TCG won't + # advertise an invariant TSC, so the kernel forces the TSC clocksource on for this + # case (gated in kernel.zig) and runs the per-AP warp check across the 4 cores; + # their TSCs are synchronized, so it stays on the TSC (no HPET fallback). The rest + # of the suite exercises the HPET fallback instead. See docs/timers.md. + {"name": "tsc-sync", + "smp": 4, + "expect": r"DANOS-TEST-RESULT: PASS", + "fail": r"DANOS-TEST-RESULT: FAIL"}, {"name": "fault-ud", "expect": r"invalid opcode \(vector 6\)"}, {"name": "fault-pf", "expect": r"page fault \(vector 14\)"}, {"name": "fault-df", "expect": r"double fault \(vector 8\)"}, @@ -284,7 +293,7 @@ CASES = [ "fail": r"acpi-parse: mismatch|DANOS-TEST-RESULT: FAIL"}, # M20.3: the flip — ps2-bus now comes up from the acpi service's report, not # a kernel-built node. Ordered: report -> spawn -> the driver attaches its - # keyboard, proving discovery runs entirely in ring 3 (docs/m19-m20-plan.md). + # keyboard, proving discovery runs entirely in ring 3 (docs/discovery.md). {"name": "acpi-ps2", "smp": 4, "timeout": 150, @@ -294,7 +303,7 @@ CASES = [ "fail": r"DANOS-TEST-RESULT: FAIL"}, # M21.1: the SCI + power button. Boot the manager (which spawns the acpi # service); ~4s in, QMP system_powerdown raises the ACPI power-button fixed - # event; the service's SCI handler must log the press (docs/m21-plan.md). + # event; the service's SCI handler must log the press (docs/acpi.md). {"name": "power-button", "smp": 4, "timeout": 60, @@ -305,7 +314,7 @@ CASES = [ # ~5s in, QMP system_powerdown raises the power button; the acpi service # publishes it, init stops its children then requests S5, and QEMU exits. # The ordered regex proves button -> shutting-down -> entering-S5; the case - # passes on QEMU's self-exit through S5 (docs/m21-plan.md). + # passes on QEMU's self-exit through S5 (docs/power.md). {"name": "orderly-shutdown", "smp": 4, "timeout": 90, @@ -316,7 +325,7 @@ CASES = [ "fail": r"power: S5 write did not take|DANOS-TEST-RESULT: FAIL"}, # M20.2: the acpi service evaluates _CRS/_STA in ring 3 and registers + # reports its _HID devices — the two PS/2 nodes must appear with resources - # (keyboard: io 0x60/0x64 + IRQ = 3; mouse: IRQ = 1) (docs/m19-m20-plan.md). + # (keyboard: io 0x60/0x64 + IRQ = 3; mouse: IRQ = 1) (docs/discovery.md). {"name": "acpi-report", "smp": 4, "timeout": 150, @@ -378,27 +387,21 @@ CASES = [ {"name": "input", "expect": r"DANOS-TEST-RESULT: PASS", "fail": r"DANOS-TEST-RESULT: FAIL"}, - # IO passthrough + IRQ-as-IPC: a user-space HPET driver maps device MMIO into - # its own address space, binds the device's interrupt to an IPC endpoint, and - # is woken by the hardware five times while blocked (never polling). - {"name": "hpet", - "expect": r"DANOS-TEST-RESULT: PASS", - "fail": r"DANOS-TEST-RESULT: FAIL"}, - # Device manager: a ring-3 service enumerates /system/devices and matches each - # device to a driver (discovery + policy in user space). This increment logs the - # decision; spawning follows. + # Device manager: a ring-3 service enumerates /system/devices, matches the PCI host + # bridge to pci-bus, and spawns it — end-to-end proof of discover -> match -> spawn + # -> driver-up (the spawned pci-bus logs " functions found"). {"name": "device-manager", "expect": r"DANOS-TEST-RESULT: PASS", "fail": r"DANOS-TEST-RESULT: FAIL"}, - # Bus driver: a user process claims a device, enumerates its children from the - # hardware, and publishes each with dev_register — and the kernel refuses a child - # whose window escapes the parent's (else dev_register maps arbitrary memory). - {"name": "bus", + # device_register containment (in-kernel): registering a child whose MMIO window + # escapes its parent's grant is refused (NotContained) — else dev_register would map + # arbitrary physical memory — while an identical re-register stays idempotent. + {"name": "containment", "expect": r"DANOS-TEST-RESULT: PASS", "fail": r"DANOS-TEST-RESULT: FAIL"}, # IRQ teardown: an exiting driver's line is masked and its slot cleared (so no # ISR notifies a freed endpoint), and a sibling owner sharing that endpoint - # keeps its own binding. The path hpet never takes, since it runs forever. + # keeps its own binding. A long-running driver never reaches this teardown path. {"name": "irqfree", "expect": r"DANOS-TEST-RESULT: PASS", "fail": r"DANOS-TEST-RESULT: FAIL"}, @@ -450,7 +453,7 @@ def qmp_send(path, command): """One QMP command: connect, capabilities handshake, execute. Raises on any failure — the caller retries until the guest's socket is ready. This is how a case injects a host-side event (system_powerdown = the ACPI power button) - into the running guest (docs/m21-plan.md).""" + into the running guest (docs/power.md).""" sock = socket.socket(socket.AF_UNIX, socket.SOCK_STREAM) sock.settimeout(5) try: