Compare commits
3
Commits
7dec1b0767
...
452080e997
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
452080e997 | ||
|
|
5b63a841ba | ||
|
|
4b9507bd59 |
@@ -82,6 +82,10 @@ straight into CI.
|
|||||||
Design notes explaining *why* behind the code live in
|
Design notes explaining *why* behind the code live in
|
||||||
[`docs/`](docs/README.md) — start with [`docs/README.md`](docs/README.md).
|
[`docs/`](docs/README.md) — start with [`docs/README.md`](docs/README.md).
|
||||||
|
|
||||||
|
For the hardware needed to run DanOS — minimum specs plus a plain-language guide
|
||||||
|
matching Intel/AMD CPU generations by name — see
|
||||||
|
[`docs/system-requirements.md`](docs/system-requirements.md).
|
||||||
|
|
||||||
## Logo
|
## Logo
|
||||||
|
|
||||||
San Serif Text "Dan OS" with a black karate belt around it.
|
San Serif Text "Dan OS" with a black karate belt around it.
|
||||||
|
|||||||
@@ -132,8 +132,8 @@ pub fn build(b: *std.Build) void {
|
|||||||
// ACPI/PnP hardware-ID (_HID) names — the flat analog of pci-class for acpi_device
|
// ACPI/PnP hardware-ID (_HID) names — the flat analog of pci-class for acpi_device
|
||||||
// nodes. Also shared reference data.
|
// nodes. Also shared reference data.
|
||||||
// The AML interpreter, a build module so the ring-3 acpi service can run the
|
// The AML interpreter, a build module so the ring-3 acpi service can run the
|
||||||
// same parser the kernel does (docs/m19-m20-plan.md decision 1). Pure Zig,
|
// same parser the kernel does (docs/discovery.md — the shared AML module).
|
||||||
// no kernel imports — one source, two builds.
|
// Pure Zig, no kernel imports — one source, two builds.
|
||||||
const aml_module = b.addModule("aml", .{
|
const aml_module = b.addModule("aml", .{
|
||||||
.root_source_file = b.path("system/devices/aml/aml.zig"),
|
.root_source_file = b.path("system/devices/aml/aml.zig"),
|
||||||
});
|
});
|
||||||
@@ -226,7 +226,7 @@ pub fn build(b: *std.Build) void {
|
|||||||
});
|
});
|
||||||
runtime_module.addImport("device-manager-protocol", device_manager_protocol_module);
|
runtime_module.addImport("device-manager-protocol", device_manager_protocol_module);
|
||||||
|
|
||||||
// The power protocol: system power's domain-named surface (docs/m21-plan.md).
|
// The power protocol: system power's domain-named surface (docs/power.md).
|
||||||
const power_protocol_module = b.addModule("power-protocol", .{
|
const power_protocol_module = b.addModule("power-protocol", .{
|
||||||
.root_source_file = b.path("system/services/power/protocol.zig"),
|
.root_source_file = b.path("system/services/power/protocol.zig"),
|
||||||
});
|
});
|
||||||
@@ -346,8 +346,6 @@ pub fn build(b: *std.Build) void {
|
|||||||
// which unpacks it and spawns each program (system/initial-ramdisk.zig).
|
// which unpacks it and spawns each program (system/initial-ramdisk.zig).
|
||||||
const vfs_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "vfs", "system/services/vfs/vfs.zig");
|
const vfs_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "vfs", "system/services/vfs/vfs.zig");
|
||||||
const vfstest_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "vfs-test", "system/services/vfs/vfs-test.zig");
|
const vfstest_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "vfs-test", "system/services/vfs/vfs-test.zig");
|
||||||
const hpet_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "hpet", "system/drivers/hpet/hpet.zig");
|
|
||||||
const bus_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "bus", "system/drivers/bus/bus.zig");
|
|
||||||
const ps2_bus_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-bus", "system/drivers/ps2-bus/ps2-bus.zig");
|
const ps2_bus_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-bus", "system/drivers/ps2-bus/ps2-bus.zig");
|
||||||
const ps2_keyboard_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-keyboard", "system/drivers/ps2-bus/keyboard.zig");
|
const ps2_keyboard_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-keyboard", "system/drivers/ps2-bus/keyboard.zig");
|
||||||
const ps2_mouse_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-mouse", "system/drivers/ps2-bus/mouse.zig");
|
const ps2_mouse_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-mouse", "system/drivers/ps2-bus/mouse.zig");
|
||||||
@@ -361,7 +359,7 @@ pub fn build(b: *std.Build) void {
|
|||||||
const crash_test_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "crash-test", "system/services/crash-test/crash-test.zig");
|
const crash_test_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "crash-test", "system/services/crash-test/crash-test.zig");
|
||||||
const device_list_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "device-list", "system/services/device-list/device-list.zig");
|
const device_list_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "device-list", "system/services/device-list/device-list.zig");
|
||||||
// The discovery service: one swappable process per firmware
|
// The discovery service: one swappable process per firmware
|
||||||
// (docs/m19-m20-plan.md decision 7), bundled under the neutral ramdisk name
|
// (docs/discovery.md), bundled under the neutral ramdisk name
|
||||||
// "discovery" so the device manager never learns which firmware it is on.
|
// "discovery" so the device manager never learns which firmware it is on.
|
||||||
// x86 boots describe hardware with ACPI; the Raspberry Pis hand over a
|
// x86 boots describe hardware with ACPI; the Raspberry Pis hand over a
|
||||||
// flattened device tree — the aarch64 target flips the default when it
|
// flattened device tree — the aarch64 target flips the default when it
|
||||||
@@ -396,10 +394,6 @@ pub fn build(b: *std.Build) void {
|
|||||||
mk_run.addFileArg(vfs_exe.getEmittedBin());
|
mk_run.addFileArg(vfs_exe.getEmittedBin());
|
||||||
mk_run.addArg("vfs-test");
|
mk_run.addArg("vfs-test");
|
||||||
mk_run.addFileArg(vfstest_exe.getEmittedBin());
|
mk_run.addFileArg(vfstest_exe.getEmittedBin());
|
||||||
mk_run.addArg("hpet");
|
|
||||||
mk_run.addFileArg(hpet_exe.getEmittedBin());
|
|
||||||
mk_run.addArg("bus");
|
|
||||||
mk_run.addFileArg(bus_exe.getEmittedBin());
|
|
||||||
mk_run.addArg("ps2-bus");
|
mk_run.addArg("ps2-bus");
|
||||||
mk_run.addFileArg(ps2_bus_exe.getEmittedBin());
|
mk_run.addFileArg(ps2_bus_exe.getEmittedBin());
|
||||||
mk_run.addArg("ps2-keyboard");
|
mk_run.addArg("ps2-keyboard");
|
||||||
@@ -435,8 +429,6 @@ pub fn build(b: *std.Build) void {
|
|||||||
.{ vfs_exe, "system/services" },
|
.{ vfs_exe, "system/services" },
|
||||||
.{ device_manager_exe, "system/services" },
|
.{ device_manager_exe, "system/services" },
|
||||||
.{ input_exe, "system/services" },
|
.{ input_exe, "system/services" },
|
||||||
.{ hpet_exe, "system/drivers" },
|
|
||||||
.{ bus_exe, "system/drivers" },
|
|
||||||
.{ ps2_bus_exe, "system/drivers" },
|
.{ ps2_bus_exe, "system/drivers" },
|
||||||
.{ ps2_keyboard_exe, "system/drivers" },
|
.{ ps2_keyboard_exe, "system/drivers" },
|
||||||
.{ ps2_mouse_exe, "system/drivers" },
|
.{ ps2_mouse_exe, "system/drivers" },
|
||||||
@@ -615,6 +607,21 @@ pub fn build(b: *std.Build) void {
|
|||||||
});
|
});
|
||||||
test_step.dependOn(&b.addRunArtifact(xkb_tests).step);
|
test_step.dependOn(&b.addRunArtifact(xkb_tests).step);
|
||||||
|
|
||||||
|
// runtime.time's Instant/Duration arithmetic. time.zig pulls in system.zig (the
|
||||||
|
// syscall wrappers), which needs the `abi` module, so it doesn't fit the plain
|
||||||
|
// loop above.
|
||||||
|
const time_tests = b.addTest(.{
|
||||||
|
.root_module = b.createModule(.{
|
||||||
|
.root_source_file = b.path("library/runtime/time.zig"),
|
||||||
|
.target = target,
|
||||||
|
.optimize = optimize,
|
||||||
|
.imports = &.{
|
||||||
|
.{ .name = "abi", .module = abi_module },
|
||||||
|
},
|
||||||
|
}),
|
||||||
|
});
|
||||||
|
test_step.dependOn(&b.addRunArtifact(time_tests).step);
|
||||||
|
|
||||||
// Convenience: `zig build gen-xkeyboard-config` regenerates the layout tables from the
|
// Convenience: `zig build gen-xkeyboard-config` regenerates the layout tables from the
|
||||||
// vendored data (offline). `fetch` (the network step) stays a manual script run.
|
// vendored data (offline). `fetch` (the network step) stays a manual script run.
|
||||||
const gen_xkb = b.addSystemCommand(&.{ "python3", "tools/make-xkeyboard-config.py", "generate" });
|
const gen_xkb = b.addSystemCommand(&.{ "python3", "tools/make-xkeyboard-config.py", "generate" });
|
||||||
|
|||||||
+16
-3
@@ -85,6 +85,10 @@ Start with the north star:
|
|||||||
|
|
||||||
Cutting across all of these:
|
Cutting across all of these:
|
||||||
|
|
||||||
|
- **[system-requirements.md](system-requirements.md) — system requirements.** The
|
||||||
|
hardware needed to run danos: minimum specs (UEFI x86-64, ACPI, PCIe ECAM,
|
||||||
|
xHCI, ~128 MiB RAM) grounded in what the boot path actually assumes, plus a
|
||||||
|
plain-language guide matching Intel/AMD CPU generations by name.
|
||||||
- **[arch.md](arch.md) — the architecture split.** How CPU-specific code is kept
|
- **[arch.md](arch.md) — the architecture split.** How CPU-specific code is kept
|
||||||
behind a build-time `arch` module so the generic kernel never names x86_64,
|
behind a build-time `arch` module so the generic kernel never names x86_64,
|
||||||
leaving room for other systems (e.g. an AArch64 Raspberry Pi) later.
|
leaving room for other systems (e.g. an AArch64 Raspberry Pi) later.
|
||||||
@@ -96,7 +100,16 @@ Cutting across all of these:
|
|||||||
when to build it, and how to keep it architecture-agnostic.
|
when to build it, and how to keep it architecture-agnostic.
|
||||||
- **[acpi.md](acpi.md) — finding the ACPI tables.** The concrete x86 locator chain:
|
- **[acpi.md](acpi.md) — finding the ACPI tables.** The concrete x86 locator chain:
|
||||||
how the loader captures the **RSDP**, hands its physical address across in `BootInfo`,
|
how the loader captures the **RSDP**, hands its physical address across in `BootInfo`,
|
||||||
and how the platform derives the **RSDT/XSDT** from it and walks the SDTs.
|
and how the platform derives the **RSDT/XSDT** from it and walks the SDTs — plus the
|
||||||
|
live event side (the SCI, the power button, GPE/Notify) the ring-3 acpi service runs.
|
||||||
|
- **[power.md](power.md) — the power service.** System power as a domain-named
|
||||||
|
service: button/lid/battery events published to subscribers, and init's orderly
|
||||||
|
shutdown composing the [lifecycle](process-lifecycle.md) stop sequence with an ACPI
|
||||||
|
S5 write. Firmware-neutral — a PSCI backend drops in on ARM.
|
||||||
|
- **[timers.md](timers.md) — timers and time.** The ring-3 surface for reading the
|
||||||
|
clock and waiting: why `now()` is a syscall rather than a service, and the one-shot
|
||||||
|
timer notification (`timer_bind`) that gives supervisors a timed wait — built on the
|
||||||
|
LAPIC heartbeat and calibrated TSC of [device-interrupts.md](device-interrupts.md).
|
||||||
- **[smp.md](smp.md) — multiple cores.** A design/research note on how microkernels
|
- **[smp.md](smp.md) — multiple cores.** A design/research note on how microkernels
|
||||||
(L4, seL4) handle SMP — big kernel lock vs per-CPU vs multikernel — and how the
|
(L4, seL4) handle SMP — big kernel lock vs per-CPU vs multikernel — and how the
|
||||||
right choice depends on whether danos is chasing real-time or resilience.
|
right choice depends on whether danos is chasing real-time or resilience.
|
||||||
@@ -153,7 +166,7 @@ addressed as **`system/services/init`** — the repeated leaf resolves away:
|
|||||||
| Source (root file) | Addressed as (module / binary / FHS path) |
|
| Source (root file) | Addressed as (module / binary / FHS path) |
|
||||||
|----------------------------------------|--------------------------------------------|
|
|----------------------------------------|--------------------------------------------|
|
||||||
| `system/services/init/init.zig` | `system/services/init` → `/system/services/init` |
|
| `system/services/init/init.zig` | `system/services/init` → `/system/services/init` |
|
||||||
| `system/drivers/hpet/hpet.zig` | `system/drivers/hpet` → `/system/drivers/hpet` |
|
| `system/drivers/ps2-bus/ps2-bus.zig` | `system/drivers/ps2-bus` → `/system/drivers/ps2-bus` |
|
||||||
| `library/runtime/runtime.zig` | `library/runtime` (the `runtime` module) |
|
| `library/runtime/runtime.zig` | `library/runtime` (the `runtime` module) |
|
||||||
|
|
||||||
In **source**, a sub-project is a directory so it can hold many files — the entry is
|
In **source**, a sub-project is a directory so it can hold many files — the entry is
|
||||||
@@ -219,6 +232,6 @@ appears in the private-ABI path.
|
|||||||
| danos-native runtime (`runtime`): syscall wrappers, heap, IPC, device access — the stable application ABI | `library/runtime/` |
|
| danos-native runtime (`runtime`): syscall wrappers, heap, IPC, device access — the stable application ABI | `library/runtime/` |
|
||||||
| POSIX/C compatibility (`posix`): unistd, stdio — the one place POSIX names are allowed | `library/posix/` |
|
| POSIX/C compatibility (`posix`): unistd, stdio — the one place POSIX names are allowed | `library/posix/` |
|
||||||
| System services (init, the VFS server + `protocol`, the device-manager) | `system/services/` |
|
| System services (init, the VFS server + `protocol`, the device-manager) | `system/services/` |
|
||||||
| Device drivers, one sub-project each (`hpet` leaf driver, `bus` bus driver) | `system/drivers/` |
|
| Device drivers, one sub-project each (`pci-bus`, `ps2-bus`, `usb-xhci-bus` bus drivers) | `system/drivers/` |
|
||||||
| Build + `run-x86-64` (QEMU/OVMF) | `build.zig` |
|
| Build + `run-x86-64` (QEMU/OVMF) | `build.zig` |
|
||||||
| QEMU integration test harness | `test/qemu_test.py` |
|
| QEMU integration test harness | `test/qemu_test.py` |
|
||||||
|
|||||||
+55
-1
@@ -107,12 +107,66 @@ firmware-agnostic [device model](discovery.md) gets populated; this note stops a
|
|||||||
part that answers "where are the tables?" — everything past the RSDP is just following
|
part that answers "where are the tables?" — everything past the RSDP is just following
|
||||||
more pointers the tables themselves provide.
|
more pointers the tables themselves provide.
|
||||||
|
|
||||||
|
## ACPI events: the SCI, the power button, and GPEs (M21)
|
||||||
|
|
||||||
|
The tables above are static description; ACPI is also a *live* channel. Hardware
|
||||||
|
raises the **SCI** (System Control Interrupt) — one shared, level-triggered line
|
||||||
|
whose vector the FADT names — and the OS reads status registers to learn what
|
||||||
|
happened: a fixed event like the power button, or a **General-Purpose Event**
|
||||||
|
(GPE) whose handler is an AML method. Since [discovery](discovery.md) moved AML
|
||||||
|
to ring 3, the event side lives there too, in the same **acpi service** — the
|
||||||
|
device discoverer and the event source are one process, because both need the
|
||||||
|
namespace and the port grant.
|
||||||
|
|
||||||
|
**The kernel hands the service what it needs and no more.** Reading PM1 event
|
||||||
|
blocks and GPE blocks requires the FADT, which the kernel already parses for its
|
||||||
|
own `\_S5` poweroff. Rather than re-parse, the kernel appends the **FADT as one
|
||||||
|
more memory resource** on the `acpi-tables` node; the service tells it apart
|
||||||
|
from the AML blob resources by signature — the FADT keeps its intact `"FACP"`
|
||||||
|
header, while the blob resources are header-stripped bytecode that starts with
|
||||||
|
no signature. The kernel's own FADT parse is untouched; the service reads the
|
||||||
|
PM1 *event* blocks (which the kernel never parsed — it only needs PM1 *control*
|
||||||
|
for `\_S5`) and the GPE0/GPE1 blocks straight from its copy. The **SCI itself**
|
||||||
|
arrives as the node's one `len == 1` irq resource (distinct from the broad
|
||||||
|
`[0, 256)` window that covers children's legacy lines), which is how the service
|
||||||
|
finds the line to `irq_bind`.
|
||||||
|
|
||||||
|
With those in hand the service enables ACPI mode (only if `SCI_EN` is clear —
|
||||||
|
some firmwares boot with it already set), sets `PWRBTN_EN`, and on each SCI:
|
||||||
|
|
||||||
|
- **The power button** is a *fixed* event: a set `PWRBTN_STS` bit in PM1 status.
|
||||||
|
The handler clears it (write-1-to-clear), logs the press, and publishes a
|
||||||
|
[`power`](power.md) `power_button` event to subscribers.
|
||||||
|
- **GPEs** are the general path: for each set-and-enabled GPE bit `n`, the
|
||||||
|
service evaluates its `\_GPE._L%02X` (level) or `_E%02X` (edge) handler
|
||||||
|
method, drains the **Notify** queue that method produced, maps each notified
|
||||||
|
device to an event (battery, AC, lid, or a generic `notify` with its code),
|
||||||
|
and clears the status bit. A missing handler method is clear-and-log, not an
|
||||||
|
error. Making GPEs work required teaching the interpreter one opcode it never
|
||||||
|
handled — `Notify` (`0x86`) — which it now folds into a bounded queue drained
|
||||||
|
per evaluation; everything else a handler needs (field access, control flow,
|
||||||
|
method calls) was already proven by the ring-3 `_STA`/`_CRS` work.
|
||||||
|
|
||||||
|
**How this is tested.** QEMU cannot raise GPEs deterministically on this config,
|
||||||
|
so GPE/Notify correctness is proven by **host unit tests** — hand-encoded AML
|
||||||
|
with a `Notify` inside a method body, run under `zig build test`. The QEMU
|
||||||
|
`power-button` scenario proves the fixed-event path end to end: a QMP
|
||||||
|
`system_powerdown` injects a real ACPI power-button press, and the service's SCI
|
||||||
|
handler must log it. Battery/AC/lid and the embedded controller's `_Qxx` queries
|
||||||
|
are interface-complete but validated on real hardware later.
|
||||||
|
|
||||||
|
The service surface these events are *published on* — subscription, the event
|
||||||
|
vocabulary, and orderly shutdown — is the power service, [power.md](power.md).
|
||||||
|
|
||||||
## Related
|
## Related
|
||||||
|
|
||||||
- [efi.md](efi.md) — the loader that captures the RSDP before `ExitBootServices`.
|
- [efi.md](efi.md) — the loader that captures the RSDP before `ExitBootServices`.
|
||||||
- [memory-map.md](memory-map.md) — the same loader-captures / kernel-consumes seam, and
|
- [memory-map.md](memory-map.md) — the same loader-captures / kernel-consumes seam, and
|
||||||
the ACPI-reclaim memory the RSDP lives in.
|
the ACPI-reclaim memory the RSDP lives in.
|
||||||
- [discovery.md](discovery.md) — the broader (still-evolving) plan for turning these
|
- [discovery.md](discovery.md) — the broader (still-evolving) plan for turning these
|
||||||
tables into one neutral device model shared with the ARM device-tree path.
|
tables into one neutral device model shared with the ARM device-tree path, and how
|
||||||
|
ACPI enumeration and events moved to the ring-3 acpi service.
|
||||||
|
- [power.md](power.md) — the domain-named power service the ACPI event side publishes
|
||||||
|
to (button, lid, battery) and its orderly-shutdown path into S5.
|
||||||
- [arch.md](arch.md) — why the kernel reaches the device code through a `platform`
|
- [arch.md](arch.md) — why the kernel reaches the device code through a `platform`
|
||||||
module and never names ACPI directly.
|
module and never names ACPI directly.
|
||||||
|
|||||||
@@ -96,7 +96,7 @@ That's all — no Unix-abbreviation exception. The source directories are full w
|
|||||||
(`system`, `library`, not `src`/`lib`), and there is no daemon `d` suffix: a driver
|
(`system`, `library`, not `src`/`lib`), and there is no daemon `d` suffix: a driver
|
||||||
lives in `system/drivers/` and a service in `system/services/`, so the *location*
|
lives in `system/drivers/` and a service in `system/services/`, so the *location*
|
||||||
already says what it is. Encoding the role in the name too (`busd`, `vfsd`) is
|
already says what it is. Encoding the role in the name too (`busd`, `vfsd`) is
|
||||||
redundant — the program is just `bus`, `vfs`. Don't put in a name what its directory
|
redundant — the program is just `ps2-bus`, `vfs`. Don't put in a name what its directory
|
||||||
already tells you.
|
already tells you.
|
||||||
|
|
||||||
## A note on collisions
|
## A note on collisions
|
||||||
@@ -138,7 +138,7 @@ single word or acronym needs no hyphen: `scheduler.zig`, `paging.zig`, `apic.zig
|
|||||||
conventions above — `snake_case` — because it's an identifier, not a filename.)
|
conventions above — `snake_case` — because it's an identifier, not a filename.)
|
||||||
|
|
||||||
**A sub-project's entry point repeats its directory's name** — `init/init.zig`,
|
**A sub-project's entry point repeats its directory's name** — `init/init.zig`,
|
||||||
`runtime/runtime.zig`, `hpet/hpet.zig` — and the sub-project is addressed by the
|
`runtime/runtime.zig`, `ps2-bus/ps2-bus.zig` — and the sub-project is addressed by the
|
||||||
*directory* (`system/services/init`, `library/runtime`), with the repeated leaf
|
*directory* (`system/services/init`, `library/runtime`), with the repeated leaf
|
||||||
resolving away. See the repository-layout section of [README.md](README.md).
|
resolving away. See the repository-layout section of [README.md](README.md).
|
||||||
|
|
||||||
|
|||||||
@@ -17,7 +17,7 @@ Most modern Unix and Unix-like operating systems follow the FHS. DanOS has its o
|
|||||||
| /srv | Site-specific data served by this system, such as data and scripts for web servers, data offered by FTP servers, and repositories for version control systems |
|
| /srv | Site-specific data served by this system, such as data and scripts for web servers, data offered by FTP servers, and repositories for version control systems |
|
||||||
| /system | DanOS operating system files (similar idea to C:\Windows). A true representation of danos — its layout mirrors the source tree, so `/system` is what danos *is*. |
|
| /system | DanOS operating system files (similar idea to C:\Windows). A true representation of danos — its layout mirrors the source tree, so `/system` is what danos *is*. |
|
||||||
| /system/devices | danos virtual device tree e.g. similar to /sys on linux but with danos device tree conventions (the structures in the devices module) |
|
| /system/devices | danos virtual device tree e.g. similar to /sys on linux but with danos device tree conventions (the structures in the devices module) |
|
||||||
| /system/drivers | driver binaries, one sub-project each (e.g. /system/drivers/hpet) |
|
| /system/drivers | driver binaries, one sub-project each (e.g. /system/drivers/pci-bus, /system/drivers/ps2-bus) |
|
||||||
| /system/services | system-service binaries — the VFS server, init, and other user-mode servers (e.g. /system/services/vfs, /system/services/init) |
|
| /system/services | system-service binaries — the VFS server, init, and other user-mode servers (e.g. /system/services/vfs, /system/services/init) |
|
||||||
| /system/kernel | the kernel image |
|
| /system/kernel | the kernel image |
|
||||||
| /tmp | Directory for temporary files (see also /var/tmp). Often not preserved between system reboots and may be severely size-restricted. |
|
| /tmp | Directory for temporary files (see also /var/tmp). Often not preserved between system reboots and may be severely size-restricted. |
|
||||||
@@ -61,8 +61,8 @@ to the driver in the order written, and a read consumes what is there. Terminals
|
|||||||
serial lines, keyboards and mice are all of this shape. These are the natural first
|
serial lines, keyboards and mice are all of this shape. These are the natural first
|
||||||
device nodes in danos, because a character driver needs nothing the kernel doesn't
|
device nodes in danos, because a character driver needs nothing the kernel doesn't
|
||||||
already provide — it claims its device, maps its registers with `mmio_map`, and blocks
|
already provide — it claims its device, maps its registers with `mmio_map`, and blocks
|
||||||
on `replyWait` for either an interrupt or a client request. `system/drivers/hpet/hpet.zig` is already
|
on `replyWait` for either an interrupt or a client request. `system/drivers/ps2-bus/ps2-bus.zig`
|
||||||
that program, minus the client half.
|
is already that program, minus the file-node client half.
|
||||||
|
|
||||||
The obstacle was never the file type; it is which hardware a ring-3 driver can reach.
|
The obstacle was never the file type; it is which hardware a ring-3 driver can reach.
|
||||||
Direct `in`/`out` from user space is still a #GP (no TSS I/O bitmap, IOPL never raised),
|
Direct `in`/`out` from user space is still a #GP (no TSS I/O bitmap, IOPL never raised),
|
||||||
|
|||||||
@@ -78,6 +78,40 @@ preemption and wakeups (1 ms granularity); the **TSC** is the resolution you rea
|
|||||||
time at. Making `sleep` itself sub-millisecond would take a tickless one-shot
|
time at. Making `sleep` itself sub-millisecond would take a tickless one-shot
|
||||||
timer — a later step.
|
timer — a later step.
|
||||||
|
|
||||||
|
### Is the TSC trustworthy? Invariant, and synchronized
|
||||||
|
|
||||||
|
A cycle counter is only a valid *clock* if two things hold, and danos checks both,
|
||||||
|
because they decide whether we read time with a cheap `rdtsc` or fall back to the HPET.
|
||||||
|
|
||||||
|
**Invariant.** An old TSC counted core clock cycles, so it sped up and slowed down with
|
||||||
|
frequency scaling — useless as wall time. Modern CPUs (all of danos's targets) provide an
|
||||||
|
**invariant TSC**: a constant rate across P/C-states that never stops. The guarantee is a
|
||||||
|
CPUID bit — leaf `0x80000007`, EDX bit 8 — on both Intel *and* AMD. danos reads it in
|
||||||
|
`calibrate`, and a TSC that doesn't advertise it is not used as the clocksource. AMD is
|
||||||
|
why this matters in practice: it doesn't populate the Intel leaf `0x15` that enumerates
|
||||||
|
the TSC *frequency*, so danos already measures AMD's rate against the HPET — but a
|
||||||
|
measured frequency without the invariance guarantee is not enough.
|
||||||
|
|
||||||
|
**Synchronized.** Each core has its own TSC. Even invariant ones can start at different
|
||||||
|
values (a second socket, some firmware), so a thread migrating from a core reading
|
||||||
|
`1_000_000` to one reading `999_000` would see time jump *backward*. danos runs a **warp
|
||||||
|
check** as each application processor comes online (`checkWarpSource`, adapted from
|
||||||
|
Linux's): the waking core and the BSP hammer a shared "highest seen" TSC under a lock,
|
||||||
|
and if either ever reads below it, the cores' TSCs are skewed. It's pairwise because APs
|
||||||
|
come up one at a time ([smp.md](smp.md)).
|
||||||
|
|
||||||
|
**The fallback.** When the TSC fails either test — non-invariant (a bare VM such as the
|
||||||
|
default qemu64), or warped between cores — danos moves the monotonic clock onto the
|
||||||
|
**HPET** main counter: one fixed-rate counter, so it can neither skew between cores nor
|
||||||
|
drift with frequency. It costs a memory-mapped read instead of a register read, but it
|
||||||
|
keeps time *accurate*, which is the whole point. The switch preserves the current value,
|
||||||
|
so the clock never jumps. The boot log names the outcome:
|
||||||
|
|
||||||
|
```
|
||||||
|
/system/kernel: clocksource tsc (TSC invariant: yes, synchronized: yes) # real Intel/AMD
|
||||||
|
/system/kernel: clocksource hpet (TSC invariant: no, synchronized: yes) # a bare VM (TCG)
|
||||||
|
```
|
||||||
|
|
||||||
## Two kinds of vector, one dispatch
|
## Two kinds of vector, one dispatch
|
||||||
|
|
||||||
The IDT now installs gates `0-47`: the 32 exceptions plus the device range. Every
|
The IDT now installs gates `0-47`: the 32 exceptions plus the device range. Every
|
||||||
|
|||||||
+21
-5
@@ -51,7 +51,16 @@ enumeration is a **pci-bus driver**: the manager spawns it against the host brid
|
|||||||
like any bus reports children. ACPI becomes an **acpi service** that interprets the
|
like any bus reports children. ACPI becomes an **acpi service** that interprets the
|
||||||
tables and reports the namespace. The manager only orchestrates and merges. Moving
|
tables and reports the namespace. The manager only orchestrates and merges. Moving
|
||||||
AML interpretation out of ring 0 is its own project on its own track; nothing here
|
AML interpretation out of ring 0 is its own project on its own track; nothing here
|
||||||
depends on when it lands.
|
depends on when it lands. (It landed: [discovery.md](discovery.md), M19–M20.)
|
||||||
|
|
||||||
|
`device_register` is **idempotent on exact match**: a re-registration with an
|
||||||
|
identical (parent, class, identity, resources) tuple returns the existing id
|
||||||
|
instead of appending a duplicate. The kernel table has no unregister, so without
|
||||||
|
this a restarted registering bus would re-report its children as fresh nodes on
|
||||||
|
every respawn. Idempotence is what makes restart-and-re-report sound for *every*
|
||||||
|
reporting bus — pci-bus, the acpi service, a future fdt service — not just one,
|
||||||
|
and it is why supervision (below) can prune a dead bus's subtree and trust the
|
||||||
|
restarted instance to rebuild exactly the same ids.
|
||||||
|
|
||||||
## The protocol
|
## The protocol
|
||||||
|
|
||||||
@@ -136,10 +145,17 @@ published exit events, signals + `runtime.process`). On top of those:
|
|||||||
the mouse and keyboard QEMU already hangs off it.
|
the mouse and keyboard QEMU already hangs off it.
|
||||||
7. **App surface**: `enumerate`/`subscribe` over IPC; `device_enumerate` retreats
|
7. **App surface**: `enumerate`/`subscribe` over IPC; `device_enumerate` retreats
|
||||||
to a manager-internal seam.
|
to a manager-internal seam.
|
||||||
8. **Discovery migration** — DONE (M19–M20, 2026-07-13): pci-bus driver (M19)
|
8. **Discovery migration** — DONE (M19–M20, 2026-07-13): enumeration moved to
|
||||||
then the acpi service (M20) moved enumeration to ring 3; the kernel seeds
|
ring 3 as swappable per-firmware discoverers — the pci-bus driver (M19) then
|
||||||
only the host bridge and the acpi-tables node. See
|
the acpi service (M20), see [discovery.md](discovery.md); the kernel seeds
|
||||||
[m19-m20-plan.md](m19-m20-plan.md).
|
only the host bridge and the acpi-tables node. Matching moved with it:
|
||||||
|
`child_added` grew a `device_id` (the kernel-registered id, `no_device` for
|
||||||
|
unregistered leaves like USB ports) and a firmware `hid`, and the manager now
|
||||||
|
matches drivers from those **reports** rather than its boot-time snapshot. The
|
||||||
|
PCI arm flipped in M19.3, the ACPI arm (ps2-bus matched from `_HID`) in M20.3
|
||||||
|
— each in a single phase so no device is ever matched from both sources at
|
||||||
|
once. The acpi service reports only the non-PCI `_HID` devices, since pci-bus
|
||||||
|
already reports PCI functions (M20.2).
|
||||||
|
|
||||||
## Settled questions (2026-07-12)
|
## Settled questions (2026-07-12)
|
||||||
|
|
||||||
|
|||||||
@@ -191,3 +191,57 @@ and registers + reports each `_HID` device — the device manager matches driver
|
|||||||
(ps2-bus) from those reports. With M19's pci-bus driver, discovery now runs
|
(ps2-bus) from those reports. With M19's pci-bus driver, discovery now runs
|
||||||
entirely in user space; the kernel seeds only the host bridge and the
|
entirely in user space; the kernel seeds only the host bridge and the
|
||||||
acpi-tables node.
|
acpi-tables node.
|
||||||
|
|
||||||
|
## Discovery is a swappable process per firmware (M19–M20)
|
||||||
|
|
||||||
|
Moving PCI and ACPI enumeration out of ring 0 was not just a relocation — it
|
||||||
|
made discovery **firmware-neutral by construction**, which is the whole reason
|
||||||
|
to do it before the second architecture rather than after. Everything at and
|
||||||
|
above the [device-manager](device-manager.md) protocol — descriptors,
|
||||||
|
containment, reports, matching, supervision — is generic and may never become
|
||||||
|
x86-specific. Discovery is the single firmware-specific piece, and it is
|
||||||
|
isolated as **one swappable process per firmware**:
|
||||||
|
|
||||||
|
- **x86** boots describe hardware with ACPI, so the discoverer is the **acpi
|
||||||
|
service** ([acpi.md](acpi.md)): it claims the `acpi-tables` node and runs AML.
|
||||||
|
- **The Raspberry Pis** hand over a flattened device tree, so the discoverer is
|
||||||
|
an **fdt service**: it claims a `devicetree-blob` node and walks the tree —
|
||||||
|
pure data, no bytecode, so it needs neither a port grant nor an interpreter,
|
||||||
|
strictly simpler than ACPI. (A placeholder until the [aarch64](arm.md)
|
||||||
|
bring-up fills it in.)
|
||||||
|
|
||||||
|
The device manager spawns the discoverer under the **neutral ramdisk name
|
||||||
|
`discovery`** and never learns which firmware it is on; the build's
|
||||||
|
`-Ddiscovery=acpi|fdt` option fills that slot (x86 defaults to `acpi`, the
|
||||||
|
aarch64 target flips the default when it lands). The manager owns the device
|
||||||
|
tree as *data* and touches no hardware, ever — firmware bytecode runs only
|
||||||
|
inside the crashable, supervised discoverer, so an AML fault can never take
|
||||||
|
down the supervisor.
|
||||||
|
|
||||||
|
Two consequences of neutrality bind on later work:
|
||||||
|
|
||||||
|
- **Cross-firmware surfaces are named by domain, not firmware.** System power is
|
||||||
|
a [`power`](power.md) protocol, not an "ACPI events" protocol: on x86 the acpi
|
||||||
|
service registers it, on ARM a PSCI/mailbox service registers the same
|
||||||
|
`ServiceId.power`, and subscribers never learn the difference.
|
||||||
|
- **Identity must widen before the fdt service exists.** `DeviceDescriptor`'s
|
||||||
|
8-byte `hid` holds an EISA id but cannot hold an FDT `compatible` string
|
||||||
|
(`"brcm,bcm2835-aux-uart"`); the identity field grows before the ARM path can
|
||||||
|
report a real node.
|
||||||
|
|
||||||
|
Two supporting decisions keep the kernel's remaining slice honest:
|
||||||
|
|
||||||
|
- **The AML interpreter is a shared build module**, compiled into both the
|
||||||
|
kernel and the acpi service — one source, two builds, no fork. The kernel
|
||||||
|
links it for the `\_S5` poweroff evaluation, the service links it for
|
||||||
|
everything else, and the `acpi-parse` test asserts the two produce the same
|
||||||
|
device count across the ring-3 move.
|
||||||
|
- **Bridge apertures come from the firmware memory map, not AML.** Registered
|
||||||
|
PCI functions carry BAR resources, and `device_register` containment demands
|
||||||
|
the bridge own windows that cover them. Those apertures are derived
|
||||||
|
kernel-side from the boot memory map's MMIO holes (regions that are neither
|
||||||
|
RAM nor tables) — mechanical, AML-free, and available at boot regardless of
|
||||||
|
what later moved to user space. The acpi service's authority is likewise
|
||||||
|
exactly one node: the `acpi-tables` node, whose broad io_port grant is the
|
||||||
|
documented trust boundary for the one process allowed to run firmware
|
||||||
|
bytecode.
|
||||||
|
|||||||
+10
-9
@@ -57,8 +57,9 @@ is not an address window. Discovery is trusted; user space is not.
|
|||||||
|
|
||||||
### What a bus driver looks like
|
### What a bus driver looks like
|
||||||
|
|
||||||
`system/drivers/bus/bus.zig` is the smallest honest one. Its "bus" is the HPET's register block and
|
danos ships no demo bus driver — the real ones are `pci-bus`, `ps2-bus`, and
|
||||||
its "devices" are the block's comparators:
|
`usb-xhci-bus`. The smallest *honest* shape, illustrated here with an HPET register block
|
||||||
|
as the "bus" and its comparators as the "devices", is:
|
||||||
|
|
||||||
```zig
|
```zig
|
||||||
_ = dev.claim(bus.id); // 1. own the bus
|
_ = dev.claim(bus.id); // 1. own the bus
|
||||||
@@ -78,8 +79,8 @@ for (0..n) |i| { // 3. publish each child
|
|||||||
|
|
||||||
Each child is left **unclaimed**, which is the handoff: a comparator driver can now
|
Each child is left **unclaimed**, which is the handoff: a comparator driver can now
|
||||||
`device_claim` one and `mmio_map` it, and will see only its own 0x20-byte window. A child
|
`device_claim` one and `mmio_map` it, and will see only its own 0x20-byte window. A child
|
||||||
whose window escapes the bus is refused — `bus` asserts that, and the `bus` test
|
whose window escapes the bus is refused; the in-kernel `containment` test asserts the
|
||||||
asserts the kernel's table upholds it.
|
kernel's table upholds that ([drivers.md](drivers.md)).
|
||||||
|
|
||||||
A USB device has *no* resources at all: `resource_count = 0`, because it's addressed
|
A USB device has *no* resources at all: `resource_count = 0`, because it's addressed
|
||||||
through its controller, not by MMIO. That case is allowed and is the common one.
|
through its controller, not by MMIO. That case is allowed and is the common one.
|
||||||
@@ -143,7 +144,7 @@ If a class driver needs `mmio`, it has become an HCD and should be one.
|
|||||||
physically-contiguous, pinned, uncacheable, reclaim-on-teardown buffers with the
|
physically-contiguous, pinned, uncacheable, reclaim-on-teardown buffers with the
|
||||||
physical address exposed (`pmm.allocContiguous`, a DMA arena, `mapUserDmaInto`).
|
physical address exposed (`pmm.allocContiguous`, a DMA arena, `mapUserDmaInto`).
|
||||||
`dma_below_4g` caps the address for legacy engines; `dma_write_combining` is accepted
|
`dma_below_4g` caps the address for legacy engines; `dma_write_combining` is accepted
|
||||||
but falls back to coherent until PAT is programmed. hpet is refactored onto `/lib/mmio`;
|
but falls back to coherent until PAT is programmed. The bus drivers use `/lib/mmio`;
|
||||||
no DMA driver consumes `dma_alloc` yet.
|
no DMA driver consumes `dma_alloc` yet.
|
||||||
- **M15** — interrupts for PCI devices, the MSI half. Discovery now gives every PCI
|
- **M15** — interrupts for PCI devices, the MSI half. Discovery now gives every PCI
|
||||||
function its 4 KiB ECAM config space as resource 0 (unblocking the capability walk
|
function its 4 KiB ECAM config space as resource 0 (unblocking the capability walk
|
||||||
@@ -302,8 +303,8 @@ rather than an out-struct. The rest of this section is the original design note.
|
|||||||
|
|
||||||
**The blocker, and it's a hard one.** No PCI device can take an interrupt today.
|
**The blocker, and it's a hard one.** No PCI device can take an interrupt today.
|
||||||
[`addBars`](system/devices/acpi.zig) records `.memory` and `.io_port` BARs and never an
|
[`addBars`](system/devices/acpi.zig) records `.memory` and `.io_port` BARs and never an
|
||||||
`.irq`; there is no `_PRT` parsing anywhere in the tree. `hpet` only works because the
|
`.irq`; there is no `_PRT` parsing anywhere in the tree. The HPET is the one exception —
|
||||||
HPET advertises its own routing options in its own registers — a privilege no ordinary
|
it advertises its own interrupt routing in its own registers, a privilege no ordinary
|
||||||
device has.
|
device has.
|
||||||
|
|
||||||
**The fix, in two halves.**
|
**The fix, in two halves.**
|
||||||
@@ -326,7 +327,7 @@ which means **discovery should give each `pci_device` a `.memory` resource for i
|
|||||||
4 KiB ECAM slot**. That's a small change to `parseMcfg` and it unblocks the whole
|
4 KiB ECAM slot**. That's a small change to `parseMcfg` and it unblocks the whole
|
||||||
capability walk (MSI, MSI-X, PCIe extended caps) without any new syscall.
|
capability walk (MSI, MSI-X, PCIe extended caps) without any new syscall.
|
||||||
|
|
||||||
Note QEMU's HPET reports `Tn_FSB_INT_DEL_CAP = 0` — no MSI — so `hpet` can never
|
Note QEMU's HPET reports `Tn_FSB_INT_DEL_CAP = 0` — no MSI — so an HPET timer could never
|
||||||
exercise this path. The first MSI driver will be the first PCI driver.
|
exercise this path. The first MSI driver will be the first PCI driver.
|
||||||
|
|
||||||
## M16 — the IOMMU, and the honest caveat ◑ detection done, enforcement pending
|
## M16 — the IOMMU, and the honest caveat ◑ detection done, enforcement pending
|
||||||
@@ -353,7 +354,7 @@ gap should be named rather than implied.
|
|||||||
|
|
||||||
`M13` (capability passing) is independent of `M14`/`M15` and is the cheapest. It
|
`M13` (capability passing) is independent of `M14`/`M15` and is the cheapest. It
|
||||||
unlocks class drivers, which are the shape with no hardware requirements at all — you
|
unlocks class drivers, which are the shape with no hardware requirements at all — you
|
||||||
could write a real one against `bus`'s comparators tomorrow.
|
could write a real one against any device a bus driver publishes tomorrow.
|
||||||
|
|
||||||
`M14` and `M15` together unlock the first HCD. `M14`'s barrier layer is worth landing
|
`M14` and `M15` together unlock the first HCD. `M14`'s barrier layer is worth landing
|
||||||
on its own regardless: it's small, obviously correct, and stops every future driver
|
on its own regardless: it's small, obviously correct, and stops every future driver
|
||||||
|
|||||||
+40
-29
@@ -22,12 +22,12 @@ say.*
|
|||||||
|
|
||||||
## How a driver gets started: discover, match, spawn
|
## How a driver gets started: discover, match, spawn
|
||||||
|
|
||||||
Nothing in the kernel decides that the HPET needs the `hpet` driver — that is policy,
|
Nothing in the kernel decides that the PCI host bridge needs the `pci-bus` driver — that
|
||||||
and policy lives in user space. Boot brings user space up as a three-level supervision
|
is policy, and policy lives in user space. Boot brings user space up as a three-level
|
||||||
hierarchy, each level owning one job:
|
supervision hierarchy, each level owning one job:
|
||||||
|
|
||||||
```
|
```
|
||||||
kernel ──spawns──► init (PID 1) ──spawns──► device-manager ──spawns──► hpet
|
kernel ──spawns──► init (PID 1) ──spawns──► device-manager ──spawns──► pci-bus
|
||||||
| | |
|
| | |
|
||||||
spawns only init, the service supervisor: the driver supervisor: enumerates
|
spawns only init, the service supervisor: the driver supervisor: enumerates
|
||||||
publishes the starts the system /system/devices, matches each device
|
publishes the starts the system /system/devices, matches each device
|
||||||
@@ -184,7 +184,11 @@ Two properties worth knowing:
|
|||||||
|
|
||||||
## A whole driver
|
## A whole driver
|
||||||
|
|
||||||
`system/drivers/hpet/hpet.zig` is ~150 lines and does all of it. The shape:
|
A minimal leaf driver is only ~150 lines and does all of it. danos ships **no such
|
||||||
|
example binary** — the driver model is proven by the real drivers (`pci-bus`, `ps2-bus`,
|
||||||
|
`usb-xhci-bus`), and a teaching example belongs here, in the docs, rather than as a
|
||||||
|
compiled program nobody runs. Illustrated with a hypothetical HPET timer driver, the
|
||||||
|
shape is:
|
||||||
|
|
||||||
```zig
|
```zig
|
||||||
const hpet = findHpet(buf) orelse return; // device_enumerate, look for
|
const hpet = findHpet(buf) orelse return; // device_enumerate, look for
|
||||||
@@ -209,8 +213,8 @@ while (...) {
|
|||||||
}
|
}
|
||||||
```
|
```
|
||||||
|
|
||||||
The HPET is a good first driver for a reason that isn't obvious. Its *counter* is a
|
The HPET makes a good illustration for a reason that isn't obvious. Its *counter* is a
|
||||||
clocksource — the only way to use it is to read it, so it proved `mmio_map` without
|
clocksource — the only way to use it is to read it, so it exercises `mmio_map` without
|
||||||
needing interrupts at all. Its *comparators* are a clockevent, and can be configured
|
needing interrupts at all. Its *comparators* are a clockevent, and can be configured
|
||||||
**level-triggered** (`Tn_INT_TYPE_CNF`), which asserts a bit in `GENERAL_INT_STATUS`
|
**level-triggered** (`Tn_INT_TYPE_CNF`), which asserts a bit in `GENERAL_INT_STATUS`
|
||||||
that the driver must write-1-to-clear. That's a genuine deassert step, so the full
|
that the driver must write-1-to-clear. That's a genuine deassert step, so the full
|
||||||
@@ -252,9 +256,10 @@ bus driver may only ever subdivide what it already owns.
|
|||||||
A device with **no resources** is legal and common. A USB device is reached through its
|
A device with **no resources** is legal and common. A USB device is reached through its
|
||||||
controller, not by MMIO, so it gets `resource_count = 0`.
|
controller, not by MMIO, so it gets `resource_count = 0`.
|
||||||
|
|
||||||
See [`system/drivers/bus/bus.zig`](../system/drivers/bus/bus.zig) for a complete one, and
|
See [`system/drivers/pci-bus/pci-bus.zig`](../system/drivers/pci-bus/pci-bus.zig) for a
|
||||||
[driver-model.md](driver-model.md) for how bus drivers, class drivers and host
|
real one — it claims a PCI host bridge, maps its ECAM window, and publishes each function
|
||||||
controller drivers fit together.
|
it finds as a child — and [driver-model.md](driver-model.md) for how bus drivers, class
|
||||||
|
drivers and host controller drivers fit together.
|
||||||
|
|
||||||
## What the kernel does not do for you
|
## What the kernel does not do for you
|
||||||
|
|
||||||
@@ -313,32 +318,38 @@ uncacheable, physical address exposed), and **memory barriers** (`/lib/mmio`'s
|
|||||||
|
|
||||||
## Verifying it
|
## Verifying it
|
||||||
|
|
||||||
The `hpet` test spawns `hpet` from the initial ramdisk and watches the serial log. The driver
|
No demo driver ships to prove this end to end; the *real* drivers do, so the tests
|
||||||
prints `hpet: ok` only after being woken five times, and its loop's only exit is
|
target them and the kernel primitives directly:
|
||||||
through `replyWait` returning a notification — it cannot reach that line by polling.
|
|
||||||
|
|
||||||
The last check doesn't trust the driver's self-report at all: the kernel reads the I/O
|
- **`device-manager`** — boots only the device manager, which discovers the PCI host
|
||||||
APIC redirection entry back and asserts the line really is routed to a device vector,
|
bridge, matches `pci-bus`, and `system_spawn`s it. The test reads kernel state — the
|
||||||
really is level-triggered, and really was left unmasked by the driver's final
|
process table and the device tree — to confirm pci-bus came up and registered the
|
||||||
`irq_ack`.
|
functions it enumerated: the whole discover → match → spawn → driver-up chain.
|
||||||
|
- **`acpi-ps2`** — a user-space driver (`ps2-bus`) is woken by its device's IRQ,
|
||||||
|
delivered as an IPC notification, and attaches the keyboard: IRQ-as-IPC, end to end.
|
||||||
|
- **`pci-scan`** — a user-space driver (`pci-bus`) maps its device's MMIO (the ECAM
|
||||||
|
window) and walks it: `mmio_map`, end to end.
|
||||||
|
- **`containment`** — the kernel refuses a `device_register` whose child window escapes
|
||||||
|
the parent's grant (else it would be a syscall for mapping arbitrary memory), while an
|
||||||
|
identical re-register stays idempotent. Asserted in-kernel, straight against the broker.
|
||||||
|
- **`irqfree`** — the teardown path. Binds two owners to one shared endpoint, releases
|
||||||
|
one, and reads the I/O APIC back: the departing owner's line is masked, the sibling's
|
||||||
|
is not. That second half is why bindings are keyed on the owning *task* and not on the
|
||||||
|
endpoint pointer — endpoints are shared, so releasing "everything pointing at this
|
||||||
|
endpoint" would silently mask a live driver's device.
|
||||||
|
- **`iopass`** — the `device_grant` teardown rule, so destroying a driver's address
|
||||||
|
space never returns MMIO frames to the RAM pool.
|
||||||
|
|
||||||
```
|
```
|
||||||
$ python3 test/qemu_test.py hpet irqfree iopass
|
$ python3 test/qemu_test.py device-manager acpi-ps2 pci-scan containment irqfree iopass
|
||||||
hpet ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
device-manager ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||||
|
acpi-ps2 ... PASS
|
||||||
|
pci-scan ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||||
|
containment ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||||
irqfree ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
irqfree ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||||
iopass ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
iopass ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||||
```
|
```
|
||||||
|
|
||||||
Two companions cover what `hpet` can't, because it never exits:
|
|
||||||
|
|
||||||
- **`irqfree`** — the teardown path. Binds two owners to one shared endpoint, releases
|
|
||||||
one, and reads the I/O APIC back: the departing owner's line is masked, the sibling's
|
|
||||||
is not. That second half is why bindings are keyed on the owning *task* and not on
|
|
||||||
the endpoint pointer — endpoints are shared, so releasing "everything pointing at
|
|
||||||
this endpoint" would silently mask a live driver's device.
|
|
||||||
- **`iopass`** — the `device_grant` teardown rule, so destroying a driver's address
|
|
||||||
space never returns MMIO frames to the RAM pool.
|
|
||||||
|
|
||||||
## What's next (not done here)
|
## What's next (not done here)
|
||||||
|
|
||||||
The big driver-model pieces — capability passing (class drivers), DMA + barriers, MSI,
|
The big driver-model pieces — capability passing (class drivers), DMA + barriers, MSI,
|
||||||
|
|||||||
@@ -1,210 +0,0 @@
|
|||||||
# M17–M18 execution plan: process lifecycle + device manager
|
|
||||||
|
|
||||||
**Archived — completed 2026-07-13** (every item checked; suite ended 54/54).
|
|
||||||
Kept as the record of how M17–M18 landed; the successor is
|
|
||||||
[m19-m20-plan.md](m19-m20-plan.md).
|
|
||||||
|
|
||||||
The operational plan for building [process-lifecycle.md](process-lifecycle.md)
|
|
||||||
(M17) and [device-manager.md](device-manager.md) increments 5–7 (M18). Design is
|
|
||||||
settled in those documents; this file is the build order — one phase at a time,
|
|
||||||
each phase green before the next starts. Delete or archive this file when M18
|
|
||||||
lands.
|
|
||||||
|
|
||||||
**Definition of green, every phase:** `zig build` clean, `zig build test` clean,
|
|
||||||
`python3 test/qemu_test.py` passes (existing scenarios plus the phase's new one),
|
|
||||||
and the relevant design doc's "known gaps" / status lines updated. Commit per
|
|
||||||
green phase (no co-author trailers).
|
|
||||||
|
|
||||||
**Workflow (settled 2026-07-12):** work happens in a dedicated git worktree, on
|
|
||||||
feature branches cut from `main` — `feat/process-lifecycle` (M17.1–17.4),
|
|
||||||
`feat/device-manager` (M18.1), `feat/usb-xhci-bus` (M18.2–18.3). When a branch's
|
|
||||||
phases are all green it is **auto-merged into `main`**; branches are kept after
|
|
||||||
merge, not deleted. Merges and branches are pushed to origin. Phase 0 (once):
|
|
||||||
commit the design docs, merge the outstanding `feat/usb` work into `main`, and
|
|
||||||
run the existing QEMU suite green before any new work starts.
|
|
||||||
|
|
||||||
**Numbering note:** continues the milestone sequence (driver track ended at M16).
|
|
||||||
|
|
||||||
## Status
|
|
||||||
|
|
||||||
The loop marks a phase `[x]` in the same commit that lands it. A phase is marked
|
|
||||||
only when its definition of green holds.
|
|
||||||
|
|
||||||
- [x] **Phase 0** — baseline: docs committed, feat/usb merged to main, pushed;
|
|
||||||
`usb-xhci-libary.zig` renamed to `usb-xhci-library.zig`; existing QEMU
|
|
||||||
suite green from the worktree (48/48, 2026-07-12).
|
|
||||||
- [x] **M17.1** — kernel releases claims/MSI on death (claims: `releaseAllOwnedBy`
|
|
||||||
in the reap; MSI was already swept by `irq.releaseOwner`; `claim-release`
|
|
||||||
test; suite 49/49)
|
|
||||||
- [x] **M17.2** — exit reasons (`ExitReason` recorded at exit/fault/kill before
|
|
||||||
the notification; `process_exit_reason` supervisor-gated;
|
|
||||||
`runtime.process.exitReason`; kernel + ring-3 assertions; suite 49/49)
|
|
||||||
- [x] **M17.3** — published exit events + VFS subscriber (`process_subscribe`,
|
|
||||||
bounded ref-counted table, publish on every death;
|
|
||||||
`runtime.process.subscribeExits`; VFS handles carry owners and are swept on
|
|
||||||
the owner's death; `vfs-client-death` test; suite 50/50)
|
|
||||||
- [x] **M17.4** — signals, timer notifications, `runtime.process`, the service
|
|
||||||
harness (signal_bind/process_signal + coalescing pending mask; timer_bind
|
|
||||||
on the tick; bindSignals/signalsFrom/sendSignal/stop + timerOnce;
|
|
||||||
runtime.service.run with the zero-length ping; VFS converted; `signals`
|
|
||||||
scenario; suite 51/51)
|
|
||||||
- [x] **merge** `feat/process-lifecycle` → main, push (merged 2026-07-13)
|
|
||||||
- [x] **M18.1** — device-manager protocol: hello + restart policy
|
|
||||||
(device-manager-protocol module; the manager as a harness service:
|
|
||||||
supervised spawns, hello deadline via timer sweep, restart with
|
|
||||||
300/600/1200ms backoff, exit reasons deciding restart-vs-stopped,
|
|
||||||
crash-loop cap; usb-xhci-bus first conforming driver; crash-test fixture
|
|
||||||
re-proving claim release each respawn; `driver-restart` scenario;
|
|
||||||
maximum_tasks 16→32 — the sweep was overflowing the pool; suite 52/52)
|
|
||||||
- [x] **merge** `feat/device-manager` → main, push (merged 2026-07-13)
|
|
||||||
- [x] **M18.2** — xHCI port scan + tree reports (child_added/child_removed in
|
|
||||||
the protocol; the manager's child mirror with death-pruning; xHCI maps the
|
|
||||||
register BAR — resource 0 is ECAM — reads CAPLENGTH/HCSPARAMS1, scans
|
|
||||||
PORTSC, reports connected ports with speed-class identity; `usb-report`
|
|
||||||
scenario proves report → prune → respawn → re-report; suite 53/53)
|
|
||||||
- [x] **M18.3** — app surface: enumerate/subscribe over IPC (subscriber
|
|
||||||
endpoint rides as the call's capability; events are the same structs the
|
|
||||||
buses send); device-list first client; protocol capped at the kernel's
|
|
||||||
IPC MESSAGE_MAXIMUM (256); the startUserTask debug print removed — it
|
|
||||||
sheared concurrent serial lines and was the scenario-flake root cause;
|
|
||||||
`device-list` scenario; suite 54/54)
|
|
||||||
- [x] **merge** `feat/usb-xhci-bus` → main, push (merged 2026-07-13) — **plan complete**
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## M17.1 — the kernel releases a dead process's claims
|
|
||||||
|
|
||||||
The cleanup half of iron rule 1; the prerequisite for every restart story.
|
|
||||||
|
|
||||||
- `system/kernel/devices-broker.zig`: `releaseAllOwnedBy(owner: u32)` — clear
|
|
||||||
every `claimed[]` slot holding this task id.
|
|
||||||
- `system/kernel/process.zig`: call it from the reap path, alongside the existing
|
|
||||||
IRQ-binding release (the ordering comment there says why IRQs go first — claims
|
|
||||||
slot in after them, before the exit notification).
|
|
||||||
- MSI vectors: find where `msi_bind` records per-device vectors (interrupts
|
|
||||||
module) and release those by owner in the same pass.
|
|
||||||
- Docs: remove the claims bullet from process-management.md "Known gaps".
|
|
||||||
|
|
||||||
**Test:** new QEMU scenario `claim-release` — a test child claims an unclaimed
|
|
||||||
device, is killed, is respawned, and claims the same device again successfully;
|
|
||||||
assert both claims in the serial log. Kernel-side unit coverage in
|
|
||||||
`system/kernel/tests.zig` for `releaseAllOwnedBy` (claim two devices as two owners,
|
|
||||||
release one owner, verify exactly its claims freed).
|
|
||||||
|
|
||||||
## M17.2 — exit reasons
|
|
||||||
|
|
||||||
- `system/abi.zig`: `ExitReason` (exited, aborted, segmentation_fault,
|
|
||||||
illegal_instruction, arithmetic_fault, killed).
|
|
||||||
- Kernel: record the reason at every death site — clean exit path, each fault
|
|
||||||
class in `onException`, the kill path. Bounded recent-exits table (ids are never
|
|
||||||
reused, so a small ring keyed by id is enough).
|
|
||||||
- New system call `process_exit_reason(id)` — supervisor-gated, like kill; returns
|
|
||||||
the recorded reason or `-ESRCH` once evicted.
|
|
||||||
- `library/runtime/process.zig`: `ExitReason` + `exitReason(id: u32)`.
|
|
||||||
- Docs: remove the no-exit-status bullet from process-management.md.
|
|
||||||
|
|
||||||
**Test:** extend the `supervision` scenario — three children: one exits cleanly,
|
|
||||||
one faults (the fault-recovery pattern), one is killed; the supervisor asserts all
|
|
||||||
three reasons.
|
|
||||||
|
|
||||||
## M17.3 — published exit events
|
|
||||||
|
|
||||||
- Kernel: bounded subscriber table (endpoints); new system call
|
|
||||||
`process_subscribe(endpoint)` (ungated, like `process_enumerate`); every death
|
|
||||||
posts `notify_exit_bit | id` to each subscriber — the same post the supervisor
|
|
||||||
path already uses.
|
|
||||||
- `library/runtime/process.zig`: `subscribeExits(endpoint)`.
|
|
||||||
- VFS becomes the first subscriber: on an exit event, release every handle keyed
|
|
||||||
by that task id (badges already are task ids). Log the release.
|
|
||||||
- Docs: note the convention in ipc.md (exit events reuse the exit-notification
|
|
||||||
badge encoding).
|
|
||||||
|
|
||||||
**Test:** new QEMU scenario `vfs-client-death` — a client opens a file and is
|
|
||||||
killed without closing; assert the VFS logs the handle release and its open-handle
|
|
||||||
count returns to baseline.
|
|
||||||
|
|
||||||
## M17.4 — signals and the service harness
|
|
||||||
|
|
||||||
- Kernel: per-task pending mask + bound endpoint; system calls
|
|
||||||
`signal_bind(endpoint)` and `process_signal(id, signal)` (supervisor-or-self
|
|
||||||
gated); delivery posts `notify_signal_bit | pending mask`, coalescing; pending
|
|
||||||
signals with no bound endpoint pend silently.
|
|
||||||
- `library/runtime/process.zig`: `Signal`, `SignalSet`, `bindSignals`,
|
|
||||||
`signalsFrom`, `sendSignal`, `stop(id, deadline_ms)` (terminate → wait for exit
|
|
||||||
notification → kill). Implement `terminate`, `reload`, `user_1`, `user_2`;
|
|
||||||
`interrupt`/`quit` are enum members with no sender yet; `alarm` stays unbuilt.
|
|
||||||
- Kernel: **one-shot timer notifications** — `timer_bind(endpoint, ms)` posts a
|
|
||||||
notification badge when the deadline lands (IRQ-as-IPC again, on the timer
|
|
||||||
wheel `sleep` already uses). This is the missing timed-wait primitive:
|
|
||||||
`replyWait` blocks forever and `sleep` blocks the whole process, but `stop()`'s
|
|
||||||
escalation, the device manager's `hello` deadline (M18.1), and restart backoff
|
|
||||||
all need a deadline while staying responsive. It is also the mechanism `alarm`
|
|
||||||
gets for free later.
|
|
||||||
- New `library/runtime/service.zig`: the harness — `run(callbacks)` owning the
|
|
||||||
replyWait loop, folding protocol messages, signals, and child-exit notifications
|
|
||||||
into `init` / `on_message` / `on_reload` / `on_terminate`; answers the common
|
|
||||||
`ping` automatically. Define the reserved `ping` request encoding here and
|
|
||||||
document it in ipc.md (one obvious encoding; smallest that cannot collide with
|
|
||||||
existing protocols).
|
|
||||||
- Convert one existing service (input-source or hpet) to the harness as proof it
|
|
||||||
subtracts code rather than adding it.
|
|
||||||
|
|
||||||
**Test:** extend `supervision` — a harness-built child: `sendSignal(reload)`
|
|
||||||
observed in its log, `ping` answered, `stop()` produces a clean exit with reason
|
|
||||||
`exited`; a second child that ignores signals (no bind) is killed by `stop()`'s
|
|
||||||
deadline with reason `killed`.
|
|
||||||
|
|
||||||
## M18.1 — device-manager protocol: hello + restart policy
|
|
||||||
|
|
||||||
- New `system/services/device-manager/device-manager-protocol.zig` module
|
|
||||||
(vfs-protocol pattern): `hello { version, role, device_id }`; version constant;
|
|
||||||
reserved fields.
|
|
||||||
- Device manager: register the `.device_manager` endpoint; spawn drivers with its
|
|
||||||
exit endpoint; enforce the hello deadline; restart policy — backoff, crash-loop
|
|
||||||
cap (three fast deaths → mark failed, log, stop), reasons from M17.2 deciding
|
|
||||||
restart vs not.
|
|
||||||
- usb-xhci-bus: adopt the harness + send hello. hpet/ps2-bus follow only if the
|
|
||||||
conversion is mechanical; otherwise they keep working unconverted (the manager
|
|
||||||
only enforces hello on drivers spawned with an assignment).
|
|
||||||
- build.zig: test-loop entry for the protocol module if it grows pure logic.
|
|
||||||
|
|
||||||
**Test:** new QEMU scenario `driver-restart` — the xHCI driver takes a test-only
|
|
||||||
argv flag to fault after hello on its first run; assert: fault, exit reason
|
|
||||||
recorded, manager respawns with backoff, second run claims the controller
|
|
||||||
(M17.1) and hellos clean. Assert the crash-loop cap by a driver that always
|
|
||||||
faults (a tiny test driver, not xhci).
|
|
||||||
|
|
||||||
## M18.2 — bus tree reports
|
|
||||||
|
|
||||||
- Protocol: `child_added { parent, identity, resources }` / `child_removed { id }`.
|
|
||||||
- usb-xhci-bus: bring-up to **port scan only** — map the MMIO window (claimed in
|
|
||||||
M16-era work), controller reset/start per xHCI spec, walk the port registers,
|
|
||||||
report one `child_added` per connected port with speed + port number as
|
|
||||||
identity. **No transfer rings, no descriptors** — reading device/interface
|
|
||||||
descriptors (and therefore USB class triples for matching) is the follow-on USB
|
|
||||||
track, not this plan.
|
|
||||||
- Device manager: mirror reports into its tree; prune the subtree (emitting
|
|
||||||
`child_removed`) when a bus driver dies; assert re-report on restart.
|
|
||||||
|
|
||||||
**Test:** QEMU already attaches usb-kbd + usb-mouse on xhci.0 — assert two
|
|
||||||
`child_added` events reach the manager and appear in its tree dump; kill the
|
|
||||||
driver, assert two `child_removed` then two fresh `child_added` after respawn.
|
|
||||||
|
|
||||||
## M18.3 — the application surface
|
|
||||||
|
|
||||||
- Protocol: `enumerate` (tree snapshot) + `subscribe` (published add/remove
|
|
||||||
events, input-service pattern).
|
|
||||||
- A small client (`device-list`, the `ps` analog) exercising both; the manager
|
|
||||||
becomes the one answer to "what devices exist" for user space.
|
|
||||||
`device_enumerate` stays for drivers/kernel seeding — its retreat is tied to the
|
|
||||||
discovery migration, out of this plan.
|
|
||||||
|
|
||||||
**Test:** QEMU scenario — `device-list` shows the tree including USB children;
|
|
||||||
during a driver restart the subscribing client logs remove + add events.
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
**Explicitly out of scope** (own tracks, after M18): discovery migration (pci-bus
|
|
||||||
driver, acpi service, retiring the kernel scan), USB control transfers +
|
|
||||||
descriptors + class-driver matching, the musl layer, `interrupt`/`quit` senders
|
|
||||||
(needs a console), job control.
|
|
||||||
@@ -1,196 +0,0 @@
|
|||||||
# M19–M20 execution plan: discovery migration
|
|
||||||
|
|
||||||
The operational plan for [device-manager.md](device-manager.md)'s increment 8:
|
|
||||||
discovery leaves the kernel — a **pci-bus driver** (M19) and an **acpi service**
|
|
||||||
(M20), with the kernel's device enumeration retired behind them. Same rules as
|
|
||||||
[m17-m18-plan.md](m17-m18-plan.md): one phase at a time, each green before the
|
|
||||||
next; this file is the build order and the checklist.
|
|
||||||
|
|
||||||
**Definition of green, every phase:** `zig build` clean, `zig build test` clean,
|
|
||||||
`python3 test/qemu_test.py` passes (existing scenarios plus the phase's new
|
|
||||||
one), and the relevant design doc updated. Commit per green phase (no co-author
|
|
||||||
trailers). The full suite is the regression net — the existing
|
|
||||||
`driver-restart` / `usb-report` / `device-list` / `input` scenarios must stay
|
|
||||||
green *through* the migration, which is the whole point: the system must not be
|
|
||||||
able to tell who enumerated it.
|
|
||||||
|
|
||||||
**Workflow:** dedicated worktree; branches off `main` — `feat/pci-bus`
|
|
||||||
(M19.0–19.3), `feat/acpi-service` (M20.1–20.3); auto-merge to main when a
|
|
||||||
branch is green; keep branches; push everything.
|
|
||||||
|
|
||||||
## Settled decisions (2026-07-13 — veto before the loop starts)
|
|
||||||
|
|
||||||
1. **What "retiring the kernel scan" means.** The kernel keeps, forever, the
|
|
||||||
parses it needs before user space exists: RSDP/XSDT location, MADT (SMP),
|
|
||||||
the HPET table (the tick), FADT + the AML `\_S5` evaluation (poweroff — the
|
|
||||||
power tests prove it), and MCFG (the host bridge node). What retires is
|
|
||||||
**device enumeration**: the ECAM function walk (M19.3) and the DSDT/SSDT
|
|
||||||
namespace walk that builds device nodes (M20.3). The AML module stays a
|
|
||||||
shared build module compiled into both the kernel (for `\_S5`) and the acpi
|
|
||||||
service (for everything else) — same source, two builds, no fork.
|
|
||||||
2. **Bridge apertures come from the firmware memory map, not AML.** Registered
|
|
||||||
PCI functions carry BAR resources, and containment demands the bridge own
|
|
||||||
windows that cover them. The apertures are derived kernel-side from the
|
|
||||||
boot memory map's MMIO holes (regions that are neither RAM nor tables) —
|
|
||||||
mechanical, AML-free, and available at boot regardless of what later moved
|
|
||||||
to user space. (The bridge today carries only ECAM + bus range; this is the
|
|
||||||
prerequisite M19.0 exists for.)
|
|
||||||
3. **`device_register` becomes idempotent on exact match.** A re-registration
|
|
||||||
with identical (parent, class, resources) returns the existing id instead
|
|
||||||
of appending. The kernel table has no unregister, so without this a
|
|
||||||
restarted registering bus would duplicate its children on every respawn —
|
|
||||||
idempotence makes restart-and-re-report safe for every future bus, not just
|
|
||||||
PCI.
|
|
||||||
4. **The manager matches from reports.** `ChildAdded` gains a `device_id`
|
|
||||||
field (the kernel-registered id, `no_device` for unregistered leaves like
|
|
||||||
USB ports). After the M19.3 flip, PCI driver matching keys off reported
|
|
||||||
identity (the class triple) instead of the manager's boot-time snapshot —
|
|
||||||
the snapshot match remains only for what the kernel still seeds. One flip
|
|
||||||
phase changes both sides at once so no device is ever matched twice.
|
|
||||||
5. **The acpi service's authority is one node.** The kernel publishes an
|
|
||||||
`acpi-tables` device: memory resources covering the table blobs plus a
|
|
||||||
broad `io_port` resource — the documented trust grant to exactly one
|
|
||||||
process (AML OperationRegions reach EC/PM ports; the claim-gated
|
|
||||||
io_read/io_write calls already exist). The service claims it, maps the
|
|
||||||
tables, and runs the shared AML module in ring 3 behind a `Hal` backed by
|
|
||||||
`mmio_map` + `io_read`/`io_write`.
|
|
||||||
6. **Both new processes are protocol drivers** under the manager: hello,
|
|
||||||
supervision, restart with backoff — all inherited from M18.1 for free.
|
|
||||||
Registration idempotence (decision 3) is what makes their restarts sound.
|
|
||||||
7. **Firmware neutrality is the contract** (2026-07-13). The generic layer is
|
|
||||||
everything at and above the device-manager protocol — descriptors,
|
|
||||||
containment, reports, matching, supervision — and none of it may become
|
|
||||||
x86-specific. Discovery is one swappable process per firmware: the acpi
|
|
||||||
service on x86; an **fdt service** on the Raspberry Pis (claims a
|
|
||||||
`devicetree-blob` node, reports children from the flattened device tree —
|
|
||||||
pure data, no bytecode, no port grant, strictly simpler than ACPI). The
|
|
||||||
manager owns the tree as *data* and touches no hardware, ever — AML runs in
|
|
||||||
a crashable, supervised discoverer precisely so a firmware-bytecode fault
|
|
||||||
can never take down the supervisor. Two consequences recorded now:
|
|
||||||
`DeviceDescriptor`'s 8-byte `hid` cannot hold an FDT `compatible` string
|
|
||||||
("brcm,bcm2835-aux-uart") — identity widens before the fdt service exists;
|
|
||||||
and cross-firmware surfaces are named by **domain, not firmware** (M21
|
|
||||||
defines a *power* protocol, not an "ACPI events" protocol — PSCI/mailbox
|
|
||||||
sources feed the same subscribers on ARM). **Landed early (2026-07-13):**
|
|
||||||
both services exist as placeholders (system/services/acpi, system/services/
|
|
||||||
fdt) and the build's `-Ddiscovery=acpi|fdt` option fills the ramdisk's
|
|
||||||
neutral `discovery` slot — the manager will spawn "discovery" by that name
|
|
||||||
in M20.3 and never learn which firmware it is on.
|
|
||||||
|
|
||||||
## Status
|
|
||||||
|
|
||||||
- [x] **M19.0** — prerequisites (bridge apertures from the memory map's
|
|
||||||
*gaps* — the single-hole rule died on OVMF's flash at the top of 4 GiB,
|
|
||||||
caught by the new every-BAR-contained assert in `discovery`; idempotent
|
|
||||||
`device_register` proven in `bus`; `ChildAdded.device_id`;
|
|
||||||
m17-m18-plan.md archived; suite 54/54).
|
|
||||||
- [x] **M19.1** — pci-bus driver, scan only (claims the bridge, maps ECAM
|
|
||||||
through its grant, brute-force walk with the multifunction rule; the
|
|
||||||
manager matches pci_host_bridge → pci-bus per device with the full
|
|
||||||
protocol contract; `pci-scan` builds its expected marker from the
|
|
||||||
kernel's own count — equivalence on the first run; suite 55/55).
|
|
||||||
- [x] **M19.2** — register + report (BAR probe mirrored byte-for-byte from the
|
|
||||||
kernel's addBars so dedupe returns the kernel's node ids during
|
|
||||||
coexistence; the bridge gained the io_port aperture I/O BARs need;
|
|
||||||
reports carry the registered device_id; pci-scan drills a forced restart
|
|
||||||
and asserts the PCI node count never grows — plus harness hardening: a
|
|
||||||
failing case now preserves its serial as <case>-failed-serial.log, and
|
|
||||||
the heavy scenarios run at 150s; suite 55/55).
|
|
||||||
- [x] **M19.3** — the flip: kernel `enumeratePci`/`addBars`/`PciHeader` all
|
|
||||||
deleted (bridge node stays); manager matches PCI drivers from reported
|
|
||||||
identity, deduped by registered id. Surfaced and fixed a real SMP race the
|
|
||||||
flip created — ring-3 device_register made the broker table concurrent, so
|
|
||||||
mmio_map's lock-free read intermittently tore hpet's resource length
|
|
||||||
(user fault) and overflowed `r.len-1` into a kernel panic; now the broker
|
|
||||||
read is under the big lock and the arithmetic is guarded, and pci-bus
|
|
||||||
skips size-0 BARs. discovery.md updated; suite 55/55 (driver-restart
|
|
||||||
hammered 6×).
|
|
||||||
- [x] **merge** `feat/pci-bus` → main, push (merged 2026-07-13).
|
|
||||||
- [x] **M20.1** — acpi service, parse only: the AML interpreter is now a build
|
|
||||||
module compiled into both kernel and service; the kernel publishes the
|
|
||||||
`acpi-tables` node (AML blobs as memory resources, the broad io_port grant,
|
|
||||||
the SCI); the service claims it, maps the blobs, runs the shared parser in
|
|
||||||
ring 3, and self-verifies its Device count against the kernel's (34 = 34,
|
|
||||||
deterministic via argv, no log-scraping); the manager spawns `discovery`
|
|
||||||
at startup. Parse-only touches no hardware. Suite 56/56.
|
|
||||||
|
|
||||||
- [x] **M20.2** — register + report: the service evaluates `_STA`/`_CRS` in
|
|
||||||
ring 3 (interpreter Hal = port I/O over the claimed node; a scratch page
|
|
||||||
backs SystemMemory maps so a stray region can't fault it) and registers +
|
|
||||||
reports each present `_HID` device under `acpi-tables`. Containment: the
|
|
||||||
broker's irq check became range-based (len-1 == the old equality) so the
|
|
||||||
node's broad irq window covers children's legacy lines; io ports fall in
|
|
||||||
the broad io grant. ChildAdded gained `hid`. Matching stays off. The
|
|
||||||
`acpi-report` scenario asserts the PS/2 keyboard (3 resources) and mouse
|
|
||||||
(1 resource) among the reports. Suite 57/57.
|
|
||||||
- [x] **M20.3** — the flip: the kernel's `wireAcpiDevices` call is gone (the
|
|
||||||
device-building helpers are retained-but-dead pending a focused sweep,
|
|
||||||
spawned as a task; static tables + `\_S5` + the acpi-tables node stay).
|
|
||||||
The manager matches ps2-bus from ACPI `_HID` reports; the service
|
|
||||||
registers all devices before reporting any (no keyboard-before-mouse
|
|
||||||
race). The `acpi-ps2` scenario proves report → spawn → ps2-bus attaches
|
|
||||||
its keyboard; `ioport` retargeted to the acpi-tables I/O window (the
|
|
||||||
kernel-built PS/2 node is gone). Suite 58/58.
|
|
||||||
- [x] **merge** `feat/acpi-service` → main, push (merged 2026-07-13) — **discovery migration complete**.
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Phase notes
|
|
||||||
|
|
||||||
**M19.0 apertures:** the boot memory map already crosses the handoff
|
|
||||||
([boot-handoff]), but discovery never sees it today — expect a small
|
|
||||||
pass-through (kernel init hands the map to the platform layer) before the
|
|
||||||
holes computation, which belongs where the bridge node is built
|
|
||||||
(`parseMcfg`). Sanity-check on QEMU q35: the xHCI BAR (`0xc0000000`-region
|
|
||||||
values seen in the M18 logs) must land inside a derived aperture, asserted in
|
|
||||||
the kernel unit test.
|
|
||||||
|
|
||||||
**M19.1 scanning without owning config access twice:** the driver reads config
|
|
||||||
space through its ECAM mmio_map grant of the *bridge* window — the same bytes
|
|
||||||
the kernel walk read. Vendor-id `0xFFFF` skip, header-type multifunction rule,
|
|
||||||
no bridge recursion (matches the kernel's current single-segment walk).
|
|
||||||
|
|
||||||
**M19.2 BAR sizing:** the classic size probe (write all-ones, read mask,
|
|
||||||
restore) is deferred — the BARs' current programmed values and types are
|
|
||||||
enough for containment-checked registration at bring-up; sizing lands with the
|
|
||||||
first driver that needs to *move* a BAR. Log what is registered so the
|
|
||||||
scenario can assert it.
|
|
||||||
|
|
||||||
**M19.3 what the manager still seeds from the snapshot:** everything the
|
|
||||||
kernel still enumerates (timers, ACPI nodes until M20.3). The PCI arm of
|
|
||||||
`pciDriverFor` switches source; `driverFor` doesn't move until M20.3.
|
|
||||||
|
|
||||||
**M20.1 spawn and identity (pre-settled 2026-07-13):** the manager spawns
|
|
||||||
`discovery` by its neutral ramdisk name at startup, as an ordinary protocol
|
|
||||||
driver (hello, supervision) — from M20.1 on, on every boot. For reporting ACPI
|
|
||||||
devices, `ChildAdded` gains `hid: [8]u8` (EISA ids fit; zero = none):
|
|
||||||
firmware *string* identity travels beside the numeric `identity` field until
|
|
||||||
the FDT-driven widening replaces both (decision 7).
|
|
||||||
|
|
||||||
**M20.1 Hal in ring 3:** `mapMmio` → `device.mmioMap` over the claimed
|
|
||||||
acpi-tables node (plus a table-offset map for blobs); `pioRead`/`pioWrite` →
|
|
||||||
`device.ioRead`/`ioWrite` against its io_port resource. The interpreter cannot
|
|
||||||
tell it moved — that is the assertion of `acpi-parse`.
|
|
||||||
|
|
||||||
**M20.2 containment for `_CRS`:** io ports fall inside the node's broad
|
|
||||||
io_port resource; MMIO windows (HPET, LAPIC ranges some firmwares list) fall
|
|
||||||
inside the memory-map holes added to the node in M20.1. Anything that doesn't
|
|
||||||
fit is logged and skipped, loudly — bring-up honesty over silent drops.
|
|
||||||
|
|
||||||
**M20.3 ps2 ordering:** ps2-bus binds nodes the acpi service now reports, so
|
|
||||||
its spawn moves behind the report (the manager's matching handles this once
|
|
||||||
the source flips); the `input` scenario proves the keyboard still types.
|
|
||||||
|
|
||||||
**Explicitly out of scope:** PCI bridge recursion (single segment, flat bus
|
|
||||||
walk stays); BAR reprogramming/sizing; disk/PCIe hotplug; interrupt routing
|
|
||||||
changes (`_PRT` stays wherever it is today); the USB descriptor track;
|
|
||||||
multi-segment ECAM; per-device power states (D-states, `_PSx`/`_PRx`,
|
|
||||||
suspend/resume — a future *lifecycle-vocabulary* extension, since "suspend"
|
|
||||||
has the shape of a signal every driver must answer, and it has no consumer
|
|
||||||
until laptop sleep); CPU P/C-states.
|
|
||||||
|
|
||||||
## M21 — ACPI events + system power — DONE
|
|
||||||
|
|
||||||
Built and merged (docs/m21-plan.md, 2026-07-13): the SCI + power button, Notify/GPE
|
|
||||||
dispatch, and orderly shutdown (init's stop cascade into a ring-3 S5 write).
|
|
||||||
See that plan for the phase record.
|
|
||||||
@@ -1,138 +0,0 @@
|
|||||||
# M21 execution plan: ACPI events + system power
|
|
||||||
|
|
||||||
The operational plan for the event side of the acpi service and orderly
|
|
||||||
shutdown — the capstone [m19-m20-plan.md](m19-m20-plan.md) previewed. Same
|
|
||||||
rules as its predecessors: one phase at a time, each green before the next;
|
|
||||||
this file is the build order and the checklist.
|
|
||||||
|
|
||||||
**Definition of green, every phase:** `zig build` clean, `zig build test`
|
|
||||||
clean, `python3 test/qemu_test.py` passes (existing scenarios plus the
|
|
||||||
phase's new one), and the relevant design doc updated. Commit per green phase
|
|
||||||
(no co-author trailers). Failing cases preserve their serial logs
|
|
||||||
(`<case>-failed-serial.log`).
|
|
||||||
|
|
||||||
**Workflow:** dedicated worktree; branch `feat/power-events` off `main`;
|
|
||||||
auto-merge to main when the branch is green; keep the branch; push everything.
|
|
||||||
|
|
||||||
## Settled decisions (2026-07-13, approved)
|
|
||||||
|
|
||||||
1. **S5 is executed by the acpi service from ring 3.** No new syscall: the
|
|
||||||
broad port grant (M20 decision 5) already made this physically possible —
|
|
||||||
the service holds the PM1 control ports in its io grant and derives `_S5`
|
|
||||||
from its own namespace (`aml.sleepState`). Formalizing it adds no
|
|
||||||
authority. The kernel keeps `power.zig` for its own test paths and
|
|
||||||
panic-time use.
|
|
||||||
2. **The power surface is domain-named** (decision 7 of the last plan): a
|
|
||||||
`power-protocol` module + `ServiceId.power = 5`, registered by the acpi
|
|
||||||
service — on ARM, a PSCI/mailbox service registers the same id and
|
|
||||||
subscribers never know the difference. Messages: `subscribe` (endpoint as
|
|
||||||
the call's capability, the input/manager pattern), `shutdown` (accepted
|
|
||||||
only from PID 1 — init), and events published as buffered messages:
|
|
||||||
`power_button`, `lid`, `ac`, `battery`, generic `notify` with a code.
|
|
||||||
3. **The service learns event ports from its own FADT copy**: the kernel adds
|
|
||||||
the FADT as one more memory resource on the acpi-tables node; the service
|
|
||||||
tells it apart from the AML blobs by signature ("FACP" header — the blob
|
|
||||||
resources are header-stripped bytecode and start with no signature). The
|
|
||||||
kernel's own FADT parse is untouched.
|
|
||||||
4. **The acpi service converts to the harness** (`runtime.service.run`):
|
|
||||||
protocol messages (subscribe/shutdown), the SCI notification, and the
|
|
||||||
existing report flow fold into one loop — the shape it was always meant
|
|
||||||
to have.
|
|
||||||
5. **GPE/Notify correctness is proven by host unit tests** (synthetic AML
|
|
||||||
with a Notify inside a method body; aml.zig joins the `zig build test`
|
|
||||||
loop). The QEMU scenario proves the power button — a *fixed* event,
|
|
||||||
deterministically injectable via QMP `system_powerdown` — because QEMU
|
|
||||||
cannot raise GPEs deterministically on this config. Battery/AC/lid and the
|
|
||||||
embedded controller (`_Qxx`) are interface-complete here and validated on
|
|
||||||
real hardware (the laptop) later.
|
|
||||||
|
|
||||||
## Ground truth the phases build on (verified 2026-07-13)
|
|
||||||
|
|
||||||
- `system/devices/power.zig` `shutdown()` is the kernel's S5 write
|
|
||||||
(SLP_TYP|SLP_EN to PM1a/PM1b control); there is no power syscall.
|
|
||||||
- init (`system/services/init/init.zig`) spawns vfs/input/device-manager
|
|
||||||
fire-and-forget — no child ids kept, no signals, no event loop. The whole
|
|
||||||
stop toolkit exists in `runtime.process` (stop/sendSignal/bindSignals).
|
|
||||||
- `test/qemu_test.py` has no QMP channel (serial is a one-way file).
|
|
||||||
- The kernel parses PM1 *control* blocks and SCI_INT from the FADT; the PM1
|
|
||||||
**event** blocks (offsets 56/60, len at 88) and **GPE0/GPE1** blocks
|
|
||||||
(offsets 80/84, lens 92/93) are unparsed — the service reads them from its
|
|
||||||
FADT copy (decision 3).
|
|
||||||
- The acpi-tables node carries the SCI as its only `len == 1` irq resource
|
|
||||||
(the broad window is len 256) — that is how the service finds it to
|
|
||||||
`irqBind`.
|
|
||||||
- `notify_opcode = 0x86` exists in `system/devices/aml/opcodes.zig` but the
|
|
||||||
interpreter never handles it — a GPE `_Lxx` body containing Notify fails
|
|
||||||
evaluation today. Everything else a GPE handler needs (field access,
|
|
||||||
control flow, method calls) is proven by the ring-3 `_STA`/`_CRS` work.
|
|
||||||
- The dead-code sweep (spawned task) also edits `system/devices/acpi.zig`;
|
|
||||||
M21.0 checks whether it landed and rebases before touching that file.
|
|
||||||
|
|
||||||
## Status
|
|
||||||
|
|
||||||
- [x] **M21.0** — baseline (dead-code sweep confirmed landed on main — no
|
|
||||||
acpi.zig conflict; `feat/power-events` cut; QMP channel in the harness:
|
|
||||||
always-on unix socket, client with the capabilities handshake, per-case
|
|
||||||
`qmp_after` hook, and a hook-must-deliver pass gate that the smoke case
|
|
||||||
now proves with a harmless query-status; suite 58/58).
|
|
||||||
- [x] **M21.1** — SCI + the power button (kernel appends the FADT as an
|
|
||||||
acpi-tables memory resource, tagged by its "FACP" header; `power-protocol`
|
|
||||||
module + `ServiceId.power = 5`; the acpi service converted to
|
|
||||||
`runtime.service.run`, registers `.power`, reads PM1 event/control + GPE
|
|
||||||
ports from its FADT copy, enables ACPI mode if SCI_EN is clear, binds the
|
|
||||||
SCI (the len-1 irq), sets PWRBTN_EN; the SCI handler clears PM1_STS,
|
|
||||||
logs `power: button pressed`, publishes `power_button`, acks. Scenario
|
|
||||||
`power-button` injects a real `system_powerdown` via QMP; initial-ramdisk
|
|
||||||
timeout 30→60s for the service's added boot work; suite 59/59).
|
|
||||||
- [x] **M21.2** — Notify + GPE dispatch (interpreter handles `notify_opcode`
|
|
||||||
into a bounded queue, cleared per-evaluate, drained via
|
|
||||||
`takeNotifications`; the service walks GPE status/enable bytes, evaluates
|
|
||||||
`\_GPE._Lxx`/`_Exx` per active bit, maps notified nodes to events
|
|
||||||
(battery/ac/lid/generic), clears GPE_STS write-1, acks. EC `_Qxx` out.
|
|
||||||
Host unit test with hand-encoded AML proves the queue; aml.zig joined the
|
|
||||||
`zig build test` loop. QEMU raises no GPEs — suite is regression net,
|
|
||||||
59/59).
|
|
||||||
- [x] **M21.3** — orderly shutdown (init supervises its children on one
|
|
||||||
endpoint that also carries signals, power events, and a re-arming
|
|
||||||
heartbeat timer; on `power_button` or a `terminate` signal it logs
|
|
||||||
`init: shutting down`, runs `stop(child, 2000, endpoint)` in reverse
|
|
||||||
order, then requests `.power` shutdown; the acpi service honors shutdown
|
|
||||||
from a subscriber — init is the one subscriber, a soft gate that survives
|
|
||||||
testing where PID 1 isn't init — and writes SLP_TYP|SLP_EN from ring 3.
|
|
||||||
`orderly-shutdown` scenario proves button → shutting-down → S5 → QEMU
|
|
||||||
exit; suite 60/60).
|
|
||||||
- [x] **merge** `feat/power-events` → main, push, keep the branch (merged 2026-07-13) — **M21 complete**.
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Phase notes
|
|
||||||
|
|
||||||
**M21.0 QMP:** open the unix socket after Popen, complete the
|
|
||||||
`qmp_capabilities` handshake, then send the hook's command (for these
|
|
||||||
scenarios: `{"execute": "system_powerdown"}`). The socket is additive — no
|
|
||||||
existing case may notice it. Note e3fe3f3 recently reworked how the harness
|
|
||||||
boots; adapt to its current shape rather than the pre-rework description.
|
|
||||||
|
|
||||||
**M21.1 SCI details:** PM1_STS is at the event block base (write-1-to-clear);
|
|
||||||
PM1_EN at base + block_len/2; PWRBTN bit is 8 in both. If PM1b exists, mirror
|
|
||||||
reads/writes to both blocks. Enable ACPI mode only when SCI_EN (PM1 control
|
|
||||||
bit 0) is clear — OVMF boots may already have it set. The publish path reuses
|
|
||||||
the manager's subscriber table pattern (bounded, drop-on-failed-send).
|
|
||||||
|
|
||||||
**M21.2 GPE walk:** GPE0_STS bytes live at the GPE0 block base, GPE0_EN in
|
|
||||||
the block's upper half; for a set+enabled bit n, the handler method is
|
|
||||||
`_L%02X` (level) or `_E%02X` (edge) under `\_GPE`. Evaluate, drain the notify
|
|
||||||
queue, clear the status bit, ack. A missing handler method is clear-and-log,
|
|
||||||
not an error.
|
|
||||||
|
|
||||||
**M21.3 ordering:** init subscribes with retries — the acpi service registers
|
|
||||||
`.power` well after init starts. The stop sequence runs vfs last (other
|
|
||||||
services may flush through it). The S5 write mirrors `power.zig`'s
|
|
||||||
`sleepValue` (SLP_TYP bits [12:10], SLP_EN bit 13); if the write returns, log
|
|
||||||
`power: S5 write did not take` so the scenario fails loudly instead of
|
|
||||||
hanging.
|
|
||||||
|
|
||||||
**Explicitly out of scope:** the embedded controller and `_Qxx` queries,
|
|
||||||
battery `_BST`/`_BIF` evaluation beyond the interface stubs, lid/AC on QEMU
|
|
||||||
(no emulation), reboot over the power protocol, S3 sleep, per-device D-states
|
|
||||||
(a future lifecycle-vocabulary extension), thermal zones.
|
|
||||||
+128
@@ -0,0 +1,128 @@
|
|||||||
|
# The power service: events and shutdown
|
||||||
|
|
||||||
|
A laptop lid closes, a battery drains, someone presses the power button — and
|
||||||
|
several parts of the system might care: a session manager dims the screen, a
|
||||||
|
logger notes it, and ultimately *something* has to turn the machine off. None of
|
||||||
|
them owns the hardware that reported the event, and the reporter should not know
|
||||||
|
who is listening. So system power is a **service**: an event source **publishes**
|
||||||
|
button/lid/battery/AC events, interested processes **subscribe**, and one
|
||||||
|
privileged caller — init — can ask it to power the machine off. It is the same
|
||||||
|
publish/subscribe shape as the [input service](input.md), applied to power.
|
||||||
|
|
||||||
|
## Why a service, and why it is named for the domain, not the firmware
|
||||||
|
|
||||||
|
Where the events come from is firmware-specific — on x86 they ride the ACPI SCI
|
||||||
|
([acpi.md](acpi.md)); on a Raspberry Pi they would come from PSCI or a mailbox.
|
||||||
|
What subscribers want is not: *the lid closed* means the same thing regardless of
|
||||||
|
who noticed. So the surface is **domain-named**. There is a `power-protocol`
|
||||||
|
module and a well-known `ServiceId.power = 5`; on x86 the **acpi service**
|
||||||
|
registers it, and on ARM a PSCI/mailbox service will register the *same* id.
|
||||||
|
Subscribers call `runtime.ipc.lookup(.power)` and never learn which firmware they
|
||||||
|
are on — the neutrality the whole [discovery](discovery.md) migration exists to
|
||||||
|
preserve, carried one layer up into a running-system surface.
|
||||||
|
|
||||||
|
This is why the protocol is `power`, not "ACPI events": naming a cross-firmware
|
||||||
|
surface after one firmware would leak x86 into code the ARM port must reuse
|
||||||
|
unchanged.
|
||||||
|
|
||||||
|
## The protocol
|
||||||
|
|
||||||
|
The `power-protocol` module ([system/services/power/protocol.zig](../system/services/power/protocol.zig))
|
||||||
|
follows the vfs-protocol pattern — extern-struct messages, a version, reserved
|
||||||
|
fields. Three operations:
|
||||||
|
|
||||||
|
| Direction | Operation | Purpose |
|
||||||
|
|---|---|---|
|
||||||
|
| subscriber → service | `subscribe` | receive published events; the subscriber's endpoint rides as the call's **capability** (the input/device-manager pattern) |
|
||||||
|
| init → service | `shutdown` | orderly shutdown's last step: enter S5 (soft off) |
|
||||||
|
| service → subscriber | `event` | a published `EventMessage`, delivered as a buffered message (never sent *to* the service) |
|
||||||
|
|
||||||
|
Events are published, not polled: like the input service, the service holds
|
||||||
|
subscriber endpoints as capabilities and `ipc_send`s each event as a buffered
|
||||||
|
message, so a slow or dead subscriber can never wedge the source. The event
|
||||||
|
vocabulary is hardware-neutral:
|
||||||
|
|
||||||
|
- `power_button` — the button was pressed (a fixed ACPI event on x86).
|
||||||
|
- `lid`, `ac`, `battery` — the named GPE-driven events.
|
||||||
|
- `notify` — a device notification that maps to none of the above; its `code`
|
||||||
|
(the ACPI `Notify` argument) and the notifying device's `hid` say which device
|
||||||
|
and what happened.
|
||||||
|
|
||||||
|
An `EventMessage` carries the `event` tag plus `code` and an 8-byte `hid`, so a
|
||||||
|
generic `notify` is fully described without a second round trip.
|
||||||
|
|
||||||
|
**`shutdown` is authority, not information.** It is the only operation that
|
||||||
|
*does* something irreversible, so it is gated: the contract is that only init
|
||||||
|
(PID 1) may request it, because init is the process that has already run the stop
|
||||||
|
sequence over everything else. The acpi service implements this as a **soft
|
||||||
|
gate** — it honors `shutdown` only from a process that is a *subscriber*, and
|
||||||
|
init is the one subscriber. That stands in for "only the system supervisor may
|
||||||
|
power off" without hard-coding a pid, so it still holds under tests where PID 1
|
||||||
|
is not init.
|
||||||
|
|
||||||
|
## Orderly shutdown
|
||||||
|
|
||||||
|
Powering off cleanly is where the power service, the [process
|
||||||
|
lifecycle](process-lifecycle.md), and [ACPI events](acpi.md) compose. init
|
||||||
|
already supervises the services it starts; for shutdown it runs **one event loop
|
||||||
|
over one endpoint** that carries three things at once: its children's exit
|
||||||
|
notifications, the lifecycle **signals** it can receive (`terminate`), and the
|
||||||
|
**power events** it subscribes to — plus a re-arming heartbeat timer proving PID
|
||||||
|
1 is alive. (init subscribes with retries, because the power service registers
|
||||||
|
`.power` well after init starts; a missing power service is not fatal — a
|
||||||
|
`terminate` signal drives the same path.)
|
||||||
|
|
||||||
|
On a `power_button` event or a `terminate` signal, init:
|
||||||
|
|
||||||
|
1. logs that it is shutting down,
|
||||||
|
2. runs the standard stop sequence — `runtime.process.stop(child, deadline,
|
||||||
|
endpoint)` — over its children **in reverse spawn order**, so the VFS stops
|
||||||
|
last (other services may flush through it), each child getting the
|
||||||
|
*terminate → deadline → kill* escalation from
|
||||||
|
[process-lifecycle.md](process-lifecycle.md), and
|
||||||
|
3. requests `.power` `shutdown`.
|
||||||
|
|
||||||
|
The service then enters **S5** (soft off) by writing `SLP_TYP | SLP_EN` to the
|
||||||
|
PM1 control register(s) from ring 3, mirroring the kernel's own
|
||||||
|
`system/devices/power.zig` `sleepValue`. If the write returns instead of powering
|
||||||
|
the machine off, it logs loudly so a test fails rather than hangs.
|
||||||
|
|
||||||
|
**No new system call was needed for S5.** The broad io_port grant on the
|
||||||
|
`acpi-tables` node ([discovery.md](discovery.md)) already put the PM1 control
|
||||||
|
ports in the acpi service's hands, so writing S5 from ring 3 is something it
|
||||||
|
could physically already do; formalizing it as a protocol operation added a
|
||||||
|
contract, not authority. The kernel keeps `power.zig` for its own test paths and
|
||||||
|
panic-time poweroff, where no user space is available to ask.
|
||||||
|
|
||||||
|
## Verifying it
|
||||||
|
|
||||||
|
Two QEMU scenarios exercise the path, both injecting a real ACPI power-button
|
||||||
|
press via QMP `system_powerdown` (there is no other deterministic power event on
|
||||||
|
this config):
|
||||||
|
|
||||||
|
- `power-button` proves the source: the acpi service's SCI handler logs the
|
||||||
|
press and publishes `power_button` (the ACPI half is in [acpi.md](acpi.md)).
|
||||||
|
- `orderly-shutdown` proves the whole composition: button → init logs shutting
|
||||||
|
down → children stopped → the service enters S5 → QEMU exits. The ordered
|
||||||
|
regex is the proof, and QEMU's self-exit through S5 is the pass.
|
||||||
|
|
||||||
|
## Scope
|
||||||
|
|
||||||
|
Interface-complete but validated on real hardware (the author's laptop) later,
|
||||||
|
because QEMU does not emulate them: battery `_BST`/`_BIF` evaluation beyond the
|
||||||
|
interface stubs, lid and AC events, and the embedded controller's `_Qxx`
|
||||||
|
queries. Deliberately out of scope for now: reboot over the power protocol, S3
|
||||||
|
sleep, per-device D-states (a future lifecycle-vocabulary extension, since
|
||||||
|
"suspend" has the shape of a signal every driver must answer and has no consumer
|
||||||
|
until laptop sleep), and thermal zones.
|
||||||
|
|
||||||
|
## See also
|
||||||
|
|
||||||
|
- [acpi.md](acpi.md) — where the events come from on x86: the SCI, the power
|
||||||
|
button fixed event, and GPE/Notify dispatch in the acpi service.
|
||||||
|
- [discovery.md](discovery.md) — why the surface is domain-named, and the
|
||||||
|
firmware neutrality that makes a PSCI backend drop-in on ARM.
|
||||||
|
- [process-lifecycle.md](process-lifecycle.md) — the stop sequence
|
||||||
|
(`terminate → deadline → kill`) and signals init composes into shutdown.
|
||||||
|
- [device-manager.md](device-manager.md) — the supervision model init mirrors for
|
||||||
|
its own children.
|
||||||
+1
-1
@@ -11,7 +11,7 @@ because the kernel releases a dead process's claims. The `driver-restart` and
|
|||||||
`usb-report` scenarios prove kill → release → respawn → re-claim → re-report
|
`usb-report` scenarios prove kill → release → respawn → re-claim → re-report
|
||||||
end to end. What remains of this document's ladder is scope, not mechanism:
|
end to end. What remains of this document's ladder is scope, not mechanism:
|
||||||
more of the system moved into restartable processes (the discovery migration,
|
more of the system moved into restartable processes (the discovery migration,
|
||||||
[m19-m20-plan.md](m19-m20-plan.md), is the next rung). This is the property danos is really chasing:
|
[discovery.md](discovery.md), is the next rung). This is the property danos is really chasing:
|
||||||
**if a part of the OS breaks, isolate it, and re-initialise it — without rebooting.**
|
**if a part of the OS breaks, isolate it, and re-initialise it — without rebooting.**
|
||||||
A crashed driver gets restarted; a wedged service gets killed and brought back. It's
|
A crashed driver gets restarted; a wedged service gets killed and brought back. It's
|
||||||
the reason the [microkernel](vision.md) shape was chosen, and it's a *separate* goal
|
the reason the [microkernel](vision.md) shape was chosen, and it's a *separate* goal
|
||||||
|
|||||||
+117
@@ -0,0 +1,117 @@
|
|||||||
|
# Timers and time
|
||||||
|
|
||||||
|
Two different needs hide under the word "timer", and danos keeps them apart:
|
||||||
|
|
||||||
|
- **Reading the clock** — *what time is it?* A read of a free-running counter.
|
||||||
|
- **Waiting** — *wake me in N milliseconds*, or *notify me when a deadline passes.*
|
||||||
|
|
||||||
|
Both are answered by the **kernel**, because the kernel already owns a timer: it has
|
||||||
|
to, to preempt tasks. The LAPIC heartbeat and the calibrated TSC that back all of this
|
||||||
|
are built in [device-interrupts.md](device-interrupts.md); the scheduler's blocking and
|
||||||
|
wait queues are in [scheduling.md](scheduling.md). This page is about the surface a
|
||||||
|
ring-3 program actually uses, and one deliberate absence: **there is no user-space time
|
||||||
|
service.**
|
||||||
|
|
||||||
|
## Why time is a syscall, not a service
|
||||||
|
|
||||||
|
The tempting microkernel move is to put a timer *driver* in user space and have
|
||||||
|
applications ask it for the time over IPC. For a **monotonic clock that is wrong** —
|
||||||
|
reading `now()` should never cost an IPC round trip. The kernel is already holding the
|
||||||
|
answer: it computes the current time every time it schedules, from the TSC, in a couple
|
||||||
|
of instructions. Surfacing that as a system call is pure mechanism; routing it through a
|
||||||
|
message to another process would be slower *and* redundant, and a device like the HPET
|
||||||
|
(uncacheable MMIO reads) is a particularly bad thing to read on every `now()`.
|
||||||
|
|
||||||
|
This is the same conclusion every serious system reaches: Linux and Zircon read the
|
||||||
|
counter in the vDSO, L4 exposes a clock field in a shared kernel page, seL4 reads the
|
||||||
|
cycle counter directly. None of them make a clock read an IPC. danos makes it a syscall.
|
||||||
|
|
||||||
|
That "from the TSC" hides a portability question, because the TSC is only a valid clock
|
||||||
|
when the CPU guarantees it is *invariant* and when every core's TSC is *synchronized*.
|
||||||
|
danos checks both — the invariant-TSC CPUID bit (`0x80000007` EDX[8], set on Intel and
|
||||||
|
AMD), and a cross-core "warp" check as the cores come up — and falls back to the HPET
|
||||||
|
counter when either fails. So `now()` stays accurate on a real Intel box, a real AMD box,
|
||||||
|
and inside a VM alike; only the source behind it differs. The mechanism is in
|
||||||
|
[device-interrupts.md](device-interrupts.md).
|
||||||
|
|
||||||
|
So the timer hardware lives in the kernel, and there is **no `hpet` driver and no time
|
||||||
|
server** to consume. (An earlier HPET driver existed only to *demonstrate* the driver
|
||||||
|
model; that role now lives in [drivers.md](drivers.md), as documentation.) The one place
|
||||||
|
a user-space time service *is* justified — **wall-clock / calendar time** — is discussed
|
||||||
|
at the end; it is deliberately not built yet.
|
||||||
|
|
||||||
|
## The three system calls
|
||||||
|
|
||||||
|
Time and waiting are three entries in the small syscall table ([syscall.md](syscall.md)):
|
||||||
|
|
||||||
|
- **`clock` (#23)** → monotonic nanoseconds since boot. It only moves forward. Not
|
||||||
|
wall-clock: no date, no timezone. Backed by `architecture.nanos()` (TSC, scaled with a
|
||||||
|
128-bit intermediate so a long uptime can't overflow) — a few nanoseconds of
|
||||||
|
resolution, and just an `rdtsc` plus a multiply.
|
||||||
|
- **`sleep` (#3)** → block the caller for N milliseconds. The scheduler records a wake
|
||||||
|
deadline and the tick sweep wakes it (`scheduler.sleep`).
|
||||||
|
- **`timer_bind` (#31)** → arm a one-shot timer that, after N milliseconds, posts a
|
||||||
|
**timer notification** to an IPC endpoint. Unlike `sleep` it does **not** block: a
|
||||||
|
service can keep answering messages on the same endpoint while a deadline is pending.
|
||||||
|
This is the timed wait that stop-sequence escalation, hello deadlines, and restart
|
||||||
|
backoff are built from ([process-lifecycle.md](process-lifecycle.md),
|
||||||
|
[device-manager.md](device-manager.md)).
|
||||||
|
|
||||||
|
The kernel's own scheduling timer (the LAPIC, vector 32) is never exposed to user space;
|
||||||
|
programs read the TSC through `clock` and get timed wakeups through `sleep`/`timer_bind`,
|
||||||
|
both riding the scheduler tick.
|
||||||
|
|
||||||
|
## `runtime.time` — the generic interface
|
||||||
|
|
||||||
|
Applications don't call the syscalls directly; they use `runtime.time`
|
||||||
|
(`library/runtime/time.zig`), a thin `Instant`/`Duration` layer over them — an ergonomic
|
||||||
|
front door, not new mechanism.
|
||||||
|
|
||||||
|
```zig
|
||||||
|
const time = @import("runtime").time;
|
||||||
|
|
||||||
|
const start = time.now(); // Instant — monotonic
|
||||||
|
doWork();
|
||||||
|
const took = start.elapsed(); // Duration
|
||||||
|
time.sleep(time.Duration.fromMillis(5)); // block ~5 ms
|
||||||
|
|
||||||
|
// A deadline delivered as a notification, so a service keeps serving meanwhile:
|
||||||
|
_ = time.after(endpoint, time.Duration.fromMillis(200));
|
||||||
|
```
|
||||||
|
|
||||||
|
- `Duration` is nanoseconds under the hood, with `fromNanos/fromMicros/fromMillis/
|
||||||
|
fromSeconds` and `asNanos/asMillis`. `ceilMillis` rounds *up* to the kernel's
|
||||||
|
millisecond granularity, so a sub-millisecond `sleep` never rounds down to zero and
|
||||||
|
returns early. All arithmetic saturates rather than wraps.
|
||||||
|
- `Instant` is a point on the monotonic clock: `since`, `elapsed`, `plus`, `reached` —
|
||||||
|
built for deadline loops (`while (!deadline.reached()) …`).
|
||||||
|
- `now()` / `monotonicNanos()` wrap `clock`. `available()` reports whether the clock is
|
||||||
|
calibrated at all (the kernel returns 0 until the TSC frequency is known, so a caller
|
||||||
|
that needs real time can treat 0 as "unavailable" rather than assume it advances).
|
||||||
|
- `sleep(d)` wraps `sleep`; `spin(d)` busy-polls `now()` for the sub-millisecond delays
|
||||||
|
the millisecond tick can't express; `after(endpoint, d)` wraps `timer_bind`.
|
||||||
|
|
||||||
|
The raw wrappers (`system.clock`, `system.sleep`, `system.timerOnce`) stay in
|
||||||
|
`library/runtime/system.zig`; `runtime.time` is the layer meant for everyday use.
|
||||||
|
|
||||||
|
## Wall-clock time (not built)
|
||||||
|
|
||||||
|
Everything above is **monotonic**: elapsed time since boot, perfect for timeouts and
|
||||||
|
measurement, useless for "what is the date?" Calendar time — a real-time clock, time
|
||||||
|
zones, leap seconds — is genuinely a **user-space** concern, and it *is* the case a time
|
||||||
|
service is for. It would be backed by an **RTC** driver (the CMOS real-time clock), not
|
||||||
|
the HPET, and exposed as a `CLOCK_REALTIME`-style service alongside the monotonic
|
||||||
|
syscall. It is deferred until something needs it; the monotonic clock the kernel already
|
||||||
|
owns covers every current use.
|
||||||
|
|
||||||
|
## Verifying it
|
||||||
|
|
||||||
|
`runtime.time`'s `Instant`/`Duration` arithmetic has unit tests that run on the host:
|
||||||
|
|
||||||
|
```
|
||||||
|
$ zig build test # includes library/runtime/time.zig
|
||||||
|
```
|
||||||
|
|
||||||
|
End to end, the proof the clock is real is that it *advances*: read `now()`, `sleep` a
|
||||||
|
`Duration`, read `now()` again, and the second reading is later — the kernel's timer
|
||||||
|
driving a ring-3 program with no service in between.
|
||||||
@@ -12,6 +12,9 @@
|
|||||||
//! (arguments arrive via `init`).
|
//! (arguments arrive via `init`).
|
||||||
|
|
||||||
pub const system = @import("system.zig");
|
pub const system = @import("system.zig");
|
||||||
|
/// Monotonic time, delays, and deadlines over the kernel clock/sleep/timer syscalls
|
||||||
|
/// — an `Instant`/`Duration` front door, no time service (docs/timers.md).
|
||||||
|
pub const time = @import("time.zig");
|
||||||
pub const heap = @import("heap.zig");
|
pub const heap = @import("heap.zig");
|
||||||
pub const ipc = @import("ipc.zig");
|
pub const ipc = @import("ipc.zig");
|
||||||
pub const start = @import("start.zig");
|
pub const start = @import("start.zig");
|
||||||
@@ -21,7 +24,7 @@ pub const vfs_protocol = @import("vfs-protocol");
|
|||||||
/// The device-manager protocol: hello + tree reports (docs/device-manager.md).
|
/// The device-manager protocol: hello + tree reports (docs/device-manager.md).
|
||||||
pub const device_manager_protocol = @import("device-manager-protocol");
|
pub const device_manager_protocol = @import("device-manager-protocol");
|
||||||
|
|
||||||
/// The power protocol: events (button, lid, battery) + shutdown (docs/m21-plan.md).
|
/// The power protocol: events (button, lid, battery) + shutdown (docs/power.md).
|
||||||
pub const power_protocol = @import("power-protocol");
|
pub const power_protocol = @import("power-protocol");
|
||||||
/// Keyboard-event listening (subscribe/next) and broadcasting (publish), over the input
|
/// Keyboard-event listening (subscribe/next) and broadcasting (publish), over the input
|
||||||
/// service. See library/runtime/input.zig and system/services/input/.
|
/// service. See library/runtime/input.zig and system/services/input/.
|
||||||
|
|||||||
@@ -0,0 +1,169 @@
|
|||||||
|
//! The danos time interface — monotonic time, delays, and deadlines for user space.
|
||||||
|
//!
|
||||||
|
//! There is no time *service*: the kernel already owns the scheduling timer and
|
||||||
|
//! surfaces it directly, so reading the clock is one system call (an `rdtsc` and a
|
||||||
|
//! scale), never an IPC round trip (docs/timers.md explains why). This module is a
|
||||||
|
//! thin, generic layer over the `clock`/`sleep`/`timer_bind` wrappers in `system.zig`
|
||||||
|
//! — an ergonomic `Instant`/`Duration` front door, not new mechanism.
|
||||||
|
//!
|
||||||
|
//! It is **monotonic** time only: nanoseconds since boot, moving forward, no date or
|
||||||
|
//! timezone. Wall-clock/calendar time is a separate user-space service (an RTC-backed
|
||||||
|
//! CLOCK_REALTIME) layered on top later.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const system = @import("system.zig");
|
||||||
|
|
||||||
|
const nanos_per_micro: u64 = 1_000;
|
||||||
|
const nanos_per_milli: u64 = 1_000_000;
|
||||||
|
const nanos_per_second: u64 = 1_000_000_000;
|
||||||
|
|
||||||
|
/// A span of time, held as nanoseconds. Constructors name their unit; accessors
|
||||||
|
/// truncate toward zero. `ceilMillis` rounds *up*, since `sleep`/`after` land on the
|
||||||
|
/// kernel's millisecond granularity and rounding down could return early.
|
||||||
|
pub const Duration = struct {
|
||||||
|
ns: u64,
|
||||||
|
|
||||||
|
pub fn fromNanos(n: u64) Duration {
|
||||||
|
return .{ .ns = n };
|
||||||
|
}
|
||||||
|
pub fn fromMicros(n: u64) Duration {
|
||||||
|
return .{ .ns = n *| nanos_per_micro };
|
||||||
|
}
|
||||||
|
pub fn fromMillis(n: u64) Duration {
|
||||||
|
return .{ .ns = n *| nanos_per_milli };
|
||||||
|
}
|
||||||
|
pub fn fromSeconds(n: u64) Duration {
|
||||||
|
return .{ .ns = n *| nanos_per_second };
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn asNanos(d: Duration) u64 {
|
||||||
|
return d.ns;
|
||||||
|
}
|
||||||
|
pub fn asMicros(d: Duration) u64 {
|
||||||
|
return d.ns / nanos_per_micro;
|
||||||
|
}
|
||||||
|
pub fn asMillis(d: Duration) u64 {
|
||||||
|
return d.ns / nanos_per_milli;
|
||||||
|
}
|
||||||
|
pub fn asSeconds(d: Duration) u64 {
|
||||||
|
return d.ns / nanos_per_second;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whole milliseconds, rounded up — the argument `sleep`/`after` pass the kernel.
|
||||||
|
/// A non-zero sub-millisecond duration becomes 1 ms rather than 0.
|
||||||
|
pub fn ceilMillis(d: Duration) u64 {
|
||||||
|
return (d.ns +| (nanos_per_milli - 1)) / nanos_per_milli;
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn plus(a: Duration, b: Duration) Duration {
|
||||||
|
return .{ .ns = a.ns +| b.ns };
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// A point on the monotonic clock — nanoseconds since boot. Compare and subtract
|
||||||
|
/// instants to measure elapsed time; it never runs backward, so `since` is safe to
|
||||||
|
/// saturate at zero rather than wrap.
|
||||||
|
pub const Instant = struct {
|
||||||
|
ns: u64,
|
||||||
|
|
||||||
|
/// The span from `earlier` to `self`, saturating at zero if `earlier` is later
|
||||||
|
/// (which the monotonic clock should never produce, but callers may pass any pair).
|
||||||
|
pub fn since(self: Instant, earlier: Instant) Duration {
|
||||||
|
return .{ .ns = self.ns -| earlier.ns };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// How long since this instant, sampled now.
|
||||||
|
pub fn elapsed(self: Instant) Duration {
|
||||||
|
return now().since(self);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// This instant advanced by `d` (a deadline, `d` from here).
|
||||||
|
pub fn plus(self: Instant, d: Duration) Instant {
|
||||||
|
return .{ .ns = self.ns +| d.ns };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether the monotonic clock has reached this instant (used as a deadline).
|
||||||
|
pub fn reached(deadline: Instant) bool {
|
||||||
|
return now().ns >= deadline.ns;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The current monotonic time.
|
||||||
|
pub fn now() Instant {
|
||||||
|
return .{ .ns = system.clock() };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Monotonic nanoseconds since boot — the raw `clock()` reading, for callers that
|
||||||
|
/// want a plain integer instead of an `Instant`.
|
||||||
|
pub fn monotonicNanos() u64 {
|
||||||
|
return system.clock();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether the monotonic clock is usable. The kernel returns 0 until the TSC is
|
||||||
|
/// calibrated (`tsc_hz == 0`); a caller that needs real time can treat that as
|
||||||
|
/// "unavailable" instead of assuming the clock advances.
|
||||||
|
pub fn available() bool {
|
||||||
|
return system.clock() != 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Block the caller for at least `d`, rounded up to the kernel's millisecond
|
||||||
|
/// granularity. For sub-millisecond precision the scheduler cannot express, use
|
||||||
|
/// `spin`.
|
||||||
|
pub fn sleep(d: Duration) void {
|
||||||
|
system.sleep(d.ceilMillis());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Block the caller for `ms` milliseconds — the coarse, allocation-free form.
|
||||||
|
pub fn sleepMillis(ms: u64) void {
|
||||||
|
system.sleep(ms);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Busy-wait until `d` has elapsed, polling the monotonic clock. This burns the CPU
|
||||||
|
/// on purpose, to hit sub-millisecond delays the scheduler's millisecond tick cannot.
|
||||||
|
/// Prefer `sleep` for anything at or above a millisecond.
|
||||||
|
pub fn spin(d: Duration) void {
|
||||||
|
const deadline = now().plus(d);
|
||||||
|
while (!deadline.reached()) {}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Arm a one-shot timer against `endpoint` (a handle from `ipc.createIpcEndpoint`):
|
||||||
|
/// after `d` the kernel posts a timer notification (`ipc.Received.isTimer`) there.
|
||||||
|
/// Unlike `sleep`, this does not block — a service can keep serving IPC on the same
|
||||||
|
/// endpoint while the deadline is pending. Rounds `d` up to milliseconds; returns
|
||||||
|
/// false if the timer could not be armed. See `system.timerOnce`.
|
||||||
|
pub fn after(endpoint: usize, d: Duration) bool {
|
||||||
|
return system.timerOnce(endpoint, d.ceilMillis());
|
||||||
|
}
|
||||||
|
|
||||||
|
test "Duration unit conversions round toward zero" {
|
||||||
|
try std.testing.expectEqual(@as(u64, 1_000_000_000), Duration.fromSeconds(1).asNanos());
|
||||||
|
try std.testing.expectEqual(@as(u64, 1_500), Duration.fromNanos(1_500).asNanos());
|
||||||
|
try std.testing.expectEqual(@as(u64, 2), Duration.fromMillis(2).asMillis());
|
||||||
|
try std.testing.expectEqual(@as(u64, 1), Duration.fromNanos(1_999_999).asMillis());
|
||||||
|
try std.testing.expectEqual(@as(u64, 250), Duration.fromMicros(250).asMicros());
|
||||||
|
}
|
||||||
|
|
||||||
|
test "ceilMillis rounds up, and never turns a nonzero span into zero" {
|
||||||
|
try std.testing.expectEqual(@as(u64, 0), Duration.fromNanos(0).ceilMillis());
|
||||||
|
try std.testing.expectEqual(@as(u64, 1), Duration.fromNanos(1).ceilMillis());
|
||||||
|
try std.testing.expectEqual(@as(u64, 1), Duration.fromMillis(1).ceilMillis());
|
||||||
|
try std.testing.expectEqual(@as(u64, 2), Duration.fromNanos(nanos_per_milli + 1).ceilMillis());
|
||||||
|
try std.testing.expectEqual(@as(u64, 5), Duration.fromMillis(5).ceilMillis());
|
||||||
|
}
|
||||||
|
|
||||||
|
test "Instant arithmetic: since saturates, plus/reached form deadlines" {
|
||||||
|
const t0 = Instant{ .ns = 1_000 };
|
||||||
|
const t1 = Instant{ .ns = 4_000 };
|
||||||
|
try std.testing.expectEqual(@as(u64, 3_000), t1.since(t0).asNanos());
|
||||||
|
// earlier-than-self can't happen on a monotonic clock; saturate rather than wrap.
|
||||||
|
try std.testing.expectEqual(@as(u64, 0), t0.since(t1).asNanos());
|
||||||
|
const deadline = t0.plus(Duration.fromNanos(2_500));
|
||||||
|
try std.testing.expectEqual(@as(u64, 3_500), deadline.ns);
|
||||||
|
}
|
||||||
|
|
||||||
|
test "saturating arithmetic does not overflow at the u64 ceiling" {
|
||||||
|
const big = Duration.fromSeconds(std.math.maxInt(u64));
|
||||||
|
try std.testing.expectEqual(@as(u64, std.math.maxInt(u64)), big.asNanos());
|
||||||
|
const late = Instant{ .ns = std.math.maxInt(u64) };
|
||||||
|
try std.testing.expectEqual(@as(u64, std.math.maxInt(u64)), late.plus(Duration.fromSeconds(10)).ns);
|
||||||
|
}
|
||||||
+1
-1
@@ -177,7 +177,7 @@ pub const ServiceId = enum(u32) {
|
|||||||
input = 2,
|
input = 2,
|
||||||
ps2_bus = 3, // the 8042 owner; child device drivers attach here for raw bytes
|
ps2_bus = 3, // the 8042 owner; child device drivers attach here for raw bytes
|
||||||
device_manager = 4, // the tree, the matcher, the supervisor (docs/device-manager.md)
|
device_manager = 4, // the tree, the matcher, the supervisor (docs/device-manager.md)
|
||||||
power = 5, // system power: events (button, lid, battery) + shutdown (docs/m21-plan.md; domain-named per decision 7 — the acpi service registers it on x86, a PSCI service will on ARM)
|
power = 5, // system power: events (button, lid, battery) + shutdown (docs/power.md; domain-named per docs/discovery.md — the acpi service registers it on x86, a PSCI service will on ARM)
|
||||||
_,
|
_,
|
||||||
};
|
};
|
||||||
|
|
||||||
|
|||||||
@@ -159,7 +159,7 @@ pub var dsdt_physical: u64 = 0;
|
|||||||
|
|
||||||
/// The FADT itself (physical + length), published on the acpi-tables node so
|
/// The FADT itself (physical + length), published on the acpi-tables node so
|
||||||
/// the ring-3 acpi service can read the PM1 event and GPE blocks it needs for
|
/// the ring-3 acpi service can read the PM1 event and GPE blocks it needs for
|
||||||
/// the event side (docs/m21-plan.md decision 3). Distinguished from the AML
|
/// the event side (docs/acpi.md — ACPI events). Distinguished from the AML
|
||||||
/// blob resources by its intact "FACP" header — the blobs are header-stripped.
|
/// blob resources by its intact "FACP" header — the blobs are header-stripped.
|
||||||
var fadt_physical: u64 = 0;
|
var fadt_physical: u64 = 0;
|
||||||
var fadt_length: u64 = 0;
|
var fadt_length: u64 = 0;
|
||||||
@@ -423,7 +423,7 @@ pub fn discover(rsdp_physical: u64, memory_regions: []const boot_handoff.MemoryR
|
|||||||
// whatever the FADT alone provided.
|
// whatever the FADT alone provided.
|
||||||
}
|
}
|
||||||
|
|
||||||
// Publish the acpi-tables node (docs/m19-m20-plan.md M20): the AML blobs as
|
// Publish the acpi-tables node (docs/discovery.md): the AML blobs as
|
||||||
// memory resources for the acpi service to map and parse in ring 3, a broad
|
// memory resources for the acpi service to map and parse in ring 3, a broad
|
||||||
// io_port grant for the OperationRegion access its interpreter needs, and
|
// io_port grant for the OperationRegion access its interpreter needs, and
|
||||||
// the SCI for the events track (M21). Exactly one node, one trusted
|
// the SCI for the events track (M21). Exactly one node, one trusted
|
||||||
@@ -448,7 +448,7 @@ fn publishAcpiTablesNode(device_tree: *DeviceTree) !void {
|
|||||||
// The broad I/O grant: OperationRegions name whatever ports the firmware
|
// The broad I/O grant: OperationRegions name whatever ports the firmware
|
||||||
// chose (EC, PM1, GPE, SMBus); which ports cannot be known before the AML
|
// chose (EC, PM1, GPE, SMBus); which ports cannot be known before the AML
|
||||||
// that names them is parsed, so the grant is the whole space — the honest
|
// that names them is parsed, so the grant is the whole space — the honest
|
||||||
// trust boundary of docs/m19-m20-plan.md decision 5.
|
// trust boundary of docs/discovery.md (the acpi service's one trusted node).
|
||||||
_ = node.addResource(.io_port, 0, 1 << 16);
|
_ = node.addResource(.io_port, 0, 1 << 16);
|
||||||
// A broad interrupt window: ACPI _CRS names legacy ISA IRQs (the PS/2 lines
|
// A broad interrupt window: ACPI _CRS names legacy ISA IRQs (the PS/2 lines
|
||||||
// 1 and 12, the RTC, …), and the service registers those devices under this
|
// 1 and 12, the RTC, …), and the service registers those devices under this
|
||||||
@@ -610,7 +610,7 @@ fn parseMcfg(device_tree: *DeviceTree, header: *const SystemDescriptorTableHeade
|
|||||||
var boot_memory_regions: []const boot_handoff.MemoryRegion = &.{};
|
var boot_memory_regions: []const boot_handoff.MemoryRegion = &.{};
|
||||||
|
|
||||||
/// The bridge's MMIO apertures, derived from the boot memory map's holes
|
/// The bridge's MMIO apertures, derived from the boot memory map's holes
|
||||||
/// (docs/m19-m20-plan.md decision 2): registered PCI functions carry BAR
|
/// (docs/discovery.md — apertures from the memory map): registered PCI functions carry BAR
|
||||||
/// resources, and `device_register` containment demands the bridge own windows
|
/// resources, and `device_register` containment demands the bridge own windows
|
||||||
/// that cover them. Everything the firmware described is "not hole"; the low
|
/// that cover them. Everything the firmware described is "not hole"; the low
|
||||||
/// aperture runs from the end of the described space below 4 GiB up to the
|
/// aperture runs from the end of the described space below 4 GiB up to the
|
||||||
|
|||||||
@@ -56,7 +56,7 @@ pub fn parse(allocator: std.mem.Allocator, blocks: []const []const u8) !ParseRes
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Count the Device objects in a parsed namespace — what the acpi service
|
/// Count the Device objects in a parsed namespace — what the acpi service
|
||||||
/// (docs/m19-m20-plan.md M20) reports, and what the kernel's own parse counts
|
/// (docs/discovery.md) reports, and what the kernel's own parse counts
|
||||||
/// so the two can be checked equal across the ring-3 move.
|
/// so the two can be checked equal across the ring-3 move.
|
||||||
pub fn deviceCount(namespace: *const Namespace) usize {
|
pub fn deviceCount(namespace: *const Namespace) usize {
|
||||||
return countKind(namespace.root, .device);
|
return countKind(namespace.root, .device);
|
||||||
|
|||||||
@@ -29,7 +29,7 @@ pub const DeviceClass = enum(u32) {
|
|||||||
/// hardware ID (`_HID`) and, where static, current resource settings (`_CRS`).
|
/// hardware ID (`_HID`) and, where static, current resource settings (`_CRS`).
|
||||||
acpi_device,
|
acpi_device,
|
||||||
/// The ACPI tables themselves, published as one node for the user-space acpi
|
/// The ACPI tables themselves, published as one node for the user-space acpi
|
||||||
/// service (docs/m19-m20-plan.md M20): memory resources over the AML blobs,
|
/// service (docs/discovery.md): memory resources over the AML blobs,
|
||||||
/// a broad io_port grant for OperationRegion access, and the SCI interrupt.
|
/// a broad io_port grant for OperationRegion access, and the SCI interrupt.
|
||||||
/// The one node whose claimant is trusted to run firmware bytecode.
|
/// The one node whose claimant is trusted to run firmware bytecode.
|
||||||
acpi_tables,
|
acpi_tables,
|
||||||
|
|||||||
@@ -1,211 +0,0 @@
|
|||||||
//! /system/drivers/bus — a user-space **bus driver**, and the smallest honest example of one.
|
|
||||||
//!
|
|
||||||
//! A bus driver owns a device that *contains other devices*, enumerates them by some
|
|
||||||
//! bus-specific protocol, and publishes each one into the kernel's device table so a
|
|
||||||
//! class driver can claim it. PCI walks configuration space; USB walks hub descriptors. Here
|
|
||||||
//! the "bus" is the HPET's register block and the "devices" are its comparators, each
|
|
||||||
//! a 0x20-byte window at 0x100 + 0x20*n that can be driven independently.
|
|
||||||
//!
|
|
||||||
//! It's a toy bus, but nothing about the mechanism is: `bus` reads how many children
|
|
||||||
//! exist from the hardware (GENERAL_CAP bits [12:8]), publishes one `DeviceDescriptor` per
|
|
||||||
//! child with a sub-window of its own MMIO plus the shared IRQ, and the kernel checks
|
|
||||||
//! every one of those resources is contained in what `bus` was granted. A comparator
|
|
||||||
//! driver then claims a child and maps only *its* registers — not the whole block.
|
|
||||||
//!
|
|
||||||
//! It also proves the negative: registering a child whose window escapes the parent's
|
|
||||||
//! is refused. Without that check, `device_register` would be a system_call for mapping
|
|
||||||
//! arbitrary physical memory.
|
|
||||||
|
|
||||||
const std = @import("std");
|
|
||||||
const runtime = @import("runtime");
|
|
||||||
const device = runtime.device;
|
|
||||||
|
|
||||||
const register_general_cap = 0x000;
|
|
||||||
|
|
||||||
/// Comparator n's registers: configuration+comparator+FSB route, 0x20 bytes.
|
|
||||||
fn timerWindow(hpet_base: u64, n: u64) device.ResourceDescriptor {
|
|
||||||
return .{
|
|
||||||
.kind = @intFromEnum(device.ResourceKind.memory),
|
|
||||||
.start = hpet_base + 0x100 + 0x20 * n,
|
|
||||||
.len = 0x20,
|
|
||||||
};
|
|
||||||
}
|
|
||||||
|
|
||||||
fn findHpet(buffer: []device.DeviceDescriptor) ?device.DeviceDescriptor {
|
|
||||||
const total = device.enumerate(buffer);
|
|
||||||
const n = @min(total, buffer.len);
|
|
||||||
for (buffer[0..n]) |d| {
|
|
||||||
if (d.class != @intFromEnum(device.DeviceClass.timer)) continue;
|
|
||||||
if (d.parent != device.no_parent) continue; // the block, not a comparator child
|
|
||||||
for (0..d.resource_count) |j| {
|
|
||||||
if (d.resources[j].kind == @intFromEnum(device.ResourceKind.memory)) return d;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return null;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// The parent's MMIO resource, and its IRQ if it has one.
|
|
||||||
fn resourcesOf(d: device.DeviceDescriptor) struct { mmio: device.ResourceDescriptor, irq: ?device.ResourceDescriptor } {
|
|
||||||
var mmio: device.ResourceDescriptor = undefined;
|
|
||||||
var irq: ?device.ResourceDescriptor = null;
|
|
||||||
for (0..d.resource_count) |j| {
|
|
||||||
const r = d.resources[j];
|
|
||||||
if (r.kind == @intFromEnum(device.ResourceKind.memory)) mmio = r;
|
|
||||||
if (r.kind == @intFromEnum(device.ResourceKind.irq)) irq = r;
|
|
||||||
}
|
|
||||||
return .{ .mmio = mmio, .irq = irq };
|
|
||||||
}
|
|
||||||
|
|
||||||
fn firstChildOf(buffer: []device.DeviceDescriptor, total: usize, parent_id: u64) ?u64 {
|
|
||||||
for (buffer[0..@min(total, buffer.len)]) |d| {
|
|
||||||
if (d.parent == parent_id) return d.id;
|
|
||||||
}
|
|
||||||
return null;
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn main() void {
|
|
||||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
|
||||||
_ = runtime.system.write("bus: out of memory\n");
|
|
||||||
return;
|
|
||||||
};
|
|
||||||
|
|
||||||
const parent = findHpet(buffer) orelse {
|
|
||||||
_ = runtime.system.write("bus: no HPET\n");
|
|
||||||
return;
|
|
||||||
};
|
|
||||||
const resource = resourcesOf(parent);
|
|
||||||
|
|
||||||
// Claim the bus. Everything below is subdivision of what this claim granted.
|
|
||||||
//
|
|
||||||
// Claims are exclusive, and at a normal boot the kernel spawns every initial_ramdisk
|
|
||||||
// binary — so hpet may own the HPET already. That's not an error, it's the
|
|
||||||
// capability model working: exit quietly and leave the device to its owner. The
|
|
||||||
// `bus` test spawns bus alone, so there it wins the claim.
|
|
||||||
if (!device.claim(parent.id)) {
|
|
||||||
_ = runtime.system.write("bus: HPET already claimed by another driver, nothing to do\n");
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
// Enumerate the bus: ask the hardware how many children it has.
|
|
||||||
const base = device.mmioMap(parent.id, 0) orelse {
|
|
||||||
_ = runtime.system.write("bus: mmio_map failed\n");
|
|
||||||
return;
|
|
||||||
};
|
|
||||||
const cap: *volatile u64 = @ptrFromInt(base + register_general_cap);
|
|
||||||
const n_children = ((cap.* >> 8) & 0x1F) + 1;
|
|
||||||
|
|
||||||
// Publish one child per comparator, each owning only its own window.
|
|
||||||
var published: u64 = 0;
|
|
||||||
var n: u64 = 0;
|
|
||||||
while (n < n_children) : (n += 1) {
|
|
||||||
var child = std.mem.zeroes(device.DeviceDescriptor);
|
|
||||||
child.class = @intFromEnum(device.DeviceClass.timer);
|
|
||||||
child.pci_class = device.no_pci_class;
|
|
||||||
child.hid_len = 6;
|
|
||||||
child.hid[0..6].* = "hpet-t".*;
|
|
||||||
child.resource_count = 1;
|
|
||||||
child.resources[0] = timerWindow(resource.mmio.start, n);
|
|
||||||
// Comparators share the block's interrupt line; only one child can bind it,
|
|
||||||
// but all of them may legitimately name it.
|
|
||||||
if (resource.irq) |i| {
|
|
||||||
child.resources[child.resource_count] = i;
|
|
||||||
child.resource_count += 1;
|
|
||||||
}
|
|
||||||
|
|
||||||
if (device.register(parent.id, &child) == null) {
|
|
||||||
_ = runtime.system.write("bus: register failed\n");
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
published += 1;
|
|
||||||
}
|
|
||||||
|
|
||||||
// The negative case. A window one byte past the end of the parent's must be
|
|
||||||
// refused — otherwise device_register would be "map any physical page you like".
|
|
||||||
// Confirm the table did not grow, not merely that the call returned null: null
|
|
||||||
// also means NoSpace/BadParent, so a size check is what actually proves the
|
|
||||||
// *containment* rule fired.
|
|
||||||
const before = device.enumerate(buffer);
|
|
||||||
var rogue = std.mem.zeroes(device.DeviceDescriptor);
|
|
||||||
rogue.class = @intFromEnum(device.DeviceClass.unknown);
|
|
||||||
rogue.resource_count = 1;
|
|
||||||
rogue.resources[0] = .{
|
|
||||||
.kind = @intFromEnum(device.ResourceKind.memory),
|
|
||||||
.start = resource.mmio.start + resource.mmio.len,
|
|
||||||
.len = 0x1000,
|
|
||||||
};
|
|
||||||
if (device.register(parent.id, &rogue) != null) {
|
|
||||||
_ = runtime.system.write("bus: FAIL out-of-window child was accepted\n");
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
if (device.enumerate(buffer) != before) {
|
|
||||||
_ = runtime.system.write("bus: FAIL rogue child leaked into the table\n");
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
// And confirm the children came back with the right parent and a *narrower*
|
|
||||||
// window than the bus — read from the table, not from our own memory.
|
|
||||||
const total = device.enumerate(buffer);
|
|
||||||
var seen: u64 = 0;
|
|
||||||
for (buffer[0..@min(total, buffer.len)]) |d| {
|
|
||||||
if (d.parent != parent.id) continue;
|
|
||||||
const w = d.resources[0];
|
|
||||||
if (w.start < resource.mmio.start or w.len >= resource.mmio.len) {
|
|
||||||
_ = runtime.system.write("bus: FAIL child window is not inside the bus\n");
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
seen += 1;
|
|
||||||
}
|
|
||||||
if (seen != published) {
|
|
||||||
_ = runtime.system.write("bus: FAIL child count mismatch\n");
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
// Delegation, end to end: claim a child and map *it*. A real class driver would be
|
|
||||||
// a different process; here bus plays both parts, which exercises the same path.
|
|
||||||
// The child's window is 0x20 bytes at parent+0x100, so the register it sees at
|
|
||||||
// offset 0 must be the same timer-0 configuration register the bus sees at 0x100.
|
|
||||||
//
|
|
||||||
// (mmio_map rounds to a page, so the child's mapping physically covers the whole
|
|
||||||
// 4 KiB the HPET lives in — the granularity limit documented in docs/drivers.md.
|
|
||||||
// The *resource* is narrow even though the page isn't.)
|
|
||||||
const child_id = firstChildOf(buffer, device.enumerate(buffer), parent.id) orelse {
|
|
||||||
_ = runtime.system.write("bus: FAIL no child to claim\n");
|
|
||||||
return;
|
|
||||||
};
|
|
||||||
if (!device.claim(child_id)) {
|
|
||||||
_ = runtime.system.write("bus: FAIL could not claim own child\n");
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
const child_base = device.mmioMap(child_id, 0) orelse {
|
|
||||||
_ = runtime.system.write("bus: FAIL child mmio_map refused\n");
|
|
||||||
return;
|
|
||||||
};
|
|
||||||
const via_child: *volatile u64 = @ptrFromInt(child_base);
|
|
||||||
const via_bus: *volatile u64 = @ptrFromInt(base + 0x100);
|
|
||||||
if (via_child.* != via_bus.*) {
|
|
||||||
_ = runtime.system.write("bus: FAIL child window does not alias the bus register\n");
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
// A descriptor pointer into an unmapped page must fail the call, not fault the
|
|
||||||
// kernel. Grab a page, free it, and register through the stale address: if the
|
|
||||||
// kernel dereferenced it raw (rather than copying in through the page tables) this
|
|
||||||
// would triple-fault QEMU and the test would time out instead of printing ok.
|
|
||||||
const scratch = runtime.system.mmap(0x1000, runtime.system.PROT_READ | runtime.system.PROT_WRITE);
|
|
||||||
if (!runtime.system.mmapFailed(scratch)) {
|
|
||||||
_ = runtime.system.munmap(scratch, 0x1000);
|
|
||||||
const descriptor: *const device.DeviceDescriptor = @ptrFromInt(scratch);
|
|
||||||
if (device.register(parent.id, descriptor) != null) {
|
|
||||||
_ = runtime.system.write("bus: FAIL register accepted an unmapped descriptor\n");
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
_ = runtime.system.write("bus: ok\n");
|
|
||||||
while (true) runtime.system.sleep(1000);
|
|
||||||
}
|
|
||||||
|
|
||||||
pub const panic = runtime.panic;
|
|
||||||
comptime {
|
|
||||||
_ = &runtime.start._start;
|
|
||||||
}
|
|
||||||
@@ -1,195 +0,0 @@
|
|||||||
//! /system/drivers/hpet — a user-space HPET driver. It proves the whole driver model end to
|
|
||||||
//! end: enumerate the device table, find the HPET, claim it, map its registers into
|
|
||||||
//! this ring-3 address space (strong-uncacheable), **bind its interrupt to an IPC
|
|
||||||
//! endpoint**, then sit blocked in `replyWait` until the hardware wakes it.
|
|
||||||
//!
|
|
||||||
//! Nothing here polls. Between interrupts the process is `.blocked` and off every
|
|
||||||
//! scheduler queue; the core runs other work or idles. That is the point of the
|
|
||||||
//! exercise — a driver is a process that sleeps until its device has something to
|
|
||||||
//! say (see docs/drivers.md).
|
|
||||||
//!
|
|
||||||
//! The comparator is configured **level-triggered** on purpose. Edge would be
|
|
||||||
//! simpler, but level is the discipline every real device line needs, and it forces
|
|
||||||
//! the full cycle to be correct:
|
|
||||||
//!
|
|
||||||
//! kernel ISR mask the GSI -> EOI -> notify this endpoint
|
|
||||||
//! hpet wake, clear GENERAL_INT_STATUS (deasserts the line), re-arm
|
|
||||||
//! hpet irq_ack -> kernel unmasks the GSI
|
|
||||||
//!
|
|
||||||
//! Clear the status bit *before* acking, or the line is still asserted when the
|
|
||||||
//! kernel unmasks and the I/O APIC redelivers forever.
|
|
||||||
//!
|
|
||||||
//! Register map (HPET spec 1.0a):
|
|
||||||
//! 0x000 GENERAL_CAP [63:32] fs per tick, [12:8] number timers - 1
|
|
||||||
//! 0x010 GENERAL_CONFIGURATION bit0 ENABLE_CNF, bit1 LEG_RT_CNF
|
|
||||||
//! 0x020 GENERAL_INT_STATUS bit n = timer n asserted (write 1 to clear)
|
|
||||||
//! 0x0F0 MAIN_COUNTER
|
|
||||||
//! 0x100 TIMER0_CONFIGURATION bit1 INT_TYPE(1=level) bit2 INT_ENB bit3 TYPE(periodic)
|
|
||||||
//! bits[13:9] INT_ROUTE, [63:32] INT_ROUTE_CAP
|
|
||||||
//! 0x108 TIMER0_COMPARATOR
|
|
||||||
|
|
||||||
const runtime = @import("runtime");
|
|
||||||
const mmio = @import("mmio");
|
|
||||||
const device = runtime.device;
|
|
||||||
const ipc = runtime.ipc;
|
|
||||||
|
|
||||||
const register_general_cap = 0x000;
|
|
||||||
const register_general_configuration = 0x010;
|
|
||||||
const register_int_status = 0x020;
|
|
||||||
const register_main_counter = 0x0F0;
|
|
||||||
const register_timer0_configuration = 0x100;
|
|
||||||
const register_timer0_comparator = 0x108;
|
|
||||||
|
|
||||||
const configuration_enable: u64 = 1 << 0; // GENERAL_CONFIGURATION.ENABLE_CNF
|
|
||||||
const configuration_leg_rt: u64 = 1 << 1; // GENERAL_CONFIGURATION.LEG_RT_CNF
|
|
||||||
const tn_int_type_level: u64 = 1 << 1;
|
|
||||||
const tn_int_enb: u64 = 1 << 2;
|
|
||||||
const tn_type_periodic: u64 = 1 << 3;
|
|
||||||
const tn_route_shift = 9;
|
|
||||||
const tn_route_mask: u64 = 0x1F << tn_route_shift;
|
|
||||||
|
|
||||||
/// Interrupts to observe before declaring victory.
|
|
||||||
const target_ticks = 5;
|
|
||||||
|
|
||||||
/// Read/write a 64-bit HPET register through the typed volatile MMIO layer (/lib/mmio).
|
|
||||||
/// The HPET is pure MMIO with no DMA, and on x86 its grant is strong-uncacheable (so
|
|
||||||
/// UC writes are already ordered) — no barriers are needed here; the point is the
|
|
||||||
/// typed, arch-portable access every driver should use.
|
|
||||||
inline fn rd(base: usize, off: usize) u64 {
|
|
||||||
return mmio.read(u64, base + off);
|
|
||||||
}
|
|
||||||
inline fn wr(base: usize, off: usize, value: u64) void {
|
|
||||||
mmio.write(u64, base + off, value);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// A timer-class device exposing both an MMIO window and an IRQ: its id, the two
|
|
||||||
/// resource indices, and the GSI discovery chose out of `Tn_INT_ROUTE_CAP`.
|
|
||||||
const Found = struct { device_id: u64, mmio: u64, irq: u64, gsi: u64 };
|
|
||||||
|
|
||||||
fn findHpet(buffer: []device.DeviceDescriptor) ?Found {
|
|
||||||
const total = device.enumerate(buffer);
|
|
||||||
const n = @min(total, buffer.len);
|
|
||||||
for (buffer[0..n]) |d| {
|
|
||||||
if (d.class != @intFromEnum(device.DeviceClass.timer)) continue;
|
|
||||||
// Skip comparator children a bus driver may have published below the block
|
|
||||||
// (see system/drivers/bus/bus.zig) — we want the register block itself.
|
|
||||||
if (d.parent != device.no_parent) continue;
|
|
||||||
var mmio_index: ?u64 = null;
|
|
||||||
var irq: ?u64 = null;
|
|
||||||
for (0..d.resource_count) |j| {
|
|
||||||
switch (d.resources[j].kind) {
|
|
||||||
@intFromEnum(device.ResourceKind.memory) => mmio_index = mmio_index orelse j,
|
|
||||||
@intFromEnum(device.ResourceKind.irq) => irq = irq orelse j,
|
|
||||||
else => {},
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (mmio_index) |m| if (irq) |i| {
|
|
||||||
return .{ .device_id = d.id, .mmio = m, .irq = i, .gsi = d.resources[i].start };
|
|
||||||
};
|
|
||||||
}
|
|
||||||
return null;
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn main() void {
|
|
||||||
// Enumerate into a heap buffer (too big for the one-page user stack).
|
|
||||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 32) catch {
|
|
||||||
_ = runtime.system.write("system/drivers/hpet: out of memory\n");
|
|
||||||
return;
|
|
||||||
};
|
|
||||||
|
|
||||||
const hpet = findHpet(buffer) orelse {
|
|
||||||
_ = runtime.system.write("system/drivers/hpet: no HPET with an IRQ\n");
|
|
||||||
return;
|
|
||||||
};
|
|
||||||
|
|
||||||
if (!device.claim(hpet.device_id)) {
|
|
||||||
_ = runtime.system.write("system/drivers/hpet: claim failed\n");
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
const base = device.mmioMap(hpet.device_id, hpet.mmio) orelse {
|
|
||||||
_ = runtime.system.write("system/drivers/hpet: mmio_map failed\n");
|
|
||||||
return;
|
|
||||||
};
|
|
||||||
|
|
||||||
// The GSI discovery picked for us out of Tn_INT_ROUTE_CAP. Program the comparator
|
|
||||||
// to raise exactly this line — the kernel will only bind the one it recorded.
|
|
||||||
const gsi = hpet.gsi;
|
|
||||||
|
|
||||||
const endpoint = ipc.createIpcEndpoint() orelse {
|
|
||||||
_ = runtime.system.write("system/drivers/hpet: create_ipc_endpoint failed\n");
|
|
||||||
return;
|
|
||||||
};
|
|
||||||
|
|
||||||
// --- program the hardware ------------------------------------------------
|
|
||||||
// Counter period, so we can arm the comparator a fixed wall-clock distance out.
|
|
||||||
const femtos_per_tick = rd(base, register_general_cap) >> 32;
|
|
||||||
if (femtos_per_tick == 0) {
|
|
||||||
_ = runtime.system.write("system/drivers/hpet: bad HPET period\n");
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
const ticks_per_ms = 1_000_000_000_000 / femtos_per_tick;
|
|
||||||
|
|
||||||
// Stop the counter and take the legacy route off while we reconfigure.
|
|
||||||
wr(base, register_general_configuration, rd(base, register_general_configuration) & ~(configuration_enable | configuration_leg_rt));
|
|
||||||
|
|
||||||
// Timer 0: one-shot, level-triggered, routed to our GSI, interrupt enabled.
|
|
||||||
// One-shot (not periodic) sidesteps the HPET's Tn_value_SET accumulator quirk —
|
|
||||||
// we simply re-arm from the driver on each interrupt, which is what a tickless
|
|
||||||
// timer driver does anyway.
|
|
||||||
var t0 = rd(base, register_timer0_configuration);
|
|
||||||
t0 &= ~(tn_route_mask | tn_type_periodic);
|
|
||||||
t0 |= tn_int_type_level | tn_int_enb | (gsi << tn_route_shift);
|
|
||||||
wr(base, register_timer0_configuration, t0);
|
|
||||||
|
|
||||||
// Clear any stale assertion, then arm ~100 ms out and start the counter.
|
|
||||||
wr(base, register_int_status, 1);
|
|
||||||
wr(base, register_timer0_comparator, rd(base, register_main_counter) + ticks_per_ms * 100);
|
|
||||||
wr(base, register_general_configuration, rd(base, register_general_configuration) | configuration_enable);
|
|
||||||
|
|
||||||
if (!device.irqBind(hpet.device_id, hpet.irq, endpoint)) {
|
|
||||||
_ = runtime.system.write("system/drivers/hpet: irq_bind failed\n");
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
_ = runtime.system.write("system/drivers/hpet: bound, sleeping until the hardware speaks\n");
|
|
||||||
|
|
||||||
// --- the driver loop -----------------------------------------------------
|
|
||||||
// Blocked in replyWait. No polling, no spinning: the next line of this function
|
|
||||||
// runs only because an interrupt fired.
|
|
||||||
var receive: [64]u8 = undefined;
|
|
||||||
var count: usize = 0;
|
|
||||||
while (count < target_ticks) {
|
|
||||||
// Blocked here. The task is `.blocked` and off every scheduler queue; the
|
|
||||||
// next line runs only because the HPET raised its line.
|
|
||||||
const r = ipc.replyWait(endpoint, &.{}, &receive, null);
|
|
||||||
if (!r.isNotification()) continue; // a client request, not our IRQ
|
|
||||||
|
|
||||||
// Quiet the device: write 1 to timer 0's status bit. Until this lands, the
|
|
||||||
// line is still asserted and unmasking would refire immediately.
|
|
||||||
wr(base, register_int_status, 1);
|
|
||||||
count += 1;
|
|
||||||
|
|
||||||
if (count < target_ticks) {
|
|
||||||
wr(base, register_timer0_comparator, rd(base, register_main_counter) + ticks_per_ms * 100);
|
|
||||||
} else {
|
|
||||||
// Last one: stop the source rather than re-arming, so the line is left
|
|
||||||
// both quiet *and* unmasked by the ack below. Re-arming here would leave
|
|
||||||
// a pending interrupt that nobody is waiting for, and the ISR would mask
|
|
||||||
// the line again a moment later.
|
|
||||||
wr(base, register_timer0_configuration, rd(base, register_timer0_configuration) & ~tn_int_enb);
|
|
||||||
}
|
|
||||||
|
|
||||||
_ = runtime.system.write("system/drivers/hpet: irq\n");
|
|
||||||
if (!device.irqAck(hpet.device_id, hpet.irq)) {
|
|
||||||
_ = runtime.system.write("system/drivers/hpet: irq_ack failed\n");
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
_ = runtime.system.write("system/drivers/hpet: ok\n");
|
|
||||||
while (true) runtime.system.sleep(1000);
|
|
||||||
}
|
|
||||||
|
|
||||||
pub const panic = runtime.panic;
|
|
||||||
comptime {
|
|
||||||
_ = &runtime.start._start;
|
|
||||||
}
|
|
||||||
@@ -1,5 +1,5 @@
|
|||||||
//! /system/drivers/pci-bus — the PCI bus driver: enumeration moved out of ring 0
|
//! /system/drivers/pci-bus — the PCI bus driver: enumeration moved out of ring 0
|
||||||
//! (docs/m19-m20-plan.md, M19). The device manager matches the `pci_host_bridge`
|
//! (docs/discovery.md). The device manager matches the `pci_host_bridge`
|
||||||
//! node and spawns one instance per bridge, the bridge's device id as argv[1] —
|
//! node and spawns one instance per bridge, the bridge's device id as argv[1] —
|
||||||
//! the same per-device contract as usb-xhci-bus.
|
//! the same per-device contract as usb-xhci-bus.
|
||||||
//!
|
//!
|
||||||
|
|||||||
@@ -80,6 +80,35 @@ var timer_hz: u32 = 0;
|
|||||||
var tsc_hz: u64 = 0;
|
var tsc_hz: u64 = 0;
|
||||||
var tsc_base: u64 = 0;
|
var tsc_base: u64 = 0;
|
||||||
|
|
||||||
|
/// Whether the TSC is architecturally **invariant** — a constant rate regardless of
|
||||||
|
/// P/C-state transitions, and thus valid as a clocksource (CPUID leaf 0x80000007,
|
||||||
|
/// EDX bit 8). AMD and modern Intel set it; the bare qemu64 model does not. Measured
|
||||||
|
/// frequency alone is not enough: a non-invariant TSC speeds up and slows down with
|
||||||
|
/// the core clock, so reading it as wall time would drift.
|
||||||
|
var tsc_invariant: bool = false;
|
||||||
|
/// Cleared if the cross-core warp check (checkWarpSource) ever sees the TSC read
|
||||||
|
/// lower on one core than the max another core has already published — i.e. the
|
||||||
|
/// per-core TSCs are not synchronized, and a task migrating cores could see time go
|
||||||
|
/// backward. Starts true (assume synchronized until proven otherwise).
|
||||||
|
var tsc_synced: bool = true;
|
||||||
|
/// The worst backward skew the warp check observed, in TSC cycles (0 = none).
|
||||||
|
var tsc_warp_cycles: u64 = 0;
|
||||||
|
|
||||||
|
/// The monotonic clock's source. The TSC when it is invariant *and* synchronized —
|
||||||
|
/// the fast `rdtsc` path taken on real Intel/AMD and modern VMs. Otherwise the HPET
|
||||||
|
/// main counter: a single fixed-rate counter, immune to both per-core skew and
|
||||||
|
/// frequency scaling, so it stays accurate on a bare VM or a warped machine.
|
||||||
|
const ClockSource = enum { tsc, hpet };
|
||||||
|
var clock_source: ClockSource = .tsc;
|
||||||
|
|
||||||
|
/// HPET standby clocksource, set up in calibrate() whenever an HPET exists (whether
|
||||||
|
/// or not calibration itself measured against it): its frequency, the counter value
|
||||||
|
/// chosen as the zero point, and its width mask. Only a 64-bit HPET is used as a
|
||||||
|
/// clocksource — a 32-bit one wraps too fast to be monotonic without accumulation.
|
||||||
|
var hpet_clock_hz: u64 = 0;
|
||||||
|
var hpet_clock_base: u64 = 0;
|
||||||
|
var hpet_clock_mask: u64 = ~@as(u64, 0);
|
||||||
|
|
||||||
/// Read the 64-bit Time Stamp Counter.
|
/// Read the 64-bit Time Stamp Counter.
|
||||||
fn rdtsc() u64 {
|
fn rdtsc() u64 {
|
||||||
var low: u32 = undefined;
|
var low: u32 = undefined;
|
||||||
@@ -220,6 +249,31 @@ pub fn calibrate() void {
|
|||||||
}
|
}
|
||||||
|
|
||||||
tsc_base = rdtsc(); // the clock's zero point (boot)
|
tsc_base = rdtsc(); // the clock's zero point (boot)
|
||||||
|
|
||||||
|
// Decide whether the TSC is trustworthy as a clocksource. Frequency (measured
|
||||||
|
// above, possibly against the HPET/PIT) is necessary but not sufficient: the TSC
|
||||||
|
// must also be *invariant* (CPUID 0x80000007 EDX[8]). AMD and modern Intel set
|
||||||
|
// this; the bare qemu64 model does not.
|
||||||
|
tsc_invariant = tscIsInvariant();
|
||||||
|
|
||||||
|
// Bring up the HPET as a standby clocksource whenever one exists — even on the
|
||||||
|
// CPUID-0x15 path where calibration never touched it — so a non-invariant TSC
|
||||||
|
// (here) or an unsynchronized one (checkWarpSource, during SMP bring-up) can fall
|
||||||
|
// back to a source that is immune to both. hpetHz() maps + enables the counter
|
||||||
|
// and is idempotent if calibration already used it.
|
||||||
|
if (configuration_hpet_base != 0) {
|
||||||
|
if (hpetHz()) |hz| {
|
||||||
|
hpet_clock_mask = hpetMask();
|
||||||
|
if (hpet_clock_mask == ~@as(u64, 0)) { // only a 64-bit HPET is monotonic enough
|
||||||
|
hpet_clock_hz = hz;
|
||||||
|
hpet_clock_base = readHpet();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Select the source: the fast TSC when invariant, else the HPET if we have one.
|
||||||
|
// (checkWarpSource may still demote TSC -> HPET later if the cores' TSCs skew.)
|
||||||
|
if (!tsc_invariant and hpet_clock_hz != 0) clock_source = .hpet;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Run the LAPIC timer one-shot from its maximum count while a monotonic reference
|
/// Run the LAPIC timer one-shot from its maximum count while a monotonic reference
|
||||||
@@ -275,7 +329,7 @@ fn calibratePit() void {
|
|||||||
// --- reference clocks ------------------------------------------------------
|
// --- reference clocks ------------------------------------------------------
|
||||||
|
|
||||||
/// TSC frequency from CPUID leaf 0x15 (crystal_hz * numerator / denominator), or
|
/// TSC frequency from CPUID leaf 0x15 (crystal_hz * numerator / denominator), or
|
||||||
/// null if the CPU doesn't enumerate it (common under QEMU).
|
/// null if the CPU doesn't enumerate it (common under QEMU, and on AMD).
|
||||||
fn cpuidTscHz() ?u64 {
|
fn cpuidTscHz() ?u64 {
|
||||||
if (cpuid(0).eax < 0x15) return null;
|
if (cpuid(0).eax < 0x15) return null;
|
||||||
const r = cpuid(0x15);
|
const r = cpuid(0x15);
|
||||||
@@ -283,6 +337,15 @@ fn cpuidTscHz() ?u64 {
|
|||||||
return @as(u64, r.ecx) * r.ebx / r.eax;
|
return @as(u64, r.ecx) * r.ebx / r.eax;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Whether the CPU advertises an **invariant** TSC (CPUID leaf 0x80000007, EDX
|
||||||
|
/// bit 8) — the architectural guarantee, on both Intel and AMD, that the TSC ticks
|
||||||
|
/// at a constant rate across P/C-states and never stops. Requires the extended-leaf
|
||||||
|
/// range to reach 0x80000007 first.
|
||||||
|
fn tscIsInvariant() bool {
|
||||||
|
if (cpuid(0x80000000).eax < 0x80000007) return false;
|
||||||
|
return (cpuid(0x80000007).edx & (1 << 8)) != 0;
|
||||||
|
}
|
||||||
|
|
||||||
const CpuidRegs = struct { eax: u32, ebx: u32, ecx: u32, edx: u32 };
|
const CpuidRegs = struct { eax: u32, ebx: u32, ecx: u32, edx: u32 };
|
||||||
|
|
||||||
fn cpuid(leaf: u32) CpuidRegs {
|
fn cpuid(leaf: u32) CpuidRegs {
|
||||||
@@ -310,11 +373,20 @@ fn hpetWrite64(off: usize, value: u64) void {
|
|||||||
@as(*volatile u64, @ptrFromInt(configuration_hpet_base + off)).* = value;
|
@as(*volatile u64, @ptrFromInt(configuration_hpet_base + off)).* = value;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Whether the HPET has been mapped into the physmap yet, so `configuration_hpet_base`
|
||||||
|
/// already holds the virtual address. `hpetHz` is called more than once (calibration
|
||||||
|
/// may use the HPET, and the standby-clocksource setup asks for it again), and mapping
|
||||||
|
/// an already-mapped base a second time would double-offset it into an overflow.
|
||||||
|
var hpet_mapped: bool = false;
|
||||||
|
|
||||||
/// Map + enable the HPET and return its tick frequency, or null if unusable.
|
/// Map + enable the HPET and return its tick frequency, or null if unusable.
|
||||||
/// Maps the HPET into the physmap and switches configuration_hpet_base to that virtual
|
/// Maps the HPET into the physmap and switches configuration_hpet_base to that virtual
|
||||||
/// address, so the register accessors reach it without the identity map.
|
/// address, so the register accessors reach it without the identity map. Idempotent.
|
||||||
fn hpetHz() ?u64 {
|
fn hpetHz() ?u64 {
|
||||||
configuration_hpet_base = paging.mapMmio(configuration_hpet_base, 0x400, true);
|
if (!hpet_mapped) {
|
||||||
|
configuration_hpet_base = paging.mapMmio(configuration_hpet_base, 0x400, true);
|
||||||
|
hpet_mapped = true;
|
||||||
|
}
|
||||||
const caps = hpetRead64(0x00);
|
const caps = hpetRead64(0x00);
|
||||||
const period_fs = caps >> 32; // femtoseconds per tick
|
const period_fs = caps >> 32; // femtoseconds per tick
|
||||||
if (period_fs == 0) return null;
|
if (period_fs == 0) return null;
|
||||||
@@ -364,24 +436,169 @@ pub fn tscHz() u64 {
|
|||||||
return tsc_hz;
|
return tsc_hz;
|
||||||
}
|
}
|
||||||
|
|
||||||
// Monotonic high-resolution clock, from the TSC. A function per resolution, each
|
// Monotonic high-resolution clock. A function per resolution, each scaling the
|
||||||
// scaling the cycle delta directly at its unit (the 128-bit intermediate avoids
|
// counter delta directly at its unit (the 128-bit intermediate avoids overflow
|
||||||
// overflow across a long uptime). nanos() resolves to a few ns; millis() is what
|
// across a long uptime). nanos() resolves to a few ns on the TSC; millis() is what
|
||||||
// the scheduler uses for sleep deadlines.
|
// the scheduler uses for sleep deadlines. The source is the TSC when it is invariant
|
||||||
|
// and synchronized, else the HPET counter (see clock_source) — the branch is one
|
||||||
|
// global load and the TSC path is unchanged from before.
|
||||||
|
|
||||||
|
/// The selected source's counter delta since its zero point.
|
||||||
|
fn clockCount() u64 {
|
||||||
|
return switch (clock_source) {
|
||||||
|
.tsc => rdtsc() -% tsc_base,
|
||||||
|
// A 64-bit HPET (the only kind we select) never wraps in any realistic
|
||||||
|
// uptime, so the wrapping subtraction is exact.
|
||||||
|
.hpet => readHpet() -% hpet_clock_base,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The selected source's frequency (0 if the clock is unavailable/uncalibrated).
|
||||||
|
fn clockHertz() u64 {
|
||||||
|
return switch (clock_source) {
|
||||||
|
.tsc => tsc_hz,
|
||||||
|
.hpet => hpet_clock_hz,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
pub fn nanos() u64 {
|
pub fn nanos() u64 {
|
||||||
if (tsc_hz == 0) return 0;
|
const hz = clockHertz();
|
||||||
return @intCast(@as(u128, rdtsc() -% tsc_base) * 1_000_000_000 / tsc_hz);
|
if (hz == 0) return 0;
|
||||||
|
return @intCast(@as(u128, clockCount()) * 1_000_000_000 / hz);
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn micros() u64 {
|
pub fn micros() u64 {
|
||||||
if (tsc_hz == 0) return 0;
|
const hz = clockHertz();
|
||||||
return @intCast(@as(u128, rdtsc() -% tsc_base) * 1_000_000 / tsc_hz);
|
if (hz == 0) return 0;
|
||||||
|
return @intCast(@as(u128, clockCount()) * 1_000_000 / hz);
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn millis() u64 {
|
pub fn millis() u64 {
|
||||||
if (tsc_hz == 0) return 0;
|
const hz = clockHertz();
|
||||||
return @intCast(@as(u128, rdtsc() -% tsc_base) * 1_000 / tsc_hz);
|
if (hz == 0) return 0;
|
||||||
|
return @intCast(@as(u128, clockCount()) * 1_000 / hz);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether the CPU advertises an invariant TSC (CPUID 0x80000007 EDX[8]).
|
||||||
|
pub fn tscInvariant() bool {
|
||||||
|
return tsc_invariant;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Test hook: force the TSC clocksource on, as if the CPU had advertised an invariant
|
||||||
|
/// TSC. QEMU's TCG accelerator (the only one for an x86 guest on an Apple-Silicon
|
||||||
|
/// host) does not expose the invariant-TSC bit — its emulated TSC isn't invariant — so
|
||||||
|
/// the tsc-sync test can't reach the real-Intel/AMD/KVM path through CPUID. This lets
|
||||||
|
/// that test exercise the TSC clocksource and the cross-core warp check anyway. tsc_base
|
||||||
|
/// is left as-is so the switch from the HPET is continuous.
|
||||||
|
pub fn forceTscClocksourceForTest() void {
|
||||||
|
tsc_invariant = true;
|
||||||
|
clock_source = .tsc;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// How many per-AP warp checks actually ran (a rendezvous completed) — lets a test
|
||||||
|
/// confirm the cross-core check executed rather than being skipped.
|
||||||
|
pub fn warpChecksRun() u32 {
|
||||||
|
return warp_checks;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether the per-core TSCs are synchronized (no backward warp seen at bring-up).
|
||||||
|
pub fn tscSynced() bool {
|
||||||
|
return tsc_synced;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The active monotonic clocksource, for the boot log and tests.
|
||||||
|
pub fn clockSourceName() []const u8 {
|
||||||
|
return switch (clock_source) {
|
||||||
|
.tsc => "tsc",
|
||||||
|
.hpet => "hpet",
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- cross-core TSC synchronization ("warp") check -------------------------
|
||||||
|
// Two cores hammer a shared "max seen" TSC value under a lock; if either reads a
|
||||||
|
// value below that max, its TSC lags the other's, and time would run backward for a
|
||||||
|
// task migrating between them (Linux calls this a warp). danos brings APs up one at a
|
||||||
|
// time, so this runs pairwise: the BSP (source) against each AP (target) as it comes
|
||||||
|
// online. It only matters — and only runs — while the TSC is the clocksource; on a
|
||||||
|
// machine already on the HPET (a bare VM) the whole rendezvous is skipped.
|
||||||
|
|
||||||
|
var warp_lock: u32 = 0;
|
||||||
|
var warp_last: u64 = 0;
|
||||||
|
var warp_bsp_ready: u32 = 0;
|
||||||
|
var warp_ap_ready: u32 = 0;
|
||||||
|
var warp_stop: u32 = 0;
|
||||||
|
var warp_checks: u32 = 0; // completed per-AP rendezvous count (for the tsc-sync test)
|
||||||
|
|
||||||
|
const warp_rounds: u32 = 1 << 20; // locked reads on the BSP: ~1 ms at GHz rates
|
||||||
|
const warp_spin_limit: u64 = 1 << 32; // bound every rendezvous wait so a lost core can't hang boot
|
||||||
|
|
||||||
|
fn warpTick() void {
|
||||||
|
while (@cmpxchgWeak(u32, &warp_lock, 0, 1, .acquire, .monotonic) != null) asm volatile ("pause");
|
||||||
|
const t = rdtsc();
|
||||||
|
if (t < warp_last) {
|
||||||
|
const delta = warp_last - t;
|
||||||
|
if (delta > tsc_warp_cycles) tsc_warp_cycles = delta;
|
||||||
|
tsc_synced = false;
|
||||||
|
} else {
|
||||||
|
warp_last = t;
|
||||||
|
}
|
||||||
|
@atomicStore(u32, &warp_lock, 0, .release);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Spin (bounded) until `flag` is nonzero; false on timeout.
|
||||||
|
fn warpAwait(flag: *u32) bool {
|
||||||
|
var spins: u64 = 0;
|
||||||
|
while (@atomicLoad(u32, flag, .acquire) == 0) : (spins += 1) {
|
||||||
|
if (spins >= warp_spin_limit) return false;
|
||||||
|
asm volatile ("pause");
|
||||||
|
}
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// BSP side of the pairwise TSC warp check, run once per AP as it reports in. No-op
|
||||||
|
/// unless the TSC is the active clocksource. If the AP's TSC proves to lag, demote
|
||||||
|
/// the monotonic clock to the HPET without a discontinuity.
|
||||||
|
pub fn checkWarpSource() void {
|
||||||
|
if (clock_source != .tsc) return;
|
||||||
|
warp_last = 0;
|
||||||
|
@atomicStore(u32, &warp_stop, 0, .release);
|
||||||
|
@atomicStore(u32, &warp_ap_ready, 0, .release);
|
||||||
|
@atomicStore(u32, &warp_bsp_ready, 1, .release);
|
||||||
|
if (!warpAwait(&warp_ap_ready)) { // AP never joined the rendezvous; skip, don't hang
|
||||||
|
@atomicStore(u32, &warp_bsp_ready, 0, .release);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
var i: u32 = 0;
|
||||||
|
while (i < warp_rounds) : (i += 1) warpTick();
|
||||||
|
@atomicStore(u32, &warp_stop, 1, .release);
|
||||||
|
@atomicStore(u32, &warp_bsp_ready, 0, .release);
|
||||||
|
warp_checks += 1;
|
||||||
|
|
||||||
|
if (!tsc_synced and hpet_clock_hz != 0) demoteToHpet();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// AP side: join the BSP's warp check, then return so the core can enter the
|
||||||
|
/// scheduler. Bounded so a missing BSP can't strand the core.
|
||||||
|
pub fn checkWarpTarget() void {
|
||||||
|
if (clock_source != .tsc) return;
|
||||||
|
if (!warpAwait(&warp_bsp_ready)) return;
|
||||||
|
@atomicStore(u32, &warp_ap_ready, 1, .release);
|
||||||
|
var spins: u64 = 0;
|
||||||
|
while (@atomicLoad(u32, &warp_stop, .acquire) == 0) : (spins += 1) {
|
||||||
|
if (spins >= warp_spin_limit) return;
|
||||||
|
warpTick();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Switch the clocksource from the TSC to the HPET without a discontinuity: choose
|
||||||
|
/// the HPET zero point so it reads the same nanosecond value the TSC does right now,
|
||||||
|
/// so time neither jumps nor runs backward across the switch. Called when the warp
|
||||||
|
/// check proves the per-core TSCs unsynchronized.
|
||||||
|
fn demoteToHpet() void {
|
||||||
|
const now_ns = @as(u128, rdtsc() -% tsc_base) * 1_000_000_000 / tsc_hz;
|
||||||
|
const equivalent_ticks: u64 = @intCast(now_ns * hpet_clock_hz / 1_000_000_000);
|
||||||
|
hpet_clock_base = readHpet() -% equivalent_ticks;
|
||||||
|
clock_source = .hpet;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Acknowledge the current interrupt so the LAPIC will deliver the next one.
|
/// Acknowledge the current interrupt so the LAPIC will deliver the next one.
|
||||||
|
|||||||
@@ -469,6 +469,35 @@ pub fn clockHz() u64 {
|
|||||||
return apic.tscHz();
|
return apic.tscHz();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Whether the CPU guarantees an **invariant** TSC (CPUID 0x80000007 EDX[8] on
|
||||||
|
/// x86; the analogous architectural guarantee elsewhere). When false the TSC is not
|
||||||
|
/// used as the clocksource.
|
||||||
|
pub fn clockInvariant() bool {
|
||||||
|
return apic.tscInvariant();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether the per-core clock counters are synchronized (no backward warp observed
|
||||||
|
/// at SMP bring-up). When false the clock falls back off the TSC.
|
||||||
|
pub fn clockSynchronized() bool {
|
||||||
|
return apic.tscSynced();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The active monotonic clocksource, for the boot log ("tsc" or "hpet" on x86).
|
||||||
|
pub fn clockSourceName() []const u8 {
|
||||||
|
return apic.clockSourceName();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Test hook: force the TSC clocksource on, to exercise the TSC + warp-check path on
|
||||||
|
/// a hypervisor that won't advertise an invariant TSC (see apic.forceTscClocksourceForTest).
|
||||||
|
pub fn forceTscClocksourceForTest() void {
|
||||||
|
apic.forceTscClocksourceForTest();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// How many per-AP TSC warp checks completed (for the tsc-sync test).
|
||||||
|
pub fn warpChecksRun() u32 {
|
||||||
|
return apic.warpChecksRun();
|
||||||
|
}
|
||||||
|
|
||||||
/// Unmask maskable interrupts (`sti`) so device interrupts get delivered.
|
/// Unmask maskable interrupts (`sti`) so device interrupts get delivered.
|
||||||
pub fn enableInterrupts() void {
|
pub fn enableInterrupts() void {
|
||||||
asm volatile ("sti");
|
asm volatile ("sti");
|
||||||
|
|||||||
@@ -148,7 +148,14 @@ pub fn startAp(apic_id: u32, stack_top: usize, percpu: usize, index: usize, cr3:
|
|||||||
// Wait up to 100 ms for the AP to reach apEntry and set the flag.
|
// Wait up to 100 ms for the AP to reach apEntry and set the flag.
|
||||||
const deadline = apic.millis() + 100;
|
const deadline = apic.millis() + 100;
|
||||||
while (apic.millis() < deadline) {
|
while (apic.millis() < deadline) {
|
||||||
if (@atomicLoad(u32, &ap_alive, .acquire) != 0) return true;
|
if (@atomicLoad(u32, &ap_alive, .acquire) != 0) {
|
||||||
|
// Cross-check this core's TSC against the BSP's before it joins the run
|
||||||
|
// loop: an unsynchronized TSC must be caught before any task can migrate
|
||||||
|
// onto this core and observe time going backward. No-op unless the TSC is
|
||||||
|
// the clocksource (apic.checkWarpSource).
|
||||||
|
apic.checkWarpSource();
|
||||||
|
return true;
|
||||||
|
}
|
||||||
asm volatile ("pause");
|
asm volatile ("pause");
|
||||||
}
|
}
|
||||||
return false;
|
return false;
|
||||||
@@ -177,6 +184,11 @@ fn apEntry(percpu: usize) callconv(.c) noreturn {
|
|||||||
|
|
||||||
@atomicStore(u32, &ap_alive, 1, .release); // "architecture state up" — BSP is polling this
|
@atomicStore(u32, &ap_alive, 1, .release); // "architecture state up" — BSP is polling this
|
||||||
|
|
||||||
|
// Rendezvous with the BSP for the TSC warp check (no-op unless the TSC is the
|
||||||
|
// clocksource) before joining the run loop, so this core's clock is vetted before
|
||||||
|
// it can run any task.
|
||||||
|
apic.checkWarpTarget();
|
||||||
|
|
||||||
if (secondary_entry) |enterScheduler| enterScheduler(); // joins the run loop
|
if (secondary_entry) |enterScheduler| enterScheduler(); // joins the run loop
|
||||||
while (true) asm volatile ("hlt"); // (only if no entry was registered)
|
while (true) asm volatile ("hlt"); // (only if no entry was registered)
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -189,7 +189,7 @@ pub fn register(parent_id: u64, owner: u32, descriptor: *const device_abi.Device
|
|||||||
if (!ok) return error.NotContained;
|
if (!ok) return error.NotContained;
|
||||||
}
|
}
|
||||||
|
|
||||||
// Idempotent on exact match (docs/m19-m20-plan.md decision 3): a restarted
|
// Idempotent on exact match (docs/device-manager.md): a restarted
|
||||||
// registering bus re-registers what it rediscovers, and the table has no
|
// registering bus re-registers what it rediscovers, and the table has no
|
||||||
// unregister — an identical (class, identity, resources) child under the
|
// unregister — an identical (class, identity, resources) child under the
|
||||||
// same parent returns the existing id instead of appending a duplicate.
|
// same parent returns the existing id instead of appending a duplicate.
|
||||||
|
|||||||
@@ -257,10 +257,30 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
|||||||
log.checkpoint(cp_timer);
|
log.checkpoint(cp_timer);
|
||||||
log.print("/system/kernel: timer online ({d} Hz tick; timer clock {d} MHz, clock {d} MHz; calibrated via {s})\n", .{ architecture.timer_hz, architecture.timerClockHz() / 1_000_000, architecture.clockHz() / 1_000_000, architecture.timerCalibrationSource() });
|
log.print("/system/kernel: timer online ({d} Hz tick; timer clock {d} MHz, clock {d} MHz; calibrated via {s})\n", .{ architecture.timer_hz, architecture.timerClockHz() / 1_000_000, architecture.clockHz() / 1_000_000, architecture.timerCalibrationSource() });
|
||||||
|
|
||||||
|
// The tsc-sync test forces the TSC clocksource on before the cores come up, so the
|
||||||
|
// TSC + warp-check path is exercised even under TCG (which won't advertise an
|
||||||
|
// invariant TSC). Inert in a normal build (docs/timers.md).
|
||||||
|
if (build_options.test_case) |tc| {
|
||||||
|
if (std.mem.eql(u8, tc, "tsc-sync")) architecture.forceTscClocksourceForTest();
|
||||||
|
}
|
||||||
|
|
||||||
// Wake the other cores (application processors). A no-op on a single-core
|
// Wake the other cores (application processors). A no-op on a single-core
|
||||||
// machine; on SMP each AP climbs to long mode and reports in (docs/smp.md).
|
// machine; on SMP each AP climbs to long mode and reports in (docs/smp.md). The
|
||||||
|
// per-core TSC warp check rides this: each AP is vetted before it joins the run
|
||||||
|
// loop (docs/timers.md).
|
||||||
bringUpSecondaries();
|
bringUpSecondaries();
|
||||||
|
|
||||||
|
// Report the monotonic clock's final reliability, now the warp check has run on
|
||||||
|
// every core. On real Intel/AMD this is the invariant, synchronized TSC; a bare
|
||||||
|
// VM (no invariant bit) or a machine whose cores' TSCs skew uses the HPET instead.
|
||||||
|
log.print("/system/kernel: clocksource {s} (TSC invariant: {s}, synchronized: {s})\n", .{
|
||||||
|
architecture.clockSourceName(),
|
||||||
|
if (architecture.clockInvariant()) "yes" else "no",
|
||||||
|
if (architecture.clockSynchronized()) "yes" else "no",
|
||||||
|
});
|
||||||
|
if (!architecture.clockSynchronized())
|
||||||
|
log.write("/system/kernel: WARNING: per-core TSCs are not synchronized; monotonic clock moved off the TSC\n");
|
||||||
|
|
||||||
// In a test build (`zig build -Dtest-case=<name>`), run that case and stop.
|
// In a test build (`zig build -Dtest-case=<name>`), run that case and stop.
|
||||||
// Normal builds fall through to the idle halt.
|
// Normal builds fall through to the idle halt.
|
||||||
if (build_options.test_case) |case| {
|
if (build_options.test_case) |case| {
|
||||||
|
|||||||
+143
-179
@@ -102,6 +102,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
|
|||||||
stressTest();
|
stressTest();
|
||||||
} else if (eql(case, "smp-retry")) {
|
} else if (eql(case, "smp-retry")) {
|
||||||
smpRetryTest();
|
smpRetryTest();
|
||||||
|
} else if (eql(case, "tsc-sync")) {
|
||||||
|
tscSyncTest();
|
||||||
} else if (eql(case, "fault-ud")) {
|
} else if (eql(case, "fault-ud")) {
|
||||||
faultInvalidOpcode();
|
faultInvalidOpcode();
|
||||||
} else if (eql(case, "fault-pf")) {
|
} else if (eql(case, "fault-pf")) {
|
||||||
@@ -162,14 +164,12 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
|
|||||||
vfsTest(boot_information);
|
vfsTest(boot_information);
|
||||||
} else if (eql(case, "input")) {
|
} else if (eql(case, "input")) {
|
||||||
inputTest(boot_information);
|
inputTest(boot_information);
|
||||||
} else if (eql(case, "hpet")) {
|
|
||||||
hpetTest(boot_information);
|
|
||||||
} else if (eql(case, "iopass")) {
|
} else if (eql(case, "iopass")) {
|
||||||
ioPassTest();
|
ioPassTest();
|
||||||
} else if (eql(case, "irqfree")) {
|
} else if (eql(case, "irqfree")) {
|
||||||
irqFreeTest();
|
irqFreeTest();
|
||||||
} else if (eql(case, "bus")) {
|
} else if (eql(case, "containment")) {
|
||||||
busTest(boot_information);
|
containmentTest();
|
||||||
} else if (eql(case, "device-manager")) {
|
} else if (eql(case, "device-manager")) {
|
||||||
deviceManagerTest(boot_information);
|
deviceManagerTest(boot_information);
|
||||||
} else if (eql(case, "poweroff")) {
|
} else if (eql(case, "poweroff")) {
|
||||||
@@ -213,7 +213,7 @@ fn eql(a: []const u8, b: []const u8) bool {
|
|||||||
|
|
||||||
/// Whether the captured last-write buffer *contains* `needle`. Markers are
|
/// Whether the captured last-write buffer *contains* `needle`. Markers are
|
||||||
/// matched as substrings, not prefixes, so a service's source-path debug prefix
|
/// matched as substrings, not prefixes, so a service's source-path debug prefix
|
||||||
/// (`system/drivers/hpet: ok`) still satisfies a marker like `hpet: ok`.
|
/// (`system/drivers/pci-bus: ...`) still satisfies a marker like `pci-bus: `.
|
||||||
fn bufferHas(needle: []const u8) bool {
|
fn bufferHas(needle: []const u8) bool {
|
||||||
return std.mem.indexOf(u8, process.write_buffer[0..process.write_len], needle) != null;
|
return std.mem.indexOf(u8, process.write_buffer[0..process.write_len], needle) != null;
|
||||||
}
|
}
|
||||||
@@ -1211,6 +1211,35 @@ fn clockTest() void {
|
|||||||
result();
|
result();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The TSC clocksource + cross-core warp check (`-smp 4`). This is the real
|
||||||
|
/// Intel/AMD / KVM path — an invariant, synchronized TSC. TCG (the only x86
|
||||||
|
/// accelerator on an Apple-Silicon host) won't advertise an invariant TSC, so the
|
||||||
|
/// boot forces the TSC clocksource on (kernel.zig, gated on this case) to exercise
|
||||||
|
/// the machinery: the kernel must run the per-AP warp check as each core came up,
|
||||||
|
/// find the cores' TSCs synchronized, and keep the clock on the TSC (no HPET
|
||||||
|
/// fallback). The default suite (no force) exercises the HPET fallback instead.
|
||||||
|
fn tscSyncTest() void {
|
||||||
|
log("DANOS-TEST-BEGIN: tsc-sync\n", .{});
|
||||||
|
check("clocksource is the TSC (forced-invariant path)", eql(architecture.clockSourceName(), "tsc"));
|
||||||
|
check("the cross-core warp check ran on the APs", architecture.warpChecksRun() >= 1);
|
||||||
|
check("per-core TSCs synchronized (no warp, no HPET fallback)", architecture.clockSynchronized());
|
||||||
|
|
||||||
|
// A warp that slipped past the bring-up check would surface as a backward reading.
|
||||||
|
var last = architecture.nanos();
|
||||||
|
var monotonic = true;
|
||||||
|
var advanced = false;
|
||||||
|
var i: u32 = 0;
|
||||||
|
while (i < 1_000_000) : (i += 1) {
|
||||||
|
const t = architecture.nanos();
|
||||||
|
if (t < last) monotonic = false;
|
||||||
|
if (t > last) advanced = true;
|
||||||
|
last = t;
|
||||||
|
}
|
||||||
|
check("monotonic clock advanced", advanced);
|
||||||
|
check("monotonic clock never ran backward", monotonic);
|
||||||
|
result();
|
||||||
|
}
|
||||||
|
|
||||||
var proc_worker_run: bool = true;
|
var proc_worker_run: bool = true;
|
||||||
var proc_worker_ran: bool = false;
|
var proc_worker_ran: bool = false;
|
||||||
|
|
||||||
@@ -2258,68 +2287,7 @@ fn spawnNamed(rd: initial_ramdisk.Reader, name: []const u8) bool {
|
|||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// IO passthrough + IRQ-as-IPC: a user-space driver drives real hardware and is
|
/// The GSI discovery recorded for the HPET, from the same device table drivers see.
|
||||||
/// *woken by it*. Spawn hpet, which claims the HPET, maps its registers into its
|
|
||||||
/// own ring-3 address space, arms a level-triggered comparator, binds the interrupt
|
|
||||||
/// to an IPC endpoint, and then blocks. It prints "hpet: ok" only after being woken
|
|
||||||
/// `target_ticks` times — it cannot reach that line by polling, because the loop's
|
|
||||||
/// only exit is through `replyWait` returning a notification badge.
|
|
||||||
///
|
|
||||||
/// The interesting assertion is the last one, which doesn't trust hpet at all: it
|
|
||||||
/// reads the I/O APIC's redirection entry back and checks the kernel really routed
|
|
||||||
/// the line (our vector, level-triggered) and really left it unmasked after the
|
|
||||||
/// driver's final `irq_ack`. hpet disables its comparator on the last interrupt, so
|
|
||||||
/// that state is quiescent and not a race.
|
|
||||||
fn hpetTest(boot_information: *const BootInformation) void {
|
|
||||||
log("DANOS-TEST-BEGIN: hpet\n", .{});
|
|
||||||
if (boot_information.initial_ramdisk_len == 0) {
|
|
||||||
check("bootloader handed over an initial_ramdisk", false);
|
|
||||||
result();
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
|
||||||
const rd = initial_ramdisk.Reader.init(image) orelse {
|
|
||||||
check("initial_ramdisk image is valid", false);
|
|
||||||
result();
|
|
||||||
return;
|
|
||||||
};
|
|
||||||
|
|
||||||
process.write_count = 0;
|
|
||||||
process.write_from_user = false;
|
|
||||||
check("hpet spawned from the initial_ramdisk", spawnNamed(rd, "hpet"));
|
|
||||||
|
|
||||||
const prefix = "hpet: ok";
|
|
||||||
scheduler.setPriority(1);
|
|
||||||
const deadline = architecture.millis() + 10000;
|
|
||||||
while (architecture.millis() < deadline) {
|
|
||||||
if (bufferHas(prefix) and process.write_count >= 2) break;
|
|
||||||
scheduler.yield();
|
|
||||||
}
|
|
||||||
scheduler.setPriority(4);
|
|
||||||
|
|
||||||
const ok = bufferHas(prefix);
|
|
||||||
check("user driver mapped HPET MMIO and was woken by its interrupt", ok);
|
|
||||||
check("driver syscalls came from user mode (CPL 3)", process.write_from_user);
|
|
||||||
check("kernel routed and re-armed the HPET's line at the I/O APIC", hpetRouteOk());
|
|
||||||
result();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Read back the I/O APIC redirection entry for the HPET's GSI and confirm the
|
|
||||||
/// kernel programmed it: a vector in the device window, level-triggered, unmasked.
|
|
||||||
/// Independent of anything the driver reported about itself.
|
|
||||||
fn hpetRouteOk() bool {
|
|
||||||
const gsi = hpetGsi() orelse return false;
|
|
||||||
if (gsi >= architecture.irqRouteCount()) return false;
|
|
||||||
const low = architecture.irqRouteRaw(gsi); // entry index == GSI (this I/O APIC's gsi_base is 0)
|
|
||||||
const vector: u8 = @truncate(low & 0xFF);
|
|
||||||
const masked = low & (1 << 16) != 0;
|
|
||||||
const level = low & (1 << 15) != 0;
|
|
||||||
return vector >= architecture.irq_vector_base and
|
|
||||||
vector < architecture.irq_vector_base + architecture.irq_vector_count and
|
|
||||||
level and !masked;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// The GSI discovery recorded for the HPET, from the same device table the driver saw.
|
|
||||||
fn hpetGsi() ?u32 {
|
fn hpetGsi() ?u32 {
|
||||||
var buffer: [16]device_abi.DeviceDescriptor = undefined;
|
var buffer: [16]device_abi.DeviceDescriptor = undefined;
|
||||||
const n = @min(devices_broker.enumerate(&buffer), buffer.len);
|
const n = @min(devices_broker.enumerate(&buffer), buffer.len);
|
||||||
@@ -2334,87 +2302,87 @@ fn hpetGsi() ?u32 {
|
|||||||
return null;
|
return null;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Bus driver: a user process claims a device that contains other devices, enumerates
|
/// `device_register` containment — a kernel security property, tested directly against
|
||||||
/// them from the hardware, and publishes each as a child via `device_register` — the
|
/// the broker (no user-space demo driver). A bus driver publishes children of a device
|
||||||
/// primitive a PCI bridge or USB hub driver is built from.
|
/// it owns; the kernel must **refuse** any child whose resource escapes the parent's
|
||||||
///
|
/// grant, or `device_register` would become a system call for mapping arbitrary physical
|
||||||
/// `bus` treats the HPET's register block as a bus and its comparators as children,
|
/// memory. This is the property the old `bus` demo driver proved end-to-end; with the
|
||||||
/// giving each a 0x20 sub-window. It checks its own work (children come back from the
|
/// demo gone, the property is asserted where it lives — in the kernel. Also checks the
|
||||||
/// table with the right parent and a strictly narrower window) and, importantly, that
|
/// idempotence rule (M19.0): re-registering an identical child returns the same id
|
||||||
/// the kernel **refuses** a child whose window escapes the parent's — without that,
|
/// instead of appending a duplicate.
|
||||||
/// `device_register` would be a system_call for mapping arbitrary physical memory. It prints
|
fn containmentTest() void {
|
||||||
/// "bus: ok" only if all of that holds.
|
log("DANOS-TEST-BEGIN: containment\n", .{});
|
||||||
///
|
|
||||||
/// The kernel-side check here is the one bus can't make: that the children really did
|
|
||||||
/// land in the device table with the containment invariant intact.
|
|
||||||
fn busTest(boot_information: *const BootInformation) void {
|
|
||||||
log("DANOS-TEST-BEGIN: bus\n", .{});
|
|
||||||
|
|
||||||
// M19.0: device_register is idempotent on exact match — a restarted
|
const me = scheduler.currentId();
|
||||||
// registering bus must not duplicate its children. Driven directly against
|
var buffer: [64]device_abi.DeviceDescriptor = undefined;
|
||||||
// the broker: claim an unclaimed node, register the same (class, hid,
|
check("device tree is seeded", devices_broker.enumerate(&buffer) >= 1);
|
||||||
// resourceless) child twice, expect one id and one table entry.
|
|
||||||
{
|
|
||||||
const me = scheduler.currentId();
|
|
||||||
var probe: [1]device_abi.DeviceDescriptor = undefined;
|
|
||||||
const total = devices_broker.enumerate(&probe);
|
|
||||||
check("device tree is seeded for the idempotence check", total >= 1);
|
|
||||||
if (devices_broker.ownerOf(0) == null) {
|
|
||||||
check("claimed device 0 for the idempotence check", devices_broker.claim(0, me));
|
|
||||||
var child = std.mem.zeroes(device_abi.DeviceDescriptor);
|
|
||||||
child.class = @intFromEnum(device_abi.DeviceClass.unknown);
|
|
||||||
child.pci_class = device_abi.no_pci_class;
|
|
||||||
child.hid_len = 4;
|
|
||||||
child.hid[0..4].* = "idem".*;
|
|
||||||
const first = devices_broker.register(0, me, &child) catch 0;
|
|
||||||
check("first register succeeded", first != 0);
|
|
||||||
const before = devices_broker.enumerate(&probe);
|
|
||||||
const second = devices_broker.register(0, me, &child) catch 0;
|
|
||||||
check("re-register returned the same id", second == first);
|
|
||||||
check("re-register grew nothing", devices_broker.enumerate(&probe) == before);
|
|
||||||
devices_broker.releaseAllOwnedBy(me);
|
|
||||||
} else {
|
|
||||||
check("device 0 unexpectedly claimed before the idempotence check", false);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
if (boot_information.initial_ramdisk_len == 0) {
|
// The kernel-seeded HPET timer block is a device with a memory resource — a natural
|
||||||
check("bootloader handed over an initial_ramdisk", false);
|
// parent to publish sub-window children under, as a PCI bridge or USB hub would.
|
||||||
|
const parent_id = hpetDeviceId() orelse {
|
||||||
|
check("found a device with a memory window to parent children under", false);
|
||||||
result();
|
result();
|
||||||
return;
|
return;
|
||||||
|
};
|
||||||
|
const parent = buffer[@intCast(parent_id)];
|
||||||
|
var window: ?device_abi.ResourceDescriptor = null;
|
||||||
|
for (0..parent.resource_count) |j| {
|
||||||
|
if (parent.resources[j].kind == @intFromEnum(device_abi.ResourceKind.memory)) window = parent.resources[j];
|
||||||
}
|
}
|
||||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
const parent_window = window orelse {
|
||||||
const rd = initial_ramdisk.Reader.init(image) orelse {
|
check("parent exposes a memory window", false);
|
||||||
check("initial_ramdisk image is valid", false);
|
|
||||||
result();
|
result();
|
||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
|
|
||||||
process.write_count = 0;
|
if (devices_broker.ownerOf(parent_id) != null) {
|
||||||
process.write_from_user = false;
|
check("parent device was unclaimed at the start of the test", false);
|
||||||
check("bus spawned from the initial_ramdisk", spawnNamed(rd, "bus"));
|
result();
|
||||||
|
return;
|
||||||
const prefix = "bus: ok";
|
|
||||||
scheduler.setPriority(1);
|
|
||||||
const deadline = architecture.millis() + 10000;
|
|
||||||
while (architecture.millis() < deadline) {
|
|
||||||
if (bufferHas(prefix)) break;
|
|
||||||
scheduler.yield();
|
|
||||||
}
|
}
|
||||||
scheduler.setPriority(4);
|
check("claimed the parent device", devices_broker.claim(parent_id, me));
|
||||||
|
defer devices_broker.releaseAllOwnedBy(me);
|
||||||
|
|
||||||
|
// A child whose window lies inside the parent's is accepted.
|
||||||
|
var fits = childDescriptor("cfit", parent_window.start, 0x20);
|
||||||
|
const before = devices_broker.enumerate(&buffer);
|
||||||
|
const good = devices_broker.register(parent_id, me, &fits) catch 0;
|
||||||
|
check("a contained child is registered", good != 0);
|
||||||
|
check("the contained child was appended to the table", devices_broker.enumerate(&buffer) == before + 1);
|
||||||
|
|
||||||
|
// A child whose window escapes the parent's is refused with NotContained.
|
||||||
|
var escapes = childDescriptor("cesc", parent_window.start, parent_window.len + 0x1000);
|
||||||
|
const refused = if (devices_broker.register(parent_id, me, &escapes)) |_| false else |err| err == error.NotContained;
|
||||||
|
check("an out-of-window child is refused (NotContained)", refused);
|
||||||
|
check("the refused child left the table unchanged", devices_broker.enumerate(&buffer) == before + 1);
|
||||||
|
|
||||||
|
// Idempotent on exact match: re-registering the accepted child returns its id and
|
||||||
|
// appends nothing (M19.0 — a restarted bus re-reports what it rediscovers).
|
||||||
|
const again = devices_broker.register(parent_id, me, &fits) catch 0;
|
||||||
|
check("re-registering an identical child returns the same id", again != 0 and again == good);
|
||||||
|
check("re-registering grew nothing", devices_broker.enumerate(&buffer) == before + 1);
|
||||||
|
|
||||||
const ok = bufferHas(prefix);
|
|
||||||
check("bus driver published children and the kernel refused an out-of-window one", ok);
|
|
||||||
check("driver syscalls came from user mode (CPL 3)", process.write_from_user);
|
|
||||||
check("every registered child is contained in its parent", childrenContained());
|
|
||||||
result();
|
result();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// A minimal child descriptor with one memory resource, for the containment test.
|
||||||
|
fn childDescriptor(hid: []const u8, start: u64, len: u64) device_abi.DeviceDescriptor {
|
||||||
|
var child = std.mem.zeroes(device_abi.DeviceDescriptor);
|
||||||
|
child.class = @intFromEnum(device_abi.DeviceClass.unknown);
|
||||||
|
child.pci_class = device_abi.no_pci_class;
|
||||||
|
child.hid_len = @intCast(hid.len);
|
||||||
|
@memcpy(child.hid[0..hid.len], hid);
|
||||||
|
child.resource_count = 1;
|
||||||
|
child.resources[0] = .{ .kind = @intFromEnum(device_abi.ResourceKind.memory), .start = start, .len = len };
|
||||||
|
return child;
|
||||||
|
}
|
||||||
|
|
||||||
/// The device manager (a ring-3 service) enumerates /system/devices, matches each
|
/// The device manager (a ring-3 service) enumerates /system/devices, matches each
|
||||||
/// device to a driver, and — eventually — spawns it. This increment only checks the
|
/// device to a driver, and spawns it. Proof of the whole discover -> match -> spawn ->
|
||||||
/// discovery+matching half: it must find the HPET (a timer) and decide `hpet` serves
|
/// driver-up chain: boot only the device-manager; it must discover the PCI host bridge,
|
||||||
/// it, printing "device-manager: ok". It uses no special privilege — the same
|
/// match `pci-bus`, and spawn it (with the bridge id as its argument) — and the spawned
|
||||||
/// `device_enumerate` any process could call. (Spawning is the next increment.)
|
/// pci-bus must reach its own live marker. It uses no special privilege — the same
|
||||||
|
/// `device_enumerate` any process could call.
|
||||||
fn deviceManagerTest(boot_information: *const BootInformation) void {
|
fn deviceManagerTest(boot_information: *const BootInformation) void {
|
||||||
log("DANOS-TEST-BEGIN: device-manager\n", .{});
|
log("DANOS-TEST-BEGIN: device-manager\n", .{});
|
||||||
if (boot_information.initial_ramdisk_len == 0) {
|
if (boot_information.initial_ramdisk_len == 0) {
|
||||||
@@ -2430,70 +2398,65 @@ fn deviceManagerTest(boot_information: *const BootInformation) void {
|
|||||||
};
|
};
|
||||||
|
|
||||||
// Let `system_spawn` find bundled binaries by name (the normal boot path does
|
// Let `system_spawn` find bundled binaries by name (the normal boot path does
|
||||||
// this too). Only the device-manager is spawned here — so if `hpet` runs at all,
|
// this too). Only the device-manager is spawned here — so if `pci-bus` runs at
|
||||||
// it's because the manager discovered the timer, matched, and spawned it.
|
// all, it's because the manager discovered the PCI host bridge, matched, and
|
||||||
|
// spawned it.
|
||||||
process.setInitialRamdisk(image);
|
process.setInitialRamdisk(image);
|
||||||
|
|
||||||
process.write_count = 0;
|
process.write_count = 0;
|
||||||
process.write_from_user = false;
|
process.write_from_user = false;
|
||||||
check("device-manager spawned from the initial_ramdisk", spawnNamed(rd, "device-manager"));
|
check("device-manager spawned from the initial_ramdisk", spawnNamed(rd, "device-manager"));
|
||||||
|
|
||||||
// End-to-end proof: the driver the manager spawned reaches its own live marker.
|
// End-to-end proof, read from kernel state — not the racy last-write serial buffer,
|
||||||
// `hpet: ok` is hpet's final, stable message (it claims the timer, maps its MMIO,
|
// since many services keep logging after pci-bus. The manager must discover the PCI
|
||||||
// binds its IRQ, services one, then sleeps) — nothing overwrites the buffer after,
|
// host bridge, match pci-bus, and spawn it, and pci-bus must come up: claim the
|
||||||
// so it's race-free to poll for. Its arrival means the whole
|
// bridge, map its ECAM, and register the functions it enumerates as children in the
|
||||||
// discover -> match -> system_spawn -> driver-up chain worked.
|
// device tree.
|
||||||
const prefix = "hpet: ok";
|
|
||||||
scheduler.setPriority(1);
|
scheduler.setPriority(1);
|
||||||
const deadline = architecture.millis() + 10000;
|
const deadline = architecture.millis() + 10000;
|
||||||
|
var spawned = false;
|
||||||
while (architecture.millis() < deadline) {
|
while (architecture.millis() < deadline) {
|
||||||
if (bufferHas(prefix)) break;
|
if (processRunning("pci-bus")) spawned = true;
|
||||||
|
if (spawned and pciFunctionsRegistered()) break;
|
||||||
scheduler.yield();
|
scheduler.yield();
|
||||||
}
|
}
|
||||||
scheduler.setPriority(4);
|
scheduler.setPriority(4);
|
||||||
|
|
||||||
const ok = bufferHas(prefix);
|
check("device manager discovered the PCI host bridge and spawned pci-bus", spawned);
|
||||||
check("device manager matched the timer and system_spawn'd hpet, which came up", ok);
|
check("pci-bus came up and registered the functions it enumerated", pciFunctionsRegistered());
|
||||||
check("its syscalls came from user mode (CPL 3)", process.write_from_user);
|
check("its syscalls came from user mode (CPL 3)", process.write_from_user);
|
||||||
result();
|
result();
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Every child `bus` registered must have each of its resources inside a parent
|
/// Whether a live task was spawned under `name` (its argv[0]) — read from the kernel
|
||||||
/// resource of the same kind — the invariant `device_register` exists to maintain,
|
/// task table, the same snapshot `process_enumerate` exposes.
|
||||||
/// checked from the kernel's own table rather than the driver's word for it.
|
fn processRunning(name: []const u8) bool {
|
||||||
///
|
var table: [64]abi.ProcessDescriptor = undefined;
|
||||||
/// Only *registered* children are checked, not the whole tree. Firmware topology is
|
const total = scheduler.enumerate(&table);
|
||||||
/// trusted and doesn't obey containment: a PCI function's BAR is not inside its host
|
for (table[0..@min(total, table.len)]) |d| {
|
||||||
/// bridge's `bus_range`, because a bus-number range isn't an address window.
|
if (std.mem.eql(u8, d.name[0..d.name_length], name)) return true;
|
||||||
fn childrenContained() bool {
|
|
||||||
var buffer: [64]device_abi.DeviceDescriptor = undefined;
|
|
||||||
const n = @min(devices_broker.enumerate(&buffer), buffer.len);
|
|
||||||
|
|
||||||
const bus_id = hpetDeviceId() orelse return false;
|
|
||||||
const p = buffer[@intCast(bus_id)];
|
|
||||||
|
|
||||||
var children: usize = 0;
|
|
||||||
for (buffer[0..n]) |d| {
|
|
||||||
if (d.parent != bus_id) continue;
|
|
||||||
children += 1;
|
|
||||||
for (0..d.resource_count) |i| {
|
|
||||||
const r = d.resources[i];
|
|
||||||
var ok = false;
|
|
||||||
for (0..p.resource_count) |j| {
|
|
||||||
const pr = p.resources[j];
|
|
||||||
if (pr.kind != r.kind) continue;
|
|
||||||
if (r.kind == @intFromEnum(device_abi.ResourceKind.irq)) {
|
|
||||||
if (pr.start == r.start) ok = true;
|
|
||||||
} else if (r.len != 0 and r.start >= pr.start and
|
|
||||||
r.start + r.len <= pr.start + pr.len) ok = true;
|
|
||||||
}
|
|
||||||
if (!ok) return false;
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
return children > 0; // bus must have published at least one
|
return false;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Device id of the HPET (the bus bus claims), from the same table drivers see.
|
/// Whether pci-bus registered at least one function under the PCI host bridge — proof
|
||||||
|
/// it came up, claimed the bridge, mapped its ECAM, and walked configuration space.
|
||||||
|
fn pciFunctionsRegistered() bool {
|
||||||
|
var buffer: [64]device_abi.DeviceDescriptor = undefined;
|
||||||
|
const n = @min(devices_broker.enumerate(&buffer), buffer.len);
|
||||||
|
var bridge_id: ?u64 = null;
|
||||||
|
for (buffer[0..n]) |d| {
|
||||||
|
if (d.class == @intFromEnum(device_abi.DeviceClass.pci_host_bridge)) bridge_id = d.id;
|
||||||
|
}
|
||||||
|
const bid = bridge_id orelse return false;
|
||||||
|
for (buffer[0..n]) |d| {
|
||||||
|
if (d.parent == bid) return true;
|
||||||
|
}
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Device id of the kernel-seeded HPET timer block (the node with a memory resource),
|
||||||
|
/// from the same device table drivers see. Used as a containment-test parent.
|
||||||
fn hpetDeviceId() ?u64 {
|
fn hpetDeviceId() ?u64 {
|
||||||
var buffer: [64]device_abi.DeviceDescriptor = undefined;
|
var buffer: [64]device_abi.DeviceDescriptor = undefined;
|
||||||
const n = @min(devices_broker.enumerate(&buffer), buffer.len);
|
const n = @min(devices_broker.enumerate(&buffer), buffer.len);
|
||||||
@@ -2511,8 +2474,9 @@ fn hpetDeviceId() ?u64 {
|
|||||||
/// (so a dead driver's device goes quiet instead of storming) and the slot cleared
|
/// (so a dead driver's device goes quiet instead of storming) and the slot cleared
|
||||||
/// (so an ISR never posts a notification into the endpoint that is about to be freed).
|
/// (so an ISR never posts a notification into the endpoint that is about to be freed).
|
||||||
///
|
///
|
||||||
/// This is the path `hpet` never takes — it runs forever — so it gets its own test.
|
/// A long-running driver that never exits wouldn't reach this teardown path, so it
|
||||||
/// Two properties, both read back from the hardware rather than from our own state:
|
/// gets its own test that binds and releases directly. Two properties, both read back
|
||||||
|
/// from the hardware rather than from our own state:
|
||||||
///
|
///
|
||||||
/// 1. A bound GSI is routed and unmasked.
|
/// 1. A bound GSI is routed and unmasked.
|
||||||
/// 2. After `releaseOwner` for the binding's owner, that same entry is masked again.
|
/// 2. After `releaseOwner` for the binding's owner, that same entry is masked again.
|
||||||
|
|||||||
@@ -1,5 +1,5 @@
|
|||||||
//! /system/services/acpi — the ACPI discovery service: the x86 firmware
|
//! /system/services/acpi — the ACPI discovery service: the x86 firmware
|
||||||
//! interpreter, moved out of ring 0 (docs/m19-m20-plan.md, M20). Claims the
|
//! interpreter, moved out of ring 0 (docs/discovery.md). Claims the
|
||||||
//! `acpi-tables` node the kernel publishes (the AML blobs, the broad io_port
|
//! `acpi-tables` node the kernel publishes (the AML blobs, the broad io_port
|
||||||
//! grant, a broad irq window, the SCI), and runs the **shared AML module** in
|
//! grant, a broad irq window, the SCI), and runs the **shared AML module** in
|
||||||
//! ring 3 — the same parser and interpreter the kernel uses.
|
//! ring 3 — the same parser and interpreter the kernel uses.
|
||||||
@@ -321,7 +321,7 @@ fn onSci() void {
|
|||||||
/// handler method (`_Lxx` level / `_Exx` edge), drain the Notify queue the
|
/// handler method (`_Lxx` level / `_Exx` edge), drain the Notify queue the
|
||||||
/// method produced, and publish an event per notified device. Then clear the
|
/// method produced, and publish an event per notified device. Then clear the
|
||||||
/// status bit. QEMU raises no GPEs on this config, so this path is exercised by
|
/// status bit. QEMU raises no GPEs on this config, so this path is exercised by
|
||||||
/// host unit tests (docs/m21-plan.md decision 5); on real hardware it carries
|
/// host unit tests (docs/acpi.md — ACPI events); on real hardware it carries
|
||||||
/// battery/AC/lid. The embedded controller's `_Qxx` queries are out of scope.
|
/// battery/AC/lid. The embedded controller's `_Qxx` queries are out of scope.
|
||||||
fn handleGpe() void {
|
fn handleGpe() void {
|
||||||
handleGpeBlock(gpe0_blk, gpe0_len, 0);
|
handleGpeBlock(gpe0_blk, gpe0_len, 0);
|
||||||
@@ -476,7 +476,7 @@ fn walkDevices(node: *aml.Node, interpreter: *aml.Interpreter) void {
|
|||||||
|
|
||||||
if (readHid(c, interpreter)) |hid| {
|
if (readHid(c, interpreter)) |hid| {
|
||||||
// Skip PCI roots — pci-bus already reports PCI functions; ACPI adds
|
// Skip PCI roots — pci-bus already reports PCI functions; ACPI adds
|
||||||
// only the non-PCI _HID devices (docs/m19-m20-plan.md M20.2). The two
|
// only the non-PCI _HID devices (docs/device-manager.md — matching). The two
|
||||||
// roots are named through the shared registry, not bare _HID strings.
|
// roots are named through the shared registry, not bare _HID strings.
|
||||||
const id = acpi_ids.HardwareId.fromHid(hid[0..7]);
|
const id = acpi_ids.HardwareId.fromHid(hid[0..7]);
|
||||||
if (id != .pci_bus and id != .pci_express_root_bridge) {
|
if (id != .pci_bus and id != .pci_express_root_bridge) {
|
||||||
|
|||||||
@@ -31,17 +31,6 @@ fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
|||||||
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The driver that serves each device — the policy table. In a fuller system
|
|
||||||
/// this comes from a manifest (docs/device-manager.md: the third bus type
|
|
||||||
/// triggers it); for now a static map. `null` = no driver for this class yet.
|
|
||||||
fn driverFor(d: device.DeviceDescriptor) ?[]const u8 {
|
|
||||||
// The HPET timer node is still kernel-seeded (from the HPET table, not AML).
|
|
||||||
// PS/2 and other _HID devices now arrive as acpi-service reports and match
|
|
||||||
// in onChildAdded (M20.3), not from this boot snapshot.
|
|
||||||
if (d.class == @intFromEnum(device.DeviceClass.timer)) return "hpet";
|
|
||||||
return null;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// The PCI class/subclass/prog-IF triple of an xHCI (USB 3) host controller —
|
/// The PCI class/subclass/prog-IF triple of an xHCI (USB 3) host controller —
|
||||||
/// Serial Bus Controller / USB Controller / XHCI — named from pci-class.zig rather
|
/// Serial Bus Controller / USB Controller / XHCI — named from pci-class.zig rather
|
||||||
/// than written as the bare 0x0C0330 (docs/coding-standards.md, "Named values").
|
/// than written as the bare 0x0C0330 (docs/coding-standards.md, "Named values").
|
||||||
@@ -107,7 +96,7 @@ const Driver = struct {
|
|||||||
// The assigned device id (becomes argv[1]), or protocol.no_device.
|
// The assigned device id (becomes argv[1]), or protocol.no_device.
|
||||||
device_id: u64 = protocol.no_device,
|
device_id: u64 = protocol.no_device,
|
||||||
// Whether this driver speaks the protocol (hello expected, deadline
|
// Whether this driver speaks the protocol (hello expected, deadline
|
||||||
// enforced). Legacy drivers (hpet, ps2-bus) are supervised and restarted
|
// enforced). Legacy drivers (e.g. ps2-bus) are supervised and restarted
|
||||||
// but not yet required to hello.
|
// but not yet required to hello.
|
||||||
speaks_protocol: bool = false,
|
speaks_protocol: bool = false,
|
||||||
process_id: u32 = 0,
|
process_id: u32 = 0,
|
||||||
@@ -346,19 +335,15 @@ fn initialise(endpoint: runtime.ipc.Handle) bool {
|
|||||||
addDriver("pci-bus", descriptor.id, true);
|
addDriver("pci-bus", descriptor.id, true);
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
// PCI functions no longer appear in the boot snapshot (M19.3): the
|
// Nothing else is matched from the boot snapshot today. The kernel-seeded
|
||||||
// pci-bus driver reports them, and onChildAdded matches from reports.
|
// HPET timer node is served by the kernel's own clock (docs/timers.md), not
|
||||||
const driver_name = driverFor(descriptor) orelse continue;
|
// a user-space driver; PCI functions and PS/2 _HID devices arrive later as
|
||||||
matched += 1;
|
// pci-bus / acpi-service reports and match in onChildAdded (docs/discovery.md).
|
||||||
// Skip a singleton that is already alive (the initial-ramdisk sweep test
|
// A fuller system's static class->driver manifest (docs/device-manager.md)
|
||||||
// starts every bundled binary bare, this manager included) — spawning a
|
// would slot in here.
|
||||||
// second instance would only lose the claim race and churn the log.
|
|
||||||
if (!alreadySupervised(driver_name) and !system.isProcessRunning(driver_name)) {
|
|
||||||
addDriver(driver_name, protocol.no_device, false);
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// The discovery service (docs/m19-m20-plan.md M20): one per firmware, packed
|
// The discovery service (docs/discovery.md): one per firmware, packed
|
||||||
// under the neutral name "discovery", spawned once at startup. It finds and
|
// under the neutral name "discovery", spawned once at startup. It finds and
|
||||||
// claims the acpi-tables (or devicetree-blob) node itself. Not a per-device
|
// claims the acpi-tables (or devicetree-blob) node itself. Not a per-device
|
||||||
// match — it is the discoverer, not a driver bound to one device.
|
// match — it is the discoverer, not a driver bound to one device.
|
||||||
|
|||||||
@@ -1,5 +1,5 @@
|
|||||||
//! /system/services/fdt — the devicetree discovery service: the ARM twin of the
|
//! /system/services/fdt — the devicetree discovery service: the ARM twin of the
|
||||||
//! acpi service (docs/m19-m20-plan.md decision 7). **Placeholder: not
|
//! acpi service (docs/discovery.md — firmware neutrality). **Placeholder: not
|
||||||
//! implemented.** It exists so the build's `-Ddiscovery` option has both of its
|
//! implemented.** It exists so the build's `-Ddiscovery` option has both of its
|
||||||
//! values from day one; the implementation lands with the Raspberry Pi
|
//! values from day one; the implementation lands with the Raspberry Pi
|
||||||
//! bring-up (docs/arm.md).
|
//! bring-up (docs/arm.md).
|
||||||
@@ -15,7 +15,7 @@
|
|||||||
//! resident under the manager's supervision (hello, restart, the usual
|
//! resident under the manager's supervision (hello, restart, the usual
|
||||||
//! contract).
|
//! contract).
|
||||||
//!
|
//!
|
||||||
//! Known prerequisite recorded in the plan: `DeviceDescriptor`'s 8-byte `hid`
|
//! Known prerequisite recorded in docs/discovery.md: `DeviceDescriptor`'s 8-byte `hid`
|
||||||
//! cannot hold an FDT `compatible` string ("brcm,bcm2835-aux-uart") — identity
|
//! cannot hold an FDT `compatible` string ("brcm,bcm2835-aux-uart") — identity
|
||||||
//! widens before this file grows a body.
|
//! widens before this file grows a body.
|
||||||
|
|
||||||
|
|||||||
@@ -1,8 +1,8 @@
|
|||||||
//! The power protocol (docs/m21-plan.md): system power's domain-named surface,
|
//! The power protocol (docs/power.md): system power's domain-named surface,
|
||||||
//! registered under `ServiceId.power`. On x86 the acpi service serves it; on
|
//! registered under `ServiceId.power`. On x86 the acpi service serves it; on
|
||||||
//! ARM a PSCI/mailbox service will register the same id — subscribers never
|
//! ARM a PSCI/mailbox service will register the same id — subscribers never
|
||||||
//! learn which firmware they are on (m19-m20-plan.md decision 7). The
|
//! learn which firmware they are on (docs/discovery.md — firmware neutrality).
|
||||||
//! vfs-protocol pattern: extern-struct messages, a version, reserved fields.
|
//! The vfs-protocol pattern: extern-struct messages, a version, reserved fields.
|
||||||
|
|
||||||
/// The protocol version a client states nowhere yet — reserved for the day a
|
/// The protocol version a client states nowhere yet — reserved for the day a
|
||||||
/// handshake needs it; requests carry it so a mismatch can be refused loudly.
|
/// handshake needs it; requests carry it so a mismatch can be refused loudly.
|
||||||
|
|||||||
+22
-19
@@ -177,6 +177,15 @@ CASES = [
|
|||||||
"smp": 4,
|
"smp": 4,
|
||||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
|
# TSC clocksource + cross-core warp check (real Intel/AMD / KVM path). TCG won't
|
||||||
|
# advertise an invariant TSC, so the kernel forces the TSC clocksource on for this
|
||||||
|
# case (gated in kernel.zig) and runs the per-AP warp check across the 4 cores;
|
||||||
|
# their TSCs are synchronized, so it stays on the TSC (no HPET fallback). The rest
|
||||||
|
# of the suite exercises the HPET fallback instead. See docs/timers.md.
|
||||||
|
{"name": "tsc-sync",
|
||||||
|
"smp": 4,
|
||||||
|
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||||
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
{"name": "fault-ud", "expect": r"invalid opcode \(vector 6\)"},
|
{"name": "fault-ud", "expect": r"invalid opcode \(vector 6\)"},
|
||||||
{"name": "fault-pf", "expect": r"page fault \(vector 14\)"},
|
{"name": "fault-pf", "expect": r"page fault \(vector 14\)"},
|
||||||
{"name": "fault-df", "expect": r"double fault \(vector 8\)"},
|
{"name": "fault-df", "expect": r"double fault \(vector 8\)"},
|
||||||
@@ -284,7 +293,7 @@ CASES = [
|
|||||||
"fail": r"acpi-parse: mismatch|DANOS-TEST-RESULT: FAIL"},
|
"fail": r"acpi-parse: mismatch|DANOS-TEST-RESULT: FAIL"},
|
||||||
# M20.3: the flip — ps2-bus now comes up from the acpi service's report, not
|
# M20.3: the flip — ps2-bus now comes up from the acpi service's report, not
|
||||||
# a kernel-built node. Ordered: report -> spawn -> the driver attaches its
|
# a kernel-built node. Ordered: report -> spawn -> the driver attaches its
|
||||||
# keyboard, proving discovery runs entirely in ring 3 (docs/m19-m20-plan.md).
|
# keyboard, proving discovery runs entirely in ring 3 (docs/discovery.md).
|
||||||
{"name": "acpi-ps2",
|
{"name": "acpi-ps2",
|
||||||
"smp": 4,
|
"smp": 4,
|
||||||
"timeout": 150,
|
"timeout": 150,
|
||||||
@@ -294,7 +303,7 @@ CASES = [
|
|||||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
# M21.1: the SCI + power button. Boot the manager (which spawns the acpi
|
# M21.1: the SCI + power button. Boot the manager (which spawns the acpi
|
||||||
# service); ~4s in, QMP system_powerdown raises the ACPI power-button fixed
|
# service); ~4s in, QMP system_powerdown raises the ACPI power-button fixed
|
||||||
# event; the service's SCI handler must log the press (docs/m21-plan.md).
|
# event; the service's SCI handler must log the press (docs/acpi.md).
|
||||||
{"name": "power-button",
|
{"name": "power-button",
|
||||||
"smp": 4,
|
"smp": 4,
|
||||||
"timeout": 60,
|
"timeout": 60,
|
||||||
@@ -305,7 +314,7 @@ CASES = [
|
|||||||
# ~5s in, QMP system_powerdown raises the power button; the acpi service
|
# ~5s in, QMP system_powerdown raises the power button; the acpi service
|
||||||
# publishes it, init stops its children then requests S5, and QEMU exits.
|
# publishes it, init stops its children then requests S5, and QEMU exits.
|
||||||
# The ordered regex proves button -> shutting-down -> entering-S5; the case
|
# The ordered regex proves button -> shutting-down -> entering-S5; the case
|
||||||
# passes on QEMU's self-exit through S5 (docs/m21-plan.md).
|
# passes on QEMU's self-exit through S5 (docs/power.md).
|
||||||
{"name": "orderly-shutdown",
|
{"name": "orderly-shutdown",
|
||||||
"smp": 4,
|
"smp": 4,
|
||||||
"timeout": 90,
|
"timeout": 90,
|
||||||
@@ -316,7 +325,7 @@ CASES = [
|
|||||||
"fail": r"power: S5 write did not take|DANOS-TEST-RESULT: FAIL"},
|
"fail": r"power: S5 write did not take|DANOS-TEST-RESULT: FAIL"},
|
||||||
# M20.2: the acpi service evaluates _CRS/_STA in ring 3 and registers +
|
# M20.2: the acpi service evaluates _CRS/_STA in ring 3 and registers +
|
||||||
# reports its _HID devices — the two PS/2 nodes must appear with resources
|
# reports its _HID devices — the two PS/2 nodes must appear with resources
|
||||||
# (keyboard: io 0x60/0x64 + IRQ = 3; mouse: IRQ = 1) (docs/m19-m20-plan.md).
|
# (keyboard: io 0x60/0x64 + IRQ = 3; mouse: IRQ = 1) (docs/discovery.md).
|
||||||
{"name": "acpi-report",
|
{"name": "acpi-report",
|
||||||
"smp": 4,
|
"smp": 4,
|
||||||
"timeout": 150,
|
"timeout": 150,
|
||||||
@@ -378,27 +387,21 @@ CASES = [
|
|||||||
{"name": "input",
|
{"name": "input",
|
||||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
# IO passthrough + IRQ-as-IPC: a user-space HPET driver maps device MMIO into
|
# Device manager: a ring-3 service enumerates /system/devices, matches the PCI host
|
||||||
# its own address space, binds the device's interrupt to an IPC endpoint, and
|
# bridge to pci-bus, and spawns it — end-to-end proof of discover -> match -> spawn
|
||||||
# is woken by the hardware five times while blocked (never polling).
|
# -> driver-up (the spawned pci-bus logs "<N> functions found").
|
||||||
{"name": "hpet",
|
|
||||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
|
||||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
|
||||||
# Device manager: a ring-3 service enumerates /system/devices and matches each
|
|
||||||
# device to a driver (discovery + policy in user space). This increment logs the
|
|
||||||
# decision; spawning follows.
|
|
||||||
{"name": "device-manager",
|
{"name": "device-manager",
|
||||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
# Bus driver: a user process claims a device, enumerates its children from the
|
# device_register containment (in-kernel): registering a child whose MMIO window
|
||||||
# hardware, and publishes each with dev_register — and the kernel refuses a child
|
# escapes its parent's grant is refused (NotContained) — else dev_register would map
|
||||||
# whose window escapes the parent's (else dev_register maps arbitrary memory).
|
# arbitrary physical memory — while an identical re-register stays idempotent.
|
||||||
{"name": "bus",
|
{"name": "containment",
|
||||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
# IRQ teardown: an exiting driver's line is masked and its slot cleared (so no
|
# IRQ teardown: an exiting driver's line is masked and its slot cleared (so no
|
||||||
# ISR notifies a freed endpoint), and a sibling owner sharing that endpoint
|
# ISR notifies a freed endpoint), and a sibling owner sharing that endpoint
|
||||||
# keeps its own binding. The path hpet never takes, since it runs forever.
|
# keeps its own binding. A long-running driver never reaches this teardown path.
|
||||||
{"name": "irqfree",
|
{"name": "irqfree",
|
||||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
@@ -450,7 +453,7 @@ def qmp_send(path, command):
|
|||||||
"""One QMP command: connect, capabilities handshake, execute. Raises on any
|
"""One QMP command: connect, capabilities handshake, execute. Raises on any
|
||||||
failure — the caller retries until the guest's socket is ready. This is how
|
failure — the caller retries until the guest's socket is ready. This is how
|
||||||
a case injects a host-side event (system_powerdown = the ACPI power button)
|
a case injects a host-side event (system_powerdown = the ACPI power button)
|
||||||
into the running guest (docs/m21-plan.md)."""
|
into the running guest (docs/power.md)."""
|
||||||
sock = socket.socket(socket.AF_UNIX, socket.SOCK_STREAM)
|
sock = socket.socket(socket.AF_UNIX, socket.SOCK_STREAM)
|
||||||
sock.settimeout(5)
|
sock.settimeout(5)
|
||||||
try:
|
try:
|
||||||
|
|||||||
Reference in New Issue
Block a user