Compare commits
60
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
452080e997 | ||
|
|
5b63a841ba | ||
|
|
4b9507bd59 | ||
|
|
7dec1b0767 | ||
|
|
45b8fd8614 | ||
|
|
dd22bfbc48 | ||
|
|
6e60daed6a | ||
|
|
446f655c69 | ||
|
|
a785efa4a3 | ||
|
|
767a2a9a7c | ||
|
|
dd044fb115 | ||
|
|
3ec14509a0 | ||
|
|
1f2c60b3ec | ||
|
|
d71a5f25d3 | ||
|
|
2a0f17ae86 | ||
|
|
9ef61a0844 | ||
|
|
688b9101e8 | ||
|
|
1d7ba814dc | ||
|
|
8aba86b4ce | ||
|
|
a0c83f4b3f | ||
|
|
1ea48ed5d6 | ||
|
|
dfc7d6a609 | ||
|
|
77a3ccd33d | ||
|
|
07da27dc39 | ||
|
|
d89657d0a4 | ||
|
|
849b4b62d4 | ||
|
|
8589bf713b | ||
|
|
738f6aa697 | ||
|
|
01e56e3f36 | ||
|
|
d5d15cefcb | ||
|
|
fd96a35eb9 | ||
|
|
e3fe3f3f45 | ||
|
|
60da667b42 | ||
|
|
36145e623b | ||
|
|
565415327d | ||
|
|
bf6bdb389d | ||
|
|
0628944b15 | ||
|
|
e6d0bb7ef0 | ||
|
|
5ca804d827 | ||
|
|
a299363b59 | ||
|
|
d8dd62c639 | ||
|
|
d106b6e8dc | ||
|
|
af2c766f42 | ||
|
|
d26262bf56 | ||
|
|
10b89c06ff | ||
|
|
a2a05d0b3d | ||
|
|
75d62660b0 | ||
|
|
a53c2b0193 | ||
|
|
bf481c080c | ||
|
|
3a78dcab3f | ||
|
|
470f93a83d | ||
|
|
7798706b41 | ||
|
|
ad40de03c2 | ||
|
|
d8778b4b70 | ||
|
|
79d859a111 | ||
|
|
37fb09f75e | ||
|
|
34ebeb968d | ||
|
|
3cc1d38dd0 | ||
|
|
36e804b848 | ||
|
|
be83a42d42 |
@@ -82,6 +82,10 @@ straight into CI.
|
||||
Design notes explaining *why* behind the code live in
|
||||
[`docs/`](docs/README.md) — start with [`docs/README.md`](docs/README.md).
|
||||
|
||||
For the hardware needed to run DanOS — minimum specs plus a plain-language guide
|
||||
matching Intel/AMD CPU generations by name — see
|
||||
[`docs/system-requirements.md`](docs/system-requirements.md).
|
||||
|
||||
## Logo
|
||||
|
||||
San Serif Text "Dan OS" with a black karate belt around it.
|
||||
|
||||
+6
-6
@@ -29,7 +29,7 @@ pub fn main() uefi.Status {
|
||||
// report the reason (boot services are still up) and park the machine so the
|
||||
// message stays on screen.
|
||||
boot() catch |err| {
|
||||
log("\r\ndanos: boot failed: ");
|
||||
log("\r\nEFI: boot failed: ");
|
||||
logBytes(@errorName(err));
|
||||
log("\r\n");
|
||||
while (true) asm volatile ("hlt");
|
||||
@@ -65,14 +65,14 @@ fn boot() !noreturn {
|
||||
|
||||
// Best effort: a volume without /system/services/init still boots (kernel-only).
|
||||
loadInit(bs, &boot_information) catch |err| {
|
||||
log("danos: no /system/services/init (");
|
||||
log("EFI: no /system/services/init (");
|
||||
logBytes(@errorName(err));
|
||||
log(") - booting without user space\r\n");
|
||||
};
|
||||
|
||||
// Best effort: the initial_ramdisk (VFS server + drivers) is optional too.
|
||||
loadInitialRamdisk(bs, &boot_information) catch |err| {
|
||||
log("danos: no initial_ramdisk (");
|
||||
log("EFI: no initial_ramdisk (");
|
||||
logBytes(@errorName(err));
|
||||
log(")\r\n");
|
||||
};
|
||||
@@ -84,7 +84,7 @@ fn boot() !noreturn {
|
||||
// the map and exiting would invalidate the map key.
|
||||
const cr3 = try buildBootstrapTables(bs, &boot_information);
|
||||
|
||||
log("danos: kernel loaded, exiting boot services\r\n");
|
||||
log("EFI: kernel loaded, exiting boot services\r\n");
|
||||
boot_information.memory_map = try exitBootServices(bs);
|
||||
|
||||
// Switch onto our tables and jump to the kernel in one uninterruptible step.
|
||||
@@ -395,7 +395,7 @@ fn loadInit(bs: *uefi.tables.BootServices, boot_information: *BootInformation) !
|
||||
const image = try loadFile(bs, init_file_name);
|
||||
boot_information.init_base = @intFromPtr(image.ptr);
|
||||
boot_information.init_len = image.len;
|
||||
log("danos: /system/services/init loaded\r\n");
|
||||
log("EFI: /system/services/init loaded\r\n");
|
||||
}
|
||||
|
||||
/// Ferry the initial_ramdisk (the VFS server + drivers) to the kernel, same as init.
|
||||
@@ -403,7 +403,7 @@ fn loadInitialRamdisk(bs: *uefi.tables.BootServices, boot_information: *BootInfo
|
||||
const image = try loadFile(bs, initial_ramdisk_file_name);
|
||||
boot_information.initial_ramdisk_base = @intFromPtr(image.ptr);
|
||||
boot_information.initial_ramdisk_len = image.len;
|
||||
log("danos: initial_ramdisk loaded\r\n");
|
||||
log("EFI: initial_ramdisk loaded\r\n");
|
||||
}
|
||||
|
||||
/// Validate the ELF, copy every PT_LOAD segment to its physical address, and
|
||||
|
||||
@@ -131,6 +131,13 @@ pub fn build(b: *std.Build) void {
|
||||
});
|
||||
// ACPI/PnP hardware-ID (_HID) names — the flat analog of pci-class for acpi_device
|
||||
// nodes. Also shared reference data.
|
||||
// The AML interpreter, a build module so the ring-3 acpi service can run the
|
||||
// same parser the kernel does (docs/discovery.md — the shared AML module).
|
||||
// Pure Zig, no kernel imports — one source, two builds.
|
||||
const aml_module = b.addModule("aml", .{
|
||||
.root_source_file = b.path("system/devices/aml/aml.zig"),
|
||||
});
|
||||
|
||||
const acpi_ids_module = b.addModule("acpi-ids", .{
|
||||
.root_source_file = b.path("system/devices/acpi-ids.zig"),
|
||||
});
|
||||
@@ -212,6 +219,19 @@ pub fn build(b: *std.Build) void {
|
||||
},
|
||||
});
|
||||
|
||||
// The device-manager protocol: hello + (M18.2) tree reports, exposed as its
|
||||
// own module like the other protocol modules. Imported through the runtime.
|
||||
const device_manager_protocol_module = b.addModule("device-manager-protocol", .{
|
||||
.root_source_file = b.path("system/services/device-manager/device-manager-protocol.zig"),
|
||||
});
|
||||
runtime_module.addImport("device-manager-protocol", device_manager_protocol_module);
|
||||
|
||||
// The power protocol: system power's domain-named surface (docs/power.md).
|
||||
const power_protocol_module = b.addModule("power-protocol", .{
|
||||
.root_source_file = b.path("system/services/power/protocol.zig"),
|
||||
});
|
||||
runtime_module.addImport("power-protocol", power_protocol_module);
|
||||
|
||||
// Typed volatile MMIO register access + memory-ordering barriers, for drivers on
|
||||
// top of an mmio_map grant. Depends only on `builtin` (arch-conditional barriers);
|
||||
// no target set, so it inherits each driver's. See library/mmio/mmio.zig.
|
||||
@@ -326,13 +346,36 @@ pub fn build(b: *std.Build) void {
|
||||
// which unpacks it and spawns each program (system/initial-ramdisk.zig).
|
||||
const vfs_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "vfs", "system/services/vfs/vfs.zig");
|
||||
const vfstest_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "vfs-test", "system/services/vfs/vfs-test.zig");
|
||||
const hpet_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "hpet", "system/drivers/hpet/hpet.zig");
|
||||
const bus_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "bus", "system/drivers/bus/bus.zig");
|
||||
const ps2_bus_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-bus", "system/drivers/ps2-bus/ps2-bus.zig");
|
||||
const ps2_keyboard_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-keyboard", "system/drivers/ps2-bus/keyboard.zig");
|
||||
const ps2_mouse_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-mouse", "system/drivers/ps2-bus/mouse.zig");
|
||||
const usb_xhci_bus_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "usb-xhci-bus", "system/drivers/usb-xhci-bus/usb-xhci-bus.zig");
|
||||
const pci_bus_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "pci-bus", "system/drivers/pci-bus/pci-bus.zig");
|
||||
// The PCI bus driver decodes each function's class triple to human names in its
|
||||
// boot log (class/subclass/prog-IF), so pull in the shared pci-class reference.
|
||||
pci_bus_exe.root_module.addImport("pci-class", pci_class_module);
|
||||
// A test fixture, not a real driver: hellos to the device manager, then faults —
|
||||
// what the driver-restart scenario drives the crash-loop cap with.
|
||||
const crash_test_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "crash-test", "system/services/crash-test/crash-test.zig");
|
||||
const device_list_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "device-list", "system/services/device-list/device-list.zig");
|
||||
// The discovery service: one swappable process per firmware
|
||||
// (docs/discovery.md), bundled under the neutral ramdisk name
|
||||
// "discovery" so the device manager never learns which firmware it is on.
|
||||
// x86 boots describe hardware with ACPI; the Raspberry Pis hand over a
|
||||
// flattened device tree — the aarch64 target flips the default when it
|
||||
// lands (docs/arm.md). Both are placeholders until M20.1 (acpi) and the
|
||||
// ARM bring-up (fdt).
|
||||
const Discovery = enum { acpi, fdt };
|
||||
const discovery = b.option(Discovery, "discovery", "Which discovery service fills the ramdisk's 'discovery' slot (default: acpi)") orelse Discovery.acpi;
|
||||
const discovery_source: []const u8 = switch (discovery) {
|
||||
.acpi => "system/services/acpi/acpi.zig",
|
||||
.fdt => "system/services/fdt/fdt.zig",
|
||||
};
|
||||
const discovery_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "discovery", discovery_source);
|
||||
if (discovery == .acpi) discovery_exe.root_module.addImport("aml", aml_module);
|
||||
const device_manager_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "device-manager", "system/services/device-manager/device-manager.zig");
|
||||
// Names the xHCI PCI class triple from the shared taxonomy instead of a bare 0x0C0330.
|
||||
device_manager_exe.root_module.addImport("pci-class", pci_class_module);
|
||||
// The input service and its exercisers: the fan-out server, a hardware-free synthetic
|
||||
// source, and a subscriber that doubles as the `input` test's oracle. See docs/input.md.
|
||||
const input_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "input", "system/services/input/input.zig");
|
||||
@@ -351,10 +394,6 @@ pub fn build(b: *std.Build) void {
|
||||
mk_run.addFileArg(vfs_exe.getEmittedBin());
|
||||
mk_run.addArg("vfs-test");
|
||||
mk_run.addFileArg(vfstest_exe.getEmittedBin());
|
||||
mk_run.addArg("hpet");
|
||||
mk_run.addFileArg(hpet_exe.getEmittedBin());
|
||||
mk_run.addArg("bus");
|
||||
mk_run.addFileArg(bus_exe.getEmittedBin());
|
||||
mk_run.addArg("ps2-bus");
|
||||
mk_run.addFileArg(ps2_bus_exe.getEmittedBin());
|
||||
mk_run.addArg("ps2-keyboard");
|
||||
@@ -363,6 +402,14 @@ pub fn build(b: *std.Build) void {
|
||||
mk_run.addFileArg(ps2_mouse_exe.getEmittedBin());
|
||||
mk_run.addArg("usb-xhci-bus");
|
||||
mk_run.addFileArg(usb_xhci_bus_exe.getEmittedBin());
|
||||
mk_run.addArg("pci-bus");
|
||||
mk_run.addFileArg(pci_bus_exe.getEmittedBin());
|
||||
mk_run.addArg("crash-test");
|
||||
mk_run.addFileArg(crash_test_exe.getEmittedBin());
|
||||
mk_run.addArg("device-list");
|
||||
mk_run.addFileArg(device_list_exe.getEmittedBin());
|
||||
mk_run.addArg("discovery");
|
||||
mk_run.addFileArg(discovery_exe.getEmittedBin());
|
||||
mk_run.addArg("device-manager");
|
||||
mk_run.addFileArg(device_manager_exe.getEmittedBin());
|
||||
mk_run.addArg("input");
|
||||
@@ -382,8 +429,6 @@ pub fn build(b: *std.Build) void {
|
||||
.{ vfs_exe, "system/services" },
|
||||
.{ device_manager_exe, "system/services" },
|
||||
.{ input_exe, "system/services" },
|
||||
.{ hpet_exe, "system/drivers" },
|
||||
.{ bus_exe, "system/drivers" },
|
||||
.{ ps2_bus_exe, "system/drivers" },
|
||||
.{ ps2_keyboard_exe, "system/drivers" },
|
||||
.{ ps2_mouse_exe, "system/drivers" },
|
||||
@@ -530,6 +575,7 @@ pub fn build(b: *std.Build) void {
|
||||
"system/devices/device-abi.zig",
|
||||
"system/devices/pci-class.zig", // class/subclass/prog-IF name decoding
|
||||
"system/devices/acpi-ids.zig", // _HID name decoding
|
||||
"system/devices/aml/aml.zig", // AML parse + interpret, incl. Notify dispatch (M21)
|
||||
"system/devices/usb-abi.zig", // wire sizes + bit packings + set-up packet encodings
|
||||
"system/devices/usb-ids.zig", // class/subclass/protocol code assignments
|
||||
"library/mmio/mmio.zig", // barriers assemble + registers round-trip
|
||||
@@ -561,6 +607,21 @@ pub fn build(b: *std.Build) void {
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(xkb_tests).step);
|
||||
|
||||
// runtime.time's Instant/Duration arithmetic. time.zig pulls in system.zig (the
|
||||
// syscall wrappers), which needs the `abi` module, so it doesn't fit the plain
|
||||
// loop above.
|
||||
const time_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("library/runtime/time.zig"),
|
||||
.target = target,
|
||||
.optimize = optimize,
|
||||
.imports = &.{
|
||||
.{ .name = "abi", .module = abi_module },
|
||||
},
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(time_tests).step);
|
||||
|
||||
// Convenience: `zig build gen-xkeyboard-config` regenerates the layout tables from the
|
||||
// vendored data (offline). `fetch` (the network step) stays a manual script run.
|
||||
const gen_xkb = b.addSystemCommand(&.{ "python3", "tools/make-xkeyboard-config.py", "generate" });
|
||||
|
||||
+24
-9
@@ -45,20 +45,22 @@ rather than restate it. Roughly in the order things happen at runtime:
|
||||
until its hardware interrupts it**. The claim is the capability; `irq_ack` is the
|
||||
unmask.
|
||||
14. **[driver-model.md](driver-model.md) — buses, classes and host controllers.** How
|
||||
real driver stacks factor into three shapes, how families share code, and the
|
||||
proposed ABI for the three primitives still missing (capability passing, DMA +
|
||||
memory barriers, MSI).
|
||||
real driver stacks factor into three shapes and how families share code. The
|
||||
three primitives it proposed are long since built (M13 capability passing,
|
||||
M14 DMA + barriers, M15 MSI), and the driver *contract* on top of them —
|
||||
hello, supervision, restart — is built too (device-manager.md, M18).
|
||||
15. **[process-management.md](process-management.md) — process management.** The
|
||||
microkernel's `ps`/`kill`/SIGCHLD: enumerate as a table snapshot, the
|
||||
supervision link as the kill authority, and child-exit notifications over the
|
||||
same endpoints IRQs arrive on.
|
||||
16. **[process-lifecycle.md](process-lifecycle.md) — the process lifecycle.** Design:
|
||||
signals over IPC as the one lifecycle vocabulary every process speaks — the
|
||||
16. **[process-lifecycle.md](process-lifecycle.md) — the process lifecycle.** Built
|
||||
(M17): signals over IPC as the one lifecycle vocabulary every process speaks — the
|
||||
POSIX.1-1990 words with message delivery instead of stack hijack, the stable
|
||||
`runtime.process` interface, exit reasons, published exit events any stateful
|
||||
service can subscribe to (the VFS releasing dead clients' handles), and the two
|
||||
iron rules (cleanup is the kernel's job; kill is not a signal).
|
||||
17. **[device-manager.md](device-manager.md) — the device manager.** Design: the
|
||||
17. **[device-manager.md](device-manager.md) — the device manager.** Built (M18,
|
||||
through the app surface): the
|
||||
tree, the matcher, and the supervisor. Tree structure lives in the manager,
|
||||
authority stays in the kernel; bus drivers report what they see; drivers are
|
||||
restarted through the lifecycle vocabulary — the plan that turns
|
||||
@@ -83,6 +85,10 @@ Start with the north star:
|
||||
|
||||
Cutting across all of these:
|
||||
|
||||
- **[system-requirements.md](system-requirements.md) — system requirements.** The
|
||||
hardware needed to run danos: minimum specs (UEFI x86-64, ACPI, PCIe ECAM,
|
||||
xHCI, ~128 MiB RAM) grounded in what the boot path actually assumes, plus a
|
||||
plain-language guide matching Intel/AMD CPU generations by name.
|
||||
- **[arch.md](arch.md) — the architecture split.** How CPU-specific code is kept
|
||||
behind a build-time `arch` module so the generic kernel never names x86_64,
|
||||
leaving room for other systems (e.g. an AArch64 Raspberry Pi) later.
|
||||
@@ -94,7 +100,16 @@ Cutting across all of these:
|
||||
when to build it, and how to keep it architecture-agnostic.
|
||||
- **[acpi.md](acpi.md) — finding the ACPI tables.** The concrete x86 locator chain:
|
||||
how the loader captures the **RSDP**, hands its physical address across in `BootInfo`,
|
||||
and how the platform derives the **RSDT/XSDT** from it and walks the SDTs.
|
||||
and how the platform derives the **RSDT/XSDT** from it and walks the SDTs — plus the
|
||||
live event side (the SCI, the power button, GPE/Notify) the ring-3 acpi service runs.
|
||||
- **[power.md](power.md) — the power service.** System power as a domain-named
|
||||
service: button/lid/battery events published to subscribers, and init's orderly
|
||||
shutdown composing the [lifecycle](process-lifecycle.md) stop sequence with an ACPI
|
||||
S5 write. Firmware-neutral — a PSCI backend drops in on ARM.
|
||||
- **[timers.md](timers.md) — timers and time.** The ring-3 surface for reading the
|
||||
clock and waiting: why `now()` is a syscall rather than a service, and the one-shot
|
||||
timer notification (`timer_bind`) that gives supervisors a timed wait — built on the
|
||||
LAPIC heartbeat and calibrated TSC of [device-interrupts.md](device-interrupts.md).
|
||||
- **[smp.md](smp.md) — multiple cores.** A design/research note on how microkernels
|
||||
(L4, seL4) handle SMP — big kernel lock vs per-CPU vs multikernel — and how the
|
||||
right choice depends on whether danos is chasing real-time or resilience.
|
||||
@@ -151,7 +166,7 @@ addressed as **`system/services/init`** — the repeated leaf resolves away:
|
||||
| Source (root file) | Addressed as (module / binary / FHS path) |
|
||||
|----------------------------------------|--------------------------------------------|
|
||||
| `system/services/init/init.zig` | `system/services/init` → `/system/services/init` |
|
||||
| `system/drivers/hpet/hpet.zig` | `system/drivers/hpet` → `/system/drivers/hpet` |
|
||||
| `system/drivers/ps2-bus/ps2-bus.zig` | `system/drivers/ps2-bus` → `/system/drivers/ps2-bus` |
|
||||
| `library/runtime/runtime.zig` | `library/runtime` (the `runtime` module) |
|
||||
|
||||
In **source**, a sub-project is a directory so it can hold many files — the entry is
|
||||
@@ -217,6 +232,6 @@ appears in the private-ABI path.
|
||||
| danos-native runtime (`runtime`): syscall wrappers, heap, IPC, device access — the stable application ABI | `library/runtime/` |
|
||||
| POSIX/C compatibility (`posix`): unistd, stdio — the one place POSIX names are allowed | `library/posix/` |
|
||||
| System services (init, the VFS server + `protocol`, the device-manager) | `system/services/` |
|
||||
| Device drivers, one sub-project each (`hpet` leaf driver, `bus` bus driver) | `system/drivers/` |
|
||||
| Device drivers, one sub-project each (`pci-bus`, `ps2-bus`, `usb-xhci-bus` bus drivers) | `system/drivers/` |
|
||||
| Build + `run-x86-64` (QEMU/OVMF) | `build.zig` |
|
||||
| QEMU integration test harness | `test/qemu_test.py` |
|
||||
|
||||
+55
-1
@@ -107,12 +107,66 @@ firmware-agnostic [device model](discovery.md) gets populated; this note stops a
|
||||
part that answers "where are the tables?" — everything past the RSDP is just following
|
||||
more pointers the tables themselves provide.
|
||||
|
||||
## ACPI events: the SCI, the power button, and GPEs (M21)
|
||||
|
||||
The tables above are static description; ACPI is also a *live* channel. Hardware
|
||||
raises the **SCI** (System Control Interrupt) — one shared, level-triggered line
|
||||
whose vector the FADT names — and the OS reads status registers to learn what
|
||||
happened: a fixed event like the power button, or a **General-Purpose Event**
|
||||
(GPE) whose handler is an AML method. Since [discovery](discovery.md) moved AML
|
||||
to ring 3, the event side lives there too, in the same **acpi service** — the
|
||||
device discoverer and the event source are one process, because both need the
|
||||
namespace and the port grant.
|
||||
|
||||
**The kernel hands the service what it needs and no more.** Reading PM1 event
|
||||
blocks and GPE blocks requires the FADT, which the kernel already parses for its
|
||||
own `\_S5` poweroff. Rather than re-parse, the kernel appends the **FADT as one
|
||||
more memory resource** on the `acpi-tables` node; the service tells it apart
|
||||
from the AML blob resources by signature — the FADT keeps its intact `"FACP"`
|
||||
header, while the blob resources are header-stripped bytecode that starts with
|
||||
no signature. The kernel's own FADT parse is untouched; the service reads the
|
||||
PM1 *event* blocks (which the kernel never parsed — it only needs PM1 *control*
|
||||
for `\_S5`) and the GPE0/GPE1 blocks straight from its copy. The **SCI itself**
|
||||
arrives as the node's one `len == 1` irq resource (distinct from the broad
|
||||
`[0, 256)` window that covers children's legacy lines), which is how the service
|
||||
finds the line to `irq_bind`.
|
||||
|
||||
With those in hand the service enables ACPI mode (only if `SCI_EN` is clear —
|
||||
some firmwares boot with it already set), sets `PWRBTN_EN`, and on each SCI:
|
||||
|
||||
- **The power button** is a *fixed* event: a set `PWRBTN_STS` bit in PM1 status.
|
||||
The handler clears it (write-1-to-clear), logs the press, and publishes a
|
||||
[`power`](power.md) `power_button` event to subscribers.
|
||||
- **GPEs** are the general path: for each set-and-enabled GPE bit `n`, the
|
||||
service evaluates its `\_GPE._L%02X` (level) or `_E%02X` (edge) handler
|
||||
method, drains the **Notify** queue that method produced, maps each notified
|
||||
device to an event (battery, AC, lid, or a generic `notify` with its code),
|
||||
and clears the status bit. A missing handler method is clear-and-log, not an
|
||||
error. Making GPEs work required teaching the interpreter one opcode it never
|
||||
handled — `Notify` (`0x86`) — which it now folds into a bounded queue drained
|
||||
per evaluation; everything else a handler needs (field access, control flow,
|
||||
method calls) was already proven by the ring-3 `_STA`/`_CRS` work.
|
||||
|
||||
**How this is tested.** QEMU cannot raise GPEs deterministically on this config,
|
||||
so GPE/Notify correctness is proven by **host unit tests** — hand-encoded AML
|
||||
with a `Notify` inside a method body, run under `zig build test`. The QEMU
|
||||
`power-button` scenario proves the fixed-event path end to end: a QMP
|
||||
`system_powerdown` injects a real ACPI power-button press, and the service's SCI
|
||||
handler must log it. Battery/AC/lid and the embedded controller's `_Qxx` queries
|
||||
are interface-complete but validated on real hardware later.
|
||||
|
||||
The service surface these events are *published on* — subscription, the event
|
||||
vocabulary, and orderly shutdown — is the power service, [power.md](power.md).
|
||||
|
||||
## Related
|
||||
|
||||
- [efi.md](efi.md) — the loader that captures the RSDP before `ExitBootServices`.
|
||||
- [memory-map.md](memory-map.md) — the same loader-captures / kernel-consumes seam, and
|
||||
the ACPI-reclaim memory the RSDP lives in.
|
||||
- [discovery.md](discovery.md) — the broader (still-evolving) plan for turning these
|
||||
tables into one neutral device model shared with the ARM device-tree path.
|
||||
tables into one neutral device model shared with the ARM device-tree path, and how
|
||||
ACPI enumeration and events moved to the ring-3 acpi service.
|
||||
- [power.md](power.md) — the domain-named power service the ACPI event side publishes
|
||||
to (button, lid, battery) and its orderly-shutdown path into S5.
|
||||
- [arch.md](arch.md) — why the kernel reaches the device code through a `platform`
|
||||
module and never names ACPI directly.
|
||||
|
||||
@@ -96,7 +96,7 @@ That's all — no Unix-abbreviation exception. The source directories are full w
|
||||
(`system`, `library`, not `src`/`lib`), and there is no daemon `d` suffix: a driver
|
||||
lives in `system/drivers/` and a service in `system/services/`, so the *location*
|
||||
already says what it is. Encoding the role in the name too (`busd`, `vfsd`) is
|
||||
redundant — the program is just `bus`, `vfs`. Don't put in a name what its directory
|
||||
redundant — the program is just `ps2-bus`, `vfs`. Don't put in a name what its directory
|
||||
already tells you.
|
||||
|
||||
## A note on collisions
|
||||
@@ -138,10 +138,36 @@ single word or acronym needs no hyphen: `scheduler.zig`, `paging.zig`, `apic.zig
|
||||
conventions above — `snake_case` — because it's an identifier, not a filename.)
|
||||
|
||||
**A sub-project's entry point repeats its directory's name** — `init/init.zig`,
|
||||
`runtime/runtime.zig`, `hpet/hpet.zig` — and the sub-project is addressed by the
|
||||
`runtime/runtime.zig`, `ps2-bus/ps2-bus.zig` — and the sub-project is addressed by the
|
||||
*directory* (`system/services/init`, `library/runtime`), with the repeated leaf
|
||||
resolving away. See the repository-layout section of [README.md](README.md).
|
||||
|
||||
## Named values, not magic numbers
|
||||
|
||||
The naming rule has a twin: **a value with meaning gets a name, too.** The same
|
||||
principle drives both — a reader should never have to leave the code to understand it.
|
||||
An abbreviated *name* forces a reader to guess; a bare *number* forces them worse, out
|
||||
to a spec or a header or a comment three files away, to learn what the value even *is*.
|
||||
If `0x0C` is the PCI serial-bus class, the code says `BaseClass.serial_bus`, not `0x0C`;
|
||||
if `0x04` is the ACPI IRQ resource descriptor, it says `SmallResourceType.irq`, not
|
||||
`0x04`. The number is an implementation detail of the name — recorded once, where the
|
||||
name is defined, and never spelled again at a use site.
|
||||
|
||||
**Prefer an `enum`** when the values form a set (device classes, AML opcodes, resource
|
||||
descriptor types, states): the type then also says *which* set a value belongs to, and
|
||||
the compiler rejects a value from the wrong one. A lone `pub const` with a descriptive
|
||||
name suffices for a one-off (`const large_descriptor_bit = 0x80`). Reach for the enum
|
||||
the moment code elsewhere compares against, packs, or produces the value — a packed PCI
|
||||
class triple is written from named parts (`.serial_bus`, `.usb`, `.xhci`), never as
|
||||
`0x0C_03_30` under a comment that decodes the bytes.
|
||||
|
||||
The exceptions are the numbers that carry no hidden meaning: `0` and `1` as plain zero
|
||||
and one, an index step, a field width, a bit shift. `x + 1`, `buffer[0]`, and `<< 8`
|
||||
need no christening — there is nothing to look up. The test is exactly the naming test:
|
||||
*would a reader have to look this up to know what it means?* If yes, name it. This is
|
||||
what `opcodes.zig`'s `*_opcode` constants, `acpi-ids`'s `HardwareId`, and `pci-class`'s
|
||||
class enums already are — reference data defined once and named everywhere it is used.
|
||||
|
||||
## Why acronyms are the line
|
||||
|
||||
Because an acronym has no letters to restore. `MMIO` doesn't become "memory mapped
|
||||
|
||||
@@ -17,7 +17,7 @@ Most modern Unix and Unix-like operating systems follow the FHS. DanOS has its o
|
||||
| /srv | Site-specific data served by this system, such as data and scripts for web servers, data offered by FTP servers, and repositories for version control systems |
|
||||
| /system | DanOS operating system files (similar idea to C:\Windows). A true representation of danos — its layout mirrors the source tree, so `/system` is what danos *is*. |
|
||||
| /system/devices | danos virtual device tree e.g. similar to /sys on linux but with danos device tree conventions (the structures in the devices module) |
|
||||
| /system/drivers | driver binaries, one sub-project each (e.g. /system/drivers/hpet) |
|
||||
| /system/drivers | driver binaries, one sub-project each (e.g. /system/drivers/pci-bus, /system/drivers/ps2-bus) |
|
||||
| /system/services | system-service binaries — the VFS server, init, and other user-mode servers (e.g. /system/services/vfs, /system/services/init) |
|
||||
| /system/kernel | the kernel image |
|
||||
| /tmp | Directory for temporary files (see also /var/tmp). Often not preserved between system reboots and may be severely size-restricted. |
|
||||
@@ -61,8 +61,8 @@ to the driver in the order written, and a read consumes what is there. Terminals
|
||||
serial lines, keyboards and mice are all of this shape. These are the natural first
|
||||
device nodes in danos, because a character driver needs nothing the kernel doesn't
|
||||
already provide — it claims its device, maps its registers with `mmio_map`, and blocks
|
||||
on `replyWait` for either an interrupt or a client request. `system/drivers/hpet/hpet.zig` is already
|
||||
that program, minus the client half.
|
||||
on `replyWait` for either an interrupt or a client request. `system/drivers/ps2-bus/ps2-bus.zig`
|
||||
is already that program, minus the file-node client half.
|
||||
|
||||
The obstacle was never the file type; it is which hardware a ring-3 driver can reach.
|
||||
Direct `in`/`out` from user space is still a #GP (no TSS I/O bitmap, IOPL never raised),
|
||||
|
||||
@@ -78,6 +78,40 @@ preemption and wakeups (1 ms granularity); the **TSC** is the resolution you rea
|
||||
time at. Making `sleep` itself sub-millisecond would take a tickless one-shot
|
||||
timer — a later step.
|
||||
|
||||
### Is the TSC trustworthy? Invariant, and synchronized
|
||||
|
||||
A cycle counter is only a valid *clock* if two things hold, and danos checks both,
|
||||
because they decide whether we read time with a cheap `rdtsc` or fall back to the HPET.
|
||||
|
||||
**Invariant.** An old TSC counted core clock cycles, so it sped up and slowed down with
|
||||
frequency scaling — useless as wall time. Modern CPUs (all of danos's targets) provide an
|
||||
**invariant TSC**: a constant rate across P/C-states that never stops. The guarantee is a
|
||||
CPUID bit — leaf `0x80000007`, EDX bit 8 — on both Intel *and* AMD. danos reads it in
|
||||
`calibrate`, and a TSC that doesn't advertise it is not used as the clocksource. AMD is
|
||||
why this matters in practice: it doesn't populate the Intel leaf `0x15` that enumerates
|
||||
the TSC *frequency*, so danos already measures AMD's rate against the HPET — but a
|
||||
measured frequency without the invariance guarantee is not enough.
|
||||
|
||||
**Synchronized.** Each core has its own TSC. Even invariant ones can start at different
|
||||
values (a second socket, some firmware), so a thread migrating from a core reading
|
||||
`1_000_000` to one reading `999_000` would see time jump *backward*. danos runs a **warp
|
||||
check** as each application processor comes online (`checkWarpSource`, adapted from
|
||||
Linux's): the waking core and the BSP hammer a shared "highest seen" TSC under a lock,
|
||||
and if either ever reads below it, the cores' TSCs are skewed. It's pairwise because APs
|
||||
come up one at a time ([smp.md](smp.md)).
|
||||
|
||||
**The fallback.** When the TSC fails either test — non-invariant (a bare VM such as the
|
||||
default qemu64), or warped between cores — danos moves the monotonic clock onto the
|
||||
**HPET** main counter: one fixed-rate counter, so it can neither skew between cores nor
|
||||
drift with frequency. It costs a memory-mapped read instead of a register read, but it
|
||||
keeps time *accurate*, which is the whole point. The switch preserves the current value,
|
||||
so the clock never jumps. The boot log names the outcome:
|
||||
|
||||
```
|
||||
/system/kernel: clocksource tsc (TSC invariant: yes, synchronized: yes) # real Intel/AMD
|
||||
/system/kernel: clocksource hpet (TSC invariant: no, synchronized: yes) # a bare VM (TCG)
|
||||
```
|
||||
|
||||
## Two kinds of vector, one dispatch
|
||||
|
||||
The IDT now installs gates `0-47`: the 32 exceptions plus the device range. Every
|
||||
|
||||
+32
-4
@@ -1,6 +1,16 @@
|
||||
# The device manager
|
||||
|
||||
**Status: design.** The primitives this builds on are real ([process-management.md](process-management.md):
|
||||
**Status: the protocol and supervision are built** (M18.1, 2026-07-13): `hello`
|
||||
with its deadline, supervised spawn, restart with backoff, and the crash-loop
|
||||
cap are in — usb-xhci-bus is the first conforming driver, and the
|
||||
`driver-restart` scenario proves fault → backoff → re-claim → cap end to end.
|
||||
Tree reports are built too (M18.2, 2026-07-13): the xHCI driver scans its
|
||||
root-hub ports and reports each connected device (`child_added`); the manager
|
||||
mirrors them and prunes a dead reporter's children, and the `usb-report`
|
||||
scenario proves report → prune → respawn → re-report. The application surface is built (M18.3, 2026-07-13):
|
||||
`enumerate` and `subscribe` over IPC, with `device-list` as the first client —
|
||||
the manager is now the one answer to "what devices exist" for applications.
|
||||
The primitives underneath are real ([process-management.md](process-management.md):
|
||||
spawn/supervise/kill/exit-notification; [driver-model.md](driver-model.md): the device
|
||||
table as a capability system; [drivers.md](drivers.md): claim/map/IRQ), and the first
|
||||
per-device driver spawn works (the device manager matches the xHCI controller by PCI
|
||||
@@ -41,7 +51,16 @@ enumeration is a **pci-bus driver**: the manager spawns it against the host brid
|
||||
like any bus reports children. ACPI becomes an **acpi service** that interprets the
|
||||
tables and reports the namespace. The manager only orchestrates and merges. Moving
|
||||
AML interpretation out of ring 0 is its own project on its own track; nothing here
|
||||
depends on when it lands.
|
||||
depends on when it lands. (It landed: [discovery.md](discovery.md), M19–M20.)
|
||||
|
||||
`device_register` is **idempotent on exact match**: a re-registration with an
|
||||
identical (parent, class, identity, resources) tuple returns the existing id
|
||||
instead of appending a duplicate. The kernel table has no unregister, so without
|
||||
this a restarted registering bus would re-report its children as fresh nodes on
|
||||
every respawn. Idempotence is what makes restart-and-re-report sound for *every*
|
||||
reporting bus — pci-bus, the acpi service, a future fdt service — not just one,
|
||||
and it is why supervision (below) can prune a dead bus's subtree and trust the
|
||||
restarted instance to rebuild exactly the same ids.
|
||||
|
||||
## The protocol
|
||||
|
||||
@@ -126,8 +145,17 @@ published exit events, signals + `runtime.process`). On top of those:
|
||||
the mouse and keyboard QEMU already hangs off it.
|
||||
7. **App surface**: `enumerate`/`subscribe` over IPC; `device_enumerate` retreats
|
||||
to a manager-internal seam.
|
||||
8. **Discovery migration**: pci-bus driver first, acpi service second, kernel scan
|
||||
retired last. (AML-in-user-space is its own track.)
|
||||
8. **Discovery migration** — DONE (M19–M20, 2026-07-13): enumeration moved to
|
||||
ring 3 as swappable per-firmware discoverers — the pci-bus driver (M19) then
|
||||
the acpi service (M20), see [discovery.md](discovery.md); the kernel seeds
|
||||
only the host bridge and the acpi-tables node. Matching moved with it:
|
||||
`child_added` grew a `device_id` (the kernel-registered id, `no_device` for
|
||||
unregistered leaves like USB ports) and a firmware `hid`, and the manager now
|
||||
matches drivers from those **reports** rather than its boot-time snapshot. The
|
||||
PCI arm flipped in M19.3, the ACPI arm (ps2-bus matched from `_HID`) in M20.3
|
||||
— each in a single phase so no device is ever matched from both sources at
|
||||
once. The acpi service reports only the non-PCI `_HID` devices, since pci-bus
|
||||
already reports PCI functions (M20.2).
|
||||
|
||||
## Settled questions (2026-07-12)
|
||||
|
||||
|
||||
@@ -167,3 +167,81 @@ free; discovery on x86 is partly about *finding* what ARM just tells you.
|
||||
- [ipc.md](ipc.md) — the channels that interrupts-as-messages and the device manager
|
||||
will ride on.
|
||||
- [vision.md](vision.md) — why drivers belong in isolated user space at all.
|
||||
|
||||
## Update (M19.3, 2026-07-13): PCI enumeration left the kernel
|
||||
|
||||
The kernel now seeds only the `pci_host_bridge` node (ECAM window, MMIO
|
||||
apertures derived from the memory map's holes, bus range, and the 16-bit I/O
|
||||
window). The per-function walk moved to the ring-3 `pci-bus` driver
|
||||
([device-manager.md](device-manager.md)): it claims the bridge, repeats the
|
||||
ECAM scan through its mmio grant, and `device_register`s what it finds, which
|
||||
the device manager mirrors and matches. The ACPI namespace walk follows in M20;
|
||||
the static tables (MADT, HPET, MCFG, FADT + `\\_S5`) stay kernel-side.
|
||||
|
||||
## Update (M20.3, 2026-07-13): ACPI enumeration left the kernel too
|
||||
|
||||
The kernel no longer folds the AML namespace's Device objects into the device
|
||||
tree. It still parses the *static* tables (MADT for SMP, HPET for the tick, MCFG
|
||||
for the host bridge, FADT) and still builds the AML namespace — but only to read
|
||||
the `\\_S5` sleep type for poweroff. Device discovery is the ring-3 **acpi
|
||||
service** ([device-manager.md](device-manager.md)): it claims the `acpi-tables`
|
||||
node the kernel publishes (the AML blobs, a broad io_port grant, the SCI),
|
||||
re-parses the same blobs with the shared AML module, evaluates `_STA`/`_CRS`,
|
||||
and registers + reports each `_HID` device — the device manager matches drivers
|
||||
(ps2-bus) from those reports. With M19's pci-bus driver, discovery now runs
|
||||
entirely in user space; the kernel seeds only the host bridge and the
|
||||
acpi-tables node.
|
||||
|
||||
## Discovery is a swappable process per firmware (M19–M20)
|
||||
|
||||
Moving PCI and ACPI enumeration out of ring 0 was not just a relocation — it
|
||||
made discovery **firmware-neutral by construction**, which is the whole reason
|
||||
to do it before the second architecture rather than after. Everything at and
|
||||
above the [device-manager](device-manager.md) protocol — descriptors,
|
||||
containment, reports, matching, supervision — is generic and may never become
|
||||
x86-specific. Discovery is the single firmware-specific piece, and it is
|
||||
isolated as **one swappable process per firmware**:
|
||||
|
||||
- **x86** boots describe hardware with ACPI, so the discoverer is the **acpi
|
||||
service** ([acpi.md](acpi.md)): it claims the `acpi-tables` node and runs AML.
|
||||
- **The Raspberry Pis** hand over a flattened device tree, so the discoverer is
|
||||
an **fdt service**: it claims a `devicetree-blob` node and walks the tree —
|
||||
pure data, no bytecode, so it needs neither a port grant nor an interpreter,
|
||||
strictly simpler than ACPI. (A placeholder until the [aarch64](arm.md)
|
||||
bring-up fills it in.)
|
||||
|
||||
The device manager spawns the discoverer under the **neutral ramdisk name
|
||||
`discovery`** and never learns which firmware it is on; the build's
|
||||
`-Ddiscovery=acpi|fdt` option fills that slot (x86 defaults to `acpi`, the
|
||||
aarch64 target flips the default when it lands). The manager owns the device
|
||||
tree as *data* and touches no hardware, ever — firmware bytecode runs only
|
||||
inside the crashable, supervised discoverer, so an AML fault can never take
|
||||
down the supervisor.
|
||||
|
||||
Two consequences of neutrality bind on later work:
|
||||
|
||||
- **Cross-firmware surfaces are named by domain, not firmware.** System power is
|
||||
a [`power`](power.md) protocol, not an "ACPI events" protocol: on x86 the acpi
|
||||
service registers it, on ARM a PSCI/mailbox service registers the same
|
||||
`ServiceId.power`, and subscribers never learn the difference.
|
||||
- **Identity must widen before the fdt service exists.** `DeviceDescriptor`'s
|
||||
8-byte `hid` holds an EISA id but cannot hold an FDT `compatible` string
|
||||
(`"brcm,bcm2835-aux-uart"`); the identity field grows before the ARM path can
|
||||
report a real node.
|
||||
|
||||
Two supporting decisions keep the kernel's remaining slice honest:
|
||||
|
||||
- **The AML interpreter is a shared build module**, compiled into both the
|
||||
kernel and the acpi service — one source, two builds, no fork. The kernel
|
||||
links it for the `\_S5` poweroff evaluation, the service links it for
|
||||
everything else, and the `acpi-parse` test asserts the two produce the same
|
||||
device count across the ring-3 move.
|
||||
- **Bridge apertures come from the firmware memory map, not AML.** Registered
|
||||
PCI functions carry BAR resources, and `device_register` containment demands
|
||||
the bridge own windows that cover them. Those apertures are derived
|
||||
kernel-side from the boot memory map's MMIO holes (regions that are neither
|
||||
RAM nor tables) — mechanical, AML-free, and available at boot regardless of
|
||||
what later moved to user space. The acpi service's authority is likewise
|
||||
exactly one node: the `acpi-tables` node, whose broad io_port grant is the
|
||||
documented trust boundary for the one process allowed to run firmware
|
||||
bytecode.
|
||||
|
||||
+10
-9
@@ -57,8 +57,9 @@ is not an address window. Discovery is trusted; user space is not.
|
||||
|
||||
### What a bus driver looks like
|
||||
|
||||
`system/drivers/bus/bus.zig` is the smallest honest one. Its "bus" is the HPET's register block and
|
||||
its "devices" are the block's comparators:
|
||||
danos ships no demo bus driver — the real ones are `pci-bus`, `ps2-bus`, and
|
||||
`usb-xhci-bus`. The smallest *honest* shape, illustrated here with an HPET register block
|
||||
as the "bus" and its comparators as the "devices", is:
|
||||
|
||||
```zig
|
||||
_ = dev.claim(bus.id); // 1. own the bus
|
||||
@@ -78,8 +79,8 @@ for (0..n) |i| { // 3. publish each child
|
||||
|
||||
Each child is left **unclaimed**, which is the handoff: a comparator driver can now
|
||||
`device_claim` one and `mmio_map` it, and will see only its own 0x20-byte window. A child
|
||||
whose window escapes the bus is refused — `bus` asserts that, and the `bus` test
|
||||
asserts the kernel's table upholds it.
|
||||
whose window escapes the bus is refused; the in-kernel `containment` test asserts the
|
||||
kernel's table upholds that ([drivers.md](drivers.md)).
|
||||
|
||||
A USB device has *no* resources at all: `resource_count = 0`, because it's addressed
|
||||
through its controller, not by MMIO. That case is allowed and is the common one.
|
||||
@@ -143,7 +144,7 @@ If a class driver needs `mmio`, it has become an HCD and should be one.
|
||||
physically-contiguous, pinned, uncacheable, reclaim-on-teardown buffers with the
|
||||
physical address exposed (`pmm.allocContiguous`, a DMA arena, `mapUserDmaInto`).
|
||||
`dma_below_4g` caps the address for legacy engines; `dma_write_combining` is accepted
|
||||
but falls back to coherent until PAT is programmed. hpet is refactored onto `/lib/mmio`;
|
||||
but falls back to coherent until PAT is programmed. The bus drivers use `/lib/mmio`;
|
||||
no DMA driver consumes `dma_alloc` yet.
|
||||
- **M15** — interrupts for PCI devices, the MSI half. Discovery now gives every PCI
|
||||
function its 4 KiB ECAM config space as resource 0 (unblocking the capability walk
|
||||
@@ -302,8 +303,8 @@ rather than an out-struct. The rest of this section is the original design note.
|
||||
|
||||
**The blocker, and it's a hard one.** No PCI device can take an interrupt today.
|
||||
[`addBars`](system/devices/acpi.zig) records `.memory` and `.io_port` BARs and never an
|
||||
`.irq`; there is no `_PRT` parsing anywhere in the tree. `hpet` only works because the
|
||||
HPET advertises its own routing options in its own registers — a privilege no ordinary
|
||||
`.irq`; there is no `_PRT` parsing anywhere in the tree. The HPET is the one exception —
|
||||
it advertises its own interrupt routing in its own registers, a privilege no ordinary
|
||||
device has.
|
||||
|
||||
**The fix, in two halves.**
|
||||
@@ -326,7 +327,7 @@ which means **discovery should give each `pci_device` a `.memory` resource for i
|
||||
4 KiB ECAM slot**. That's a small change to `parseMcfg` and it unblocks the whole
|
||||
capability walk (MSI, MSI-X, PCIe extended caps) without any new syscall.
|
||||
|
||||
Note QEMU's HPET reports `Tn_FSB_INT_DEL_CAP = 0` — no MSI — so `hpet` can never
|
||||
Note QEMU's HPET reports `Tn_FSB_INT_DEL_CAP = 0` — no MSI — so an HPET timer could never
|
||||
exercise this path. The first MSI driver will be the first PCI driver.
|
||||
|
||||
## M16 — the IOMMU, and the honest caveat ◑ detection done, enforcement pending
|
||||
@@ -353,7 +354,7 @@ gap should be named rather than implied.
|
||||
|
||||
`M13` (capability passing) is independent of `M14`/`M15` and is the cheapest. It
|
||||
unlocks class drivers, which are the shape with no hardware requirements at all — you
|
||||
could write a real one against `bus`'s comparators tomorrow.
|
||||
could write a real one against any device a bus driver publishes tomorrow.
|
||||
|
||||
`M14` and `M15` together unlock the first HCD. `M14`'s barrier layer is worth landing
|
||||
on its own regardless: it's small, obviously correct, and stops every future driver
|
||||
|
||||
+60
-29
@@ -22,12 +22,12 @@ say.*
|
||||
|
||||
## How a driver gets started: discover, match, spawn
|
||||
|
||||
Nothing in the kernel decides that the HPET needs the `hpet` driver — that is policy,
|
||||
and policy lives in user space. Boot brings user space up as a three-level supervision
|
||||
hierarchy, each level owning one job:
|
||||
Nothing in the kernel decides that the PCI host bridge needs the `pci-bus` driver — that
|
||||
is policy, and policy lives in user space. Boot brings user space up as a three-level
|
||||
supervision hierarchy, each level owning one job:
|
||||
|
||||
```
|
||||
kernel ──spawns──► init (PID 1) ──spawns──► device-manager ──spawns──► hpet
|
||||
kernel ──spawns──► init (PID 1) ──spawns──► device-manager ──spawns──► pci-bus
|
||||
| | |
|
||||
spawns only init, the service supervisor: the driver supervisor: enumerates
|
||||
publishes the starts the system /system/devices, matches each device
|
||||
@@ -184,7 +184,11 @@ Two properties worth knowing:
|
||||
|
||||
## A whole driver
|
||||
|
||||
`system/drivers/hpet/hpet.zig` is ~150 lines and does all of it. The shape:
|
||||
A minimal leaf driver is only ~150 lines and does all of it. danos ships **no such
|
||||
example binary** — the driver model is proven by the real drivers (`pci-bus`, `ps2-bus`,
|
||||
`usb-xhci-bus`), and a teaching example belongs here, in the docs, rather than as a
|
||||
compiled program nobody runs. Illustrated with a hypothetical HPET timer driver, the
|
||||
shape is:
|
||||
|
||||
```zig
|
||||
const hpet = findHpet(buf) orelse return; // device_enumerate, look for
|
||||
@@ -209,8 +213,8 @@ while (...) {
|
||||
}
|
||||
```
|
||||
|
||||
The HPET is a good first driver for a reason that isn't obvious. Its *counter* is a
|
||||
clocksource — the only way to use it is to read it, so it proved `mmio_map` without
|
||||
The HPET makes a good illustration for a reason that isn't obvious. Its *counter* is a
|
||||
clocksource — the only way to use it is to read it, so it exercises `mmio_map` without
|
||||
needing interrupts at all. Its *comparators* are a clockevent, and can be configured
|
||||
**level-triggered** (`Tn_INT_TYPE_CNF`), which asserts a bit in `GENERAL_INT_STATUS`
|
||||
that the driver must write-1-to-clear. That's a genuine deassert step, so the full
|
||||
@@ -252,9 +256,10 @@ bus driver may only ever subdivide what it already owns.
|
||||
A device with **no resources** is legal and common. A USB device is reached through its
|
||||
controller, not by MMIO, so it gets `resource_count = 0`.
|
||||
|
||||
See [`system/drivers/bus/bus.zig`](../system/drivers/bus/bus.zig) for a complete one, and
|
||||
[driver-model.md](driver-model.md) for how bus drivers, class drivers and host
|
||||
controller drivers fit together.
|
||||
See [`system/drivers/pci-bus/pci-bus.zig`](../system/drivers/pci-bus/pci-bus.zig) for a
|
||||
real one — it claims a PCI host bridge, maps its ECAM window, and publishes each function
|
||||
it finds as a child — and [driver-model.md](driver-model.md) for how bus drivers, class
|
||||
drivers and host controller drivers fit together.
|
||||
|
||||
## What the kernel does not do for you
|
||||
|
||||
@@ -313,32 +318,38 @@ uncacheable, physical address exposed), and **memory barriers** (`/lib/mmio`'s
|
||||
|
||||
## Verifying it
|
||||
|
||||
The `hpet` test spawns `hpet` from the initial ramdisk and watches the serial log. The driver
|
||||
prints `hpet: ok` only after being woken five times, and its loop's only exit is
|
||||
through `replyWait` returning a notification — it cannot reach that line by polling.
|
||||
No demo driver ships to prove this end to end; the *real* drivers do, so the tests
|
||||
target them and the kernel primitives directly:
|
||||
|
||||
The last check doesn't trust the driver's self-report at all: the kernel reads the I/O
|
||||
APIC redirection entry back and asserts the line really is routed to a device vector,
|
||||
really is level-triggered, and really was left unmasked by the driver's final
|
||||
`irq_ack`.
|
||||
- **`device-manager`** — boots only the device manager, which discovers the PCI host
|
||||
bridge, matches `pci-bus`, and `system_spawn`s it. The test reads kernel state — the
|
||||
process table and the device tree — to confirm pci-bus came up and registered the
|
||||
functions it enumerated: the whole discover → match → spawn → driver-up chain.
|
||||
- **`acpi-ps2`** — a user-space driver (`ps2-bus`) is woken by its device's IRQ,
|
||||
delivered as an IPC notification, and attaches the keyboard: IRQ-as-IPC, end to end.
|
||||
- **`pci-scan`** — a user-space driver (`pci-bus`) maps its device's MMIO (the ECAM
|
||||
window) and walks it: `mmio_map`, end to end.
|
||||
- **`containment`** — the kernel refuses a `device_register` whose child window escapes
|
||||
the parent's grant (else it would be a syscall for mapping arbitrary memory), while an
|
||||
identical re-register stays idempotent. Asserted in-kernel, straight against the broker.
|
||||
- **`irqfree`** — the teardown path. Binds two owners to one shared endpoint, releases
|
||||
one, and reads the I/O APIC back: the departing owner's line is masked, the sibling's
|
||||
is not. That second half is why bindings are keyed on the owning *task* and not on the
|
||||
endpoint pointer — endpoints are shared, so releasing "everything pointing at this
|
||||
endpoint" would silently mask a live driver's device.
|
||||
- **`iopass`** — the `device_grant` teardown rule, so destroying a driver's address
|
||||
space never returns MMIO frames to the RAM pool.
|
||||
|
||||
```
|
||||
$ python3 test/qemu_test.py hpet irqfree iopass
|
||||
hpet ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||
$ python3 test/qemu_test.py device-manager acpi-ps2 pci-scan containment irqfree iopass
|
||||
device-manager ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||
acpi-ps2 ... PASS
|
||||
pci-scan ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||
containment ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||
irqfree ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||
iopass ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||
```
|
||||
|
||||
Two companions cover what `hpet` can't, because it never exits:
|
||||
|
||||
- **`irqfree`** — the teardown path. Binds two owners to one shared endpoint, releases
|
||||
one, and reads the I/O APIC back: the departing owner's line is masked, the sibling's
|
||||
is not. That second half is why bindings are keyed on the owning *task* and not on
|
||||
the endpoint pointer — endpoints are shared, so releasing "everything pointing at
|
||||
this endpoint" would silently mask a live driver's device.
|
||||
- **`iopass`** — the `device_grant` teardown rule, so destroying a driver's address
|
||||
space never returns MMIO frames to the RAM pool.
|
||||
|
||||
## What's next (not done here)
|
||||
|
||||
The big driver-model pieces — capability passing (class drivers), DMA + barriers, MSI,
|
||||
@@ -362,3 +373,23 @@ the first DMA driver to protect and test against) and these smaller items:
|
||||
- **Interrupt priority / threaded IRQ latency.** `notifyFromIsr` enqueues the woken
|
||||
driver but doesn't preempt (`wakeLocked` deliberately leaves that to the caller), so
|
||||
a woken driver waits for the next scheduling point.
|
||||
|
||||
## The driver contract (M17–M18)
|
||||
|
||||
Claiming and mapping is half of being a danos driver; the other half is the
|
||||
**lifecycle and protocol contract**, and the runtime makes it nearly free:
|
||||
|
||||
- Build on `runtime.service.run` — one replyWait loop folding protocol
|
||||
requests, signals, and notifications into callbacks. The harness answers the
|
||||
universal zero-length ping and turns `terminate` into a clean exit for you
|
||||
([process-lifecycle.md](process-lifecycle.md)).
|
||||
- A driver spawned with an assignment (its device id as argv[1]) sends the
|
||||
versioned `hello` to the device manager inside the deadline, and a **bus**
|
||||
driver reports what it discovers with `child_added`
|
||||
([device-manager.md](device-manager.md); usb-xhci-bus is the reference
|
||||
implementation).
|
||||
- Crash freely — that is the design. The kernel releases your claims, IRQ
|
||||
bindings, and MSI vectors at death; the manager reads your exit reason,
|
||||
prunes what you reported, restarts you with backoff, and your fresh instance
|
||||
re-claims and re-reports. Never depend on your own cleanup running
|
||||
(iron rule 1).
|
||||
|
||||
@@ -1,191 +0,0 @@
|
||||
# M17–M18 execution plan: process lifecycle + device manager
|
||||
|
||||
The operational plan for building [process-lifecycle.md](process-lifecycle.md)
|
||||
(M17) and [device-manager.md](device-manager.md) increments 5–7 (M18). Design is
|
||||
settled in those documents; this file is the build order — one phase at a time,
|
||||
each phase green before the next starts. Delete or archive this file when M18
|
||||
lands.
|
||||
|
||||
**Definition of green, every phase:** `zig build` clean, `zig build test` clean,
|
||||
`python3 test/qemu_test.py` passes (existing scenarios plus the phase's new one),
|
||||
and the relevant design doc's "known gaps" / status lines updated. Commit per
|
||||
green phase (no co-author trailers).
|
||||
|
||||
**Workflow (settled 2026-07-12):** work happens in a dedicated git worktree, on
|
||||
feature branches cut from `main` — `feat/process-lifecycle` (M17.1–17.4),
|
||||
`feat/device-manager` (M18.1), `feat/usb-xhci-bus` (M18.2–18.3). When a branch's
|
||||
phases are all green it is **auto-merged into `main`**; branches are kept after
|
||||
merge, not deleted. Merges and branches are pushed to origin. Phase 0 (once):
|
||||
commit the design docs, merge the outstanding `feat/usb` work into `main`, and
|
||||
run the existing QEMU suite green before any new work starts.
|
||||
|
||||
**Numbering note:** continues the milestone sequence (driver track ended at M16).
|
||||
|
||||
## Status
|
||||
|
||||
The loop marks a phase `[x]` in the same commit that lands it. A phase is marked
|
||||
only when its definition of green holds.
|
||||
|
||||
- [x] **Phase 0** — baseline: docs committed, feat/usb merged to main, pushed;
|
||||
`usb-xhci-libary.zig` renamed to `usb-xhci-library.zig`; existing QEMU
|
||||
suite green from the worktree (48/48, 2026-07-12).
|
||||
- [x] **M17.1** — kernel releases claims/MSI on death (claims: `releaseAllOwnedBy`
|
||||
in the reap; MSI was already swept by `irq.releaseOwner`; `claim-release`
|
||||
test; suite 49/49)
|
||||
- [x] **M17.2** — exit reasons (`ExitReason` recorded at exit/fault/kill before
|
||||
the notification; `process_exit_reason` supervisor-gated;
|
||||
`runtime.process.exitReason`; kernel + ring-3 assertions; suite 49/49)
|
||||
- [x] **M17.3** — published exit events + VFS subscriber (`process_subscribe`,
|
||||
bounded ref-counted table, publish on every death;
|
||||
`runtime.process.subscribeExits`; VFS handles carry owners and are swept on
|
||||
the owner's death; `vfs-client-death` test; suite 50/50)
|
||||
- [x] **M17.4** — signals, timer notifications, `runtime.process`, the service
|
||||
harness (signal_bind/process_signal + coalescing pending mask; timer_bind
|
||||
on the tick; bindSignals/signalsFrom/sendSignal/stop + timerOnce;
|
||||
runtime.service.run with the zero-length ping; VFS converted; `signals`
|
||||
scenario; suite 51/51)
|
||||
- [ ] **merge** `feat/process-lifecycle` → main, push
|
||||
- [ ] **M18.1** — device-manager protocol: hello + restart policy (branch `feat/device-manager`)
|
||||
- [ ] **merge** `feat/device-manager` → main, push
|
||||
- [ ] **M18.2** — xHCI port scan + tree reports (branch `feat/usb-xhci-bus`)
|
||||
- [ ] **M18.3** — app surface: enumerate/subscribe + device-list
|
||||
- [ ] **merge** `feat/usb-xhci-bus` → main, push — **loop ends here**
|
||||
|
||||
---
|
||||
|
||||
## M17.1 — the kernel releases a dead process's claims
|
||||
|
||||
The cleanup half of iron rule 1; the prerequisite for every restart story.
|
||||
|
||||
- `system/kernel/devices-broker.zig`: `releaseAllOwnedBy(owner: u32)` — clear
|
||||
every `claimed[]` slot holding this task id.
|
||||
- `system/kernel/process.zig`: call it from the reap path, alongside the existing
|
||||
IRQ-binding release (the ordering comment there says why IRQs go first — claims
|
||||
slot in after them, before the exit notification).
|
||||
- MSI vectors: find where `msi_bind` records per-device vectors (interrupts
|
||||
module) and release those by owner in the same pass.
|
||||
- Docs: remove the claims bullet from process-management.md "Known gaps".
|
||||
|
||||
**Test:** new QEMU scenario `claim-release` — a test child claims an unclaimed
|
||||
device, is killed, is respawned, and claims the same device again successfully;
|
||||
assert both claims in the serial log. Kernel-side unit coverage in
|
||||
`system/kernel/tests.zig` for `releaseAllOwnedBy` (claim two devices as two owners,
|
||||
release one owner, verify exactly its claims freed).
|
||||
|
||||
## M17.2 — exit reasons
|
||||
|
||||
- `system/abi.zig`: `ExitReason` (exited, aborted, segmentation_fault,
|
||||
illegal_instruction, arithmetic_fault, killed).
|
||||
- Kernel: record the reason at every death site — clean exit path, each fault
|
||||
class in `onException`, the kill path. Bounded recent-exits table (ids are never
|
||||
reused, so a small ring keyed by id is enough).
|
||||
- New system call `process_exit_reason(id)` — supervisor-gated, like kill; returns
|
||||
the recorded reason or `-ESRCH` once evicted.
|
||||
- `library/runtime/process.zig`: `ExitReason` + `exitReason(id: u32)`.
|
||||
- Docs: remove the no-exit-status bullet from process-management.md.
|
||||
|
||||
**Test:** extend the `supervision` scenario — three children: one exits cleanly,
|
||||
one faults (the fault-recovery pattern), one is killed; the supervisor asserts all
|
||||
three reasons.
|
||||
|
||||
## M17.3 — published exit events
|
||||
|
||||
- Kernel: bounded subscriber table (endpoints); new system call
|
||||
`process_subscribe(endpoint)` (ungated, like `process_enumerate`); every death
|
||||
posts `notify_exit_bit | id` to each subscriber — the same post the supervisor
|
||||
path already uses.
|
||||
- `library/runtime/process.zig`: `subscribeExits(endpoint)`.
|
||||
- VFS becomes the first subscriber: on an exit event, release every handle keyed
|
||||
by that task id (badges already are task ids). Log the release.
|
||||
- Docs: note the convention in ipc.md (exit events reuse the exit-notification
|
||||
badge encoding).
|
||||
|
||||
**Test:** new QEMU scenario `vfs-client-death` — a client opens a file and is
|
||||
killed without closing; assert the VFS logs the handle release and its open-handle
|
||||
count returns to baseline.
|
||||
|
||||
## M17.4 — signals and the service harness
|
||||
|
||||
- Kernel: per-task pending mask + bound endpoint; system calls
|
||||
`signal_bind(endpoint)` and `process_signal(id, signal)` (supervisor-or-self
|
||||
gated); delivery posts `notify_signal_bit | pending mask`, coalescing; pending
|
||||
signals with no bound endpoint pend silently.
|
||||
- `library/runtime/process.zig`: `Signal`, `SignalSet`, `bindSignals`,
|
||||
`signalsFrom`, `sendSignal`, `stop(id, deadline_ms)` (terminate → wait for exit
|
||||
notification → kill). Implement `terminate`, `reload`, `user_1`, `user_2`;
|
||||
`interrupt`/`quit` are enum members with no sender yet; `alarm` stays unbuilt.
|
||||
- Kernel: **one-shot timer notifications** — `timer_bind(endpoint, ms)` posts a
|
||||
notification badge when the deadline lands (IRQ-as-IPC again, on the timer
|
||||
wheel `sleep` already uses). This is the missing timed-wait primitive:
|
||||
`replyWait` blocks forever and `sleep` blocks the whole process, but `stop()`'s
|
||||
escalation, the device manager's `hello` deadline (M18.1), and restart backoff
|
||||
all need a deadline while staying responsive. It is also the mechanism `alarm`
|
||||
gets for free later.
|
||||
- New `library/runtime/service.zig`: the harness — `run(callbacks)` owning the
|
||||
replyWait loop, folding protocol messages, signals, and child-exit notifications
|
||||
into `init` / `on_message` / `on_reload` / `on_terminate`; answers the common
|
||||
`ping` automatically. Define the reserved `ping` request encoding here and
|
||||
document it in ipc.md (one obvious encoding; smallest that cannot collide with
|
||||
existing protocols).
|
||||
- Convert one existing service (input-source or hpet) to the harness as proof it
|
||||
subtracts code rather than adding it.
|
||||
|
||||
**Test:** extend `supervision` — a harness-built child: `sendSignal(reload)`
|
||||
observed in its log, `ping` answered, `stop()` produces a clean exit with reason
|
||||
`exited`; a second child that ignores signals (no bind) is killed by `stop()`'s
|
||||
deadline with reason `killed`.
|
||||
|
||||
## M18.1 — device-manager protocol: hello + restart policy
|
||||
|
||||
- New `system/services/device-manager/device-manager-protocol.zig` module
|
||||
(vfs-protocol pattern): `hello { version, role, device_id }`; version constant;
|
||||
reserved fields.
|
||||
- Device manager: register the `.device_manager` endpoint; spawn drivers with its
|
||||
exit endpoint; enforce the hello deadline; restart policy — backoff, crash-loop
|
||||
cap (three fast deaths → mark failed, log, stop), reasons from M17.2 deciding
|
||||
restart vs not.
|
||||
- usb-xhci-bus: adopt the harness + send hello. hpet/ps2-bus follow only if the
|
||||
conversion is mechanical; otherwise they keep working unconverted (the manager
|
||||
only enforces hello on drivers spawned with an assignment).
|
||||
- build.zig: test-loop entry for the protocol module if it grows pure logic.
|
||||
|
||||
**Test:** new QEMU scenario `driver-restart` — the xHCI driver takes a test-only
|
||||
argv flag to fault after hello on its first run; assert: fault, exit reason
|
||||
recorded, manager respawns with backoff, second run claims the controller
|
||||
(M17.1) and hellos clean. Assert the crash-loop cap by a driver that always
|
||||
faults (a tiny test driver, not xhci).
|
||||
|
||||
## M18.2 — bus tree reports
|
||||
|
||||
- Protocol: `child_added { parent, identity, resources }` / `child_removed { id }`.
|
||||
- usb-xhci-bus: bring-up to **port scan only** — map the MMIO window (claimed in
|
||||
M16-era work), controller reset/start per xHCI spec, walk the port registers,
|
||||
report one `child_added` per connected port with speed + port number as
|
||||
identity. **No transfer rings, no descriptors** — reading device/interface
|
||||
descriptors (and therefore USB class triples for matching) is the follow-on USB
|
||||
track, not this plan.
|
||||
- Device manager: mirror reports into its tree; prune the subtree (emitting
|
||||
`child_removed`) when a bus driver dies; assert re-report on restart.
|
||||
|
||||
**Test:** QEMU already attaches usb-kbd + usb-mouse on xhci.0 — assert two
|
||||
`child_added` events reach the manager and appear in its tree dump; kill the
|
||||
driver, assert two `child_removed` then two fresh `child_added` after respawn.
|
||||
|
||||
## M18.3 — the application surface
|
||||
|
||||
- Protocol: `enumerate` (tree snapshot) + `subscribe` (published add/remove
|
||||
events, input-service pattern).
|
||||
- A small client (`device-list`, the `ps` analog) exercising both; the manager
|
||||
becomes the one answer to "what devices exist" for user space.
|
||||
`device_enumerate` stays for drivers/kernel seeding — its retreat is tied to the
|
||||
discovery migration, out of this plan.
|
||||
|
||||
**Test:** QEMU scenario — `device-list` shows the tree including USB children;
|
||||
during a driver restart the subscribing client logs remove + add events.
|
||||
|
||||
---
|
||||
|
||||
**Explicitly out of scope** (own tracks, after M18): discovery migration (pci-bus
|
||||
driver, acpi service, retiring the kernel scan), USB control transfers +
|
||||
descriptors + class-driver matching, the musl layer, `interrupt`/`quit` senders
|
||||
(needs a console), job control.
|
||||
+128
@@ -0,0 +1,128 @@
|
||||
# The power service: events and shutdown
|
||||
|
||||
A laptop lid closes, a battery drains, someone presses the power button — and
|
||||
several parts of the system might care: a session manager dims the screen, a
|
||||
logger notes it, and ultimately *something* has to turn the machine off. None of
|
||||
them owns the hardware that reported the event, and the reporter should not know
|
||||
who is listening. So system power is a **service**: an event source **publishes**
|
||||
button/lid/battery/AC events, interested processes **subscribe**, and one
|
||||
privileged caller — init — can ask it to power the machine off. It is the same
|
||||
publish/subscribe shape as the [input service](input.md), applied to power.
|
||||
|
||||
## Why a service, and why it is named for the domain, not the firmware
|
||||
|
||||
Where the events come from is firmware-specific — on x86 they ride the ACPI SCI
|
||||
([acpi.md](acpi.md)); on a Raspberry Pi they would come from PSCI or a mailbox.
|
||||
What subscribers want is not: *the lid closed* means the same thing regardless of
|
||||
who noticed. So the surface is **domain-named**. There is a `power-protocol`
|
||||
module and a well-known `ServiceId.power = 5`; on x86 the **acpi service**
|
||||
registers it, and on ARM a PSCI/mailbox service will register the *same* id.
|
||||
Subscribers call `runtime.ipc.lookup(.power)` and never learn which firmware they
|
||||
are on — the neutrality the whole [discovery](discovery.md) migration exists to
|
||||
preserve, carried one layer up into a running-system surface.
|
||||
|
||||
This is why the protocol is `power`, not "ACPI events": naming a cross-firmware
|
||||
surface after one firmware would leak x86 into code the ARM port must reuse
|
||||
unchanged.
|
||||
|
||||
## The protocol
|
||||
|
||||
The `power-protocol` module ([system/services/power/protocol.zig](../system/services/power/protocol.zig))
|
||||
follows the vfs-protocol pattern — extern-struct messages, a version, reserved
|
||||
fields. Three operations:
|
||||
|
||||
| Direction | Operation | Purpose |
|
||||
|---|---|---|
|
||||
| subscriber → service | `subscribe` | receive published events; the subscriber's endpoint rides as the call's **capability** (the input/device-manager pattern) |
|
||||
| init → service | `shutdown` | orderly shutdown's last step: enter S5 (soft off) |
|
||||
| service → subscriber | `event` | a published `EventMessage`, delivered as a buffered message (never sent *to* the service) |
|
||||
|
||||
Events are published, not polled: like the input service, the service holds
|
||||
subscriber endpoints as capabilities and `ipc_send`s each event as a buffered
|
||||
message, so a slow or dead subscriber can never wedge the source. The event
|
||||
vocabulary is hardware-neutral:
|
||||
|
||||
- `power_button` — the button was pressed (a fixed ACPI event on x86).
|
||||
- `lid`, `ac`, `battery` — the named GPE-driven events.
|
||||
- `notify` — a device notification that maps to none of the above; its `code`
|
||||
(the ACPI `Notify` argument) and the notifying device's `hid` say which device
|
||||
and what happened.
|
||||
|
||||
An `EventMessage` carries the `event` tag plus `code` and an 8-byte `hid`, so a
|
||||
generic `notify` is fully described without a second round trip.
|
||||
|
||||
**`shutdown` is authority, not information.** It is the only operation that
|
||||
*does* something irreversible, so it is gated: the contract is that only init
|
||||
(PID 1) may request it, because init is the process that has already run the stop
|
||||
sequence over everything else. The acpi service implements this as a **soft
|
||||
gate** — it honors `shutdown` only from a process that is a *subscriber*, and
|
||||
init is the one subscriber. That stands in for "only the system supervisor may
|
||||
power off" without hard-coding a pid, so it still holds under tests where PID 1
|
||||
is not init.
|
||||
|
||||
## Orderly shutdown
|
||||
|
||||
Powering off cleanly is where the power service, the [process
|
||||
lifecycle](process-lifecycle.md), and [ACPI events](acpi.md) compose. init
|
||||
already supervises the services it starts; for shutdown it runs **one event loop
|
||||
over one endpoint** that carries three things at once: its children's exit
|
||||
notifications, the lifecycle **signals** it can receive (`terminate`), and the
|
||||
**power events** it subscribes to — plus a re-arming heartbeat timer proving PID
|
||||
1 is alive. (init subscribes with retries, because the power service registers
|
||||
`.power` well after init starts; a missing power service is not fatal — a
|
||||
`terminate` signal drives the same path.)
|
||||
|
||||
On a `power_button` event or a `terminate` signal, init:
|
||||
|
||||
1. logs that it is shutting down,
|
||||
2. runs the standard stop sequence — `runtime.process.stop(child, deadline,
|
||||
endpoint)` — over its children **in reverse spawn order**, so the VFS stops
|
||||
last (other services may flush through it), each child getting the
|
||||
*terminate → deadline → kill* escalation from
|
||||
[process-lifecycle.md](process-lifecycle.md), and
|
||||
3. requests `.power` `shutdown`.
|
||||
|
||||
The service then enters **S5** (soft off) by writing `SLP_TYP | SLP_EN` to the
|
||||
PM1 control register(s) from ring 3, mirroring the kernel's own
|
||||
`system/devices/power.zig` `sleepValue`. If the write returns instead of powering
|
||||
the machine off, it logs loudly so a test fails rather than hangs.
|
||||
|
||||
**No new system call was needed for S5.** The broad io_port grant on the
|
||||
`acpi-tables` node ([discovery.md](discovery.md)) already put the PM1 control
|
||||
ports in the acpi service's hands, so writing S5 from ring 3 is something it
|
||||
could physically already do; formalizing it as a protocol operation added a
|
||||
contract, not authority. The kernel keeps `power.zig` for its own test paths and
|
||||
panic-time poweroff, where no user space is available to ask.
|
||||
|
||||
## Verifying it
|
||||
|
||||
Two QEMU scenarios exercise the path, both injecting a real ACPI power-button
|
||||
press via QMP `system_powerdown` (there is no other deterministic power event on
|
||||
this config):
|
||||
|
||||
- `power-button` proves the source: the acpi service's SCI handler logs the
|
||||
press and publishes `power_button` (the ACPI half is in [acpi.md](acpi.md)).
|
||||
- `orderly-shutdown` proves the whole composition: button → init logs shutting
|
||||
down → children stopped → the service enters S5 → QEMU exits. The ordered
|
||||
regex is the proof, and QEMU's self-exit through S5 is the pass.
|
||||
|
||||
## Scope
|
||||
|
||||
Interface-complete but validated on real hardware (the author's laptop) later,
|
||||
because QEMU does not emulate them: battery `_BST`/`_BIF` evaluation beyond the
|
||||
interface stubs, lid and AC events, and the embedded controller's `_Qxx`
|
||||
queries. Deliberately out of scope for now: reboot over the power protocol, S3
|
||||
sleep, per-device D-states (a future lifecycle-vocabulary extension, since
|
||||
"suspend" has the shape of a signal every driver must answer and has no consumer
|
||||
until laptop sleep), and thermal zones.
|
||||
|
||||
## See also
|
||||
|
||||
- [acpi.md](acpi.md) — where the events come from on x86: the SCI, the power
|
||||
button fixed event, and GPE/Notify dispatch in the acpi service.
|
||||
- [discovery.md](discovery.md) — why the surface is domain-named, and the
|
||||
firmware neutrality that makes a PSCI backend drop-in on ARM.
|
||||
- [process-lifecycle.md](process-lifecycle.md) — the stop sequence
|
||||
(`terminate → deadline → kill`) and signals init composes into shutdown.
|
||||
- [device-manager.md](device-manager.md) — the supervision model init mirrors for
|
||||
its own children.
|
||||
+12
-5
@@ -1,10 +1,17 @@
|
||||
# Resilience: fault isolation and live restart
|
||||
|
||||
Steps 1–2 of the ordering below are **built**: user-mode isolation, and fault →
|
||||
kill the process → keep the core (`onException` in `system/kernel/kernel.zig`; the
|
||||
`fault-recovery` test proves a crashing ring-3 process dies alone while the system
|
||||
keeps running). The supervisor notification and restart policy (steps 3+) are
|
||||
still design. This is the property danos is really chasing:
|
||||
Steps 1–4 of the ordering below are **built** (M17–M18, 2026-07-13): user-mode
|
||||
isolation; fault → kill the process → keep the core (`onException`; the
|
||||
`fault-recovery` test); the supervisor notification **with exit reasons**
|
||||
([process-lifecycle.md](process-lifecycle.md) — clean exit, fault class, or
|
||||
killed, recorded before the notice posts); and the **restart policy itself**
|
||||
([device-manager.md](device-manager.md)): the device manager supervises every
|
||||
driver, restarts crashes with backoff, caps crash loops, and re-claims work
|
||||
because the kernel releases a dead process's claims. The `driver-restart` and
|
||||
`usb-report` scenarios prove kill → release → respawn → re-claim → re-report
|
||||
end to end. What remains of this document's ladder is scope, not mechanism:
|
||||
more of the system moved into restartable processes (the discovery migration,
|
||||
[discovery.md](discovery.md), is the next rung). This is the property danos is really chasing:
|
||||
**if a part of the OS breaks, isolate it, and re-initialise it — without rebooting.**
|
||||
A crashed driver gets restarted; a wedged service gets killed and brought back. It's
|
||||
the reason the [microkernel](vision.md) shape was chosen, and it's a *separate* goal
|
||||
|
||||
@@ -0,0 +1,227 @@
|
||||
# System Requirements
|
||||
|
||||
Minimum and recommended hardware for running danos. Every requirement below is
|
||||
grounded in what the current code actually assumes at boot — this is a
|
||||
description of the real target, not an aspirational one.
|
||||
|
||||
## Summary
|
||||
|
||||
danos targets a **modern UEFI x86-64 PC with ACPI and PCIe**. The practical
|
||||
minimum is:
|
||||
|
||||
- 64-bit x86-64 CPU with SSE2, APIC, and `syscall`/`sysret`
|
||||
- UEFI firmware (no BIOS / legacy boot)
|
||||
- ACPI tables: MADT, MCFG, FADT
|
||||
- PCIe with an ECAM (MMConfig) window
|
||||
- **128 MiB RAM** (target); see [Memory](#memory) for the breakdown
|
||||
- USB via **xHCI only**
|
||||
|
||||
There is no support for legacy BIOS boot, x2APIC, port-IO PCI configuration, or
|
||||
any USB host controller other than xHCI.
|
||||
|
||||
## Plain-language hardware guide
|
||||
|
||||
If you don't want to cross-reference chipset datasheets, here's roughly what era
|
||||
of PC works. These are **guidance based on when the required features became
|
||||
standard**, not a list of tested machines — the authoritative rules are in the
|
||||
technical sections below.
|
||||
|
||||
The feature that sets the floor is **built-in xHCI USB** (danos supports no other
|
||||
USB controller) combined with **UEFI firmware**. Both became standard on
|
||||
mainstream desktops and laptops around **2012**.
|
||||
|
||||
| | Known-good baseline | Comfortable recommendation |
|
||||
|---|---|---|
|
||||
| **Intel** | 3rd-gen Core "Ivy Bridge" (2012) with a 7-series "Panther Point" chipset — Intel's first chipset with xHCI built in | 6th-gen Core "Skylake" (2015) or newer |
|
||||
| **AMD** | A-series "Llano" APU with an A75 FCH (2011) — the industry's first chipset with built-in xHCI | Any AM4 platform, i.e. Ryzen (2017) or newer |
|
||||
|
||||
**AMD is not behind Intel here — it was first.** AMD's A75 FCH shipped with
|
||||
native xHCI in April 2011, about a year *ahead* of Intel's 7-series (2012); AMD
|
||||
was the first vendor to earn USB-IF certification for chipset-level USB 3.0. The
|
||||
two "comfortable recommendation" dates differ only because they name convenient,
|
||||
long-supported product lines (Skylake, Ryzen) — not because of any USB
|
||||
capability gap. Every AMD desktop platform from the A75 FCH (2011) and FM2/AM3+
|
||||
era onward has built-in xHCI, and any of them qualifies as a baseline.
|
||||
|
||||
Older 64-bit machines (e.g. Intel Core 2, Nehalem, Sandy Bridge) meet the CPU
|
||||
requirements but typically **lack built-in xHCI and/or ship with BIOS instead of
|
||||
UEFI**, so they are not supported.
|
||||
|
||||
### Matching your CPU by name
|
||||
|
||||
If you know your chip's marketing name or codename, find it here. Everything from
|
||||
the **Supported** rows down works; the **Too old** row does not.
|
||||
|
||||
**Intel Core** (the "-lake"/"-bridge"/"-well" codenames):
|
||||
|
||||
| Status | Generation | Codename(s) | Year |
|
||||
|---|---|---|---|
|
||||
| Too old | 2nd gen | Sandy Bridge | 2011 |
|
||||
| Supported (baseline) | 3rd gen | Ivy Bridge | 2012 |
|
||||
| Supported | 4th–5th gen | Haswell, Broadwell | 2013–2014 |
|
||||
| **Recommended** | 6th–9th gen | **Skylake**, Kaby Lake, Coffee Lake | 2015–2018 |
|
||||
| Recommended | 10th–11th gen | Comet Lake, Ice Lake, Tiger Lake, Rocket Lake | 2019–2021 |
|
||||
| Recommended | 12th gen+ | Alder Lake, Raptor Lake | 2021–2023 |
|
||||
| Recommended | Core Ultra | Meteor Lake, Arrow Lake, Lunar Lake | 2023+ |
|
||||
|
||||
**AMD:**
|
||||
|
||||
| Status | Family | Codename(s) | Year |
|
||||
|---|---|---|---|
|
||||
| Supported (baseline) | A-series APU (A75/A85 FCH) | Llano, Trinity, Richland, Kaveri | 2011–2014 |
|
||||
| Supported | FX (AM3+) | Bulldozer, Piledriver | 2011–2012 |
|
||||
| **Recommended** | **Ryzen** 1000–5000 (AM4) | Summit/Pinnacle Ridge, Matisse, Vermeer (Zen–Zen 3) | 2017–2020 |
|
||||
| Recommended | Ryzen 7000+ (AM5) | Raphael, Granite Ridge (Zen 4 / Zen 5) | 2022+ |
|
||||
| Recommended | Threadripper / EPYC | Zen and later | 2017+ |
|
||||
|
||||
(These map generations to the era their platforms shipped built-in xHCI + UEFI;
|
||||
they are guidance, not a tested-hardware list.)
|
||||
|
||||
**Two caveats that matter regardless of CPU:**
|
||||
|
||||
- **Firmware must be UEFI.** Many 2011-era machines could do either UEFI or
|
||||
legacy BIOS — danos needs it set to UEFI. There is no BIOS boot path.
|
||||
- **Input is PS/2 only, for now.** danos does not yet support USB
|
||||
keyboards/mice. This is fine on most **laptops** (their built-in keyboards are
|
||||
wired to a PS/2-style i8042 controller) but means a **desktop with only USB
|
||||
ports** currently has no usable keyboard. USB HID input is planned.
|
||||
|
||||
Virtual machines are the easiest way to meet every requirement: QEMU (with OVMF/
|
||||
UEFI, a `qemu-xhci` controller, and the default Q35 machine type), or any
|
||||
hypervisor configured for UEFI firmware and an xHCI USB controller.
|
||||
|
||||
## CPU / architecture
|
||||
|
||||
| Requirement | Detail | Source |
|
||||
|---|---|---|
|
||||
| **x86-64, 64-bit only** | Kernel and loader are built exclusively for `x86_64`; the loader rejects any non-x86-64 kernel ELF (`error.WrongArchitecture`). | `build.zig:285`, `boot/efi.zig:418` |
|
||||
| **Long mode + PAE + NX** | AP trampoline sets `CR4.PAE`, `EFER.LME`, `EFER.NXE`; NX is used in kernel page-table entries. | `system/kernel/architecture/x86_64/trampoline.s:62` |
|
||||
| **SSE / SSE2** | Baseline: the compiler emits SSE for ordinary struct copies. Trampoline enables `CR4.OSFXSR` + `OSXMMEXCPT` and clears `CR0.EM`. | `build.zig:282`, `trampoline.s:62` |
|
||||
| **`syscall` / `sysret`** | Primary user↔kernel entry path. `EFER.SCE` enabled; `STAR`/`LSTAR`/`SFMASK` programmed per core. (`int 0x80` exists as a parallel gate.) | `architecture/x86_64/per-cpu.zig:59`, `isr.s:169` |
|
||||
| **Local APIC (xAPIC)** | LAPIC accessed via MMIO at `0xFEE00000`. LAPIC ID read as a `u8` — classic xAPIC. **x2APIC is not supported** (no MSR path). | `apic.zig:62`, `apic.zig:414` |
|
||||
| **CPUID + RDTSC** | CPUID leaf `0x15` for TSC frequency; RDTSC is the monotonic clock. | `apic.zig:279`, `apic.zig:84` |
|
||||
| **SMP (optional)** | Multi-core supported via INIT–SIPI–SIPI; ceiling `maximum_cpus = 128`. Single core is fine. Cores beyond the ceiling are parked. | `system/parameters.zig:16`, `apic.zig:144` |
|
||||
|
||||
## Firmware / boot
|
||||
|
||||
- **UEFI only.** A custom UEFI application loader is installed to
|
||||
`\EFI\BOOT\BOOTX64.efi`. There is **no BIOS, multiboot, or limine** path. The
|
||||
loader tolerates UEFI Class-3 machines with no legacy PIC/PIT.
|
||||
(`build.zig:464`, `boot/efi.zig`)
|
||||
- **ACPI is the hardware-discovery mechanism.** The RSDP is taken from the UEFI
|
||||
configuration table (ACPI 2.0 GUID preferred, 1.0 fallback). Without a valid
|
||||
RSDP there is **no device discovery** — no SMP, no IOAPIC routing, no PCI/USB.
|
||||
(`efi.zig:578`, `boot-handoff.zig:144`)
|
||||
- **Required ACPI tables:** MADT (interrupt topology), MCFG (PCIe ECAM base),
|
||||
FADT (power / PM timer). Optionally consumed: HPET, DMAR, SPCR.
|
||||
(`system/devices/acpi.zig:3`)
|
||||
- The loader reads `/system/kernel`, `/system/services/init`, and
|
||||
`/boot/initial-ramdisk.img` off the FAT boot volume. The kernel can boot
|
||||
"kernel-only" without init or the ramdisk. (`efi.zig:14`, `efi.zig:66`)
|
||||
|
||||
## Interrupt controller
|
||||
|
||||
- **Local APIC + I/O APIC required.** I/O APIC base, GSI base, and MADT
|
||||
interrupt-source overrides come from ACPI. (`cpu.zig:365`, `apic.zig:119`)
|
||||
- **MSI supported** — edge-triggered, keyed by vector, no I/O APIC mask cycle.
|
||||
Vector window 33–46, timer on 32, spurious on 47. (`system/kernel/irq.zig:70`,
|
||||
`cpu.zig:397`)
|
||||
- The legacy 8259 PIC is remapped and masked **only if present** (MADT
|
||||
`PCAT_COMPAT`); it is not required. (`apic.zig:103`)
|
||||
|
||||
## PCI / PCIe
|
||||
|
||||
- **PCIe with ECAM (MMConfig) required.** The PCI bus driver maps the host
|
||||
bridge's ECAM window (1 MiB config space per bus) and computes config
|
||||
addresses directly. **There is no legacy CF8/CFC port-IO config path** — the
|
||||
driver bails if the bridge exposes no ECAM window. The ECAM base comes from
|
||||
the ACPI MCFG table. (`system/drivers/pci-bus/pci-bus.zig:41`, `acpi.zig:6`)
|
||||
|
||||
## USB
|
||||
|
||||
- **xHCI only.** The sole USB driver is `usb-xhci-bus`, and the device manager
|
||||
binds it strictly to PCI prog-IF `0x30` (xHCI). UHCI / OHCI / EHCI exist only
|
||||
as report strings with no driver behind them — **USB 1.x/2.0-only controllers
|
||||
are not supported.** (`system/drivers/usb-xhci-bus/`,
|
||||
`system/services/device-manager/device-manager.zig:34`)
|
||||
- USB input (keyboard/mouse over HID) is future work; the current input stack is
|
||||
PS/2. See [Buses & devices](#buses--devices).
|
||||
|
||||
## Timers
|
||||
|
||||
Calibration prefers, in order: (1) CPUID leaf `0x15` TSC frequency, (2) HPET,
|
||||
(3) ACPI PM timer (3.579545 MHz, from FADT), (4) legacy PIT. Any one suffices —
|
||||
HPET/PM-timer/PIT are optional fallbacks when CPUID `0x15` is absent.
|
||||
(`apic.zig:180`)
|
||||
|
||||
- **TSC** — monotonic high-resolution clock.
|
||||
- **LAPIC timer** — scheduler heartbeat, periodic at `timer_hz = 1000 Hz`.
|
||||
(`parameters.zig:39`)
|
||||
|
||||
## Memory
|
||||
|
||||
**Target: 128 MiB RAM.** The system uses 4 KiB pages and a bitmap physical-frame
|
||||
allocator built from the firmware memory map. There is no hardcoded minimum-RAM
|
||||
constant — the allocator only panics if there is no usable region, or none large
|
||||
enough to hold its own bitmap. (`system/kernel/pmm.zig:13`, `pmm.zig:77`)
|
||||
|
||||
Where the budget goes:
|
||||
|
||||
| Consumer | Size | Source |
|
||||
|---|---|---|
|
||||
| Kernel heap (cap, grown one page at a time) | up to **64 MiB** | `system/kernel/heap.zig:26` |
|
||||
| Kernel stack, per CPU | 16 KiB | `parameters.zig:26` |
|
||||
| IST stack, per CPU | 16 KiB | `parameters.zig:36` |
|
||||
| User stack, per task | 8 pages / 32 KiB | `parameters.zig:32` |
|
||||
| Max concurrent tasks | 32 | `parameters.zig:23` |
|
||||
| Boot page-table pool | 64 frames / 256 KiB | `efi.zig:299` |
|
||||
|
||||
The 64 MiB heap cap plus kernel image, per-CPU stacks, task stacks, the frame
|
||||
bitmap, and DMA-contiguous allocations fit comfortably within 128 MiB on a
|
||||
single- or low-core-count machine. Very high core counts (toward the 128-CPU
|
||||
ceiling) add per-CPU stack overhead and push toward more RAM.
|
||||
|
||||
**Note on the 4 GiB physmap:** the loader identity-maps and physmaps the low
|
||||
4 GiB of address space with 2 MiB leaves. This is *virtual address* reach, not a
|
||||
RAM requirement — RAM above 4 GiB simply needs an extra mapping window and is not
|
||||
needed to boot. (`efi.zig:305`)
|
||||
|
||||
Virtual-memory layout (`boot-handoff.zig:47`):
|
||||
|
||||
| Region | Base |
|
||||
|---|---|
|
||||
| User space | `0x0000_7000_0000_0000` |
|
||||
| Kernel heap | `0xFFFF_8000_0000_0000` |
|
||||
| Physmap | `0xFFFF_8800_0000_0000` |
|
||||
| Kernel image | `0xFFFF_FFFF_8000_0000` |
|
||||
|
||||
## Buses & devices
|
||||
|
||||
Buses with real drivers today:
|
||||
|
||||
- **PCIe** via ECAM (`pci-bus`)
|
||||
- **xHCI USB** (`usb-xhci-bus`)
|
||||
- **PS/2** keyboard + mouse (`ps2-bus`) — the current input stack
|
||||
- **Serial UART** (16550/16450), configured from the ACPI SPCR table
|
||||
|
||||
**No storage driver exists yet.** AHCI / NVMe / IDE are named for reporting only;
|
||||
there is no block-device driver. Persistent storage is future work.
|
||||
|
||||
## IOMMU
|
||||
|
||||
**Detection only; enforcement deferred.** The ACPI DMAR table is parsed for the
|
||||
first VT-d DRHD unit and its capabilities are exposed via `PlatformInfo`
|
||||
(`iommu_present`, `iommu_base`, `iommu_version`). No DMA-remapping tables are
|
||||
programmed and no translation is enforced. An IOMMU is therefore **not required**
|
||||
and does not currently constrain devices. (`system/devices/acpi.zig:96`)
|
||||
|
||||
## What is explicitly NOT supported
|
||||
|
||||
- Legacy BIOS / multiboot / limine boot
|
||||
- 32-bit x86
|
||||
- x2APIC
|
||||
- Legacy port-IO (CF8/CFC) PCI configuration
|
||||
- Non-xHCI USB (UHCI / OHCI / EHCI)
|
||||
- Machines without ACPI (no device discovery)
|
||||
- Persistent storage (no AHCI / NVMe / IDE driver yet)
|
||||
- USB HID input (PS/2 only for now)
|
||||
+117
@@ -0,0 +1,117 @@
|
||||
# Timers and time
|
||||
|
||||
Two different needs hide under the word "timer", and danos keeps them apart:
|
||||
|
||||
- **Reading the clock** — *what time is it?* A read of a free-running counter.
|
||||
- **Waiting** — *wake me in N milliseconds*, or *notify me when a deadline passes.*
|
||||
|
||||
Both are answered by the **kernel**, because the kernel already owns a timer: it has
|
||||
to, to preempt tasks. The LAPIC heartbeat and the calibrated TSC that back all of this
|
||||
are built in [device-interrupts.md](device-interrupts.md); the scheduler's blocking and
|
||||
wait queues are in [scheduling.md](scheduling.md). This page is about the surface a
|
||||
ring-3 program actually uses, and one deliberate absence: **there is no user-space time
|
||||
service.**
|
||||
|
||||
## Why time is a syscall, not a service
|
||||
|
||||
The tempting microkernel move is to put a timer *driver* in user space and have
|
||||
applications ask it for the time over IPC. For a **monotonic clock that is wrong** —
|
||||
reading `now()` should never cost an IPC round trip. The kernel is already holding the
|
||||
answer: it computes the current time every time it schedules, from the TSC, in a couple
|
||||
of instructions. Surfacing that as a system call is pure mechanism; routing it through a
|
||||
message to another process would be slower *and* redundant, and a device like the HPET
|
||||
(uncacheable MMIO reads) is a particularly bad thing to read on every `now()`.
|
||||
|
||||
This is the same conclusion every serious system reaches: Linux and Zircon read the
|
||||
counter in the vDSO, L4 exposes a clock field in a shared kernel page, seL4 reads the
|
||||
cycle counter directly. None of them make a clock read an IPC. danos makes it a syscall.
|
||||
|
||||
That "from the TSC" hides a portability question, because the TSC is only a valid clock
|
||||
when the CPU guarantees it is *invariant* and when every core's TSC is *synchronized*.
|
||||
danos checks both — the invariant-TSC CPUID bit (`0x80000007` EDX[8], set on Intel and
|
||||
AMD), and a cross-core "warp" check as the cores come up — and falls back to the HPET
|
||||
counter when either fails. So `now()` stays accurate on a real Intel box, a real AMD box,
|
||||
and inside a VM alike; only the source behind it differs. The mechanism is in
|
||||
[device-interrupts.md](device-interrupts.md).
|
||||
|
||||
So the timer hardware lives in the kernel, and there is **no `hpet` driver and no time
|
||||
server** to consume. (An earlier HPET driver existed only to *demonstrate* the driver
|
||||
model; that role now lives in [drivers.md](drivers.md), as documentation.) The one place
|
||||
a user-space time service *is* justified — **wall-clock / calendar time** — is discussed
|
||||
at the end; it is deliberately not built yet.
|
||||
|
||||
## The three system calls
|
||||
|
||||
Time and waiting are three entries in the small syscall table ([syscall.md](syscall.md)):
|
||||
|
||||
- **`clock` (#23)** → monotonic nanoseconds since boot. It only moves forward. Not
|
||||
wall-clock: no date, no timezone. Backed by `architecture.nanos()` (TSC, scaled with a
|
||||
128-bit intermediate so a long uptime can't overflow) — a few nanoseconds of
|
||||
resolution, and just an `rdtsc` plus a multiply.
|
||||
- **`sleep` (#3)** → block the caller for N milliseconds. The scheduler records a wake
|
||||
deadline and the tick sweep wakes it (`scheduler.sleep`).
|
||||
- **`timer_bind` (#31)** → arm a one-shot timer that, after N milliseconds, posts a
|
||||
**timer notification** to an IPC endpoint. Unlike `sleep` it does **not** block: a
|
||||
service can keep answering messages on the same endpoint while a deadline is pending.
|
||||
This is the timed wait that stop-sequence escalation, hello deadlines, and restart
|
||||
backoff are built from ([process-lifecycle.md](process-lifecycle.md),
|
||||
[device-manager.md](device-manager.md)).
|
||||
|
||||
The kernel's own scheduling timer (the LAPIC, vector 32) is never exposed to user space;
|
||||
programs read the TSC through `clock` and get timed wakeups through `sleep`/`timer_bind`,
|
||||
both riding the scheduler tick.
|
||||
|
||||
## `runtime.time` — the generic interface
|
||||
|
||||
Applications don't call the syscalls directly; they use `runtime.time`
|
||||
(`library/runtime/time.zig`), a thin `Instant`/`Duration` layer over them — an ergonomic
|
||||
front door, not new mechanism.
|
||||
|
||||
```zig
|
||||
const time = @import("runtime").time;
|
||||
|
||||
const start = time.now(); // Instant — monotonic
|
||||
doWork();
|
||||
const took = start.elapsed(); // Duration
|
||||
time.sleep(time.Duration.fromMillis(5)); // block ~5 ms
|
||||
|
||||
// A deadline delivered as a notification, so a service keeps serving meanwhile:
|
||||
_ = time.after(endpoint, time.Duration.fromMillis(200));
|
||||
```
|
||||
|
||||
- `Duration` is nanoseconds under the hood, with `fromNanos/fromMicros/fromMillis/
|
||||
fromSeconds` and `asNanos/asMillis`. `ceilMillis` rounds *up* to the kernel's
|
||||
millisecond granularity, so a sub-millisecond `sleep` never rounds down to zero and
|
||||
returns early. All arithmetic saturates rather than wraps.
|
||||
- `Instant` is a point on the monotonic clock: `since`, `elapsed`, `plus`, `reached` —
|
||||
built for deadline loops (`while (!deadline.reached()) …`).
|
||||
- `now()` / `monotonicNanos()` wrap `clock`. `available()` reports whether the clock is
|
||||
calibrated at all (the kernel returns 0 until the TSC frequency is known, so a caller
|
||||
that needs real time can treat 0 as "unavailable" rather than assume it advances).
|
||||
- `sleep(d)` wraps `sleep`; `spin(d)` busy-polls `now()` for the sub-millisecond delays
|
||||
the millisecond tick can't express; `after(endpoint, d)` wraps `timer_bind`.
|
||||
|
||||
The raw wrappers (`system.clock`, `system.sleep`, `system.timerOnce`) stay in
|
||||
`library/runtime/system.zig`; `runtime.time` is the layer meant for everyday use.
|
||||
|
||||
## Wall-clock time (not built)
|
||||
|
||||
Everything above is **monotonic**: elapsed time since boot, perfect for timeouts and
|
||||
measurement, useless for "what is the date?" Calendar time — a real-time clock, time
|
||||
zones, leap seconds — is genuinely a **user-space** concern, and it *is* the case a time
|
||||
service is for. It would be backed by an **RTC** driver (the CMOS real-time clock), not
|
||||
the HPET, and exposed as a `CLOCK_REALTIME`-style service alongside the monotonic
|
||||
syscall. It is deferred until something needs it; the monotonic clock the kernel already
|
||||
owns covers every current use.
|
||||
|
||||
## Verifying it
|
||||
|
||||
`runtime.time`'s `Instant`/`Duration` arithmetic has unit tests that run on the host:
|
||||
|
||||
```
|
||||
$ zig build test # includes library/runtime/time.zig
|
||||
```
|
||||
|
||||
End to end, the proof the clock is real is that it *advances*: read `now()`, `sleep` a
|
||||
`Duration`, read `now()` again, and the second reading is later — the kernel's timer
|
||||
driving a ring-3 program with no service in between.
|
||||
@@ -12,11 +12,20 @@
|
||||
//! (arguments arrive via `init`).
|
||||
|
||||
pub const system = @import("system.zig");
|
||||
/// Monotonic time, delays, and deadlines over the kernel clock/sleep/timer syscalls
|
||||
/// — an `Instant`/`Duration` front door, no time service (docs/timers.md).
|
||||
pub const time = @import("time.zig");
|
||||
pub const heap = @import("heap.zig");
|
||||
pub const ipc = @import("ipc.zig");
|
||||
pub const start = @import("start.zig");
|
||||
/// The VFS wire protocol (shared with the VFS server).
|
||||
pub const vfs_protocol = @import("vfs-protocol");
|
||||
|
||||
/// The device-manager protocol: hello + tree reports (docs/device-manager.md).
|
||||
pub const device_manager_protocol = @import("device-manager-protocol");
|
||||
|
||||
/// The power protocol: events (button, lid, battery) + shutdown (docs/power.md).
|
||||
pub const power_protocol = @import("power-protocol");
|
||||
/// Keyboard-event listening (subscribe/next) and broadcasting (publish), over the input
|
||||
/// service. See library/runtime/input.zig and system/services/input/.
|
||||
pub const input = @import("input.zig");
|
||||
|
||||
@@ -22,8 +22,10 @@ pub const Callbacks = struct {
|
||||
/// Return false to abort startup (the process exits).
|
||||
init: ?*const fn (endpoint: ipc.Handle) bool = null,
|
||||
/// One protocol request from `sender` (a task id): write the reply into
|
||||
/// `reply`, return its length. The zero-length ping never reaches this.
|
||||
on_message: *const fn (message: []const u8, reply: []u8, sender: u32) usize,
|
||||
/// `reply`, return its length. `capability` is the handle the request
|
||||
/// carried, if any (M13 cap passing — how a subscriber hands over its
|
||||
/// endpoint). The zero-length ping never reaches this.
|
||||
on_message: *const fn (message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize,
|
||||
/// A notification that is not a signal — a subscribed exit event, a bound
|
||||
/// IRQ, a timer landing. The raw badge; decode with the ipc helpers.
|
||||
on_notification: ?*const fn (badge: u64) void = null,
|
||||
@@ -76,6 +78,6 @@ pub fn run(comptime maximum_message: usize, callbacks: Callbacks) void {
|
||||
reply_len = 0; // the universal ping: a zero-length reply, from the harness
|
||||
continue;
|
||||
}
|
||||
reply_len = callbacks.on_message(receive[0..got.len], &reply_buffer, got.senderTaskId());
|
||||
reply_len = callbacks.on_message(receive[0..got.len], &reply_buffer, got.senderTaskId(), got.cap);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,169 @@
|
||||
//! The danos time interface — monotonic time, delays, and deadlines for user space.
|
||||
//!
|
||||
//! There is no time *service*: the kernel already owns the scheduling timer and
|
||||
//! surfaces it directly, so reading the clock is one system call (an `rdtsc` and a
|
||||
//! scale), never an IPC round trip (docs/timers.md explains why). This module is a
|
||||
//! thin, generic layer over the `clock`/`sleep`/`timer_bind` wrappers in `system.zig`
|
||||
//! — an ergonomic `Instant`/`Duration` front door, not new mechanism.
|
||||
//!
|
||||
//! It is **monotonic** time only: nanoseconds since boot, moving forward, no date or
|
||||
//! timezone. Wall-clock/calendar time is a separate user-space service (an RTC-backed
|
||||
//! CLOCK_REALTIME) layered on top later.
|
||||
|
||||
const std = @import("std");
|
||||
const system = @import("system.zig");
|
||||
|
||||
const nanos_per_micro: u64 = 1_000;
|
||||
const nanos_per_milli: u64 = 1_000_000;
|
||||
const nanos_per_second: u64 = 1_000_000_000;
|
||||
|
||||
/// A span of time, held as nanoseconds. Constructors name their unit; accessors
|
||||
/// truncate toward zero. `ceilMillis` rounds *up*, since `sleep`/`after` land on the
|
||||
/// kernel's millisecond granularity and rounding down could return early.
|
||||
pub const Duration = struct {
|
||||
ns: u64,
|
||||
|
||||
pub fn fromNanos(n: u64) Duration {
|
||||
return .{ .ns = n };
|
||||
}
|
||||
pub fn fromMicros(n: u64) Duration {
|
||||
return .{ .ns = n *| nanos_per_micro };
|
||||
}
|
||||
pub fn fromMillis(n: u64) Duration {
|
||||
return .{ .ns = n *| nanos_per_milli };
|
||||
}
|
||||
pub fn fromSeconds(n: u64) Duration {
|
||||
return .{ .ns = n *| nanos_per_second };
|
||||
}
|
||||
|
||||
pub fn asNanos(d: Duration) u64 {
|
||||
return d.ns;
|
||||
}
|
||||
pub fn asMicros(d: Duration) u64 {
|
||||
return d.ns / nanos_per_micro;
|
||||
}
|
||||
pub fn asMillis(d: Duration) u64 {
|
||||
return d.ns / nanos_per_milli;
|
||||
}
|
||||
pub fn asSeconds(d: Duration) u64 {
|
||||
return d.ns / nanos_per_second;
|
||||
}
|
||||
|
||||
/// Whole milliseconds, rounded up — the argument `sleep`/`after` pass the kernel.
|
||||
/// A non-zero sub-millisecond duration becomes 1 ms rather than 0.
|
||||
pub fn ceilMillis(d: Duration) u64 {
|
||||
return (d.ns +| (nanos_per_milli - 1)) / nanos_per_milli;
|
||||
}
|
||||
|
||||
pub fn plus(a: Duration, b: Duration) Duration {
|
||||
return .{ .ns = a.ns +| b.ns };
|
||||
}
|
||||
};
|
||||
|
||||
/// A point on the monotonic clock — nanoseconds since boot. Compare and subtract
|
||||
/// instants to measure elapsed time; it never runs backward, so `since` is safe to
|
||||
/// saturate at zero rather than wrap.
|
||||
pub const Instant = struct {
|
||||
ns: u64,
|
||||
|
||||
/// The span from `earlier` to `self`, saturating at zero if `earlier` is later
|
||||
/// (which the monotonic clock should never produce, but callers may pass any pair).
|
||||
pub fn since(self: Instant, earlier: Instant) Duration {
|
||||
return .{ .ns = self.ns -| earlier.ns };
|
||||
}
|
||||
|
||||
/// How long since this instant, sampled now.
|
||||
pub fn elapsed(self: Instant) Duration {
|
||||
return now().since(self);
|
||||
}
|
||||
|
||||
/// This instant advanced by `d` (a deadline, `d` from here).
|
||||
pub fn plus(self: Instant, d: Duration) Instant {
|
||||
return .{ .ns = self.ns +| d.ns };
|
||||
}
|
||||
|
||||
/// Whether the monotonic clock has reached this instant (used as a deadline).
|
||||
pub fn reached(deadline: Instant) bool {
|
||||
return now().ns >= deadline.ns;
|
||||
}
|
||||
};
|
||||
|
||||
/// The current monotonic time.
|
||||
pub fn now() Instant {
|
||||
return .{ .ns = system.clock() };
|
||||
}
|
||||
|
||||
/// Monotonic nanoseconds since boot — the raw `clock()` reading, for callers that
|
||||
/// want a plain integer instead of an `Instant`.
|
||||
pub fn monotonicNanos() u64 {
|
||||
return system.clock();
|
||||
}
|
||||
|
||||
/// Whether the monotonic clock is usable. The kernel returns 0 until the TSC is
|
||||
/// calibrated (`tsc_hz == 0`); a caller that needs real time can treat that as
|
||||
/// "unavailable" instead of assuming the clock advances.
|
||||
pub fn available() bool {
|
||||
return system.clock() != 0;
|
||||
}
|
||||
|
||||
/// Block the caller for at least `d`, rounded up to the kernel's millisecond
|
||||
/// granularity. For sub-millisecond precision the scheduler cannot express, use
|
||||
/// `spin`.
|
||||
pub fn sleep(d: Duration) void {
|
||||
system.sleep(d.ceilMillis());
|
||||
}
|
||||
|
||||
/// Block the caller for `ms` milliseconds — the coarse, allocation-free form.
|
||||
pub fn sleepMillis(ms: u64) void {
|
||||
system.sleep(ms);
|
||||
}
|
||||
|
||||
/// Busy-wait until `d` has elapsed, polling the monotonic clock. This burns the CPU
|
||||
/// on purpose, to hit sub-millisecond delays the scheduler's millisecond tick cannot.
|
||||
/// Prefer `sleep` for anything at or above a millisecond.
|
||||
pub fn spin(d: Duration) void {
|
||||
const deadline = now().plus(d);
|
||||
while (!deadline.reached()) {}
|
||||
}
|
||||
|
||||
/// Arm a one-shot timer against `endpoint` (a handle from `ipc.createIpcEndpoint`):
|
||||
/// after `d` the kernel posts a timer notification (`ipc.Received.isTimer`) there.
|
||||
/// Unlike `sleep`, this does not block — a service can keep serving IPC on the same
|
||||
/// endpoint while the deadline is pending. Rounds `d` up to milliseconds; returns
|
||||
/// false if the timer could not be armed. See `system.timerOnce`.
|
||||
pub fn after(endpoint: usize, d: Duration) bool {
|
||||
return system.timerOnce(endpoint, d.ceilMillis());
|
||||
}
|
||||
|
||||
test "Duration unit conversions round toward zero" {
|
||||
try std.testing.expectEqual(@as(u64, 1_000_000_000), Duration.fromSeconds(1).asNanos());
|
||||
try std.testing.expectEqual(@as(u64, 1_500), Duration.fromNanos(1_500).asNanos());
|
||||
try std.testing.expectEqual(@as(u64, 2), Duration.fromMillis(2).asMillis());
|
||||
try std.testing.expectEqual(@as(u64, 1), Duration.fromNanos(1_999_999).asMillis());
|
||||
try std.testing.expectEqual(@as(u64, 250), Duration.fromMicros(250).asMicros());
|
||||
}
|
||||
|
||||
test "ceilMillis rounds up, and never turns a nonzero span into zero" {
|
||||
try std.testing.expectEqual(@as(u64, 0), Duration.fromNanos(0).ceilMillis());
|
||||
try std.testing.expectEqual(@as(u64, 1), Duration.fromNanos(1).ceilMillis());
|
||||
try std.testing.expectEqual(@as(u64, 1), Duration.fromMillis(1).ceilMillis());
|
||||
try std.testing.expectEqual(@as(u64, 2), Duration.fromNanos(nanos_per_milli + 1).ceilMillis());
|
||||
try std.testing.expectEqual(@as(u64, 5), Duration.fromMillis(5).ceilMillis());
|
||||
}
|
||||
|
||||
test "Instant arithmetic: since saturates, plus/reached form deadlines" {
|
||||
const t0 = Instant{ .ns = 1_000 };
|
||||
const t1 = Instant{ .ns = 4_000 };
|
||||
try std.testing.expectEqual(@as(u64, 3_000), t1.since(t0).asNanos());
|
||||
// earlier-than-self can't happen on a monotonic clock; saturate rather than wrap.
|
||||
try std.testing.expectEqual(@as(u64, 0), t0.since(t1).asNanos());
|
||||
const deadline = t0.plus(Duration.fromNanos(2_500));
|
||||
try std.testing.expectEqual(@as(u64, 3_500), deadline.ns);
|
||||
}
|
||||
|
||||
test "saturating arithmetic does not overflow at the u64 ceiling" {
|
||||
const big = Duration.fromSeconds(std.math.maxInt(u64));
|
||||
try std.testing.expectEqual(@as(u64, std.math.maxInt(u64)), big.asNanos());
|
||||
const late = Instant{ .ns = std.math.maxInt(u64) };
|
||||
try std.testing.expectEqual(@as(u64, std.math.maxInt(u64)), late.plus(Duration.fromSeconds(10)).ns);
|
||||
}
|
||||
@@ -176,6 +176,8 @@ pub const ServiceId = enum(u32) {
|
||||
vfs = 1,
|
||||
input = 2,
|
||||
ps2_bus = 3, // the 8042 owner; child device drivers attach here for raw bytes
|
||||
device_manager = 4, // the tree, the matcher, the supervisor (docs/device-manager.md)
|
||||
power = 5, // system power: events (button, lid, battery) + shutdown (docs/power.md; domain-named per docs/discovery.md — the acpi service registers it on x86, a PSCI service will on ARM)
|
||||
_,
|
||||
};
|
||||
|
||||
|
||||
+138
-504
@@ -17,7 +17,6 @@
|
||||
const std = @import("std");
|
||||
const boot_handoff = @import("boot-handoff");
|
||||
const abi = @import("abi");
|
||||
const acpi_ids = @import("acpi-ids");
|
||||
const parameters = @import("parameters");
|
||||
const device_model = @import("device-model.zig");
|
||||
const aml = @import("aml/aml.zig");
|
||||
@@ -41,6 +40,9 @@ pub const RegisterAccess = struct {
|
||||
/// Everything the power subsystem needs, extracted from the FADT and the AML
|
||||
/// sleep packages during discovery. Populated by `discover`, read by `power`.
|
||||
pub const PowerInformation = struct {
|
||||
/// The System Control Interrupt's GSI (FADT SCI_INT) — the line ACPI events
|
||||
/// (power button, GPEs) arrive on. Published to the acpi service for M21.
|
||||
sci_interrupt: u16 = 0,
|
||||
/// The SMM command port and the value that switches the platform into ACPI mode.
|
||||
smi_cmd: u16 = 0,
|
||||
acpi_enable: u8 = 0,
|
||||
@@ -155,6 +157,13 @@ pub var namespace: ?aml.Namespace = null;
|
||||
/// Physical address of the DSDT the FADT points at, or 0.
|
||||
pub var dsdt_physical: u64 = 0;
|
||||
|
||||
/// The FADT itself (physical + length), published on the acpi-tables node so
|
||||
/// the ring-3 acpi service can read the PM1 event and GPE blocks it needs for
|
||||
/// the event side (docs/acpi.md — ACPI events). Distinguished from the AML
|
||||
/// blob resources by its intact "FACP" header — the blobs are header-stripped.
|
||||
var fadt_physical: u64 = 0;
|
||||
var fadt_length: u64 = 0;
|
||||
|
||||
// AML blocks (DSDT + any SSDTs) collected during the table walk, as physical
|
||||
// address + length of each table's post-header bytecode. Scanned after the walk
|
||||
// for the sleep-state (`_Sx`) packages.
|
||||
@@ -360,35 +369,19 @@ const Hpet = extern struct {
|
||||
page_protection: u8,
|
||||
};
|
||||
|
||||
// --- PCI configuration-space header (first 64 bytes, common fields) ---------
|
||||
|
||||
const PciHeader = extern struct {
|
||||
vendor_id: u16 align(1),
|
||||
device_id: u16 align(1),
|
||||
command: u16 align(1),
|
||||
status: u16 align(1),
|
||||
revision_id: u8,
|
||||
prog_if: u8,
|
||||
subclass: u8,
|
||||
class_code: u8,
|
||||
cache_line_size: u8,
|
||||
latency_timer: u8,
|
||||
/// bit 7 set => multi-function device.
|
||||
header_type: u8,
|
||||
bist: u8,
|
||||
// 0x10 onward (BARs, etc.) depends on header_type; read separately.
|
||||
};
|
||||
|
||||
// --- Entry point ------------------------------------------------------------
|
||||
|
||||
/// Discover hardware from the ACPI tables rooted at `rsdp_physical` and populate
|
||||
/// `device_tree`. `hal` provides MMIO mapping (for PCIe ECAM) and port I/O. Also parses the
|
||||
/// FADT and the AML sleep-state (`_Sx`) packages into `power_information` for the power service.
|
||||
pub fn discover(rsdp_physical: u64, device_tree: *DeviceTree, hal: Hal) !void {
|
||||
pub fn discover(rsdp_physical: u64, memory_regions: []const boot_handoff.MemoryRegion, device_tree: *DeviceTree, hal: Hal) !void {
|
||||
if (rsdp_physical == 0) return error.NoRsdp;
|
||||
boot_memory_regions = memory_regions;
|
||||
|
||||
// Start clean so a re-run doesn't accumulate stale state.
|
||||
power_information = .{};
|
||||
fadt_physical = 0;
|
||||
fadt_length = 0;
|
||||
platform_information = .{};
|
||||
aml_stats = .{};
|
||||
namespace = null;
|
||||
@@ -420,12 +413,58 @@ pub fn discover(rsdp_physical: u64, device_tree: *DeviceTree, hal: Hal) !void {
|
||||
aml_stats = .{ .nodes = namespace.?.nodeCount(), .consumed = pr.consumed, .total = pr.total };
|
||||
power_information.s5 = aml.sleepState(&namespace.?, 5);
|
||||
power_information.s3 = aml.sleepState(&namespace.?, 3);
|
||||
// Fold the namespace's Device objects into the generic tree.
|
||||
wireAcpiDevices(device_tree, &namespace.?, hal) catch {};
|
||||
// The namespace's Device objects are no longer folded into the kernel
|
||||
// tree (M20.3): the ring-3 acpi service claims the acpi-tables node
|
||||
// (published below), re-parses the same blobs, and registers + reports
|
||||
// the _HID devices itself. The kernel keeps the namespace only for the
|
||||
// \_S5 sleep type above.
|
||||
} else |_| {
|
||||
// AML parse failed (e.g. out of memory); power stays best-effort with
|
||||
// whatever the FADT alone provided.
|
||||
}
|
||||
|
||||
// Publish the acpi-tables node (docs/discovery.md): the AML blobs as
|
||||
// memory resources for the acpi service to map and parse in ring 3, a broad
|
||||
// io_port grant for the OperationRegion access its interpreter needs, and
|
||||
// the SCI for the events track (M21). Exactly one node, one trusted
|
||||
// claimant. Kept even when the kernel-side device building (above) retires
|
||||
// in M20.3 — the kernel still owns the *static* tables and \_S5.
|
||||
publishAcpiTablesNode(device_tree) catch {};
|
||||
}
|
||||
|
||||
/// Build the acpi-tables node (see the call site in discover). Best-effort: a
|
||||
/// failure here leaves the kernel-seeded tree working, only the ring-3 service
|
||||
/// finds nothing to claim.
|
||||
fn publishAcpiTablesNode(device_tree: *DeviceTree) !void {
|
||||
const node = try device_tree.addChild(device_tree.root, .acpi_tables, "acpi-tables");
|
||||
// One memory resource per AML block — page-aligned base down, length padded
|
||||
// up to cover the bytecode, so mmio_map hands the service a pointer into it.
|
||||
var i: usize = 0;
|
||||
while (i < aml_block_count and i < device_model.maximum_resources - 2) : (i += 1) {
|
||||
// mmio_map preserves the sub-page offset, so the service maps this and
|
||||
// gets a pointer straight to the bytecode.
|
||||
_ = node.addResource(.memory, aml_block_physical[i], aml_block_len[i]);
|
||||
}
|
||||
// The broad I/O grant: OperationRegions name whatever ports the firmware
|
||||
// chose (EC, PM1, GPE, SMBus); which ports cannot be known before the AML
|
||||
// that names them is parsed, so the grant is the whole space — the honest
|
||||
// trust boundary of docs/discovery.md (the acpi service's one trusted node).
|
||||
_ = node.addResource(.io_port, 0, 1 << 16);
|
||||
// A broad interrupt window: ACPI _CRS names legacy ISA IRQs (the PS/2 lines
|
||||
// 1 and 12, the RTC, …), and the service registers those devices under this
|
||||
// node, so it must own a superset. The range [0, 256) covers every GSI; the
|
||||
// SCI (recorded first, len 1) stays distinct so M21 can pick it out.
|
||||
if (power_information.sci_interrupt != 0) _ = node.addResource(.irq, power_information.sci_interrupt, 1);
|
||||
_ = node.addResource(.irq, 0, 256);
|
||||
// The FADT rides along (M21): the service reads the PM1 event / GPE blocks
|
||||
// from its own copy, telling it apart from the AML blobs by signature.
|
||||
if (fadt_physical != 0) _ = node.addResource(.memory, fadt_physical, fadt_length);
|
||||
}
|
||||
|
||||
/// The number of Device objects in the namespace built during discovery, or 0.
|
||||
pub fn amlDeviceCount() usize {
|
||||
if (namespace) |*ns| return aml.deviceCount(ns);
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// Walk the RSDT (Entry = u32) or XSDT (Entry = u64): validate it, then dispatch
|
||||
@@ -451,10 +490,12 @@ fn handleTable(device_tree: *DeviceTree, hal: Hal, sdt_physical: u64) !void {
|
||||
if (std.mem.eql(u8, &sig, &APIC)) {
|
||||
try parseMadt(device_tree, header);
|
||||
} else if (std.mem.eql(u8, &sig, &MCFG)) {
|
||||
try parseMcfg(device_tree, hal, header);
|
||||
try parseMcfg(device_tree, header);
|
||||
} else if (std.mem.eql(u8, &sig, &HPET)) {
|
||||
try parseHpet(device_tree, hal, header);
|
||||
} else if (std.mem.eql(u8, &sig, &FACP)) {
|
||||
fadt_physical = sdt_physical;
|
||||
fadt_length = header.length;
|
||||
parseFadt(header);
|
||||
} else if (std.mem.eql(u8, &sig, &SPCR)) {
|
||||
parseSpcr(header);
|
||||
@@ -536,7 +577,7 @@ fn parseMadt(device_tree: *DeviceTree, header: *const SystemDescriptorTableHeade
|
||||
}
|
||||
|
||||
/// MCFG -> a pci_host_bridge per ECAM segment, then a PCI enumeration underneath.
|
||||
fn parseMcfg(device_tree: *DeviceTree, hal: Hal, header: *const SystemDescriptorTableHeader) !void {
|
||||
fn parseMcfg(device_tree: *DeviceTree, header: *const SystemDescriptorTableHeader) !void {
|
||||
const total: usize = header.length;
|
||||
const base: [*]const u8 = @ptrCast(header);
|
||||
|
||||
@@ -551,109 +592,85 @@ fn parseMcfg(device_tree: *DeviceTree, hal: Hal, header: *const SystemDescriptor
|
||||
// ECAM window: 1 MiB of configuration space per bus.
|
||||
_ = bridge.addResource(.memory, alloc.base_address, bus_count << 20);
|
||||
_ = bridge.addResource(.bus_range, alloc.start_bus, bus_count);
|
||||
addBridgeApertures(bridge);
|
||||
// The bridge decodes the whole 16-bit I/O space toward its bus — the
|
||||
// window functions' I/O BARs must register-contain within (M19.2).
|
||||
_ = bridge.addResource(.io_port, 0, 1 << 16);
|
||||
|
||||
try enumeratePci(device_tree, bridge, hal, alloc.*);
|
||||
// The function walk itself retired to ring 3 (M19.3): the pci-bus
|
||||
// driver claims this bridge, repeats the scan through its ECAM grant,
|
||||
// and device_registers what it finds — the kernel seeds only the
|
||||
// bridge. The scan's equivalence was proven before the hand-off
|
||||
// (pci-scan), and the walk's history is in git if archaeology calls.
|
||||
}
|
||||
}
|
||||
|
||||
/// Brute-force scan the ECAM window's bus range for present PCI functions. No
|
||||
/// bridge recursion yet: on the ECAM path the host bridge decodes every bus in
|
||||
/// the window, so scanning the declared range finds everything QEMU exposes.
|
||||
fn enumeratePci(
|
||||
device_tree: *DeviceTree,
|
||||
bridge: *device_model.Device,
|
||||
hal: Hal,
|
||||
alloc: McfgAllocation,
|
||||
) !void {
|
||||
var bus: u16 = alloc.start_bus;
|
||||
while (bus <= alloc.end_bus) : (bus += 1) {
|
||||
var device: u8 = 0;
|
||||
while (device < 32) : (device += 1) {
|
||||
const h0: *align(1) const PciHeader = @ptrCast(pciConfigurationPtr(alloc, hal, @intCast(bus), device, 0));
|
||||
if (h0.vendor_id == 0xFFFF) continue; // no function 0 => slot empty
|
||||
/// The boot memory map, stored at discover() entry for the aperture derivation
|
||||
/// below (and, in M20, for the acpi-tables node's containment windows).
|
||||
var boot_memory_regions: []const boot_handoff.MemoryRegion = &.{};
|
||||
|
||||
const funcs: u8 = if (h0.header_type & 0x80 != 0) 8 else 1;
|
||||
var function: u8 = 0;
|
||||
while (function < funcs) : (function += 1) {
|
||||
const configuration = pciConfigurationPtr(alloc, hal, @intCast(bus), device, function);
|
||||
const h: *align(1) const PciHeader = @ptrCast(configuration);
|
||||
if (h.vendor_id == 0xFFFF) continue;
|
||||
|
||||
var nb: [24]u8 = undefined;
|
||||
const nm = std.fmt.bufPrint(&nb, "{s}:{x:0>2}:{x:0>2}.{d}", .{
|
||||
bridge.name(), bus, device, function,
|
||||
}) catch "pcidev";
|
||||
const node = try device_tree.addChild(bridge, .pci_device, nm);
|
||||
// Resource 0 is the function's own 4 KiB ECAM configuration space. A
|
||||
// claimed PCI driver mmio_maps this to reach its command register,
|
||||
// BARs, and — the point — its capability list (MSI/MSI-X, PCIe
|
||||
// extended caps), without any new syscall. Physical address per the
|
||||
// ECAM formula (same as pciConfigurationPtr).
|
||||
const config_physical = alloc.base_address +
|
||||
(@as(u64, @as(u8, @intCast(bus)) - alloc.start_bus) << 20) +
|
||||
(@as(u64, device) << 15) + (@as(u64, function) << 12);
|
||||
_ = node.addResource(.memory, config_physical, abi.page_size);
|
||||
node.ids.pci_vendor = h.vendor_id;
|
||||
node.ids.pci_device = h.device_id;
|
||||
node.ids.pci_class = (@as(u24, h.class_code) << 16) |
|
||||
(@as(u24, h.subclass) << 8) | h.prog_if;
|
||||
node.ids.pci_bdf = (@as(u16, @intCast(bus)) << 8) | (@as(u16, device) << 3) | function;
|
||||
|
||||
// BARs only exist in header type 0 (normal devices), not bridges.
|
||||
if (h.header_type & 0x7F == 0) addBars(node, configuration);
|
||||
/// The bridge's MMIO apertures, derived from the boot memory map's holes
|
||||
/// (docs/discovery.md — apertures from the memory map): registered PCI functions carry BAR
|
||||
/// resources, and `device_register` containment demands the bridge own windows
|
||||
/// that cover them. Everything the firmware described is "not hole"; the low
|
||||
/// aperture runs from the end of the described space below 4 GiB up to the
|
||||
/// I/O-APIC region, the high one from 4 GiB (or the end of RAM above it) to
|
||||
/// the 46-bit line. Coarse, mechanical, and AML-free — available at boot no
|
||||
/// matter what later moved to user space.
|
||||
fn addBridgeApertures(bridge: *device_model.Device) void {
|
||||
// Below 4 GiB the described regions are sparse (RAM low, firmware flash
|
||||
// and tables high), so the holes are the *gaps between* them — a single
|
||||
// "after the last region" rule dies on OVMF's flash at the very top.
|
||||
// Sort-merge the described ranges, then keep the three largest gaps
|
||||
// (resource slots are bounded at 8 per device; ECAM + bus range + 3 + the
|
||||
// high aperture fits). Above 4 GiB one aperture runs from the end of the
|
||||
// described space to the 46-bit line.
|
||||
const Range = struct { base: u64, end: u64 };
|
||||
var below: [64]Range = undefined;
|
||||
var below_count: usize = 0;
|
||||
var high_end: u64 = 1 << 32;
|
||||
for (boot_memory_regions) |region| {
|
||||
const end = region.base + region.pages * 4096;
|
||||
// Above 4 GiB only *usable RAM* blocks the aperture: OVMF describes
|
||||
// its own 64-bit PCI window as a reserved region and then programs
|
||||
// BARs inside it — honoring reserved there would exclude the very
|
||||
// space BARs live in. Below 4 GiB every described region blocks (the
|
||||
// kernel image, the tables, the ramdisk all live there). Bring-up
|
||||
// trust: only the bridge's claimant can register into the aperture.
|
||||
if (region.kind == .usable and end > high_end) high_end = end;
|
||||
if (region.base >= (1 << 32) or below_count == below.len) continue;
|
||||
below[below_count] = .{ .base = region.base, .end = @min(end, 1 << 32) };
|
||||
below_count += 1;
|
||||
}
|
||||
// Insertion sort by base (the map is small and this runs once at boot).
|
||||
for (1..below_count) |i| {
|
||||
const key = below[i];
|
||||
var j = i;
|
||||
while (j > 0 and below[j - 1].base > key.base) : (j -= 1) below[j] = below[j - 1];
|
||||
below[j] = key;
|
||||
}
|
||||
// Walk the sorted ranges, collecting inter-region gaps of at least 1 MiB.
|
||||
var gaps: [3]Range = .{Range{ .base = 0, .end = 0 }} ** 3;
|
||||
var cursor: u64 = 0;
|
||||
var index: usize = 0;
|
||||
while (index <= below_count) : (index += 1) {
|
||||
const gap_end = if (index == below_count) (1 << 32) else below[index].base;
|
||||
if (gap_end > cursor and gap_end - cursor >= (1 << 20)) {
|
||||
// Keep the three largest, replacing the smallest kept so far.
|
||||
var smallest: usize = 0;
|
||||
for (gaps, 0..) |gap, gi| {
|
||||
if (gap.end - gap.base < gaps[smallest].end - gaps[smallest].base) smallest = gi;
|
||||
}
|
||||
if (gap_end - cursor > gaps[smallest].end - gaps[smallest].base) {
|
||||
gaps[smallest] = .{ .base = cursor, .end = gap_end };
|
||||
}
|
||||
}
|
||||
if (index < below_count and below[index].end > cursor) cursor = below[index].end;
|
||||
}
|
||||
}
|
||||
|
||||
/// Record and size the memory/IO windows named by a device's Base Address
|
||||
/// Registers. Sizing is the standard probe: disable decode, write all-ones, read
|
||||
/// back the writable (address) bits, restore. `size = ~mask + 1`.
|
||||
fn addBars(node: *device_model.Device, configuration: [*]align(1) u8) void {
|
||||
// Stop the device decoding its BARs while we transiently write all-ones.
|
||||
const command = rd(u16, configuration, 0x04);
|
||||
wr(u16, configuration, 0x04, command & ~@as(u16, 0b11));
|
||||
|
||||
var i: usize = 0;
|
||||
while (i < 6) : (i += 1) {
|
||||
const off = 0x10 + i * 4;
|
||||
const orig = rd(u32, configuration, off);
|
||||
if (orig == 0) continue;
|
||||
|
||||
if (orig & 1 != 0) {
|
||||
// I/O-space BAR (16-bit address space on x86).
|
||||
wr(u32, configuration, off, 0xFFFF_FFFF);
|
||||
const readback = rd(u32, configuration, off);
|
||||
wr(u32, configuration, off, orig);
|
||||
const mask = readback & 0xFFFF_FFFC;
|
||||
const size: u32 = if (mask == 0) 0 else (~mask +% 1) & 0xFFFF;
|
||||
_ = node.addResource(.io_port, orig & 0xFFFF_FFFC, size);
|
||||
} else if ((orig >> 1) & 0x3 == 2) {
|
||||
// 64-bit memory BAR: this BAR pair spans two configuration slots.
|
||||
const orig_hi = rd(u32, configuration, off + 4);
|
||||
wr(u32, configuration, off, 0xFFFF_FFFF);
|
||||
wr(u32, configuration, off + 4, 0xFFFF_FFFF);
|
||||
const lo = rd(u32, configuration, off);
|
||||
const hi = rd(u32, configuration, off + 4);
|
||||
wr(u32, configuration, off, orig);
|
||||
wr(u32, configuration, off + 4, orig_hi);
|
||||
const readback = (@as(u64, hi) << 32) | (lo & 0xFFFF_FFF0);
|
||||
const size: u64 = if (readback == 0) 0 else ~readback +% 1;
|
||||
const address = (@as(u64, orig_hi) << 32) | (orig & 0xFFFF_FFF0);
|
||||
_ = node.addResource(.memory, address, size);
|
||||
i += 1; // consumed the high half
|
||||
} else {
|
||||
// 32-bit memory BAR.
|
||||
wr(u32, configuration, off, 0xFFFF_FFFF);
|
||||
const readback = rd(u32, configuration, off);
|
||||
wr(u32, configuration, off, orig);
|
||||
const mask = readback & 0xFFFF_FFF0;
|
||||
const size: u32 = if (mask == 0) 0 else ~mask +% 1;
|
||||
_ = node.addResource(.memory, orig & 0xFFFF_FFF0, size);
|
||||
}
|
||||
for (gaps) |gap| {
|
||||
if (gap.end > gap.base) _ = bridge.addResource(.memory, gap.base, gap.end - gap.base);
|
||||
}
|
||||
|
||||
wr(u16, configuration, 0x04, command); // restore decode
|
||||
_ = bridge.addResource(.memory, high_end, (@as(u64, 1) << 46) - high_end);
|
||||
}
|
||||
|
||||
/// HPET -> a timer node with its register block as an MMIO resource, plus the GSI
|
||||
@@ -712,6 +729,7 @@ const fadt_pm1a_cnt_blk = 64; // u32 (I/O port)
|
||||
const fadt_pm1b_cnt_blk = 68; // u32 (I/O port)
|
||||
const fadt_pm_tmr_blk = 76; // u32 (I/O port) — the PM timer counter
|
||||
const fadt_pm1_cnt_len = 89; // u8 (bytes)
|
||||
const fadt_sci_int = 46; // u16 (the SCI's GSI)
|
||||
const fadt_flags = 112; // u32
|
||||
const fadt_reset_register = 116; // GAS (12 bytes)
|
||||
const fadt_reset_value = 128; // u8
|
||||
@@ -729,6 +747,7 @@ fn parseFadt(header: *const SystemDescriptorTableHeader) void {
|
||||
const len: usize = header.length;
|
||||
const pi = &power_information;
|
||||
|
||||
pi.sci_interrupt = @truncate(fadt(u16, base, len, fadt_sci_int) orelse 0);
|
||||
pi.smi_cmd = @truncate(fadt(u32, base, len, fadt_smi_cmd) orelse 0);
|
||||
pi.acpi_enable = fadt(u8, base, len, fadt_acpi_enable) orelse 0;
|
||||
pi.acpi_disable = fadt(u8, base, len, fadt_acpi_disable) orelse 0;
|
||||
@@ -808,342 +827,6 @@ fn parseDmar(hal: Hal, header: *const SystemDescriptorTableHeader) void {
|
||||
}
|
||||
}
|
||||
|
||||
// --- AML namespace -> generic device tree -----------------------------------
|
||||
|
||||
/// The PCI bus context while descending the ACPI namespace: the generic host
|
||||
/// bridge whose children ACPI address (`_ADR`) devices resolve against, and the bus number.
|
||||
const PciContext = struct { bridge: *device_model.Device, bus: u8 };
|
||||
|
||||
/// Mirror the ACPI namespace's Device objects into the generic tree, *merging*
|
||||
/// them with the PCI-enumerated nodes: a PCI root bridge (`PNP0A03`/`PNP0A08`)
|
||||
/// folds onto the existing `pci_host_bridge`, and each addressed (`_ADR`) device folds onto
|
||||
/// the matching PCI function (annotating it with the ACPI hardware ID (`_HID`) and nesting the
|
||||
/// ACPI-only children — keyboard, RTC, … — beneath it). Namespace devices with no
|
||||
/// PCI match land under a synthetic `acpi` node.
|
||||
fn wireAcpiDevices(device_tree: *DeviceTree, aml_namespace: *aml.Namespace, hal: Hal) !void {
|
||||
var arena = std.heap.ArenaAllocator.init(device_tree.allocator);
|
||||
defer arena.deinit();
|
||||
var interpreter = aml.Interpreter.init(aml_namespace, .{
|
||||
.mapMmio = hal.mapMmio,
|
||||
.pioRead = hal.pioRead,
|
||||
.pioWrite = hal.pioWrite,
|
||||
}, arena.allocator());
|
||||
|
||||
const acpi_root = try device_tree.addChild(device_tree.root, .unknown, "acpi");
|
||||
try mirrorDevices(device_tree, aml_namespace.root, acpi_root, null, &interpreter);
|
||||
}
|
||||
|
||||
fn mirrorDevices(device_tree: *DeviceTree, node: *aml.Node, parent_device: *device_model.Device, context: ?PciContext, interpreter: *aml.Interpreter) (error{OutOfMemory})!void {
|
||||
var child = node.first_child;
|
||||
while (child) |c| : (child = c.next_sibling) {
|
||||
if (c.kind != .device) {
|
||||
// A scope — the System Bus (\_SB), General Purpose Events (\_GPE), … —
|
||||
// descend without adding a node.
|
||||
try mirrorDevices(device_tree, c, parent_device, context, interpreter);
|
||||
continue;
|
||||
}
|
||||
|
||||
// Skip devices the firmware reports as not present (via a device-status (`_STA`) method),
|
||||
// along with their whole subtree — per the ACPI rules.
|
||||
if (!devicePresent(interpreter, c)) continue;
|
||||
|
||||
var mirrored_device: *device_model.Device = undefined;
|
||||
var child_context = context;
|
||||
|
||||
if (isPciRootNode(c)) {
|
||||
// The PCI root bridge folds onto the generic host bridge.
|
||||
mirrored_device = matchHostBridge(device_tree) orelse
|
||||
try device_tree.addChild(parent_device, .acpi_device, &c.segment);
|
||||
child_context = .{ .bridge = mirrored_device, .bus = 0 };
|
||||
} else {
|
||||
// An addressed device folds onto its matching PCI function; anything
|
||||
// else becomes a fresh node under the current parent.
|
||||
mirrored_device = pick: {
|
||||
if (context) |pc| {
|
||||
if (readAdr(c)) |adr| {
|
||||
if (findPciNode(pc.bridge, pc.bus, adr)) |pnode| break :pick pnode;
|
||||
}
|
||||
}
|
||||
break :pick try device_tree.addChild(parent_device, .acpi_device, &c.segment);
|
||||
};
|
||||
}
|
||||
|
||||
applyHid(mirrored_device, c, interpreter);
|
||||
applyCrs(mirrored_device, c, interpreter);
|
||||
try mirrorDevices(device_tree, c, mirrored_device, child_context, interpreter);
|
||||
}
|
||||
}
|
||||
|
||||
/// Evaluate a device's status (`_STA`) to decide if it is present. An absent status
|
||||
/// (`_STA`) means present by default; an evaluation failure is treated as present too (we'd
|
||||
/// rather over-report than hide a device we couldn't introspect).
|
||||
fn devicePresent(interpreter: *aml.Interpreter, node: *aml.Node) bool {
|
||||
const sta = aml.Namespace.childOf(node, seg4("_STA")) orelse return true;
|
||||
const obj = interpreter.evaluate(sta, &.{}) catch return true;
|
||||
const status = obj.asInteger() catch return true;
|
||||
return (status & 0x01) != 0; // bit 0 = present
|
||||
}
|
||||
|
||||
/// The first PCI host bridge in the generic tree (segment 0).
|
||||
fn matchHostBridge(device_tree: *DeviceTree) ?*device_model.Device {
|
||||
var c = device_tree.root.first_child;
|
||||
while (c) |ch| : (c = ch.next_sibling) {
|
||||
if (ch.class == .pci_host_bridge) return ch;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// The PCI function node under `bridge` at the address the device's address object
|
||||
/// (`_ADR`) names (device/function on
|
||||
/// `bus`), or null.
|
||||
fn findPciNode(bridge: *device_model.Device, bus: u8, adr: u32) ?*device_model.Device {
|
||||
const device: u16 = @truncate((adr >> 16) & 0x1F);
|
||||
const function: u16 = @truncate(adr & 0x7);
|
||||
const target: u16 = (@as(u16, bus) << 8) | (device << 3) | function;
|
||||
var c = bridge.first_child;
|
||||
while (c) |ch| : (c = ch.next_sibling) {
|
||||
if (ch.ids.pci_bdf) |bdf| {
|
||||
if (bdf == target) return ch;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// A device's address (`_ADR`) — a static integer Name — or null.
|
||||
fn readAdr(node: *aml.Node) ?u32 {
|
||||
const n = aml.Namespace.childOf(node, seg4("_ADR")) orelse return null;
|
||||
if (n.kind != .name) return null;
|
||||
var p: usize = 0;
|
||||
return @truncate(readIntObj(n.value, &p) orelse return null);
|
||||
}
|
||||
|
||||
/// Whether a `_HID` string names a PCI(e) host bridge.
|
||||
fn isPciRootHid(hid: []const u8) bool {
|
||||
const id = acpi_ids.HardwareId.fromHid(hid) orelse return false;
|
||||
return id == .pci_bus or id == .pci_express_root_bridge;
|
||||
}
|
||||
|
||||
/// Whether a namespace device is a PCI(e) host bridge. A packed EISA id is decoded
|
||||
/// to its string form first, so both encodings answer through the one registry.
|
||||
fn isPciRootNode(node: *aml.Node) bool {
|
||||
const hid = aml.Namespace.childOf(node, seg4("_HID")) orelse return false;
|
||||
if (hid.kind != .name or hid.value.len == 0) return false;
|
||||
switch (hid.value[0]) {
|
||||
0x00, 0x01, 0xFF, 0x0A, 0x0B, 0x0C, 0x0E => {
|
||||
var p: usize = 0;
|
||||
const n = readIntObj(hid.value, &p) orelse return false;
|
||||
var buffer: [8]u8 = undefined;
|
||||
return isPciRootHid(eisaIdToStr(@truncate(n), &buffer));
|
||||
},
|
||||
0x0D => return isPciRootHid(cstr(hid.value[1..])),
|
||||
else => return false,
|
||||
}
|
||||
}
|
||||
|
||||
/// Read a device's hardware ID (`_HID`) into the generic device: an integer decodes as an EISA
|
||||
/// id ("PNP0A03"), a string is taken verbatim. Handles both the common static
|
||||
/// Name form and a Method form (evaluated).
|
||||
fn applyHid(device: *device_model.Device, node: *aml.Node, interpreter: *aml.Interpreter) void {
|
||||
const hid = aml.Namespace.childOf(node, seg4("_HID")) orelse return;
|
||||
if (hid.kind == .method) {
|
||||
const obj = interpreter.evaluate(hid, &.{}) catch return;
|
||||
switch (obj) {
|
||||
.integer => |n| setEisaHid(device, @truncate(n)),
|
||||
.string => |s| device.setHid(s),
|
||||
else => {},
|
||||
}
|
||||
return;
|
||||
}
|
||||
if (hid.kind != .name or hid.value.len == 0) return;
|
||||
const v = hid.value;
|
||||
switch (v[0]) {
|
||||
0x00, 0x01, 0xFF, 0x0A, 0x0B, 0x0C, 0x0E => {
|
||||
var p: usize = 0;
|
||||
const n = readIntObj(v, &p) orelse return;
|
||||
setEisaHid(device, @truncate(n));
|
||||
},
|
||||
0x0D => device.setHid(cstr(v[1..])), // StringPrefix
|
||||
else => {},
|
||||
}
|
||||
}
|
||||
|
||||
fn setEisaHid(device: *device_model.Device, id: u32) void {
|
||||
device.ids.acpi_hid = id;
|
||||
var buffer: [8]u8 = undefined;
|
||||
device.setHid(eisaIdToStr(id, &buffer));
|
||||
}
|
||||
|
||||
/// Parse a device's current resource settings (`_CRS`). The evaluator handles both the static
|
||||
/// `Buffer` form (a `Name`) and the method form uniformly, yielding the
|
||||
/// ResourceTemplate bytes we then decode.
|
||||
fn applyCrs(device: *device_model.Device, node: *aml.Node, interpreter: *aml.Interpreter) void {
|
||||
const crs = aml.Namespace.childOf(node, seg4("_CRS")) orelse return;
|
||||
const obj = interpreter.evaluate(crs, &.{}) catch return;
|
||||
const buffer = switch (obj) {
|
||||
.buffer => |b| b,
|
||||
else => return,
|
||||
};
|
||||
parseResourceTemplate(device, buffer);
|
||||
}
|
||||
|
||||
/// Walk a ResourceTemplate byte list, adding recognised descriptors as resources.
|
||||
fn parseResourceTemplate(device: *device_model.Device, bytes: []const u8) void {
|
||||
var i: usize = 0;
|
||||
while (i < bytes.len) {
|
||||
const tag = bytes[i];
|
||||
if (tag & 0x80 == 0) {
|
||||
// Small descriptor: length in low 3 bits, type in bits [6:3].
|
||||
const len: usize = tag & 0x07;
|
||||
const body = i + 1;
|
||||
if (body + len > bytes.len) break;
|
||||
switch ((tag >> 3) & 0x0F) {
|
||||
0x04 => if (len >= 2) { // IRQ: a 16-bit mask, one resource per set bit
|
||||
const mask = @as(u16, bytes[body]) | (@as(u16, bytes[body + 1]) << 8);
|
||||
var b: usize = 0;
|
||||
while (b < 16) : (b += 1) {
|
||||
if (mask & (@as(u16, 1) << @intCast(b)) != 0) _ = device.addResource(.irq, b, 1);
|
||||
}
|
||||
},
|
||||
0x08 => if (len >= 7) { // IO port: minimum at +1, length at +6
|
||||
_ = device.addResource(.io_port, rd16(bytes, body + 1), bytes[body + 6]);
|
||||
},
|
||||
0x09 => if (len >= 3) { // Fixed IO: base at +0, length at +2
|
||||
_ = device.addResource(.io_port, rd16(bytes, body), bytes[body + 2]);
|
||||
},
|
||||
0x0F => break, // EndTag
|
||||
else => {},
|
||||
}
|
||||
i = body + len;
|
||||
} else {
|
||||
// Large descriptor: 16-bit length follows the tag.
|
||||
if (i + 3 > bytes.len) break;
|
||||
const len: usize = @intCast(rd16(bytes, i + 1));
|
||||
const body = i + 3;
|
||||
if (body + len > bytes.len) break;
|
||||
switch (tag) {
|
||||
0x85 => if (len >= 17) { // Memory32: minimum at +1, length at +13
|
||||
_ = device.addResource(.memory, rd32(bytes, body + 1), rd32(bytes, body + 13));
|
||||
},
|
||||
0x86 => if (len >= 9) { // Memory32Fixed: base at +1, length at +5
|
||||
_ = device.addResource(.memory, rd32(bytes, body + 1), rd32(bytes, body + 5));
|
||||
},
|
||||
0x89 => if (len >= 2) { // Extended IRQ: count at +1, then count u32s
|
||||
const count = bytes[body + 1];
|
||||
var k: usize = 0;
|
||||
while (k < count and body + 2 + k * 4 + 4 <= body + len) : (k += 1) {
|
||||
_ = device.addResource(.irq, rd32(bytes, body + 2 + k * 4), 1);
|
||||
}
|
||||
},
|
||||
0x87, 0x88, 0x8A => parseAddressSpace(device, tag, bytes[body .. body + len]),
|
||||
else => {},
|
||||
}
|
||||
i = body + len;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Word/DWord/QWord address-space descriptors: resource type at [0], then
|
||||
/// granularity/minimum/maximum/translation/length, each of width `w`.
|
||||
fn parseAddressSpace(device: *device_model.Device, tag: u8, body: []const u8) void {
|
||||
const w: usize = switch (tag) {
|
||||
0x88 => 2, // Word
|
||||
0x87 => 4, // DWord
|
||||
else => 8, // QWord (0x8A)
|
||||
};
|
||||
if (body.len < 3 + 5 * w) return;
|
||||
const minimum = readN(body, 3 + w, w);
|
||||
const length = readN(body, 3 + 4 * w, w);
|
||||
const kind: device_model.ResourceKind = switch (body[0]) {
|
||||
0 => .memory,
|
||||
1 => .io_port,
|
||||
else => .bus_range,
|
||||
};
|
||||
_ = device.addResource(kind, minimum, length);
|
||||
}
|
||||
|
||||
/// Decode a packed EISA id into its 7-char string (e.g. 0x030AD041 -> "PNP0A03").
|
||||
fn eisaIdToStr(id: u32, buffer: *[8]u8) []const u8 {
|
||||
const b0: u16 = @intCast(id & 0xFF);
|
||||
const b1: u16 = @intCast((id >> 8) & 0xFF);
|
||||
const b2: u8 = @truncate(id >> 16);
|
||||
const b3: u8 = @truncate(id >> 24);
|
||||
const mfg = (b0 << 8) | b1;
|
||||
buffer[0] = '@' + @as(u8, @intCast((mfg >> 10) & 0x1F));
|
||||
buffer[1] = '@' + @as(u8, @intCast((mfg >> 5) & 0x1F));
|
||||
buffer[2] = '@' + @as(u8, @intCast(mfg & 0x1F));
|
||||
buffer[3] = hexDigit((b2 >> 4) & 0xF);
|
||||
buffer[4] = hexDigit(b2 & 0xF);
|
||||
buffer[5] = hexDigit((b3 >> 4) & 0xF);
|
||||
buffer[6] = hexDigit(b3 & 0xF);
|
||||
return buffer[0..7];
|
||||
}
|
||||
|
||||
fn hexDigit(n: u8) u8 {
|
||||
return if (n < 10) '0' + n else 'A' + (n - 10);
|
||||
}
|
||||
|
||||
fn seg4(comptime s: *const [4:0]u8) [4]u8 {
|
||||
return s[0..4].*;
|
||||
}
|
||||
|
||||
fn cstr(bytes: []const u8) []const u8 {
|
||||
const index = std.mem.indexOfScalar(u8, bytes, 0) orelse bytes.len;
|
||||
return bytes[0..index];
|
||||
}
|
||||
|
||||
const PkgLen = struct { value: usize, size: usize };
|
||||
|
||||
fn packageLength(bytes: []const u8, p: usize) ?PkgLen {
|
||||
if (p >= bytes.len) return null;
|
||||
const lead = bytes[p];
|
||||
const follow: usize = lead >> 6;
|
||||
if (p + 1 + follow > bytes.len) return null;
|
||||
if (follow == 0) return .{ .value = lead & 0x3F, .size = 1 };
|
||||
var value: usize = lead & 0x0F;
|
||||
var i: usize = 0;
|
||||
while (i < follow) : (i += 1) value |= @as(usize, bytes[p + 1 + i]) << @intCast(4 + i * 8);
|
||||
return .{ .value = value, .size = 1 + follow };
|
||||
}
|
||||
|
||||
/// Read an AML integer object at `p`, advancing `p` past it.
|
||||
fn readIntObj(bytes: []const u8, p: *usize) ?u64 {
|
||||
if (p.* >= bytes.len) return null;
|
||||
const opcode = bytes[p.*];
|
||||
p.* += 1;
|
||||
return switch (opcode) {
|
||||
0x00 => 0,
|
||||
0x01 => 1,
|
||||
0xFF => 0xFF,
|
||||
0x0A => readLE(bytes, p, 1),
|
||||
0x0B => readLE(bytes, p, 2),
|
||||
0x0C => readLE(bytes, p, 4),
|
||||
0x0E => readLE(bytes, p, 8),
|
||||
else => null,
|
||||
};
|
||||
}
|
||||
|
||||
fn readLE(bytes: []const u8, p: *usize, n: usize) ?u64 {
|
||||
if (p.* + n > bytes.len) return null;
|
||||
const v = readN(bytes, p.*, n);
|
||||
p.* += n;
|
||||
return v;
|
||||
}
|
||||
|
||||
fn readN(bytes: []const u8, off: usize, n: usize) u64 {
|
||||
var v: u64 = 0;
|
||||
var k: usize = 0;
|
||||
while (k < n and off + k < bytes.len) : (k += 1) v |= @as(u64, bytes[off + k]) << @intCast(k * 8);
|
||||
return v;
|
||||
}
|
||||
|
||||
fn rd16(bytes: []const u8, off: usize) u64 {
|
||||
return readN(bytes, off, 2);
|
||||
}
|
||||
|
||||
fn rd32(bytes: []const u8, off: usize) u64 {
|
||||
return readN(bytes, off, 4);
|
||||
}
|
||||
|
||||
// --- helpers ----------------------------------------------------------------
|
||||
|
||||
/// Sum `len` bytes; an ACPI table/pointer is valid when the low 8 bits are zero.
|
||||
@@ -1184,58 +867,9 @@ fn readCntRegister(base: [*]align(1) const u8, len: usize, xoff: usize, legacy_o
|
||||
return .{ .mmio = false, .address = port, .width = width };
|
||||
}
|
||||
|
||||
/// The mapped configuration space of one PCI function (its 4 KiB ECAM page). Mapped
|
||||
/// writable so BAR sizing can probe it; reads and writes both go through here.
|
||||
fn pciConfigurationPtr(alloc: McfgAllocation, hal: Hal, bus: u8, device: u8, function: u8) [*]align(1) u8 {
|
||||
const physical = alloc.base_address +
|
||||
(@as(u64, bus - alloc.start_bus) << 20) +
|
||||
(@as(u64, device) << 15) +
|
||||
(@as(u64, function) << 12);
|
||||
// Map the configuration page (writable, for BAR sizing) and use the virtual
|
||||
// address the HAL hands back.
|
||||
return @ptrFromInt(hal.mapMmio(physical, abi.page_size, true));
|
||||
}
|
||||
|
||||
/// Read a little-endian integer at `off` from a (possibly unaligned) byte pointer.
|
||||
/// x86 is little-endian and native, so an unaligned load suffices.
|
||||
fn rd(comptime T: type, bytes: [*]align(1) const u8, off: usize) T {
|
||||
const p: *align(1) const T = @ptrCast(bytes + off);
|
||||
return p.*;
|
||||
}
|
||||
|
||||
/// Write a little-endian integer at `off` through a (possibly unaligned) pointer.
|
||||
fn wr(comptime T: type, bytes: [*]align(1) u8, off: usize, value: T) void {
|
||||
const p: *align(1) T = @ptrCast(bytes + off);
|
||||
p.* = value;
|
||||
}
|
||||
|
||||
// --- tests ------------------------------------------------------------------
|
||||
|
||||
test "eisaIdToStr decodes a packed EISA id" {
|
||||
var buffer: [8]u8 = undefined;
|
||||
// 0x030AD041 is the well-known encoding of "PNP0A03" (PCI root bridge).
|
||||
try std.testing.expectEqualStrings("PNP0A03", eisaIdToStr(0x030AD041, &buffer));
|
||||
}
|
||||
|
||||
test "parseResourceTemplate extracts IO, IRQ, and fixed memory" {
|
||||
// ResourceTemplate { IO(minimum 0x60, len 8), IRQ(4), Memory32Fixed(0xFED00000, 0x1000) }
|
||||
const runtime = [_]u8{
|
||||
0x47, 0x01, 0x60, 0x00, 0x60, 0x00, 0x01, 0x08, // small IO descriptor
|
||||
0x22, 0x10, 0x00, // small IRQ descriptor (mask bit 4 -> IRQ 4)
|
||||
0x86, 0x09, 0x00, 0x01, 0x00, 0x00, 0xD0, 0xFE, 0x00, 0x10, 0x00, 0x00, // Memory32Fixed
|
||||
0x79, 0x00, // EndTag
|
||||
};
|
||||
var device = device_model.Device{};
|
||||
parseResourceTemplate(&device, &runtime);
|
||||
|
||||
try std.testing.expectEqual(@as(u8, 3), device.resource_count);
|
||||
const rs = device.resources[0..device.resource_count];
|
||||
try std.testing.expectEqual(device_model.ResourceKind.io_port, rs[0].kind);
|
||||
try std.testing.expectEqual(@as(u64, 0x60), rs[0].start);
|
||||
try std.testing.expectEqual(@as(u64, 8), rs[0].len);
|
||||
try std.testing.expectEqual(device_model.ResourceKind.irq, rs[1].kind);
|
||||
try std.testing.expectEqual(@as(u64, 4), rs[1].start);
|
||||
try std.testing.expectEqual(device_model.ResourceKind.memory, rs[2].kind);
|
||||
try std.testing.expectEqual(@as(u64, 0xFED00000), rs[2].start);
|
||||
try std.testing.expectEqual(@as(u64, 0x1000), rs[2].len);
|
||||
}
|
||||
|
||||
@@ -12,6 +12,11 @@ const std = @import("std");
|
||||
const opcode = @import("opcodes.zig");
|
||||
const parser = @import("parser.zig");
|
||||
|
||||
/// The named AML opcode/prefix bytes (`zero_opcode`, `byte_prefix`, …). Re-exported so
|
||||
/// callers that decode raw AML bytes — e.g. the acpi service reading a `_HID` integer —
|
||||
/// name the opcodes instead of writing bare 0x0A/0x0B/… literals (docs/coding-standards.md).
|
||||
pub const opcodes = @import("opcodes.zig");
|
||||
|
||||
pub const Namespace = @import("namespace.zig").Namespace;
|
||||
pub const Node = @import("namespace.zig").Node;
|
||||
pub const NodeKind = @import("namespace.zig").NodeKind;
|
||||
@@ -50,6 +55,20 @@ pub fn parse(allocator: std.mem.Allocator, blocks: []const []const u8) !ParseRes
|
||||
return .{ .namespace = namespace, .consumed = consumed, .total = total };
|
||||
}
|
||||
|
||||
/// Count the Device objects in a parsed namespace — what the acpi service
|
||||
/// (docs/discovery.md) reports, and what the kernel's own parse counts
|
||||
/// so the two can be checked equal across the ring-3 move.
|
||||
pub fn deviceCount(namespace: *const Namespace) usize {
|
||||
return countKind(namespace.root, .device);
|
||||
}
|
||||
|
||||
fn countKind(node: *const Node, kind: NodeKind) usize {
|
||||
var n: usize = if (node.kind == kind) 1 else 0;
|
||||
var c = node.first_child;
|
||||
while (c) |child| : (c = child.next_sibling) n += countKind(child, kind);
|
||||
return n;
|
||||
}
|
||||
|
||||
/// Look up the `\_S{state}` sleep package in a parsed namespace and return its
|
||||
/// first two integer elements (SLP_TYP for PM1a / PM1b), or null if absent.
|
||||
pub fn sleepState(namespace: *Namespace, state: u8) ?SleepType {
|
||||
@@ -211,3 +230,31 @@ test "interpreter runs a method with args, arithmetic, and control flow" {
|
||||
const lo = try interpreter.evaluate(tst, &.{.{ .integer = 2 }}); // 2+5=7 !> 10 -> 0
|
||||
try std.testing.expectEqual(@as(u64, 0), try lo.asInteger());
|
||||
}
|
||||
|
||||
test "interpreter records Notify(device, code)" {
|
||||
// Device(DEV_) { Name(_HID, 0x030AD041) } // PNP0A03-ish placeholder
|
||||
// Method(TST_, 0) { Notify(DEV_, 0x80); Return(Zero) }
|
||||
// Encoded: a Device holding a Name, then a Method issuing Notify on it.
|
||||
const blob = [_]u8{
|
||||
0x5B, 0x82, 0x0F, 0x44, 0x45, 0x56, 0x5F, // Device(DEV_) len=0x0F (pkglen + DEV_ + Name)
|
||||
0x08, 0x5F, 0x48, 0x49, 0x44, 0x0C, 0x41, 0xD0, 0x0A, 0x03, // Name(_HID, DWord 0x030AD041)
|
||||
0x14, 0x0F, 0x54, 0x53, 0x54, 0x5F, 0x00, // Method(TST_, 0) len=0x0F (pkglen + TST_ + flags + body)
|
||||
0x86, 0x44, 0x45, 0x56, 0x5F, 0x0A, 0x80, // Notify(DEV_, 0x80)
|
||||
0xA4, 0x00, // Return(Zero)
|
||||
};
|
||||
|
||||
var arena = std.heap.ArenaAllocator.init(std.testing.allocator);
|
||||
defer arena.deinit();
|
||||
var result = try parse(arena.allocator(), &.{&blob});
|
||||
const namespace = &result.namespace;
|
||||
const tst = namespace.resolve(namespace.root, false, 0, &.{.{ 'T', 'S', 'T', '_' }}) orelse return error.NoMethod;
|
||||
const dev = namespace.resolve(namespace.root, false, 0, &.{.{ 'D', 'E', 'V', '_' }}) orelse return error.NoDevice;
|
||||
|
||||
var interpreter = Interpreter.init(namespace, .{ .mapMmio = noMap, .pioRead = noRead, .pioWrite = noWrite }, arena.allocator());
|
||||
_ = try interpreter.evaluate(tst, &.{});
|
||||
|
||||
const events = interpreter.takeNotifications();
|
||||
try std.testing.expectEqual(@as(usize, 1), events.len);
|
||||
try std.testing.expectEqual(dev, events[0].node);
|
||||
try std.testing.expectEqual(@as(u64, 0x80), events[0].code);
|
||||
}
|
||||
|
||||
@@ -141,6 +141,9 @@ const Frame = struct {
|
||||
/// A CreateField binding: a name that indexes into a buffer object.
|
||||
const BufferField = struct { buffer: *Node, byte_off: usize, bit_width: u32 };
|
||||
|
||||
/// One Notify(device, code) the interpreter executed.
|
||||
pub const NotifyEvent = struct { node: *Node, code: u64 };
|
||||
|
||||
pub const Interpreter = struct {
|
||||
namespace: *Namespace,
|
||||
hal: Hal,
|
||||
@@ -149,6 +152,11 @@ pub const Interpreter = struct {
|
||||
dynamic_overrides: std.AutoHashMapUnmanaged(*Node, Object) = .{},
|
||||
/// CreateField bindings active for the current evaluation.
|
||||
fields: std.AutoHashMapUnmanaged(*Node, BufferField) = .{},
|
||||
/// Notify(device, code) operations the last evaluation executed — a GPE or
|
||||
/// EC handler tells the OS "look at this device" this way. Bounded; the
|
||||
/// caller drains it with `takeNotifications` after `evaluate` (M21).
|
||||
notify_queue: [16]NotifyEvent = undefined,
|
||||
notify_count: usize = 0,
|
||||
|
||||
pub fn init(namespace: *Namespace, hal: Hal, arena: std.mem.Allocator) Interpreter {
|
||||
return .{ .namespace = namespace, .hal = hal, .arena = arena };
|
||||
@@ -157,6 +165,7 @@ pub const Interpreter = struct {
|
||||
/// Evaluate a namespace object: invoke a Method, read a Name's value, or read a
|
||||
/// Field. Resets per-evaluation runtime state first.
|
||||
pub fn evaluate(self: *Interpreter, node: *Node, args: []const Object) Error!Object {
|
||||
self.notify_count = 0;
|
||||
self.dynamic_overrides.clearRetainingCapacity();
|
||||
self.fields.clearRetainingCapacity();
|
||||
return self.invoke(node, args);
|
||||
@@ -267,6 +276,8 @@ pub const Interpreter = struct {
|
||||
},
|
||||
opcode.to_buffer_opcode => try self.passThroughUnary(current, frame),
|
||||
|
||||
opcode.notify_opcode => try self.notify(current, frame),
|
||||
|
||||
opcode.extended_opcode_prefix => try self.ext(current, frame),
|
||||
|
||||
// CreateXField: source, index, name (bit widths differ by op)
|
||||
@@ -542,6 +553,36 @@ pub const Interpreter = struct {
|
||||
try self.storeInto(current, frame, value);
|
||||
}
|
||||
|
||||
/// Notify(SuperName, NotifyValue): resolve the named device, evaluate the
|
||||
/// code, and record the pair for the caller to dispatch. AML control flow
|
||||
/// continues (Notify returns nothing).
|
||||
fn notify(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||
const lead = current.peek() orelse return error.Truncated;
|
||||
var target: ?*Node = null;
|
||||
if (isNameStart(lead)) {
|
||||
const name_path = try current.nameString();
|
||||
target = self.namespace.resolve(frame.scope, name_path.rooted, name_path.parents, name_path.slice());
|
||||
} else {
|
||||
// A non-name SuperName (Local/Arg holding a reference).
|
||||
const obj = try self.term(current, frame);
|
||||
if (obj == .reference) target = obj.reference;
|
||||
}
|
||||
const code = try self.evaluateInteger(current, frame);
|
||||
if (target) |node| {
|
||||
if (self.notify_count < self.notify_queue.len) {
|
||||
self.notify_queue[self.notify_count] = .{ .node = node, .code = code };
|
||||
self.notify_count += 1;
|
||||
}
|
||||
}
|
||||
return .uninitialized;
|
||||
}
|
||||
|
||||
/// The Notify events the last `evaluate` produced. Valid until the next
|
||||
/// `evaluate` clears the queue.
|
||||
pub fn takeNotifications(self: *Interpreter) []const NotifyEvent {
|
||||
return self.notify_queue[0..self.notify_count];
|
||||
}
|
||||
|
||||
fn storeInto(self: *Interpreter, current: *Cursor, frame: *Frame, value: Object) Error!void {
|
||||
const lead = current.peek() orelse return error.Truncated;
|
||||
if (isNameStart(lead)) {
|
||||
|
||||
@@ -28,6 +28,11 @@ pub const DeviceClass = enum(u32) {
|
||||
/// A device named in the ACPI namespace (from the DSDT/SSDT), carrying a
|
||||
/// hardware ID (`_HID`) and, where static, current resource settings (`_CRS`).
|
||||
acpi_device,
|
||||
/// The ACPI tables themselves, published as one node for the user-space acpi
|
||||
/// service (docs/discovery.md): memory resources over the AML blobs,
|
||||
/// a broad io_port grant for OperationRegion access, and the SCI interrupt.
|
||||
/// The one node whose claimant is trusted to run firmware bytecode.
|
||||
acpi_tables,
|
||||
unknown,
|
||||
};
|
||||
|
||||
|
||||
+496
-189
@@ -7,6 +7,16 @@
|
||||
//! apart. Pure reference data (from the PCI spec; see https://wiki.osdev.org/PCI) — no
|
||||
//! hardware access — so it is shared by kernel discovery (the device-tree dump) and any
|
||||
//! user-space tool (a future lspci, driver matching).
|
||||
//!
|
||||
//! The taxonomy is named, not numbered (docs/coding-standards.md, "Named values"): the
|
||||
//! base class is a `BaseClass` enum, and each class with defined subclasses gets a
|
||||
//! namespace holding its `SubClass` enum (and, where the spec defines them, per-subclass
|
||||
//! `ProgIf` enums) — the same shape as `usb-ids.zig`. Code that *means* a specific class
|
||||
//! names it (`BaseClass.serial_bus`, `serial_bus.usb.ProgIf.xhci`) rather than writing a
|
||||
//! bare 0x0C/0x03/0x30. The `className`/`subclassName`/`progIfName` functions still take
|
||||
//! the raw bytes a function reports in its header, because that is what hardware hands us.
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
/// The three bytes of a PCI class code, unpacked from the `0xCCSSPP` value discovery
|
||||
/// records in `Device.ids.pci_class` (CC = base class, SS = subclass, PP = prog-IF).
|
||||
@@ -22,148 +32,465 @@ pub const ClassCode = struct {
|
||||
.prog_if = @intCast(packed_code & 0xFF),
|
||||
};
|
||||
}
|
||||
|
||||
/// Re-pack the triple into the `0xCCSSPP` form. Lets code name a whole class code
|
||||
/// from its parts — `pack(.{ .base = @intFromEnum(BaseClass.serial_bus), … })` —
|
||||
/// instead of writing the literal 0x0C0330.
|
||||
pub fn pack(self: ClassCode) u24 {
|
||||
return (@as(u24, self.base) << 16) | (@as(u24, self.subclass) << 8) | self.prog_if;
|
||||
}
|
||||
};
|
||||
|
||||
/// Base class (config byte 0x0B). Non-exhaustive: an unlisted code is a real but
|
||||
/// unnamed class, decoded as "Unknown" rather than rejected.
|
||||
pub const BaseClass = enum(u8) {
|
||||
unclassified = 0x00,
|
||||
mass_storage = 0x01,
|
||||
network = 0x02,
|
||||
display = 0x03,
|
||||
multimedia = 0x04,
|
||||
memory = 0x05,
|
||||
bridge = 0x06,
|
||||
simple_communication = 0x07,
|
||||
base_system_peripheral = 0x08,
|
||||
input_device = 0x09,
|
||||
docking_station = 0x0A,
|
||||
processor = 0x0B,
|
||||
serial_bus = 0x0C,
|
||||
wireless = 0x0D,
|
||||
intelligent = 0x0E,
|
||||
satellite_communication = 0x0F,
|
||||
encryption = 0x10,
|
||||
signal_processing = 0x11,
|
||||
processing_accelerator = 0x12,
|
||||
non_essential_instrumentation = 0x13,
|
||||
co_processor = 0x40,
|
||||
unassigned = 0xFF,
|
||||
_,
|
||||
|
||||
pub fn name(self: BaseClass) []const u8 {
|
||||
return switch (self) {
|
||||
.unclassified => "Unclassified",
|
||||
.mass_storage => "Mass Storage Controller",
|
||||
.network => "Network Controller",
|
||||
.display => "Display Controller",
|
||||
.multimedia => "Multimedia Controller",
|
||||
.memory => "Memory Controller",
|
||||
.bridge => "Bridge",
|
||||
.simple_communication => "Simple Communication Controller",
|
||||
.base_system_peripheral => "Base System Peripheral",
|
||||
.input_device => "Input Device Controller",
|
||||
.docking_station => "Docking Station",
|
||||
.processor => "Processor",
|
||||
.serial_bus => "Serial Bus Controller",
|
||||
.wireless => "Wireless Controller",
|
||||
.intelligent => "Intelligent Controller",
|
||||
.satellite_communication => "Satellite Communication Controller",
|
||||
.encryption => "Encryption Controller",
|
||||
.signal_processing => "Signal Processing Controller",
|
||||
.processing_accelerator => "Processing Accelerator",
|
||||
.non_essential_instrumentation => "Non-Essential Instrumentation",
|
||||
.co_processor => "Co-Processor",
|
||||
.unassigned => "Unassigned Class (Vendor specific)",
|
||||
_ => "Unknown",
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
// --- Per-class subclass (and prog-IF) taxonomies --------------------------------------
|
||||
// One namespace per base class that has defined subclasses, named after the class. Each
|
||||
// holds an exhaustive `SubClass` enum (so an unlisted code decodes to the class default,
|
||||
// not a wrong name), and, where the spec assigns them, per-subclass `ProgIf` enums.
|
||||
|
||||
pub const mass_storage = struct {
|
||||
pub const SubClass = enum(u8) {
|
||||
scsi_bus = 0x00,
|
||||
ide = 0x01,
|
||||
floppy = 0x02,
|
||||
ipi_bus = 0x03,
|
||||
raid = 0x04,
|
||||
ata = 0x05,
|
||||
serial_ata = 0x06,
|
||||
serial_attached_scsi = 0x07,
|
||||
non_volatile_memory = 0x08,
|
||||
|
||||
pub fn name(self: SubClass) []const u8 {
|
||||
return switch (self) {
|
||||
.scsi_bus => "SCSI Bus Controller",
|
||||
.ide => "IDE Controller",
|
||||
.floppy => "Floppy Disk Controller",
|
||||
.ipi_bus => "IPI Bus Controller",
|
||||
.raid => "RAID Controller",
|
||||
.ata => "ATA Controller",
|
||||
.serial_ata => "Serial ATA Controller",
|
||||
.serial_attached_scsi => "Serial Attached SCSI Controller",
|
||||
.non_volatile_memory => "Non-Volatile Memory Controller",
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
pub const serial_ata = struct {
|
||||
pub const ProgIf = enum(u8) {
|
||||
vendor_specific = 0x00,
|
||||
ahci = 0x01,
|
||||
serial_storage_bus = 0x02,
|
||||
|
||||
pub fn name(self: ProgIf) []const u8 {
|
||||
return switch (self) {
|
||||
.vendor_specific => "Vendor Specific Interface",
|
||||
.ahci => "AHCI 1.0",
|
||||
.serial_storage_bus => "Serial Storage Bus",
|
||||
};
|
||||
}
|
||||
};
|
||||
};
|
||||
|
||||
pub const non_volatile_memory = struct {
|
||||
pub const ProgIf = enum(u8) {
|
||||
nvmhci = 0x01,
|
||||
nvm_express = 0x02,
|
||||
|
||||
pub fn name(self: ProgIf) []const u8 {
|
||||
return switch (self) {
|
||||
.nvmhci => "NVMHCI",
|
||||
.nvm_express => "NVM Express",
|
||||
};
|
||||
}
|
||||
};
|
||||
};
|
||||
};
|
||||
|
||||
pub const network = struct {
|
||||
pub const SubClass = enum(u8) {
|
||||
ethernet = 0x00,
|
||||
token_ring = 0x01,
|
||||
fddi = 0x02,
|
||||
atm = 0x03,
|
||||
isdn = 0x04,
|
||||
picmg_multi_computing = 0x06,
|
||||
infiniband = 0x07,
|
||||
fabric = 0x08,
|
||||
|
||||
pub fn name(self: SubClass) []const u8 {
|
||||
return switch (self) {
|
||||
.ethernet => "Ethernet Controller",
|
||||
.token_ring => "Token Ring Controller",
|
||||
.fddi => "FDDI Controller",
|
||||
.atm => "ATM Controller",
|
||||
.isdn => "ISDN Controller",
|
||||
.picmg_multi_computing => "PICMG 2.14 Multi Computing Controller",
|
||||
.infiniband => "Infiniband Controller",
|
||||
.fabric => "Fabric Controller",
|
||||
};
|
||||
}
|
||||
};
|
||||
};
|
||||
|
||||
pub const display = struct {
|
||||
pub const SubClass = enum(u8) {
|
||||
vga_compatible = 0x00,
|
||||
xga = 0x01,
|
||||
three_dimensional = 0x02,
|
||||
|
||||
pub fn name(self: SubClass) []const u8 {
|
||||
return switch (self) {
|
||||
.vga_compatible => "VGA Compatible Controller",
|
||||
.xga => "XGA Controller",
|
||||
.three_dimensional => "3D Controller (Not VGA-Compatible)",
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
pub const vga_compatible = struct {
|
||||
pub const ProgIf = enum(u8) {
|
||||
vga = 0x00,
|
||||
compatible_8514 = 0x01,
|
||||
|
||||
pub fn name(self: ProgIf) []const u8 {
|
||||
return switch (self) {
|
||||
.vga => "VGA Controller",
|
||||
.compatible_8514 => "8514-Compatible Controller",
|
||||
};
|
||||
}
|
||||
};
|
||||
};
|
||||
};
|
||||
|
||||
pub const multimedia = struct {
|
||||
pub const SubClass = enum(u8) {
|
||||
video = 0x00,
|
||||
audio = 0x01,
|
||||
telephony = 0x02,
|
||||
audio_device = 0x03,
|
||||
|
||||
pub fn name(self: SubClass) []const u8 {
|
||||
return switch (self) {
|
||||
.video => "Multimedia Video Controller",
|
||||
.audio => "Multimedia Audio Controller",
|
||||
.telephony => "Computer Telephony Device",
|
||||
.audio_device => "Audio Device",
|
||||
};
|
||||
}
|
||||
};
|
||||
};
|
||||
|
||||
pub const memory = struct {
|
||||
pub const SubClass = enum(u8) {
|
||||
ram = 0x00,
|
||||
flash = 0x01,
|
||||
|
||||
pub fn name(self: SubClass) []const u8 {
|
||||
return switch (self) {
|
||||
.ram => "RAM Controller",
|
||||
.flash => "Flash Controller",
|
||||
};
|
||||
}
|
||||
};
|
||||
};
|
||||
|
||||
pub const bridge = struct {
|
||||
pub const SubClass = enum(u8) {
|
||||
host = 0x00,
|
||||
isa = 0x01,
|
||||
eisa = 0x02,
|
||||
mca = 0x03,
|
||||
pci_to_pci = 0x04,
|
||||
pcmcia = 0x05,
|
||||
nubus = 0x06,
|
||||
cardbus = 0x07,
|
||||
raceway = 0x08,
|
||||
pci_to_pci_semi_transparent = 0x09,
|
||||
infiniband_to_pci = 0x0A,
|
||||
|
||||
pub fn name(self: SubClass) []const u8 {
|
||||
return switch (self) {
|
||||
.host => "Host Bridge",
|
||||
.isa => "ISA Bridge",
|
||||
.eisa => "EISA Bridge",
|
||||
.mca => "MCA Bridge",
|
||||
.pci_to_pci => "PCI-to-PCI Bridge",
|
||||
.pcmcia => "PCMCIA Bridge",
|
||||
.nubus => "NuBus Bridge",
|
||||
.cardbus => "CardBus Bridge",
|
||||
.raceway => "RACEway Bridge",
|
||||
.pci_to_pci_semi_transparent => "PCI-to-PCI Bridge (Semi-Transparent)",
|
||||
.infiniband_to_pci => "InfiniBand-to-PCI Host Bridge",
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
pub const pci_to_pci = struct {
|
||||
pub const ProgIf = enum(u8) {
|
||||
normal_decode = 0x00,
|
||||
subtractive_decode = 0x01,
|
||||
|
||||
pub fn name(self: ProgIf) []const u8 {
|
||||
return switch (self) {
|
||||
.normal_decode => "Normal Decode",
|
||||
.subtractive_decode => "Subtractive Decode",
|
||||
};
|
||||
}
|
||||
};
|
||||
};
|
||||
};
|
||||
|
||||
pub const simple_communication = struct {
|
||||
pub const SubClass = enum(u8) {
|
||||
serial = 0x00,
|
||||
parallel = 0x01,
|
||||
multiport_serial = 0x02,
|
||||
modem = 0x03,
|
||||
gpib = 0x04,
|
||||
smart_card = 0x05,
|
||||
|
||||
pub fn name(self: SubClass) []const u8 {
|
||||
return switch (self) {
|
||||
.serial => "Serial Controller",
|
||||
.parallel => "Parallel Controller",
|
||||
.multiport_serial => "Multiport Serial Controller",
|
||||
.modem => "Modem",
|
||||
.gpib => "IEEE 488.1/2 (GPIB) Controller",
|
||||
.smart_card => "Smart Card Controller",
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
pub const serial = struct {
|
||||
pub const ProgIf = enum(u8) {
|
||||
compatible_8250 = 0x00,
|
||||
compatible_16450 = 0x01,
|
||||
compatible_16550 = 0x02,
|
||||
compatible_16650 = 0x03,
|
||||
compatible_16750 = 0x04,
|
||||
compatible_16850 = 0x05,
|
||||
compatible_16950 = 0x06,
|
||||
|
||||
pub fn name(self: ProgIf) []const u8 {
|
||||
return switch (self) {
|
||||
.compatible_8250 => "8250-Compatible (Generic XT)",
|
||||
.compatible_16450 => "16450-Compatible",
|
||||
.compatible_16550 => "16550-Compatible",
|
||||
.compatible_16650 => "16650-Compatible",
|
||||
.compatible_16750 => "16750-Compatible",
|
||||
.compatible_16850 => "16850-Compatible",
|
||||
.compatible_16950 => "16950-Compatible",
|
||||
};
|
||||
}
|
||||
};
|
||||
};
|
||||
};
|
||||
|
||||
pub const base_system_peripheral = struct {
|
||||
pub const SubClass = enum(u8) {
|
||||
pic = 0x00,
|
||||
dma = 0x01,
|
||||
timer = 0x02,
|
||||
rtc = 0x03,
|
||||
pci_hot_plug = 0x04,
|
||||
sd_host = 0x05,
|
||||
iommu = 0x06,
|
||||
|
||||
pub fn name(self: SubClass) []const u8 {
|
||||
return switch (self) {
|
||||
.pic => "PIC",
|
||||
.dma => "DMA Controller",
|
||||
.timer => "Timer",
|
||||
.rtc => "RTC Controller",
|
||||
.pci_hot_plug => "PCI Hot-Plug Controller",
|
||||
.sd_host => "SD Host Controller",
|
||||
.iommu => "IOMMU",
|
||||
};
|
||||
}
|
||||
};
|
||||
};
|
||||
|
||||
pub const input_device = struct {
|
||||
pub const SubClass = enum(u8) {
|
||||
keyboard = 0x00,
|
||||
digitizer_pen = 0x01,
|
||||
mouse = 0x02,
|
||||
scanner = 0x03,
|
||||
gameport = 0x04,
|
||||
|
||||
pub fn name(self: SubClass) []const u8 {
|
||||
return switch (self) {
|
||||
.keyboard => "Keyboard Controller",
|
||||
.digitizer_pen => "Digitizer Pen",
|
||||
.mouse => "Mouse Controller",
|
||||
.scanner => "Scanner Controller",
|
||||
.gameport => "Gameport Controller",
|
||||
};
|
||||
}
|
||||
};
|
||||
};
|
||||
|
||||
pub const serial_bus = struct {
|
||||
pub const SubClass = enum(u8) {
|
||||
firewire = 0x00,
|
||||
access_bus = 0x01,
|
||||
ssa = 0x02,
|
||||
usb = 0x03,
|
||||
fibre_channel = 0x04,
|
||||
smbus = 0x05,
|
||||
infiniband = 0x06,
|
||||
ipmi = 0x07,
|
||||
sercos = 0x08,
|
||||
canbus = 0x09,
|
||||
|
||||
pub fn name(self: SubClass) []const u8 {
|
||||
return switch (self) {
|
||||
.firewire => "FireWire (IEEE 1394) Controller",
|
||||
.access_bus => "ACCESS Bus Controller",
|
||||
.ssa => "SSA",
|
||||
.usb => "USB Controller",
|
||||
.fibre_channel => "Fibre Channel",
|
||||
.smbus => "SMBus Controller",
|
||||
.infiniband => "InfiniBand Controller",
|
||||
.ipmi => "IPMI Interface",
|
||||
.sercos => "SERCOS Interface (IEC 61491)",
|
||||
.canbus => "CANbus Controller",
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
pub const usb = struct {
|
||||
pub const ProgIf = enum(u8) {
|
||||
uhci = 0x00,
|
||||
ohci = 0x10,
|
||||
ehci = 0x20,
|
||||
xhci = 0x30,
|
||||
unspecified = 0x80,
|
||||
device = 0xFE,
|
||||
|
||||
pub fn name(self: ProgIf) []const u8 {
|
||||
return switch (self) {
|
||||
.uhci => "UHCI Controller",
|
||||
.ohci => "OHCI Controller",
|
||||
.ehci => "EHCI (USB2) Controller",
|
||||
.xhci => "XHCI (USB3) Controller",
|
||||
.unspecified => "Unspecified",
|
||||
.device => "USB Device (not a host controller)",
|
||||
};
|
||||
}
|
||||
};
|
||||
};
|
||||
};
|
||||
|
||||
pub const wireless = struct {
|
||||
pub const SubClass = enum(u8) {
|
||||
irda = 0x00,
|
||||
consumer_ir = 0x01,
|
||||
rf = 0x10,
|
||||
bluetooth = 0x11,
|
||||
broadband = 0x12,
|
||||
ethernet_802_1a = 0x20,
|
||||
ethernet_802_1b = 0x21,
|
||||
|
||||
pub fn name(self: SubClass) []const u8 {
|
||||
return switch (self) {
|
||||
.irda => "iRDA Compatible Controller",
|
||||
.consumer_ir => "Consumer IR Controller",
|
||||
.rf => "RF Controller",
|
||||
.bluetooth => "Bluetooth Controller",
|
||||
.broadband => "Broadband Controller",
|
||||
.ethernet_802_1a => "Ethernet Controller (802.1a)",
|
||||
.ethernet_802_1b => "Ethernet Controller (802.1b)",
|
||||
};
|
||||
}
|
||||
};
|
||||
};
|
||||
|
||||
// --- Raw-byte decoding (what a function reports in its header) -------------------------
|
||||
|
||||
/// The name of an exhaustive class-code enum member, or null if `value` is not one — the
|
||||
/// bridge from a raw config byte to a named taxonomy above.
|
||||
fn enumName(comptime Enum: type, value: u8) ?[]const u8 {
|
||||
return (std.enums.fromInt(Enum, value) orelse return null).name();
|
||||
}
|
||||
|
||||
/// Name of the base class (byte 0x0B), e.g. `0x06` -> "Bridge".
|
||||
pub fn className(base: u8) []const u8 {
|
||||
return switch (base) {
|
||||
0x00 => "Unclassified",
|
||||
0x01 => "Mass Storage Controller",
|
||||
0x02 => "Network Controller",
|
||||
0x03 => "Display Controller",
|
||||
0x04 => "Multimedia Controller",
|
||||
0x05 => "Memory Controller",
|
||||
0x06 => "Bridge",
|
||||
0x07 => "Simple Communication Controller",
|
||||
0x08 => "Base System Peripheral",
|
||||
0x09 => "Input Device Controller",
|
||||
0x0A => "Docking Station",
|
||||
0x0B => "Processor",
|
||||
0x0C => "Serial Bus Controller",
|
||||
0x0D => "Wireless Controller",
|
||||
0x0E => "Intelligent Controller",
|
||||
0x0F => "Satellite Communication Controller",
|
||||
0x10 => "Encryption Controller",
|
||||
0x11 => "Signal Processing Controller",
|
||||
0x12 => "Processing Accelerator",
|
||||
0x13 => "Non-Essential Instrumentation",
|
||||
0x40 => "Co-Processor",
|
||||
0xFF => "Unassigned Class (Vendor specific)",
|
||||
else => "Unknown",
|
||||
};
|
||||
return @as(BaseClass, @enumFromInt(base)).name();
|
||||
}
|
||||
|
||||
/// Name of the subclass within its base class, e.g. `(0x06, 0x01)` -> "ISA Bridge".
|
||||
/// Subclass `0x80` is "Other" by PCI convention; anything unlisted is "Unknown".
|
||||
pub fn subclassName(base: u8, subclass: u8) []const u8 {
|
||||
return switch (base) {
|
||||
0x01 => switch (subclass) {
|
||||
0x00 => "SCSI Bus Controller",
|
||||
0x01 => "IDE Controller",
|
||||
0x02 => "Floppy Disk Controller",
|
||||
0x03 => "IPI Bus Controller",
|
||||
0x04 => "RAID Controller",
|
||||
0x05 => "ATA Controller",
|
||||
0x06 => "Serial ATA Controller",
|
||||
0x07 => "Serial Attached SCSI Controller",
|
||||
0x08 => "Non-Volatile Memory Controller",
|
||||
else => defaultSubclass(subclass),
|
||||
},
|
||||
0x02 => switch (subclass) {
|
||||
0x00 => "Ethernet Controller",
|
||||
0x01 => "Token Ring Controller",
|
||||
0x02 => "FDDI Controller",
|
||||
0x03 => "ATM Controller",
|
||||
0x04 => "ISDN Controller",
|
||||
0x06 => "PICMG 2.14 Multi Computing Controller",
|
||||
0x07 => "Infiniband Controller",
|
||||
0x08 => "Fabric Controller",
|
||||
else => defaultSubclass(subclass),
|
||||
},
|
||||
0x03 => switch (subclass) {
|
||||
0x00 => "VGA Compatible Controller",
|
||||
0x01 => "XGA Controller",
|
||||
0x02 => "3D Controller (Not VGA-Compatible)",
|
||||
else => defaultSubclass(subclass),
|
||||
},
|
||||
0x04 => switch (subclass) {
|
||||
0x00 => "Multimedia Video Controller",
|
||||
0x01 => "Multimedia Audio Controller",
|
||||
0x02 => "Computer Telephony Device",
|
||||
0x03 => "Audio Device",
|
||||
else => defaultSubclass(subclass),
|
||||
},
|
||||
0x05 => switch (subclass) {
|
||||
0x00 => "RAM Controller",
|
||||
0x01 => "Flash Controller",
|
||||
else => defaultSubclass(subclass),
|
||||
},
|
||||
0x06 => switch (subclass) {
|
||||
0x00 => "Host Bridge",
|
||||
0x01 => "ISA Bridge",
|
||||
0x02 => "EISA Bridge",
|
||||
0x03 => "MCA Bridge",
|
||||
0x04 => "PCI-to-PCI Bridge",
|
||||
0x05 => "PCMCIA Bridge",
|
||||
0x06 => "NuBus Bridge",
|
||||
0x07 => "CardBus Bridge",
|
||||
0x08 => "RACEway Bridge",
|
||||
0x09 => "PCI-to-PCI Bridge (Semi-Transparent)",
|
||||
0x0A => "InfiniBand-to-PCI Host Bridge",
|
||||
else => defaultSubclass(subclass),
|
||||
},
|
||||
0x07 => switch (subclass) {
|
||||
0x00 => "Serial Controller",
|
||||
0x01 => "Parallel Controller",
|
||||
0x02 => "Multiport Serial Controller",
|
||||
0x03 => "Modem",
|
||||
0x04 => "IEEE 488.1/2 (GPIB) Controller",
|
||||
0x05 => "Smart Card Controller",
|
||||
else => defaultSubclass(subclass),
|
||||
},
|
||||
0x08 => switch (subclass) {
|
||||
0x00 => "PIC",
|
||||
0x01 => "DMA Controller",
|
||||
0x02 => "Timer",
|
||||
0x03 => "RTC Controller",
|
||||
0x04 => "PCI Hot-Plug Controller",
|
||||
0x05 => "SD Host Controller",
|
||||
0x06 => "IOMMU",
|
||||
else => defaultSubclass(subclass),
|
||||
},
|
||||
0x09 => switch (subclass) {
|
||||
0x00 => "Keyboard Controller",
|
||||
0x01 => "Digitizer Pen",
|
||||
0x02 => "Mouse Controller",
|
||||
0x03 => "Scanner Controller",
|
||||
0x04 => "Gameport Controller",
|
||||
else => defaultSubclass(subclass),
|
||||
},
|
||||
0x0C => switch (subclass) {
|
||||
0x00 => "FireWire (IEEE 1394) Controller",
|
||||
0x01 => "ACCESS Bus Controller",
|
||||
0x02 => "SSA",
|
||||
0x03 => "USB Controller",
|
||||
0x04 => "Fibre Channel",
|
||||
0x05 => "SMBus Controller",
|
||||
0x06 => "InfiniBand Controller",
|
||||
0x07 => "IPMI Interface",
|
||||
0x08 => "SERCOS Interface (IEC 61491)",
|
||||
0x09 => "CANbus Controller",
|
||||
else => defaultSubclass(subclass),
|
||||
},
|
||||
0x0D => switch (subclass) {
|
||||
0x00 => "iRDA Compatible Controller",
|
||||
0x01 => "Consumer IR Controller",
|
||||
0x10 => "RF Controller",
|
||||
0x11 => "Bluetooth Controller",
|
||||
0x12 => "Broadband Controller",
|
||||
0x20 => "Ethernet Controller (802.1a)",
|
||||
0x21 => "Ethernet Controller (802.1b)",
|
||||
else => defaultSubclass(subclass),
|
||||
},
|
||||
else => defaultSubclass(subclass),
|
||||
const named: ?[]const u8 = switch (@as(BaseClass, @enumFromInt(base))) {
|
||||
.mass_storage => enumName(mass_storage.SubClass, subclass),
|
||||
.network => enumName(network.SubClass, subclass),
|
||||
.display => enumName(display.SubClass, subclass),
|
||||
.multimedia => enumName(multimedia.SubClass, subclass),
|
||||
.memory => enumName(memory.SubClass, subclass),
|
||||
.bridge => enumName(bridge.SubClass, subclass),
|
||||
.simple_communication => enumName(simple_communication.SubClass, subclass),
|
||||
.base_system_peripheral => enumName(base_system_peripheral.SubClass, subclass),
|
||||
.input_device => enumName(input_device.SubClass, subclass),
|
||||
.serial_bus => enumName(serial_bus.SubClass, subclass),
|
||||
.wireless => enumName(wireless.SubClass, subclass),
|
||||
else => null,
|
||||
};
|
||||
return named orelse defaultSubclass(subclass);
|
||||
}
|
||||
|
||||
fn defaultSubclass(subclass: u8) []const u8 {
|
||||
@@ -175,68 +502,34 @@ fn defaultSubclass(subclass: u8) []const u8 {
|
||||
/// Returns "" when the prog-IF carries no standard meaning for this class/subclass —
|
||||
/// callers just print the hex byte in that case.
|
||||
pub fn progIfName(base: u8, subclass: u8, prog_if: u8) []const u8 {
|
||||
return switch (base) {
|
||||
0x01 => switch (subclass) {
|
||||
0x06 => switch (prog_if) { // Serial ATA
|
||||
0x00 => "Vendor Specific Interface",
|
||||
0x01 => "AHCI 1.0",
|
||||
0x02 => "Serial Storage Bus",
|
||||
else => "",
|
||||
},
|
||||
0x08 => switch (prog_if) { // Non-Volatile Memory
|
||||
0x01 => "NVMHCI",
|
||||
0x02 => "NVM Express",
|
||||
else => "",
|
||||
},
|
||||
else => "",
|
||||
const named: ?[]const u8 = switch (@as(BaseClass, @enumFromInt(base))) {
|
||||
.mass_storage => switch (std.enums.fromInt(mass_storage.SubClass, subclass) orelse return "") {
|
||||
.serial_ata => enumName(mass_storage.serial_ata.ProgIf, prog_if),
|
||||
.non_volatile_memory => enumName(mass_storage.non_volatile_memory.ProgIf, prog_if),
|
||||
else => null,
|
||||
},
|
||||
0x03 => switch (subclass) {
|
||||
0x00 => switch (prog_if) { // VGA Compatible
|
||||
0x00 => "VGA Controller",
|
||||
0x01 => "8514-Compatible Controller",
|
||||
else => "",
|
||||
},
|
||||
else => "",
|
||||
.display => switch (std.enums.fromInt(display.SubClass, subclass) orelse return "") {
|
||||
.vga_compatible => enumName(display.vga_compatible.ProgIf, prog_if),
|
||||
else => null,
|
||||
},
|
||||
0x06 => switch (subclass) {
|
||||
0x04 => switch (prog_if) { // PCI-to-PCI Bridge
|
||||
0x00 => "Normal Decode",
|
||||
0x01 => "Subtractive Decode",
|
||||
else => "",
|
||||
},
|
||||
else => "",
|
||||
.bridge => switch (std.enums.fromInt(bridge.SubClass, subclass) orelse return "") {
|
||||
.pci_to_pci => enumName(bridge.pci_to_pci.ProgIf, prog_if),
|
||||
else => null,
|
||||
},
|
||||
0x07 => switch (subclass) {
|
||||
0x00 => switch (prog_if) { // Serial Controller
|
||||
0x00 => "8250-Compatible (Generic XT)",
|
||||
0x01 => "16450-Compatible",
|
||||
0x02 => "16550-Compatible",
|
||||
0x03 => "16650-Compatible",
|
||||
0x04 => "16750-Compatible",
|
||||
0x05 => "16850-Compatible",
|
||||
0x06 => "16950-Compatible",
|
||||
else => "",
|
||||
},
|
||||
else => "",
|
||||
.simple_communication => switch (std.enums.fromInt(simple_communication.SubClass, subclass) orelse return "") {
|
||||
.serial => enumName(simple_communication.serial.ProgIf, prog_if),
|
||||
else => null,
|
||||
},
|
||||
0x0C => switch (subclass) {
|
||||
0x03 => switch (prog_if) { // USB Controller
|
||||
0x00 => "UHCI Controller",
|
||||
0x10 => "OHCI Controller",
|
||||
0x20 => "EHCI (USB2) Controller",
|
||||
0x30 => "XHCI (USB3) Controller",
|
||||
0x80 => "Unspecified",
|
||||
0xFE => "USB Device (not a host controller)",
|
||||
else => "",
|
||||
},
|
||||
else => "",
|
||||
.serial_bus => switch (std.enums.fromInt(serial_bus.SubClass, subclass) orelse return "") {
|
||||
.usb => enumName(serial_bus.usb.ProgIf, prog_if),
|
||||
else => null,
|
||||
},
|
||||
else => "",
|
||||
else => null,
|
||||
};
|
||||
return named orelse "";
|
||||
}
|
||||
|
||||
test "decodes the common class codes" {
|
||||
const std = @import("std");
|
||||
const eq = std.testing.expectEqualStrings;
|
||||
|
||||
const isa = ClassCode.unpack(0x06_01_00);
|
||||
@@ -251,11 +544,25 @@ test "decodes the common class codes" {
|
||||
try eq("AHCI 1.0", progIfName(ahci.base, ahci.subclass, ahci.prog_if));
|
||||
|
||||
const xhci = ClassCode.unpack(0x0C_03_30);
|
||||
try eq("Serial Bus Controller", className(xhci.base));
|
||||
try eq("USB Controller", subclassName(xhci.base, xhci.subclass));
|
||||
try eq("XHCI (USB3) Controller", progIfName(xhci.base, xhci.subclass, xhci.prog_if));
|
||||
}
|
||||
|
||||
// Unknowns and the "Other" convention.
|
||||
try eq("Other", subclassName(0x02, 0x80));
|
||||
try eq("Unknown", subclassName(0x06, 0x7E));
|
||||
try eq("", progIfName(0x06, 0x00, 0x00)); // host bridge: prog-IF has no standard name
|
||||
test "unlisted codes fall back without a wrong name" {
|
||||
const eq = std.testing.expectEqualStrings;
|
||||
try eq("Unknown", className(0x77)); // no such base class
|
||||
try eq("Other", subclassName(0x01, 0x80)); // 0x80 is the PCI "Other" convention
|
||||
try eq("Unknown", subclassName(0x01, 0x7A)); // unlisted mass-storage subclass
|
||||
try eq("", progIfName(0x01, 0x06, 0x7F)); // no standard SATA prog-IF for 0x7F
|
||||
try eq("", progIfName(0x02, 0x00, 0x00)); // class with no prog-IF taxonomy at all
|
||||
}
|
||||
|
||||
test "named parts pack to the raw triple" {
|
||||
const xhci = ClassCode{
|
||||
.base = @intFromEnum(BaseClass.serial_bus),
|
||||
.subclass = @intFromEnum(serial_bus.SubClass.usb),
|
||||
.prog_if = @intFromEnum(serial_bus.usb.ProgIf.xhci),
|
||||
};
|
||||
try std.testing.expectEqual(@as(u24, 0x0C_03_30), xhci.pack());
|
||||
}
|
||||
|
||||
@@ -40,6 +40,13 @@ pub fn platformInformation() PlatformInformation {
|
||||
}
|
||||
|
||||
/// AML parse integrity/diagnostics (namespace node count, bytes consumed).
|
||||
/// The number of Device objects in the kernel's own AML namespace, or 0 if the
|
||||
/// parse produced none — the `acpi-parse` test compares the ring-3 service's
|
||||
/// count against this.
|
||||
pub fn amlDeviceCount() usize {
|
||||
return acpi.amlDeviceCount();
|
||||
}
|
||||
|
||||
pub fn amlStats() AmlStats {
|
||||
return acpi.aml_stats;
|
||||
}
|
||||
@@ -72,7 +79,8 @@ pub fn discover(
|
||||
var device_tree = try DeviceTree.init(allocator);
|
||||
|
||||
if (boot_information.acpi_rsdp != 0) {
|
||||
try acpi.discover(boot_information.acpi_rsdp, &device_tree, hal);
|
||||
const memory_regions = @as([*]const boot_handoff.MemoryRegion, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.memory_map.regions)))[0..boot_information.memory_map.len];
|
||||
try acpi.discover(boot_information.acpi_rsdp, memory_regions, &device_tree, hal);
|
||||
} else {
|
||||
// No ACPI RSDP. A device-tree boot would parse its blob here; today that
|
||||
// path is a stub, so this reports the machine described itself no way we
|
||||
|
||||
@@ -1,211 +0,0 @@
|
||||
//! /system/drivers/bus — a user-space **bus driver**, and the smallest honest example of one.
|
||||
//!
|
||||
//! A bus driver owns a device that *contains other devices*, enumerates them by some
|
||||
//! bus-specific protocol, and publishes each one into the kernel's device table so a
|
||||
//! class driver can claim it. PCI walks configuration space; USB walks hub descriptors. Here
|
||||
//! the "bus" is the HPET's register block and the "devices" are its comparators, each
|
||||
//! a 0x20-byte window at 0x100 + 0x20*n that can be driven independently.
|
||||
//!
|
||||
//! It's a toy bus, but nothing about the mechanism is: `bus` reads how many children
|
||||
//! exist from the hardware (GENERAL_CAP bits [12:8]), publishes one `DeviceDescriptor` per
|
||||
//! child with a sub-window of its own MMIO plus the shared IRQ, and the kernel checks
|
||||
//! every one of those resources is contained in what `bus` was granted. A comparator
|
||||
//! driver then claims a child and maps only *its* registers — not the whole block.
|
||||
//!
|
||||
//! It also proves the negative: registering a child whose window escapes the parent's
|
||||
//! is refused. Without that check, `device_register` would be a system_call for mapping
|
||||
//! arbitrary physical memory.
|
||||
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
const device = runtime.device;
|
||||
|
||||
const register_general_cap = 0x000;
|
||||
|
||||
/// Comparator n's registers: configuration+comparator+FSB route, 0x20 bytes.
|
||||
fn timerWindow(hpet_base: u64, n: u64) device.ResourceDescriptor {
|
||||
return .{
|
||||
.kind = @intFromEnum(device.ResourceKind.memory),
|
||||
.start = hpet_base + 0x100 + 0x20 * n,
|
||||
.len = 0x20,
|
||||
};
|
||||
}
|
||||
|
||||
fn findHpet(buffer: []device.DeviceDescriptor) ?device.DeviceDescriptor {
|
||||
const total = device.enumerate(buffer);
|
||||
const n = @min(total, buffer.len);
|
||||
for (buffer[0..n]) |d| {
|
||||
if (d.class != @intFromEnum(device.DeviceClass.timer)) continue;
|
||||
if (d.parent != device.no_parent) continue; // the block, not a comparator child
|
||||
for (0..d.resource_count) |j| {
|
||||
if (d.resources[j].kind == @intFromEnum(device.ResourceKind.memory)) return d;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// The parent's MMIO resource, and its IRQ if it has one.
|
||||
fn resourcesOf(d: device.DeviceDescriptor) struct { mmio: device.ResourceDescriptor, irq: ?device.ResourceDescriptor } {
|
||||
var mmio: device.ResourceDescriptor = undefined;
|
||||
var irq: ?device.ResourceDescriptor = null;
|
||||
for (0..d.resource_count) |j| {
|
||||
const r = d.resources[j];
|
||||
if (r.kind == @intFromEnum(device.ResourceKind.memory)) mmio = r;
|
||||
if (r.kind == @intFromEnum(device.ResourceKind.irq)) irq = r;
|
||||
}
|
||||
return .{ .mmio = mmio, .irq = irq };
|
||||
}
|
||||
|
||||
fn firstChildOf(buffer: []device.DeviceDescriptor, total: usize, parent_id: u64) ?u64 {
|
||||
for (buffer[0..@min(total, buffer.len)]) |d| {
|
||||
if (d.parent == parent_id) return d.id;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
pub fn main() void {
|
||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||
_ = runtime.system.write("bus: out of memory\n");
|
||||
return;
|
||||
};
|
||||
|
||||
const parent = findHpet(buffer) orelse {
|
||||
_ = runtime.system.write("bus: no HPET\n");
|
||||
return;
|
||||
};
|
||||
const resource = resourcesOf(parent);
|
||||
|
||||
// Claim the bus. Everything below is subdivision of what this claim granted.
|
||||
//
|
||||
// Claims are exclusive, and at a normal boot the kernel spawns every initial_ramdisk
|
||||
// binary — so hpet may own the HPET already. That's not an error, it's the
|
||||
// capability model working: exit quietly and leave the device to its owner. The
|
||||
// `bus` test spawns bus alone, so there it wins the claim.
|
||||
if (!device.claim(parent.id)) {
|
||||
_ = runtime.system.write("bus: HPET already claimed by another driver, nothing to do\n");
|
||||
return;
|
||||
}
|
||||
|
||||
// Enumerate the bus: ask the hardware how many children it has.
|
||||
const base = device.mmioMap(parent.id, 0) orelse {
|
||||
_ = runtime.system.write("bus: mmio_map failed\n");
|
||||
return;
|
||||
};
|
||||
const cap: *volatile u64 = @ptrFromInt(base + register_general_cap);
|
||||
const n_children = ((cap.* >> 8) & 0x1F) + 1;
|
||||
|
||||
// Publish one child per comparator, each owning only its own window.
|
||||
var published: u64 = 0;
|
||||
var n: u64 = 0;
|
||||
while (n < n_children) : (n += 1) {
|
||||
var child = std.mem.zeroes(device.DeviceDescriptor);
|
||||
child.class = @intFromEnum(device.DeviceClass.timer);
|
||||
child.pci_class = device.no_pci_class;
|
||||
child.hid_len = 6;
|
||||
child.hid[0..6].* = "hpet-t".*;
|
||||
child.resource_count = 1;
|
||||
child.resources[0] = timerWindow(resource.mmio.start, n);
|
||||
// Comparators share the block's interrupt line; only one child can bind it,
|
||||
// but all of them may legitimately name it.
|
||||
if (resource.irq) |i| {
|
||||
child.resources[child.resource_count] = i;
|
||||
child.resource_count += 1;
|
||||
}
|
||||
|
||||
if (device.register(parent.id, &child) == null) {
|
||||
_ = runtime.system.write("bus: register failed\n");
|
||||
return;
|
||||
}
|
||||
published += 1;
|
||||
}
|
||||
|
||||
// The negative case. A window one byte past the end of the parent's must be
|
||||
// refused — otherwise device_register would be "map any physical page you like".
|
||||
// Confirm the table did not grow, not merely that the call returned null: null
|
||||
// also means NoSpace/BadParent, so a size check is what actually proves the
|
||||
// *containment* rule fired.
|
||||
const before = device.enumerate(buffer);
|
||||
var rogue = std.mem.zeroes(device.DeviceDescriptor);
|
||||
rogue.class = @intFromEnum(device.DeviceClass.unknown);
|
||||
rogue.resource_count = 1;
|
||||
rogue.resources[0] = .{
|
||||
.kind = @intFromEnum(device.ResourceKind.memory),
|
||||
.start = resource.mmio.start + resource.mmio.len,
|
||||
.len = 0x1000,
|
||||
};
|
||||
if (device.register(parent.id, &rogue) != null) {
|
||||
_ = runtime.system.write("bus: FAIL out-of-window child was accepted\n");
|
||||
return;
|
||||
}
|
||||
if (device.enumerate(buffer) != before) {
|
||||
_ = runtime.system.write("bus: FAIL rogue child leaked into the table\n");
|
||||
return;
|
||||
}
|
||||
|
||||
// And confirm the children came back with the right parent and a *narrower*
|
||||
// window than the bus — read from the table, not from our own memory.
|
||||
const total = device.enumerate(buffer);
|
||||
var seen: u64 = 0;
|
||||
for (buffer[0..@min(total, buffer.len)]) |d| {
|
||||
if (d.parent != parent.id) continue;
|
||||
const w = d.resources[0];
|
||||
if (w.start < resource.mmio.start or w.len >= resource.mmio.len) {
|
||||
_ = runtime.system.write("bus: FAIL child window is not inside the bus\n");
|
||||
return;
|
||||
}
|
||||
seen += 1;
|
||||
}
|
||||
if (seen != published) {
|
||||
_ = runtime.system.write("bus: FAIL child count mismatch\n");
|
||||
return;
|
||||
}
|
||||
|
||||
// Delegation, end to end: claim a child and map *it*. A real class driver would be
|
||||
// a different process; here bus plays both parts, which exercises the same path.
|
||||
// The child's window is 0x20 bytes at parent+0x100, so the register it sees at
|
||||
// offset 0 must be the same timer-0 configuration register the bus sees at 0x100.
|
||||
//
|
||||
// (mmio_map rounds to a page, so the child's mapping physically covers the whole
|
||||
// 4 KiB the HPET lives in — the granularity limit documented in docs/drivers.md.
|
||||
// The *resource* is narrow even though the page isn't.)
|
||||
const child_id = firstChildOf(buffer, device.enumerate(buffer), parent.id) orelse {
|
||||
_ = runtime.system.write("bus: FAIL no child to claim\n");
|
||||
return;
|
||||
};
|
||||
if (!device.claim(child_id)) {
|
||||
_ = runtime.system.write("bus: FAIL could not claim own child\n");
|
||||
return;
|
||||
}
|
||||
const child_base = device.mmioMap(child_id, 0) orelse {
|
||||
_ = runtime.system.write("bus: FAIL child mmio_map refused\n");
|
||||
return;
|
||||
};
|
||||
const via_child: *volatile u64 = @ptrFromInt(child_base);
|
||||
const via_bus: *volatile u64 = @ptrFromInt(base + 0x100);
|
||||
if (via_child.* != via_bus.*) {
|
||||
_ = runtime.system.write("bus: FAIL child window does not alias the bus register\n");
|
||||
return;
|
||||
}
|
||||
|
||||
// A descriptor pointer into an unmapped page must fail the call, not fault the
|
||||
// kernel. Grab a page, free it, and register through the stale address: if the
|
||||
// kernel dereferenced it raw (rather than copying in through the page tables) this
|
||||
// would triple-fault QEMU and the test would time out instead of printing ok.
|
||||
const scratch = runtime.system.mmap(0x1000, runtime.system.PROT_READ | runtime.system.PROT_WRITE);
|
||||
if (!runtime.system.mmapFailed(scratch)) {
|
||||
_ = runtime.system.munmap(scratch, 0x1000);
|
||||
const descriptor: *const device.DeviceDescriptor = @ptrFromInt(scratch);
|
||||
if (device.register(parent.id, descriptor) != null) {
|
||||
_ = runtime.system.write("bus: FAIL register accepted an unmapped descriptor\n");
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
_ = runtime.system.write("bus: ok\n");
|
||||
while (true) runtime.system.sleep(1000);
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start;
|
||||
}
|
||||
@@ -1,195 +0,0 @@
|
||||
//! /system/drivers/hpet — a user-space HPET driver. It proves the whole driver model end to
|
||||
//! end: enumerate the device table, find the HPET, claim it, map its registers into
|
||||
//! this ring-3 address space (strong-uncacheable), **bind its interrupt to an IPC
|
||||
//! endpoint**, then sit blocked in `replyWait` until the hardware wakes it.
|
||||
//!
|
||||
//! Nothing here polls. Between interrupts the process is `.blocked` and off every
|
||||
//! scheduler queue; the core runs other work or idles. That is the point of the
|
||||
//! exercise — a driver is a process that sleeps until its device has something to
|
||||
//! say (see docs/drivers.md).
|
||||
//!
|
||||
//! The comparator is configured **level-triggered** on purpose. Edge would be
|
||||
//! simpler, but level is the discipline every real device line needs, and it forces
|
||||
//! the full cycle to be correct:
|
||||
//!
|
||||
//! kernel ISR mask the GSI -> EOI -> notify this endpoint
|
||||
//! hpet wake, clear GENERAL_INT_STATUS (deasserts the line), re-arm
|
||||
//! hpet irq_ack -> kernel unmasks the GSI
|
||||
//!
|
||||
//! Clear the status bit *before* acking, or the line is still asserted when the
|
||||
//! kernel unmasks and the I/O APIC redelivers forever.
|
||||
//!
|
||||
//! Register map (HPET spec 1.0a):
|
||||
//! 0x000 GENERAL_CAP [63:32] fs per tick, [12:8] number timers - 1
|
||||
//! 0x010 GENERAL_CONFIGURATION bit0 ENABLE_CNF, bit1 LEG_RT_CNF
|
||||
//! 0x020 GENERAL_INT_STATUS bit n = timer n asserted (write 1 to clear)
|
||||
//! 0x0F0 MAIN_COUNTER
|
||||
//! 0x100 TIMER0_CONFIGURATION bit1 INT_TYPE(1=level) bit2 INT_ENB bit3 TYPE(periodic)
|
||||
//! bits[13:9] INT_ROUTE, [63:32] INT_ROUTE_CAP
|
||||
//! 0x108 TIMER0_COMPARATOR
|
||||
|
||||
const runtime = @import("runtime");
|
||||
const mmio = @import("mmio");
|
||||
const device = runtime.device;
|
||||
const ipc = runtime.ipc;
|
||||
|
||||
const register_general_cap = 0x000;
|
||||
const register_general_configuration = 0x010;
|
||||
const register_int_status = 0x020;
|
||||
const register_main_counter = 0x0F0;
|
||||
const register_timer0_configuration = 0x100;
|
||||
const register_timer0_comparator = 0x108;
|
||||
|
||||
const configuration_enable: u64 = 1 << 0; // GENERAL_CONFIGURATION.ENABLE_CNF
|
||||
const configuration_leg_rt: u64 = 1 << 1; // GENERAL_CONFIGURATION.LEG_RT_CNF
|
||||
const tn_int_type_level: u64 = 1 << 1;
|
||||
const tn_int_enb: u64 = 1 << 2;
|
||||
const tn_type_periodic: u64 = 1 << 3;
|
||||
const tn_route_shift = 9;
|
||||
const tn_route_mask: u64 = 0x1F << tn_route_shift;
|
||||
|
||||
/// Interrupts to observe before declaring victory.
|
||||
const target_ticks = 5;
|
||||
|
||||
/// Read/write a 64-bit HPET register through the typed volatile MMIO layer (/lib/mmio).
|
||||
/// The HPET is pure MMIO with no DMA, and on x86 its grant is strong-uncacheable (so
|
||||
/// UC writes are already ordered) — no barriers are needed here; the point is the
|
||||
/// typed, arch-portable access every driver should use.
|
||||
inline fn rd(base: usize, off: usize) u64 {
|
||||
return mmio.read(u64, base + off);
|
||||
}
|
||||
inline fn wr(base: usize, off: usize, value: u64) void {
|
||||
mmio.write(u64, base + off, value);
|
||||
}
|
||||
|
||||
/// A timer-class device exposing both an MMIO window and an IRQ: its id, the two
|
||||
/// resource indices, and the GSI discovery chose out of `Tn_INT_ROUTE_CAP`.
|
||||
const Found = struct { device_id: u64, mmio: u64, irq: u64, gsi: u64 };
|
||||
|
||||
fn findHpet(buffer: []device.DeviceDescriptor) ?Found {
|
||||
const total = device.enumerate(buffer);
|
||||
const n = @min(total, buffer.len);
|
||||
for (buffer[0..n]) |d| {
|
||||
if (d.class != @intFromEnum(device.DeviceClass.timer)) continue;
|
||||
// Skip comparator children a bus driver may have published below the block
|
||||
// (see system/drivers/bus/bus.zig) — we want the register block itself.
|
||||
if (d.parent != device.no_parent) continue;
|
||||
var mmio_index: ?u64 = null;
|
||||
var irq: ?u64 = null;
|
||||
for (0..d.resource_count) |j| {
|
||||
switch (d.resources[j].kind) {
|
||||
@intFromEnum(device.ResourceKind.memory) => mmio_index = mmio_index orelse j,
|
||||
@intFromEnum(device.ResourceKind.irq) => irq = irq orelse j,
|
||||
else => {},
|
||||
}
|
||||
}
|
||||
if (mmio_index) |m| if (irq) |i| {
|
||||
return .{ .device_id = d.id, .mmio = m, .irq = i, .gsi = d.resources[i].start };
|
||||
};
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
pub fn main() void {
|
||||
// Enumerate into a heap buffer (too big for the one-page user stack).
|
||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 32) catch {
|
||||
_ = runtime.system.write("hpet: out of memory\n");
|
||||
return;
|
||||
};
|
||||
|
||||
const hpet = findHpet(buffer) orelse {
|
||||
_ = runtime.system.write("hpet: no HPET with an IRQ\n");
|
||||
return;
|
||||
};
|
||||
|
||||
if (!device.claim(hpet.device_id)) {
|
||||
_ = runtime.system.write("hpet: claim failed\n");
|
||||
return;
|
||||
}
|
||||
const base = device.mmioMap(hpet.device_id, hpet.mmio) orelse {
|
||||
_ = runtime.system.write("hpet: mmio_map failed\n");
|
||||
return;
|
||||
};
|
||||
|
||||
// The GSI discovery picked for us out of Tn_INT_ROUTE_CAP. Program the comparator
|
||||
// to raise exactly this line — the kernel will only bind the one it recorded.
|
||||
const gsi = hpet.gsi;
|
||||
|
||||
const endpoint = ipc.createIpcEndpoint() orelse {
|
||||
_ = runtime.system.write("hpet: create_ipc_endpoint failed\n");
|
||||
return;
|
||||
};
|
||||
|
||||
// --- program the hardware ------------------------------------------------
|
||||
// Counter period, so we can arm the comparator a fixed wall-clock distance out.
|
||||
const femtos_per_tick = rd(base, register_general_cap) >> 32;
|
||||
if (femtos_per_tick == 0) {
|
||||
_ = runtime.system.write("hpet: bad HPET period\n");
|
||||
return;
|
||||
}
|
||||
const ticks_per_ms = 1_000_000_000_000 / femtos_per_tick;
|
||||
|
||||
// Stop the counter and take the legacy route off while we reconfigure.
|
||||
wr(base, register_general_configuration, rd(base, register_general_configuration) & ~(configuration_enable | configuration_leg_rt));
|
||||
|
||||
// Timer 0: one-shot, level-triggered, routed to our GSI, interrupt enabled.
|
||||
// One-shot (not periodic) sidesteps the HPET's Tn_value_SET accumulator quirk —
|
||||
// we simply re-arm from the driver on each interrupt, which is what a tickless
|
||||
// timer driver does anyway.
|
||||
var t0 = rd(base, register_timer0_configuration);
|
||||
t0 &= ~(tn_route_mask | tn_type_periodic);
|
||||
t0 |= tn_int_type_level | tn_int_enb | (gsi << tn_route_shift);
|
||||
wr(base, register_timer0_configuration, t0);
|
||||
|
||||
// Clear any stale assertion, then arm ~100 ms out and start the counter.
|
||||
wr(base, register_int_status, 1);
|
||||
wr(base, register_timer0_comparator, rd(base, register_main_counter) + ticks_per_ms * 100);
|
||||
wr(base, register_general_configuration, rd(base, register_general_configuration) | configuration_enable);
|
||||
|
||||
if (!device.irqBind(hpet.device_id, hpet.irq, endpoint)) {
|
||||
_ = runtime.system.write("hpet: irq_bind failed\n");
|
||||
return;
|
||||
}
|
||||
_ = runtime.system.write("hpet: bound, sleeping until the hardware speaks\n");
|
||||
|
||||
// --- the driver loop -----------------------------------------------------
|
||||
// Blocked in replyWait. No polling, no spinning: the next line of this function
|
||||
// runs only because an interrupt fired.
|
||||
var receive: [64]u8 = undefined;
|
||||
var count: usize = 0;
|
||||
while (count < target_ticks) {
|
||||
// Blocked here. The task is `.blocked` and off every scheduler queue; the
|
||||
// next line runs only because the HPET raised its line.
|
||||
const r = ipc.replyWait(endpoint, &.{}, &receive, null);
|
||||
if (!r.isNotification()) continue; // a client request, not our IRQ
|
||||
|
||||
// Quiet the device: write 1 to timer 0's status bit. Until this lands, the
|
||||
// line is still asserted and unmasking would refire immediately.
|
||||
wr(base, register_int_status, 1);
|
||||
count += 1;
|
||||
|
||||
if (count < target_ticks) {
|
||||
wr(base, register_timer0_comparator, rd(base, register_main_counter) + ticks_per_ms * 100);
|
||||
} else {
|
||||
// Last one: stop the source rather than re-arming, so the line is left
|
||||
// both quiet *and* unmasked by the ack below. Re-arming here would leave
|
||||
// a pending interrupt that nobody is waiting for, and the ISR would mask
|
||||
// the line again a moment later.
|
||||
wr(base, register_timer0_configuration, rd(base, register_timer0_configuration) & ~tn_int_enb);
|
||||
}
|
||||
|
||||
_ = runtime.system.write("hpet: irq\n");
|
||||
if (!device.irqAck(hpet.device_id, hpet.irq)) {
|
||||
_ = runtime.system.write("hpet: irq_ack failed\n");
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
_ = runtime.system.write("hpet: ok\n");
|
||||
while (true) runtime.system.sleep(1000);
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start;
|
||||
}
|
||||
@@ -0,0 +1,270 @@
|
||||
//! /system/drivers/pci-bus — the PCI bus driver: enumeration moved out of ring 0
|
||||
//! (docs/discovery.md). The device manager matches the `pci_host_bridge`
|
||||
//! node and spawns one instance per bridge, the bridge's device id as argv[1] —
|
||||
//! the same per-device contract as usb-xhci-bus.
|
||||
//!
|
||||
//! M19.1 (this increment): claim the bridge, map its ECAM window (resource 0;
|
||||
//! the bus range and the MMIO apertures follow it), walk every
|
||||
//! bus/device/function config header, and log what the walk finds — ending
|
||||
//! with "/system/drivers/pci-bus: N functions found", which the `pci-scan` scenario compares
|
||||
//! against the kernel's own enumeration. Registration and reports (M19.2), and
|
||||
//! the kernel walk's retirement (M19.3), build on this proven-equivalent scan.
|
||||
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
const protocol = runtime.device_manager_protocol;
|
||||
const device = runtime.device;
|
||||
const pci_class = @import("pci-class");
|
||||
|
||||
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
var line: [128]u8 = undefined;
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
/// Log a discovered function with its (class / subclass / prog-IF) triple decoded
|
||||
/// to human names — the boot-log breadcrumb that says *what* the hardware is, so
|
||||
/// "class 0x01 (Mass Storage Controller) subclass 0x06 (Serial ATA Controller)
|
||||
/// progif 0x01 (AHCI 1.0)" reads straight off the log when writing a new driver.
|
||||
/// A dedicated wider buffer than `writeLine`'s, since the decoded names are long.
|
||||
fn logFunction(bus: u64, dev: u64, function: u64, class_triple: u32) void {
|
||||
const cc = pci_class.ClassCode.unpack(@truncate(class_triple));
|
||||
const pif = pci_class.progIfName(cc.base, cc.subclass, cc.prog_if);
|
||||
var line: [200]u8 = undefined;
|
||||
const text = if (pif.len != 0)
|
||||
std.fmt.bufPrint(&line, "/system/drivers/pci-bus: {d}:{d}.{d} class 0x{x:0>2} ({s}) subclass 0x{x:0>2} ({s}) progif 0x{x:0>2} ({s})\n", .{ bus, dev, function, cc.base, pci_class.className(cc.base), cc.subclass, pci_class.subclassName(cc.base, cc.subclass), cc.prog_if, pif }) catch return
|
||||
else
|
||||
std.fmt.bufPrint(&line, "/system/drivers/pci-bus: {d}:{d}.{d} class 0x{x:0>2} ({s}) subclass 0x{x:0>2} ({s}) progif 0x{x:0>2}\n", .{ bus, dev, function, cc.base, pci_class.className(cc.base), cc.subclass, pci_class.subclassName(cc.base, cc.subclass), cc.prog_if }) catch return;
|
||||
_ = runtime.system.write(text);
|
||||
}
|
||||
|
||||
var bridge_id: u64 = protocol.no_device;
|
||||
var ecam_base: usize = 0;
|
||||
var ecam_physical: u64 = 0;
|
||||
var start_bus: u64 = 0;
|
||||
var bus_count: u64 = 0;
|
||||
var manager_handle: runtime.ipc.Handle = 0;
|
||||
|
||||
/// One aligned 32-bit read from a function's configuration space.
|
||||
fn configRead(bus: u64, dev: u64, function: u64, offset: u64) u32 {
|
||||
const address = ecam_base + (((bus - start_bus) << 20) | (dev << 15) | (function << 12) | offset);
|
||||
const register: *volatile u32 = @ptrFromInt(address);
|
||||
return register.*;
|
||||
}
|
||||
|
||||
fn configWrite(bus: u64, dev: u64, function: u64, offset: u64, value: u32) void {
|
||||
const address = ecam_base + (((bus - start_bus) << 20) | (dev << 15) | (function << 12) | offset);
|
||||
const register: *volatile u32 = @ptrFromInt(address);
|
||||
register.* = value;
|
||||
}
|
||||
|
||||
fn configRead16(bus: u64, dev: u64, function: u64, offset: u64) u16 {
|
||||
const word = configRead(bus, dev, function, offset & ~@as(u64, 3));
|
||||
return @truncate(word >> @intCast((offset & 3) * 8));
|
||||
}
|
||||
|
||||
fn configWrite16(bus: u64, dev: u64, function: u64, offset: u64, value: u16) void {
|
||||
const aligned = offset & ~@as(u64, 3);
|
||||
const shift: u5 = @intCast((offset & 3) * 8);
|
||||
const word = configRead(bus, dev, function, aligned);
|
||||
const mask = @as(u32, 0xFFFF) << shift;
|
||||
configWrite(bus, dev, function, aligned, (word & ~mask) | (@as(u32, value) << shift));
|
||||
}
|
||||
|
||||
/// Claim the bridge, map the ECAM, hello the manager, then scan.
|
||||
fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||
_ = endpoint;
|
||||
if (!device.claim(bridge_id)) {
|
||||
writeLine("/system/drivers/pci-bus: unable to claim bridge device {d}\n", .{bridge_id});
|
||||
return false;
|
||||
}
|
||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||
_ = runtime.system.write("/system/drivers/pci-bus: out of memory\n");
|
||||
return false;
|
||||
};
|
||||
const total = device.enumerate(buffer);
|
||||
const descriptor = for (buffer[0..@min(total, buffer.len)]) |d| {
|
||||
if (d.id == bridge_id) break d;
|
||||
} else {
|
||||
writeLine("/system/drivers/pci-bus: device {d} not in the device tree\n", .{bridge_id});
|
||||
return false;
|
||||
};
|
||||
// Resource 0 is the ECAM window (1 MiB of config space per bus); the bus
|
||||
// range rides beside it. The MMIO apertures (M19.0) come after both.
|
||||
if (descriptor.resource_count < 2 or descriptor.resources[0].kind != @intFromEnum(device.ResourceKind.memory)) {
|
||||
_ = runtime.system.write("/system/drivers/pci-bus: bridge has no ECAM window\n");
|
||||
return false;
|
||||
}
|
||||
const bus_range = for (descriptor.resources[0..@intCast(descriptor.resource_count)]) |resource| {
|
||||
if (resource.kind == @intFromEnum(device.ResourceKind.bus_range)) break resource;
|
||||
} else {
|
||||
_ = runtime.system.write("/system/drivers/pci-bus: bridge has no bus range\n");
|
||||
return false;
|
||||
};
|
||||
start_bus = bus_range.start;
|
||||
bus_count = bus_range.len;
|
||||
ecam_physical = descriptor.resources[0].start;
|
||||
ecam_base = device.mmioMap(bridge_id, 0) orelse {
|
||||
_ = runtime.system.write("/system/drivers/pci-bus: ECAM mmio_map failed\n");
|
||||
return false;
|
||||
};
|
||||
|
||||
// The handshake, then the scan (reports join in M19.2).
|
||||
var manager: ?runtime.ipc.Handle = null;
|
||||
var tries: u32 = 0;
|
||||
while (manager == null and tries < 100) : (tries += 1) {
|
||||
manager = runtime.ipc.lookup(.device_manager);
|
||||
if (manager == null) runtime.system.sleep(20);
|
||||
}
|
||||
const h = manager orelse {
|
||||
_ = runtime.system.write("/system/drivers/pci-bus: no device manager to hello\n");
|
||||
return false;
|
||||
};
|
||||
const hello = protocol.Hello{ .role = @intFromEnum(protocol.Role.bus), .device_id = bridge_id };
|
||||
var reply: [protocol.message_maximum]u8 = undefined;
|
||||
const n = runtime.ipc.call(h, std.mem.asBytes(&hello), &reply) catch {
|
||||
_ = runtime.system.write("/system/drivers/pci-bus: hello call failed\n");
|
||||
return false;
|
||||
};
|
||||
if (n < protocol.reply_size or std.mem.bytesToValue(protocol.HelloReply, reply[0..protocol.reply_size]).status != 0) {
|
||||
_ = runtime.system.write("/system/drivers/pci-bus: hello refused\n");
|
||||
return false;
|
||||
}
|
||||
manager_handle = h;
|
||||
|
||||
scan();
|
||||
return true;
|
||||
}
|
||||
|
||||
/// The brute-force walk the kernel does today, from ring 3: every bus in the
|
||||
/// range, 32 devices, 8 functions; vendor id FFFFh means nothing decodes there,
|
||||
/// and only multifunction devices get their functions 1..7 probed.
|
||||
fn scan() void {
|
||||
var found: u32 = 0;
|
||||
var bus: u64 = start_bus;
|
||||
while (bus < start_bus + bus_count) : (bus += 1) {
|
||||
var dev: u64 = 0;
|
||||
while (dev < 32) : (dev += 1) {
|
||||
const first = configRead(bus, dev, 0, 0);
|
||||
if (first & 0xFFFF == 0xFFFF) continue;
|
||||
const multifunction = (configRead(bus, dev, 0, 0x0C) >> 16) & 0x80 != 0;
|
||||
var function: u64 = 0;
|
||||
while (function < 8) : (function += 1) {
|
||||
if (function != 0 and !multifunction) break;
|
||||
const vendor_device = configRead(bus, dev, function, 0);
|
||||
if (vendor_device & 0xFFFF == 0xFFFF) continue;
|
||||
const class_revision = configRead(bus, dev, function, 0x08);
|
||||
found += 1;
|
||||
logFunction(bus, dev, function, class_revision >> 8);
|
||||
registerAndReport(bus, dev, function, class_revision >> 8);
|
||||
}
|
||||
}
|
||||
}
|
||||
writeLine("/system/drivers/pci-bus: {d} functions found\n", .{found});
|
||||
}
|
||||
|
||||
/// Register one function under the bridge and report it to the manager. The
|
||||
/// descriptor mirrors the kernel's own recording byte for byte — config slice
|
||||
/// as resource 0, then the sized BARs — so during coexistence the idempotent
|
||||
/// device_register (M19.0) returns the kernel's existing node id rather than
|
||||
/// growing a duplicate, and the report carries the id drivers already use.
|
||||
fn registerAndReport(bus: u64, dev: u64, function: u64, class_triple: u32) void {
|
||||
var descriptor = std.mem.zeroes(device.DeviceDescriptor);
|
||||
descriptor.class = @intFromEnum(device.DeviceClass.pci_device);
|
||||
descriptor.pci_class = class_triple;
|
||||
descriptor.resources[0] = .{
|
||||
.kind = @intFromEnum(device.ResourceKind.memory),
|
||||
.start = ecam_physical + (((bus - start_bus) << 20) | (dev << 15) | (function << 12)),
|
||||
.len = 4096,
|
||||
};
|
||||
descriptor.resource_count = 1;
|
||||
|
||||
// The standard BAR-sizing probe, exactly as the kernel does it: decode off,
|
||||
// write all-ones, read the writable mask back, restore. Header type 0 only.
|
||||
const header_type = (configRead(bus, dev, function, 0x0C) >> 16) & 0x7F;
|
||||
if (header_type == 0) {
|
||||
const command = configRead16(bus, dev, function, 0x04);
|
||||
configWrite16(bus, dev, function, 0x04, command & ~@as(u16, 0b11));
|
||||
var i: u64 = 0;
|
||||
while (i < 6) : (i += 1) {
|
||||
if (descriptor.resource_count >= 8) break;
|
||||
const off = 0x10 + i * 4;
|
||||
const original = configRead(bus, dev, function, off);
|
||||
if (original == 0) continue;
|
||||
const slot: usize = @intCast(descriptor.resource_count);
|
||||
if (original & 1 != 0) {
|
||||
configWrite(bus, dev, function, off, 0xFFFF_FFFF);
|
||||
const readback = configRead(bus, dev, function, off);
|
||||
configWrite(bus, dev, function, off, original);
|
||||
const mask = readback & 0xFFFF_FFFC;
|
||||
const size: u32 = if (mask == 0) 0 else (~mask +% 1) & 0xFFFF;
|
||||
if (size == 0) continue; // unimplemented BAR — nothing to register
|
||||
descriptor.resources[slot] = .{ .kind = @intFromEnum(device.ResourceKind.io_port), .start = original & 0xFFFF_FFFC, .len = size };
|
||||
descriptor.resource_count += 1;
|
||||
} else if ((original >> 1) & 0x3 == 2) {
|
||||
const original_high = configRead(bus, dev, function, off + 4);
|
||||
configWrite(bus, dev, function, off, 0xFFFF_FFFF);
|
||||
configWrite(bus, dev, function, off + 4, 0xFFFF_FFFF);
|
||||
const lo = configRead(bus, dev, function, off);
|
||||
const hi = configRead(bus, dev, function, off + 4);
|
||||
configWrite(bus, dev, function, off, original);
|
||||
configWrite(bus, dev, function, off + 4, original_high);
|
||||
const readback = (@as(u64, hi) << 32) | (lo & 0xFFFF_FFF0);
|
||||
const size: u64 = if (readback == 0) 0 else ~readback +% 1;
|
||||
i += 1; // consumed the high half regardless
|
||||
if (size == 0) continue;
|
||||
descriptor.resources[slot] = .{ .kind = @intFromEnum(device.ResourceKind.memory), .start = (@as(u64, original_high) << 32) | (original & 0xFFFF_FFF0), .len = size };
|
||||
descriptor.resource_count += 1;
|
||||
} else {
|
||||
configWrite(bus, dev, function, off, 0xFFFF_FFFF);
|
||||
const readback = configRead(bus, dev, function, off);
|
||||
configWrite(bus, dev, function, off, original);
|
||||
const mask = readback & 0xFFFF_FFF0;
|
||||
const size: u32 = if (mask == 0) 0 else ~mask +% 1;
|
||||
if (size == 0) continue;
|
||||
descriptor.resources[slot] = .{ .kind = @intFromEnum(device.ResourceKind.memory), .start = original & 0xFFFF_FFF0, .len = size };
|
||||
descriptor.resource_count += 1;
|
||||
}
|
||||
}
|
||||
configWrite16(bus, dev, function, 0x04, command);
|
||||
}
|
||||
|
||||
const registered = device.register(bridge_id, &descriptor) orelse {
|
||||
writeLine("/system/drivers/pci-bus: register refused for {d}:{d}.{d}\n", .{ bus, dev, function });
|
||||
return;
|
||||
};
|
||||
const report = protocol.ChildAdded{
|
||||
.parent = bridge_id,
|
||||
.bus_address = (bus << 8) | (dev << 3) | function,
|
||||
.identity = class_triple,
|
||||
.device_id = registered,
|
||||
};
|
||||
var reply: [protocol.message_maximum]u8 = undefined;
|
||||
_ = runtime.ipc.call(manager_handle, std.mem.asBytes(&report), &reply) catch {
|
||||
writeLine("/system/drivers/pci-bus: child report for {d}:{d}.{d} failed\n", .{ bus, dev, function });
|
||||
};
|
||||
}
|
||||
|
||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?runtime.ipc.Handle) usize {
|
||||
_ = message;
|
||||
_ = reply;
|
||||
_ = sender;
|
||||
_ = capability;
|
||||
return 0;
|
||||
}
|
||||
|
||||
pub fn main(init: runtime.process.Init) void {
|
||||
const argument = init.arguments.get(1) orelse return; // bare (ramdisk sweep): stay silent
|
||||
bridge_id = std.fmt.parseInt(u64, argument, 10) catch {
|
||||
writeLine("/system/drivers/pci-bus: malformed bridge device id '{s}'\n", .{argument});
|
||||
return;
|
||||
};
|
||||
runtime.service.run(protocol.message_maximum, .{
|
||||
.init = initialise,
|
||||
.on_message = onMessage,
|
||||
});
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||
}
|
||||
@@ -72,17 +72,17 @@ fn modifierWord(modifiers: scancode.ModifierSnapshot) u32 {
|
||||
pub fn main(init: runtime.process.Init) void {
|
||||
const hid = init.arguments.get(1).?;
|
||||
if (hid.len == 0) {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus/keyboard: no HID argument\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus/keyboard: no HID argument\n");
|
||||
return;
|
||||
}
|
||||
writeLine("system/drivers/ps2-bus/keyboard: starting for hid {s}\n", .{hid});
|
||||
writeLine("/system/drivers/ps2-bus/keyboard: starting for hid {s}\n", .{hid});
|
||||
|
||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus/keyboard: out of memory\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus/keyboard: out of memory\n");
|
||||
return;
|
||||
};
|
||||
if (device.findDeviceDescriptorByHid(buffer, hid) == null) {
|
||||
writeLine("system/drivers/ps2-bus/keyboard: no device for hid {s}\n", .{hid});
|
||||
writeLine("/system/drivers/ps2-bus/keyboard: no device for hid {s}\n", .{hid});
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -90,38 +90,38 @@ pub fn main(init: runtime.process.Init) void {
|
||||
// absent (as today) it defaults to us.
|
||||
const layout_name = init.arguments.get(2) orelse "us";
|
||||
const layout = xkb.byName(layout_name) orelse xkb.us;
|
||||
writeLine("system/drivers/ps2-bus/keyboard: layout {s}\n", .{layout.name});
|
||||
writeLine("/system/drivers/ps2-bus/keyboard: layout {s}\n", .{layout.name});
|
||||
|
||||
// Attach to the bus: hand it our endpoint, and it forwards every byte the
|
||||
// keyboard sends (it owns the controller; we own the decoding).
|
||||
const bus = lookupBus() orelse {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus/keyboard: ps2-bus service unavailable\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus/keyboard: ps2-bus service unavailable\n");
|
||||
return;
|
||||
};
|
||||
const endpoint = ipc.createIpcEndpoint() orelse {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus/keyboard: no endpoint\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus/keyboard: no endpoint\n");
|
||||
return;
|
||||
};
|
||||
var attach = ps2.AttachRequest{ .device_type = @intFromEnum(ps2.DeviceType.keyboard) };
|
||||
var attach_reply: [@sizeOf(ps2.AttachReply)]u8 = undefined;
|
||||
const attached = ipc.callCap(bus, std.mem.asBytes(&attach), &attach_reply, endpoint) catch {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus/keyboard: attach call failed\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus/keyboard: attach call failed\n");
|
||||
return;
|
||||
};
|
||||
if (attached.len < @sizeOf(ps2.AttachReply) or
|
||||
std.mem.bytesToValue(ps2.AttachReply, attach_reply[0..@sizeOf(ps2.AttachReply)]).status != @intFromEnum(ps2.AttachStatus.ok))
|
||||
{
|
||||
_ = runtime.system.write("system/drivers/ps2-bus/keyboard: attach refused\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus/keyboard: attach refused\n");
|
||||
return;
|
||||
}
|
||||
|
||||
// Broadcast keyboard events through the input service so programs can listen
|
||||
// for them (docs/input.md).
|
||||
var source = runtime.input.connectSource() orelse {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus/keyboard: input service unavailable\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus/keyboard: input service unavailable\n");
|
||||
return;
|
||||
};
|
||||
_ = runtime.system.write("system/drivers/ps2-bus/keyboard: ok\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus/keyboard: ok\n");
|
||||
|
||||
var decoder = scancode.Decoder{};
|
||||
var state = scancode.KeyboardState{};
|
||||
|
||||
@@ -51,50 +51,50 @@ pub fn main(init: runtime.process.Init) void {
|
||||
const hid = init.arguments.get(1).?;
|
||||
|
||||
if (hid.len == 0) {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus/mouse: no HID argument\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus/mouse: no HID argument\n");
|
||||
return;
|
||||
}
|
||||
writeLine("system/drivers/ps2-bus/mouse: starting for hid {s}\n", .{hid});
|
||||
writeLine("/system/drivers/ps2-bus/mouse: starting for hid {s}\n", .{hid});
|
||||
|
||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus/mouse: out of memory\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus/mouse: out of memory\n");
|
||||
return;
|
||||
};
|
||||
if (device.findDeviceDescriptorByHid(buffer, hid) == null) {
|
||||
writeLine("system/drivers/ps2-bus/mouse: no device for hid {s}\n", .{hid});
|
||||
writeLine("/system/drivers/ps2-bus/mouse: no device for hid {s}\n", .{hid});
|
||||
return;
|
||||
}
|
||||
|
||||
// Attach to the bus: hand it our endpoint, and it forwards every byte the
|
||||
// mouse sends (it owns the controller; we own the decoding).
|
||||
const bus = lookupBus() orelse {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus/mouse: ps2-bus service unavailable\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus/mouse: ps2-bus service unavailable\n");
|
||||
return;
|
||||
};
|
||||
const endpoint = ipc.createIpcEndpoint() orelse {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus/mouse: no endpoint\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus/mouse: no endpoint\n");
|
||||
return;
|
||||
};
|
||||
var attach = ps2.AttachRequest{ .device_type = @intFromEnum(ps2.DeviceType.mouse) };
|
||||
var attach_reply: [@sizeOf(ps2.AttachReply)]u8 = undefined;
|
||||
const attached = ipc.callCap(bus, std.mem.asBytes(&attach), &attach_reply, endpoint) catch {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus/mouse: attach call failed\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus/mouse: attach call failed\n");
|
||||
return;
|
||||
};
|
||||
if (attached.len < @sizeOf(ps2.AttachReply) or
|
||||
std.mem.bytesToValue(ps2.AttachReply, attach_reply[0..@sizeOf(ps2.AttachReply)]).status != @intFromEnum(ps2.AttachStatus.ok))
|
||||
{
|
||||
_ = runtime.system.write("system/drivers/ps2-bus/mouse: attach refused\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus/mouse: attach refused\n");
|
||||
return;
|
||||
}
|
||||
|
||||
// Broadcast mouse events through the input service so programs can listen
|
||||
// for them (docs/input.md).
|
||||
var source = runtime.input.connectSource() orelse {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus/mouse: input service unavailable\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus/mouse: input service unavailable\n");
|
||||
return;
|
||||
};
|
||||
_ = runtime.system.write("system/drivers/ps2-bus/mouse: ok\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus/mouse: ok\n");
|
||||
|
||||
var assembler = mouse_packet.Assembler{};
|
||||
var buttons: u32 = 0;
|
||||
|
||||
@@ -30,19 +30,19 @@ fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
/// attaches, or null if nothing was spawned.
|
||||
fn spawnIdentifiedDriver(controller: ps2.Controller, port: ps2.Port) ?ps2.DeviceType {
|
||||
const device_type = controller.identifyDevice(port) orelse {
|
||||
writeLine("system/drivers/ps2-bus: identify timed out on port {s}\n", .{@tagName(port)});
|
||||
writeLine("/system/drivers/ps2-bus: identify timed out on port {s}\n", .{@tagName(port)});
|
||||
return null;
|
||||
};
|
||||
const driver_name = device_type.driverName() orelse {
|
||||
writeLine("system/drivers/ps2-bus: unrecognized device on port {s}\n", .{@tagName(port)});
|
||||
writeLine("/system/drivers/ps2-bus: unrecognized device on port {s}\n", .{@tagName(port)});
|
||||
return null;
|
||||
};
|
||||
const hid = device_type.hid() orelse "";
|
||||
if (runtime.system.spawnWithArguments(driver_name, &.{hid}) != null) {
|
||||
writeLine("system/drivers/ps2-bus: port {s} is a {s}, spawned {s}\n", .{ @tagName(port), hid, driver_name });
|
||||
writeLine("/system/drivers/ps2-bus: port {s} is a {s}, spawned {s}\n", .{ @tagName(port), hid, driver_name });
|
||||
return device_type;
|
||||
}
|
||||
writeLine("system/drivers/ps2-bus: failed to spawn {s}\n", .{driver_name});
|
||||
writeLine("/system/drivers/ps2-bus: failed to spawn {s}\n", .{driver_name});
|
||||
return null;
|
||||
}
|
||||
|
||||
@@ -83,7 +83,7 @@ fn handleAttach(message: []const u8, got: ipc.Received, out: []u8) usize {
|
||||
const device_type = maybe_type orelse continue;
|
||||
if (@intFromEnum(device_type) != request.device_type) continue;
|
||||
port_endpoints[port_index] = endpoint;
|
||||
writeLine("system/drivers/ps2-bus: {s} driver attached\n", .{@tagName(device_type)});
|
||||
writeLine("/system/drivers/ps2-bus: {s} driver attached\n", .{@tagName(device_type)});
|
||||
return reply.write(out, .ok);
|
||||
}
|
||||
return reply.write(out, .no_such_device);
|
||||
@@ -91,7 +91,7 @@ fn handleAttach(message: []const u8, got: ipc.Received, out: []u8) usize {
|
||||
|
||||
pub fn main() void {
|
||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: out of memory\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: out of memory\n");
|
||||
return;
|
||||
};
|
||||
|
||||
@@ -103,16 +103,16 @@ pub fn main() void {
|
||||
// is on which port is decided later by identify, not by this HID.
|
||||
const maybe_controller_device_descriptor = device.findDeviceDescriptorByHid(buffer, acpi_ids.HardwareId.ps2_keyboard.hid());
|
||||
if (maybe_controller_device_descriptor) |controller_device_descriptor| {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: found PS/2 controller\n");
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: initializing controller\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: found PS/2 controller\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: initializing controller\n");
|
||||
|
||||
if (!device.claim(controller_device_descriptor.id)) {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: unable to claim controller \n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: unable to claim controller \n");
|
||||
return;
|
||||
}
|
||||
|
||||
const controller = ps2.Controller.init(controller_device_descriptor) orelse {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: controller is missing its IO ports\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: controller is missing its IO ports\n");
|
||||
return;
|
||||
};
|
||||
maybe_controller = controller;
|
||||
@@ -123,7 +123,7 @@ pub fn main() void {
|
||||
controller.flushOutputBuffer();
|
||||
|
||||
const current = controller.readConfigurationByte() orelse {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: controller configuration timed out\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: controller configuration timed out\n");
|
||||
return;
|
||||
};
|
||||
|
||||
@@ -132,49 +132,49 @@ pub fn main() void {
|
||||
ps2.configuration_first_port_translation);
|
||||
|
||||
if (controller.writeConfigurationByte(update) == null) {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: controller configuration timed out\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: controller configuration timed out\n");
|
||||
return;
|
||||
}
|
||||
|
||||
if (controller.performSelfTest()) |reply| {
|
||||
if (reply != ps2.response_controller_test_passed) {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: perform controller self test failed\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: perform controller self test failed\n");
|
||||
return;
|
||||
}
|
||||
} else {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: controller self test timed out\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: controller self test timed out\n");
|
||||
return;
|
||||
}
|
||||
|
||||
has_two_channels = controller.hasTwoChannels() orelse {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: controller channels timed out\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: controller channels timed out\n");
|
||||
return;
|
||||
};
|
||||
|
||||
if (has_two_channels) {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: has two channels\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: has two channels\n");
|
||||
// keep the bus quiet until we have tested the ports and are ready to use them
|
||||
controller.disablePort(.two);
|
||||
} else {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: has one channel\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: has one channel\n");
|
||||
}
|
||||
|
||||
// interface tests: always test port 1, test port 2 only if it exists
|
||||
const port_one_works = (controller.testPort(.one) orelse {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: port 1 test timed out\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: port 1 test timed out\n");
|
||||
return;
|
||||
}) == ps2.response_port_test_passed;
|
||||
|
||||
var port_two_works = false;
|
||||
if (has_two_channels) {
|
||||
port_two_works = (controller.testPort(.two) orelse {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: port 2 test timed out\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: port 2 test timed out\n");
|
||||
return;
|
||||
}) == ps2.response_port_test_passed;
|
||||
}
|
||||
|
||||
if (!port_one_works and !port_two_works) {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: no usable ports\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: no usable ports\n");
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -188,16 +188,16 @@ pub fn main() void {
|
||||
// abort bring-up of the other one
|
||||
if (port_one_works) {
|
||||
if (controller.resetDevice(.one)) |passed| {
|
||||
if (!passed) _ = runtime.system.write("system/drivers/ps2-bus: port 1 device reset failed\n");
|
||||
if (!passed) _ = runtime.system.write("/system/drivers/ps2-bus: port 1 device reset failed\n");
|
||||
} else {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: port 1 device reset timed out\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: port 1 device reset timed out\n");
|
||||
}
|
||||
}
|
||||
if (port_two_works) {
|
||||
if (controller.resetDevice(.two)) |passed| {
|
||||
if (!passed) _ = runtime.system.write("system/drivers/ps2-bus: port 2 device reset failed\n");
|
||||
if (!passed) _ = runtime.system.write("/system/drivers/ps2-bus: port 2 device reset failed\n");
|
||||
} else {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: port 2 device reset timed out\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: port 2 device reset timed out\n");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -207,13 +207,13 @@ pub fn main() void {
|
||||
if (port_one_works) port_device_types[@intFromEnum(ps2.Port.one)] = spawnIdentifiedDriver(controller, .one);
|
||||
if (port_two_works) port_device_types[@intFromEnum(ps2.Port.two)] = spawnIdentifiedDriver(controller, .two);
|
||||
} else {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: no PS/2 controller found\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: no PS/2 controller found\n");
|
||||
return;
|
||||
}
|
||||
|
||||
const controller = maybe_controller.?;
|
||||
const interrupt_index = maybe_interrupt_index orelse {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: controller is missing its IRQ\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: controller is missing its IRQ\n");
|
||||
return;
|
||||
};
|
||||
|
||||
@@ -221,11 +221,11 @@ pub fn main() void {
|
||||
// well-known id so the children can find it, the way input subscribers find
|
||||
// the input service.
|
||||
const endpoint = ipc.createIpcEndpoint() orelse {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: no endpoint\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: no endpoint\n");
|
||||
return;
|
||||
};
|
||||
if (!ipc.register(.ps2_bus, endpoint)) {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: register failed\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: register failed\n");
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -234,7 +234,7 @@ pub fn main() void {
|
||||
// let the controller raise them — an interrupt with nobody bound is lost.
|
||||
controller.drainOutputBuffer();
|
||||
if (!device.irqBind(controller.device_id, interrupt_index, endpoint)) {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: irq_bind failed\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: irq_bind failed\n");
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -253,21 +253,21 @@ pub fn main() void {
|
||||
.gsi = descriptor.resources[auxiliary_index].start,
|
||||
};
|
||||
} else {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: auxiliary irq_bind failed\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: auxiliary irq_bind failed\n");
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
var configuration = controller.readConfigurationByte() orelse {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: controller configuration timed out\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: controller configuration timed out\n");
|
||||
return;
|
||||
};
|
||||
if (port_device_types[@intFromEnum(ps2.Port.one)] != null) configuration |= ps2.Port.one.interruptBit();
|
||||
if (maybe_auxiliary_interrupt != null) configuration |= ps2.Port.two.interruptBit();
|
||||
_ = controller.writeConfigurationByte(configuration);
|
||||
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: ok\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: ok\n");
|
||||
|
||||
// The forwarding loop: an IRQ1 notification drains the output buffer, routing
|
||||
// each byte to the attached driver of the port it came from; a client message
|
||||
|
||||
@@ -1,68 +1,188 @@
|
||||
//! /system/drivers/usb-xhci-bus — the xHCI (USB 3) host-controller bus driver.
|
||||
//! The device manager spawns **one instance per controller** it discovers (a machine
|
||||
//! can carry several), passing the controller's device-tree id as argv[1]; this
|
||||
//! instance claims that device and no other, so multiple instances never fight over
|
||||
//! hardware. This increment proves the plumbing: parse the id, claim the controller,
|
||||
//! and report its MMIO window. The next increments map the registers and bring the
|
||||
//! controller up (reset, rings, port scan), then enumerate the USB devices on the
|
||||
//! bus with the usb-abi request builders and publish each with `device_register`.
|
||||
//! The device manager spawns **one instance per controller** it discovers (a
|
||||
//! machine can carry several), passing the controller's device-tree id as
|
||||
//! argv[1]; this instance claims that device and no other, so multiple
|
||||
//! instances never fight over hardware.
|
||||
//!
|
||||
//! M18.2 (this increment): after the hello, real hardware — map the xHC's
|
||||
//! register window (the first memory BAR; resource 0 is the ECAM config
|
||||
//! space), read the capability registers, and walk the root-hub ports: one
|
||||
//! `child_added` report to the manager per connected port, carrying the port
|
||||
//! number and the PORTSC speed class as identity. No transfer rings yet —
|
||||
//! descriptors and USB class matching are the USB track; the connect bit and
|
||||
//! speed come straight from PORTSC, which reflects hardware state whether or
|
||||
//! not the controller is running.
|
||||
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
const protocol = runtime.device_manager_protocol;
|
||||
const device = runtime.device;
|
||||
|
||||
/// Format one whole log line and emit it in a single `debug_write`, so concurrent
|
||||
/// instances (one per controller) can never interleave mid-line.
|
||||
/// Format one whole log line and emit it in a single `debug_write`, so
|
||||
/// concurrent instances (one per controller) can never interleave mid-line.
|
||||
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
var line: [128]u8 = undefined;
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
pub fn main(init: runtime.process.Init) void {
|
||||
const argument = init.arguments.get(1) orelse {
|
||||
_ = runtime.system.write("usb-xhci-bus: missing controller device id (argv[1])\n");
|
||||
return;
|
||||
};
|
||||
const controller_id = std.fmt.parseInt(u64, argument, 10) catch {
|
||||
writeLine("usb-xhci-bus: malformed controller device id '{s}'\n", .{argument});
|
||||
return;
|
||||
};
|
||||
var controller_id: u64 = protocol.no_device;
|
||||
|
||||
/// Claim the assigned controller, find its register window, and hello the
|
||||
/// manager. Any failure returns false: the process exits cleanly, which the
|
||||
/// manager reads as "meant to stop" — a missing assignment is not a crash loop.
|
||||
fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||
_ = endpoint;
|
||||
if (!device.claim(controller_id)) {
|
||||
writeLine("usb-xhci-bus: unable to claim controller device {d}\n", .{controller_id});
|
||||
return;
|
||||
writeLine("/system/drivers/usb-xhci-bus: unable to claim controller device {d}\n", .{controller_id});
|
||||
return false;
|
||||
}
|
||||
|
||||
// Fetch our own descriptor back for the controller's resources.
|
||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||
_ = runtime.system.write("usb-xhci-bus: out of memory\n");
|
||||
return;
|
||||
_ = runtime.system.write("/system/drivers/usb-xhci-bus: out of memory\n");
|
||||
return false;
|
||||
};
|
||||
const total = device.enumerate(buffer);
|
||||
const descriptor = for (buffer[0..@min(total, buffer.len)]) |d| {
|
||||
if (d.id == controller_id) break d;
|
||||
} else {
|
||||
writeLine("usb-xhci-bus: device {d} not in the device tree\n", .{controller_id});
|
||||
return;
|
||||
writeLine("/system/drivers/usb-xhci-bus: device {d} not in the device tree\n", .{controller_id});
|
||||
return false;
|
||||
};
|
||||
|
||||
// The controller's operational registers live behind BAR0, enumerated as the
|
||||
// device's first memory resource.
|
||||
const register_window = for (descriptor.resources[0..@intCast(descriptor.resource_count)]) |resource| {
|
||||
if (resource.kind == @intFromEnum(device.ResourceKind.memory)) break resource;
|
||||
// The xHC's registers live behind the first memory BAR. Resource 0 is the
|
||||
// function's ECAM configuration space (M15), so the walk starts at 1.
|
||||
var register_index: u64 = 0;
|
||||
const register_window = for (descriptor.resources[1..@intCast(descriptor.resource_count)], 1..) |resource, index| {
|
||||
if (resource.kind == @intFromEnum(device.ResourceKind.memory)) {
|
||||
register_index = index;
|
||||
break resource;
|
||||
}
|
||||
} else {
|
||||
writeLine("usb-xhci-bus: controller device {d} has no MMIO window\n", .{controller_id});
|
||||
return;
|
||||
writeLine("/system/drivers/usb-xhci-bus: controller device {d} has no register BAR\n", .{controller_id});
|
||||
return false;
|
||||
};
|
||||
writeLine("usb-xhci-bus: claimed controller device {d} (registers at 0x{x}, {d} bytes)\n", .{
|
||||
writeLine("/system/drivers/usb-xhci-bus: claimed controller device {d} (registers at 0x{x}, {d} bytes)\n", .{
|
||||
controller_id,
|
||||
register_window.start,
|
||||
register_window.len,
|
||||
});
|
||||
register_base = device.mmioMap(controller_id, register_index) orelse {
|
||||
_ = runtime.system.write("/system/drivers/usb-xhci-bus: mmio_map failed\n");
|
||||
return false;
|
||||
};
|
||||
|
||||
// Controller bring-up (map the window, reset, rings, port scan) is the next
|
||||
// increment; stay resident as the bus's supervisor in the meantime.
|
||||
while (true) runtime.system.sleep(1000);
|
||||
// The handshake: role, protocol version, assignment — inside the manager's
|
||||
// deadline (the lookup retries cover the manager still registering).
|
||||
var manager: ?runtime.ipc.Handle = null;
|
||||
var tries: u32 = 0;
|
||||
while (manager == null and tries < 100) : (tries += 1) {
|
||||
manager = runtime.ipc.lookup(.device_manager);
|
||||
if (manager == null) runtime.system.sleep(20);
|
||||
}
|
||||
const h = manager orelse {
|
||||
_ = runtime.system.write("/system/drivers/usb-xhci-bus: no device manager to hello\n");
|
||||
return false;
|
||||
};
|
||||
const hello = protocol.Hello{ .role = @intFromEnum(protocol.Role.bus), .device_id = controller_id };
|
||||
var reply: [protocol.message_maximum]u8 = undefined;
|
||||
const n = runtime.ipc.call(h, std.mem.asBytes(&hello), &reply) catch {
|
||||
_ = runtime.system.write("/system/drivers/usb-xhci-bus: hello call failed\n");
|
||||
return false;
|
||||
};
|
||||
if (n < protocol.reply_size or std.mem.bytesToValue(protocol.HelloReply, reply[0..protocol.reply_size]).status != 0) {
|
||||
_ = runtime.system.write("/system/drivers/usb-xhci-bus: hello refused\n");
|
||||
return false;
|
||||
}
|
||||
_ = runtime.system.write("/system/drivers/usb-xhci-bus: hello acknowledged\n");
|
||||
|
||||
scanPorts(h);
|
||||
return true;
|
||||
}
|
||||
|
||||
var register_base: usize = 0;
|
||||
|
||||
/// One 32-bit volatile register read at `offset` from the mapped window.
|
||||
fn readRegister(offset: usize) u32 {
|
||||
const register: *volatile u32 = @ptrFromInt(register_base + offset);
|
||||
return register.*;
|
||||
}
|
||||
|
||||
/// The xHCI default Protocol Speed IDs (the PORTSC port-speed field, bits 13:10)
|
||||
/// decoded to human names — the boot-log breadcrumb for what actually enumerated on
|
||||
/// a port, the USB analog of the pci-bus class-code line. A controller may redefine
|
||||
/// these through its Supported Protocol capability, but the defaults cover every
|
||||
/// speed QEMU and real hardware report at this (pre-descriptor) stage.
|
||||
fn speedName(speed: u32) []const u8 {
|
||||
return switch (speed) {
|
||||
1 => "Full-speed (USB 2.0, 12 Mb/s)",
|
||||
2 => "Low-speed (USB 2.0, 1.5 Mb/s)",
|
||||
3 => "High-speed (USB 2.0, 480 Mb/s)",
|
||||
4 => "SuperSpeed (USB 3.0, 5 Gb/s)",
|
||||
5 => "SuperSpeedPlus (USB 3.1, 10 Gb/s)",
|
||||
else => "unknown speed",
|
||||
};
|
||||
}
|
||||
|
||||
/// The root-hub port scan: read the capability registers for the port count
|
||||
/// and the operational-register offset, then one PORTSC per port. The connect
|
||||
/// bit (CCS) and the speed field reflect hardware state directly — no
|
||||
/// controller reset or run needed to *see* the devices; driving them needs the
|
||||
/// rings (the USB track).
|
||||
fn scanPorts(manager: runtime.ipc.Handle) void {
|
||||
// Capability registers: CAPLENGTH is byte 0 of the first dword; HCSPARAMS1
|
||||
// carries MaxPorts in bits 31:24.
|
||||
const capability_length = readRegister(0) & 0xFF;
|
||||
const structural = readRegister(0x04);
|
||||
const maximum_ports: u32 = structural >> 24;
|
||||
writeLine("/system/drivers/usb-xhci-bus: {d} root-hub ports\n", .{maximum_ports});
|
||||
|
||||
// PORTSC registers: operational base + 0x400 + 0x10 per port (1-based).
|
||||
var port: u32 = 1;
|
||||
var connected: u32 = 0;
|
||||
while (port <= maximum_ports) : (port += 1) {
|
||||
const port_status = readRegister(capability_length + 0x400 + 0x10 * (port - 1));
|
||||
if (port_status & 1 == 0) continue; // CCS: nothing connected
|
||||
connected += 1;
|
||||
const speed = (port_status >> 10) & 0xF; // the PORTSC port-speed class
|
||||
writeLine("/system/drivers/usb-xhci-bus: port {d} connected — {s} (speed class {d})\n", .{ port, speedName(speed), speed });
|
||||
|
||||
const report = protocol.ChildAdded{
|
||||
.parent = controller_id,
|
||||
.bus_address = port,
|
||||
.identity = speed,
|
||||
};
|
||||
var reply: [protocol.message_maximum]u8 = undefined;
|
||||
_ = runtime.ipc.call(manager, std.mem.asBytes(&report), &reply) catch {
|
||||
writeLine("/system/drivers/usb-xhci-bus: child report for port {d} failed\n", .{port});
|
||||
continue;
|
||||
};
|
||||
}
|
||||
if (connected == 0) _ = runtime.system.write("/system/drivers/usb-xhci-bus: no devices connected\n");
|
||||
}
|
||||
|
||||
/// No bus protocol to serve yet — transfer requests arrive with the USB track.
|
||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?runtime.ipc.Handle) usize {
|
||||
_ = message;
|
||||
_ = reply;
|
||||
_ = sender;
|
||||
_ = capability;
|
||||
return 0;
|
||||
}
|
||||
|
||||
pub fn main(init: runtime.process.Init) void {
|
||||
const argument = init.arguments.get(1) orelse {
|
||||
_ = runtime.system.write("/system/drivers/usb-xhci-bus: missing controller device id (argv[1])\n");
|
||||
return;
|
||||
};
|
||||
controller_id = std.fmt.parseInt(u64, argument, 10) catch {
|
||||
writeLine("/system/drivers/usb-xhci-bus: malformed controller device id '{s}'\n", .{argument});
|
||||
return;
|
||||
};
|
||||
runtime.service.run(protocol.message_maximum, .{
|
||||
.init = initialise,
|
||||
.on_message = onMessage,
|
||||
});
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
|
||||
@@ -80,6 +80,35 @@ var timer_hz: u32 = 0;
|
||||
var tsc_hz: u64 = 0;
|
||||
var tsc_base: u64 = 0;
|
||||
|
||||
/// Whether the TSC is architecturally **invariant** — a constant rate regardless of
|
||||
/// P/C-state transitions, and thus valid as a clocksource (CPUID leaf 0x80000007,
|
||||
/// EDX bit 8). AMD and modern Intel set it; the bare qemu64 model does not. Measured
|
||||
/// frequency alone is not enough: a non-invariant TSC speeds up and slows down with
|
||||
/// the core clock, so reading it as wall time would drift.
|
||||
var tsc_invariant: bool = false;
|
||||
/// Cleared if the cross-core warp check (checkWarpSource) ever sees the TSC read
|
||||
/// lower on one core than the max another core has already published — i.e. the
|
||||
/// per-core TSCs are not synchronized, and a task migrating cores could see time go
|
||||
/// backward. Starts true (assume synchronized until proven otherwise).
|
||||
var tsc_synced: bool = true;
|
||||
/// The worst backward skew the warp check observed, in TSC cycles (0 = none).
|
||||
var tsc_warp_cycles: u64 = 0;
|
||||
|
||||
/// The monotonic clock's source. The TSC when it is invariant *and* synchronized —
|
||||
/// the fast `rdtsc` path taken on real Intel/AMD and modern VMs. Otherwise the HPET
|
||||
/// main counter: a single fixed-rate counter, immune to both per-core skew and
|
||||
/// frequency scaling, so it stays accurate on a bare VM or a warped machine.
|
||||
const ClockSource = enum { tsc, hpet };
|
||||
var clock_source: ClockSource = .tsc;
|
||||
|
||||
/// HPET standby clocksource, set up in calibrate() whenever an HPET exists (whether
|
||||
/// or not calibration itself measured against it): its frequency, the counter value
|
||||
/// chosen as the zero point, and its width mask. Only a 64-bit HPET is used as a
|
||||
/// clocksource — a 32-bit one wraps too fast to be monotonic without accumulation.
|
||||
var hpet_clock_hz: u64 = 0;
|
||||
var hpet_clock_base: u64 = 0;
|
||||
var hpet_clock_mask: u64 = ~@as(u64, 0);
|
||||
|
||||
/// Read the 64-bit Time Stamp Counter.
|
||||
fn rdtsc() u64 {
|
||||
var low: u32 = undefined;
|
||||
@@ -220,6 +249,31 @@ pub fn calibrate() void {
|
||||
}
|
||||
|
||||
tsc_base = rdtsc(); // the clock's zero point (boot)
|
||||
|
||||
// Decide whether the TSC is trustworthy as a clocksource. Frequency (measured
|
||||
// above, possibly against the HPET/PIT) is necessary but not sufficient: the TSC
|
||||
// must also be *invariant* (CPUID 0x80000007 EDX[8]). AMD and modern Intel set
|
||||
// this; the bare qemu64 model does not.
|
||||
tsc_invariant = tscIsInvariant();
|
||||
|
||||
// Bring up the HPET as a standby clocksource whenever one exists — even on the
|
||||
// CPUID-0x15 path where calibration never touched it — so a non-invariant TSC
|
||||
// (here) or an unsynchronized one (checkWarpSource, during SMP bring-up) can fall
|
||||
// back to a source that is immune to both. hpetHz() maps + enables the counter
|
||||
// and is idempotent if calibration already used it.
|
||||
if (configuration_hpet_base != 0) {
|
||||
if (hpetHz()) |hz| {
|
||||
hpet_clock_mask = hpetMask();
|
||||
if (hpet_clock_mask == ~@as(u64, 0)) { // only a 64-bit HPET is monotonic enough
|
||||
hpet_clock_hz = hz;
|
||||
hpet_clock_base = readHpet();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Select the source: the fast TSC when invariant, else the HPET if we have one.
|
||||
// (checkWarpSource may still demote TSC -> HPET later if the cores' TSCs skew.)
|
||||
if (!tsc_invariant and hpet_clock_hz != 0) clock_source = .hpet;
|
||||
}
|
||||
|
||||
/// Run the LAPIC timer one-shot from its maximum count while a monotonic reference
|
||||
@@ -275,7 +329,7 @@ fn calibratePit() void {
|
||||
// --- reference clocks ------------------------------------------------------
|
||||
|
||||
/// TSC frequency from CPUID leaf 0x15 (crystal_hz * numerator / denominator), or
|
||||
/// null if the CPU doesn't enumerate it (common under QEMU).
|
||||
/// null if the CPU doesn't enumerate it (common under QEMU, and on AMD).
|
||||
fn cpuidTscHz() ?u64 {
|
||||
if (cpuid(0).eax < 0x15) return null;
|
||||
const r = cpuid(0x15);
|
||||
@@ -283,6 +337,15 @@ fn cpuidTscHz() ?u64 {
|
||||
return @as(u64, r.ecx) * r.ebx / r.eax;
|
||||
}
|
||||
|
||||
/// Whether the CPU advertises an **invariant** TSC (CPUID leaf 0x80000007, EDX
|
||||
/// bit 8) — the architectural guarantee, on both Intel and AMD, that the TSC ticks
|
||||
/// at a constant rate across P/C-states and never stops. Requires the extended-leaf
|
||||
/// range to reach 0x80000007 first.
|
||||
fn tscIsInvariant() bool {
|
||||
if (cpuid(0x80000000).eax < 0x80000007) return false;
|
||||
return (cpuid(0x80000007).edx & (1 << 8)) != 0;
|
||||
}
|
||||
|
||||
const CpuidRegs = struct { eax: u32, ebx: u32, ecx: u32, edx: u32 };
|
||||
|
||||
fn cpuid(leaf: u32) CpuidRegs {
|
||||
@@ -310,11 +373,20 @@ fn hpetWrite64(off: usize, value: u64) void {
|
||||
@as(*volatile u64, @ptrFromInt(configuration_hpet_base + off)).* = value;
|
||||
}
|
||||
|
||||
/// Whether the HPET has been mapped into the physmap yet, so `configuration_hpet_base`
|
||||
/// already holds the virtual address. `hpetHz` is called more than once (calibration
|
||||
/// may use the HPET, and the standby-clocksource setup asks for it again), and mapping
|
||||
/// an already-mapped base a second time would double-offset it into an overflow.
|
||||
var hpet_mapped: bool = false;
|
||||
|
||||
/// Map + enable the HPET and return its tick frequency, or null if unusable.
|
||||
/// Maps the HPET into the physmap and switches configuration_hpet_base to that virtual
|
||||
/// address, so the register accessors reach it without the identity map.
|
||||
/// address, so the register accessors reach it without the identity map. Idempotent.
|
||||
fn hpetHz() ?u64 {
|
||||
configuration_hpet_base = paging.mapMmio(configuration_hpet_base, 0x400, true);
|
||||
if (!hpet_mapped) {
|
||||
configuration_hpet_base = paging.mapMmio(configuration_hpet_base, 0x400, true);
|
||||
hpet_mapped = true;
|
||||
}
|
||||
const caps = hpetRead64(0x00);
|
||||
const period_fs = caps >> 32; // femtoseconds per tick
|
||||
if (period_fs == 0) return null;
|
||||
@@ -364,24 +436,169 @@ pub fn tscHz() u64 {
|
||||
return tsc_hz;
|
||||
}
|
||||
|
||||
// Monotonic high-resolution clock, from the TSC. A function per resolution, each
|
||||
// scaling the cycle delta directly at its unit (the 128-bit intermediate avoids
|
||||
// overflow across a long uptime). nanos() resolves to a few ns; millis() is what
|
||||
// the scheduler uses for sleep deadlines.
|
||||
// Monotonic high-resolution clock. A function per resolution, each scaling the
|
||||
// counter delta directly at its unit (the 128-bit intermediate avoids overflow
|
||||
// across a long uptime). nanos() resolves to a few ns on the TSC; millis() is what
|
||||
// the scheduler uses for sleep deadlines. The source is the TSC when it is invariant
|
||||
// and synchronized, else the HPET counter (see clock_source) — the branch is one
|
||||
// global load and the TSC path is unchanged from before.
|
||||
|
||||
/// The selected source's counter delta since its zero point.
|
||||
fn clockCount() u64 {
|
||||
return switch (clock_source) {
|
||||
.tsc => rdtsc() -% tsc_base,
|
||||
// A 64-bit HPET (the only kind we select) never wraps in any realistic
|
||||
// uptime, so the wrapping subtraction is exact.
|
||||
.hpet => readHpet() -% hpet_clock_base,
|
||||
};
|
||||
}
|
||||
|
||||
/// The selected source's frequency (0 if the clock is unavailable/uncalibrated).
|
||||
fn clockHertz() u64 {
|
||||
return switch (clock_source) {
|
||||
.tsc => tsc_hz,
|
||||
.hpet => hpet_clock_hz,
|
||||
};
|
||||
}
|
||||
|
||||
pub fn nanos() u64 {
|
||||
if (tsc_hz == 0) return 0;
|
||||
return @intCast(@as(u128, rdtsc() -% tsc_base) * 1_000_000_000 / tsc_hz);
|
||||
const hz = clockHertz();
|
||||
if (hz == 0) return 0;
|
||||
return @intCast(@as(u128, clockCount()) * 1_000_000_000 / hz);
|
||||
}
|
||||
|
||||
pub fn micros() u64 {
|
||||
if (tsc_hz == 0) return 0;
|
||||
return @intCast(@as(u128, rdtsc() -% tsc_base) * 1_000_000 / tsc_hz);
|
||||
const hz = clockHertz();
|
||||
if (hz == 0) return 0;
|
||||
return @intCast(@as(u128, clockCount()) * 1_000_000 / hz);
|
||||
}
|
||||
|
||||
pub fn millis() u64 {
|
||||
if (tsc_hz == 0) return 0;
|
||||
return @intCast(@as(u128, rdtsc() -% tsc_base) * 1_000 / tsc_hz);
|
||||
const hz = clockHertz();
|
||||
if (hz == 0) return 0;
|
||||
return @intCast(@as(u128, clockCount()) * 1_000 / hz);
|
||||
}
|
||||
|
||||
/// Whether the CPU advertises an invariant TSC (CPUID 0x80000007 EDX[8]).
|
||||
pub fn tscInvariant() bool {
|
||||
return tsc_invariant;
|
||||
}
|
||||
|
||||
/// Test hook: force the TSC clocksource on, as if the CPU had advertised an invariant
|
||||
/// TSC. QEMU's TCG accelerator (the only one for an x86 guest on an Apple-Silicon
|
||||
/// host) does not expose the invariant-TSC bit — its emulated TSC isn't invariant — so
|
||||
/// the tsc-sync test can't reach the real-Intel/AMD/KVM path through CPUID. This lets
|
||||
/// that test exercise the TSC clocksource and the cross-core warp check anyway. tsc_base
|
||||
/// is left as-is so the switch from the HPET is continuous.
|
||||
pub fn forceTscClocksourceForTest() void {
|
||||
tsc_invariant = true;
|
||||
clock_source = .tsc;
|
||||
}
|
||||
|
||||
/// How many per-AP warp checks actually ran (a rendezvous completed) — lets a test
|
||||
/// confirm the cross-core check executed rather than being skipped.
|
||||
pub fn warpChecksRun() u32 {
|
||||
return warp_checks;
|
||||
}
|
||||
|
||||
/// Whether the per-core TSCs are synchronized (no backward warp seen at bring-up).
|
||||
pub fn tscSynced() bool {
|
||||
return tsc_synced;
|
||||
}
|
||||
|
||||
/// The active monotonic clocksource, for the boot log and tests.
|
||||
pub fn clockSourceName() []const u8 {
|
||||
return switch (clock_source) {
|
||||
.tsc => "tsc",
|
||||
.hpet => "hpet",
|
||||
};
|
||||
}
|
||||
|
||||
// --- cross-core TSC synchronization ("warp") check -------------------------
|
||||
// Two cores hammer a shared "max seen" TSC value under a lock; if either reads a
|
||||
// value below that max, its TSC lags the other's, and time would run backward for a
|
||||
// task migrating between them (Linux calls this a warp). danos brings APs up one at a
|
||||
// time, so this runs pairwise: the BSP (source) against each AP (target) as it comes
|
||||
// online. It only matters — and only runs — while the TSC is the clocksource; on a
|
||||
// machine already on the HPET (a bare VM) the whole rendezvous is skipped.
|
||||
|
||||
var warp_lock: u32 = 0;
|
||||
var warp_last: u64 = 0;
|
||||
var warp_bsp_ready: u32 = 0;
|
||||
var warp_ap_ready: u32 = 0;
|
||||
var warp_stop: u32 = 0;
|
||||
var warp_checks: u32 = 0; // completed per-AP rendezvous count (for the tsc-sync test)
|
||||
|
||||
const warp_rounds: u32 = 1 << 20; // locked reads on the BSP: ~1 ms at GHz rates
|
||||
const warp_spin_limit: u64 = 1 << 32; // bound every rendezvous wait so a lost core can't hang boot
|
||||
|
||||
fn warpTick() void {
|
||||
while (@cmpxchgWeak(u32, &warp_lock, 0, 1, .acquire, .monotonic) != null) asm volatile ("pause");
|
||||
const t = rdtsc();
|
||||
if (t < warp_last) {
|
||||
const delta = warp_last - t;
|
||||
if (delta > tsc_warp_cycles) tsc_warp_cycles = delta;
|
||||
tsc_synced = false;
|
||||
} else {
|
||||
warp_last = t;
|
||||
}
|
||||
@atomicStore(u32, &warp_lock, 0, .release);
|
||||
}
|
||||
|
||||
/// Spin (bounded) until `flag` is nonzero; false on timeout.
|
||||
fn warpAwait(flag: *u32) bool {
|
||||
var spins: u64 = 0;
|
||||
while (@atomicLoad(u32, flag, .acquire) == 0) : (spins += 1) {
|
||||
if (spins >= warp_spin_limit) return false;
|
||||
asm volatile ("pause");
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/// BSP side of the pairwise TSC warp check, run once per AP as it reports in. No-op
|
||||
/// unless the TSC is the active clocksource. If the AP's TSC proves to lag, demote
|
||||
/// the monotonic clock to the HPET without a discontinuity.
|
||||
pub fn checkWarpSource() void {
|
||||
if (clock_source != .tsc) return;
|
||||
warp_last = 0;
|
||||
@atomicStore(u32, &warp_stop, 0, .release);
|
||||
@atomicStore(u32, &warp_ap_ready, 0, .release);
|
||||
@atomicStore(u32, &warp_bsp_ready, 1, .release);
|
||||
if (!warpAwait(&warp_ap_ready)) { // AP never joined the rendezvous; skip, don't hang
|
||||
@atomicStore(u32, &warp_bsp_ready, 0, .release);
|
||||
return;
|
||||
}
|
||||
var i: u32 = 0;
|
||||
while (i < warp_rounds) : (i += 1) warpTick();
|
||||
@atomicStore(u32, &warp_stop, 1, .release);
|
||||
@atomicStore(u32, &warp_bsp_ready, 0, .release);
|
||||
warp_checks += 1;
|
||||
|
||||
if (!tsc_synced and hpet_clock_hz != 0) demoteToHpet();
|
||||
}
|
||||
|
||||
/// AP side: join the BSP's warp check, then return so the core can enter the
|
||||
/// scheduler. Bounded so a missing BSP can't strand the core.
|
||||
pub fn checkWarpTarget() void {
|
||||
if (clock_source != .tsc) return;
|
||||
if (!warpAwait(&warp_bsp_ready)) return;
|
||||
@atomicStore(u32, &warp_ap_ready, 1, .release);
|
||||
var spins: u64 = 0;
|
||||
while (@atomicLoad(u32, &warp_stop, .acquire) == 0) : (spins += 1) {
|
||||
if (spins >= warp_spin_limit) return;
|
||||
warpTick();
|
||||
}
|
||||
}
|
||||
|
||||
/// Switch the clocksource from the TSC to the HPET without a discontinuity: choose
|
||||
/// the HPET zero point so it reads the same nanosecond value the TSC does right now,
|
||||
/// so time neither jumps nor runs backward across the switch. Called when the warp
|
||||
/// check proves the per-core TSCs unsynchronized.
|
||||
fn demoteToHpet() void {
|
||||
const now_ns = @as(u128, rdtsc() -% tsc_base) * 1_000_000_000 / tsc_hz;
|
||||
const equivalent_ticks: u64 = @intCast(now_ns * hpet_clock_hz / 1_000_000_000);
|
||||
hpet_clock_base = readHpet() -% equivalent_ticks;
|
||||
clock_source = .hpet;
|
||||
}
|
||||
|
||||
/// Acknowledge the current interrupt so the LAPIC will deliver the next one.
|
||||
|
||||
@@ -469,6 +469,35 @@ pub fn clockHz() u64 {
|
||||
return apic.tscHz();
|
||||
}
|
||||
|
||||
/// Whether the CPU guarantees an **invariant** TSC (CPUID 0x80000007 EDX[8] on
|
||||
/// x86; the analogous architectural guarantee elsewhere). When false the TSC is not
|
||||
/// used as the clocksource.
|
||||
pub fn clockInvariant() bool {
|
||||
return apic.tscInvariant();
|
||||
}
|
||||
|
||||
/// Whether the per-core clock counters are synchronized (no backward warp observed
|
||||
/// at SMP bring-up). When false the clock falls back off the TSC.
|
||||
pub fn clockSynchronized() bool {
|
||||
return apic.tscSynced();
|
||||
}
|
||||
|
||||
/// The active monotonic clocksource, for the boot log ("tsc" or "hpet" on x86).
|
||||
pub fn clockSourceName() []const u8 {
|
||||
return apic.clockSourceName();
|
||||
}
|
||||
|
||||
/// Test hook: force the TSC clocksource on, to exercise the TSC + warp-check path on
|
||||
/// a hypervisor that won't advertise an invariant TSC (see apic.forceTscClocksourceForTest).
|
||||
pub fn forceTscClocksourceForTest() void {
|
||||
apic.forceTscClocksourceForTest();
|
||||
}
|
||||
|
||||
/// How many per-AP TSC warp checks completed (for the tsc-sync test).
|
||||
pub fn warpChecksRun() u32 {
|
||||
return apic.warpChecksRun();
|
||||
}
|
||||
|
||||
/// Unmask maskable interrupts (`sti`) so device interrupts get delivered.
|
||||
pub fn enableInterrupts() void {
|
||||
asm volatile ("sti");
|
||||
|
||||
@@ -148,7 +148,14 @@ pub fn startAp(apic_id: u32, stack_top: usize, percpu: usize, index: usize, cr3:
|
||||
// Wait up to 100 ms for the AP to reach apEntry and set the flag.
|
||||
const deadline = apic.millis() + 100;
|
||||
while (apic.millis() < deadline) {
|
||||
if (@atomicLoad(u32, &ap_alive, .acquire) != 0) return true;
|
||||
if (@atomicLoad(u32, &ap_alive, .acquire) != 0) {
|
||||
// Cross-check this core's TSC against the BSP's before it joins the run
|
||||
// loop: an unsynchronized TSC must be caught before any task can migrate
|
||||
// onto this core and observe time going backward. No-op unless the TSC is
|
||||
// the clocksource (apic.checkWarpSource).
|
||||
apic.checkWarpSource();
|
||||
return true;
|
||||
}
|
||||
asm volatile ("pause");
|
||||
}
|
||||
return false;
|
||||
@@ -177,6 +184,11 @@ fn apEntry(percpu: usize) callconv(.c) noreturn {
|
||||
|
||||
@atomicStore(u32, &ap_alive, 1, .release); // "architecture state up" — BSP is polling this
|
||||
|
||||
// Rendezvous with the BSP for the TSC warp check (no-op unless the TSC is the
|
||||
// clocksource) before joining the run loop, so this core's clock is vetted before
|
||||
// it can run any task.
|
||||
apic.checkWarpTarget();
|
||||
|
||||
if (secondary_entry) |enterScheduler| enterScheduler(); // joins the run loop
|
||||
while (true) asm volatile ("hlt"); // (only if no entry was registered)
|
||||
}
|
||||
|
||||
@@ -131,7 +131,16 @@ pub fn resourceOf(id: u64, index: u64) ?device_abi.ResourceDescriptor {
|
||||
/// and would otherwise vacuously "fit" anywhere.
|
||||
fn contains(parent: device_abi.ResourceDescriptor, child: device_abi.ResourceDescriptor) bool {
|
||||
if (parent.kind != child.kind) return false;
|
||||
if (child.kind == @intFromEnum(device_abi.ResourceKind.irq)) return parent.start == child.start;
|
||||
if (child.kind == @intFromEnum(device_abi.ResourceKind.irq)) {
|
||||
// Range containment: an interrupt line is still indivisible (a child owns
|
||||
// exactly one GSI), but a parent may own a *range* of lines so a broad
|
||||
// owner — the acpi-tables node, whose firmware names any legacy IRQ —
|
||||
// can contain its children's specific lines. A length-1 parent range is
|
||||
// exactly the old equality rule, so existing single-IRQ parents are
|
||||
// unaffected.
|
||||
const span = if (parent.len == 0) 1 else parent.len;
|
||||
return child.start >= parent.start and child.start < parent.start + span;
|
||||
}
|
||||
if (child.len == 0 or parent.len == 0) return false;
|
||||
// No overflow: a resource that wraps the address space is not containable.
|
||||
const child_end = std.math.add(u64, child.start, child.len) catch return false;
|
||||
@@ -180,6 +189,26 @@ pub fn register(parent_id: u64, owner: u32, descriptor: *const device_abi.Device
|
||||
if (!ok) return error.NotContained;
|
||||
}
|
||||
|
||||
// Idempotent on exact match (docs/device-manager.md): a restarted
|
||||
// registering bus re-registers what it rediscovers, and the table has no
|
||||
// unregister — an identical (class, identity, resources) child under the
|
||||
// same parent returns the existing id instead of appending a duplicate.
|
||||
for (devices[0..count]) |*existing| {
|
||||
if (existing.parent != parent_id) continue;
|
||||
if (existing.class != descriptor.class) continue;
|
||||
if (existing.pci_class != descriptor.pci_class) continue;
|
||||
if (existing.hid_len != descriptor.hid_len) continue;
|
||||
if (!std.mem.eql(u8, existing.hid[0..@intCast(existing.hid_len)], descriptor.hid[0..@intCast(descriptor.hid_len)])) continue;
|
||||
if (existing.resource_count != descriptor.resource_count) continue;
|
||||
var same = true;
|
||||
for (0..@intCast(descriptor.resource_count)) |i| {
|
||||
const a = existing.resources[i];
|
||||
const b = descriptor.resources[i];
|
||||
if (a.kind != b.kind or a.start != b.start or a.len != b.len) same = false;
|
||||
}
|
||||
if (same) return existing.id;
|
||||
}
|
||||
|
||||
var d = std.mem.zeroes(device_abi.DeviceDescriptor);
|
||||
d.id = count;
|
||||
d.parent = parent_id;
|
||||
|
||||
+47
-27
@@ -77,12 +77,12 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
architecture.setFaultHandler(onException);
|
||||
architecture.init();
|
||||
|
||||
status("danos: initialising kernel...\n");
|
||||
status("/system/kernel: initialising kernel...\n");
|
||||
log.write(if (console.present())
|
||||
"danos: framebuffer console online (bootstrap; graphics driver later)\n"
|
||||
"/system/kernel: framebuffer console online (bootstrap; graphics driver later)\n"
|
||||
else
|
||||
"danos: no framebuffer (headless) -> logging to serial/debugcon only\n");
|
||||
log.write("danos: cpu tables online (GDT, IDT, TSS)\n");
|
||||
"/system/kernel: no framebuffer (headless) -> logging to serial/debugcon only\n");
|
||||
log.write("/system/kernel: cpu tables online (GDT, IDT, TSS)\n");
|
||||
log.print(" resolution : {d}x{d}\n", .{ fb.width, fb.height });
|
||||
log.print(" pitch : {d} bytes\n", .{fb.pitch});
|
||||
log.print(" format : {s}\n", .{@tagName(fb.format)});
|
||||
@@ -105,7 +105,7 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
const total_bytes = total_pages * abi.page_size;
|
||||
const gib = 1 << 30;
|
||||
|
||||
log.write("\ndanos: physical memory\n");
|
||||
log.write("\n/system/kernel: physical memory\n");
|
||||
log.print(" total RAM : {d}.{d:0>2} GiB ({d} MiB) - RAM the firmware reported\n", .{ total_bytes / gib, (total_bytes % gib) * 100 / gib, mib(total_pages) });
|
||||
log.print(" usable : {d} MiB - free RAM (incl. reclaimed boot-services memory)\n", .{mib(usable_pages)});
|
||||
log.print(" reserved : {d} MiB - kernel image, boot stack, ACPI, runtime services\n", .{mib(reserved_pages)});
|
||||
@@ -119,7 +119,7 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
// until SMP bring-up; 0 means none was available (we stay uniprocessor).
|
||||
ap_trampoline_page = pmm.allocBelow(0x100000) orelse 0;
|
||||
const s1 = pmm.stats();
|
||||
log.print("\ndanos: frame allocator online\n", .{});
|
||||
log.print("\n/system/kernel: frame allocator online\n", .{});
|
||||
log.print(" free frames: {d} ({d} MiB)\n", .{ s1.free_frames, mib(s1.free_frames) });
|
||||
const f0 = pmm.alloc();
|
||||
const f1 = pmm.alloc();
|
||||
@@ -133,14 +133,14 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
// Switch off the firmware's page tables onto our own (with real permissions).
|
||||
architecture.enablePaging(pmm.alloc, pmm.free, boot_information);
|
||||
log.checkpoint(cp_paging);
|
||||
log.print("\ndanos: paging enabled\n", .{});
|
||||
log.print("\n/system/kernel: paging enabled\n", .{});
|
||||
log.print(" page tables: root = 0x{x:0>16}\n", .{architecture.activePageTable()});
|
||||
log.print(" kernel segs: {d} (mapped with W^X permissions)\n", .{boot_information.kernel_segment_count});
|
||||
|
||||
// Bring up the kernel heap (dynamic allocation), built on the VMM.
|
||||
heap.init();
|
||||
log.checkpoint(cp_heap);
|
||||
log.write("\ndanos: kernel heap online\n");
|
||||
log.write("\n/system/kernel: kernel heap online\n");
|
||||
// Measure the amount of resources the kernel is actually using
|
||||
const s2 = pmm.stats();
|
||||
log.print(" Kernel footprint: {d} KiB\n", .{kib(s1.free_frames - s2.free_frames)});
|
||||
@@ -156,7 +156,7 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
};
|
||||
if (platform.discover(boot_information, heap.allocator(), hal)) |devtree| {
|
||||
var device_tree = devtree;
|
||||
log.write("\ndanos: device discovery online\n");
|
||||
log.write("\n/system/kernel: device discovery online\n");
|
||||
device_tree.dump(log.write);
|
||||
|
||||
// Snapshot the device tree for user-space drivers (device_enumerate/claim/
|
||||
@@ -164,7 +164,7 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
devices_broker.init(&device_tree);
|
||||
if (devices_broker.dropped > 0) {
|
||||
// Otherwise entirely silent: drivers would just never see that hardware.
|
||||
log.print("danos: WARNING {d} device(s) dropped — table full\n", .{devices_broker.dropped});
|
||||
log.print("/system/kernel: WARNING {d} device(s) dropped — table full\n", .{devices_broker.dropped});
|
||||
}
|
||||
|
||||
// Install the device-IRQ trampolines, so a driver's irq_bind has vectors to
|
||||
@@ -173,7 +173,7 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
|
||||
// Power register map extracted from the FADT + AML, for confidence it parsed.
|
||||
const pw = platform.powerInformation();
|
||||
log.write("danos: power\n");
|
||||
log.write("/system/kernel: power\n");
|
||||
log.print(" pm1a_cnt : {s} 0x{x} (width {d})\n", .{ if (pw.pm1a_cnt.mmio) "mmio" else "io", pw.pm1a_cnt.address, pw.pm1a_cnt.width });
|
||||
if (pw.s5) |s| {
|
||||
log.print(" S5 slp_typ : a={d} b={d}\n", .{ s.slp_typ_a, s.slp_typ_b });
|
||||
@@ -221,7 +221,7 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
});
|
||||
if (pinfo.spcr_uart) |u| architecture.serialReconfigure(u.mmio, u.address);
|
||||
|
||||
log.write("danos: platform\n");
|
||||
log.write("/system/kernel: platform\n");
|
||||
log.print(" 8259 PIC : {s}\n", .{if (pinfo.pic_present) "present" else "absent"});
|
||||
log.print(" lapic base : 0x{x}\n", .{pinfo.lapic_base});
|
||||
log.print(" hpet base : 0x{x}\n", .{hpet_base});
|
||||
@@ -237,7 +237,7 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
if (platform.cpusDropped() > 0)
|
||||
log.print(" cpus : WARNING {d} core(s) beyond pool cap dropped\n", .{platform.cpusDropped()});
|
||||
} else |err| {
|
||||
log.print("\ndanos: device discovery failed: {s}\n", .{@errorName(err)});
|
||||
log.print("\n/system/kernel: device discovery failed: {s}\n", .{@errorName(err)});
|
||||
}
|
||||
log.checkpoint(cp_discovery);
|
||||
|
||||
@@ -248,19 +248,39 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
// Register the current context as the first task before enabling preemption.
|
||||
scheduler.init(4);
|
||||
log.checkpoint(cp_scheduler);
|
||||
log.write("\ndanos: scheduler online\n");
|
||||
log.write("\n/system/kernel: scheduler online\n");
|
||||
|
||||
// Start the timer and unmask interrupts — the kernel now has a heartbeat, and
|
||||
// the timer preempts among tasks.
|
||||
architecture.startTimer();
|
||||
architecture.enableInterrupts();
|
||||
log.checkpoint(cp_timer);
|
||||
log.print("danos: timer online ({d} Hz tick; timer clock {d} MHz, clock {d} MHz; calibrated via {s})\n", .{ architecture.timer_hz, architecture.timerClockHz() / 1_000_000, architecture.clockHz() / 1_000_000, architecture.timerCalibrationSource() });
|
||||
log.print("/system/kernel: timer online ({d} Hz tick; timer clock {d} MHz, clock {d} MHz; calibrated via {s})\n", .{ architecture.timer_hz, architecture.timerClockHz() / 1_000_000, architecture.clockHz() / 1_000_000, architecture.timerCalibrationSource() });
|
||||
|
||||
// The tsc-sync test forces the TSC clocksource on before the cores come up, so the
|
||||
// TSC + warp-check path is exercised even under TCG (which won't advertise an
|
||||
// invariant TSC). Inert in a normal build (docs/timers.md).
|
||||
if (build_options.test_case) |tc| {
|
||||
if (std.mem.eql(u8, tc, "tsc-sync")) architecture.forceTscClocksourceForTest();
|
||||
}
|
||||
|
||||
// Wake the other cores (application processors). A no-op on a single-core
|
||||
// machine; on SMP each AP climbs to long mode and reports in (docs/smp.md).
|
||||
// machine; on SMP each AP climbs to long mode and reports in (docs/smp.md). The
|
||||
// per-core TSC warp check rides this: each AP is vetted before it joins the run
|
||||
// loop (docs/timers.md).
|
||||
bringUpSecondaries();
|
||||
|
||||
// Report the monotonic clock's final reliability, now the warp check has run on
|
||||
// every core. On real Intel/AMD this is the invariant, synchronized TSC; a bare
|
||||
// VM (no invariant bit) or a machine whose cores' TSCs skew uses the HPET instead.
|
||||
log.print("/system/kernel: clocksource {s} (TSC invariant: {s}, synchronized: {s})\n", .{
|
||||
architecture.clockSourceName(),
|
||||
if (architecture.clockInvariant()) "yes" else "no",
|
||||
if (architecture.clockSynchronized()) "yes" else "no",
|
||||
});
|
||||
if (!architecture.clockSynchronized())
|
||||
log.write("/system/kernel: WARNING: per-core TSCs are not synchronized; monotonic clock moved off the TSC\n");
|
||||
|
||||
// In a test build (`zig build -Dtest-case=<name>`), run that case and stop.
|
||||
// Normal builds fall through to the idle halt.
|
||||
if (build_options.test_case) |case| {
|
||||
@@ -269,7 +289,7 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
}
|
||||
|
||||
log.checkpoint(cp_running);
|
||||
status("kernel initialised.\n");
|
||||
status("/system/kernel: initialised.\n");
|
||||
|
||||
// Publish the initial-ramdisk so user space can `system_spawn` its bundled
|
||||
// binaries by name. The kernel no longer launches them itself: init is the
|
||||
@@ -282,10 +302,10 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
// manager then discovers the hardware and spawns each driver. init runs on its own
|
||||
// address space, preemptively — this boot context becomes the BSP's idle loop.
|
||||
if (boot_information.init_len != 0) {
|
||||
status("starting /system/services/init...\n");
|
||||
status("/system/kernel: starting /system/services/init...\n");
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.init_base)))[0..boot_information.init_len];
|
||||
process.spawnProcess(image, 4, &.{"/system/services/init"}) catch |err| {
|
||||
statusPrint("/system/services/init failed to load: {s}\n", .{@errorName(err)});
|
||||
statusPrint("/system/kernel: /system/services/init failed to load: {s}\n", .{@errorName(err)});
|
||||
};
|
||||
} else {
|
||||
status("no /system/services/init on the boot volume.\n");
|
||||
@@ -294,7 +314,7 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
// Become the idle task: drop below every real task and halt until an
|
||||
// interrupt. The timer keeps preempting into init and any other work.
|
||||
scheduler.setPriority(0);
|
||||
status("\nkernel idle; user space is running.\n");
|
||||
status("\n/system/kernel: kernel idle; user space is running.\n");
|
||||
architecture.halt();
|
||||
}
|
||||
|
||||
@@ -321,7 +341,7 @@ fn bringUpSecondaries() void {
|
||||
// vector addresses it). It's kept for the system's life — armed only during a
|
||||
// wake, inert (zeroed, non-executable) otherwise — so cores can be re-woken later.
|
||||
if (ap_trampoline_page == 0) {
|
||||
log.write("danos: smp: no low page for the AP trampoline; staying uniprocessor\n");
|
||||
log.write("/system/kernel: smp: no low page for the AP trampoline; staying uniprocessor\n");
|
||||
return;
|
||||
}
|
||||
architecture.setTrampolinePage(ap_trampoline_page);
|
||||
@@ -333,7 +353,7 @@ fn bringUpSecondaries() void {
|
||||
if (std.mem.eql(u8, tc, "smp-retry")) architecture.testFailNextWakes(1);
|
||||
}
|
||||
|
||||
log.print("\ndanos: bringing up {d} application processor(s)\n", .{cores.len - 1});
|
||||
log.print("\n/system/kernel: bringing up {d} application processor(s)\n", .{cores.len - 1});
|
||||
const maximum_wake_attempts = 3; // a core that misses the first INIT-SIPI-SIPI gets retried
|
||||
for (cores[1..], 1..) |core, index| {
|
||||
const stack = heap.allocator().alloc(u8, parameters.kernel_stack_size) catch {
|
||||
@@ -344,7 +364,7 @@ fn bringUpSecondaries() void {
|
||||
// This core's dedicated fault stack — allocated only now that the core is
|
||||
// real, rather than reserved statically for every possible core.
|
||||
const fault_stack = heap.allocator().alloc(u8, architecture.fault_stack_size) catch {
|
||||
log.print(" cpu apic_id {d}: no fault stack; skipped\n", .{core.apic_id});
|
||||
log.print("/system/kernel: cpu apic_id {d}: no fault stack; skipped\n", .{core.apic_id});
|
||||
continue;
|
||||
};
|
||||
architecture.setFaultStack(index, (@intFromPtr(fault_stack.ptr) + fault_stack.len) & ~@as(usize, 15));
|
||||
@@ -353,14 +373,14 @@ fn bringUpSecondaries() void {
|
||||
while (attempt <= maximum_wake_attempts) : (attempt += 1) {
|
||||
if (architecture.startSecondary(core.apic_id, stack_top, @intFromPtr(pc), index)) {
|
||||
pc.online = true;
|
||||
log.print(" cpu apic_id {d}: online (attempt {d})\n", .{ core.apic_id, attempt });
|
||||
log.print("/system/kernel: cpu apic_id {d}: online (attempt {d})\n", .{ core.apic_id, attempt });
|
||||
break;
|
||||
}
|
||||
if (attempt == maximum_wake_attempts)
|
||||
log.print(" cpu apic_id {d}: no response after {d} attempts (parked)\n", .{ core.apic_id, maximum_wake_attempts });
|
||||
log.print("/system/kernel: cpu apic_id {d}: no response after {d} attempts (parked)\n", .{ core.apic_id, maximum_wake_attempts });
|
||||
}
|
||||
}
|
||||
log.print("danos: {d}/{d} cores online\n", .{ scheduler.onlineCount(), cores.len });
|
||||
log.print("/system/kernel: {d}/{d} cores online\n", .{ scheduler.onlineCount(), cores.len });
|
||||
}
|
||||
|
||||
/// A user-facing status line: to the diagnostic `log` *and* the on-screen console
|
||||
@@ -427,7 +447,7 @@ fn exitReasonForVector(vector: u64) abi.ExitReason {
|
||||
|
||||
fn onException(state: *const architecture.CpuState) noreturn {
|
||||
if (architecture.fromUser(state) and scheduler.currentIsUserProcess() and recoverableFault(state.vector)) {
|
||||
statusPrint("\ndanos: process {d} ({s}) killed by {s} (vector {d}) on core {d}\n", .{ scheduler.currentId(), scheduler.current().name(), architecture.exceptionName(state.vector), state.vector, scheduler.currentCpuIndex() });
|
||||
statusPrint("\n/system/kernel: process {d} ({s}) killed by {s} (vector {d}) on core {d}\n", .{ scheduler.currentId(), scheduler.current().name(), architecture.exceptionName(state.vector), state.vector, scheduler.currentCpuIndex() });
|
||||
statusPrint(" error code : 0x{x}\n", .{state.error_code});
|
||||
statusPrint(" IP : 0x{x:0>16}\n", .{architecture.instructionPointer(state)});
|
||||
if (architecture.faultAddress(state)) |address| statusPrint(" fault addr : 0x{x:0>16}\n", .{address});
|
||||
|
||||
@@ -298,6 +298,8 @@ fn systemDeviceEnumerate(state: *architecture.CpuState) void {
|
||||
|
||||
/// device_claim(id) -> 0/-1: take exclusive ownership of a device for this process.
|
||||
fn systemDeviceClaim(state: *architecture.CpuState) void {
|
||||
const claim_flags = sync.enter();
|
||||
defer sync.leave(claim_flags);
|
||||
if (devices_broker.claim(architecture.systemCallArg(state, 0), scheduler.current().id))
|
||||
architecture.setSystemCallResult(state, 0)
|
||||
else
|
||||
@@ -312,10 +314,22 @@ fn systemMmioMap(state: *architecture.CpuState) void {
|
||||
const resource_index = architecture.systemCallArg(state, 1);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
const owner = devices_broker.ownerOf(device_id) orelse return fail(state);
|
||||
if (owner != t.id) return fail(state); // not claimed by this process
|
||||
const r = devices_broker.resourceOf(device_id, resource_index) orelse return fail(state);
|
||||
// Read the broker table under the lock: ring-3 device_register (M19) now
|
||||
// mutates it concurrently on other cores, so a lock-free read here could
|
||||
// see a torn resource (and a torn length used to panic the arithmetic
|
||||
// below on integer overflow).
|
||||
const r = blk: {
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
const owner = devices_broker.ownerOf(device_id) orelse return fail(state);
|
||||
if (owner != t.id) return fail(state); // not claimed by this process
|
||||
break :blk devices_broker.resourceOf(device_id, resource_index) orelse return fail(state);
|
||||
};
|
||||
if (r.kind != @intFromEnum(device_abi.ResourceKind.memory)) return fail(state);
|
||||
// A zero-length or wrapping window is not mappable — fail cleanly rather
|
||||
// than underflow `r.len - 1`.
|
||||
if (r.len == 0) return fail(state);
|
||||
if (@addWithOverflow(r.start, r.len)[1] != 0) return fail(state);
|
||||
|
||||
if (t.device_map_next == 0) t.device_map_next = device_arena_base;
|
||||
const first = r.start & ~@as(u64, page_size - 1);
|
||||
@@ -451,6 +465,11 @@ fn systemDeviceRegister(state: *architecture.CpuState) void {
|
||||
var descriptor: device_abi.DeviceDescriptor = undefined;
|
||||
if (!ipc.copyFromUser(t.aspace, descriptor_ptr, std.mem.asBytes(&descriptor))) return fail(state);
|
||||
|
||||
// Under the big kernel lock: the broker's table is also mutated by the
|
||||
// death sweep (releaseAllOwnedBy) and read by enumerate on other cores —
|
||||
// ring-3 registration (M19) made those genuinely concurrent.
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
const id = devices_broker.register(parent_id, t.id, &descriptor) catch return fail(state);
|
||||
architecture.setSystemCallResult(state, id);
|
||||
}
|
||||
|
||||
@@ -350,8 +350,9 @@ pub fn spawnUserLocked(aspace: u64, entry: u64, user_sp: u64, priority: Priority
|
||||
/// context switch and lock release.
|
||||
fn startUserTask() void {
|
||||
const t = current();
|
||||
var buffer: [96]u8 = undefined;
|
||||
architecture.serialWrite(std.fmt.bufPrint(&buffer, "DBG startUserTask ip=0x{x} sp=0x{x} aspace=0x{x} kstack=0x{x}\n", .{ t.user_ip, t.user_sp, t.aspace, t.kstack_top }) catch "");
|
||||
// No serial chatter here: this runs on every spawn, unserialized against
|
||||
// user-space writes, and its output used to shear concurrent log lines in
|
||||
// half — the largest source of corrupted markers in the QEMU scenarios.
|
||||
architecture.jumpToUser(t.user_ip, t.user_sp); // noreturn
|
||||
}
|
||||
|
||||
|
||||
+548
-180
@@ -102,6 +102,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
|
||||
stressTest();
|
||||
} else if (eql(case, "smp-retry")) {
|
||||
smpRetryTest();
|
||||
} else if (eql(case, "tsc-sync")) {
|
||||
tscSyncTest();
|
||||
} else if (eql(case, "fault-ud")) {
|
||||
faultInvalidOpcode();
|
||||
} else if (eql(case, "fault-pf")) {
|
||||
@@ -138,20 +140,36 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
|
||||
vfsClientDeathTest(boot_information);
|
||||
} else if (eql(case, "signals")) {
|
||||
signalsTest(boot_information);
|
||||
} else if (eql(case, "driver-restart")) {
|
||||
driverRestartTest(boot_information);
|
||||
} else if (eql(case, "usb-report")) {
|
||||
usbReportTest(boot_information);
|
||||
} else if (eql(case, "device-list")) {
|
||||
deviceListTest(boot_information);
|
||||
} else if (eql(case, "pci-scan")) {
|
||||
pciScanTest(boot_information);
|
||||
} else if (eql(case, "acpi-parse")) {
|
||||
acpiParseTest(boot_information);
|
||||
} else if (eql(case, "acpi-report")) {
|
||||
acpiReportTest(boot_information);
|
||||
} else if (eql(case, "acpi-ps2")) {
|
||||
acpiReportTest(boot_information); // same spawn; the harness regex differs
|
||||
} else if (eql(case, "power-button")) {
|
||||
acpiReportTest(boot_information); // boot the manager (spawns the acpi service); harness injects the button
|
||||
} else if (eql(case, "orderly-shutdown")) {
|
||||
orderlyShutdownTest(boot_information);
|
||||
} else if (eql(case, "initial-ramdisk")) {
|
||||
initialRamdiskTest(boot_information);
|
||||
} else if (eql(case, "vfs")) {
|
||||
vfsTest(boot_information);
|
||||
} else if (eql(case, "input")) {
|
||||
inputTest(boot_information);
|
||||
} else if (eql(case, "hpet")) {
|
||||
hpetTest(boot_information);
|
||||
} else if (eql(case, "iopass")) {
|
||||
ioPassTest();
|
||||
} else if (eql(case, "irqfree")) {
|
||||
irqFreeTest();
|
||||
} else if (eql(case, "bus")) {
|
||||
busTest(boot_information);
|
||||
} else if (eql(case, "containment")) {
|
||||
containmentTest();
|
||||
} else if (eql(case, "device-manager")) {
|
||||
deviceManagerTest(boot_information);
|
||||
} else if (eql(case, "poweroff")) {
|
||||
@@ -193,6 +211,13 @@ fn eql(a: []const u8, b: []const u8) bool {
|
||||
return std.mem.eql(u8, a, b);
|
||||
}
|
||||
|
||||
/// Whether the captured last-write buffer *contains* `needle`. Markers are
|
||||
/// matched as substrings, not prefixes, so a service's source-path debug prefix
|
||||
/// (`system/drivers/pci-bus: ...`) still satisfies a marker like `pci-bus: `.
|
||||
fn bufferHas(needle: []const u8) bool {
|
||||
return std.mem.indexOf(u8, process.write_buffer[0..process.write_len], needle) != null;
|
||||
}
|
||||
|
||||
/// Non-destructive checks of the memory map and frame allocator.
|
||||
fn smoke(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: smoke\n", .{});
|
||||
@@ -262,20 +287,56 @@ fn discoveryTest() void {
|
||||
|
||||
// M15: every PCI function now carries its own 4 KiB ECAM configuration space as
|
||||
// resource 0 — the window a driver mmio_maps to walk its capability list (MSI etc).
|
||||
// M19.3: the kernel seeds only the bridge; functions arrive by the ring-3
|
||||
// scan (proven equivalent in pci-scan before the walk retired).
|
||||
var buffer: [64]device_abi.DeviceDescriptor = undefined;
|
||||
const n = @min(devices_broker.enumerate(&buffer), buffer.len);
|
||||
var pci_functions: u32 = 0;
|
||||
var pci_config_ok = true;
|
||||
var bridges: u32 = 0;
|
||||
var bridge_shape_ok = false;
|
||||
for (buffer[0..n]) |d| {
|
||||
if (d.class != @intFromEnum(device_abi.DeviceClass.pci_host_bridge)) continue;
|
||||
bridges += 1;
|
||||
var has_bus_range = false;
|
||||
var has_io = false;
|
||||
var memory_windows: u32 = 0;
|
||||
for (d.resources[0..@intCast(d.resource_count)]) |resource| {
|
||||
if (resource.kind == @intFromEnum(device_abi.ResourceKind.bus_range)) has_bus_range = true;
|
||||
if (resource.kind == @intFromEnum(device_abi.ResourceKind.io_port)) has_io = true;
|
||||
if (resource.kind == @intFromEnum(device_abi.ResourceKind.memory)) memory_windows += 1;
|
||||
}
|
||||
// ECAM plus at least one MMIO aperture, the bus range, the I/O window.
|
||||
if (has_bus_range and has_io and memory_windows >= 2) bridge_shape_ok = true;
|
||||
}
|
||||
check("a PCI host bridge was seeded (MCFG)", bridges >= 1);
|
||||
check("the bridge carries ECAM, apertures, bus range, and the I/O window", bridge_shape_ok);
|
||||
|
||||
// M19.0: every PCI memory resource (config slice and BARs alike) must be
|
||||
// contained in one of its parent bridge's windows — the aperture derivation
|
||||
// from the memory map is what makes a future user-space device_register of
|
||||
// these functions pass containment. This is the assert that catches a
|
||||
// too-coarse hole computation before M19.2 would.
|
||||
var bars_contained = true;
|
||||
for (buffer[0..n]) |d| {
|
||||
if (d.class != @intFromEnum(device_abi.DeviceClass.pci_device)) continue;
|
||||
pci_functions += 1;
|
||||
const has_config = d.resource_count >= 1 and
|
||||
d.resources[0].kind == @intFromEnum(device_abi.ResourceKind.memory) and
|
||||
d.resources[0].len == abi.page_size;
|
||||
if (!has_config) pci_config_ok = false;
|
||||
if (d.parent >= n) {
|
||||
bars_contained = false;
|
||||
continue;
|
||||
}
|
||||
const bridge = buffer[@intCast(d.parent)];
|
||||
for (d.resources[0..@intCast(d.resource_count)]) |r| {
|
||||
if (r.kind != @intFromEnum(device_abi.ResourceKind.memory)) continue;
|
||||
var inside = false;
|
||||
for (bridge.resources[0..@intCast(bridge.resource_count)]) |w| {
|
||||
if (w.kind != @intFromEnum(device_abi.ResourceKind.memory)) continue;
|
||||
if (r.start >= w.start and r.start + r.len <= w.start + w.len) inside = true;
|
||||
}
|
||||
if (!inside) {
|
||||
bars_contained = false;
|
||||
log(" escaping BAR: 0x{x}+0x{x} on device {d}\n", .{ r.start, r.len, d.id });
|
||||
}
|
||||
}
|
||||
}
|
||||
check("PCI functions were enumerated (MCFG/ECAM)", pci_functions >= 1);
|
||||
check("each PCI function exposes its ECAM config space as resource 0", pci_config_ok);
|
||||
check("every PCI BAR lies inside a bridge aperture (M19.0)", bars_contained);
|
||||
|
||||
result();
|
||||
}
|
||||
@@ -1086,12 +1147,18 @@ fn ioPortTest() void {
|
||||
var buffer: [64]device_abi.DeviceDescriptor = undefined;
|
||||
const n = @min(devices_broker.enumerate(&buffer), buffer.len);
|
||||
|
||||
// Post-M20.3 the PS/2 node is registered at runtime by the ring-3 acpi
|
||||
// service, so it is absent from this boot snapshot. Exercise the same
|
||||
// io_port claim/resolve mechanism against the acpi-tables node's broad I/O
|
||||
// grant — the window that now carries port authority (the service uses it
|
||||
// for exactly this). The PS/2 status port 0x64 is offset 0x64 within it.
|
||||
var found_id: ?u64 = null;
|
||||
var found_res: u64 = 0;
|
||||
outer: for (buffer[0..n]) |d| {
|
||||
if (d.class != @intFromEnum(device_abi.DeviceClass.acpi_tables)) continue;
|
||||
for (0..d.resource_count) |ri| {
|
||||
const r = d.resources[ri];
|
||||
if (r.kind == @intFromEnum(device_abi.ResourceKind.io_port) and r.start == 0x64 and r.len >= 1) {
|
||||
if (r.kind == @intFromEnum(device_abi.ResourceKind.io_port) and r.start == 0 and r.len > 0x64) {
|
||||
found_id = d.id;
|
||||
found_res = ri;
|
||||
break :outer;
|
||||
@@ -1099,18 +1166,18 @@ fn ioPortTest() void {
|
||||
}
|
||||
}
|
||||
const id = found_id orelse {
|
||||
check("discovered the PS/2 status port (io_port 0x64)", false);
|
||||
check("discovered the acpi-tables I/O window", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
check("discovered the PS/2 status port (io_port 0x64)", true);
|
||||
check("discovered the acpi-tables I/O window", true);
|
||||
|
||||
const me = scheduler.current();
|
||||
check("claimed the io_port device", devices_broker.claim(id, me.id));
|
||||
check("an in-range access resolves to port 0x64", process.resolveIoPort(me, id, found_res, 0, 1) == 0x64);
|
||||
check("an over-wide access is refused", process.resolveIoPort(me, id, found_res, 0, 2) == null);
|
||||
check("an out-of-range offset is refused", process.resolveIoPort(me, id, found_res, 1, 1) == null);
|
||||
check("an unclaimed device id is refused", process.resolveIoPort(me, 0xDEAD_BEEF, found_res, 0, 1) == null);
|
||||
check("an in-range access resolves to port 0x64", process.resolveIoPort(me, id, found_res, 0x64, 1) == 0x64);
|
||||
check("a 4-byte access at the last port is refused", process.resolveIoPort(me, id, found_res, 0xFFFF, 4) == null);
|
||||
check("an out-of-range offset is refused", process.resolveIoPort(me, id, found_res, 0x10000, 1) == null);
|
||||
check("an unclaimed device id is refused", process.resolveIoPort(me, 0xDEAD_BEEF, found_res, 0x64, 1) == null);
|
||||
|
||||
// The kernel actually issues the `in`. Reaching this line at all proves it didn't
|
||||
// fault; a width-1 read must return a single byte.
|
||||
@@ -1144,6 +1211,35 @@ fn clockTest() void {
|
||||
result();
|
||||
}
|
||||
|
||||
/// The TSC clocksource + cross-core warp check (`-smp 4`). This is the real
|
||||
/// Intel/AMD / KVM path — an invariant, synchronized TSC. TCG (the only x86
|
||||
/// accelerator on an Apple-Silicon host) won't advertise an invariant TSC, so the
|
||||
/// boot forces the TSC clocksource on (kernel.zig, gated on this case) to exercise
|
||||
/// the machinery: the kernel must run the per-AP warp check as each core came up,
|
||||
/// find the cores' TSCs synchronized, and keep the clock on the TSC (no HPET
|
||||
/// fallback). The default suite (no force) exercises the HPET fallback instead.
|
||||
fn tscSyncTest() void {
|
||||
log("DANOS-TEST-BEGIN: tsc-sync\n", .{});
|
||||
check("clocksource is the TSC (forced-invariant path)", eql(architecture.clockSourceName(), "tsc"));
|
||||
check("the cross-core warp check ran on the APs", architecture.warpChecksRun() >= 1);
|
||||
check("per-core TSCs synchronized (no warp, no HPET fallback)", architecture.clockSynchronized());
|
||||
|
||||
// A warp that slipped past the bring-up check would surface as a backward reading.
|
||||
var last = architecture.nanos();
|
||||
var monotonic = true;
|
||||
var advanced = false;
|
||||
var i: u32 = 0;
|
||||
while (i < 1_000_000) : (i += 1) {
|
||||
const t = architecture.nanos();
|
||||
if (t < last) monotonic = false;
|
||||
if (t > last) advanced = true;
|
||||
last = t;
|
||||
}
|
||||
check("monotonic clock advanced", advanced);
|
||||
check("monotonic clock never ran backward", monotonic);
|
||||
result();
|
||||
}
|
||||
|
||||
var proc_worker_run: bool = true;
|
||||
var proc_worker_ran: bool = false;
|
||||
|
||||
@@ -1320,7 +1416,7 @@ fn initTest(boot_information: *const BootInformation) void {
|
||||
scheduler.setPriority(4);
|
||||
|
||||
const prefix = "init: heartbeat";
|
||||
const beat_ok = process.write_len >= prefix.len and eql(process.write_buffer[0..prefix.len], prefix);
|
||||
const beat_ok = bufferHas(prefix);
|
||||
check("init produced repeated heartbeats (>=2)", process.write_count >= 2);
|
||||
check("heartbeat text arrived intact", beat_ok);
|
||||
check("heartbeats came from user mode (CPL 3)", process.write_from_user);
|
||||
@@ -1582,11 +1678,11 @@ fn vfsClientDeathTest(boot_information: *const BootInformation) void {
|
||||
scheduler.setPriority(1);
|
||||
var deadline = architecture.millis() + 10000;
|
||||
while (architecture.millis() < deadline) {
|
||||
if (process.write_len >= parked.len and eql(process.write_buffer[0..parked.len], parked)) break;
|
||||
if (bufferHas(parked)) break;
|
||||
scheduler.yield();
|
||||
}
|
||||
scheduler.setPriority(4);
|
||||
check("client parked holding an open handle", process.write_len >= parked.len and eql(process.write_buffer[0..parked.len], parked));
|
||||
check("client parked holding an open handle", bufferHas(parked));
|
||||
|
||||
check("the kill is accepted", process.killProcess(me, client) == 0);
|
||||
var badge: u64 = 0;
|
||||
@@ -1599,11 +1695,11 @@ fn vfsClientDeathTest(boot_information: *const BootInformation) void {
|
||||
scheduler.setPriority(1);
|
||||
deadline = architecture.millis() + 10000;
|
||||
while (architecture.millis() < deadline) {
|
||||
if (process.write_len >= released.len and eql(process.write_buffer[0..released.len], released)) break;
|
||||
if (bufferHas(released)) break;
|
||||
scheduler.yield();
|
||||
}
|
||||
scheduler.setPriority(4);
|
||||
check("the VFS released the dead client's handle", process.write_len >= released.len and eql(process.write_buffer[0..released.len], released));
|
||||
check("the VFS released the dead client's handle", bufferHas(released));
|
||||
result();
|
||||
}
|
||||
|
||||
@@ -1645,8 +1741,8 @@ fn signalsTest(boot_information: *const BootInformation) void {
|
||||
var saw_pass = false;
|
||||
var saw_fail = false;
|
||||
while (architecture.millis() < deadline and !saw_pass and !saw_fail) {
|
||||
if (process.write_len >= pass_marker.len and eql(process.write_buffer[0..pass_marker.len], pass_marker)) saw_pass = true;
|
||||
if (process.write_len >= fail_marker.len and eql(process.write_buffer[0..fail_marker.len], fail_marker)) saw_fail = true;
|
||||
if (bufferHas(pass_marker)) saw_pass = true;
|
||||
if (bufferHas(fail_marker)) saw_fail = true;
|
||||
scheduler.yield();
|
||||
}
|
||||
scheduler.setPriority(4);
|
||||
@@ -1654,6 +1750,314 @@ fn signalsTest(boot_information: *const BootInformation) void {
|
||||
result();
|
||||
}
|
||||
|
||||
/// M18.1: the device manager's restart machinery, end to end. In test-restart
|
||||
/// mode the manager also supervises crash-test: a fixture that claims device 0,
|
||||
/// hellos, and faults. The scenario asserts three markers in order — the real
|
||||
/// xHCI driver hellos clean and stays; crash-test is restarted with backoff
|
||||
/// (each respawn re-claiming the device the dead instance held, M17.1 through
|
||||
/// the manager's path); the crash loop caps and the manager gives up.
|
||||
fn driverRestartTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: driver-restart\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over an initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
|
||||
process.setInitialRamdisk(image); // the manager system_spawns drivers by name
|
||||
process.write_count = 0;
|
||||
var manager: u32 = 0;
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
if (!eql(item.name, "device-manager")) continue;
|
||||
manager = process.spawnProcessSupervised(item.blob, 4, &.{ "device-manager", "test-restart" }, scheduler.currentId(), null) catch 0;
|
||||
break;
|
||||
}
|
||||
check("device-manager spawned in test-restart mode", manager != 0);
|
||||
// The assertions live in the harness: its expect regex requires, in order,
|
||||
// the xHCI hello ack, a crash-test restart, and the crash-loop cap — read
|
||||
// from the whole serial capture, immune to the transient-line races a
|
||||
// write_buffer poll would have here (many processes log concurrently).
|
||||
result();
|
||||
}
|
||||
|
||||
/// M18.2: bus tree reports, end to end. The manager (test-usb-restart mode)
|
||||
/// spawns the xHCI driver; the driver maps its BAR, scans the root-hub ports,
|
||||
/// and reports the two QEMU devices; the manager mirrors them, kills the
|
||||
/// reporter (the test trigger), prunes both children, restarts the driver with
|
||||
/// backoff, and the respawned instance re-claims, re-scans, and re-reports.
|
||||
/// The harness's ordered expect regex is the assertion; this test only
|
||||
/// orchestrates the spawn.
|
||||
fn usbReportTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: usb-report\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over an initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
|
||||
process.setInitialRamdisk(image);
|
||||
var manager: u32 = 0;
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
if (!eql(item.name, "device-manager")) continue;
|
||||
manager = process.spawnProcessSupervised(item.blob, 4, &.{ "device-manager", "test-usb-restart" }, scheduler.currentId(), null) catch 0;
|
||||
break;
|
||||
}
|
||||
check("device-manager spawned in test-usb-restart mode", manager != 0);
|
||||
result();
|
||||
}
|
||||
|
||||
/// M18.3: the application surface. device-list enumerates the manager's tree
|
||||
/// over IPC, subscribes with its endpoint as a capability, and prints every
|
||||
/// published event; the manager's delayed test-kill of the reporter produces a
|
||||
/// removed/added storm the subscriber must observe. The harness's ordered
|
||||
/// expect regex is the assertion.
|
||||
fn deviceListTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: device-list\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over an initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
|
||||
process.setInitialRamdisk(image);
|
||||
var manager: u32 = 0;
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
if (!eql(item.name, "device-manager")) continue;
|
||||
manager = process.spawnProcessSupervised(item.blob, 4, &.{ "device-manager", "test-usb-restart" }, scheduler.currentId(), null) catch 0;
|
||||
break;
|
||||
}
|
||||
check("device-manager spawned in test-usb-restart mode", manager != 0);
|
||||
check("device-list spawned", spawnNamed(rd, "device-list"));
|
||||
result();
|
||||
}
|
||||
|
||||
/// M19.1: the ring-3 PCI scan agrees with the kernel's. The manager spawns
|
||||
/// pci-bus for the host bridge; the driver walks the same ECAM window through
|
||||
/// its mmio_map grant and must find exactly the functions the kernel's own
|
||||
/// enumeration recorded — the equivalence that licenses retiring the kernel
|
||||
/// walk in M19.3.
|
||||
fn pciScanTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: pci-scan\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over an initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
|
||||
// Post-flip (M19.3) ground truth: the kernel no longer enumerates PCI
|
||||
// functions, so equivalence inverts — the broker's function count after
|
||||
// the scan must equal what the driver itself reported finding.
|
||||
var buffer: [64]device_abi.DeviceDescriptor = undefined;
|
||||
const n = @min(devices_broker.enumerate(&buffer), buffer.len);
|
||||
var boot_pci: u32 = 0;
|
||||
for (buffer[0..n]) |d| {
|
||||
if (d.class == @intFromEnum(device_abi.DeviceClass.pci_device)) boot_pci += 1;
|
||||
}
|
||||
check("the kernel seeded no PCI functions (the walk retired)", boot_pci == 0);
|
||||
|
||||
process.setInitialRamdisk(image);
|
||||
process.write_count = 0;
|
||||
var manager: u32 = 0;
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
if (!eql(item.name, "device-manager")) continue;
|
||||
manager = process.spawnProcessSupervised(item.blob, 4, &.{ "device-manager", "test-pci-restart" }, scheduler.currentId(), null) catch 0;
|
||||
break;
|
||||
}
|
||||
check("device-manager spawned (test-pci-restart mode)", manager != 0);
|
||||
|
||||
// First scan: wait for the driver's count line and parse the number.
|
||||
const count_prefix = "pci-bus: ";
|
||||
const count_suffix = " functions found";
|
||||
var reported: u32 = 0;
|
||||
scheduler.setPriority(1);
|
||||
var deadline = architecture.millis() + 15000;
|
||||
while (architecture.millis() < deadline and reported == 0) {
|
||||
const line = process.write_buffer[0..process.write_len];
|
||||
if (std.mem.indexOf(u8, line, count_prefix)) |start| {
|
||||
if (std.mem.indexOf(u8, line, count_suffix)) |digits_end| {
|
||||
reported = std.fmt.parseInt(u32, line[start + count_prefix.len .. digits_end], 10) catch 0;
|
||||
}
|
||||
}
|
||||
scheduler.yield();
|
||||
}
|
||||
scheduler.setPriority(4);
|
||||
check("the ring-3 scan reported a function count", reported >= 1);
|
||||
|
||||
// Every reported function was registered: the broker holds exactly them.
|
||||
var registered: [64]device_abi.DeviceDescriptor = undefined;
|
||||
const r = @min(devices_broker.enumerate(®istered), registered.len);
|
||||
var registered_pci: u32 = 0;
|
||||
for (registered[0..r]) |d| {
|
||||
if (d.class == @intFromEnum(device_abi.DeviceClass.pci_device)) registered_pci += 1;
|
||||
}
|
||||
check("the broker holds exactly the reported functions", registered_pci == reported);
|
||||
const kernel_count = reported; // the no-duplicate check below reuses it
|
||||
|
||||
// The restart drill: the manager kills pci-bus after its reports; the
|
||||
// respawn re-claims, re-scans, and re-registers.
|
||||
const restart_marker = "device-manager: restarting pci-bus";
|
||||
scheduler.setPriority(1);
|
||||
deadline = architecture.millis() + 15000;
|
||||
var restarted = false;
|
||||
while (architecture.millis() < deadline and !restarted) {
|
||||
if (bufferHas(restart_marker)) restarted = true;
|
||||
scheduler.yield();
|
||||
}
|
||||
scheduler.setPriority(4);
|
||||
check("the manager restarted pci-bus", restarted);
|
||||
|
||||
var marker_buffer: [48]u8 = undefined;
|
||||
const marker = std.fmt.bufPrint(&marker_buffer, "pci-bus: {d} functions found", .{reported}) catch "";
|
||||
scheduler.setPriority(1);
|
||||
deadline = architecture.millis() + 15000;
|
||||
var seen = false;
|
||||
while (architecture.millis() < deadline and !seen) {
|
||||
if (bufferHas(marker)) seen = true;
|
||||
scheduler.yield();
|
||||
}
|
||||
scheduler.setPriority(4);
|
||||
check("the respawned scan reported the same count", seen);
|
||||
|
||||
// No duplicates: the registrations deduped against the kernel's own nodes
|
||||
// on the first pass, and against themselves on the second.
|
||||
var after: [64]device_abi.DeviceDescriptor = undefined;
|
||||
const m = @min(devices_broker.enumerate(&after), after.len);
|
||||
var after_count: u32 = 0;
|
||||
for (after[0..m]) |d| {
|
||||
if (d.class == @intFromEnum(device_abi.DeviceClass.pci_device)) after_count += 1;
|
||||
}
|
||||
check("no duplicate PCI nodes after register + restart + re-register", after_count == kernel_count);
|
||||
result();
|
||||
}
|
||||
|
||||
/// M21.3 capstone: orderly shutdown. Boot init with the initial-ramdisk
|
||||
/// published, so init spawns the full service tree (vfs, input, device-manager
|
||||
/// -> discovery/acpi); the harness injects a real power-button event via QMP;
|
||||
/// the acpi service publishes it; init runs the stop sequence over its children
|
||||
/// and asks the power service for S5; the machine powers off (QEMU exits). The
|
||||
/// kernel test only spawns init — the ordered chain is the harness assertion.
|
||||
fn orderlyShutdownTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: orderly-shutdown\n", .{});
|
||||
if (boot_information.init_len == 0 or boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over init and the initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const ramdisk = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
process.setInitialRamdisk(ramdisk);
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.init_base)))[0..boot_information.init_len];
|
||||
const spawned = if (process.spawnProcess(image, 4, &.{"/system/services/init"})) true else |_| false;
|
||||
check("init spawned as PID root of user space", spawned);
|
||||
result();
|
||||
}
|
||||
|
||||
/// M20.2: the acpi service registers + reports its _HID devices. Boot normally
|
||||
/// (the manager spawns discovery); the harness's expect regex requires the two
|
||||
/// PS/2 nodes among the service's report lines, each with its _CRS resources —
|
||||
/// the ring-3 _CRS/_STA evaluation working end to end. The kernel test only
|
||||
/// starts the manager.
|
||||
fn acpiReportTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: acpi-report\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over an initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
process.setInitialRamdisk(image);
|
||||
var spawned = false;
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
if (!eql(item.name, "device-manager")) continue;
|
||||
_ = process.spawnProcessSupervised(item.blob, 4, &.{"device-manager"}, scheduler.currentId(), null) catch 0;
|
||||
spawned = true;
|
||||
break;
|
||||
}
|
||||
check("device-manager spawned", spawned);
|
||||
result();
|
||||
}
|
||||
|
||||
/// M20.1: the ring-3 AML parse agrees with the kernel's. The manager spawns
|
||||
/// the discovery service (the acpi build variant); it claims the acpi-tables
|
||||
/// node, maps the blobs, parses them, and logs its Device count — which must
|
||||
/// equal what the kernel's own parse produced (the equivalence that licenses
|
||||
/// retiring the kernel's device build in M20.3).
|
||||
fn acpiParseTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: acpi-parse\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over an initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
|
||||
// The kernel's own count, from the namespace it already built for \_S5.
|
||||
const kernel_devices = platform.amlDeviceCount();
|
||||
check("the kernel namespace has devices to compare against", kernel_devices >= 1);
|
||||
|
||||
// Spawn the discovery service directly with that count as argv: it parses
|
||||
// the same blobs in ring 3 and self-verifies, printing "acpi-parse: ok" iff
|
||||
// the counts match. The harness's expect regex is that marker — deterministic,
|
||||
// no racing the shared serial buffer.
|
||||
process.setInitialRamdisk(image);
|
||||
var count_text: [16]u8 = undefined;
|
||||
const count_arg = std.fmt.bufPrint(&count_text, "{d}", .{kernel_devices}) catch "0";
|
||||
var spawned = false;
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
if (!eql(item.name, "discovery")) continue;
|
||||
_ = process.spawnProcessSupervised(item.blob, 4, &.{ "discovery", count_arg }, scheduler.currentId(), null) catch 0;
|
||||
spawned = true;
|
||||
break;
|
||||
}
|
||||
check("discovery service spawned", spawned);
|
||||
result();
|
||||
}
|
||||
|
||||
/// The whole user-side surface at once: spawn process-test's supervisor role,
|
||||
/// which — entirely from ring 3 — creates an exit endpoint, spawns its two
|
||||
/// children supervised, sees them in process_enumerate, kills them (one blocked,
|
||||
@@ -1690,12 +2094,12 @@ fn supervisionTest(boot_information: *const BootInformation) void {
|
||||
scheduler.setPriority(1);
|
||||
const deadline = architecture.millis() + 10000;
|
||||
while (architecture.millis() < deadline) {
|
||||
if (process.write_len >= marker.len and eql(process.write_buffer[0..marker.len], marker)) break;
|
||||
if (bufferHas(marker)) break;
|
||||
scheduler.yield();
|
||||
}
|
||||
scheduler.setPriority(4);
|
||||
|
||||
const ok = process.write_len >= marker.len and eql(process.write_buffer[0..marker.len], marker);
|
||||
const ok = bufferHas(marker);
|
||||
if (!ok and process.write_len > 0) log("DANOS-SUPERVISION: got \"{s}\"\n", .{process.write_buffer[0..process.write_len]});
|
||||
check("the supervisor completed every step (spawn/list/kill/notify)", ok);
|
||||
check("it ran in user mode (CPL 3)", process.write_from_user);
|
||||
@@ -1775,12 +2179,12 @@ fn vfsTest(boot_information: *const BootInformation) void {
|
||||
scheduler.setPriority(1);
|
||||
const deadline = architecture.millis() + 10000;
|
||||
while (architecture.millis() < deadline) {
|
||||
if (process.write_len >= prefix.len and eql(process.write_buffer[0..prefix.len], prefix) and process.write_count >= 2) break;
|
||||
if (bufferHas(prefix) and process.write_count >= 2) break;
|
||||
scheduler.yield();
|
||||
}
|
||||
scheduler.setPriority(4);
|
||||
|
||||
const ok = process.write_len >= prefix.len and eql(process.write_buffer[0..prefix.len], prefix);
|
||||
const ok = bufferHas(prefix);
|
||||
check("client completed the VFS round trip (open/write/read matched)", ok);
|
||||
check("the round trip ran repeatedly (server stays up)", process.write_count >= 2);
|
||||
check("client syscalls came from user mode (CPL 3)", process.write_from_user);
|
||||
@@ -1820,12 +2224,12 @@ fn inputTest(boot_information: *const BootInformation) void {
|
||||
scheduler.setPriority(1);
|
||||
const deadline = architecture.millis() + 12000;
|
||||
while (architecture.millis() < deadline) {
|
||||
if (process.write_len >= prefix.len and eql(process.write_buffer[0..prefix.len], prefix) and process.write_count >= 2) break;
|
||||
if (bufferHas(prefix) and process.write_count >= 2) break;
|
||||
scheduler.yield();
|
||||
}
|
||||
scheduler.setPriority(4);
|
||||
|
||||
const ok = process.write_len >= prefix.len and eql(process.write_buffer[0..prefix.len], prefix);
|
||||
const ok = bufferHas(prefix);
|
||||
check("a subscriber received a broadcast key event over IPC (source -> service -> subscriber)", ok);
|
||||
check("events kept flowing (service + async send stay up)", process.write_count >= 2);
|
||||
check("client syscalls came from user mode (CPL 3)", process.write_from_user);
|
||||
@@ -1883,68 +2287,7 @@ fn spawnNamed(rd: initial_ramdisk.Reader, name: []const u8) bool {
|
||||
return false;
|
||||
}
|
||||
|
||||
/// IO passthrough + IRQ-as-IPC: a user-space driver drives real hardware and is
|
||||
/// *woken by it*. Spawn hpet, which claims the HPET, maps its registers into its
|
||||
/// own ring-3 address space, arms a level-triggered comparator, binds the interrupt
|
||||
/// to an IPC endpoint, and then blocks. It prints "hpet: ok" only after being woken
|
||||
/// `target_ticks` times — it cannot reach that line by polling, because the loop's
|
||||
/// only exit is through `replyWait` returning a notification badge.
|
||||
///
|
||||
/// The interesting assertion is the last one, which doesn't trust hpet at all: it
|
||||
/// reads the I/O APIC's redirection entry back and checks the kernel really routed
|
||||
/// the line (our vector, level-triggered) and really left it unmasked after the
|
||||
/// driver's final `irq_ack`. hpet disables its comparator on the last interrupt, so
|
||||
/// that state is quiescent and not a race.
|
||||
fn hpetTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: hpet\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over an initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
|
||||
process.write_count = 0;
|
||||
process.write_from_user = false;
|
||||
check("hpet spawned from the initial_ramdisk", spawnNamed(rd, "hpet"));
|
||||
|
||||
const prefix = "hpet: ok";
|
||||
scheduler.setPriority(1);
|
||||
const deadline = architecture.millis() + 10000;
|
||||
while (architecture.millis() < deadline) {
|
||||
if (process.write_len >= prefix.len and eql(process.write_buffer[0..prefix.len], prefix) and process.write_count >= 2) break;
|
||||
scheduler.yield();
|
||||
}
|
||||
scheduler.setPriority(4);
|
||||
|
||||
const ok = process.write_len >= prefix.len and eql(process.write_buffer[0..prefix.len], prefix);
|
||||
check("user driver mapped HPET MMIO and was woken by its interrupt", ok);
|
||||
check("driver syscalls came from user mode (CPL 3)", process.write_from_user);
|
||||
check("kernel routed and re-armed the HPET's line at the I/O APIC", hpetRouteOk());
|
||||
result();
|
||||
}
|
||||
|
||||
/// Read back the I/O APIC redirection entry for the HPET's GSI and confirm the
|
||||
/// kernel programmed it: a vector in the device window, level-triggered, unmasked.
|
||||
/// Independent of anything the driver reported about itself.
|
||||
fn hpetRouteOk() bool {
|
||||
const gsi = hpetGsi() orelse return false;
|
||||
if (gsi >= architecture.irqRouteCount()) return false;
|
||||
const low = architecture.irqRouteRaw(gsi); // entry index == GSI (this I/O APIC's gsi_base is 0)
|
||||
const vector: u8 = @truncate(low & 0xFF);
|
||||
const masked = low & (1 << 16) != 0;
|
||||
const level = low & (1 << 15) != 0;
|
||||
return vector >= architecture.irq_vector_base and
|
||||
vector < architecture.irq_vector_base + architecture.irq_vector_count and
|
||||
level and !masked;
|
||||
}
|
||||
|
||||
/// The GSI discovery recorded for the HPET, from the same device table the driver saw.
|
||||
/// The GSI discovery recorded for the HPET, from the same device table drivers see.
|
||||
fn hpetGsi() ?u32 {
|
||||
var buffer: [16]device_abi.DeviceDescriptor = undefined;
|
||||
const n = @min(devices_broker.enumerate(&buffer), buffer.len);
|
||||
@@ -1959,58 +2302,87 @@ fn hpetGsi() ?u32 {
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Bus driver: a user process claims a device that contains other devices, enumerates
|
||||
/// them from the hardware, and publishes each as a child via `device_register` — the
|
||||
/// primitive a PCI bridge or USB hub driver is built from.
|
||||
///
|
||||
/// `bus` treats the HPET's register block as a bus and its comparators as children,
|
||||
/// giving each a 0x20 sub-window. It checks its own work (children come back from the
|
||||
/// table with the right parent and a strictly narrower window) and, importantly, that
|
||||
/// the kernel **refuses** a child whose window escapes the parent's — without that,
|
||||
/// `device_register` would be a system_call for mapping arbitrary physical memory. It prints
|
||||
/// "bus: ok" only if all of that holds.
|
||||
///
|
||||
/// The kernel-side check here is the one bus can't make: that the children really did
|
||||
/// land in the device table with the containment invariant intact.
|
||||
fn busTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: bus\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over an initial_ramdisk", false);
|
||||
/// `device_register` containment — a kernel security property, tested directly against
|
||||
/// the broker (no user-space demo driver). A bus driver publishes children of a device
|
||||
/// it owns; the kernel must **refuse** any child whose resource escapes the parent's
|
||||
/// grant, or `device_register` would become a system call for mapping arbitrary physical
|
||||
/// memory. This is the property the old `bus` demo driver proved end-to-end; with the
|
||||
/// demo gone, the property is asserted where it lives — in the kernel. Also checks the
|
||||
/// idempotence rule (M19.0): re-registering an identical child returns the same id
|
||||
/// instead of appending a duplicate.
|
||||
fn containmentTest() void {
|
||||
log("DANOS-TEST-BEGIN: containment\n", .{});
|
||||
|
||||
const me = scheduler.currentId();
|
||||
var buffer: [64]device_abi.DeviceDescriptor = undefined;
|
||||
check("device tree is seeded", devices_broker.enumerate(&buffer) >= 1);
|
||||
|
||||
// The kernel-seeded HPET timer block is a device with a memory resource — a natural
|
||||
// parent to publish sub-window children under, as a PCI bridge or USB hub would.
|
||||
const parent_id = hpetDeviceId() orelse {
|
||||
check("found a device with a memory window to parent children under", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
const parent = buffer[@intCast(parent_id)];
|
||||
var window: ?device_abi.ResourceDescriptor = null;
|
||||
for (0..parent.resource_count) |j| {
|
||||
if (parent.resources[j].kind == @intFromEnum(device_abi.ResourceKind.memory)) window = parent.resources[j];
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
const parent_window = window orelse {
|
||||
check("parent exposes a memory window", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
|
||||
process.write_count = 0;
|
||||
process.write_from_user = false;
|
||||
check("bus spawned from the initial_ramdisk", spawnNamed(rd, "bus"));
|
||||
|
||||
const prefix = "bus: ok";
|
||||
scheduler.setPriority(1);
|
||||
const deadline = architecture.millis() + 10000;
|
||||
while (architecture.millis() < deadline) {
|
||||
if (process.write_len >= prefix.len and eql(process.write_buffer[0..prefix.len], prefix)) break;
|
||||
scheduler.yield();
|
||||
if (devices_broker.ownerOf(parent_id) != null) {
|
||||
check("parent device was unclaimed at the start of the test", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
scheduler.setPriority(4);
|
||||
check("claimed the parent device", devices_broker.claim(parent_id, me));
|
||||
defer devices_broker.releaseAllOwnedBy(me);
|
||||
|
||||
// A child whose window lies inside the parent's is accepted.
|
||||
var fits = childDescriptor("cfit", parent_window.start, 0x20);
|
||||
const before = devices_broker.enumerate(&buffer);
|
||||
const good = devices_broker.register(parent_id, me, &fits) catch 0;
|
||||
check("a contained child is registered", good != 0);
|
||||
check("the contained child was appended to the table", devices_broker.enumerate(&buffer) == before + 1);
|
||||
|
||||
// A child whose window escapes the parent's is refused with NotContained.
|
||||
var escapes = childDescriptor("cesc", parent_window.start, parent_window.len + 0x1000);
|
||||
const refused = if (devices_broker.register(parent_id, me, &escapes)) |_| false else |err| err == error.NotContained;
|
||||
check("an out-of-window child is refused (NotContained)", refused);
|
||||
check("the refused child left the table unchanged", devices_broker.enumerate(&buffer) == before + 1);
|
||||
|
||||
// Idempotent on exact match: re-registering the accepted child returns its id and
|
||||
// appends nothing (M19.0 — a restarted bus re-reports what it rediscovers).
|
||||
const again = devices_broker.register(parent_id, me, &fits) catch 0;
|
||||
check("re-registering an identical child returns the same id", again != 0 and again == good);
|
||||
check("re-registering grew nothing", devices_broker.enumerate(&buffer) == before + 1);
|
||||
|
||||
const ok = process.write_len >= prefix.len and eql(process.write_buffer[0..prefix.len], prefix);
|
||||
check("bus driver published children and the kernel refused an out-of-window one", ok);
|
||||
check("driver syscalls came from user mode (CPL 3)", process.write_from_user);
|
||||
check("every registered child is contained in its parent", childrenContained());
|
||||
result();
|
||||
}
|
||||
|
||||
/// A minimal child descriptor with one memory resource, for the containment test.
|
||||
fn childDescriptor(hid: []const u8, start: u64, len: u64) device_abi.DeviceDescriptor {
|
||||
var child = std.mem.zeroes(device_abi.DeviceDescriptor);
|
||||
child.class = @intFromEnum(device_abi.DeviceClass.unknown);
|
||||
child.pci_class = device_abi.no_pci_class;
|
||||
child.hid_len = @intCast(hid.len);
|
||||
@memcpy(child.hid[0..hid.len], hid);
|
||||
child.resource_count = 1;
|
||||
child.resources[0] = .{ .kind = @intFromEnum(device_abi.ResourceKind.memory), .start = start, .len = len };
|
||||
return child;
|
||||
}
|
||||
|
||||
/// The device manager (a ring-3 service) enumerates /system/devices, matches each
|
||||
/// device to a driver, and — eventually — spawns it. This increment only checks the
|
||||
/// discovery+matching half: it must find the HPET (a timer) and decide `hpet` serves
|
||||
/// it, printing "device-manager: ok". It uses no special privilege — the same
|
||||
/// `device_enumerate` any process could call. (Spawning is the next increment.)
|
||||
/// device to a driver, and spawns it. Proof of the whole discover -> match -> spawn ->
|
||||
/// driver-up chain: boot only the device-manager; it must discover the PCI host bridge,
|
||||
/// match `pci-bus`, and spawn it (with the bridge id as its argument) — and the spawned
|
||||
/// pci-bus must reach its own live marker. It uses no special privilege — the same
|
||||
/// `device_enumerate` any process could call.
|
||||
fn deviceManagerTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: device-manager\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
@@ -2026,70 +2398,65 @@ fn deviceManagerTest(boot_information: *const BootInformation) void {
|
||||
};
|
||||
|
||||
// Let `system_spawn` find bundled binaries by name (the normal boot path does
|
||||
// this too). Only the device-manager is spawned here — so if `hpet` runs at all,
|
||||
// it's because the manager discovered the timer, matched, and spawned it.
|
||||
// this too). Only the device-manager is spawned here — so if `pci-bus` runs at
|
||||
// all, it's because the manager discovered the PCI host bridge, matched, and
|
||||
// spawned it.
|
||||
process.setInitialRamdisk(image);
|
||||
|
||||
process.write_count = 0;
|
||||
process.write_from_user = false;
|
||||
check("device-manager spawned from the initial_ramdisk", spawnNamed(rd, "device-manager"));
|
||||
|
||||
// End-to-end proof: the driver the manager spawned reaches its own live marker.
|
||||
// `hpet: ok` is hpet's final, stable message (it claims the timer, maps its MMIO,
|
||||
// binds its IRQ, services one, then sleeps) — nothing overwrites the buffer after,
|
||||
// so it's race-free to poll for. Its arrival means the whole
|
||||
// discover -> match -> system_spawn -> driver-up chain worked.
|
||||
const prefix = "hpet: ok";
|
||||
// End-to-end proof, read from kernel state — not the racy last-write serial buffer,
|
||||
// since many services keep logging after pci-bus. The manager must discover the PCI
|
||||
// host bridge, match pci-bus, and spawn it, and pci-bus must come up: claim the
|
||||
// bridge, map its ECAM, and register the functions it enumerates as children in the
|
||||
// device tree.
|
||||
scheduler.setPriority(1);
|
||||
const deadline = architecture.millis() + 10000;
|
||||
var spawned = false;
|
||||
while (architecture.millis() < deadline) {
|
||||
if (process.write_len >= prefix.len and eql(process.write_buffer[0..prefix.len], prefix)) break;
|
||||
if (processRunning("pci-bus")) spawned = true;
|
||||
if (spawned and pciFunctionsRegistered()) break;
|
||||
scheduler.yield();
|
||||
}
|
||||
scheduler.setPriority(4);
|
||||
|
||||
const ok = process.write_len >= prefix.len and eql(process.write_buffer[0..prefix.len], prefix);
|
||||
check("device manager matched the timer and system_spawn'd hpet, which came up", ok);
|
||||
check("device manager discovered the PCI host bridge and spawned pci-bus", spawned);
|
||||
check("pci-bus came up and registered the functions it enumerated", pciFunctionsRegistered());
|
||||
check("its syscalls came from user mode (CPL 3)", process.write_from_user);
|
||||
result();
|
||||
}
|
||||
|
||||
/// Every child `bus` registered must have each of its resources inside a parent
|
||||
/// resource of the same kind — the invariant `device_register` exists to maintain,
|
||||
/// checked from the kernel's own table rather than the driver's word for it.
|
||||
///
|
||||
/// Only *registered* children are checked, not the whole tree. Firmware topology is
|
||||
/// trusted and doesn't obey containment: a PCI function's BAR is not inside its host
|
||||
/// bridge's `bus_range`, because a bus-number range isn't an address window.
|
||||
fn childrenContained() bool {
|
||||
var buffer: [64]device_abi.DeviceDescriptor = undefined;
|
||||
const n = @min(devices_broker.enumerate(&buffer), buffer.len);
|
||||
|
||||
const bus_id = hpetDeviceId() orelse return false;
|
||||
const p = buffer[@intCast(bus_id)];
|
||||
|
||||
var children: usize = 0;
|
||||
for (buffer[0..n]) |d| {
|
||||
if (d.parent != bus_id) continue;
|
||||
children += 1;
|
||||
for (0..d.resource_count) |i| {
|
||||
const r = d.resources[i];
|
||||
var ok = false;
|
||||
for (0..p.resource_count) |j| {
|
||||
const pr = p.resources[j];
|
||||
if (pr.kind != r.kind) continue;
|
||||
if (r.kind == @intFromEnum(device_abi.ResourceKind.irq)) {
|
||||
if (pr.start == r.start) ok = true;
|
||||
} else if (r.len != 0 and r.start >= pr.start and
|
||||
r.start + r.len <= pr.start + pr.len) ok = true;
|
||||
}
|
||||
if (!ok) return false;
|
||||
}
|
||||
/// Whether a live task was spawned under `name` (its argv[0]) — read from the kernel
|
||||
/// task table, the same snapshot `process_enumerate` exposes.
|
||||
fn processRunning(name: []const u8) bool {
|
||||
var table: [64]abi.ProcessDescriptor = undefined;
|
||||
const total = scheduler.enumerate(&table);
|
||||
for (table[0..@min(total, table.len)]) |d| {
|
||||
if (std.mem.eql(u8, d.name[0..d.name_length], name)) return true;
|
||||
}
|
||||
return children > 0; // bus must have published at least one
|
||||
return false;
|
||||
}
|
||||
|
||||
/// Device id of the HPET (the bus bus claims), from the same table drivers see.
|
||||
/// Whether pci-bus registered at least one function under the PCI host bridge — proof
|
||||
/// it came up, claimed the bridge, mapped its ECAM, and walked configuration space.
|
||||
fn pciFunctionsRegistered() bool {
|
||||
var buffer: [64]device_abi.DeviceDescriptor = undefined;
|
||||
const n = @min(devices_broker.enumerate(&buffer), buffer.len);
|
||||
var bridge_id: ?u64 = null;
|
||||
for (buffer[0..n]) |d| {
|
||||
if (d.class == @intFromEnum(device_abi.DeviceClass.pci_host_bridge)) bridge_id = d.id;
|
||||
}
|
||||
const bid = bridge_id orelse return false;
|
||||
for (buffer[0..n]) |d| {
|
||||
if (d.parent == bid) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/// Device id of the kernel-seeded HPET timer block (the node with a memory resource),
|
||||
/// from the same device table drivers see. Used as a containment-test parent.
|
||||
fn hpetDeviceId() ?u64 {
|
||||
var buffer: [64]device_abi.DeviceDescriptor = undefined;
|
||||
const n = @min(devices_broker.enumerate(&buffer), buffer.len);
|
||||
@@ -2107,8 +2474,9 @@ fn hpetDeviceId() ?u64 {
|
||||
/// (so a dead driver's device goes quiet instead of storming) and the slot cleared
|
||||
/// (so an ISR never posts a notification into the endpoint that is about to be freed).
|
||||
///
|
||||
/// This is the path `hpet` never takes — it runs forever — so it gets its own test.
|
||||
/// Two properties, both read back from the hardware rather than from our own state:
|
||||
/// A long-running driver that never exits wouldn't reach this teardown path, so it
|
||||
/// gets its own test that binds and releases directly. Two properties, both read back
|
||||
/// from the hardware rather than from our own state:
|
||||
///
|
||||
/// 1. A bound GSI is routed and unmasked.
|
||||
/// 2. After `releaseOwner` for the binding's owner, that same entry is masked again.
|
||||
|
||||
@@ -16,8 +16,11 @@
|
||||
pub const maximum_cpus = 128;
|
||||
|
||||
/// Maximum tasks (kernel threads) alive at once — the static task-table size. Each
|
||||
/// online core consumes one slot for its idle task, plus task 0 on the BSP.
|
||||
pub const maximum_tasks = 16;
|
||||
/// online core consumes one slot for its idle task, plus task 0 on the BSP. Sized
|
||||
/// for the initial-ramdisk sweep (15 bundled binaries spawned at once) plus the
|
||||
/// device manager's supervised children with room to grow — at 16 the sweep
|
||||
/// started failing spawns once the bundle passed a dozen binaries.
|
||||
pub const maximum_tasks = 32;
|
||||
|
||||
/// Each task's kernel stack (also each AP's bring-up stack), in bytes.
|
||||
pub const kernel_stack_size = 16 * 1024;
|
||||
|
||||
@@ -0,0 +1,699 @@
|
||||
//! /system/services/acpi — the ACPI discovery service: the x86 firmware
|
||||
//! interpreter, moved out of ring 0 (docs/discovery.md). Claims the
|
||||
//! `acpi-tables` node the kernel publishes (the AML blobs, the broad io_port
|
||||
//! grant, a broad irq window, the SCI), and runs the **shared AML module** in
|
||||
//! ring 3 — the same parser and interpreter the kernel uses.
|
||||
//!
|
||||
//! It also owns the **event side** (M21): it registers the domain-named `.power`
|
||||
//! service, binds the SCI (System Control Interrupt), and on a power-button
|
||||
//! fixed event publishes `power_button` to subscribers — and on init's request
|
||||
//! writes S5 to power the machine off. The device discovery (M20) and the event
|
||||
//! handling both run in one `runtime.service.run` loop.
|
||||
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
const aml = @import("aml");
|
||||
const acpi_ids = @import("acpi-ids");
|
||||
const device = runtime.device;
|
||||
const protocol = runtime.device_manager_protocol;
|
||||
const power = runtime.power_protocol;
|
||||
/// AML opcode/prefix bytes by name (`zero_opcode`, `byte_prefix`, …) — so the `_HID`
|
||||
/// integer decode names the opcodes instead of bare 0x0A/0x0B/… (docs/coding-standards.md).
|
||||
const opcodes = aml.opcodes;
|
||||
|
||||
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
var line: [128]u8 = undefined;
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
// The claimed acpi-tables node and the resource index of its broad io_port
|
||||
// window — the Hal routes every port access through this one claim.
|
||||
var node_id: u64 = 0;
|
||||
var io_resource_index: u64 = 0;
|
||||
// The SCI's irq resource index on the node (the len-1 irq, distinct from the
|
||||
// broad [0,256) window), for irqBind / irqAck.
|
||||
var sci_resource_index: u64 = 0;
|
||||
var has_sci = false;
|
||||
|
||||
// PM1 event/control and GPE register ports, read from the FADT copy the kernel
|
||||
// publishes on the node (M21). Port 0 means absent.
|
||||
var pm1a_evt: u16 = 0;
|
||||
var pm1b_evt: u16 = 0;
|
||||
var pm1_evt_len: u8 = 0;
|
||||
var pm1a_cnt: u16 = 0;
|
||||
var pm1b_cnt: u16 = 0;
|
||||
var gpe0_blk: u16 = 0;
|
||||
var gpe0_len: u8 = 0;
|
||||
var gpe1_blk: u16 = 0;
|
||||
var gpe1_len: u8 = 0;
|
||||
var smi_cmd: u16 = 0;
|
||||
var acpi_enable_value: u8 = 0;
|
||||
var s5_slp_typ_a: u8 = 0;
|
||||
var s5_slp_typ_b: u8 = 0;
|
||||
var s5_valid = false;
|
||||
|
||||
// PM1 event-register bits (ACPI): PWRBTN in the status/enable word is bit 8;
|
||||
// the control word's SCI_EN is bit 0; SLP_EN is bit 13.
|
||||
const pwrbtn_bit: u16 = 1 << 8;
|
||||
const sci_en_bit: u32 = 1 << 0;
|
||||
const slp_en: u32 = 1 << 13;
|
||||
|
||||
// The `.power` subscribers: endpoints handed over as capabilities, each
|
||||
// receiving events as buffered messages. Dropped on a failed send. The
|
||||
// subscriber's task id is kept too — a shutdown request is honored only from a
|
||||
// subscriber (init subscribes; a stray process does not), the soft gate that
|
||||
// stands in for "only the system supervisor may power off" without hardcoding
|
||||
// a pid the kernel's idle tasks would have taken.
|
||||
const maximum_subscribers = 8;
|
||||
var subscribers: [maximum_subscribers]?runtime.ipc.Handle = .{null} ** maximum_subscribers;
|
||||
var subscriber_tasks: [maximum_subscribers]u32 = .{0} ** maximum_subscribers;
|
||||
|
||||
// Pass-1 registration record (see main): what pass 2 reports.
|
||||
const Registered = struct { hid: [8]u8 = .{0} ** 8, hid_len: usize = 0, device_id: u64 = 0, resource_count: u64 = 0 };
|
||||
var registered: [64]Registered = undefined;
|
||||
var registered_count: usize = 0;
|
||||
|
||||
// A scratch page returned for SystemMemory OperationRegion maps: the service
|
||||
// cannot map arbitrary physical memory from ring 3, so such regions are
|
||||
// unsupported and degrade to harmless zeros rather than faulting. The M20.2
|
||||
// targets (ps2, the legacy devices) use SystemIO and static templates.
|
||||
var mmio_scratch: [4096]u8 align(4096) = .{0} ** 4096;
|
||||
|
||||
fn halMapMmio(physical: u64, len: u64, writable: bool) u64 {
|
||||
_ = physical;
|
||||
_ = len;
|
||||
_ = writable;
|
||||
return @intFromPtr(&mmio_scratch);
|
||||
}
|
||||
|
||||
fn halPioRead(width: u8, port: u16) u32 {
|
||||
return device.ioRead(node_id, io_resource_index, port, width) orelse 0;
|
||||
}
|
||||
|
||||
fn halPioWrite(width: u8, port: u16, value: u32) void {
|
||||
_ = device.ioWrite(node_id, io_resource_index, port, width, value);
|
||||
}
|
||||
|
||||
fn findTablesNode(buffer: []device.DeviceDescriptor) ?device.DeviceDescriptor {
|
||||
const total = device.enumerate(buffer);
|
||||
for (buffer[0..@min(total, buffer.len)]) |d| {
|
||||
if (d.class == @intFromEnum(device.DeviceClass.acpi_tables)) return d;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
pub fn main(init: runtime.process.Init) void {
|
||||
// When the acpi-parse scenario spawns this directly, argv[1] is the kernel's
|
||||
// own device count to self-verify against — deterministic, no log-scraping.
|
||||
const expected: ?usize = if (init.arguments.get(1)) |a| (std.fmt.parseInt(usize, a, 10) catch null) else null;
|
||||
|
||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||
_ = runtime.system.write("/system/services/acpi: out of memory\n");
|
||||
return;
|
||||
};
|
||||
const node = findTablesNode(buffer) orelse {
|
||||
_ = runtime.system.write("/system/services/acpi: no acpi-tables node to claim\n");
|
||||
return;
|
||||
};
|
||||
node_id = node.id;
|
||||
if (!device.claim(node_id)) {
|
||||
_ = runtime.system.write("/system/services/acpi: unable to claim acpi-tables\n");
|
||||
return;
|
||||
}
|
||||
|
||||
// Map the node's resources: the AML blobs (bytecode), the FADT (intact
|
||||
// "FACP" header — decision 3), the io_port grant, and the SCI irq.
|
||||
var blocks: [8][]const u8 = undefined;
|
||||
var block_count: usize = 0;
|
||||
var found_io = false;
|
||||
var fadt: ?[]const u8 = null;
|
||||
for (node.resources[0..@intCast(node.resource_count)], 0..) |resource, index| {
|
||||
if (resource.kind == @intFromEnum(device.ResourceKind.io_port) and !found_io) {
|
||||
io_resource_index = index;
|
||||
found_io = true;
|
||||
continue;
|
||||
}
|
||||
if (resource.kind == @intFromEnum(device.ResourceKind.irq) and resource.len == 1) {
|
||||
sci_resource_index = index;
|
||||
has_sci = true;
|
||||
continue;
|
||||
}
|
||||
if (resource.kind != @intFromEnum(device.ResourceKind.memory)) continue;
|
||||
const base = device.mmioMap(node_id, index) orelse continue;
|
||||
const pointer: [*]const u8 = @ptrFromInt(base);
|
||||
const bytes = pointer[0..@intCast(resource.len)];
|
||||
if (bytes.len >= 4 and std.mem.eql(u8, bytes[0..4], "FACP")) {
|
||||
fadt = bytes;
|
||||
continue;
|
||||
}
|
||||
if (block_count == blocks.len) continue;
|
||||
blocks[block_count] = bytes;
|
||||
block_count += 1;
|
||||
}
|
||||
if (block_count == 0) {
|
||||
_ = runtime.system.write("/system/services/acpi: no AML blobs on the node\n");
|
||||
return;
|
||||
}
|
||||
|
||||
const result = aml.parse(runtime.allocator(), blocks[0..block_count]) catch {
|
||||
_ = runtime.system.write("/system/services/acpi: AML parse failed\n");
|
||||
return;
|
||||
};
|
||||
var namespace = result.namespace;
|
||||
const devices = aml.deviceCount(&namespace);
|
||||
writeLine("/system/services/acpi: parsed {d} AML blob(s), {d} namespace devices\n", .{ block_count, devices });
|
||||
if (expected) |want| {
|
||||
if (devices == want) {
|
||||
_ = runtime.system.write("acpi-parse: ok\n");
|
||||
} else {
|
||||
writeLine("acpi-parse: mismatch (ring-3 {d} vs kernel {d})\n", .{ devices, want });
|
||||
}
|
||||
// Self-verify mode is standalone (no manager); stop before reporting.
|
||||
while (true) runtime.system.sleep(1000);
|
||||
}
|
||||
|
||||
// Register + report the present _HID devices (M20), then set up the power
|
||||
// event side (M21), then serve — all in one harness loop. The interpreter
|
||||
// and namespace outlive this frame (static), so the harness callbacks can
|
||||
// reach them.
|
||||
interpreter_arena = std.heap.ArenaAllocator.init(runtime.allocator());
|
||||
persistent_namespace = namespace;
|
||||
global_interpreter = aml.Interpreter.init(&persistent_namespace, .{
|
||||
.mapMmio = halMapMmio,
|
||||
.pioRead = halPioRead,
|
||||
.pioWrite = halPioWrite,
|
||||
}, interpreter_arena.allocator());
|
||||
|
||||
readFadt(fadt);
|
||||
s5_valid = readSleepS5(&persistent_namespace);
|
||||
|
||||
runtime.service.run(power.message_maximum, .{
|
||||
.service = .power,
|
||||
.init = onInit,
|
||||
.on_message = onMessage,
|
||||
.on_notification = onNotification,
|
||||
});
|
||||
}
|
||||
|
||||
// Static so the harness callbacks (which run after main's stack frame is gone)
|
||||
// can reach the namespace and interpreter.
|
||||
var persistent_namespace: aml.Namespace = undefined;
|
||||
var global_interpreter: aml.Interpreter = undefined;
|
||||
var interpreter_arena: std.heap.ArenaAllocator = undefined;
|
||||
|
||||
/// Startup under the harness: register + report the discovered devices to the
|
||||
/// manager (M20), then enable ACPI mode and arm the power button (M21).
|
||||
fn onInit(endpoint: runtime.ipc.Handle) bool {
|
||||
registered_count = 0;
|
||||
walkDevices(persistent_namespace.root, &global_interpreter);
|
||||
|
||||
const manager = runtime.ipc.lookup(.device_manager);
|
||||
var i: usize = 0;
|
||||
while (i < registered_count) : (i += 1) {
|
||||
const entry = registered[i];
|
||||
const hid = entry.hid[0..entry.hid_len];
|
||||
const desc = acpi_ids.description(hid);
|
||||
if (desc.len != 0)
|
||||
writeLine("/system/services/acpi: reported {s} (device {d}, {d} resources) — {s}\n", .{ hid, entry.device_id, entry.resource_count, desc })
|
||||
else
|
||||
writeLine("/system/services/acpi: reported {s} (device {d}, {d} resources)\n", .{ hid, entry.device_id, entry.resource_count });
|
||||
if (manager) |h| {
|
||||
var report = protocol.ChildAdded{ .parent = node_id, .bus_address = entry.device_id, .identity = 0, .device_id = entry.device_id };
|
||||
@memcpy(report.hid[0..entry.hid_len], entry.hid[0..entry.hid_len]);
|
||||
var reply: [protocol.message_maximum]u8 = undefined;
|
||||
_ = runtime.ipc.call(h, std.mem.asBytes(&report), &reply) catch {};
|
||||
}
|
||||
}
|
||||
writeLine("/system/services/acpi: reported {d} device(s) to the manager\n", .{registered_count});
|
||||
|
||||
armPowerButton(endpoint);
|
||||
return true;
|
||||
}
|
||||
|
||||
// --- power event side (M21) ---------------------------------------------------
|
||||
|
||||
/// Read the PM1 event/control and GPE register ports plus the SMI enable pair
|
||||
/// from the FADT copy on the node. Offsets are from the FADT table start (the
|
||||
/// SDT header is the first 36 bytes). Prefers the 32-bit port fields; QEMU's
|
||||
/// FADT populates them.
|
||||
fn readFadt(fadt: ?[]const u8) void {
|
||||
const f = fadt orelse {
|
||||
_ = runtime.system.write("acpi: no FADT on the node — power events off\n");
|
||||
return;
|
||||
};
|
||||
smi_cmd = @truncate(rd32(f, 48));
|
||||
acpi_enable_value = f[52];
|
||||
pm1a_evt = @truncate(rd32(f, 56));
|
||||
pm1b_evt = @truncate(rd32(f, 60));
|
||||
pm1a_cnt = @truncate(rd32(f, 64));
|
||||
pm1b_cnt = @truncate(rd32(f, 68));
|
||||
gpe0_blk = @truncate(rd32(f, 80));
|
||||
gpe1_blk = @truncate(rd32(f, 84));
|
||||
pm1_evt_len = if (f.len > 88) f[88] else 4;
|
||||
gpe0_len = if (f.len > 92) f[92] else 0;
|
||||
gpe1_len = if (f.len > 93) f[93] else 0;
|
||||
}
|
||||
|
||||
fn readSleepS5(ns: *aml.Namespace) bool {
|
||||
const st = aml.sleepState(ns, 5) orelse return false;
|
||||
s5_slp_typ_a = st.slp_typ_a;
|
||||
s5_slp_typ_b = st.slp_typ_b;
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Enable ACPI mode if the firmware isn't already in it, then bind the SCI and
|
||||
/// set PWRBTN_EN so the power button raises an interrupt we can see.
|
||||
fn armPowerButton(endpoint: runtime.ipc.Handle) void {
|
||||
if (pm1a_cnt != 0 and (halPioRead(2, pm1a_cnt) & sci_en_bit) == 0 and smi_cmd != 0) {
|
||||
// Switch to ACPI mode: write ACPI_ENABLE to the SMI command port, then
|
||||
// spin (bounded) until SCI_EN latches.
|
||||
halPioWrite(1, smi_cmd, acpi_enable_value);
|
||||
var tries: u32 = 0;
|
||||
while (tries < 1000 and (halPioRead(2, pm1a_cnt) & sci_en_bit) == 0) : (tries += 1) {
|
||||
runtime.system.sleep(1);
|
||||
}
|
||||
}
|
||||
if (!has_sci) {
|
||||
_ = runtime.system.write("acpi: no SCI resource — power button unavailable\n");
|
||||
return;
|
||||
}
|
||||
if (!device.irqBind(node_id, sci_resource_index, endpoint)) {
|
||||
_ = runtime.system.write("acpi: SCI irq_bind failed\n");
|
||||
return;
|
||||
}
|
||||
// PWRBTN_EN lives in the PM1 enable register at evt_blk + evt_len/2.
|
||||
if (pm1a_evt != 0) {
|
||||
const en_port = pm1a_evt + pm1_evt_len / 2;
|
||||
halPioWrite(2, en_port, @as(u16, @truncate(halPioRead(2, en_port))) | pwrbtn_bit);
|
||||
}
|
||||
if (pm1b_evt != 0) {
|
||||
const en_port = pm1b_evt + pm1_evt_len / 2;
|
||||
halPioWrite(2, en_port, @as(u16, @truncate(halPioRead(2, en_port))) | pwrbtn_bit);
|
||||
}
|
||||
_ = runtime.system.write("acpi: power button armed\n");
|
||||
}
|
||||
|
||||
/// The SCI fired. Read PM1 status; a set PWRBTN_STS is the power button — clear
|
||||
/// it (write-1), publish, log. Any other set status is cleared and logged
|
||||
/// (GPE/Notify dispatch is M21.2). Always re-arm the line.
|
||||
fn onSci() void {
|
||||
var handled = false;
|
||||
inline for (.{ pm1a_evt, pm1b_evt }) |evt_port| {
|
||||
if (evt_port != 0) {
|
||||
const sts: u16 = @truncate(halPioRead(2, evt_port));
|
||||
if (sts & pwrbtn_bit != 0) {
|
||||
halPioWrite(2, evt_port, pwrbtn_bit); // write-1-to-clear
|
||||
handled = true;
|
||||
} else if (sts != 0) {
|
||||
halPioWrite(2, evt_port, sts); // clear whatever else latched
|
||||
}
|
||||
}
|
||||
}
|
||||
if (handled) {
|
||||
_ = runtime.system.write("power: button pressed\n");
|
||||
publishButton();
|
||||
}
|
||||
handleGpe();
|
||||
_ = device.irqAck(node_id, sci_resource_index);
|
||||
}
|
||||
|
||||
/// General-purpose events: for each set+enabled GPE bit, evaluate its `\_GPE`
|
||||
/// handler method (`_Lxx` level / `_Exx` edge), drain the Notify queue the
|
||||
/// method produced, and publish an event per notified device. Then clear the
|
||||
/// status bit. QEMU raises no GPEs on this config, so this path is exercised by
|
||||
/// host unit tests (docs/acpi.md — ACPI events); on real hardware it carries
|
||||
/// battery/AC/lid. The embedded controller's `_Qxx` queries are out of scope.
|
||||
fn handleGpe() void {
|
||||
handleGpeBlock(gpe0_blk, gpe0_len, 0);
|
||||
handleGpeBlock(gpe1_blk, gpe1_len, gpe0_len * 4);
|
||||
}
|
||||
|
||||
fn handleGpeBlock(blk: u16, len: u8, gpe_base: u32) void {
|
||||
if (blk == 0 or len == 0) return;
|
||||
const status_bytes = len / 2; // status half, then enable half
|
||||
var byte_index: u8 = 0;
|
||||
while (byte_index < status_bytes) : (byte_index += 1) {
|
||||
const sts: u8 = @truncate(halPioRead(1, blk + byte_index));
|
||||
const en: u8 = @truncate(halPioRead(1, blk + status_bytes + byte_index));
|
||||
const active = sts & en;
|
||||
if (active == 0) continue;
|
||||
var bit: u3 = 0;
|
||||
while (true) : (bit += 1) {
|
||||
if (active & (@as(u8, 1) << bit) != 0) {
|
||||
dispatchGpe(gpe_base + @as(u32, byte_index) * 8 + bit);
|
||||
}
|
||||
if (bit == 7) break;
|
||||
}
|
||||
halPioWrite(1, blk + byte_index, active); // write-1-to-clear the serviced bits
|
||||
}
|
||||
}
|
||||
|
||||
/// Evaluate the `\_GPE._L%02X` or `_E%02X` handler for GPE number `n`, then
|
||||
/// publish an event for each device it notified.
|
||||
fn dispatchGpe(n: u32) void {
|
||||
const gpe_scope = aml.Namespace.resolve(&persistent_namespace, persistent_namespace.root, true, 0, &.{seg4("_GPE")}) orelse return;
|
||||
var name: [4]u8 = .{ '_', 'L', 0, 0 };
|
||||
writeHex2(name[2..4], n);
|
||||
var method = aml.Namespace.childOf(gpe_scope, name);
|
||||
if (method == null) {
|
||||
name[1] = 'E';
|
||||
method = aml.Namespace.childOf(gpe_scope, name);
|
||||
}
|
||||
const m = method orelse return; // no handler — the status bit was already cleared
|
||||
_ = global_interpreter.evaluate(m, &.{}) catch return;
|
||||
for (global_interpreter.takeNotifications()) |event| publishNotify(event.node, event.code);
|
||||
}
|
||||
|
||||
fn publishNotify(node: *aml.Node, code: u64) void {
|
||||
// Map the notified device's _HID to a domain event where we recognize it.
|
||||
var hid: [8]u8 = .{0} ** 8;
|
||||
if (readHid(node, &global_interpreter)) |h| hid = h;
|
||||
const which: power.Event = if (std.mem.eql(u8, hid[0..7], "PNP0C0A")) .battery else if (std.mem.eql(u8, hid[0..7], "ACPI0003")) .ac else if (std.mem.eql(u8, hid[0..7], "PNP0C0D")) .lid else .notify;
|
||||
var event = power.EventMessage{ .event = @intFromEnum(which), .code = @truncate(code) };
|
||||
event.hid = hid;
|
||||
writeLine("power: notify {s} code {d}\n", .{ hid[0..7], code });
|
||||
publishEvent(std.mem.asBytes(&event));
|
||||
}
|
||||
|
||||
/// Two lowercase hex digits of `n` into `out[0..2]`.
|
||||
fn writeHex2(out: []u8, n: u32) void {
|
||||
const digits = "0123456789ABCDEF";
|
||||
out[0] = digits[(n >> 4) & 0xF];
|
||||
out[1] = digits[n & 0xF];
|
||||
}
|
||||
|
||||
fn publishButton() void {
|
||||
const event = power.EventMessage{ .event = @intFromEnum(power.Event.power_button) };
|
||||
publishEvent(std.mem.asBytes(&event));
|
||||
}
|
||||
|
||||
fn publishEvent(bytes: []const u8) void {
|
||||
for (&subscribers) |*slot| {
|
||||
if (slot.*) |handle| {
|
||||
if (!runtime.ipc.send(handle, bytes)) slot.* = null;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn isSubscriber(task: u32) bool {
|
||||
for (&subscribers, 0..) |*slot, si| {
|
||||
if (slot.* != null and subscriber_tasks[si] == task) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/// Enter S5 (soft off): write SLP_TYP|SLP_EN to the PM1 control register(s).
|
||||
/// Mirrors the kernel's power.zig sleepValue. Only reached from a PID-1
|
||||
/// shutdown request (M21.3).
|
||||
fn enterS5() void {
|
||||
if (!s5_valid or pm1a_cnt == 0) {
|
||||
_ = runtime.system.write("power: S5 unavailable\n");
|
||||
return;
|
||||
}
|
||||
_ = runtime.system.write("power: entering S5\n");
|
||||
halPioWrite(2, pm1a_cnt, (@as(u32, s5_slp_typ_a & 0x7) << 10) | slp_en);
|
||||
if (pm1b_cnt != 0) halPioWrite(2, pm1b_cnt, (@as(u32, s5_slp_typ_b & 0x7) << 10) | slp_en);
|
||||
// If control returns, the write did not take — say so instead of hanging.
|
||||
runtime.system.sleep(500);
|
||||
_ = runtime.system.write("power: S5 write did not take\n");
|
||||
}
|
||||
|
||||
// --- harness callbacks --------------------------------------------------------
|
||||
|
||||
fn onNotification(badge: u64) void {
|
||||
// The only notification the service binds is the SCI (an IRQ badge).
|
||||
_ = badge;
|
||||
onSci();
|
||||
}
|
||||
|
||||
/// The `.power` protocol: subscribe (endpoint as the call's capability),
|
||||
/// shutdown (PID 1 only). Device discovery uses a different endpoint (the
|
||||
/// device manager's), so nothing here handles ChildAdded.
|
||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?runtime.ipc.Handle) usize {
|
||||
if (message.len < 1) return 0;
|
||||
switch (message[0]) {
|
||||
@intFromEnum(power.Operation.subscribe) => {
|
||||
var status: i32 = -1;
|
||||
if (capability) |handle| {
|
||||
for (&subscribers, 0..) |*slot, si| {
|
||||
if (slot.* == null) {
|
||||
slot.* = handle;
|
||||
subscriber_tasks[si] = sender;
|
||||
status = 0;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
const r = power.Reply{ .status = status };
|
||||
@memcpy(reply[0..@sizeOf(power.Reply)], std.mem.asBytes(&r));
|
||||
return @sizeOf(power.Reply);
|
||||
},
|
||||
@intFromEnum(power.Operation.shutdown) => {
|
||||
// Honored only from a power subscriber — init, which has already run
|
||||
// the stop sequence over everything else. The power service is
|
||||
// mechanism (write S5); deciding *when* to shut down and stopping
|
||||
// the rest of the system first is init's policy.
|
||||
const allowed = isSubscriber(sender);
|
||||
const r = power.Reply{ .status = if (allowed) 0 else -1 };
|
||||
@memcpy(reply[0..@sizeOf(power.Reply)], std.mem.asBytes(&r));
|
||||
if (allowed) enterS5();
|
||||
return @sizeOf(power.Reply);
|
||||
},
|
||||
else => return 0,
|
||||
}
|
||||
}
|
||||
|
||||
/// Depth-first walk: register + report each present device with a _HID, then
|
||||
/// descend. Scopes (\_SB, \_GPE …) are descended without producing a node.
|
||||
fn walkDevices(node: *aml.Node, interpreter: *aml.Interpreter) void {
|
||||
var child = node.first_child;
|
||||
while (child) |c| : (child = c.next_sibling) {
|
||||
if (c.kind != .device) {
|
||||
walkDevices(c, interpreter);
|
||||
continue;
|
||||
}
|
||||
if (!devicePresent(interpreter, c)) continue; // absent: skip it and its subtree
|
||||
|
||||
if (readHid(c, interpreter)) |hid| {
|
||||
// Skip PCI roots — pci-bus already reports PCI functions; ACPI adds
|
||||
// only the non-PCI _HID devices (docs/device-manager.md — matching). The two
|
||||
// roots are named through the shared registry, not bare _HID strings.
|
||||
const id = acpi_ids.HardwareId.fromHid(hid[0..7]);
|
||||
if (id != .pci_bus and id != .pci_express_root_bridge) {
|
||||
registerDevice(c, hid, interpreter);
|
||||
}
|
||||
}
|
||||
walkDevices(c, interpreter);
|
||||
}
|
||||
}
|
||||
|
||||
fn registerDevice(node: *aml.Node, hid: [8]u8, interpreter: *aml.Interpreter) void {
|
||||
if (registered_count >= registered.len) return;
|
||||
var descriptor = std.mem.zeroes(device.DeviceDescriptor);
|
||||
descriptor.class = @intFromEnum(device.DeviceClass.acpi_device);
|
||||
descriptor.pci_class = device.no_pci_class;
|
||||
const hid_len: u64 = std.mem.indexOfScalar(u8, &hid, 0) orelse hid.len;
|
||||
descriptor.hid_len = hid_len;
|
||||
@memcpy(descriptor.hid[0..@intCast(hid_len)], hid[0..@intCast(hid_len)]);
|
||||
applyCrs(&descriptor, node, interpreter);
|
||||
|
||||
const id = device.register(node_id, &descriptor) orelse {
|
||||
writeLine("/system/services/acpi: register refused for {s}\n", .{hid[0..@intCast(hid_len)]});
|
||||
return;
|
||||
};
|
||||
registered[registered_count] = .{ .hid = hid, .hid_len = @intCast(hid_len), .device_id = id, .resource_count = descriptor.resource_count };
|
||||
registered_count += 1;
|
||||
}
|
||||
|
||||
/// _STA bit 0 (present); absent method or a failed evaluation is treated as
|
||||
/// present, per the ACPI rules.
|
||||
fn devicePresent(interpreter: *aml.Interpreter, node: *aml.Node) bool {
|
||||
const sta = aml.Namespace.childOf(node, seg4("_STA")) orelse return true;
|
||||
const obj = interpreter.evaluate(sta, &.{}) catch return true;
|
||||
const status = obj.asInteger() catch return true;
|
||||
return (status & 0x01) != 0;
|
||||
}
|
||||
|
||||
/// The device's EISA-decoded _HID (e.g. "PNP0303"), or null.
|
||||
fn readHid(node: *aml.Node, interpreter: *aml.Interpreter) ?[8]u8 {
|
||||
const hid = aml.Namespace.childOf(node, seg4("_HID")) orelse return null;
|
||||
var buffer: [8]u8 = .{0} ** 8;
|
||||
if (hid.kind == .method) {
|
||||
const obj = interpreter.evaluate(hid, &.{}) catch return null;
|
||||
switch (obj) {
|
||||
.integer => |n| {
|
||||
_ = eisaIdToStr(@truncate(n), &buffer);
|
||||
return buffer;
|
||||
},
|
||||
else => return null,
|
||||
}
|
||||
}
|
||||
if (hid.kind != .name or hid.value.len == 0) return null;
|
||||
const v = hid.value;
|
||||
switch (v[0]) {
|
||||
// A static _HID names an integer EISA id: Zero/One/Ones or a Byte/Word/DWord/
|
||||
// QWord integer prefix. Anything else is not an integer we can EISA-decode.
|
||||
opcodes.zero_opcode, opcodes.one_opcode, opcodes.ones_opcode, opcodes.byte_prefix, opcodes.word_prefix, opcodes.dword_prefix, opcodes.qword_prefix => {
|
||||
var p: usize = 0;
|
||||
const n = readIntObj(v, &p) orelse return null;
|
||||
_ = eisaIdToStr(@truncate(n), &buffer);
|
||||
return buffer;
|
||||
},
|
||||
else => return null,
|
||||
}
|
||||
}
|
||||
|
||||
// --- _CRS resource-template decode (ported from the kernel's acpi.zig) --------
|
||||
|
||||
/// A resource template is a byte list of descriptors. Each starts with a tag byte whose
|
||||
/// high bit picks the encoding: a *small* descriptor carries its type in bits [6:3] and
|
||||
/// its length in bits [2:0]; a *large* descriptor is the whole tag byte, followed by a
|
||||
/// 16-bit length. These are the descriptor types danos decodes into resources — named so
|
||||
/// the walk below reads by descriptor, not by 0x04/0x85/… (docs/coding-standards.md).
|
||||
const large_descriptor_bit: u8 = 0x80; // set in a tag byte => large descriptor
|
||||
const small_length_mask: u8 = 0x07; // low 3 bits of a small tag = body length
|
||||
const small_type_shift: u3 = 3; // small type sits in bits [6:3]
|
||||
|
||||
/// Small resource descriptor types (tag bits [6:3]). Non-exhaustive: an unhandled type
|
||||
/// is skipped by its length, not misread.
|
||||
const SmallResourceType = enum(u8) {
|
||||
irq = 0x04,
|
||||
io_port = 0x08,
|
||||
fixed_io_port = 0x09,
|
||||
end_tag = 0x0F,
|
||||
_,
|
||||
};
|
||||
|
||||
/// Large resource descriptor types (the whole tag byte). Non-exhaustive for the same reason.
|
||||
const LargeResourceType = enum(u8) {
|
||||
memory32 = 0x85,
|
||||
memory32_fixed = 0x86,
|
||||
extended_irq = 0x89,
|
||||
_,
|
||||
};
|
||||
|
||||
fn applyCrs(descriptor: *device.DeviceDescriptor, node: *aml.Node, interpreter: *aml.Interpreter) void {
|
||||
const crs = aml.Namespace.childOf(node, seg4("_CRS")) orelse return;
|
||||
const obj = interpreter.evaluate(crs, &.{}) catch return;
|
||||
const bytes = switch (obj) {
|
||||
.buffer => |b| b,
|
||||
else => return,
|
||||
};
|
||||
var i: usize = 0;
|
||||
while (i < bytes.len) {
|
||||
const tag = bytes[i];
|
||||
if (tag & large_descriptor_bit == 0) {
|
||||
const len: usize = tag & small_length_mask;
|
||||
const body = i + 1;
|
||||
if (body + len > bytes.len) break;
|
||||
switch (@as(SmallResourceType, @enumFromInt((tag >> small_type_shift) & 0x0F))) {
|
||||
.irq => if (len >= 2) { // IRQ mask
|
||||
const mask = @as(u16, bytes[body]) | (@as(u16, bytes[body + 1]) << 8);
|
||||
var b: usize = 0;
|
||||
while (b < 16) : (b += 1) {
|
||||
if (mask & (@as(u16, 1) << @intCast(b)) != 0) addResource(descriptor, .irq, b, 1);
|
||||
}
|
||||
},
|
||||
.io_port => if (len >= 7) addResource(descriptor, .io_port, rd16(bytes, body + 1), bytes[body + 6]),
|
||||
.fixed_io_port => if (len >= 3) addResource(descriptor, .io_port, rd16(bytes, body), bytes[body + 2]),
|
||||
.end_tag => break,
|
||||
else => {},
|
||||
}
|
||||
i = body + len;
|
||||
} else {
|
||||
if (i + 3 > bytes.len) break;
|
||||
const len: usize = @intCast(rd16(bytes, i + 1));
|
||||
const body = i + 3;
|
||||
if (body + len > bytes.len) break;
|
||||
switch (@as(LargeResourceType, @enumFromInt(tag))) {
|
||||
.memory32 => if (len >= 17) addResource(descriptor, .memory, rd32(bytes, body + 1), rd32(bytes, body + 13)),
|
||||
.memory32_fixed => if (len >= 9) addResource(descriptor, .memory, rd32(bytes, body + 1), rd32(bytes, body + 5)),
|
||||
.extended_irq => if (len >= 2) {
|
||||
const count = bytes[body + 1];
|
||||
var k: usize = 0;
|
||||
while (k < count and body + 2 + k * 4 + 4 <= body + len) : (k += 1) {
|
||||
addResource(descriptor, .irq, rd32(bytes, body + 2 + k * 4), 1);
|
||||
}
|
||||
},
|
||||
else => {},
|
||||
}
|
||||
i = body + len;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn addResource(descriptor: *device.DeviceDescriptor, kind: device.ResourceKind, start: u64, len: u64) void {
|
||||
if (descriptor.resource_count >= descriptor.resources.len) return;
|
||||
descriptor.resources[@intCast(descriptor.resource_count)] = .{ .kind = @intFromEnum(kind), .start = start, .len = len };
|
||||
descriptor.resource_count += 1;
|
||||
}
|
||||
|
||||
// --- small helpers ported verbatim from the kernel's acpi.zig ----------------
|
||||
|
||||
fn seg4(comptime s: *const [4:0]u8) [4]u8 {
|
||||
return s[0..4].*;
|
||||
}
|
||||
|
||||
fn hexDigit(n: u8) u8 {
|
||||
return if (n < 10) '0' + n else 'A' + (n - 10);
|
||||
}
|
||||
|
||||
fn eisaIdToStr(id: u32, buffer: *[8]u8) []const u8 {
|
||||
const b0: u16 = @intCast(id & 0xFF);
|
||||
const b1: u16 = @intCast((id >> 8) & 0xFF);
|
||||
const b2: u8 = @truncate(id >> 16);
|
||||
const b3: u8 = @truncate(id >> 24);
|
||||
const mfg = (b0 << 8) | b1;
|
||||
buffer[0] = '@' + @as(u8, @intCast((mfg >> 10) & 0x1F));
|
||||
buffer[1] = '@' + @as(u8, @intCast((mfg >> 5) & 0x1F));
|
||||
buffer[2] = '@' + @as(u8, @intCast(mfg & 0x1F));
|
||||
buffer[3] = hexDigit((b2 >> 4) & 0xF);
|
||||
buffer[4] = hexDigit(b2 & 0xF);
|
||||
buffer[5] = hexDigit((b3 >> 4) & 0xF);
|
||||
buffer[6] = hexDigit(b3 & 0xF);
|
||||
buffer[7] = 0;
|
||||
return buffer[0..7];
|
||||
}
|
||||
|
||||
fn readIntObj(bytes: []const u8, p: *usize) ?u64 {
|
||||
if (p.* >= bytes.len) return null;
|
||||
const op = bytes[p.*];
|
||||
p.* += 1;
|
||||
switch (op) {
|
||||
opcodes.zero_opcode => return 0,
|
||||
opcodes.one_opcode => return 1,
|
||||
opcodes.ones_opcode => return 1,
|
||||
opcodes.byte_prefix => {
|
||||
if (p.* >= bytes.len) return null;
|
||||
const v = bytes[p.*];
|
||||
p.* += 1;
|
||||
return v;
|
||||
},
|
||||
opcodes.word_prefix => {
|
||||
if (p.* + 2 > bytes.len) return null;
|
||||
const v = rd16(bytes, p.*);
|
||||
p.* += 2;
|
||||
return v;
|
||||
},
|
||||
opcodes.dword_prefix => {
|
||||
if (p.* + 4 > bytes.len) return null;
|
||||
const v = rd32(bytes, p.*);
|
||||
p.* += 4;
|
||||
return v;
|
||||
},
|
||||
else => return null,
|
||||
}
|
||||
}
|
||||
|
||||
fn rd16(bytes: []const u8, off: usize) u64 {
|
||||
return @as(u64, bytes[off]) | (@as(u64, bytes[off + 1]) << 8);
|
||||
}
|
||||
|
||||
fn rd32(bytes: []const u8, off: usize) u64 {
|
||||
return rd16(bytes, off) | (rd16(bytes, off + 2) << 16);
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||
}
|
||||
@@ -0,0 +1,44 @@
|
||||
//! crash-test — a test fixture, not a driver: claims the device it is assigned,
|
||||
//! hellos the device manager, announces itself, then faults on purpose. The
|
||||
//! driver-restart scenario drives the manager's whole restart machinery with
|
||||
//! it: fault → exit reason → backoff → respawn → the **same claim succeeding
|
||||
//! again** (claim release on death, M17.1, through the manager's path) → the
|
||||
//! crash-loop cap. Spawned bare (the initial-ramdisk sweep starts every bundled
|
||||
//! binary), it exits silently so it cannot derange other tests.
|
||||
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
const protocol = runtime.device_manager_protocol;
|
||||
|
||||
pub fn main(init: runtime.process.Init) void {
|
||||
const argument = init.arguments.get(1) orelse return; // bare: stay silent
|
||||
const assigned = std.fmt.parseInt(u64, argument, 10) catch return;
|
||||
|
||||
// The respawn only reaches this line because the kernel released the
|
||||
// previous instance's claim at death. A failed claim exits cleanly — the
|
||||
// manager reads "meant to stop" and the scenario fails loudly by silence.
|
||||
if (!runtime.device.claim(assigned)) {
|
||||
_ = runtime.system.write("crash-test: claim failed\n");
|
||||
return;
|
||||
}
|
||||
|
||||
var manager: ?runtime.ipc.Handle = null;
|
||||
var tries: u32 = 0;
|
||||
while (manager == null and tries < 100) : (tries += 1) {
|
||||
manager = runtime.ipc.lookup(.device_manager);
|
||||
if (manager == null) runtime.system.sleep(20);
|
||||
}
|
||||
const h = manager orelse return;
|
||||
const hello = protocol.Hello{ .role = @intFromEnum(protocol.Role.device), .device_id = assigned };
|
||||
var reply: [protocol.message_maximum]u8 = undefined;
|
||||
_ = runtime.ipc.call(h, std.mem.asBytes(&hello), &reply) catch return;
|
||||
|
||||
_ = runtime.system.write("crash-test: faulting now\n");
|
||||
const poison: *volatile u32 = @ptrFromInt(0xdead0000);
|
||||
poison.* = 1; // the restart machinery's fuel: a real segmentation fault
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||
}
|
||||
@@ -0,0 +1,88 @@
|
||||
//! device-list — the `ps` analog for the device tree (docs/device-manager.md
|
||||
//! M18.3): asks the device manager for the tree over IPC, prints it, then
|
||||
//! subscribes and prints every published add/remove event. The manager is the
|
||||
//! one answer to "what devices exist" for user space; nothing here touches a
|
||||
//! device_* system call.
|
||||
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
const protocol = runtime.device_manager_protocol;
|
||||
|
||||
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
var line: [96]u8 = undefined;
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
pub fn main() void {
|
||||
var manager: ?runtime.ipc.Handle = null;
|
||||
var tries: u32 = 0;
|
||||
while (manager == null and tries < 200) : (tries += 1) {
|
||||
manager = runtime.ipc.lookup(.device_manager);
|
||||
if (manager == null) runtime.system.sleep(20);
|
||||
}
|
||||
const h = manager orelse {
|
||||
_ = runtime.system.write("device-list: no device manager\n");
|
||||
return;
|
||||
};
|
||||
|
||||
// The snapshot — polled briefly, because at boot the bus drivers may still
|
||||
// be scanning: an empty first answer usually just means "too early".
|
||||
var reply: [protocol.message_maximum]u8 = undefined;
|
||||
var count: u32 = 0;
|
||||
var length: usize = 0;
|
||||
tries = 0;
|
||||
while (tries < 20) : (tries += 1) {
|
||||
const request = protocol.Enumerate{};
|
||||
length = runtime.ipc.call(h, std.mem.asBytes(&request), &reply) catch 0;
|
||||
if (length >= @sizeOf(protocol.EnumerateReply)) {
|
||||
count = std.mem.bytesToValue(protocol.EnumerateReply, reply[0..@sizeOf(protocol.EnumerateReply)]).count;
|
||||
if (count != 0) break;
|
||||
}
|
||||
runtime.system.sleep(100);
|
||||
}
|
||||
writeLine("device-list: {d} devices\n", .{count});
|
||||
var offset: usize = @sizeOf(protocol.EnumerateReply);
|
||||
var index: u32 = 0;
|
||||
while (index < count and offset + @sizeOf(protocol.ChildEntry) <= length) : (index += 1) {
|
||||
const entry = std.mem.bytesToValue(protocol.ChildEntry, reply[offset..][0..@sizeOf(protocol.ChildEntry)]);
|
||||
writeLine("device-list: device {d} port {d} identity {d}\n", .{ entry.parent, entry.bus_address, entry.identity });
|
||||
offset += @sizeOf(protocol.ChildEntry);
|
||||
}
|
||||
|
||||
// The subscription: our endpoint rides as the call's capability; events
|
||||
// arrive as buffered messages carrying the same structs the bus sends.
|
||||
const endpoint = runtime.ipc.createIpcEndpoint() orelse {
|
||||
_ = runtime.system.write("device-list: no endpoint\n");
|
||||
return;
|
||||
};
|
||||
const subscribe = protocol.Subscribe{};
|
||||
_ = runtime.ipc.callCap(h, std.mem.asBytes(&subscribe), &reply, endpoint) catch {
|
||||
_ = runtime.system.write("device-list: subscribe failed\n");
|
||||
return;
|
||||
};
|
||||
_ = runtime.system.write("device-list: subscribed\n");
|
||||
|
||||
var receive: [protocol.message_maximum]u8 = undefined;
|
||||
while (true) {
|
||||
const got = runtime.ipc.replyWait(endpoint, &.{}, &receive, null);
|
||||
if (!got.isMessage() or got.len < 1) continue;
|
||||
switch (receive[0]) {
|
||||
@intFromEnum(protocol.Operation.child_added) => {
|
||||
if (got.len < protocol.child_added_size) continue;
|
||||
const event = std.mem.bytesToValue(protocol.ChildAdded, receive[0..protocol.child_added_size]);
|
||||
writeLine("device-list: added (device {d} port {d})\n", .{ event.parent, event.bus_address });
|
||||
},
|
||||
@intFromEnum(protocol.Operation.child_removed) => {
|
||||
if (got.len < protocol.child_removed_size) continue;
|
||||
const event = std.mem.bytesToValue(protocol.ChildRemoved, receive[0..protocol.child_removed_size]);
|
||||
writeLine("device-list: removed (device {d} port {d})\n", .{ event.parent, event.bus_address });
|
||||
},
|
||||
else => {},
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||
}
|
||||
@@ -0,0 +1,148 @@
|
||||
//! The device-manager protocol (docs/device-manager.md): what drivers and
|
||||
//! applications say to the device manager over its well-known endpoint. The
|
||||
//! vfs-protocol pattern — extern-struct messages, a version in the handshake,
|
||||
//! reserved fields — so both sides depend on the contract by name. Deliberately
|
||||
//! contains nothing lifecycle-shaped: stopping, liveness (the zero-length ping),
|
||||
//! and exit reasons are the universal vocabulary of
|
||||
//! docs/process-lifecycle.md, not this protocol.
|
||||
|
||||
/// The protocol version a driver states in its hello. A manager that cannot
|
||||
/// serve a driver's version refuses the hello, and the mismatch is loud at
|
||||
/// startup instead of quiet corruption later.
|
||||
pub const version: u16 = 1;
|
||||
|
||||
/// What kind of driver is talking (docs/driver-model.md's shapes).
|
||||
pub const Role = enum(u8) {
|
||||
/// Owns a controller and reports the devices behind it (`child_added`).
|
||||
bus = 1,
|
||||
/// Serves one device, reached through a bus's transfer protocol.
|
||||
device = 2,
|
||||
};
|
||||
|
||||
/// The message kinds.
|
||||
pub const Operation = enum(u8) {
|
||||
hello = 1,
|
||||
child_added = 2,
|
||||
child_removed = 3,
|
||||
enumerate = 4,
|
||||
subscribe = 5,
|
||||
};
|
||||
|
||||
/// `Hello.device_id` for a driver that serves no enumerated device (a test
|
||||
/// fixture, a synthetic source).
|
||||
pub const no_device: u64 = ~@as(u64, 0);
|
||||
|
||||
/// The handshake, sent once by every driver the manager spawns — the manager's
|
||||
/// one self-enforced deadline: spawned and silent past it means wrong binary,
|
||||
/// wrong version, or wedged before main, and the stop sequence follows.
|
||||
pub const Hello = extern struct {
|
||||
operation: u8 = @intFromEnum(Operation.hello),
|
||||
/// A Role value.
|
||||
role: u8,
|
||||
/// The protocol version this driver was built against (`version`).
|
||||
version: u16 = version,
|
||||
reserved: u32 = 0,
|
||||
/// The device this driver was assigned (its argv[1]), or `no_device`.
|
||||
device_id: u64,
|
||||
};
|
||||
|
||||
pub const hello_size = @sizeOf(Hello);
|
||||
|
||||
/// The manager's answer to a hello. Nonzero status = refused (version mismatch,
|
||||
/// unknown sender); a refused driver should exit cleanly.
|
||||
pub const HelloReply = extern struct {
|
||||
status: i32,
|
||||
reserved: u32 = 0,
|
||||
};
|
||||
|
||||
pub const reply_size = @sizeOf(HelloReply);
|
||||
|
||||
/// A bus driver reporting one device it discovered behind its controller
|
||||
/// (docs/device-manager.md "the tree"). Identity is the bus's native language —
|
||||
/// for USB a port-speed class; the (class, subclass, protocol) triple joins it
|
||||
/// once control transfers exist (the USB track). The manager mirrors the child
|
||||
/// into its tree; when the reporting driver dies, the manager prunes everything
|
||||
/// it reported (the children describe protocol state that died with it) and the
|
||||
/// restarted instance rediscovers and re-reports.
|
||||
pub const ChildAdded = extern struct {
|
||||
operation: u8 = @intFromEnum(Operation.child_added),
|
||||
reserved0: u8 = 0,
|
||||
reserved1: u16 = 0,
|
||||
reserved2: u32 = 0,
|
||||
/// The reporting driver's own device (the controller) — the child's parent.
|
||||
parent: u64,
|
||||
/// Where on the bus (for USB: the root port number, 1-based).
|
||||
bus_address: u64,
|
||||
/// Bus-specific identity (for USB: the PORTSC port-speed class; for PCI:
|
||||
/// the class triple; for ACPI devices, 0 — identity is the hid below).
|
||||
identity: u64,
|
||||
/// The kernel device id this child was `device_register`ed as — what the
|
||||
/// manager hands a matched driver as its argv assignment — or `no_device`
|
||||
/// for an unregistered leaf (a USB port before the descriptor track).
|
||||
device_id: u64 = no_device,
|
||||
/// The ACPI hardware id (`_HID`), EISA-decoded (e.g. "PNP0303"), for devices
|
||||
/// discovered by firmware string rather than a numeric bus identity. Empty
|
||||
/// (all zero) otherwise. Widens for FDT `compatible` strings later.
|
||||
hid: [8]u8 = .{0} ** 8,
|
||||
};
|
||||
|
||||
pub const child_added_size = @sizeOf(ChildAdded);
|
||||
|
||||
/// A bus driver reporting a device gone (hot-unplug). Not yet sent by any
|
||||
/// driver — the port scan has no unplug interrupt — but the manager handles it;
|
||||
/// death-pruning covers removal until hotplug lands.
|
||||
pub const ChildRemoved = extern struct {
|
||||
operation: u8 = @intFromEnum(Operation.child_removed),
|
||||
reserved0: u8 = 0,
|
||||
reserved1: u16 = 0,
|
||||
reserved2: u32 = 0,
|
||||
parent: u64,
|
||||
bus_address: u64,
|
||||
};
|
||||
|
||||
pub const child_removed_size = @sizeOf(ChildRemoved);
|
||||
|
||||
/// The manager's answer to a tree report.
|
||||
pub const ReportReply = extern struct {
|
||||
status: i32,
|
||||
reserved: u32 = 0,
|
||||
};
|
||||
|
||||
/// An application asking for the tree (M18.3): the reply is an EnumerateReply
|
||||
/// header followed by `count` ChildEntry records.
|
||||
pub const Enumerate = extern struct {
|
||||
operation: u8 = @intFromEnum(Operation.enumerate),
|
||||
reserved0: u8 = 0,
|
||||
reserved1: u16 = 0,
|
||||
reserved2: u32 = 0,
|
||||
};
|
||||
|
||||
pub const EnumerateReply = extern struct {
|
||||
status: i32,
|
||||
/// ChildEntry records following this header.
|
||||
count: u32,
|
||||
};
|
||||
|
||||
pub const ChildEntry = extern struct {
|
||||
parent: u64,
|
||||
bus_address: u64,
|
||||
identity: u64,
|
||||
};
|
||||
|
||||
/// An application subscribing to published add/remove events (the input-service
|
||||
/// pattern): the subscriber's endpoint rides as the call's **capability**, and
|
||||
/// events arrive on it as buffered messages whose payload is the same
|
||||
/// ChildAdded / ChildRemoved struct the bus drivers send — one encoding, both
|
||||
/// directions.
|
||||
pub const Subscribe = extern struct {
|
||||
operation: u8 = @intFromEnum(Operation.subscribe),
|
||||
reserved0: u8 = 0,
|
||||
reserved1: u16 = 0,
|
||||
reserved2: u32 = 0,
|
||||
};
|
||||
|
||||
/// Upper bound on any message in this protocol — sizes the endpoint buffers.
|
||||
/// Capped by the kernel's IPC MESSAGE_MAXIMUM (256): an EnumerateReply carries
|
||||
/// up to ten ChildEntry records per call, plenty for the mirror's current
|
||||
/// bounds; paging joins the protocol if a tree ever outgrows one message.
|
||||
pub const message_maximum = 256;
|
||||
@@ -1,20 +1,25 @@
|
||||
//! /system/services/device-manager — the ring-3 process that turns the device
|
||||
//! tree into a running system. The kernel enumerates the hardware and enforces the
|
||||
//! claim capability (mechanism); this decides *which driver serves which device*
|
||||
//! and, eventually, spawns it (policy). Keeping that split in user space is the
|
||||
//! whole point of the microkernel: the manager is an ordinary, restartable process
|
||||
//! with no special privilege — it uses the same `device_*` system calls any process
|
||||
//! could ([drivers.md](../../../docs/drivers.md), [driver-model.md]).
|
||||
//! tree into a running system: **the matcher and the supervisor**
|
||||
//! (docs/device-manager.md). The kernel enumerates the hardware and enforces the
|
||||
//! claim capability (mechanism); this decides which driver serves which device,
|
||||
//! spawns it, and keeps it alive (policy). Keeping that split in user space is
|
||||
//! the whole point of the microkernel: the manager is an ordinary, restartable
|
||||
//! process with no special privilege.
|
||||
//!
|
||||
//! Increment 2 (this file): enumerate /system/devices, *match* each device to a
|
||||
//! driver, and *spawn* it with `system_spawn` — the kernel loads the named binary
|
||||
//! from the initial-ramdisk as a fresh ring-3 process. On QEMU this discovers the
|
||||
//! HPET, decides `hpet` serves it, and brings that driver all the way up. (The
|
||||
//! kernel still auto-spawns the whole initial-ramdisk at boot; increment 3 removes
|
||||
//! that redundancy so the manager is the sole owner of driver spawning.)
|
||||
//! M18.1 (this increment): the manager is a harness service on the well-known
|
||||
//! `.device_manager` endpoint. Every driver is spawned **supervised** — exit
|
||||
//! notifications land in the same loop as protocol messages. Drivers with an
|
||||
//! assignment must `hello` within a deadline or be stopped; a driver that dies
|
||||
//! is restarted with backoff, and a crash loop (three fast deaths) marks it
|
||||
//! failed instead of respawning forever. Exit reasons (M17.2) drive the
|
||||
//! decision: a clean exit meant to stop; only faults and missed deadlines
|
||||
//! restart. Tree reports (`child_added`) land in M18.2.
|
||||
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
const acpi_ids = @import("acpi-ids");
|
||||
const pci_class = @import("pci-class");
|
||||
const protocol = runtime.device_manager_protocol;
|
||||
const device = runtime.device;
|
||||
const system = runtime.system;
|
||||
|
||||
@@ -26,87 +31,503 @@ fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
/// The driver that serves each device — the policy table. In a fuller system
|
||||
/// this comes from the drivers describing what they bind (or a manifest under
|
||||
/// /system/drivers); for now it is a small static map, which is enough to prove the
|
||||
/// manager reads the tree and decides. `null` = no driver for this class yet.
|
||||
fn driverFor(d: device.DeviceDescriptor) ?[]const u8 {
|
||||
// detect device via DeviceClass
|
||||
if (d.class == @intFromEnum(device.DeviceClass.timer)) return "hpet";
|
||||
// detect device via hid
|
||||
const hid = d.hid[0..@intCast(d.hid_len)];
|
||||
const id = acpi_ids.HardwareId.fromHid(hid) orelse return null;
|
||||
return switch (id) {
|
||||
.ps2_keyboard, .ps2_mouse => "ps2-bus",
|
||||
else => null,
|
||||
};
|
||||
}
|
||||
/// The PCI class/subclass/prog-IF triple of an xHCI (USB 3) host controller —
|
||||
/// Serial Bus Controller / USB Controller / XHCI — named from pci-class.zig rather
|
||||
/// than written as the bare 0x0C0330 (docs/coding-standards.md, "Named values").
|
||||
const xhci_pci_class: u64 = pci_class.ClassCode.pack(.{
|
||||
.base = @intFromEnum(pci_class.BaseClass.serial_bus),
|
||||
.subclass = @intFromEnum(pci_class.serial_bus.SubClass.usb),
|
||||
.prog_if = @intFromEnum(pci_class.serial_bus.usb.ProgIf.xhci),
|
||||
});
|
||||
|
||||
/// The PCI class/subclass/prog-IF triple of an xHCI (USB 3) host controller:
|
||||
/// Serial Bus Controller (0x0C) / USB Controller (0x03) / XHCI (0x30) — the names
|
||||
/// pci-class.zig decodes.
|
||||
const xhci_pci_class: u64 = 0x0C_03_30;
|
||||
|
||||
/// The bus driver that serves a PCI function, or null. Unlike the singleton drivers
|
||||
/// in `driverFor`, a machine can carry several identical controllers — so the caller
|
||||
/// spawns one driver instance *per device*, passing the device id as argv[1] for the
|
||||
/// instance to claim.
|
||||
fn pciDriverFor(d: device.DeviceDescriptor) ?[]const u8 {
|
||||
if (d.class != @intFromEnum(device.DeviceClass.pci_device)) return null;
|
||||
return switch (d.pci_class) {
|
||||
/// The driver that serves a *reported* PCI function (M19.3: matching moved
|
||||
/// from the boot snapshot to the bus reports), or null. A machine can carry
|
||||
/// several identical controllers — one driver instance per reported device,
|
||||
/// its registered id as argv[1].
|
||||
fn pciDriverForIdentity(identity: u64) ?[]const u8 {
|
||||
return switch (identity) {
|
||||
xhci_pci_class => "usb-xhci-bus",
|
||||
else => null,
|
||||
};
|
||||
}
|
||||
|
||||
/// Spawn one instance of `driver_name` to serve the specific device `id` — the id
|
||||
/// arrives as argv[1]. No isProcessRunning gate here: the name alone cannot tell two
|
||||
/// instances apart, and this manager is the sole spawner of drivers.
|
||||
fn spawnForDevice(driver_name: []const u8, id: u64) void {
|
||||
var text: [20]u8 = undefined;
|
||||
const id_text = std.fmt.bufPrint(&text, "{d}", .{id}) catch return;
|
||||
if (system.spawnWithArguments(driver_name, &.{id_text}) != null) {
|
||||
writeLine("device-manager: spawned {s} for device {d}\n", .{ driver_name, id });
|
||||
} else {
|
||||
writeLine("device-manager: failed to spawn {s} for device {d}\n", .{ driver_name, id });
|
||||
/// The driver that serves a *reported* ACPI device by its `_HID` (M20.3:
|
||||
/// ps2-bus now binds the PS/2 nodes the acpi service reports, not boot-snapshot
|
||||
/// nodes the kernel used to build). ps2-bus is a singleton that finds both its
|
||||
/// devices by hid once spawned, so keyboard and mouse map to the same name.
|
||||
fn hidDriverFor(hid: []const u8) ?[]const u8 {
|
||||
if (std.mem.eql(u8, hid, "PNP0303")) return "ps2-bus"; // PS/2 keyboard
|
||||
if (std.mem.eql(u8, hid, "PNP0F13")) return "ps2-bus"; // PS/2 mouse
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Whether some driver entry already serves registered device `device_id` —
|
||||
/// a re-report after a bus restart must not spawn a second instance.
|
||||
fn driverForDevice(device_id: u64) bool {
|
||||
for (&drivers) |*driver| {
|
||||
if (driver.used and driver.device_id == device_id) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
// --- supervision -------------------------------------------------------------
|
||||
|
||||
/// How long a protocol driver has to hello after its spawn.
|
||||
const hello_deadline_ms: u64 = 3000;
|
||||
/// Deaths faster than this count toward the crash loop; slower ones reset it.
|
||||
const fast_death_ns: u64 = 2_000_000_000;
|
||||
/// Consecutive fast deaths before the manager gives up on a driver.
|
||||
const crash_loop_cap: u32 = 3;
|
||||
/// Restart backoff: base << (restarts - 1), so 300 ms, 600 ms, 1200 ms.
|
||||
const backoff_base_ms: u64 = 300;
|
||||
|
||||
const DriverState = enum {
|
||||
awaiting_hello, // spawned; the deadline is armed (protocol drivers only)
|
||||
running,
|
||||
restarting, // dead; respawn due at restart_due_ns
|
||||
stopped, // exited cleanly — it meant to; not restarted
|
||||
failed, // crash loop, or unspawnable; the manager gave up
|
||||
};
|
||||
|
||||
const Driver = struct {
|
||||
used: bool = false,
|
||||
name_buffer: [24]u8 = undefined,
|
||||
name_len: usize = 0,
|
||||
// The assigned device id (becomes argv[1]), or protocol.no_device.
|
||||
device_id: u64 = protocol.no_device,
|
||||
// Whether this driver speaks the protocol (hello expected, deadline
|
||||
// enforced). Legacy drivers (e.g. ps2-bus) are supervised and restarted
|
||||
// but not yet required to hello.
|
||||
speaks_protocol: bool = false,
|
||||
process_id: u32 = 0,
|
||||
state: DriverState = .running,
|
||||
restarts: u32 = 0,
|
||||
spawn_ns: u64 = 0,
|
||||
hello_deadline_ns: u64 = 0,
|
||||
restart_due_ns: u64 = 0,
|
||||
|
||||
fn name(driver: *const Driver) []const u8 {
|
||||
return driver.name_buffer[0..driver.name_len];
|
||||
}
|
||||
};
|
||||
|
||||
const maximum_drivers = 16;
|
||||
var drivers: [maximum_drivers]Driver = .{Driver{}} ** maximum_drivers;
|
||||
var manager_endpoint: runtime.ipc.Handle = 0;
|
||||
var test_restart_mode = false;
|
||||
var test_usb_restart_mode = false;
|
||||
var test_usb_killed = false;
|
||||
var test_pci_restart_mode = false;
|
||||
var test_kill_pid: u32 = 0;
|
||||
var test_kill_due_ns: u64 = 0;
|
||||
|
||||
/// The application subscribers (M18.3, the input-service pattern): endpoints
|
||||
/// handed over as capabilities, each receiving every child add/remove as a
|
||||
/// buffered message. A subscriber whose endpoint stops accepting (it died) is
|
||||
/// dropped on the failed send.
|
||||
const maximum_subscribers = 8;
|
||||
var subscribers: [maximum_subscribers]?runtime.ipc.Handle = .{null} ** maximum_subscribers;
|
||||
|
||||
/// Publish one event (a ChildAdded or ChildRemoved struct, the same encoding
|
||||
/// the bus drivers send) to every subscriber.
|
||||
fn publishEvent(event: []const u8) void {
|
||||
for (&subscribers) |*slot| {
|
||||
if (slot.*) |handle| {
|
||||
if (!runtime.ipc.send(handle, event)) slot.* = null; // dead subscriber
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub fn main() void {
|
||||
/// The manager's mirror of what bus drivers report (docs/device-manager.md "the
|
||||
/// tree"): the children, keyed by (parent, bus address), each remembering which
|
||||
/// driver instance reported it — that is what death-pruning sweeps by.
|
||||
const Child = struct {
|
||||
used: bool = false,
|
||||
parent: u64 = 0,
|
||||
bus_address: u64 = 0,
|
||||
identity: u64 = 0,
|
||||
// The kernel device id (registered by the reporter), or protocol.no_device.
|
||||
device_id: u64 = 0,
|
||||
reporter: u32 = 0, // the reporting driver instance's process id
|
||||
};
|
||||
|
||||
const maximum_children = 64; // ACPI adds ~34 device nodes (M20.2), plus PCI + USB
|
||||
var children: [maximum_children]Child = .{Child{}} ** maximum_children;
|
||||
|
||||
/// Record (or refresh) a reported child. Refreshing matters: a restarted bus
|
||||
/// driver re-reports what it rediscovers, and the same (parent, port) must not
|
||||
/// duplicate.
|
||||
fn addChild(parent: u64, bus_address: u64, identity: u64, device_id: u64, reporter: u32) bool {
|
||||
var free: ?*Child = null;
|
||||
for (&children) |*child| {
|
||||
if (child.used and child.parent == parent and child.bus_address == bus_address) {
|
||||
child.identity = identity;
|
||||
child.device_id = device_id;
|
||||
child.reporter = reporter;
|
||||
return true;
|
||||
}
|
||||
if (!child.used and free == null) free = child;
|
||||
}
|
||||
const slot = free orelse return false;
|
||||
slot.* = .{ .used = true, .parent = parent, .bus_address = bus_address, .identity = identity, .device_id = device_id, .reporter = reporter };
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Prune every child a dead driver instance reported: the children describe
|
||||
/// protocol state (slots, rings) that died with the process — keeping the nodes
|
||||
/// would be keeping a lie. The restarted instance rediscovers and re-reports.
|
||||
/// Watchers hear the honest story: removed now, added again on rediscovery.
|
||||
fn pruneChildrenOf(reporter: u32) void {
|
||||
for (&children) |*child| {
|
||||
if (child.used and child.reporter == reporter) {
|
||||
writeLine("/system/services/device-manager: child removed (device {d} port {d})\n", .{ child.parent, child.bus_address });
|
||||
child.used = false;
|
||||
const event = protocol.ChildRemoved{ .parent = child.parent, .bus_address = child.bus_address };
|
||||
publishEvent(std.mem.asBytes(&event));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// How many children a driver instance has reported (the test-usb-restart
|
||||
/// trigger counts these).
|
||||
fn childCountOf(reporter: u32) u32 {
|
||||
var n: u32 = 0;
|
||||
for (&children) |*child| {
|
||||
if (child.used and child.reporter == reporter) n += 1;
|
||||
}
|
||||
return n;
|
||||
}
|
||||
|
||||
fn driverByProcess(process_id: u32) ?*Driver {
|
||||
for (&drivers) |*driver| {
|
||||
if (driver.used and driver.process_id == process_id) return driver;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Whether a singleton driver is already in the table (two ACPI nodes can both
|
||||
/// map to ps2-bus; one instance serves both).
|
||||
fn alreadySupervised(name: []const u8) bool {
|
||||
for (&drivers) |*driver| {
|
||||
if (driver.used and std.mem.eql(u8, driver.name(), name)) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/// Record a driver in the table and spawn its first instance.
|
||||
fn addDriver(name: []const u8, device_id: u64, speaks_protocol: bool) void {
|
||||
for (&drivers) |*driver| {
|
||||
if (driver.used) continue;
|
||||
const n = @min(name.len, driver.name_buffer.len);
|
||||
@memcpy(driver.name_buffer[0..n], name[0..n]);
|
||||
driver.name_len = n;
|
||||
driver.device_id = device_id;
|
||||
driver.speaks_protocol = speaks_protocol;
|
||||
driver.used = true;
|
||||
spawnDriver(driver);
|
||||
return;
|
||||
}
|
||||
writeLine("/system/services/device-manager: driver table full; cannot supervise {s}\n", .{name});
|
||||
}
|
||||
|
||||
/// (Re)spawn a driver instance: supervised on the manager's own endpoint, the
|
||||
/// device id as argv[1] when it has one, the hello deadline armed when it
|
||||
/// speaks the protocol.
|
||||
fn spawnDriver(driver: *Driver) void {
|
||||
var id_text: [20]u8 = undefined;
|
||||
var arguments: [1][]const u8 = undefined;
|
||||
var argument_count: usize = 0;
|
||||
if (driver.device_id != protocol.no_device) {
|
||||
arguments[0] = std.fmt.bufPrint(&id_text, "{d}", .{driver.device_id}) catch return;
|
||||
argument_count = 1;
|
||||
}
|
||||
const child = system.spawnSupervised(driver.name(), arguments[0..argument_count], manager_endpoint) orelse {
|
||||
writeLine("/system/services/device-manager: failed to spawn {s}\n", .{driver.name()});
|
||||
driver.state = .failed;
|
||||
return;
|
||||
};
|
||||
driver.process_id = child;
|
||||
driver.spawn_ns = system.clock();
|
||||
if (driver.speaks_protocol) {
|
||||
driver.state = .awaiting_hello;
|
||||
driver.hello_deadline_ns = driver.spawn_ns + hello_deadline_ms * 1_000_000;
|
||||
_ = system.timerOnce(manager_endpoint, hello_deadline_ms + 100);
|
||||
} else {
|
||||
driver.state = .running;
|
||||
}
|
||||
if (driver.device_id != protocol.no_device) {
|
||||
writeLine("/system/services/device-manager: spawned {s} for device {d}\n", .{ driver.name(), driver.device_id });
|
||||
} else {
|
||||
writeLine("/system/services/device-manager: spawned {s}\n", .{driver.name()});
|
||||
}
|
||||
}
|
||||
|
||||
/// A driver died. Prune what it reported first — then the exit reason (M17.2)
|
||||
/// is the whole restart decision: a clean exit meant to stop; anything else
|
||||
/// restarts with backoff until the crash-loop cap.
|
||||
fn onDriverExit(driver: *Driver) void {
|
||||
pruneChildrenOf(driver.process_id);
|
||||
const reason = runtime.process.exitReason(driver.process_id) orelse .fault;
|
||||
if (reason == .exited) {
|
||||
driver.state = .stopped;
|
||||
writeLine("/system/services/device-manager: {s} exited cleanly; not restarting\n", .{driver.name()});
|
||||
return;
|
||||
}
|
||||
const now = system.clock();
|
||||
const alive_ns = now - driver.spawn_ns;
|
||||
driver.restarts = if (alive_ns < fast_death_ns) driver.restarts + 1 else 1;
|
||||
if (driver.restarts >= crash_loop_cap) {
|
||||
driver.state = .failed;
|
||||
writeLine("/system/services/device-manager: {s} is failing repeatedly (crash loop); giving up\n", .{driver.name()});
|
||||
return;
|
||||
}
|
||||
const delay_ms = backoff_base_ms << @intCast(driver.restarts - 1);
|
||||
driver.state = .restarting;
|
||||
driver.restart_due_ns = now + delay_ms * 1_000_000;
|
||||
writeLine("/system/services/device-manager: restarting {s} in {d} ms (died: {s})\n", .{ driver.name(), delay_ms, @tagName(reason) });
|
||||
_ = system.timerOnce(manager_endpoint, delay_ms + 50);
|
||||
}
|
||||
|
||||
/// A timer landed: sweep every deadline. Overdue hellos are killed (the exit
|
||||
/// notification then routes through the normal restart policy); due restarts
|
||||
/// respawn. Timers carry no id on purpose — the table is the state, and one
|
||||
/// sweep serves every armed deadline.
|
||||
fn sweepDeadlines() void {
|
||||
const now = system.clock();
|
||||
if (test_kill_pid != 0 and now >= test_kill_due_ns) {
|
||||
writeLine("/system/services/device-manager: test mode: killing the reporter\n", .{});
|
||||
_ = system.kill(test_kill_pid);
|
||||
test_kill_pid = 0;
|
||||
}
|
||||
for (&drivers) |*driver| {
|
||||
if (!driver.used) continue;
|
||||
switch (driver.state) {
|
||||
.awaiting_hello => if (now >= driver.hello_deadline_ns) {
|
||||
writeLine("/system/services/device-manager: {s} missed its hello deadline\n", .{driver.name()});
|
||||
_ = system.kill(driver.process_id);
|
||||
// The exit notification finishes the job via onDriverExit.
|
||||
},
|
||||
.restarting => if (now >= driver.restart_due_ns) spawnDriver(driver),
|
||||
else => {},
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- the harness callbacks -----------------------------------------------------
|
||||
|
||||
fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||
manager_endpoint = endpoint;
|
||||
|
||||
// Enumerate into a heap buffer (too big for the one-page user stack).
|
||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||
_ = runtime.system.write("device-manager: out of memory\n");
|
||||
return;
|
||||
_ = runtime.system.write("/system/services/device-manager: out of memory\n");
|
||||
return false;
|
||||
};
|
||||
const total = device.enumerate(buffer);
|
||||
const count = @min(total, buffer.len);
|
||||
|
||||
var matched: usize = 0;
|
||||
for (buffer[0..count]) |descriptor| {
|
||||
if (pciDriverFor(descriptor)) |driver_name| {
|
||||
if (descriptor.class == @intFromEnum(device.DeviceClass.pci_host_bridge)) {
|
||||
// The PCI bus driver: enumeration in ring 3 (M19), one instance
|
||||
// per bridge, the bridge id as its assignment.
|
||||
matched += 1;
|
||||
spawnForDevice(driver_name, descriptor.id);
|
||||
addDriver("pci-bus", descriptor.id, true);
|
||||
continue;
|
||||
}
|
||||
const driver_name = driverFor(descriptor) orelse continue;
|
||||
matched += 1;
|
||||
if (!system.isProcessRunning(driver_name)) {
|
||||
if (runtime.system.spawn(driver_name) != null) {
|
||||
writeLine("device-manager: spawned {s}\n", .{driver_name});
|
||||
} else {
|
||||
writeLine("device-manager: failed to spawn {s}\n", .{driver_name});
|
||||
}
|
||||
} else {
|
||||
writeLine("device-manager: already spawned {s}\n", .{driver_name});
|
||||
}
|
||||
// Nothing else is matched from the boot snapshot today. The kernel-seeded
|
||||
// HPET timer node is served by the kernel's own clock (docs/timers.md), not
|
||||
// a user-space driver; PCI functions and PS/2 _HID devices arrive later as
|
||||
// pci-bus / acpi-service reports and match in onChildAdded (docs/discovery.md).
|
||||
// A fuller system's static class->driver manifest (docs/device-manager.md)
|
||||
// would slot in here.
|
||||
}
|
||||
|
||||
// The discovery service (docs/discovery.md): one per firmware, packed
|
||||
// under the neutral name "discovery", spawned once at startup. It finds and
|
||||
// claims the acpi-tables (or devicetree-blob) node itself. Not a per-device
|
||||
// match — it is the discoverer, not a driver bound to one device.
|
||||
addDriver("discovery", protocol.no_device, false);
|
||||
|
||||
if (test_restart_mode) {
|
||||
// The driver-restart scenario's fixture: claims device 0 (the tree
|
||||
// root, otherwise unclaimed), hellos, then faults — driving backoff,
|
||||
// re-claim-after-death, and the crash-loop cap deterministically.
|
||||
addDriver("crash-test", 0, true);
|
||||
}
|
||||
|
||||
if (matched == 0) {
|
||||
_ = runtime.system.write("device-manager: no matchable devices\n");
|
||||
_ = runtime.system.write("/system/services/device-manager: no matchable devices\n");
|
||||
} else {
|
||||
_ = runtime.system.write("/system/services/device-manager: ok\n");
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?runtime.ipc.Handle) usize {
|
||||
if (message.len < 1) return 0;
|
||||
switch (message[0]) {
|
||||
@intFromEnum(protocol.Operation.child_added) => return onChildAdded(message, reply, sender),
|
||||
@intFromEnum(protocol.Operation.child_removed) => return onChildRemoved(message, reply, sender),
|
||||
@intFromEnum(protocol.Operation.enumerate) => return onEnumerate(reply),
|
||||
@intFromEnum(protocol.Operation.subscribe) => return onSubscribe(reply, capability),
|
||||
@intFromEnum(protocol.Operation.hello) => {},
|
||||
else => return 0,
|
||||
}
|
||||
if (message.len < protocol.hello_size) return 0;
|
||||
const hello = std.mem.bytesToValue(protocol.Hello, message[0..protocol.hello_size]);
|
||||
|
||||
var status: i32 = 0;
|
||||
if (hello.version != protocol.version) {
|
||||
status = -1;
|
||||
writeLine("/system/services/device-manager: refused hello (version {d}) from process {d}\n", .{ hello.version, sender });
|
||||
} else if (driverByProcess(sender)) |driver| {
|
||||
driver.state = .running;
|
||||
writeLine("/system/services/device-manager: hello from {s} (device {d})\n", .{ driver.name(), hello.device_id });
|
||||
} else {
|
||||
status = -1;
|
||||
writeLine("/system/services/device-manager: hello from unknown process {d}\n", .{sender});
|
||||
}
|
||||
const hello_reply = protocol.HelloReply{ .status = status };
|
||||
@memcpy(reply[0..protocol.reply_size], std.mem.asBytes(&hello_reply));
|
||||
return protocol.reply_size;
|
||||
}
|
||||
|
||||
/// A bus driver reported a discovered device: mirror it, and in
|
||||
/// test-usb-restart mode kill the reporter once after its second child — the
|
||||
/// deterministic trigger for prune -> backoff -> respawn -> re-report.
|
||||
fn onChildAdded(message: []const u8, reply: []u8, sender: u32) usize {
|
||||
if (message.len < protocol.child_added_size) return 0;
|
||||
const report = std.mem.bytesToValue(protocol.ChildAdded, message[0..protocol.child_added_size]);
|
||||
var status: i32 = 0;
|
||||
if (driverByProcess(sender)) |driver| {
|
||||
if (!addChild(report.parent, report.bus_address, report.identity, report.device_id, sender)) status = -1;
|
||||
writeLine("/system/services/device-manager: child added (device {d} port {d}, identity {d}) by {s}\n", .{ report.parent, report.bus_address, report.identity, driver.name() });
|
||||
if (status == 0) publishEvent(message[0..protocol.child_added_size]);
|
||||
// Matching from reports (M19.3): a registered child whose identity
|
||||
// names a driver gets one, once — re-reports after a bus restart
|
||||
// dedupe on the registered id, exactly like the registrations do.
|
||||
if (status == 0 and report.device_id != protocol.no_device) {
|
||||
if (pciDriverForIdentity(report.identity)) |child_driver| {
|
||||
if (!driverForDevice(report.device_id)) addDriver(child_driver, report.device_id, true);
|
||||
}
|
||||
// ACPI _HID match (M20.3): ps2-bus is a singleton that finds its own
|
||||
// devices by hid, so spawn it once, without a device assignment.
|
||||
const hid_len = std.mem.indexOfScalar(u8, &report.hid, 0) orelse report.hid.len;
|
||||
if (hid_len != 0) {
|
||||
if (hidDriverFor(report.hid[0..hid_len])) |hid_driver| {
|
||||
if (!alreadySupervised(hid_driver)) addDriver(hid_driver, protocol.no_device, false);
|
||||
}
|
||||
}
|
||||
}
|
||||
} else {
|
||||
status = -1;
|
||||
}
|
||||
const report_reply = protocol.ReportReply{ .status = status };
|
||||
@memcpy(reply[0..@sizeOf(protocol.ReportReply)], std.mem.asBytes(&report_reply));
|
||||
if (test_pci_restart_mode and !test_usb_killed) {
|
||||
if (driverByProcess(sender)) |driver| {
|
||||
if (std.mem.eql(u8, driver.name(), "pci-bus") and childCountOf(sender) >= 3) {
|
||||
// The pci restart drill: kill the enumerator after it has
|
||||
// reported; the respawn must re-register without duplicates
|
||||
// (M19.0 idempotence, proven end to end by pci-scan).
|
||||
test_usb_killed = true;
|
||||
test_kill_pid = sender;
|
||||
test_kill_due_ns = system.clock() + 1_000_000_000;
|
||||
_ = system.timerOnce(manager_endpoint, 1100);
|
||||
}
|
||||
}
|
||||
}
|
||||
if (test_usb_restart_mode and !test_usb_killed and childCountOf(sender) >= 2) {
|
||||
// Only the xHCI reporter is the drill's victim — pci-bus also reports
|
||||
// now, and whichever finishes second must not trigger the kill.
|
||||
if (driverByProcess(sender)) |driver| {
|
||||
if (std.mem.eql(u8, driver.name(), "usb-xhci-bus")) {
|
||||
// Delayed, not immediate: the device-list scenario's subscriber
|
||||
// needs a window to enumerate and subscribe before the events.
|
||||
test_usb_killed = true;
|
||||
test_kill_pid = sender;
|
||||
test_kill_due_ns = system.clock() + 2_000_000_000;
|
||||
_ = system.timerOnce(manager_endpoint, 2100);
|
||||
}
|
||||
}
|
||||
}
|
||||
return @sizeOf(protocol.ReportReply);
|
||||
}
|
||||
|
||||
/// A bus driver reported a device gone (hot-unplug; no sender exists yet, but
|
||||
/// the handler is protocol-complete — death-pruning covers removal until then).
|
||||
fn onChildRemoved(message: []const u8, reply: []u8, sender: u32) usize {
|
||||
if (message.len < protocol.child_removed_size) return 0;
|
||||
const report = std.mem.bytesToValue(protocol.ChildRemoved, message[0..protocol.child_removed_size]);
|
||||
var status: i32 = -1;
|
||||
for (&children) |*child| {
|
||||
if (child.used and child.parent == report.parent and child.bus_address == report.bus_address and child.reporter == sender) {
|
||||
writeLine("/system/services/device-manager: child removed (device {d} port {d})\n", .{ child.parent, child.bus_address });
|
||||
child.used = false;
|
||||
status = 0;
|
||||
}
|
||||
}
|
||||
const report_reply = protocol.ReportReply{ .status = status };
|
||||
@memcpy(reply[0..@sizeOf(protocol.ReportReply)], std.mem.asBytes(&report_reply));
|
||||
return @sizeOf(protocol.ReportReply);
|
||||
}
|
||||
|
||||
/// An application asked for the tree: the mirror, as a header plus entries.
|
||||
fn onEnumerate(reply: []u8) usize {
|
||||
var count: u32 = 0;
|
||||
var offset: usize = @sizeOf(protocol.EnumerateReply);
|
||||
for (&children) |*child| {
|
||||
if (!child.used) continue;
|
||||
if (offset + @sizeOf(protocol.ChildEntry) > reply.len) break;
|
||||
const entry = protocol.ChildEntry{ .parent = child.parent, .bus_address = child.bus_address, .identity = child.identity };
|
||||
@memcpy(reply[offset..][0..@sizeOf(protocol.ChildEntry)], std.mem.asBytes(&entry));
|
||||
offset += @sizeOf(protocol.ChildEntry);
|
||||
count += 1;
|
||||
}
|
||||
const header = protocol.EnumerateReply{ .status = 0, .count = count };
|
||||
@memcpy(reply[0..@sizeOf(protocol.EnumerateReply)], std.mem.asBytes(&header));
|
||||
return offset;
|
||||
}
|
||||
|
||||
/// An application subscribed: its endpoint arrived as the call's capability.
|
||||
fn onSubscribe(reply: []u8, capability: ?runtime.ipc.Handle) usize {
|
||||
var status: i32 = -1;
|
||||
if (capability) |handle| {
|
||||
for (&subscribers) |*slot| {
|
||||
if (slot.* == null) {
|
||||
slot.* = handle;
|
||||
status = 0;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
const report_reply = protocol.ReportReply{ .status = status };
|
||||
@memcpy(reply[0..@sizeOf(protocol.ReportReply)], std.mem.asBytes(&report_reply));
|
||||
return @sizeOf(protocol.ReportReply);
|
||||
}
|
||||
|
||||
fn onNotification(badge: u64) void {
|
||||
if (badge & runtime.ipc.notify_exit_bit != 0) {
|
||||
const dead: u32 = @intCast(badge & ~(runtime.ipc.notify_badge_bit | runtime.ipc.notify_exit_bit));
|
||||
if (driverByProcess(dead)) |driver| onDriverExit(driver);
|
||||
return;
|
||||
}
|
||||
_ = runtime.system.write("device-manager: ok\n");
|
||||
while (true) runtime.system.sleep(1000);
|
||||
if (badge & runtime.ipc.notify_timer_bit != 0) sweepDeadlines();
|
||||
}
|
||||
|
||||
pub fn main(init: runtime.process.Init) void {
|
||||
if (init.arguments.get(1)) |mode| {
|
||||
test_restart_mode = std.mem.eql(u8, mode, "test-restart");
|
||||
test_usb_restart_mode = std.mem.eql(u8, mode, "test-usb-restart");
|
||||
test_pci_restart_mode = std.mem.eql(u8, mode, "test-pci-restart");
|
||||
}
|
||||
runtime.service.run(protocol.message_maximum, .{
|
||||
.service = .device_manager,
|
||||
.init = initialise,
|
||||
.on_message = onMessage,
|
||||
.on_notification = onNotification,
|
||||
});
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
|
||||
@@ -0,0 +1,35 @@
|
||||
//! /system/services/fdt — the devicetree discovery service: the ARM twin of the
|
||||
//! acpi service (docs/discovery.md — firmware neutrality). **Placeholder: not
|
||||
//! implemented.** It exists so the build's `-Ddiscovery` option has both of its
|
||||
//! values from day one; the implementation lands with the Raspberry Pi
|
||||
//! bring-up (docs/arm.md).
|
||||
//!
|
||||
//! What it becomes: the per-firmware discoverer for boots that hand over a
|
||||
//! flattened device tree instead of ACPI tables. It claims the
|
||||
//! `devicetree-blob` node the kernel publishes (the FDT the loader received),
|
||||
//! walks the tree — pure data, no bytecode, so unlike the acpi service it
|
||||
//! needs no port grant and no interpreter — and, like any bus-shaped driver:
|
||||
//! `device_register`s what it finds (containment against the blob node's
|
||||
//! recorded apertures), reports each child to the device manager
|
||||
//! (`child_added`, identity = the node's `compatible` string), and stays
|
||||
//! resident under the manager's supervision (hello, restart, the usual
|
||||
//! contract).
|
||||
//!
|
||||
//! Known prerequisite recorded in docs/discovery.md: `DeviceDescriptor`'s 8-byte `hid`
|
||||
//! cannot hold an FDT `compatible` string ("brcm,bcm2835-aux-uart") — identity
|
||||
//! widens before this file grows a body.
|
||||
|
||||
const runtime = @import("runtime");
|
||||
|
||||
pub fn main(init: runtime.process.Init) void {
|
||||
_ = init;
|
||||
// Not implemented: exit cleanly and silently (a bare spawn by the
|
||||
// initial-ramdisk sweep must not derange other tests' markers). The
|
||||
// supervisor reads a clean exit as "meant to stop" — correct for a
|
||||
// placeholder.
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||
}
|
||||
@@ -6,12 +6,20 @@
|
||||
//!
|
||||
//! It proves the C-convention heap works, then — as PID 1 — acts as the system's
|
||||
//! **service supervisor**: it spawns the user-space services danos brings up at boot
|
||||
//! (the VFS server, the device manager), and settles into a heartbeat so it stays
|
||||
//! alive as the root of user space. Drivers are *not* its job: the device manager
|
||||
//! discovers the hardware and spawns those. This is the service half of the
|
||||
//! service/driver spawn split (docs/driver-model.md).
|
||||
//! (the VFS server, the device manager), and settles into an event loop as the root
|
||||
//! of user space. Drivers are *not* its job: the device manager discovers the
|
||||
//! hardware and spawns those. This is the service half of the service/driver spawn
|
||||
//! split (docs/driver-model.md).
|
||||
//!
|
||||
//! M21: init also owns **orderly shutdown**. It supervises its children (keeping
|
||||
//! their ids and an exit endpoint), subscribes to the power service, and on a
|
||||
//! power-button event runs the stop sequence over its children in reverse order
|
||||
//! before asking the power service to enter S5 — lifecycle (M17) and events (M21)
|
||||
//! composing into a clean poweroff.
|
||||
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
const power = runtime.power_protocol;
|
||||
|
||||
/// The system services init brings up at boot, in order. This is init's policy — the
|
||||
/// microkernel keeps such choices in user space, not the kernel. Drivers are absent
|
||||
@@ -19,6 +27,10 @@ const runtime = @import("runtime");
|
||||
/// manifest under /system/services instead of a hardcoded list.)
|
||||
const boot_services = [_][]const u8{ "vfs", "input", "device-manager" };
|
||||
|
||||
var children: [boot_services.len]u32 = .{0} ** boot_services.len;
|
||||
var child_count: usize = 0;
|
||||
var supervision_endpoint: runtime.ipc.Handle = 0;
|
||||
|
||||
pub fn main() void {
|
||||
// Prove the heap end to end: allocate through the runtime allocator (which
|
||||
// mmaps pages from the kernel and carves them with the free list), write into
|
||||
@@ -28,25 +40,100 @@ pub fn main() void {
|
||||
// the extern malloc/free symbols; Zig code uses this allocator.)
|
||||
const gpa = runtime.allocator();
|
||||
if (gpa.alloc(u8, 64)) |buffer| {
|
||||
const message = "init: heap ok\n";
|
||||
const message = "/system/services/init: heap ok\n";
|
||||
@memcpy(buffer[0..message.len], message);
|
||||
_ = runtime.system.write(buffer[0..message.len]);
|
||||
gpa.free(buffer);
|
||||
} else |_| {}
|
||||
|
||||
// Bring up the boot services. Best-effort and silent: each service announces its
|
||||
// own readiness (`vfs: ready`, ...), and in an isolation test that runs init with
|
||||
// no initial-ramdisk the spawns simply no-op rather than deranging the heartbeat.
|
||||
// One endpoint carries everything init waits on: children's exit
|
||||
// notifications (they are spawned supervised against it), init's own
|
||||
// signals, and power events it subscribes to. All arrive in the loop below.
|
||||
supervision_endpoint = runtime.ipc.createIpcEndpoint() orelse {
|
||||
_ = runtime.system.write("/system/services/init: no endpoint\n");
|
||||
return;
|
||||
};
|
||||
_ = runtime.process.bindSignals(supervision_endpoint);
|
||||
|
||||
// Bring up the boot services, supervised so init can stop them cleanly.
|
||||
// Best-effort and silent: each service announces its own readiness, and in
|
||||
// an isolation test with no initial-ramdisk the spawns simply no-op.
|
||||
for (boot_services) |service| {
|
||||
_ = runtime.system.spawn(service);
|
||||
if (runtime.system.spawnSupervised(service, &.{}, supervision_endpoint)) |id| {
|
||||
children[child_count] = id;
|
||||
child_count += 1;
|
||||
}
|
||||
}
|
||||
|
||||
// Subscribe to power events (retry: the power service registers well after
|
||||
// init starts). Best-effort — without it, a `terminate` signal still
|
||||
// triggers the same shutdown path.
|
||||
subscribePower();
|
||||
|
||||
// A re-arming timer drives the liveness heartbeat: proof PID 1 is alive
|
||||
// (the init test's marker) while the loop stays free to receive signals,
|
||||
// power events, and children's exit notifications.
|
||||
_ = runtime.system.timerOnce(supervision_endpoint, 1000);
|
||||
|
||||
var receive: [power.message_maximum]u8 = undefined;
|
||||
while (true) {
|
||||
_ = runtime.system.write("init: heartbeat\n");
|
||||
runtime.system.sleep(1000);
|
||||
const got = runtime.ipc.replyWait(supervision_endpoint, &.{}, &receive, null);
|
||||
if (runtime.process.signalsFrom(got.badge)) |signals| {
|
||||
if (signals.has(.terminate)) shutDown();
|
||||
continue;
|
||||
}
|
||||
if (got.isTimer()) {
|
||||
_ = runtime.system.write("/system/services/init: heartbeat\n");
|
||||
_ = runtime.system.timerOnce(supervision_endpoint, 1000);
|
||||
continue;
|
||||
}
|
||||
if (got.isMessage() and got.len >= 2 and receive[0] == @intFromEnum(power.Operation.event)) {
|
||||
// A power event (the only buffered messages init receives).
|
||||
if (receive[1] == @intFromEnum(power.Event.power_button)) shutDown();
|
||||
continue;
|
||||
}
|
||||
// Child-exit notifications and anything else: keep waiting.
|
||||
if (got.isNotification()) continue;
|
||||
}
|
||||
}
|
||||
|
||||
/// Look up the power service and subscribe our endpoint (handed over as the
|
||||
/// call's capability) so events arrive as buffered messages here.
|
||||
fn subscribePower() void {
|
||||
var handle: ?runtime.ipc.Handle = null;
|
||||
var tries: u32 = 0;
|
||||
while (handle == null and tries < 200) : (tries += 1) {
|
||||
handle = runtime.ipc.lookup(.power);
|
||||
if (handle == null) runtime.system.sleep(20);
|
||||
}
|
||||
// A missing power service is not fatal — init proceeds to its heartbeat and
|
||||
// a `terminate` signal still drives shutdown. Silent so the no-ramdisk init
|
||||
// test's heartbeat marker is the next line written.
|
||||
const h = handle orelse return;
|
||||
const request = power.Subscribe{};
|
||||
var reply: [power.message_maximum]u8 = undefined;
|
||||
_ = runtime.ipc.callCap(h, std.mem.asBytes(&request), &reply, supervision_endpoint) catch {};
|
||||
}
|
||||
|
||||
/// The stop sequence: terminate each child in reverse spawn order (vfs last —
|
||||
/// other services may flush through it), waiting up to a deadline for each to
|
||||
/// exit before killing it, then ask the power service to enter S5.
|
||||
fn shutDown() void {
|
||||
_ = runtime.system.write("/system/services/init: shutting down\n");
|
||||
var i = child_count;
|
||||
while (i > 0) {
|
||||
i -= 1;
|
||||
if (children[i] != 0) runtime.process.stop(children[i], 2000, supervision_endpoint);
|
||||
}
|
||||
if (runtime.ipc.lookup(.power)) |h| {
|
||||
const request = power.Shutdown{};
|
||||
var reply: [power.message_maximum]u8 = undefined;
|
||||
_ = runtime.ipc.call(h, std.mem.asBytes(&request), &reply) catch {};
|
||||
}
|
||||
// If S5 did not take, init has nothing left to do but idle.
|
||||
while (true) runtime.system.sleep(1000);
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||
|
||||
@@ -115,14 +115,14 @@ fn handle(message: []const u8, got: ipc.Received, out: []u8) usize {
|
||||
|
||||
pub fn main() void {
|
||||
const endpoint = ipc.createIpcEndpoint() orelse {
|
||||
_ = system.write("input: no endpoint\n");
|
||||
_ = system.write("/system/services/input: no endpoint\n");
|
||||
return;
|
||||
};
|
||||
if (!ipc.register(.input, endpoint)) {
|
||||
_ = system.write("input: register failed\n");
|
||||
_ = system.write("/system/services/input: register failed\n");
|
||||
return;
|
||||
}
|
||||
_ = system.write("input: ready\n");
|
||||
_ = system.write("/system/services/input: ready\n");
|
||||
|
||||
var reply_buffer: [protocol.reply_size]u8 = undefined;
|
||||
var reply_len: usize = 0;
|
||||
|
||||
@@ -0,0 +1,68 @@
|
||||
//! The power protocol (docs/power.md): system power's domain-named surface,
|
||||
//! registered under `ServiceId.power`. On x86 the acpi service serves it; on
|
||||
//! ARM a PSCI/mailbox service will register the same id — subscribers never
|
||||
//! learn which firmware they are on (docs/discovery.md — firmware neutrality).
|
||||
//! The vfs-protocol pattern: extern-struct messages, a version, reserved fields.
|
||||
|
||||
/// The protocol version a client states nowhere yet — reserved for the day a
|
||||
/// handshake needs it; requests carry it so a mismatch can be refused loudly.
|
||||
pub const version: u16 = 1;
|
||||
|
||||
pub const Operation = enum(u8) {
|
||||
/// Subscribe to power events: the subscriber's endpoint rides as the
|
||||
/// call's capability (the input/device-manager pattern); events arrive on
|
||||
/// it as buffered messages carrying an `EventMessage`.
|
||||
subscribe = 1,
|
||||
/// Orderly shutdown's last step: enter S5. Accepted only from PID 1
|
||||
/// (init) — the process that has already run the stop sequence over
|
||||
/// everything else.
|
||||
shutdown = 2,
|
||||
/// The published event payload (never sent *to* the service).
|
||||
event = 3,
|
||||
};
|
||||
|
||||
/// What happened. The vocabulary is hardware-neutral: a lid is a lid whether
|
||||
/// ACPI or a PSCI mailbox reported it.
|
||||
pub const Event = enum(u8) {
|
||||
power_button = 1,
|
||||
lid = 2,
|
||||
ac = 3,
|
||||
battery = 4,
|
||||
/// A device notification that maps to none of the named events — the
|
||||
/// `code` and `hid` fields say which device and what code.
|
||||
notify = 5,
|
||||
};
|
||||
|
||||
pub const Subscribe = extern struct {
|
||||
operation: u8 = @intFromEnum(Operation.subscribe),
|
||||
reserved0: u8 = 0,
|
||||
version: u16 = version,
|
||||
reserved1: u32 = 0,
|
||||
};
|
||||
|
||||
pub const Shutdown = extern struct {
|
||||
operation: u8 = @intFromEnum(Operation.shutdown),
|
||||
reserved0: u8 = 0,
|
||||
version: u16 = version,
|
||||
reserved1: u32 = 0,
|
||||
};
|
||||
|
||||
/// A published event, as the buffered-message payload subscribers receive.
|
||||
pub const EventMessage = extern struct {
|
||||
operation: u8 = @intFromEnum(Operation.event),
|
||||
/// An Event value.
|
||||
event: u8,
|
||||
reserved0: u16 = 0,
|
||||
/// The device notification code (Notify's second argument), or 0.
|
||||
code: u32 = 0,
|
||||
/// The notifying device's hardware id (EISA-decoded), or all zero.
|
||||
hid: [8]u8 = .{0} ** 8,
|
||||
};
|
||||
|
||||
pub const Reply = extern struct {
|
||||
status: i32,
|
||||
reserved: u32 = 0,
|
||||
};
|
||||
|
||||
/// Upper bound on any message in this protocol — sizes endpoint buffers.
|
||||
pub const message_maximum = 64;
|
||||
@@ -51,8 +51,9 @@ fn awaitChildExit(endpoint: runtime.ipc.Handle) u32 {
|
||||
/// The harness-run child of the signals test: echoes requests, logs the two
|
||||
/// signals it handles. Terminate makes run() return, and returning from main is
|
||||
/// the clean exit the parent reads as ExitReason.exited.
|
||||
fn echo(message: []const u8, reply: []u8, sender: u32) usize {
|
||||
fn echo(message: []const u8, reply: []u8, sender: u32, capability: ?runtime.ipc.Handle) usize {
|
||||
_ = sender;
|
||||
_ = capability;
|
||||
const n = @min(message.len, reply.len);
|
||||
@memcpy(reply[0..n], message[0..n]);
|
||||
return n;
|
||||
|
||||
@@ -87,11 +87,12 @@ fn releaseClientHandles(client: u32) void {
|
||||
released += 1;
|
||||
}
|
||||
}
|
||||
if (released != 0) writeLine("vfs: released {d} handle(s) for dead client {d}\n", .{ released, client });
|
||||
if (released != 0) writeLine("/system/services/vfs: released {d} handle(s) for dead client {d}\n", .{ released, client });
|
||||
}
|
||||
|
||||
/// Handle one request from `sender`; write the reply into `out`, return its length.
|
||||
fn handle(message: []const u8, out: []u8, sender: u32) usize {
|
||||
fn handle(message: []const u8, out: []u8, sender: u32, capability: ?runtime.ipc.Handle) usize {
|
||||
_ = capability;
|
||||
if (message.len < protocol.request_size) return fail(out);
|
||||
const request = std.mem.bytesToValue(protocol.Request, message[0..protocol.request_size]);
|
||||
const payload = message[protocol.request_size..];
|
||||
@@ -143,9 +144,9 @@ fn handle(message: []const u8, out: []u8, sender: u32) usize {
|
||||
/// to release them (docs/process-lifecycle.md).
|
||||
fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||
if (!runtime.process.subscribeExits(endpoint)) {
|
||||
_ = runtime.system.write("vfs: exit subscription failed\n");
|
||||
_ = runtime.system.write("/system/services/vfs: exit subscription failed\n");
|
||||
}
|
||||
_ = runtime.system.write("vfs: ready\n");
|
||||
_ = runtime.system.write("/system/services/vfs: ready\n");
|
||||
return true;
|
||||
}
|
||||
|
||||
|
||||
+179
-45
@@ -18,9 +18,11 @@ Usage:
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import shutil
|
||||
import socket
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
@@ -55,20 +57,16 @@ ARCHES = {
|
||||
"/opt/homebrew/share/qemu/edk2-i386-vars.fd", # macOS Homebrew (Apple Silicon)
|
||||
"/usr/local/share/qemu/edk2-i386-vars.fd", # macOS Homebrew (Intel)
|
||||
],
|
||||
# zig-out is a FHS-shaped image and the boot volume; the harness copies the
|
||||
# boot-critical files from their FHS paths into a fresh ESP with the same
|
||||
# layout. (dest in ESP, source path under zig-out) — identical here.
|
||||
"efi_app": ("EFI/BOOT/BOOTX64.efi", "EFI/BOOT/BOOTX64.efi"),
|
||||
"kernel": ("system/kernel", "system/kernel"),
|
||||
# The init user program and the initial-ramdisk (VFS server + drivers).
|
||||
"extra": [("system/services/init", "system/services/init"),
|
||||
("boot/initial-ramdisk.img", "boot/initial-ramdisk.img")],
|
||||
# zig-out is itself the FHS-shaped boot volume (docs/efi.md): the build
|
||||
# installs BOOTX64.efi, the kernel, init, and the initial-ramdisk at their
|
||||
# boot paths. The harness presents zig-out to the guest directly — exactly
|
||||
# as `zig build run-x86-64` does — so there is no separate ESP to assemble.
|
||||
# Built as a function so we can splice in per-run paths.
|
||||
"qemu_args": lambda a, esp, vars_fd, serial: [
|
||||
"qemu_args": lambda a, boot_volume, vars_fd, serial: [
|
||||
"-machine", "q35", "-m", "128M",
|
||||
"-drive", f"if=pflash,format=raw,readonly=on,file={a['ovmf_code']}",
|
||||
"-drive", f"if=pflash,format=raw,file={vars_fd}",
|
||||
"-drive", f"format=raw,file=fat:rw:{esp}",
|
||||
"-drive", f"format=raw,file=fat:rw:{boot_volume}",
|
||||
"-net", "none",
|
||||
"-vga", "none", "-device", "VGA,edid=on,xres=1280,yres=720",
|
||||
"-display", "none",
|
||||
@@ -84,7 +82,10 @@ ARCHES = {
|
||||
# `expect`: a regex that must appear in serial output => pass.
|
||||
# `fail`: optional regex whose appearance => immediate fail.
|
||||
CASES = [
|
||||
# smoke also proves the QMP channel: the harmless query must be delivered
|
||||
# (handshake + command) before the case may pass — see run_case.
|
||||
{"name": "smoke",
|
||||
"qmp_after": {"delay": 2, "command": "query-status"},
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
{"name": "discovery",
|
||||
@@ -168,7 +169,7 @@ CASES = [
|
||||
# Stress the big kernel lock across cores; heavier, so a longer timeout.
|
||||
{"name": "smp-stress",
|
||||
"smp": 4,
|
||||
"timeout": 90,
|
||||
"timeout": 150,
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# Retry: a forced first-wake failure must still bring every core online.
|
||||
@@ -176,6 +177,15 @@ CASES = [
|
||||
"smp": 4,
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# TSC clocksource + cross-core warp check (real Intel/AMD / KVM path). TCG won't
|
||||
# advertise an invariant TSC, so the kernel forces the TSC clocksource on for this
|
||||
# case (gated in kernel.zig) and runs the per-AP warp check across the 4 cores;
|
||||
# their TSCs are synchronized, so it stays on the TSC (no HPET fallback). The rest
|
||||
# of the suite exercises the HPET fallback instead. See docs/timers.md.
|
||||
{"name": "tsc-sync",
|
||||
"smp": 4,
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
{"name": "fault-ud", "expect": r"invalid opcode \(vector 6\)"},
|
||||
{"name": "fault-pf", "expect": r"page fault \(vector 14\)"},
|
||||
{"name": "fault-df", "expect": r"double fault \(vector 8\)"},
|
||||
@@ -258,9 +268,112 @@ CASES = [
|
||||
"smp": 4,
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# M18.2: bus tree reports — the xHCI driver scans its root-hub ports and
|
||||
# reports both QEMU devices; the manager mirrors, prunes on the reporter's
|
||||
# death, and the respawned driver re-reports (docs/device-manager.md).
|
||||
{"name": "usb-report",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"qemu_extra": ["-device", "qemu-xhci,id=xhci",
|
||||
"-device", "usb-kbd,bus=xhci.0",
|
||||
"-device", "usb-mouse,bus=xhci.0"],
|
||||
"expect": r"device-manager: child added[\s\S]*"
|
||||
r"device-manager: child added[\s\S]*"
|
||||
r"device-manager: test mode: killing the reporter[\s\S]*"
|
||||
r"device-manager: child removed[\s\S]*"
|
||||
r"device-manager: restarting usb-xhci-bus[\s\S]*"
|
||||
r"device-manager: child added",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# M20.1: the ring-3 AML parse (the acpi service maps the blobs and parses
|
||||
# them) finds exactly the Device count the kernel's own parse produced.
|
||||
{"name": "acpi-parse",
|
||||
"smp": 4,
|
||||
"timeout": 60,
|
||||
"expect": r"acpi-parse: ok",
|
||||
"fail": r"acpi-parse: mismatch|DANOS-TEST-RESULT: FAIL"},
|
||||
# M20.3: the flip — ps2-bus now comes up from the acpi service's report, not
|
||||
# a kernel-built node. Ordered: report -> spawn -> the driver attaches its
|
||||
# keyboard, proving discovery runs entirely in ring 3 (docs/discovery.md).
|
||||
{"name": "acpi-ps2",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"expect": r"acpi: reported PNP0303[\s\S]*"
|
||||
r"device-manager: spawned ps2-bus[\s\S]*"
|
||||
r"ps2-bus: keyboard driver attached",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# M21.1: the SCI + power button. Boot the manager (which spawns the acpi
|
||||
# service); ~4s in, QMP system_powerdown raises the ACPI power-button fixed
|
||||
# event; the service's SCI handler must log the press (docs/acpi.md).
|
||||
{"name": "power-button",
|
||||
"smp": 4,
|
||||
"timeout": 60,
|
||||
"qmp_after": {"delay": 4, "command": "system_powerdown"},
|
||||
"expect": r"power: button pressed",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# M21.3 capstone: orderly shutdown. Boot init (the full tree comes up);
|
||||
# ~5s in, QMP system_powerdown raises the power button; the acpi service
|
||||
# publishes it, init stops its children then requests S5, and QEMU exits.
|
||||
# The ordered regex proves button -> shutting-down -> entering-S5; the case
|
||||
# passes on QEMU's self-exit through S5 (docs/power.md).
|
||||
{"name": "orderly-shutdown",
|
||||
"smp": 4,
|
||||
"timeout": 90,
|
||||
"qmp_after": {"delay": 5, "command": "system_powerdown"},
|
||||
"expect": r"power: button pressed[\s\S]*"
|
||||
r"init: shutting down[\s\S]*"
|
||||
r"power: entering S5",
|
||||
"fail": r"power: S5 write did not take|DANOS-TEST-RESULT: FAIL"},
|
||||
# M20.2: the acpi service evaluates _CRS/_STA in ring 3 and registers +
|
||||
# reports its _HID devices — the two PS/2 nodes must appear with resources
|
||||
# (keyboard: io 0x60/0x64 + IRQ = 3; mouse: IRQ = 1) (docs/discovery.md).
|
||||
{"name": "acpi-report",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"qemu_extra": ["-device", "qemu-xhci,id=xhci",
|
||||
"-device", "usb-kbd,bus=xhci.0",
|
||||
"-device", "usb-mouse,bus=xhci.0"],
|
||||
"expect": r"acpi: reported PNP0303 \(device \d+, 3 resources\)[\s\S]*"
|
||||
r"acpi: reported PNP0F13 \(device \d+, 1 resources\)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# M19.1: the ring-3 PCI scan (pci-bus walks the ECAM through its mmio_map
|
||||
# grant) finds exactly the functions the kernel's own walk recorded.
|
||||
{"name": "pci-scan",
|
||||
"smp": 4,
|
||||
"timeout": 60,
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# M18.3: the application surface — device-list enumerates the tree over IPC,
|
||||
# subscribes (endpoint as capability), and observes the removed/added events
|
||||
# the reporter's test-kill produces (docs/device-manager.md).
|
||||
{"name": "device-list",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"qemu_extra": ["-device", "qemu-xhci,id=xhci",
|
||||
"-device", "usb-kbd,bus=xhci.0",
|
||||
"-device", "usb-mouse,bus=xhci.0"],
|
||||
"expect": r"device-list: \d+ devices[\s\S]*"
|
||||
r"device-list: subscribed[\s\S]*"
|
||||
r"device-manager: test mode: killing the reporter[\s\S]*"
|
||||
r"device-list: removed \(device[\s\S]*"
|
||||
r"device-list: added \(device",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# M18.1: the device manager's hello + restart policy — xHCI hellos clean and
|
||||
# stays; crash-test faults, is restarted with backoff (re-claiming its device
|
||||
# each time), and hits the crash-loop cap (docs/device-manager.md).
|
||||
{"name": "driver-restart",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"qemu_extra": ["-device", "qemu-xhci,id=xhci",
|
||||
"-device", "usb-kbd,bus=xhci.0",
|
||||
"-device", "usb-mouse,bus=xhci.0"],
|
||||
"expect": r"usb-xhci-bus: hello acknowledged[\s\S]*"
|
||||
r"device-manager: restarting crash-test[\s\S]*"
|
||||
r"device-manager: crash-test is failing repeatedly",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# The initial_ramdisk: the loader ferries a bundle of user binaries; the kernel parses
|
||||
# it and spawns each as a ring-3 process (here the VFS-server stub heartbeats).
|
||||
{"name": "initial-ramdisk",
|
||||
"timeout": 60, # the acpi service's boot-time SCI setup can push the marker past 30s under load
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# The user-space VFS: a client opens/writes/reads a file through the rt file
|
||||
@@ -274,27 +387,21 @@ CASES = [
|
||||
{"name": "input",
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# IO passthrough + IRQ-as-IPC: a user-space HPET driver maps device MMIO into
|
||||
# its own address space, binds the device's interrupt to an IPC endpoint, and
|
||||
# is woken by the hardware five times while blocked (never polling).
|
||||
{"name": "hpet",
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# Device manager: a ring-3 service enumerates /system/devices and matches each
|
||||
# device to a driver (discovery + policy in user space). This increment logs the
|
||||
# decision; spawning follows.
|
||||
# Device manager: a ring-3 service enumerates /system/devices, matches the PCI host
|
||||
# bridge to pci-bus, and spawns it — end-to-end proof of discover -> match -> spawn
|
||||
# -> driver-up (the spawned pci-bus logs "<N> functions found").
|
||||
{"name": "device-manager",
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# Bus driver: a user process claims a device, enumerates its children from the
|
||||
# hardware, and publishes each with dev_register — and the kernel refuses a child
|
||||
# whose window escapes the parent's (else dev_register maps arbitrary memory).
|
||||
{"name": "bus",
|
||||
# device_register containment (in-kernel): registering a child whose MMIO window
|
||||
# escapes its parent's grant is refused (NotContained) — else dev_register would map
|
||||
# arbitrary physical memory — while an identical re-register stays idempotent.
|
||||
{"name": "containment",
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# IRQ teardown: an exiting driver's line is masked and its slot cleared (so no
|
||||
# ISR notifies a freed endpoint), and a sibling owner sharing that endpoint
|
||||
# keeps its own binding. The path hpet never takes, since it runs forever.
|
||||
# keeps its own binding. A long-running driver never reaches this teardown path.
|
||||
{"name": "irqfree",
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
@@ -324,24 +431,6 @@ def build(arch, case):
|
||||
return None
|
||||
|
||||
|
||||
def make_esp(arch):
|
||||
"""Assemble a fresh EFI System Partition from the freshly built binaries."""
|
||||
esp = os.path.join(WORK, "esp")
|
||||
if os.path.exists(esp):
|
||||
shutil.rmtree(esp)
|
||||
efi_dest, efi_src = arch["efi_app"]
|
||||
kern_dest, kern_src = arch["kernel"]
|
||||
fhs = os.path.join(REPO, "zig-out") # zig-out is the FHS image
|
||||
os.makedirs(os.path.join(esp, os.path.dirname(efi_dest)), exist_ok=True)
|
||||
os.makedirs(os.path.join(esp, os.path.dirname(kern_dest)), exist_ok=True)
|
||||
shutil.copy(os.path.join(fhs, efi_src), os.path.join(esp, efi_dest))
|
||||
shutil.copy(os.path.join(fhs, kern_src), os.path.join(esp, kern_dest))
|
||||
for dest, src in arch.get("extra", []):
|
||||
os.makedirs(os.path.join(esp, os.path.dirname(dest)), exist_ok=True)
|
||||
shutil.copy(os.path.join(fhs, src), os.path.join(esp, dest))
|
||||
return esp
|
||||
|
||||
|
||||
def resolve_firmware(arch):
|
||||
"""Collapse the ovmf_code/ovmf_vars candidate lists to the first path that
|
||||
exists on this machine. Mutates `arch` in place; idempotent (a resolved
|
||||
@@ -360,12 +449,34 @@ def resolve_firmware(arch):
|
||||
+ "\nInstall OVMF (edk2-ovmf / ovmf) or add its path above.")
|
||||
|
||||
|
||||
def qmp_send(path, command):
|
||||
"""One QMP command: connect, capabilities handshake, execute. Raises on any
|
||||
failure — the caller retries until the guest's socket is ready. This is how
|
||||
a case injects a host-side event (system_powerdown = the ACPI power button)
|
||||
into the running guest (docs/power.md)."""
|
||||
sock = socket.socket(socket.AF_UNIX, socket.SOCK_STREAM)
|
||||
sock.settimeout(5)
|
||||
try:
|
||||
sock.connect(path)
|
||||
stream = sock.makefile("rw")
|
||||
stream.readline() # the QMP greeting
|
||||
stream.write(json.dumps({"execute": "qmp_capabilities"}) + "\n")
|
||||
stream.flush()
|
||||
stream.readline() # {"return": {}}
|
||||
stream.write(json.dumps({"execute": command}) + "\n")
|
||||
stream.flush()
|
||||
stream.readline()
|
||||
finally:
|
||||
sock.close()
|
||||
|
||||
|
||||
def run_case(arch, case):
|
||||
err = build(arch, case["name"])
|
||||
if err:
|
||||
return False, "build failed:\n" + err
|
||||
|
||||
esp = make_esp(arch)
|
||||
# zig-out is the FHS boot volume; hand it to the guest as-is (see qemu_args).
|
||||
boot_volume = os.path.join(REPO, "zig-out")
|
||||
vars_fd = os.path.join(WORK, "vars.fd")
|
||||
shutil.copy(arch["ovmf_vars"], vars_fd)
|
||||
serial = os.path.join(WORK, "serial.log")
|
||||
@@ -375,17 +486,32 @@ def run_case(arch, case):
|
||||
expect = re.compile(case["expect"])
|
||||
fail = re.compile(case["fail"]) if case.get("fail") else None
|
||||
|
||||
cmd = [arch["qemu"]] + arch["qemu_args"](arch, esp, vars_fd, serial)
|
||||
cmd = [arch["qemu"]] + arch["qemu_args"](arch, boot_volume, vars_fd, serial)
|
||||
if case.get("smp"): # some cases need more than one core (e.g. parallelism)
|
||||
cmd += ["-smp", str(case["smp"])]
|
||||
if case.get("qemu_extra"): # extra qemu args, e.g. -device intel-iommu for the IOMMU case
|
||||
cmd += case["qemu_extra"]
|
||||
# A QMP control socket, always present (additive): how a case's `qmp_after`
|
||||
# hook injects host-side events into the guest mid-run.
|
||||
qmp_path = os.path.join(WORK, "qmp.sock")
|
||||
if os.path.exists(qmp_path):
|
||||
os.remove(qmp_path)
|
||||
cmd += ["-qmp", f"unix:{qmp_path},server,nowait"]
|
||||
qmp_after = case.get("qmp_after") # {"delay": seconds, "command": "..."}
|
||||
qmp_sent = False
|
||||
started = time.monotonic()
|
||||
qemu = subprocess.Popen(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)
|
||||
try:
|
||||
timeout = case.get("timeout", TIMEOUT)
|
||||
deadline = time.monotonic() + timeout
|
||||
while time.monotonic() < deadline:
|
||||
time.sleep(0.2)
|
||||
if qmp_after and not qmp_sent and time.monotonic() - started >= qmp_after["delay"]:
|
||||
try:
|
||||
qmp_send(qmp_path, qmp_after["command"])
|
||||
qmp_sent = True
|
||||
except OSError:
|
||||
pass # socket not up yet; retry next tick
|
||||
text = ""
|
||||
if os.path.exists(serial):
|
||||
with open(serial, "r", errors="replace") as f:
|
||||
@@ -393,6 +519,8 @@ def run_case(arch, case):
|
||||
if fail and fail.search(text):
|
||||
return False, "hit failure marker"
|
||||
if expect.search(text):
|
||||
if qmp_after and not qmp_sent:
|
||||
continue # the hook must deliver before the case may pass
|
||||
return True, "matched " + repr(case["expect"])
|
||||
if qemu.poll() is not None: # QEMU exited on its own
|
||||
if expect.search(text):
|
||||
@@ -428,6 +556,12 @@ def main():
|
||||
for case in selected:
|
||||
print(f" {case['name']:<12} ... ", end="", flush=True)
|
||||
ok, detail = run_case(arch, case)
|
||||
if not ok:
|
||||
# Keep the evidence: serial.log is otherwise overwritten by the
|
||||
# next case, and an intermittent failure's log is unrecoverable.
|
||||
source = os.path.join(WORK, "serial.log")
|
||||
if os.path.exists(source):
|
||||
shutil.copy(source, os.path.join(WORK, f"{case['name']}-failed-serial.log"))
|
||||
print(("PASS" if ok else "FAIL") + f" ({detail})")
|
||||
if not ok:
|
||||
failures += 1
|
||||
|
||||
Executable
+22
@@ -0,0 +1,22 @@
|
||||
#!/usr/bin/env bash
|
||||
#
|
||||
# sort-lines-group-by-start — cluster lines that share their first
|
||||
# whitespace-separated field ($1). Keys appear in first-seen order, and lines
|
||||
# within a key keep their original order. It groups; it does NOT sort.
|
||||
#
|
||||
# Pass the log file as an argument; result is written to stdout:
|
||||
#
|
||||
# tools/sort-lines-group-by-start.sh filename.log
|
||||
#
|
||||
# Useful for a serial/boot log where several sources interleave and each line is
|
||||
# prefixed with its source (the first field): this pulls every source's lines
|
||||
# back together, in the order the sources first appeared, without reordering
|
||||
# within a source.
|
||||
#
|
||||
# input output
|
||||
# pci-bus: scan start pci-bus: scan start
|
||||
# acpi: reported PNP0303 pci-bus: 5 functions
|
||||
# pci-bus: 5 functions acpi: reported PNP0303
|
||||
# acpi: reported PNP0501 acpi: reported PNP0501
|
||||
|
||||
awk '{lines[$1] = lines[$1] ? lines[$1] ORS $0 : $0; if (!seen[$1]++) order[++count] = $1} END {for (i=1; i<=count; i++) print lines[order[i]]}' "$@"
|
||||
Reference in New Issue
Block a user