Compare commits
34
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
0628944b15 | ||
|
|
e6d0bb7ef0 | ||
|
|
5ca804d827 | ||
|
|
a299363b59 | ||
|
|
d8dd62c639 | ||
|
|
d106b6e8dc | ||
|
|
af2c766f42 | ||
|
|
d26262bf56 | ||
|
|
10b89c06ff | ||
|
|
a2a05d0b3d | ||
|
|
75d62660b0 | ||
|
|
a53c2b0193 | ||
|
|
bf481c080c | ||
|
|
3a78dcab3f | ||
|
|
470f93a83d | ||
|
|
7798706b41 | ||
|
|
ad40de03c2 | ||
|
|
d8778b4b70 | ||
|
|
79d859a111 | ||
|
|
37fb09f75e | ||
|
|
34ebeb968d | ||
|
|
3cc1d38dd0 | ||
|
|
36e804b848 | ||
|
|
be83a42d42 | ||
|
|
650a1b1595 | ||
|
|
d8c55c6f2f | ||
|
|
2ebfb0c3b0 | ||
|
|
888eaa74e1 | ||
|
|
ed76cbbc79 | ||
|
|
140229b88d | ||
|
|
cb2379fd06 | ||
|
|
1665b239b0 | ||
|
|
70ed0337f8 | ||
|
|
116b8f6c41 |
@@ -2,12 +2,31 @@
|
|||||||
Codename: Shodan
|
Codename: Shodan
|
||||||
Version: 1
|
Version: 1
|
||||||
|
|
||||||
A small operating system, written from scratch in Zig — a bootloader (`boot/`)
|
A small resilient operating system, written from scratch in Zig.
|
||||||
and a microkernel (`system/kernel/`), sharing a neutral handoff contract (`system/boot-handoff.zig`).
|
|
||||||
It boots x86-64 via UEFI, and so far has a framebuffer console, a physical frame
|
## Zen of DanOS:
|
||||||
allocator, its own paging with W^X permissions, interrupt/exception handling, a
|
|
||||||
LAPIC timer, a kernel heap, a fixed-priority preemptive scheduler, and in-kernel IPC
|
- Resilient Micro-Kernel Architecture.
|
||||||
channels. See [`docs/`](docs/README.md) for how each piece works.
|
- Every process run in an isolated user space not kernel space.
|
||||||
|
- Processes cannot take down the entire OS with it when they die or is killed
|
||||||
|
- Stable public runtime library, private OS ABI.
|
||||||
|
- Keeps a stable runtime for user space processes between OS versions (great for backwards compatibility)
|
||||||
|
- Allows the underlying OS to be changed without effecting applications
|
||||||
|
- Provides a boundary to enable compatibility between OS's e.g. POSIX, MUSL etc
|
||||||
|
- Drivers are just isolated processes in user space.
|
||||||
|
- Thin binaries that can be restarted like applications.
|
||||||
|
- Useful during driver development.
|
||||||
|
- Drivers can claim MMIO / ports
|
||||||
|
- Driver resources (e.g. IRQ/Port/MMIO) claims are automatically cleaned up if the driver dies or is killed
|
||||||
|
- Drivers can also hook into the process lifecyle to clean up or reset hardware
|
||||||
|
- No legacy to deal with
|
||||||
|
- Zig code uses a clean coding style (Zen of Zig)
|
||||||
|
- Favor reading code over writing code.
|
||||||
|
- No magic numbers.
|
||||||
|
- No shortend names unless its for ABI compatibility or acronyms
|
||||||
|
- Inter-Process Communication (IPC)
|
||||||
|
- Publish and subscribe to Asynchronous Messages
|
||||||
|
- Talk to services and processes synchronously
|
||||||
|
|
||||||
## Prerequisites
|
## Prerequisites
|
||||||
|
|
||||||
@@ -60,7 +79,7 @@ straight into CI.
|
|||||||
|
|
||||||
## Documentation
|
## Documentation
|
||||||
|
|
||||||
Design notes explaining the *why* behind the code live in
|
Design notes explaining *why* behind the code live in
|
||||||
[`docs/`](docs/README.md) — start with [`docs/README.md`](docs/README.md).
|
[`docs/`](docs/README.md) — start with [`docs/README.md`](docs/README.md).
|
||||||
|
|
||||||
## Logo
|
## Logo
|
||||||
|
|||||||
@@ -131,6 +131,13 @@ pub fn build(b: *std.Build) void {
|
|||||||
});
|
});
|
||||||
// ACPI/PnP hardware-ID (_HID) names — the flat analog of pci-class for acpi_device
|
// ACPI/PnP hardware-ID (_HID) names — the flat analog of pci-class for acpi_device
|
||||||
// nodes. Also shared reference data.
|
// nodes. Also shared reference data.
|
||||||
|
// The AML interpreter, a build module so the ring-3 acpi service can run the
|
||||||
|
// same parser the kernel does (docs/m19-m20-plan.md decision 1). Pure Zig,
|
||||||
|
// no kernel imports — one source, two builds.
|
||||||
|
const aml_module = b.addModule("aml", .{
|
||||||
|
.root_source_file = b.path("system/devices/aml/aml.zig"),
|
||||||
|
});
|
||||||
|
|
||||||
const acpi_ids_module = b.addModule("acpi-ids", .{
|
const acpi_ids_module = b.addModule("acpi-ids", .{
|
||||||
.root_source_file = b.path("system/devices/acpi-ids.zig"),
|
.root_source_file = b.path("system/devices/acpi-ids.zig"),
|
||||||
});
|
});
|
||||||
@@ -212,6 +219,13 @@ pub fn build(b: *std.Build) void {
|
|||||||
},
|
},
|
||||||
});
|
});
|
||||||
|
|
||||||
|
// The device-manager protocol: hello + (M18.2) tree reports, exposed as its
|
||||||
|
// own module like the other protocol modules. Imported through the runtime.
|
||||||
|
const device_manager_protocol_module = b.addModule("device-manager-protocol", .{
|
||||||
|
.root_source_file = b.path("system/services/device-manager/device-manager-protocol.zig"),
|
||||||
|
});
|
||||||
|
runtime_module.addImport("device-manager-protocol", device_manager_protocol_module);
|
||||||
|
|
||||||
// Typed volatile MMIO register access + memory-ordering barriers, for drivers on
|
// Typed volatile MMIO register access + memory-ordering barriers, for drivers on
|
||||||
// top of an mmio_map grant. Depends only on `builtin` (arch-conditional barriers);
|
// top of an mmio_map grant. Depends only on `builtin` (arch-conditional barriers);
|
||||||
// no target set, so it inherits each driver's. See library/mmio/mmio.zig.
|
// no target set, so it inherits each driver's. See library/mmio/mmio.zig.
|
||||||
@@ -332,6 +346,26 @@ pub fn build(b: *std.Build) void {
|
|||||||
const ps2_keyboard_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-keyboard", "system/drivers/ps2-bus/keyboard.zig");
|
const ps2_keyboard_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-keyboard", "system/drivers/ps2-bus/keyboard.zig");
|
||||||
const ps2_mouse_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-mouse", "system/drivers/ps2-bus/mouse.zig");
|
const ps2_mouse_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-mouse", "system/drivers/ps2-bus/mouse.zig");
|
||||||
const usb_xhci_bus_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "usb-xhci-bus", "system/drivers/usb-xhci-bus/usb-xhci-bus.zig");
|
const usb_xhci_bus_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "usb-xhci-bus", "system/drivers/usb-xhci-bus/usb-xhci-bus.zig");
|
||||||
|
const pci_bus_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "pci-bus", "system/drivers/pci-bus/pci-bus.zig");
|
||||||
|
// A test fixture, not a real driver: hellos to the device manager, then faults —
|
||||||
|
// what the driver-restart scenario drives the crash-loop cap with.
|
||||||
|
const crash_test_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "crash-test", "system/services/crash-test/crash-test.zig");
|
||||||
|
const device_list_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "device-list", "system/services/device-list/device-list.zig");
|
||||||
|
// The discovery service: one swappable process per firmware
|
||||||
|
// (docs/m19-m20-plan.md decision 7), bundled under the neutral ramdisk name
|
||||||
|
// "discovery" so the device manager never learns which firmware it is on.
|
||||||
|
// x86 boots describe hardware with ACPI; the Raspberry Pis hand over a
|
||||||
|
// flattened device tree — the aarch64 target flips the default when it
|
||||||
|
// lands (docs/arm.md). Both are placeholders until M20.1 (acpi) and the
|
||||||
|
// ARM bring-up (fdt).
|
||||||
|
const Discovery = enum { acpi, fdt };
|
||||||
|
const discovery = b.option(Discovery, "discovery", "Which discovery service fills the ramdisk's 'discovery' slot (default: acpi)") orelse Discovery.acpi;
|
||||||
|
const discovery_source: []const u8 = switch (discovery) {
|
||||||
|
.acpi => "system/services/acpi/acpi.zig",
|
||||||
|
.fdt => "system/services/fdt/fdt.zig",
|
||||||
|
};
|
||||||
|
const discovery_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "discovery", discovery_source);
|
||||||
|
if (discovery == .acpi) discovery_exe.root_module.addImport("aml", aml_module);
|
||||||
const device_manager_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "device-manager", "system/services/device-manager/device-manager.zig");
|
const device_manager_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "device-manager", "system/services/device-manager/device-manager.zig");
|
||||||
// The input service and its exercisers: the fan-out server, a hardware-free synthetic
|
// The input service and its exercisers: the fan-out server, a hardware-free synthetic
|
||||||
// source, and a subscriber that doubles as the `input` test's oracle. See docs/input.md.
|
// source, and a subscriber that doubles as the `input` test's oracle. See docs/input.md.
|
||||||
@@ -363,6 +397,14 @@ pub fn build(b: *std.Build) void {
|
|||||||
mk_run.addFileArg(ps2_mouse_exe.getEmittedBin());
|
mk_run.addFileArg(ps2_mouse_exe.getEmittedBin());
|
||||||
mk_run.addArg("usb-xhci-bus");
|
mk_run.addArg("usb-xhci-bus");
|
||||||
mk_run.addFileArg(usb_xhci_bus_exe.getEmittedBin());
|
mk_run.addFileArg(usb_xhci_bus_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("pci-bus");
|
||||||
|
mk_run.addFileArg(pci_bus_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("crash-test");
|
||||||
|
mk_run.addFileArg(crash_test_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("device-list");
|
||||||
|
mk_run.addFileArg(device_list_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("discovery");
|
||||||
|
mk_run.addFileArg(discovery_exe.getEmittedBin());
|
||||||
mk_run.addArg("device-manager");
|
mk_run.addArg("device-manager");
|
||||||
mk_run.addFileArg(device_manager_exe.getEmittedBin());
|
mk_run.addFileArg(device_manager_exe.getEmittedBin());
|
||||||
mk_run.addArg("input");
|
mk_run.addArg("input");
|
||||||
|
|||||||
+8
-6
@@ -45,20 +45,22 @@ rather than restate it. Roughly in the order things happen at runtime:
|
|||||||
until its hardware interrupts it**. The claim is the capability; `irq_ack` is the
|
until its hardware interrupts it**. The claim is the capability; `irq_ack` is the
|
||||||
unmask.
|
unmask.
|
||||||
14. **[driver-model.md](driver-model.md) — buses, classes and host controllers.** How
|
14. **[driver-model.md](driver-model.md) — buses, classes and host controllers.** How
|
||||||
real driver stacks factor into three shapes, how families share code, and the
|
real driver stacks factor into three shapes and how families share code. The
|
||||||
proposed ABI for the three primitives still missing (capability passing, DMA +
|
three primitives it proposed are long since built (M13 capability passing,
|
||||||
memory barriers, MSI).
|
M14 DMA + barriers, M15 MSI), and the driver *contract* on top of them —
|
||||||
|
hello, supervision, restart — is built too (device-manager.md, M18).
|
||||||
15. **[process-management.md](process-management.md) — process management.** The
|
15. **[process-management.md](process-management.md) — process management.** The
|
||||||
microkernel's `ps`/`kill`/SIGCHLD: enumerate as a table snapshot, the
|
microkernel's `ps`/`kill`/SIGCHLD: enumerate as a table snapshot, the
|
||||||
supervision link as the kill authority, and child-exit notifications over the
|
supervision link as the kill authority, and child-exit notifications over the
|
||||||
same endpoints IRQs arrive on.
|
same endpoints IRQs arrive on.
|
||||||
16. **[process-lifecycle.md](process-lifecycle.md) — the process lifecycle.** Design:
|
16. **[process-lifecycle.md](process-lifecycle.md) — the process lifecycle.** Built
|
||||||
signals over IPC as the one lifecycle vocabulary every process speaks — the
|
(M17): signals over IPC as the one lifecycle vocabulary every process speaks — the
|
||||||
POSIX.1-1990 words with message delivery instead of stack hijack, the stable
|
POSIX.1-1990 words with message delivery instead of stack hijack, the stable
|
||||||
`runtime.process` interface, exit reasons, published exit events any stateful
|
`runtime.process` interface, exit reasons, published exit events any stateful
|
||||||
service can subscribe to (the VFS releasing dead clients' handles), and the two
|
service can subscribe to (the VFS releasing dead clients' handles), and the two
|
||||||
iron rules (cleanup is the kernel's job; kill is not a signal).
|
iron rules (cleanup is the kernel's job; kill is not a signal).
|
||||||
17. **[device-manager.md](device-manager.md) — the device manager.** Design: the
|
17. **[device-manager.md](device-manager.md) — the device manager.** Built (M18,
|
||||||
|
through the app surface): the
|
||||||
tree, the matcher, and the supervisor. Tree structure lives in the manager,
|
tree, the matcher, and the supervisor. Tree structure lives in the manager,
|
||||||
authority stays in the kernel; bus drivers report what they see; drivers are
|
authority stays in the kernel; bus drivers report what they see; drivers are
|
||||||
restarted through the lifecycle vocabulary — the plan that turns
|
restarted through the lifecycle vocabulary — the plan that turns
|
||||||
|
|||||||
@@ -0,0 +1,156 @@
|
|||||||
|
# The device manager
|
||||||
|
|
||||||
|
**Status: the protocol and supervision are built** (M18.1, 2026-07-13): `hello`
|
||||||
|
with its deadline, supervised spawn, restart with backoff, and the crash-loop
|
||||||
|
cap are in — usb-xhci-bus is the first conforming driver, and the
|
||||||
|
`driver-restart` scenario proves fault → backoff → re-claim → cap end to end.
|
||||||
|
Tree reports are built too (M18.2, 2026-07-13): the xHCI driver scans its
|
||||||
|
root-hub ports and reports each connected device (`child_added`); the manager
|
||||||
|
mirrors them and prunes a dead reporter's children, and the `usb-report`
|
||||||
|
scenario proves report → prune → respawn → re-report. The application surface is built (M18.3, 2026-07-13):
|
||||||
|
`enumerate` and `subscribe` over IPC, with `device-list` as the first client —
|
||||||
|
the manager is now the one answer to "what devices exist" for applications.
|
||||||
|
The primitives underneath are real ([process-management.md](process-management.md):
|
||||||
|
spawn/supervise/kill/exit-notification; [driver-model.md](driver-model.md): the device
|
||||||
|
table as a capability system; [drivers.md](drivers.md): claim/map/IRQ), and the first
|
||||||
|
per-device driver spawn works (the device manager matches the xHCI controller by PCI
|
||||||
|
class and spawns `usb-xhci-bus` with the device id as argv[1]). This document designs
|
||||||
|
the rest: the device manager as **the tree, the matcher, and the supervisor** — the
|
||||||
|
policy process that turns [resilience.md](resilience.md)'s restart goal into practice
|
||||||
|
for drivers.
|
||||||
|
|
||||||
|
How processes stop, reload, and report their deaths is deliberately **not** in this
|
||||||
|
document: that is the universal lifecycle every danos process speaks —
|
||||||
|
[process-lifecycle.md](process-lifecycle.md), signals over IPC and the stable
|
||||||
|
`runtime.process` interface. The device manager is that design's first serious
|
||||||
|
customer, not its owner. Its own protocol contains nothing lifecycle-shaped; a
|
||||||
|
driver is stopped, health-checked, and buried exactly like any other process.
|
||||||
|
|
||||||
|
## The tree: structure in the manager, authority in the kernel
|
||||||
|
|
||||||
|
The device tree is two things fused: *information* (what exists, how it nests) and
|
||||||
|
*authority* (a descriptor is a licence to map physical memory). They separate:
|
||||||
|
|
||||||
|
- The **kernel keeps the capability system** — device, I/O-port, and interrupt
|
||||||
|
claims, resource containment on `device_register`, the
|
||||||
|
`mmio_map`/`irq_bind`/`msi_bind` gates — and **cleans all of it up when a process
|
||||||
|
dies** (settled; it is increment 1 of
|
||||||
|
[process-lifecycle.md](process-lifecycle.md)). The three invariants in
|
||||||
|
[driver-model.md](driver-model.md) stay exactly where they are. A device manager
|
||||||
|
that could mint MMIO mappings by its own say-so would be a second kernel, and a
|
||||||
|
buggy one would un-earn everything the microkernel bought.
|
||||||
|
- The **device manager owns the tree as data** — identity, topology, naming, driver
|
||||||
|
matching, hotplug events, and being the one process everything else asks about
|
||||||
|
devices. Firmware discovery seeds it (today via the kernel's snapshot); **bus
|
||||||
|
drivers grow it** by reporting what they see; applications query and watch it.
|
||||||
|
`device_enumerate` fades to a manager-internal (then deleted) seam.
|
||||||
|
|
||||||
|
Long-term, discovery itself leaves the kernel — but not *into* the manager. PCI
|
||||||
|
enumeration is a **pci-bus driver**: the manager spawns it against the host bridge
|
||||||
|
(already a device with the ECAM window as a resource), it scans, it reports functions
|
||||||
|
like any bus reports children. ACPI becomes an **acpi service** that interprets the
|
||||||
|
tables and reports the namespace. The manager only orchestrates and merges. Moving
|
||||||
|
AML interpretation out of ring 0 is its own project on its own track; nothing here
|
||||||
|
depends on when it lands.
|
||||||
|
|
||||||
|
## The protocol
|
||||||
|
|
||||||
|
A `device-manager-protocol` module (the vfs-protocol pattern): extern-struct
|
||||||
|
messages, a version in the handshake, reserved fields everywhere. The manager is a
|
||||||
|
well-known endpoint (`ipc.register(.device_manager)`); the badge tells it who is
|
||||||
|
talking; the same endpoint receives its children's exit notifications — one loop,
|
||||||
|
one world.
|
||||||
|
|
||||||
|
| Direction | Message | Purpose |
|
||||||
|
|---|---|---|
|
||||||
|
| driver → manager | `hello { version, role, device_id }` | confirms the argv assignment, starts the deadline clock |
|
||||||
|
| bus → manager | `child_added { parent, identity, resources }` | one node the bus discovered |
|
||||||
|
| bus → manager | `child_removed { id }` | unplug, or the bus lost it |
|
||||||
|
| app → manager | `enumerate` | snapshot of the tree (read-only) |
|
||||||
|
| app → manager | `subscribe` | receive published add/remove events |
|
||||||
|
|
||||||
|
`hello` is the one deadline the manager enforces itself: spawned and silent past the
|
||||||
|
deadline means wrong binary, wrong protocol version, or wedged before main — apply
|
||||||
|
the stop sequence and the restart policy. Everything else lifecycle-shaped
|
||||||
|
(terminate, the common `ping` liveness call, exit reasons) arrives through
|
||||||
|
[process-lifecycle.md](process-lifecycle.md)'s vocabulary, not this protocol.
|
||||||
|
|
||||||
|
Assignment stays argv (`usb-xhci-bus <device id>`) for now — simple, and it works.
|
||||||
|
The step after `hello` exists is delegation: the manager claims (or is granted) the
|
||||||
|
devices and passes the claim to the driver over IPC (the M13 capability-transfer
|
||||||
|
mechanism), replacing first-come-first-served `device_claim` with policy. Identity in
|
||||||
|
`child_added` is per-bus: PCI children carry the class triple (`pci_class`, as the
|
||||||
|
xHCI match already uses); USB children carry the (class, subclass, protocol) triple
|
||||||
|
from usb-ids.zig — each bus's native language, decoded by the shared ids modules.
|
||||||
|
|
||||||
|
## Supervision and restart
|
||||||
|
|
||||||
|
Every driver is spawned with the manager's exit endpoint (`spawnSupervised` — built).
|
||||||
|
On a death notification:
|
||||||
|
|
||||||
|
1. **Read the reason** ([process-lifecycle.md](process-lifecycle.md) increment 2).
|
||||||
|
Clean exit → it meant to; don't restart. Fault or missed `hello` deadline →
|
||||||
|
restart with **backoff**, and a crash-loop cap (three fast deaths → mark failed,
|
||||||
|
stop respawning, log loudly; a later `reload` to the manager can retry).
|
||||||
|
2. **Prune the subtree** the dead bus driver reported. Its children describe
|
||||||
|
protocol state (xHCI slot ids, transfer rings) that died with the process;
|
||||||
|
keeping the nodes would be keeping a lie. Watchers receive `child_removed` — the
|
||||||
|
input service losing, then regaining, a keyboard is the *honest* description of
|
||||||
|
what happened. The restarted instance rediscovers and re-reports.
|
||||||
|
3. **The claim is already free** because the kernel released it at death — the
|
||||||
|
restarted instance claims the same controller and comes up.
|
||||||
|
|
||||||
|
Who supervises the supervisor: **init** (PID 1), which already supervises the
|
||||||
|
services it starts. If the manager dies, drivers keep running (they hold their
|
||||||
|
claims; the kernel doesn't care who their supervisor was — though their exit
|
||||||
|
notifications now dangle harmlessly). The restarted manager re-learns the world:
|
||||||
|
kernel snapshot, then a re-`hello` round — drivers answer a broadcast or are stopped
|
||||||
|
and respawned. Full state handoff is deliberately not attempted.
|
||||||
|
|
||||||
|
## Thin drivers, class protocols
|
||||||
|
|
||||||
|
The [driver-model.md](driver-model.md) three-shape split, restated as processes:
|
||||||
|
|
||||||
|
- A **bus driver** (usb-xhci-bus) owns its controller — claim, MMIO, IRQ/MSI, DMA
|
||||||
|
rings — and offers a *transfer* protocol ("submit a control transfer to device N",
|
||||||
|
built from the usb-abi request constructors) plus tree reports to the manager.
|
||||||
|
- A **class driver** (usb-hid, usb-storage) owns nothing: it is matched to a reported
|
||||||
|
child by its identity triple, speaks the bus's transfer protocol downward and its
|
||||||
|
service's protocol upward — HID reports to the input service, blocks to the block
|
||||||
|
service. It works unchanged over any controller.
|
||||||
|
- **Services** (input, display, block) aggregate class drivers and face applications.
|
||||||
|
|
||||||
|
Each arrow is a protocol module. The manager routes none of the data plane — it
|
||||||
|
introduces the parties (matching), supervises them (lifecycle), and gets out of the
|
||||||
|
way.
|
||||||
|
|
||||||
|
## Increments
|
||||||
|
|
||||||
|
Increments 1–4 are the lifecycle prerequisites and live in
|
||||||
|
[process-lifecycle.md](process-lifecycle.md) (claim cleanup on death, exit reasons,
|
||||||
|
published exit events, signals + `runtime.process`). On top of those:
|
||||||
|
|
||||||
|
5. **device-manager-protocol**: `hello`, supervised spawn with restart policy;
|
||||||
|
usb-xhci-bus becomes the first conforming driver.
|
||||||
|
6. **Tree reports**: `child_added`/`child_removed`; the manager mirrors; xHCI reports
|
||||||
|
the mouse and keyboard QEMU already hangs off it.
|
||||||
|
7. **App surface**: `enumerate`/`subscribe` over IPC; `device_enumerate` retreats
|
||||||
|
to a manager-internal seam.
|
||||||
|
8. **Discovery migration** — DONE (M19–M20, 2026-07-13): pci-bus driver (M19)
|
||||||
|
then the acpi service (M20) moved enumeration to ring 3; the kernel seeds
|
||||||
|
only the host bridge and the acpi-tables node. See
|
||||||
|
[m19-m20-plan.md](m19-m20-plan.md).
|
||||||
|
|
||||||
|
## Settled questions (2026-07-12)
|
||||||
|
|
||||||
|
- **Stateful buses**: pruning the subtree on bus-driver death is right for USB. A
|
||||||
|
future storage bus with in-flight writes wants drain-before-terminate — which is
|
||||||
|
exactly the `deadline_ms` parameter `stop()` already has; a per-driver deadline
|
||||||
|
is one value in the manager's policy table when such a bus arrives. No design
|
||||||
|
change.
|
||||||
|
- **Manager death**: drivers survive the manager; the restarted manager re-learns
|
||||||
|
the world (above). Checkpointing driver state with the manager is deferred until
|
||||||
|
something demonstrates the need.
|
||||||
|
- **Matching stays code until the third bus.** `driverFor`/`pciDriverFor` are
|
||||||
|
honest at two bus types; the third triggers the manifest (a driver declares what
|
||||||
|
it binds: a PCI class triple, a USB class triple, an ACPI `_HID`).
|
||||||
@@ -167,3 +167,27 @@ free; discovery on x86 is partly about *finding* what ARM just tells you.
|
|||||||
- [ipc.md](ipc.md) — the channels that interrupts-as-messages and the device manager
|
- [ipc.md](ipc.md) — the channels that interrupts-as-messages and the device manager
|
||||||
will ride on.
|
will ride on.
|
||||||
- [vision.md](vision.md) — why drivers belong in isolated user space at all.
|
- [vision.md](vision.md) — why drivers belong in isolated user space at all.
|
||||||
|
|
||||||
|
## Update (M19.3, 2026-07-13): PCI enumeration left the kernel
|
||||||
|
|
||||||
|
The kernel now seeds only the `pci_host_bridge` node (ECAM window, MMIO
|
||||||
|
apertures derived from the memory map's holes, bus range, and the 16-bit I/O
|
||||||
|
window). The per-function walk moved to the ring-3 `pci-bus` driver
|
||||||
|
([device-manager.md](device-manager.md)): it claims the bridge, repeats the
|
||||||
|
ECAM scan through its mmio grant, and `device_register`s what it finds, which
|
||||||
|
the device manager mirrors and matches. The ACPI namespace walk follows in M20;
|
||||||
|
the static tables (MADT, HPET, MCFG, FADT + `\\_S5`) stay kernel-side.
|
||||||
|
|
||||||
|
## Update (M20.3, 2026-07-13): ACPI enumeration left the kernel too
|
||||||
|
|
||||||
|
The kernel no longer folds the AML namespace's Device objects into the device
|
||||||
|
tree. It still parses the *static* tables (MADT for SMP, HPET for the tick, MCFG
|
||||||
|
for the host bridge, FADT) and still builds the AML namespace — but only to read
|
||||||
|
the `\\_S5` sleep type for poweroff. Device discovery is the ring-3 **acpi
|
||||||
|
service** ([device-manager.md](device-manager.md)): it claims the `acpi-tables`
|
||||||
|
node the kernel publishes (the AML blobs, a broad io_port grant, the SCI),
|
||||||
|
re-parses the same blobs with the shared AML module, evaluates `_STA`/`_CRS`,
|
||||||
|
and registers + reports each `_HID` device — the device manager matches drivers
|
||||||
|
(ps2-bus) from those reports. With M19's pci-bus driver, discovery now runs
|
||||||
|
entirely in user space; the kernel seeds only the host bridge and the
|
||||||
|
acpi-tables node.
|
||||||
|
|||||||
@@ -362,3 +362,23 @@ the first DMA driver to protect and test against) and these smaller items:
|
|||||||
- **Interrupt priority / threaded IRQ latency.** `notifyFromIsr` enqueues the woken
|
- **Interrupt priority / threaded IRQ latency.** `notifyFromIsr` enqueues the woken
|
||||||
driver but doesn't preempt (`wakeLocked` deliberately leaves that to the caller), so
|
driver but doesn't preempt (`wakeLocked` deliberately leaves that to the caller), so
|
||||||
a woken driver waits for the next scheduling point.
|
a woken driver waits for the next scheduling point.
|
||||||
|
|
||||||
|
## The driver contract (M17–M18)
|
||||||
|
|
||||||
|
Claiming and mapping is half of being a danos driver; the other half is the
|
||||||
|
**lifecycle and protocol contract**, and the runtime makes it nearly free:
|
||||||
|
|
||||||
|
- Build on `runtime.service.run` — one replyWait loop folding protocol
|
||||||
|
requests, signals, and notifications into callbacks. The harness answers the
|
||||||
|
universal zero-length ping and turns `terminate` into a clean exit for you
|
||||||
|
([process-lifecycle.md](process-lifecycle.md)).
|
||||||
|
- A driver spawned with an assignment (its device id as argv[1]) sends the
|
||||||
|
versioned `hello` to the device manager inside the deadline, and a **bus**
|
||||||
|
driver reports what it discovers with `child_added`
|
||||||
|
([device-manager.md](device-manager.md); usb-xhci-bus is the reference
|
||||||
|
implementation).
|
||||||
|
- Crash freely — that is the design. The kernel releases your claims, IRQ
|
||||||
|
bindings, and MSI vectors at death; the manager reads your exit reason,
|
||||||
|
prunes what you reported, restarts you with backoff, and your fresh instance
|
||||||
|
re-claims and re-reports. Never depend on your own cleanup running
|
||||||
|
(iron rule 1).
|
||||||
|
|||||||
+19
@@ -103,3 +103,22 @@ This is what makes a user-space driver possible at all, and it's the subject of
|
|||||||
the oldest (discrete messages, not a coalescing level like the notification ring).
|
the oldest (discrete messages, not a coalescing level like the notification ring).
|
||||||
- **A bounded reply.** `MSG_MAX` is 256 bytes and the copy runs under the big kernel
|
- **A bounded reply.** `MSG_MAX` is 256 bytes and the copy runs under the big kernel
|
||||||
lock; a bulk transfer wants shared pages, not a copy.
|
lock; a bulk transfer wants shared pages, not a copy.
|
||||||
|
|
||||||
|
## Lifecycle conventions over IPC (M17)
|
||||||
|
|
||||||
|
Three conventions from [process-lifecycle.md](process-lifecycle.md) ride the
|
||||||
|
notification mechanism:
|
||||||
|
|
||||||
|
- **Signals** arrive as notifications on the endpoint a process nominated with
|
||||||
|
`signal_bind` (`runtime.process.bindSignals`): badge = the signal bit plus the
|
||||||
|
coalesced pending mask (`runtime.process.signalsFrom` decodes). Statements,
|
||||||
|
never questions; no payload, no reply.
|
||||||
|
- **One-shot timers** (`timer_bind`, `runtime.system.timerOnce`) land as a
|
||||||
|
timer-bit notification — the timed wait: a service arms a deadline and keeps
|
||||||
|
serving, instead of blocking in sleep.
|
||||||
|
- **The universal ping**: a **zero-length request is the liveness probe**,
|
||||||
|
answered with a zero-length reply by the service harness itself
|
||||||
|
(`runtime.service.run`). No protocol's requests start at length zero, so the
|
||||||
|
encoding cannot collide, and a wedged service simply fails to answer — which
|
||||||
|
is the diagnosis. Deep health ("can I reach my hardware?") stays a per-service
|
||||||
|
protocol message.
|
||||||
|
|||||||
@@ -0,0 +1,210 @@
|
|||||||
|
# M17–M18 execution plan: process lifecycle + device manager
|
||||||
|
|
||||||
|
**Archived — completed 2026-07-13** (every item checked; suite ended 54/54).
|
||||||
|
Kept as the record of how M17–M18 landed; the successor is
|
||||||
|
[m19-m20-plan.md](m19-m20-plan.md).
|
||||||
|
|
||||||
|
The operational plan for building [process-lifecycle.md](process-lifecycle.md)
|
||||||
|
(M17) and [device-manager.md](device-manager.md) increments 5–7 (M18). Design is
|
||||||
|
settled in those documents; this file is the build order — one phase at a time,
|
||||||
|
each phase green before the next starts. Delete or archive this file when M18
|
||||||
|
lands.
|
||||||
|
|
||||||
|
**Definition of green, every phase:** `zig build` clean, `zig build test` clean,
|
||||||
|
`python3 test/qemu_test.py` passes (existing scenarios plus the phase's new one),
|
||||||
|
and the relevant design doc's "known gaps" / status lines updated. Commit per
|
||||||
|
green phase (no co-author trailers).
|
||||||
|
|
||||||
|
**Workflow (settled 2026-07-12):** work happens in a dedicated git worktree, on
|
||||||
|
feature branches cut from `main` — `feat/process-lifecycle` (M17.1–17.4),
|
||||||
|
`feat/device-manager` (M18.1), `feat/usb-xhci-bus` (M18.2–18.3). When a branch's
|
||||||
|
phases are all green it is **auto-merged into `main`**; branches are kept after
|
||||||
|
merge, not deleted. Merges and branches are pushed to origin. Phase 0 (once):
|
||||||
|
commit the design docs, merge the outstanding `feat/usb` work into `main`, and
|
||||||
|
run the existing QEMU suite green before any new work starts.
|
||||||
|
|
||||||
|
**Numbering note:** continues the milestone sequence (driver track ended at M16).
|
||||||
|
|
||||||
|
## Status
|
||||||
|
|
||||||
|
The loop marks a phase `[x]` in the same commit that lands it. A phase is marked
|
||||||
|
only when its definition of green holds.
|
||||||
|
|
||||||
|
- [x] **Phase 0** — baseline: docs committed, feat/usb merged to main, pushed;
|
||||||
|
`usb-xhci-libary.zig` renamed to `usb-xhci-library.zig`; existing QEMU
|
||||||
|
suite green from the worktree (48/48, 2026-07-12).
|
||||||
|
- [x] **M17.1** — kernel releases claims/MSI on death (claims: `releaseAllOwnedBy`
|
||||||
|
in the reap; MSI was already swept by `irq.releaseOwner`; `claim-release`
|
||||||
|
test; suite 49/49)
|
||||||
|
- [x] **M17.2** — exit reasons (`ExitReason` recorded at exit/fault/kill before
|
||||||
|
the notification; `process_exit_reason` supervisor-gated;
|
||||||
|
`runtime.process.exitReason`; kernel + ring-3 assertions; suite 49/49)
|
||||||
|
- [x] **M17.3** — published exit events + VFS subscriber (`process_subscribe`,
|
||||||
|
bounded ref-counted table, publish on every death;
|
||||||
|
`runtime.process.subscribeExits`; VFS handles carry owners and are swept on
|
||||||
|
the owner's death; `vfs-client-death` test; suite 50/50)
|
||||||
|
- [x] **M17.4** — signals, timer notifications, `runtime.process`, the service
|
||||||
|
harness (signal_bind/process_signal + coalescing pending mask; timer_bind
|
||||||
|
on the tick; bindSignals/signalsFrom/sendSignal/stop + timerOnce;
|
||||||
|
runtime.service.run with the zero-length ping; VFS converted; `signals`
|
||||||
|
scenario; suite 51/51)
|
||||||
|
- [x] **merge** `feat/process-lifecycle` → main, push (merged 2026-07-13)
|
||||||
|
- [x] **M18.1** — device-manager protocol: hello + restart policy
|
||||||
|
(device-manager-protocol module; the manager as a harness service:
|
||||||
|
supervised spawns, hello deadline via timer sweep, restart with
|
||||||
|
300/600/1200ms backoff, exit reasons deciding restart-vs-stopped,
|
||||||
|
crash-loop cap; usb-xhci-bus first conforming driver; crash-test fixture
|
||||||
|
re-proving claim release each respawn; `driver-restart` scenario;
|
||||||
|
maximum_tasks 16→32 — the sweep was overflowing the pool; suite 52/52)
|
||||||
|
- [x] **merge** `feat/device-manager` → main, push (merged 2026-07-13)
|
||||||
|
- [x] **M18.2** — xHCI port scan + tree reports (child_added/child_removed in
|
||||||
|
the protocol; the manager's child mirror with death-pruning; xHCI maps the
|
||||||
|
register BAR — resource 0 is ECAM — reads CAPLENGTH/HCSPARAMS1, scans
|
||||||
|
PORTSC, reports connected ports with speed-class identity; `usb-report`
|
||||||
|
scenario proves report → prune → respawn → re-report; suite 53/53)
|
||||||
|
- [x] **M18.3** — app surface: enumerate/subscribe over IPC (subscriber
|
||||||
|
endpoint rides as the call's capability; events are the same structs the
|
||||||
|
buses send); device-list first client; protocol capped at the kernel's
|
||||||
|
IPC MESSAGE_MAXIMUM (256); the startUserTask debug print removed — it
|
||||||
|
sheared concurrent serial lines and was the scenario-flake root cause;
|
||||||
|
`device-list` scenario; suite 54/54)
|
||||||
|
- [x] **merge** `feat/usb-xhci-bus` → main, push (merged 2026-07-13) — **plan complete**
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## M17.1 — the kernel releases a dead process's claims
|
||||||
|
|
||||||
|
The cleanup half of iron rule 1; the prerequisite for every restart story.
|
||||||
|
|
||||||
|
- `system/kernel/devices-broker.zig`: `releaseAllOwnedBy(owner: u32)` — clear
|
||||||
|
every `claimed[]` slot holding this task id.
|
||||||
|
- `system/kernel/process.zig`: call it from the reap path, alongside the existing
|
||||||
|
IRQ-binding release (the ordering comment there says why IRQs go first — claims
|
||||||
|
slot in after them, before the exit notification).
|
||||||
|
- MSI vectors: find where `msi_bind` records per-device vectors (interrupts
|
||||||
|
module) and release those by owner in the same pass.
|
||||||
|
- Docs: remove the claims bullet from process-management.md "Known gaps".
|
||||||
|
|
||||||
|
**Test:** new QEMU scenario `claim-release` — a test child claims an unclaimed
|
||||||
|
device, is killed, is respawned, and claims the same device again successfully;
|
||||||
|
assert both claims in the serial log. Kernel-side unit coverage in
|
||||||
|
`system/kernel/tests.zig` for `releaseAllOwnedBy` (claim two devices as two owners,
|
||||||
|
release one owner, verify exactly its claims freed).
|
||||||
|
|
||||||
|
## M17.2 — exit reasons
|
||||||
|
|
||||||
|
- `system/abi.zig`: `ExitReason` (exited, aborted, segmentation_fault,
|
||||||
|
illegal_instruction, arithmetic_fault, killed).
|
||||||
|
- Kernel: record the reason at every death site — clean exit path, each fault
|
||||||
|
class in `onException`, the kill path. Bounded recent-exits table (ids are never
|
||||||
|
reused, so a small ring keyed by id is enough).
|
||||||
|
- New system call `process_exit_reason(id)` — supervisor-gated, like kill; returns
|
||||||
|
the recorded reason or `-ESRCH` once evicted.
|
||||||
|
- `library/runtime/process.zig`: `ExitReason` + `exitReason(id: u32)`.
|
||||||
|
- Docs: remove the no-exit-status bullet from process-management.md.
|
||||||
|
|
||||||
|
**Test:** extend the `supervision` scenario — three children: one exits cleanly,
|
||||||
|
one faults (the fault-recovery pattern), one is killed; the supervisor asserts all
|
||||||
|
three reasons.
|
||||||
|
|
||||||
|
## M17.3 — published exit events
|
||||||
|
|
||||||
|
- Kernel: bounded subscriber table (endpoints); new system call
|
||||||
|
`process_subscribe(endpoint)` (ungated, like `process_enumerate`); every death
|
||||||
|
posts `notify_exit_bit | id` to each subscriber — the same post the supervisor
|
||||||
|
path already uses.
|
||||||
|
- `library/runtime/process.zig`: `subscribeExits(endpoint)`.
|
||||||
|
- VFS becomes the first subscriber: on an exit event, release every handle keyed
|
||||||
|
by that task id (badges already are task ids). Log the release.
|
||||||
|
- Docs: note the convention in ipc.md (exit events reuse the exit-notification
|
||||||
|
badge encoding).
|
||||||
|
|
||||||
|
**Test:** new QEMU scenario `vfs-client-death` — a client opens a file and is
|
||||||
|
killed without closing; assert the VFS logs the handle release and its open-handle
|
||||||
|
count returns to baseline.
|
||||||
|
|
||||||
|
## M17.4 — signals and the service harness
|
||||||
|
|
||||||
|
- Kernel: per-task pending mask + bound endpoint; system calls
|
||||||
|
`signal_bind(endpoint)` and `process_signal(id, signal)` (supervisor-or-self
|
||||||
|
gated); delivery posts `notify_signal_bit | pending mask`, coalescing; pending
|
||||||
|
signals with no bound endpoint pend silently.
|
||||||
|
- `library/runtime/process.zig`: `Signal`, `SignalSet`, `bindSignals`,
|
||||||
|
`signalsFrom`, `sendSignal`, `stop(id, deadline_ms)` (terminate → wait for exit
|
||||||
|
notification → kill). Implement `terminate`, `reload`, `user_1`, `user_2`;
|
||||||
|
`interrupt`/`quit` are enum members with no sender yet; `alarm` stays unbuilt.
|
||||||
|
- Kernel: **one-shot timer notifications** — `timer_bind(endpoint, ms)` posts a
|
||||||
|
notification badge when the deadline lands (IRQ-as-IPC again, on the timer
|
||||||
|
wheel `sleep` already uses). This is the missing timed-wait primitive:
|
||||||
|
`replyWait` blocks forever and `sleep` blocks the whole process, but `stop()`'s
|
||||||
|
escalation, the device manager's `hello` deadline (M18.1), and restart backoff
|
||||||
|
all need a deadline while staying responsive. It is also the mechanism `alarm`
|
||||||
|
gets for free later.
|
||||||
|
- New `library/runtime/service.zig`: the harness — `run(callbacks)` owning the
|
||||||
|
replyWait loop, folding protocol messages, signals, and child-exit notifications
|
||||||
|
into `init` / `on_message` / `on_reload` / `on_terminate`; answers the common
|
||||||
|
`ping` automatically. Define the reserved `ping` request encoding here and
|
||||||
|
document it in ipc.md (one obvious encoding; smallest that cannot collide with
|
||||||
|
existing protocols).
|
||||||
|
- Convert one existing service (input-source or hpet) to the harness as proof it
|
||||||
|
subtracts code rather than adding it.
|
||||||
|
|
||||||
|
**Test:** extend `supervision` — a harness-built child: `sendSignal(reload)`
|
||||||
|
observed in its log, `ping` answered, `stop()` produces a clean exit with reason
|
||||||
|
`exited`; a second child that ignores signals (no bind) is killed by `stop()`'s
|
||||||
|
deadline with reason `killed`.
|
||||||
|
|
||||||
|
## M18.1 — device-manager protocol: hello + restart policy
|
||||||
|
|
||||||
|
- New `system/services/device-manager/device-manager-protocol.zig` module
|
||||||
|
(vfs-protocol pattern): `hello { version, role, device_id }`; version constant;
|
||||||
|
reserved fields.
|
||||||
|
- Device manager: register the `.device_manager` endpoint; spawn drivers with its
|
||||||
|
exit endpoint; enforce the hello deadline; restart policy — backoff, crash-loop
|
||||||
|
cap (three fast deaths → mark failed, log, stop), reasons from M17.2 deciding
|
||||||
|
restart vs not.
|
||||||
|
- usb-xhci-bus: adopt the harness + send hello. hpet/ps2-bus follow only if the
|
||||||
|
conversion is mechanical; otherwise they keep working unconverted (the manager
|
||||||
|
only enforces hello on drivers spawned with an assignment).
|
||||||
|
- build.zig: test-loop entry for the protocol module if it grows pure logic.
|
||||||
|
|
||||||
|
**Test:** new QEMU scenario `driver-restart` — the xHCI driver takes a test-only
|
||||||
|
argv flag to fault after hello on its first run; assert: fault, exit reason
|
||||||
|
recorded, manager respawns with backoff, second run claims the controller
|
||||||
|
(M17.1) and hellos clean. Assert the crash-loop cap by a driver that always
|
||||||
|
faults (a tiny test driver, not xhci).
|
||||||
|
|
||||||
|
## M18.2 — bus tree reports
|
||||||
|
|
||||||
|
- Protocol: `child_added { parent, identity, resources }` / `child_removed { id }`.
|
||||||
|
- usb-xhci-bus: bring-up to **port scan only** — map the MMIO window (claimed in
|
||||||
|
M16-era work), controller reset/start per xHCI spec, walk the port registers,
|
||||||
|
report one `child_added` per connected port with speed + port number as
|
||||||
|
identity. **No transfer rings, no descriptors** — reading device/interface
|
||||||
|
descriptors (and therefore USB class triples for matching) is the follow-on USB
|
||||||
|
track, not this plan.
|
||||||
|
- Device manager: mirror reports into its tree; prune the subtree (emitting
|
||||||
|
`child_removed`) when a bus driver dies; assert re-report on restart.
|
||||||
|
|
||||||
|
**Test:** QEMU already attaches usb-kbd + usb-mouse on xhci.0 — assert two
|
||||||
|
`child_added` events reach the manager and appear in its tree dump; kill the
|
||||||
|
driver, assert two `child_removed` then two fresh `child_added` after respawn.
|
||||||
|
|
||||||
|
## M18.3 — the application surface
|
||||||
|
|
||||||
|
- Protocol: `enumerate` (tree snapshot) + `subscribe` (published add/remove
|
||||||
|
events, input-service pattern).
|
||||||
|
- A small client (`device-list`, the `ps` analog) exercising both; the manager
|
||||||
|
becomes the one answer to "what devices exist" for user space.
|
||||||
|
`device_enumerate` stays for drivers/kernel seeding — its retreat is tied to the
|
||||||
|
discovery migration, out of this plan.
|
||||||
|
|
||||||
|
**Test:** QEMU scenario — `device-list` shows the tree including USB children;
|
||||||
|
during a driver restart the subscribing client logs remove + add events.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
**Explicitly out of scope** (own tracks, after M18): discovery migration (pci-bus
|
||||||
|
driver, acpi service, retiring the kernel scan), USB control transfers +
|
||||||
|
descriptors + class-driver matching, the musl layer, `interrupt`/`quit` senders
|
||||||
|
(needs a console), job control.
|
||||||
@@ -0,0 +1,212 @@
|
|||||||
|
# M19–M20 execution plan: discovery migration
|
||||||
|
|
||||||
|
The operational plan for [device-manager.md](device-manager.md)'s increment 8:
|
||||||
|
discovery leaves the kernel — a **pci-bus driver** (M19) and an **acpi service**
|
||||||
|
(M20), with the kernel's device enumeration retired behind them. Same rules as
|
||||||
|
[m17-m18-plan.md](m17-m18-plan.md): one phase at a time, each green before the
|
||||||
|
next; this file is the build order and the checklist.
|
||||||
|
|
||||||
|
**Definition of green, every phase:** `zig build` clean, `zig build test` clean,
|
||||||
|
`python3 test/qemu_test.py` passes (existing scenarios plus the phase's new
|
||||||
|
one), and the relevant design doc updated. Commit per green phase (no co-author
|
||||||
|
trailers). The full suite is the regression net — the existing
|
||||||
|
`driver-restart` / `usb-report` / `device-list` / `input` scenarios must stay
|
||||||
|
green *through* the migration, which is the whole point: the system must not be
|
||||||
|
able to tell who enumerated it.
|
||||||
|
|
||||||
|
**Workflow:** dedicated worktree; branches off `main` — `feat/pci-bus`
|
||||||
|
(M19.0–19.3), `feat/acpi-service` (M20.1–20.3); auto-merge to main when a
|
||||||
|
branch is green; keep branches; push everything.
|
||||||
|
|
||||||
|
## Settled decisions (2026-07-13 — veto before the loop starts)
|
||||||
|
|
||||||
|
1. **What "retiring the kernel scan" means.** The kernel keeps, forever, the
|
||||||
|
parses it needs before user space exists: RSDP/XSDT location, MADT (SMP),
|
||||||
|
the HPET table (the tick), FADT + the AML `\_S5` evaluation (poweroff — the
|
||||||
|
power tests prove it), and MCFG (the host bridge node). What retires is
|
||||||
|
**device enumeration**: the ECAM function walk (M19.3) and the DSDT/SSDT
|
||||||
|
namespace walk that builds device nodes (M20.3). The AML module stays a
|
||||||
|
shared build module compiled into both the kernel (for `\_S5`) and the acpi
|
||||||
|
service (for everything else) — same source, two builds, no fork.
|
||||||
|
2. **Bridge apertures come from the firmware memory map, not AML.** Registered
|
||||||
|
PCI functions carry BAR resources, and containment demands the bridge own
|
||||||
|
windows that cover them. The apertures are derived kernel-side from the
|
||||||
|
boot memory map's MMIO holes (regions that are neither RAM nor tables) —
|
||||||
|
mechanical, AML-free, and available at boot regardless of what later moved
|
||||||
|
to user space. (The bridge today carries only ECAM + bus range; this is the
|
||||||
|
prerequisite M19.0 exists for.)
|
||||||
|
3. **`device_register` becomes idempotent on exact match.** A re-registration
|
||||||
|
with identical (parent, class, resources) returns the existing id instead
|
||||||
|
of appending. The kernel table has no unregister, so without this a
|
||||||
|
restarted registering bus would duplicate its children on every respawn —
|
||||||
|
idempotence makes restart-and-re-report safe for every future bus, not just
|
||||||
|
PCI.
|
||||||
|
4. **The manager matches from reports.** `ChildAdded` gains a `device_id`
|
||||||
|
field (the kernel-registered id, `no_device` for unregistered leaves like
|
||||||
|
USB ports). After the M19.3 flip, PCI driver matching keys off reported
|
||||||
|
identity (the class triple) instead of the manager's boot-time snapshot —
|
||||||
|
the snapshot match remains only for what the kernel still seeds. One flip
|
||||||
|
phase changes both sides at once so no device is ever matched twice.
|
||||||
|
5. **The acpi service's authority is one node.** The kernel publishes an
|
||||||
|
`acpi-tables` device: memory resources covering the table blobs plus a
|
||||||
|
broad `io_port` resource — the documented trust grant to exactly one
|
||||||
|
process (AML OperationRegions reach EC/PM ports; the claim-gated
|
||||||
|
io_read/io_write calls already exist). The service claims it, maps the
|
||||||
|
tables, and runs the shared AML module in ring 3 behind a `Hal` backed by
|
||||||
|
`mmio_map` + `io_read`/`io_write`.
|
||||||
|
6. **Both new processes are protocol drivers** under the manager: hello,
|
||||||
|
supervision, restart with backoff — all inherited from M18.1 for free.
|
||||||
|
Registration idempotence (decision 3) is what makes their restarts sound.
|
||||||
|
7. **Firmware neutrality is the contract** (2026-07-13). The generic layer is
|
||||||
|
everything at and above the device-manager protocol — descriptors,
|
||||||
|
containment, reports, matching, supervision — and none of it may become
|
||||||
|
x86-specific. Discovery is one swappable process per firmware: the acpi
|
||||||
|
service on x86; an **fdt service** on the Raspberry Pis (claims a
|
||||||
|
`devicetree-blob` node, reports children from the flattened device tree —
|
||||||
|
pure data, no bytecode, no port grant, strictly simpler than ACPI). The
|
||||||
|
manager owns the tree as *data* and touches no hardware, ever — AML runs in
|
||||||
|
a crashable, supervised discoverer precisely so a firmware-bytecode fault
|
||||||
|
can never take down the supervisor. Two consequences recorded now:
|
||||||
|
`DeviceDescriptor`'s 8-byte `hid` cannot hold an FDT `compatible` string
|
||||||
|
("brcm,bcm2835-aux-uart") — identity widens before the fdt service exists;
|
||||||
|
and cross-firmware surfaces are named by **domain, not firmware** (M21
|
||||||
|
defines a *power* protocol, not an "ACPI events" protocol — PSCI/mailbox
|
||||||
|
sources feed the same subscribers on ARM). **Landed early (2026-07-13):**
|
||||||
|
both services exist as placeholders (system/services/acpi, system/services/
|
||||||
|
fdt) and the build's `-Ddiscovery=acpi|fdt` option fills the ramdisk's
|
||||||
|
neutral `discovery` slot — the manager will spawn "discovery" by that name
|
||||||
|
in M20.3 and never learn which firmware it is on.
|
||||||
|
|
||||||
|
## Status
|
||||||
|
|
||||||
|
- [x] **M19.0** — prerequisites (bridge apertures from the memory map's
|
||||||
|
*gaps* — the single-hole rule died on OVMF's flash at the top of 4 GiB,
|
||||||
|
caught by the new every-BAR-contained assert in `discovery`; idempotent
|
||||||
|
`device_register` proven in `bus`; `ChildAdded.device_id`;
|
||||||
|
m17-m18-plan.md archived; suite 54/54).
|
||||||
|
- [x] **M19.1** — pci-bus driver, scan only (claims the bridge, maps ECAM
|
||||||
|
through its grant, brute-force walk with the multifunction rule; the
|
||||||
|
manager matches pci_host_bridge → pci-bus per device with the full
|
||||||
|
protocol contract; `pci-scan` builds its expected marker from the
|
||||||
|
kernel's own count — equivalence on the first run; suite 55/55).
|
||||||
|
- [x] **M19.2** — register + report (BAR probe mirrored byte-for-byte from the
|
||||||
|
kernel's addBars so dedupe returns the kernel's node ids during
|
||||||
|
coexistence; the bridge gained the io_port aperture I/O BARs need;
|
||||||
|
reports carry the registered device_id; pci-scan drills a forced restart
|
||||||
|
and asserts the PCI node count never grows — plus harness hardening: a
|
||||||
|
failing case now preserves its serial as <case>-failed-serial.log, and
|
||||||
|
the heavy scenarios run at 150s; suite 55/55).
|
||||||
|
- [x] **M19.3** — the flip: kernel `enumeratePci`/`addBars`/`PciHeader` all
|
||||||
|
deleted (bridge node stays); manager matches PCI drivers from reported
|
||||||
|
identity, deduped by registered id. Surfaced and fixed a real SMP race the
|
||||||
|
flip created — ring-3 device_register made the broker table concurrent, so
|
||||||
|
mmio_map's lock-free read intermittently tore hpet's resource length
|
||||||
|
(user fault) and overflowed `r.len-1` into a kernel panic; now the broker
|
||||||
|
read is under the big lock and the arithmetic is guarded, and pci-bus
|
||||||
|
skips size-0 BARs. discovery.md updated; suite 55/55 (driver-restart
|
||||||
|
hammered 6×).
|
||||||
|
- [x] **merge** `feat/pci-bus` → main, push (merged 2026-07-13).
|
||||||
|
- [x] **M20.1** — acpi service, parse only: the AML interpreter is now a build
|
||||||
|
module compiled into both kernel and service; the kernel publishes the
|
||||||
|
`acpi-tables` node (AML blobs as memory resources, the broad io_port grant,
|
||||||
|
the SCI); the service claims it, maps the blobs, runs the shared parser in
|
||||||
|
ring 3, and self-verifies its Device count against the kernel's (34 = 34,
|
||||||
|
deterministic via argv, no log-scraping); the manager spawns `discovery`
|
||||||
|
at startup. Parse-only touches no hardware. Suite 56/56.
|
||||||
|
|
||||||
|
- [x] **M20.2** — register + report: the service evaluates `_STA`/`_CRS` in
|
||||||
|
ring 3 (interpreter Hal = port I/O over the claimed node; a scratch page
|
||||||
|
backs SystemMemory maps so a stray region can't fault it) and registers +
|
||||||
|
reports each present `_HID` device under `acpi-tables`. Containment: the
|
||||||
|
broker's irq check became range-based (len-1 == the old equality) so the
|
||||||
|
node's broad irq window covers children's legacy lines; io ports fall in
|
||||||
|
the broad io grant. ChildAdded gained `hid`. Matching stays off. The
|
||||||
|
`acpi-report` scenario asserts the PS/2 keyboard (3 resources) and mouse
|
||||||
|
(1 resource) among the reports. Suite 57/57.
|
||||||
|
- [x] **M20.3** — the flip: the kernel's `wireAcpiDevices` call is gone (the
|
||||||
|
device-building helpers are retained-but-dead pending a focused sweep,
|
||||||
|
spawned as a task; static tables + `\_S5` + the acpi-tables node stay).
|
||||||
|
The manager matches ps2-bus from ACPI `_HID` reports; the service
|
||||||
|
registers all devices before reporting any (no keyboard-before-mouse
|
||||||
|
race). The `acpi-ps2` scenario proves report → spawn → ps2-bus attaches
|
||||||
|
its keyboard; `ioport` retargeted to the acpi-tables I/O window (the
|
||||||
|
kernel-built PS/2 node is gone). Suite 58/58.
|
||||||
|
- [ ] **merge** `feat/acpi-service` → main, push — **loop ends here**.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Phase notes
|
||||||
|
|
||||||
|
**M19.0 apertures:** the boot memory map already crosses the handoff
|
||||||
|
([boot-handoff]), but discovery never sees it today — expect a small
|
||||||
|
pass-through (kernel init hands the map to the platform layer) before the
|
||||||
|
holes computation, which belongs where the bridge node is built
|
||||||
|
(`parseMcfg`). Sanity-check on QEMU q35: the xHCI BAR (`0xc0000000`-region
|
||||||
|
values seen in the M18 logs) must land inside a derived aperture, asserted in
|
||||||
|
the kernel unit test.
|
||||||
|
|
||||||
|
**M19.1 scanning without owning config access twice:** the driver reads config
|
||||||
|
space through its ECAM mmio_map grant of the *bridge* window — the same bytes
|
||||||
|
the kernel walk read. Vendor-id `0xFFFF` skip, header-type multifunction rule,
|
||||||
|
no bridge recursion (matches the kernel's current single-segment walk).
|
||||||
|
|
||||||
|
**M19.2 BAR sizing:** the classic size probe (write all-ones, read mask,
|
||||||
|
restore) is deferred — the BARs' current programmed values and types are
|
||||||
|
enough for containment-checked registration at bring-up; sizing lands with the
|
||||||
|
first driver that needs to *move* a BAR. Log what is registered so the
|
||||||
|
scenario can assert it.
|
||||||
|
|
||||||
|
**M19.3 what the manager still seeds from the snapshot:** everything the
|
||||||
|
kernel still enumerates (timers, ACPI nodes until M20.3). The PCI arm of
|
||||||
|
`pciDriverFor` switches source; `driverFor` doesn't move until M20.3.
|
||||||
|
|
||||||
|
**M20.1 spawn and identity (pre-settled 2026-07-13):** the manager spawns
|
||||||
|
`discovery` by its neutral ramdisk name at startup, as an ordinary protocol
|
||||||
|
driver (hello, supervision) — from M20.1 on, on every boot. For reporting ACPI
|
||||||
|
devices, `ChildAdded` gains `hid: [8]u8` (EISA ids fit; zero = none):
|
||||||
|
firmware *string* identity travels beside the numeric `identity` field until
|
||||||
|
the FDT-driven widening replaces both (decision 7).
|
||||||
|
|
||||||
|
**M20.1 Hal in ring 3:** `mapMmio` → `device.mmioMap` over the claimed
|
||||||
|
acpi-tables node (plus a table-offset map for blobs); `pioRead`/`pioWrite` →
|
||||||
|
`device.ioRead`/`ioWrite` against its io_port resource. The interpreter cannot
|
||||||
|
tell it moved — that is the assertion of `acpi-parse`.
|
||||||
|
|
||||||
|
**M20.2 containment for `_CRS`:** io ports fall inside the node's broad
|
||||||
|
io_port resource; MMIO windows (HPET, LAPIC ranges some firmwares list) fall
|
||||||
|
inside the memory-map holes added to the node in M20.1. Anything that doesn't
|
||||||
|
fit is logged and skipped, loudly — bring-up honesty over silent drops.
|
||||||
|
|
||||||
|
**M20.3 ps2 ordering:** ps2-bus binds nodes the acpi service now reports, so
|
||||||
|
its spawn moves behind the report (the manager's matching handles this once
|
||||||
|
the source flips); the `input` scenario proves the keyboard still types.
|
||||||
|
|
||||||
|
**Explicitly out of scope:** PCI bridge recursion (single segment, flat bus
|
||||||
|
walk stays); BAR reprogramming/sizing; disk/PCIe hotplug; interrupt routing
|
||||||
|
changes (`_PRT` stays wherever it is today); the USB descriptor track;
|
||||||
|
multi-segment ECAM; per-device power states (D-states, `_PSx`/`_PRx`,
|
||||||
|
suspend/resume — a future *lifecycle-vocabulary* extension, since "suspend"
|
||||||
|
has the shape of a signal every driver must answer, and it has no consumer
|
||||||
|
until laptop sleep); CPU P/C-states.
|
||||||
|
|
||||||
|
## M21 preview — ACPI events + system power (planned next, not in this loop)
|
||||||
|
|
||||||
|
The acpi service grows the event side (settled direction 2026-07-13; detailed
|
||||||
|
phases when M20 lands):
|
||||||
|
|
||||||
|
- **21.1 SCI + fixed events**: irq_bind the SCI (the resource M20.1 already
|
||||||
|
records), read/clear PM1 status, publish the power-button event to
|
||||||
|
subscribers (the same pub/sub shape the manager uses).
|
||||||
|
- **21.2 GPE + Notify**: Notify dispatch in the shared AML interpreter, GPE
|
||||||
|
block handling, `Notify(device, code)` published per reported node. The
|
||||||
|
acpi service is a **bus** here: battery (PNP0C0A), AC (ACPI0003), and lid
|
||||||
|
(PNP0C0D) nodes are reported children; small class drivers bind them and
|
||||||
|
speak an evaluate/subscribe protocol to the service — the xHCI split,
|
||||||
|
repeated. The embedded controller (`_Qxx` queries) rides this phase;
|
||||||
|
QEMU emulates no battery/EC, so those paths are interface-complete and
|
||||||
|
validated on real hardware (the laptop is the win condition), while the
|
||||||
|
plumbing is proven by the power button.
|
||||||
|
- **21.3 the capstone**: QEMU `system_powerdown` → acpi service event → init
|
||||||
|
runs the M17 stop sequence over its children → kernel `\_S5` — orderly
|
||||||
|
shutdown as the scenario that proves lifecycle + events compose. (The
|
||||||
|
harness grows a QMP poke to inject the event.)
|
||||||
@@ -0,0 +1,327 @@
|
|||||||
|
# Process lifecycle: signals over IPC
|
||||||
|
|
||||||
|
**Status: increments 1–4 built** (2026-07-12): claim release on death, exit
|
||||||
|
reasons, published exit events, and signals + one-shot timers + the service
|
||||||
|
harness are all in — the interface below is as-built. The primitives underneath
|
||||||
|
predate this design ([process-management.md](process-management.md):
|
||||||
|
spawn, the supervision link, kill, child-exit notifications); this document designs
|
||||||
|
the layer above them — the standard vocabulary a danos process speaks about its own
|
||||||
|
life, and the stable `runtime.process` interface that carries it. Nothing here is
|
||||||
|
device- or driver-specific: a driver, the VFS, and a user application all stop,
|
||||||
|
reload, and die the same way. The device manager is simply this design's first
|
||||||
|
serious customer ([device-manager.md](device-manager.md)).
|
||||||
|
|
||||||
|
**"POSIX" in this document means the concepts, never the letter of the standard.**
|
||||||
|
danos borrows the ideas and the hard-won lessons (what SIGTERM *means*, why SIGPIPE
|
||||||
|
was a mistake) without inheriting the mechanism, the API, or the names. The naming
|
||||||
|
rule is danos's own and it is strict: plain words that communicate intent
|
||||||
|
(`terminate`, `reload`, `exited`) and the IPC vocabulary the system already speaks
|
||||||
|
(`bind`, `subscribe`, `publish`, `endpoint`) — never `SIG*`, never a second word for
|
||||||
|
a concept that already has one. Literal POSIX arrives later and lives elsewhere: a
|
||||||
|
**musl-based C layer** (growing out of library/posix) that wires C programs to the
|
||||||
|
danos runtime — musl's syscall surface retargeted at danos system calls and IPC
|
||||||
|
protocols (files onto the VFS protocol, `sigaction`/`wait` onto this lifecycle,
|
||||||
|
sockets onto whatever networking becomes). Ported programs see POSIX; the system
|
||||||
|
underneath never does.
|
||||||
|
|
||||||
|
## Why a standard vocabulary
|
||||||
|
|
||||||
|
A supervisor can only manage processes it has never heard of if "please exit" means
|
||||||
|
the same thing to all of them. That is the one thing POSIX signals got deeply right:
|
||||||
|
`SIGTERM` means the same thing to nginx and to a five-line script, which is why
|
||||||
|
process supervision on Unix (init systems, container runtimes) is possible at all.
|
||||||
|
danos wants that property from day one, because supervision-and-restart is the
|
||||||
|
system's core motivation ([resilience.md](resilience.md)).
|
||||||
|
|
||||||
|
What POSIX got wrong — for a system like this — is the **delivery mechanism**:
|
||||||
|
asynchronous control-flow hijack. A Unix handler runs on a stolen stack at an
|
||||||
|
arbitrary instruction boundary, which is why the async-signal-safe function list
|
||||||
|
exists, why `errno` must be saved, and why the canonical signal bug is a SIGTERM
|
||||||
|
handler innocently calling `printf` mid-`malloc`. That entire bug class comes from
|
||||||
|
the mechanism, not the vocabulary, and none of it is worth importing.
|
||||||
|
|
||||||
|
A microkernel already has the right channel: **a signal is a message.** QNX delivers
|
||||||
|
POSIX signals over its message passing; seL4 has notification objects; Erlang turned
|
||||||
|
"death is a message to whoever linked" into a reliability philosophy. danos has
|
||||||
|
already done it once without naming it: a child's death arrives as a notification
|
||||||
|
badge on the supervisor's endpoint — the microkernel's SIGCHLD, the IRQ-as-IPC
|
||||||
|
pattern reused. Signals are the same pattern reused a third time.
|
||||||
|
|
||||||
|
## The mechanism
|
||||||
|
|
||||||
|
- **`signal_bind(endpoint)`** — a process nominates the endpoint its signals arrive
|
||||||
|
on, exactly as `irq_bind` nominates where a device's interrupts land. The runtime
|
||||||
|
does this at startup for any program that opts in.
|
||||||
|
- **`process_signal(id, signal)`** — posts the signal as an asynchronous
|
||||||
|
notification to the target's bound endpoint: badge = `notify_badge_bit |
|
||||||
|
notify_signal_bit | pending signals`. Non-blocking for the sender, always.
|
||||||
|
- **Pending signals coalesce** in a per-process bitmask until the target next waits
|
||||||
|
— exactly like interrupt notifications, and exactly POSIX's own semantics for
|
||||||
|
non-realtime signals (two pending SIGTERMs are one SIGTERM). The bitmask *is* the
|
||||||
|
design: signals carry no payload. Anything with a payload is a protocol message.
|
||||||
|
- **Authority**: the supervisor may signal its children — the same link that is
|
||||||
|
already the kill authority. A process may signal itself. Anything broader waits
|
||||||
|
for transferable process handles.
|
||||||
|
- **No binding, no problem**: a process that never calls `signal_bind` is not
|
||||||
|
broken — its signals pend unread and only `process_kill` works on it. Simple
|
||||||
|
programs stay simple; the vocabulary is opt-in, the kill authority is not.
|
||||||
|
|
||||||
|
Because delivery is a message into the process's own event loop, there is no
|
||||||
|
async-signal-safe list in danos: a handler is ordinary code running at a point the
|
||||||
|
process chose. The bug class is gone by construction, not by discipline.
|
||||||
|
|
||||||
|
## The vocabulary: POSIX.1-1990, sorted honestly
|
||||||
|
|
||||||
|
The full 1990 set, and what each becomes. Two intrinsically problematic cases get a
|
||||||
|
defense below the table.
|
||||||
|
|
||||||
|
| POSIX.1-1990 | danos disposition | Notes |
|
||||||
|
|---|---|---|
|
||||||
|
| SIGTERM | signal `terminate` | finish up and exit; the supervisor's polite half |
|
||||||
|
| SIGHUP | signal `reload` | re-read configuration / re-scan |
|
||||||
|
| SIGINT | signal `interrupt` | interactive interrupt; meaningful once a console can send it, in the vocabulary now so numbering is stable |
|
||||||
|
| SIGQUIT | signal `quit` | as SIGINT, without the core-dump baggage |
|
||||||
|
| SIGALRM | signal `alarm` | timer expiry as a message; the Unix SIGALRM+`longjmp` timeout hacks are impossible here. In the vocabulary, unbuilt: no consumer yet, and when one appears it is runtime sugar over the existing timer — zero kernel work |
|
||||||
|
| SIGUSR1, SIGUSR2 | signals `user_1`, `user_2` | service-defined |
|
||||||
|
| SIGCHLD | **already exists** — the exit notification | the badge carries the child id, dodging the classic coalescing bug (Unix code must loop `waitpid`) |
|
||||||
|
| SIGKILL | `process_kill` — kernel mechanism | its definition is "cannot be handled"; it was never really a signal |
|
||||||
|
| SIGABRT | exit reason `abort` | `abort()` is synchronous self-termination, not an event |
|
||||||
|
| SIGSEGV, SIGILL, SIGFPE | exit reasons, **never delivered** | see below |
|
||||||
|
| SIGPIPE | **an error return**, not a signal | see below |
|
||||||
|
| SIGSTOP, SIGTSTP, SIGTTIN, SIGTTOU, SIGCONT | deferred | job control needs terminals, sessions, and process groups; stop/continue is scheduler territory |
|
||||||
|
|
||||||
|
**The fault signals (SIGSEGV, SIGILL, SIGFPE) are intrinsically wrong for messages.**
|
||||||
|
They are *synchronous* — raised at a specific faulting instruction, not "sometime
|
||||||
|
soon". A message cannot be delivered to a process whose next instruction re-faults;
|
||||||
|
it never reaches its event loop to read it. POSIX only makes fault handlers "work"
|
||||||
|
via the async hijack (run the handler *instead of* the instruction), and even there,
|
||||||
|
returning from a SIGSEGV handler without curing the cause is undefined behavior.
|
||||||
|
danos's architecture already has the better answer: fault → the kernel kills the
|
||||||
|
process ([resilience.md](resilience.md) step 2, built) → the supervisor reads the
|
||||||
|
reason → restart. Recovery is restart, not a handler. This is also truer to the 1990
|
||||||
|
standard than handling is: the standard's default action for all three was
|
||||||
|
"terminate the process".
|
||||||
|
|
||||||
|
**SIGPIPE deserves special contempt.** Its default kills a process that writes to a
|
||||||
|
closed pipe — which is why "the whole server died because one client disconnected"
|
||||||
|
is roughly every network daemon's first production bug, and why every mature codebase
|
||||||
|
contains the same fix: ignore SIGPIPE, handle the `EPIPE` error return. danos made
|
||||||
|
the right choice natively already — a reply owed to a dead peer fails with `-EPEER`.
|
||||||
|
Errors from operations are error returns from those operations. The posix layer can
|
||||||
|
synthesize SIGPIPE for ported code that expects it.
|
||||||
|
|
||||||
|
### Statements, not questions
|
||||||
|
|
||||||
|
A signal and a protocol message both travel over IPC — the difference is the
|
||||||
|
**contract**, not the transport. danos IPC has two primitives, both already in
|
||||||
|
daily use: the **asynchronous notification** (a badge — bits that coalesce into a
|
||||||
|
pending mask; the sender never blocks; no payload, *no reply path*; how IRQs and
|
||||||
|
exit events arrive) and the **synchronous call** (a rendezvous — payload both
|
||||||
|
ways, the caller waits for the reply; how VFS requests work). A signal is the
|
||||||
|
first kind: a *statement*. `terminate` wants no reply — the exit notification is
|
||||||
|
its acknowledgement.
|
||||||
|
|
||||||
|
A health probe is the second kind: a *question*, worthless without its answer —
|
||||||
|
and the answer's absence within a deadline is the very thing being measured.
|
||||||
|
Asked as a signal it has no reply channel (a coalescing bit can't carry an answer,
|
||||||
|
and the authority rule forbids a child signalling its supervisor back); asked as a
|
||||||
|
call, the timeout-is-the-diagnosis semantics come free. So there is no `health`
|
||||||
|
signal. Liveness is the common **`ping`**: a reserved request every harness-run
|
||||||
|
service answers automatically on its main endpoint — still free for the service
|
||||||
|
author, still one obvious way — and a supervisor's probe is a `ping` call with a
|
||||||
|
deadline.
|
||||||
|
|
||||||
|
## The two iron rules
|
||||||
|
|
||||||
|
1. **Cleanup is the kernel's job.** A process can die with no warning — fault,
|
||||||
|
kill, power. Correctness must never depend on a `terminate` handler running. On
|
||||||
|
any death the kernel releases the address space, IPC handles, IRQ bindings, and
|
||||||
|
owed replies (built), and must also release **device, I/O-port, and interrupt
|
||||||
|
claims and MSI vectors** (the known gap in
|
||||||
|
[process-management.md](process-management.md); increment 1). A signal handler is
|
||||||
|
for *graceful* work — flushing, deregistering, saving — never for *necessary*
|
||||||
|
work.
|
||||||
|
2. **Kill is not a signal, and exit reasons are load-bearing.** The standard stop
|
||||||
|
sequence is *terminate → deadline → `process_kill`*; the unhandleable kill stays
|
||||||
|
a kernel mechanism. And a supervisor deciding whether to restart must know *how*
|
||||||
|
the child died: clean exit (meant to — don't restart), fault (restart with
|
||||||
|
backoff), killed (the supervisor did it). The exit notification today carries
|
||||||
|
only the id; it grows a reason. Restart policy cannot be written without it.
|
||||||
|
|
||||||
|
## Who learns of a death
|
||||||
|
|
||||||
|
A death has three audiences, and conflating them is how systems end up with either
|
||||||
|
zombie state or privileged snooping:
|
||||||
|
|
||||||
|
1. **The supervisor** — gets the exit notification on the endpoint it gave at spawn
|
||||||
|
(built), which grows the `ExitReason` (increment 2). The supervisor is the only
|
||||||
|
audience that needs the *reason*, because it is the only one deciding whether to
|
||||||
|
restart.
|
||||||
|
2. **The peer owed a reply** — already built: a client that dies mid-request fails
|
||||||
|
the server's reply with `-EPEER`; a server that dies fails its waiting clients
|
||||||
|
the same way. This covers the *synchronous* case only.
|
||||||
|
3. **The subscribers** — the new piece, and it is the input service's
|
||||||
|
publish/subscribe shape ([input.md](input.md)) applied to exits. A stateful
|
||||||
|
service accumulates per-client state across many requests: the VFS holds a dead
|
||||||
|
client's open file handles, the input service holds its subscriptions, a future
|
||||||
|
network stack holds its sockets. None of these are the client's supervisor, and
|
||||||
|
none learn anything from a failed reply if the client simply never calls again.
|
||||||
|
So the kernel **publishes every exit** to whoever subscribed:
|
||||||
|
`process_subscribe(endpoint)` adds a subscriber, and each death posts a
|
||||||
|
notification to every subscriber (badge = `notify_exit_bit | process id` — the
|
||||||
|
same encoding supervisors already decode, the IRQ-as-IPC pattern once more). The
|
||||||
|
subscriber filters for ids it holds state for and releases what the dead client
|
||||||
|
held. Correlating is free of bookkeeping: an IPC sender's badge already *is* its
|
||||||
|
task id (`runtime.ipc.Received`), so the id a service has been keying client
|
||||||
|
state by all along is the id the exit event carries.
|
||||||
|
|
||||||
|
Subscription, not broadcast-to-everyone: only processes that asked receive
|
||||||
|
events, the kernel keeps a bounded subscriber table, and delivery is the same
|
||||||
|
non-blocking coalescing notification as everything else — a dying process never
|
||||||
|
waits on its mourners. Subscribing is ungated, like `process_enumerate`: what is
|
||||||
|
running (and dying) is not a secret between cooperating processes. Subscribers
|
||||||
|
do not receive the exit reason — the VFS does not care *why* the client died.
|
||||||
|
|
||||||
|
This is the service-side mirror of iron rule 1: **a service must never depend on
|
||||||
|
its clients cleaning up after themselves.** Handle release on client death is the
|
||||||
|
service's job, triggered by the published exit event — never by a courtesy
|
||||||
|
"closing now" message that a crashed client will never send.
|
||||||
|
|
||||||
|
## The stable interface: `runtime.process`
|
||||||
|
|
||||||
|
`runtime.process` already owns what a process receives at birth (`Init`, the
|
||||||
|
argv contract). It grows to own the other end of life.
|
||||||
|
|
||||||
|
**The runtime is the stable interface; the numbers are not.** danos applications do
|
||||||
|
not make system calls — they call the runtime library, and the system-call numbers,
|
||||||
|
notification bits, and signal bit positions beneath it are a **private kernel ↔
|
||||||
|
runtime contract** that may change at any time (settled 2026-07-12). This is why
|
||||||
|
the runtime exists. Today kernel and runtime ship from one tree in one image, so
|
||||||
|
"stability" is simply building them together. When driver binaries start shipping
|
||||||
|
as separately-versioned applications — the whole point of the restart design — the
|
||||||
|
binary's embedded runtime version becomes compatibility metadata (the same idea as
|
||||||
|
the protocol version in the device manager's `hello`), and the kernel refuses what
|
||||||
|
it cannot serve. Signals therefore need no reserved numbering scheme: the enum
|
||||||
|
below is vocabulary, not ABI.
|
||||||
|
|
||||||
|
```zig
|
||||||
|
/// The signal vocabulary. The value is the bit position in the pending mask — a
|
||||||
|
/// private kernel/runtime detail, free to change while they ship together.
|
||||||
|
pub const Signal = enum(u5) {
|
||||||
|
terminate = 0, // SIGTERM: finish up and exit
|
||||||
|
reload = 1, // SIGHUP: re-read configuration
|
||||||
|
interrupt = 2, // SIGINT
|
||||||
|
quit = 3, // SIGQUIT
|
||||||
|
alarm = 4, // SIGALRM
|
||||||
|
user_1 = 5, // SIGUSR1
|
||||||
|
user_2 = 6, // SIGUSR2
|
||||||
|
};
|
||||||
|
|
||||||
|
/// A decoded pending mask: the coalesced set of signals a notification delivered.
|
||||||
|
pub const SignalSet = struct {
|
||||||
|
pending: u32,
|
||||||
|
pub fn has(set: SignalSet, signal: Signal) bool { ... }
|
||||||
|
pub fn iterate(set: SignalSet) Iterator { ... }
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Nominate `endpoint` as this process's signal endpoint (signal_bind). The
|
||||||
|
/// runtime's service harness calls this; a bare program may call it directly and
|
||||||
|
/// fold signals into its own replyWait loop.
|
||||||
|
pub fn bindSignals(endpoint: usize) bool { ... }
|
||||||
|
|
||||||
|
/// Decode a received badge into signals, or null if the badge is not a signal
|
||||||
|
/// notification (mirrors ipc.Received.isChildExit).
|
||||||
|
pub fn signalsFrom(badge: usize) ?SignalSet { ... }
|
||||||
|
|
||||||
|
/// Send `signal` to process `id`. Supervisor-gated, like kill; non-blocking.
|
||||||
|
pub fn sendSignal(id: u32, signal: Signal) bool { ... }
|
||||||
|
|
||||||
|
/// The standard stop sequence: terminate, wait up to `deadline_ms` for the exit
|
||||||
|
/// notification, then process_kill. The one call a supervisor needs.
|
||||||
|
pub fn stop(id: u32, deadline_ms: u64) void { ... }
|
||||||
|
|
||||||
|
/// Subscribe `endpoint` to published exit events (process_subscribe). Every
|
||||||
|
/// process death posts an asynchronous notification: badge = notify_exit_bit |
|
||||||
|
/// process id — the same encoding a supervisor's exit notification uses, decoded
|
||||||
|
/// by the same ipc.Received helpers. For stateful services: release what the dead
|
||||||
|
/// client held (file handles, subscriptions, sockets). Ungated, like
|
||||||
|
/// process_enumerate.
|
||||||
|
pub fn subscribeExits(endpoint: usize) bool { ... }
|
||||||
|
|
||||||
|
/// How a process ended — queried after the exit notification (the kernel records
|
||||||
|
/// it first, so the two never race). What restart policy reads. (Built in M17.2.)
|
||||||
|
pub const ExitReason = enum(u8) {
|
||||||
|
exited, // returned from main / clean exit
|
||||||
|
aborted, // abort() — deliberate self-termination (SIGABRT's ghost; reserved)
|
||||||
|
segmentation_fault, // SIGSEGV's ghost
|
||||||
|
illegal_instruction, // SIGILL's ghost
|
||||||
|
arithmetic_fault, // SIGFPE's ghost
|
||||||
|
protection_fault, // general protection fault
|
||||||
|
fault, // any other CPU exception
|
||||||
|
killed, // process_kill
|
||||||
|
};
|
||||||
|
```
|
||||||
|
|
||||||
|
Two deliberate absences. There is no `mask`/`block` API — a process that is not
|
||||||
|
ready for a signal simply has not waited on its endpoint yet; the pending mask *is*
|
||||||
|
the blocked set. And there is no per-signal handler registration at this layer —
|
||||||
|
dispatch is the process's own `switch` over `SignalSet`, or the service harness's
|
||||||
|
callbacks (`on_terminate`, `on_reload`) for programs that want defaults.
|
||||||
|
|
||||||
|
### The service harness
|
||||||
|
|
||||||
|
`runtime.service` owns the `replyWait` loop and folds every event source — signals,
|
||||||
|
child exits, protocol messages — into callbacks, with the vocabulary's defaults:
|
||||||
|
`terminate` returns from the loop (clean exit), the common `ping` is answered automatically,
|
||||||
|
`reload` is ignored unless overridden. One loop, no locking, nothing reentrant. A
|
||||||
|
service author writes domain logic; the lifecycle contract is satisfied by the
|
||||||
|
harness. A process that bypasses the harness and ignores its signals meets the
|
||||||
|
deadline-then-kill escalation — you cannot force a process to implement an
|
||||||
|
interface, but you can make compliance free and non-compliance fatal.
|
||||||
|
|
||||||
|
### The musl layer later
|
||||||
|
|
||||||
|
The POSIX C layer is a **musl port**: musl's arch/syscall layer retargeted so that
|
||||||
|
what musl believes are kernel syscalls become danos runtime calls and IPC — `open`
|
||||||
|
and `read` onto the VFS protocol, `kill`/`sigaction`/`waitpid` onto this document's
|
||||||
|
vocabulary, `exit` onto the runtime's exit path. `sigaction` handlers registered
|
||||||
|
through it are invoked by the runtime's loop when the signal message arrives —
|
||||||
|
synchronous underneath, async-looking to ported code, delivered at wait boundaries
|
||||||
|
the way most Unix programs already experience signals (at syscalls). No stack hijack
|
||||||
|
ever happens, `SA_RESTART` semantics come free because nothing was interrupted, and
|
||||||
|
SIGPIPE can be synthesized from `-EPEER` for the programs that expect it. C programs
|
||||||
|
get POSIX; danos-native programs never pay for it.
|
||||||
|
|
||||||
|
## Increments
|
||||||
|
|
||||||
|
1. **Kernel: release device/port/IRQ claims and MSI vectors on death** — the
|
||||||
|
cleanup half of iron rule 1, and the prerequisite for any restart story. Test:
|
||||||
|
kill a claiming driver, spawn it again, the claim succeeds.
|
||||||
|
2. **Exit reason in the death notification** (`ExitReason` above).
|
||||||
|
3. **Exit events**: `process_subscribe` in the kernel (bounded subscriber table,
|
||||||
|
publishes on every death), `runtime.process.subscribeExits`; the VFS becomes the
|
||||||
|
first subscriber — releasing a dead client's handles is its proof test.
|
||||||
|
4. **Signals**: `signal_bind` + `process_signal` + the pending mask in the kernel;
|
||||||
|
`runtime.process` grows the interface above; the service harness handles
|
||||||
|
`terminate` and answers the common `ping`; `stop()` for supervisors.
|
||||||
|
|
||||||
|
[device-manager.md](device-manager.md) builds directly on all four.
|
||||||
|
|
||||||
|
## Settled questions (2026-07-12)
|
||||||
|
|
||||||
|
- **Signal numbering is not ABI**: the runtime is the stable interface; the numbers
|
||||||
|
beneath it are a private kernel ↔ runtime contract (see "The stable interface").
|
||||||
|
- **Liveness is a `ping` call, not a signal**: signals are statements, questions
|
||||||
|
are synchronous calls (see "Statements, not questions"). A service wanting *deep*
|
||||||
|
health ("can I reach my hardware?") defines its own protocol message on top.
|
||||||
|
- **Process handles: deferred.** Pids + the supervisor gate cover everything
|
||||||
|
planned; transferable handles (Fuchsia-style, delegating signalling without
|
||||||
|
delegating kill) wait for the capability table to grow types beyond endpoints.
|
||||||
|
- **`alarm`: in the vocabulary, unbuilt.** No consumer yet; when one appears it is
|
||||||
|
runtime sugar over the existing timer (arm a timer that posts your own signal) —
|
||||||
|
zero kernel work, so deferring costs nothing.
|
||||||
|
- **Subscription granularity: all exits**, subscriber-side filtering — one
|
||||||
|
subscription per service, a bounded kernel table. Per-id subscriptions only if
|
||||||
|
event volume ever matters (hundreds of processes, not before).
|
||||||
|
- **Client identity across the exit boundary: no convention needed** — an IPC
|
||||||
|
sender's badge already is its task id (see "Who learns of a death").
|
||||||
@@ -95,11 +95,18 @@ the architecture layer calls up into `tick`.
|
|||||||
|
|
||||||
## Known gaps (bring-up honesty)
|
## Known gaps (bring-up honesty)
|
||||||
|
|
||||||
- Device **claims** are not released on death (pre-existing: the fault path has
|
- ~~Device claims are not released on death~~ Closed (M17.1): every path out of a
|
||||||
the same gap) — a killed driver's device stays claimed until reboot.
|
process releases its device claims alongside its IRQ and MSI bindings
|
||||||
|
(`releaseTaskResourcesLocked`), so a restarted driver can claim its hardware
|
||||||
|
again — the cleanup half of [process-lifecycle.md](process-lifecycle.md)'s iron
|
||||||
|
rule 1. The `claim-release` test proves the kill → release → re-claim cycle.
|
||||||
- Kernel stacks of dead tasks are leaked, as on every exit path (no reaper yet).
|
- Kernel stacks of dead tasks are leaked, as on every exit path (no reaper yet).
|
||||||
- There is no exit *status* in the notification, only the id; a supervisor that
|
- ~~There is no exit status in the notification~~ Closed (M17.2): the kernel
|
||||||
needs the code can grow a wait-style call later.
|
records how every process ends — exited, a fault class, or killed — before it
|
||||||
|
posts the exit notification, and the supervisor reads it with
|
||||||
|
`process_exit_reason` (`runtime.process.exitReason`). This is the input to
|
||||||
|
restart policy ([process-lifecycle.md](process-lifecycle.md)); an exit *code*
|
||||||
|
for the clean case can still ride alongside later.
|
||||||
- Enumerate writes through the caller's raw pointer under the bring-up trust
|
- Enumerate writes through the caller's raw pointer under the bring-up trust
|
||||||
model, like `device_enumerate` (an unmapped page is a self-DoS, not an
|
model, like `device_enumerate` (an unmapped page is a self-DoS, not an
|
||||||
isolation break).
|
isolation break).
|
||||||
@@ -109,4 +116,5 @@ the architecture layer calls up into `tick`.
|
|||||||
`process-list` (enumerate), `process-kill` (kernel-level kill paths, refusals,
|
`process-list` (enumerate), `process-kill` (kernel-level kill paths, refusals,
|
||||||
notifications), `supervision` (the whole user-side surface via the process-test
|
notifications), `supervision` (the whole user-side surface via the process-test
|
||||||
service: spawn supervised → enumerate → kill blocked and spinning children →
|
service: spawn supervised → enumerate → kill blocked and spinning children →
|
||||||
notifications → gone). See test/qemu_test.py.
|
notifications → gone), `claim-release` (a killed claim-holder's device is
|
||||||
|
claimable again). See test/qemu_test.py.
|
||||||
|
|||||||
+12
-5
@@ -1,10 +1,17 @@
|
|||||||
# Resilience: fault isolation and live restart
|
# Resilience: fault isolation and live restart
|
||||||
|
|
||||||
Steps 1–2 of the ordering below are **built**: user-mode isolation, and fault →
|
Steps 1–4 of the ordering below are **built** (M17–M18, 2026-07-13): user-mode
|
||||||
kill the process → keep the core (`onException` in `system/kernel/kernel.zig`; the
|
isolation; fault → kill the process → keep the core (`onException`; the
|
||||||
`fault-recovery` test proves a crashing ring-3 process dies alone while the system
|
`fault-recovery` test); the supervisor notification **with exit reasons**
|
||||||
keeps running). The supervisor notification and restart policy (steps 3+) are
|
([process-lifecycle.md](process-lifecycle.md) — clean exit, fault class, or
|
||||||
still design. This is the property danos is really chasing:
|
killed, recorded before the notice posts); and the **restart policy itself**
|
||||||
|
([device-manager.md](device-manager.md)): the device manager supervises every
|
||||||
|
driver, restarts crashes with backoff, caps crash loops, and re-claims work
|
||||||
|
because the kernel releases a dead process's claims. The `driver-restart` and
|
||||||
|
`usb-report` scenarios prove kill → release → respawn → re-claim → re-report
|
||||||
|
end to end. What remains of this document's ladder is scope, not mechanism:
|
||||||
|
more of the system moved into restartable processes (the discovery migration,
|
||||||
|
[m19-m20-plan.md](m19-m20-plan.md), is the next rung). This is the property danos is really chasing:
|
||||||
**if a part of the OS breaks, isolate it, and re-initialise it — without rebooting.**
|
**if a part of the OS breaks, isolate it, and re-initialise it — without rebooting.**
|
||||||
A crashed driver gets restarted; a wedged service gets killed and brought back. It's
|
A crashed driver gets restarted; a wedged service gets killed and brought back. It's
|
||||||
the reason the [microkernel](vision.md) shape was chosen, and it's a *separate* goal
|
the reason the [microkernel](vision.md) shape was chosen, and it's a *separate* goal
|
||||||
|
|||||||
@@ -94,6 +94,15 @@ pub fn send(h: Handle, message: []const u8) bool {
|
|||||||
/// GSI. See `isNotification`.
|
/// GSI. See `isNotification`.
|
||||||
pub const notify_badge_bit: u64 = abi.notify_badge_bit;
|
pub const notify_badge_bit: u64 = abi.notify_badge_bit;
|
||||||
|
|
||||||
|
/// Set alongside `notify_badge_bit` when the notification is a **signal** — the
|
||||||
|
/// lifecycle vocabulary of docs/process-lifecycle.md, delivered to the endpoint
|
||||||
|
/// nominated with `process.bindSignals`. Decode with `process.signalsFrom`.
|
||||||
|
pub const notify_signal_bit: u64 = abi.notify_signal_bit;
|
||||||
|
|
||||||
|
/// Set alongside `notify_badge_bit` when the notification is a **one-shot timer**
|
||||||
|
/// landing (`system.timerOnce`).
|
||||||
|
pub const notify_timer_bit: u64 = abi.notify_timer_bit;
|
||||||
|
|
||||||
/// Set alongside `notify_badge_bit` when the notification is a **child-exit
|
/// Set alongside `notify_badge_bit` when the notification is a **child-exit
|
||||||
/// notice** — a process this one spawned (with an exit endpoint) has ended —
|
/// notice** — a process this one spawned (with an exit endpoint) has ended —
|
||||||
/// rather than a device interrupt. The low bits carry the child's process id.
|
/// rather than a device interrupt. The low bits carry the child's process id.
|
||||||
@@ -133,6 +142,17 @@ pub const Received = struct {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// The task id of whoever posted a buffered message, meaningful only when
|
/// The task id of whoever posted a buffered message, meaningful only when
|
||||||
|
/// Whether this arrival is a signal notification — decode the set with
|
||||||
|
/// `process.signalsFrom(badge)`.
|
||||||
|
pub fn isSignal(self: Received) bool {
|
||||||
|
return self.isNotification() and self.badge & notify_signal_bit != 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether this arrival is a one-shot timer landing (`system.timerOnce`).
|
||||||
|
pub fn isTimer(self: Received) bool {
|
||||||
|
return self.isNotification() and self.badge & notify_timer_bit != 0;
|
||||||
|
}
|
||||||
|
|
||||||
/// `isMessage`. (The badge's low bits, with the three high marker bits masked off.)
|
/// `isMessage`. (The badge's low bits, with the three high marker bits masked off.)
|
||||||
pub fn senderTaskId(self: Received) u32 {
|
pub fn senderTaskId(self: Received) u32 {
|
||||||
return @intCast(self.badge & ~(notify_badge_bit | notify_exit_bit | notify_message_bit));
|
return @intCast(self.badge & ~(notify_badge_bit | notify_exit_bit | notify_message_bit));
|
||||||
|
|||||||
@@ -1,8 +1,15 @@
|
|||||||
//! Process-level runtime types: what a user program receives at entry. Mirrors
|
//! Process-level runtime types: what a user program receives at entry (`Init`,
|
||||||
|
//! the argv contract) and the process end of the lifecycle
|
||||||
|
//! (docs/process-lifecycle.md) — today the exit reason a supervisor reads to
|
||||||
|
//! decide restart; signals and the stop sequence land here with M17.4. Mirrors
|
||||||
//! the spirit of `std.process.Init.Minimal` in danos terms — std's `Args` holds
|
//! the spirit of `std.process.Init.Minimal` in danos terms — std's `Args` holds
|
||||||
//! no data on freestanding targets, so the type is danos's own.
|
//! no data on freestanding targets, so the type is danos's own.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
|
const abi = @import("abi");
|
||||||
|
const sc = @import("system-call.zig");
|
||||||
|
const ipc = @import("ipc.zig");
|
||||||
|
const system = @import("system.zig");
|
||||||
|
|
||||||
/// Everything a program receives at entry. Passed to
|
/// Everything a program receives at entry. Passed to
|
||||||
/// `pub fn main(init: runtime.process.Init)`; programs that need nothing keep
|
/// `pub fn main(init: runtime.process.Init)`; programs that need nothing keep
|
||||||
@@ -45,3 +52,86 @@ pub const Arguments = struct {
|
|||||||
}
|
}
|
||||||
};
|
};
|
||||||
};
|
};
|
||||||
|
|
||||||
|
/// How a process ended — what a supervisor's restart policy reads: a clean exit
|
||||||
|
/// meant to stop, a fault wants a restart with backoff, killed means the
|
||||||
|
/// supervisor did it itself (docs/process-lifecycle.md).
|
||||||
|
pub const ExitReason = abi.ExitReason;
|
||||||
|
|
||||||
|
/// How dead child `id` ended. Ask after the exit notification arrives — the
|
||||||
|
/// kernel records the reason before it posts the notification, so this never
|
||||||
|
/// races it. Returns null for an id that never lived, is still alive, was
|
||||||
|
/// evicted from the kernel's bounded record, or is not this process's child
|
||||||
|
/// (the same authority gate as `kill`).
|
||||||
|
pub fn exitReason(id: u32) ?ExitReason {
|
||||||
|
const r = sc.systemCall1(.process_exit_reason, id);
|
||||||
|
if (r > ~@as(usize, 0) - 4095) return null; // a wrapped -errno
|
||||||
|
return @enumFromInt(r);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The signal vocabulary (docs/process-lifecycle.md): POSIX's concepts, danos's
|
||||||
|
/// names, message delivery. A signal is a one-way coalescing statement — never a
|
||||||
|
/// question (liveness is the zero-length ping call) and never kill (that is
|
||||||
|
/// `system.kill`, unhandleable by definition).
|
||||||
|
pub const Signal = abi.Signal;
|
||||||
|
|
||||||
|
/// The coalesced set of signals one notification delivered: two pending
|
||||||
|
/// terminates arrive as one. Decode a received badge with `signalsFrom`.
|
||||||
|
pub const SignalSet = struct {
|
||||||
|
pending: u32,
|
||||||
|
|
||||||
|
pub fn has(set: SignalSet, signal: Signal) bool {
|
||||||
|
return set.pending & (@as(u32, 1) << @intFromEnum(signal)) != 0;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Nominate `endpoint` as this process's signal endpoint. Signals posted while
|
||||||
|
/// unbound have pended; they are delivered immediately on bind, coalesced.
|
||||||
|
pub fn bindSignals(endpoint: usize) bool {
|
||||||
|
return sc.systemCall1(.signal_bind, endpoint) == 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Decode a received badge into the signals it delivered, or null if it is not
|
||||||
|
/// a signal notification.
|
||||||
|
pub fn signalsFrom(badge: u64) ?SignalSet {
|
||||||
|
if (badge & abi.notify_badge_bit == 0 or badge & abi.notify_signal_bit == 0) return null;
|
||||||
|
return .{ .pending = @truncate(badge & ~(abi.notify_badge_bit | abi.notify_signal_bit)) };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Post `signal` to child `id` (or to yourself). Supervisor-gated, like kill;
|
||||||
|
/// non-blocking, always — a statement, not a conversation.
|
||||||
|
pub fn sendSignal(id: u32, signal: Signal) bool {
|
||||||
|
return sc.systemCall2(.process_signal, id, @intFromEnum(signal)) == 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The standard stop sequence (docs/process-lifecycle.md): terminate, wait up to
|
||||||
|
/// `deadline_ms` for the exit notification on `exit_endpoint` (the endpoint the
|
||||||
|
/// child was spawned with), then kill. Any *other* notifications arriving on
|
||||||
|
/// that endpoint while stopping are consumed and dropped — a supervisor with
|
||||||
|
/// concurrent traffic implements the same sequence inside its own event loop
|
||||||
|
/// (arm `system.timerOnce`, keep serving) instead of calling this.
|
||||||
|
pub fn stop(id: u32, deadline_ms: u64, exit_endpoint: usize) void {
|
||||||
|
_ = sendSignal(id, .terminate);
|
||||||
|
_ = system.timerOnce(exit_endpoint, deadline_ms);
|
||||||
|
var receive: [8]u8 = undefined;
|
||||||
|
while (true) {
|
||||||
|
const got = ipc.replyWait(exit_endpoint, &.{}, &receive, null);
|
||||||
|
if (got.isChildExit() and got.childProcessId() == id) return;
|
||||||
|
if (got.isTimer()) break; // the deadline passed first — escalate
|
||||||
|
}
|
||||||
|
_ = system.kill(id);
|
||||||
|
while (true) {
|
||||||
|
const got = ipc.replyWait(exit_endpoint, &.{}, &receive, null);
|
||||||
|
if (got.isChildExit() and got.childProcessId() == id) return;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Subscribe `endpoint` to published exit events: every process death posts an
|
||||||
|
/// asynchronous notification with the same badge encoding as a supervisor's exit
|
||||||
|
/// notice (decode with `ipc.Received.isChildExit`/`childProcessId`). For stateful
|
||||||
|
/// services: release what the dead client held — file handles, subscriptions —
|
||||||
|
/// because a service must never depend on clients cleaning up after themselves
|
||||||
|
/// (docs/process-lifecycle.md). Ungated, like `system.processes`.
|
||||||
|
pub fn subscribeExits(endpoint: usize) bool {
|
||||||
|
return sc.systemCall1(.process_subscribe, endpoint) == 0;
|
||||||
|
}
|
||||||
|
|||||||
@@ -17,6 +17,9 @@ pub const ipc = @import("ipc.zig");
|
|||||||
pub const start = @import("start.zig");
|
pub const start = @import("start.zig");
|
||||||
/// The VFS wire protocol (shared with the VFS server).
|
/// The VFS wire protocol (shared with the VFS server).
|
||||||
pub const vfs_protocol = @import("vfs-protocol");
|
pub const vfs_protocol = @import("vfs-protocol");
|
||||||
|
|
||||||
|
/// The device-manager protocol: hello + tree reports (docs/device-manager.md).
|
||||||
|
pub const device_manager_protocol = @import("device-manager-protocol");
|
||||||
/// Keyboard-event listening (subscribe/next) and broadcasting (publish), over the input
|
/// Keyboard-event listening (subscribe/next) and broadcasting (publish), over the input
|
||||||
/// service. See library/runtime/input.zig and system/services/input/.
|
/// service. See library/runtime/input.zig and system/services/input/.
|
||||||
pub const input = @import("input.zig");
|
pub const input = @import("input.zig");
|
||||||
@@ -35,5 +38,9 @@ pub const panic = start.panic;
|
|||||||
/// Process entry types: the `Init` handed to `main`, and its `Arguments`.
|
/// Process entry types: the `Init` handed to `main`, and its `Arguments`.
|
||||||
pub const process = @import("process.zig");
|
pub const process = @import("process.zig");
|
||||||
|
|
||||||
|
/// The service harness: one replyWait loop folding requests, signals, and
|
||||||
|
/// notifications into callbacks (docs/process-lifecycle.md).
|
||||||
|
pub const service = @import("service.zig");
|
||||||
|
|
||||||
/// The heap as a `std.mem.Allocator`, for Zig `std` containers in user code.
|
/// The heap as a `std.mem.Allocator`, for Zig `std` containers in user code.
|
||||||
pub const allocator = heap.allocator;
|
pub const allocator = heap.allocator;
|
||||||
|
|||||||
@@ -0,0 +1,83 @@
|
|||||||
|
//! The service harness (docs/process-lifecycle.md): one replyWait loop that
|
||||||
|
//! folds protocol requests, signals, and subscribed notifications into
|
||||||
|
//! callbacks — so the lifecycle contract ("answers ping, exits on terminate")
|
||||||
|
//! is satisfied by construction and a service author writes domain logic only.
|
||||||
|
//! Nothing is asynchronous inside the process: a callback runs at a point the
|
||||||
|
//! loop chose, never on a hijacked stack — the whole reason signals are
|
||||||
|
//! messages.
|
||||||
|
//!
|
||||||
|
//! The liveness probe: a **zero-length request is the universal ping**, answered
|
||||||
|
//! with a zero-length reply by the harness itself. No protocol's requests start
|
||||||
|
//! at length zero, so the encoding cannot collide, and there is nothing for a
|
||||||
|
//! service author to implement — a wedged service simply fails to answer, which
|
||||||
|
//! is the diagnosis (see docs/ipc.md).
|
||||||
|
|
||||||
|
const abi = @import("abi");
|
||||||
|
const ipc = @import("ipc.zig");
|
||||||
|
const process = @import("process.zig");
|
||||||
|
|
||||||
|
pub const Callbacks = struct {
|
||||||
|
/// Called once with the service's endpoint before the loop starts — the
|
||||||
|
/// place to subscribe to exit events, bind IRQs, or announce readiness.
|
||||||
|
/// Return false to abort startup (the process exits).
|
||||||
|
init: ?*const fn (endpoint: ipc.Handle) bool = null,
|
||||||
|
/// One protocol request from `sender` (a task id): write the reply into
|
||||||
|
/// `reply`, return its length. `capability` is the handle the request
|
||||||
|
/// carried, if any (M13 cap passing — how a subscriber hands over its
|
||||||
|
/// endpoint). The zero-length ping never reaches this.
|
||||||
|
on_message: *const fn (message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize,
|
||||||
|
/// A notification that is not a signal — a subscribed exit event, a bound
|
||||||
|
/// IRQ, a timer landing. The raw badge; decode with the ipc helpers.
|
||||||
|
on_notification: ?*const fn (badge: u64) void = null,
|
||||||
|
/// The reload signal. Default: ignored.
|
||||||
|
on_reload: ?*const fn () void = null,
|
||||||
|
/// The terminate signal, called before the loop returns. The clean exit is
|
||||||
|
/// the return itself — never put *necessary* work here (iron rule 1: a kill
|
||||||
|
/// arrives with no warning; this is for graceful extras only).
|
||||||
|
on_terminate: ?*const fn () void = null,
|
||||||
|
/// Publish the endpoint under a well-known service id at startup.
|
||||||
|
service: ?abi.ServiceId = null,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Run the service: create and (optionally) register the endpoint, bind signals
|
||||||
|
/// to it, call `init`, then serve until `terminate` arrives — at which point the
|
||||||
|
/// loop returns and main's return is the clean exit the supervisor reads as
|
||||||
|
/// `ExitReason.exited`. `maximum_message` sizes the receive and reply buffers
|
||||||
|
/// (a service passes its protocol's message maximum).
|
||||||
|
pub fn run(comptime maximum_message: usize, callbacks: Callbacks) void {
|
||||||
|
const endpoint = ipc.createIpcEndpoint() orelse return;
|
||||||
|
if (callbacks.service) |id| {
|
||||||
|
if (!ipc.register(id, endpoint)) return;
|
||||||
|
}
|
||||||
|
_ = process.bindSignals(endpoint);
|
||||||
|
if (callbacks.init) |initialise| {
|
||||||
|
if (!initialise(endpoint)) return;
|
||||||
|
}
|
||||||
|
|
||||||
|
var reply_buffer: [maximum_message]u8 = undefined;
|
||||||
|
var reply_len: usize = 0;
|
||||||
|
var receive: [maximum_message]u8 = undefined;
|
||||||
|
while (true) {
|
||||||
|
const got = ipc.replyWait(endpoint, reply_buffer[0..reply_len], &receive, null);
|
||||||
|
if (got.isNotification()) {
|
||||||
|
reply_len = 0; // nothing owed for a notification
|
||||||
|
if (process.signalsFrom(got.badge)) |signals| {
|
||||||
|
if (signals.has(.reload)) {
|
||||||
|
if (callbacks.on_reload) |onReload| onReload();
|
||||||
|
}
|
||||||
|
if (signals.has(.terminate)) {
|
||||||
|
if (callbacks.on_terminate) |onTerminate| onTerminate();
|
||||||
|
return; // the loop's return IS the clean exit
|
||||||
|
}
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if (callbacks.on_notification) |onNotification| onNotification(got.badge);
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if (got.len == 0) {
|
||||||
|
reply_len = 0; // the universal ping: a zero-length reply, from the harness
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
reply_len = callbacks.on_message(receive[0..got.len], &reply_buffer, got.senderTaskId(), got.cap);
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -32,6 +32,15 @@ pub fn sleep(ms: usize) void {
|
|||||||
_ = sc.systemCall1(.sleep, ms);
|
_ = sc.systemCall1(.sleep, ms);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Arm a one-shot timer: after `ms` milliseconds the kernel posts a timer
|
||||||
|
/// notification (`ipc.Received.isTimer`) to `endpoint`. The timed wait of
|
||||||
|
/// docs/process-lifecycle.md — a service arms a deadline and keeps serving,
|
||||||
|
/// instead of blocking in sleep; what stop-sequence escalation, hello deadlines,
|
||||||
|
/// and restart backoff are built from.
|
||||||
|
pub fn timerOnce(endpoint: usize, ms: u64) bool {
|
||||||
|
return sc.systemCall2(.timer_bind, endpoint, ms) == 0;
|
||||||
|
}
|
||||||
|
|
||||||
/// Monotonic nanoseconds since boot — a time source for timeouts and short delays. It
|
/// Monotonic nanoseconds since boot — a time source for timeouts and short delays. It
|
||||||
/// only ever moves forward. This is *not* wall-clock time (no date, no timezone — that
|
/// only ever moves forward. This is *not* wall-clock time (no date, no timezone — that
|
||||||
/// is a user-space service layered on top). Deadline pattern for a bounded poll loop:
|
/// is a user-space service layered on top). Deadline pattern for a bounded poll loop:
|
||||||
|
|||||||
@@ -53,9 +53,31 @@ pub const SystemCall = enum(u64) {
|
|||||||
process_enumerate = 24, // process_enumerate(buffer, maximum) -> total: snapshot the task table
|
process_enumerate = 24, // process_enumerate(buffer, maximum) -> total: snapshot the task table
|
||||||
process_kill = 25, // process_kill(id) -> 0/-errno: end a process this process spawned
|
process_kill = 25, // process_kill(id) -> 0/-errno: end a process this process spawned
|
||||||
ipc_send = 26, // ipc_send(handle, message_ptr, message_len) -> 0/-errno: post a payload to an endpoint's async queue without blocking
|
ipc_send = 26, // ipc_send(handle, message_ptr, message_len) -> 0/-errno: post a payload to an endpoint's async queue without blocking
|
||||||
|
process_exit_reason = 27, // process_exit_reason(id) -> ExitReason/-errno: how a dead child ended (its supervisor only)
|
||||||
|
process_subscribe = 28, // process_subscribe(endpoint) -> 0/-errno: subscribe to published exit events — every death posts a notification
|
||||||
|
signal_bind = 29, // signal_bind(endpoint) -> 0/-errno: nominate the endpoint this process's signals arrive on
|
||||||
|
process_signal = 30, // process_signal(id, signal) -> 0/-errno: post a signal to a child (or to yourself)
|
||||||
|
timer_bind = 31, // timer_bind(endpoint, ms) -> 0/-errno: one-shot timer — posts a notification when ms elapse
|
||||||
_,
|
_,
|
||||||
};
|
};
|
||||||
|
|
||||||
|
/// How a process ended — recorded by the kernel at death, queried by the
|
||||||
|
/// supervisor with `process_exit_reason`, and the input to its restart decision
|
||||||
|
/// (docs/process-lifecycle.md): a clean exit meant to stop, a fault wants a
|
||||||
|
/// restart with backoff, killed means the supervisor did it itself. The faults
|
||||||
|
/// mirror the CPU exceptions a ring-3 process can die of; they are exit reasons,
|
||||||
|
/// never delivered to the faulting process (recovery is restart, not a handler).
|
||||||
|
pub const ExitReason = enum(u8) {
|
||||||
|
exited = 0, // returned from main / called exit
|
||||||
|
aborted = 1, // deliberate self-termination (reserved: no abort path yet)
|
||||||
|
segmentation_fault = 2, // page fault
|
||||||
|
illegal_instruction = 3, // invalid opcode
|
||||||
|
arithmetic_fault = 4, // divide error, x87 or SIMD fault
|
||||||
|
protection_fault = 5, // general protection fault
|
||||||
|
fault = 6, // any other CPU exception
|
||||||
|
killed = 7, // process_kill
|
||||||
|
};
|
||||||
|
|
||||||
/// The x86 MSI message address base (`0xFEE0_0000`): a device raises an MSI by writing
|
/// The x86 MSI message address base (`0xFEE0_0000`): a device raises an MSI by writing
|
||||||
/// `data` to this address, which the Local APIC turns into an interrupt at the vector
|
/// `data` to this address, which the Local APIC turns into an interrupt at the vector
|
||||||
/// in `data`. The kernel returns the concrete (address, data) from `msi_bind`; this is
|
/// in `data`. The kernel returns the concrete (address, data) from `msi_bind`; this is
|
||||||
@@ -93,6 +115,35 @@ pub const notify_exit_bit: u64 = 1 << 62;
|
|||||||
/// broadcasts where a rendezvous is the wrong shape (the input service is the first user).
|
/// broadcasts where a rendezvous is the wrong shape (the input service is the first user).
|
||||||
pub const notify_message_bit: u64 = 1 << 61;
|
pub const notify_message_bit: u64 = 1 << 61;
|
||||||
|
|
||||||
|
/// Set (alongside `notify_badge_bit`) in the badge of a **signal notification** —
|
||||||
|
/// the process-lifecycle vocabulary of docs/process-lifecycle.md, delivered to the
|
||||||
|
/// endpoint the process nominated with `signal_bind`. The low bits carry the
|
||||||
|
/// coalesced pending mask (bit positions = `Signal` values): signals are
|
||||||
|
/// statements, not questions, and two pending terminates are one terminate.
|
||||||
|
pub const notify_signal_bit: u64 = 1 << 60;
|
||||||
|
|
||||||
|
/// Set (alongside `notify_badge_bit`) in the badge of a **timer notification** —
|
||||||
|
/// a one-shot `timer_bind` deadline landing. No payload bits: what to do when the
|
||||||
|
/// deadline fires is whatever the receiver armed it for (a stop-sequence
|
||||||
|
/// escalation, a restart backoff, an alarm).
|
||||||
|
pub const notify_timer_bit: u64 = 1 << 59;
|
||||||
|
|
||||||
|
/// The signal vocabulary (docs/process-lifecycle.md): POSIX's concepts, danos's
|
||||||
|
/// names, message delivery. The value is the bit position in the pending mask — a
|
||||||
|
/// private kernel/runtime detail, free to change while they ship together. Kill
|
||||||
|
/// is not here (it is `process_kill`, unhandleable by definition); faults are not
|
||||||
|
/// here (they are `ExitReason`s — recovery is restart, not a handler); liveness is
|
||||||
|
/// not here (a question, asked as the zero-length ping call, not a statement).
|
||||||
|
pub const Signal = enum(u5) {
|
||||||
|
terminate = 0, // finish up and exit (the polite half of the stop sequence)
|
||||||
|
reload = 1, // re-read configuration / re-scan
|
||||||
|
interrupt = 2, // interactive interrupt (no sender until a console exists)
|
||||||
|
quit = 3, // as interrupt, by convention more final
|
||||||
|
alarm = 4, // a timer the process armed for itself (unbuilt: no consumer yet)
|
||||||
|
user_1 = 5, // service-defined
|
||||||
|
user_2 = 6, // service-defined
|
||||||
|
};
|
||||||
|
|
||||||
/// Capacity of `ProcessDescriptor.name` — matches the longest name `system_spawn`
|
/// Capacity of `ProcessDescriptor.name` — matches the longest name `system_spawn`
|
||||||
/// accepts, so a process's recorded name (its argv[0]) is never truncated.
|
/// accepts, so a process's recorded name (its argv[0]) is never truncated.
|
||||||
pub const maximum_process_name = 64;
|
pub const maximum_process_name = 64;
|
||||||
@@ -125,6 +176,7 @@ pub const ServiceId = enum(u32) {
|
|||||||
vfs = 1,
|
vfs = 1,
|
||||||
input = 2,
|
input = 2,
|
||||||
ps2_bus = 3, // the 8042 owner; child device drivers attach here for raw bytes
|
ps2_bus = 3, // the 8042 owner; child device drivers attach here for raw bytes
|
||||||
|
device_manager = 4, // the tree, the matcher, the supervisor (docs/device-manager.md)
|
||||||
_,
|
_,
|
||||||
};
|
};
|
||||||
|
|
||||||
|
|||||||
+126
-134
@@ -41,6 +41,9 @@ pub const RegisterAccess = struct {
|
|||||||
/// Everything the power subsystem needs, extracted from the FADT and the AML
|
/// Everything the power subsystem needs, extracted from the FADT and the AML
|
||||||
/// sleep packages during discovery. Populated by `discover`, read by `power`.
|
/// sleep packages during discovery. Populated by `discover`, read by `power`.
|
||||||
pub const PowerInformation = struct {
|
pub const PowerInformation = struct {
|
||||||
|
/// The System Control Interrupt's GSI (FADT SCI_INT) — the line ACPI events
|
||||||
|
/// (power button, GPEs) arrive on. Published to the acpi service for M21.
|
||||||
|
sci_interrupt: u16 = 0,
|
||||||
/// The SMM command port and the value that switches the platform into ACPI mode.
|
/// The SMM command port and the value that switches the platform into ACPI mode.
|
||||||
smi_cmd: u16 = 0,
|
smi_cmd: u16 = 0,
|
||||||
acpi_enable: u8 = 0,
|
acpi_enable: u8 = 0,
|
||||||
@@ -360,32 +363,14 @@ const Hpet = extern struct {
|
|||||||
page_protection: u8,
|
page_protection: u8,
|
||||||
};
|
};
|
||||||
|
|
||||||
// --- PCI configuration-space header (first 64 bytes, common fields) ---------
|
|
||||||
|
|
||||||
const PciHeader = extern struct {
|
|
||||||
vendor_id: u16 align(1),
|
|
||||||
device_id: u16 align(1),
|
|
||||||
command: u16 align(1),
|
|
||||||
status: u16 align(1),
|
|
||||||
revision_id: u8,
|
|
||||||
prog_if: u8,
|
|
||||||
subclass: u8,
|
|
||||||
class_code: u8,
|
|
||||||
cache_line_size: u8,
|
|
||||||
latency_timer: u8,
|
|
||||||
/// bit 7 set => multi-function device.
|
|
||||||
header_type: u8,
|
|
||||||
bist: u8,
|
|
||||||
// 0x10 onward (BARs, etc.) depends on header_type; read separately.
|
|
||||||
};
|
|
||||||
|
|
||||||
// --- Entry point ------------------------------------------------------------
|
// --- Entry point ------------------------------------------------------------
|
||||||
|
|
||||||
/// Discover hardware from the ACPI tables rooted at `rsdp_physical` and populate
|
/// Discover hardware from the ACPI tables rooted at `rsdp_physical` and populate
|
||||||
/// `device_tree`. `hal` provides MMIO mapping (for PCIe ECAM) and port I/O. Also parses the
|
/// `device_tree`. `hal` provides MMIO mapping (for PCIe ECAM) and port I/O. Also parses the
|
||||||
/// FADT and the AML sleep-state (`_Sx`) packages into `power_information` for the power service.
|
/// FADT and the AML sleep-state (`_Sx`) packages into `power_information` for the power service.
|
||||||
pub fn discover(rsdp_physical: u64, device_tree: *DeviceTree, hal: Hal) !void {
|
pub fn discover(rsdp_physical: u64, memory_regions: []const boot_handoff.MemoryRegion, device_tree: *DeviceTree, hal: Hal) !void {
|
||||||
if (rsdp_physical == 0) return error.NoRsdp;
|
if (rsdp_physical == 0) return error.NoRsdp;
|
||||||
|
boot_memory_regions = memory_regions;
|
||||||
|
|
||||||
// Start clean so a re-run doesn't accumulate stale state.
|
// Start clean so a re-run doesn't accumulate stale state.
|
||||||
power_information = .{};
|
power_information = .{};
|
||||||
@@ -420,12 +405,57 @@ pub fn discover(rsdp_physical: u64, device_tree: *DeviceTree, hal: Hal) !void {
|
|||||||
aml_stats = .{ .nodes = namespace.?.nodeCount(), .consumed = pr.consumed, .total = pr.total };
|
aml_stats = .{ .nodes = namespace.?.nodeCount(), .consumed = pr.consumed, .total = pr.total };
|
||||||
power_information.s5 = aml.sleepState(&namespace.?, 5);
|
power_information.s5 = aml.sleepState(&namespace.?, 5);
|
||||||
power_information.s3 = aml.sleepState(&namespace.?, 3);
|
power_information.s3 = aml.sleepState(&namespace.?, 3);
|
||||||
// Fold the namespace's Device objects into the generic tree.
|
// The namespace's Device objects are no longer folded into the kernel
|
||||||
wireAcpiDevices(device_tree, &namespace.?, hal) catch {};
|
// tree (M20.3): the ring-3 acpi service claims the acpi-tables node
|
||||||
|
// (published below), re-parses the same blobs, and registers + reports
|
||||||
|
// the _HID devices itself. The kernel keeps the namespace only for the
|
||||||
|
// \_S5 sleep type above. The device-building helpers below
|
||||||
|
// (wireAcpiDevices and friends) are retained but unreferenced — a
|
||||||
|
// focused dead-code sweep follows the migration.
|
||||||
} else |_| {
|
} else |_| {
|
||||||
// AML parse failed (e.g. out of memory); power stays best-effort with
|
// AML parse failed (e.g. out of memory); power stays best-effort with
|
||||||
// whatever the FADT alone provided.
|
// whatever the FADT alone provided.
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Publish the acpi-tables node (docs/m19-m20-plan.md M20): the AML blobs as
|
||||||
|
// memory resources for the acpi service to map and parse in ring 3, a broad
|
||||||
|
// io_port grant for the OperationRegion access its interpreter needs, and
|
||||||
|
// the SCI for the events track (M21). Exactly one node, one trusted
|
||||||
|
// claimant. Kept even when the kernel-side device building (above) retires
|
||||||
|
// in M20.3 — the kernel still owns the *static* tables and \_S5.
|
||||||
|
publishAcpiTablesNode(device_tree) catch {};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Build the acpi-tables node (see the call site in discover). Best-effort: a
|
||||||
|
/// failure here leaves the kernel-seeded tree working, only the ring-3 service
|
||||||
|
/// finds nothing to claim.
|
||||||
|
fn publishAcpiTablesNode(device_tree: *DeviceTree) !void {
|
||||||
|
const node = try device_tree.addChild(device_tree.root, .acpi_tables, "acpi-tables");
|
||||||
|
// One memory resource per AML block — page-aligned base down, length padded
|
||||||
|
// up to cover the bytecode, so mmio_map hands the service a pointer into it.
|
||||||
|
var i: usize = 0;
|
||||||
|
while (i < aml_block_count and i < device_model.maximum_resources - 2) : (i += 1) {
|
||||||
|
// mmio_map preserves the sub-page offset, so the service maps this and
|
||||||
|
// gets a pointer straight to the bytecode.
|
||||||
|
_ = node.addResource(.memory, aml_block_physical[i], aml_block_len[i]);
|
||||||
|
}
|
||||||
|
// The broad I/O grant: OperationRegions name whatever ports the firmware
|
||||||
|
// chose (EC, PM1, GPE, SMBus); which ports cannot be known before the AML
|
||||||
|
// that names them is parsed, so the grant is the whole space — the honest
|
||||||
|
// trust boundary of docs/m19-m20-plan.md decision 5.
|
||||||
|
_ = node.addResource(.io_port, 0, 1 << 16);
|
||||||
|
// A broad interrupt window: ACPI _CRS names legacy ISA IRQs (the PS/2 lines
|
||||||
|
// 1 and 12, the RTC, …), and the service registers those devices under this
|
||||||
|
// node, so it must own a superset. The range [0, 256) covers every GSI; the
|
||||||
|
// SCI (recorded first, len 1) stays distinct so M21 can pick it out.
|
||||||
|
if (power_information.sci_interrupt != 0) _ = node.addResource(.irq, power_information.sci_interrupt, 1);
|
||||||
|
_ = node.addResource(.irq, 0, 256);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The number of Device objects in the namespace built during discovery, or 0.
|
||||||
|
pub fn amlDeviceCount() usize {
|
||||||
|
if (namespace) |*ns| return aml.deviceCount(ns);
|
||||||
|
return 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Walk the RSDT (Entry = u32) or XSDT (Entry = u64): validate it, then dispatch
|
/// Walk the RSDT (Entry = u32) or XSDT (Entry = u64): validate it, then dispatch
|
||||||
@@ -451,7 +481,7 @@ fn handleTable(device_tree: *DeviceTree, hal: Hal, sdt_physical: u64) !void {
|
|||||||
if (std.mem.eql(u8, &sig, &APIC)) {
|
if (std.mem.eql(u8, &sig, &APIC)) {
|
||||||
try parseMadt(device_tree, header);
|
try parseMadt(device_tree, header);
|
||||||
} else if (std.mem.eql(u8, &sig, &MCFG)) {
|
} else if (std.mem.eql(u8, &sig, &MCFG)) {
|
||||||
try parseMcfg(device_tree, hal, header);
|
try parseMcfg(device_tree, header);
|
||||||
} else if (std.mem.eql(u8, &sig, &HPET)) {
|
} else if (std.mem.eql(u8, &sig, &HPET)) {
|
||||||
try parseHpet(device_tree, hal, header);
|
try parseHpet(device_tree, hal, header);
|
||||||
} else if (std.mem.eql(u8, &sig, &FACP)) {
|
} else if (std.mem.eql(u8, &sig, &FACP)) {
|
||||||
@@ -536,7 +566,7 @@ fn parseMadt(device_tree: *DeviceTree, header: *const SystemDescriptorTableHeade
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// MCFG -> a pci_host_bridge per ECAM segment, then a PCI enumeration underneath.
|
/// MCFG -> a pci_host_bridge per ECAM segment, then a PCI enumeration underneath.
|
||||||
fn parseMcfg(device_tree: *DeviceTree, hal: Hal, header: *const SystemDescriptorTableHeader) !void {
|
fn parseMcfg(device_tree: *DeviceTree, header: *const SystemDescriptorTableHeader) !void {
|
||||||
const total: usize = header.length;
|
const total: usize = header.length;
|
||||||
const base: [*]const u8 = @ptrCast(header);
|
const base: [*]const u8 = @ptrCast(header);
|
||||||
|
|
||||||
@@ -551,109 +581,85 @@ fn parseMcfg(device_tree: *DeviceTree, hal: Hal, header: *const SystemDescriptor
|
|||||||
// ECAM window: 1 MiB of configuration space per bus.
|
// ECAM window: 1 MiB of configuration space per bus.
|
||||||
_ = bridge.addResource(.memory, alloc.base_address, bus_count << 20);
|
_ = bridge.addResource(.memory, alloc.base_address, bus_count << 20);
|
||||||
_ = bridge.addResource(.bus_range, alloc.start_bus, bus_count);
|
_ = bridge.addResource(.bus_range, alloc.start_bus, bus_count);
|
||||||
|
addBridgeApertures(bridge);
|
||||||
|
// The bridge decodes the whole 16-bit I/O space toward its bus — the
|
||||||
|
// window functions' I/O BARs must register-contain within (M19.2).
|
||||||
|
_ = bridge.addResource(.io_port, 0, 1 << 16);
|
||||||
|
|
||||||
try enumeratePci(device_tree, bridge, hal, alloc.*);
|
// The function walk itself retired to ring 3 (M19.3): the pci-bus
|
||||||
|
// driver claims this bridge, repeats the scan through its ECAM grant,
|
||||||
|
// and device_registers what it finds — the kernel seeds only the
|
||||||
|
// bridge. The scan's equivalence was proven before the hand-off
|
||||||
|
// (pci-scan), and the walk's history is in git if archaeology calls.
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Brute-force scan the ECAM window's bus range for present PCI functions. No
|
/// The boot memory map, stored at discover() entry for the aperture derivation
|
||||||
/// bridge recursion yet: on the ECAM path the host bridge decodes every bus in
|
/// below (and, in M20, for the acpi-tables node's containment windows).
|
||||||
/// the window, so scanning the declared range finds everything QEMU exposes.
|
var boot_memory_regions: []const boot_handoff.MemoryRegion = &.{};
|
||||||
fn enumeratePci(
|
|
||||||
device_tree: *DeviceTree,
|
|
||||||
bridge: *device_model.Device,
|
|
||||||
hal: Hal,
|
|
||||||
alloc: McfgAllocation,
|
|
||||||
) !void {
|
|
||||||
var bus: u16 = alloc.start_bus;
|
|
||||||
while (bus <= alloc.end_bus) : (bus += 1) {
|
|
||||||
var device: u8 = 0;
|
|
||||||
while (device < 32) : (device += 1) {
|
|
||||||
const h0: *align(1) const PciHeader = @ptrCast(pciConfigurationPtr(alloc, hal, @intCast(bus), device, 0));
|
|
||||||
if (h0.vendor_id == 0xFFFF) continue; // no function 0 => slot empty
|
|
||||||
|
|
||||||
const funcs: u8 = if (h0.header_type & 0x80 != 0) 8 else 1;
|
/// The bridge's MMIO apertures, derived from the boot memory map's holes
|
||||||
var function: u8 = 0;
|
/// (docs/m19-m20-plan.md decision 2): registered PCI functions carry BAR
|
||||||
while (function < funcs) : (function += 1) {
|
/// resources, and `device_register` containment demands the bridge own windows
|
||||||
const configuration = pciConfigurationPtr(alloc, hal, @intCast(bus), device, function);
|
/// that cover them. Everything the firmware described is "not hole"; the low
|
||||||
const h: *align(1) const PciHeader = @ptrCast(configuration);
|
/// aperture runs from the end of the described space below 4 GiB up to the
|
||||||
if (h.vendor_id == 0xFFFF) continue;
|
/// I/O-APIC region, the high one from 4 GiB (or the end of RAM above it) to
|
||||||
|
/// the 46-bit line. Coarse, mechanical, and AML-free — available at boot no
|
||||||
var nb: [24]u8 = undefined;
|
/// matter what later moved to user space.
|
||||||
const nm = std.fmt.bufPrint(&nb, "{s}:{x:0>2}:{x:0>2}.{d}", .{
|
fn addBridgeApertures(bridge: *device_model.Device) void {
|
||||||
bridge.name(), bus, device, function,
|
// Below 4 GiB the described regions are sparse (RAM low, firmware flash
|
||||||
}) catch "pcidev";
|
// and tables high), so the holes are the *gaps between* them — a single
|
||||||
const node = try device_tree.addChild(bridge, .pci_device, nm);
|
// "after the last region" rule dies on OVMF's flash at the very top.
|
||||||
// Resource 0 is the function's own 4 KiB ECAM configuration space. A
|
// Sort-merge the described ranges, then keep the three largest gaps
|
||||||
// claimed PCI driver mmio_maps this to reach its command register,
|
// (resource slots are bounded at 8 per device; ECAM + bus range + 3 + the
|
||||||
// BARs, and — the point — its capability list (MSI/MSI-X, PCIe
|
// high aperture fits). Above 4 GiB one aperture runs from the end of the
|
||||||
// extended caps), without any new syscall. Physical address per the
|
// described space to the 46-bit line.
|
||||||
// ECAM formula (same as pciConfigurationPtr).
|
const Range = struct { base: u64, end: u64 };
|
||||||
const config_physical = alloc.base_address +
|
var below: [64]Range = undefined;
|
||||||
(@as(u64, @as(u8, @intCast(bus)) - alloc.start_bus) << 20) +
|
var below_count: usize = 0;
|
||||||
(@as(u64, device) << 15) + (@as(u64, function) << 12);
|
var high_end: u64 = 1 << 32;
|
||||||
_ = node.addResource(.memory, config_physical, abi.page_size);
|
for (boot_memory_regions) |region| {
|
||||||
node.ids.pci_vendor = h.vendor_id;
|
const end = region.base + region.pages * 4096;
|
||||||
node.ids.pci_device = h.device_id;
|
// Above 4 GiB only *usable RAM* blocks the aperture: OVMF describes
|
||||||
node.ids.pci_class = (@as(u24, h.class_code) << 16) |
|
// its own 64-bit PCI window as a reserved region and then programs
|
||||||
(@as(u24, h.subclass) << 8) | h.prog_if;
|
// BARs inside it — honoring reserved there would exclude the very
|
||||||
node.ids.pci_bdf = (@as(u16, @intCast(bus)) << 8) | (@as(u16, device) << 3) | function;
|
// space BARs live in. Below 4 GiB every described region blocks (the
|
||||||
|
// kernel image, the tables, the ramdisk all live there). Bring-up
|
||||||
// BARs only exist in header type 0 (normal devices), not bridges.
|
// trust: only the bridge's claimant can register into the aperture.
|
||||||
if (h.header_type & 0x7F == 0) addBars(node, configuration);
|
if (region.kind == .usable and end > high_end) high_end = end;
|
||||||
|
if (region.base >= (1 << 32) or below_count == below.len) continue;
|
||||||
|
below[below_count] = .{ .base = region.base, .end = @min(end, 1 << 32) };
|
||||||
|
below_count += 1;
|
||||||
|
}
|
||||||
|
// Insertion sort by base (the map is small and this runs once at boot).
|
||||||
|
for (1..below_count) |i| {
|
||||||
|
const key = below[i];
|
||||||
|
var j = i;
|
||||||
|
while (j > 0 and below[j - 1].base > key.base) : (j -= 1) below[j] = below[j - 1];
|
||||||
|
below[j] = key;
|
||||||
|
}
|
||||||
|
// Walk the sorted ranges, collecting inter-region gaps of at least 1 MiB.
|
||||||
|
var gaps: [3]Range = .{Range{ .base = 0, .end = 0 }} ** 3;
|
||||||
|
var cursor: u64 = 0;
|
||||||
|
var index: usize = 0;
|
||||||
|
while (index <= below_count) : (index += 1) {
|
||||||
|
const gap_end = if (index == below_count) (1 << 32) else below[index].base;
|
||||||
|
if (gap_end > cursor and gap_end - cursor >= (1 << 20)) {
|
||||||
|
// Keep the three largest, replacing the smallest kept so far.
|
||||||
|
var smallest: usize = 0;
|
||||||
|
for (gaps, 0..) |gap, gi| {
|
||||||
|
if (gap.end - gap.base < gaps[smallest].end - gaps[smallest].base) smallest = gi;
|
||||||
|
}
|
||||||
|
if (gap_end - cursor > gaps[smallest].end - gaps[smallest].base) {
|
||||||
|
gaps[smallest] = .{ .base = cursor, .end = gap_end };
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
if (index < below_count and below[index].end > cursor) cursor = below[index].end;
|
||||||
}
|
}
|
||||||
|
for (gaps) |gap| {
|
||||||
|
if (gap.end > gap.base) _ = bridge.addResource(.memory, gap.base, gap.end - gap.base);
|
||||||
}
|
}
|
||||||
|
_ = bridge.addResource(.memory, high_end, (@as(u64, 1) << 46) - high_end);
|
||||||
/// Record and size the memory/IO windows named by a device's Base Address
|
|
||||||
/// Registers. Sizing is the standard probe: disable decode, write all-ones, read
|
|
||||||
/// back the writable (address) bits, restore. `size = ~mask + 1`.
|
|
||||||
fn addBars(node: *device_model.Device, configuration: [*]align(1) u8) void {
|
|
||||||
// Stop the device decoding its BARs while we transiently write all-ones.
|
|
||||||
const command = rd(u16, configuration, 0x04);
|
|
||||||
wr(u16, configuration, 0x04, command & ~@as(u16, 0b11));
|
|
||||||
|
|
||||||
var i: usize = 0;
|
|
||||||
while (i < 6) : (i += 1) {
|
|
||||||
const off = 0x10 + i * 4;
|
|
||||||
const orig = rd(u32, configuration, off);
|
|
||||||
if (orig == 0) continue;
|
|
||||||
|
|
||||||
if (orig & 1 != 0) {
|
|
||||||
// I/O-space BAR (16-bit address space on x86).
|
|
||||||
wr(u32, configuration, off, 0xFFFF_FFFF);
|
|
||||||
const readback = rd(u32, configuration, off);
|
|
||||||
wr(u32, configuration, off, orig);
|
|
||||||
const mask = readback & 0xFFFF_FFFC;
|
|
||||||
const size: u32 = if (mask == 0) 0 else (~mask +% 1) & 0xFFFF;
|
|
||||||
_ = node.addResource(.io_port, orig & 0xFFFF_FFFC, size);
|
|
||||||
} else if ((orig >> 1) & 0x3 == 2) {
|
|
||||||
// 64-bit memory BAR: this BAR pair spans two configuration slots.
|
|
||||||
const orig_hi = rd(u32, configuration, off + 4);
|
|
||||||
wr(u32, configuration, off, 0xFFFF_FFFF);
|
|
||||||
wr(u32, configuration, off + 4, 0xFFFF_FFFF);
|
|
||||||
const lo = rd(u32, configuration, off);
|
|
||||||
const hi = rd(u32, configuration, off + 4);
|
|
||||||
wr(u32, configuration, off, orig);
|
|
||||||
wr(u32, configuration, off + 4, orig_hi);
|
|
||||||
const readback = (@as(u64, hi) << 32) | (lo & 0xFFFF_FFF0);
|
|
||||||
const size: u64 = if (readback == 0) 0 else ~readback +% 1;
|
|
||||||
const address = (@as(u64, orig_hi) << 32) | (orig & 0xFFFF_FFF0);
|
|
||||||
_ = node.addResource(.memory, address, size);
|
|
||||||
i += 1; // consumed the high half
|
|
||||||
} else {
|
|
||||||
// 32-bit memory BAR.
|
|
||||||
wr(u32, configuration, off, 0xFFFF_FFFF);
|
|
||||||
const readback = rd(u32, configuration, off);
|
|
||||||
wr(u32, configuration, off, orig);
|
|
||||||
const mask = readback & 0xFFFF_FFF0;
|
|
||||||
const size: u32 = if (mask == 0) 0 else ~mask +% 1;
|
|
||||||
_ = node.addResource(.memory, orig & 0xFFFF_FFF0, size);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
wr(u16, configuration, 0x04, command); // restore decode
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// HPET -> a timer node with its register block as an MMIO resource, plus the GSI
|
/// HPET -> a timer node with its register block as an MMIO resource, plus the GSI
|
||||||
@@ -712,6 +718,7 @@ const fadt_pm1a_cnt_blk = 64; // u32 (I/O port)
|
|||||||
const fadt_pm1b_cnt_blk = 68; // u32 (I/O port)
|
const fadt_pm1b_cnt_blk = 68; // u32 (I/O port)
|
||||||
const fadt_pm_tmr_blk = 76; // u32 (I/O port) — the PM timer counter
|
const fadt_pm_tmr_blk = 76; // u32 (I/O port) — the PM timer counter
|
||||||
const fadt_pm1_cnt_len = 89; // u8 (bytes)
|
const fadt_pm1_cnt_len = 89; // u8 (bytes)
|
||||||
|
const fadt_sci_int = 46; // u16 (the SCI's GSI)
|
||||||
const fadt_flags = 112; // u32
|
const fadt_flags = 112; // u32
|
||||||
const fadt_reset_register = 116; // GAS (12 bytes)
|
const fadt_reset_register = 116; // GAS (12 bytes)
|
||||||
const fadt_reset_value = 128; // u8
|
const fadt_reset_value = 128; // u8
|
||||||
@@ -729,6 +736,7 @@ fn parseFadt(header: *const SystemDescriptorTableHeader) void {
|
|||||||
const len: usize = header.length;
|
const len: usize = header.length;
|
||||||
const pi = &power_information;
|
const pi = &power_information;
|
||||||
|
|
||||||
|
pi.sci_interrupt = @truncate(fadt(u16, base, len, fadt_sci_int) orelse 0);
|
||||||
pi.smi_cmd = @truncate(fadt(u32, base, len, fadt_smi_cmd) orelse 0);
|
pi.smi_cmd = @truncate(fadt(u32, base, len, fadt_smi_cmd) orelse 0);
|
||||||
pi.acpi_enable = fadt(u8, base, len, fadt_acpi_enable) orelse 0;
|
pi.acpi_enable = fadt(u8, base, len, fadt_acpi_enable) orelse 0;
|
||||||
pi.acpi_disable = fadt(u8, base, len, fadt_acpi_disable) orelse 0;
|
pi.acpi_disable = fadt(u8, base, len, fadt_acpi_disable) orelse 0;
|
||||||
@@ -1186,16 +1194,6 @@ fn readCntRegister(base: [*]align(1) const u8, len: usize, xoff: usize, legacy_o
|
|||||||
|
|
||||||
/// The mapped configuration space of one PCI function (its 4 KiB ECAM page). Mapped
|
/// The mapped configuration space of one PCI function (its 4 KiB ECAM page). Mapped
|
||||||
/// writable so BAR sizing can probe it; reads and writes both go through here.
|
/// writable so BAR sizing can probe it; reads and writes both go through here.
|
||||||
fn pciConfigurationPtr(alloc: McfgAllocation, hal: Hal, bus: u8, device: u8, function: u8) [*]align(1) u8 {
|
|
||||||
const physical = alloc.base_address +
|
|
||||||
(@as(u64, bus - alloc.start_bus) << 20) +
|
|
||||||
(@as(u64, device) << 15) +
|
|
||||||
(@as(u64, function) << 12);
|
|
||||||
// Map the configuration page (writable, for BAR sizing) and use the virtual
|
|
||||||
// address the HAL hands back.
|
|
||||||
return @ptrFromInt(hal.mapMmio(physical, abi.page_size, true));
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Read a little-endian integer at `off` from a (possibly unaligned) byte pointer.
|
/// Read a little-endian integer at `off` from a (possibly unaligned) byte pointer.
|
||||||
/// x86 is little-endian and native, so an unaligned load suffices.
|
/// x86 is little-endian and native, so an unaligned load suffices.
|
||||||
fn rd(comptime T: type, bytes: [*]align(1) const u8, off: usize) T {
|
fn rd(comptime T: type, bytes: [*]align(1) const u8, off: usize) T {
|
||||||
@@ -1203,12 +1201,6 @@ fn rd(comptime T: type, bytes: [*]align(1) const u8, off: usize) T {
|
|||||||
return p.*;
|
return p.*;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Write a little-endian integer at `off` through a (possibly unaligned) pointer.
|
|
||||||
fn wr(comptime T: type, bytes: [*]align(1) u8, off: usize, value: T) void {
|
|
||||||
const p: *align(1) T = @ptrCast(bytes + off);
|
|
||||||
p.* = value;
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- tests ------------------------------------------------------------------
|
// --- tests ------------------------------------------------------------------
|
||||||
|
|
||||||
test "eisaIdToStr decodes a packed EISA id" {
|
test "eisaIdToStr decodes a packed EISA id" {
|
||||||
|
|||||||
@@ -50,6 +50,20 @@ pub fn parse(allocator: std.mem.Allocator, blocks: []const []const u8) !ParseRes
|
|||||||
return .{ .namespace = namespace, .consumed = consumed, .total = total };
|
return .{ .namespace = namespace, .consumed = consumed, .total = total };
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Count the Device objects in a parsed namespace — what the acpi service
|
||||||
|
/// (docs/m19-m20-plan.md M20) reports, and what the kernel's own parse counts
|
||||||
|
/// so the two can be checked equal across the ring-3 move.
|
||||||
|
pub fn deviceCount(namespace: *const Namespace) usize {
|
||||||
|
return countKind(namespace.root, .device);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn countKind(node: *const Node, kind: NodeKind) usize {
|
||||||
|
var n: usize = if (node.kind == kind) 1 else 0;
|
||||||
|
var c = node.first_child;
|
||||||
|
while (c) |child| : (c = child.next_sibling) n += countKind(child, kind);
|
||||||
|
return n;
|
||||||
|
}
|
||||||
|
|
||||||
/// Look up the `\_S{state}` sleep package in a parsed namespace and return its
|
/// Look up the `\_S{state}` sleep package in a parsed namespace and return its
|
||||||
/// first two integer elements (SLP_TYP for PM1a / PM1b), or null if absent.
|
/// first two integer elements (SLP_TYP for PM1a / PM1b), or null if absent.
|
||||||
pub fn sleepState(namespace: *Namespace, state: u8) ?SleepType {
|
pub fn sleepState(namespace: *Namespace, state: u8) ?SleepType {
|
||||||
|
|||||||
@@ -28,6 +28,11 @@ pub const DeviceClass = enum(u32) {
|
|||||||
/// A device named in the ACPI namespace (from the DSDT/SSDT), carrying a
|
/// A device named in the ACPI namespace (from the DSDT/SSDT), carrying a
|
||||||
/// hardware ID (`_HID`) and, where static, current resource settings (`_CRS`).
|
/// hardware ID (`_HID`) and, where static, current resource settings (`_CRS`).
|
||||||
acpi_device,
|
acpi_device,
|
||||||
|
/// The ACPI tables themselves, published as one node for the user-space acpi
|
||||||
|
/// service (docs/m19-m20-plan.md M20): memory resources over the AML blobs,
|
||||||
|
/// a broad io_port grant for OperationRegion access, and the SCI interrupt.
|
||||||
|
/// The one node whose claimant is trusted to run firmware bytecode.
|
||||||
|
acpi_tables,
|
||||||
unknown,
|
unknown,
|
||||||
};
|
};
|
||||||
|
|
||||||
|
|||||||
@@ -40,6 +40,13 @@ pub fn platformInformation() PlatformInformation {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// AML parse integrity/diagnostics (namespace node count, bytes consumed).
|
/// AML parse integrity/diagnostics (namespace node count, bytes consumed).
|
||||||
|
/// The number of Device objects in the kernel's own AML namespace, or 0 if the
|
||||||
|
/// parse produced none — the `acpi-parse` test compares the ring-3 service's
|
||||||
|
/// count against this.
|
||||||
|
pub fn amlDeviceCount() usize {
|
||||||
|
return acpi.amlDeviceCount();
|
||||||
|
}
|
||||||
|
|
||||||
pub fn amlStats() AmlStats {
|
pub fn amlStats() AmlStats {
|
||||||
return acpi.aml_stats;
|
return acpi.aml_stats;
|
||||||
}
|
}
|
||||||
@@ -72,7 +79,8 @@ pub fn discover(
|
|||||||
var device_tree = try DeviceTree.init(allocator);
|
var device_tree = try DeviceTree.init(allocator);
|
||||||
|
|
||||||
if (boot_information.acpi_rsdp != 0) {
|
if (boot_information.acpi_rsdp != 0) {
|
||||||
try acpi.discover(boot_information.acpi_rsdp, &device_tree, hal);
|
const memory_regions = @as([*]const boot_handoff.MemoryRegion, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.memory_map.regions)))[0..boot_information.memory_map.len];
|
||||||
|
try acpi.discover(boot_information.acpi_rsdp, memory_regions, &device_tree, hal);
|
||||||
} else {
|
} else {
|
||||||
// No ACPI RSDP. A device-tree boot would parse its blob here; today that
|
// No ACPI RSDP. A device-tree boot would parse its blob here; today that
|
||||||
// path is a stub, so this reports the machine described itself no way we
|
// path is a stub, so this reports the machine described itself no way we
|
||||||
|
|||||||
@@ -0,0 +1,253 @@
|
|||||||
|
//! /system/drivers/pci-bus — the PCI bus driver: enumeration moved out of ring 0
|
||||||
|
//! (docs/m19-m20-plan.md, M19). The device manager matches the `pci_host_bridge`
|
||||||
|
//! node and spawns one instance per bridge, the bridge's device id as argv[1] —
|
||||||
|
//! the same per-device contract as usb-xhci-bus.
|
||||||
|
//!
|
||||||
|
//! M19.1 (this increment): claim the bridge, map its ECAM window (resource 0;
|
||||||
|
//! the bus range and the MMIO apertures follow it), walk every
|
||||||
|
//! bus/device/function config header, and log what the walk finds — ending
|
||||||
|
//! with "pci-bus: N functions found", which the `pci-scan` scenario compares
|
||||||
|
//! against the kernel's own enumeration. Registration and reports (M19.2), and
|
||||||
|
//! the kernel walk's retirement (M19.3), build on this proven-equivalent scan.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const runtime = @import("runtime");
|
||||||
|
const protocol = runtime.device_manager_protocol;
|
||||||
|
const device = runtime.device;
|
||||||
|
|
||||||
|
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||||
|
var line: [128]u8 = undefined;
|
||||||
|
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||||
|
}
|
||||||
|
|
||||||
|
var bridge_id: u64 = protocol.no_device;
|
||||||
|
var ecam_base: usize = 0;
|
||||||
|
var ecam_physical: u64 = 0;
|
||||||
|
var start_bus: u64 = 0;
|
||||||
|
var bus_count: u64 = 0;
|
||||||
|
var manager_handle: runtime.ipc.Handle = 0;
|
||||||
|
|
||||||
|
/// One aligned 32-bit read from a function's configuration space.
|
||||||
|
fn configRead(bus: u64, dev: u64, function: u64, offset: u64) u32 {
|
||||||
|
const address = ecam_base + (((bus - start_bus) << 20) | (dev << 15) | (function << 12) | offset);
|
||||||
|
const register: *volatile u32 = @ptrFromInt(address);
|
||||||
|
return register.*;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn configWrite(bus: u64, dev: u64, function: u64, offset: u64, value: u32) void {
|
||||||
|
const address = ecam_base + (((bus - start_bus) << 20) | (dev << 15) | (function << 12) | offset);
|
||||||
|
const register: *volatile u32 = @ptrFromInt(address);
|
||||||
|
register.* = value;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn configRead16(bus: u64, dev: u64, function: u64, offset: u64) u16 {
|
||||||
|
const word = configRead(bus, dev, function, offset & ~@as(u64, 3));
|
||||||
|
return @truncate(word >> @intCast((offset & 3) * 8));
|
||||||
|
}
|
||||||
|
|
||||||
|
fn configWrite16(bus: u64, dev: u64, function: u64, offset: u64, value: u16) void {
|
||||||
|
const aligned = offset & ~@as(u64, 3);
|
||||||
|
const shift: u5 = @intCast((offset & 3) * 8);
|
||||||
|
const word = configRead(bus, dev, function, aligned);
|
||||||
|
const mask = @as(u32, 0xFFFF) << shift;
|
||||||
|
configWrite(bus, dev, function, aligned, (word & ~mask) | (@as(u32, value) << shift));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Claim the bridge, map the ECAM, hello the manager, then scan.
|
||||||
|
fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||||
|
_ = endpoint;
|
||||||
|
if (!device.claim(bridge_id)) {
|
||||||
|
writeLine("pci-bus: unable to claim bridge device {d}\n", .{bridge_id});
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||||
|
_ = runtime.system.write("pci-bus: out of memory\n");
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
const total = device.enumerate(buffer);
|
||||||
|
const descriptor = for (buffer[0..@min(total, buffer.len)]) |d| {
|
||||||
|
if (d.id == bridge_id) break d;
|
||||||
|
} else {
|
||||||
|
writeLine("pci-bus: device {d} not in the device tree\n", .{bridge_id});
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
// Resource 0 is the ECAM window (1 MiB of config space per bus); the bus
|
||||||
|
// range rides beside it. The MMIO apertures (M19.0) come after both.
|
||||||
|
if (descriptor.resource_count < 2 or descriptor.resources[0].kind != @intFromEnum(device.ResourceKind.memory)) {
|
||||||
|
_ = runtime.system.write("pci-bus: bridge has no ECAM window\n");
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
const bus_range = for (descriptor.resources[0..@intCast(descriptor.resource_count)]) |resource| {
|
||||||
|
if (resource.kind == @intFromEnum(device.ResourceKind.bus_range)) break resource;
|
||||||
|
} else {
|
||||||
|
_ = runtime.system.write("pci-bus: bridge has no bus range\n");
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
start_bus = bus_range.start;
|
||||||
|
bus_count = bus_range.len;
|
||||||
|
ecam_physical = descriptor.resources[0].start;
|
||||||
|
ecam_base = device.mmioMap(bridge_id, 0) orelse {
|
||||||
|
_ = runtime.system.write("pci-bus: ECAM mmio_map failed\n");
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
|
||||||
|
// The handshake, then the scan (reports join in M19.2).
|
||||||
|
var manager: ?runtime.ipc.Handle = null;
|
||||||
|
var tries: u32 = 0;
|
||||||
|
while (manager == null and tries < 100) : (tries += 1) {
|
||||||
|
manager = runtime.ipc.lookup(.device_manager);
|
||||||
|
if (manager == null) runtime.system.sleep(20);
|
||||||
|
}
|
||||||
|
const h = manager orelse {
|
||||||
|
_ = runtime.system.write("pci-bus: no device manager to hello\n");
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
const hello = protocol.Hello{ .role = @intFromEnum(protocol.Role.bus), .device_id = bridge_id };
|
||||||
|
var reply: [protocol.message_maximum]u8 = undefined;
|
||||||
|
const n = runtime.ipc.call(h, std.mem.asBytes(&hello), &reply) catch {
|
||||||
|
_ = runtime.system.write("pci-bus: hello call failed\n");
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
if (n < protocol.reply_size or std.mem.bytesToValue(protocol.HelloReply, reply[0..protocol.reply_size]).status != 0) {
|
||||||
|
_ = runtime.system.write("pci-bus: hello refused\n");
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
manager_handle = h;
|
||||||
|
|
||||||
|
scan();
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The brute-force walk the kernel does today, from ring 3: every bus in the
|
||||||
|
/// range, 32 devices, 8 functions; vendor id FFFFh means nothing decodes there,
|
||||||
|
/// and only multifunction devices get their functions 1..7 probed.
|
||||||
|
fn scan() void {
|
||||||
|
var found: u32 = 0;
|
||||||
|
var bus: u64 = start_bus;
|
||||||
|
while (bus < start_bus + bus_count) : (bus += 1) {
|
||||||
|
var dev: u64 = 0;
|
||||||
|
while (dev < 32) : (dev += 1) {
|
||||||
|
const first = configRead(bus, dev, 0, 0);
|
||||||
|
if (first & 0xFFFF == 0xFFFF) continue;
|
||||||
|
const multifunction = (configRead(bus, dev, 0, 0x0C) >> 16) & 0x80 != 0;
|
||||||
|
var function: u64 = 0;
|
||||||
|
while (function < 8) : (function += 1) {
|
||||||
|
if (function != 0 and !multifunction) break;
|
||||||
|
const vendor_device = configRead(bus, dev, function, 0);
|
||||||
|
if (vendor_device & 0xFFFF == 0xFFFF) continue;
|
||||||
|
const class_revision = configRead(bus, dev, function, 0x08);
|
||||||
|
found += 1;
|
||||||
|
writeLine("pci-bus: {d}:{d}.{d} class 0x{x:0>6}\n", .{ bus, dev, function, class_revision >> 8 });
|
||||||
|
registerAndReport(bus, dev, function, class_revision >> 8);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
writeLine("pci-bus: {d} functions found\n", .{found});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Register one function under the bridge and report it to the manager. The
|
||||||
|
/// descriptor mirrors the kernel's own recording byte for byte — config slice
|
||||||
|
/// as resource 0, then the sized BARs — so during coexistence the idempotent
|
||||||
|
/// device_register (M19.0) returns the kernel's existing node id rather than
|
||||||
|
/// growing a duplicate, and the report carries the id drivers already use.
|
||||||
|
fn registerAndReport(bus: u64, dev: u64, function: u64, class_triple: u32) void {
|
||||||
|
var descriptor = std.mem.zeroes(device.DeviceDescriptor);
|
||||||
|
descriptor.class = @intFromEnum(device.DeviceClass.pci_device);
|
||||||
|
descriptor.pci_class = class_triple;
|
||||||
|
descriptor.resources[0] = .{
|
||||||
|
.kind = @intFromEnum(device.ResourceKind.memory),
|
||||||
|
.start = ecam_physical + (((bus - start_bus) << 20) | (dev << 15) | (function << 12)),
|
||||||
|
.len = 4096,
|
||||||
|
};
|
||||||
|
descriptor.resource_count = 1;
|
||||||
|
|
||||||
|
// The standard BAR-sizing probe, exactly as the kernel does it: decode off,
|
||||||
|
// write all-ones, read the writable mask back, restore. Header type 0 only.
|
||||||
|
const header_type = (configRead(bus, dev, function, 0x0C) >> 16) & 0x7F;
|
||||||
|
if (header_type == 0) {
|
||||||
|
const command = configRead16(bus, dev, function, 0x04);
|
||||||
|
configWrite16(bus, dev, function, 0x04, command & ~@as(u16, 0b11));
|
||||||
|
var i: u64 = 0;
|
||||||
|
while (i < 6) : (i += 1) {
|
||||||
|
if (descriptor.resource_count >= 8) break;
|
||||||
|
const off = 0x10 + i * 4;
|
||||||
|
const original = configRead(bus, dev, function, off);
|
||||||
|
if (original == 0) continue;
|
||||||
|
const slot: usize = @intCast(descriptor.resource_count);
|
||||||
|
if (original & 1 != 0) {
|
||||||
|
configWrite(bus, dev, function, off, 0xFFFF_FFFF);
|
||||||
|
const readback = configRead(bus, dev, function, off);
|
||||||
|
configWrite(bus, dev, function, off, original);
|
||||||
|
const mask = readback & 0xFFFF_FFFC;
|
||||||
|
const size: u32 = if (mask == 0) 0 else (~mask +% 1) & 0xFFFF;
|
||||||
|
if (size == 0) continue; // unimplemented BAR — nothing to register
|
||||||
|
descriptor.resources[slot] = .{ .kind = @intFromEnum(device.ResourceKind.io_port), .start = original & 0xFFFF_FFFC, .len = size };
|
||||||
|
descriptor.resource_count += 1;
|
||||||
|
} else if ((original >> 1) & 0x3 == 2) {
|
||||||
|
const original_high = configRead(bus, dev, function, off + 4);
|
||||||
|
configWrite(bus, dev, function, off, 0xFFFF_FFFF);
|
||||||
|
configWrite(bus, dev, function, off + 4, 0xFFFF_FFFF);
|
||||||
|
const lo = configRead(bus, dev, function, off);
|
||||||
|
const hi = configRead(bus, dev, function, off + 4);
|
||||||
|
configWrite(bus, dev, function, off, original);
|
||||||
|
configWrite(bus, dev, function, off + 4, original_high);
|
||||||
|
const readback = (@as(u64, hi) << 32) | (lo & 0xFFFF_FFF0);
|
||||||
|
const size: u64 = if (readback == 0) 0 else ~readback +% 1;
|
||||||
|
i += 1; // consumed the high half regardless
|
||||||
|
if (size == 0) continue;
|
||||||
|
descriptor.resources[slot] = .{ .kind = @intFromEnum(device.ResourceKind.memory), .start = (@as(u64, original_high) << 32) | (original & 0xFFFF_FFF0), .len = size };
|
||||||
|
descriptor.resource_count += 1;
|
||||||
|
} else {
|
||||||
|
configWrite(bus, dev, function, off, 0xFFFF_FFFF);
|
||||||
|
const readback = configRead(bus, dev, function, off);
|
||||||
|
configWrite(bus, dev, function, off, original);
|
||||||
|
const mask = readback & 0xFFFF_FFF0;
|
||||||
|
const size: u32 = if (mask == 0) 0 else ~mask +% 1;
|
||||||
|
if (size == 0) continue;
|
||||||
|
descriptor.resources[slot] = .{ .kind = @intFromEnum(device.ResourceKind.memory), .start = original & 0xFFFF_FFF0, .len = size };
|
||||||
|
descriptor.resource_count += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
configWrite16(bus, dev, function, 0x04, command);
|
||||||
|
}
|
||||||
|
|
||||||
|
const registered = device.register(bridge_id, &descriptor) orelse {
|
||||||
|
writeLine("pci-bus: register refused for {d}:{d}.{d}\n", .{ bus, dev, function });
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
const report = protocol.ChildAdded{
|
||||||
|
.parent = bridge_id,
|
||||||
|
.bus_address = (bus << 8) | (dev << 3) | function,
|
||||||
|
.identity = class_triple,
|
||||||
|
.device_id = registered,
|
||||||
|
};
|
||||||
|
var reply: [protocol.message_maximum]u8 = undefined;
|
||||||
|
_ = runtime.ipc.call(manager_handle, std.mem.asBytes(&report), &reply) catch {
|
||||||
|
writeLine("pci-bus: child report for {d}:{d}.{d} failed\n", .{ bus, dev, function });
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?runtime.ipc.Handle) usize {
|
||||||
|
_ = message;
|
||||||
|
_ = reply;
|
||||||
|
_ = sender;
|
||||||
|
_ = capability;
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn main(init: runtime.process.Init) void {
|
||||||
|
const argument = init.arguments.get(1) orelse return; // bare (ramdisk sweep): stay silent
|
||||||
|
bridge_id = std.fmt.parseInt(u64, argument, 10) catch {
|
||||||
|
writeLine("pci-bus: malformed bridge device id '{s}'\n", .{argument});
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
runtime.service.run(protocol.message_maximum, .{
|
||||||
|
.init = initialise,
|
||||||
|
.on_message = onMessage,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
pub const panic = runtime.panic;
|
||||||
|
comptime {
|
||||||
|
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||||
|
}
|
||||||
@@ -1,68 +1,172 @@
|
|||||||
//! /system/drivers/usb-xhci-bus — the xHCI (USB 3) host-controller bus driver.
|
//! /system/drivers/usb-xhci-bus — the xHCI (USB 3) host-controller bus driver.
|
||||||
//! The device manager spawns **one instance per controller** it discovers (a machine
|
//! The device manager spawns **one instance per controller** it discovers (a
|
||||||
//! can carry several), passing the controller's device-tree id as argv[1]; this
|
//! machine can carry several), passing the controller's device-tree id as
|
||||||
//! instance claims that device and no other, so multiple instances never fight over
|
//! argv[1]; this instance claims that device and no other, so multiple
|
||||||
//! hardware. This increment proves the plumbing: parse the id, claim the controller,
|
//! instances never fight over hardware.
|
||||||
//! and report its MMIO window. The next increments map the registers and bring the
|
//!
|
||||||
//! controller up (reset, rings, port scan), then enumerate the USB devices on the
|
//! M18.2 (this increment): after the hello, real hardware — map the xHC's
|
||||||
//! bus with the usb-abi request builders and publish each with `device_register`.
|
//! register window (the first memory BAR; resource 0 is the ECAM config
|
||||||
|
//! space), read the capability registers, and walk the root-hub ports: one
|
||||||
|
//! `child_added` report to the manager per connected port, carrying the port
|
||||||
|
//! number and the PORTSC speed class as identity. No transfer rings yet —
|
||||||
|
//! descriptors and USB class matching are the USB track; the connect bit and
|
||||||
|
//! speed come straight from PORTSC, which reflects hardware state whether or
|
||||||
|
//! not the controller is running.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const runtime = @import("runtime");
|
const runtime = @import("runtime");
|
||||||
|
const protocol = runtime.device_manager_protocol;
|
||||||
const device = runtime.device;
|
const device = runtime.device;
|
||||||
|
|
||||||
/// Format one whole log line and emit it in a single `debug_write`, so concurrent
|
/// Format one whole log line and emit it in a single `debug_write`, so
|
||||||
/// instances (one per controller) can never interleave mid-line.
|
/// concurrent instances (one per controller) can never interleave mid-line.
|
||||||
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||||
var line: [128]u8 = undefined;
|
var line: [128]u8 = undefined;
|
||||||
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
var controller_id: u64 = protocol.no_device;
|
||||||
|
|
||||||
|
/// Claim the assigned controller, find its register window, and hello the
|
||||||
|
/// manager. Any failure returns false: the process exits cleanly, which the
|
||||||
|
/// manager reads as "meant to stop" — a missing assignment is not a crash loop.
|
||||||
|
fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||||
|
_ = endpoint;
|
||||||
|
if (!device.claim(controller_id)) {
|
||||||
|
writeLine("usb-xhci-bus: unable to claim controller device {d}\n", .{controller_id});
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Fetch our own descriptor back for the controller's resources.
|
||||||
|
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||||
|
_ = runtime.system.write("usb-xhci-bus: out of memory\n");
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
const total = device.enumerate(buffer);
|
||||||
|
const descriptor = for (buffer[0..@min(total, buffer.len)]) |d| {
|
||||||
|
if (d.id == controller_id) break d;
|
||||||
|
} else {
|
||||||
|
writeLine("usb-xhci-bus: device {d} not in the device tree\n", .{controller_id});
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
|
||||||
|
// The xHC's registers live behind the first memory BAR. Resource 0 is the
|
||||||
|
// function's ECAM configuration space (M15), so the walk starts at 1.
|
||||||
|
var register_index: u64 = 0;
|
||||||
|
const register_window = for (descriptor.resources[1..@intCast(descriptor.resource_count)], 1..) |resource, index| {
|
||||||
|
if (resource.kind == @intFromEnum(device.ResourceKind.memory)) {
|
||||||
|
register_index = index;
|
||||||
|
break resource;
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
writeLine("usb-xhci-bus: controller device {d} has no register BAR\n", .{controller_id});
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
writeLine("usb-xhci-bus: claimed controller device {d} (registers at 0x{x}, {d} bytes)\n", .{
|
||||||
|
controller_id,
|
||||||
|
register_window.start,
|
||||||
|
register_window.len,
|
||||||
|
});
|
||||||
|
register_base = device.mmioMap(controller_id, register_index) orelse {
|
||||||
|
_ = runtime.system.write("usb-xhci-bus: mmio_map failed\n");
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
|
||||||
|
// The handshake: role, protocol version, assignment — inside the manager's
|
||||||
|
// deadline (the lookup retries cover the manager still registering).
|
||||||
|
var manager: ?runtime.ipc.Handle = null;
|
||||||
|
var tries: u32 = 0;
|
||||||
|
while (manager == null and tries < 100) : (tries += 1) {
|
||||||
|
manager = runtime.ipc.lookup(.device_manager);
|
||||||
|
if (manager == null) runtime.system.sleep(20);
|
||||||
|
}
|
||||||
|
const h = manager orelse {
|
||||||
|
_ = runtime.system.write("usb-xhci-bus: no device manager to hello\n");
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
const hello = protocol.Hello{ .role = @intFromEnum(protocol.Role.bus), .device_id = controller_id };
|
||||||
|
var reply: [protocol.message_maximum]u8 = undefined;
|
||||||
|
const n = runtime.ipc.call(h, std.mem.asBytes(&hello), &reply) catch {
|
||||||
|
_ = runtime.system.write("usb-xhci-bus: hello call failed\n");
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
if (n < protocol.reply_size or std.mem.bytesToValue(protocol.HelloReply, reply[0..protocol.reply_size]).status != 0) {
|
||||||
|
_ = runtime.system.write("usb-xhci-bus: hello refused\n");
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
_ = runtime.system.write("usb-xhci-bus: hello acknowledged\n");
|
||||||
|
|
||||||
|
scanPorts(h);
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
var register_base: usize = 0;
|
||||||
|
|
||||||
|
/// One 32-bit volatile register read at `offset` from the mapped window.
|
||||||
|
fn readRegister(offset: usize) u32 {
|
||||||
|
const register: *volatile u32 = @ptrFromInt(register_base + offset);
|
||||||
|
return register.*;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The root-hub port scan: read the capability registers for the port count
|
||||||
|
/// and the operational-register offset, then one PORTSC per port. The connect
|
||||||
|
/// bit (CCS) and the speed field reflect hardware state directly — no
|
||||||
|
/// controller reset or run needed to *see* the devices; driving them needs the
|
||||||
|
/// rings (the USB track).
|
||||||
|
fn scanPorts(manager: runtime.ipc.Handle) void {
|
||||||
|
// Capability registers: CAPLENGTH is byte 0 of the first dword; HCSPARAMS1
|
||||||
|
// carries MaxPorts in bits 31:24.
|
||||||
|
const capability_length = readRegister(0) & 0xFF;
|
||||||
|
const structural = readRegister(0x04);
|
||||||
|
const maximum_ports: u32 = structural >> 24;
|
||||||
|
writeLine("usb-xhci-bus: {d} root-hub ports\n", .{maximum_ports});
|
||||||
|
|
||||||
|
// PORTSC registers: operational base + 0x400 + 0x10 per port (1-based).
|
||||||
|
var port: u32 = 1;
|
||||||
|
var connected: u32 = 0;
|
||||||
|
while (port <= maximum_ports) : (port += 1) {
|
||||||
|
const port_status = readRegister(capability_length + 0x400 + 0x10 * (port - 1));
|
||||||
|
if (port_status & 1 == 0) continue; // CCS: nothing connected
|
||||||
|
connected += 1;
|
||||||
|
const speed = (port_status >> 10) & 0xF; // the PORTSC port-speed class
|
||||||
|
writeLine("usb-xhci-bus: port {d} connected (speed class {d})\n", .{ port, speed });
|
||||||
|
|
||||||
|
const report = protocol.ChildAdded{
|
||||||
|
.parent = controller_id,
|
||||||
|
.bus_address = port,
|
||||||
|
.identity = speed,
|
||||||
|
};
|
||||||
|
var reply: [protocol.message_maximum]u8 = undefined;
|
||||||
|
_ = runtime.ipc.call(manager, std.mem.asBytes(&report), &reply) catch {
|
||||||
|
writeLine("usb-xhci-bus: child report for port {d} failed\n", .{port});
|
||||||
|
continue;
|
||||||
|
};
|
||||||
|
}
|
||||||
|
if (connected == 0) _ = runtime.system.write("usb-xhci-bus: no devices connected\n");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// No bus protocol to serve yet — transfer requests arrive with the USB track.
|
||||||
|
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?runtime.ipc.Handle) usize {
|
||||||
|
_ = message;
|
||||||
|
_ = reply;
|
||||||
|
_ = sender;
|
||||||
|
_ = capability;
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
|
||||||
pub fn main(init: runtime.process.Init) void {
|
pub fn main(init: runtime.process.Init) void {
|
||||||
const argument = init.arguments.get(1) orelse {
|
const argument = init.arguments.get(1) orelse {
|
||||||
_ = runtime.system.write("usb-xhci-bus: missing controller device id (argv[1])\n");
|
_ = runtime.system.write("usb-xhci-bus: missing controller device id (argv[1])\n");
|
||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
const controller_id = std.fmt.parseInt(u64, argument, 10) catch {
|
controller_id = std.fmt.parseInt(u64, argument, 10) catch {
|
||||||
writeLine("usb-xhci-bus: malformed controller device id '{s}'\n", .{argument});
|
writeLine("usb-xhci-bus: malformed controller device id '{s}'\n", .{argument});
|
||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
|
runtime.service.run(protocol.message_maximum, .{
|
||||||
if (!device.claim(controller_id)) {
|
.init = initialise,
|
||||||
writeLine("usb-xhci-bus: unable to claim controller device {d}\n", .{controller_id});
|
.on_message = onMessage,
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
// Fetch our own descriptor back for the controller's resources.
|
|
||||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
|
||||||
_ = runtime.system.write("usb-xhci-bus: out of memory\n");
|
|
||||||
return;
|
|
||||||
};
|
|
||||||
const total = device.enumerate(buffer);
|
|
||||||
const descriptor = for (buffer[0..@min(total, buffer.len)]) |d| {
|
|
||||||
if (d.id == controller_id) break d;
|
|
||||||
} else {
|
|
||||||
writeLine("usb-xhci-bus: device {d} not in the device tree\n", .{controller_id});
|
|
||||||
return;
|
|
||||||
};
|
|
||||||
|
|
||||||
// The controller's operational registers live behind BAR0, enumerated as the
|
|
||||||
// device's first memory resource.
|
|
||||||
const register_window = for (descriptor.resources[0..@intCast(descriptor.resource_count)]) |resource| {
|
|
||||||
if (resource.kind == @intFromEnum(device.ResourceKind.memory)) break resource;
|
|
||||||
} else {
|
|
||||||
writeLine("usb-xhci-bus: controller device {d} has no MMIO window\n", .{controller_id});
|
|
||||||
return;
|
|
||||||
};
|
|
||||||
writeLine("usb-xhci-bus: claimed controller device {d} (registers at 0x{x}, {d} bytes)\n", .{
|
|
||||||
controller_id,
|
|
||||||
register_window.start,
|
|
||||||
register_window.len,
|
|
||||||
});
|
});
|
||||||
|
|
||||||
// Controller bring-up (map the window, reset, rings, port scan) is the next
|
|
||||||
// increment; stay resident as the bus's supervisor in the meantime.
|
|
||||||
while (true) runtime.system.sleep(1000);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
pub const panic = runtime.panic;
|
pub const panic = runtime.panic;
|
||||||
|
|||||||
@@ -104,6 +104,19 @@ pub fn ownerOf(id: u64) ?u32 {
|
|||||||
return claimed[@intCast(id)];
|
return claimed[@intCast(id)];
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Release every claim held by `owner` — called by the process layer on every
|
||||||
|
/// path out of a process (exit, fault, kill), so a restarted driver can claim its
|
||||||
|
/// hardware again (docs/process-lifecycle.md iron rule 1: cleanup is the kernel's
|
||||||
|
/// job). The devices stay in the table — they describe hardware, which did not go
|
||||||
|
/// away — only their ownership clears.
|
||||||
|
pub fn releaseAllOwnedBy(owner: u32) void {
|
||||||
|
for (claimed[0..count]) |*slot| {
|
||||||
|
if (slot.*) |o| {
|
||||||
|
if (o == owner) slot.* = null;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// Resource `index` of device `id`, or null if out of range.
|
/// Resource `index` of device `id`, or null if out of range.
|
||||||
pub fn resourceOf(id: u64, index: u64) ?device_abi.ResourceDescriptor {
|
pub fn resourceOf(id: u64, index: u64) ?device_abi.ResourceDescriptor {
|
||||||
if (id >= count) return null;
|
if (id >= count) return null;
|
||||||
@@ -118,7 +131,16 @@ pub fn resourceOf(id: u64, index: u64) ?device_abi.ResourceDescriptor {
|
|||||||
/// and would otherwise vacuously "fit" anywhere.
|
/// and would otherwise vacuously "fit" anywhere.
|
||||||
fn contains(parent: device_abi.ResourceDescriptor, child: device_abi.ResourceDescriptor) bool {
|
fn contains(parent: device_abi.ResourceDescriptor, child: device_abi.ResourceDescriptor) bool {
|
||||||
if (parent.kind != child.kind) return false;
|
if (parent.kind != child.kind) return false;
|
||||||
if (child.kind == @intFromEnum(device_abi.ResourceKind.irq)) return parent.start == child.start;
|
if (child.kind == @intFromEnum(device_abi.ResourceKind.irq)) {
|
||||||
|
// Range containment: an interrupt line is still indivisible (a child owns
|
||||||
|
// exactly one GSI), but a parent may own a *range* of lines so a broad
|
||||||
|
// owner — the acpi-tables node, whose firmware names any legacy IRQ —
|
||||||
|
// can contain its children's specific lines. A length-1 parent range is
|
||||||
|
// exactly the old equality rule, so existing single-IRQ parents are
|
||||||
|
// unaffected.
|
||||||
|
const span = if (parent.len == 0) 1 else parent.len;
|
||||||
|
return child.start >= parent.start and child.start < parent.start + span;
|
||||||
|
}
|
||||||
if (child.len == 0 or parent.len == 0) return false;
|
if (child.len == 0 or parent.len == 0) return false;
|
||||||
// No overflow: a resource that wraps the address space is not containable.
|
// No overflow: a resource that wraps the address space is not containable.
|
||||||
const child_end = std.math.add(u64, child.start, child.len) catch return false;
|
const child_end = std.math.add(u64, child.start, child.len) catch return false;
|
||||||
@@ -167,6 +189,26 @@ pub fn register(parent_id: u64, owner: u32, descriptor: *const device_abi.Device
|
|||||||
if (!ok) return error.NotContained;
|
if (!ok) return error.NotContained;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Idempotent on exact match (docs/m19-m20-plan.md decision 3): a restarted
|
||||||
|
// registering bus re-registers what it rediscovers, and the table has no
|
||||||
|
// unregister — an identical (class, identity, resources) child under the
|
||||||
|
// same parent returns the existing id instead of appending a duplicate.
|
||||||
|
for (devices[0..count]) |*existing| {
|
||||||
|
if (existing.parent != parent_id) continue;
|
||||||
|
if (existing.class != descriptor.class) continue;
|
||||||
|
if (existing.pci_class != descriptor.pci_class) continue;
|
||||||
|
if (existing.hid_len != descriptor.hid_len) continue;
|
||||||
|
if (!std.mem.eql(u8, existing.hid[0..@intCast(existing.hid_len)], descriptor.hid[0..@intCast(descriptor.hid_len)])) continue;
|
||||||
|
if (existing.resource_count != descriptor.resource_count) continue;
|
||||||
|
var same = true;
|
||||||
|
for (0..@intCast(descriptor.resource_count)) |i| {
|
||||||
|
const a = existing.resources[i];
|
||||||
|
const b = descriptor.resources[i];
|
||||||
|
if (a.kind != b.kind or a.start != b.start or a.len != b.len) same = false;
|
||||||
|
}
|
||||||
|
if (same) return existing.id;
|
||||||
|
}
|
||||||
|
|
||||||
var d = std.mem.zeroes(device_abi.DeviceDescriptor);
|
var d = std.mem.zeroes(device_abi.DeviceDescriptor);
|
||||||
d.id = count;
|
d.id = count;
|
||||||
d.parent = parent_id;
|
d.parent = parent_id;
|
||||||
|
|||||||
@@ -412,13 +412,26 @@ fn recoverableFault(vector: u64) bool {
|
|||||||
/// plus a POST code and a persistent breadcrumb. (A ring-3 fault on a *borrowed*
|
/// plus a POST code and a persistent breadcrumb. (A ring-3 fault on a *borrowed*
|
||||||
/// kernel thread — process.run, the user-pf isolation probe — also lands here: there
|
/// kernel thread — process.run, the user-pf isolation probe — also lands here: there
|
||||||
/// is no scheduled process to kill.)
|
/// is no scheduled process to kill.)
|
||||||
|
/// Classify a CPU exception vector as the ExitReason a supervisor reads — the
|
||||||
|
/// fault classes of docs/process-lifecycle.md. Faults are exit reasons, never
|
||||||
|
/// signals delivered to the faulting process: recovery is restart, not a handler.
|
||||||
|
fn exitReasonForVector(vector: u64) abi.ExitReason {
|
||||||
|
return switch (vector) {
|
||||||
|
14 => .segmentation_fault, // page fault
|
||||||
|
6 => .illegal_instruction, // invalid opcode
|
||||||
|
0, 16, 19 => .arithmetic_fault, // divide error, x87, SIMD
|
||||||
|
13 => .protection_fault, // general protection
|
||||||
|
else => .fault,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
fn onException(state: *const architecture.CpuState) noreturn {
|
fn onException(state: *const architecture.CpuState) noreturn {
|
||||||
if (architecture.fromUser(state) and scheduler.currentIsUserProcess() and recoverableFault(state.vector)) {
|
if (architecture.fromUser(state) and scheduler.currentIsUserProcess() and recoverableFault(state.vector)) {
|
||||||
statusPrint("\ndanos: process {d} ({s}) killed by {s} (vector {d}) on core {d}\n", .{ scheduler.currentId(), scheduler.current().name(), architecture.exceptionName(state.vector), state.vector, scheduler.currentCpuIndex() });
|
statusPrint("\ndanos: process {d} ({s}) killed by {s} (vector {d}) on core {d}\n", .{ scheduler.currentId(), scheduler.current().name(), architecture.exceptionName(state.vector), state.vector, scheduler.currentCpuIndex() });
|
||||||
statusPrint(" error code : 0x{x}\n", .{state.error_code});
|
statusPrint(" error code : 0x{x}\n", .{state.error_code});
|
||||||
statusPrint(" IP : 0x{x:0>16}\n", .{architecture.instructionPointer(state)});
|
statusPrint(" IP : 0x{x:0>16}\n", .{architecture.instructionPointer(state)});
|
||||||
if (architecture.faultAddress(state)) |address| statusPrint(" fault addr : 0x{x:0>16}\n", .{address});
|
if (architecture.faultAddress(state)) |address| statusPrint(" fault addr : 0x{x:0>16}\n", .{address});
|
||||||
process.killCurrentProcess(); // reclaims everything, reschedules; never returns
|
process.killCurrentProcess(exitReasonForVector(state.vector)); // reclaims everything, reschedules; never returns
|
||||||
}
|
}
|
||||||
|
|
||||||
log.checkpoint(cp_exception);
|
log.checkpoint(cp_exception);
|
||||||
|
|||||||
+225
-4
@@ -137,6 +137,7 @@ pub fn init() void {
|
|||||||
architecture.setSystemCallHandler(system_call);
|
architecture.setSystemCallHandler(system_call);
|
||||||
scheduler.terminate_current_hook = terminateCurrentLocked;
|
scheduler.terminate_current_hook = terminateCurrentLocked;
|
||||||
scheduler.reap_task_hook = reapTaskLocked;
|
scheduler.reap_task_hook = reapTaskLocked;
|
||||||
|
scheduler.timer_tick_hook = timerSweepLocked;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Return -1 (as an unsigned bit pattern) in the system_call result register.
|
/// Return -1 (as an unsigned bit pattern) in the system_call result register.
|
||||||
@@ -164,6 +165,7 @@ fn system_call(state: *architecture.CpuState) void {
|
|||||||
// A scheduled process tears down fully (terminateCurrent); a borrowed
|
// A scheduled process tears down fully (terminateCurrent); a borrowed
|
||||||
// test thread unwinds back to the kernel that entered it.
|
// test thread unwinds back to the kernel that entered it.
|
||||||
if (scheduler.currentIsUserProcess()) {
|
if (scheduler.currentIsUserProcess()) {
|
||||||
|
scheduler.current().exit_reason = .exited;
|
||||||
terminateCurrent();
|
terminateCurrent();
|
||||||
} else architecture.userExit();
|
} else architecture.userExit();
|
||||||
},
|
},
|
||||||
@@ -199,6 +201,11 @@ fn system_call(state: *architecture.CpuState) void {
|
|||||||
.clock => systemClock(state),
|
.clock => systemClock(state),
|
||||||
.process_enumerate => systemProcessEnumerate(state),
|
.process_enumerate => systemProcessEnumerate(state),
|
||||||
.process_kill => systemProcessKill(state),
|
.process_kill => systemProcessKill(state),
|
||||||
|
.process_exit_reason => systemProcessExitReason(state),
|
||||||
|
.process_subscribe => systemProcessSubscribe(state),
|
||||||
|
.signal_bind => systemSignalBind(state),
|
||||||
|
.process_signal => systemProcessSignal(state),
|
||||||
|
.timer_bind => systemTimerBind(state),
|
||||||
_ => fail(state),
|
_ => fail(state),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -291,6 +298,8 @@ fn systemDeviceEnumerate(state: *architecture.CpuState) void {
|
|||||||
|
|
||||||
/// device_claim(id) -> 0/-1: take exclusive ownership of a device for this process.
|
/// device_claim(id) -> 0/-1: take exclusive ownership of a device for this process.
|
||||||
fn systemDeviceClaim(state: *architecture.CpuState) void {
|
fn systemDeviceClaim(state: *architecture.CpuState) void {
|
||||||
|
const claim_flags = sync.enter();
|
||||||
|
defer sync.leave(claim_flags);
|
||||||
if (devices_broker.claim(architecture.systemCallArg(state, 0), scheduler.current().id))
|
if (devices_broker.claim(architecture.systemCallArg(state, 0), scheduler.current().id))
|
||||||
architecture.setSystemCallResult(state, 0)
|
architecture.setSystemCallResult(state, 0)
|
||||||
else
|
else
|
||||||
@@ -305,10 +314,22 @@ fn systemMmioMap(state: *architecture.CpuState) void {
|
|||||||
const resource_index = architecture.systemCallArg(state, 1);
|
const resource_index = architecture.systemCallArg(state, 1);
|
||||||
const t = scheduler.current();
|
const t = scheduler.current();
|
||||||
if (t.aspace == 0) return fail(state);
|
if (t.aspace == 0) return fail(state);
|
||||||
|
// Read the broker table under the lock: ring-3 device_register (M19) now
|
||||||
|
// mutates it concurrently on other cores, so a lock-free read here could
|
||||||
|
// see a torn resource (and a torn length used to panic the arithmetic
|
||||||
|
// below on integer overflow).
|
||||||
|
const r = blk: {
|
||||||
|
const flags = sync.enter();
|
||||||
|
defer sync.leave(flags);
|
||||||
const owner = devices_broker.ownerOf(device_id) orelse return fail(state);
|
const owner = devices_broker.ownerOf(device_id) orelse return fail(state);
|
||||||
if (owner != t.id) return fail(state); // not claimed by this process
|
if (owner != t.id) return fail(state); // not claimed by this process
|
||||||
const r = devices_broker.resourceOf(device_id, resource_index) orelse return fail(state);
|
break :blk devices_broker.resourceOf(device_id, resource_index) orelse return fail(state);
|
||||||
|
};
|
||||||
if (r.kind != @intFromEnum(device_abi.ResourceKind.memory)) return fail(state);
|
if (r.kind != @intFromEnum(device_abi.ResourceKind.memory)) return fail(state);
|
||||||
|
// A zero-length or wrapping window is not mappable — fail cleanly rather
|
||||||
|
// than underflow `r.len - 1`.
|
||||||
|
if (r.len == 0) return fail(state);
|
||||||
|
if (@addWithOverflow(r.start, r.len)[1] != 0) return fail(state);
|
||||||
|
|
||||||
if (t.device_map_next == 0) t.device_map_next = device_arena_base;
|
if (t.device_map_next == 0) t.device_map_next = device_arena_base;
|
||||||
const first = r.start & ~@as(u64, page_size - 1);
|
const first = r.start & ~@as(u64, page_size - 1);
|
||||||
@@ -444,6 +465,11 @@ fn systemDeviceRegister(state: *architecture.CpuState) void {
|
|||||||
var descriptor: device_abi.DeviceDescriptor = undefined;
|
var descriptor: device_abi.DeviceDescriptor = undefined;
|
||||||
if (!ipc.copyFromUser(t.aspace, descriptor_ptr, std.mem.asBytes(&descriptor))) return fail(state);
|
if (!ipc.copyFromUser(t.aspace, descriptor_ptr, std.mem.asBytes(&descriptor))) return fail(state);
|
||||||
|
|
||||||
|
// Under the big kernel lock: the broker's table is also mutated by the
|
||||||
|
// death sweep (releaseAllOwnedBy) and read by enumerate on other cores —
|
||||||
|
// ring-3 registration (M19) made those genuinely concurrent.
|
||||||
|
const flags = sync.enter();
|
||||||
|
defer sync.leave(flags);
|
||||||
const id = devices_broker.register(parent_id, t.id, &descriptor) catch return fail(state);
|
const id = devices_broker.register(parent_id, t.id, &descriptor) catch return fail(state);
|
||||||
architecture.setSystemCallResult(state, id);
|
architecture.setSystemCallResult(state, id);
|
||||||
}
|
}
|
||||||
@@ -552,7 +578,11 @@ pub var fault_kill_count: u64 = 0;
|
|||||||
/// endpoint reference destroys the Endpoint, and a still-bound GSI would have an
|
/// endpoint reference destroys the Endpoint, and a still-bound GSI would have an
|
||||||
/// ISR call notifyFromIsr on freed memory the next time the device fired.
|
/// ISR call notifyFromIsr on freed memory the next time the device fired.
|
||||||
/// `releaseOwner` also leaves the line masked, so a dead driver's device goes
|
/// `releaseOwner` also leaves the line masked, so a dead driver's device goes
|
||||||
/// quiet rather than storming.
|
/// quiet rather than storming. (It drops MSI vectors by the same owner sweep.)
|
||||||
|
/// - Device claims are released with the IRQ bindings, so a restarted driver can
|
||||||
|
/// claim the same hardware again — the cleanup half of process-lifecycle.md's
|
||||||
|
/// iron rule 1. Claims hold no pointers, so ordering is free; they go here so
|
||||||
|
/// the exit notification (below, last) observes a fully-released child.
|
||||||
/// - A client this task still owes a reply to (it died between receive and reply)
|
/// - A client this task still owes a reply to (it died between receive and reply)
|
||||||
/// is failed with -EPEER rather than left blocked forever — a dead server must
|
/// is failed with -EPEER rather than left blocked forever — a dead server must
|
||||||
/// not hang its callers.
|
/// not hang its callers.
|
||||||
@@ -565,7 +595,33 @@ pub var fault_kill_count: u64 = 0;
|
|||||||
/// reference taken at spawn is dropped with it.
|
/// reference taken at spawn is dropped with it.
|
||||||
/// Precondition: the big kernel lock is held.
|
/// Precondition: the big kernel lock is held.
|
||||||
fn releaseTaskResourcesLocked(t: *scheduler.Task) void {
|
fn releaseTaskResourcesLocked(t: *scheduler.Task) void {
|
||||||
|
recordExitLocked(t);
|
||||||
irq.releaseOwner(t.id);
|
irq.releaseOwner(t.id);
|
||||||
|
devices_broker.releaseAllOwnedBy(t.id);
|
||||||
|
// The dying task's signal endpoint and one-shot timers go with it.
|
||||||
|
if (t.signal_endpoint) |raw| {
|
||||||
|
ipc.dropRef(@ptrCast(@alignCast(raw)));
|
||||||
|
t.signal_endpoint = null;
|
||||||
|
}
|
||||||
|
t.pending_signals = 0;
|
||||||
|
for (&one_shot_timers) |*slot| {
|
||||||
|
if (slot.*) |timer| {
|
||||||
|
if (timer.owner == t.id) {
|
||||||
|
ipc.dropRef(timer.endpoint);
|
||||||
|
slot.* = null;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// A dead subscriber's own subscriptions go first: it must not hear about
|
||||||
|
// itself, and the slots' endpoint references drop with it.
|
||||||
|
for (&exit_subscribers) |*slot| {
|
||||||
|
if (slot.*) |subscriber| {
|
||||||
|
if (subscriber.owner == t.id) {
|
||||||
|
ipc.dropRef(subscriber.endpoint);
|
||||||
|
slot.* = null;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
if (t.ipc_client) |client| {
|
if (t.ipc_client) |client| {
|
||||||
t.ipc_client = null;
|
t.ipc_client = null;
|
||||||
client.ipc_status = -ipc.EPEER;
|
client.ipc_status = -ipc.EPEER;
|
||||||
@@ -575,6 +631,12 @@ fn releaseTaskResourcesLocked(t: *scheduler.Task) void {
|
|||||||
scheduler.removeFromWaitQueueLocked(t);
|
scheduler.removeFromWaitQueueLocked(t);
|
||||||
scheduler.forgetIpcClientLocked(t);
|
scheduler.forgetIpcClientLocked(t);
|
||||||
ipc.closeHandles(t);
|
ipc.closeHandles(t);
|
||||||
|
// Publish the exit to every subscriber (docs/process-lifecycle.md): the same
|
||||||
|
// badge encoding as the supervisor's notification, and equally late, so a
|
||||||
|
// subscriber also observes a fully-released child.
|
||||||
|
for (&exit_subscribers) |*slot| {
|
||||||
|
if (slot.*) |subscriber| ipc.notifyLocked(subscriber.endpoint, abi.notify_exit_bit | t.id);
|
||||||
|
}
|
||||||
if (t.exit_endpoint) |raw| {
|
if (t.exit_endpoint) |raw| {
|
||||||
const endpoint: *ipc.Endpoint = @ptrCast(@alignCast(raw));
|
const endpoint: *ipc.Endpoint = @ptrCast(@alignCast(raw));
|
||||||
t.exit_endpoint = null;
|
t.exit_endpoint = null;
|
||||||
@@ -628,6 +690,7 @@ pub fn killProcess(caller_id: u32, target_id: u32) i64 {
|
|||||||
const target = scheduler.taskByIdLocked(target_id) orelse return -ipc.ESRCH;
|
const target = scheduler.taskByIdLocked(target_id) orelse return -ipc.ESRCH;
|
||||||
if (target.aspace == 0) return -ipc.ESRCH; // kernel tasks are not processes
|
if (target.aspace == 0) return -ipc.ESRCH; // kernel tasks are not processes
|
||||||
if (target.supervisor != caller_id) return -ipc.EPERM;
|
if (target.supervisor != caller_id) return -ipc.EPERM;
|
||||||
|
target.exit_reason = .killed;
|
||||||
if (target.state == .running) {
|
if (target.state == .running) {
|
||||||
target.kill_pending = true;
|
target.kill_pending = true;
|
||||||
} else {
|
} else {
|
||||||
@@ -640,12 +703,170 @@ pub fn killProcess(caller_id: u32, target_id: u32) i64 {
|
|||||||
/// The fault is confined to the process — the kernel trapped it on the task's own
|
/// The fault is confined to the process — the kernel trapped it on the task's own
|
||||||
/// kernel stack and is intact — so everything the process held is reclaimed and the
|
/// kernel stack and is intact — so everything the process held is reclaimed and the
|
||||||
/// core reschedules. The system keeps running; only the faulting process dies
|
/// core reschedules. The system keeps running; only the faulting process dies
|
||||||
/// (docs/resilience.md: fault -> kill -> continue).
|
/// (docs/resilience.md: fault -> kill -> continue). `reason` is the fault class
|
||||||
pub fn killCurrentProcess() noreturn {
|
/// (from the vector), recorded for the supervisor's `process_exit_reason`.
|
||||||
|
pub fn killCurrentProcess(reason: abi.ExitReason) noreturn {
|
||||||
|
scheduler.current().exit_reason = reason;
|
||||||
fault_kill_count += 1;
|
fault_kill_count += 1;
|
||||||
terminateCurrent();
|
terminateCurrent();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The bounded record of recent deaths, for `process_exit_reason`: ids are never
|
||||||
|
/// reused, so a ring keyed by id is enough — a record evicted by wraparound reads
|
||||||
|
/// as -ESRCH, the same as an id that never lived, which a supervisor treats as
|
||||||
|
/// "too late to ask". Written under the big kernel lock by the reap.
|
||||||
|
const exit_record_capacity = 64;
|
||||||
|
const ExitRecord = struct { id: u32 = 0, supervisor: u32 = 0, reason: abi.ExitReason = .exited, valid: bool = false };
|
||||||
|
var exit_records: [exit_record_capacity]ExitRecord = .{ExitRecord{}} ** exit_record_capacity;
|
||||||
|
var exit_record_next: usize = 0;
|
||||||
|
|
||||||
|
/// Record a dying task's (id, supervisor, reason) — called by the reap before the
|
||||||
|
/// exit notification is posted, so a supervisor that hears the notification can
|
||||||
|
/// always still query the reason. Precondition: the big kernel lock is held.
|
||||||
|
fn recordExitLocked(t: *scheduler.Task) void {
|
||||||
|
exit_records[exit_record_next] = .{ .id = t.id, .supervisor = t.supervisor, .reason = t.exit_reason, .valid = true };
|
||||||
|
exit_record_next = (exit_record_next + 1) % exit_record_capacity;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// How dead process `id` ended, for `caller` — the kernel half of the
|
||||||
|
/// process_exit_reason system call. Returns the ExitReason value, -ESRCH (never
|
||||||
|
/// lived, still alive, or evicted from the ring), or -EPERM (the caller was not
|
||||||
|
/// its supervisor — the same authority gate as process_kill).
|
||||||
|
pub fn exitReasonOf(caller_id: u32, target_id: u32) i64 {
|
||||||
|
const flags = sync.enter();
|
||||||
|
defer sync.leave(flags);
|
||||||
|
for (&exit_records) |*record| {
|
||||||
|
if (record.valid and record.id == target_id) {
|
||||||
|
if (record.supervisor != caller_id) return -ipc.EPERM;
|
||||||
|
return @intFromEnum(record.reason);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return -ipc.ESRCH;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The published exit events' subscribers (docs/process-lifecycle.md "Who learns
|
||||||
|
/// of a death"): stateful services — the VFS's file handles, input's
|
||||||
|
/// subscriptions — that must release what a dead client held and cannot learn it
|
||||||
|
/// any other way (a client that simply never calls again looks like silence).
|
||||||
|
/// Bounded like every kernel table; each entry holds its own endpoint reference.
|
||||||
|
const exit_subscriber_capacity = 8;
|
||||||
|
const ExitSubscriber = struct { endpoint: *ipc.Endpoint, owner: u32 };
|
||||||
|
var exit_subscribers: [exit_subscriber_capacity]?ExitSubscriber = .{null} ** exit_subscriber_capacity;
|
||||||
|
|
||||||
|
/// process_subscribe(endpoint): subscribe the caller's endpoint to published exit
|
||||||
|
/// events. Ungated, like process_enumerate — what is running (and dying) is not a
|
||||||
|
/// secret between cooperating processes. -ENOSPC when the table is full.
|
||||||
|
fn systemProcessSubscribe(state: *architecture.CpuState) void {
|
||||||
|
const t = scheduler.current();
|
||||||
|
if (t.aspace == 0) return fail(state);
|
||||||
|
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
|
||||||
|
const flags = sync.enter();
|
||||||
|
defer sync.leave(flags);
|
||||||
|
for (&exit_subscribers) |*slot| {
|
||||||
|
if (slot.* == null) {
|
||||||
|
endpoint.refcount += 1; // the slot's own reference, dropped on unsubscribe-by-death
|
||||||
|
slot.* = .{ .endpoint = endpoint, .owner = t.id };
|
||||||
|
return architecture.setSystemCallResult(state, 0);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
failErr(state, ipc.ENOSPC);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// signal_bind(endpoint): nominate where this process's signals arrive — the
|
||||||
|
/// IRQ-as-IPC pattern a fourth time (docs/process-lifecycle.md). Replacing a
|
||||||
|
/// binding drops the old reference; signals that pended while unbound are
|
||||||
|
/// delivered immediately on bind, coalesced into one notification.
|
||||||
|
fn systemSignalBind(state: *architecture.CpuState) void {
|
||||||
|
const t = scheduler.current();
|
||||||
|
if (t.aspace == 0) return fail(state);
|
||||||
|
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
|
||||||
|
const flags = sync.enter();
|
||||||
|
defer sync.leave(flags);
|
||||||
|
if (t.signal_endpoint) |raw| ipc.dropRef(@ptrCast(@alignCast(raw)));
|
||||||
|
endpoint.refcount += 1;
|
||||||
|
t.signal_endpoint = @ptrCast(endpoint);
|
||||||
|
if (t.pending_signals != 0) {
|
||||||
|
ipc.notifyLocked(endpoint, abi.notify_signal_bit | t.pending_signals);
|
||||||
|
t.pending_signals = 0;
|
||||||
|
}
|
||||||
|
architecture.setSystemCallResult(state, 0);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// process_signal(id, signal): post a signal — a one-way, coalescing statement,
|
||||||
|
/// never a question (docs/process-lifecycle.md). The authority gate is the
|
||||||
|
/// supervision link, like kill; a process may also signal itself. Unbound
|
||||||
|
/// targets accumulate the signal in their pending mask.
|
||||||
|
fn systemProcessSignal(state: *architecture.CpuState) void {
|
||||||
|
const t = scheduler.current();
|
||||||
|
if (t.aspace == 0) return fail(state);
|
||||||
|
const id = architecture.systemCallArg(state, 0);
|
||||||
|
const signal = architecture.systemCallArg(state, 1);
|
||||||
|
if (id > std.math.maxInt(u32)) return failErr(state, ipc.ESRCH);
|
||||||
|
if (signal > 31) return failErr(state, ipc.EBADF); // not a Signal bit position
|
||||||
|
const flags = sync.enter();
|
||||||
|
defer sync.leave(flags);
|
||||||
|
const target = scheduler.taskByIdLocked(@intCast(id)) orelse return failErr(state, ipc.ESRCH);
|
||||||
|
if (target.aspace == 0) return failErr(state, ipc.ESRCH);
|
||||||
|
if (target.supervisor != t.id and target.id != t.id) return failErr(state, ipc.EPERM);
|
||||||
|
target.pending_signals |= @as(u32, 1) << @intCast(signal);
|
||||||
|
if (target.signal_endpoint) |raw| {
|
||||||
|
const endpoint: *ipc.Endpoint = @ptrCast(@alignCast(raw));
|
||||||
|
ipc.notifyLocked(endpoint, abi.notify_signal_bit | target.pending_signals);
|
||||||
|
target.pending_signals = 0;
|
||||||
|
}
|
||||||
|
architecture.setSystemCallResult(state, 0);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The one-shot timers of timer_bind: the missing timed wait. A service arms a
|
||||||
|
/// deadline and keeps serving; the expiry arrives in the same replyWait as
|
||||||
|
/// everything else (notify_timer_bit). What stop-sequence escalation, hello
|
||||||
|
/// deadlines, and restart backoff are built from — and later, `alarm`.
|
||||||
|
const timer_capacity = 16;
|
||||||
|
const OneShotTimer = struct { deadline: u64, endpoint: *ipc.Endpoint, owner: u32 };
|
||||||
|
var one_shot_timers: [timer_capacity]?OneShotTimer = .{null} ** timer_capacity;
|
||||||
|
|
||||||
|
/// Sweep expired timers — hung on scheduler.timer_tick_hook, so it runs on every
|
||||||
|
/// tick with the big kernel lock held, like the sleeper wake it rides beside.
|
||||||
|
fn timerSweepLocked() void {
|
||||||
|
const now = architecture.millis();
|
||||||
|
for (&one_shot_timers) |*slot| {
|
||||||
|
if (slot.*) |timer| {
|
||||||
|
if (now >= timer.deadline) {
|
||||||
|
ipc.notifyLocked(timer.endpoint, abi.notify_timer_bit);
|
||||||
|
ipc.dropRef(timer.endpoint);
|
||||||
|
slot.* = null;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// timer_bind(endpoint, ms): arm a one-shot timer. -ENOSPC when the table is full.
|
||||||
|
fn systemTimerBind(state: *architecture.CpuState) void {
|
||||||
|
const t = scheduler.current();
|
||||||
|
if (t.aspace == 0) return fail(state);
|
||||||
|
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
|
||||||
|
const ms = architecture.systemCallArg(state, 1);
|
||||||
|
const flags = sync.enter();
|
||||||
|
defer sync.leave(flags);
|
||||||
|
for (&one_shot_timers) |*slot| {
|
||||||
|
if (slot.* == null) {
|
||||||
|
endpoint.refcount += 1;
|
||||||
|
slot.* = .{ .deadline = architecture.millis() + ms, .endpoint = endpoint, .owner = t.id };
|
||||||
|
return architecture.setSystemCallResult(state, 0);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
failErr(state, ipc.ENOSPC);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn systemProcessExitReason(state: *architecture.CpuState) void {
|
||||||
|
const t = scheduler.current();
|
||||||
|
if (t.aspace == 0) return fail(state);
|
||||||
|
const id = architecture.systemCallArg(state, 0);
|
||||||
|
if (id > std.math.maxInt(u32)) return failErr(state, ipc.ESRCH);
|
||||||
|
const r = exitReasonOf(t.id, @intCast(id));
|
||||||
|
architecture.setSystemCallResult(state, @bitCast(r));
|
||||||
|
}
|
||||||
|
|
||||||
/// Resolve `(device_id, resource_index)` to a GSI this process is entitled to bind, or null.
|
/// Resolve `(device_id, resource_index)` to a GSI this process is entitled to bind, or null.
|
||||||
/// The two checks are the whole security story: the device must be *claimed* by the
|
/// The two checks are the whole security story: the device must be *claimed* by the
|
||||||
/// caller, and the resource must be one of that device's `irq` resources as recorded
|
/// caller, and the resource must be one of that device's `irq` resources as recorded
|
||||||
|
|||||||
@@ -50,6 +50,17 @@ pub const Task = struct {
|
|||||||
// null. Holds its own reference, dropped when the notification is posted.
|
// null. Holds its own reference, dropped when the notification is posted.
|
||||||
// Opaque here for the same reason as `handles` below.
|
// Opaque here for the same reason as `handles` below.
|
||||||
exit_endpoint: ?*anyopaque = null,
|
exit_endpoint: ?*anyopaque = null,
|
||||||
|
// How this process ended — set by the death paths (exit, fault, kill) just
|
||||||
|
// before the reap records it for `process_exit_reason`. Meaningless while
|
||||||
|
// the task lives.
|
||||||
|
exit_reason: abi.ExitReason = .exited,
|
||||||
|
// Endpoint this process's signals arrive on (signal_bind), or null — same
|
||||||
|
// ownership rules as exit_endpoint (holds a reference; opaque here).
|
||||||
|
signal_endpoint: ?*anyopaque = null,
|
||||||
|
// Signals posted but not yet delivered: the coalescing pending mask
|
||||||
|
// (docs/process-lifecycle.md). Bits are abi.Signal values. Signals pend here
|
||||||
|
// until an endpoint is bound; two pending terminates are one terminate.
|
||||||
|
pending_signals: u32 = 0,
|
||||||
// Set by process_kill on a task that is running on another core; the kernel
|
// Set by process_kill on a task that is running on another core; the kernel
|
||||||
// finishes the kill at that task's next system call or timer tick.
|
// finishes the kill at that task's next system call or timer tick.
|
||||||
kill_pending: bool = false,
|
kill_pending: bool = false,
|
||||||
@@ -339,8 +350,9 @@ pub fn spawnUserLocked(aspace: u64, entry: u64, user_sp: u64, priority: Priority
|
|||||||
/// context switch and lock release.
|
/// context switch and lock release.
|
||||||
fn startUserTask() void {
|
fn startUserTask() void {
|
||||||
const t = current();
|
const t = current();
|
||||||
var buffer: [96]u8 = undefined;
|
// No serial chatter here: this runs on every spawn, unserialized against
|
||||||
architecture.serialWrite(std.fmt.bufPrint(&buffer, "DBG startUserTask ip=0x{x} sp=0x{x} aspace=0x{x} kstack=0x{x}\n", .{ t.user_ip, t.user_sp, t.aspace, t.kstack_top }) catch "");
|
// user-space writes, and its output used to shear concurrent log lines in
|
||||||
|
// half — the largest source of corrupted markers in the QEMU scenarios.
|
||||||
architecture.jumpToUser(t.user_ip, t.user_sp); // noreturn
|
architecture.jumpToUser(t.user_ip, t.user_sp); // noreturn
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -643,9 +655,15 @@ fn reapKillPendingLocked() void {
|
|||||||
/// other critical section, but releases it *without* touching the interrupt flag
|
/// other critical section, but releases it *without* touching the interrupt flag
|
||||||
/// — the handler's `iretq` restores the interrupted context's flags, so
|
/// — the handler's `iretq` restores the interrupted context's flags, so
|
||||||
/// re-enabling here would open a nested-interrupt window before the return.
|
/// re-enabling here would open a nested-interrupt window before the return.
|
||||||
|
/// Called from the tick with the big kernel lock held — process.zig hangs the
|
||||||
|
/// one-shot timer sweep here (timer_bind), the same call-up pattern as the
|
||||||
|
/// teardown hooks below.
|
||||||
|
pub var timer_tick_hook: ?*const fn () void = null;
|
||||||
|
|
||||||
pub fn tick() void {
|
pub fn tick() void {
|
||||||
_ = sync.enter();
|
_ = sync.enter();
|
||||||
wakeExpired();
|
wakeExpired();
|
||||||
|
if (timer_tick_hook) |hook| hook();
|
||||||
reapKillPendingLocked();
|
reapKillPendingLocked();
|
||||||
if (preemption_enabled) schedule();
|
if (preemption_enabled) schedule();
|
||||||
sync.leaveIsr();
|
sync.leaveIsr();
|
||||||
|
|||||||
+599
-25
@@ -132,6 +132,26 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
|
|||||||
processKillTest(boot_information);
|
processKillTest(boot_information);
|
||||||
} else if (eql(case, "supervision")) {
|
} else if (eql(case, "supervision")) {
|
||||||
supervisionTest(boot_information);
|
supervisionTest(boot_information);
|
||||||
|
} else if (eql(case, "claim-release")) {
|
||||||
|
claimReleaseTest(boot_information);
|
||||||
|
} else if (eql(case, "vfs-client-death")) {
|
||||||
|
vfsClientDeathTest(boot_information);
|
||||||
|
} else if (eql(case, "signals")) {
|
||||||
|
signalsTest(boot_information);
|
||||||
|
} else if (eql(case, "driver-restart")) {
|
||||||
|
driverRestartTest(boot_information);
|
||||||
|
} else if (eql(case, "usb-report")) {
|
||||||
|
usbReportTest(boot_information);
|
||||||
|
} else if (eql(case, "device-list")) {
|
||||||
|
deviceListTest(boot_information);
|
||||||
|
} else if (eql(case, "pci-scan")) {
|
||||||
|
pciScanTest(boot_information);
|
||||||
|
} else if (eql(case, "acpi-parse")) {
|
||||||
|
acpiParseTest(boot_information);
|
||||||
|
} else if (eql(case, "acpi-report")) {
|
||||||
|
acpiReportTest(boot_information);
|
||||||
|
} else if (eql(case, "acpi-ps2")) {
|
||||||
|
acpiReportTest(boot_information); // same spawn; the harness regex differs
|
||||||
} else if (eql(case, "initial-ramdisk")) {
|
} else if (eql(case, "initial-ramdisk")) {
|
||||||
initialRamdiskTest(boot_information);
|
initialRamdiskTest(boot_information);
|
||||||
} else if (eql(case, "vfs")) {
|
} else if (eql(case, "vfs")) {
|
||||||
@@ -256,20 +276,56 @@ fn discoveryTest() void {
|
|||||||
|
|
||||||
// M15: every PCI function now carries its own 4 KiB ECAM configuration space as
|
// M15: every PCI function now carries its own 4 KiB ECAM configuration space as
|
||||||
// resource 0 — the window a driver mmio_maps to walk its capability list (MSI etc).
|
// resource 0 — the window a driver mmio_maps to walk its capability list (MSI etc).
|
||||||
|
// M19.3: the kernel seeds only the bridge; functions arrive by the ring-3
|
||||||
|
// scan (proven equivalent in pci-scan before the walk retired).
|
||||||
var buffer: [64]device_abi.DeviceDescriptor = undefined;
|
var buffer: [64]device_abi.DeviceDescriptor = undefined;
|
||||||
const n = @min(devices_broker.enumerate(&buffer), buffer.len);
|
const n = @min(devices_broker.enumerate(&buffer), buffer.len);
|
||||||
var pci_functions: u32 = 0;
|
var bridges: u32 = 0;
|
||||||
var pci_config_ok = true;
|
var bridge_shape_ok = false;
|
||||||
|
for (buffer[0..n]) |d| {
|
||||||
|
if (d.class != @intFromEnum(device_abi.DeviceClass.pci_host_bridge)) continue;
|
||||||
|
bridges += 1;
|
||||||
|
var has_bus_range = false;
|
||||||
|
var has_io = false;
|
||||||
|
var memory_windows: u32 = 0;
|
||||||
|
for (d.resources[0..@intCast(d.resource_count)]) |resource| {
|
||||||
|
if (resource.kind == @intFromEnum(device_abi.ResourceKind.bus_range)) has_bus_range = true;
|
||||||
|
if (resource.kind == @intFromEnum(device_abi.ResourceKind.io_port)) has_io = true;
|
||||||
|
if (resource.kind == @intFromEnum(device_abi.ResourceKind.memory)) memory_windows += 1;
|
||||||
|
}
|
||||||
|
// ECAM plus at least one MMIO aperture, the bus range, the I/O window.
|
||||||
|
if (has_bus_range and has_io and memory_windows >= 2) bridge_shape_ok = true;
|
||||||
|
}
|
||||||
|
check("a PCI host bridge was seeded (MCFG)", bridges >= 1);
|
||||||
|
check("the bridge carries ECAM, apertures, bus range, and the I/O window", bridge_shape_ok);
|
||||||
|
|
||||||
|
// M19.0: every PCI memory resource (config slice and BARs alike) must be
|
||||||
|
// contained in one of its parent bridge's windows — the aperture derivation
|
||||||
|
// from the memory map is what makes a future user-space device_register of
|
||||||
|
// these functions pass containment. This is the assert that catches a
|
||||||
|
// too-coarse hole computation before M19.2 would.
|
||||||
|
var bars_contained = true;
|
||||||
for (buffer[0..n]) |d| {
|
for (buffer[0..n]) |d| {
|
||||||
if (d.class != @intFromEnum(device_abi.DeviceClass.pci_device)) continue;
|
if (d.class != @intFromEnum(device_abi.DeviceClass.pci_device)) continue;
|
||||||
pci_functions += 1;
|
if (d.parent >= n) {
|
||||||
const has_config = d.resource_count >= 1 and
|
bars_contained = false;
|
||||||
d.resources[0].kind == @intFromEnum(device_abi.ResourceKind.memory) and
|
continue;
|
||||||
d.resources[0].len == abi.page_size;
|
|
||||||
if (!has_config) pci_config_ok = false;
|
|
||||||
}
|
}
|
||||||
check("PCI functions were enumerated (MCFG/ECAM)", pci_functions >= 1);
|
const bridge = buffer[@intCast(d.parent)];
|
||||||
check("each PCI function exposes its ECAM config space as resource 0", pci_config_ok);
|
for (d.resources[0..@intCast(d.resource_count)]) |r| {
|
||||||
|
if (r.kind != @intFromEnum(device_abi.ResourceKind.memory)) continue;
|
||||||
|
var inside = false;
|
||||||
|
for (bridge.resources[0..@intCast(bridge.resource_count)]) |w| {
|
||||||
|
if (w.kind != @intFromEnum(device_abi.ResourceKind.memory)) continue;
|
||||||
|
if (r.start >= w.start and r.start + r.len <= w.start + w.len) inside = true;
|
||||||
|
}
|
||||||
|
if (!inside) {
|
||||||
|
bars_contained = false;
|
||||||
|
log(" escaping BAR: 0x{x}+0x{x} on device {d}\n", .{ r.start, r.len, d.id });
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
check("every PCI BAR lies inside a bridge aperture (M19.0)", bars_contained);
|
||||||
|
|
||||||
result();
|
result();
|
||||||
}
|
}
|
||||||
@@ -1080,12 +1136,18 @@ fn ioPortTest() void {
|
|||||||
var buffer: [64]device_abi.DeviceDescriptor = undefined;
|
var buffer: [64]device_abi.DeviceDescriptor = undefined;
|
||||||
const n = @min(devices_broker.enumerate(&buffer), buffer.len);
|
const n = @min(devices_broker.enumerate(&buffer), buffer.len);
|
||||||
|
|
||||||
|
// Post-M20.3 the PS/2 node is registered at runtime by the ring-3 acpi
|
||||||
|
// service, so it is absent from this boot snapshot. Exercise the same
|
||||||
|
// io_port claim/resolve mechanism against the acpi-tables node's broad I/O
|
||||||
|
// grant — the window that now carries port authority (the service uses it
|
||||||
|
// for exactly this). The PS/2 status port 0x64 is offset 0x64 within it.
|
||||||
var found_id: ?u64 = null;
|
var found_id: ?u64 = null;
|
||||||
var found_res: u64 = 0;
|
var found_res: u64 = 0;
|
||||||
outer: for (buffer[0..n]) |d| {
|
outer: for (buffer[0..n]) |d| {
|
||||||
|
if (d.class != @intFromEnum(device_abi.DeviceClass.acpi_tables)) continue;
|
||||||
for (0..d.resource_count) |ri| {
|
for (0..d.resource_count) |ri| {
|
||||||
const r = d.resources[ri];
|
const r = d.resources[ri];
|
||||||
if (r.kind == @intFromEnum(device_abi.ResourceKind.io_port) and r.start == 0x64 and r.len >= 1) {
|
if (r.kind == @intFromEnum(device_abi.ResourceKind.io_port) and r.start == 0 and r.len > 0x64) {
|
||||||
found_id = d.id;
|
found_id = d.id;
|
||||||
found_res = ri;
|
found_res = ri;
|
||||||
break :outer;
|
break :outer;
|
||||||
@@ -1093,18 +1155,18 @@ fn ioPortTest() void {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
const id = found_id orelse {
|
const id = found_id orelse {
|
||||||
check("discovered the PS/2 status port (io_port 0x64)", false);
|
check("discovered the acpi-tables I/O window", false);
|
||||||
result();
|
result();
|
||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
check("discovered the PS/2 status port (io_port 0x64)", true);
|
check("discovered the acpi-tables I/O window", true);
|
||||||
|
|
||||||
const me = scheduler.current();
|
const me = scheduler.current();
|
||||||
check("claimed the io_port device", devices_broker.claim(id, me.id));
|
check("claimed the io_port device", devices_broker.claim(id, me.id));
|
||||||
check("an in-range access resolves to port 0x64", process.resolveIoPort(me, id, found_res, 0, 1) == 0x64);
|
check("an in-range access resolves to port 0x64", process.resolveIoPort(me, id, found_res, 0x64, 1) == 0x64);
|
||||||
check("an over-wide access is refused", process.resolveIoPort(me, id, found_res, 0, 2) == null);
|
check("a 4-byte access at the last port is refused", process.resolveIoPort(me, id, found_res, 0xFFFF, 4) == null);
|
||||||
check("an out-of-range offset is refused", process.resolveIoPort(me, id, found_res, 1, 1) == null);
|
check("an out-of-range offset is refused", process.resolveIoPort(me, id, found_res, 0x10000, 1) == null);
|
||||||
check("an unclaimed device id is refused", process.resolveIoPort(me, 0xDEAD_BEEF, found_res, 0, 1) == null);
|
check("an unclaimed device id is refused", process.resolveIoPort(me, 0xDEAD_BEEF, found_res, 0x64, 1) == null);
|
||||||
|
|
||||||
// The kernel actually issues the `in`. Reaching this line at all proves it didn't
|
// The kernel actually issues the `in`. Reaching this line at all proves it didn't
|
||||||
// fault; a width-1 read must return a single byte.
|
// fault; a width-1 read must return a single byte.
|
||||||
@@ -1207,15 +1269,15 @@ fn userPfTest() void {
|
|||||||
/// hand — address space, code page RO+X, stack page RW+NX — because the blob is a
|
/// hand — address space, code page RO+X, stack page RW+NX — because the blob is a
|
||||||
/// raw code fragment, not an ELF `spawnProcess` could load. Returns false if any
|
/// raw code fragment, not an ELF `spawnProcess` could load. Returns false if any
|
||||||
/// allocation fails.
|
/// allocation fails.
|
||||||
fn spawnFaultingProcess() bool {
|
fn spawnFaultingProcess() ?u32 {
|
||||||
const blob = process.pfBlob();
|
const blob = process.pfBlob();
|
||||||
const flags = sync.enter();
|
const flags = sync.enter();
|
||||||
defer sync.leave(flags);
|
defer sync.leave(flags);
|
||||||
|
|
||||||
const aspace = architecture.createAddressSpace() orelse return false;
|
const aspace = architecture.createAddressSpace() orelse return null;
|
||||||
const code_frame = pmm.alloc() orelse {
|
const code_frame = pmm.alloc() orelse {
|
||||||
architecture.destroyAddressSpace(aspace);
|
architecture.destroyAddressSpace(aspace);
|
||||||
return false;
|
return null;
|
||||||
};
|
};
|
||||||
// Fill through the physmap (the user mapping is read-only); pad with int3 so a
|
// Fill through the physmap (the user mapping is read-only); pad with int3 so a
|
||||||
// stray jump traps instead of sliding.
|
// stray jump traps instead of sliding.
|
||||||
@@ -1226,15 +1288,16 @@ fn spawnFaultingProcess() bool {
|
|||||||
|
|
||||||
const stack_frame = pmm.alloc() orelse {
|
const stack_frame = pmm.alloc() orelse {
|
||||||
architecture.destroyAddressSpace(aspace); // frees code_frame too — it's mapped
|
architecture.destroyAddressSpace(aspace); // frees code_frame too — it's mapped
|
||||||
return false;
|
return null;
|
||||||
};
|
};
|
||||||
architecture.mapUserPageInto(aspace, process.stack_base_virtual, stack_frame, true, false); // RW + NX
|
architecture.mapUserPageInto(aspace, process.stack_base_virtual, stack_frame, true, false); // RW + NX
|
||||||
|
|
||||||
if (scheduler.spawnUserLocked(aspace, process.code_virtual, process.stack_base_virtual + abi.page_size, 4, "fault-probe", 0, null) == null) {
|
// Supervised by the calling test task, so exitReasonOf can read the verdict.
|
||||||
|
const id = scheduler.spawnUserLocked(aspace, process.code_virtual, process.stack_base_virtual + abi.page_size, 4, "fault-probe", scheduler.currentId(), null) orelse {
|
||||||
architecture.destroyAddressSpace(aspace);
|
architecture.destroyAddressSpace(aspace);
|
||||||
return false;
|
return null;
|
||||||
}
|
};
|
||||||
return true;
|
return id;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Fault recovery (docs/resilience.md step 2): a scheduled ring-3 process that
|
/// Fault recovery (docs/resilience.md step 2): a scheduled ring-3 process that
|
||||||
@@ -1263,7 +1326,8 @@ fn faultRecoveryTest(boot_information: *const BootInformation) void {
|
|||||||
scheduler.setPriority(4);
|
scheduler.setPriority(4);
|
||||||
check("init heartbeat before the fault", process.write_count >= 1);
|
check("init heartbeat before the fault", process.write_count >= 1);
|
||||||
|
|
||||||
check("faulting process spawned", spawnFaultingProcess());
|
const probe = spawnFaultingProcess() orelse 0;
|
||||||
|
check("faulting process spawned", probe != 0);
|
||||||
|
|
||||||
// The kill: the faulting process #PFs on its first instruction and the kernel
|
// The kill: the faulting process #PFs on its first instruction and the kernel
|
||||||
// reaps it instead of halting.
|
// reaps it instead of halting.
|
||||||
@@ -1272,6 +1336,7 @@ fn faultRecoveryTest(boot_information: *const BootInformation) void {
|
|||||||
while (process.fault_kill_count < 1 and architecture.millis() < deadline) scheduler.yield();
|
while (process.fault_kill_count < 1 and architecture.millis() < deadline) scheduler.yield();
|
||||||
scheduler.setPriority(4);
|
scheduler.setPriority(4);
|
||||||
check("faulting process was killed (not the machine)", process.fault_kill_count == 1);
|
check("faulting process was killed (not the machine)", process.fault_kill_count == 1);
|
||||||
|
check("the probe's reason reads segmentation_fault", process.exitReasonOf(scheduler.currentId(), probe) == @intFromEnum(abi.ExitReason.segmentation_fault));
|
||||||
|
|
||||||
// Life after the kill: init must keep beating on the same core.
|
// Life after the kill: init must keep beating on the same core.
|
||||||
const beats_at_kill = process.write_count;
|
const beats_at_kill = process.write_count;
|
||||||
@@ -1423,6 +1488,12 @@ fn processKillTest(boot_information: *const BootInformation) void {
|
|||||||
check("the sleeper's exit notification arrived (length 0)", r == 0);
|
check("the sleeper's exit notification arrived (length 0)", r == 0);
|
||||||
check("its badge carries the exit bit and the child id", badge == abi.notify_badge_bit | abi.notify_exit_bit | sleeper);
|
check("its badge carries the exit bit and the child id", badge == abi.notify_badge_bit | abi.notify_exit_bit | sleeper);
|
||||||
|
|
||||||
|
// M17.2: the recorded reason — the notification is the fence, so it is
|
||||||
|
// already readable, and gated by the same supervisor check as the kill.
|
||||||
|
check("the sleeper's reason reads killed", process.exitReasonOf(me, sleeper) == @intFromEnum(abi.ExitReason.killed));
|
||||||
|
check("a non-supervisor may not read the reason (-EPERM)", process.exitReasonOf(me + 12345, sleeper) == -ipcsync.EPERM);
|
||||||
|
check("an unknown id has no reason (-ESRCH)", process.exitReasonOf(me, 0xFFFF_FF00) == -ipcsync.ESRCH);
|
||||||
|
|
||||||
const beats_at_kill = process.write_count;
|
const beats_at_kill = process.write_count;
|
||||||
scheduler.sleep(1500); // more than one heartbeat period
|
scheduler.sleep(1500); // more than one heartbeat period
|
||||||
check("the heartbeat stopped with the kill", process.write_count == beats_at_kill);
|
check("the heartbeat stopped with the kill", process.write_count == beats_at_kill);
|
||||||
@@ -1443,6 +1514,21 @@ fn processKillTest(boot_information: *const BootInformation) void {
|
|||||||
check("the spinner's exit notification arrived (length 0)", r == 0);
|
check("the spinner's exit notification arrived (length 0)", r == 0);
|
||||||
check("its badge carries the exit bit and the child id", badge == abi.notify_badge_bit | abi.notify_exit_bit | spinner);
|
check("its badge carries the exit bit and the child id", badge == abi.notify_badge_bit | abi.notify_exit_bit | spinner);
|
||||||
|
|
||||||
|
// M17.2: a child that ends on its own must read exited, not killed —
|
||||||
|
// args-echo with arguments echoes once and returns from main.
|
||||||
|
var clean: u32 = 0;
|
||||||
|
i = 0;
|
||||||
|
while (i < rd.count) : (i += 1) {
|
||||||
|
const item = rd.entry(i) orelse continue;
|
||||||
|
if (!eql(item.name, "args-echo")) continue;
|
||||||
|
clean = process.spawnProcessSupervised(item.blob, 4, &.{ "args-echo", "clean-exit" }, me, endpoint) catch 0;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
check("args-echo spawned as the clean-exit child", clean != 0);
|
||||||
|
r = ipcsync.replyWait(endpoint, 0, 0, 0, 0, abi.no_cap, &badge, &received_cap);
|
||||||
|
check("the clean child's exit notification arrived", badge == abi.notify_badge_bit | abi.notify_exit_bit | clean);
|
||||||
|
check("the clean child's reason reads exited", process.exitReasonOf(me, clean) == @intFromEnum(abi.ExitReason.exited));
|
||||||
|
|
||||||
var table: [32]abi.ProcessDescriptor = undefined;
|
var table: [32]abi.ProcessDescriptor = undefined;
|
||||||
const total = scheduler.enumerate(&table);
|
const total = scheduler.enumerate(&table);
|
||||||
var still_listed = false;
|
var still_listed = false;
|
||||||
@@ -1454,6 +1540,465 @@ fn processKillTest(boot_information: *const BootInformation) void {
|
|||||||
result();
|
result();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// M17.1: a dead process's device claims are released by the reap, so a restarted
|
||||||
|
/// driver can claim its hardware again (docs/process-lifecycle.md iron rule 1).
|
||||||
|
/// First the broker release in isolation — two owners, one released, the other's
|
||||||
|
/// claim must survive. Then the death-path wiring with a real child: the claim is
|
||||||
|
/// made on the child's behalf (the broker is kernel-callable), the child is
|
||||||
|
/// killed, and once the exit notification arrives — posted last, after release —
|
||||||
|
/// the device must be unclaimed and claimable again.
|
||||||
|
fn claimReleaseTest(boot_information: *const BootInformation) void {
|
||||||
|
log("DANOS-TEST-BEGIN: claim-release\n", .{});
|
||||||
|
|
||||||
|
var buffer: [2]device_abi.DeviceDescriptor = undefined;
|
||||||
|
const total = devices_broker.enumerate(&buffer);
|
||||||
|
check("the device tree is seeded (>= 2 devices)", total >= 2);
|
||||||
|
if (total < 2) {
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
// The broker release in isolation.
|
||||||
|
check("device 0 claimed by owner 111", devices_broker.claim(0, 111));
|
||||||
|
check("device 1 claimed by owner 222", devices_broker.claim(1, 222));
|
||||||
|
devices_broker.releaseAllOwnedBy(111);
|
||||||
|
check("owner 111's claim is released", devices_broker.ownerOf(0) == null);
|
||||||
|
check("owner 222's claim survives", (devices_broker.ownerOf(1) orelse 0) == 222);
|
||||||
|
devices_broker.releaseAllOwnedBy(222);
|
||||||
|
check("cleanup released owner 222", devices_broker.ownerOf(1) == null);
|
||||||
|
|
||||||
|
// The death-path wiring: a real process dies holding a claim.
|
||||||
|
check("bootloader handed over /system/services/init", boot_information.init_len != 0);
|
||||||
|
if (boot_information.init_len == 0) {
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.init_base)))[0..boot_information.init_len];
|
||||||
|
const me = scheduler.currentId();
|
||||||
|
const endpoint = ipcsync.createIpcEndpoint() orelse {
|
||||||
|
check("exit endpoint allocated", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
const child = process.spawnProcessSupervised(image, 4, &.{"/system/services/init"}, me, endpoint) catch 0;
|
||||||
|
check("supervised child spawned", child != 0);
|
||||||
|
check("device 0 claimed on the child's behalf", devices_broker.claim(0, child));
|
||||||
|
|
||||||
|
check("the kill is accepted", process.killProcess(me, child) == 0);
|
||||||
|
var badge: u64 = 0;
|
||||||
|
var received_cap: u64 = 0;
|
||||||
|
_ = ipcsync.replyWait(endpoint, 0, 0, 0, 0, abi.no_cap, &badge, &received_cap);
|
||||||
|
check("the exit notification arrived", badge == abi.notify_badge_bit | abi.notify_exit_bit | child);
|
||||||
|
check("death released the child's claim", devices_broker.ownerOf(0) == null);
|
||||||
|
check("the device is claimable again", devices_broker.claim(0, me));
|
||||||
|
devices_broker.releaseAllOwnedBy(me);
|
||||||
|
result();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// M17.3: the published exit events, proven by their first subscriber. The VFS
|
||||||
|
/// subscribes at startup; a client opens a file and parks holding the handle;
|
||||||
|
/// the kill posts the exit event to the VFS's endpoint; the VFS releases the
|
||||||
|
/// dead client's handle and says so — the service-side mirror of iron rule 1
|
||||||
|
/// (a service must never depend on clients cleaning up after themselves).
|
||||||
|
fn vfsClientDeathTest(boot_information: *const BootInformation) void {
|
||||||
|
log("DANOS-TEST-BEGIN: vfs-client-death\n", .{});
|
||||||
|
if (boot_information.initial_ramdisk_len == 0) {
|
||||||
|
check("bootloader handed over an initial_ramdisk", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||||
|
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||||
|
check("initial_ramdisk image is valid", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
|
||||||
|
process.write_count = 0;
|
||||||
|
check("vfs spawned", spawnNamed(rd, "vfs"));
|
||||||
|
|
||||||
|
const me = scheduler.currentId();
|
||||||
|
const endpoint = ipcsync.createIpcEndpoint() orelse {
|
||||||
|
check("exit endpoint allocated", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
var client: u32 = 0;
|
||||||
|
var i: u32 = 0;
|
||||||
|
while (i < rd.count) : (i += 1) {
|
||||||
|
const item = rd.entry(i) orelse continue;
|
||||||
|
if (!eql(item.name, "vfs-test")) continue;
|
||||||
|
client = process.spawnProcessSupervised(item.blob, 4, &.{ "vfs-test", "park" }, me, endpoint) catch 0;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
check("parked client spawned (supervised)", client != 0);
|
||||||
|
|
||||||
|
// Its heartbeat is the fence: once it beats, the handle is open.
|
||||||
|
const parked = "vfstest: parked";
|
||||||
|
scheduler.setPriority(1);
|
||||||
|
var deadline = architecture.millis() + 10000;
|
||||||
|
while (architecture.millis() < deadline) {
|
||||||
|
if (process.write_len >= parked.len and eql(process.write_buffer[0..parked.len], parked)) break;
|
||||||
|
scheduler.yield();
|
||||||
|
}
|
||||||
|
scheduler.setPriority(4);
|
||||||
|
check("client parked holding an open handle", process.write_len >= parked.len and eql(process.write_buffer[0..parked.len], parked));
|
||||||
|
|
||||||
|
check("the kill is accepted", process.killProcess(me, client) == 0);
|
||||||
|
var badge: u64 = 0;
|
||||||
|
var received_cap: u64 = 0;
|
||||||
|
_ = ipcsync.replyWait(endpoint, 0, 0, 0, 0, abi.no_cap, &badge, &received_cap);
|
||||||
|
check("the exit notification arrived", badge == abi.notify_badge_bit | abi.notify_exit_bit | client);
|
||||||
|
|
||||||
|
// The VFS heard the same published event; its release line is the proof.
|
||||||
|
const released = "vfs: released 1 handle(s) for dead client";
|
||||||
|
scheduler.setPriority(1);
|
||||||
|
deadline = architecture.millis() + 10000;
|
||||||
|
while (architecture.millis() < deadline) {
|
||||||
|
if (process.write_len >= released.len and eql(process.write_buffer[0..released.len], released)) break;
|
||||||
|
scheduler.yield();
|
||||||
|
}
|
||||||
|
scheduler.setPriority(4);
|
||||||
|
check("the VFS released the dead client's handle", process.write_len >= released.len and eql(process.write_buffer[0..released.len], released));
|
||||||
|
result();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// M17.4 from ring 3: process-test's signal-run role drives the whole lifecycle
|
||||||
|
/// surface — the zero-length ping (answered by the harness), signals as
|
||||||
|
/// statements (reload logged, terminate = clean exit), the one-shot timer, and
|
||||||
|
/// both endings of the stop sequence (polite -> exited, deaf -> killed at the
|
||||||
|
/// deadline). Its "process-test: signals ok" is the pass marker.
|
||||||
|
fn signalsTest(boot_information: *const BootInformation) void {
|
||||||
|
log("DANOS-TEST-BEGIN: signals\n", .{});
|
||||||
|
if (boot_information.initial_ramdisk_len == 0) {
|
||||||
|
check("bootloader handed over an initial_ramdisk", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||||
|
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||||
|
check("initial_ramdisk image is valid", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
|
||||||
|
process.setInitialRamdisk(image); // the parent system_spawns its children by name
|
||||||
|
process.write_count = 0;
|
||||||
|
var runner: u32 = 0;
|
||||||
|
var i: u32 = 0;
|
||||||
|
while (i < rd.count) : (i += 1) {
|
||||||
|
const item = rd.entry(i) orelse continue;
|
||||||
|
if (!eql(item.name, "process-test")) continue;
|
||||||
|
runner = process.spawnProcessSupervised(item.blob, 4, &.{ "process-test", "signal-run" }, scheduler.currentId(), null) catch 0;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
check("signal-run parent spawned", runner != 0);
|
||||||
|
|
||||||
|
const pass_marker = "process-test: signals ok";
|
||||||
|
const fail_marker = "process-test: FAIL";
|
||||||
|
scheduler.setPriority(1);
|
||||||
|
const deadline = architecture.millis() + 15000;
|
||||||
|
var saw_pass = false;
|
||||||
|
var saw_fail = false;
|
||||||
|
while (architecture.millis() < deadline and !saw_pass and !saw_fail) {
|
||||||
|
if (process.write_len >= pass_marker.len and eql(process.write_buffer[0..pass_marker.len], pass_marker)) saw_pass = true;
|
||||||
|
if (process.write_len >= fail_marker.len and eql(process.write_buffer[0..fail_marker.len], fail_marker)) saw_fail = true;
|
||||||
|
scheduler.yield();
|
||||||
|
}
|
||||||
|
scheduler.setPriority(4);
|
||||||
|
check("the signal-run parent reported ok", saw_pass and !saw_fail);
|
||||||
|
result();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// M18.1: the device manager's restart machinery, end to end. In test-restart
|
||||||
|
/// mode the manager also supervises crash-test: a fixture that claims device 0,
|
||||||
|
/// hellos, and faults. The scenario asserts three markers in order — the real
|
||||||
|
/// xHCI driver hellos clean and stays; crash-test is restarted with backoff
|
||||||
|
/// (each respawn re-claiming the device the dead instance held, M17.1 through
|
||||||
|
/// the manager's path); the crash loop caps and the manager gives up.
|
||||||
|
fn driverRestartTest(boot_information: *const BootInformation) void {
|
||||||
|
log("DANOS-TEST-BEGIN: driver-restart\n", .{});
|
||||||
|
if (boot_information.initial_ramdisk_len == 0) {
|
||||||
|
check("bootloader handed over an initial_ramdisk", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||||
|
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||||
|
check("initial_ramdisk image is valid", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
|
||||||
|
process.setInitialRamdisk(image); // the manager system_spawns drivers by name
|
||||||
|
process.write_count = 0;
|
||||||
|
var manager: u32 = 0;
|
||||||
|
var i: u32 = 0;
|
||||||
|
while (i < rd.count) : (i += 1) {
|
||||||
|
const item = rd.entry(i) orelse continue;
|
||||||
|
if (!eql(item.name, "device-manager")) continue;
|
||||||
|
manager = process.spawnProcessSupervised(item.blob, 4, &.{ "device-manager", "test-restart" }, scheduler.currentId(), null) catch 0;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
check("device-manager spawned in test-restart mode", manager != 0);
|
||||||
|
// The assertions live in the harness: its expect regex requires, in order,
|
||||||
|
// the xHCI hello ack, a crash-test restart, and the crash-loop cap — read
|
||||||
|
// from the whole serial capture, immune to the transient-line races a
|
||||||
|
// write_buffer poll would have here (many processes log concurrently).
|
||||||
|
result();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// M18.2: bus tree reports, end to end. The manager (test-usb-restart mode)
|
||||||
|
/// spawns the xHCI driver; the driver maps its BAR, scans the root-hub ports,
|
||||||
|
/// and reports the two QEMU devices; the manager mirrors them, kills the
|
||||||
|
/// reporter (the test trigger), prunes both children, restarts the driver with
|
||||||
|
/// backoff, and the respawned instance re-claims, re-scans, and re-reports.
|
||||||
|
/// The harness's ordered expect regex is the assertion; this test only
|
||||||
|
/// orchestrates the spawn.
|
||||||
|
fn usbReportTest(boot_information: *const BootInformation) void {
|
||||||
|
log("DANOS-TEST-BEGIN: usb-report\n", .{});
|
||||||
|
if (boot_information.initial_ramdisk_len == 0) {
|
||||||
|
check("bootloader handed over an initial_ramdisk", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||||
|
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||||
|
check("initial_ramdisk image is valid", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
|
||||||
|
process.setInitialRamdisk(image);
|
||||||
|
var manager: u32 = 0;
|
||||||
|
var i: u32 = 0;
|
||||||
|
while (i < rd.count) : (i += 1) {
|
||||||
|
const item = rd.entry(i) orelse continue;
|
||||||
|
if (!eql(item.name, "device-manager")) continue;
|
||||||
|
manager = process.spawnProcessSupervised(item.blob, 4, &.{ "device-manager", "test-usb-restart" }, scheduler.currentId(), null) catch 0;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
check("device-manager spawned in test-usb-restart mode", manager != 0);
|
||||||
|
result();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// M18.3: the application surface. device-list enumerates the manager's tree
|
||||||
|
/// over IPC, subscribes with its endpoint as a capability, and prints every
|
||||||
|
/// published event; the manager's delayed test-kill of the reporter produces a
|
||||||
|
/// removed/added storm the subscriber must observe. The harness's ordered
|
||||||
|
/// expect regex is the assertion.
|
||||||
|
fn deviceListTest(boot_information: *const BootInformation) void {
|
||||||
|
log("DANOS-TEST-BEGIN: device-list\n", .{});
|
||||||
|
if (boot_information.initial_ramdisk_len == 0) {
|
||||||
|
check("bootloader handed over an initial_ramdisk", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||||
|
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||||
|
check("initial_ramdisk image is valid", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
|
||||||
|
process.setInitialRamdisk(image);
|
||||||
|
var manager: u32 = 0;
|
||||||
|
var i: u32 = 0;
|
||||||
|
while (i < rd.count) : (i += 1) {
|
||||||
|
const item = rd.entry(i) orelse continue;
|
||||||
|
if (!eql(item.name, "device-manager")) continue;
|
||||||
|
manager = process.spawnProcessSupervised(item.blob, 4, &.{ "device-manager", "test-usb-restart" }, scheduler.currentId(), null) catch 0;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
check("device-manager spawned in test-usb-restart mode", manager != 0);
|
||||||
|
check("device-list spawned", spawnNamed(rd, "device-list"));
|
||||||
|
result();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// M19.1: the ring-3 PCI scan agrees with the kernel's. The manager spawns
|
||||||
|
/// pci-bus for the host bridge; the driver walks the same ECAM window through
|
||||||
|
/// its mmio_map grant and must find exactly the functions the kernel's own
|
||||||
|
/// enumeration recorded — the equivalence that licenses retiring the kernel
|
||||||
|
/// walk in M19.3.
|
||||||
|
fn pciScanTest(boot_information: *const BootInformation) void {
|
||||||
|
log("DANOS-TEST-BEGIN: pci-scan\n", .{});
|
||||||
|
if (boot_information.initial_ramdisk_len == 0) {
|
||||||
|
check("bootloader handed over an initial_ramdisk", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||||
|
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||||
|
check("initial_ramdisk image is valid", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
|
||||||
|
// Post-flip (M19.3) ground truth: the kernel no longer enumerates PCI
|
||||||
|
// functions, so equivalence inverts — the broker's function count after
|
||||||
|
// the scan must equal what the driver itself reported finding.
|
||||||
|
var buffer: [64]device_abi.DeviceDescriptor = undefined;
|
||||||
|
const n = @min(devices_broker.enumerate(&buffer), buffer.len);
|
||||||
|
var boot_pci: u32 = 0;
|
||||||
|
for (buffer[0..n]) |d| {
|
||||||
|
if (d.class == @intFromEnum(device_abi.DeviceClass.pci_device)) boot_pci += 1;
|
||||||
|
}
|
||||||
|
check("the kernel seeded no PCI functions (the walk retired)", boot_pci == 0);
|
||||||
|
|
||||||
|
process.setInitialRamdisk(image);
|
||||||
|
process.write_count = 0;
|
||||||
|
var manager: u32 = 0;
|
||||||
|
var i: u32 = 0;
|
||||||
|
while (i < rd.count) : (i += 1) {
|
||||||
|
const item = rd.entry(i) orelse continue;
|
||||||
|
if (!eql(item.name, "device-manager")) continue;
|
||||||
|
manager = process.spawnProcessSupervised(item.blob, 4, &.{ "device-manager", "test-pci-restart" }, scheduler.currentId(), null) catch 0;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
check("device-manager spawned (test-pci-restart mode)", manager != 0);
|
||||||
|
|
||||||
|
// First scan: wait for the driver's count line and parse the number.
|
||||||
|
const count_prefix = "pci-bus: ";
|
||||||
|
const count_suffix = " functions found";
|
||||||
|
var reported: u32 = 0;
|
||||||
|
scheduler.setPriority(1);
|
||||||
|
var deadline = architecture.millis() + 15000;
|
||||||
|
while (architecture.millis() < deadline and reported == 0) {
|
||||||
|
if (process.write_len > count_prefix.len + count_suffix.len and eql(process.write_buffer[0..count_prefix.len], count_prefix)) {
|
||||||
|
const line = process.write_buffer[0..process.write_len];
|
||||||
|
const digits_end = std.mem.indexOf(u8, line, count_suffix) orelse {
|
||||||
|
scheduler.yield();
|
||||||
|
continue;
|
||||||
|
};
|
||||||
|
reported = std.fmt.parseInt(u32, line[count_prefix.len..digits_end], 10) catch 0;
|
||||||
|
}
|
||||||
|
scheduler.yield();
|
||||||
|
}
|
||||||
|
scheduler.setPriority(4);
|
||||||
|
check("the ring-3 scan reported a function count", reported >= 1);
|
||||||
|
|
||||||
|
// Every reported function was registered: the broker holds exactly them.
|
||||||
|
var registered: [64]device_abi.DeviceDescriptor = undefined;
|
||||||
|
const r = @min(devices_broker.enumerate(®istered), registered.len);
|
||||||
|
var registered_pci: u32 = 0;
|
||||||
|
for (registered[0..r]) |d| {
|
||||||
|
if (d.class == @intFromEnum(device_abi.DeviceClass.pci_device)) registered_pci += 1;
|
||||||
|
}
|
||||||
|
check("the broker holds exactly the reported functions", registered_pci == reported);
|
||||||
|
const kernel_count = reported; // the no-duplicate check below reuses it
|
||||||
|
|
||||||
|
// The restart drill: the manager kills pci-bus after its reports; the
|
||||||
|
// respawn re-claims, re-scans, and re-registers.
|
||||||
|
const restart_marker = "device-manager: restarting pci-bus";
|
||||||
|
scheduler.setPriority(1);
|
||||||
|
deadline = architecture.millis() + 15000;
|
||||||
|
var restarted = false;
|
||||||
|
while (architecture.millis() < deadline and !restarted) {
|
||||||
|
if (process.write_len >= restart_marker.len and eql(process.write_buffer[0..restart_marker.len], restart_marker)) restarted = true;
|
||||||
|
scheduler.yield();
|
||||||
|
}
|
||||||
|
scheduler.setPriority(4);
|
||||||
|
check("the manager restarted pci-bus", restarted);
|
||||||
|
|
||||||
|
var marker_buffer: [48]u8 = undefined;
|
||||||
|
const marker = std.fmt.bufPrint(&marker_buffer, "pci-bus: {d} functions found", .{reported}) catch "";
|
||||||
|
scheduler.setPriority(1);
|
||||||
|
deadline = architecture.millis() + 15000;
|
||||||
|
var seen = false;
|
||||||
|
while (architecture.millis() < deadline and !seen) {
|
||||||
|
if (process.write_len >= marker.len and eql(process.write_buffer[0..marker.len], marker)) seen = true;
|
||||||
|
scheduler.yield();
|
||||||
|
}
|
||||||
|
scheduler.setPriority(4);
|
||||||
|
check("the respawned scan reported the same count", seen);
|
||||||
|
|
||||||
|
// No duplicates: the registrations deduped against the kernel's own nodes
|
||||||
|
// on the first pass, and against themselves on the second.
|
||||||
|
var after: [64]device_abi.DeviceDescriptor = undefined;
|
||||||
|
const m = @min(devices_broker.enumerate(&after), after.len);
|
||||||
|
var after_count: u32 = 0;
|
||||||
|
for (after[0..m]) |d| {
|
||||||
|
if (d.class == @intFromEnum(device_abi.DeviceClass.pci_device)) after_count += 1;
|
||||||
|
}
|
||||||
|
check("no duplicate PCI nodes after register + restart + re-register", after_count == kernel_count);
|
||||||
|
result();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// M20.2: the acpi service registers + reports its _HID devices. Boot normally
|
||||||
|
/// (the manager spawns discovery); the harness's expect regex requires the two
|
||||||
|
/// PS/2 nodes among the service's report lines, each with its _CRS resources —
|
||||||
|
/// the ring-3 _CRS/_STA evaluation working end to end. The kernel test only
|
||||||
|
/// starts the manager.
|
||||||
|
fn acpiReportTest(boot_information: *const BootInformation) void {
|
||||||
|
log("DANOS-TEST-BEGIN: acpi-report\n", .{});
|
||||||
|
if (boot_information.initial_ramdisk_len == 0) {
|
||||||
|
check("bootloader handed over an initial_ramdisk", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||||
|
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||||
|
check("initial_ramdisk image is valid", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
process.setInitialRamdisk(image);
|
||||||
|
var spawned = false;
|
||||||
|
var i: u32 = 0;
|
||||||
|
while (i < rd.count) : (i += 1) {
|
||||||
|
const item = rd.entry(i) orelse continue;
|
||||||
|
if (!eql(item.name, "device-manager")) continue;
|
||||||
|
_ = process.spawnProcessSupervised(item.blob, 4, &.{"device-manager"}, scheduler.currentId(), null) catch 0;
|
||||||
|
spawned = true;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
check("device-manager spawned", spawned);
|
||||||
|
result();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// M20.1: the ring-3 AML parse agrees with the kernel's. The manager spawns
|
||||||
|
/// the discovery service (the acpi build variant); it claims the acpi-tables
|
||||||
|
/// node, maps the blobs, parses them, and logs its Device count — which must
|
||||||
|
/// equal what the kernel's own parse produced (the equivalence that licenses
|
||||||
|
/// retiring the kernel's device build in M20.3).
|
||||||
|
fn acpiParseTest(boot_information: *const BootInformation) void {
|
||||||
|
log("DANOS-TEST-BEGIN: acpi-parse\n", .{});
|
||||||
|
if (boot_information.initial_ramdisk_len == 0) {
|
||||||
|
check("bootloader handed over an initial_ramdisk", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||||
|
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||||
|
check("initial_ramdisk image is valid", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
|
||||||
|
// The kernel's own count, from the namespace it already built for \_S5.
|
||||||
|
const kernel_devices = platform.amlDeviceCount();
|
||||||
|
check("the kernel namespace has devices to compare against", kernel_devices >= 1);
|
||||||
|
|
||||||
|
// Spawn the discovery service directly with that count as argv: it parses
|
||||||
|
// the same blobs in ring 3 and self-verifies, printing "acpi-parse: ok" iff
|
||||||
|
// the counts match. The harness's expect regex is that marker — deterministic,
|
||||||
|
// no racing the shared serial buffer.
|
||||||
|
process.setInitialRamdisk(image);
|
||||||
|
var count_text: [16]u8 = undefined;
|
||||||
|
const count_arg = std.fmt.bufPrint(&count_text, "{d}", .{kernel_devices}) catch "0";
|
||||||
|
var spawned = false;
|
||||||
|
var i: u32 = 0;
|
||||||
|
while (i < rd.count) : (i += 1) {
|
||||||
|
const item = rd.entry(i) orelse continue;
|
||||||
|
if (!eql(item.name, "discovery")) continue;
|
||||||
|
_ = process.spawnProcessSupervised(item.blob, 4, &.{ "discovery", count_arg }, scheduler.currentId(), null) catch 0;
|
||||||
|
spawned = true;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
check("discovery service spawned", spawned);
|
||||||
|
result();
|
||||||
|
}
|
||||||
|
|
||||||
/// The whole user-side surface at once: spawn process-test's supervisor role,
|
/// The whole user-side surface at once: spawn process-test's supervisor role,
|
||||||
/// which — entirely from ring 3 — creates an exit endpoint, spawns its two
|
/// which — entirely from ring 3 — creates an exit endpoint, spawns its two
|
||||||
/// children supervised, sees them in process_enumerate, kills them (one blocked,
|
/// children supervised, sees them in process_enumerate, kills them (one blocked,
|
||||||
@@ -1774,6 +2319,35 @@ fn hpetGsi() ?u32 {
|
|||||||
/// land in the device table with the containment invariant intact.
|
/// land in the device table with the containment invariant intact.
|
||||||
fn busTest(boot_information: *const BootInformation) void {
|
fn busTest(boot_information: *const BootInformation) void {
|
||||||
log("DANOS-TEST-BEGIN: bus\n", .{});
|
log("DANOS-TEST-BEGIN: bus\n", .{});
|
||||||
|
|
||||||
|
// M19.0: device_register is idempotent on exact match — a restarted
|
||||||
|
// registering bus must not duplicate its children. Driven directly against
|
||||||
|
// the broker: claim an unclaimed node, register the same (class, hid,
|
||||||
|
// resourceless) child twice, expect one id and one table entry.
|
||||||
|
{
|
||||||
|
const me = scheduler.currentId();
|
||||||
|
var probe: [1]device_abi.DeviceDescriptor = undefined;
|
||||||
|
const total = devices_broker.enumerate(&probe);
|
||||||
|
check("device tree is seeded for the idempotence check", total >= 1);
|
||||||
|
if (devices_broker.ownerOf(0) == null) {
|
||||||
|
check("claimed device 0 for the idempotence check", devices_broker.claim(0, me));
|
||||||
|
var child = std.mem.zeroes(device_abi.DeviceDescriptor);
|
||||||
|
child.class = @intFromEnum(device_abi.DeviceClass.unknown);
|
||||||
|
child.pci_class = device_abi.no_pci_class;
|
||||||
|
child.hid_len = 4;
|
||||||
|
child.hid[0..4].* = "idem".*;
|
||||||
|
const first = devices_broker.register(0, me, &child) catch 0;
|
||||||
|
check("first register succeeded", first != 0);
|
||||||
|
const before = devices_broker.enumerate(&probe);
|
||||||
|
const second = devices_broker.register(0, me, &child) catch 0;
|
||||||
|
check("re-register returned the same id", second == first);
|
||||||
|
check("re-register grew nothing", devices_broker.enumerate(&probe) == before);
|
||||||
|
devices_broker.releaseAllOwnedBy(me);
|
||||||
|
} else {
|
||||||
|
check("device 0 unexpectedly claimed before the idempotence check", false);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
if (boot_information.initial_ramdisk_len == 0) {
|
if (boot_information.initial_ramdisk_len == 0) {
|
||||||
check("bootloader handed over an initial_ramdisk", false);
|
check("bootloader handed over an initial_ramdisk", false);
|
||||||
result();
|
result();
|
||||||
|
|||||||
@@ -16,8 +16,11 @@
|
|||||||
pub const maximum_cpus = 128;
|
pub const maximum_cpus = 128;
|
||||||
|
|
||||||
/// Maximum tasks (kernel threads) alive at once — the static task-table size. Each
|
/// Maximum tasks (kernel threads) alive at once — the static task-table size. Each
|
||||||
/// online core consumes one slot for its idle task, plus task 0 on the BSP.
|
/// online core consumes one slot for its idle task, plus task 0 on the BSP. Sized
|
||||||
pub const maximum_tasks = 16;
|
/// for the initial-ramdisk sweep (15 bundled binaries spawned at once) plus the
|
||||||
|
/// device manager's supervised children with room to grow — at 16 the sweep
|
||||||
|
/// started failing spawns once the bundle passed a dozen binaries.
|
||||||
|
pub const maximum_tasks = 32;
|
||||||
|
|
||||||
/// Each task's kernel stack (also each AP's bring-up stack), in bytes.
|
/// Each task's kernel stack (also each AP's bring-up stack), in bytes.
|
||||||
pub const kernel_stack_size = 16 * 1024;
|
pub const kernel_stack_size = 16 * 1024;
|
||||||
|
|||||||
@@ -0,0 +1,364 @@
|
|||||||
|
//! /system/services/acpi — the ACPI discovery service: the x86 firmware
|
||||||
|
//! interpreter, moved out of ring 0 (docs/m19-m20-plan.md, M20). Claims the
|
||||||
|
//! `acpi-tables` node the kernel publishes (the AML blobs, the broad io_port
|
||||||
|
//! grant, a broad irq window, the SCI), and runs the **shared AML module** in
|
||||||
|
//! ring 3 — the same parser and interpreter the kernel uses.
|
||||||
|
//!
|
||||||
|
//! M20.2 (this increment): after parsing, walk the namespace and, for each
|
||||||
|
//! present Device with a hardware id (`_HID`), evaluate its current resource
|
||||||
|
//! settings (`_CRS`) through a ring-3 `Hal` (port I/O over the claimed node),
|
||||||
|
//! register it under the acpi-tables node (its I/O ports and IRQs contained by
|
||||||
|
//! the node's broad grants), and report it to the device manager with its
|
||||||
|
//! EISA-decoded hid as identity. Matching those reports to drivers (ps2-bus)
|
||||||
|
//! and retiring the kernel's own device build follow in M20.3.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const runtime = @import("runtime");
|
||||||
|
const aml = @import("aml");
|
||||||
|
const device = runtime.device;
|
||||||
|
const protocol = runtime.device_manager_protocol;
|
||||||
|
|
||||||
|
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||||
|
var line: [128]u8 = undefined;
|
||||||
|
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||||
|
}
|
||||||
|
|
||||||
|
// The claimed acpi-tables node and the resource index of its broad io_port
|
||||||
|
// window — the Hal routes every port access through this one claim.
|
||||||
|
var node_id: u64 = 0;
|
||||||
|
var io_resource_index: u64 = 0;
|
||||||
|
|
||||||
|
// Pass-1 registration record (see main): what pass 2 reports.
|
||||||
|
const Registered = struct { hid: [8]u8 = .{0} ** 8, hid_len: usize = 0, device_id: u64 = 0, resource_count: u64 = 0 };
|
||||||
|
var registered: [64]Registered = undefined;
|
||||||
|
var registered_count: usize = 0;
|
||||||
|
|
||||||
|
// A scratch page returned for SystemMemory OperationRegion maps: the service
|
||||||
|
// cannot map arbitrary physical memory from ring 3, so such regions are
|
||||||
|
// unsupported and degrade to harmless zeros rather than faulting. The M20.2
|
||||||
|
// targets (ps2, the legacy devices) use SystemIO and static templates.
|
||||||
|
var mmio_scratch: [4096]u8 align(4096) = .{0} ** 4096;
|
||||||
|
|
||||||
|
fn halMapMmio(physical: u64, len: u64, writable: bool) u64 {
|
||||||
|
_ = physical;
|
||||||
|
_ = len;
|
||||||
|
_ = writable;
|
||||||
|
return @intFromPtr(&mmio_scratch);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn halPioRead(width: u8, port: u16) u32 {
|
||||||
|
return device.ioRead(node_id, io_resource_index, port, width) orelse 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn halPioWrite(width: u8, port: u16, value: u32) void {
|
||||||
|
_ = device.ioWrite(node_id, io_resource_index, port, width, value);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn findTablesNode(buffer: []device.DeviceDescriptor) ?device.DeviceDescriptor {
|
||||||
|
const total = device.enumerate(buffer);
|
||||||
|
for (buffer[0..@min(total, buffer.len)]) |d| {
|
||||||
|
if (d.class == @intFromEnum(device.DeviceClass.acpi_tables)) return d;
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn main(init: runtime.process.Init) void {
|
||||||
|
// When the acpi-parse scenario spawns this directly, argv[1] is the kernel's
|
||||||
|
// own device count to self-verify against — deterministic, no log-scraping.
|
||||||
|
const expected: ?usize = if (init.arguments.get(1)) |a| (std.fmt.parseInt(usize, a, 10) catch null) else null;
|
||||||
|
|
||||||
|
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||||
|
_ = runtime.system.write("acpi: out of memory\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
const node = findTablesNode(buffer) orelse {
|
||||||
|
_ = runtime.system.write("acpi: no acpi-tables node to claim\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
node_id = node.id;
|
||||||
|
if (!device.claim(node_id)) {
|
||||||
|
_ = runtime.system.write("acpi: unable to claim acpi-tables\n");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Map each memory resource (an AML blob) and note the io_port resource.
|
||||||
|
var blocks: [8][]const u8 = undefined;
|
||||||
|
var block_count: usize = 0;
|
||||||
|
var found_io = false;
|
||||||
|
for (node.resources[0..@intCast(node.resource_count)], 0..) |resource, index| {
|
||||||
|
if (resource.kind == @intFromEnum(device.ResourceKind.io_port) and !found_io) {
|
||||||
|
io_resource_index = index;
|
||||||
|
found_io = true;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if (resource.kind != @intFromEnum(device.ResourceKind.memory)) continue;
|
||||||
|
const base = device.mmioMap(node_id, index) orelse continue;
|
||||||
|
const pointer: [*]const u8 = @ptrFromInt(base);
|
||||||
|
blocks[block_count] = pointer[0..@intCast(resource.len)];
|
||||||
|
block_count += 1;
|
||||||
|
if (block_count == blocks.len) break;
|
||||||
|
}
|
||||||
|
if (block_count == 0) {
|
||||||
|
_ = runtime.system.write("acpi: no AML blobs on the node\n");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
const result = aml.parse(runtime.allocator(), blocks[0..block_count]) catch {
|
||||||
|
_ = runtime.system.write("acpi: AML parse failed\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
var namespace = result.namespace;
|
||||||
|
const devices = aml.deviceCount(&namespace);
|
||||||
|
writeLine("acpi: parsed {d} AML blob(s), {d} namespace devices\n", .{ block_count, devices });
|
||||||
|
if (expected) |want| {
|
||||||
|
if (devices == want) {
|
||||||
|
_ = runtime.system.write("acpi-parse: ok\n");
|
||||||
|
} else {
|
||||||
|
writeLine("acpi-parse: mismatch (ring-3 {d} vs kernel {d})\n", .{ devices, want });
|
||||||
|
}
|
||||||
|
// Self-verify mode is standalone (no manager); stop before reporting.
|
||||||
|
while (true) runtime.system.sleep(1000);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Register + report the present _HID devices (M20.2).
|
||||||
|
var arena = std.heap.ArenaAllocator.init(runtime.allocator());
|
||||||
|
var interpreter = aml.Interpreter.init(&namespace, .{
|
||||||
|
.mapMmio = halMapMmio,
|
||||||
|
.pioRead = halPioRead,
|
||||||
|
.pioWrite = halPioWrite,
|
||||||
|
}, arena.allocator());
|
||||||
|
|
||||||
|
// Pass 1: register every present _HID device under acpi-tables, remembering
|
||||||
|
// each (hid, device id). Pass 2: report them all. Registering before any
|
||||||
|
// report reaches the manager means a driver it spawns on the first report
|
||||||
|
// already sees the whole set (no keyboard-before-mouse race for ps2-bus).
|
||||||
|
registered_count = 0;
|
||||||
|
walkDevices(namespace.root, &interpreter);
|
||||||
|
|
||||||
|
const manager = runtime.ipc.lookup(.device_manager);
|
||||||
|
var i: usize = 0;
|
||||||
|
while (i < registered_count) : (i += 1) {
|
||||||
|
const entry = registered[i];
|
||||||
|
writeLine("acpi: reported {s} (device {d}, {d} resources)\n", .{ entry.hid[0..entry.hid_len], entry.device_id, entry.resource_count });
|
||||||
|
if (manager) |h| {
|
||||||
|
var report = protocol.ChildAdded{
|
||||||
|
.parent = node_id,
|
||||||
|
.bus_address = entry.device_id,
|
||||||
|
.identity = 0,
|
||||||
|
.device_id = entry.device_id,
|
||||||
|
};
|
||||||
|
@memcpy(report.hid[0..entry.hid_len], entry.hid[0..entry.hid_len]);
|
||||||
|
var reply: [protocol.message_maximum]u8 = undefined;
|
||||||
|
_ = runtime.ipc.call(h, std.mem.asBytes(&report), &reply) catch {};
|
||||||
|
}
|
||||||
|
}
|
||||||
|
writeLine("acpi: reported {d} device(s) to the manager\n", .{registered_count});
|
||||||
|
|
||||||
|
// Stay resident: the claim holds, and the service is here to grow into the
|
||||||
|
// supervised discoverer (M20.3, then the M21 event side on the SCI).
|
||||||
|
while (true) runtime.system.sleep(1000);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Depth-first walk: register + report each present device with a _HID, then
|
||||||
|
/// descend. Scopes (\_SB, \_GPE …) are descended without producing a node.
|
||||||
|
fn walkDevices(node: *aml.Node, interpreter: *aml.Interpreter) void {
|
||||||
|
var child = node.first_child;
|
||||||
|
while (child) |c| : (child = c.next_sibling) {
|
||||||
|
if (c.kind != .device) {
|
||||||
|
walkDevices(c, interpreter);
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if (!devicePresent(interpreter, c)) continue; // absent: skip it and its subtree
|
||||||
|
|
||||||
|
if (readHid(c, interpreter)) |hid| {
|
||||||
|
// Skip PCI roots — pci-bus already reports PCI functions; ACPI adds
|
||||||
|
// only the non-PCI _HID devices (docs/m19-m20-plan.md M20.2).
|
||||||
|
if (!std.mem.eql(u8, hid[0..7], "PNP0A03") and !std.mem.eql(u8, hid[0..7], "PNP0A08")) {
|
||||||
|
registerDevice(c, hid, interpreter);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
walkDevices(c, interpreter);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn registerDevice(node: *aml.Node, hid: [8]u8, interpreter: *aml.Interpreter) void {
|
||||||
|
if (registered_count >= registered.len) return;
|
||||||
|
var descriptor = std.mem.zeroes(device.DeviceDescriptor);
|
||||||
|
descriptor.class = @intFromEnum(device.DeviceClass.acpi_device);
|
||||||
|
descriptor.pci_class = device.no_pci_class;
|
||||||
|
const hid_len: u64 = std.mem.indexOfScalar(u8, &hid, 0) orelse hid.len;
|
||||||
|
descriptor.hid_len = hid_len;
|
||||||
|
@memcpy(descriptor.hid[0..@intCast(hid_len)], hid[0..@intCast(hid_len)]);
|
||||||
|
applyCrs(&descriptor, node, interpreter);
|
||||||
|
|
||||||
|
const id = device.register(node_id, &descriptor) orelse {
|
||||||
|
writeLine("acpi: register refused for {s}\n", .{hid[0..@intCast(hid_len)]});
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
registered[registered_count] = .{ .hid = hid, .hid_len = @intCast(hid_len), .device_id = id, .resource_count = descriptor.resource_count };
|
||||||
|
registered_count += 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// _STA bit 0 (present); absent method or a failed evaluation is treated as
|
||||||
|
/// present, per the ACPI rules.
|
||||||
|
fn devicePresent(interpreter: *aml.Interpreter, node: *aml.Node) bool {
|
||||||
|
const sta = aml.Namespace.childOf(node, seg4("_STA")) orelse return true;
|
||||||
|
const obj = interpreter.evaluate(sta, &.{}) catch return true;
|
||||||
|
const status = obj.asInteger() catch return true;
|
||||||
|
return (status & 0x01) != 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The device's EISA-decoded _HID (e.g. "PNP0303"), or null.
|
||||||
|
fn readHid(node: *aml.Node, interpreter: *aml.Interpreter) ?[8]u8 {
|
||||||
|
const hid = aml.Namespace.childOf(node, seg4("_HID")) orelse return null;
|
||||||
|
var buffer: [8]u8 = .{0} ** 8;
|
||||||
|
if (hid.kind == .method) {
|
||||||
|
const obj = interpreter.evaluate(hid, &.{}) catch return null;
|
||||||
|
switch (obj) {
|
||||||
|
.integer => |n| {
|
||||||
|
_ = eisaIdToStr(@truncate(n), &buffer);
|
||||||
|
return buffer;
|
||||||
|
},
|
||||||
|
else => return null,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (hid.kind != .name or hid.value.len == 0) return null;
|
||||||
|
const v = hid.value;
|
||||||
|
switch (v[0]) {
|
||||||
|
0x00, 0x01, 0xFF, 0x0A, 0x0B, 0x0C, 0x0E => {
|
||||||
|
var p: usize = 0;
|
||||||
|
const n = readIntObj(v, &p) orelse return null;
|
||||||
|
_ = eisaIdToStr(@truncate(n), &buffer);
|
||||||
|
return buffer;
|
||||||
|
},
|
||||||
|
else => return null,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- _CRS resource-template decode (ported from the kernel's acpi.zig) --------
|
||||||
|
|
||||||
|
fn applyCrs(descriptor: *device.DeviceDescriptor, node: *aml.Node, interpreter: *aml.Interpreter) void {
|
||||||
|
const crs = aml.Namespace.childOf(node, seg4("_CRS")) orelse return;
|
||||||
|
const obj = interpreter.evaluate(crs, &.{}) catch return;
|
||||||
|
const bytes = switch (obj) {
|
||||||
|
.buffer => |b| b,
|
||||||
|
else => return,
|
||||||
|
};
|
||||||
|
var i: usize = 0;
|
||||||
|
while (i < bytes.len) {
|
||||||
|
const tag = bytes[i];
|
||||||
|
if (tag & 0x80 == 0) {
|
||||||
|
const len: usize = tag & 0x07;
|
||||||
|
const body = i + 1;
|
||||||
|
if (body + len > bytes.len) break;
|
||||||
|
switch ((tag >> 3) & 0x0F) {
|
||||||
|
0x04 => if (len >= 2) { // IRQ mask
|
||||||
|
const mask = @as(u16, bytes[body]) | (@as(u16, bytes[body + 1]) << 8);
|
||||||
|
var b: usize = 0;
|
||||||
|
while (b < 16) : (b += 1) {
|
||||||
|
if (mask & (@as(u16, 1) << @intCast(b)) != 0) addResource(descriptor, .irq, b, 1);
|
||||||
|
}
|
||||||
|
},
|
||||||
|
0x08 => if (len >= 7) addResource(descriptor, .io_port, rd16(bytes, body + 1), bytes[body + 6]),
|
||||||
|
0x09 => if (len >= 3) addResource(descriptor, .io_port, rd16(bytes, body), bytes[body + 2]),
|
||||||
|
0x0F => break,
|
||||||
|
else => {},
|
||||||
|
}
|
||||||
|
i = body + len;
|
||||||
|
} else {
|
||||||
|
if (i + 3 > bytes.len) break;
|
||||||
|
const len: usize = @intCast(rd16(bytes, i + 1));
|
||||||
|
const body = i + 3;
|
||||||
|
if (body + len > bytes.len) break;
|
||||||
|
switch (tag) {
|
||||||
|
0x85 => if (len >= 17) addResource(descriptor, .memory, rd32(bytes, body + 1), rd32(bytes, body + 13)),
|
||||||
|
0x86 => if (len >= 9) addResource(descriptor, .memory, rd32(bytes, body + 1), rd32(bytes, body + 5)),
|
||||||
|
0x89 => if (len >= 2) {
|
||||||
|
const count = bytes[body + 1];
|
||||||
|
var k: usize = 0;
|
||||||
|
while (k < count and body + 2 + k * 4 + 4 <= body + len) : (k += 1) {
|
||||||
|
addResource(descriptor, .irq, rd32(bytes, body + 2 + k * 4), 1);
|
||||||
|
}
|
||||||
|
},
|
||||||
|
else => {},
|
||||||
|
}
|
||||||
|
i = body + len;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn addResource(descriptor: *device.DeviceDescriptor, kind: device.ResourceKind, start: u64, len: u64) void {
|
||||||
|
if (descriptor.resource_count >= descriptor.resources.len) return;
|
||||||
|
descriptor.resources[@intCast(descriptor.resource_count)] = .{ .kind = @intFromEnum(kind), .start = start, .len = len };
|
||||||
|
descriptor.resource_count += 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- small helpers ported verbatim from the kernel's acpi.zig ----------------
|
||||||
|
|
||||||
|
fn seg4(comptime s: *const [4:0]u8) [4]u8 {
|
||||||
|
return s[0..4].*;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn hexDigit(n: u8) u8 {
|
||||||
|
return if (n < 10) '0' + n else 'A' + (n - 10);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn eisaIdToStr(id: u32, buffer: *[8]u8) []const u8 {
|
||||||
|
const b0: u16 = @intCast(id & 0xFF);
|
||||||
|
const b1: u16 = @intCast((id >> 8) & 0xFF);
|
||||||
|
const b2: u8 = @truncate(id >> 16);
|
||||||
|
const b3: u8 = @truncate(id >> 24);
|
||||||
|
const mfg = (b0 << 8) | b1;
|
||||||
|
buffer[0] = '@' + @as(u8, @intCast((mfg >> 10) & 0x1F));
|
||||||
|
buffer[1] = '@' + @as(u8, @intCast((mfg >> 5) & 0x1F));
|
||||||
|
buffer[2] = '@' + @as(u8, @intCast(mfg & 0x1F));
|
||||||
|
buffer[3] = hexDigit((b2 >> 4) & 0xF);
|
||||||
|
buffer[4] = hexDigit(b2 & 0xF);
|
||||||
|
buffer[5] = hexDigit((b3 >> 4) & 0xF);
|
||||||
|
buffer[6] = hexDigit(b3 & 0xF);
|
||||||
|
buffer[7] = 0;
|
||||||
|
return buffer[0..7];
|
||||||
|
}
|
||||||
|
|
||||||
|
fn readIntObj(bytes: []const u8, p: *usize) ?u64 {
|
||||||
|
if (p.* >= bytes.len) return null;
|
||||||
|
const op = bytes[p.*];
|
||||||
|
p.* += 1;
|
||||||
|
switch (op) {
|
||||||
|
0x00 => return 0,
|
||||||
|
0x01 => return 1,
|
||||||
|
0xFF => return 1,
|
||||||
|
0x0A => {
|
||||||
|
if (p.* >= bytes.len) return null;
|
||||||
|
const v = bytes[p.*];
|
||||||
|
p.* += 1;
|
||||||
|
return v;
|
||||||
|
},
|
||||||
|
0x0B => {
|
||||||
|
if (p.* + 2 > bytes.len) return null;
|
||||||
|
const v = rd16(bytes, p.*);
|
||||||
|
p.* += 2;
|
||||||
|
return v;
|
||||||
|
},
|
||||||
|
0x0C => {
|
||||||
|
if (p.* + 4 > bytes.len) return null;
|
||||||
|
const v = rd32(bytes, p.*);
|
||||||
|
p.* += 4;
|
||||||
|
return v;
|
||||||
|
},
|
||||||
|
else => return null,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn rd16(bytes: []const u8, off: usize) u64 {
|
||||||
|
return @as(u64, bytes[off]) | (@as(u64, bytes[off + 1]) << 8);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn rd32(bytes: []const u8, off: usize) u64 {
|
||||||
|
return rd16(bytes, off) | (rd16(bytes, off + 2) << 16);
|
||||||
|
}
|
||||||
|
|
||||||
|
pub const panic = runtime.panic;
|
||||||
|
comptime {
|
||||||
|
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||||
|
}
|
||||||
@@ -0,0 +1,44 @@
|
|||||||
|
//! crash-test — a test fixture, not a driver: claims the device it is assigned,
|
||||||
|
//! hellos the device manager, announces itself, then faults on purpose. The
|
||||||
|
//! driver-restart scenario drives the manager's whole restart machinery with
|
||||||
|
//! it: fault → exit reason → backoff → respawn → the **same claim succeeding
|
||||||
|
//! again** (claim release on death, M17.1, through the manager's path) → the
|
||||||
|
//! crash-loop cap. Spawned bare (the initial-ramdisk sweep starts every bundled
|
||||||
|
//! binary), it exits silently so it cannot derange other tests.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const runtime = @import("runtime");
|
||||||
|
const protocol = runtime.device_manager_protocol;
|
||||||
|
|
||||||
|
pub fn main(init: runtime.process.Init) void {
|
||||||
|
const argument = init.arguments.get(1) orelse return; // bare: stay silent
|
||||||
|
const assigned = std.fmt.parseInt(u64, argument, 10) catch return;
|
||||||
|
|
||||||
|
// The respawn only reaches this line because the kernel released the
|
||||||
|
// previous instance's claim at death. A failed claim exits cleanly — the
|
||||||
|
// manager reads "meant to stop" and the scenario fails loudly by silence.
|
||||||
|
if (!runtime.device.claim(assigned)) {
|
||||||
|
_ = runtime.system.write("crash-test: claim failed\n");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
var manager: ?runtime.ipc.Handle = null;
|
||||||
|
var tries: u32 = 0;
|
||||||
|
while (manager == null and tries < 100) : (tries += 1) {
|
||||||
|
manager = runtime.ipc.lookup(.device_manager);
|
||||||
|
if (manager == null) runtime.system.sleep(20);
|
||||||
|
}
|
||||||
|
const h = manager orelse return;
|
||||||
|
const hello = protocol.Hello{ .role = @intFromEnum(protocol.Role.device), .device_id = assigned };
|
||||||
|
var reply: [protocol.message_maximum]u8 = undefined;
|
||||||
|
_ = runtime.ipc.call(h, std.mem.asBytes(&hello), &reply) catch return;
|
||||||
|
|
||||||
|
_ = runtime.system.write("crash-test: faulting now\n");
|
||||||
|
const poison: *volatile u32 = @ptrFromInt(0xdead0000);
|
||||||
|
poison.* = 1; // the restart machinery's fuel: a real segmentation fault
|
||||||
|
}
|
||||||
|
|
||||||
|
pub const panic = runtime.panic;
|
||||||
|
comptime {
|
||||||
|
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||||
|
}
|
||||||
@@ -0,0 +1,88 @@
|
|||||||
|
//! device-list — the `ps` analog for the device tree (docs/device-manager.md
|
||||||
|
//! M18.3): asks the device manager for the tree over IPC, prints it, then
|
||||||
|
//! subscribes and prints every published add/remove event. The manager is the
|
||||||
|
//! one answer to "what devices exist" for user space; nothing here touches a
|
||||||
|
//! device_* system call.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const runtime = @import("runtime");
|
||||||
|
const protocol = runtime.device_manager_protocol;
|
||||||
|
|
||||||
|
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||||
|
var line: [96]u8 = undefined;
|
||||||
|
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn main() void {
|
||||||
|
var manager: ?runtime.ipc.Handle = null;
|
||||||
|
var tries: u32 = 0;
|
||||||
|
while (manager == null and tries < 200) : (tries += 1) {
|
||||||
|
manager = runtime.ipc.lookup(.device_manager);
|
||||||
|
if (manager == null) runtime.system.sleep(20);
|
||||||
|
}
|
||||||
|
const h = manager orelse {
|
||||||
|
_ = runtime.system.write("device-list: no device manager\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
|
||||||
|
// The snapshot — polled briefly, because at boot the bus drivers may still
|
||||||
|
// be scanning: an empty first answer usually just means "too early".
|
||||||
|
var reply: [protocol.message_maximum]u8 = undefined;
|
||||||
|
var count: u32 = 0;
|
||||||
|
var length: usize = 0;
|
||||||
|
tries = 0;
|
||||||
|
while (tries < 20) : (tries += 1) {
|
||||||
|
const request = protocol.Enumerate{};
|
||||||
|
length = runtime.ipc.call(h, std.mem.asBytes(&request), &reply) catch 0;
|
||||||
|
if (length >= @sizeOf(protocol.EnumerateReply)) {
|
||||||
|
count = std.mem.bytesToValue(protocol.EnumerateReply, reply[0..@sizeOf(protocol.EnumerateReply)]).count;
|
||||||
|
if (count != 0) break;
|
||||||
|
}
|
||||||
|
runtime.system.sleep(100);
|
||||||
|
}
|
||||||
|
writeLine("device-list: {d} devices\n", .{count});
|
||||||
|
var offset: usize = @sizeOf(protocol.EnumerateReply);
|
||||||
|
var index: u32 = 0;
|
||||||
|
while (index < count and offset + @sizeOf(protocol.ChildEntry) <= length) : (index += 1) {
|
||||||
|
const entry = std.mem.bytesToValue(protocol.ChildEntry, reply[offset..][0..@sizeOf(protocol.ChildEntry)]);
|
||||||
|
writeLine("device-list: device {d} port {d} identity {d}\n", .{ entry.parent, entry.bus_address, entry.identity });
|
||||||
|
offset += @sizeOf(protocol.ChildEntry);
|
||||||
|
}
|
||||||
|
|
||||||
|
// The subscription: our endpoint rides as the call's capability; events
|
||||||
|
// arrive as buffered messages carrying the same structs the bus sends.
|
||||||
|
const endpoint = runtime.ipc.createIpcEndpoint() orelse {
|
||||||
|
_ = runtime.system.write("device-list: no endpoint\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
const subscribe = protocol.Subscribe{};
|
||||||
|
_ = runtime.ipc.callCap(h, std.mem.asBytes(&subscribe), &reply, endpoint) catch {
|
||||||
|
_ = runtime.system.write("device-list: subscribe failed\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
_ = runtime.system.write("device-list: subscribed\n");
|
||||||
|
|
||||||
|
var receive: [protocol.message_maximum]u8 = undefined;
|
||||||
|
while (true) {
|
||||||
|
const got = runtime.ipc.replyWait(endpoint, &.{}, &receive, null);
|
||||||
|
if (!got.isMessage() or got.len < 1) continue;
|
||||||
|
switch (receive[0]) {
|
||||||
|
@intFromEnum(protocol.Operation.child_added) => {
|
||||||
|
if (got.len < protocol.child_added_size) continue;
|
||||||
|
const event = std.mem.bytesToValue(protocol.ChildAdded, receive[0..protocol.child_added_size]);
|
||||||
|
writeLine("device-list: added (device {d} port {d})\n", .{ event.parent, event.bus_address });
|
||||||
|
},
|
||||||
|
@intFromEnum(protocol.Operation.child_removed) => {
|
||||||
|
if (got.len < protocol.child_removed_size) continue;
|
||||||
|
const event = std.mem.bytesToValue(protocol.ChildRemoved, receive[0..protocol.child_removed_size]);
|
||||||
|
writeLine("device-list: removed (device {d} port {d})\n", .{ event.parent, event.bus_address });
|
||||||
|
},
|
||||||
|
else => {},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub const panic = runtime.panic;
|
||||||
|
comptime {
|
||||||
|
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||||
|
}
|
||||||
@@ -0,0 +1,148 @@
|
|||||||
|
//! The device-manager protocol (docs/device-manager.md): what drivers and
|
||||||
|
//! applications say to the device manager over its well-known endpoint. The
|
||||||
|
//! vfs-protocol pattern — extern-struct messages, a version in the handshake,
|
||||||
|
//! reserved fields — so both sides depend on the contract by name. Deliberately
|
||||||
|
//! contains nothing lifecycle-shaped: stopping, liveness (the zero-length ping),
|
||||||
|
//! and exit reasons are the universal vocabulary of
|
||||||
|
//! docs/process-lifecycle.md, not this protocol.
|
||||||
|
|
||||||
|
/// The protocol version a driver states in its hello. A manager that cannot
|
||||||
|
/// serve a driver's version refuses the hello, and the mismatch is loud at
|
||||||
|
/// startup instead of quiet corruption later.
|
||||||
|
pub const version: u16 = 1;
|
||||||
|
|
||||||
|
/// What kind of driver is talking (docs/driver-model.md's shapes).
|
||||||
|
pub const Role = enum(u8) {
|
||||||
|
/// Owns a controller and reports the devices behind it (`child_added`).
|
||||||
|
bus = 1,
|
||||||
|
/// Serves one device, reached through a bus's transfer protocol.
|
||||||
|
device = 2,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The message kinds.
|
||||||
|
pub const Operation = enum(u8) {
|
||||||
|
hello = 1,
|
||||||
|
child_added = 2,
|
||||||
|
child_removed = 3,
|
||||||
|
enumerate = 4,
|
||||||
|
subscribe = 5,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// `Hello.device_id` for a driver that serves no enumerated device (a test
|
||||||
|
/// fixture, a synthetic source).
|
||||||
|
pub const no_device: u64 = ~@as(u64, 0);
|
||||||
|
|
||||||
|
/// The handshake, sent once by every driver the manager spawns — the manager's
|
||||||
|
/// one self-enforced deadline: spawned and silent past it means wrong binary,
|
||||||
|
/// wrong version, or wedged before main, and the stop sequence follows.
|
||||||
|
pub const Hello = extern struct {
|
||||||
|
operation: u8 = @intFromEnum(Operation.hello),
|
||||||
|
/// A Role value.
|
||||||
|
role: u8,
|
||||||
|
/// The protocol version this driver was built against (`version`).
|
||||||
|
version: u16 = version,
|
||||||
|
reserved: u32 = 0,
|
||||||
|
/// The device this driver was assigned (its argv[1]), or `no_device`.
|
||||||
|
device_id: u64,
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const hello_size = @sizeOf(Hello);
|
||||||
|
|
||||||
|
/// The manager's answer to a hello. Nonzero status = refused (version mismatch,
|
||||||
|
/// unknown sender); a refused driver should exit cleanly.
|
||||||
|
pub const HelloReply = extern struct {
|
||||||
|
status: i32,
|
||||||
|
reserved: u32 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const reply_size = @sizeOf(HelloReply);
|
||||||
|
|
||||||
|
/// A bus driver reporting one device it discovered behind its controller
|
||||||
|
/// (docs/device-manager.md "the tree"). Identity is the bus's native language —
|
||||||
|
/// for USB a port-speed class; the (class, subclass, protocol) triple joins it
|
||||||
|
/// once control transfers exist (the USB track). The manager mirrors the child
|
||||||
|
/// into its tree; when the reporting driver dies, the manager prunes everything
|
||||||
|
/// it reported (the children describe protocol state that died with it) and the
|
||||||
|
/// restarted instance rediscovers and re-reports.
|
||||||
|
pub const ChildAdded = extern struct {
|
||||||
|
operation: u8 = @intFromEnum(Operation.child_added),
|
||||||
|
reserved0: u8 = 0,
|
||||||
|
reserved1: u16 = 0,
|
||||||
|
reserved2: u32 = 0,
|
||||||
|
/// The reporting driver's own device (the controller) — the child's parent.
|
||||||
|
parent: u64,
|
||||||
|
/// Where on the bus (for USB: the root port number, 1-based).
|
||||||
|
bus_address: u64,
|
||||||
|
/// Bus-specific identity (for USB: the PORTSC port-speed class; for PCI:
|
||||||
|
/// the class triple; for ACPI devices, 0 — identity is the hid below).
|
||||||
|
identity: u64,
|
||||||
|
/// The kernel device id this child was `device_register`ed as — what the
|
||||||
|
/// manager hands a matched driver as its argv assignment — or `no_device`
|
||||||
|
/// for an unregistered leaf (a USB port before the descriptor track).
|
||||||
|
device_id: u64 = no_device,
|
||||||
|
/// The ACPI hardware id (`_HID`), EISA-decoded (e.g. "PNP0303"), for devices
|
||||||
|
/// discovered by firmware string rather than a numeric bus identity. Empty
|
||||||
|
/// (all zero) otherwise. Widens for FDT `compatible` strings later.
|
||||||
|
hid: [8]u8 = .{0} ** 8,
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const child_added_size = @sizeOf(ChildAdded);
|
||||||
|
|
||||||
|
/// A bus driver reporting a device gone (hot-unplug). Not yet sent by any
|
||||||
|
/// driver — the port scan has no unplug interrupt — but the manager handles it;
|
||||||
|
/// death-pruning covers removal until hotplug lands.
|
||||||
|
pub const ChildRemoved = extern struct {
|
||||||
|
operation: u8 = @intFromEnum(Operation.child_removed),
|
||||||
|
reserved0: u8 = 0,
|
||||||
|
reserved1: u16 = 0,
|
||||||
|
reserved2: u32 = 0,
|
||||||
|
parent: u64,
|
||||||
|
bus_address: u64,
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const child_removed_size = @sizeOf(ChildRemoved);
|
||||||
|
|
||||||
|
/// The manager's answer to a tree report.
|
||||||
|
pub const ReportReply = extern struct {
|
||||||
|
status: i32,
|
||||||
|
reserved: u32 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// An application asking for the tree (M18.3): the reply is an EnumerateReply
|
||||||
|
/// header followed by `count` ChildEntry records.
|
||||||
|
pub const Enumerate = extern struct {
|
||||||
|
operation: u8 = @intFromEnum(Operation.enumerate),
|
||||||
|
reserved0: u8 = 0,
|
||||||
|
reserved1: u16 = 0,
|
||||||
|
reserved2: u32 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const EnumerateReply = extern struct {
|
||||||
|
status: i32,
|
||||||
|
/// ChildEntry records following this header.
|
||||||
|
count: u32,
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const ChildEntry = extern struct {
|
||||||
|
parent: u64,
|
||||||
|
bus_address: u64,
|
||||||
|
identity: u64,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// An application subscribing to published add/remove events (the input-service
|
||||||
|
/// pattern): the subscriber's endpoint rides as the call's **capability**, and
|
||||||
|
/// events arrive on it as buffered messages whose payload is the same
|
||||||
|
/// ChildAdded / ChildRemoved struct the bus drivers send — one encoding, both
|
||||||
|
/// directions.
|
||||||
|
pub const Subscribe = extern struct {
|
||||||
|
operation: u8 = @intFromEnum(Operation.subscribe),
|
||||||
|
reserved0: u8 = 0,
|
||||||
|
reserved1: u16 = 0,
|
||||||
|
reserved2: u32 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Upper bound on any message in this protocol — sizes the endpoint buffers.
|
||||||
|
/// Capped by the kernel's IPC MESSAGE_MAXIMUM (256): an EnumerateReply carries
|
||||||
|
/// up to ten ChildEntry records per call, plenty for the mirror's current
|
||||||
|
/// bounds; paging joins the protocol if a tree ever outgrows one message.
|
||||||
|
pub const message_maximum = 256;
|
||||||
@@ -1,20 +1,24 @@
|
|||||||
//! /system/services/device-manager — the ring-3 process that turns the device
|
//! /system/services/device-manager — the ring-3 process that turns the device
|
||||||
//! tree into a running system. The kernel enumerates the hardware and enforces the
|
//! tree into a running system: **the matcher and the supervisor**
|
||||||
//! claim capability (mechanism); this decides *which driver serves which device*
|
//! (docs/device-manager.md). The kernel enumerates the hardware and enforces the
|
||||||
//! and, eventually, spawns it (policy). Keeping that split in user space is the
|
//! claim capability (mechanism); this decides which driver serves which device,
|
||||||
//! whole point of the microkernel: the manager is an ordinary, restartable process
|
//! spawns it, and keeps it alive (policy). Keeping that split in user space is
|
||||||
//! with no special privilege — it uses the same `device_*` system calls any process
|
//! the whole point of the microkernel: the manager is an ordinary, restartable
|
||||||
//! could ([drivers.md](../../../docs/drivers.md), [driver-model.md]).
|
//! process with no special privilege.
|
||||||
//!
|
//!
|
||||||
//! Increment 2 (this file): enumerate /system/devices, *match* each device to a
|
//! M18.1 (this increment): the manager is a harness service on the well-known
|
||||||
//! driver, and *spawn* it with `system_spawn` — the kernel loads the named binary
|
//! `.device_manager` endpoint. Every driver is spawned **supervised** — exit
|
||||||
//! from the initial-ramdisk as a fresh ring-3 process. On QEMU this discovers the
|
//! notifications land in the same loop as protocol messages. Drivers with an
|
||||||
//! HPET, decides `hpet` serves it, and brings that driver all the way up. (The
|
//! assignment must `hello` within a deadline or be stopped; a driver that dies
|
||||||
//! kernel still auto-spawns the whole initial-ramdisk at boot; increment 3 removes
|
//! is restarted with backoff, and a crash loop (three fast deaths) marks it
|
||||||
//! that redundancy so the manager is the sole owner of driver spawning.)
|
//! failed instead of respawning forever. Exit reasons (M17.2) drive the
|
||||||
|
//! decision: a clean exit meant to stop; only faults and missed deadlines
|
||||||
|
//! restart. Tree reports (`child_added`) land in M18.2.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const runtime = @import("runtime");
|
const runtime = @import("runtime");
|
||||||
const acpi_ids = @import("acpi-ids");
|
const acpi_ids = @import("acpi-ids");
|
||||||
|
const protocol = runtime.device_manager_protocol;
|
||||||
const device = runtime.device;
|
const device = runtime.device;
|
||||||
const system = runtime.system;
|
const system = runtime.system;
|
||||||
|
|
||||||
@@ -27,19 +31,14 @@ fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// The driver that serves each device — the policy table. In a fuller system
|
/// The driver that serves each device — the policy table. In a fuller system
|
||||||
/// this comes from the drivers describing what they bind (or a manifest under
|
/// this comes from a manifest (docs/device-manager.md: the third bus type
|
||||||
/// /system/drivers); for now it is a small static map, which is enough to prove the
|
/// triggers it); for now a static map. `null` = no driver for this class yet.
|
||||||
/// manager reads the tree and decides. `null` = no driver for this class yet.
|
|
||||||
fn driverFor(d: device.DeviceDescriptor) ?[]const u8 {
|
fn driverFor(d: device.DeviceDescriptor) ?[]const u8 {
|
||||||
// detect device via DeviceClass
|
// The HPET timer node is still kernel-seeded (from the HPET table, not AML).
|
||||||
|
// PS/2 and other _HID devices now arrive as acpi-service reports and match
|
||||||
|
// in onChildAdded (M20.3), not from this boot snapshot.
|
||||||
if (d.class == @intFromEnum(device.DeviceClass.timer)) return "hpet";
|
if (d.class == @intFromEnum(device.DeviceClass.timer)) return "hpet";
|
||||||
// detect device via hid
|
return null;
|
||||||
const hid = d.hid[0..@intCast(d.hid_len)];
|
|
||||||
const id = acpi_ids.HardwareId.fromHid(hid) orelse return null;
|
|
||||||
return switch (id) {
|
|
||||||
.ps2_keyboard, .ps2_mouse => "ps2-bus",
|
|
||||||
else => null,
|
|
||||||
};
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The PCI class/subclass/prog-IF triple of an xHCI (USB 3) host controller:
|
/// The PCI class/subclass/prog-IF triple of an xHCI (USB 3) host controller:
|
||||||
@@ -47,66 +46,498 @@ fn driverFor(d: device.DeviceDescriptor) ?[]const u8 {
|
|||||||
/// pci-class.zig decodes.
|
/// pci-class.zig decodes.
|
||||||
const xhci_pci_class: u64 = 0x0C_03_30;
|
const xhci_pci_class: u64 = 0x0C_03_30;
|
||||||
|
|
||||||
/// The bus driver that serves a PCI function, or null. Unlike the singleton drivers
|
/// The driver that serves a *reported* PCI function (M19.3: matching moved
|
||||||
/// in `driverFor`, a machine can carry several identical controllers — so the caller
|
/// from the boot snapshot to the bus reports), or null. A machine can carry
|
||||||
/// spawns one driver instance *per device*, passing the device id as argv[1] for the
|
/// several identical controllers — one driver instance per reported device,
|
||||||
/// instance to claim.
|
/// its registered id as argv[1].
|
||||||
fn pciDriverFor(d: device.DeviceDescriptor) ?[]const u8 {
|
fn pciDriverForIdentity(identity: u64) ?[]const u8 {
|
||||||
if (d.class != @intFromEnum(device.DeviceClass.pci_device)) return null;
|
return switch (identity) {
|
||||||
return switch (d.pci_class) {
|
|
||||||
xhci_pci_class => "usb-xhci-bus",
|
xhci_pci_class => "usb-xhci-bus",
|
||||||
else => null,
|
else => null,
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Spawn one instance of `driver_name` to serve the specific device `id` — the id
|
/// The driver that serves a *reported* ACPI device by its `_HID` (M20.3:
|
||||||
/// arrives as argv[1]. No isProcessRunning gate here: the name alone cannot tell two
|
/// ps2-bus now binds the PS/2 nodes the acpi service reports, not boot-snapshot
|
||||||
/// instances apart, and this manager is the sole spawner of drivers.
|
/// nodes the kernel used to build). ps2-bus is a singleton that finds both its
|
||||||
fn spawnForDevice(driver_name: []const u8, id: u64) void {
|
/// devices by hid once spawned, so keyboard and mouse map to the same name.
|
||||||
var text: [20]u8 = undefined;
|
fn hidDriverFor(hid: []const u8) ?[]const u8 {
|
||||||
const id_text = std.fmt.bufPrint(&text, "{d}", .{id}) catch return;
|
if (std.mem.eql(u8, hid, "PNP0303")) return "ps2-bus"; // PS/2 keyboard
|
||||||
if (system.spawnWithArguments(driver_name, &.{id_text}) != null) {
|
if (std.mem.eql(u8, hid, "PNP0F13")) return "ps2-bus"; // PS/2 mouse
|
||||||
writeLine("device-manager: spawned {s} for device {d}\n", .{ driver_name, id });
|
return null;
|
||||||
} else {
|
}
|
||||||
writeLine("device-manager: failed to spawn {s} for device {d}\n", .{ driver_name, id });
|
|
||||||
|
/// Whether some driver entry already serves registered device `device_id` —
|
||||||
|
/// a re-report after a bus restart must not spawn a second instance.
|
||||||
|
fn driverForDevice(device_id: u64) bool {
|
||||||
|
for (&drivers) |*driver| {
|
||||||
|
if (driver.used and driver.device_id == device_id) return true;
|
||||||
|
}
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- supervision -------------------------------------------------------------
|
||||||
|
|
||||||
|
/// How long a protocol driver has to hello after its spawn.
|
||||||
|
const hello_deadline_ms: u64 = 3000;
|
||||||
|
/// Deaths faster than this count toward the crash loop; slower ones reset it.
|
||||||
|
const fast_death_ns: u64 = 2_000_000_000;
|
||||||
|
/// Consecutive fast deaths before the manager gives up on a driver.
|
||||||
|
const crash_loop_cap: u32 = 3;
|
||||||
|
/// Restart backoff: base << (restarts - 1), so 300 ms, 600 ms, 1200 ms.
|
||||||
|
const backoff_base_ms: u64 = 300;
|
||||||
|
|
||||||
|
const DriverState = enum {
|
||||||
|
awaiting_hello, // spawned; the deadline is armed (protocol drivers only)
|
||||||
|
running,
|
||||||
|
restarting, // dead; respawn due at restart_due_ns
|
||||||
|
stopped, // exited cleanly — it meant to; not restarted
|
||||||
|
failed, // crash loop, or unspawnable; the manager gave up
|
||||||
|
};
|
||||||
|
|
||||||
|
const Driver = struct {
|
||||||
|
used: bool = false,
|
||||||
|
name_buffer: [24]u8 = undefined,
|
||||||
|
name_len: usize = 0,
|
||||||
|
// The assigned device id (becomes argv[1]), or protocol.no_device.
|
||||||
|
device_id: u64 = protocol.no_device,
|
||||||
|
// Whether this driver speaks the protocol (hello expected, deadline
|
||||||
|
// enforced). Legacy drivers (hpet, ps2-bus) are supervised and restarted
|
||||||
|
// but not yet required to hello.
|
||||||
|
speaks_protocol: bool = false,
|
||||||
|
process_id: u32 = 0,
|
||||||
|
state: DriverState = .running,
|
||||||
|
restarts: u32 = 0,
|
||||||
|
spawn_ns: u64 = 0,
|
||||||
|
hello_deadline_ns: u64 = 0,
|
||||||
|
restart_due_ns: u64 = 0,
|
||||||
|
|
||||||
|
fn name(driver: *const Driver) []const u8 {
|
||||||
|
return driver.name_buffer[0..driver.name_len];
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
const maximum_drivers = 16;
|
||||||
|
var drivers: [maximum_drivers]Driver = .{Driver{}} ** maximum_drivers;
|
||||||
|
var manager_endpoint: runtime.ipc.Handle = 0;
|
||||||
|
var test_restart_mode = false;
|
||||||
|
var test_usb_restart_mode = false;
|
||||||
|
var test_usb_killed = false;
|
||||||
|
var test_pci_restart_mode = false;
|
||||||
|
var test_kill_pid: u32 = 0;
|
||||||
|
var test_kill_due_ns: u64 = 0;
|
||||||
|
|
||||||
|
/// The application subscribers (M18.3, the input-service pattern): endpoints
|
||||||
|
/// handed over as capabilities, each receiving every child add/remove as a
|
||||||
|
/// buffered message. A subscriber whose endpoint stops accepting (it died) is
|
||||||
|
/// dropped on the failed send.
|
||||||
|
const maximum_subscribers = 8;
|
||||||
|
var subscribers: [maximum_subscribers]?runtime.ipc.Handle = .{null} ** maximum_subscribers;
|
||||||
|
|
||||||
|
/// Publish one event (a ChildAdded or ChildRemoved struct, the same encoding
|
||||||
|
/// the bus drivers send) to every subscriber.
|
||||||
|
fn publishEvent(event: []const u8) void {
|
||||||
|
for (&subscribers) |*slot| {
|
||||||
|
if (slot.*) |handle| {
|
||||||
|
if (!runtime.ipc.send(handle, event)) slot.* = null; // dead subscriber
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn main() void {
|
/// The manager's mirror of what bus drivers report (docs/device-manager.md "the
|
||||||
|
/// tree"): the children, keyed by (parent, bus address), each remembering which
|
||||||
|
/// driver instance reported it — that is what death-pruning sweeps by.
|
||||||
|
const Child = struct {
|
||||||
|
used: bool = false,
|
||||||
|
parent: u64 = 0,
|
||||||
|
bus_address: u64 = 0,
|
||||||
|
identity: u64 = 0,
|
||||||
|
// The kernel device id (registered by the reporter), or protocol.no_device.
|
||||||
|
device_id: u64 = 0,
|
||||||
|
reporter: u32 = 0, // the reporting driver instance's process id
|
||||||
|
};
|
||||||
|
|
||||||
|
const maximum_children = 64; // ACPI adds ~34 device nodes (M20.2), plus PCI + USB
|
||||||
|
var children: [maximum_children]Child = .{Child{}} ** maximum_children;
|
||||||
|
|
||||||
|
/// Record (or refresh) a reported child. Refreshing matters: a restarted bus
|
||||||
|
/// driver re-reports what it rediscovers, and the same (parent, port) must not
|
||||||
|
/// duplicate.
|
||||||
|
fn addChild(parent: u64, bus_address: u64, identity: u64, device_id: u64, reporter: u32) bool {
|
||||||
|
var free: ?*Child = null;
|
||||||
|
for (&children) |*child| {
|
||||||
|
if (child.used and child.parent == parent and child.bus_address == bus_address) {
|
||||||
|
child.identity = identity;
|
||||||
|
child.device_id = device_id;
|
||||||
|
child.reporter = reporter;
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
if (!child.used and free == null) free = child;
|
||||||
|
}
|
||||||
|
const slot = free orelse return false;
|
||||||
|
slot.* = .{ .used = true, .parent = parent, .bus_address = bus_address, .identity = identity, .device_id = device_id, .reporter = reporter };
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Prune every child a dead driver instance reported: the children describe
|
||||||
|
/// protocol state (slots, rings) that died with the process — keeping the nodes
|
||||||
|
/// would be keeping a lie. The restarted instance rediscovers and re-reports.
|
||||||
|
/// Watchers hear the honest story: removed now, added again on rediscovery.
|
||||||
|
fn pruneChildrenOf(reporter: u32) void {
|
||||||
|
for (&children) |*child| {
|
||||||
|
if (child.used and child.reporter == reporter) {
|
||||||
|
writeLine("device-manager: child removed (device {d} port {d})\n", .{ child.parent, child.bus_address });
|
||||||
|
child.used = false;
|
||||||
|
const event = protocol.ChildRemoved{ .parent = child.parent, .bus_address = child.bus_address };
|
||||||
|
publishEvent(std.mem.asBytes(&event));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// How many children a driver instance has reported (the test-usb-restart
|
||||||
|
/// trigger counts these).
|
||||||
|
fn childCountOf(reporter: u32) u32 {
|
||||||
|
var n: u32 = 0;
|
||||||
|
for (&children) |*child| {
|
||||||
|
if (child.used and child.reporter == reporter) n += 1;
|
||||||
|
}
|
||||||
|
return n;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn driverByProcess(process_id: u32) ?*Driver {
|
||||||
|
for (&drivers) |*driver| {
|
||||||
|
if (driver.used and driver.process_id == process_id) return driver;
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether a singleton driver is already in the table (two ACPI nodes can both
|
||||||
|
/// map to ps2-bus; one instance serves both).
|
||||||
|
fn alreadySupervised(name: []const u8) bool {
|
||||||
|
for (&drivers) |*driver| {
|
||||||
|
if (driver.used and std.mem.eql(u8, driver.name(), name)) return true;
|
||||||
|
}
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Record a driver in the table and spawn its first instance.
|
||||||
|
fn addDriver(name: []const u8, device_id: u64, speaks_protocol: bool) void {
|
||||||
|
for (&drivers) |*driver| {
|
||||||
|
if (driver.used) continue;
|
||||||
|
const n = @min(name.len, driver.name_buffer.len);
|
||||||
|
@memcpy(driver.name_buffer[0..n], name[0..n]);
|
||||||
|
driver.name_len = n;
|
||||||
|
driver.device_id = device_id;
|
||||||
|
driver.speaks_protocol = speaks_protocol;
|
||||||
|
driver.used = true;
|
||||||
|
spawnDriver(driver);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
writeLine("device-manager: driver table full; cannot supervise {s}\n", .{name});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// (Re)spawn a driver instance: supervised on the manager's own endpoint, the
|
||||||
|
/// device id as argv[1] when it has one, the hello deadline armed when it
|
||||||
|
/// speaks the protocol.
|
||||||
|
fn spawnDriver(driver: *Driver) void {
|
||||||
|
var id_text: [20]u8 = undefined;
|
||||||
|
var arguments: [1][]const u8 = undefined;
|
||||||
|
var argument_count: usize = 0;
|
||||||
|
if (driver.device_id != protocol.no_device) {
|
||||||
|
arguments[0] = std.fmt.bufPrint(&id_text, "{d}", .{driver.device_id}) catch return;
|
||||||
|
argument_count = 1;
|
||||||
|
}
|
||||||
|
const child = system.spawnSupervised(driver.name(), arguments[0..argument_count], manager_endpoint) orelse {
|
||||||
|
writeLine("device-manager: failed to spawn {s}\n", .{driver.name()});
|
||||||
|
driver.state = .failed;
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
driver.process_id = child;
|
||||||
|
driver.spawn_ns = system.clock();
|
||||||
|
if (driver.speaks_protocol) {
|
||||||
|
driver.state = .awaiting_hello;
|
||||||
|
driver.hello_deadline_ns = driver.spawn_ns + hello_deadline_ms * 1_000_000;
|
||||||
|
_ = system.timerOnce(manager_endpoint, hello_deadline_ms + 100);
|
||||||
|
} else {
|
||||||
|
driver.state = .running;
|
||||||
|
}
|
||||||
|
if (driver.device_id != protocol.no_device) {
|
||||||
|
writeLine("device-manager: spawned {s} for device {d}\n", .{ driver.name(), driver.device_id });
|
||||||
|
} else {
|
||||||
|
writeLine("device-manager: spawned {s}\n", .{driver.name()});
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A driver died. Prune what it reported first — then the exit reason (M17.2)
|
||||||
|
/// is the whole restart decision: a clean exit meant to stop; anything else
|
||||||
|
/// restarts with backoff until the crash-loop cap.
|
||||||
|
fn onDriverExit(driver: *Driver) void {
|
||||||
|
pruneChildrenOf(driver.process_id);
|
||||||
|
const reason = runtime.process.exitReason(driver.process_id) orelse .fault;
|
||||||
|
if (reason == .exited) {
|
||||||
|
driver.state = .stopped;
|
||||||
|
writeLine("device-manager: {s} exited cleanly; not restarting\n", .{driver.name()});
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const now = system.clock();
|
||||||
|
const alive_ns = now - driver.spawn_ns;
|
||||||
|
driver.restarts = if (alive_ns < fast_death_ns) driver.restarts + 1 else 1;
|
||||||
|
if (driver.restarts >= crash_loop_cap) {
|
||||||
|
driver.state = .failed;
|
||||||
|
writeLine("device-manager: {s} is failing repeatedly (crash loop); giving up\n", .{driver.name()});
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const delay_ms = backoff_base_ms << @intCast(driver.restarts - 1);
|
||||||
|
driver.state = .restarting;
|
||||||
|
driver.restart_due_ns = now + delay_ms * 1_000_000;
|
||||||
|
writeLine("device-manager: restarting {s} in {d} ms (died: {s})\n", .{ driver.name(), delay_ms, @tagName(reason) });
|
||||||
|
_ = system.timerOnce(manager_endpoint, delay_ms + 50);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A timer landed: sweep every deadline. Overdue hellos are killed (the exit
|
||||||
|
/// notification then routes through the normal restart policy); due restarts
|
||||||
|
/// respawn. Timers carry no id on purpose — the table is the state, and one
|
||||||
|
/// sweep serves every armed deadline.
|
||||||
|
fn sweepDeadlines() void {
|
||||||
|
const now = system.clock();
|
||||||
|
if (test_kill_pid != 0 and now >= test_kill_due_ns) {
|
||||||
|
writeLine("device-manager: test mode: killing the reporter\n", .{});
|
||||||
|
_ = system.kill(test_kill_pid);
|
||||||
|
test_kill_pid = 0;
|
||||||
|
}
|
||||||
|
for (&drivers) |*driver| {
|
||||||
|
if (!driver.used) continue;
|
||||||
|
switch (driver.state) {
|
||||||
|
.awaiting_hello => if (now >= driver.hello_deadline_ns) {
|
||||||
|
writeLine("device-manager: {s} missed its hello deadline\n", .{driver.name()});
|
||||||
|
_ = system.kill(driver.process_id);
|
||||||
|
// The exit notification finishes the job via onDriverExit.
|
||||||
|
},
|
||||||
|
.restarting => if (now >= driver.restart_due_ns) spawnDriver(driver),
|
||||||
|
else => {},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- the harness callbacks -----------------------------------------------------
|
||||||
|
|
||||||
|
fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||||
|
manager_endpoint = endpoint;
|
||||||
|
|
||||||
// Enumerate into a heap buffer (too big for the one-page user stack).
|
// Enumerate into a heap buffer (too big for the one-page user stack).
|
||||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||||
_ = runtime.system.write("device-manager: out of memory\n");
|
_ = runtime.system.write("device-manager: out of memory\n");
|
||||||
return;
|
return false;
|
||||||
};
|
};
|
||||||
const total = device.enumerate(buffer);
|
const total = device.enumerate(buffer);
|
||||||
const count = @min(total, buffer.len);
|
const count = @min(total, buffer.len);
|
||||||
|
|
||||||
var matched: usize = 0;
|
var matched: usize = 0;
|
||||||
for (buffer[0..count]) |descriptor| {
|
for (buffer[0..count]) |descriptor| {
|
||||||
if (pciDriverFor(descriptor)) |driver_name| {
|
if (descriptor.class == @intFromEnum(device.DeviceClass.pci_host_bridge)) {
|
||||||
|
// The PCI bus driver: enumeration in ring 3 (M19), one instance
|
||||||
|
// per bridge, the bridge id as its assignment.
|
||||||
matched += 1;
|
matched += 1;
|
||||||
spawnForDevice(driver_name, descriptor.id);
|
addDriver("pci-bus", descriptor.id, true);
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
|
// PCI functions no longer appear in the boot snapshot (M19.3): the
|
||||||
|
// pci-bus driver reports them, and onChildAdded matches from reports.
|
||||||
const driver_name = driverFor(descriptor) orelse continue;
|
const driver_name = driverFor(descriptor) orelse continue;
|
||||||
matched += 1;
|
matched += 1;
|
||||||
if (!system.isProcessRunning(driver_name)) {
|
// Skip a singleton that is already alive (the initial-ramdisk sweep test
|
||||||
if (runtime.system.spawn(driver_name) != null) {
|
// starts every bundled binary bare, this manager included) — spawning a
|
||||||
writeLine("device-manager: spawned {s}\n", .{driver_name});
|
// second instance would only lose the claim race and churn the log.
|
||||||
} else {
|
if (!alreadySupervised(driver_name) and !system.isProcessRunning(driver_name)) {
|
||||||
writeLine("device-manager: failed to spawn {s}\n", .{driver_name});
|
addDriver(driver_name, protocol.no_device, false);
|
||||||
}
|
}
|
||||||
} else {
|
|
||||||
writeLine("device-manager: already spawned {s}\n", .{driver_name});
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// The discovery service (docs/m19-m20-plan.md M20): one per firmware, packed
|
||||||
|
// under the neutral name "discovery", spawned once at startup. It finds and
|
||||||
|
// claims the acpi-tables (or devicetree-blob) node itself. Not a per-device
|
||||||
|
// match — it is the discoverer, not a driver bound to one device.
|
||||||
|
addDriver("discovery", protocol.no_device, false);
|
||||||
|
|
||||||
|
if (test_restart_mode) {
|
||||||
|
// The driver-restart scenario's fixture: claims device 0 (the tree
|
||||||
|
// root, otherwise unclaimed), hellos, then faults — driving backoff,
|
||||||
|
// re-claim-after-death, and the crash-loop cap deterministically.
|
||||||
|
addDriver("crash-test", 0, true);
|
||||||
}
|
}
|
||||||
|
|
||||||
if (matched == 0) {
|
if (matched == 0) {
|
||||||
_ = runtime.system.write("device-manager: no matchable devices\n");
|
_ = runtime.system.write("device-manager: no matchable devices\n");
|
||||||
|
} else {
|
||||||
|
_ = runtime.system.write("device-manager: ok\n");
|
||||||
|
}
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?runtime.ipc.Handle) usize {
|
||||||
|
if (message.len < 1) return 0;
|
||||||
|
switch (message[0]) {
|
||||||
|
@intFromEnum(protocol.Operation.child_added) => return onChildAdded(message, reply, sender),
|
||||||
|
@intFromEnum(protocol.Operation.child_removed) => return onChildRemoved(message, reply, sender),
|
||||||
|
@intFromEnum(protocol.Operation.enumerate) => return onEnumerate(reply),
|
||||||
|
@intFromEnum(protocol.Operation.subscribe) => return onSubscribe(reply, capability),
|
||||||
|
@intFromEnum(protocol.Operation.hello) => {},
|
||||||
|
else => return 0,
|
||||||
|
}
|
||||||
|
if (message.len < protocol.hello_size) return 0;
|
||||||
|
const hello = std.mem.bytesToValue(protocol.Hello, message[0..protocol.hello_size]);
|
||||||
|
|
||||||
|
var status: i32 = 0;
|
||||||
|
if (hello.version != protocol.version) {
|
||||||
|
status = -1;
|
||||||
|
writeLine("device-manager: refused hello (version {d}) from process {d}\n", .{ hello.version, sender });
|
||||||
|
} else if (driverByProcess(sender)) |driver| {
|
||||||
|
driver.state = .running;
|
||||||
|
writeLine("device-manager: hello from {s} (device {d})\n", .{ driver.name(), hello.device_id });
|
||||||
|
} else {
|
||||||
|
status = -1;
|
||||||
|
writeLine("device-manager: hello from unknown process {d}\n", .{sender});
|
||||||
|
}
|
||||||
|
const hello_reply = protocol.HelloReply{ .status = status };
|
||||||
|
@memcpy(reply[0..protocol.reply_size], std.mem.asBytes(&hello_reply));
|
||||||
|
return protocol.reply_size;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A bus driver reported a discovered device: mirror it, and in
|
||||||
|
/// test-usb-restart mode kill the reporter once after its second child — the
|
||||||
|
/// deterministic trigger for prune -> backoff -> respawn -> re-report.
|
||||||
|
fn onChildAdded(message: []const u8, reply: []u8, sender: u32) usize {
|
||||||
|
if (message.len < protocol.child_added_size) return 0;
|
||||||
|
const report = std.mem.bytesToValue(protocol.ChildAdded, message[0..protocol.child_added_size]);
|
||||||
|
var status: i32 = 0;
|
||||||
|
if (driverByProcess(sender)) |driver| {
|
||||||
|
if (!addChild(report.parent, report.bus_address, report.identity, report.device_id, sender)) status = -1;
|
||||||
|
writeLine("device-manager: child added (device {d} port {d}, identity {d}) by {s}\n", .{ report.parent, report.bus_address, report.identity, driver.name() });
|
||||||
|
if (status == 0) publishEvent(message[0..protocol.child_added_size]);
|
||||||
|
// Matching from reports (M19.3): a registered child whose identity
|
||||||
|
// names a driver gets one, once — re-reports after a bus restart
|
||||||
|
// dedupe on the registered id, exactly like the registrations do.
|
||||||
|
if (status == 0 and report.device_id != protocol.no_device) {
|
||||||
|
if (pciDriverForIdentity(report.identity)) |child_driver| {
|
||||||
|
if (!driverForDevice(report.device_id)) addDriver(child_driver, report.device_id, true);
|
||||||
|
}
|
||||||
|
// ACPI _HID match (M20.3): ps2-bus is a singleton that finds its own
|
||||||
|
// devices by hid, so spawn it once, without a device assignment.
|
||||||
|
const hid_len = std.mem.indexOfScalar(u8, &report.hid, 0) orelse report.hid.len;
|
||||||
|
if (hid_len != 0) {
|
||||||
|
if (hidDriverFor(report.hid[0..hid_len])) |hid_driver| {
|
||||||
|
if (!alreadySupervised(hid_driver)) addDriver(hid_driver, protocol.no_device, false);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
status = -1;
|
||||||
|
}
|
||||||
|
const report_reply = protocol.ReportReply{ .status = status };
|
||||||
|
@memcpy(reply[0..@sizeOf(protocol.ReportReply)], std.mem.asBytes(&report_reply));
|
||||||
|
if (test_pci_restart_mode and !test_usb_killed) {
|
||||||
|
if (driverByProcess(sender)) |driver| {
|
||||||
|
if (std.mem.eql(u8, driver.name(), "pci-bus") and childCountOf(sender) >= 3) {
|
||||||
|
// The pci restart drill: kill the enumerator after it has
|
||||||
|
// reported; the respawn must re-register without duplicates
|
||||||
|
// (M19.0 idempotence, proven end to end by pci-scan).
|
||||||
|
test_usb_killed = true;
|
||||||
|
test_kill_pid = sender;
|
||||||
|
test_kill_due_ns = system.clock() + 1_000_000_000;
|
||||||
|
_ = system.timerOnce(manager_endpoint, 1100);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (test_usb_restart_mode and !test_usb_killed and childCountOf(sender) >= 2) {
|
||||||
|
// Only the xHCI reporter is the drill's victim — pci-bus also reports
|
||||||
|
// now, and whichever finishes second must not trigger the kill.
|
||||||
|
if (driverByProcess(sender)) |driver| {
|
||||||
|
if (std.mem.eql(u8, driver.name(), "usb-xhci-bus")) {
|
||||||
|
// Delayed, not immediate: the device-list scenario's subscriber
|
||||||
|
// needs a window to enumerate and subscribe before the events.
|
||||||
|
test_usb_killed = true;
|
||||||
|
test_kill_pid = sender;
|
||||||
|
test_kill_due_ns = system.clock() + 2_000_000_000;
|
||||||
|
_ = system.timerOnce(manager_endpoint, 2100);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return @sizeOf(protocol.ReportReply);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A bus driver reported a device gone (hot-unplug; no sender exists yet, but
|
||||||
|
/// the handler is protocol-complete — death-pruning covers removal until then).
|
||||||
|
fn onChildRemoved(message: []const u8, reply: []u8, sender: u32) usize {
|
||||||
|
if (message.len < protocol.child_removed_size) return 0;
|
||||||
|
const report = std.mem.bytesToValue(protocol.ChildRemoved, message[0..protocol.child_removed_size]);
|
||||||
|
var status: i32 = -1;
|
||||||
|
for (&children) |*child| {
|
||||||
|
if (child.used and child.parent == report.parent and child.bus_address == report.bus_address and child.reporter == sender) {
|
||||||
|
writeLine("device-manager: child removed (device {d} port {d})\n", .{ child.parent, child.bus_address });
|
||||||
|
child.used = false;
|
||||||
|
status = 0;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
const report_reply = protocol.ReportReply{ .status = status };
|
||||||
|
@memcpy(reply[0..@sizeOf(protocol.ReportReply)], std.mem.asBytes(&report_reply));
|
||||||
|
return @sizeOf(protocol.ReportReply);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// An application asked for the tree: the mirror, as a header plus entries.
|
||||||
|
fn onEnumerate(reply: []u8) usize {
|
||||||
|
var count: u32 = 0;
|
||||||
|
var offset: usize = @sizeOf(protocol.EnumerateReply);
|
||||||
|
for (&children) |*child| {
|
||||||
|
if (!child.used) continue;
|
||||||
|
if (offset + @sizeOf(protocol.ChildEntry) > reply.len) break;
|
||||||
|
const entry = protocol.ChildEntry{ .parent = child.parent, .bus_address = child.bus_address, .identity = child.identity };
|
||||||
|
@memcpy(reply[offset..][0..@sizeOf(protocol.ChildEntry)], std.mem.asBytes(&entry));
|
||||||
|
offset += @sizeOf(protocol.ChildEntry);
|
||||||
|
count += 1;
|
||||||
|
}
|
||||||
|
const header = protocol.EnumerateReply{ .status = 0, .count = count };
|
||||||
|
@memcpy(reply[0..@sizeOf(protocol.EnumerateReply)], std.mem.asBytes(&header));
|
||||||
|
return offset;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// An application subscribed: its endpoint arrived as the call's capability.
|
||||||
|
fn onSubscribe(reply: []u8, capability: ?runtime.ipc.Handle) usize {
|
||||||
|
var status: i32 = -1;
|
||||||
|
if (capability) |handle| {
|
||||||
|
for (&subscribers) |*slot| {
|
||||||
|
if (slot.* == null) {
|
||||||
|
slot.* = handle;
|
||||||
|
status = 0;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
const report_reply = protocol.ReportReply{ .status = status };
|
||||||
|
@memcpy(reply[0..@sizeOf(protocol.ReportReply)], std.mem.asBytes(&report_reply));
|
||||||
|
return @sizeOf(protocol.ReportReply);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn onNotification(badge: u64) void {
|
||||||
|
if (badge & runtime.ipc.notify_exit_bit != 0) {
|
||||||
|
const dead: u32 = @intCast(badge & ~(runtime.ipc.notify_badge_bit | runtime.ipc.notify_exit_bit));
|
||||||
|
if (driverByProcess(dead)) |driver| onDriverExit(driver);
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
_ = runtime.system.write("device-manager: ok\n");
|
if (badge & runtime.ipc.notify_timer_bit != 0) sweepDeadlines();
|
||||||
while (true) runtime.system.sleep(1000);
|
}
|
||||||
|
|
||||||
|
pub fn main(init: runtime.process.Init) void {
|
||||||
|
if (init.arguments.get(1)) |mode| {
|
||||||
|
test_restart_mode = std.mem.eql(u8, mode, "test-restart");
|
||||||
|
test_usb_restart_mode = std.mem.eql(u8, mode, "test-usb-restart");
|
||||||
|
test_pci_restart_mode = std.mem.eql(u8, mode, "test-pci-restart");
|
||||||
|
}
|
||||||
|
runtime.service.run(protocol.message_maximum, .{
|
||||||
|
.service = .device_manager,
|
||||||
|
.init = initialise,
|
||||||
|
.on_message = onMessage,
|
||||||
|
.on_notification = onNotification,
|
||||||
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
pub const panic = runtime.panic;
|
pub const panic = runtime.panic;
|
||||||
|
|||||||
@@ -0,0 +1,35 @@
|
|||||||
|
//! /system/services/fdt — the devicetree discovery service: the ARM twin of the
|
||||||
|
//! acpi service (docs/m19-m20-plan.md decision 7). **Placeholder: not
|
||||||
|
//! implemented.** It exists so the build's `-Ddiscovery` option has both of its
|
||||||
|
//! values from day one; the implementation lands with the Raspberry Pi
|
||||||
|
//! bring-up (docs/arm.md).
|
||||||
|
//!
|
||||||
|
//! What it becomes: the per-firmware discoverer for boots that hand over a
|
||||||
|
//! flattened device tree instead of ACPI tables. It claims the
|
||||||
|
//! `devicetree-blob` node the kernel publishes (the FDT the loader received),
|
||||||
|
//! walks the tree — pure data, no bytecode, so unlike the acpi service it
|
||||||
|
//! needs no port grant and no interpreter — and, like any bus-shaped driver:
|
||||||
|
//! `device_register`s what it finds (containment against the blob node's
|
||||||
|
//! recorded apertures), reports each child to the device manager
|
||||||
|
//! (`child_added`, identity = the node's `compatible` string), and stays
|
||||||
|
//! resident under the manager's supervision (hello, restart, the usual
|
||||||
|
//! contract).
|
||||||
|
//!
|
||||||
|
//! Known prerequisite recorded in the plan: `DeviceDescriptor`'s 8-byte `hid`
|
||||||
|
//! cannot hold an FDT `compatible` string ("brcm,bcm2835-aux-uart") — identity
|
||||||
|
//! widens before this file grows a body.
|
||||||
|
|
||||||
|
const runtime = @import("runtime");
|
||||||
|
|
||||||
|
pub fn main(init: runtime.process.Init) void {
|
||||||
|
_ = init;
|
||||||
|
// Not implemented: exit cleanly and silently (a bare spawn by the
|
||||||
|
// initial-ramdisk sweep must not derange other tests' markers). The
|
||||||
|
// supervisor reads a clean exit as "meant to stop" — correct for a
|
||||||
|
// placeholder.
|
||||||
|
}
|
||||||
|
|
||||||
|
pub const panic = runtime.panic;
|
||||||
|
comptime {
|
||||||
|
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||||
|
}
|
||||||
@@ -48,11 +48,92 @@ fn awaitChildExit(endpoint: runtime.ipc.Handle) u32 {
|
|||||||
return received.childProcessId();
|
return received.childProcessId();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The harness-run child of the signals test: echoes requests, logs the two
|
||||||
|
/// signals it handles. Terminate makes run() return, and returning from main is
|
||||||
|
/// the clean exit the parent reads as ExitReason.exited.
|
||||||
|
fn echo(message: []const u8, reply: []u8, sender: u32, capability: ?runtime.ipc.Handle) usize {
|
||||||
|
_ = sender;
|
||||||
|
_ = capability;
|
||||||
|
const n = @min(message.len, reply.len);
|
||||||
|
@memcpy(reply[0..n], message[0..n]);
|
||||||
|
return n;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn onReload() void {
|
||||||
|
_ = runtime.system.write("process-test: reloaded\n");
|
||||||
|
}
|
||||||
|
|
||||||
|
fn onTerminate() void {
|
||||||
|
_ = runtime.system.write("process-test: terminating\n");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The parent of the signals test: drives ping, echo, reload, the one-shot
|
||||||
|
/// timer, and both endings of the stop sequence (polite -> exited; deaf ->
|
||||||
|
/// killed at the deadline). Prints "process-test: signals ok" as the marker.
|
||||||
|
fn signalRun() void {
|
||||||
|
const endpoint = runtime.ipc.createIpcEndpoint() orelse fail("create exit endpoint");
|
||||||
|
const child = runtime.system.spawnSupervised("process-test", &.{"service"}, endpoint) orelse fail("spawn service child");
|
||||||
|
|
||||||
|
// Reach the child's endpoint through the registry (retry: it may not be up).
|
||||||
|
var service_handle: ?runtime.ipc.Handle = null;
|
||||||
|
var tries: u32 = 0;
|
||||||
|
while (service_handle == null and tries < 200) : (tries += 1) {
|
||||||
|
service_handle = runtime.ipc.lookup(.input);
|
||||||
|
if (service_handle == null) runtime.system.sleep(20);
|
||||||
|
}
|
||||||
|
const h = service_handle orelse fail("service child never registered");
|
||||||
|
|
||||||
|
// The universal ping: a zero-length call answered zero-length by the harness.
|
||||||
|
var reply: [16]u8 = undefined;
|
||||||
|
const pong = runtime.ipc.call(h, &.{}, &reply) catch fail("ping call failed");
|
||||||
|
if (pong != 0) fail("ping reply not empty");
|
||||||
|
|
||||||
|
// An ordinary request still reaches on_message.
|
||||||
|
const n = runtime.ipc.call(h, "echo!", &reply) catch fail("echo call failed");
|
||||||
|
if (n != 5 or !std.mem.eql(u8, reply[0..5], "echo!")) fail("echo mismatch");
|
||||||
|
|
||||||
|
// reload: a statement — the child logs it; the kernel test reads the serial.
|
||||||
|
if (!runtime.process.sendSignal(child, .reload)) fail("send reload");
|
||||||
|
runtime.system.sleep(200);
|
||||||
|
|
||||||
|
// The one-shot timer: armed on our endpoint, lands as isTimer.
|
||||||
|
if (!runtime.system.timerOnce(endpoint, 100)) fail("arm timer");
|
||||||
|
var scratch: [8]u8 = undefined;
|
||||||
|
const landing = runtime.ipc.replyWait(endpoint, scratch[0..0], &scratch, null);
|
||||||
|
if (!landing.isTimer()) fail("expected the timer landing");
|
||||||
|
|
||||||
|
// The stop sequence, polite path: terminate, clean exit inside the deadline.
|
||||||
|
runtime.process.stop(child, 2000, endpoint);
|
||||||
|
if ((runtime.process.exitReason(child) orelse .killed) != .exited) fail("service child reason not exited");
|
||||||
|
|
||||||
|
// The deaf child: binds nothing, hears nothing — the deadline kills it.
|
||||||
|
const deaf = runtime.system.spawnSupervised("process-test", &.{"sleeper"}, endpoint) orelse fail("spawn deaf child");
|
||||||
|
runtime.system.sleep(50); // let it reach its sleep
|
||||||
|
runtime.process.stop(deaf, 300, endpoint);
|
||||||
|
if ((runtime.process.exitReason(deaf) orelse .exited) != .killed) fail("deaf child reason not killed");
|
||||||
|
|
||||||
|
_ = runtime.system.write("process-test: signals ok\n");
|
||||||
|
}
|
||||||
|
|
||||||
pub fn main(init: runtime.process.Init) void {
|
pub fn main(init: runtime.process.Init) void {
|
||||||
const role = init.arguments.get(1) orelse return; // spawned bare (ramdisk sweep): stay silent
|
const role = init.arguments.get(1) orelse return; // spawned bare (ramdisk sweep): stay silent
|
||||||
if (std.mem.eql(u8, role, "sleeper")) {
|
if (std.mem.eql(u8, role, "sleeper")) {
|
||||||
while (true) runtime.system.sleep(500);
|
while (true) runtime.system.sleep(500);
|
||||||
}
|
}
|
||||||
|
if (std.mem.eql(u8, role, "service")) {
|
||||||
|
// Borrowed well-known id: the input service is not part of this scenario.
|
||||||
|
runtime.service.run(64, .{
|
||||||
|
.service = .input,
|
||||||
|
.on_message = echo,
|
||||||
|
.on_reload = onReload,
|
||||||
|
.on_terminate = onTerminate,
|
||||||
|
});
|
||||||
|
return; // terminate arrived; returning is the clean exit
|
||||||
|
}
|
||||||
|
if (std.mem.eql(u8, role, "signal-run")) {
|
||||||
|
signalRun();
|
||||||
|
return;
|
||||||
|
}
|
||||||
if (std.mem.eql(u8, role, "spinner")) {
|
if (std.mem.eql(u8, role, "spinner")) {
|
||||||
var beat: u64 = 0;
|
var beat: u64 = 0;
|
||||||
const touch: *volatile u64 = &beat;
|
const touch: *volatile u64 = &beat;
|
||||||
@@ -89,6 +170,12 @@ pub fn main(init: runtime.process.Init) void {
|
|||||||
if (listed(sleeper, "process-test")) fail("sleeper still listed after kill");
|
if (listed(sleeper, "process-test")) fail("sleeper still listed after kill");
|
||||||
if (listed(spinner, "process-test")) fail("spinner still listed after kill");
|
if (listed(spinner, "process-test")) fail("spinner still listed after kill");
|
||||||
|
|
||||||
|
// M17.2: both children were killed by us, and the reason says so — the whole
|
||||||
|
// restart-policy input, read through the runtime like a real supervisor would.
|
||||||
|
if ((runtime.process.exitReason(sleeper) orelse .exited) != .killed) fail("sleeper reason not killed");
|
||||||
|
if ((runtime.process.exitReason(spinner) orelse .exited) != .killed) fail("spinner reason not killed");
|
||||||
|
if (runtime.process.exitReason(0xFFFF_FFF0) != null) fail("unknown id had a reason");
|
||||||
|
|
||||||
_ = runtime.system.write("process-test: ok\n");
|
_ = runtime.system.write("process-test: ok\n");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -6,10 +6,30 @@
|
|||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const runtime = @import("runtime");
|
const runtime = @import("runtime");
|
||||||
|
|
||||||
pub fn main() void {
|
pub fn main(init: runtime.process.Init) void {
|
||||||
const u = @import("posix").unistd;
|
const u = @import("posix").unistd;
|
||||||
const payload = "hello-vfs";
|
const payload = "hello-vfs";
|
||||||
|
|
||||||
|
// The "park" role (the vfs-client-death test): open a file, then hold the
|
||||||
|
// handle forever without closing — the kill and the VFS's release-on-death
|
||||||
|
// are the point.
|
||||||
|
if (init.arguments.count > 1) {
|
||||||
|
var fd: i32 = -1;
|
||||||
|
var tries: u32 = 0;
|
||||||
|
while (fd < 0 and tries < 200) : (tries += 1) {
|
||||||
|
fd = u.open("parked", u.O_CREAT);
|
||||||
|
if (fd < 0) runtime.system.sleep(20);
|
||||||
|
}
|
||||||
|
if (fd < 0) {
|
||||||
|
_ = runtime.system.write("vfstest: park open failed\n");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
while (true) {
|
||||||
|
_ = runtime.system.write("vfstest: parked\n");
|
||||||
|
runtime.system.sleep(500);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// The VFS server may not have registered yet — retry open until it's up.
|
// The VFS server may not have registered yet — retry open until it's up.
|
||||||
var fd: i32 = -1;
|
var fd: i32 = -1;
|
||||||
var tries: u32 = 0;
|
var tries: u32 = 0;
|
||||||
|
|||||||
+54
-19
@@ -23,6 +23,10 @@ const Node = struct {
|
|||||||
const OpenFile = struct {
|
const OpenFile = struct {
|
||||||
used: bool = false,
|
used: bool = false,
|
||||||
node: usize = 0,
|
node: usize = 0,
|
||||||
|
// The client (task id — an IPC badge is one) that opened this handle. What
|
||||||
|
// release-on-death sweeps by: a service must never depend on its clients
|
||||||
|
// cleaning up after themselves (docs/process-lifecycle.md).
|
||||||
|
owner: u32 = 0,
|
||||||
};
|
};
|
||||||
|
|
||||||
var nodes = [_]Node{.{}} ** 8;
|
var nodes = [_]Node{.{}} ** 8;
|
||||||
@@ -65,8 +69,30 @@ fn fail(out: []u8) usize {
|
|||||||
return writeReply(out, .{ .status = -1 }, &.{});
|
return writeReply(out, .{ .status = -1 }, &.{});
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Handle one request; write the reply into `out`, return its length.
|
/// Format one whole log line and emit it in a single `debug_write`, so lines from
|
||||||
fn handle(message: []const u8, out: []u8) usize {
|
/// concurrent processes can never land in the middle of it.
|
||||||
|
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||||
|
var line: [96]u8 = undefined;
|
||||||
|
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Release every open handle `client` held — called on that client's published
|
||||||
|
/// exit event. The nodes (the files) stay: ramfs contents outlive their writers,
|
||||||
|
/// only the dead client's handles go.
|
||||||
|
fn releaseClientHandles(client: u32) void {
|
||||||
|
var released: u32 = 0;
|
||||||
|
for (&opens) |*o| {
|
||||||
|
if (o.used and o.owner == client) {
|
||||||
|
o.used = false;
|
||||||
|
released += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (released != 0) writeLine("vfs: released {d} handle(s) for dead client {d}\n", .{ released, client });
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Handle one request from `sender`; write the reply into `out`, return its length.
|
||||||
|
fn handle(message: []const u8, out: []u8, sender: u32, capability: ?runtime.ipc.Handle) usize {
|
||||||
|
_ = capability;
|
||||||
if (message.len < protocol.request_size) return fail(out);
|
if (message.len < protocol.request_size) return fail(out);
|
||||||
const request = std.mem.bytesToValue(protocol.Request, message[0..protocol.request_size]);
|
const request = std.mem.bytesToValue(protocol.Request, message[0..protocol.request_size]);
|
||||||
const payload = message[protocol.request_size..];
|
const payload = message[protocol.request_size..];
|
||||||
@@ -77,7 +103,7 @@ fn handle(message: []const u8, out: []u8) usize {
|
|||||||
const ni = findNode(name) orelse createNode(name) orelse return fail(out);
|
const ni = findNode(name) orelse createNode(name) orelse return fail(out);
|
||||||
for (&opens, 0..) |*o, i| {
|
for (&opens, 0..) |*o, i| {
|
||||||
if (!o.used) {
|
if (!o.used) {
|
||||||
o.* = .{ .used = true, .node = ni };
|
o.* = .{ .used = true, .node = ni, .owner = sender };
|
||||||
return writeReply(out, .{ .status = 0, .node = i }, &.{});
|
return writeReply(out, .{ .status = 0, .node = i }, &.{});
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -113,25 +139,34 @@ fn handle(message: []const u8, out: []u8) usize {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn main() void {
|
/// Startup, under the harness: subscribe to the published exit events — when a
|
||||||
const endpoint = runtime.ipc.createIpcEndpoint() orelse {
|
/// client dies holding open handles, the exit notification is how the VFS learns
|
||||||
_ = runtime.system.write("vfs: no endpoint\n");
|
/// to release them (docs/process-lifecycle.md).
|
||||||
return;
|
fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||||
};
|
if (!runtime.process.subscribeExits(endpoint)) {
|
||||||
if (!runtime.ipc.register(.vfs, endpoint)) {
|
_ = runtime.system.write("vfs: exit subscription failed\n");
|
||||||
_ = runtime.system.write("vfs: register failed\n");
|
|
||||||
return;
|
|
||||||
}
|
}
|
||||||
_ = runtime.system.write("vfs: ready\n");
|
_ = runtime.system.write("vfs: ready\n");
|
||||||
|
return true;
|
||||||
var reply_buffer: [protocol.message_maximum]u8 = undefined;
|
|
||||||
var reply_len: usize = 0;
|
|
||||||
var receive: [protocol.message_maximum]u8 = undefined;
|
|
||||||
while (true) {
|
|
||||||
const got = runtime.ipc.replyWait(endpoint, reply_buffer[0..reply_len], &receive, null);
|
|
||||||
// Ignore notifications (none expected here); handle a request.
|
|
||||||
reply_len = handle(receive[0..got.len], &reply_buffer);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// A non-signal notification: the only kind the VFS subscribes to is exit events.
|
||||||
|
fn onNotification(badge: u64) void {
|
||||||
|
if (badge & runtime.ipc.notify_exit_bit != 0) {
|
||||||
|
releaseClientHandles(@intCast(badge & ~(runtime.ipc.notify_badge_bit | runtime.ipc.notify_exit_bit)));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn main() void {
|
||||||
|
// The harness owns the loop: requests dispatch to handle(), exit events to
|
||||||
|
// onNotification(), ping and terminate are answered for free — this service
|
||||||
|
// gained the whole lifecycle contract by deleting its hand-rolled loop.
|
||||||
|
runtime.service.run(protocol.message_maximum, .{
|
||||||
|
.service = .vfs,
|
||||||
|
.init = initialise,
|
||||||
|
.on_message = handle,
|
||||||
|
.on_notification = onNotification,
|
||||||
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
pub const panic = runtime.panic;
|
pub const panic = runtime.panic;
|
||||||
|
|||||||
+105
-1
@@ -168,7 +168,7 @@ CASES = [
|
|||||||
# Stress the big kernel lock across cores; heavier, so a longer timeout.
|
# Stress the big kernel lock across cores; heavier, so a longer timeout.
|
||||||
{"name": "smp-stress",
|
{"name": "smp-stress",
|
||||||
"smp": 4,
|
"smp": 4,
|
||||||
"timeout": 90,
|
"timeout": 150,
|
||||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
# Retry: a forced first-wake failure must still bring every core online.
|
# Retry: a forced first-wake failure must still bring every core online.
|
||||||
@@ -240,6 +240,104 @@ CASES = [
|
|||||||
"smp": 4,
|
"smp": 4,
|
||||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
|
# M17.1: a dead process's device claims are released by the reap — kill a child
|
||||||
|
# holding a claim, the device must be claimable again (process-lifecycle.md).
|
||||||
|
{"name": "claim-release",
|
||||||
|
"smp": 4,
|
||||||
|
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||||
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
|
# M17.3: published exit events — the VFS subscribes, a client dies holding an
|
||||||
|
# open handle, and the VFS releases it (process-lifecycle.md "Who learns of a death").
|
||||||
|
{"name": "vfs-client-death",
|
||||||
|
"smp": 4,
|
||||||
|
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||||
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
|
# M17.4: signals over IPC — ping, reload, terminate (clean exit), the one-shot
|
||||||
|
# timer, and the stop sequence's two endings, all driven from ring 3.
|
||||||
|
{"name": "signals",
|
||||||
|
"smp": 4,
|
||||||
|
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||||
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
|
# M18.2: bus tree reports — the xHCI driver scans its root-hub ports and
|
||||||
|
# reports both QEMU devices; the manager mirrors, prunes on the reporter's
|
||||||
|
# death, and the respawned driver re-reports (docs/device-manager.md).
|
||||||
|
{"name": "usb-report",
|
||||||
|
"smp": 4,
|
||||||
|
"timeout": 150,
|
||||||
|
"qemu_extra": ["-device", "qemu-xhci,id=xhci",
|
||||||
|
"-device", "usb-kbd,bus=xhci.0",
|
||||||
|
"-device", "usb-mouse,bus=xhci.0"],
|
||||||
|
"expect": r"device-manager: child added[\s\S]*"
|
||||||
|
r"device-manager: child added[\s\S]*"
|
||||||
|
r"device-manager: test mode: killing the reporter[\s\S]*"
|
||||||
|
r"device-manager: child removed[\s\S]*"
|
||||||
|
r"device-manager: restarting usb-xhci-bus[\s\S]*"
|
||||||
|
r"device-manager: child added",
|
||||||
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
|
# M20.1: the ring-3 AML parse (the acpi service maps the blobs and parses
|
||||||
|
# them) finds exactly the Device count the kernel's own parse produced.
|
||||||
|
{"name": "acpi-parse",
|
||||||
|
"smp": 4,
|
||||||
|
"timeout": 60,
|
||||||
|
"expect": r"acpi-parse: ok",
|
||||||
|
"fail": r"acpi-parse: mismatch|DANOS-TEST-RESULT: FAIL"},
|
||||||
|
# M20.3: the flip — ps2-bus now comes up from the acpi service's report, not
|
||||||
|
# a kernel-built node. Ordered: report -> spawn -> the driver attaches its
|
||||||
|
# keyboard, proving discovery runs entirely in ring 3 (docs/m19-m20-plan.md).
|
||||||
|
{"name": "acpi-ps2",
|
||||||
|
"smp": 4,
|
||||||
|
"timeout": 150,
|
||||||
|
"expect": r"acpi: reported PNP0303[\s\S]*"
|
||||||
|
r"device-manager: spawned ps2-bus[\s\S]*"
|
||||||
|
r"ps2-bus: keyboard driver attached",
|
||||||
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
|
# M20.2: the acpi service evaluates _CRS/_STA in ring 3 and registers +
|
||||||
|
# reports its _HID devices — the two PS/2 nodes must appear with resources
|
||||||
|
# (keyboard: io 0x60/0x64 + IRQ = 3; mouse: IRQ = 1) (docs/m19-m20-plan.md).
|
||||||
|
{"name": "acpi-report",
|
||||||
|
"smp": 4,
|
||||||
|
"timeout": 150,
|
||||||
|
"qemu_extra": ["-device", "qemu-xhci,id=xhci",
|
||||||
|
"-device", "usb-kbd,bus=xhci.0",
|
||||||
|
"-device", "usb-mouse,bus=xhci.0"],
|
||||||
|
"expect": r"acpi: reported PNP0303 \(device \d+, 3 resources\)[\s\S]*"
|
||||||
|
r"acpi: reported PNP0F13 \(device \d+, 1 resources\)",
|
||||||
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
|
# M19.1: the ring-3 PCI scan (pci-bus walks the ECAM through its mmio_map
|
||||||
|
# grant) finds exactly the functions the kernel's own walk recorded.
|
||||||
|
{"name": "pci-scan",
|
||||||
|
"smp": 4,
|
||||||
|
"timeout": 60,
|
||||||
|
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||||
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
|
# M18.3: the application surface — device-list enumerates the tree over IPC,
|
||||||
|
# subscribes (endpoint as capability), and observes the removed/added events
|
||||||
|
# the reporter's test-kill produces (docs/device-manager.md).
|
||||||
|
{"name": "device-list",
|
||||||
|
"smp": 4,
|
||||||
|
"timeout": 150,
|
||||||
|
"qemu_extra": ["-device", "qemu-xhci,id=xhci",
|
||||||
|
"-device", "usb-kbd,bus=xhci.0",
|
||||||
|
"-device", "usb-mouse,bus=xhci.0"],
|
||||||
|
"expect": r"device-list: \d+ devices[\s\S]*"
|
||||||
|
r"device-list: subscribed[\s\S]*"
|
||||||
|
r"device-manager: test mode: killing the reporter[\s\S]*"
|
||||||
|
r"device-list: removed \(device[\s\S]*"
|
||||||
|
r"device-list: added \(device",
|
||||||
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
|
# M18.1: the device manager's hello + restart policy — xHCI hellos clean and
|
||||||
|
# stays; crash-test faults, is restarted with backoff (re-claiming its device
|
||||||
|
# each time), and hits the crash-loop cap (docs/device-manager.md).
|
||||||
|
{"name": "driver-restart",
|
||||||
|
"smp": 4,
|
||||||
|
"timeout": 150,
|
||||||
|
"qemu_extra": ["-device", "qemu-xhci,id=xhci",
|
||||||
|
"-device", "usb-kbd,bus=xhci.0",
|
||||||
|
"-device", "usb-mouse,bus=xhci.0"],
|
||||||
|
"expect": r"usb-xhci-bus: hello acknowledged[\s\S]*"
|
||||||
|
r"device-manager: restarting crash-test[\s\S]*"
|
||||||
|
r"device-manager: crash-test is failing repeatedly",
|
||||||
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
# The initial_ramdisk: the loader ferries a bundle of user binaries; the kernel parses
|
# The initial_ramdisk: the loader ferries a bundle of user binaries; the kernel parses
|
||||||
# it and spawns each as a ring-3 process (here the VFS-server stub heartbeats).
|
# it and spawns each as a ring-3 process (here the VFS-server stub heartbeats).
|
||||||
{"name": "initial-ramdisk",
|
{"name": "initial-ramdisk",
|
||||||
@@ -410,6 +508,12 @@ def main():
|
|||||||
for case in selected:
|
for case in selected:
|
||||||
print(f" {case['name']:<12} ... ", end="", flush=True)
|
print(f" {case['name']:<12} ... ", end="", flush=True)
|
||||||
ok, detail = run_case(arch, case)
|
ok, detail = run_case(arch, case)
|
||||||
|
if not ok:
|
||||||
|
# Keep the evidence: serial.log is otherwise overwritten by the
|
||||||
|
# next case, and an intermittent failure's log is unrecoverable.
|
||||||
|
source = os.path.join(WORK, "serial.log")
|
||||||
|
if os.path.exists(source):
|
||||||
|
shutil.copy(source, os.path.join(WORK, f"{case['name']}-failed-serial.log"))
|
||||||
print(("PASS" if ok else "FAIL") + f" ({detail})")
|
print(("PASS" if ok else "FAIL") + f" ({detail})")
|
||||||
if not ok:
|
if not ok:
|
||||||
failures += 1
|
failures += 1
|
||||||
|
|||||||
Reference in New Issue
Block a user