From 3cc1d38dd02fa71614dc48de67c2350a8d969ff2 Mon Sep 17 00:00:00 2001 From: Daniel Samson <12231216+daniel-samson@users.noreply.github.com> Date: Mon, 13 Jul 2026 00:19:30 +0100 Subject: [PATCH] The device manager supervises: hello, backoff, and the crash-loop cap (M18.1) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The manager is now a harness service on the well-known .device_manager endpoint. Every driver spawns supervised; drivers with an assignment must hello (device-manager-protocol, versioned) within a deadline enforced by a timer sweep. Exit reasons drive the restart decision: clean exits stay down, faults restart with 300/600/1200ms backoff, and three fast deaths mark a driver failed instead of respawning forever. usb-xhci-bus is the first conforming driver; the crash-test fixture claims a device, hellos, and faults on purpose — each respawn re-proving claim release on death through the manager's own path. maximum_tasks grows 16 -> 32: the initial-ramdisk sweep (15 binaries at once) was intermittently overflowing the static pool. --- build.zig | 12 + docs/device-manager.md | 7 +- docs/m17-m18-plan.md | 8 +- library/runtime/runtime.zig | 3 + system/abi.zig | 1 + system/drivers/usb-xhci-bus/usb-xhci-bus.zig | 99 +++++-- system/kernel/tests.zig | 40 +++ system/parameters.zig | 7 +- system/services/crash-test/crash-test.zig | 44 +++ .../device-manager-protocol.zig | 58 ++++ .../device-manager/device-manager.zig | 275 +++++++++++++++--- test/qemu_test.py | 13 + 12 files changed, 495 insertions(+), 72 deletions(-) create mode 100644 system/services/crash-test/crash-test.zig create mode 100644 system/services/device-manager/device-manager-protocol.zig diff --git a/build.zig b/build.zig index 269bef1..bae8552 100644 --- a/build.zig +++ b/build.zig @@ -212,6 +212,13 @@ pub fn build(b: *std.Build) void { }, }); + // The device-manager protocol: hello + (M18.2) tree reports, exposed as its + // own module like the other protocol modules. Imported through the runtime. + const device_manager_protocol_module = b.addModule("device-manager-protocol", .{ + .root_source_file = b.path("system/services/device-manager/device-manager-protocol.zig"), + }); + runtime_module.addImport("device-manager-protocol", device_manager_protocol_module); + // Typed volatile MMIO register access + memory-ordering barriers, for drivers on // top of an mmio_map grant. Depends only on `builtin` (arch-conditional barriers); // no target set, so it inherits each driver's. See library/mmio/mmio.zig. @@ -332,6 +339,9 @@ pub fn build(b: *std.Build) void { const ps2_keyboard_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-keyboard", "system/drivers/ps2-bus/keyboard.zig"); const ps2_mouse_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-mouse", "system/drivers/ps2-bus/mouse.zig"); const usb_xhci_bus_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "usb-xhci-bus", "system/drivers/usb-xhci-bus/usb-xhci-bus.zig"); + // A test fixture, not a real driver: hellos to the device manager, then faults — + // what the driver-restart scenario drives the crash-loop cap with. + const crash_test_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "crash-test", "system/services/crash-test/crash-test.zig"); const device_manager_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "device-manager", "system/services/device-manager/device-manager.zig"); // The input service and its exercisers: the fan-out server, a hardware-free synthetic // source, and a subscriber that doubles as the `input` test's oracle. See docs/input.md. @@ -363,6 +373,8 @@ pub fn build(b: *std.Build) void { mk_run.addFileArg(ps2_mouse_exe.getEmittedBin()); mk_run.addArg("usb-xhci-bus"); mk_run.addFileArg(usb_xhci_bus_exe.getEmittedBin()); + mk_run.addArg("crash-test"); + mk_run.addFileArg(crash_test_exe.getEmittedBin()); mk_run.addArg("device-manager"); mk_run.addFileArg(device_manager_exe.getEmittedBin()); mk_run.addArg("input"); diff --git a/docs/device-manager.md b/docs/device-manager.md index aecc3ee..e38eea0 100644 --- a/docs/device-manager.md +++ b/docs/device-manager.md @@ -1,6 +1,11 @@ # The device manager -**Status: design.** The primitives this builds on are real ([process-management.md](process-management.md): +**Status: the protocol and supervision are built** (M18.1, 2026-07-13): `hello` +with its deadline, supervised spawn, restart with backoff, and the crash-loop +cap are in — usb-xhci-bus is the first conforming driver, and the +`driver-restart` scenario proves fault → backoff → re-claim → cap end to end. +Tree reports (M18.2) and the application surface (M18.3) remain design. The +primitives underneath are real ([process-management.md](process-management.md): spawn/supervise/kill/exit-notification; [driver-model.md](driver-model.md): the device table as a capability system; [drivers.md](drivers.md): claim/map/IRQ), and the first per-device driver spawn works (the device manager matches the xHCI controller by PCI diff --git a/docs/m17-m18-plan.md b/docs/m17-m18-plan.md index 53dd664..552e414 100644 --- a/docs/m17-m18-plan.md +++ b/docs/m17-m18-plan.md @@ -45,7 +45,13 @@ only when its definition of green holds. runtime.service.run with the zero-length ping; VFS converted; `signals` scenario; suite 51/51) - [x] **merge** `feat/process-lifecycle` → main, push (merged 2026-07-13) -- [ ] **M18.1** — device-manager protocol: hello + restart policy (branch `feat/device-manager`) +- [x] **M18.1** — device-manager protocol: hello + restart policy + (device-manager-protocol module; the manager as a harness service: + supervised spawns, hello deadline via timer sweep, restart with + 300/600/1200ms backoff, exit reasons deciding restart-vs-stopped, + crash-loop cap; usb-xhci-bus first conforming driver; crash-test fixture + re-proving claim release each respawn; `driver-restart` scenario; + maximum_tasks 16→32 — the sweep was overflowing the pool; suite 52/52) - [ ] **merge** `feat/device-manager` → main, push - [ ] **M18.2** — xHCI port scan + tree reports (branch `feat/usb-xhci-bus`) - [ ] **M18.3** — app surface: enumerate/subscribe + device-list diff --git a/library/runtime/runtime.zig b/library/runtime/runtime.zig index 2ff1e5c..29b1895 100644 --- a/library/runtime/runtime.zig +++ b/library/runtime/runtime.zig @@ -17,6 +17,9 @@ pub const ipc = @import("ipc.zig"); pub const start = @import("start.zig"); /// The VFS wire protocol (shared with the VFS server). pub const vfs_protocol = @import("vfs-protocol"); + +/// The device-manager protocol: hello + tree reports (docs/device-manager.md). +pub const device_manager_protocol = @import("device-manager-protocol"); /// Keyboard-event listening (subscribe/next) and broadcasting (publish), over the input /// service. See library/runtime/input.zig and system/services/input/. pub const input = @import("input.zig"); diff --git a/system/abi.zig b/system/abi.zig index 78b2b35..6e6e985 100644 --- a/system/abi.zig +++ b/system/abi.zig @@ -176,6 +176,7 @@ pub const ServiceId = enum(u32) { vfs = 1, input = 2, ps2_bus = 3, // the 8042 owner; child device drivers attach here for raw bytes + device_manager = 4, // the tree, the matcher, the supervisor (docs/device-manager.md) _, }; diff --git a/system/drivers/usb-xhci-bus/usb-xhci-bus.zig b/system/drivers/usb-xhci-bus/usb-xhci-bus.zig index f13694a..cb23480 100644 --- a/system/drivers/usb-xhci-bus/usb-xhci-bus.zig +++ b/system/drivers/usb-xhci-bus/usb-xhci-bus.zig @@ -1,58 +1,59 @@ //! /system/drivers/usb-xhci-bus — the xHCI (USB 3) host-controller bus driver. -//! The device manager spawns **one instance per controller** it discovers (a machine -//! can carry several), passing the controller's device-tree id as argv[1]; this -//! instance claims that device and no other, so multiple instances never fight over -//! hardware. This increment proves the plumbing: parse the id, claim the controller, -//! and report its MMIO window. The next increments map the registers and bring the -//! controller up (reset, rings, port scan), then enumerate the USB devices on the -//! bus with the usb-abi request builders and publish each with `device_register`. +//! The device manager spawns **one instance per controller** it discovers (a +//! machine can carry several), passing the controller's device-tree id as +//! argv[1]; this instance claims that device and no other, so multiple +//! instances never fight over hardware. +//! +//! M18.1 (this increment): a harness service and the first conforming driver of +//! the device-manager protocol — claim the controller, `hello` the manager +//! (role, version, assignment) inside its deadline, then serve. Controller +//! bring-up (map the MMIO window, reset, port scan) and tree reports +//! (`child_added` for each connected port) land in M18.2. const std = @import("std"); const runtime = @import("runtime"); +const protocol = runtime.device_manager_protocol; const device = runtime.device; -/// Format one whole log line and emit it in a single `debug_write`, so concurrent -/// instances (one per controller) can never interleave mid-line. +/// Format one whole log line and emit it in a single `debug_write`, so +/// concurrent instances (one per controller) can never interleave mid-line. fn writeLine(comptime fmt: []const u8, arguments: anytype) void { var line: [128]u8 = undefined; _ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return); } -pub fn main(init: runtime.process.Init) void { - const argument = init.arguments.get(1) orelse { - _ = runtime.system.write("usb-xhci-bus: missing controller device id (argv[1])\n"); - return; - }; - const controller_id = std.fmt.parseInt(u64, argument, 10) catch { - writeLine("usb-xhci-bus: malformed controller device id '{s}'\n", .{argument}); - return; - }; +var controller_id: u64 = protocol.no_device; +/// Claim the assigned controller, find its register window, and hello the +/// manager. Any failure returns false: the process exits cleanly, which the +/// manager reads as "meant to stop" — a missing assignment is not a crash loop. +fn initialise(endpoint: runtime.ipc.Handle) bool { + _ = endpoint; if (!device.claim(controller_id)) { writeLine("usb-xhci-bus: unable to claim controller device {d}\n", .{controller_id}); - return; + return false; } // Fetch our own descriptor back for the controller's resources. const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch { _ = runtime.system.write("usb-xhci-bus: out of memory\n"); - return; + return false; }; const total = device.enumerate(buffer); const descriptor = for (buffer[0..@min(total, buffer.len)]) |d| { if (d.id == controller_id) break d; } else { writeLine("usb-xhci-bus: device {d} not in the device tree\n", .{controller_id}); - return; + return false; }; - // The controller's operational registers live behind BAR0, enumerated as the - // device's first memory resource. + // The controller's operational registers live behind BAR0, enumerated as + // the device's first memory resource. const register_window = for (descriptor.resources[0..@intCast(descriptor.resource_count)]) |resource| { if (resource.kind == @intFromEnum(device.ResourceKind.memory)) break resource; } else { writeLine("usb-xhci-bus: controller device {d} has no MMIO window\n", .{controller_id}); - return; + return false; }; writeLine("usb-xhci-bus: claimed controller device {d} (registers at 0x{x}, {d} bytes)\n", .{ controller_id, @@ -60,9 +61,53 @@ pub fn main(init: runtime.process.Init) void { register_window.len, }); - // Controller bring-up (map the window, reset, rings, port scan) is the next - // increment; stay resident as the bus's supervisor in the meantime. - while (true) runtime.system.sleep(1000); + // The handshake: role, protocol version, assignment — inside the manager's + // deadline (the lookup retries cover the manager still registering). + var manager: ?runtime.ipc.Handle = null; + var tries: u32 = 0; + while (manager == null and tries < 100) : (tries += 1) { + manager = runtime.ipc.lookup(.device_manager); + if (manager == null) runtime.system.sleep(20); + } + const h = manager orelse { + _ = runtime.system.write("usb-xhci-bus: no device manager to hello\n"); + return false; + }; + const hello = protocol.Hello{ .role = @intFromEnum(protocol.Role.bus), .device_id = controller_id }; + var reply: [protocol.message_maximum]u8 = undefined; + const n = runtime.ipc.call(h, std.mem.asBytes(&hello), &reply) catch { + _ = runtime.system.write("usb-xhci-bus: hello call failed\n"); + return false; + }; + if (n < protocol.reply_size or std.mem.bytesToValue(protocol.HelloReply, reply[0..protocol.reply_size]).status != 0) { + _ = runtime.system.write("usb-xhci-bus: hello refused\n"); + return false; + } + _ = runtime.system.write("usb-xhci-bus: hello acknowledged\n"); + return true; +} + +/// No bus protocol to serve yet — transfer requests arrive with the USB track. +fn onMessage(message: []const u8, reply: []u8, sender: u32) usize { + _ = message; + _ = reply; + _ = sender; + return 0; +} + +pub fn main(init: runtime.process.Init) void { + const argument = init.arguments.get(1) orelse { + _ = runtime.system.write("usb-xhci-bus: missing controller device id (argv[1])\n"); + return; + }; + controller_id = std.fmt.parseInt(u64, argument, 10) catch { + writeLine("usb-xhci-bus: malformed controller device id '{s}'\n", .{argument}); + return; + }; + runtime.service.run(protocol.message_maximum, .{ + .init = initialise, + .on_message = onMessage, + }); } pub const panic = runtime.panic; diff --git a/system/kernel/tests.zig b/system/kernel/tests.zig index 499ad0a..44cd12b 100644 --- a/system/kernel/tests.zig +++ b/system/kernel/tests.zig @@ -138,6 +138,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void { vfsClientDeathTest(boot_information); } else if (eql(case, "signals")) { signalsTest(boot_information); + } else if (eql(case, "driver-restart")) { + driverRestartTest(boot_information); } else if (eql(case, "initial-ramdisk")) { initialRamdiskTest(boot_information); } else if (eql(case, "vfs")) { @@ -1654,6 +1656,44 @@ fn signalsTest(boot_information: *const BootInformation) void { result(); } +/// M18.1: the device manager's restart machinery, end to end. In test-restart +/// mode the manager also supervises crash-test: a fixture that claims device 0, +/// hellos, and faults. The scenario asserts three markers in order — the real +/// xHCI driver hellos clean and stays; crash-test is restarted with backoff +/// (each respawn re-claiming the device the dead instance held, M17.1 through +/// the manager's path); the crash loop caps and the manager gives up. +fn driverRestartTest(boot_information: *const BootInformation) void { + log("DANOS-TEST-BEGIN: driver-restart\n", .{}); + if (boot_information.initial_ramdisk_len == 0) { + check("bootloader handed over an initial_ramdisk", false); + result(); + return; + } + const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len]; + const rd = initial_ramdisk.Reader.init(image) orelse { + check("initial_ramdisk image is valid", false); + result(); + return; + }; + + process.setInitialRamdisk(image); // the manager system_spawns drivers by name + process.write_count = 0; + var manager: u32 = 0; + var i: u32 = 0; + while (i < rd.count) : (i += 1) { + const item = rd.entry(i) orelse continue; + if (!eql(item.name, "device-manager")) continue; + manager = process.spawnProcessSupervised(item.blob, 4, &.{ "device-manager", "test-restart" }, scheduler.currentId(), null) catch 0; + break; + } + check("device-manager spawned in test-restart mode", manager != 0); + // The assertions live in the harness: its expect regex requires, in order, + // the xHCI hello ack, a crash-test restart, and the crash-loop cap — read + // from the whole serial capture, immune to the transient-line races a + // write_buffer poll would have here (many processes log concurrently). + result(); +} + /// The whole user-side surface at once: spawn process-test's supervisor role, /// which — entirely from ring 3 — creates an exit endpoint, spawns its two /// children supervised, sees them in process_enumerate, kills them (one blocked, diff --git a/system/parameters.zig b/system/parameters.zig index a17e556..e6517af 100644 --- a/system/parameters.zig +++ b/system/parameters.zig @@ -16,8 +16,11 @@ pub const maximum_cpus = 128; /// Maximum tasks (kernel threads) alive at once — the static task-table size. Each -/// online core consumes one slot for its idle task, plus task 0 on the BSP. -pub const maximum_tasks = 16; +/// online core consumes one slot for its idle task, plus task 0 on the BSP. Sized +/// for the initial-ramdisk sweep (15 bundled binaries spawned at once) plus the +/// device manager's supervised children with room to grow — at 16 the sweep +/// started failing spawns once the bundle passed a dozen binaries. +pub const maximum_tasks = 32; /// Each task's kernel stack (also each AP's bring-up stack), in bytes. pub const kernel_stack_size = 16 * 1024; diff --git a/system/services/crash-test/crash-test.zig b/system/services/crash-test/crash-test.zig new file mode 100644 index 0000000..9155b24 --- /dev/null +++ b/system/services/crash-test/crash-test.zig @@ -0,0 +1,44 @@ +//! crash-test — a test fixture, not a driver: claims the device it is assigned, +//! hellos the device manager, announces itself, then faults on purpose. The +//! driver-restart scenario drives the manager's whole restart machinery with +//! it: fault → exit reason → backoff → respawn → the **same claim succeeding +//! again** (claim release on death, M17.1, through the manager's path) → the +//! crash-loop cap. Spawned bare (the initial-ramdisk sweep starts every bundled +//! binary), it exits silently so it cannot derange other tests. + +const std = @import("std"); +const runtime = @import("runtime"); +const protocol = runtime.device_manager_protocol; + +pub fn main(init: runtime.process.Init) void { + const argument = init.arguments.get(1) orelse return; // bare: stay silent + const assigned = std.fmt.parseInt(u64, argument, 10) catch return; + + // The respawn only reaches this line because the kernel released the + // previous instance's claim at death. A failed claim exits cleanly — the + // manager reads "meant to stop" and the scenario fails loudly by silence. + if (!runtime.device.claim(assigned)) { + _ = runtime.system.write("crash-test: claim failed\n"); + return; + } + + var manager: ?runtime.ipc.Handle = null; + var tries: u32 = 0; + while (manager == null and tries < 100) : (tries += 1) { + manager = runtime.ipc.lookup(.device_manager); + if (manager == null) runtime.system.sleep(20); + } + const h = manager orelse return; + const hello = protocol.Hello{ .role = @intFromEnum(protocol.Role.device), .device_id = assigned }; + var reply: [protocol.message_maximum]u8 = undefined; + _ = runtime.ipc.call(h, std.mem.asBytes(&hello), &reply) catch return; + + _ = runtime.system.write("crash-test: faulting now\n"); + const poison: *volatile u32 = @ptrFromInt(0xdead0000); + poison.* = 1; // the restart machinery's fuel: a real segmentation fault +} + +pub const panic = runtime.panic; +comptime { + _ = &runtime.start._start; // pull the runtime entry shim into the image +} diff --git a/system/services/device-manager/device-manager-protocol.zig b/system/services/device-manager/device-manager-protocol.zig new file mode 100644 index 0000000..d53a50f --- /dev/null +++ b/system/services/device-manager/device-manager-protocol.zig @@ -0,0 +1,58 @@ +//! The device-manager protocol (docs/device-manager.md): what drivers and +//! applications say to the device manager over its well-known endpoint. The +//! vfs-protocol pattern — extern-struct messages, a version in the handshake, +//! reserved fields — so both sides depend on the contract by name. Deliberately +//! contains nothing lifecycle-shaped: stopping, liveness (the zero-length ping), +//! and exit reasons are the universal vocabulary of +//! docs/process-lifecycle.md, not this protocol. + +/// The protocol version a driver states in its hello. A manager that cannot +/// serve a driver's version refuses the hello, and the mismatch is loud at +/// startup instead of quiet corruption later. +pub const version: u16 = 1; + +/// What kind of driver is talking (docs/driver-model.md's shapes). +pub const Role = enum(u8) { + /// Owns a controller and reports the devices behind it (`child_added`). + bus = 1, + /// Serves one device, reached through a bus's transfer protocol. + device = 2, +}; + +/// The message kinds. `child_added`/`child_removed` land in M18.2; +/// `enumerate`/`subscribe` in M18.3. +pub const Operation = enum(u8) { + hello = 1, +}; + +/// `Hello.device_id` for a driver that serves no enumerated device (a test +/// fixture, a synthetic source). +pub const no_device: u64 = ~@as(u64, 0); + +/// The handshake, sent once by every driver the manager spawns — the manager's +/// one self-enforced deadline: spawned and silent past it means wrong binary, +/// wrong version, or wedged before main, and the stop sequence follows. +pub const Hello = extern struct { + operation: u8 = @intFromEnum(Operation.hello), + /// A Role value. + role: u8, + /// The protocol version this driver was built against (`version`). + version: u16 = version, + reserved: u32 = 0, + /// The device this driver was assigned (its argv[1]), or `no_device`. + device_id: u64, +}; + +pub const hello_size = @sizeOf(Hello); + +/// The manager's answer to a hello. Nonzero status = refused (version mismatch, +/// unknown sender); a refused driver should exit cleanly. +pub const HelloReply = extern struct { + status: i32, + reserved: u32 = 0, +}; + +pub const reply_size = @sizeOf(HelloReply); + +/// Upper bound on any message in this protocol — sizes the endpoint buffers. +pub const message_maximum = 64; diff --git a/system/services/device-manager/device-manager.zig b/system/services/device-manager/device-manager.zig index 82310c3..63c0ff2 100644 --- a/system/services/device-manager/device-manager.zig +++ b/system/services/device-manager/device-manager.zig @@ -1,20 +1,24 @@ //! /system/services/device-manager — the ring-3 process that turns the device -//! tree into a running system. The kernel enumerates the hardware and enforces the -//! claim capability (mechanism); this decides *which driver serves which device* -//! and, eventually, spawns it (policy). Keeping that split in user space is the -//! whole point of the microkernel: the manager is an ordinary, restartable process -//! with no special privilege — it uses the same `device_*` system calls any process -//! could ([drivers.md](../../../docs/drivers.md), [driver-model.md]). +//! tree into a running system: **the matcher and the supervisor** +//! (docs/device-manager.md). The kernel enumerates the hardware and enforces the +//! claim capability (mechanism); this decides which driver serves which device, +//! spawns it, and keeps it alive (policy). Keeping that split in user space is +//! the whole point of the microkernel: the manager is an ordinary, restartable +//! process with no special privilege. //! -//! Increment 2 (this file): enumerate /system/devices, *match* each device to a -//! driver, and *spawn* it with `system_spawn` — the kernel loads the named binary -//! from the initial-ramdisk as a fresh ring-3 process. On QEMU this discovers the -//! HPET, decides `hpet` serves it, and brings that driver all the way up. (The -//! kernel still auto-spawns the whole initial-ramdisk at boot; increment 3 removes -//! that redundancy so the manager is the sole owner of driver spawning.) +//! M18.1 (this increment): the manager is a harness service on the well-known +//! `.device_manager` endpoint. Every driver is spawned **supervised** — exit +//! notifications land in the same loop as protocol messages. Drivers with an +//! assignment must `hello` within a deadline or be stopped; a driver that dies +//! is restarted with backoff, and a crash loop (three fast deaths) marks it +//! failed instead of respawning forever. Exit reasons (M17.2) drive the +//! decision: a clean exit meant to stop; only faults and missed deadlines +//! restart. Tree reports (`child_added`) land in M18.2. + const std = @import("std"); const runtime = @import("runtime"); const acpi_ids = @import("acpi-ids"); +const protocol = runtime.device_manager_protocol; const device = runtime.device; const system = runtime.system; @@ -27,9 +31,8 @@ fn writeLine(comptime fmt: []const u8, arguments: anytype) void { } /// The driver that serves each device — the policy table. In a fuller system -/// this comes from the drivers describing what they bind (or a manifest under -/// /system/drivers); for now it is a small static map, which is enough to prove the -/// manager reads the tree and decides. `null` = no driver for this class yet. +/// this comes from a manifest (docs/device-manager.md: the third bus type +/// triggers it); for now a static map. `null` = no driver for this class yet. fn driverFor(d: device.DeviceDescriptor) ?[]const u8 { // detect device via DeviceClass if (d.class == @intFromEnum(device.DeviceClass.timer)) return "hpet"; @@ -47,10 +50,9 @@ fn driverFor(d: device.DeviceDescriptor) ?[]const u8 { /// pci-class.zig decodes. const xhci_pci_class: u64 = 0x0C_03_30; -/// The bus driver that serves a PCI function, or null. Unlike the singleton drivers -/// in `driverFor`, a machine can carry several identical controllers — so the caller -/// spawns one driver instance *per device*, passing the device id as argv[1] for the -/// instance to claim. +/// The bus driver that serves a PCI function, or null. A machine can carry +/// several identical controllers — one driver instance per device, the id as +/// argv[1]. These drivers speak the protocol: a hello is expected. fn pciDriverFor(d: device.DeviceDescriptor) ?[]const u8 { if (d.class != @intFromEnum(device.DeviceClass.pci_device)) return null; return switch (d.pci_class) { @@ -59,24 +61,169 @@ fn pciDriverFor(d: device.DeviceDescriptor) ?[]const u8 { }; } -/// Spawn one instance of `driver_name` to serve the specific device `id` — the id -/// arrives as argv[1]. No isProcessRunning gate here: the name alone cannot tell two -/// instances apart, and this manager is the sole spawner of drivers. -fn spawnForDevice(driver_name: []const u8, id: u64) void { - var text: [20]u8 = undefined; - const id_text = std.fmt.bufPrint(&text, "{d}", .{id}) catch return; - if (system.spawnWithArguments(driver_name, &.{id_text}) != null) { - writeLine("device-manager: spawned {s} for device {d}\n", .{ driver_name, id }); +// --- supervision ------------------------------------------------------------- + +/// How long a protocol driver has to hello after its spawn. +const hello_deadline_ms: u64 = 3000; +/// Deaths faster than this count toward the crash loop; slower ones reset it. +const fast_death_ns: u64 = 2_000_000_000; +/// Consecutive fast deaths before the manager gives up on a driver. +const crash_loop_cap: u32 = 3; +/// Restart backoff: base << (restarts - 1), so 300 ms, 600 ms, 1200 ms. +const backoff_base_ms: u64 = 300; + +const DriverState = enum { + awaiting_hello, // spawned; the deadline is armed (protocol drivers only) + running, + restarting, // dead; respawn due at restart_due_ns + stopped, // exited cleanly — it meant to; not restarted + failed, // crash loop, or unspawnable; the manager gave up +}; + +const Driver = struct { + used: bool = false, + name_buffer: [24]u8 = undefined, + name_len: usize = 0, + // The assigned device id (becomes argv[1]), or protocol.no_device. + device_id: u64 = protocol.no_device, + // Whether this driver speaks the protocol (hello expected, deadline + // enforced). Legacy drivers (hpet, ps2-bus) are supervised and restarted + // but not yet required to hello. + speaks_protocol: bool = false, + process_id: u32 = 0, + state: DriverState = .running, + restarts: u32 = 0, + spawn_ns: u64 = 0, + hello_deadline_ns: u64 = 0, + restart_due_ns: u64 = 0, + + fn name(driver: *const Driver) []const u8 { + return driver.name_buffer[0..driver.name_len]; + } +}; + +const maximum_drivers = 16; +var drivers: [maximum_drivers]Driver = .{Driver{}} ** maximum_drivers; +var manager_endpoint: runtime.ipc.Handle = 0; +var test_restart_mode = false; + +fn driverByProcess(process_id: u32) ?*Driver { + for (&drivers) |*driver| { + if (driver.used and driver.process_id == process_id) return driver; + } + return null; +} + +/// Whether a singleton driver is already in the table (two ACPI nodes can both +/// map to ps2-bus; one instance serves both). +fn alreadySupervised(name: []const u8) bool { + for (&drivers) |*driver| { + if (driver.used and std.mem.eql(u8, driver.name(), name)) return true; + } + return false; +} + +/// Record a driver in the table and spawn its first instance. +fn addDriver(name: []const u8, device_id: u64, speaks_protocol: bool) void { + for (&drivers) |*driver| { + if (driver.used) continue; + const n = @min(name.len, driver.name_buffer.len); + @memcpy(driver.name_buffer[0..n], name[0..n]); + driver.name_len = n; + driver.device_id = device_id; + driver.speaks_protocol = speaks_protocol; + driver.used = true; + spawnDriver(driver); + return; + } + writeLine("device-manager: driver table full; cannot supervise {s}\n", .{name}); +} + +/// (Re)spawn a driver instance: supervised on the manager's own endpoint, the +/// device id as argv[1] when it has one, the hello deadline armed when it +/// speaks the protocol. +fn spawnDriver(driver: *Driver) void { + var id_text: [20]u8 = undefined; + var arguments: [1][]const u8 = undefined; + var argument_count: usize = 0; + if (driver.device_id != protocol.no_device) { + arguments[0] = std.fmt.bufPrint(&id_text, "{d}", .{driver.device_id}) catch return; + argument_count = 1; + } + const child = system.spawnSupervised(driver.name(), arguments[0..argument_count], manager_endpoint) orelse { + writeLine("device-manager: failed to spawn {s}\n", .{driver.name()}); + driver.state = .failed; + return; + }; + driver.process_id = child; + driver.spawn_ns = system.clock(); + if (driver.speaks_protocol) { + driver.state = .awaiting_hello; + driver.hello_deadline_ns = driver.spawn_ns + hello_deadline_ms * 1_000_000; + _ = system.timerOnce(manager_endpoint, hello_deadline_ms + 100); } else { - writeLine("device-manager: failed to spawn {s} for device {d}\n", .{ driver_name, id }); + driver.state = .running; + } + if (driver.device_id != protocol.no_device) { + writeLine("device-manager: spawned {s} for device {d}\n", .{ driver.name(), driver.device_id }); + } else { + writeLine("device-manager: spawned {s}\n", .{driver.name()}); } } -pub fn main() void { +/// A driver died. The exit reason (M17.2) is the whole decision: a clean exit +/// meant to stop; anything else restarts with backoff until the crash-loop cap. +fn onDriverExit(driver: *Driver) void { + const reason = runtime.process.exitReason(driver.process_id) orelse .fault; + if (reason == .exited) { + driver.state = .stopped; + writeLine("device-manager: {s} exited cleanly; not restarting\n", .{driver.name()}); + return; + } + const now = system.clock(); + const alive_ns = now - driver.spawn_ns; + driver.restarts = if (alive_ns < fast_death_ns) driver.restarts + 1 else 1; + if (driver.restarts >= crash_loop_cap) { + driver.state = .failed; + writeLine("device-manager: {s} is failing repeatedly (crash loop); giving up\n", .{driver.name()}); + return; + } + const delay_ms = backoff_base_ms << @intCast(driver.restarts - 1); + driver.state = .restarting; + driver.restart_due_ns = now + delay_ms * 1_000_000; + writeLine("device-manager: restarting {s} in {d} ms (died: {s})\n", .{ driver.name(), delay_ms, @tagName(reason) }); + _ = system.timerOnce(manager_endpoint, delay_ms + 50); +} + +/// A timer landed: sweep every deadline. Overdue hellos are killed (the exit +/// notification then routes through the normal restart policy); due restarts +/// respawn. Timers carry no id on purpose — the table is the state, and one +/// sweep serves every armed deadline. +fn sweepDeadlines() void { + const now = system.clock(); + for (&drivers) |*driver| { + if (!driver.used) continue; + switch (driver.state) { + .awaiting_hello => if (now >= driver.hello_deadline_ns) { + writeLine("device-manager: {s} missed its hello deadline\n", .{driver.name()}); + _ = system.kill(driver.process_id); + // The exit notification finishes the job via onDriverExit. + }, + .restarting => if (now >= driver.restart_due_ns) spawnDriver(driver), + else => {}, + } + } +} + +// --- the harness callbacks ----------------------------------------------------- + +fn initialise(endpoint: runtime.ipc.Handle) bool { + manager_endpoint = endpoint; + // Enumerate into a heap buffer (too big for the one-page user stack). const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch { _ = runtime.system.write("device-manager: out of memory\n"); - return; + return false; }; const total = device.enumerate(buffer); const count = @min(total, buffer.len); @@ -85,28 +232,74 @@ pub fn main() void { for (buffer[0..count]) |descriptor| { if (pciDriverFor(descriptor)) |driver_name| { matched += 1; - spawnForDevice(driver_name, descriptor.id); + addDriver(driver_name, descriptor.id, true); continue; } const driver_name = driverFor(descriptor) orelse continue; matched += 1; - if (!system.isProcessRunning(driver_name)) { - if (runtime.system.spawn(driver_name) != null) { - writeLine("device-manager: spawned {s}\n", .{driver_name}); - } else { - writeLine("device-manager: failed to spawn {s}\n", .{driver_name}); - } - } else { - writeLine("device-manager: already spawned {s}\n", .{driver_name}); + // Skip a singleton that is already alive (the initial-ramdisk sweep test + // starts every bundled binary bare, this manager included) — spawning a + // second instance would only lose the claim race and churn the log. + if (!alreadySupervised(driver_name) and !system.isProcessRunning(driver_name)) { + addDriver(driver_name, protocol.no_device, false); } } + if (test_restart_mode) { + // The driver-restart scenario's fixture: claims device 0 (the tree + // root, otherwise unclaimed), hellos, then faults — driving backoff, + // re-claim-after-death, and the crash-loop cap deterministically. + addDriver("crash-test", 0, true); + } + if (matched == 0) { _ = runtime.system.write("device-manager: no matchable devices\n"); + } else { + _ = runtime.system.write("device-manager: ok\n"); + } + return true; +} + +fn onMessage(message: []const u8, reply: []u8, sender: u32) usize { + if (message.len < protocol.hello_size) return 0; + const hello = std.mem.bytesToValue(protocol.Hello, message[0..protocol.hello_size]); + if (hello.operation != @intFromEnum(protocol.Operation.hello)) return 0; + + var status: i32 = 0; + if (hello.version != protocol.version) { + status = -1; + writeLine("device-manager: refused hello (version {d}) from process {d}\n", .{ hello.version, sender }); + } else if (driverByProcess(sender)) |driver| { + driver.state = .running; + writeLine("device-manager: hello from {s} (device {d})\n", .{ driver.name(), hello.device_id }); + } else { + status = -1; + writeLine("device-manager: hello from unknown process {d}\n", .{sender}); + } + const hello_reply = protocol.HelloReply{ .status = status }; + @memcpy(reply[0..protocol.reply_size], std.mem.asBytes(&hello_reply)); + return protocol.reply_size; +} + +fn onNotification(badge: u64) void { + if (badge & runtime.ipc.notify_exit_bit != 0) { + const dead: u32 = @intCast(badge & ~(runtime.ipc.notify_badge_bit | runtime.ipc.notify_exit_bit)); + if (driverByProcess(dead)) |driver| onDriverExit(driver); return; } - _ = runtime.system.write("device-manager: ok\n"); - while (true) runtime.system.sleep(1000); + if (badge & runtime.ipc.notify_timer_bit != 0) sweepDeadlines(); +} + +pub fn main(init: runtime.process.Init) void { + if (init.arguments.get(1)) |mode| { + test_restart_mode = std.mem.eql(u8, mode, "test-restart"); + } + runtime.service.run(protocol.message_maximum, .{ + .service = .device_manager, + .init = initialise, + .on_message = onMessage, + .on_notification = onNotification, + }); } pub const panic = runtime.panic; diff --git a/test/qemu_test.py b/test/qemu_test.py index 40007c3..e9eec0c 100644 --- a/test/qemu_test.py +++ b/test/qemu_test.py @@ -258,6 +258,19 @@ CASES = [ "smp": 4, "expect": r"DANOS-TEST-RESULT: PASS", "fail": r"DANOS-TEST-RESULT: FAIL"}, + # M18.1: the device manager's hello + restart policy — xHCI hellos clean and + # stays; crash-test faults, is restarted with backoff (re-claiming its device + # each time), and hits the crash-loop cap (docs/device-manager.md). + {"name": "driver-restart", + "smp": 4, + "timeout": 90, + "qemu_extra": ["-device", "qemu-xhci,id=xhci", + "-device", "usb-kbd,bus=xhci.0", + "-device", "usb-mouse,bus=xhci.0"], + "expect": r"usb-xhci-bus: hello acknowledged[\s\S]*" + r"device-manager: restarting crash-test[\s\S]*" + r"device-manager: crash-test is failing repeatedly", + "fail": r"DANOS-TEST-RESULT: FAIL"}, # The initial_ramdisk: the loader ferries a bundle of user binaries; the kernel parses # it and spawns each as a ring-3 process (here the VFS-server stub heartbeats). {"name": "initial-ramdisk",