Files
danos/system/services/device-manager/device-manager.zig
T
Daniel Samson 5dc966838a volume-manager: rebuild a volume when its storage driver dies (S5)
The V4 review's open edge: a storage driver that crashes while its device
stays in the tree left fat wedged on a dead channel — device-presence
polling (a device-manager enumerate) still reported the device present,
so nothing reaped it. pollTick now also probes channelAlive(dev), a
geometry() on the block channel that fails fast on the dead endpoint; a
present device with a dead channel is reaped like a pull, and the adopt
loop re-adopts it on the restarted driver's fresh channel — the rebuild.
The manager's own liveness probe makes fat self-detection unnecessary:
it rebuilds regardless of the wedged filesystem's state.

The drill: the device manager gains a test-storage-restart mode that
kills usb-storage once, ~2s after its hello (post-mount); a new
volume-driver-restart kernel case boots a manual tree with it, and the
QEMU case asserts a SECOND mount of the same id-path after the reap —
the rebuild. A pre-S5 manager, checking only device presence, never
reaps, so the second mount never appears. The manually-spawned volume
manager needed kernel-supervisor protocol grants (bind its name, open
the device manager), as the other manual-tree services already have.

Full suite 133/133.
2026-08-10 05:21:48 +01:00

788 lines
38 KiB
Zig

//! /system/services/device-manager — the ring-3 process that turns the device
//! tree into a running system: **the matcher and the supervisor**
//! (docs/device-manager.md). The kernel enumerates the hardware and enforces the
//! claim capability (mechanism); this decides which driver serves which device,
//! spawns it, and keeps it alive (policy). Keeping that split in user space is
//! the whole point of the microkernel: the manager is an ordinary, restartable
//! process with no special privilege.
//!
//! M18.1 (this increment): the manager is a harness service on the well-known
//! `.device_manager` endpoint. Every driver is spawned **supervised** — exit
//! notifications land in the same loop as protocol messages. Drivers with an
//! assignment must `hello` within a deadline or be stopped; a driver that dies
//! is restarted with backoff, and a crash loop (three fast deaths) marks it
//! failed instead of respawning forever. Exit reasons (M17.2) drive the
//! decision: a clean exit meant to stop; only faults and missed deadlines
//! restart. Tree reports (`child_added`) land in M18.2.
const std = @import("std");
const device = @import("driver");
const ipc = @import("ipc");
const process = @import("process");
const service = @import("service");
const time = @import("time");
const memory = @import("memory");
const logging = @import("logging");
const device_manager_protocol = @import("device-manager-protocol");
const envelope = @import("envelope");
const registry = @import("device-registry");
/// The generated device-manager dispatch, plus the subscriber machinery the
/// harness owns (P4c): the watcher table, the reserved `subscribe` verb, the
/// exit sweep, and the fan-out. One manager per system, so the handler context is
/// empty and the tables stay in this file's globals.
const Serve = service.Subscribers(device_manager_protocol.Protocol, void);
const Invocation = envelope.Invocation;
const Answer = envelope.Answer;
const fs = @import("file-system");
// --- the device registry ------------------------------------------------------
// Driver matching is data-driven and authoritative: /system/configuration/devices.csv (parsed by
// the device-registry module) names, per bus, which driver binds a reported
// device, the most-specific match winning. There is no compiled-in fallback — a
// device no row matches goes unbound and is logged. This retired the hand-kept
// pciDriverForIdentity / hidDriverFor / usbDriverForIdentity switch tables
// (docs/device-manager.md: "matching stays code until the third bus").
/// The CSV bytes, held for the life of the process because the parsed rules'
/// string fields (hid, driver) slice into this buffer.
var registry_source: [8192]u8 = undefined;
var registry_rules: [64]registry.Rule = undefined;
var registry_count: usize = 0;
/// Read and parse /system/configuration/devices.csv once at boot. The file lives in the initial
/// ramdisk, which the kernel serves directly — no filesystem service need be up
/// (fat is spawned after the manager), so this is a plain fs.open + read.
fn loadRegistry() void {
var file = fs.open("/system/configuration/devices.csv", .{}) orelse {
_ = logging.write("/system/services/device-manager: /system/configuration/devices.csv missing — nothing will match\n");
return;
};
defer file.close();
var used: usize = 0;
while (used < registry_source.len) {
const n = file.read(registry_source[used..]) orelse break;
if (n == 0) break;
used += n;
}
const result = registry.parse(registry_source[0..used], &registry_rules);
registry_count = result.count;
if (result.malformed != 0) std.log.info("/system/configuration/devices.csv: {d} malformed line(s) skipped", .{result.malformed});
if (result.truncated) _ = logging.write("/system/services/device-manager: /system/configuration/devices.csv has more rules than the table holds\n");
std.log.info("/system/configuration/devices.csv: {d} rule(s) loaded", .{registry_count});
}
/// Build a registry Identity from a bus driver's report: the bus it named, the
/// class triple unpacked from `identity` (0xCCSSPP — the same packing for a PCI
/// class code and a USB class triple), the widened numeric ids, and the ACPI hid.
fn identityFromReport(report: device_manager_protocol.ChildAdded) registry.Identity {
const bus: registry.Bus = switch (report.bus) {
@intFromEnum(device_manager_protocol.BusKind.pci) => .pci,
@intFromEnum(device_manager_protocol.BusKind.usb) => .usb,
@intFromEnum(device_manager_protocol.BusKind.acpi) => .acpi,
else => .unknown,
};
const hid_len = std.mem.indexOfScalar(u8, &report.hid, 0) orelse report.hid.len;
return .{
.bus = bus,
.base = @truncate(report.identity >> 16),
.subclass = @truncate(report.identity >> 8),
.prog_if = @truncate(report.identity),
.vendor = report.vendor,
.device = report.device,
.subsystem = report.subsystem,
.hid = report.hid[0..hid_len],
};
}
/// Whether some driver entry already serves registered device `device_id` —
/// a re-report after a bus restart must not spawn a second instance.
fn driverForDevice(device_id: u64) bool {
return driverEntryForDevice(device_id) != null;
}
/// The driver entry BOUND TO a device — matched and spawned for it. Distinct
/// from the device's reporter: a mouse's provider is the bus that reported it
/// (lineage via `children[].reporter`), while a volume's provider is the
/// storage driver spawned FOR it — which is what a `.consumer` hello asks for.
fn driverEntryForDevice(device_id: u64) ?*Driver {
if (device_id == device_manager_protocol.no_device) return null;
for (&drivers) |*driver| {
if (driver.used and driver.device_id == device_id) return driver;
}
return null;
}
// --- supervision -------------------------------------------------------------
/// How long a protocol driver has to hello after its spawn.
const hello_deadline_ms: u64 = 3000;
/// Deaths faster than this count toward the crash loop; slower ones reset it.
const fast_death_ns: u64 = 2_000_000_000;
/// Consecutive fast deaths before the manager gives up on a driver.
const crash_loop_cap: u32 = 3;
/// Restart backoff: base << (restarts - 1), so 300 ms, 600 ms, 1200 ms.
const backoff_base_ms: u64 = 300;
const DriverState = enum {
awaiting_hello, // spawned; the deadline is armed (protocol drivers only)
running,
restarting, // dead; respawn due at restart_due_ns
stopped, // exited cleanly — it meant to; not restarted
failed, // crash loop, or unspawnable; the manager gave up
};
const Driver = struct {
used: bool = false,
name_buffer: [64]u8 = undefined, // fits a full binary path (abi.maximum_process_name)
name_len: usize = 0,
// The assigned device id (becomes argv[1]), or device_manager_protocol.no_device.
device_id: u64 = device_manager_protocol.no_device,
// Whether this driver speaks the protocol (hello expected, deadline
// enforced). Legacy drivers (e.g. ps2-bus) are supervised and restarted
// but not yet required to hello.
speaks_protocol: bool = false,
process_id: u32 = 0,
state: DriverState = .running,
restarts: u32 = 0,
spawn_ns: u64 = 0,
hello_deadline_ns: u64 = 0,
restart_due_ns: u64 = 0,
// The serving endpoint this instance handed up in its hello — what consumer
// hellos for its reported children are answered with (establishment by
// lineage, communication.md "Establishment: two planes"). A refcounted
// handle in OUR table: onDriverExit must close it, or every restart leaks a
// slot of the manager's 32 until no capability can arrive at all.
endpoint: ?ipc.Handle = null,
fn name(driver: *const Driver) []const u8 {
return driver.name_buffer[0..driver.name_len];
}
};
const maximum_drivers = 16;
var drivers: [maximum_drivers]Driver = .{Driver{}} ** maximum_drivers;
var manager_endpoint: ipc.Handle = 0;
var test_restart_mode = false;
var test_usb_restart_mode = false;
var test_usb_killed = false;
var test_pci_restart_mode = false;
var test_scanout_restart_mode = false;
var test_scanout_killed = false;
var test_storage_restart_mode = false;
var test_storage_killed = false;
var test_kill_pid: u32 = 0;
var test_kill_due_ns: u64 = 0;
/// The manager's mirror of what bus drivers report (docs/device-manager.md "the
/// tree"): the children, keyed by (parent, bus address), each remembering which
/// driver instance reported it — that is what death-pruning sweeps by.
const Child = struct {
used: bool = false,
parent: u64 = 0,
bus_address: u64 = 0,
identity: u64 = 0,
// The kernel device id (registered by the reporter), or device_manager_protocol.no_device.
device_id: u64 = 0,
reporter: u32 = 0, // the reporting driver instance's process id
};
const maximum_children = 64; // ACPI adds ~34 device nodes (M20.2), plus PCI + USB
var children: [maximum_children]Child = .{Child{}} ** maximum_children;
/// Record (or refresh) a reported child. Refreshing matters: a restarted bus
/// driver re-reports what it rediscovers, and the same (parent, port) must not
/// duplicate.
fn addChild(parent: u64, bus_address: u64, identity: u64, device_id: u64, reporter: u32) bool {
var free: ?*Child = null;
for (&children) |*child| {
if (child.used and child.parent == parent and child.bus_address == bus_address) {
child.identity = identity;
child.device_id = device_id;
child.reporter = reporter;
return true;
}
if (!child.used and free == null) free = child;
}
const slot = free orelse return false;
slot.* = .{ .used = true, .parent = parent, .bus_address = bus_address, .identity = identity, .device_id = device_id, .reporter = reporter };
return true;
}
/// Prune every child a dead driver instance reported: the children describe
/// protocol state (slots, rings) that died with the process — keeping the nodes
/// would be keeping a lie. The restarted instance rediscovers and re-reports.
/// Watchers hear the honest story: removed now, added again on rediscovery.
///
/// **A reporter's death reaps its subtree.** The class drivers spawned for the
/// pruned children hold channels into the dead process; they cannot observe
/// the death themselves (an HID driver blocks on interrupt reports that will
/// simply never come — a silent zombie), and their still-`used` entries would
/// make the matcher's dedupe refuse the respawn when the re-report arrives.
/// So each is killed and its entry cleared: the re-report spawns a fresh
/// instance, whose hello fetches the successor's channel. That is the restart
/// story working — kill a bus driver and only its subtree blinks
/// (docs/establishment-planes-plan.md P3).
fn pruneChildrenOf(reporter: u32) void {
for (&children) |*child| {
if (child.used and child.reporter == reporter) {
std.log.info("child removed (device {d} port {d})", .{ child.parent, child.bus_address });
reapDriverBoundTo(child.device_id);
child.used = false;
Serve.publish(.child_removed, 0, .{ .parent = child.parent, .bus_address = child.bus_address });
}
}
}
/// Kill the driver bound to a pruned child and clear its entry — the exit
/// notification that follows finds no entry and is ignored, so the death is
/// never double-counted, and the freed entry is what lets the re-report
/// respawn. The device it was given returns to us by the kernel's loan rule.
fn reapDriverBoundTo(device_id: u64) void {
const driver = driverEntryForDevice(device_id) orelse return;
if (driver.process_id != 0) {
std.log.info("reaping {s} (its provider died)", .{driver.name()});
_ = process.kill(driver.process_id);
}
if (driver.endpoint) |endpoint| _ = ipc.close(endpoint);
driver.* = .{};
}
/// How many children a driver instance has reported (the test-usb-restart
/// trigger counts these).
fn childCountOf(reporter: u32) u32 {
var n: u32 = 0;
for (&children) |*child| {
if (child.used and child.reporter == reporter) n += 1;
}
return n;
}
/// The child a kernel device id belongs to — the lineage lookup: its
/// `.reporter` names the driver instance that provides it, which is how a
/// consumer's hello is routed to the right provider (communication.md
/// "Establishment: two planes"). Null when nothing reported it, or its
/// reporter died and pruned it — a retryable gap, not a verdict.
fn childByDevice(device_id: u64) ?*Child {
if (device_id == device_manager_protocol.no_device) return null;
for (&children) |*child| {
if (child.used and child.device_id == device_id) return child;
}
return null;
}
/// The driver entry a live process id belongs to. Zero is not a process id here:
/// it is what `onDriverExit` writes back to retire an id it has already acted on,
/// so a second notification for the same death matches nothing.
fn driverByName(name: []const u8) ?*Driver {
for (&drivers) |*driver| {
if (driver.used and std.mem.eql(u8, driver.name(), name)) return driver;
}
return null;
}
fn driverByProcess(process_id: u32) ?*Driver {
if (process_id == 0) return null;
for (&drivers) |*driver| {
if (driver.used and driver.process_id == process_id) return driver;
}
return null;
}
/// Whether a singleton driver is already in the table (two ACPI nodes can both
/// map to ps2-bus; one instance serves both).
fn alreadySupervised(name: []const u8) bool {
for (&drivers) |*driver| {
if (driver.used and std.mem.eql(u8, driver.name(), name)) return true;
}
return false;
}
/// Record a driver in the table and spawn its first instance.
fn addDriver(name: []const u8, device_id: u64, speaks_protocol: bool) void {
for (&drivers) |*driver| {
if (driver.used) continue;
const n = @min(name.len, driver.name_buffer.len);
@memcpy(driver.name_buffer[0..n], name[0..n]);
driver.name_len = n;
driver.device_id = device_id;
driver.speaks_protocol = speaks_protocol;
driver.used = true;
spawnDriver(driver);
return;
}
std.log.info("driver table full; cannot supervise {s}", .{name});
}
/// (Re)spawn a driver instance: supervised on the manager's own endpoint, the
/// device id as argv[1] when it has one, the hello deadline armed when it
/// speaks the protocol.
fn spawnDriver(driver: *Driver) void {
// Take the device before the driver exists, so there is no window in which anyone
// else could claim it — which is the whole of what makes the handover authoritative
// rather than advisory. Re-claiming across a restart is expected to say
// AlreadyClaimed once the manager already holds it, and that is fine: it means the
// device never left our hands while the driver was dead.
if (driver.device_id != device_manager_protocol.no_device) {
device.claim(driver.device_id) catch |e| switch (e) {
error.AlreadyClaimed => {}, // ours already, from a previous spawn of this driver
else => {
std.log.warn("cannot hold device {d} for {s}: {s}", .{ driver.device_id, driver.name(), @errorName(e) });
driver.state = .failed;
return;
},
};
}
var id_text: [20]u8 = undefined;
var arguments: [1][]const u8 = undefined;
var argument_count: usize = 0;
if (driver.device_id != device_manager_protocol.no_device) {
arguments[0] = std.fmt.bufPrint(&id_text, "{d}", .{driver.device_id}) catch return;
argument_count = 1;
}
// The device rides the spawn, so the driver holds it before its first instruction.
// A transfer *after* spawning would leave a window in which the child is running
// without its hardware — closed on one machine, open on another
// (docs/bounds-track-plan.md, "the grant rides system_spawn").
const give = driver.device_id;
if (give != device_manager_protocol.no_device)
std.log.info("delegated device {d} to {s}", .{ give, driver.name() });
const child = process.spawnSupervisedWithDevice(driver.name(), arguments[0..argument_count], manager_endpoint, give) orelse {
std.log.info("failed to spawn {s}", .{driver.name()});
driver.state = .failed;
return;
};
driver.process_id = child;
driver.spawn_ns = time.clock();
if (driver.speaks_protocol) {
driver.state = .awaiting_hello;
driver.hello_deadline_ns = driver.spawn_ns + hello_deadline_ms * 1_000_000;
_ = time.timerOnce(manager_endpoint, hello_deadline_ms + 100);
} else {
driver.state = .running;
}
if (driver.device_id != device_manager_protocol.no_device) {
std.log.info("spawned {s} for device {d}", .{ driver.name(), driver.device_id });
} else {
std.log.info("spawned {s}", .{driver.name()});
}
}
/// A driver died. Prune what it reported first — then the exit reason (M17.2)
/// is the whole restart decision: a clean exit meant to stop; anything else
/// restarts with backoff until the crash-loop cap.
fn onDriverExit(driver: *Driver) void {
const dead = driver.process_id;
// One death, two notifications: the manager is this driver's supervisor (its
// spawn named this endpoint) *and*, since P4c put the watcher table in the
// harness, a subscriber to published exits. Both badges carry the same id, and
// the ring delivers them separately — so the id is retired here, before any
// decision is taken, and the second notification finds no driver to act on.
// Without this the backoff would count one death twice and the crash-loop cap
// would fire at half the deaths it names.
driver.process_id = 0;
// The dead instance's serving endpoint is stale the moment it died — the
// kernel marked the endpoint dead, but our refcounted handle would sit in
// the 32-slot table forever. The successor's hello stores a fresh one.
if (driver.endpoint) |stale| {
_ = ipc.close(stale);
driver.endpoint = null;
}
pruneChildrenOf(dead);
const reason = process.exitReason(dead) orelse .fault;
if (reason == .exited) {
driver.state = .stopped;
std.log.info("{s} exited cleanly; not restarting", .{driver.name()});
return;
}
const now = time.clock();
const alive_ns = now - driver.spawn_ns;
driver.restarts = if (alive_ns < fast_death_ns) driver.restarts + 1 else 1;
if (driver.restarts >= crash_loop_cap) {
driver.state = .failed;
std.log.info("{s} is failing repeatedly (crash loop); giving up", .{driver.name()});
return;
}
const delay_ms = backoff_base_ms << @intCast(driver.restarts - 1);
driver.state = .restarting;
driver.restart_due_ns = now + delay_ms * 1_000_000;
std.log.info("restarting {s} in {d} ms (died: {s})", .{ driver.name(), delay_ms, @tagName(reason) });
_ = time.timerOnce(manager_endpoint, delay_ms + 50);
}
/// A timer landed: sweep every deadline. Overdue hellos are killed (the exit
/// notification then routes through the normal restart policy); due restarts
/// respawn. Timers carry no id on purpose — the table is the state, and one
/// sweep serves every armed deadline.
fn sweepDeadlines() void {
const now = time.clock();
if (test_kill_pid != 0 and now >= test_kill_due_ns) {
std.log.info("test mode: killing the reporter", .{});
_ = process.kill(test_kill_pid);
test_kill_pid = 0;
}
for (&drivers) |*driver| {
if (!driver.used) continue;
switch (driver.state) {
.awaiting_hello => if (now >= driver.hello_deadline_ns) {
std.log.info("{s} missed its hello deadline", .{driver.name()});
_ = process.kill(driver.process_id);
// The exit notification finishes the job via onDriverExit.
},
.restarting => if (now >= driver.restart_due_ns) spawnDriver(driver),
else => {},
}
}
}
// --- the harness callbacks -----------------------------------------------------
fn initialise(endpoint: ipc.Handle) bool {
manager_endpoint = endpoint;
// Load the authoritative driver-match registry before any bus driver can
// report a device to match against it.
loadRegistry();
// Enumerate into a heap buffer (too big for the one-page user stack).
const buffer = memory.allocator().alloc(device.DeviceDescriptor, 64) catch {
_ = logging.write("/system/services/device-manager: out of memory\n");
return false;
};
const total = device.enumerate(buffer);
const count = @min(total, buffer.len);
// The node discovery needs: the kernel seeds it, so it is in this same snapshot
// and can be handed over like any other assignment. Discovery used to find and
// claim it itself — the last driver that acquired hardware by naming it rather
// than being given it (docs/os-development/device-authority.md).
var tables_node: u64 = device_manager_protocol.no_device;
for (buffer[0..count]) |descriptor| {
if (descriptor.class == @intFromEnum(device.DeviceClass.acpi_tables)) {
tables_node = descriptor.id;
break;
}
}
// **Hold the firmware-discovered hardware, so none of it is left lying around.**
// A device nobody holds can be claimed by anyone, so every seeded device that
// carries mappable resources is taken here whether or not a driver wants it — the
// HPET most of all, which has an MMIO window and an IRQ and no user-space driver.
// Held by the manager it is inert; unheld it was there for the taking.
//
// Two deliberate exclusions:
// - the loader's framebuffer, which the compositor claims and which is not
// hardware anyone is delegated (the manager starts before display, so taking
// it here would break the boot screen);
// - anything with no resources, which grants nothing and so is not worth holding.
//
// This covers the boot snapshot only. A device *reported* later and matched to no
// driver stays claimable — the pci-cap and iommu-fault fixtures rely on exactly
// that to reach an unmatched NIC. Narrowing it further is a separate change with
// those fixtures in scope.
for (buffer[0..count]) |descriptor| {
if (descriptor.resource_count == 0) continue;
if (descriptor.class == @intFromEnum(device.DeviceClass.display)) continue;
device.claim(descriptor.id) catch continue; // already held, or not ours to take
}
var matched: usize = 0;
for (buffer[0..count]) |descriptor| {
if (descriptor.class == @intFromEnum(device.DeviceClass.pci_host_bridge)) {
// The PCI bus driver: enumeration in ring 3 (M19), one instance
// per bridge, the bridge id as its assignment.
matched += 1;
addDriver("pci-bus", descriptor.id, true);
continue;
}
// Nothing else is matched from the boot snapshot today. The kernel-seeded
// HPET timer node is served by the kernel's own clock (docs/timers.md), not
// a user-space driver; PCI functions and PS/2 _HID devices arrive later as
// pci-bus / acpi-service reports and match in onChildAdded (docs/discovery.md).
// A fuller system's static class->driver manifest (docs/device-manager.md)
// would slot in here.
}
// The discovery service (docs/discovery.md): one per firmware, packed
// under the neutral name "discovery", spawned once at startup. It finds and
// claims the acpi-tables (or devicetree-blob) node itself. Not a per-device
// match — it is the discoverer, not a driver bound to one device.
addDriver("discovery", tables_node, false);
if (test_restart_mode) {
// The driver-restart scenario's fixture: claims device 0 (the tree
// root, otherwise unclaimed), hellos, then faults — driving backoff,
// re-claim-after-death, and the crash-loop cap deterministically.
addDriver("crash-test", 0, true);
}
if (matched == 0) {
_ = logging.write("/system/services/device-manager: no matchable devices\n");
} else {
_ = logging.write("/system/services/device-manager: ok\n");
}
return true;
}
fn onMessage(message: []const u8, reply: []u8, sender: u32, arrived: *ipc.Arrival) usize {
return Serve.dispatch({}, handlers, message, sender, arrived, reply);
}
/// `subscribe` and `unsubscribe` are absent on purpose: the harness answers both,
/// and its table is what `publish` fans out over.
const handlers = Serve.Handlers{
.hello = onHello,
.child_added = onChildAdded,
.child_removed = onChildRemoved,
.enumerate = onEnumerate,
};
/// The handshake. The device this driver was assigned is the packet's target.
fn onHello(_: void, invocation: Invocation(device_manager_protocol.Hello), _: Answer(void)) isize {
if (invocation.request.version != device_manager_protocol.version) {
std.log.info("refused hello (version {d}) from process {d}", .{ invocation.request.version, invocation.sender });
return -envelope.EPROTO;
}
// A consumer is not a spawned driver: no entry, no deadline, no state —
// just establishment. It asks for the channel of the driver BOUND TO its
// target (fat asking for its volume's block provider). No channel is a
// retryable ack, exactly as for a device-role consumer.
//
// The residual, stated plainly: any process granted `open device-manager`
// can ask. The grant rows are the gate today, as they were when the block
// name was open-granted; a finer per-channel policy belongs to the same
// future as the spawn capability (device-authority.md).
if (invocation.request.role == @intFromEnum(device_manager_protocol.Role.consumer)) {
if (invocation.request.wants_channel != 0) {
if (driverEntryForDevice(invocation.target)) |provider| {
if (provider.endpoint) |serving| service.replyWithCapability(serving);
}
}
return 0;
}
const driver = driverByProcess(invocation.sender) orelse {
std.log.info("hello from unknown process {d}", .{invocation.sender});
return -envelope.EPERM;
};
driver.state = .running;
// A provider's hello carries its serving endpoint — the channel consumer
// hellos for its reported children are answered with. Stored on the entry
// (claimed from the turn, so the harness's defer leaves it alone); a
// re-hello replaces, closing the old handle rather than leaking the slot.
if (invocation.capability) |serving| {
if (driver.endpoint) |previous| _ = ipc.close(previous);
driver.endpoint = serving;
Serve.claimArrival();
}
// A consumer's hello asks for its device's provider: the child's reporter
// is the routing fact (establishment by lineage). No channel is not a
// refusal — the provider may be mid-restart and its re-report on the way —
// so the hello still acks and the consumer retries. Nothing is nominated
// unless asked: a capability sent to a caller that never reads one is a
// leaked slot in ITS table.
if (invocation.request.wants_channel != 0) {
if (childByDevice(invocation.target)) |child| {
if (driverByProcess(child.reporter)) |provider| {
if (provider.endpoint) |serving| service.replyWithCapability(serving);
}
}
}
std.log.info("hello from {s} (device {d})", .{ driver.name(), invocation.target });
// Resilience drill (V6): once, kill the virtio-gpu driver a moment after it hellos, so
// the normal restart policy respawns it — the compositor must survive and re-attach.
if (test_scanout_restart_mode and !test_scanout_killed and std.mem.eql(u8, driver.name(), "/system/drivers/virtio-gpu")) {
test_scanout_killed = true;
test_kill_pid = invocation.sender;
test_kill_due_ns = time.clock() + 1_500_000_000;
_ = time.timerOnce(manager_endpoint, 1600);
}
// Storage-driver-crash drill (S5): once, a moment after usb-storage hellos —
// long enough that its volume has mounted — kill it. The manager re-delegates
// the still-present device to a restarted driver on a fresh channel; the volume
// manager's channel-liveness probe must notice the dead channel and rebuild.
if (test_storage_restart_mode and !test_storage_killed and std.mem.eql(u8, driver.name(), "/system/drivers/usb-storage")) {
test_storage_killed = true;
test_kill_pid = invocation.sender;
test_kill_due_ns = time.clock() + 2_000_000_000; // after the ~0.6s mount
_ = time.timerOnce(manager_endpoint, 2100);
}
return 0;
}
/// A bus driver reported a discovered device: mirror it, publish it, match a
/// driver for it — and in the restart drills kill the reporter once, the
/// deterministic trigger for prune -> backoff -> respawn -> re-report.
fn onChildAdded(_: void, invocation: Invocation(device_manager_protocol.ChildAdded), _: Answer(void)) isize {
const report = invocation.request;
const sender = invocation.sender;
// The registered kernel device id is the packet's target, not a field: what
// the manager hands a matched driver as its argv assignment.
const device_id = invocation.target;
var status: isize = 0;
if (driverByProcess(sender)) |driver| {
if (!addChild(report.parent, report.bus_address, report.identity, device_id, sender)) status = -envelope.ENOSPC;
std.log.info("child added (device {d} port {d}, identity {d}) by {s}", .{ report.parent, report.bus_address, report.identity, driver.name() });
if (status == 0) Serve.publish(.child_added, device_id, report);
// Matching from reports (M19.3), now data-driven via the /system/configuration/devices.csv
// registry: a registered child gets the most-specific driver its identity
// matches, once — re-reports after a bus restart dedupe on the registered
// id, exactly like the registrations do.
if (status == 0 and device_id != device_manager_protocol.no_device) {
const id = identityFromReport(report);
if (registry.matchDriver(registry_rules[0..registry_count], id)) |match| {
if (match.ambiguous)
std.log.info("/system/configuration/devices.csv: multiple equally-specific rules match the device {s} reported; binding {s}", .{ driver.name(), match.driver });
if (id.bus == .acpi) {
// An hid-matched driver is a singleton over one piece of hardware
// described by several nodes: the 8042 is a single controller whose
// I/O ports live under the keyboard node (PNP0303) while the mouse
// is a second node (PNP0F13). Two processes would fight over the
// same 0x60/0x64 registers, so there is exactly one instance — and
// it needs *every* matching device, however many the machine has
// (some have none, some one port, some two).
//
// The first rides the spawn; the rest are transferred to the
// running instance. Late arrival is fine here because the order is
// natural: the instance needs the controller node immediately and
// reaches the mouse only after the 8042 handshakes and identify.
if (!alreadySupervised(match.driver)) {
addDriver(match.driver, device_id, false);
} else if (driverByName(match.driver)) |running| {
if (running.process_id != 0) {
device.claim(device_id) catch |e| switch (e) {
error.AlreadyClaimed => {},
else => {
std.log.warn("cannot hold device {d} for {s}: {s}", .{ device_id, match.driver, @errorName(e) });
return status;
},
};
device.transfer(device_id, running.process_id) catch |e| {
std.log.warn("could not give device {d} to {s}: {s}", .{ device_id, match.driver, @errorName(e) });
return status;
};
std.log.info("delegated device {d} to {s} (already running)", .{ device_id, match.driver });
}
}
} else {
// A per-device driver: one instance, the registered id as argv[1].
if (!driverForDevice(device_id)) addDriver(match.driver, device_id, true);
}
}
}
} else {
status = -envelope.EPERM;
}
if (test_pci_restart_mode and !test_usb_killed) {
if (driverByProcess(sender)) |driver| {
if (std.mem.eql(u8, driver.name(), "pci-bus") and childCountOf(sender) >= 3) {
// The pci restart drill: kill the enumerator after it has
// reported; the respawn must re-register without duplicates
// (M19.0 idempotence, proven end to end by pci-scan).
test_usb_killed = true;
test_kill_pid = sender;
test_kill_due_ns = time.clock() + 1_000_000_000;
_ = time.timerOnce(manager_endpoint, 1100);
}
}
}
if (test_usb_restart_mode and !test_usb_killed and childCountOf(sender) >= 2) {
// Only the xHCI reporter is the drill's victim — pci-bus also reports
// now, and whichever finishes second must not trigger the kill.
if (driverByProcess(sender)) |driver| {
if (std.mem.eql(u8, driver.name(), "/system/drivers/usb-xhci-bus")) {
// Delayed, not immediate: the device-list scenario's subscriber
// needs a window to enumerate and subscribe before the events.
test_usb_killed = true;
test_kill_pid = sender;
test_kill_due_ns = time.clock() + 2_000_000_000;
_ = time.timerOnce(manager_endpoint, 2100);
}
}
}
return status;
}
/// A bus driver reported a device gone (hot-unplug). Addressed by the composite
/// (parent, bus address) the reporter knows, which is why that pair is the
/// packet's body rather than its target.
fn onChildRemoved(_: void, invocation: Invocation(device_manager_protocol.ChildRemoved), _: Answer(void)) isize {
const report = invocation.request;
var status: isize = -envelope.ENOENT;
for (&children) |*child| {
if (child.used and child.parent == report.parent and child.bus_address == report.bus_address and child.reporter == invocation.sender) {
std.log.info("child removed (device {d} port {d})", .{ child.parent, child.bus_address });
// Unplug reaps exactly like reporter death (pruneChildrenOf): the
// bound driver's device is gone and it cannot observe that — it
// blocks on reports that will never come — and its stale entry
// would make the dedupe refuse the respawn when the device is
// PLUGGED BACK IN. Reap now, and a replug's re-report rebinds.
reapDriverBoundTo(child.device_id);
child.used = false;
status = 0;
}
}
return status;
}
/// The reserved `enumerate` verb: the mirror, one `ChildEntry` per known child,
/// packed into the reply's tail. How many arrived is the reply's own length —
/// `Status.len` — so no count header is spent saying it twice.
fn onEnumerate(_: void, invocation: Invocation(void), answer: Answer(void)) isize {
const entry_size = @sizeOf(device_manager_protocol.ChildEntry);
const tail = answer.tail();
// `target` is the page cursor: skip that many known children first. One
// reply holds only a handful of entries, and a real tree (a dozen ACPI
// nodes before the first USB child) outgrew one packet — the storage
// child silently never fit, which is precisely the truncation shape the
// bounds audit exists to forbid. A caller pages until a short page.
var skip = invocation.target;
var written: usize = 0;
for (&children) |*child| {
if (!child.used) continue;
if (skip > 0) {
skip -= 1;
continue;
}
if (written + entry_size > tail.len) break;
const entry = device_manager_protocol.ChildEntry{ .parent = child.parent, .bus_address = child.bus_address, .identity = child.identity, .device_id = child.device_id };
@memcpy(tail[written..][0..entry_size], std.mem.asBytes(&entry));
written += entry_size;
}
return @intCast(written);
}
fn onNotification(badge: u64) void {
if (badge & ipc.notify_exit_bit != 0) {
const dead: u32 = @intCast(badge & ~(ipc.notify_badge_bit | ipc.notify_exit_bit));
// The harness has already swept the watcher table for this death; what is
// left is the manager's own concern, its supervised drivers.
if (driverByProcess(dead)) |driver| onDriverExit(driver);
return;
}
if (badge & ipc.notify_timer_bit != 0) sweepDeadlines();
}
pub fn main(init: process.Init) void {
if (init.arguments.get(1)) |mode| {
test_restart_mode = std.mem.eql(u8, mode, "test-restart");
test_usb_restart_mode = std.mem.eql(u8, mode, "test-usb-restart");
test_pci_restart_mode = std.mem.eql(u8, mode, "test-pci-restart");
test_scanout_restart_mode = std.mem.eql(u8, mode, "test-scanout-restart");
test_storage_restart_mode = std.mem.eql(u8, mode, "test-storage-restart");
}
service.run(device_manager_protocol.message_maximum, .{
.service = "device-manager",
.init = initialise,
.on_message = onMessage,
.on_notification = onNotification,
.subscribers = Serve.hooks,
});
}