library: the harness keeps the subscribers, and an id belongs to whoever opened it

Three services had each written the same thing and got it three different
ways: input polled the process list to notice a dead subscriber, and only
when someone else subscribed; the power service never noticed at all; the
device manager noticed drivers but not subscribers. The harness owns the
table now, driven by the events a protocol declares — it registers on the
reserved verb, frames each event once, posts to everyone interested without
waiting on any of them, and reclaims a slot when the kernel says its owner
died. Interest masks moved to the envelope, so a subscriber that wants only
mice asks the same way everywhere.

Two consequences the plan had not foreseen. The device manager now hears a
supervised child's death twice, once as its supervisor and once as a
subscriber, so restart backoff counted every crash twice and gave up after
half as many; it retires the id before counting. And the kernel's published
exit table had eight slots for what is now six subscriptions in a plain
boot, so it holds sixteen.

The other half is a hole the design named early and left standing: a
backend handed out a small integer and then honoured it from anyone. A
process that guessed a file's node id read another client's file; a display
layer had no owner at all, so any client could reconfigure or destroy any
layer; a USB device token was never checked against the client that opened
it. Each is now bound to the task that opened it, and a wrong owner gets
exactly what an unknown id gets — the refusal must not become the oracle
the identical answers elsewhere were designed to remove. Closing a file
changed with it: it used to succeed unconditionally, which would have told
a caller which ids existed.

Suite 111/111, with a new case in which one process holds a file and a
layer, hands both ids to a second process, and finds them untouched after
that process has tried everything with them.
This commit is contained in:
Daniel Samson
2026-08-01 09:05:26 +01:00
parent 2719b93530
commit 1b1c587c14
29 changed files with 1072 additions and 335 deletions
@@ -27,9 +27,11 @@ const device_manager_protocol = @import("device-manager-protocol");
const envelope = @import("envelope");
const registry = @import("device-registry");
/// The generated device-manager dispatch. One manager per system, so the handler
/// context is empty and the tables stay in this file's globals.
const Serve = device_manager_protocol.Protocol.Provider(void);
/// The generated device-manager dispatch, plus the subscriber machinery the
/// harness owns (P4c): the watcher table, the reserved `subscribe` verb, the
/// exit sweep, and the fan-out. One manager per system, so the handler context is
/// empty and the tables stay in this file's globals.
const Serve = service.Subscribers(device_manager_protocol.Protocol, void);
const Invocation = envelope.Invocation;
const Answer = envelope.Answer;
@@ -156,31 +158,6 @@ var test_scanout_killed = false;
var test_kill_pid: u32 = 0;
var test_kill_due_ns: u64 = 0;
/// The application subscribers (M18.3, the input-service pattern): endpoints
/// handed over as capabilities, each receiving every child add/remove as a
/// buffered message. A subscriber whose endpoint stops accepting (it died) is
/// dropped on the failed send.
const maximum_subscribers = 8;
var subscribers: [maximum_subscribers]?ipc.Handle = .{null} ** maximum_subscribers;
/// Push one event to every subscriber: the same struct a bus driver *called*
/// with, framed as an event instead — one encoding, both directions, told apart
/// by the packet's verb rather than by anything inside it. A subscriber whose
/// endpoint stops accepting (it died) is dropped on the failed send.
fn publish(
comptime event: device_manager_protocol.Event,
target: u64,
payload: device_manager_protocol.Protocol.PayloadOf(event),
) void {
var packet: [envelope.post_maximum]u8 = undefined;
const framed = device_manager_protocol.Protocol.encodeEvent(event, target, payload, &packet) orelse return;
for (&subscribers) |*slot| {
if (slot.*) |handle| {
if (!ipc.send(handle, framed)) slot.* = null; // dead subscriber
}
}
}
/// The manager's mirror of what bus drivers report (docs/device-manager.md "the
/// tree"): the children, keyed by (parent, bus address), each remembering which
/// driver instance reported it — that is what death-pruning sweeps by.
@@ -225,7 +202,7 @@ fn pruneChildrenOf(reporter: u32) void {
if (child.used and child.reporter == reporter) {
std.log.info("child removed (device {d} port {d})", .{ child.parent, child.bus_address });
child.used = false;
publish(.child_removed, 0, .{ .parent = child.parent, .bus_address = child.bus_address });
Serve.publish(.child_removed, 0, .{ .parent = child.parent, .bus_address = child.bus_address });
}
}
}
@@ -240,7 +217,11 @@ fn childCountOf(reporter: u32) u32 {
return n;
}
/// The driver entry a live process id belongs to. Zero is not a process id here:
/// it is what `onDriverExit` writes back to retire an id it has already acted on,
/// so a second notification for the same death matches nothing.
fn driverByProcess(process_id: u32) ?*Driver {
if (process_id == 0) return null;
for (&drivers) |*driver| {
if (driver.used and driver.process_id == process_id) return driver;
}
@@ -308,8 +289,17 @@ fn spawnDriver(driver: *Driver) void {
/// is the whole restart decision: a clean exit meant to stop; anything else
/// restarts with backoff until the crash-loop cap.
fn onDriverExit(driver: *Driver) void {
pruneChildrenOf(driver.process_id);
const reason = process.exitReason(driver.process_id) orelse .fault;
const dead = driver.process_id;
// One death, two notifications: the manager is this driver's supervisor (its
// spawn named this endpoint) *and*, since P4c put the watcher table in the
// harness, a subscriber to published exits. Both badges carry the same id, and
// the ring delivers them separately — so the id is retired here, before any
// decision is taken, and the second notification finds no driver to act on.
// Without this the backoff would count one death twice and the crash-loop cap
// would fire at half the deaths it names.
driver.process_id = 0;
pruneChildrenOf(dead);
const reason = process.exitReason(dead) orelse .fault;
if (reason == .exited) {
driver.state = .stopped;
std.log.info("{s} exited cleanly; not restarting", .{driver.name()});
@@ -410,27 +400,17 @@ fn initialise(endpoint: ipc.Handle) bool {
return true;
}
/// Set by `onSubscribe` when the subscriber table has taken ownership of the
/// capability the call carried, and read by `onMessage`, where the turn's
/// `Arrival` lives. The generated dispatch hands a handler the raw handle rather
/// than the `Arrival` — deliberately, since a handler has no business closing the
/// turn's property — so the *claim* travels back out this way. One turn, one
/// handler, one thread: there is nothing here to race.
var capability_claimed = false;
fn onMessage(message: []const u8, reply: []u8, sender: u32, arrived: *ipc.Arrival) usize {
capability_claimed = false;
const written = Serve.dispatch({}, handlers, message, sender, arrived.peek(), reply);
if (capability_claimed) _ = arrived.take();
return written;
return Serve.dispatch({}, handlers, message, sender, arrived, reply);
}
/// `subscribe` and `unsubscribe` are absent on purpose: the harness answers both,
/// and its table is what `publish` fans out over.
const handlers = Serve.Handlers{
.hello = onHello,
.child_added = onChildAdded,
.child_removed = onChildRemoved,
.enumerate = onEnumerate,
.subscribe = onSubscribe,
};
/// The handshake. The device this driver was assigned is the packet's target.
@@ -470,7 +450,7 @@ fn onChildAdded(_: void, invocation: Invocation(device_manager_protocol.ChildAdd
if (driverByProcess(sender)) |driver| {
if (!addChild(report.parent, report.bus_address, report.identity, device_id, sender)) status = -envelope.ENOSPC;
std.log.info("child added (device {d} port {d}, identity {d}) by {s}", .{ report.parent, report.bus_address, report.identity, driver.name() });
if (status == 0) publish(.child_added, device_id, report);
if (status == 0) Serve.publish(.child_added, device_id, report);
// Matching from reports (M19.3), now data-driven via the /system/configuration/devices.csv
// registry: a registered child gets the most-specific driver its identity
// matches, once — re-reports after a bus restart dedupe on the registered
@@ -556,24 +536,11 @@ fn onEnumerate(_: void, _: Invocation(void), answer: Answer(void)) isize {
return @intCast(written);
}
/// The reserved `subscribe` verb: an application's endpoint arrived as the call's
/// capability. The table taking a slot is what claims it; a full table refuses
/// and lets the turn close it, so a subscribe storm cannot spend the handle table.
fn onSubscribe(_: void, invocation: Invocation(void), _: Answer(void)) isize {
const endpoint = invocation.capability orelse return -envelope.EPROTO;
for (&subscribers) |*slot| {
if (slot.* == null) {
slot.* = endpoint;
capability_claimed = true; // the table holds it from here
return 0;
}
}
return -envelope.ENOSPC;
}
fn onNotification(badge: u64) void {
if (badge & ipc.notify_exit_bit != 0) {
const dead: u32 = @intCast(badge & ~(ipc.notify_badge_bit | ipc.notify_exit_bit));
// The harness has already swept the watcher table for this death; what is
// left is the manager's own concern, its supervised drivers.
if (driverByProcess(dead)) |driver| onDriverExit(driver);
return;
}
@@ -592,5 +559,6 @@ pub fn main(init: process.Init) void {
.init = initialise,
.on_message = onMessage,
.on_notification = onNotification,
.subscribers = Serve.hooks,
});
}