library: the harness keeps the subscribers, and an id belongs to whoever opened it

Three services had each written the same thing and got it three different
ways: input polled the process list to notice a dead subscriber, and only
when someone else subscribed; the power service never noticed at all; the
device manager noticed drivers but not subscribers. The harness owns the
table now, driven by the events a protocol declares — it registers on the
reserved verb, frames each event once, posts to everyone interested without
waiting on any of them, and reclaims a slot when the kernel says its owner
died. Interest masks moved to the envelope, so a subscriber that wants only
mice asks the same way everywhere.

Two consequences the plan had not foreseen. The device manager now hears a
supervised child's death twice, once as its supervisor and once as a
subscriber, so restart backoff counted every crash twice and gave up after
half as many; it retires the id before counting. And the kernel's published
exit table had eight slots for what is now six subscriptions in a plain
boot, so it holds sixteen.

The other half is a hole the design named early and left standing: a
backend handed out a small integer and then honoured it from anyone. A
process that guessed a file's node id read another client's file; a display
layer had no owner at all, so any client could reconfigure or destroy any
layer; a USB device token was never checked against the client that opened
it. Each is now bound to the task that opened it, and a wrong owner gets
exactly what an unknown id gets — the refusal must not become the oracle
the identical answers elsewhere were designed to remove. Closing a file
changed with it: it used to succeed unconditionally, which would have told
a caller which ids existed.

Suite 111/111, with a new case in which one process holds a file and a
layer, hands both ids to a second process, and finds them untouched after
that process has tried everything with them.
This commit is contained in:
Daniel Samson
2026-08-01 09:05:26 +01:00
parent 2719b93530
commit 1b1c587c14
29 changed files with 1072 additions and 335 deletions
+75 -9
View File
@@ -61,22 +61,33 @@ fn timerInterval() u64 {
return if (msi_vector != null) reconcile_interval_ms else poll_interval_ms;
}
/// The class driver endpoints that opened each device, so interrupt reports can
/// be pushed back to them. Keyed by the device token (the interface's device id).
/// The class driver that opened each device: the endpoint interrupt reports are
/// pushed back to, and **the task that opened it** — the kernel-stamped badge, so
/// a device token is scoped to the client that was given it. Keyed by the device
/// token (the interface's device id).
///
/// The scoping is the point (docs/os-development/protocol-namespace.md: handles
/// are validated against the badge). A token is a small registered-device id any
/// process could name, and every packet that carries one used to be honoured from
/// anyone: a stranger could run control transfers on another driver's device, arm
/// interrupt polling on it, or redirect its reports.
const Open = struct {
used: bool = false,
device_token: u64 = 0,
owner: u32 = 0,
report_endpoint: usize = 0,
};
var opens = [_]Open{.{}} ** 16;
/// Remember (or replace) the endpoint that reports for `device_token`. Returns
/// whether the table kept the handle — false means the caller still owns it and
/// must dispose of it. A re-open supersedes the previous endpoint, and the one
/// it displaced is closed here: the table holds exactly one reference per slot.
fn recordOpen(device_token: u64, report_endpoint: usize) bool {
/// Remember (or replace) the endpoint task `owner` receives reports for
/// `device_token` on. Returns whether the table kept the handle — false means the
/// caller still owns it and must dispose of it. A re-open by the **same** client
/// supersedes its previous endpoint, and the one it displaced is closed here: the
/// table holds exactly one reference per slot.
fn recordOpen(device_token: u64, owner: u32, report_endpoint: usize) bool {
for (&opens) |*open| {
if (open.used and open.device_token == device_token) {
if (open.owner != owner) return false; // someone else's device; nothing kept
if (open.report_endpoint != report_endpoint) _ = ipc.close(open.report_endpoint);
open.report_endpoint = report_endpoint;
return true;
@@ -84,13 +95,33 @@ fn recordOpen(device_token: u64, report_endpoint: usize) bool {
}
for (&opens) |*open| {
if (!open.used) {
open.* = .{ .used = true, .device_token = device_token, .report_endpoint = report_endpoint };
open.* = .{ .used = true, .device_token = device_token, .owner = owner, .report_endpoint = report_endpoint };
return true;
}
}
return false; // table full: not kept
}
/// Whether `device_token` is open to task `owner`. An open device belonging to
/// someone else answers exactly as one that was never opened, so a prober cannot
/// tell another driver's device from an absent one.
fn openedBy(device_token: u64, owner: u32) bool {
for (&opens) |*open| {
if (open.used and open.device_token == device_token) return open.owner == owner;
}
return false;
}
/// Whether `device_token` is open at all. Asked only after `openedBy` has said
/// the caller is not the holder, so an answer of true means *someone else* holds
/// it — one device, one class driver.
fn heldByAnother(device_token: u64) bool {
for (&opens) |*open| {
if (open.used and open.device_token == device_token) return true;
}
return false;
}
fn reportEndpointFor(device_token: u64) ?usize {
for (&opens) |*open| {
if (open.used and open.device_token == device_token) return open.report_endpoint;
@@ -98,6 +129,23 @@ fn reportEndpointFor(device_token: u64) ?usize {
return null;
}
/// Release every device a dead client held: its slot, and the report endpoint
/// capability in it. Driven by published process exits — the same sweep idiom the
/// FAT server uses for open files and the harness uses for subscribers — which is
/// also what lets a restarted class driver re-open the device its predecessor had.
fn releaseOpensOf(dead: u32) void {
for (&opens) |*open| {
if (open.used and open.owner == dead) {
// The engine first: it holds this endpoint's handle number per
// subscription, and the close below frees that number for reuse.
if (controller) |*engine| engine.releaseSubscriptions(open.device_token);
_ = ipc.close(open.report_endpoint);
std.log.info("released device {d} for dead client {d}", .{ open.device_token, dead });
open.* = .{};
}
}
}
var controller_id: u64 = device_manager_protocol.no_device;
/// Claim the assigned controller, find its register window, and hello the
@@ -105,6 +153,10 @@ var controller_id: u64 = device_manager_protocol.no_device;
/// manager reads as "meant to stop" — a missing assignment is not a crash loop.
fn initialise(endpoint: ipc.Handle) bool {
service_endpoint = endpoint;
// Device tokens are per-client state, so this driver needs deaths for the
// same reason the FAT server does: a class driver that crashes must not keep
// its device open, or its restarted instance could never claim it back.
_ = process.subscribeExits(endpoint);
// The transfer contract, bound by hand rather than through the harness's
// `.service`, because **losing it is not fatal here**. One machine can carry
@@ -543,11 +595,16 @@ const handlers = Serve.Handlers{
fn onOpen(_: void, invocation: Invocation(void), answer: Answer(usb_transfer_protocol.Opened)) isize {
const engine = if (controller) |*c| c else return refused;
const found = engine.findInterface(invocation.target) orelse return refused;
// A device another live client holds is refused exactly as an absent one: one
// device, one class driver. The predecessor's slot is released by the exit
// sweep, and notifications are delivered ahead of requests, so a *restarted*
// driver's open always finds the device free.
if (!openedBy(invocation.target, invocation.sender) and heldByAnother(invocation.target)) return refused;
// The report endpoint is claimed only if the open table actually keeps it;
// a full table leaves it to the turn to close.
if (invocation.capability) |endpoint| {
if (recordOpen(invocation.target, endpoint)) capability_claimed = true;
if (recordOpen(invocation.target, invocation.sender, endpoint)) capability_claimed = true;
}
var opened = usb_transfer_protocol.Opened{
@@ -576,6 +633,7 @@ fn onOpen(_: void, invocation: Invocation(void), answer: Answer(usb_transfer_pro
/// written into `answer.tail()` — which is what `Status.len` then reports.
fn onControl(_: void, invocation: Invocation(usb_transfer_protocol.Control), answer: Answer(void)) isize {
const engine = if (controller) |*c| c else return refused;
if (!openedBy(invocation.target, invocation.sender)) return refused;
const found = engine.findInterface(invocation.target) orelse return refused;
const setup = std.mem.bytesToValue(usb_abi.Request, &invocation.request.setup);
@@ -600,6 +658,7 @@ fn onControl(_: void, invocation: Invocation(usb_transfer_protocol.Control), ans
/// to the endpoint this device's `open` handed over.
fn onInterruptSubscribe(_: void, invocation: Invocation(usb_transfer_protocol.InterruptSubscribe), _: Answer(void)) isize {
const engine = if (controller) |*c| c else return refused;
if (!openedBy(invocation.target, invocation.sender)) return refused;
const found = engine.findInterface(invocation.target) orelse return refused;
const endpoint = library.Controller.endpointForAddress(found.interface, invocation.request.endpoint_address) orelse return refused;
const report_endpoint = reportEndpointFor(invocation.target) orelse return refused;
@@ -610,6 +669,7 @@ fn onInterruptSubscribe(_: void, invocation: Invocation(usb_transfer_protocol.In
/// address), so sector-sized data never crosses IPC.
fn onBulk(_: void, invocation: Invocation(usb_transfer_protocol.Bulk), answer: Answer(usb_transfer_protocol.Transferred)) isize {
const engine = if (controller) |*c| c else return refused;
if (!openedBy(invocation.target, invocation.sender)) return refused;
const found = engine.findInterface(invocation.target) orelse return refused;
const endpoint = library.Controller.endpointForAddress(found.interface, invocation.request.endpoint_address) orelse return refused;
const transferred = engine.bulkTransfer(found.device, endpoint, invocation.request.physical_address, invocation.request.length) orelse return refused;
@@ -633,6 +693,12 @@ fn onDmaAttach(_: void, invocation: Invocation(void), _: Answer(void)) isize {
/// arriving after the drain takes IP 0→1 and fires a fresh edge instead of being
/// swallowed until the reconcile tick.
fn onNotification(badge: u64) void {
if (badge & ipc.notify_exit_bit != 0) {
// A class driver died: release the devices it held, so its successor can
// open them and no report is aimed at an endpoint that is gone.
releaseOpensOf(@intCast(badge & ~(ipc.notify_badge_bit | ipc.notify_exit_bit)));
return;
}
if (badge & ipc.notify_timer_bit != 0) {
serviceController();
_ = time.timerOnce(service_endpoint, timerInterval());
@@ -1498,6 +1498,24 @@ pub const Controller = struct {
return true;
}
/// Stop reporting for `device_token`: deactivate every subscription tagged
/// with it, so no further report is queued and the slot can be reused. Called
/// when the class driver that owned the token dies — its endpoint handle is
/// closed with it, and a queued report would then be aimed at a handle number
/// the bus driver has since given to something else. In-process hub
/// subscriptions carry no token and are never touched.
///
/// One completion may already be in flight; it finds no active subscription
/// and is dropped as a foreign transfer event, which is exactly what it is.
pub fn releaseSubscriptions(self: *Controller, device_token: u64) void {
for (&self.subscriptions) |*subscription| {
if (!subscription.active or subscription.hub != null) continue;
if (subscription.device_token != device_token) continue;
subscription.active = false;
subscription.report_endpoint = 0;
}
}
// Arm (or re-arm) a subscription's endpoint with a Normal TRB pointing at its
// report buffer, and ring the endpoint's doorbell so the controller polls it.
fn armInterrupt(self: *Controller, subscription: *Subscription) void {