library: the harness keeps the subscribers, and an id belongs to whoever opened it

Three services had each written the same thing and got it three different
ways: input polled the process list to notice a dead subscriber, and only
when someone else subscribed; the power service never noticed at all; the
device manager noticed drivers but not subscribers. The harness owns the
table now, driven by the events a protocol declares — it registers on the
reserved verb, frames each event once, posts to everyone interested without
waiting on any of them, and reclaims a slot when the kernel says its owner
died. Interest masks moved to the envelope, so a subscriber that wants only
mice asks the same way everywhere.

Two consequences the plan had not foreseen. The device manager now hears a
supervised child's death twice, once as its supervisor and once as a
subscriber, so restart backoff counted every crash twice and gave up after
half as many; it retires the id before counting. And the kernel's published
exit table had eight slots for what is now six subscriptions in a plain
boot, so it holds sixteen.

The other half is a hole the design named early and left standing: a
backend handed out a small integer and then honoured it from anyone. A
process that guessed a file's node id read another client's file; a display
layer had no owner at all, so any client could reconfigure or destroy any
layer; a USB device token was never checked against the client that opened
it. Each is now bound to the task that opened it, and a wrong owner gets
exactly what an unknown id gets — the refusal must not become the oracle
the identical answers elsewhere were designed to remove. Closing a file
changed with it: it used to succeed unconditionally, which would have told
a caller which ids existed.

Suite 111/111, with a new case in which one process holds a file and a
layer, hands both ids to a second process, and finds them untouched after
that process has tried everything with them.
This commit is contained in:
Daniel Samson
2026-08-01 09:05:26 +01:00
parent 2719b93530
commit 1b1c587c14
29 changed files with 1072 additions and 335 deletions
+33 -12
View File
@@ -66,8 +66,9 @@ var ipc_block: IpcBlock = undefined;
var device_dirty: bool = false;
var filesystem: engine.FileSystem = undefined;
// Open handles the VFS holds against this backend: each maps a node id to a
// resolved engine node.
// Open handles clients hold against this backend: each maps a node id to a
// resolved engine node, and to the client that opened it. `owner` is the
// kernel-stamped badge of the opening task — the only source identity there is.
const OpenNode = struct { used: bool = false, node: engine.Node = undefined, owner: u32 = 0 };
var open_nodes = [_]OpenNode{.{}} ** 32;
@@ -78,16 +79,31 @@ fn allocOpen() ?usize {
return null;
}
fn openAt(id: u64) ?*OpenNode {
/// The open node `id` names **for `owner`** — null unless the id is in range, in
/// use, and this client's own. Node ids are small integers drawn from a table of
/// thirty-two, so they are trivially guessable; before this check every client
/// honoured every other client's ids, which is the hole
/// docs/os-development/protocol-namespace.md names ("handles must be scoped per
/// client — validated against the badge"). Nothing else about them changed: they
/// are still per-session, still swept when their owner dies.
///
/// The owner is a *task*, not a process, because the badge is: a threaded client
/// reads and writes a node from the thread that opened it, exactly as the exit
/// sweep already released a worker thread's handles when that thread died.
fn openFor(id: u64, owner: u32) ?*OpenNode {
if (id >= open_nodes.len) return null;
const o = &open_nodes[@intCast(id)];
return if (o.used) o else null;
if (!o.used or o.owner != owner) return null;
return o;
}
/// What a handler returns when the thing asked for is not there — a bad node id,
/// a path that does not resolve, a mutation the volume refused. One errno for all
/// of them, because a filesystem's failures are all "no such thing" as far as the
/// file API can act on them.
/// a node that is someone else's, a path that does not resolve, a mutation the
/// volume refused. One errno for all of them, because a filesystem's failures are
/// all "no such thing" as far as the file API can act on them — and because
/// *someone else's* must be indistinguishable from *nobody's*, or the refusal
/// would itself tell a prober which ids are live (the same discipline the
/// protocol namespace's refused open follows).
const refused: isize = -envelope.ENOENT;
/// How often to look for a block device while none is mounted. Storage arriving
@@ -230,14 +246,14 @@ fn onOpen(_: void, invocation: Invocation(vfs_protocol.Open), answer: Answer(vfs
}
fn onRead(_: void, invocation: Invocation(vfs_protocol.Read), answer: Answer(void)) isize {
const o = openAt(invocation.target) orelse return refused;
const o = openFor(invocation.target, invocation.sender) orelse return refused;
const into = answer.tail();
const want = @min(@as(usize, invocation.request.len), into.len);
return @intCast(filesystem.readFile(o.node, @intCast(invocation.request.offset), into[0..want]));
}
fn onWrite(_: void, invocation: Invocation(vfs_protocol.Write), answer: Answer(vfs_protocol.Written)) isize {
const o = openAt(invocation.target) orelse return refused;
const o = openFor(invocation.target, invocation.sender) orelse return refused;
const data = invocation.tail[0..@min(invocation.tail.len, invocation.request.len)];
const n = filesystem.writeFile(&o.node, @intCast(invocation.request.offset), data);
answer.set(.{ .count = @intCast(n) });
@@ -245,7 +261,7 @@ fn onWrite(_: void, invocation: Invocation(vfs_protocol.Write), answer: Answer(v
}
fn onStatus(_: void, invocation: Invocation(void), answer: Answer(vfs_protocol.FileStatus)) isize {
const o = openAt(invocation.target) orelse return refused;
const o = openFor(invocation.target, invocation.sender) orelse return refused;
const kind: vfs_protocol.NodeKind = if (o.node.is_directory) .directory else .regular;
answer.set(.{ .size = o.node.size, .kind = @intFromEnum(kind), .mtime = o.node.mtime });
return 0;
@@ -255,7 +271,7 @@ fn onStatus(_: void, invocation: Invocation(void), answer: Answer(vfs_protocol.F
/// cursor past the last child — is an entry with no name, which is how the
/// protocol spells it now that the reply's length always counts the fixed part.
fn onReaddir(_: void, invocation: Invocation(vfs_protocol.Readdir), answer: Answer(vfs_protocol.DirectoryEntry)) isize {
const o = openAt(invocation.target) orelse return refused;
const o = openFor(invocation.target, invocation.sender) orelse return refused;
if (!o.node.is_directory) {
answer.set(.{});
return 0;
@@ -272,8 +288,13 @@ fn onReaddir(_: void, invocation: Invocation(vfs_protocol.Readdir), answer: Answ
return @intCast(name_len);
}
/// Closing is an operation on a node like any other, so it is scoped like any
/// other: a client may release its own handles and nobody else's. An id that is
/// not the caller's — free, out of range, or another client's — is refused
/// identically, so a close cannot be used to ask which ids are live either.
fn onClose(_: void, invocation: Invocation(void), _: Answer(void)) isize {
if (openAt(invocation.target)) |o| o.used = false;
const o = openFor(invocation.target, invocation.sender) orelse return refused;
o.used = false;
// Durable-on-close: if any block reached the device since the last flush,
// commit its cache to stable media now (best-effort). This is what makes
// init's shutdown log flush survive a real power-off, and is the right