init: /protocol replaces the ServiceId registry

A protocol is reached by name now, not by a compile-time integer. Init is
PID 1 and already knows which binary it started, so init serves /protocol
as a vfs backend: bind claims a contract with the provider's endpoint
attached, open answers with that endpoint as the reply's capability, and
readdir lists what is bound with the task and binary behind it. The kernel
reserves the prefix — nothing may mount over it, under it, or unmount it —
and ServiceId, ipc_register and ipc_lookup are gone, their syscall numbers
left vacant.

A bind is authorized by who the caller *is*: the kernel-stamped binary
together with the supervising task's identity, matched against
/system/configuration/protocol.csv. Identity, not spelling — spawn is
ungated, so an attacker can run any bundled binary, and a name-only rule
would have let it launder grants through an init of its own making. A name
a live process holds is refused to everyone else; a dead one's is released.

Three review rounds against a hostile ring-3 process found what 108 green
tests could not, because the suite contains no attacker. Publishing init's
supervision endpoint as the registry put PID 1's mailbox in every process's
hands, where two forged bytes reached the shutdown path: privileged traffic
is now believed only from the task that holds the contract it speaks for.
A capability arriving on a request outlived every path that ignored it,
one handle per call until the table was full — in init, and in the harness
ten services share — so the arriving capability is owned by the turn and
released unless a handler says otherwise. And the kernel let anyone holding
an endpoint handle aim signals, timers, exit notices and interrupts at it:
binding now requires having created it.

Suite 108/108. The new protocol-registry case asserts eleven properties,
each one an attack that must fail.
This commit is contained in:
Daniel Samson
2026-08-01 02:39:07 +01:00
parent 1ff0991452
commit 1379b699f3
66 changed files with 2191 additions and 392 deletions
+79 -36
View File
@@ -40,16 +40,13 @@ const Task = scheduler.Task;
pub const MESSAGE_MAXIMUM: usize = 256;
pub const maximum_handles = scheduler.ipc_maximum_handles;
// The name registry is indexed directly by ServiceId, so this must exceed the
// largest id (currently fat = 8). Sized with headroom for new services.
pub const maximum_services = 16;
/// Errno-style failures, returned as `-value` in the system_call result register.
pub const EBADF: i64 = 1; // bad handle
pub const E2BIG: i64 = 2; // message exceeds MESSAGE_MAXIMUM
pub const EFAULT: i64 = 3; // buffer unmapped / out of the user half
pub const ENOENT: i64 = 4; // no such registered service
pub const ENOSPC: i64 = 5; // handle table or registry full
pub const ENOENT: i64 = 4; // no such name
pub const ENOSPC: i64 = 5; // handle table full
pub const ENOMEM: i64 = 6; // out of memory
pub const EPEER: i64 = 7; // peer died before replying (its process exited or was killed)
pub const ESRCH: i64 = 8; // no such process (process_kill of an unknown/dead id)
@@ -90,12 +87,21 @@ const PostSlot = struct {
const user_half_end: u64 = user_memory.user_half_end;
/// A rendezvous endpoint. Allocated from the kernel heap; referenced by handle
/// (per process) and/or by a registry slot, counted by `refcount`.
/// (per process) and by whoever a capability was passed to, counted by `refcount`.
pub const Endpoint = struct {
refcount: u32 = 1,
/// Next in the list of every live endpoint. Endpoints are otherwise reachable
/// only through the handle tables that name them, and the death path has to
/// find a dying task's endpoints without one — see `live_endpoints`.
next_live: ?*Endpoint = null,
// The task that created it. When that task dies, the endpoint is marked `dead` so a caller
// gets -EPEER instead of blocking forever on a service that will never reply again (V6).
owner: u32 = 0,
// The *process* that created it — `owner`'s leader, snapshotted at creation so the
// answer survives the creating thread. `owner` alone cannot answer "is this mine?"
// for a threaded service, and the question has to be answerable after that thread is
// gone; see `ownedBy`.
owner_leader: u32 = 0,
dead: bool = false,
// Callers blocked in `call`, awaiting receive, in FIFO order (threaded via
// Task.next; each such task is .blocked and in no scheduler queue).
@@ -115,29 +121,78 @@ pub const Endpoint = struct {
post_tail: u16 = 0,
};
/// Every live endpoint, singly linked through `next_live`. The list exists for
/// exactly one purpose: the death path must mark a dying task's endpoints dead,
/// and a handle table only answers the other question (which endpoints does this
/// task *hold*). Mutated under the big kernel lock, like every other IPC global.
var live_endpoints: ?*Endpoint = null;
pub fn createIpcEndpoint() ?*Endpoint {
const creator = scheduler.current();
const endpoint = heap.allocator().create(Endpoint) catch return null;
endpoint.* = .{ .owner = scheduler.currentId() };
endpoint.* = .{ .owner = creator.id, .owner_leader = creator.leader, .next_live = live_endpoints };
live_endpoints = endpoint;
return endpoint;
}
/// A task is dying: kill the endpoints it registered as services. Mark each `dead` (so a later
/// `call` returns -EPEER rather than blocking on a reply that will never come), wake anyone
/// already parked sending to it with that error, and vacate its registry slot. Only *registered*
/// endpoints are reachable from here; unregistered ones drop with the task's handle table. The
/// caller holds the big kernel lock (this runs on the death path). See docs/display-v2.md (V6).
/// Whether `t` may have the kernel post **notifications** — signals, timer
/// landings, exit notices, interrupts — into `endpoint`: whether the endpoint is
/// its process's own.
///
/// Holding a *handle* to an endpoint is not ownership of it. `fs_resolve`
/// installs a mounted backend's capability in any caller's table
/// (`installHandleDeduped`), and any capability may be passed along a call, so a
/// sendable handle means only "you may talk to this". A kernel notification is
/// different in kind: it makes the kernel speak *into* someone else's mailbox
/// with a badge that receiver cannot distinguish from one it asked for — a
/// genuine signal badge, a genuine timer landing. That is how a forged
/// `terminate` reached PID 1's shutdown path: the attacker aimed **its own**
/// signal delivery at init's endpoint with `signal_bind` and then signalled
/// itself, and every bit the kernel stamped was authentic. Refusing the *bind*
/// is the only place the distinction still exists.
///
/// Threads: ownership is the **process's**, not the task's, so any thread may
/// bind an endpoint a sibling created — the same normalization `process_signal`
/// and `process_kill` perform when they resolve a member to its leader. The
/// creating task's own id is honoured too, which is what keeps kernel tasks
/// (leader 0) from being treated as one process.
pub fn ownedBy(endpoint: *const Endpoint, t: *const Task) bool {
if (endpoint.owner == t.id) return true;
return t.leader != 0 and endpoint.owner_leader == t.leader;
}
/// Unlink a freed endpoint from the live list. O(n) in the number of live
/// endpoints, which is tens.
fn forgetEndpoint(endpoint: *Endpoint) void {
var link = &live_endpoints;
while (link.*) |current| {
if (current == endpoint) {
link.* = current.next_live;
return;
}
link = &current.next_live;
}
}
/// A task is dying: kill every endpoint it created. Mark each `dead` (so a later
/// `call` returns -EPEER rather than blocking on a reply that will never come) and wake
/// anyone already parked sending to it with that error. This is what makes a provider's
/// death visible to the clients holding its capability — the naming layer's restart
/// story (a client re-resolves on -EPEER) rests on it, as does the VFS router's lazy
/// unmount of a backend that died. The endpoint object itself lives until the last
/// handle naming it drops. The caller holds the big kernel lock (this runs on the death
/// path). See docs/display-v2.md (V6).
pub fn killOwnedEndpointsLocked(task_id: u32) void {
for (&registry) |*slot| {
const endpoint = slot.* orelse continue;
if (endpoint.owner != task_id) continue;
var current = live_endpoints;
while (current) |endpoint| {
current = endpoint.next_live;
if (endpoint.owner != task_id or endpoint.dead) continue;
endpoint.dead = true;
while (dequeueSender(endpoint)) |sender| {
sender.ipc_status = -EPEER;
sender.ipc_received_cap = abi.no_cap;
scheduler.readyLocked(sender);
}
slot.* = null;
dropRef(endpoint);
}
}
@@ -147,6 +202,7 @@ pub fn dropRef(endpoint: *Endpoint) void {
if (endpoint.refcount > 1) {
endpoint.refcount -= 1;
} else {
forgetEndpoint(endpoint);
heap.allocator().destroy(endpoint);
}
}
@@ -633,22 +689,9 @@ fn dropEntry(entry: scheduler.HandleObject) void {
}
}
var registry: [maximum_services]?*Endpoint = .{null} ** maximum_services;
/// Publish `endpoint` under well-known `id` (takes a reference). Returns 0 or -errno.
pub fn register(id: u32, endpoint: *Endpoint) i64 {
if (id >= maximum_services) return -ENOENT;
if (registry[id]) |old| dropRef(old);
endpoint.refcount += 1;
registry[id] = endpoint;
return 0;
}
/// Find the endpoint published under `id`, taking a reference for the caller to
/// install in its handle table. Null if nothing is registered there.
pub fn lookup(id: u32) ?*Endpoint {
if (id >= maximum_services) return null;
const endpoint = registry[id] orelse return null;
endpoint.refcount += 1;
return endpoint;
}
// The flat `ServiceId` registry lived here — a 16-slot table any process could
// write, indexed by a compile-time enum. Naming is user-space's job now: init
// serves `/protocol` and decides who may claim a name
// (docs/os-development/protocol-namespace.md). The kernel keeps only what is
// genuinely kernel work — moving capabilities and telling clients their provider
// died (`killOwnedEndpointsLocked`).
+2 -2
View File
@@ -47,8 +47,8 @@ pub const maximum_gsi = 24;
var bound: [maximum_gsi]?*ipc_sync.Endpoint = .{null} ** maximum_gsi;
/// Task that owns each binding. Teardown is keyed on *this*, not on the endpoint
/// pointer: an endpoint can be shared between processes (ipc_register/ipc_lookup hand
/// out extra references), so "every GSI pointing at this endpoint" is not the same
/// pointer: an endpoint can be shared between processes (a capability passed in a message
/// hands out extra references), so "every GSI pointing at this endpoint" is not the same
/// set as "every GSI this process bound", and releasing the former on exit would mask
/// a live sibling's device line.
var bound_owner: [maximum_gsi]u32 = .{0} ** maximum_gsi;
+50 -41
View File
@@ -229,8 +229,6 @@ fn system_call(state: *architecture.CpuState) void {
.mmap => systemMmap(state),
.munmap => systemMunmap(state),
.create_ipc_endpoint => systemCreateIpcEndpoint(state),
.ipc_register => systemIpcRegister(state),
.ipc_lookup => systemIpcLookup(state),
.ipc_call => systemIpcCall(state),
.ipc_reply_wait => systemIpcReplyWait(state),
.ipc_send => systemIpcSend(state),
@@ -320,36 +318,6 @@ fn systemCreateIpcEndpoint(state: *architecture.CpuState) void {
architecture.setSystemCallResult(state, @intCast(h));
}
/// ipc_register(service_id, handle): publish the caller's endpoint under a
/// well-known id so other processes can find it.
fn systemIpcRegister(state: *architecture.CpuState) void {
// Under the big kernel lock: mutates the global service registry and endpoint
// refcounts, which threads of the same (or another) process can race.
const flags = sync.enter();
defer sync.leave(flags);
const id: u32 = @truncate(architecture.systemCallArg(state, 0));
const endpoint = ipc.resolveHandle(scheduler.current(), architecture.systemCallArg(state, 1)) orelse return failErr(state, ipc.EBADF);
architecture.setSystemCallResult(state, @bitCast(ipc.register(id, endpoint)));
}
/// ipc_lookup(service_id) -> handle: find a published endpoint and install a
/// handle to it in the caller.
fn systemIpcLookup(state: *architecture.CpuState) void {
// Under the big kernel lock: reads the global registry, takes an endpoint reference,
// and installs a handle — all racy against concurrent threads (this is the path the
// display's mouse-listener thread takes to reach the compositor endpoint).
const flags = sync.enter();
defer sync.leave(flags);
const id: u32 = @truncate(architecture.systemCallArg(state, 0));
const endpoint = ipc.lookup(id) orelse return failErr(state, ipc.ENOENT);
const h = ipc.installHandle(scheduler.current(), endpoint);
if (h < 0) {
ipc.dropRef(endpoint);
return failErr(state, ipc.ENOSPC);
}
architecture.setSystemCallResult(state, @intCast(h));
}
/// ipc_call(handle, message_ptr, message_len, reply_ptr, reply_cap) -> reply_len.
/// Blocks until the server replies; the trap frame lives on this task's kernel
/// stack, so it survives the block and receives the result on resume.
@@ -989,10 +957,17 @@ fn systemSpawn(state: *architecture.CpuState) void {
if (len == 0 or len > scheduler.maximum_task_name or ptr >= user_half_end or ptr + len > user_half_end) return fail(state);
if (arguments_len > maximum_argument_bytes) return fail(state);
if (arguments_len != 0 and (arguments_ptr >= user_half_end or arguments_ptr + arguments_len > user_half_end)) return fail(state);
// The exit endpoint is a notification binding like signal_bind's and
// timer_bind's, so it obeys the same rule: the caller's own mailbox, never a
// stranger's. Otherwise any process could have the kernel post child-exit
// badges into PID 1 by spawning throwaway children against init's endpoint.
const exit_endpoint: ?*ipc.Endpoint = if (exit_handle == abi.no_cap)
null
else
ipc.resolveHandle(t, exit_handle) orelse return failErr(state, ipc.EBADF);
else block: {
const endpoint = ipc.resolveHandle(t, exit_handle) orelse return failErr(state, ipc.EBADF);
if (!ipc.ownedBy(endpoint, t)) return failErr(state, ipc.EPERM);
break :block endpoint;
};
const image = ramdisk_image orelse return fail(state);
const rd = initial_ramdisk.Reader.init(image) orelse return fail(state);
@@ -1040,11 +1015,15 @@ fn systemThreadSpawn(state: *architecture.CpuState) void {
if (t.address_space == 0) return fail(state); // kernel tasks own no address space to share
if (entry == 0 or entry >= user_half_end) return fail(state);
if (stack_top == 0 or stack_top > user_half_end) return fail(state);
// The endpoint the thread notifies on exit (how join waits), or none.
// The endpoint the thread notifies on exit (how join waits), or none — the
// caller's own, like every other notification binding.
const exit_endpoint: ?*ipc.Endpoint = if (exit_handle == abi.no_cap)
null
else
ipc.resolveHandle(t, exit_handle) orelse return failErr(state, ipc.EBADF);
else block: {
const endpoint = ipc.resolveHandle(t, exit_handle) orelse return failErr(state, ipc.EBADF);
if (!ipc.ownedBy(endpoint, t)) return failErr(state, ipc.EPERM);
break :block endpoint;
};
const tid = spawnThreadSupervised(t.address_space, entry, stack_top, arg, t.priority, t.id, exit_endpoint, t.leader);
if (tid == -ipc.ESRCH) return failErr(state, ipc.ESRCH); // dying group admits no member
if (tid < 0) return fail(state);
@@ -1538,13 +1517,20 @@ const exit_subscriber_capacity = 8;
const ExitSubscriber = struct { endpoint: *ipc.Endpoint, owner: u32 };
var exit_subscribers: [exit_subscriber_capacity]?ExitSubscriber = .{null} ** exit_subscriber_capacity;
/// process_subscribe(endpoint): subscribe the caller's endpoint to published exit
/// events. Ungated, like process_enumerate — what is running (and dying) is not a
/// secret between cooperating processes. -ENOSPC when the table is full.
/// process_subscribe(endpoint): subscribe the **caller's own** endpoint to
/// published exit events. *Which* deaths one may hear of is ungated, like
/// process_enumerate — what is running (and dying) is not a secret between
/// cooperating processes. *Whose mailbox* they land in is not: the endpoint must
/// be the caller's (`ipc.ownedBy`), or any process could aim the firehose at a
/// stranger — filling PID 1's mailbox with exit notices it reads as its own
/// children's, and spending the eight-slot table so the services that need
/// deaths (the VFS's handle sweep) cannot subscribe at all. -EPERM otherwise,
/// -ENOSPC when the table is full.
fn systemProcessSubscribe(state: *architecture.CpuState) void {
const t = scheduler.current();
if (t.address_space == 0) return fail(state);
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
if (!ipc.ownedBy(endpoint, t)) return failErr(state, ipc.EPERM);
const flags = sync.enter();
defer sync.leave(flags);
for (&exit_subscribers) |*slot| {
@@ -1561,10 +1547,19 @@ fn systemProcessSubscribe(state: *architecture.CpuState) void {
/// IRQ-as-IPC pattern a fourth time (docs/process-lifecycle.md). Replacing a
/// binding drops the old reference; signals that pended while unbound are
/// delivered immediately on bind, coalesced into one notification.
///
/// The endpoint must be the caller's own (`ipc.ownedBy`), or `signal_bind`
/// becomes a signal *forgery* primitive: `process_signal` is deliberately loose
/// about the target (a task may always signal itself) because the delivery point
/// was assumed to be the target's own mailbox. Aim it elsewhere and a stranger
/// signalling itself makes the kernel stamp a genuine `terminate` badge into
/// somebody else's queue — which is a shutdown request PID 1 has no way to
/// disbelieve.
fn systemSignalBind(state: *architecture.CpuState) void {
const t = scheduler.current();
if (t.address_space == 0) return fail(state);
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
if (!ipc.ownedBy(endpoint, t)) return failErr(state, ipc.EPERM);
const flags = sync.enter();
defer sync.leave(flags);
if (t.signal_endpoint) |raw| ipc.dropRef(@ptrCast(@alignCast(raw)));
@@ -1633,11 +1628,19 @@ fn timerSweepLocked() void {
}
}
/// timer_bind(endpoint, ms): arm a one-shot timer. -ENOSPC when the table is full.
/// timer_bind(endpoint, ms): arm a one-shot timer on an endpoint of the caller's
/// own (`ipc.ownedBy`; -EPERM otherwise). A timer landing carries no identity —
/// that is the whole reason a service may keep exactly one in flight — so a
/// timer armed on someone else's endpoint is indistinguishable from one they
/// armed themselves, and a loop that re-arms on every landing (init's heartbeat)
/// multiplies: N forged timers leave N+1 self-perpetuating beats. The
/// sixteen-slot table is a shared resource on top of that. -ENOSPC when it is
/// full.
fn systemTimerBind(state: *architecture.CpuState) void {
const t = scheduler.current();
if (t.address_space == 0) return fail(state);
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
if (!ipc.ownedBy(endpoint, t)) return failErr(state, ipc.EPERM);
const ms = architecture.systemCallArg(state, 1);
const flags = sync.enter();
defer sync.leave(flags);
@@ -1677,12 +1680,17 @@ fn ownedGsi(t: *scheduler.Task, device_id: u64, resource_index: u64) ?u32 {
/// irq_bind(device_id, resource_index, endpoint) -> 0/-1: deliver that device's IRQ to the
/// endpoint as an asynchronous IPC notification. The driver then blocks in
/// IPC_ReplyWait and is woken by the ISR; see system/kernel/irq.zig for the cycle.
/// Two gates, both necessary: the device must be *claimed* by the caller
/// (`ownedGsi`), and the endpoint must be the caller's own (`ipc.ownedBy`) — a
/// claim entitles a driver to its own interrupts, not to post them into a
/// stranger's mailbox.
fn systemIrqBind(state: *architecture.CpuState) void {
const t = scheduler.current();
if (t.address_space == 0) return fail(state);
const gsi = ownedGsi(t, architecture.systemCallArg(state, 0), architecture.systemCallArg(state, 1)) orelse
return fail(state);
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 2)) orelse return fail(state);
if (!ipc.ownedBy(endpoint, t)) return failErr(state, ipc.EPERM);
const flags = sync.enter();
defer sync.leave(flags);
@@ -1703,6 +1711,7 @@ fn systemMsiBind(state: *architecture.CpuState) void {
const owner = devices_broker.ownerOf(device_id) orelse return fail(state);
if (owner != t.id) return fail(state); // not claimed by this process
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 1)) orelse return failErr(state, ipc.EBADF);
if (!ipc.ownedBy(endpoint, t)) return failErr(state, ipc.EPERM); // interrupts land in your own mailbox
const flags = sync.enter();
defer sync.leave(flags);
+116 -17
View File
@@ -246,6 +246,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
containmentTest();
} else if (eql(case, "device-manager")) {
deviceManagerTest(boot_information);
} else if (eql(case, "protocol-registry")) {
protocolRegistryTest(boot_information);
} else if (eql(case, "reboot")) {
rebootTest();
} else {
@@ -2209,7 +2211,7 @@ fn processKillTest(boot_information: *const BootInformation) void {
while (i < rd.count) : (i += 1) {
const item = rd.entry(i) orelse continue;
if (!eql(initial_ramdisk.basename(item.name), "process-test")) continue;
spinner = process.spawnProcessSupervised(item.blob, 4, &.{ "process-test", "spinner" }, me, endpoint) catch 0;
spinner = process.spawnProcessSupervised(item.blob, 4, &.{ item.name, "spinner" }, me, endpoint) catch 0;
break;
}
check("process-test spawned as the supervised spinner victim", spinner != 0);
@@ -2395,13 +2397,14 @@ fn signalsTest(boot_information: *const BootInformation) void {
};
process.setInitialRamdisk(image); // the parent system_spawns its children by name
_ = spawnRegistry(rd); // the service child binds /protocol/test/process
process.write_count = 0;
var runner: u32 = 0;
var i: u32 = 0;
while (i < rd.count) : (i += 1) {
const item = rd.entry(i) orelse continue;
if (!eql(initial_ramdisk.basename(item.name), "process-test")) continue;
runner = process.spawnProcessSupervised(item.blob, 4, &.{ "process-test", "signal-run" }, scheduler.currentId(), null) catch 0;
runner = process.spawnProcessSupervised(item.blob, 4, &.{ item.name, "signal-run" }, scheduler.currentId(), null) catch 0;
break;
}
check("signal-run parent spawned", runner != 0);
@@ -2443,13 +2446,14 @@ fn driverRestartTest(boot_information: *const BootInformation) void {
};
process.setInitialRamdisk(image); // the manager system_spawns drivers by name
_ = spawnRegistry(rd); // the drivers bind their contracts
process.write_count = 0;
var manager: u32 = 0;
var i: u32 = 0;
while (i < rd.count) : (i += 1) {
const item = rd.entry(i) orelse continue;
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
manager = process.spawnProcessSupervised(item.blob, 4, &.{ "device-manager", "test-restart" }, scheduler.currentId(), null) catch 0;
manager = process.spawnProcessSupervised(item.blob, 4, &.{ item.name, "test-restart" }, scheduler.currentId(), null) catch 0;
break;
}
check("device-manager spawned in test-restart mode", manager != 0);
@@ -2482,12 +2486,13 @@ fn usbReportTest(boot_information: *const BootInformation) void {
};
process.setInitialRamdisk(image);
_ = spawnRegistry(rd); // the xhci driver binds /protocol/usb-transfer
var manager: u32 = 0;
var i: u32 = 0;
while (i < rd.count) : (i += 1) {
const item = rd.entry(i) orelse continue;
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
manager = process.spawnProcessSupervised(item.blob, 4, &.{ "device-manager", "test-usb-restart" }, scheduler.currentId(), null) catch 0;
manager = process.spawnProcessSupervised(item.blob, 4, &.{ item.name, "test-usb-restart" }, scheduler.currentId(), null) catch 0;
break;
}
check("device-manager spawned in test-usb-restart mode", manager != 0);
@@ -2514,12 +2519,13 @@ fn deviceListTest(boot_information: *const BootInformation) void {
};
process.setInitialRamdisk(image);
_ = spawnRegistry(rd); // the fixture opens /protocol/device-manager
var manager: u32 = 0;
var i: u32 = 0;
while (i < rd.count) : (i += 1) {
const item = rd.entry(i) orelse continue;
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
manager = process.spawnProcessSupervised(item.blob, 4, &.{ "device-manager", "test-usb-restart" }, scheduler.currentId(), null) catch 0;
manager = process.spawnProcessSupervised(item.blob, 4, &.{ item.name, "test-usb-restart" }, scheduler.currentId(), null) catch 0;
break;
}
check("device-manager spawned in test-usb-restart mode", manager != 0);
@@ -2548,13 +2554,14 @@ fn pciCapsTest(boot_information: *const BootInformation) void {
};
process.setInitialRamdisk(image);
_ = spawnRegistry(rd); // the manager binds /protocol/device-manager
// Plain mode — no restart drill, whose kill would race the fixture's claim.
var manager: u32 = 0;
var i: u32 = 0;
while (i < rd.count) : (i += 1) {
const item = rd.entry(i) orelse continue;
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
manager = process.spawnProcessSupervised(item.blob, 4, &.{"device-manager"}, scheduler.currentId(), null) catch 0;
manager = process.spawnProcessSupervised(item.blob, 4, &.{item.name}, scheduler.currentId(), null) catch 0;
break;
}
check("device-manager spawned", manager != 0);
@@ -2582,12 +2589,13 @@ fn iommuFaultTest(boot_information: *const BootInformation) void {
check("IOMMU enabled for the enforcement test", iommu.enabled());
process.setInitialRamdisk(image);
_ = spawnRegistry(rd); // the manager binds /protocol/device-manager
var manager: u32 = 0;
var i: u32 = 0;
while (i < rd.count) : (i += 1) {
const item = rd.entry(i) orelse continue;
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
manager = process.spawnProcessSupervised(item.blob, 4, &.{"device-manager"}, scheduler.currentId(), null) catch 0;
manager = process.spawnProcessSupervised(item.blob, 4, &.{item.name}, scheduler.currentId(), null) catch 0;
break;
}
check("device-manager spawned", manager != 0);
@@ -2623,12 +2631,13 @@ fn pciScanTest(boot_information: *const BootInformation) void {
check("the kernel seeded no PCI functions (the walk retired)", brokerPciCount(&buffer) == 0);
process.setInitialRamdisk(image);
_ = spawnRegistry(rd); // the manager binds /protocol/device-manager
var manager: u32 = 0;
var i: u32 = 0;
while (i < rd.count) : (i += 1) {
const item = rd.entry(i) orelse continue;
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
manager = process.spawnProcessSupervised(item.blob, 4, &.{ "device-manager", "test-pci-restart" }, scheduler.currentId(), null) catch 0;
manager = process.spawnProcessSupervised(item.blob, 4, &.{ item.name, "test-pci-restart" }, scheduler.currentId(), null) catch 0;
break;
}
check("device-manager spawned (test-pci-restart mode)", manager != 0);
@@ -2776,12 +2785,13 @@ fn acpiReportTest(boot_information: *const BootInformation) void {
return;
};
process.setInitialRamdisk(image);
_ = spawnRegistry(rd); // the manager and the acpi service bind theirs
var spawned = false;
var i: u32 = 0;
while (i < rd.count) : (i += 1) {
const item = rd.entry(i) orelse continue;
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
_ = process.spawnProcessSupervised(item.blob, 4, &.{"device-manager"}, scheduler.currentId(), null) catch 0;
_ = process.spawnProcessSupervised(item.blob, 4, &.{item.name}, scheduler.currentId(), null) catch 0;
spawned = true;
break;
}
@@ -2819,7 +2829,7 @@ fn acpiParseTest(boot_information: *const BootInformation) void {
while (i < rd.count) : (i += 1) {
const item = rd.entry(i) orelse continue;
if (!eql(initial_ramdisk.basename(item.name), "discovery")) continue;
_ = process.spawnProcessSupervised(item.blob, 4, &.{ "discovery", "1" }, scheduler.currentId(), null) catch 0;
_ = process.spawnProcessSupervised(item.blob, 4, &.{ item.name, "1" }, scheduler.currentId(), null) catch 0;
spawned = true;
break;
}
@@ -2854,7 +2864,7 @@ fn supervisionTest(boot_information: *const BootInformation) void {
while (i < rd.count) : (i += 1) {
const item = rd.entry(i) orelse continue;
if (!eql(initial_ramdisk.basename(item.name), "process-test")) continue;
started = if (process.spawnProcess(item.blob, 4, &.{ "process-test", "run" })) true else |_| false;
started = if (process.spawnProcess(item.blob, 4, &.{ item.name, "run" })) true else |_| false;
break;
}
check("process-test spawned as the user-space supervisor", started);
@@ -2996,6 +3006,9 @@ fn inputTest(boot_information: *const BootInformation) void {
process.write_count = 0;
process.write_from_user = false;
// init (the registry, below) reads its manifests through the kernel VFS.
process.setInitialRamdisk(image);
_ = spawnRegistry(rd); // the input service binds /protocol/input
_ = spawnNamed(rd, "input"); // the fan-out service
_ = spawnNamed(rd, "input-source"); // a synthetic keyboard publishing events
_ = spawnNamed(rd, "input-test"); // the subscriber whose "ok" line is the marker
@@ -3037,6 +3050,10 @@ fn displayServiceTest(boot_information: *const BootInformation) void {
return;
};
// init (the registry) reads its manifests through the kernel VFS.
process.setInitialRamdisk(image);
_ = spawnRegistry(rd); // the compositor binds /protocol/display
// Spawn the compositor and hand it the core. Its own serial heartbeats — `display:
// online WxH` and `display: presented frame 0` — are what the harness matches (it
// reads serial directly, like the fault cases). We don't poll for them in-kernel: a
@@ -3073,6 +3090,9 @@ fn displayCursorTest(boot_information: *const BootInformation) void {
return;
};
// init (the registry, below) reads its manifests through the kernel VFS.
process.setInitialRamdisk(image);
_ = spawnRegistry(rd); // input and display bind theirs
if (!spawnNamed(rd, "input")) {
log("display-cursor: could not spawn the input service\n", .{});
result();
@@ -3113,6 +3133,9 @@ fn displayDemoTest(boot_information: *const BootInformation) void {
return;
};
// init (the registry, below) reads its manifests through the kernel VFS.
process.setInitialRamdisk(image);
_ = spawnRegistry(rd); // the compositor binds /protocol/display
if (!spawnNamed(rd, "display")) {
log("display-demo: could not spawn the display service\n", .{});
result();
@@ -3148,6 +3171,9 @@ fn sharedMemoryTest(boot_information: *const BootInformation) void {
return;
};
// init (the registry, below) reads its manifests through the kernel VFS.
process.setInitialRamdisk(image);
_ = spawnRegistry(rd); // the server binds /protocol/test/shared-memory
if (!spawnNamed(rd, "shared-memory-server")) {
log("shared-memory: could not spawn shared-memory-server\n", .{});
result();
@@ -3185,12 +3211,13 @@ fn virtioGpuTest(boot_information: *const BootInformation) void {
// from the kernel device tree, spawns pci-bus, and matches the virtio-gpu class triple to
// spawn our driver with the function's device id as argv[1].
process.setInitialRamdisk(image);
_ = spawnRegistry(rd); // the driver binds /protocol/scanout
var manager: u32 = 0;
var i: u32 = 0;
while (i < rd.count) : (i += 1) {
const item = rd.entry(i) orelse continue;
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
manager = process.spawnProcessSupervised(item.blob, 4, &.{"device-manager"}, scheduler.currentId(), null) catch 0;
manager = process.spawnProcessSupervised(item.blob, 4, &.{item.name}, scheduler.currentId(), null) catch 0;
break;
}
if (manager == 0) {
@@ -3226,12 +3253,13 @@ fn displayNativeTest(boot_information: *const BootInformation) void {
};
process.setInitialRamdisk(image);
_ = spawnRegistry(rd); // display, the manager, and the driver bind theirs
var manager: u32 = 0;
var i: u32 = 0;
while (i < rd.count) : (i += 1) {
const item = rd.entry(i) orelse continue;
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
manager = process.spawnProcessSupervised(item.blob, 4, &.{"device-manager"}, scheduler.currentId(), null) catch 0;
manager = process.spawnProcessSupervised(item.blob, 4, &.{item.name}, scheduler.currentId(), null) catch 0;
break;
}
if (manager == 0) {
@@ -3270,12 +3298,13 @@ fn displayReattachTest(boot_information: *const BootInformation) void {
};
process.setInitialRamdisk(image);
_ = spawnRegistry(rd); // display and the restarted driver bind theirs
var manager: u32 = 0;
var i: u32 = 0;
while (i < rd.count) : (i += 1) {
const item = rd.entry(i) orelse continue;
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
manager = process.spawnProcessSupervised(item.blob, 4, &.{ "device-manager", "test-scanout-restart" }, scheduler.currentId(), null) catch 0;
manager = process.spawnProcessSupervised(item.blob, 4, &.{ item.name, "test-scanout-restart" }, scheduler.currentId(), null) catch 0;
break;
}
if (manager == 0) {
@@ -3625,6 +3654,21 @@ fn threadTestMarkerCase(boot_information: *const BootInformation, case_name: []c
result();
}
/// Bring up the protocol namespace for a scenario that spawns its providers
/// itself. `/protocol` is served by init, PID 1 — but a scenario case wants the
/// naming layer without init's whole service list underneath it, so init is
/// started in its `registry` role: it mounts `/protocol`, reads the grants, and
/// spawns nothing (docs/os-development/protocol-namespace.md; the plan's
/// decision 9). Providers retry their bind, so racing the mount is survivable —
/// but calling this first makes the race rare.
///
/// The caller must have published the initial ramdisk already
/// (`process.setInitialRamdisk`): init reads its manifests out of it, and every
/// `/protocol` resolve goes through the same kernel VFS.
fn spawnRegistry(rd: initial_ramdisk.Reader) bool {
return spawnNamedWithArg(rd, "init", "registry");
}
fn spawnNamed(rd: initial_ramdisk.Reader, name: []const u8) bool {
var i: u32 = 0;
while (i < rd.count) : (i += 1) {
@@ -3745,6 +3789,60 @@ fn childDescriptor(hid: []const u8, start: u64, len: u64) device_abi.DeviceDescr
/// match `pci-bus`, and spawn it (with the bridge id as its argument) — and the spawned
/// pci-bus must reach its own live marker. It uses no special privilege — the same
/// `device_enumerate` any process could call.
/// P2 — the registrar (docs/os-development/protocol-namespace.md). Bring up
/// `/protocol` (init in its registry role) and hand the fixture the core: it
/// asserts that an ungranted bind is refused, that the kernel's reserved prefix
/// holds, that a name a live provider holds cannot be taken, and that killing a
/// provider makes its channel fail while re-resolving the same name reaches the
/// restarted instance.
///
/// It doubles as the security case for PID 1's shared mailbox, since resolving
/// `/protocol` hands every process a sendable handle to it: a forged power
/// payload, a redirected terminate signal, a timer or exit subscription armed on
/// a foreign endpoint, and capability-carrying ping storms against both PID 1 and
/// a harness-run service. Those assertions kill the boot when they regress rather
/// than printing anything, which is the strongest form available here.
///
/// The fixture's `protocol-registry: ok` is the marker; each step also prints its
/// own line, which the harness's ordered regex reads.
fn protocolRegistryTest(boot_information: *const BootInformation) void {
log("DANOS-TEST-BEGIN: protocol-registry\n", .{});
if (boot_information.initial_ramdisk_len == 0) {
check("bootloader handed over an initial_ramdisk", false);
result();
return;
}
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
const rd = initial_ramdisk.Reader.init(image) orelse {
check("initial_ramdisk image is valid", false);
result();
return;
};
// The fixture spawns its own providers by name, so the ramdisk must be
// published; init then mounts /protocol over the same kernel VFS.
process.setInitialRamdisk(image);
check("registry (init) spawned", spawnRegistry(rd));
check("protocol-registry-test spawned", spawnNamedWithArg(rd, "protocol-registry-test", "run"));
const pass_marker = "protocol-registry: ok";
const fail_marker = "protocol-registry: FAIL";
scheduler.setPriority(1);
const deadline = architecture.millis() + 20000;
var saw_pass = false;
var saw_fail = false;
while (architecture.millis() < deadline and !saw_pass and !saw_fail) {
if (bufferHas(pass_marker)) saw_pass = true;
if (bufferHas(fail_marker)) saw_fail = true;
scheduler.yield();
}
scheduler.setPriority(4);
check("no step of the registry contract failed", !saw_fail);
check("the fixture completed every registry assertion", saw_pass);
result();
}
fn deviceManagerTest(boot_information: *const BootInformation) void {
log("DANOS-TEST-BEGIN: device-manager\n", .{});
if (boot_information.initial_ramdisk_len == 0) {
@@ -3764,6 +3862,7 @@ fn deviceManagerTest(boot_information: *const BootInformation) void {
// all, it's because the manager discovered the PCI host bridge, matched, and
// spawned it.
process.setInitialRamdisk(image);
_ = spawnRegistry(rd); // the manager binds /protocol/device-manager
process.write_count = 0;
process.write_from_user = false;
@@ -3845,9 +3944,9 @@ fn hpetDeviceId() ?u64 {
/// 2. After `releaseOwner` for the binding's owner, that same entry is masked again.
///
/// And one property that can only be checked from kernel state: a *different* owner's
/// binding on the same endpoint survives. Endpoints are shared (ipc_register hands out
/// references), so teardown keyed on the endpoint pointer rather than the owning task
/// would mask a live sibling driver's device line.
/// binding on the same endpoint survives. Endpoints are shared (a capability passed in a
/// message hands out extra references), so teardown keyed on the endpoint pointer rather
/// than the owning task would mask a live sibling driver's device line.
fn irqFreeTest() void {
log("DANOS-TEST-BEGIN: irqfree\n", .{});
+33
View File
@@ -168,6 +168,11 @@ fn installMount(prefix: []const u8, kind: MountKind, backend: ?*ipc.Endpoint, re
var slot: ?*Mount = null;
for (&mounts) |*m| {
if (m.used and std.mem.eql(u8, m.prefixSlice(), prefix)) {
// ...except the protocol namespace. Remount-replace is how a
// restarted FAT retakes /volumes/usb; letting it retake /protocol
// would hand the whole naming layer to whoever asked second.
// First mount wins, and init (PID 1) is always first.
if (std.mem.eql(u8, prefix, protocol_root)) return;
if (m.backend) |old| ipc.dropRef(old);
slot = m;
break;
@@ -346,6 +351,30 @@ fn isInitrdCarveOut(prefix: []const u8) bool {
return false;
}
/// The protocol namespace's root — a reserved prefix, like the initrd trees.
/// Init (PID 1) mounts the registry here once at boot and the prefix then
/// refuses everything: a second mount at it, any mount *under* it (which would
/// shadow one contract), and its unmount. That is the whole kernel-side residue
/// of the naming layer — the registrar authority itself never leaves init
/// (docs/os-development/protocol-namespace.md).
const protocol_root = "/protocol";
fn protocolBound() bool {
for (&mounts) |*m| {
if (m.used and std.mem.eql(u8, m.prefixSlice(), protocol_root)) return true;
}
return false;
}
/// Whether mounting at `prefix` would touch the protocol namespace. Exactly
/// `/protocol` is allowed once — while nothing holds it; anything under it,
/// ever, is refused.
fn refusesProtocolMount(prefix: []const u8) bool {
const relative = underMount(prefix, protocol_root) orelse return false;
if (relative.len != 1) return true; // strictly under /protocol: never
return protocolBound(); // /protocol itself: first mount wins
}
/// Mount `backend` at `prefix` with an optional backend-side `rewrite` prefix.
/// The endpoint reference is taken by the caller (process.zig bumps it); refuses
/// shadowing or replacing the initrd trees (/system, /test) — except the two
@@ -353,6 +382,7 @@ fn isInitrdCarveOut(prefix: []const u8) bool {
pub fn mountBackend(prefix: []const u8, backend: *ipc.Endpoint, rewrite: []const u8) bool {
if (!isAbsolute(prefix) or prefix.len < 2 or prefix.len > maximum_prefix) return false;
if (rewrite.len > maximum_rewrite) return false;
if (refusesProtocolMount(prefix)) return false; // the registry's prefix is claimed once
for (&mounts) |*m| { // the initrd trees are not shadowable (carve-outs aside)
if (m.used and m.kind == .kernel_initrd and underMount(prefix, m.prefixSlice()) != null) {
if (!isInitrdCarveOut(prefix)) return false;
@@ -363,6 +393,9 @@ pub fn mountBackend(prefix: []const u8, backend: *ipc.Endpoint, rewrite: []const
}
pub fn unmount(prefix: []const u8) bool {
// Unmounting /protocol would delete the naming layer for everyone; nobody
// may, init included. The mount lasts the boot.
if (std.mem.eql(u8, prefix, protocol_root)) return false;
for (&mounts) |*m| {
if (m.used and m.kind == .backend and std.mem.eql(u8, m.prefixSlice(), prefix)) {
if (m.backend) |endpoint| ipc.dropRef(endpoint);