init: /protocol replaces the ServiceId registry
A protocol is reached by name now, not by a compile-time integer. Init is PID 1 and already knows which binary it started, so init serves /protocol as a vfs backend: bind claims a contract with the provider's endpoint attached, open answers with that endpoint as the reply's capability, and readdir lists what is bound with the task and binary behind it. The kernel reserves the prefix — nothing may mount over it, under it, or unmount it — and ServiceId, ipc_register and ipc_lookup are gone, their syscall numbers left vacant. A bind is authorized by who the caller *is*: the kernel-stamped binary together with the supervising task's identity, matched against /system/configuration/protocol.csv. Identity, not spelling — spawn is ungated, so an attacker can run any bundled binary, and a name-only rule would have let it launder grants through an init of its own making. A name a live process holds is refused to everyone else; a dead one's is released. Three review rounds against a hostile ring-3 process found what 108 green tests could not, because the suite contains no attacker. Publishing init's supervision endpoint as the registry put PID 1's mailbox in every process's hands, where two forged bytes reached the shutdown path: privileged traffic is now believed only from the task that holds the contract it speaks for. A capability arriving on a request outlived every path that ignored it, one handle per call until the table was full — in init, and in the harness ten services share — so the arriving capability is owned by the turn and released unless a handler says otherwise. And the kernel let anyone holding an endpoint handle aim signals, timers, exit notices and interrupts at it: binding now requires having created it. Suite 108/108. The new protocol-registry case asserts eleven properties, each one an attack that must fail.
This commit is contained in:
+50
-41
@@ -229,8 +229,6 @@ fn system_call(state: *architecture.CpuState) void {
|
||||
.mmap => systemMmap(state),
|
||||
.munmap => systemMunmap(state),
|
||||
.create_ipc_endpoint => systemCreateIpcEndpoint(state),
|
||||
.ipc_register => systemIpcRegister(state),
|
||||
.ipc_lookup => systemIpcLookup(state),
|
||||
.ipc_call => systemIpcCall(state),
|
||||
.ipc_reply_wait => systemIpcReplyWait(state),
|
||||
.ipc_send => systemIpcSend(state),
|
||||
@@ -320,36 +318,6 @@ fn systemCreateIpcEndpoint(state: *architecture.CpuState) void {
|
||||
architecture.setSystemCallResult(state, @intCast(h));
|
||||
}
|
||||
|
||||
/// ipc_register(service_id, handle): publish the caller's endpoint under a
|
||||
/// well-known id so other processes can find it.
|
||||
fn systemIpcRegister(state: *architecture.CpuState) void {
|
||||
// Under the big kernel lock: mutates the global service registry and endpoint
|
||||
// refcounts, which threads of the same (or another) process can race.
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
const id: u32 = @truncate(architecture.systemCallArg(state, 0));
|
||||
const endpoint = ipc.resolveHandle(scheduler.current(), architecture.systemCallArg(state, 1)) orelse return failErr(state, ipc.EBADF);
|
||||
architecture.setSystemCallResult(state, @bitCast(ipc.register(id, endpoint)));
|
||||
}
|
||||
|
||||
/// ipc_lookup(service_id) -> handle: find a published endpoint and install a
|
||||
/// handle to it in the caller.
|
||||
fn systemIpcLookup(state: *architecture.CpuState) void {
|
||||
// Under the big kernel lock: reads the global registry, takes an endpoint reference,
|
||||
// and installs a handle — all racy against concurrent threads (this is the path the
|
||||
// display's mouse-listener thread takes to reach the compositor endpoint).
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
const id: u32 = @truncate(architecture.systemCallArg(state, 0));
|
||||
const endpoint = ipc.lookup(id) orelse return failErr(state, ipc.ENOENT);
|
||||
const h = ipc.installHandle(scheduler.current(), endpoint);
|
||||
if (h < 0) {
|
||||
ipc.dropRef(endpoint);
|
||||
return failErr(state, ipc.ENOSPC);
|
||||
}
|
||||
architecture.setSystemCallResult(state, @intCast(h));
|
||||
}
|
||||
|
||||
/// ipc_call(handle, message_ptr, message_len, reply_ptr, reply_cap) -> reply_len.
|
||||
/// Blocks until the server replies; the trap frame lives on this task's kernel
|
||||
/// stack, so it survives the block and receives the result on resume.
|
||||
@@ -989,10 +957,17 @@ fn systemSpawn(state: *architecture.CpuState) void {
|
||||
if (len == 0 or len > scheduler.maximum_task_name or ptr >= user_half_end or ptr + len > user_half_end) return fail(state);
|
||||
if (arguments_len > maximum_argument_bytes) return fail(state);
|
||||
if (arguments_len != 0 and (arguments_ptr >= user_half_end or arguments_ptr + arguments_len > user_half_end)) return fail(state);
|
||||
// The exit endpoint is a notification binding like signal_bind's and
|
||||
// timer_bind's, so it obeys the same rule: the caller's own mailbox, never a
|
||||
// stranger's. Otherwise any process could have the kernel post child-exit
|
||||
// badges into PID 1 by spawning throwaway children against init's endpoint.
|
||||
const exit_endpoint: ?*ipc.Endpoint = if (exit_handle == abi.no_cap)
|
||||
null
|
||||
else
|
||||
ipc.resolveHandle(t, exit_handle) orelse return failErr(state, ipc.EBADF);
|
||||
else block: {
|
||||
const endpoint = ipc.resolveHandle(t, exit_handle) orelse return failErr(state, ipc.EBADF);
|
||||
if (!ipc.ownedBy(endpoint, t)) return failErr(state, ipc.EPERM);
|
||||
break :block endpoint;
|
||||
};
|
||||
const image = ramdisk_image orelse return fail(state);
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse return fail(state);
|
||||
|
||||
@@ -1040,11 +1015,15 @@ fn systemThreadSpawn(state: *architecture.CpuState) void {
|
||||
if (t.address_space == 0) return fail(state); // kernel tasks own no address space to share
|
||||
if (entry == 0 or entry >= user_half_end) return fail(state);
|
||||
if (stack_top == 0 or stack_top > user_half_end) return fail(state);
|
||||
// The endpoint the thread notifies on exit (how join waits), or none.
|
||||
// The endpoint the thread notifies on exit (how join waits), or none — the
|
||||
// caller's own, like every other notification binding.
|
||||
const exit_endpoint: ?*ipc.Endpoint = if (exit_handle == abi.no_cap)
|
||||
null
|
||||
else
|
||||
ipc.resolveHandle(t, exit_handle) orelse return failErr(state, ipc.EBADF);
|
||||
else block: {
|
||||
const endpoint = ipc.resolveHandle(t, exit_handle) orelse return failErr(state, ipc.EBADF);
|
||||
if (!ipc.ownedBy(endpoint, t)) return failErr(state, ipc.EPERM);
|
||||
break :block endpoint;
|
||||
};
|
||||
const tid = spawnThreadSupervised(t.address_space, entry, stack_top, arg, t.priority, t.id, exit_endpoint, t.leader);
|
||||
if (tid == -ipc.ESRCH) return failErr(state, ipc.ESRCH); // dying group admits no member
|
||||
if (tid < 0) return fail(state);
|
||||
@@ -1538,13 +1517,20 @@ const exit_subscriber_capacity = 8;
|
||||
const ExitSubscriber = struct { endpoint: *ipc.Endpoint, owner: u32 };
|
||||
var exit_subscribers: [exit_subscriber_capacity]?ExitSubscriber = .{null} ** exit_subscriber_capacity;
|
||||
|
||||
/// process_subscribe(endpoint): subscribe the caller's endpoint to published exit
|
||||
/// events. Ungated, like process_enumerate — what is running (and dying) is not a
|
||||
/// secret between cooperating processes. -ENOSPC when the table is full.
|
||||
/// process_subscribe(endpoint): subscribe the **caller's own** endpoint to
|
||||
/// published exit events. *Which* deaths one may hear of is ungated, like
|
||||
/// process_enumerate — what is running (and dying) is not a secret between
|
||||
/// cooperating processes. *Whose mailbox* they land in is not: the endpoint must
|
||||
/// be the caller's (`ipc.ownedBy`), or any process could aim the firehose at a
|
||||
/// stranger — filling PID 1's mailbox with exit notices it reads as its own
|
||||
/// children's, and spending the eight-slot table so the services that need
|
||||
/// deaths (the VFS's handle sweep) cannot subscribe at all. -EPERM otherwise,
|
||||
/// -ENOSPC when the table is full.
|
||||
fn systemProcessSubscribe(state: *architecture.CpuState) void {
|
||||
const t = scheduler.current();
|
||||
if (t.address_space == 0) return fail(state);
|
||||
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
|
||||
if (!ipc.ownedBy(endpoint, t)) return failErr(state, ipc.EPERM);
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
for (&exit_subscribers) |*slot| {
|
||||
@@ -1561,10 +1547,19 @@ fn systemProcessSubscribe(state: *architecture.CpuState) void {
|
||||
/// IRQ-as-IPC pattern a fourth time (docs/process-lifecycle.md). Replacing a
|
||||
/// binding drops the old reference; signals that pended while unbound are
|
||||
/// delivered immediately on bind, coalesced into one notification.
|
||||
///
|
||||
/// The endpoint must be the caller's own (`ipc.ownedBy`), or `signal_bind`
|
||||
/// becomes a signal *forgery* primitive: `process_signal` is deliberately loose
|
||||
/// about the target (a task may always signal itself) because the delivery point
|
||||
/// was assumed to be the target's own mailbox. Aim it elsewhere and a stranger
|
||||
/// signalling itself makes the kernel stamp a genuine `terminate` badge into
|
||||
/// somebody else's queue — which is a shutdown request PID 1 has no way to
|
||||
/// disbelieve.
|
||||
fn systemSignalBind(state: *architecture.CpuState) void {
|
||||
const t = scheduler.current();
|
||||
if (t.address_space == 0) return fail(state);
|
||||
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
|
||||
if (!ipc.ownedBy(endpoint, t)) return failErr(state, ipc.EPERM);
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
if (t.signal_endpoint) |raw| ipc.dropRef(@ptrCast(@alignCast(raw)));
|
||||
@@ -1633,11 +1628,19 @@ fn timerSweepLocked() void {
|
||||
}
|
||||
}
|
||||
|
||||
/// timer_bind(endpoint, ms): arm a one-shot timer. -ENOSPC when the table is full.
|
||||
/// timer_bind(endpoint, ms): arm a one-shot timer on an endpoint of the caller's
|
||||
/// own (`ipc.ownedBy`; -EPERM otherwise). A timer landing carries no identity —
|
||||
/// that is the whole reason a service may keep exactly one in flight — so a
|
||||
/// timer armed on someone else's endpoint is indistinguishable from one they
|
||||
/// armed themselves, and a loop that re-arms on every landing (init's heartbeat)
|
||||
/// multiplies: N forged timers leave N+1 self-perpetuating beats. The
|
||||
/// sixteen-slot table is a shared resource on top of that. -ENOSPC when it is
|
||||
/// full.
|
||||
fn systemTimerBind(state: *architecture.CpuState) void {
|
||||
const t = scheduler.current();
|
||||
if (t.address_space == 0) return fail(state);
|
||||
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
|
||||
if (!ipc.ownedBy(endpoint, t)) return failErr(state, ipc.EPERM);
|
||||
const ms = architecture.systemCallArg(state, 1);
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
@@ -1677,12 +1680,17 @@ fn ownedGsi(t: *scheduler.Task, device_id: u64, resource_index: u64) ?u32 {
|
||||
/// irq_bind(device_id, resource_index, endpoint) -> 0/-1: deliver that device's IRQ to the
|
||||
/// endpoint as an asynchronous IPC notification. The driver then blocks in
|
||||
/// IPC_ReplyWait and is woken by the ISR; see system/kernel/irq.zig for the cycle.
|
||||
/// Two gates, both necessary: the device must be *claimed* by the caller
|
||||
/// (`ownedGsi`), and the endpoint must be the caller's own (`ipc.ownedBy`) — a
|
||||
/// claim entitles a driver to its own interrupts, not to post them into a
|
||||
/// stranger's mailbox.
|
||||
fn systemIrqBind(state: *architecture.CpuState) void {
|
||||
const t = scheduler.current();
|
||||
if (t.address_space == 0) return fail(state);
|
||||
const gsi = ownedGsi(t, architecture.systemCallArg(state, 0), architecture.systemCallArg(state, 1)) orelse
|
||||
return fail(state);
|
||||
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 2)) orelse return fail(state);
|
||||
if (!ipc.ownedBy(endpoint, t)) return failErr(state, ipc.EPERM);
|
||||
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
@@ -1703,6 +1711,7 @@ fn systemMsiBind(state: *architecture.CpuState) void {
|
||||
const owner = devices_broker.ownerOf(device_id) orelse return fail(state);
|
||||
if (owner != t.id) return fail(state); // not claimed by this process
|
||||
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 1)) orelse return failErr(state, ipc.EBADF);
|
||||
if (!ipc.ownedBy(endpoint, t)) return failErr(state, ipc.EPERM); // interrupts land in your own mailbox
|
||||
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
|
||||
Reference in New Issue
Block a user