init: /protocol replaces the ServiceId registry
A protocol is reached by name now, not by a compile-time integer. Init is PID 1 and already knows which binary it started, so init serves /protocol as a vfs backend: bind claims a contract with the provider's endpoint attached, open answers with that endpoint as the reply's capability, and readdir lists what is bound with the task and binary behind it. The kernel reserves the prefix — nothing may mount over it, under it, or unmount it — and ServiceId, ipc_register and ipc_lookup are gone, their syscall numbers left vacant. A bind is authorized by who the caller *is*: the kernel-stamped binary together with the supervising task's identity, matched against /system/configuration/protocol.csv. Identity, not spelling — spawn is ungated, so an attacker can run any bundled binary, and a name-only rule would have let it launder grants through an init of its own making. A name a live process holds is refused to everyone else; a dead one's is released. Three review rounds against a hostile ring-3 process found what 108 green tests could not, because the suite contains no attacker. Publishing init's supervision endpoint as the registry put PID 1's mailbox in every process's hands, where two forged bytes reached the shutdown path: privileged traffic is now believed only from the task that holds the contract it speaks for. A capability arriving on a request outlived every path that ignored it, one handle per call until the table was full — in init, and in the harness ten services share — so the arriving capability is owned by the turn and released unless a handler says otherwise. And the kernel let anyone holding an endpoint handle aim signals, timers, exit notices and interrupts at it: binding now requires having created it. Suite 108/108. The new protocol-registry case asserts eleven properties, each one an attack that must fail.
This commit is contained in:
+674
-54
@@ -16,6 +16,28 @@
|
||||
//! power-button event runs the stop sequence over its children in reverse order
|
||||
//! before asking the power service to enter S5 — lifecycle (M17) and events (M21)
|
||||
//! composing into a clean poweroff.
|
||||
//!
|
||||
//! P2: init is also the **registrar** — it serves `/protocol`, the namespace where
|
||||
//! a program finds everything it talks to
|
||||
//! (docs/os-development/protocol-namespace.md). It is the natural home: it already
|
||||
//! spawns the services and already holds the supervision link to each, so it is the
|
||||
//! process that *knows* which binary is which. Registry traffic rides the same
|
||||
//! endpoint as supervision, because one thread can only wait in one place — the
|
||||
//! loop below answers vfs `open`/`readdir`/`bind` alongside signals, timers, power
|
||||
//! events, and children's deaths.
|
||||
//!
|
||||
//! Which sets the security posture of this file. `fs_resolve` installs a mounted
|
||||
//! backend's endpoint capability in *any* caller's handle table, so sharing the
|
||||
//! mailbox means **every ring-3 process can send into PID 1**. Two rules follow,
|
||||
//! and both are structural here rather than remembered per branch:
|
||||
//!
|
||||
//! - **Privileged action requires an attested sender.** What arrives is a
|
||||
//! stranger's bytes; the only identity on it is the task id the kernel stamps.
|
||||
//! Content never authorizes (`onPowerEvent`), and neither does a name — the
|
||||
//! registrar attests a caller's supervision by task id (`supervisorSatisfies`).
|
||||
//! - **A capability that arrives is closed unless it is claimed** (`Arrival`),
|
||||
//! because PID 1's thirty-two handle slots are a resource an unauthenticated
|
||||
//! caller would otherwise be able to spend.
|
||||
|
||||
const std = @import("std");
|
||||
const ipc = @import("ipc");
|
||||
@@ -24,6 +46,8 @@ const time = @import("time");
|
||||
const memory = @import("memory");
|
||||
const logging = @import("logging");
|
||||
const power_protocol = @import("power-protocol");
|
||||
const vfs_protocol = @import("vfs-protocol");
|
||||
const envelope = @import("envelope");
|
||||
const build_options = @import("build_options");
|
||||
const fs = @import("file-system");
|
||||
const csv = @import("csv");
|
||||
@@ -73,17 +97,10 @@ var supervision_endpoint: ipc.Handle = 0;
|
||||
/// uses for /system/configuration/devices.csv. A missing file means no services (the no-ramdisk
|
||||
/// isolation test): loud, but not fatal.
|
||||
fn loadServices() void {
|
||||
var file = fs.open("/system/configuration/init.csv", .{}) orelse {
|
||||
const used = readConfiguration("/system/configuration/init.csv", &init_csv) orelse {
|
||||
_ = logging.write("/system/services/init: /system/configuration/init.csv missing — no services started\n");
|
||||
return;
|
||||
};
|
||||
defer file.close();
|
||||
var used: usize = 0;
|
||||
while (used < init_csv.len) {
|
||||
const n = file.read(init_csv[used..]) orelse break;
|
||||
if (n == 0) break;
|
||||
used += n;
|
||||
}
|
||||
var lines = std.mem.splitScalar(u8, init_csv[0..used], '\n');
|
||||
while (lines.next()) |line| {
|
||||
const body = csv.stripComment(line);
|
||||
@@ -107,11 +124,500 @@ fn loadServices() void {
|
||||
}
|
||||
}
|
||||
|
||||
/// Read a whole configuration file into `into`, returning the byte count. Both of
|
||||
/// init's manifests live in the initial ramdisk the kernel serves directly, so this
|
||||
/// works before any filesystem service exists. A file that fills the buffer exactly
|
||||
/// is reported: a manifest silently losing its last rows is a policy change nobody
|
||||
/// asked for, and the symptom (one service refused a name) points nowhere near it.
|
||||
fn readConfiguration(path: []const u8, into: []u8) ?usize {
|
||||
var file = fs.open(path, .{}) orelse return null;
|
||||
defer file.close();
|
||||
var used: usize = 0;
|
||||
while (used < into.len) {
|
||||
const n = file.read(into[used..]) orelse break;
|
||||
if (n == 0) break;
|
||||
used += n;
|
||||
}
|
||||
if (used == into.len) std.log.info("{s} filled the read buffer — rows past {d} bytes are lost", .{ path, used });
|
||||
return used;
|
||||
}
|
||||
|
||||
/// Give up restarting a service after this many crashes — a crash-loop cap, so a service
|
||||
/// that faults immediately on every spawn doesn't respawn forever.
|
||||
const maximum_restarts = 3;
|
||||
|
||||
pub fn main() void {
|
||||
// --- the registry: /protocol ------------------------------------------------
|
||||
|
||||
/// Longest contract name the namespace admits (`display`, `test/shared-memory`)
|
||||
/// and the most that may be bound at once. Both static, like everything else
|
||||
/// init holds.
|
||||
const maximum_name = 64;
|
||||
const maximum_bindings = 16;
|
||||
const maximum_grants = 48;
|
||||
|
||||
/// One bound contract: the name, the provider's endpoint (a capability init
|
||||
/// holds and hands to whoever opens the name), and the provenance a diagnostic
|
||||
/// listing answers "who serves this?" with.
|
||||
const Binding = struct {
|
||||
used: bool = false,
|
||||
name: [maximum_name]u8 = undefined,
|
||||
name_len: usize = 0,
|
||||
endpoint: ipc.Handle = 0,
|
||||
task: u32 = 0,
|
||||
binary: [64]u8 = undefined,
|
||||
binary_len: usize = 0,
|
||||
|
||||
fn nameSlice(self: *const Binding) []const u8 {
|
||||
return self.name[0..self.name_len];
|
||||
}
|
||||
fn binarySlice(self: *const Binding) []const u8 {
|
||||
return self.binary[0..self.binary_len];
|
||||
}
|
||||
};
|
||||
|
||||
var bindings: [maximum_bindings]Binding = .{Binding{}} ** maximum_bindings;
|
||||
|
||||
/// What a grant row permits: claiming a name, or reaching one. `open` rows are
|
||||
/// parsed and held but not yet enforced — every open resolves in P2, and P3 is
|
||||
/// the milestone that turns these into refusals (docs/security-track-plan.md).
|
||||
const Permission = enum { bind, open };
|
||||
|
||||
/// One row of `/system/configuration/protocol.csv`. Every field may end in `*`,
|
||||
/// which matches any tail — the subtree scoping the design doc describes, and
|
||||
/// what lets one row grant the whole `/test/` family its `test/...` names.
|
||||
const Grant = struct {
|
||||
binary: []const u8 = "",
|
||||
supervisor: []const u8 = "",
|
||||
permission: Permission = .bind,
|
||||
name: []const u8 = "",
|
||||
};
|
||||
|
||||
/// Roomier than init.csv's: this manifest carries a row per provider per spawn
|
||||
/// path, its own format documentation, and grows again with the open grants.
|
||||
var protocol_csv: [8192]u8 = undefined;
|
||||
var grants: [maximum_grants]Grant = .{Grant{}} ** maximum_grants;
|
||||
var grant_count: usize = 0;
|
||||
|
||||
/// Parse `/system/configuration/protocol.csv` — the grant manifest. Separate from
|
||||
/// init.csv because every field there after the path is argv, and overloading that
|
||||
/// would be ambiguous; separate *files* also means a grant exists for binaries init
|
||||
/// never spawns (the drivers, which the device manager owns).
|
||||
fn loadGrants() void {
|
||||
const used = readConfiguration("/system/configuration/protocol.csv", &protocol_csv) orelse {
|
||||
_ = logging.write("/system/services/init: /system/configuration/protocol.csv missing — no protocol may be bound\n");
|
||||
return;
|
||||
};
|
||||
var lines = std.mem.splitScalar(u8, protocol_csv[0..used], '\n');
|
||||
while (lines.next()) |line| {
|
||||
const body = csv.stripComment(line);
|
||||
if (body.len == 0) continue;
|
||||
if (grant_count >= grants.len) {
|
||||
_ = logging.write("/system/services/init: /system/configuration/protocol.csv has more rows than the table holds\n");
|
||||
break;
|
||||
}
|
||||
var it = csv.fields(body);
|
||||
const binary = it.next() orelse continue;
|
||||
const supervisor = it.next() orelse continue;
|
||||
const permission = it.next() orelse continue;
|
||||
const name = it.next() orelse continue;
|
||||
if (binary.len == 0 or supervisor.len == 0 or name.len == 0) continue;
|
||||
const kind: Permission = if (std.mem.eql(u8, permission, "bind"))
|
||||
.bind
|
||||
else if (std.mem.eql(u8, permission, "open"))
|
||||
.open
|
||||
else
|
||||
continue; // an unreadable row grants nothing rather than something wrong
|
||||
grants[grant_count] = .{ .binary = binary, .supervisor = supervisor, .permission = kind, .name = name };
|
||||
grant_count += 1;
|
||||
}
|
||||
}
|
||||
|
||||
/// Match a manifest field against a value: exact, or a trailing `*` matching any
|
||||
/// tail. The wildcard is how a subtree is granted whole (`/test/*` for every test
|
||||
/// fixture, `test/*` for every name they may claim).
|
||||
fn matches(pattern: []const u8, value: []const u8) bool {
|
||||
if (pattern.len != 0 and pattern[pattern.len - 1] == '*') {
|
||||
const prefix = pattern[0 .. pattern.len - 1];
|
||||
return value.len >= prefix.len and std.mem.eql(u8, value[0..prefix.len], prefix);
|
||||
}
|
||||
return std.mem.eql(u8, pattern, value);
|
||||
}
|
||||
|
||||
/// A snapshot of the kernel's process records — the only identity in the system
|
||||
/// that cannot be forged, because the kernel stamps it at spawn. Refreshed per
|
||||
/// authorization; binds are rare, so the copy costs nothing that matters.
|
||||
var process_table: [64]process.ProcessDescriptor = undefined;
|
||||
var process_count: usize = 0;
|
||||
var process_truncated = false;
|
||||
|
||||
fn refreshProcessTable() void {
|
||||
const total = process.processes(&process_table);
|
||||
process_count = @min(total, process_table.len);
|
||||
process_truncated = total > process_table.len;
|
||||
}
|
||||
|
||||
/// Whether task `id` is still alive, as the last snapshot saw it. A snapshot that
|
||||
/// did not fit answers "alive" for anything it did not list: refusing a bind is
|
||||
/// recoverable, stealing a live provider's name is not.
|
||||
fn taskAlive(id: u32) bool {
|
||||
return descriptorOf(id) != null or process_truncated;
|
||||
}
|
||||
|
||||
fn descriptorOf(id: u32) ?*const process.ProcessDescriptor {
|
||||
for (process_table[0..process_count]) |*descriptor| {
|
||||
if (descriptor.id == id) return descriptor;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
fn nameOf(descriptor: *const process.ProcessDescriptor) []const u8 {
|
||||
const length = @min(@as(usize, descriptor.name_length), descriptor.name.len);
|
||||
return descriptor.name[0..length];
|
||||
}
|
||||
|
||||
/// The name a kernel task answers to in a grant row. Kernel tasks carry no
|
||||
/// binary, so the manifest spells the harness's parentage `kernel`.
|
||||
const kernel_supervisor = "kernel";
|
||||
|
||||
/// init's own task id, read once at startup. Ids are monotonic and never reused
|
||||
/// (system/kernel/process.zig), so an id comparison is an *identity* test where a
|
||||
/// name comparison is only a resemblance test — the whole basis of the
|
||||
/// attestation below.
|
||||
var own_task: u32 = 0;
|
||||
|
||||
/// Whether `id` is a process THIS init spawned: a lookup in its own child table,
|
||||
/// which is the one record of "I started that one" nobody else can write.
|
||||
fn spawnedByUs(id: u32) bool {
|
||||
if (id == 0) return false;
|
||||
for (child_ids[0..service_count]) |child| {
|
||||
if (child == id) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/// Who is asking, attested by the kernel: the caller's binary, and the **task**
|
||||
/// that spawned it — an id, not a name.
|
||||
///
|
||||
/// A name alone is not identity: `spawn` is ungated, so a hostile process can
|
||||
/// start a granted binary itself and would inherit its grants. Neither is the
|
||||
/// supervisor's *name* enough, and this is the trap the first cut fell into —
|
||||
/// init and the device manager are ordinary bundled binaries, so an attacker
|
||||
/// spawns its own `/system/services/init` and lets that instance spawn
|
||||
/// `/system/services/input`. Both kernel-stamped names then match the grant row
|
||||
/// exactly, and walking to the root of the chain does not help either: the
|
||||
/// laundered chain still roots at the real PID 1. What refuses it is asking
|
||||
/// *which task* the supervisor is, and only accepting one init can vouch for.
|
||||
const Identity = struct {
|
||||
/// The caller's binary path, exactly as the kernel stamped it at spawn.
|
||||
binary: []const u8,
|
||||
/// The supervising task's id. 0 means the kernel spawned the caller, which
|
||||
/// no ring-3 process can arrange: every `system_spawn` stamps the caller as
|
||||
/// the child's supervisor (system/kernel/process.zig `systemSpawn`).
|
||||
supervisor_task: u32,
|
||||
/// The supervising task's kernel-stamped binary — the grant row's supervisor
|
||||
/// column is matched against this, and the refusal log prints it. `kernel`
|
||||
/// when there is no supervising task.
|
||||
supervisor_binary: []const u8,
|
||||
/// Whether init can vouch for how the supervising task came to exist: it is
|
||||
/// this init, a process this init spawned, or a process the KERNEL spawned.
|
||||
/// A supervisor init cannot vouch for satisfies no row, however well its
|
||||
/// name reads — that is the laundering deputy's refusal.
|
||||
supervisor_vouched: bool,
|
||||
};
|
||||
|
||||
/// The process a task belongs to. A thread resolves to its leader: threads share
|
||||
/// a binary (a thread's own record is named `thread`), and the supervision link
|
||||
/// that matters is the process's.
|
||||
fn leaderOf(descriptor: *const process.ProcessDescriptor) *const process.ProcessDescriptor {
|
||||
if (descriptor.leader == descriptor.id) return descriptor;
|
||||
return descriptorOf(descriptor.leader) orelse descriptor;
|
||||
}
|
||||
|
||||
/// Resolve the badge on a request into an identity, one hop up the supervision
|
||||
/// chain in the kernel's records — one hop is enough because the hop is attested
|
||||
/// by id (see `supervisorSatisfies`), and every id in the chain init accepts is
|
||||
/// one init or the kernel created.
|
||||
fn identify(task: u32) ?Identity {
|
||||
const caller = descriptorOf(task) orelse return null;
|
||||
const leader = leaderOf(caller);
|
||||
if (leader.supervisor == 0) return .{
|
||||
.binary = nameOf(leader),
|
||||
.supervisor_task = 0,
|
||||
.supervisor_binary = kernel_supervisor,
|
||||
.supervisor_vouched = true, // the kernel is the root of trust, not a claimant
|
||||
};
|
||||
// The supervising *task* may be a worker thread of the supervising process;
|
||||
// its process is what the manifest names and what init recorded at spawn.
|
||||
const supervisor = leaderOf(descriptorOf(leader.supervisor) orelse return null); // unattestable: refuse
|
||||
return .{
|
||||
.binary = nameOf(leader),
|
||||
.supervisor_task = supervisor.id,
|
||||
.supervisor_binary = nameOf(supervisor),
|
||||
.supervisor_vouched = supervisor.id == own_task or
|
||||
spawnedByUs(supervisor.id) or
|
||||
supervisor.supervisor == 0,
|
||||
};
|
||||
}
|
||||
|
||||
/// Whether the caller's supervising task satisfies a grant row's supervisor
|
||||
/// column. The column names *the authorized supervising task*, matched by
|
||||
/// identity — the binary it must be, plus proof that this instance of that
|
||||
/// binary is the authorized one:
|
||||
///
|
||||
/// - `kernel` is satisfied only by a genuinely kernel-spawned caller
|
||||
/// (supervisor id 0). A ring-3 process cannot manufacture that: user
|
||||
/// `system_spawn` always stamps the caller (system/kernel/process.zig).
|
||||
/// - init's own binary is satisfied only when the supervising task IS this
|
||||
/// init (`own_task`).
|
||||
/// - any other binary — the device manager, a test fixture spawning another —
|
||||
/// is satisfied only when the supervising task is one init spawned itself
|
||||
/// (its own child table) or one the kernel spawned. Everything init and the
|
||||
/// kernel start is therefore reachable; a chain that passes through a
|
||||
/// process *neither* of them started is not.
|
||||
fn supervisorSatisfies(column: []const u8, identity: Identity) bool {
|
||||
if (std.mem.eql(u8, column, kernel_supervisor)) return identity.supervisor_task == 0;
|
||||
if (identity.supervisor_task == 0) return false; // a kernel task answers to no binary column
|
||||
if (!matches(column, identity.supervisor_binary)) return false;
|
||||
return identity.supervisor_vouched;
|
||||
}
|
||||
|
||||
/// Whether `identity` is granted `permission` on `name`.
|
||||
fn granted(identity: Identity, permission: Permission, name: []const u8) bool {
|
||||
for (grants[0..grant_count]) |grant| {
|
||||
if (grant.permission != permission) continue;
|
||||
if (!matches(grant.binary, identity.binary)) continue;
|
||||
if (!supervisorSatisfies(grant.supervisor, identity)) continue;
|
||||
if (!matches(grant.name, name)) continue;
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
fn findBinding(name: []const u8) ?*Binding {
|
||||
for (&bindings) |*binding| {
|
||||
if (binding.used and std.mem.eql(u8, binding.nameSlice(), name)) return binding;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Release a binding: the provider's endpoint capability goes back to the handle
|
||||
/// table, and the name is free for the next claimant. init's own cached power
|
||||
/// channel goes with it — a closed handle number is reused by the next capability
|
||||
/// that arrives, and a stale copy would quietly aim the shutdown call at a
|
||||
/// stranger. So does the authorized power *task*: nothing may speak for a
|
||||
/// contract nobody holds.
|
||||
fn releaseBinding(binding: *Binding) void {
|
||||
if (std.mem.eql(u8, binding.nameSlice(), power_contract)) {
|
||||
power_endpoint = null;
|
||||
power_task = null;
|
||||
power_pending = false;
|
||||
}
|
||||
_ = ipc.close(binding.endpoint);
|
||||
binding.* = .{};
|
||||
}
|
||||
|
||||
/// Drop every name a dead process held. Called when a supervised child dies (so
|
||||
/// the restarted instance can bind again) and whenever a bind finds the current
|
||||
/// owner gone — providers init does not supervise need the second path.
|
||||
fn unbindTask(task: u32) void {
|
||||
for (&bindings) |*binding| {
|
||||
if (binding.used and binding.task == task) {
|
||||
std.log.info("/protocol/{s} released ({s} is gone)", .{ binding.nameSlice(), binding.binarySlice() });
|
||||
releaseBinding(binding);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// A contract name as the namespace spells it: the mount-relative path a resolve
|
||||
/// hands us ("/display") and the name a bind sends ("display") are the same thing
|
||||
/// with and without a leading slash, so one normaliser serves both. Empty or
|
||||
/// longer than the namespace admits is not a name.
|
||||
fn contractName(raw: []const u8) ?[]const u8 {
|
||||
const name = if (raw.len != 0 and raw[0] == '/') raw[1..] else raw;
|
||||
if (name.len == 0 or name.len > maximum_name) return null;
|
||||
return name;
|
||||
}
|
||||
|
||||
/// The provider's endpoint that will ride the *next* reply, when the request was
|
||||
/// an `open` that found its contract.
|
||||
var pending_capability: ?ipc.Handle = null;
|
||||
|
||||
/// The ownership rule for a capability that arrives with a turn of the loop —
|
||||
/// **the turn owns it until a handler takes it, and closes whatever is left** —
|
||||
/// lives in `ipc.Arrival`, next to `replyWait`, because it is not PID 1's rule:
|
||||
/// the service harness every other service runs (library/kernel/service.zig) had
|
||||
/// the identical hole and now states the identical contract.
|
||||
const Arrival = ipc.Arrival;
|
||||
|
||||
/// The one contract init is itself a client of. It never resolves the name — it
|
||||
/// *is* the registry, so it reads its own table; the binding is what hands it the
|
||||
/// channel.
|
||||
const power_contract = "power";
|
||||
|
||||
/// Set when `power` is bound: init subscribes to it on the next turn of the loop,
|
||||
/// never inside the bind — the provider is blocked on our reply until then, so
|
||||
/// calling it here would deadlock the pair.
|
||||
var power_pending = false;
|
||||
var power_endpoint: ?ipc.Handle = null;
|
||||
|
||||
/// The one task authorized to deliver power events: whoever holds the `power`
|
||||
/// binding. Recorded at the bind and cleared with the binding, so a provider that
|
||||
/// dies and rebinds re-derives it with no further ceremony.
|
||||
///
|
||||
/// This is the *authentication* for the shutdown path. init's registry endpoint
|
||||
/// is its supervision endpoint, and `fs_resolve("/protocol")` installs a sendable
|
||||
/// handle to it in any caller's table — so after P2 every ring-3 process can post
|
||||
/// into PID 1's mailbox. A power event may therefore never be believed on the
|
||||
/// strength of its payload; it is believed because the kernel stamped the
|
||||
/// sender's task id on it and that id is the provider's.
|
||||
var power_task: ?u32 = null;
|
||||
|
||||
/// Whether the heartbeat's re-arming timer is running. A timer landing carries no
|
||||
/// identity, so the loop cannot tell one timer from another — which means exactly
|
||||
/// one may ever be in flight, or every landing re-arms and the beat doubles. (It
|
||||
/// did: two beats a second is enough extra chatter to cut a driver's echoed line
|
||||
/// in half on the shared serial stream.) So the deferred power subscribe borrows
|
||||
/// the heartbeat's tick when there is one, and arms its own only when there is not.
|
||||
var heartbeat_running = false;
|
||||
|
||||
/// Answer one registry request. Writes a vfs-protocol reply into `reply` and
|
||||
/// returns its length; a capability the reply must carry lands in
|
||||
/// `pending_capability`. `arrived` is the capability the *request* carried, owned
|
||||
/// by the turn — nothing here has to close it, only `bind` has to claim it.
|
||||
fn serveRegistry(request_bytes: []const u8, reply: []u8, sender: u32, arrived: *Arrival) usize {
|
||||
if (request_bytes.len < vfs_protocol.request_size)
|
||||
return answer(reply, -envelope.EPROTO, 0, 0);
|
||||
// The header is read field by field rather than reinterpreted whole: the
|
||||
// operation is an enum on the wire and the bytes come from anyone at all, so
|
||||
// a value outside it must be a refusal, never a decoded enum.
|
||||
const operation = std.mem.readInt(u32, request_bytes[0..4], .little);
|
||||
const cursor = std.mem.readInt(u64, request_bytes[16..24], .little);
|
||||
const declared = std.mem.readInt(u32, request_bytes[24..28], .little);
|
||||
const payload_len = @min(@as(usize, declared), request_bytes.len - vfs_protocol.request_size);
|
||||
const payload = request_bytes[vfs_protocol.request_size..][0..payload_len];
|
||||
|
||||
if (operation == @intFromEnum(vfs_protocol.Operation.bind))
|
||||
return answer(reply, onBind(sender, payload, arrived), 0, 0);
|
||||
// Only `bind` claims a capability; one attached to anything else is closed by
|
||||
// the turn's `defer` in the loop, along with the ones sent to a request that
|
||||
// was too short to name a verb at all.
|
||||
if (operation == @intFromEnum(vfs_protocol.Operation.open)) return onOpen(reply, payload);
|
||||
if (operation == @intFromEnum(vfs_protocol.Operation.readdir)) return onReaddir(reply, cursor);
|
||||
// Everything else a filesystem answers is meaningless here: `/protocol` holds
|
||||
// contracts, not bytes.
|
||||
return answer(reply, -envelope.ENOSYS, 0, 0);
|
||||
}
|
||||
|
||||
/// Lay down a vfs reply header (and say how many payload bytes follow it).
|
||||
fn answer(reply: []u8, status: i32, node: u64, payload_len: usize) usize {
|
||||
const header = vfs_protocol.Reply{ .status = status, .node = node, .len = @intCast(payload_len) };
|
||||
@memcpy(reply[0..vfs_protocol.reply_size], std.mem.asBytes(&header));
|
||||
return vfs_protocol.reply_size + payload_len;
|
||||
}
|
||||
|
||||
/// `bind(name, capability = the provider's endpoint)`. The capability is the
|
||||
/// point of the call, so a bind without one is malformed. Every refusal below
|
||||
/// simply returns: the endpoint stays the turn's, and the turn closes it — which
|
||||
/// is why there is not one `ipc.close` on the way out of any of the six of them.
|
||||
/// The success path is the only one that says anything about ownership, because
|
||||
/// it is the only one that keeps the capability.
|
||||
fn onBind(sender: u32, raw_name: []const u8, arrived: *Arrival) i32 {
|
||||
if (arrived.peek() == null) return -envelope.EPROTO;
|
||||
const name = contractName(raw_name) orelse return -envelope.ENOENT;
|
||||
refreshProcessTable();
|
||||
const identity = identify(sender) orelse return -envelope.EPERM;
|
||||
if (!granted(identity, .bind, name)) {
|
||||
std.log.info("refused bind of /protocol/{s} by {s} (pid {d}, supervisor {s} pid {d})", .{
|
||||
name,
|
||||
identity.binary,
|
||||
sender,
|
||||
identity.supervisor_binary,
|
||||
identity.supervisor_task,
|
||||
});
|
||||
return -envelope.EPERM;
|
||||
}
|
||||
if (findBinding(name)) |existing| {
|
||||
// Collision is an error — never last-writer-wins — unless the incumbent
|
||||
// is dead, which is how a restarted provider retakes its own name.
|
||||
if (taskAlive(existing.task)) {
|
||||
std.log.info("refused bind of /protocol/{s}: held by {s} (pid {d})", .{ name, existing.binarySlice(), existing.task });
|
||||
return -envelope.EBUSY;
|
||||
}
|
||||
releaseBinding(existing);
|
||||
}
|
||||
const slot = for (&bindings) |*binding| {
|
||||
if (!binding.used) break binding;
|
||||
} else return -envelope.ENOSPC;
|
||||
|
||||
// Claimed: the binding owns the endpoint from here, and `releaseBinding` is
|
||||
// what closes it.
|
||||
const endpoint = arrived.take().?;
|
||||
slot.* = .{ .used = true, .endpoint = endpoint, .task = sender };
|
||||
@memcpy(slot.name[0..name.len], name);
|
||||
slot.name_len = name.len;
|
||||
const binary_len = @min(identity.binary.len, slot.binary.len);
|
||||
@memcpy(slot.binary[0..binary_len], identity.binary[0..binary_len]);
|
||||
slot.binary_len = binary_len;
|
||||
|
||||
// Provenance, at the moment it becomes true: name -> pid -> binary path.
|
||||
std.log.info("/protocol/{s} -> pid {d} {s}", .{ name, sender, slot.binarySlice() });
|
||||
if (std.mem.eql(u8, name, power_contract)) {
|
||||
power_endpoint = endpoint;
|
||||
// The bind is also the authentication: whoever holds `power` is the one
|
||||
// task whose power events init will act on (see `onPowerEvent`).
|
||||
power_task = sender;
|
||||
power_pending = true;
|
||||
// Wake ourselves once the reply has gone out; the subscribe call cannot
|
||||
// happen while the power service is still blocked on it. The heartbeat's
|
||||
// tick is that wake when it is running — see `heartbeat_running`.
|
||||
if (!heartbeat_running) _ = time.timerOnce(supervision_endpoint, 1);
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// `open(name)` -> the provider's endpoint, delivered as the reply's capability.
|
||||
/// A name nothing has bound is `-ENOENT`; in P3 an ungranted one becomes the same
|
||||
/// answer, because absence and refusal are deliberately indistinguishable.
|
||||
fn onOpen(reply: []u8, raw_name: []const u8) usize {
|
||||
const name = contractName(raw_name) orelse return answer(reply, -envelope.ENOENT, 0, 0);
|
||||
const binding = findBinding(name) orelse return answer(reply, -envelope.ENOENT, 0, 0);
|
||||
pending_capability = binding.endpoint;
|
||||
return answer(reply, 0, 0, 0);
|
||||
}
|
||||
|
||||
/// `readdir(cursor)` — the namespace, browsable. One entry per turn, as the vfs
|
||||
/// protocol lists any directory: kind `protocol`, the contract's name, and the
|
||||
/// provider's task id in `size`, so a plain listing answers "who serves this?".
|
||||
fn onReaddir(reply: []u8, cursor: u64) usize {
|
||||
var index: u64 = 0;
|
||||
for (&bindings) |*binding| {
|
||||
if (!binding.used) continue;
|
||||
if (index != cursor) {
|
||||
index += 1;
|
||||
continue;
|
||||
}
|
||||
const name = binding.nameSlice();
|
||||
const entry = vfs_protocol.DirectoryEntry{
|
||||
.kind = @intFromEnum(vfs_protocol.NodeKind.protocol),
|
||||
.name_len = @intCast(name.len),
|
||||
.size = binding.task,
|
||||
};
|
||||
const total = vfs_protocol.directory_entry_size + name.len;
|
||||
if (vfs_protocol.reply_size + total > reply.len) return answer(reply, -envelope.EPROTO, 0, 0);
|
||||
@memcpy(reply[vfs_protocol.reply_size..][0..vfs_protocol.directory_entry_size], std.mem.asBytes(&entry));
|
||||
@memcpy(reply[vfs_protocol.reply_size + vfs_protocol.directory_entry_size ..][0..name.len], name);
|
||||
return answer(reply, 0, 0, total);
|
||||
}
|
||||
return answer(reply, 0, 0, 0); // end of directory
|
||||
}
|
||||
|
||||
pub fn main(startup: process.Init) void {
|
||||
// `registry` is the scenario mode: serve /protocol and nothing else. The
|
||||
// kernel test harness spawns its own providers directly, so it wants the
|
||||
// naming layer up without init's whole service list underneath it
|
||||
// (docs/security-track-plan.md, decision 9).
|
||||
const registry_only = if (startup.arguments.get(1)) |role| std.mem.eql(u8, role, "registry") else false;
|
||||
|
||||
// Prove the heap end to end: allocate through the runtime allocator (which
|
||||
// mmaps pages from the kernel and carves them with the free list), write into
|
||||
// that heap buffer (exercising the widened debug_write bounds check), and
|
||||
@@ -128,65 +634,175 @@ pub fn main() void {
|
||||
|
||||
// One endpoint carries everything init waits on: children's exit
|
||||
// notifications (they are spawned supervised against it), init's own
|
||||
// signals, and power events it subscribes to. All arrive in the loop below.
|
||||
// signals, power events it subscribes to, and — since PID 1 is the registrar
|
||||
// — every /protocol request. One thread can wait in one place, so they share
|
||||
// a mailbox and the loop below tells them apart.
|
||||
supervision_endpoint = ipc.createIpcEndpoint() orelse {
|
||||
_ = logging.write("/system/services/init: no endpoint\n");
|
||||
return;
|
||||
};
|
||||
_ = process.bindSignals(supervision_endpoint);
|
||||
|
||||
// Our own id, before anything can ask us a question. It is half of the
|
||||
// registrar's authority: a grant row naming init as the supervisor is
|
||||
// satisfied by *this* task and no other instance of this binary
|
||||
// (`supervisorSatisfies`).
|
||||
own_task = process.taskId();
|
||||
|
||||
// The namespace goes up BEFORE anything is spawned, so a service's first
|
||||
// bind lands rather than retrying. The kernel reserves the prefix: this
|
||||
// mount is the only one it will ever hold.
|
||||
loadGrants();
|
||||
if (!fs.mount("/protocol", supervision_endpoint)) {
|
||||
_ = logging.write("/system/services/init: /protocol already mounted — not the registrar\n");
|
||||
}
|
||||
|
||||
// Load the service list, then bring each up supervised so init can stop them
|
||||
// cleanly. Best-effort and silent: each service announces its own readiness,
|
||||
// and with no /system/configuration/init.csv (an isolation test) the loop starts nothing.
|
||||
loadServices();
|
||||
for (services[0..service_count], 0..) |*service, i| {
|
||||
if (process.spawnSupervised(service.path, service.arguments(), supervision_endpoint)) |id| child_ids[i] = id;
|
||||
if (!registry_only) {
|
||||
loadServices();
|
||||
for (services[0..service_count], 0..) |*service, i| {
|
||||
if (process.spawnSupervised(service.path, service.arguments(), supervision_endpoint)) |id| child_ids[i] = id;
|
||||
}
|
||||
}
|
||||
|
||||
// Subscribe to power events (retry: the power service registers well after
|
||||
// init starts). Best-effort — without it, a `terminate` signal still
|
||||
// triggers the same shutdown path.
|
||||
subscribePower();
|
||||
|
||||
// A re-arming timer drives the liveness heartbeat — proof PID 1 is alive (the
|
||||
// init test's marker) and a -Dserial diagnostic. It is a serial/test-build-only
|
||||
// concern: a flashable (serial-off) image runs a purely event-driven PID 1 that
|
||||
// wakes only for real work (signals, power events, children's exits), never for a
|
||||
// periodic beat. `build_options.serial` is comptime, so the heartbeat — its timer
|
||||
// and the handler below — folds away entirely when serial is off.
|
||||
if (build_options.serial) _ = time.timerOnce(supervision_endpoint, 1000);
|
||||
heartbeat_running = build_options.serial and !registry_only;
|
||||
if (heartbeat_running) _ = time.timerOnce(supervision_endpoint, 1000);
|
||||
|
||||
var receive: [power_protocol.message_maximum]u8 = undefined;
|
||||
var receive: [vfs_protocol.message_maximum]u8 = undefined;
|
||||
var reply_buffer: [vfs_protocol.message_maximum]u8 = undefined;
|
||||
var reply_len: usize = 0;
|
||||
var reply_capability: ?ipc.Handle = null;
|
||||
while (true) {
|
||||
const got = ipc.replyWait(supervision_endpoint, &.{}, &receive, null);
|
||||
if (process.signalsFrom(got.badge)) |signals| {
|
||||
if (signals.has(.terminate)) shutDown();
|
||||
continue;
|
||||
const got = ipc.replyWait(supervision_endpoint, reply_buffer[0..reply_len], &receive, reply_capability);
|
||||
reply_len = 0; // nothing owed until this turn's request says otherwise
|
||||
reply_capability = null;
|
||||
|
||||
// Whatever capability came with this turn is the turn's, and the turn
|
||||
// closes it unless a handler claims it. Structural on purpose — see
|
||||
// `Arrival`; it is what keeps a zero-length call from spending a handle
|
||||
// slot of PID 1's per call.
|
||||
var arrived: Arrival = .{ .handle = got.cap };
|
||||
defer arrived.release();
|
||||
|
||||
if (got.isNotification()) {
|
||||
// Every badge on this branch is stamped by the KERNEL, and a stranger
|
||||
// cannot stamp one: `ipc_send` — the only way a ring-3 process puts
|
||||
// something in this mailbox with no reply owed — sets exactly
|
||||
// `notify_badge_bit | notify_message_bit` and fills the low bits with
|
||||
// the sender's own task id (system/kernel/ipc-synchronous.zig,
|
||||
// `sendLocked`).
|
||||
//
|
||||
// That is a statement about `ipc_send`, and on its own it proved far
|
||||
// too little: an attacker does not use `ipc_send` to forge a signal,
|
||||
// it asks the kernel to deliver a real one *here*. `fs_resolve`
|
||||
// hands any process a sendable handle to this endpoint, and
|
||||
// `signal_bind`/`timer_bind`/`process_subscribe`/spawn's exit
|
||||
// endpoint all used to accept any handle the caller held — so a
|
||||
// stranger could point its own signal delivery at PID 1 and signal
|
||||
// itself, and the terminate badge landing here was genuine in every
|
||||
// bit. What makes these branches trustworthy is therefore in the
|
||||
// KERNEL, not in this comment: binding a kernel notification to an
|
||||
// endpoint now requires *owning* that endpoint (`ipc.ownedBy`), so a
|
||||
// signal here comes only from our supervisor or our own group, a
|
||||
// timer landing only from a timer we armed, and a child-exit notice
|
||||
// only from a child we spawned. The buffered-message branch below is
|
||||
// the one still carrying a stranger's bytes, and it is the one that
|
||||
// authenticates its sender.
|
||||
if (process.signalsFrom(got.badge)) |signals| {
|
||||
if (signals.has(.terminate)) shutDown();
|
||||
continue;
|
||||
}
|
||||
if (got.isTimer()) {
|
||||
// The pending power subscription rides any timer landing: by the
|
||||
// time one arrives, the bind's reply has left and the power
|
||||
// service is serving again.
|
||||
if (power_pending) {
|
||||
power_pending = false;
|
||||
subscribePower();
|
||||
}
|
||||
if (heartbeat_running) {
|
||||
_ = logging.write("/system/services/init: heartbeat\n");
|
||||
_ = time.timerOnce(supervision_endpoint, 1000);
|
||||
}
|
||||
continue;
|
||||
}
|
||||
if (got.isMessage()) {
|
||||
// A buffered message: the only thing here an anonymous stranger
|
||||
// can put in front of PID 1. Authenticated by sender, never by
|
||||
// payload — see `onPowerEvent`.
|
||||
onPowerEvent(got.senderTaskId(), receive[0..got.len]);
|
||||
continue;
|
||||
}
|
||||
if (got.isChildExit()) {
|
||||
restartChild(got.childProcessId());
|
||||
continue;
|
||||
}
|
||||
continue; // anything else: keep waiting
|
||||
}
|
||||
if (build_options.serial and got.isTimer()) {
|
||||
_ = logging.write("/system/services/init: heartbeat\n");
|
||||
_ = time.timerOnce(supervision_endpoint, 1000);
|
||||
continue;
|
||||
}
|
||||
if (got.isMessage() and got.len >= 2 and receive[0] == @intFromEnum(power_protocol.Operation.event)) {
|
||||
// A power event (the only buffered messages init receives).
|
||||
if (receive[1] == @intFromEnum(power_protocol.Event.power_button)) shutDown();
|
||||
continue;
|
||||
}
|
||||
if (got.isChildExit()) {
|
||||
restartChild(got.childProcessId());
|
||||
continue;
|
||||
}
|
||||
// Anything else: keep waiting.
|
||||
if (got.isNotification()) continue;
|
||||
// The universal ping, answered by the empty reply. A ping may still carry
|
||||
// a capability — the kernel installs one regardless of length — and this
|
||||
// `continue` disposes of it through the turn's `defer`, which is exactly
|
||||
// what it failed to do when the close lived in the branches.
|
||||
if (got.len == 0) continue;
|
||||
reply_len = serveRegistry(receive[0..got.len], &reply_buffer, got.senderTaskId(), &arrived);
|
||||
reply_capability = pending_capability;
|
||||
pending_capability = null;
|
||||
}
|
||||
}
|
||||
|
||||
/// A buffered message claiming to be a power event.
|
||||
///
|
||||
/// **Privileged control traffic is authenticated by sender, never by content.**
|
||||
/// init's registry endpoint is its supervision endpoint, and `fs_resolve` installs
|
||||
/// a sendable handle to any mount's backend in *any* caller's table
|
||||
/// (system/kernel/process.zig), so after P2 every ring-3 process holds a handle it
|
||||
/// can `ipc_send` into. Two payload bytes were once enough to reach `shutDown()`
|
||||
/// from here — which stops every service and parks PID 1 in its final sleep,
|
||||
/// destroying the registry for the rest of the boot, and does it for any process
|
||||
/// that cares to ask.
|
||||
///
|
||||
/// The sender's task id is the fix, because it is not the sender's to choose: the
|
||||
/// kernel stamps it into the badge's low bits as it copies the message into the
|
||||
/// ring. Init is the registry, so it knows exactly which task holds `power`, and
|
||||
/// that task alone is believed. A provider that dies and rebinds moves the
|
||||
/// authorization with the binding; a name nothing holds authorizes nobody. Task
|
||||
/// ids are never reused, so even a dead provider's id cannot be inherited.
|
||||
///
|
||||
/// (One task, not one process: the ACPI service is single-threaded and publishes
|
||||
/// from the same task that bound the name. A threaded provider would want its
|
||||
/// leader compared instead — which is a change to make when one appears, not a
|
||||
/// looser rule to leave lying around for it.)
|
||||
fn onPowerEvent(sender: u32, payload: []const u8) void {
|
||||
const authorized = power_task orelse {
|
||||
std.log.info("ignored a power event from pid {d}: nothing holds /protocol/power", .{sender});
|
||||
return;
|
||||
};
|
||||
if (sender != authorized) {
|
||||
std.log.info("ignored a power event from pid {d}: /protocol/power is pid {d}", .{ sender, authorized });
|
||||
return;
|
||||
}
|
||||
if (payload.len < 2) return;
|
||||
if (payload[0] != @intFromEnum(power_protocol.Operation.event)) return;
|
||||
if (payload[1] == @intFromEnum(power_protocol.Event.power_button)) shutDown();
|
||||
}
|
||||
|
||||
/// A supervised boot service died. Find which one and restart it — unless it exited
|
||||
/// cleanly (it chose to stop, e.g. a driver with no hardware) or has hit the crash-loop
|
||||
/// cap. Reclaiming the dead process is already the kernel's job (docs/process-lifecycle.md
|
||||
/// iron rule 1); init only decides whether to bring it back.
|
||||
fn restartChild(id: u32) void {
|
||||
// Whatever it served, it serves no longer: the name goes back before the
|
||||
// replacement asks for it, so the restarted instance binds rather than
|
||||
// colliding with its own corpse.
|
||||
unbindTask(id);
|
||||
if (shutting_down) return; // deaths during the stop sequence are expected, not crashes
|
||||
for (services[0..service_count], 0..) |*service, i| {
|
||||
if (child_ids[i] != id) continue;
|
||||
@@ -209,22 +825,26 @@ fn restartChild(id: u32) void {
|
||||
// An untracked child (e.g. the log-flush one-shot): nothing to restart.
|
||||
}
|
||||
|
||||
/// Look up the power service and subscribe our endpoint (handed over as the
|
||||
/// call's capability) so events arrive as buffered messages here.
|
||||
/// Subscribe our endpoint (handed over as the call's capability) to the power
|
||||
/// service, so events arrive as buffered messages here. init is the registry, so
|
||||
/// it never resolves `/protocol/power` — it reads its own table, which is also
|
||||
/// what makes this reachable at all: the subscription is armed by the bind that
|
||||
/// put the endpoint there.
|
||||
///
|
||||
/// **This is the only place PID 1 blocks on another process, and it is the one
|
||||
/// hazard the registrar has.** One thread serves both the namespace and this
|
||||
/// call, so while it is outstanding init answers nobody: if the callee were
|
||||
/// itself blocked asking init to resolve a name, the pair would never move. Two
|
||||
/// things keep that from happening — the call is deferred to the next turn of
|
||||
/// the loop (so the provider has its bind reply and is on its way to
|
||||
/// `replyWait`), and the power provider resolves every name it needs *before* it
|
||||
/// binds (system/services/acpi/acpi.zig, `manager_channel`). Any future service
|
||||
/// init calls owes the same discipline.
|
||||
fn subscribePower() void {
|
||||
var handle: ?ipc.Handle = null;
|
||||
var tries: u32 = 0;
|
||||
while (handle == null and tries < 200) : (tries += 1) {
|
||||
handle = ipc.lookup(.power);
|
||||
if (handle == null) time.sleepMillis(20);
|
||||
}
|
||||
// A missing power service is not fatal — init proceeds to its heartbeat and
|
||||
// a `terminate` signal still drives shutdown. Silent so the no-ramdisk init
|
||||
// test's heartbeat marker is the next line written.
|
||||
const h = handle orelse return;
|
||||
const handle = power_endpoint orelse return;
|
||||
const request = power_protocol.Subscribe{};
|
||||
var reply: [power_protocol.message_maximum]u8 = undefined;
|
||||
_ = ipc.callCap(h, std.mem.asBytes(&request), &reply, supervision_endpoint) catch {};
|
||||
_ = ipc.callCap(handle, std.mem.asBytes(&request), &reply, supervision_endpoint) catch {};
|
||||
}
|
||||
|
||||
/// The stop sequence: persist the log while storage is still up, then terminate
|
||||
@@ -242,7 +862,7 @@ fn shutDown() void {
|
||||
i -= 1;
|
||||
if (child_ids[i] != 0) process.stop(child_ids[i], 2000, supervision_endpoint);
|
||||
}
|
||||
if (ipc.lookup(.power)) |h| {
|
||||
if (power_endpoint) |h| {
|
||||
const request = power_protocol.Shutdown{};
|
||||
var reply: [power_protocol.message_maximum]u8 = undefined;
|
||||
_ = ipc.call(h, std.mem.asBytes(&request), &reply) catch {};
|
||||
|
||||
Reference in New Issue
Block a user