danos/system/services/init/init.zig

993 lines
51 KiB
Zig

//! /system/services/init — the first user-space program, PID 1. Built as its own
//! freestanding binary (see build.zig), shipped on the boot volume at /system/services/init,
//! loaded by the bootloader, and started in ring 3 as a scheduled process by the
//! kernel (system/kernel/process.zig). It links against the shared user runtime
//! library `runtime` and talks to the kernel only through `runtime`'s system_call wrappers.
//!
//! It proves the C-convention heap works, then — as PID 1 — acts as the system's
//! **service supervisor**: it spawns the user-space services danos brings up at boot
//! (the VFS server, the device manager), and settles into an event loop as the root
//! of user space. Drivers are *not* its job: the device manager discovers the
//! hardware and spawns those. This is the service half of the service/driver spawn
//! split (docs/driver-model.md).
//!
//! M21: init also owns **orderly shutdown**. It supervises its children (keeping
//! their ids and an exit endpoint), subscribes to the power service, and on a
//! power-button event runs the stop sequence over its children in reverse order
//! before asking the power service to enter S5 — lifecycle (M17) and events (M21)
//! composing into a clean poweroff.
//!
//! P2: init is also the **registrar** — it serves `/protocol`, the namespace where
//! a program finds everything it talks to
//! (docs/os-development/protocol-namespace.md). It is the natural home: it already
//! spawns the services and already holds the supervision link to each, so it is the
//! process that *knows* which binary is which. Registry traffic rides the same
//! endpoint as supervision, because one thread can only wait in one place — the
//! loop below answers vfs `open`/`readdir`/`bind` alongside signals, timers, power
//! events, and children's deaths.
//!
//! Which sets the security posture of this file. `fs_resolve` installs a mounted
//! backend's endpoint capability in *any* caller's handle table, so sharing the
//! mailbox means **every ring-3 process can send into PID 1**. Two rules follow,
//! and both are structural here rather than remembered per branch:
//!
//! - **Privileged action requires an attested sender.** What arrives is a
//! stranger's bytes; the only identity on it is the task id the kernel stamps.
//! Content never authorizes (`onPowerEvent`), and neither does a name — the
//! registrar attests a caller's supervision by task id (`supervisorSatisfies`).
//! - **Absence is the enforcement.** P3: `open` consults the manifest with the
//! same attested identity a `bind` does, and a caller with no grant is told
//! exactly what a caller asking for a name nobody bound is told — `-ENOENT`,
//! and no capability (`onOpen`). Restriction stage one of
//! docs/os-development/protocol-namespace.md: what a process cannot open does
//! not exist for it, so there is no "permission denied" to distinguish.
//! - **A capability that arrives is closed unless it is claimed** (`Arrival`),
//! because PID 1's thirty-two handle slots are a resource an unauthenticated
//! caller would otherwise be able to spend.
const std = @import("std");
const ipc = @import("ipc");
const process = @import("process");
const time = @import("time");
const memory = @import("memory");
const logging = @import("logging");
const power_protocol = @import("power-protocol");
const vfs_protocol = @import("vfs-protocol");
const envelope = @import("envelope");
const build_options = @import("build_options");
const fs = @import("file-system");
const csv = @import("csv");
/// The system services init brings up at boot are init's policy, not the kernel's —
/// and that policy is now data: `/system/configuration/init.csv` (see `loadServices`), read at
/// startup instead of a hardcoded list. Drivers are absent on purpose: the device
/// manager owns those.
///
/// The most services `/system/configuration/init.csv` can list, and the most argv entries (beyond the
/// path) each may carry. Fixed caps because init parses the list into static storage —
/// the freestanding, no-allocator counterpart to the device manager's registry table.
const max_services = 16;
const max_service_args = 4;
/// One service init starts, parsed from a row of `/system/configuration/init.csv`: its binary path
/// and argv, both slices into `init_csv` (held for the life of the process).
const Service = struct {
path: []const u8 = "",
arg_buffer: [max_service_args][]const u8 = undefined,
arg_count: usize = 0,
fn arguments(self: *const Service) []const []const u8 {
return self.arg_buffer[0..self.arg_count];
}
};
/// The `/system/configuration/init.csv` bytes, held because the parsed services slice into them.
var init_csv: [4096]u8 = undefined;
var services: [max_services]Service = .{Service{}} ** max_services;
var service_count: usize = 0;
/// The live process id of each service (0 = not running) and its restart count,
/// indexed by position in `services`. init supervises these: it spawns them against
/// `supervision_endpoint` and, on a child's death, restarts it (up to
/// `maximum_restarts`) — the reincarnation half of resilience (docs/resilience.md),
/// the service-level counterpart to the device manager's driver restarts.
var child_ids: [max_services]u32 = .{0} ** max_services;
var restart_counts: [max_services]u32 = .{0} ** max_services;
var shutting_down = false;
var supervision_endpoint: ipc.Handle = 0;
/// Parse `/system/configuration/init.csv` into `services`, in file order (startup order; shutdown is
/// the reverse). Each row is a binary path followed by its argv, comma-separated;
/// `#` comments and blank lines are ignored. The file lives in the initial ramdisk,
/// which the kernel serves directly, so init — PID 1, running before any filesystem
/// service — reads it with a plain fs.open, the same mechanism the device manager
/// uses for /system/configuration/devices.csv. A missing file means no services (the no-ramdisk
/// isolation test): loud, but not fatal.
fn loadServices() void {
const used = readConfiguration("/system/configuration/init.csv", &init_csv) orelse {
_ = logging.write("/system/services/init: /system/configuration/init.csv missing — no services started\n");
return;
};
var lines = std.mem.splitScalar(u8, init_csv[0..used], '\n');
while (lines.next()) |line| {
const body = csv.stripComment(line);
if (body.len == 0) continue;
if (service_count >= services.len) {
_ = logging.write("/system/services/init: /system/configuration/init.csv has more services than the table holds\n");
break;
}
var it = csv.fields(body);
const path = it.next() orelse continue;
if (path.len == 0) continue;
var service: Service = .{ .path = path };
while (it.next()) |argument| {
if (argument.len == 0) continue; // padding, or a trailing comma
if (service.arg_count >= max_service_args) break;
service.arg_buffer[service.arg_count] = argument;
service.arg_count += 1;
}
services[service_count] = service;
service_count += 1;
}
}
/// Read a whole configuration file into `into`, returning the byte count. Both of
/// init's manifests live in the initial ramdisk the kernel serves directly, so this
/// works before any filesystem service exists. A file that fills the buffer exactly
/// is reported: a manifest silently losing its last rows is a policy change nobody
/// asked for, and the symptom (one service refused a name) points nowhere near it.
fn readConfiguration(path: []const u8, into: []u8) ?usize {
var file = fs.open(path, .{}) orelse return null;
defer file.close();
var used: usize = 0;
while (used < into.len) {
const n = file.read(into[used..]) orelse break;
if (n == 0) break;
used += n;
}
if (used == into.len) std.log.info("{s} filled the read buffer — rows past {d} bytes are lost", .{ path, used });
return used;
}
/// Give up restarting a service after this many crashes — a crash-loop cap, so a service
/// that faults immediately on every spawn doesn't respawn forever.
const maximum_restarts = 3;
// --- the registry: /protocol ------------------------------------------------
/// Longest contract name the namespace admits (`display`, `test/shared-memory`)
/// and the most that may be bound at once. Both static, like everything else
/// init holds.
const maximum_name = 64;
const maximum_bindings = 16;
const maximum_grants = 64;
/// One bound contract: the name, the provider's endpoint (a capability init
/// holds and hands to whoever opens the name), and the provenance a diagnostic
/// listing answers "who serves this?" with.
const Binding = struct {
used: bool = false,
name: [maximum_name]u8 = undefined,
name_len: usize = 0,
endpoint: ipc.Handle = 0,
task: u32 = 0,
binary: [64]u8 = undefined,
binary_len: usize = 0,
fn nameSlice(self: *const Binding) []const u8 {
return self.name[0..self.name_len];
}
fn binarySlice(self: *const Binding) []const u8 {
return self.binary[0..self.binary_len];
}
};
var bindings: [maximum_bindings]Binding = .{Binding{}} ** maximum_bindings;
/// What a grant row permits.
///
/// - `bind` — claim the name, i.e. provide the contract.
/// - `open` — reach the name, i.e. speak the contract to whoever provides it.
/// - `supervise` — stand in a third task's supervision chain: a task running this
/// binary, under this supervisor, may be the supervising task named by an
/// `open` row for this contract. It grants the *delegate* nothing itself.
///
/// `supervise` exists because attestation is deliberately one hop deep
/// (`supervisorSatisfies`): init vouches only for tasks it or the kernel started.
/// The driver tree is deeper than that — the device manager starts the PS/2 bus,
/// and the bus starts the keyboard and mouse drivers — so without a way to say
/// "this task is an authorized supervisor", a legitimate grandchild would be
/// indistinguishable from a laundering deputy. Naming the delegate in the
/// manifest is what tells them apart, and it is the same shape as every other
/// row: a binary, the supervisor it must have, and the contract it concerns.
/// Deliberately `open`-only — a delegate may vouch for what its children may
/// *reach*, never for what they may *claim* — so the bind path's attestation is
/// exactly what P2 shipped and every refusal it makes still holds.
const Permission = enum { bind, open, supervise };
/// One row of `/system/configuration/protocol.csv`. Every field may end in `*`,
/// which matches any tail — the subtree scoping the design doc describes, and
/// what lets one row grant the whole `/test/` family its `test/...` names.
const Grant = struct {
binary: []const u8 = "",
supervisor: []const u8 = "",
permission: Permission = .bind,
name: []const u8 = "",
};
/// Roomier than init.csv's: this manifest carries a row per provider per spawn
/// path, a row per client per contract it reaches, and its own format
/// documentation — which is most of the bytes, and is the point of the file.
var protocol_csv: [16384]u8 = undefined;
var grants: [maximum_grants]Grant = .{Grant{}} ** maximum_grants;
var grant_count: usize = 0;
/// Parse `/system/configuration/protocol.csv` — the grant manifest. Separate from
/// init.csv because every field there after the path is argv, and overloading that
/// would be ambiguous; separate *files* also means a grant exists for binaries init
/// never spawns (the drivers, which the device manager owns).
fn loadGrants() void {
const used = readConfiguration("/system/configuration/protocol.csv", &protocol_csv) orelse {
_ = logging.write("/system/services/init: /system/configuration/protocol.csv missing — no protocol may be bound\n");
return;
};
var lines = std.mem.splitScalar(u8, protocol_csv[0..used], '\n');
while (lines.next()) |line| {
const body = csv.stripComment(line);
if (body.len == 0) continue;
if (grant_count >= grants.len) {
_ = logging.write("/system/services/init: /system/configuration/protocol.csv has more rows than the table holds\n");
break;
}
var it = csv.fields(body);
const binary = it.next() orelse continue;
const supervisor = it.next() orelse continue;
const permission = it.next() orelse continue;
const name = it.next() orelse continue;
if (binary.len == 0 or supervisor.len == 0 or name.len == 0) continue;
const kind: Permission = if (std.mem.eql(u8, permission, "bind"))
.bind
else if (std.mem.eql(u8, permission, "open"))
.open
else if (std.mem.eql(u8, permission, "supervise"))
.supervise
else
continue; // an unreadable row grants nothing rather than something wrong
grants[grant_count] = .{ .binary = binary, .supervisor = supervisor, .permission = kind, .name = name };
grant_count += 1;
}
}
/// Match a manifest field against a value: exact, or a trailing `*` matching any
/// tail. The wildcard is how a subtree is granted whole (`/test/*` for every test
/// fixture, `test/*` for every name they may claim).
fn matches(pattern: []const u8, value: []const u8) bool {
if (pattern.len != 0 and pattern[pattern.len - 1] == '*') {
const prefix = pattern[0 .. pattern.len - 1];
return value.len >= prefix.len and std.mem.eql(u8, value[0..prefix.len], prefix);
}
return std.mem.eql(u8, pattern, value);
}
/// A snapshot of the kernel's process records — the only identity in the system
/// that cannot be forged, because the kernel stamps it at spawn. Refreshed per
/// authorization; binds are rare, so the copy costs nothing that matters.
var process_table: [64]process.ProcessDescriptor = undefined;
var process_count: usize = 0;
var process_truncated = false;
fn refreshProcessTable() void {
const total = process.processes(&process_table);
process_count = @min(total, process_table.len);
process_truncated = total > process_table.len;
}
/// Whether task `id` is still alive, as the last snapshot saw it. A snapshot that
/// did not fit answers "alive" for anything it did not list: refusing a bind is
/// recoverable, stealing a live provider's name is not.
fn taskAlive(id: u32) bool {
return descriptorOf(id) != null or process_truncated;
}
fn descriptorOf(id: u32) ?*const process.ProcessDescriptor {
for (process_table[0..process_count]) |*descriptor| {
if (descriptor.id == id) return descriptor;
}
return null;
}
fn nameOf(descriptor: *const process.ProcessDescriptor) []const u8 {
const length = @min(@as(usize, descriptor.name_length), descriptor.name.len);
return descriptor.name[0..length];
}
/// The name a kernel task answers to in a grant row. Kernel tasks carry no
/// binary, so the manifest spells the harness's parentage `kernel`.
const kernel_supervisor = "kernel";
/// init's own task id, read once at startup. Ids are monotonic and never reused
/// (system/kernel/process.zig), so an id comparison is an *identity* test where a
/// name comparison is only a resemblance test — the whole basis of the
/// attestation below.
var own_task: u32 = 0;
/// Whether `id` is a process THIS init spawned: a lookup in its own child table,
/// which is the one record of "I started that one" nobody else can write.
fn spawnedByUs(id: u32) bool {
if (id == 0) return false;
for (child_ids[0..service_count]) |child| {
if (child == id) return true;
}
return false;
}
/// Who is asking, attested by the kernel: the caller's binary, and the **task**
/// that spawned it — an id, not a name.
///
/// A name alone is not identity: `spawn` is ungated, so a hostile process can
/// start a granted binary itself and would inherit its grants. Neither is the
/// supervisor's *name* enough, and this is the trap the first cut fell into —
/// init and the device manager are ordinary bundled binaries, so an attacker
/// spawns its own `/system/services/init` and lets that instance spawn
/// `/system/services/input`. Both kernel-stamped names then match the grant row
/// exactly, and walking to the root of the chain does not help either: the
/// laundered chain still roots at the real PID 1. What refuses it is asking
/// *which task* the supervisor is, and only accepting one init can vouch for.
const Identity = struct {
/// The caller's binary path, exactly as the kernel stamped it at spawn.
binary: []const u8,
/// The supervising task's id. 0 means the kernel spawned the caller, which
/// no ring-3 process can arrange: every `system_spawn` stamps the caller as
/// the child's supervisor (system/kernel/process.zig `systemSpawn`).
supervisor_task: u32,
/// The supervising task's kernel-stamped binary — the grant row's supervisor
/// column is matched against this, and the refusal log prints it. `kernel`
/// when there is no supervising task.
supervisor_binary: []const u8,
/// Whether init can vouch for how the supervising task came to exist: it is
/// this init, a process this init spawned, or a process the KERNEL spawned.
/// A supervisor init cannot vouch for satisfies no row, however well its
/// name reads — that is the laundering deputy's refusal.
supervisor_vouched: bool,
};
/// The process a task belongs to. A thread resolves to its leader: threads share
/// a binary (a thread's own record is named `thread`), and the supervision link
/// that matters is the process's.
fn leaderOf(descriptor: *const process.ProcessDescriptor) *const process.ProcessDescriptor {
if (descriptor.leader == descriptor.id) return descriptor;
return descriptorOf(descriptor.leader) orelse descriptor;
}
/// Resolve the badge on a request into an identity, one hop up the supervision
/// chain in the kernel's records — one hop is enough because the hop is attested
/// by id (see `supervisorSatisfies`), and every id in the chain init accepts is
/// one init or the kernel created.
fn identify(task: u32) ?Identity {
const caller = descriptorOf(task) orelse return null;
const leader = leaderOf(caller);
if (leader.supervisor == 0) return .{
.binary = nameOf(leader),
.supervisor_task = 0,
.supervisor_binary = kernel_supervisor,
.supervisor_vouched = true, // the kernel is the root of trust, not a claimant
};
// The supervising *task* may be a worker thread of the supervising process;
// its process is what the manifest names and what init recorded at spawn.
const supervisor = leaderOf(descriptorOf(leader.supervisor) orelse return null); // unattestable: refuse
return .{
.binary = nameOf(leader),
.supervisor_task = supervisor.id,
.supervisor_binary = nameOf(supervisor),
.supervisor_vouched = supervisor.id == own_task or
spawnedByUs(supervisor.id) or
supervisor.supervisor == 0,
};
}
/// Whether the caller's supervising task satisfies a grant row's supervisor
/// column. The column names *the authorized supervising task*, matched by
/// identity — the binary it must be, plus proof that this instance of that
/// binary is the authorized one:
///
/// - `kernel` is satisfied only by a genuinely kernel-spawned caller
/// (supervisor id 0). A ring-3 process cannot manufacture that: user
/// `system_spawn` always stamps the caller (system/kernel/process.zig).
/// - init's own binary is satisfied only when the supervising task IS this
/// init (`own_task`).
/// - any other binary — the device manager, a test fixture spawning another —
/// is satisfied only when the supervising task is one init spawned itself
/// (its own child table) or one the kernel spawned. Everything init and the
/// kernel start is therefore reachable; a chain that passes through a
/// process *neither* of them started is not.
fn supervisorSatisfies(column: []const u8, identity: Identity) bool {
if (std.mem.eql(u8, column, kernel_supervisor)) return identity.supervisor_task == 0;
if (identity.supervisor_task == 0) return false; // a kernel task answers to no binary column
if (!matches(column, identity.supervisor_binary)) return false;
return identity.supervisor_vouched;
}
/// Whether `identity` is granted `permission` on `name`.
fn granted(identity: Identity, permission: Permission, name: []const u8) bool {
for (grants[0..grant_count]) |grant| {
if (grant.permission != permission) continue;
if (!matches(grant.binary, identity.binary)) continue;
if (!supervisorSatisfies(grant.supervisor, identity)) continue;
if (!matches(grant.name, name)) continue;
return true;
}
return false;
}
/// Whether `identity` may reach `name` — `granted(.open, …)`, plus the one hop
/// `open` takes that `bind` does not (`Permission.supervise`).
///
/// The hop is needed because the driver tree is three deep and attestation is
/// one: the PS/2 keyboard driver's supervising task is the PS/2 bus driver,
/// which the device manager started, which init started. Init cannot vouch for
/// the bus by acquaintance — it never met it — so the manifest says so instead,
/// and says it per contract: `ps2-bus` may be the supervisor named in an `open`
/// grant for `ps2-bus` and for `input`, and for nothing else.
fn mayOpen(identity: Identity, name: []const u8) bool {
if (granted(identity, .open, name)) return true;
return delegatedOpen(identity, name);
}
/// The delegated `open`: the row's supervisor column names the caller's actual
/// supervising task by binary, that task is one init cannot vouch for directly,
/// and a `supervise` row authorizes it for exactly this contract.
///
/// The delegate itself is attested the ordinary way (`granted` → strict
/// `supervisorSatisfies`), so the chain is still anchored one hop above it in
/// init or the kernel and the recursion stops there. Two hops of manifest, never
/// an unbounded walk — a laundering deputy is refused at the first hop nobody
/// wrote a row for.
fn delegatedOpen(identity: Identity, name: []const u8) bool {
if (identity.supervisor_task == 0) return false; // a kernel-spawned caller needs no delegate
if (identity.supervisor_vouched) return false; // already answered by `granted` above
const delegate = identify(identity.supervisor_task) orelse return false;
if (!granted(delegate, .supervise, name)) return false;
for (grants[0..grant_count]) |grant| {
if (grant.permission != .open) continue;
if (!matches(grant.binary, identity.binary)) continue;
if (!matches(grant.supervisor, identity.supervisor_binary)) continue;
if (!matches(grant.name, name)) continue;
return true;
}
return false;
}
fn findBinding(name: []const u8) ?*Binding {
for (&bindings) |*binding| {
if (binding.used and std.mem.eql(u8, binding.nameSlice(), name)) return binding;
}
return null;
}
/// Release a binding: the provider's endpoint capability goes back to the handle
/// table, and the name is free for the next claimant. init's own cached power
/// channel goes with it — a closed handle number is reused by the next capability
/// that arrives, and a stale copy would quietly aim the shutdown call at a
/// stranger. So does the authorized power *task*: nothing may speak for a
/// contract nobody holds.
fn releaseBinding(binding: *Binding) void {
if (std.mem.eql(u8, binding.nameSlice(), power_contract)) {
power_endpoint = null;
power_task = null;
power_pending = false;
}
_ = ipc.close(binding.endpoint);
binding.* = .{};
}
/// Drop every name a dead process held. Called when a supervised child dies (so
/// the restarted instance can bind again) and whenever a bind finds the current
/// owner gone — providers init does not supervise need the second path.
fn unbindTask(task: u32) void {
for (&bindings) |*binding| {
if (binding.used and binding.task == task) {
std.log.info("/protocol/{s} released ({s} is gone)", .{ binding.nameSlice(), binding.binarySlice() });
releaseBinding(binding);
}
}
}
/// A contract name as the namespace spells it: the mount-relative path a resolve
/// hands us ("/display") and the name a bind sends ("display") are the same thing
/// with and without a leading slash, so one normaliser serves both. Empty or
/// longer than the namespace admits is not a name.
fn contractName(raw: []const u8) ?[]const u8 {
const name = if (raw.len != 0 and raw[0] == '/') raw[1..] else raw;
if (name.len == 0 or name.len > maximum_name) return null;
return name;
}
/// The provider's endpoint that will ride the *next* reply, when the request was
/// an `open` that found its contract.
var pending_capability: ?ipc.Handle = null;
/// The ownership rule for a capability that arrives with a turn of the loop —
/// **the turn owns it until a handler takes it, and closes whatever is left** —
/// lives in `ipc.Arrival`, next to `replyWait`, because it is not PID 1's rule:
/// the service harness every other service runs (library/kernel/service.zig) had
/// the identical hole and now states the identical contract.
const Arrival = ipc.Arrival;
/// The one contract init is itself a client of. It never resolves the name — it
/// *is* the registry, so it reads its own table; the binding is what hands it the
/// channel.
const power_contract = "power";
/// Set when `power` is bound: init subscribes to it on the next turn of the loop,
/// never inside the bind — the provider is blocked on our reply until then, so
/// calling it here would deadlock the pair.
var power_pending = false;
var power_endpoint: ?ipc.Handle = null;
/// The one task authorized to deliver power events: whoever holds the `power`
/// binding. Recorded at the bind and cleared with the binding, so a provider that
/// dies and rebinds re-derives it with no further ceremony.
///
/// This is the *authentication* for the shutdown path. init's registry endpoint
/// is its supervision endpoint, and `fs_resolve("/protocol")` installs a sendable
/// handle to it in any caller's table — so after P2 every ring-3 process can post
/// into PID 1's mailbox. A power event may therefore never be believed on the
/// strength of its payload; it is believed because the kernel stamped the
/// sender's task id on it and that id is the provider's.
var power_task: ?u32 = null;
/// Whether the heartbeat's re-arming timer is running. A timer landing carries no
/// identity, so the loop cannot tell one timer from another — which means exactly
/// one may ever be in flight, or every landing re-arms and the beat doubles. (It
/// did: two beats a second is enough extra chatter to cut a driver's echoed line
/// in half on the shared serial stream.) So the deferred power subscribe borrows
/// the heartbeat's tick when there is one, and arms its own only when there is not.
var heartbeat_running = false;
/// Answer one registry request. Writes a vfs-protocol reply into `reply` and
/// returns its length; a capability the reply must carry lands in
/// `pending_capability`. `arrived` is the capability the *request* carried, owned
/// by the turn — nothing here has to close it, only `bind` has to claim it.
fn serveRegistry(request_bytes: []const u8, reply: []u8, sender: u32, arrived: *Arrival) usize {
if (request_bytes.len < envelope.prefix_size)
return answer(reply, -envelope.EPROTO, 0);
// The header is read field by field rather than reinterpreted whole, and the
// verb is compared as a number rather than decoded into the generated
// `Operation`: the bytes come from anyone at all, so a value outside the enum
// must be a refusal, never an `@enumFromInt`. This is deliberately NOT
// `Protocol.Provider.dispatch` for the same reason — PID 1 reads a stranger's
// packet, and it reads it by hand.
const operation = std.mem.readInt(u32, request_bytes[0..4], .little);
const body = request_bytes[envelope.prefix_size..];
if (operation == @intFromEnum(vfs_protocol.Operation.bind))
return answer(reply, onBind(sender, body, arrived), 0);
// Only `bind` claims a capability; one attached to anything else is closed by
// the turn's `defer` in the loop, along with the ones sent to a request that
// was too short to name a verb at all.
if (operation == @intFromEnum(vfs_protocol.Operation.open)) {
// `open`'s fixed part is the flags word, which means nothing to a
// namespace; the name follows it as the packet's tail.
if (body.len < @sizeOf(vfs_protocol.Open)) return answer(reply, -envelope.EPROTO, 0);
return onOpen(reply, sender, body[@sizeOf(vfs_protocol.Open)..]);
}
if (operation == @intFromEnum(vfs_protocol.Operation.readdir)) {
if (body.len < @sizeOf(vfs_protocol.Readdir)) return answer(reply, -envelope.EPROTO, 0);
return onReaddir(reply, std.mem.readInt(u64, body[0..8], .little));
}
// Everything else a filesystem answers is meaningless here: `/protocol` holds
// contracts, not bytes.
return answer(reply, -envelope.ENOSYS, 0);
}
/// Lay down the envelope's reply prefix (and say how many payload bytes the
/// caller has already written after it).
fn answer(reply: []u8, status: i32, payload_len: usize) usize {
const header = envelope.Status{ .status = status, .len = @intCast(payload_len) };
@memcpy(reply[0..envelope.prefix_size], std.mem.asBytes(&header));
return envelope.prefix_size + payload_len;
}
/// `bind(name, capability = the provider's endpoint)`. The capability is the
/// point of the call, so a bind without one is malformed. Every refusal below
/// simply returns: the endpoint stays the turn's, and the turn closes it — which
/// is why there is not one `ipc.close` on the way out of any of the six of them.
/// The success path is the only one that says anything about ownership, because
/// it is the only one that keeps the capability.
fn onBind(sender: u32, raw_name: []const u8, arrived: *Arrival) i32 {
if (arrived.peek() == null) return -envelope.EPROTO;
const name = contractName(raw_name) orelse return -envelope.ENOENT;
refreshProcessTable();
const identity = identify(sender) orelse return -envelope.EPERM;
if (!granted(identity, .bind, name)) {
std.log.info("refused bind of /protocol/{s} by {s} (pid {d}, supervisor {s} pid {d})", .{
name,
identity.binary,
sender,
identity.supervisor_binary,
identity.supervisor_task,
});
return -envelope.EPERM;
}
if (findBinding(name)) |existing| {
// Collision is an error — never last-writer-wins — unless the incumbent
// is dead, which is how a restarted provider retakes its own name.
if (taskAlive(existing.task)) {
std.log.info("refused bind of /protocol/{s}: held by {s} (pid {d})", .{ name, existing.binarySlice(), existing.task });
return -envelope.EBUSY;
}
releaseBinding(existing);
}
const slot = for (&bindings) |*binding| {
if (!binding.used) break binding;
} else return -envelope.ENOSPC;
// Claimed: the binding owns the endpoint from here, and `releaseBinding` is
// what closes it.
const endpoint = arrived.take().?;
slot.* = .{ .used = true, .endpoint = endpoint, .task = sender };
@memcpy(slot.name[0..name.len], name);
slot.name_len = name.len;
const binary_len = @min(identity.binary.len, slot.binary.len);
@memcpy(slot.binary[0..binary_len], identity.binary[0..binary_len]);
slot.binary_len = binary_len;
// Provenance, at the moment it becomes true: name -> pid -> binary path.
std.log.info("/protocol/{s} -> pid {d} {s}", .{ name, sender, slot.binarySlice() });
if (std.mem.eql(u8, name, power_contract)) {
power_endpoint = endpoint;
// The bind is also the authentication: whoever holds `power` is the one
// task whose power events init will act on (see `onPowerEvent`).
power_task = sender;
power_pending = true;
// Wake ourselves once the reply has gone out; the subscribe call cannot
// happen while the power service is still blocked on it. The heartbeat's
// tick is that wake when it is running — see `heartbeat_running`.
if (!heartbeat_running) _ = time.timerOnce(supervision_endpoint, 1);
}
return 0;
}
/// `open(name)` -> the provider's endpoint, delivered as the reply's capability.
///
/// **A refusal and an absence are the same answer, and that is the whole point.**
/// The namespace is the restriction (docs/os-development/protocol-namespace.md):
/// what a process may open is what exists for it, so "you may not have this" and
/// "there is no such thing" collapse into one reply — `-ENOENT`, no payload, no
/// capability. A caller therefore has no oracle: it cannot use `open` to learn
/// that a contract it lacks is bound, and — the reason this matters beyond
/// tidiness — stage two's supervisor can refuse, stall for a human, or substitute
/// a fake without the child being able to tell which happened.
///
/// Indistinguishable is a claim about *work done*, not only about the bytes, so
/// both questions are asked on every open whatever the first one answers: the
/// process table is refreshed, the caller identified, the grants scanned and the
/// bindings scanned, and only then is the single verdict formed. Nothing here
/// logs, either — `klog_read` is ungated (system/kernel/process.zig), so a line
/// written on one branch is a line the refused caller can read, and a serial line
/// costs milliseconds it could time. The operator's diagnosis is the pair the
/// namespace already publishes on purpose: `readdir` over `/protocol` says what is
/// bound, `/system/configuration/protocol.csv` says who may reach it, and the
/// client's own retry loop says which one it wanted.
///
/// (Not constant-time in the cryptographic sense, and not claimed to be: the two
/// scans stop at the row they match, and the optimiser is free to sink a pure
/// table walk past a branch that discards it. What is removed is the difference a
/// caller could actually measure or read — a syscall on one branch and not the
/// other, a line in a world-readable log ring, or a serial write costing
/// milliseconds.)
fn onOpen(reply: []u8, sender: u32, raw_name: []const u8) usize {
const name = contractName(raw_name) orelse return answer(reply, -envelope.ENOENT, 0);
refreshProcessTable();
const identity = identify(sender);
const permitted = if (identity) |who| mayOpen(who, name) else false;
const binding = findBinding(name);
if (!permitted) return answer(reply, -envelope.ENOENT, 0);
const found = binding orelse return answer(reply, -envelope.ENOENT, 0);
pending_capability = found.endpoint;
// A contract node has no node id — the capability is the whole answer — but
// the protocol says an `open` reply carries one, so it carries a zero.
const opened = vfs_protocol.Opened{ .node = 0 };
@memcpy(reply[envelope.prefix_size..][0..@sizeOf(vfs_protocol.Opened)], std.mem.asBytes(&opened));
return answer(reply, 0, @sizeOf(vfs_protocol.Opened));
}
/// `readdir(cursor)` — the namespace, browsable. One entry per turn, as the vfs
/// protocol lists any directory: kind `protocol`, the contract's name, and the
/// provider's task id in `size`, so a plain listing answers "who serves this?".
fn onReaddir(reply: []u8, cursor: u64) usize {
var index: u64 = 0;
for (&bindings) |*binding| {
if (!binding.used) continue;
if (index != cursor) {
index += 1;
continue;
}
const name = binding.nameSlice();
return writeEntry(reply, .{
.kind = @intFromEnum(vfs_protocol.NodeKind.protocol),
.name_len = @intCast(name.len),
.size = binding.task,
}, name);
}
// End of directory, which the envelope spells as an entry with no name: the
// reply's own length cannot say it any more, because the fixed reply part
// always travels.
return writeEntry(reply, .{}, &.{});
}
/// One `readdir` reply: the entry, then its name inline.
fn writeEntry(reply: []u8, entry: vfs_protocol.DirectoryEntry, name: []const u8) usize {
const total = vfs_protocol.directory_entry_size + name.len;
if (envelope.prefix_size + total > reply.len) return answer(reply, -envelope.EPROTO, 0);
@memcpy(reply[envelope.prefix_size..][0..vfs_protocol.directory_entry_size], std.mem.asBytes(&entry));
@memcpy(reply[envelope.prefix_size + vfs_protocol.directory_entry_size ..][0..name.len], name);
return answer(reply, 0, total);
}
pub fn main(startup: process.Init) void {
// `registry` is the scenario mode: serve /protocol and nothing else. The
// kernel test harness spawns its own providers directly, so it wants the
// naming layer up without init's whole service list underneath it
// (docs/security-track-plan.md, decision 9).
const registry_only = if (startup.arguments.get(1)) |role| std.mem.eql(u8, role, "registry") else false;
// Prove the heap end to end: allocate through the runtime allocator (which
// mmaps pages from the kernel and carves them with the free list), write into
// that heap buffer (exercising the widened debug_write bounds check), and
// free it. A fault here would kill init before it heartbeats — so the init
// test doubles as the heap regression test. (C code links the same heap via
// the extern malloc/free symbols; Zig code uses this allocator.)
const gpa = memory.allocator();
if (gpa.alloc(u8, 64)) |buffer| {
const message = "/system/services/init: heap ok\n";
@memcpy(buffer[0..message.len], message);
_ = logging.write(buffer[0..message.len]);
gpa.free(buffer);
} else |_| {}
// One endpoint carries everything init waits on: children's exit
// notifications (they are spawned supervised against it), init's own
// signals, power events it subscribes to, and — since PID 1 is the registrar
// — every /protocol request. One thread can wait in one place, so they share
// a mailbox and the loop below tells them apart.
supervision_endpoint = ipc.createIpcEndpoint() orelse {
_ = logging.write("/system/services/init: no endpoint\n");
return;
};
_ = process.bindSignals(supervision_endpoint);
// Our own id, before anything can ask us a question. It is half of the
// registrar's authority: a grant row naming init as the supervisor is
// satisfied by *this* task and no other instance of this binary
// (`supervisorSatisfies`).
own_task = process.taskId();
// The namespace goes up BEFORE anything is spawned, so a service's first
// bind lands rather than retrying. The kernel reserves the prefix: this
// mount is the only one it will ever hold.
loadGrants();
if (!fs.mount("/protocol", supervision_endpoint)) {
_ = logging.write("/system/services/init: /protocol already mounted — not the registrar\n");
}
// Load the service list, then bring each up supervised so init can stop them
// cleanly. Best-effort and silent: each service announces its own readiness,
// and with no /system/configuration/init.csv (an isolation test) the loop starts nothing.
if (!registry_only) {
loadServices();
for (services[0..service_count], 0..) |*service, i| {
if (process.spawnSupervised(service.path, service.arguments(), supervision_endpoint)) |id| child_ids[i] = id;
}
}
// A re-arming timer drives the liveness heartbeat — proof PID 1 is alive (the
// init test's marker) and a -Dserial diagnostic. It is a serial/test-build-only
// concern: a flashable (serial-off) image runs a purely event-driven PID 1 that
// wakes only for real work (signals, power events, children's exits), never for a
// periodic beat. `build_options.serial` is comptime, so the heartbeat — its timer
// and the handler below — folds away entirely when serial is off.
heartbeat_running = build_options.serial and !registry_only;
if (heartbeat_running) _ = time.timerOnce(supervision_endpoint, 1000);
var receive: [vfs_protocol.message_maximum]u8 = undefined;
var reply_buffer: [vfs_protocol.message_maximum]u8 = undefined;
var reply_len: usize = 0;
var reply_capability: ?ipc.Handle = null;
while (true) {
const got = ipc.replyWait(supervision_endpoint, reply_buffer[0..reply_len], &receive, reply_capability);
reply_len = 0; // nothing owed until this turn's request says otherwise
reply_capability = null;
// Whatever capability came with this turn is the turn's, and the turn
// closes it unless a handler claims it. Structural on purpose — see
// `Arrival`; it is what keeps a zero-length call from spending a handle
// slot of PID 1's per call.
var arrived: Arrival = .{ .handle = got.cap };
defer arrived.release();
if (got.isNotification()) {
// Every badge on this branch is stamped by the KERNEL, and a stranger
// cannot stamp one: `ipc_send` — the only way a ring-3 process puts
// something in this mailbox with no reply owed — sets exactly
// `notify_badge_bit | notify_message_bit` and fills the low bits with
// the sender's own task id (system/kernel/ipc-synchronous.zig,
// `sendLocked`).
//
// That is a statement about `ipc_send`, and on its own it proved far
// too little: an attacker does not use `ipc_send` to forge a signal,
// it asks the kernel to deliver a real one *here*. `fs_resolve`
// hands any process a sendable handle to this endpoint, and
// `signal_bind`/`timer_bind`/`process_subscribe`/spawn's exit
// endpoint all used to accept any handle the caller held — so a
// stranger could point its own signal delivery at PID 1 and signal
// itself, and the terminate badge landing here was genuine in every
// bit. What makes these branches trustworthy is therefore in the
// KERNEL, not in this comment: binding a kernel notification to an
// endpoint now requires *owning* that endpoint (`ipc.ownedBy`), so a
// signal here comes only from our supervisor or our own group, a
// timer landing only from a timer we armed, and a child-exit notice
// only from a child we spawned. The buffered-message branch below is
// the one still carrying a stranger's bytes, and it is the one that
// authenticates its sender.
if (process.signalsFrom(got.badge)) |signals| {
if (signals.has(.terminate)) shutDown();
continue;
}
if (got.isTimer()) {
// The pending power subscription rides any timer landing: by the
// time one arrives, the bind's reply has left and the power
// service is serving again.
if (power_pending) {
power_pending = false;
subscribePower();
}
if (heartbeat_running) {
_ = logging.write("/system/services/init: heartbeat\n");
_ = time.timerOnce(supervision_endpoint, 1000);
}
continue;
}
if (got.isMessage()) {
// A buffered message: the only thing here an anonymous stranger
// can put in front of PID 1. Authenticated by sender, never by
// payload — see `onPowerEvent`.
onPowerEvent(got.senderTaskId(), receive[0..got.len]);
continue;
}
if (got.isChildExit()) {
restartChild(got.childProcessId());
continue;
}
continue; // anything else: keep waiting
}
// The universal ping, answered by the empty reply. A ping may still carry
// a capability — the kernel installs one regardless of length — and this
// `continue` disposes of it through the turn's `defer`, which is exactly
// what it failed to do when the close lived in the branches.
if (got.len == 0) continue;
reply_len = serveRegistry(receive[0..got.len], &reply_buffer, got.senderTaskId(), &arrived);
reply_capability = pending_capability;
pending_capability = null;
}
}
/// A buffered message claiming to be a power event.
///
/// **Privileged control traffic is authenticated by sender, never by content.**
/// init's registry endpoint is its supervision endpoint, and `fs_resolve` installs
/// a sendable handle to any mount's backend in *any* caller's table
/// (system/kernel/process.zig), so after P2 every ring-3 process holds a handle it
/// can `ipc_send` into. Two payload bytes were once enough to reach `shutDown()`
/// from here — which stops every service and parks PID 1 in its final sleep,
/// destroying the registry for the rest of the boot, and does it for any process
/// that cares to ask.
///
/// The sender's task id is the fix, because it is not the sender's to choose: the
/// kernel stamps it into the badge's low bits as it copies the message into the
/// ring. Init is the registry, so it knows exactly which task holds `power`, and
/// that task alone is believed. A provider that dies and rebinds moves the
/// authorization with the binding; a name nothing holds authorizes nobody. Task
/// ids are never reused, so even a dead provider's id cannot be inherited.
///
/// (One task, not one process: the ACPI service is single-threaded and publishes
/// from the same task that bound the name. A threaded provider would want its
/// leader compared instead — which is a change to make when one appears, not a
/// looser rule to leave lying around for it.)
fn onPowerEvent(sender: u32, payload: []const u8) void {
const authorized = power_task orelse {
std.log.info("ignored a power event from pid {d}: nothing holds /protocol/power", .{sender});
return;
};
if (sender != authorized) {
std.log.info("ignored a power event from pid {d}: /protocol/power is pid {d}", .{ sender, authorized });
return;
}
// Read as an envelope packet, never by byte offset: the kind IS the packet's
// verb, so a power event is decoded exactly the way every other event in the
// system is. A packet whose operation is not one of this protocol's events —
// anything else that lands in this mailbox — decodes to null and is dropped.
const kind = power_protocol.Protocol.eventOf(payload) orelse return;
if (kind == .power_button) shutDown();
}
/// A supervised boot service died. Find which one and restart it — unless it exited
/// cleanly (it chose to stop, e.g. a driver with no hardware) or has hit the crash-loop
/// cap. Reclaiming the dead process is already the kernel's job (docs/process-lifecycle.md
/// iron rule 1); init only decides whether to bring it back.
fn restartChild(id: u32) void {
// Whatever it served, it serves no longer: the name goes back before the
// replacement asks for it, so the restarted instance binds rather than
// colliding with its own corpse.
unbindTask(id);
if (shutting_down) return; // deaths during the stop sequence are expected, not crashes
for (services[0..service_count], 0..) |*service, i| {
if (child_ids[i] != id) continue;
child_ids[i] = 0;
// An unknown reason (the record aged out) is treated as a crash worth restarting.
const reason = process.exitReason(id) orelse .fault;
if (reason == .exited) {
std.log.info("{s} exited cleanly; not restarting", .{service.path});
return;
}
restart_counts[i] += 1;
if (restart_counts[i] > maximum_restarts) {
std.log.info("{s} keeps crashing; giving up after {d} restarts", .{ service.path, maximum_restarts });
return;
}
std.log.info("{s} died ({s}); restarting ({d}/{d})", .{ service.path, @tagName(reason), restart_counts[i], maximum_restarts });
if (process.spawnSupervised(service.path, service.arguments(), supervision_endpoint)) |new_id| child_ids[i] = new_id;
return;
}
// An untracked child (e.g. the log-flush one-shot): nothing to restart.
}
/// Subscribe our endpoint (handed over as the call's capability) to the power
/// service, so events arrive as buffered messages here. init is the registry, so
/// it never resolves `/protocol/power` — it reads its own table, which is also
/// what makes this reachable at all: the subscription is armed by the bind that
/// put the endpoint there.
///
/// **This is the only place PID 1 blocks on another process, and it is the one
/// hazard the registrar has.** One thread serves both the namespace and this
/// call, so while it is outstanding init answers nobody: if the callee were
/// itself blocked asking init to resolve a name, the pair would never move. Two
/// things keep that from happening — the call is deferred to the next turn of
/// the loop (so the provider has its bind reply and is on its way to
/// `replyWait`), and the power provider resolves every name it needs *before* it
/// binds (system/services/acpi/acpi.zig, `manager_channel`). Any future service
/// init calls owes the same discipline.
fn subscribePower() void {
const handle = power_endpoint orelse return;
// The reserved `subscribe` verb: nothing but the header, with our own
// endpoint riding as the call's capability.
const header = envelope.Header{ .operation = envelope.operation_subscribe };
var reply: [power_protocol.message_maximum]u8 = undefined;
_ = ipc.callCap(handle, std.mem.asBytes(&header), &reply, supervision_endpoint) catch {};
}
/// The stop sequence: persist the log while storage is still up, then terminate
/// each child in reverse spawn order (vfs last — other services may flush through
/// it), waiting up to a deadline for each to exit before killing it, then ask the
/// power service to enter S5.
fn shutDown() void {
shutting_down = true; // the stop loop below kills children — those deaths aren't crashes
_ = logging.write("/system/services/init: shutting down\n");
// Log persistence is the logger service's job: it is the LAST boot service,
// so the reverse-order stop below terminates it first and its final drain
// runs while the whole storage chain is still alive.
var i = service_count;
while (i > 0) {
i -= 1;
if (child_ids[i] != 0) process.stop(child_ids[i], 2000, supervision_endpoint);
}
if (power_endpoint) |h| {
var packet: [power_protocol.message_maximum]u8 = undefined;
if (power_protocol.Protocol.encodeRequest(.shutdown, 0, {}, &.{}, &packet)) |framed| {
var reply: [power_protocol.message_maximum]u8 = undefined;
_ = ipc.call(h, framed, &reply) catch {};
}
}
// If S5 did not take, init has nothing left to do but idle.
while (true) time.sleepMillis(1000);
}