init: /protocol replaces the ServiceId registry

A protocol is reached by name now, not by a compile-time integer. Init is
PID 1 and already knows which binary it started, so init serves /protocol
as a vfs backend: bind claims a contract with the provider's endpoint
attached, open answers with that endpoint as the reply's capability, and
readdir lists what is bound with the task and binary behind it. The kernel
reserves the prefix — nothing may mount over it, under it, or unmount it —
and ServiceId, ipc_register and ipc_lookup are gone, their syscall numbers
left vacant.

A bind is authorized by who the caller *is*: the kernel-stamped binary
together with the supervising task's identity, matched against
/system/configuration/protocol.csv. Identity, not spelling — spawn is
ungated, so an attacker can run any bundled binary, and a name-only rule
would have let it launder grants through an init of its own making. A name
a live process holds is refused to everyone else; a dead one's is released.

Three review rounds against a hostile ring-3 process found what 108 green
tests could not, because the suite contains no attacker. Publishing init's
supervision endpoint as the registry put PID 1's mailbox in every process's
hands, where two forged bytes reached the shutdown path: privileged traffic
is now believed only from the task that holds the contract it speaks for.
A capability arriving on a request outlived every path that ignored it,
one handle per call until the table was full — in init, and in the harness
ten services share — so the arriving capability is owned by the turn and
released unless a handler says otherwise. And the kernel let anyone holding
an endpoint handle aim signals, timers, exit notices and interrupts at it:
binding now requires having created it.

Suite 108/108. The new protocol-registry case asserts eleven properties,
each one an attack that must fail.
This commit is contained in:
Daniel Samson
2026-08-01 02:39:07 +01:00
parent 1ff0991452
commit 1379b699f3
66 changed files with 2191 additions and 392 deletions
+6
View File
@@ -12,10 +12,15 @@ pub fn build(b: *std.Build) void {
const ipc = kernel.module("ipc");
const time = kernel.module("time");
// Every client reaches its service by name now: resolve `/protocol/<name>`,
// open it, and take the provider's endpoint out of the reply
// (docs/os-development/protocol-namespace.md).
const channel = kernel.module("channel");
_ = b.addModule("display-client", .{
.root_source_file = b.path("display/display-client.zig"),
.imports = &.{
.{ .name = "channel", .module = channel },
.{ .name = "ipc", .module = ipc },
.{ .name = "time", .module = time },
.{ .name = "display-protocol", .module = protocol.module("display-protocol") },
@@ -24,6 +29,7 @@ pub fn build(b: *std.Build) void {
_ = b.addModule("input-client", .{
.root_source_file = b.path("input/input-client.zig"),
.imports = &.{
.{ .name = "channel", .module = channel },
.{ .name = "ipc", .module = ipc },
.{ .name = "time", .module = time },
.{ .name = "input-protocol", .module = protocol.module("input-protocol") },
+5 -4
View File
@@ -1,9 +1,10 @@
//! User-space display client: talk to the display service (query the mode, and — from D3
//! — create layers, draw, and present) without hand-rolling the IPC. The `runtime.block`
//! shape: a cached `.display` lookup with a boot-race retry, then extern-struct request/
//! shape: a cached `/protocol/display` open with a boot-race retry, then extern-struct request/
//! reply marshalling. See system/services/display/ and docs/display.md.
const std = @import("std");
const channel = @import("channel");
const ipc = @import("ipc");
const time = @import("time");
const display_protocol = @import("display-protocol");
@@ -19,13 +20,13 @@ pub const Info = struct {
/// The service endpoint, looked up once and cached.
var handle: ?ipc.Handle = null;
/// Look up the display service, retrying while it comes up (a client races its
/// registration at boot). Returns the endpoint, or null if it never appears.
/// Open `/protocol/display`, retrying while it comes up (a client races the
/// service's bind at boot). Returns the endpoint, or null if it never appears.
fn service() ?ipc.Handle {
if (handle) |h| return h;
var attempts: usize = 0;
while (attempts < 100) : (attempts += 1) {
if (ipc.lookup(.display)) |h| {
if (channel.openEndpoint("display")) |h| {
handle = h;
return h;
}
+5 -5
View File
@@ -22,7 +22,7 @@
//! }
const std = @import("std");
const abi = @import("abi");
const channel = @import("channel");
const ipc = @import("ipc");
const time = @import("time");
const input_protocol = @import("input-protocol");
@@ -44,13 +44,13 @@ pub const device_mouse = input_protocol.device_mouse;
pub const device_joystick = input_protocol.device_joystick;
pub const device_all = input_protocol.device_all;
/// Look up the input service, retrying while it is still coming up. Both a subscriber and
/// a source race the service's registration at boot, so both wait for it here rather than
/// failing. Returns the service endpoint handle, or null if it never appears.
/// Open `/protocol/input`, retrying while it is still coming up. Both a subscriber and
/// a source race the service's bind at boot, so both wait for it here rather than
/// failing. Returns the provider's endpoint handle, or null if it never appears.
fn lookupService() ?ipc.Handle {
var attempts: usize = 0;
while (attempts < 100) : (attempts += 1) {
if (ipc.lookup(.input)) |handle| return handle;
if (channel.openEndpoint("input")) |handle| return handle;
time.sleepMillis(50);
}
return null;
+5 -4
View File
@@ -8,6 +8,7 @@
//! limit — the same handoff usb-storage uses toward the controller.
const std = @import("std");
const channel = @import("channel");
const ipc = @import("ipc");
const time = @import("time");
const block_protocol = @import("block-protocol");
@@ -66,14 +67,14 @@ pub const Device = struct {
}
};
/// One lookup attempt, no waiting — for a server that retries on its own
/// One open attempt, no waiting — for a server that retries on its own
/// timer (the fat service) instead of blocking its harness in here.
pub fn tryOpen() ?Device {
if (ipc.lookup(.block)) |handle| return .{ .endpoint = handle };
if (channel.openEndpoint("block")) |handle| return .{ .endpoint = handle };
return null;
}
/// Look up the block device, retrying generously while the USB storage chain
/// Open `/protocol/block`, retrying generously while the USB storage chain
/// (controller reset, enumeration, mass-storage bring-up) comes up.
pub fn open() ?Device {
// Patient: the whole USB storage chain (firmware discovery, xHCI reset and
@@ -84,7 +85,7 @@ pub fn open() ?Device {
// completed at ~24 s); a machine whose stick genuinely failed setup should
// not sit a further minute pretending otherwise.
while (attempts < 600) : (attempts += 1) {
if (ipc.lookup(.block)) |handle| return .{ .endpoint = handle };
if (channel.openEndpoint("block")) |handle| return .{ .endpoint = handle };
time.sleepMillis(50);
}
return null;
+7
View File
@@ -14,6 +14,10 @@ pub fn build(b: *std.Build) void {
const system_call = kernel.module("system-call");
const ipc = kernel.module("ipc");
const time = kernel.module("time");
// A driver finds the bus it attaches to by name — `/protocol/device-manager`,
// `/protocol/usb-transfer`, `/protocol/block`
// (docs/os-development/protocol-namespace.md).
const channel = kernel.module("channel");
// The devices sub-project's public interface (the flat wire types),
// importable by user space, unlike the kernel-internal device model it
@@ -55,6 +59,7 @@ pub fn build(b: *std.Build) void {
.root_source_file = b.path("driver/driver.zig"),
.imports = &.{
.{ .name = "abi", .module = abi },
.{ .name = "channel", .module = channel },
.{ .name = "device-abi", .module = device_abi },
.{ .name = "system-call", .module = system_call },
.{ .name = "ipc", .module = ipc },
@@ -81,6 +86,7 @@ pub fn build(b: *std.Build) void {
_ = b.addModule("usb", .{
.root_source_file = b.path("usb/usb.zig"),
.imports = &.{
.{ .name = "channel", .module = channel },
.{ .name = "ipc", .module = ipc },
.{ .name = "time", .module = time },
.{ .name = "usb-transfer-protocol", .module = protocol.module("usb-transfer-protocol") },
@@ -92,6 +98,7 @@ pub fn build(b: *std.Build) void {
_ = b.addModule("block", .{
.root_source_file = b.path("block/block.zig"),
.imports = &.{
.{ .name = "channel", .module = channel },
.{ .name = "ipc", .module = ipc },
.{ .name = "time", .module = time },
.{ .name = "block-protocol", .module = protocol.module("block-protocol") },
+2 -1
View File
@@ -7,6 +7,7 @@ const std = @import("std");
const abi = @import("abi");
const device_abi = @import("device-abi");
const sc = @import("system-call");
const channel = @import("channel");
const ipc = @import("ipc");
const time = @import("time");
const device_manager_protocol = @import("device-manager-protocol");
@@ -170,7 +171,7 @@ const lookup_pause_ms: u64 = 20;
pub fn hello(role: Role, device_id: u64) ?ipc.Handle {
var attempts: u32 = 0;
const manager = while (attempts < lookup_attempts) : (attempts += 1) {
if (ipc.lookup(.device_manager)) |handle| break handle;
if (channel.openEndpoint("device-manager")) |handle| break handle;
time.sleepMillis(lookup_pause_ms);
} else {
std.log.info("no device manager to hello", .{});
+7 -4
View File
@@ -16,6 +16,7 @@
//! the service harness drops buffered-message payloads — see service.zig).
const std = @import("std");
const channel = @import("channel");
const ipc = @import("ipc");
const time = @import("time");
const usb_transfer_protocol = @import("usb-transfer-protocol");
@@ -130,13 +131,15 @@ pub const Device = struct {
}
};
/// Look up the USB bus and open the device with the assigned id, handing over a
/// freshly created endpoint for asynchronous interrupt reports. Retries while the
/// bus is still coming up (a class driver races the bus driver at boot).
/// Open `/protocol/usb-transfer` and, on that channel, open the device with the
/// assigned id, handing over a freshly created endpoint for asynchronous interrupt
/// reports. Retries while the bus is still coming up (a class driver races the bus
/// driver at boot). Two opens, deliberately: the first names the contract, the
/// second names an object within it.
pub fn open(device_id: u64) ?Device {
var attempts: usize = 0;
const bus = while (attempts < 100) : (attempts += 1) {
if (ipc.lookup(.usb_bus)) |handle| break handle;
if (channel.openEndpoint("usb-transfer")) |handle| break handle;
time.sleepMillis(20);
} else return null;
+7 -2
View File
@@ -65,10 +65,11 @@ pub fn build(b: *std.Build) void {
// it needs the namespace (file-system, to resolve a /protocol name) and the
// transport (ipc) both, which is why it lives here rather than in a protocol
// module — those import nothing.
_ = b.addModule("channel", .{
const channel = b.addModule("channel", .{
.root_source_file = b.path("channel.zig"),
.imports = &.{
.{ .name = "ipc", .module = ipc },
.{ .name = "time", .module = time },
.{ .name = "file-system", .module = file_system },
.{ .name = "vfs-protocol", .module = protocol.module("vfs-protocol") },
.{ .name = "envelope", .module = protocol.module("envelope") },
@@ -83,10 +84,13 @@ pub fn build(b: *std.Build) void {
.{ .name = "thread", .module = thread },
},
});
// The harness binds the service's contract name at startup, which is a
// conversation with the registry — hence channel (and time, for the patience
// a provider that beat init to the mount needs).
_ = b.addModule("service", .{
.root_source_file = b.path("service.zig"),
.imports = &.{
.{ .name = "abi", .module = abi },
.{ .name = "channel", .module = channel },
.{ .name = "ipc", .module = ipc },
.{ .name = "process", .module = process },
},
@@ -124,6 +128,7 @@ pub fn build(b: *std.Build) void {
.target = b.resolveTargetQuery(.{}),
.imports = &.{
.{ .name = "ipc", .module = ipc },
.{ .name = "time", .module = time },
.{ .name = "file-system", .module = file_system },
.{ .name = "vfs-protocol", .module = protocol.module("vfs-protocol") },
.{ .name = "envelope", .module = protocol.module("envelope") },
+162 -39
View File
@@ -19,14 +19,14 @@
//! falls out of the naming layer for free; no protocol needs a reconnect verb.
//!
//! `open` resolves a `/protocol/<name>` path through the kernel VFS router and
//! takes the provider's endpoint from the open reply's capability. **No registry
//! is mounted at `/protocol` yet** — init grows one in P2
//! (docs/os-development/protocol-namespace.md) — so `open` cannot succeed today;
//! it is here so the client half of the conversation is written once, against
//! the shape the registry will answer with.
//! takes the provider's endpoint from the open reply's capability. The registry
//! answering it is init, PID 1, which mounts `/protocol` before it spawns anyone
//! (docs/os-development/protocol-namespace.md); `bind` below is the other half —
//! how a provider claims the name in the first place.
const std = @import("std");
const ipc = @import("ipc");
const time = @import("time");
const file_system = @import("file-system");
const vfs_protocol = @import("vfs-protocol");
const envelope = @import("envelope");
@@ -36,6 +36,15 @@ const envelope = @import("envelope");
/// on the stack of whoever opens.
pub const path_maximum: usize = 224;
/// Where the protocol namespace is rooted — the one path prefix in the system
/// that names contracts rather than files. Spelled once, here, so no caller
/// builds it by hand (docs/file-system-development/file-system-hierarchy.md).
pub const root: []const u8 = "/protocol";
/// Longest contract name — the part after `/protocol/`. Short by construction:
/// a leaf like `display`, or a subtree leaf like `test/shared-memory`.
pub const name_maximum: usize = 64;
/// What a `call` came back with: the provider's status, the reply payload (the
/// bytes after the `Status`, in the caller's own buffer), and any capability the
/// reply carried.
@@ -74,41 +83,13 @@ pub const Channel = struct {
/// The path is spoken exactly once, here. Everything afterwards is integers
/// in the packet header.
pub fn open(path: []const u8) ?Channel {
var relative: [path_maximum]u8 = undefined;
const route = file_system.fsResolve(path, 0, &relative) orelse return null;
// A protocol name must land on a mounted backend. The kernel-served
// `/system` tree answers with a node token and knows nothing of
// channels, so that route is simply the wrong path.
const backend = switch (route) {
.kernel => return null,
.backend => |b| b,
};
// The registry handle is deduplicated by the kernel across resolves and
// shared with every other user of that mount, so it is not ours to
// close — only the provider endpoint below belongs to this channel.
const name = relative[0..backend.path_len];
return .{ .endpoint = openPath(path) orelse return null };
}
var request: [vfs_protocol.message_maximum]u8 = undefined;
const header = vfs_protocol.Request{
.operation = .open,
.node = 0,
.offset = 0,
.len = @intCast(name.len),
.flags = 0,
};
if (vfs_protocol.request_size + name.len > request.len) return null;
@memcpy(request[0..vfs_protocol.request_size], std.mem.asBytes(&header));
@memcpy(request[vfs_protocol.request_size..][0..name.len], name);
var reply: [vfs_protocol.message_maximum]u8 = undefined;
const answer = ipc.callCap(backend.handle, request[0 .. vfs_protocol.request_size + name.len], &reply, null) catch return null;
if (answer.len < vfs_protocol.reply_size) return null;
const decoded = std.mem.bytesToValue(vfs_protocol.Reply, reply[0..vfs_protocol.reply_size]);
if (decoded.status != 0) return null;
// The capability *is* the channel — an open that succeeds without one
// was answered by a file backend, which does not speak protocols.
const provider = answer.cap orelse return null;
return .{ .endpoint = provider };
/// Establish a channel by contract name — `open` with `/protocol/` supplied,
/// which is how every caller in the system spells it.
pub fn connect(name: []const u8) ?Channel {
return .{ .endpoint = openEndpoint(name) orelse return null };
}
/// Send one request packet and block for the reply: `[Header][request]` out,
@@ -178,6 +159,140 @@ pub const Channel = struct {
}
};
// --- the namespace: resolving, opening, and claiming a contract name ---------
/// Where a `/protocol/...` path routed: the registry's endpoint, plus the path
/// rewritten mount-relative (`/display` for `/protocol/display`). The handle is
/// deduplicated by the kernel across resolves and shared with every other user
/// of that mount, so it is never ours to close.
const Registry = struct {
handle: ipc.Handle,
relative: [path_maximum]u8,
relative_len: usize,
fn path(self: *const Registry) []const u8 {
return self.relative[0..self.relative_len];
}
};
/// Route `path` to whatever backend serves it. Null when nothing is mounted
/// there — under `/protocol` that means the registry is not up yet, which is a
/// *retry*, not a refusal. A kernel-served route (the read-only `/system` tree)
/// is the wrong path, not a channel, and is refused here.
fn reach(path: []const u8) ?Registry {
var out: Registry = .{ .handle = 0, .relative = undefined, .relative_len = 0 };
const route = file_system.fsResolve(path, 0, &out.relative) orelse return null;
switch (route) {
.kernel => return null,
.backend => |b| {
out.handle = b.handle;
out.relative_len = b.path_len;
return out;
},
}
}
/// One vfs-protocol round trip at a backend: fixed header, inline payload, and
/// an optional capability in each direction.
fn transact(
handle: ipc.Handle,
operation: vfs_protocol.Operation,
payload: []const u8,
send_capability: ?ipc.Handle,
) ?struct { reply: vfs_protocol.Reply, capability: ?ipc.Handle } {
var request: [vfs_protocol.message_maximum]u8 = undefined;
if (vfs_protocol.request_size + payload.len > request.len) return null;
const header = vfs_protocol.Request{
.operation = operation,
.node = 0,
.offset = 0,
.len = @intCast(payload.len),
.flags = 0,
};
@memcpy(request[0..vfs_protocol.request_size], std.mem.asBytes(&header));
@memcpy(request[vfs_protocol.request_size..][0..payload.len], payload);
var reply: [vfs_protocol.message_maximum]u8 = undefined;
const answer = ipc.callCap(handle, request[0 .. vfs_protocol.request_size + payload.len], &reply, send_capability) catch return null;
if (answer.len < vfs_protocol.reply_size) return null;
return .{
.reply = std.mem.bytesToValue(vfs_protocol.Reply, reply[0..vfs_protocol.reply_size]),
.capability = answer.cap,
};
}
/// Resolve an absolute `/protocol/...` path and take the provider's endpoint out
/// of the open reply's capability.
fn openPath(path: []const u8) ?ipc.Handle {
const registry = reach(path) orelse return null;
const answered = transact(registry.handle, .open, registry.path(), null) orelse return null;
if (answered.reply.status != 0) return null;
// The capability *is* the channel — an open that succeeds without one was
// answered by a file backend, which does not speak protocols.
return answered.capability;
}
/// The provider's raw endpoint behind `/protocol/<name>`. The transitional form,
/// for the clients that still marshal their protocol's bytes by hand; P4 moves
/// them onto `Channel` proper and this shrinks back to `connect`.
///
/// Null covers both "no such contract" and "you may not have it" — deliberately
/// the same answer (protocol-namespace.md: enforcement is absence), and also
/// "the registry is not mounted yet", which is why every caller retries.
pub fn openEndpoint(name: []const u8) ?ipc.Handle {
var path: [path_maximum]u8 = undefined;
const full = join(name, &path) orelse return null;
return openPath(full);
}
/// Claim `/protocol/<name>` for `endpoint`: the registry records the name
/// against this process and hands the endpoint to whoever opens it afterwards.
/// The endpoint rides the call as its capability, the one direction-crossing
/// move kernel-ipc offers.
///
/// Three-valued on purpose. **Null** is "the registry could not be reached" —
/// it is not mounted yet, which happens when a provider starts before init has
/// finished coming up, and the answer is to retry. A **value** is the registry's
/// verdict and is final: 0 bound, `-EPERM` this binary is not granted that name,
/// `-EBUSY` a live provider already holds it.
pub fn bind(name: []const u8, endpoint: ipc.Handle) ?i32 {
const registry = reach(root) orelse return null;
const answered = transact(registry.handle, .bind, name, endpoint) orelse return null;
return answered.reply.status;
}
/// How long a provider keeps offering itself before giving up. The registry is
/// init, which mounts `/protocol` before it spawns anyone, so in a normal boot
/// the first try lands; a provider the kernel test harness starts may well beat
/// init to the mount, which is what the patience is for. Four seconds of 20 ms
/// tries — the same cadence every client in the tree spends finding a service.
const bind_attempts: u32 = 200;
const bind_retry_ms: u64 = 20;
/// `bind`, waiting out a registry that is not mounted yet. Only unreachability
/// is retried: a registry that *answered* has decided, and asking again cannot
/// change its mind. True when the name is ours.
pub fn bindPatiently(name: []const u8, endpoint: ipc.Handle) bool {
var attempt: u32 = 0;
while (attempt < bind_attempts) : (attempt += 1) {
if (bind(name, endpoint)) |status| return status == 0;
time.sleepMillis(bind_retry_ms);
}
return false;
}
/// `/protocol/` + `name`, in the caller's buffer. Null if the name is empty or
/// longer than the namespace admits.
fn join(name: []const u8, buffer: []u8) ?[]u8 {
if (name.len == 0 or name.len > name_maximum) return null;
const total = root.len + 1 + name.len;
if (total > buffer.len) return null;
@memcpy(buffer[0..root.len], root);
buffer[root.len] = '/';
@memcpy(buffer[root.len + 1 ..][0..name.len], name);
return buffer[0..total];
}
/// Lay a packet down: the folded header first, then the protocol's bytes. Null
/// when it would not fit the buffer — the same rule as `envelope`'s framing,
/// applied where the buffer is the transport's, not the protocol's.
@@ -209,6 +324,14 @@ test "a framed packet is the header followed by the protocol's bytes" {
try testing.expectEqualStrings("body", packet[envelope.prefix_size..]);
}
test "a contract name joins the namespace root exactly once" {
var buffer: [path_maximum]u8 = undefined;
try testing.expectEqualStrings("/protocol/display", join("display", &buffer).?);
try testing.expectEqualStrings("/protocol/test/shared-memory", join("test/shared-memory", &buffer).?);
try testing.expect(join("", &buffer) == null);
try testing.expect(join("x" ** (name_maximum + 1), &buffer) == null);
}
test "framing refuses a packet that would not fit rather than truncating it" {
var post: [envelope.post_maximum]u8 = undefined;
const header = envelope.Header{ .operation = envelope.first_protocol_operation };
+54 -11
View File
@@ -29,10 +29,10 @@ pub fn createIpcEndpoint() ?Handle {
return if (failed(r)) null else r;
}
/// Publish endpoint `h` under a well-known service id so other processes find it.
pub fn register(id: abi.ServiceId, h: Handle) bool {
return !failed(sc.systemCall2(.ipc_register, @intFromEnum(id), h));
}
// `register`/`lookup` lived here — the two wrappers over the flat ServiceId
// registry. Naming is not a system call any more: a provider binds its contract
// name at the registry and a client resolves and opens `/protocol/<name>`, both
// through `channel` (docs/os-development/protocol-namespace.md).
/// Drop a capability handle (endpoint, shared-memory, or DMA-region) and free its table
/// slot. A forwarding hop closes a cap it passed on; a binder closes a DMA-region cap
@@ -42,13 +42,6 @@ pub fn close(h: Handle) bool {
return !failed(sc.systemCall1(.handle_close, h));
}
/// Find the endpoint published under `id`, installing a handle to it in this
/// process.
pub fn lookup(id: abi.ServiceId) ?Handle {
const r = sc.systemCall1(.ipc_lookup, @intFromEnum(id));
return if (failed(r)) null else r;
}
pub const CallError = error{Failed};
/// The result of a capability-passing `callCap`: the reply length, and the handle of
@@ -178,6 +171,56 @@ pub const Received = struct {
}
};
/// A capability that arrived with one turn of a receive loop, and the ownership
/// rule for it: **the turn owns it until a handler takes it, and closes whatever
/// is left.**
///
/// The kernel installs a sent capability in the receiver's handle table whenever
/// the caller attached one, *independent of the message's length or kind*
/// (system/kernel/ipc-synchronous.zig `replyWait`), so every path out of a loop
/// has to dispose of one — including the paths that never look at the message.
/// The table is thirty-two slots, and `ipc_call` does not dedupe, so a client
/// looping on `callCap(server, &.{}, endpoint)` spends one slot per call: about
/// thirty-two zero-length pings and the service can never accept another
/// capability, which means no subscribe and no shared-memory handover, for the
/// rest of the boot. It is unauthenticated and it is two lines to write.
///
/// So ownership is structural rather than a close per branch — the per-branch
/// version has already failed twice in this tree, in PID 1's ping path and in
/// every `service.run` callback that simply ignored its capability argument.
/// Written this way, forgetting **closes**, and *keeping* a capability is the
/// thing a handler has to say out loud:
///
/// ```zig
/// var arrived: ipc.Arrival = .{ .handle = got.cap };
/// defer arrived.release(); // every exit path, including `continue`
/// ...
/// const kept = arrived.take().?; // claimed: mine to hold or close
/// ```
pub const Arrival = struct {
handle: ?Handle = null,
/// Look without claiming — a handler that may still refuse wants no close of
/// its own on the refusal paths.
pub fn peek(self: *const Arrival) ?Handle {
return self.handle;
}
/// Claim ownership: from here the capability is the taker's to keep or close,
/// and the turn will not touch it.
pub fn take(self: *Arrival) ?Handle {
defer self.handle = null;
return self.handle;
}
/// Close whatever nobody claimed. Idempotent, so it is safe as a `defer` next
/// to any number of `take`s.
pub fn release(self: *Arrival) void {
if (self.handle) |handle| _ = close(handle);
self.handle = null;
}
};
/// Server side of IPC_ReplyWait: deliver `reply` to the client last received (if any,
/// optionally handing it `send_cap`), then block until the next request arrives in
/// `receive`. Returns its length, the sender badge, and any capability the request
+12
View File
@@ -142,6 +142,18 @@ pub fn subscribeExits(endpoint: usize) bool {
/// snapshot buffer without importing `abi` itself.
pub const ProcessDescriptor = abi.ProcessDescriptor;
/// The calling task's own kernel id — its row in the process table, and the value
/// every other process sees as this one's `supervisor` after it spawns them. For a
/// single-threaded program that is its process id; in a threaded one it is the
/// calling thread's id (`Thread.getCurrentId` is the same system call, named for
/// the threading vocabulary). Ids are monotonic and never reused
/// (system/kernel/process.zig), which is what makes comparing one an identity
/// test where comparing a *name* is only a resemblance test — the registrar in
/// init leans on exactly that.
pub fn taskId() u32 {
return @intCast(sc.systemCall0(.thread_self));
}
/// Give up the rest of this quantum.
pub fn yield() void {
_ = sc.systemCall0(.yield);
+50 -16
View File
@@ -6,13 +6,19 @@
//! loop chose, never on a hijacked stack — the whole reason signals are
//! messages.
//!
//! One rule a service author does have to know, and it is stated on
//! `Callbacks.on_message`: **a capability that arrives belongs to the turn** —
//! the loop closes it unless the callback claims it with `take()`. Forgetting is
//! therefore safe, and keeping is explicit; the opposite arrangement quietly
//! spends a handle-table slot per request.
//!
//! The liveness probe: a **zero-length request is the universal ping**, answered
//! with a zero-length reply by the harness itself. No protocol's requests start
//! at length zero, so the encoding cannot collide, and there is nothing for a
//! service author to implement — a wedged service simply fails to answer, which
//! is the diagnosis (see docs/ipc.md).
const abi = @import("abi");
const channel = @import("channel");
const ipc = @import("ipc");
const process = @import("process");
@@ -22,10 +28,22 @@ pub const Callbacks = struct {
/// Return false to abort startup (the process exits).
init: ?*const fn (endpoint: ipc.Handle) bool = null,
/// One protocol request from `sender` (a task id): write the reply into
/// `reply`, return its length. `capability` is the handle the request
/// carried, if any (M13 cap passing — how a subscriber hands over its
/// endpoint). The zero-length ping never reaches this.
on_message: *const fn (message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize,
/// `reply`, return its length. The zero-length ping never reaches this.
///
/// `arrived` is the capability the request carried (M13 cap passing — how a
/// subscriber hands over its endpoint), and it comes with **an ownership
/// rule: the turn owns it, and a handler that wants to keep it must say so
/// with `take()`.** Whatever is left when this returns, the loop closes.
/// `peek()` reads it without claiming, which is what a handler that may
/// still refuse wants — no close of its own on the refusal paths.
///
/// The rule is stated here, in the contract, because the alternative has
/// failed in practice: an implementation that simply ignored a `?ipc.Handle`
/// argument leaked a handle table slot per request, and every operation
/// except a subscribe ignores it. Thirty-two such requests — zero-length
/// pings will do, and they need no authorization — and the service can never
/// accept another capability for the rest of the boot. See `ipc.Arrival`.
on_message: *const fn (message: []const u8, reply: []u8, sender: u32, arrived: *ipc.Arrival) usize,
/// A notification that is not a signal — a subscribed exit event, a bound
/// IRQ, a timer landing. The raw badge; decode with the ipc helpers.
on_notification: ?*const fn (badge: u64) void = null,
@@ -35,19 +53,25 @@ pub const Callbacks = struct {
/// the return itself — never put *necessary* work here (iron rule 1: a kill
/// arrives with no warning; this is for graceful extras only).
on_terminate: ?*const fn () void = null,
/// Publish the endpoint under a well-known service id at startup.
service: ?abi.ServiceId = null,
/// The contract this service provides: a name under `/protocol`, mirroring
/// the `library/protocol/` module that defines the wire format — a program
/// imports `display-protocol` and the provider binds `"display"`
/// (docs/os-development/protocol-namespace.md). Bound at startup, before
/// `init` runs, so the service is reachable the moment it serves. A refusal
/// (not granted, or a live provider already holds the name) aborts startup.
service: ?[]const u8 = null,
};
/// Run the service: create and (optionally) register the endpoint, bind signals
/// to it, call `init`, then serve until `terminate` arrives — at which point the
/// loop returns and main's return is the clean exit the supervisor reads as
/// `ExitReason.exited`. `maximum_message` sizes the receive and reply buffers
/// (a service passes its protocol's message maximum).
/// Run the service: create the endpoint, bind it under the service's contract
/// name (if it has one), bind signals to it, call `init`, then serve until
/// `terminate` arrives — at which point the loop returns and main's return is
/// the clean exit the supervisor reads as `ExitReason.exited`.
/// `maximum_message` sizes the receive and reply buffers (a service passes its
/// protocol's message maximum).
pub fn run(comptime maximum_message: usize, callbacks: Callbacks) void {
const endpoint = ipc.createIpcEndpoint() orelse return;
if (callbacks.service) |id| {
if (!ipc.register(id, endpoint)) return;
if (callbacks.service) |name| {
if (!channel.bindPatiently(name, endpoint)) return;
}
_ = process.bindSignals(endpoint);
if (callbacks.init) |initialise| {
@@ -59,6 +83,16 @@ pub fn run(comptime maximum_message: usize, callbacks: Callbacks) void {
var receive: [maximum_message]u8 = undefined;
while (true) {
const got = ipc.replyWait(endpoint, reply_buffer[0..reply_len], &receive, null);
// Whatever capability came with this turn is the turn's, and the turn
// closes it unless a callback claims it (`ipc.Arrival`). Structural
// rather than a close per branch, because the branches are exactly what
// gets forgotten: the ping's `continue` below, and every `on_message`
// that has no use for a capability — which is every operation but a
// subscribe. A `defer` in a loop body runs on `continue` and on the
// `return` that ends the loop, so this covers all four exits.
var arrived: ipc.Arrival = .{ .handle = got.cap };
defer arrived.release();
if (got.isNotification()) {
reply_len = 0; // nothing owed for a notification
if (process.signalsFrom(got.badge)) |signals| {
@@ -76,8 +110,8 @@ pub fn run(comptime maximum_message: usize, callbacks: Callbacks) void {
}
if (got.len == 0) {
reply_len = 0; // the universal ping: a zero-length reply, from the harness
continue;
continue; // any capability it carried goes out through the turn's `defer`
}
reply_len = callbacks.on_message(receive[0..got.len], &reply_buffer, got.senderTaskId(), got.cap);
reply_len = callbacks.on_message(receive[0..got.len], &reply_buffer, got.senderTaskId(), &arrived);
}
}
+9
View File
@@ -119,6 +119,15 @@ pub fn fitsPost(comptime T: type) bool {
/// Positive here, sent negated in `Status.status`, as the kernel spells it.
pub const ENOSYS: i32 = 10; // this protocol has no such operation
pub const EPROTO: i32 = 11; // malformed packet: shorter than the verb it names
pub const EBUSY: i32 = 12; // the thing asked for is held by someone still alive
/// Restated from the kernel's half of the numbering, because a provider refuses
/// too and userspace has no other place to read these from: `ENOENT` is "no such
/// name", `EPERM` "not permitted". The protocol registry answers an ungranted
/// bind with the second and a name a live provider already holds with `EBUSY`.
pub const ENOENT: i32 = 4;
pub const ENOSPC: i32 = 5;
pub const EPERM: i32 = 9;
// --- framing ----------------------------------------------------------------
+5 -3
View File
@@ -1,7 +1,9 @@
//! The power protocol (docs/power.md): system power's domain-named surface,
//! registered under `ServiceId.power`. On x86 the acpi service serves it; on
//! ARM a PSCI/mailbox service will register the same id — subscribers never
//! learn which firmware they are on (docs/discovery.md — firmware neutrality).
//! bound at `/protocol/power`. On x86 the acpi service provides it; on ARM a
//! PSCI/mailbox service will bind the same name — subscribers never learn which
//! firmware they are on (docs/discovery.md — firmware neutrality), which is the
//! whole point of naming the contract rather than the provider
//! (docs/os-development/protocol-namespace.md).
//! The vfs-protocol pattern: extern-struct messages, a version, reserved fields.
/// The protocol version a client states nowhere yet — reserved for the day a
+9
View File
@@ -40,6 +40,12 @@ pub const Operation = enum(u32) {
// rename: the payload is the old path, a single 0x00 separator, then the new
// path. Same-directory rename only (the router requires both under one mount).
rename, // rename(old\0new payload) -> status
// Appended for the protocol namespace (P2). The registry is a synthetic
// backend mounted at /protocol: `open` establishes a channel and `readdir`
// lists the bound names like any directory, so those two verbs need nothing
// new — but *claiming* a name does. A file backend refuses it, alongside the
// router verbs it does not implement either; only the registry implements it.
bind, // bind(name payload, capability = the provider's endpoint) -> status
};
/// The type of a filesystem node, aligned to the node-kind table
@@ -132,4 +138,7 @@ test "protocol struct sizes and node kinds" {
try std.testing.expectEqual(@as(u32, 0), @intFromEnum(Operation.open));
try std.testing.expectEqual(@as(u32, 4), @intFromEnum(Operation.status));
try std.testing.expectEqual(@as(u32, 5), @intFromEnum(Operation.readdir));
try std.testing.expectEqual(@as(u32, 10), @intFromEnum(Operation.rename));
// The registry's claim verb, appended last with the protocol namespace.
try std.testing.expectEqual(@as(u32, 11), @intFromEnum(Operation.bind));
}