init: /protocol replaces the ServiceId registry
A protocol is reached by name now, not by a compile-time integer. Init is PID 1 and already knows which binary it started, so init serves /protocol as a vfs backend: bind claims a contract with the provider's endpoint attached, open answers with that endpoint as the reply's capability, and readdir lists what is bound with the task and binary behind it. The kernel reserves the prefix — nothing may mount over it, under it, or unmount it — and ServiceId, ipc_register and ipc_lookup are gone, their syscall numbers left vacant. A bind is authorized by who the caller *is*: the kernel-stamped binary together with the supervising task's identity, matched against /system/configuration/protocol.csv. Identity, not spelling — spawn is ungated, so an attacker can run any bundled binary, and a name-only rule would have let it launder grants through an init of its own making. A name a live process holds is refused to everyone else; a dead one's is released. Three review rounds against a hostile ring-3 process found what 108 green tests could not, because the suite contains no attacker. Publishing init's supervision endpoint as the registry put PID 1's mailbox in every process's hands, where two forged bytes reached the shutdown path: privileged traffic is now believed only from the task that holds the contract it speaks for. A capability arriving on a request outlived every path that ignored it, one handle per call until the table was full — in init, and in the harness ten services share — so the arriving capability is owned by the turn and released unless a handler says otherwise. And the kernel let anyone holding an endpoint handle aim signals, timers, exit notices and interrupts at it: binding now requires having created it. Suite 108/108. The new protocol-registry case asserts eleven properties, each one an attack that must fail.
This commit is contained in:
+162
-39
@@ -19,14 +19,14 @@
|
||||
//! falls out of the naming layer for free; no protocol needs a reconnect verb.
|
||||
//!
|
||||
//! `open` resolves a `/protocol/<name>` path through the kernel VFS router and
|
||||
//! takes the provider's endpoint from the open reply's capability. **No registry
|
||||
//! is mounted at `/protocol` yet** — init grows one in P2
|
||||
//! (docs/os-development/protocol-namespace.md) — so `open` cannot succeed today;
|
||||
//! it is here so the client half of the conversation is written once, against
|
||||
//! the shape the registry will answer with.
|
||||
//! takes the provider's endpoint from the open reply's capability. The registry
|
||||
//! answering it is init, PID 1, which mounts `/protocol` before it spawns anyone
|
||||
//! (docs/os-development/protocol-namespace.md); `bind` below is the other half —
|
||||
//! how a provider claims the name in the first place.
|
||||
|
||||
const std = @import("std");
|
||||
const ipc = @import("ipc");
|
||||
const time = @import("time");
|
||||
const file_system = @import("file-system");
|
||||
const vfs_protocol = @import("vfs-protocol");
|
||||
const envelope = @import("envelope");
|
||||
@@ -36,6 +36,15 @@ const envelope = @import("envelope");
|
||||
/// on the stack of whoever opens.
|
||||
pub const path_maximum: usize = 224;
|
||||
|
||||
/// Where the protocol namespace is rooted — the one path prefix in the system
|
||||
/// that names contracts rather than files. Spelled once, here, so no caller
|
||||
/// builds it by hand (docs/file-system-development/file-system-hierarchy.md).
|
||||
pub const root: []const u8 = "/protocol";
|
||||
|
||||
/// Longest contract name — the part after `/protocol/`. Short by construction:
|
||||
/// a leaf like `display`, or a subtree leaf like `test/shared-memory`.
|
||||
pub const name_maximum: usize = 64;
|
||||
|
||||
/// What a `call` came back with: the provider's status, the reply payload (the
|
||||
/// bytes after the `Status`, in the caller's own buffer), and any capability the
|
||||
/// reply carried.
|
||||
@@ -74,41 +83,13 @@ pub const Channel = struct {
|
||||
/// The path is spoken exactly once, here. Everything afterwards is integers
|
||||
/// in the packet header.
|
||||
pub fn open(path: []const u8) ?Channel {
|
||||
var relative: [path_maximum]u8 = undefined;
|
||||
const route = file_system.fsResolve(path, 0, &relative) orelse return null;
|
||||
// A protocol name must land on a mounted backend. The kernel-served
|
||||
// `/system` tree answers with a node token and knows nothing of
|
||||
// channels, so that route is simply the wrong path.
|
||||
const backend = switch (route) {
|
||||
.kernel => return null,
|
||||
.backend => |b| b,
|
||||
};
|
||||
// The registry handle is deduplicated by the kernel across resolves and
|
||||
// shared with every other user of that mount, so it is not ours to
|
||||
// close — only the provider endpoint below belongs to this channel.
|
||||
const name = relative[0..backend.path_len];
|
||||
return .{ .endpoint = openPath(path) orelse return null };
|
||||
}
|
||||
|
||||
var request: [vfs_protocol.message_maximum]u8 = undefined;
|
||||
const header = vfs_protocol.Request{
|
||||
.operation = .open,
|
||||
.node = 0,
|
||||
.offset = 0,
|
||||
.len = @intCast(name.len),
|
||||
.flags = 0,
|
||||
};
|
||||
if (vfs_protocol.request_size + name.len > request.len) return null;
|
||||
@memcpy(request[0..vfs_protocol.request_size], std.mem.asBytes(&header));
|
||||
@memcpy(request[vfs_protocol.request_size..][0..name.len], name);
|
||||
|
||||
var reply: [vfs_protocol.message_maximum]u8 = undefined;
|
||||
const answer = ipc.callCap(backend.handle, request[0 .. vfs_protocol.request_size + name.len], &reply, null) catch return null;
|
||||
if (answer.len < vfs_protocol.reply_size) return null;
|
||||
const decoded = std.mem.bytesToValue(vfs_protocol.Reply, reply[0..vfs_protocol.reply_size]);
|
||||
if (decoded.status != 0) return null;
|
||||
// The capability *is* the channel — an open that succeeds without one
|
||||
// was answered by a file backend, which does not speak protocols.
|
||||
const provider = answer.cap orelse return null;
|
||||
return .{ .endpoint = provider };
|
||||
/// Establish a channel by contract name — `open` with `/protocol/` supplied,
|
||||
/// which is how every caller in the system spells it.
|
||||
pub fn connect(name: []const u8) ?Channel {
|
||||
return .{ .endpoint = openEndpoint(name) orelse return null };
|
||||
}
|
||||
|
||||
/// Send one request packet and block for the reply: `[Header][request]` out,
|
||||
@@ -178,6 +159,140 @@ pub const Channel = struct {
|
||||
}
|
||||
};
|
||||
|
||||
// --- the namespace: resolving, opening, and claiming a contract name ---------
|
||||
|
||||
/// Where a `/protocol/...` path routed: the registry's endpoint, plus the path
|
||||
/// rewritten mount-relative (`/display` for `/protocol/display`). The handle is
|
||||
/// deduplicated by the kernel across resolves and shared with every other user
|
||||
/// of that mount, so it is never ours to close.
|
||||
const Registry = struct {
|
||||
handle: ipc.Handle,
|
||||
relative: [path_maximum]u8,
|
||||
relative_len: usize,
|
||||
|
||||
fn path(self: *const Registry) []const u8 {
|
||||
return self.relative[0..self.relative_len];
|
||||
}
|
||||
};
|
||||
|
||||
/// Route `path` to whatever backend serves it. Null when nothing is mounted
|
||||
/// there — under `/protocol` that means the registry is not up yet, which is a
|
||||
/// *retry*, not a refusal. A kernel-served route (the read-only `/system` tree)
|
||||
/// is the wrong path, not a channel, and is refused here.
|
||||
fn reach(path: []const u8) ?Registry {
|
||||
var out: Registry = .{ .handle = 0, .relative = undefined, .relative_len = 0 };
|
||||
const route = file_system.fsResolve(path, 0, &out.relative) orelse return null;
|
||||
switch (route) {
|
||||
.kernel => return null,
|
||||
.backend => |b| {
|
||||
out.handle = b.handle;
|
||||
out.relative_len = b.path_len;
|
||||
return out;
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
/// One vfs-protocol round trip at a backend: fixed header, inline payload, and
|
||||
/// an optional capability in each direction.
|
||||
fn transact(
|
||||
handle: ipc.Handle,
|
||||
operation: vfs_protocol.Operation,
|
||||
payload: []const u8,
|
||||
send_capability: ?ipc.Handle,
|
||||
) ?struct { reply: vfs_protocol.Reply, capability: ?ipc.Handle } {
|
||||
var request: [vfs_protocol.message_maximum]u8 = undefined;
|
||||
if (vfs_protocol.request_size + payload.len > request.len) return null;
|
||||
const header = vfs_protocol.Request{
|
||||
.operation = operation,
|
||||
.node = 0,
|
||||
.offset = 0,
|
||||
.len = @intCast(payload.len),
|
||||
.flags = 0,
|
||||
};
|
||||
@memcpy(request[0..vfs_protocol.request_size], std.mem.asBytes(&header));
|
||||
@memcpy(request[vfs_protocol.request_size..][0..payload.len], payload);
|
||||
|
||||
var reply: [vfs_protocol.message_maximum]u8 = undefined;
|
||||
const answer = ipc.callCap(handle, request[0 .. vfs_protocol.request_size + payload.len], &reply, send_capability) catch return null;
|
||||
if (answer.len < vfs_protocol.reply_size) return null;
|
||||
return .{
|
||||
.reply = std.mem.bytesToValue(vfs_protocol.Reply, reply[0..vfs_protocol.reply_size]),
|
||||
.capability = answer.cap,
|
||||
};
|
||||
}
|
||||
|
||||
/// Resolve an absolute `/protocol/...` path and take the provider's endpoint out
|
||||
/// of the open reply's capability.
|
||||
fn openPath(path: []const u8) ?ipc.Handle {
|
||||
const registry = reach(path) orelse return null;
|
||||
const answered = transact(registry.handle, .open, registry.path(), null) orelse return null;
|
||||
if (answered.reply.status != 0) return null;
|
||||
// The capability *is* the channel — an open that succeeds without one was
|
||||
// answered by a file backend, which does not speak protocols.
|
||||
return answered.capability;
|
||||
}
|
||||
|
||||
/// The provider's raw endpoint behind `/protocol/<name>`. The transitional form,
|
||||
/// for the clients that still marshal their protocol's bytes by hand; P4 moves
|
||||
/// them onto `Channel` proper and this shrinks back to `connect`.
|
||||
///
|
||||
/// Null covers both "no such contract" and "you may not have it" — deliberately
|
||||
/// the same answer (protocol-namespace.md: enforcement is absence), and also
|
||||
/// "the registry is not mounted yet", which is why every caller retries.
|
||||
pub fn openEndpoint(name: []const u8) ?ipc.Handle {
|
||||
var path: [path_maximum]u8 = undefined;
|
||||
const full = join(name, &path) orelse return null;
|
||||
return openPath(full);
|
||||
}
|
||||
|
||||
/// Claim `/protocol/<name>` for `endpoint`: the registry records the name
|
||||
/// against this process and hands the endpoint to whoever opens it afterwards.
|
||||
/// The endpoint rides the call as its capability, the one direction-crossing
|
||||
/// move kernel-ipc offers.
|
||||
///
|
||||
/// Three-valued on purpose. **Null** is "the registry could not be reached" —
|
||||
/// it is not mounted yet, which happens when a provider starts before init has
|
||||
/// finished coming up, and the answer is to retry. A **value** is the registry's
|
||||
/// verdict and is final: 0 bound, `-EPERM` this binary is not granted that name,
|
||||
/// `-EBUSY` a live provider already holds it.
|
||||
pub fn bind(name: []const u8, endpoint: ipc.Handle) ?i32 {
|
||||
const registry = reach(root) orelse return null;
|
||||
const answered = transact(registry.handle, .bind, name, endpoint) orelse return null;
|
||||
return answered.reply.status;
|
||||
}
|
||||
|
||||
/// How long a provider keeps offering itself before giving up. The registry is
|
||||
/// init, which mounts `/protocol` before it spawns anyone, so in a normal boot
|
||||
/// the first try lands; a provider the kernel test harness starts may well beat
|
||||
/// init to the mount, which is what the patience is for. Four seconds of 20 ms
|
||||
/// tries — the same cadence every client in the tree spends finding a service.
|
||||
const bind_attempts: u32 = 200;
|
||||
const bind_retry_ms: u64 = 20;
|
||||
|
||||
/// `bind`, waiting out a registry that is not mounted yet. Only unreachability
|
||||
/// is retried: a registry that *answered* has decided, and asking again cannot
|
||||
/// change its mind. True when the name is ours.
|
||||
pub fn bindPatiently(name: []const u8, endpoint: ipc.Handle) bool {
|
||||
var attempt: u32 = 0;
|
||||
while (attempt < bind_attempts) : (attempt += 1) {
|
||||
if (bind(name, endpoint)) |status| return status == 0;
|
||||
time.sleepMillis(bind_retry_ms);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/// `/protocol/` + `name`, in the caller's buffer. Null if the name is empty or
|
||||
/// longer than the namespace admits.
|
||||
fn join(name: []const u8, buffer: []u8) ?[]u8 {
|
||||
if (name.len == 0 or name.len > name_maximum) return null;
|
||||
const total = root.len + 1 + name.len;
|
||||
if (total > buffer.len) return null;
|
||||
@memcpy(buffer[0..root.len], root);
|
||||
buffer[root.len] = '/';
|
||||
@memcpy(buffer[root.len + 1 ..][0..name.len], name);
|
||||
return buffer[0..total];
|
||||
}
|
||||
|
||||
/// Lay a packet down: the folded header first, then the protocol's bytes. Null
|
||||
/// when it would not fit the buffer — the same rule as `envelope`'s framing,
|
||||
/// applied where the buffer is the transport's, not the protocol's.
|
||||
@@ -209,6 +324,14 @@ test "a framed packet is the header followed by the protocol's bytes" {
|
||||
try testing.expectEqualStrings("body", packet[envelope.prefix_size..]);
|
||||
}
|
||||
|
||||
test "a contract name joins the namespace root exactly once" {
|
||||
var buffer: [path_maximum]u8 = undefined;
|
||||
try testing.expectEqualStrings("/protocol/display", join("display", &buffer).?);
|
||||
try testing.expectEqualStrings("/protocol/test/shared-memory", join("test/shared-memory", &buffer).?);
|
||||
try testing.expect(join("", &buffer) == null);
|
||||
try testing.expect(join("x" ** (name_maximum + 1), &buffer) == null);
|
||||
}
|
||||
|
||||
test "framing refuses a packet that would not fit rather than truncating it" {
|
||||
var post: [envelope.post_maximum]u8 = undefined;
|
||||
const header = envelope.Header{ .operation = envelope.first_protocol_operation };
|
||||
|
||||
Reference in New Issue
Block a user