A protocol is reached by name now, not by a compile-time integer. Init is PID 1 and already knows which binary it started, so init serves /protocol as a vfs backend: bind claims a contract with the provider's endpoint attached, open answers with that endpoint as the reply's capability, and readdir lists what is bound with the task and binary behind it. The kernel reserves the prefix — nothing may mount over it, under it, or unmount it — and ServiceId, ipc_register and ipc_lookup are gone, their syscall numbers left vacant. A bind is authorized by who the caller *is*: the kernel-stamped binary together with the supervising task's identity, matched against /system/configuration/protocol.csv. Identity, not spelling — spawn is ungated, so an attacker can run any bundled binary, and a name-only rule would have let it launder grants through an init of its own making. A name a live process holds is refused to everyone else; a dead one's is released. Three review rounds against a hostile ring-3 process found what 108 green tests could not, because the suite contains no attacker. Publishing init's supervision endpoint as the registry put PID 1's mailbox in every process's hands, where two forged bytes reached the shutdown path: privileged traffic is now believed only from the task that holds the contract it speaks for. A capability arriving on a request outlived every path that ignored it, one handle per call until the table was full — in init, and in the harness ten services share — so the arriving capability is owned by the turn and released unless a handler says otherwise. And the kernel let anyone holding an endpoint handle aim signals, timers, exit notices and interrupts at it: binding now requires having created it. Suite 108/108. The new protocol-registry case asserts eleven properties, each one an attack that must fail.
425 lines
18 KiB
Zig
425 lines
18 KiB
Zig
//! The kernel-resident VFS root: the mount table and the kernel-backed nodes.
|
|
//!
|
|
//! The kernel's job here is NAMING, never data plumbing to userspace backends —
|
|
//! the mechanism is **resolve + redirect**:
|
|
//!
|
|
//! - `fs_resolve(path)` walks the mount table. A path under a KERNEL-backed
|
|
//! mount (the initrd trees at /system and /test, the scratch ram nodes) resolves to a
|
|
//! stateless node TOKEN served directly by `fs_node` (read/status/readdir
|
|
//! with copy-out). A path under a USERSPACE mount (the fat server at
|
|
//! /volumes/usb, /system/configuration, and /system/logs) resolves to the
|
|
//! backend's ENDPOINT: the kernel
|
|
//! installs a (deduplicated) handle in the caller's table, rewrites the
|
|
//! path mount-relative, and the caller speaks the unchanged vfs-protocol
|
|
//! to the backend over the ordinary ipc_call rendezvous. The kernel never
|
|
//! blocks on a userspace server.
|
|
//!
|
|
//! - Kernel node tokens are PERMANENT for a boot: the initrd is immutable and
|
|
//! ram nodes are never reclaimed — no open-handle state, no close, no sweep
|
|
//! on client death. Backend file state lives in the backend, which sweeps
|
|
//! dead clients itself via the published exit events.
|
|
//!
|
|
//! Mounting is `fs_mount(prefix, backend_handle, rewrite)`: possession of the
|
|
//! backend endpoint handle is the capability, exactly the trust of the old
|
|
//! userspace router's op-6 cap-pass. An optional REWRITE prefix maps the mount
|
|
//! into the backend's namespace ("/system/logs" -> the boot volume's
|
|
//! identically-named subtree while the same backend also serves "/volumes/usb"
|
|
//! from its root), so hierarchy paths stay decoupled from which volume happens
|
|
//! to carry them.
|
|
|
|
const std = @import("std");
|
|
const abi = @import("abi");
|
|
const initial_ramdisk = @import("initial-ramdisk");
|
|
const ipc = @import("ipc-synchronous.zig");
|
|
|
|
// --- node tokens -------------------------------------------------------------
|
|
|
|
/// Kind lives in the top byte of a token; the index below. Tokens are permanent
|
|
/// for a boot, so userspace may cache them freely.
|
|
pub const token_kind_shift = 56;
|
|
pub const token_kind_initrd_file: u64 = 1;
|
|
pub const token_kind_initrd_directory: u64 = 2;
|
|
pub const token_kind_ram: u64 = 3;
|
|
|
|
fn token(kind: u64, index: u64) u64 {
|
|
return (kind << token_kind_shift) | index;
|
|
}
|
|
|
|
fn tokenKind(t: u64) u64 {
|
|
return t >> token_kind_shift;
|
|
}
|
|
|
|
fn tokenIndex(t: u64) u64 {
|
|
return t & ((@as(u64, 1) << token_kind_shift) - 1);
|
|
}
|
|
|
|
// --- the mount table ---------------------------------------------------------
|
|
|
|
pub const maximum_mounts = 8;
|
|
const maximum_prefix = 64;
|
|
const maximum_rewrite = 32;
|
|
|
|
const MountKind = enum(u8) { kernel_initrd, backend };
|
|
|
|
const Mount = struct {
|
|
used: bool = false,
|
|
prefix: [maximum_prefix]u8 = undefined,
|
|
prefix_len: usize = 0,
|
|
kind: MountKind = .backend,
|
|
backend: ?*ipc.Endpoint = null, // referenced while mounted
|
|
rewrite: [maximum_rewrite]u8 = undefined,
|
|
rewrite_len: usize = 0,
|
|
|
|
fn prefixSlice(self: *const Mount) []const u8 {
|
|
return self.prefix[0..self.prefix_len];
|
|
}
|
|
fn rewriteSlice(self: *const Mount) []const u8 {
|
|
return self.rewrite[0..self.rewrite_len];
|
|
}
|
|
};
|
|
|
|
var mounts: [maximum_mounts]Mount = @splat(.{});
|
|
|
|
/// The initrd image (set once at boot) and its derived directory table.
|
|
var ramdisk_image: ?[]const u8 = null;
|
|
|
|
const maximum_directories = 8;
|
|
const Directory = struct {
|
|
path: [maximum_prefix]u8 = undefined,
|
|
path_len: usize = 0,
|
|
parent: usize = 0, // index into `directories`; a top-level tree root is its own parent
|
|
|
|
fn slice(self: *const Directory) []const u8 {
|
|
return self.path[0..self.path_len];
|
|
}
|
|
};
|
|
var directories: [maximum_directories]Directory = @splat(.{});
|
|
var directory_count: usize = 0;
|
|
|
|
// --- pure path helpers (ported from the userspace router, with its tests) ----
|
|
|
|
/// If `path` lies under `mount_prefix` — equal to it, or the prefix followed by
|
|
/// a path separator — return the path relative to the mount ("/" for an exact
|
|
/// match, otherwise the tail beginning with '/'). Null when not under the
|
|
/// mount, so "/volumes/usb" never captures "/volumes/usbextra".
|
|
pub fn underMount(path: []const u8, mount_prefix: []const u8) ?[]const u8 {
|
|
if (path.len < mount_prefix.len) return null;
|
|
if (!std.mem.eql(u8, path[0..mount_prefix.len], mount_prefix)) return null;
|
|
if (path.len == mount_prefix.len) return "/";
|
|
if (path[mount_prefix.len] != '/') return null;
|
|
return path[mount_prefix.len..];
|
|
}
|
|
|
|
pub fn isAbsolute(path: []const u8) bool {
|
|
return path.len > 0 and path[0] == '/';
|
|
}
|
|
|
|
/// The parent directory portion of an initrd path ("/system/services/fat" ->
|
|
/// "/system/services").
|
|
fn parentOf(path: []const u8) []const u8 {
|
|
const slash = std.mem.lastIndexOfScalar(u8, path, '/') orelse return path[0..0];
|
|
if (slash == 0) return path[0..1];
|
|
return path[0..slash];
|
|
}
|
|
|
|
// --- boot wiring -------------------------------------------------------------
|
|
|
|
/// Publish the initrd as kernel-backed mounts — one per top-level tree its
|
|
/// entry paths name (/system, /test) — and derive the bounded directory table
|
|
/// (every ancestor directory of the entry paths). Called once at boot.
|
|
pub fn setInitialRamdisk(image: []const u8) void {
|
|
ramdisk_image = image;
|
|
directory_count = 0;
|
|
|
|
const rd = initial_ramdisk.Reader.init(image) orelse return;
|
|
var i: u32 = 0;
|
|
while (i < rd.count) : (i += 1) {
|
|
const item = rd.entry(i) orelse continue;
|
|
// Register every ancestor directory, top-level tree roots included.
|
|
var parent = parentOf(item.name);
|
|
while (parent.len > 1) : (parent = parentOf(parent)) {
|
|
if (directoryIndex(parent) == null and directory_count < maximum_directories) {
|
|
var d = &directories[directory_count];
|
|
@memcpy(d.path[0..parent.len], parent);
|
|
d.path_len = parent.len;
|
|
d.parent = directory_count; // fixed up below once all exist
|
|
directory_count += 1;
|
|
}
|
|
}
|
|
}
|
|
// Parent links (a second pass so out-of-order registration doesn't matter).
|
|
// A top-level root keeps itself as parent and becomes an initrd mount.
|
|
for (directories[0..directory_count], 0..) |*d, index| {
|
|
const parent = parentOf(d.slice());
|
|
d.parent = directoryIndex(parent) orelse index;
|
|
if (parent.len == 1) installMount(d.slice(), .kernel_initrd, null, "");
|
|
}
|
|
}
|
|
|
|
fn directoryIndex(path: []const u8) ?usize {
|
|
for (directories[0..directory_count], 0..) |*d, i| {
|
|
if (std.mem.eql(u8, d.slice(), path)) return i;
|
|
}
|
|
return null;
|
|
}
|
|
|
|
fn installMount(prefix: []const u8, kind: MountKind, backend: ?*ipc.Endpoint, rewrite: []const u8) void {
|
|
// Remount replaces: a restarted backend re-mounts its prefix.
|
|
var slot: ?*Mount = null;
|
|
for (&mounts) |*m| {
|
|
if (m.used and std.mem.eql(u8, m.prefixSlice(), prefix)) {
|
|
// ...except the protocol namespace. Remount-replace is how a
|
|
// restarted FAT retakes /volumes/usb; letting it retake /protocol
|
|
// would hand the whole naming layer to whoever asked second.
|
|
// First mount wins, and init (PID 1) is always first.
|
|
if (std.mem.eql(u8, prefix, protocol_root)) return;
|
|
if (m.backend) |old| ipc.dropRef(old);
|
|
slot = m;
|
|
break;
|
|
}
|
|
if (slot == null and !m.used) slot = m;
|
|
}
|
|
const m = slot orelse return;
|
|
m.* = .{ .used = true, .kind = kind, .backend = backend };
|
|
@memcpy(m.prefix[0..prefix.len], prefix);
|
|
m.prefix_len = prefix.len;
|
|
@memcpy(m.rewrite[0..rewrite.len], rewrite);
|
|
m.rewrite_len = rewrite.len;
|
|
}
|
|
|
|
// --- resolve -----------------------------------------------------------------
|
|
|
|
/// Longest rewritten mount-relative path a backend resolution can carry — the
|
|
/// size `fs_resolve`'s caller has to have room for, so it is named rather than
|
|
/// spelled out at the one place that builds it.
|
|
pub const maximum_backend_path = maximum_rewrite + maximum_prefix + 160;
|
|
|
|
pub const Resolved = union(enum) {
|
|
/// Kernel-served: a permanent node token.
|
|
kernel_node: u64,
|
|
/// Backend-served: the endpoint plus the rewritten mount-relative path.
|
|
backend: struct { endpoint: *ipc.Endpoint, path: [maximum_backend_path]u8, path_len: usize },
|
|
not_found: void,
|
|
};
|
|
|
|
/// Longest-prefix match over the mount table, then per-kind resolution.
|
|
/// `create`-intent on the immutable initrd trees fails here (EROFS-style).
|
|
pub fn resolvePath(path: []const u8, wants_create: bool) Resolved {
|
|
if (!isAbsolute(path)) {
|
|
return .{ .not_found = {} }; // bare names have no kernel namespace (ramfs retired)
|
|
}
|
|
var best: ?*Mount = null;
|
|
var best_relative: []const u8 = undefined;
|
|
for (&mounts) |*m| {
|
|
if (!m.used) continue;
|
|
const relative = underMount(path, m.prefixSlice()) orelse continue;
|
|
if (best == null or m.prefix_len > best.?.prefix_len) {
|
|
best = m;
|
|
best_relative = relative;
|
|
}
|
|
}
|
|
const m = best orelse return .{ .not_found = {} };
|
|
switch (m.kind) {
|
|
.kernel_initrd => {
|
|
if (wants_create) return .{ .not_found = {} }; // read-only
|
|
return resolveInitrd(path);
|
|
},
|
|
.backend => {
|
|
const endpoint = m.backend orelse return .{ .not_found = {} };
|
|
if (endpoint.dead) {
|
|
// The backend died: treat the mount as gone (it re-mounts on
|
|
// restart) and release our reference lazily.
|
|
ipc.dropRef(endpoint);
|
|
m.backend = null;
|
|
m.used = false;
|
|
return .{ .not_found = {} };
|
|
}
|
|
var out: Resolved = .{ .backend = .{ .endpoint = endpoint, .path = undefined, .path_len = 0 } };
|
|
const rewrite = m.rewriteSlice();
|
|
const tail = if (std.mem.eql(u8, best_relative, "/") and rewrite.len != 0) "" else best_relative;
|
|
const total = rewrite.len + tail.len;
|
|
if (total > out.backend.path.len or total == 0) {
|
|
if (rewrite.len == 0 and tail.len == 0) return .{ .not_found = {} };
|
|
if (total > out.backend.path.len) return .{ .not_found = {} };
|
|
}
|
|
@memcpy(out.backend.path[0..rewrite.len], rewrite);
|
|
@memcpy(out.backend.path[rewrite.len..][0..tail.len], tail);
|
|
out.backend.path_len = total;
|
|
return out;
|
|
},
|
|
}
|
|
}
|
|
|
|
fn resolveInitrd(path: []const u8) Resolved {
|
|
if (directoryIndex(path)) |index| return .{ .kernel_node = token(token_kind_initrd_directory, index) };
|
|
const image = ramdisk_image orelse return .{ .not_found = {} };
|
|
const rd = initial_ramdisk.Reader.init(image) orelse return .{ .not_found = {} };
|
|
var i: u32 = 0;
|
|
while (i < rd.count) : (i += 1) {
|
|
const item = rd.entry(i) orelse continue;
|
|
if (std.mem.eql(u8, item.name, path)) return .{ .kernel_node = token(token_kind_initrd_file, i) };
|
|
}
|
|
return .{ .not_found = {} };
|
|
}
|
|
|
|
// --- fs_node: serving kernel-backed nodes ------------------------------------
|
|
|
|
/// Read `out.len` bytes of an initrd file at `offset`. Returns bytes copied
|
|
/// (0 at EOF) or null for a bad token. Lock-free: the initrd is immutable.
|
|
pub fn nodeRead(node_token: u64, offset: u64, out: []u8) ?usize {
|
|
if (tokenKind(node_token) != token_kind_initrd_file) return null;
|
|
const image = ramdisk_image orelse return null;
|
|
const rd = initial_ramdisk.Reader.init(image) orelse return null;
|
|
const item = rd.entry(@intCast(tokenIndex(node_token))) orelse return null;
|
|
if (offset >= item.blob.len) return 0;
|
|
const n = @min(out.len, item.blob.len - @as(usize, @intCast(offset)));
|
|
@memcpy(out[0..n], item.blob[@intCast(offset)..][0..n]);
|
|
return n;
|
|
}
|
|
|
|
/// A node's metadata in vfs-protocol FileStatus shape (size, kind, mtime).
|
|
pub fn nodeStatus(node_token: u64) ?abi.FileAttributes {
|
|
switch (tokenKind(node_token)) {
|
|
token_kind_initrd_file => {
|
|
const image = ramdisk_image orelse return null;
|
|
const rd = initial_ramdisk.Reader.init(image) orelse return null;
|
|
const item = rd.entry(@intCast(tokenIndex(node_token))) orelse return null;
|
|
return .{ .size = item.blob.len, .kind = abi.file_kind_regular };
|
|
},
|
|
token_kind_initrd_directory => {
|
|
if (tokenIndex(node_token) >= directory_count) return null;
|
|
return .{ .size = 0, .kind = abi.file_kind_directory };
|
|
},
|
|
else => return null,
|
|
}
|
|
}
|
|
|
|
/// The `cursor`th child of an initrd directory: fills `name_out`, returns the
|
|
/// entry header, or null past the end / bad token. Cursor enumerates
|
|
/// subdirectories first, then files whose parent is this directory — stable,
|
|
/// because the initrd is immutable.
|
|
pub fn nodeReaddir(node_token: u64, cursor: u64, name_out: []u8) ?struct { header: abi.DirectoryEntryHeader, name_len: usize } {
|
|
if (tokenKind(node_token) != token_kind_initrd_directory) return null;
|
|
const directory_index = tokenIndex(node_token);
|
|
if (directory_index >= directory_count) return null;
|
|
const self_path = directories[@intCast(directory_index)].slice();
|
|
|
|
var index: u64 = 0;
|
|
// Subdirectories whose parent is this directory (roots are self-parented,
|
|
// so they never appear as anyone's child).
|
|
for (directories[0..directory_count], 0..) |*d, i| {
|
|
if (i == directory_index) continue;
|
|
if (d.parent != directory_index) continue;
|
|
if (index == cursor) {
|
|
const name = d.slice()[self_path.len + 1 ..];
|
|
const n = @min(name.len, name_out.len);
|
|
@memcpy(name_out[0..n], name[0..n]);
|
|
return .{ .header = .{ .kind = abi.file_kind_directory, .name_len = @intCast(n), .size = 0 }, .name_len = n };
|
|
}
|
|
index += 1;
|
|
}
|
|
// Files directly inside this directory.
|
|
const image = ramdisk_image orelse return null;
|
|
const rd = initial_ramdisk.Reader.init(image) orelse return null;
|
|
var i: u32 = 0;
|
|
while (i < rd.count) : (i += 1) {
|
|
const item = rd.entry(i) orelse continue;
|
|
if (!std.mem.eql(u8, parentOf(item.name), self_path)) continue;
|
|
if (index == cursor) {
|
|
const name = item.name[self_path.len + 1 ..];
|
|
const n = @min(name.len, name_out.len);
|
|
@memcpy(name_out[0..n], name[0..n]);
|
|
return .{ .header = .{ .kind = abi.file_kind_regular, .name_len = @intCast(n), .size = item.blob.len }, .name_len = n };
|
|
}
|
|
index += 1;
|
|
}
|
|
return null;
|
|
}
|
|
|
|
// --- mount/unmount (syscall bodies; caller resolved the handle) --------------
|
|
|
|
/// The writable subtrees a backend may mount beneath an initrd tree — exactly
|
|
/// these two, nothing else. Longest-prefix resolution then routes them to the
|
|
/// volume while every other /system and /test path stays initrd-served, so no
|
|
/// bundled binary can ever be shadowed.
|
|
const initrd_carve_outs = [_][]const u8{ "/system/configuration", "/system/logs" };
|
|
|
|
fn isInitrdCarveOut(prefix: []const u8) bool {
|
|
for (initrd_carve_outs) |allowed| {
|
|
if (std.mem.eql(u8, prefix, allowed)) return true;
|
|
}
|
|
return false;
|
|
}
|
|
|
|
/// The protocol namespace's root — a reserved prefix, like the initrd trees.
|
|
/// Init (PID 1) mounts the registry here once at boot and the prefix then
|
|
/// refuses everything: a second mount at it, any mount *under* it (which would
|
|
/// shadow one contract), and its unmount. That is the whole kernel-side residue
|
|
/// of the naming layer — the registrar authority itself never leaves init
|
|
/// (docs/os-development/protocol-namespace.md).
|
|
const protocol_root = "/protocol";
|
|
|
|
fn protocolBound() bool {
|
|
for (&mounts) |*m| {
|
|
if (m.used and std.mem.eql(u8, m.prefixSlice(), protocol_root)) return true;
|
|
}
|
|
return false;
|
|
}
|
|
|
|
/// Whether mounting at `prefix` would touch the protocol namespace. Exactly
|
|
/// `/protocol` is allowed once — while nothing holds it; anything under it,
|
|
/// ever, is refused.
|
|
fn refusesProtocolMount(prefix: []const u8) bool {
|
|
const relative = underMount(prefix, protocol_root) orelse return false;
|
|
if (relative.len != 1) return true; // strictly under /protocol: never
|
|
return protocolBound(); // /protocol itself: first mount wins
|
|
}
|
|
|
|
/// Mount `backend` at `prefix` with an optional backend-side `rewrite` prefix.
|
|
/// The endpoint reference is taken by the caller (process.zig bumps it); refuses
|
|
/// shadowing or replacing the initrd trees (/system, /test) — except the two
|
|
/// carve-outs in `initrd_carve_outs`, the writable configuration/log subtrees.
|
|
pub fn mountBackend(prefix: []const u8, backend: *ipc.Endpoint, rewrite: []const u8) bool {
|
|
if (!isAbsolute(prefix) or prefix.len < 2 or prefix.len > maximum_prefix) return false;
|
|
if (rewrite.len > maximum_rewrite) return false;
|
|
if (refusesProtocolMount(prefix)) return false; // the registry's prefix is claimed once
|
|
for (&mounts) |*m| { // the initrd trees are not shadowable (carve-outs aside)
|
|
if (m.used and m.kind == .kernel_initrd and underMount(prefix, m.prefixSlice()) != null) {
|
|
if (!isInitrdCarveOut(prefix)) return false;
|
|
}
|
|
}
|
|
installMount(prefix, .backend, backend, rewrite);
|
|
return true;
|
|
}
|
|
|
|
pub fn unmount(prefix: []const u8) bool {
|
|
// Unmounting /protocol would delete the naming layer for everyone; nobody
|
|
// may, init included. The mount lasts the boot.
|
|
if (std.mem.eql(u8, prefix, protocol_root)) return false;
|
|
for (&mounts) |*m| {
|
|
if (m.used and m.kind == .backend and std.mem.eql(u8, m.prefixSlice(), prefix)) {
|
|
if (m.backend) |endpoint| ipc.dropRef(endpoint);
|
|
m.* = .{};
|
|
return true;
|
|
}
|
|
}
|
|
return false;
|
|
}
|
|
|
|
// --- tests (host) ------------------------------------------------------------
|
|
|
|
test "underMount matches only at path boundaries" {
|
|
try std.testing.expectEqualStrings("/", underMount("/volumes/usb", "/volumes/usb").?);
|
|
try std.testing.expectEqualStrings("/system/kernel", underMount("/volumes/usb/system/kernel", "/volumes/usb").?);
|
|
try std.testing.expect(underMount("/volumes/usbextra", "/volumes/usb") == null);
|
|
try std.testing.expect(underMount("/volumes", "/volumes/usb") == null);
|
|
try std.testing.expect(underMount("/other", "/volumes/usb") == null);
|
|
try std.testing.expect(underMount("greeting", "/volumes/usb") == null);
|
|
}
|
|
|
|
test "parentOf walks toward the root" {
|
|
try std.testing.expectEqualStrings("/system/services", parentOf("/system/services/fat"));
|
|
try std.testing.expectEqualStrings("/system", parentOf("/system/services"));
|
|
try std.testing.expectEqualStrings("/", parentOf("/system"));
|
|
}
|