kernel: VFS root — mount table, /system from the initrd, fs syscalls

system/kernel/vfs.zig is the resolve+redirect router: fs_resolve (#46)
walks the kernel mount table; a path under the kernel-backed /system
mount (the initrd, seeded by setInitialRamdisk with a derived directory
table) yields a permanent stateless node token served by fs_node (#47)
— read/status/readdir with copy-out, initrd reads lock-free — while a
path under a userspace mount yields the backend's endpoint (installed
in the caller's table, DEDUPLICATED so 16 slots can't be exhausted by
repeated resolves) plus the rewritten mount-relative path; the caller
then speaks the unchanged vfs-protocol rendezvous directly. The kernel
never blocks on a userspace filesystem, holds no open-file state, and
refuses create-intent on the immutable /system.

fs_mount (#48) is the syscall form of the old router's op-6 cap-pass
(possession of the backend handle is the capability; an optional
rewrite prefix maps the mount into the backend's namespace — how /var
will reach the flash volume); fs_unmount (#49) removes one. A dead
backend's mount clears lazily on resolve.

Dormant this milestone: the userspace vfs still serves runtime.fs
unchanged; the kvfs QEMU case covers the kernel side (resolution, ELF
magic read-through, /system listing, read-only + unknown refusals)
until the M-G cutover exercises the syscalls end-to-end.
This commit is contained in:
Daniel Samson
2026-07-21 16:15:05 +01:00
parent 127ea2dad9
commit 7c5645fe48
6 changed files with 619 additions and 0 deletions
+127
View File
@@ -34,6 +34,7 @@ const ipc = @import("ipc-synchronous.zig");
const devices_broker = @import("devices-broker.zig");
const irq = @import("irq.zig");
const initial_ramdisk = @import("initial-ramdisk");
const vfs = @import("vfs.zig");
const log = @import("log.zig");
const wall_clock = @import("wall-clock.zig");
@@ -138,6 +139,7 @@ var ramdisk_image: ?[]const u8 = null;
/// handoff) so a user-space supervisor can `system_spawn` binaries out of it.
pub fn setInitialRamdisk(image: []const u8) void {
ramdisk_image = image;
vfs.setInitialRamdisk(image); // the kernel VFS serves the same bytes at /system
}
/// Spawn a bundled binary from the kernel by path. Used exactly once, to start
@@ -235,6 +237,10 @@ fn system_call(state: *architecture.CpuState) void {
.timer_bind => systemTimerBind(state),
.klog_read => systemKlogRead(state),
.klog_status => systemKlogStatus(state),
.fs_resolve => systemFsResolve(state),
.fs_node => systemFsNode(state),
.fs_mount => systemFsMount(state),
.fs_unmount => systemFsUnmount(state),
.wall_clock => systemWallClock(state),
.shm_create => systemShmCreate(state),
.shm_map => systemShmMap(state),
@@ -1299,6 +1305,127 @@ fn systemKlogStatus(state: *architecture.CpuState) void {
}
}
/// fs_resolve(path_ptr, path_len, flags, out_ptr, out_cap): route a path
/// through the kernel mount table (docs/vfs-protocol.md). Kernel-served ->
/// rax=fs_route_kernel, rdx=node token. Backend-served -> rax=fs_route_backend,
/// rdx=an endpoint handle in the caller's table (deduplicated), and the
/// rewritten mount-relative path copied to `out` with its length in the third
/// result register. Fails for unknown paths, create-intent on /system, or an
/// undersized out buffer.
fn systemFsResolve(state: *architecture.CpuState) void {
const path_ptr = architecture.systemCallArg(state, 0);
const path_len = architecture.systemCallArg(state, 1);
const flags = architecture.systemCallArg(state, 2);
const out_ptr = architecture.systemCallArg(state, 3);
const out_cap = architecture.systemCallArg(state, 4);
if (path_len == 0 or path_len > 224 or path_ptr >= user_half_end or path_ptr + path_len > user_half_end) return fail(state);
if (out_cap != 0 and (out_ptr >= user_half_end or out_ptr + out_cap > user_half_end)) return fail(state);
const path = @as([*]const u8, @ptrFromInt(path_ptr))[0..path_len];
const t = scheduler.current();
const flags_lock = sync.enter();
defer sync.leave(flags_lock);
switch (vfs.resolvePath(path, flags & abi.fs_flag_create != 0)) {
.kernel_node => |node_token| {
architecture.setSystemCallResult(state, abi.fs_route_kernel);
architecture.setSystemCallResult2(state, node_token);
},
.backend => |*backend| {
if (backend.path_len > out_cap) return fail(state);
const handle = ipc.installHandleDeduped(t, backend.endpoint);
if (handle < 0) return fail(state);
const destination: [*]u8 = @ptrFromInt(out_ptr);
@memcpy(destination[0..backend.path_len], backend.path[0..backend.path_len]);
architecture.setSystemCallResult(state, abi.fs_route_backend);
architecture.setSystemCallResult2(state, @intCast(handle));
architecture.setSystemCallResult3(state, backend.path_len);
},
.not_found => fail(state),
}
}
/// fs_node(op, node_token, offset, buf_ptr, buf_len) -> bytes/0/-errno: serve a
/// kernel-backed node. read copies file bytes; status copies a FileAttributes;
/// readdir copies [DirectoryEntryHeader][name] for the `offset`th child. Reads
/// of the immutable initrd never take the kernel lock.
fn systemFsNode(state: *architecture.CpuState) void {
const operation = architecture.systemCallArg(state, 0);
const node_token = architecture.systemCallArg(state, 1);
const offset = architecture.systemCallArg(state, 2);
const buf_ptr = architecture.systemCallArg(state, 3);
const buf_len = architecture.systemCallArg(state, 4);
if (buf_ptr >= user_half_end or buf_len > user_half_end - buf_ptr) return fail(state);
const capped = @min(buf_len, 64 * 1024); // bound any single copy
const destination: [*]u8 = @ptrFromInt(buf_ptr);
switch (operation) {
abi.fs_node_read => {
const n = vfs.nodeRead(node_token, offset, destination[0..capped]) orelse return fail(state);
architecture.setSystemCallResult(state, n);
},
abi.fs_node_status => {
var attributes = vfs.nodeStatus(node_token) orelse return fail(state);
if (capped < @sizeOf(abi.FileAttributes)) return fail(state);
@memcpy(destination[0..@sizeOf(abi.FileAttributes)], std.mem.asBytes(&attributes));
architecture.setSystemCallResult(state, @sizeOf(abi.FileAttributes));
},
abi.fs_node_readdir => {
const header_size = @sizeOf(abi.DirectoryEntryHeader);
if (capped < header_size) return fail(state);
var name_buffer: [64]u8 = undefined;
const result = vfs.nodeReaddir(node_token, offset, &name_buffer) orelse {
architecture.setSystemCallResult(state, 0); // past the end
return;
};
var header = result.header;
const total = header_size + @min(result.name_len, capped - header_size);
@memcpy(destination[0..header_size], std.mem.asBytes(&header));
@memcpy(destination[header_size..total], name_buffer[0 .. total - header_size]);
architecture.setSystemCallResult(state, total);
},
else => fail(state),
}
}
/// fs_mount(prefix_ptr, prefix_len, backend_handle, rewrite_ptr, rewrite_len):
/// mount a userspace filesystem at an absolute prefix. Possession of the
/// backend endpoint handle is the capability — the same trust as the old
/// router's cap-passing mount. The mount takes its own endpoint reference.
fn systemFsMount(state: *architecture.CpuState) void {
const prefix_ptr = architecture.systemCallArg(state, 0);
const prefix_len = architecture.systemCallArg(state, 1);
const backend_handle = architecture.systemCallArg(state, 2);
const rewrite_ptr = architecture.systemCallArg(state, 3);
const rewrite_len = architecture.systemCallArg(state, 4);
if (prefix_len == 0 or prefix_len > 64 or prefix_ptr >= user_half_end or prefix_ptr + prefix_len > user_half_end) return fail(state);
if (rewrite_len > 32) return fail(state);
if (rewrite_len != 0 and (rewrite_ptr >= user_half_end or rewrite_ptr + rewrite_len > user_half_end)) return fail(state);
const t = scheduler.current();
const prefix = @as([*]const u8, @ptrFromInt(prefix_ptr))[0..prefix_len];
const rewrite = if (rewrite_len == 0) "" else @as([*]const u8, @ptrFromInt(rewrite_ptr))[0..rewrite_len];
const flags = sync.enter();
defer sync.leave(flags);
const endpoint = ipc.resolveHandle(t, backend_handle) orelse return failErr(state, ipc.EBADF);
endpoint.refcount += 1; // the mount table's reference
if (!vfs.mountBackend(prefix, endpoint, rewrite)) {
ipc.dropRef(endpoint);
return fail(state);
}
architecture.setSystemCallResult(state, 0);
}
/// fs_unmount(prefix_ptr, prefix_len): remove a backend mount.
fn systemFsUnmount(state: *architecture.CpuState) void {
const prefix_ptr = architecture.systemCallArg(state, 0);
const prefix_len = architecture.systemCallArg(state, 1);
if (prefix_len == 0 or prefix_len > 64 or prefix_ptr >= user_half_end or prefix_ptr + prefix_len > user_half_end) return fail(state);
const prefix = @as([*]const u8, @ptrFromInt(prefix_ptr))[0..prefix_len];
const flags = sync.enter();
defer sync.leave(flags);
if (!vfs.unmount(prefix)) return fail(state);
architecture.setSystemCallResult(state, 0);
}
/// mmap(len, prot) -> base: grant `len` bytes (rounded up to whole pages) of
/// fresh, zeroed, writable+NX memory in the caller's mmap arena, and return the
/// base virtual address. `prot` is accepted but not yet honoured (grants are