kernel: VFS root — mount table, /system from the initrd, fs syscalls

system/kernel/vfs.zig is the resolve+redirect router: fs_resolve (#46)
walks the kernel mount table; a path under the kernel-backed /system
mount (the initrd, seeded by setInitialRamdisk with a derived directory
table) yields a permanent stateless node token served by fs_node (#47)
— read/status/readdir with copy-out, initrd reads lock-free — while a
path under a userspace mount yields the backend's endpoint (installed
in the caller's table, DEDUPLICATED so 16 slots can't be exhausted by
repeated resolves) plus the rewritten mount-relative path; the caller
then speaks the unchanged vfs-protocol rendezvous directly. The kernel
never blocks on a userspace filesystem, holds no open-file state, and
refuses create-intent on the immutable /system.

fs_mount (#48) is the syscall form of the old router's op-6 cap-pass
(possession of the backend handle is the capability; an optional
rewrite prefix maps the mount into the backend's namespace — how /var
will reach the flash volume); fs_unmount (#49) removes one. A dead
backend's mount clears lazily on resolve.

Dormant this milestone: the userspace vfs still serves runtime.fs
unchanged; the kvfs QEMU case covers the kernel side (resolution, ELF
magic read-through, /system listing, read-only + unknown refusals)
until the M-G cutover exercises the syscalls end-to-end.
This commit is contained in:
Daniel Samson
2026-07-21 16:15:05 +01:00
parent 127ea2dad9
commit 7c5645fe48
6 changed files with 619 additions and 0 deletions
+43
View File
@@ -72,6 +72,10 @@ pub const SystemCall = enum(u64) {
thread_join = 43, // thread_join(tid) -> 0: block until the thread with id `tid` has exited (runtime.Thread.join; no per-thread IPC endpoint) (docs/threading.md)
set_thread_pointer = 44, // set_thread_pointer(addr) -> 0: set the caller's thread pointer (user-space TLS base; x86_64 IA32_FS_BASE, aarch64 TPIDR_EL0); restored per task across context switches (docs/threading-plan.md M10)
klog_status = 45, // klog_status(ptr) -> 0: copy a KlogStatus (ring cursors + the boot wall-clock anchor) out to a user buffer
fs_resolve = 46, // fs_resolve(path_ptr, path_len, flags, out_ptr, out_cap) -> route tag (rax: fs_route_*) + node token or backend handle (rdx); a backend resolve writes the rewritten mount-relative path into out (length in r8 via third result)
fs_node = 47, // fs_node(op, node_token, offset, buf_ptr, buf_len) -> bytes/0/-errno: read/status/readdir on a kernel-served node (op values mirror the vfs-protocol Operation numbers)
fs_mount = 48, // fs_mount(prefix_ptr, prefix_len, backend_handle, rewrite_ptr, rewrite_len) -> 0/-errno: mount a userspace filesystem's endpoint at an absolute prefix (possession of the handle is the capability)
fs_unmount = 49, // fs_unmount(prefix_ptr, prefix_len) -> 0/-errno: remove a backend mount
_,
};
@@ -225,6 +229,45 @@ pub const klog_record_alignment: usize = 8;
/// Per-record payload cap (one line; longer emitter lines are truncated).
pub const klog_maximum_message: usize = 256;
// --- the kernel VFS root (resolve + redirect) --------------------------------
// fs_resolve routes a path through the kernel mount table. Kernel-backed mounts
// (the initrd at /system) resolve to a permanent node TOKEN served by fs_node;
// userspace mounts resolve to the backend's endpoint handle (installed in the
// caller's table, deduplicated) plus the rewritten mount-relative path — the
// caller then speaks the vfs-protocol to the backend directly. The kernel never
// blocks on a userspace filesystem.
/// fs_resolve result tags (rax).
pub const fs_route_kernel: u64 = 0; // rdx = node token; serve via fs_node
pub const fs_route_backend: u64 = 1; // rdx = endpoint handle; speak vfs-protocol
/// fs_node operations — the same numbers as the vfs-protocol Operation enum, so
/// client code shares one vocabulary.
pub const fs_node_read: u64 = 2;
pub const fs_node_status: u64 = 4;
pub const fs_node_readdir: u64 = 5;
/// fs_resolve flags (same values as the vfs-protocol open flags).
pub const fs_flag_create: u64 = 1;
/// FileStatus-shaped node metadata (matches the vfs-protocol payload layout).
pub const file_kind_regular: u32 = 0;
pub const file_kind_directory: u32 = 1;
pub const FileAttributes = extern struct {
size: u64,
kind: u32,
_pad: u32 = 0,
mtime: u64 = 0,
};
/// One fs_node readdir result: the header, followed by `name_len` name bytes in
/// the caller's buffer (matches the vfs-protocol DirectoryEntry layout).
pub const DirectoryEntryHeader = extern struct {
kind: u32,
name_len: u32,
size: u64,
};
/// The klog_status copy-out: the ring's live cursors plus the wall-clock time
/// of boot — the anchor a log persister names its per-boot directory with and
/// combines with record timestamps for wall-clock line stamps.
+14
View File
@@ -516,6 +516,20 @@ pub fn installShmHandle(t: *Task, shm: *ShmObject) i64 {
return installEntry(t, .{ .kind = handle_kind_shm, .ptr = @ptrCast(shm) });
}
/// Install an endpoint handle, reusing an existing slot that already names this
/// endpoint (no new reference taken in that case). For callers that install per
/// operation — fs_resolve — so a 16-slot table can't be exhausted by repeats.
/// Any subsystem installing handles per-call should come through here.
pub fn installHandleDeduped(t: *Task, endpoint: *Endpoint) i64 {
for (t.handles, 0..) |slot, i| {
const entry = slot orelse continue;
if (entry.kind == handle_kind_endpoint and entry.ptr == @as(*anyopaque, @ptrCast(endpoint))) return @intCast(i);
}
const h = installHandle(t, endpoint);
if (h >= 0) endpoint.refcount += 1; // the table entry owns a reference
return h;
}
/// Resolve a handle to its endpoint, or null if out of range, unused, or a different kind
/// (e.g. an shm handle used where an endpoint is expected).
pub fn resolveHandle(t: *Task, h: u64) ?*Endpoint {
+127
View File
@@ -34,6 +34,7 @@ const ipc = @import("ipc-synchronous.zig");
const devices_broker = @import("devices-broker.zig");
const irq = @import("irq.zig");
const initial_ramdisk = @import("initial-ramdisk");
const vfs = @import("vfs.zig");
const log = @import("log.zig");
const wall_clock = @import("wall-clock.zig");
@@ -138,6 +139,7 @@ var ramdisk_image: ?[]const u8 = null;
/// handoff) so a user-space supervisor can `system_spawn` binaries out of it.
pub fn setInitialRamdisk(image: []const u8) void {
ramdisk_image = image;
vfs.setInitialRamdisk(image); // the kernel VFS serves the same bytes at /system
}
/// Spawn a bundled binary from the kernel by path. Used exactly once, to start
@@ -235,6 +237,10 @@ fn system_call(state: *architecture.CpuState) void {
.timer_bind => systemTimerBind(state),
.klog_read => systemKlogRead(state),
.klog_status => systemKlogStatus(state),
.fs_resolve => systemFsResolve(state),
.fs_node => systemFsNode(state),
.fs_mount => systemFsMount(state),
.fs_unmount => systemFsUnmount(state),
.wall_clock => systemWallClock(state),
.shm_create => systemShmCreate(state),
.shm_map => systemShmMap(state),
@@ -1299,6 +1305,127 @@ fn systemKlogStatus(state: *architecture.CpuState) void {
}
}
/// fs_resolve(path_ptr, path_len, flags, out_ptr, out_cap): route a path
/// through the kernel mount table (docs/vfs-protocol.md). Kernel-served ->
/// rax=fs_route_kernel, rdx=node token. Backend-served -> rax=fs_route_backend,
/// rdx=an endpoint handle in the caller's table (deduplicated), and the
/// rewritten mount-relative path copied to `out` with its length in the third
/// result register. Fails for unknown paths, create-intent on /system, or an
/// undersized out buffer.
fn systemFsResolve(state: *architecture.CpuState) void {
const path_ptr = architecture.systemCallArg(state, 0);
const path_len = architecture.systemCallArg(state, 1);
const flags = architecture.systemCallArg(state, 2);
const out_ptr = architecture.systemCallArg(state, 3);
const out_cap = architecture.systemCallArg(state, 4);
if (path_len == 0 or path_len > 224 or path_ptr >= user_half_end or path_ptr + path_len > user_half_end) return fail(state);
if (out_cap != 0 and (out_ptr >= user_half_end or out_ptr + out_cap > user_half_end)) return fail(state);
const path = @as([*]const u8, @ptrFromInt(path_ptr))[0..path_len];
const t = scheduler.current();
const flags_lock = sync.enter();
defer sync.leave(flags_lock);
switch (vfs.resolvePath(path, flags & abi.fs_flag_create != 0)) {
.kernel_node => |node_token| {
architecture.setSystemCallResult(state, abi.fs_route_kernel);
architecture.setSystemCallResult2(state, node_token);
},
.backend => |*backend| {
if (backend.path_len > out_cap) return fail(state);
const handle = ipc.installHandleDeduped(t, backend.endpoint);
if (handle < 0) return fail(state);
const destination: [*]u8 = @ptrFromInt(out_ptr);
@memcpy(destination[0..backend.path_len], backend.path[0..backend.path_len]);
architecture.setSystemCallResult(state, abi.fs_route_backend);
architecture.setSystemCallResult2(state, @intCast(handle));
architecture.setSystemCallResult3(state, backend.path_len);
},
.not_found => fail(state),
}
}
/// fs_node(op, node_token, offset, buf_ptr, buf_len) -> bytes/0/-errno: serve a
/// kernel-backed node. read copies file bytes; status copies a FileAttributes;
/// readdir copies [DirectoryEntryHeader][name] for the `offset`th child. Reads
/// of the immutable initrd never take the kernel lock.
fn systemFsNode(state: *architecture.CpuState) void {
const operation = architecture.systemCallArg(state, 0);
const node_token = architecture.systemCallArg(state, 1);
const offset = architecture.systemCallArg(state, 2);
const buf_ptr = architecture.systemCallArg(state, 3);
const buf_len = architecture.systemCallArg(state, 4);
if (buf_ptr >= user_half_end or buf_len > user_half_end - buf_ptr) return fail(state);
const capped = @min(buf_len, 64 * 1024); // bound any single copy
const destination: [*]u8 = @ptrFromInt(buf_ptr);
switch (operation) {
abi.fs_node_read => {
const n = vfs.nodeRead(node_token, offset, destination[0..capped]) orelse return fail(state);
architecture.setSystemCallResult(state, n);
},
abi.fs_node_status => {
var attributes = vfs.nodeStatus(node_token) orelse return fail(state);
if (capped < @sizeOf(abi.FileAttributes)) return fail(state);
@memcpy(destination[0..@sizeOf(abi.FileAttributes)], std.mem.asBytes(&attributes));
architecture.setSystemCallResult(state, @sizeOf(abi.FileAttributes));
},
abi.fs_node_readdir => {
const header_size = @sizeOf(abi.DirectoryEntryHeader);
if (capped < header_size) return fail(state);
var name_buffer: [64]u8 = undefined;
const result = vfs.nodeReaddir(node_token, offset, &name_buffer) orelse {
architecture.setSystemCallResult(state, 0); // past the end
return;
};
var header = result.header;
const total = header_size + @min(result.name_len, capped - header_size);
@memcpy(destination[0..header_size], std.mem.asBytes(&header));
@memcpy(destination[header_size..total], name_buffer[0 .. total - header_size]);
architecture.setSystemCallResult(state, total);
},
else => fail(state),
}
}
/// fs_mount(prefix_ptr, prefix_len, backend_handle, rewrite_ptr, rewrite_len):
/// mount a userspace filesystem at an absolute prefix. Possession of the
/// backend endpoint handle is the capability — the same trust as the old
/// router's cap-passing mount. The mount takes its own endpoint reference.
fn systemFsMount(state: *architecture.CpuState) void {
const prefix_ptr = architecture.systemCallArg(state, 0);
const prefix_len = architecture.systemCallArg(state, 1);
const backend_handle = architecture.systemCallArg(state, 2);
const rewrite_ptr = architecture.systemCallArg(state, 3);
const rewrite_len = architecture.systemCallArg(state, 4);
if (prefix_len == 0 or prefix_len > 64 or prefix_ptr >= user_half_end or prefix_ptr + prefix_len > user_half_end) return fail(state);
if (rewrite_len > 32) return fail(state);
if (rewrite_len != 0 and (rewrite_ptr >= user_half_end or rewrite_ptr + rewrite_len > user_half_end)) return fail(state);
const t = scheduler.current();
const prefix = @as([*]const u8, @ptrFromInt(prefix_ptr))[0..prefix_len];
const rewrite = if (rewrite_len == 0) "" else @as([*]const u8, @ptrFromInt(rewrite_ptr))[0..rewrite_len];
const flags = sync.enter();
defer sync.leave(flags);
const endpoint = ipc.resolveHandle(t, backend_handle) orelse return failErr(state, ipc.EBADF);
endpoint.refcount += 1; // the mount table's reference
if (!vfs.mountBackend(prefix, endpoint, rewrite)) {
ipc.dropRef(endpoint);
return fail(state);
}
architecture.setSystemCallResult(state, 0);
}
/// fs_unmount(prefix_ptr, prefix_len): remove a backend mount.
fn systemFsUnmount(state: *architecture.CpuState) void {
const prefix_ptr = architecture.systemCallArg(state, 0);
const prefix_len = architecture.systemCallArg(state, 1);
if (prefix_len == 0 or prefix_len > 64 or prefix_ptr >= user_half_end or prefix_ptr + prefix_len > user_half_end) return fail(state);
const prefix = @as([*]const u8, @ptrFromInt(prefix_ptr))[0..prefix_len];
const flags = sync.enter();
defer sync.leave(flags);
if (!vfs.unmount(prefix)) return fail(state);
architecture.setSystemCallResult(state, 0);
}
/// mmap(len, prot) -> base: grant `len` bytes (rounded up to whole pages) of
/// fresh, zeroed, writable+NX memory in the caller's mmap arena, and return the
/// base virtual address. `prot` is accepted but not yet honoured (grants are
+61
View File
@@ -27,6 +27,7 @@ const sync = @import("sync.zig");
const process = @import("process.zig");
const initial_ramdisk = @import("initial-ramdisk");
const kernel_log = @import("log.zig");
const kernel_vfs = @import("vfs.zig");
/// Formatted test-marker write. Goes through the kernel log (not straight to
/// serial): the log lock is what keeps marker lines from interleaving with
@@ -209,6 +210,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
initialRamdiskTest(boot_information);
} else if (eql(case, "vfs")) {
vfsTest(boot_information);
} else if (eql(case, "kvfs")) {
kernelVfsTest(boot_information);
} else if (eql(case, "input")) {
inputTest(boot_information);
} else if (eql(case, "iopass")) {
@@ -3073,6 +3076,64 @@ fn bundledInit(boot_information: *const BootInformation) ?[]const u8 {
return item.blob;
}
/// The kernel VFS root (M-F): resolve initrd paths to node tokens, read an ELF
/// header through nodeRead, enumerate /system's derived directory table, and
/// verify the read-only + unknown-path refusals. Pure kernel-side — the
/// syscall surface gets its end-to-end coverage when runtime.fs cuts over.
fn kernelVfsTest(boot_information: *const BootInformation) void {
log("DANOS-TEST-BEGIN: kvfs\n", .{});
if (boot_information.initial_ramdisk_len == 0) {
check("bootloader handed over the initial_ramdisk", false);
result();
return;
}
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
process.setInitialRamdisk(image); // also seeds the kernel VFS /system mount
// A file resolves to a kernel node token; its status and bytes are served.
const resolved = kernel_vfs.resolvePath("/system/services/init", false);
const is_file = resolved == .kernel_node;
check("/system/services/init resolves to a kernel node", is_file);
if (is_file) {
const status = kernel_vfs.nodeStatus(resolved.kernel_node);
check("its status is a non-empty regular file", status != null and status.?.kind == abi.file_kind_regular and status.?.size > 0);
var header: [4]u8 = undefined;
const n = kernel_vfs.nodeRead(resolved.kernel_node, 0, &header) orelse 0;
check("its first bytes are an ELF magic", n == 4 and header[0] == 0x7f and header[1] == 'E' and header[2] == 'L' and header[3] == 'F');
}
// Directories resolve and enumerate: /system lists services/drivers/tests.
const root_directory = kernel_vfs.resolvePath("/system", false);
check("/system resolves to a directory node", root_directory == .kernel_node);
var saw_services = false;
var saw_drivers = false;
var saw_files_in_services = false;
if (root_directory == .kernel_node) {
var cursor: u64 = 0;
var name: [64]u8 = undefined;
while (kernel_vfs.nodeReaddir(root_directory.kernel_node, cursor, &name)) |entry| : (cursor += 1) {
if (eql(name[0..entry.name_len], "services")) saw_services = true;
if (eql(name[0..entry.name_len], "drivers")) saw_drivers = true;
}
}
check("readdir /system yields services and drivers", saw_services and saw_drivers);
const services = kernel_vfs.resolvePath("/system/services", false);
if (services == .kernel_node) {
var cursor: u64 = 0;
var name: [64]u8 = undefined;
while (kernel_vfs.nodeReaddir(services.kernel_node, cursor, &name)) |entry| : (cursor += 1) {
if (eql(name[0..entry.name_len], "init")) saw_files_in_services = true;
}
}
check("readdir /system/services yields init", saw_files_in_services);
// Refusals: unknown paths, and create-intent on the immutable initrd.
check("an unknown path does not resolve", kernel_vfs.resolvePath("/system/services/no-such", false) == .not_found);
check("an unmounted absolute path does not resolve", kernel_vfs.resolvePath("/elsewhere", false) == .not_found);
check("create on /system is refused (read-only)", kernel_vfs.resolvePath("/system/services/new-file", true) == .not_found);
result();
}
fn spawnNamed(rd: initial_ramdisk.Reader, name: []const u8) bool {
var i: u32 = 0;
while (i < rd.count) : (i += 1) {
+368
View File
@@ -0,0 +1,368 @@
//! The kernel-resident VFS root: the mount table and the kernel-backed nodes.
//!
//! The kernel's job here is NAMING, never data plumbing to userspace backends —
//! the mechanism is **resolve + redirect**:
//!
//! - `fs_resolve(path)` walks the mount table. A path under a KERNEL-backed
//! mount (the initrd at /system, the scratch ram nodes) resolves to a
//! stateless node TOKEN served directly by `fs_node` (read/status/readdir
//! with copy-out). A path under a USERSPACE mount (the fat server at
//! /mnt/usb and /var) resolves to the backend's ENDPOINT: the kernel
//! installs a (deduplicated) handle in the caller's table, rewrites the
//! path mount-relative, and the caller speaks the unchanged vfs-protocol
//! to the backend over the ordinary ipc_call rendezvous. The kernel never
//! blocks on a userspace server.
//!
//! - Kernel node tokens are PERMANENT for a boot: the initrd is immutable and
//! ram nodes are never reclaimed — no open-handle state, no close, no sweep
//! on client death. Backend file state lives in the backend, which sweeps
//! dead clients itself via the published exit events.
//!
//! Mounting is `fs_mount(prefix, backend_handle, rewrite)`: possession of the
//! backend endpoint handle is the capability, exactly the trust of the old
//! userspace router's op-6 cap-pass. An optional REWRITE prefix maps the mount
//! into the backend's namespace ("/var" -> fat's "/var" subtree while the same
//! backend also serves "/mnt/usb" from its root), so FHS paths stay decoupled
//! from which volume happens to carry them.
const std = @import("std");
const abi = @import("abi");
const initial_ramdisk = @import("initial-ramdisk");
const ipc = @import("ipc-synchronous.zig");
// --- node tokens -------------------------------------------------------------
/// Kind lives in the top byte of a token; the index below. Tokens are permanent
/// for a boot, so userspace may cache them freely.
pub const token_kind_shift = 56;
pub const token_kind_initrd_file: u64 = 1;
pub const token_kind_initrd_directory: u64 = 2;
pub const token_kind_ram: u64 = 3;
fn token(kind: u64, index: u64) u64 {
return (kind << token_kind_shift) | index;
}
fn tokenKind(t: u64) u64 {
return t >> token_kind_shift;
}
fn tokenIndex(t: u64) u64 {
return t & ((@as(u64, 1) << token_kind_shift) - 1);
}
// --- the mount table ---------------------------------------------------------
pub const maximum_mounts = 8;
const maximum_prefix = 64;
const maximum_rewrite = 32;
const MountKind = enum(u8) { kernel_initrd, backend };
const Mount = struct {
used: bool = false,
prefix: [maximum_prefix]u8 = undefined,
prefix_len: usize = 0,
kind: MountKind = .backend,
backend: ?*ipc.Endpoint = null, // referenced while mounted
rewrite: [maximum_rewrite]u8 = undefined,
rewrite_len: usize = 0,
fn prefixSlice(self: *const Mount) []const u8 {
return self.prefix[0..self.prefix_len];
}
fn rewriteSlice(self: *const Mount) []const u8 {
return self.rewrite[0..self.rewrite_len];
}
};
var mounts: [maximum_mounts]Mount = @splat(.{});
/// The initrd image (set once at boot) and its derived directory table.
var ramdisk_image: ?[]const u8 = null;
const maximum_directories = 8;
const Directory = struct {
path: [maximum_prefix]u8 = undefined,
path_len: usize = 0,
parent: usize = 0, // index into `directories`; 0 is /system itself
fn slice(self: *const Directory) []const u8 {
return self.path[0..self.path_len];
}
};
var directories: [maximum_directories]Directory = @splat(.{});
var directory_count: usize = 0;
// --- pure path helpers (ported from the userspace router, with its tests) ----
/// If `path` lies under `mount_prefix` — equal to it, or the prefix followed by
/// a path separator — return the path relative to the mount ("/" for an exact
/// match, otherwise the tail beginning with '/'). Null when not under the
/// mount, so "/mnt/usb" never captures "/mnt/usbextra".
pub fn underMount(path: []const u8, mount_prefix: []const u8) ?[]const u8 {
if (path.len < mount_prefix.len) return null;
if (!std.mem.eql(u8, path[0..mount_prefix.len], mount_prefix)) return null;
if (path.len == mount_prefix.len) return "/";
if (path[mount_prefix.len] != '/') return null;
return path[mount_prefix.len..];
}
pub fn isAbsolute(path: []const u8) bool {
return path.len > 0 and path[0] == '/';
}
/// The parent directory portion of an initrd path ("/system/services/fat" ->
/// "/system/services").
fn parentOf(path: []const u8) []const u8 {
const slash = std.mem.lastIndexOfScalar(u8, path, '/') orelse return path[0..0];
if (slash == 0) return path[0..1];
return path[0..slash];
}
// --- boot wiring -------------------------------------------------------------
/// Publish the initrd as the kernel-backed /system mount and derive its bounded
/// directory table (the unique parents of the entry paths). Called once at boot.
pub fn setInitialRamdisk(image: []const u8) void {
ramdisk_image = image;
installMount("/system", .kernel_initrd, null, "");
// Directory 0 is /system itself.
directories[0] = .{ .parent = 0 };
@memcpy(directories[0].path[0..7], "/system");
directories[0].path_len = 7;
directory_count = 1;
const rd = initial_ramdisk.Reader.init(image) orelse return;
var i: u32 = 0;
while (i < rd.count) : (i += 1) {
const item = rd.entry(i) orelse continue;
// Register every ancestor directory strictly below /system.
var parent = parentOf(item.name);
while (parent.len > 7) : (parent = parentOf(parent)) {
if (directoryIndex(parent) == null and directory_count < maximum_directories) {
var d = &directories[directory_count];
@memcpy(d.path[0..parent.len], parent);
d.path_len = parent.len;
d.parent = 0; // fixed up below once all exist
directory_count += 1;
}
}
}
// Parent links (a second pass so out-of-order registration doesn't matter).
for (directories[1..directory_count]) |*d| {
d.parent = directoryIndex(parentOf(d.slice())) orelse 0;
}
}
fn directoryIndex(path: []const u8) ?usize {
for (directories[0..directory_count], 0..) |*d, i| {
if (std.mem.eql(u8, d.slice(), path)) return i;
}
return null;
}
fn installMount(prefix: []const u8, kind: MountKind, backend: ?*ipc.Endpoint, rewrite: []const u8) void {
// Remount replaces: a restarted backend re-mounts its prefix.
var slot: ?*Mount = null;
for (&mounts) |*m| {
if (m.used and std.mem.eql(u8, m.prefixSlice(), prefix)) {
if (m.backend) |old| ipc.dropRef(old);
slot = m;
break;
}
if (slot == null and !m.used) slot = m;
}
const m = slot orelse return;
m.* = .{ .used = true, .kind = kind, .backend = backend };
@memcpy(m.prefix[0..prefix.len], prefix);
m.prefix_len = prefix.len;
@memcpy(m.rewrite[0..rewrite.len], rewrite);
m.rewrite_len = rewrite.len;
}
// --- resolve -----------------------------------------------------------------
pub const Resolved = union(enum) {
/// Kernel-served: a permanent node token.
kernel_node: u64,
/// Backend-served: the endpoint plus the rewritten mount-relative path.
backend: struct { endpoint: *ipc.Endpoint, path: [maximum_rewrite + maximum_prefix + 160]u8, path_len: usize },
not_found: void,
};
/// Longest-prefix match over the mount table, then per-kind resolution.
/// `create`-intent on the immutable /system fails here (EROFS-style).
pub fn resolvePath(path: []const u8, wants_create: bool) Resolved {
if (!isAbsolute(path)) {
return .{ .not_found = {} }; // bare names have no kernel namespace (ramfs retired)
}
var best: ?*Mount = null;
var best_relative: []const u8 = undefined;
for (&mounts) |*m| {
if (!m.used) continue;
const relative = underMount(path, m.prefixSlice()) orelse continue;
if (best == null or m.prefix_len > best.?.prefix_len) {
best = m;
best_relative = relative;
}
}
const m = best orelse return .{ .not_found = {} };
switch (m.kind) {
.kernel_initrd => {
if (wants_create) return .{ .not_found = {} }; // read-only
return resolveInitrd(path);
},
.backend => {
const endpoint = m.backend orelse return .{ .not_found = {} };
if (endpoint.dead) {
// The backend died: treat the mount as gone (it re-mounts on
// restart) and release our reference lazily.
ipc.dropRef(endpoint);
m.backend = null;
m.used = false;
return .{ .not_found = {} };
}
var out: Resolved = .{ .backend = .{ .endpoint = endpoint, .path = undefined, .path_len = 0 } };
const rewrite = m.rewriteSlice();
const tail = if (std.mem.eql(u8, best_relative, "/") and rewrite.len != 0) "" else best_relative;
const total = rewrite.len + tail.len;
if (total > out.backend.path.len or total == 0) {
if (rewrite.len == 0 and tail.len == 0) return .{ .not_found = {} };
if (total > out.backend.path.len) return .{ .not_found = {} };
}
@memcpy(out.backend.path[0..rewrite.len], rewrite);
@memcpy(out.backend.path[rewrite.len..][0..tail.len], tail);
out.backend.path_len = total;
return out;
},
}
}
fn resolveInitrd(path: []const u8) Resolved {
if (directoryIndex(path)) |index| return .{ .kernel_node = token(token_kind_initrd_directory, index) };
const image = ramdisk_image orelse return .{ .not_found = {} };
const rd = initial_ramdisk.Reader.init(image) orelse return .{ .not_found = {} };
var i: u32 = 0;
while (i < rd.count) : (i += 1) {
const item = rd.entry(i) orelse continue;
if (std.mem.eql(u8, item.name, path)) return .{ .kernel_node = token(token_kind_initrd_file, i) };
}
return .{ .not_found = {} };
}
// --- fs_node: serving kernel-backed nodes ------------------------------------
/// Read `out.len` bytes of an initrd file at `offset`. Returns bytes copied
/// (0 at EOF) or null for a bad token. Lock-free: the initrd is immutable.
pub fn nodeRead(node_token: u64, offset: u64, out: []u8) ?usize {
if (tokenKind(node_token) != token_kind_initrd_file) return null;
const image = ramdisk_image orelse return null;
const rd = initial_ramdisk.Reader.init(image) orelse return null;
const item = rd.entry(@intCast(tokenIndex(node_token))) orelse return null;
if (offset >= item.blob.len) return 0;
const n = @min(out.len, item.blob.len - @as(usize, @intCast(offset)));
@memcpy(out[0..n], item.blob[@intCast(offset)..][0..n]);
return n;
}
/// A node's metadata in vfs-protocol FileStatus shape (size, kind, mtime).
pub fn nodeStatus(node_token: u64) ?abi.FileAttributes {
switch (tokenKind(node_token)) {
token_kind_initrd_file => {
const image = ramdisk_image orelse return null;
const rd = initial_ramdisk.Reader.init(image) orelse return null;
const item = rd.entry(@intCast(tokenIndex(node_token))) orelse return null;
return .{ .size = item.blob.len, .kind = abi.file_kind_regular };
},
token_kind_initrd_directory => {
if (tokenIndex(node_token) >= directory_count) return null;
return .{ .size = 0, .kind = abi.file_kind_directory };
},
else => return null,
}
}
/// The `cursor`th child of an initrd directory: fills `name_out`, returns the
/// entry header, or null past the end / bad token. Cursor enumerates
/// subdirectories first, then files whose parent is this directory — stable,
/// because the initrd is immutable.
pub fn nodeReaddir(node_token: u64, cursor: u64, name_out: []u8) ?struct { header: abi.DirectoryEntryHeader, name_len: usize } {
if (tokenKind(node_token) != token_kind_initrd_directory) return null;
const directory_index = tokenIndex(node_token);
if (directory_index >= directory_count) return null;
const self_path = directories[@intCast(directory_index)].slice();
var index: u64 = 0;
// Subdirectories whose parent is this directory.
for (directories[0..directory_count], 0..) |*d, i| {
if (i == directory_index) continue;
if (d.parent != directory_index) continue;
if (i == 0) continue;
if (index == cursor) {
const name = d.slice()[self_path.len + 1 ..];
const n = @min(name.len, name_out.len);
@memcpy(name_out[0..n], name[0..n]);
return .{ .header = .{ .kind = abi.file_kind_directory, .name_len = @intCast(n), .size = 0 }, .name_len = n };
}
index += 1;
}
// Files directly inside this directory.
const image = ramdisk_image orelse return null;
const rd = initial_ramdisk.Reader.init(image) orelse return null;
var i: u32 = 0;
while (i < rd.count) : (i += 1) {
const item = rd.entry(i) orelse continue;
if (!std.mem.eql(u8, parentOf(item.name), self_path)) continue;
if (index == cursor) {
const name = item.name[self_path.len + 1 ..];
const n = @min(name.len, name_out.len);
@memcpy(name_out[0..n], name[0..n]);
return .{ .header = .{ .kind = abi.file_kind_regular, .name_len = @intCast(n), .size = item.blob.len }, .name_len = n };
}
index += 1;
}
return null;
}
// --- mount/unmount (syscall bodies; caller resolved the handle) --------------
/// Mount `backend` at `prefix` with an optional backend-side `rewrite` prefix.
/// The endpoint reference is taken by the caller (process.zig bumps it); refuses
/// shadowing or replacing /system.
pub fn mountBackend(prefix: []const u8, backend: *ipc.Endpoint, rewrite: []const u8) bool {
if (!isAbsolute(prefix) or prefix.len < 2 or prefix.len > maximum_prefix) return false;
if (rewrite.len > maximum_rewrite) return false;
if (underMount(prefix, "/system") != null) return false; // the initrd is not shadowable
installMount(prefix, .backend, backend, rewrite);
return true;
}
pub fn unmount(prefix: []const u8) bool {
for (&mounts) |*m| {
if (m.used and m.kind == .backend and std.mem.eql(u8, m.prefixSlice(), prefix)) {
if (m.backend) |endpoint| ipc.dropRef(endpoint);
m.* = .{};
return true;
}
}
return false;
}
// --- tests (host) ------------------------------------------------------------
test "underMount matches only at path boundaries" {
try std.testing.expectEqualStrings("/", underMount("/mnt/usb", "/mnt/usb").?);
try std.testing.expectEqualStrings("/system/kernel", underMount("/mnt/usb/system/kernel", "/mnt/usb").?);
try std.testing.expect(underMount("/mnt/usbextra", "/mnt/usb") == null);
try std.testing.expect(underMount("/mnt", "/mnt/usb") == null);
try std.testing.expect(underMount("/other", "/mnt/usb") == null);
try std.testing.expect(underMount("greeting", "/mnt/usb") == null);
}
test "parentOf walks toward the root" {
try std.testing.expectEqualStrings("/system/services", parentOf("/system/services/fat"));
try std.testing.expectEqualStrings("/system", parentOf("/system/services"));
try std.testing.expectEqualStrings("/", parentOf("/system"));
}
+6
View File
@@ -624,6 +624,12 @@ CASES = [
"fail": r"DANOS-TEST-RESULT: FAIL"},
# The user-space VFS: a client opens/writes/reads a file through the rt file
# API, which IPCs the VFS server process; the round trip must match.
# The kernel VFS root (M-F): the mount table serves the initrd at /system —
# path resolution, node status/read (an ELF magic), and directory listing,
# asserted kernel-side.
{"name": "kvfs",
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
{"name": "vfs",
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},