vfs: the root moves into the kernel — resolve + redirect cutover

runtime.fs now routes every path through fs_resolve: kernel-served
/system nodes are read via fs_node (tokens, no open state); everything
under a userspace mount goes straight to the owning backend's endpoint
with the kernel-rewritten mount-relative path — one syscall of naming,
then the unchanged vfs-protocol rendezvous, public API untouched. mkdir/
unlink/rename resolve-then-forward (rename checks both paths land on
the SAME backend); mount is the fs_mount syscall.

The fat server mounts twice — /mnt/usb from the volume root and /var
from its /var subtree — so the logger now writes the FHS path
/var/log/<boot-stamp>/... and swapping the persistent medium later
touches only fat's two mount calls. With clients holding fat's node ids
directly, fat records each handle's owner, checks it, and sweeps a dead
client's handles via the published exit events (the old router's
pattern, now where the state actually lives).

The userspace vfs server and its router die; ServiceId.vfs=1 stays
reserved-retired; protocol.zig moves to system/vfs-protocol.zig (the
wire contract is backend-only now). vfs-test becomes the ring-3 proof
of the kernel VFS (own-binary ELF magic through /system, read-only
refusals, listing); vfs-client-death becomes the fat sweep test over
the full storage chain, with a ring-scanning check (the last-write
buffer is too racy under a chattering tree).
This commit is contained in:
Daniel Samson
2026-07-21 16:27:06 +01:00
parent 7c5645fe48
commit e186858315
15 changed files with 379 additions and 590 deletions
+110 -49
View File
@@ -1,5 +1,5 @@
//! runtime.fs — the danos-native file API. A program opens, reads, writes, and
//! lists files served by the user-space VFS (system/services/vfs), each call
//! lists files through the kernel VFS root (resolve + redirect), each call
//! marshalling a vfs-protocol request over IPC. This is the danos-native layer
//! danos programs use directly; it is also where the file operations that later
//! become `std.os.danos` are staged (see docs/zig-self-hosting.md). It replaces
@@ -12,6 +12,7 @@
const std = @import("std");
const ipc = @import("ipc.zig");
const system = @import("system.zig");
const protocol = @import("vfs-protocol");
/// The kind of a filesystem node — re-exported so a caller need not import the
@@ -60,23 +61,37 @@ pub const OpenOptions = struct {
}
};
// The VFS server endpoint, looked up once by well-known id and cached.
var vfs_handle: ipc.Handle = 0;
var vfs_resolved = false;
fn vfs() ?ipc.Handle {
if (!vfs_resolved) {
vfs_handle = ipc.lookup(.vfs) orelse return null;
vfs_resolved = true;
// The route to a path: the kernel resolves (fs_resolve) and either serves the
// node itself (the initrd at /system — a permanent token) or redirects us to
// the owning filesystem backend's endpoint, to which we speak the vfs-protocol
// rendezvous directly with the rewritten mount-relative path.
const Route = union(enum) {
kernel: u64,
backend: struct { handle: ipc.Handle, path: [224]u8, path_len: usize },
fn backendPath(self: *const Route) []const u8 {
return self.backend.path[0..self.backend.path_len];
}
};
fn resolve(path: []const u8, flags: usize) ?Route {
var out: [224]u8 = undefined;
const route = system.fsResolve(path, flags, &out) orelse return null;
switch (route) {
.kernel => |token| return .{ .kernel = token },
.backend => |b| {
var r: Route = .{ .backend = .{ .handle = b.handle, .path = undefined, .path_len = b.path_len } };
@memcpy(r.backend.path[0..b.path_len], out[0..b.path_len]);
return r;
},
}
return vfs_handle;
}
const Result = struct { reply: protocol.Reply, payload: []u8 };
// One request/reply round trip: [Request header][send payload] -> VFS ->
// One request/reply round trip: [Request header][send payload] -> backend ->
// [Reply header][receive payload]. The receive payload lands in `out`.
fn transact(request: protocol.Request, send: []const u8, out: []u8) ?Result {
const h = vfs() orelse return null;
fn transact(h: ipc.Handle, request: protocol.Request, send: []const u8, out: []u8) ?Result {
var message: [protocol.message_maximum]u8 = undefined;
@memcpy(message[0..protocol.request_size], std.mem.asBytes(&request));
const slen = @min(send.len, protocol.maximum_payload);
@@ -95,13 +110,21 @@ fn transact(request: protocol.Request, send: []const u8, out: []u8) ?Result {
pub const File = struct {
node: u64,
offset: u64 = 0,
/// The owning backend's endpoint, or null for a kernel-served node (the
/// read-only /system tree), whose `node` is a permanent fs_node token.
backend: ?ipc.Handle = null,
/// Read up to `buffer.len` bytes at the current offset; returns the count, or
/// null on error.
pub fn read(self: *File, buffer: []u8) ?usize {
const h = self.backend orelse {
const n = system.fsNodeRead(self.node, self.offset, buffer) orelse return null;
self.offset += n;
return n;
};
const want: u32 = @intCast(@min(buffer.len, protocol.maximum_payload));
const request = protocol.Request{ .operation = .read, .node = self.node, .offset = self.offset, .len = want, .flags = 0 };
const r = transact(request, &.{}, buffer) orelse return null;
const r = transact(h, request, &.{}, buffer) orelse return null;
if (r.reply.status != 0) return null;
self.offset += r.reply.len;
return r.reply.len;
@@ -109,11 +132,13 @@ pub const File = struct {
/// Write `data` at the current offset; returns the count written. A single
/// call is capped at the VFS payload size, so the return may be short — use
/// `writeAll` to write the whole slice. Null on error.
/// `writeAll` to write the whole slice. Null on error (kernel-served nodes
/// are read-only).
pub fn write(self: *File, data: []const u8) ?usize {
const h = self.backend orelse return null;
const want: u32 = @intCast(@min(data.len, protocol.maximum_payload));
const request = protocol.Request{ .operation = .write, .node = self.node, .offset = self.offset, .len = want, .flags = 0 };
const r = transact(request, data[0..want], &.{}) orelse return null;
const r = transact(h, request, data[0..want], &.{}) orelse return null;
if (r.reply.status != 0) return null;
self.offset += r.reply.len;
return r.reply.len;
@@ -138,27 +163,40 @@ pub const File = struct {
/// This file's metadata.
pub fn attributes(self: *File) ?Attributes {
const h = self.backend orelse {
const a = system.fsNodeStatus(self.node) orelse return null;
return .{ .size = a.size, .kind = if (a.kind == system.file_kind_directory) .directory else .regular, .mtime = a.mtime };
};
const request = protocol.Request{ .operation = .status, .node = self.node, .offset = 0, .len = 0, .flags = 0 };
var buffer: [@sizeOf(protocol.FileStatus)]u8 = undefined;
const r = transact(request, &.{}, &buffer) orelse return null;
const r = transact(h, request, &.{}, &buffer) orelse return null;
if (r.reply.status != 0 or r.payload.len < @sizeOf(protocol.FileStatus)) return null;
const status = std.mem.bytesToValue(protocol.FileStatus, buffer[0..@sizeOf(protocol.FileStatus)]);
return .{ .size = status.size, .kind = kindFromWire(status.kind), .mtime = status.mtime };
}
/// Release the VFS's open handle for this file.
/// Release the backend's open handle for this file. Kernel-served node
/// tokens are permanent — nothing to release.
pub fn close(self: *File) void {
const h = self.backend orelse return;
const request = protocol.Request{ .operation = .close, .node = self.node, .offset = 0, .len = 0, .flags = 0 };
_ = transact(request, &.{}, &.{});
_ = transact(h, request, &.{}, &.{});
}
};
/// Open (or create, with `.create`) `path`. Returns the open file, or null.
pub fn open(path: []const u8, options: OpenOptions) ?File {
const request = protocol.Request{ .operation = .open, .node = 0, .offset = 0, .len = @intCast(path.len), .flags = options.wireFlags() };
const r = transact(request, path, &.{}) orelse return null;
if (r.reply.status != 0) return null;
return .{ .node = r.reply.node };
const route = resolve(path, options.wireFlags()) orelse return null;
switch (route) {
.kernel => |token| return .{ .node = token, .backend = null },
.backend => |b| {
const relative = route.backendPath();
const request = protocol.Request{ .operation = .open, .node = 0, .offset = 0, .len = @intCast(relative.len), .flags = options.wireFlags() };
const r = transact(b.handle, request, relative, &.{}) orelse return null;
if (r.reply.status != 0) return null;
return .{ .node = r.reply.node, .backend = b.handle };
},
}
}
/// A path's metadata without keeping it open (open -> status -> close).
@@ -190,13 +228,27 @@ pub const Entry = struct {
pub const Directory = struct {
node: u64,
cursor: u64 = 0,
backend: ?ipc.Handle = null,
/// Fill `entry` with the next directory entry; false at end of directory or
/// on error.
pub fn next(self: *Directory, entry: *Entry) bool {
const h = self.backend orelse {
var buffer: [@sizeOf(system.DirectoryEntryHeader) + 64]u8 = undefined;
const n = system.fsNodeReaddir(self.node, self.cursor, &buffer) orelse return false;
if (n < @sizeOf(system.DirectoryEntryHeader)) return false; // end
const header = std.mem.bytesToValue(system.DirectoryEntryHeader, buffer[0..@sizeOf(system.DirectoryEntryHeader)]);
entry.kind = if (header.kind == system.file_kind_directory) .directory else .regular;
entry.size = header.size;
const nlen = @min(@as(usize, header.name_len), entry.name_buffer.len);
@memcpy(entry.name_buffer[0..nlen], buffer[@sizeOf(system.DirectoryEntryHeader)..][0..nlen]);
entry.name_len = nlen;
self.cursor += 1;
return true;
};
const request = protocol.Request{ .operation = .readdir, .node = self.node, .offset = self.cursor, .len = 0, .flags = 0 };
var buffer: [protocol.message_maximum]u8 = undefined;
const r = transact(request, &.{}, &buffer) orelse return false;
const r = transact(h, request, &.{}, &buffer) orelse return false;
if (r.reply.status != 0 or r.reply.len == 0) return false; // error or EOF
if (r.payload.len < protocol.directory_entry_size) return false;
const header = std.mem.bytesToValue(protocol.DirectoryEntry, r.payload[0..protocol.directory_entry_size]);
@@ -210,9 +262,9 @@ pub const Directory = struct {
return true;
}
/// Release the VFS's open handle for this directory.
/// Release the backend's open handle for this directory.
pub fn close(self: *Directory) void {
var f = File{ .node = self.node };
var f = File{ .node = self.node, .backend = self.backend };
f.close();
}
};
@@ -220,13 +272,18 @@ pub const Directory = struct {
/// Open `path` as a directory for listing. Returns null if it isn't one / on error.
pub fn openDirectory(path: []const u8) ?Directory {
const file = open(path, .{ .directory = true }) orelse return null;
return .{ .node = file.node };
return .{ .node = file.node, .backend = file.backend };
}
// A path-based request that returns only a status (mkdir, unlink).
// A path-based request that returns only a status (mkdir, unlink). Kernel-served
// paths (the read-only /system) refuse mutation by construction: the resolve
// must land on a backend.
fn pathOperation(operation: protocol.Operation, path: []const u8) bool {
const request = protocol.Request{ .operation = operation, .node = 0, .offset = 0, .len = @intCast(path.len), .flags = 0 };
const r = transact(request, path, &.{}) orelse return false;
const route = resolve(path, 0) orelse return false;
if (route != .backend) return false;
const relative = route.backendPath();
const request = protocol.Request{ .operation = operation, .node = 0, .offset = 0, .len = @intCast(relative.len), .flags = 0 };
const r = transact(route.backend.handle, request, relative, &.{}) orelse return false;
return r.reply.status == 0;
}
@@ -261,33 +318,37 @@ pub fn remove(path: []const u8) bool {
return pathOperation(.unlink, path);
}
/// Rename `old_path` to `new_path`. Both must be in the same directory (same-
/// directory, 8.3-name rename only for now). Returns true on success.
/// Rename `old_path` to `new_path`. Both must resolve to the SAME filesystem
/// backend (same-directory, 8.3-name rename only for now). Returns true on
/// success.
pub fn rename(old_path: []const u8, new_path: []const u8) bool {
const total = old_path.len + 1 + new_path.len;
const old_route = resolve(old_path, 0) orelse return false;
const new_route = resolve(new_path, 0) orelse return false;
if (old_route != .backend or new_route != .backend) return false;
if (old_route.backend.handle != new_route.backend.handle) return false; // cross-filesystem
const old_relative = old_route.backendPath();
const new_relative = new_route.backendPath();
const total = old_relative.len + 1 + new_relative.len;
if (total > protocol.maximum_payload) return false;
var payload: [protocol.maximum_payload]u8 = undefined;
@memcpy(payload[0..old_path.len], old_path);
payload[old_path.len] = 0;
@memcpy(payload[old_path.len + 1 ..][0..new_path.len], new_path);
@memcpy(payload[0..old_relative.len], old_relative);
payload[old_relative.len] = 0;
@memcpy(payload[old_relative.len + 1 ..][0..new_relative.len], new_relative);
const request = protocol.Request{ .operation = .rename, .node = 0, .offset = 0, .len = @intCast(total), .flags = 0 };
const r = transact(request, payload[0..total], &.{}) orelse return false;
const r = transact(old_route.backend.handle, request, payload[0..total], &.{}) orelse return false;
return r.reply.status == 0;
}
/// Mount a filesystem backend (its server endpoint) at absolute path `target`;
/// the VFS then routes everything under `target` to that backend. This is the one
/// call that hands the VFS a capability (the backend endpoint). Returns true on
/// success.
/// the kernel VFS then routes everything under `target` to that backend.
/// Possession of the endpoint handle is the capability. Returns true on success.
pub fn mount(target: []const u8, backend: ipc.Handle) bool {
const h = vfs() orelse return false;
const request = protocol.Request{ .operation = .mount, .node = 0, .offset = 0, .len = @intCast(target.len), .flags = 0 };
var message: [protocol.message_maximum]u8 = undefined;
@memcpy(message[0..protocol.request_size], std.mem.asBytes(&request));
const tlen = @min(target.len, protocol.maximum_payload);
@memcpy(message[protocol.request_size..][0..tlen], target[0..tlen]);
var rbuf: [protocol.message_maximum]u8 = undefined;
const result = ipc.callCap(h, message[0 .. protocol.request_size + tlen], &rbuf, backend) catch return false;
if (result.len < protocol.reply_size) return false;
return std.mem.bytesToValue(protocol.Reply, rbuf[0..protocol.reply_size]).status == 0;
return system.fsMount(target, backend, "");
}
/// As `mount`, with a backend-side rewrite prefix: a path under `target` reaches
/// the backend as `rewrite` + the mount-relative tail. How one volume serves two
/// mounts ("/mnt/usb" from its root, "/var" from its /var subtree).
pub fn mountRewritten(target: []const u8, backend: ipc.Handle, rewrite: []const u8) bool {
return system.fsMount(target, backend, rewrite);
}