vfs: the root moves into the kernel — resolve + redirect cutover

runtime.fs now routes every path through fs_resolve: kernel-served
/system nodes are read via fs_node (tokens, no open state); everything
under a userspace mount goes straight to the owning backend's endpoint
with the kernel-rewritten mount-relative path — one syscall of naming,
then the unchanged vfs-protocol rendezvous, public API untouched. mkdir/
unlink/rename resolve-then-forward (rename checks both paths land on
the SAME backend); mount is the fs_mount syscall.

The fat server mounts twice — /mnt/usb from the volume root and /var
from its /var subtree — so the logger now writes the FHS path
/var/log/<boot-stamp>/... and swapping the persistent medium later
touches only fat's two mount calls. With clients holding fat's node ids
directly, fat records each handle's owner, checks it, and sweeps a dead
client's handles via the published exit events (the old router's
pattern, now where the state actually lives).

The userspace vfs server and its router die; ServiceId.vfs=1 stays
reserved-retired; protocol.zig moves to system/vfs-protocol.zig (the
wire contract is backend-only now). vfs-test becomes the ring-3 proof
of the kernel VFS (own-binary ELF magic through /system, read-only
refusals, listing); vfs-client-death becomes the fat sweep test over
the full storage chain, with a ring-scanning check (the last-write
buffer is too racy under a chattering tree).
This commit is contained in:
Daniel Samson
2026-07-21 16:27:06 +01:00
parent 7c5645fe48
commit e186858315
15 changed files with 379 additions and 590 deletions
+49 -16
View File
@@ -265,6 +265,31 @@ fn bufferHas(needle: []const u8) bool {
return std.mem.indexOf(u8, process.write_buffer[0..process.write_len], needle) != null;
}
/// Substring search across the whole retained log ring (record payloads are
/// contiguous in the stream, so a one-line needle always matches if present).
/// Unlike `bufferHas` (the single LAST write), this survives busy-tree chatter.
fn ringHas(needle: []const u8) bool {
var chunk: [1024]u8 = undefined;
var overlap: [128]u8 = undefined;
var overlap_len: usize = 0;
var offset = kernel_log.status().tail;
while (true) {
const n = kernel_log.readAt(offset, &chunk) orelse return false;
if (n == 0) return false;
offset += n;
// Search the previous tail glued to this chunk, then the chunk itself.
if (overlap_len != 0) {
var glued: [1152]u8 = undefined;
@memcpy(glued[0..overlap_len], overlap[0..overlap_len]);
const m = @min(n, glued.len - overlap_len);
@memcpy(glued[overlap_len..][0..m], chunk[0..m]);
if (std.mem.indexOf(u8, glued[0 .. overlap_len + m], needle) != null) return true;
} else if (std.mem.indexOf(u8, chunk[0..n], needle) != null) return true;
overlap_len = @min(n, @min(overlap.len, needle.len));
@memcpy(overlap[0..overlap_len], chunk[n - overlap_len ..][0..overlap_len]);
}
}
/// How many `pci_device` functions the devices broker currently holds. A durable
/// snapshot, unlike a `bufferHas` poll of the single-latest write_buffer line, so a
/// test can wait on it without racing transient log output. `scratch` is
@@ -2120,11 +2145,11 @@ fn claimReleaseTest(boot_information: *const BootInformation) void {
result();
}
/// M17.3: the published exit events, proven by their first subscriber. The VFS
/// subscribes at startup; a client opens a file and parks holding the handle;
/// the kill posts the exit event to the VFS's endpoint; the VFS releases the
/// dead client's handle and says so — the service-side mirror of iron rule 1
/// (a service must never depend on clients cleaning up after themselves).
/// M17.3: the published exit events, proven by a stateful server. The fat
/// server subscribes at startup; a client opens a file on the volume and parks
/// holding the handle; the kill posts the exit event to fat's endpoint; fat
/// releases the dead client's handle and says so — the service-side mirror of
/// iron rule 1 (a service must never depend on clients cleaning up).
fn vfsClientDeathTest(boot_information: *const BootInformation) void {
log("DANOS-TEST-BEGIN: vfs-client-death\n", .{});
if (boot_information.initial_ramdisk_len == 0) {
@@ -2140,7 +2165,11 @@ fn vfsClientDeathTest(boot_information: *const BootInformation) void {
};
process.write_count = 0;
check("vfs spawned", spawnNamed(rd, "vfs"));
// The full tree: the storage chain must come up for /mnt/usb to exist —
// the fat server (not a router) now owns client file state and its sweep.
process.setInitialRamdisk(image);
const init_ok = if (process.spawnBundled("/system/services/init")) true else |_| false;
check("init spawned (boots the storage chain)", init_ok);
const me = scheduler.currentId();
const endpoint = ipcsync.createIpcEndpoint() orelse {
@@ -2158,16 +2187,18 @@ fn vfsClientDeathTest(boot_information: *const BootInformation) void {
}
check("parked client spawned (supervised)", client != 0);
// Its heartbeat is the fence: once it beats, the handle is open.
// Its heartbeat is the fence: once it beats, the handle is open. The park
// waits out the whole USB->block->fat chain, so give it room; with the full
// tree chattering, the ring (not the last-write buffer) is the evidence.
const parked = "vfstest: parked";
scheduler.setPriority(1);
var deadline = architecture.millis() + 10000;
var deadline = architecture.millis() + 30000;
while (architecture.millis() < deadline) {
if (bufferHas(parked)) break;
if (ringHas(parked)) break;
scheduler.yield();
}
scheduler.setPriority(4);
check("client parked holding an open handle", bufferHas(parked));
check("client parked holding an open handle", ringHas(parked));
check("the kill is accepted", process.killProcess(me, client) == 0);
var badge: u64 = 0;
@@ -2175,16 +2206,17 @@ fn vfsClientDeathTest(boot_information: *const BootInformation) void {
_ = ipcsync.replyWait(endpoint, 0, 0, 0, 0, abi.no_cap, &badge, &received_cap);
check("the exit notification arrived", badge == abi.notify_badge_bit | abi.notify_exit_bit | client);
// The VFS heard the same published event; its release line is the proof.
// The fat server heard the same published event; its release line in the
// ring is the proof.
const released = "released 1 handle(s) for dead client";
scheduler.setPriority(1);
deadline = architecture.millis() + 10000;
while (architecture.millis() < deadline) {
if (bufferHas(released)) break;
if (ringHas(released)) break;
scheduler.yield();
}
scheduler.setPriority(4);
check("the VFS released the dead client's handle", bufferHas(released));
check("the fat server released the dead client's handle", ringHas(released));
result();
}
@@ -2683,9 +2715,10 @@ fn vfsTest(boot_information: *const BootInformation) void {
process.write_count = 0;
process.write_from_user = false;
// Spawn just the server and its client (other initial_ramdisk binaries would write to
// the shared evidence buffer and confuse the marker check).
_ = spawnNamed(rd, "vfs");
// Seed the kernel VFS (/system) — the router the client exercises.
process.setInitialRamdisk(image);
// Spawn just the client: the kernel itself is the VFS root it exercises
// (resolve + fs_node over /system through the plain runtime.fs API).
_ = spawnNamed(rd, "vfs-test");
// Wait for the client's success heartbeat (it round-trips, then beats ~1/s).