vfs: the root moves into the kernel — resolve + redirect cutover

runtime.fs now routes every path through fs_resolve: kernel-served
/system nodes are read via fs_node (tokens, no open state); everything
under a userspace mount goes straight to the owning backend's endpoint
with the kernel-rewritten mount-relative path — one syscall of naming,
then the unchanged vfs-protocol rendezvous, public API untouched. mkdir/
unlink/rename resolve-then-forward (rename checks both paths land on
the SAME backend); mount is the fs_mount syscall.

The fat server mounts twice — /mnt/usb from the volume root and /var
from its /var subtree — so the logger now writes the FHS path
/var/log/<boot-stamp>/... and swapping the persistent medium later
touches only fat's two mount calls. With clients holding fat's node ids
directly, fat records each handle's owner, checks it, and sweeps a dead
client's handles via the published exit events (the old router's
pattern, now where the state actually lives).

The userspace vfs server and its router die; ServiceId.vfs=1 stays
reserved-retired; protocol.zig moves to system/vfs-protocol.zig (the
wire contract is backend-only now). vfs-test becomes the ring-3 proof
of the kernel VFS (own-binary ELF magic through /system, read-only
refusals, listing); vfs-client-death becomes the fat sweep test over
the full storage chain, with a ring-scanning check (the last-write
buffer is too racy under a chattering tree).
This commit is contained in:
Daniel Samson
2026-07-21 16:27:06 +01:00
parent 7c5645fe48
commit e186858315
15 changed files with 379 additions and 590 deletions
+39 -15
View File
@@ -101,18 +101,42 @@ fn initialise(endpoint: runtime.ipc.Handle) bool {
};
std.log.info("mounted FAT ({s}, {d} clusters, partition lba {d})", .{ @tagName(filesystem.geometry.fat_type), filesystem.geometry.cluster_count, filesystem.base_lba });
// Mount ourselves into the VFS namespace at /mnt/usb (retry while the VFS
// comes up). From here the VFS routes /mnt/usb/... to this server.
var tries: u32 = 0;
while (tries < 100) : (tries += 1) {
if (runtime.fs.mount(mount_point, endpoint)) {
std.log.info("mounted {s}", .{mount_point});
return true;
}
runtime.system.sleep(50);
// With the router in the kernel, clients hold OUR node ids directly; sweep
// a dead client's open handles via the published exit events (the pattern
// the old userspace router used for its own table).
_ = runtime.process.subscribeExits(endpoint);
// Mount ourselves into the kernel VFS at /mnt/usb — and serve /var from the
// volume's /var subtree, so FHS paths (the logger's /var/log) stay decoupled
// from which volume carries them. A mount is one syscall now; no retry
// needed (the kernel's table exists before any service).
if (runtime.fs.mount(mount_point, endpoint)) {
std.log.info("mounted {s}", .{mount_point});
} else {
_ = runtime.system.write("/system/services/fat: could not mount /mnt/usb\n");
}
_ = runtime.system.write("/system/services/fat: could not mount into the VFS\n");
return true; // still serve directly, even if the namespace mount didn't take
if (runtime.fs.mountRewritten("/var", endpoint, "/var")) {
std.log.info("mounted /var", .{});
} else {
_ = runtime.system.write("/system/services/fat: could not mount /var\n");
}
return true;
}
/// A subscribed process-exit event: release every open handle the dead client
/// held, so a crashed reader can't pin table slots (or, later, locks).
fn onNotification(badge: u64) void {
const got = runtime.ipc.Received{ .len = 0, .badge = badge, .cap = null };
if (!got.isChildExit()) return;
const dead = got.childProcessId();
var released: u32 = 0;
for (&open_nodes) |*o| {
if (o.used and o.owner == dead) {
o.* = .{};
released += 1;
}
}
if (released != 0) std.log.info("released {d} handle(s) for dead client {d}", .{ released, dead });
}
const ParentLeaf = struct { parent: []const u8, leaf: []const u8 };
@@ -127,7 +151,7 @@ fn splitParent(path: []const u8) ParentLeaf {
};
}
fn handleOpen(out: []u8, path: []const u8, flags: u32) usize {
fn handleOpen(out: []u8, path: []const u8, flags: u32, sender: u32) usize {
var node = filesystem.resolve(path);
if (node == null and flags & protocol.create != 0) {
const split = splitParent(path);
@@ -141,13 +165,12 @@ fn handleOpen(out: []u8, path: []const u8, flags: u32) usize {
filesystem.truncate(&resolved);
}
const index = allocOpen() orelse return fail(out);
open_nodes[index] = .{ .used = true, .node = resolved };
open_nodes[index] = .{ .used = true, .node = resolved, .owner = sender };
return writeReply(out, .{ .status = 0, .node = index }, &.{});
}
fn onMessage(message: []const u8, out: []u8, sender: u32, capability: ?runtime.ipc.Handle) usize {
_ = capability;
_ = sender;
if (message.len < protocol.request_size) return fail(out);
const request = std.mem.bytesToValue(protocol.Request, message[0..protocol.request_size]);
const payload = message[protocol.request_size..];
@@ -157,7 +180,7 @@ fn onMessage(message: []const u8, out: []u8, sender: u32, capability: ?runtime.i
filesystem.current_time_epoch = runtime.system.wallClock();
switch (request.operation) {
.open => return handleOpen(out, payload[0..@min(payload.len, request.len)], request.flags),
.open => return handleOpen(out, payload[0..@min(payload.len, request.len)], request.flags, sender),
.read => {
const o = openAt(request.node) orelse return fail(out);
var buffer: [protocol.maximum_payload]u8 = undefined;
@@ -237,5 +260,6 @@ pub fn main() void {
.service = .fat,
.init = initialise,
.on_message = onMessage,
.on_notification = onNotification,
});
}
-1
View File
@@ -28,7 +28,6 @@ const build_options = @import("build_options");
/// future init reads this from a manifest under /system/services instead of a
/// hardcoded list.)
const boot_services = [_][]const u8{
"/system/services/vfs",
"/system/services/input",
"/system/services/device-manager",
"/system/services/fat",
+8 -6
View File
@@ -35,9 +35,11 @@ const runtime = @import("runtime");
const system = runtime.system;
const fs = runtime.fs;
/// Where log trees live. Flips to "/var/log" when the kernel VFS routes /var
/// to the flash volume (M-G); today the FAT service mounts at /mnt/usb only.
const base = "/mnt/usb/var/log";
/// Where log trees live: the FHS path. The kernel VFS routes /var to whatever
/// volume the fat server mounted there (today: the /var subtree of the USB
/// flash volume) — swapping the persistent medium later touches fat's two
/// mount calls, never this constant.
const base = "/var/log";
/// Drain cadence and the quiet period after which files are closed (flushed).
const tick_ms = 250;
@@ -119,9 +121,9 @@ fn onTerminate() void {
fn tick() void {
if (!storage_ready) {
// Probe the mount; the ring buffers until it appears. `exists` on the
// mount root is the documented readiness check.
if (!fs.exists("/mnt/usb")) return;
// makePath doubles as the readiness probe: while /var is unmounted the
// resolve fails fast (no storage round trip) and the ring buffers; the
// first success creates the whole per-boot tree.
if (!fs.makePath(boot_directory[0..boot_directory_len])) return;
storage_ready = true;
if (!announced) {
+89
View File
@@ -0,0 +1,89 @@
//! /system/tests/vfs-test — a ring-3 client that proves the kernel VFS end to
//! end through the plain `runtime.fs` API: resolve its OWN binary under the
//! kernel-served /system mount, check its metadata, read its ELF magic, and
//! list /system/services. On success it heartbeats "vfstest: ok" so the kernel
//! test can observe it; on failure it reports what went wrong.
//!
//! The "park" role (the fat-client-death test): open a file on the FAT volume,
//! then hold the handle forever without closing — the kill and the fat
//! server's release-on-death sweep are the point.
const std = @import("std");
const runtime = @import("runtime");
const fs = runtime.fs;
pub fn main(init: runtime.process.Init) void {
if (init.arguments.count > 1) {
park();
return;
}
// Our own binary, resolved through the kernel mount table.
const self_path = "/system/tests/vfs-test";
var file = fs.open(self_path, .{}) orelse {
_ = runtime.system.write("vfstest: open of own binary failed\n");
return;
};
defer file.close();
const attributes = file.attributes() orelse {
_ = runtime.system.write("vfstest: attributes failed\n");
return;
};
if (attributes.kind != .regular or attributes.size == 0) {
_ = runtime.system.write("vfstest: bad attributes\n");
return;
}
var header: [4]u8 = undefined;
const n = file.read(&header) orelse 0;
if (n != 4 or header[0] != 0x7f or header[1] != 'E' or header[2] != 'L' or header[3] != 'F') {
_ = runtime.system.write("vfstest: ELF magic mismatch\n");
return;
}
// The write refusal: /system is read-only by construction.
if (file.write("x") != null or fs.open("/system/tests/new-file", .{ .create = true }) != null) {
_ = runtime.system.write("vfstest: /system accepted a write\n");
return;
}
// Listing: /system/services contains init.
var saw_init = false;
if (fs.openDirectory("/system/services")) |listing| {
var directory = listing;
defer directory.close();
var entry: fs.Entry = .{};
while (directory.next(&entry)) {
if (std.mem.eql(u8, entry.name(), "init")) saw_init = true;
}
}
if (!saw_init) {
_ = runtime.system.write("vfstest: /system/services listing missed init\n");
return;
}
while (true) {
_ = runtime.system.write("vfstest: ok\n");
runtime.system.sleep(1000);
}
}
fn park() void {
// The storage chain (usb -> block -> fat -> mounts) takes a few seconds;
// retry until the volume appears.
var parked: ?fs.File = null;
var tries: u32 = 0;
while (parked == null and tries < 1000) : (tries += 1) {
parked = fs.open("/mnt/usb/parked", .{ .create = true });
if (parked == null) runtime.system.sleep(20);
}
if (parked == null) {
_ = runtime.system.write("vfstest: park open failed\n");
return;
}
while (true) {
_ = runtime.system.write("vfstest: parked\n");
runtime.system.sleep(500);
}
}
-39
View File
@@ -1,39 +0,0 @@
//! Pure path utilities for the VFS mount router — no IPC, no state, so they are
//! host-testable in isolation. The router uses these to decide whether an opened
//! path lies under a mount point and, if so, what it looks like relative to that
//! mount.
const std = @import("std");
/// If `path` lies under `mount_prefix` — equal to it, or the prefix followed by a
/// path separator — return the path relative to the mount ("/" for an exact
/// match, otherwise the tail beginning with '/'). Returns null when `path` is not
/// under the mount, so a prefix like "/mnt/usb" never captures "/mnt/usbextra".
pub fn underMount(path: []const u8, mount_prefix: []const u8) ?[]const u8 {
if (path.len < mount_prefix.len) return null;
if (!std.mem.eql(u8, path[0..mount_prefix.len], mount_prefix)) return null;
if (path.len == mount_prefix.len) return "/";
if (path[mount_prefix.len] != '/') return null;
return path[mount_prefix.len..];
}
/// Whether `path` is absolute (rooted at '/'). Bare names — what the flat ramfs
/// uses — are relative and never route through a mount.
pub fn isAbsolute(path: []const u8) bool {
return path.len > 0 and path[0] == '/';
}
test "underMount matches only at path boundaries" {
try std.testing.expectEqualStrings("/", underMount("/mnt/usb", "/mnt/usb").?);
try std.testing.expectEqualStrings("/system/kernel", underMount("/mnt/usb/system/kernel", "/mnt/usb").?);
try std.testing.expect(underMount("/mnt/usbextra", "/mnt/usb") == null); // not a boundary
try std.testing.expect(underMount("/mnt", "/mnt/usb") == null); // shorter than the prefix
try std.testing.expect(underMount("/other", "/mnt/usb") == null);
try std.testing.expect(underMount("greeting", "/mnt/usb") == null); // a bare name
}
test "isAbsolute distinguishes paths from bare names" {
try std.testing.expect(isAbsolute("/mnt/usb"));
try std.testing.expect(!isAbsolute("greeting"));
try std.testing.expect(!isAbsolute(""));
}
-114
View File
@@ -1,114 +0,0 @@
//! The VFS wire protocol — the message format spoken between a client (via the file
//! API) and the user-space VFS server over IPC. A request is a fixed `Request` header
//! followed by an inline payload (a path, or write bytes); a reply is a fixed `Reply`
//! header followed by an inline payload (read bytes, or a FileStatus). Everything fits
//! in one IPC message (<= ipc MESSAGE_MAXIMUM = 256 bytes).
//!
//! This is a danos-native contract, so it uses danos names throughout. The client
//! side is `runtime.fs` (library/runtime/fs.zig), which programs use directly.
//!
//! This is user-space only — the kernel knows nothing of files or paths; it only moves the bytes.
//! Shared by library/runtime/fs.zig (client) and system/services/vfs/vfs.zig (server).
pub const Operation = enum(u32) {
open, // open(path) -> node id
close, // close(node)
read, // read(node, offset, len) -> bytes
write, // write(node, offset, bytes) -> count
status, // status(node) -> FileStatus
// Appended for the mount router (M5). Values stay stable, so existing clients
// and the flat-ramfs tests are unaffected.
readdir, // readdir(dir_node, cursor=offset) -> one DirectoryEntry (len==0 => EOF)
mount, // mount(prefix payload, capability = backend endpoint)
unmount, // unmount(prefix payload)
// Appended for filesystem mutation (Phase 2). Path-based (the path is the
// payload); a mounted backend handles them, the flat ramfs refuses them.
mkdir, // mkdir(path payload) -> status
unlink, // unlink(path payload) -> status
// rename: the payload is the old path, a single 0x00 separator, then the new
// path. Same-directory rename only (the router requires both under one mount).
rename, // rename(old\0new payload) -> status
};
/// The type of a filesystem node, aligned to the FSH file-type table
/// (docs/danos-file-system-hierarchy-FSH.md). Fills `FileStatus.kind` and
/// `DirectoryEntry.kind`; `regular = 0` keeps the historical hardcoded value.
pub const NodeKind = enum(u32) {
regular = 0,
directory = 1,
character_device = 2,
block_device = 3,
symbolic_link = 4,
fifo = 5,
socket = 6,
};
/// One directory entry, returned by `readdir`: a fixed header followed inline in
/// the reply payload by `name_len` bytes of name. A zero-length reply is EOF.
pub const DirectoryEntry = extern struct {
kind: u32, // a NodeKind
name_len: u32,
size: u64,
};
pub const directory_entry_size: usize = @sizeOf(DirectoryEntry);
/// Request header. `node` is the server-side open-file id (from a prior open);
/// for `open` the path is the payload and `len` is its length. `offset`/`len`
/// carry the read/write position and count.
pub const Request = extern struct {
operation: Operation,
node: u64,
offset: u64,
len: u32,
flags: u32,
};
/// Reply header. `status` is 0 on success or a negative errno; `node` is the new
/// open-file id (for `open`); `len` is the payload length (bytes read, or the
/// FileStatus size).
pub const Reply = extern struct {
status: i32,
_padding: u32 = 0,
node: u64 = 0,
len: u32 = 0,
_padding2: u32 = 0,
};
/// A file's metadata (the danos-native answer to a `status` request). The POSIX
/// layer maps this onto `struct stat`.
pub const FileStatus = extern struct {
size: u64,
kind: u32,
_padding: u32 = 0,
/// Modification time — Unix epoch seconds, UTC. 0 if the backend has none (the
/// flat ramfs). Filled from the FAT directory entry's write date/time.
mtime: u64 = 0,
};
pub const message_maximum: usize = 256;
pub const request_size: usize = @sizeOf(Request);
pub const reply_size: usize = @sizeOf(Reply);
/// Largest inline payload that still fits one IPC message alongside a header.
pub const maximum_payload: usize = message_maximum - request_size;
/// Open flags (danos-native; `runtime.fs.OpenOptions` maps its booleans onto these).
pub const create: u32 = 1;
/// Open a directory (for readdir) rather than a file. A mounted backend uses
/// this to open a directory node; the flat ramfs ignores it.
pub const directory: u32 = 2;
/// Truncate the file to zero length on open (O_TRUNC): replace its contents rather
/// than overwriting in place, so a shorter new file leaves no stale tail. A mounted
/// backend frees the old cluster chain; the flat ramfs ignores it.
pub const truncate: u32 = 4;
test "protocol struct sizes and node kinds" {
const std = @import("std");
try std.testing.expectEqual(@as(u32, 0), @intFromEnum(NodeKind.regular));
try std.testing.expectEqual(@as(u32, 1), @intFromEnum(NodeKind.directory));
try std.testing.expectEqual(@as(usize, 16), @sizeOf(DirectoryEntry));
// The appended operations keep the original values.
try std.testing.expectEqual(@as(u32, 0), @intFromEnum(Operation.open));
try std.testing.expectEqual(@as(u32, 4), @intFromEnum(Operation.status));
try std.testing.expectEqual(@as(u32, 5), @intFromEnum(Operation.readdir));
}
-62
View File
@@ -1,62 +0,0 @@
//! /system/services/vfs/vfs-test — a client that proves the VFS round trip end to end: open a
//! file through the `runtime.fs` file API, write to it, seek back, read it, and compare.
//! On success it heartbeats "vfstest: ok" so the kernel test can observe it;
//! on failure it reports what went wrong. Shipped in the initial_ramdisk alongside vfs.
const std = @import("std");
const runtime = @import("runtime");
const fs = runtime.fs;
pub fn main(init: runtime.process.Init) void {
const payload = "hello-vfs";
// The "park" role (the vfs-client-death test): open a file, then hold the
// handle forever without closing — the kill and the VFS's release-on-death
// are the point.
if (init.arguments.count > 1) {
var parked: ?fs.File = null;
var tries: u32 = 0;
while (parked == null and tries < 200) : (tries += 1) {
parked = fs.open("parked", .{ .create = true });
if (parked == null) runtime.system.sleep(20);
}
if (parked == null) {
_ = runtime.system.write("vfstest: park open failed\n");
return;
}
while (true) {
_ = runtime.system.write("vfstest: parked\n");
runtime.system.sleep(500);
}
}
// The VFS server may not have registered yet — retry open until it's up.
var opened: ?fs.File = null;
var tries: u32 = 0;
while (opened == null and tries < 200) : (tries += 1) {
opened = fs.open("greeting", .{ .create = true });
if (opened == null) runtime.system.sleep(20);
}
var greeting = opened orelse {
_ = runtime.system.write("vfstest: open failed\n");
return;
};
if ((greeting.write(payload) orelse 0) != payload.len) {
_ = runtime.system.write("vfstest: write failed\n");
return;
}
greeting.seekTo(0);
var buffer: [32]u8 = undefined;
const n = greeting.read(&buffer) orelse 0;
greeting.close();
if (n == payload.len and std.mem.eql(u8, buffer[0..n], payload)) {
while (true) {
_ = runtime.system.write("vfstest: ok\n");
runtime.system.sleep(1000);
}
}
_ = runtime.system.write("vfstest: mismatch\n");
}
-389
View File
@@ -1,389 +0,0 @@
//! system/services/vfs — the user-space VFS server. Shipped in the initial_ramdisk, spawned as a
//! ring-3 process, and reached by every other process through IPC (the `runtime`
//! file API marshals open/read/write/stat/close into calls to this server's
//! endpoint, published under the well-known `vfs` service id).
//!
//! Two namespaces meet here (M5):
//! - a small in-memory **ramfs** — opening a bare name creates it — enough to
//! prove the round trip and to back the existing tests;
//! - **mounted filesystems**: a mount table maps an absolute path prefix (e.g.
//! `/mnt/usb`) to a backend server's endpoint. An open of a path under a mount
//! is *forwarded* to that backend (which speaks this same protocol), and every
//! later read/write/status/readdir/close on the resulting handle is relayed to
//! it. The VFS is the router; a filesystem (FAT) is the backend.
//!
//! A path routes through a mount only when it is absolute and lies under a mount
//! prefix; bare names always resolve in the flat ramfs — the backward-compat
//! contract the `vfs` / `vfs-client-death` tests rely on.
const std = @import("std");
const runtime = @import("runtime");
const protocol = runtime.vfs_protocol;
const path = @import("path.zig");
const ipc = runtime.ipc;
const Node = struct {
used: bool = false,
name: [24]u8 = undefined,
name_len: usize = 0,
data: [512]u8 = undefined,
size: usize = 0,
};
const OpenFile = struct {
used: bool = false,
// For a local handle: an index into `nodes`. For a forwarding handle: the
// node id the backend returned. (usize == u64 here, so it holds either.)
node: usize = 0,
// Non-null for a handle that forwards to a mounted backend.
backend: ?ipc.Handle = null,
// The client (task id — an IPC badge is one) that opened this handle. What
// release-on-death sweeps by: a service must never depend on its clients
// cleaning up after themselves (docs/process-lifecycle.md).
owner: u32 = 0,
};
// One mounted filesystem: an absolute path prefix and the backend endpoint that
// serves everything under it.
const Mount = struct {
used: bool = false,
prefix: [64]u8 = undefined,
prefix_len: usize = 0,
backend: ipc.Handle = 0,
};
var nodes = [_]Node{.{}} ** 8;
var opens = [_]OpenFile{.{}} ** 16;
var mounts = [_]Mount{.{}} ** 8;
fn findNode(name: []const u8) ?usize {
for (&nodes, 0..) |*n, i| {
if (n.used and std.mem.eql(u8, n.name[0..n.name_len], name)) return i;
}
return null;
}
fn createNode(name: []const u8) ?usize {
for (&nodes, 0..) |*n, i| {
if (!n.used) {
const l = @min(name.len, n.name.len);
@memcpy(n.name[0..l], name[0..l]);
n.* = .{ .used = true, .name = n.name, .name_len = l, .size = 0 };
return i;
}
}
return null;
}
fn openAt(id: u64) ?*OpenFile {
if (id >= opens.len) return null;
const o = &opens[@intCast(id)];
return if (o.used) o else null;
}
/// The mount whose prefix most specifically contains `name`, and the path
/// relative to it. Only absolute paths route; bare names never match.
const MountMatch = struct { backend: ipc.Handle, relative: []const u8 };
fn longestMount(name: []const u8) ?MountMatch {
if (!path.isAbsolute(name)) return null;
var best: ?MountMatch = null;
var best_len: usize = 0;
for (&mounts) |*m| {
if (!m.used) continue;
const prefix = m.prefix[0..m.prefix_len];
if (path.underMount(name, prefix)) |relative| {
if (best == null or prefix.len >= best_len) {
best_len = prefix.len;
best = .{ .backend = m.backend, .relative = relative };
}
}
}
return best;
}
/// Serialise a reply header + payload into `out`; returns the total length.
fn writeReply(out: []u8, reply: protocol.Reply, payload: []const u8) usize {
@memcpy(out[0..protocol.reply_size], std.mem.asBytes(&reply));
const n = @min(payload.len, out.len - protocol.reply_size);
@memcpy(out[protocol.reply_size..][0..n], payload[0..n]);
return protocol.reply_size + n;
}
fn fail(out: []u8) usize {
return writeReply(out, .{ .status = -1 }, &.{});
}
// --- mount routing ----------------------------------------------------------
/// Forward an open under a mount to its backend and, on success, allocate a local
/// forwarding handle that remembers the backend's node id.
fn forwardOpen(out: []u8, backend: ipc.Handle, relative: []const u8, flags: u32, sender: u32) usize {
const request = protocol.Request{ .operation = .open, .node = 0, .offset = 0, .len = @intCast(relative.len), .flags = flags };
var message: [protocol.message_maximum]u8 = undefined;
@memcpy(message[0..protocol.request_size], std.mem.asBytes(&request));
const rel = relative[0..@min(relative.len, protocol.maximum_payload)];
@memcpy(message[protocol.request_size..][0..rel.len], rel);
var reply: [protocol.message_maximum]u8 = undefined;
const n = ipc.call(backend, message[0 .. protocol.request_size + rel.len], &reply) catch return fail(out);
if (n < protocol.reply_size) return fail(out);
const backend_reply = std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]);
if (backend_reply.status != 0) return writeReply(out, .{ .status = backend_reply.status }, &.{});
for (&opens, 0..) |*o, i| {
if (!o.used) {
o.* = .{ .used = true, .node = @intCast(backend_reply.node), .backend = backend, .owner = sender };
return writeReply(out, .{ .status = 0, .node = i }, &.{});
}
}
return fail(out);
}
/// Relay a read/write/status/readdir/close on a forwarding handle to the backend
/// (the node already rewritten to the backend's id) and copy its reply out.
fn forwardRequest(out: []u8, backend: ipc.Handle, request: protocol.Request, payload: []const u8) usize {
var message: [protocol.message_maximum]u8 = undefined;
@memcpy(message[0..protocol.request_size], std.mem.asBytes(&request));
const plen = @min(payload.len, protocol.maximum_payload);
@memcpy(message[protocol.request_size..][0..plen], payload[0..plen]);
var reply: [protocol.message_maximum]u8 = undefined;
const n = ipc.call(backend, message[0 .. protocol.request_size + plen], &reply) catch return fail(out);
const copy = @min(n, out.len);
@memcpy(out[0..copy], reply[0..copy]);
return copy;
}
/// Forward a path-based operation (mkdir, unlink) under a mount to its backend and
/// relay the reply. No handle is created — these operate by path and return only a
/// status.
fn forwardPath(out: []u8, backend: ipc.Handle, operation: protocol.Operation, relative: []const u8) usize {
const request = protocol.Request{ .operation = operation, .node = 0, .offset = 0, .len = @intCast(relative.len), .flags = 0 };
var message: [protocol.message_maximum]u8 = undefined;
@memcpy(message[0..protocol.request_size], std.mem.asBytes(&request));
const rel = relative[0..@min(relative.len, protocol.maximum_payload)];
@memcpy(message[protocol.request_size..][0..rel.len], rel);
var reply: [protocol.message_maximum]u8 = undefined;
const n = ipc.call(backend, message[0 .. protocol.request_size + rel.len], &reply) catch return fail(out);
const copy = @min(n, out.len);
@memcpy(out[0..copy], reply[0..copy]);
return copy;
}
/// Forward a rename to its backend: the payload is the mount-relative old path, a
/// 0x00 separator, then the mount-relative new path. Relays the backend's reply.
fn forwardRename(out: []u8, backend: ipc.Handle, old_relative: []const u8, new_relative: []const u8) usize {
const total = old_relative.len + 1 + new_relative.len;
var message: [protocol.message_maximum]u8 = undefined;
if (protocol.request_size + total > message.len) return fail(out);
const request = protocol.Request{ .operation = .rename, .node = 0, .offset = 0, .len = @intCast(total), .flags = 0 };
@memcpy(message[0..protocol.request_size], std.mem.asBytes(&request));
var p = protocol.request_size;
@memcpy(message[p..][0..old_relative.len], old_relative);
p += old_relative.len;
message[p] = 0;
p += 1;
@memcpy(message[p..][0..new_relative.len], new_relative);
p += new_relative.len;
var reply: [protocol.message_maximum]u8 = undefined;
const n = ipc.call(backend, message[0..p], &reply) catch return fail(out);
const copy = @min(n, out.len);
@memcpy(out[0..copy], reply[0..copy]);
return copy;
}
/// Best-effort close of a backend node (used when a dead client's forwarding
/// handles are swept — the backend must not leak the vfs's opens).
fn forwardClose(backend: ipc.Handle, backend_node: u64) void {
const request = protocol.Request{ .operation = .close, .node = backend_node, .offset = 0, .len = 0, .flags = 0 };
var reply: [protocol.message_maximum]u8 = undefined;
_ = ipc.call(backend, std.mem.asBytes(&request), &reply) catch {};
}
fn doMount(out: []u8, prefix: []const u8, backend: ipc.Handle) usize {
for (&mounts) |*m| {
if (m.used and std.mem.eql(u8, m.prefix[0..m.prefix_len], prefix)) {
m.backend = backend;
std.log.info("remounted {s}", .{prefix});
return writeReply(out, .{ .status = 0 }, &.{});
}
}
for (&mounts) |*m| {
if (!m.used) {
const l = @min(prefix.len, m.prefix.len);
m.used = true;
@memcpy(m.prefix[0..l], prefix[0..l]);
m.prefix_len = l;
m.backend = backend;
std.log.info("mounted {s}", .{prefix[0..l]});
return writeReply(out, .{ .status = 0 }, &.{});
}
}
return fail(out);
}
fn doUnmount(out: []u8, prefix: []const u8) usize {
for (&mounts) |*m| {
if (m.used and std.mem.eql(u8, m.prefix[0..m.prefix_len], prefix)) {
m.used = false;
std.log.info("unmounted {s}", .{prefix});
return writeReply(out, .{ .status = 0 }, &.{});
}
}
return fail(out);
}
/// Release every open handle `client` held — called on that client's published
/// exit event. Forwarding handles also tell their backend to release; local
/// nodes (the ramfs files) stay, since ramfs contents outlive their writers.
fn releaseClientHandles(client: u32) void {
var released: u32 = 0;
for (&opens) |*o| {
if (o.used and o.owner == client) {
if (o.backend) |backend| forwardClose(backend, o.node);
o.used = false;
released += 1;
}
}
if (released != 0) std.log.info("released {d} handle(s) for dead client {d}", .{ released, client });
}
/// Handle one request from `sender`; write the reply into `out`, return its length.
fn handle(message: []const u8, out: []u8, sender: u32, capability: ?ipc.Handle) usize {
if (message.len < protocol.request_size) return fail(out);
const request = std.mem.bytesToValue(protocol.Request, message[0..protocol.request_size]);
const payload = message[protocol.request_size..];
switch (request.operation) {
.mount => {
const prefix = payload[0..@min(payload.len, request.len)];
const backend = capability orelse return fail(out);
return doMount(out, prefix, backend);
},
.unmount => {
const prefix = payload[0..@min(payload.len, request.len)];
return doUnmount(out, prefix);
},
.open => {
const name = payload[0..@min(payload.len, request.len)];
if (longestMount(name)) |m| return forwardOpen(out, m.backend, m.relative, request.flags, sender);
// An absolute path with no matching mount is simply not found — only
// bare names live in the flat ramfs. (Else /mnt/usb would be silently
// created as a flat file when its filesystem is not yet mounted.)
if (path.isAbsolute(name)) return fail(out);
const ni = findNode(name) orelse createNode(name) orelse return fail(out);
for (&opens, 0..) |*o, i| {
if (!o.used) {
o.* = .{ .used = true, .node = ni, .backend = null, .owner = sender };
return writeReply(out, .{ .status = 0, .node = i }, &.{});
}
}
return fail(out);
},
.read => {
const of = openAt(request.node) orelse return fail(out);
if (of.backend) |backend| {
var forwarded = request;
forwarded.node = of.node;
return forwardRequest(out, backend, forwarded, payload);
}
const nd = &nodes[@intCast(of.node)];
const off: usize = @intCast(request.offset);
if (off >= nd.size) return writeReply(out, .{ .status = 0, .len = 0 }, &.{}); // EOF
const n = @min(@min(nd.size - off, request.len), protocol.maximum_payload);
return writeReply(out, .{ .status = 0, .len = @intCast(n) }, nd.data[off .. off + n]);
},
.write => {
const of = openAt(request.node) orelse return fail(out);
if (of.backend) |backend| {
var forwarded = request;
forwarded.node = of.node;
return forwardRequest(out, backend, forwarded, payload);
}
const nd = &nodes[@intCast(of.node)];
const off: usize = @intCast(request.offset);
if (off > nd.data.len) return fail(out);
const n = @min(@min(payload.len, request.len), nd.data.len - off);
@memcpy(nd.data[off .. off + n], payload[0..n]);
if (off + n > nd.size) nd.size = off + n;
return writeReply(out, .{ .status = 0, .len = @intCast(n) }, &.{});
},
.status => {
const of = openAt(request.node) orelse return fail(out);
if (of.backend) |backend| {
var forwarded = request;
forwarded.node = of.node;
return forwardRequest(out, backend, forwarded, payload);
}
const st = protocol.FileStatus{ .size = nodes[@intCast(of.node)].size, .kind = @intFromEnum(protocol.NodeKind.regular) };
return writeReply(out, .{ .status = 0, .len = @sizeOf(protocol.FileStatus) }, std.mem.asBytes(&st));
},
.readdir => {
const of = openAt(request.node) orelse return fail(out);
if (of.backend) |backend| {
var forwarded = request;
forwarded.node = of.node;
return forwardRequest(out, backend, forwarded, payload);
}
// The flat ramfs has no directories: report EOF.
return writeReply(out, .{ .status = 0, .len = 0 }, &.{});
},
.close => {
const of = openAt(request.node);
if (of) |o| {
if (o.backend) |backend| forwardClose(backend, o.node);
o.used = false;
}
return writeReply(out, .{ .status = 0 }, &.{});
},
.mkdir, .unlink => {
const name = payload[0..@min(payload.len, request.len)];
if (longestMount(name)) |m| return forwardPath(out, m.backend, request.operation, m.relative);
// Only a mounted backend has real directories; the flat ramfs cannot
// create or remove them (and a bare-name path is not a mount target).
return fail(out);
},
.rename => {
const both = payload[0..@min(payload.len, request.len)];
const sep = std.mem.indexOfScalar(u8, both, 0) orelse return fail(out);
const old_path = both[0..sep];
const new_path = both[sep + 1 ..];
const mo = longestMount(old_path) orelse return fail(out);
const mn = longestMount(new_path) orelse return fail(out);
// Both paths must live under the same mount — cross-filesystem rename is
// not supported.
if (mo.backend != mn.backend) return fail(out);
return forwardRename(out, mo.backend, mo.relative, mn.relative);
},
}
}
/// Startup, under the harness: subscribe to the published exit events — when a
/// client dies holding open handles, the exit notification is how the VFS learns
/// to release them (docs/process-lifecycle.md).
fn initialise(endpoint: ipc.Handle) bool {
if (!runtime.process.subscribeExits(endpoint)) {
_ = runtime.system.write("/system/services/vfs: exit subscription failed\n");
}
_ = runtime.system.write("/system/services/vfs: ready\n");
return true;
}
/// A non-signal notification: the only kind the VFS subscribes to is exit events.
fn onNotification(badge: u64) void {
if (badge & ipc.notify_exit_bit != 0) {
releaseClientHandles(@intCast(badge & ~(ipc.notify_badge_bit | ipc.notify_exit_bit)));
}
}
pub fn main() void {
// The harness owns the loop: requests dispatch to handle(), exit events to
// onNotification(), ping and terminate are answered for free — this service
// gained the whole lifecycle contract by deleting its hand-rolled loop.
runtime.service.run(protocol.message_maximum, .{
.service = .vfs,
.init = initialise,
.on_message = handle,
.on_notification = onNotification,
});
}