file-system: extract the serving harness from fat — V1

fat was one binary doing four jobs; the three that are not FAT-specific move
to library/kernel/file-system-harness, a Server(comptime Engine) generic over
the engine type: the badge-scoped open-node table, the nine vfs handlers, the
not-mounted politeness, the exit sweep, mount registration, and durable-on-
close. A filesystem is now an engine plus a main that hands the harness a
mounted volume; a second engine reuses the harness wholesale.

Placement note: the plan said library/file-system, but the harness is a
specialization of `service` (its sibling) and needs nothing from the device
domain, so it lives beside service in library/kernel and stays block-free —
durability rides a caller closure (Volume.flush), no backwards kernel->device
dependency, no new-domain scaffolding. The engine type is inferred from
resolve()'s return, so engine.zig is untouched (its Node stays module-scope).

fat keeps only its FAT-specific bring-up (acquireVolume, DMA, engine.mount,
the attach/detach round trip) and the three mount prefixes as data. Behavior-
neutral: 13/13 across the fat/vfs/logger/IOMMU surface, nothing observable
changed. This lands first so every later phase touches the harness once.
This commit is contained in:
Daniel Samson
2026-08-09 16:55:48 +01:00
parent b59f981c58
commit d63a008148
4 changed files with 409 additions and 318 deletions
+336
View File
@@ -0,0 +1,336 @@
//! The filesystem serving harness: the block-client-and-engine-agnostic half of
//! a filesystem service (docs/file-system-development/storage-architecture.md).
//! Everything a filesystem process does that is NOT its on-disk format lives
//! here — establishment, the badge-scoped open-node table, the nine vfs-protocol
//! handlers, mount registration, the not-mounted-yet politeness, the exit sweep,
//! and the durable-on-close flush. A filesystem is then an ENGINE (the pure,
//! host-testable format code behind a small method set) plus a `main` that wires
//! it in, so a second filesystem reuses this wholesale — the reason it is a
//! shared library rather than per-filesystem code.
//!
//! Placement: `library/kernel`, beside its sibling `service` (the generic
//! serving harness this specializes for the vfs protocol). It is block-free —
//! the caller's `Volume.flush` closure owns durability — so it needs nothing
//! from the device domain and introduces no backwards dependency.
//!
//! `Server(Engine)` is generic over the engine TYPE, checked at compile time by
//! the calls below. An engine must expose:
//! - `pub const Node` with fields `is_directory: bool`, `size`, `mtime`;
//! - `pub const Listing` with `name_buffer`, `name_len`, `is_directory`, `size`;
//! - `current_time_epoch` a settable field (the harness stamps it per turn);
//! - resolve, createFile, createDirectory, removeFile, rename, truncate,
//! readFile, writeFile, listEntry — the signatures fat's engine.zig already has.
const std = @import("std");
const ipc = @import("ipc");
const process = @import("process");
const service = @import("service");
const time = @import("time");
const file_system = @import("file-system");
const envelope = @import("envelope");
const vfs_protocol = @import("vfs-protocol");
const logging = @import("logging");
/// One prefix this filesystem mounts into the kernel mount table. `rewrite` is
/// the backend-relative prefix a path is rewritten to before it reaches the
/// engine (empty = mount the volume root at `prefix`, the common case).
pub const MountSpec = struct { prefix: []const u8, rewrite: []const u8 = "" };
/// The engine's node type, inferred from `resolve`'s return (`?Node`) so the
/// engine need not re-export it as a member — engine.zig keeps `Node` at module
/// scope, and this harness stays purely additive on the engine side.
fn NodeType(comptime Engine: type) type {
return @typeInfo(@typeInfo(@TypeOf(Engine.resolve)).@"fn".return_type.?).optional.child;
}
pub fn Server(comptime Engine: type) type {
return struct {
const Node = NodeType(Engine);
/// What a bring-up produces: the mounted engine (a stable pointer the
/// caller owns), the prefixes to install, and a durability closure the
/// harness calls on every close (the caller checks its own dirty state).
pub const Volume = struct {
engine: *Engine,
mounts: []const MountSpec,
flush: *const fn () void,
};
pub const Callbacks = struct {
/// Acquire and mount the volume, or null to retry on the timer. The
/// caller does the filesystem-specific bring-up (find the block
/// device, set up DMA, mount the engine) and returns a `Volume`.
bringUp: *const fn (endpoint: ipc.Handle) ?Volume,
/// The vfs contract name to bind. A filesystem serving one volume
/// binds "vfs" today; the volume-manager era hands each per-volume
/// process its own establishment and this fades.
service_name: ?[]const u8 = "vfs",
};
// --- the harness's own state, one set per instantiation ---------------
// A filesystem binary instantiates Server once, so these globals are the
// one server's state, exactly where fat's file-scoped globals were.
const Serve = vfs_protocol.Protocol.Provider(void);
const Invocation = envelope.Invocation;
const Answer = envelope.Answer;
/// What a handler returns when the thing asked for is not there — a bad
/// node id, someone else's node, an unresolved path, a refused mutation.
/// One errno for all: a filesystem's failures are all "no such thing" to
/// the file API, and *someone else's* must be indistinguishable from
/// *nobody's*, or the refusal would leak which ids are live.
const refused: isize = -envelope.ENOENT;
/// How often to retry bring-up while unmounted. Storage arriving is
/// event-shaped (the usb chain registering, maybe after a restart), but
/// there is no subscription; a slow poll keeps the service responsive
/// (ping, terminate) while it waits and alive to catch late storage.
const mount_retry_ms = 500;
const OpenNode = struct { used: bool = false, node: Node = undefined, owner: u32 = 0 };
var open_nodes = [_]OpenNode{.{}} ** 32;
var callbacks: Callbacks = undefined;
var service_endpoint: ipc.Handle = 0;
var engine_ptr: ?*Engine = null;
var volume_flush: *const fn () void = undefined;
var mounted: bool = false;
fn allocOpen() ?usize {
for (&open_nodes, 0..) |*o, i| {
if (!o.used) return i;
}
return null;
}
/// The open node `id` names **for `owner`** — null unless in range, in
/// use, and this client's own. Ids are small integers from a table of 32,
/// trivially guessable, so this badge check is the scope
/// (docs/os-development/protocol-namespace.md). The owner is a TASK, not a
/// process, because the badge is: a threaded client reads a node from the
/// thread that opened it, and the exit sweep releases a worker's handles.
fn openFor(id: u64, owner: u32) ?*OpenNode {
if (id >= open_nodes.len) return null;
const o = &open_nodes[@intCast(id)];
if (!o.used or o.owner != owner) return null;
return o;
}
const ParentLeaf = struct { parent: []const u8, leaf: []const u8 };
// Split a path: "/a/b" -> ("/a", "b"); "/b" -> ("/", "b"); "b" -> ("/", "b").
fn splitParent(path: []const u8) ParentLeaf {
const slash = std.mem.lastIndexOfScalar(u8, path, '/');
return .{
.parent = if (slash) |s| (if (s == 0) "/" else path[0..s]) else "/",
.leaf = if (slash) |s| path[s + 1 ..] else path,
};
}
fn onOpen(_: void, invocation: Invocation(vfs_protocol.Open), answer: Answer(vfs_protocol.Opened)) isize {
const fs = engine_ptr orelse return refused;
const path = invocation.tail;
const flags = invocation.request.flags;
var node = fs.resolve(path);
if (node == null and flags & vfs_protocol.create != 0) {
const split = splitParent(path);
const parent = fs.resolve(split.parent) orelse return refused;
node = fs.createFile(parent, split.leaf);
}
var resolved = node orelse return refused;
// O_TRUNC: replace contents rather than overwrite in place (frees the
// old chain, so a shorter rewrite leaves no stale tail).
if (flags & vfs_protocol.truncate != 0 and !resolved.is_directory) {
fs.truncate(&resolved);
}
const index = allocOpen() orelse return refused;
open_nodes[index] = .{ .used = true, .node = resolved, .owner = invocation.sender };
answer.set(.{ .node = index });
return 0;
}
fn onRead(_: void, invocation: Invocation(vfs_protocol.Read), answer: Answer(void)) isize {
const fs = engine_ptr orelse return refused;
const o = openFor(invocation.target, invocation.sender) orelse return refused;
const into = answer.tail();
const want = @min(@as(usize, invocation.request.len), into.len);
return @intCast(fs.readFile(o.node, @intCast(invocation.request.offset), into[0..want]));
}
fn onWrite(_: void, invocation: Invocation(vfs_protocol.Write), answer: Answer(vfs_protocol.Written)) isize {
const fs = engine_ptr orelse return refused;
const o = openFor(invocation.target, invocation.sender) orelse return refused;
const data = invocation.tail[0..@min(invocation.tail.len, invocation.request.len)];
const n = fs.writeFile(&o.node, @intCast(invocation.request.offset), data);
answer.set(.{ .count = @intCast(n) });
return 0;
}
fn onStatus(_: void, invocation: Invocation(void), answer: Answer(vfs_protocol.FileStatus)) isize {
const o = openFor(invocation.target, invocation.sender) orelse return refused;
const kind: vfs_protocol.NodeKind = if (o.node.is_directory) .directory else .regular;
answer.set(.{ .size = o.node.size, .kind = @intFromEnum(kind), .mtime = o.node.mtime });
return 0;
}
/// One entry per call. End of directory — not a directory, or a cursor
/// past the last child — is an entry with no name.
fn onReaddir(_: void, invocation: Invocation(vfs_protocol.Readdir), answer: Answer(vfs_protocol.DirectoryEntry)) isize {
const fs = engine_ptr orelse return refused;
const o = openFor(invocation.target, invocation.sender) orelse return refused;
if (!o.node.is_directory) {
answer.set(.{});
return 0;
}
const listing = fs.listEntry(o.node, @intCast(invocation.request.cursor)) orelse {
answer.set(.{});
return 0;
};
const kind: vfs_protocol.NodeKind = if (listing.is_directory) .directory else .regular;
const into = answer.tail();
const name_len = @min(listing.name_len, into.len);
@memcpy(into[0..name_len], listing.name_buffer[0..name_len]);
answer.set(.{ .kind = @intFromEnum(kind), .name_len = @intCast(name_len), .size = listing.size });
return @intCast(name_len);
}
/// Closing is scoped like any other node operation: a client releases its
/// own handles and nobody else's, and a foreign/free/out-of-range id is
/// refused identically so a close cannot probe which ids are live.
fn onClose(_: void, invocation: Invocation(void), _: Answer(void)) isize {
const o = openFor(invocation.target, invocation.sender) orelse return refused;
o.used = false;
// Durable-on-close: the caller's flush commits any device write cache
// to stable media now. This is what makes init's shutdown log flush
// survive a real power-off, and the right default for removable media.
volume_flush();
return 0;
}
fn onMakeDirectory(_: void, invocation: Invocation(void), _: Answer(void)) isize {
const fs = engine_ptr orelse return refused;
const path = invocation.tail;
if (fs.resolve(path) != null) return refused; // already exists
const split = splitParent(path);
const parent = fs.resolve(split.parent) orelse return refused;
if (fs.createDirectory(parent, split.leaf) == null) return refused;
return 0;
}
fn onUnlink(_: void, invocation: Invocation(void), _: Answer(void)) isize {
const fs = engine_ptr orelse return refused;
const split = splitParent(invocation.tail);
const parent = fs.resolve(split.parent) orelse return refused;
if (!fs.removeFile(parent, split.leaf)) return refused;
return 0;
}
fn onRename(_: void, invocation: Invocation(void), _: Answer(void)) isize {
const fs = engine_ptr orelse return refused;
const both = invocation.tail;
const separator = std.mem.indexOfScalar(u8, both, 0) orelse return refused;
const old_split = splitParent(both[0..separator]);
const new_split = splitParent(both[separator + 1 ..]);
if (!std.mem.eql(u8, old_split.parent, new_split.parent)) return refused; // same-directory only
const parent = fs.resolve(old_split.parent) orelse return refused;
if (!fs.rename(parent, old_split.leaf, new_split.leaf)) return refused;
return 0;
}
/// The verbs this backend implements. `mount`/`unmount`/`bind` are absent
/// on purpose — path routing is the kernel's, and only init serves `bind`.
const handlers = Serve.Handlers{
.open = onOpen,
.close = onClose,
.read = onRead,
.write = onWrite,
.status = onStatus,
.readdir = onReaddir,
.mkdir = onMakeDirectory,
.unlink = onUnlink,
.rename = onRename,
};
/// The vfs protocol carries no capability, so `arrived` is never claimed
/// — the harness's ownership rule then closes whatever a caller attached,
/// so a request carrying one cannot spend a slot of this server's table.
fn onMessage(message: []const u8, out: []u8, sender: u32, arrived: *ipc.Arrival) usize {
_ = arrived;
const fs = engine_ptr;
// Storage not up yet: fail politely, whatever was asked — clients retry.
if (!mounted or fs == null) {
const status = envelope.Status{ .status = refused, .len = 0 };
@memcpy(out[0..envelope.prefix_size], std.mem.asBytes(&status));
return envelope.prefix_size;
}
// Stamp create/write with the current wall-clock time (mtime): cheap,
// and it keeps the engine pure (it takes the time as data, not a call).
fs.?.current_time_epoch = time.wallClock();
return Serve.dispatch({}, handlers, message, sender, null, out);
}
/// A process-exit event releases every open handle the dead client held,
/// so a crashed reader cannot pin table slots.
fn onNotification(badge: u64) void {
const got = ipc.Received{ .len = 0, .badge = badge, .cap = null };
if (got.isTimer()) {
tryBringUp();
if (!mounted) _ = time.timerOnce(service_endpoint, mount_retry_ms);
return;
}
if (!got.isChildExit()) return;
const dead = got.childProcessId();
var released: u32 = 0;
for (&open_nodes) |*o| {
if (o.used and o.owner == dead) {
o.* = .{};
released += 1;
}
}
if (released != 0) std.log.info("released {d} handle(s) for dead client {d}", .{ released, dead });
}
/// One bring-up attempt: ask the caller for a mounted volume, and on
/// success install its mounts and go live. A failure leaves everything
/// untouched for the next tick.
fn tryBringUp() void {
if (mounted) return;
const volume = callbacks.bringUp(service_endpoint) orelse return;
engine_ptr = volume.engine;
volume_flush = volume.flush;
for (volume.mounts) |m| {
const ok = if (m.rewrite.len == 0)
file_system.mount(m.prefix, service_endpoint)
else
file_system.mountRewritten(m.prefix, service_endpoint, m.rewrite);
if (ok) {
std.log.info("mounted {s}", .{m.prefix});
} else {
std.log.info("could not mount {s}", .{m.prefix});
}
}
mounted = true;
}
fn initialise(endpoint: ipc.Handle) bool {
service_endpoint = endpoint;
// Sweep a dead client's open handles via the published exit events —
// clients hold OUR node ids directly, so a crash must not pin slots.
_ = process.subscribeExits(endpoint);
tryBringUp();
if (!mounted) _ = time.timerOnce(endpoint, mount_retry_ms);
return true; // serve regardless: requests fail politely until storage mounts
}
pub fn run(cb: Callbacks) void {
callbacks = cb;
service.run(vfs_protocol.message_maximum, .{
.service = cb.service_name,
.init = initialise,
.on_message = onMessage,
.on_notification = onNotification,
});
}
};
}