//! The filesystem serving harness: the block-client-and-engine-agnostic half of //! a filesystem service (docs/file-system-development/storage-architecture.md). //! Everything a filesystem process does that is NOT its on-disk format lives //! here — establishment, the badge-scoped open-node table, the nine vfs-protocol //! handlers, mount registration, the not-mounted-yet politeness, the exit sweep, //! and the durable-on-close flush. A filesystem is then an ENGINE (the pure, //! host-testable format code behind a small method set) plus a `main` that wires //! it in, so a second filesystem reuses this wholesale — the reason it is a //! shared library rather than per-filesystem code. //! //! Placement: `library/kernel`, beside its sibling `service` (the generic //! serving harness this specializes for the vfs protocol). It is block-free — //! the caller's `Volume.flush` closure owns durability — so it needs nothing //! from the device domain and introduces no backwards dependency. //! //! `Server(Engine)` is generic over the engine TYPE, checked at compile time by //! the calls below. An engine must expose: //! - `pub const Node` with fields `is_directory: bool`, `size`, `mtime`; //! - `pub const Listing` with `name_buffer`, `name_len`, `is_directory`, `size`; //! - `current_time_epoch` a settable field (the harness stamps it per turn); //! - resolve, createFile, createDirectory, removeFile, rename, truncate, //! readFile, writeFile, listEntry — the signatures fat's engine.zig already has. const std = @import("std"); const ipc = @import("ipc"); const process = @import("process"); const service = @import("service"); const time = @import("time"); const file_system = @import("file-system"); const envelope = @import("envelope"); const vfs_protocol = @import("vfs-protocol"); const logging = @import("logging"); /// One prefix this filesystem mounts into the kernel mount table. `rewrite` is /// the backend-relative prefix a path is rewritten to before it reaches the /// engine (empty = mount the volume root at `prefix`, the common case). pub const MountSpec = struct { prefix: []const u8, rewrite: []const u8 = "" }; /// The engine's node type, inferred from `resolve`'s return (`?Node`) so the /// engine need not re-export it as a member — engine.zig keeps `Node` at module /// scope, and this harness stays purely additive on the engine side. fn NodeType(comptime Engine: type) type { return @typeInfo(@typeInfo(@TypeOf(Engine.resolve)).@"fn".return_type.?).optional.child; } pub fn Server(comptime Engine: type) type { return struct { const Node = NodeType(Engine); /// What a bring-up produces: the mounted engine (a stable pointer the /// caller owns), the prefixes to install, and a durability closure the /// harness calls on every close (the caller checks its own dirty state). pub const Volume = struct { engine: *Engine, mounts: []const MountSpec, flush: *const fn () void, }; pub const Callbacks = struct { /// Acquire and mount the volume, or null to retry on the timer. The /// caller does the filesystem-specific bring-up (find the block /// device, set up DMA, mount the engine) and returns a `Volume`. bringUp: *const fn (endpoint: ipc.Handle) ?Volume, /// The vfs contract name to bind. A filesystem serving one volume /// binds "vfs" today; the volume-manager era hands each per-volume /// process its own establishment and this fades. service_name: ?[]const u8 = "vfs", }; // --- the harness's own state, one set per instantiation --------------- // A filesystem binary instantiates Server once, so these globals are the // one server's state, exactly where fat's file-scoped globals were. const Serve = vfs_protocol.Protocol.Provider(void); const Invocation = envelope.Invocation; const Answer = envelope.Answer; /// What a handler returns when the thing asked for is not there — a bad /// node id, someone else's node, an unresolved path, a refused mutation. /// One errno for all: a filesystem's failures are all "no such thing" to /// the file API, and *someone else's* must be indistinguishable from /// *nobody's*, or the refusal would leak which ids are live. const refused: isize = -envelope.ENOENT; /// How often to retry bring-up while unmounted. Storage arriving is /// event-shaped (the usb chain registering, maybe after a restart), but /// there is no subscription; a slow poll keeps the service responsive /// (ping, terminate) while it waits and alive to catch late storage. const mount_retry_ms = 500; const OpenNode = struct { used: bool = false, node: Node = undefined, owner: u32 = 0 }; var open_nodes = [_]OpenNode{.{}} ** 32; var callbacks: Callbacks = undefined; var service_endpoint: ipc.Handle = 0; var engine_ptr: ?*Engine = null; var volume_flush: *const fn () void = undefined; var mounted: bool = false; fn allocOpen() ?usize { for (&open_nodes, 0..) |*o, i| { if (!o.used) return i; } return null; } /// The open node `id` names **for `owner`** — null unless in range, in /// use, and this client's own. Ids are small integers from a table of 32, /// trivially guessable, so this badge check is the scope /// (docs/os-development/protocol-namespace.md). The owner is a TASK, not a /// process, because the badge is: a threaded client reads a node from the /// thread that opened it, and the exit sweep releases a worker's handles. fn openFor(id: u64, owner: u32) ?*OpenNode { if (id >= open_nodes.len) return null; const o = &open_nodes[@intCast(id)]; if (!o.used or o.owner != owner) return null; return o; } const ParentLeaf = struct { parent: []const u8, leaf: []const u8 }; // Split a path: "/a/b" -> ("/a", "b"); "/b" -> ("/", "b"); "b" -> ("/", "b"). fn splitParent(path: []const u8) ParentLeaf { const slash = std.mem.lastIndexOfScalar(u8, path, '/'); return .{ .parent = if (slash) |s| (if (s == 0) "/" else path[0..s]) else "/", .leaf = if (slash) |s| path[s + 1 ..] else path, }; } fn onOpen(_: void, invocation: Invocation(vfs_protocol.Open), answer: Answer(vfs_protocol.Opened)) isize { const fs = engine_ptr orelse return refused; const path = invocation.tail; const flags = invocation.request.flags; var node = fs.resolve(path); if (node == null and flags & vfs_protocol.create != 0) { const split = splitParent(path); const parent = fs.resolve(split.parent) orelse return refused; node = fs.createFile(parent, split.leaf); } var resolved = node orelse return refused; // O_TRUNC: replace contents rather than overwrite in place (frees the // old chain, so a shorter rewrite leaves no stale tail). if (flags & vfs_protocol.truncate != 0 and !resolved.is_directory) { fs.truncate(&resolved); } const index = allocOpen() orelse return refused; open_nodes[index] = .{ .used = true, .node = resolved, .owner = invocation.sender }; answer.set(.{ .node = index }); return 0; } fn onRead(_: void, invocation: Invocation(vfs_protocol.Read), answer: Answer(void)) isize { const fs = engine_ptr orelse return refused; const o = openFor(invocation.target, invocation.sender) orelse return refused; const into = answer.tail(); const want = @min(@as(usize, invocation.request.len), into.len); return @intCast(fs.readFile(o.node, @intCast(invocation.request.offset), into[0..want])); } fn onWrite(_: void, invocation: Invocation(vfs_protocol.Write), answer: Answer(vfs_protocol.Written)) isize { const fs = engine_ptr orelse return refused; const o = openFor(invocation.target, invocation.sender) orelse return refused; const data = invocation.tail[0..@min(invocation.tail.len, invocation.request.len)]; const n = fs.writeFile(&o.node, @intCast(invocation.request.offset), data); answer.set(.{ .count = @intCast(n) }); return 0; } fn onStatus(_: void, invocation: Invocation(void), answer: Answer(vfs_protocol.FileStatus)) isize { const o = openFor(invocation.target, invocation.sender) orelse return refused; const kind: vfs_protocol.NodeKind = if (o.node.is_directory) .directory else .regular; answer.set(.{ .size = o.node.size, .kind = @intFromEnum(kind), .mtime = o.node.mtime }); return 0; } /// One entry per call. End of directory — not a directory, or a cursor /// past the last child — is an entry with no name. fn onReaddir(_: void, invocation: Invocation(vfs_protocol.Readdir), answer: Answer(vfs_protocol.DirectoryEntry)) isize { const fs = engine_ptr orelse return refused; const o = openFor(invocation.target, invocation.sender) orelse return refused; if (!o.node.is_directory) { answer.set(.{}); return 0; } const listing = fs.listEntry(o.node, @intCast(invocation.request.cursor)) orelse { answer.set(.{}); return 0; }; const kind: vfs_protocol.NodeKind = if (listing.is_directory) .directory else .regular; const into = answer.tail(); const name_len = @min(listing.name_len, into.len); @memcpy(into[0..name_len], listing.name_buffer[0..name_len]); answer.set(.{ .kind = @intFromEnum(kind), .name_len = @intCast(name_len), .size = listing.size }); return @intCast(name_len); } /// Closing is scoped like any other node operation: a client releases its /// own handles and nobody else's, and a foreign/free/out-of-range id is /// refused identically so a close cannot probe which ids are live. fn onClose(_: void, invocation: Invocation(void), _: Answer(void)) isize { const o = openFor(invocation.target, invocation.sender) orelse return refused; o.used = false; // Durable-on-close: the caller's flush commits any device write cache // to stable media now. This is what makes init's shutdown log flush // survive a real power-off, and the right default for removable media. volume_flush(); return 0; } fn onMakeDirectory(_: void, invocation: Invocation(void), _: Answer(void)) isize { const fs = engine_ptr orelse return refused; const path = invocation.tail; if (fs.resolve(path) != null) return refused; // already exists const split = splitParent(path); const parent = fs.resolve(split.parent) orelse return refused; if (fs.createDirectory(parent, split.leaf) == null) return refused; return 0; } fn onUnlink(_: void, invocation: Invocation(void), _: Answer(void)) isize { const fs = engine_ptr orelse return refused; const split = splitParent(invocation.tail); const parent = fs.resolve(split.parent) orelse return refused; if (!fs.removeFile(parent, split.leaf)) return refused; return 0; } fn onRename(_: void, invocation: Invocation(void), _: Answer(void)) isize { const fs = engine_ptr orelse return refused; const both = invocation.tail; const separator = std.mem.indexOfScalar(u8, both, 0) orelse return refused; const old_split = splitParent(both[0..separator]); const new_split = splitParent(both[separator + 1 ..]); if (!std.mem.eql(u8, old_split.parent, new_split.parent)) return refused; // same-directory only const parent = fs.resolve(old_split.parent) orelse return refused; if (!fs.rename(parent, old_split.leaf, new_split.leaf)) return refused; return 0; } /// The verbs this backend implements. `mount`/`unmount`/`bind` are absent /// on purpose — path routing is the kernel's, and only init serves `bind`. const handlers = Serve.Handlers{ .open = onOpen, .close = onClose, .read = onRead, .write = onWrite, .status = onStatus, .readdir = onReaddir, .mkdir = onMakeDirectory, .unlink = onUnlink, .rename = onRename, }; /// The vfs protocol carries no capability, so `arrived` is never claimed /// — the harness's ownership rule then closes whatever a caller attached, /// so a request carrying one cannot spend a slot of this server's table. fn onMessage(message: []const u8, out: []u8, sender: u32, arrived: *ipc.Arrival) usize { _ = arrived; const fs = engine_ptr; // Storage not up yet: fail politely, whatever was asked — clients retry. if (!mounted or fs == null) { const status = envelope.Status{ .status = refused, .len = 0 }; @memcpy(out[0..envelope.prefix_size], std.mem.asBytes(&status)); return envelope.prefix_size; } // Stamp create/write with the current wall-clock time (mtime): cheap, // and it keeps the engine pure (it takes the time as data, not a call). fs.?.current_time_epoch = time.wallClock(); return Serve.dispatch({}, handlers, message, sender, null, out); } /// A process-exit event releases every open handle the dead client held, /// so a crashed reader cannot pin table slots. fn onNotification(badge: u64) void { const got = ipc.Received{ .len = 0, .badge = badge, .cap = null }; if (got.isTimer()) { tryBringUp(); if (!mounted) _ = time.timerOnce(service_endpoint, mount_retry_ms); return; } if (!got.isChildExit()) return; const dead = got.childProcessId(); var released: u32 = 0; for (&open_nodes) |*o| { if (o.used and o.owner == dead) { o.* = .{}; released += 1; } } if (released != 0) std.log.info("released {d} handle(s) for dead client {d}", .{ released, dead }); } /// One bring-up attempt: ask the caller for a mounted volume, and on /// success install its mounts and go live. A failure leaves everything /// untouched for the next tick. fn tryBringUp() void { if (mounted) return; const volume = callbacks.bringUp(service_endpoint) orelse return; engine_ptr = volume.engine; volume_flush = volume.flush; for (volume.mounts) |m| { const ok = if (m.rewrite.len == 0) file_system.mount(m.prefix, service_endpoint) else file_system.mountRewritten(m.prefix, service_endpoint, m.rewrite); if (ok) { std.log.info("mounted {s}", .{m.prefix}); } else { std.log.info("could not mount {s}", .{m.prefix}); } } mounted = true; } fn initialise(endpoint: ipc.Handle) bool { service_endpoint = endpoint; // Sweep a dead client's open handles via the published exit events — // clients hold OUR node ids directly, so a crash must not pin slots. _ = process.subscribeExits(endpoint); tryBringUp(); if (!mounted) _ = time.timerOnce(endpoint, mount_retry_ms); return true; // serve regardless: requests fail politely until storage mounts } pub fn run(cb: Callbacks) void { callbacks = cb; service.run(vfs_protocol.message_maximum, .{ .service = cb.service_name, .init = initialise, .on_message = onMessage, .on_notification = onNotification, }); } }; }