From d63a00814867794c3b7b10a1dd7df0e309bb88b8 Mon Sep 17 00:00:00 2001 From: Daniel Samson <12231216+daniel-samson@users.noreply.github.com> Date: Sun, 9 Aug 2026 16:55:48 +0100 Subject: [PATCH] =?UTF-8?q?file-system:=20extract=20the=20serving=20harnes?= =?UTF-8?q?s=20from=20fat=20=E2=80=94=20V1?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit fat was one binary doing four jobs; the three that are not FAT-specific move to library/kernel/file-system-harness, a Server(comptime Engine) generic over the engine type: the badge-scoped open-node table, the nine vfs handlers, the not-mounted politeness, the exit sweep, mount registration, and durable-on- close. A filesystem is now an engine plus a main that hands the harness a mounted volume; a second engine reuses the harness wholesale. Placement note: the plan said library/file-system, but the harness is a specialization of `service` (its sibling) and needs nothing from the device domain, so it lives beside service in library/kernel and stays block-free — durability rides a caller closure (Volume.flush), no backwards kernel->device dependency, no new-domain scaffolding. The engine type is inferred from resolve()'s return, so engine.zig is untouched (its Node stays module-scope). fat keeps only its FAT-specific bring-up (acquireVolume, DMA, engine.mount, the attach/detach round trip) and the three mount prefixes as data. Behavior- neutral: 13/13 across the fat/vfs/logger/IOMMU surface, nothing observable changed. This lands first so every later phase touches the harness once. --- library/kernel/build.zig | 19 +- library/kernel/file-system-harness.zig | 336 +++++++++++++++++++++++ system/services/fat/build.zig | 7 +- system/services/fat/fat.zig | 365 ++++--------------------- 4 files changed, 409 insertions(+), 318 deletions(-) create mode 100644 library/kernel/file-system-harness.zig diff --git a/library/kernel/build.zig b/library/kernel/build.zig index 4361821..0524050 100644 --- a/library/kernel/build.zig +++ b/library/kernel/build.zig @@ -90,7 +90,7 @@ pub fn build(b: *std.Build) void { // a provider that beat init to the mount needs). It also owns the subscriber // table and the fan-out, which are expressed in the envelope's vocabulary // (the reserved subscribe verb, the push floor) — hence envelope. - _ = b.addModule("service", .{ + const service = b.addModule("service", .{ .root_source_file = b.path("service.zig"), .imports = &.{ .{ .name = "channel", .module = channel }, @@ -99,6 +99,23 @@ pub fn build(b: *std.Build) void { .{ .name = "process", .module = process }, }, }); + // The filesystem serving harness (docs/file-system-development/storage-architecture.md): + // the engine-agnostic half of a filesystem service. It specializes `service` + // for the vfs protocol and is block-free (durability rides a caller closure), + // so it needs nothing from the device domain — it sits beside its sibling. + _ = b.addModule("file-system-harness", .{ + .root_source_file = b.path("file-system-harness.zig"), + .imports = &.{ + .{ .name = "ipc", .module = ipc }, + .{ .name = "process", .module = process }, + .{ .name = "service", .module = service }, + .{ .name = "time", .module = time }, + .{ .name = "file-system", .module = file_system }, + .{ .name = "envelope", .module = protocol.module("envelope") }, + .{ .name = "vfs-protocol", .module = protocol.module("vfs-protocol") }, + .{ .name = "logging", .module = logging }, + }, + }); _ = b.addModule("start", .{ .root_source_file = b.path("start.zig"), .imports = &.{ .{ .name = "process", .module = process }, .{ .name = "logging", .module = logging } }, diff --git a/library/kernel/file-system-harness.zig b/library/kernel/file-system-harness.zig new file mode 100644 index 0000000..404795d --- /dev/null +++ b/library/kernel/file-system-harness.zig @@ -0,0 +1,336 @@ +//! The filesystem serving harness: the block-client-and-engine-agnostic half of +//! a filesystem service (docs/file-system-development/storage-architecture.md). +//! Everything a filesystem process does that is NOT its on-disk format lives +//! here — establishment, the badge-scoped open-node table, the nine vfs-protocol +//! handlers, mount registration, the not-mounted-yet politeness, the exit sweep, +//! and the durable-on-close flush. A filesystem is then an ENGINE (the pure, +//! host-testable format code behind a small method set) plus a `main` that wires +//! it in, so a second filesystem reuses this wholesale — the reason it is a +//! shared library rather than per-filesystem code. +//! +//! Placement: `library/kernel`, beside its sibling `service` (the generic +//! serving harness this specializes for the vfs protocol). It is block-free — +//! the caller's `Volume.flush` closure owns durability — so it needs nothing +//! from the device domain and introduces no backwards dependency. +//! +//! `Server(Engine)` is generic over the engine TYPE, checked at compile time by +//! the calls below. An engine must expose: +//! - `pub const Node` with fields `is_directory: bool`, `size`, `mtime`; +//! - `pub const Listing` with `name_buffer`, `name_len`, `is_directory`, `size`; +//! - `current_time_epoch` a settable field (the harness stamps it per turn); +//! - resolve, createFile, createDirectory, removeFile, rename, truncate, +//! readFile, writeFile, listEntry — the signatures fat's engine.zig already has. + +const std = @import("std"); +const ipc = @import("ipc"); +const process = @import("process"); +const service = @import("service"); +const time = @import("time"); +const file_system = @import("file-system"); +const envelope = @import("envelope"); +const vfs_protocol = @import("vfs-protocol"); +const logging = @import("logging"); + +/// One prefix this filesystem mounts into the kernel mount table. `rewrite` is +/// the backend-relative prefix a path is rewritten to before it reaches the +/// engine (empty = mount the volume root at `prefix`, the common case). +pub const MountSpec = struct { prefix: []const u8, rewrite: []const u8 = "" }; + +/// The engine's node type, inferred from `resolve`'s return (`?Node`) so the +/// engine need not re-export it as a member — engine.zig keeps `Node` at module +/// scope, and this harness stays purely additive on the engine side. +fn NodeType(comptime Engine: type) type { + return @typeInfo(@typeInfo(@TypeOf(Engine.resolve)).@"fn".return_type.?).optional.child; +} + +pub fn Server(comptime Engine: type) type { + return struct { + const Node = NodeType(Engine); + + /// What a bring-up produces: the mounted engine (a stable pointer the + /// caller owns), the prefixes to install, and a durability closure the + /// harness calls on every close (the caller checks its own dirty state). + pub const Volume = struct { + engine: *Engine, + mounts: []const MountSpec, + flush: *const fn () void, + }; + + pub const Callbacks = struct { + /// Acquire and mount the volume, or null to retry on the timer. The + /// caller does the filesystem-specific bring-up (find the block + /// device, set up DMA, mount the engine) and returns a `Volume`. + bringUp: *const fn (endpoint: ipc.Handle) ?Volume, + /// The vfs contract name to bind. A filesystem serving one volume + /// binds "vfs" today; the volume-manager era hands each per-volume + /// process its own establishment and this fades. + service_name: ?[]const u8 = "vfs", + }; + + // --- the harness's own state, one set per instantiation --------------- + // A filesystem binary instantiates Server once, so these globals are the + // one server's state, exactly where fat's file-scoped globals were. + + const Serve = vfs_protocol.Protocol.Provider(void); + const Invocation = envelope.Invocation; + const Answer = envelope.Answer; + + /// What a handler returns when the thing asked for is not there — a bad + /// node id, someone else's node, an unresolved path, a refused mutation. + /// One errno for all: a filesystem's failures are all "no such thing" to + /// the file API, and *someone else's* must be indistinguishable from + /// *nobody's*, or the refusal would leak which ids are live. + const refused: isize = -envelope.ENOENT; + + /// How often to retry bring-up while unmounted. Storage arriving is + /// event-shaped (the usb chain registering, maybe after a restart), but + /// there is no subscription; a slow poll keeps the service responsive + /// (ping, terminate) while it waits and alive to catch late storage. + const mount_retry_ms = 500; + + const OpenNode = struct { used: bool = false, node: Node = undefined, owner: u32 = 0 }; + var open_nodes = [_]OpenNode{.{}} ** 32; + + var callbacks: Callbacks = undefined; + var service_endpoint: ipc.Handle = 0; + var engine_ptr: ?*Engine = null; + var volume_flush: *const fn () void = undefined; + var mounted: bool = false; + + fn allocOpen() ?usize { + for (&open_nodes, 0..) |*o, i| { + if (!o.used) return i; + } + return null; + } + + /// The open node `id` names **for `owner`** — null unless in range, in + /// use, and this client's own. Ids are small integers from a table of 32, + /// trivially guessable, so this badge check is the scope + /// (docs/os-development/protocol-namespace.md). The owner is a TASK, not a + /// process, because the badge is: a threaded client reads a node from the + /// thread that opened it, and the exit sweep releases a worker's handles. + fn openFor(id: u64, owner: u32) ?*OpenNode { + if (id >= open_nodes.len) return null; + const o = &open_nodes[@intCast(id)]; + if (!o.used or o.owner != owner) return null; + return o; + } + + const ParentLeaf = struct { parent: []const u8, leaf: []const u8 }; + + // Split a path: "/a/b" -> ("/a", "b"); "/b" -> ("/", "b"); "b" -> ("/", "b"). + fn splitParent(path: []const u8) ParentLeaf { + const slash = std.mem.lastIndexOfScalar(u8, path, '/'); + return .{ + .parent = if (slash) |s| (if (s == 0) "/" else path[0..s]) else "/", + .leaf = if (slash) |s| path[s + 1 ..] else path, + }; + } + + fn onOpen(_: void, invocation: Invocation(vfs_protocol.Open), answer: Answer(vfs_protocol.Opened)) isize { + const fs = engine_ptr orelse return refused; + const path = invocation.tail; + const flags = invocation.request.flags; + var node = fs.resolve(path); + if (node == null and flags & vfs_protocol.create != 0) { + const split = splitParent(path); + const parent = fs.resolve(split.parent) orelse return refused; + node = fs.createFile(parent, split.leaf); + } + var resolved = node orelse return refused; + // O_TRUNC: replace contents rather than overwrite in place (frees the + // old chain, so a shorter rewrite leaves no stale tail). + if (flags & vfs_protocol.truncate != 0 and !resolved.is_directory) { + fs.truncate(&resolved); + } + const index = allocOpen() orelse return refused; + open_nodes[index] = .{ .used = true, .node = resolved, .owner = invocation.sender }; + answer.set(.{ .node = index }); + return 0; + } + + fn onRead(_: void, invocation: Invocation(vfs_protocol.Read), answer: Answer(void)) isize { + const fs = engine_ptr orelse return refused; + const o = openFor(invocation.target, invocation.sender) orelse return refused; + const into = answer.tail(); + const want = @min(@as(usize, invocation.request.len), into.len); + return @intCast(fs.readFile(o.node, @intCast(invocation.request.offset), into[0..want])); + } + + fn onWrite(_: void, invocation: Invocation(vfs_protocol.Write), answer: Answer(vfs_protocol.Written)) isize { + const fs = engine_ptr orelse return refused; + const o = openFor(invocation.target, invocation.sender) orelse return refused; + const data = invocation.tail[0..@min(invocation.tail.len, invocation.request.len)]; + const n = fs.writeFile(&o.node, @intCast(invocation.request.offset), data); + answer.set(.{ .count = @intCast(n) }); + return 0; + } + + fn onStatus(_: void, invocation: Invocation(void), answer: Answer(vfs_protocol.FileStatus)) isize { + const o = openFor(invocation.target, invocation.sender) orelse return refused; + const kind: vfs_protocol.NodeKind = if (o.node.is_directory) .directory else .regular; + answer.set(.{ .size = o.node.size, .kind = @intFromEnum(kind), .mtime = o.node.mtime }); + return 0; + } + + /// One entry per call. End of directory — not a directory, or a cursor + /// past the last child — is an entry with no name. + fn onReaddir(_: void, invocation: Invocation(vfs_protocol.Readdir), answer: Answer(vfs_protocol.DirectoryEntry)) isize { + const fs = engine_ptr orelse return refused; + const o = openFor(invocation.target, invocation.sender) orelse return refused; + if (!o.node.is_directory) { + answer.set(.{}); + return 0; + } + const listing = fs.listEntry(o.node, @intCast(invocation.request.cursor)) orelse { + answer.set(.{}); + return 0; + }; + const kind: vfs_protocol.NodeKind = if (listing.is_directory) .directory else .regular; + const into = answer.tail(); + const name_len = @min(listing.name_len, into.len); + @memcpy(into[0..name_len], listing.name_buffer[0..name_len]); + answer.set(.{ .kind = @intFromEnum(kind), .name_len = @intCast(name_len), .size = listing.size }); + return @intCast(name_len); + } + + /// Closing is scoped like any other node operation: a client releases its + /// own handles and nobody else's, and a foreign/free/out-of-range id is + /// refused identically so a close cannot probe which ids are live. + fn onClose(_: void, invocation: Invocation(void), _: Answer(void)) isize { + const o = openFor(invocation.target, invocation.sender) orelse return refused; + o.used = false; + // Durable-on-close: the caller's flush commits any device write cache + // to stable media now. This is what makes init's shutdown log flush + // survive a real power-off, and the right default for removable media. + volume_flush(); + return 0; + } + + fn onMakeDirectory(_: void, invocation: Invocation(void), _: Answer(void)) isize { + const fs = engine_ptr orelse return refused; + const path = invocation.tail; + if (fs.resolve(path) != null) return refused; // already exists + const split = splitParent(path); + const parent = fs.resolve(split.parent) orelse return refused; + if (fs.createDirectory(parent, split.leaf) == null) return refused; + return 0; + } + + fn onUnlink(_: void, invocation: Invocation(void), _: Answer(void)) isize { + const fs = engine_ptr orelse return refused; + const split = splitParent(invocation.tail); + const parent = fs.resolve(split.parent) orelse return refused; + if (!fs.removeFile(parent, split.leaf)) return refused; + return 0; + } + + fn onRename(_: void, invocation: Invocation(void), _: Answer(void)) isize { + const fs = engine_ptr orelse return refused; + const both = invocation.tail; + const separator = std.mem.indexOfScalar(u8, both, 0) orelse return refused; + const old_split = splitParent(both[0..separator]); + const new_split = splitParent(both[separator + 1 ..]); + if (!std.mem.eql(u8, old_split.parent, new_split.parent)) return refused; // same-directory only + const parent = fs.resolve(old_split.parent) orelse return refused; + if (!fs.rename(parent, old_split.leaf, new_split.leaf)) return refused; + return 0; + } + + /// The verbs this backend implements. `mount`/`unmount`/`bind` are absent + /// on purpose — path routing is the kernel's, and only init serves `bind`. + const handlers = Serve.Handlers{ + .open = onOpen, + .close = onClose, + .read = onRead, + .write = onWrite, + .status = onStatus, + .readdir = onReaddir, + .mkdir = onMakeDirectory, + .unlink = onUnlink, + .rename = onRename, + }; + + /// The vfs protocol carries no capability, so `arrived` is never claimed + /// — the harness's ownership rule then closes whatever a caller attached, + /// so a request carrying one cannot spend a slot of this server's table. + fn onMessage(message: []const u8, out: []u8, sender: u32, arrived: *ipc.Arrival) usize { + _ = arrived; + const fs = engine_ptr; + // Storage not up yet: fail politely, whatever was asked — clients retry. + if (!mounted or fs == null) { + const status = envelope.Status{ .status = refused, .len = 0 }; + @memcpy(out[0..envelope.prefix_size], std.mem.asBytes(&status)); + return envelope.prefix_size; + } + // Stamp create/write with the current wall-clock time (mtime): cheap, + // and it keeps the engine pure (it takes the time as data, not a call). + fs.?.current_time_epoch = time.wallClock(); + return Serve.dispatch({}, handlers, message, sender, null, out); + } + + /// A process-exit event releases every open handle the dead client held, + /// so a crashed reader cannot pin table slots. + fn onNotification(badge: u64) void { + const got = ipc.Received{ .len = 0, .badge = badge, .cap = null }; + if (got.isTimer()) { + tryBringUp(); + if (!mounted) _ = time.timerOnce(service_endpoint, mount_retry_ms); + return; + } + if (!got.isChildExit()) return; + const dead = got.childProcessId(); + var released: u32 = 0; + for (&open_nodes) |*o| { + if (o.used and o.owner == dead) { + o.* = .{}; + released += 1; + } + } + if (released != 0) std.log.info("released {d} handle(s) for dead client {d}", .{ released, dead }); + } + + /// One bring-up attempt: ask the caller for a mounted volume, and on + /// success install its mounts and go live. A failure leaves everything + /// untouched for the next tick. + fn tryBringUp() void { + if (mounted) return; + const volume = callbacks.bringUp(service_endpoint) orelse return; + engine_ptr = volume.engine; + volume_flush = volume.flush; + for (volume.mounts) |m| { + const ok = if (m.rewrite.len == 0) + file_system.mount(m.prefix, service_endpoint) + else + file_system.mountRewritten(m.prefix, service_endpoint, m.rewrite); + if (ok) { + std.log.info("mounted {s}", .{m.prefix}); + } else { + std.log.info("could not mount {s}", .{m.prefix}); + } + } + mounted = true; + } + + fn initialise(endpoint: ipc.Handle) bool { + service_endpoint = endpoint; + // Sweep a dead client's open handles via the published exit events — + // clients hold OUR node ids directly, so a crash must not pin slots. + _ = process.subscribeExits(endpoint); + tryBringUp(); + if (!mounted) _ = time.timerOnce(endpoint, mount_retry_ms); + return true; // serve regardless: requests fail politely until storage mounts + } + + pub fn run(cb: Callbacks) void { + callbacks = cb; + service.run(vfs_protocol.message_maximum, .{ + .service = cb.service_name, + .init = initialise, + .on_message = onMessage, + .on_notification = onNotification, + }); + } + }; +} diff --git a/system/services/fat/build.zig b/system/services/fat/build.zig index 039af65..b398bfc 100644 --- a/system/services/fat/build.zig +++ b/system/services/fat/build.zig @@ -10,10 +10,9 @@ pub fn build(b: *std.Build) void { .name = "fat", .root_source_file = b.path("fat.zig"), .imports = &.{ - "block", "channel", "device-manager-protocol", "driver", - "envelope", "file-system", "ipc", "logging", - "memory", "process", "service", "time", - "vfs-protocol", + "block", "channel", "device-manager-protocol", + "driver", "envelope", "file-system-harness", + "ipc", "logging", "memory", }, }); b.installArtifact(exe); diff --git a/system/services/fat/fat.zig b/system/services/fat/fat.zig index 30d4626..700ab91 100644 --- a/system/services/fat/fat.zig +++ b/system/services/fat/fat.zig @@ -1,40 +1,30 @@ -//! system/services/fat — the FAT filesystem server. Spawned as a boot service, it -//! opens the block device (a USB stick via usb-storage) under `.block`, mounts the -//! FAT filesystem on it (the pure engine in engine.zig), and mounts itself into -//! the VFS at /volumes/usb. From then on the VFS forwards every open/read/write/ -//! status/readdir/close under /volumes/usb to this server, which serves the same -//! vfs-protocol as a backend — turning block reads into file reads. +//! system/services/fat — the FAT filesystem service. This is FAT's FAT-specific +//! half: it finds its block device, sets up the DMA bounce buffer, mounts the +//! FAT engine on it, and hands the mounted volume to the shared filesystem +//! harness (library/kernel/file-system-harness), which owns everything else — +//! the vfs-protocol serving, the open-node table, mount registration, the exit +//! sweep, durable-on-close. The engine (engine.zig) is the pure, host-testable +//! format code; on-disk.zig its byte layout. A second filesystem reuses the +//! harness and supplies its own engine +//! (docs/file-system-development/storage-architecture.md). //! //! The block data path never crosses IPC: a DMA bounce buffer is handed to the -//! block driver by physical address, and the engine copies sectors in and out of -//! it. +//! block driver by physical address, and the engine copies sectors in and out. const std = @import("std"); const channel = @import("channel"); const device_manager_protocol = @import("device-manager-protocol"); const driver = @import("driver"); const ipc = @import("ipc"); -const process = @import("process"); -const service = @import("service"); -const time = @import("time"); const block = @import("block"); -const file_system = @import("file-system"); const memory = @import("memory"); const logging = @import("logging"); const engine = @import("engine.zig"); -const on_disk = @import("on-disk.zig"); const envelope = @import("envelope"); -const vfs_protocol = @import("vfs-protocol"); +const harness = @import("file-system-harness"); -/// The generated vfs dispatch, bound to this server. There is one FAT volume per -/// process, so the handler context is empty and the state stays where it was: in -/// this file's globals. -const Serve = vfs_protocol.Protocol.Provider(void); - -const Invocation = envelope.Invocation; -const Answer = envelope.Answer; - -const mount_point = "/volumes/usb"; +/// The serving harness, specialized for the FAT engine. One volume per process. +const Harness = harness.Server(engine.FileSystem); // The engine's BlockDevice, backed by the `.block` driver plus a DMA bounce // buffer the driver reads/writes by physical address. @@ -68,69 +58,28 @@ var ipc_block: IpcBlock = undefined; // file close — so writes are committed to stable media before a power-off. var device_dirty: bool = false; var filesystem: engine.FileSystem = undefined; - -// Open handles clients hold against this backend: each maps a node id to a -// resolved engine node, and to the client that opened it. `owner` is the -// kernel-stamped badge of the opening task — the only source identity there is. -const OpenNode = struct { used: bool = false, node: engine.Node = undefined, owner: u32 = 0 }; -var open_nodes = [_]OpenNode{.{}} ** 32; - -fn allocOpen() ?usize { - for (&open_nodes, 0..) |*o, i| { - if (!o.used) return i; - } - return null; -} - -/// The open node `id` names **for `owner`** — null unless the id is in range, in -/// use, and this client's own. Node ids are small integers drawn from a table of -/// thirty-two, so they are trivially guessable; before this check every client -/// honoured every other client's ids, which is the hole -/// docs/os-development/protocol-namespace.md names ("handles must be scoped per -/// client — validated against the badge"). Nothing else about them changed: they -/// are still per-session, still swept when their owner dies. -/// -/// The owner is a *task*, not a process, because the badge is: a threaded client -/// reads and writes a node from the thread that opened it, exactly as the exit -/// sweep already released a worker thread's handles when that thread died. -fn openFor(id: u64, owner: u32) ?*OpenNode { - if (id >= open_nodes.len) return null; - const o = &open_nodes[@intCast(id)]; - if (!o.used or o.owner != owner) return null; - return o; -} - -/// What a handler returns when the thing asked for is not there — a bad node id, -/// a node that is someone else's, a path that does not resolve, a mutation the -/// volume refused. One errno for all of them, because a filesystem's failures are -/// all "no such thing" as far as the file API can act on them — and because -/// *someone else's* must be indistinguishable from *nobody's*, or the refusal -/// would itself tell a prober which ids are live (the same discipline the -/// protocol namespace's refused open follows). -const refused: isize = -envelope.ENOENT; - -/// How often to look for a block device while none is mounted. Storage arriving -/// is EVENT-shaped (the usb chain registering, possibly after a driver restart), -/// but the registry has no subscription — a slow poll from our own harness loop -/// keeps the service responsive (ping, terminate) while it waits, and keeps it -/// alive to catch storage that appears LATE (a restarted usb-storage after a -/// transient failure — the resilience half of docs/logging.md's storage story). -const mount_retry_ms = 500; - -var mounted = false; -var service_endpoint: ipc.Handle = 0; /// The one channel to the device manager, opened on first need and kept — the /// poll retries on it, never spending a handle-table slot per attempt. var manager_handle: ?ipc.Handle = null; +/// The prefixes this volume installs: /volumes/usb from the volume root, plus +/// the two hierarchy subtrees the boot volume carries (rewrite == prefix), so +/// hierarchy paths (the logger's /system/logs) stay decoupled from which volume +/// backs them. +const fat_mounts = [_]harness.MountSpec{ + .{ .prefix = "/volumes/usb" }, + .{ .prefix = "/system/configuration", .rewrite = "/system/configuration" }, + .{ .prefix = "/system/logs", .rewrite = "/system/logs" }, +}; + /// Find the volume's provider through the device manager (establishment by /// lineage, communication.md "Establishment: two planes" — `block` is not a /// registry name; one storage process serves each stick): enumerate the /// manager's tree, take the FIRST usb mass-storage child by enumeration order /// (deterministic within a boot; single-volume by construction, and choosing -/// the BOOT volume by content when two sticks are present is the M21 remount +/// the BOOT volume by content when two sticks are present is the volume-manager /// track), and consumer-hello for the channel of the driver bound to it. -/// Null until the chain is up — the caller's poll retries. +/// Null until the chain is up — the harness's poll retries. fn acquireVolume() ?block.Device { const manager = manager_handle orelse opened: { const handle = channel.openEndpoint("device-manager") orelse return null; @@ -170,47 +119,45 @@ fn acquireVolume() ?block.Device { } } -fn initialise(endpoint: ipc.Handle) bool { - service_endpoint = endpoint; - _ = logging.write("/system/services/fat: starting, waiting for a block device\n"); - // With the router in the kernel, clients hold OUR node ids directly; sweep - // a dead client's open handles via the published exit events (the pattern - // the old userspace router used for its own table). - _ = process.subscribeExits(endpoint); - tryBringUp(); - if (!mounted) _ = time.timerOnce(endpoint, mount_retry_ms); - return true; // serve regardless: requests fail politely until storage mounts +/// Durable-on-close: commit the device write cache if any block reached it since +/// the last flush. The harness calls this on every close; the dirty check keeps +/// it cheap. `device_dirty` lives here because `IpcBlock.writeBlocks` sets it. +fn flushIfDirty() void { + if (device_dirty) { + _ = ipc_block.device.flush(); + device_dirty = false; + } } -/// One storage bring-up attempt: block device -> FAT mount -> VFS mounts. Sets -/// `mounted` on success; a failure leaves everything untouched for the next tick. -fn tryBringUp() void { - if (mounted) return; - const device = acquireVolume() orelse return; +/// FAT bring-up: find the block device, set up DMA, mount the engine, and hand +/// the volume to the harness — or null to retry on the harness's timer. +fn fatBringUp(endpoint: ipc.Handle) ?Harness.Volume { + _ = endpoint; + const device = acquireVolume() orelse return null; const geometry = device.geometry() orelse { _ = logging.write("/system/services/fat: block geometry unavailable\n"); - return; + return null; }; - // Shareable so the buffer's capability can be attached down the chain (block server - // -> controller), making its physical addresses reachable by the device under an - // enforcing IOMMU. No-op binding otherwise. - const bounce = memory.dmaAlloc(engine.max_transfer_sectors * 512, memory.dma_coherent | memory.dma_shareable) orelse return; + // Shareable so the buffer's capability can be attached down the chain (block + // server -> controller), making its physical addresses reachable by the + // device under an enforcing IOMMU. No-op binding otherwise. + const bounce = memory.dmaAlloc(engine.max_transfer_sectors * 512, memory.dma_coherent | memory.dma_shareable) orelse return null; if (bounce.handle) |handle| { // Attach, detach, and attach again: the round trip exercises BOTH verbs - // of the DMA-window lifecycle through the whole chain (fat → storage → - // bus → kernel) on every boot, so a broken detach fails every fat case + // of the DMA-window lifecycle through the whole chain (fat -> storage -> + // bus -> kernel) on every boot, so a broken detach fails every fat case // rather than lying dormant until the first buffer replacement. if (!device.attach(handle)) { _ = logging.write("/system/services/fat: could not attach the DMA bounce buffer\n"); - return; + return null; } if (!device.detach(handle)) { _ = logging.write("/system/services/fat: could not detach the DMA bounce buffer\n"); - return; + return null; } if (!device.attach(handle)) { _ = logging.write("/system/services/fat: could not re-attach the DMA bounce buffer\n"); - return; + return null; } _ = ipc.close(handle); // the binding holds its own reference now } @@ -225,222 +172,14 @@ fn tryBringUp() void { }; filesystem = engine.FileSystem.mount(block_device) orelse { _ = logging.write("/system/services/fat: not a FAT filesystem\n"); - return; + return null; }; std.log.info("mounted FAT ({s}, {d} clusters, partition lba {d})", .{ @tagName(filesystem.geometry.fat_type), filesystem.geometry.cluster_count, filesystem.base_lba }); - // Mount ourselves into the kernel VFS at /volumes/usb — and serve - // /system/configuration and /system/logs from the volume's identically-named - // subtrees (the boot volume is hierarchy-shaped, so rewrite == prefix), so - // hierarchy paths (the logger's /system/logs) stay decoupled from which - // volume carries them. - if (file_system.mount(mount_point, endpointForMount())) { - std.log.info("mounted {s}", .{mount_point}); - } else { - _ = logging.write("/system/services/fat: could not mount /volumes/usb\n"); - } - if (file_system.mountRewritten("/system/configuration", endpointForMount(), "/system/configuration")) { - std.log.info("mounted /system/configuration", .{}); - } else { - _ = logging.write("/system/services/fat: could not mount /system/configuration\n"); - } - if (file_system.mountRewritten("/system/logs", endpointForMount(), "/system/logs")) { - std.log.info("mounted /system/logs", .{}); - } else { - _ = logging.write("/system/services/fat: could not mount /system/logs\n"); - } - mounted = true; -} - -fn endpointForMount() ipc.Handle { - return service_endpoint; -} - -/// A subscribed process-exit event: release every open handle the dead client -/// held, so a crashed reader can't pin table slots (or, later, locks). -fn onNotification(badge: u64) void { - const got = ipc.Received{ .len = 0, .badge = badge, .cap = null }; - if (got.isTimer()) { - tryBringUp(); - if (!mounted) _ = time.timerOnce(service_endpoint, mount_retry_ms); - return; - } - if (!got.isChildExit()) return; - const dead = got.childProcessId(); - var released: u32 = 0; - for (&open_nodes) |*o| { - if (o.used and o.owner == dead) { - o.* = .{}; - released += 1; - } - } - if (released != 0) std.log.info("released {d} handle(s) for dead client {d}", .{ released, dead }); -} - -const ParentLeaf = struct { parent: []const u8, leaf: []const u8 }; - -// Split a path into its parent directory and final component: "/a/b" -> ("/a", -// "b"); "/b" -> ("/", "b"); "b" -> ("/", "b"). -fn splitParent(path: []const u8) ParentLeaf { - const slash = std.mem.lastIndexOfScalar(u8, path, '/'); - return .{ - .parent = if (slash) |s| (if (s == 0) "/" else path[0..s]) else "/", - .leaf = if (slash) |s| path[s + 1 ..] else path, - }; -} - -fn onOpen(_: void, invocation: Invocation(vfs_protocol.Open), answer: Answer(vfs_protocol.Opened)) isize { - const path = invocation.tail; - const flags = invocation.request.flags; - var node = filesystem.resolve(path); - if (node == null and flags & vfs_protocol.create != 0) { - const split = splitParent(path); - const parent = filesystem.resolve(split.parent) orelse return refused; - node = filesystem.createFile(parent, split.leaf); - } - var resolved = node orelse return refused; - // O_TRUNC: replace an existing file's contents rather than overwriting in place - // (frees the old chain, so a shorter rewrite leaves no stale tail). - if (flags & vfs_protocol.truncate != 0 and !resolved.is_directory) { - filesystem.truncate(&resolved); - } - const index = allocOpen() orelse return refused; - open_nodes[index] = .{ .used = true, .node = resolved, .owner = invocation.sender }; - answer.set(.{ .node = index }); - return 0; -} - -fn onRead(_: void, invocation: Invocation(vfs_protocol.Read), answer: Answer(void)) isize { - const o = openFor(invocation.target, invocation.sender) orelse return refused; - const into = answer.tail(); - const want = @min(@as(usize, invocation.request.len), into.len); - return @intCast(filesystem.readFile(o.node, @intCast(invocation.request.offset), into[0..want])); -} - -fn onWrite(_: void, invocation: Invocation(vfs_protocol.Write), answer: Answer(vfs_protocol.Written)) isize { - const o = openFor(invocation.target, invocation.sender) orelse return refused; - const data = invocation.tail[0..@min(invocation.tail.len, invocation.request.len)]; - const n = filesystem.writeFile(&o.node, @intCast(invocation.request.offset), data); - answer.set(.{ .count = @intCast(n) }); - return 0; -} - -fn onStatus(_: void, invocation: Invocation(void), answer: Answer(vfs_protocol.FileStatus)) isize { - const o = openFor(invocation.target, invocation.sender) orelse return refused; - const kind: vfs_protocol.NodeKind = if (o.node.is_directory) .directory else .regular; - answer.set(.{ .size = o.node.size, .kind = @intFromEnum(kind), .mtime = o.node.mtime }); - return 0; -} - -/// One entry per call. End of directory — a node that is not a directory, or a -/// cursor past the last child — is an entry with no name, which is how the -/// protocol spells it now that the reply's length always counts the fixed part. -fn onReaddir(_: void, invocation: Invocation(vfs_protocol.Readdir), answer: Answer(vfs_protocol.DirectoryEntry)) isize { - const o = openFor(invocation.target, invocation.sender) orelse return refused; - if (!o.node.is_directory) { - answer.set(.{}); - return 0; - } - const listing = filesystem.listEntry(o.node, @intCast(invocation.request.cursor)) orelse { - answer.set(.{}); - return 0; - }; - const kind: vfs_protocol.NodeKind = if (listing.is_directory) .directory else .regular; - const into = answer.tail(); - const name_len = @min(listing.name_len, into.len); - @memcpy(into[0..name_len], listing.name_buffer[0..name_len]); - answer.set(.{ .kind = @intFromEnum(kind), .name_len = @intCast(name_len), .size = listing.size }); - return @intCast(name_len); -} - -/// Closing is an operation on a node like any other, so it is scoped like any -/// other: a client may release its own handles and nobody else's. An id that is -/// not the caller's — free, out of range, or another client's — is refused -/// identically, so a close cannot be used to ask which ids are live either. -fn onClose(_: void, invocation: Invocation(void), _: Answer(void)) isize { - const o = openFor(invocation.target, invocation.sender) orelse return refused; - o.used = false; - // Durable-on-close: if any block reached the device since the last flush, - // commit its cache to stable media now (best-effort). This is what makes - // init's shutdown log flush survive a real power-off, and is the right - // default for removable media the user may unplug. - if (device_dirty) { - _ = ipc_block.device.flush(); - device_dirty = false; - } - return 0; -} - -fn onMakeDirectory(_: void, invocation: Invocation(void), _: Answer(void)) isize { - const path = invocation.tail; - if (filesystem.resolve(path) != null) return refused; // already exists — no duplicate entries - const split = splitParent(path); - const parent = filesystem.resolve(split.parent) orelse return refused; - if (filesystem.createDirectory(parent, split.leaf) == null) return refused; - return 0; -} - -fn onUnlink(_: void, invocation: Invocation(void), _: Answer(void)) isize { - const split = splitParent(invocation.tail); - const parent = filesystem.resolve(split.parent) orelse return refused; - if (!filesystem.removeFile(parent, split.leaf)) return refused; - return 0; -} - -fn onRename(_: void, invocation: Invocation(void), _: Answer(void)) isize { - const both = invocation.tail; - const separator = std.mem.indexOfScalar(u8, both, 0) orelse return refused; - const old_split = splitParent(both[0..separator]); - const new_split = splitParent(both[separator + 1 ..]); - // Same-directory rename only. - if (!std.mem.eql(u8, old_split.parent, new_split.parent)) return refused; - const parent = filesystem.resolve(old_split.parent) orelse return refused; - if (!filesystem.rename(parent, old_split.leaf, new_split.leaf)) return refused; - return 0; -} - -/// The verbs this backend implements. The three it leaves out — `mount`, -/// `unmount`, `bind` — answer `-ENOSYS` from the generated dispatch, which is -/// exactly right: path routing is the kernel's now, and only init implements -/// `bind` (docs/os-development/protocol-namespace.md). `describe` is the -/// envelope's own. -const handlers = Serve.Handlers{ - .open = onOpen, - .close = onClose, - .read = onRead, - .write = onWrite, - .status = onStatus, - .readdir = onReaddir, - .mkdir = onMakeDirectory, - .unlink = onUnlink, - .rename = onRename, -}; - -/// The vfs protocol has no operation that takes a capability, so `arrived` is -/// never claimed here — which, under the harness's ownership rule, means the -/// loop closes whatever a caller attached. That is the point of the rule: this -/// callback used to discard a `?ipc.Handle` and every request carrying one — a -/// legal thing for any client to do — spent a slot of the VFS server's -/// thirty-two until it could accept no capability at all. -fn onMessage(message: []const u8, out: []u8, sender: u32, arrived: *ipc.Arrival) usize { - _ = arrived; - // Storage not up (yet): fail politely, whatever was asked — clients retry. - if (!mounted) { - const status = envelope.Status{ .status = refused, .len = 0 }; - @memcpy(out[0..envelope.prefix_size], std.mem.asBytes(&status)); - return envelope.prefix_size; - } - // Stamp create/write with the current wall-clock time (mtime). Cheap, and it - // keeps the engine pure (it takes the time as data, not a syscall). - filesystem.current_time_epoch = time.wallClock(); - return Serve.dispatch({}, handlers, message, sender, null, out); + return .{ .engine = &filesystem, .mounts = &fat_mounts, .flush = flushIfDirty }; } pub fn main() void { - service.run(vfs_protocol.message_maximum, .{ - .service = "vfs", - .init = initialise, - .on_message = onMessage, - .on_notification = onNotification, - }); + _ = logging.write("/system/services/fat: starting, waiting for a block device\n"); + Harness.run(.{ .bringUp = fatBringUp }); }