file-system: extract the serving harness from fat — V1
fat was one binary doing four jobs; the three that are not FAT-specific move to library/kernel/file-system-harness, a Server(comptime Engine) generic over the engine type: the badge-scoped open-node table, the nine vfs handlers, the not-mounted politeness, the exit sweep, mount registration, and durable-on- close. A filesystem is now an engine plus a main that hands the harness a mounted volume; a second engine reuses the harness wholesale. Placement note: the plan said library/file-system, but the harness is a specialization of `service` (its sibling) and needs nothing from the device domain, so it lives beside service in library/kernel and stays block-free — durability rides a caller closure (Volume.flush), no backwards kernel->device dependency, no new-domain scaffolding. The engine type is inferred from resolve()'s return, so engine.zig is untouched (its Node stays module-scope). fat keeps only its FAT-specific bring-up (acquireVolume, DMA, engine.mount, the attach/detach round trip) and the three mount prefixes as data. Behavior- neutral: 13/13 across the fat/vfs/logger/IOMMU surface, nothing observable changed. This lands first so every later phase touches the harness once.
This commit is contained in:
@@ -90,7 +90,7 @@ pub fn build(b: *std.Build) void {
|
||||
// a provider that beat init to the mount needs). It also owns the subscriber
|
||||
// table and the fan-out, which are expressed in the envelope's vocabulary
|
||||
// (the reserved subscribe verb, the push floor) — hence envelope.
|
||||
_ = b.addModule("service", .{
|
||||
const service = b.addModule("service", .{
|
||||
.root_source_file = b.path("service.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "channel", .module = channel },
|
||||
@@ -99,6 +99,23 @@ pub fn build(b: *std.Build) void {
|
||||
.{ .name = "process", .module = process },
|
||||
},
|
||||
});
|
||||
// The filesystem serving harness (docs/file-system-development/storage-architecture.md):
|
||||
// the engine-agnostic half of a filesystem service. It specializes `service`
|
||||
// for the vfs protocol and is block-free (durability rides a caller closure),
|
||||
// so it needs nothing from the device domain — it sits beside its sibling.
|
||||
_ = b.addModule("file-system-harness", .{
|
||||
.root_source_file = b.path("file-system-harness.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "ipc", .module = ipc },
|
||||
.{ .name = "process", .module = process },
|
||||
.{ .name = "service", .module = service },
|
||||
.{ .name = "time", .module = time },
|
||||
.{ .name = "file-system", .module = file_system },
|
||||
.{ .name = "envelope", .module = protocol.module("envelope") },
|
||||
.{ .name = "vfs-protocol", .module = protocol.module("vfs-protocol") },
|
||||
.{ .name = "logging", .module = logging },
|
||||
},
|
||||
});
|
||||
_ = b.addModule("start", .{
|
||||
.root_source_file = b.path("start.zig"),
|
||||
.imports = &.{ .{ .name = "process", .module = process }, .{ .name = "logging", .module = logging } },
|
||||
|
||||
@@ -0,0 +1,336 @@
|
||||
//! The filesystem serving harness: the block-client-and-engine-agnostic half of
|
||||
//! a filesystem service (docs/file-system-development/storage-architecture.md).
|
||||
//! Everything a filesystem process does that is NOT its on-disk format lives
|
||||
//! here — establishment, the badge-scoped open-node table, the nine vfs-protocol
|
||||
//! handlers, mount registration, the not-mounted-yet politeness, the exit sweep,
|
||||
//! and the durable-on-close flush. A filesystem is then an ENGINE (the pure,
|
||||
//! host-testable format code behind a small method set) plus a `main` that wires
|
||||
//! it in, so a second filesystem reuses this wholesale — the reason it is a
|
||||
//! shared library rather than per-filesystem code.
|
||||
//!
|
||||
//! Placement: `library/kernel`, beside its sibling `service` (the generic
|
||||
//! serving harness this specializes for the vfs protocol). It is block-free —
|
||||
//! the caller's `Volume.flush` closure owns durability — so it needs nothing
|
||||
//! from the device domain and introduces no backwards dependency.
|
||||
//!
|
||||
//! `Server(Engine)` is generic over the engine TYPE, checked at compile time by
|
||||
//! the calls below. An engine must expose:
|
||||
//! - `pub const Node` with fields `is_directory: bool`, `size`, `mtime`;
|
||||
//! - `pub const Listing` with `name_buffer`, `name_len`, `is_directory`, `size`;
|
||||
//! - `current_time_epoch` a settable field (the harness stamps it per turn);
|
||||
//! - resolve, createFile, createDirectory, removeFile, rename, truncate,
|
||||
//! readFile, writeFile, listEntry — the signatures fat's engine.zig already has.
|
||||
|
||||
const std = @import("std");
|
||||
const ipc = @import("ipc");
|
||||
const process = @import("process");
|
||||
const service = @import("service");
|
||||
const time = @import("time");
|
||||
const file_system = @import("file-system");
|
||||
const envelope = @import("envelope");
|
||||
const vfs_protocol = @import("vfs-protocol");
|
||||
const logging = @import("logging");
|
||||
|
||||
/// One prefix this filesystem mounts into the kernel mount table. `rewrite` is
|
||||
/// the backend-relative prefix a path is rewritten to before it reaches the
|
||||
/// engine (empty = mount the volume root at `prefix`, the common case).
|
||||
pub const MountSpec = struct { prefix: []const u8, rewrite: []const u8 = "" };
|
||||
|
||||
/// The engine's node type, inferred from `resolve`'s return (`?Node`) so the
|
||||
/// engine need not re-export it as a member — engine.zig keeps `Node` at module
|
||||
/// scope, and this harness stays purely additive on the engine side.
|
||||
fn NodeType(comptime Engine: type) type {
|
||||
return @typeInfo(@typeInfo(@TypeOf(Engine.resolve)).@"fn".return_type.?).optional.child;
|
||||
}
|
||||
|
||||
pub fn Server(comptime Engine: type) type {
|
||||
return struct {
|
||||
const Node = NodeType(Engine);
|
||||
|
||||
/// What a bring-up produces: the mounted engine (a stable pointer the
|
||||
/// caller owns), the prefixes to install, and a durability closure the
|
||||
/// harness calls on every close (the caller checks its own dirty state).
|
||||
pub const Volume = struct {
|
||||
engine: *Engine,
|
||||
mounts: []const MountSpec,
|
||||
flush: *const fn () void,
|
||||
};
|
||||
|
||||
pub const Callbacks = struct {
|
||||
/// Acquire and mount the volume, or null to retry on the timer. The
|
||||
/// caller does the filesystem-specific bring-up (find the block
|
||||
/// device, set up DMA, mount the engine) and returns a `Volume`.
|
||||
bringUp: *const fn (endpoint: ipc.Handle) ?Volume,
|
||||
/// The vfs contract name to bind. A filesystem serving one volume
|
||||
/// binds "vfs" today; the volume-manager era hands each per-volume
|
||||
/// process its own establishment and this fades.
|
||||
service_name: ?[]const u8 = "vfs",
|
||||
};
|
||||
|
||||
// --- the harness's own state, one set per instantiation ---------------
|
||||
// A filesystem binary instantiates Server once, so these globals are the
|
||||
// one server's state, exactly where fat's file-scoped globals were.
|
||||
|
||||
const Serve = vfs_protocol.Protocol.Provider(void);
|
||||
const Invocation = envelope.Invocation;
|
||||
const Answer = envelope.Answer;
|
||||
|
||||
/// What a handler returns when the thing asked for is not there — a bad
|
||||
/// node id, someone else's node, an unresolved path, a refused mutation.
|
||||
/// One errno for all: a filesystem's failures are all "no such thing" to
|
||||
/// the file API, and *someone else's* must be indistinguishable from
|
||||
/// *nobody's*, or the refusal would leak which ids are live.
|
||||
const refused: isize = -envelope.ENOENT;
|
||||
|
||||
/// How often to retry bring-up while unmounted. Storage arriving is
|
||||
/// event-shaped (the usb chain registering, maybe after a restart), but
|
||||
/// there is no subscription; a slow poll keeps the service responsive
|
||||
/// (ping, terminate) while it waits and alive to catch late storage.
|
||||
const mount_retry_ms = 500;
|
||||
|
||||
const OpenNode = struct { used: bool = false, node: Node = undefined, owner: u32 = 0 };
|
||||
var open_nodes = [_]OpenNode{.{}} ** 32;
|
||||
|
||||
var callbacks: Callbacks = undefined;
|
||||
var service_endpoint: ipc.Handle = 0;
|
||||
var engine_ptr: ?*Engine = null;
|
||||
var volume_flush: *const fn () void = undefined;
|
||||
var mounted: bool = false;
|
||||
|
||||
fn allocOpen() ?usize {
|
||||
for (&open_nodes, 0..) |*o, i| {
|
||||
if (!o.used) return i;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// The open node `id` names **for `owner`** — null unless in range, in
|
||||
/// use, and this client's own. Ids are small integers from a table of 32,
|
||||
/// trivially guessable, so this badge check is the scope
|
||||
/// (docs/os-development/protocol-namespace.md). The owner is a TASK, not a
|
||||
/// process, because the badge is: a threaded client reads a node from the
|
||||
/// thread that opened it, and the exit sweep releases a worker's handles.
|
||||
fn openFor(id: u64, owner: u32) ?*OpenNode {
|
||||
if (id >= open_nodes.len) return null;
|
||||
const o = &open_nodes[@intCast(id)];
|
||||
if (!o.used or o.owner != owner) return null;
|
||||
return o;
|
||||
}
|
||||
|
||||
const ParentLeaf = struct { parent: []const u8, leaf: []const u8 };
|
||||
|
||||
// Split a path: "/a/b" -> ("/a", "b"); "/b" -> ("/", "b"); "b" -> ("/", "b").
|
||||
fn splitParent(path: []const u8) ParentLeaf {
|
||||
const slash = std.mem.lastIndexOfScalar(u8, path, '/');
|
||||
return .{
|
||||
.parent = if (slash) |s| (if (s == 0) "/" else path[0..s]) else "/",
|
||||
.leaf = if (slash) |s| path[s + 1 ..] else path,
|
||||
};
|
||||
}
|
||||
|
||||
fn onOpen(_: void, invocation: Invocation(vfs_protocol.Open), answer: Answer(vfs_protocol.Opened)) isize {
|
||||
const fs = engine_ptr orelse return refused;
|
||||
const path = invocation.tail;
|
||||
const flags = invocation.request.flags;
|
||||
var node = fs.resolve(path);
|
||||
if (node == null and flags & vfs_protocol.create != 0) {
|
||||
const split = splitParent(path);
|
||||
const parent = fs.resolve(split.parent) orelse return refused;
|
||||
node = fs.createFile(parent, split.leaf);
|
||||
}
|
||||
var resolved = node orelse return refused;
|
||||
// O_TRUNC: replace contents rather than overwrite in place (frees the
|
||||
// old chain, so a shorter rewrite leaves no stale tail).
|
||||
if (flags & vfs_protocol.truncate != 0 and !resolved.is_directory) {
|
||||
fs.truncate(&resolved);
|
||||
}
|
||||
const index = allocOpen() orelse return refused;
|
||||
open_nodes[index] = .{ .used = true, .node = resolved, .owner = invocation.sender };
|
||||
answer.set(.{ .node = index });
|
||||
return 0;
|
||||
}
|
||||
|
||||
fn onRead(_: void, invocation: Invocation(vfs_protocol.Read), answer: Answer(void)) isize {
|
||||
const fs = engine_ptr orelse return refused;
|
||||
const o = openFor(invocation.target, invocation.sender) orelse return refused;
|
||||
const into = answer.tail();
|
||||
const want = @min(@as(usize, invocation.request.len), into.len);
|
||||
return @intCast(fs.readFile(o.node, @intCast(invocation.request.offset), into[0..want]));
|
||||
}
|
||||
|
||||
fn onWrite(_: void, invocation: Invocation(vfs_protocol.Write), answer: Answer(vfs_protocol.Written)) isize {
|
||||
const fs = engine_ptr orelse return refused;
|
||||
const o = openFor(invocation.target, invocation.sender) orelse return refused;
|
||||
const data = invocation.tail[0..@min(invocation.tail.len, invocation.request.len)];
|
||||
const n = fs.writeFile(&o.node, @intCast(invocation.request.offset), data);
|
||||
answer.set(.{ .count = @intCast(n) });
|
||||
return 0;
|
||||
}
|
||||
|
||||
fn onStatus(_: void, invocation: Invocation(void), answer: Answer(vfs_protocol.FileStatus)) isize {
|
||||
const o = openFor(invocation.target, invocation.sender) orelse return refused;
|
||||
const kind: vfs_protocol.NodeKind = if (o.node.is_directory) .directory else .regular;
|
||||
answer.set(.{ .size = o.node.size, .kind = @intFromEnum(kind), .mtime = o.node.mtime });
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// One entry per call. End of directory — not a directory, or a cursor
|
||||
/// past the last child — is an entry with no name.
|
||||
fn onReaddir(_: void, invocation: Invocation(vfs_protocol.Readdir), answer: Answer(vfs_protocol.DirectoryEntry)) isize {
|
||||
const fs = engine_ptr orelse return refused;
|
||||
const o = openFor(invocation.target, invocation.sender) orelse return refused;
|
||||
if (!o.node.is_directory) {
|
||||
answer.set(.{});
|
||||
return 0;
|
||||
}
|
||||
const listing = fs.listEntry(o.node, @intCast(invocation.request.cursor)) orelse {
|
||||
answer.set(.{});
|
||||
return 0;
|
||||
};
|
||||
const kind: vfs_protocol.NodeKind = if (listing.is_directory) .directory else .regular;
|
||||
const into = answer.tail();
|
||||
const name_len = @min(listing.name_len, into.len);
|
||||
@memcpy(into[0..name_len], listing.name_buffer[0..name_len]);
|
||||
answer.set(.{ .kind = @intFromEnum(kind), .name_len = @intCast(name_len), .size = listing.size });
|
||||
return @intCast(name_len);
|
||||
}
|
||||
|
||||
/// Closing is scoped like any other node operation: a client releases its
|
||||
/// own handles and nobody else's, and a foreign/free/out-of-range id is
|
||||
/// refused identically so a close cannot probe which ids are live.
|
||||
fn onClose(_: void, invocation: Invocation(void), _: Answer(void)) isize {
|
||||
const o = openFor(invocation.target, invocation.sender) orelse return refused;
|
||||
o.used = false;
|
||||
// Durable-on-close: the caller's flush commits any device write cache
|
||||
// to stable media now. This is what makes init's shutdown log flush
|
||||
// survive a real power-off, and the right default for removable media.
|
||||
volume_flush();
|
||||
return 0;
|
||||
}
|
||||
|
||||
fn onMakeDirectory(_: void, invocation: Invocation(void), _: Answer(void)) isize {
|
||||
const fs = engine_ptr orelse return refused;
|
||||
const path = invocation.tail;
|
||||
if (fs.resolve(path) != null) return refused; // already exists
|
||||
const split = splitParent(path);
|
||||
const parent = fs.resolve(split.parent) orelse return refused;
|
||||
if (fs.createDirectory(parent, split.leaf) == null) return refused;
|
||||
return 0;
|
||||
}
|
||||
|
||||
fn onUnlink(_: void, invocation: Invocation(void), _: Answer(void)) isize {
|
||||
const fs = engine_ptr orelse return refused;
|
||||
const split = splitParent(invocation.tail);
|
||||
const parent = fs.resolve(split.parent) orelse return refused;
|
||||
if (!fs.removeFile(parent, split.leaf)) return refused;
|
||||
return 0;
|
||||
}
|
||||
|
||||
fn onRename(_: void, invocation: Invocation(void), _: Answer(void)) isize {
|
||||
const fs = engine_ptr orelse return refused;
|
||||
const both = invocation.tail;
|
||||
const separator = std.mem.indexOfScalar(u8, both, 0) orelse return refused;
|
||||
const old_split = splitParent(both[0..separator]);
|
||||
const new_split = splitParent(both[separator + 1 ..]);
|
||||
if (!std.mem.eql(u8, old_split.parent, new_split.parent)) return refused; // same-directory only
|
||||
const parent = fs.resolve(old_split.parent) orelse return refused;
|
||||
if (!fs.rename(parent, old_split.leaf, new_split.leaf)) return refused;
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// The verbs this backend implements. `mount`/`unmount`/`bind` are absent
|
||||
/// on purpose — path routing is the kernel's, and only init serves `bind`.
|
||||
const handlers = Serve.Handlers{
|
||||
.open = onOpen,
|
||||
.close = onClose,
|
||||
.read = onRead,
|
||||
.write = onWrite,
|
||||
.status = onStatus,
|
||||
.readdir = onReaddir,
|
||||
.mkdir = onMakeDirectory,
|
||||
.unlink = onUnlink,
|
||||
.rename = onRename,
|
||||
};
|
||||
|
||||
/// The vfs protocol carries no capability, so `arrived` is never claimed
|
||||
/// — the harness's ownership rule then closes whatever a caller attached,
|
||||
/// so a request carrying one cannot spend a slot of this server's table.
|
||||
fn onMessage(message: []const u8, out: []u8, sender: u32, arrived: *ipc.Arrival) usize {
|
||||
_ = arrived;
|
||||
const fs = engine_ptr;
|
||||
// Storage not up yet: fail politely, whatever was asked — clients retry.
|
||||
if (!mounted or fs == null) {
|
||||
const status = envelope.Status{ .status = refused, .len = 0 };
|
||||
@memcpy(out[0..envelope.prefix_size], std.mem.asBytes(&status));
|
||||
return envelope.prefix_size;
|
||||
}
|
||||
// Stamp create/write with the current wall-clock time (mtime): cheap,
|
||||
// and it keeps the engine pure (it takes the time as data, not a call).
|
||||
fs.?.current_time_epoch = time.wallClock();
|
||||
return Serve.dispatch({}, handlers, message, sender, null, out);
|
||||
}
|
||||
|
||||
/// A process-exit event releases every open handle the dead client held,
|
||||
/// so a crashed reader cannot pin table slots.
|
||||
fn onNotification(badge: u64) void {
|
||||
const got = ipc.Received{ .len = 0, .badge = badge, .cap = null };
|
||||
if (got.isTimer()) {
|
||||
tryBringUp();
|
||||
if (!mounted) _ = time.timerOnce(service_endpoint, mount_retry_ms);
|
||||
return;
|
||||
}
|
||||
if (!got.isChildExit()) return;
|
||||
const dead = got.childProcessId();
|
||||
var released: u32 = 0;
|
||||
for (&open_nodes) |*o| {
|
||||
if (o.used and o.owner == dead) {
|
||||
o.* = .{};
|
||||
released += 1;
|
||||
}
|
||||
}
|
||||
if (released != 0) std.log.info("released {d} handle(s) for dead client {d}", .{ released, dead });
|
||||
}
|
||||
|
||||
/// One bring-up attempt: ask the caller for a mounted volume, and on
|
||||
/// success install its mounts and go live. A failure leaves everything
|
||||
/// untouched for the next tick.
|
||||
fn tryBringUp() void {
|
||||
if (mounted) return;
|
||||
const volume = callbacks.bringUp(service_endpoint) orelse return;
|
||||
engine_ptr = volume.engine;
|
||||
volume_flush = volume.flush;
|
||||
for (volume.mounts) |m| {
|
||||
const ok = if (m.rewrite.len == 0)
|
||||
file_system.mount(m.prefix, service_endpoint)
|
||||
else
|
||||
file_system.mountRewritten(m.prefix, service_endpoint, m.rewrite);
|
||||
if (ok) {
|
||||
std.log.info("mounted {s}", .{m.prefix});
|
||||
} else {
|
||||
std.log.info("could not mount {s}", .{m.prefix});
|
||||
}
|
||||
}
|
||||
mounted = true;
|
||||
}
|
||||
|
||||
fn initialise(endpoint: ipc.Handle) bool {
|
||||
service_endpoint = endpoint;
|
||||
// Sweep a dead client's open handles via the published exit events —
|
||||
// clients hold OUR node ids directly, so a crash must not pin slots.
|
||||
_ = process.subscribeExits(endpoint);
|
||||
tryBringUp();
|
||||
if (!mounted) _ = time.timerOnce(endpoint, mount_retry_ms);
|
||||
return true; // serve regardless: requests fail politely until storage mounts
|
||||
}
|
||||
|
||||
pub fn run(cb: Callbacks) void {
|
||||
callbacks = cb;
|
||||
service.run(vfs_protocol.message_maximum, .{
|
||||
.service = cb.service_name,
|
||||
.init = initialise,
|
||||
.on_message = onMessage,
|
||||
.on_notification = onNotification,
|
||||
});
|
||||
}
|
||||
};
|
||||
}
|
||||
@@ -10,10 +10,9 @@ pub fn build(b: *std.Build) void {
|
||||
.name = "fat",
|
||||
.root_source_file = b.path("fat.zig"),
|
||||
.imports = &.{
|
||||
"block", "channel", "device-manager-protocol", "driver",
|
||||
"envelope", "file-system", "ipc", "logging",
|
||||
"memory", "process", "service", "time",
|
||||
"vfs-protocol",
|
||||
"block", "channel", "device-manager-protocol",
|
||||
"driver", "envelope", "file-system-harness",
|
||||
"ipc", "logging", "memory",
|
||||
},
|
||||
});
|
||||
b.installArtifact(exe);
|
||||
|
||||
+52
-313
@@ -1,40 +1,30 @@
|
||||
//! system/services/fat — the FAT filesystem server. Spawned as a boot service, it
|
||||
//! opens the block device (a USB stick via usb-storage) under `.block`, mounts the
|
||||
//! FAT filesystem on it (the pure engine in engine.zig), and mounts itself into
|
||||
//! the VFS at /volumes/usb. From then on the VFS forwards every open/read/write/
|
||||
//! status/readdir/close under /volumes/usb to this server, which serves the same
|
||||
//! vfs-protocol as a backend — turning block reads into file reads.
|
||||
//! system/services/fat — the FAT filesystem service. This is FAT's FAT-specific
|
||||
//! half: it finds its block device, sets up the DMA bounce buffer, mounts the
|
||||
//! FAT engine on it, and hands the mounted volume to the shared filesystem
|
||||
//! harness (library/kernel/file-system-harness), which owns everything else —
|
||||
//! the vfs-protocol serving, the open-node table, mount registration, the exit
|
||||
//! sweep, durable-on-close. The engine (engine.zig) is the pure, host-testable
|
||||
//! format code; on-disk.zig its byte layout. A second filesystem reuses the
|
||||
//! harness and supplies its own engine
|
||||
//! (docs/file-system-development/storage-architecture.md).
|
||||
//!
|
||||
//! The block data path never crosses IPC: a DMA bounce buffer is handed to the
|
||||
//! block driver by physical address, and the engine copies sectors in and out of
|
||||
//! it.
|
||||
//! block driver by physical address, and the engine copies sectors in and out.
|
||||
|
||||
const std = @import("std");
|
||||
const channel = @import("channel");
|
||||
const device_manager_protocol = @import("device-manager-protocol");
|
||||
const driver = @import("driver");
|
||||
const ipc = @import("ipc");
|
||||
const process = @import("process");
|
||||
const service = @import("service");
|
||||
const time = @import("time");
|
||||
const block = @import("block");
|
||||
const file_system = @import("file-system");
|
||||
const memory = @import("memory");
|
||||
const logging = @import("logging");
|
||||
const engine = @import("engine.zig");
|
||||
const on_disk = @import("on-disk.zig");
|
||||
const envelope = @import("envelope");
|
||||
const vfs_protocol = @import("vfs-protocol");
|
||||
const harness = @import("file-system-harness");
|
||||
|
||||
/// The generated vfs dispatch, bound to this server. There is one FAT volume per
|
||||
/// process, so the handler context is empty and the state stays where it was: in
|
||||
/// this file's globals.
|
||||
const Serve = vfs_protocol.Protocol.Provider(void);
|
||||
|
||||
const Invocation = envelope.Invocation;
|
||||
const Answer = envelope.Answer;
|
||||
|
||||
const mount_point = "/volumes/usb";
|
||||
/// The serving harness, specialized for the FAT engine. One volume per process.
|
||||
const Harness = harness.Server(engine.FileSystem);
|
||||
|
||||
// The engine's BlockDevice, backed by the `.block` driver plus a DMA bounce
|
||||
// buffer the driver reads/writes by physical address.
|
||||
@@ -68,69 +58,28 @@ var ipc_block: IpcBlock = undefined;
|
||||
// file close — so writes are committed to stable media before a power-off.
|
||||
var device_dirty: bool = false;
|
||||
var filesystem: engine.FileSystem = undefined;
|
||||
|
||||
// Open handles clients hold against this backend: each maps a node id to a
|
||||
// resolved engine node, and to the client that opened it. `owner` is the
|
||||
// kernel-stamped badge of the opening task — the only source identity there is.
|
||||
const OpenNode = struct { used: bool = false, node: engine.Node = undefined, owner: u32 = 0 };
|
||||
var open_nodes = [_]OpenNode{.{}} ** 32;
|
||||
|
||||
fn allocOpen() ?usize {
|
||||
for (&open_nodes, 0..) |*o, i| {
|
||||
if (!o.used) return i;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// The open node `id` names **for `owner`** — null unless the id is in range, in
|
||||
/// use, and this client's own. Node ids are small integers drawn from a table of
|
||||
/// thirty-two, so they are trivially guessable; before this check every client
|
||||
/// honoured every other client's ids, which is the hole
|
||||
/// docs/os-development/protocol-namespace.md names ("handles must be scoped per
|
||||
/// client — validated against the badge"). Nothing else about them changed: they
|
||||
/// are still per-session, still swept when their owner dies.
|
||||
///
|
||||
/// The owner is a *task*, not a process, because the badge is: a threaded client
|
||||
/// reads and writes a node from the thread that opened it, exactly as the exit
|
||||
/// sweep already released a worker thread's handles when that thread died.
|
||||
fn openFor(id: u64, owner: u32) ?*OpenNode {
|
||||
if (id >= open_nodes.len) return null;
|
||||
const o = &open_nodes[@intCast(id)];
|
||||
if (!o.used or o.owner != owner) return null;
|
||||
return o;
|
||||
}
|
||||
|
||||
/// What a handler returns when the thing asked for is not there — a bad node id,
|
||||
/// a node that is someone else's, a path that does not resolve, a mutation the
|
||||
/// volume refused. One errno for all of them, because a filesystem's failures are
|
||||
/// all "no such thing" as far as the file API can act on them — and because
|
||||
/// *someone else's* must be indistinguishable from *nobody's*, or the refusal
|
||||
/// would itself tell a prober which ids are live (the same discipline the
|
||||
/// protocol namespace's refused open follows).
|
||||
const refused: isize = -envelope.ENOENT;
|
||||
|
||||
/// How often to look for a block device while none is mounted. Storage arriving
|
||||
/// is EVENT-shaped (the usb chain registering, possibly after a driver restart),
|
||||
/// but the registry has no subscription — a slow poll from our own harness loop
|
||||
/// keeps the service responsive (ping, terminate) while it waits, and keeps it
|
||||
/// alive to catch storage that appears LATE (a restarted usb-storage after a
|
||||
/// transient failure — the resilience half of docs/logging.md's storage story).
|
||||
const mount_retry_ms = 500;
|
||||
|
||||
var mounted = false;
|
||||
var service_endpoint: ipc.Handle = 0;
|
||||
/// The one channel to the device manager, opened on first need and kept — the
|
||||
/// poll retries on it, never spending a handle-table slot per attempt.
|
||||
var manager_handle: ?ipc.Handle = null;
|
||||
|
||||
/// The prefixes this volume installs: /volumes/usb from the volume root, plus
|
||||
/// the two hierarchy subtrees the boot volume carries (rewrite == prefix), so
|
||||
/// hierarchy paths (the logger's /system/logs) stay decoupled from which volume
|
||||
/// backs them.
|
||||
const fat_mounts = [_]harness.MountSpec{
|
||||
.{ .prefix = "/volumes/usb" },
|
||||
.{ .prefix = "/system/configuration", .rewrite = "/system/configuration" },
|
||||
.{ .prefix = "/system/logs", .rewrite = "/system/logs" },
|
||||
};
|
||||
|
||||
/// Find the volume's provider through the device manager (establishment by
|
||||
/// lineage, communication.md "Establishment: two planes" — `block` is not a
|
||||
/// registry name; one storage process serves each stick): enumerate the
|
||||
/// manager's tree, take the FIRST usb mass-storage child by enumeration order
|
||||
/// (deterministic within a boot; single-volume by construction, and choosing
|
||||
/// the BOOT volume by content when two sticks are present is the M21 remount
|
||||
/// the BOOT volume by content when two sticks are present is the volume-manager
|
||||
/// track), and consumer-hello for the channel of the driver bound to it.
|
||||
/// Null until the chain is up — the caller's poll retries.
|
||||
/// Null until the chain is up — the harness's poll retries.
|
||||
fn acquireVolume() ?block.Device {
|
||||
const manager = manager_handle orelse opened: {
|
||||
const handle = channel.openEndpoint("device-manager") orelse return null;
|
||||
@@ -170,47 +119,45 @@ fn acquireVolume() ?block.Device {
|
||||
}
|
||||
}
|
||||
|
||||
fn initialise(endpoint: ipc.Handle) bool {
|
||||
service_endpoint = endpoint;
|
||||
_ = logging.write("/system/services/fat: starting, waiting for a block device\n");
|
||||
// With the router in the kernel, clients hold OUR node ids directly; sweep
|
||||
// a dead client's open handles via the published exit events (the pattern
|
||||
// the old userspace router used for its own table).
|
||||
_ = process.subscribeExits(endpoint);
|
||||
tryBringUp();
|
||||
if (!mounted) _ = time.timerOnce(endpoint, mount_retry_ms);
|
||||
return true; // serve regardless: requests fail politely until storage mounts
|
||||
/// Durable-on-close: commit the device write cache if any block reached it since
|
||||
/// the last flush. The harness calls this on every close; the dirty check keeps
|
||||
/// it cheap. `device_dirty` lives here because `IpcBlock.writeBlocks` sets it.
|
||||
fn flushIfDirty() void {
|
||||
if (device_dirty) {
|
||||
_ = ipc_block.device.flush();
|
||||
device_dirty = false;
|
||||
}
|
||||
}
|
||||
|
||||
/// One storage bring-up attempt: block device -> FAT mount -> VFS mounts. Sets
|
||||
/// `mounted` on success; a failure leaves everything untouched for the next tick.
|
||||
fn tryBringUp() void {
|
||||
if (mounted) return;
|
||||
const device = acquireVolume() orelse return;
|
||||
/// FAT bring-up: find the block device, set up DMA, mount the engine, and hand
|
||||
/// the volume to the harness — or null to retry on the harness's timer.
|
||||
fn fatBringUp(endpoint: ipc.Handle) ?Harness.Volume {
|
||||
_ = endpoint;
|
||||
const device = acquireVolume() orelse return null;
|
||||
const geometry = device.geometry() orelse {
|
||||
_ = logging.write("/system/services/fat: block geometry unavailable\n");
|
||||
return;
|
||||
return null;
|
||||
};
|
||||
// Shareable so the buffer's capability can be attached down the chain (block server
|
||||
// -> controller), making its physical addresses reachable by the device under an
|
||||
// enforcing IOMMU. No-op binding otherwise.
|
||||
const bounce = memory.dmaAlloc(engine.max_transfer_sectors * 512, memory.dma_coherent | memory.dma_shareable) orelse return;
|
||||
// Shareable so the buffer's capability can be attached down the chain (block
|
||||
// server -> controller), making its physical addresses reachable by the
|
||||
// device under an enforcing IOMMU. No-op binding otherwise.
|
||||
const bounce = memory.dmaAlloc(engine.max_transfer_sectors * 512, memory.dma_coherent | memory.dma_shareable) orelse return null;
|
||||
if (bounce.handle) |handle| {
|
||||
// Attach, detach, and attach again: the round trip exercises BOTH verbs
|
||||
// of the DMA-window lifecycle through the whole chain (fat → storage →
|
||||
// bus → kernel) on every boot, so a broken detach fails every fat case
|
||||
// of the DMA-window lifecycle through the whole chain (fat -> storage ->
|
||||
// bus -> kernel) on every boot, so a broken detach fails every fat case
|
||||
// rather than lying dormant until the first buffer replacement.
|
||||
if (!device.attach(handle)) {
|
||||
_ = logging.write("/system/services/fat: could not attach the DMA bounce buffer\n");
|
||||
return;
|
||||
return null;
|
||||
}
|
||||
if (!device.detach(handle)) {
|
||||
_ = logging.write("/system/services/fat: could not detach the DMA bounce buffer\n");
|
||||
return;
|
||||
return null;
|
||||
}
|
||||
if (!device.attach(handle)) {
|
||||
_ = logging.write("/system/services/fat: could not re-attach the DMA bounce buffer\n");
|
||||
return;
|
||||
return null;
|
||||
}
|
||||
_ = ipc.close(handle); // the binding holds its own reference now
|
||||
}
|
||||
@@ -225,222 +172,14 @@ fn tryBringUp() void {
|
||||
};
|
||||
filesystem = engine.FileSystem.mount(block_device) orelse {
|
||||
_ = logging.write("/system/services/fat: not a FAT filesystem\n");
|
||||
return;
|
||||
return null;
|
||||
};
|
||||
std.log.info("mounted FAT ({s}, {d} clusters, partition lba {d})", .{ @tagName(filesystem.geometry.fat_type), filesystem.geometry.cluster_count, filesystem.base_lba });
|
||||
|
||||
// Mount ourselves into the kernel VFS at /volumes/usb — and serve
|
||||
// /system/configuration and /system/logs from the volume's identically-named
|
||||
// subtrees (the boot volume is hierarchy-shaped, so rewrite == prefix), so
|
||||
// hierarchy paths (the logger's /system/logs) stay decoupled from which
|
||||
// volume carries them.
|
||||
if (file_system.mount(mount_point, endpointForMount())) {
|
||||
std.log.info("mounted {s}", .{mount_point});
|
||||
} else {
|
||||
_ = logging.write("/system/services/fat: could not mount /volumes/usb\n");
|
||||
}
|
||||
if (file_system.mountRewritten("/system/configuration", endpointForMount(), "/system/configuration")) {
|
||||
std.log.info("mounted /system/configuration", .{});
|
||||
} else {
|
||||
_ = logging.write("/system/services/fat: could not mount /system/configuration\n");
|
||||
}
|
||||
if (file_system.mountRewritten("/system/logs", endpointForMount(), "/system/logs")) {
|
||||
std.log.info("mounted /system/logs", .{});
|
||||
} else {
|
||||
_ = logging.write("/system/services/fat: could not mount /system/logs\n");
|
||||
}
|
||||
mounted = true;
|
||||
}
|
||||
|
||||
fn endpointForMount() ipc.Handle {
|
||||
return service_endpoint;
|
||||
}
|
||||
|
||||
/// A subscribed process-exit event: release every open handle the dead client
|
||||
/// held, so a crashed reader can't pin table slots (or, later, locks).
|
||||
fn onNotification(badge: u64) void {
|
||||
const got = ipc.Received{ .len = 0, .badge = badge, .cap = null };
|
||||
if (got.isTimer()) {
|
||||
tryBringUp();
|
||||
if (!mounted) _ = time.timerOnce(service_endpoint, mount_retry_ms);
|
||||
return;
|
||||
}
|
||||
if (!got.isChildExit()) return;
|
||||
const dead = got.childProcessId();
|
||||
var released: u32 = 0;
|
||||
for (&open_nodes) |*o| {
|
||||
if (o.used and o.owner == dead) {
|
||||
o.* = .{};
|
||||
released += 1;
|
||||
}
|
||||
}
|
||||
if (released != 0) std.log.info("released {d} handle(s) for dead client {d}", .{ released, dead });
|
||||
}
|
||||
|
||||
const ParentLeaf = struct { parent: []const u8, leaf: []const u8 };
|
||||
|
||||
// Split a path into its parent directory and final component: "/a/b" -> ("/a",
|
||||
// "b"); "/b" -> ("/", "b"); "b" -> ("/", "b").
|
||||
fn splitParent(path: []const u8) ParentLeaf {
|
||||
const slash = std.mem.lastIndexOfScalar(u8, path, '/');
|
||||
return .{
|
||||
.parent = if (slash) |s| (if (s == 0) "/" else path[0..s]) else "/",
|
||||
.leaf = if (slash) |s| path[s + 1 ..] else path,
|
||||
};
|
||||
}
|
||||
|
||||
fn onOpen(_: void, invocation: Invocation(vfs_protocol.Open), answer: Answer(vfs_protocol.Opened)) isize {
|
||||
const path = invocation.tail;
|
||||
const flags = invocation.request.flags;
|
||||
var node = filesystem.resolve(path);
|
||||
if (node == null and flags & vfs_protocol.create != 0) {
|
||||
const split = splitParent(path);
|
||||
const parent = filesystem.resolve(split.parent) orelse return refused;
|
||||
node = filesystem.createFile(parent, split.leaf);
|
||||
}
|
||||
var resolved = node orelse return refused;
|
||||
// O_TRUNC: replace an existing file's contents rather than overwriting in place
|
||||
// (frees the old chain, so a shorter rewrite leaves no stale tail).
|
||||
if (flags & vfs_protocol.truncate != 0 and !resolved.is_directory) {
|
||||
filesystem.truncate(&resolved);
|
||||
}
|
||||
const index = allocOpen() orelse return refused;
|
||||
open_nodes[index] = .{ .used = true, .node = resolved, .owner = invocation.sender };
|
||||
answer.set(.{ .node = index });
|
||||
return 0;
|
||||
}
|
||||
|
||||
fn onRead(_: void, invocation: Invocation(vfs_protocol.Read), answer: Answer(void)) isize {
|
||||
const o = openFor(invocation.target, invocation.sender) orelse return refused;
|
||||
const into = answer.tail();
|
||||
const want = @min(@as(usize, invocation.request.len), into.len);
|
||||
return @intCast(filesystem.readFile(o.node, @intCast(invocation.request.offset), into[0..want]));
|
||||
}
|
||||
|
||||
fn onWrite(_: void, invocation: Invocation(vfs_protocol.Write), answer: Answer(vfs_protocol.Written)) isize {
|
||||
const o = openFor(invocation.target, invocation.sender) orelse return refused;
|
||||
const data = invocation.tail[0..@min(invocation.tail.len, invocation.request.len)];
|
||||
const n = filesystem.writeFile(&o.node, @intCast(invocation.request.offset), data);
|
||||
answer.set(.{ .count = @intCast(n) });
|
||||
return 0;
|
||||
}
|
||||
|
||||
fn onStatus(_: void, invocation: Invocation(void), answer: Answer(vfs_protocol.FileStatus)) isize {
|
||||
const o = openFor(invocation.target, invocation.sender) orelse return refused;
|
||||
const kind: vfs_protocol.NodeKind = if (o.node.is_directory) .directory else .regular;
|
||||
answer.set(.{ .size = o.node.size, .kind = @intFromEnum(kind), .mtime = o.node.mtime });
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// One entry per call. End of directory — a node that is not a directory, or a
|
||||
/// cursor past the last child — is an entry with no name, which is how the
|
||||
/// protocol spells it now that the reply's length always counts the fixed part.
|
||||
fn onReaddir(_: void, invocation: Invocation(vfs_protocol.Readdir), answer: Answer(vfs_protocol.DirectoryEntry)) isize {
|
||||
const o = openFor(invocation.target, invocation.sender) orelse return refused;
|
||||
if (!o.node.is_directory) {
|
||||
answer.set(.{});
|
||||
return 0;
|
||||
}
|
||||
const listing = filesystem.listEntry(o.node, @intCast(invocation.request.cursor)) orelse {
|
||||
answer.set(.{});
|
||||
return 0;
|
||||
};
|
||||
const kind: vfs_protocol.NodeKind = if (listing.is_directory) .directory else .regular;
|
||||
const into = answer.tail();
|
||||
const name_len = @min(listing.name_len, into.len);
|
||||
@memcpy(into[0..name_len], listing.name_buffer[0..name_len]);
|
||||
answer.set(.{ .kind = @intFromEnum(kind), .name_len = @intCast(name_len), .size = listing.size });
|
||||
return @intCast(name_len);
|
||||
}
|
||||
|
||||
/// Closing is an operation on a node like any other, so it is scoped like any
|
||||
/// other: a client may release its own handles and nobody else's. An id that is
|
||||
/// not the caller's — free, out of range, or another client's — is refused
|
||||
/// identically, so a close cannot be used to ask which ids are live either.
|
||||
fn onClose(_: void, invocation: Invocation(void), _: Answer(void)) isize {
|
||||
const o = openFor(invocation.target, invocation.sender) orelse return refused;
|
||||
o.used = false;
|
||||
// Durable-on-close: if any block reached the device since the last flush,
|
||||
// commit its cache to stable media now (best-effort). This is what makes
|
||||
// init's shutdown log flush survive a real power-off, and is the right
|
||||
// default for removable media the user may unplug.
|
||||
if (device_dirty) {
|
||||
_ = ipc_block.device.flush();
|
||||
device_dirty = false;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
fn onMakeDirectory(_: void, invocation: Invocation(void), _: Answer(void)) isize {
|
||||
const path = invocation.tail;
|
||||
if (filesystem.resolve(path) != null) return refused; // already exists — no duplicate entries
|
||||
const split = splitParent(path);
|
||||
const parent = filesystem.resolve(split.parent) orelse return refused;
|
||||
if (filesystem.createDirectory(parent, split.leaf) == null) return refused;
|
||||
return 0;
|
||||
}
|
||||
|
||||
fn onUnlink(_: void, invocation: Invocation(void), _: Answer(void)) isize {
|
||||
const split = splitParent(invocation.tail);
|
||||
const parent = filesystem.resolve(split.parent) orelse return refused;
|
||||
if (!filesystem.removeFile(parent, split.leaf)) return refused;
|
||||
return 0;
|
||||
}
|
||||
|
||||
fn onRename(_: void, invocation: Invocation(void), _: Answer(void)) isize {
|
||||
const both = invocation.tail;
|
||||
const separator = std.mem.indexOfScalar(u8, both, 0) orelse return refused;
|
||||
const old_split = splitParent(both[0..separator]);
|
||||
const new_split = splitParent(both[separator + 1 ..]);
|
||||
// Same-directory rename only.
|
||||
if (!std.mem.eql(u8, old_split.parent, new_split.parent)) return refused;
|
||||
const parent = filesystem.resolve(old_split.parent) orelse return refused;
|
||||
if (!filesystem.rename(parent, old_split.leaf, new_split.leaf)) return refused;
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// The verbs this backend implements. The three it leaves out — `mount`,
|
||||
/// `unmount`, `bind` — answer `-ENOSYS` from the generated dispatch, which is
|
||||
/// exactly right: path routing is the kernel's now, and only init implements
|
||||
/// `bind` (docs/os-development/protocol-namespace.md). `describe` is the
|
||||
/// envelope's own.
|
||||
const handlers = Serve.Handlers{
|
||||
.open = onOpen,
|
||||
.close = onClose,
|
||||
.read = onRead,
|
||||
.write = onWrite,
|
||||
.status = onStatus,
|
||||
.readdir = onReaddir,
|
||||
.mkdir = onMakeDirectory,
|
||||
.unlink = onUnlink,
|
||||
.rename = onRename,
|
||||
};
|
||||
|
||||
/// The vfs protocol has no operation that takes a capability, so `arrived` is
|
||||
/// never claimed here — which, under the harness's ownership rule, means the
|
||||
/// loop closes whatever a caller attached. That is the point of the rule: this
|
||||
/// callback used to discard a `?ipc.Handle` and every request carrying one — a
|
||||
/// legal thing for any client to do — spent a slot of the VFS server's
|
||||
/// thirty-two until it could accept no capability at all.
|
||||
fn onMessage(message: []const u8, out: []u8, sender: u32, arrived: *ipc.Arrival) usize {
|
||||
_ = arrived;
|
||||
// Storage not up (yet): fail politely, whatever was asked — clients retry.
|
||||
if (!mounted) {
|
||||
const status = envelope.Status{ .status = refused, .len = 0 };
|
||||
@memcpy(out[0..envelope.prefix_size], std.mem.asBytes(&status));
|
||||
return envelope.prefix_size;
|
||||
}
|
||||
// Stamp create/write with the current wall-clock time (mtime). Cheap, and it
|
||||
// keeps the engine pure (it takes the time as data, not a syscall).
|
||||
filesystem.current_time_epoch = time.wallClock();
|
||||
return Serve.dispatch({}, handlers, message, sender, null, out);
|
||||
return .{ .engine = &filesystem, .mounts = &fat_mounts, .flush = flushIfDirty };
|
||||
}
|
||||
|
||||
pub fn main() void {
|
||||
service.run(vfs_protocol.message_maximum, .{
|
||||
.service = "vfs",
|
||||
.init = initialise,
|
||||
.on_message = onMessage,
|
||||
.on_notification = onNotification,
|
||||
});
|
||||
_ = logging.write("/system/services/fat: starting, waiting for a block device\n");
|
||||
Harness.run(.{ .bringUp = fatBringUp });
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user