From e186858315fc13703c5fcf25d05c9708be4369ca Mon Sep 17 00:00:00 2001 From: Daniel Samson <12231216+daniel-samson@users.noreply.github.com> Date: Tue, 21 Jul 2026 16:27:06 +0100 Subject: [PATCH] =?UTF-8?q?vfs:=20the=20root=20moves=20into=20the=20kernel?= =?UTF-8?q?=20=E2=80=94=20resolve=20+=20redirect=20cutover?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit runtime.fs now routes every path through fs_resolve: kernel-served /system nodes are read via fs_node (tokens, no open state); everything under a userspace mount goes straight to the owning backend's endpoint with the kernel-rewritten mount-relative path — one syscall of naming, then the unchanged vfs-protocol rendezvous, public API untouched. mkdir/ unlink/rename resolve-then-forward (rename checks both paths land on the SAME backend); mount is the fs_mount syscall. The fat server mounts twice — /mnt/usb from the volume root and /var from its /var subtree — so the logger now writes the FHS path /var/log//... and swapping the persistent medium later touches only fat's two mount calls. With clients holding fat's node ids directly, fat records each handle's owner, checks it, and sweeps a dead client's handles via the published exit events (the old router's pattern, now where the state actually lives). The userspace vfs server and its router die; ServiceId.vfs=1 stays reserved-retired; protocol.zig moves to system/vfs-protocol.zig (the wire contract is backend-only now). vfs-test becomes the ring-3 proof of the kernel VFS (own-binary ELF magic through /system, read-only refusals, listing); vfs-client-death becomes the fat sweep test over the full storage chain, with a ring-scanning check (the last-write buffer is too racy under a chattering tree). --- build.zig | 9 +- library/runtime/fs.zig | 159 ++++--- library/runtime/system.zig | 70 ++++ system/abi.zig | 2 +- system/kernel/process.zig | 13 +- system/kernel/tests.zig | 65 ++- system/services/fat/fat.zig | 54 ++- system/services/init/init.zig | 1 - system/services/logger/logger.zig | 14 +- system/services/vfs-test/vfs-test.zig | 89 ++++ system/services/vfs/path.zig | 39 -- system/services/vfs/vfs-test.zig | 62 --- system/services/vfs/vfs.zig | 389 ------------------ .../vfs/protocol.zig => vfs-protocol.zig} | 0 test/qemu_test.py | 3 +- 15 files changed, 379 insertions(+), 590 deletions(-) create mode 100644 system/services/vfs-test/vfs-test.zig delete mode 100644 system/services/vfs/path.zig delete mode 100644 system/services/vfs/vfs-test.zig delete mode 100644 system/services/vfs/vfs.zig rename system/{services/vfs/protocol.zig => vfs-protocol.zig} (100%) diff --git a/build.zig b/build.zig index a5343de..0e312c1 100644 --- a/build.zig +++ b/build.zig @@ -364,7 +364,7 @@ pub fn build(b: *std.Build) void { // is the first "protocol module" (see docs/driver-model.md); usb/block will // expose theirs the same way. const vfs_protocol_module = b.addModule("vfs-protocol", .{ - .root_source_file = b.path("system/services/vfs/protocol.zig"), + .root_source_file = b.path("system/vfs-protocol.zig"), }); // The input wire protocol: the input service's public interface, exposed as its own @@ -511,8 +511,7 @@ pub fn build(b: *std.Build) void { // Each is built by the same user-binary recipe and laid out at its FHS path on // the boot volume (see `bundled` below). The EFI loader walks the tree at boot // and hands the kernel an in-RAM initial_ramdisk of it (system/initial-ramdisk.zig). - const vfs_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "vfs", "system/services/vfs/vfs.zig"); - const vfstest_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "vfs-test", "system/services/vfs/vfs-test.zig"); + const vfstest_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "vfs-test", "system/services/vfs-test/vfs-test.zig"); const ps2_bus_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-bus", "system/drivers/ps2-bus/ps2-bus.zig"); const ps2_keyboard_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-keyboard", "system/drivers/ps2-bus/keyboard.zig"); const ps2_mouse_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-mouse", "system/drivers/ps2-bus/mouse.zig"); @@ -594,7 +593,6 @@ pub fn build(b: *std.Build) void { // are these paths with a leading slash. const bundled = [_]BundledBinary{ .{ .path = "system/services/init", .binary = init_exe.getEmittedBin() }, - .{ .path = "system/services/vfs", .binary = vfs_exe.getEmittedBin() }, .{ .path = "system/services/fat", .binary = fat_exe.getEmittedBin() }, .{ .path = "system/services/display", .binary = display_exe.getEmittedBin() }, .{ .path = "system/services/display-demo", .binary = display_demo_exe.getEmittedBin() }, @@ -888,8 +886,7 @@ pub fn build(b: *std.Build) void { "system/drivers/usb-hid/hid-report.zig", // HID boot-report keyboard/mouse decode "system/drivers/usb-storage/bulk-only-transport.zig", // CBW/CSW wrapper sizes "system/drivers/usb-storage/scsi.zig", // SCSI CDB encodings (big-endian) - "system/services/vfs/path.zig", // mount-prefix path matching - "system/services/vfs/protocol.zig", // NodeKind / DirectoryEntry sizes + op values + "system/vfs-protocol.zig", // NodeKind / DirectoryEntry sizes + op values "system/services/fat/on-disk.zig", // FAT on-disk struct sizes + type detection "system/services/fat/engine.zig", // FAT read/write over a RAM-backed image "system/services/display/compositor.zig", // Rect math + fill/composite/blit-tile diff --git a/library/runtime/fs.zig b/library/runtime/fs.zig index 84dd8f1..267bf4c 100644 --- a/library/runtime/fs.zig +++ b/library/runtime/fs.zig @@ -1,5 +1,5 @@ //! runtime.fs — the danos-native file API. A program opens, reads, writes, and -//! lists files served by the user-space VFS (system/services/vfs), each call +//! lists files through the kernel VFS root (resolve + redirect), each call //! marshalling a vfs-protocol request over IPC. This is the danos-native layer //! danos programs use directly; it is also where the file operations that later //! become `std.os.danos` are staged (see docs/zig-self-hosting.md). It replaces @@ -12,6 +12,7 @@ const std = @import("std"); const ipc = @import("ipc.zig"); +const system = @import("system.zig"); const protocol = @import("vfs-protocol"); /// The kind of a filesystem node — re-exported so a caller need not import the @@ -60,23 +61,37 @@ pub const OpenOptions = struct { } }; -// The VFS server endpoint, looked up once by well-known id and cached. -var vfs_handle: ipc.Handle = 0; -var vfs_resolved = false; -fn vfs() ?ipc.Handle { - if (!vfs_resolved) { - vfs_handle = ipc.lookup(.vfs) orelse return null; - vfs_resolved = true; +// The route to a path: the kernel resolves (fs_resolve) and either serves the +// node itself (the initrd at /system — a permanent token) or redirects us to +// the owning filesystem backend's endpoint, to which we speak the vfs-protocol +// rendezvous directly with the rewritten mount-relative path. +const Route = union(enum) { + kernel: u64, + backend: struct { handle: ipc.Handle, path: [224]u8, path_len: usize }, + + fn backendPath(self: *const Route) []const u8 { + return self.backend.path[0..self.backend.path_len]; + } +}; + +fn resolve(path: []const u8, flags: usize) ?Route { + var out: [224]u8 = undefined; + const route = system.fsResolve(path, flags, &out) orelse return null; + switch (route) { + .kernel => |token| return .{ .kernel = token }, + .backend => |b| { + var r: Route = .{ .backend = .{ .handle = b.handle, .path = undefined, .path_len = b.path_len } }; + @memcpy(r.backend.path[0..b.path_len], out[0..b.path_len]); + return r; + }, } - return vfs_handle; } const Result = struct { reply: protocol.Reply, payload: []u8 }; -// One request/reply round trip: [Request header][send payload] -> VFS -> +// One request/reply round trip: [Request header][send payload] -> backend -> // [Reply header][receive payload]. The receive payload lands in `out`. -fn transact(request: protocol.Request, send: []const u8, out: []u8) ?Result { - const h = vfs() orelse return null; +fn transact(h: ipc.Handle, request: protocol.Request, send: []const u8, out: []u8) ?Result { var message: [protocol.message_maximum]u8 = undefined; @memcpy(message[0..protocol.request_size], std.mem.asBytes(&request)); const slen = @min(send.len, protocol.maximum_payload); @@ -95,13 +110,21 @@ fn transact(request: protocol.Request, send: []const u8, out: []u8) ?Result { pub const File = struct { node: u64, offset: u64 = 0, + /// The owning backend's endpoint, or null for a kernel-served node (the + /// read-only /system tree), whose `node` is a permanent fs_node token. + backend: ?ipc.Handle = null, /// Read up to `buffer.len` bytes at the current offset; returns the count, or /// null on error. pub fn read(self: *File, buffer: []u8) ?usize { + const h = self.backend orelse { + const n = system.fsNodeRead(self.node, self.offset, buffer) orelse return null; + self.offset += n; + return n; + }; const want: u32 = @intCast(@min(buffer.len, protocol.maximum_payload)); const request = protocol.Request{ .operation = .read, .node = self.node, .offset = self.offset, .len = want, .flags = 0 }; - const r = transact(request, &.{}, buffer) orelse return null; + const r = transact(h, request, &.{}, buffer) orelse return null; if (r.reply.status != 0) return null; self.offset += r.reply.len; return r.reply.len; @@ -109,11 +132,13 @@ pub const File = struct { /// Write `data` at the current offset; returns the count written. A single /// call is capped at the VFS payload size, so the return may be short — use - /// `writeAll` to write the whole slice. Null on error. + /// `writeAll` to write the whole slice. Null on error (kernel-served nodes + /// are read-only). pub fn write(self: *File, data: []const u8) ?usize { + const h = self.backend orelse return null; const want: u32 = @intCast(@min(data.len, protocol.maximum_payload)); const request = protocol.Request{ .operation = .write, .node = self.node, .offset = self.offset, .len = want, .flags = 0 }; - const r = transact(request, data[0..want], &.{}) orelse return null; + const r = transact(h, request, data[0..want], &.{}) orelse return null; if (r.reply.status != 0) return null; self.offset += r.reply.len; return r.reply.len; @@ -138,27 +163,40 @@ pub const File = struct { /// This file's metadata. pub fn attributes(self: *File) ?Attributes { + const h = self.backend orelse { + const a = system.fsNodeStatus(self.node) orelse return null; + return .{ .size = a.size, .kind = if (a.kind == system.file_kind_directory) .directory else .regular, .mtime = a.mtime }; + }; const request = protocol.Request{ .operation = .status, .node = self.node, .offset = 0, .len = 0, .flags = 0 }; var buffer: [@sizeOf(protocol.FileStatus)]u8 = undefined; - const r = transact(request, &.{}, &buffer) orelse return null; + const r = transact(h, request, &.{}, &buffer) orelse return null; if (r.reply.status != 0 or r.payload.len < @sizeOf(protocol.FileStatus)) return null; const status = std.mem.bytesToValue(protocol.FileStatus, buffer[0..@sizeOf(protocol.FileStatus)]); return .{ .size = status.size, .kind = kindFromWire(status.kind), .mtime = status.mtime }; } - /// Release the VFS's open handle for this file. + /// Release the backend's open handle for this file. Kernel-served node + /// tokens are permanent — nothing to release. pub fn close(self: *File) void { + const h = self.backend orelse return; const request = protocol.Request{ .operation = .close, .node = self.node, .offset = 0, .len = 0, .flags = 0 }; - _ = transact(request, &.{}, &.{}); + _ = transact(h, request, &.{}, &.{}); } }; /// Open (or create, with `.create`) `path`. Returns the open file, or null. pub fn open(path: []const u8, options: OpenOptions) ?File { - const request = protocol.Request{ .operation = .open, .node = 0, .offset = 0, .len = @intCast(path.len), .flags = options.wireFlags() }; - const r = transact(request, path, &.{}) orelse return null; - if (r.reply.status != 0) return null; - return .{ .node = r.reply.node }; + const route = resolve(path, options.wireFlags()) orelse return null; + switch (route) { + .kernel => |token| return .{ .node = token, .backend = null }, + .backend => |b| { + const relative = route.backendPath(); + const request = protocol.Request{ .operation = .open, .node = 0, .offset = 0, .len = @intCast(relative.len), .flags = options.wireFlags() }; + const r = transact(b.handle, request, relative, &.{}) orelse return null; + if (r.reply.status != 0) return null; + return .{ .node = r.reply.node, .backend = b.handle }; + }, + } } /// A path's metadata without keeping it open (open -> status -> close). @@ -190,13 +228,27 @@ pub const Entry = struct { pub const Directory = struct { node: u64, cursor: u64 = 0, + backend: ?ipc.Handle = null, /// Fill `entry` with the next directory entry; false at end of directory or /// on error. pub fn next(self: *Directory, entry: *Entry) bool { + const h = self.backend orelse { + var buffer: [@sizeOf(system.DirectoryEntryHeader) + 64]u8 = undefined; + const n = system.fsNodeReaddir(self.node, self.cursor, &buffer) orelse return false; + if (n < @sizeOf(system.DirectoryEntryHeader)) return false; // end + const header = std.mem.bytesToValue(system.DirectoryEntryHeader, buffer[0..@sizeOf(system.DirectoryEntryHeader)]); + entry.kind = if (header.kind == system.file_kind_directory) .directory else .regular; + entry.size = header.size; + const nlen = @min(@as(usize, header.name_len), entry.name_buffer.len); + @memcpy(entry.name_buffer[0..nlen], buffer[@sizeOf(system.DirectoryEntryHeader)..][0..nlen]); + entry.name_len = nlen; + self.cursor += 1; + return true; + }; const request = protocol.Request{ .operation = .readdir, .node = self.node, .offset = self.cursor, .len = 0, .flags = 0 }; var buffer: [protocol.message_maximum]u8 = undefined; - const r = transact(request, &.{}, &buffer) orelse return false; + const r = transact(h, request, &.{}, &buffer) orelse return false; if (r.reply.status != 0 or r.reply.len == 0) return false; // error or EOF if (r.payload.len < protocol.directory_entry_size) return false; const header = std.mem.bytesToValue(protocol.DirectoryEntry, r.payload[0..protocol.directory_entry_size]); @@ -210,9 +262,9 @@ pub const Directory = struct { return true; } - /// Release the VFS's open handle for this directory. + /// Release the backend's open handle for this directory. pub fn close(self: *Directory) void { - var f = File{ .node = self.node }; + var f = File{ .node = self.node, .backend = self.backend }; f.close(); } }; @@ -220,13 +272,18 @@ pub const Directory = struct { /// Open `path` as a directory for listing. Returns null if it isn't one / on error. pub fn openDirectory(path: []const u8) ?Directory { const file = open(path, .{ .directory = true }) orelse return null; - return .{ .node = file.node }; + return .{ .node = file.node, .backend = file.backend }; } -// A path-based request that returns only a status (mkdir, unlink). +// A path-based request that returns only a status (mkdir, unlink). Kernel-served +// paths (the read-only /system) refuse mutation by construction: the resolve +// must land on a backend. fn pathOperation(operation: protocol.Operation, path: []const u8) bool { - const request = protocol.Request{ .operation = operation, .node = 0, .offset = 0, .len = @intCast(path.len), .flags = 0 }; - const r = transact(request, path, &.{}) orelse return false; + const route = resolve(path, 0) orelse return false; + if (route != .backend) return false; + const relative = route.backendPath(); + const request = protocol.Request{ .operation = operation, .node = 0, .offset = 0, .len = @intCast(relative.len), .flags = 0 }; + const r = transact(route.backend.handle, request, relative, &.{}) orelse return false; return r.reply.status == 0; } @@ -261,33 +318,37 @@ pub fn remove(path: []const u8) bool { return pathOperation(.unlink, path); } -/// Rename `old_path` to `new_path`. Both must be in the same directory (same- -/// directory, 8.3-name rename only for now). Returns true on success. +/// Rename `old_path` to `new_path`. Both must resolve to the SAME filesystem +/// backend (same-directory, 8.3-name rename only for now). Returns true on +/// success. pub fn rename(old_path: []const u8, new_path: []const u8) bool { - const total = old_path.len + 1 + new_path.len; + const old_route = resolve(old_path, 0) orelse return false; + const new_route = resolve(new_path, 0) orelse return false; + if (old_route != .backend or new_route != .backend) return false; + if (old_route.backend.handle != new_route.backend.handle) return false; // cross-filesystem + const old_relative = old_route.backendPath(); + const new_relative = new_route.backendPath(); + const total = old_relative.len + 1 + new_relative.len; if (total > protocol.maximum_payload) return false; var payload: [protocol.maximum_payload]u8 = undefined; - @memcpy(payload[0..old_path.len], old_path); - payload[old_path.len] = 0; - @memcpy(payload[old_path.len + 1 ..][0..new_path.len], new_path); + @memcpy(payload[0..old_relative.len], old_relative); + payload[old_relative.len] = 0; + @memcpy(payload[old_relative.len + 1 ..][0..new_relative.len], new_relative); const request = protocol.Request{ .operation = .rename, .node = 0, .offset = 0, .len = @intCast(total), .flags = 0 }; - const r = transact(request, payload[0..total], &.{}) orelse return false; + const r = transact(old_route.backend.handle, request, payload[0..total], &.{}) orelse return false; return r.reply.status == 0; } /// Mount a filesystem backend (its server endpoint) at absolute path `target`; -/// the VFS then routes everything under `target` to that backend. This is the one -/// call that hands the VFS a capability (the backend endpoint). Returns true on -/// success. +/// the kernel VFS then routes everything under `target` to that backend. +/// Possession of the endpoint handle is the capability. Returns true on success. pub fn mount(target: []const u8, backend: ipc.Handle) bool { - const h = vfs() orelse return false; - const request = protocol.Request{ .operation = .mount, .node = 0, .offset = 0, .len = @intCast(target.len), .flags = 0 }; - var message: [protocol.message_maximum]u8 = undefined; - @memcpy(message[0..protocol.request_size], std.mem.asBytes(&request)); - const tlen = @min(target.len, protocol.maximum_payload); - @memcpy(message[protocol.request_size..][0..tlen], target[0..tlen]); - var rbuf: [protocol.message_maximum]u8 = undefined; - const result = ipc.callCap(h, message[0 .. protocol.request_size + tlen], &rbuf, backend) catch return false; - if (result.len < protocol.reply_size) return false; - return std.mem.bytesToValue(protocol.Reply, rbuf[0..protocol.reply_size]).status == 0; + return system.fsMount(target, backend, ""); +} + +/// As `mount`, with a backend-side rewrite prefix: a path under `target` reaches +/// the backend as `rewrite` + the mount-relative tail. How one volume serves two +/// mounts ("/mnt/usb" from its root, "/var" from its /var subtree). +pub fn mountRewritten(target: []const u8, backend: ipc.Handle, rewrite: []const u8) bool { + return system.fsMount(target, backend, rewrite); } diff --git a/library/runtime/system.zig b/library/runtime/system.zig index 71b85ec..ffaf7fb 100644 --- a/library/runtime/system.zig +++ b/library/runtime/system.zig @@ -32,6 +32,10 @@ pub const klog_record_magic = abi.klog_record_magic; pub const klog_flag_truncated = abi.klog_flag_truncated; pub const klog_maximum_message = abi.klog_maximum_message; pub const maximum_process_name = abi.maximum_process_name; +pub const FileAttributes = abi.FileAttributes; +pub const DirectoryEntryHeader = abi.DirectoryEntryHeader; +pub const file_kind_regular = abi.file_kind_regular; +pub const file_kind_directory = abi.file_kind_directory; /// Write raw bytes to the kernel log (bring-up/panic diagnostics; ordinary /// output goes through std.log -> writeRecord). The kernel stamps the record @@ -100,6 +104,72 @@ pub fn klogStatus() ?KlogStatus { return status; } +/// Where fs_resolve routed a path: served by the kernel (a permanent node +/// token for fs_node) or by a userspace filesystem backend (an endpoint handle +/// plus the rewritten mount-relative path, returned in the caller's buffer). +pub const FsRoute = union(enum) { + kernel: u64, + backend: struct { handle: usize, path_len: usize }, +}; + +/// Route `path` through the kernel VFS. For a backend route the rewritten +/// mount-relative path lands in `out` (behind a kernel-written length prefix, +/// already stripped here: out[0..path_len] is the path). +pub fn fsResolve(path: []const u8, flags: usize, out: []u8) ?FsRoute { + var rax: usize = undefined; + var rdx: usize = flags; // in: flags (arg #3); out: node token / backend handle + asm volatile ("syscall" + : [rax] "={rax}" (rax), + [rdx] "+{rdx}" (rdx), + : [n] "{rax}" (@intFromEnum(abi.SystemCall.fs_resolve)), + [a0] "{rdi}" (@intFromPtr(path.ptr)), + [a1] "{rsi}" (path.len), + [a3] "{r10}" (@intFromPtr(out.ptr)), + [a4] "{r8}" (out.len), + : .{ .rcx = true, .r11 = true, .memory = true }); + if (@as(isize, @bitCast(rax)) < 0) return null; + if (rax == abi.fs_route_kernel) return .{ .kernel = rdx }; + if (rax != abi.fs_route_backend) return null; + const path_len = @as(usize, out[0]) | (@as(usize, out[1]) << 8); + if (path_len + 2 > out.len) return null; + std.mem.copyForwards(u8, out[0..path_len], out[2..][0..path_len]); + return .{ .backend = .{ .handle = rdx, .path_len = path_len } }; +} + +/// Read `out.len` bytes of a kernel-served node at `offset` (fs_node read). +pub fn fsNodeRead(node_token: u64, offset: u64, out: []u8) ?usize { + const r = sc.systemCall5(.fs_node, abi.fs_node_read, node_token, offset, @intFromPtr(out.ptr), out.len); + if (@as(isize, @bitCast(r)) < 0) return null; + return r; +} + +/// A kernel-served node's metadata (fs_node status). +pub fn fsNodeStatus(node_token: u64) ?abi.FileAttributes { + var attributes: abi.FileAttributes = undefined; + const r = sc.systemCall5(.fs_node, abi.fs_node_status, node_token, 0, @intFromPtr(&attributes), @sizeOf(abi.FileAttributes)); + if (@as(isize, @bitCast(r)) < 0) return null; + return attributes; +} + +/// The `cursor`th child of a kernel-served directory (fs_node readdir): fills +/// `out` with [DirectoryEntryHeader][name]; returns total bytes (0 = end). +pub fn fsNodeReaddir(node_token: u64, cursor: u64, out: []u8) ?usize { + const r = sc.systemCall5(.fs_node, abi.fs_node_readdir, node_token, cursor, @intFromPtr(out.ptr), out.len); + if (@as(isize, @bitCast(r)) < 0) return null; + return r; +} + +/// Mount a userspace filesystem's endpoint at `prefix`, with an optional +/// backend-side `rewrite` prefix ("" = none). Possession of the endpoint +/// handle is the capability. +pub fn fsMount(prefix: []const u8, backend: usize, rewrite: []const u8) bool { + return sc.systemCall5(.fs_mount, @intFromPtr(prefix.ptr), prefix.len, backend, @intFromPtr(rewrite.ptr), rewrite.len) == 0; +} + +pub fn fsUnmount(prefix: []const u8) bool { + return sc.systemCall2(.fs_unmount, @intFromPtr(prefix.ptr), prefix.len) == 0; +} + /// End the process. Never returns. pub fn exit(code: usize) noreturn { _ = sc.systemCall1(.exit, code); diff --git a/system/abi.zig b/system/abi.zig index f510004..3da6204 100644 --- a/system/abi.zig +++ b/system/abi.zig @@ -282,7 +282,7 @@ pub const KlogStatus = extern struct { /// ipc_register/ipc_lookup). Small integers, so no string interning is needed /// during bring-up. The VFS server registers under `vfs`; clients look it up. pub const ServiceId = enum(u32) { - vfs = 1, + vfs = 1, // RETIRED: the router moved into the kernel (fs_resolve); the slot stays reserved input = 2, ps2_bus = 3, // the 8042 owner; child device drivers attach here for raw bytes device_manager = 4, // the tree, the matcher, the supervisor (docs/device-manager.md) diff --git a/system/kernel/process.zig b/system/kernel/process.zig index 8bc5ff4..6972143 100644 --- a/system/kernel/process.zig +++ b/system/kernel/process.zig @@ -1309,8 +1309,7 @@ fn systemKlogStatus(state: *architecture.CpuState) void { /// through the kernel mount table (docs/vfs-protocol.md). Kernel-served -> /// rax=fs_route_kernel, rdx=node token. Backend-served -> rax=fs_route_backend, /// rdx=an endpoint handle in the caller's table (deduplicated), and the -/// rewritten mount-relative path copied to `out` with its length in the third -/// result register. Fails for unknown paths, create-intent on /system, or an +/// rewritten mount-relative path copied into `out` behind a u16 length prefix. Fails for unknown paths, create-intent on /system, or an /// undersized out buffer. fn systemFsResolve(state: *architecture.CpuState) void { const path_ptr = architecture.systemCallArg(state, 0); @@ -1331,14 +1330,18 @@ fn systemFsResolve(state: *architecture.CpuState) void { architecture.setSystemCallResult2(state, node_token); }, .backend => |*backend| { - if (backend.path_len > out_cap) return fail(state); + // The rewritten path goes back in the out buffer behind a u16 + // length prefix (a third result register would collide with r8's + // argument role in the userspace stub). + if (backend.path_len + 2 > out_cap) return fail(state); const handle = ipc.installHandleDeduped(t, backend.endpoint); if (handle < 0) return fail(state); const destination: [*]u8 = @ptrFromInt(out_ptr); - @memcpy(destination[0..backend.path_len], backend.path[0..backend.path_len]); + destination[0] = @intCast(backend.path_len & 0xFF); + destination[1] = @intCast(backend.path_len >> 8); + @memcpy(destination[2..][0..backend.path_len], backend.path[0..backend.path_len]); architecture.setSystemCallResult(state, abi.fs_route_backend); architecture.setSystemCallResult2(state, @intCast(handle)); - architecture.setSystemCallResult3(state, backend.path_len); }, .not_found => fail(state), } diff --git a/system/kernel/tests.zig b/system/kernel/tests.zig index 5ea3de9..52fc1b7 100644 --- a/system/kernel/tests.zig +++ b/system/kernel/tests.zig @@ -265,6 +265,31 @@ fn bufferHas(needle: []const u8) bool { return std.mem.indexOf(u8, process.write_buffer[0..process.write_len], needle) != null; } +/// Substring search across the whole retained log ring (record payloads are +/// contiguous in the stream, so a one-line needle always matches if present). +/// Unlike `bufferHas` (the single LAST write), this survives busy-tree chatter. +fn ringHas(needle: []const u8) bool { + var chunk: [1024]u8 = undefined; + var overlap: [128]u8 = undefined; + var overlap_len: usize = 0; + var offset = kernel_log.status().tail; + while (true) { + const n = kernel_log.readAt(offset, &chunk) orelse return false; + if (n == 0) return false; + offset += n; + // Search the previous tail glued to this chunk, then the chunk itself. + if (overlap_len != 0) { + var glued: [1152]u8 = undefined; + @memcpy(glued[0..overlap_len], overlap[0..overlap_len]); + const m = @min(n, glued.len - overlap_len); + @memcpy(glued[overlap_len..][0..m], chunk[0..m]); + if (std.mem.indexOf(u8, glued[0 .. overlap_len + m], needle) != null) return true; + } else if (std.mem.indexOf(u8, chunk[0..n], needle) != null) return true; + overlap_len = @min(n, @min(overlap.len, needle.len)); + @memcpy(overlap[0..overlap_len], chunk[n - overlap_len ..][0..overlap_len]); + } +} + /// How many `pci_device` functions the devices broker currently holds. A durable /// snapshot, unlike a `bufferHas` poll of the single-latest write_buffer line, so a /// test can wait on it without racing transient log output. `scratch` is @@ -2120,11 +2145,11 @@ fn claimReleaseTest(boot_information: *const BootInformation) void { result(); } -/// M17.3: the published exit events, proven by their first subscriber. The VFS -/// subscribes at startup; a client opens a file and parks holding the handle; -/// the kill posts the exit event to the VFS's endpoint; the VFS releases the -/// dead client's handle and says so — the service-side mirror of iron rule 1 -/// (a service must never depend on clients cleaning up after themselves). +/// M17.3: the published exit events, proven by a stateful server. The fat +/// server subscribes at startup; a client opens a file on the volume and parks +/// holding the handle; the kill posts the exit event to fat's endpoint; fat +/// releases the dead client's handle and says so — the service-side mirror of +/// iron rule 1 (a service must never depend on clients cleaning up). fn vfsClientDeathTest(boot_information: *const BootInformation) void { log("DANOS-TEST-BEGIN: vfs-client-death\n", .{}); if (boot_information.initial_ramdisk_len == 0) { @@ -2140,7 +2165,11 @@ fn vfsClientDeathTest(boot_information: *const BootInformation) void { }; process.write_count = 0; - check("vfs spawned", spawnNamed(rd, "vfs")); + // The full tree: the storage chain must come up for /mnt/usb to exist — + // the fat server (not a router) now owns client file state and its sweep. + process.setInitialRamdisk(image); + const init_ok = if (process.spawnBundled("/system/services/init")) true else |_| false; + check("init spawned (boots the storage chain)", init_ok); const me = scheduler.currentId(); const endpoint = ipcsync.createIpcEndpoint() orelse { @@ -2158,16 +2187,18 @@ fn vfsClientDeathTest(boot_information: *const BootInformation) void { } check("parked client spawned (supervised)", client != 0); - // Its heartbeat is the fence: once it beats, the handle is open. + // Its heartbeat is the fence: once it beats, the handle is open. The park + // waits out the whole USB->block->fat chain, so give it room; with the full + // tree chattering, the ring (not the last-write buffer) is the evidence. const parked = "vfstest: parked"; scheduler.setPriority(1); - var deadline = architecture.millis() + 10000; + var deadline = architecture.millis() + 30000; while (architecture.millis() < deadline) { - if (bufferHas(parked)) break; + if (ringHas(parked)) break; scheduler.yield(); } scheduler.setPriority(4); - check("client parked holding an open handle", bufferHas(parked)); + check("client parked holding an open handle", ringHas(parked)); check("the kill is accepted", process.killProcess(me, client) == 0); var badge: u64 = 0; @@ -2175,16 +2206,17 @@ fn vfsClientDeathTest(boot_information: *const BootInformation) void { _ = ipcsync.replyWait(endpoint, 0, 0, 0, 0, abi.no_cap, &badge, &received_cap); check("the exit notification arrived", badge == abi.notify_badge_bit | abi.notify_exit_bit | client); - // The VFS heard the same published event; its release line is the proof. + // The fat server heard the same published event; its release line in the + // ring is the proof. const released = "released 1 handle(s) for dead client"; scheduler.setPriority(1); deadline = architecture.millis() + 10000; while (architecture.millis() < deadline) { - if (bufferHas(released)) break; + if (ringHas(released)) break; scheduler.yield(); } scheduler.setPriority(4); - check("the VFS released the dead client's handle", bufferHas(released)); + check("the fat server released the dead client's handle", ringHas(released)); result(); } @@ -2683,9 +2715,10 @@ fn vfsTest(boot_information: *const BootInformation) void { process.write_count = 0; process.write_from_user = false; - // Spawn just the server and its client (other initial_ramdisk binaries would write to - // the shared evidence buffer and confuse the marker check). - _ = spawnNamed(rd, "vfs"); + // Seed the kernel VFS (/system) — the router the client exercises. + process.setInitialRamdisk(image); + // Spawn just the client: the kernel itself is the VFS root it exercises + // (resolve + fs_node over /system through the plain runtime.fs API). _ = spawnNamed(rd, "vfs-test"); // Wait for the client's success heartbeat (it round-trips, then beats ~1/s). diff --git a/system/services/fat/fat.zig b/system/services/fat/fat.zig index 3949302..fcd53eb 100644 --- a/system/services/fat/fat.zig +++ b/system/services/fat/fat.zig @@ -101,18 +101,42 @@ fn initialise(endpoint: runtime.ipc.Handle) bool { }; std.log.info("mounted FAT ({s}, {d} clusters, partition lba {d})", .{ @tagName(filesystem.geometry.fat_type), filesystem.geometry.cluster_count, filesystem.base_lba }); - // Mount ourselves into the VFS namespace at /mnt/usb (retry while the VFS - // comes up). From here the VFS routes /mnt/usb/... to this server. - var tries: u32 = 0; - while (tries < 100) : (tries += 1) { - if (runtime.fs.mount(mount_point, endpoint)) { - std.log.info("mounted {s}", .{mount_point}); - return true; - } - runtime.system.sleep(50); + // With the router in the kernel, clients hold OUR node ids directly; sweep + // a dead client's open handles via the published exit events (the pattern + // the old userspace router used for its own table). + _ = runtime.process.subscribeExits(endpoint); + + // Mount ourselves into the kernel VFS at /mnt/usb — and serve /var from the + // volume's /var subtree, so FHS paths (the logger's /var/log) stay decoupled + // from which volume carries them. A mount is one syscall now; no retry + // needed (the kernel's table exists before any service). + if (runtime.fs.mount(mount_point, endpoint)) { + std.log.info("mounted {s}", .{mount_point}); + } else { + _ = runtime.system.write("/system/services/fat: could not mount /mnt/usb\n"); } - _ = runtime.system.write("/system/services/fat: could not mount into the VFS\n"); - return true; // still serve directly, even if the namespace mount didn't take + if (runtime.fs.mountRewritten("/var", endpoint, "/var")) { + std.log.info("mounted /var", .{}); + } else { + _ = runtime.system.write("/system/services/fat: could not mount /var\n"); + } + return true; +} + +/// A subscribed process-exit event: release every open handle the dead client +/// held, so a crashed reader can't pin table slots (or, later, locks). +fn onNotification(badge: u64) void { + const got = runtime.ipc.Received{ .len = 0, .badge = badge, .cap = null }; + if (!got.isChildExit()) return; + const dead = got.childProcessId(); + var released: u32 = 0; + for (&open_nodes) |*o| { + if (o.used and o.owner == dead) { + o.* = .{}; + released += 1; + } + } + if (released != 0) std.log.info("released {d} handle(s) for dead client {d}", .{ released, dead }); } const ParentLeaf = struct { parent: []const u8, leaf: []const u8 }; @@ -127,7 +151,7 @@ fn splitParent(path: []const u8) ParentLeaf { }; } -fn handleOpen(out: []u8, path: []const u8, flags: u32) usize { +fn handleOpen(out: []u8, path: []const u8, flags: u32, sender: u32) usize { var node = filesystem.resolve(path); if (node == null and flags & protocol.create != 0) { const split = splitParent(path); @@ -141,13 +165,12 @@ fn handleOpen(out: []u8, path: []const u8, flags: u32) usize { filesystem.truncate(&resolved); } const index = allocOpen() orelse return fail(out); - open_nodes[index] = .{ .used = true, .node = resolved }; + open_nodes[index] = .{ .used = true, .node = resolved, .owner = sender }; return writeReply(out, .{ .status = 0, .node = index }, &.{}); } fn onMessage(message: []const u8, out: []u8, sender: u32, capability: ?runtime.ipc.Handle) usize { _ = capability; - _ = sender; if (message.len < protocol.request_size) return fail(out); const request = std.mem.bytesToValue(protocol.Request, message[0..protocol.request_size]); const payload = message[protocol.request_size..]; @@ -157,7 +180,7 @@ fn onMessage(message: []const u8, out: []u8, sender: u32, capability: ?runtime.i filesystem.current_time_epoch = runtime.system.wallClock(); switch (request.operation) { - .open => return handleOpen(out, payload[0..@min(payload.len, request.len)], request.flags), + .open => return handleOpen(out, payload[0..@min(payload.len, request.len)], request.flags, sender), .read => { const o = openAt(request.node) orelse return fail(out); var buffer: [protocol.maximum_payload]u8 = undefined; @@ -237,5 +260,6 @@ pub fn main() void { .service = .fat, .init = initialise, .on_message = onMessage, + .on_notification = onNotification, }); } diff --git a/system/services/init/init.zig b/system/services/init/init.zig index 25efbe5..1a06839 100644 --- a/system/services/init/init.zig +++ b/system/services/init/init.zig @@ -28,7 +28,6 @@ const build_options = @import("build_options"); /// future init reads this from a manifest under /system/services instead of a /// hardcoded list.) const boot_services = [_][]const u8{ - "/system/services/vfs", "/system/services/input", "/system/services/device-manager", "/system/services/fat", diff --git a/system/services/logger/logger.zig b/system/services/logger/logger.zig index b2a8ff3..268b5b1 100644 --- a/system/services/logger/logger.zig +++ b/system/services/logger/logger.zig @@ -35,9 +35,11 @@ const runtime = @import("runtime"); const system = runtime.system; const fs = runtime.fs; -/// Where log trees live. Flips to "/var/log" when the kernel VFS routes /var -/// to the flash volume (M-G); today the FAT service mounts at /mnt/usb only. -const base = "/mnt/usb/var/log"; +/// Where log trees live: the FHS path. The kernel VFS routes /var to whatever +/// volume the fat server mounted there (today: the /var subtree of the USB +/// flash volume) — swapping the persistent medium later touches fat's two +/// mount calls, never this constant. +const base = "/var/log"; /// Drain cadence and the quiet period after which files are closed (flushed). const tick_ms = 250; @@ -119,9 +121,9 @@ fn onTerminate() void { fn tick() void { if (!storage_ready) { - // Probe the mount; the ring buffers until it appears. `exists` on the - // mount root is the documented readiness check. - if (!fs.exists("/mnt/usb")) return; + // makePath doubles as the readiness probe: while /var is unmounted the + // resolve fails fast (no storage round trip) and the ring buffers; the + // first success creates the whole per-boot tree. if (!fs.makePath(boot_directory[0..boot_directory_len])) return; storage_ready = true; if (!announced) { diff --git a/system/services/vfs-test/vfs-test.zig b/system/services/vfs-test/vfs-test.zig new file mode 100644 index 0000000..f0d00b0 --- /dev/null +++ b/system/services/vfs-test/vfs-test.zig @@ -0,0 +1,89 @@ +//! /system/tests/vfs-test — a ring-3 client that proves the kernel VFS end to +//! end through the plain `runtime.fs` API: resolve its OWN binary under the +//! kernel-served /system mount, check its metadata, read its ELF magic, and +//! list /system/services. On success it heartbeats "vfstest: ok" so the kernel +//! test can observe it; on failure it reports what went wrong. +//! +//! The "park" role (the fat-client-death test): open a file on the FAT volume, +//! then hold the handle forever without closing — the kill and the fat +//! server's release-on-death sweep are the point. + +const std = @import("std"); +const runtime = @import("runtime"); +const fs = runtime.fs; + +pub fn main(init: runtime.process.Init) void { + if (init.arguments.count > 1) { + park(); + return; + } + + // Our own binary, resolved through the kernel mount table. + const self_path = "/system/tests/vfs-test"; + var file = fs.open(self_path, .{}) orelse { + _ = runtime.system.write("vfstest: open of own binary failed\n"); + return; + }; + defer file.close(); + + const attributes = file.attributes() orelse { + _ = runtime.system.write("vfstest: attributes failed\n"); + return; + }; + if (attributes.kind != .regular or attributes.size == 0) { + _ = runtime.system.write("vfstest: bad attributes\n"); + return; + } + + var header: [4]u8 = undefined; + const n = file.read(&header) orelse 0; + if (n != 4 or header[0] != 0x7f or header[1] != 'E' or header[2] != 'L' or header[3] != 'F') { + _ = runtime.system.write("vfstest: ELF magic mismatch\n"); + return; + } + + // The write refusal: /system is read-only by construction. + if (file.write("x") != null or fs.open("/system/tests/new-file", .{ .create = true }) != null) { + _ = runtime.system.write("vfstest: /system accepted a write\n"); + return; + } + + // Listing: /system/services contains init. + var saw_init = false; + if (fs.openDirectory("/system/services")) |listing| { + var directory = listing; + defer directory.close(); + var entry: fs.Entry = .{}; + while (directory.next(&entry)) { + if (std.mem.eql(u8, entry.name(), "init")) saw_init = true; + } + } + if (!saw_init) { + _ = runtime.system.write("vfstest: /system/services listing missed init\n"); + return; + } + + while (true) { + _ = runtime.system.write("vfstest: ok\n"); + runtime.system.sleep(1000); + } +} + +fn park() void { + // The storage chain (usb -> block -> fat -> mounts) takes a few seconds; + // retry until the volume appears. + var parked: ?fs.File = null; + var tries: u32 = 0; + while (parked == null and tries < 1000) : (tries += 1) { + parked = fs.open("/mnt/usb/parked", .{ .create = true }); + if (parked == null) runtime.system.sleep(20); + } + if (parked == null) { + _ = runtime.system.write("vfstest: park open failed\n"); + return; + } + while (true) { + _ = runtime.system.write("vfstest: parked\n"); + runtime.system.sleep(500); + } +} diff --git a/system/services/vfs/path.zig b/system/services/vfs/path.zig deleted file mode 100644 index 028de25..0000000 --- a/system/services/vfs/path.zig +++ /dev/null @@ -1,39 +0,0 @@ -//! Pure path utilities for the VFS mount router — no IPC, no state, so they are -//! host-testable in isolation. The router uses these to decide whether an opened -//! path lies under a mount point and, if so, what it looks like relative to that -//! mount. - -const std = @import("std"); - -/// If `path` lies under `mount_prefix` — equal to it, or the prefix followed by a -/// path separator — return the path relative to the mount ("/" for an exact -/// match, otherwise the tail beginning with '/'). Returns null when `path` is not -/// under the mount, so a prefix like "/mnt/usb" never captures "/mnt/usbextra". -pub fn underMount(path: []const u8, mount_prefix: []const u8) ?[]const u8 { - if (path.len < mount_prefix.len) return null; - if (!std.mem.eql(u8, path[0..mount_prefix.len], mount_prefix)) return null; - if (path.len == mount_prefix.len) return "/"; - if (path[mount_prefix.len] != '/') return null; - return path[mount_prefix.len..]; -} - -/// Whether `path` is absolute (rooted at '/'). Bare names — what the flat ramfs -/// uses — are relative and never route through a mount. -pub fn isAbsolute(path: []const u8) bool { - return path.len > 0 and path[0] == '/'; -} - -test "underMount matches only at path boundaries" { - try std.testing.expectEqualStrings("/", underMount("/mnt/usb", "/mnt/usb").?); - try std.testing.expectEqualStrings("/system/kernel", underMount("/mnt/usb/system/kernel", "/mnt/usb").?); - try std.testing.expect(underMount("/mnt/usbextra", "/mnt/usb") == null); // not a boundary - try std.testing.expect(underMount("/mnt", "/mnt/usb") == null); // shorter than the prefix - try std.testing.expect(underMount("/other", "/mnt/usb") == null); - try std.testing.expect(underMount("greeting", "/mnt/usb") == null); // a bare name -} - -test "isAbsolute distinguishes paths from bare names" { - try std.testing.expect(isAbsolute("/mnt/usb")); - try std.testing.expect(!isAbsolute("greeting")); - try std.testing.expect(!isAbsolute("")); -} diff --git a/system/services/vfs/vfs-test.zig b/system/services/vfs/vfs-test.zig deleted file mode 100644 index 8a7caf3..0000000 --- a/system/services/vfs/vfs-test.zig +++ /dev/null @@ -1,62 +0,0 @@ -//! /system/services/vfs/vfs-test — a client that proves the VFS round trip end to end: open a -//! file through the `runtime.fs` file API, write to it, seek back, read it, and compare. -//! On success it heartbeats "vfstest: ok" so the kernel test can observe it; -//! on failure it reports what went wrong. Shipped in the initial_ramdisk alongside vfs. - -const std = @import("std"); -const runtime = @import("runtime"); -const fs = runtime.fs; - -pub fn main(init: runtime.process.Init) void { - const payload = "hello-vfs"; - - // The "park" role (the vfs-client-death test): open a file, then hold the - // handle forever without closing — the kill and the VFS's release-on-death - // are the point. - if (init.arguments.count > 1) { - var parked: ?fs.File = null; - var tries: u32 = 0; - while (parked == null and tries < 200) : (tries += 1) { - parked = fs.open("parked", .{ .create = true }); - if (parked == null) runtime.system.sleep(20); - } - if (parked == null) { - _ = runtime.system.write("vfstest: park open failed\n"); - return; - } - while (true) { - _ = runtime.system.write("vfstest: parked\n"); - runtime.system.sleep(500); - } - } - - // The VFS server may not have registered yet — retry open until it's up. - var opened: ?fs.File = null; - var tries: u32 = 0; - while (opened == null and tries < 200) : (tries += 1) { - opened = fs.open("greeting", .{ .create = true }); - if (opened == null) runtime.system.sleep(20); - } - var greeting = opened orelse { - _ = runtime.system.write("vfstest: open failed\n"); - return; - }; - - if ((greeting.write(payload) orelse 0) != payload.len) { - _ = runtime.system.write("vfstest: write failed\n"); - return; - } - greeting.seekTo(0); - - var buffer: [32]u8 = undefined; - const n = greeting.read(&buffer) orelse 0; - greeting.close(); - - if (n == payload.len and std.mem.eql(u8, buffer[0..n], payload)) { - while (true) { - _ = runtime.system.write("vfstest: ok\n"); - runtime.system.sleep(1000); - } - } - _ = runtime.system.write("vfstest: mismatch\n"); -} diff --git a/system/services/vfs/vfs.zig b/system/services/vfs/vfs.zig deleted file mode 100644 index 3b2223c..0000000 --- a/system/services/vfs/vfs.zig +++ /dev/null @@ -1,389 +0,0 @@ -//! system/services/vfs — the user-space VFS server. Shipped in the initial_ramdisk, spawned as a -//! ring-3 process, and reached by every other process through IPC (the `runtime` -//! file API marshals open/read/write/stat/close into calls to this server's -//! endpoint, published under the well-known `vfs` service id). -//! -//! Two namespaces meet here (M5): -//! - a small in-memory **ramfs** — opening a bare name creates it — enough to -//! prove the round trip and to back the existing tests; -//! - **mounted filesystems**: a mount table maps an absolute path prefix (e.g. -//! `/mnt/usb`) to a backend server's endpoint. An open of a path under a mount -//! is *forwarded* to that backend (which speaks this same protocol), and every -//! later read/write/status/readdir/close on the resulting handle is relayed to -//! it. The VFS is the router; a filesystem (FAT) is the backend. -//! -//! A path routes through a mount only when it is absolute and lies under a mount -//! prefix; bare names always resolve in the flat ramfs — the backward-compat -//! contract the `vfs` / `vfs-client-death` tests rely on. - -const std = @import("std"); -const runtime = @import("runtime"); -const protocol = runtime.vfs_protocol; -const path = @import("path.zig"); -const ipc = runtime.ipc; - -const Node = struct { - used: bool = false, - name: [24]u8 = undefined, - name_len: usize = 0, - data: [512]u8 = undefined, - size: usize = 0, -}; - -const OpenFile = struct { - used: bool = false, - // For a local handle: an index into `nodes`. For a forwarding handle: the - // node id the backend returned. (usize == u64 here, so it holds either.) - node: usize = 0, - // Non-null for a handle that forwards to a mounted backend. - backend: ?ipc.Handle = null, - // The client (task id — an IPC badge is one) that opened this handle. What - // release-on-death sweeps by: a service must never depend on its clients - // cleaning up after themselves (docs/process-lifecycle.md). - owner: u32 = 0, -}; - -// One mounted filesystem: an absolute path prefix and the backend endpoint that -// serves everything under it. -const Mount = struct { - used: bool = false, - prefix: [64]u8 = undefined, - prefix_len: usize = 0, - backend: ipc.Handle = 0, -}; - -var nodes = [_]Node{.{}} ** 8; -var opens = [_]OpenFile{.{}} ** 16; -var mounts = [_]Mount{.{}} ** 8; - -fn findNode(name: []const u8) ?usize { - for (&nodes, 0..) |*n, i| { - if (n.used and std.mem.eql(u8, n.name[0..n.name_len], name)) return i; - } - return null; -} - -fn createNode(name: []const u8) ?usize { - for (&nodes, 0..) |*n, i| { - if (!n.used) { - const l = @min(name.len, n.name.len); - @memcpy(n.name[0..l], name[0..l]); - n.* = .{ .used = true, .name = n.name, .name_len = l, .size = 0 }; - return i; - } - } - return null; -} - -fn openAt(id: u64) ?*OpenFile { - if (id >= opens.len) return null; - const o = &opens[@intCast(id)]; - return if (o.used) o else null; -} - -/// The mount whose prefix most specifically contains `name`, and the path -/// relative to it. Only absolute paths route; bare names never match. -const MountMatch = struct { backend: ipc.Handle, relative: []const u8 }; -fn longestMount(name: []const u8) ?MountMatch { - if (!path.isAbsolute(name)) return null; - var best: ?MountMatch = null; - var best_len: usize = 0; - for (&mounts) |*m| { - if (!m.used) continue; - const prefix = m.prefix[0..m.prefix_len]; - if (path.underMount(name, prefix)) |relative| { - if (best == null or prefix.len >= best_len) { - best_len = prefix.len; - best = .{ .backend = m.backend, .relative = relative }; - } - } - } - return best; -} - -/// Serialise a reply header + payload into `out`; returns the total length. -fn writeReply(out: []u8, reply: protocol.Reply, payload: []const u8) usize { - @memcpy(out[0..protocol.reply_size], std.mem.asBytes(&reply)); - const n = @min(payload.len, out.len - protocol.reply_size); - @memcpy(out[protocol.reply_size..][0..n], payload[0..n]); - return protocol.reply_size + n; -} - -fn fail(out: []u8) usize { - return writeReply(out, .{ .status = -1 }, &.{}); -} - -// --- mount routing ---------------------------------------------------------- - -/// Forward an open under a mount to its backend and, on success, allocate a local -/// forwarding handle that remembers the backend's node id. -fn forwardOpen(out: []u8, backend: ipc.Handle, relative: []const u8, flags: u32, sender: u32) usize { - const request = protocol.Request{ .operation = .open, .node = 0, .offset = 0, .len = @intCast(relative.len), .flags = flags }; - var message: [protocol.message_maximum]u8 = undefined; - @memcpy(message[0..protocol.request_size], std.mem.asBytes(&request)); - const rel = relative[0..@min(relative.len, protocol.maximum_payload)]; - @memcpy(message[protocol.request_size..][0..rel.len], rel); - - var reply: [protocol.message_maximum]u8 = undefined; - const n = ipc.call(backend, message[0 .. protocol.request_size + rel.len], &reply) catch return fail(out); - if (n < protocol.reply_size) return fail(out); - const backend_reply = std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]); - if (backend_reply.status != 0) return writeReply(out, .{ .status = backend_reply.status }, &.{}); - - for (&opens, 0..) |*o, i| { - if (!o.used) { - o.* = .{ .used = true, .node = @intCast(backend_reply.node), .backend = backend, .owner = sender }; - return writeReply(out, .{ .status = 0, .node = i }, &.{}); - } - } - return fail(out); -} - -/// Relay a read/write/status/readdir/close on a forwarding handle to the backend -/// (the node already rewritten to the backend's id) and copy its reply out. -fn forwardRequest(out: []u8, backend: ipc.Handle, request: protocol.Request, payload: []const u8) usize { - var message: [protocol.message_maximum]u8 = undefined; - @memcpy(message[0..protocol.request_size], std.mem.asBytes(&request)); - const plen = @min(payload.len, protocol.maximum_payload); - @memcpy(message[protocol.request_size..][0..plen], payload[0..plen]); - - var reply: [protocol.message_maximum]u8 = undefined; - const n = ipc.call(backend, message[0 .. protocol.request_size + plen], &reply) catch return fail(out); - const copy = @min(n, out.len); - @memcpy(out[0..copy], reply[0..copy]); - return copy; -} - -/// Forward a path-based operation (mkdir, unlink) under a mount to its backend and -/// relay the reply. No handle is created — these operate by path and return only a -/// status. -fn forwardPath(out: []u8, backend: ipc.Handle, operation: protocol.Operation, relative: []const u8) usize { - const request = protocol.Request{ .operation = operation, .node = 0, .offset = 0, .len = @intCast(relative.len), .flags = 0 }; - var message: [protocol.message_maximum]u8 = undefined; - @memcpy(message[0..protocol.request_size], std.mem.asBytes(&request)); - const rel = relative[0..@min(relative.len, protocol.maximum_payload)]; - @memcpy(message[protocol.request_size..][0..rel.len], rel); - var reply: [protocol.message_maximum]u8 = undefined; - const n = ipc.call(backend, message[0 .. protocol.request_size + rel.len], &reply) catch return fail(out); - const copy = @min(n, out.len); - @memcpy(out[0..copy], reply[0..copy]); - return copy; -} - -/// Forward a rename to its backend: the payload is the mount-relative old path, a -/// 0x00 separator, then the mount-relative new path. Relays the backend's reply. -fn forwardRename(out: []u8, backend: ipc.Handle, old_relative: []const u8, new_relative: []const u8) usize { - const total = old_relative.len + 1 + new_relative.len; - var message: [protocol.message_maximum]u8 = undefined; - if (protocol.request_size + total > message.len) return fail(out); - const request = protocol.Request{ .operation = .rename, .node = 0, .offset = 0, .len = @intCast(total), .flags = 0 }; - @memcpy(message[0..protocol.request_size], std.mem.asBytes(&request)); - var p = protocol.request_size; - @memcpy(message[p..][0..old_relative.len], old_relative); - p += old_relative.len; - message[p] = 0; - p += 1; - @memcpy(message[p..][0..new_relative.len], new_relative); - p += new_relative.len; - var reply: [protocol.message_maximum]u8 = undefined; - const n = ipc.call(backend, message[0..p], &reply) catch return fail(out); - const copy = @min(n, out.len); - @memcpy(out[0..copy], reply[0..copy]); - return copy; -} - -/// Best-effort close of a backend node (used when a dead client's forwarding -/// handles are swept — the backend must not leak the vfs's opens). -fn forwardClose(backend: ipc.Handle, backend_node: u64) void { - const request = protocol.Request{ .operation = .close, .node = backend_node, .offset = 0, .len = 0, .flags = 0 }; - var reply: [protocol.message_maximum]u8 = undefined; - _ = ipc.call(backend, std.mem.asBytes(&request), &reply) catch {}; -} - -fn doMount(out: []u8, prefix: []const u8, backend: ipc.Handle) usize { - for (&mounts) |*m| { - if (m.used and std.mem.eql(u8, m.prefix[0..m.prefix_len], prefix)) { - m.backend = backend; - std.log.info("remounted {s}", .{prefix}); - return writeReply(out, .{ .status = 0 }, &.{}); - } - } - for (&mounts) |*m| { - if (!m.used) { - const l = @min(prefix.len, m.prefix.len); - m.used = true; - @memcpy(m.prefix[0..l], prefix[0..l]); - m.prefix_len = l; - m.backend = backend; - std.log.info("mounted {s}", .{prefix[0..l]}); - return writeReply(out, .{ .status = 0 }, &.{}); - } - } - return fail(out); -} - -fn doUnmount(out: []u8, prefix: []const u8) usize { - for (&mounts) |*m| { - if (m.used and std.mem.eql(u8, m.prefix[0..m.prefix_len], prefix)) { - m.used = false; - std.log.info("unmounted {s}", .{prefix}); - return writeReply(out, .{ .status = 0 }, &.{}); - } - } - return fail(out); -} - -/// Release every open handle `client` held — called on that client's published -/// exit event. Forwarding handles also tell their backend to release; local -/// nodes (the ramfs files) stay, since ramfs contents outlive their writers. -fn releaseClientHandles(client: u32) void { - var released: u32 = 0; - for (&opens) |*o| { - if (o.used and o.owner == client) { - if (o.backend) |backend| forwardClose(backend, o.node); - o.used = false; - released += 1; - } - } - if (released != 0) std.log.info("released {d} handle(s) for dead client {d}", .{ released, client }); -} - -/// Handle one request from `sender`; write the reply into `out`, return its length. -fn handle(message: []const u8, out: []u8, sender: u32, capability: ?ipc.Handle) usize { - if (message.len < protocol.request_size) return fail(out); - const request = std.mem.bytesToValue(protocol.Request, message[0..protocol.request_size]); - const payload = message[protocol.request_size..]; - - switch (request.operation) { - .mount => { - const prefix = payload[0..@min(payload.len, request.len)]; - const backend = capability orelse return fail(out); - return doMount(out, prefix, backend); - }, - .unmount => { - const prefix = payload[0..@min(payload.len, request.len)]; - return doUnmount(out, prefix); - }, - .open => { - const name = payload[0..@min(payload.len, request.len)]; - if (longestMount(name)) |m| return forwardOpen(out, m.backend, m.relative, request.flags, sender); - // An absolute path with no matching mount is simply not found — only - // bare names live in the flat ramfs. (Else /mnt/usb would be silently - // created as a flat file when its filesystem is not yet mounted.) - if (path.isAbsolute(name)) return fail(out); - const ni = findNode(name) orelse createNode(name) orelse return fail(out); - for (&opens, 0..) |*o, i| { - if (!o.used) { - o.* = .{ .used = true, .node = ni, .backend = null, .owner = sender }; - return writeReply(out, .{ .status = 0, .node = i }, &.{}); - } - } - return fail(out); - }, - .read => { - const of = openAt(request.node) orelse return fail(out); - if (of.backend) |backend| { - var forwarded = request; - forwarded.node = of.node; - return forwardRequest(out, backend, forwarded, payload); - } - const nd = &nodes[@intCast(of.node)]; - const off: usize = @intCast(request.offset); - if (off >= nd.size) return writeReply(out, .{ .status = 0, .len = 0 }, &.{}); // EOF - const n = @min(@min(nd.size - off, request.len), protocol.maximum_payload); - return writeReply(out, .{ .status = 0, .len = @intCast(n) }, nd.data[off .. off + n]); - }, - .write => { - const of = openAt(request.node) orelse return fail(out); - if (of.backend) |backend| { - var forwarded = request; - forwarded.node = of.node; - return forwardRequest(out, backend, forwarded, payload); - } - const nd = &nodes[@intCast(of.node)]; - const off: usize = @intCast(request.offset); - if (off > nd.data.len) return fail(out); - const n = @min(@min(payload.len, request.len), nd.data.len - off); - @memcpy(nd.data[off .. off + n], payload[0..n]); - if (off + n > nd.size) nd.size = off + n; - return writeReply(out, .{ .status = 0, .len = @intCast(n) }, &.{}); - }, - .status => { - const of = openAt(request.node) orelse return fail(out); - if (of.backend) |backend| { - var forwarded = request; - forwarded.node = of.node; - return forwardRequest(out, backend, forwarded, payload); - } - const st = protocol.FileStatus{ .size = nodes[@intCast(of.node)].size, .kind = @intFromEnum(protocol.NodeKind.regular) }; - return writeReply(out, .{ .status = 0, .len = @sizeOf(protocol.FileStatus) }, std.mem.asBytes(&st)); - }, - .readdir => { - const of = openAt(request.node) orelse return fail(out); - if (of.backend) |backend| { - var forwarded = request; - forwarded.node = of.node; - return forwardRequest(out, backend, forwarded, payload); - } - // The flat ramfs has no directories: report EOF. - return writeReply(out, .{ .status = 0, .len = 0 }, &.{}); - }, - .close => { - const of = openAt(request.node); - if (of) |o| { - if (o.backend) |backend| forwardClose(backend, o.node); - o.used = false; - } - return writeReply(out, .{ .status = 0 }, &.{}); - }, - .mkdir, .unlink => { - const name = payload[0..@min(payload.len, request.len)]; - if (longestMount(name)) |m| return forwardPath(out, m.backend, request.operation, m.relative); - // Only a mounted backend has real directories; the flat ramfs cannot - // create or remove them (and a bare-name path is not a mount target). - return fail(out); - }, - .rename => { - const both = payload[0..@min(payload.len, request.len)]; - const sep = std.mem.indexOfScalar(u8, both, 0) orelse return fail(out); - const old_path = both[0..sep]; - const new_path = both[sep + 1 ..]; - const mo = longestMount(old_path) orelse return fail(out); - const mn = longestMount(new_path) orelse return fail(out); - // Both paths must live under the same mount — cross-filesystem rename is - // not supported. - if (mo.backend != mn.backend) return fail(out); - return forwardRename(out, mo.backend, mo.relative, mn.relative); - }, - } -} - -/// Startup, under the harness: subscribe to the published exit events — when a -/// client dies holding open handles, the exit notification is how the VFS learns -/// to release them (docs/process-lifecycle.md). -fn initialise(endpoint: ipc.Handle) bool { - if (!runtime.process.subscribeExits(endpoint)) { - _ = runtime.system.write("/system/services/vfs: exit subscription failed\n"); - } - _ = runtime.system.write("/system/services/vfs: ready\n"); - return true; -} - -/// A non-signal notification: the only kind the VFS subscribes to is exit events. -fn onNotification(badge: u64) void { - if (badge & ipc.notify_exit_bit != 0) { - releaseClientHandles(@intCast(badge & ~(ipc.notify_badge_bit | ipc.notify_exit_bit))); - } -} - -pub fn main() void { - // The harness owns the loop: requests dispatch to handle(), exit events to - // onNotification(), ping and terminate are answered for free — this service - // gained the whole lifecycle contract by deleting its hand-rolled loop. - runtime.service.run(protocol.message_maximum, .{ - .service = .vfs, - .init = initialise, - .on_message = handle, - .on_notification = onNotification, - }); -} diff --git a/system/services/vfs/protocol.zig b/system/vfs-protocol.zig similarity index 100% rename from system/services/vfs/protocol.zig rename to system/vfs-protocol.zig diff --git a/test/qemu_test.py b/test/qemu_test.py index 0ea1849..0fc76a2 100644 --- a/test/qemu_test.py +++ b/test/qemu_test.py @@ -427,6 +427,7 @@ CASES = [ # open handle, and the VFS releases it (process-lifecycle.md "Who learns of a death"). {"name": "vfs-client-death", "smp": 4, + "timeout": 90, # the park client waits out the whole USB->block->fat chain "expect": r"DANOS-TEST-RESULT: PASS", "fail": r"DANOS-TEST-RESULT: FAIL"}, # M17.4: signals over IPC — ping, reload, terminate (clean exit), the one-shot @@ -562,7 +563,7 @@ CASES = [ "smp": 4, "timeout": 150, "qmp_after": {"delay": 8, "command": "system_powerdown"}, - "expect": r"logger: logging to /mnt/usb/var/log/\d{4}-\d{2}-\d{2}T\d{6}Z[\s\S]*" + "expect": r"logger: logging to /var/log/\d{4}-\d{2}-\d{2}T\d{6}Z[\s\S]*" r"init: shutting down[\s\S]*" r"logger: flushed through sequence \d+[\s\S]*" r"power: entering S5",