diff --git a/system/abi.zig b/system/abi.zig index 0f3ee8f..f510004 100644 --- a/system/abi.zig +++ b/system/abi.zig @@ -72,6 +72,10 @@ pub const SystemCall = enum(u64) { thread_join = 43, // thread_join(tid) -> 0: block until the thread with id `tid` has exited (runtime.Thread.join; no per-thread IPC endpoint) (docs/threading.md) set_thread_pointer = 44, // set_thread_pointer(addr) -> 0: set the caller's thread pointer (user-space TLS base; x86_64 IA32_FS_BASE, aarch64 TPIDR_EL0); restored per task across context switches (docs/threading-plan.md M10) klog_status = 45, // klog_status(ptr) -> 0: copy a KlogStatus (ring cursors + the boot wall-clock anchor) out to a user buffer + fs_resolve = 46, // fs_resolve(path_ptr, path_len, flags, out_ptr, out_cap) -> route tag (rax: fs_route_*) + node token or backend handle (rdx); a backend resolve writes the rewritten mount-relative path into out (length in r8 via third result) + fs_node = 47, // fs_node(op, node_token, offset, buf_ptr, buf_len) -> bytes/0/-errno: read/status/readdir on a kernel-served node (op values mirror the vfs-protocol Operation numbers) + fs_mount = 48, // fs_mount(prefix_ptr, prefix_len, backend_handle, rewrite_ptr, rewrite_len) -> 0/-errno: mount a userspace filesystem's endpoint at an absolute prefix (possession of the handle is the capability) + fs_unmount = 49, // fs_unmount(prefix_ptr, prefix_len) -> 0/-errno: remove a backend mount _, }; @@ -225,6 +229,45 @@ pub const klog_record_alignment: usize = 8; /// Per-record payload cap (one line; longer emitter lines are truncated). pub const klog_maximum_message: usize = 256; +// --- the kernel VFS root (resolve + redirect) -------------------------------- +// fs_resolve routes a path through the kernel mount table. Kernel-backed mounts +// (the initrd at /system) resolve to a permanent node TOKEN served by fs_node; +// userspace mounts resolve to the backend's endpoint handle (installed in the +// caller's table, deduplicated) plus the rewritten mount-relative path — the +// caller then speaks the vfs-protocol to the backend directly. The kernel never +// blocks on a userspace filesystem. + +/// fs_resolve result tags (rax). +pub const fs_route_kernel: u64 = 0; // rdx = node token; serve via fs_node +pub const fs_route_backend: u64 = 1; // rdx = endpoint handle; speak vfs-protocol + +/// fs_node operations — the same numbers as the vfs-protocol Operation enum, so +/// client code shares one vocabulary. +pub const fs_node_read: u64 = 2; +pub const fs_node_status: u64 = 4; +pub const fs_node_readdir: u64 = 5; + +/// fs_resolve flags (same values as the vfs-protocol open flags). +pub const fs_flag_create: u64 = 1; + +/// FileStatus-shaped node metadata (matches the vfs-protocol payload layout). +pub const file_kind_regular: u32 = 0; +pub const file_kind_directory: u32 = 1; +pub const FileAttributes = extern struct { + size: u64, + kind: u32, + _pad: u32 = 0, + mtime: u64 = 0, +}; + +/// One fs_node readdir result: the header, followed by `name_len` name bytes in +/// the caller's buffer (matches the vfs-protocol DirectoryEntry layout). +pub const DirectoryEntryHeader = extern struct { + kind: u32, + name_len: u32, + size: u64, +}; + /// The klog_status copy-out: the ring's live cursors plus the wall-clock time /// of boot — the anchor a log persister names its per-boot directory with and /// combines with record timestamps for wall-clock line stamps. diff --git a/system/kernel/ipc-synchronous.zig b/system/kernel/ipc-synchronous.zig index 34476bf..2293a4b 100644 --- a/system/kernel/ipc-synchronous.zig +++ b/system/kernel/ipc-synchronous.zig @@ -516,6 +516,20 @@ pub fn installShmHandle(t: *Task, shm: *ShmObject) i64 { return installEntry(t, .{ .kind = handle_kind_shm, .ptr = @ptrCast(shm) }); } +/// Install an endpoint handle, reusing an existing slot that already names this +/// endpoint (no new reference taken in that case). For callers that install per +/// operation — fs_resolve — so a 16-slot table can't be exhausted by repeats. +/// Any subsystem installing handles per-call should come through here. +pub fn installHandleDeduped(t: *Task, endpoint: *Endpoint) i64 { + for (t.handles, 0..) |slot, i| { + const entry = slot orelse continue; + if (entry.kind == handle_kind_endpoint and entry.ptr == @as(*anyopaque, @ptrCast(endpoint))) return @intCast(i); + } + const h = installHandle(t, endpoint); + if (h >= 0) endpoint.refcount += 1; // the table entry owns a reference + return h; +} + /// Resolve a handle to its endpoint, or null if out of range, unused, or a different kind /// (e.g. an shm handle used where an endpoint is expected). pub fn resolveHandle(t: *Task, h: u64) ?*Endpoint { diff --git a/system/kernel/process.zig b/system/kernel/process.zig index cc2ac47..8bc5ff4 100644 --- a/system/kernel/process.zig +++ b/system/kernel/process.zig @@ -34,6 +34,7 @@ const ipc = @import("ipc-synchronous.zig"); const devices_broker = @import("devices-broker.zig"); const irq = @import("irq.zig"); const initial_ramdisk = @import("initial-ramdisk"); +const vfs = @import("vfs.zig"); const log = @import("log.zig"); const wall_clock = @import("wall-clock.zig"); @@ -138,6 +139,7 @@ var ramdisk_image: ?[]const u8 = null; /// handoff) so a user-space supervisor can `system_spawn` binaries out of it. pub fn setInitialRamdisk(image: []const u8) void { ramdisk_image = image; + vfs.setInitialRamdisk(image); // the kernel VFS serves the same bytes at /system } /// Spawn a bundled binary from the kernel by path. Used exactly once, to start @@ -235,6 +237,10 @@ fn system_call(state: *architecture.CpuState) void { .timer_bind => systemTimerBind(state), .klog_read => systemKlogRead(state), .klog_status => systemKlogStatus(state), + .fs_resolve => systemFsResolve(state), + .fs_node => systemFsNode(state), + .fs_mount => systemFsMount(state), + .fs_unmount => systemFsUnmount(state), .wall_clock => systemWallClock(state), .shm_create => systemShmCreate(state), .shm_map => systemShmMap(state), @@ -1299,6 +1305,127 @@ fn systemKlogStatus(state: *architecture.CpuState) void { } } +/// fs_resolve(path_ptr, path_len, flags, out_ptr, out_cap): route a path +/// through the kernel mount table (docs/vfs-protocol.md). Kernel-served -> +/// rax=fs_route_kernel, rdx=node token. Backend-served -> rax=fs_route_backend, +/// rdx=an endpoint handle in the caller's table (deduplicated), and the +/// rewritten mount-relative path copied to `out` with its length in the third +/// result register. Fails for unknown paths, create-intent on /system, or an +/// undersized out buffer. +fn systemFsResolve(state: *architecture.CpuState) void { + const path_ptr = architecture.systemCallArg(state, 0); + const path_len = architecture.systemCallArg(state, 1); + const flags = architecture.systemCallArg(state, 2); + const out_ptr = architecture.systemCallArg(state, 3); + const out_cap = architecture.systemCallArg(state, 4); + if (path_len == 0 or path_len > 224 or path_ptr >= user_half_end or path_ptr + path_len > user_half_end) return fail(state); + if (out_cap != 0 and (out_ptr >= user_half_end or out_ptr + out_cap > user_half_end)) return fail(state); + const path = @as([*]const u8, @ptrFromInt(path_ptr))[0..path_len]; + const t = scheduler.current(); + + const flags_lock = sync.enter(); + defer sync.leave(flags_lock); + switch (vfs.resolvePath(path, flags & abi.fs_flag_create != 0)) { + .kernel_node => |node_token| { + architecture.setSystemCallResult(state, abi.fs_route_kernel); + architecture.setSystemCallResult2(state, node_token); + }, + .backend => |*backend| { + if (backend.path_len > out_cap) return fail(state); + const handle = ipc.installHandleDeduped(t, backend.endpoint); + if (handle < 0) return fail(state); + const destination: [*]u8 = @ptrFromInt(out_ptr); + @memcpy(destination[0..backend.path_len], backend.path[0..backend.path_len]); + architecture.setSystemCallResult(state, abi.fs_route_backend); + architecture.setSystemCallResult2(state, @intCast(handle)); + architecture.setSystemCallResult3(state, backend.path_len); + }, + .not_found => fail(state), + } +} + +/// fs_node(op, node_token, offset, buf_ptr, buf_len) -> bytes/0/-errno: serve a +/// kernel-backed node. read copies file bytes; status copies a FileAttributes; +/// readdir copies [DirectoryEntryHeader][name] for the `offset`th child. Reads +/// of the immutable initrd never take the kernel lock. +fn systemFsNode(state: *architecture.CpuState) void { + const operation = architecture.systemCallArg(state, 0); + const node_token = architecture.systemCallArg(state, 1); + const offset = architecture.systemCallArg(state, 2); + const buf_ptr = architecture.systemCallArg(state, 3); + const buf_len = architecture.systemCallArg(state, 4); + if (buf_ptr >= user_half_end or buf_len > user_half_end - buf_ptr) return fail(state); + const capped = @min(buf_len, 64 * 1024); // bound any single copy + const destination: [*]u8 = @ptrFromInt(buf_ptr); + switch (operation) { + abi.fs_node_read => { + const n = vfs.nodeRead(node_token, offset, destination[0..capped]) orelse return fail(state); + architecture.setSystemCallResult(state, n); + }, + abi.fs_node_status => { + var attributes = vfs.nodeStatus(node_token) orelse return fail(state); + if (capped < @sizeOf(abi.FileAttributes)) return fail(state); + @memcpy(destination[0..@sizeOf(abi.FileAttributes)], std.mem.asBytes(&attributes)); + architecture.setSystemCallResult(state, @sizeOf(abi.FileAttributes)); + }, + abi.fs_node_readdir => { + const header_size = @sizeOf(abi.DirectoryEntryHeader); + if (capped < header_size) return fail(state); + var name_buffer: [64]u8 = undefined; + const result = vfs.nodeReaddir(node_token, offset, &name_buffer) orelse { + architecture.setSystemCallResult(state, 0); // past the end + return; + }; + var header = result.header; + const total = header_size + @min(result.name_len, capped - header_size); + @memcpy(destination[0..header_size], std.mem.asBytes(&header)); + @memcpy(destination[header_size..total], name_buffer[0 .. total - header_size]); + architecture.setSystemCallResult(state, total); + }, + else => fail(state), + } +} + +/// fs_mount(prefix_ptr, prefix_len, backend_handle, rewrite_ptr, rewrite_len): +/// mount a userspace filesystem at an absolute prefix. Possession of the +/// backend endpoint handle is the capability — the same trust as the old +/// router's cap-passing mount. The mount takes its own endpoint reference. +fn systemFsMount(state: *architecture.CpuState) void { + const prefix_ptr = architecture.systemCallArg(state, 0); + const prefix_len = architecture.systemCallArg(state, 1); + const backend_handle = architecture.systemCallArg(state, 2); + const rewrite_ptr = architecture.systemCallArg(state, 3); + const rewrite_len = architecture.systemCallArg(state, 4); + if (prefix_len == 0 or prefix_len > 64 or prefix_ptr >= user_half_end or prefix_ptr + prefix_len > user_half_end) return fail(state); + if (rewrite_len > 32) return fail(state); + if (rewrite_len != 0 and (rewrite_ptr >= user_half_end or rewrite_ptr + rewrite_len > user_half_end)) return fail(state); + const t = scheduler.current(); + const prefix = @as([*]const u8, @ptrFromInt(prefix_ptr))[0..prefix_len]; + const rewrite = if (rewrite_len == 0) "" else @as([*]const u8, @ptrFromInt(rewrite_ptr))[0..rewrite_len]; + + const flags = sync.enter(); + defer sync.leave(flags); + const endpoint = ipc.resolveHandle(t, backend_handle) orelse return failErr(state, ipc.EBADF); + endpoint.refcount += 1; // the mount table's reference + if (!vfs.mountBackend(prefix, endpoint, rewrite)) { + ipc.dropRef(endpoint); + return fail(state); + } + architecture.setSystemCallResult(state, 0); +} + +/// fs_unmount(prefix_ptr, prefix_len): remove a backend mount. +fn systemFsUnmount(state: *architecture.CpuState) void { + const prefix_ptr = architecture.systemCallArg(state, 0); + const prefix_len = architecture.systemCallArg(state, 1); + if (prefix_len == 0 or prefix_len > 64 or prefix_ptr >= user_half_end or prefix_ptr + prefix_len > user_half_end) return fail(state); + const prefix = @as([*]const u8, @ptrFromInt(prefix_ptr))[0..prefix_len]; + const flags = sync.enter(); + defer sync.leave(flags); + if (!vfs.unmount(prefix)) return fail(state); + architecture.setSystemCallResult(state, 0); +} + /// mmap(len, prot) -> base: grant `len` bytes (rounded up to whole pages) of /// fresh, zeroed, writable+NX memory in the caller's mmap arena, and return the /// base virtual address. `prot` is accepted but not yet honoured (grants are diff --git a/system/kernel/tests.zig b/system/kernel/tests.zig index 0cf6db1..5ea3de9 100644 --- a/system/kernel/tests.zig +++ b/system/kernel/tests.zig @@ -27,6 +27,7 @@ const sync = @import("sync.zig"); const process = @import("process.zig"); const initial_ramdisk = @import("initial-ramdisk"); const kernel_log = @import("log.zig"); +const kernel_vfs = @import("vfs.zig"); /// Formatted test-marker write. Goes through the kernel log (not straight to /// serial): the log lock is what keeps marker lines from interleaving with @@ -209,6 +210,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void { initialRamdiskTest(boot_information); } else if (eql(case, "vfs")) { vfsTest(boot_information); + } else if (eql(case, "kvfs")) { + kernelVfsTest(boot_information); } else if (eql(case, "input")) { inputTest(boot_information); } else if (eql(case, "iopass")) { @@ -3073,6 +3076,64 @@ fn bundledInit(boot_information: *const BootInformation) ?[]const u8 { return item.blob; } +/// The kernel VFS root (M-F): resolve initrd paths to node tokens, read an ELF +/// header through nodeRead, enumerate /system's derived directory table, and +/// verify the read-only + unknown-path refusals. Pure kernel-side — the +/// syscall surface gets its end-to-end coverage when runtime.fs cuts over. +fn kernelVfsTest(boot_information: *const BootInformation) void { + log("DANOS-TEST-BEGIN: kvfs\n", .{}); + if (boot_information.initial_ramdisk_len == 0) { + check("bootloader handed over the initial_ramdisk", false); + result(); + return; + } + const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len]; + process.setInitialRamdisk(image); // also seeds the kernel VFS /system mount + + // A file resolves to a kernel node token; its status and bytes are served. + const resolved = kernel_vfs.resolvePath("/system/services/init", false); + const is_file = resolved == .kernel_node; + check("/system/services/init resolves to a kernel node", is_file); + if (is_file) { + const status = kernel_vfs.nodeStatus(resolved.kernel_node); + check("its status is a non-empty regular file", status != null and status.?.kind == abi.file_kind_regular and status.?.size > 0); + var header: [4]u8 = undefined; + const n = kernel_vfs.nodeRead(resolved.kernel_node, 0, &header) orelse 0; + check("its first bytes are an ELF magic", n == 4 and header[0] == 0x7f and header[1] == 'E' and header[2] == 'L' and header[3] == 'F'); + } + + // Directories resolve and enumerate: /system lists services/drivers/tests. + const root_directory = kernel_vfs.resolvePath("/system", false); + check("/system resolves to a directory node", root_directory == .kernel_node); + var saw_services = false; + var saw_drivers = false; + var saw_files_in_services = false; + if (root_directory == .kernel_node) { + var cursor: u64 = 0; + var name: [64]u8 = undefined; + while (kernel_vfs.nodeReaddir(root_directory.kernel_node, cursor, &name)) |entry| : (cursor += 1) { + if (eql(name[0..entry.name_len], "services")) saw_services = true; + if (eql(name[0..entry.name_len], "drivers")) saw_drivers = true; + } + } + check("readdir /system yields services and drivers", saw_services and saw_drivers); + const services = kernel_vfs.resolvePath("/system/services", false); + if (services == .kernel_node) { + var cursor: u64 = 0; + var name: [64]u8 = undefined; + while (kernel_vfs.nodeReaddir(services.kernel_node, cursor, &name)) |entry| : (cursor += 1) { + if (eql(name[0..entry.name_len], "init")) saw_files_in_services = true; + } + } + check("readdir /system/services yields init", saw_files_in_services); + + // Refusals: unknown paths, and create-intent on the immutable initrd. + check("an unknown path does not resolve", kernel_vfs.resolvePath("/system/services/no-such", false) == .not_found); + check("an unmounted absolute path does not resolve", kernel_vfs.resolvePath("/elsewhere", false) == .not_found); + check("create on /system is refused (read-only)", kernel_vfs.resolvePath("/system/services/new-file", true) == .not_found); + result(); +} + fn spawnNamed(rd: initial_ramdisk.Reader, name: []const u8) bool { var i: u32 = 0; while (i < rd.count) : (i += 1) { diff --git a/system/kernel/vfs.zig b/system/kernel/vfs.zig new file mode 100644 index 0000000..9ca296c --- /dev/null +++ b/system/kernel/vfs.zig @@ -0,0 +1,368 @@ +//! The kernel-resident VFS root: the mount table and the kernel-backed nodes. +//! +//! The kernel's job here is NAMING, never data plumbing to userspace backends — +//! the mechanism is **resolve + redirect**: +//! +//! - `fs_resolve(path)` walks the mount table. A path under a KERNEL-backed +//! mount (the initrd at /system, the scratch ram nodes) resolves to a +//! stateless node TOKEN served directly by `fs_node` (read/status/readdir +//! with copy-out). A path under a USERSPACE mount (the fat server at +//! /mnt/usb and /var) resolves to the backend's ENDPOINT: the kernel +//! installs a (deduplicated) handle in the caller's table, rewrites the +//! path mount-relative, and the caller speaks the unchanged vfs-protocol +//! to the backend over the ordinary ipc_call rendezvous. The kernel never +//! blocks on a userspace server. +//! +//! - Kernel node tokens are PERMANENT for a boot: the initrd is immutable and +//! ram nodes are never reclaimed — no open-handle state, no close, no sweep +//! on client death. Backend file state lives in the backend, which sweeps +//! dead clients itself via the published exit events. +//! +//! Mounting is `fs_mount(prefix, backend_handle, rewrite)`: possession of the +//! backend endpoint handle is the capability, exactly the trust of the old +//! userspace router's op-6 cap-pass. An optional REWRITE prefix maps the mount +//! into the backend's namespace ("/var" -> fat's "/var" subtree while the same +//! backend also serves "/mnt/usb" from its root), so FHS paths stay decoupled +//! from which volume happens to carry them. + +const std = @import("std"); +const abi = @import("abi"); +const initial_ramdisk = @import("initial-ramdisk"); +const ipc = @import("ipc-synchronous.zig"); + +// --- node tokens ------------------------------------------------------------- + +/// Kind lives in the top byte of a token; the index below. Tokens are permanent +/// for a boot, so userspace may cache them freely. +pub const token_kind_shift = 56; +pub const token_kind_initrd_file: u64 = 1; +pub const token_kind_initrd_directory: u64 = 2; +pub const token_kind_ram: u64 = 3; + +fn token(kind: u64, index: u64) u64 { + return (kind << token_kind_shift) | index; +} + +fn tokenKind(t: u64) u64 { + return t >> token_kind_shift; +} + +fn tokenIndex(t: u64) u64 { + return t & ((@as(u64, 1) << token_kind_shift) - 1); +} + +// --- the mount table --------------------------------------------------------- + +pub const maximum_mounts = 8; +const maximum_prefix = 64; +const maximum_rewrite = 32; + +const MountKind = enum(u8) { kernel_initrd, backend }; + +const Mount = struct { + used: bool = false, + prefix: [maximum_prefix]u8 = undefined, + prefix_len: usize = 0, + kind: MountKind = .backend, + backend: ?*ipc.Endpoint = null, // referenced while mounted + rewrite: [maximum_rewrite]u8 = undefined, + rewrite_len: usize = 0, + + fn prefixSlice(self: *const Mount) []const u8 { + return self.prefix[0..self.prefix_len]; + } + fn rewriteSlice(self: *const Mount) []const u8 { + return self.rewrite[0..self.rewrite_len]; + } +}; + +var mounts: [maximum_mounts]Mount = @splat(.{}); + +/// The initrd image (set once at boot) and its derived directory table. +var ramdisk_image: ?[]const u8 = null; + +const maximum_directories = 8; +const Directory = struct { + path: [maximum_prefix]u8 = undefined, + path_len: usize = 0, + parent: usize = 0, // index into `directories`; 0 is /system itself + + fn slice(self: *const Directory) []const u8 { + return self.path[0..self.path_len]; + } +}; +var directories: [maximum_directories]Directory = @splat(.{}); +var directory_count: usize = 0; + +// --- pure path helpers (ported from the userspace router, with its tests) ---- + +/// If `path` lies under `mount_prefix` — equal to it, or the prefix followed by +/// a path separator — return the path relative to the mount ("/" for an exact +/// match, otherwise the tail beginning with '/'). Null when not under the +/// mount, so "/mnt/usb" never captures "/mnt/usbextra". +pub fn underMount(path: []const u8, mount_prefix: []const u8) ?[]const u8 { + if (path.len < mount_prefix.len) return null; + if (!std.mem.eql(u8, path[0..mount_prefix.len], mount_prefix)) return null; + if (path.len == mount_prefix.len) return "/"; + if (path[mount_prefix.len] != '/') return null; + return path[mount_prefix.len..]; +} + +pub fn isAbsolute(path: []const u8) bool { + return path.len > 0 and path[0] == '/'; +} + +/// The parent directory portion of an initrd path ("/system/services/fat" -> +/// "/system/services"). +fn parentOf(path: []const u8) []const u8 { + const slash = std.mem.lastIndexOfScalar(u8, path, '/') orelse return path[0..0]; + if (slash == 0) return path[0..1]; + return path[0..slash]; +} + +// --- boot wiring ------------------------------------------------------------- + +/// Publish the initrd as the kernel-backed /system mount and derive its bounded +/// directory table (the unique parents of the entry paths). Called once at boot. +pub fn setInitialRamdisk(image: []const u8) void { + ramdisk_image = image; + installMount("/system", .kernel_initrd, null, ""); + + // Directory 0 is /system itself. + directories[0] = .{ .parent = 0 }; + @memcpy(directories[0].path[0..7], "/system"); + directories[0].path_len = 7; + directory_count = 1; + + const rd = initial_ramdisk.Reader.init(image) orelse return; + var i: u32 = 0; + while (i < rd.count) : (i += 1) { + const item = rd.entry(i) orelse continue; + // Register every ancestor directory strictly below /system. + var parent = parentOf(item.name); + while (parent.len > 7) : (parent = parentOf(parent)) { + if (directoryIndex(parent) == null and directory_count < maximum_directories) { + var d = &directories[directory_count]; + @memcpy(d.path[0..parent.len], parent); + d.path_len = parent.len; + d.parent = 0; // fixed up below once all exist + directory_count += 1; + } + } + } + // Parent links (a second pass so out-of-order registration doesn't matter). + for (directories[1..directory_count]) |*d| { + d.parent = directoryIndex(parentOf(d.slice())) orelse 0; + } +} + +fn directoryIndex(path: []const u8) ?usize { + for (directories[0..directory_count], 0..) |*d, i| { + if (std.mem.eql(u8, d.slice(), path)) return i; + } + return null; +} + +fn installMount(prefix: []const u8, kind: MountKind, backend: ?*ipc.Endpoint, rewrite: []const u8) void { + // Remount replaces: a restarted backend re-mounts its prefix. + var slot: ?*Mount = null; + for (&mounts) |*m| { + if (m.used and std.mem.eql(u8, m.prefixSlice(), prefix)) { + if (m.backend) |old| ipc.dropRef(old); + slot = m; + break; + } + if (slot == null and !m.used) slot = m; + } + const m = slot orelse return; + m.* = .{ .used = true, .kind = kind, .backend = backend }; + @memcpy(m.prefix[0..prefix.len], prefix); + m.prefix_len = prefix.len; + @memcpy(m.rewrite[0..rewrite.len], rewrite); + m.rewrite_len = rewrite.len; +} + +// --- resolve ----------------------------------------------------------------- + +pub const Resolved = union(enum) { + /// Kernel-served: a permanent node token. + kernel_node: u64, + /// Backend-served: the endpoint plus the rewritten mount-relative path. + backend: struct { endpoint: *ipc.Endpoint, path: [maximum_rewrite + maximum_prefix + 160]u8, path_len: usize }, + not_found: void, +}; + +/// Longest-prefix match over the mount table, then per-kind resolution. +/// `create`-intent on the immutable /system fails here (EROFS-style). +pub fn resolvePath(path: []const u8, wants_create: bool) Resolved { + if (!isAbsolute(path)) { + return .{ .not_found = {} }; // bare names have no kernel namespace (ramfs retired) + } + var best: ?*Mount = null; + var best_relative: []const u8 = undefined; + for (&mounts) |*m| { + if (!m.used) continue; + const relative = underMount(path, m.prefixSlice()) orelse continue; + if (best == null or m.prefix_len > best.?.prefix_len) { + best = m; + best_relative = relative; + } + } + const m = best orelse return .{ .not_found = {} }; + switch (m.kind) { + .kernel_initrd => { + if (wants_create) return .{ .not_found = {} }; // read-only + return resolveInitrd(path); + }, + .backend => { + const endpoint = m.backend orelse return .{ .not_found = {} }; + if (endpoint.dead) { + // The backend died: treat the mount as gone (it re-mounts on + // restart) and release our reference lazily. + ipc.dropRef(endpoint); + m.backend = null; + m.used = false; + return .{ .not_found = {} }; + } + var out: Resolved = .{ .backend = .{ .endpoint = endpoint, .path = undefined, .path_len = 0 } }; + const rewrite = m.rewriteSlice(); + const tail = if (std.mem.eql(u8, best_relative, "/") and rewrite.len != 0) "" else best_relative; + const total = rewrite.len + tail.len; + if (total > out.backend.path.len or total == 0) { + if (rewrite.len == 0 and tail.len == 0) return .{ .not_found = {} }; + if (total > out.backend.path.len) return .{ .not_found = {} }; + } + @memcpy(out.backend.path[0..rewrite.len], rewrite); + @memcpy(out.backend.path[rewrite.len..][0..tail.len], tail); + out.backend.path_len = total; + return out; + }, + } +} + +fn resolveInitrd(path: []const u8) Resolved { + if (directoryIndex(path)) |index| return .{ .kernel_node = token(token_kind_initrd_directory, index) }; + const image = ramdisk_image orelse return .{ .not_found = {} }; + const rd = initial_ramdisk.Reader.init(image) orelse return .{ .not_found = {} }; + var i: u32 = 0; + while (i < rd.count) : (i += 1) { + const item = rd.entry(i) orelse continue; + if (std.mem.eql(u8, item.name, path)) return .{ .kernel_node = token(token_kind_initrd_file, i) }; + } + return .{ .not_found = {} }; +} + +// --- fs_node: serving kernel-backed nodes ------------------------------------ + +/// Read `out.len` bytes of an initrd file at `offset`. Returns bytes copied +/// (0 at EOF) or null for a bad token. Lock-free: the initrd is immutable. +pub fn nodeRead(node_token: u64, offset: u64, out: []u8) ?usize { + if (tokenKind(node_token) != token_kind_initrd_file) return null; + const image = ramdisk_image orelse return null; + const rd = initial_ramdisk.Reader.init(image) orelse return null; + const item = rd.entry(@intCast(tokenIndex(node_token))) orelse return null; + if (offset >= item.blob.len) return 0; + const n = @min(out.len, item.blob.len - @as(usize, @intCast(offset))); + @memcpy(out[0..n], item.blob[@intCast(offset)..][0..n]); + return n; +} + +/// A node's metadata in vfs-protocol FileStatus shape (size, kind, mtime). +pub fn nodeStatus(node_token: u64) ?abi.FileAttributes { + switch (tokenKind(node_token)) { + token_kind_initrd_file => { + const image = ramdisk_image orelse return null; + const rd = initial_ramdisk.Reader.init(image) orelse return null; + const item = rd.entry(@intCast(tokenIndex(node_token))) orelse return null; + return .{ .size = item.blob.len, .kind = abi.file_kind_regular }; + }, + token_kind_initrd_directory => { + if (tokenIndex(node_token) >= directory_count) return null; + return .{ .size = 0, .kind = abi.file_kind_directory }; + }, + else => return null, + } +} + +/// The `cursor`th child of an initrd directory: fills `name_out`, returns the +/// entry header, or null past the end / bad token. Cursor enumerates +/// subdirectories first, then files whose parent is this directory — stable, +/// because the initrd is immutable. +pub fn nodeReaddir(node_token: u64, cursor: u64, name_out: []u8) ?struct { header: abi.DirectoryEntryHeader, name_len: usize } { + if (tokenKind(node_token) != token_kind_initrd_directory) return null; + const directory_index = tokenIndex(node_token); + if (directory_index >= directory_count) return null; + const self_path = directories[@intCast(directory_index)].slice(); + + var index: u64 = 0; + // Subdirectories whose parent is this directory. + for (directories[0..directory_count], 0..) |*d, i| { + if (i == directory_index) continue; + if (d.parent != directory_index) continue; + if (i == 0) continue; + if (index == cursor) { + const name = d.slice()[self_path.len + 1 ..]; + const n = @min(name.len, name_out.len); + @memcpy(name_out[0..n], name[0..n]); + return .{ .header = .{ .kind = abi.file_kind_directory, .name_len = @intCast(n), .size = 0 }, .name_len = n }; + } + index += 1; + } + // Files directly inside this directory. + const image = ramdisk_image orelse return null; + const rd = initial_ramdisk.Reader.init(image) orelse return null; + var i: u32 = 0; + while (i < rd.count) : (i += 1) { + const item = rd.entry(i) orelse continue; + if (!std.mem.eql(u8, parentOf(item.name), self_path)) continue; + if (index == cursor) { + const name = item.name[self_path.len + 1 ..]; + const n = @min(name.len, name_out.len); + @memcpy(name_out[0..n], name[0..n]); + return .{ .header = .{ .kind = abi.file_kind_regular, .name_len = @intCast(n), .size = item.blob.len }, .name_len = n }; + } + index += 1; + } + return null; +} + +// --- mount/unmount (syscall bodies; caller resolved the handle) -------------- + +/// Mount `backend` at `prefix` with an optional backend-side `rewrite` prefix. +/// The endpoint reference is taken by the caller (process.zig bumps it); refuses +/// shadowing or replacing /system. +pub fn mountBackend(prefix: []const u8, backend: *ipc.Endpoint, rewrite: []const u8) bool { + if (!isAbsolute(prefix) or prefix.len < 2 or prefix.len > maximum_prefix) return false; + if (rewrite.len > maximum_rewrite) return false; + if (underMount(prefix, "/system") != null) return false; // the initrd is not shadowable + installMount(prefix, .backend, backend, rewrite); + return true; +} + +pub fn unmount(prefix: []const u8) bool { + for (&mounts) |*m| { + if (m.used and m.kind == .backend and std.mem.eql(u8, m.prefixSlice(), prefix)) { + if (m.backend) |endpoint| ipc.dropRef(endpoint); + m.* = .{}; + return true; + } + } + return false; +} + +// --- tests (host) ------------------------------------------------------------ + +test "underMount matches only at path boundaries" { + try std.testing.expectEqualStrings("/", underMount("/mnt/usb", "/mnt/usb").?); + try std.testing.expectEqualStrings("/system/kernel", underMount("/mnt/usb/system/kernel", "/mnt/usb").?); + try std.testing.expect(underMount("/mnt/usbextra", "/mnt/usb") == null); + try std.testing.expect(underMount("/mnt", "/mnt/usb") == null); + try std.testing.expect(underMount("/other", "/mnt/usb") == null); + try std.testing.expect(underMount("greeting", "/mnt/usb") == null); +} + +test "parentOf walks toward the root" { + try std.testing.expectEqualStrings("/system/services", parentOf("/system/services/fat")); + try std.testing.expectEqualStrings("/system", parentOf("/system/services")); + try std.testing.expectEqualStrings("/", parentOf("/system")); +} diff --git a/test/qemu_test.py b/test/qemu_test.py index d65e8a6..0ea1849 100644 --- a/test/qemu_test.py +++ b/test/qemu_test.py @@ -624,6 +624,12 @@ CASES = [ "fail": r"DANOS-TEST-RESULT: FAIL"}, # The user-space VFS: a client opens/writes/reads a file through the rt file # API, which IPCs the VFS server process; the round trip must match. + # The kernel VFS root (M-F): the mount table serves the initrd at /system — + # path resolution, node status/read (an ELF magic), and directory listing, + # asserted kernel-side. + {"name": "kvfs", + "expect": r"DANOS-TEST-RESULT: PASS", + "fail": r"DANOS-TEST-RESULT: FAIL"}, {"name": "vfs", "expect": r"DANOS-TEST-RESULT: PASS", "fail": r"DANOS-TEST-RESULT: FAIL"},