kernel: VFS root — mount table, /system from the initrd, fs syscalls
system/kernel/vfs.zig is the resolve+redirect router: fs_resolve (#46) walks the kernel mount table; a path under the kernel-backed /system mount (the initrd, seeded by setInitialRamdisk with a derived directory table) yields a permanent stateless node token served by fs_node (#47) — read/status/readdir with copy-out, initrd reads lock-free — while a path under a userspace mount yields the backend's endpoint (installed in the caller's table, DEDUPLICATED so 16 slots can't be exhausted by repeated resolves) plus the rewritten mount-relative path; the caller then speaks the unchanged vfs-protocol rendezvous directly. The kernel never blocks on a userspace filesystem, holds no open-file state, and refuses create-intent on the immutable /system. fs_mount (#48) is the syscall form of the old router's op-6 cap-pass (possession of the backend handle is the capability; an optional rewrite prefix maps the mount into the backend's namespace — how /var will reach the flash volume); fs_unmount (#49) removes one. A dead backend's mount clears lazily on resolve. Dormant this milestone: the userspace vfs still serves runtime.fs unchanged; the kvfs QEMU case covers the kernel side (resolution, ELF magic read-through, /system listing, read-only + unknown refusals) until the M-G cutover exercises the syscalls end-to-end.
This commit is contained in:
@@ -34,6 +34,7 @@ const ipc = @import("ipc-synchronous.zig");
|
||||
const devices_broker = @import("devices-broker.zig");
|
||||
const irq = @import("irq.zig");
|
||||
const initial_ramdisk = @import("initial-ramdisk");
|
||||
const vfs = @import("vfs.zig");
|
||||
const log = @import("log.zig");
|
||||
const wall_clock = @import("wall-clock.zig");
|
||||
|
||||
@@ -138,6 +139,7 @@ var ramdisk_image: ?[]const u8 = null;
|
||||
/// handoff) so a user-space supervisor can `system_spawn` binaries out of it.
|
||||
pub fn setInitialRamdisk(image: []const u8) void {
|
||||
ramdisk_image = image;
|
||||
vfs.setInitialRamdisk(image); // the kernel VFS serves the same bytes at /system
|
||||
}
|
||||
|
||||
/// Spawn a bundled binary from the kernel by path. Used exactly once, to start
|
||||
@@ -235,6 +237,10 @@ fn system_call(state: *architecture.CpuState) void {
|
||||
.timer_bind => systemTimerBind(state),
|
||||
.klog_read => systemKlogRead(state),
|
||||
.klog_status => systemKlogStatus(state),
|
||||
.fs_resolve => systemFsResolve(state),
|
||||
.fs_node => systemFsNode(state),
|
||||
.fs_mount => systemFsMount(state),
|
||||
.fs_unmount => systemFsUnmount(state),
|
||||
.wall_clock => systemWallClock(state),
|
||||
.shm_create => systemShmCreate(state),
|
||||
.shm_map => systemShmMap(state),
|
||||
@@ -1299,6 +1305,127 @@ fn systemKlogStatus(state: *architecture.CpuState) void {
|
||||
}
|
||||
}
|
||||
|
||||
/// fs_resolve(path_ptr, path_len, flags, out_ptr, out_cap): route a path
|
||||
/// through the kernel mount table (docs/vfs-protocol.md). Kernel-served ->
|
||||
/// rax=fs_route_kernel, rdx=node token. Backend-served -> rax=fs_route_backend,
|
||||
/// rdx=an endpoint handle in the caller's table (deduplicated), and the
|
||||
/// rewritten mount-relative path copied to `out` with its length in the third
|
||||
/// result register. Fails for unknown paths, create-intent on /system, or an
|
||||
/// undersized out buffer.
|
||||
fn systemFsResolve(state: *architecture.CpuState) void {
|
||||
const path_ptr = architecture.systemCallArg(state, 0);
|
||||
const path_len = architecture.systemCallArg(state, 1);
|
||||
const flags = architecture.systemCallArg(state, 2);
|
||||
const out_ptr = architecture.systemCallArg(state, 3);
|
||||
const out_cap = architecture.systemCallArg(state, 4);
|
||||
if (path_len == 0 or path_len > 224 or path_ptr >= user_half_end or path_ptr + path_len > user_half_end) return fail(state);
|
||||
if (out_cap != 0 and (out_ptr >= user_half_end or out_ptr + out_cap > user_half_end)) return fail(state);
|
||||
const path = @as([*]const u8, @ptrFromInt(path_ptr))[0..path_len];
|
||||
const t = scheduler.current();
|
||||
|
||||
const flags_lock = sync.enter();
|
||||
defer sync.leave(flags_lock);
|
||||
switch (vfs.resolvePath(path, flags & abi.fs_flag_create != 0)) {
|
||||
.kernel_node => |node_token| {
|
||||
architecture.setSystemCallResult(state, abi.fs_route_kernel);
|
||||
architecture.setSystemCallResult2(state, node_token);
|
||||
},
|
||||
.backend => |*backend| {
|
||||
if (backend.path_len > out_cap) return fail(state);
|
||||
const handle = ipc.installHandleDeduped(t, backend.endpoint);
|
||||
if (handle < 0) return fail(state);
|
||||
const destination: [*]u8 = @ptrFromInt(out_ptr);
|
||||
@memcpy(destination[0..backend.path_len], backend.path[0..backend.path_len]);
|
||||
architecture.setSystemCallResult(state, abi.fs_route_backend);
|
||||
architecture.setSystemCallResult2(state, @intCast(handle));
|
||||
architecture.setSystemCallResult3(state, backend.path_len);
|
||||
},
|
||||
.not_found => fail(state),
|
||||
}
|
||||
}
|
||||
|
||||
/// fs_node(op, node_token, offset, buf_ptr, buf_len) -> bytes/0/-errno: serve a
|
||||
/// kernel-backed node. read copies file bytes; status copies a FileAttributes;
|
||||
/// readdir copies [DirectoryEntryHeader][name] for the `offset`th child. Reads
|
||||
/// of the immutable initrd never take the kernel lock.
|
||||
fn systemFsNode(state: *architecture.CpuState) void {
|
||||
const operation = architecture.systemCallArg(state, 0);
|
||||
const node_token = architecture.systemCallArg(state, 1);
|
||||
const offset = architecture.systemCallArg(state, 2);
|
||||
const buf_ptr = architecture.systemCallArg(state, 3);
|
||||
const buf_len = architecture.systemCallArg(state, 4);
|
||||
if (buf_ptr >= user_half_end or buf_len > user_half_end - buf_ptr) return fail(state);
|
||||
const capped = @min(buf_len, 64 * 1024); // bound any single copy
|
||||
const destination: [*]u8 = @ptrFromInt(buf_ptr);
|
||||
switch (operation) {
|
||||
abi.fs_node_read => {
|
||||
const n = vfs.nodeRead(node_token, offset, destination[0..capped]) orelse return fail(state);
|
||||
architecture.setSystemCallResult(state, n);
|
||||
},
|
||||
abi.fs_node_status => {
|
||||
var attributes = vfs.nodeStatus(node_token) orelse return fail(state);
|
||||
if (capped < @sizeOf(abi.FileAttributes)) return fail(state);
|
||||
@memcpy(destination[0..@sizeOf(abi.FileAttributes)], std.mem.asBytes(&attributes));
|
||||
architecture.setSystemCallResult(state, @sizeOf(abi.FileAttributes));
|
||||
},
|
||||
abi.fs_node_readdir => {
|
||||
const header_size = @sizeOf(abi.DirectoryEntryHeader);
|
||||
if (capped < header_size) return fail(state);
|
||||
var name_buffer: [64]u8 = undefined;
|
||||
const result = vfs.nodeReaddir(node_token, offset, &name_buffer) orelse {
|
||||
architecture.setSystemCallResult(state, 0); // past the end
|
||||
return;
|
||||
};
|
||||
var header = result.header;
|
||||
const total = header_size + @min(result.name_len, capped - header_size);
|
||||
@memcpy(destination[0..header_size], std.mem.asBytes(&header));
|
||||
@memcpy(destination[header_size..total], name_buffer[0 .. total - header_size]);
|
||||
architecture.setSystemCallResult(state, total);
|
||||
},
|
||||
else => fail(state),
|
||||
}
|
||||
}
|
||||
|
||||
/// fs_mount(prefix_ptr, prefix_len, backend_handle, rewrite_ptr, rewrite_len):
|
||||
/// mount a userspace filesystem at an absolute prefix. Possession of the
|
||||
/// backend endpoint handle is the capability — the same trust as the old
|
||||
/// router's cap-passing mount. The mount takes its own endpoint reference.
|
||||
fn systemFsMount(state: *architecture.CpuState) void {
|
||||
const prefix_ptr = architecture.systemCallArg(state, 0);
|
||||
const prefix_len = architecture.systemCallArg(state, 1);
|
||||
const backend_handle = architecture.systemCallArg(state, 2);
|
||||
const rewrite_ptr = architecture.systemCallArg(state, 3);
|
||||
const rewrite_len = architecture.systemCallArg(state, 4);
|
||||
if (prefix_len == 0 or prefix_len > 64 or prefix_ptr >= user_half_end or prefix_ptr + prefix_len > user_half_end) return fail(state);
|
||||
if (rewrite_len > 32) return fail(state);
|
||||
if (rewrite_len != 0 and (rewrite_ptr >= user_half_end or rewrite_ptr + rewrite_len > user_half_end)) return fail(state);
|
||||
const t = scheduler.current();
|
||||
const prefix = @as([*]const u8, @ptrFromInt(prefix_ptr))[0..prefix_len];
|
||||
const rewrite = if (rewrite_len == 0) "" else @as([*]const u8, @ptrFromInt(rewrite_ptr))[0..rewrite_len];
|
||||
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
const endpoint = ipc.resolveHandle(t, backend_handle) orelse return failErr(state, ipc.EBADF);
|
||||
endpoint.refcount += 1; // the mount table's reference
|
||||
if (!vfs.mountBackend(prefix, endpoint, rewrite)) {
|
||||
ipc.dropRef(endpoint);
|
||||
return fail(state);
|
||||
}
|
||||
architecture.setSystemCallResult(state, 0);
|
||||
}
|
||||
|
||||
/// fs_unmount(prefix_ptr, prefix_len): remove a backend mount.
|
||||
fn systemFsUnmount(state: *architecture.CpuState) void {
|
||||
const prefix_ptr = architecture.systemCallArg(state, 0);
|
||||
const prefix_len = architecture.systemCallArg(state, 1);
|
||||
if (prefix_len == 0 or prefix_len > 64 or prefix_ptr >= user_half_end or prefix_ptr + prefix_len > user_half_end) return fail(state);
|
||||
const prefix = @as([*]const u8, @ptrFromInt(prefix_ptr))[0..prefix_len];
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
if (!vfs.unmount(prefix)) return fail(state);
|
||||
architecture.setSystemCallResult(state, 0);
|
||||
}
|
||||
|
||||
/// mmap(len, prot) -> base: grant `len` bytes (rounded up to whole pages) of
|
||||
/// fresh, zeroed, writable+NX memory in the caller's mmap arena, and return the
|
||||
/// base virtual address. `prot` is accepted but not yet honoured (grants are
|
||||
|
||||
Reference in New Issue
Block a user