From 60b41c0e8245c34c88cd6f0616425e516aeb3d71 Mon Sep 17 00:00:00 2001 From: Daniel Samson <12231216+daniel-samson@users.noreply.github.com> Date: Sun, 9 Aug 2026 16:16:14 +0100 Subject: [PATCH] =?UTF-8?q?kernel:=20mounts=20have=20owners=20=E2=80=94=20?= =?UTF-8?q?V0=20of=20the=20volume-manager=20plan?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit fs_unmount was gated by nothing but the /protocol carve-out: any process could unmount any prefix — latent with one mount owner, an obvious cross-tenant hole once volumes multiply. Each backend mount now records the mounting task, and the syscall layer enforces two rules that keep the restart story intact: only the owner unmounts (a dead owner's mount is swept lazily by resolution — strangers gain nothing by racing that), and a mount may be REPLACED only by its live owner or after its owner died (the respawned-filesystem path; displacement of a live mount would be worse than unmounting it). Kernel-installed mounts are never displaceable. The vfs-test park role is the discrimination: with the volume provably mounted it attempts the foreign unmount, requires the refusal AND the subtree still resolving, and withholds its "parked" marker otherwise — against the ungated kernel the unmount was ALLOWED and vfs-client-death fails; with the gate, green. (Its verification handle closes immediately: the kernel test string-matches "released 1 handle(s)".) --- system/kernel/process.zig | 17 ++++++++++- system/kernel/vfs.zig | 34 ++++++++++++++++++---- test/system/services/vfs-test/vfs-test.zig | 22 ++++++++++++++ 3 files changed, 66 insertions(+), 7 deletions(-) diff --git a/system/kernel/process.zig b/system/kernel/process.zig index 6fa297f..9897365 100644 --- a/system/kernel/process.zig +++ b/system/kernel/process.zig @@ -2146,9 +2146,18 @@ fn systemFsMount(state: *architecture.CpuState) void { const flags = sync.enter(); defer sync.leave(flags); + // The ownership gate. A LIVE owner's mount is its own: nobody else may + // replace it — displacement-by-remount would be worse than unmounting. + // A DEAD owner's mount is replaceable by anyone with a backend: that is + // the restart story (a respawned filesystem is a new task retaking its + // prefix). Kernel-installed mounts (owner 0) are never displaceable. + if (vfs.mountOwner(prefix)) |owner| { + const displaceable = owner != 0 and (owner == t.id or scheduler.taskByIdLocked(owner) == null); + if (!displaceable) return failErr(state, ipc.EPERM); + } const endpoint = ipc.resolveHandle(t, backend_handle) orelse return failErr(state, ipc.EBADF); endpoint.refcount += 1; // the mount table's reference - if (!vfs.mountBackend(prefix, endpoint, rewrite)) { + if (!vfs.mountBackend(prefix, endpoint, rewrite, t.id)) { ipc.dropRef(endpoint); return fail(state); } @@ -2166,6 +2175,12 @@ fn systemFsUnmount(state: *architecture.CpuState) void { if (!user_memory.copyFromUser(t.address_space, prefix_ptr, prefix)) return failErr(state, ipc.EFAULT); const flags = sync.enter(); defer sync.leave(flags); + // Only the mounting task unmounts. No dead-owner exception here: a dead + // owner's mount is already being swept lazily by resolution, and a + // stranger gains nothing legitimate by racing that — the restart story + // goes through remount-replace, never through unmount. + const owner = vfs.mountOwner(prefix) orelse return fail(state); + if (owner != t.id) return failErr(state, ipc.EPERM); if (!vfs.unmount(prefix)) return fail(state); architecture.setSystemCallResult(state, 0); } diff --git a/system/kernel/vfs.zig b/system/kernel/vfs.zig index 4cdca03..c3842f7 100644 --- a/system/kernel/vfs.zig +++ b/system/kernel/vfs.zig @@ -69,6 +69,10 @@ const Mount = struct { backend: ?*ipc.Endpoint = null, // referenced while mounted rewrite: [maximum_rewrite]u8 = undefined, rewrite_len: usize = 0, + // The task that mounted this prefix — the ownership `fs_unmount` and + // remount-replace are gated on (storage-architecture.md, the lifecycle + // rule). Zero for kernel-installed mounts, which no task may displace. + owner: u32 = 0, fn prefixSlice(self: *const Mount) []const u8 { return self.prefix[0..self.prefix_len]; @@ -375,11 +379,25 @@ fn refusesProtocolMount(prefix: []const u8) bool { return protocolBound(); // /protocol itself: first mount wins } -/// Mount `backend` at `prefix` with an optional backend-side `rewrite` prefix. -/// The endpoint reference is taken by the caller (process.zig bumps it); refuses -/// shadowing or replacing the initrd trees (/system, /test) — except the two -/// carve-outs in `initrd_carve_outs`, the writable configuration/log subtrees. -pub fn mountBackend(prefix: []const u8, backend: *ipc.Endpoint, rewrite: []const u8) bool { +/// The task a backend mount at exactly `prefix` is recorded against, or null +/// when nothing backend-shaped is mounted there. The syscall layer consults +/// this before allowing a replace or an unmount — the ownership gate lives +/// there, where the task table is; this table only remembers the fact. +pub fn mountOwner(prefix: []const u8) ?u32 { + for (&mounts) |*m| { + if (m.used and m.kind == .backend and std.mem.eql(u8, m.prefixSlice(), prefix)) return m.owner; + } + return null; +} + +/// Mount `backend` at `prefix` with an optional backend-side `rewrite` prefix, +/// recorded against `owner`. The endpoint reference is taken by the caller +/// (process.zig bumps it); refuses shadowing or replacing the initrd trees +/// (/system, /test) — except the two carve-outs in `initrd_carve_outs`, the +/// writable configuration/log subtrees. The replace-vs-refuse decision for an +/// already-mounted prefix is the CALLER's (it can see task liveness); by the +/// time this runs, replacing is decided. +pub fn mountBackend(prefix: []const u8, backend: *ipc.Endpoint, rewrite: []const u8, owner: u32) bool { if (!isAbsolute(prefix) or prefix.len < 2 or prefix.len > maximum_prefix) return false; if (rewrite.len > maximum_rewrite) return false; if (refusesProtocolMount(prefix)) return false; // the registry's prefix is claimed once @@ -389,12 +407,16 @@ pub fn mountBackend(prefix: []const u8, backend: *ipc.Endpoint, rewrite: []const } } installMount(prefix, .backend, backend, rewrite); + for (&mounts) |*m| { + if (m.used and std.mem.eql(u8, m.prefixSlice(), prefix)) m.owner = owner; + } return true; } pub fn unmount(prefix: []const u8) bool { // Unmounting /protocol would delete the naming layer for everyone; nobody - // may, init included. The mount lasts the boot. + // may, init included. The mount lasts the boot. Ownership is checked by + // the syscall layer (mountOwner) before this runs. if (std.mem.eql(u8, prefix, protocol_root)) return false; for (&mounts) |*m| { if (m.used and m.kind == .backend and std.mem.eql(u8, m.prefixSlice(), prefix)) { diff --git a/test/system/services/vfs-test/vfs-test.zig b/test/system/services/vfs-test/vfs-test.zig index 3b89f94..be7b95f 100644 --- a/test/system/services/vfs-test/vfs-test.zig +++ b/test/system/services/vfs-test/vfs-test.zig @@ -84,6 +84,28 @@ fn park() void { _ = logging.write("vfstest: park open failed\n"); return; } + + // Mount ownership (V0, docs/volume-manager-plan.md): the volume is + // provably mounted (the parked file just opened on it), it is FAT's mount, + // and this process is not fat — unmounting it must be REFUSED and the + // subtree must still resolve afterwards. Bailing here withholds the + // "parked" marker, which fails the vfs-client-death case: before the + // ownership gate existed, any process could unmount any prefix, and this + // fixture would have deleted the volume out from under the whole boot. + if (fs.fsUnmount("/volumes/usb")) { + _ = logging.write("vfstest: foreign unmount was ALLOWED\n"); + return; + } + if (fs.open("/volumes/usb/parked", .{})) |resolved| { + var verification = resolved; + verification.close(); // the park below must be the client's ONLY open + // handle — the kernel test string-matches "released 1 handle(s)". + } else { + _ = logging.write("vfstest: /volumes/usb gone after refused unmount\n"); + return; + } + _ = logging.write("vfstest: foreign unmount refused\n"); + while (true) { _ = logging.write("vfstest: parked\n"); time.sleepMillis(500);