kernel: mounts have owners — V0 of the volume-manager plan

fs_unmount was gated by nothing but the /protocol carve-out: any process
could unmount any prefix — latent with one mount owner, an obvious
cross-tenant hole once volumes multiply. Each backend mount now records the
mounting task, and the syscall layer enforces two rules that keep the
restart story intact: only the owner unmounts (a dead owner's mount is
swept lazily by resolution — strangers gain nothing by racing that), and a
mount may be REPLACED only by its live owner or after its owner died (the
respawned-filesystem path; displacement of a live mount would be worse than
unmounting it). Kernel-installed mounts are never displaceable.

The vfs-test park role is the discrimination: with the volume provably
mounted it attempts the foreign unmount, requires the refusal AND the
subtree still resolving, and withholds its "parked" marker otherwise —
against the ungated kernel the unmount was ALLOWED and vfs-client-death
fails; with the gate, green. (Its verification handle closes immediately:
the kernel test string-matches "released 1 handle(s)".)
This commit is contained in:
Daniel Samson
2026-08-09 16:16:14 +01:00
parent 451abba000
commit 60b41c0e82
3 changed files with 66 additions and 7 deletions
+16 -1
View File
@@ -2146,9 +2146,18 @@ fn systemFsMount(state: *architecture.CpuState) void {
const flags = sync.enter();
defer sync.leave(flags);
// The ownership gate. A LIVE owner's mount is its own: nobody else may
// replace it — displacement-by-remount would be worse than unmounting.
// A DEAD owner's mount is replaceable by anyone with a backend: that is
// the restart story (a respawned filesystem is a new task retaking its
// prefix). Kernel-installed mounts (owner 0) are never displaceable.
if (vfs.mountOwner(prefix)) |owner| {
const displaceable = owner != 0 and (owner == t.id or scheduler.taskByIdLocked(owner) == null);
if (!displaceable) return failErr(state, ipc.EPERM);
}
const endpoint = ipc.resolveHandle(t, backend_handle) orelse return failErr(state, ipc.EBADF);
endpoint.refcount += 1; // the mount table's reference
if (!vfs.mountBackend(prefix, endpoint, rewrite)) {
if (!vfs.mountBackend(prefix, endpoint, rewrite, t.id)) {
ipc.dropRef(endpoint);
return fail(state);
}
@@ -2166,6 +2175,12 @@ fn systemFsUnmount(state: *architecture.CpuState) void {
if (!user_memory.copyFromUser(t.address_space, prefix_ptr, prefix)) return failErr(state, ipc.EFAULT);
const flags = sync.enter();
defer sync.leave(flags);
// Only the mounting task unmounts. No dead-owner exception here: a dead
// owner's mount is already being swept lazily by resolution, and a
// stranger gains nothing legitimate by racing that — the restart story
// goes through remount-replace, never through unmount.
const owner = vfs.mountOwner(prefix) orelse return fail(state);
if (owner != t.id) return failErr(state, ipc.EPERM);
if (!vfs.unmount(prefix)) return fail(state);
architecture.setSystemCallResult(state, 0);
}
+28 -6
View File
@@ -69,6 +69,10 @@ const Mount = struct {
backend: ?*ipc.Endpoint = null, // referenced while mounted
rewrite: [maximum_rewrite]u8 = undefined,
rewrite_len: usize = 0,
// The task that mounted this prefix — the ownership `fs_unmount` and
// remount-replace are gated on (storage-architecture.md, the lifecycle
// rule). Zero for kernel-installed mounts, which no task may displace.
owner: u32 = 0,
fn prefixSlice(self: *const Mount) []const u8 {
return self.prefix[0..self.prefix_len];
@@ -375,11 +379,25 @@ fn refusesProtocolMount(prefix: []const u8) bool {
return protocolBound(); // /protocol itself: first mount wins
}
/// Mount `backend` at `prefix` with an optional backend-side `rewrite` prefix.
/// The endpoint reference is taken by the caller (process.zig bumps it); refuses
/// shadowing or replacing the initrd trees (/system, /test) — except the two
/// carve-outs in `initrd_carve_outs`, the writable configuration/log subtrees.
pub fn mountBackend(prefix: []const u8, backend: *ipc.Endpoint, rewrite: []const u8) bool {
/// The task a backend mount at exactly `prefix` is recorded against, or null
/// when nothing backend-shaped is mounted there. The syscall layer consults
/// this before allowing a replace or an unmount — the ownership gate lives
/// there, where the task table is; this table only remembers the fact.
pub fn mountOwner(prefix: []const u8) ?u32 {
for (&mounts) |*m| {
if (m.used and m.kind == .backend and std.mem.eql(u8, m.prefixSlice(), prefix)) return m.owner;
}
return null;
}
/// Mount `backend` at `prefix` with an optional backend-side `rewrite` prefix,
/// recorded against `owner`. The endpoint reference is taken by the caller
/// (process.zig bumps it); refuses shadowing or replacing the initrd trees
/// (/system, /test) — except the two carve-outs in `initrd_carve_outs`, the
/// writable configuration/log subtrees. The replace-vs-refuse decision for an
/// already-mounted prefix is the CALLER's (it can see task liveness); by the
/// time this runs, replacing is decided.
pub fn mountBackend(prefix: []const u8, backend: *ipc.Endpoint, rewrite: []const u8, owner: u32) bool {
if (!isAbsolute(prefix) or prefix.len < 2 or prefix.len > maximum_prefix) return false;
if (rewrite.len > maximum_rewrite) return false;
if (refusesProtocolMount(prefix)) return false; // the registry's prefix is claimed once
@@ -389,12 +407,16 @@ pub fn mountBackend(prefix: []const u8, backend: *ipc.Endpoint, rewrite: []const
}
}
installMount(prefix, .backend, backend, rewrite);
for (&mounts) |*m| {
if (m.used and std.mem.eql(u8, m.prefixSlice(), prefix)) m.owner = owner;
}
return true;
}
pub fn unmount(prefix: []const u8) bool {
// Unmounting /protocol would delete the naming layer for everyone; nobody
// may, init included. The mount lasts the boot.
// may, init included. The mount lasts the boot. Ownership is checked by
// the syscall layer (mountOwner) before this runs.
if (std.mem.eql(u8, prefix, protocol_root)) return false;
for (&mounts) |*m| {
if (m.used and m.kind == .backend and std.mem.eql(u8, m.prefixSlice(), prefix)) {