kernel: mounts have owners — V0 of the volume-manager plan
fs_unmount was gated by nothing but the /protocol carve-out: any process could unmount any prefix — latent with one mount owner, an obvious cross-tenant hole once volumes multiply. Each backend mount now records the mounting task, and the syscall layer enforces two rules that keep the restart story intact: only the owner unmounts (a dead owner's mount is swept lazily by resolution — strangers gain nothing by racing that), and a mount may be REPLACED only by its live owner or after its owner died (the respawned-filesystem path; displacement of a live mount would be worse than unmounting it). Kernel-installed mounts are never displaceable. The vfs-test park role is the discrimination: with the volume provably mounted it attempts the foreign unmount, requires the refusal AND the subtree still resolving, and withholds its "parked" marker otherwise — against the ungated kernel the unmount was ALLOWED and vfs-client-death fails; with the gate, green. (Its verification handle closes immediately: the kernel test string-matches "released 1 handle(s)".)
This commit is contained in:
@@ -2146,9 +2146,18 @@ fn systemFsMount(state: *architecture.CpuState) void {
|
||||
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
// The ownership gate. A LIVE owner's mount is its own: nobody else may
|
||||
// replace it — displacement-by-remount would be worse than unmounting.
|
||||
// A DEAD owner's mount is replaceable by anyone with a backend: that is
|
||||
// the restart story (a respawned filesystem is a new task retaking its
|
||||
// prefix). Kernel-installed mounts (owner 0) are never displaceable.
|
||||
if (vfs.mountOwner(prefix)) |owner| {
|
||||
const displaceable = owner != 0 and (owner == t.id or scheduler.taskByIdLocked(owner) == null);
|
||||
if (!displaceable) return failErr(state, ipc.EPERM);
|
||||
}
|
||||
const endpoint = ipc.resolveHandle(t, backend_handle) orelse return failErr(state, ipc.EBADF);
|
||||
endpoint.refcount += 1; // the mount table's reference
|
||||
if (!vfs.mountBackend(prefix, endpoint, rewrite)) {
|
||||
if (!vfs.mountBackend(prefix, endpoint, rewrite, t.id)) {
|
||||
ipc.dropRef(endpoint);
|
||||
return fail(state);
|
||||
}
|
||||
@@ -2166,6 +2175,12 @@ fn systemFsUnmount(state: *architecture.CpuState) void {
|
||||
if (!user_memory.copyFromUser(t.address_space, prefix_ptr, prefix)) return failErr(state, ipc.EFAULT);
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
// Only the mounting task unmounts. No dead-owner exception here: a dead
|
||||
// owner's mount is already being swept lazily by resolution, and a
|
||||
// stranger gains nothing legitimate by racing that — the restart story
|
||||
// goes through remount-replace, never through unmount.
|
||||
const owner = vfs.mountOwner(prefix) orelse return fail(state);
|
||||
if (owner != t.id) return failErr(state, ipc.EPERM);
|
||||
if (!vfs.unmount(prefix)) return fail(state);
|
||||
architecture.setSystemCallResult(state, 0);
|
||||
}
|
||||
|
||||
+28
-6
@@ -69,6 +69,10 @@ const Mount = struct {
|
||||
backend: ?*ipc.Endpoint = null, // referenced while mounted
|
||||
rewrite: [maximum_rewrite]u8 = undefined,
|
||||
rewrite_len: usize = 0,
|
||||
// The task that mounted this prefix — the ownership `fs_unmount` and
|
||||
// remount-replace are gated on (storage-architecture.md, the lifecycle
|
||||
// rule). Zero for kernel-installed mounts, which no task may displace.
|
||||
owner: u32 = 0,
|
||||
|
||||
fn prefixSlice(self: *const Mount) []const u8 {
|
||||
return self.prefix[0..self.prefix_len];
|
||||
@@ -375,11 +379,25 @@ fn refusesProtocolMount(prefix: []const u8) bool {
|
||||
return protocolBound(); // /protocol itself: first mount wins
|
||||
}
|
||||
|
||||
/// Mount `backend` at `prefix` with an optional backend-side `rewrite` prefix.
|
||||
/// The endpoint reference is taken by the caller (process.zig bumps it); refuses
|
||||
/// shadowing or replacing the initrd trees (/system, /test) — except the two
|
||||
/// carve-outs in `initrd_carve_outs`, the writable configuration/log subtrees.
|
||||
pub fn mountBackend(prefix: []const u8, backend: *ipc.Endpoint, rewrite: []const u8) bool {
|
||||
/// The task a backend mount at exactly `prefix` is recorded against, or null
|
||||
/// when nothing backend-shaped is mounted there. The syscall layer consults
|
||||
/// this before allowing a replace or an unmount — the ownership gate lives
|
||||
/// there, where the task table is; this table only remembers the fact.
|
||||
pub fn mountOwner(prefix: []const u8) ?u32 {
|
||||
for (&mounts) |*m| {
|
||||
if (m.used and m.kind == .backend and std.mem.eql(u8, m.prefixSlice(), prefix)) return m.owner;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Mount `backend` at `prefix` with an optional backend-side `rewrite` prefix,
|
||||
/// recorded against `owner`. The endpoint reference is taken by the caller
|
||||
/// (process.zig bumps it); refuses shadowing or replacing the initrd trees
|
||||
/// (/system, /test) — except the two carve-outs in `initrd_carve_outs`, the
|
||||
/// writable configuration/log subtrees. The replace-vs-refuse decision for an
|
||||
/// already-mounted prefix is the CALLER's (it can see task liveness); by the
|
||||
/// time this runs, replacing is decided.
|
||||
pub fn mountBackend(prefix: []const u8, backend: *ipc.Endpoint, rewrite: []const u8, owner: u32) bool {
|
||||
if (!isAbsolute(prefix) or prefix.len < 2 or prefix.len > maximum_prefix) return false;
|
||||
if (rewrite.len > maximum_rewrite) return false;
|
||||
if (refusesProtocolMount(prefix)) return false; // the registry's prefix is claimed once
|
||||
@@ -389,12 +407,16 @@ pub fn mountBackend(prefix: []const u8, backend: *ipc.Endpoint, rewrite: []const
|
||||
}
|
||||
}
|
||||
installMount(prefix, .backend, backend, rewrite);
|
||||
for (&mounts) |*m| {
|
||||
if (m.used and std.mem.eql(u8, m.prefixSlice(), prefix)) m.owner = owner;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
pub fn unmount(prefix: []const u8) bool {
|
||||
// Unmounting /protocol would delete the naming layer for everyone; nobody
|
||||
// may, init included. The mount lasts the boot.
|
||||
// may, init included. The mount lasts the boot. Ownership is checked by
|
||||
// the syscall layer (mountOwner) before this runs.
|
||||
if (std.mem.eql(u8, prefix, protocol_root)) return false;
|
||||
for (&mounts) |*m| {
|
||||
if (m.used and m.kind == .backend and std.mem.eql(u8, m.prefixSlice(), prefix)) {
|
||||
|
||||
@@ -84,6 +84,28 @@ fn park() void {
|
||||
_ = logging.write("vfstest: park open failed\n");
|
||||
return;
|
||||
}
|
||||
|
||||
// Mount ownership (V0, docs/volume-manager-plan.md): the volume is
|
||||
// provably mounted (the parked file just opened on it), it is FAT's mount,
|
||||
// and this process is not fat — unmounting it must be REFUSED and the
|
||||
// subtree must still resolve afterwards. Bailing here withholds the
|
||||
// "parked" marker, which fails the vfs-client-death case: before the
|
||||
// ownership gate existed, any process could unmount any prefix, and this
|
||||
// fixture would have deleted the volume out from under the whole boot.
|
||||
if (fs.fsUnmount("/volumes/usb")) {
|
||||
_ = logging.write("vfstest: foreign unmount was ALLOWED\n");
|
||||
return;
|
||||
}
|
||||
if (fs.open("/volumes/usb/parked", .{})) |resolved| {
|
||||
var verification = resolved;
|
||||
verification.close(); // the park below must be the client's ONLY open
|
||||
// handle — the kernel test string-matches "released 1 handle(s)".
|
||||
} else {
|
||||
_ = logging.write("vfstest: /volumes/usb gone after refused unmount\n");
|
||||
return;
|
||||
}
|
||||
_ = logging.write("vfstest: foreign unmount refused\n");
|
||||
|
||||
while (true) {
|
||||
_ = logging.write("vfstest: parked\n");
|
||||
time.sleepMillis(500);
|
||||
|
||||
Reference in New Issue
Block a user