Files
danos/system/services/volume-manager/volume-manager.zig
T
Daniel Samson 7c6ed2ca09 volume-manager: harden the probe and supervision from the V3 review
Five confirmed defects from the boundary review:

1. (security) The VM never checked a partition fit inside the device, so a
   crafted MBR could hand the driver a range whose base+lba wraps past a u32
   — panicking usb-storage in a loop, and at multi-volume overlapping a
   neighbour. This is the exact invariant the clamp's overflow-safety rests
   on. partition.firstVolume now skips any entry that runs past the device
   (host-tested), establishing the invariant where the untrusted bytes are
   first read.
2. (leak) The probe re-acquired a fresh block channel on every 500 ms retry,
   leaking a handle each time on a medium-absent device. The channel is now
   acquired once and kept.
3. (wedge) A failed spawn or defineRange stranded the volume with no retry;
   both now arm a backoff restart.
4. (loop) fat respawn had no exit-reason gate, no backoff, no crash-loop cap
   — a faulting filesystem respawned in a zero-delay loop, and a clean exit
   was resurrected. Supervision now mirrors the device manager: a clean exit
   is not restarted, a fault backs off, three fast deaths give up.
5. (removable) A device that parsed to no volume was terminal; it now keeps
   polling so an inserted medium is picked up — the removal-lifecycle trigger.

Known limitation (noted, not fixed here): if the VM itself crashes and init
restarts it, the orphaned fat keeps serving vfs while the new VM spawns a
second fat whose bind is refused — the same "manager restart re-learns the
world" gap the device manager also defers. The old fat keeps storage working.

Neutral: partition unit tests + fat-mount, volume-probe, block-range, logger
all green.
2026-08-09 18:50:01 +01:00

259 lines
11 KiB
Zig

//! system/services/volume-manager — the storage layer's policy home
//! (docs/file-system-development/storage-architecture.md). Beside the device
//! manager: that owns the DEVICE tree, this owns the VOLUME layer. It probes a
//! storage provider's partition table, confines each filesystem to its
//! partition, spawns one filesystem per volume, and answers that filesystem's
//! startup hello with the range-confined block channel — so the filesystem
//! never finds its storage by name and never sees the whole device. It
//! supervises the filesystems it spawns, exactly as the device manager
//! supervises drivers.
//!
//! This increment (V3b) is the flip: the FAT service stops acquiring its own
//! volume and is spawned here instead, confined to its partition, and handed
//! its channel over the volume-manager protocol. Single volume for now; the
//! mount map (volumes.csv) and multi-volume land next.
const std = @import("std");
const channel = @import("channel");
const device_manager_protocol = @import("device-manager-protocol");
const volume_manager_protocol = @import("volume-manager-protocol");
const driver = @import("driver");
const ipc = @import("ipc");
const block = @import("block");
const memory = @import("memory");
const logging = @import("logging");
const process = @import("process");
const service = @import("service");
const time = @import("time");
const envelope = @import("envelope");
const partition = @import("partition.zig");
const Serve = volume_manager_protocol.Protocol.Provider(void);
const Invocation = envelope.Invocation;
const Answer = envelope.Answer;
/// The single volume this increment handles: its provider channel, its block
/// sub-range, its identity, the id it is addressed by, and the filesystem
/// process serving it (0 until spawned; reset on death for respawn).
const Volume = struct {
storage: block.Device,
base_lba: u64,
block_count: u64,
identity: u64,
id: u64,
filesystem_pid: u32 = 0,
};
/// The filesystem binary a probed volume is served by. The signature->binary
/// map (filesystems.csv) lands with the identity ladder; for now every FAT-shaped
/// volume gets the FAT service.
const filesystem_binary = "/system/services/fat";
const volume_id: u64 = 1;
var service_endpoint: ipc.Handle = 0;
var manager_handle: ?ipc.Handle = null;
var bounce: memory.DmaRegion = undefined;
var bounce_ready = false;
var probed = false;
var volume: ?Volume = null;
/// The storage channel, acquired ONCE and kept — re-acquiring on every probe
/// retry would leak a handle per attempt on a medium-absent device.
var storage_device: ?block.Device = null;
var logged_no_volume = false;
const probe_retry_ms = 500;
// Filesystem supervision, mirroring the device manager's (device-manager.zig):
// a clean exit is not restarted, a fault restarts with backoff, and a fast
// crash loop gives up rather than spinning. Without this a faulting filesystem
// respawns in a zero-delay loop.
const fast_death_ns: u64 = 2_000_000_000;
const crash_loop_cap: u32 = 3;
const backoff_base_ms: u64 = 300;
var fs_restarts: u32 = 0;
var fs_spawn_ns: u64 = 0;
var fs_failed = false;
/// Set when a backoff timer is pending so its tick respawns rather than probes.
var restart_pending = false;
fn acquireStorage() ?block.Device {
const manager = manager_handle orelse opened: {
const handle = channel.openEndpoint("device-manager") orelse return null;
manager_handle = handle;
break :opened handle;
};
const Entry = device_manager_protocol.ChildEntry;
var start: u64 = 0;
while (true) {
const enumerate = envelope.Header{ .operation = envelope.operation_enumerate, .target = start };
var reply: [device_manager_protocol.message_maximum]u8 = undefined;
const length = ipc.call(manager, std.mem.asBytes(&enumerate), &reply) catch return null;
const status = envelope.statusOf(reply[0..length]) orelse return null;
if (status.status != 0) return null;
const carried = @min(@as(usize, status.len), length -| envelope.prefix_size);
const tail = reply[envelope.prefix_size..][0..carried];
const count = tail.len / @sizeOf(Entry);
if (count == 0) return null;
var index: usize = 0;
while (index < count) : (index += 1) {
const entry = std.mem.bytesToValue(Entry, tail[index * @sizeOf(Entry) ..][0..@sizeOf(Entry)]);
if (entry.device_id == device_manager_protocol.no_device) continue;
if ((entry.identity >> 16) & 0xff != 0x08 or (entry.identity >> 8) & 0xff != 0x06) continue;
const exchanged = driver.helloOn(manager, .consumer, entry.device_id, null, true) orelse return null;
const provider = exchanged.channel orelse continue;
return .{ .endpoint = provider };
}
start += count;
}
}
/// Spawn the filesystem for `v`, confine it to the volume's range, and record
/// its pid. The confinement is defined for the fresh pid BEFORE the filesystem
/// runs, so its first read is already bounded; the volume manager is the
/// confinement controller (it defines the first range on the device).
fn spawnFilesystem(v: *Volume) void {
if (fs_failed) return;
const pid = process.spawnSupervised(filesystem_binary, &.{"1"}, service_endpoint) orelse {
_ = logging.write("volume-manager: could not spawn the filesystem; retrying\n");
armRestart();
return;
};
if (!v.storage.defineRange(pid, v.base_lba, v.block_count)) {
_ = logging.write("volume-manager: could not confine the filesystem to its volume; retrying\n");
_ = process.kill(pid);
armRestart();
return;
}
v.filesystem_pid = pid;
fs_spawn_ns = time.clock();
std.log.info("volume 0x{x} -> {s} (pid {d}), lba {d}, {d} blocks", .{ v.identity, filesystem_binary, pid, v.base_lba, v.block_count });
}
/// Arm a one-shot timer to (re)spawn the filesystem after backoff — used both
/// when a spawn step fails and when a running filesystem faults. Distinguished
/// from the probe timer by `restart_pending`.
fn armRestart() void {
const delay = if (fs_restarts == 0) backoff_base_ms else backoff_base_ms << @intCast(@min(fs_restarts - 1, 5));
restart_pending = true;
_ = time.timerOnce(service_endpoint, delay);
}
fn tryProbe() void {
if (probed) return;
if (!bounce_ready) {
bounce = memory.dmaAlloc(512, memory.dma_coherent | memory.dma_shareable) orelse return;
bounce_ready = true;
}
// Acquire the storage channel once and keep it: a fresh consumer-hello per
// retry would leak a handle every 500 ms on a device whose medium is absent.
const device = storage_device orelse acquired: {
const d = acquireStorage() orelse return;
storage_device = d;
break :acquired d;
};
if (bounce.handle) |handle| {
if (!device.attach(handle)) return;
_ = ipc.close(handle);
bounce.handle = null;
}
const geometry = device.geometry() orelse return;
if (!device.read(0, 1, bounce.physical)) return;
const sector: [*]const u8 = @ptrFromInt(bounce.virtual);
const found = partition.firstVolume(sector[0..512], geometry.block_count) orelse {
// No volume yet. On removable media this can mean no medium is present —
// keep polling so an inserted medium is picked up (the removal-lifecycle
// trigger). Log once, and do NOT terminate the probe.
if (!logged_no_volume) {
_ = logging.write("volume-manager: no volume on the storage device yet\n");
logged_no_volume = true;
}
return;
};
volume = .{ .storage = device, .base_lba = found.base_lba, .block_count = found.block_count, .identity = found.identity, .id = volume_id };
probed = true;
spawnFilesystem(&volume.?);
}
/// A filesystem announces itself for the volume it was spawned to serve. Reply
/// with that volume's block channel (already range-confined to this filesystem's
/// badge) as the call's returned capability. No channel means the volume is not
/// ready — the filesystem retries.
fn onHello(_: void, invocation: Invocation(volume_manager_protocol.Hello), _: Answer(void)) isize {
const v = volume orelse return 0; // not probed yet — retryable, no cap
if (invocation.target != v.id) return 0; // unknown volume — retryable
if (invocation.sender != v.filesystem_pid) {
// Not the filesystem we spawned for this volume. Refuse: only the
// confined filesystem gets the channel.
std.log.info("refused hello for volume {d} from process {d}", .{ invocation.target, invocation.sender });
return -envelope.EPERM;
}
service.replyWithCapability(v.storage.endpoint);
std.log.info("handed volume {d} to pid {d}", .{ v.id, invocation.sender });
return 0;
}
const handlers = Serve.Handlers{ .hello = onHello };
fn onMessage(message: []const u8, out: []u8, sender: u32, arrived: *ipc.Arrival) usize {
// No verb takes a capability up, so the turn closes whatever arrives.
return Serve.dispatch({}, handlers, message, sender, arrived.peek(), out);
}
fn initialise(endpoint: ipc.Handle) bool {
service_endpoint = endpoint;
_ = logging.write("volume-manager: starting, waiting for a storage device\n");
_ = process.subscribeExits(endpoint);
tryProbe();
if (!probed) _ = time.timerOnce(endpoint, probe_retry_ms);
return true;
}
fn onNotification(badge: u64) void {
const got = ipc.Received{ .len = 0, .badge = badge, .cap = null };
if (got.isTimer()) {
// One timer signal, two jobs, told apart by state: a pending backoff
// restart, otherwise the probe retry.
if (restart_pending) {
restart_pending = false;
if (volume) |*v| spawnFilesystem(v);
return;
}
tryProbe();
if (!probed) _ = time.timerOnce(service_endpoint, probe_retry_ms);
return;
}
// A filesystem died. The exit reason drives the decision, exactly as the
// device manager supervises drivers: a clean exit meant to stop; a fault
// restarts with backoff until a fast crash loop gives up. The old range is
// reclaimed by the driver on the same death; the respawn confines afresh.
if (got.isChildExit()) {
const dead = got.childProcessId();
const v = &(volume orelse return);
if (v.filesystem_pid != dead) return;
v.filesystem_pid = 0;
const reason = process.exitReason(dead) orelse .fault;
if (reason == .exited) {
std.log.info("filesystem for volume {d} exited cleanly; not restarting", .{v.id});
return;
}
const alive = time.clock() -| fs_spawn_ns;
fs_restarts = if (alive < fast_death_ns) fs_restarts + 1 else 1;
if (fs_restarts >= crash_loop_cap) {
fs_failed = true;
std.log.info("filesystem for volume {d} is failing repeatedly; giving up", .{v.id});
return;
}
std.log.info("filesystem for volume {d} died ({s}); restarting", .{ v.id, @tagName(reason) });
armRestart();
}
}
pub fn main(init: process.Init) void {
_ = init;
service.run(volume_manager_protocol.message_maximum, .{
.service = "volume-manager",
.init = initialise,
.on_message = onMessage,
.on_notification = onNotification,
});
}