Files
danos/system/services/volume-manager/volume-manager.zig
T
Daniel Samson e3ec9fa668 volume-manager: try every mass-storage entry, watch the one we opened
The full suite caught a V4 regression: under AMD-Vi the device-manager tree
carries more than one mass-storage-identity entry (a phantom no driver is
bound to, which answers a consumer hello with NO channel). V4 split presence
from acquisition and picked the FIRST identity match blindly, so it kept
helloing the phantom (device 27) and never reached the real storage (device
31). V3's inline loop had skipped no-channel entries with `orelse continue`;
the split lost that.

Restore it: openAnyStorage tries each matching entry and takes the first whose
channel opens, recording its device id. Removal detection then watches THAT
specific device id leave the tree (isDevicePresent), not "any mass-storage" —
so a phantom that never leaves cannot mask a real removal. Both are bare
enumerates; the hello only happens while bringing a volume up.

Green: amd-iommu-usb-storage, fat-mount, volume-removal.
2026-08-09 20:08:30 +01:00

336 lines
15 KiB
Zig

//! system/services/volume-manager — the storage layer's policy home
//! (docs/file-system-development/storage-architecture.md). Beside the device
//! manager: that owns the DEVICE tree, this owns the VOLUME layer. It probes a
//! storage provider's partition table, confines each filesystem to its
//! partition, spawns one filesystem per volume, and answers that filesystem's
//! startup hello with the range-confined block channel — so the filesystem
//! never finds its storage by name and never sees the whole device. It
//! supervises the filesystems it spawns, exactly as the device manager
//! supervises drivers.
//!
//! This increment (V3b) is the flip: the FAT service stops acquiring its own
//! volume and is spawned here instead, confined to its partition, and handed
//! its channel over the volume-manager protocol. Single volume for now; the
//! mount map (volumes.csv) and multi-volume land next.
const std = @import("std");
const channel = @import("channel");
const device_manager_protocol = @import("device-manager-protocol");
const volume_manager_protocol = @import("volume-manager-protocol");
const driver = @import("driver");
const ipc = @import("ipc");
const block = @import("block");
const memory = @import("memory");
const logging = @import("logging");
const process = @import("process");
const service = @import("service");
const time = @import("time");
const envelope = @import("envelope");
const partition = @import("partition.zig");
const Serve = volume_manager_protocol.Protocol.Provider(void);
const Invocation = envelope.Invocation;
const Answer = envelope.Answer;
/// The single volume this increment handles: its provider channel, its block
/// sub-range, its identity, the id it is addressed by, and the filesystem
/// process serving it (0 until spawned; reset on death for respawn).
const Volume = struct {
storage: block.Device,
storage_device_id: u64, // the device-manager id this volume's provider serves
base_lba: u64,
block_count: u64,
identity: u64,
id: u64,
filesystem_pid: u32 = 0,
};
/// The filesystem binary a probed volume is served by. The signature->binary
/// map (filesystems.csv) lands with the identity ladder; for now every FAT-shaped
/// volume gets the FAT service.
const filesystem_binary = "/system/services/fat";
const volume_id: u64 = 1;
var service_endpoint: ipc.Handle = 0;
var manager_handle: ?ipc.Handle = null;
var bounce: memory.DmaRegion = undefined;
var bounce_ready = false;
/// The currently-mounted volume, or null while no storage is present. The whole
/// removal lifecycle is this field going null and back: the poll sees the
/// storage provider leave the device tree (a pulled stick), kills the filesystem
/// and clears this; when it returns, the poll re-acquires and re-mounts.
var volume: ?Volume = null;
var logged_no_volume = false;
/// How often the poll checks whether the storage provider is present. Fast
/// enough that an unplug unmounts promptly; the poll is a bare device-manager
/// enumerate, no channel work, so it is cheap to run continuously.
const poll_interval_ms = 500;
// Filesystem supervision, mirroring the device manager's (device-manager.zig):
// a clean exit is not restarted, a fault restarts with backoff, and a fast
// crash loop gives up rather than spinning. Without this a faulting filesystem
// respawns in a zero-delay loop.
const fast_death_ns: u64 = 2_000_000_000;
const crash_loop_cap: u32 = 3;
const backoff_base_ms: u64 = 300;
var fs_restarts: u32 = 0;
var fs_spawn_ns: u64 = 0;
var fs_failed = false;
/// A fat restart is due at `restart_due_ns`; the poll loop performs it once the
/// backoff has elapsed (one timer, folded into the poll — no second timer).
var restart_pending = false;
var restart_due_ns: u64 = 0;
fn deviceManager() ?ipc.Handle {
if (manager_handle) |h| return h;
const handle = channel.openEndpoint("device-manager") orelse return null;
manager_handle = handle;
return handle;
}
const OpenedStorage = struct { device_id: u64, device: block.Device };
/// The first mass-storage provider whose block channel actually opens, with its
/// device id. A device-manager tree can carry more than one entry of the
/// mass-storage identity — a phantom that no driver is bound to answers a
/// consumer hello with NO channel — so this tries each and takes the first that
/// yields a channel, exactly as a filesystem's own acquisition loop does.
/// Called only when there is no volume (an insertion), so the hellos it makes
/// are not per-poll churn.
fn openAnyStorage() ?OpenedStorage {
const manager = deviceManager() orelse return null;
const Entry = device_manager_protocol.ChildEntry;
var start: u64 = 0;
while (true) {
const enumerate = envelope.Header{ .operation = envelope.operation_enumerate, .target = start };
var reply: [device_manager_protocol.message_maximum]u8 = undefined;
const length = ipc.call(manager, std.mem.asBytes(&enumerate), &reply) catch return null;
const status = envelope.statusOf(reply[0..length]) orelse return null;
if (status.status != 0) return null;
const carried = @min(@as(usize, status.len), length -| envelope.prefix_size);
const tail = reply[envelope.prefix_size..][0..carried];
const count = tail.len / @sizeOf(Entry);
if (count == 0) return null;
var index: usize = 0;
while (index < count) : (index += 1) {
const entry = std.mem.bytesToValue(Entry, tail[index * @sizeOf(Entry) ..][0..@sizeOf(Entry)]);
if (entry.device_id == device_manager_protocol.no_device) continue;
if ((entry.identity >> 16) & 0xff != 0x08 or (entry.identity >> 8) & 0xff != 0x06) continue;
const exchanged = driver.helloOn(manager, .consumer, entry.device_id, null, true) orelse continue;
const provider = exchanged.channel orelse continue; // a phantom / not-yet-bound entry
return .{ .device_id = entry.device_id, .device = .{ .endpoint = provider } };
}
start += count;
}
}
/// Whether `device_id` is still in the device-manager tree — a bare enumerate,
/// no consumer-hello, so it is cheap to call every poll. This is how removal is
/// detected: the specific device the mounted volume sits on disappears.
fn isDevicePresent(device_id: u64) bool {
const manager = deviceManager() orelse return false;
const Entry = device_manager_protocol.ChildEntry;
var start: u64 = 0;
while (true) {
const enumerate = envelope.Header{ .operation = envelope.operation_enumerate, .target = start };
var reply: [device_manager_protocol.message_maximum]u8 = undefined;
const length = ipc.call(manager, std.mem.asBytes(&enumerate), &reply) catch return false;
const status = envelope.statusOf(reply[0..length]) orelse return false;
if (status.status != 0) return false;
const carried = @min(@as(usize, status.len), length -| envelope.prefix_size);
const tail = reply[envelope.prefix_size..][0..carried];
const count = tail.len / @sizeOf(Entry);
if (count == 0) return false;
var index: usize = 0;
while (index < count) : (index += 1) {
const entry = std.mem.bytesToValue(Entry, tail[index * @sizeOf(Entry) ..][0..@sizeOf(Entry)]);
if (entry.device_id == device_id) return true;
}
start += count;
}
}
/// Spawn the filesystem for `v`, confine it to the volume's range, and record
/// its pid. The confinement is defined for the fresh pid BEFORE the filesystem
/// runs, so its first read is already bounded; the volume manager is the
/// confinement controller (it defines the first range on the device).
fn spawnFilesystem(v: *Volume) void {
if (fs_failed) return;
const pid = process.spawnSupervised(filesystem_binary, &.{"1"}, service_endpoint) orelse {
_ = logging.write("volume-manager: could not spawn the filesystem; retrying\n");
armRestart();
return;
};
if (!v.storage.defineRange(pid, v.base_lba, v.block_count)) {
_ = logging.write("volume-manager: could not confine the filesystem to its volume; retrying\n");
_ = process.kill(pid);
armRestart();
return;
}
v.filesystem_pid = pid;
fs_spawn_ns = time.clock();
std.log.info("volume 0x{x} -> {s} (pid {d}), lba {d}, {d} blocks", .{ v.identity, filesystem_binary, pid, v.base_lba, v.block_count });
}
/// Schedule a fat restart after backoff; the poll loop performs it once due.
fn armRestart() void {
const delay = if (fs_restarts == 0) backoff_base_ms else backoff_base_ms << @intCast(@min(fs_restarts - 1, 5));
restart_due_ns = time.clock() + delay * 1_000_000;
restart_pending = true;
}
/// A storage provider just appeared: open its channel, read block 0, parse the
/// volume, and spawn its filesystem. On any failure the channel is closed (so a
/// present-but-unreadable device does not leak a handle every poll) and `volume`
/// stays null — the next poll retries. A fresh medium gets a fresh supervision
/// budget.
fn bringUpVolume() void {
if (!bounce_ready) {
bounce = memory.dmaAlloc(512, memory.dma_coherent | memory.dma_shareable) orelse return;
bounce_ready = true;
}
const opened = openAnyStorage() orelse return;
const device = opened.device;
// Attach the read buffer to THIS device (a no-op without an enforcing IOMMU).
// The handle is kept, not closed, so it can be re-attached to the next
// device after a replug.
if (bounce.handle) |handle| {
if (!device.attach(handle)) {
_ = ipc.close(device.endpoint);
return;
}
}
const geometry = device.geometry() orelse {
_ = ipc.close(device.endpoint);
return;
};
if (!device.read(0, 1, bounce.physical)) {
_ = ipc.close(device.endpoint);
return;
}
const sector: [*]const u8 = @ptrFromInt(bounce.virtual);
const found = partition.firstVolume(sector[0..512], geometry.block_count) orelse {
if (!logged_no_volume) {
_ = logging.write("volume-manager: storage present but no recognizable volume\n");
logged_no_volume = true;
}
_ = ipc.close(device.endpoint);
return;
};
logged_no_volume = false;
fs_restarts = 0;
fs_failed = false;
restart_pending = false;
volume = .{ .storage = device, .storage_device_id = opened.device_id, .base_lba = found.base_lba, .block_count = found.block_count, .identity = found.identity, .id = volume_id };
spawnFilesystem(&volume.?);
}
/// The storage provider left the device tree (a pulled stick): kill the
/// filesystem so its mounts are retired (the kernel sweeps a dead backend's
/// mounts), drop the now-dead channel, and clear the volume. The next poll that
/// sees storage return will re-mount.
fn removeVolume() void {
const v = volume orelse return;
std.log.info("storage for volume {d} removed; unmounting", .{v.id});
if (v.filesystem_pid != 0) _ = process.kill(v.filesystem_pid);
_ = ipc.close(v.storage.endpoint);
volume = null;
restart_pending = false;
fs_restarts = 0;
fs_failed = false;
}
/// One poll tick: perform a due restart, else reconcile presence — mount a newly
/// present volume, unmount a departed one.
fn pollTick() void {
if (restart_pending and time.clock() >= restart_due_ns) {
restart_pending = false;
if (volume) |*v| spawnFilesystem(v);
return;
}
if (volume) |v| {
// Serving: watch for the specific device leaving (a pulled stick).
if (!isDevicePresent(v.storage_device_id)) removeVolume();
} else {
// Idle: try to bring a present storage device up.
bringUpVolume();
}
}
/// A filesystem announces itself for the volume it was spawned to serve. Reply
/// with that volume's block channel (already range-confined to this filesystem's
/// badge) as the call's returned capability. No channel means the volume is not
/// ready — the filesystem retries.
fn onHello(_: void, invocation: Invocation(volume_manager_protocol.Hello), _: Answer(void)) isize {
const v = volume orelse return 0; // not probed yet — retryable, no cap
if (invocation.target != v.id) return 0; // unknown volume — retryable
if (invocation.sender != v.filesystem_pid) {
// Not the filesystem we spawned for this volume. Refuse: only the
// confined filesystem gets the channel.
std.log.info("refused hello for volume {d} from process {d}", .{ invocation.target, invocation.sender });
return -envelope.EPERM;
}
service.replyWithCapability(v.storage.endpoint);
std.log.info("handed volume {d} to pid {d}", .{ v.id, invocation.sender });
return 0;
}
const handlers = Serve.Handlers{ .hello = onHello };
fn onMessage(message: []const u8, out: []u8, sender: u32, arrived: *ipc.Arrival) usize {
// No verb takes a capability up, so the turn closes whatever arrives.
return Serve.dispatch({}, handlers, message, sender, arrived.peek(), out);
}
fn initialise(endpoint: ipc.Handle) bool {
service_endpoint = endpoint;
_ = logging.write("volume-manager: starting, waiting for a storage device\n");
_ = process.subscribeExits(endpoint);
pollTick();
_ = time.timerOnce(endpoint, poll_interval_ms); // the poll runs for the life of the boot
return true;
}
fn onNotification(badge: u64) void {
const got = ipc.Received{ .len = 0, .badge = badge, .cap = null };
if (got.isTimer()) {
pollTick();
_ = time.timerOnce(service_endpoint, poll_interval_ms); // always re-arm: presence is watched continuously
return;
}
// A filesystem died. The exit reason drives the decision, exactly as the
// device manager supervises drivers: a clean exit meant to stop; a fault
// restarts with backoff until a fast crash loop gives up. The old range is
// reclaimed by the driver on the same death; the respawn confines afresh.
if (got.isChildExit()) {
const dead = got.childProcessId();
const v = &(volume orelse return);
if (v.filesystem_pid != dead) return;
v.filesystem_pid = 0;
const reason = process.exitReason(dead) orelse .fault;
if (reason == .exited) {
std.log.info("filesystem for volume {d} exited cleanly; not restarting", .{v.id});
return;
}
const alive = time.clock() -| fs_spawn_ns;
fs_restarts = if (alive < fast_death_ns) fs_restarts + 1 else 1;
if (fs_restarts >= crash_loop_cap) {
fs_failed = true;
std.log.info("filesystem for volume {d} is failing repeatedly; giving up", .{v.id});
return;
}
std.log.info("filesystem for volume {d} died ({s}); restarting", .{ v.id, @tagName(reason) });
armRestart();
}
}
pub fn main(init: process.Init) void {
_ = init;
service.run(volume_manager_protocol.message_maximum, .{
.service = "volume-manager",
.init = initialise,
.on_message = onMessage,
.on_notification = onNotification,
});
}