Files
danos/system/services/volume-manager/volume-manager.zig
T
Daniel Samson f1e79d0eeb volume-manager: the volumes query verb — read a volume's id, path, and label (S2)
The mechanism the id/label split needs: a `volumes` verb whose reply packs the
mounted volume's {id, mount_path, label} into the tail (VolumeInfo.encode/decode
— three length-prefixed strings). Software keys on the id (the mount path is
/volumes/<id>); a shell or file manager shows the label — the database id/name
split made a query. The VM's onVolumes answers from the mounted volume, empty
reply if none. Two host round-trip tests (encode/decode; too-small buffer and
short-tail rejection). No runtime consumer yet — the first is a userspace shell;
the hello handshake is unaffected (fat-mount/volume-probe green).
2026-08-10 00:17:22 +01:00

459 lines
21 KiB
Zig

//! system/services/volume-manager — the storage layer's policy home
//! (docs/file-system-development/storage-architecture.md). Beside the device
//! manager: that owns the DEVICE tree, this owns the VOLUME layer. It probes a
//! storage provider's partition table, confines each filesystem to its
//! partition, spawns one filesystem per volume, and answers that filesystem's
//! startup hello with the range-confined block channel — so the filesystem
//! never finds its storage by name and never sees the whole device. It
//! supervises the filesystems it spawns, exactly as the device manager
//! supervises drivers.
//!
//! This increment (V3b) is the flip: the FAT service stops acquiring its own
//! volume and is spawned here instead, confined to its partition, and handed
//! its channel over the volume-manager protocol. Single volume for now; the
//! mount map (volumes.csv) and multi-volume land next.
const std = @import("std");
const channel = @import("channel");
const device_manager_protocol = @import("device-manager-protocol");
const volume_manager_protocol = @import("volume-manager-protocol");
const driver = @import("driver");
const ipc = @import("ipc");
const block = @import("block");
const memory = @import("memory");
const logging = @import("logging");
const process = @import("process");
const service = @import("service");
const time = @import("time");
const envelope = @import("envelope");
const fs = @import("file-system");
const partition = @import("partition.zig");
const filesystem_map = @import("filesystem-map.zig");
const volume_map = @import("volume-map.zig");
const Serve = volume_manager_protocol.Protocol.Provider(void);
const Invocation = envelope.Invocation;
const Answer = envelope.Answer;
/// The single volume this increment handles: its provider channel, its block
/// sub-range, its identity, the id it is addressed by, and the filesystem
/// process serving it (0 until spawned; reset on death for respawn).
const Volume = struct {
storage: block.Device,
storage_device_id: u64, // the device-manager id this volume's provider serves
base_lba: u64,
block_count: u64,
identity: partition.Identity,
id: u64,
binary: []const u8, // the service binary, from filesystems.csv by signature
mount_prefix: []const u8, // the volume-root mount path (its id-path, or a volumes.csv override)
filesystem_pid: u32 = 0,
};
const volume_id: u64 = 1;
// The mount map, read from configuration at boot (the policy home, storage-
// architecture.md): filesystems.csv (content signature -> service binary) and
// volumes.csv (an optional id -> mount-prefix override). The sources are held
// for the process life so the parsed rules' slices into them stay valid.
/// bound: bytes of filesystems.csv / volumes.csv the manager reads
/// decided-by: ours
/// protects: the config source buffers below
/// at-limit: truncate - a longer file is cut; a row split by the cut is malformed
/// observed-by: the per-file "malformed/truncated" log line
const config_source_bytes = 2048;
var filesystems_source: [config_source_bytes]u8 = undefined;
var volumes_source: [config_source_bytes]u8 = undefined;
/// bound: filesystem-map rules held (one per content signature)
/// decided-by: ours
/// protects: the filesystem_rules table
/// at-limit: truncate - extra rows are dropped and the "truncated" note logged
/// observed-by: the "truncated" log line
const maximum_filesystem_rules = 8;
/// bound: volumes.csv override rows held (one per pinned volume id)
/// decided-by: ours
/// protects: the volume_rules table
/// at-limit: truncate - extra rows are dropped and the "truncated" note logged
/// observed-by: the "truncated" log line
const maximum_volume_rules = 64;
var filesystem_rules: [maximum_filesystem_rules]filesystem_map.Rule = undefined;
var filesystem_rule_count: usize = 0;
var volume_rules: [maximum_volume_rules]volume_map.Override = undefined;
var volume_rule_count: usize = 0;
/// The composed default mount path (/volumes/<id>) for the current volume; a
/// volumes.csv override is used in place and needs no buffer (it is already a
/// slice into volumes_source). One buffer suffices while the manager serves one
/// volume (multi-volume gives each its own in S3).
/// bound: bytes of a composed /volumes/<id> mount path
/// decided-by: ours
/// protects: the mount_prefix_buf below
/// at-limit: truncate - bufPrint fails; the volume mounts at a fallback path (logged)
/// observed-by: the fallback path in the log
const mount_path_maximum = 64;
var mount_prefix_buf: [mount_path_maximum]u8 = undefined;
var service_endpoint: ipc.Handle = 0;
var manager_handle: ?ipc.Handle = null;
var bounce: memory.DmaRegion = undefined;
var bounce_ready = false;
/// The currently-mounted volume, or null while no storage is present. The whole
/// removal lifecycle is this field going null and back: the poll sees the
/// storage provider leave the device tree (a pulled stick), kills the filesystem
/// and clears this; when it returns, the poll re-acquires and re-mounts.
var volume: ?Volume = null;
var logged_no_volume = false;
/// How often the poll checks whether the storage provider is present. Fast
/// enough that an unplug unmounts promptly; the poll is a bare device-manager
/// enumerate, no channel work, so it is cheap to run continuously.
const poll_interval_ms = 500;
// Filesystem supervision, mirroring the device manager's (device-manager.zig):
// a clean exit is not restarted, a fault restarts with backoff, and a fast
// crash loop gives up rather than spinning. Without this a faulting filesystem
// respawns in a zero-delay loop.
const fast_death_ns: u64 = 2_000_000_000;
const crash_loop_cap: u32 = 3;
const backoff_base_ms: u64 = 300;
var fs_restarts: u32 = 0;
var fs_spawn_ns: u64 = 0;
var fs_failed = false;
/// A fat restart is due at `restart_due_ns`; the poll loop performs it once the
/// backoff has elapsed (one timer, folded into the poll — no second timer).
var restart_pending = false;
var restart_due_ns: u64 = 0;
fn deviceManager() ?ipc.Handle {
if (manager_handle) |h| return h;
const handle = channel.openEndpoint("device-manager") orelse return null;
manager_handle = handle;
return handle;
}
const OpenedStorage = struct { device_id: u64, device: block.Device };
/// The first mass-storage provider whose block channel actually opens, with its
/// device id. A device-manager tree can carry more than one entry of the
/// mass-storage identity — a phantom that no driver is bound to answers a
/// consumer hello with NO channel — so this tries each and takes the first that
/// yields a channel, exactly as a filesystem's own acquisition loop does.
/// Called only when there is no volume (an insertion), so the hellos it makes
/// are not per-poll churn.
fn openAnyStorage() ?OpenedStorage {
const manager = deviceManager() orelse return null;
const Entry = device_manager_protocol.ChildEntry;
var start: u64 = 0;
while (true) {
const enumerate = envelope.Header{ .operation = envelope.operation_enumerate, .target = start };
var reply: [device_manager_protocol.message_maximum]u8 = undefined;
const length = ipc.call(manager, std.mem.asBytes(&enumerate), &reply) catch return null;
const status = envelope.statusOf(reply[0..length]) orelse return null;
if (status.status != 0) return null;
const carried = @min(@as(usize, status.len), length -| envelope.prefix_size);
const tail = reply[envelope.prefix_size..][0..carried];
const count = tail.len / @sizeOf(Entry);
if (count == 0) return null;
var index: usize = 0;
while (index < count) : (index += 1) {
const entry = std.mem.bytesToValue(Entry, tail[index * @sizeOf(Entry) ..][0..@sizeOf(Entry)]);
if (entry.device_id == device_manager_protocol.no_device) continue;
if ((entry.identity >> 16) & 0xff != 0x08 or (entry.identity >> 8) & 0xff != 0x06) continue;
const exchanged = driver.helloOn(manager, .consumer, entry.device_id, null, true) orelse continue;
const provider = exchanged.channel orelse continue; // a phantom / not-yet-bound entry
return .{ .device_id = entry.device_id, .device = .{ .endpoint = provider } };
}
start += count;
}
}
/// Whether `device_id` is still in the device-manager tree — a bare enumerate,
/// no consumer-hello, so it is cheap to call every poll. This is how removal is
/// detected: the specific device the mounted volume sits on disappears.
fn isDevicePresent(device_id: u64) bool {
const manager = deviceManager() orelse return false;
const Entry = device_manager_protocol.ChildEntry;
var start: u64 = 0;
while (true) {
const enumerate = envelope.Header{ .operation = envelope.operation_enumerate, .target = start };
var reply: [device_manager_protocol.message_maximum]u8 = undefined;
const length = ipc.call(manager, std.mem.asBytes(&enumerate), &reply) catch return false;
const status = envelope.statusOf(reply[0..length]) orelse return false;
if (status.status != 0) return false;
const carried = @min(@as(usize, status.len), length -| envelope.prefix_size);
const tail = reply[envelope.prefix_size..][0..carried];
const count = tail.len / @sizeOf(Entry);
if (count == 0) return false;
var index: usize = 0;
while (index < count) : (index += 1) {
const entry = std.mem.bytesToValue(Entry, tail[index * @sizeOf(Entry) ..][0..@sizeOf(Entry)]);
if (entry.device_id == device_id) return true;
}
start += count;
}
}
/// Spawn the filesystem for `v`, confine it to the volume's range, and record
/// its pid. The confinement is defined for the fresh pid BEFORE the filesystem
/// runs, so its first read is already bounded; the volume manager is the
/// confinement controller (it defines the first range on the device).
fn spawnFilesystem(v: *Volume) void {
if (fs_failed) return;
const pid = process.spawnSupervised(v.binary, &.{ "1", v.mount_prefix }, service_endpoint) orelse {
_ = logging.write("volume-manager: could not spawn the filesystem; retrying\n");
armRestart();
return;
};
if (!v.storage.defineRange(pid, v.base_lba, v.block_count)) {
_ = logging.write("volume-manager: could not confine the filesystem to its volume; retrying\n");
_ = process.kill(pid);
armRestart();
return;
}
v.filesystem_pid = pid;
fs_spawn_ns = time.clock();
std.log.info("volume 0x{x} -> {s} (pid {d}), lba {d}, {d} blocks", .{ v.identity.key, v.binary, pid, v.base_lba, v.block_count });
}
/// Schedule a fat restart after backoff; the poll loop performs it once due.
fn armRestart() void {
const delay = if (fs_restarts == 0) backoff_base_ms else backoff_base_ms << @intCast(@min(fs_restarts - 1, 5));
restart_due_ns = time.clock() + delay * 1_000_000;
restart_pending = true;
}
/// A storage provider just appeared: open its channel, read block 0, parse the
/// volume, and spawn its filesystem. On any failure the channel is closed (so a
/// present-but-unreadable device does not leak a handle every poll) and `volume`
/// stays null — the next poll retries. A fresh medium gets a fresh supervision
/// budget.
fn bringUpVolume() void {
if (!bounce_ready) {
bounce = memory.dmaAlloc(512, memory.dma_coherent | memory.dma_shareable) orelse return;
bounce_ready = true;
}
const opened = openAnyStorage() orelse return;
const device = opened.device;
// Attach the read buffer to THIS device (a no-op without an enforcing IOMMU).
// The handle is kept, not closed, so it can be re-attached to the next
// device after a replug.
if (bounce.handle) |handle| {
if (!device.attach(handle)) {
_ = ipc.close(device.endpoint);
return;
}
}
const geometry = device.geometry() orelse {
_ = ipc.close(device.endpoint);
return;
};
const ProbeReader = struct {
device: block.Device,
fn readSector(context: *anyopaque, lba: u64, buffer: *[partition.sector_bytes]u8) bool {
const self: *@This() = @ptrCast(@alignCast(context));
if (!self.device.read(lba, 1, bounce.physical)) return false;
const src: [*]const u8 = @ptrFromInt(bounce.virtual);
@memcpy(buffer, src[0..partition.sector_bytes]);
return true;
}
};
var probe = ProbeReader{ .device = device };
const reader = partition.SectorReader{ .context = &probe, .readFn = ProbeReader.readSector };
const found = partition.firstVolume(reader, geometry.block_count) orelse {
if (!logged_no_volume) {
_ = logging.write("volume-manager: storage present but no recognizable volume\n");
logged_no_volume = true;
}
_ = ipc.close(device.endpoint);
return;
};
// Pick the service binary from the volume's content signature. A signature
// no filesystems.csv row serves goes unserved (logged), like an unbound
// device — the manager does not guess.
const binary = filesystem_map.match(filesystem_rules[0..filesystem_rule_count], found.signature) orelse {
if (!logged_no_volume) {
_ = logging.write("volume-manager: no filesystem serves this volume's content; unserved\n");
logged_no_volume = true;
}
_ = ipc.close(device.endpoint);
return;
};
// The mount path is the volume's identity id (/volumes/<id>), or a
// volumes.csv override pinning it to a chosen path. The id is content-derived,
// so the path is stable and never a port or a label.
var id_buf: [volume_map.id_maximum]u8 = undefined;
const id = volume_map.idString(found.identity, &id_buf);
const mount_prefix = volume_map.overrideFor(volume_rules[0..volume_rule_count], id) orelse
(std.fmt.bufPrint(&mount_prefix_buf, "/volumes/{s}", .{id}) catch "/volumes/unknown");
logged_no_volume = false;
fs_restarts = 0;
fs_failed = false;
restart_pending = false;
volume = .{ .storage = device, .storage_device_id = opened.device_id, .base_lba = found.base_lba, .block_count = found.block_count, .identity = found.identity, .id = volume_id, .binary = binary, .mount_prefix = mount_prefix };
spawnFilesystem(&volume.?);
}
/// The storage provider left the device tree (a pulled stick): kill the
/// filesystem so its mounts are retired. Retirement is lazy, not an eager
/// death-time sweep — killing the process marks the filesystem's backend
/// endpoint dead, and the VFS router drops each mount that endpoint backed on
/// the next path resolution under it (that resolve frees the slot and returns
/// not_found). Then drop the now-dead channel and clear the volume; the next
/// poll that sees storage return re-mounts.
fn removeVolume() void {
const v = volume orelse return;
std.log.info("storage for volume {d} removed; unmounting", .{v.id});
if (v.filesystem_pid != 0) _ = process.kill(v.filesystem_pid);
_ = ipc.close(v.storage.endpoint);
volume = null;
restart_pending = false;
fs_restarts = 0;
fs_failed = false;
}
/// One poll tick. Removal is checked FIRST and supersedes a pending restart: if
/// the device is gone there is nothing to restart fat onto, and respawning it
/// against the dead channel would just churn until the crash cap. Only once the
/// device is confirmed present does a due restart fire.
fn pollTick() void {
if (volume) |v| {
// Serving: watch for the specific device leaving (a pulled stick).
if (!isDevicePresent(v.storage_device_id)) {
removeVolume();
return;
}
if (restart_pending and time.clock() >= restart_due_ns) {
restart_pending = false;
spawnFilesystem(&volume.?);
}
} else {
// Idle: try to bring a present storage device up.
bringUpVolume();
}
}
/// A filesystem announces itself for the volume it was spawned to serve. Reply
/// with that volume's block channel (already range-confined to this filesystem's
/// badge) as the call's returned capability. No channel means the volume is not
/// ready — the filesystem retries.
fn onHello(_: void, invocation: Invocation(volume_manager_protocol.Hello), _: Answer(void)) isize {
const v = volume orelse return 0; // not probed yet — retryable, no cap
if (invocation.target != v.id) return 0; // unknown volume — retryable
if (invocation.sender != v.filesystem_pid) {
// Not the filesystem we spawned for this volume. Refuse: only the
// confined filesystem gets the channel.
std.log.info("refused hello for volume {d} from process {d}", .{ invocation.target, invocation.sender });
return -envelope.EPERM;
}
service.replyWithCapability(v.storage.endpoint);
std.log.info("handed volume {d} to pid {d}", .{ v.id, invocation.sender });
return 0;
}
/// Answer a `volumes` query with the mounted volume's descriptor — its id (its
/// mount path is /volumes/<id> unless overridden), its actual mount path, and
/// its display label. This is how a shell or file manager reads a volume's
/// friendly name: software keys on the id, a UI shows the label. An empty reply
/// means no volume is mounted.
fn onVolumes(_: void, _: Invocation(volume_manager_protocol.Volumes), answer: Answer(void)) isize {
const v = volume orelse return 0;
var id_buf: [volume_map.id_maximum]u8 = undefined;
const info = volume_manager_protocol.VolumeInfo{
.id = volume_map.idString(v.identity, &id_buf),
.mount_path = v.mount_prefix,
.label = v.identity.labelSlice(),
};
const encoded = info.encode(answer.tail()) orelse return 0;
return @intCast(encoded.len);
}
const handlers = Serve.Handlers{ .hello = onHello, .volumes = onVolumes };
fn onMessage(message: []const u8, out: []u8, sender: u32, arrived: *ipc.Arrival) usize {
// No verb takes a capability up, so the turn closes whatever arrives.
return Serve.dispatch({}, handlers, message, sender, arrived.peek(), out);
}
/// Read a config file into `buf`, returning the byte count (0 if missing).
fn readConfig(path: []const u8, buf: []u8) usize {
var file = fs.open(path, .{}) orelse {
std.log.info("volume-manager: {s} missing", .{path});
return 0;
};
defer file.close();
var used: usize = 0;
while (used < buf.len) {
const n = file.read(buf[used..]) orelse break;
if (n == 0) break;
used += n;
}
return used;
}
/// Load the mount map from configuration once at boot (mirrors the device
/// manager's registry load). A missing or empty filesystems.csv means no volume
/// is served; volumes.csv is optional — no rows means every volume takes its
/// default /volumes/<id> path.
fn loadTables() void {
const fs_used = readConfig("/system/configuration/filesystems.csv", &filesystems_source);
const fr = filesystem_map.parse(filesystems_source[0..fs_used], &filesystem_rules);
filesystem_rule_count = fr.count;
if (fr.malformed != 0 or fr.truncated) std.log.info("filesystems.csv: {d} malformed, truncated={}", .{ fr.malformed, fr.truncated });
const vol_used = readConfig("/system/configuration/volumes.csv", &volumes_source);
const vr = volume_map.parse(volumes_source[0..vol_used], &volume_rules);
volume_rule_count = vr.count;
if (vr.malformed != 0 or vr.truncated) std.log.info("volumes.csv: {d} malformed, truncated={}", .{ vr.malformed, vr.truncated });
}
fn initialise(endpoint: ipc.Handle) bool {
service_endpoint = endpoint;
_ = logging.write("volume-manager: starting, waiting for a storage device\n");
loadTables();
_ = process.subscribeExits(endpoint);
pollTick();
_ = time.timerOnce(endpoint, poll_interval_ms); // the poll runs for the life of the boot
return true;
}
fn onNotification(badge: u64) void {
const got = ipc.Received{ .len = 0, .badge = badge, .cap = null };
if (got.isTimer()) {
pollTick();
_ = time.timerOnce(service_endpoint, poll_interval_ms); // always re-arm: presence is watched continuously
return;
}
// A filesystem died. The exit reason drives the decision, exactly as the
// device manager supervises drivers: a clean exit meant to stop; a fault
// restarts with backoff until a fast crash loop gives up. The old range is
// reclaimed by the driver on the same death; the respawn confines afresh.
if (got.isChildExit()) {
const dead = got.childProcessId();
const v = &(volume orelse return);
if (v.filesystem_pid != dead) return;
v.filesystem_pid = 0;
const reason = process.exitReason(dead) orelse .fault;
if (reason == .exited) {
std.log.info("filesystem for volume {d} exited cleanly; not restarting", .{v.id});
return;
}
const alive = time.clock() -| fs_spawn_ns;
fs_restarts = if (alive < fast_death_ns) fs_restarts + 1 else 1;
if (fs_restarts >= crash_loop_cap) {
fs_failed = true;
std.log.info("filesystem for volume {d} is failing repeatedly; giving up", .{v.id});
return;
}
std.log.info("filesystem for volume {d} died ({s}); restarting", .{ v.id, @tagName(reason) });
armRestart();
}
}
pub fn main(init: process.Init) void {
_ = init;
service.run(volume_manager_protocol.message_maximum, .{
.service = "volume-manager",
.init = initialise,
.on_message = onMessage,
.on_notification = onNotification,
});
}