//! system/services/volume-manager — the storage layer's policy home //! (docs/file-system-development/storage-architecture.md). Beside the device //! manager: that owns the DEVICE tree, this owns the VOLUME layer. It probes a //! storage provider's partition table, confines each filesystem to its //! partition, spawns one filesystem per volume, and answers that filesystem's //! startup hello with the range-confined block channel — so the filesystem //! never finds its storage by name and never sees the whole device. It //! supervises the filesystems it spawns, exactly as the device manager //! supervises drivers. //! //! The manager holds a table of adopted storage DEVICES and a table of the //! VOLUMES on them: it adopts every storage device the device-manager tree //! carries, probes each one's whole partition table, and spawns one filesystem //! process per volume — each confined to its partition's badge-scoped block //! range, each supervised with its own budget. A device leaving the tree takes //! its volumes with it. const std = @import("std"); const channel = @import("channel"); const device_manager_protocol = @import("device-manager-protocol"); const volume_manager_protocol = @import("volume-manager-protocol"); const driver = @import("driver"); const ipc = @import("ipc"); const block = @import("block"); const memory = @import("memory"); const logging = @import("logging"); const process = @import("process"); const service = @import("service"); const time = @import("time"); const envelope = @import("envelope"); const fs = @import("file-system"); const partition = @import("partition.zig"); const filesystem_map = @import("filesystem-map.zig"); const volume_map = @import("volume-map.zig"); const Serve = volume_manager_protocol.Protocol.Provider(void); const Invocation = envelope.Invocation; const Answer = envelope.Answer; /// One adopted storage device: the block channel to its provider (opened once and /// shared — refcounted per confined filesystem via the hello reply) and the /// device-manager id it serves. A device leaving the tree takes its volumes. const StorageDevice = struct { used: bool = false, device_id: u64 = 0, channel: block.Device = undefined, }; /// One volume: which device serves it, its block sub-range, its content /// identity, the id it is addressed by, the service binary + mount path it was /// spawned with, the filesystem process serving it, and its own supervision /// budget (so one volume's crash loop never touches another's). const Volume = struct { used: bool = false, device_id: u64 = 0, base_lba: u64 = 0, block_count: u64 = 0, identity: partition.Identity = .{ .rung = .anonymous }, id: u64 = 0, binary: []const u8 = "", mount_prefix: []const u8 = "", filesystem_pid: u32 = 0, // Per-volume supervision, mirroring the device manager's: a clean exit is not // restarted, a fault restarts with backoff, a fast crash loop gives up. restarts: u32 = 0, spawn_ns: u64 = 0, failed: bool = false, restart_pending: bool = false, restart_due_ns: u64 = 0, }; // The mount map, read from configuration at boot (the policy home, storage- // architecture.md): filesystems.csv (content signature -> service binary) and // volumes.csv (an optional id -> mount-prefix override). The sources are held // for the process life so the parsed rules' slices into them stay valid. /// bound: bytes of filesystems.csv / volumes.csv the manager reads /// decided-by: ours /// protects: the config source buffers below /// at-limit: truncate - a longer file is cut; a row split by the cut is malformed /// observed-by: the per-file "malformed/truncated" log line const config_source_bytes = 2048; var filesystems_source: [config_source_bytes]u8 = undefined; var volumes_source: [config_source_bytes]u8 = undefined; /// bound: filesystem-map rules held (one per content signature) /// decided-by: ours /// protects: the filesystem_rules table /// at-limit: truncate - extra rows are dropped and the "truncated" note logged /// observed-by: the "truncated" log line const maximum_filesystem_rules = 8; /// bound: volumes.csv override rows held (one per pinned volume id) /// decided-by: ours /// protects: the volume_rules table /// at-limit: truncate - extra rows are dropped and the "truncated" note logged /// observed-by: the "truncated" log line const maximum_volume_rules = 64; var filesystem_rules: [maximum_filesystem_rules]filesystem_map.Rule = undefined; var filesystem_rule_count: usize = 0; var volume_rules: [maximum_volume_rules]volume_map.Override = undefined; var volume_rule_count: usize = 0; /// bound: bytes of a composed /volumes/ mount path /// decided-by: ours /// protects: the per-volume mount_prefix buffers below /// at-limit: truncate - bufPrint fails; the volume mounts at a fallback path (logged) /// observed-by: the fallback path in the log const mount_path_maximum = 64; /// bound: volumes the manager serves at once /// decided-by: ours /// protects: the volumes table and its per-volume mount-path buffers /// at-limit: truncate - a further partition is left unserved and logged (real /// machines carry a handful of volumes, far under this) /// observed-by: the "volume table full" log line const maximum_volumes = 16; /// bound: storage devices the manager adopts at once /// decided-by: ours /// protects: the devices table /// at-limit: truncate - a further device is left unadopted and logged /// observed-by: the "device table full" log line const maximum_devices = 8; var devices = [_]StorageDevice{.{}} ** maximum_devices; var volumes = [_]Volume{.{}} ** maximum_volumes; /// Each volume's composed default mount path lives in its slot's buffer; a /// volumes.csv override is used in place (a slice into volumes_source, no buffer). var mount_prefix_bufs: [maximum_volumes][mount_path_maximum]u8 = undefined; var next_volume_id: u64 = 1; // monotonic — never reused, so a stale id can't address the wrong child var service_endpoint: ipc.Handle = 0; var manager_handle: ?ipc.Handle = null; var bounce: memory.DmaRegion = undefined; var bounce_ready = false; /// How often the poll checks device presence and fires due restarts. Fast enough /// that an unplug unmounts promptly; the poll is a bare device-manager enumerate, /// no channel work, so it is cheap to run continuously. const poll_interval_ms = 500; // Filesystem supervision, mirroring the device manager's (device-manager.zig). const fast_death_ns: u64 = 2_000_000_000; const crash_loop_cap: u32 = 3; const backoff_base_ms: u64 = 300; /// bytes to format a u64 volume id as decimal (20 digits fit) const id_decimal_bytes = 24; // --- table lookups ----------------------------------------------------------- fn deviceById(id: u64) ?*StorageDevice { for (&devices) |*d| if (d.used and d.device_id == id) return d; return null; } fn claimDevice() ?*StorageDevice { for (&devices) |*d| if (!d.used) return d; return null; } fn volumeById(id: u64) ?*Volume { for (&volumes) |*v| if (v.used and v.id == id) return v; return null; } fn volumeByPid(pid: u32) ?*Volume { for (&volumes) |*v| if (v.used and v.filesystem_pid == pid) return v; return null; } fn firstUsedVolume() ?*Volume { for (&volumes) |*v| if (v.used) return v; return null; } fn claimVolumeIndex() ?usize { for (&volumes, 0..) |*v, i| if (!v.used) return i; return null; } // --- device-manager plumbing ------------------------------------------------- fn deviceManager() ?ipc.Handle { if (manager_handle) |h| return h; const handle = channel.openEndpoint("device-manager") orelse return null; manager_handle = handle; return handle; } const OpenedStorage = struct { device_id: u64, device: block.Device }; /// The first mass-storage provider whose block channel opens and is NOT already /// adopted, with its device id. A device-manager tree can carry more than one /// entry of the mass-storage identity — a phantom that no driver is bound to /// answers a consumer hello with NO channel — so this tries each and takes the /// first that yields a channel. Skips already-adopted devices so a re-poll does /// not re-open a device it already serves. fn openAnyStorage() ?OpenedStorage { const manager = deviceManager() orelse return null; const Entry = device_manager_protocol.ChildEntry; var start: u64 = 0; while (true) { const enumerate = envelope.Header{ .operation = envelope.operation_enumerate, .target = start }; var reply: [device_manager_protocol.message_maximum]u8 = undefined; const length = ipc.call(manager, std.mem.asBytes(&enumerate), &reply) catch return null; const status = envelope.statusOf(reply[0..length]) orelse return null; if (status.status != 0) return null; const carried = @min(@as(usize, status.len), length -| envelope.prefix_size); const tail = reply[envelope.prefix_size..][0..carried]; const count = tail.len / @sizeOf(Entry); if (count == 0) return null; var index: usize = 0; while (index < count) : (index += 1) { const entry = std.mem.bytesToValue(Entry, tail[index * @sizeOf(Entry) ..][0..@sizeOf(Entry)]); if (entry.device_id == device_manager_protocol.no_device) continue; if ((entry.identity >> 16) & 0xff != 0x08 or (entry.identity >> 8) & 0xff != 0x06) continue; if (deviceById(entry.device_id) != null) continue; // already adopted const exchanged = driver.helloOn(manager, .consumer, entry.device_id, null, true) orelse continue; const provider = exchanged.channel orelse continue; // a phantom / not-yet-bound entry return .{ .device_id = entry.device_id, .device = .{ .endpoint = provider } }; } start += count; } } /// Whether `device_id` is still in the device-manager tree — a bare enumerate, /// no consumer-hello, so it is cheap to call every poll. This is how removal is /// detected: the specific device a mounted volume sits on disappears. fn isDevicePresent(device_id: u64) bool { const manager = deviceManager() orelse return false; const Entry = device_manager_protocol.ChildEntry; var start: u64 = 0; while (true) { const enumerate = envelope.Header{ .operation = envelope.operation_enumerate, .target = start }; var reply: [device_manager_protocol.message_maximum]u8 = undefined; const length = ipc.call(manager, std.mem.asBytes(&enumerate), &reply) catch return false; const status = envelope.statusOf(reply[0..length]) orelse return false; if (status.status != 0) return false; const carried = @min(@as(usize, status.len), length -| envelope.prefix_size); const tail = reply[envelope.prefix_size..][0..carried]; const count = tail.len / @sizeOf(Entry); if (count == 0) return false; var index: usize = 0; while (index < count) : (index += 1) { const entry = std.mem.bytesToValue(Entry, tail[index * @sizeOf(Entry) ..][0..@sizeOf(Entry)]); if (entry.device_id == device_id) return true; } start += count; } } // --- lifecycle --------------------------------------------------------------- /// Spawn the filesystem for `v`, confine it to the volume's range on its device's /// channel, and record its pid. The confinement is defined for the fresh pid /// BEFORE the filesystem runs, so its first read is already bounded; the volume /// manager is the confinement controller (it defines the first range on the /// device). fn spawnFilesystem(v: *Volume) void { if (v.failed) return; const dev = deviceById(v.device_id) orelse return; // its device left — poll will clean up var id_str_buf: [id_decimal_bytes]u8 = undefined; const id_str = std.fmt.bufPrint(&id_str_buf, "{d}", .{v.id}) catch "1"; const pid = process.spawnSupervised(v.binary, &.{ id_str, v.mount_prefix }, service_endpoint) orelse { _ = logging.write("volume-manager: could not spawn the filesystem; retrying\n"); armRestart(v); return; }; if (!dev.channel.defineRange(pid, v.base_lba, v.block_count)) { _ = logging.write("volume-manager: could not confine the filesystem to its volume; retrying\n"); _ = process.kill(pid); armRestart(v); return; } v.filesystem_pid = pid; v.spawn_ns = time.clock(); std.log.info("volume 0x{x} -> {s} (pid {d}), lba {d}, {d} blocks", .{ v.identity.key, v.binary, pid, v.base_lba, v.block_count }); } /// Schedule a restart for `v` after backoff; the poll loop performs it once due. fn armRestart(v: *Volume) void { const delay = if (v.restarts == 0) backoff_base_ms else backoff_base_ms << @intCast(@min(v.restarts - 1, 5)); v.restart_due_ns = time.clock() + delay * 1_000_000; v.restart_pending = true; } /// Compose a volume's mount path (its id-path `/volumes/`, or a volumes.csv /// override) into its slot's buffer, and return the slice. fn composeMountPrefix(slot: usize, identity: partition.Identity) []const u8 { var id_buf: [volume_map.id_maximum]u8 = undefined; const id = volume_map.idString(identity, &id_buf); return volume_map.overrideFor(volume_rules[0..volume_rule_count], id) orelse (std.fmt.bufPrint(&mount_prefix_bufs[slot], "/volumes/{s}", .{id}) catch "/volumes/unknown"); } /// Adopt the next present, not-yet-adopted storage device: take its channel, /// probe its whole partition table, and spawn a filesystem per volume it carries. /// Returns true when it consumed a device (so the caller can loop to adopt every /// present device in one tick), false when none remain or the device table is full. /// /// A device is adopted exactly once and kept until it leaves the tree — even when /// it carries no volume we can serve, or its geometry cannot be read. Keeping the /// empty/unreadable device adopted (rather than dropping and re-probing) is what /// lets openAnyStorage advance PAST it to the devices behind it; dropping it would /// make openAnyStorage hand back the same unservable device every tick and starve /// the rest. A genuine removal frees the slot (removeDevice); a re-insert gets a /// fresh device id and is probed anew. fn bringUpVolume() bool { if (!bounce_ready) { bounce = memory.dmaAlloc(512, memory.dma_coherent | memory.dma_shareable) orelse return false; bounce_ready = true; } const opened = openAnyStorage() orelse return false; const dev = claimDevice() orelse { _ = logging.write("volume-manager: device table full; a storage device is left unadopted\n"); _ = ipc.close(opened.device.endpoint); return false; }; dev.* = .{ .used = true, .device_id = opened.device_id, .channel = opened.device }; // Consume this device's medium_changed events (the second of the removal // lifecycle's two triggers: the device stays in the tree while its medium // leaves — a card reader, an eject). Best effort: a provider that never // publishes the event simply never wakes us, and device-pull is still caught // by the presence poll. _ = opened.device.subscribeMedium(service_endpoint); const device = opened.device; // Attach the read buffer to THIS device (a no-op without an enforcing IOMMU). // The handle is kept, not closed, so it can be re-attached after a replug. A // failed attach or geometry read leaves the device adopted but empty — we just // cannot read it, and the slot still watches it for removal. if (bounce.handle) |handle| { if (!device.attach(handle)) { _ = logging.write("volume-manager: could not attach the read buffer to a storage device; no volume served\n"); return true; } } const geometry = device.geometry() orelse { _ = logging.write("volume-manager: could not read a storage device's geometry; no volume served\n"); return true; }; const ProbeReader = struct { device: block.Device, fn readSector(context: *anyopaque, lba: u64, buffer: *[partition.sector_bytes]u8) bool { const self: *@This() = @ptrCast(@alignCast(context)); if (!self.device.read(lba, 1, bounce.physical)) return false; const src: [*]const u8 = @ptrFromInt(bounce.virtual); @memcpy(buffer, src[0..partition.sector_bytes]); return true; } }; var probe = ProbeReader{ .device = device }; const reader = partition.SectorReader{ .context = &probe, .readFn = ProbeReader.readSector }; var found: [maximum_volumes]partition.Volume = undefined; const n = partition.allVolumes(reader, geometry.block_count, found[0..]); if (n == 0) { std.log.info("device {d} present but carries no recognizable volume", .{dev.device_id}); return true; } for (found[0..n]) |fv| { // Pick the service binary from the volume's content signature. A signature // no filesystems.csv row serves goes unserved (logged), like an unbound // device — the manager does not guess. const binary = filesystem_map.match(filesystem_rules[0..filesystem_rule_count], fv.signature) orelse { _ = logging.write("volume-manager: no filesystem serves this volume's content; unserved\n"); continue; }; const slot = claimVolumeIndex() orelse { _ = logging.write("volume-manager: volume table full; a volume is left unserved\n"); break; }; volumes[slot] = .{ .used = true, .device_id = dev.device_id, .base_lba = fv.base_lba, .block_count = fv.block_count, .identity = fv.identity, .id = next_volume_id, .binary = binary, .mount_prefix = composeMountPrefix(slot, fv.identity), }; next_volume_id += 1; spawnFilesystem(&volumes[slot]); } return true; } /// Close a device's channel and free its slot. No volumes are touched (the caller /// ensures none remain, or there never were any). fn dropDevice(dev: *StorageDevice) void { // Free the driver's subscriber slot before the channel closes. On a still-live // channel (a medium eject) this frees the slot; on a dead one (a device pull) // the call fails fast and the exit sweep frees it anyway. _ = dev.channel.unsubscribeMedium(); _ = ipc.close(dev.channel.endpoint); dev.* = .{}; } /// Retire one volume: kill its filesystem so its mounts are retired. Retirement /// is lazy, not an eager death-time sweep — killing the process marks the /// filesystem's backend endpoint dead, and the VFS router drops each mount that /// endpoint backed on the next path resolution under it (that resolve frees the /// slot and returns not_found). Then free the volume slot. fn removeVolumeState(v: *Volume) void { std.log.info("storage for volume {d} removed; unmounting", .{v.id}); if (v.filesystem_pid != 0) _ = process.kill(v.filesystem_pid); v.* = .{}; } /// A storage device left the tree (a pulled stick): retire every volume it served /// and drop its channel. One removal path, whether the device is pulled cleanly /// or vanishes. fn removeDevice(dev: *StorageDevice) void { for (&volumes) |*v| { if (v.used and v.device_id == dev.device_id) removeVolumeState(v); } dropDevice(dev); } /// One poll tick. Device removal is reconciled FIRST and supersedes a pending /// restart: a volume whose device left is retired before its restart could fire, /// so nothing respawns against a dead channel. Then due restarts fire for present /// volumes; then, if no device is adopted, a present device is brought up. fn pollTick() void { for (&devices) |*dev| { if (dev.used and !isDevicePresent(dev.device_id)) removeDevice(dev); } for (&volumes) |*v| { if (v.used and v.restart_pending and time.clock() >= v.restart_due_ns) { v.restart_pending = false; spawnFilesystem(v); } } // Adopt every present, not-yet-adopted storage device. Each call consumes at // most one device (openAnyStorage skips the adopted), so the loop terminates // once none remain; the maximum_devices guard is insurance against a logic // slip, never the normal exit. var adopted: usize = 0; while (adopted < maximum_devices and bringUpVolume()) : (adopted += 1) {} } /// A filesystem announces itself for the volume it was spawned to serve. Reply /// with that volume's block channel (already range-confined to this filesystem's /// badge) as the call's returned capability. No channel means the volume is not /// ready — the filesystem retries. fn onHello(_: void, invocation: Invocation(volume_manager_protocol.Hello), _: Answer(void)) isize { const v = volumeById(invocation.target) orelse return 0; // not probed yet — retryable, no cap if (invocation.sender != v.filesystem_pid) { // Not the filesystem we spawned for this volume. Refuse: only the confined // filesystem gets the channel. std.log.info("refused hello for volume {d} from process {d}", .{ invocation.target, invocation.sender }); return -envelope.EPERM; } const dev = deviceById(v.device_id) orelse return 0; // its device left — retryable service.replyWithCapability(dev.channel.endpoint); std.log.info("handed volume {d} to pid {d}", .{ v.id, invocation.sender }); return 0; } /// Answer a `volumes` query with a mounted volume's descriptor — its id (its /// mount path is /volumes/ unless overridden), its actual mount path, and its /// display label. Software keys on the id; a UI shows the label. Returns the first /// mounted volume for now; a full enumerate is a later refinement. Empty reply /// means no volume is mounted. fn onVolumes(_: void, _: Invocation(volume_manager_protocol.Volumes), answer: Answer(void)) isize { const v = firstUsedVolume() orelse return 0; var id_buf: [volume_map.id_maximum]u8 = undefined; const info = volume_manager_protocol.VolumeInfo{ .id = volume_map.idString(v.identity, &id_buf), .mount_path = v.mount_prefix, .label = v.identity.labelSlice(), }; const encoded = info.encode(answer.tail()) orelse return 0; return @intCast(encoded.len); } const handlers = Serve.Handlers{ .hello = onHello, .volumes = onVolumes }; fn onMessage(message: []const u8, out: []u8, sender: u32, arrived: *ipc.Arrival) usize { // No verb takes a capability up, so the turn closes whatever arrives. return Serve.dispatch({}, handlers, message, sender, arrived.peek(), out); } /// Read a config file into `buf`, returning the byte count (0 if missing). fn readConfig(path: []const u8, buf: []u8) usize { var file = fs.open(path, .{}) orelse { std.log.info("volume-manager: {s} missing", .{path}); return 0; }; defer file.close(); var used: usize = 0; while (used < buf.len) { const nn = file.read(buf[used..]) orelse break; if (nn == 0) break; used += nn; } return used; } /// Load the mount map from configuration once at boot (mirrors the device /// manager's registry load). A missing or empty filesystems.csv means no volume /// is served; volumes.csv is optional — no rows means every volume takes its /// default /volumes/ path. fn loadTables() void { const fs_used = readConfig("/system/configuration/filesystems.csv", &filesystems_source); const fr = filesystem_map.parse(filesystems_source[0..fs_used], &filesystem_rules); filesystem_rule_count = fr.count; if (fr.malformed != 0 or fr.truncated) std.log.info("filesystems.csv: {d} malformed, truncated={}", .{ fr.malformed, fr.truncated }); const vol_used = readConfig("/system/configuration/volumes.csv", &volumes_source); const vr = volume_map.parse(volumes_source[0..vol_used], &volume_rules); volume_rule_count = vr.count; if (vr.malformed != 0 or vr.truncated) std.log.info("volumes.csv: {d} malformed, truncated={}", .{ vr.malformed, vr.truncated }); } fn initialise(endpoint: ipc.Handle) bool { service_endpoint = endpoint; _ = logging.write("volume-manager: starting, waiting for a storage device\n"); loadTables(); _ = process.subscribeExits(endpoint); pollTick(); _ = time.timerOnce(endpoint, poll_interval_ms); // the poll runs for the life of the boot return true; } fn onNotification(badge: u64) void { const got = ipc.Received{ .len = 0, .badge = badge, .cap = null }; if (got.isTimer()) { pollTick(); _ = time.timerOnce(service_endpoint, poll_interval_ms); // always re-arm: presence is watched continuously return; } // A filesystem died. The exit reason drives the decision, exactly as the // device manager supervises drivers: a clean exit meant to stop; a fault // restarts with backoff until a fast crash loop gives up. The old range is // reclaimed by the driver on the same death; the respawn confines afresh. if (got.isChildExit()) { const dead = got.childProcessId(); const v = volumeByPid(dead) orelse return; v.filesystem_pid = 0; const reason = process.exitReason(dead) orelse .fault; if (reason == .exited) { std.log.info("filesystem for volume {d} exited cleanly; not restarting", .{v.id}); return; } const alive = time.clock() -| v.spawn_ns; v.restarts = if (alive < fast_death_ns) v.restarts + 1 else 1; if (v.restarts >= crash_loop_cap) { v.failed = true; std.log.info("filesystem for volume {d} is failing repeatedly; giving up", .{v.id}); return; } std.log.info("filesystem for volume {d} died ({s}); restarting", .{ v.id, @tagName(reason) }); armRestart(v); } } var last_medium_change: u32 = 0; fn anyVolumeOn(device_id: u64) bool { for (&volumes) |*v| if (v.used and v.device_id == device_id) return true; return false; } /// A storage device published `medium_changed` — the second removal trigger: the /// device stays in the tree while its medium leaves or returns (a card reader, an /// eject). This arrives as a buffered async message, NOT a protocol request, so it /// never reaches `Serve.dispatch` (its event op number collides with the manager's /// own `hello`); it is decoded here by hand. Single-volume scope: the event names /// no device, so `absent` retires every adopted device (its volumes unmount and /// the poll re-adopts the still-present device with its now-empty medium), and /// `present` frees any empty adopted device so the poll re-probes and remounts it. fn onMediumEvent(payload: []const u8) void { const event = block.decodeMediumChanged(payload) orelse return; if (event.change_count == last_medium_change) return; // a coalesced or re-delivered edge last_medium_change = event.change_count; if (event.present == 0) { std.log.info("medium left a storage device; unmounting its volume(s)", .{}); for (&devices) |*dev| { if (dev.used) removeDevice(dev); } } else { for (&devices) |*dev| { if (dev.used and !anyVolumeOn(dev.device_id)) removeDevice(dev); } } } pub fn main(init: process.Init) void { _ = init; service.run(volume_manager_protocol.message_maximum, .{ .service = "volume-manager", .init = initialise, .on_message = onMessage, .on_notification = onNotification, .on_buffered_message = onMediumEvent, }); }