//! system/services/volume-manager — the storage layer's policy home //! (docs/file-system-development/storage-architecture.md). Beside the device //! manager: that owns the DEVICE tree, this owns the VOLUME layer. It probes a //! storage provider's partition table, confines each filesystem to its //! partition, spawns one filesystem per volume, and answers that filesystem's //! startup hello with the range-confined block channel — so the filesystem //! never finds its storage by name and never sees the whole device. It //! supervises the filesystems it spawns, exactly as the device manager //! supervises drivers. //! //! This increment (V3b) is the flip: the FAT service stops acquiring its own //! volume and is spawned here instead, confined to its partition, and handed //! its channel over the volume-manager protocol. Single volume for now; the //! mount map (volumes.csv) and multi-volume land next. const std = @import("std"); const channel = @import("channel"); const device_manager_protocol = @import("device-manager-protocol"); const volume_manager_protocol = @import("volume-manager-protocol"); const driver = @import("driver"); const ipc = @import("ipc"); const block = @import("block"); const memory = @import("memory"); const logging = @import("logging"); const process = @import("process"); const service = @import("service"); const time = @import("time"); const envelope = @import("envelope"); const partition = @import("partition.zig"); const Serve = volume_manager_protocol.Protocol.Provider(void); const Invocation = envelope.Invocation; const Answer = envelope.Answer; /// The single volume this increment handles: its provider channel, its block /// sub-range, its identity, the id it is addressed by, and the filesystem /// process serving it (0 until spawned; reset on death for respawn). const Volume = struct { storage: block.Device, storage_device_id: u64, // the device-manager id this volume's provider serves base_lba: u64, block_count: u64, identity: partition.Identity, id: u64, filesystem_pid: u32 = 0, }; /// The filesystem binary a probed volume is served by. The signature->binary /// map (filesystems.csv) lands with the identity ladder; for now every FAT-shaped /// volume gets the FAT service. const filesystem_binary = "/system/services/fat"; const volume_id: u64 = 1; var service_endpoint: ipc.Handle = 0; var manager_handle: ?ipc.Handle = null; var bounce: memory.DmaRegion = undefined; var bounce_ready = false; /// The currently-mounted volume, or null while no storage is present. The whole /// removal lifecycle is this field going null and back: the poll sees the /// storage provider leave the device tree (a pulled stick), kills the filesystem /// and clears this; when it returns, the poll re-acquires and re-mounts. var volume: ?Volume = null; var logged_no_volume = false; /// How often the poll checks whether the storage provider is present. Fast /// enough that an unplug unmounts promptly; the poll is a bare device-manager /// enumerate, no channel work, so it is cheap to run continuously. const poll_interval_ms = 500; // Filesystem supervision, mirroring the device manager's (device-manager.zig): // a clean exit is not restarted, a fault restarts with backoff, and a fast // crash loop gives up rather than spinning. Without this a faulting filesystem // respawns in a zero-delay loop. const fast_death_ns: u64 = 2_000_000_000; const crash_loop_cap: u32 = 3; const backoff_base_ms: u64 = 300; var fs_restarts: u32 = 0; var fs_spawn_ns: u64 = 0; var fs_failed = false; /// A fat restart is due at `restart_due_ns`; the poll loop performs it once the /// backoff has elapsed (one timer, folded into the poll — no second timer). var restart_pending = false; var restart_due_ns: u64 = 0; fn deviceManager() ?ipc.Handle { if (manager_handle) |h| return h; const handle = channel.openEndpoint("device-manager") orelse return null; manager_handle = handle; return handle; } const OpenedStorage = struct { device_id: u64, device: block.Device }; /// The first mass-storage provider whose block channel actually opens, with its /// device id. A device-manager tree can carry more than one entry of the /// mass-storage identity — a phantom that no driver is bound to answers a /// consumer hello with NO channel — so this tries each and takes the first that /// yields a channel, exactly as a filesystem's own acquisition loop does. /// Called only when there is no volume (an insertion), so the hellos it makes /// are not per-poll churn. fn openAnyStorage() ?OpenedStorage { const manager = deviceManager() orelse return null; const Entry = device_manager_protocol.ChildEntry; var start: u64 = 0; while (true) { const enumerate = envelope.Header{ .operation = envelope.operation_enumerate, .target = start }; var reply: [device_manager_protocol.message_maximum]u8 = undefined; const length = ipc.call(manager, std.mem.asBytes(&enumerate), &reply) catch return null; const status = envelope.statusOf(reply[0..length]) orelse return null; if (status.status != 0) return null; const carried = @min(@as(usize, status.len), length -| envelope.prefix_size); const tail = reply[envelope.prefix_size..][0..carried]; const count = tail.len / @sizeOf(Entry); if (count == 0) return null; var index: usize = 0; while (index < count) : (index += 1) { const entry = std.mem.bytesToValue(Entry, tail[index * @sizeOf(Entry) ..][0..@sizeOf(Entry)]); if (entry.device_id == device_manager_protocol.no_device) continue; if ((entry.identity >> 16) & 0xff != 0x08 or (entry.identity >> 8) & 0xff != 0x06) continue; const exchanged = driver.helloOn(manager, .consumer, entry.device_id, null, true) orelse continue; const provider = exchanged.channel orelse continue; // a phantom / not-yet-bound entry return .{ .device_id = entry.device_id, .device = .{ .endpoint = provider } }; } start += count; } } /// Whether `device_id` is still in the device-manager tree — a bare enumerate, /// no consumer-hello, so it is cheap to call every poll. This is how removal is /// detected: the specific device the mounted volume sits on disappears. fn isDevicePresent(device_id: u64) bool { const manager = deviceManager() orelse return false; const Entry = device_manager_protocol.ChildEntry; var start: u64 = 0; while (true) { const enumerate = envelope.Header{ .operation = envelope.operation_enumerate, .target = start }; var reply: [device_manager_protocol.message_maximum]u8 = undefined; const length = ipc.call(manager, std.mem.asBytes(&enumerate), &reply) catch return false; const status = envelope.statusOf(reply[0..length]) orelse return false; if (status.status != 0) return false; const carried = @min(@as(usize, status.len), length -| envelope.prefix_size); const tail = reply[envelope.prefix_size..][0..carried]; const count = tail.len / @sizeOf(Entry); if (count == 0) return false; var index: usize = 0; while (index < count) : (index += 1) { const entry = std.mem.bytesToValue(Entry, tail[index * @sizeOf(Entry) ..][0..@sizeOf(Entry)]); if (entry.device_id == device_id) return true; } start += count; } } /// Spawn the filesystem for `v`, confine it to the volume's range, and record /// its pid. The confinement is defined for the fresh pid BEFORE the filesystem /// runs, so its first read is already bounded; the volume manager is the /// confinement controller (it defines the first range on the device). fn spawnFilesystem(v: *Volume) void { if (fs_failed) return; const pid = process.spawnSupervised(filesystem_binary, &.{"1"}, service_endpoint) orelse { _ = logging.write("volume-manager: could not spawn the filesystem; retrying\n"); armRestart(); return; }; if (!v.storage.defineRange(pid, v.base_lba, v.block_count)) { _ = logging.write("volume-manager: could not confine the filesystem to its volume; retrying\n"); _ = process.kill(pid); armRestart(); return; } v.filesystem_pid = pid; fs_spawn_ns = time.clock(); std.log.info("volume 0x{x} -> {s} (pid {d}), lba {d}, {d} blocks", .{ v.identity.key, filesystem_binary, pid, v.base_lba, v.block_count }); } /// Schedule a fat restart after backoff; the poll loop performs it once due. fn armRestart() void { const delay = if (fs_restarts == 0) backoff_base_ms else backoff_base_ms << @intCast(@min(fs_restarts - 1, 5)); restart_due_ns = time.clock() + delay * 1_000_000; restart_pending = true; } /// A storage provider just appeared: open its channel, read block 0, parse the /// volume, and spawn its filesystem. On any failure the channel is closed (so a /// present-but-unreadable device does not leak a handle every poll) and `volume` /// stays null — the next poll retries. A fresh medium gets a fresh supervision /// budget. fn bringUpVolume() void { if (!bounce_ready) { bounce = memory.dmaAlloc(512, memory.dma_coherent | memory.dma_shareable) orelse return; bounce_ready = true; } const opened = openAnyStorage() orelse return; const device = opened.device; // Attach the read buffer to THIS device (a no-op without an enforcing IOMMU). // The handle is kept, not closed, so it can be re-attached to the next // device after a replug. if (bounce.handle) |handle| { if (!device.attach(handle)) { _ = ipc.close(device.endpoint); return; } } const geometry = device.geometry() orelse { _ = ipc.close(device.endpoint); return; }; const ProbeReader = struct { device: block.Device, fn readSector(context: *anyopaque, lba: u64, buffer: *[partition.sector_bytes]u8) bool { const self: *@This() = @ptrCast(@alignCast(context)); if (!self.device.read(lba, 1, bounce.physical)) return false; const src: [*]const u8 = @ptrFromInt(bounce.virtual); @memcpy(buffer, src[0..partition.sector_bytes]); return true; } }; var probe = ProbeReader{ .device = device }; const reader = partition.SectorReader{ .context = &probe, .readFn = ProbeReader.readSector }; const found = partition.firstVolume(reader, geometry.block_count) orelse { if (!logged_no_volume) { _ = logging.write("volume-manager: storage present but no recognizable volume\n"); logged_no_volume = true; } _ = ipc.close(device.endpoint); return; }; logged_no_volume = false; fs_restarts = 0; fs_failed = false; restart_pending = false; volume = .{ .storage = device, .storage_device_id = opened.device_id, .base_lba = found.base_lba, .block_count = found.block_count, .identity = found.identity, .id = volume_id }; spawnFilesystem(&volume.?); } /// The storage provider left the device tree (a pulled stick): kill the /// filesystem so its mounts are retired. Retirement is lazy, not an eager /// death-time sweep — killing the process marks the filesystem's backend /// endpoint dead, and the VFS router drops each mount that endpoint backed on /// the next path resolution under it (that resolve frees the slot and returns /// not_found). Then drop the now-dead channel and clear the volume; the next /// poll that sees storage return re-mounts. fn removeVolume() void { const v = volume orelse return; std.log.info("storage for volume {d} removed; unmounting", .{v.id}); if (v.filesystem_pid != 0) _ = process.kill(v.filesystem_pid); _ = ipc.close(v.storage.endpoint); volume = null; restart_pending = false; fs_restarts = 0; fs_failed = false; } /// One poll tick. Removal is checked FIRST and supersedes a pending restart: if /// the device is gone there is nothing to restart fat onto, and respawning it /// against the dead channel would just churn until the crash cap. Only once the /// device is confirmed present does a due restart fire. fn pollTick() void { if (volume) |v| { // Serving: watch for the specific device leaving (a pulled stick). if (!isDevicePresent(v.storage_device_id)) { removeVolume(); return; } if (restart_pending and time.clock() >= restart_due_ns) { restart_pending = false; spawnFilesystem(&volume.?); } } else { // Idle: try to bring a present storage device up. bringUpVolume(); } } /// A filesystem announces itself for the volume it was spawned to serve. Reply /// with that volume's block channel (already range-confined to this filesystem's /// badge) as the call's returned capability. No channel means the volume is not /// ready — the filesystem retries. fn onHello(_: void, invocation: Invocation(volume_manager_protocol.Hello), _: Answer(void)) isize { const v = volume orelse return 0; // not probed yet — retryable, no cap if (invocation.target != v.id) return 0; // unknown volume — retryable if (invocation.sender != v.filesystem_pid) { // Not the filesystem we spawned for this volume. Refuse: only the // confined filesystem gets the channel. std.log.info("refused hello for volume {d} from process {d}", .{ invocation.target, invocation.sender }); return -envelope.EPERM; } service.replyWithCapability(v.storage.endpoint); std.log.info("handed volume {d} to pid {d}", .{ v.id, invocation.sender }); return 0; } const handlers = Serve.Handlers{ .hello = onHello }; fn onMessage(message: []const u8, out: []u8, sender: u32, arrived: *ipc.Arrival) usize { // No verb takes a capability up, so the turn closes whatever arrives. return Serve.dispatch({}, handlers, message, sender, arrived.peek(), out); } fn initialise(endpoint: ipc.Handle) bool { service_endpoint = endpoint; _ = logging.write("volume-manager: starting, waiting for a storage device\n"); _ = process.subscribeExits(endpoint); pollTick(); _ = time.timerOnce(endpoint, poll_interval_ms); // the poll runs for the life of the boot return true; } fn onNotification(badge: u64) void { const got = ipc.Received{ .len = 0, .badge = badge, .cap = null }; if (got.isTimer()) { pollTick(); _ = time.timerOnce(service_endpoint, poll_interval_ms); // always re-arm: presence is watched continuously return; } // A filesystem died. The exit reason drives the decision, exactly as the // device manager supervises drivers: a clean exit meant to stop; a fault // restarts with backoff until a fast crash loop gives up. The old range is // reclaimed by the driver on the same death; the respawn confines afresh. if (got.isChildExit()) { const dead = got.childProcessId(); const v = &(volume orelse return); if (v.filesystem_pid != dead) return; v.filesystem_pid = 0; const reason = process.exitReason(dead) orelse .fault; if (reason == .exited) { std.log.info("filesystem for volume {d} exited cleanly; not restarting", .{v.id}); return; } const alive = time.clock() -| fs_spawn_ns; fs_restarts = if (alive < fast_death_ns) fs_restarts + 1 else 1; if (fs_restarts >= crash_loop_cap) { fs_failed = true; std.log.info("filesystem for volume {d} is failing repeatedly; giving up", .{v.id}); return; } std.log.info("filesystem for volume {d} died ({s}); restarting", .{ v.id, @tagName(reason) }); armRestart(); } } pub fn main(init: process.Init) void { _ = init; service.run(volume_manager_protocol.message_maximum, .{ .service = "volume-manager", .init = initialise, .on_message = onMessage, .on_notification = onNotification, }); }