From 7efe7b72d8ad77dfb500c0d0dbeb6268cb7826ee Mon Sep 17 00:00:00 2001 From: Daniel Samson <12231216+daniel-samson@users.noreply.github.com> Date: Mon, 10 Aug 2026 01:13:52 +0100 Subject: [PATCH] volume-manager: N-volume device+volume tables, behavior-preserving (S3) Replace the single `var volume: ?Volume` and file-global supervision state with two fixed tables: devices[maximum_devices] owning each adopted block channel once, and volumes[maximum_volumes] each carrying its own identity, id, mount prefix, and supervision fields (restarts, spawn_ns, failed, restart_pending, restart_due_ns). A monotonic next_volume_id never reuses ids, so a stale hello can't address the wrong child. Lookups (deviceById, volumeById, volumeByPid, firstUsedVolume) and claims (claimDevice, claimVolumeIndex) replace the ad-hoc singletons. pollTick reconciles devices first (removeDevice drops their volumes), then per-volume restarts, then idle bring-up. This step stays one-device/one-volume on purpose: bringUpVolume adopts the first device and caps allVolumes to a single partition, so behavior is identical and the full suite stays 128/128. Uncapping and adopt-all land next. --- .../volume-manager/volume-manager.zig | 393 +++++++++++------- 1 file changed, 240 insertions(+), 153 deletions(-) diff --git a/system/services/volume-manager/volume-manager.zig b/system/services/volume-manager/volume-manager.zig index 9423e58..6917ac3 100644 --- a/system/services/volume-manager/volume-manager.zig +++ b/system/services/volume-manager/volume-manager.zig @@ -8,10 +8,13 @@ //! supervises the filesystems it spawns, exactly as the device manager //! supervises drivers. //! -//! This increment (V3b) is the flip: the FAT service stops acquiring its own -//! volume and is spawned here instead, confined to its partition, and handed -//! its channel over the volume-manager protocol. Single volume for now; the -//! mount map (volumes.csv) and multi-volume land next. +//! The manager holds a table of adopted storage DEVICES and a table of the +//! VOLUMES on them: one filesystem process per volume, each confined to its +//! partition's badge-scoped block range, each supervised with its own budget. A +//! device leaving the tree takes its volumes with it. (This S3 increment lays the +//! tables in; multi-device adoption and per-partition spawn land in the next +//! step — for now it adopts one device and its first volume, behavior-identical +//! to before.) const std = @import("std"); const channel = @import("channel"); @@ -35,22 +38,37 @@ const Serve = volume_manager_protocol.Protocol.Provider(void); const Invocation = envelope.Invocation; const Answer = envelope.Answer; -/// The single volume this increment handles: its provider channel, its block -/// sub-range, its identity, the id it is addressed by, and the filesystem -/// process serving it (0 until spawned; reset on death for respawn). -const Volume = struct { - storage: block.Device, - storage_device_id: u64, // the device-manager id this volume's provider serves - base_lba: u64, - block_count: u64, - identity: partition.Identity, - id: u64, - binary: []const u8, // the service binary, from filesystems.csv by signature - mount_prefix: []const u8, // the volume-root mount path (its id-path, or a volumes.csv override) - filesystem_pid: u32 = 0, +/// One adopted storage device: the block channel to its provider (opened once and +/// shared — refcounted per confined filesystem via the hello reply) and the +/// device-manager id it serves. A device leaving the tree takes its volumes. +const StorageDevice = struct { + used: bool = false, + device_id: u64 = 0, + channel: block.Device = undefined, }; -const volume_id: u64 = 1; +/// One volume: which device serves it, its block sub-range, its content +/// identity, the id it is addressed by, the service binary + mount path it was +/// spawned with, the filesystem process serving it, and its own supervision +/// budget (so one volume's crash loop never touches another's). +const Volume = struct { + used: bool = false, + device_id: u64 = 0, + base_lba: u64 = 0, + block_count: u64 = 0, + identity: partition.Identity = .{ .rung = .anonymous }, + id: u64 = 0, + binary: []const u8 = "", + mount_prefix: []const u8 = "", + filesystem_pid: u32 = 0, + // Per-volume supervision, mirroring the device manager's: a clean exit is not + // restarted, a fault restarts with backoff, a fast crash loop gives up. + restarts: u32 = 0, + spawn_ns: u64 = 0, + failed: bool = false, + restart_pending: bool = false, + restart_due_ns: u64 = 0, +}; // The mount map, read from configuration at boot (the policy home, storage- // architecture.md): filesystems.csv (content signature -> service binary) and @@ -80,47 +98,82 @@ var filesystem_rules: [maximum_filesystem_rules]filesystem_map.Rule = undefined; var filesystem_rule_count: usize = 0; var volume_rules: [maximum_volume_rules]volume_map.Override = undefined; var volume_rule_count: usize = 0; -/// The composed default mount path (/volumes/) for the current volume; a -/// volumes.csv override is used in place and needs no buffer (it is already a -/// slice into volumes_source). One buffer suffices while the manager serves one -/// volume (multi-volume gives each its own in S3). /// bound: bytes of a composed /volumes/ mount path /// decided-by: ours -/// protects: the mount_prefix_buf below +/// protects: the per-volume mount_prefix buffers below /// at-limit: truncate - bufPrint fails; the volume mounts at a fallback path (logged) /// observed-by: the fallback path in the log const mount_path_maximum = 64; -var mount_prefix_buf: [mount_path_maximum]u8 = undefined; + +/// bound: volumes the manager serves at once +/// decided-by: ours +/// protects: the volumes table and its per-volume mount-path buffers +/// at-limit: truncate - a further partition is left unserved and logged (real +/// machines carry a handful of volumes, far under this) +/// observed-by: the "volume table full" log line +const maximum_volumes = 16; +/// bound: storage devices the manager adopts at once +/// decided-by: ours +/// protects: the devices table +/// at-limit: truncate - a further device is left unadopted and logged +/// observed-by: the "device table full" log line +const maximum_devices = 8; +var devices = [_]StorageDevice{.{}} ** maximum_devices; +var volumes = [_]Volume{.{}} ** maximum_volumes; +/// Each volume's composed default mount path lives in its slot's buffer; a +/// volumes.csv override is used in place (a slice into volumes_source, no buffer). +var mount_prefix_bufs: [maximum_volumes][mount_path_maximum]u8 = undefined; +var next_volume_id: u64 = 1; // monotonic — never reused, so a stale id can't address the wrong child var service_endpoint: ipc.Handle = 0; var manager_handle: ?ipc.Handle = null; var bounce: memory.DmaRegion = undefined; var bounce_ready = false; -/// The currently-mounted volume, or null while no storage is present. The whole -/// removal lifecycle is this field going null and back: the poll sees the -/// storage provider leave the device tree (a pulled stick), kills the filesystem -/// and clears this; when it returns, the poll re-acquires and re-mounts. -var volume: ?Volume = null; var logged_no_volume = false; -/// How often the poll checks whether the storage provider is present. Fast -/// enough that an unplug unmounts promptly; the poll is a bare device-manager -/// enumerate, no channel work, so it is cheap to run continuously. +/// How often the poll checks device presence and fires due restarts. Fast enough +/// that an unplug unmounts promptly; the poll is a bare device-manager enumerate, +/// no channel work, so it is cheap to run continuously. const poll_interval_ms = 500; -// Filesystem supervision, mirroring the device manager's (device-manager.zig): -// a clean exit is not restarted, a fault restarts with backoff, and a fast -// crash loop gives up rather than spinning. Without this a faulting filesystem -// respawns in a zero-delay loop. +// Filesystem supervision, mirroring the device manager's (device-manager.zig). const fast_death_ns: u64 = 2_000_000_000; const crash_loop_cap: u32 = 3; const backoff_base_ms: u64 = 300; -var fs_restarts: u32 = 0; -var fs_spawn_ns: u64 = 0; -var fs_failed = false; -/// A fat restart is due at `restart_due_ns`; the poll loop performs it once the -/// backoff has elapsed (one timer, folded into the poll — no second timer). -var restart_pending = false; -var restart_due_ns: u64 = 0; +/// bytes to format a u64 volume id as decimal (20 digits fit) +const id_decimal_bytes = 24; + +// --- table lookups ----------------------------------------------------------- + +fn deviceById(id: u64) ?*StorageDevice { + for (&devices) |*d| if (d.used and d.device_id == id) return d; + return null; +} +fn claimDevice() ?*StorageDevice { + for (&devices) |*d| if (!d.used) return d; + return null; +} +fn anyDeviceUsed() bool { + for (&devices) |*d| if (d.used) return true; + return false; +} +fn volumeById(id: u64) ?*Volume { + for (&volumes) |*v| if (v.used and v.id == id) return v; + return null; +} +fn volumeByPid(pid: u32) ?*Volume { + for (&volumes) |*v| if (v.used and v.filesystem_pid == pid) return v; + return null; +} +fn firstUsedVolume() ?*Volume { + for (&volumes) |*v| if (v.used) return v; + return null; +} +fn claimVolumeIndex() ?usize { + for (&volumes, 0..) |*v, i| if (!v.used) return i; + return null; +} + +// --- device-manager plumbing ------------------------------------------------- fn deviceManager() ?ipc.Handle { if (manager_handle) |h| return h; @@ -131,13 +184,12 @@ fn deviceManager() ?ipc.Handle { const OpenedStorage = struct { device_id: u64, device: block.Device }; -/// The first mass-storage provider whose block channel actually opens, with its -/// device id. A device-manager tree can carry more than one entry of the -/// mass-storage identity — a phantom that no driver is bound to answers a -/// consumer hello with NO channel — so this tries each and takes the first that -/// yields a channel, exactly as a filesystem's own acquisition loop does. -/// Called only when there is no volume (an insertion), so the hellos it makes -/// are not per-poll churn. +/// The first mass-storage provider whose block channel opens and is NOT already +/// adopted, with its device id. A device-manager tree can carry more than one +/// entry of the mass-storage identity — a phantom that no driver is bound to +/// answers a consumer hello with NO channel — so this tries each and takes the +/// first that yields a channel. Skips already-adopted devices so a re-poll does +/// not re-open a device it already serves. fn openAnyStorage() ?OpenedStorage { const manager = deviceManager() orelse return null; const Entry = device_manager_protocol.ChildEntry; @@ -157,6 +209,7 @@ fn openAnyStorage() ?OpenedStorage { const entry = std.mem.bytesToValue(Entry, tail[index * @sizeOf(Entry) ..][0..@sizeOf(Entry)]); if (entry.device_id == device_manager_protocol.no_device) continue; if ((entry.identity >> 16) & 0xff != 0x08 or (entry.identity >> 8) & 0xff != 0x06) continue; + if (deviceById(entry.device_id) != null) continue; // already adopted const exchanged = driver.helloOn(manager, .consumer, entry.device_id, null, true) orelse continue; const provider = exchanged.channel orelse continue; // a phantom / not-yet-bound entry return .{ .device_id = entry.device_id, .device = .{ .endpoint = provider } }; @@ -167,7 +220,7 @@ fn openAnyStorage() ?OpenedStorage { /// Whether `device_id` is still in the device-manager tree — a bare enumerate, /// no consumer-hello, so it is cheap to call every poll. This is how removal is -/// detected: the specific device the mounted volume sits on disappears. +/// detected: the specific device a mounted volume sits on disappears. fn isDevicePresent(device_id: u64) bool { const manager = deviceManager() orelse return false; const Entry = device_manager_protocol.ChildEntry; @@ -191,58 +244,79 @@ fn isDevicePresent(device_id: u64) bool { } } -/// Spawn the filesystem for `v`, confine it to the volume's range, and record -/// its pid. The confinement is defined for the fresh pid BEFORE the filesystem -/// runs, so its first read is already bounded; the volume manager is the -/// confinement controller (it defines the first range on the device). +// --- lifecycle --------------------------------------------------------------- + +/// Spawn the filesystem for `v`, confine it to the volume's range on its device's +/// channel, and record its pid. The confinement is defined for the fresh pid +/// BEFORE the filesystem runs, so its first read is already bounded; the volume +/// manager is the confinement controller (it defines the first range on the +/// device). fn spawnFilesystem(v: *Volume) void { - if (fs_failed) return; - const pid = process.spawnSupervised(v.binary, &.{ "1", v.mount_prefix }, service_endpoint) orelse { + if (v.failed) return; + const dev = deviceById(v.device_id) orelse return; // its device left — poll will clean up + var id_str_buf: [id_decimal_bytes]u8 = undefined; + const id_str = std.fmt.bufPrint(&id_str_buf, "{d}", .{v.id}) catch "1"; + const pid = process.spawnSupervised(v.binary, &.{ id_str, v.mount_prefix }, service_endpoint) orelse { _ = logging.write("volume-manager: could not spawn the filesystem; retrying\n"); - armRestart(); + armRestart(v); return; }; - if (!v.storage.defineRange(pid, v.base_lba, v.block_count)) { + if (!dev.channel.defineRange(pid, v.base_lba, v.block_count)) { _ = logging.write("volume-manager: could not confine the filesystem to its volume; retrying\n"); _ = process.kill(pid); - armRestart(); + armRestart(v); return; } v.filesystem_pid = pid; - fs_spawn_ns = time.clock(); + v.spawn_ns = time.clock(); std.log.info("volume 0x{x} -> {s} (pid {d}), lba {d}, {d} blocks", .{ v.identity.key, v.binary, pid, v.base_lba, v.block_count }); } -/// Schedule a fat restart after backoff; the poll loop performs it once due. -fn armRestart() void { - const delay = if (fs_restarts == 0) backoff_base_ms else backoff_base_ms << @intCast(@min(fs_restarts - 1, 5)); - restart_due_ns = time.clock() + delay * 1_000_000; - restart_pending = true; +/// Schedule a restart for `v` after backoff; the poll loop performs it once due. +fn armRestart(v: *Volume) void { + const delay = if (v.restarts == 0) backoff_base_ms else backoff_base_ms << @intCast(@min(v.restarts - 1, 5)); + v.restart_due_ns = time.clock() + delay * 1_000_000; + v.restart_pending = true; } -/// A storage provider just appeared: open its channel, read block 0, parse the -/// volume, and spawn its filesystem. On any failure the channel is closed (so a -/// present-but-unreadable device does not leak a handle every poll) and `volume` -/// stays null — the next poll retries. A fresh medium gets a fresh supervision -/// budget. +/// Compose a volume's mount path (its id-path `/volumes/`, or a volumes.csv +/// override) into its slot's buffer, and return the slice. +fn composeMountPrefix(slot: usize, identity: partition.Identity) []const u8 { + var id_buf: [volume_map.id_maximum]u8 = undefined; + const id = volume_map.idString(identity, &id_buf); + return volume_map.overrideFor(volume_rules[0..volume_rule_count], id) orelse + (std.fmt.bufPrint(&mount_prefix_bufs[slot], "/volumes/{s}", .{id}) catch "/volumes/unknown"); +} + +/// A storage device appeared: adopt its channel, probe its partition table, and +/// spawn a filesystem per volume it carries. On any failure the channel is +/// dropped (so a present-but-unreadable device does not leak a handle every poll) +/// and the device stays unadopted — the next poll retries. This increment probes +/// only the first volume (behavior-identical to before); the next step lifts the +/// cap. fn bringUpVolume() void { if (!bounce_ready) { bounce = memory.dmaAlloc(512, memory.dma_coherent | memory.dma_shareable) orelse return; bounce_ready = true; } const opened = openAnyStorage() orelse return; + const dev = claimDevice() orelse { + _ = logging.write("volume-manager: device table full; a storage device is left unadopted\n"); + _ = ipc.close(opened.device.endpoint); + return; + }; + dev.* = .{ .used = true, .device_id = opened.device_id, .channel = opened.device }; const device = opened.device; // Attach the read buffer to THIS device (a no-op without an enforcing IOMMU). - // The handle is kept, not closed, so it can be re-attached to the next - // device after a replug. + // The handle is kept, not closed, so it can be re-attached after a replug. if (bounce.handle) |handle| { if (!device.attach(handle)) { - _ = ipc.close(device.endpoint); + dropDevice(dev); return; } } const geometry = device.geometry() orelse { - _ = ipc.close(device.endpoint); + dropDevice(dev); return; }; const ProbeReader = struct { @@ -257,77 +331,91 @@ fn bringUpVolume() void { }; var probe = ProbeReader{ .device = device }; const reader = partition.SectorReader{ .context = &probe, .readFn = ProbeReader.readSector }; - const found = partition.firstVolume(reader, geometry.block_count) orelse { + var found: [maximum_volumes]partition.Volume = undefined; + const n = partition.allVolumes(reader, geometry.block_count, found[0..1]); // cap 1 this step + if (n == 0) { if (!logged_no_volume) { _ = logging.write("volume-manager: storage present but no recognizable volume\n"); logged_no_volume = true; } - _ = ipc.close(device.endpoint); + dropDevice(dev); return; - }; - // Pick the service binary from the volume's content signature. A signature - // no filesystems.csv row serves goes unserved (logged), like an unbound - // device — the manager does not guess. - const binary = filesystem_map.match(filesystem_rules[0..filesystem_rule_count], found.signature) orelse { - if (!logged_no_volume) { - _ = logging.write("volume-manager: no filesystem serves this volume's content; unserved\n"); - logged_no_volume = true; - } - _ = ipc.close(device.endpoint); - return; - }; - // The mount path is the volume's identity id (/volumes/), or a - // volumes.csv override pinning it to a chosen path. The id is content-derived, - // so the path is stable and never a port or a label. - var id_buf: [volume_map.id_maximum]u8 = undefined; - const id = volume_map.idString(found.identity, &id_buf); - const mount_prefix = volume_map.overrideFor(volume_rules[0..volume_rule_count], id) orelse - (std.fmt.bufPrint(&mount_prefix_buf, "/volumes/{s}", .{id}) catch "/volumes/unknown"); + } logged_no_volume = false; - fs_restarts = 0; - fs_failed = false; - restart_pending = false; - volume = .{ .storage = device, .storage_device_id = opened.device_id, .base_lba = found.base_lba, .block_count = found.block_count, .identity = found.identity, .id = volume_id, .binary = binary, .mount_prefix = mount_prefix }; - spawnFilesystem(&volume.?); + var spawned = false; + for (found[0..n]) |fv| { + // Pick the service binary from the volume's content signature. A signature + // no filesystems.csv row serves goes unserved (logged), like an unbound + // device — the manager does not guess. + const binary = filesystem_map.match(filesystem_rules[0..filesystem_rule_count], fv.signature) orelse { + _ = logging.write("volume-manager: no filesystem serves this volume's content; unserved\n"); + continue; + }; + const slot = claimVolumeIndex() orelse { + _ = logging.write("volume-manager: volume table full; a volume is left unserved\n"); + break; + }; + volumes[slot] = .{ + .used = true, + .device_id = dev.device_id, + .base_lba = fv.base_lba, + .block_count = fv.block_count, + .identity = fv.identity, + .id = next_volume_id, + .binary = binary, + .mount_prefix = composeMountPrefix(slot, fv.identity), + }; + next_volume_id += 1; + spawnFilesystem(&volumes[slot]); + spawned = true; + } + // The device carried nothing we could serve — drop it so a re-poll retries. + if (!spawned) dropDevice(dev); } -/// The storage provider left the device tree (a pulled stick): kill the -/// filesystem so its mounts are retired. Retirement is lazy, not an eager -/// death-time sweep — killing the process marks the filesystem's backend -/// endpoint dead, and the VFS router drops each mount that endpoint backed on -/// the next path resolution under it (that resolve frees the slot and returns -/// not_found). Then drop the now-dead channel and clear the volume; the next -/// poll that sees storage return re-mounts. -fn removeVolume() void { - const v = volume orelse return; +/// Close a device's channel and free its slot. No volumes are touched (the caller +/// ensures none remain, or there never were any). +fn dropDevice(dev: *StorageDevice) void { + _ = ipc.close(dev.channel.endpoint); + dev.* = .{}; +} + +/// Retire one volume: kill its filesystem so its mounts are retired. Retirement +/// is lazy, not an eager death-time sweep — killing the process marks the +/// filesystem's backend endpoint dead, and the VFS router drops each mount that +/// endpoint backed on the next path resolution under it (that resolve frees the +/// slot and returns not_found). Then free the volume slot. +fn removeVolumeState(v: *Volume) void { std.log.info("storage for volume {d} removed; unmounting", .{v.id}); if (v.filesystem_pid != 0) _ = process.kill(v.filesystem_pid); - _ = ipc.close(v.storage.endpoint); - volume = null; - restart_pending = false; - fs_restarts = 0; - fs_failed = false; + v.* = .{}; } -/// One poll tick. Removal is checked FIRST and supersedes a pending restart: if -/// the device is gone there is nothing to restart fat onto, and respawning it -/// against the dead channel would just churn until the crash cap. Only once the -/// device is confirmed present does a due restart fire. -fn pollTick() void { - if (volume) |v| { - // Serving: watch for the specific device leaving (a pulled stick). - if (!isDevicePresent(v.storage_device_id)) { - removeVolume(); - return; - } - if (restart_pending and time.clock() >= restart_due_ns) { - restart_pending = false; - spawnFilesystem(&volume.?); - } - } else { - // Idle: try to bring a present storage device up. - bringUpVolume(); +/// A storage device left the tree (a pulled stick): retire every volume it served +/// and drop its channel. One removal path, whether the device is pulled cleanly +/// or vanishes. +fn removeDevice(dev: *StorageDevice) void { + for (&volumes) |*v| { + if (v.used and v.device_id == dev.device_id) removeVolumeState(v); } + dropDevice(dev); +} + +/// One poll tick. Device removal is reconciled FIRST and supersedes a pending +/// restart: a volume whose device left is retired before its restart could fire, +/// so nothing respawns against a dead channel. Then due restarts fire for present +/// volumes; then, if no device is adopted, a present device is brought up. +fn pollTick() void { + for (&devices) |*dev| { + if (dev.used and !isDevicePresent(dev.device_id)) removeDevice(dev); + } + for (&volumes) |*v| { + if (v.used and v.restart_pending and time.clock() >= v.restart_due_ns) { + v.restart_pending = false; + spawnFilesystem(v); + } + } + if (!anyDeviceUsed()) bringUpVolume(); } /// A filesystem announces itself for the volume it was spawned to serve. Reply @@ -335,26 +423,26 @@ fn pollTick() void { /// badge) as the call's returned capability. No channel means the volume is not /// ready — the filesystem retries. fn onHello(_: void, invocation: Invocation(volume_manager_protocol.Hello), _: Answer(void)) isize { - const v = volume orelse return 0; // not probed yet — retryable, no cap - if (invocation.target != v.id) return 0; // unknown volume — retryable + const v = volumeById(invocation.target) orelse return 0; // not probed yet — retryable, no cap if (invocation.sender != v.filesystem_pid) { - // Not the filesystem we spawned for this volume. Refuse: only the - // confined filesystem gets the channel. + // Not the filesystem we spawned for this volume. Refuse: only the confined + // filesystem gets the channel. std.log.info("refused hello for volume {d} from process {d}", .{ invocation.target, invocation.sender }); return -envelope.EPERM; } - service.replyWithCapability(v.storage.endpoint); + const dev = deviceById(v.device_id) orelse return 0; // its device left — retryable + service.replyWithCapability(dev.channel.endpoint); std.log.info("handed volume {d} to pid {d}", .{ v.id, invocation.sender }); return 0; } -/// Answer a `volumes` query with the mounted volume's descriptor — its id (its -/// mount path is /volumes/ unless overridden), its actual mount path, and -/// its display label. This is how a shell or file manager reads a volume's -/// friendly name: software keys on the id, a UI shows the label. An empty reply +/// Answer a `volumes` query with a mounted volume's descriptor — its id (its +/// mount path is /volumes/ unless overridden), its actual mount path, and its +/// display label. Software keys on the id; a UI shows the label. Returns the first +/// mounted volume for now; a full enumerate is a later refinement. Empty reply /// means no volume is mounted. fn onVolumes(_: void, _: Invocation(volume_manager_protocol.Volumes), answer: Answer(void)) isize { - const v = volume orelse return 0; + const v = firstUsedVolume() orelse return 0; var id_buf: [volume_map.id_maximum]u8 = undefined; const info = volume_manager_protocol.VolumeInfo{ .id = volume_map.idString(v.identity, &id_buf), @@ -381,9 +469,9 @@ fn readConfig(path: []const u8, buf: []u8) usize { defer file.close(); var used: usize = 0; while (used < buf.len) { - const n = file.read(buf[used..]) orelse break; - if (n == 0) break; - used += n; + const nn = file.read(buf[used..]) orelse break; + if (nn == 0) break; + used += nn; } return used; } @@ -427,23 +515,22 @@ fn onNotification(badge: u64) void { // reclaimed by the driver on the same death; the respawn confines afresh. if (got.isChildExit()) { const dead = got.childProcessId(); - const v = &(volume orelse return); - if (v.filesystem_pid != dead) return; + const v = volumeByPid(dead) orelse return; v.filesystem_pid = 0; const reason = process.exitReason(dead) orelse .fault; if (reason == .exited) { std.log.info("filesystem for volume {d} exited cleanly; not restarting", .{v.id}); return; } - const alive = time.clock() -| fs_spawn_ns; - fs_restarts = if (alive < fast_death_ns) fs_restarts + 1 else 1; - if (fs_restarts >= crash_loop_cap) { - fs_failed = true; + const alive = time.clock() -| v.spawn_ns; + v.restarts = if (alive < fast_death_ns) v.restarts + 1 else 1; + if (v.restarts >= crash_loop_cap) { + v.failed = true; std.log.info("filesystem for volume {d} is failing repeatedly; giving up", .{v.id}); return; } std.log.info("filesystem for volume {d} died ({s}); restarting", .{ v.id, @tagName(reason) }); - armRestart(); + armRestart(v); } }