volume-manager: rebuild a volume when its storage driver dies (S5)

The V4 review's open edge: a storage driver that crashes while its device
stays in the tree left fat wedged on a dead channel — device-presence
polling (a device-manager enumerate) still reported the device present,
so nothing reaped it. pollTick now also probes channelAlive(dev), a
geometry() on the block channel that fails fast on the dead endpoint; a
present device with a dead channel is reaped like a pull, and the adopt
loop re-adopts it on the restarted driver's fresh channel — the rebuild.
The manager's own liveness probe makes fat self-detection unnecessary:
it rebuilds regardless of the wedged filesystem's state.

The drill: the device manager gains a test-storage-restart mode that
kills usb-storage once, ~2s after its hello (post-mount); a new
volume-driver-restart kernel case boots a manual tree with it, and the
QEMU case asserts a SECOND mount of the same id-path after the reap —
the rebuild. A pre-S5 manager, checking only device presence, never
reaps, so the second mount never appears. The manually-spawned volume
manager needed kernel-supervisor protocol grants (bind its name, open
the device manager), as the other manual-tree services already have.

Full suite 133/133.
This commit is contained in:
Daniel Samson
2026-08-10 05:21:48 +01:00
parent 700452dc4e
commit 5dc966838a
5 changed files with 90 additions and 4 deletions
+40
View File
@@ -229,6 +229,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
fatMountTest(boot_information);
} else if (eql(case, "exfat-volume")) {
exfatVolumeTest(boot_information);
} else if (eql(case, "volume-driver-restart")) {
volumeDriverRestartTest(boot_information);
} else if (eql(case, "device-list")) {
deviceListTest(boot_information);
} else if (eql(case, "pci-scan")) {
@@ -3685,6 +3687,44 @@ fn displayReattachTest(boot_information: *const BootInformation) void {
while (true) scheduler.yield();
}
/// Storage-driver-crash rebuild (S5): the device manager runs in
/// "test-storage-restart" mode and kills the usb-storage driver once, a moment
/// after its volume has mounted. The driver's device stays in the tree, so the
/// volume manager's presence poll alone would miss the death and leave fat wedged
/// on a dead channel; its channel-liveness probe must notice, reap the volume, and
/// rebuild on the restarted driver's fresh channel — a SECOND mount of the same
/// id-path is the proof. (A pre-S5 manager, checking only device presence, never
/// reaps, so the second mount never appears.)
fn volumeDriverRestartTest(boot_information: *const BootInformation) void {
log("DANOS-TEST-BEGIN: volume-driver-restart\n", .{});
if (boot_information.initial_ramdisk_len == 0) {
check("bootloader handed over an initial_ramdisk", false);
result();
return;
}
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
const rd = initial_ramdisk.Reader.init(image) orelse {
check("initial_ramdisk image is valid", false);
result();
return;
};
process.setInitialRamdisk(image);
_ = spawnRegistry(rd);
var manager: u32 = 0;
var i: u32 = 0;
while (i < rd.count) : (i += 1) {
const item = rd.entry(i) orelse continue;
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
manager = process.spawnProcessSupervised(item.blob, 4, &.{ item.name, "test-storage-restart" }, scheduler.currentId(), null) catch 0;
break;
}
check("device-manager spawned (test-storage-restart mode)", manager != 0);
check("volume-manager spawned", spawnNamed(rd, "volume-manager"));
check("fat-test client spawned", spawnNamed(rd, "fat-test"));
scheduler.setPriority(1); // below the tree, so it runs
while (true) scheduler.yield();
}
/// Process arguments, end to end: spawn args-echo bare (its argv[0] is the
/// initial-ramdisk name). Instance 1 sees argc == 1 and respawns itself through
/// `system_spawn` with the extra arguments "alpha beta-42" — the syscall argument