volume-manager: rebuild a volume when its storage driver dies (S5)
The V4 review's open edge: a storage driver that crashes while its device stays in the tree left fat wedged on a dead channel — device-presence polling (a device-manager enumerate) still reported the device present, so nothing reaped it. pollTick now also probes channelAlive(dev), a geometry() on the block channel that fails fast on the dead endpoint; a present device with a dead channel is reaped like a pull, and the adopt loop re-adopts it on the restarted driver's fresh channel — the rebuild. The manager's own liveness probe makes fat self-detection unnecessary: it rebuilds regardless of the wedged filesystem's state. The drill: the device manager gains a test-storage-restart mode that kills usb-storage once, ~2s after its hello (post-mount); a new volume-driver-restart kernel case boots a manual tree with it, and the QEMU case asserts a SECOND mount of the same id-path after the reap — the rebuild. A pre-S5 manager, checking only device presence, never reaps, so the second mount never appears. The manually-spawned volume manager needed kernel-supervisor protocol grants (bind its name, open the device manager), as the other manual-tree services already have. Full suite 133/133.
This commit is contained in:
@@ -79,6 +79,7 @@
|
||||
# 'kernel' as the supervisor. Nothing else changes: the binary must still match.
|
||||
/system/services/input, kernel, bind, input
|
||||
/system/services/device-manager, kernel, bind, device-manager
|
||||
/system/services/volume-manager, kernel, bind, volume-manager
|
||||
/system/services/fat, kernel, bind, vfs
|
||||
/system/services/display, kernel, bind, display
|
||||
/system/services/discovery, kernel, bind, power
|
||||
@@ -107,6 +108,9 @@
|
||||
# The volume manager reaches the device manager to be routed to each storage
|
||||
# provider's block channel, then confines a filesystem to each volume.
|
||||
/system/services/volume-manager, /system/services/init, open, device-manager
|
||||
# ...and again under the kernel supervisor for the manual-tree drills (S5's
|
||||
# volume-driver-restart spawns the volume manager directly, not via init).
|
||||
/system/services/volume-manager, kernel, open, device-manager
|
||||
/system/services/display, /system/services/init, open, scanout
|
||||
/system/services/display, /system/services/init, open, display
|
||||
/system/services/display, /system/services/init, open, input
|
||||
|
||||
|
Can't render this file because it contains an unexpected character in line 12 and column 15.
|
@@ -229,6 +229,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
|
||||
fatMountTest(boot_information);
|
||||
} else if (eql(case, "exfat-volume")) {
|
||||
exfatVolumeTest(boot_information);
|
||||
} else if (eql(case, "volume-driver-restart")) {
|
||||
volumeDriverRestartTest(boot_information);
|
||||
} else if (eql(case, "device-list")) {
|
||||
deviceListTest(boot_information);
|
||||
} else if (eql(case, "pci-scan")) {
|
||||
@@ -3685,6 +3687,44 @@ fn displayReattachTest(boot_information: *const BootInformation) void {
|
||||
while (true) scheduler.yield();
|
||||
}
|
||||
|
||||
/// Storage-driver-crash rebuild (S5): the device manager runs in
|
||||
/// "test-storage-restart" mode and kills the usb-storage driver once, a moment
|
||||
/// after its volume has mounted. The driver's device stays in the tree, so the
|
||||
/// volume manager's presence poll alone would miss the death and leave fat wedged
|
||||
/// on a dead channel; its channel-liveness probe must notice, reap the volume, and
|
||||
/// rebuild on the restarted driver's fresh channel — a SECOND mount of the same
|
||||
/// id-path is the proof. (A pre-S5 manager, checking only device presence, never
|
||||
/// reaps, so the second mount never appears.)
|
||||
fn volumeDriverRestartTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: volume-driver-restart\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over an initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
process.setInitialRamdisk(image);
|
||||
_ = spawnRegistry(rd);
|
||||
var manager: u32 = 0;
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
|
||||
manager = process.spawnProcessSupervised(item.blob, 4, &.{ item.name, "test-storage-restart" }, scheduler.currentId(), null) catch 0;
|
||||
break;
|
||||
}
|
||||
check("device-manager spawned (test-storage-restart mode)", manager != 0);
|
||||
check("volume-manager spawned", spawnNamed(rd, "volume-manager"));
|
||||
check("fat-test client spawned", spawnNamed(rd, "fat-test"));
|
||||
scheduler.setPriority(1); // below the tree, so it runs
|
||||
while (true) scheduler.yield();
|
||||
}
|
||||
|
||||
/// Process arguments, end to end: spawn args-echo bare (its argv[0] is the
|
||||
/// initial-ramdisk name). Instance 1 sees argc == 1 and respawns itself through
|
||||
/// `system_spawn` with the extra arguments "alpha beta-42" — the syscall argument
|
||||
|
||||
@@ -170,6 +170,8 @@ var test_usb_killed = false;
|
||||
var test_pci_restart_mode = false;
|
||||
var test_scanout_restart_mode = false;
|
||||
var test_scanout_killed = false;
|
||||
var test_storage_restart_mode = false;
|
||||
var test_storage_killed = false;
|
||||
var test_kill_pid: u32 = 0;
|
||||
var test_kill_due_ns: u64 = 0;
|
||||
|
||||
@@ -600,6 +602,16 @@ fn onHello(_: void, invocation: Invocation(device_manager_protocol.Hello), _: An
|
||||
test_kill_due_ns = time.clock() + 1_500_000_000;
|
||||
_ = time.timerOnce(manager_endpoint, 1600);
|
||||
}
|
||||
// Storage-driver-crash drill (S5): once, a moment after usb-storage hellos —
|
||||
// long enough that its volume has mounted — kill it. The manager re-delegates
|
||||
// the still-present device to a restarted driver on a fresh channel; the volume
|
||||
// manager's channel-liveness probe must notice the dead channel and rebuild.
|
||||
if (test_storage_restart_mode and !test_storage_killed and std.mem.eql(u8, driver.name(), "/system/drivers/usb-storage")) {
|
||||
test_storage_killed = true;
|
||||
test_kill_pid = invocation.sender;
|
||||
test_kill_due_ns = time.clock() + 2_000_000_000; // after the ~0.6s mount
|
||||
_ = time.timerOnce(manager_endpoint, 2100);
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -763,6 +775,7 @@ pub fn main(init: process.Init) void {
|
||||
test_usb_restart_mode = std.mem.eql(u8, mode, "test-usb-restart");
|
||||
test_pci_restart_mode = std.mem.eql(u8, mode, "test-pci-restart");
|
||||
test_scanout_restart_mode = std.mem.eql(u8, mode, "test-scanout-restart");
|
||||
test_storage_restart_mode = std.mem.eql(u8, mode, "test-storage-restart");
|
||||
}
|
||||
service.run(device_manager_protocol.message_maximum, .{
|
||||
.service = "device-manager",
|
||||
|
||||
@@ -405,13 +405,25 @@ fn removeDevice(dev: *StorageDevice) void {
|
||||
dropDevice(dev);
|
||||
}
|
||||
|
||||
/// Whether a device's block channel still answers — a geometry() probe. A storage
|
||||
/// driver that DIED while its device stays in the tree (it crashed; the device
|
||||
/// manager will re-delegate the device to a restarted driver on a FRESH channel)
|
||||
/// leaves a dead channel here, even though isDevicePresent still reports the device
|
||||
/// present. geometry() on the dead endpoint fails fast, so this catches the crash
|
||||
/// that presence-polling alone cannot — the V4 review's open edge.
|
||||
fn channelAlive(dev: *StorageDevice) bool {
|
||||
return dev.channel.geometry() != null;
|
||||
}
|
||||
|
||||
/// One poll tick. Device removal is reconciled FIRST and supersedes a pending
|
||||
/// restart: a volume whose device left is retired before its restart could fire,
|
||||
/// so nothing respawns against a dead channel. Then due restarts fire for present
|
||||
/// volumes; then, if no device is adopted, a present device is brought up.
|
||||
/// restart: a volume whose device left (a pull) OR whose driver died on a channel
|
||||
/// that no longer answers is retired before its restart could fire, so nothing
|
||||
/// respawns against a dead channel. Dropping the device frees its slot, so the
|
||||
/// adopt loop below re-adopts the still-present device on the restarted driver's
|
||||
/// fresh channel — the rebuild. Then due restarts fire for present volumes.
|
||||
fn pollTick() void {
|
||||
for (&devices) |*dev| {
|
||||
if (dev.used and !isDevicePresent(dev.device_id)) removeDevice(dev);
|
||||
if (dev.used and (!isDevicePresent(dev.device_id) or !channelAlive(dev))) removeDevice(dev);
|
||||
}
|
||||
for (&volumes) |*v| {
|
||||
if (v.used and v.restart_pending and time.clock() >= v.restart_due_ns) {
|
||||
|
||||
@@ -815,6 +815,23 @@ CASES = [
|
||||
r"[\s\S]*volume-manager: medium left a storage device; unmounting"
|
||||
r"[\s\S]*volume-manager: storage for volume \d+ removed; unmounting",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# S5 storage-driver-crash rebuild. The device manager (test-storage-restart
|
||||
# mode) kills usb-storage once, ~2s in — after its volume mounted. The device
|
||||
# stays in the tree, so device-presence polling alone would leave fat wedged on
|
||||
# the dead channel; the volume manager's channel-liveness probe (a geometry()
|
||||
# that fails on the dead endpoint) must notice, reap the volume, and rebuild on
|
||||
# the restarted driver's fresh channel — a SECOND mount of the same id-path.
|
||||
# Discrimination: a pre-S5 manager checks only isDevicePresent (still true), so
|
||||
# it never reaps and the second mount never appears (it would restart fat on
|
||||
# the stale channel and crash-loop).
|
||||
{"name": "volume-driver-restart",
|
||||
"build_case": "volume-driver-restart",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"expect": r"(?s)fat: mounted /volumes/fat-12345678"
|
||||
r"[\s\S]*volume-manager: storage for volume \d+ removed; unmounting"
|
||||
r"[\s\S]*fat: mounted /volumes/fat-12345678",
|
||||
"fail": r"failing repeatedly; giving up|\[FAIL\]|DANOS-TEST-RESULT: FAIL"},
|
||||
# Volume-manager discovery + probe (V3a, docs/volume-manager-plan.md). Reuses
|
||||
# the fat-mount kernel build (the default boot now spawns the volume manager
|
||||
# from init.csv). It acquires the mass-storage block channel through the
|
||||
|
||||
Reference in New Issue
Block a user