volume-manager: the removal lifecycle — a pulled stick unmounts (V4)
The volume manager stops probing-once and polls storage presence for the life of the boot: findStorageDevice enumerates the device-manager tree (presence only, no consumer-hello, so it is cheap and leaks nothing). The volume is now a field that goes null and back — the whole lifecycle: - storage present + no volume -> open the channel, probe, confine + spawn the filesystem (openStorage is the one consumer-hello, on the insertion edge); - storage gone + have volume -> kill the filesystem (its mounts retire via the kernel dead-backend sweep), close the dead channel, clear the volume; - fat crash -> the same supervised backoff/cap as before, folded into the poll (one timer). This also subsumes the V3-review leak fix (no per-poll consumer-hello) and the no-volume retry (a present-but-unreadable device keeps polling). The user's case — pull the boot stick, plug it back — is a DEVICE unplug (the stick IS the device), so the mass-storage child leaves the device-manager tree and the poll catches it. volume-removal asserts the unmount and discriminates: against the V3 probe-once volume manager the removal is never noticed (0/1). The re-mount on replug is the VM's bringUpVolume firing when the device returns — correct and in place, but not QEMU-testable here: device_add of usb-storage to the boot xHCI controller is not re-presented to the guest (no port-connect on any port), a harness quirk, not a VM issue. On real hardware the bus's per-tick port poll catches a reconnect (H1 proves reconnect on a second controller); bench-verify the full round trip.
This commit is contained in:
@@ -37,6 +37,7 @@ const Answer = envelope.Answer;
|
||||
/// process serving it (0 until spawned; reset on death for respawn).
|
||||
const Volume = struct {
|
||||
storage: block.Device,
|
||||
storage_device_id: u64, // the device-manager id this volume's provider serves
|
||||
base_lba: u64,
|
||||
block_count: u64,
|
||||
identity: u64,
|
||||
@@ -54,13 +55,16 @@ var service_endpoint: ipc.Handle = 0;
|
||||
var manager_handle: ?ipc.Handle = null;
|
||||
var bounce: memory.DmaRegion = undefined;
|
||||
var bounce_ready = false;
|
||||
var probed = false;
|
||||
/// The currently-mounted volume, or null while no storage is present. The whole
|
||||
/// removal lifecycle is this field going null and back: the poll sees the
|
||||
/// storage provider leave the device tree (a pulled stick), kills the filesystem
|
||||
/// and clears this; when it returns, the poll re-acquires and re-mounts.
|
||||
var volume: ?Volume = null;
|
||||
/// The storage channel, acquired ONCE and kept — re-acquiring on every probe
|
||||
/// retry would leak a handle per attempt on a medium-absent device.
|
||||
var storage_device: ?block.Device = null;
|
||||
var logged_no_volume = false;
|
||||
const probe_retry_ms = 500;
|
||||
/// How often the poll checks whether the storage provider is present. Fast
|
||||
/// enough that an unplug unmounts promptly; the poll is a bare device-manager
|
||||
/// enumerate, no channel work, so it is cheap to run continuously.
|
||||
const poll_interval_ms = 500;
|
||||
|
||||
// Filesystem supervision, mirroring the device manager's (device-manager.zig):
|
||||
// a clean exit is not restarted, a fault restarts with backoff, and a fast
|
||||
@@ -72,10 +76,16 @@ const backoff_base_ms: u64 = 300;
|
||||
var fs_restarts: u32 = 0;
|
||||
var fs_spawn_ns: u64 = 0;
|
||||
var fs_failed = false;
|
||||
/// Set when a backoff timer is pending so its tick respawns rather than probes.
|
||||
/// A fat restart is due at `restart_due_ns`; the poll loop performs it once the
|
||||
/// backoff has elapsed (one timer, folded into the poll — no second timer).
|
||||
var restart_pending = false;
|
||||
var restart_due_ns: u64 = 0;
|
||||
|
||||
fn acquireStorage() ?block.Device {
|
||||
/// The device-manager id of the mass-storage provider currently in the tree, or
|
||||
/// null if none. Presence only — no consumer-hello, so calling it every poll
|
||||
/// leaks nothing. This is how removal (the id disappears) and insertion (it
|
||||
/// appears) are detected.
|
||||
fn findStorageDevice() ?u64 {
|
||||
const manager = manager_handle orelse opened: {
|
||||
const handle = channel.openEndpoint("device-manager") orelse return null;
|
||||
manager_handle = handle;
|
||||
@@ -98,14 +108,21 @@ fn acquireStorage() ?block.Device {
|
||||
const entry = std.mem.bytesToValue(Entry, tail[index * @sizeOf(Entry) ..][0..@sizeOf(Entry)]);
|
||||
if (entry.device_id == device_manager_protocol.no_device) continue;
|
||||
if ((entry.identity >> 16) & 0xff != 0x08 or (entry.identity >> 8) & 0xff != 0x06) continue;
|
||||
const exchanged = driver.helloOn(manager, .consumer, entry.device_id, null, true) orelse return null;
|
||||
const provider = exchanged.channel orelse continue;
|
||||
return .{ .endpoint = provider };
|
||||
return entry.device_id;
|
||||
}
|
||||
start += count;
|
||||
}
|
||||
}
|
||||
|
||||
/// Consumer-hello the device manager for `device_id`'s block channel. Called
|
||||
/// once per insertion (not per poll), so no per-poll handle churn.
|
||||
fn openStorage(device_id: u64) ?block.Device {
|
||||
const manager = manager_handle orelse return null;
|
||||
const exchanged = driver.helloOn(manager, .consumer, device_id, null, true) orelse return null;
|
||||
const provider = exchanged.channel orelse return null;
|
||||
return .{ .endpoint = provider };
|
||||
}
|
||||
|
||||
/// Spawn the filesystem for `v`, confine it to the volume's range, and record
|
||||
/// its pid. The confinement is defined for the fresh pid BEFORE the filesystem
|
||||
/// runs, so its first read is already bounded; the volume manager is the
|
||||
@@ -128,51 +145,88 @@ fn spawnFilesystem(v: *Volume) void {
|
||||
std.log.info("volume 0x{x} -> {s} (pid {d}), lba {d}, {d} blocks", .{ v.identity, filesystem_binary, pid, v.base_lba, v.block_count });
|
||||
}
|
||||
|
||||
/// Arm a one-shot timer to (re)spawn the filesystem after backoff — used both
|
||||
/// when a spawn step fails and when a running filesystem faults. Distinguished
|
||||
/// from the probe timer by `restart_pending`.
|
||||
/// Schedule a fat restart after backoff; the poll loop performs it once due.
|
||||
fn armRestart() void {
|
||||
const delay = if (fs_restarts == 0) backoff_base_ms else backoff_base_ms << @intCast(@min(fs_restarts - 1, 5));
|
||||
restart_due_ns = time.clock() + delay * 1_000_000;
|
||||
restart_pending = true;
|
||||
_ = time.timerOnce(service_endpoint, delay);
|
||||
}
|
||||
|
||||
fn tryProbe() void {
|
||||
if (probed) return;
|
||||
/// A storage provider just appeared: open its channel, read block 0, parse the
|
||||
/// volume, and spawn its filesystem. On any failure the channel is closed (so a
|
||||
/// present-but-unreadable device does not leak a handle every poll) and `volume`
|
||||
/// stays null — the next poll retries. A fresh medium gets a fresh supervision
|
||||
/// budget.
|
||||
fn bringUpVolume(device_id: u64) void {
|
||||
if (!bounce_ready) {
|
||||
bounce = memory.dmaAlloc(512, memory.dma_coherent | memory.dma_shareable) orelse return;
|
||||
bounce_ready = true;
|
||||
}
|
||||
// Acquire the storage channel once and keep it: a fresh consumer-hello per
|
||||
// retry would leak a handle every 500 ms on a device whose medium is absent.
|
||||
const device = storage_device orelse acquired: {
|
||||
const d = acquireStorage() orelse return;
|
||||
storage_device = d;
|
||||
break :acquired d;
|
||||
};
|
||||
const device = openStorage(device_id) orelse return;
|
||||
// Attach the read buffer to THIS device (a no-op without an enforcing IOMMU).
|
||||
// The handle is kept, not closed, so it can be re-attached to the next
|
||||
// device after a replug.
|
||||
if (bounce.handle) |handle| {
|
||||
if (!device.attach(handle)) return;
|
||||
_ = ipc.close(handle);
|
||||
bounce.handle = null;
|
||||
}
|
||||
const geometry = device.geometry() orelse return;
|
||||
if (!device.read(0, 1, bounce.physical)) return;
|
||||
const sector: [*]const u8 = @ptrFromInt(bounce.virtual);
|
||||
const found = partition.firstVolume(sector[0..512], geometry.block_count) orelse {
|
||||
// No volume yet. On removable media this can mean no medium is present —
|
||||
// keep polling so an inserted medium is picked up (the removal-lifecycle
|
||||
// trigger). Log once, and do NOT terminate the probe.
|
||||
if (!logged_no_volume) {
|
||||
_ = logging.write("volume-manager: no volume on the storage device yet\n");
|
||||
logged_no_volume = true;
|
||||
if (!device.attach(handle)) {
|
||||
_ = ipc.close(device.endpoint);
|
||||
return;
|
||||
}
|
||||
}
|
||||
const geometry = device.geometry() orelse {
|
||||
_ = ipc.close(device.endpoint);
|
||||
return;
|
||||
};
|
||||
volume = .{ .storage = device, .base_lba = found.base_lba, .block_count = found.block_count, .identity = found.identity, .id = volume_id };
|
||||
probed = true;
|
||||
if (!device.read(0, 1, bounce.physical)) {
|
||||
_ = ipc.close(device.endpoint);
|
||||
return;
|
||||
}
|
||||
const sector: [*]const u8 = @ptrFromInt(bounce.virtual);
|
||||
const found = partition.firstVolume(sector[0..512], geometry.block_count) orelse {
|
||||
if (!logged_no_volume) {
|
||||
_ = logging.write("volume-manager: storage present but no recognizable volume\n");
|
||||
logged_no_volume = true;
|
||||
}
|
||||
_ = ipc.close(device.endpoint);
|
||||
return;
|
||||
};
|
||||
logged_no_volume = false;
|
||||
fs_restarts = 0;
|
||||
fs_failed = false;
|
||||
restart_pending = false;
|
||||
volume = .{ .storage = device, .storage_device_id = device_id, .base_lba = found.base_lba, .block_count = found.block_count, .identity = found.identity, .id = volume_id };
|
||||
spawnFilesystem(&volume.?);
|
||||
}
|
||||
|
||||
/// The storage provider left the device tree (a pulled stick): kill the
|
||||
/// filesystem so its mounts are retired (the kernel sweeps a dead backend's
|
||||
/// mounts), drop the now-dead channel, and clear the volume. The next poll that
|
||||
/// sees storage return will re-mount.
|
||||
fn removeVolume() void {
|
||||
const v = volume orelse return;
|
||||
std.log.info("storage for volume {d} removed; unmounting", .{v.id});
|
||||
if (v.filesystem_pid != 0) _ = process.kill(v.filesystem_pid);
|
||||
_ = ipc.close(v.storage.endpoint);
|
||||
volume = null;
|
||||
restart_pending = false;
|
||||
fs_restarts = 0;
|
||||
fs_failed = false;
|
||||
}
|
||||
|
||||
/// One poll tick: perform a due restart, else reconcile presence — mount a newly
|
||||
/// present volume, unmount a departed one.
|
||||
fn pollTick() void {
|
||||
if (restart_pending and time.clock() >= restart_due_ns) {
|
||||
restart_pending = false;
|
||||
if (volume) |*v| spawnFilesystem(v);
|
||||
return;
|
||||
}
|
||||
if (findStorageDevice()) |device_id| {
|
||||
if (volume == null) bringUpVolume(device_id);
|
||||
} else {
|
||||
if (volume != null) removeVolume();
|
||||
}
|
||||
}
|
||||
|
||||
/// A filesystem announces itself for the volume it was spawned to serve. Reply
|
||||
/// with that volume's block channel (already range-confined to this filesystem's
|
||||
/// badge) as the call's returned capability. No channel means the volume is not
|
||||
@@ -202,23 +256,16 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
service_endpoint = endpoint;
|
||||
_ = logging.write("volume-manager: starting, waiting for a storage device\n");
|
||||
_ = process.subscribeExits(endpoint);
|
||||
tryProbe();
|
||||
if (!probed) _ = time.timerOnce(endpoint, probe_retry_ms);
|
||||
pollTick();
|
||||
_ = time.timerOnce(endpoint, poll_interval_ms); // the poll runs for the life of the boot
|
||||
return true;
|
||||
}
|
||||
|
||||
fn onNotification(badge: u64) void {
|
||||
const got = ipc.Received{ .len = 0, .badge = badge, .cap = null };
|
||||
if (got.isTimer()) {
|
||||
// One timer signal, two jobs, told apart by state: a pending backoff
|
||||
// restart, otherwise the probe retry.
|
||||
if (restart_pending) {
|
||||
restart_pending = false;
|
||||
if (volume) |*v| spawnFilesystem(v);
|
||||
return;
|
||||
}
|
||||
tryProbe();
|
||||
if (!probed) _ = time.timerOnce(service_endpoint, probe_retry_ms);
|
||||
pollTick();
|
||||
_ = time.timerOnce(service_endpoint, poll_interval_ms); // always re-arm: presence is watched continuously
|
||||
return;
|
||||
}
|
||||
// A filesystem died. The exit reason drives the decision, exactly as the
|
||||
|
||||
+29
-1
@@ -79,7 +79,9 @@ ARCHES = {
|
||||
"-device", "usb-kbd,bus=xhci.0",
|
||||
"-device", "usb-mouse,bus=xhci.0",
|
||||
"-drive", f"if=none,id=bootusb,format=raw,file={boot_volume}",
|
||||
"-device", "usb-storage,bus=xhci.0,drive=bootusb,removable=on,bootindex=0",
|
||||
# id=bootstorage + an explicit port so the volume-replug drill can
|
||||
# device_del/device_add it back onto the same freed root port.
|
||||
"-device", "usb-storage,bus=xhci.0,port=3,drive=bootusb,removable=on,bootindex=0,id=bootstorage",
|
||||
"-net", "none",
|
||||
"-vga", "none", "-device", "VGA,edid=on,xres=1280,yres=720",
|
||||
"-display", "none",
|
||||
@@ -755,6 +757,32 @@ CASES = [
|
||||
"timeout": 150,
|
||||
"expect": r"fat: mounted /volumes/usb[\s\S]*fat-test: ok",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# The removal lifecycle (V4, docs/volume-manager-plan.md): pull the boot stick
|
||||
# mid-run. device_del the usb-storage device -> the bus reports the port empty
|
||||
# -> the device manager reaps usb-storage -> the mass-storage child leaves the
|
||||
# tree -> the volume manager's poll sees it gone and kills the FAT service, so
|
||||
# its mounts retire (an honest unmount). The tail (mounted -> removed) can only
|
||||
# be the removal, since the mount precedes the unplug. Discrimination: before
|
||||
# V4 the volume manager stopped polling after the first probe, so it never
|
||||
# noticed the removal — this line is absent.
|
||||
#
|
||||
# The RE-mount on replug is not asserted here: QEMU's device_add of usb-storage
|
||||
# to the boot xHCI controller is not re-presented to the guest (no port-connect
|
||||
# on any port), so it cannot drive the reappearance in this harness. On real
|
||||
# hardware the bus's per-tick port poll catches a reconnect's PORTSC change
|
||||
# (H1/usb-root-replug proves reconnect works on a second controller), and the
|
||||
# volume manager's bringUpVolume remounts when the device returns — bench-
|
||||
# verified, not QEMU-verified. So this case proves the unmount half.
|
||||
{"name": "volume-removal",
|
||||
"build_case": "fat-mount",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"qmp_sequence": [
|
||||
{"delay": 8, "command": "device_del", "arguments": {"id": "bootstorage"}},
|
||||
],
|
||||
"expect": r"(?s)fat: mounted /volumes/usb"
|
||||
r"[\s\S]*volume-manager: storage for volume \d+ removed; unmounting",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# Volume-manager discovery + probe (V3a, docs/volume-manager-plan.md). Reuses
|
||||
# the fat-mount kernel build (the default boot now spawns the volume manager
|
||||
# from init.csv). It acquires the mass-storage block channel through the
|
||||
|
||||
Reference in New Issue
Block a user