volume-manager: harden the probe and supervision from the V3 review
Five confirmed defects from the boundary review: 1. (security) The VM never checked a partition fit inside the device, so a crafted MBR could hand the driver a range whose base+lba wraps past a u32 — panicking usb-storage in a loop, and at multi-volume overlapping a neighbour. This is the exact invariant the clamp's overflow-safety rests on. partition.firstVolume now skips any entry that runs past the device (host-tested), establishing the invariant where the untrusted bytes are first read. 2. (leak) The probe re-acquired a fresh block channel on every 500 ms retry, leaking a handle each time on a medium-absent device. The channel is now acquired once and kept. 3. (wedge) A failed spawn or defineRange stranded the volume with no retry; both now arm a backoff restart. 4. (loop) fat respawn had no exit-reason gate, no backoff, no crash-loop cap — a faulting filesystem respawned in a zero-delay loop, and a clean exit was resurrected. Supervision now mirrors the device manager: a clean exit is not restarted, a fault backs off, three fast deaths give up. 5. (removable) A device that parsed to no volume was terminal; it now keeps polling so an inserted medium is picked up — the removal-lifecycle trigger. Known limitation (noted, not fixed here): if the VM itself crashes and init restarts it, the orphaned fat keeps serving vfs while the new VM spawns a second fat whose bind is refused — the same "manager restart re-learns the world" gap the device manager also defers. The old fat keeps storage working. Neutral: partition unit tests + fat-mount, volume-probe, block-range, logger all green.
This commit is contained in:
@@ -56,6 +56,13 @@ pub fn firstVolume(block0: []const u8, device_blocks: u64) ?Volume {
|
|||||||
const start = std.mem.readInt(u32, entry[8..12], .little);
|
const start = std.mem.readInt(u32, entry[8..12], .little);
|
||||||
const size = std.mem.readInt(u32, entry[12..16], .little);
|
const size = std.mem.readInt(u32, entry[12..16], .little);
|
||||||
if (kind == 0 or start == 0 or size == 0) continue;
|
if (kind == 0 or start == 0 or size == 0) continue;
|
||||||
|
// These bytes come off an untrusted removable medium. A partition that
|
||||||
|
// does not fit inside the device is not a partition — skip it. This is
|
||||||
|
// where the driver's confinement-safety invariant is established: the
|
||||||
|
// clamp's overflow-safety rests on base + count staying inside the
|
||||||
|
// device (usb-storage.zig resolveTransfer), which only holds because the
|
||||||
|
// range handed down is validated here. The subtraction cannot overflow.
|
||||||
|
if (start > device_blocks or device_blocks - start < size) continue;
|
||||||
return .{ .base_lba = start, .block_count = size, .identity = identityOf(block0, index) };
|
return .{ .base_lba = start, .block_count = size, .identity = identityOf(block0, index) };
|
||||||
}
|
}
|
||||||
// No partition entries: a bare FAT spanning the device.
|
// No partition entries: a bare FAT spanning the device.
|
||||||
@@ -90,3 +97,20 @@ test "no boot signature is no volume" {
|
|||||||
const block0 = [_]u8{0} ** 512;
|
const block0 = [_]u8{0} ** 512;
|
||||||
try std.testing.expect(firstVolume(&block0, 65536) == null);
|
try std.testing.expect(firstVolume(&block0, 65536) == null);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
test "a partition that runs past the device is skipped, not trusted" {
|
||||||
|
var block0 = [_]u8{0} ** 512;
|
||||||
|
block0[510] = 0x55;
|
||||||
|
block0[511] = 0xAA;
|
||||||
|
// partition 0: start 0xFFFFFF00, size 0x400 — far past a 200000-block device.
|
||||||
|
block0[446 + 4] = 0x0c;
|
||||||
|
std.mem.writeInt(u32, block0[446 + 8 ..][0..4], 0xFFFFFF00, .little);
|
||||||
|
std.mem.writeInt(u32, block0[446 + 12 ..][0..4], 0x400, .little);
|
||||||
|
// partition 1: start 2048, size 1000 — fits.
|
||||||
|
block0[462 + 4] = 0x0c;
|
||||||
|
std.mem.writeInt(u32, block0[462 + 8 ..][0..4], 2048, .little);
|
||||||
|
std.mem.writeInt(u32, block0[462 + 12 ..][0..4], 1000, .little);
|
||||||
|
const v = firstVolume(&block0, 200000).?;
|
||||||
|
try std.testing.expectEqual(@as(u64, 2048), v.base_lba); // the fitting one, not the overflowing one
|
||||||
|
try std.testing.expectEqual(@as(u64, 1000), v.block_count);
|
||||||
|
}
|
||||||
|
|||||||
@@ -56,8 +56,25 @@ var bounce: memory.DmaRegion = undefined;
|
|||||||
var bounce_ready = false;
|
var bounce_ready = false;
|
||||||
var probed = false;
|
var probed = false;
|
||||||
var volume: ?Volume = null;
|
var volume: ?Volume = null;
|
||||||
|
/// The storage channel, acquired ONCE and kept — re-acquiring on every probe
|
||||||
|
/// retry would leak a handle per attempt on a medium-absent device.
|
||||||
|
var storage_device: ?block.Device = null;
|
||||||
|
var logged_no_volume = false;
|
||||||
const probe_retry_ms = 500;
|
const probe_retry_ms = 500;
|
||||||
|
|
||||||
|
// Filesystem supervision, mirroring the device manager's (device-manager.zig):
|
||||||
|
// a clean exit is not restarted, a fault restarts with backoff, and a fast
|
||||||
|
// crash loop gives up rather than spinning. Without this a faulting filesystem
|
||||||
|
// respawns in a zero-delay loop.
|
||||||
|
const fast_death_ns: u64 = 2_000_000_000;
|
||||||
|
const crash_loop_cap: u32 = 3;
|
||||||
|
const backoff_base_ms: u64 = 300;
|
||||||
|
var fs_restarts: u32 = 0;
|
||||||
|
var fs_spawn_ns: u64 = 0;
|
||||||
|
var fs_failed = false;
|
||||||
|
/// Set when a backoff timer is pending so its tick respawns rather than probes.
|
||||||
|
var restart_pending = false;
|
||||||
|
|
||||||
fn acquireStorage() ?block.Device {
|
fn acquireStorage() ?block.Device {
|
||||||
const manager = manager_handle orelse opened: {
|
const manager = manager_handle orelse opened: {
|
||||||
const handle = channel.openEndpoint("device-manager") orelse return null;
|
const handle = channel.openEndpoint("device-manager") orelse return null;
|
||||||
@@ -94,26 +111,45 @@ fn acquireStorage() ?block.Device {
|
|||||||
/// runs, so its first read is already bounded; the volume manager is the
|
/// runs, so its first read is already bounded; the volume manager is the
|
||||||
/// confinement controller (it defines the first range on the device).
|
/// confinement controller (it defines the first range on the device).
|
||||||
fn spawnFilesystem(v: *Volume) void {
|
fn spawnFilesystem(v: *Volume) void {
|
||||||
|
if (fs_failed) return;
|
||||||
const pid = process.spawnSupervised(filesystem_binary, &.{"1"}, service_endpoint) orelse {
|
const pid = process.spawnSupervised(filesystem_binary, &.{"1"}, service_endpoint) orelse {
|
||||||
_ = logging.write("volume-manager: could not spawn the filesystem\n");
|
_ = logging.write("volume-manager: could not spawn the filesystem; retrying\n");
|
||||||
|
armRestart();
|
||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
if (!v.storage.defineRange(pid, v.base_lba, v.block_count)) {
|
if (!v.storage.defineRange(pid, v.base_lba, v.block_count)) {
|
||||||
_ = logging.write("volume-manager: could not confine the filesystem to its volume\n");
|
_ = logging.write("volume-manager: could not confine the filesystem to its volume; retrying\n");
|
||||||
_ = process.kill(pid);
|
_ = process.kill(pid);
|
||||||
|
armRestart();
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
v.filesystem_pid = pid;
|
v.filesystem_pid = pid;
|
||||||
|
fs_spawn_ns = time.clock();
|
||||||
std.log.info("volume 0x{x} -> {s} (pid {d}), lba {d}, {d} blocks", .{ v.identity, filesystem_binary, pid, v.base_lba, v.block_count });
|
std.log.info("volume 0x{x} -> {s} (pid {d}), lba {d}, {d} blocks", .{ v.identity, filesystem_binary, pid, v.base_lba, v.block_count });
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Arm a one-shot timer to (re)spawn the filesystem after backoff — used both
|
||||||
|
/// when a spawn step fails and when a running filesystem faults. Distinguished
|
||||||
|
/// from the probe timer by `restart_pending`.
|
||||||
|
fn armRestart() void {
|
||||||
|
const delay = if (fs_restarts == 0) backoff_base_ms else backoff_base_ms << @intCast(@min(fs_restarts - 1, 5));
|
||||||
|
restart_pending = true;
|
||||||
|
_ = time.timerOnce(service_endpoint, delay);
|
||||||
|
}
|
||||||
|
|
||||||
fn tryProbe() void {
|
fn tryProbe() void {
|
||||||
if (probed) return;
|
if (probed) return;
|
||||||
if (!bounce_ready) {
|
if (!bounce_ready) {
|
||||||
bounce = memory.dmaAlloc(512, memory.dma_coherent | memory.dma_shareable) orelse return;
|
bounce = memory.dmaAlloc(512, memory.dma_coherent | memory.dma_shareable) orelse return;
|
||||||
bounce_ready = true;
|
bounce_ready = true;
|
||||||
}
|
}
|
||||||
const device = acquireStorage() orelse return;
|
// Acquire the storage channel once and keep it: a fresh consumer-hello per
|
||||||
|
// retry would leak a handle every 500 ms on a device whose medium is absent.
|
||||||
|
const device = storage_device orelse acquired: {
|
||||||
|
const d = acquireStorage() orelse return;
|
||||||
|
storage_device = d;
|
||||||
|
break :acquired d;
|
||||||
|
};
|
||||||
if (bounce.handle) |handle| {
|
if (bounce.handle) |handle| {
|
||||||
if (!device.attach(handle)) return;
|
if (!device.attach(handle)) return;
|
||||||
_ = ipc.close(handle);
|
_ = ipc.close(handle);
|
||||||
@@ -123,8 +159,13 @@ fn tryProbe() void {
|
|||||||
if (!device.read(0, 1, bounce.physical)) return;
|
if (!device.read(0, 1, bounce.physical)) return;
|
||||||
const sector: [*]const u8 = @ptrFromInt(bounce.virtual);
|
const sector: [*]const u8 = @ptrFromInt(bounce.virtual);
|
||||||
const found = partition.firstVolume(sector[0..512], geometry.block_count) orelse {
|
const found = partition.firstVolume(sector[0..512], geometry.block_count) orelse {
|
||||||
_ = logging.write("volume-manager: no volume found on the storage device\n");
|
// No volume yet. On removable media this can mean no medium is present —
|
||||||
probed = true;
|
// keep polling so an inserted medium is picked up (the removal-lifecycle
|
||||||
|
// trigger). Log once, and do NOT terminate the probe.
|
||||||
|
if (!logged_no_volume) {
|
||||||
|
_ = logging.write("volume-manager: no volume on the storage device yet\n");
|
||||||
|
logged_no_volume = true;
|
||||||
|
}
|
||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
volume = .{ .storage = device, .base_lba = found.base_lba, .block_count = found.block_count, .identity = found.identity, .id = volume_id };
|
volume = .{ .storage = device, .base_lba = found.base_lba, .block_count = found.block_count, .identity = found.identity, .id = volume_id };
|
||||||
@@ -169,22 +210,40 @@ fn initialise(endpoint: ipc.Handle) bool {
|
|||||||
fn onNotification(badge: u64) void {
|
fn onNotification(badge: u64) void {
|
||||||
const got = ipc.Received{ .len = 0, .badge = badge, .cap = null };
|
const got = ipc.Received{ .len = 0, .badge = badge, .cap = null };
|
||||||
if (got.isTimer()) {
|
if (got.isTimer()) {
|
||||||
|
// One timer signal, two jobs, told apart by state: a pending backoff
|
||||||
|
// restart, otherwise the probe retry.
|
||||||
|
if (restart_pending) {
|
||||||
|
restart_pending = false;
|
||||||
|
if (volume) |*v| spawnFilesystem(v);
|
||||||
|
return;
|
||||||
|
}
|
||||||
tryProbe();
|
tryProbe();
|
||||||
if (!probed) _ = time.timerOnce(service_endpoint, probe_retry_ms);
|
if (!probed) _ = time.timerOnce(service_endpoint, probe_retry_ms);
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
// A filesystem died. Its old range is reclaimed by the driver on the same
|
// A filesystem died. The exit reason drives the decision, exactly as the
|
||||||
// death; respawn it, confined afresh to the same volume (a fresh pid, a
|
// device manager supervises drivers: a clean exit meant to stop; a fault
|
||||||
// fresh range). The reap-and-rebuild the device manager proved, one layer up.
|
// restarts with backoff until a fast crash loop gives up. The old range is
|
||||||
|
// reclaimed by the driver on the same death; the respawn confines afresh.
|
||||||
if (got.isChildExit()) {
|
if (got.isChildExit()) {
|
||||||
const dead = got.childProcessId();
|
const dead = got.childProcessId();
|
||||||
if (volume) |*v| {
|
const v = &(volume orelse return);
|
||||||
if (v.filesystem_pid == dead) {
|
if (v.filesystem_pid != dead) return;
|
||||||
v.filesystem_pid = 0;
|
v.filesystem_pid = 0;
|
||||||
std.log.info("filesystem for volume {d} died; respawning", .{v.id});
|
const reason = process.exitReason(dead) orelse .fault;
|
||||||
spawnFilesystem(v);
|
if (reason == .exited) {
|
||||||
}
|
std.log.info("filesystem for volume {d} exited cleanly; not restarting", .{v.id});
|
||||||
|
return;
|
||||||
}
|
}
|
||||||
|
const alive = time.clock() -| fs_spawn_ns;
|
||||||
|
fs_restarts = if (alive < fast_death_ns) fs_restarts + 1 else 1;
|
||||||
|
if (fs_restarts >= crash_loop_cap) {
|
||||||
|
fs_failed = true;
|
||||||
|
std.log.info("filesystem for volume {d} is failing repeatedly; giving up", .{v.id});
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
std.log.info("filesystem for volume {d} died ({s}); restarting", .{ v.id, @tagName(reason) });
|
||||||
|
armRestart();
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user