usb/fat: transfer events matched by slot+endpoint; storage failures heal

B2 — the 1-in-3 boot-time READ CAPACITY failure, root-caused: the xHCI
library's awaitTransfer claimed ANY unclaimed transfer event as its own
completion. An interrupt-endpoint event whose TRB pointer no longer
matched the armed subscription (an error or stale completion from the
keyboard/mouse polling concurrently with storage bring-up) fell through
and was misread as the bulk transfer's completion — desynchronizing the
mass-storage bulk protocol in controller state that SURVIVED driver
restarts, so every retry failed too. Awaited transfers now match the
event's slot id and endpoint DCI; foreign events are dropped and named.
Twelve consecutive runs of the previously-flaky cases pass; the full
suite is green with none of its old intermittents.

B1 — and when storage does fail transiently, the system now heals
instead of giving up forever: a nonzero exit maps to ExitReason.aborted
(a deliberate FAILURE exit — supervisors restart those with backoff,
unlike a clean .exited), usb-storage exits nonzero when a PRESENT
device fails bring-up, and the fat service no longer blocks its harness
polling for a block device and then dies — it serves immediately
(requests fail politely), retries on a 500 ms timer, and mounts
whenever storage appears, including after a driver restart.
This commit is contained in:
Daniel Samson
2026-07-21 20:27:55 +01:00
parent 7082699f5f
commit 1638845a4b
6 changed files with 92 additions and 29 deletions
+44 -17
View File
@@ -76,17 +76,40 @@ fn fail(out: []u8) usize {
return writeReply(out, .{ .status = -1 }, &.{});
}
/// How often to look for a block device while none is mounted. Storage arriving
/// is EVENT-shaped (the usb chain registering, possibly after a driver restart),
/// but the registry has no subscription — a slow poll from our own harness loop
/// keeps the service responsive (ping, terminate) while it waits, and keeps it
/// alive to catch storage that appears LATE (a restarted usb-storage after a
/// transient failure — the resilience half of docs/logging.md's storage story).
const mount_retry_ms = 500;
var mounted = false;
var service_endpoint: runtime.ipc.Handle = 0;
fn initialise(endpoint: runtime.ipc.Handle) bool {
service_endpoint = endpoint;
_ = runtime.system.write("/system/services/fat: starting, waiting for a block device\n");
const device = runtime.block.open() orelse {
_ = runtime.system.write("/system/services/fat: no block device (no storage attached)\n");
return false; // clean exit: nothing to serve
};
// With the router in the kernel, clients hold OUR node ids directly; sweep
// a dead client's open handles via the published exit events (the pattern
// the old userspace router used for its own table).
_ = runtime.process.subscribeExits(endpoint);
tryBringUp();
if (!mounted) _ = runtime.system.timerOnce(endpoint, mount_retry_ms);
return true; // serve regardless: requests fail politely until storage mounts
}
/// One storage bring-up attempt: block device -> FAT mount -> VFS mounts. Sets
/// `mounted` on success; a failure leaves everything untouched for the next tick.
fn tryBringUp() void {
if (mounted) return;
const device = runtime.block.tryOpen() orelse return;
const geometry = device.geometry() orelse {
_ = runtime.system.write("/system/services/fat: block geometry unavailable\n");
return false;
return;
};
ipc_block = .{ .device = device, .bounce = dma.alloc(4096, dma.coherent) orelse return false };
const bounce = dma.alloc(4096, dma.coherent) orelse return;
ipc_block = .{ .device = device, .bounce = bounce };
const block_device = engine.BlockDevice{
.context = &ipc_block,
@@ -97,36 +120,39 @@ fn initialise(endpoint: runtime.ipc.Handle) bool {
};
filesystem = engine.FileSystem.mount(block_device) orelse {
_ = runtime.system.write("/system/services/fat: not a FAT filesystem\n");
return false;
return;
};
std.log.info("mounted FAT ({s}, {d} clusters, partition lba {d})", .{ @tagName(filesystem.geometry.fat_type), filesystem.geometry.cluster_count, filesystem.base_lba });
// With the router in the kernel, clients hold OUR node ids directly; sweep
// a dead client's open handles via the published exit events (the pattern
// the old userspace router used for its own table).
_ = runtime.process.subscribeExits(endpoint);
// Mount ourselves into the kernel VFS at /mnt/usb — and serve /var from the
// volume's /var subtree, so FHS paths (the logger's /var/log) stay decoupled
// from which volume carries them. A mount is one syscall now; no retry
// needed (the kernel's table exists before any service).
if (runtime.fs.mount(mount_point, endpoint)) {
// from which volume carries them.
if (runtime.fs.mount(mount_point, endpointForMount())) {
std.log.info("mounted {s}", .{mount_point});
} else {
_ = runtime.system.write("/system/services/fat: could not mount /mnt/usb\n");
}
if (runtime.fs.mountRewritten("/var", endpoint, "/var")) {
if (runtime.fs.mountRewritten("/var", endpointForMount(), "/var")) {
std.log.info("mounted /var", .{});
} else {
_ = runtime.system.write("/system/services/fat: could not mount /var\n");
}
return true;
mounted = true;
}
fn endpointForMount() runtime.ipc.Handle {
return service_endpoint;
}
/// A subscribed process-exit event: release every open handle the dead client
/// held, so a crashed reader can't pin table slots (or, later, locks).
fn onNotification(badge: u64) void {
const got = runtime.ipc.Received{ .len = 0, .badge = badge, .cap = null };
if (got.isTimer()) {
tryBringUp();
if (!mounted) _ = runtime.system.timerOnce(service_endpoint, mount_retry_ms);
return;
}
if (!got.isChildExit()) return;
const dead = got.childProcessId();
var released: u32 = 0;
@@ -171,6 +197,7 @@ fn handleOpen(out: []u8, path: []const u8, flags: u32, sender: u32) usize {
fn onMessage(message: []const u8, out: []u8, sender: u32, capability: ?runtime.ipc.Handle) usize {
_ = capability;
if (!mounted) return fail(out); // storage not up (yet): fail politely, clients retry
if (message.len < protocol.request_size) return fail(out);
const request = std.mem.bytesToValue(protocol.Request, message[0..protocol.request_size]);
const payload = message[protocol.request_size..];