The naive bound `lba + count > r.count` wraps for an lba near u64 max: the sum overflows to a small value, sails under the check, and `base + lba` wraps to an absolute block OUTSIDE the range. Calibrated, it is a real confinement escape — a process confined to [1,3) reads absolute block 0 (the boot sector) with lba = maxInt(u64), since base + lba wraps to 0. The bound is rewritten as two subtractions that cannot overflow: lba within the range, and count within what remains. block-range gains a wrap-refused assertion calibrated to be exploitable against the naive form — it FAILS against the old bound (reads block 0) and passes against the fix (verified by reverting the clamp). Caught pre-emptively before the V2 boundary review.
365 lines
17 KiB
Zig
365 lines
17 KiB
Zig
//! USB mass-storage class driver (Bulk-Only Transport + transparent SCSI).
|
|
//!
|
|
//! Spawned by the device manager when the xHCI bus driver reports a mass-storage
|
|
//! / SCSI / bulk-only interface (class 8, subclass 6, protocol 0x50); its device
|
|
//! id arrives as argv[1]. It owns no hardware: it opens its device through the
|
|
//! USB transfer protocol (`usb`), then drives it with the BOT command
|
|
//! cycle — CBW out, an optional data stage, CSW in — carrying SCSI commands
|
|
//! (READ CAPACITY, READ(10), WRITE(10)). Upward it is a block device: it serves
|
|
//! the block protocol under `.block`, the storage a FAT filesystem sits on.
|
|
//!
|
|
//! Block data never crosses IPC: read/write name a caller-owned DMA buffer by
|
|
//! physical address, which the data stage DMAs straight to/from.
|
|
|
|
const std = @import("std");
|
|
const ipc = @import("ipc");
|
|
const process = @import("process");
|
|
const service = @import("service");
|
|
const time = @import("time");
|
|
const device_manager = @import("driver");
|
|
const memory = @import("memory");
|
|
const logging = @import("logging");
|
|
const usb = @import("usb");
|
|
const scsi = @import("scsi.zig");
|
|
const bot = @import("bulk-only-transport.zig");
|
|
const envelope = @import("envelope");
|
|
const block_protocol = @import("block-protocol");
|
|
|
|
/// The generated block dispatch plus the subscriber machinery the harness owns
|
|
/// (subscribe/unsubscribe, the exit sweep, the fan-out) — usb-storage publishes
|
|
/// `medium_changed`, so it is a Subscribers provider, not a bare Provider. One
|
|
/// device per process, so the handler context is empty.
|
|
const Serve = service.Subscribers(block_protocol.Protocol, void);
|
|
|
|
const Invocation = envelope.Invocation;
|
|
const Answer = envelope.Answer;
|
|
|
|
var device_id: u64 = 0;
|
|
var device: usb.Device = undefined;
|
|
var bulk_in: usb.Endpoint = undefined;
|
|
var bulk_out: usb.Endpoint = undefined;
|
|
|
|
// DMA buffers for the transport: the 31-byte CBW, the 13-byte CSW, and a page
|
|
// for the small command data (INQUIRY / READ CAPACITY / the self-check sector).
|
|
var command_wrapper: memory.DmaRegion = undefined;
|
|
var status_wrapper: memory.DmaRegion = undefined;
|
|
var command_data: memory.DmaRegion = undefined;
|
|
|
|
var next_tag: u32 = 1;
|
|
var block_size: u32 = 512;
|
|
var block_count: u64 = 0;
|
|
|
|
// --- medium presence --------------------------------------------------------
|
|
//
|
|
// A slow TEST UNIT READY poll tracks whether the medium is present; on a
|
|
// transition the driver publishes `medium_changed` to its subscribers (the
|
|
// volume manager). This is the second removal trigger — the DEVICE stays while
|
|
// the MEDIUM leaves (a card reader, an ATAPI tray) — which channel death cannot
|
|
// see (docs/file-system-development/storage-architecture.md). Presence only,
|
|
// never content. A device that is genuinely unplugged is reaped by the device
|
|
// manager instead; a poll failure just before that death publishes absent
|
|
// harmlessly.
|
|
var service_endpoint: ipc.Handle = 0;
|
|
var medium_present: bool = true; // a successful bring-up means the medium is here
|
|
var medium_change_count: u32 = 0;
|
|
const presence_poll_ms = 1000;
|
|
|
|
/// One Bulk-Only-Transport command: send the CBW, run the data stage (to/from
|
|
/// `data_physical`), read and validate the CSW. Returns true on a passed status.
|
|
fn transact(cdb: []const u8, direction_in: bool, data_physical: u64, data_length: u32) bool {
|
|
const tag = next_tag;
|
|
next_tag +%= 1;
|
|
|
|
const wrapper: *bot.CommandBlockWrapper = @ptrFromInt(command_wrapper.virtual);
|
|
wrapper.* = .{
|
|
.tag = tag,
|
|
.data_transfer_length = data_length,
|
|
.flags = if (direction_in) bot.flag_data_in else 0,
|
|
.lun = 0,
|
|
.cdb_length = @intCast(cdb.len),
|
|
};
|
|
@memcpy(wrapper.cdb[0..cdb.len], cdb);
|
|
|
|
if (device.bulk(bulk_out.address, command_wrapper.physical, @sizeOf(bot.CommandBlockWrapper)) == null) return false;
|
|
if (data_length > 0) {
|
|
const endpoint = if (direction_in) bulk_in.address else bulk_out.address;
|
|
if (device.bulk(endpoint, data_physical, data_length) == null) return false;
|
|
}
|
|
if (device.bulk(bulk_in.address, status_wrapper.physical, @sizeOf(bot.CommandStatusWrapper)) == null) return false;
|
|
|
|
const status: *const bot.CommandStatusWrapper = @ptrFromInt(status_wrapper.virtual);
|
|
if (status.signature != bot.csw_signature or status.tag != tag) return false;
|
|
return status.status == @intFromEnum(bot.CommandStatus.passed);
|
|
}
|
|
|
|
/// Set when bring-up failed with the device PRESENT (an opened device that then
|
|
/// failed a step): main exits nonzero, and the device manager restarts us with
|
|
/// backoff — a transient failure heals instead of leaving storage down forever.
|
|
/// Device-absent paths stay clean exits: nothing to serve, nothing to retry.
|
|
var bring_up_failed = false;
|
|
|
|
fn initialise(endpoint: ipc.Handle) bool {
|
|
service_endpoint = endpoint;
|
|
// One hello, both directions: the block-serving endpoint goes UP (the
|
|
// manager routes fat's consumer hello here — this driver serves one
|
|
// volume, one process per stick, so `block` is never a registry name),
|
|
// and the channel to THIS device's controller comes DOWN, routed by
|
|
// lineage. A second stick used to die silently on the exclusive bind;
|
|
// now every instance is reachable through its lineage.
|
|
const bus = device_manager.helloForChannel(.device, device_id, endpoint) orelse return false;
|
|
device = usb.open(bus, device_id) orelse {
|
|
std.log.info("could not open device {d}", .{device_id});
|
|
return false;
|
|
};
|
|
bulk_in = device.findEndpoint(usb.transfer_type_bulk, true) orelse {
|
|
_ = logging.write("/system/drivers/usb-storage: no bulk-IN endpoint\n");
|
|
bring_up_failed = true;
|
|
return false;
|
|
};
|
|
bulk_out = device.findEndpoint(usb.transfer_type_bulk, false) orelse {
|
|
_ = logging.write("/system/drivers/usb-storage: no bulk-OUT endpoint\n");
|
|
bring_up_failed = true;
|
|
return false;
|
|
};
|
|
// Shareable, so each buffer's capability can be handed to the controller: usb-storage
|
|
// owns no device, so its buffers are not auto-bound anywhere — the controller reaches
|
|
// them only once attached. (No-op binding when no IOMMU is enforcing.)
|
|
command_wrapper = memory.dmaAlloc(4096, memory.dma_coherent | memory.dma_shareable) orelse return false;
|
|
status_wrapper = memory.dmaAlloc(4096, memory.dma_coherent | memory.dma_shareable) orelse return false;
|
|
command_data = memory.dmaAlloc(4096, memory.dma_coherent | memory.dma_shareable) orelse return false;
|
|
for ([_]memory.DmaRegion{ command_wrapper, status_wrapper, command_data }) |region| {
|
|
if (region.handle) |handle| {
|
|
if (!device.attachDma(handle)) {
|
|
_ = logging.write("/system/drivers/usb-storage: could not attach a DMA buffer to the controller\n");
|
|
bring_up_failed = true;
|
|
return false;
|
|
}
|
|
_ = ipc.close(handle); // the binding holds its own reference now
|
|
}
|
|
}
|
|
|
|
// Bring the LUN up: wait for it to be ready (clearing the initial unit-attention
|
|
// with REQUEST SENSE), identify it, and read its capacity.
|
|
var tries: u32 = 0;
|
|
while (tries < 10) : (tries += 1) {
|
|
const ready = scsi.testUnitReady();
|
|
if (transact(&ready, false, 0, 0)) break;
|
|
const sense = scsi.requestSense(18);
|
|
_ = transact(&sense, true, command_data.physical, 18);
|
|
time.sleepMillis(50);
|
|
}
|
|
const inquiry = scsi.inquiry(36);
|
|
_ = transact(&inquiry, true, command_data.physical, 36);
|
|
|
|
const capacity_command = scsi.readCapacity10();
|
|
if (!transact(&capacity_command, true, command_data.physical, 8)) {
|
|
_ = logging.write("/system/drivers/usb-storage: READ CAPACITY failed\n");
|
|
bring_up_failed = true;
|
|
return false;
|
|
}
|
|
var capacity_bytes: [8]u8 = undefined;
|
|
const capacity_source: [*]const u8 = @ptrFromInt(command_data.virtual);
|
|
@memcpy(&capacity_bytes, capacity_source[0..8]);
|
|
const capacity = scsi.parseCapacity(capacity_bytes);
|
|
block_size = capacity.block_size;
|
|
block_count = @as(u64, capacity.last_lba) + 1;
|
|
std.log.info("ready ({d} blocks x {d} bytes)", .{ block_count, block_size });
|
|
|
|
// Self-check: read block 0 and log its trailing signature (0x55AA for a boot
|
|
// sector) — proof READ(10) works end to end over the bulk path.
|
|
const read0 = scsi.read10(0, 1);
|
|
if (block_size <= 4096 and transact(&read0, true, command_data.physical, block_size)) {
|
|
const sector: [*]const u8 = @ptrFromInt(command_data.virtual);
|
|
std.log.info("block 0 signature 0x{x:0>2}{x:0>2}", .{ sector[510], sector[511] });
|
|
}
|
|
// Bring-up succeeded, so the medium is present; start the presence poll.
|
|
_ = time.timerOnce(endpoint, presence_poll_ms);
|
|
return true;
|
|
}
|
|
|
|
// --- serving the block protocol ---------------------------------------------
|
|
//
|
|
// Geometry, and whole-block read/write to and from the caller's DMA buffer
|
|
// (named by physical address). One device per process, so `Header.target` is
|
|
// always 0 and no handler reads it.
|
|
|
|
/// A transfer the device refused. Every failure here is the same one — the SCSI
|
|
/// command did not complete — so there is one errno for all of them.
|
|
const refused: isize = -envelope.ENOENT;
|
|
|
|
// --- per-sender range confinement -------------------------------------------
|
|
//
|
|
// The volume manager confines each filesystem to the partition it mounts
|
|
// (define_range); a confined sender addresses volume-relative LBAs from 0 and
|
|
// the driver translates and bounds-checks against its range. A sender with no
|
|
// range is unconfined — the whole device — which is the default until a range
|
|
// is defined (behaviour-neutral for a single-volume boot), and is what the
|
|
// volume manager itself uses to probe partitions before it confines anyone.
|
|
|
|
/// bound: filesystem processes confined to sub-ranges of this device at once
|
|
/// decided-by: ours
|
|
/// protects: the per-badge range table below
|
|
/// at-limit: refuse - define_range past it returns -ENOSPC; a runaway detector for
|
|
/// a compromised volume manager, not a real-partition limit (real disks carry a
|
|
/// handful of volumes, far under this)
|
|
/// observed-by: the -ENOSPC a define_range caller gets when the table is full
|
|
const maximum_ranges = 64;
|
|
|
|
const Range = struct { used: bool = false, badge: u32 = 0, base: u64 = 0, count: u64 = 0 };
|
|
var ranges = [_]Range{.{}} ** maximum_ranges;
|
|
|
|
fn rangeFor(badge: u32) ?*Range {
|
|
for (&ranges) |*r| {
|
|
if (r.used and r.badge == badge) return r;
|
|
}
|
|
return null;
|
|
}
|
|
|
|
/// Resolve a caller's transfer to an absolute LBA, or null if it falls outside
|
|
/// the caller's confinement. Unconfined callers (no range) pass through against
|
|
/// the whole device.
|
|
///
|
|
/// The bound is written to survive a hostile confined caller: `lba + count`
|
|
/// would WRAP for an `lba` near u64 max, sail under a naive `> r.count` check,
|
|
/// and translate to a wild absolute block — so the check is phrased as two
|
|
/// subtractions that cannot overflow (`lba` within the range, and `count`
|
|
/// within what remains). `r.base + lba` cannot overflow once `lba <= r.count`,
|
|
/// because the volume manager sets `base + count` inside the device.
|
|
fn resolveTransfer(sender: u32, lba: u64, count: u32) ?u64 {
|
|
const r = rangeFor(sender) orelse return lba; // unconfined: whole device
|
|
if (lba > r.count or r.count - lba < count) return null; // past the volume's end
|
|
return r.base + lba;
|
|
}
|
|
|
|
fn onGeometry(_: void, invocation: Invocation(void), answer: Answer(block_protocol.Geometry)) isize {
|
|
// A confined caller sees ITS volume's size, not the device's — so a
|
|
// filesystem mounts against the geometry it is actually allowed to touch.
|
|
if (rangeFor(invocation.sender)) |r| {
|
|
answer.set(.{ .block_size = block_size, .block_count = r.count });
|
|
return 0;
|
|
}
|
|
answer.set(.{ .block_size = block_size, .block_count = block_count });
|
|
return 0;
|
|
}
|
|
|
|
fn onRead(_: void, invocation: Invocation(block_protocol.Transfer), answer: Answer(block_protocol.Transferred)) isize {
|
|
const request = invocation.request;
|
|
const abs = resolveTransfer(invocation.sender, request.lba, request.count) orelse return refused;
|
|
const cdb = scsi.read10(@intCast(abs), @intCast(request.count));
|
|
if (!transact(&cdb, true, request.physical, request.count * block_size)) return refused;
|
|
answer.set(.{ .count = request.count });
|
|
return 0;
|
|
}
|
|
|
|
fn onWrite(_: void, invocation: Invocation(block_protocol.Transfer), answer: Answer(block_protocol.Transferred)) isize {
|
|
const request = invocation.request;
|
|
const abs = resolveTransfer(invocation.sender, request.lba, request.count) orelse return refused;
|
|
const cdb = scsi.write10(@intCast(abs), @intCast(request.count));
|
|
if (!transact(&cdb, false, request.physical, request.count * block_size)) return refused;
|
|
answer.set(.{ .count = request.count });
|
|
return 0;
|
|
}
|
|
|
|
/// Confine a sender to a block sub-range (the volume manager's per-volume grant).
|
|
/// Refused if the CALLER is itself confined — a filesystem cannot widen its own
|
|
/// range or confine anyone; only an unconfined party (the volume manager) may.
|
|
fn onDefineRange(_: void, invocation: Invocation(block_protocol.DefineRange), _: Answer(void)) isize {
|
|
if (rangeFor(invocation.sender) != null) return -envelope.EPERM;
|
|
const request = invocation.request;
|
|
const slot = rangeFor(request.badge) orelse free: {
|
|
for (&ranges) |*r| {
|
|
if (!r.used) break :free r;
|
|
}
|
|
break :free null;
|
|
} orelse return -envelope.ENOSPC;
|
|
slot.* = .{ .used = true, .badge = request.badge, .base = request.base_lba, .count = request.block_count };
|
|
return 0;
|
|
}
|
|
|
|
/// SYNCHRONIZE CACHE: commit the device's write cache to flash. No data stage.
|
|
/// Makes prior writes durable before a caller (init at shutdown) cuts power. A
|
|
/// device without a volatile cache reports success anyway.
|
|
fn onFlush(_: void, _: Invocation(void), _: Answer(void)) isize {
|
|
const cdb = scsi.synchronizeCache10();
|
|
return if (transact(&cdb, false, 0, 0)) 0 else refused;
|
|
}
|
|
|
|
/// The filesystem's DMA buffer: forward its capability to the controller so the
|
|
/// device can reach it. Never claimed — the binding holds its own reference, so
|
|
/// our copy is the turn's to close, on this path and on the refusal alike.
|
|
fn onAttach(_: void, invocation: Invocation(void), _: Answer(void)) isize {
|
|
const handle = invocation.capability orelse return -envelope.EPROTO;
|
|
return if (device.attachDma(handle)) 0 else refused;
|
|
}
|
|
|
|
/// The reverse: forward the same region capability so the controller unbinds
|
|
/// the buffer. As with attach, our copy stays the turn's to close.
|
|
fn onDetach(_: void, invocation: Invocation(void), _: Answer(void)) isize {
|
|
const handle = invocation.capability orelse return -envelope.EPROTO;
|
|
return if (device.detachDma(handle)) 0 else refused;
|
|
}
|
|
|
|
const handlers = Serve.Handlers{
|
|
.geometry = onGeometry,
|
|
.read = onRead,
|
|
.write = onWrite,
|
|
.flush = onFlush,
|
|
.attach = onAttach,
|
|
.detach = onDetach,
|
|
.define_range = onDefineRange,
|
|
};
|
|
|
|
fn onMessage(message: []const u8, reply: []u8, sender: u32, arrived: *ipc.Arrival) usize {
|
|
// The harness peeks the arrival and takes it only if a handler (subscribe)
|
|
// claimed it; attach/detach forward the capability without claiming, so the
|
|
// turn still closes their copy after the controller took its own reference.
|
|
return Serve.dispatch({}, handlers, message, sender, arrived, reply);
|
|
}
|
|
|
|
/// A slow TEST UNIT READY poll: success means the medium is present, failure
|
|
/// means it is not. On a transition, bump the counter and publish. (Sense-key
|
|
/// inspection to tell "medium absent" from other transport errors is a
|
|
/// refinement; a clean eject — what QEMU and a card reader produce — makes
|
|
/// TEST UNIT READY report not-ready, which this reads correctly.)
|
|
fn pollPresence() void {
|
|
const ready = scsi.testUnitReady();
|
|
const now = transact(&ready, false, 0, 0);
|
|
if (now == medium_present) return;
|
|
medium_present = now;
|
|
medium_change_count +%= 1;
|
|
std.log.info("medium {s}", .{if (now) "present" else "absent"});
|
|
Serve.publish(.medium_changed, 0, .{ .present = @intFromBool(now), .change_count = medium_change_count });
|
|
}
|
|
|
|
fn onNotification(badge: u64) void {
|
|
const got = ipc.Received{ .len = 0, .badge = badge, .cap = null };
|
|
if (got.isTimer()) {
|
|
pollPresence();
|
|
_ = time.timerOnce(service_endpoint, presence_poll_ms);
|
|
}
|
|
// Subscriber deaths are swept by the harness (Serve.hooks); nothing else here.
|
|
}
|
|
|
|
pub fn main(init: process.Init) void {
|
|
const argument = init.arguments.get(1) orelse {
|
|
_ = logging.write("/system/drivers/usb-storage: missing device id (argv[1])\n");
|
|
return;
|
|
};
|
|
device_id = std.fmt.parseInt(u64, argument, 10) catch {
|
|
std.log.info("malformed device id '{s}'", .{argument});
|
|
return;
|
|
};
|
|
// No `.service` name: several instances provide the block contract (one
|
|
// per stick), so consumers are routed here by the device manager's
|
|
// lineage, never by a registry bind — the endpoint goes up in the hello.
|
|
service.run(block_protocol.message_maximum, .{
|
|
.init = initialise,
|
|
.on_message = onMessage,
|
|
.on_notification = onNotification,
|
|
.subscribers = Serve.hooks,
|
|
});
|
|
// A failure exit (nonzero -> .aborted) tells the device manager to restart
|
|
// us with backoff; a clean return means there was nothing to serve.
|
|
if (bring_up_failed) process.exit(1);
|
|
}
|