block: per-sender range confinement — V2a mechanism
The block protocol gains define_range (appended, numbers hold): confine the process named by `badge` to blocks [base, base+count). usb-storage keeps a per-badge range table and, in read/write, translates volume-relative LBAs (base added) and refuses any transfer past the volume end. geometry returns the confined size, so a filesystem mounts against what it may actually touch. The security seam (decision 4, settled): the clamp lives at the PROVIDER, so a channel carries exactly the authority it grants — handing a filesystem the whole disk plus a base offset would let it reach the neighbouring partition. The gate: a confined caller may NOT call define_range, so a filesystem cannot widen its own range or confine anyone; only an unconfined party (the volume manager, whole-device) may. The volume manager defines a filesystem's range before handing it the channel, so the ordering holds by construction. Default (no range for a badge) is the whole device — behaviour-neutral for a single-volume boot and what the volume manager itself uses to probe partitions. The range table is declared through bounds.md as a runaway detector (ours, refuse at limit), not a real-partition cap. Neutral: fat-mount, usb-storage, iommu-usb-storage green. The discrimination fixture (a confined process reads past its range and is refused) follows next.
This commit is contained in:
@@ -167,14 +167,58 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
/// command did not complete — so there is one errno for all of them.
|
||||
const refused: isize = -envelope.ENOENT;
|
||||
|
||||
fn onGeometry(_: void, _: Invocation(void), answer: Answer(block_protocol.Geometry)) isize {
|
||||
// --- per-sender range confinement -------------------------------------------
|
||||
//
|
||||
// The volume manager confines each filesystem to the partition it mounts
|
||||
// (define_range); a confined sender addresses volume-relative LBAs from 0 and
|
||||
// the driver translates and bounds-checks against its range. A sender with no
|
||||
// range is unconfined — the whole device — which is the default until a range
|
||||
// is defined (behaviour-neutral for a single-volume boot), and is what the
|
||||
// volume manager itself uses to probe partitions before it confines anyone.
|
||||
|
||||
/// bound: filesystem processes confined to sub-ranges of this device at once
|
||||
/// decided-by: ours
|
||||
/// protects: the per-badge range table below
|
||||
/// at-limit: refuse - define_range past it returns -ENOSPC; a runaway detector for
|
||||
/// a compromised volume manager, not a real-partition limit (real disks carry a
|
||||
/// handful of volumes, far under this)
|
||||
/// observed-by: the -ENOSPC a define_range caller gets when the table is full
|
||||
const maximum_ranges = 64;
|
||||
|
||||
const Range = struct { used: bool = false, badge: u32 = 0, base: u64 = 0, count: u64 = 0 };
|
||||
var ranges = [_]Range{.{}} ** maximum_ranges;
|
||||
|
||||
fn rangeFor(badge: u32) ?*Range {
|
||||
for (&ranges) |*r| {
|
||||
if (r.used and r.badge == badge) return r;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Resolve a caller's transfer to an absolute LBA, or null if it falls outside
|
||||
/// the caller's confinement. Unconfined callers (no range) pass through against
|
||||
/// the whole device.
|
||||
fn resolveTransfer(sender: u32, lba: u64, count: u32) ?u64 {
|
||||
const r = rangeFor(sender) orelse return lba; // unconfined: whole device
|
||||
if (lba + count > r.count) return null; // past the volume's end
|
||||
return r.base + lba;
|
||||
}
|
||||
|
||||
fn onGeometry(_: void, invocation: Invocation(void), answer: Answer(block_protocol.Geometry)) isize {
|
||||
// A confined caller sees ITS volume's size, not the device's — so a
|
||||
// filesystem mounts against the geometry it is actually allowed to touch.
|
||||
if (rangeFor(invocation.sender)) |r| {
|
||||
answer.set(.{ .block_size = block_size, .block_count = r.count });
|
||||
return 0;
|
||||
}
|
||||
answer.set(.{ .block_size = block_size, .block_count = block_count });
|
||||
return 0;
|
||||
}
|
||||
|
||||
fn onRead(_: void, invocation: Invocation(block_protocol.Transfer), answer: Answer(block_protocol.Transferred)) isize {
|
||||
const request = invocation.request;
|
||||
const cdb = scsi.read10(@intCast(request.lba), @intCast(request.count));
|
||||
const abs = resolveTransfer(invocation.sender, request.lba, request.count) orelse return refused;
|
||||
const cdb = scsi.read10(@intCast(abs), @intCast(request.count));
|
||||
if (!transact(&cdb, true, request.physical, request.count * block_size)) return refused;
|
||||
answer.set(.{ .count = request.count });
|
||||
return 0;
|
||||
@@ -182,12 +226,29 @@ fn onRead(_: void, invocation: Invocation(block_protocol.Transfer), answer: Answ
|
||||
|
||||
fn onWrite(_: void, invocation: Invocation(block_protocol.Transfer), answer: Answer(block_protocol.Transferred)) isize {
|
||||
const request = invocation.request;
|
||||
const cdb = scsi.write10(@intCast(request.lba), @intCast(request.count));
|
||||
const abs = resolveTransfer(invocation.sender, request.lba, request.count) orelse return refused;
|
||||
const cdb = scsi.write10(@intCast(abs), @intCast(request.count));
|
||||
if (!transact(&cdb, false, request.physical, request.count * block_size)) return refused;
|
||||
answer.set(.{ .count = request.count });
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// Confine a sender to a block sub-range (the volume manager's per-volume grant).
|
||||
/// Refused if the CALLER is itself confined — a filesystem cannot widen its own
|
||||
/// range or confine anyone; only an unconfined party (the volume manager) may.
|
||||
fn onDefineRange(_: void, invocation: Invocation(block_protocol.DefineRange), _: Answer(void)) isize {
|
||||
if (rangeFor(invocation.sender) != null) return -envelope.EPERM;
|
||||
const request = invocation.request;
|
||||
const slot = rangeFor(request.badge) orelse free: {
|
||||
for (&ranges) |*r| {
|
||||
if (!r.used) break :free r;
|
||||
}
|
||||
break :free null;
|
||||
} orelse return -envelope.ENOSPC;
|
||||
slot.* = .{ .used = true, .badge = request.badge, .base = request.base_lba, .count = request.block_count };
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// SYNCHRONIZE CACHE: commit the device's write cache to flash. No data stage.
|
||||
/// Makes prior writes durable before a caller (init at shutdown) cuts power. A
|
||||
/// device without a volatile cache reports success anyway.
|
||||
@@ -218,6 +279,7 @@ const handlers = Serve.Handlers{
|
||||
.flush = onFlush,
|
||||
.attach = onAttach,
|
||||
.detach = onDetach,
|
||||
.define_range = onDefineRange,
|
||||
};
|
||||
|
||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, arrived: *ipc.Arrival) usize {
|
||||
|
||||
Reference in New Issue
Block a user