block: per-sender range confinement — V2a mechanism

The block protocol gains define_range (appended, numbers hold): confine the
process named by `badge` to blocks [base, base+count). usb-storage keeps a
per-badge range table and, in read/write, translates volume-relative LBAs
(base added) and refuses any transfer past the volume end. geometry returns
the confined size, so a filesystem mounts against what it may actually touch.

The security seam (decision 4, settled): the clamp lives at the PROVIDER, so
a channel carries exactly the authority it grants — handing a filesystem the
whole disk plus a base offset would let it reach the neighbouring partition.
The gate: a confined caller may NOT call define_range, so a filesystem cannot
widen its own range or confine anyone; only an unconfined party (the volume
manager, whole-device) may. The volume manager defines a filesystem's range
before handing it the channel, so the ordering holds by construction.

Default (no range for a badge) is the whole device — behaviour-neutral for a
single-volume boot and what the volume manager itself uses to probe
partitions. The range table is declared through bounds.md as a runaway
detector (ours, refuse at limit), not a real-partition cap. Neutral:
fat-mount, usb-storage, iommu-usb-storage green. The discrimination fixture
(a confined process reads past its range and is refused) follows next.
This commit is contained in:
Daniel Samson
2026-08-09 17:13:26 +01:00
parent 7ea54a84e5
commit f1bdce25e0
3 changed files with 98 additions and 3 deletions
+9
View File
@@ -63,6 +63,15 @@ pub const Device = struct {
return self.call(.flush, {}, null, &reply) != null;
}
/// Confine the process `badge` to blocks `[base_lba, base_lba + block_count)`
/// on this device — the volume manager's per-volume grant to a filesystem.
/// The caller must itself be unconfined (whole-device); a confined caller is
/// refused, so a filesystem cannot widen its own range. See `DefineRange`.
pub fn defineRange(self: Device, badge: u32, base_lba: u64, block_count: u64) bool {
var reply: [block_protocol.message_maximum]u8 = undefined;
return self.call(.define_range, .{ .badge = badge, .base_lba = base_lba, .block_count = block_count }, null, &reply) != null;
}
/// One request at the driver. `target` is always 0: one endpoint per device, so
/// there is no object within the peer to address.
fn call(
+24
View File
@@ -37,6 +37,26 @@ pub const Transfer = extern struct {
/// How many blocks a transfer actually moved.
pub const Transferred = extern struct { count: u32 };
/// `define_range(badge, base_lba, block_count)`: confine the sender identified by
/// `badge` to blocks `[base_lba, base_lba + block_count)`. The volume manager
/// calls this for each filesystem process it hands a channel to — the badge is
/// the filesystem's kernel-stamped task id, and the range is the partition it
/// mounts. A confined sender's read/write LBAs are then volume-relative (the
/// driver adds `base_lba`) and a transfer past `block_count` is refused. A
/// sender with no range is unconfined (the whole device), the default until the
/// volume manager defines one. The clamp lives at the provider because a channel
/// must carry exactly the authority it grants (storage-architecture.md): handing
/// a filesystem the whole disk plus a base offset would let it reach the
/// neighbouring partition. **A confined caller may not call this** — a filesystem
/// cannot redefine its own range and escape; only an unconfined party (the
/// volume manager) confines others.
pub const DefineRange = extern struct {
badge: u32,
_padding: u32 = 0,
base_lba: u64,
block_count: u64,
};
pub const Protocol = envelope.Define(.{
.name = "block",
.version = 1,
@@ -60,6 +80,10 @@ pub const Protocol = envelope.Define(.{
// makes is revocable by the granter while alive; death remains the
// mechanical backstop (storage-architecture.md, the lifecycle rule).
.{ .name = "detach" },
// define_range(): confine a sender to a block sub-range — the partition
// it mounts. See `DefineRange`. Appended, so every verb above keeps its
// number.
.{ .name = "define_range", .request = DefineRange },
},
});
+65 -3
View File
@@ -167,14 +167,58 @@ fn initialise(endpoint: ipc.Handle) bool {
/// command did not complete — so there is one errno for all of them.
const refused: isize = -envelope.ENOENT;
fn onGeometry(_: void, _: Invocation(void), answer: Answer(block_protocol.Geometry)) isize {
// --- per-sender range confinement -------------------------------------------
//
// The volume manager confines each filesystem to the partition it mounts
// (define_range); a confined sender addresses volume-relative LBAs from 0 and
// the driver translates and bounds-checks against its range. A sender with no
// range is unconfined — the whole device — which is the default until a range
// is defined (behaviour-neutral for a single-volume boot), and is what the
// volume manager itself uses to probe partitions before it confines anyone.
/// bound: filesystem processes confined to sub-ranges of this device at once
/// decided-by: ours
/// protects: the per-badge range table below
/// at-limit: refuse - define_range past it returns -ENOSPC; a runaway detector for
/// a compromised volume manager, not a real-partition limit (real disks carry a
/// handful of volumes, far under this)
/// observed-by: the -ENOSPC a define_range caller gets when the table is full
const maximum_ranges = 64;
const Range = struct { used: bool = false, badge: u32 = 0, base: u64 = 0, count: u64 = 0 };
var ranges = [_]Range{.{}} ** maximum_ranges;
fn rangeFor(badge: u32) ?*Range {
for (&ranges) |*r| {
if (r.used and r.badge == badge) return r;
}
return null;
}
/// Resolve a caller's transfer to an absolute LBA, or null if it falls outside
/// the caller's confinement. Unconfined callers (no range) pass through against
/// the whole device.
fn resolveTransfer(sender: u32, lba: u64, count: u32) ?u64 {
const r = rangeFor(sender) orelse return lba; // unconfined: whole device
if (lba + count > r.count) return null; // past the volume's end
return r.base + lba;
}
fn onGeometry(_: void, invocation: Invocation(void), answer: Answer(block_protocol.Geometry)) isize {
// A confined caller sees ITS volume's size, not the device's — so a
// filesystem mounts against the geometry it is actually allowed to touch.
if (rangeFor(invocation.sender)) |r| {
answer.set(.{ .block_size = block_size, .block_count = r.count });
return 0;
}
answer.set(.{ .block_size = block_size, .block_count = block_count });
return 0;
}
fn onRead(_: void, invocation: Invocation(block_protocol.Transfer), answer: Answer(block_protocol.Transferred)) isize {
const request = invocation.request;
const cdb = scsi.read10(@intCast(request.lba), @intCast(request.count));
const abs = resolveTransfer(invocation.sender, request.lba, request.count) orelse return refused;
const cdb = scsi.read10(@intCast(abs), @intCast(request.count));
if (!transact(&cdb, true, request.physical, request.count * block_size)) return refused;
answer.set(.{ .count = request.count });
return 0;
@@ -182,12 +226,29 @@ fn onRead(_: void, invocation: Invocation(block_protocol.Transfer), answer: Answ
fn onWrite(_: void, invocation: Invocation(block_protocol.Transfer), answer: Answer(block_protocol.Transferred)) isize {
const request = invocation.request;
const cdb = scsi.write10(@intCast(request.lba), @intCast(request.count));
const abs = resolveTransfer(invocation.sender, request.lba, request.count) orelse return refused;
const cdb = scsi.write10(@intCast(abs), @intCast(request.count));
if (!transact(&cdb, false, request.physical, request.count * block_size)) return refused;
answer.set(.{ .count = request.count });
return 0;
}
/// Confine a sender to a block sub-range (the volume manager's per-volume grant).
/// Refused if the CALLER is itself confined — a filesystem cannot widen its own
/// range or confine anyone; only an unconfined party (the volume manager) may.
fn onDefineRange(_: void, invocation: Invocation(block_protocol.DefineRange), _: Answer(void)) isize {
if (rangeFor(invocation.sender) != null) return -envelope.EPERM;
const request = invocation.request;
const slot = rangeFor(request.badge) orelse free: {
for (&ranges) |*r| {
if (!r.used) break :free r;
}
break :free null;
} orelse return -envelope.ENOSPC;
slot.* = .{ .used = true, .badge = request.badge, .base = request.base_lba, .count = request.block_count };
return 0;
}
/// SYNCHRONIZE CACHE: commit the device's write cache to flash. No data stage.
/// Makes prior writes durable before a caller (init at shutdown) cuts power. A
/// device without a volatile cache reports success anyway.
@@ -218,6 +279,7 @@ const handlers = Serve.Handlers{
.flush = onFlush,
.attach = onAttach,
.detach = onDetach,
.define_range = onDefineRange,
};
fn onMessage(message: []const u8, reply: []u8, sender: u32, arrived: *ipc.Arrival) usize {