block: per-sender range confinement — V2a mechanism
The block protocol gains define_range (appended, numbers hold): confine the process named by `badge` to blocks [base, base+count). usb-storage keeps a per-badge range table and, in read/write, translates volume-relative LBAs (base added) and refuses any transfer past the volume end. geometry returns the confined size, so a filesystem mounts against what it may actually touch. The security seam (decision 4, settled): the clamp lives at the PROVIDER, so a channel carries exactly the authority it grants — handing a filesystem the whole disk plus a base offset would let it reach the neighbouring partition. The gate: a confined caller may NOT call define_range, so a filesystem cannot widen its own range or confine anyone; only an unconfined party (the volume manager, whole-device) may. The volume manager defines a filesystem's range before handing it the channel, so the ordering holds by construction. Default (no range for a badge) is the whole device — behaviour-neutral for a single-volume boot and what the volume manager itself uses to probe partitions. The range table is declared through bounds.md as a runaway detector (ours, refuse at limit), not a real-partition cap. Neutral: fat-mount, usb-storage, iommu-usb-storage green. The discrimination fixture (a confined process reads past its range and is refused) follows next.
This commit is contained in:
@@ -63,6 +63,15 @@ pub const Device = struct {
|
||||
return self.call(.flush, {}, null, &reply) != null;
|
||||
}
|
||||
|
||||
/// Confine the process `badge` to blocks `[base_lba, base_lba + block_count)`
|
||||
/// on this device — the volume manager's per-volume grant to a filesystem.
|
||||
/// The caller must itself be unconfined (whole-device); a confined caller is
|
||||
/// refused, so a filesystem cannot widen its own range. See `DefineRange`.
|
||||
pub fn defineRange(self: Device, badge: u32, base_lba: u64, block_count: u64) bool {
|
||||
var reply: [block_protocol.message_maximum]u8 = undefined;
|
||||
return self.call(.define_range, .{ .badge = badge, .base_lba = base_lba, .block_count = block_count }, null, &reply) != null;
|
||||
}
|
||||
|
||||
/// One request at the driver. `target` is always 0: one endpoint per device, so
|
||||
/// there is no object within the peer to address.
|
||||
fn call(
|
||||
|
||||
@@ -37,6 +37,26 @@ pub const Transfer = extern struct {
|
||||
/// How many blocks a transfer actually moved.
|
||||
pub const Transferred = extern struct { count: u32 };
|
||||
|
||||
/// `define_range(badge, base_lba, block_count)`: confine the sender identified by
|
||||
/// `badge` to blocks `[base_lba, base_lba + block_count)`. The volume manager
|
||||
/// calls this for each filesystem process it hands a channel to — the badge is
|
||||
/// the filesystem's kernel-stamped task id, and the range is the partition it
|
||||
/// mounts. A confined sender's read/write LBAs are then volume-relative (the
|
||||
/// driver adds `base_lba`) and a transfer past `block_count` is refused. A
|
||||
/// sender with no range is unconfined (the whole device), the default until the
|
||||
/// volume manager defines one. The clamp lives at the provider because a channel
|
||||
/// must carry exactly the authority it grants (storage-architecture.md): handing
|
||||
/// a filesystem the whole disk plus a base offset would let it reach the
|
||||
/// neighbouring partition. **A confined caller may not call this** — a filesystem
|
||||
/// cannot redefine its own range and escape; only an unconfined party (the
|
||||
/// volume manager) confines others.
|
||||
pub const DefineRange = extern struct {
|
||||
badge: u32,
|
||||
_padding: u32 = 0,
|
||||
base_lba: u64,
|
||||
block_count: u64,
|
||||
};
|
||||
|
||||
pub const Protocol = envelope.Define(.{
|
||||
.name = "block",
|
||||
.version = 1,
|
||||
@@ -60,6 +80,10 @@ pub const Protocol = envelope.Define(.{
|
||||
// makes is revocable by the granter while alive; death remains the
|
||||
// mechanical backstop (storage-architecture.md, the lifecycle rule).
|
||||
.{ .name = "detach" },
|
||||
// define_range(): confine a sender to a block sub-range — the partition
|
||||
// it mounts. See `DefineRange`. Appended, so every verb above keeps its
|
||||
// number.
|
||||
.{ .name = "define_range", .request = DefineRange },
|
||||
},
|
||||
});
|
||||
|
||||
|
||||
@@ -167,14 +167,58 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
/// command did not complete — so there is one errno for all of them.
|
||||
const refused: isize = -envelope.ENOENT;
|
||||
|
||||
fn onGeometry(_: void, _: Invocation(void), answer: Answer(block_protocol.Geometry)) isize {
|
||||
// --- per-sender range confinement -------------------------------------------
|
||||
//
|
||||
// The volume manager confines each filesystem to the partition it mounts
|
||||
// (define_range); a confined sender addresses volume-relative LBAs from 0 and
|
||||
// the driver translates and bounds-checks against its range. A sender with no
|
||||
// range is unconfined — the whole device — which is the default until a range
|
||||
// is defined (behaviour-neutral for a single-volume boot), and is what the
|
||||
// volume manager itself uses to probe partitions before it confines anyone.
|
||||
|
||||
/// bound: filesystem processes confined to sub-ranges of this device at once
|
||||
/// decided-by: ours
|
||||
/// protects: the per-badge range table below
|
||||
/// at-limit: refuse - define_range past it returns -ENOSPC; a runaway detector for
|
||||
/// a compromised volume manager, not a real-partition limit (real disks carry a
|
||||
/// handful of volumes, far under this)
|
||||
/// observed-by: the -ENOSPC a define_range caller gets when the table is full
|
||||
const maximum_ranges = 64;
|
||||
|
||||
const Range = struct { used: bool = false, badge: u32 = 0, base: u64 = 0, count: u64 = 0 };
|
||||
var ranges = [_]Range{.{}} ** maximum_ranges;
|
||||
|
||||
fn rangeFor(badge: u32) ?*Range {
|
||||
for (&ranges) |*r| {
|
||||
if (r.used and r.badge == badge) return r;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Resolve a caller's transfer to an absolute LBA, or null if it falls outside
|
||||
/// the caller's confinement. Unconfined callers (no range) pass through against
|
||||
/// the whole device.
|
||||
fn resolveTransfer(sender: u32, lba: u64, count: u32) ?u64 {
|
||||
const r = rangeFor(sender) orelse return lba; // unconfined: whole device
|
||||
if (lba + count > r.count) return null; // past the volume's end
|
||||
return r.base + lba;
|
||||
}
|
||||
|
||||
fn onGeometry(_: void, invocation: Invocation(void), answer: Answer(block_protocol.Geometry)) isize {
|
||||
// A confined caller sees ITS volume's size, not the device's — so a
|
||||
// filesystem mounts against the geometry it is actually allowed to touch.
|
||||
if (rangeFor(invocation.sender)) |r| {
|
||||
answer.set(.{ .block_size = block_size, .block_count = r.count });
|
||||
return 0;
|
||||
}
|
||||
answer.set(.{ .block_size = block_size, .block_count = block_count });
|
||||
return 0;
|
||||
}
|
||||
|
||||
fn onRead(_: void, invocation: Invocation(block_protocol.Transfer), answer: Answer(block_protocol.Transferred)) isize {
|
||||
const request = invocation.request;
|
||||
const cdb = scsi.read10(@intCast(request.lba), @intCast(request.count));
|
||||
const abs = resolveTransfer(invocation.sender, request.lba, request.count) orelse return refused;
|
||||
const cdb = scsi.read10(@intCast(abs), @intCast(request.count));
|
||||
if (!transact(&cdb, true, request.physical, request.count * block_size)) return refused;
|
||||
answer.set(.{ .count = request.count });
|
||||
return 0;
|
||||
@@ -182,12 +226,29 @@ fn onRead(_: void, invocation: Invocation(block_protocol.Transfer), answer: Answ
|
||||
|
||||
fn onWrite(_: void, invocation: Invocation(block_protocol.Transfer), answer: Answer(block_protocol.Transferred)) isize {
|
||||
const request = invocation.request;
|
||||
const cdb = scsi.write10(@intCast(request.lba), @intCast(request.count));
|
||||
const abs = resolveTransfer(invocation.sender, request.lba, request.count) orelse return refused;
|
||||
const cdb = scsi.write10(@intCast(abs), @intCast(request.count));
|
||||
if (!transact(&cdb, false, request.physical, request.count * block_size)) return refused;
|
||||
answer.set(.{ .count = request.count });
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// Confine a sender to a block sub-range (the volume manager's per-volume grant).
|
||||
/// Refused if the CALLER is itself confined — a filesystem cannot widen its own
|
||||
/// range or confine anyone; only an unconfined party (the volume manager) may.
|
||||
fn onDefineRange(_: void, invocation: Invocation(block_protocol.DefineRange), _: Answer(void)) isize {
|
||||
if (rangeFor(invocation.sender) != null) return -envelope.EPERM;
|
||||
const request = invocation.request;
|
||||
const slot = rangeFor(request.badge) orelse free: {
|
||||
for (&ranges) |*r| {
|
||||
if (!r.used) break :free r;
|
||||
}
|
||||
break :free null;
|
||||
} orelse return -envelope.ENOSPC;
|
||||
slot.* = .{ .used = true, .badge = request.badge, .base = request.base_lba, .count = request.block_count };
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// SYNCHRONIZE CACHE: commit the device's write cache to flash. No data stage.
|
||||
/// Makes prior writes durable before a caller (init at shutdown) cuts power. A
|
||||
/// device without a volatile cache reports success anyway.
|
||||
@@ -218,6 +279,7 @@ const handlers = Serve.Handlers{
|
||||
.flush = onFlush,
|
||||
.attach = onAttach,
|
||||
.detach = onDetach,
|
||||
.define_range = onDefineRange,
|
||||
};
|
||||
|
||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, arrived: *ipc.Arrival) usize {
|
||||
|
||||
Reference in New Issue
Block a user