iommu: DMA-region capabilities — per-grant reachability, protocol flag-day

Replaces L2's interim DMA pool (every buffer reachable by every claimed
device) with true per-grant confinement: a device reaches only buffers
whose capability was delegated to its driver.

Kernel:
- DmaRegionObject (handle kind 2): a delegation token naming a dma_alloc'd
  region, passable across processes on the IPC cap slot like an endpoint
  or shared-memory object. Frames stay owned by the allocating address
  space (freed on dma_free/teardown as before); the token carries a `dead`
  flag so a stale downstream handle can no longer bind a freed region.
- dma_alloc gains the dma_shareable flag: it returns a capability handle
  in r8 and every region is tracked in a registry. A task's own regions
  auto-bind into the devices it claims (its rings just work); foreign
  buffers are bound explicitly.
- dma_bind / dma_unbind / handle_close syscalls (51-53). dma_bind maps a
  held region (or shared-memory) capability into a claimed device's domain;
  it is idempotent. handle_close reclaims a table slot (raised 16 -> 32).
- dma_free and task death unmap a region from every domain and invalidate
  BEFORE its frames return to the allocator — the stale-IOTLB use-after-
  free window, closed structurally.

Protocols (flag-day): block gains attach, usb-transfer gains dma_attach —
each carries a region capability on the cap slot. fat allocates its bounce
buffer shareable and attaches it; usb-storage allocates its transport
buffers shareable, attaches them to the controller, and forwards fat's
capability downstream; usb-xhci-bus binds and closes; virtio-gpu binds its
shared scanout surface. The physical addresses on the wire are unchanged
(identity IOVA), so no register-programming code moved.

Cross-process DMA (fat -> usb-storage -> xHC) now flows only through
delegated capabilities. iommu-usb-storage / iommu-usb-hid / iommu-fault
all green under per-grant enforcement; 104/104 overall (fail-open paths
unchanged).
This commit is contained in:
Daniel Samson
2026-07-26 18:31:27 +01:00
parent e94adcfc02
commit 4e7cbc9792
17 changed files with 414 additions and 64 deletions
+39 -39
View File
@@ -128,29 +128,18 @@ pub fn init() void {
const Confined = struct { active: bool = false, owner: u32 = 0, bdf: u16 = 0, domain: u16 = invalid_domain };
var confined: [maximum_domains]Confined = .{Confined{}} ** maximum_domains;
/// The interim DMA pool: the set of `dma_alloc`'d regions. Every such region is mapped
/// into every claimed device's domain, so a device reaches DMA buffers (including one
/// another's — the honest limit of this stage) but NOT the kernel, page tables, process
/// heaps, or arbitrary RAM. The capability layer (L3/L4) narrows this to per-grant.
const PoolRegion = struct { active: bool = false, physical: u64 = 0, len: u64 = 0 };
var pool: [maximum_pool]PoolRegion = .{PoolRegion{}} ** maximum_pool;
const maximum_pool = 256;
/// Place a just-claimed PCI function under IOMMU translation on behalf of `owner`: give
/// it a private empty domain, seed it with the current DMA pool and the device's own
/// firmware reserved region, and attach. false only if a domain can't be allocated —
/// the caller rolls the claim back (a claim that can't be confined must not stand). No-op
/// success when no IOMMU exists (fail-open).
/// it a private empty domain, seed it with the device's own firmware reserved region,
/// and attach. Its DMA buffers arrive afterward as explicit grants — the owner's own
/// `dma_alloc`'d regions are bound by the claim path (`mapForDevice`), and cross-process
/// buffers by `dma_bind`. false only if a domain can't be allocated — the caller rolls
/// the claim back (a claim that can't be confined must not stand). No-op success when no
/// IOMMU exists (fail-open).
pub fn confineDevice(device_id: u64, bdf: u16, owner: u32) bool {
if (kind == .none) return true;
if (device_id >= confined.len) return true; // unusual id; leave it to fail-open
const domain = domainCreate(owner, bdf) orelse return false;
// Seed with every pooled DMA region so the device's own rings/buffers and the
// cross-process buffers handed to it (fat's bounce buffer via usb-storage) resolve.
for (&pool) |*r| {
if (r.active) _ = map(domain, r.physical, r.len);
}
// Firmware reserved region for this device, if any (real hardware; QEMU has none).
const info = platform.platformInformation();
var i: usize = 0;
@@ -164,34 +153,45 @@ pub fn confineDevice(device_id: u64, bdf: u16, owner: u32) bool {
return true;
}
/// Record a newly `dma_alloc`'d region and map it into every claimed device's domain
/// (the interim pool rule). Called from the dma_alloc syscall.
pub fn poolAdd(physical: u64, len: u64) void {
/// The confined record for `device_id`, or null if the device is not confined.
fn confinedOf(device_id: u64) ?*Confined {
if (device_id >= confined.len) return null;
const c = &confined[@intCast(device_id)];
return if (c.active) c else null;
}
/// Map a DMA region into a specific claimed device's domain (the device owner binding a
/// granted buffer). false if the device is not confined. No-op success without an IOMMU.
pub fn mapForDevice(device_id: u64, physical: u64, len: u64) bool {
if (kind == .none) return true;
const c = confinedOf(device_id) orelse return false;
return map(c.domain, physical, len);
}
/// Unmap a DMA region from a specific claimed device's domain. No-op if not confined.
pub fn unmapForDevice(device_id: u64, physical: u64, len: u64) void {
if (kind == .none) return;
const c = confinedOf(device_id) orelse return;
unmap(c.domain, physical, len);
}
/// Map a region into every claimed device owned by `owner` — the auto-bind of a task's
/// own freshly-`dma_alloc`'d buffer into the devices it drives.
pub fn mapRegionForOwner(owner: u32, physical: u64, len: u64) void {
if (kind == .none) return;
for (&pool) |*r| {
if (!r.active) {
r.* = .{ .active = true, .physical = physical, .len = len };
break;
}
} else return; // pool full; region stays unmapped and its device DMA will fault
for (&confined) |*c| {
if (c.active) _ = map(c.domain, physical, len);
if (c.active and c.owner == owner) _ = map(c.domain, physical, len);
}
}
/// Unmap a freed DMA region from every claimed device's domain and forget it. MUST run
/// before the frames return to pmm — a device translating to a reallocated frame is the
/// use-after-free this prevents. Called from the dma_free syscall.
pub fn poolRemove(physical: u64, len: u64) void {
/// Unmap a region from EVERY claimed device's domain — the freed-region sweep. MUST run
/// before the frames return to pmm: a device translating to a reallocated frame is the
/// use-after-free this prevents. Cross-device because a granted buffer may be bound in a
/// domain other than its owner's.
pub fn unmapRegionEverywhere(physical: u64, len: u64) void {
if (kind == .none) return;
for (&pool) |*r| {
if (r.active and r.physical == physical and r.len == len) {
for (&confined) |*c| {
if (c.active) unmap(c.domain, physical, len);
}
r.* = .{};
return;
}
for (&confined) |*c| {
if (c.active) unmap(c.domain, physical, len);
}
}
+65
View File
@@ -154,6 +154,66 @@ pub fn dropRef(endpoint: *Endpoint) void {
/// Defined here (not in scheduler) because the meaning is the IPC/capability layer's.
pub const handle_kind_endpoint: u8 = 0;
pub const handle_kind_shared_memory: u8 = 1;
pub const handle_kind_dma_region: u8 = 2;
/// A DMA buffer's delegation token: names the `pages` contiguous frames at `phys` that
/// `dma_alloc` handed its `owner`, and can be passed across processes as a capability so
/// the driver that owns a device can bind it into that device's IOMMU domain
/// (`dma_bind`). Unlike `SharedMemoryObject` this does NOT own the frames — the
/// allocating address space still does, and frees them on `dma_free` or teardown — so
/// this is a pure token: `dead` is set when the allocator frees the region, after which
/// a stale handle can no longer bind it. `refcount` counts the allocator's registry
/// entry plus every outstanding handle; the object is freed when the last drops.
pub const DmaRegionObject = struct {
refcount: u32 = 1,
phys: u64,
pages: usize,
owner: u32,
dead: bool = false,
};
/// Create a DMA-region token for `pages` frames at `phys` owned by task `owner`. The
/// frames are already allocated and mapped by the caller; this only wraps them for
/// delegation. null if the heap is out of room.
pub fn createDmaRegion(phys: u64, pages: usize, owner: u32) ?*DmaRegionObject {
const region = heap.allocator().create(DmaRegionObject) catch return null;
region.* = .{ .phys = phys, .pages = pages, .owner = owner };
return region;
}
/// Drop a DMA-region reference; free the token when the last (registry + handles) goes.
/// Never frees frames — the allocator owns those.
pub fn dropDmaRegionReference(region: *DmaRegionObject) void {
if (region.refcount > 1) {
region.refcount -= 1;
} else {
heap.allocator().destroy(region);
}
}
/// Resolve a handle to its DMA-region token, or null if out of range, unused, or a
/// different kind.
pub fn resolveDmaRegion(t: *Task, h: u64) ?*DmaRegionObject {
if (h >= t.handles.len) return null;
const entry = t.handles[@intCast(h)] orelse return null;
if (entry.kind != handle_kind_dma_region) return null;
return @ptrCast(@alignCast(entry.ptr));
}
/// Install a DMA-region handle in task `t`'s table (the slot owns a reference).
pub fn installDmaRegionHandle(t: *Task, region: *DmaRegionObject) i64 {
return installEntry(t, .{ .kind = handle_kind_dma_region, .ptr = @ptrCast(region) });
}
/// Drop the handle at slot `h` of task `t` (handle_close): release its reference and
/// free the slot. Returns 0 or -EBADF.
pub fn closeHandle(t: *Task, h: u64) i64 {
if (h >= t.handles.len) return -EBADF;
const entry = t.handles[@intCast(h)] orelse return -EBADF;
dropEntry(entry);
t.handles[@intCast(h)] = null;
return 0;
}
/// A page-aligned block of **shared cacheable RAM** (docs/display-v2.md), referenced by
/// capability handles across processes and freed when the last one drops. `phys` is its
@@ -307,6 +367,10 @@ fn shareCapability(from: *Task, to: *Task, cap: u64) i64 {
const s: *SharedMemoryObject = @ptrCast(@alignCast(entry.ptr));
s.refcount += 1;
},
handle_kind_dma_region => {
const r: *DmaRegionObject = @ptrCast(@alignCast(entry.ptr));
r.refcount += 1;
},
else => return -EBADF,
}
const handle = installEntry(to, entry);
@@ -574,6 +638,7 @@ fn dropEntry(entry: scheduler.HandleObject) void {
switch (entry.kind) {
handle_kind_endpoint => dropRef(@ptrCast(@alignCast(entry.ptr))),
handle_kind_shared_memory => dropSharedMemoryReference(@ptrCast(@alignCast(entry.ptr))),
handle_kind_dma_region => dropDmaRegionReference(@ptrCast(@alignCast(entry.ptr))),
else => {},
}
}
+159 -12
View File
@@ -252,6 +252,9 @@ fn system_call(state: *architecture.CpuState) void {
.fs_mount => systemFsMount(state),
.fs_unmount => systemFsUnmount(state),
.iommu_fault_drain => systemIommuFaultDrain(state),
.dma_bind => systemDmaBind(state),
.dma_unbind => systemDmaUnbind(state),
.handle_close => systemHandleClose(state),
.wall_clock => systemWallClock(state),
.shared_memory_create => systemSharedMemoryCreate(state),
.shared_memory_map => systemSharedMemoryMap(state),
@@ -398,10 +401,14 @@ fn systemDeviceClaim(state: *architecture.CpuState) void {
// since the whole point is that claiming a DMA device is no longer equivalent
// to ring 0. No-op when no IOMMU exists (fail-open).
if (devices_broker.pciAddressOf(device_id)) |bdf| {
if (!iommu.confineDevice(device_id, bdf, scheduler.current().id)) {
_ = devices_broker.unclaim(device_id, scheduler.current().id);
const owner = scheduler.current().id;
if (!iommu.confineDevice(device_id, bdf, owner)) {
_ = devices_broker.unclaim(device_id, owner);
return fail(state);
}
// Bind the buffers this task allocated before claiming the device (a driver
// that dma_alloc'd its rings, then claimed the controller).
dmaBindOwnerRegionsInto(owner, device_id);
}
// A display service just took the framebuffer — quiesce the bootstrap console
// so the kernel and the service don't scribble over each other's pixels. The
@@ -510,6 +517,72 @@ fn systemIoWrite(state: *architecture.CpuState) void {
/// their physical address is never disclosed. `dma_below_4g` caps the physical address
/// for legacy engines; `dma_write_combining` is accepted but falls back to coherent
/// until PAT is programmed. See docs/driver-model.md (M14).
// --- DMA-region registry ---------------------------------------------------------------
// Every dma_alloc'd region is tracked here so it can be (a) auto-bound into the devices
// its owner claims, (b) delegated across processes as a capability and bound into a
// device's IOMMU domain by dma_bind, and (c) unmapped from every domain before its frames
// return to the allocator. Only *shareable* regions carry a heap `object` (the delegation
// token); a driver's private rings are tracked without one. All access under the big lock.
const DmaRegistryEntry = struct {
active: bool = false,
object: ?*ipc.DmaRegionObject = null,
physical: u64 = 0,
len: u64 = 0,
owner: u32 = 0,
};
const maximum_dma_regions = 256;
var dma_registry: [maximum_dma_regions]DmaRegistryEntry = .{DmaRegistryEntry{}} ** maximum_dma_regions;
fn dmaRegistryAdd(object: ?*ipc.DmaRegionObject, physical: u64, len: u64, owner: u32) void {
for (&dma_registry) |*e| {
if (!e.active) {
e.* = .{ .active = true, .object = object, .physical = physical, .len = len, .owner = owner };
return;
}
}
}
/// Retire the region based at `physical` owned by `owner`: unmap it from every device
/// domain (before the frames are freed), mark its token dead so a stale downstream handle
/// can no longer bind it, drop the registry's reference, and clear the slot.
fn dmaRegistryRemove(owner: u32, physical: u64) void {
for (&dma_registry) |*e| {
if (e.active and e.owner == owner and e.physical == physical) {
iommu.unmapRegionEverywhere(e.physical, e.len);
if (e.object) |object| {
object.dead = true;
ipc.dropDmaRegionReference(object);
}
e.* = .{};
return;
}
}
}
/// Retire every region owned by `owner` (task death) — same discipline as a per-region
/// free, run before the address space is torn down and its DMA frames reclaimed.
fn dmaRegistryReleaseOwner(owner: u32) void {
for (&dma_registry) |*e| {
if (e.active and e.owner == owner) {
iommu.unmapRegionEverywhere(e.physical, e.len);
if (e.object) |object| {
object.dead = true;
ipc.dropDmaRegionReference(object);
}
e.* = .{};
}
}
}
/// Bind every region `owner` allocated into the domain of the device it just claimed
/// (its rings allocated before the claim). Regions allocated *after* the claim are bound
/// by dma_alloc's own auto-bind.
fn dmaBindOwnerRegionsInto(owner: u32, device_id: u64) void {
for (&dma_registry) |*e| {
if (e.active and e.owner == owner) _ = iommu.mapForDevice(device_id, e.physical, e.len);
}
}
/// iommu_fault_drain() -> count: drain and log any pending IOMMU translation faults,
/// returning how many were seen. A diagnostic hook — a driver (or a test) that suspects
/// its device faulted can force the fault records to be logged now rather than waiting
@@ -521,6 +594,60 @@ fn systemIommuFaultDrain(state: *architecture.CpuState) void {
architecture.setSystemCallResult(state, count);
}
/// Resolve a capability handle the caller holds to the physical range it names — either
/// a DMA-region token or a shared-memory object (both are bindable buffers). null if the
/// handle is neither, or names a region already freed by its allocator.
fn bindableRange(t: *scheduler.Task, handle: u64) ?struct { physical: u64, len: u64 } {
if (ipc.resolveDmaRegion(t, handle)) |region| {
if (region.dead) return null;
return .{ .physical = region.phys, .len = region.pages * page_size };
}
if (ipc.resolveSharedMemory(t, handle)) |shared| {
return .{ .physical = shared.phys, .len = shared.pages * page_size };
}
return null;
}
/// dma_bind(device_id, handle) -> 0/-errno: map the buffer named by `handle` into the
/// claimed device's IOMMU domain. The caller must own the device (the mmio_map gate) and
/// hold the handle. Idempotent: re-binding is a harmless success, so a driver may re-bind
/// after a restart without tracking what it already bound. Success (no-op) without an IOMMU.
fn systemDmaBind(state: *architecture.CpuState) void {
const device_id = architecture.systemCallArg(state, 0);
const handle = architecture.systemCallArg(state, 1);
const t = scheduler.current();
const flags = sync.enter();
defer sync.leave(flags);
if (devices_broker.ownerOf(device_id) != t.id) return fail(state);
const range = bindableRange(t, handle) orelse return fail(state);
if (!iommu.mapForDevice(device_id, range.physical, range.len)) return fail(state);
architecture.setSystemCallResult(state, 0);
}
/// dma_unbind(device_id, handle) -> 0/-errno: unmap a previously bound buffer from the
/// device's domain and invalidate.
fn systemDmaUnbind(state: *architecture.CpuState) void {
const device_id = architecture.systemCallArg(state, 0);
const handle = architecture.systemCallArg(state, 1);
const t = scheduler.current();
const flags = sync.enter();
defer sync.leave(flags);
if (devices_broker.ownerOf(device_id) != t.id) return fail(state);
const range = bindableRange(t, handle) orelse return fail(state);
iommu.unmapForDevice(device_id, range.physical, range.len);
architecture.setSystemCallResult(state, 0);
}
/// handle_close(handle) -> 0/-errno: drop one capability handle and free its slot.
fn systemHandleClose(state: *architecture.CpuState) void {
const handle = architecture.systemCallArg(state, 0);
const t = scheduler.current();
const flags = sync.enter();
defer sync.leave(flags);
if (ipc.closeHandle(t, handle) < 0) return fail(state);
architecture.setSystemCallResult(state, 0);
}
fn systemDmaAlloc(state: *architecture.CpuState) void {
const len = architecture.systemCallArg(state, 0);
const flags = architecture.systemCallArg(state, 1);
@@ -567,15 +694,33 @@ fn systemDmaAlloc(state: *architecture.CpuState) void {
architecture.mapUserDmaInto(t.address_space, base_v + i * page_size, phys + i * page_size, page_size);
sync.leave(lock_flags);
}
// Publish the region to the DMA pool: it becomes reachable to every claimed device
// (the interim rule until DMA-region capabilities land). No-op without an IOMMU.
// Register the region and bind it into every device this task already drives (its
// own buffers reach its own devices). When `dma_shareable` is set, also wrap it in a
// capability token and return a handle so it can be delegated to another driver and
// dma_bound there. All under the lock (the registry + IOMMU tables are shared).
var handle: u64 = abi.no_cap;
{
const lock_flags = sync.enter();
iommu.poolAdd(phys, pages * page_size);
sync.leave(lock_flags);
defer sync.leave(lock_flags);
var object: ?*ipc.DmaRegionObject = null;
if (flags & abi.dma_shareable != 0) {
if (ipc.createDmaRegion(phys, pages, t.id)) |region| {
const h = ipc.installDmaRegionHandle(t, region);
if (h >= 0) {
region.refcount += 1; // the handle's reference (registry holds the first)
handle = @intCast(h);
object = region;
} else {
ipc.dropDmaRegionReference(region); // no table slot; drop it
}
}
}
dmaRegistryAdd(object, phys, pages * page_size, t.id);
iommu.mapRegionForOwner(t.id, phys, pages * page_size);
}
architecture.setSystemCallResult(state, base_v); // virtual address for the CPU
architecture.setSystemCallResult2(state, phys); // physical address for the device
architecture.setSystemCallResult3(state, handle); // capability handle (no_cap unless shareable)
}
/// dma_free(virtual_address, len) -> 0: release a prior `dma_alloc`. Bounded to the DMA arena so
@@ -591,14 +736,13 @@ fn systemDmaFree(state: *architecture.CpuState) void {
const pages: usize = @intCast((len + page_size - 1) / page_size);
if (base_v < dma_arena_base or base_v + pages * page_size > dma_arena_end) return fail(state);
// Pull the region out of every device's domain and invalidate BEFORE any frame
// returns to the allocator — a device still translating to a reallocated frame is a
// use-after-free. dma_alloc's frames are contiguous, so the base translation names
// the whole region.
// Retire the region — unmap it from every device domain and mark its token dead —
// BEFORE any frame returns to the allocator, so no device can still translate to a
// reallocated frame. dma_alloc's frames are contiguous, so the base names the region.
{
const lock_flags = sync.enter();
if (architecture.translate(t.address_space, base_v)) |base_phys|
iommu.poolRemove(base_phys, pages * page_size);
dmaRegistryRemove(t.id, base_phys);
sync.leave(lock_flags);
}
@@ -1036,8 +1180,11 @@ fn releaseTaskResourcesLocked(t: *scheduler.Task) void {
// Detach the task's devices from their IOMMU domains BEFORE the broker clears the
// claims (the detach reads ownership) and before the address space is torn down and
// its DMA frames return to the allocator — a device must stop translating to a frame
// before that frame can be handed to someone else.
// before that frame can be handed to someone else. Then retire the buffers this task
// allocated, unmapping them from any *other* driver's domain they were granted into,
// before those frames are freed too.
iommu.releaseAllOwnedBy(t.id);
dmaRegistryReleaseOwner(t.id);
devices_broker.releaseAllOwnedBy(t.id);
// If that dropped the framebuffer claim (this task was the display service), let the
// bootstrap console draw again — the screen is nobody's now, so panics/status land.
+4 -1
View File
@@ -149,7 +149,10 @@ pub const maximum_task_name = abi.maximum_process_name;
/// Size of each task's IPC handle table. Kept here (not in ipc_sync.zig) because
/// it dimensions a field of `Task`; ipc_sync.zig re-exports it.
pub const ipc_maximum_handles = 16;
// Raised from 16 with DMA-region capabilities: a driver now holds its per-device
// channel endpoints plus received DMA-region handles (a storage driver forwards several
// buffer caps), and repeated cap-passing consumes slots until handle_close.
pub const ipc_maximum_handles = 32;
/// One handle-table entry: a capability object plus a `kind` tag saying what `ptr` points
/// at (an ipc endpoint or a shared-memory object), so a task's exit path and the