iommu: DMA-region capabilities — per-grant reachability, protocol flag-day

Replaces L2's interim DMA pool (every buffer reachable by every claimed
device) with true per-grant confinement: a device reaches only buffers
whose capability was delegated to its driver.

Kernel:
- DmaRegionObject (handle kind 2): a delegation token naming a dma_alloc'd
  region, passable across processes on the IPC cap slot like an endpoint
  or shared-memory object. Frames stay owned by the allocating address
  space (freed on dma_free/teardown as before); the token carries a `dead`
  flag so a stale downstream handle can no longer bind a freed region.
- dma_alloc gains the dma_shareable flag: it returns a capability handle
  in r8 and every region is tracked in a registry. A task's own regions
  auto-bind into the devices it claims (its rings just work); foreign
  buffers are bound explicitly.
- dma_bind / dma_unbind / handle_close syscalls (51-53). dma_bind maps a
  held region (or shared-memory) capability into a claimed device's domain;
  it is idempotent. handle_close reclaims a table slot (raised 16 -> 32).
- dma_free and task death unmap a region from every domain and invalidate
  BEFORE its frames return to the allocator — the stale-IOTLB use-after-
  free window, closed structurally.

Protocols (flag-day): block gains attach, usb-transfer gains dma_attach —
each carries a region capability on the cap slot. fat allocates its bounce
buffer shareable and attaches it; usb-storage allocates its transport
buffers shareable, attaches them to the controller, and forwards fat's
capability downstream; usb-xhci-bus binds and closes; virtio-gpu binds its
shared scanout surface. The physical addresses on the wire are unchanged
(identity IOVA), so no register-programming code moved.

Cross-process DMA (fat -> usb-storage -> xHC) now flows only through
delegated capabilities. iommu-usb-storage / iommu-usb-hid / iommu-fault
all green under per-grant enforcement; 104/104 overall (fail-open paths
unchanged).
This commit is contained in:
Daniel Samson
2026-07-26 18:31:27 +01:00
parent e94adcfc02
commit 4e7cbc9792
17 changed files with 414 additions and 64 deletions
+159 -12
View File
@@ -252,6 +252,9 @@ fn system_call(state: *architecture.CpuState) void {
.fs_mount => systemFsMount(state),
.fs_unmount => systemFsUnmount(state),
.iommu_fault_drain => systemIommuFaultDrain(state),
.dma_bind => systemDmaBind(state),
.dma_unbind => systemDmaUnbind(state),
.handle_close => systemHandleClose(state),
.wall_clock => systemWallClock(state),
.shared_memory_create => systemSharedMemoryCreate(state),
.shared_memory_map => systemSharedMemoryMap(state),
@@ -398,10 +401,14 @@ fn systemDeviceClaim(state: *architecture.CpuState) void {
// since the whole point is that claiming a DMA device is no longer equivalent
// to ring 0. No-op when no IOMMU exists (fail-open).
if (devices_broker.pciAddressOf(device_id)) |bdf| {
if (!iommu.confineDevice(device_id, bdf, scheduler.current().id)) {
_ = devices_broker.unclaim(device_id, scheduler.current().id);
const owner = scheduler.current().id;
if (!iommu.confineDevice(device_id, bdf, owner)) {
_ = devices_broker.unclaim(device_id, owner);
return fail(state);
}
// Bind the buffers this task allocated before claiming the device (a driver
// that dma_alloc'd its rings, then claimed the controller).
dmaBindOwnerRegionsInto(owner, device_id);
}
// A display service just took the framebuffer — quiesce the bootstrap console
// so the kernel and the service don't scribble over each other's pixels. The
@@ -510,6 +517,72 @@ fn systemIoWrite(state: *architecture.CpuState) void {
/// their physical address is never disclosed. `dma_below_4g` caps the physical address
/// for legacy engines; `dma_write_combining` is accepted but falls back to coherent
/// until PAT is programmed. See docs/driver-model.md (M14).
// --- DMA-region registry ---------------------------------------------------------------
// Every dma_alloc'd region is tracked here so it can be (a) auto-bound into the devices
// its owner claims, (b) delegated across processes as a capability and bound into a
// device's IOMMU domain by dma_bind, and (c) unmapped from every domain before its frames
// return to the allocator. Only *shareable* regions carry a heap `object` (the delegation
// token); a driver's private rings are tracked without one. All access under the big lock.
const DmaRegistryEntry = struct {
active: bool = false,
object: ?*ipc.DmaRegionObject = null,
physical: u64 = 0,
len: u64 = 0,
owner: u32 = 0,
};
const maximum_dma_regions = 256;
var dma_registry: [maximum_dma_regions]DmaRegistryEntry = .{DmaRegistryEntry{}} ** maximum_dma_regions;
fn dmaRegistryAdd(object: ?*ipc.DmaRegionObject, physical: u64, len: u64, owner: u32) void {
for (&dma_registry) |*e| {
if (!e.active) {
e.* = .{ .active = true, .object = object, .physical = physical, .len = len, .owner = owner };
return;
}
}
}
/// Retire the region based at `physical` owned by `owner`: unmap it from every device
/// domain (before the frames are freed), mark its token dead so a stale downstream handle
/// can no longer bind it, drop the registry's reference, and clear the slot.
fn dmaRegistryRemove(owner: u32, physical: u64) void {
for (&dma_registry) |*e| {
if (e.active and e.owner == owner and e.physical == physical) {
iommu.unmapRegionEverywhere(e.physical, e.len);
if (e.object) |object| {
object.dead = true;
ipc.dropDmaRegionReference(object);
}
e.* = .{};
return;
}
}
}
/// Retire every region owned by `owner` (task death) — same discipline as a per-region
/// free, run before the address space is torn down and its DMA frames reclaimed.
fn dmaRegistryReleaseOwner(owner: u32) void {
for (&dma_registry) |*e| {
if (e.active and e.owner == owner) {
iommu.unmapRegionEverywhere(e.physical, e.len);
if (e.object) |object| {
object.dead = true;
ipc.dropDmaRegionReference(object);
}
e.* = .{};
}
}
}
/// Bind every region `owner` allocated into the domain of the device it just claimed
/// (its rings allocated before the claim). Regions allocated *after* the claim are bound
/// by dma_alloc's own auto-bind.
fn dmaBindOwnerRegionsInto(owner: u32, device_id: u64) void {
for (&dma_registry) |*e| {
if (e.active and e.owner == owner) _ = iommu.mapForDevice(device_id, e.physical, e.len);
}
}
/// iommu_fault_drain() -> count: drain and log any pending IOMMU translation faults,
/// returning how many were seen. A diagnostic hook — a driver (or a test) that suspects
/// its device faulted can force the fault records to be logged now rather than waiting
@@ -521,6 +594,60 @@ fn systemIommuFaultDrain(state: *architecture.CpuState) void {
architecture.setSystemCallResult(state, count);
}
/// Resolve a capability handle the caller holds to the physical range it names — either
/// a DMA-region token or a shared-memory object (both are bindable buffers). null if the
/// handle is neither, or names a region already freed by its allocator.
fn bindableRange(t: *scheduler.Task, handle: u64) ?struct { physical: u64, len: u64 } {
if (ipc.resolveDmaRegion(t, handle)) |region| {
if (region.dead) return null;
return .{ .physical = region.phys, .len = region.pages * page_size };
}
if (ipc.resolveSharedMemory(t, handle)) |shared| {
return .{ .physical = shared.phys, .len = shared.pages * page_size };
}
return null;
}
/// dma_bind(device_id, handle) -> 0/-errno: map the buffer named by `handle` into the
/// claimed device's IOMMU domain. The caller must own the device (the mmio_map gate) and
/// hold the handle. Idempotent: re-binding is a harmless success, so a driver may re-bind
/// after a restart without tracking what it already bound. Success (no-op) without an IOMMU.
fn systemDmaBind(state: *architecture.CpuState) void {
const device_id = architecture.systemCallArg(state, 0);
const handle = architecture.systemCallArg(state, 1);
const t = scheduler.current();
const flags = sync.enter();
defer sync.leave(flags);
if (devices_broker.ownerOf(device_id) != t.id) return fail(state);
const range = bindableRange(t, handle) orelse return fail(state);
if (!iommu.mapForDevice(device_id, range.physical, range.len)) return fail(state);
architecture.setSystemCallResult(state, 0);
}
/// dma_unbind(device_id, handle) -> 0/-errno: unmap a previously bound buffer from the
/// device's domain and invalidate.
fn systemDmaUnbind(state: *architecture.CpuState) void {
const device_id = architecture.systemCallArg(state, 0);
const handle = architecture.systemCallArg(state, 1);
const t = scheduler.current();
const flags = sync.enter();
defer sync.leave(flags);
if (devices_broker.ownerOf(device_id) != t.id) return fail(state);
const range = bindableRange(t, handle) orelse return fail(state);
iommu.unmapForDevice(device_id, range.physical, range.len);
architecture.setSystemCallResult(state, 0);
}
/// handle_close(handle) -> 0/-errno: drop one capability handle and free its slot.
fn systemHandleClose(state: *architecture.CpuState) void {
const handle = architecture.systemCallArg(state, 0);
const t = scheduler.current();
const flags = sync.enter();
defer sync.leave(flags);
if (ipc.closeHandle(t, handle) < 0) return fail(state);
architecture.setSystemCallResult(state, 0);
}
fn systemDmaAlloc(state: *architecture.CpuState) void {
const len = architecture.systemCallArg(state, 0);
const flags = architecture.systemCallArg(state, 1);
@@ -567,15 +694,33 @@ fn systemDmaAlloc(state: *architecture.CpuState) void {
architecture.mapUserDmaInto(t.address_space, base_v + i * page_size, phys + i * page_size, page_size);
sync.leave(lock_flags);
}
// Publish the region to the DMA pool: it becomes reachable to every claimed device
// (the interim rule until DMA-region capabilities land). No-op without an IOMMU.
// Register the region and bind it into every device this task already drives (its
// own buffers reach its own devices). When `dma_shareable` is set, also wrap it in a
// capability token and return a handle so it can be delegated to another driver and
// dma_bound there. All under the lock (the registry + IOMMU tables are shared).
var handle: u64 = abi.no_cap;
{
const lock_flags = sync.enter();
iommu.poolAdd(phys, pages * page_size);
sync.leave(lock_flags);
defer sync.leave(lock_flags);
var object: ?*ipc.DmaRegionObject = null;
if (flags & abi.dma_shareable != 0) {
if (ipc.createDmaRegion(phys, pages, t.id)) |region| {
const h = ipc.installDmaRegionHandle(t, region);
if (h >= 0) {
region.refcount += 1; // the handle's reference (registry holds the first)
handle = @intCast(h);
object = region;
} else {
ipc.dropDmaRegionReference(region); // no table slot; drop it
}
}
}
dmaRegistryAdd(object, phys, pages * page_size, t.id);
iommu.mapRegionForOwner(t.id, phys, pages * page_size);
}
architecture.setSystemCallResult(state, base_v); // virtual address for the CPU
architecture.setSystemCallResult2(state, phys); // physical address for the device
architecture.setSystemCallResult3(state, handle); // capability handle (no_cap unless shareable)
}
/// dma_free(virtual_address, len) -> 0: release a prior `dma_alloc`. Bounded to the DMA arena so
@@ -591,14 +736,13 @@ fn systemDmaFree(state: *architecture.CpuState) void {
const pages: usize = @intCast((len + page_size - 1) / page_size);
if (base_v < dma_arena_base or base_v + pages * page_size > dma_arena_end) return fail(state);
// Pull the region out of every device's domain and invalidate BEFORE any frame
// returns to the allocator — a device still translating to a reallocated frame is a
// use-after-free. dma_alloc's frames are contiguous, so the base translation names
// the whole region.
// Retire the region — unmap it from every device domain and mark its token dead —
// BEFORE any frame returns to the allocator, so no device can still translate to a
// reallocated frame. dma_alloc's frames are contiguous, so the base names the region.
{
const lock_flags = sync.enter();
if (architecture.translate(t.address_space, base_v)) |base_phys|
iommu.poolRemove(base_phys, pages * page_size);
dmaRegistryRemove(t.id, base_phys);
sync.leave(lock_flags);
}
@@ -1036,8 +1180,11 @@ fn releaseTaskResourcesLocked(t: *scheduler.Task) void {
// Detach the task's devices from their IOMMU domains BEFORE the broker clears the
// claims (the detach reads ownership) and before the address space is torn down and
// its DMA frames return to the allocator — a device must stop translating to a frame
// before that frame can be handed to someone else.
// before that frame can be handed to someone else. Then retire the buffers this task
// allocated, unmapping them from any *other* driver's domain they were granted into,
// before those frames are freed too.
iommu.releaseAllOwnedBy(t.id);
dmaRegistryReleaseOwner(t.id);
devices_broker.releaseAllOwnedBy(t.id);
// If that dropped the framebuffer claim (this task was the display service), let the
// bootstrap console draw again — the screen is nobody's now, so panics/status land.