iommu: DMA-region capabilities — per-grant reachability, protocol flag-day

Replaces L2's interim DMA pool (every buffer reachable by every claimed
device) with true per-grant confinement: a device reaches only buffers
whose capability was delegated to its driver.

Kernel:
- DmaRegionObject (handle kind 2): a delegation token naming a dma_alloc'd
  region, passable across processes on the IPC cap slot like an endpoint
  or shared-memory object. Frames stay owned by the allocating address
  space (freed on dma_free/teardown as before); the token carries a `dead`
  flag so a stale downstream handle can no longer bind a freed region.
- dma_alloc gains the dma_shareable flag: it returns a capability handle
  in r8 and every region is tracked in a registry. A task's own regions
  auto-bind into the devices it claims (its rings just work); foreign
  buffers are bound explicitly.
- dma_bind / dma_unbind / handle_close syscalls (51-53). dma_bind maps a
  held region (or shared-memory) capability into a claimed device's domain;
  it is idempotent. handle_close reclaims a table slot (raised 16 -> 32).
- dma_free and task death unmap a region from every domain and invalidate
  BEFORE its frames return to the allocator — the stale-IOTLB use-after-
  free window, closed structurally.

Protocols (flag-day): block gains attach, usb-transfer gains dma_attach —
each carries a region capability on the cap slot. fat allocates its bounce
buffer shareable and attaches it; usb-storage allocates its transport
buffers shareable, attaches them to the controller, and forwards fat's
capability downstream; usb-xhci-bus binds and closes; virtio-gpu binds its
shared scanout surface. The physical addresses on the wire are unchanged
(identity IOVA), so no register-programming code moved.

Cross-process DMA (fat -> usb-storage -> xHC) now flows only through
delegated capabilities. iommu-usb-storage / iommu-usb-hid / iommu-fault
all green under per-grant enforcement; 104/104 overall (fail-open paths
unchanged).
This commit is contained in:
Daniel Samson
2026-07-26 18:31:27 +01:00
parent e94adcfc02
commit 4e7cbc9792
17 changed files with 414 additions and 64 deletions
+39 -39
View File
@@ -128,29 +128,18 @@ pub fn init() void {
const Confined = struct { active: bool = false, owner: u32 = 0, bdf: u16 = 0, domain: u16 = invalid_domain };
var confined: [maximum_domains]Confined = .{Confined{}} ** maximum_domains;
/// The interim DMA pool: the set of `dma_alloc`'d regions. Every such region is mapped
/// into every claimed device's domain, so a device reaches DMA buffers (including one
/// another's — the honest limit of this stage) but NOT the kernel, page tables, process
/// heaps, or arbitrary RAM. The capability layer (L3/L4) narrows this to per-grant.
const PoolRegion = struct { active: bool = false, physical: u64 = 0, len: u64 = 0 };
var pool: [maximum_pool]PoolRegion = .{PoolRegion{}} ** maximum_pool;
const maximum_pool = 256;
/// Place a just-claimed PCI function under IOMMU translation on behalf of `owner`: give
/// it a private empty domain, seed it with the current DMA pool and the device's own
/// firmware reserved region, and attach. false only if a domain can't be allocated —
/// the caller rolls the claim back (a claim that can't be confined must not stand). No-op
/// success when no IOMMU exists (fail-open).
/// it a private empty domain, seed it with the device's own firmware reserved region,
/// and attach. Its DMA buffers arrive afterward as explicit grants — the owner's own
/// `dma_alloc`'d regions are bound by the claim path (`mapForDevice`), and cross-process
/// buffers by `dma_bind`. false only if a domain can't be allocated — the caller rolls
/// the claim back (a claim that can't be confined must not stand). No-op success when no
/// IOMMU exists (fail-open).
pub fn confineDevice(device_id: u64, bdf: u16, owner: u32) bool {
if (kind == .none) return true;
if (device_id >= confined.len) return true; // unusual id; leave it to fail-open
const domain = domainCreate(owner, bdf) orelse return false;
// Seed with every pooled DMA region so the device's own rings/buffers and the
// cross-process buffers handed to it (fat's bounce buffer via usb-storage) resolve.
for (&pool) |*r| {
if (r.active) _ = map(domain, r.physical, r.len);
}
// Firmware reserved region for this device, if any (real hardware; QEMU has none).
const info = platform.platformInformation();
var i: usize = 0;
@@ -164,34 +153,45 @@ pub fn confineDevice(device_id: u64, bdf: u16, owner: u32) bool {
return true;
}
/// Record a newly `dma_alloc`'d region and map it into every claimed device's domain
/// (the interim pool rule). Called from the dma_alloc syscall.
pub fn poolAdd(physical: u64, len: u64) void {
/// The confined record for `device_id`, or null if the device is not confined.
fn confinedOf(device_id: u64) ?*Confined {
if (device_id >= confined.len) return null;
const c = &confined[@intCast(device_id)];
return if (c.active) c else null;
}
/// Map a DMA region into a specific claimed device's domain (the device owner binding a
/// granted buffer). false if the device is not confined. No-op success without an IOMMU.
pub fn mapForDevice(device_id: u64, physical: u64, len: u64) bool {
if (kind == .none) return true;
const c = confinedOf(device_id) orelse return false;
return map(c.domain, physical, len);
}
/// Unmap a DMA region from a specific claimed device's domain. No-op if not confined.
pub fn unmapForDevice(device_id: u64, physical: u64, len: u64) void {
if (kind == .none) return;
const c = confinedOf(device_id) orelse return;
unmap(c.domain, physical, len);
}
/// Map a region into every claimed device owned by `owner` — the auto-bind of a task's
/// own freshly-`dma_alloc`'d buffer into the devices it drives.
pub fn mapRegionForOwner(owner: u32, physical: u64, len: u64) void {
if (kind == .none) return;
for (&pool) |*r| {
if (!r.active) {
r.* = .{ .active = true, .physical = physical, .len = len };
break;
}
} else return; // pool full; region stays unmapped and its device DMA will fault
for (&confined) |*c| {
if (c.active) _ = map(c.domain, physical, len);
if (c.active and c.owner == owner) _ = map(c.domain, physical, len);
}
}
/// Unmap a freed DMA region from every claimed device's domain and forget it. MUST run
/// before the frames return to pmm — a device translating to a reallocated frame is the
/// use-after-free this prevents. Called from the dma_free syscall.
pub fn poolRemove(physical: u64, len: u64) void {
/// Unmap a region from EVERY claimed device's domain — the freed-region sweep. MUST run
/// before the frames return to pmm: a device translating to a reallocated frame is the
/// use-after-free this prevents. Cross-device because a granted buffer may be bound in a
/// domain other than its owner's.
pub fn unmapRegionEverywhere(physical: u64, len: u64) void {
if (kind == .none) return;
for (&pool) |*r| {
if (r.active and r.physical == physical and r.len == len) {
for (&confined) |*c| {
if (c.active) unmap(c.domain, physical, len);
}
r.* = .{};
return;
}
for (&confined) |*c| {
if (c.active) unmap(c.domain, physical, len);
}
}