iommu: per-device domains with interim DMA-pool enforcement

Replaces L1's shared blanket identity domain with a private translation
domain per claimed PCI function. A device now reaches only:
  - the DMA pool: every dma_alloc'd region, mapped into every claimed
    device's domain (poolAdd/poolRemove, driven from the dma_alloc and
    dma_free syscalls). This keeps the cross-process buffer handoff
    working (fat's bounce buffer reaches the xHC) while blocking the
    kernel, page tables, process heaps, MMIO, and unallocated RAM.
  - its own firmware reserved region (RMRR), seeded at confine time.
The pool is the honest interim: devices can still reach one another's
DMA buffers. The DMA-region capability layer (next) narrows it to
per-grant reachability.

dma_free unmaps from every domain and invalidates BEFORE the frames
return to the allocator, closing the stale-IOTLB use-after-free window.
Driver death tears down its domains (detach + free tables) before the
broker claims and DMA frames are released.

New iommu_fault_drain syscall (+ driver.iommuFaultDrain) forces pending
fault records to the log on demand. The new iommu-fault case proves it:
a claimed e1000e is programmed to DMA-fetch its TX ring from an unmapped
page; VT-d faults the access (bdf 00:03.0 addr 0x1000 reason 0x6) and the
system stays alive. 104/104.
This commit is contained in:
Daniel Samson
2026-07-26 18:31:27 +01:00
parent f477ef7d9f
commit e94adcfc02
8 changed files with 280 additions and 43 deletions
+72 -43
View File
@@ -115,66 +115,95 @@ pub fn init() void {
return;
}
// L1 (enabled-but-invisible): every device is placed in a single blanket domain
// that identity-maps all of RAM plus the firmware reserved regions, so translation
// is on but nothing's DMA changes. L2 replaces this per device: on claim a device
// is detached from the blanket and attached to its own empty domain, so it reaches
// only what is explicitly mapped for it.
buildBlanketDomain(info);
// The root table starts empty: every device's context entry is not-present, so any
// DMA faults until the device's driver claims it (confineDevice gives it a private
// domain). PCI functions are enumerated post-boot by the ring-3 pci-bus driver, so
// there is nothing to attach at init anyway.
intel.enable();
logEnabled(info);
}
/// The shared identity domain claimed devices are placed in (L1). L2 replaces this with
/// a private empty domain per device plus explicitly-granted mappings.
var blanket: u16 = invalid_domain;
/// PCI functions are enumerated post-boot by the ring-3 pci-bus driver, so at init the
/// device tree has none — a device is attached to the blanket when its driver *claims*
/// it (confineDevice), which is always before the driver programs any DMA. Until then
/// its context entry is not-present and its DMA faults (only stale firmware bus-
/// mastering would hit that, which is the evidence M16 exists to surface).
fn buildBlanketDomain(info: platform.PlatformInformation) void {
const domain = domainCreate(0, 0) orelse return;
blanket = domain;
// Identity-map all of physical RAM (2 MiB leaves keep the table small even on a
// 64 GiB machine), then each firmware reserved region in case it sits outside the
// RAM extent (RMRRs must stay reachable under translation).
const top_of_ram = @as(u64, pmm.stats().total_frames) * page_size;
_ = map(domain, 0, top_of_ram);
var i: usize = 0;
while (i < info.rmrr_count) : (i += 1) {
const region = info.rmrr[i];
_ = map(domain, region.base, region.limit - region.base + 1);
}
}
/// Per-claimed-device record, so a driver's death detaches exactly the devices it held.
const Confined = struct { active: bool = false, owner: u32 = 0, bdf: u16 = 0 };
/// Per-claimed-device record: its private domain, so a driver's death tears down
/// exactly the domains it held.
const Confined = struct { active: bool = false, owner: u32 = 0, bdf: u16 = 0, domain: u16 = invalid_domain };
var confined: [maximum_domains]Confined = .{Confined{}} ** maximum_domains;
/// Place a just-claimed PCI function under IOMMU translation on behalf of `owner`. L1:
/// attach it to the blanket identity domain (its DMA works, but through real second-
/// level walks). false only if the machinery is unexpectedly unavailable — the caller
/// rolls the claim back. No-op success when no IOMMU exists (fail-open).
/// The interim DMA pool: the set of `dma_alloc`'d regions. Every such region is mapped
/// into every claimed device's domain, so a device reaches DMA buffers (including one
/// another's — the honest limit of this stage) but NOT the kernel, page tables, process
/// heaps, or arbitrary RAM. The capability layer (L3/L4) narrows this to per-grant.
const PoolRegion = struct { active: bool = false, physical: u64 = 0, len: u64 = 0 };
var pool: [maximum_pool]PoolRegion = .{PoolRegion{}} ** maximum_pool;
const maximum_pool = 256;
/// Place a just-claimed PCI function under IOMMU translation on behalf of `owner`: give
/// it a private empty domain, seed it with the current DMA pool and the device's own
/// firmware reserved region, and attach. false only if a domain can't be allocated —
/// the caller rolls the claim back (a claim that can't be confined must not stand). No-op
/// success when no IOMMU exists (fail-open).
pub fn confineDevice(device_id: u64, bdf: u16, owner: u32) bool {
if (kind == .none) return true;
if (blanket == invalid_domain) return false;
if (device_id >= confined.len) return true; // unusual id; leave it to fail-open
attachDevice(blanket, bdf);
confined[@intCast(device_id)] = .{ .active = true, .owner = owner, .bdf = bdf };
const domain = domainCreate(owner, bdf) orelse return false;
// Seed with every pooled DMA region so the device's own rings/buffers and the
// cross-process buffers handed to it (fat's bounce buffer via usb-storage) resolve.
for (&pool) |*r| {
if (r.active) _ = map(domain, r.physical, r.len);
}
// Firmware reserved region for this device, if any (real hardware; QEMU has none).
const info = platform.platformInformation();
var i: usize = 0;
while (i < info.rmrr_count) : (i += 1) {
if (info.rmrr[i].bdf == bdf)
_ = map(domain, info.rmrr[i].base, info.rmrr[i].limit - info.rmrr[i].base + 1);
}
attachDevice(domain, bdf);
confined[@intCast(device_id)] = .{ .active = true, .owner = owner, .bdf = bdf, .domain = domain };
return true;
}
/// A driver died or released its devices: detach every device it held so their DMA is
/// blocked again (a restarted driver re-claims and re-confines). Runs BEFORE the frames
/// and broker claims are released.
/// Record a newly `dma_alloc`'d region and map it into every claimed device's domain
/// (the interim pool rule). Called from the dma_alloc syscall.
pub fn poolAdd(physical: u64, len: u64) void {
if (kind == .none) return;
for (&pool) |*r| {
if (!r.active) {
r.* = .{ .active = true, .physical = physical, .len = len };
break;
}
} else return; // pool full; region stays unmapped and its device DMA will fault
for (&confined) |*c| {
if (c.active) _ = map(c.domain, physical, len);
}
}
/// Unmap a freed DMA region from every claimed device's domain and forget it. MUST run
/// before the frames return to pmm — a device translating to a reallocated frame is the
/// use-after-free this prevents. Called from the dma_free syscall.
pub fn poolRemove(physical: u64, len: u64) void {
if (kind == .none) return;
for (&pool) |*r| {
if (r.active and r.physical == physical and r.len == len) {
for (&confined) |*c| {
if (c.active) unmap(c.domain, physical, len);
}
r.* = .{};
return;
}
}
}
/// A driver died or released its devices: tear down every domain it held (detach the
/// device, free the tables) so their DMA is blocked again and a restarted driver
/// re-claims cleanly. Runs BEFORE the broker claims and the DMA frames are released.
pub fn releaseAllOwnedBy(owner: u32) void {
if (kind == .none) return;
for (&confined) |*c| {
if (c.active and c.owner == owner) {
detachDevice(c.bdf);
domainDestroy(c.domain);
c.* = .{};
}
}