iommu: per-device domains with interim DMA-pool enforcement
Replaces L1's shared blanket identity domain with a private translation
domain per claimed PCI function. A device now reaches only:
- the DMA pool: every dma_alloc'd region, mapped into every claimed
device's domain (poolAdd/poolRemove, driven from the dma_alloc and
dma_free syscalls). This keeps the cross-process buffer handoff
working (fat's bounce buffer reaches the xHC) while blocking the
kernel, page tables, process heaps, MMIO, and unallocated RAM.
- its own firmware reserved region (RMRR), seeded at confine time.
The pool is the honest interim: devices can still reach one another's
DMA buffers. The DMA-region capability layer (next) narrows it to
per-grant reachability.
dma_free unmaps from every domain and invalidates BEFORE the frames
return to the allocator, closing the stale-IOTLB use-after-free window.
Driver death tears down its domains (detach + free tables) before the
broker claims and DMA frames are released.
New iommu_fault_drain syscall (+ driver.iommuFaultDrain) forces pending
fault records to the log on demand. The new iommu-fault case proves it:
a claimed e1000e is programmed to DMA-fetch its TX ring from an unmapped
page; VT-d faults the access (bdf 00:03.0 addr 0x1000 reason 0x6) and the
system stays alive. 104/104.
This commit is contained in:
+72
-43
@@ -115,66 +115,95 @@ pub fn init() void {
|
||||
return;
|
||||
}
|
||||
|
||||
// L1 (enabled-but-invisible): every device is placed in a single blanket domain
|
||||
// that identity-maps all of RAM plus the firmware reserved regions, so translation
|
||||
// is on but nothing's DMA changes. L2 replaces this per device: on claim a device
|
||||
// is detached from the blanket and attached to its own empty domain, so it reaches
|
||||
// only what is explicitly mapped for it.
|
||||
buildBlanketDomain(info);
|
||||
// The root table starts empty: every device's context entry is not-present, so any
|
||||
// DMA faults until the device's driver claims it (confineDevice gives it a private
|
||||
// domain). PCI functions are enumerated post-boot by the ring-3 pci-bus driver, so
|
||||
// there is nothing to attach at init anyway.
|
||||
intel.enable();
|
||||
logEnabled(info);
|
||||
}
|
||||
|
||||
/// The shared identity domain claimed devices are placed in (L1). L2 replaces this with
|
||||
/// a private empty domain per device plus explicitly-granted mappings.
|
||||
var blanket: u16 = invalid_domain;
|
||||
|
||||
/// PCI functions are enumerated post-boot by the ring-3 pci-bus driver, so at init the
|
||||
/// device tree has none — a device is attached to the blanket when its driver *claims*
|
||||
/// it (confineDevice), which is always before the driver programs any DMA. Until then
|
||||
/// its context entry is not-present and its DMA faults (only stale firmware bus-
|
||||
/// mastering would hit that, which is the evidence M16 exists to surface).
|
||||
fn buildBlanketDomain(info: platform.PlatformInformation) void {
|
||||
const domain = domainCreate(0, 0) orelse return;
|
||||
blanket = domain;
|
||||
|
||||
// Identity-map all of physical RAM (2 MiB leaves keep the table small even on a
|
||||
// 64 GiB machine), then each firmware reserved region in case it sits outside the
|
||||
// RAM extent (RMRRs must stay reachable under translation).
|
||||
const top_of_ram = @as(u64, pmm.stats().total_frames) * page_size;
|
||||
_ = map(domain, 0, top_of_ram);
|
||||
var i: usize = 0;
|
||||
while (i < info.rmrr_count) : (i += 1) {
|
||||
const region = info.rmrr[i];
|
||||
_ = map(domain, region.base, region.limit - region.base + 1);
|
||||
}
|
||||
}
|
||||
|
||||
/// Per-claimed-device record, so a driver's death detaches exactly the devices it held.
|
||||
const Confined = struct { active: bool = false, owner: u32 = 0, bdf: u16 = 0 };
|
||||
/// Per-claimed-device record: its private domain, so a driver's death tears down
|
||||
/// exactly the domains it held.
|
||||
const Confined = struct { active: bool = false, owner: u32 = 0, bdf: u16 = 0, domain: u16 = invalid_domain };
|
||||
var confined: [maximum_domains]Confined = .{Confined{}} ** maximum_domains;
|
||||
|
||||
/// Place a just-claimed PCI function under IOMMU translation on behalf of `owner`. L1:
|
||||
/// attach it to the blanket identity domain (its DMA works, but through real second-
|
||||
/// level walks). false only if the machinery is unexpectedly unavailable — the caller
|
||||
/// rolls the claim back. No-op success when no IOMMU exists (fail-open).
|
||||
/// The interim DMA pool: the set of `dma_alloc`'d regions. Every such region is mapped
|
||||
/// into every claimed device's domain, so a device reaches DMA buffers (including one
|
||||
/// another's — the honest limit of this stage) but NOT the kernel, page tables, process
|
||||
/// heaps, or arbitrary RAM. The capability layer (L3/L4) narrows this to per-grant.
|
||||
const PoolRegion = struct { active: bool = false, physical: u64 = 0, len: u64 = 0 };
|
||||
var pool: [maximum_pool]PoolRegion = .{PoolRegion{}} ** maximum_pool;
|
||||
const maximum_pool = 256;
|
||||
|
||||
/// Place a just-claimed PCI function under IOMMU translation on behalf of `owner`: give
|
||||
/// it a private empty domain, seed it with the current DMA pool and the device's own
|
||||
/// firmware reserved region, and attach. false only if a domain can't be allocated —
|
||||
/// the caller rolls the claim back (a claim that can't be confined must not stand). No-op
|
||||
/// success when no IOMMU exists (fail-open).
|
||||
pub fn confineDevice(device_id: u64, bdf: u16, owner: u32) bool {
|
||||
if (kind == .none) return true;
|
||||
if (blanket == invalid_domain) return false;
|
||||
if (device_id >= confined.len) return true; // unusual id; leave it to fail-open
|
||||
attachDevice(blanket, bdf);
|
||||
confined[@intCast(device_id)] = .{ .active = true, .owner = owner, .bdf = bdf };
|
||||
const domain = domainCreate(owner, bdf) orelse return false;
|
||||
|
||||
// Seed with every pooled DMA region so the device's own rings/buffers and the
|
||||
// cross-process buffers handed to it (fat's bounce buffer via usb-storage) resolve.
|
||||
for (&pool) |*r| {
|
||||
if (r.active) _ = map(domain, r.physical, r.len);
|
||||
}
|
||||
// Firmware reserved region for this device, if any (real hardware; QEMU has none).
|
||||
const info = platform.platformInformation();
|
||||
var i: usize = 0;
|
||||
while (i < info.rmrr_count) : (i += 1) {
|
||||
if (info.rmrr[i].bdf == bdf)
|
||||
_ = map(domain, info.rmrr[i].base, info.rmrr[i].limit - info.rmrr[i].base + 1);
|
||||
}
|
||||
|
||||
attachDevice(domain, bdf);
|
||||
confined[@intCast(device_id)] = .{ .active = true, .owner = owner, .bdf = bdf, .domain = domain };
|
||||
return true;
|
||||
}
|
||||
|
||||
/// A driver died or released its devices: detach every device it held so their DMA is
|
||||
/// blocked again (a restarted driver re-claims and re-confines). Runs BEFORE the frames
|
||||
/// and broker claims are released.
|
||||
/// Record a newly `dma_alloc`'d region and map it into every claimed device's domain
|
||||
/// (the interim pool rule). Called from the dma_alloc syscall.
|
||||
pub fn poolAdd(physical: u64, len: u64) void {
|
||||
if (kind == .none) return;
|
||||
for (&pool) |*r| {
|
||||
if (!r.active) {
|
||||
r.* = .{ .active = true, .physical = physical, .len = len };
|
||||
break;
|
||||
}
|
||||
} else return; // pool full; region stays unmapped and its device DMA will fault
|
||||
for (&confined) |*c| {
|
||||
if (c.active) _ = map(c.domain, physical, len);
|
||||
}
|
||||
}
|
||||
|
||||
/// Unmap a freed DMA region from every claimed device's domain and forget it. MUST run
|
||||
/// before the frames return to pmm — a device translating to a reallocated frame is the
|
||||
/// use-after-free this prevents. Called from the dma_free syscall.
|
||||
pub fn poolRemove(physical: u64, len: u64) void {
|
||||
if (kind == .none) return;
|
||||
for (&pool) |*r| {
|
||||
if (r.active and r.physical == physical and r.len == len) {
|
||||
for (&confined) |*c| {
|
||||
if (c.active) unmap(c.domain, physical, len);
|
||||
}
|
||||
r.* = .{};
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// A driver died or released its devices: tear down every domain it held (detach the
|
||||
/// device, free the tables) so their DMA is blocked again and a restarted driver
|
||||
/// re-claims cleanly. Runs BEFORE the broker claims and the DMA frames are released.
|
||||
pub fn releaseAllOwnedBy(owner: u32) void {
|
||||
if (kind == .none) return;
|
||||
for (&confined) |*c| {
|
||||
if (c.active and c.owner == owner) {
|
||||
detachDevice(c.bdf);
|
||||
domainDestroy(c.domain);
|
||||
c.* = .{};
|
||||
}
|
||||
}
|
||||
|
||||
@@ -251,6 +251,7 @@ fn system_call(state: *architecture.CpuState) void {
|
||||
.fs_node => systemFsNode(state),
|
||||
.fs_mount => systemFsMount(state),
|
||||
.fs_unmount => systemFsUnmount(state),
|
||||
.iommu_fault_drain => systemIommuFaultDrain(state),
|
||||
.wall_clock => systemWallClock(state),
|
||||
.shared_memory_create => systemSharedMemoryCreate(state),
|
||||
.shared_memory_map => systemSharedMemoryMap(state),
|
||||
@@ -509,6 +510,17 @@ fn systemIoWrite(state: *architecture.CpuState) void {
|
||||
/// their physical address is never disclosed. `dma_below_4g` caps the physical address
|
||||
/// for legacy engines; `dma_write_combining` is accepted but falls back to coherent
|
||||
/// until PAT is programmed. See docs/driver-model.md (M14).
|
||||
/// iommu_fault_drain() -> count: drain and log any pending IOMMU translation faults,
|
||||
/// returning how many were seen. A diagnostic hook — a driver (or a test) that suspects
|
||||
/// its device faulted can force the fault records to be logged now rather than waiting
|
||||
/// for the next device-release drain. Harmless without an IOMMU (returns 0).
|
||||
fn systemIommuFaultDrain(state: *architecture.CpuState) void {
|
||||
const flags = sync.enter();
|
||||
const count = iommu.faultDrain();
|
||||
sync.leave(flags);
|
||||
architecture.setSystemCallResult(state, count);
|
||||
}
|
||||
|
||||
fn systemDmaAlloc(state: *architecture.CpuState) void {
|
||||
const len = architecture.systemCallArg(state, 0);
|
||||
const flags = architecture.systemCallArg(state, 1);
|
||||
@@ -555,6 +567,13 @@ fn systemDmaAlloc(state: *architecture.CpuState) void {
|
||||
architecture.mapUserDmaInto(t.address_space, base_v + i * page_size, phys + i * page_size, page_size);
|
||||
sync.leave(lock_flags);
|
||||
}
|
||||
// Publish the region to the DMA pool: it becomes reachable to every claimed device
|
||||
// (the interim rule until DMA-region capabilities land). No-op without an IOMMU.
|
||||
{
|
||||
const lock_flags = sync.enter();
|
||||
iommu.poolAdd(phys, pages * page_size);
|
||||
sync.leave(lock_flags);
|
||||
}
|
||||
architecture.setSystemCallResult(state, base_v); // virtual address for the CPU
|
||||
architecture.setSystemCallResult2(state, phys); // physical address for the device
|
||||
}
|
||||
@@ -572,6 +591,17 @@ fn systemDmaFree(state: *architecture.CpuState) void {
|
||||
const pages: usize = @intCast((len + page_size - 1) / page_size);
|
||||
if (base_v < dma_arena_base or base_v + pages * page_size > dma_arena_end) return fail(state);
|
||||
|
||||
// Pull the region out of every device's domain and invalidate BEFORE any frame
|
||||
// returns to the allocator — a device still translating to a reallocated frame is a
|
||||
// use-after-free. dma_alloc's frames are contiguous, so the base translation names
|
||||
// the whole region.
|
||||
{
|
||||
const lock_flags = sync.enter();
|
||||
if (architecture.translate(t.address_space, base_v)) |base_phys|
|
||||
iommu.poolRemove(base_phys, pages * page_size);
|
||||
sync.leave(lock_flags);
|
||||
}
|
||||
|
||||
for (0..pages) |i| {
|
||||
const va = base_v + i * page_size;
|
||||
// Per-page lock hold: the translate/unmap walks the shared page tables
|
||||
|
||||
@@ -215,6 +215,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
|
||||
pciScanTest(boot_information);
|
||||
} else if (eql(case, "pci-caps")) {
|
||||
pciCapsTest(boot_information);
|
||||
} else if (eql(case, "iommu-fault")) {
|
||||
iommuFaultTest(boot_information);
|
||||
} else if (eql(case, "acpi-parse")) {
|
||||
acpiParseTest(boot_information);
|
||||
} else if (eql(case, "acpi-report")) {
|
||||
@@ -2453,6 +2455,39 @@ fn pciCapsTest(boot_information: *const BootInformation) void {
|
||||
result();
|
||||
}
|
||||
|
||||
/// IOMMU enforcement, the negative proof: boot with VT-d on and an unclaimed e1000e.
|
||||
/// The manager spawns pci-bus, the fixture claims the NIC and fires a DMA at an
|
||||
/// unmapped page; the unit must fault it and the system survive. Substance is asserted
|
||||
/// by the harness on the kernel's DANOS-IOMMU-FAULT line and the fixture's markers.
|
||||
fn iommuFaultTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: iommu-fault\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over an initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
|
||||
check("IOMMU enabled for the enforcement test", iommu.enabled());
|
||||
process.setInitialRamdisk(image);
|
||||
var manager: u32 = 0;
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
|
||||
manager = process.spawnProcessSupervised(item.blob, 4, &.{"device-manager"}, scheduler.currentId(), null) catch 0;
|
||||
break;
|
||||
}
|
||||
check("device-manager spawned", manager != 0);
|
||||
check("iommu-fault-test spawned", spawnNamed(rd, "iommu-fault-test"));
|
||||
result();
|
||||
}
|
||||
|
||||
/// M19.1: the ring-3 PCI scan agrees with the kernel's. The manager spawns
|
||||
/// pci-bus for the host bridge; the driver walks the same ECAM window through
|
||||
/// its mmio_map grant and must find exactly the functions the kernel's own
|
||||
|
||||
Reference in New Issue
Block a user