iommu: per-device domains with interim DMA-pool enforcement

Replaces L1's shared blanket identity domain with a private translation
domain per claimed PCI function. A device now reaches only:
  - the DMA pool: every dma_alloc'd region, mapped into every claimed
    device's domain (poolAdd/poolRemove, driven from the dma_alloc and
    dma_free syscalls). This keeps the cross-process buffer handoff
    working (fat's bounce buffer reaches the xHC) while blocking the
    kernel, page tables, process heaps, MMIO, and unallocated RAM.
  - its own firmware reserved region (RMRR), seeded at confine time.
The pool is the honest interim: devices can still reach one another's
DMA buffers. The DMA-region capability layer (next) narrows it to
per-grant reachability.

dma_free unmaps from every domain and invalidates BEFORE the frames
return to the allocator, closing the stale-IOTLB use-after-free window.
Driver death tears down its domains (detach + free tables) before the
broker claims and DMA frames are released.

New iommu_fault_drain syscall (+ driver.iommuFaultDrain) forces pending
fault records to the log on demand. The new iommu-fault case proves it:
a claimed e1000e is programmed to DMA-fetch its TX ring from an unmapped
page; VT-d faults the access (bdf 00:03.0 addr 0x1000 reason 0x6) and the
system stays alive. 104/104.
This commit is contained in:
Daniel Samson
2026-07-26 18:31:27 +01:00
parent f477ef7d9f
commit e94adcfc02
8 changed files with 280 additions and 43 deletions
+6
View File
@@ -737,6 +737,11 @@ pub fn build(b: *std.Build) void {
const pci_cap_test_exe = addUserBinary(b, kernel_target, &default_imports, "pci-cap-test", "test/system/services/pci-cap-test/pci-cap-test.zig"); const pci_cap_test_exe = addUserBinary(b, kernel_target, &default_imports, "pci-cap-test", "test/system/services/pci-cap-test/pci-cap-test.zig");
programModule(pci_cap_test_exe).addImport("pci", pci_module); programModule(pci_cap_test_exe).addImport("pci", pci_module);
programModule(pci_cap_test_exe).addImport("pci-class", pci_class_module); programModule(pci_cap_test_exe).addImport("pci-class", pci_class_module);
// The IOMMU-enforcement negative test: claims an unclaimed e1000e and fires a rogue
// DMA that VT-d must fault. Same PCI building blocks as pci-cap-test.
const iommu_fault_test_exe = addUserBinary(b, kernel_target, &default_imports, "iommu-fault-test", "test/system/services/iommu-fault-test/iommu-fault-test.zig");
programModule(iommu_fault_test_exe).addImport("pci", pci_module);
programModule(iommu_fault_test_exe).addImport("pci-class", pci_class_module);
// The discovery service: one swappable process per firmware // The discovery service: one swappable process per firmware
// (docs/discovery.md), bundled under the neutral ramdisk name // (docs/discovery.md), bundled under the neutral ramdisk name
// "discovery" so the device manager never learns which firmware it is on. // "discovery" so the device manager never learns which firmware it is on.
@@ -822,6 +827,7 @@ pub fn build(b: *std.Build) void {
.{ .path = "test/system/services/crash-test", .binary = crash_test_exe.getEmittedBin() }, .{ .path = "test/system/services/crash-test", .binary = crash_test_exe.getEmittedBin() },
.{ .path = "test/system/services/device-list", .binary = device_list_exe.getEmittedBin() }, .{ .path = "test/system/services/device-list", .binary = device_list_exe.getEmittedBin() },
.{ .path = "test/system/services/pci-cap-test", .binary = pci_cap_test_exe.getEmittedBin() }, .{ .path = "test/system/services/pci-cap-test", .binary = pci_cap_test_exe.getEmittedBin() },
.{ .path = "test/system/services/iommu-fault-test", .binary = iommu_fault_test_exe.getEmittedBin() },
.{ .path = "test/system/services/input-source", .binary = input_source_exe.getEmittedBin() }, .{ .path = "test/system/services/input-source", .binary = input_source_exe.getEmittedBin() },
.{ .path = "test/system/services/input-test", .binary = input_test_exe.getEmittedBin() }, .{ .path = "test/system/services/input-test", .binary = input_test_exe.getEmittedBin() },
.{ .path = "test/system/services/args-echo", .binary = args_echo_exe.getEmittedBin() }, .{ .path = "test/system/services/args-echo", .binary = args_echo_exe.getEmittedBin() },
+8
View File
@@ -99,6 +99,14 @@ pub fn msiBind(device_id: u64, endpoint: usize) ?Msi {
return .{ .address = rax, .data = @intCast(rdx) }; return .{ .address = rax, .data = @intCast(rdx) };
} }
/// Drain and log any pending IOMMU translation faults, returning the count seen. A
/// diagnostic: a driver that suspects its device attempted an out-of-domain DMA (or a
/// test proving enforcement) forces the hardware's fault records to the log now. Returns
/// 0 when no IOMMU is present.
pub fn iommuFaultDrain() usize {
return sc.systemCall0(.iommu_fault_drain);
}
/// Read `width` bytes (1, 2, or 4) from a port in a claimed device's `io_port` /// Read `width` bytes (1, 2, or 4) from a port in a claimed device's `io_port`
/// resource, at byte `offset` within it. Ring 3 has no direct `in`/`out`, so a legacy /// resource, at byte `offset` within it. Ring 3 has no direct `in`/`out`, so a legacy
/// driver (PS/2, 16550 UART) reaches its ports through this claim-gated call — each /// driver (PS/2, 16550 UART) reaches its ports through this claim-gated call — each
+1
View File
@@ -76,6 +76,7 @@ pub const SystemCall = enum(u64) {
fs_node = 47, // fs_node(op, node_token, offset, buf_ptr, buf_len) -> bytes/0/-errno: read/status/readdir on a kernel-served node (op values mirror the vfs-protocol Operation numbers) fs_node = 47, // fs_node(op, node_token, offset, buf_ptr, buf_len) -> bytes/0/-errno: read/status/readdir on a kernel-served node (op values mirror the vfs-protocol Operation numbers)
fs_mount = 48, // fs_mount(prefix_ptr, prefix_len, backend_handle, rewrite_ptr, rewrite_len) -> 0/-errno: mount a userspace filesystem's endpoint at an absolute prefix (possession of the handle is the capability) fs_mount = 48, // fs_mount(prefix_ptr, prefix_len, backend_handle, rewrite_ptr, rewrite_len) -> 0/-errno: mount a userspace filesystem's endpoint at an absolute prefix (possession of the handle is the capability)
fs_unmount = 49, // fs_unmount(prefix_ptr, prefix_len) -> 0/-errno: remove a backend mount fs_unmount = 49, // fs_unmount(prefix_ptr, prefix_len) -> 0/-errno: remove a backend mount
iommu_fault_drain = 50, // iommu_fault_drain() -> count: drain + log pending IOMMU translation faults (a diagnostic; the count of faults seen this call)
_, _,
}; };
+72 -43
View File
@@ -115,66 +115,95 @@ pub fn init() void {
return; return;
} }
// L1 (enabled-but-invisible): every device is placed in a single blanket domain // The root table starts empty: every device's context entry is not-present, so any
// that identity-maps all of RAM plus the firmware reserved regions, so translation // DMA faults until the device's driver claims it (confineDevice gives it a private
// is on but nothing's DMA changes. L2 replaces this per device: on claim a device // domain). PCI functions are enumerated post-boot by the ring-3 pci-bus driver, so
// is detached from the blanket and attached to its own empty domain, so it reaches // there is nothing to attach at init anyway.
// only what is explicitly mapped for it.
buildBlanketDomain(info);
intel.enable(); intel.enable();
logEnabled(info); logEnabled(info);
} }
/// The shared identity domain claimed devices are placed in (L1). L2 replaces this with /// Per-claimed-device record: its private domain, so a driver's death tears down
/// a private empty domain per device plus explicitly-granted mappings. /// exactly the domains it held.
var blanket: u16 = invalid_domain; const Confined = struct { active: bool = false, owner: u32 = 0, bdf: u16 = 0, domain: u16 = invalid_domain };
/// PCI functions are enumerated post-boot by the ring-3 pci-bus driver, so at init the
/// device tree has none — a device is attached to the blanket when its driver *claims*
/// it (confineDevice), which is always before the driver programs any DMA. Until then
/// its context entry is not-present and its DMA faults (only stale firmware bus-
/// mastering would hit that, which is the evidence M16 exists to surface).
fn buildBlanketDomain(info: platform.PlatformInformation) void {
const domain = domainCreate(0, 0) orelse return;
blanket = domain;
// Identity-map all of physical RAM (2 MiB leaves keep the table small even on a
// 64 GiB machine), then each firmware reserved region in case it sits outside the
// RAM extent (RMRRs must stay reachable under translation).
const top_of_ram = @as(u64, pmm.stats().total_frames) * page_size;
_ = map(domain, 0, top_of_ram);
var i: usize = 0;
while (i < info.rmrr_count) : (i += 1) {
const region = info.rmrr[i];
_ = map(domain, region.base, region.limit - region.base + 1);
}
}
/// Per-claimed-device record, so a driver's death detaches exactly the devices it held.
const Confined = struct { active: bool = false, owner: u32 = 0, bdf: u16 = 0 };
var confined: [maximum_domains]Confined = .{Confined{}} ** maximum_domains; var confined: [maximum_domains]Confined = .{Confined{}} ** maximum_domains;
/// Place a just-claimed PCI function under IOMMU translation on behalf of `owner`. L1: /// The interim DMA pool: the set of `dma_alloc`'d regions. Every such region is mapped
/// attach it to the blanket identity domain (its DMA works, but through real second- /// into every claimed device's domain, so a device reaches DMA buffers (including one
/// level walks). false only if the machinery is unexpectedly unavailable — the caller /// another's — the honest limit of this stage) but NOT the kernel, page tables, process
/// rolls the claim back. No-op success when no IOMMU exists (fail-open). /// heaps, or arbitrary RAM. The capability layer (L3/L4) narrows this to per-grant.
const PoolRegion = struct { active: bool = false, physical: u64 = 0, len: u64 = 0 };
var pool: [maximum_pool]PoolRegion = .{PoolRegion{}} ** maximum_pool;
const maximum_pool = 256;
/// Place a just-claimed PCI function under IOMMU translation on behalf of `owner`: give
/// it a private empty domain, seed it with the current DMA pool and the device's own
/// firmware reserved region, and attach. false only if a domain can't be allocated —
/// the caller rolls the claim back (a claim that can't be confined must not stand). No-op
/// success when no IOMMU exists (fail-open).
pub fn confineDevice(device_id: u64, bdf: u16, owner: u32) bool { pub fn confineDevice(device_id: u64, bdf: u16, owner: u32) bool {
if (kind == .none) return true; if (kind == .none) return true;
if (blanket == invalid_domain) return false;
if (device_id >= confined.len) return true; // unusual id; leave it to fail-open if (device_id >= confined.len) return true; // unusual id; leave it to fail-open
attachDevice(blanket, bdf); const domain = domainCreate(owner, bdf) orelse return false;
confined[@intCast(device_id)] = .{ .active = true, .owner = owner, .bdf = bdf };
// Seed with every pooled DMA region so the device's own rings/buffers and the
// cross-process buffers handed to it (fat's bounce buffer via usb-storage) resolve.
for (&pool) |*r| {
if (r.active) _ = map(domain, r.physical, r.len);
}
// Firmware reserved region for this device, if any (real hardware; QEMU has none).
const info = platform.platformInformation();
var i: usize = 0;
while (i < info.rmrr_count) : (i += 1) {
if (info.rmrr[i].bdf == bdf)
_ = map(domain, info.rmrr[i].base, info.rmrr[i].limit - info.rmrr[i].base + 1);
}
attachDevice(domain, bdf);
confined[@intCast(device_id)] = .{ .active = true, .owner = owner, .bdf = bdf, .domain = domain };
return true; return true;
} }
/// A driver died or released its devices: detach every device it held so their DMA is /// Record a newly `dma_alloc`'d region and map it into every claimed device's domain
/// blocked again (a restarted driver re-claims and re-confines). Runs BEFORE the frames /// (the interim pool rule). Called from the dma_alloc syscall.
/// and broker claims are released. pub fn poolAdd(physical: u64, len: u64) void {
if (kind == .none) return;
for (&pool) |*r| {
if (!r.active) {
r.* = .{ .active = true, .physical = physical, .len = len };
break;
}
} else return; // pool full; region stays unmapped and its device DMA will fault
for (&confined) |*c| {
if (c.active) _ = map(c.domain, physical, len);
}
}
/// Unmap a freed DMA region from every claimed device's domain and forget it. MUST run
/// before the frames return to pmm — a device translating to a reallocated frame is the
/// use-after-free this prevents. Called from the dma_free syscall.
pub fn poolRemove(physical: u64, len: u64) void {
if (kind == .none) return;
for (&pool) |*r| {
if (r.active and r.physical == physical and r.len == len) {
for (&confined) |*c| {
if (c.active) unmap(c.domain, physical, len);
}
r.* = .{};
return;
}
}
}
/// A driver died or released its devices: tear down every domain it held (detach the
/// device, free the tables) so their DMA is blocked again and a restarted driver
/// re-claims cleanly. Runs BEFORE the broker claims and the DMA frames are released.
pub fn releaseAllOwnedBy(owner: u32) void { pub fn releaseAllOwnedBy(owner: u32) void {
if (kind == .none) return; if (kind == .none) return;
for (&confined) |*c| { for (&confined) |*c| {
if (c.active and c.owner == owner) { if (c.active and c.owner == owner) {
detachDevice(c.bdf); detachDevice(c.bdf);
domainDestroy(c.domain);
c.* = .{}; c.* = .{};
} }
} }
+30
View File
@@ -251,6 +251,7 @@ fn system_call(state: *architecture.CpuState) void {
.fs_node => systemFsNode(state), .fs_node => systemFsNode(state),
.fs_mount => systemFsMount(state), .fs_mount => systemFsMount(state),
.fs_unmount => systemFsUnmount(state), .fs_unmount => systemFsUnmount(state),
.iommu_fault_drain => systemIommuFaultDrain(state),
.wall_clock => systemWallClock(state), .wall_clock => systemWallClock(state),
.shared_memory_create => systemSharedMemoryCreate(state), .shared_memory_create => systemSharedMemoryCreate(state),
.shared_memory_map => systemSharedMemoryMap(state), .shared_memory_map => systemSharedMemoryMap(state),
@@ -509,6 +510,17 @@ fn systemIoWrite(state: *architecture.CpuState) void {
/// their physical address is never disclosed. `dma_below_4g` caps the physical address /// their physical address is never disclosed. `dma_below_4g` caps the physical address
/// for legacy engines; `dma_write_combining` is accepted but falls back to coherent /// for legacy engines; `dma_write_combining` is accepted but falls back to coherent
/// until PAT is programmed. See docs/driver-model.md (M14). /// until PAT is programmed. See docs/driver-model.md (M14).
/// iommu_fault_drain() -> count: drain and log any pending IOMMU translation faults,
/// returning how many were seen. A diagnostic hook — a driver (or a test) that suspects
/// its device faulted can force the fault records to be logged now rather than waiting
/// for the next device-release drain. Harmless without an IOMMU (returns 0).
fn systemIommuFaultDrain(state: *architecture.CpuState) void {
const flags = sync.enter();
const count = iommu.faultDrain();
sync.leave(flags);
architecture.setSystemCallResult(state, count);
}
fn systemDmaAlloc(state: *architecture.CpuState) void { fn systemDmaAlloc(state: *architecture.CpuState) void {
const len = architecture.systemCallArg(state, 0); const len = architecture.systemCallArg(state, 0);
const flags = architecture.systemCallArg(state, 1); const flags = architecture.systemCallArg(state, 1);
@@ -555,6 +567,13 @@ fn systemDmaAlloc(state: *architecture.CpuState) void {
architecture.mapUserDmaInto(t.address_space, base_v + i * page_size, phys + i * page_size, page_size); architecture.mapUserDmaInto(t.address_space, base_v + i * page_size, phys + i * page_size, page_size);
sync.leave(lock_flags); sync.leave(lock_flags);
} }
// Publish the region to the DMA pool: it becomes reachable to every claimed device
// (the interim rule until DMA-region capabilities land). No-op without an IOMMU.
{
const lock_flags = sync.enter();
iommu.poolAdd(phys, pages * page_size);
sync.leave(lock_flags);
}
architecture.setSystemCallResult(state, base_v); // virtual address for the CPU architecture.setSystemCallResult(state, base_v); // virtual address for the CPU
architecture.setSystemCallResult2(state, phys); // physical address for the device architecture.setSystemCallResult2(state, phys); // physical address for the device
} }
@@ -572,6 +591,17 @@ fn systemDmaFree(state: *architecture.CpuState) void {
const pages: usize = @intCast((len + page_size - 1) / page_size); const pages: usize = @intCast((len + page_size - 1) / page_size);
if (base_v < dma_arena_base or base_v + pages * page_size > dma_arena_end) return fail(state); if (base_v < dma_arena_base or base_v + pages * page_size > dma_arena_end) return fail(state);
// Pull the region out of every device's domain and invalidate BEFORE any frame
// returns to the allocator — a device still translating to a reallocated frame is a
// use-after-free. dma_alloc's frames are contiguous, so the base translation names
// the whole region.
{
const lock_flags = sync.enter();
if (architecture.translate(t.address_space, base_v)) |base_phys|
iommu.poolRemove(base_phys, pages * page_size);
sync.leave(lock_flags);
}
for (0..pages) |i| { for (0..pages) |i| {
const va = base_v + i * page_size; const va = base_v + i * page_size;
// Per-page lock hold: the translate/unmap walks the shared page tables // Per-page lock hold: the translate/unmap walks the shared page tables
+35
View File
@@ -215,6 +215,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
pciScanTest(boot_information); pciScanTest(boot_information);
} else if (eql(case, "pci-caps")) { } else if (eql(case, "pci-caps")) {
pciCapsTest(boot_information); pciCapsTest(boot_information);
} else if (eql(case, "iommu-fault")) {
iommuFaultTest(boot_information);
} else if (eql(case, "acpi-parse")) { } else if (eql(case, "acpi-parse")) {
acpiParseTest(boot_information); acpiParseTest(boot_information);
} else if (eql(case, "acpi-report")) { } else if (eql(case, "acpi-report")) {
@@ -2453,6 +2455,39 @@ fn pciCapsTest(boot_information: *const BootInformation) void {
result(); result();
} }
/// IOMMU enforcement, the negative proof: boot with VT-d on and an unclaimed e1000e.
/// The manager spawns pci-bus, the fixture claims the NIC and fires a DMA at an
/// unmapped page; the unit must fault it and the system survive. Substance is asserted
/// by the harness on the kernel's DANOS-IOMMU-FAULT line and the fixture's markers.
fn iommuFaultTest(boot_information: *const BootInformation) void {
log("DANOS-TEST-BEGIN: iommu-fault\n", .{});
if (boot_information.initial_ramdisk_len == 0) {
check("bootloader handed over an initial_ramdisk", false);
result();
return;
}
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
const rd = initial_ramdisk.Reader.init(image) orelse {
check("initial_ramdisk image is valid", false);
result();
return;
};
check("IOMMU enabled for the enforcement test", iommu.enabled());
process.setInitialRamdisk(image);
var manager: u32 = 0;
var i: u32 = 0;
while (i < rd.count) : (i += 1) {
const item = rd.entry(i) orelse continue;
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
manager = process.spawnProcessSupervised(item.blob, 4, &.{"device-manager"}, scheduler.currentId(), null) catch 0;
break;
}
check("device-manager spawned", manager != 0);
check("iommu-fault-test spawned", spawnNamed(rd, "iommu-fault-test"));
result();
}
/// M19.1: the ring-3 PCI scan agrees with the kernel's. The manager spawns /// M19.1: the ring-3 PCI scan agrees with the kernel's. The manager spawns
/// pci-bus for the host bridge; the driver walks the same ECAM window through /// pci-bus for the host bridge; the driver walks the same ECAM window through
/// its mmio_map grant and must find exactly the functions the kernel's own /// its mmio_map grant and must find exactly the functions the kernel's own
+9
View File
@@ -184,6 +184,15 @@ CASES = [
"qemu_extra": ["-device", "intel-iommu,intremap=off"], "qemu_extra": ["-device", "intel-iommu,intremap=off"],
"expect": r"(?s)(?=.*/system/kernel: iommu online)(?=.*usb-hid-keyboard: ok)(?=.*usb-hid-mouse: ok)", "expect": r"(?s)(?=.*/system/kernel: iommu online)(?=.*usb-hid-keyboard: ok)(?=.*usb-hid-mouse: ok)",
"fail": r"DANOS-TEST-RESULT: FAIL|DANOS-IOMMU-FAULT"}, "fail": r"DANOS-TEST-RESULT: FAIL|DANOS-IOMMU-FAULT"},
# Enforcement, the negative proof: a claimed e1000e fires a DMA at an unmapped page;
# VT-d must fault it (logged) and the system must stay alive. The fault line is the
# point here, so unlike the positive cases it appears in `expect`, not `fail`.
{"name": "iommu-fault",
"smp": 4,
"timeout": 120,
"qemu_extra": ["-device", "intel-iommu,intremap=off", "-device", "e1000e"],
"expect": r"(?s)(?=.*DANOS-IOMMU-FAULT: bdf=)(?=.*iommu-fault-test: system alive)(?=.*DANOS-TEST-RESULT: PASS)",
"fail": r"iommu-fault-test: FAIL|DANOS-TEST-RESULT: FAIL|CPU EXCEPTION|KERNEL PANIC"},
# Port I/O grants: a claimed device's io_port resource lets a driver read/write its # Port I/O grants: a claimed device's io_port resource lets a driver read/write its
# ports (PS/2 status 0x64), gated by the claim; out-of-range/unclaimed is refused. # ports (PS/2 status 0x64), gated by the claim; out-of-range/unclaimed is refused.
{"name": "ioport", {"name": "ioport",
@@ -0,0 +1,119 @@
//! iommu-fault-test — the negative proof for IOMMU enforcement. The iommu-fault QEMU
//! case boots with VT-d enabled and an extra e1000e NIC no danos driver claims; this
//! fixture claims it, then deliberately programs its transmit engine to DMA from a
//! physical address that was never dma_alloc'd (so it is in no device's domain). The
//! IOMMU must fault that access — the descriptor fetch never reaches memory — and the
//! system must stay alive. A kernel `DANOS-IOMMU-FAULT` line plus this fixture's
//! `system alive` marker is the pass.
//!
//! The rogue target is the e1000e's transmit descriptor RING base itself: the very first
//! DMA the engine issues on a doorbell write is the descriptor fetch from that base, so
//! pointing the ring at an unmapped page makes the first access the faulting one — no
//! valid descriptor need be crafted.
const std = @import("std");
const device = @import("driver");
const time = @import("time");
const logging = @import("logging");
const mmio = @import("mmio");
const pci = @import("pci");
const pci_class = @import("pci-class");
const intel_vendor: u16 = 0x8086;
const e1000e_device: u16 = 0x10D3;
const ethernet_class: u64 = pci_class.ClassCode.pack(.{
.base = @intFromEnum(pci_class.BaseClass.network),
.subclass = @intFromEnum(pci_class.network.SubClass.ethernet),
.prog_if = 0,
});
// e1000e transmit-engine registers (Intel 82574L datasheet §Register Descriptions),
// byte offsets within BAR0. VERIFY-AGAINST-SPEC held on first bring-up: these are the
// legacy TX ring registers.
const reg_tctl = 0x0400; // Transmit Control
const reg_tdbal = 0x3800; // TX Descriptor Base Address Low
const reg_tdbah = 0x3804; // TX Descriptor Base Address High
const reg_tdlen = 0x3808; // TX Descriptor Length (bytes, 128-byte aligned)
const reg_tdh = 0x3810; // TX Descriptor Head
const reg_tdt = 0x3818; // TX Descriptor Tail
const tctl_en: u32 = 1 << 1; // Transmit Enable
const tctl_psp: u32 = 1 << 3; // Pad Short Packets
/// A low physical page that user DMA never touches — never returned by dma_alloc (whose
/// arena is far higher), so it is in no device's IOMMU domain. The e1000e's descriptor
/// fetch from here is exactly the out-of-domain access the unit must block.
const rogue_physical: u64 = 0x1000;
var descriptor: device.DeviceDescriptor = undefined;
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
var line: [128]u8 = undefined;
_ = logging.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
}
pub fn main() void {
const nic_id: u64 = found: {
var tries: u32 = 0;
while (tries < 150) : (tries += 1) {
var descriptors: [64]device.DeviceDescriptor = undefined;
const total = device.enumerate(&descriptors);
for (descriptors[0..@min(total, descriptors.len)]) |*entry| {
if (entry.class == @intFromEnum(device.DeviceClass.pci_device) and entry.pci_class == ethernet_class) {
descriptor = entry.*;
break :found entry.id;
}
}
time.sleepMillis(100);
}
_ = logging.write("iommu-fault-test: FAIL no ethernet function found\n");
return;
};
if (!device.claim(nic_id)) {
_ = logging.write("iommu-fault-test: FAIL claim\n");
return;
}
var function = pci.Function.map(nic_id, &descriptor) orelse {
_ = logging.write("iommu-fault-test: FAIL config-space map\n");
return;
};
if (function.vendorId() != intel_vendor or function.deviceId() != e1000e_device) {
_ = logging.write("iommu-fault-test: FAIL not an e1000e\n");
return;
}
function.enableMemoryAndBusMaster();
const bar0 = function.mapBar(0) orelse {
_ = logging.write("iommu-fault-test: FAIL map BAR0\n");
return;
};
// Point the TX ring at the rogue page and kick the engine: TDBA = rogue, a non-zero
// length, head=0, enable, then tail=1 so the engine fetches descriptor 0 — a DMA
// read from the rogue page, which the IOMMU must fault.
_ = logging.write("iommu-fault-test: pointing e1000e TX ring at an unmapped page\n");
mmio.writeRegister(u32, bar0 + reg_tdbal, @truncate(rogue_physical));
mmio.writeRegister(u32, bar0 + reg_tdbah, @intCast(rogue_physical >> 32));
mmio.writeRegister(u32, bar0 + reg_tdlen, 128);
mmio.writeRegister(u32, bar0 + reg_tdh, 0);
mmio.writeRegister(u32, bar0 + reg_tctl, tctl_en | tctl_psp);
mmio.writeMemoryBarrier();
mmio.writeRegister(u32, bar0 + reg_tdt, 1); // doorbell: fetch descriptor 0
// Give the engine time to attempt the fetch, then force the fault records to the log.
time.sleepMillis(200);
const faults = device.iommuFaultDrain();
writeLine("iommu-fault-test: drained {d} iommu fault(s)\n", .{faults});
if (faults == 0) {
_ = logging.write("iommu-fault-test: FAIL rogue DMA was not blocked\n");
return;
}
// Liveness: the system survived the blocked DMA — read our own config space back.
if (function.vendorId() != intel_vendor) {
_ = logging.write("iommu-fault-test: FAIL device unreadable after fault\n");
return;
}
_ = logging.write("iommu-fault-test: system alive\n");
}