diff --git a/build.zig b/build.zig index 8cee9a9..422281e 100644 --- a/build.zig +++ b/build.zig @@ -737,6 +737,11 @@ pub fn build(b: *std.Build) void { const pci_cap_test_exe = addUserBinary(b, kernel_target, &default_imports, "pci-cap-test", "test/system/services/pci-cap-test/pci-cap-test.zig"); programModule(pci_cap_test_exe).addImport("pci", pci_module); programModule(pci_cap_test_exe).addImport("pci-class", pci_class_module); + // The IOMMU-enforcement negative test: claims an unclaimed e1000e and fires a rogue + // DMA that VT-d must fault. Same PCI building blocks as pci-cap-test. + const iommu_fault_test_exe = addUserBinary(b, kernel_target, &default_imports, "iommu-fault-test", "test/system/services/iommu-fault-test/iommu-fault-test.zig"); + programModule(iommu_fault_test_exe).addImport("pci", pci_module); + programModule(iommu_fault_test_exe).addImport("pci-class", pci_class_module); // The discovery service: one swappable process per firmware // (docs/discovery.md), bundled under the neutral ramdisk name // "discovery" so the device manager never learns which firmware it is on. @@ -822,6 +827,7 @@ pub fn build(b: *std.Build) void { .{ .path = "test/system/services/crash-test", .binary = crash_test_exe.getEmittedBin() }, .{ .path = "test/system/services/device-list", .binary = device_list_exe.getEmittedBin() }, .{ .path = "test/system/services/pci-cap-test", .binary = pci_cap_test_exe.getEmittedBin() }, + .{ .path = "test/system/services/iommu-fault-test", .binary = iommu_fault_test_exe.getEmittedBin() }, .{ .path = "test/system/services/input-source", .binary = input_source_exe.getEmittedBin() }, .{ .path = "test/system/services/input-test", .binary = input_test_exe.getEmittedBin() }, .{ .path = "test/system/services/args-echo", .binary = args_echo_exe.getEmittedBin() }, diff --git a/library/device/driver/driver.zig b/library/device/driver/driver.zig index b6e89f2..4d65a21 100644 --- a/library/device/driver/driver.zig +++ b/library/device/driver/driver.zig @@ -99,6 +99,14 @@ pub fn msiBind(device_id: u64, endpoint: usize) ?Msi { return .{ .address = rax, .data = @intCast(rdx) }; } +/// Drain and log any pending IOMMU translation faults, returning the count seen. A +/// diagnostic: a driver that suspects its device attempted an out-of-domain DMA (or a +/// test proving enforcement) forces the hardware's fault records to the log now. Returns +/// 0 when no IOMMU is present. +pub fn iommuFaultDrain() usize { + return sc.systemCall0(.iommu_fault_drain); +} + /// Read `width` bytes (1, 2, or 4) from a port in a claimed device's `io_port` /// resource, at byte `offset` within it. Ring 3 has no direct `in`/`out`, so a legacy /// driver (PS/2, 16550 UART) reaches its ports through this claim-gated call — each diff --git a/system/abi.zig b/system/abi.zig index 47f53b6..7531996 100644 --- a/system/abi.zig +++ b/system/abi.zig @@ -76,6 +76,7 @@ pub const SystemCall = enum(u64) { fs_node = 47, // fs_node(op, node_token, offset, buf_ptr, buf_len) -> bytes/0/-errno: read/status/readdir on a kernel-served node (op values mirror the vfs-protocol Operation numbers) fs_mount = 48, // fs_mount(prefix_ptr, prefix_len, backend_handle, rewrite_ptr, rewrite_len) -> 0/-errno: mount a userspace filesystem's endpoint at an absolute prefix (possession of the handle is the capability) fs_unmount = 49, // fs_unmount(prefix_ptr, prefix_len) -> 0/-errno: remove a backend mount + iommu_fault_drain = 50, // iommu_fault_drain() -> count: drain + log pending IOMMU translation faults (a diagnostic; the count of faults seen this call) _, }; diff --git a/system/kernel/iommu.zig b/system/kernel/iommu.zig index 6e64f33..c841ba3 100644 --- a/system/kernel/iommu.zig +++ b/system/kernel/iommu.zig @@ -115,66 +115,95 @@ pub fn init() void { return; } - // L1 (enabled-but-invisible): every device is placed in a single blanket domain - // that identity-maps all of RAM plus the firmware reserved regions, so translation - // is on but nothing's DMA changes. L2 replaces this per device: on claim a device - // is detached from the blanket and attached to its own empty domain, so it reaches - // only what is explicitly mapped for it. - buildBlanketDomain(info); + // The root table starts empty: every device's context entry is not-present, so any + // DMA faults until the device's driver claims it (confineDevice gives it a private + // domain). PCI functions are enumerated post-boot by the ring-3 pci-bus driver, so + // there is nothing to attach at init anyway. intel.enable(); logEnabled(info); } -/// The shared identity domain claimed devices are placed in (L1). L2 replaces this with -/// a private empty domain per device plus explicitly-granted mappings. -var blanket: u16 = invalid_domain; - -/// PCI functions are enumerated post-boot by the ring-3 pci-bus driver, so at init the -/// device tree has none — a device is attached to the blanket when its driver *claims* -/// it (confineDevice), which is always before the driver programs any DMA. Until then -/// its context entry is not-present and its DMA faults (only stale firmware bus- -/// mastering would hit that, which is the evidence M16 exists to surface). -fn buildBlanketDomain(info: platform.PlatformInformation) void { - const domain = domainCreate(0, 0) orelse return; - blanket = domain; - - // Identity-map all of physical RAM (2 MiB leaves keep the table small even on a - // 64 GiB machine), then each firmware reserved region in case it sits outside the - // RAM extent (RMRRs must stay reachable under translation). - const top_of_ram = @as(u64, pmm.stats().total_frames) * page_size; - _ = map(domain, 0, top_of_ram); - var i: usize = 0; - while (i < info.rmrr_count) : (i += 1) { - const region = info.rmrr[i]; - _ = map(domain, region.base, region.limit - region.base + 1); - } -} - -/// Per-claimed-device record, so a driver's death detaches exactly the devices it held. -const Confined = struct { active: bool = false, owner: u32 = 0, bdf: u16 = 0 }; +/// Per-claimed-device record: its private domain, so a driver's death tears down +/// exactly the domains it held. +const Confined = struct { active: bool = false, owner: u32 = 0, bdf: u16 = 0, domain: u16 = invalid_domain }; var confined: [maximum_domains]Confined = .{Confined{}} ** maximum_domains; -/// Place a just-claimed PCI function under IOMMU translation on behalf of `owner`. L1: -/// attach it to the blanket identity domain (its DMA works, but through real second- -/// level walks). false only if the machinery is unexpectedly unavailable — the caller -/// rolls the claim back. No-op success when no IOMMU exists (fail-open). +/// The interim DMA pool: the set of `dma_alloc`'d regions. Every such region is mapped +/// into every claimed device's domain, so a device reaches DMA buffers (including one +/// another's — the honest limit of this stage) but NOT the kernel, page tables, process +/// heaps, or arbitrary RAM. The capability layer (L3/L4) narrows this to per-grant. +const PoolRegion = struct { active: bool = false, physical: u64 = 0, len: u64 = 0 }; +var pool: [maximum_pool]PoolRegion = .{PoolRegion{}} ** maximum_pool; +const maximum_pool = 256; + +/// Place a just-claimed PCI function under IOMMU translation on behalf of `owner`: give +/// it a private empty domain, seed it with the current DMA pool and the device's own +/// firmware reserved region, and attach. false only if a domain can't be allocated — +/// the caller rolls the claim back (a claim that can't be confined must not stand). No-op +/// success when no IOMMU exists (fail-open). pub fn confineDevice(device_id: u64, bdf: u16, owner: u32) bool { if (kind == .none) return true; - if (blanket == invalid_domain) return false; if (device_id >= confined.len) return true; // unusual id; leave it to fail-open - attachDevice(blanket, bdf); - confined[@intCast(device_id)] = .{ .active = true, .owner = owner, .bdf = bdf }; + const domain = domainCreate(owner, bdf) orelse return false; + + // Seed with every pooled DMA region so the device's own rings/buffers and the + // cross-process buffers handed to it (fat's bounce buffer via usb-storage) resolve. + for (&pool) |*r| { + if (r.active) _ = map(domain, r.physical, r.len); + } + // Firmware reserved region for this device, if any (real hardware; QEMU has none). + const info = platform.platformInformation(); + var i: usize = 0; + while (i < info.rmrr_count) : (i += 1) { + if (info.rmrr[i].bdf == bdf) + _ = map(domain, info.rmrr[i].base, info.rmrr[i].limit - info.rmrr[i].base + 1); + } + + attachDevice(domain, bdf); + confined[@intCast(device_id)] = .{ .active = true, .owner = owner, .bdf = bdf, .domain = domain }; return true; } -/// A driver died or released its devices: detach every device it held so their DMA is -/// blocked again (a restarted driver re-claims and re-confines). Runs BEFORE the frames -/// and broker claims are released. +/// Record a newly `dma_alloc`'d region and map it into every claimed device's domain +/// (the interim pool rule). Called from the dma_alloc syscall. +pub fn poolAdd(physical: u64, len: u64) void { + if (kind == .none) return; + for (&pool) |*r| { + if (!r.active) { + r.* = .{ .active = true, .physical = physical, .len = len }; + break; + } + } else return; // pool full; region stays unmapped and its device DMA will fault + for (&confined) |*c| { + if (c.active) _ = map(c.domain, physical, len); + } +} + +/// Unmap a freed DMA region from every claimed device's domain and forget it. MUST run +/// before the frames return to pmm — a device translating to a reallocated frame is the +/// use-after-free this prevents. Called from the dma_free syscall. +pub fn poolRemove(physical: u64, len: u64) void { + if (kind == .none) return; + for (&pool) |*r| { + if (r.active and r.physical == physical and r.len == len) { + for (&confined) |*c| { + if (c.active) unmap(c.domain, physical, len); + } + r.* = .{}; + return; + } + } +} + +/// A driver died or released its devices: tear down every domain it held (detach the +/// device, free the tables) so their DMA is blocked again and a restarted driver +/// re-claims cleanly. Runs BEFORE the broker claims and the DMA frames are released. pub fn releaseAllOwnedBy(owner: u32) void { if (kind == .none) return; for (&confined) |*c| { if (c.active and c.owner == owner) { detachDevice(c.bdf); + domainDestroy(c.domain); c.* = .{}; } } diff --git a/system/kernel/process.zig b/system/kernel/process.zig index f434ff5..1bc0e5e 100644 --- a/system/kernel/process.zig +++ b/system/kernel/process.zig @@ -251,6 +251,7 @@ fn system_call(state: *architecture.CpuState) void { .fs_node => systemFsNode(state), .fs_mount => systemFsMount(state), .fs_unmount => systemFsUnmount(state), + .iommu_fault_drain => systemIommuFaultDrain(state), .wall_clock => systemWallClock(state), .shared_memory_create => systemSharedMemoryCreate(state), .shared_memory_map => systemSharedMemoryMap(state), @@ -509,6 +510,17 @@ fn systemIoWrite(state: *architecture.CpuState) void { /// their physical address is never disclosed. `dma_below_4g` caps the physical address /// for legacy engines; `dma_write_combining` is accepted but falls back to coherent /// until PAT is programmed. See docs/driver-model.md (M14). +/// iommu_fault_drain() -> count: drain and log any pending IOMMU translation faults, +/// returning how many were seen. A diagnostic hook — a driver (or a test) that suspects +/// its device faulted can force the fault records to be logged now rather than waiting +/// for the next device-release drain. Harmless without an IOMMU (returns 0). +fn systemIommuFaultDrain(state: *architecture.CpuState) void { + const flags = sync.enter(); + const count = iommu.faultDrain(); + sync.leave(flags); + architecture.setSystemCallResult(state, count); +} + fn systemDmaAlloc(state: *architecture.CpuState) void { const len = architecture.systemCallArg(state, 0); const flags = architecture.systemCallArg(state, 1); @@ -555,6 +567,13 @@ fn systemDmaAlloc(state: *architecture.CpuState) void { architecture.mapUserDmaInto(t.address_space, base_v + i * page_size, phys + i * page_size, page_size); sync.leave(lock_flags); } + // Publish the region to the DMA pool: it becomes reachable to every claimed device + // (the interim rule until DMA-region capabilities land). No-op without an IOMMU. + { + const lock_flags = sync.enter(); + iommu.poolAdd(phys, pages * page_size); + sync.leave(lock_flags); + } architecture.setSystemCallResult(state, base_v); // virtual address for the CPU architecture.setSystemCallResult2(state, phys); // physical address for the device } @@ -572,6 +591,17 @@ fn systemDmaFree(state: *architecture.CpuState) void { const pages: usize = @intCast((len + page_size - 1) / page_size); if (base_v < dma_arena_base or base_v + pages * page_size > dma_arena_end) return fail(state); + // Pull the region out of every device's domain and invalidate BEFORE any frame + // returns to the allocator — a device still translating to a reallocated frame is a + // use-after-free. dma_alloc's frames are contiguous, so the base translation names + // the whole region. + { + const lock_flags = sync.enter(); + if (architecture.translate(t.address_space, base_v)) |base_phys| + iommu.poolRemove(base_phys, pages * page_size); + sync.leave(lock_flags); + } + for (0..pages) |i| { const va = base_v + i * page_size; // Per-page lock hold: the translate/unmap walks the shared page tables diff --git a/system/kernel/tests.zig b/system/kernel/tests.zig index 1073f64..eeea730 100644 --- a/system/kernel/tests.zig +++ b/system/kernel/tests.zig @@ -215,6 +215,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void { pciScanTest(boot_information); } else if (eql(case, "pci-caps")) { pciCapsTest(boot_information); + } else if (eql(case, "iommu-fault")) { + iommuFaultTest(boot_information); } else if (eql(case, "acpi-parse")) { acpiParseTest(boot_information); } else if (eql(case, "acpi-report")) { @@ -2453,6 +2455,39 @@ fn pciCapsTest(boot_information: *const BootInformation) void { result(); } +/// IOMMU enforcement, the negative proof: boot with VT-d on and an unclaimed e1000e. +/// The manager spawns pci-bus, the fixture claims the NIC and fires a DMA at an +/// unmapped page; the unit must fault it and the system survive. Substance is asserted +/// by the harness on the kernel's DANOS-IOMMU-FAULT line and the fixture's markers. +fn iommuFaultTest(boot_information: *const BootInformation) void { + log("DANOS-TEST-BEGIN: iommu-fault\n", .{}); + if (boot_information.initial_ramdisk_len == 0) { + check("bootloader handed over an initial_ramdisk", false); + result(); + return; + } + const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len]; + const rd = initial_ramdisk.Reader.init(image) orelse { + check("initial_ramdisk image is valid", false); + result(); + return; + }; + + check("IOMMU enabled for the enforcement test", iommu.enabled()); + process.setInitialRamdisk(image); + var manager: u32 = 0; + var i: u32 = 0; + while (i < rd.count) : (i += 1) { + const item = rd.entry(i) orelse continue; + if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue; + manager = process.spawnProcessSupervised(item.blob, 4, &.{"device-manager"}, scheduler.currentId(), null) catch 0; + break; + } + check("device-manager spawned", manager != 0); + check("iommu-fault-test spawned", spawnNamed(rd, "iommu-fault-test")); + result(); +} + /// M19.1: the ring-3 PCI scan agrees with the kernel's. The manager spawns /// pci-bus for the host bridge; the driver walks the same ECAM window through /// its mmio_map grant and must find exactly the functions the kernel's own diff --git a/test/qemu_test.py b/test/qemu_test.py index 34ee48c..d73d151 100644 --- a/test/qemu_test.py +++ b/test/qemu_test.py @@ -184,6 +184,15 @@ CASES = [ "qemu_extra": ["-device", "intel-iommu,intremap=off"], "expect": r"(?s)(?=.*/system/kernel: iommu online)(?=.*usb-hid-keyboard: ok)(?=.*usb-hid-mouse: ok)", "fail": r"DANOS-TEST-RESULT: FAIL|DANOS-IOMMU-FAULT"}, + # Enforcement, the negative proof: a claimed e1000e fires a DMA at an unmapped page; + # VT-d must fault it (logged) and the system must stay alive. The fault line is the + # point here, so unlike the positive cases it appears in `expect`, not `fail`. + {"name": "iommu-fault", + "smp": 4, + "timeout": 120, + "qemu_extra": ["-device", "intel-iommu,intremap=off", "-device", "e1000e"], + "expect": r"(?s)(?=.*DANOS-IOMMU-FAULT: bdf=)(?=.*iommu-fault-test: system alive)(?=.*DANOS-TEST-RESULT: PASS)", + "fail": r"iommu-fault-test: FAIL|DANOS-TEST-RESULT: FAIL|CPU EXCEPTION|KERNEL PANIC"}, # Port I/O grants: a claimed device's io_port resource lets a driver read/write its # ports (PS/2 status 0x64), gated by the claim; out-of-range/unclaimed is refused. {"name": "ioport", diff --git a/test/system/services/iommu-fault-test/iommu-fault-test.zig b/test/system/services/iommu-fault-test/iommu-fault-test.zig new file mode 100644 index 0000000..44b183e --- /dev/null +++ b/test/system/services/iommu-fault-test/iommu-fault-test.zig @@ -0,0 +1,119 @@ +//! iommu-fault-test — the negative proof for IOMMU enforcement. The iommu-fault QEMU +//! case boots with VT-d enabled and an extra e1000e NIC no danos driver claims; this +//! fixture claims it, then deliberately programs its transmit engine to DMA from a +//! physical address that was never dma_alloc'd (so it is in no device's domain). The +//! IOMMU must fault that access — the descriptor fetch never reaches memory — and the +//! system must stay alive. A kernel `DANOS-IOMMU-FAULT` line plus this fixture's +//! `system alive` marker is the pass. +//! +//! The rogue target is the e1000e's transmit descriptor RING base itself: the very first +//! DMA the engine issues on a doorbell write is the descriptor fetch from that base, so +//! pointing the ring at an unmapped page makes the first access the faulting one — no +//! valid descriptor need be crafted. + +const std = @import("std"); +const device = @import("driver"); +const time = @import("time"); +const logging = @import("logging"); +const mmio = @import("mmio"); +const pci = @import("pci"); +const pci_class = @import("pci-class"); + +const intel_vendor: u16 = 0x8086; +const e1000e_device: u16 = 0x10D3; + +const ethernet_class: u64 = pci_class.ClassCode.pack(.{ + .base = @intFromEnum(pci_class.BaseClass.network), + .subclass = @intFromEnum(pci_class.network.SubClass.ethernet), + .prog_if = 0, +}); + +// e1000e transmit-engine registers (Intel 82574L datasheet §Register Descriptions), +// byte offsets within BAR0. VERIFY-AGAINST-SPEC held on first bring-up: these are the +// legacy TX ring registers. +const reg_tctl = 0x0400; // Transmit Control +const reg_tdbal = 0x3800; // TX Descriptor Base Address Low +const reg_tdbah = 0x3804; // TX Descriptor Base Address High +const reg_tdlen = 0x3808; // TX Descriptor Length (bytes, 128-byte aligned) +const reg_tdh = 0x3810; // TX Descriptor Head +const reg_tdt = 0x3818; // TX Descriptor Tail + +const tctl_en: u32 = 1 << 1; // Transmit Enable +const tctl_psp: u32 = 1 << 3; // Pad Short Packets + +/// A low physical page that user DMA never touches — never returned by dma_alloc (whose +/// arena is far higher), so it is in no device's IOMMU domain. The e1000e's descriptor +/// fetch from here is exactly the out-of-domain access the unit must block. +const rogue_physical: u64 = 0x1000; + +var descriptor: device.DeviceDescriptor = undefined; + +fn writeLine(comptime fmt: []const u8, arguments: anytype) void { + var line: [128]u8 = undefined; + _ = logging.write(std.fmt.bufPrint(&line, fmt, arguments) catch return); +} + +pub fn main() void { + const nic_id: u64 = found: { + var tries: u32 = 0; + while (tries < 150) : (tries += 1) { + var descriptors: [64]device.DeviceDescriptor = undefined; + const total = device.enumerate(&descriptors); + for (descriptors[0..@min(total, descriptors.len)]) |*entry| { + if (entry.class == @intFromEnum(device.DeviceClass.pci_device) and entry.pci_class == ethernet_class) { + descriptor = entry.*; + break :found entry.id; + } + } + time.sleepMillis(100); + } + _ = logging.write("iommu-fault-test: FAIL no ethernet function found\n"); + return; + }; + + if (!device.claim(nic_id)) { + _ = logging.write("iommu-fault-test: FAIL claim\n"); + return; + } + var function = pci.Function.map(nic_id, &descriptor) orelse { + _ = logging.write("iommu-fault-test: FAIL config-space map\n"); + return; + }; + if (function.vendorId() != intel_vendor or function.deviceId() != e1000e_device) { + _ = logging.write("iommu-fault-test: FAIL not an e1000e\n"); + return; + } + function.enableMemoryAndBusMaster(); + const bar0 = function.mapBar(0) orelse { + _ = logging.write("iommu-fault-test: FAIL map BAR0\n"); + return; + }; + + // Point the TX ring at the rogue page and kick the engine: TDBA = rogue, a non-zero + // length, head=0, enable, then tail=1 so the engine fetches descriptor 0 — a DMA + // read from the rogue page, which the IOMMU must fault. + _ = logging.write("iommu-fault-test: pointing e1000e TX ring at an unmapped page\n"); + mmio.writeRegister(u32, bar0 + reg_tdbal, @truncate(rogue_physical)); + mmio.writeRegister(u32, bar0 + reg_tdbah, @intCast(rogue_physical >> 32)); + mmio.writeRegister(u32, bar0 + reg_tdlen, 128); + mmio.writeRegister(u32, bar0 + reg_tdh, 0); + mmio.writeRegister(u32, bar0 + reg_tctl, tctl_en | tctl_psp); + mmio.writeMemoryBarrier(); + mmio.writeRegister(u32, bar0 + reg_tdt, 1); // doorbell: fetch descriptor 0 + + // Give the engine time to attempt the fetch, then force the fault records to the log. + time.sleepMillis(200); + const faults = device.iommuFaultDrain(); + writeLine("iommu-fault-test: drained {d} iommu fault(s)\n", .{faults}); + if (faults == 0) { + _ = logging.write("iommu-fault-test: FAIL rogue DMA was not blocked\n"); + return; + } + + // Liveness: the system survived the blocked DMA — read our own config space back. + if (function.vendorId() != intel_vendor) { + _ = logging.write("iommu-fault-test: FAIL device unreadable after fault\n"); + return; + } + _ = logging.write("iommu-fault-test: system alive\n"); +}