Replaces L1's shared blanket identity domain with a private translation
domain per claimed PCI function. A device now reaches only:
- the DMA pool: every dma_alloc'd region, mapped into every claimed
device's domain (poolAdd/poolRemove, driven from the dma_alloc and
dma_free syscalls). This keeps the cross-process buffer handoff
working (fat's bounce buffer reaches the xHC) while blocking the
kernel, page tables, process heaps, MMIO, and unallocated RAM.
- its own firmware reserved region (RMRR), seeded at confine time.
The pool is the honest interim: devices can still reach one another's
DMA buffers. The DMA-region capability layer (next) narrows it to
per-grant reachability.
dma_free unmaps from every domain and invalidates BEFORE the frames
return to the allocator, closing the stale-IOTLB use-after-free window.
Driver death tears down its domains (detach + free tables) before the
broker claims and DMA frames are released.
New iommu_fault_drain syscall (+ driver.iommuFaultDrain) forces pending
fault records to the log on demand. The new iommu-fault case proves it:
a claimed e1000e is programmed to DMA-fetch its TX ring from an unmapped
page; VT-d faults the access (bdf 00:03.0 addr 0x1000 reason 0x6) and the
system stays alive. 104/104.
433 lines
18 KiB
Zig
433 lines
18 KiB
Zig
//! system/kernel/iommu.zig — vendor-neutral IOMMU core: per-device DMA translation
|
|
//! domains over an Intel VT-d or AMD-Vi backend.
|
|
//!
|
|
//! The problem this closes: without an IOMMU, a claimed bus-mastering device can DMA to
|
|
//! ANY physical address, so a compromised or buggy driver reaches all of memory through
|
|
//! its device — driver isolation stops at the CPU's MMU. This core gives each claimed
|
|
//! PCI function its own translation domain; a device reaches only the physical ranges
|
|
//! mapped into its domain, and nothing else (kernel, page tables, other processes) is
|
|
//! visible to it.
|
|
//!
|
|
//! Design:
|
|
//! - **Identity mappings** (IOVA == physical). `dma_alloc` already hands drivers the
|
|
//! physical address they program into hardware; a domain simply makes that same
|
|
//! address the ONLY thing the device can reach. No IOVA allocator, and every
|
|
//! driver's register-programming code is untouched.
|
|
//! - **Vendor-neutral**: this file owns the domain table and a shared 512-entry
|
|
//! page-table walker; a `Backend` vtable supplies the hardware specifics (VT-d in
|
|
//! iommu-intel.zig, AMD-Vi in iommu-amd.zig) — the entry-bit encodings, the
|
|
//! enable/invalidate register dances, and the fault drain.
|
|
//! - **Fail-open**: when no IOMMU is found, `kind == .none` and every entry point is a
|
|
//! success no-op, so callers in process.zig stay unconditional and behavior is
|
|
//! byte-for-byte the pre-IOMMU kernel. The boot log states the posture.
|
|
//!
|
|
//! All entry points run under the big kernel lock (the caller holds it); no internal
|
|
//! locking. All memory comes from `pmm` reached through the physmap, like paging.zig.
|
|
|
|
const std = @import("std");
|
|
const abi = @import("abi");
|
|
const boot_handoff = @import("boot-handoff");
|
|
const pmm = @import("pmm.zig");
|
|
const platform = @import("platform");
|
|
const devices_broker = @import("devices-broker.zig");
|
|
const log = @import("log.zig");
|
|
const intel = @import("iommu-intel.zig");
|
|
|
|
const page_size: u64 = abi.page_size;
|
|
const page_mask: u64 = page_size - 1;
|
|
const huge_page_size: u64 = 2 * 1024 * 1024;
|
|
|
|
pub const Kind = enum { none, intel_vtd, amd_vi };
|
|
|
|
/// One domain per claimed PCI function. 64 mirrors devices-broker's device cap.
|
|
pub const maximum_domains = 64;
|
|
pub const invalid_domain: u16 = 0xFFFF;
|
|
|
|
/// The bit encodings and hardware operations a backend supplies to the shared core.
|
|
/// Entry helpers build the raw page-table entries for the backend's format; the core
|
|
/// walks the tree with them. The hardware ops act on a whole domain (identified by its
|
|
/// hardware domain id = core index + 1) or device (by requester id / bdf).
|
|
pub const Backend = struct {
|
|
/// Number of page-table levels (3 or 4) the backend selected from hardware caps.
|
|
levels: u8,
|
|
/// Largest leaf the walker may emit: 4 KiB always, 2 MiB when the backend allows.
|
|
supports_huge_pages: bool,
|
|
|
|
/// Raw entry bits for a leaf mapping `physical` (with the given size), and for a
|
|
/// non-leaf entry pointing at `table_physical` at `level` (level counts down to 1
|
|
/// at the leaf's parent). `isPresent` tests a read-back entry.
|
|
makeLeaf: *const fn (physical: u64, huge: bool) u64,
|
|
makeTable: *const fn (table_physical: u64, level: u8) u64,
|
|
isPresent: *const fn (entry: u64) bool,
|
|
/// Flush a cache line holding IOMMU structures the hardware reads non-coherently
|
|
/// (VT-d with ECAP.C==0). A no-op where the unit snoops caches.
|
|
flushStructure: *const fn (address: usize) void,
|
|
|
|
/// Point `bdf`'s translation structure at `domain` (hardware id) and invalidate the
|
|
/// context/device caches so the change takes effect.
|
|
attach: *const fn (bdf: u16, domain: u16, page_table_root: u64) void,
|
|
/// Return `bdf`'s translation structure to not-present + invalidate — all its DMA
|
|
/// faults afterward.
|
|
detach: *const fn (bdf: u16) void,
|
|
/// Invalidate cached translations for `domain` (after a map or unmap).
|
|
invalidateDomain: *const fn (domain: u16) void,
|
|
/// Pull pending faults out of the hardware, log them (rate-limited), return the
|
|
/// count seen this call.
|
|
faultDrain: *const fn () usize,
|
|
};
|
|
|
|
const Domain = struct {
|
|
in_use: bool = false,
|
|
owner: u32 = 0, // task that owns the attached device
|
|
bdf: u16 = 0, // requester id of the attached device
|
|
page_table_root: u64 = 0, // physical address of the top-level table
|
|
rmrr: bool = false, // a firmware reserved-region domain (persists across claims)
|
|
};
|
|
|
|
var kind: Kind = .none;
|
|
var backend: Backend = undefined;
|
|
var domains: [maximum_domains]Domain = .{Domain{}} ** maximum_domains;
|
|
|
|
pub fn kindOf() Kind {
|
|
return kind;
|
|
}
|
|
pub fn enabled() bool {
|
|
return kind != .none;
|
|
}
|
|
|
|
/// Detect the IOMMU, pick a backend, pre-map firmware reserved regions, and enable
|
|
/// translation. Fail-open (kind stays .none) when no unit exists — the caller logs the
|
|
/// posture. Must run after platform discovery and before any user process starts.
|
|
pub fn init() void {
|
|
const info = platform.platformInformation();
|
|
if (!info.iommu_present) {
|
|
kind = .none;
|
|
return;
|
|
}
|
|
// Only Intel VT-d for now; AMD-Vi (iommu-amd.zig) selects here when its detection
|
|
// (IVRS) lands. A present-but-unsupported unit stays fail-open with a logged reason.
|
|
if (intel.detect(info)) |be| {
|
|
backend = be;
|
|
kind = .intel_vtd;
|
|
} else {
|
|
kind = .none;
|
|
log.write("/system/kernel: WARNING IOMMU present but unusable — staying fail-open\n");
|
|
return;
|
|
}
|
|
|
|
// The root table starts empty: every device's context entry is not-present, so any
|
|
// DMA faults until the device's driver claims it (confineDevice gives it a private
|
|
// domain). PCI functions are enumerated post-boot by the ring-3 pci-bus driver, so
|
|
// there is nothing to attach at init anyway.
|
|
intel.enable();
|
|
logEnabled(info);
|
|
}
|
|
|
|
/// Per-claimed-device record: its private domain, so a driver's death tears down
|
|
/// exactly the domains it held.
|
|
const Confined = struct { active: bool = false, owner: u32 = 0, bdf: u16 = 0, domain: u16 = invalid_domain };
|
|
var confined: [maximum_domains]Confined = .{Confined{}} ** maximum_domains;
|
|
|
|
/// The interim DMA pool: the set of `dma_alloc`'d regions. Every such region is mapped
|
|
/// into every claimed device's domain, so a device reaches DMA buffers (including one
|
|
/// another's — the honest limit of this stage) but NOT the kernel, page tables, process
|
|
/// heaps, or arbitrary RAM. The capability layer (L3/L4) narrows this to per-grant.
|
|
const PoolRegion = struct { active: bool = false, physical: u64 = 0, len: u64 = 0 };
|
|
var pool: [maximum_pool]PoolRegion = .{PoolRegion{}} ** maximum_pool;
|
|
const maximum_pool = 256;
|
|
|
|
/// Place a just-claimed PCI function under IOMMU translation on behalf of `owner`: give
|
|
/// it a private empty domain, seed it with the current DMA pool and the device's own
|
|
/// firmware reserved region, and attach. false only if a domain can't be allocated —
|
|
/// the caller rolls the claim back (a claim that can't be confined must not stand). No-op
|
|
/// success when no IOMMU exists (fail-open).
|
|
pub fn confineDevice(device_id: u64, bdf: u16, owner: u32) bool {
|
|
if (kind == .none) return true;
|
|
if (device_id >= confined.len) return true; // unusual id; leave it to fail-open
|
|
const domain = domainCreate(owner, bdf) orelse return false;
|
|
|
|
// Seed with every pooled DMA region so the device's own rings/buffers and the
|
|
// cross-process buffers handed to it (fat's bounce buffer via usb-storage) resolve.
|
|
for (&pool) |*r| {
|
|
if (r.active) _ = map(domain, r.physical, r.len);
|
|
}
|
|
// Firmware reserved region for this device, if any (real hardware; QEMU has none).
|
|
const info = platform.platformInformation();
|
|
var i: usize = 0;
|
|
while (i < info.rmrr_count) : (i += 1) {
|
|
if (info.rmrr[i].bdf == bdf)
|
|
_ = map(domain, info.rmrr[i].base, info.rmrr[i].limit - info.rmrr[i].base + 1);
|
|
}
|
|
|
|
attachDevice(domain, bdf);
|
|
confined[@intCast(device_id)] = .{ .active = true, .owner = owner, .bdf = bdf, .domain = domain };
|
|
return true;
|
|
}
|
|
|
|
/// Record a newly `dma_alloc`'d region and map it into every claimed device's domain
|
|
/// (the interim pool rule). Called from the dma_alloc syscall.
|
|
pub fn poolAdd(physical: u64, len: u64) void {
|
|
if (kind == .none) return;
|
|
for (&pool) |*r| {
|
|
if (!r.active) {
|
|
r.* = .{ .active = true, .physical = physical, .len = len };
|
|
break;
|
|
}
|
|
} else return; // pool full; region stays unmapped and its device DMA will fault
|
|
for (&confined) |*c| {
|
|
if (c.active) _ = map(c.domain, physical, len);
|
|
}
|
|
}
|
|
|
|
/// Unmap a freed DMA region from every claimed device's domain and forget it. MUST run
|
|
/// before the frames return to pmm — a device translating to a reallocated frame is the
|
|
/// use-after-free this prevents. Called from the dma_free syscall.
|
|
pub fn poolRemove(physical: u64, len: u64) void {
|
|
if (kind == .none) return;
|
|
for (&pool) |*r| {
|
|
if (r.active and r.physical == physical and r.len == len) {
|
|
for (&confined) |*c| {
|
|
if (c.active) unmap(c.domain, physical, len);
|
|
}
|
|
r.* = .{};
|
|
return;
|
|
}
|
|
}
|
|
}
|
|
|
|
/// A driver died or released its devices: tear down every domain it held (detach the
|
|
/// device, free the tables) so their DMA is blocked again and a restarted driver
|
|
/// re-claims cleanly. Runs BEFORE the broker claims and the DMA frames are released.
|
|
pub fn releaseAllOwnedBy(owner: u32) void {
|
|
if (kind == .none) return;
|
|
for (&confined) |*c| {
|
|
if (c.active and c.owner == owner) {
|
|
detachDevice(c.bdf);
|
|
domainDestroy(c.domain);
|
|
c.* = .{};
|
|
}
|
|
}
|
|
_ = faultDrain(); // log any faults a mid-DMA device raised as it was cut off
|
|
}
|
|
|
|
/// Allocate an empty domain (an empty top-level table). null when the table is full.
|
|
pub fn domainCreate(owner: u32, bdf: u16) ?u16 {
|
|
if (kind == .none) return 0; // fail-open: a dummy id the no-op ops ignore
|
|
for (&domains, 0..) |*d, index| {
|
|
if (d.in_use) continue;
|
|
const root = allocTable() orelse return null;
|
|
d.* = .{ .in_use = true, .owner = owner, .bdf = bdf, .page_table_root = root };
|
|
return @intCast(index);
|
|
}
|
|
return null;
|
|
}
|
|
|
|
/// Free a domain's page-table frames and its slot. Precondition: no device attached
|
|
/// (detach first).
|
|
pub fn domainDestroy(domain: u16) void {
|
|
if (kind == .none) return;
|
|
const d = &domains[domain];
|
|
if (!d.in_use) return;
|
|
freeTables(d.page_table_root, backend.levels);
|
|
d.* = .{};
|
|
}
|
|
|
|
/// Attach `bdf`'s device to `domain` and pre-load any RMRR range recorded for it.
|
|
pub fn attachDevice(domain: u16, bdf: u16) void {
|
|
if (kind == .none) return;
|
|
const d = &domains[domain];
|
|
d.bdf = bdf;
|
|
backend.attach(bdf, hardwareId(domain), d.page_table_root);
|
|
}
|
|
|
|
/// Return `bdf`'s device to not-present + invalidate.
|
|
pub fn detachDevice(bdf: u16) void {
|
|
if (kind == .none) return;
|
|
backend.detach(bdf);
|
|
}
|
|
|
|
/// Identity-map [physical, physical+len) into `domain` (read+write) and invalidate.
|
|
/// Unconditional domain-selective invalidation after every map — correct under VT-d
|
|
/// caching-mode and free otherwise.
|
|
pub fn map(domain: u16, physical: u64, len: u64) bool {
|
|
if (kind == .none) return true;
|
|
const d = &domains[domain];
|
|
if (!d.in_use) return false;
|
|
if (!mapRange(d.page_table_root, physical, len)) return false;
|
|
backend.invalidateDomain(hardwareId(domain));
|
|
return true;
|
|
}
|
|
|
|
/// Unmap [physical, physical+len) from `domain` and invalidate. MUST finish its
|
|
/// invalidation before the caller returns the frames to pmm — a stale IOTLB entry
|
|
/// pointing at a reallocated frame is the use-after-free this ordering prevents.
|
|
pub fn unmap(domain: u16, physical: u64, len: u64) void {
|
|
if (kind == .none) return;
|
|
const d = &domains[domain];
|
|
if (!d.in_use) return;
|
|
unmapRange(d.page_table_root, physical, len);
|
|
backend.invalidateDomain(hardwareId(domain));
|
|
}
|
|
|
|
/// Poll the hardware for translation faults, log them, return the count. Called by the
|
|
/// IOMMU test case and opportunistically after a device detaches.
|
|
pub fn faultDrain() usize {
|
|
if (kind == .none) return 0;
|
|
return backend.faultDrain();
|
|
}
|
|
|
|
/// The physical address `virtual` maps to in `domain`, or null if unmapped — a test
|
|
/// helper that walks the domain's page tables (identity mappings return `virtual`).
|
|
pub fn translationOf(domain: u16, virtual: u64) ?u64 {
|
|
if (kind == .none) return virtual;
|
|
const d = &domains[domain];
|
|
if (!d.in_use) return null;
|
|
var table = d.page_table_root;
|
|
var level = backend.levels;
|
|
while (level > 1) : (level -= 1) {
|
|
const entry = tableAt(table)[indexAt(virtual, level)];
|
|
if (!backend.isPresent(entry)) return null;
|
|
if (level == 2 and isHugeLeaf(entry))
|
|
return (entry & address_mask) | (virtual & (huge_page_size - 1));
|
|
table = entry & address_mask;
|
|
}
|
|
const leaf = tableAt(table)[indexAt(virtual, 1)];
|
|
if (!backend.isPresent(leaf)) return null;
|
|
return (leaf & address_mask) | (virtual & page_mask);
|
|
}
|
|
|
|
// --- the shared page-table walker -----------------------------------------------------
|
|
// 512-entry, 9-bits-per-level, 4 KiB tables reached through the physmap — the shape both
|
|
// VT-d second-level and AMD-Vi native tables share. The backend supplies the entry bits.
|
|
|
|
fn tableAt(physical: u64) [*]volatile u64 {
|
|
return @ptrFromInt(boot_handoff.physicalToVirtual(physical));
|
|
}
|
|
|
|
fn allocTable() ?u64 {
|
|
const frame = pmm.alloc() orelse return null;
|
|
const table = tableAt(frame);
|
|
var i: usize = 0;
|
|
while (i < 512) : (i += 1) table[i] = 0;
|
|
return frame;
|
|
}
|
|
|
|
const address_mask: u64 = 0x000F_FFFF_FFFF_F000;
|
|
|
|
fn indexAt(virtual: u64, level: u8) usize {
|
|
// level 1 is the leaf table; shift = 12 + 9*(level-1).
|
|
const shift: u6 = @intCast(12 + 9 * (@as(u32, level) - 1));
|
|
return @intCast((virtual >> shift) & 0x1FF);
|
|
}
|
|
|
|
/// Descend to (allocating) the next-level table below `entry_ptr`, returning its
|
|
/// physical base. null on out-of-memory.
|
|
fn descend(entry_ptr: *volatile u64, level: u8) ?u64 {
|
|
const entry = entry_ptr.*;
|
|
if (backend.isPresent(entry)) return entry & address_mask;
|
|
const table = allocTable() orelse return null;
|
|
backend.flushStructure(@intFromPtr(tableAt(table)));
|
|
entry_ptr.* = backend.makeTable(table, level);
|
|
backend.flushStructure(@intFromPtr(entry_ptr));
|
|
return table;
|
|
}
|
|
|
|
fn mapRange(root: u64, physical: u64, len: u64) bool {
|
|
const start = physical & ~page_mask;
|
|
const end = (physical + len + page_mask) & ~page_mask;
|
|
var addr = start;
|
|
while (addr < end) {
|
|
// 2 MiB leaf when the backend allows it and both address and remaining span are
|
|
// huge-aligned — keeps table memory sane for the blanket-identity and real-PC
|
|
// cases without a separate superpage path per backend.
|
|
const huge = backend.supports_huge_pages and
|
|
addr % huge_page_size == 0 and (end - addr) >= huge_page_size;
|
|
if (!mapOne(root, addr, huge)) return false;
|
|
addr += if (huge) huge_page_size else page_size;
|
|
}
|
|
return true;
|
|
}
|
|
|
|
fn mapOne(root: u64, addr: u64, huge: bool) bool {
|
|
const leaf_level: u8 = if (huge) 2 else 1;
|
|
var table = root;
|
|
var level = backend.levels;
|
|
while (level > leaf_level) : (level -= 1) {
|
|
const entry_ptr = &tableAt(table)[indexAt(addr, level)];
|
|
table = descend(entry_ptr, level) orelse return false;
|
|
}
|
|
const leaf_ptr = &tableAt(table)[indexAt(addr, leaf_level)];
|
|
leaf_ptr.* = backend.makeLeaf(addr, huge);
|
|
backend.flushStructure(@intFromPtr(leaf_ptr));
|
|
return true;
|
|
}
|
|
|
|
fn unmapRange(root: u64, physical: u64, len: u64) void {
|
|
const start = physical & ~page_mask;
|
|
const end = (physical + len + page_mask) & ~page_mask;
|
|
var addr = start;
|
|
while (addr < end) {
|
|
const huge = backend.supports_huge_pages and
|
|
addr % huge_page_size == 0 and (end - addr) >= huge_page_size;
|
|
unmapOne(root, addr, huge);
|
|
addr += if (huge) huge_page_size else page_size;
|
|
}
|
|
}
|
|
|
|
fn unmapOne(root: u64, addr: u64, huge: bool) void {
|
|
const leaf_level: u8 = if (huge) 2 else 1;
|
|
var table = root;
|
|
var level = backend.levels;
|
|
while (level > leaf_level) : (level -= 1) {
|
|
const entry = tableAt(table)[indexAt(addr, level)];
|
|
if (!backend.isPresent(entry)) return; // nothing mapped here
|
|
table = entry & address_mask;
|
|
}
|
|
const leaf_ptr = &tableAt(table)[indexAt(addr, leaf_level)];
|
|
leaf_ptr.* = 0;
|
|
backend.flushStructure(@intFromPtr(leaf_ptr));
|
|
}
|
|
|
|
/// Post-order free of a domain's whole table tree.
|
|
fn freeTables(root: u64, level: u8) void {
|
|
if (level > 1) {
|
|
const table = tableAt(root);
|
|
var i: usize = 0;
|
|
while (i < 512) : (i += 1) {
|
|
const entry = table[i];
|
|
if (!backend.isPresent(entry)) continue;
|
|
// A 2 MiB leaf sits at level 2 and points at RAM, not a sub-table.
|
|
if (level == 2 and isHugeLeaf(entry)) continue;
|
|
freeTables(entry & address_mask, level - 1);
|
|
}
|
|
}
|
|
pmm.free(root);
|
|
}
|
|
|
|
fn isHugeLeaf(entry: u64) bool {
|
|
// Both backends set a page-size bit (VT-d bit 7, AMD leaf next-level=0 at level 2).
|
|
// The backend's makeLeaf encodes it; the walker only needs "is this a leaf, not a
|
|
// pointer" at level 2, which huge leaves are by construction.
|
|
return entry & huge_leaf_bit != 0;
|
|
}
|
|
|
|
/// The size-bit the backends set on a 2 MiB leaf (VT-d SL-PTE PS bit 7; AMD encodes a
|
|
/// leaf as next-level 0, so the core marks huge leaves with this software bit — an
|
|
/// ignored bit in both formats — to tell them apart from table pointers when freeing).
|
|
pub const huge_leaf_bit: u64 = 1 << 7;
|
|
|
|
fn hardwareId(domain: u16) u16 {
|
|
return domain + 1; // id 0 is reserved by both architectures
|
|
}
|
|
|
|
fn logEnabled(info: platform.PlatformInformation) void {
|
|
log.write("/system/kernel: iommu online (Intel VT-d)\n");
|
|
log.print(" version : 0x{x}\n", .{info.iommu_version});
|
|
log.print(" agaw : {d} levels\n", .{backend.levels});
|
|
log.print(" rmrr : {d} region(s) premapped\n", .{info.rmrr_count});
|
|
if (info.rmrr_skipped > 0)
|
|
log.print(" rmrr : WARNING {d} scope(s) skipped — a device keeps an unmapped firmware buffer\n", .{info.rmrr_skipped});
|
|
if (info.iommu_extra_units > 0)
|
|
log.print(" units : WARNING {d} other DRHD(s) — their scoped devices are NOT translated\n", .{info.iommu_extra_units});
|
|
}
|