iommu: enable Intel VT-d translation with per-claim device confinement
First enforcement step of the IOVA track. A vendor-neutral IOMMU core (iommu.zig) drives an Intel VT-d backend (iommu-intel.zig) to give DMA a real translation layer instead of the fail-open free-for-all M16 left. - Boot posture is now stated explicitly: "iommu online (Intel VT-d)" with version/agaw/rmrr, or "none present - DMA fail-open (unisolated)". - DMAR parsing extended to select the INCLUDE_PCI_ALL unit (real Intel PCs put an iGPU-scoped unit first) and record single-path-endpoint RMRRs; multi-hop scopes and extra DRHDs are counted and warned, never silently dropped. - Translation is enabled at boot into a blanket identity domain (all RAM + RMRRs, 2 MiB leaves). PCI functions are enumerated post-boot by the ring-3 pci-bus driver, so a device is attached to the domain when its driver claims it (confineDevice, with claim rollback if confinement fails) and detached on driver death, before broker release and DMA frame teardown. Unclaimed devices are non-present: their DMA faults. - Interrupt remapping stays off, so MSI writes to 0xFEE00000 bypass translation and the interrupt-driven xHC keeps working. - devices-broker gains pciAddressOf (derives BDF from the config-space ECAM offset), unclaim, and forEachPciFunction. Faults are drained and logged rate-limited as DANOS-IOMMU-FAULT. Cases: iommu extended (translation on, scratch-domain map/resolve/unmap, zero idle faults); new iommu-usb-storage and iommu-usb-hid run the full storage + input stacks through translated DMA with MSI intact. 103/103.
This commit is contained in:
@@ -0,0 +1,403 @@
|
||||
//! system/kernel/iommu.zig — vendor-neutral IOMMU core: per-device DMA translation
|
||||
//! domains over an Intel VT-d or AMD-Vi backend.
|
||||
//!
|
||||
//! The problem this closes: without an IOMMU, a claimed bus-mastering device can DMA to
|
||||
//! ANY physical address, so a compromised or buggy driver reaches all of memory through
|
||||
//! its device — driver isolation stops at the CPU's MMU. This core gives each claimed
|
||||
//! PCI function its own translation domain; a device reaches only the physical ranges
|
||||
//! mapped into its domain, and nothing else (kernel, page tables, other processes) is
|
||||
//! visible to it.
|
||||
//!
|
||||
//! Design:
|
||||
//! - **Identity mappings** (IOVA == physical). `dma_alloc` already hands drivers the
|
||||
//! physical address they program into hardware; a domain simply makes that same
|
||||
//! address the ONLY thing the device can reach. No IOVA allocator, and every
|
||||
//! driver's register-programming code is untouched.
|
||||
//! - **Vendor-neutral**: this file owns the domain table and a shared 512-entry
|
||||
//! page-table walker; a `Backend` vtable supplies the hardware specifics (VT-d in
|
||||
//! iommu-intel.zig, AMD-Vi in iommu-amd.zig) — the entry-bit encodings, the
|
||||
//! enable/invalidate register dances, and the fault drain.
|
||||
//! - **Fail-open**: when no IOMMU is found, `kind == .none` and every entry point is a
|
||||
//! success no-op, so callers in process.zig stay unconditional and behavior is
|
||||
//! byte-for-byte the pre-IOMMU kernel. The boot log states the posture.
|
||||
//!
|
||||
//! All entry points run under the big kernel lock (the caller holds it); no internal
|
||||
//! locking. All memory comes from `pmm` reached through the physmap, like paging.zig.
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const boot_handoff = @import("boot-handoff");
|
||||
const pmm = @import("pmm.zig");
|
||||
const platform = @import("platform");
|
||||
const devices_broker = @import("devices-broker.zig");
|
||||
const log = @import("log.zig");
|
||||
const intel = @import("iommu-intel.zig");
|
||||
|
||||
const page_size: u64 = abi.page_size;
|
||||
const page_mask: u64 = page_size - 1;
|
||||
const huge_page_size: u64 = 2 * 1024 * 1024;
|
||||
|
||||
pub const Kind = enum { none, intel_vtd, amd_vi };
|
||||
|
||||
/// One domain per claimed PCI function. 64 mirrors devices-broker's device cap.
|
||||
pub const maximum_domains = 64;
|
||||
pub const invalid_domain: u16 = 0xFFFF;
|
||||
|
||||
/// The bit encodings and hardware operations a backend supplies to the shared core.
|
||||
/// Entry helpers build the raw page-table entries for the backend's format; the core
|
||||
/// walks the tree with them. The hardware ops act on a whole domain (identified by its
|
||||
/// hardware domain id = core index + 1) or device (by requester id / bdf).
|
||||
pub const Backend = struct {
|
||||
/// Number of page-table levels (3 or 4) the backend selected from hardware caps.
|
||||
levels: u8,
|
||||
/// Largest leaf the walker may emit: 4 KiB always, 2 MiB when the backend allows.
|
||||
supports_huge_pages: bool,
|
||||
|
||||
/// Raw entry bits for a leaf mapping `physical` (with the given size), and for a
|
||||
/// non-leaf entry pointing at `table_physical` at `level` (level counts down to 1
|
||||
/// at the leaf's parent). `isPresent` tests a read-back entry.
|
||||
makeLeaf: *const fn (physical: u64, huge: bool) u64,
|
||||
makeTable: *const fn (table_physical: u64, level: u8) u64,
|
||||
isPresent: *const fn (entry: u64) bool,
|
||||
/// Flush a cache line holding IOMMU structures the hardware reads non-coherently
|
||||
/// (VT-d with ECAP.C==0). A no-op where the unit snoops caches.
|
||||
flushStructure: *const fn (address: usize) void,
|
||||
|
||||
/// Point `bdf`'s translation structure at `domain` (hardware id) and invalidate the
|
||||
/// context/device caches so the change takes effect.
|
||||
attach: *const fn (bdf: u16, domain: u16, page_table_root: u64) void,
|
||||
/// Return `bdf`'s translation structure to not-present + invalidate — all its DMA
|
||||
/// faults afterward.
|
||||
detach: *const fn (bdf: u16) void,
|
||||
/// Invalidate cached translations for `domain` (after a map or unmap).
|
||||
invalidateDomain: *const fn (domain: u16) void,
|
||||
/// Pull pending faults out of the hardware, log them (rate-limited), return the
|
||||
/// count seen this call.
|
||||
faultDrain: *const fn () usize,
|
||||
};
|
||||
|
||||
const Domain = struct {
|
||||
in_use: bool = false,
|
||||
owner: u32 = 0, // task that owns the attached device
|
||||
bdf: u16 = 0, // requester id of the attached device
|
||||
page_table_root: u64 = 0, // physical address of the top-level table
|
||||
rmrr: bool = false, // a firmware reserved-region domain (persists across claims)
|
||||
};
|
||||
|
||||
var kind: Kind = .none;
|
||||
var backend: Backend = undefined;
|
||||
var domains: [maximum_domains]Domain = .{Domain{}} ** maximum_domains;
|
||||
|
||||
pub fn kindOf() Kind {
|
||||
return kind;
|
||||
}
|
||||
pub fn enabled() bool {
|
||||
return kind != .none;
|
||||
}
|
||||
|
||||
/// Detect the IOMMU, pick a backend, pre-map firmware reserved regions, and enable
|
||||
/// translation. Fail-open (kind stays .none) when no unit exists — the caller logs the
|
||||
/// posture. Must run after platform discovery and before any user process starts.
|
||||
pub fn init() void {
|
||||
const info = platform.platformInformation();
|
||||
if (!info.iommu_present) {
|
||||
kind = .none;
|
||||
return;
|
||||
}
|
||||
// Only Intel VT-d for now; AMD-Vi (iommu-amd.zig) selects here when its detection
|
||||
// (IVRS) lands. A present-but-unsupported unit stays fail-open with a logged reason.
|
||||
if (intel.detect(info)) |be| {
|
||||
backend = be;
|
||||
kind = .intel_vtd;
|
||||
} else {
|
||||
kind = .none;
|
||||
log.write("/system/kernel: WARNING IOMMU present but unusable — staying fail-open\n");
|
||||
return;
|
||||
}
|
||||
|
||||
// L1 (enabled-but-invisible): every device is placed in a single blanket domain
|
||||
// that identity-maps all of RAM plus the firmware reserved regions, so translation
|
||||
// is on but nothing's DMA changes. L2 replaces this per device: on claim a device
|
||||
// is detached from the blanket and attached to its own empty domain, so it reaches
|
||||
// only what is explicitly mapped for it.
|
||||
buildBlanketDomain(info);
|
||||
intel.enable();
|
||||
logEnabled(info);
|
||||
}
|
||||
|
||||
/// The shared identity domain claimed devices are placed in (L1). L2 replaces this with
|
||||
/// a private empty domain per device plus explicitly-granted mappings.
|
||||
var blanket: u16 = invalid_domain;
|
||||
|
||||
/// PCI functions are enumerated post-boot by the ring-3 pci-bus driver, so at init the
|
||||
/// device tree has none — a device is attached to the blanket when its driver *claims*
|
||||
/// it (confineDevice), which is always before the driver programs any DMA. Until then
|
||||
/// its context entry is not-present and its DMA faults (only stale firmware bus-
|
||||
/// mastering would hit that, which is the evidence M16 exists to surface).
|
||||
fn buildBlanketDomain(info: platform.PlatformInformation) void {
|
||||
const domain = domainCreate(0, 0) orelse return;
|
||||
blanket = domain;
|
||||
|
||||
// Identity-map all of physical RAM (2 MiB leaves keep the table small even on a
|
||||
// 64 GiB machine), then each firmware reserved region in case it sits outside the
|
||||
// RAM extent (RMRRs must stay reachable under translation).
|
||||
const top_of_ram = @as(u64, pmm.stats().total_frames) * page_size;
|
||||
_ = map(domain, 0, top_of_ram);
|
||||
var i: usize = 0;
|
||||
while (i < info.rmrr_count) : (i += 1) {
|
||||
const region = info.rmrr[i];
|
||||
_ = map(domain, region.base, region.limit - region.base + 1);
|
||||
}
|
||||
}
|
||||
|
||||
/// Per-claimed-device record, so a driver's death detaches exactly the devices it held.
|
||||
const Confined = struct { active: bool = false, owner: u32 = 0, bdf: u16 = 0 };
|
||||
var confined: [maximum_domains]Confined = .{Confined{}} ** maximum_domains;
|
||||
|
||||
/// Place a just-claimed PCI function under IOMMU translation on behalf of `owner`. L1:
|
||||
/// attach it to the blanket identity domain (its DMA works, but through real second-
|
||||
/// level walks). false only if the machinery is unexpectedly unavailable — the caller
|
||||
/// rolls the claim back. No-op success when no IOMMU exists (fail-open).
|
||||
pub fn confineDevice(device_id: u64, bdf: u16, owner: u32) bool {
|
||||
if (kind == .none) return true;
|
||||
if (blanket == invalid_domain) return false;
|
||||
if (device_id >= confined.len) return true; // unusual id; leave it to fail-open
|
||||
attachDevice(blanket, bdf);
|
||||
confined[@intCast(device_id)] = .{ .active = true, .owner = owner, .bdf = bdf };
|
||||
return true;
|
||||
}
|
||||
|
||||
/// A driver died or released its devices: detach every device it held so their DMA is
|
||||
/// blocked again (a restarted driver re-claims and re-confines). Runs BEFORE the frames
|
||||
/// and broker claims are released.
|
||||
pub fn releaseAllOwnedBy(owner: u32) void {
|
||||
if (kind == .none) return;
|
||||
for (&confined) |*c| {
|
||||
if (c.active and c.owner == owner) {
|
||||
detachDevice(c.bdf);
|
||||
c.* = .{};
|
||||
}
|
||||
}
|
||||
_ = faultDrain(); // log any faults a mid-DMA device raised as it was cut off
|
||||
}
|
||||
|
||||
/// Allocate an empty domain (an empty top-level table). null when the table is full.
|
||||
pub fn domainCreate(owner: u32, bdf: u16) ?u16 {
|
||||
if (kind == .none) return 0; // fail-open: a dummy id the no-op ops ignore
|
||||
for (&domains, 0..) |*d, index| {
|
||||
if (d.in_use) continue;
|
||||
const root = allocTable() orelse return null;
|
||||
d.* = .{ .in_use = true, .owner = owner, .bdf = bdf, .page_table_root = root };
|
||||
return @intCast(index);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Free a domain's page-table frames and its slot. Precondition: no device attached
|
||||
/// (detach first).
|
||||
pub fn domainDestroy(domain: u16) void {
|
||||
if (kind == .none) return;
|
||||
const d = &domains[domain];
|
||||
if (!d.in_use) return;
|
||||
freeTables(d.page_table_root, backend.levels);
|
||||
d.* = .{};
|
||||
}
|
||||
|
||||
/// Attach `bdf`'s device to `domain` and pre-load any RMRR range recorded for it.
|
||||
pub fn attachDevice(domain: u16, bdf: u16) void {
|
||||
if (kind == .none) return;
|
||||
const d = &domains[domain];
|
||||
d.bdf = bdf;
|
||||
backend.attach(bdf, hardwareId(domain), d.page_table_root);
|
||||
}
|
||||
|
||||
/// Return `bdf`'s device to not-present + invalidate.
|
||||
pub fn detachDevice(bdf: u16) void {
|
||||
if (kind == .none) return;
|
||||
backend.detach(bdf);
|
||||
}
|
||||
|
||||
/// Identity-map [physical, physical+len) into `domain` (read+write) and invalidate.
|
||||
/// Unconditional domain-selective invalidation after every map — correct under VT-d
|
||||
/// caching-mode and free otherwise.
|
||||
pub fn map(domain: u16, physical: u64, len: u64) bool {
|
||||
if (kind == .none) return true;
|
||||
const d = &domains[domain];
|
||||
if (!d.in_use) return false;
|
||||
if (!mapRange(d.page_table_root, physical, len)) return false;
|
||||
backend.invalidateDomain(hardwareId(domain));
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Unmap [physical, physical+len) from `domain` and invalidate. MUST finish its
|
||||
/// invalidation before the caller returns the frames to pmm — a stale IOTLB entry
|
||||
/// pointing at a reallocated frame is the use-after-free this ordering prevents.
|
||||
pub fn unmap(domain: u16, physical: u64, len: u64) void {
|
||||
if (kind == .none) return;
|
||||
const d = &domains[domain];
|
||||
if (!d.in_use) return;
|
||||
unmapRange(d.page_table_root, physical, len);
|
||||
backend.invalidateDomain(hardwareId(domain));
|
||||
}
|
||||
|
||||
/// Poll the hardware for translation faults, log them, return the count. Called by the
|
||||
/// IOMMU test case and opportunistically after a device detaches.
|
||||
pub fn faultDrain() usize {
|
||||
if (kind == .none) return 0;
|
||||
return backend.faultDrain();
|
||||
}
|
||||
|
||||
/// The physical address `virtual` maps to in `domain`, or null if unmapped — a test
|
||||
/// helper that walks the domain's page tables (identity mappings return `virtual`).
|
||||
pub fn translationOf(domain: u16, virtual: u64) ?u64 {
|
||||
if (kind == .none) return virtual;
|
||||
const d = &domains[domain];
|
||||
if (!d.in_use) return null;
|
||||
var table = d.page_table_root;
|
||||
var level = backend.levels;
|
||||
while (level > 1) : (level -= 1) {
|
||||
const entry = tableAt(table)[indexAt(virtual, level)];
|
||||
if (!backend.isPresent(entry)) return null;
|
||||
if (level == 2 and isHugeLeaf(entry))
|
||||
return (entry & address_mask) | (virtual & (huge_page_size - 1));
|
||||
table = entry & address_mask;
|
||||
}
|
||||
const leaf = tableAt(table)[indexAt(virtual, 1)];
|
||||
if (!backend.isPresent(leaf)) return null;
|
||||
return (leaf & address_mask) | (virtual & page_mask);
|
||||
}
|
||||
|
||||
// --- the shared page-table walker -----------------------------------------------------
|
||||
// 512-entry, 9-bits-per-level, 4 KiB tables reached through the physmap — the shape both
|
||||
// VT-d second-level and AMD-Vi native tables share. The backend supplies the entry bits.
|
||||
|
||||
fn tableAt(physical: u64) [*]volatile u64 {
|
||||
return @ptrFromInt(boot_handoff.physicalToVirtual(physical));
|
||||
}
|
||||
|
||||
fn allocTable() ?u64 {
|
||||
const frame = pmm.alloc() orelse return null;
|
||||
const table = tableAt(frame);
|
||||
var i: usize = 0;
|
||||
while (i < 512) : (i += 1) table[i] = 0;
|
||||
return frame;
|
||||
}
|
||||
|
||||
const address_mask: u64 = 0x000F_FFFF_FFFF_F000;
|
||||
|
||||
fn indexAt(virtual: u64, level: u8) usize {
|
||||
// level 1 is the leaf table; shift = 12 + 9*(level-1).
|
||||
const shift: u6 = @intCast(12 + 9 * (@as(u32, level) - 1));
|
||||
return @intCast((virtual >> shift) & 0x1FF);
|
||||
}
|
||||
|
||||
/// Descend to (allocating) the next-level table below `entry_ptr`, returning its
|
||||
/// physical base. null on out-of-memory.
|
||||
fn descend(entry_ptr: *volatile u64, level: u8) ?u64 {
|
||||
const entry = entry_ptr.*;
|
||||
if (backend.isPresent(entry)) return entry & address_mask;
|
||||
const table = allocTable() orelse return null;
|
||||
backend.flushStructure(@intFromPtr(tableAt(table)));
|
||||
entry_ptr.* = backend.makeTable(table, level);
|
||||
backend.flushStructure(@intFromPtr(entry_ptr));
|
||||
return table;
|
||||
}
|
||||
|
||||
fn mapRange(root: u64, physical: u64, len: u64) bool {
|
||||
const start = physical & ~page_mask;
|
||||
const end = (physical + len + page_mask) & ~page_mask;
|
||||
var addr = start;
|
||||
while (addr < end) {
|
||||
// 2 MiB leaf when the backend allows it and both address and remaining span are
|
||||
// huge-aligned — keeps table memory sane for the blanket-identity and real-PC
|
||||
// cases without a separate superpage path per backend.
|
||||
const huge = backend.supports_huge_pages and
|
||||
addr % huge_page_size == 0 and (end - addr) >= huge_page_size;
|
||||
if (!mapOne(root, addr, huge)) return false;
|
||||
addr += if (huge) huge_page_size else page_size;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
fn mapOne(root: u64, addr: u64, huge: bool) bool {
|
||||
const leaf_level: u8 = if (huge) 2 else 1;
|
||||
var table = root;
|
||||
var level = backend.levels;
|
||||
while (level > leaf_level) : (level -= 1) {
|
||||
const entry_ptr = &tableAt(table)[indexAt(addr, level)];
|
||||
table = descend(entry_ptr, level) orelse return false;
|
||||
}
|
||||
const leaf_ptr = &tableAt(table)[indexAt(addr, leaf_level)];
|
||||
leaf_ptr.* = backend.makeLeaf(addr, huge);
|
||||
backend.flushStructure(@intFromPtr(leaf_ptr));
|
||||
return true;
|
||||
}
|
||||
|
||||
fn unmapRange(root: u64, physical: u64, len: u64) void {
|
||||
const start = physical & ~page_mask;
|
||||
const end = (physical + len + page_mask) & ~page_mask;
|
||||
var addr = start;
|
||||
while (addr < end) {
|
||||
const huge = backend.supports_huge_pages and
|
||||
addr % huge_page_size == 0 and (end - addr) >= huge_page_size;
|
||||
unmapOne(root, addr, huge);
|
||||
addr += if (huge) huge_page_size else page_size;
|
||||
}
|
||||
}
|
||||
|
||||
fn unmapOne(root: u64, addr: u64, huge: bool) void {
|
||||
const leaf_level: u8 = if (huge) 2 else 1;
|
||||
var table = root;
|
||||
var level = backend.levels;
|
||||
while (level > leaf_level) : (level -= 1) {
|
||||
const entry = tableAt(table)[indexAt(addr, level)];
|
||||
if (!backend.isPresent(entry)) return; // nothing mapped here
|
||||
table = entry & address_mask;
|
||||
}
|
||||
const leaf_ptr = &tableAt(table)[indexAt(addr, leaf_level)];
|
||||
leaf_ptr.* = 0;
|
||||
backend.flushStructure(@intFromPtr(leaf_ptr));
|
||||
}
|
||||
|
||||
/// Post-order free of a domain's whole table tree.
|
||||
fn freeTables(root: u64, level: u8) void {
|
||||
if (level > 1) {
|
||||
const table = tableAt(root);
|
||||
var i: usize = 0;
|
||||
while (i < 512) : (i += 1) {
|
||||
const entry = table[i];
|
||||
if (!backend.isPresent(entry)) continue;
|
||||
// A 2 MiB leaf sits at level 2 and points at RAM, not a sub-table.
|
||||
if (level == 2 and isHugeLeaf(entry)) continue;
|
||||
freeTables(entry & address_mask, level - 1);
|
||||
}
|
||||
}
|
||||
pmm.free(root);
|
||||
}
|
||||
|
||||
fn isHugeLeaf(entry: u64) bool {
|
||||
// Both backends set a page-size bit (VT-d bit 7, AMD leaf next-level=0 at level 2).
|
||||
// The backend's makeLeaf encodes it; the walker only needs "is this a leaf, not a
|
||||
// pointer" at level 2, which huge leaves are by construction.
|
||||
return entry & huge_leaf_bit != 0;
|
||||
}
|
||||
|
||||
/// The size-bit the backends set on a 2 MiB leaf (VT-d SL-PTE PS bit 7; AMD encodes a
|
||||
/// leaf as next-level 0, so the core marks huge leaves with this software bit — an
|
||||
/// ignored bit in both formats — to tell them apart from table pointers when freeing).
|
||||
pub const huge_leaf_bit: u64 = 1 << 7;
|
||||
|
||||
fn hardwareId(domain: u16) u16 {
|
||||
return domain + 1; // id 0 is reserved by both architectures
|
||||
}
|
||||
|
||||
fn logEnabled(info: platform.PlatformInformation) void {
|
||||
log.write("/system/kernel: iommu online (Intel VT-d)\n");
|
||||
log.print(" version : 0x{x}\n", .{info.iommu_version});
|
||||
log.print(" agaw : {d} levels\n", .{backend.levels});
|
||||
log.print(" rmrr : {d} region(s) premapped\n", .{info.rmrr_count});
|
||||
if (info.rmrr_skipped > 0)
|
||||
log.print(" rmrr : WARNING {d} scope(s) skipped — a device keeps an unmapped firmware buffer\n", .{info.rmrr_skipped});
|
||||
if (info.iommu_extra_units > 0)
|
||||
log.print(" units : WARNING {d} other DRHD(s) — their scoped devices are NOT translated\n", .{info.iommu_extra_units});
|
||||
}
|
||||
Reference in New Issue
Block a user