449 lines
19 KiB
Zig
449 lines
19 KiB
Zig
//! system/kernel/iommu.zig — vendor-neutral IOMMU core: per-device DMA translation
|
|
//! domains over an Intel VT-d or AMD-Vi backend.
|
|
//!
|
|
//! The problem this closes: without an IOMMU, a claimed bus-mastering device can DMA to
|
|
//! ANY physical address, so a compromised or buggy driver reaches all of memory through
|
|
//! its device — driver isolation stops at the CPU's MMU. This core gives each claimed
|
|
//! PCI function its own translation domain; a device reaches only the physical ranges
|
|
//! mapped into its domain, and nothing else (kernel, page tables, other processes) is
|
|
//! visible to it.
|
|
//!
|
|
//! Design:
|
|
//! - **Identity mappings** (IOVA == physical). `dma_alloc` already hands drivers the
|
|
//! physical address they program into hardware; a domain simply makes that same
|
|
//! address the ONLY thing the device can reach. No IOVA allocator, and every
|
|
//! driver's register-programming code is untouched.
|
|
//! - **Vendor-neutral**: this file owns the domain table and a shared 512-entry
|
|
//! page-table walker; a `Backend` vtable supplies the hardware specifics (VT-d in
|
|
//! iommu-intel.zig, AMD-Vi in iommu-amd.zig) — the entry-bit encodings, the
|
|
//! enable/invalidate register dances, and the fault drain.
|
|
//! - **Fail-open**: when no IOMMU is found, `kind == .none` and every entry point is a
|
|
//! success no-op, so callers in process.zig stay unconditional and behavior is
|
|
//! byte-for-byte the pre-IOMMU kernel. The boot log states the posture.
|
|
//!
|
|
//! All entry points run under the big kernel lock (the caller holds it); no internal
|
|
//! locking. All memory comes from `pmm` reached through the physmap, like paging.zig.
|
|
|
|
const std = @import("std");
|
|
const abi = @import("abi");
|
|
const boot_handoff = @import("boot-handoff");
|
|
const pmm = @import("pmm.zig");
|
|
const platform = @import("platform");
|
|
const devices_broker = @import("devices-broker.zig");
|
|
const log = @import("log.zig");
|
|
const intel = @import("iommu-intel.zig");
|
|
const amd = @import("iommu-amd.zig");
|
|
|
|
const page_size: u64 = abi.page_size;
|
|
const page_mask: u64 = page_size - 1;
|
|
const huge_page_size: u64 = 2 * 1024 * 1024;
|
|
|
|
pub const Kind = enum { none, intel_vtd, amd_vi };
|
|
|
|
/// One domain per claimed PCI function. 64 mirrors devices-broker's device cap.
|
|
pub const maximum_domains = 64;
|
|
pub const invalid_domain: u16 = 0xFFFF;
|
|
|
|
/// The bit encodings and hardware operations a backend supplies to the shared core.
|
|
/// Entry helpers build the raw page-table entries for the backend's format; the core
|
|
/// walks the tree with them. The hardware ops act on a whole domain (identified by its
|
|
/// hardware domain id = core index + 1) or device (by requester id / bdf).
|
|
pub const Backend = struct {
|
|
/// Number of page-table levels (3 or 4) the backend selected from hardware caps.
|
|
levels: u8,
|
|
/// Largest leaf the walker may emit: 4 KiB always, 2 MiB when the backend allows.
|
|
supports_huge_pages: bool,
|
|
|
|
/// Raw entry bits for a leaf mapping `physical` (with the given size), and for a
|
|
/// non-leaf entry pointing at `table_physical` at `level` (level counts down to 1
|
|
/// at the leaf's parent). `isPresent` tests a read-back entry.
|
|
makeLeaf: *const fn (physical: u64, huge: bool) u64,
|
|
makeTable: *const fn (table_physical: u64, level: u8) u64,
|
|
isPresent: *const fn (entry: u64) bool,
|
|
/// Flush a cache line holding IOMMU structures the hardware reads non-coherently
|
|
/// (VT-d with ECAP.C==0). A no-op where the unit snoops caches.
|
|
flushStructure: *const fn (address: usize) void,
|
|
|
|
/// Point `bdf`'s translation structure at `domain` (hardware id) and invalidate the
|
|
/// context/device caches so the change takes effect.
|
|
attach: *const fn (bdf: u16, domain: u16, page_table_root: u64) void,
|
|
/// Return `bdf`'s translation structure to not-present + invalidate — all its DMA
|
|
/// faults afterward.
|
|
detach: *const fn (bdf: u16) void,
|
|
/// Invalidate cached translations for `domain` (after a map or unmap).
|
|
invalidateDomain: *const fn (domain: u16) void,
|
|
/// Pull pending faults out of the hardware, log them (rate-limited), return the
|
|
/// count seen this call.
|
|
faultDrain: *const fn () usize,
|
|
};
|
|
|
|
const Domain = struct {
|
|
in_use: bool = false,
|
|
owner: u32 = 0, // task that owns the attached device
|
|
bdf: u16 = 0, // requester id of the attached device
|
|
page_table_root: u64 = 0, // physical address of the top-level table
|
|
rmrr: bool = false, // a firmware reserved-region domain (persists across claims)
|
|
};
|
|
|
|
var kind: Kind = .none;
|
|
var backend: Backend = undefined;
|
|
var domains: [maximum_domains]Domain = .{Domain{}} ** maximum_domains;
|
|
|
|
pub fn kindOf() Kind {
|
|
return kind;
|
|
}
|
|
pub fn enabled() bool {
|
|
return kind != .none;
|
|
}
|
|
|
|
/// Detect the IOMMU, pick a backend, pre-map firmware reserved regions, and enable
|
|
/// translation. Fail-open (kind stays .none) when no unit exists — the caller logs the
|
|
/// posture. Must run after platform discovery and before any user process starts.
|
|
pub fn init() void {
|
|
const info = platform.platformInformation();
|
|
if (!info.iommu_present) {
|
|
kind = .none;
|
|
return;
|
|
}
|
|
// Pick the backend by vendor. A present-but-unusable unit stays fail-open with a
|
|
// logged reason rather than half-enabling.
|
|
if (info.iommu_is_amd) {
|
|
if (amd.detect(info)) |be| {
|
|
backend = be;
|
|
kind = .amd_vi;
|
|
} else {
|
|
kind = .none;
|
|
log.write("/system/kernel: WARNING AMD-Vi present but unusable — staying fail-open\n");
|
|
return;
|
|
}
|
|
} else {
|
|
if (intel.detect(info)) |be| {
|
|
backend = be;
|
|
kind = .intel_vtd;
|
|
} else {
|
|
kind = .none;
|
|
log.write("/system/kernel: WARNING IOMMU present but unusable — staying fail-open\n");
|
|
return;
|
|
}
|
|
}
|
|
|
|
// The translation structures start empty: every device is denied until its driver
|
|
// claims it (confineDevice gives it a private domain). PCI functions are enumerated
|
|
// post-boot by the ring-3 pci-bus driver, so there is nothing to attach at init.
|
|
if (kind == .amd_vi) amd.enable() else intel.enable();
|
|
logEnabled(info);
|
|
}
|
|
|
|
/// Per-claimed-device record: its private domain, so a driver's death tears down
|
|
/// exactly the domains it held.
|
|
const Confined = struct { active: bool = false, owner: u32 = 0, bdf: u16 = 0, domain: u16 = invalid_domain };
|
|
var confined: [maximum_domains]Confined = .{Confined{}} ** maximum_domains;
|
|
|
|
/// Place a just-claimed PCI function under IOMMU translation on behalf of `owner`: give
|
|
/// it a private empty domain, seed it with the device's own firmware reserved region,
|
|
/// and attach. Its DMA buffers arrive afterward as explicit grants — the owner's own
|
|
/// `dma_alloc`'d regions are bound by the claim path (`mapForDevice`), and cross-process
|
|
/// buffers by `dma_bind`. false only if a domain can't be allocated — the caller rolls
|
|
/// the claim back (a claim that can't be confined must not stand). No-op success when no
|
|
/// IOMMU exists (fail-open).
|
|
pub fn confineDevice(device_id: u64, bdf: u16, owner: u32) bool {
|
|
if (kind == .none) return true;
|
|
if (device_id >= confined.len) return true; // unusual id; leave it to fail-open
|
|
const domain = domainCreate(owner, bdf) orelse return false;
|
|
|
|
// Firmware reserved region for this device, if any (real hardware; QEMU has none).
|
|
const info = platform.platformInformation();
|
|
var i: usize = 0;
|
|
while (i < info.rmrr_count) : (i += 1) {
|
|
if (info.rmrr[i].bdf == bdf)
|
|
_ = map(domain, info.rmrr[i].base, info.rmrr[i].limit - info.rmrr[i].base + 1);
|
|
}
|
|
|
|
attachDevice(domain, bdf);
|
|
confined[@intCast(device_id)] = .{ .active = true, .owner = owner, .bdf = bdf, .domain = domain };
|
|
return true;
|
|
}
|
|
|
|
/// The confined record for `device_id`, or null if the device is not confined.
|
|
fn confinedOf(device_id: u64) ?*Confined {
|
|
if (device_id >= confined.len) return null;
|
|
const c = &confined[@intCast(device_id)];
|
|
return if (c.active) c else null;
|
|
}
|
|
|
|
/// Map a DMA region into a specific claimed device's domain (the device owner binding a
|
|
/// granted buffer). false if the device is not confined. No-op success without an IOMMU.
|
|
pub fn mapForDevice(device_id: u64, physical: u64, len: u64) bool {
|
|
if (kind == .none) return true;
|
|
const c = confinedOf(device_id) orelse return false;
|
|
return map(c.domain, physical, len);
|
|
}
|
|
|
|
/// Unmap a DMA region from a specific claimed device's domain. No-op if not confined.
|
|
pub fn unmapForDevice(device_id: u64, physical: u64, len: u64) void {
|
|
if (kind == .none) return;
|
|
const c = confinedOf(device_id) orelse return;
|
|
unmap(c.domain, physical, len);
|
|
}
|
|
|
|
/// Map a region into every claimed device owned by `owner` — the auto-bind of a task's
|
|
/// own freshly-`dma_alloc`'d buffer into the devices it drives.
|
|
pub fn mapRegionForOwner(owner: u32, physical: u64, len: u64) void {
|
|
if (kind == .none) return;
|
|
for (&confined) |*c| {
|
|
if (c.active and c.owner == owner) _ = map(c.domain, physical, len);
|
|
}
|
|
}
|
|
|
|
/// Unmap a region from EVERY claimed device's domain — the freed-region sweep. MUST run
|
|
/// before the frames return to pmm: a device translating to a reallocated frame is the
|
|
/// use-after-free this prevents. Cross-device because a granted buffer may be bound in a
|
|
/// domain other than its owner's.
|
|
pub fn unmapRegionEverywhere(physical: u64, len: u64) void {
|
|
if (kind == .none) return;
|
|
for (&confined) |*c| {
|
|
if (c.active) unmap(c.domain, physical, len);
|
|
}
|
|
}
|
|
|
|
/// A driver died or released its devices: tear down every domain it held (detach the
|
|
/// device, free the tables) so their DMA is blocked again and a restarted driver
|
|
/// re-claims cleanly. Runs BEFORE the broker claims and the DMA frames are released.
|
|
pub fn releaseAllOwnedBy(owner: u32) void {
|
|
if (kind == .none) return;
|
|
for (&confined) |*c| {
|
|
if (c.active and c.owner == owner) {
|
|
detachDevice(c.bdf);
|
|
domainDestroy(c.domain);
|
|
c.* = .{};
|
|
}
|
|
}
|
|
_ = faultDrain(); // log any faults a mid-DMA device raised as it was cut off
|
|
}
|
|
|
|
/// Allocate an empty domain (an empty top-level table). null when the table is full.
|
|
pub fn domainCreate(owner: u32, bdf: u16) ?u16 {
|
|
if (kind == .none) return 0; // fail-open: a dummy id the no-op ops ignore
|
|
for (&domains, 0..) |*d, index| {
|
|
if (d.in_use) continue;
|
|
const root = allocTable() orelse return null;
|
|
d.* = .{ .in_use = true, .owner = owner, .bdf = bdf, .page_table_root = root };
|
|
return @intCast(index);
|
|
}
|
|
return null;
|
|
}
|
|
|
|
/// Free a domain's page-table frames and its slot. Precondition: no device attached
|
|
/// (detach first).
|
|
pub fn domainDestroy(domain: u16) void {
|
|
if (kind == .none) return;
|
|
const d = &domains[domain];
|
|
if (!d.in_use) return;
|
|
freeTables(d.page_table_root, backend.levels);
|
|
d.* = .{};
|
|
}
|
|
|
|
/// Attach `bdf`'s device to `domain` and pre-load any RMRR range recorded for it.
|
|
pub fn attachDevice(domain: u16, bdf: u16) void {
|
|
if (kind == .none) return;
|
|
const d = &domains[domain];
|
|
d.bdf = bdf;
|
|
backend.attach(bdf, hardwareId(domain), d.page_table_root);
|
|
}
|
|
|
|
/// Return `bdf`'s device to not-present + invalidate.
|
|
pub fn detachDevice(bdf: u16) void {
|
|
if (kind == .none) return;
|
|
backend.detach(bdf);
|
|
}
|
|
|
|
/// Identity-map [physical, physical+len) into `domain` (read+write) and invalidate.
|
|
/// Unconditional domain-selective invalidation after every map — correct under VT-d
|
|
/// caching-mode and free otherwise.
|
|
pub fn map(domain: u16, physical: u64, len: u64) bool {
|
|
if (kind == .none) return true;
|
|
const d = &domains[domain];
|
|
if (!d.in_use) return false;
|
|
if (!mapRange(d.page_table_root, physical, len)) return false;
|
|
backend.invalidateDomain(hardwareId(domain));
|
|
return true;
|
|
}
|
|
|
|
/// Unmap [physical, physical+len) from `domain` and invalidate. MUST finish its
|
|
/// invalidation before the caller returns the frames to pmm — a stale IOTLB entry
|
|
/// pointing at a reallocated frame is the use-after-free this ordering prevents.
|
|
pub fn unmap(domain: u16, physical: u64, len: u64) void {
|
|
if (kind == .none) return;
|
|
const d = &domains[domain];
|
|
if (!d.in_use) return;
|
|
unmapRange(d.page_table_root, physical, len);
|
|
backend.invalidateDomain(hardwareId(domain));
|
|
}
|
|
|
|
/// Poll the hardware for translation faults, log them, return the count. Called by the
|
|
/// IOMMU test case and opportunistically after a device detaches.
|
|
pub fn faultDrain() usize {
|
|
if (kind == .none) return 0;
|
|
return backend.faultDrain();
|
|
}
|
|
|
|
/// The physical address `virtual` maps to in `domain`, or null if unmapped — a test
|
|
/// helper that walks the domain's page tables (identity mappings return `virtual`).
|
|
pub fn translationOf(domain: u16, virtual: u64) ?u64 {
|
|
if (kind == .none) return virtual;
|
|
const d = &domains[domain];
|
|
if (!d.in_use) return null;
|
|
var table = d.page_table_root;
|
|
var level = backend.levels;
|
|
while (level > 1) : (level -= 1) {
|
|
const entry = tableAt(table)[indexAt(virtual, level)];
|
|
if (!backend.isPresent(entry)) return null;
|
|
if (level == 2 and isHugeLeaf(entry))
|
|
return (entry & address_mask) | (virtual & (huge_page_size - 1));
|
|
table = entry & address_mask;
|
|
}
|
|
const leaf = tableAt(table)[indexAt(virtual, 1)];
|
|
if (!backend.isPresent(leaf)) return null;
|
|
return (leaf & address_mask) | (virtual & page_mask);
|
|
}
|
|
|
|
// --- the shared page-table walker -----------------------------------------------------
|
|
// 512-entry, 9-bits-per-level, 4 KiB tables reached through the physmap — the shape both
|
|
// VT-d second-level and AMD-Vi native tables share. The backend supplies the entry bits.
|
|
|
|
fn tableAt(physical: u64) [*]volatile u64 {
|
|
return @ptrFromInt(boot_handoff.physicalToVirtual(physical));
|
|
}
|
|
|
|
fn allocTable() ?u64 {
|
|
const frame = pmm.alloc() orelse return null;
|
|
const table = tableAt(frame);
|
|
var i: usize = 0;
|
|
while (i < 512) : (i += 1) table[i] = 0;
|
|
return frame;
|
|
}
|
|
|
|
const address_mask: u64 = 0x000F_FFFF_FFFF_F000;
|
|
|
|
fn indexAt(virtual: u64, level: u8) usize {
|
|
// level 1 is the leaf table; shift = 12 + 9*(level-1).
|
|
const shift: u6 = @intCast(12 + 9 * (@as(u32, level) - 1));
|
|
return @intCast((virtual >> shift) & 0x1FF);
|
|
}
|
|
|
|
/// Descend to (allocating) the next-level table below `entry_ptr`, returning its
|
|
/// physical base. null on out-of-memory.
|
|
fn descend(entry_ptr: *volatile u64, level: u8) ?u64 {
|
|
const entry = entry_ptr.*;
|
|
if (backend.isPresent(entry)) return entry & address_mask;
|
|
const table = allocTable() orelse return null;
|
|
backend.flushStructure(@intFromPtr(tableAt(table)));
|
|
entry_ptr.* = backend.makeTable(table, level);
|
|
backend.flushStructure(@intFromPtr(entry_ptr));
|
|
return table;
|
|
}
|
|
|
|
fn mapRange(root: u64, physical: u64, len: u64) bool {
|
|
const start = physical & ~page_mask;
|
|
const end = (physical + len + page_mask) & ~page_mask;
|
|
var addr = start;
|
|
while (addr < end) {
|
|
// 2 MiB leaf when the backend allows it and both address and remaining span are
|
|
// huge-aligned — keeps table memory sane for the blanket-identity and real-PC
|
|
// cases without a separate superpage path per backend.
|
|
const huge = backend.supports_huge_pages and
|
|
addr % huge_page_size == 0 and (end - addr) >= huge_page_size;
|
|
if (!mapOne(root, addr, huge)) return false;
|
|
addr += if (huge) huge_page_size else page_size;
|
|
}
|
|
return true;
|
|
}
|
|
|
|
fn mapOne(root: u64, addr: u64, huge: bool) bool {
|
|
const leaf_level: u8 = if (huge) 2 else 1;
|
|
var table = root;
|
|
var level = backend.levels;
|
|
while (level > leaf_level) : (level -= 1) {
|
|
const entry_ptr = &tableAt(table)[indexAt(addr, level)];
|
|
table = descend(entry_ptr, level) orelse return false;
|
|
}
|
|
const leaf_ptr = &tableAt(table)[indexAt(addr, leaf_level)];
|
|
leaf_ptr.* = backend.makeLeaf(addr, huge);
|
|
backend.flushStructure(@intFromPtr(leaf_ptr));
|
|
return true;
|
|
}
|
|
|
|
fn unmapRange(root: u64, physical: u64, len: u64) void {
|
|
const start = physical & ~page_mask;
|
|
const end = (physical + len + page_mask) & ~page_mask;
|
|
var addr = start;
|
|
while (addr < end) {
|
|
const huge = backend.supports_huge_pages and
|
|
addr % huge_page_size == 0 and (end - addr) >= huge_page_size;
|
|
unmapOne(root, addr, huge);
|
|
addr += if (huge) huge_page_size else page_size;
|
|
}
|
|
}
|
|
|
|
fn unmapOne(root: u64, addr: u64, huge: bool) void {
|
|
const leaf_level: u8 = if (huge) 2 else 1;
|
|
var table = root;
|
|
var level = backend.levels;
|
|
while (level > leaf_level) : (level -= 1) {
|
|
const entry = tableAt(table)[indexAt(addr, level)];
|
|
if (!backend.isPresent(entry)) return; // nothing mapped here
|
|
table = entry & address_mask;
|
|
}
|
|
const leaf_ptr = &tableAt(table)[indexAt(addr, leaf_level)];
|
|
leaf_ptr.* = 0;
|
|
backend.flushStructure(@intFromPtr(leaf_ptr));
|
|
}
|
|
|
|
/// Post-order free of a domain's whole table tree.
|
|
fn freeTables(root: u64, level: u8) void {
|
|
if (level > 1) {
|
|
const table = tableAt(root);
|
|
var i: usize = 0;
|
|
while (i < 512) : (i += 1) {
|
|
const entry = table[i];
|
|
if (!backend.isPresent(entry)) continue;
|
|
// A 2 MiB leaf sits at level 2 and points at RAM, not a sub-table.
|
|
if (level == 2 and isHugeLeaf(entry)) continue;
|
|
freeTables(entry & address_mask, level - 1);
|
|
}
|
|
}
|
|
pmm.free(root);
|
|
}
|
|
|
|
fn isHugeLeaf(entry: u64) bool {
|
|
// Both backends set a page-size bit (VT-d bit 7, AMD leaf next-level=0 at level 2).
|
|
// The backend's makeLeaf encodes it; the walker only needs "is this a leaf, not a
|
|
// pointer" at level 2, which huge leaves are by construction.
|
|
return entry & huge_leaf_bit != 0;
|
|
}
|
|
|
|
/// The size-bit the backends set on a 2 MiB leaf (VT-d SL-PTE PS bit 7; AMD encodes a
|
|
/// leaf as next-level 0, so the core marks huge leaves with this software bit — an
|
|
/// ignored bit in both formats — to tell them apart from table pointers when freeing).
|
|
pub const huge_leaf_bit: u64 = 1 << 7;
|
|
|
|
fn hardwareId(domain: u16) u16 {
|
|
return domain + 1; // id 0 is reserved by both architectures
|
|
}
|
|
|
|
fn logEnabled(info: platform.PlatformInformation) void {
|
|
if (kind == .amd_vi) {
|
|
log.write("/system/kernel: iommu online (AMD-Vi) — UNTESTED on real AMD hardware (QEMU-verified only)\n");
|
|
log.print(" levels : {d} (48-bit)\n", .{backend.levels});
|
|
return;
|
|
}
|
|
log.write("/system/kernel: iommu online (Intel VT-d)\n");
|
|
log.print(" version : 0x{x}\n", .{info.iommu_version});
|
|
log.print(" agaw : {d} levels\n", .{backend.levels});
|
|
log.print(" rmrr : {d} region(s) premapped\n", .{info.rmrr_count});
|
|
if (info.rmrr_skipped > 0)
|
|
log.print(" rmrr : WARNING {d} scope(s) skipped — a device keeps an unmapped firmware buffer\n", .{info.rmrr_skipped});
|
|
if (info.iommu_extra_units > 0)
|
|
log.print(" units : WARNING {d} other DRHD(s) — their scoped devices are NOT translated\n", .{info.iommu_extra_units});
|
|
}
|