kernel: the IOMMU backends move behind the architecture boundary
VT-d and AMD-Vi are x86 hardware, but lived in the architecture-neutral kernel tree and leaked further: the core's public Kind enum named both vendors, and the ACPI parser read the VT-d version/capability registers (raw volatile MMIO inside table discovery). Now the vendor backends live in architecture/x86_64/ behind architecture.iommu — the core hands over the discovery facts plus an injected environment (frame allocation + the log sink, the same pattern enablePaging uses) and receives the hardware vtable back, so the backends never import kernel internals and an ARM port supplies its SMMU with no core change. Discovery keeps table facts only; the live-unit register check moved into VT-d detect (version reading zero now stays fail-open). The unused kindOf() is gone. Log shapes the harness pins (iommu online, DANOS-IOMMU-FAULT) are unchanged; all five IOMMU QEMU cases pass.
This commit is contained in:
@@ -17,6 +17,11 @@ const io = @import("io.zig");
|
||||
const smp = @import("smp.zig");
|
||||
const pcpu = @import("per-cpu.zig");
|
||||
|
||||
/// The x86-64 IOMMU backends (Intel VT-d, AMD-Vi) behind their dispatch
|
||||
/// surface — the architecture-neutral IOMMU core (system/kernel/iommu.zig)
|
||||
/// reaches the hardware only through this.
|
||||
pub const iommu = @import("iommu.zig");
|
||||
|
||||
/// The saved register/trap frame passed to a fault handler.
|
||||
pub const CpuState = idt.CpuState;
|
||||
|
||||
|
||||
@@ -0,0 +1,265 @@
|
||||
//! AMD-Vi (AMD I/O Virtualization) backend for the IOMMU core, behind the
|
||||
//! architecture boundary. The AMD analogue of iommu-intel.zig: it supplies
|
||||
//! the architecture-neutral core's `Backend` vtable (iommu.zig beside this file)
|
||||
//! with AMD-Vi's page-table entry bits and drives the device table, command buffer, and
|
||||
//! event log.
|
||||
//!
|
||||
//! **UNTESTED on real AMD hardware.** danos is developed on Intel; this backend is
|
||||
//! validated only against QEMU's `-device amd-iommu,dma-remap=on`, whose AMD-Vi
|
||||
//! emulation is far less exercised than its Intel one. Every code path here should be
|
||||
//! read as "QEMU-verified, real-AMD-unverified" until an AMD machine confirms it.
|
||||
//!
|
||||
//! Interrupt remapping is left off (the DTE forwards interrupts unmapped), so MSI writes
|
||||
//! to the 0xFEE00000 range reach the APIC untranslated, exactly as on the Intel path.
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const boot_handoff = @import("boot-handoff");
|
||||
const paging = @import("paging.zig");
|
||||
const iommu = @import("iommu.zig");
|
||||
|
||||
const page_size = abi.page_size;
|
||||
|
||||
// MMIO register offsets from the IOMMU control-register base.
|
||||
const reg_device_table_base = 0x00; // base | (pages-1) in bits 8:0
|
||||
const reg_command_buffer_base = 0x08; // base | (ComLen << 56)
|
||||
const reg_event_log_base = 0x10; // base | (EventLen << 56)
|
||||
const reg_control = 0x18;
|
||||
const reg_command_head = 0x2000;
|
||||
const reg_command_tail = 0x2008;
|
||||
const reg_event_head = 0x2010;
|
||||
const reg_event_tail = 0x2018;
|
||||
const reg_status = 0x2020;
|
||||
|
||||
const control_iommu_enable: u64 = 1 << 0;
|
||||
const control_event_log_enable: u64 = 1 << 2;
|
||||
const control_command_buffer_enable: u64 = 1 << 12;
|
||||
|
||||
// Device table: one 32-byte DTE (4 qwords) per requester id, indexed by bdf.
|
||||
const device_table_pages = 512; // 2 MiB = 65536 entries (a full 256-bus segment)
|
||||
const dte_qwords = 4;
|
||||
const dte_valid: u64 = 1 << 0; // V
|
||||
const dte_translation_valid: u64 = 1 << 1; // TV
|
||||
const dte_mode_shift = 9; // bits 11:9 — page-table levels
|
||||
const dte_read: u64 = 1 << 61; // IR
|
||||
const dte_write: u64 = 1 << 62; // IW
|
||||
const dte_intctl_forward: u64 = @as(u64, 1) << 60; // qword2 bits 61:60 = 01b: forward interrupts unmapped
|
||||
|
||||
// Page-table entry bits (AMD native format).
|
||||
const pte_present: u64 = 1 << 0; // PR
|
||||
const pte_next_level_shift = 9; // bits 11:9: 0 = leaf, N = pointer to a level-N table
|
||||
const pte_read: u64 = 1 << 61; // IR
|
||||
const pte_write: u64 = 1 << 62; // IW
|
||||
const address_mask: u64 = 0x000F_FFFF_FFFF_F000;
|
||||
|
||||
// Command buffer / event log: one 4 KiB frame each = 256 entries.
|
||||
const ring_entries = 256;
|
||||
const ring_length_code: u64 = 8; // log2(256), the ComLen/EventLen field value
|
||||
const command_opcode_shift = 60; // opcode in bits 63:60 of qword 0
|
||||
const command_completion_wait: u64 = 0x01;
|
||||
const command_invalidate_devtab: u64 = 0x02;
|
||||
const command_invalidate_pages: u64 = 0x03;
|
||||
const completion_wait_store: u64 = 1 << 1; // S: store `data` to the supplied address
|
||||
const invalidate_pages_all: u64 = 0x000F_FFFF_FFFF_F000 | 1; // address bits 51:12 all-ones + S
|
||||
|
||||
const levels: u8 = 4; // 48-bit IOVA, matching the Intel 4-level path
|
||||
|
||||
var register_base: usize = 0;
|
||||
var device_table: u64 = 0; // physical base of the device table
|
||||
var command_buffer: u64 = 0;
|
||||
var event_log: u64 = 0;
|
||||
var completion_frame: u64 = 0; // COMPLETION_WAIT store target
|
||||
var command_tail: u32 = 0; // our software copy of the command tail (bytes)
|
||||
|
||||
var fault_log_budget: u32 = 32;
|
||||
var completion_warned = false;
|
||||
|
||||
fn read64(offset: usize) u64 {
|
||||
return @as(*const volatile u64, @ptrFromInt(register_base + offset)).*;
|
||||
}
|
||||
fn write64(offset: usize, value: u64) void {
|
||||
@as(*volatile u64, @ptrFromInt(register_base + offset)).* = value;
|
||||
}
|
||||
fn ram(physical: u64) [*]volatile u64 {
|
||||
return @ptrFromInt(boot_handoff.physicalToVirtual(physical));
|
||||
}
|
||||
|
||||
/// Map the register window, allocate the device table / command buffer / event log.
|
||||
/// Returns the vtable, or null if the boot-time allocations fail.
|
||||
pub fn detect(discovery: iommu.Discovery) ?iommu.Backend {
|
||||
register_base = paging.mapMmio(discovery.register_base, 16 * 1024, true);
|
||||
|
||||
device_table = iommu.environment.allocateContiguous(device_table_pages, ~@as(u64, 0)) orelse return null;
|
||||
zero(device_table, device_table_pages); // all-zero DTE = V=0 = deny every device
|
||||
command_buffer = allocZeroedFrame() orelse return null;
|
||||
event_log = allocZeroedFrame() orelse return null;
|
||||
completion_frame = allocZeroedFrame() orelse return null;
|
||||
|
||||
return iommu.Backend{
|
||||
.levels = levels,
|
||||
.supports_huge_pages = false, // 4 KiB leaves only (AMD superpage encoding deferred)
|
||||
.enable = enable,
|
||||
.makeLeaf = makeLeaf,
|
||||
.makeTable = makeTable,
|
||||
.isPresent = isPresent,
|
||||
.flushStructure = flushStructure,
|
||||
.attach = attach,
|
||||
.detach = detach,
|
||||
.invalidateDomain = invalidateDomain,
|
||||
.faultDrain = faultDrain,
|
||||
};
|
||||
}
|
||||
|
||||
/// Program the base registers and enable translation. The device table is already
|
||||
/// zeroed (every device denied) except any entries `attach` wrote for RMRR/claimed
|
||||
/// devices, so turning translation on blocks all other DMA and logs it.
|
||||
fn enable() void {
|
||||
write64(reg_device_table_base, (device_table & address_mask) | (device_table_pages - 1));
|
||||
write64(reg_command_buffer_base, (command_buffer & address_mask) | (ring_length_code << 56));
|
||||
write64(reg_event_log_base, (event_log & address_mask) | (ring_length_code << 56));
|
||||
write64(reg_command_head, 0);
|
||||
write64(reg_command_tail, 0);
|
||||
write64(reg_event_head, 0);
|
||||
write64(reg_event_tail, 0);
|
||||
command_tail = 0;
|
||||
// Buffers first, then the master enable.
|
||||
write64(reg_control, control_command_buffer_enable | control_event_log_enable);
|
||||
write64(reg_control, control_command_buffer_enable | control_event_log_enable | control_iommu_enable);
|
||||
|
||||
iommu.environment.write("/system/kernel: iommu online (AMD-Vi) — UNTESTED on real AMD hardware (QEMU-verified only)\n");
|
||||
var buffer: [48]u8 = undefined;
|
||||
if (std.fmt.bufPrint(&buffer, " levels : {d} (48-bit)\n", .{levels})) |line|
|
||||
iommu.environment.write(line)
|
||||
else |_| {}
|
||||
}
|
||||
|
||||
// --- Backend vtable ------------------------------------------------------------------
|
||||
|
||||
fn makeLeaf(physical: u64, huge: bool) u64 {
|
||||
_ = huge; // 4 KiB only
|
||||
return (physical & address_mask) | pte_present | pte_read | pte_write; // Next Level 0 = leaf
|
||||
}
|
||||
fn makeTable(table_physical: u64, level: u8) u64 {
|
||||
// This entry (at `level`) points to a table one level down; AMD's Next Level names
|
||||
// the pointed-to table's level.
|
||||
const next_level: u64 = @as(u64, level) - 1;
|
||||
return (table_physical & address_mask) | pte_present | pte_read | pte_write | (next_level << pte_next_level_shift);
|
||||
}
|
||||
fn isPresent(entry: u64) bool {
|
||||
return (entry & pte_present) != 0;
|
||||
}
|
||||
fn flushStructure(address: usize) void {
|
||||
_ = address; // AMD-Vi reads its structures coherently
|
||||
}
|
||||
|
||||
fn attach(bdf: u16, domain: u16, page_table_root: u64) void {
|
||||
const dte = ram(device_table) + @as(usize, bdf) * dte_qwords;
|
||||
dte[0] = (page_table_root & address_mask) | dte_valid | dte_translation_valid |
|
||||
(@as(u64, levels) << dte_mode_shift) | dte_read | dte_write;
|
||||
dte[1] = @as(u64, domain); // DomainID in bits 15:0
|
||||
dte[2] = dte_intctl_forward; // forward interrupts unmapped (no remapping)
|
||||
dte[3] = 0;
|
||||
invalidateDevice(bdf);
|
||||
invalidateDomain(domain);
|
||||
}
|
||||
|
||||
fn detach(bdf: u16) void {
|
||||
const dte = ram(device_table) + @as(usize, bdf) * dte_qwords;
|
||||
dte[0] = 0; // V=0: deny
|
||||
dte[1] = 0;
|
||||
dte[2] = 0;
|
||||
dte[3] = 0;
|
||||
invalidateDevice(bdf);
|
||||
}
|
||||
|
||||
fn invalidateDomain(domain: u16) void {
|
||||
submitCommand((command_invalidate_pages << command_opcode_shift) | (@as(u64, domain) << 32), invalidate_pages_all);
|
||||
completeAndWait();
|
||||
}
|
||||
|
||||
fn faultDrain() usize {
|
||||
const tail = read64(reg_event_tail) & 0xFFFF_FFF0;
|
||||
var head = read64(reg_event_head) & 0xFFFF_FFF0;
|
||||
if (head == tail) return 0;
|
||||
var seen: usize = 0;
|
||||
while (head != tail) {
|
||||
const entry = ram(event_log) + (head / 8);
|
||||
const code: u4 = @truncate(entry[0] >> command_opcode_shift);
|
||||
if (code == 0x2) { // IO_PAGE_FAULT
|
||||
const source: u16 = @truncate(entry[0]);
|
||||
logFault(source, entry[1]);
|
||||
}
|
||||
seen += 1;
|
||||
head += 16;
|
||||
if (head >= ring_entries * 16) head = 0;
|
||||
}
|
||||
write64(reg_event_head, head);
|
||||
return seen;
|
||||
}
|
||||
|
||||
fn logFault(source: u16, address: u64) void {
|
||||
if (fault_log_budget == 0) return;
|
||||
fault_log_budget -= 1;
|
||||
var buffer: [128]u8 = undefined;
|
||||
if (std.fmt.bufPrint(&buffer, "DANOS-IOMMU-FAULT: bdf={x:0>2}:{x:0>2}.{d} addr=0x{x} reason=amd-io-page-fault\n", .{
|
||||
source >> 8,
|
||||
(source >> 3) & 0x1F,
|
||||
source & 0x7,
|
||||
address,
|
||||
})) |line| iommu.environment.write(line) else |_| {}
|
||||
if (fault_log_budget == 0) iommu.environment.write("DANOS-IOMMU-FAULT: (further faults suppressed)\n");
|
||||
}
|
||||
|
||||
// --- command ring --------------------------------------------------------------------
|
||||
|
||||
fn invalidateDevice(bdf: u16) void {
|
||||
submitCommand((command_invalidate_devtab << command_opcode_shift) | @as(u64, bdf), 0);
|
||||
completeAndWait();
|
||||
}
|
||||
|
||||
/// Append a 128-bit command (two qwords) to the ring and advance the tail.
|
||||
fn submitCommand(qword0: u64, qword1: u64) void {
|
||||
const slot = ram(command_buffer) + (command_tail / 8);
|
||||
slot[0] = qword0;
|
||||
slot[1] = qword1;
|
||||
command_tail += 16;
|
||||
if (command_tail >= ring_entries * 16) command_tail = 0;
|
||||
write64(reg_command_tail, command_tail);
|
||||
}
|
||||
|
||||
/// Append a COMPLETION_WAIT (store form) and spin until the IOMMU writes our sentinel to
|
||||
/// the completion frame. QEMU consumes the command buffer synchronously on the tail-
|
||||
/// register write, so by the time we poll the prior invalidation is already applied; the
|
||||
/// store confirmation is belt-and-suspenders for real hardware. If it never lands
|
||||
/// (QEMU's amd-iommu does not implement the store form), warn ONCE and proceed — the
|
||||
/// invalidation itself has happened.
|
||||
fn completeAndWait() void {
|
||||
const sentinel: u64 = 0xC0FFEE;
|
||||
ram(completion_frame)[0] = 0;
|
||||
submitCommand(
|
||||
(command_completion_wait << command_opcode_shift) | (completion_frame & 0x000F_FFFF_FFFF_FFF8) | completion_wait_store,
|
||||
sentinel,
|
||||
);
|
||||
var spins: u64 = 0;
|
||||
while (@as(*const volatile u64, @ptrFromInt(boot_handoff.physicalToVirtual(completion_frame))).* != sentinel) {
|
||||
spins += 1;
|
||||
if (spins > 100_000) {
|
||||
if (!completion_warned) {
|
||||
completion_warned = true;
|
||||
iommu.environment.write("/system/kernel: AMD-Vi COMPLETION_WAIT store not observed — proceeding (QEMU processes commands synchronously)\n");
|
||||
}
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn allocZeroedFrame() ?u64 {
|
||||
const frame = iommu.environment.allocateFrame() orelse return null;
|
||||
zero(frame, 1);
|
||||
return frame;
|
||||
}
|
||||
fn zero(physical: u64, pages: usize) void {
|
||||
const words = ram(physical);
|
||||
var i: usize = 0;
|
||||
while (i < pages * page_size / 8) : (i += 1) words[i] = 0;
|
||||
}
|
||||
@@ -0,0 +1,327 @@
|
||||
//! Intel VT-d backend for the IOMMU core, behind the architecture boundary.
|
||||
//!
|
||||
//! Provides the architecture-neutral core (system/kernel/iommu.zig) with the VT-d
|
||||
//! hardware specifics behind the `Backend` vtable (iommu.zig beside this file): second-level page-table entry bits, the root/context table structure, the
|
||||
//! translation-enable and invalidation register sequences, and the fault drain. The
|
||||
//! core owns the domain table and the page-table walk; this file owns the registers.
|
||||
//!
|
||||
//! Register model (VT-d spec §10-11): offsets from the DRHD register base. The unit is
|
||||
//! programmed once at enable (root table + Translation Enable), then touched only for
|
||||
//! per-device context changes, per-domain invalidations, and fault draining. Interrupt
|
||||
//! remapping is deliberately left OFF (GCMD.IRE stays 0): with it off, upstream writes
|
||||
//! to 0xFEE0_0000-0xFEEF_FFFF are treated as interrupt requests and bypass second-level
|
||||
//! translation, so the existing MSI contract survives unchanged.
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const boot_handoff = @import("boot-handoff");
|
||||
const paging = @import("paging.zig");
|
||||
const iommu = @import("iommu.zig");
|
||||
|
||||
const page_size = abi.page_size;
|
||||
|
||||
// Register offsets from the unit's base.
|
||||
const reg_cap = 0x08; // Capability (64)
|
||||
const reg_ecap = 0x10; // Extended Capability (64)
|
||||
const reg_gcmd = 0x18; // Global Command (32, write-only)
|
||||
const reg_gsts = 0x1C; // Global Status (32, read-only)
|
||||
const reg_rtaddr = 0x20; // Root Table Address (64)
|
||||
const reg_ccmd = 0x28; // Context Command (64)
|
||||
const reg_fsts = 0x34; // Fault Status (32)
|
||||
|
||||
const gcmd_te: u32 = 1 << 31; // Translation Enable
|
||||
const gcmd_srtp: u32 = 1 << 30; // Set Root Table Pointer
|
||||
const gsts_tes: u32 = 1 << 31; // Translation Enable Status
|
||||
const gsts_rtps: u32 = 1 << 30; // Root Table Pointer Status
|
||||
|
||||
const cap_cm: u64 = 1 << 7; // Caching Mode
|
||||
const cap_sagaw_shift = 8; // Supported Adjusted Guest Address Widths, bits 12:8
|
||||
const cap_sagaw_39bit: u64 = 1 << 9; // 3-level
|
||||
const cap_sagaw_48bit: u64 = 1 << 10; // 4-level
|
||||
const cap_fro_shift = 24; // Fault-Recording Register Offset, bits 33:24 (×16)
|
||||
const cap_nfr_shift = 40; // Number of Fault Recording regs, bits 47:40 (+1)
|
||||
const ecap_coherent: u64 = 1 << 0; // hardware snoops CPU caches reading its structures
|
||||
const ecap_iro_shift = 8; // IOTLB Register Offset, bits 17:8 (×16)
|
||||
|
||||
const ccmd_icc: u64 = 1 << 63; // Invalidate Context-Cache
|
||||
const ccmd_cirg_global: u64 = @as(u64, 1) << 61; // global granularity
|
||||
const ccmd_cirg_device: u64 = @as(u64, 3) << 61; // device-selective
|
||||
|
||||
const iotlb_ivt: u64 = 1 << 63; // Invalidate IOTLB
|
||||
const iotlb_iirg_global: u64 = @as(u64, 1) << 60;
|
||||
const iotlb_iirg_domain: u64 = @as(u64, 2) << 60;
|
||||
const iotlb_dr: u64 = 1 << 49; // drain reads
|
||||
const iotlb_dw: u64 = 1 << 48; // drain writes
|
||||
|
||||
const fsts_ppf: u32 = 1 << 1; // Primary Pending Fault
|
||||
|
||||
// Second-level PTE bits.
|
||||
const slpte_read: u64 = 1 << 0;
|
||||
const slpte_write: u64 = 1 << 1;
|
||||
const slpte_page_size: u64 = 1 << 7; // a 2 MiB leaf (== iommu.huge_leaf_bit)
|
||||
|
||||
const address_mask: u64 = 0x000F_FFFF_FFFF_F000;
|
||||
|
||||
var register_base: usize = 0;
|
||||
var version: u32 = 0;
|
||||
var capabilities: u64 = 0;
|
||||
var extended_capabilities: u64 = 0;
|
||||
var coherent: bool = true; // ECAP.C — whether clflush is unnecessary
|
||||
var levels: u8 = 4;
|
||||
var context_aw: u64 = 2; // context-entry AW field (001=3-level, 010=4-level)
|
||||
var gcmd_shadow: u32 = 0; // sticky GCMD bits (TE etc.), for the write-only register
|
||||
|
||||
var root_table: u64 = 0; // physical base of the 256-entry root table
|
||||
var context_table: [256]u64 = .{0} ** 256; // per-bus context table physical, 0 = none
|
||||
|
||||
var fault_log_budget: u32 = 32; // rate-limit: log this many faults, then just count
|
||||
var faults_suppressed: u64 = 0;
|
||||
|
||||
fn read32(offset: usize) u32 {
|
||||
return @as(*const volatile u32, @ptrFromInt(register_base + offset)).*;
|
||||
}
|
||||
fn write32(offset: usize, value: u32) void {
|
||||
@as(*volatile u32, @ptrFromInt(register_base + offset)).* = value;
|
||||
}
|
||||
fn read64(offset: usize) u64 {
|
||||
return @as(*const volatile u64, @ptrFromInt(register_base + offset)).*;
|
||||
}
|
||||
fn write64(offset: usize, value: u64) void {
|
||||
@as(*volatile u64, @ptrFromInt(register_base + offset)).* = value;
|
||||
}
|
||||
|
||||
fn tableAt(physical: u64) [*]volatile u64 {
|
||||
return @ptrFromInt(boot_handoff.physicalToVirtual(physical));
|
||||
}
|
||||
|
||||
/// Map the register window, read caps, pick the address width. Returns the vtable, or
|
||||
/// null when the unit is not live or advertises no address width danos can drive.
|
||||
pub fn detect(discovery: iommu.Discovery) ?iommu.Backend {
|
||||
// Map 16 KiB: FRCD and IOTLB registers can sit past the first page (CAP.FRO /
|
||||
// ECAP.IRO are 16-byte-unit offsets).
|
||||
register_base = paging.mapMmio(discovery.register_base, 16 * 1024, true);
|
||||
// The Version register's low byte is major.minor; reading it back nonzero is
|
||||
// the live-mappable-unit sanity check (previously a kernel-test assertion).
|
||||
version = read32(0x00);
|
||||
if (version == 0) return null;
|
||||
capabilities = read64(reg_cap);
|
||||
extended_capabilities = read64(reg_ecap);
|
||||
coherent = (extended_capabilities & ecap_coherent) != 0;
|
||||
|
||||
const sagaw = capabilities >> cap_sagaw_shift;
|
||||
if (sagaw & cap_sagaw_48bit != 0) {
|
||||
levels = 4;
|
||||
context_aw = 2; // 010b
|
||||
} else if (sagaw & cap_sagaw_39bit != 0) {
|
||||
levels = 3;
|
||||
context_aw = 1; // 001b
|
||||
} else {
|
||||
return null; // no width we build tables for
|
||||
}
|
||||
|
||||
root_table = allocZeroed() orelse return null;
|
||||
|
||||
return iommu.Backend{
|
||||
.levels = levels,
|
||||
.supports_huge_pages = true,
|
||||
.enable = enable,
|
||||
.makeLeaf = makeLeaf,
|
||||
.makeTable = makeTable,
|
||||
.isPresent = isPresent,
|
||||
.flushStructure = flushStructure,
|
||||
.attach = attach,
|
||||
.detach = detach,
|
||||
.invalidateDomain = invalidateDomain,
|
||||
.faultDrain = faultDrain,
|
||||
};
|
||||
}
|
||||
|
||||
/// Program the root table and turn Translation Enable on. The core has already created
|
||||
/// and populated the RMRR domains (their context entries are live via `attach`), so at
|
||||
/// this instant every OTHER device's context entry is not-present and will fault — which
|
||||
/// for stale firmware bus-mastering is the desired evidence, not a bug.
|
||||
fn enable() void {
|
||||
write64(reg_rtaddr, root_table); // legacy mode (bits 11:10 = 00)
|
||||
setGlobalCommand(gcmd_srtp);
|
||||
spinStatus(gsts_rtps);
|
||||
globalInvalidate();
|
||||
setGlobalCommand(gcmd_te);
|
||||
spinStatus(gsts_tes);
|
||||
gcmd_shadow |= gcmd_te;
|
||||
|
||||
iommu.environment.write("/system/kernel: iommu online (Intel VT-d)\n");
|
||||
var buffer: [64]u8 = undefined;
|
||||
if (std.fmt.bufPrint(&buffer, " version : 0x{x}\n", .{version})) |line|
|
||||
iommu.environment.write(line)
|
||||
else |_| {}
|
||||
if (std.fmt.bufPrint(&buffer, " agaw : {d} levels\n", .{levels})) |line|
|
||||
iommu.environment.write(line)
|
||||
else |_| {}
|
||||
}
|
||||
|
||||
// --- Backend vtable ------------------------------------------------------------------
|
||||
|
||||
fn makeLeaf(physical: u64, huge: bool) u64 {
|
||||
return (physical & address_mask) | slpte_read | slpte_write | (if (huge) slpte_page_size else 0);
|
||||
}
|
||||
fn makeTable(table_physical: u64, level: u8) u64 {
|
||||
_ = level;
|
||||
return (table_physical & address_mask) | slpte_read | slpte_write;
|
||||
}
|
||||
fn isPresent(entry: u64) bool {
|
||||
return (entry & (slpte_read | slpte_write)) != 0;
|
||||
}
|
||||
|
||||
fn flushStructure(address: usize) void {
|
||||
if (coherent) return; // the unit snoops CPU caches; no flush needed (QEMU)
|
||||
asm volatile ("clflush (%[p])"
|
||||
:
|
||||
: [p] "r" (address),
|
||||
: .{ .memory = true });
|
||||
}
|
||||
|
||||
fn attach(bdf: u16, domain: u16, page_table_root: u64) void {
|
||||
const bus: u8 = @intCast(bdf >> 8);
|
||||
const devfn: u8 = @intCast(bdf & 0xFF);
|
||||
|
||||
// Lazily allocate this bus's context table and link it into the root table.
|
||||
if (context_table[bus] == 0) {
|
||||
const table = allocZeroed() orelse return;
|
||||
context_table[bus] = table;
|
||||
const root_entry = &tableAt(root_table)[@as(usize, bus) * 2]; // 16-byte entries
|
||||
root_entry.* = (table & address_mask) | 1; // present
|
||||
flushStructure(@intFromPtr(root_entry));
|
||||
}
|
||||
|
||||
const context = tableAt(context_table[bus]);
|
||||
const low = &context[@as(usize, devfn) * 2];
|
||||
const high = &context[@as(usize, devfn) * 2 + 1];
|
||||
high.* = (context_aw & 0x7) | (@as(u64, domain) << 8); // AW + DID
|
||||
low.* = (page_table_root & address_mask) | 1; // present, TT=00 (use second-level)
|
||||
flushStructure(@intFromPtr(high));
|
||||
flushStructure(@intFromPtr(low));
|
||||
|
||||
invalidateContextDevice(bdf, domain);
|
||||
invalidateDomain(domain);
|
||||
}
|
||||
|
||||
fn detach(bdf: u16) void {
|
||||
const bus: u8 = @intCast(bdf >> 8);
|
||||
const devfn: u8 = @intCast(bdf & 0xFF);
|
||||
if (context_table[bus] == 0) return;
|
||||
const context = tableAt(context_table[bus]);
|
||||
context[@as(usize, devfn) * 2] = 0; // not present
|
||||
context[@as(usize, devfn) * 2 + 1] = 0;
|
||||
flushStructure(@intFromPtr(&context[@as(usize, devfn) * 2]));
|
||||
invalidateContextDevice(bdf, 0);
|
||||
globalIotlb();
|
||||
}
|
||||
|
||||
fn invalidateDomain(domain: u16) void {
|
||||
const iotlb_offset = iotlbOffset();
|
||||
write64(iotlb_offset, iotlb_ivt | iotlb_iirg_domain | iotlb_dr | iotlb_dw | (@as(u64, domain) << 32));
|
||||
spin64(iotlb_offset, iotlb_ivt);
|
||||
}
|
||||
|
||||
fn faultDrain() usize {
|
||||
const fsts = read32(reg_fsts);
|
||||
if (fsts & fsts_ppf == 0) return 0;
|
||||
|
||||
const fro = (capabilities >> cap_fro_shift) & 0x3FF;
|
||||
const nfr = ((capabilities >> cap_nfr_shift) & 0xFF) + 1;
|
||||
const frcd_base = @as(usize, @intCast(fro)) * 16;
|
||||
|
||||
var seen: usize = 0;
|
||||
var i: usize = 0;
|
||||
while (i < nfr) : (i += 1) {
|
||||
const off = frcd_base + i * 16;
|
||||
const high = read64(off + 8);
|
||||
if (high & (@as(u64, 1) << 63) == 0) continue; // F: no fault recorded here
|
||||
const low = read64(off);
|
||||
const address = low & ~@as(u64, 0xFFF);
|
||||
const source: u16 = @intCast(high & 0xFFFF);
|
||||
const reason: u8 = @intCast((high >> 32) & 0xFF);
|
||||
const is_read = (high >> 62) & 1; // T: 1 = read request
|
||||
logFault(source, address, reason, is_read == 1);
|
||||
write64(off + 8, @as(u64, 1) << 63); // RW1C: clear F
|
||||
seen += 1;
|
||||
}
|
||||
write32(reg_fsts, fsts); // clear PPF/PFO (write-1-to-clear)
|
||||
return seen;
|
||||
}
|
||||
|
||||
fn logFault(source: u16, address: u64, reason: u8, is_read: bool) void {
|
||||
if (fault_log_budget > 0) {
|
||||
fault_log_budget -= 1;
|
||||
var buffer: [128]u8 = undefined;
|
||||
if (std.fmt.bufPrint(&buffer, "DANOS-IOMMU-FAULT: bdf={x:0>2}:{x:0>2}.{d} addr=0x{x} reason=0x{x} write={d}\n", .{
|
||||
source >> 8,
|
||||
(source >> 3) & 0x1F,
|
||||
source & 0x7,
|
||||
address,
|
||||
reason,
|
||||
@intFromBool(!is_read),
|
||||
})) |line| iommu.environment.write(line) else |_| {}
|
||||
if (fault_log_budget == 0)
|
||||
iommu.environment.write("DANOS-IOMMU-FAULT: (further faults suppressed)\n");
|
||||
} else {
|
||||
faults_suppressed += 1;
|
||||
}
|
||||
}
|
||||
|
||||
// --- register helpers ----------------------------------------------------------------
|
||||
|
||||
fn setGlobalCommand(one_shot: u32) void {
|
||||
// GCMD is write-only: every write must carry the full sticky state plus the one-shot
|
||||
// bit being requested, or a set sticky bit (TE) would be cleared as a side effect.
|
||||
write32(reg_gcmd, gcmd_shadow | one_shot);
|
||||
}
|
||||
|
||||
fn spinStatus(bit: u32) void {
|
||||
var spins: u64 = 0;
|
||||
while (read32(reg_gsts) & bit == 0) {
|
||||
spins += 1;
|
||||
if (spins > 10_000_000) {
|
||||
iommu.environment.write("/system/kernel: WARNING VT-d status bit never set — translation may be incomplete\n");
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn spin64(offset: usize, bit: u64) void {
|
||||
var spins: u64 = 0;
|
||||
while (read64(offset) & bit != 0) {
|
||||
spins += 1;
|
||||
if (spins > 10_000_000) return;
|
||||
}
|
||||
}
|
||||
|
||||
fn globalInvalidate() void {
|
||||
write64(reg_ccmd, ccmd_icc | ccmd_cirg_global);
|
||||
spin64(reg_ccmd, ccmd_icc);
|
||||
globalIotlb();
|
||||
}
|
||||
|
||||
fn globalIotlb() void {
|
||||
const iotlb_offset = iotlbOffset();
|
||||
write64(iotlb_offset, iotlb_ivt | iotlb_iirg_global | iotlb_dr | iotlb_dw);
|
||||
spin64(iotlb_offset, iotlb_ivt);
|
||||
}
|
||||
|
||||
fn invalidateContextDevice(bdf: u16, domain: u16) void {
|
||||
write64(reg_ccmd, ccmd_icc | ccmd_cirg_device | (@as(u64, bdf) << 16) | domain);
|
||||
spin64(reg_ccmd, ccmd_icc);
|
||||
}
|
||||
|
||||
fn iotlbOffset() usize {
|
||||
const iro = (extended_capabilities >> ecap_iro_shift) & 0x3FF;
|
||||
return @as(usize, @intCast(iro)) * 16 + 8; // IOTLB register sits at IRO*16 + 8
|
||||
}
|
||||
|
||||
fn allocZeroed() ?u64 {
|
||||
const frame = iommu.environment.allocateFrame() orelse return null;
|
||||
const table = tableAt(frame);
|
||||
var i: usize = 0;
|
||||
while (i < 512) : (i += 1) table[i] = 0;
|
||||
return frame;
|
||||
}
|
||||
@@ -0,0 +1,77 @@
|
||||
//! x86-64 IOMMU backends, behind the architecture boundary: Intel VT-d
|
||||
//! (iommu-intel.zig) and AMD-Vi (iommu-amd.zig). The architecture-neutral
|
||||
//! core (system/kernel/iommu.zig) owns the domain table and the shared
|
||||
//! page-table walker; it hands this file the firmware discovery facts and an
|
||||
//! environment (frame allocation + the log sink, injected the same way
|
||||
//! enablePaging receives its frame hooks), and gets back a hardware vtable.
|
||||
//! A new architecture supplies its own unit (ARM: the SMMU) from its own
|
||||
//! directory with no core change.
|
||||
|
||||
const intel = @import("iommu-intel.zig");
|
||||
const amd = @import("iommu-amd.zig");
|
||||
|
||||
/// What the platform's firmware tables reported: where the unit's registers
|
||||
/// live, and which programming model its table implies (an IVRS table
|
||||
/// describes AMD-Vi; a DMAR table describes Intel VT-d).
|
||||
pub const Discovery = struct {
|
||||
register_base: u64,
|
||||
amd: bool,
|
||||
};
|
||||
|
||||
/// What the backends need from the generic kernel, injected at detect so this
|
||||
/// module never imports kernel internals: physical-frame allocation for the
|
||||
/// hardware structures, and the kernel log sink (fault reports, warnings, the
|
||||
/// enable banner).
|
||||
pub const Environment = struct {
|
||||
allocateFrame: *const fn () ?u64,
|
||||
allocateContiguous: *const fn (count: usize, max_physical: u64) ?u64,
|
||||
write: *const fn (bytes: []const u8) void,
|
||||
};
|
||||
|
||||
/// The bit encodings and hardware operations a backend supplies to the shared
|
||||
/// core. Entry helpers build the raw page-table entries for the backend's
|
||||
/// format; the core walks the tree with them. The hardware ops act on a whole
|
||||
/// domain (identified by its hardware domain id = core index + 1) or device
|
||||
/// (by requester id / bdf).
|
||||
pub const Backend = struct {
|
||||
/// Number of page-table levels (3 or 4) the backend selected from hardware caps.
|
||||
levels: u8,
|
||||
/// Largest leaf the walker may emit: 4 KiB always, 2 MiB when the backend allows.
|
||||
supports_huge_pages: bool,
|
||||
|
||||
/// Raw entry bits for a leaf mapping `physical` (with the given size), and for a
|
||||
/// non-leaf entry pointing at `table_physical` at `level` (level counts down to 1
|
||||
/// at the leaf's parent). `isPresent` tests a read-back entry.
|
||||
makeLeaf: *const fn (physical: u64, huge: bool) u64,
|
||||
makeTable: *const fn (table_physical: u64, level: u8) u64,
|
||||
isPresent: *const fn (entry: u64) bool,
|
||||
/// Flush a cache line holding IOMMU structures the hardware reads non-coherently
|
||||
/// (VT-d with ECAP.C==0). A no-op where the unit snoops caches.
|
||||
flushStructure: *const fn (address: usize) void,
|
||||
|
||||
/// Turn translation on (the core has already seeded any pre-claim domains)
|
||||
/// and write the unit's identity lines to the log — the core follows with
|
||||
/// the neutral posture lines.
|
||||
enable: *const fn () void,
|
||||
/// Point `bdf`'s translation structure at `domain` (hardware id) and invalidate the
|
||||
/// context/device caches so the change takes effect.
|
||||
attach: *const fn (bdf: u16, domain: u16, page_table_root: u64) void,
|
||||
/// Return `bdf`'s translation structure to not-present + invalidate — all its DMA
|
||||
/// faults afterward.
|
||||
detach: *const fn (bdf: u16) void,
|
||||
/// Invalidate cached translations for `domain` (after a map or unmap).
|
||||
invalidateDomain: *const fn (domain: u16) void,
|
||||
/// Pull pending faults out of the hardware, log them (rate-limited), return the
|
||||
/// count seen this call.
|
||||
faultDrain: *const fn () usize,
|
||||
};
|
||||
|
||||
/// The injected kernel services, stored for the backends at detect time.
|
||||
pub var environment: Environment = undefined;
|
||||
|
||||
/// Probe the discovered unit and return its vtable, or null when it is
|
||||
/// unusable (the core stays fail-open and says so).
|
||||
pub fn detect(discovery: Discovery, injected: Environment) ?Backend {
|
||||
environment = injected;
|
||||
return if (discovery.amd) amd.detect(discovery) else intel.detect(discovery);
|
||||
}
|
||||
Reference in New Issue
Block a user