266 lines
11 KiB
Zig
266 lines
11 KiB
Zig
//! AMD-Vi (AMD I/O Virtualization) backend for the IOMMU core, behind the
|
|
//! architecture boundary. The AMD analogue of iommu-intel.zig: it supplies
|
|
//! the architecture-neutral core's `Backend` vtable (iommu.zig beside this file)
|
|
//! with AMD-Vi's page-table entry bits and drives the device table, command buffer, and
|
|
//! event log.
|
|
//!
|
|
//! **UNTESTED on real AMD hardware.** danos is developed on Intel; this backend is
|
|
//! validated only against QEMU's `-device amd-iommu,dma-remap=on`, whose AMD-Vi
|
|
//! emulation is far less exercised than its Intel one. Every code path here should be
|
|
//! read as "QEMU-verified, real-AMD-unverified" until an AMD machine confirms it.
|
|
//!
|
|
//! Interrupt remapping is left off (the DTE forwards interrupts unmapped), so MSI writes
|
|
//! to the 0xFEE00000 range reach the APIC untranslated, exactly as on the Intel path.
|
|
|
|
const std = @import("std");
|
|
const abi = @import("abi");
|
|
const boot_handoff = @import("boot-handoff");
|
|
const paging = @import("paging.zig");
|
|
const iommu = @import("iommu.zig");
|
|
|
|
const page_size = abi.page_size;
|
|
|
|
// MMIO register offsets from the IOMMU control-register base.
|
|
const reg_device_table_base = 0x00; // base | (pages-1) in bits 8:0
|
|
const reg_command_buffer_base = 0x08; // base | (ComLen << 56)
|
|
const reg_event_log_base = 0x10; // base | (EventLen << 56)
|
|
const reg_control = 0x18;
|
|
const reg_command_head = 0x2000;
|
|
const reg_command_tail = 0x2008;
|
|
const reg_event_head = 0x2010;
|
|
const reg_event_tail = 0x2018;
|
|
const reg_status = 0x2020;
|
|
|
|
const control_iommu_enable: u64 = 1 << 0;
|
|
const control_event_log_enable: u64 = 1 << 2;
|
|
const control_command_buffer_enable: u64 = 1 << 12;
|
|
|
|
// Device table: one 32-byte DTE (4 qwords) per requester id, indexed by bdf.
|
|
const device_table_pages = 512; // 2 MiB = 65536 entries (a full 256-bus segment)
|
|
const dte_qwords = 4;
|
|
const dte_valid: u64 = 1 << 0; // V
|
|
const dte_translation_valid: u64 = 1 << 1; // TV
|
|
const dte_mode_shift = 9; // bits 11:9 — page-table levels
|
|
const dte_read: u64 = 1 << 61; // IR
|
|
const dte_write: u64 = 1 << 62; // IW
|
|
const dte_intctl_forward: u64 = @as(u64, 1) << 60; // qword2 bits 61:60 = 01b: forward interrupts unmapped
|
|
|
|
// Page-table entry bits (AMD native format).
|
|
const pte_present: u64 = 1 << 0; // PR
|
|
const pte_next_level_shift = 9; // bits 11:9: 0 = leaf, N = pointer to a level-N table
|
|
const pte_read: u64 = 1 << 61; // IR
|
|
const pte_write: u64 = 1 << 62; // IW
|
|
const address_mask: u64 = 0x000F_FFFF_FFFF_F000;
|
|
|
|
// Command buffer / event log: one 4 KiB frame each = 256 entries.
|
|
const ring_entries = 256;
|
|
const ring_length_code: u64 = 8; // log2(256), the ComLen/EventLen field value
|
|
const command_opcode_shift = 60; // opcode in bits 63:60 of qword 0
|
|
const command_completion_wait: u64 = 0x01;
|
|
const command_invalidate_devtab: u64 = 0x02;
|
|
const command_invalidate_pages: u64 = 0x03;
|
|
const completion_wait_store: u64 = 1 << 1; // S: store `data` to the supplied address
|
|
const invalidate_pages_all: u64 = 0x000F_FFFF_FFFF_F000 | 1; // address bits 51:12 all-ones + S
|
|
|
|
const levels: u8 = 4; // 48-bit IOVA, matching the Intel 4-level path
|
|
|
|
var register_base: usize = 0;
|
|
var device_table: u64 = 0; // physical base of the device table
|
|
var command_buffer: u64 = 0;
|
|
var event_log: u64 = 0;
|
|
var completion_frame: u64 = 0; // COMPLETION_WAIT store target
|
|
var command_tail: u32 = 0; // our software copy of the command tail (bytes)
|
|
|
|
var fault_log_budget: u32 = 32;
|
|
var completion_warned = false;
|
|
|
|
fn read64(offset: usize) u64 {
|
|
return @as(*const volatile u64, @ptrFromInt(register_base + offset)).*;
|
|
}
|
|
fn write64(offset: usize, value: u64) void {
|
|
@as(*volatile u64, @ptrFromInt(register_base + offset)).* = value;
|
|
}
|
|
fn ram(physical: u64) [*]volatile u64 {
|
|
return @ptrFromInt(boot_handoff.physicalToVirtual(physical));
|
|
}
|
|
|
|
/// Map the register window, allocate the device table / command buffer / event log.
|
|
/// Returns the vtable, or null if the boot-time allocations fail.
|
|
pub fn detect(discovery: iommu.Discovery) ?iommu.Backend {
|
|
register_base = paging.mapMmio(discovery.register_base, 16 * 1024, true);
|
|
|
|
device_table = iommu.environment.allocateContiguous(device_table_pages, ~@as(u64, 0)) orelse return null;
|
|
zero(device_table, device_table_pages); // all-zero DTE = V=0 = deny every device
|
|
command_buffer = allocZeroedFrame() orelse return null;
|
|
event_log = allocZeroedFrame() orelse return null;
|
|
completion_frame = allocZeroedFrame() orelse return null;
|
|
|
|
return iommu.Backend{
|
|
.levels = levels,
|
|
.supports_huge_pages = false, // 4 KiB leaves only (AMD superpage encoding deferred)
|
|
.enable = enable,
|
|
.makeLeaf = makeLeaf,
|
|
.makeTable = makeTable,
|
|
.isPresent = isPresent,
|
|
.flushStructure = flushStructure,
|
|
.attach = attach,
|
|
.detach = detach,
|
|
.invalidateDomain = invalidateDomain,
|
|
.faultDrain = faultDrain,
|
|
};
|
|
}
|
|
|
|
/// Program the base registers and enable translation. The device table is already
|
|
/// zeroed (every device denied) except any entries `attach` wrote for RMRR/claimed
|
|
/// devices, so turning translation on blocks all other DMA and logs it.
|
|
fn enable() void {
|
|
write64(reg_device_table_base, (device_table & address_mask) | (device_table_pages - 1));
|
|
write64(reg_command_buffer_base, (command_buffer & address_mask) | (ring_length_code << 56));
|
|
write64(reg_event_log_base, (event_log & address_mask) | (ring_length_code << 56));
|
|
write64(reg_command_head, 0);
|
|
write64(reg_command_tail, 0);
|
|
write64(reg_event_head, 0);
|
|
write64(reg_event_tail, 0);
|
|
command_tail = 0;
|
|
// Buffers first, then the master enable.
|
|
write64(reg_control, control_command_buffer_enable | control_event_log_enable);
|
|
write64(reg_control, control_command_buffer_enable | control_event_log_enable | control_iommu_enable);
|
|
|
|
iommu.environment.write("/system/kernel: iommu online (AMD-Vi) — UNTESTED on real AMD hardware (QEMU-verified only)\n");
|
|
var buffer: [48]u8 = undefined;
|
|
if (std.fmt.bufPrint(&buffer, " levels : {d} (48-bit)\n", .{levels})) |line|
|
|
iommu.environment.write(line)
|
|
else |_| {}
|
|
}
|
|
|
|
// --- Backend vtable ------------------------------------------------------------------
|
|
|
|
fn makeLeaf(physical: u64, huge: bool) u64 {
|
|
_ = huge; // 4 KiB only
|
|
return (physical & address_mask) | pte_present | pte_read | pte_write; // Next Level 0 = leaf
|
|
}
|
|
fn makeTable(table_physical: u64, level: u8) u64 {
|
|
// This entry (at `level`) points to a table one level down; AMD's Next Level names
|
|
// the pointed-to table's level.
|
|
const next_level: u64 = @as(u64, level) - 1;
|
|
return (table_physical & address_mask) | pte_present | pte_read | pte_write | (next_level << pte_next_level_shift);
|
|
}
|
|
fn isPresent(entry: u64) bool {
|
|
return (entry & pte_present) != 0;
|
|
}
|
|
fn flushStructure(address: usize) void {
|
|
_ = address; // AMD-Vi reads its structures coherently
|
|
}
|
|
|
|
fn attach(bdf: u16, domain: u16, page_table_root: u64) void {
|
|
const dte = ram(device_table) + @as(usize, bdf) * dte_qwords;
|
|
dte[0] = (page_table_root & address_mask) | dte_valid | dte_translation_valid |
|
|
(@as(u64, levels) << dte_mode_shift) | dte_read | dte_write;
|
|
dte[1] = @as(u64, domain); // DomainID in bits 15:0
|
|
dte[2] = dte_intctl_forward; // forward interrupts unmapped (no remapping)
|
|
dte[3] = 0;
|
|
invalidateDevice(bdf);
|
|
invalidateDomain(domain);
|
|
}
|
|
|
|
fn detach(bdf: u16) void {
|
|
const dte = ram(device_table) + @as(usize, bdf) * dte_qwords;
|
|
dte[0] = 0; // V=0: deny
|
|
dte[1] = 0;
|
|
dte[2] = 0;
|
|
dte[3] = 0;
|
|
invalidateDevice(bdf);
|
|
}
|
|
|
|
fn invalidateDomain(domain: u16) void {
|
|
submitCommand((command_invalidate_pages << command_opcode_shift) | (@as(u64, domain) << 32), invalidate_pages_all);
|
|
completeAndWait();
|
|
}
|
|
|
|
fn faultDrain() usize {
|
|
const tail = read64(reg_event_tail) & 0xFFFF_FFF0;
|
|
var head = read64(reg_event_head) & 0xFFFF_FFF0;
|
|
if (head == tail) return 0;
|
|
var seen: usize = 0;
|
|
while (head != tail) {
|
|
const entry = ram(event_log) + (head / 8);
|
|
const code: u4 = @truncate(entry[0] >> command_opcode_shift);
|
|
if (code == 0x2) { // IO_PAGE_FAULT
|
|
const source: u16 = @truncate(entry[0]);
|
|
logFault(source, entry[1]);
|
|
}
|
|
seen += 1;
|
|
head += 16;
|
|
if (head >= ring_entries * 16) head = 0;
|
|
}
|
|
write64(reg_event_head, head);
|
|
return seen;
|
|
}
|
|
|
|
fn logFault(source: u16, address: u64) void {
|
|
if (fault_log_budget == 0) return;
|
|
fault_log_budget -= 1;
|
|
var buffer: [128]u8 = undefined;
|
|
if (std.fmt.bufPrint(&buffer, "DANOS-IOMMU-FAULT: bdf={x:0>2}:{x:0>2}.{d} addr=0x{x} reason=amd-io-page-fault\n", .{
|
|
source >> 8,
|
|
(source >> 3) & 0x1F,
|
|
source & 0x7,
|
|
address,
|
|
})) |line| iommu.environment.write(line) else |_| {}
|
|
if (fault_log_budget == 0) iommu.environment.write("DANOS-IOMMU-FAULT: (further faults suppressed)\n");
|
|
}
|
|
|
|
// --- command ring --------------------------------------------------------------------
|
|
|
|
fn invalidateDevice(bdf: u16) void {
|
|
submitCommand((command_invalidate_devtab << command_opcode_shift) | @as(u64, bdf), 0);
|
|
completeAndWait();
|
|
}
|
|
|
|
/// Append a 128-bit command (two qwords) to the ring and advance the tail.
|
|
fn submitCommand(qword0: u64, qword1: u64) void {
|
|
const slot = ram(command_buffer) + (command_tail / 8);
|
|
slot[0] = qword0;
|
|
slot[1] = qword1;
|
|
command_tail += 16;
|
|
if (command_tail >= ring_entries * 16) command_tail = 0;
|
|
write64(reg_command_tail, command_tail);
|
|
}
|
|
|
|
/// Append a COMPLETION_WAIT (store form) and spin until the IOMMU writes our sentinel to
|
|
/// the completion frame. QEMU consumes the command buffer synchronously on the tail-
|
|
/// register write, so by the time we poll the prior invalidation is already applied; the
|
|
/// store confirmation is belt-and-suspenders for real hardware. If it never lands
|
|
/// (QEMU's amd-iommu does not implement the store form), warn ONCE and proceed — the
|
|
/// invalidation itself has happened.
|
|
fn completeAndWait() void {
|
|
const sentinel: u64 = 0xC0FFEE;
|
|
ram(completion_frame)[0] = 0;
|
|
submitCommand(
|
|
(command_completion_wait << command_opcode_shift) | (completion_frame & 0x000F_FFFF_FFFF_FFF8) | completion_wait_store,
|
|
sentinel,
|
|
);
|
|
var spins: u64 = 0;
|
|
while (@as(*const volatile u64, @ptrFromInt(boot_handoff.physicalToVirtual(completion_frame))).* != sentinel) {
|
|
spins += 1;
|
|
if (spins > 100_000) {
|
|
if (!completion_warned) {
|
|
completion_warned = true;
|
|
iommu.environment.write("/system/kernel: AMD-Vi COMPLETION_WAIT store not observed — proceeding (QEMU processes commands synchronously)\n");
|
|
}
|
|
return;
|
|
}
|
|
}
|
|
}
|
|
|
|
fn allocZeroedFrame() ?u64 {
|
|
const frame = iommu.environment.allocateFrame() orelse return null;
|
|
zero(frame, 1);
|
|
return frame;
|
|
}
|
|
fn zero(physical: u64, pages: usize) void {
|
|
const words = ram(physical);
|
|
var i: usize = 0;
|
|
while (i < pages * page_size / 8) : (i += 1) words[i] = 0;
|
|
}
|