diff --git a/system/kernel/acpi.zig b/system/kernel/acpi.zig index afed61e..b2b7656 100644 --- a/system/kernel/acpi.zig +++ b/system/kernel/acpi.zig @@ -95,18 +95,49 @@ pub const PlatformInformation = struct { override_count: usize = 0, /// Whether an IOMMU (VT-d DMA-remapping unit) was found in the ACPI DMAR table. /// When false, `device_claim` on a DMA-capable device is equivalent to granting - /// ring 0 — a device can DMA to any physical address (docs/driver-model.md M16). - /// Detection is the first step; per-device domain enforcement lands with the first - /// DMA driver. + /// ring 0 — a device can DMA to any physical address (docs/driver-model.md M16), + /// and the kernel says so at every boot (the fail-open platform log line). When + /// true, the IOMMU core builds per-device translation domains from this record. iommu_present: bool = false, - /// MMIO base of the first DMA-remapping hardware unit (DMAR DRHD), when present. + /// MMIO base of the selected DMA-remapping hardware unit (the DRHD with + /// INCLUDE_PCI_ALL — the catch-all unit covering every device not scoped to a + /// more specific one; falls back to the first unit when none carries the flag). iommu_base: u64 = 0, /// The unit's Version register (offset 0x00) — its low byte is major.minor; /// reading it back nonzero confirms a real, mappable VT-d unit. iommu_version: u32 = 0, /// The unit's Capability register (offset 0x08): supported address widths, number - /// of domains, etc. Recorded now; consumed when enforcement is built. + /// of domains, etc. Consumed by the IOMMU core when it enables translation. iommu_capabilities: u64 = 0, + /// Whether the selected unit carries INCLUDE_PCI_ALL. False means every unit is + /// device-scoped (unusual) — the core still enables on the selected unit but + /// devices outside its scope remain untranslated. + iommu_include_all: bool = false, + /// DRHD units in the DMAR beyond the selected one. Devices scoped to those units + /// (typically the integrated GPU) are NOT translated by v1 — the boot log warns. + iommu_extra_units: u8 = 0, + /// Reserved-memory regions (DMAR RMRRs): firmware-owned buffers a named device + /// keeps DMAing into across the OS handoff (classically the xHC keyboard-emulation + /// buffer). These must be identity-mapped in the device's domain BEFORE translation + /// enables, or platform firmware breaks. Only single-path endpoint scopes are + /// recorded; anything fancier is skipped with a loud log at parse time. + rmrr: [maximum_rmrr]RmrrRegion = undefined, + rmrr_count: usize = 0, + /// RMRR device scopes the parser could not record (multi-hop paths, sub-hierarchy + /// types, or table overflow). Non-zero means a device keeps an unmapped firmware + /// buffer — the kernel boot log warns loudly (the platform module itself is + /// log-free by design; it records, the kernel reports). + rmrr_skipped: u8 = 0, +}; + +pub const maximum_rmrr = 8; + +/// One recorded RMRR: the device (requester id) and the inclusive physical range it +/// must always be allowed to reach. +pub const RmrrRegion = struct { + bdf: u16, + base: u64, + limit: u64, }; /// Filled in by `discover`; the architecture layer reads it during bring-up. @@ -757,18 +788,33 @@ fn parseSpcr(header: *const SystemDescriptorTableHeader) void { // DMAR remapping-structure layout (Intel VT-d spec §8): the DMAR-specific header is 12 // bytes (host-address-width, flags, 10 reserved), then a list of {type u16, length u16} -// structures. Type 0 is a DRHD (DMA Remapping Hardware Unit Definition), whose 64-bit -// register base sits at offset 8 within it. +// structures. Type 0 is a DRHD (DMA Remapping Hardware Unit Definition): flags byte at +// offset 4 (bit 0 = INCLUDE_PCI_ALL, the catch-all unit), 64-bit register base at +// offset 8. Type 1 is an RMRR (Reserved Memory Region Reporting): a physical range at +// offsets 8/16 (base / inclusive limit) that the device(s) named by the trailing +// device-scope entries keep DMAing into across the firmware→OS handoff. const dmar_structures_offset = 48; // 36-byte ACPI header + 12-byte DMAR header const dmar_type_drhd: u16 = 0; +const dmar_type_rmrr: u16 = 1; +const drhd_flags_offset = 4; +const drhd_include_pci_all: u8 = 1; const drhd_register_base_offset = 8; +const rmrr_base_offset = 8; +const rmrr_limit_offset = 16; +const rmrr_scopes_offset = 24; +// Device-scope entry (within DRHD/RMRR structures): type 1 = PCI endpoint; the path is +// (device, function) byte pairs from offset 6, one pair per bridge hop plus the leaf. +const scope_type_pci_endpoint: u8 = 1; +const scope_start_bus_offset = 5; +const scope_path_offset = 6; -/// DMAR -> detect the IOMMU. Find the first DMA-remapping hardware unit, map its -/// register block, and record its version and capabilities. This is *detection only*: -/// it tells the system an IOMMU exists (so `device_claim` on a DMA device could one day -/// be gated by a per-device translation domain), but no domains are programmed yet — -/// enforcement is built with the first DMA driver, which is what there is to protect and -/// test against. See docs/driver-model.md (M16), the honest caveat. +/// DMAR -> the VT-d unit(s) and reserved memory regions. Walks every remapping +/// structure: selects the INCLUDE_PCI_ALL DRHD (the catch-all covering all devices not +/// scoped elsewhere — commonly the SECOND unit on real machines, after an iGPU-scoped +/// one), counts the rest so the boot log can warn that their devices stay untranslated, +/// and records single-path endpoint RMRRs for the IOMMU core to pre-map before it +/// enables translation. Multi-hop RMRR scopes are skipped loudly: better a named gap +/// than a silent one. fn parseDmar(hal: Hal, header: *const SystemDescriptorTableHeader) void { const base: [*]align(1) const u8 = @ptrCast(header); const total: usize = header.length; @@ -780,19 +826,65 @@ fn parseDmar(hal: Hal, header: *const SystemDescriptorTableHeader) void { if (length < 4 or off + length > total) break; // malformed; stop rather than loop if (kind == dmar_type_drhd) { const register_base = fadt(u64, base, total, off + drhd_register_base_offset) orelse 0; + const include_all = ((fadt(u8, base, total, off + drhd_flags_offset) orelse 0) & drhd_include_pci_all) != 0; if (register_base != 0) { - const regs = hal.mapMmio(register_base, abi.page_size, true); - platform_information.iommu_present = true; - platform_information.iommu_base = register_base; - platform_information.iommu_version = @as(*const volatile u32, @ptrFromInt(regs + 0x00)).*; - platform_information.iommu_capabilities = @as(*const volatile u64, @ptrFromInt(regs + 0x08)).*; - return; // first unit is enough for detection; multi-unit is future + // Selection: the INCLUDE_PCI_ALL unit wins; otherwise keep the first + // seen. A later catch-all replaces an earlier scoped unit. + const replace = !platform_information.iommu_present or + (include_all and !platform_information.iommu_include_all); + if (replace) { + if (platform_information.iommu_present) platform_information.iommu_extra_units += 1; + const regs = hal.mapMmio(register_base, abi.page_size, true); + platform_information.iommu_present = true; + platform_information.iommu_base = register_base; + platform_information.iommu_include_all = include_all; + platform_information.iommu_version = @as(*const volatile u32, @ptrFromInt(regs + 0x00)).*; + platform_information.iommu_capabilities = @as(*const volatile u64, @ptrFromInt(regs + 0x08)).*; + } else { + platform_information.iommu_extra_units += 1; + } } + } else if (kind == dmar_type_rmrr) { + parseRmrr(base, total, off, length); } off += length; } } +/// One RMRR structure: record a {bdf, base, limit} per single-path endpoint scope. +fn parseRmrr(base: [*]align(1) const u8, total: usize, off: usize, length: usize) void { + const range_base = fadt(u64, base, total, off + rmrr_base_offset) orelse return; + const range_limit = fadt(u64, base, total, off + rmrr_limit_offset) orelse return; + if (range_limit < range_base) return; + + var scope = off + rmrr_scopes_offset; + const end = off + length; + while (scope + 6 <= end) { + const scope_type = fadt(u8, base, total, scope) orelse break; + const scope_length = fadt(u8, base, total, scope + 1) orelse break; + if (scope_length < 6 or scope + scope_length > end) break; + if (scope_type == scope_type_pci_endpoint and scope_length == scope_path_offset + 2) { + // Single (device, function) pair: a directly-reachable endpoint. + const bus = fadt(u8, base, total, scope + scope_start_bus_offset) orelse 0; + const device = fadt(u8, base, total, scope + scope_path_offset) orelse 0; + const function = fadt(u8, base, total, scope + scope_path_offset + 1) orelse 0; + if (platform_information.rmrr_count < maximum_rmrr) { + platform_information.rmrr[platform_information.rmrr_count] = .{ + .bdf = (@as(u16, bus) << 8) | (@as(u16, device) << 3) | function, + .base = range_base, + .limit = range_limit, + }; + platform_information.rmrr_count += 1; + } else { + platform_information.rmrr_skipped +|= 1; // table full + } + } else { + platform_information.rmrr_skipped +|= 1; // multi-hop path or non-endpoint scope + } + scope += scope_length; + } +} + // --- helpers ---------------------------------------------------------------- /// Sum `len` bytes; an ACPI table/pointer is valid when the low 8 bits are zero. diff --git a/system/kernel/devices-broker.zig b/system/kernel/devices-broker.zig index cec0008..001ea93 100644 --- a/system/kernel/devices-broker.zig +++ b/system/kernel/devices-broker.zig @@ -167,6 +167,20 @@ pub fn releaseAllOwnedBy(owner: u32) void { } } +/// Release the claim on `id` iff `owner` holds it — the rollback for a claim that +/// cannot be confined (the IOMMU domain could not be created/attached). Returns true +/// when a claim was actually cleared. +pub fn unclaim(id: u64, owner: u32) bool { + if (id >= count) return false; + if (claimed[@intCast(id)]) |o| { + if (o == owner) { + claimed[@intCast(id)] = null; + return true; + } + } + return false; +} + /// Resource `index` of device `id`, or null if out of range. pub fn resourceOf(id: u64, index: u64) ?device_abi.ResourceDescriptor { if (id >= count) return null; @@ -175,6 +189,50 @@ pub fn resourceOf(id: u64, index: u64) ?device_abi.ResourceDescriptor { return d.resources[@intCast(index)]; } +/// The PCI requester id (bus<<8 | device<<3 | function) of device `id`, derived from +/// its config-space slice against its host bridge's ECAM window — the identity a VT-d +/// context entry / AMD-Vi DTE is keyed by. null when `id` is not a PCI function or the +/// geometry doesn't decode. The kernel never stored the BDF (the descriptor has no such +/// field); pci-bus encodes it into resource 0's physical base as +/// `ecam_base + ((bus - start_bus) << 20 | device << 15 | function << 12)`, and the +/// requester id the device emits uses the absolute bus, so we add `start_bus << 8` back. +pub fn pciAddressOf(id: u64) ?u16 { + if (id >= count) return null; + const d = &devices[@intCast(id)]; + if (d.class != @intFromEnum(device_abi.DeviceClass.pci_device)) return null; + if (d.resource_count == 0) return null; + const config = d.resources[0]; + if (config.kind != @intFromEnum(device_abi.ResourceKind.memory) or config.len != 4096) return null; + + // Walk up to the host bridge, whose resource 0 is the segment's ECAM window and + // resource 1 the bus_range (start_bus, bus_count). + var parent = d.parent; + while (parent != device_abi.no_parent and parent < count) { + const p = &devices[@intCast(parent)]; + if (p.class == @intFromEnum(device_abi.DeviceClass.pci_host_bridge)) { + if (p.resource_count < 2) return null; + const ecam = p.resources[0]; + const bus_range = p.resources[1]; + if (config.start < ecam.start or config.start >= ecam.start + ecam.len) return null; + const offset = config.start - ecam.start; + const start_bus: u16 = @intCast(bus_range.start & 0xFF); + return @intCast((offset >> 12) + (@as(u64, start_bus) << 8)); + } + parent = p.parent; + } + return null; +} + +/// Call `visit(id, bdf)` for every PCI function in the table — the IOMMU core's boot +/// sweep to place every device under a domain. Only functions whose BDF decodes are +/// visited. +pub fn forEachPciFunction(visit: *const fn (id: u64, bdf: u16) void) void { + var id: u64 = 0; + while (id < count) : (id += 1) { + if (pciAddressOf(id)) |bdf| visit(id, bdf); + } +} + /// Is `child` wholly inside `parent`? For a range (memory, io_port, bus_range) that's /// interval containment; for an irq it's equality, since an interrupt line is not /// divisible. Zero-length child ranges are refused — an empty window is meaningless diff --git a/system/kernel/iommu-intel.zig b/system/kernel/iommu-intel.zig new file mode 100644 index 0000000..a715f0e --- /dev/null +++ b/system/kernel/iommu-intel.zig @@ -0,0 +1,314 @@ +//! system/kernel/iommu-intel.zig — Intel VT-d backend for the IOMMU core. +//! +//! Provides the core (iommu.zig) with the VT-d hardware specifics behind its `Backend` +//! vtable: second-level page-table entry bits, the root/context table structure, the +//! translation-enable and invalidation register sequences, and the fault drain. The +//! core owns the domain table and the page-table walk; this file owns the registers. +//! +//! Register model (VT-d spec §10-11): offsets from the DRHD register base. The unit is +//! programmed once at enable (root table + Translation Enable), then touched only for +//! per-device context changes, per-domain invalidations, and fault draining. Interrupt +//! remapping is deliberately left OFF (GCMD.IRE stays 0): with it off, upstream writes +//! to 0xFEE0_0000-0xFEEF_FFFF are treated as interrupt requests and bypass second-level +//! translation, so the existing MSI contract survives unchanged. + +const std = @import("std"); +const abi = @import("abi"); +const boot_handoff = @import("boot-handoff"); +const pmm = @import("pmm.zig"); +const platform = @import("platform"); +const architecture = @import("architecture"); +const log = @import("log.zig"); +const iommu = @import("iommu.zig"); + +const page_size = abi.page_size; + +// Register offsets from the unit's base. +const reg_cap = 0x08; // Capability (64) +const reg_ecap = 0x10; // Extended Capability (64) +const reg_gcmd = 0x18; // Global Command (32, write-only) +const reg_gsts = 0x1C; // Global Status (32, read-only) +const reg_rtaddr = 0x20; // Root Table Address (64) +const reg_ccmd = 0x28; // Context Command (64) +const reg_fsts = 0x34; // Fault Status (32) + +const gcmd_te: u32 = 1 << 31; // Translation Enable +const gcmd_srtp: u32 = 1 << 30; // Set Root Table Pointer +const gsts_tes: u32 = 1 << 31; // Translation Enable Status +const gsts_rtps: u32 = 1 << 30; // Root Table Pointer Status + +const cap_cm: u64 = 1 << 7; // Caching Mode +const cap_sagaw_shift = 8; // Supported Adjusted Guest Address Widths, bits 12:8 +const cap_sagaw_39bit: u64 = 1 << 9; // 3-level +const cap_sagaw_48bit: u64 = 1 << 10; // 4-level +const cap_fro_shift = 24; // Fault-Recording Register Offset, bits 33:24 (×16) +const cap_nfr_shift = 40; // Number of Fault Recording regs, bits 47:40 (+1) +const ecap_coherent: u64 = 1 << 0; // hardware snoops CPU caches reading its structures +const ecap_iro_shift = 8; // IOTLB Register Offset, bits 17:8 (×16) + +const ccmd_icc: u64 = 1 << 63; // Invalidate Context-Cache +const ccmd_cirg_global: u64 = @as(u64, 1) << 61; // global granularity +const ccmd_cirg_device: u64 = @as(u64, 3) << 61; // device-selective + +const iotlb_ivt: u64 = 1 << 63; // Invalidate IOTLB +const iotlb_iirg_global: u64 = @as(u64, 1) << 60; +const iotlb_iirg_domain: u64 = @as(u64, 2) << 60; +const iotlb_dr: u64 = 1 << 49; // drain reads +const iotlb_dw: u64 = 1 << 48; // drain writes + +const fsts_ppf: u32 = 1 << 1; // Primary Pending Fault + +// Second-level PTE bits. +const slpte_read: u64 = 1 << 0; +const slpte_write: u64 = 1 << 1; +const slpte_page_size: u64 = 1 << 7; // a 2 MiB leaf (== iommu.huge_leaf_bit) + +const address_mask: u64 = 0x000F_FFFF_FFFF_F000; + +var register_base: usize = 0; +var capabilities: u64 = 0; +var extended_capabilities: u64 = 0; +var coherent: bool = true; // ECAP.C — whether clflush is unnecessary +var levels: u8 = 4; +var context_aw: u64 = 2; // context-entry AW field (001=3-level, 010=4-level) +var gcmd_shadow: u32 = 0; // sticky GCMD bits (TE etc.), for the write-only register + +var root_table: u64 = 0; // physical base of the 256-entry root table +var context_table: [256]u64 = .{0} ** 256; // per-bus context table physical, 0 = none + +var fault_log_budget: u32 = 32; // rate-limit: log this many faults, then just count +var faults_suppressed: u64 = 0; + +fn read32(offset: usize) u32 { + return @as(*const volatile u32, @ptrFromInt(register_base + offset)).*; +} +fn write32(offset: usize, value: u32) void { + @as(*volatile u32, @ptrFromInt(register_base + offset)).* = value; +} +fn read64(offset: usize) u64 { + return @as(*const volatile u64, @ptrFromInt(register_base + offset)).*; +} +fn write64(offset: usize, value: u64) void { + @as(*volatile u64, @ptrFromInt(register_base + offset)).* = value; +} + +fn tableAt(physical: u64) [*]volatile u64 { + return @ptrFromInt(boot_handoff.physicalToVirtual(physical)); +} + +/// Map the register window, read caps, pick the address width. Returns the vtable, or +/// null when the unit advertises no address width danos can drive. +pub fn detect(info: platform.PlatformInformation) ?iommu.Backend { + // Remap 16 KiB: FRCD and IOTLB registers can sit past the first page (CAP.FRO / + // ECAP.IRO are 16-byte-unit offsets). Idempotent with the detection-time mapping. + register_base = architecture.mapMmio(info.iommu_base, 16 * 1024, true); + capabilities = read64(reg_cap); + extended_capabilities = read64(reg_ecap); + coherent = (extended_capabilities & ecap_coherent) != 0; + + const sagaw = capabilities >> cap_sagaw_shift; + if (sagaw & cap_sagaw_48bit != 0) { + levels = 4; + context_aw = 2; // 010b + } else if (sagaw & cap_sagaw_39bit != 0) { + levels = 3; + context_aw = 1; // 001b + } else { + return null; // no width we build tables for + } + + root_table = allocZeroed() orelse return null; + + return iommu.Backend{ + .levels = levels, + .supports_huge_pages = true, + .makeLeaf = makeLeaf, + .makeTable = makeTable, + .isPresent = isPresent, + .flushStructure = flushStructure, + .attach = attach, + .detach = detach, + .invalidateDomain = invalidateDomain, + .faultDrain = faultDrain, + }; +} + +/// Program the root table and turn Translation Enable on. The core has already created +/// and populated the RMRR domains (their context entries are live via `attach`), so at +/// this instant every OTHER device's context entry is not-present and will fault — which +/// for stale firmware bus-mastering is the desired evidence, not a bug. +pub fn enable() void { + write64(reg_rtaddr, root_table); // legacy mode (bits 11:10 = 00) + setGlobalCommand(gcmd_srtp); + spinStatus(gsts_rtps); + globalInvalidate(); + setGlobalCommand(gcmd_te); + spinStatus(gsts_tes); + gcmd_shadow |= gcmd_te; +} + +// --- Backend vtable ------------------------------------------------------------------ + +fn makeLeaf(physical: u64, huge: bool) u64 { + return (physical & address_mask) | slpte_read | slpte_write | (if (huge) slpte_page_size else 0); +} +fn makeTable(table_physical: u64, level: u8) u64 { + _ = level; + return (table_physical & address_mask) | slpte_read | slpte_write; +} +fn isPresent(entry: u64) bool { + return (entry & (slpte_read | slpte_write)) != 0; +} + +fn flushStructure(address: usize) void { + if (coherent) return; // the unit snoops CPU caches; no flush needed (QEMU) + asm volatile ("clflush (%[p])" + : + : [p] "r" (address), + : .{ .memory = true }); +} + +fn attach(bdf: u16, domain: u16, page_table_root: u64) void { + const bus: u8 = @intCast(bdf >> 8); + const devfn: u8 = @intCast(bdf & 0xFF); + + // Lazily allocate this bus's context table and link it into the root table. + if (context_table[bus] == 0) { + const table = allocZeroed() orelse return; + context_table[bus] = table; + const root_entry = &tableAt(root_table)[@as(usize, bus) * 2]; // 16-byte entries + root_entry.* = (table & address_mask) | 1; // present + flushStructure(@intFromPtr(root_entry)); + } + + const context = tableAt(context_table[bus]); + const low = &context[@as(usize, devfn) * 2]; + const high = &context[@as(usize, devfn) * 2 + 1]; + high.* = (context_aw & 0x7) | (@as(u64, domain) << 8); // AW + DID + low.* = (page_table_root & address_mask) | 1; // present, TT=00 (use second-level) + flushStructure(@intFromPtr(high)); + flushStructure(@intFromPtr(low)); + + invalidateContextDevice(bdf, domain); + invalidateDomain(domain); +} + +fn detach(bdf: u16) void { + const bus: u8 = @intCast(bdf >> 8); + const devfn: u8 = @intCast(bdf & 0xFF); + if (context_table[bus] == 0) return; + const context = tableAt(context_table[bus]); + context[@as(usize, devfn) * 2] = 0; // not present + context[@as(usize, devfn) * 2 + 1] = 0; + flushStructure(@intFromPtr(&context[@as(usize, devfn) * 2])); + invalidateContextDevice(bdf, 0); + globalIotlb(); +} + +fn invalidateDomain(domain: u16) void { + const iotlb_offset = iotlbOffset(); + write64(iotlb_offset, iotlb_ivt | iotlb_iirg_domain | iotlb_dr | iotlb_dw | (@as(u64, domain) << 32)); + spin64(iotlb_offset, iotlb_ivt); +} + +fn faultDrain() usize { + const fsts = read32(reg_fsts); + if (fsts & fsts_ppf == 0) return 0; + + const fro = (capabilities >> cap_fro_shift) & 0x3FF; + const nfr = ((capabilities >> cap_nfr_shift) & 0xFF) + 1; + const frcd_base = @as(usize, @intCast(fro)) * 16; + + var seen: usize = 0; + var i: usize = 0; + while (i < nfr) : (i += 1) { + const off = frcd_base + i * 16; + const high = read64(off + 8); + if (high & (@as(u64, 1) << 63) == 0) continue; // F: no fault recorded here + const low = read64(off); + const address = low & ~@as(u64, 0xFFF); + const source: u16 = @intCast(high & 0xFFFF); + const reason: u8 = @intCast((high >> 32) & 0xFF); + const is_read = (high >> 62) & 1; // T: 1 = read request + logFault(source, address, reason, is_read == 1); + write64(off + 8, @as(u64, 1) << 63); // RW1C: clear F + seen += 1; + } + write32(reg_fsts, fsts); // clear PPF/PFO (write-1-to-clear) + return seen; +} + +fn logFault(source: u16, address: u64, reason: u8, is_read: bool) void { + if (fault_log_budget > 0) { + fault_log_budget -= 1; + log.print("DANOS-IOMMU-FAULT: bdf={x:0>2}:{x:0>2}.{d} addr=0x{x} reason=0x{x} write={d}\n", .{ + source >> 8, + (source >> 3) & 0x1F, + source & 0x7, + address, + reason, + @intFromBool(!is_read), + }); + if (fault_log_budget == 0) + log.write("DANOS-IOMMU-FAULT: (further faults suppressed)\n"); + } else { + faults_suppressed += 1; + } +} + +// --- register helpers ---------------------------------------------------------------- + +fn setGlobalCommand(one_shot: u32) void { + // GCMD is write-only: every write must carry the full sticky state plus the one-shot + // bit being requested, or a set sticky bit (TE) would be cleared as a side effect. + write32(reg_gcmd, gcmd_shadow | one_shot); +} + +fn spinStatus(bit: u32) void { + var spins: u64 = 0; + while (read32(reg_gsts) & bit == 0) { + spins += 1; + if (spins > 10_000_000) { + log.write("/system/kernel: WARNING VT-d status bit never set — translation may be incomplete\n"); + return; + } + } +} + +fn spin64(offset: usize, bit: u64) void { + var spins: u64 = 0; + while (read64(offset) & bit != 0) { + spins += 1; + if (spins > 10_000_000) return; + } +} + +fn globalInvalidate() void { + write64(reg_ccmd, ccmd_icc | ccmd_cirg_global); + spin64(reg_ccmd, ccmd_icc); + globalIotlb(); +} + +fn globalIotlb() void { + const iotlb_offset = iotlbOffset(); + write64(iotlb_offset, iotlb_ivt | iotlb_iirg_global | iotlb_dr | iotlb_dw); + spin64(iotlb_offset, iotlb_ivt); +} + +fn invalidateContextDevice(bdf: u16, domain: u16) void { + write64(reg_ccmd, ccmd_icc | ccmd_cirg_device | (@as(u64, bdf) << 16) | domain); + spin64(reg_ccmd, ccmd_icc); +} + +fn iotlbOffset() usize { + const iro = (extended_capabilities >> ecap_iro_shift) & 0x3FF; + return @as(usize, @intCast(iro)) * 16 + 8; // IOTLB register sits at IRO*16 + 8 +} + +fn allocZeroed() ?u64 { + const frame = pmm.alloc() orelse return null; + const table = tableAt(frame); + var i: usize = 0; + while (i < 512) : (i += 1) table[i] = 0; + return frame; +} diff --git a/system/kernel/iommu.zig b/system/kernel/iommu.zig new file mode 100644 index 0000000..6e64f33 --- /dev/null +++ b/system/kernel/iommu.zig @@ -0,0 +1,403 @@ +//! system/kernel/iommu.zig — vendor-neutral IOMMU core: per-device DMA translation +//! domains over an Intel VT-d or AMD-Vi backend. +//! +//! The problem this closes: without an IOMMU, a claimed bus-mastering device can DMA to +//! ANY physical address, so a compromised or buggy driver reaches all of memory through +//! its device — driver isolation stops at the CPU's MMU. This core gives each claimed +//! PCI function its own translation domain; a device reaches only the physical ranges +//! mapped into its domain, and nothing else (kernel, page tables, other processes) is +//! visible to it. +//! +//! Design: +//! - **Identity mappings** (IOVA == physical). `dma_alloc` already hands drivers the +//! physical address they program into hardware; a domain simply makes that same +//! address the ONLY thing the device can reach. No IOVA allocator, and every +//! driver's register-programming code is untouched. +//! - **Vendor-neutral**: this file owns the domain table and a shared 512-entry +//! page-table walker; a `Backend` vtable supplies the hardware specifics (VT-d in +//! iommu-intel.zig, AMD-Vi in iommu-amd.zig) — the entry-bit encodings, the +//! enable/invalidate register dances, and the fault drain. +//! - **Fail-open**: when no IOMMU is found, `kind == .none` and every entry point is a +//! success no-op, so callers in process.zig stay unconditional and behavior is +//! byte-for-byte the pre-IOMMU kernel. The boot log states the posture. +//! +//! All entry points run under the big kernel lock (the caller holds it); no internal +//! locking. All memory comes from `pmm` reached through the physmap, like paging.zig. + +const std = @import("std"); +const abi = @import("abi"); +const boot_handoff = @import("boot-handoff"); +const pmm = @import("pmm.zig"); +const platform = @import("platform"); +const devices_broker = @import("devices-broker.zig"); +const log = @import("log.zig"); +const intel = @import("iommu-intel.zig"); + +const page_size: u64 = abi.page_size; +const page_mask: u64 = page_size - 1; +const huge_page_size: u64 = 2 * 1024 * 1024; + +pub const Kind = enum { none, intel_vtd, amd_vi }; + +/// One domain per claimed PCI function. 64 mirrors devices-broker's device cap. +pub const maximum_domains = 64; +pub const invalid_domain: u16 = 0xFFFF; + +/// The bit encodings and hardware operations a backend supplies to the shared core. +/// Entry helpers build the raw page-table entries for the backend's format; the core +/// walks the tree with them. The hardware ops act on a whole domain (identified by its +/// hardware domain id = core index + 1) or device (by requester id / bdf). +pub const Backend = struct { + /// Number of page-table levels (3 or 4) the backend selected from hardware caps. + levels: u8, + /// Largest leaf the walker may emit: 4 KiB always, 2 MiB when the backend allows. + supports_huge_pages: bool, + + /// Raw entry bits for a leaf mapping `physical` (with the given size), and for a + /// non-leaf entry pointing at `table_physical` at `level` (level counts down to 1 + /// at the leaf's parent). `isPresent` tests a read-back entry. + makeLeaf: *const fn (physical: u64, huge: bool) u64, + makeTable: *const fn (table_physical: u64, level: u8) u64, + isPresent: *const fn (entry: u64) bool, + /// Flush a cache line holding IOMMU structures the hardware reads non-coherently + /// (VT-d with ECAP.C==0). A no-op where the unit snoops caches. + flushStructure: *const fn (address: usize) void, + + /// Point `bdf`'s translation structure at `domain` (hardware id) and invalidate the + /// context/device caches so the change takes effect. + attach: *const fn (bdf: u16, domain: u16, page_table_root: u64) void, + /// Return `bdf`'s translation structure to not-present + invalidate — all its DMA + /// faults afterward. + detach: *const fn (bdf: u16) void, + /// Invalidate cached translations for `domain` (after a map or unmap). + invalidateDomain: *const fn (domain: u16) void, + /// Pull pending faults out of the hardware, log them (rate-limited), return the + /// count seen this call. + faultDrain: *const fn () usize, +}; + +const Domain = struct { + in_use: bool = false, + owner: u32 = 0, // task that owns the attached device + bdf: u16 = 0, // requester id of the attached device + page_table_root: u64 = 0, // physical address of the top-level table + rmrr: bool = false, // a firmware reserved-region domain (persists across claims) +}; + +var kind: Kind = .none; +var backend: Backend = undefined; +var domains: [maximum_domains]Domain = .{Domain{}} ** maximum_domains; + +pub fn kindOf() Kind { + return kind; +} +pub fn enabled() bool { + return kind != .none; +} + +/// Detect the IOMMU, pick a backend, pre-map firmware reserved regions, and enable +/// translation. Fail-open (kind stays .none) when no unit exists — the caller logs the +/// posture. Must run after platform discovery and before any user process starts. +pub fn init() void { + const info = platform.platformInformation(); + if (!info.iommu_present) { + kind = .none; + return; + } + // Only Intel VT-d for now; AMD-Vi (iommu-amd.zig) selects here when its detection + // (IVRS) lands. A present-but-unsupported unit stays fail-open with a logged reason. + if (intel.detect(info)) |be| { + backend = be; + kind = .intel_vtd; + } else { + kind = .none; + log.write("/system/kernel: WARNING IOMMU present but unusable — staying fail-open\n"); + return; + } + + // L1 (enabled-but-invisible): every device is placed in a single blanket domain + // that identity-maps all of RAM plus the firmware reserved regions, so translation + // is on but nothing's DMA changes. L2 replaces this per device: on claim a device + // is detached from the blanket and attached to its own empty domain, so it reaches + // only what is explicitly mapped for it. + buildBlanketDomain(info); + intel.enable(); + logEnabled(info); +} + +/// The shared identity domain claimed devices are placed in (L1). L2 replaces this with +/// a private empty domain per device plus explicitly-granted mappings. +var blanket: u16 = invalid_domain; + +/// PCI functions are enumerated post-boot by the ring-3 pci-bus driver, so at init the +/// device tree has none — a device is attached to the blanket when its driver *claims* +/// it (confineDevice), which is always before the driver programs any DMA. Until then +/// its context entry is not-present and its DMA faults (only stale firmware bus- +/// mastering would hit that, which is the evidence M16 exists to surface). +fn buildBlanketDomain(info: platform.PlatformInformation) void { + const domain = domainCreate(0, 0) orelse return; + blanket = domain; + + // Identity-map all of physical RAM (2 MiB leaves keep the table small even on a + // 64 GiB machine), then each firmware reserved region in case it sits outside the + // RAM extent (RMRRs must stay reachable under translation). + const top_of_ram = @as(u64, pmm.stats().total_frames) * page_size; + _ = map(domain, 0, top_of_ram); + var i: usize = 0; + while (i < info.rmrr_count) : (i += 1) { + const region = info.rmrr[i]; + _ = map(domain, region.base, region.limit - region.base + 1); + } +} + +/// Per-claimed-device record, so a driver's death detaches exactly the devices it held. +const Confined = struct { active: bool = false, owner: u32 = 0, bdf: u16 = 0 }; +var confined: [maximum_domains]Confined = .{Confined{}} ** maximum_domains; + +/// Place a just-claimed PCI function under IOMMU translation on behalf of `owner`. L1: +/// attach it to the blanket identity domain (its DMA works, but through real second- +/// level walks). false only if the machinery is unexpectedly unavailable — the caller +/// rolls the claim back. No-op success when no IOMMU exists (fail-open). +pub fn confineDevice(device_id: u64, bdf: u16, owner: u32) bool { + if (kind == .none) return true; + if (blanket == invalid_domain) return false; + if (device_id >= confined.len) return true; // unusual id; leave it to fail-open + attachDevice(blanket, bdf); + confined[@intCast(device_id)] = .{ .active = true, .owner = owner, .bdf = bdf }; + return true; +} + +/// A driver died or released its devices: detach every device it held so their DMA is +/// blocked again (a restarted driver re-claims and re-confines). Runs BEFORE the frames +/// and broker claims are released. +pub fn releaseAllOwnedBy(owner: u32) void { + if (kind == .none) return; + for (&confined) |*c| { + if (c.active and c.owner == owner) { + detachDevice(c.bdf); + c.* = .{}; + } + } + _ = faultDrain(); // log any faults a mid-DMA device raised as it was cut off +} + +/// Allocate an empty domain (an empty top-level table). null when the table is full. +pub fn domainCreate(owner: u32, bdf: u16) ?u16 { + if (kind == .none) return 0; // fail-open: a dummy id the no-op ops ignore + for (&domains, 0..) |*d, index| { + if (d.in_use) continue; + const root = allocTable() orelse return null; + d.* = .{ .in_use = true, .owner = owner, .bdf = bdf, .page_table_root = root }; + return @intCast(index); + } + return null; +} + +/// Free a domain's page-table frames and its slot. Precondition: no device attached +/// (detach first). +pub fn domainDestroy(domain: u16) void { + if (kind == .none) return; + const d = &domains[domain]; + if (!d.in_use) return; + freeTables(d.page_table_root, backend.levels); + d.* = .{}; +} + +/// Attach `bdf`'s device to `domain` and pre-load any RMRR range recorded for it. +pub fn attachDevice(domain: u16, bdf: u16) void { + if (kind == .none) return; + const d = &domains[domain]; + d.bdf = bdf; + backend.attach(bdf, hardwareId(domain), d.page_table_root); +} + +/// Return `bdf`'s device to not-present + invalidate. +pub fn detachDevice(bdf: u16) void { + if (kind == .none) return; + backend.detach(bdf); +} + +/// Identity-map [physical, physical+len) into `domain` (read+write) and invalidate. +/// Unconditional domain-selective invalidation after every map — correct under VT-d +/// caching-mode and free otherwise. +pub fn map(domain: u16, physical: u64, len: u64) bool { + if (kind == .none) return true; + const d = &domains[domain]; + if (!d.in_use) return false; + if (!mapRange(d.page_table_root, physical, len)) return false; + backend.invalidateDomain(hardwareId(domain)); + return true; +} + +/// Unmap [physical, physical+len) from `domain` and invalidate. MUST finish its +/// invalidation before the caller returns the frames to pmm — a stale IOTLB entry +/// pointing at a reallocated frame is the use-after-free this ordering prevents. +pub fn unmap(domain: u16, physical: u64, len: u64) void { + if (kind == .none) return; + const d = &domains[domain]; + if (!d.in_use) return; + unmapRange(d.page_table_root, physical, len); + backend.invalidateDomain(hardwareId(domain)); +} + +/// Poll the hardware for translation faults, log them, return the count. Called by the +/// IOMMU test case and opportunistically after a device detaches. +pub fn faultDrain() usize { + if (kind == .none) return 0; + return backend.faultDrain(); +} + +/// The physical address `virtual` maps to in `domain`, or null if unmapped — a test +/// helper that walks the domain's page tables (identity mappings return `virtual`). +pub fn translationOf(domain: u16, virtual: u64) ?u64 { + if (kind == .none) return virtual; + const d = &domains[domain]; + if (!d.in_use) return null; + var table = d.page_table_root; + var level = backend.levels; + while (level > 1) : (level -= 1) { + const entry = tableAt(table)[indexAt(virtual, level)]; + if (!backend.isPresent(entry)) return null; + if (level == 2 and isHugeLeaf(entry)) + return (entry & address_mask) | (virtual & (huge_page_size - 1)); + table = entry & address_mask; + } + const leaf = tableAt(table)[indexAt(virtual, 1)]; + if (!backend.isPresent(leaf)) return null; + return (leaf & address_mask) | (virtual & page_mask); +} + +// --- the shared page-table walker ----------------------------------------------------- +// 512-entry, 9-bits-per-level, 4 KiB tables reached through the physmap — the shape both +// VT-d second-level and AMD-Vi native tables share. The backend supplies the entry bits. + +fn tableAt(physical: u64) [*]volatile u64 { + return @ptrFromInt(boot_handoff.physicalToVirtual(physical)); +} + +fn allocTable() ?u64 { + const frame = pmm.alloc() orelse return null; + const table = tableAt(frame); + var i: usize = 0; + while (i < 512) : (i += 1) table[i] = 0; + return frame; +} + +const address_mask: u64 = 0x000F_FFFF_FFFF_F000; + +fn indexAt(virtual: u64, level: u8) usize { + // level 1 is the leaf table; shift = 12 + 9*(level-1). + const shift: u6 = @intCast(12 + 9 * (@as(u32, level) - 1)); + return @intCast((virtual >> shift) & 0x1FF); +} + +/// Descend to (allocating) the next-level table below `entry_ptr`, returning its +/// physical base. null on out-of-memory. +fn descend(entry_ptr: *volatile u64, level: u8) ?u64 { + const entry = entry_ptr.*; + if (backend.isPresent(entry)) return entry & address_mask; + const table = allocTable() orelse return null; + backend.flushStructure(@intFromPtr(tableAt(table))); + entry_ptr.* = backend.makeTable(table, level); + backend.flushStructure(@intFromPtr(entry_ptr)); + return table; +} + +fn mapRange(root: u64, physical: u64, len: u64) bool { + const start = physical & ~page_mask; + const end = (physical + len + page_mask) & ~page_mask; + var addr = start; + while (addr < end) { + // 2 MiB leaf when the backend allows it and both address and remaining span are + // huge-aligned — keeps table memory sane for the blanket-identity and real-PC + // cases without a separate superpage path per backend. + const huge = backend.supports_huge_pages and + addr % huge_page_size == 0 and (end - addr) >= huge_page_size; + if (!mapOne(root, addr, huge)) return false; + addr += if (huge) huge_page_size else page_size; + } + return true; +} + +fn mapOne(root: u64, addr: u64, huge: bool) bool { + const leaf_level: u8 = if (huge) 2 else 1; + var table = root; + var level = backend.levels; + while (level > leaf_level) : (level -= 1) { + const entry_ptr = &tableAt(table)[indexAt(addr, level)]; + table = descend(entry_ptr, level) orelse return false; + } + const leaf_ptr = &tableAt(table)[indexAt(addr, leaf_level)]; + leaf_ptr.* = backend.makeLeaf(addr, huge); + backend.flushStructure(@intFromPtr(leaf_ptr)); + return true; +} + +fn unmapRange(root: u64, physical: u64, len: u64) void { + const start = physical & ~page_mask; + const end = (physical + len + page_mask) & ~page_mask; + var addr = start; + while (addr < end) { + const huge = backend.supports_huge_pages and + addr % huge_page_size == 0 and (end - addr) >= huge_page_size; + unmapOne(root, addr, huge); + addr += if (huge) huge_page_size else page_size; + } +} + +fn unmapOne(root: u64, addr: u64, huge: bool) void { + const leaf_level: u8 = if (huge) 2 else 1; + var table = root; + var level = backend.levels; + while (level > leaf_level) : (level -= 1) { + const entry = tableAt(table)[indexAt(addr, level)]; + if (!backend.isPresent(entry)) return; // nothing mapped here + table = entry & address_mask; + } + const leaf_ptr = &tableAt(table)[indexAt(addr, leaf_level)]; + leaf_ptr.* = 0; + backend.flushStructure(@intFromPtr(leaf_ptr)); +} + +/// Post-order free of a domain's whole table tree. +fn freeTables(root: u64, level: u8) void { + if (level > 1) { + const table = tableAt(root); + var i: usize = 0; + while (i < 512) : (i += 1) { + const entry = table[i]; + if (!backend.isPresent(entry)) continue; + // A 2 MiB leaf sits at level 2 and points at RAM, not a sub-table. + if (level == 2 and isHugeLeaf(entry)) continue; + freeTables(entry & address_mask, level - 1); + } + } + pmm.free(root); +} + +fn isHugeLeaf(entry: u64) bool { + // Both backends set a page-size bit (VT-d bit 7, AMD leaf next-level=0 at level 2). + // The backend's makeLeaf encodes it; the walker only needs "is this a leaf, not a + // pointer" at level 2, which huge leaves are by construction. + return entry & huge_leaf_bit != 0; +} + +/// The size-bit the backends set on a 2 MiB leaf (VT-d SL-PTE PS bit 7; AMD encodes a +/// leaf as next-level 0, so the core marks huge leaves with this software bit — an +/// ignored bit in both formats — to tell them apart from table pointers when freeing). +pub const huge_leaf_bit: u64 = 1 << 7; + +fn hardwareId(domain: u16) u16 { + return domain + 1; // id 0 is reserved by both architectures +} + +fn logEnabled(info: platform.PlatformInformation) void { + log.write("/system/kernel: iommu online (Intel VT-d)\n"); + log.print(" version : 0x{x}\n", .{info.iommu_version}); + log.print(" agaw : {d} levels\n", .{backend.levels}); + log.print(" rmrr : {d} region(s) premapped\n", .{info.rmrr_count}); + if (info.rmrr_skipped > 0) + log.print(" rmrr : WARNING {d} scope(s) skipped — a device keeps an unmapped firmware buffer\n", .{info.rmrr_skipped}); + if (info.iommu_extra_units > 0) + log.print(" units : WARNING {d} other DRHD(s) — their scoped devices are NOT translated\n", .{info.iommu_extra_units}); +} diff --git a/system/kernel/kernel.zig b/system/kernel/kernel.zig index 6bfa459..b9e1e29 100644 --- a/system/kernel/kernel.zig +++ b/system/kernel/kernel.zig @@ -13,6 +13,7 @@ const sync = @import("sync.zig"); const process = @import("process.zig"); const devices_broker = @import("devices-broker.zig"); const irq = @import("irq.zig"); +const iommu = @import("iommu.zig"); const platform = @import("platform"); const tests = @import("tests.zig"); const build_options = @import("build_options"); @@ -215,6 +216,12 @@ fn kmain(boot_information: *const BootInformation) noreturn { // land on. Every line stays masked until something binds it (ioapic.init). irq.init(); + // Bring up DMA translation: build the IOMMU domains and enable it (or record + // fail-open when no unit exists). Must run before any driver claims a device — + // an unclaimed device's DMA is blocked once translation is on. Logs its own + // enable block; the fail-open posture is stated in the platform block below. + iommu.init(); + // Power register map, from the FADT (the SLP_TYP sleep values live in AML, // which the kernel doesn't parse — the ring-3 acpi service owns soft-off). const pw = platform.powerInformation(); @@ -272,6 +279,11 @@ fn kmain(boot_information: *const BootInformation) noreturn { log.print(" cpus : {d} usable core(s); 1 running (BSP), {d} AP(s) parked (SMP bring-up pending)\n", .{ cores.len, if (cores.len > 0) cores.len - 1 else 0 }); if (platform.cpusDropped() > 0) log.print(" cpus : WARNING {d} core(s) beyond pool cap dropped\n", .{platform.cpusDropped()}); + // The DMA-isolation posture, stated plainly at every boot. When a unit exists, + // iommu.init already logged its enable block above; here we only state the + // fail-open case, so a boot without the line is a boot with translation on. + if (!iommu.enabled()) + log.write(" iommu : none present - DMA fail-open (unisolated)\n"); } else |err| { log.print("\n/system/kernel: device discovery failed: {s}\n", .{@errorName(err)}); } diff --git a/system/kernel/process.zig b/system/kernel/process.zig index bff36a2..f434ff5 100644 --- a/system/kernel/process.zig +++ b/system/kernel/process.zig @@ -33,6 +33,7 @@ const sync = @import("sync.zig"); const ipc = @import("ipc-synchronous.zig"); const devices_broker = @import("devices-broker.zig"); const irq = @import("irq.zig"); +const iommu = @import("iommu.zig"); const initial_ramdisk = @import("initial-ramdisk"); const vfs = @import("vfs.zig"); const log = @import("log.zig"); @@ -390,6 +391,17 @@ fn systemDeviceClaim(state: *architecture.CpuState) void { const claim_flags = sync.enter(); defer sync.leave(claim_flags); if (devices_broker.claim(device_id, scheduler.current().id)) { + // Confine the device's DMA before the driver can program it: a PCI function + // becomes reachable to the IOMMU only once claimed (until now its DMA is + // blocked). A claim that cannot be confined must not stand — roll it back — + // since the whole point is that claiming a DMA device is no longer equivalent + // to ring 0. No-op when no IOMMU exists (fail-open). + if (devices_broker.pciAddressOf(device_id)) |bdf| { + if (!iommu.confineDevice(device_id, bdf, scheduler.current().id)) { + _ = devices_broker.unclaim(device_id, scheduler.current().id); + return fail(state); + } + } // A display service just took the framebuffer — quiesce the bootstrap console // so the kernel and the service don't scribble over each other's pixels. The // claim releases (and the console resumes) automatically if the service dies; @@ -991,6 +1003,11 @@ pub var fault_kill_count: u64 = 0; fn releaseTaskResourcesLocked(t: *scheduler.Task) void { recordExitLocked(t); irq.releaseOwner(t.id); + // Detach the task's devices from their IOMMU domains BEFORE the broker clears the + // claims (the detach reads ownership) and before the address space is torn down and + // its DMA frames return to the allocator — a device must stop translating to a frame + // before that frame can be handed to someone else. + iommu.releaseAllOwnedBy(t.id); devices_broker.releaseAllOwnedBy(t.id); // If that dropped the framebuffer claim (this task was the display service), let the // bootstrap console draw again — the screen is nobody's now, so panics/status land. diff --git a/system/kernel/tests.zig b/system/kernel/tests.zig index c119cd9..1073f64 100644 --- a/system/kernel/tests.zig +++ b/system/kernel/tests.zig @@ -18,6 +18,7 @@ const wall_clock = @import("wall-clock.zig"); const devices_broker = @import("devices-broker.zig"); const platform = @import("platform"); const pmm = @import("pmm.zig"); +const iommu = @import("iommu.zig"); const heap = @import("heap.zig"); const scheduler = @import("scheduler.zig"); const ipc = @import("ipc.zig"); @@ -1166,6 +1167,11 @@ fn dmaTest() void { log("DANOS-TEST-BEGIN: dma\n", .{}); const base_free = pmm.stats().free_frames; + // This case boots WITHOUT an IOMMU device, so it is the explicit witness of the + // fail-open posture: no unit found, and the kernel said so at boot (the harness + // asserts the boot line; this check pins the recorded state to it). + check("no IOMMU present: DMA runs fail-open", !platform.platformInformation().iommu_present); + // A contiguous run: aligned, and it consumed exactly that many frames. const frames = 4; const phys = pmm.allocContiguous(frames, ~@as(u64, 0)) orelse { @@ -1236,9 +1242,10 @@ fn msiTest() void { /// IOMMU (M16): with an emulated VT-d unit present (the harness boots this case with /// `-device intel-iommu`), danos must find it in the ACPI DMAR table, map its register -/// block, and read back a real version. This is *detection*, the honest first step — -/// no translation domains are programmed yet, so DMA is still unprotected; enforcement -/// lands with the first DMA driver (docs/driver-model.md M16). +/// block, and read back a real version. Detection was M16's honest first step; the +/// IOVA-enforcement track extends this case milestone by milestone (translation +/// enabled, then per-device domains) — see the plan in docs and the fail-open witness +/// in dmaTest. fn iommuTest() void { log("DANOS-TEST-BEGIN: iommu\n", .{}); const pinfo = platform.platformInformation(); @@ -1246,6 +1253,27 @@ fn iommuTest() void { check("VT-d unit has a register base", pinfo.iommu_base != 0); check("VT-d version register reads back nonzero (real, mappable unit)", pinfo.iommu_version != 0); log("DANOS-IOMMU: base=0x{x} version=0x{x} capabilities=0x{x}\n", .{ pinfo.iommu_base, pinfo.iommu_version, pinfo.iommu_capabilities }); + + // Translation was enabled at boot (kernel.zig: iommu.init before any driver claims + // a device). The blanket domain keeps every device identity-mapped, so DMA still + // works, but the unit is live — and with only the boot-time mappings present, no + // device should have faulted yet. + check("IOMMU enabled (translation on)", iommu.enabled()); + check("no spurious translation faults at idle", iommu.faultDrain() == 0); + + // A scratch domain proves the walker + invalidation path end to end: create it, + // identity-map a page, and confirm the mapping resolves; then unmap and destroy. + if (iommu.domainCreate(0, 0)) |scratch| { + const scratch_phys: u64 = 0x0010_0000; // 1 MiB, page-aligned + check("map into a scratch domain succeeds", iommu.map(scratch, scratch_phys, abi.page_size)); + check("scratch domain resolves the mapping", iommu.translationOf(scratch, scratch_phys) == scratch_phys); + iommu.unmap(scratch, scratch_phys, abi.page_size); + check("scratch domain drops the mapping", iommu.translationOf(scratch, scratch_phys) == null); + iommu.domainDestroy(scratch); + } else { + check("scratch domain allocated", false); + } + log("DANOS-IOMMU: enabled base=0x{x} domains active\n", .{pinfo.iommu_base}); result(); } diff --git a/test/qemu_test.py b/test/qemu_test.py index 2798152..34ee48c 100644 --- a/test/qemu_test.py +++ b/test/qemu_test.py @@ -146,20 +146,44 @@ CASES = [ "fail": r"DANOS-TEST-RESULT: FAIL"}, # DMA memory (M14): contiguous frame allocation, below-4G cap, coherent mapping, # and reclaim on teardown. + # Boots with no IOMMU device, so it also asserts the explicit fail-open boot line + # (the DMA-isolation posture must be stated, never silent). {"name": "dma", - "expect": r"DANOS-TEST-RESULT: PASS", + "expect": r"(?s)(?=.*iommu : none present - DMA fail-open \(unisolated\))(?=.*DANOS-TEST-RESULT: PASS)", "fail": r"DANOS-TEST-RESULT: FAIL"}, # MSI (M15): allocate a per-device vector and deliver it as a notification (a # self-IPI stands in for the device's MSI write, since the HPET has no MSI). {"name": "msi", "expect": r"DANOS-TEST-RESULT: PASS", "fail": r"DANOS-TEST-RESULT: FAIL"}, - # IOMMU (M16): boot with an emulated VT-d unit and confirm danos parses the DMAR - # table and reads the unit's registers. Detection only — enforcement is future. + # IOMMU: boot with an emulated VT-d unit; danos parses the DMAR table, enables + # translation, and proves the domain walker (map/resolve/unmap on a scratch domain) + # with no spurious faults. The `enabled` line is a lookahead so a silently-dead unit + # cannot fake a pass. {"name": "iommu", "qemu_extra": ["-device", "intel-iommu,intremap=off"], - "expect": r"DANOS-TEST-RESULT: PASS", - "fail": r"DANOS-TEST-RESULT: FAIL"}, + "expect": r"(?s)(?=.*DANOS-IOMMU: enabled base=0x[0-9a-f]+ domains active)(?=.*DANOS-TEST-RESULT: PASS)", + "fail": r"DANOS-TEST-RESULT: FAIL|DANOS-IOMMU-FAULT"}, + # DMA under translation: the full USB storage stack (xHC ring DMA + BOT/SCSI + fat's + # cross-process bounce buffer) runs with VT-d enabled. Every device is identity- + # mapped in the blanket domain, so DMA works, but through real second-level walks. + {"name": "iommu-usb-storage", + "build_case": "fat-mount", + "smp": 4, + "timeout": 150, + "qemu_extra": ["-device", "intel-iommu,intremap=off"], + "expect": r"(?s)(?=.*/system/kernel: iommu online)(?=.*fat: mounted /mnt/usb)(?=.*fat-test: ok)", + "fail": r"DANOS-TEST-RESULT: FAIL|DANOS-IOMMU-FAULT"}, + # DMA + MSI under translation: interrupt-IN reports arrive through translated DMA and + # the xHC's MSI/MSI-X still delivers (the 0xFEE00000 interrupt window bypasses second- + # level translation with interrupt remapping off). + {"name": "iommu-usb-hid", + "build_case": "usb-hid", + "smp": 4, + "timeout": 150, + "qemu_extra": ["-device", "intel-iommu,intremap=off"], + "expect": r"(?s)(?=.*/system/kernel: iommu online)(?=.*usb-hid-keyboard: ok)(?=.*usb-hid-mouse: ok)", + "fail": r"DANOS-TEST-RESULT: FAIL|DANOS-IOMMU-FAULT"}, # Port I/O grants: a claimed device's io_port resource lets a driver read/write its # ports (PS/2 status 0x64), gated by the claim; out-of-range/unclaimed is refused. {"name": "ioport",