diff --git a/system/kernel/architecture/x86_64/iommu-amd.zig b/system/kernel/architecture/x86_64/iommu-amd.zig index f096905..f3fab5b 100644 --- a/system/kernel/architecture/x86_64/iommu-amd.zig +++ b/system/kernel/architecture/x86_64/iommu-amd.zig @@ -59,8 +59,15 @@ const command_opcode_shift = 60; // opcode in bits 63:60 of qword 0 const command_completion_wait: u64 = 0x01; const command_invalidate_devtab: u64 = 0x02; const command_invalidate_pages: u64 = 0x03; -const completion_wait_store: u64 = 1 << 1; // S: store `data` to the supplied address -const invalidate_pages_all: u64 = 0x000F_FFFF_FFFF_F000 | 1; // address bits 51:12 all-ones + S +// COMPLETION_WAIT qword 0: bit 0 is S (store `data` to the supplied address), bit 1 is +// I (raise an interrupt). An earlier revision set bit 1 and then blamed QEMU for the +// sentinel never landing — with I instead of S the IOMMU is never *asked* to store, on +// QEMU or on silicon, and completeAndWait was no barrier at all. +const completion_wait_store: u64 = 1 << 0; +// INVALIDATE_IOMMU_PAGES address qword, S=1: the architected invalidate-everything +// encoding is bits 62:12 all-ones (the spec's literal 0x7FFF_FFFF_FFFF_F000). The +// previous 51:12 value is a non-architected range real silicon is free to misread. +const invalidate_pages_all: u64 = 0x7FFF_FFFF_FFFF_F000 | 1; const levels: u8 = 4; // 48-bit IOVA, matching the Intel 4-level path @@ -228,11 +235,14 @@ fn submitCommand(qword0: u64, qword1: u64) void { } /// Append a COMPLETION_WAIT (store form) and spin until the IOMMU writes our sentinel to -/// the completion frame. QEMU consumes the command buffer synchronously on the tail- -/// register write, so by the time we poll the prior invalidation is already applied; the -/// store confirmation is belt-and-suspenders for real hardware. If it never lands -/// (QEMU's amd-iommu does not implement the store form), warn ONCE and proceed — the -/// invalidation itself has happened. +/// the completion frame. This is the driver's only ordering barrier: real hardware +/// fetches commands asynchronously, so a preceding invalidation has not happened until +/// this store lands — the callers that free DMA frames after an unmap depend on it. +/// (QEMU consumes the ring synchronously on the tail write and implements the store +/// form fine; the warning below once fired there only because the command carried the +/// I bit instead of S, so no store was ever requested.) If the store never lands, warn +/// ONCE and proceed rather than wedge the claim path — but on real silicon that line +/// means invalidations are unconfirmed and must be treated as a bug report. fn completeAndWait() void { const sentinel: u64 = 0xC0FFEE; ram(completion_frame)[0] = 0; @@ -246,7 +256,7 @@ fn completeAndWait() void { if (spins > 100_000) { if (!completion_warned) { completion_warned = true; - iommu.environment.write("/system/kernel: AMD-Vi COMPLETION_WAIT store not observed — proceeding (QEMU processes commands synchronously)\n"); + iommu.environment.write("/system/kernel: AMD-Vi COMPLETION_WAIT store not observed — proceeding WITHOUT an invalidation barrier\n"); } return; }