Compare commits
6
Commits
e376c9e908
...
feat/igpu
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
198767bb30 | ||
|
|
6a687fbc2b | ||
|
|
4e7cbc9792 | ||
|
|
e94adcfc02 | ||
|
|
f477ef7d9f | ||
|
|
9e649178bf |
@@ -531,15 +531,18 @@ pub fn build(b: *std.Build) void {
|
||||
},
|
||||
});
|
||||
// A device driver's view of its claimed PCI function: config-space header fields, BAR
|
||||
// decode + map, and the capability walk (library/device/pci/pci.zig). The generic PCI
|
||||
// decode + map, capability walks (legacy + extended), MSI/MSI-X programming, power
|
||||
// state, and function-level reset (library/device/pci/pci.zig). The generic PCI
|
||||
// mechanics every leaf PCI driver used to re-derive inline. Imports the driver (device
|
||||
// access) client + mmio + the pci-class data module (config-space layout constants).
|
||||
// access) client + mmio + the pci-class data module (config-space layout constants) +
|
||||
// time (the D0 and FLR settle delays).
|
||||
const pci_module = b.addModule("pci", .{
|
||||
.root_source_file = b.path("library/device/pci/pci.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "driver", .module = driver_module },
|
||||
.{ .name = "mmio", .module = mmio_module },
|
||||
.{ .name = "pci-class", .module = pci_class_module },
|
||||
.{ .name = "time", .module = time_module },
|
||||
},
|
||||
});
|
||||
|
||||
@@ -677,6 +680,8 @@ pub fn build(b: *std.Build) void {
|
||||
programModule(ps2_mouse_exe).addImport("input-protocol", input_protocol_module);
|
||||
const usb_xhci_bus_exe = addUserBinary(b, kernel_target, &default_imports, "usb-xhci-bus", "system/drivers/usb-xhci-bus/usb-xhci-bus.zig");
|
||||
programModule(usb_xhci_bus_exe).addImport("device-manager-protocol", device_manager_protocol_module);
|
||||
// MSI setup: the claimed-function view (enable bits + MSI capability programming).
|
||||
programModule(usb_xhci_bus_exe).addImport("pci", pci_module);
|
||||
// The xHCI bus driver builds chapter-9 requests and decodes descriptors from
|
||||
// usb-abi, and reports each interface's (class,subclass,protocol) identity via
|
||||
// usb-ids.packTriple.
|
||||
@@ -727,6 +732,16 @@ pub fn build(b: *std.Build) void {
|
||||
programModule(crash_test_exe).addImport("device-manager-protocol", device_manager_protocol_module);
|
||||
const device_list_exe = addUserBinary(b, kernel_target, &default_imports, "device-list", "test/system/services/device-list/device-list.zig");
|
||||
programModule(device_list_exe).addImport("device-manager-protocol", device_manager_protocol_module);
|
||||
// A test fixture: claims the pci-caps case's extra unclaimed NIC and exercises the
|
||||
// driver-side PCI library surface (capabilities, MSI, MSI-X, power, FLR) against it.
|
||||
const pci_cap_test_exe = addUserBinary(b, kernel_target, &default_imports, "pci-cap-test", "test/system/services/pci-cap-test/pci-cap-test.zig");
|
||||
programModule(pci_cap_test_exe).addImport("pci", pci_module);
|
||||
programModule(pci_cap_test_exe).addImport("pci-class", pci_class_module);
|
||||
// The IOMMU-enforcement negative test: claims an unclaimed e1000e and fires a rogue
|
||||
// DMA that VT-d must fault. Same PCI building blocks as pci-cap-test.
|
||||
const iommu_fault_test_exe = addUserBinary(b, kernel_target, &default_imports, "iommu-fault-test", "test/system/services/iommu-fault-test/iommu-fault-test.zig");
|
||||
programModule(iommu_fault_test_exe).addImport("pci", pci_module);
|
||||
programModule(iommu_fault_test_exe).addImport("pci-class", pci_class_module);
|
||||
// The discovery service: one swappable process per firmware
|
||||
// (docs/discovery.md), bundled under the neutral ramdisk name
|
||||
// "discovery" so the device manager never learns which firmware it is on.
|
||||
@@ -811,6 +826,8 @@ pub fn build(b: *std.Build) void {
|
||||
.{ .path = "test/system/services/shared-memory-client", .binary = shared_memory_client_exe.getEmittedBin() },
|
||||
.{ .path = "test/system/services/crash-test", .binary = crash_test_exe.getEmittedBin() },
|
||||
.{ .path = "test/system/services/device-list", .binary = device_list_exe.getEmittedBin() },
|
||||
.{ .path = "test/system/services/pci-cap-test", .binary = pci_cap_test_exe.getEmittedBin() },
|
||||
.{ .path = "test/system/services/iommu-fault-test", .binary = iommu_fault_test_exe.getEmittedBin() },
|
||||
.{ .path = "test/system/services/input-source", .binary = input_source_exe.getEmittedBin() },
|
||||
.{ .path = "test/system/services/input-test", .binary = input_test_exe.getEmittedBin() },
|
||||
.{ .path = "test/system/services/args-echo", .binary = args_echo_exe.getEmittedBin() },
|
||||
|
||||
+3
-1
@@ -55,7 +55,9 @@ rather than restate it. Roughly in the order things happen at runtime:
|
||||
15. **[drivers.md](device-driver-development/drivers.md) — writing a driver.** The payoff: a driver is an
|
||||
ordinary ring-3 process that claims a device, maps its registers, and **sleeps
|
||||
until its hardware interrupts it**. The claim is the capability; `irq_ack` is the
|
||||
unmask.
|
||||
unmask. The condensed version: the
|
||||
[new-driver checklist](device-driver-development/new-driver-checklist.md) —
|
||||
the minimum steps from boot-log line to mapped registers.
|
||||
16. **[driver-model.md](device-driver-development/driver-model.md) — buses, classes and host controllers.** How
|
||||
real driver stacks factor into three shapes and how families share code. The
|
||||
three primitives it proposed are long since built (M13 capability passing,
|
||||
|
||||
@@ -4,7 +4,9 @@ In a monolithic kernel a driver is a function call away from everything: it runs
|
||||
ring 0, dereferences any physical address, and its interrupt handler *is* the ISR. In
|
||||
danos a driver is **an ordinary ring-3 process**. It has its own address space, it
|
||||
can crash without taking the kernel with it, and — the point of this document — it
|
||||
can be restarted ([resilience](../os-development/resilience.md)).
|
||||
can be restarted ([resilience](../os-development/resilience.md)). This document is
|
||||
the reasoning; the condensed do-this-then-that version is the
|
||||
[new-driver checklist](new-driver-checklist.md).
|
||||
|
||||
That leaves three questions the kernel has to answer, because a process can't answer
|
||||
them for itself:
|
||||
|
||||
@@ -0,0 +1,339 @@
|
||||
Vendor ID,Device ID,Graphics Family,GPU Name
|
||||
0x8086,0x0152,HD Graphics,Xeon E3-1200 v2/3rd Gen Core GT1
|
||||
0x8086,0x0155,HD Graphics,Xeon E3-1200 v2/3rd Gen Core
|
||||
0x8086,0x0156,HD Graphics,Ivy Bridge mobile GT1
|
||||
0x8086,0x0157,HD Graphics,Ivy Bridge mobile GT1
|
||||
0x8086,0x015a,HD Graphics,Xeon E3-1200 v2/Ivy Bridge
|
||||
0x8086,0x0162,HD Graphics,Ivy Bridge GT2 (HD Graphics 4000)
|
||||
0x8086,0x0166,HD Graphics,Ivy Bridge mobile GT2 (HD Graphics 4000)
|
||||
0x8086,0x016a,HD Graphics,Xeon E3-1200 v2/3rd Gen Core GT3
|
||||
0x8086,0x0402,HD Graphics,Xeon E3-1200 v3/4th Gen Core GT1
|
||||
0x8086,0x0406,HD Graphics,Haswell GT1
|
||||
0x8086,0x040a,HD Graphics,Xeon E3-1200 v3 GT1
|
||||
0x8086,0x040b,HD Graphics,Haswell GT1
|
||||
0x8086,0x040e,HD Graphics,Haswell GT1
|
||||
0x8086,0x0412,HD Graphics,Xeon E3-1200 v3/4th Gen Core GT2
|
||||
0x8086,0x0416,HD Graphics,4th Gen Core GT2
|
||||
0x8086,0x041a,HD Graphics,Xeon E3-1200 v3 GT2
|
||||
0x8086,0x041b,HD Graphics,Haswell GT2
|
||||
0x8086,0x041e,HD Graphics,4th Gen Core Family GT2
|
||||
0x8086,0x0422,HD Graphics,Haswell GT3
|
||||
0x8086,0x0426,HD Graphics,Haswell GT3
|
||||
0x8086,0x042a,HD Graphics,Haswell GT3
|
||||
0x8086,0x042b,HD Graphics,Haswell GT3
|
||||
0x8086,0x042e,HD Graphics,Haswell GT3
|
||||
0x8086,0x0a02,HD Graphics,Haswell-ULT GT1
|
||||
0x8086,0x0a06,HD Graphics,Haswell-ULT GT1
|
||||
0x8086,0x0a0a,HD Graphics,Haswell-ULT GT1
|
||||
0x8086,0x0a0b,HD Graphics,Haswell-ULT GT1
|
||||
0x8086,0x0a0e,HD Graphics,Haswell-ULT GT1
|
||||
0x8086,0x0a12,HD Graphics,Haswell-ULT GT2
|
||||
0x8086,0x0a16,HD Graphics,Haswell-ULT GT2
|
||||
0x8086,0x0a1a,HD Graphics,Haswell-ULT GT2
|
||||
0x8086,0x0a1b,HD Graphics,Haswell-ULT GT2
|
||||
0x8086,0x0a1e,HD Graphics,Haswell-ULT GT2
|
||||
0x8086,0x0a22,HD Graphics,Haswell-ULT GT2
|
||||
0x8086,0x0a26,HD Graphics,Haswell-ULT GT2
|
||||
0x8086,0x0a2a,HD Graphics,Haswell-ULT GT2
|
||||
0x8086,0x0a2b,HD Graphics,Haswell-ULT GT2
|
||||
0x8086,0x0a2e,HD Graphics,Haswell-ULT GT2
|
||||
0x8086,0x0d02,Iris Pro Graphics,Crystal Well GT1
|
||||
0x8086,0x0d06,Iris Pro Graphics,Crystal Well GT2
|
||||
0x8086,0x0d0a,Iris Pro Graphics,Crystal Well GT2
|
||||
0x8086,0x0d0b,Iris Pro Graphics,Crystal Well GT2
|
||||
0x8086,0x0d0e,Iris Pro Graphics,Crystal Well GT2
|
||||
0x8086,0x0d12,Iris Pro Graphics,Crystal Well GT3 (Iris Pro 5200)
|
||||
0x8086,0x0d16,Iris Pro Graphics,Crystal Well GT3
|
||||
0x8086,0x0d1a,Iris Pro Graphics,Crystal Well GT3
|
||||
0x8086,0x0d1b,Iris Pro Graphics,Crystal Well GT3
|
||||
0x8086,0x0d1e,Iris Pro Graphics,Crystal Well GT3
|
||||
0x8086,0x0d22,Iris Pro Graphics,Crystal Well (Iris Pro 5200)
|
||||
0x8086,0x0d26,Iris Pro Graphics,Crystal Well GT3
|
||||
0x8086,0x0d2a,Iris Pro Graphics,Crystal Well GT3
|
||||
0x8086,0x0d2b,Iris Pro Graphics,Crystal Well GT3
|
||||
0x8086,0x0d2e,Iris Pro Graphics,Crystal Well GT3
|
||||
0x8086,0x0d32,Iris Pro Graphics,Crystal Well GT3
|
||||
0x8086,0x0d36,Iris Pro Graphics,Crystal Well GT3
|
||||
0x8086,0x0d3a,Iris Pro Graphics,Crystal Well GT3
|
||||
0x8086,0x1602,HD Graphics,Broadwell-U GT1
|
||||
0x8086,0x1606,HD Graphics,Broadwell-U GT1
|
||||
0x8086,0x160a,HD Graphics,Broadwell-U GT1
|
||||
0x8086,0x160b,HD Graphics,Broadwell-U GT1
|
||||
0x8086,0x160d,HD Graphics,Broadwell-U GT1
|
||||
0x8086,0x160e,HD Graphics,Broadwell-U GT1
|
||||
0x8086,0x1612,HD Graphics,Broadwell-H GT2 (HD Graphics 5600)
|
||||
0x8086,0x1616,HD Graphics,Broadwell-U GT2 (HD Graphics 5500)
|
||||
0x8086,0x161a,HD Graphics,Broadwell-U GT2
|
||||
0x8086,0x161b,HD Graphics,Broadwell-U GT2
|
||||
0x8086,0x161d,HD Graphics,Broadwell-U GT2
|
||||
0x8086,0x161e,HD Graphics,Broadwell-Y GT2 (HD Graphics 5300)
|
||||
0x8086,0x1622,Iris Pro Graphics,Broadwell-DT/H GT3 (Iris Pro 6200)
|
||||
0x8086,0x1626,HD Graphics,Broadwell-U GT3 (HD Graphics 6000)
|
||||
0x8086,0x162a,Iris Pro Graphics,Broadwell-DT GT3 (Iris Pro P6300)
|
||||
0x8086,0x162b,Iris Graphics,Broadwell-U GT3 (Iris 6100)
|
||||
0x8086,0x162d,HD Graphics,Broadwell-U GT3
|
||||
0x8086,0x162e,HD Graphics,Broadwell-U GT3
|
||||
0x8086,0x1632,HD Graphics,Broadwell-U GT3
|
||||
0x8086,0x1636,HD Graphics,Broadwell-U GT3
|
||||
0x8086,0x163a,HD Graphics,Broadwell-U GT3
|
||||
0x8086,0x163b,HD Graphics,Broadwell-U GT3
|
||||
0x8086,0x163d,HD Graphics,Broadwell-U GT3
|
||||
0x8086,0x163e,HD Graphics,Broadwell-U GT3
|
||||
0x8086,0x1902,HD Graphics,Skylake-S GT1 (HD Graphics 510)
|
||||
0x8086,0x1906,HD Graphics,Skylake-U GT1 (HD Graphics 510)
|
||||
0x8086,0x190a,HD Graphics,Skylake-DT GT1
|
||||
0x8086,0x190b,HD Graphics,Skylake GT1 (HD Graphics 510)
|
||||
0x8086,0x190e,HD Graphics,Skylake GT1
|
||||
0x8086,0x1912,HD Graphics,Skylake-S GT2 (HD Graphics 530)
|
||||
0x8086,0x1913,HD Graphics,Skylake GT2
|
||||
0x8086,0x1915,HD Graphics,Skylake GT2
|
||||
0x8086,0x1916,HD Graphics,Skylake-U GT2 (HD Graphics 520)
|
||||
0x8086,0x1917,HD Graphics,Skylake GT2
|
||||
0x8086,0x191a,HD Graphics,Skylake GT2
|
||||
0x8086,0x191b,HD Graphics,Skylake-H GT2 (HD Graphics 530)
|
||||
0x8086,0x191d,HD Graphics,Skylake-DT/H GT2 (HD Graphics P530)
|
||||
0x8086,0x191e,HD Graphics,Skylake-Y GT2 (HD Graphics 515)
|
||||
0x8086,0x1921,HD Graphics,Skylake GT2 (HD Graphics 520)
|
||||
0x8086,0x1923,HD Graphics,Skylake GT2 (HD Graphics 535)
|
||||
0x8086,0x1926,Iris Graphics,Skylake-U GT3 (Iris Graphics 540)
|
||||
0x8086,0x1927,Iris Graphics,Skylake-U GT3 (Iris Graphics 550)
|
||||
0x8086,0x192a,Iris Graphics,Skylake GT3
|
||||
0x8086,0x192b,Iris Graphics,Skylake GT3 (Iris Graphics 555)
|
||||
0x8086,0x192d,Iris Graphics,Skylake-H GT3 (Iris Graphics P555)
|
||||
0x8086,0x1932,Iris Pro Graphics,Skylake GT4 (Iris Pro 580)
|
||||
0x8086,0x193a,Iris Pro Graphics,Skylake-H GT4 (Iris Pro P580)
|
||||
0x8086,0x193b,Iris Pro Graphics,Skylake-H GT4 (Iris Pro 580)
|
||||
0x8086,0x193d,Iris Pro Graphics,Skylake-H GT4 (Iris Pro P580)
|
||||
0x8086,0x1a84,UHD Graphics,Skylake-DT
|
||||
0x8086,0x1a85,UHD Graphics,Skylake-DT
|
||||
0x8086,0x5902,HD Graphics,Kaby Lake-S GT1 (HD Graphics 610)
|
||||
0x8086,0x5906,HD Graphics,Kaby Lake-U GT1 (HD Graphics 610)
|
||||
0x8086,0x5908,HD Graphics,Kaby Lake GT1
|
||||
0x8086,0x590a,HD Graphics,Kaby Lake GT1
|
||||
0x8086,0x590b,HD Graphics,Kaby Lake GT1 (HD Graphics 610)
|
||||
0x8086,0x590e,HD Graphics,Kaby Lake GT1
|
||||
0x8086,0x5912,HD Graphics,Kaby Lake-S GT2 (HD Graphics 630)
|
||||
0x8086,0x5913,HD Graphics,Kaby Lake GT2
|
||||
0x8086,0x5915,HD Graphics,Kaby Lake GT2
|
||||
0x8086,0x5916,HD Graphics,Kaby Lake-U GT2 (HD Graphics 620)
|
||||
0x8086,0x5917,UHD Graphics,Kaby Lake-R GT2 (UHD Graphics 620)
|
||||
0x8086,0x591a,HD Graphics,Kaby Lake GT2
|
||||
0x8086,0x591b,HD Graphics,Kaby Lake-H GT2 (HD Graphics 630)
|
||||
0x8086,0x591c,UHD Graphics,Kaby Lake GT2 (UHD Graphics 615)
|
||||
0x8086,0x591d,HD Graphics,Kaby Lake-DT GT2 (HD Graphics P630)
|
||||
0x8086,0x591e,HD Graphics,Kaby Lake-Y GT2 (HD Graphics 615)
|
||||
0x8086,0x5921,HD Graphics,Kaby Lake GT2 (HD Graphics 620)
|
||||
0x8086,0x5923,HD Graphics,Kaby Lake GT2 (HD Graphics 635)
|
||||
0x8086,0x5926,Iris Plus Graphics,Kaby Lake-U GT3 (Iris Plus 640)
|
||||
0x8086,0x5927,Iris Plus Graphics,Kaby Lake-U GT3 (Iris Plus 650)
|
||||
0x8086,0x592a,Iris Plus Graphics,Kaby Lake GT3
|
||||
0x8086,0x592b,Iris Plus Graphics,Kaby Lake GT3
|
||||
0x8086,0x5932,Iris Plus Graphics,Kaby Lake GT3
|
||||
0x8086,0x593a,Iris Plus Graphics,Kaby Lake GT3
|
||||
0x8086,0x593b,Iris Plus Graphics,Kaby Lake GT3
|
||||
0x8086,0x593d,Iris Plus Graphics,Kaby Lake GT3
|
||||
0x8086,0x5a40,Intel Graphics,Apollolake
|
||||
0x8086,0x5a41,Intel Graphics,Apollolake
|
||||
0x8086,0x5a42,Intel Graphics,Apollolake
|
||||
0x8086,0x5a44,Intel Graphics,Apollolake
|
||||
0x8086,0x5a49,Intel Graphics,Apollolake
|
||||
0x8086,0x5a4a,Intel Graphics,Apollolake
|
||||
0x8086,0x5a4c,Intel Graphics,Apollolake
|
||||
0x8086,0x5a50,Intel Graphics,Apollolake
|
||||
0x8086,0x5a51,Intel Graphics,Apollolake
|
||||
0x8086,0x5a52,Intel Graphics,Apollolake
|
||||
0x8086,0x5a54,Intel Graphics,Apollolake
|
||||
0x8086,0x5a59,Intel Graphics,Apollolake
|
||||
0x8086,0x5a5a,Intel Graphics,Apollolake
|
||||
0x8086,0x5a5c,Intel Graphics,Apollolake
|
||||
0x8086,0x5a71,Intel Graphics,Apollolake
|
||||
0x8086,0x5a79,Intel Graphics,Apollolake
|
||||
0x8086,0x5a84,HD Graphics,Apollo Lake GT1 (HD Graphics 505)
|
||||
0x8086,0x5a85,HD Graphics,Apollo Lake GT1 (HD Graphics 500)
|
||||
0x8086,0x3184,UHD Graphics,GeminiLake (UHD Graphics 605)
|
||||
0x8086,0x3185,UHD Graphics,GeminiLake (UHD Graphics 600)
|
||||
0x8086,0x3e90,UHD Graphics,Coffee Lake-S GT1 (UHD Graphics 610)
|
||||
0x8086,0x3e91,UHD Graphics,Coffee Lake-S GT2 (UHD Graphics 630)
|
||||
0x8086,0x3e92,UHD Graphics,Coffee Lake-S GT2 (UHD Graphics 630)
|
||||
0x8086,0x3e93,UHD Graphics,Coffee Lake-S GT1 (UHD Graphics 610)
|
||||
0x8086,0x3e94,UHD Graphics,Coffee Lake-S GT2 (UHD Graphics P630)
|
||||
0x8086,0x3e96,UHD Graphics,Coffee Lake-S GT2 (UHD Graphics P630)
|
||||
0x8086,0x3e98,UHD Graphics,Coffee Lake-S GT2 (UHD Graphics 630)
|
||||
0x8086,0x3e99,UHD Graphics,Coffee Lake GT2
|
||||
0x8086,0x3e9a,UHD Graphics,Coffee Lake-S GT2 (UHD Graphics P630)
|
||||
0x8086,0x3e9b,UHD Graphics,Coffee Lake-H GT2 (UHD Graphics 630)
|
||||
0x8086,0x3e9c,UHD Graphics,Coffee Lake-S GT1 (UHD Graphics 610)
|
||||
0x8086,0x3ea0,UHD Graphics,Whiskey Lake-U GT2 (UHD Graphics 620)
|
||||
0x8086,0x3ea1,UHD Graphics,Whiskey Lake-U GT1 (UHD Graphics 610)
|
||||
0x8086,0x3ea2,UHD Graphics,Whiskey Lake GT1
|
||||
0x8086,0x3ea3,UHD Graphics,Whiskey Lake GT1
|
||||
0x8086,0x3ea4,UHD Graphics,Whiskey Lake GT1
|
||||
0x8086,0x3ea5,Iris Plus Graphics,Coffee Lake-U GT3e (Iris Plus 655)
|
||||
0x8086,0x3ea6,Iris Plus Graphics,Coffee Lake-U GT3 (Iris Plus 645)
|
||||
0x8086,0x3ea7,Iris Plus Graphics,Whiskey Lake GT3
|
||||
0x8086,0x3ea8,Iris Plus Graphics,Coffee Lake-U GT3 (Iris Plus 655)
|
||||
0x8086,0x3ea9,UHD Graphics,Coffee Lake-U GT2 (UHD Graphics 620)
|
||||
0x8086,0x87c0,UHD Graphics,9th Gen Core (UHD Graphics 617)
|
||||
0x8086,0x87ca,UHD Graphics,9th Gen Core (UHD Graphics 617)
|
||||
0x8086,0x8a50,Iris Plus Graphics,Ice Lake Gen1
|
||||
0x8086,0x8a51,Iris Plus Graphics,Ice Lake GT2 (Iris Plus G7)
|
||||
0x8086,0x8a52,Iris Plus Graphics,Ice Lake GT2 (Iris Plus G7)
|
||||
0x8086,0x8a53,Iris Plus Graphics,Ice Lake GT2 (Iris Plus G7)
|
||||
0x8086,0x8a54,Iris Plus Graphics,Ice Lake GT2
|
||||
0x8086,0x8a56,Iris Plus Graphics,Ice Lake GT1 (Iris Plus G1)
|
||||
0x8086,0x8a57,Iris Plus Graphics,Ice Lake GT1
|
||||
0x8086,0x8a58,UHD Graphics,Ice Lake-Y GT1 (UHD Graphics G1)
|
||||
0x8086,0x8a59,UHD Graphics,Ice Lake GT1
|
||||
0x8086,0x8a5a,Iris Plus Graphics,Ice Lake GT4 (Iris Plus G4)
|
||||
0x8086,0x8a5b,Iris Plus Graphics,Ice Lake GT4
|
||||
0x8086,0x8a5c,Iris Plus Graphics,Ice Lake GT4 (Iris Plus G4)
|
||||
0x8086,0x8a5d,Iris Plus Graphics,Ice Lake GT4
|
||||
0x8086,0x8a70,Iris Plus Graphics,Ice Lake
|
||||
0x8086,0x8a71,Iris Plus Graphics,Ice Lake
|
||||
0x8086,0x9a40,Iris Xe Graphics,Tiger Lake-UP4 GT2
|
||||
0x8086,0x9a49,Iris Xe Graphics,Tiger Lake-LP GT2
|
||||
0x8086,0x9a59,Iris Xe Graphics,Tiger Lake GT2
|
||||
0x8086,0x9a60,UHD Graphics,Tiger Lake-H GT1
|
||||
0x8086,0x9a68,UHD Graphics,Tiger Lake-H GT1
|
||||
0x8086,0x9a70,UHD Graphics,Tiger Lake-H GT1
|
||||
0x8086,0x9a78,UHD Graphics,Tiger Lake-LP GT2 (UHD Graphics G4)
|
||||
0x8086,0x9ac0,Iris Xe Graphics,Tiger Lake
|
||||
0x8086,0x9ac9,Iris Xe Graphics,Tiger Lake
|
||||
0x8086,0x9ad9,Iris Xe Graphics,Tiger Lake
|
||||
0x8086,0x9af8,Iris Xe Graphics,Tiger Lake
|
||||
0x8086,0x9b21,UHD Graphics,Comet Lake-U GT2 (UHD Graphics 620)
|
||||
0x8086,0x9b41,UHD Graphics,Comet Lake-U GT2
|
||||
0x8086,0x9ba0,UHD Graphics,Comet Lake GT1
|
||||
0x8086,0x9ba2,UHD Graphics,Comet Lake GT1
|
||||
0x8086,0x9ba4,UHD Graphics,Comet Lake-H GT1 (UHD Graphics 610)
|
||||
0x8086,0x9ba5,UHD Graphics,Comet Lake GT1
|
||||
0x8086,0x9ba8,UHD Graphics,Comet Lake-S GT1 (UHD Graphics 610)
|
||||
0x8086,0x9baa,UHD Graphics,Comet Lake GT1
|
||||
0x8086,0x9bab,UHD Graphics,Comet Lake GT1
|
||||
0x8086,0x9bac,UHD Graphics,Comet Lake GT1
|
||||
0x8086,0x9bc0,UHD Graphics,Comet Lake GT2
|
||||
0x8086,0x9bc2,UHD Graphics,Comet Lake GT2
|
||||
0x8086,0x9bc4,UHD Graphics,Comet Lake-H GT2
|
||||
0x8086,0x9bc5,UHD Graphics,Comet Lake-S GT2 (UHD Graphics 630)
|
||||
0x8086,0x9bc6,UHD Graphics,Comet Lake-S GT2 (UHD Graphics P630)
|
||||
0x8086,0x9bc8,UHD Graphics,Comet Lake-S GT2 (UHD Graphics 630)
|
||||
0x8086,0x9bca,UHD Graphics,Comet Lake GT2
|
||||
0x8086,0x9bcb,UHD Graphics,Comet Lake GT2
|
||||
0x8086,0x9bcc,UHD Graphics,Comet Lake GT2
|
||||
0x8086,0x9be6,UHD Graphics,Comet Lake-S GT2 (UHD Graphics P630)
|
||||
0x8086,0x9bf6,UHD Graphics,Coffee Lake-S GT2 (UHD Graphics P630)
|
||||
0x8086,0x4555,UHD Graphics,Elkhart Lake GT2 (UHD Graphics Gen11 16EU)
|
||||
0x8086,0x4571,UHD Graphics,Elkhart Lake GT2 (UHD Graphics Gen11 32EU)
|
||||
0x8086,0x4500,Intel Graphics,Elkhart Lake GT1
|
||||
0x8086,0x4541,Intel Graphics,Elkhart Lake GT1
|
||||
0x8086,0x4551,Intel Graphics,Elkhart Lake GT1
|
||||
0x8086,0x4557,Intel Graphics,Elkhart Lake GT1
|
||||
0x8086,0x4570,Intel Graphics,Elkhart Lake GT1
|
||||
0x8086,0x4c80,Intel Graphics,Rocket Lake
|
||||
0x8086,0x4c8a,UHD Graphics,Rocket Lake-S GT1 (UHD Graphics 750)
|
||||
0x8086,0x4c8b,UHD Graphics,Rocket Lake-S GT1 (UHD Graphics 730)
|
||||
0x8086,0x4c8c,Intel Graphics,Rocket Lake GT1
|
||||
0x8086,0x4c90,UHD Graphics,Rocket Lake-S GT1 (UHD Graphics P750)
|
||||
0x8086,0x4c9a,UHD Graphics,Rocket Lake-S
|
||||
0x8086,0x4e51,Intel Graphics,Jasper Lake GT1
|
||||
0x8086,0x4e55,UHD Graphics,Jasper Lake
|
||||
0x8086,0x4e57,Intel Graphics,Jasper Lake GT1
|
||||
0x8086,0x4e61,UHD Graphics,Jasper Lake
|
||||
0x8086,0x4e71,UHD Graphics,Jasper Lake
|
||||
0x8086,0x4626,Intel Graphics,Alder Lake-P GT1
|
||||
0x8086,0x4628,UHD Graphics,Alder Lake-UP3 GT2
|
||||
0x8086,0x462a,Intel Graphics,Alder Lake-P GT1
|
||||
0x8086,0x4680,UHD Graphics,Alder Lake-S GT1 (UHD Graphics 770)
|
||||
0x8086,0x4681,Intel Graphics,Alder Lake-S GT1
|
||||
0x8086,0x4682,UHD Graphics,Alder Lake-S GT1 (UHD Graphics 730)
|
||||
0x8086,0x4683,Intel Graphics,Alder Lake-S GT1
|
||||
0x8086,0x4688,UHD Graphics,Alder Lake-HX GT1 (UHD Graphics 770)
|
||||
0x8086,0x4689,Intel Graphics,Alder Lake-HX GT1
|
||||
0x8086,0x468a,Intel Graphics,Alder Lake-S
|
||||
0x8086,0x468b,Intel Graphics,Alder Lake-S
|
||||
0x8086,0x4690,UHD Graphics,Alder Lake-S GT1 (UHD Graphics 770)
|
||||
0x8086,0x4691,Intel Graphics,Alder Lake-S GT1
|
||||
0x8086,0x4692,UHD Graphics,Alder Lake-S GT1 (UHD Graphics 730)
|
||||
0x8086,0x4693,UHD Graphics,Alder Lake-S GT1 (UHD Graphics 710)
|
||||
0x8086,0x46a0,Intel Graphics,Alder Lake-P GT2
|
||||
0x8086,0x46a1,UHD Graphics,Alder Lake-P GT1
|
||||
0x8086,0x46a2,Intel Graphics,Alder Lake-P GT1
|
||||
0x8086,0x46a3,UHD Graphics,Alder Lake-P GT1
|
||||
0x8086,0x46a6,Iris Xe Graphics,Alder Lake-P GT2
|
||||
0x8086,0x46a8,Iris Xe Graphics,Alder Lake-UP3 GT2
|
||||
0x8086,0x46aa,Iris Xe Graphics,Alder Lake-UP4 GT2
|
||||
0x8086,0x46b0,Iris Xe Graphics,Alder Lake-P
|
||||
0x8086,0x46b1,Iris Xe Graphics,Alder Lake-P
|
||||
0x8086,0x46b2,Intel Graphics,Alder Lake-P GT1
|
||||
0x8086,0x46b3,UHD Graphics,Alder Lake-UP3 GT1
|
||||
0x8086,0x46c0,Intel Graphics,Alder Lake-M GT1
|
||||
0x8086,0x46c1,Iris Xe Graphics,Alder Lake-M
|
||||
0x8086,0x46c2,Intel Graphics,Alder Lake-M GT1
|
||||
0x8086,0x46c3,UHD Graphics,Alder Lake-UP4 GT1
|
||||
0x8086,0x46d0,UHD Graphics,Alder Lake-N
|
||||
0x8086,0x46d1,UHD Graphics,Alder Lake-N
|
||||
0x8086,0x46d2,UHD Graphics,Alder Lake-N
|
||||
0x8086,0x46d3,Intel Graphics,Alder Lake-N
|
||||
0x8086,0x46d4,Intel Graphics,Alder Lake-N
|
||||
0x8086,0x7d40,Intel Graphics,Meteor Lake-M
|
||||
0x8086,0x7d41,Intel Graphics,Arrow Lake-U
|
||||
0x8086,0x7d45,Intel Graphics,Meteor Lake-P
|
||||
0x8086,0x7d51,Arc Pro Graphics,Arrow Lake-P (Arc Pro 130T/140T)
|
||||
0x8086,0x7d55,Intel Arc Graphics,Meteor Lake-P
|
||||
0x8086,0x7d60,Intel Graphics,Meteor Lake-M
|
||||
0x8086,0x7d67,Intel Graphics,Arrow Lake-S
|
||||
0x8086,0x7dd1,Intel Graphics,Arrow Lake-P
|
||||
0x8086,0x7dd5,Intel Graphics,Meteor Lake-P
|
||||
0x8086,0xa720,UHD Graphics,Raptor Lake-P
|
||||
0x8086,0xa721,UHD Graphics,Raptor Lake-P
|
||||
0x8086,0xa780,UHD Graphics,Raptor Lake-S GT1 (UHD Graphics 770)
|
||||
0x8086,0xa781,UHD Graphics,Raptor Lake-S
|
||||
0x8086,0xa782,UHD Graphics,Raptor Lake-S
|
||||
0x8086,0xa783,UHD Graphics,Raptor Lake-S
|
||||
0x8086,0xa788,UHD Graphics,Raptor Lake-S
|
||||
0x8086,0xa789,UHD Graphics,Raptor Lake-S
|
||||
0x8086,0xa78a,UHD Graphics,Raptor Lake-S
|
||||
0x8086,0xa78b,UHD Graphics,Raptor Lake-S
|
||||
0x8086,0xa7a0,Iris Xe Graphics,Raptor Lake-P
|
||||
0x8086,0xa7a1,Iris Xe Graphics,Raptor Lake-P
|
||||
0x8086,0xa7a8,UHD Graphics,Raptor Lake-P
|
||||
0x8086,0xa7a9,UHD Graphics,Raptor Lake-P
|
||||
0x8086,0xa7aa,Intel Graphics,Raptor Lake-P
|
||||
0x8086,0xa7ab,Intel Graphics,Raptor Lake-P
|
||||
0x8086,0xa7ac,Intel Graphics,Raptor Lake-U
|
||||
0x8086,0xa7ad,Intel Graphics,Raptor Lake-U
|
||||
0x8086,0xb640,Intel Graphics,Arrow Lake-H
|
||||
0x8086,0x4905,Iris Xe MAX Graphics,DG1
|
||||
0x8086,0x4906,Iris Xe Graphics,DG1
|
||||
0x8086,0x4907,Intel Graphics,DG1 Server
|
||||
0x8086,0x4908,Iris Xe Graphics,DG1
|
||||
0x8086,0x4909,Iris Xe MAX Graphics,DG1 (Iris Xe MAX 100)
|
||||
0x8086,0x5690,Arc Graphics,DG2 (Arc A770M)
|
||||
0x8086,0x5691,Arc Graphics,DG2 (Arc A730M)
|
||||
0x8086,0x5692,Arc Graphics,DG2 (Arc A550M)
|
||||
0x8086,0x5693,Arc Graphics,DG2 (Arc A370M)
|
||||
0x8086,0x5694,Arc Graphics,DG2 (Arc A350M)
|
||||
0x8086,0x5695,Iris Xe MAX Graphics,DG2 (Iris Xe MAX A200M)
|
||||
0x8086,0x5696,Arc Graphics,DG2 (Arc A570M)
|
||||
0x8086,0x5697,Arc Graphics,DG2 (Arc A530M)
|
||||
0x8086,0x5698,Arc Graphics,DG2 (Arc Xe Graphics)
|
||||
0x8086,0x56a0,Arc Graphics,DG2 (Arc A770)
|
||||
0x8086,0x56a1,Arc Graphics,DG2 (Arc A750)
|
||||
0x8086,0x56a2,Arc Graphics,DG2 (Arc A580)
|
||||
0x8086,0x56a3,Arc Graphics,DG2 (Arc Xe Graphics)
|
||||
0x8086,0x56a4,Arc Graphics,DG2 (Arc Xe Graphics)
|
||||
0x8086,0x56a5,Arc Graphics,DG2 (Arc A380)
|
||||
0x8086,0x56a6,Arc Graphics,DG2 (Arc A310)
|
||||
0x8086,0x56b0,Arc Pro Graphics,DG2 (Arc Pro A30M)
|
||||
0x8086,0x56b1,Arc Pro Graphics,DG2 (Arc Pro A40/A50)
|
||||
0x8086,0x56b2,Arc Pro Graphics,DG2 (Arc Pro A60M)
|
||||
0x8086,0x56b3,Arc Pro Graphics,DG2 (Arc Pro A60)
|
||||
0x8086,0x56ba,Arc Graphics,DG2 (Arc A380E)
|
||||
0x8086,0x56bb,Arc Graphics,DG2 (Arc A310E)
|
||||
0x8086,0x56bc,Arc Graphics,DG2 (Arc A370E)
|
||||
0x8086,0x56bd,Arc Graphics,DG2 (Arc A350E)
|
||||
0x8086,0x56be,Arc Graphics,DG2 (Arc A750E)
|
||||
0x8086,0x56bf,Arc Graphics,DG2 (Arc A580E)
|
||||
0x8086,0x56c0,Arc Graphics,DG2 (Data Center GPU Flex 170)
|
||||
0x8086,0x56c1,Arc Graphics,DG2 (Data Center GPU Flex 140)
|
||||
0x8086,0x56c2,Arc Graphics,DG2 (Data Center GPU Flex 170V)
|
||||
|
@@ -25,6 +25,7 @@
|
||||
#
|
||||
# bus base class prog_if vendor device subsystem hid driver
|
||||
pci, 0C, 03, 30, *, *, *, *, /system/drivers/usb-xhci-bus
|
||||
pci, 03, 00, 00, 8086, 4C8A, *, *, /system/drivers/intel-uhd-graphics-750
|
||||
pci, 03, 80, *, 1AF4, 1050, *, *, /system/drivers/virtio-gpu
|
||||
usb, 03, 01, 01, *, *, *, *, /system/drivers/usb-hid-keyboard
|
||||
usb, 03, 01, 02, *, *, *, *, /system/drivers/usb-hid-mouse
|
||||
|
||||
|
@@ -28,6 +28,18 @@ pub const Device = struct {
|
||||
return .{ .block_size = result.block_size, .block_count = result.block_count };
|
||||
}
|
||||
|
||||
/// Hand the block server a DMA-region capability (`handle` — from a `shareable`
|
||||
/// dma_alloc) so it forwards it to the controller and the buffer's physical
|
||||
/// addresses become reachable by the device. Call once per buffer before naming it
|
||||
/// in `read`/`write`. Harmless success when no IOMMU is enforcing.
|
||||
pub fn attach(self: Device, handle: ipc.Handle) bool {
|
||||
var request = block_protocol.Request{ .operation = @intFromEnum(block_protocol.Operation.attach), .lba = 0, .count = 0, .physical = 0 };
|
||||
var reply: [block_protocol.reply_size]u8 = undefined;
|
||||
const result = ipc.callCap(self.endpoint, std.mem.asBytes(&request), &reply, handle) catch return false;
|
||||
if (result.len < block_protocol.reply_size) return false;
|
||||
return std.mem.bytesToValue(block_protocol.Reply, reply[0..block_protocol.reply_size]).status == 0;
|
||||
}
|
||||
|
||||
/// Read `count` blocks starting at `lba` into the DMA buffer at `physical`.
|
||||
pub fn read(self: Device, lba: u64, count: u32, physical: u64) bool {
|
||||
return self.transfer(.read, lba, count, physical);
|
||||
|
||||
@@ -99,6 +99,27 @@ pub fn msiBind(device_id: u64, endpoint: usize) ?Msi {
|
||||
return .{ .address = rax, .data = @intCast(rdx) };
|
||||
}
|
||||
|
||||
/// Map a delegated DMA-region (or shared-memory) capability into a claimed device's
|
||||
/// IOMMU domain, so the device may DMA to that buffer. The caller must own `device_id`
|
||||
/// and hold `handle` (received over IPC or from its own `dma.alloc(.. | shareable)`).
|
||||
/// Idempotent. Returns true on success (and trivially when no IOMMU is present).
|
||||
pub fn dmaBind(device_id: u64, handle: usize) bool {
|
||||
return !failed(sc.systemCall2(.dma_bind, device_id, handle));
|
||||
}
|
||||
|
||||
/// Unmap a previously `dmaBind`'d buffer from the device's domain.
|
||||
pub fn dmaUnbind(device_id: u64, handle: usize) bool {
|
||||
return !failed(sc.systemCall2(.dma_unbind, device_id, handle));
|
||||
}
|
||||
|
||||
/// Drain and log any pending IOMMU translation faults, returning the count seen. A
|
||||
/// diagnostic: a driver that suspects its device attempted an out-of-domain DMA (or a
|
||||
/// test proving enforcement) forces the hardware's fault records to the log now. Returns
|
||||
/// 0 when no IOMMU is present.
|
||||
pub fn iommuFaultDrain() usize {
|
||||
return sc.systemCall0(.iommu_fault_drain);
|
||||
}
|
||||
|
||||
/// Read `width` bytes (1, 2, or 4) from a port in a claimed device's `io_port`
|
||||
/// resource, at byte `offset` within it. Ring 3 has no direct `in`/`out`, so a legacy
|
||||
/// driver (PS/2, 16550 UART) reaches its ports through this claim-gated call — each
|
||||
|
||||
@@ -52,16 +52,134 @@ pub const config_vendor_id: usize = 0x00;
|
||||
pub const config_device_id: usize = 0x02;
|
||||
pub const config_command: usize = 0x04;
|
||||
pub const config_status: usize = 0x06;
|
||||
pub const config_capabilities_pointer: usize = 0x34;
|
||||
pub const config_revision_id: usize = 0x08;
|
||||
pub const config_class_code: usize = 0x09; // 3 bytes: prog-IF 0x09, subclass 0x0A, base class 0x0B
|
||||
pub const config_bar0: usize = 0x10; // BAR0; BAR n is at config_bar0 + n*4
|
||||
pub const config_subsystem_vendor_id: usize = 0x2C;
|
||||
pub const config_subsystem_id: usize = 0x2E;
|
||||
pub const config_expansion_rom: usize = 0x30;
|
||||
pub const config_capabilities_pointer: usize = 0x34;
|
||||
pub const config_interrupt_line: usize = 0x3C;
|
||||
pub const config_interrupt_pin: usize = 0x3D; // 0 = none, 1..4 = INTA..INTD
|
||||
|
||||
/// Command register: Memory-Space enable (bit 1) | Bus-Master enable (bit 2).
|
||||
pub const command_memory_and_bus_master: u16 = 0x06;
|
||||
/// Command register bits.
|
||||
pub const command_io_space: u16 = 0x0001; // bit 0: I/O-space decode enable
|
||||
pub const command_memory_space: u16 = 0x0002; // bit 1: memory-space decode enable
|
||||
pub const command_bus_master: u16 = 0x0004; // bit 2: bus-master (DMA) enable
|
||||
pub const command_interrupt_disable: u16 = 0x0400; // bit 10: suppress legacy INTx (MSI/MSI-X unaffected)
|
||||
/// The pair a bus-mastering driver enables together: decode my BARs, let me DMA.
|
||||
pub const command_memory_and_bus_master: u16 = command_memory_space | command_bus_master;
|
||||
|
||||
/// Status register bit 3: legacy INTx is asserted (upstream of the command bit-10 gate).
|
||||
pub const status_interrupt: u16 = 0x0008;
|
||||
/// Status register bit 4: a capability list is present at config_capabilities_pointer.
|
||||
pub const status_capabilities_list: u16 = 0x10;
|
||||
/// Capability pointers are dword-aligned; the low two bits are reserved.
|
||||
pub const capability_pointer_mask: u8 = 0xFC;
|
||||
|
||||
/// Capability IDs — the first byte of each entry in the legacy capability list.
|
||||
/// Non-exhaustive: hardware may report IDs not named here.
|
||||
pub const CapabilityId = enum(u8) {
|
||||
power_management = 0x01,
|
||||
msi = 0x05,
|
||||
vendor_specific = 0x09,
|
||||
pci_express = 0x10,
|
||||
msix = 0x11,
|
||||
_,
|
||||
};
|
||||
|
||||
/// MSI capability (id 0x05) register layout. Offsets are relative to the capability
|
||||
/// header; whether the address is one or two dwords (and therefore where the data word
|
||||
/// sits) depends on `control_64bit_capable`.
|
||||
pub const msi = struct {
|
||||
pub const control: usize = 0x02; // u16 Message Control
|
||||
pub const control_enable: u16 = 0x0001;
|
||||
pub const control_multiple_message_capable_mask: u16 = 0x000E; // bits 3:1, log2(vectors requested)
|
||||
pub const control_multiple_message_enable_mask: u16 = 0x0070; // bits 6:4, log2(vectors granted)
|
||||
pub const control_64bit_capable: u16 = 0x0080; // bit 7: address is 64-bit (layout shifts)
|
||||
pub const control_per_vector_masking: u16 = 0x0100; // bit 8
|
||||
pub const address: usize = 0x04; // u32 low address dword (both layouts)
|
||||
pub const address_high: usize = 0x08; // u32, present only when 64-bit capable
|
||||
pub const data_32: usize = 0x08; // u16 message data, 32-bit layout
|
||||
pub const data_64: usize = 0x0C; // u16 message data, 64-bit layout
|
||||
pub const mask_bits_32: usize = 0x0C; // u32, only with per-vector masking
|
||||
pub const mask_bits_64: usize = 0x10;
|
||||
};
|
||||
|
||||
/// MSI-X capability (id 0x11) register layout, plus the 16-byte vector table entry that
|
||||
/// lives in BAR space (not configuration space) at the decoded (BIR, offset).
|
||||
pub const msix = struct {
|
||||
pub const control: usize = 0x02; // u16 Message Control
|
||||
pub const control_table_size_mask: u16 = 0x07FF; // bits 10:0, encoded as N-1
|
||||
pub const control_function_mask: u16 = 0x4000; // bit 14: mask every vector
|
||||
pub const control_enable: u16 = 0x8000; // bit 15
|
||||
pub const table_offset_word: usize = 0x04; // u32: BIR in bits 2:0, table offset in bits 31:3
|
||||
pub const pba_offset_word: usize = 0x08; // u32: same encoding, pending-bit array
|
||||
pub const bir_mask: u32 = 0x0000_0007;
|
||||
pub const offset_mask: u32 = 0xFFFF_FFF8;
|
||||
pub const entry_size: usize = 16; // table entry stride; offsets within an entry:
|
||||
pub const entry_address: usize = 0x0; // u32 low
|
||||
pub const entry_address_high: usize = 0x4; // u32 high
|
||||
pub const entry_data: usize = 0x8; // u32
|
||||
pub const entry_vector_control: usize = 0xC; // u32
|
||||
pub const entry_vector_control_masked: u32 = 0x1; // bit 0; entries reset to masked
|
||||
|
||||
/// Where the table (or pending-bit array) lives, decoded from its offset/BIR dword.
|
||||
pub const TableLocation = struct { bar: u8, offset: u32 };
|
||||
pub fn tableLocation(word: u32) TableLocation {
|
||||
return .{ .bar = @intCast(word & bir_mask), .offset = word & offset_mask };
|
||||
}
|
||||
/// Number of table entries (the control field encodes N-1).
|
||||
pub fn tableSize(control_value: u16) u16 {
|
||||
return (control_value & control_table_size_mask) + 1;
|
||||
}
|
||||
};
|
||||
|
||||
/// Power-management capability (id 0x01) register layout.
|
||||
pub const power_management = struct {
|
||||
pub const capabilities: usize = 0x02; // u16 PMC (read-only: version, D-state support)
|
||||
pub const control_status: usize = 0x04; // u16 PMCSR
|
||||
pub const control_status_power_state_mask: u16 = 0x0003; // bits 1:0
|
||||
pub const power_state_d0: u16 = 0x0;
|
||||
pub const power_state_d3_hot: u16 = 0x3;
|
||||
pub const control_status_pme_enable: u16 = 0x0100; // bit 8: plain RW — preserve on writes
|
||||
pub const control_status_pme_status: u16 = 0x8000; // bit 15: RW1C — write 0 or you clear it
|
||||
};
|
||||
|
||||
/// PCI Express capability (id 0x10) register layout — the slice function-level reset
|
||||
/// needs; the full capability is much larger.
|
||||
pub const pci_express = struct {
|
||||
pub const capabilities: usize = 0x02; // u16 PCIe Capabilities register
|
||||
pub const device_capabilities: usize = 0x04; // u32
|
||||
pub const device_capabilities_flr: u32 = 1 << 28; // Function Level Reset supported
|
||||
pub const device_control: usize = 0x08; // u16
|
||||
pub const device_control_initiate_flr: u16 = 1 << 15;
|
||||
pub const device_status: usize = 0x0A; // u16
|
||||
pub const device_status_transactions_pending: u16 = 1 << 5;
|
||||
};
|
||||
|
||||
/// Extended (PCI Express) capabilities start here in the 4 KiB configuration space; a
|
||||
/// conventional-PCI function has nothing there (the space reads as all-ones).
|
||||
pub const extended_capability_start: usize = 0x100;
|
||||
/// Extended-capability next pointers are dword-aligned within the 4 KiB space.
|
||||
pub const extended_capability_pointer_mask: u16 = 0xFFC;
|
||||
|
||||
/// The 32-bit header at the start of each extended capability: ID in bits 15:0,
|
||||
/// version in 19:16, next offset in 31:20 (0 = end of list).
|
||||
pub const ExtendedCapabilityHeader = struct {
|
||||
id: u16,
|
||||
version: u4,
|
||||
next: u16,
|
||||
|
||||
pub fn decode(word: u32) ExtendedCapabilityHeader {
|
||||
return .{
|
||||
.id = @truncate(word),
|
||||
.version = @truncate(word >> 16),
|
||||
.next = @intCast((word >> 20) & extended_capability_pointer_mask),
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
/// BAR bit layout: bit 0 selects I/O (1) vs memory (0) space; for a memory BAR, bits 2:1
|
||||
/// give the type (00 = 32-bit, 10 = 64-bit spanning the next BAR), and the base address is
|
||||
/// the dword with the low 4 flag bits masked off.
|
||||
@@ -595,3 +713,37 @@ test "named parts pack to the raw triple" {
|
||||
};
|
||||
try std.testing.expectEqual(@as(u24, 0x0C_03_30), xhci.pack());
|
||||
}
|
||||
|
||||
test "MSI-X table word decodes to BIR and offset" {
|
||||
const eq = std.testing.expectEqual;
|
||||
// BIR 3, table at 0x2000 within that BAR.
|
||||
try eq(msix.TableLocation{ .bar = 3, .offset = 0x2000 }, msix.tableLocation(0x0000_2003));
|
||||
// BIR 0, offset 0 — the degenerate-but-common "table at BAR start" case.
|
||||
try eq(msix.TableLocation{ .bar = 0, .offset = 0 }, msix.tableLocation(0));
|
||||
// Table size encodes N-1 in bits 10:0; enable/function-mask bits must not leak in.
|
||||
try eq(@as(u16, 11), msix.tableSize(msix.control_enable | 0x000A));
|
||||
try eq(@as(u16, 1), msix.tableSize(0));
|
||||
try eq(@as(u16, 2048), msix.tableSize(msix.control_table_size_mask));
|
||||
}
|
||||
|
||||
test "extended capability header unpacks id, version, next" {
|
||||
const eq = std.testing.expectEqual;
|
||||
// AER (id 0x0001), version 1, next capability at 0x140.
|
||||
const aer = ExtendedCapabilityHeader.decode(0x1401_0001);
|
||||
try eq(@as(u16, 0x0001), aer.id);
|
||||
try eq(@as(u4, 1), aer.version);
|
||||
try eq(@as(u16, 0x140), aer.next);
|
||||
// A zero header is the "nothing here" terminator.
|
||||
const none = ExtendedCapabilityHeader.decode(0);
|
||||
try eq(@as(u16, 0), none.id);
|
||||
try eq(@as(u16, 0), none.next);
|
||||
}
|
||||
|
||||
test "command bits and capability ids compose" {
|
||||
const eq = std.testing.expectEqual;
|
||||
try eq(command_memory_space | command_bus_master, command_memory_and_bus_master);
|
||||
try eq(@as(u8, 0x05), @intFromEnum(CapabilityId.msi));
|
||||
try eq(@as(u8, 0x11), @intFromEnum(CapabilityId.msix));
|
||||
try eq(@as(u8, 0x01), @intFromEnum(CapabilityId.power_management));
|
||||
try eq(@as(u8, 0x10), @intFromEnum(CapabilityId.pci_express));
|
||||
}
|
||||
|
||||
+283
-7
@@ -1,7 +1,8 @@
|
||||
//! library/device/pci/pci.zig — a device driver's view of the ONE PCI function it has
|
||||
//! claimed. Config space is mapped as resource 0; this gives header-field accessors, BAR
|
||||
//! decode + map, and a capability-list iterator, so a driver never re-derives the
|
||||
//! config-space layout by hand.
|
||||
//! claimed. Config space is mapped as resource 0 (a full 4 KiB ECAM page); this gives
|
||||
//! header-field accessors, BAR decode + map, capability walks (legacy and extended),
|
||||
//! MSI/MSI-X programming, power-state handling, and function-level reset, so a driver
|
||||
//! never re-derives the config-space layout by hand.
|
||||
//!
|
||||
//! This is the *device-owned* view: read my own function's live config, map my own BARs.
|
||||
//! The bus enumerator's view — probing arbitrary, not-yet-claimed functions and sizing
|
||||
@@ -13,6 +14,18 @@ const std = @import("std");
|
||||
const mmio = @import("mmio");
|
||||
const pci_class = @import("pci-class");
|
||||
const device = @import("driver");
|
||||
const time = @import("time");
|
||||
|
||||
/// Spec recovery time after a D3hot -> D0 transition.
|
||||
const d0_recovery_millis: u64 = 10;
|
||||
/// How long to wait for in-flight transactions to drain before a function-level reset
|
||||
/// (then reset anyway — resetting a stuck function is the point of FLR).
|
||||
const flr_pending_timeout_millis: u64 = 100;
|
||||
/// The spec's maximum FLR completion time.
|
||||
const flr_settle_millis: u64 = 100;
|
||||
/// How long to wait for the function to become readable again after an FLR.
|
||||
const flr_ready_timeout_millis: u64 = 1000;
|
||||
const flr_poll_interval_millis: u64 = 10;
|
||||
|
||||
/// A claimed PCI function whose configuration space is mapped (resource 0). `descriptor`
|
||||
/// must outlive the Function — the driver's `device.enumerate` buffer does, for the whole
|
||||
@@ -42,12 +55,61 @@ pub const Function = struct {
|
||||
pub fn status(self: *const Function) u16 {
|
||||
return mmio.readRegister(u16, self.config + pci_class.config_status);
|
||||
}
|
||||
pub fn revisionId(self: *const Function) u8 {
|
||||
return mmio.readRegister(u8, self.config + pci_class.config_revision_id);
|
||||
}
|
||||
/// Subsystem vendor ID (config 0x2C) — with `subsystemId`, the standard key for
|
||||
/// board-level quirk matching.
|
||||
pub fn subsystemVendorId(self: *const Function) u16 {
|
||||
return mmio.readRegister(u16, self.config + pci_class.config_subsystem_vendor_id);
|
||||
}
|
||||
pub fn subsystemId(self: *const Function) u16 {
|
||||
return mmio.readRegister(u16, self.config + pci_class.config_subsystem_id);
|
||||
}
|
||||
/// Interrupt pin (config 0x3D): 0 = none, 1..4 = INTA..INTD.
|
||||
pub fn interruptPin(self: *const Function) u8 {
|
||||
return mmio.readRegister(u8, self.config + pci_class.config_interrupt_pin);
|
||||
}
|
||||
/// The live class-code triple (config 0x09..0x0B), same shape discovery records.
|
||||
pub fn classCode(self: *const Function) pci_class.ClassCode {
|
||||
return .{
|
||||
.prog_if = mmio.readRegister(u8, self.config + pci_class.config_class_code),
|
||||
.subclass = mmio.readRegister(u8, self.config + pci_class.config_class_code + 1),
|
||||
.base = mmio.readRegister(u8, self.config + pci_class.config_class_code + 2),
|
||||
};
|
||||
}
|
||||
|
||||
/// Set Memory-Space + Bus-Master enable in the command register. Firmware often leaves
|
||||
/// a secondary display's decode off; a bus-mastering device must enable both.
|
||||
pub fn enableMemoryAndBusMaster(self: *const Function) void {
|
||||
fn commandSetBits(self: *const Function, bits: u16) void {
|
||||
const at = self.config + pci_class.config_command;
|
||||
mmio.writeRegister(u16, at, mmio.readRegister(u16, at) | pci_class.command_memory_and_bus_master);
|
||||
mmio.writeRegister(u16, at, mmio.readRegister(u16, at) | bits);
|
||||
}
|
||||
fn commandClearBits(self: *const Function, bits: u16) void {
|
||||
const at = self.config + pci_class.config_command;
|
||||
mmio.writeRegister(u16, at, mmio.readRegister(u16, at) & ~bits);
|
||||
}
|
||||
|
||||
/// Set Memory-Space + Bus-Master Enable in the command register. Firmware only enables
|
||||
/// memory decode on devices it used at boot; any other device has dead BARs until its
|
||||
/// driver sets it. Bus mastering is separately required for the device to do DMA.
|
||||
pub fn enableMemoryAndBusMaster(self: *const Function) void {
|
||||
self.commandSetBits(pci_class.command_memory_and_bus_master);
|
||||
}
|
||||
|
||||
/// Clear Bus-Master Enable — stop the device initiating DMA. The quiesce half of a
|
||||
/// driver's shutdown (or a supervisor restart): after this the device can no longer
|
||||
/// write memory the process is about to stop owning.
|
||||
pub fn disableBusMaster(self: *const Function) void {
|
||||
self.commandClearBits(pci_class.command_bus_master);
|
||||
}
|
||||
|
||||
/// Set command bit 10: suppress legacy INTx assertion. MSI/MSI-X are unaffected —
|
||||
/// set this when enabling either, so the device cannot also raise the shared pin.
|
||||
pub fn setInterruptDisable(self: *const Function) void {
|
||||
self.commandSetBits(pci_class.command_interrupt_disable);
|
||||
}
|
||||
/// Clear command bit 10, re-allowing legacy INTx assertion.
|
||||
pub fn clearInterruptDisable(self: *const Function) void {
|
||||
self.commandClearBits(pci_class.command_interrupt_disable);
|
||||
}
|
||||
|
||||
/// Decode BAR `bar` (0..5) and map it: read the BAR register, reject I/O-space BARs,
|
||||
@@ -86,6 +148,129 @@ pub const Function = struct {
|
||||
0;
|
||||
return .{ .config = self.config, .cursor = first };
|
||||
}
|
||||
|
||||
/// First capability with `id`, or null.
|
||||
pub fn findCapability(self: *const Function, id: pci_class.CapabilityId) ?Capability {
|
||||
var walk = self.capabilities();
|
||||
while (walk.next()) |capability| {
|
||||
if (capability.id == @intFromEnum(id)) return capability;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Program the MSI capability with the kernel's `msi_bind` result and enable it —
|
||||
/// one vector (multiple-message-enable 0, matching the kernel's single-vector
|
||||
/// grant), INTx suppressed. false if the function has no MSI capability.
|
||||
pub fn programMsi(self: *const Function, message: device.Msi) bool {
|
||||
const cap = self.findCapability(.msi) orelse return false;
|
||||
const control_at = cap.offset + pci_class.msi.control;
|
||||
const control = mmio.readRegister(u16, control_at);
|
||||
// Program the registers while the capability is disabled.
|
||||
mmio.writeRegister(u16, control_at, control & ~pci_class.msi.control_enable);
|
||||
mmio.writeRegister(u32, cap.offset + pci_class.msi.address, @truncate(message.address));
|
||||
const data_offset = if (control & pci_class.msi.control_64bit_capable != 0) offset: {
|
||||
mmio.writeRegister(u32, cap.offset + pci_class.msi.address_high, @intCast(message.address >> 32));
|
||||
break :offset pci_class.msi.data_64;
|
||||
} else pci_class.msi.data_32;
|
||||
// Message data is a 16-bit register in both layouts.
|
||||
mmio.writeRegister(u16, cap.offset + data_offset, @truncate(message.data));
|
||||
mmio.writeRegister(u16, control_at, (control & ~pci_class.msi.control_multiple_message_enable_mask) | pci_class.msi.control_enable);
|
||||
self.setInterruptDisable();
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Clear the MSI enable bit. No-op if the function has no MSI capability.
|
||||
pub fn disableMsi(self: *const Function) void {
|
||||
const cap = self.findCapability(.msi) orelse return;
|
||||
const control_at = cap.offset + pci_class.msi.control;
|
||||
mmio.writeRegister(u16, control_at, mmio.readRegister(u16, control_at) & ~pci_class.msi.control_enable);
|
||||
}
|
||||
|
||||
/// The function's MSI-X capability with its vector table mapped: the table's BIR is
|
||||
/// resolved through `mapBar` (a free cache hit when it is a BAR the driver already
|
||||
/// mapped). null if the capability is absent or the table's BAR cannot be mapped.
|
||||
pub fn msix(self: *Function) ?MsiX {
|
||||
const cap = self.findCapability(.msix) orelse return null;
|
||||
const control = mmio.readRegister(u16, cap.offset + pci_class.msix.control);
|
||||
const word = mmio.readRegister(u32, cap.offset + pci_class.msix.table_offset_word);
|
||||
const location = pci_class.msix.tableLocation(word);
|
||||
const bar_base = self.mapBar(location.bar) orelse return null;
|
||||
return .{
|
||||
.capability = cap.offset,
|
||||
.table = bar_base + location.offset,
|
||||
.entry_count = pci_class.msix.tableSize(control),
|
||||
};
|
||||
}
|
||||
|
||||
/// Bring the function to D0. Firmware can leave a non-boot device in D3hot, where
|
||||
/// its BARs and MSI registers do not decode; call this before touching either. No
|
||||
/// power-management capability means the function is always at D0: nothing to do.
|
||||
/// Preserves PME-Enable and never clears the write-1-to-clear PME-Status bit.
|
||||
pub fn ensurePowerStateD0(self: *const Function) void {
|
||||
const cap = self.findCapability(.power_management) orelse return;
|
||||
const at = cap.offset + pci_class.power_management.control_status;
|
||||
const pmcsr = mmio.readRegister(u16, at);
|
||||
if (pmcsr & pci_class.power_management.control_status_power_state_mask == pci_class.power_management.power_state_d0) return;
|
||||
// PME-Status is RW1C: echoing a read 1 back would clear it, so write it as 0.
|
||||
mmio.writeRegister(u16, at, (pmcsr & ~pci_class.power_management.control_status_power_state_mask & ~pci_class.power_management.control_status_pme_status) | pci_class.power_management.power_state_d0);
|
||||
time.sleepMillis(d0_recovery_millis);
|
||||
}
|
||||
|
||||
/// Function Level Reset via the PCI Express capability: return the hardware to a
|
||||
/// known state (a supervisor re-claiming a device after its driver died, or a driver
|
||||
/// recovering a wedged function). The six BAR dwords are saved and restored — FLR
|
||||
/// clears them, and the bus enumerator's assignment must survive for the descriptor
|
||||
/// correlation and `mapBar` cache to stay valid. Everything else is reset: command
|
||||
/// enables and MSI/MSI-X programming are gone, so the caller re-runs its whole
|
||||
/// bring-up afterwards. false if the function has no PCI Express capability, does
|
||||
/// not advertise FLR (conventional-PCI Advanced Features FLR is a possible
|
||||
/// follow-up), or never became readable again. Blocks for at least 100 ms.
|
||||
pub fn functionLevelReset(self: *const Function) bool {
|
||||
const cap = self.findCapability(.pci_express) orelse return false;
|
||||
const device_capabilities = mmio.readRegister(u32, cap.offset + pci_class.pci_express.device_capabilities);
|
||||
if (device_capabilities & pci_class.pci_express.device_capabilities_flr == 0) return false;
|
||||
|
||||
// Stop new DMA, then give in-flight transactions a bounded chance to drain —
|
||||
// and reset anyway on timeout, since resetting a stuck function is the point.
|
||||
self.disableBusMaster();
|
||||
var waited: u64 = 0;
|
||||
while (mmio.readRegister(u16, cap.offset + pci_class.pci_express.device_status) & pci_class.pci_express.device_status_transactions_pending != 0) {
|
||||
if (waited >= flr_pending_timeout_millis) break;
|
||||
time.sleepMillis(flr_poll_interval_millis);
|
||||
waited += flr_poll_interval_millis;
|
||||
}
|
||||
|
||||
var bars: [6]u32 = undefined;
|
||||
for (&bars, 0..) |*bar, index| bar.* = mmio.readRegister(u32, self.config + pci_class.config_bar0 + index * 4);
|
||||
|
||||
const control_at = cap.offset + pci_class.pci_express.device_control;
|
||||
mmio.writeRegister(u16, control_at, mmio.readRegister(u16, control_at) | pci_class.pci_express.device_control_initiate_flr);
|
||||
time.sleepMillis(flr_settle_millis);
|
||||
|
||||
waited = 0;
|
||||
while (self.vendorId() == 0xFFFF) {
|
||||
if (waited >= flr_ready_timeout_millis) return false;
|
||||
time.sleepMillis(flr_poll_interval_millis);
|
||||
waited += flr_poll_interval_millis;
|
||||
}
|
||||
for (bars, 0..) |bar, index| mmio.writeRegister(u32, self.config + pci_class.config_bar0 + index * 4, bar);
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Iterate the extended (PCI Express) capability list at 0x100.. in the 4 KiB ECAM
|
||||
/// page. Empty on a conventional-PCI function (the space reads as all-ones).
|
||||
pub fn extendedCapabilities(self: *const Function) ExtendedCapabilityIterator {
|
||||
return .{ .config = self.config };
|
||||
}
|
||||
|
||||
/// First extended capability with `id`, or null.
|
||||
pub fn findExtendedCapability(self: *const Function, id: u16) ?ExtendedCapability {
|
||||
var walk = self.extendedCapabilities();
|
||||
while (walk.next()) |capability| {
|
||||
if (capability.id == id) return capability;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
};
|
||||
|
||||
/// One capability header. `offset` is the ABSOLUTE virtual address of the header, so the
|
||||
@@ -106,3 +291,94 @@ pub const CapabilityIterator = struct {
|
||||
return .{ .id = id, .offset = at };
|
||||
}
|
||||
};
|
||||
|
||||
/// A resolved MSI-X capability from `Function.msix`: `capability` is the absolute
|
||||
/// virtual address of the config-space header, `table` of vector-table entry 0 (in BAR
|
||||
/// space — table writes are MMIO, not config space). Entries reset masked; bring-up
|
||||
/// order is programEntry per vector, unmaskEntry per used vector, `enable`, then
|
||||
/// `Function.setInterruptDisable`.
|
||||
pub const MsiX = struct {
|
||||
capability: usize,
|
||||
table: usize,
|
||||
entry_count: u16,
|
||||
|
||||
/// Write `message` into table entry `entry`, leaving the entry masked (its reset
|
||||
/// state) — the spec requires masking while address/data change. false if `entry`
|
||||
/// is out of range.
|
||||
pub fn programEntry(self: *const MsiX, entry: u16, message: device.Msi) bool {
|
||||
if (entry >= self.entry_count) return false;
|
||||
const at = self.table + @as(usize, entry) * pci_class.msix.entry_size;
|
||||
mmio.writeRegister(u32, at + pci_class.msix.entry_vector_control, pci_class.msix.entry_vector_control_masked);
|
||||
mmio.writeRegister(u32, at + pci_class.msix.entry_address, @truncate(message.address));
|
||||
mmio.writeRegister(u32, at + pci_class.msix.entry_address_high, @intCast(message.address >> 32));
|
||||
mmio.writeRegister(u32, at + pci_class.msix.entry_data, message.data);
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Set the entry's vector-control mask bit — its interrupt is held off (pended in
|
||||
/// the PBA, not lost). false if `entry` is out of range.
|
||||
pub fn maskEntry(self: *const MsiX, entry: u16) bool {
|
||||
return self.writeEntryMask(entry, true);
|
||||
}
|
||||
/// Clear the entry's vector-control mask bit. false if `entry` is out of range.
|
||||
pub fn unmaskEntry(self: *const MsiX, entry: u16) bool {
|
||||
return self.writeEntryMask(entry, false);
|
||||
}
|
||||
fn writeEntryMask(self: *const MsiX, entry: u16, masked: bool) bool {
|
||||
if (entry >= self.entry_count) return false;
|
||||
const at = self.table + @as(usize, entry) * pci_class.msix.entry_size + pci_class.msix.entry_vector_control;
|
||||
const control = mmio.readRegister(u32, at);
|
||||
mmio.writeRegister(u32, at, if (masked)
|
||||
control | pci_class.msix.entry_vector_control_masked
|
||||
else
|
||||
control & ~pci_class.msix.entry_vector_control_masked);
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Set the function-mask control bit: every vector masked regardless of entry bits.
|
||||
pub fn setFunctionMask(self: *const MsiX) void {
|
||||
self.writeControl(pci_class.msix.control_function_mask, true);
|
||||
}
|
||||
/// Clear the function-mask control bit.
|
||||
pub fn clearFunctionMask(self: *const MsiX) void {
|
||||
self.writeControl(pci_class.msix.control_function_mask, false);
|
||||
}
|
||||
/// Set MSI-X Enable. The caller also calls `Function.setInterruptDisable` (INTx off).
|
||||
pub fn enable(self: *const MsiX) void {
|
||||
self.writeControl(pci_class.msix.control_enable, true);
|
||||
}
|
||||
/// Clear MSI-X Enable.
|
||||
pub fn disable(self: *const MsiX) void {
|
||||
self.writeControl(pci_class.msix.control_enable, false);
|
||||
}
|
||||
fn writeControl(self: *const MsiX, bit: u16, set: bool) void {
|
||||
const at = self.capability + pci_class.msix.control;
|
||||
const control = mmio.readRegister(u16, at);
|
||||
mmio.writeRegister(u16, at, if (set) control | bit else control & ~bit);
|
||||
}
|
||||
};
|
||||
|
||||
/// One extended capability. `offset` is the ABSOLUTE virtual address of its header,
|
||||
/// like `Capability.offset`.
|
||||
pub const ExtendedCapability = struct { id: u16, version: u4, offset: usize };
|
||||
|
||||
pub const ExtendedCapabilityIterator = struct {
|
||||
config: usize,
|
||||
cursor: u16 = @intCast(pci_class.extended_capability_start),
|
||||
guard: u32 = 0, // bounds a malformed chain (480 = the 0xF00-byte space / 8-byte minimum spacing)
|
||||
|
||||
pub fn next(self: *ExtendedCapabilityIterator) ?ExtendedCapability {
|
||||
if (self.cursor == 0 or self.guard >= 480) return null;
|
||||
self.guard += 1;
|
||||
const at = self.config + self.cursor;
|
||||
const header = pci_class.ExtendedCapabilityHeader.decode(mmio.readRegister(u32, at));
|
||||
// Id 0 marks an empty list; all-ones is a conventional-PCI function (no
|
||||
// extended space — reads come back as FFs).
|
||||
if (header.id == 0 or header.id == 0xFFFF) return null;
|
||||
// A next pointer below 0x100 would walk into the legacy header; treat it as the
|
||||
// terminator it must be (0 is the normal one). The 0xFFC decode mask already
|
||||
// keeps `config + cursor + 4` inside the 4 KiB page.
|
||||
self.cursor = if (header.next >= pci_class.extended_capability_start) header.next else 0;
|
||||
return .{ .id = header.id, .version = header.version, .offset = at };
|
||||
}
|
||||
};
|
||||
|
||||
@@ -99,6 +99,19 @@ pub const Device = struct {
|
||||
return std.mem.bytesToValue(usb_transfer_protocol.InterruptSubscribeReply, reply[0..@sizeOf(usb_transfer_protocol.InterruptSubscribeReply)]).status == 0;
|
||||
}
|
||||
|
||||
/// Hand the controller a DMA-region capability (`handle` — from a `shareable`
|
||||
/// dma_alloc, or forwarded from another process) so it binds that buffer into its
|
||||
/// IOMMU domain. Must be called for every buffer whose physical address this device
|
||||
/// will name in a `bulk` transfer, before the transfer. Harmless (and a no-op
|
||||
/// success) when no IOMMU is enforcing. Returns false on failure.
|
||||
pub fn attachDma(self: *Device, handle: ipc.Handle) bool {
|
||||
var request = usb_transfer_protocol.DmaAttachRequest{ .device_token = self.token };
|
||||
var reply: [@sizeOf(usb_transfer_protocol.DmaAttachReply)]u8 = undefined;
|
||||
const result = ipc.callCap(self.bus, std.mem.asBytes(&request), &reply, handle) catch return false;
|
||||
if (result.len < @sizeOf(usb_transfer_protocol.DmaAttachReply)) return false;
|
||||
return std.mem.bytesToValue(usb_transfer_protocol.DmaAttachReply, reply[0..@sizeOf(usb_transfer_protocol.DmaAttachReply)]).status == 0;
|
||||
}
|
||||
|
||||
/// One bulk transfer (IN or OUT per `endpoint_address`'s direction bit) to or
|
||||
/// from the caller's own DMA buffer at `physical`. Returns the bytes moved.
|
||||
pub fn bulk(self: *Device, endpoint_address: u8, physical: u64, length: u32) ?u32 {
|
||||
|
||||
@@ -34,6 +34,14 @@ pub fn register(id: abi.ServiceId, h: Handle) bool {
|
||||
return !failed(sc.systemCall2(.ipc_register, @intFromEnum(id), h));
|
||||
}
|
||||
|
||||
/// Drop a capability handle (endpoint, shared-memory, or DMA-region) and free its table
|
||||
/// slot. A forwarding hop closes a cap it passed on; a binder closes a DMA-region cap
|
||||
/// once the binding holds its own reference — the 32-slot table is otherwise consumed by
|
||||
/// repeated cap-passing.
|
||||
pub fn close(h: Handle) bool {
|
||||
return !failed(sc.systemCall1(.handle_close, h));
|
||||
}
|
||||
|
||||
/// Find the endpoint published under `id`, installing a handle to it in this
|
||||
/// process.
|
||||
pub fn lookup(id: abi.ServiceId) ?Handle {
|
||||
|
||||
@@ -12,12 +12,18 @@ const sc = @import("system-call");
|
||||
pub const coherent: usize = abi.dma_coherent;
|
||||
pub const write_combining: usize = abi.dma_write_combining;
|
||||
pub const below_4g: usize = abi.dma_below_4g;
|
||||
/// Ask for a capability handle (in `Region.handle`) so the buffer can be delegated to
|
||||
/// another driver and bound into a device's IOMMU domain (`driver.dmaBind`). A driver's
|
||||
/// private rings don't need it; a buffer whose physical address crosses IPC does.
|
||||
pub const shareable: usize = abi.dma_shareable;
|
||||
|
||||
/// A DMA allocation: the `virtual` address the CPU touches, and the `physical` address
|
||||
/// to program into the device's descriptor-ring / base registers.
|
||||
/// A DMA allocation: the `virtual` address the CPU touches, the `physical` address to
|
||||
/// program into the device's registers, and — when `shareable` was requested — a
|
||||
/// capability `handle` naming the region for delegation (null otherwise).
|
||||
pub const Region = struct {
|
||||
virtual: usize,
|
||||
physical: usize,
|
||||
handle: ?usize = null,
|
||||
};
|
||||
|
||||
inline fn failed(r: usize) bool {
|
||||
@@ -25,21 +31,23 @@ inline fn failed(r: usize) bool {
|
||||
}
|
||||
|
||||
/// Allocate `len` bytes of DMA-capable memory with `flags` (e.g. `coherent`, or
|
||||
/// `coherent | below_4g`). Returns the virtual/physical pair, or null on failure. Two
|
||||
/// return values — the virtual address in rax, the physical address in rdx — so it
|
||||
/// needs a hand-written stub.
|
||||
/// `coherent | shareable`). Returns virtual/physical (and a handle when `shareable`), or
|
||||
/// null on failure. Three return values — virtual in rax, physical in rdx, handle in r8
|
||||
/// — so it needs a hand-written stub.
|
||||
pub fn alloc(len: usize, flags: usize) ?Region {
|
||||
var rax: usize = undefined;
|
||||
var rdx: usize = undefined; // out: physical address
|
||||
var r8: usize = undefined; // out: capability handle (abi.no_cap unless shareable)
|
||||
asm volatile ("syscall"
|
||||
: [rax] "={rax}" (rax),
|
||||
[rdx] "={rdx}" (rdx),
|
||||
[r8] "={r8}" (r8),
|
||||
: [n] "{rax}" (@intFromEnum(abi.SystemCall.dma_alloc)),
|
||||
[a0] "{rdi}" (len),
|
||||
[a1] "{rsi}" (flags),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
if (failed(rax)) return null;
|
||||
return .{ .virtual = rax, .physical = rdx };
|
||||
return .{ .virtual = rax, .physical = rdx, .handle = if (r8 == abi.no_cap) null else r8 };
|
||||
}
|
||||
|
||||
/// Release a region from a prior `alloc` (`virtual` and the same `len`).
|
||||
|
||||
@@ -38,6 +38,7 @@ pub const DmaRegion = dma.Region;
|
||||
pub const dma_coherent = dma.coherent;
|
||||
pub const dma_write_combining = dma.write_combining;
|
||||
pub const dma_below_4g = dma.below_4g;
|
||||
pub const dma_shareable = dma.shareable;
|
||||
pub const dmaAlloc = dma.alloc;
|
||||
pub const dmaFree = dma.free;
|
||||
|
||||
|
||||
@@ -7,7 +7,9 @@
|
||||
//! buffer**, named by its physical address — the same physical-address handoff
|
||||
//! usb-storage already uses toward the controller, one layer up. So a 512-byte
|
||||
//! sector never has to cross the 256-byte IPC boundary; only the small request /
|
||||
//! reply headers do. (Safe while the IOMMU is unenforced; see docs/driver-model.md.)
|
||||
//! reply headers do. Under an enforcing IOMMU the buffer's physical addresses are
|
||||
//! only reachable by the device once the filesystem has `attach`ed the buffer's
|
||||
//! capability (the block server forwards it to the controller); see docs/driver-model.md.
|
||||
|
||||
pub const Operation = enum(u32) {
|
||||
/// geometry() -> { block_size, block_count }
|
||||
@@ -20,6 +22,11 @@ pub const Operation = enum(u32) {
|
||||
/// A filesystem calls this to make prior writes durable — e.g. before power-off,
|
||||
/// so a shutdown-time write isn't lost in the USB flash controller's cache.
|
||||
flush = 3,
|
||||
/// attach(): the caller's DMA-region capability rides the call's cap slot; the
|
||||
/// block server forwards it to the controller so the buffer's physical addresses
|
||||
/// (named in later read/write) are reachable by the device under an enforcing
|
||||
/// IOMMU. Call once per buffer before using it in a transfer.
|
||||
attach = 4,
|
||||
};
|
||||
|
||||
pub const Request = extern struct {
|
||||
|
||||
@@ -45,6 +45,11 @@ pub const Operation = enum(u32) {
|
||||
control = 1,
|
||||
interrupt_subscribe = 2,
|
||||
bulk = 3,
|
||||
/// dma_attach: a class driver hands the controller a DMA-region capability (riding
|
||||
/// the call's cap slot) so the controller binds that buffer into its IOMMU domain
|
||||
/// and may then DMA to the physical addresses inside it. Needed once per buffer the
|
||||
/// class driver will name in a `bulk` transfer (its own, or one forwarded to it).
|
||||
dma_attach = 4,
|
||||
};
|
||||
|
||||
/// The endpoint facts a class driver needs, lifted from the endpoint descriptor
|
||||
@@ -138,6 +143,19 @@ pub const BulkReply = extern struct {
|
||||
actual_length: u32,
|
||||
};
|
||||
|
||||
/// dma_attach: the region capability rides the call's cap slot; the body only carries
|
||||
/// the device token (scoping) so the controller knows which caller is attaching.
|
||||
pub const DmaAttachRequest = extern struct {
|
||||
operation: u32 = @intFromEnum(Operation.dma_attach),
|
||||
reserved: u32 = 0,
|
||||
device_token: u64,
|
||||
};
|
||||
|
||||
pub const DmaAttachReply = extern struct {
|
||||
status: i32,
|
||||
reserved: u32 = 0,
|
||||
};
|
||||
|
||||
/// An asynchronous interrupt report, pushed with `ipc.send` to a subscriber's
|
||||
/// endpoint. `Received.isMessage()` is set; there is no reply owed.
|
||||
pub const InterruptReport = extern struct {
|
||||
|
||||
@@ -76,6 +76,10 @@ pub const SystemCall = enum(u64) {
|
||||
fs_node = 47, // fs_node(op, node_token, offset, buf_ptr, buf_len) -> bytes/0/-errno: read/status/readdir on a kernel-served node (op values mirror the vfs-protocol Operation numbers)
|
||||
fs_mount = 48, // fs_mount(prefix_ptr, prefix_len, backend_handle, rewrite_ptr, rewrite_len) -> 0/-errno: mount a userspace filesystem's endpoint at an absolute prefix (possession of the handle is the capability)
|
||||
fs_unmount = 49, // fs_unmount(prefix_ptr, prefix_len) -> 0/-errno: remove a backend mount
|
||||
iommu_fault_drain = 50, // iommu_fault_drain() -> count: drain + log pending IOMMU translation faults (a diagnostic; the count of faults seen this call)
|
||||
dma_bind = 51, // dma_bind(device_id, region_handle) -> 0/-errno: map a DMA-region capability into the claimed device's IOMMU domain (idempotent). The caller must own the device and hold the handle
|
||||
dma_unbind = 52, // dma_unbind(device_id, region_handle) -> 0/-errno: unmap a previously bound region from the device's domain and invalidate
|
||||
handle_close = 53, // handle_close(handle) -> 0/-errno: drop one capability handle and free its table slot (endpoints, shared-memory, DMA regions)
|
||||
_,
|
||||
};
|
||||
|
||||
@@ -113,6 +117,7 @@ pub const msi_address_base: u64 = 0xFEE0_0000;
|
||||
pub const dma_coherent: u64 = 1; // strong-uncacheable — the default, the only portable one
|
||||
pub const dma_write_combining: u64 = 2; // write-combining (framebuffers); needs PAT
|
||||
pub const dma_below_4g: u64 = 4; // physical address must fit 32 bits (legacy DMA engines)
|
||||
pub const dma_shareable: u64 = 8; // return a capability handle (r8) so the region can be delegated + dma_bound
|
||||
|
||||
/// Set in the badge returned by `ipc_reply_wait` when what arrived is an
|
||||
/// **asynchronous notification** (a device interrupt bound with `irq_bind`, or a
|
||||
|
||||
@@ -0,0 +1,82 @@
|
||||
//! /system/drivers/intel-uhd-graphics-750 — spawned by the device manager with
|
||||
//! the device-tree id as argv[1]; claims that device and no other.
|
||||
|
||||
const std = @import("std");
|
||||
const device = @import("driver");
|
||||
const ipc = @import("ipc");
|
||||
const memory = @import("memory");
|
||||
const process = @import("process");
|
||||
const service = @import("service");
|
||||
|
||||
/// No protocol yet: the kernel's IPC ceiling (MESSAGE_MAXIMUM) sizes the buffers.
|
||||
const message_maximum = 256;
|
||||
|
||||
var controller_id: u64 = 0;
|
||||
var register_base: usize = 0;
|
||||
|
||||
fn initialise(endpoint: ipc.Handle) bool {
|
||||
_ = endpoint; // needed later, for irq binding and timers
|
||||
|
||||
if (!device.claim(controller_id)) {
|
||||
std.log.err("unable to claim device {d}", .{controller_id});
|
||||
return false;
|
||||
}
|
||||
|
||||
// Fetch our own descriptor back for the device's resources.
|
||||
const buffer = memory.allocator().alloc(device.DeviceDescriptor, 64) catch return false;
|
||||
defer memory.allocator().free(buffer);
|
||||
const total = device.enumerate(buffer);
|
||||
const descriptor = for (buffer[0..@min(total, buffer.len)]) |d| {
|
||||
if (d.id == controller_id) break d;
|
||||
} else {
|
||||
std.log.err("device {d} not in the device tree", .{controller_id});
|
||||
return false;
|
||||
};
|
||||
|
||||
// Log every resource BEFORE choosing one (see step 5).
|
||||
var register_index: u64 = 0;
|
||||
for (descriptor.resources[0..@intCast(descriptor.resource_count)], 0..) |resource, index| {
|
||||
std.log.info("resource {d}: kind={d} start=0x{x} len=0x{x}", .{
|
||||
index, resource.kind, resource.start, resource.len,
|
||||
});
|
||||
// The 16 MiB window is GTTMMADR, the register BAR (this device also has
|
||||
// a 256 MiB memory BAR, GMADR — "first memory resource" would be wrong).
|
||||
if (resource.kind == @intFromEnum(device.ResourceKind.memory) and
|
||||
resource.len == 16 * 1024 * 1024) register_index = index;
|
||||
}
|
||||
if (register_index == 0) {
|
||||
std.log.err("register BAR not found", .{});
|
||||
return false;
|
||||
}
|
||||
|
||||
register_base = device.mmioMap(controller_id, register_index) orelse {
|
||||
std.log.err("mmio_map failed", .{});
|
||||
return false;
|
||||
};
|
||||
std.log.info("registers mapped at 0x{x}", .{register_base});
|
||||
return true;
|
||||
}
|
||||
|
||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
|
||||
_ = message;
|
||||
_ = reply;
|
||||
_ = sender;
|
||||
_ = capability;
|
||||
return 0; // no protocol yet; the zero-length ping is answered by the harness
|
||||
}
|
||||
|
||||
pub fn main(init: process.Init) void {
|
||||
const argument = init.arguments.get(1) orelse {
|
||||
std.log.err("missing device id (argv[1])", .{});
|
||||
return;
|
||||
};
|
||||
controller_id = std.fmt.parseInt(u64, argument, 10) catch {
|
||||
std.log.err("malformed device id '{s}'", .{argument});
|
||||
return;
|
||||
};
|
||||
service.run(message_maximum, .{
|
||||
.init = initialise,
|
||||
.on_message = onMessage,
|
||||
// .on_notification only once an IRQ or timer is bound
|
||||
});
|
||||
}
|
||||
@@ -90,9 +90,22 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
bring_up_failed = true;
|
||||
return false;
|
||||
};
|
||||
command_wrapper = memory.dmaAlloc(4096, memory.dma_coherent) orelse return false;
|
||||
status_wrapper = memory.dmaAlloc(4096, memory.dma_coherent) orelse return false;
|
||||
command_data = memory.dmaAlloc(4096, memory.dma_coherent) orelse return false;
|
||||
// Shareable, so each buffer's capability can be handed to the controller: usb-storage
|
||||
// owns no device, so its buffers are not auto-bound anywhere — the controller reaches
|
||||
// them only once attached. (No-op binding when no IOMMU is enforcing.)
|
||||
command_wrapper = memory.dmaAlloc(4096, memory.dma_coherent | memory.dma_shareable) orelse return false;
|
||||
status_wrapper = memory.dmaAlloc(4096, memory.dma_coherent | memory.dma_shareable) orelse return false;
|
||||
command_data = memory.dmaAlloc(4096, memory.dma_coherent | memory.dma_shareable) orelse return false;
|
||||
for ([_]memory.DmaRegion{ command_wrapper, status_wrapper, command_data }) |region| {
|
||||
if (region.handle) |handle| {
|
||||
if (!device.attachDma(handle)) {
|
||||
_ = logging.write("/system/drivers/usb-storage: could not attach a DMA buffer to the controller\n");
|
||||
bring_up_failed = true;
|
||||
return false;
|
||||
}
|
||||
_ = ipc.close(handle); // the binding holds its own reference now
|
||||
}
|
||||
}
|
||||
|
||||
// Bring the LUN up: wait for it to be ready (clearing the initial unit-attention
|
||||
// with REQUEST SENSE), identify it, and read its capacity.
|
||||
@@ -135,10 +148,17 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
/// caller's DMA buffer (named by physical address).
|
||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
|
||||
_ = sender;
|
||||
_ = capability;
|
||||
if (message.len < block_protocol.request_size) return 0;
|
||||
const request = std.mem.bytesToValue(block_protocol.Request, message[0..block_protocol.request_size]);
|
||||
switch (request.operation) {
|
||||
@intFromEnum(block_protocol.Operation.attach) => {
|
||||
// The filesystem's DMA buffer: forward its capability to the controller so
|
||||
// the device can reach it, then release our copy (the binding holds a ref).
|
||||
const handle = capability orelse return writeReply(reply, .{ .status = -1, .block_size = 0, .block_count = 0 });
|
||||
const ok = device.attachDma(handle);
|
||||
_ = ipc.close(handle);
|
||||
return writeReply(reply, .{ .status = if (ok) 0 else -1, .block_size = 0, .block_count = 0 });
|
||||
},
|
||||
@intFromEnum(block_protocol.Operation.geometry) => {
|
||||
return writeReply(reply, .{ .status = 0, .block_size = block_size, .block_count = block_count });
|
||||
},
|
||||
|
||||
@@ -28,18 +28,37 @@ const usb_ids = @import("usb-ids");
|
||||
const usb_abi = @import("usb-abi");
|
||||
const usb_transfer_protocol = @import("usb-transfer-protocol");
|
||||
const library = @import("usb-xhci-library.zig");
|
||||
const pci = @import("pci");
|
||||
|
||||
/// The controller engine (reset, rings, transfers), stood up in `initialise`.
|
||||
var controller: ?library.Controller = null;
|
||||
|
||||
/// This driver's service endpoint (registered as `.usb_bus`), where class-driver
|
||||
/// requests, signals, and the interrupt-poll timer all arrive.
|
||||
/// requests, signals, MSI notifications, and the poll/reconcile timer all arrive.
|
||||
var service_endpoint: ipc.Handle = 0;
|
||||
|
||||
/// How often the driver drains the event ring for interrupt reports (~125 Hz),
|
||||
/// re-armed each tick. Frequent enough for responsive input.
|
||||
/// How often the driver drains the event ring for interrupt reports (~125 Hz) when
|
||||
/// polling, re-armed each tick. Frequent enough for responsive input.
|
||||
const poll_interval_ms: u64 = 8;
|
||||
|
||||
/// The timer interval in MSI mode: the ring is drained at interrupt time, and the tick
|
||||
/// only reconciles root ports (real hardware delivers late USB2 companion-hub debounce
|
||||
/// with no reliable port-change event — see onNotification) and un-wedges a lost MSI
|
||||
/// edge (edge-triggered, no kernel mask/ack: a missed IP clear stalls, never storms).
|
||||
const reconcile_interval_ms: u64 = 250;
|
||||
|
||||
/// The controller's own descriptor, kept at file scope because `pci.Function` holds a
|
||||
/// pointer to it for the whole bring-up.
|
||||
var controller_descriptor: device.DeviceDescriptor = undefined;
|
||||
|
||||
/// Non-null iff MSI mode is active: the vector whose notification badge means "the
|
||||
/// controller interrupted". Null means the 8 ms polling fallback is running.
|
||||
var msi_vector: ?u32 = null;
|
||||
|
||||
fn timerInterval() u64 {
|
||||
return if (msi_vector != null) reconcile_interval_ms else poll_interval_ms;
|
||||
}
|
||||
|
||||
/// The class driver endpoints that opened each device, so interrupt reports can
|
||||
/// be pushed back to them. Keyed by the device token (the interface's device id).
|
||||
const Open = struct {
|
||||
@@ -95,6 +114,7 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
std.log.info("device {d} not in the device tree", .{controller_id});
|
||||
return false;
|
||||
};
|
||||
controller_descriptor = descriptor;
|
||||
|
||||
// The xHC's registers live behind the first memory BAR. Resource 0 is the
|
||||
// function's ECAM configuration space (M15), so the walk starts at 1.
|
||||
@@ -118,6 +138,14 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
return false;
|
||||
};
|
||||
|
||||
// Message-signalled interrupt setup comes BEFORE the controller bring-up, not
|
||||
// after: Controller.init writes IMAN.IE, and QEMU's xhci only registers the MSI-X
|
||||
// vector as in-use when that write happens with MSI-X already enabled (its
|
||||
// intr_update callback early-outs on !msix_enabled, and msix_notify silently
|
||||
// drops interrupts for an unused vector). Real hardware does not care about the
|
||||
// order; QEMU requires it.
|
||||
setupMsi();
|
||||
|
||||
// Bring the controller up: reset it, stand up the command and event rings,
|
||||
// and start it running (the hardware half lives in usb-xhci-library.zig).
|
||||
controller = library.Controller.init(register_base) orelse {
|
||||
@@ -146,12 +174,52 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
|
||||
scanPorts(handle);
|
||||
|
||||
// Arm the poll timer that drains interrupt reports from the event ring. It is
|
||||
// re-armed on each tick in onNotification; class drivers subscribe later.
|
||||
_ = time.timerOnce(service_endpoint, poll_interval_ms);
|
||||
// Arm the timer: in polling mode it drains the event ring; in MSI mode it is the
|
||||
// slower port-reconcile/safety-net tick. Re-armed on each tick in onNotification.
|
||||
_ = time.timerOnce(service_endpoint, timerInterval());
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Switch the event ring from timer polling to message-signalled interrupts, if the
|
||||
/// whole path is available: map the function's config space, bind a vector, program
|
||||
/// the MSI capability — or, on an MSI-X-only function (QEMU's qemu-xhci is one: it
|
||||
/// advertises MSI-X and PCIe but no plain MSI), entry 0 of the MSI-X table, which
|
||||
/// takes the same kernel (address, data) pair (xHCI interrupter 0 raises vector 0).
|
||||
/// Any step failing leaves `msi_vector` null and the 8 ms polling path exactly as it
|
||||
/// was. The controller side needs nothing extra — IMAN.IE and USBCMD.INTE are already
|
||||
/// set (see Controller.init: QEMU only writes runtime events with the interrupter
|
||||
/// enabled).
|
||||
///
|
||||
/// After a supervised kill, the kernel drops the vector binding but the device still
|
||||
/// has the interrupt enabled and fires the stale vector; the kernel EOIs it
|
||||
/// harmlessly, and the respawned driver re-runs this with its fresh vector.
|
||||
fn setupMsi() void {
|
||||
var function = pci.Function.map(controller_id, &controller_descriptor) orelse {
|
||||
std.log.info("config-space map failed; polling at {d} ms", .{poll_interval_ms});
|
||||
return;
|
||||
};
|
||||
function.enableMemoryAndBusMaster();
|
||||
const message = device.msiBind(controller_id, service_endpoint) orelse {
|
||||
std.log.info("msi_bind unavailable; polling at {d} ms", .{poll_interval_ms});
|
||||
return;
|
||||
};
|
||||
if (function.programMsi(message)) {
|
||||
msi_vector = message.data;
|
||||
std.log.info("msi active (vector {d}); reconcile tick at {d} ms", .{ message.data, reconcile_interval_ms });
|
||||
return;
|
||||
}
|
||||
if (function.msix()) |table| {
|
||||
if (table.programEntry(0, message) and table.unmaskEntry(0)) {
|
||||
table.enable();
|
||||
function.setInterruptDisable();
|
||||
msi_vector = message.data;
|
||||
std.log.info("msix active (vector {d}); reconcile tick at {d} ms", .{ message.data, reconcile_interval_ms });
|
||||
return;
|
||||
}
|
||||
}
|
||||
std.log.info("no msi/msi-x capability; polling at {d} ms", .{poll_interval_ms});
|
||||
}
|
||||
|
||||
var register_base: usize = 0;
|
||||
|
||||
/// The xHCI default Protocol Speed IDs (the PORTSC port-speed field, bits 13:10)
|
||||
@@ -416,10 +484,22 @@ fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Han
|
||||
@intFromEnum(usb_transfer_protocol.Operation.control) => handleControl(message, reply),
|
||||
@intFromEnum(usb_transfer_protocol.Operation.interrupt_subscribe) => handleSubscribe(message, reply),
|
||||
@intFromEnum(usb_transfer_protocol.Operation.bulk) => handleBulk(message, reply),
|
||||
@intFromEnum(usb_transfer_protocol.Operation.dma_attach) => handleDmaAttach(message, reply, capability),
|
||||
else => 0,
|
||||
};
|
||||
}
|
||||
|
||||
/// dma_attach: bind the class driver's DMA-region capability into the controller's IOMMU
|
||||
/// domain, so the controller may DMA to the physical addresses inside that buffer. The
|
||||
/// binding holds its own kernel reference, so the forwarded capability is closed here.
|
||||
fn handleDmaAttach(message: []const u8, reply: []u8, capability: ?ipc.Handle) usize {
|
||||
if (message.len < @sizeOf(usb_transfer_protocol.DmaAttachRequest)) return writeReply(reply, usb_transfer_protocol.DmaAttachReply{ .status = -1 });
|
||||
const handle = capability orelse return writeReply(reply, usb_transfer_protocol.DmaAttachReply{ .status = -1 });
|
||||
const ok = device.dmaBind(controller_id, handle);
|
||||
_ = ipc.close(handle);
|
||||
return writeReply(reply, usb_transfer_protocol.DmaAttachReply{ .status = if (ok) 0 else -1 });
|
||||
}
|
||||
|
||||
fn writeReply(reply: []u8, value: anytype) usize {
|
||||
const bytes = std.mem.asBytes(&value);
|
||||
@memcpy(reply[0..bytes.len], bytes);
|
||||
@@ -501,10 +581,27 @@ fn handleBulk(message: []const u8, reply: []u8) usize {
|
||||
return writeReply(reply, usb_transfer_protocol.BulkReply{ .status = if (transferred != null) 0 else -1, .actual_length = transferred orelse 0 });
|
||||
}
|
||||
|
||||
/// The poll timer landed: drain any interrupt reports off the event ring and push
|
||||
/// each to the class driver that subscribed, then re-arm the timer.
|
||||
/// A timer tick or an MSI landed: drain the event ring, reconcile ports, and fan out.
|
||||
/// The timer arm re-arms itself (8 ms drain when polling, 250 ms reconcile under MSI);
|
||||
/// the MSI arm clears the interrupter's pending bit FIRST, then drains — so an event
|
||||
/// arriving after the drain takes IP 0→1 and fires a fresh edge instead of being
|
||||
/// swallowed until the reconcile tick.
|
||||
fn onNotification(badge: u64) void {
|
||||
if (badge & ipc.notify_timer_bit == 0) return;
|
||||
if (badge & ipc.notify_timer_bit != 0) {
|
||||
serviceController();
|
||||
_ = time.timerOnce(service_endpoint, timerInterval());
|
||||
return;
|
||||
}
|
||||
const vector = msi_vector orelse return;
|
||||
if (badge & ~ipc.notify_badge_bit != vector) return;
|
||||
if (controller) |*engine| engine.acknowledgeInterrupt();
|
||||
serviceController();
|
||||
}
|
||||
|
||||
/// Everything one servicing pass does, shared verbatim by the poll/reconcile tick and
|
||||
/// the MSI notification: drain the event ring, reconcile root ports, service hub
|
||||
/// changes, and push interrupt reports to their class drivers.
|
||||
fn serviceController() void {
|
||||
if (controller) |*engine| {
|
||||
engine.pump();
|
||||
// Poll every root port and reconcile — a device present but not yet
|
||||
@@ -564,7 +661,6 @@ fn onNotification(badge: u64) void {
|
||||
_ = ipc.send(report.report_endpoint, std.mem.asBytes(&message));
|
||||
}
|
||||
}
|
||||
_ = time.timerOnce(service_endpoint, poll_interval_ms);
|
||||
}
|
||||
|
||||
pub fn main(init: process.Init) void {
|
||||
|
||||
@@ -1563,6 +1563,15 @@ pub const Controller = struct {
|
||||
return report;
|
||||
}
|
||||
|
||||
/// Clear interrupter 0's pending bit (IMAN.IP). IP is write-1-to-clear, and the
|
||||
/// read-back carries IE (plain read-write) through unchanged. In MSI mode the
|
||||
/// driver clears IP **before** draining the ring: an event that lands after the
|
||||
/// drain then takes IP 0→1 and fires a fresh edge, where clearing afterwards would
|
||||
/// leave a race in which a new event finds IP already set and raises nothing.
|
||||
pub fn acknowledgeInterrupt(self: *const Controller) void {
|
||||
write32(self.interrupter(interrupter_management), read32(self.interrupter(interrupter_management)) | 1);
|
||||
}
|
||||
|
||||
/// Drain any events currently on the event ring: interrupt reports into the
|
||||
/// report queue, PORT STATUS CHANGES into the port-change queue (hot-plug —
|
||||
/// these were silently dropped before M20). Non-blocking — called on the
|
||||
|
||||
@@ -323,6 +323,15 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
std.log.info("could not resolve the scanout surface physical address", .{});
|
||||
return false;
|
||||
};
|
||||
// Bind the shared surface into this device's IOMMU domain so the GPU may DMA the
|
||||
// framebuffer (attach_backing points it here). The ring/command buffers are
|
||||
// dma_alloc'd and auto-bound; a shared-memory surface needs an explicit bind. We keep
|
||||
// the handle (it is also passed to the display service), so do not close it. No-op
|
||||
// without an IOMMU.
|
||||
if (!device.dmaBind(device_id, surface.handle)) {
|
||||
std.log.info("could not bind the scanout surface for DMA", .{});
|
||||
return false;
|
||||
}
|
||||
{
|
||||
const request = requestAt(vg.ResourceAttachBacking);
|
||||
request.* = .{
|
||||
|
||||
+147
-14
@@ -95,18 +95,52 @@ pub const PlatformInformation = struct {
|
||||
override_count: usize = 0,
|
||||
/// Whether an IOMMU (VT-d DMA-remapping unit) was found in the ACPI DMAR table.
|
||||
/// When false, `device_claim` on a DMA-capable device is equivalent to granting
|
||||
/// ring 0 — a device can DMA to any physical address (docs/driver-model.md M16).
|
||||
/// Detection is the first step; per-device domain enforcement lands with the first
|
||||
/// DMA driver.
|
||||
/// ring 0 — a device can DMA to any physical address (docs/driver-model.md M16),
|
||||
/// and the kernel says so at every boot (the fail-open platform log line). When
|
||||
/// true, the IOMMU core builds per-device translation domains from this record.
|
||||
iommu_present: bool = false,
|
||||
/// MMIO base of the first DMA-remapping hardware unit (DMAR DRHD), when present.
|
||||
/// true when the present unit is AMD-Vi (from IVRS) rather than Intel VT-d (DMAR).
|
||||
/// The two are mutually exclusive on real hardware; the IOMMU core picks the backend.
|
||||
iommu_is_amd: bool = false,
|
||||
/// MMIO base of the selected DMA-remapping hardware unit — the VT-d DRHD with
|
||||
/// INCLUDE_PCI_ALL (the catch-all unit; falls back to the first), or the AMD-Vi
|
||||
/// IOMMU's control-register base from the first IVHD.
|
||||
iommu_base: u64 = 0,
|
||||
/// The unit's Version register (offset 0x00) — its low byte is major.minor;
|
||||
/// reading it back nonzero confirms a real, mappable VT-d unit.
|
||||
iommu_version: u32 = 0,
|
||||
/// The unit's Capability register (offset 0x08): supported address widths, number
|
||||
/// of domains, etc. Recorded now; consumed when enforcement is built.
|
||||
/// of domains, etc. Consumed by the IOMMU core when it enables translation.
|
||||
iommu_capabilities: u64 = 0,
|
||||
/// Whether the selected unit carries INCLUDE_PCI_ALL. False means every unit is
|
||||
/// device-scoped (unusual) — the core still enables on the selected unit but
|
||||
/// devices outside its scope remain untranslated.
|
||||
iommu_include_all: bool = false,
|
||||
/// DRHD units in the DMAR beyond the selected one. Devices scoped to those units
|
||||
/// (typically the integrated GPU) are NOT translated by v1 — the boot log warns.
|
||||
iommu_extra_units: u8 = 0,
|
||||
/// Reserved-memory regions (DMAR RMRRs): firmware-owned buffers a named device
|
||||
/// keeps DMAing into across the OS handoff (classically the xHC keyboard-emulation
|
||||
/// buffer). These must be identity-mapped in the device's domain BEFORE translation
|
||||
/// enables, or platform firmware breaks. Only single-path endpoint scopes are
|
||||
/// recorded; anything fancier is skipped with a loud log at parse time.
|
||||
rmrr: [maximum_rmrr]RmrrRegion = undefined,
|
||||
rmrr_count: usize = 0,
|
||||
/// RMRR device scopes the parser could not record (multi-hop paths, sub-hierarchy
|
||||
/// types, or table overflow). Non-zero means a device keeps an unmapped firmware
|
||||
/// buffer — the kernel boot log warns loudly (the platform module itself is
|
||||
/// log-free by design; it records, the kernel reports).
|
||||
rmrr_skipped: u8 = 0,
|
||||
};
|
||||
|
||||
pub const maximum_rmrr = 8;
|
||||
|
||||
/// One recorded RMRR: the device (requester id) and the inclusive physical range it
|
||||
/// must always be allowed to reach.
|
||||
pub const RmrrRegion = struct {
|
||||
bdf: u16,
|
||||
base: u64,
|
||||
limit: u64,
|
||||
};
|
||||
|
||||
/// Filled in by `discover`; the architecture layer reads it during bring-up.
|
||||
@@ -248,6 +282,8 @@ const SLIT: [4]u8 = "SLIT".*;
|
||||
const SRAT: [4]u8 = "SRAT".*;
|
||||
/// Secondary System Description Table (SSDT)
|
||||
const DMAR: [4]u8 = "DMAR".*;
|
||||
/// I/O Virtualization Reporting Structure (IVRS) — the AMD-Vi analogue of DMAR.
|
||||
const IVRS: [4]u8 = "IVRS".*;
|
||||
const SSDT: [4]u8 = "SSDT".*;
|
||||
/// Serial Port Console Redirection table (SPCR) — the firmware's console UART.
|
||||
const SPCR: [4]u8 = "SPCR".*;
|
||||
@@ -467,6 +503,8 @@ fn handleTable(device_tree: *DeviceTree, hal: Hal, sdt_physical: u64) !void {
|
||||
parseSpcr(header);
|
||||
} else if (std.mem.eql(u8, &sig, &DMAR)) {
|
||||
parseDmar(hal, header);
|
||||
} else if (std.mem.eql(u8, &sig, &IVRS)) {
|
||||
parseIvrs(header);
|
||||
} else if (std.mem.eql(u8, &sig, &SSDT)) {
|
||||
// Secondary namespace bytecode — collect it to publish for the ring-3 parse.
|
||||
addAmlBlock(sdt_physical);
|
||||
@@ -757,18 +795,33 @@ fn parseSpcr(header: *const SystemDescriptorTableHeader) void {
|
||||
|
||||
// DMAR remapping-structure layout (Intel VT-d spec §8): the DMAR-specific header is 12
|
||||
// bytes (host-address-width, flags, 10 reserved), then a list of {type u16, length u16}
|
||||
// structures. Type 0 is a DRHD (DMA Remapping Hardware Unit Definition), whose 64-bit
|
||||
// register base sits at offset 8 within it.
|
||||
// structures. Type 0 is a DRHD (DMA Remapping Hardware Unit Definition): flags byte at
|
||||
// offset 4 (bit 0 = INCLUDE_PCI_ALL, the catch-all unit), 64-bit register base at
|
||||
// offset 8. Type 1 is an RMRR (Reserved Memory Region Reporting): a physical range at
|
||||
// offsets 8/16 (base / inclusive limit) that the device(s) named by the trailing
|
||||
// device-scope entries keep DMAing into across the firmware→OS handoff.
|
||||
const dmar_structures_offset = 48; // 36-byte ACPI header + 12-byte DMAR header
|
||||
const dmar_type_drhd: u16 = 0;
|
||||
const dmar_type_rmrr: u16 = 1;
|
||||
const drhd_flags_offset = 4;
|
||||
const drhd_include_pci_all: u8 = 1;
|
||||
const drhd_register_base_offset = 8;
|
||||
const rmrr_base_offset = 8;
|
||||
const rmrr_limit_offset = 16;
|
||||
const rmrr_scopes_offset = 24;
|
||||
// Device-scope entry (within DRHD/RMRR structures): type 1 = PCI endpoint; the path is
|
||||
// (device, function) byte pairs from offset 6, one pair per bridge hop plus the leaf.
|
||||
const scope_type_pci_endpoint: u8 = 1;
|
||||
const scope_start_bus_offset = 5;
|
||||
const scope_path_offset = 6;
|
||||
|
||||
/// DMAR -> detect the IOMMU. Find the first DMA-remapping hardware unit, map its
|
||||
/// register block, and record its version and capabilities. This is *detection only*:
|
||||
/// it tells the system an IOMMU exists (so `device_claim` on a DMA device could one day
|
||||
/// be gated by a per-device translation domain), but no domains are programmed yet —
|
||||
/// enforcement is built with the first DMA driver, which is what there is to protect and
|
||||
/// test against. See docs/driver-model.md (M16), the honest caveat.
|
||||
/// DMAR -> the VT-d unit(s) and reserved memory regions. Walks every remapping
|
||||
/// structure: selects the INCLUDE_PCI_ALL DRHD (the catch-all covering all devices not
|
||||
/// scoped elsewhere — commonly the SECOND unit on real machines, after an iGPU-scoped
|
||||
/// one), counts the rest so the boot log can warn that their devices stay untranslated,
|
||||
/// and records single-path endpoint RMRRs for the IOMMU core to pre-map before it
|
||||
/// enables translation. Multi-hop RMRR scopes are skipped loudly: better a named gap
|
||||
/// than a silent one.
|
||||
fn parseDmar(hal: Hal, header: *const SystemDescriptorTableHeader) void {
|
||||
const base: [*]align(1) const u8 = @ptrCast(header);
|
||||
const total: usize = header.length;
|
||||
@@ -780,13 +833,93 @@ fn parseDmar(hal: Hal, header: *const SystemDescriptorTableHeader) void {
|
||||
if (length < 4 or off + length > total) break; // malformed; stop rather than loop
|
||||
if (kind == dmar_type_drhd) {
|
||||
const register_base = fadt(u64, base, total, off + drhd_register_base_offset) orelse 0;
|
||||
const include_all = ((fadt(u8, base, total, off + drhd_flags_offset) orelse 0) & drhd_include_pci_all) != 0;
|
||||
if (register_base != 0) {
|
||||
// Selection: the INCLUDE_PCI_ALL unit wins; otherwise keep the first
|
||||
// seen. A later catch-all replaces an earlier scoped unit.
|
||||
const replace = !platform_information.iommu_present or
|
||||
(include_all and !platform_information.iommu_include_all);
|
||||
if (replace) {
|
||||
if (platform_information.iommu_present) platform_information.iommu_extra_units += 1;
|
||||
const regs = hal.mapMmio(register_base, abi.page_size, true);
|
||||
platform_information.iommu_present = true;
|
||||
platform_information.iommu_base = register_base;
|
||||
platform_information.iommu_include_all = include_all;
|
||||
platform_information.iommu_version = @as(*const volatile u32, @ptrFromInt(regs + 0x00)).*;
|
||||
platform_information.iommu_capabilities = @as(*const volatile u64, @ptrFromInt(regs + 0x08)).*;
|
||||
return; // first unit is enough for detection; multi-unit is future
|
||||
} else {
|
||||
platform_information.iommu_extra_units += 1;
|
||||
}
|
||||
}
|
||||
} else if (kind == dmar_type_rmrr) {
|
||||
parseRmrr(base, total, off, length);
|
||||
}
|
||||
off += length;
|
||||
}
|
||||
}
|
||||
|
||||
/// One RMRR structure: record a {bdf, base, limit} per single-path endpoint scope.
|
||||
fn parseRmrr(base: [*]align(1) const u8, total: usize, off: usize, length: usize) void {
|
||||
const range_base = fadt(u64, base, total, off + rmrr_base_offset) orelse return;
|
||||
const range_limit = fadt(u64, base, total, off + rmrr_limit_offset) orelse return;
|
||||
if (range_limit < range_base) return;
|
||||
|
||||
var scope = off + rmrr_scopes_offset;
|
||||
const end = off + length;
|
||||
while (scope + 6 <= end) {
|
||||
const scope_type = fadt(u8, base, total, scope) orelse break;
|
||||
const scope_length = fadt(u8, base, total, scope + 1) orelse break;
|
||||
if (scope_length < 6 or scope + scope_length > end) break;
|
||||
if (scope_type == scope_type_pci_endpoint and scope_length == scope_path_offset + 2) {
|
||||
// Single (device, function) pair: a directly-reachable endpoint.
|
||||
const bus = fadt(u8, base, total, scope + scope_start_bus_offset) orelse 0;
|
||||
const device = fadt(u8, base, total, scope + scope_path_offset) orelse 0;
|
||||
const function = fadt(u8, base, total, scope + scope_path_offset + 1) orelse 0;
|
||||
if (platform_information.rmrr_count < maximum_rmrr) {
|
||||
platform_information.rmrr[platform_information.rmrr_count] = .{
|
||||
.bdf = (@as(u16, bus) << 8) | (@as(u16, device) << 3) | function,
|
||||
.base = range_base,
|
||||
.limit = range_limit,
|
||||
};
|
||||
platform_information.rmrr_count += 1;
|
||||
} else {
|
||||
platform_information.rmrr_skipped +|= 1; // table full
|
||||
}
|
||||
} else {
|
||||
platform_information.rmrr_skipped +|= 1; // multi-hop path or non-endpoint scope
|
||||
}
|
||||
scope += scope_length;
|
||||
}
|
||||
}
|
||||
|
||||
// IVRS layout (AMD I/O Virtualization spec): 36-byte ACPI header, IVinfo u32 @36,
|
||||
// 8 reserved @40, then IVHD/IVMD blocks from @48. An IVHD common header is type u8 @0,
|
||||
// flags u8 @1, length u16 @2, device id u16 @4, capability offset u16 @6, IOMMU base
|
||||
// address u64 @8, PCI segment u16 @16, IOMMU info u16 @18.
|
||||
const ivrs_blocks_offset = 48;
|
||||
const ivhd_type_10: u8 = 0x10;
|
||||
const ivhd_type_11: u8 = 0x11;
|
||||
const ivhd_base_offset = 8;
|
||||
|
||||
/// IVRS -> detect an AMD-Vi IOMMU. Record the control-register base from the first IVHD
|
||||
/// of type 0x10/0x11. Per-device entries and IVMD (the AMD analogue of RMRR) are ignored
|
||||
/// in v1 — the default-deny device table is what we build anyway, and QEMU emits no IVMD;
|
||||
/// a real machine that needs them is flagged untested on AMD regardless.
|
||||
fn parseIvrs(header: *const SystemDescriptorTableHeader) void {
|
||||
const base: [*]align(1) const u8 = @ptrCast(header);
|
||||
const total: usize = header.length;
|
||||
var off: usize = ivrs_blocks_offset;
|
||||
while (off + 4 <= total) {
|
||||
const kind = fadt(u8, base, total, off) orelse break;
|
||||
const length = fadt(u16, base, total, off + 2) orelse break;
|
||||
if (length < 4 or off + length > total) break;
|
||||
if (kind == ivhd_type_10 or kind == ivhd_type_11) {
|
||||
const iommu_base = fadt(u64, base, total, off + ivhd_base_offset) orelse 0;
|
||||
if (iommu_base != 0) {
|
||||
platform_information.iommu_present = true;
|
||||
platform_information.iommu_is_amd = true;
|
||||
platform_information.iommu_base = iommu_base;
|
||||
return; // first IVHD is enough; multi-unit is future work
|
||||
}
|
||||
}
|
||||
off += length;
|
||||
|
||||
@@ -167,6 +167,20 @@ pub fn releaseAllOwnedBy(owner: u32) void {
|
||||
}
|
||||
}
|
||||
|
||||
/// Release the claim on `id` iff `owner` holds it — the rollback for a claim that
|
||||
/// cannot be confined (the IOMMU domain could not be created/attached). Returns true
|
||||
/// when a claim was actually cleared.
|
||||
pub fn unclaim(id: u64, owner: u32) bool {
|
||||
if (id >= count) return false;
|
||||
if (claimed[@intCast(id)]) |o| {
|
||||
if (o == owner) {
|
||||
claimed[@intCast(id)] = null;
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/// Resource `index` of device `id`, or null if out of range.
|
||||
pub fn resourceOf(id: u64, index: u64) ?device_abi.ResourceDescriptor {
|
||||
if (id >= count) return null;
|
||||
@@ -175,6 +189,50 @@ pub fn resourceOf(id: u64, index: u64) ?device_abi.ResourceDescriptor {
|
||||
return d.resources[@intCast(index)];
|
||||
}
|
||||
|
||||
/// The PCI requester id (bus<<8 | device<<3 | function) of device `id`, derived from
|
||||
/// its config-space slice against its host bridge's ECAM window — the identity a VT-d
|
||||
/// context entry / AMD-Vi DTE is keyed by. null when `id` is not a PCI function or the
|
||||
/// geometry doesn't decode. The kernel never stored the BDF (the descriptor has no such
|
||||
/// field); pci-bus encodes it into resource 0's physical base as
|
||||
/// `ecam_base + ((bus - start_bus) << 20 | device << 15 | function << 12)`, and the
|
||||
/// requester id the device emits uses the absolute bus, so we add `start_bus << 8` back.
|
||||
pub fn pciAddressOf(id: u64) ?u16 {
|
||||
if (id >= count) return null;
|
||||
const d = &devices[@intCast(id)];
|
||||
if (d.class != @intFromEnum(device_abi.DeviceClass.pci_device)) return null;
|
||||
if (d.resource_count == 0) return null;
|
||||
const config = d.resources[0];
|
||||
if (config.kind != @intFromEnum(device_abi.ResourceKind.memory) or config.len != 4096) return null;
|
||||
|
||||
// Walk up to the host bridge, whose resource 0 is the segment's ECAM window and
|
||||
// resource 1 the bus_range (start_bus, bus_count).
|
||||
var parent = d.parent;
|
||||
while (parent != device_abi.no_parent and parent < count) {
|
||||
const p = &devices[@intCast(parent)];
|
||||
if (p.class == @intFromEnum(device_abi.DeviceClass.pci_host_bridge)) {
|
||||
if (p.resource_count < 2) return null;
|
||||
const ecam = p.resources[0];
|
||||
const bus_range = p.resources[1];
|
||||
if (config.start < ecam.start or config.start >= ecam.start + ecam.len) return null;
|
||||
const offset = config.start - ecam.start;
|
||||
const start_bus: u16 = @intCast(bus_range.start & 0xFF);
|
||||
return @intCast((offset >> 12) + (@as(u64, start_bus) << 8));
|
||||
}
|
||||
parent = p.parent;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Call `visit(id, bdf)` for every PCI function in the table — the IOMMU core's boot
|
||||
/// sweep to place every device under a domain. Only functions whose BDF decodes are
|
||||
/// visited.
|
||||
pub fn forEachPciFunction(visit: *const fn (id: u64, bdf: u16) void) void {
|
||||
var id: u64 = 0;
|
||||
while (id < count) : (id += 1) {
|
||||
if (pciAddressOf(id)) |bdf| visit(id, bdf);
|
||||
}
|
||||
}
|
||||
|
||||
/// Is `child` wholly inside `parent`? For a range (memory, io_port, bus_range) that's
|
||||
/// interval containment; for an irq it's equality, since an interrupt line is not
|
||||
/// divisible. Zero-length child ranges are refused — an empty window is meaningless
|
||||
|
||||
@@ -0,0 +1,259 @@
|
||||
//! system/kernel/iommu-amd.zig — AMD-Vi (AMD I/O Virtualization) backend for the IOMMU
|
||||
//! core. The AMD analogue of iommu-intel.zig: it supplies the core's `Backend` vtable
|
||||
//! with AMD-Vi's page-table entry bits and drives the device table, command buffer, and
|
||||
//! event log.
|
||||
//!
|
||||
//! **UNTESTED on real AMD hardware.** danos is developed on Intel; this backend is
|
||||
//! validated only against QEMU's `-device amd-iommu,dma-remap=on`, whose AMD-Vi
|
||||
//! emulation is far less exercised than its Intel one. Every code path here should be
|
||||
//! read as "QEMU-verified, real-AMD-unverified" until an AMD machine confirms it.
|
||||
//!
|
||||
//! Interrupt remapping is left off (the DTE forwards interrupts unmapped), so MSI writes
|
||||
//! to the 0xFEE00000 range reach the APIC untranslated, exactly as on the Intel path.
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const boot_handoff = @import("boot-handoff");
|
||||
const pmm = @import("pmm.zig");
|
||||
const platform = @import("platform");
|
||||
const architecture = @import("architecture");
|
||||
const log = @import("log.zig");
|
||||
const iommu = @import("iommu.zig");
|
||||
|
||||
const page_size = abi.page_size;
|
||||
|
||||
// MMIO register offsets from the IOMMU control-register base.
|
||||
const reg_device_table_base = 0x00; // base | (pages-1) in bits 8:0
|
||||
const reg_command_buffer_base = 0x08; // base | (ComLen << 56)
|
||||
const reg_event_log_base = 0x10; // base | (EventLen << 56)
|
||||
const reg_control = 0x18;
|
||||
const reg_command_head = 0x2000;
|
||||
const reg_command_tail = 0x2008;
|
||||
const reg_event_head = 0x2010;
|
||||
const reg_event_tail = 0x2018;
|
||||
const reg_status = 0x2020;
|
||||
|
||||
const control_iommu_enable: u64 = 1 << 0;
|
||||
const control_event_log_enable: u64 = 1 << 2;
|
||||
const control_command_buffer_enable: u64 = 1 << 12;
|
||||
|
||||
// Device table: one 32-byte DTE (4 qwords) per requester id, indexed by bdf.
|
||||
const device_table_pages = 512; // 2 MiB = 65536 entries (a full 256-bus segment)
|
||||
const dte_qwords = 4;
|
||||
const dte_valid: u64 = 1 << 0; // V
|
||||
const dte_translation_valid: u64 = 1 << 1; // TV
|
||||
const dte_mode_shift = 9; // bits 11:9 — page-table levels
|
||||
const dte_read: u64 = 1 << 61; // IR
|
||||
const dte_write: u64 = 1 << 62; // IW
|
||||
const dte_intctl_forward: u64 = @as(u64, 1) << 60; // qword2 bits 61:60 = 01b: forward interrupts unmapped
|
||||
|
||||
// Page-table entry bits (AMD native format).
|
||||
const pte_present: u64 = 1 << 0; // PR
|
||||
const pte_next_level_shift = 9; // bits 11:9: 0 = leaf, N = pointer to a level-N table
|
||||
const pte_read: u64 = 1 << 61; // IR
|
||||
const pte_write: u64 = 1 << 62; // IW
|
||||
const address_mask: u64 = 0x000F_FFFF_FFFF_F000;
|
||||
|
||||
// Command buffer / event log: one 4 KiB frame each = 256 entries.
|
||||
const ring_entries = 256;
|
||||
const ring_length_code: u64 = 8; // log2(256), the ComLen/EventLen field value
|
||||
const command_opcode_shift = 60; // opcode in bits 63:60 of qword 0
|
||||
const command_completion_wait: u64 = 0x01;
|
||||
const command_invalidate_devtab: u64 = 0x02;
|
||||
const command_invalidate_pages: u64 = 0x03;
|
||||
const completion_wait_store: u64 = 1 << 1; // S: store `data` to the supplied address
|
||||
const invalidate_pages_all: u64 = 0x000F_FFFF_FFFF_F000 | 1; // address bits 51:12 all-ones + S
|
||||
|
||||
const levels: u8 = 4; // 48-bit IOVA, matching the Intel 4-level path
|
||||
|
||||
var register_base: usize = 0;
|
||||
var device_table: u64 = 0; // physical base of the device table
|
||||
var command_buffer: u64 = 0;
|
||||
var event_log: u64 = 0;
|
||||
var completion_frame: u64 = 0; // COMPLETION_WAIT store target
|
||||
var command_tail: u32 = 0; // our software copy of the command tail (bytes)
|
||||
|
||||
var fault_log_budget: u32 = 32;
|
||||
var completion_warned = false;
|
||||
|
||||
fn read64(offset: usize) u64 {
|
||||
return @as(*const volatile u64, @ptrFromInt(register_base + offset)).*;
|
||||
}
|
||||
fn write64(offset: usize, value: u64) void {
|
||||
@as(*volatile u64, @ptrFromInt(register_base + offset)).* = value;
|
||||
}
|
||||
fn ram(physical: u64) [*]volatile u64 {
|
||||
return @ptrFromInt(boot_handoff.physicalToVirtual(physical));
|
||||
}
|
||||
|
||||
/// Map the register window, allocate the device table / command buffer / event log.
|
||||
/// Returns the vtable, or null if the boot-time allocations fail.
|
||||
pub fn detect(info: platform.PlatformInformation) ?iommu.Backend {
|
||||
register_base = architecture.mapMmio(info.iommu_base, 16 * 1024, true);
|
||||
|
||||
device_table = pmm.allocContiguous(device_table_pages, ~@as(u64, 0)) orelse return null;
|
||||
zero(device_table, device_table_pages); // all-zero DTE = V=0 = deny every device
|
||||
command_buffer = allocZeroedFrame() orelse return null;
|
||||
event_log = allocZeroedFrame() orelse return null;
|
||||
completion_frame = allocZeroedFrame() orelse return null;
|
||||
|
||||
return iommu.Backend{
|
||||
.levels = levels,
|
||||
.supports_huge_pages = false, // 4 KiB leaves only (AMD superpage encoding deferred)
|
||||
.makeLeaf = makeLeaf,
|
||||
.makeTable = makeTable,
|
||||
.isPresent = isPresent,
|
||||
.flushStructure = flushStructure,
|
||||
.attach = attach,
|
||||
.detach = detach,
|
||||
.invalidateDomain = invalidateDomain,
|
||||
.faultDrain = faultDrain,
|
||||
};
|
||||
}
|
||||
|
||||
/// Program the base registers and enable translation. The device table is already
|
||||
/// zeroed (every device denied) except any entries `attach` wrote for RMRR/claimed
|
||||
/// devices, so turning translation on blocks all other DMA and logs it.
|
||||
pub fn enable() void {
|
||||
write64(reg_device_table_base, (device_table & address_mask) | (device_table_pages - 1));
|
||||
write64(reg_command_buffer_base, (command_buffer & address_mask) | (ring_length_code << 56));
|
||||
write64(reg_event_log_base, (event_log & address_mask) | (ring_length_code << 56));
|
||||
write64(reg_command_head, 0);
|
||||
write64(reg_command_tail, 0);
|
||||
write64(reg_event_head, 0);
|
||||
write64(reg_event_tail, 0);
|
||||
command_tail = 0;
|
||||
// Buffers first, then the master enable.
|
||||
write64(reg_control, control_command_buffer_enable | control_event_log_enable);
|
||||
write64(reg_control, control_command_buffer_enable | control_event_log_enable | control_iommu_enable);
|
||||
}
|
||||
|
||||
// --- Backend vtable ------------------------------------------------------------------
|
||||
|
||||
fn makeLeaf(physical: u64, huge: bool) u64 {
|
||||
_ = huge; // 4 KiB only
|
||||
return (physical & address_mask) | pte_present | pte_read | pte_write; // Next Level 0 = leaf
|
||||
}
|
||||
fn makeTable(table_physical: u64, level: u8) u64 {
|
||||
// This entry (at `level`) points to a table one level down; AMD's Next Level names
|
||||
// the pointed-to table's level.
|
||||
const next_level: u64 = @as(u64, level) - 1;
|
||||
return (table_physical & address_mask) | pte_present | pte_read | pte_write | (next_level << pte_next_level_shift);
|
||||
}
|
||||
fn isPresent(entry: u64) bool {
|
||||
return (entry & pte_present) != 0;
|
||||
}
|
||||
fn flushStructure(address: usize) void {
|
||||
_ = address; // AMD-Vi reads its structures coherently
|
||||
}
|
||||
|
||||
fn attach(bdf: u16, domain: u16, page_table_root: u64) void {
|
||||
const dte = ram(device_table) + @as(usize, bdf) * dte_qwords;
|
||||
dte[0] = (page_table_root & address_mask) | dte_valid | dte_translation_valid |
|
||||
(@as(u64, levels) << dte_mode_shift) | dte_read | dte_write;
|
||||
dte[1] = @as(u64, domain); // DomainID in bits 15:0
|
||||
dte[2] = dte_intctl_forward; // forward interrupts unmapped (no remapping)
|
||||
dte[3] = 0;
|
||||
invalidateDevice(bdf);
|
||||
invalidateDomain(domain);
|
||||
}
|
||||
|
||||
fn detach(bdf: u16) void {
|
||||
const dte = ram(device_table) + @as(usize, bdf) * dte_qwords;
|
||||
dte[0] = 0; // V=0: deny
|
||||
dte[1] = 0;
|
||||
dte[2] = 0;
|
||||
dte[3] = 0;
|
||||
invalidateDevice(bdf);
|
||||
}
|
||||
|
||||
fn invalidateDomain(domain: u16) void {
|
||||
submitCommand((command_invalidate_pages << command_opcode_shift) | (@as(u64, domain) << 32), invalidate_pages_all);
|
||||
completeAndWait();
|
||||
}
|
||||
|
||||
fn faultDrain() usize {
|
||||
const tail = read64(reg_event_tail) & 0xFFFF_FFF0;
|
||||
var head = read64(reg_event_head) & 0xFFFF_FFF0;
|
||||
if (head == tail) return 0;
|
||||
var seen: usize = 0;
|
||||
while (head != tail) {
|
||||
const entry = ram(event_log) + (head / 8);
|
||||
const code: u4 = @truncate(entry[0] >> command_opcode_shift);
|
||||
if (code == 0x2) { // IO_PAGE_FAULT
|
||||
const source: u16 = @truncate(entry[0]);
|
||||
logFault(source, entry[1]);
|
||||
}
|
||||
seen += 1;
|
||||
head += 16;
|
||||
if (head >= ring_entries * 16) head = 0;
|
||||
}
|
||||
write64(reg_event_head, head);
|
||||
return seen;
|
||||
}
|
||||
|
||||
fn logFault(source: u16, address: u64) void {
|
||||
if (fault_log_budget == 0) return;
|
||||
fault_log_budget -= 1;
|
||||
log.print("DANOS-IOMMU-FAULT: bdf={x:0>2}:{x:0>2}.{d} addr=0x{x} reason=amd-io-page-fault\n", .{
|
||||
source >> 8,
|
||||
(source >> 3) & 0x1F,
|
||||
source & 0x7,
|
||||
address,
|
||||
});
|
||||
if (fault_log_budget == 0) log.write("DANOS-IOMMU-FAULT: (further faults suppressed)\n");
|
||||
}
|
||||
|
||||
// --- command ring --------------------------------------------------------------------
|
||||
|
||||
fn invalidateDevice(bdf: u16) void {
|
||||
submitCommand((command_invalidate_devtab << command_opcode_shift) | @as(u64, bdf), 0);
|
||||
completeAndWait();
|
||||
}
|
||||
|
||||
/// Append a 128-bit command (two qwords) to the ring and advance the tail.
|
||||
fn submitCommand(qword0: u64, qword1: u64) void {
|
||||
const slot = ram(command_buffer) + (command_tail / 8);
|
||||
slot[0] = qword0;
|
||||
slot[1] = qword1;
|
||||
command_tail += 16;
|
||||
if (command_tail >= ring_entries * 16) command_tail = 0;
|
||||
write64(reg_command_tail, command_tail);
|
||||
}
|
||||
|
||||
/// Append a COMPLETION_WAIT (store form) and spin until the IOMMU writes our sentinel to
|
||||
/// the completion frame. QEMU consumes the command buffer synchronously on the tail-
|
||||
/// register write, so by the time we poll the prior invalidation is already applied; the
|
||||
/// store confirmation is belt-and-suspenders for real hardware. If it never lands
|
||||
/// (QEMU's amd-iommu does not implement the store form), warn ONCE and proceed — the
|
||||
/// invalidation itself has happened.
|
||||
fn completeAndWait() void {
|
||||
const sentinel: u64 = 0xC0FFEE;
|
||||
ram(completion_frame)[0] = 0;
|
||||
submitCommand(
|
||||
(command_completion_wait << command_opcode_shift) | (completion_frame & 0x000F_FFFF_FFFF_FFF8) | completion_wait_store,
|
||||
sentinel,
|
||||
);
|
||||
var spins: u64 = 0;
|
||||
while (@as(*const volatile u64, @ptrFromInt(boot_handoff.physicalToVirtual(completion_frame))).* != sentinel) {
|
||||
spins += 1;
|
||||
if (spins > 100_000) {
|
||||
if (!completion_warned) {
|
||||
completion_warned = true;
|
||||
log.write("/system/kernel: AMD-Vi COMPLETION_WAIT store not observed — proceeding (QEMU processes commands synchronously)\n");
|
||||
}
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn allocZeroedFrame() ?u64 {
|
||||
const frame = pmm.alloc() orelse return null;
|
||||
zero(frame, 1);
|
||||
return frame;
|
||||
}
|
||||
fn zero(physical: u64, pages: usize) void {
|
||||
const words = ram(physical);
|
||||
var i: usize = 0;
|
||||
while (i < pages * page_size / 8) : (i += 1) words[i] = 0;
|
||||
}
|
||||
@@ -0,0 +1,314 @@
|
||||
//! system/kernel/iommu-intel.zig — Intel VT-d backend for the IOMMU core.
|
||||
//!
|
||||
//! Provides the core (iommu.zig) with the VT-d hardware specifics behind its `Backend`
|
||||
//! vtable: second-level page-table entry bits, the root/context table structure, the
|
||||
//! translation-enable and invalidation register sequences, and the fault drain. The
|
||||
//! core owns the domain table and the page-table walk; this file owns the registers.
|
||||
//!
|
||||
//! Register model (VT-d spec §10-11): offsets from the DRHD register base. The unit is
|
||||
//! programmed once at enable (root table + Translation Enable), then touched only for
|
||||
//! per-device context changes, per-domain invalidations, and fault draining. Interrupt
|
||||
//! remapping is deliberately left OFF (GCMD.IRE stays 0): with it off, upstream writes
|
||||
//! to 0xFEE0_0000-0xFEEF_FFFF are treated as interrupt requests and bypass second-level
|
||||
//! translation, so the existing MSI contract survives unchanged.
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const boot_handoff = @import("boot-handoff");
|
||||
const pmm = @import("pmm.zig");
|
||||
const platform = @import("platform");
|
||||
const architecture = @import("architecture");
|
||||
const log = @import("log.zig");
|
||||
const iommu = @import("iommu.zig");
|
||||
|
||||
const page_size = abi.page_size;
|
||||
|
||||
// Register offsets from the unit's base.
|
||||
const reg_cap = 0x08; // Capability (64)
|
||||
const reg_ecap = 0x10; // Extended Capability (64)
|
||||
const reg_gcmd = 0x18; // Global Command (32, write-only)
|
||||
const reg_gsts = 0x1C; // Global Status (32, read-only)
|
||||
const reg_rtaddr = 0x20; // Root Table Address (64)
|
||||
const reg_ccmd = 0x28; // Context Command (64)
|
||||
const reg_fsts = 0x34; // Fault Status (32)
|
||||
|
||||
const gcmd_te: u32 = 1 << 31; // Translation Enable
|
||||
const gcmd_srtp: u32 = 1 << 30; // Set Root Table Pointer
|
||||
const gsts_tes: u32 = 1 << 31; // Translation Enable Status
|
||||
const gsts_rtps: u32 = 1 << 30; // Root Table Pointer Status
|
||||
|
||||
const cap_cm: u64 = 1 << 7; // Caching Mode
|
||||
const cap_sagaw_shift = 8; // Supported Adjusted Guest Address Widths, bits 12:8
|
||||
const cap_sagaw_39bit: u64 = 1 << 9; // 3-level
|
||||
const cap_sagaw_48bit: u64 = 1 << 10; // 4-level
|
||||
const cap_fro_shift = 24; // Fault-Recording Register Offset, bits 33:24 (×16)
|
||||
const cap_nfr_shift = 40; // Number of Fault Recording regs, bits 47:40 (+1)
|
||||
const ecap_coherent: u64 = 1 << 0; // hardware snoops CPU caches reading its structures
|
||||
const ecap_iro_shift = 8; // IOTLB Register Offset, bits 17:8 (×16)
|
||||
|
||||
const ccmd_icc: u64 = 1 << 63; // Invalidate Context-Cache
|
||||
const ccmd_cirg_global: u64 = @as(u64, 1) << 61; // global granularity
|
||||
const ccmd_cirg_device: u64 = @as(u64, 3) << 61; // device-selective
|
||||
|
||||
const iotlb_ivt: u64 = 1 << 63; // Invalidate IOTLB
|
||||
const iotlb_iirg_global: u64 = @as(u64, 1) << 60;
|
||||
const iotlb_iirg_domain: u64 = @as(u64, 2) << 60;
|
||||
const iotlb_dr: u64 = 1 << 49; // drain reads
|
||||
const iotlb_dw: u64 = 1 << 48; // drain writes
|
||||
|
||||
const fsts_ppf: u32 = 1 << 1; // Primary Pending Fault
|
||||
|
||||
// Second-level PTE bits.
|
||||
const slpte_read: u64 = 1 << 0;
|
||||
const slpte_write: u64 = 1 << 1;
|
||||
const slpte_page_size: u64 = 1 << 7; // a 2 MiB leaf (== iommu.huge_leaf_bit)
|
||||
|
||||
const address_mask: u64 = 0x000F_FFFF_FFFF_F000;
|
||||
|
||||
var register_base: usize = 0;
|
||||
var capabilities: u64 = 0;
|
||||
var extended_capabilities: u64 = 0;
|
||||
var coherent: bool = true; // ECAP.C — whether clflush is unnecessary
|
||||
var levels: u8 = 4;
|
||||
var context_aw: u64 = 2; // context-entry AW field (001=3-level, 010=4-level)
|
||||
var gcmd_shadow: u32 = 0; // sticky GCMD bits (TE etc.), for the write-only register
|
||||
|
||||
var root_table: u64 = 0; // physical base of the 256-entry root table
|
||||
var context_table: [256]u64 = .{0} ** 256; // per-bus context table physical, 0 = none
|
||||
|
||||
var fault_log_budget: u32 = 32; // rate-limit: log this many faults, then just count
|
||||
var faults_suppressed: u64 = 0;
|
||||
|
||||
fn read32(offset: usize) u32 {
|
||||
return @as(*const volatile u32, @ptrFromInt(register_base + offset)).*;
|
||||
}
|
||||
fn write32(offset: usize, value: u32) void {
|
||||
@as(*volatile u32, @ptrFromInt(register_base + offset)).* = value;
|
||||
}
|
||||
fn read64(offset: usize) u64 {
|
||||
return @as(*const volatile u64, @ptrFromInt(register_base + offset)).*;
|
||||
}
|
||||
fn write64(offset: usize, value: u64) void {
|
||||
@as(*volatile u64, @ptrFromInt(register_base + offset)).* = value;
|
||||
}
|
||||
|
||||
fn tableAt(physical: u64) [*]volatile u64 {
|
||||
return @ptrFromInt(boot_handoff.physicalToVirtual(physical));
|
||||
}
|
||||
|
||||
/// Map the register window, read caps, pick the address width. Returns the vtable, or
|
||||
/// null when the unit advertises no address width danos can drive.
|
||||
pub fn detect(info: platform.PlatformInformation) ?iommu.Backend {
|
||||
// Remap 16 KiB: FRCD and IOTLB registers can sit past the first page (CAP.FRO /
|
||||
// ECAP.IRO are 16-byte-unit offsets). Idempotent with the detection-time mapping.
|
||||
register_base = architecture.mapMmio(info.iommu_base, 16 * 1024, true);
|
||||
capabilities = read64(reg_cap);
|
||||
extended_capabilities = read64(reg_ecap);
|
||||
coherent = (extended_capabilities & ecap_coherent) != 0;
|
||||
|
||||
const sagaw = capabilities >> cap_sagaw_shift;
|
||||
if (sagaw & cap_sagaw_48bit != 0) {
|
||||
levels = 4;
|
||||
context_aw = 2; // 010b
|
||||
} else if (sagaw & cap_sagaw_39bit != 0) {
|
||||
levels = 3;
|
||||
context_aw = 1; // 001b
|
||||
} else {
|
||||
return null; // no width we build tables for
|
||||
}
|
||||
|
||||
root_table = allocZeroed() orelse return null;
|
||||
|
||||
return iommu.Backend{
|
||||
.levels = levels,
|
||||
.supports_huge_pages = true,
|
||||
.makeLeaf = makeLeaf,
|
||||
.makeTable = makeTable,
|
||||
.isPresent = isPresent,
|
||||
.flushStructure = flushStructure,
|
||||
.attach = attach,
|
||||
.detach = detach,
|
||||
.invalidateDomain = invalidateDomain,
|
||||
.faultDrain = faultDrain,
|
||||
};
|
||||
}
|
||||
|
||||
/// Program the root table and turn Translation Enable on. The core has already created
|
||||
/// and populated the RMRR domains (their context entries are live via `attach`), so at
|
||||
/// this instant every OTHER device's context entry is not-present and will fault — which
|
||||
/// for stale firmware bus-mastering is the desired evidence, not a bug.
|
||||
pub fn enable() void {
|
||||
write64(reg_rtaddr, root_table); // legacy mode (bits 11:10 = 00)
|
||||
setGlobalCommand(gcmd_srtp);
|
||||
spinStatus(gsts_rtps);
|
||||
globalInvalidate();
|
||||
setGlobalCommand(gcmd_te);
|
||||
spinStatus(gsts_tes);
|
||||
gcmd_shadow |= gcmd_te;
|
||||
}
|
||||
|
||||
// --- Backend vtable ------------------------------------------------------------------
|
||||
|
||||
fn makeLeaf(physical: u64, huge: bool) u64 {
|
||||
return (physical & address_mask) | slpte_read | slpte_write | (if (huge) slpte_page_size else 0);
|
||||
}
|
||||
fn makeTable(table_physical: u64, level: u8) u64 {
|
||||
_ = level;
|
||||
return (table_physical & address_mask) | slpte_read | slpte_write;
|
||||
}
|
||||
fn isPresent(entry: u64) bool {
|
||||
return (entry & (slpte_read | slpte_write)) != 0;
|
||||
}
|
||||
|
||||
fn flushStructure(address: usize) void {
|
||||
if (coherent) return; // the unit snoops CPU caches; no flush needed (QEMU)
|
||||
asm volatile ("clflush (%[p])"
|
||||
:
|
||||
: [p] "r" (address),
|
||||
: .{ .memory = true });
|
||||
}
|
||||
|
||||
fn attach(bdf: u16, domain: u16, page_table_root: u64) void {
|
||||
const bus: u8 = @intCast(bdf >> 8);
|
||||
const devfn: u8 = @intCast(bdf & 0xFF);
|
||||
|
||||
// Lazily allocate this bus's context table and link it into the root table.
|
||||
if (context_table[bus] == 0) {
|
||||
const table = allocZeroed() orelse return;
|
||||
context_table[bus] = table;
|
||||
const root_entry = &tableAt(root_table)[@as(usize, bus) * 2]; // 16-byte entries
|
||||
root_entry.* = (table & address_mask) | 1; // present
|
||||
flushStructure(@intFromPtr(root_entry));
|
||||
}
|
||||
|
||||
const context = tableAt(context_table[bus]);
|
||||
const low = &context[@as(usize, devfn) * 2];
|
||||
const high = &context[@as(usize, devfn) * 2 + 1];
|
||||
high.* = (context_aw & 0x7) | (@as(u64, domain) << 8); // AW + DID
|
||||
low.* = (page_table_root & address_mask) | 1; // present, TT=00 (use second-level)
|
||||
flushStructure(@intFromPtr(high));
|
||||
flushStructure(@intFromPtr(low));
|
||||
|
||||
invalidateContextDevice(bdf, domain);
|
||||
invalidateDomain(domain);
|
||||
}
|
||||
|
||||
fn detach(bdf: u16) void {
|
||||
const bus: u8 = @intCast(bdf >> 8);
|
||||
const devfn: u8 = @intCast(bdf & 0xFF);
|
||||
if (context_table[bus] == 0) return;
|
||||
const context = tableAt(context_table[bus]);
|
||||
context[@as(usize, devfn) * 2] = 0; // not present
|
||||
context[@as(usize, devfn) * 2 + 1] = 0;
|
||||
flushStructure(@intFromPtr(&context[@as(usize, devfn) * 2]));
|
||||
invalidateContextDevice(bdf, 0);
|
||||
globalIotlb();
|
||||
}
|
||||
|
||||
fn invalidateDomain(domain: u16) void {
|
||||
const iotlb_offset = iotlbOffset();
|
||||
write64(iotlb_offset, iotlb_ivt | iotlb_iirg_domain | iotlb_dr | iotlb_dw | (@as(u64, domain) << 32));
|
||||
spin64(iotlb_offset, iotlb_ivt);
|
||||
}
|
||||
|
||||
fn faultDrain() usize {
|
||||
const fsts = read32(reg_fsts);
|
||||
if (fsts & fsts_ppf == 0) return 0;
|
||||
|
||||
const fro = (capabilities >> cap_fro_shift) & 0x3FF;
|
||||
const nfr = ((capabilities >> cap_nfr_shift) & 0xFF) + 1;
|
||||
const frcd_base = @as(usize, @intCast(fro)) * 16;
|
||||
|
||||
var seen: usize = 0;
|
||||
var i: usize = 0;
|
||||
while (i < nfr) : (i += 1) {
|
||||
const off = frcd_base + i * 16;
|
||||
const high = read64(off + 8);
|
||||
if (high & (@as(u64, 1) << 63) == 0) continue; // F: no fault recorded here
|
||||
const low = read64(off);
|
||||
const address = low & ~@as(u64, 0xFFF);
|
||||
const source: u16 = @intCast(high & 0xFFFF);
|
||||
const reason: u8 = @intCast((high >> 32) & 0xFF);
|
||||
const is_read = (high >> 62) & 1; // T: 1 = read request
|
||||
logFault(source, address, reason, is_read == 1);
|
||||
write64(off + 8, @as(u64, 1) << 63); // RW1C: clear F
|
||||
seen += 1;
|
||||
}
|
||||
write32(reg_fsts, fsts); // clear PPF/PFO (write-1-to-clear)
|
||||
return seen;
|
||||
}
|
||||
|
||||
fn logFault(source: u16, address: u64, reason: u8, is_read: bool) void {
|
||||
if (fault_log_budget > 0) {
|
||||
fault_log_budget -= 1;
|
||||
log.print("DANOS-IOMMU-FAULT: bdf={x:0>2}:{x:0>2}.{d} addr=0x{x} reason=0x{x} write={d}\n", .{
|
||||
source >> 8,
|
||||
(source >> 3) & 0x1F,
|
||||
source & 0x7,
|
||||
address,
|
||||
reason,
|
||||
@intFromBool(!is_read),
|
||||
});
|
||||
if (fault_log_budget == 0)
|
||||
log.write("DANOS-IOMMU-FAULT: (further faults suppressed)\n");
|
||||
} else {
|
||||
faults_suppressed += 1;
|
||||
}
|
||||
}
|
||||
|
||||
// --- register helpers ----------------------------------------------------------------
|
||||
|
||||
fn setGlobalCommand(one_shot: u32) void {
|
||||
// GCMD is write-only: every write must carry the full sticky state plus the one-shot
|
||||
// bit being requested, or a set sticky bit (TE) would be cleared as a side effect.
|
||||
write32(reg_gcmd, gcmd_shadow | one_shot);
|
||||
}
|
||||
|
||||
fn spinStatus(bit: u32) void {
|
||||
var spins: u64 = 0;
|
||||
while (read32(reg_gsts) & bit == 0) {
|
||||
spins += 1;
|
||||
if (spins > 10_000_000) {
|
||||
log.write("/system/kernel: WARNING VT-d status bit never set — translation may be incomplete\n");
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn spin64(offset: usize, bit: u64) void {
|
||||
var spins: u64 = 0;
|
||||
while (read64(offset) & bit != 0) {
|
||||
spins += 1;
|
||||
if (spins > 10_000_000) return;
|
||||
}
|
||||
}
|
||||
|
||||
fn globalInvalidate() void {
|
||||
write64(reg_ccmd, ccmd_icc | ccmd_cirg_global);
|
||||
spin64(reg_ccmd, ccmd_icc);
|
||||
globalIotlb();
|
||||
}
|
||||
|
||||
fn globalIotlb() void {
|
||||
const iotlb_offset = iotlbOffset();
|
||||
write64(iotlb_offset, iotlb_ivt | iotlb_iirg_global | iotlb_dr | iotlb_dw);
|
||||
spin64(iotlb_offset, iotlb_ivt);
|
||||
}
|
||||
|
||||
fn invalidateContextDevice(bdf: u16, domain: u16) void {
|
||||
write64(reg_ccmd, ccmd_icc | ccmd_cirg_device | (@as(u64, bdf) << 16) | domain);
|
||||
spin64(reg_ccmd, ccmd_icc);
|
||||
}
|
||||
|
||||
fn iotlbOffset() usize {
|
||||
const iro = (extended_capabilities >> ecap_iro_shift) & 0x3FF;
|
||||
return @as(usize, @intCast(iro)) * 16 + 8; // IOTLB register sits at IRO*16 + 8
|
||||
}
|
||||
|
||||
fn allocZeroed() ?u64 {
|
||||
const frame = pmm.alloc() orelse return null;
|
||||
const table = tableAt(frame);
|
||||
var i: usize = 0;
|
||||
while (i < 512) : (i += 1) table[i] = 0;
|
||||
return frame;
|
||||
}
|
||||
@@ -0,0 +1,448 @@
|
||||
//! system/kernel/iommu.zig — vendor-neutral IOMMU core: per-device DMA translation
|
||||
//! domains over an Intel VT-d or AMD-Vi backend.
|
||||
//!
|
||||
//! The problem this closes: without an IOMMU, a claimed bus-mastering device can DMA to
|
||||
//! ANY physical address, so a compromised or buggy driver reaches all of memory through
|
||||
//! its device — driver isolation stops at the CPU's MMU. This core gives each claimed
|
||||
//! PCI function its own translation domain; a device reaches only the physical ranges
|
||||
//! mapped into its domain, and nothing else (kernel, page tables, other processes) is
|
||||
//! visible to it.
|
||||
//!
|
||||
//! Design:
|
||||
//! - **Identity mappings** (IOVA == physical). `dma_alloc` already hands drivers the
|
||||
//! physical address they program into hardware; a domain simply makes that same
|
||||
//! address the ONLY thing the device can reach. No IOVA allocator, and every
|
||||
//! driver's register-programming code is untouched.
|
||||
//! - **Vendor-neutral**: this file owns the domain table and a shared 512-entry
|
||||
//! page-table walker; a `Backend` vtable supplies the hardware specifics (VT-d in
|
||||
//! iommu-intel.zig, AMD-Vi in iommu-amd.zig) — the entry-bit encodings, the
|
||||
//! enable/invalidate register dances, and the fault drain.
|
||||
//! - **Fail-open**: when no IOMMU is found, `kind == .none` and every entry point is a
|
||||
//! success no-op, so callers in process.zig stay unconditional and behavior is
|
||||
//! byte-for-byte the pre-IOMMU kernel. The boot log states the posture.
|
||||
//!
|
||||
//! All entry points run under the big kernel lock (the caller holds it); no internal
|
||||
//! locking. All memory comes from `pmm` reached through the physmap, like paging.zig.
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const boot_handoff = @import("boot-handoff");
|
||||
const pmm = @import("pmm.zig");
|
||||
const platform = @import("platform");
|
||||
const devices_broker = @import("devices-broker.zig");
|
||||
const log = @import("log.zig");
|
||||
const intel = @import("iommu-intel.zig");
|
||||
const amd = @import("iommu-amd.zig");
|
||||
|
||||
const page_size: u64 = abi.page_size;
|
||||
const page_mask: u64 = page_size - 1;
|
||||
const huge_page_size: u64 = 2 * 1024 * 1024;
|
||||
|
||||
pub const Kind = enum { none, intel_vtd, amd_vi };
|
||||
|
||||
/// One domain per claimed PCI function. 64 mirrors devices-broker's device cap.
|
||||
pub const maximum_domains = 64;
|
||||
pub const invalid_domain: u16 = 0xFFFF;
|
||||
|
||||
/// The bit encodings and hardware operations a backend supplies to the shared core.
|
||||
/// Entry helpers build the raw page-table entries for the backend's format; the core
|
||||
/// walks the tree with them. The hardware ops act on a whole domain (identified by its
|
||||
/// hardware domain id = core index + 1) or device (by requester id / bdf).
|
||||
pub const Backend = struct {
|
||||
/// Number of page-table levels (3 or 4) the backend selected from hardware caps.
|
||||
levels: u8,
|
||||
/// Largest leaf the walker may emit: 4 KiB always, 2 MiB when the backend allows.
|
||||
supports_huge_pages: bool,
|
||||
|
||||
/// Raw entry bits for a leaf mapping `physical` (with the given size), and for a
|
||||
/// non-leaf entry pointing at `table_physical` at `level` (level counts down to 1
|
||||
/// at the leaf's parent). `isPresent` tests a read-back entry.
|
||||
makeLeaf: *const fn (physical: u64, huge: bool) u64,
|
||||
makeTable: *const fn (table_physical: u64, level: u8) u64,
|
||||
isPresent: *const fn (entry: u64) bool,
|
||||
/// Flush a cache line holding IOMMU structures the hardware reads non-coherently
|
||||
/// (VT-d with ECAP.C==0). A no-op where the unit snoops caches.
|
||||
flushStructure: *const fn (address: usize) void,
|
||||
|
||||
/// Point `bdf`'s translation structure at `domain` (hardware id) and invalidate the
|
||||
/// context/device caches so the change takes effect.
|
||||
attach: *const fn (bdf: u16, domain: u16, page_table_root: u64) void,
|
||||
/// Return `bdf`'s translation structure to not-present + invalidate — all its DMA
|
||||
/// faults afterward.
|
||||
detach: *const fn (bdf: u16) void,
|
||||
/// Invalidate cached translations for `domain` (after a map or unmap).
|
||||
invalidateDomain: *const fn (domain: u16) void,
|
||||
/// Pull pending faults out of the hardware, log them (rate-limited), return the
|
||||
/// count seen this call.
|
||||
faultDrain: *const fn () usize,
|
||||
};
|
||||
|
||||
const Domain = struct {
|
||||
in_use: bool = false,
|
||||
owner: u32 = 0, // task that owns the attached device
|
||||
bdf: u16 = 0, // requester id of the attached device
|
||||
page_table_root: u64 = 0, // physical address of the top-level table
|
||||
rmrr: bool = false, // a firmware reserved-region domain (persists across claims)
|
||||
};
|
||||
|
||||
var kind: Kind = .none;
|
||||
var backend: Backend = undefined;
|
||||
var domains: [maximum_domains]Domain = .{Domain{}} ** maximum_domains;
|
||||
|
||||
pub fn kindOf() Kind {
|
||||
return kind;
|
||||
}
|
||||
pub fn enabled() bool {
|
||||
return kind != .none;
|
||||
}
|
||||
|
||||
/// Detect the IOMMU, pick a backend, pre-map firmware reserved regions, and enable
|
||||
/// translation. Fail-open (kind stays .none) when no unit exists — the caller logs the
|
||||
/// posture. Must run after platform discovery and before any user process starts.
|
||||
pub fn init() void {
|
||||
const info = platform.platformInformation();
|
||||
if (!info.iommu_present) {
|
||||
kind = .none;
|
||||
return;
|
||||
}
|
||||
// Pick the backend by vendor. A present-but-unusable unit stays fail-open with a
|
||||
// logged reason rather than half-enabling.
|
||||
if (info.iommu_is_amd) {
|
||||
if (amd.detect(info)) |be| {
|
||||
backend = be;
|
||||
kind = .amd_vi;
|
||||
} else {
|
||||
kind = .none;
|
||||
log.write("/system/kernel: WARNING AMD-Vi present but unusable — staying fail-open\n");
|
||||
return;
|
||||
}
|
||||
} else {
|
||||
if (intel.detect(info)) |be| {
|
||||
backend = be;
|
||||
kind = .intel_vtd;
|
||||
} else {
|
||||
kind = .none;
|
||||
log.write("/system/kernel: WARNING IOMMU present but unusable — staying fail-open\n");
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
// The translation structures start empty: every device is denied until its driver
|
||||
// claims it (confineDevice gives it a private domain). PCI functions are enumerated
|
||||
// post-boot by the ring-3 pci-bus driver, so there is nothing to attach at init.
|
||||
if (kind == .amd_vi) amd.enable() else intel.enable();
|
||||
logEnabled(info);
|
||||
}
|
||||
|
||||
/// Per-claimed-device record: its private domain, so a driver's death tears down
|
||||
/// exactly the domains it held.
|
||||
const Confined = struct { active: bool = false, owner: u32 = 0, bdf: u16 = 0, domain: u16 = invalid_domain };
|
||||
var confined: [maximum_domains]Confined = .{Confined{}} ** maximum_domains;
|
||||
|
||||
/// Place a just-claimed PCI function under IOMMU translation on behalf of `owner`: give
|
||||
/// it a private empty domain, seed it with the device's own firmware reserved region,
|
||||
/// and attach. Its DMA buffers arrive afterward as explicit grants — the owner's own
|
||||
/// `dma_alloc`'d regions are bound by the claim path (`mapForDevice`), and cross-process
|
||||
/// buffers by `dma_bind`. false only if a domain can't be allocated — the caller rolls
|
||||
/// the claim back (a claim that can't be confined must not stand). No-op success when no
|
||||
/// IOMMU exists (fail-open).
|
||||
pub fn confineDevice(device_id: u64, bdf: u16, owner: u32) bool {
|
||||
if (kind == .none) return true;
|
||||
if (device_id >= confined.len) return true; // unusual id; leave it to fail-open
|
||||
const domain = domainCreate(owner, bdf) orelse return false;
|
||||
|
||||
// Firmware reserved region for this device, if any (real hardware; QEMU has none).
|
||||
const info = platform.platformInformation();
|
||||
var i: usize = 0;
|
||||
while (i < info.rmrr_count) : (i += 1) {
|
||||
if (info.rmrr[i].bdf == bdf)
|
||||
_ = map(domain, info.rmrr[i].base, info.rmrr[i].limit - info.rmrr[i].base + 1);
|
||||
}
|
||||
|
||||
attachDevice(domain, bdf);
|
||||
confined[@intCast(device_id)] = .{ .active = true, .owner = owner, .bdf = bdf, .domain = domain };
|
||||
return true;
|
||||
}
|
||||
|
||||
/// The confined record for `device_id`, or null if the device is not confined.
|
||||
fn confinedOf(device_id: u64) ?*Confined {
|
||||
if (device_id >= confined.len) return null;
|
||||
const c = &confined[@intCast(device_id)];
|
||||
return if (c.active) c else null;
|
||||
}
|
||||
|
||||
/// Map a DMA region into a specific claimed device's domain (the device owner binding a
|
||||
/// granted buffer). false if the device is not confined. No-op success without an IOMMU.
|
||||
pub fn mapForDevice(device_id: u64, physical: u64, len: u64) bool {
|
||||
if (kind == .none) return true;
|
||||
const c = confinedOf(device_id) orelse return false;
|
||||
return map(c.domain, physical, len);
|
||||
}
|
||||
|
||||
/// Unmap a DMA region from a specific claimed device's domain. No-op if not confined.
|
||||
pub fn unmapForDevice(device_id: u64, physical: u64, len: u64) void {
|
||||
if (kind == .none) return;
|
||||
const c = confinedOf(device_id) orelse return;
|
||||
unmap(c.domain, physical, len);
|
||||
}
|
||||
|
||||
/// Map a region into every claimed device owned by `owner` — the auto-bind of a task's
|
||||
/// own freshly-`dma_alloc`'d buffer into the devices it drives.
|
||||
pub fn mapRegionForOwner(owner: u32, physical: u64, len: u64) void {
|
||||
if (kind == .none) return;
|
||||
for (&confined) |*c| {
|
||||
if (c.active and c.owner == owner) _ = map(c.domain, physical, len);
|
||||
}
|
||||
}
|
||||
|
||||
/// Unmap a region from EVERY claimed device's domain — the freed-region sweep. MUST run
|
||||
/// before the frames return to pmm: a device translating to a reallocated frame is the
|
||||
/// use-after-free this prevents. Cross-device because a granted buffer may be bound in a
|
||||
/// domain other than its owner's.
|
||||
pub fn unmapRegionEverywhere(physical: u64, len: u64) void {
|
||||
if (kind == .none) return;
|
||||
for (&confined) |*c| {
|
||||
if (c.active) unmap(c.domain, physical, len);
|
||||
}
|
||||
}
|
||||
|
||||
/// A driver died or released its devices: tear down every domain it held (detach the
|
||||
/// device, free the tables) so their DMA is blocked again and a restarted driver
|
||||
/// re-claims cleanly. Runs BEFORE the broker claims and the DMA frames are released.
|
||||
pub fn releaseAllOwnedBy(owner: u32) void {
|
||||
if (kind == .none) return;
|
||||
for (&confined) |*c| {
|
||||
if (c.active and c.owner == owner) {
|
||||
detachDevice(c.bdf);
|
||||
domainDestroy(c.domain);
|
||||
c.* = .{};
|
||||
}
|
||||
}
|
||||
_ = faultDrain(); // log any faults a mid-DMA device raised as it was cut off
|
||||
}
|
||||
|
||||
/// Allocate an empty domain (an empty top-level table). null when the table is full.
|
||||
pub fn domainCreate(owner: u32, bdf: u16) ?u16 {
|
||||
if (kind == .none) return 0; // fail-open: a dummy id the no-op ops ignore
|
||||
for (&domains, 0..) |*d, index| {
|
||||
if (d.in_use) continue;
|
||||
const root = allocTable() orelse return null;
|
||||
d.* = .{ .in_use = true, .owner = owner, .bdf = bdf, .page_table_root = root };
|
||||
return @intCast(index);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Free a domain's page-table frames and its slot. Precondition: no device attached
|
||||
/// (detach first).
|
||||
pub fn domainDestroy(domain: u16) void {
|
||||
if (kind == .none) return;
|
||||
const d = &domains[domain];
|
||||
if (!d.in_use) return;
|
||||
freeTables(d.page_table_root, backend.levels);
|
||||
d.* = .{};
|
||||
}
|
||||
|
||||
/// Attach `bdf`'s device to `domain` and pre-load any RMRR range recorded for it.
|
||||
pub fn attachDevice(domain: u16, bdf: u16) void {
|
||||
if (kind == .none) return;
|
||||
const d = &domains[domain];
|
||||
d.bdf = bdf;
|
||||
backend.attach(bdf, hardwareId(domain), d.page_table_root);
|
||||
}
|
||||
|
||||
/// Return `bdf`'s device to not-present + invalidate.
|
||||
pub fn detachDevice(bdf: u16) void {
|
||||
if (kind == .none) return;
|
||||
backend.detach(bdf);
|
||||
}
|
||||
|
||||
/// Identity-map [physical, physical+len) into `domain` (read+write) and invalidate.
|
||||
/// Unconditional domain-selective invalidation after every map — correct under VT-d
|
||||
/// caching-mode and free otherwise.
|
||||
pub fn map(domain: u16, physical: u64, len: u64) bool {
|
||||
if (kind == .none) return true;
|
||||
const d = &domains[domain];
|
||||
if (!d.in_use) return false;
|
||||
if (!mapRange(d.page_table_root, physical, len)) return false;
|
||||
backend.invalidateDomain(hardwareId(domain));
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Unmap [physical, physical+len) from `domain` and invalidate. MUST finish its
|
||||
/// invalidation before the caller returns the frames to pmm — a stale IOTLB entry
|
||||
/// pointing at a reallocated frame is the use-after-free this ordering prevents.
|
||||
pub fn unmap(domain: u16, physical: u64, len: u64) void {
|
||||
if (kind == .none) return;
|
||||
const d = &domains[domain];
|
||||
if (!d.in_use) return;
|
||||
unmapRange(d.page_table_root, physical, len);
|
||||
backend.invalidateDomain(hardwareId(domain));
|
||||
}
|
||||
|
||||
/// Poll the hardware for translation faults, log them, return the count. Called by the
|
||||
/// IOMMU test case and opportunistically after a device detaches.
|
||||
pub fn faultDrain() usize {
|
||||
if (kind == .none) return 0;
|
||||
return backend.faultDrain();
|
||||
}
|
||||
|
||||
/// The physical address `virtual` maps to in `domain`, or null if unmapped — a test
|
||||
/// helper that walks the domain's page tables (identity mappings return `virtual`).
|
||||
pub fn translationOf(domain: u16, virtual: u64) ?u64 {
|
||||
if (kind == .none) return virtual;
|
||||
const d = &domains[domain];
|
||||
if (!d.in_use) return null;
|
||||
var table = d.page_table_root;
|
||||
var level = backend.levels;
|
||||
while (level > 1) : (level -= 1) {
|
||||
const entry = tableAt(table)[indexAt(virtual, level)];
|
||||
if (!backend.isPresent(entry)) return null;
|
||||
if (level == 2 and isHugeLeaf(entry))
|
||||
return (entry & address_mask) | (virtual & (huge_page_size - 1));
|
||||
table = entry & address_mask;
|
||||
}
|
||||
const leaf = tableAt(table)[indexAt(virtual, 1)];
|
||||
if (!backend.isPresent(leaf)) return null;
|
||||
return (leaf & address_mask) | (virtual & page_mask);
|
||||
}
|
||||
|
||||
// --- the shared page-table walker -----------------------------------------------------
|
||||
// 512-entry, 9-bits-per-level, 4 KiB tables reached through the physmap — the shape both
|
||||
// VT-d second-level and AMD-Vi native tables share. The backend supplies the entry bits.
|
||||
|
||||
fn tableAt(physical: u64) [*]volatile u64 {
|
||||
return @ptrFromInt(boot_handoff.physicalToVirtual(physical));
|
||||
}
|
||||
|
||||
fn allocTable() ?u64 {
|
||||
const frame = pmm.alloc() orelse return null;
|
||||
const table = tableAt(frame);
|
||||
var i: usize = 0;
|
||||
while (i < 512) : (i += 1) table[i] = 0;
|
||||
return frame;
|
||||
}
|
||||
|
||||
const address_mask: u64 = 0x000F_FFFF_FFFF_F000;
|
||||
|
||||
fn indexAt(virtual: u64, level: u8) usize {
|
||||
// level 1 is the leaf table; shift = 12 + 9*(level-1).
|
||||
const shift: u6 = @intCast(12 + 9 * (@as(u32, level) - 1));
|
||||
return @intCast((virtual >> shift) & 0x1FF);
|
||||
}
|
||||
|
||||
/// Descend to (allocating) the next-level table below `entry_ptr`, returning its
|
||||
/// physical base. null on out-of-memory.
|
||||
fn descend(entry_ptr: *volatile u64, level: u8) ?u64 {
|
||||
const entry = entry_ptr.*;
|
||||
if (backend.isPresent(entry)) return entry & address_mask;
|
||||
const table = allocTable() orelse return null;
|
||||
backend.flushStructure(@intFromPtr(tableAt(table)));
|
||||
entry_ptr.* = backend.makeTable(table, level);
|
||||
backend.flushStructure(@intFromPtr(entry_ptr));
|
||||
return table;
|
||||
}
|
||||
|
||||
fn mapRange(root: u64, physical: u64, len: u64) bool {
|
||||
const start = physical & ~page_mask;
|
||||
const end = (physical + len + page_mask) & ~page_mask;
|
||||
var addr = start;
|
||||
while (addr < end) {
|
||||
// 2 MiB leaf when the backend allows it and both address and remaining span are
|
||||
// huge-aligned — keeps table memory sane for the blanket-identity and real-PC
|
||||
// cases without a separate superpage path per backend.
|
||||
const huge = backend.supports_huge_pages and
|
||||
addr % huge_page_size == 0 and (end - addr) >= huge_page_size;
|
||||
if (!mapOne(root, addr, huge)) return false;
|
||||
addr += if (huge) huge_page_size else page_size;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
fn mapOne(root: u64, addr: u64, huge: bool) bool {
|
||||
const leaf_level: u8 = if (huge) 2 else 1;
|
||||
var table = root;
|
||||
var level = backend.levels;
|
||||
while (level > leaf_level) : (level -= 1) {
|
||||
const entry_ptr = &tableAt(table)[indexAt(addr, level)];
|
||||
table = descend(entry_ptr, level) orelse return false;
|
||||
}
|
||||
const leaf_ptr = &tableAt(table)[indexAt(addr, leaf_level)];
|
||||
leaf_ptr.* = backend.makeLeaf(addr, huge);
|
||||
backend.flushStructure(@intFromPtr(leaf_ptr));
|
||||
return true;
|
||||
}
|
||||
|
||||
fn unmapRange(root: u64, physical: u64, len: u64) void {
|
||||
const start = physical & ~page_mask;
|
||||
const end = (physical + len + page_mask) & ~page_mask;
|
||||
var addr = start;
|
||||
while (addr < end) {
|
||||
const huge = backend.supports_huge_pages and
|
||||
addr % huge_page_size == 0 and (end - addr) >= huge_page_size;
|
||||
unmapOne(root, addr, huge);
|
||||
addr += if (huge) huge_page_size else page_size;
|
||||
}
|
||||
}
|
||||
|
||||
fn unmapOne(root: u64, addr: u64, huge: bool) void {
|
||||
const leaf_level: u8 = if (huge) 2 else 1;
|
||||
var table = root;
|
||||
var level = backend.levels;
|
||||
while (level > leaf_level) : (level -= 1) {
|
||||
const entry = tableAt(table)[indexAt(addr, level)];
|
||||
if (!backend.isPresent(entry)) return; // nothing mapped here
|
||||
table = entry & address_mask;
|
||||
}
|
||||
const leaf_ptr = &tableAt(table)[indexAt(addr, leaf_level)];
|
||||
leaf_ptr.* = 0;
|
||||
backend.flushStructure(@intFromPtr(leaf_ptr));
|
||||
}
|
||||
|
||||
/// Post-order free of a domain's whole table tree.
|
||||
fn freeTables(root: u64, level: u8) void {
|
||||
if (level > 1) {
|
||||
const table = tableAt(root);
|
||||
var i: usize = 0;
|
||||
while (i < 512) : (i += 1) {
|
||||
const entry = table[i];
|
||||
if (!backend.isPresent(entry)) continue;
|
||||
// A 2 MiB leaf sits at level 2 and points at RAM, not a sub-table.
|
||||
if (level == 2 and isHugeLeaf(entry)) continue;
|
||||
freeTables(entry & address_mask, level - 1);
|
||||
}
|
||||
}
|
||||
pmm.free(root);
|
||||
}
|
||||
|
||||
fn isHugeLeaf(entry: u64) bool {
|
||||
// Both backends set a page-size bit (VT-d bit 7, AMD leaf next-level=0 at level 2).
|
||||
// The backend's makeLeaf encodes it; the walker only needs "is this a leaf, not a
|
||||
// pointer" at level 2, which huge leaves are by construction.
|
||||
return entry & huge_leaf_bit != 0;
|
||||
}
|
||||
|
||||
/// The size-bit the backends set on a 2 MiB leaf (VT-d SL-PTE PS bit 7; AMD encodes a
|
||||
/// leaf as next-level 0, so the core marks huge leaves with this software bit — an
|
||||
/// ignored bit in both formats — to tell them apart from table pointers when freeing).
|
||||
pub const huge_leaf_bit: u64 = 1 << 7;
|
||||
|
||||
fn hardwareId(domain: u16) u16 {
|
||||
return domain + 1; // id 0 is reserved by both architectures
|
||||
}
|
||||
|
||||
fn logEnabled(info: platform.PlatformInformation) void {
|
||||
if (kind == .amd_vi) {
|
||||
log.write("/system/kernel: iommu online (AMD-Vi) — UNTESTED on real AMD hardware (QEMU-verified only)\n");
|
||||
log.print(" levels : {d} (48-bit)\n", .{backend.levels});
|
||||
return;
|
||||
}
|
||||
log.write("/system/kernel: iommu online (Intel VT-d)\n");
|
||||
log.print(" version : 0x{x}\n", .{info.iommu_version});
|
||||
log.print(" agaw : {d} levels\n", .{backend.levels});
|
||||
log.print(" rmrr : {d} region(s) premapped\n", .{info.rmrr_count});
|
||||
if (info.rmrr_skipped > 0)
|
||||
log.print(" rmrr : WARNING {d} scope(s) skipped — a device keeps an unmapped firmware buffer\n", .{info.rmrr_skipped});
|
||||
if (info.iommu_extra_units > 0)
|
||||
log.print(" units : WARNING {d} other DRHD(s) — their scoped devices are NOT translated\n", .{info.iommu_extra_units});
|
||||
}
|
||||
@@ -154,6 +154,66 @@ pub fn dropRef(endpoint: *Endpoint) void {
|
||||
/// Defined here (not in scheduler) because the meaning is the IPC/capability layer's.
|
||||
pub const handle_kind_endpoint: u8 = 0;
|
||||
pub const handle_kind_shared_memory: u8 = 1;
|
||||
pub const handle_kind_dma_region: u8 = 2;
|
||||
|
||||
/// A DMA buffer's delegation token: names the `pages` contiguous frames at `phys` that
|
||||
/// `dma_alloc` handed its `owner`, and can be passed across processes as a capability so
|
||||
/// the driver that owns a device can bind it into that device's IOMMU domain
|
||||
/// (`dma_bind`). Unlike `SharedMemoryObject` this does NOT own the frames — the
|
||||
/// allocating address space still does, and frees them on `dma_free` or teardown — so
|
||||
/// this is a pure token: `dead` is set when the allocator frees the region, after which
|
||||
/// a stale handle can no longer bind it. `refcount` counts the allocator's registry
|
||||
/// entry plus every outstanding handle; the object is freed when the last drops.
|
||||
pub const DmaRegionObject = struct {
|
||||
refcount: u32 = 1,
|
||||
phys: u64,
|
||||
pages: usize,
|
||||
owner: u32,
|
||||
dead: bool = false,
|
||||
};
|
||||
|
||||
/// Create a DMA-region token for `pages` frames at `phys` owned by task `owner`. The
|
||||
/// frames are already allocated and mapped by the caller; this only wraps them for
|
||||
/// delegation. null if the heap is out of room.
|
||||
pub fn createDmaRegion(phys: u64, pages: usize, owner: u32) ?*DmaRegionObject {
|
||||
const region = heap.allocator().create(DmaRegionObject) catch return null;
|
||||
region.* = .{ .phys = phys, .pages = pages, .owner = owner };
|
||||
return region;
|
||||
}
|
||||
|
||||
/// Drop a DMA-region reference; free the token when the last (registry + handles) goes.
|
||||
/// Never frees frames — the allocator owns those.
|
||||
pub fn dropDmaRegionReference(region: *DmaRegionObject) void {
|
||||
if (region.refcount > 1) {
|
||||
region.refcount -= 1;
|
||||
} else {
|
||||
heap.allocator().destroy(region);
|
||||
}
|
||||
}
|
||||
|
||||
/// Resolve a handle to its DMA-region token, or null if out of range, unused, or a
|
||||
/// different kind.
|
||||
pub fn resolveDmaRegion(t: *Task, h: u64) ?*DmaRegionObject {
|
||||
if (h >= t.handles.len) return null;
|
||||
const entry = t.handles[@intCast(h)] orelse return null;
|
||||
if (entry.kind != handle_kind_dma_region) return null;
|
||||
return @ptrCast(@alignCast(entry.ptr));
|
||||
}
|
||||
|
||||
/// Install a DMA-region handle in task `t`'s table (the slot owns a reference).
|
||||
pub fn installDmaRegionHandle(t: *Task, region: *DmaRegionObject) i64 {
|
||||
return installEntry(t, .{ .kind = handle_kind_dma_region, .ptr = @ptrCast(region) });
|
||||
}
|
||||
|
||||
/// Drop the handle at slot `h` of task `t` (handle_close): release its reference and
|
||||
/// free the slot. Returns 0 or -EBADF.
|
||||
pub fn closeHandle(t: *Task, h: u64) i64 {
|
||||
if (h >= t.handles.len) return -EBADF;
|
||||
const entry = t.handles[@intCast(h)] orelse return -EBADF;
|
||||
dropEntry(entry);
|
||||
t.handles[@intCast(h)] = null;
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// A page-aligned block of **shared cacheable RAM** (docs/display-v2.md), referenced by
|
||||
/// capability handles across processes and freed when the last one drops. `phys` is its
|
||||
@@ -307,6 +367,10 @@ fn shareCapability(from: *Task, to: *Task, cap: u64) i64 {
|
||||
const s: *SharedMemoryObject = @ptrCast(@alignCast(entry.ptr));
|
||||
s.refcount += 1;
|
||||
},
|
||||
handle_kind_dma_region => {
|
||||
const r: *DmaRegionObject = @ptrCast(@alignCast(entry.ptr));
|
||||
r.refcount += 1;
|
||||
},
|
||||
else => return -EBADF,
|
||||
}
|
||||
const handle = installEntry(to, entry);
|
||||
@@ -574,6 +638,7 @@ fn dropEntry(entry: scheduler.HandleObject) void {
|
||||
switch (entry.kind) {
|
||||
handle_kind_endpoint => dropRef(@ptrCast(@alignCast(entry.ptr))),
|
||||
handle_kind_shared_memory => dropSharedMemoryReference(@ptrCast(@alignCast(entry.ptr))),
|
||||
handle_kind_dma_region => dropDmaRegionReference(@ptrCast(@alignCast(entry.ptr))),
|
||||
else => {},
|
||||
}
|
||||
}
|
||||
|
||||
@@ -13,6 +13,7 @@ const sync = @import("sync.zig");
|
||||
const process = @import("process.zig");
|
||||
const devices_broker = @import("devices-broker.zig");
|
||||
const irq = @import("irq.zig");
|
||||
const iommu = @import("iommu.zig");
|
||||
const platform = @import("platform");
|
||||
const tests = @import("tests.zig");
|
||||
const build_options = @import("build_options");
|
||||
@@ -215,6 +216,12 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
// land on. Every line stays masked until something binds it (ioapic.init).
|
||||
irq.init();
|
||||
|
||||
// Bring up DMA translation: build the IOMMU domains and enable it (or record
|
||||
// fail-open when no unit exists). Must run before any driver claims a device —
|
||||
// an unclaimed device's DMA is blocked once translation is on. Logs its own
|
||||
// enable block; the fail-open posture is stated in the platform block below.
|
||||
iommu.init();
|
||||
|
||||
// Power register map, from the FADT (the SLP_TYP sleep values live in AML,
|
||||
// which the kernel doesn't parse — the ring-3 acpi service owns soft-off).
|
||||
const pw = platform.powerInformation();
|
||||
@@ -272,6 +279,11 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
log.print(" cpus : {d} usable core(s); 1 running (BSP), {d} AP(s) parked (SMP bring-up pending)\n", .{ cores.len, if (cores.len > 0) cores.len - 1 else 0 });
|
||||
if (platform.cpusDropped() > 0)
|
||||
log.print(" cpus : WARNING {d} core(s) beyond pool cap dropped\n", .{platform.cpusDropped()});
|
||||
// The DMA-isolation posture, stated plainly at every boot. When a unit exists,
|
||||
// iommu.init already logged its enable block above; here we only state the
|
||||
// fail-open case, so a boot without the line is a boot with translation on.
|
||||
if (!iommu.enabled())
|
||||
log.write(" iommu : none present - DMA fail-open (unisolated)\n");
|
||||
} else |err| {
|
||||
log.print("\n/system/kernel: device discovery failed: {s}\n", .{@errorName(err)});
|
||||
}
|
||||
|
||||
@@ -33,6 +33,7 @@ const sync = @import("sync.zig");
|
||||
const ipc = @import("ipc-synchronous.zig");
|
||||
const devices_broker = @import("devices-broker.zig");
|
||||
const irq = @import("irq.zig");
|
||||
const iommu = @import("iommu.zig");
|
||||
const initial_ramdisk = @import("initial-ramdisk");
|
||||
const vfs = @import("vfs.zig");
|
||||
const log = @import("log.zig");
|
||||
@@ -250,6 +251,10 @@ fn system_call(state: *architecture.CpuState) void {
|
||||
.fs_node => systemFsNode(state),
|
||||
.fs_mount => systemFsMount(state),
|
||||
.fs_unmount => systemFsUnmount(state),
|
||||
.iommu_fault_drain => systemIommuFaultDrain(state),
|
||||
.dma_bind => systemDmaBind(state),
|
||||
.dma_unbind => systemDmaUnbind(state),
|
||||
.handle_close => systemHandleClose(state),
|
||||
.wall_clock => systemWallClock(state),
|
||||
.shared_memory_create => systemSharedMemoryCreate(state),
|
||||
.shared_memory_map => systemSharedMemoryMap(state),
|
||||
@@ -390,6 +395,21 @@ fn systemDeviceClaim(state: *architecture.CpuState) void {
|
||||
const claim_flags = sync.enter();
|
||||
defer sync.leave(claim_flags);
|
||||
if (devices_broker.claim(device_id, scheduler.current().id)) {
|
||||
// Confine the device's DMA before the driver can program it: a PCI function
|
||||
// becomes reachable to the IOMMU only once claimed (until now its DMA is
|
||||
// blocked). A claim that cannot be confined must not stand — roll it back —
|
||||
// since the whole point is that claiming a DMA device is no longer equivalent
|
||||
// to ring 0. No-op when no IOMMU exists (fail-open).
|
||||
if (devices_broker.pciAddressOf(device_id)) |bdf| {
|
||||
const owner = scheduler.current().id;
|
||||
if (!iommu.confineDevice(device_id, bdf, owner)) {
|
||||
_ = devices_broker.unclaim(device_id, owner);
|
||||
return fail(state);
|
||||
}
|
||||
// Bind the buffers this task allocated before claiming the device (a driver
|
||||
// that dma_alloc'd its rings, then claimed the controller).
|
||||
dmaBindOwnerRegionsInto(owner, device_id);
|
||||
}
|
||||
// A display service just took the framebuffer — quiesce the bootstrap console
|
||||
// so the kernel and the service don't scribble over each other's pixels. The
|
||||
// claim releases (and the console resumes) automatically if the service dies;
|
||||
@@ -497,6 +517,137 @@ fn systemIoWrite(state: *architecture.CpuState) void {
|
||||
/// their physical address is never disclosed. `dma_below_4g` caps the physical address
|
||||
/// for legacy engines; `dma_write_combining` is accepted but falls back to coherent
|
||||
/// until PAT is programmed. See docs/driver-model.md (M14).
|
||||
// --- DMA-region registry ---------------------------------------------------------------
|
||||
// Every dma_alloc'd region is tracked here so it can be (a) auto-bound into the devices
|
||||
// its owner claims, (b) delegated across processes as a capability and bound into a
|
||||
// device's IOMMU domain by dma_bind, and (c) unmapped from every domain before its frames
|
||||
// return to the allocator. Only *shareable* regions carry a heap `object` (the delegation
|
||||
// token); a driver's private rings are tracked without one. All access under the big lock.
|
||||
const DmaRegistryEntry = struct {
|
||||
active: bool = false,
|
||||
object: ?*ipc.DmaRegionObject = null,
|
||||
physical: u64 = 0,
|
||||
len: u64 = 0,
|
||||
owner: u32 = 0,
|
||||
};
|
||||
const maximum_dma_regions = 256;
|
||||
var dma_registry: [maximum_dma_regions]DmaRegistryEntry = .{DmaRegistryEntry{}} ** maximum_dma_regions;
|
||||
|
||||
fn dmaRegistryAdd(object: ?*ipc.DmaRegionObject, physical: u64, len: u64, owner: u32) void {
|
||||
for (&dma_registry) |*e| {
|
||||
if (!e.active) {
|
||||
e.* = .{ .active = true, .object = object, .physical = physical, .len = len, .owner = owner };
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Retire the region based at `physical` owned by `owner`: unmap it from every device
|
||||
/// domain (before the frames are freed), mark its token dead so a stale downstream handle
|
||||
/// can no longer bind it, drop the registry's reference, and clear the slot.
|
||||
fn dmaRegistryRemove(owner: u32, physical: u64) void {
|
||||
for (&dma_registry) |*e| {
|
||||
if (e.active and e.owner == owner and e.physical == physical) {
|
||||
iommu.unmapRegionEverywhere(e.physical, e.len);
|
||||
if (e.object) |object| {
|
||||
object.dead = true;
|
||||
ipc.dropDmaRegionReference(object);
|
||||
}
|
||||
e.* = .{};
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Retire every region owned by `owner` (task death) — same discipline as a per-region
|
||||
/// free, run before the address space is torn down and its DMA frames reclaimed.
|
||||
fn dmaRegistryReleaseOwner(owner: u32) void {
|
||||
for (&dma_registry) |*e| {
|
||||
if (e.active and e.owner == owner) {
|
||||
iommu.unmapRegionEverywhere(e.physical, e.len);
|
||||
if (e.object) |object| {
|
||||
object.dead = true;
|
||||
ipc.dropDmaRegionReference(object);
|
||||
}
|
||||
e.* = .{};
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Bind every region `owner` allocated into the domain of the device it just claimed
|
||||
/// (its rings allocated before the claim). Regions allocated *after* the claim are bound
|
||||
/// by dma_alloc's own auto-bind.
|
||||
fn dmaBindOwnerRegionsInto(owner: u32, device_id: u64) void {
|
||||
for (&dma_registry) |*e| {
|
||||
if (e.active and e.owner == owner) _ = iommu.mapForDevice(device_id, e.physical, e.len);
|
||||
}
|
||||
}
|
||||
|
||||
/// iommu_fault_drain() -> count: drain and log any pending IOMMU translation faults,
|
||||
/// returning how many were seen. A diagnostic hook — a driver (or a test) that suspects
|
||||
/// its device faulted can force the fault records to be logged now rather than waiting
|
||||
/// for the next device-release drain. Harmless without an IOMMU (returns 0).
|
||||
fn systemIommuFaultDrain(state: *architecture.CpuState) void {
|
||||
const flags = sync.enter();
|
||||
const count = iommu.faultDrain();
|
||||
sync.leave(flags);
|
||||
architecture.setSystemCallResult(state, count);
|
||||
}
|
||||
|
||||
/// Resolve a capability handle the caller holds to the physical range it names — either
|
||||
/// a DMA-region token or a shared-memory object (both are bindable buffers). null if the
|
||||
/// handle is neither, or names a region already freed by its allocator.
|
||||
fn bindableRange(t: *scheduler.Task, handle: u64) ?struct { physical: u64, len: u64 } {
|
||||
if (ipc.resolveDmaRegion(t, handle)) |region| {
|
||||
if (region.dead) return null;
|
||||
return .{ .physical = region.phys, .len = region.pages * page_size };
|
||||
}
|
||||
if (ipc.resolveSharedMemory(t, handle)) |shared| {
|
||||
return .{ .physical = shared.phys, .len = shared.pages * page_size };
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// dma_bind(device_id, handle) -> 0/-errno: map the buffer named by `handle` into the
|
||||
/// claimed device's IOMMU domain. The caller must own the device (the mmio_map gate) and
|
||||
/// hold the handle. Idempotent: re-binding is a harmless success, so a driver may re-bind
|
||||
/// after a restart without tracking what it already bound. Success (no-op) without an IOMMU.
|
||||
fn systemDmaBind(state: *architecture.CpuState) void {
|
||||
const device_id = architecture.systemCallArg(state, 0);
|
||||
const handle = architecture.systemCallArg(state, 1);
|
||||
const t = scheduler.current();
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
if (devices_broker.ownerOf(device_id) != t.id) return fail(state);
|
||||
const range = bindableRange(t, handle) orelse return fail(state);
|
||||
if (!iommu.mapForDevice(device_id, range.physical, range.len)) return fail(state);
|
||||
architecture.setSystemCallResult(state, 0);
|
||||
}
|
||||
|
||||
/// dma_unbind(device_id, handle) -> 0/-errno: unmap a previously bound buffer from the
|
||||
/// device's domain and invalidate.
|
||||
fn systemDmaUnbind(state: *architecture.CpuState) void {
|
||||
const device_id = architecture.systemCallArg(state, 0);
|
||||
const handle = architecture.systemCallArg(state, 1);
|
||||
const t = scheduler.current();
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
if (devices_broker.ownerOf(device_id) != t.id) return fail(state);
|
||||
const range = bindableRange(t, handle) orelse return fail(state);
|
||||
iommu.unmapForDevice(device_id, range.physical, range.len);
|
||||
architecture.setSystemCallResult(state, 0);
|
||||
}
|
||||
|
||||
/// handle_close(handle) -> 0/-errno: drop one capability handle and free its slot.
|
||||
fn systemHandleClose(state: *architecture.CpuState) void {
|
||||
const handle = architecture.systemCallArg(state, 0);
|
||||
const t = scheduler.current();
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
if (ipc.closeHandle(t, handle) < 0) return fail(state);
|
||||
architecture.setSystemCallResult(state, 0);
|
||||
}
|
||||
|
||||
fn systemDmaAlloc(state: *architecture.CpuState) void {
|
||||
const len = architecture.systemCallArg(state, 0);
|
||||
const flags = architecture.systemCallArg(state, 1);
|
||||
@@ -543,8 +694,33 @@ fn systemDmaAlloc(state: *architecture.CpuState) void {
|
||||
architecture.mapUserDmaInto(t.address_space, base_v + i * page_size, phys + i * page_size, page_size);
|
||||
sync.leave(lock_flags);
|
||||
}
|
||||
// Register the region and bind it into every device this task already drives (its
|
||||
// own buffers reach its own devices). When `dma_shareable` is set, also wrap it in a
|
||||
// capability token and return a handle so it can be delegated to another driver and
|
||||
// dma_bound there. All under the lock (the registry + IOMMU tables are shared).
|
||||
var handle: u64 = abi.no_cap;
|
||||
{
|
||||
const lock_flags = sync.enter();
|
||||
defer sync.leave(lock_flags);
|
||||
var object: ?*ipc.DmaRegionObject = null;
|
||||
if (flags & abi.dma_shareable != 0) {
|
||||
if (ipc.createDmaRegion(phys, pages, t.id)) |region| {
|
||||
const h = ipc.installDmaRegionHandle(t, region);
|
||||
if (h >= 0) {
|
||||
region.refcount += 1; // the handle's reference (registry holds the first)
|
||||
handle = @intCast(h);
|
||||
object = region;
|
||||
} else {
|
||||
ipc.dropDmaRegionReference(region); // no table slot; drop it
|
||||
}
|
||||
}
|
||||
}
|
||||
dmaRegistryAdd(object, phys, pages * page_size, t.id);
|
||||
iommu.mapRegionForOwner(t.id, phys, pages * page_size);
|
||||
}
|
||||
architecture.setSystemCallResult(state, base_v); // virtual address for the CPU
|
||||
architecture.setSystemCallResult2(state, phys); // physical address for the device
|
||||
architecture.setSystemCallResult3(state, handle); // capability handle (no_cap unless shareable)
|
||||
}
|
||||
|
||||
/// dma_free(virtual_address, len) -> 0: release a prior `dma_alloc`. Bounded to the DMA arena so
|
||||
@@ -560,6 +736,16 @@ fn systemDmaFree(state: *architecture.CpuState) void {
|
||||
const pages: usize = @intCast((len + page_size - 1) / page_size);
|
||||
if (base_v < dma_arena_base or base_v + pages * page_size > dma_arena_end) return fail(state);
|
||||
|
||||
// Retire the region — unmap it from every device domain and mark its token dead —
|
||||
// BEFORE any frame returns to the allocator, so no device can still translate to a
|
||||
// reallocated frame. dma_alloc's frames are contiguous, so the base names the region.
|
||||
{
|
||||
const lock_flags = sync.enter();
|
||||
if (architecture.translate(t.address_space, base_v)) |base_phys|
|
||||
dmaRegistryRemove(t.id, base_phys);
|
||||
sync.leave(lock_flags);
|
||||
}
|
||||
|
||||
for (0..pages) |i| {
|
||||
const va = base_v + i * page_size;
|
||||
// Per-page lock hold: the translate/unmap walks the shared page tables
|
||||
@@ -991,6 +1177,14 @@ pub var fault_kill_count: u64 = 0;
|
||||
fn releaseTaskResourcesLocked(t: *scheduler.Task) void {
|
||||
recordExitLocked(t);
|
||||
irq.releaseOwner(t.id);
|
||||
// Detach the task's devices from their IOMMU domains BEFORE the broker clears the
|
||||
// claims (the detach reads ownership) and before the address space is torn down and
|
||||
// its DMA frames return to the allocator — a device must stop translating to a frame
|
||||
// before that frame can be handed to someone else. Then retire the buffers this task
|
||||
// allocated, unmapping them from any *other* driver's domain they were granted into,
|
||||
// before those frames are freed too.
|
||||
iommu.releaseAllOwnedBy(t.id);
|
||||
dmaRegistryReleaseOwner(t.id);
|
||||
devices_broker.releaseAllOwnedBy(t.id);
|
||||
// If that dropped the framebuffer claim (this task was the display service), let the
|
||||
// bootstrap console draw again — the screen is nobody's now, so panics/status land.
|
||||
|
||||
@@ -149,7 +149,10 @@ pub const maximum_task_name = abi.maximum_process_name;
|
||||
|
||||
/// Size of each task's IPC handle table. Kept here (not in ipc_sync.zig) because
|
||||
/// it dimensions a field of `Task`; ipc_sync.zig re-exports it.
|
||||
pub const ipc_maximum_handles = 16;
|
||||
// Raised from 16 with DMA-region capabilities: a driver now holds its per-device
|
||||
// channel endpoints plus received DMA-region handles (a storage driver forwards several
|
||||
// buffer caps), and repeated cap-passing consumes slots until handle_close.
|
||||
pub const ipc_maximum_handles = 32;
|
||||
|
||||
/// One handle-table entry: a capability object plus a `kind` tag saying what `ptr` points
|
||||
/// at (an ipc endpoint or a shared-memory object), so a task's exit path and the
|
||||
|
||||
+108
-5
@@ -18,6 +18,7 @@ const wall_clock = @import("wall-clock.zig");
|
||||
const devices_broker = @import("devices-broker.zig");
|
||||
const platform = @import("platform");
|
||||
const pmm = @import("pmm.zig");
|
||||
const iommu = @import("iommu.zig");
|
||||
const heap = @import("heap.zig");
|
||||
const scheduler = @import("scheduler.zig");
|
||||
const ipc = @import("ipc.zig");
|
||||
@@ -212,6 +213,10 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
|
||||
deviceListTest(boot_information);
|
||||
} else if (eql(case, "pci-scan")) {
|
||||
pciScanTest(boot_information);
|
||||
} else if (eql(case, "pci-caps")) {
|
||||
pciCapsTest(boot_information);
|
||||
} else if (eql(case, "iommu-fault")) {
|
||||
iommuFaultTest(boot_information);
|
||||
} else if (eql(case, "acpi-parse")) {
|
||||
acpiParseTest(boot_information);
|
||||
} else if (eql(case, "acpi-report")) {
|
||||
@@ -1164,6 +1169,11 @@ fn dmaTest() void {
|
||||
log("DANOS-TEST-BEGIN: dma\n", .{});
|
||||
const base_free = pmm.stats().free_frames;
|
||||
|
||||
// This case boots WITHOUT an IOMMU device, so it is the explicit witness of the
|
||||
// fail-open posture: no unit found, and the kernel said so at boot (the harness
|
||||
// asserts the boot line; this check pins the recorded state to it).
|
||||
check("no IOMMU present: DMA runs fail-open", !platform.platformInformation().iommu_present);
|
||||
|
||||
// A contiguous run: aligned, and it consumed exactly that many frames.
|
||||
const frames = 4;
|
||||
const phys = pmm.allocContiguous(frames, ~@as(u64, 0)) orelse {
|
||||
@@ -1234,16 +1244,41 @@ fn msiTest() void {
|
||||
|
||||
/// IOMMU (M16): with an emulated VT-d unit present (the harness boots this case with
|
||||
/// `-device intel-iommu`), danos must find it in the ACPI DMAR table, map its register
|
||||
/// block, and read back a real version. This is *detection*, the honest first step —
|
||||
/// no translation domains are programmed yet, so DMA is still unprotected; enforcement
|
||||
/// lands with the first DMA driver (docs/driver-model.md M16).
|
||||
/// block, and read back a real version. Detection was M16's honest first step; the
|
||||
/// IOVA-enforcement track extends this case milestone by milestone (translation
|
||||
/// enabled, then per-device domains) — see the plan in docs and the fail-open witness
|
||||
/// in dmaTest.
|
||||
fn iommuTest() void {
|
||||
log("DANOS-TEST-BEGIN: iommu\n", .{});
|
||||
const pinfo = platform.platformInformation();
|
||||
check("IOMMU found in the DMAR table", pinfo.iommu_present);
|
||||
check("VT-d unit has a register base", pinfo.iommu_base != 0);
|
||||
check("IOMMU found in the firmware tables", pinfo.iommu_present);
|
||||
check("IOMMU unit has a register base", pinfo.iommu_base != 0);
|
||||
// The VT-d version register is a live-unit sanity check; AMD-Vi (from IVRS) records
|
||||
// no version, so gate it on the vendor.
|
||||
if (!pinfo.iommu_is_amd)
|
||||
check("VT-d version register reads back nonzero (real, mappable unit)", pinfo.iommu_version != 0);
|
||||
log("DANOS-IOMMU: base=0x{x} version=0x{x} capabilities=0x{x}\n", .{ pinfo.iommu_base, pinfo.iommu_version, pinfo.iommu_capabilities });
|
||||
|
||||
// Translation was enabled at boot (kernel.zig: iommu.init before any driver claims
|
||||
// a device). The blanket domain keeps every device identity-mapped, so DMA still
|
||||
// works, but the unit is live — and with only the boot-time mappings present, no
|
||||
// device should have faulted yet.
|
||||
check("IOMMU enabled (translation on)", iommu.enabled());
|
||||
check("no spurious translation faults at idle", iommu.faultDrain() == 0);
|
||||
|
||||
// A scratch domain proves the walker + invalidation path end to end: create it,
|
||||
// identity-map a page, and confirm the mapping resolves; then unmap and destroy.
|
||||
if (iommu.domainCreate(0, 0)) |scratch| {
|
||||
const scratch_phys: u64 = 0x0010_0000; // 1 MiB, page-aligned
|
||||
check("map into a scratch domain succeeds", iommu.map(scratch, scratch_phys, abi.page_size));
|
||||
check("scratch domain resolves the mapping", iommu.translationOf(scratch, scratch_phys) == scratch_phys);
|
||||
iommu.unmap(scratch, scratch_phys, abi.page_size);
|
||||
check("scratch domain drops the mapping", iommu.translationOf(scratch, scratch_phys) == null);
|
||||
iommu.domainDestroy(scratch);
|
||||
} else {
|
||||
check("scratch domain allocated", false);
|
||||
}
|
||||
log("DANOS-IOMMU: enabled base=0x{x} domains active\n", .{pinfo.iommu_base});
|
||||
result();
|
||||
}
|
||||
|
||||
@@ -2388,6 +2423,74 @@ fn deviceListTest(boot_information: *const BootInformation) void {
|
||||
result();
|
||||
}
|
||||
|
||||
/// The driver-side PCI library against a real function: the pci-caps QEMU case adds an
|
||||
/// e1000e NIC no danos driver claims; the pci-cap-test fixture claims it and exercises
|
||||
/// header accessors, command bits, the capability walks, MSI programming (the first
|
||||
/// driver-side `msi_bind` use), the MSI-X table, power state, and FLR. The kernel side
|
||||
/// only spawns the manager (which spawns pci-bus itself) and the fixture; the substance
|
||||
/// is asserted by the harness on the fixture's own serial lines.
|
||||
fn pciCapsTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: pci-caps\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over an initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
|
||||
process.setInitialRamdisk(image);
|
||||
// Plain mode — no restart drill, whose kill would race the fixture's claim.
|
||||
var manager: u32 = 0;
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
|
||||
manager = process.spawnProcessSupervised(item.blob, 4, &.{"device-manager"}, scheduler.currentId(), null) catch 0;
|
||||
break;
|
||||
}
|
||||
check("device-manager spawned", manager != 0);
|
||||
check("pci-cap-test spawned", spawnNamed(rd, "pci-cap-test"));
|
||||
result();
|
||||
}
|
||||
|
||||
/// IOMMU enforcement, the negative proof: boot with VT-d on and an unclaimed e1000e.
|
||||
/// The manager spawns pci-bus, the fixture claims the NIC and fires a DMA at an
|
||||
/// unmapped page; the unit must fault it and the system survive. Substance is asserted
|
||||
/// by the harness on the kernel's DANOS-IOMMU-FAULT line and the fixture's markers.
|
||||
fn iommuFaultTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: iommu-fault\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over an initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
|
||||
check("IOMMU enabled for the enforcement test", iommu.enabled());
|
||||
process.setInitialRamdisk(image);
|
||||
var manager: u32 = 0;
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
|
||||
manager = process.spawnProcessSupervised(item.blob, 4, &.{"device-manager"}, scheduler.currentId(), null) catch 0;
|
||||
break;
|
||||
}
|
||||
check("device-manager spawned", manager != 0);
|
||||
check("iommu-fault-test spawned", spawnNamed(rd, "iommu-fault-test"));
|
||||
result();
|
||||
}
|
||||
|
||||
/// M19.1: the ring-3 PCI scan agrees with the kernel's. The manager spawns
|
||||
/// pci-bus for the host bridge; the driver walks the same ECAM window through
|
||||
/// its mmio_map grant and must find exactly the functions the kernel's own
|
||||
|
||||
@@ -118,7 +118,17 @@ fn tryBringUp() void {
|
||||
_ = logging.write("/system/services/fat: block geometry unavailable\n");
|
||||
return;
|
||||
};
|
||||
const bounce = memory.dmaAlloc(engine.max_transfer_sectors * 512, memory.dma_coherent) orelse return;
|
||||
// Shareable so the buffer's capability can be attached down the chain (block server
|
||||
// -> controller), making its physical addresses reachable by the device under an
|
||||
// enforcing IOMMU. No-op binding otherwise.
|
||||
const bounce = memory.dmaAlloc(engine.max_transfer_sectors * 512, memory.dma_coherent | memory.dma_shareable) orelse return;
|
||||
if (bounce.handle) |handle| {
|
||||
if (!device.attach(handle)) {
|
||||
_ = logging.write("/system/services/fat: could not attach the DMA bounce buffer\n");
|
||||
return;
|
||||
}
|
||||
_ = ipc.close(handle); // the binding holds its own reference now
|
||||
}
|
||||
ipc_block = .{ .device = device, .bounce = bounce };
|
||||
|
||||
const block_device = engine.BlockDevice{
|
||||
|
||||
+67
-5
@@ -146,20 +146,70 @@ CASES = [
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# DMA memory (M14): contiguous frame allocation, below-4G cap, coherent mapping,
|
||||
# and reclaim on teardown.
|
||||
# Boots with no IOMMU device, so it also asserts the explicit fail-open boot line
|
||||
# (the DMA-isolation posture must be stated, never silent).
|
||||
{"name": "dma",
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"expect": r"(?s)(?=.*iommu : none present - DMA fail-open \(unisolated\))(?=.*DANOS-TEST-RESULT: PASS)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# MSI (M15): allocate a per-device vector and deliver it as a notification (a
|
||||
# self-IPI stands in for the device's MSI write, since the HPET has no MSI).
|
||||
{"name": "msi",
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# IOMMU (M16): boot with an emulated VT-d unit and confirm danos parses the DMAR
|
||||
# table and reads the unit's registers. Detection only — enforcement is future.
|
||||
# IOMMU: boot with an emulated VT-d unit; danos parses the DMAR table, enables
|
||||
# translation, and proves the domain walker (map/resolve/unmap on a scratch domain)
|
||||
# with no spurious faults. The `enabled` line is a lookahead so a silently-dead unit
|
||||
# cannot fake a pass.
|
||||
{"name": "iommu",
|
||||
"qemu_extra": ["-device", "intel-iommu,intremap=off"],
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
"expect": r"(?s)(?=.*DANOS-IOMMU: enabled base=0x[0-9a-f]+ domains active)(?=.*DANOS-TEST-RESULT: PASS)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL|DANOS-IOMMU-FAULT"},
|
||||
# DMA under translation: the full USB storage stack (xHC ring DMA + BOT/SCSI + fat's
|
||||
# cross-process bounce buffer) runs with VT-d enabled. Every device is identity-
|
||||
# mapped in the blanket domain, so DMA works, but through real second-level walks.
|
||||
{"name": "iommu-usb-storage",
|
||||
"build_case": "fat-mount",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"qemu_extra": ["-device", "intel-iommu,intremap=off"],
|
||||
"expect": r"(?s)(?=.*/system/kernel: iommu online)(?=.*fat: mounted /mnt/usb)(?=.*fat-test: ok)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL|DANOS-IOMMU-FAULT"},
|
||||
# DMA + MSI under translation: interrupt-IN reports arrive through translated DMA and
|
||||
# the xHC's MSI/MSI-X still delivers (the 0xFEE00000 interrupt window bypasses second-
|
||||
# level translation with interrupt remapping off).
|
||||
{"name": "iommu-usb-hid",
|
||||
"build_case": "usb-hid",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"qemu_extra": ["-device", "intel-iommu,intremap=off"],
|
||||
"expect": r"(?s)(?=.*/system/kernel: iommu online)(?=.*usb-hid-keyboard: ok)(?=.*usb-hid-mouse: ok)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL|DANOS-IOMMU-FAULT"},
|
||||
# Enforcement, the negative proof: a claimed e1000e fires a DMA at an unmapped page;
|
||||
# VT-d must fault it (logged) and the system must stay alive. The fault line is the
|
||||
# point here, so unlike the positive cases it appears in `expect`, not `fail`.
|
||||
{"name": "iommu-fault",
|
||||
"smp": 4,
|
||||
"timeout": 120,
|
||||
"qemu_extra": ["-device", "intel-iommu,intremap=off", "-device", "e1000e"],
|
||||
"expect": r"(?s)(?=.*DANOS-IOMMU-FAULT: bdf=)(?=.*iommu-fault-test: system alive)(?=.*DANOS-TEST-RESULT: PASS)",
|
||||
"fail": r"iommu-fault-test: FAIL|DANOS-TEST-RESULT: FAIL|CPU EXCEPTION|KERNEL PANIC"},
|
||||
# AMD-Vi: the same detection + scratch-domain walker proof as the `iommu` case, but on
|
||||
# the AMD backend (IVRS parse, device table, command buffer). QEMU's amd-iommu needs
|
||||
# dma-remap=on (default off = translation silently ignored). UNTESTED on real AMD.
|
||||
{"name": "amd-iommu",
|
||||
"build_case": "iommu",
|
||||
"qemu_extra": ["-device", "amd-iommu,dma-remap=on,intremap=off"],
|
||||
"expect": r"(?s)(?=.*iommu online \(AMD-Vi\))(?=.*DANOS-IOMMU: enabled base=0x[0-9a-f]+ domains active)(?=.*DANOS-TEST-RESULT: PASS)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL|DANOS-IOMMU-FAULT"},
|
||||
# DMA under AMD-Vi translation: the full USB storage stack through AMD device-table
|
||||
# translation. Tentative — land depends on QEMU amd-iommu behaving. UNTESTED on real AMD.
|
||||
{"name": "amd-iommu-usb-storage",
|
||||
"build_case": "fat-mount",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"qemu_extra": ["-device", "amd-iommu,dma-remap=on,intremap=off"],
|
||||
"expect": r"(?s)(?=.*iommu online \(AMD-Vi\))(?=.*fat: mounted /mnt/usb)(?=.*fat-test: ok)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL|DANOS-IOMMU-FAULT"},
|
||||
# Port I/O grants: a claimed device's io_port resource lets a driver read/write its
|
||||
# ports (PS/2 status 0x64), gated by the claim; out-of-range/unclaimed is refused.
|
||||
{"name": "ioport",
|
||||
@@ -687,6 +737,18 @@ CASES = [
|
||||
# duplicates); this ordered regex asserts the drill itself over the whole serial
|
||||
# log — the backreference requires the respawn to re-scan the same count, and the
|
||||
# full-capture match is immune to the transient-line races an in-kernel poll hits.
|
||||
# The driver-side PCI library (library/device/pci) against a real function: an extra
|
||||
# e1000e NIC — PM + MSI + PCIe + MSI-X capabilities, claimed by no danos driver — is
|
||||
# claimed by the pci-cap-test fixture, which exercises the capability walks, MSI
|
||||
# programming (the first driver-side msi_bind use), the MSI-X table, power state,
|
||||
# and FLR, printing a marker per check.
|
||||
{"name": "pci-caps",
|
||||
"smp": 4,
|
||||
"timeout": 120,
|
||||
"qemu_extra": ["-device", "e1000e"],
|
||||
"expect": r"(?s)(?=.*DANOS-TEST-RESULT: PASS)(?=.*pci-cap-test: all checks passed)",
|
||||
"fail": r"pci-cap-test: FAIL|DANOS-TEST-RESULT: FAIL"},
|
||||
|
||||
{"name": "pci-scan",
|
||||
"smp": 4,
|
||||
"timeout": 60,
|
||||
|
||||
@@ -0,0 +1,119 @@
|
||||
//! iommu-fault-test — the negative proof for IOMMU enforcement. The iommu-fault QEMU
|
||||
//! case boots with VT-d enabled and an extra e1000e NIC no danos driver claims; this
|
||||
//! fixture claims it, then deliberately programs its transmit engine to DMA from a
|
||||
//! physical address that was never dma_alloc'd (so it is in no device's domain). The
|
||||
//! IOMMU must fault that access — the descriptor fetch never reaches memory — and the
|
||||
//! system must stay alive. A kernel `DANOS-IOMMU-FAULT` line plus this fixture's
|
||||
//! `system alive` marker is the pass.
|
||||
//!
|
||||
//! The rogue target is the e1000e's transmit descriptor RING base itself: the very first
|
||||
//! DMA the engine issues on a doorbell write is the descriptor fetch from that base, so
|
||||
//! pointing the ring at an unmapped page makes the first access the faulting one — no
|
||||
//! valid descriptor need be crafted.
|
||||
|
||||
const std = @import("std");
|
||||
const device = @import("driver");
|
||||
const time = @import("time");
|
||||
const logging = @import("logging");
|
||||
const mmio = @import("mmio");
|
||||
const pci = @import("pci");
|
||||
const pci_class = @import("pci-class");
|
||||
|
||||
const intel_vendor: u16 = 0x8086;
|
||||
const e1000e_device: u16 = 0x10D3;
|
||||
|
||||
const ethernet_class: u64 = pci_class.ClassCode.pack(.{
|
||||
.base = @intFromEnum(pci_class.BaseClass.network),
|
||||
.subclass = @intFromEnum(pci_class.network.SubClass.ethernet),
|
||||
.prog_if = 0,
|
||||
});
|
||||
|
||||
// e1000e transmit-engine registers (Intel 82574L datasheet §Register Descriptions),
|
||||
// byte offsets within BAR0. VERIFY-AGAINST-SPEC held on first bring-up: these are the
|
||||
// legacy TX ring registers.
|
||||
const reg_tctl = 0x0400; // Transmit Control
|
||||
const reg_tdbal = 0x3800; // TX Descriptor Base Address Low
|
||||
const reg_tdbah = 0x3804; // TX Descriptor Base Address High
|
||||
const reg_tdlen = 0x3808; // TX Descriptor Length (bytes, 128-byte aligned)
|
||||
const reg_tdh = 0x3810; // TX Descriptor Head
|
||||
const reg_tdt = 0x3818; // TX Descriptor Tail
|
||||
|
||||
const tctl_en: u32 = 1 << 1; // Transmit Enable
|
||||
const tctl_psp: u32 = 1 << 3; // Pad Short Packets
|
||||
|
||||
/// A low physical page that user DMA never touches — never returned by dma_alloc (whose
|
||||
/// arena is far higher), so it is in no device's IOMMU domain. The e1000e's descriptor
|
||||
/// fetch from here is exactly the out-of-domain access the unit must block.
|
||||
const rogue_physical: u64 = 0x1000;
|
||||
|
||||
var descriptor: device.DeviceDescriptor = undefined;
|
||||
|
||||
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
var line: [128]u8 = undefined;
|
||||
_ = logging.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
pub fn main() void {
|
||||
const nic_id: u64 = found: {
|
||||
var tries: u32 = 0;
|
||||
while (tries < 150) : (tries += 1) {
|
||||
var descriptors: [64]device.DeviceDescriptor = undefined;
|
||||
const total = device.enumerate(&descriptors);
|
||||
for (descriptors[0..@min(total, descriptors.len)]) |*entry| {
|
||||
if (entry.class == @intFromEnum(device.DeviceClass.pci_device) and entry.pci_class == ethernet_class) {
|
||||
descriptor = entry.*;
|
||||
break :found entry.id;
|
||||
}
|
||||
}
|
||||
time.sleepMillis(100);
|
||||
}
|
||||
_ = logging.write("iommu-fault-test: FAIL no ethernet function found\n");
|
||||
return;
|
||||
};
|
||||
|
||||
if (!device.claim(nic_id)) {
|
||||
_ = logging.write("iommu-fault-test: FAIL claim\n");
|
||||
return;
|
||||
}
|
||||
var function = pci.Function.map(nic_id, &descriptor) orelse {
|
||||
_ = logging.write("iommu-fault-test: FAIL config-space map\n");
|
||||
return;
|
||||
};
|
||||
if (function.vendorId() != intel_vendor or function.deviceId() != e1000e_device) {
|
||||
_ = logging.write("iommu-fault-test: FAIL not an e1000e\n");
|
||||
return;
|
||||
}
|
||||
function.enableMemoryAndBusMaster();
|
||||
const bar0 = function.mapBar(0) orelse {
|
||||
_ = logging.write("iommu-fault-test: FAIL map BAR0\n");
|
||||
return;
|
||||
};
|
||||
|
||||
// Point the TX ring at the rogue page and kick the engine: TDBA = rogue, a non-zero
|
||||
// length, head=0, enable, then tail=1 so the engine fetches descriptor 0 — a DMA
|
||||
// read from the rogue page, which the IOMMU must fault.
|
||||
_ = logging.write("iommu-fault-test: pointing e1000e TX ring at an unmapped page\n");
|
||||
mmio.writeRegister(u32, bar0 + reg_tdbal, @truncate(rogue_physical));
|
||||
mmio.writeRegister(u32, bar0 + reg_tdbah, @intCast(rogue_physical >> 32));
|
||||
mmio.writeRegister(u32, bar0 + reg_tdlen, 128);
|
||||
mmio.writeRegister(u32, bar0 + reg_tdh, 0);
|
||||
mmio.writeRegister(u32, bar0 + reg_tctl, tctl_en | tctl_psp);
|
||||
mmio.writeMemoryBarrier();
|
||||
mmio.writeRegister(u32, bar0 + reg_tdt, 1); // doorbell: fetch descriptor 0
|
||||
|
||||
// Give the engine time to attempt the fetch, then force the fault records to the log.
|
||||
time.sleepMillis(200);
|
||||
const faults = device.iommuFaultDrain();
|
||||
writeLine("iommu-fault-test: drained {d} iommu fault(s)\n", .{faults});
|
||||
if (faults == 0) {
|
||||
_ = logging.write("iommu-fault-test: FAIL rogue DMA was not blocked\n");
|
||||
return;
|
||||
}
|
||||
|
||||
// Liveness: the system survived the blocked DMA — read our own config space back.
|
||||
if (function.vendorId() != intel_vendor) {
|
||||
_ = logging.write("iommu-fault-test: FAIL device unreadable after fault\n");
|
||||
return;
|
||||
}
|
||||
_ = logging.write("iommu-fault-test: system alive\n");
|
||||
}
|
||||
@@ -0,0 +1,178 @@
|
||||
//! pci-cap-test — QEMU fixture for the driver-side PCI library (library/device/pci).
|
||||
//! The pci-caps test case boots with an extra `-device e1000e` NIC that no danos driver
|
||||
//! claims; this fixture claims it and exercises the whole claimed-function surface
|
||||
//! against real (emulated) hardware: header accessors, command bits, the capability
|
||||
//! walk, MSI programming (the first driver-side `msi_bind` use), the MSI-X table,
|
||||
//! extended capabilities, power state, and — where offered — function-level reset.
|
||||
//! Every check prints `pci-cap-test: <check> ok` or `pci-cap-test: FAIL <check>`; the
|
||||
//! harness asserts on the final `all checks passed` marker (test/qemu_test.py).
|
||||
|
||||
const std = @import("std");
|
||||
const device = @import("driver");
|
||||
const ipc = @import("ipc");
|
||||
const time = @import("time");
|
||||
const logging = @import("logging");
|
||||
const mmio = @import("mmio");
|
||||
const pci = @import("pci");
|
||||
const pci_class = @import("pci-class");
|
||||
|
||||
/// QEMU's e1000e: Intel 82574L.
|
||||
const intel_vendor: u16 = 0x8086;
|
||||
const e1000e_device: u16 = 0x10D3;
|
||||
|
||||
const ethernet_class: u64 = pci_class.ClassCode.pack(.{
|
||||
.base = @intFromEnum(pci_class.BaseClass.network),
|
||||
.subclass = @intFromEnum(pci_class.network.SubClass.ethernet),
|
||||
.prog_if = 0,
|
||||
});
|
||||
|
||||
/// `pci.Function` keeps a pointer to the descriptor, so it must outlive the stack frame
|
||||
/// that found it.
|
||||
var descriptor: device.DeviceDescriptor = undefined;
|
||||
|
||||
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
var line: [128]u8 = undefined;
|
||||
_ = logging.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
/// Print the check's verdict; the caller returns on false to stop at the first failure.
|
||||
fn check(comptime name: []const u8, ok: bool) bool {
|
||||
if (ok) {
|
||||
_ = logging.write("pci-cap-test: " ++ name ++ " ok\n");
|
||||
} else {
|
||||
_ = logging.write("pci-cap-test: FAIL " ++ name ++ "\n");
|
||||
}
|
||||
return ok;
|
||||
}
|
||||
|
||||
pub fn main() void {
|
||||
// The bus scan runs in another process; poll until the NIC shows up.
|
||||
const nic_id: u64 = found: {
|
||||
var tries: u32 = 0;
|
||||
while (tries < 150) : (tries += 1) {
|
||||
var descriptors: [64]device.DeviceDescriptor = undefined;
|
||||
const total = device.enumerate(&descriptors);
|
||||
for (descriptors[0..@min(total, descriptors.len)]) |*entry| {
|
||||
if (entry.class == @intFromEnum(device.DeviceClass.pci_device) and entry.pci_class == ethernet_class) {
|
||||
descriptor = entry.*;
|
||||
break :found entry.id;
|
||||
}
|
||||
}
|
||||
time.sleepMillis(100);
|
||||
}
|
||||
_ = logging.write("pci-cap-test: FAIL no ethernet function found\n");
|
||||
return;
|
||||
};
|
||||
writeLine("pci-cap-test: claiming ethernet function (device {d})\n", .{nic_id});
|
||||
if (!check("claim", device.claim(nic_id))) return;
|
||||
var function = pci.Function.map(nic_id, &descriptor) orelse {
|
||||
_ = logging.write("pci-cap-test: FAIL config-space map\n");
|
||||
return;
|
||||
};
|
||||
|
||||
// Identity: the header accessors against known e1000e values.
|
||||
if (!check("vendor/device id", function.vendorId() == intel_vendor and function.deviceId() == e1000e_device)) return;
|
||||
if (!check("class code", function.classCode().pack() == ethernet_class)) return;
|
||||
if (!check("subsystem ids readable", function.subsystemVendorId() != 0xFFFF and function.subsystemId() != 0xFFFF)) return;
|
||||
|
||||
// Command bits: enable, read back, quiesce, read back, re-enable.
|
||||
function.enableMemoryAndBusMaster();
|
||||
if (!check("memory+bus-master enable", function.command() & pci_class.command_memory_and_bus_master == pci_class.command_memory_and_bus_master)) return;
|
||||
function.disableBusMaster();
|
||||
if (!check("bus-master disable", function.command() & pci_class.command_bus_master == 0)) return;
|
||||
function.enableMemoryAndBusMaster();
|
||||
|
||||
// The capability walk: e1000e advertises PM, MSI, PCIe, and MSI-X.
|
||||
var seen_power = false;
|
||||
var seen_msi = false;
|
||||
var seen_pci_express = false;
|
||||
var seen_msix = false;
|
||||
var walk = function.capabilities();
|
||||
while (walk.next()) |capability| {
|
||||
switch (capability.id) {
|
||||
@intFromEnum(pci_class.CapabilityId.power_management) => seen_power = true,
|
||||
@intFromEnum(pci_class.CapabilityId.msi) => seen_msi = true,
|
||||
@intFromEnum(pci_class.CapabilityId.pci_express) => seen_pci_express = true,
|
||||
@intFromEnum(pci_class.CapabilityId.msix) => seen_msix = true,
|
||||
else => {},
|
||||
}
|
||||
}
|
||||
if (!check("capability walk", seen_power and seen_msi and seen_pci_express and seen_msix)) return;
|
||||
if (!check("findCapability", function.findCapability(.msi) != null and function.findCapability(.pci_express) != null)) return;
|
||||
|
||||
// MSI: bind a vector (the syscall's first driver-side use), program the capability,
|
||||
// and read the registers straight back.
|
||||
const endpoint = ipc.createIpcEndpoint() orelse {
|
||||
_ = logging.write("pci-cap-test: FAIL endpoint creation\n");
|
||||
return;
|
||||
};
|
||||
const message = device.msiBind(nic_id, endpoint) orelse {
|
||||
_ = logging.write("pci-cap-test: FAIL msi_bind\n");
|
||||
return;
|
||||
};
|
||||
if (!check("msi_bind address", message.address == 0xFEE0_0000)) return;
|
||||
if (!check("programMsi", function.programMsi(message))) return;
|
||||
const msi_cap = function.findCapability(.msi).?;
|
||||
const msi_control = mmio.readRegister(u16, msi_cap.offset + pci_class.msi.control);
|
||||
const msi_data_offset: usize = if (msi_control & pci_class.msi.control_64bit_capable != 0) pci_class.msi.data_64 else pci_class.msi.data_32;
|
||||
if (!check("msi registers read back", msi_control & pci_class.msi.control_enable != 0 and
|
||||
msi_control & pci_class.msi.control_multiple_message_enable_mask == 0 and
|
||||
mmio.readRegister(u32, msi_cap.offset + pci_class.msi.address) == @as(u32, @truncate(message.address)) and
|
||||
mmio.readRegister(u16, msi_cap.offset + msi_data_offset) == @as(u16, @truncate(message.data)))) return;
|
||||
if (!check("intx disabled with msi", function.command() & pci_class.command_interrupt_disable != 0)) return;
|
||||
function.disableMsi();
|
||||
if (!check("disableMsi", mmio.readRegister(u16, msi_cap.offset + pci_class.msi.control) & pci_class.msi.control_enable == 0)) return;
|
||||
|
||||
// MSI-X: map the table, program entry 0, exercise the masks. Never enable — this
|
||||
// proves the programming surface, not delivery.
|
||||
const msix_table = function.msix() orelse {
|
||||
_ = logging.write("pci-cap-test: FAIL msix table map\n");
|
||||
return;
|
||||
};
|
||||
writeLine("pci-cap-test: msix table has {d} entries\n", .{msix_table.entry_count});
|
||||
if (!check("msix entry count", msix_table.entry_count >= 1)) return;
|
||||
if (!check("msix programEntry", msix_table.programEntry(0, message))) return;
|
||||
if (!check("msix entry reads back", mmio.readRegister(u32, msix_table.table + pci_class.msix.entry_address) == @as(u32, @truncate(message.address)) and
|
||||
mmio.readRegister(u32, msix_table.table + pci_class.msix.entry_data) == message.data and
|
||||
mmio.readRegister(u32, msix_table.table + pci_class.msix.entry_vector_control) & pci_class.msix.entry_vector_control_masked != 0)) return;
|
||||
if (!check("msix unmask entry", msix_table.unmaskEntry(0) and
|
||||
mmio.readRegister(u32, msix_table.table + pci_class.msix.entry_vector_control) & pci_class.msix.entry_vector_control_masked == 0)) return;
|
||||
if (!check("msix re-mask entry", msix_table.maskEntry(0) and
|
||||
mmio.readRegister(u32, msix_table.table + pci_class.msix.entry_vector_control) & pci_class.msix.entry_vector_control_masked != 0)) return;
|
||||
msix_table.setFunctionMask();
|
||||
if (!check("msix function mask", mmio.readRegister(u16, msix_table.capability + pci_class.msix.control) & pci_class.msix.control_function_mask != 0)) return;
|
||||
msix_table.clearFunctionMask();
|
||||
if (!check("msix function unmask", mmio.readRegister(u16, msix_table.capability + pci_class.msix.control) & pci_class.msix.control_function_mask == 0)) return;
|
||||
if (!check("msix out-of-range rejected", !msix_table.programEntry(msix_table.entry_count, message))) return;
|
||||
|
||||
// Extended capabilities: the walk must terminate cleanly; the count is informative
|
||||
// (don't hard-bind to QEMU's exact extended-capability set).
|
||||
var extended_count: u32 = 0;
|
||||
var extended = function.extendedCapabilities();
|
||||
while (extended.next()) |_| extended_count += 1;
|
||||
writeLine("pci-cap-test: {d} extended capabilities\n", .{extended_count});
|
||||
if (!check("extended walk terminates", extended_count < 480)) return;
|
||||
|
||||
// Power: QEMU leaves the function in D0; ensurePowerStateD0 must agree and not
|
||||
// disturb the PMCSR.
|
||||
const power_cap = function.findCapability(.power_management).?;
|
||||
const pmcsr_before = mmio.readRegister(u16, power_cap.offset + pci_class.power_management.control_status);
|
||||
if (!check("power state is D0", pmcsr_before & pci_class.power_management.control_status_power_state_mask == pci_class.power_management.power_state_d0)) return;
|
||||
function.ensurePowerStateD0();
|
||||
if (!check("ensurePowerStateD0 is a no-op at D0", mmio.readRegister(u16, power_cap.offset + pci_class.power_management.control_status) == pmcsr_before)) return;
|
||||
|
||||
// Function-level reset, where the device offers it: afterwards the function must be
|
||||
// readable with its identity intact, and bring-up must work again.
|
||||
const express_cap = function.findCapability(.pci_express).?;
|
||||
const device_capabilities = mmio.readRegister(u32, express_cap.offset + pci_class.pci_express.device_capabilities);
|
||||
if (device_capabilities & pci_class.pci_express.device_capabilities_flr != 0) {
|
||||
if (!check("functionLevelReset", function.functionLevelReset())) return;
|
||||
if (!check("identity after flr", function.vendorId() == intel_vendor and function.deviceId() == e1000e_device)) return;
|
||||
function.enableMemoryAndBusMaster();
|
||||
if (!check("re-enable after flr", function.command() & pci_class.command_memory_and_bus_master == pci_class.command_memory_and_bus_master)) return;
|
||||
} else {
|
||||
_ = logging.write("pci-cap-test: flr not offered, skipped\n");
|
||||
}
|
||||
|
||||
_ = logging.write("pci-cap-test: all checks passed\n");
|
||||
}
|
||||
Reference in New Issue
Block a user