From 9e649178bfe3d8a3ed78d17906747b902eb45d5b Mon Sep 17 00:00:00 2001 From: Daniel Samson <12231216+daniel-samson@users.noreply.github.com> Date: Thu, 23 Jul 2026 18:53:33 +0100 Subject: [PATCH] pci: full driver-side library (MSI/MSI-X, power, FLR, extended caps); xhci goes interrupt-driven MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit library/device/pci is now the complete generic floor a leaf PCI driver needs, instead of just what virtio-gpu used: - pci-class: capability IDs, MSI/MSI-X/power-management/PCI-Express register layouts, extended-capability header decode (host-tested), per-bit command constants, remaining header offsets. - pci.Function: header accessors, disableBusMaster + interrupt-disable helpers, findCapability, programMsi/disableMsi, MsiX vector-table struct, ensurePowerStateD0, functionLevelReset (BAR save/restore), extended-capability iterator. Proven by the new pci-caps QEMU case: a pci-cap-test fixture claims an extra e1000e (PM+MSI+PCIe+MSI-X, no danos driver) and readback-verifies every surface, including the first driver-side use of msi_bind. usb-xhci-bus converts from 8 ms event-ring polling to message-signalled interrupts: plain MSI where offered (real Intel xHC), MSI-X entry 0 otherwise (qemu-xhci has no MSI capability), byte-identical polling as fallback. The timer survives as a 250 ms port-reconcile/lost-edge tick — real-hardware USB2 hub debounce still needs it. MSI setup runs BEFORE controller bring-up: QEMU's xhci only registers the MSI-X vector as used when IMAN.IE is written while MSI-X is already enabled; interrupts are silently dropped otherwise (real hardware does not care about the order). 101/101 QEMU cases green; real-hardware smoke passed (mouse works, boot 2026-07-23T174805Z, plain-MSI branch, vector 33). --- build.zig | 15 +- library/device/pci/pci-class.zig | 158 +++++++++- library/device/pci/pci.zig | 290 +++++++++++++++++- system/drivers/usb-xhci-bus/usb-xhci-bus.zig | 104 ++++++- .../drivers/usb-xhci-bus/usb-xhci-library.zig | 9 + system/kernel/tests.zig | 37 +++ test/qemu_test.py | 12 + .../services/pci-cap-test/pci-cap-test.zig | 178 +++++++++++ 8 files changed, 781 insertions(+), 22 deletions(-) create mode 100644 test/system/services/pci-cap-test/pci-cap-test.zig diff --git a/build.zig b/build.zig index 1f3992e..8cee9a9 100644 --- a/build.zig +++ b/build.zig @@ -531,15 +531,18 @@ pub fn build(b: *std.Build) void { }, }); // A device driver's view of its claimed PCI function: config-space header fields, BAR - // decode + map, and the capability walk (library/device/pci/pci.zig). The generic PCI + // decode + map, capability walks (legacy + extended), MSI/MSI-X programming, power + // state, and function-level reset (library/device/pci/pci.zig). The generic PCI // mechanics every leaf PCI driver used to re-derive inline. Imports the driver (device - // access) client + mmio + the pci-class data module (config-space layout constants). + // access) client + mmio + the pci-class data module (config-space layout constants) + + // time (the D0 and FLR settle delays). const pci_module = b.addModule("pci", .{ .root_source_file = b.path("library/device/pci/pci.zig"), .imports = &.{ .{ .name = "driver", .module = driver_module }, .{ .name = "mmio", .module = mmio_module }, .{ .name = "pci-class", .module = pci_class_module }, + .{ .name = "time", .module = time_module }, }, }); @@ -677,6 +680,8 @@ pub fn build(b: *std.Build) void { programModule(ps2_mouse_exe).addImport("input-protocol", input_protocol_module); const usb_xhci_bus_exe = addUserBinary(b, kernel_target, &default_imports, "usb-xhci-bus", "system/drivers/usb-xhci-bus/usb-xhci-bus.zig"); programModule(usb_xhci_bus_exe).addImport("device-manager-protocol", device_manager_protocol_module); + // MSI setup: the claimed-function view (enable bits + MSI capability programming). + programModule(usb_xhci_bus_exe).addImport("pci", pci_module); // The xHCI bus driver builds chapter-9 requests and decodes descriptors from // usb-abi, and reports each interface's (class,subclass,protocol) identity via // usb-ids.packTriple. @@ -727,6 +732,11 @@ pub fn build(b: *std.Build) void { programModule(crash_test_exe).addImport("device-manager-protocol", device_manager_protocol_module); const device_list_exe = addUserBinary(b, kernel_target, &default_imports, "device-list", "test/system/services/device-list/device-list.zig"); programModule(device_list_exe).addImport("device-manager-protocol", device_manager_protocol_module); + // A test fixture: claims the pci-caps case's extra unclaimed NIC and exercises the + // driver-side PCI library surface (capabilities, MSI, MSI-X, power, FLR) against it. + const pci_cap_test_exe = addUserBinary(b, kernel_target, &default_imports, "pci-cap-test", "test/system/services/pci-cap-test/pci-cap-test.zig"); + programModule(pci_cap_test_exe).addImport("pci", pci_module); + programModule(pci_cap_test_exe).addImport("pci-class", pci_class_module); // The discovery service: one swappable process per firmware // (docs/discovery.md), bundled under the neutral ramdisk name // "discovery" so the device manager never learns which firmware it is on. @@ -811,6 +821,7 @@ pub fn build(b: *std.Build) void { .{ .path = "test/system/services/shared-memory-client", .binary = shared_memory_client_exe.getEmittedBin() }, .{ .path = "test/system/services/crash-test", .binary = crash_test_exe.getEmittedBin() }, .{ .path = "test/system/services/device-list", .binary = device_list_exe.getEmittedBin() }, + .{ .path = "test/system/services/pci-cap-test", .binary = pci_cap_test_exe.getEmittedBin() }, .{ .path = "test/system/services/input-source", .binary = input_source_exe.getEmittedBin() }, .{ .path = "test/system/services/input-test", .binary = input_test_exe.getEmittedBin() }, .{ .path = "test/system/services/args-echo", .binary = args_echo_exe.getEmittedBin() }, diff --git a/library/device/pci/pci-class.zig b/library/device/pci/pci-class.zig index 3a903ea..82c36b3 100644 --- a/library/device/pci/pci-class.zig +++ b/library/device/pci/pci-class.zig @@ -52,16 +52,134 @@ pub const config_vendor_id: usize = 0x00; pub const config_device_id: usize = 0x02; pub const config_command: usize = 0x04; pub const config_status: usize = 0x06; -pub const config_capabilities_pointer: usize = 0x34; +pub const config_revision_id: usize = 0x08; +pub const config_class_code: usize = 0x09; // 3 bytes: prog-IF 0x09, subclass 0x0A, base class 0x0B pub const config_bar0: usize = 0x10; // BAR0; BAR n is at config_bar0 + n*4 +pub const config_subsystem_vendor_id: usize = 0x2C; +pub const config_subsystem_id: usize = 0x2E; +pub const config_expansion_rom: usize = 0x30; +pub const config_capabilities_pointer: usize = 0x34; +pub const config_interrupt_line: usize = 0x3C; +pub const config_interrupt_pin: usize = 0x3D; // 0 = none, 1..4 = INTA..INTD -/// Command register: Memory-Space enable (bit 1) | Bus-Master enable (bit 2). -pub const command_memory_and_bus_master: u16 = 0x06; +/// Command register bits. +pub const command_io_space: u16 = 0x0001; // bit 0: I/O-space decode enable +pub const command_memory_space: u16 = 0x0002; // bit 1: memory-space decode enable +pub const command_bus_master: u16 = 0x0004; // bit 2: bus-master (DMA) enable +pub const command_interrupt_disable: u16 = 0x0400; // bit 10: suppress legacy INTx (MSI/MSI-X unaffected) +/// The pair a bus-mastering driver enables together: decode my BARs, let me DMA. +pub const command_memory_and_bus_master: u16 = command_memory_space | command_bus_master; + +/// Status register bit 3: legacy INTx is asserted (upstream of the command bit-10 gate). +pub const status_interrupt: u16 = 0x0008; /// Status register bit 4: a capability list is present at config_capabilities_pointer. pub const status_capabilities_list: u16 = 0x10; /// Capability pointers are dword-aligned; the low two bits are reserved. pub const capability_pointer_mask: u8 = 0xFC; +/// Capability IDs — the first byte of each entry in the legacy capability list. +/// Non-exhaustive: hardware may report IDs not named here. +pub const CapabilityId = enum(u8) { + power_management = 0x01, + msi = 0x05, + vendor_specific = 0x09, + pci_express = 0x10, + msix = 0x11, + _, +}; + +/// MSI capability (id 0x05) register layout. Offsets are relative to the capability +/// header; whether the address is one or two dwords (and therefore where the data word +/// sits) depends on `control_64bit_capable`. +pub const msi = struct { + pub const control: usize = 0x02; // u16 Message Control + pub const control_enable: u16 = 0x0001; + pub const control_multiple_message_capable_mask: u16 = 0x000E; // bits 3:1, log2(vectors requested) + pub const control_multiple_message_enable_mask: u16 = 0x0070; // bits 6:4, log2(vectors granted) + pub const control_64bit_capable: u16 = 0x0080; // bit 7: address is 64-bit (layout shifts) + pub const control_per_vector_masking: u16 = 0x0100; // bit 8 + pub const address: usize = 0x04; // u32 low address dword (both layouts) + pub const address_high: usize = 0x08; // u32, present only when 64-bit capable + pub const data_32: usize = 0x08; // u16 message data, 32-bit layout + pub const data_64: usize = 0x0C; // u16 message data, 64-bit layout + pub const mask_bits_32: usize = 0x0C; // u32, only with per-vector masking + pub const mask_bits_64: usize = 0x10; +}; + +/// MSI-X capability (id 0x11) register layout, plus the 16-byte vector table entry that +/// lives in BAR space (not configuration space) at the decoded (BIR, offset). +pub const msix = struct { + pub const control: usize = 0x02; // u16 Message Control + pub const control_table_size_mask: u16 = 0x07FF; // bits 10:0, encoded as N-1 + pub const control_function_mask: u16 = 0x4000; // bit 14: mask every vector + pub const control_enable: u16 = 0x8000; // bit 15 + pub const table_offset_word: usize = 0x04; // u32: BIR in bits 2:0, table offset in bits 31:3 + pub const pba_offset_word: usize = 0x08; // u32: same encoding, pending-bit array + pub const bir_mask: u32 = 0x0000_0007; + pub const offset_mask: u32 = 0xFFFF_FFF8; + pub const entry_size: usize = 16; // table entry stride; offsets within an entry: + pub const entry_address: usize = 0x0; // u32 low + pub const entry_address_high: usize = 0x4; // u32 high + pub const entry_data: usize = 0x8; // u32 + pub const entry_vector_control: usize = 0xC; // u32 + pub const entry_vector_control_masked: u32 = 0x1; // bit 0; entries reset to masked + + /// Where the table (or pending-bit array) lives, decoded from its offset/BIR dword. + pub const TableLocation = struct { bar: u8, offset: u32 }; + pub fn tableLocation(word: u32) TableLocation { + return .{ .bar = @intCast(word & bir_mask), .offset = word & offset_mask }; + } + /// Number of table entries (the control field encodes N-1). + pub fn tableSize(control_value: u16) u16 { + return (control_value & control_table_size_mask) + 1; + } +}; + +/// Power-management capability (id 0x01) register layout. +pub const power_management = struct { + pub const capabilities: usize = 0x02; // u16 PMC (read-only: version, D-state support) + pub const control_status: usize = 0x04; // u16 PMCSR + pub const control_status_power_state_mask: u16 = 0x0003; // bits 1:0 + pub const power_state_d0: u16 = 0x0; + pub const power_state_d3_hot: u16 = 0x3; + pub const control_status_pme_enable: u16 = 0x0100; // bit 8: plain RW — preserve on writes + pub const control_status_pme_status: u16 = 0x8000; // bit 15: RW1C — write 0 or you clear it +}; + +/// PCI Express capability (id 0x10) register layout — the slice function-level reset +/// needs; the full capability is much larger. +pub const pci_express = struct { + pub const capabilities: usize = 0x02; // u16 PCIe Capabilities register + pub const device_capabilities: usize = 0x04; // u32 + pub const device_capabilities_flr: u32 = 1 << 28; // Function Level Reset supported + pub const device_control: usize = 0x08; // u16 + pub const device_control_initiate_flr: u16 = 1 << 15; + pub const device_status: usize = 0x0A; // u16 + pub const device_status_transactions_pending: u16 = 1 << 5; +}; + +/// Extended (PCI Express) capabilities start here in the 4 KiB configuration space; a +/// conventional-PCI function has nothing there (the space reads as all-ones). +pub const extended_capability_start: usize = 0x100; +/// Extended-capability next pointers are dword-aligned within the 4 KiB space. +pub const extended_capability_pointer_mask: u16 = 0xFFC; + +/// The 32-bit header at the start of each extended capability: ID in bits 15:0, +/// version in 19:16, next offset in 31:20 (0 = end of list). +pub const ExtendedCapabilityHeader = struct { + id: u16, + version: u4, + next: u16, + + pub fn decode(word: u32) ExtendedCapabilityHeader { + return .{ + .id = @truncate(word), + .version = @truncate(word >> 16), + .next = @intCast((word >> 20) & extended_capability_pointer_mask), + }; + } +}; + /// BAR bit layout: bit 0 selects I/O (1) vs memory (0) space; for a memory BAR, bits 2:1 /// give the type (00 = 32-bit, 10 = 64-bit spanning the next BAR), and the base address is /// the dword with the low 4 flag bits masked off. @@ -595,3 +713,37 @@ test "named parts pack to the raw triple" { }; try std.testing.expectEqual(@as(u24, 0x0C_03_30), xhci.pack()); } + +test "MSI-X table word decodes to BIR and offset" { + const eq = std.testing.expectEqual; + // BIR 3, table at 0x2000 within that BAR. + try eq(msix.TableLocation{ .bar = 3, .offset = 0x2000 }, msix.tableLocation(0x0000_2003)); + // BIR 0, offset 0 — the degenerate-but-common "table at BAR start" case. + try eq(msix.TableLocation{ .bar = 0, .offset = 0 }, msix.tableLocation(0)); + // Table size encodes N-1 in bits 10:0; enable/function-mask bits must not leak in. + try eq(@as(u16, 11), msix.tableSize(msix.control_enable | 0x000A)); + try eq(@as(u16, 1), msix.tableSize(0)); + try eq(@as(u16, 2048), msix.tableSize(msix.control_table_size_mask)); +} + +test "extended capability header unpacks id, version, next" { + const eq = std.testing.expectEqual; + // AER (id 0x0001), version 1, next capability at 0x140. + const aer = ExtendedCapabilityHeader.decode(0x1401_0001); + try eq(@as(u16, 0x0001), aer.id); + try eq(@as(u4, 1), aer.version); + try eq(@as(u16, 0x140), aer.next); + // A zero header is the "nothing here" terminator. + const none = ExtendedCapabilityHeader.decode(0); + try eq(@as(u16, 0), none.id); + try eq(@as(u16, 0), none.next); +} + +test "command bits and capability ids compose" { + const eq = std.testing.expectEqual; + try eq(command_memory_space | command_bus_master, command_memory_and_bus_master); + try eq(@as(u8, 0x05), @intFromEnum(CapabilityId.msi)); + try eq(@as(u8, 0x11), @intFromEnum(CapabilityId.msix)); + try eq(@as(u8, 0x01), @intFromEnum(CapabilityId.power_management)); + try eq(@as(u8, 0x10), @intFromEnum(CapabilityId.pci_express)); +} diff --git a/library/device/pci/pci.zig b/library/device/pci/pci.zig index 2e72357..ca8b7b5 100644 --- a/library/device/pci/pci.zig +++ b/library/device/pci/pci.zig @@ -1,7 +1,8 @@ //! library/device/pci/pci.zig — a device driver's view of the ONE PCI function it has -//! claimed. Config space is mapped as resource 0; this gives header-field accessors, BAR -//! decode + map, and a capability-list iterator, so a driver never re-derives the -//! config-space layout by hand. +//! claimed. Config space is mapped as resource 0 (a full 4 KiB ECAM page); this gives +//! header-field accessors, BAR decode + map, capability walks (legacy and extended), +//! MSI/MSI-X programming, power-state handling, and function-level reset, so a driver +//! never re-derives the config-space layout by hand. //! //! This is the *device-owned* view: read my own function's live config, map my own BARs. //! The bus enumerator's view — probing arbitrary, not-yet-claimed functions and sizing @@ -13,6 +14,18 @@ const std = @import("std"); const mmio = @import("mmio"); const pci_class = @import("pci-class"); const device = @import("driver"); +const time = @import("time"); + +/// Spec recovery time after a D3hot -> D0 transition. +const d0_recovery_millis: u64 = 10; +/// How long to wait for in-flight transactions to drain before a function-level reset +/// (then reset anyway — resetting a stuck function is the point of FLR). +const flr_pending_timeout_millis: u64 = 100; +/// The spec's maximum FLR completion time. +const flr_settle_millis: u64 = 100; +/// How long to wait for the function to become readable again after an FLR. +const flr_ready_timeout_millis: u64 = 1000; +const flr_poll_interval_millis: u64 = 10; /// A claimed PCI function whose configuration space is mapped (resource 0). `descriptor` /// must outlive the Function — the driver's `device.enumerate` buffer does, for the whole @@ -42,12 +55,61 @@ pub const Function = struct { pub fn status(self: *const Function) u16 { return mmio.readRegister(u16, self.config + pci_class.config_status); } + pub fn revisionId(self: *const Function) u8 { + return mmio.readRegister(u8, self.config + pci_class.config_revision_id); + } + /// Subsystem vendor ID (config 0x2C) — with `subsystemId`, the standard key for + /// board-level quirk matching. + pub fn subsystemVendorId(self: *const Function) u16 { + return mmio.readRegister(u16, self.config + pci_class.config_subsystem_vendor_id); + } + pub fn subsystemId(self: *const Function) u16 { + return mmio.readRegister(u16, self.config + pci_class.config_subsystem_id); + } + /// Interrupt pin (config 0x3D): 0 = none, 1..4 = INTA..INTD. + pub fn interruptPin(self: *const Function) u8 { + return mmio.readRegister(u8, self.config + pci_class.config_interrupt_pin); + } + /// The live class-code triple (config 0x09..0x0B), same shape discovery records. + pub fn classCode(self: *const Function) pci_class.ClassCode { + return .{ + .prog_if = mmio.readRegister(u8, self.config + pci_class.config_class_code), + .subclass = mmio.readRegister(u8, self.config + pci_class.config_class_code + 1), + .base = mmio.readRegister(u8, self.config + pci_class.config_class_code + 2), + }; + } - /// Set Memory-Space + Bus-Master enable in the command register. Firmware often leaves - /// a secondary display's decode off; a bus-mastering device must enable both. - pub fn enableMemoryAndBusMaster(self: *const Function) void { + fn commandSetBits(self: *const Function, bits: u16) void { const at = self.config + pci_class.config_command; - mmio.writeRegister(u16, at, mmio.readRegister(u16, at) | pci_class.command_memory_and_bus_master); + mmio.writeRegister(u16, at, mmio.readRegister(u16, at) | bits); + } + fn commandClearBits(self: *const Function, bits: u16) void { + const at = self.config + pci_class.config_command; + mmio.writeRegister(u16, at, mmio.readRegister(u16, at) & ~bits); + } + + /// Set Memory-Space + Bus-Master Enable in the command register. Firmware only enables + /// memory decode on devices it used at boot; any other device has dead BARs until its + /// driver sets it. Bus mastering is separately required for the device to do DMA. + pub fn enableMemoryAndBusMaster(self: *const Function) void { + self.commandSetBits(pci_class.command_memory_and_bus_master); + } + + /// Clear Bus-Master Enable — stop the device initiating DMA. The quiesce half of a + /// driver's shutdown (or a supervisor restart): after this the device can no longer + /// write memory the process is about to stop owning. + pub fn disableBusMaster(self: *const Function) void { + self.commandClearBits(pci_class.command_bus_master); + } + + /// Set command bit 10: suppress legacy INTx assertion. MSI/MSI-X are unaffected — + /// set this when enabling either, so the device cannot also raise the shared pin. + pub fn setInterruptDisable(self: *const Function) void { + self.commandSetBits(pci_class.command_interrupt_disable); + } + /// Clear command bit 10, re-allowing legacy INTx assertion. + pub fn clearInterruptDisable(self: *const Function) void { + self.commandClearBits(pci_class.command_interrupt_disable); } /// Decode BAR `bar` (0..5) and map it: read the BAR register, reject I/O-space BARs, @@ -86,6 +148,129 @@ pub const Function = struct { 0; return .{ .config = self.config, .cursor = first }; } + + /// First capability with `id`, or null. + pub fn findCapability(self: *const Function, id: pci_class.CapabilityId) ?Capability { + var walk = self.capabilities(); + while (walk.next()) |capability| { + if (capability.id == @intFromEnum(id)) return capability; + } + return null; + } + + /// Program the MSI capability with the kernel's `msi_bind` result and enable it — + /// one vector (multiple-message-enable 0, matching the kernel's single-vector + /// grant), INTx suppressed. false if the function has no MSI capability. + pub fn programMsi(self: *const Function, message: device.Msi) bool { + const cap = self.findCapability(.msi) orelse return false; + const control_at = cap.offset + pci_class.msi.control; + const control = mmio.readRegister(u16, control_at); + // Program the registers while the capability is disabled. + mmio.writeRegister(u16, control_at, control & ~pci_class.msi.control_enable); + mmio.writeRegister(u32, cap.offset + pci_class.msi.address, @truncate(message.address)); + const data_offset = if (control & pci_class.msi.control_64bit_capable != 0) offset: { + mmio.writeRegister(u32, cap.offset + pci_class.msi.address_high, @intCast(message.address >> 32)); + break :offset pci_class.msi.data_64; + } else pci_class.msi.data_32; + // Message data is a 16-bit register in both layouts. + mmio.writeRegister(u16, cap.offset + data_offset, @truncate(message.data)); + mmio.writeRegister(u16, control_at, (control & ~pci_class.msi.control_multiple_message_enable_mask) | pci_class.msi.control_enable); + self.setInterruptDisable(); + return true; + } + + /// Clear the MSI enable bit. No-op if the function has no MSI capability. + pub fn disableMsi(self: *const Function) void { + const cap = self.findCapability(.msi) orelse return; + const control_at = cap.offset + pci_class.msi.control; + mmio.writeRegister(u16, control_at, mmio.readRegister(u16, control_at) & ~pci_class.msi.control_enable); + } + + /// The function's MSI-X capability with its vector table mapped: the table's BIR is + /// resolved through `mapBar` (a free cache hit when it is a BAR the driver already + /// mapped). null if the capability is absent or the table's BAR cannot be mapped. + pub fn msix(self: *Function) ?MsiX { + const cap = self.findCapability(.msix) orelse return null; + const control = mmio.readRegister(u16, cap.offset + pci_class.msix.control); + const word = mmio.readRegister(u32, cap.offset + pci_class.msix.table_offset_word); + const location = pci_class.msix.tableLocation(word); + const bar_base = self.mapBar(location.bar) orelse return null; + return .{ + .capability = cap.offset, + .table = bar_base + location.offset, + .entry_count = pci_class.msix.tableSize(control), + }; + } + + /// Bring the function to D0. Firmware can leave a non-boot device in D3hot, where + /// its BARs and MSI registers do not decode; call this before touching either. No + /// power-management capability means the function is always at D0: nothing to do. + /// Preserves PME-Enable and never clears the write-1-to-clear PME-Status bit. + pub fn ensurePowerStateD0(self: *const Function) void { + const cap = self.findCapability(.power_management) orelse return; + const at = cap.offset + pci_class.power_management.control_status; + const pmcsr = mmio.readRegister(u16, at); + if (pmcsr & pci_class.power_management.control_status_power_state_mask == pci_class.power_management.power_state_d0) return; + // PME-Status is RW1C: echoing a read 1 back would clear it, so write it as 0. + mmio.writeRegister(u16, at, (pmcsr & ~pci_class.power_management.control_status_power_state_mask & ~pci_class.power_management.control_status_pme_status) | pci_class.power_management.power_state_d0); + time.sleepMillis(d0_recovery_millis); + } + + /// Function Level Reset via the PCI Express capability: return the hardware to a + /// known state (a supervisor re-claiming a device after its driver died, or a driver + /// recovering a wedged function). The six BAR dwords are saved and restored — FLR + /// clears them, and the bus enumerator's assignment must survive for the descriptor + /// correlation and `mapBar` cache to stay valid. Everything else is reset: command + /// enables and MSI/MSI-X programming are gone, so the caller re-runs its whole + /// bring-up afterwards. false if the function has no PCI Express capability, does + /// not advertise FLR (conventional-PCI Advanced Features FLR is a possible + /// follow-up), or never became readable again. Blocks for at least 100 ms. + pub fn functionLevelReset(self: *const Function) bool { + const cap = self.findCapability(.pci_express) orelse return false; + const device_capabilities = mmio.readRegister(u32, cap.offset + pci_class.pci_express.device_capabilities); + if (device_capabilities & pci_class.pci_express.device_capabilities_flr == 0) return false; + + // Stop new DMA, then give in-flight transactions a bounded chance to drain — + // and reset anyway on timeout, since resetting a stuck function is the point. + self.disableBusMaster(); + var waited: u64 = 0; + while (mmio.readRegister(u16, cap.offset + pci_class.pci_express.device_status) & pci_class.pci_express.device_status_transactions_pending != 0) { + if (waited >= flr_pending_timeout_millis) break; + time.sleepMillis(flr_poll_interval_millis); + waited += flr_poll_interval_millis; + } + + var bars: [6]u32 = undefined; + for (&bars, 0..) |*bar, index| bar.* = mmio.readRegister(u32, self.config + pci_class.config_bar0 + index * 4); + + const control_at = cap.offset + pci_class.pci_express.device_control; + mmio.writeRegister(u16, control_at, mmio.readRegister(u16, control_at) | pci_class.pci_express.device_control_initiate_flr); + time.sleepMillis(flr_settle_millis); + + waited = 0; + while (self.vendorId() == 0xFFFF) { + if (waited >= flr_ready_timeout_millis) return false; + time.sleepMillis(flr_poll_interval_millis); + waited += flr_poll_interval_millis; + } + for (bars, 0..) |bar, index| mmio.writeRegister(u32, self.config + pci_class.config_bar0 + index * 4, bar); + return true; + } + + /// Iterate the extended (PCI Express) capability list at 0x100.. in the 4 KiB ECAM + /// page. Empty on a conventional-PCI function (the space reads as all-ones). + pub fn extendedCapabilities(self: *const Function) ExtendedCapabilityIterator { + return .{ .config = self.config }; + } + + /// First extended capability with `id`, or null. + pub fn findExtendedCapability(self: *const Function, id: u16) ?ExtendedCapability { + var walk = self.extendedCapabilities(); + while (walk.next()) |capability| { + if (capability.id == id) return capability; + } + return null; + } }; /// One capability header. `offset` is the ABSOLUTE virtual address of the header, so the @@ -106,3 +291,94 @@ pub const CapabilityIterator = struct { return .{ .id = id, .offset = at }; } }; + +/// A resolved MSI-X capability from `Function.msix`: `capability` is the absolute +/// virtual address of the config-space header, `table` of vector-table entry 0 (in BAR +/// space — table writes are MMIO, not config space). Entries reset masked; bring-up +/// order is programEntry per vector, unmaskEntry per used vector, `enable`, then +/// `Function.setInterruptDisable`. +pub const MsiX = struct { + capability: usize, + table: usize, + entry_count: u16, + + /// Write `message` into table entry `entry`, leaving the entry masked (its reset + /// state) — the spec requires masking while address/data change. false if `entry` + /// is out of range. + pub fn programEntry(self: *const MsiX, entry: u16, message: device.Msi) bool { + if (entry >= self.entry_count) return false; + const at = self.table + @as(usize, entry) * pci_class.msix.entry_size; + mmio.writeRegister(u32, at + pci_class.msix.entry_vector_control, pci_class.msix.entry_vector_control_masked); + mmio.writeRegister(u32, at + pci_class.msix.entry_address, @truncate(message.address)); + mmio.writeRegister(u32, at + pci_class.msix.entry_address_high, @intCast(message.address >> 32)); + mmio.writeRegister(u32, at + pci_class.msix.entry_data, message.data); + return true; + } + + /// Set the entry's vector-control mask bit — its interrupt is held off (pended in + /// the PBA, not lost). false if `entry` is out of range. + pub fn maskEntry(self: *const MsiX, entry: u16) bool { + return self.writeEntryMask(entry, true); + } + /// Clear the entry's vector-control mask bit. false if `entry` is out of range. + pub fn unmaskEntry(self: *const MsiX, entry: u16) bool { + return self.writeEntryMask(entry, false); + } + fn writeEntryMask(self: *const MsiX, entry: u16, masked: bool) bool { + if (entry >= self.entry_count) return false; + const at = self.table + @as(usize, entry) * pci_class.msix.entry_size + pci_class.msix.entry_vector_control; + const control = mmio.readRegister(u32, at); + mmio.writeRegister(u32, at, if (masked) + control | pci_class.msix.entry_vector_control_masked + else + control & ~pci_class.msix.entry_vector_control_masked); + return true; + } + + /// Set the function-mask control bit: every vector masked regardless of entry bits. + pub fn setFunctionMask(self: *const MsiX) void { + self.writeControl(pci_class.msix.control_function_mask, true); + } + /// Clear the function-mask control bit. + pub fn clearFunctionMask(self: *const MsiX) void { + self.writeControl(pci_class.msix.control_function_mask, false); + } + /// Set MSI-X Enable. The caller also calls `Function.setInterruptDisable` (INTx off). + pub fn enable(self: *const MsiX) void { + self.writeControl(pci_class.msix.control_enable, true); + } + /// Clear MSI-X Enable. + pub fn disable(self: *const MsiX) void { + self.writeControl(pci_class.msix.control_enable, false); + } + fn writeControl(self: *const MsiX, bit: u16, set: bool) void { + const at = self.capability + pci_class.msix.control; + const control = mmio.readRegister(u16, at); + mmio.writeRegister(u16, at, if (set) control | bit else control & ~bit); + } +}; + +/// One extended capability. `offset` is the ABSOLUTE virtual address of its header, +/// like `Capability.offset`. +pub const ExtendedCapability = struct { id: u16, version: u4, offset: usize }; + +pub const ExtendedCapabilityIterator = struct { + config: usize, + cursor: u16 = @intCast(pci_class.extended_capability_start), + guard: u32 = 0, // bounds a malformed chain (480 = the 0xF00-byte space / 8-byte minimum spacing) + + pub fn next(self: *ExtendedCapabilityIterator) ?ExtendedCapability { + if (self.cursor == 0 or self.guard >= 480) return null; + self.guard += 1; + const at = self.config + self.cursor; + const header = pci_class.ExtendedCapabilityHeader.decode(mmio.readRegister(u32, at)); + // Id 0 marks an empty list; all-ones is a conventional-PCI function (no + // extended space — reads come back as FFs). + if (header.id == 0 or header.id == 0xFFFF) return null; + // A next pointer below 0x100 would walk into the legacy header; treat it as the + // terminator it must be (0 is the normal one). The 0xFFC decode mask already + // keeps `config + cursor + 4` inside the 4 KiB page. + self.cursor = if (header.next >= pci_class.extended_capability_start) header.next else 0; + return .{ .id = header.id, .version = header.version, .offset = at }; + } +}; diff --git a/system/drivers/usb-xhci-bus/usb-xhci-bus.zig b/system/drivers/usb-xhci-bus/usb-xhci-bus.zig index af514f0..7bd020d 100644 --- a/system/drivers/usb-xhci-bus/usb-xhci-bus.zig +++ b/system/drivers/usb-xhci-bus/usb-xhci-bus.zig @@ -28,18 +28,37 @@ const usb_ids = @import("usb-ids"); const usb_abi = @import("usb-abi"); const usb_transfer_protocol = @import("usb-transfer-protocol"); const library = @import("usb-xhci-library.zig"); +const pci = @import("pci"); /// The controller engine (reset, rings, transfers), stood up in `initialise`. var controller: ?library.Controller = null; /// This driver's service endpoint (registered as `.usb_bus`), where class-driver -/// requests, signals, and the interrupt-poll timer all arrive. +/// requests, signals, MSI notifications, and the poll/reconcile timer all arrive. var service_endpoint: ipc.Handle = 0; -/// How often the driver drains the event ring for interrupt reports (~125 Hz), -/// re-armed each tick. Frequent enough for responsive input. +/// How often the driver drains the event ring for interrupt reports (~125 Hz) when +/// polling, re-armed each tick. Frequent enough for responsive input. const poll_interval_ms: u64 = 8; +/// The timer interval in MSI mode: the ring is drained at interrupt time, and the tick +/// only reconciles root ports (real hardware delivers late USB2 companion-hub debounce +/// with no reliable port-change event — see onNotification) and un-wedges a lost MSI +/// edge (edge-triggered, no kernel mask/ack: a missed IP clear stalls, never storms). +const reconcile_interval_ms: u64 = 250; + +/// The controller's own descriptor, kept at file scope because `pci.Function` holds a +/// pointer to it for the whole bring-up. +var controller_descriptor: device.DeviceDescriptor = undefined; + +/// Non-null iff MSI mode is active: the vector whose notification badge means "the +/// controller interrupted". Null means the 8 ms polling fallback is running. +var msi_vector: ?u32 = null; + +fn timerInterval() u64 { + return if (msi_vector != null) reconcile_interval_ms else poll_interval_ms; +} + /// The class driver endpoints that opened each device, so interrupt reports can /// be pushed back to them. Keyed by the device token (the interface's device id). const Open = struct { @@ -95,6 +114,7 @@ fn initialise(endpoint: ipc.Handle) bool { std.log.info("device {d} not in the device tree", .{controller_id}); return false; }; + controller_descriptor = descriptor; // The xHC's registers live behind the first memory BAR. Resource 0 is the // function's ECAM configuration space (M15), so the walk starts at 1. @@ -118,6 +138,14 @@ fn initialise(endpoint: ipc.Handle) bool { return false; }; + // Message-signalled interrupt setup comes BEFORE the controller bring-up, not + // after: Controller.init writes IMAN.IE, and QEMU's xhci only registers the MSI-X + // vector as in-use when that write happens with MSI-X already enabled (its + // intr_update callback early-outs on !msix_enabled, and msix_notify silently + // drops interrupts for an unused vector). Real hardware does not care about the + // order; QEMU requires it. + setupMsi(); + // Bring the controller up: reset it, stand up the command and event rings, // and start it running (the hardware half lives in usb-xhci-library.zig). controller = library.Controller.init(register_base) orelse { @@ -146,12 +174,52 @@ fn initialise(endpoint: ipc.Handle) bool { scanPorts(handle); - // Arm the poll timer that drains interrupt reports from the event ring. It is - // re-armed on each tick in onNotification; class drivers subscribe later. - _ = time.timerOnce(service_endpoint, poll_interval_ms); + // Arm the timer: in polling mode it drains the event ring; in MSI mode it is the + // slower port-reconcile/safety-net tick. Re-armed on each tick in onNotification. + _ = time.timerOnce(service_endpoint, timerInterval()); return true; } +/// Switch the event ring from timer polling to message-signalled interrupts, if the +/// whole path is available: map the function's config space, bind a vector, program +/// the MSI capability — or, on an MSI-X-only function (QEMU's qemu-xhci is one: it +/// advertises MSI-X and PCIe but no plain MSI), entry 0 of the MSI-X table, which +/// takes the same kernel (address, data) pair (xHCI interrupter 0 raises vector 0). +/// Any step failing leaves `msi_vector` null and the 8 ms polling path exactly as it +/// was. The controller side needs nothing extra — IMAN.IE and USBCMD.INTE are already +/// set (see Controller.init: QEMU only writes runtime events with the interrupter +/// enabled). +/// +/// After a supervised kill, the kernel drops the vector binding but the device still +/// has the interrupt enabled and fires the stale vector; the kernel EOIs it +/// harmlessly, and the respawned driver re-runs this with its fresh vector. +fn setupMsi() void { + var function = pci.Function.map(controller_id, &controller_descriptor) orelse { + std.log.info("config-space map failed; polling at {d} ms", .{poll_interval_ms}); + return; + }; + function.enableMemoryAndBusMaster(); + const message = device.msiBind(controller_id, service_endpoint) orelse { + std.log.info("msi_bind unavailable; polling at {d} ms", .{poll_interval_ms}); + return; + }; + if (function.programMsi(message)) { + msi_vector = message.data; + std.log.info("msi active (vector {d}); reconcile tick at {d} ms", .{ message.data, reconcile_interval_ms }); + return; + } + if (function.msix()) |table| { + if (table.programEntry(0, message) and table.unmaskEntry(0)) { + table.enable(); + function.setInterruptDisable(); + msi_vector = message.data; + std.log.info("msix active (vector {d}); reconcile tick at {d} ms", .{ message.data, reconcile_interval_ms }); + return; + } + } + std.log.info("no msi/msi-x capability; polling at {d} ms", .{poll_interval_ms}); +} + var register_base: usize = 0; /// The xHCI default Protocol Speed IDs (the PORTSC port-speed field, bits 13:10) @@ -501,10 +569,27 @@ fn handleBulk(message: []const u8, reply: []u8) usize { return writeReply(reply, usb_transfer_protocol.BulkReply{ .status = if (transferred != null) 0 else -1, .actual_length = transferred orelse 0 }); } -/// The poll timer landed: drain any interrupt reports off the event ring and push -/// each to the class driver that subscribed, then re-arm the timer. +/// A timer tick or an MSI landed: drain the event ring, reconcile ports, and fan out. +/// The timer arm re-arms itself (8 ms drain when polling, 250 ms reconcile under MSI); +/// the MSI arm clears the interrupter's pending bit FIRST, then drains — so an event +/// arriving after the drain takes IP 0→1 and fires a fresh edge instead of being +/// swallowed until the reconcile tick. fn onNotification(badge: u64) void { - if (badge & ipc.notify_timer_bit == 0) return; + if (badge & ipc.notify_timer_bit != 0) { + serviceController(); + _ = time.timerOnce(service_endpoint, timerInterval()); + return; + } + const vector = msi_vector orelse return; + if (badge & ~ipc.notify_badge_bit != vector) return; + if (controller) |*engine| engine.acknowledgeInterrupt(); + serviceController(); +} + +/// Everything one servicing pass does, shared verbatim by the poll/reconcile tick and +/// the MSI notification: drain the event ring, reconcile root ports, service hub +/// changes, and push interrupt reports to their class drivers. +fn serviceController() void { if (controller) |*engine| { engine.pump(); // Poll every root port and reconcile — a device present but not yet @@ -564,7 +649,6 @@ fn onNotification(badge: u64) void { _ = ipc.send(report.report_endpoint, std.mem.asBytes(&message)); } } - _ = time.timerOnce(service_endpoint, poll_interval_ms); } pub fn main(init: process.Init) void { diff --git a/system/drivers/usb-xhci-bus/usb-xhci-library.zig b/system/drivers/usb-xhci-bus/usb-xhci-library.zig index fa3f0fc..89e5f64 100644 --- a/system/drivers/usb-xhci-bus/usb-xhci-library.zig +++ b/system/drivers/usb-xhci-bus/usb-xhci-library.zig @@ -1563,6 +1563,15 @@ pub const Controller = struct { return report; } + /// Clear interrupter 0's pending bit (IMAN.IP). IP is write-1-to-clear, and the + /// read-back carries IE (plain read-write) through unchanged. In MSI mode the + /// driver clears IP **before** draining the ring: an event that lands after the + /// drain then takes IP 0→1 and fires a fresh edge, where clearing afterwards would + /// leave a race in which a new event finds IP already set and raises nothing. + pub fn acknowledgeInterrupt(self: *const Controller) void { + write32(self.interrupter(interrupter_management), read32(self.interrupter(interrupter_management)) | 1); + } + /// Drain any events currently on the event ring: interrupt reports into the /// report queue, PORT STATUS CHANGES into the port-change queue (hot-plug — /// these were silently dropped before M20). Non-blocking — called on the diff --git a/system/kernel/tests.zig b/system/kernel/tests.zig index 644e79c..c119cd9 100644 --- a/system/kernel/tests.zig +++ b/system/kernel/tests.zig @@ -212,6 +212,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void { deviceListTest(boot_information); } else if (eql(case, "pci-scan")) { pciScanTest(boot_information); + } else if (eql(case, "pci-caps")) { + pciCapsTest(boot_information); } else if (eql(case, "acpi-parse")) { acpiParseTest(boot_information); } else if (eql(case, "acpi-report")) { @@ -2388,6 +2390,41 @@ fn deviceListTest(boot_information: *const BootInformation) void { result(); } +/// The driver-side PCI library against a real function: the pci-caps QEMU case adds an +/// e1000e NIC no danos driver claims; the pci-cap-test fixture claims it and exercises +/// header accessors, command bits, the capability walks, MSI programming (the first +/// driver-side `msi_bind` use), the MSI-X table, power state, and FLR. The kernel side +/// only spawns the manager (which spawns pci-bus itself) and the fixture; the substance +/// is asserted by the harness on the fixture's own serial lines. +fn pciCapsTest(boot_information: *const BootInformation) void { + log("DANOS-TEST-BEGIN: pci-caps\n", .{}); + if (boot_information.initial_ramdisk_len == 0) { + check("bootloader handed over an initial_ramdisk", false); + result(); + return; + } + const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len]; + const rd = initial_ramdisk.Reader.init(image) orelse { + check("initial_ramdisk image is valid", false); + result(); + return; + }; + + process.setInitialRamdisk(image); + // Plain mode — no restart drill, whose kill would race the fixture's claim. + var manager: u32 = 0; + var i: u32 = 0; + while (i < rd.count) : (i += 1) { + const item = rd.entry(i) orelse continue; + if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue; + manager = process.spawnProcessSupervised(item.blob, 4, &.{"device-manager"}, scheduler.currentId(), null) catch 0; + break; + } + check("device-manager spawned", manager != 0); + check("pci-cap-test spawned", spawnNamed(rd, "pci-cap-test")); + result(); +} + /// M19.1: the ring-3 PCI scan agrees with the kernel's. The manager spawns /// pci-bus for the host bridge; the driver walks the same ECAM window through /// its mmio_map grant and must find exactly the functions the kernel's own diff --git a/test/qemu_test.py b/test/qemu_test.py index 117c6fb..2798152 100644 --- a/test/qemu_test.py +++ b/test/qemu_test.py @@ -687,6 +687,18 @@ CASES = [ # duplicates); this ordered regex asserts the drill itself over the whole serial # log — the backreference requires the respawn to re-scan the same count, and the # full-capture match is immune to the transient-line races an in-kernel poll hits. + # The driver-side PCI library (library/device/pci) against a real function: an extra + # e1000e NIC — PM + MSI + PCIe + MSI-X capabilities, claimed by no danos driver — is + # claimed by the pci-cap-test fixture, which exercises the capability walks, MSI + # programming (the first driver-side msi_bind use), the MSI-X table, power state, + # and FLR, printing a marker per check. + {"name": "pci-caps", + "smp": 4, + "timeout": 120, + "qemu_extra": ["-device", "e1000e"], + "expect": r"(?s)(?=.*DANOS-TEST-RESULT: PASS)(?=.*pci-cap-test: all checks passed)", + "fail": r"pci-cap-test: FAIL|DANOS-TEST-RESULT: FAIL"}, + {"name": "pci-scan", "smp": 4, "timeout": 60, diff --git a/test/system/services/pci-cap-test/pci-cap-test.zig b/test/system/services/pci-cap-test/pci-cap-test.zig new file mode 100644 index 0000000..f4ae39e --- /dev/null +++ b/test/system/services/pci-cap-test/pci-cap-test.zig @@ -0,0 +1,178 @@ +//! pci-cap-test — QEMU fixture for the driver-side PCI library (library/device/pci). +//! The pci-caps test case boots with an extra `-device e1000e` NIC that no danos driver +//! claims; this fixture claims it and exercises the whole claimed-function surface +//! against real (emulated) hardware: header accessors, command bits, the capability +//! walk, MSI programming (the first driver-side `msi_bind` use), the MSI-X table, +//! extended capabilities, power state, and — where offered — function-level reset. +//! Every check prints `pci-cap-test: ok` or `pci-cap-test: FAIL `; the +//! harness asserts on the final `all checks passed` marker (test/qemu_test.py). + +const std = @import("std"); +const device = @import("driver"); +const ipc = @import("ipc"); +const time = @import("time"); +const logging = @import("logging"); +const mmio = @import("mmio"); +const pci = @import("pci"); +const pci_class = @import("pci-class"); + +/// QEMU's e1000e: Intel 82574L. +const intel_vendor: u16 = 0x8086; +const e1000e_device: u16 = 0x10D3; + +const ethernet_class: u64 = pci_class.ClassCode.pack(.{ + .base = @intFromEnum(pci_class.BaseClass.network), + .subclass = @intFromEnum(pci_class.network.SubClass.ethernet), + .prog_if = 0, +}); + +/// `pci.Function` keeps a pointer to the descriptor, so it must outlive the stack frame +/// that found it. +var descriptor: device.DeviceDescriptor = undefined; + +fn writeLine(comptime fmt: []const u8, arguments: anytype) void { + var line: [128]u8 = undefined; + _ = logging.write(std.fmt.bufPrint(&line, fmt, arguments) catch return); +} + +/// Print the check's verdict; the caller returns on false to stop at the first failure. +fn check(comptime name: []const u8, ok: bool) bool { + if (ok) { + _ = logging.write("pci-cap-test: " ++ name ++ " ok\n"); + } else { + _ = logging.write("pci-cap-test: FAIL " ++ name ++ "\n"); + } + return ok; +} + +pub fn main() void { + // The bus scan runs in another process; poll until the NIC shows up. + const nic_id: u64 = found: { + var tries: u32 = 0; + while (tries < 150) : (tries += 1) { + var descriptors: [64]device.DeviceDescriptor = undefined; + const total = device.enumerate(&descriptors); + for (descriptors[0..@min(total, descriptors.len)]) |*entry| { + if (entry.class == @intFromEnum(device.DeviceClass.pci_device) and entry.pci_class == ethernet_class) { + descriptor = entry.*; + break :found entry.id; + } + } + time.sleepMillis(100); + } + _ = logging.write("pci-cap-test: FAIL no ethernet function found\n"); + return; + }; + writeLine("pci-cap-test: claiming ethernet function (device {d})\n", .{nic_id}); + if (!check("claim", device.claim(nic_id))) return; + var function = pci.Function.map(nic_id, &descriptor) orelse { + _ = logging.write("pci-cap-test: FAIL config-space map\n"); + return; + }; + + // Identity: the header accessors against known e1000e values. + if (!check("vendor/device id", function.vendorId() == intel_vendor and function.deviceId() == e1000e_device)) return; + if (!check("class code", function.classCode().pack() == ethernet_class)) return; + if (!check("subsystem ids readable", function.subsystemVendorId() != 0xFFFF and function.subsystemId() != 0xFFFF)) return; + + // Command bits: enable, read back, quiesce, read back, re-enable. + function.enableMemoryAndBusMaster(); + if (!check("memory+bus-master enable", function.command() & pci_class.command_memory_and_bus_master == pci_class.command_memory_and_bus_master)) return; + function.disableBusMaster(); + if (!check("bus-master disable", function.command() & pci_class.command_bus_master == 0)) return; + function.enableMemoryAndBusMaster(); + + // The capability walk: e1000e advertises PM, MSI, PCIe, and MSI-X. + var seen_power = false; + var seen_msi = false; + var seen_pci_express = false; + var seen_msix = false; + var walk = function.capabilities(); + while (walk.next()) |capability| { + switch (capability.id) { + @intFromEnum(pci_class.CapabilityId.power_management) => seen_power = true, + @intFromEnum(pci_class.CapabilityId.msi) => seen_msi = true, + @intFromEnum(pci_class.CapabilityId.pci_express) => seen_pci_express = true, + @intFromEnum(pci_class.CapabilityId.msix) => seen_msix = true, + else => {}, + } + } + if (!check("capability walk", seen_power and seen_msi and seen_pci_express and seen_msix)) return; + if (!check("findCapability", function.findCapability(.msi) != null and function.findCapability(.pci_express) != null)) return; + + // MSI: bind a vector (the syscall's first driver-side use), program the capability, + // and read the registers straight back. + const endpoint = ipc.createIpcEndpoint() orelse { + _ = logging.write("pci-cap-test: FAIL endpoint creation\n"); + return; + }; + const message = device.msiBind(nic_id, endpoint) orelse { + _ = logging.write("pci-cap-test: FAIL msi_bind\n"); + return; + }; + if (!check("msi_bind address", message.address == 0xFEE0_0000)) return; + if (!check("programMsi", function.programMsi(message))) return; + const msi_cap = function.findCapability(.msi).?; + const msi_control = mmio.readRegister(u16, msi_cap.offset + pci_class.msi.control); + const msi_data_offset: usize = if (msi_control & pci_class.msi.control_64bit_capable != 0) pci_class.msi.data_64 else pci_class.msi.data_32; + if (!check("msi registers read back", msi_control & pci_class.msi.control_enable != 0 and + msi_control & pci_class.msi.control_multiple_message_enable_mask == 0 and + mmio.readRegister(u32, msi_cap.offset + pci_class.msi.address) == @as(u32, @truncate(message.address)) and + mmio.readRegister(u16, msi_cap.offset + msi_data_offset) == @as(u16, @truncate(message.data)))) return; + if (!check("intx disabled with msi", function.command() & pci_class.command_interrupt_disable != 0)) return; + function.disableMsi(); + if (!check("disableMsi", mmio.readRegister(u16, msi_cap.offset + pci_class.msi.control) & pci_class.msi.control_enable == 0)) return; + + // MSI-X: map the table, program entry 0, exercise the masks. Never enable — this + // proves the programming surface, not delivery. + const msix_table = function.msix() orelse { + _ = logging.write("pci-cap-test: FAIL msix table map\n"); + return; + }; + writeLine("pci-cap-test: msix table has {d} entries\n", .{msix_table.entry_count}); + if (!check("msix entry count", msix_table.entry_count >= 1)) return; + if (!check("msix programEntry", msix_table.programEntry(0, message))) return; + if (!check("msix entry reads back", mmio.readRegister(u32, msix_table.table + pci_class.msix.entry_address) == @as(u32, @truncate(message.address)) and + mmio.readRegister(u32, msix_table.table + pci_class.msix.entry_data) == message.data and + mmio.readRegister(u32, msix_table.table + pci_class.msix.entry_vector_control) & pci_class.msix.entry_vector_control_masked != 0)) return; + if (!check("msix unmask entry", msix_table.unmaskEntry(0) and + mmio.readRegister(u32, msix_table.table + pci_class.msix.entry_vector_control) & pci_class.msix.entry_vector_control_masked == 0)) return; + if (!check("msix re-mask entry", msix_table.maskEntry(0) and + mmio.readRegister(u32, msix_table.table + pci_class.msix.entry_vector_control) & pci_class.msix.entry_vector_control_masked != 0)) return; + msix_table.setFunctionMask(); + if (!check("msix function mask", mmio.readRegister(u16, msix_table.capability + pci_class.msix.control) & pci_class.msix.control_function_mask != 0)) return; + msix_table.clearFunctionMask(); + if (!check("msix function unmask", mmio.readRegister(u16, msix_table.capability + pci_class.msix.control) & pci_class.msix.control_function_mask == 0)) return; + if (!check("msix out-of-range rejected", !msix_table.programEntry(msix_table.entry_count, message))) return; + + // Extended capabilities: the walk must terminate cleanly; the count is informative + // (don't hard-bind to QEMU's exact extended-capability set). + var extended_count: u32 = 0; + var extended = function.extendedCapabilities(); + while (extended.next()) |_| extended_count += 1; + writeLine("pci-cap-test: {d} extended capabilities\n", .{extended_count}); + if (!check("extended walk terminates", extended_count < 480)) return; + + // Power: QEMU leaves the function in D0; ensurePowerStateD0 must agree and not + // disturb the PMCSR. + const power_cap = function.findCapability(.power_management).?; + const pmcsr_before = mmio.readRegister(u16, power_cap.offset + pci_class.power_management.control_status); + if (!check("power state is D0", pmcsr_before & pci_class.power_management.control_status_power_state_mask == pci_class.power_management.power_state_d0)) return; + function.ensurePowerStateD0(); + if (!check("ensurePowerStateD0 is a no-op at D0", mmio.readRegister(u16, power_cap.offset + pci_class.power_management.control_status) == pmcsr_before)) return; + + // Function-level reset, where the device offers it: afterwards the function must be + // readable with its identity intact, and bring-up must work again. + const express_cap = function.findCapability(.pci_express).?; + const device_capabilities = mmio.readRegister(u32, express_cap.offset + pci_class.pci_express.device_capabilities); + if (device_capabilities & pci_class.pci_express.device_capabilities_flr != 0) { + if (!check("functionLevelReset", function.functionLevelReset())) return; + if (!check("identity after flr", function.vendorId() == intel_vendor and function.deviceId() == e1000e_device)) return; + function.enableMemoryAndBusMaster(); + if (!check("re-enable after flr", function.command() & pci_class.command_memory_and_bus_master == pci_class.command_memory_and_bus_master)) return; + } else { + _ = logging.write("pci-cap-test: flr not offered, skipped\n"); + } + + _ = logging.write("pci-cap-test: all checks passed\n"); +}