pci: full driver-side library (MSI/MSI-X, power, FLR, extended caps); xhci goes interrupt-driven
library/device/pci is now the complete generic floor a leaf PCI driver needs, instead of just what virtio-gpu used: - pci-class: capability IDs, MSI/MSI-X/power-management/PCI-Express register layouts, extended-capability header decode (host-tested), per-bit command constants, remaining header offsets. - pci.Function: header accessors, disableBusMaster + interrupt-disable helpers, findCapability, programMsi/disableMsi, MsiX vector-table struct, ensurePowerStateD0, functionLevelReset (BAR save/restore), extended-capability iterator. Proven by the new pci-caps QEMU case: a pci-cap-test fixture claims an extra e1000e (PM+MSI+PCIe+MSI-X, no danos driver) and readback-verifies every surface, including the first driver-side use of msi_bind. usb-xhci-bus converts from 8 ms event-ring polling to message-signalled interrupts: plain MSI where offered (real Intel xHC), MSI-X entry 0 otherwise (qemu-xhci has no MSI capability), byte-identical polling as fallback. The timer survives as a 250 ms port-reconcile/lost-edge tick — real-hardware USB2 hub debounce still needs it. MSI setup runs BEFORE controller bring-up: QEMU's xhci only registers the MSI-X vector as used when IMAN.IE is written while MSI-X is already enabled; interrupts are silently dropped otherwise (real hardware does not care about the order). 101/101 QEMU cases green; real-hardware smoke passed (mouse works, boot 2026-07-23T174805Z, plain-MSI branch, vector 33).
This commit is contained in:
@@ -28,18 +28,37 @@ const usb_ids = @import("usb-ids");
|
||||
const usb_abi = @import("usb-abi");
|
||||
const usb_transfer_protocol = @import("usb-transfer-protocol");
|
||||
const library = @import("usb-xhci-library.zig");
|
||||
const pci = @import("pci");
|
||||
|
||||
/// The controller engine (reset, rings, transfers), stood up in `initialise`.
|
||||
var controller: ?library.Controller = null;
|
||||
|
||||
/// This driver's service endpoint (registered as `.usb_bus`), where class-driver
|
||||
/// requests, signals, and the interrupt-poll timer all arrive.
|
||||
/// requests, signals, MSI notifications, and the poll/reconcile timer all arrive.
|
||||
var service_endpoint: ipc.Handle = 0;
|
||||
|
||||
/// How often the driver drains the event ring for interrupt reports (~125 Hz),
|
||||
/// re-armed each tick. Frequent enough for responsive input.
|
||||
/// How often the driver drains the event ring for interrupt reports (~125 Hz) when
|
||||
/// polling, re-armed each tick. Frequent enough for responsive input.
|
||||
const poll_interval_ms: u64 = 8;
|
||||
|
||||
/// The timer interval in MSI mode: the ring is drained at interrupt time, and the tick
|
||||
/// only reconciles root ports (real hardware delivers late USB2 companion-hub debounce
|
||||
/// with no reliable port-change event — see onNotification) and un-wedges a lost MSI
|
||||
/// edge (edge-triggered, no kernel mask/ack: a missed IP clear stalls, never storms).
|
||||
const reconcile_interval_ms: u64 = 250;
|
||||
|
||||
/// The controller's own descriptor, kept at file scope because `pci.Function` holds a
|
||||
/// pointer to it for the whole bring-up.
|
||||
var controller_descriptor: device.DeviceDescriptor = undefined;
|
||||
|
||||
/// Non-null iff MSI mode is active: the vector whose notification badge means "the
|
||||
/// controller interrupted". Null means the 8 ms polling fallback is running.
|
||||
var msi_vector: ?u32 = null;
|
||||
|
||||
fn timerInterval() u64 {
|
||||
return if (msi_vector != null) reconcile_interval_ms else poll_interval_ms;
|
||||
}
|
||||
|
||||
/// The class driver endpoints that opened each device, so interrupt reports can
|
||||
/// be pushed back to them. Keyed by the device token (the interface's device id).
|
||||
const Open = struct {
|
||||
@@ -95,6 +114,7 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
std.log.info("device {d} not in the device tree", .{controller_id});
|
||||
return false;
|
||||
};
|
||||
controller_descriptor = descriptor;
|
||||
|
||||
// The xHC's registers live behind the first memory BAR. Resource 0 is the
|
||||
// function's ECAM configuration space (M15), so the walk starts at 1.
|
||||
@@ -118,6 +138,14 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
return false;
|
||||
};
|
||||
|
||||
// Message-signalled interrupt setup comes BEFORE the controller bring-up, not
|
||||
// after: Controller.init writes IMAN.IE, and QEMU's xhci only registers the MSI-X
|
||||
// vector as in-use when that write happens with MSI-X already enabled (its
|
||||
// intr_update callback early-outs on !msix_enabled, and msix_notify silently
|
||||
// drops interrupts for an unused vector). Real hardware does not care about the
|
||||
// order; QEMU requires it.
|
||||
setupMsi();
|
||||
|
||||
// Bring the controller up: reset it, stand up the command and event rings,
|
||||
// and start it running (the hardware half lives in usb-xhci-library.zig).
|
||||
controller = library.Controller.init(register_base) orelse {
|
||||
@@ -146,12 +174,52 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
|
||||
scanPorts(handle);
|
||||
|
||||
// Arm the poll timer that drains interrupt reports from the event ring. It is
|
||||
// re-armed on each tick in onNotification; class drivers subscribe later.
|
||||
_ = time.timerOnce(service_endpoint, poll_interval_ms);
|
||||
// Arm the timer: in polling mode it drains the event ring; in MSI mode it is the
|
||||
// slower port-reconcile/safety-net tick. Re-armed on each tick in onNotification.
|
||||
_ = time.timerOnce(service_endpoint, timerInterval());
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Switch the event ring from timer polling to message-signalled interrupts, if the
|
||||
/// whole path is available: map the function's config space, bind a vector, program
|
||||
/// the MSI capability — or, on an MSI-X-only function (QEMU's qemu-xhci is one: it
|
||||
/// advertises MSI-X and PCIe but no plain MSI), entry 0 of the MSI-X table, which
|
||||
/// takes the same kernel (address, data) pair (xHCI interrupter 0 raises vector 0).
|
||||
/// Any step failing leaves `msi_vector` null and the 8 ms polling path exactly as it
|
||||
/// was. The controller side needs nothing extra — IMAN.IE and USBCMD.INTE are already
|
||||
/// set (see Controller.init: QEMU only writes runtime events with the interrupter
|
||||
/// enabled).
|
||||
///
|
||||
/// After a supervised kill, the kernel drops the vector binding but the device still
|
||||
/// has the interrupt enabled and fires the stale vector; the kernel EOIs it
|
||||
/// harmlessly, and the respawned driver re-runs this with its fresh vector.
|
||||
fn setupMsi() void {
|
||||
var function = pci.Function.map(controller_id, &controller_descriptor) orelse {
|
||||
std.log.info("config-space map failed; polling at {d} ms", .{poll_interval_ms});
|
||||
return;
|
||||
};
|
||||
function.enableMemoryAndBusMaster();
|
||||
const message = device.msiBind(controller_id, service_endpoint) orelse {
|
||||
std.log.info("msi_bind unavailable; polling at {d} ms", .{poll_interval_ms});
|
||||
return;
|
||||
};
|
||||
if (function.programMsi(message)) {
|
||||
msi_vector = message.data;
|
||||
std.log.info("msi active (vector {d}); reconcile tick at {d} ms", .{ message.data, reconcile_interval_ms });
|
||||
return;
|
||||
}
|
||||
if (function.msix()) |table| {
|
||||
if (table.programEntry(0, message) and table.unmaskEntry(0)) {
|
||||
table.enable();
|
||||
function.setInterruptDisable();
|
||||
msi_vector = message.data;
|
||||
std.log.info("msix active (vector {d}); reconcile tick at {d} ms", .{ message.data, reconcile_interval_ms });
|
||||
return;
|
||||
}
|
||||
}
|
||||
std.log.info("no msi/msi-x capability; polling at {d} ms", .{poll_interval_ms});
|
||||
}
|
||||
|
||||
var register_base: usize = 0;
|
||||
|
||||
/// The xHCI default Protocol Speed IDs (the PORTSC port-speed field, bits 13:10)
|
||||
@@ -501,10 +569,27 @@ fn handleBulk(message: []const u8, reply: []u8) usize {
|
||||
return writeReply(reply, usb_transfer_protocol.BulkReply{ .status = if (transferred != null) 0 else -1, .actual_length = transferred orelse 0 });
|
||||
}
|
||||
|
||||
/// The poll timer landed: drain any interrupt reports off the event ring and push
|
||||
/// each to the class driver that subscribed, then re-arm the timer.
|
||||
/// A timer tick or an MSI landed: drain the event ring, reconcile ports, and fan out.
|
||||
/// The timer arm re-arms itself (8 ms drain when polling, 250 ms reconcile under MSI);
|
||||
/// the MSI arm clears the interrupter's pending bit FIRST, then drains — so an event
|
||||
/// arriving after the drain takes IP 0→1 and fires a fresh edge instead of being
|
||||
/// swallowed until the reconcile tick.
|
||||
fn onNotification(badge: u64) void {
|
||||
if (badge & ipc.notify_timer_bit == 0) return;
|
||||
if (badge & ipc.notify_timer_bit != 0) {
|
||||
serviceController();
|
||||
_ = time.timerOnce(service_endpoint, timerInterval());
|
||||
return;
|
||||
}
|
||||
const vector = msi_vector orelse return;
|
||||
if (badge & ~ipc.notify_badge_bit != vector) return;
|
||||
if (controller) |*engine| engine.acknowledgeInterrupt();
|
||||
serviceController();
|
||||
}
|
||||
|
||||
/// Everything one servicing pass does, shared verbatim by the poll/reconcile tick and
|
||||
/// the MSI notification: drain the event ring, reconcile root ports, service hub
|
||||
/// changes, and push interrupt reports to their class drivers.
|
||||
fn serviceController() void {
|
||||
if (controller) |*engine| {
|
||||
engine.pump();
|
||||
// Poll every root port and reconcile — a device present but not yet
|
||||
@@ -564,7 +649,6 @@ fn onNotification(badge: u64) void {
|
||||
_ = ipc.send(report.report_endpoint, std.mem.asBytes(&message));
|
||||
}
|
||||
}
|
||||
_ = time.timerOnce(service_endpoint, poll_interval_ms);
|
||||
}
|
||||
|
||||
pub fn main(init: process.Init) void {
|
||||
|
||||
@@ -1563,6 +1563,15 @@ pub const Controller = struct {
|
||||
return report;
|
||||
}
|
||||
|
||||
/// Clear interrupter 0's pending bit (IMAN.IP). IP is write-1-to-clear, and the
|
||||
/// read-back carries IE (plain read-write) through unchanged. In MSI mode the
|
||||
/// driver clears IP **before** draining the ring: an event that lands after the
|
||||
/// drain then takes IP 0→1 and fires a fresh edge, where clearing afterwards would
|
||||
/// leave a race in which a new event finds IP already set and raises nothing.
|
||||
pub fn acknowledgeInterrupt(self: *const Controller) void {
|
||||
write32(self.interrupter(interrupter_management), read32(self.interrupter(interrupter_management)) | 1);
|
||||
}
|
||||
|
||||
/// Drain any events currently on the event ring: interrupt reports into the
|
||||
/// report queue, PORT STATUS CHANGES into the port-change queue (hot-plug —
|
||||
/// these were silently dropped before M20). Non-blocking — called on the
|
||||
|
||||
@@ -212,6 +212,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
|
||||
deviceListTest(boot_information);
|
||||
} else if (eql(case, "pci-scan")) {
|
||||
pciScanTest(boot_information);
|
||||
} else if (eql(case, "pci-caps")) {
|
||||
pciCapsTest(boot_information);
|
||||
} else if (eql(case, "acpi-parse")) {
|
||||
acpiParseTest(boot_information);
|
||||
} else if (eql(case, "acpi-report")) {
|
||||
@@ -2388,6 +2390,41 @@ fn deviceListTest(boot_information: *const BootInformation) void {
|
||||
result();
|
||||
}
|
||||
|
||||
/// The driver-side PCI library against a real function: the pci-caps QEMU case adds an
|
||||
/// e1000e NIC no danos driver claims; the pci-cap-test fixture claims it and exercises
|
||||
/// header accessors, command bits, the capability walks, MSI programming (the first
|
||||
/// driver-side `msi_bind` use), the MSI-X table, power state, and FLR. The kernel side
|
||||
/// only spawns the manager (which spawns pci-bus itself) and the fixture; the substance
|
||||
/// is asserted by the harness on the fixture's own serial lines.
|
||||
fn pciCapsTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: pci-caps\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over an initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
|
||||
process.setInitialRamdisk(image);
|
||||
// Plain mode — no restart drill, whose kill would race the fixture's claim.
|
||||
var manager: u32 = 0;
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
|
||||
manager = process.spawnProcessSupervised(item.blob, 4, &.{"device-manager"}, scheduler.currentId(), null) catch 0;
|
||||
break;
|
||||
}
|
||||
check("device-manager spawned", manager != 0);
|
||||
check("pci-cap-test spawned", spawnNamed(rd, "pci-cap-test"));
|
||||
result();
|
||||
}
|
||||
|
||||
/// M19.1: the ring-3 PCI scan agrees with the kernel's. The manager spawns
|
||||
/// pci-bus for the host bridge; the driver walks the same ECAM window through
|
||||
/// its mmio_map grant and must find exactly the functions the kernel's own
|
||||
|
||||
Reference in New Issue
Block a user