The flip: PCI enumeration leaves the kernel (M19.3)

enumeratePci, addBars, pciConfigurationPtr, and the PciHeader struct are
deleted; the kernel seeds only the host bridge, and the ring-3 pci-bus
driver's reports are the sole source of PCI function nodes. The manager
matches PCI drivers from reported identity, deduped by registered device
id so a bus restart never double-spawns.

The flip did its job by exposing a latent SMP race: ring-3
device_register made the broker table concurrent for the first time, and
mmio_map read it lock-free — under load a torn resource length mapped
hpet's window wrong (its user fault) and underflowed r.len-1 into a
kernel integer-overflow panic. Fixed: the broker read in mmio_map (and
claim) runs under the big kernel lock, the arithmetic rejects
zero-length and wrapping windows cleanly, and pci-bus no longer registers
unimplemented size-0 BARs. driver-restart hammered 6x, suite 55/55.
This commit is contained in:
Daniel Samson
2026-07-13 02:54:50 +01:00
parent d26262bf56
commit af2c766f42
7 changed files with 135 additions and 185 deletions
+14 -139
View File
@@ -360,25 +360,6 @@ const Hpet = extern struct {
page_protection: u8,
};
// --- PCI configuration-space header (first 64 bytes, common fields) ---------
const PciHeader = extern struct {
vendor_id: u16 align(1),
device_id: u16 align(1),
command: u16 align(1),
status: u16 align(1),
revision_id: u8,
prog_if: u8,
subclass: u8,
class_code: u8,
cache_line_size: u8,
latency_timer: u8,
/// bit 7 set => multi-function device.
header_type: u8,
bist: u8,
// 0x10 onward (BARs, etc.) depends on header_type; read separately.
};
// --- Entry point ------------------------------------------------------------
/// Discover hardware from the ACPI tables rooted at `rsdp_physical` and populate
@@ -452,7 +433,7 @@ fn handleTable(device_tree: *DeviceTree, hal: Hal, sdt_physical: u64) !void {
if (std.mem.eql(u8, &sig, &APIC)) {
try parseMadt(device_tree, header);
} else if (std.mem.eql(u8, &sig, &MCFG)) {
try parseMcfg(device_tree, hal, header);
try parseMcfg(device_tree, header);
} else if (std.mem.eql(u8, &sig, &HPET)) {
try parseHpet(device_tree, hal, header);
} else if (std.mem.eql(u8, &sig, &FACP)) {
@@ -537,7 +518,7 @@ fn parseMadt(device_tree: *DeviceTree, header: *const SystemDescriptorTableHeade
}
/// MCFG -> a pci_host_bridge per ECAM segment, then a PCI enumeration underneath.
fn parseMcfg(device_tree: *DeviceTree, hal: Hal, header: *const SystemDescriptorTableHeader) !void {
fn parseMcfg(device_tree: *DeviceTree, header: *const SystemDescriptorTableHeader) !void {
const total: usize = header.length;
const base: [*]const u8 = @ptrCast(header);
@@ -557,7 +538,11 @@ fn parseMcfg(device_tree: *DeviceTree, hal: Hal, header: *const SystemDescriptor
// window functions' I/O BARs must register-contain within (M19.2).
_ = bridge.addResource(.io_port, 0, 1 << 16);
try enumeratePci(device_tree, bridge, hal, alloc.*);
// The function walk itself retired to ring 3 (M19.3): the pci-bus
// driver claims this bridge, repeats the scan through its ECAM grant,
// and device_registers what it finds — the kernel seeds only the
// bridge. The scan's equivalence was proven before the hand-off
// (pci-scan), and the walk's history is in git if archaeology calls.
}
}
@@ -587,7 +572,13 @@ fn addBridgeApertures(bridge: *device_model.Device) void {
var high_end: u64 = 1 << 32;
for (boot_memory_regions) |region| {
const end = region.base + region.pages * 4096;
if (end > high_end) high_end = end;
// Above 4 GiB only *usable RAM* blocks the aperture: OVMF describes
// its own 64-bit PCI window as a reserved region and then programs
// BARs inside it — honoring reserved there would exclude the very
// space BARs live in. Below 4 GiB every described region blocks (the
// kernel image, the tables, the ramdisk all live there). Bring-up
// trust: only the bridge's claimant can register into the aperture.
if (region.kind == .usable and end > high_end) high_end = end;
if (region.base >= (1 << 32) or below_count == below.len) continue;
below[below_count] = .{ .base = region.base, .end = @min(end, 1 << 32) };
below_count += 1;
@@ -623,106 +614,6 @@ fn addBridgeApertures(bridge: *device_model.Device) void {
_ = bridge.addResource(.memory, high_end, (@as(u64, 1) << 46) - high_end);
}
/// Brute-force scan the ECAM window's bus range for present PCI functions. No
/// bridge recursion yet: on the ECAM path the host bridge decodes every bus in
/// the window, so scanning the declared range finds everything QEMU exposes.
fn enumeratePci(
device_tree: *DeviceTree,
bridge: *device_model.Device,
hal: Hal,
alloc: McfgAllocation,
) !void {
var bus: u16 = alloc.start_bus;
while (bus <= alloc.end_bus) : (bus += 1) {
var device: u8 = 0;
while (device < 32) : (device += 1) {
const h0: *align(1) const PciHeader = @ptrCast(pciConfigurationPtr(alloc, hal, @intCast(bus), device, 0));
if (h0.vendor_id == 0xFFFF) continue; // no function 0 => slot empty
const funcs: u8 = if (h0.header_type & 0x80 != 0) 8 else 1;
var function: u8 = 0;
while (function < funcs) : (function += 1) {
const configuration = pciConfigurationPtr(alloc, hal, @intCast(bus), device, function);
const h: *align(1) const PciHeader = @ptrCast(configuration);
if (h.vendor_id == 0xFFFF) continue;
var nb: [24]u8 = undefined;
const nm = std.fmt.bufPrint(&nb, "{s}:{x:0>2}:{x:0>2}.{d}", .{
bridge.name(), bus, device, function,
}) catch "pcidev";
const node = try device_tree.addChild(bridge, .pci_device, nm);
// Resource 0 is the function's own 4 KiB ECAM configuration space. A
// claimed PCI driver mmio_maps this to reach its command register,
// BARs, and — the point — its capability list (MSI/MSI-X, PCIe
// extended caps), without any new syscall. Physical address per the
// ECAM formula (same as pciConfigurationPtr).
const config_physical = alloc.base_address +
(@as(u64, @as(u8, @intCast(bus)) - alloc.start_bus) << 20) +
(@as(u64, device) << 15) + (@as(u64, function) << 12);
_ = node.addResource(.memory, config_physical, abi.page_size);
node.ids.pci_vendor = h.vendor_id;
node.ids.pci_device = h.device_id;
node.ids.pci_class = (@as(u24, h.class_code) << 16) |
(@as(u24, h.subclass) << 8) | h.prog_if;
node.ids.pci_bdf = (@as(u16, @intCast(bus)) << 8) | (@as(u16, device) << 3) | function;
// BARs only exist in header type 0 (normal devices), not bridges.
if (h.header_type & 0x7F == 0) addBars(node, configuration);
}
}
}
}
/// Record and size the memory/IO windows named by a device's Base Address
/// Registers. Sizing is the standard probe: disable decode, write all-ones, read
/// back the writable (address) bits, restore. `size = ~mask + 1`.
fn addBars(node: *device_model.Device, configuration: [*]align(1) u8) void {
// Stop the device decoding its BARs while we transiently write all-ones.
const command = rd(u16, configuration, 0x04);
wr(u16, configuration, 0x04, command & ~@as(u16, 0b11));
var i: usize = 0;
while (i < 6) : (i += 1) {
const off = 0x10 + i * 4;
const orig = rd(u32, configuration, off);
if (orig == 0) continue;
if (orig & 1 != 0) {
// I/O-space BAR (16-bit address space on x86).
wr(u32, configuration, off, 0xFFFF_FFFF);
const readback = rd(u32, configuration, off);
wr(u32, configuration, off, orig);
const mask = readback & 0xFFFF_FFFC;
const size: u32 = if (mask == 0) 0 else (~mask +% 1) & 0xFFFF;
_ = node.addResource(.io_port, orig & 0xFFFF_FFFC, size);
} else if ((orig >> 1) & 0x3 == 2) {
// 64-bit memory BAR: this BAR pair spans two configuration slots.
const orig_hi = rd(u32, configuration, off + 4);
wr(u32, configuration, off, 0xFFFF_FFFF);
wr(u32, configuration, off + 4, 0xFFFF_FFFF);
const lo = rd(u32, configuration, off);
const hi = rd(u32, configuration, off + 4);
wr(u32, configuration, off, orig);
wr(u32, configuration, off + 4, orig_hi);
const readback = (@as(u64, hi) << 32) | (lo & 0xFFFF_FFF0);
const size: u64 = if (readback == 0) 0 else ~readback +% 1;
const address = (@as(u64, orig_hi) << 32) | (orig & 0xFFFF_FFF0);
_ = node.addResource(.memory, address, size);
i += 1; // consumed the high half
} else {
// 32-bit memory BAR.
wr(u32, configuration, off, 0xFFFF_FFFF);
const readback = rd(u32, configuration, off);
wr(u32, configuration, off, orig);
const mask = readback & 0xFFFF_FFF0;
const size: u32 = if (mask == 0) 0 else ~mask +% 1;
_ = node.addResource(.memory, orig & 0xFFFF_FFF0, size);
}
}
wr(u16, configuration, 0x04, command); // restore decode
}
/// HPET -> a timer node with its register block as an MMIO resource, plus the GSI
/// its comparators can raise.
///
@@ -1253,16 +1144,6 @@ fn readCntRegister(base: [*]align(1) const u8, len: usize, xoff: usize, legacy_o
/// The mapped configuration space of one PCI function (its 4 KiB ECAM page). Mapped
/// writable so BAR sizing can probe it; reads and writes both go through here.
fn pciConfigurationPtr(alloc: McfgAllocation, hal: Hal, bus: u8, device: u8, function: u8) [*]align(1) u8 {
const physical = alloc.base_address +
(@as(u64, bus - alloc.start_bus) << 20) +
(@as(u64, device) << 15) +
(@as(u64, function) << 12);
// Map the configuration page (writable, for BAR sizing) and use the virtual
// address the HAL hands back.
return @ptrFromInt(hal.mapMmio(physical, abi.page_size, true));
}
/// Read a little-endian integer at `off` from a (possibly unaligned) byte pointer.
/// x86 is little-endian and native, so an unaligned load suffices.
fn rd(comptime T: type, bytes: [*]align(1) const u8, off: usize) T {
@@ -1270,12 +1151,6 @@ fn rd(comptime T: type, bytes: [*]align(1) const u8, off: usize) T {
return p.*;
}
/// Write a little-endian integer at `off` through a (possibly unaligned) pointer.
fn wr(comptime T: type, bytes: [*]align(1) u8, off: usize, value: T) void {
const p: *align(1) T = @ptrCast(bytes + off);
p.* = value;
}
// --- tests ------------------------------------------------------------------
test "eisaIdToStr decodes a packed EISA id" {