The flip: PCI enumeration leaves the kernel (M19.3)

enumeratePci, addBars, pciConfigurationPtr, and the PciHeader struct are
deleted; the kernel seeds only the host bridge, and the ring-3 pci-bus
driver's reports are the sole source of PCI function nodes. The manager
matches PCI drivers from reported identity, deduped by registered device
id so a bus restart never double-spawns.

The flip did its job by exposing a latent SMP race: ring-3
device_register made the broker table concurrent for the first time, and
mmio_map read it lock-free — under load a torn resource length mapped
hpet's window wrong (its user fault) and underflowed r.len-1 into a
kernel integer-overflow panic. Fixed: the broker read in mmio_map (and
claim) runs under the big kernel lock, the arithmetic rejects
zero-length and wrapping windows cleanly, and pci-bus no longer registers
unimplemented size-0 BARs. driver-restart hammered 6x, suite 55/55.
This commit is contained in:
Daniel Samson
2026-07-13 02:54:50 +01:00
parent d26262bf56
commit af2c766f42
7 changed files with 135 additions and 185 deletions
+22 -3
View File
@@ -298,6 +298,8 @@ fn systemDeviceEnumerate(state: *architecture.CpuState) void {
/// device_claim(id) -> 0/-1: take exclusive ownership of a device for this process.
fn systemDeviceClaim(state: *architecture.CpuState) void {
const claim_flags = sync.enter();
defer sync.leave(claim_flags);
if (devices_broker.claim(architecture.systemCallArg(state, 0), scheduler.current().id))
architecture.setSystemCallResult(state, 0)
else
@@ -312,10 +314,22 @@ fn systemMmioMap(state: *architecture.CpuState) void {
const resource_index = architecture.systemCallArg(state, 1);
const t = scheduler.current();
if (t.aspace == 0) return fail(state);
const owner = devices_broker.ownerOf(device_id) orelse return fail(state);
if (owner != t.id) return fail(state); // not claimed by this process
const r = devices_broker.resourceOf(device_id, resource_index) orelse return fail(state);
// Read the broker table under the lock: ring-3 device_register (M19) now
// mutates it concurrently on other cores, so a lock-free read here could
// see a torn resource (and a torn length used to panic the arithmetic
// below on integer overflow).
const r = blk: {
const flags = sync.enter();
defer sync.leave(flags);
const owner = devices_broker.ownerOf(device_id) orelse return fail(state);
if (owner != t.id) return fail(state); // not claimed by this process
break :blk devices_broker.resourceOf(device_id, resource_index) orelse return fail(state);
};
if (r.kind != @intFromEnum(device_abi.ResourceKind.memory)) return fail(state);
// A zero-length or wrapping window is not mappable — fail cleanly rather
// than underflow `r.len - 1`.
if (r.len == 0) return fail(state);
if (@addWithOverflow(r.start, r.len)[1] != 0) return fail(state);
if (t.device_map_next == 0) t.device_map_next = device_arena_base;
const first = r.start & ~@as(u64, page_size - 1);
@@ -451,6 +465,11 @@ fn systemDeviceRegister(state: *architecture.CpuState) void {
var descriptor: device_abi.DeviceDescriptor = undefined;
if (!ipc.copyFromUser(t.aspace, descriptor_ptr, std.mem.asBytes(&descriptor))) return fail(state);
// Under the big kernel lock: the broker's table is also mutated by the
// death sweep (releaseAllOwnedBy) and read by enumerate on other cores —
// ring-3 registration (M19) made those genuinely concurrent.
const flags = sync.enter();
defer sync.leave(flags);
const id = devices_broker.register(parent_id, t.id, &descriptor) catch return fail(state);
architecture.setSystemCallResult(state, id);
}